Compare commits

..

6 Commits

Author SHA1 Message Date
diegosouzapw
66f12222a9 docs: sync the migration count with the code (171 → 172)
The base merge brought a 172nd migration; README, AGENTS and llm.txt still
claimed 171 and the strict docs-counts gate failed on this branch. The 50
llm.txt mirrors are re-synced with the root body, as docs-sync requires.
2026-09-10 09:26:46 -03:00
diegosouzapw
4151cf50b0 fix(i18n): restore the ICU literal escape around angle placeholders
English writes '<name>' — the single quotes are ICU's literal escape, so the
text renders as <name>. The translation backend dropped them in the nine new
locales, turning the span into an unclosed ICU tag: featureFlags
OMNIROUTE_AUTO_SYNC_CLAUDE_PROFILES.description failed with UNCLOSED_TAG in all
nine, and cliTools.ccOnboardingKeyPlaceholder with INVALID_TAG in six.

Where the translated text carries its own apostrophe (Estonian OmniRoute'i,
Irish d'eochair) the apostrophe is doubled, otherwise it closes the literal
early. Guards: i18n-cc-onboarding-placeholder-12302 and
feature-flag-description-icu-parse-12505.
2026-09-10 08:44:55 -03:00
diegosouzapw
04358e11f0 Merge remote-tracking branch 'origin/release/v3.8.51' into feat/i18n-batch-eu 2026-09-10 08:33:31 -03:00
diegosouzapw
2aef14e40d docs(changelog): point the batch-1 fragment at #13044
Also re-syncs the nine new llm.txt mirrors with the root file, which moved
with the base merge. docs-sync keeps the body byte-identical to the root
minus its top heading; the mirrors carry their own heading plus the language
bar above the --- separator.
2026-09-08 09:22:36 -03:00
diegosouzapw
885e17fd52 Merge remote-tracking branch 'origin/release/v3.8.51' into feat/i18n-batch-eu 2026-09-08 09:19:28 -03:00
diegosouzapw
c1ea96e03f feat(i18n): add 9 European locales (el hr sr lt et lv sl mt ga)
Batch 1 of the locale-expansion plan: Greek, Croatian, Serbian, Lithuanian,
Estonian, Latvian, Slovenian, Maltese and Irish across every surface —
dashboard catalog, docs mirrors, CLI catalog, README, locale index and the
marketing site. OmniRoute now ships all 24 official EU languages (51 locales).

Also fixes two defects the batch exposed:

- The placeholder-parity gate matched every "{…}" pair, so an ICU plural branch
  body (other {s}) counted as an argument named "s" and any correct plural
  translation was reported as drift. The scanner now follows the ICU grammar.
  Three translations that invented a {count} argument the English source never
  defines were corrected, as was one Irish string that translated the argument
  name itself.
- Language bars linked to mirrors that do not exist: docs/guides/I18N.md is
  English-only by design yet keeps legacy mirrors, so every new locale got a
  dead link. Bars now skip locales without a mirror on disk.
2026-09-08 09:15:01 -03:00
260 changed files with 958 additions and 11075 deletions

View File

@@ -142,7 +142,6 @@ jobs:
- run: npm run check:known-symbols
- run: npm run check:route-guard-membership
- run: npm run check:test-discovery
- run: npm run check:radar-sentinels
- run: npm run check:tracked-artifacts
# (gap 30) Also lives in quality.yml's PR-only "Merge integrity" job — because the
# CHANGELOG half of that job needs a base to diff against. This half does NOT: the

View File

@@ -22,7 +22,7 @@
"node": ">=22.22.3"
},
"peerDependencies": {
"@opencode-ai/plugin": ">=1.18.29 <2"
"@opencode-ai/plugin": "*"
}
},
"node_modules/@ai-sdk/provider": {
@@ -39,9 +39,9 @@
}
},
"node_modules/@esbuild/aix-ppc64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/aix-ppc64/-/aix-ppc64-0.28.2.tgz",
"integrity": "sha512-XExcO+dvLKvVtNTibSTBej1NCAbaGhWn9Ww1ZPx80qsahhPFe/8jgWP0IchNe0F3HwkU7n8ejhH8bjonqht8mQ==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/aix-ppc64/-/aix-ppc64-0.28.1.tgz",
"integrity": "sha512-Svl7tq8k/08+p6CXPpRjQ1fKX+1odH/BQbb48fV6fj3CWHhsoIOoY87w1oHXm0qEpkIK3ZfVgp0hed3XBXzXMQ==",
"cpu": [
"ppc64"
],
@@ -56,9 +56,9 @@
}
},
"node_modules/@esbuild/android-arm": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/android-arm/-/android-arm-0.28.2.tgz",
"integrity": "sha512-kXXoiPVVGQcnIYGOeaovwOURpniDBpSq4A03qkQ+BMQqtGG6HYap3xne9C1O1yo4TR3qxlCX5IqqmX6fFo2Lqg==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/android-arm/-/android-arm-0.28.1.tgz",
"integrity": "sha512-0k2F129Xdio1TdJfzJ8sy1Q47vUD2NnwdhiAf7drUN1EBTfPf4hsFCtmMgu/6m8JSzsBrlmVjudMBQqOfG8usQ==",
"cpu": [
"arm"
],
@@ -73,9 +73,9 @@
}
},
"node_modules/@esbuild/android-arm64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/android-arm64/-/android-arm64-0.28.2.tgz",
"integrity": "sha512-5YfKeeI8qWfBZIX+u2xZC3Zlb3Os/gLS2sbEKM+I4ZOcsWmHS2WLysCcQZDAFRslDUU5Oiq44gf6PYN1vGwG5A==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/android-arm64/-/android-arm64-0.28.1.tgz",
"integrity": "sha512-34EGEbCIAgosYz6goLcopX6Mo7NyGv9tfwEM2/7Ce2VcVRk568iSvniGWcUXIy7wEDR1wzolcxcriFVrWYcwBg==",
"cpu": [
"arm64"
],
@@ -90,9 +90,9 @@
}
},
"node_modules/@esbuild/android-x64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/android-x64/-/android-x64-0.28.2.tgz",
"integrity": "sha512-O387ite7SzUyCcy3JQX4P4bLtEA7bLLkx+esve5JHnyYfNTxcVpXZo9jhdB0lTKN44gztELTdU7nS8Nr16Fs1Q==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/android-x64/-/android-x64-0.28.1.tgz",
"integrity": "sha512-dbwY7ltSMDWsRatcRpCnES4F+im88OCUgGZjy52shC7GqHRE/cYlxNbB4Z4UpJswpcc4Qxd2oE/ufM0p61IKng==",
"cpu": [
"x64"
],
@@ -107,9 +107,9 @@
}
},
"node_modules/@esbuild/darwin-arm64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/darwin-arm64/-/darwin-arm64-0.28.2.tgz",
"integrity": "sha512-n4KqkOQrraxHJcgjM1RvwbigfQKIKJVpM7xp+KsxiyUSrRdIXnt73VhrPAx0fV44hgfmIVKjxMN9J1t5jySVkw==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/darwin-arm64/-/darwin-arm64-0.28.1.tgz",
"integrity": "sha512-TZbWkQY7kvTAXbXUT7uVACR5cMHsDiSz9z7ZKAX/RTq/WJEk3QyRr0wZpNhBDX+/0CtdqUIJlOiodQcta6tY3Q==",
"cpu": [
"arm64"
],
@@ -124,9 +124,9 @@
}
},
"node_modules/@esbuild/darwin-x64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/darwin-x64/-/darwin-x64-0.28.2.tgz",
"integrity": "sha512-uq6suIWYP37qzGddBKPw5QEQPi6HiLGsO7UmkpfyaYNQ3D+rN6w6WfwH+nuqcGXWvawGwxOEroO4YGnFh95azw==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/darwin-x64/-/darwin-x64-0.28.1.tgz",
"integrity": "sha512-zfdzgK9ACBNZLI/CyHTOx81SyNbM6YXn7rxSgX97VjyiPl9W1i4Ka4fgKECEoFCKGpvBj5qArWIGgQjOwkgskQ==",
"cpu": [
"x64"
],
@@ -141,9 +141,9 @@
}
},
"node_modules/@esbuild/freebsd-arm64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/freebsd-arm64/-/freebsd-arm64-0.28.2.tgz",
"integrity": "sha512-n+I0BTSRIoy+d6RPKnEVwql5UwBJolytvY4mAOIEJorKlqgPII8ix6slVVrfZ5Tnj7glIZvloylbB/EJPMWEXw==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/freebsd-arm64/-/freebsd-arm64-0.28.1.tgz",
"integrity": "sha512-wG2EA8ENdEI0qhkSZMjfqrdY+ziCYCPMmtZjjIwOmXFjmyzEHn+UUxk5of+SYsjtfs3VpnlC7QLzSI5hY/rOAw==",
"cpu": [
"arm64"
],
@@ -158,9 +158,9 @@
}
},
"node_modules/@esbuild/freebsd-x64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/freebsd-x64/-/freebsd-x64-0.28.2.tgz",
"integrity": "sha512-78XJTJkvPs0kz2w61301PJjXl4g7q3JqiYMZ/M/yVI73EHBrCRTgkhu9oqG7vPqq+a/yadEW8aD+agKlk5xrmg==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/freebsd-x64/-/freebsd-x64-0.28.1.tgz",
"integrity": "sha512-i7dZ9vQgnvSCzi/rYCXNgtF/U+eKZNJBzu3eTQbRgHnM7tNSizLOkRFAl3qzVc/Op/u5YkHHa4pf/3DOYHthLQ==",
"cpu": [
"x64"
],
@@ -175,9 +175,9 @@
}
},
"node_modules/@esbuild/linux-arm": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/linux-arm/-/linux-arm-0.28.2.tgz",
"integrity": "sha512-XlDnu2q5yoqems+xay6wSAcg9DDD7K9RLKZEBOMZm3ckNpJBvOX20tSfby8KfrrhINDyv9V2YVZKY/SpoGJI8w==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/linux-arm/-/linux-arm-0.28.1.tgz",
"integrity": "sha512-qVXBOHQS+d5Y722GwJzJUtOLlX7km3CraOaGormF1pDtPd2C/l1SHRPgjLunLGe51Sh5YYWKMFDyV4SxgMQYTQ==",
"cpu": [
"arm"
],
@@ -192,9 +192,9 @@
}
},
"node_modules/@esbuild/linux-arm64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/linux-arm64/-/linux-arm64-0.28.2.tgz",
"integrity": "sha512-pW4AC0P3it8c7do9MVM4p51FzHzdM/TZrerurgRcHJ2WTa1VQ1CIq18xncfpBJw4ojkiZZrKW2yIBWBP92j6Ug==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/linux-arm64/-/linux-arm64-0.28.1.tgz",
"integrity": "sha512-yHs+0uc8+nvEAfAfxrWQKK5peSNzBc4PegcMO0EJ2hT71uA7vB8Ihg2e77R2P7SG5uYjPbHlLLmve4LLLRCf0g==",
"cpu": [
"arm64"
],
@@ -209,9 +209,9 @@
}
},
"node_modules/@esbuild/linux-ia32": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/linux-ia32/-/linux-ia32-0.28.2.tgz",
"integrity": "sha512-CYbnj78HsIeA+DhgUKgFCfvNsTHFhMMrinUrMZpDXJXKN8T3XViTZ/+wtHeVxEWY8ewSzTFN+nRmSwO2tZaLUQ==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/linux-ia32/-/linux-ia32-0.28.1.tgz",
"integrity": "sha512-d1z4ZuP0ajrfz/FhGT4vv278rX8KnPPJx8i5+AtK7TYbx9Le9F1hyzurZpkEyjkGa9dUGhQow4C1NmeGvqxN2w==",
"cpu": [
"ia32"
],
@@ -226,9 +226,9 @@
}
},
"node_modules/@esbuild/linux-loong64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/linux-loong64/-/linux-loong64-0.28.2.tgz",
"integrity": "sha512-buwkd8nsph4R+ajRvw0qM5Hja/TXQow3ptzWO2EbG/cqcIkHloRrdlBtQlshyYGTNFvfkfJ5tpPLVkY4DtsPfQ==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/linux-loong64/-/linux-loong64-0.28.1.tgz",
"integrity": "sha512-M5sRjUVZrkm1OAPR3dlOYzNmN+loZKGVi1VUQGrwuqLcbR6qeAz+famMhjASeH3YVKvZz+zT1jlh/keC3Rj/lg==",
"cpu": [
"loong64"
],
@@ -243,9 +243,9 @@
}
},
"node_modules/@esbuild/linux-mips64el": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/linux-mips64el/-/linux-mips64el-0.28.2.tgz",
"integrity": "sha512-ZVykbDyk7519VwiNb9Lcj9m8XM6v5V9uKPvrEMkkEedVewf+0itkhahp4HDpgERXhwLRpWFypsGbG/J8s0QjJA==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/linux-mips64el/-/linux-mips64el-0.28.1.tgz",
"integrity": "sha512-mRObBZeHh2OxcBFPWE/FjylkRgZdYuiTR3vaTozquCGOH14iP9oN4x4Ge81CoIDYQrXmIxpFumJBu5MtZpnQJQ==",
"cpu": [
"mips64el"
],
@@ -260,9 +260,9 @@
}
},
"node_modules/@esbuild/linux-ppc64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/linux-ppc64/-/linux-ppc64-0.28.2.tgz",
"integrity": "sha512-CAXl+Dtd9UUuJd8pKKdwh6MLm3MUMiqMPmhZ3tTSXPqfyQ3vDl6R5hZdZ/kYojK4ofXtdfSv1tFq8XzWx3heNQ==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/linux-ppc64/-/linux-ppc64-0.28.1.tgz",
"integrity": "sha512-slScBsMAb3GFDcdrCgLwZtPYRoH2H/youv10QiZyRjmsP48fznoveWytSgCI/R0ZcUgpc0ZhIUEx6LHts8yrfQ==",
"cpu": [
"ppc64"
],
@@ -277,9 +277,9 @@
}
},
"node_modules/@esbuild/linux-riscv64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/linux-riscv64/-/linux-riscv64-0.28.2.tgz",
"integrity": "sha512-GeXCej4IQtU1B+QlDV8W/RRvbzI3O/Stss+/bCXv4lZls5WGRtu2a+3JkA3i4qIUlMXpcHebWpF8AkJhATowuA==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/linux-riscv64/-/linux-riscv64-0.28.1.tgz",
"integrity": "sha512-kw0owk1o0GFETUJyW0jc0G4Yzs0BHZn0JDZ8JRT088vjJYX777BAs1fDGxAC+q831qOs2DTC96mNsG2opdfyyQ==",
"cpu": [
"riscv64"
],
@@ -294,9 +294,9 @@
}
},
"node_modules/@esbuild/linux-s390x": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/linux-s390x/-/linux-s390x-0.28.2.tgz",
"integrity": "sha512-3H1weTYZPxt/WOhByszQZybS9w5lKzUn1FDMsgEChbHWQwHYQQRfBxgCcZvPhjHfKyJjIievvMmEUawJrdY9Dg==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/linux-s390x/-/linux-s390x-0.28.1.tgz",
"integrity": "sha512-/lAIjX8aYFRByhh6L5rYtPEDRqa9de/4V/juOXcta5frjvzXO4/sqEtyytse0g3zZFuWu5cDN0MkLz2qRDD2Ag==",
"cpu": [
"s390x"
],
@@ -311,9 +311,7 @@
}
},
"node_modules/@esbuild/linux-x64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/linux-x64/-/linux-x64-0.28.2.tgz",
"integrity": "sha512-4xTZr1FUmSoQW4XIWmit3tzQrUTZM+N3P0XV8xROKYF50XfI7xeO90+1bZvNwxIufQ9hDQVRJH5YhgPVF8A/HQ==",
"version": "0.28.1",
"cpu": [
"x64"
],
@@ -328,9 +326,9 @@
}
},
"node_modules/@esbuild/netbsd-arm64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/netbsd-arm64/-/netbsd-arm64-0.28.2.tgz",
"integrity": "sha512-sSATRjPeDBg3pdgHoQfoYBob11Kk1FGa9lui5RIHZCoCkJa9QKlvl3/vKz2usCmYYjs7ymJR/2Nnsqe+Hjt5nw==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/netbsd-arm64/-/netbsd-arm64-0.28.1.tgz",
"integrity": "sha512-oks0DYbLwWMmaakTsCb+zL4E+aHRVLom9IJZOAthMQEPiQmydXHkziYEsGYRx0uNV/IjEKGAV941JzH02pflqw==",
"cpu": [
"arm64"
],
@@ -345,9 +343,9 @@
}
},
"node_modules/@esbuild/netbsd-x64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/netbsd-x64/-/netbsd-x64-0.28.2.tgz",
"integrity": "sha512-lqnzCV+mM0gIADaKihiCg6ifgfU2L3h5E33rNQBN1Y4MaVGnzryzmvvf7UHxprpQdE8hpqLolJ9Rl+SkIRDpyw==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/netbsd-x64/-/netbsd-x64-0.28.1.tgz",
"integrity": "sha512-aeL6lAnN89Hz43Mlh1G8ARasbuoYvSITDEx0tHh5b7jJnHcssqgjy9Yx430GDpmCa6OyrKoS0aNRjKundRizGg==",
"cpu": [
"x64"
],
@@ -362,9 +360,9 @@
}
},
"node_modules/@esbuild/openbsd-arm64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/openbsd-arm64/-/openbsd-arm64-0.28.2.tgz",
"integrity": "sha512-AL2qJILH7lNjrDmCQDvdxMfAUIv8KMNZOvrwAQ8i8//ntL9FflhOyMJ8OZSMBb8/AWXe3/5v5S20y3zCoZWKoQ==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/openbsd-arm64/-/openbsd-arm64-0.28.1.tgz",
"integrity": "sha512-MEFJe5C3R8pwXdZ5Y21oo6m7ePiS0d9pWucn99O/wvyJZChoIQKrQDxKrGeW8F5+T0okTHesAmDeiHDTIq0V/Q==",
"cpu": [
"arm64"
],
@@ -379,9 +377,9 @@
}
},
"node_modules/@esbuild/openbsd-x64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/openbsd-x64/-/openbsd-x64-0.28.2.tgz",
"integrity": "sha512-QtiuPytchRyC4rwUKhexJdQKvDuZ6hWloi3igqPQNUJCS1/v9EiO3UTOXR6A3FoMo4fnAKbWJdqaIwhOzh8qEw==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/openbsd-x64/-/openbsd-x64-0.28.1.tgz",
"integrity": "sha512-i/ZLIOafE0Z8cI/XANJAixoJL/uRAoS2xOA3rb0xN+KK0K177cMAsQYkzHtBrtMXAKuAc7HGgcWiZ/sRC1Nxgw==",
"cpu": [
"x64"
],
@@ -396,9 +394,9 @@
}
},
"node_modules/@esbuild/openharmony-arm64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/openharmony-arm64/-/openharmony-arm64-0.28.2.tgz",
"integrity": "sha512-WkhYDmpTjLvGlScA1rwjRUmhl4k8oXR3cIbtqWmELgU/dFeHHlEllxDvdWcNJV9rbzCexB5vz8gtNewWLgCT7Q==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/openharmony-arm64/-/openharmony-arm64-0.28.1.tgz",
"integrity": "sha512-ge+Z7EXFNt2BO1oAMsVpiQ8EwndV9i1xXerAeTIK7AtPs3bKFXQM7nlRxDSIUIMeueR1CNXxqztLzdNeReKBJg==",
"cpu": [
"arm64"
],
@@ -413,9 +411,9 @@
}
},
"node_modules/@esbuild/sunos-x64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/sunos-x64/-/sunos-x64-0.28.2.tgz",
"integrity": "sha512-GPMSkTOtMnv2U2F8gxe4Io6qmVs+YKyp832Etqqxr0hFngmXQ3rzwytelm3GIn7T4VviRUlf3sOgBOiTdvaf7g==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/sunos-x64/-/sunos-x64-0.28.1.tgz",
"integrity": "sha512-BEjgtECkL3vY+SaSQ6nzVfiALUeFxpawyp8Jmf5PtYhf1Ug40N1h/hxlhts+f1FvSvarEigdxS3BlSMI2PJLcQ==",
"cpu": [
"x64"
],
@@ -430,9 +428,9 @@
}
},
"node_modules/@esbuild/win32-arm64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/win32-arm64/-/win32-arm64-0.28.2.tgz",
"integrity": "sha512-PIhhEkE9uPBleRBrQEJpUn7MBnibZzbGzYWPmY3x+YoVg/95zbjB4CxPPOQ8l5tYYM4mMaCthF8/1DIfBQQyWQ==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/win32-arm64/-/win32-arm64-0.28.1.tgz",
"integrity": "sha512-lCv9eK/H6ZJWbE7bh2nw54CZ9M2nupBxJcTsdk/QQnWkdSjKGuxmmH8/GWrlT1eMmZfn4dGcCjRte397WqfQXA==",
"cpu": [
"arm64"
],
@@ -447,9 +445,9 @@
}
},
"node_modules/@esbuild/win32-ia32": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/win32-ia32/-/win32-ia32-0.28.2.tgz",
"integrity": "sha512-YmJbfTlvU7Sdn9BB+4PRES4oB6pxgS37MAONj+hBr/cpXS1aBPKXxNnDbu+QCWPj0o9dgyxeq79g6c5P8KeuYA==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/win32-ia32/-/win32-ia32-0.28.1.tgz",
"integrity": "sha512-zvb/mB2bSCoJOpoCBgYKKpX6YM6mJBlBUVUtVj41DlZJVEB6/0CKlRYxP5wWl1C1ILiCoAU5wZZ4q1P3qeS6Eg==",
"cpu": [
"ia32"
],
@@ -464,9 +462,9 @@
}
},
"node_modules/@esbuild/win32-x64": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/@esbuild/win32-x64/-/win32-x64-0.28.2.tgz",
"integrity": "sha512-5ebpxr3nWMzrL/rnUI755Jkuee0bHL/Gq0WTF9lvcpv73wAp5eu8MfBUgWK9bhWvZjj7yX8etf/8tI8Ney695g==",
"version": "0.28.1",
"resolved": "https://registry.npmjs.org/@esbuild/win32-x64/-/win32-x64-0.28.1.tgz",
"integrity": "sha512-bm4Mowrv+GXMlpWX++EcXw/iLyd1o3+bJkC2DkWXYVvgZCqD/bSj9ctZeAMC3cIxgjRVR2Dufaiu4YPxr5gW1A==",
"cpu": [
"x64"
],
@@ -1182,9 +1180,7 @@
}
},
"node_modules/esbuild": {
"version": "0.28.2",
"resolved": "https://registry.npmjs.org/esbuild/-/esbuild-0.28.2.tgz",
"integrity": "sha512-HKVLS8dvII+xoKW9kmqxbRKrnWEXfJJr/FZhhJmiqIB0e053QNYFqOBouTMO/k5sID4MvCiUCvv8b9M4h32wIA==",
"version": "0.28.1",
"dev": true,
"hasInstallScript": true,
"license": "MIT",
@@ -1195,32 +1191,32 @@
"node": ">=18"
},
"optionalDependencies": {
"@esbuild/aix-ppc64": "0.28.2",
"@esbuild/android-arm": "0.28.2",
"@esbuild/android-arm64": "0.28.2",
"@esbuild/android-x64": "0.28.2",
"@esbuild/darwin-arm64": "0.28.2",
"@esbuild/darwin-x64": "0.28.2",
"@esbuild/freebsd-arm64": "0.28.2",
"@esbuild/freebsd-x64": "0.28.2",
"@esbuild/linux-arm": "0.28.2",
"@esbuild/linux-arm64": "0.28.2",
"@esbuild/linux-ia32": "0.28.2",
"@esbuild/linux-loong64": "0.28.2",
"@esbuild/linux-mips64el": "0.28.2",
"@esbuild/linux-ppc64": "0.28.2",
"@esbuild/linux-riscv64": "0.28.2",
"@esbuild/linux-s390x": "0.28.2",
"@esbuild/linux-x64": "0.28.2",
"@esbuild/netbsd-arm64": "0.28.2",
"@esbuild/netbsd-x64": "0.28.2",
"@esbuild/openbsd-arm64": "0.28.2",
"@esbuild/openbsd-x64": "0.28.2",
"@esbuild/openharmony-arm64": "0.28.2",
"@esbuild/sunos-x64": "0.28.2",
"@esbuild/win32-arm64": "0.28.2",
"@esbuild/win32-ia32": "0.28.2",
"@esbuild/win32-x64": "0.28.2"
"@esbuild/aix-ppc64": "0.28.1",
"@esbuild/android-arm": "0.28.1",
"@esbuild/android-arm64": "0.28.1",
"@esbuild/android-x64": "0.28.1",
"@esbuild/darwin-arm64": "0.28.1",
"@esbuild/darwin-x64": "0.28.1",
"@esbuild/freebsd-arm64": "0.28.1",
"@esbuild/freebsd-x64": "0.28.1",
"@esbuild/linux-arm": "0.28.1",
"@esbuild/linux-arm64": "0.28.1",
"@esbuild/linux-ia32": "0.28.1",
"@esbuild/linux-loong64": "0.28.1",
"@esbuild/linux-mips64el": "0.28.1",
"@esbuild/linux-ppc64": "0.28.1",
"@esbuild/linux-riscv64": "0.28.1",
"@esbuild/linux-s390x": "0.28.1",
"@esbuild/linux-x64": "0.28.1",
"@esbuild/netbsd-arm64": "0.28.1",
"@esbuild/netbsd-x64": "0.28.1",
"@esbuild/openbsd-arm64": "0.28.1",
"@esbuild/openbsd-x64": "0.28.1",
"@esbuild/openharmony-arm64": "0.28.1",
"@esbuild/sunos-x64": "0.28.1",
"@esbuild/win32-arm64": "0.28.1",
"@esbuild/win32-ia32": "0.28.1",
"@esbuild/win32-x64": "0.28.1"
}
},
"node_modules/fast-check": {

View File

@@ -1745,9 +1745,9 @@
}
},
"node_modules/toml": {
"version": "4.3.0",
"resolved": "https://registry.npmjs.org/toml/-/toml-4.3.0.tgz",
"integrity": "sha512-lVb8X9BsPVuH0M4BKeS91tXAmJvCjQ5UIyAbQFaxkKGyUFK2RPkhwaFSQH8vbpl1d23eu/IBH+dwVMHWaq9A5A==",
"version": "4.1.1",
"resolved": "https://registry.npmjs.org/toml/-/toml-4.1.1.tgz",
"integrity": "sha512-EBJnVBr3dTXdA89WVFoAIPUqkBjxPMwRqsfuo1r240tKFHXv3zgca4+NJib/h6TyvGF7vOawz0jGuryJCdNHrw==",
"dev": true,
"license": "MIT",
"engines": {

View File

@@ -35,24 +35,16 @@ export function resolveOpencodeTarget(opts = {}) {
baseUrl = `http://localhost:${Number(opts.port ?? process.env.PORT ?? 20128) || 20128}`;
}
// Precedence: explicit --api-key flag > OMNIROUTE_API_KEY env var > active
// context's management token. A context's accessToken/apiKey is a CLI
// management credential (oma_live_...) with no /v1/* inference scope — it
// must never silently outrank a real inference key the caller supplied
// either as a flag or via the ambient env var (mirrors the explicit >
// ambient-env > context precedence documented in bin/cli/api.mjs's
// buildHeaders()). Only fall back to the context token when neither an
// explicit flag nor the env var is set.
let apiKey = opts.apiKey ?? opts["api-key"];
if (!apiKey) apiKey = process.env.OMNIROUTE_API_KEY || "";
if (!apiKey) {
try {
const c = resolveActiveContext(opts.context ?? process.env.OMNIROUTE_CONTEXT);
apiKey = c?.accessToken || c?.apiKey || "";
apiKey = c?.accessToken || c?.apiKey;
} catch {
/* no context auth */
}
}
if (!apiKey) apiKey = process.env.OMNIROUTE_API_KEY || "";
return { baseUrl: baseUrl.replace(/\/+$/, ""), apiKey };
}
@@ -185,17 +177,8 @@ export function registerSetupOpencode(program) {
"--allow-container-write",
"Write even when the target is inside a container and not mounted from the host"
)
.action(async (opts, cmd) => {
// Commander parses the ancestor program's own global --api-key option
// (bin/cli/program.mjs, bound to .env("OMNIROUTE_API_KEY")) against any
// occurrence of the flag in argv, so it wins the value even when the
// user typed --api-key AFTER `setup-opencode` — this local option's own
// `opts.apiKey` never sees it. cmd.optsWithGlobals() resolves to the
// correct value either way ("globals overwrite locals" is exactly the
// outcome we want here, since the global option is where the value
// always actually lands).
const resolvedOpts = { ...opts, apiKey: cmd.optsWithGlobals().apiKey ?? opts.apiKey };
const code = await runSetupOpencodeCommand(resolvedOpts);
.action(async (opts) => {
const code = await runSetupOpencodeCommand(opts);
if (code !== 0) process.exit(code);
});
}

View File

@@ -1 +0,0 @@
- **feat(providers):** advertise a `free-tier` capability in the provider plugin manifest for every provider with documented free models, so sidecars and dashboards can filter free-capable providers without reading the quota catalog ([#12786](https://github.com/diegosouzapw/OmniRoute/pull/12786)) — thanks @maxmad64bis

View File

@@ -1,11 +0,0 @@
- **feat(dashboard):** the orchestration History tab gained a "Compare runs" mode — toggling it
turns each grid cell into a 2-item selection queue (a 3rd click drops the oldest pick), and
picking two cells opens a side-by-side comparison panel instead of the usual detail drawer.
The panel fetches both runs' detail the same way the drawer does (falling back to persisted
history once a run leaves the live TTL window) and shows, per side: identity/source/state,
start time, a signed `right - left` delta for duration/cost/event count, the event timeline
aligned by index, and any memory hits. One side's fetch failing never blocks the other, and a
delta is only ever computed when both sides have a finite value — otherwise it renders "—",
never `NaN`. Comparing two runs from different sources or skills still works; a banner marks
the deltas as informational rather than hiding them, since the two runs aren't a strict
apples-to-apples pair.

View File

@@ -1 +0,0 @@
- **fix(resilience):** stop unbounded queue that hangs 6min until Aborted — gate, provider slot, and Bottleneck queue now share a per-connection `maxWaitMs` budget; `fail-closed` on exhaust (503); execution backstop `executionMaxWaitMs` overridable per connection with upstream clamp ([#12715](https://github.com/diegosouzapw/OmniRoute/pull/12715)) — thanks @maxmad64bis (with thanks to @Tushar49 for surfacing the slow-provider need in #12635)

View File

@@ -1 +0,0 @@
- **fix(auto-combo):** every mode pack now carries `quality` and `reliability` weights (`quality` 0.02, 0.03 in `quality-first`; `reliability` 0.03, 0.04 in `reliability-first`), so selecting a pack no longer silences either signal; `DEFAULT_WEIGHTS` is unchanged ([#12731](https://github.com/diegosouzapw/OmniRoute/pull/12731)) — thanks @maxmad64bis

View File

@@ -1 +0,0 @@
- **fix(models):** stop treating provider-supplied `isFree:true`/`:free`/`0/0` flags as trusted unless the provider has a documented free tier — only the shipped free-tier catalog decides otherwise; fetched `isFree:true` trusted only for a free-tier provider, custom `isFree:true` via a trusted path ([#12744](https://github.com/diegosouzapw/OmniRoute/pull/12744)) — thanks @maxmad64bis

View File

@@ -1 +0,0 @@
- fix(cli): setup-opencode no longer sends an active context's management token to `/v1/models` when `--api-key`/`OMNIROUTE_API_KEY` is supplied — an explicit flag or the env var now always outranks the context's token, and the flag itself is no longer swallowed by the parent program's global `--api-key` option (#12783)

View File

@@ -1 +0,0 @@
- **fix(providers):** single source for provider order with `xao` ranking alongside `xai-oauth` ([#12790](https://github.com/diegosouzapw/OmniRoute/pull/12790)) — thanks @maxmad64bis

View File

@@ -1 +0,0 @@
- **fix(routing):** off-table models check the free-model catalog before inheriting premium prices, latency bootstraps from the observed pool median, async tiers read live database pricing with a 90-day freshness gate, and the tier cache invalidates on every pricing write ([#12792](https://github.com/diegosouzapw/OmniRoute/pull/12792)) — thanks @maxmad64bis

View File

@@ -1 +0,0 @@
- **fix(auto-combo):** snapshot scoring uses the observed per-connection account tier and per-model quality instead of neutral constants, and circuit-open providers rank lower at build time ([#12794](https://github.com/diegosouzapw/OmniRoute/pull/12794)) — thanks @maxmad64bis

View File

@@ -1 +0,0 @@
- **fix(pool):** empty `auto/*` pools now say why they are empty, and the models listing uses the same paid check as routing ([#12795](https://github.com/diegosouzapw/OmniRoute/pull/12795)) — thanks @maxmad64bis

View File

@@ -1 +0,0 @@
- **fix(sse):** translate-mode streams now emit the estimated token counts as a trailing usage-only chunk before `[DONE]` when the upstream stays silent, so metered chat clients see totals instead of nothing ([#12828](https://github.com/diegosouzapw/OmniRoute/pull/12828)) — thanks @maxmad64bis

View File

@@ -1 +0,0 @@
- **fix(dashboard):** the Radar catalog table no longer leaves absent data unexplained — empty limits, unknown context windows, and unreported capabilities each explain themselves on hover, and a new check keeps it that way ([#12937](https://github.com/diegosouzapw/OmniRoute/pull/12937)) — thanks @maxmad64bis

View File

@@ -1 +0,0 @@
- **fix(sse):** geo-blocked opencode requests rotate to the next account proxy instead of failing, so one refused egress no longer aborts the whole chain ([#12941](https://github.com/diegosouzapw/OmniRoute/pull/12941)) — thanks @maxmad64bis

View File

@@ -1 +0,0 @@
- **fix(monitoring):** `GET /api/monitoring/health` `credentialHealth` now includes a bounded `failedConnections` list (`connectionId`, `status`, sanitized `lastError`) when the probe-cache gauge `failed>0`, plus `source: probe-cache` and a cheap `staleDbNonOkCount` for sticky SQLite `test_status` on active rows. Documents that the live gauge is not `provider_connections.test_status`.

View File

@@ -1 +0,0 @@
- **fix(resilience):** an openai-compatible multi-upstream gateway no longer marks the whole connection `credits_exhausted` when a single upstream (or the `/models` / representative-model health probe) returns 402. Quota failures stay model-scoped; true connection-wide auth failures are unchanged. Single-credential 402 key disable ([#5239](https://github.com/diegosouzapw/OmniRoute/issues/5239)) is preserved.

View File

@@ -1,17 +0,0 @@
- **fix(dashboard):** nine fixes on the Orchestration Canvas, all diagnosed in the Phase 2 reviews.
A source that fails now keeps the timestamp of its FIRST failure instead of being re-stamped
every poll — the stale line said "since the last poll" no matter how long the source had been
down, and the churn also defeated the snapshot's stable identity (it serializes the source list),
so the canvas re-rendered on every tick while anything was broken. A source that HAD data and
only then started failing is flagged too: previously only a source with no node at all got the
warning, so a source that went down mid-session kept a healthy-looking node forever. The rest are
pointwise: clicking a filter chip cancels the pending search debounce (left armed, it fired
~300ms later and silently reverted the chip); the search input carries an accessible name;
`?state=running, failed` parses like the unpadded form instead of dropping the padded value; the
CSV toggle helper is defined once in `model/urlParams.ts` rather than twice; the agents tab tells
"nothing running" apart from "the filter matched nothing", offering a clear-filters button
instead of setup links that would be wrong advice there; edges stop emitting SMIL particles above
40 simultaneously active edges, keeping the colored stroke; and the drawer's error banner clears
when a retried action succeeds.
Closes #12392

View File

@@ -1,15 +0,0 @@
- **fix(dashboard):** five follow-ups from the Phase 2 Orchestration Canvas review. The History
drawer now shows the memory section for runs that already left the live TTL window: the
persisted `memory_hits` event is parsed into `metadata.memoryHits` with the same defensive
validation the drawer applies, and — because that event is observability rather than a state
transition — it no longer leaks into the timeline, where it had been inheriting the task's
state and rendering as a duplicate transition. The A2A memory recall runs against its own 1.5s
deadline instead of inheriting the memory backend's 30s one; overshooting degrades exactly like
any other recall failure (no hits, task proceeds), and the timer is cleared on both paths.
Repeating a Conductor run carries its `requirements.cli`/`requirements.model` forward, so the
new run is pinned to the same runner profile and model rather than drifting to whatever the
fleet picks. A successful repeat from the Agents tab now focuses the run it created, instead of
leaving the operator on the finished one. And the auth test for `POST /api/conductor/tasks`
moved into the shared `ROUTES` array rather than restating the pattern.
Closes #12639

View File

@@ -1 +0,0 @@
- **docs(checks):** keep doc counts honest — headings, rankings, catalog, weights, quality gate and scoring diagram now covered ([#12507](https://github.com/diegosouzapw/OmniRoute/pull/12507)) — thanks @maxmad64bis

View File

@@ -1 +0,0 @@
- Update the documented migration count to 172 after the call-logs provider-stats indexes landed.

View File

@@ -1 +0,0 @@
- **refactor(ui):** flow surfaces (home topology, combo live studio, compression cockpit/waterfall/nodes, token-health badge) express state through the theme-aware `--orch-status-*` tokens instead of fixed dark-mode hex, so green/red/amber/grey stay legible in the light theme; dark mode is byte-identical. Categorical palettes (routing-strategy hues, compression-layer pills, provider brand colors) deliberately stay hex ([#12378](https://github.com/diegosouzapw/OmniRoute/issues/12378))

View File

@@ -411,6 +411,11 @@
"count": 1
}
},
"open-sse/services/autoCombo/__tests__/autoCombo.test.ts": {
"@typescript-eslint/no-unused-vars": {
"count": 2
}
},
"open-sse/services/autoCombo/chaosEngine.ts": {
"@typescript-eslint/no-unused-vars": {
"count": 2

View File

@@ -1,6 +1,4 @@
{
"_rebaseline_2026_09_10_12828_translate_usage_chunk": "PR #12828 own growth: open-sse/utils/stream.ts 3072->3080 (+8). Translate-mode streams now send the estimated usage as the canonical trailing usage-only chunk before [DONE] when the upstream stays silent (parity with the #12151 passthrough flush), with a latch so a finish chunk that already carried the estimate is not doubled. The chunk builder is shared with the passthrough flush in open-sse/utils/usageOnlyChunk.ts (under cap); what remains is the flush-site wiring. Covered by tests/unit/stream-translate-usage-trailing.test.ts.",
"_rebaseline_2026_09_10_12715_queue_budget": "PR #12715 own growth: open-sse/handlers/chatCore.ts 6021->6036 (+15). Hierarchical admission now resolves the per-connection queue budget before the gates and hands withRateLimit the remaining budget, the correlation id and the executor timeout context, so gate wait, provider slot and Bottleneck queue share one bound instead of stacking. Error shaping lives in open-sse/handlers/chatCore/queueBudget.ts (under cap); what remains is irreducible call-site wiring. Covered by tests/unit/rate-limit-remaining-budget.test.ts, rate-limit-manager-queue-bound.test.ts and chatcore-hierarchical-admission.test.ts.",
"_rebaseline_2026_09_06_runtime_quotagroup_nodemap": "Own growth: src/app/(dashboard)/dashboard/runtime/RuntimePageClient.tsx 1201->1222 (+21, check-file-size split-newline). QuotaGroup is a module-level sibling and was reading nodeMap from RuntimePageClient's closure; that identifier is not in scope, so a quota monitor with status error/exhausted/alerting throws ReferenceError. Fix threads nodeMap as a prop (3 call sites + parameter + ProviderNodeEntry import). Prettier wraps the long import and the three QuotaGroup JSX tags. Covered by tests/unit/ui/runtime-page-client.test.tsx (empty monitors stay green; error+exhausted fixtures mount QuotaGroup).",
"_rebaseline_2026_09_05_claude_extra_usage_preflight": "Own growth: open-sse/services/combo.ts 4080->4084 (+4). buildAutoCandidates now forwards connection.providerSpecificData into evaluateQuotaCutoff so a Claude account with blockExtraUsage=false is not dropped at the 5h bar. Irreducible at the existing cutoff call site; the helper lives in claudeExtraUsage.ts (under cap). Covered by tests/unit/quota-preflight.test.ts.",
"_rebaseline_2026_09_04_12697_combo_pin_allowlist": "PR #12697 own growth: src/sse/handlers/chat.ts 2454->2458 (+4). checkModelAvailable preflight and handleSingleModelChat now call comboPinAllowlist so a pin-only combo step cannot scan the provider pool after 502/429. Helper lives in src/lib/combos/steps.ts under cap. Covered by tests/unit/combo-pin-implicit-allowlist.test.ts (11/11).",
@@ -426,8 +424,7 @@
"open-sse/executors/codex.ts": 1505,
"open-sse/executors/cursor.ts": 1759,
"open-sse/executors/muse-spark-web.ts": 1405,
"open-sse/handlers/chatCore.ts": 6026,
"open-sse/handlers/chatCore.ts": 6036,
"open-sse/handlers/chatCore.ts": 6021,
"open-sse/handlers/imageGeneration.ts": 3259,
"open-sse/handlers/search.ts": 1789,
"open-sse/mcp-server/schemas/tools.ts": 1621,
@@ -439,7 +436,7 @@
"open-sse/translator/response/openai-responses.ts": 1466,
"open-sse/utils/cursorAgentProtobuf.ts": 1547,
"open-sse/utils/proxyFetch.ts": 1271,
"open-sse/utils/stream.ts": 3098,
"open-sse/utils/stream.ts": 3072,
"open-sse/vendor/codex-chatgpt-web/adapters/chatgpt-web/browser-worker.ts": 4398,
"open-sse/vendor/codex-chatgpt-web/bridge.ts": 1335,
"src/app/(dashboard)/dashboard/HomePageClient.tsx": 1344,
@@ -470,9 +467,7 @@
"src/sse/handlers/chat.ts": 2458,
"src/sse/services/auth.ts": 3450,
"tests/unit/account-fallback-service.test.ts": 2453,
"tests/unit/provider-validation-specialty.test.ts": 4656,
"open-sse/services/autoCombo/virtualFactory.ts": 1219,
"open-sse/services/combo/roundRobinCombo.ts": 1205
"tests/unit/provider-validation-specialty.test.ts": 4656
},
"_rebaseline_base_2026_08_10_proxyfetch": "Base-red fix (green-prs sweep, issue #9985): open-sse/utils/proxyFetch.ts 1207 > cap 1000 — new proxied-TLS fetch helper introduced by the Fal reference-image work. Owner-authorized quick rebaseline to green; structural slim tracked for v3.9.0.",
"_rebaseline_2026_07_27_v3849_train2": "Merge-train 2 (7 PRs) — owner-approved 2026-07-27. Single entry: chatCore.ts 4955->5006 (#8595, Responses multi-turn image compaction before the context hard-reject). Genuine irreducible growth at the existing compaction chokepoint in handleChatCore — the PR adds a last-resort retry against the concrete budget plus the estimateFinalInputTokens helper, both wired at the pre-existing call site rather than a new branch. Covered by tests/unit/8560-responses-image-compaction.test.ts (4 tests).",
@@ -657,8 +652,5 @@
"_rebaseline_2026_09_03_houminxi_combo_stacked": "Leva HouMinXi (#12624 #12626 #12632 #12637): open-sse/services/combo.ts 4075->4080 (+5), medido no tip com os quatro mergeados. Cada PR registrou o proprio crescimento contra o tip de onde forkou (o #12637 ja subira o cap para 4075); as 5 linhas restantes so aparecem quando eles empilham, porque mais de um toca o mesmo chokepoint de scoring reset-aware em combo.ts. Fiacao em ponto existente, sem extracao possivel sem partir a funcao de selecao de alvos. Coberto por combo-strategies e reset-aware-request-scope-12600 (119/119 focados na leva).",
"_rebaseline_2026_09_04_12641_continuation_effective_input": "PR #12641 crescimento proprio: src/sse/handlers/chat.ts 2450->2454 (+4). A continuacao por previous_response_id encadeava a partir de clientRawRequest.body.input, que e capturado ANTES da reconstrucao do proprio chat.ts; quando o turno anterior ja era uma continuacao, esse campo guarda so o delta do cliente, e o erro se acumulava a cada salto ate a reconstrucao virar itens de tool sem prefixo. Persistir o input EFETIVO exige as linhas no ponto onde a reconstrucao termina, dentro do fluxo de despacho. Coberto por tests/unit/responses-continuation-store.test.ts (22/22 focados na leva).",
"_rebaseline_2026_09_05_12671_combos_usage_guide_external_store": "combos/page.tsx 5018 -> 5066: #12671 replaces the effect-based localStorage read with useSyncExternalStore; the +48 lines are the store helpers (subscribe/getSnapshot/getServerSnapshot/emit) hoisted to module scope, which is the sanctioned shape and what let the react-hooks/set-state-in-effect suppression be dropped.",
"_rebaseline_2026_09_07_chatcore_nonstreaming_regression_fixes": "Own growth: open-sse/handlers/chatCore.ts 5984->6021 (+37). Two of my own PRs on top of #12867: #12963 pins the ok variant of the non-streaming leg result in its own binding (the discriminated-union narrowing was lost across the tool-loop reassignment, 13 TS2339 under tsconfig.typecheck-api.json), and #12990 restores four behaviours the same refactor dropped — abort classification through isLocalStreamLifecycleError, the omitted synthetic clientResponse, the claudePromptCacheLogMeta rebuild on the leg path, and the lazy fail-closed fence identity. Irreducible at the existing chokepoints: each edit sits where chatCore already owns the decision, and the helpers themselves (nonStreamingProviderLeg.ts, serverOwnedToolLoopWire.ts) are under cap. Covered by tests/unit/chatcore-translation-paths.test.ts (72/74; the 2 open are issue #13043).",
"_rebaseline_2026_09_07_virtualfactory_crosses_the_new_file_cap": "open-sse/services/autoCombo/virtualFactory.ts crosses the 1200 new-file cap for the first time (1187 on the pre-wave tip, 1219 after the wave). Growth is spread across the routing/free-tier wave, not one extractable block: #12794 feeds observed breaker state and model quality into snapshot scoring instead of neutral constants, #12792 adds the reliability factor the snapshot path was still ignoring and the pooled-latency bootstrap, and #12744 tightens the free-model predicate the factory consumes, and #12795 records which filter stage emptied an auto/* pool. FROZEN RATHER THAN SPLIT, deliberately, and this is debt: two cohesive extraction candidates are ready when someone owns the move — computeSnapshotWeights (~85 lines) and the credential-eligibility group hasUsableOAuthToken/hasProviderSpecificSessionData/isKeylessEligibleConnection/hasUsableConnectionCredential (~70 lines). Either alone clears 1200 from here. Splitting three contributors' just-merged work mid-batch was the larger risk.",
"_rebaseline_2026_09_07_streaming_wave": "Stacked growth from the SSE/streaming wave. open-sse/handlers/chatCore.ts 6021->6026 (+5): #12854 seeds the in-memory pending continuation state synchronously, before saveCallLogOperation's first await, closing the window where resolvePreviousResponseState finds nothing because the artifact write has not landed yet. open-sse/utils/stream.ts 3080->3098 (+18): #12828 emits the trailing usage-estimate chunk on the translate flush (#12151 had only covered passthrough, so translate-mode clients never saw token counts) and #12718 stops rebuilding a truncated summary from the collector's cap-dropped event array. Irreducible at the existing chokepoints — both are the flush/finalization points themselves. Covered by the continuation-store, translate-usage and collector-truncation suites.",
"_rebaseline_2026_09_07_roundrobin_crosses_new_file_cap": "open-sse/services/combo/roundRobinCombo.ts 1198->1205, crossing the 1200 new-file cap. #12884 wires the quota-skip diagnostics into the round-robin attempt path so an ALL_TARGETS_SKIPPED 503 names which windows were exhausted instead of returning an opaque skip. The file was already at 1198 when #12811 lifted it out of combo.ts, so seven lines cross it; the diagnostics themselves live in quotaSkipDiagnostics.ts, under cap. Frozen rather than split: the natural next extraction is the attempt-loop body, which #12746/#12811 just moved and should settle before being cut again."
"_rebaseline_2026_09_07_chatcore_nonstreaming_regression_fixes": "Own growth: open-sse/handlers/chatCore.ts 5984->6021 (+37). Two of my own PRs on top of #12867: #12963 pins the ok variant of the non-streaming leg result in its own binding (the discriminated-union narrowing was lost across the tool-loop reassignment, 13 TS2339 under tsconfig.typecheck-api.json), and #12990 restores four behaviours the same refactor dropped — abort classification through isLocalStreamLifecycleError, the omitted synthetic clientResponse, the claudePromptCacheLogMeta rebuild on the leg path, and the lazy fail-closed fence identity. Irreducible at the existing chokepoints: each edit sits where chatCore already owns the decision, and the helpers themselves (nonStreamingProviderLeg.ts, serverOwnedToolLoopWire.ts) are under cap. Covered by tests/unit/chatcore-translation-paths.test.ts (72/74; the 2 open are issue #13043)."
}

View File

@@ -2,7 +2,7 @@
%% Reflects: open-sse/services/autoCombo/scoring.ts (DEFAULT_WEIGHTS, sum = 1.0)
%% v3.8.50
%% svg-title: OmniRoute Auto-Combo 16-factor scoring
%% svg-description: Flow from an incoming request through eligible candidates, the 16 weighted scoring factors, descending score sort, top-N selection, and sequential dispatch.
%% svg-description: Flow from an incoming request through eligible candidates, the 15 weighted scoring factors, descending score sort, top-N selection, and sequential dispatch.
flowchart TB
Request["Incoming request"] --> Candidates["Eligible candidates<br/>(provider × model × account)"]
Candidates --> Score["Compute composite score<br/>per candidate"]
@@ -23,7 +23,6 @@ flowchart TB
f13["resetWindowAffinity (0.0000)"]
f14["connectionDensity (0.0476)"]
f15["quality (0.0300)"]
f16["reliability (0.0000 DEFAULT, 0.03 packs, 0.04 reliable)"]
end
Score --> Factors

File diff suppressed because one or more lines are too long

Before

Width:  |  Height:  |  Size: 27 KiB

After

Width:  |  Height:  |  Size: 26 KiB

View File

@@ -105,10 +105,10 @@ Per-combo:
OmniRoute exposes **two** HTTP health surfaces. They are not interchangeable for orchestrators.
| Path | Purpose | Weight | Use for |
| ---------------------------- | ------------------------------------------------------------- | --------------------------------- | ---------------------------------------------------------------- |
| `GET /healthz` | Lifecycle liveness/readiness (`ok` / `starting` / `stopping`) | Trivial (phase flag only) | Kubernetes **readiness**; soft **liveness** if you must use HTTP |
| `GET /api/monitoring/health` | Deep system + provider summary (DB, heap, catalog counts, …) | Heavy (sync DB / monitoring work) | Dashboards, blackbox deep checks, Dockers built-in healthcheck |
| Path | Purpose | Weight | Use for |
| --- | --- | --- | --- |
| `GET /healthz` | Lifecycle liveness/readiness (`ok` / `starting` / `stopping`) | Trivial (phase flag only) | Kubernetes **readiness**; soft **liveness** if you must use HTTP |
| `GET /api/monitoring/health` | Deep system + provider summary (DB, heap, catalog counts, …) | Heavy (sync DB / monitoring work) | Dashboards, blackbox deep checks, Dockers built-in healthcheck |
> **Note:** Provider health matrices, autopilot issues, quota monitors, token health, and latency detail beyond `/api/monitoring/health` are available via the **MCP tool** `observability_snapshot` or the **dashboard** pages — there are no dedicated REST routes for those.
@@ -155,41 +155,16 @@ Response:
}
```
#### `credentialHealth`: probe-cache vs SQLite `test_status`
`GET /api/monitoring/health``credentialHealth` is the **in-memory probe-cache
gauge**, not a live dump of `provider_connections.test_status`. After #12532 the
request path reads `getCachedCredentialHealthSummary()` only; background probes
refresh the cache off the event loop.
| Layer | Where | What it means |
| ------------------------ | --------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| Probe-cache gauge | `credentialHealth.total` / `healthy` / `failed` / `unknown` / `stale` | Last credential-health probe results still held in process memory. `source` is always `probe-cache`. |
| Failed connection detail | `credentialHealth.failedConnections` | Present **only when `failed > 0`**. Bounded list of cache rows with `status=error` (`connectionId`, `status`, sanitized `lastError` / `lastErrorType`). `failedOmitted` is set when the list was capped. |
| SQLite sticky status | `credentialHealth.staleDbNonOkCount` | Count of **active** (`is_active=1`) connection rows whose persisted `test_status` is a known non-ok (`error`, `expired`, `credits_exhausted`, `banned`, `deactivated`, `unavailable`). |
The two layers can disagree on purpose:
- Gauge `failed=0` while `staleDbNonOkCount>0` — SQLite still has a sticky
`test_status` (for example `expired` or `credits_exhausted`) that the latest
probe-cache snapshot does not count as `status=error`.
- Gauge `failed>0` while SQLite looks healthy — a recent probe failed and is
cached; the DB row has not been updated, or was later cleared.
Do not alert solely on `provider_connections.test_status` when scraping this
endpoint. Use `failed` + `failedConnections` for live probe failures, and
`staleDbNonOkCount` when you need the persisted sticky-status count.
### Kubernetes probe recommendations
OmniRoute is a **single Node process** (one event loop). Stock Docker `HEALTHCHECK` targets lightweight `/healthz`. `/api/monitoring/health` is **too heavy** for kubelet liveness intervals.
| Probe | Recommended target | Notes |
| --------------- | -------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| **Startup** | HTTP `GET /healthz` with a long `failureThreshold` (or large `startPeriod`) | Cold start + SQLite migration can exceed a few seconds |
| **Readiness** | HTTP `GET /healthz` | Lifecycle `ok` / `starting` / `stopping` (200 vs 503). Still flaps if the loop is CPU-blocked. A **200 in multiple seconds is not healthy** (#10303) — it means the event loop was starved before the 3-byte handler ran |
| **Liveness** | HTTP `GET /livez`, **or TCP** on the main service port (`PORT`, default `20128`) | `/livez` is process-alive only (always 200 if the handler runs). It still shares the event loop — busy ≠ dead, and it does not detect event-loop starvation (#10303) any better than TCP does. Prefer **TCP** if HTTP probes time out under catalog/compression load; do **not** kill the pod on short event-loop stalls either way |
| **Deep health** | `GET /api/monitoring/health` from an external checker | Not for kubelet `livenessProbe` / tight `readinessProbe` |
| Probe | Recommended target | Notes |
| --- | --- | --- |
| **Startup** | HTTP `GET /healthz` with a long `failureThreshold` (or large `startPeriod`) | Cold start + SQLite migration can exceed a few seconds |
| **Readiness** | HTTP `GET /healthz` | Lifecycle `ok` / `starting` / `stopping` (200 vs 503). Still flaps if the loop is CPU-blocked. A **200 in multiple seconds is not healthy** (#10303) — it means the event loop was starved before the 3-byte handler ran |
| **Liveness** | HTTP `GET /livez`, **or TCP** on the main service port (`PORT`, default `20128`) | `/livez` is process-alive only (always 200 if the handler runs). It still shares the event loop — busy ≠ dead, and it does not detect event-loop starvation (#10303) any better than TCP does. Prefer **TCP** if HTTP probes time out under catalog/compression load; do **not** kill the pod on short event-loop stalls either way |
| **Deep health** | `GET /api/monitoring/health` from an external checker | Not for kubelet `livenessProbe` / tight `readinessProbe` |
Example shape (adjust thresholds to your cold-start and compression load):
@@ -227,6 +202,7 @@ livenessProbe:
Related: [#10052](https://github.com/diegosouzapw/OmniRoute/issues/10052) (probes while the event loop is busy), [#9685](https://github.com/diegosouzapw/OmniRoute/issues/9685) / [#10055](https://github.com/diegosouzapw/OmniRoute/pull/10055) (catalog pricing hog), [#10117](https://github.com/diegosouzapw/OmniRoute/issues/10117) (compression token-count hog).
### Optional request-path work (memory, skills, token refresh)
Memory extraction, skills injection, and OAuth token refresh share the **main Node event loop** with `/healthz`. They are dashboard-toggle features (`memoryEnabled`, `skillsEnabled`), not a worker pool. See [Environment — event-loop cost](../reference/ENVIRONMENT.md#event-loop-cost-of-memory-skills-and-token-refresh-10349).
@@ -368,7 +344,9 @@ The MCP tool `observability_snapshot` returns a **complete system snapshot** for
"ageMs": 109
}
],
"quotaMonitors": {/* see above */},
"quotaMonitors": {
/* see above */
},
"uptime": 12345,
"version": "3.8.16"
}

View File

@@ -881,15 +881,15 @@ ordinary inference API keys. Credential families, scopes, and curl examples:
### Monitoring
| Endpoint | Method | Description |
| ------------------------------------ | ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `/api/sessions` | GET | Active session tracking |
| `/api/rate-limits` | GET | Per-account rate limits |
| `/api/monitoring/health` | GET | Health check + provider summary (`catalogCount`, `configuredCount`, `activeCount`, `monitoredCount`). Management view includes `credentialHealth`: probe-cache scalars, `failedConnections` when `failed>0`, and `staleDbNonOkCount` (SQLite sticky `test_status`, not the gauge). See [MONITORING_GUIDE.md](../ops/MONITORING_GUIDE.md#credentialhealth-probe-cache-vs-sqlite-test_status). |
| `/api/cache/stats` | GET/DELETE | Cache stats / clear |
| `/api/modality-bridge/stats` | GET | In-memory `attempts`, successes/`bridged`, failures, cache hits, `totalLatencyMs`, `latencySamples`, sample-denominated `averageLatencyMs`, and last-use time (reset on restart; management auth) |
| `/api/modality-bridge/video/runtime` | GET | Strict trusted-loopback check before management auth/probe; sanitized FFmpeg/ffprobe availability and versions (no-store) |
| `/api/modality-bridge/video/extract` | POST | Internal authenticated trusted-loopback byte broker; 50 MiB input, bounded queue/32 MiB output, `503` capacity, `499` disconnect, `504` deadline; not a public upload API |
| Endpoint | Method | Description |
| ------------------------------------ | ---------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `/api/sessions` | GET | Active session tracking |
| `/api/rate-limits` | GET | Per-account rate limits |
| `/api/monitoring/health` | GET | Health check + provider summary (`catalogCount`, `configuredCount`, `activeCount`, `monitoredCount`) |
| `/api/cache/stats` | GET/DELETE | Cache stats / clear |
| `/api/modality-bridge/stats` | GET | In-memory `attempts`, successes/`bridged`, failures, cache hits, `totalLatencyMs`, `latencySamples`, sample-denominated `averageLatencyMs`, and last-use time (reset on restart; management auth) |
| `/api/modality-bridge/video/runtime` | GET | Strict trusted-loopback check before management auth/probe; sanitized FFmpeg/ffprobe availability and versions (no-store) |
| `/api/modality-bridge/video/extract` | POST | Internal authenticated trusted-loopback byte broker; 50 MiB input, bounded queue/32 MiB output, `503` capacity, `499` disconnect, `504` deadline; not a public upload API |
### Backup & Export/Import

View File

@@ -74,7 +74,7 @@ purpose.
## Methodology & caveats
- Numbers are **upper-bound estimates** from each provider's documented free-tier limits as of **2026-06-17**, gathered by web research. Free tiers change constantly — re-verify before relying on a figure.
- **What an entry actually vouches for.** No entry carries a per-row confidence rating, and the API serves none — treat every figure above as an estimate of the same, unstated quality. Two facts are different, because they are curated by hand rather than inferred: 5 entries carry an independently documented hard stop, and 13 entries carry a prompt-training disclosure. `hardStopGuaranteed` is set only when the provider's own terms say that exceeding the free allowance refuses the request rather than silently starting to bill you, with the source in a comment next to the entry; it is never defaulted to `true`, and an entry nobody has verified stays unset. So a missing hard-stop flag means "not established", not "known to bill you". **STRICT mode** (the opt-in `freeAccessPolicy=strict` routing guard) only trusts entries that carry the flag; every other free tier is excluded as `no-hard-stop` (see `open-sse/services/autoCombo/strictZeroCostFilter.ts`).
- **What an entry actually vouches for.** No entry carries a per-row confidence rating, and the API serves none — treat every figure above as an estimate of the same, unstated quality. Two facts are different, because they are curated by hand rather than inferred: 5 entries carry an independently documented hard stop, and 13 entries carry a prompt-training disclosure. `hardStopGuaranteed` is set only when the provider's own terms say that exceeding the free allowance refuses the request rather than silently starting to bill you, with the source in a comment next to the entry; it is never defaulted to `true`, and an entry nobody has verified stays unset. So a missing hard-stop flag means "not established", not "known to bill you".
- `estMonthlyFreeTokens` = recurring monthly tokens only. **One-time signup credits do not recur** and count as 0. Discontinued tiers are also 0.
- Daily token cap → `monthly = daily × 30`. Only RPD documented → `RPD × ~800 output tokens × 30`. Only RPM/TPM (no daily cap) → **uncapped** (see below).
- **Permanently free, but no published token cap** (`siliconflow`, `glm-cn`, `tencent`, `baidu`, `kilo-gateway`, `opencode-zen`, `gemini`, `ollama-cloud`): these are real recurring free access, rate/concurrency-limited. We classify them `recurring-uncapped` and **never sum them** — multiplying `RPM × 24/7 × 30d` would produce a fantasy ceiling (the inflation we reject). They are listed so you know they exist.

View File

@@ -184,7 +184,7 @@ See [#7992](https://github.com/diegosouzapw/OmniRoute/issues/7992) and [#7111](h
## How It Works (Persisted Auto-Combos)
The Auto-Combo Engine dynamically selects the best provider/model for each request using a **16-factor scoring function** (defined in `open-sse/services/autoCombo/scoring.ts``DEFAULT_WEIGHTS`). The default weights sum to `1.0`; custom weights are renormalized by `normalizeScoringWeights()`. Two of the sixteen — `cacheAffinity` and `resetWindowAffinity` — carry a default weight of `0`; `reliability` carries `0` in `DEFAULT_WEIGHTS` but `0.03` in generic packs and `0.04` in `reliability-first`, and `quality` carries `0.02` in packs (`0.03` in `quality-first`): they are still computed for every candidate, and `cacheAffinity` gates prompt-cache deduplication outside the score, so the zero-default factors simply do not vote by default while packs do.
The Auto-Combo Engine dynamically selects the best provider/model for each request using a **16-factor scoring function** (defined in `open-sse/services/autoCombo/scoring.ts``DEFAULT_WEIGHTS`). The default weights sum to `1.0`; custom weights are renormalized by `normalizeScoringWeights()`. Three of the sixteen — `cacheAffinity`, `resetWindowAffinity` and `reliability` carry a default weight of `0`: they are still computed for every candidate, and `cacheAffinity` gates prompt-cache deduplication outside the score, so they are declared factors that simply do not vote by default.
![Auto-Combo 16-factor scoring](../diagrams/exported/auto-combo-scoring.svg)
@@ -217,32 +217,30 @@ The Auto-Combo Engine dynamically selects the best provider/model for each reque
| Factor | ship-fast | cost-saver | quality-first | offline-friendly | reliability-first | chaos-mode |
| :-------------------- | :--------- | :--------- | :------------ | :--------------- | :---------------- | :--------- |
| `quota` | 0.1133 | 0.1133 | 0.0752 | **0.3324** | 0.1133 | 0.0376 |
| `quota` | 0.1333 | 0.1333 | 0.0952 | **0.3524** | 0.1333 | 0.0476 |
| `health` | 0.2667 | 0.1810 | 0.1714 | 0.2667 | **0.3524** | **0.4000** |
| `costInv` | 0.0276 | **0.3324** | 0.0276 | 0.0752 | 0.0181 | 0.0140 |
| `latencyInv` | **0.3048** | 0.0476 | 0.0476 | 0.0476 | 0.0476 | 0.0186 |
| `costInv` | 0.0476 | **0.3524** | 0.0476 | 0.0952 | 0.0381 | 0.0190 |
| `latencyInv` | **0.3048** | 0.0476 | 0.0476 | 0.0476 | 0.0476 | 0.0286 |
| `taskFit` | 0.0952 | 0.0952 | **0.3524** | 0.0000 | 0.0952 | 0.1905 |
| `stability` | 0.0000 | 0.0476 | 0.1429 | 0.0952 | 0.1905 | 0.1714 |
| `tierPriority` | 0.0376 | 0.0376 | 0.0276 | 0.0376 | 0.0276 | 0.0040 |
| `tierPriority` | 0.0476 | 0.0476 | 0.0476 | 0.0476 | 0.0476 | 0.0190 |
| `tierAffinity` | 0.0000 | 0.0000 | 0.0000 | 0.0000 | 0.0000 | 0.0000 |
| `specificityMatch` | 0.0000 | 0.0000 | 0.0000 | 0.0000 | 0.0000 | 0.0000 |
| `contextAffinity` | 0.0095 | 0.0000 | 0.0000 | 0.0000 | 0.0000 | 0.0186 |
| `contextAffinity` | 0.0095 | 0.0000 | 0.0000 | 0.0000 | 0.0000 | 0.0286 |
| `sessionAvailability` | 0.0476 | 0.0476 | 0.0476 | 0.0476 | 0.0476 | 0.0476 |
| `resetWindowAffinity` | 0.0000 | 0.0000 | 0.0000 | 0.0000 | 0.0000 | 0.0000 |
| `connectionDensity` | 0.0476 | 0.0476 | 0.0476 | 0.0476 | 0.0476 | 0.0476 |
| `quality` | 0.02 | 0.02 | **0.03** | 0.02 | 0.02 | 0.02 |
| `reliability` | 0.03 | 0.03 | 0.03 | 0.03 | **0.04** | 0.03 |
Notes:
- **Packs carry `quality` and `reliability`** (`quality 0.02`, `quality-first 0.03`; `reliability 0.03`, `reliability-first 0.04`) and replace the weight map wholesale (`weights = pack`, not a merge). `DEFAULT_WEIGHTS` carries `quality 0.03 / reliability 0`; selecting `balanced`/`default` keeps those defaults, selecting a pack uses the pack's values above. On a cold pool (no observations yet, so `quality 0.5` and `reliability 1`) these two factors add `+0.04` under a generic pack (`0.03 + 0.01`), `+0.045` under `quality-first` and `+0.05` under `reliability-first`.
- **No pack sets `quality`, and a pack replaces the weight map wholesale** (`weights = pack`, not a merge). `quality` carries `0.03` in `DEFAULT_WEIGHTS`, but under any mode pack it normalizes to `0` selecting a pack silences the observed-quality signal completely. If you want quality feedback to influence routing, leave `modePack` unset and tune the weights directly. (`cacheAffinity` is also unset by every pack, but it defaults to `0` anyway, so nothing changes there.)
- `tierAffinity`, `specificityMatch` and `resetWindowAffinity` are explicitly `0` in every pack.
- Each pack's emphasis at a glance:
- **ship-fast** → latencyInv 0.3048 + health 0.2667 (low-latency, healthy connections)
- **cost-saver** → costInv 0.3324 (cheapest tokens win)
- **quality-first** → taskFit 0.3524 + stability 0.1429 + quality 0.03, the highest of any pack (best model for the task, consistent)
- **offline-friendly** → quota 0.3324 + health 0.2667 (max headroom regardless of speed/cost)
- **reliability-first** → health 0.3524 + stability 0.1905 + reliability 0.04, the highest of any pack (fewest surprises)
- **cost-saver** → costInv 0.3524 (cheapest tokens win)
- **quality-first** → taskFit 0.3524 + stability 0.1429 (best model for the task, consistent)
- **offline-friendly** → quota 0.3524 + health 0.2667 (max headroom regardless of speed/cost)
- **reliability-first** → health 0.3524 + stability 0.1905 (fewest surprises)
- **chaos-mode** → health 0.4000 + taskFit 0.1905 (fault-injection profile)
### Per-Request Controls (headers) — #6023 / #6024 / #6025 / #3470

View File

@@ -2113,9 +2113,9 @@
}
},
"node_modules/js-yaml": {
"version": "4.3.2",
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.2.tgz",
"integrity": "sha512-SFNOvSJ+Dgf/9An904Yx+CgSlIPCkIpao4qo51lpee25TIRejdH3rhR4EZMGoNx3/TP3O+wzWuiTFl4sqbltzA==",
"version": "4.3.1",
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.1.tgz",
"integrity": "sha512-CY6crGq313MX8GkwvB7tzgp99vjQxY1++5y10/BKN/GUfHqWaOGQMNZkBvqSzsZKWk/ijwHlWzzkLulsGHhjWQ==",
"funding": [
{
"type": "github",

View File

@@ -1,24 +0,0 @@
/**
* freeTierProviders.ts — registration list of providers with documented free models.
*
* Derived from `FREE_MODEL_BUDGETS` (`./freeModelCatalog.data.ts`) filtered by
* `grantsFreeAccess` (`./freeModelCatalog.ts`), so a `discontinued` entry never
* grants access. Pure data — no imports outside `open-sse/config`, no module
* state — so it cannot introduce a cycle and `open-sse/config/providerPluginManifest.ts`
* stays a light leaf. The open-sse typecheck gate forbids open-sse → src imports;
* `src/shared/utils/freeModels.ts` reads the same source for dashboard use.
*/
import { FREE_MODEL_BUDGETS } from "./freeModelCatalog.data.ts";
import { grantsFreeAccess } from "./freeModelCatalog.ts";
const FREE_BUDGETS = FREE_MODEL_BUDGETS.filter((m) => grantsFreeAccess(m.freeType));
export const FREE_TIER_PROVIDER_SET: ReadonlySet<string> = new Set(
FREE_BUDGETS.map((m) => m.provider)
);
export function hasFreeTierProvider(idOrAlias: string | undefined | null): boolean {
if (typeof idOrAlias !== "string") return false;
return FREE_TIER_PROVIDER_SET.has(idOrAlias);
}

View File

@@ -1,12 +1,10 @@
import type { RegistryEntry, RegistryModel } from "./providers/shared.ts";
import { USAGE_FETCHER_PROVIDERS } from "../services/usage/fetcherProviders.ts";
import { USAGE_SUPPORTED_PROVIDERS } from "../services/usage/supportedProviders.ts";
import { hasFreeTierProvider } from "./freeTierProviders.ts";
export type ProviderPluginCapability =
| "apikey"
| "custom-executor"
| "free-tier"
| "oauth"
| "passthrough-models"
| "responses"
@@ -158,9 +156,6 @@ function capabilitiesFor(entry: RegistryEntry, eligible: boolean): ProviderPlugi
if (USAGE_SUPPORTED_PROVIDER_SET.has(entry.id)) {
capabilities.add("usage-supported");
}
if ([entry.id, entry.alias].some(hasFreeTierProvider)) {
capabilities.add("free-tier");
}
return [...capabilities].sort();
}

View File

@@ -28,7 +28,6 @@ import {
isEmptyUpstreamRejection,
extractChatcmplId,
} from "./accountRotation.ts";
import { isOpencodeGeoBlocked, proxyKeyOf } from "./opencodeGeoBlock.ts";
import { isNetworkRotationSharedEgressGuardEnabled } from "@/shared/utils/featureFlags";
/**
@@ -279,8 +278,8 @@ export class OpencodeExecutor extends BaseExecutor {
private accounts: OpencodeAccountState[] = [
{ fingerprint: "", cooldownUntil: 0, consecutiveFails: 0, proxy: null },
];
// Not `private`: passed as the mutable rotation cursor to
// pickRotatableAccount(), which needs a plain `{ nextAccountIdx }` shape —
// Not `private`: passed as the mutable rotation cursor to the shared
// pickAccount() helper, which needs a plain `{ nextAccountIdx }` shape —
// TS's private-member nominal check rejects `this` there otherwise.
nextAccountIdx = 0;
@@ -325,11 +324,9 @@ export class OpencodeExecutor extends BaseExecutor {
if (this.nextAccountIdx >= this.accounts.length) this.nextAccountIdx = 0;
}
/** Round-robin pick, skipping non-candidates; falls back to the next index. */
private pickAccountWith(
isReady: (account: OpencodeAccountState) => boolean
): OpencodeAccountState {
return pickRotatableAccount(this.accounts, this, isReady);
/** Round-robin pick, skipping accounts in cooldown; falls back to the next index. */
private pickAccount(): OpencodeAccountState {
return pickRotatableAccount(this.accounts, this);
}
private markCooldown(
@@ -567,48 +564,9 @@ export class OpencodeExecutor extends BaseExecutor {
// through the accounts is the retry). Avoids an unbounded loop on a
// persistently malformed upstream.
const emptyRejectionBudget = this.accounts.length === 1 ? 1 : 0;
// 403-geo tried set: proxy keys already proven geo-blocked for this
// request's model. Request-local only — nothing persists past execute().
const geoTriedProxyKeys = new Set<string>();
let directTried = false;
for (let attempt = 0; attempt < this.accounts.length + emptyRejectionBudget; attempt++) {
const isProxiedCandidate = (a: OpencodeAccountState): boolean => {
if (a.cooldownUntil > Date.now()) return false;
// Without any geo evidence this pass, every cooldown-ready account
// stays eligible (preserves the plain round-robin first pick).
if (a.proxy === null) return !directTried || geoTriedProxyKeys.size === 0;
const k = proxyKeyOf(a.proxy);
return k !== null && !geoTriedProxyKeys.has(k);
};
let account = this.pickAccountWith(isProxiedCandidate);
// Last resort: a single direct attempt (distinct egress that may
// succeed) once no proxied account is a candidate — never before.
if (!isProxiedCandidate(account) && !directTried && geoTriedProxyKeys.size > 0) {
const direct = this.accounts.find(
(a) => a.proxy === null && a.cooldownUntil <= Date.now()
);
if (direct) {
account = direct;
}
}
const lastStatus = lastResult !== null ? lastResult.response.status : null;
const lastWasGeo = lastStatus === 403 || lastStatus === 451;
if (
lastResult !== null &&
geoTriedProxyKeys.size > 0 &&
!isProxiedCandidate(account) &&
!(account.proxy === null && !directTried)
) {
// Geo exhaustion (last was 403/451) → surface as-is, no success mark.
// Any other last status (e.g. 429 after 403s) → skip without a call.
if (lastWasGeo) break;
continue;
}
// Commit the last-resort direct attempt so a later exclusion breaks
// instead of retrying it. Set here (not at pick time) so the guard
// above still lets this committed attempt through.
if (account.proxy === null && geoTriedProxyKeys.size > 0) directTried = true;
const account = this.pickAccount();
const masked = maskAccountId(account.fingerprint);
if (sharedEgressGuardEnabled && sharedEgressDown && !account.proxy) {
@@ -683,28 +641,6 @@ export class OpencodeExecutor extends BaseExecutor {
continue;
}
if (status === 403 || status === 451) {
let bodyText: string | null = null;
try {
bodyText = await result.response.clone().text();
} catch {
log?.debug?.("OPENCODE", "body read failed on geo-block check");
}
if (bodyText !== null && isOpencodeGeoBlocked(status, bodyText)) {
const key = proxyKeyOf(account.proxy);
if (key !== null) geoTriedProxyKeys.add(key);
else directTried = true;
log?.warn?.(
"OPENCODE",
`geo-blocked on account ${masked} (proxy ${key ?? "direct"}), rotating to next…`
);
// Single account with a proxy: 0 retries (same egress = dead latency).
// (The fast path above already covers single-without-proxy; here length===1 WITH proxy.)
if (this.accounts.length === 1) return result;
continue;
}
}
// Empty upstream rejection (malformed 400: no error field, no real
// content, finish_reason null — see isEmptyUpstreamRejection). Rotate/
// retry instead of propagating it as a fatal success: the observed

View File

@@ -1,55 +0,0 @@
/**
* opencodeGeoBlock.ts — geo-block predicate for the opencode executor loop.
*
* Leaf module: zero internal imports (layering — errorClassifier pulls
* accountFallback + registry + DB; this file must not). The 1010 check below
* mirrors errorClassifier.isCloudflareFingerprintRejection semantics for the
* tokens this path needs; any divergence is a bug — see the parity test.
*/
// "not available in your country" is the observed opencode RegionError phrasing
// (2026-09-07 — app.log: "This model is not available in your country.");
// siblings cover the same class, not the single incident. No bare "in your
// country/region": location text without the full prefix is not a geo signal.
const GEO_SIGNALS = [
"not available in your country",
"not available in your region",
"unsupported_country",
"unsupported country",
];
// `regionerror` word-bounded: bare substring would match region_error /
// region-error variants, which are unobserved phrasings (fail closed).
const REGION_ERROR_REGEX = /(?<![A-Za-z0-9_-])regionerror(?![A-Za-z0-9_-])/i;
// Fingerprint-first: a CDN 1010 rejection says nothing about account health —
// it must never rotate as geo. Parity with errorClassifier
// isCloudflareFingerprintRejection: the bare number 1010 alone is NOT a signal
// (it occurs as port/count/model token) — only with an explicit Cloudflare key
// or the unique tokens (mirrored vectors live in the parity test below).
const CLOUDFLARE_1010_KEY_REGEX =
/(?<![A-Za-z0-9_-])error[\s_-]?code[\\"':=\s]{0,12}1010(?!\w)|(?<![A-Za-z0-9_-])error[-_]\s?1010(?!\w)\/?/i;
function isFingerprintRejection(bodyText: string): boolean {
const text = String(bodyText || "");
const lower = text.toLowerCase();
return (
CLOUDFLARE_1010_KEY_REGEX.test(text) ||
lower.includes("browser_signature_banned") ||
lower.includes("fingerprint_rejection")
);
}
export function isOpencodeGeoBlocked(status: number, bodyText: string): boolean {
if (status !== 403 && status !== 451) return false;
const text = String(bodyText || "");
if (isFingerprintRejection(text)) return false;
const lower = text.toLowerCase();
if (REGION_ERROR_REGEX.test(text)) return true;
return GEO_SIGNALS.some((signal) => lower.includes(signal));
}
export function proxyKeyOf(proxy: { host: string; port: number } | null): string | null {
if (!proxy) return null;
return `${proxy.host}:${proxy.port}`;
}

View File

@@ -51,7 +51,32 @@ import {
outcomeFromStatus,
} from "../services/routing/index.ts";
import { routingFinishReason } from "./chatCore/routingFinishReason.ts";
/**
* Best-effort finish_reason extraction from a (possibly translated) response
* body for routing-event telemetry. Returns null when the shape is unknown.
*/
function routingFinishReason(body: unknown): string | null {
if (!body || typeof body !== "object") return null;
const record = body as Record<string, unknown>;
const choices = record.choices;
if (Array.isArray(choices)) {
const first = choices[0];
if (first && typeof first === "object") {
const fr = (first as Record<string, unknown>).finish_reason;
if (typeof fr === "string") return fr;
}
}
const output = record.output;
if (Array.isArray(output)) {
for (const item of output) {
if (item && typeof item === "object") {
const fr = (item as Record<string, unknown>).finish_reason;
if (typeof fr === "string") return fr;
}
}
}
return null;
}
import {
getHeaderValueCaseInsensitive,
isNoMemoryRequested,
@@ -354,10 +379,8 @@ import {
updateFromHeaders,
updateFromResponseBody,
initializeRateLimits,
resolveRequestQueueMaxWaitMs,
} from "../services/rateLimitManager.ts";
import * as localLimiterErrors from "../services/rateLimitManager/errors.ts";
import { rethrowAdmissionError, remainingQueueBudgetMs } from "./chatCore/queueBudget.ts";
import {
acquireMany as acquireConcurrencyGates,
markBlocked as markAccountSemaphoreBlocked,
@@ -365,7 +388,6 @@ import {
import {
lockModel,
lockModelIfPerModelQuota,
hasPerModelQuota,
recordCoreOwnedAntigravityQuotaState,
shouldDeferAntigravityQuotaStateToCaller,
} from "../services/accountFallback.ts";
@@ -3094,12 +3116,6 @@ export async function handleChatCore({
stage: "waiting_account_slot",
});
}
const maxWaitMs = resolveRequestQueueMaxWaitMs(
provider,
undefined,
attemptConnectionId ?? undefined
);
const gateStartedAt = Date.now();
const releaseAccountSemaphore = await acquireConcurrencyGates(
[
{
@@ -3116,13 +3132,12 @@ export async function handleChatCore({
},
],
{
timeoutMs: maxWaitMs,
timeoutMs: resilienceSettings.requestQueue.maxWaitMs,
maxQueueSize: resilienceSettings.requestQueue.maxQueueDepth,
signal: streamController.signal,
}
).catch(rethrowAdmissionError);
const remainingAfterGate = remainingQueueBudgetMs(maxWaitMs, gateStartedAt);
trace("post_semaphore", { maxWaitMs, remainingAfterGate });
);
trace("post_semaphore");
updatePendingScope(pendingScope, {
stage: "waiting_rate_limit",
});
@@ -3171,13 +3186,7 @@ export async function handleChatCore({
),
});
},
streamController.signal,
remainingAfterGate,
correlationId ?? undefined,
{
executor: executor as unknown as { getTimeoutMs?: () => unknown },
providerSpecificData: execCreds?.providerSpecificData,
}
streamController.signal
);
const res = normalizeExecutorResult(rawExecutorResult);
trace("post_executor", { status: res?.response?.status });
@@ -4220,18 +4229,6 @@ export async function handleChatCore({
console.warn(
`[provider] Node ${errorConnectionId} probe ${errorType} (${statusCode}) — connection stays active`
);
} else if (hasPerModelQuota(provider, model)) {
// Compatible / passthrough gateways: a 402 without a model id
// still must not terminalize the whole connection. Record the
// error for operators; sibling models stay selectable.
await updateProviderConnection(errorConnectionId, {
lastErrorType: errorType,
lastError: persistentMessage,
errorCode: statusCode,
});
console.warn(
`[provider] Node ${errorConnectionId} per-model quota exhausted (${statusCode}) — connection stays active`
);
} else {
console.warn(
`[provider] Node ${errorConnectionId} banned (${statusCode}) — disabling permanently`
@@ -5993,11 +5990,6 @@ export async function handleChatCore({
clientResponseFormat,
echoModel,
responseHeaders,
// Same adaptive budget the pre-handoff readiness gate above just used —
// reasoning models that legitimately take a while to say anything keep
// that same patience for their first REAL content, not just their first
// lifecycle frame. See pipeWithDisconnect's own doc comment.
contentStallTimeoutMs: streamReadinessPolicy.timeoutMs,
});
// ── Gamification event (fire-and-forget) ──

View File

@@ -8,10 +8,8 @@
* provider connection so it survives process restarts:
* - genuine 401/403 credential rejection → record a failure (warning, then invalid at the
* threshold), always persisted.
* - 402 → terminal (insufficient balance) on single-credential providers; mark the
* current key invalid immediately (#5239), persisted on the active→invalid
* transition. openai-compatible / per-model-quota gateways keep the key —
* a 402 there is a per-model billing signal, not a dead credential.
* - 402 → terminal (insufficient balance); mark the current key invalid immediately (#5239),
* persisted on the active→invalid transition.
* - 2xx → record a success, persisted only when recovering from a warning/invalid state.
* Model availability failures remain model/routing telemetry even when an upstream reports them
* with 401/403. Any other status only refreshes the tracked extra-key set.
@@ -25,7 +23,6 @@ import {
type KeyHealth,
} from "../../services/apiKeyRotator.ts";
import { isModelUnavailableError } from "../../services/modelFamilyFallback.ts";
import { hasPerModelQuota } from "../../services/accountFallback.ts";
import { updateProviderConnection } from "@/lib/db/providers";
type KeyHealthLog = {
@@ -112,20 +109,11 @@ export function recordKeyHealthStatus(
});
}
} else if (status === 402) {
// 402 "Insufficient account balance" is terminal for this key on
// single-credential providers — the balance won't recover mid-session
// (#5239). openai-compatible / per-model-quota gateways multiplex many
// upstreams behind one key: a 402 is a per-model billing signal and must
// not invalidate the credential used by sibling models.
const provider =
typeof creds.provider === "string"
? creds.provider
: typeof connId === "string"
? connId
: null;
if (hasPerModelQuota(provider)) {
return;
}
// 402 "Insufficient account balance" is terminal for this key — the balance
// won't recover mid-session, so mark the current key invalid immediately
// (don't wait for FAILURE_THRESHOLD) so the rotator stops returning it.
// The per-connection path already terminalizes 402 via credits_exhausted;
// this closes the per-KEY gap (#5239) for API Key Round-Robin connections.
const updatedHealth = recordKeyTerminal(connId, currentKeyId);
log?.error?.(
"AUTH",

View File

@@ -1,25 +0,0 @@
import { isLocalStreamLifecycleError } from "@/shared/utils/circuitBreaker.ts";
import {
markLocalRateLimitError,
LEGACY_RATE_LIMIT_QUEUE_TIMEOUT_CODE,
} from "../../services/rateLimitManager/errors.ts";
/**
* Rethrow a hierarchical-admission failure. Aborts, stream-lifecycle errors and coded
* semaphore errors pass through unchanged (SEMAPHORE_QUEUE_FULL is a 429 admission signal
* the combo cascade reads); only a bare time-budget timeout becomes the LEGACY queue 503.
*/
export function rethrowAdmissionError(error: unknown): never {
const err = error as { name?: unknown; code?: unknown } | null;
if (err?.name === "AbortError" || err?.code === "ABORT_ERR") throw error;
if (isLocalStreamLifecycleError(error)) throw error;
if (err?.code === undefined) {
throw markLocalRateLimitError(error as Error, LEGACY_RATE_LIMIT_QUEUE_TIMEOUT_CODE);
}
throw error;
}
/** What is left of a queue-wait budget that started at `startedAt`. */
export function remainingQueueBudgetMs(budgetMs: number, startedAt: number): number {
return Math.max(0, budgetMs - (Date.now() - startedAt));
}

View File

@@ -1,26 +0,0 @@
/**
* Best-effort finish_reason extraction from a (possibly translated) response
* body for routing-event telemetry. Returns null when the shape is unknown.
*/
export function routingFinishReason(body: unknown): string | null {
if (!body || typeof body !== "object") return null;
const record = body as Record<string, unknown>;
const choices = record.choices;
if (Array.isArray(choices)) {
const first = choices[0];
if (first && typeof first === "object") {
const fr = (first as Record<string, unknown>).finish_reason;
if (typeof fr === "string") return fr;
}
}
const output = record.output;
if (Array.isArray(output)) {
for (const item of output) {
if (item && typeof item === "object") {
const fr = (item as Record<string, unknown>).finish_reason;
if (typeof fr === "string") return fr;
}
}
}
return null;
}

View File

@@ -69,12 +69,6 @@ export function assembleStreamingPipeline(
clientResponseFormat: Parameters<typeof defaultShape>[0];
echoModel: string | null | undefined;
responseHeaders: Record<string, string>;
/** See pipeWithDisconnect's own doc comment — the post-handoff "stream is
* open but never produced real output" watchdog. Threaded from the same
* adaptive streamReadinessPolicy.timeoutMs already computed for this
* request's pre-handoff readiness gate, so slow-first-content reasoning
* models keep the same generous budget in both phases. */
contentStallTimeoutMs?: number;
},
deps: StreamingPipelineDeps = DEFAULT_DEPS
) {
@@ -89,8 +83,7 @@ export function assembleStreamingPipeline(
let piiStream = deps.pipeWithDisconnect(
args.providerResponse,
args.transformStream,
args.streamController,
{ contentStallTimeoutMs: args.contentStallTimeoutMs }
args.streamController
);
if (typeof args.createPiiTransform === "function") {
piiStream = piiStream.pipeThrough((args.createPiiTransform as () => TransformStream)());

View File

@@ -22,8 +22,8 @@ function makeTarget(provider: string, model: string): ResolvedComboTarget {
describe("ManifestAdapter", () => {
describe("generateRoutingHints - trivial query", () => {
it("returns prefer-free modifier for greeting", async () => {
const hints = await generateRoutingHints([], {
it("returns prefer-free modifier for greeting", () => {
const hints = generateRoutingHints([], {
messages: [{ content: "Hello" }],
});
expect(hints.strategyModifier).toBe("prefer-free");
@@ -32,8 +32,8 @@ describe("ManifestAdapter", () => {
});
describe("generateRoutingHints - expert query", () => {
it("returns a valid modifier for complex input", async () => {
const hints = await generateRoutingHints([], {
it("returns a valid modifier for complex input", () => {
const hints = generateRoutingHints([], {
messages: [
{
content:
@@ -47,25 +47,25 @@ describe("ManifestAdapter", () => {
});
describe("generateRoutingHints - target classification", () => {
it("marks free provider as eligible for trivial query", async () => {
it("marks free provider as eligible for trivial query", () => {
const targets = [makeTarget("kiro", "claude-sonnet-4.5")];
const hints = await generateRoutingHints(targets, {
const hints = generateRoutingHints(targets, {
messages: [{ content: "Hi" }],
});
expect(hints.eligibleTargets.length).toBeGreaterThanOrEqual(0);
});
it("handles empty targets array gracefully", async () => {
const hints = await generateRoutingHints([], {
it("handles empty targets array gracefully", () => {
const hints = generateRoutingHints([], {
messages: [{ content: "Hello" }],
});
expect(hints.eligibleTargets.length).toBe(0);
expect(hints.underqualifiedTargets.length).toBe(0);
});
it("classifies mixed targets for simple query", async () => {
it("classifies mixed targets for simple query", () => {
const targets = [makeTarget("kiro", "claude-sonnet-4.5"), makeTarget("openai", "gpt-4o")];
const hints = await generateRoutingHints(targets, {
const hints = generateRoutingHints(targets, {
messages: [{ content: "Hello" }],
});
expect(hints.eligibleTargets.length).toBeGreaterThanOrEqual(0);
@@ -73,20 +73,20 @@ describe("ManifestAdapter", () => {
});
describe("compareByCostEffectiveness", () => {
it("takes 3 arguments and returns a number", async () => {
it("takes 3 arguments and returns a number", () => {
const a = makeTarget("deepseek", "deepseek-chat");
const b = makeTarget("openai", "gpt-4o");
const hints = await generateRoutingHints([a, b], {
const hints = generateRoutingHints([a, b], {
messages: [{ content: "Test" }],
});
const result = compareByCostEffectiveness(a, b, hints);
expect(typeof result).toBe("number");
});
it("returns negative when a is cheaper than b", async () => {
it("returns negative when a is cheaper than b", () => {
const a = makeTarget("deepseek", "deepseek-chat");
const b = makeTarget("openai", "gpt-4o");
const hints = await generateRoutingHints([a, b], {
const hints = generateRoutingHints([a, b], {
messages: [{ content: "Test" }],
});
const result = compareByCostEffectiveness(a, b, hints);
@@ -115,16 +115,16 @@ describe("ManifestAdapter", () => {
});
describe("edge cases", () => {
it("handles empty targets array", async () => {
const hints = await generateRoutingHints([], {
it("handles empty targets array", () => {
const hints = generateRoutingHints([], {
messages: [{ content: "Hello" }],
});
expect(hints.eligibleTargets.length).toBe(0);
expect(hints.underqualifiedTargets.length).toBe(0);
});
it("returns valid hints structure with no targets", async () => {
const hints = await generateRoutingHints([], {
it("returns valid hints structure with no targets", () => {
const hints = generateRoutingHints([], {
messages: [{ content: "Test" }],
});
expect("specificityLevel" in hints).toBe(true);

View File

@@ -2,7 +2,7 @@
* Unit tests for Auto-Combo Engine (Phase 5)
*/
import { describe, it, expect, beforeEach } from "vitest";
import { describe, it, expect, beforeEach, vi } from "vitest";
import { calculateFactors, calculateScore, DEFAULT_WEIGHTS, validateWeights } from "../scoring";
import type { ProviderCandidate, ScoringWeights } from "../scoring";
import {
@@ -242,115 +242,6 @@ describe("Mode Packs", () => {
it("undefined pack should return undefined", () => {
expect(getModePack("nonexistent")).toBeUndefined();
});
it("every pack carries quality>0 and reliability>0 and sums to 0.9999", () => {
for (const name of getModePackNames()) {
const w = getModePack(name)!;
expect(Number(w.quality)).toBeGreaterThan(0);
expect(Number(w.reliability)).toBeGreaterThan(0);
const sum = Object.values(w).reduce((a, b) => a + Number(b), 0);
expect(sum).toBeCloseTo(0.9999, 3);
}
});
});
describe("Mode pack ranking gates (cold/warm/health)", () => {
function scoreOne(candidate: ProviderCandidate, pack: ScoringWeights): number {
const factors = calculateFactors(candidate, [candidate], "coding", () => 0.5);
return calculateScore(factors, pack);
}
it("cold pool ranking unchanged (reliability 1, quality 0.5 neutrals)", () => {
const a: ProviderCandidate = {
circuitBreakerState: "CLOSED",
failureRate: undefined,
quality: undefined,
quotaRemaining: 50,
quotaTotal: 100,
costPer1MTokens: 1,
p95LatencyMs: 100,
accountTier: "pro",
latencyStdDev: 10,
};
const b: ProviderCandidate = {
circuitBreakerState: "CLOSED",
failureRate: undefined,
quality: undefined,
quotaRemaining: 50,
quotaTotal: 100,
costPer1MTokens: 1,
p95LatencyMs: 100,
accountTier: "pro",
latencyStdDev: 10,
};
const pack = MODE_PACKS["reliability-first"];
expect(scoreOne(a, pack)).toBeCloseTo(scoreOne(b, pack), 5);
});
it("warm reliability 0.01 vs 0.4 flips winner at health tie", () => {
const highFail: ProviderCandidate = {
circuitBreakerState: "CLOSED",
failureRate: 0.4,
quality: 0.5,
quotaRemaining: 50,
quotaTotal: 100,
costPer1MTokens: 1,
p95LatencyMs: 100,
accountTier: "pro",
latencyStdDev: 10,
};
const lowFail: ProviderCandidate = {
circuitBreakerState: "CLOSED",
failureRate: 0.01,
quality: 0.5,
quotaRemaining: 50,
quotaTotal: 100,
costPer1MTokens: 1,
p95LatencyMs: 100,
accountTier: "pro",
latencyStdDev: 10,
};
const pack = MODE_PACKS["reliability-first"];
expect(scoreOne(lowFail, pack)).toBeGreaterThan(scoreOne(highFail, pack));
});
it("boundedRate NaN yields reliability 1", () => {
const c: ProviderCandidate = {
circuitBreakerState: "CLOSED",
failureRate: NaN,
quotaRemaining: 50,
quotaTotal: 100,
costPer1MTokens: 1,
p95LatencyMs: 100,
accountTier: "pro",
latencyStdDev: 10,
};
const factors = calculateFactors(c, [c], "coding", () => 0.5);
expect(factors.reliability).toBe(1);
});
it("health CLOSED vs HALF_OPEN still outweighs reliability gap", () => {
const healthyHighFail: ProviderCandidate = {
circuitBreakerState: "CLOSED",
failureRate: 0.4,
quality: 0.5,
quotaRemaining: 50,
quotaTotal: 100,
costPer1MTokens: 1,
p95LatencyMs: 100,
accountTier: "pro",
latencyStdDev: 10,
};
const halfOpenLowFail: ProviderCandidate = {
circuitBreakerState: "HALF_OPEN",
failureRate: 0.01,
quality: 0.5,
quotaRemaining: 50,
quotaTotal: 100,
costPer1MTokens: 1,
p95LatencyMs: 100,
accountTier: "pro",
latencyStdDev: 10,
};
const pack = MODE_PACKS["reliability-first"];
expect(scoreOne(healthyHighFail, pack)).toBeGreaterThan(scoreOne(halfOpenLowFail, pack));
});
});
describe("SLA-aware Strategy", () => {

View File

@@ -75,11 +75,11 @@ export function classifyRequestComplexity(input: RuleInput): ComplexityClassific
* failure — fail-open, so scoring stays tier-neutral. Extracted from combo.ts to
* keep the complexity-routing logic in one module.
*/
export async function buildComplexityRoutingHint(
export function buildComplexityRoutingHint(
modelTargets: Parameters<typeof generateRoutingHints>[0],
body: { messages?: unknown; tools?: unknown; model?: unknown } | null | undefined,
log: { info: (tag: string, message: string) => void }
): Promise<RoutingHint | null> {
): RoutingHint | null {
try {
const ruleInput = {
messages: Array.isArray(body?.messages)
@@ -92,7 +92,7 @@ export async function buildComplexityRoutingHint(
: undefined,
model: typeof body?.model === "string" ? body.model : undefined,
};
const hint = await generateRoutingHints(modelTargets, ruleInput);
const hint = generateRoutingHints(modelTargets, ruleInput);
// Tool-use escalation: floor the recommended tier at "cheap" so scoring
// favors function-calling-reliable models for agentic requests.
const classification = classifyRequestComplexity(ruleInput);

View File

@@ -39,7 +39,6 @@ export interface AutoComboConfig {
* silently overspending.
*/
budgetFallback?: "cheapest" | "strict";
estimatedInputTokens?: number; // tokens the budget is computed against (default 1000)
explorationRate: number; // 0.05 = 5% exploratory
/** If set, RouterStrategy name to use for selection ('rules' | 'cost' | 'latency') */
routerStrategy?: string;
@@ -312,13 +311,9 @@ export function selectProvider(
for (const c of candidates) {
costMap.set(`${c.provider}\0${c.model}`, c.costPer1MTokens);
}
const estimatedTokens =
Number.isFinite(config.estimatedInputTokens) && config.estimatedInputTokens! > 0
? config.estimatedInputTokens!
: 1000;
const estimatedCostFor = (s: ScoredProvider) => {
const cost = costMap.get(`${s.provider}\0${s.model}`) ?? 0;
return (cost / 1_000_000) * estimatedTokens;
return (cost / 1_000_000) * 1000;
};
if (estimatedCostFor(selected) > config.budgetCap) {
const budgetOk = candidates_.filter((s) => estimatedCostFor(s) <= config.budgetCap!);

View File

@@ -20,7 +20,6 @@ import {
type UsageFetcherProvider,
} from "./../usage.ts";
import { getCachedProviderConnections } from "@/lib/db/readCache";
import { providerHasFreeModels } from "@/shared/utils/freeModels";
import { defaultLogger as log } from "@omniroute/open-sse/utils/logger";
import type { FreeAccessState } from "./strictZeroCostFilter";
import { isStateStaleForReset } from "./subscriptionLadder";
@@ -101,20 +100,16 @@ function sweepIfDue(): void {
* `quota_omniroute.py`'s own parsing) and returns `null` — never a guess —
* for anything else. `null` is treated as "not proven safe" by the filter.
*/
function extractRemainingAllowance(usage: unknown, opts?: { isFreeTier?: boolean }): number | null {
function extractRemainingAllowance(usage: unknown): number | null {
if (!usage || typeof usage !== "object") return null;
const quotas = (usage as Record<string, unknown>).quotas;
if (!quotas || typeof quotas !== "object") return null;
let worstPercent: number | null = null;
let sawUnlimited = false;
for (const raw of Object.values(quotas as Record<string, unknown>)) {
if (!raw || typeof raw !== "object") continue;
const q = raw as Record<string, unknown>;
if (q.unlimited === true) {
sawUnlimited = true;
continue;
}
if (q.unlimited === true) continue;
let pct: number | null =
typeof q.remainingPercentage === "number" ? q.remainingPercentage : null;
if (
@@ -128,7 +123,6 @@ function extractRemainingAllowance(usage: unknown, opts?: { isFreeTier?: boolean
if (pct === null) continue;
worstPercent = worstPercent === null ? pct : Math.min(worstPercent, pct);
}
if (worstPercent === null && sawUnlimited && opts?.isFreeTier === true) return 100;
return worstPercent; // percentage points; the filter's threshold is compared against this unit
}
@@ -154,9 +148,7 @@ async function refresh(provider: string, connectionId: string): Promise<void> {
connection as unknown as Parameters<typeof getUsageForProvider>[0],
{ forceRefresh: false }
);
const remaining = extractRemainingAllowance(usage, {
isFreeTier: providerHasFreeModels(provider),
});
const remaining = extractRemainingAllowance(usage);
const state: FreeAccessState = {
status: remaining === null ? "UNKNOWN" : remaining > 0 ? "SAFE" : "EXHAUSTED",
remainingFreeAllowance: remaining,

View File

@@ -4,10 +4,8 @@
* Each pack optimizes for a different priority:
* - ship-fast: Prioritize latency and health
* - cost-saver: Prioritize cost efficiency
* - quality-first: Prioritize task fitness and stability (highest quality weight, 0.03)
* - quality-first: Prioritize task fitness and stability
* - offline-friendly: Prioritize quota availability
* - reliability-first: Prioritize health+stability (highest reliability weight, 0.04)
* - chaos-mode: Fault-injection — health > stability > taskFit
*/
import type { ScoringWeights } from "./scoring";
@@ -16,95 +14,85 @@ export const MODE_PACKS: Record<string, ScoringWeights> = {
// Prioritize latency → health. tierPriority replaces 0.05 from stability.
// tierAffinity/specificityMatch stay at 0 (manifest-routing-only weights).
"ship-fast": {
quota: 0.1133,
quota: 0.1333,
health: 0.2667,
costInv: 0.0276,
costInv: 0.0476,
latencyInv: 0.3048,
taskFit: 0.0952,
stability: 0,
tierPriority: 0.0376,
tierPriority: 0.0476,
tierAffinity: 0,
specificityMatch: 0,
contextAffinity: 0.0095,
sessionAvailability: 0.0476,
resetWindowAffinity: 0,
connectionDensity: 0.0476,
quality: 0.02,
reliability: 0.03,
},
// Prioritize cost. tierPriority replaces 0.05 from stability.
"cost-saver": {
quota: 0.1133,
quota: 0.1333,
health: 0.181,
costInv: 0.3324,
costInv: 0.3524,
latencyInv: 0.0476,
taskFit: 0.0952,
stability: 0.0476,
tierPriority: 0.0376,
tierPriority: 0.0476,
tierAffinity: 0,
specificityMatch: 0,
contextAffinity: 0,
sessionAvailability: 0.0476,
resetWindowAffinity: 0,
connectionDensity: 0.0476,
quality: 0.02,
reliability: 0.03,
},
// Prioritize task fitness. tierPriority replaces 0.05 from latencyInv.
"quality-first": {
quota: 0.0752,
quota: 0.0952,
health: 0.1714,
costInv: 0.0276,
costInv: 0.0476,
latencyInv: 0.0476,
taskFit: 0.3524,
stability: 0.1429,
tierPriority: 0.0276,
tierPriority: 0.0476,
tierAffinity: 0,
specificityMatch: 0,
contextAffinity: 0,
sessionAvailability: 0.0476,
resetWindowAffinity: 0,
connectionDensity: 0.0476,
quality: 0.03,
reliability: 0.03,
},
// Prioritize quota availability. tierPriority replaces 0.05 from taskFit.
"offline-friendly": {
quota: 0.3324,
quota: 0.3524,
health: 0.2667,
costInv: 0.0752,
costInv: 0.0952,
latencyInv: 0.0476,
taskFit: 0,
stability: 0.0952,
tierPriority: 0.0376,
tierPriority: 0.0476,
tierAffinity: 0,
specificityMatch: 0,
contextAffinity: 0,
sessionAvailability: 0.0476,
resetWindowAffinity: 0,
connectionDensity: 0.0476,
quality: 0.02,
reliability: 0.03,
},
// #4235 `:reliable` — prioritize healthy, low-variance providers (high availability).
// health (circuit-breaker) + stability (latency std-dev) dominate; weights sum to ~1.0
// (re-normalized after #8940 added sessionAvailability without rebalancing — #9985).
"reliability-first": {
quota: 0.1133,
quota: 0.1333,
health: 0.3524,
costInv: 0.0181,
costInv: 0.0381,
latencyInv: 0.0476,
taskFit: 0.0952,
stability: 0.1905,
tierPriority: 0.0276,
tierPriority: 0.0476,
tierAffinity: 0,
specificityMatch: 0,
contextAffinity: 0,
sessionAvailability: 0.0476,
resetWindowAffinity: 0,
connectionDensity: 0.0476,
quality: 0.02,
reliability: 0.04,
},
// Chaos mode — priority: health > stability > taskFit > latency.
// Selects top-N healthy providers for parallel dispatch. Favors providers with
@@ -113,21 +101,19 @@ export const MODE_PACKS: Record<string, ScoringWeights> = {
// to picking the most stable providers); connectionDensity boosted slightly to
// prefer providers with multiple accounts (more resilient to per-account rate limits).
"chaos-mode": {
quota: 0.0376,
quota: 0.0476,
health: 0.4,
costInv: 0.014,
latencyInv: 0.0186,
costInv: 0.019,
latencyInv: 0.0286,
taskFit: 0.1905,
stability: 0.1714,
tierPriority: 0.004,
tierPriority: 0.019,
tierAffinity: 0,
specificityMatch: 0,
contextAffinity: 0.0186,
contextAffinity: 0.0286,
sessionAvailability: 0.0476,
resetWindowAffinity: 0,
connectionDensity: 0.0476,
quality: 0.02,
reliability: 0.03,
},
};

View File

@@ -1,72 +0,0 @@
/**
* Filter candidates whose model is locked on every usable connection.
*
* A candidate is kept when at least one of its connection ids is not locked;
* when only some are locked the candidate is kept with `allowedConnectionIds`
* rewritten to the unlocked subset (same rewrite the STRICT filter applies).
* Candidates without any usable connection id are kept untouched — the lockout
* check needs a real id to decide anything.
*/
import { isModelLocked } from "../accountFallback";
export type LockoutDiagnosis = { excludedLockout: number; total: number };
export interface LockoutCandidate {
provider: string;
model: string;
connectionId?: string | null;
allowedConnectionIds?: string[];
}
export function filterLockoutCandidates<T extends LockoutCandidate>(
pool: T[],
deps: { isModelLocked: (provider: string, connectionId: string, model: string) => boolean } = {
isModelLocked,
}
): { pool: T[]; diagnosis: LockoutDiagnosis | null } {
if (pool.length === 0) return { pool, diagnosis: null };
let excluded = 0;
const kept: T[] = [];
for (const candidate of pool) {
const ids = candidate.connectionId
? [candidate.connectionId]
: (candidate.allowedConnectionIds ?? []);
if (ids.length === 0) {
kept.push(candidate);
continue;
}
let alive: string[];
try {
alive = ids.filter((id) => {
try {
return !deps.isModelLocked(candidate.provider, id, candidate.model);
} catch {
return true;
}
});
} catch {
alive = ids;
}
if (alive.length === 0) {
excluded++;
continue;
}
if (alive.length !== ids.length) kept.push({ ...candidate, allowedConnectionIds: alive } as T);
else kept.push(candidate);
}
return {
pool: kept,
diagnosis: excluded > 0 ? { excludedLockout: excluded, total: pool.length } : null,
};
}
/** One AUTO warning when a pool stage removed candidates; silent when nothing was dropped. */
export function warnPoolDrop(
log: { warn: (tag: string, message: string) => void },
stage: string,
excluded: number | undefined,
total: number,
detail = ""
): void {
if (excluded) log.warn("AUTO", `${stage} excluded ${excluded}/${total}${detail}`);
}

View File

@@ -11,7 +11,7 @@
* Kept as a pure, dependency-light function so the filter is unit-testable in
* isolation without seeding the DB-backed virtual factory.
*/
import { isFreeForProvider } from "@/shared/utils/freeModels";
import { isFreeModel, providerHasFreeModels } from "@/shared/utils/freeModels";
interface PaidFilterCandidate {
provider: string;
@@ -22,7 +22,10 @@ interface PaidFilterCandidate {
* selected model itself qualifies as free — mirrors `shouldHidePaid` in
* `src/app/api/v1/models/catalog.ts`. */
function isFreeCandidate(candidate: PaidFilterCandidate): boolean {
return isFreeForProvider(candidate.provider, { id: candidate.model });
return (
providerHasFreeModels(candidate.provider) &&
isFreeModel(candidate.provider, { id: candidate.model })
);
}
/**
@@ -32,20 +35,10 @@ function isFreeCandidate(candidate: PaidFilterCandidate): boolean {
* the caller's existing graceful empty-pool path handles it (consistent with the
* opt-in intent — the operator asked not to route to paid models).
*/
export type PaidFilterDiagnosis = { excludedPaid: number; total: number };
export function filterPaidOnlyCandidatesWithDiagnosis<T extends PaidFilterCandidate>(
pool: T[],
hidePaidModels: boolean
): { pool: T[]; diagnosis: PaidFilterDiagnosis | null } {
if (!hidePaidModels) return { pool, diagnosis: null };
const kept = pool.filter(isFreeCandidate);
return { pool: kept, diagnosis: { excludedPaid: pool.length - kept.length, total: pool.length } };
}
export function filterPaidOnlyCandidates<T extends PaidFilterCandidate>(
pool: T[],
hidePaidModels: boolean
): T[] {
return filterPaidOnlyCandidatesWithDiagnosis(pool, hidePaidModels).pool;
if (!hidePaidModels) return pool;
return pool.filter(isFreeCandidate);
}

View File

@@ -78,12 +78,11 @@ export const DEFAULT_WEIGHTS: ScoringWeights = {
// the new quality signal (observed output quality over time) gets a real,
// if smaller, vote. Sum remains exactly 1.0.
quality: 0.03,
// Declared but silent in DEFAULT (like `cacheAffinity`/`resetWindowAffinity`):
// every candidate already carries a measured failure rate (24h of usage
// history behind a ten-sample floor, real-time metrics otherwise) and the
// scorer had no way to read it. Which weight it deserves is a product call
// backed by measurement, so DEFAULT ships at 0 and leaves the ranking as
// it was; `reliability-first` ships at 0.04 and generic packs at 0.03.
// Declared but silent, like `cacheAffinity` and `resetWindowAffinity`: every
// candidate already carries a measured failure rate (24h of usage history
// behind a ten-sample floor, real-time metrics otherwise) and the scorer had
// no way to read it. Which weight it deserves is a product call backed by
// measurement, so this ships at 0 and leaves the ranking exactly as it was.
reliability: 0,
};
@@ -208,19 +207,6 @@ export function calculateTierScore(
return Math.min(1, baseScore * 0.8 + resetBonus * 0.2);
}
/**
* Project the account tier from a provider connection row.
* Tier whitelist in ONE place: test and combo.ts import this, never mirror it.
*/
export function projectAccountTier(
row: Record<string, unknown> | undefined
): "ultra" | "pro" | "standard" | "free" | undefined {
const psd = row?.providerSpecificData as Record<string, unknown> | undefined;
const raw = row?.accountTier ?? psd?.accountTier;
const v = typeof raw === "string" ? raw.toLowerCase() : undefined;
return v === "ultra" || v === "pro" || v === "standard" || v === "free" ? v : undefined;
}
function calculateTierAffinity(
candidate: ProviderCandidate,
hint: RoutingHint | undefined | null
@@ -293,14 +279,6 @@ export function computePoolMaxima(pool: ProviderCandidate[]): PoolMaxima {
* `speedRanking.ts` so both consumers of the same signal agree, including on
* garbage input.
*/
/** Reliability factor from an observed failure (else error) rate; absent reads fully reliable. */
export function reliabilityFactor(candidate: {
failureRate?: number | null;
errorRate?: number | null;
}): number {
return clamp01(1 - boundedRate(candidate.failureRate ?? candidate.errorRate));
}
function boundedRate(value: number | null | undefined): number {
if (typeof value !== "number" || !Number.isFinite(value) || value < 0) return 0;
return Math.min(1, value);
@@ -349,7 +327,7 @@ export function calculateFactors(
// bounded BEFORE the subtraction, exactly as `toBoundedRate` does there --
// `clamp01(1 - NaN)` would be 0, i.e. "fails every call", which is the
// opposite of what corrupt telemetry should mean.
reliability: reliabilityFactor(candidate),
reliability: clamp01(1 - boundedRate(candidate.failureRate ?? candidate.errorRate)),
};
}

View File

@@ -1,31 +0,0 @@
import { getCircuitBreaker } from "../../../src/shared/utils/circuitBreaker.ts";
export type BreakerState = "CLOSED" | "HALF_OPEN" | "OPEN";
/**
* Build-time breaker state per provider. Any other state (e.g. DEGRADED) and a
* failed lookup leave the provider absent, so it scores neutral.
*/
export function readBreakerStates(providers: Iterable<string>): Map<string, BreakerState> {
const states = new Map<string, BreakerState>();
for (const provider of providers) {
try {
const state = getCircuitBreaker(provider).getStatus().state as string;
if (state === "OPEN" || state === "HALF_OPEN" || state === "CLOSED") {
states.set(provider, state);
}
} catch {
// An unreadable breaker leaves the provider neutral.
}
}
return states;
}
/** Snapshot health factor: OPEN 0, CLOSED 1, HALF_OPEN or unknown 0.5 (neutral). */
export function snapshotHealthFactor(
states: ReadonlyMap<string, BreakerState> | undefined,
provider: string
): number {
const state = states?.get(provider);
return state === "OPEN" ? 0 : state === "CLOSED" ? 1 : 0.5;
}

View File

@@ -327,31 +327,6 @@ export function filterStrictZeroCostCandidates<T extends StrictZeroCostCandidate
return changed ? kept : pool;
}
/**
* How many candidates the STRICT filter drops outright (the same verdict it keeps on),
* and how many of those only for lack of a hard-stop guarantee.
*/
export function countStrictExclusions<T extends StrictZeroCostCandidate>(
pool: T[],
options: StrictZeroCostOptions
): { excluded: number; noHardStop: number } {
let excluded = 0;
let noHardStop = 0;
for (const candidate of pool) {
const budgetEntry = findBudgetEntry(candidate, options.catalog);
const verdict = classifyStrictZeroCostCandidate(
candidate,
budgetEntry,
options.resolveFreeAccessState,
options
);
if (verdict.outcome === "safe") continue;
excluded++;
if (verdict.outcome === "no-hard-stop") noHardStop++;
}
return { excluded, noHardStop };
}
/**
* Separate, optional ToS guard — kept independent from economic safety on
* purpose (Marco's requirement): a model can be economically SAFE and still

View File

@@ -1,6 +1,6 @@
import { AutoComboConfig } from "./engine";
import { MODE_PACKS } from "./modePacks";
import { DEFAULT_WEIGHTS, reliabilityFactor, ScoringWeights } from "./scoring";
import { DEFAULT_WEIGHTS, ScoringWeights } from "./scoring";
import { getCachedProviderConnections } from "@/lib/db/readCache";
import { getSettings } from "@/lib/db/settings";
import { getProviderRegistry } from "./providerRegistryAccessor";
@@ -26,15 +26,11 @@ import {
type AutoTier,
} from "./suffixComposition";
import { classifyTier } from "../tierResolver";
import { getQualityScore } from "../routing/quality.ts";
import { readBreakerStates, snapshotHealthFactor, type BreakerState } from "./snapshotBreaker.ts";
import { resolveVirtualCost } from "../providerCostData";
import type { AutoVariant } from "./autoPrefix";
import { buildFamilyCandidateFilter, type ModelFamily } from "./modelFamily";
import { getHiddenModelsByProvider } from "@/models";
import { getSyncedAvailableModelsByConnection, getCustomModels } from "@/lib/db/models";
import { filterPaidOnlyCandidatesWithDiagnosis } from "./paidModelFilter";
import { filterLockoutCandidates, warnPoolDrop } from "./modelLockoutFilter";
import { filterPaidOnlyCandidates } from "./paidModelFilter";
import { filterModelExposureCandidates } from "./modelExposureFilter";
import {
filterSubscriptionOnlyCandidates,
@@ -43,7 +39,6 @@ import {
} from "./subscriptionLadder";
import {
classifyStrictZeroCostCandidate,
countStrictExclusions,
filterStrictZeroCostCandidates,
filterTosAvoidCandidates,
findBudgetEntry,
@@ -110,16 +105,12 @@ export interface VirtualAutoComboCandidate {
model: string;
modelStr: string; // e.g., 'openai/gpt-4o'
costPer1MTokens: number; // from providerRegistry
/** Observed failure rate 0..1 when known; null/absent reads fully reliable. */
failureRate?: number | null;
/** Build-local capability snapshot. Runtime calls rebuild it; catalog entries reuse it. */
resolvedContextLength?: number | null;
resolvedMaxOutputTokens?: number | null;
resolvedSupportsVision?: boolean;
resolvedReasoning?: boolean;
resolvedSupportsThinking?: boolean;
/** Observed feedback quality 0..1 (the same signal as ProviderCandidate.quality). */
quality?: number | null;
/**
* Why STRICT_ZERO_COST would exclude this candidate, or null when it would
* not. Only populated for the read-only inspector build (`skip`), where the
@@ -449,7 +440,7 @@ function getNoAuthCandidates(
connectionId: SYNTHETIC_NOAUTH_CONNECTION_ID,
model: modelId,
modelStr: `${routingPrefix}/${modelId}`,
costPer1MTokens: resolveVirtualCost(providerId, modelId),
costPer1MTokens: 0,
});
}
}
@@ -553,7 +544,7 @@ function yieldVirtualAutoPreparationTurn(): Promise<void> {
return new Promise((resolve) => setImmediate(resolve));
}
export async function attachPreparedCapabilityValues(
async function attachPreparedCapabilityValues(
candidates: readonly VirtualAutoComboCandidate[],
state: PreparedCapabilityState
): Promise<VirtualAutoComboCandidate[]> {
@@ -600,11 +591,7 @@ export async function attachPreparedCapabilityValues(
await yieldVirtualAutoPreparationTurn();
}
}
prepared.push({
...candidate,
...values,
quality: getQualityScore(candidate.provider, candidate.model),
});
prepared.push({ ...candidate, ...values });
}
return prepared;
}
@@ -731,7 +718,7 @@ export async function prepareVirtualAutoComboInputs(
allowedConnectionIds,
model: modelId,
modelStr: `${providerId}/${modelId}`,
costPer1MTokens: resolveVirtualCost(providerId, modelId),
costPer1MTokens: 0, // Not used in virtual auto-combo (LKGP uses session stickiness)
});
}
}
@@ -762,13 +749,8 @@ export async function prepareVirtualAutoComboInputs(
// #6512 (follow-up to #6328/#6495): when the operator opts into `hidePaidModels`,
// exclude paid-only backends from EVERY `auto/*` candidate pool.
const paid = filterPaidOnlyCandidatesWithDiagnosis(pool, settings.hidePaidModels === true);
warnPoolDrop(log, "hidePaidModels", paid.diagnosis?.excludedPaid, pool.length);
pool = paid.pool;
const lockout = skip ? null : filterLockoutCandidates(pool); // dispatch only (#9133)
warnPoolDrop(log, "lockout", lockout?.diagnosis?.excludedLockout, pool.length);
if (lockout) pool = lockout.pool;
const paidFilteredPool = filterPaidOnlyCandidates(pool, settings.hidePaidModels === true);
if (paidFilteredPool !== pool) pool = paidFilteredPool;
// #11481: mandatory mirror of the /v1/models exposure allow/deny list —
// see src/shared/utils/modelExposureList.ts for why (#6512's lesson).
@@ -790,20 +772,15 @@ export async function prepareVirtualAutoComboInputs(
maxStateAgeMs: toNumber(settings.autoRefreshProviderQuotaInterval, 180) * 1000,
};
const strictZeroCostOn = settings.freeAccessPolicy === "strict";
const strictOptions = {
const strictFilteredPool = filterStrictZeroCostCandidates(pool, {
// The read-only candidate inspector (#9133) must be able to see what the
// guard would exclude, and why — the same opt-out the resilience filter
// already honours through `skip`. Dispatch (`skip === false`) is unaffected.
enabled: strictZeroCostOn && !skip,
resolveFreeAccessState,
...strictZeroCostThresholds,
};
const strictFilteredPool = filterStrictZeroCostCandidates(pool, strictOptions);
if (strictFilteredPool !== pool) {
const s = countStrictExclusions(pool, strictOptions);
warnPoolDrop(log, "STRICT", s.excluded, pool.length, ` (no-hard-stop ${s.noHardStop})`);
pool = strictFilteredPool;
}
});
if (strictFilteredPool !== pool) pool = strictFilteredPool;
// Annotate here rather than in the handler: this is where the thresholds and
// `resolveFreeAccessState` already live. Doing it downstream would mean a second
@@ -869,8 +846,7 @@ export async function prepareVirtualAutoComboInputs(
*/
export function computeSnapshotWeights(
candidates: readonly VirtualAutoComboCandidate[],
weights: ScoringWeights,
breakerByProvider?: ReadonlyMap<string, BreakerState>
weights: ScoringWeights
): Map<string, number> {
const scores = new Map<string, number>();
for (const c of candidates) {
@@ -908,14 +884,8 @@ export function computeSnapshotWeights(
// (no runtime data at snapshot time, so equal baseline)
if (weights.latencyInv > 0) score += weights.latencyInv * 0.5;
// reliability (#12792): the same failure-rate factor as scoring.ts; absent reads as
// fully reliable. The snapshot path was the only one still ignoring it.
if (weights.reliability > 0) score += weights.reliability * reliabilityFactor(c);
// health: build-time breaker state (OPEN 0, CLOSED 1, else neutral 0.5); quota stays neutral
score += weights.health * snapshotHealthFactor(breakerByProvider, c.provider);
score += weights.quota * 0.5;
score += (weights.quality ?? 0) * (Number.isFinite(c.quality) ? Number(c.quality) : 0.5);
// health + quota: no runtime telemetry at snapshot time → neutral baseline
score += (weights.health + weights.quota) * 0.5;
scores.set(c.modelStr, Math.min(score, 1));
}
@@ -1116,8 +1086,7 @@ export async function createVirtualAutoComboFromPrepared(
}
const providerPool = [...new Set(effectivePool.map((c) => c.provider))];
const breakerStates = readBreakerStates(providerPool);
const snapshotScores = computeSnapshotWeights(effectivePool, weights, breakerStates);
const snapshotScores = computeSnapshotWeights(effectivePool, weights);
const models = effectivePool.map((candidate, index) => ({
id: `virtual-auto-${variant || "default"}-${index + 1}-${candidate.provider}`,
kind: "model" as const,

View File

@@ -4,7 +4,6 @@ import {
CODEX_SPARK_QUOTA_WEEKLY,
isCodexSparkLimitDescriptor,
} from "../config/codexQuotaScopes.ts";
import { inferWindowFamilyLabel } from "./quotaWindowLabel.ts";
type JsonRecord = Record<string, unknown>;
@@ -111,7 +110,6 @@ function buildPercentageQuota(window: JsonRecord, displayName?: string): CodexUs
// duration instead of assuming primary=session / secondary=weekly by position.
const WEEKLY_MIN_WINDOW_SECONDS = 6 * 24 * 3600; // >= ~6d
const SESSION_MAX_WINDOW_SECONDS = 6 * 3600; // <= ~6h
const MONTHLY_MIN_WINDOW_SECONDS = 20 * 24 * 3600; // >= ~20d (ChatGPT 30d plans)
/**
* A never-started window: `used_percent === 0` and the reset still spans the
@@ -135,15 +133,12 @@ function isLatentWindow(window: JsonRecord): boolean {
* e.g. a 7-day `primary_window` is labeled "Weekly" rather than "Session".
* Returns undefined for durations that don't clearly map to either bucket.
*/
function windowDurationLabel(window: JsonRecord): "Session" | "Weekly" | "Monthly" | undefined {
function windowDurationLabel(window: JsonRecord): "Session" | "Weekly" | undefined {
const limitWindow = toNumber(
getFieldValue(window, "limit_window_seconds", "limitWindowSeconds"),
0
);
if (limitWindow <= 0) return undefined;
const inferred = inferWindowFamilyLabel(limitWindow);
if (inferred === "Monthly" || inferred === "Weekly" || inferred === "Session") return inferred;
if (limitWindow >= MONTHLY_MIN_WINDOW_SECONDS) return "Monthly";
if (limitWindow >= WEEKLY_MIN_WINDOW_SECONDS) return "Weekly";
if (limitWindow <= SESSION_MAX_WINDOW_SECONDS) return "Session";
return undefined;
@@ -285,7 +280,7 @@ export function buildCodexUsageQuotas(dataValue: unknown): {
const primaryLabel = windowDurationLabel(primaryWindow);
quotas.session = buildPercentageQuota(
primaryWindow,
primaryLabel === "Weekly" || primaryLabel === "Monthly" ? primaryLabel : undefined
primaryLabel === "Weekly" ? primaryLabel : undefined
);
}

View File

@@ -33,7 +33,7 @@ import { rejectRetiredAutoComboCandidates } from "./modelLifecycle.ts";
import { createComboContext } from "./combo/context.ts";
import { phaseComboSetup } from "./combo/comboSetup.ts";
import { projectAccountTier, type ProviderCandidate } from "./autoCombo/scoring.ts";
import { type ProviderCandidate } from "./autoCombo/scoring.ts";
import { getSessionConnection } from "./sessionManager.ts";
import { getOAuthSessionAvailability } from "./oauthSessionOccupancy.ts";
@@ -251,63 +251,6 @@ function getBootstrapLatencyMs(modelId: string): number {
return DEFAULT_MODEL_P95_MS[normalized] ?? 1500;
}
export function poolMedianP95Ms(
stats: Record<string, { p95LatencyMs?: unknown }>
): number | undefined {
const vals = Object.values(stats)
.map((st) => Number(st?.p95LatencyMs))
.filter((v) => Number.isFinite(v) && v > 0)
.sort((a, b) => a - b);
return vals.length ? vals[(vals.length - 1) >> 1] : undefined;
}
const BOOTSTRAP_WARN_WINDOW_MS = 3600_000;
export let bootstrapLatencyHits = 0; // exported for testability (reset in tests)
export let bootstrapLatencyTotal = 0;
let bootstrapWarnedAt = 0;
export function resetBootstrapCounters(): void {
bootstrapLatencyHits = 0;
bootstrapLatencyTotal = 0;
bootstrapWarnedAt = 0;
}
export function bootstrapMs(model: string, poolMedian: number | undefined): number {
bootstrapLatencyTotal++;
const table = DEFAULT_MODEL_P95_MS[String(model || "").toLowerCase()];
if (table !== undefined) return table;
bootstrapLatencyHits++;
return poolMedian ?? 1500;
}
// Pure and testable without timers: the throttled 1h warn + cold-start exemption live here.
export function shouldWarnBootstrap(
hits: number,
total: number,
hasStats: boolean,
now: number,
lastWarn: number
): boolean {
if (!hasStats || total === 0) return false;
if (hits / total <= 0.3) return false;
return now - lastWarn >= BOOTSTRAP_WARN_WINDOW_MS;
}
function maybeWarnBootstrapDominant(hasStats: boolean): void {
if (
!shouldWarnBootstrap(
bootstrapLatencyHits,
bootstrapLatencyTotal,
hasStats,
Date.now(),
bootstrapWarnedAt
)
)
return;
bootstrapWarnedAt = Date.now();
console.warn(
`[combo] bootstrap latency dominant (${bootstrapLatencyHits}/${bootstrapLatencyTotal}) — scoring runs on guesses`
);
}
export async function buildAutoCandidates(
targets: ResolvedComboTarget[],
comboName: string,
@@ -335,8 +278,6 @@ export async function buildAutoCandidates(
} catch {
// keep empty stats — auto-combo will use runtime + bootstrap signals
}
const poolMedian = poolMedianP95Ms(historicalLatencyStats);
const hasStats = Object.keys(historicalLatencyStats).length > 0;
const uniqueProviders = Array.from(
new Set(
@@ -423,10 +364,10 @@ export async function buildAutoCandidates(
const p95LatencyMs = hasHistoricalSignal
? Number.isFinite(historicalP95Latency) && historicalP95Latency > 0
? historicalP95Latency
: bootstrapMs(model, poolMedian)
: getBootstrapLatencyMs(model)
: Number.isFinite(avgLatency) && avgLatency > 0
? avgLatency
: bootstrapMs(model, poolMedian);
: getBootstrapLatencyMs(model);
const errorRate = hasHistoricalSignal
? Number.isFinite(historicalSuccessRate) &&
@@ -547,15 +488,8 @@ export async function buildAutoCandidates(
latencyStdDev,
errorRate,
...speedTelemetry,
accountTier: projectAccountTier(connection as Record<string, unknown> | undefined),
quotaResetIntervalSecs: (() => {
const tierConn = connection as Record<string, unknown> | undefined;
const tierPsd = tierConn?.providerSpecificData as Record<string, unknown> | undefined;
const rawInterval = tierConn?.quotaResetIntervalSecs ?? tierPsd?.quotaResetIntervalSecs;
return typeof rawInterval === "number" && Number.isFinite(rawInterval) && rawInterval > 0
? rawInterval
: 86400;
})(),
accountTier: "standard" as const,
quotaResetIntervalSecs: 86400,
contextAffinity,
sessionAvailability,
resetWindowAffinity,
@@ -575,7 +509,6 @@ export async function buildAutoCandidates(
// Filter out candidates whose model is hidden by the user in the dashboard,
// then drop vendor-retired ids so auto-combo cannot pick them (#11625).
maybeWarnBootstrapDominant(hasStats);
return rejectRetiredAutoComboCandidates(
candidates.filter((c) => {
const hiddenModels = hiddenModelsMap.get(c.provider);

View File

@@ -157,7 +157,7 @@ export async function applyStrategyOrdering(
orderedTargets = await sortTargetsByCost(orderedTargets);
if (config.manifestRouting === true) {
try {
const manifestHint = await generateRoutingHints(
const manifestHint = generateRoutingHints(
orderedTargets.filter((t) => t.kind === "model"),
{
messages: Array.isArray(body?.messages)

View File

@@ -14,7 +14,6 @@ import type { ComboDiagnostics } from "../../utils/error.ts";
import { COMBO_FAILURE_THRESHOLD, recordComboFailure } from "./failureTracker.ts";
import { buildNoUpstreamResponseDiagnostics, buildRecoveryHint } from "./pinRecovery.ts";
import { formatExhaustedConnectionKey } from "./comboDiagFormat.ts";
import { collectQuotaWindowExclusions, formatQuotaSkipMessage } from "./quotaSkipDiagnostics.ts";
import { recordComboRequest } from "../comboMetrics.ts";
import { notifyWebhookEvent } from "../../../src/lib/webhookDispatcher.ts";
import { parseModel } from "../model.ts";
@@ -126,9 +125,6 @@ export async function dispatchWithCooldownRetry(opts: {
excluded: [
...[...state.exhaustedProviders].map((p) => ({ provider: p, reason: "exhausted" })),
...[...state.exhaustedConnections].map((c) => formatExhaustedConnectionKey(String(c))),
...(terminalReason === "all_targets_skipped"
? collectQuotaWindowExclusions(state.orderedTargets)
: []),
],
attemptOrder: state.comboAttemptOrder,
terminalReason,
@@ -412,15 +408,10 @@ export async function dispatchWithCooldownRetry(opts: {
latencyMs,
fallbackCount: state.fallbackCount,
});
const quotaSkip = formatQuotaSkipMessage(
collectQuotaWindowExclusions(state.orderedTargets)
);
return withQuotaExhaustionClassification(
errorResponseWithComboDiagnostics(
503,
quotaSkip
? `Service temporarily unavailable: all targets were skipped by pre-dispatch filters (${quotaSkip})`
: "Service temporarily unavailable: all targets were skipped by pre-dispatch filters",
"Service temporarily unavailable: all targets were skipped by pre-dispatch filters",
buildComboDiag("all_targets_skipped"),
{ code: "ALL_TARGETS_SKIPPED", type: "service_unavailable" }
),

View File

@@ -1,75 +0,0 @@
/**
* Redacted-safe quota-window skip rows for ALL_TARGETS_SKIPPED 503 bodies.
* Connection ids are prefix-only (8 chars). Labels match AUTH / usage API.
*/
import { getQuotaCache } from "@/domain/quotaCache";
import { formatQuotaWindowLabel } from "../quotaWindowLabel.ts";
export type QuotaSkipTarget = {
provider?: string;
model?: string;
modelStr?: string;
// `string | null`, not `string | undefined`: ResolvedComboTarget carries null for an
// unpinned target, and this module only reads the field. Line 29 already narrows with
// `typeof === "string"`, so null costs nothing here (TS2345 under typecheck:core).
connectionId?: string | null;
};
export type QuotaSkipExclusion = {
provider: string;
model?: string;
reason: string;
};
const EXHAUSTED_USED_PERCENT = 99;
export function collectQuotaWindowExclusions(targets: QuotaSkipTarget[]): QuotaSkipExclusion[] {
const seen = new Set<string>();
const out: QuotaSkipExclusion[] = [];
for (const target of targets) {
const connectionId = typeof target.connectionId === "string" ? target.connectionId : "";
if (!connectionId) continue;
const entry = getQuotaCache(connectionId);
if (!entry?.quotas) continue;
const provider =
typeof target.provider === "string" && target.provider ? target.provider : "unknown";
const model =
(typeof target.modelStr === "string" && target.modelStr) ||
(typeof target.model === "string" && target.model) ||
undefined;
const connPrefix = connectionId.slice(0, 8);
for (const [key, quota] of Object.entries(entry.quotas)) {
const remaining = quota.remainingPercentage;
if (typeof remaining !== "number" || !Number.isFinite(remaining)) continue;
const used = Math.max(0, Math.min(100, 100 - remaining));
if (used < EXHAUSTED_USED_PERCENT) continue;
const label = formatQuotaWindowLabel({
key,
displayName: quota.displayName,
windowSeconds: quota.windowSeconds,
});
const reason = `quota:${label} ${Math.round(used)}% conn:${connPrefix}`;
const dedupe = `${provider}|${model ?? ""}|${reason}`;
if (seen.has(dedupe)) continue;
seen.add(dedupe);
out.push({
provider,
...(model ? { model } : {}),
reason,
});
}
}
return out;
}
export function formatQuotaSkipMessage(exclusions: QuotaSkipExclusion[]): string | null {
if (exclusions.length === 0) return null;
const windows = [...new Set(exclusions.map((row) => row.reason.replace(/^quota:/, "")))];
return windows.slice(0, 4).join("; ");
}

View File

@@ -315,7 +315,7 @@ export async function resolveAutoStrategyOrder(
// specificityMatch favor candidates whose tier matches the request.
const autoManifestHint: RoutingHint | null =
config.complexityAwareRouting === true
? await buildComplexityRoutingHint(
? buildComplexityRoutingHint(
eligibleTargets.filter((t) => t.kind === "model"),
body,
log
@@ -408,7 +408,6 @@ export async function resolveAutoStrategyOrder(
modePack,
budgetCap,
budgetFallback,
estimatedInputTokens,
explorationRate,
},
routableCandidates,

View File

@@ -15,7 +15,6 @@ import {
} from "../../utils/error.ts";
import { buildRecoveryHint } from "./pinRecovery.ts";
import { formatExhaustedConnectionKey } from "./comboDiagFormat.ts";
import { collectQuotaWindowExclusions, formatQuotaSkipMessage } from "./quotaSkipDiagnostics.ts";
import { recordComboRequest } from "../comboMetrics.ts";
import {
expandComboSystemPromptIfPresent,
@@ -1153,21 +1152,16 @@ export async function handleRoundRobinCombo({
if (!lastStatus) {
if (recordedAttempts === 0) {
const quotaExcluded = collectQuotaWindowExclusions(filteredTargets);
const quotaSkip = formatQuotaSkipMessage(quotaExcluded);
return errorResponseWithComboDiagnostics(
503,
quotaSkip
? `Service temporarily unavailable: all targets were skipped by pre-dispatch filters (${quotaSkip})`
: "Service temporarily unavailable: all targets were skipped by pre-dispatch filters",
{
poolSize: filteredTargets.length,
attempted: 0,
excluded: quotaExcluded,
attemptOrder: [],
terminalReason: "all_targets_skipped",
},
{ code: "ALL_TARGETS_SKIPPED", type: "service_unavailable" }
return new Response(
JSON.stringify({
error: {
message:
"Service temporarily unavailable: all targets were skipped by pre-dispatch filters",
type: "service_unavailable",
code: "ALL_TARGETS_SKIPPED",
},
}),
{ status: 503, headers: { "Content-Type": "application/json" } }
);
}
return new Response(

View File

@@ -1,7 +1,7 @@
import type { TierAssignment, ProviderTier } from "./tierTypes";
import { PROVIDER_TIER } from "./tierTypes";
import type { SpecificityResult, SpecificityLevel } from "./specificityTypes";
import { classifyTier, classifyTierAsync } from "./tierResolver";
import { classifyTier } from "./tierResolver";
import {
analyzeSpecificity,
getSpecificityLevel,
@@ -29,24 +29,17 @@ export interface RoutingHint {
strategyModifier: StrategyModifier;
}
export async function generateRoutingHints(
export function generateRoutingHints(
targets: ResolvedComboTarget[],
input: RuleInput
): Promise<RoutingHint> {
): RoutingHint {
const tierAssignments = new Map<string, TierAssignment>();
const modelTargets = targets.filter((t) => t.kind === "model");
const settled = await Promise.all(
modelTargets.map(async (target) => {
const key = `${target.provider}::${target.modelStr}`;
try {
return [key, await classifyTierAsync(target.provider, target.modelStr)] as const;
} catch {
return [key, classifyTier(target.provider, target.modelStr)] as const;
}
})
);
for (const [key, assignment] of settled) {
if (!tierAssignments.has(key)) tierAssignments.set(key, assignment);
for (const target of targets) {
if (target.kind !== "model") continue;
const key = `${target.provider}::${target.modelStr}`;
if (!tierAssignments.has(key)) {
tierAssignments.set(key, classifyTier(target.provider, target.modelStr));
}
}
const specificity = analyzeSpecificity(input);

View File

@@ -1,5 +1,4 @@
import { getPricingForModel as getDefaultPricingForModel } from "@/shared/constants/pricing";
import { isFreeModel } from "@/shared/utils/freeModels";
import type { TierConfig } from "./tierTypes";
export interface ModelPricing {
@@ -39,16 +38,14 @@ export const KNOWN_MODEL_PRICING: Record<string, ModelPricing> = {
};
export function getModelPricing(provider: string, model: string): ModelPricing {
const normalized = String(model || "")
.split("/")
.pop()!
.toLowerCase();
const providerHit = KNOWN_MODEL_PRICING[`${provider}/${normalized}`.toLowerCase()];
if (providerHit) return providerHit;
const defaultPricing = getDefaultPricingForModel(provider, model);
if (defaultPricing) {
const inputCostPer1M = Number(defaultPricing.input);
const outputCostPer1M = Number(defaultPricing.output);
const providerKey = `${provider}/${model}`.toLowerCase();
if (KNOWN_MODEL_PRICING[providerKey]) {
return KNOWN_MODEL_PRICING[providerKey];
}
const providerPricing = getDefaultPricingForModel(provider, model);
if (providerPricing) {
const inputCostPer1M = Number(providerPricing.input);
const outputCostPer1M = Number(providerPricing.output);
if (Number.isFinite(inputCostPer1M) && Number.isFinite(outputCostPer1M)) {
return {
inputCostPer1M,
@@ -57,18 +54,13 @@ export function getModelPricing(provider: string, model: string): ModelPricing {
};
}
}
const genericHit = KNOWN_MODEL_PRICING[normalized];
if (genericHit) return genericHit;
if (isFreeModel(provider, { id: normalized }))
return { inputCostPer1M: 0, outputCostPer1M: 0, isFree: true };
const directKey = model.toLowerCase();
if (KNOWN_MODEL_PRICING[directKey]) {
return KNOWN_MODEL_PRICING[directKey];
}
return { inputCostPer1M: 5.0, outputCostPer1M: 15.0, isFree: false };
}
/** Input cost per 1M tokens a virtual auto-combo candidate is scored at. */
export function resolveVirtualCost(providerId: string, modelId: string): number {
return getModelPricing(providerId, modelId).inputCostPer1M;
}
export function isExplicitlyFree(provider: string, config: TierConfig): boolean {
return config.freeProviders.includes(provider.toLowerCase());
}

View File

@@ -17,10 +17,6 @@
* atomically.
*/
import { SlidingWindowLimiter, type RateLimitWindow } from "./slidingWindowLimiter.ts";
import {
markLocalRateLimitError,
LEGACY_RATE_LIMIT_QUEUE_TIMEOUT_CODE,
} from "./rateLimitManager/errors.ts";
// Opt-in per-provider caps. Example shape (commented — add real entries as needed):
// "some-headerless-provider": { requests: 60, windowMs: 60_000 },
@@ -132,36 +128,16 @@ export async function awaitProviderDefaultSlot(
provider: string,
connectionId: string | null,
signal: AbortSignal | null,
remainingBudgetMs?: number
maxWaitMs?: number
): Promise<void> {
const cfg = getProviderDefaultRateLimit(provider);
if (!cfg) return;
if (typeof remainingBudgetMs === "number" && remainingBudgetMs <= 0)
throw markLocalRateLimitError(
new Error(
`Queue budget exhausted before provider-default slot (remaining=${remainingBudgetMs}ms)`
),
LEGACY_RATE_LIMIT_QUEUE_TIMEOUT_CODE
);
const budget =
typeof remainingBudgetMs === "number" && Number.isFinite(remainingBudgetMs)
? Math.min(remainingBudgetMs, cfg.windowMs)
: Math.max(cfg.windowMs, 0);
const budget = Math.max(cfg.windowMs, maxWaitMs && maxWaitMs > 0 ? maxWaitMs : 0);
const start = Date.now();
for (;;) {
const waitMs = acquireProviderDefaultSlot(provider, connectionId);
if (waitMs === 0) return;
if (Date.now() - start >= budget)
throw markLocalRateLimitError(
new Error(`Provider-default slot wait exceeded budget ${budget}ms`),
LEGACY_RATE_LIMIT_QUEUE_TIMEOUT_CODE
);
const remainingBudget = budget - (Date.now() - start);
if (remainingBudget <= 0)
throw markLocalRateLimitError(
new Error(`Provider-default slot budget exhausted`),
LEGACY_RATE_LIMIT_QUEUE_TIMEOUT_CODE
);
await sleepOrAbort(Math.min(waitMs, remainingBudget), signal);
if (Date.now() - start >= budget) return; // waited the budget; let it through
await sleepOrAbort(Math.min(waitMs, budget), signal);
}
}

View File

@@ -1,73 +0,0 @@
/**
* Shared duration-aware quota window labels for AUTH preflight and /api/usage.
* Routing keys (`session`, `weekly`) stay position-based; only the display
* string follows the real window length so a 30-day primary window is not
* reported as "session (5h)".
*/
const SESSION_MAX_SECONDS = 6 * 3600;
const WEEKLY_MAX_SECONDS = 8 * 24 * 3600;
const MONTHLY_MAX_SECONDS = 40 * 24 * 3600;
const GENERIC_FAMILY = new Set(["session", "weekly", "monthly"]);
export type QuotaWindowLabelInput = {
key: string;
displayName?: string | null;
windowSeconds?: number | null;
};
function toPositiveSeconds(value: unknown): number | null {
if (typeof value === "number" && Number.isFinite(value) && value > 0) return value;
if (typeof value === "string" && value.trim().length > 0) {
const parsed = Number(value);
if (Number.isFinite(parsed) && parsed > 0) return parsed;
}
return null;
}
export function formatWindowDuration(windowSeconds: number): string {
if (windowSeconds % 86400 === 0) return `${windowSeconds / 86400}d`;
if (windowSeconds % 3600 === 0) return `${windowSeconds / 3600}h`;
return `${Math.round(windowSeconds)}s`;
}
export function inferWindowFamilyLabel(windowSeconds: number): string | undefined {
if (windowSeconds <= SESSION_MAX_SECONDS) return "Session";
if (windowSeconds <= WEEKLY_MAX_SECONDS) return "Weekly";
if (windowSeconds <= MONTHLY_MAX_SECONDS) return "Monthly";
return undefined;
}
function normalizeFamily(value: string): string {
return value.trim().toLowerCase();
}
/**
* Prefer a duration-aware label. When usage already stamped a generic family
* (`Weekly`) on a 30-day window, replace it with Monthly so AUTH and the
* usage API agree on the reset window.
*/
export function formatQuotaWindowLabel(input: QuotaWindowLabelInput): string {
const seconds = toPositiveSeconds(input.windowSeconds);
const duration = seconds != null ? formatWindowDuration(seconds) : null;
const inferred = seconds != null ? inferWindowFamilyLabel(seconds) : undefined;
const named = typeof input.displayName === "string" ? input.displayName.trim() : "";
const key = typeof input.key === "string" && input.key.trim() ? input.key.trim() : "quota";
let base = named || inferred || key;
if (inferred && named && GENERIC_FAMILY.has(normalizeFamily(named)) && inferred !== named) {
base = inferred;
}
if (duration && !base.toLowerCase().includes(duration.toLowerCase())) {
return `${base} (${duration})`;
}
return base;
}
export function formatQuotaUsageReason(
input: QuotaWindowLabelInput,
usedPercentage: number
): string {
return `${formatQuotaWindowLabel(input)} usage ${Math.round(usedPercentage)}%`;
}

View File

@@ -31,14 +31,9 @@ import {
markLocalRateLimitError,
RATE_LIMIT_EXECUTION_TIMEOUT_CODE,
RATE_LIMIT_QUEUE_WEDGED_CODE,
LEGACY_RATE_LIMIT_QUEUE_TIMEOUT_CODE,
} from "./rateLimitManager/errors";
import { LimiterWedgeWatchdog, WATCHDOG_INTERVAL_MS } from "./rateLimitManager/wedgeWatchdog";
import { toNumber } from "@/shared/utils/numeric";
import {
getExecutorTimeoutMs,
resolveConnectionTimeoutMs,
} from "../handlers/chatCore/upstreamTimeouts.ts";
interface LearnedLimitEntry {
provider: string;
@@ -198,15 +193,9 @@ export function resolveRequestQueueMaxWaitMs(
* after a job leaves QUEUED; bounds execution, never queue wait. Kept strictly
* separate from the queue-wait budget (`maxWaitMs`) so the backstop cannot
* undercut upstream fetch-start timeouts on non-incremental gateways.
* Per-connection `executionMaxWaitMs` in `provider_connections.rateLimitOverrides`
* overrides the global setting when present (same precedence as `maxWaitMs`).
*/
export function resolveExecutionMaxWaitMs(connectionId?: string): number {
const override = connectionId
? (connectionRateLimitOverrides.get(connectionId) as Record<string, number> | undefined)
?.executionMaxWaitMs
: undefined;
return resolveOverride(override, currentRequestQueueSettings.executionMaxWaitMs);
export function resolveExecutionMaxWaitMs(): number {
return currentRequestQueueSettings.executionMaxWaitMs;
}
function buildLimiterDefaults() {
@@ -577,18 +566,7 @@ function getLimiter(provider, connectionId, model = null) {
* @param {AbortSignal} signal - Optional abort signal to cancel waiting
* @returns {Promise<unknown>} Result of fn()
*/
export async function withRateLimit(
provider,
connectionId,
model,
fn,
signal = null,
remainingBudgetMs = undefined,
correlationId = undefined,
opts:
| { executor?: { getTimeoutMs?: () => unknown }; providerSpecificData?: unknown }
| undefined = undefined
) {
export async function withRateLimit(provider, connectionId, model, fn, signal = null) {
if (!enabledConnections.has(connectionId)) {
return fn();
}
@@ -601,36 +579,10 @@ export async function withRateLimit(
throw err;
}
const queueBudgetMs = resolveRequestQueueMaxWaitMs(
provider,
undefined,
connectionId ?? undefined
);
const budgetForSlot =
typeof remainingBudgetMs === "number" && Number.isFinite(remainingBudgetMs)
? remainingBudgetMs
: queueBudgetMs;
if (
typeof remainingBudgetMs === "number" &&
Number.isFinite(remainingBudgetMs) &&
remainingBudgetMs <= 0
) {
throw markLocalRateLimitError(
new Error(`Queue budget exhausted before rate-limit (remaining=${remainingBudgetMs}ms)`),
LEGACY_RATE_LIMIT_QUEUE_TIMEOUT_CODE
);
}
const slotStart = Date.now();
await awaitProviderDefaultSlot(provider, connectionId, signal, budgetForSlot);
const elapsedSlot = Date.now() - slotStart;
const remainingForQueue =
typeof remainingBudgetMs === "number" && Number.isFinite(remainingBudgetMs)
? Math.max(0, remainingBudgetMs - elapsedSlot)
: queueBudgetMs;
if (correlationId)
logRateLimit(
`[RATE-LIMIT] cid=${correlationId} provider=${provider} remainingForQueue=${remainingForQueue}ms`
);
// Proactive sliding-window fallback for header-less providers with a declared cap
// (Fase 8.2). No-op unless PROVIDER_DEFAULT_RATE_LIMITS has an entry for `provider`.
const maxWaitMs = resolveRequestQueueMaxWaitMs(provider, undefined, connectionId);
await awaitProviderDefaultSlot(provider, connectionId, signal, maxWaitMs);
const limiter = getLimiter(provider, connectionId, model);
// Bottleneck's `expiration` starts only after a job leaves QUEUED, so it
@@ -639,26 +591,7 @@ export async function withRateLimit(
// never by the queue-wait budget: non-incremental gateways legitimately run
// for minutes before first bytes, and an expiration at the queue budget
// killed them mid-flight (false 504s on opencode-go/glm-5.3-flash).
// Per-connection executionMaxWaitMs wins, but never undercuts the upstream
// fetch-start timeout — otherwise the backstop kills a healthy mid-flight
// response (regression #12025 on GLM/thinking models).
const perConnExec = resolveExecutionMaxWaitMs(connectionId ?? undefined);
const upstreamMs = opts?.executor
? getExecutorTimeoutMs(
opts.executor as unknown,
provider,
model ?? undefined,
resolveConnectionTimeoutMs(
opts.providerSpecificData as Record<string, unknown> | null | undefined
)
)
: undefined;
const executionExpirationMs = upstreamMs ? Math.max(perConnExec, upstreamMs) : perConnExec;
if (upstreamMs && perConnExec < upstreamMs) {
logRateLimit(
`[RATE-LIMIT] executionMaxWaitMs ${perConnExec}ms clamped to upstream ${upstreamMs}ms for ${provider}/${model ?? ""}`
);
}
const executionExpirationMs = resolveExecutionMaxWaitMs();
const scheduleOpts =
executionExpirationMs && executionExpirationMs > 0 ? { expiration: executionExpirationMs } : {};
@@ -677,43 +610,6 @@ export async function withRateLimit(
throw admissionErr;
}
const queueRemainingMs = remainingForQueue;
let queueTimedOut = false;
let delayId: ReturnType<typeof setTimeout> | null = null;
const queueTimeoutErr = markLocalRateLimitError(
new Error(
`Request exceeded queue budget maxWaitMs=${queueRemainingMs}ms for ${provider}/${model ?? ""} — queue budget does not bound execution (executionMaxWaitMs=${executionExpirationMs}ms)`
),
LEGACY_RATE_LIMIT_QUEUE_TIMEOUT_CODE
);
if (queueRemainingMs <= 0) throw queueTimeoutErr;
const timeoutPromise = new Promise<never>((_, reject) => {
delayId = setTimeout(() => {
queueTimedOut = true;
reject(queueTimeoutErr);
}, queueRemainingMs);
});
timeoutPromise.catch(() => {});
// Clear the queue-wait timer once the job leaves QUEUED and starts executing.
// Without this, the timer would also bound execution (queueRemainingMs ≈ 40ms
// would kill a 300ms execution that correctly left the queue immediately).
const wrappedFn = () => {
if (queueTimedOut) return Promise.reject(queueTimeoutErr);
if (delayId) {
clearTimeout(delayId);
delayId = null;
}
return (fn as unknown as (s?: AbortSignal) => Promise<unknown>)(signal ?? undefined);
};
const scheduled = limiter.schedule(scheduleOpts, wrappedFn as unknown as () => Promise<unknown>);
scheduled.catch(() => {});
// Note: if timeoutPromise wins while the job is still QUEUED (blocked by
// maxConcurrent), Bottleneck cannot cancel it — wrappedFn rejects only on
// dispatch after the slot frees. Until then counts().QUEUED stays 1 and
// maxQueueDepth admission sees an inflated depth transiently; this is
// inherent to Bottleneck (no cancelQueuedJob) and does not affect
// correctness since fnCalled stays false.
try {
if (signal) {
let abortListener: (() => void) | undefined;
@@ -739,28 +635,23 @@ export async function withRateLimit(
abortListener = onAbort;
signal.addEventListener("abort", abortListener, { once: true });
}
abortPromise.catch(() => {});
try {
return await Promise.race([scheduled, timeoutPromise, abortPromise]);
// Race the work against the abort signal. When abort wins, fn is still
// running inside Bottleneck's limiter — its eventual rejection must not
// surface as an unhandledRejection. The .catch(noop) silences only the
// orphaned branch; the real rejection comes from abortPromise.
const scheduled = limiter.schedule(scheduleOpts, fn);
scheduled.catch(() => {}); // prevent unhandledRejection when abort wins
abortPromise.catch(() => {}); // prevent unhandledRejection when scheduled wins
return await Promise.race([scheduled, abortPromise]);
} finally {
if (delayId) {
clearTimeout(delayId);
delayId = null;
}
if (abortListener) {
signal.removeEventListener("abort", abortListener);
}
}
} else {
try {
return await Promise.race([scheduled, timeoutPromise]);
} finally {
if (delayId) {
clearTimeout(delayId);
delayId = null;
}
}
return await limiter.schedule(scheduleOpts, fn);
}
} catch (err) {
// Only Bottleneck-owned failures are rewritten. Application code can throw

View File

@@ -142,40 +142,3 @@ export function getTierStats(): Record<ProviderTier, number> {
}
return stats;
}
export let tierAsyncFallbackTotal = 0; // exported for testability
export async function classifyTierAsync(provider: string, model: string): Promise<TierAssignment> {
try {
const { getPricingForModel } = await import("@/lib/db/settings");
const db = await getPricingForModel(provider, model);
const input = Number((db as { input?: unknown } | null)?.input);
const output = Number((db as { output?: unknown } | null)?.output);
if (Number.isFinite(input) && input >= 0) {
const out = Number.isFinite(output) && output >= 0 ? output : input;
// Same thresholds as the sync path below; a DB $0 lands FREE by threshold
// (the DB carries no isFree flag of its own).
let tier: ProviderTier;
if (input <= currentConfig.defaults.freeThreshold) tier = PROVIDER_TIER.FREE;
else if (input <= currentConfig.defaults.cheapThreshold) tier = PROVIDER_TIER.CHEAP;
else tier = PROVIDER_TIER.PREMIUM;
const sync = getModelPricing(provider, model); // keeps freeQuotaLimit when DB is mute
const assignment: TierAssignment = {
provider,
model,
tier,
reason: `DB cost-based: $${input}/M input`,
costPer1MInput: input,
costPer1MOutput: out,
hasFreeTier: input === 0,
freeQuotaLimit: sync.freeQuotaLimit,
};
tierCache.set(cacheKey(provider, model), assignment);
return assignment;
}
} catch {
// fall through to sync
}
tierAsyncFallbackTotal++;
return classifyTier(provider, model);
}

View File

@@ -227,7 +227,7 @@ export function openaiResponsesToOpenAIRequest(
const itemType = toString(item.type) || (item.role ? "message" : "");
if (itemType === "message") {
const role = toString(item.role) === "agent_message" ? "assistant" : toString(item.role);
const role = toString(item.role);
if (role !== "assistant") {
if (currentAssistantMsg) {
@@ -484,13 +484,6 @@ export function openaiResponsesToOpenAIRequest(
continue;
}
// Defense in depth for Responses/subagent fallback: agent_message is
// Responses-only. Normalization should already have rewritten or dropped it;
// never throw a 5xx-looking unsupported-feature error if a shape slips through.
if (itemType === "agent_message" || toString(item.role) === "agent_message") {
continue;
}
throw unsupportedFeature(
`Unsupported Responses API feature: input item type '${itemType || "missing"}' cannot be represented in Chat Completions`
);
@@ -780,10 +773,7 @@ export function openaiResponsesToOpenAIRequest(
// ("When using tool_choice, tools must be set"). Contradictory choices like "required"
// or forced functions are preserved so the upstream error remains visible.
const finalChatTools = Array.isArray(result.tools) ? result.tools : [];
if (
finalChatTools.length === 0 &&
(result.tool_choice === "auto" || result.tool_choice === "none")
) {
if (finalChatTools.length === 0 && (result.tool_choice === "auto" || result.tool_choice === "none")) {
delete result.tool_choice;
}

View File

@@ -1,20 +1,12 @@
type JsonRecord = Record<string, unknown>;
function isAgentMessageItem(item: JsonRecord): boolean {
return item.type === "agent_message" || item.role === "agent_message";
}
function normalizeAgentMessageForChat(item: JsonRecord): JsonRecord | null {
if (item.type !== "agent_message") return null;
function collectAgentMessageText(item: JsonRecord): string | null {
if (typeof item.content === "string") return item.content;
if (typeof item.text === "string") return item.text;
if (!Array.isArray(item.content)) return null;
const textParts: string[] = [];
for (const partValue of item.content) {
if (typeof partValue === "string") {
textParts.push(partValue);
continue;
}
if (!partValue || typeof partValue !== "object" || Array.isArray(partValue)) {
return null;
}
@@ -25,21 +17,12 @@ function collectAgentMessageText(item: JsonRecord): string | null {
// partial plaintext envelope or forward an opaque payload the model cannot use.
return null;
}
if (part.type !== "input_text" && part.type !== "output_text" && part.type !== "text") {
return null;
}
if (typeof part.text !== "string") return null;
if (part.type !== "input_text" || typeof part.text !== "string") return null;
textParts.push(part.text);
}
return textParts.join("\n");
}
function normalizeAgentMessageForChat(item: JsonRecord): JsonRecord | null {
if (!isAgentMessageItem(item)) return null;
const text = collectAgentMessageText(item);
if (typeof text !== "string" || !text.trim()) return null;
const text = textParts.join("\n");
if (!text.trim()) return null;
return {
type: "message",
@@ -142,7 +125,7 @@ function normalizeResponsesInputItemForChat(value: unknown): unknown {
const agentMessage = normalizeAgentMessageForChat(item);
if (agentMessage) return agentMessage;
if (isAgentMessageItem(item)) {
if (item.type === "agent_message") {
// Encrypted or malformed agent messages have no lossless Chat equivalent.
// Treat them like other Responses-only metadata instead of failing the whole turn.
return { type: "reasoning" };

View File

@@ -30,7 +30,10 @@ import { rejectEmptyChoicesStream, buildEmptyChoicesStreamError } from "./stream
import { calculateCost } from "@/lib/usage/costCalculator";
import { buildOmniRouteSseMetadataComment } from "@/domain/omnirouteResponseMeta";
import { sseCommentsEnabled } from "./sseHeartbeat.ts";
import { createStructuredSSECollector } from "./streamPayloadCollector.ts";
import {
createStructuredSSECollector,
buildStreamSummaryFromEvents,
} from "./streamPayloadCollector.ts";
import { STREAM_IDLE_TIMEOUT_MS, FETCH_BODY_TIMEOUT_MS, HTTP_STATUS } from "../config/constants.ts";
import {
OMIT_STREAMING_CHUNK_MARKER,
@@ -85,7 +88,6 @@ import { restoreClaudeToolName } from "../services/claudeCodeToolRemapper.ts";
import { normalizeFinalOpenAIStreamChunk } from "./openAIStreamChunk.ts";
import { collectClaudeDelta } from "./streamClaudeDelta.ts";
import { createStreamTiming, type StreamTiming } from "./streamTiming.ts";
import { buildUsageOnlyChunk } from "./usageOnlyChunk.ts";
/**
* Race a response body read against a timeout.
@@ -766,8 +768,6 @@ export function createSSEStream(options: StreamOptions = {}) {
let passthroughBufferedTextualToolCallContent = "";
/** Passthrough: whether a usage block was already forwarded to the client (prevents double). */
let passthroughForwardedUsage = false;
/** Translate: usage already reached the client, or no trailing usage chunk applies. */
let translateForwardedUsage = sourceFormat !== FORMATS.OPENAI || !shouldEmitDoneTerminator;
// Passthrough Responses SSE: snapshots of items seen via `response.output_item.done`,
// used to backfill `response.completed.response.output` when upstream returns it
// empty (which happens when `store: false` — see backfillResponsesCompletedOutput).
@@ -844,27 +844,6 @@ export function createSSEStream(options: StreamOptions = {}) {
});
const clientPayloadCollector = createStructuredSSECollector({
stage: "client_response",
// Live incident (2026-09-04): the OPENAI_RESPONSES-only carve-outs below
// rebuilt the client summary from clientPayloadCollector.getEvents() --
// the collector's own RETAINED (possibly cap-truncated) event array --
// even after providerPayloadCollector got a live cap-independent reducer
// for the exact same class of bug (#9315, see that collector's own
// `format:` comment above). A reasoning-heavy stream that exhausts the
// cap during the reasoning phase alone (measured live: routine, not an
// edge case) silently dropped the terminal response.completed event from
// the retained array, so the rebuilt-from-events summary permanently
// showed status "in_progress" with empty output even though the client
// itself received the real, complete reply. Wiring `format` here gives
// this collector the SAME always-live reducer providerPayloadCollector
// already has, so switching the carve-outs below from
// buildStreamSummaryFromEvents(collector.getEvents(), ...) to
// collector.getSummary() makes them cap-independent too -- strictly
// equivalent for a stream that never hits the cap, correct instead of
// silently empty for one that does. Same mode ternary as
// clientExpectsResponsesStream/clientExpectsClaudeStream above: passthrough
// forwards clientResponseFormat as-is, translate re-shapes to sourceFormat.
format: mode === STREAM_MODE.PASSTHROUGH ? clientResponseFormat : sourceFormat,
fallbackModel: model,
});
// Per-stream instances to avoid shared state with concurrent streams
const decoder = new TextDecoder();
@@ -1059,11 +1038,9 @@ export function createSSEStream(options: StreamOptions = {}) {
const estimated = estimateUsage(body, totalContentLength, sourceFormat);
itemSanitized.usage = timing.withTps(filterUsageForFormat(estimated, sourceFormat));
state.usage = estimated;
if (hasValidUsage(estimated)) translateForwardedUsage = true; // finish chunk carries it
} else if (state?.finishReason && isFinishChunk && state.usage) {
const buffered = addBufferToUsage(state.usage);
itemSanitized.usage = timing.withTps(filterUsageForFormat(buffered, sourceFormat));
translateForwardedUsage = true;
}
if (
@@ -2588,11 +2565,14 @@ export function createSSEStream(options: StreamOptions = {}) {
// upstream DID send usage (trailing or in-band), it was forwarded
// already and passthroughForwardedUsage guards this off.
if (shouldEmitDoneTerminator && !passthroughForwardedUsage && hasValidUsage(usage)) {
const usageOnlyChunk = buildUsageOnlyChunk(
passthroughLastChatId ?? passthroughResponsesId,
const usageOnlyChunk = {
id: passthroughLastChatId ?? passthroughResponsesId ?? `chatcmpl-${Date.now()}`,
object: "chat.completion.chunk",
created: Math.floor(Date.now() / 1000),
model,
timing.withTps(filterUsageForFormat(usage, sourceFormat || FORMATS.OPENAI))
);
choices: [],
usage: timing.withTps(filterUsageForFormat(usage, sourceFormat || FORMATS.OPENAI)),
};
const usageOutput = `data: ${JSON.stringify(usageOnlyChunk)}\n\n`;
reqLogger?.appendConvertedChunk?.(usageOutput);
forward(controller, encoder.encode(usageOutput));
@@ -2685,19 +2665,18 @@ export function createSSEStream(options: StreamOptions = {}) {
// #9315 switched the summary to the accumulated responseBody to avoid
// stale/truncated event data — but responseBody here is synthesized in
// chat-completion shape, which loses the Responses API `response` object.
// Keep a Responses-shaped summary for OPENAI_RESPONSES only. responseBody
// Keep the events-derived summary for OPENAI_RESPONSES only. responseBody
// itself never carries an `object` marker (it's built purely for the
// client, which doesn't need one) — the dashboard's Provider Response
// panel does, so stamp `object: "chat.completion"` on a shallow copy
// used only for this summary, leaving responseBody itself untouched.
// getSummary(), not buildStreamSummaryFromEvents(getEvents(), ...): the
// latter only sees the collector's RETAINED (possibly cap-truncated)
// events, silently losing a late response.completed event on a long
// reasoning-heavy stream (live incident 2026-09-04) -- see
// providerPayloadCollector's own `format:` construction comment above.
providerPayload: providerPayloadCollector.build(
sourceFormat === FORMATS.OPENAI_RESPONSES
? providerPayloadCollector.getSummary()
? buildStreamSummaryFromEvents(
providerPayloadCollector.getEvents(),
sourceFormat,
model
)
: { object: "chat.completion", ...responseBody },
{ includeEvents: false }
),
@@ -2708,13 +2687,14 @@ export function createSSEStream(options: StreamOptions = {}) {
// src/lib/usage/callLogs.ts is always null for a Responses-API client
// (extractResponsesId reads `clientResponse.id`, which the chat-shaped
// responseBody never has), so previous_response_id continuation lookups
// in src/lib/db/responsesContinuationStore.ts always miss. getSummary(),
// not buildStreamSummaryFromEvents(getEvents(), ...) -- same cap-truncation
// reasoning as providerPayload above; clientPayloadCollector's own `format:`
// construction above gives it the same live, cap-independent reducer.
// in src/lib/db/responsesContinuationStore.ts always miss.
clientPayload: clientPayloadCollector.build(
clientResponseFormat === FORMATS.OPENAI_RESPONSES
? clientPayloadCollector.getSummary()
? buildStreamSummaryFromEvents(
clientPayloadCollector.getEvents(),
clientResponseFormat,
model
)
: responseBody,
{ includeEvents: false }
),
@@ -2862,26 +2842,8 @@ export function createSSEStream(options: StreamOptions = {}) {
* emitted once at stream end when merged into the final translated chunk.
*/
// Estimate usage if provider didn't return valid usage (for translate mode)
if (!hasValidUsage(state?.usage) && totalContentLength > 0) {
state.usage = estimateUsage(body, totalContentLength, sourceFormat);
}
// Send [DONE] (only if not already sent during transform)
if (!doneSent) {
// Upstream stayed silent on usage: send the estimate as the canonical
// trailing usage-only chunk before [DONE], like the passthrough flush.
if (!translateForwardedUsage && hasValidUsage(state?.usage)) {
const usageOnlyChunk = buildUsageOnlyChunk(
(state as unknown as Record<string, unknown>)?.chatId,
model,
timing.withTps(filterUsageForFormat(state.usage, sourceFormat))
);
const usageOutput = `data: ${JSON.stringify(usageOnlyChunk)}\n\n`;
reqLogger?.appendConvertedChunk?.(usageOutput);
forward(controller, encoder.encode(usageOutput));
clientPayloadCollector.push(usageOnlyChunk);
}
await emitFinalSseMetadata(controller, state?.usage as Record<string, unknown> | null);
doneSent = true;
if (shouldEmitDoneTerminator) {
@@ -2892,6 +2854,11 @@ export function createSSEStream(options: StreamOptions = {}) {
}
}
// Estimate usage if provider didn't return valid usage (for translate mode)
if (!hasValidUsage(state?.usage) && totalContentLength > 0) {
state.usage = estimateUsage(body, totalContentLength, sourceFormat);
}
if (hasValidUsage(state?.usage)) {
logUsage(state.provider || targetFormat, state.usage, model, connectionId, apiKeyInfo);
} else {
@@ -2974,14 +2941,13 @@ export function createSSEStream(options: StreamOptions = {}) {
// all — stamp `object: "chat.completion"` on a shallow copy used only
// for this summary; responseBody itself (sent to the client / below)
// stays untouched.
// getSummary(), not buildStreamSummaryFromEvents(getEvents(), ...) -- the
// latter only sees the collector's RETAINED (possibly cap-truncated)
// events, silently losing a late response.completed event on a long
// reasoning-heavy stream (live incident 2026-09-04) -- see
// providerPayloadCollector's own `format:` construction comment above.
providerPayload: providerPayloadCollector.build(
targetFormat === FORMATS.OPENAI_RESPONSES
? providerPayloadCollector.getSummary()
? buildStreamSummaryFromEvents(
providerPayloadCollector.getEvents(),
targetFormat,
model
)
: { object: "chat.completion", ...responseBody },
{ includeEvents: false }
),
@@ -2992,13 +2958,14 @@ export function createSSEStream(options: StreamOptions = {}) {
// sourceFormat, ...) above confirms that direction. emitTranslatedClientItem
// already pushes every client-visible translated item into
// clientPayloadCollector unconditionally, so the events are already there;
// getSummary() (not buildStreamSummaryFromEvents(getEvents(), ...)) makes
// reading them back cap-independent -- same reasoning as providerPayload
// above; clientPayloadCollector's own `format:` construction gives it the
// same live reducer.
// this only fixes what gets built from them.
clientPayload: clientPayloadCollector.build(
sourceFormat === FORMATS.OPENAI_RESPONSES
? clientPayloadCollector.getSummary()
? buildStreamSummaryFromEvents(
clientPayloadCollector.getEvents(),
sourceFormat,
model
)
: responseBody,
{ includeEvents: false }
),

View File

@@ -857,19 +857,12 @@ export function pipeWithDisconnect(
providerResponse: Response,
transformStream: TransformStream<Uint8Array, Uint8Array>,
streamController: StreamController,
opts: { stallTimeoutMs?: number; contentStallTimeoutMs?: number; highWaterMark?: number } = {}
opts: { stallTimeoutMs?: number; highWaterMark?: number } = {}
) {
const stallTimeoutMs = opts.stallTimeoutMs ?? DEFAULT_STREAM_STALL_TIMEOUT_MS;
// Disabled unless a caller opts in with an explicit budget (chatCore wires
// the adaptive streamReadinessPolicy.timeoutMs — see its own doc comment).
// No blanket default here: an arbitrary constant picked at this layer,
// without the request's actual model/provider/payload context, would risk
// false-stalling the exact slow-first-content reasoning models the
// readiness policy already knows to grant more patience.
const contentStallTimeoutMs = opts.contentStallTimeoutMs ?? 0;
// Watchdogs disabled — preserve legacy behavior verbatim.
if ((!stallTimeoutMs || stallTimeoutMs <= 0) && contentStallTimeoutMs <= 0) {
// Watchdog disabled — preserve legacy behavior verbatim.
if (!stallTimeoutMs || stallTimeoutMs <= 0) {
const transformedBody = providerResponse.body.pipeThrough(transformStream);
return createDisconnectAwareStream(
{ readable: transformedBody, writable: createNoopAbortWritable() },
@@ -898,7 +891,6 @@ export function pipeWithDisconnect(
}
};
const armStall = () => {
if (!stallTimeoutMs || stallTimeoutMs <= 0) return;
clearStall();
stallTimer = setTimeout(() => {
stallTimer = null;
@@ -927,117 +919,48 @@ export function pipeWithDisconnect(
}, stallTimeoutMs);
};
// Second, independent watchdog: fires when the upstream keeps sending raw
// bytes (so armStall() above keeps resetting and never fires) but none of
// them ever carry real model output — only lifecycle/ping frames
// (OpenAI Responses response.in_progress/response.created, bare
// role-only start chunks, etc). ensureStreamReadiness's own gate already
// treats any one of those as "ready" and hands the connection off (see its
// own doc comment and kiro.ts's deliberate early role-only chunk — several
// providers rely on that fast handoff for UX, so tightening readiness
// itself would regress them). Once handed off there was previously nothing
// watching whether the model ever actually said anything: a stalled free
// OpenRouter model (e.g. minimax-m3:free under load) could stream nothing
// but response.in_progress pings indefinitely, relayed byte-for-byte to
// the client, until the CLIENT's own idle timeout eventually gave up --
// sometimes 120s, sometimes 900s depending on the calling task, always
// slower and less informative than OmniRoute failing this attempt itself
// with a clear error the client's own retry/fallback logic can react to
// immediately. Armed ONCE at stream start (not re-armed by lifecycle-only
// bytes, unlike armStall above) and cleared permanently the first time
// real content is observed -- reuses the exact classifier
// (createStreamContentWatcher) createDisconnectAwareStream already trusts
// for its own end-of-stream #8649 empty-content check.
let contentStallTimer: ReturnType<typeof setTimeout> | null = null;
let contentStallFired = false;
const upstreamContentWatcher = createStreamContentWatcher();
const upstreamContentDecoder = new TextDecoder();
const clearContentStall = () => {
if (contentStallTimer) {
clearTimeout(contentStallTimer);
contentStallTimer = null;
}
};
const armContentStall = () => {
if (contentStallTimeoutMs <= 0) return;
contentStallTimer = setTimeout(() => {
contentStallTimer = null;
contentStallFired = true;
const stallError = new Error(
`stream content stall: no model output within ${contentStallTimeoutMs}ms (lifecycle/heartbeat events only)`
);
try {
streamController.handleError?.(stallError);
} catch (e) {
console.debug(`[STREAM-HANDLER] content stall watchdog handleError failed:`, e);
}
try {
upstreamTapController?.error(stallError);
} catch (e) {
console.debug(`[STREAM-HANDLER] content stall watchdog upstream tap error failed:`, e);
}
try {
streamController.abort?.();
} catch (e) {
console.debug(`[STREAM-HANDLER] content stall watchdog abort failed:`, e);
}
}, contentStallTimeoutMs);
};
// Wrap controller so every termination path clears both stall timers.
// Without this, abort/complete/error/disconnect paths leave a timer armed
// Wrap controller so every termination path clears the stall timer.
// Without this, abort/complete/error/disconnect paths leave the timer armed
// and a stale abort could fire after the request has already ended.
const wrappedController: StreamController = {
...streamController,
handleComplete: () => {
clearStall();
clearContentStall();
streamController.handleComplete();
},
handleError: (e: unknown) => {
clearStall();
clearContentStall();
// A watchdog already fired its own handleError — the inner pull()
// catch sees the same error propagated through the pipeline; suppress
// the duplicate to keep onError callbacks single-fire.
if (stallFired || contentStallFired) return;
// Watchdog already fired its own handleError — the inner pull() catch
// sees the same error propagated through the pipeline; suppress the
// duplicate to keep onError callbacks single-fire.
if (stallFired) return;
streamController.handleError(e);
},
handleDisconnect: (reason?: string) => {
clearStall();
clearContentStall();
streamController.handleDisconnect(reason);
},
abort: () => {
clearStall();
clearContentStall();
streamController.abort();
},
};
// Inert tap that resets the byte-stall timer on every raw upstream chunk
// and (independently) clears the content-stall timer the first time a
// chunk carries real output. Sits between the provider body and the SSE
// transform so reasoning models that buffer many raw bytes into a single
// emitted event do not look stalled to either watchdog.
// Inert tap that resets the stall timer on every raw upstream byte chunk.
// Sits between the provider body and the SSE transform so reasoning models
// that buffer many raw bytes into a single emitted event do not look
// stalled to the watchdog.
const upstreamTap = new TransformStream<Uint8Array, Uint8Array>({
start(controller) {
upstreamTapController = controller;
armStall();
armContentStall();
},
transform(chunk, controller) {
armStall();
if (contentStallTimeoutMs > 0 && !upstreamContentWatcher.sawContent()) {
upstreamContentWatcher.note(upstreamContentDecoder.decode(chunk, { stream: true }));
if (upstreamContentWatcher.sawContent()) clearContentStall();
}
controller.enqueue(chunk);
},
flush() {
clearStall();
clearContentStall();
},
});

View File

@@ -885,28 +885,24 @@ export function compactStructuredStreamPayload(payload: unknown): unknown {
};
}
// Live incident (2026-09-02, recurred 2026-09-04): a reasoning-heavy response
// streams reasoning token-by-token as hundreds to thousands of tiny SSE
// deltas BEFORE the real output/tool_calls ever arrive. At the old defaults
// (200 events / 48KB) the cap was routinely exhausted during the reasoning
// phase alone, dropping the completion event entirely -- measured live:
// ~22% of a sample of recent successful responses hit this. For a caller
// that reconstructs its logged summary from getEvents() after the fact
// (open-sse/utils/stream.ts's buildStreamSummaryFromEvents(collector.getEvents(),
// ...) pattern) instead of reading getSummary()'s always-live reducer, a
// Live incident (2026-09-02): a reasoning-heavy response streams reasoning
// token-by-token as hundreds to thousands of tiny SSE deltas BEFORE the real
// output/tool_calls ever arrive. At the old defaults (200 events / 48KB) the
// cap was routinely exhausted during the reasoning phase alone, dropping the
// completion event entirely -- measured live: ~22% of a sample of recent
// successful responses hit this. For a caller with no `format` (no live
// reducer -- see the CollectorOptions.format doc comment), the logged
// summary is reconstructed from getEvents() (open-sse/utils/stream.ts), so a
// dropped completion event produced a served-successfully response logged
// with status "in_progress" and empty output -- which
// src/lib/db/responsesContinuationStore.ts then had nothing real to
// reconstruct a later continuation turn from (see its own fail-closed fix,
// 2026-09-02), and which /dashboard/conversations had no way to distinguish
// from a genuinely healthy conversation (see its own "stalled" badge,
// 2026-09-04). Raising the cap alone doesn't eliminate the class of bug for
// an arbitrarily long stream, only makes it less routine; the actual
// cap-independent fix is for every caller to read getSummary() (fed live on
// every push(), see below) instead of re-deriving from getEvents() -- both
// stream.ts collector instances (provider and client payload, passthrough
// and translate mode, all four OPENAI_RESPONSES-shaped build() call sites)
// now do this consistently.
// 2026-09-02). Raising the cap doesn't eliminate the class of bug for an
// arbitrarily long stream, but it removes it as a routine, everyday failure;
// the format-driven live reducer (used by providerPayloadCollector, an
// analogous prior fix) is the cap-independent fix and remains the deeper
// follow-up for a caller that still wants build()'s summary correct beyond
// any fixed cap.
export function createStructuredSSECollector(options: CollectorOptions = {}) {
const { maxEvents = 2000, maxBytes = 524288, stage, format, fallbackModel } = options;
const events: StructuredSSEEvent[] = [];
@@ -956,9 +952,9 @@ export function createStructuredSSECollector(options: CollectorOptions = {}) {
// payload (see CollectorOptions.format) — unlike
// buildStreamSummaryFromEvents(getEvents(), ...), this is correct even
// once the collector has truncated its retained event array. Returns
// undefined if no format was configured (e.g. a caller that never needs
// a reconstructed summary at all and only reads getEvents()/build()'s
// raw event log).
// undefined if no format was configured (e.g. the client-response
// collector, which builds its summary from independently-accumulated
// response state instead).
getSummary(): unknown {
return reducer?.finalize();
},

View File

@@ -1,15 +0,0 @@
/**
* Canonical OpenAI trailing usage-only chunk (`choices: []`) sent before `[DONE]`
* when the upstream never reported usage, so metered chat clients still see
* token counts (#12151). Shared by the passthrough and translate flushes.
*/
export function buildUsageOnlyChunk(id: unknown, model: unknown, usage: unknown) {
return {
id: id ?? `chatcmpl-${Date.now()}`,
object: "chat.completion.chunk",
created: Math.floor(Date.now() / 1000),
model,
choices: [],
usage,
};
}

50
package-lock.json generated
View File

@@ -8687,6 +8687,20 @@
"license": "ISC",
"optional": true
},
"node_modules/@openai/codex-security/node_modules/smol-toml": {
"version": "1.6.1",
"resolved": "https://registry.npmjs.org/smol-toml/-/smol-toml-1.6.1.tgz",
"integrity": "sha512-dWUG8F5sIIARXih1DTaQAX4SsiTXhInKf1buxdY9DIg4ZYPZK5nGM1VRIYmEbDbsHt7USo99xSLFu5Q1IqTmsg==",
"dev": true,
"license": "BSD-3-Clause",
"optional": true,
"engines": {
"node": ">= 18"
},
"funding": {
"url": "https://github.com/sponsors/cyyynthia"
}
},
"node_modules/@openai/codex-security/node_modules/type-fest": {
"version": "5.9.0",
"resolved": "https://registry.npmjs.org/type-fest/-/type-fest-5.9.0.tgz",
@@ -15203,9 +15217,9 @@
}
},
"node_modules/@yarnpkg/parsers/node_modules/js-yaml": {
"version": "4.3.2",
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.2.tgz",
"integrity": "sha512-SFNOvSJ+Dgf/9An904Yx+CgSlIPCkIpao4qo51lpee25TIRejdH3rhR4EZMGoNx3/TP3O+wzWuiTFl4sqbltzA==",
"version": "4.3.1",
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.1.tgz",
"integrity": "sha512-CY6crGq313MX8GkwvB7tzgp99vjQxY1++5y10/BKN/GUfHqWaOGQMNZkBvqSzsZKWk/ijwHlWzzkLulsGHhjWQ==",
"dev": true,
"funding": [
{
@@ -18521,9 +18535,9 @@
"license": "MIT"
},
"node_modules/csv-parse": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/csv-parse/-/csv-parse-7.0.2.tgz",
"integrity": "sha512-uKZghv9UmPkMVLYy//KZ9HFAIJsl7wkhoEdIL0+rhuSY9pZQlhaeGEDPIe+/w7eh81MOql8Q/9+inAGWG6ZHYA==",
"version": "7.0.1",
"resolved": "https://registry.npmjs.org/csv-parse/-/csv-parse-7.0.1.tgz",
"integrity": "sha512-+2z7Ar0APQ7Uu6fX4cn+pitRmxjZ1WPBcGmZFKmA74FCyi7Et/XZx8cjNQ5CjbZ4HCOxXCOpRBYvYH08Qa003A==",
"dev": true,
"license": "MIT"
},
@@ -23589,9 +23603,9 @@
"license": "MIT"
},
"node_modules/hono": {
"version": "4.13.7",
"resolved": "https://registry.npmjs.org/hono/-/hono-4.13.7.tgz",
"integrity": "sha512-c8/gF9ac8Y78/agExVocyLevgR+JlpNB444Py0FSX8pJoPdYUfUzRcXtYEYGwt6l19qIlVZPN5Mfsw9jFShmQQ==",
"version": "4.13.0",
"resolved": "https://registry.npmjs.org/hono/-/hono-4.13.0.tgz",
"integrity": "sha512-jhunvfHWxd7J5EFfSgH4xsYJzSe/lfqbUCxiyyeaQasUsXeEHXtzVid+7EOGByc5JnFa23SSFL3Y2RV/z1T+eQ==",
"license": "MIT",
"engines": {
"node": ">=16.9.0"
@@ -26102,9 +26116,9 @@
}
},
"node_modules/joi": {
"version": "18.2.8",
"resolved": "https://registry.npmjs.org/joi/-/joi-18.2.8.tgz",
"integrity": "sha512-G2TX62h58ZHuwqetJgP2F4ualakqAmZtBYe3jWen7gxQRw5xApX6crnFtuB91WC0c3ESBnva+kGSnb3+6pIQDQ==",
"version": "18.2.3",
"resolved": "https://registry.npmjs.org/joi/-/joi-18.2.3.tgz",
"integrity": "sha512-N5A3KTWQpPWT4ExxxPlUx7WmykGXRzhNidWhV41d6Abu9YfI2NyWCJuxdPnslJCPWtbRpSVOWSnSS6GakLM/Rg==",
"dev": true,
"license": "BSD-3-Clause",
"dependencies": {
@@ -27932,9 +27946,9 @@
}
},
"node_modules/lockfile-lint/node_modules/js-yaml": {
"version": "4.3.2",
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.2.tgz",
"integrity": "sha512-SFNOvSJ+Dgf/9An904Yx+CgSlIPCkIpao4qo51lpee25TIRejdH3rhR4EZMGoNx3/TP3O+wzWuiTFl4sqbltzA==",
"version": "4.3.1",
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.1.tgz",
"integrity": "sha512-CY6crGq313MX8GkwvB7tzgp99vjQxY1++5y10/BKN/GUfHqWaOGQMNZkBvqSzsZKWk/ijwHlWzzkLulsGHhjWQ==",
"dev": true,
"funding": [
{
@@ -39606,9 +39620,9 @@
}
},
"node_modules/xmlbuilder2/node_modules/js-yaml": {
"version": "4.3.2",
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.2.tgz",
"integrity": "sha512-SFNOvSJ+Dgf/9An904Yx+CgSlIPCkIpao4qo51lpee25TIRejdH3rhR4EZMGoNx3/TP3O+wzWuiTFl4sqbltzA==",
"version": "4.3.1",
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.1.tgz",
"integrity": "sha512-CY6crGq313MX8GkwvB7tzgp99vjQxY1++5y10/BKN/GUfHqWaOGQMNZkBvqSzsZKWk/ijwHlWzzkLulsGHhjWQ==",
"dev": true,
"funding": [
{

View File

@@ -181,9 +181,7 @@
"check:model-lifecycle": "node --import tsx/esm scripts/check/check-model-lifecycle.mjs",
"check:provider-consistency": "bun scripts/check/check-provider-consistency.ts",
"check:provider-assets": "node scripts/check/check-provider-assets.mjs",
"check:provider-order-sync": "node scripts/check/check-provider-order-sync.mjs",
"check:provider-asset-provenance": "node scripts/check/check-provider-asset-provenance.mjs",
"check:radar-sentinels": "node scripts/check/check-radar-sentinels.mjs",
"check:nvidia-catalog-drift": "node --import tsx/esm scripts/check/check-nvidia-catalog-drift.ts",
"check:fetch-targets": "node scripts/check/check-fetch-targets.mjs",
"check:openapi-routes": "node scripts/check/check-openapi-routes.mjs",
@@ -266,7 +264,6 @@
"coverage:report": "cross-env NODE_OPTIONS=--max-old-space-size=8192 c8 report --merge-async --output-dir coverage --exclude=tests/** --exclude=**/*.test.* --reporter=text --reporter=text-summary --reporter=html --reporter=json-summary --reporter=lcov",
"coverage:summary": "node scripts/check/test-report-summary.mjs --input coverage/coverage-summary.json --output coverage/coverage-report.md",
"check:pr-test-policy": "node scripts/check/check-pr-test-policy.mjs",
"check:pricing-freshness": "node scripts/check/check-pricing-freshness.mjs",
"coverage:report:legacy": "c8 report --output-dir coverage --exclude=open-sse --reporter=text --reporter=text-summary",
"test:all": "npm run test:unit && npm run test:vitest && npm run test:vitest:ui && npm run test:ecosystem && npm run test:e2e",
"check": "npm run lint && npm run test",
@@ -486,10 +483,7 @@
"fast-uri": "^3.1.7",
"body-parser": "^2.3.0",
"@yarnpkg/parsers": {
"js-yaml": "^4.3.2"
},
"@openai/codex-security": {
"smol-toml": "^1.8.0"
"js-yaml": "^4.3.1"
},
"jsdom": {
"undici": "^7.29.0"

View File

@@ -27,19 +27,11 @@
// (providers / MCP tools / routing strategies / free-tier pools).
import fs from "node:fs";
import { spawnSync as _spawnSync } from "node:child_process";
import { spawnSync } from "node:child_process";
import os from "node:os";
import path from "node:path";
import { fileURLToPath } from "node:url";
let _spawnSyncImpl = _spawnSync;
export function __setSpawnSyncForTest(fn) {
_spawnSyncImpl = fn;
}
export function __resetSpawnSyncForTest() {
_spawnSyncImpl = _spawnSync;
}
const __dirname = path.dirname(fileURLToPath(import.meta.url));
const ROOT = path.resolve(__dirname, "..", "..");
@@ -149,8 +141,7 @@ export function countLocales() {
// PURE: tally STRICT vs SOFT drift for a list of checks, given a content lookup.
// `getContent(file) -> string | null`. A check whose `actual` is 0 is skipped (the
// source count could not be determined). `actual==="ERR"` is a STRICT failure, not a skip.
// Returns { strict, soft, lines }.
// source count could not be determined). Returns { strict, soft, lines }.
export function tallyDrift(checks, getContent) {
let strict = 0;
let soft = 0;
@@ -158,22 +149,7 @@ export function tallyDrift(checks, getContent) {
for (const c of checks) {
const tier = c.strict ? "STRICT" : "soft";
lines.push(`\n${c.label}: ${c.actual} (real) [${tier}]`);
if (c.actual === "ERR") {
if (c.validate) {
const v = c.validate("", `code facts:${c.actual}`);
lines.push(` ${v.ok ? "✓" : c.strict ? "✗" : "⚠"} ${c.label}${v.detail}`);
if (!v.ok) {
if (c.strict) strict++;
else soft++;
}
} else {
lines.push(` ${c.strict ? "✗" : "⚠"} ${c.label} — readCodeFacts unavailable`);
if (c.strict) strict++;
else soft++;
}
continue;
}
if (c.actual === 0 || c.actual === "0") {
if (!c.actual) {
lines.push(` ⚠ could not determine ${c.docKey} count from source — skipping`);
continue;
}
@@ -202,35 +178,13 @@ export function tallyDrift(checks, getContent) {
return { strict, soft, lines };
}
// Lightweight literal claim helper — checks that the expected string appears in the file.
function makeLiteralClaimValidator(expected, opts) {
return (content) =>
content.includes(String(expected))
? { ok: true, detail: `literal "${expected}" present — ${opts?.what ?? "literal"}` }
: {
ok: false,
detail: `expected literal "${expected}" not found — ${opts?.what ?? "literal"}`,
};
}
// Reads every code-derived fact in ONE tsx subprocess — the same functions the app
// serves at runtime, never a hardcoded copy. DATA_DIR is redirected to a throwaway dir
// so importing the MCP tool modules cannot touch the operator's real SQLite file.
// Returns null when tsx is unavailable — caller must treat it as a failing check, not a skip.
// Returns null when tsx is unavailable so the gate degrades to a skip, not a false red.
function readCodeFacts() {
const script = [
'import {computeFreeModelTotals,FREE_MODEL_BUDGETS} from "./open-sse/config/freeModelCatalog.ts";',
'import {FREE_TIER_PROVIDER_SET} from "./open-sse/config/freeTierProviders.ts";',
'import {generateProviderPluginManifest} from "./open-sse/config/providerPluginManifestRegistry.ts";',
'import {REGISTRY} from "./open-sse/config/providers/index.ts";',
'import fs2 from "node:fs";',
'import path2 from "node:path";',
'const __rtxt=fs2.readFileSync(path2.join(process.cwd(),"src/lib/freeProviderRankings.ts"),"utf8");',
'const __dtxt=fs2.readFileSync(path2.join(process.cwd(),"open-sse/config/freeModelCatalog.data.ts"),"utf8");',
'const __itxt=fs2.readFileSync(path2.join(process.cwd(),"src/lib/combos/intelligentRouting.ts"),"utf8");',
'const __cat=__dtxt.match(/FREE_CATALOG_CURATED_AT\\s*=\\s*"([^"]+)"/)?.[1]??null;',
'const __sb=(__rtxt.match(/sortBy\\?\\s*:\\s*"elo"\\s*\\|\\s*"reliability"/)?"reliability":null);',
'const __ik=__itxt.match(/DEFAULT_INTELLIGENT_WEIGHTS[^=]*=\\s*\\{([\\s\\S]*?)\\n\\};/)?.[1]?.split("\\n").filter(l=>l.includes(":")).length??0;',
'import {MODE_PACKS} from "./open-sse/services/autoCombo/modePacks.ts";',
'import {ENGINE_IDS} from "./open-sse/services/compression/engineCatalog.ts";',
'import {CLI_TOOLS} from "./src/shared/constants/cliTools.ts";',
@@ -276,16 +230,13 @@ function readCodeFacts() {
"freePools:t.poolCount,engines:ENGINE_IDS.length,",
"cliTotal:cli.length,cliCode:by('code'),cliAgent:by('agent'),",
"mcpTools:countUniqueMcpTools(cols),mcpScopes:sc.size,providers:pids.size,freeForever:ff.size,",
"modePacks:Object.keys(MODE_PACKS),catalogDate:__cat,sortBy:__sb,intelligentKeys:__ik,",
"modePacks:Object.keys(MODE_PACKS),",
"hardStop:FREE_MODEL_BUDGETS.filter(e=>e.hardStopGuaranteed===true).length,",
"trainsOnPrompts:FREE_MODEL_BUDGETS.filter(e=>e.trainsOnPrompts===true).length,",
"freeTierCount:FREE_TIER_PROVIDER_SET.size,",
"freeTierReg:FREE_TIER_PROVIDER_SET.size-[...FREE_TIER_PROVIDER_SET].filter(x=>!(x in REGISTRY)).length,",
'manifestFreeTier:generateProviderPluginManifest().providers.filter(p=>p.capabilities.includes("free-tier")).length}));',
"trainsOnPrompts:FREE_MODEL_BUDGETS.filter(e=>e.trainsOnPrompts===true).length}));",
].join("");
const tmp = fs.mkdtempSync(path.join(os.tmpdir(), "docs-counts-"));
try {
const r = _spawnSyncImpl(process.execPath, ["--import", "tsx/esm", "-e", script], {
const r = spawnSync(process.execPath, ["--import", "tsx/esm", "-e", script], {
cwd: ROOT,
encoding: "utf8",
timeout: 180000,
@@ -593,14 +544,10 @@ export function buildChecks() {
return [
{
label: "Code-derived counts",
actual: "ERR",
actual: 0,
docKey: "code facts",
strict: true,
strict: false,
files: [],
validate: () => ({
ok: false,
detail: "readCodeFacts unavailable — tsx/spawnSync failed",
}),
},
];
const claim = (expected, what, opts, files) => ({
@@ -661,22 +608,6 @@ export function buildChecks() {
files: ["docs/routing/AUTO-COMBO.md"],
validate: makeModePackNamesValidator(packs),
},
{
// Every pack must pin `quality` explicitly (6 pins, 0.02/0.03) so
// no pack silently inherits a future DEFAULT. Validates live code,
// not docs — the gate loops `files` content through `validate`.
label: "mode packs pin quality explicitly (live code)",
actual: 6,
docKey: "packs quality pins",
strict: true,
files: ["open-sse/services/autoCombo/modePacks.ts"],
validate: (content) => {
const pins = (content.match(/^\s*quality:\s*0\.0\d,?\s*$/gm) ?? []).length;
return pins >= 6
? { ok: true, detail: `${pins} quality pins` }
: { ok: false, detail: `only ${pins} quality pins — every pack must pin quality` };
},
},
{
label: "Provider reference total (doc vs live modules)",
actual: f.providers,
@@ -685,23 +616,6 @@ export function buildChecks() {
files: ["docs/reference/PROVIDER_REFERENCE.md"],
validate: makeProviderReferenceValidator(f.providers),
},
// Gate: manifest emission vs catalogue intersection. Both numbers are
// live code facts from the same spawnSync computeur. The catalogue can
// name providers the registry does not serve yet (arcee-ai at
// 9d1a896c6), so the leaf raw size is informational — the gate
// compares the intersected count, never the raw size, and the script
// itself always exists so tallyDrift runs validate (skips null only).
{
label: "Manifest free-tier capability count (live code)",
actual: f.manifestFreeTier,
docKey: "free-tier capability",
strict: true,
files: ["scripts/check/check-docs-counts-sync.mjs"],
validate: () => ({
ok: f.manifestFreeTier === f.freeTierReg,
detail: `manifest ${f.manifestFreeTier} vs leaf∩registry ${f.freeTierReg} (leaf ${f.freeTierCount})`,
}),
},
{
label: "SVG canonical numbers (live code)",
actual:
@@ -808,82 +722,17 @@ export function buildChecks() {
f.trainsOnPrompts,
"training-disclosure entries",
{
// `requireClaim`: this page is the one place that states the number,
// so a reworded or deleted sentence must fail rather than pass as
// "no claim in this file" — otherwise the gate is one edit from silent.
requireClaim: true,
pattern:
/(\d+) entr(?:y|ies) (?:that )?(?:carry|carries) a (?:prompt-)?training disclosure/gi,
},
["docs/reference/FREE_TIERS.md"]
),
{
label: "Free provider rankings sortBy (live code)",
actual: f.sortBy ?? "reliability",
docKey: "rankings sortBy",
strict: true,
files: ["src/lib/freeProviderRankings.ts", "src/app/api/free-provider-rankings/route.ts"],
validate: (content) => {
// Disjunctive: the gate loops over two files with different shapes —
// freeProviderRankings.ts carries the union + branch, route.ts the z.enum.
const hasUnion = /sortBy\?\s*:\s*"elo"\s*\|\s*"reliability"/.test(content);
const hasReliabilityBranch =
/sortBy\s*===\s*"reliability"|sortBy\s*!==\s*"reliability"/.test(content);
const hasZEnum = /z\.enum\(\["elo",\s*"reliability"\]\)/.test(content);
return (hasUnion && hasReliabilityBranch) || hasZEnum
? { ok: true, detail: "union+branch (rankings) or z.enum (route) present" }
: { ok: false, detail: "ELO-only regression: union+branch and z.enum both missing" };
},
},
{
label: "FREE_CATALOG_CURATED_AT (live code)",
actual: f.catalogDate ?? 0,
docKey: "FREE_CATALOG_CURATED_AT",
strict: false,
files: ["open-sse/config/freeModelCatalog.data.ts"],
validate: makeLiteralClaimValidator(f.catalogDate, { what: "FREE_CATALOG_CURATED_AT" }),
},
// Duplicate coverage with combo-scoring-weights-schema-coverage.test.ts — soft gate only.
{
label: "INTELLIGENT vs DEFAULT (live code)",
actual: f.intelligentKeys ?? 16,
docKey: "INTELLIGENT vs DEFAULT",
strict: false,
files: ["src/lib/combos/intelligentRouting.ts", "open-sse/services/autoCombo/scoring.ts"],
validate: (content) =>
content.includes("DEFAULT_INTELLIGENT_WEIGHTS") || content.includes("DEFAULT_WEIGHTS")
? {
ok: true,
detail:
"weight constant present (strict coverage in combo-scoring-weights-schema-coverage:66)",
}
: { ok: false, detail: "no weight constant found" },
},
];
})(),
{
label: "ToS caution (16) (live docs)",
actual: 16,
docKey: "ToS caution (16)",
strict: false,
files: ["docs/reference/FREE_TIERS.md"],
validate: makeNumberClaimValidator(16, {
what: "ToS caution (16)",
pattern: /Caution[^\n]*\(\s*(16)\s*\)/gi,
requireClaim: true,
}),
},
{
label: "quality neutral prose (live docs)",
actual: "quality neutral 0.5",
docKey: "quality neutral",
strict: false,
files: ["docs/routing/AUTO-COMBO.md"],
validate: (content) =>
/quality.*neutral.*0\.5/is.test(content)
? { ok: true, detail: "quality neutral 0.5 mentioned" }
: {
ok: false,
detail: "quality neutral 0.5 not found — prose must carry quality neutral 0.5",
},
},
{
label: "Executors count",
actual: countFiles("open-sse/executors"),

View File

@@ -1,26 +0,0 @@
#!/usr/bin/env node
// STRICT gate: KNOWN_MODEL_PRICING must have been touched within PRICING_STALE_AFTER_DAYS,
// or tiers silently drift from real prices. Exits 1 on stale, 0 on fresh.
// Run: node scripts/check/check-pricing-freshness.mjs
import { execFileSync } from "node:child_process";
const PRICING_STALE_AFTER_DAYS = 90;
const TARGET = "open-sse/services/providerCostData.ts";
function lastTouchDays() {
const out = execFileSync("git", ["log", "--follow", "-1", "--format=%ct", "--", TARGET], {
encoding: "utf8",
}).trim();
const touched = Number(out);
if (!Number.isFinite(touched) || touched <= 0) return Infinity;
return (Date.now() / 1000 - touched) / 86400;
}
const days = lastTouchDays();
if (days > PRICING_STALE_AFTER_DAYS) {
console.error(
`STALE: ${TARGET} untouched for ${Math.floor(days)}d (> ${PRICING_STALE_AFTER_DAYS}d) — refresh prices or bump the gate with justification`
);
process.exit(1);
}
console.log(`pricing fresh: ${TARGET} touched ${Math.floor(days)}d ago`);

View File

@@ -1,87 +0,0 @@
#!/usr/bin/env node
// Checks the canonical provider order stays single-sourced.
//
// STRICT (blocking, exit 1): the order derivation exists exactly once
// (canonicalProviderOrder.ts); catalogOrder.ts and comboSort.ts re-export it
// (grep-negative for a local Object.keys(OAUTH derivation); re-export present);
// the xao alias stays declared under xai-oauth; the unknown-provider contract
// (?? Infinity + codeUnitCompare) stays live.
// INFORMATIVE (warn only, always exit 0): dashboard quota ranks in
// ProviderLimits/constants.ts are a separate display order — a registry id
// missing there is fine when the provider has no quota surface.
// Run: node scripts/check/check-provider-order-sync.mjs
import fs from "node:fs";
import path from "node:path";
import { fileURLToPath } from "node:url";
const __dirname = path.dirname(fileURLToPath(import.meta.url));
const ROOT = path.resolve(__dirname, "..", "..");
const read = (p) => fs.readFileSync(path.join(ROOT, p), "utf8");
let failures = 0;
const fail = (msg) => {
failures += 1;
console.error(`[provider-order-sync] FAIL ${msg}`);
};
// 1+2. Leaf uniqueness.
for (const f of ["src/app/api/v1/models/catalogOrder.ts", "src/lib/combos/comboSort.ts"]) {
if (read(f).includes("Object.keys(OAUTH")) {
fail(`${f} defines its own order derivation (import the leaf instead)`);
}
}
// 3. xao alias form.
const oauth = read("src/shared/constants/providers/oauth.ts");
if (!/"xai-oauth":\s*\{[^}]*alias:\s*"xao"/s.test(oauth)) {
fail(`oauth.ts: "xai-oauth" entry lost its alias:"xao"`);
}
// 4. Unknown contract alive.
if (!read("src/shared/constants/canonicalProviderOrder.ts").includes("?? Infinity")) {
fail(`canonicalProviderOrder.ts lost the ?? Infinity unknown contract`);
}
if (!read("src/app/api/v1/models/catalogOrder.ts").includes("codeUnitCompare")) {
fail(`catalogOrder.ts lost the codeUnitCompare unknown-branch`);
}
// 5. Re-export intact.
if (!read("src/lib/combos/comboSort.ts").includes("CANONICAL_PROVIDER_ORDER as PROVIDER_ORDER")) {
fail(`comboSort.ts lost the PROVIDER_ORDER re-export`);
}
// Informative dashboard check (never fails).
try {
const dash = read("src/app/(dashboard)/dashboard/usage/components/ProviderLimits/constants.ts");
const ids = new Set();
// NOTE: apikey/index.ts is a barrel (imports + spreads, zero literal entries),
// so glob the family files instead.
const apikeyFiles = fs
.readdirSync(path.join(ROOT, "src/shared/constants/providers/apikey"))
.filter((f) => f.endsWith(".ts") && !f.endsWith(".test.ts"))
.map((f) => `src/shared/constants/providers/apikey/${f}`);
for (const f of [
"src/shared/constants/providers/oauth.ts",
"src/shared/constants/providers/noauth.ts",
...apikeyFiles,
]) {
let src;
try {
src = read(f);
} catch {
continue;
}
for (const m of src.matchAll(/^ "([^"]+)": \{$/gm)) ids.add(m[1]);
}
for (const id of [...ids].sort()) {
if (!dash.includes(`"${id}"`) && !dash.includes(`${id}:`)) {
console.log(
`[provider-order-sync] warn: ${id} has no dashboard quota rank (ok if no quota surface)`
);
}
}
} catch (e) {
console.log(`[provider-order-sync] warn: dashboard check skipped (${String(e).slice(0, 120)})`);
}
if (failures > 0) {
process.exit(1);
}
console.log("[provider-order-sync] OK — single source + alias form + unknown contract");

View File

@@ -1,94 +0,0 @@
#!/usr/bin/env node
/**
* OmniRoute — Radar silent-cell structural gate.
*
* Asserts three structural invariants over the Radar catalog table —
* every absent-data sentinel in that table must carry an explaining title.
* (Rationale lives in the PR description; this header is self-contained on
* purpose — no local doc paths here, they would leak into the upstream diff.
* Sized to 3 known sites — extend with a 4th check if a new silent cell lands.):
* 1. limits cell wraps {formatLimits(entry)} in <span title={t("limitsUnknownHelp")}> on the "—" branch
* 2. context cell wraps the "—" branch in <span title={t("contextUnknownHelp")}>
* 3. capabilityBadge's "?" span carries title={t("capabilityUnknownHelp")}
*
* Single purpose: structural invariant over one file. i18n coverage lives in
* scripts/i18n/check-ui-keys-coverage.mjs — a red run here means a sentinel
* without explanation, never a locale threshold.
*
* Usage: npm run check:radar-sentinels (exit 1 on any broken invariant)
*/
import { readFileSync } from "node:fs";
import path from "node:path";
import process from "node:process";
import { fileURLToPath } from "node:url";
const SCRIPT_DIR = path.dirname(fileURLToPath(import.meta.url));
const ROOT = path.resolve(SCRIPT_DIR, "..", "..");
const TABLE = path.join(
ROOT,
"src",
"app",
"(dashboard)",
"dashboard",
"radar",
"RadarCatalogTable.tsx"
);
function fail(message) {
console.error(`[radar-sentinels] FAIL ${message}`);
}
function main() {
const raw = readFileSync(TABLE, "utf8");
const src = raw.replace(/\/\/.*|\/\*[\s\S]*?\*\//g, ""); // commented call sites must not count (no URLs in this file at pin)
const failures = [];
// Invariant 1 — limits cell: the "—" branch of formatLimits is wrapped with the help title.
// Sized to the implementation shape (const limits + conditional span). Window {0,500}:
// prettier rewraps must not false-red. If a 4th silent cell is added one day,
// extend here with a 4th check — this gate is deliberately sized to known sites,
// not generic (static-explicit beats magic-generic for a 1-file, 3-site gate).
if (
!/const limits = formatLimits\(entry\)[\s\S]{0,500}<span title=\{t\("limitsUnknownHelp"\)\}>\{limits\}<\/span>/.test(
src
)
) {
failures.push(
'limits cell: expected const limits + "—" branch wrapped in <span title={t("limitsUnknownHelp")}>{limits}</span>'
);
}
// Invariant 2 — context cell: the "—" branch carries the context help title.
if (!/<span title=\{t\("contextUnknownHelp"\)\}>—<\/span>/.test(src)) {
failures.push('context cell: expected <span title={t("contextUnknownHelp")}>—</span>');
}
// Invariant 3 — capabilityBadge: the unknown-state span carries the capability help title.
// Post-fix shape (param form): helper takes `unknownHelp: string`,
// ALL THREE call sites (tools/vision/thinking) pass t("capabilityUnknownHelp") —
// count, not exist (one dropped site must go red), and the span spreads
// `{...(unknown ? { title: unknownHelp } : {})}` (unconditional title would spam ✓/✕ badges).
const helpSites = src.match(/t\("capabilityUnknownHelp"\)/g) || [];
if (helpSites.length < 3) {
failures.push(
`capability badge: expected t("capabilityUnknownHelp") at all 3 call sites, found ${helpSites.length}`
);
}
if (!/title:\s*unknownHelp/.test(src)) {
failures.push(
"capability badge: expected conditional title from unknownHelp on the badge span"
);
}
if (failures.length) {
for (const f of failures) fail(f);
console.error(
`[radar-sentinels] ${failures.length} broken invariant(s) in RadarCatalogTable.tsx`
);
process.exit(1);
}
console.log("[radar-sentinels] all invariants hold");
}
main();

View File

@@ -29,8 +29,6 @@ const GATES = [
// Group B — fast (<5s)
{ name: "check:provider-consistency", cmd: ["node", "--import", "tsx", "scripts/check/check-provider-consistency.ts"] },
{ name: "check:provider-assets", cmd: ["node", "scripts/check/check-provider-assets.mjs"] },
{ name: "check:pricing-freshness", cmd: ["node", "scripts/check/check-pricing-freshness.mjs"] },
{ name: "check:provider-order-sync", cmd: ["node", "scripts/check/check-provider-order-sync.mjs"] },
{ name: "check:public-creds", cmd: ["node", "scripts/check/check-public-creds.mjs"] },
{ name: "check:error-helper", cmd: ["node", "scripts/check/check-error-helper.mjs"] },
{ name: "check:fetch-targets", cmd: ["node", "scripts/check/check-fetch-targets.mjs"] },

View File

@@ -4,7 +4,6 @@ import { useEffect, useMemo, useState } from "react";
import type { NodeTypes } from "@xyflow/react";
import { useTranslations } from "next-intl";
import { FlowCanvas } from "@/shared/components/flow/FlowCanvas";
import { FLOW_EDGE_COLORS, flowColorAlpha } from "@/shared/components/flow/edgeStyles";
import {
comboRunToFlow,
reduceComboEvent,
@@ -82,10 +81,7 @@ function FleetOverview({ comboEvents }: FleetOverviewProps) {
<span
key={p}
className="text-[11px] px-2 py-0.5 rounded-full font-medium"
style={{
backgroundColor: flowColorAlpha(FLOW_EDGE_COLORS.active, 13),
color: FLOW_EDGE_COLORS.active,
}}
style={{ backgroundColor: "#22c55e20", color: "#22c55e" }}
>
{p}
</span>
@@ -103,10 +99,7 @@ function FleetOverview({ comboEvents }: FleetOverviewProps) {
<span
key={p}
className="text-[11px] px-2 py-0.5 rounded-full font-medium"
style={{
backgroundColor: flowColorAlpha(FLOW_EDGE_COLORS.error, 13),
color: FLOW_EDGE_COLORS.error,
}}
style={{ backgroundColor: "#ef444420", color: "#ef4444" }}
>
{p}
</span>
@@ -328,10 +321,10 @@ export function ComboLiveStudio({
style={{
color:
displayRun.outcome === "succeeded"
? "var(--orch-status-success)"
? "#22c55e"
: displayRun.outcome === "exhausted"
? "var(--orch-status-error)"
: "var(--orch-status-warning)",
? "#ef4444"
: "#f59e0b",
}}
data-testid="run-outcome"
>

View File

@@ -3,7 +3,7 @@
import { Handle, Position, type NodeProps } from "@xyflow/react";
import ProviderIcon from "@/shared/components/ProviderIcon";
import { StatusDot } from "@/shared/components/flow/StatusDot";
import { FLOW_EDGE_COLORS, flowColorAlpha } from "@/shared/components/flow/edgeStyles";
import { FLOW_EDGE_COLORS } from "@/shared/components/flow/edgeStyles";
import type { TargetState, FailKind, CbState } from "../comboFlowModel";
// ── State → visual mapping ────────────────────────────────────────────────
@@ -26,11 +26,11 @@ function getStateBorderColor(state: TargetState): string {
function getStateGlow(state: TargetState): string {
switch (state) {
case "attempting":
return `0 0 12px ${flowColorAlpha(FLOW_EDGE_COLORS.last, 25)}`;
return `0 0 12px ${FLOW_EDGE_COLORS.last}40`;
case "failed":
return `0 0 12px ${flowColorAlpha(FLOW_EDGE_COLORS.error, 25)}`;
return `0 0 12px ${FLOW_EDGE_COLORS.error}40`;
case "succeeded":
return `0 0 12px ${flowColorAlpha(FLOW_EDGE_COLORS.active, 25)}`;
return `0 0 12px ${FLOW_EDGE_COLORS.active}40`;
default:
return "none";
}
@@ -190,7 +190,7 @@ export function ProviderCascadeNode({ data }: NodeProps) {
<span
className="text-[9px] font-semibold px-1.5 py-0.5 rounded"
style={{
backgroundColor: flowColorAlpha(FLOW_EDGE_COLORS.error, 13),
backgroundColor: `${FLOW_EDGE_COLORS.error}20`,
color: FLOW_EDGE_COLORS.error,
}}
data-testid="fail-kind-badge"

View File

@@ -1,7 +1,7 @@
"use client";
import { Handle, Position, type NodeProps } from "@xyflow/react";
import { FLOW_EDGE_COLORS, flowColorAlpha } from "@/shared/components/flow/edgeStyles";
import { FLOW_EDGE_COLORS } from "@/shared/components/flow/edgeStyles";
import type { ComboRunModel } from "../comboFlowModel";
// ── Node data shape ───────────────────────────────────────────────────────
@@ -53,7 +53,7 @@ export function ResponseNode({ data }: NodeProps) {
className="flex flex-col items-center gap-1.5 px-3 py-2 rounded-lg border-2 bg-bg transition-all duration-300"
style={{
borderColor: color,
boxShadow: `0 0 10px ${flowColorAlpha(color, 19)}`,
boxShadow: `0 0 10px ${color}30`,
minWidth: "100px",
}}
data-testid="response-node"

View File

@@ -213,7 +213,7 @@ export function CompressionCockpit({ run: runProp }: CompressionCockpitProps) {
<span className="text-[11px] text-muted">
{fmt(run.originalTokens)} {fmt(run.compressedTokens)} {t("tokenShort")}
</span>
<span className="text-xs font-bold" style={{ color: "var(--orch-status-success)" }}>
<span className="text-xs font-bold" style={{ color: "#22c55e" }}>
{run.savingsPercent.toFixed(1)}%
</span>
{isComplete && view === "canvas" && (

View File

@@ -5,16 +5,12 @@ import { useTranslations } from "next-intl";
// ── Helpers ───────────────────────────────────────────────────────────────
/** Savings quality ramp — same thresholds/tokens as `EngineNode.getSavingsColor`. */
function savingsColor(pct: number): string {
if (pct >= 30) return "var(--orch-status-success)";
if (pct >= 15) return "var(--orch-status-warning)";
return "var(--orch-status-muted)";
if (pct >= 30) return "#22c55e";
if (pct >= 15) return "#f59e0b";
return "#6b7280";
}
/** A skipped step reads as idle. */
const SKIPPED_COLOR = "var(--orch-status-muted)";
function pctWidth(tokIn: number, tokOut: number): string {
if (tokIn === 0) return "100%";
return `${((tokOut / tokIn) * 100).toFixed(1)}%`;
@@ -29,7 +25,7 @@ function fmt(n: number): string {
function StepRow({ step, maxTokens }: { step: CompressionEngineStep; maxTokens: number }) {
const t = useTranslations("compressionStudio");
const skipped = step.originalTokens === step.compressedTokens;
const color = skipped ? SKIPPED_COLOR : savingsColor(step.savingsPercent);
const color = skipped ? "#6b7280" : savingsColor(step.savingsPercent);
const barWidthIn = maxTokens > 0 ? (step.originalTokens / maxTokens) * 100 : 100;
const barWidthOut = maxTokens > 0 ? (step.compressedTokens / maxTokens) * 100 : 100;
@@ -40,7 +36,7 @@ function StepRow({ step, maxTokens }: { step: CompressionEngineStep; maxTokens:
>
{/* Engine label */}
<div className="w-28 shrink-0">
<span className="text-xs font-semibold" style={{ color: skipped ? SKIPPED_COLOR : color }}>
<span className="text-xs font-semibold" style={{ color: skipped ? "#6b7280" : color }}>
{step.engine}
</span>
{skipped && (
@@ -151,7 +147,7 @@ export function WaterfallInspector({ run, className = "" }: WaterfallInspectorPr
</div>
<div
className="text-[11px] font-bold"
style={{ color: "var(--orch-status-success)" }}
style={{ color: "#22c55e" }}
data-testid="waterfall-total-savings"
>
{`-${run.savingsPercent.toFixed(1)}%`}

View File

@@ -2,7 +2,6 @@
import { Handle, Position, type NodeProps } from "@xyflow/react";
import { StatusDot } from "@/shared/components/flow/StatusDot";
import { flowColorAlpha } from "@/shared/components/flow/edgeStyles";
// ── Layer pill color map (10-layer model) ─────────────────────────────────
@@ -15,12 +14,6 @@ const LAYER_COLORS: Record<string, string> = {
L9: "#ec4899", // pruning — pink
};
/**
* In-flight engine step. Mirrors the Phase-2 orchestration mapping `running ->
* var(--orch-status-warning)` (`orchestrationTypes.ts::STATE_VAR`).
*/
const RUNNING_COLOR = "var(--orch-status-warning)";
/** Map engine name → layer tags for the pill display */
const ENGINE_LAYER_MAP: Record<string, string[]> = {
rtk: ["L3", "L4"],
@@ -35,15 +28,10 @@ const ENGINE_LAYER_MAP: Record<string, string[]> = {
"rtk:standard": ["L3", "L4"],
};
/**
* Savings quality ramp — good / mediocre / none. Theme-aware `--orch-status-*` tokens
* (light values in `:root`, dark values in `.dark` of `src/app/globals.css`); the dark
* values are the previous hexes, so dark mode is byte-identical.
*/
function getSavingsColor(savingsPercent: number): string {
if (savingsPercent >= 30) return "var(--orch-status-success)";
if (savingsPercent >= 15) return "var(--orch-status-warning)";
return "var(--orch-status-muted)";
if (savingsPercent >= 30) return "#22c55e";
if (savingsPercent >= 15) return "#f59e0b";
return "#6b7280";
}
// ── Node data shape ───────────────────────────────────────────────────────
@@ -84,18 +72,14 @@ export function EngineNode({ data }: NodeProps) {
const tokOut = compressedTokens as number;
const techniques = (techniquesUsed as string[]).slice(0, 2);
const borderColor = skipped ? "var(--color-border)" : running ? RUNNING_COLOR : color;
const borderColor = skipped ? "var(--color-border)" : running ? "#f59e0b" : color;
return (
<div
className="rounded-lg border-2 bg-bg transition-all duration-300 min-w-[150px] max-w-[180px]"
style={{
borderColor,
boxShadow: running
? `0 0 14px ${flowColorAlpha(RUNNING_COLOR, 25)}`
: skipped
? "none"
: `0 0 10px ${flowColorAlpha(color, 19)}`,
boxShadow: running ? `0 0 14px #f59e0b40` : skipped ? "none" : `0 0 10px ${color}30`,
}}
>
<Handle
@@ -114,7 +98,7 @@ export function EngineNode({ data }: NodeProps) {
className="flex items-center gap-1.5 px-2.5 pt-2 pb-1"
style={{ borderBottom: "1px solid var(--color-border)" }}
>
{running && <StatusDot color={RUNNING_COLOR} sizeClass="size-1.5" />}
{running && <StatusDot color="#f59e0b" sizeClass="size-1.5" />}
<span className="text-xs font-semibold truncate flex-1" title={engine as string}>
{engine as string}
</span>

View File

@@ -55,7 +55,7 @@ export function IoNode({ data }: NodeProps) {
{!isInput && savingsPercent != null && (
<div
className="mt-1 text-[11px] font-semibold"
style={{ color: "var(--orch-status-success)" }}
style={{ color: "#22c55e" }}
data-testid="io-savings-percent"
>
{`-${(savingsPercent as number).toFixed(1)}%`}

View File

@@ -8,7 +8,7 @@ import { copyToClipboard } from "@/shared/utils/clipboard";
import RequestLoggerDetail from "@/shared/components/RequestLoggerDetail";
import useEmailPrivacyStore from "@/store/emailPrivacyStore";
import { ChatBubble } from "@/app/(dashboard)/dashboard/tools/traffic-inspector/components/chat/ChatBubble";
import { toTurn, type ConversationTurn } from "./toTurn";
import type { NormalizedBlock, NormalizedTurn } from "@/mitm/inspector/types";
interface ConversationRow {
id: string;
@@ -31,12 +31,6 @@ interface ConversationRow {
// only means the client-side content-hash tracker saw >= 2 turns
// regardless of transport (see isGenuineContinuationTurn).
isGenuineContinuation: boolean;
// The last turn never reached a clean "stop" (truncated/failed stream, or
// a tool call still unanswered) AND 5+ minutes have passed with nothing
// having continued — see resolveConversationStalledState's own doc
// comment for why a bare unanswered tool call alone doesn't count (that's
// normal seconds after it lands). Never true while isActive.
isStalled: boolean;
}
// Same spinner used for an in-flight request on /dashboard/logs
@@ -53,6 +47,17 @@ function ActiveSpinner() {
);
}
interface ConversationTurn {
seq: number;
id: string;
parentId: string | null;
role: string;
textPreview: string;
blockKind: string;
toolName: string | null;
firstSeenAt: string;
}
interface ConversationTurnsPage {
nodes: ConversationTurn[];
hasMore: boolean;
@@ -118,24 +123,43 @@ function ContinuationBadge({ isGenuine }: { isGenuine: boolean }) {
);
}
// Flags a conversation whose latest turn never reached a clean "stop" --
// a truncated/failed stream, or a tool call still unanswered 5+ minutes
// after the last activity with nothing having continued (see
// resolveConversationStalledState -- a bare unanswered tool call alone is
// completely normal seconds after it lands, so this only fires once the
// grace period has actually elapsed). Server-computed so this badge never
// disagrees with the actual persisted artifact state.
function StalledBadge({ isStalled }: { isStalled: boolean }) {
if (!isStalled) return null;
return (
<span
title="Latest turn didn't end in stop and nothing has continued for 5+ minutes"
className="inline-flex items-center gap-0.5 px-2 py-0.5 rounded-full text-[9px] font-bold bg-red-500/15 text-red-500 border border-red-500/25"
>
<span className="material-symbols-outlined text-[11px] leading-none">error</span>
stalled
</span>
);
/**
* Builds the exact NormalizedBlock (src/mitm/inspector/types.ts) the
* request-detail panel already builds from buildRequestTurns/
* buildResponseTurns, so a tool call/result renders through the very same
* ChatBubble → MessageContent → ToolCallBlock/ToolResultBlock pipeline as
* the detail view — not a parallel implementation. `textPreview` round-
* tripped through JSON for a structured tool_use/tool_result turn; parse it
* best-effort so the block gets a real object, not a JSON string.
*/
function toTurn(node: ConversationTurn): NormalizedTurn {
const role: NormalizedTurn["role"] =
node.role === "system" || node.role === "user" || node.role === "assistant"
? node.role
: "tool";
let block: NormalizedBlock;
if (node.blockKind === "tool_use") {
let input: unknown = node.textPreview;
try {
input = JSON.parse(node.textPreview);
} catch {
// Arguments weren't valid JSON — show the raw string.
}
block = { type: "tool_use", id: node.id.slice(0, 12), name: node.toolName ?? "tool", input };
} else if (node.blockKind === "tool_result") {
let content: unknown = node.textPreview;
try {
content = JSON.parse(node.textPreview);
} catch {
// Not JSON — show the raw string.
}
block = { type: "tool_result", tool_use_id: node.id.slice(0, 12), content };
} else {
block = { type: "text", text: node.textPreview || "_(empty)_" };
}
return { role, blocks: [block], timestamp: node.firstSeenAt };
}
/**
@@ -714,7 +738,6 @@ function ConversationsPageContent() {
isActive: false,
activeCallLogId: null,
isGenuineContinuation: false,
isStalled: false,
}
);
}, [initialConversationParam, loading, conversations, openConversation]);
@@ -820,7 +843,6 @@ function ConversationsPageContent() {
{row.id.slice(0, 16)}
</span>
<ContinuationBadge isGenuine={row.isGenuineContinuation} />
<StalledBadge isStalled={row.isStalled} />
</span>
<span className="font-mono text-xs text-text-muted shrink-0">
{row.turnCount} turns
@@ -881,7 +903,6 @@ function ConversationsPageContent() {
</td>
<td className="px-3 py-2">
<ContinuationBadge isGenuine={row.isGenuineContinuation} />
<StalledBadge isStalled={row.isStalled} />
</td>
<td className="px-3 py-2 text-text-main">{row.lastModel ?? "—"}</td>
<td className="px-3 py-2">

View File

@@ -1,71 +0,0 @@
import type { NormalizedBlock, NormalizedTurn } from "@/mitm/inspector/types";
export interface ConversationTurn {
seq: number;
id: string;
parentId: string | null;
role: string;
textPreview: string;
blockKind: string;
toolName: string | null;
firstSeenAt: string;
}
/**
* Builds the exact NormalizedBlock (src/mitm/inspector/types.ts) the
* request-detail panel already builds from buildRequestTurns/
* buildResponseTurns, so a tool call/result renders through the very same
* ChatBubble → MessageContent → ToolCallBlock/ToolResultBlock pipeline as
* the detail view — not a parallel implementation. `textPreview` round-
* tripped through JSON for a structured tool_use/tool_result turn; parse it
* best-effort so the block gets a real object, not a JSON string.
*
* Kept out of page.tsx (a "use client" component that pulls in ChatBubble/
* MarkdownMessage's dependency tree) so this pure ConversationTurn ->
* NormalizedTurn mapping stays unit-testable on its own.
*/
export function toTurn(node: ConversationTurn): NormalizedTurn {
const role: NormalizedTurn["role"] =
node.role === "system" || node.role === "user" || node.role === "assistant"
? node.role
: "tool";
let block: NormalizedBlock;
if (node.blockKind === "tool_use") {
let input: unknown = node.textPreview;
try {
input = JSON.parse(node.textPreview);
} catch {
// Arguments weren't valid JSON — show the raw string.
}
block = { type: "tool_use", id: node.id.slice(0, 12), name: node.toolName ?? "tool", input };
} else if (node.blockKind === "tool_result") {
let content: unknown = node.textPreview;
try {
content = JSON.parse(node.textPreview);
} catch {
// Not JSON — show the raw string.
}
block = { type: "tool_result", tool_use_id: node.id.slice(0, 12), content };
} else if (!node.textPreview) {
// The tree API records a node's identity (role/blockKind) the moment
// the turn is recorded, independent of when its display content
// resolves from the owning call-log artifact -- a request still in
// flight has a real node (any role: user, assistant, or tool) with
// nothing to show yet, and /api/conversations/[id]/tree's own
// blockKind ?? "text" fallback can't tell that apart from a
// permanently-purged artifact. Live traffic shows this lag hits every
// role, not just tool nodes (a user/assistant turn's own textPreview
// resolves through the same lazy pipeline) -- so any empty text node
// gets the same treatment. In practice this is near-always transient:
// the same poll that already refreshes this page (ConversationLogView's
// activeCallLogId-driven effect) picks up the real content within a
// tick or two once the artifact lands. Show that instead of a bare
// "(empty)" that reads as broken rather than in progress.
block = { type: "pending" };
} else {
block = { type: "text", text: node.textPreview || "_(empty)_" };
}
return { role, blocks: [block], timestamp: node.firstSeenAt };
}

View File

@@ -12,7 +12,6 @@ import { HistoryTab } from "./tabs/HistoryTab";
import { OrchestrationDrawer } from "./drawer/OrchestrationDrawer";
import { OrchestrationToolbar } from "./OrchestrationToolbar";
import { collectProviderKeys, filterSnapshot } from "./model/filterSnapshot";
import { parseCsvSet, toggleCsv } from "./model/urlParams";
import type { OrchFilter } from "./model/filterSnapshot";
import { ORCH_STATES } from "./model/orchestrationTypes";
import type { OrchSource, OrchState } from "./model/orchestrationTypes";
@@ -23,6 +22,25 @@ type Tab = (typeof TABS)[number];
const VALID_STATES: ReadonlySet<OrchState> = new Set(ORCH_STATES);
const VALID_SOURCES: ReadonlySet<OrchSource> = new Set(["cloud-agent", "a2a", "conductor"]);
/** CSV → Set, dropping empty/invalid entries (`valid` omitted accepts any non-empty token). */
function parseCsvSet<T extends string>(raw: string | null, valid?: ReadonlySet<T>): Set<T> {
const out = new Set<T>();
if (!raw) return out;
for (const v of raw.split(",")) {
if (!v) continue;
if (!valid || valid.has(v as T)) out.add(v as T);
}
return out;
}
/** Toggle `value` in `current`, returning the next CSV (or `null` to drop the param). */
function toggleCsv<T extends string>(current: ReadonlySet<T>, value: T): string | null {
const next = new Set(current);
if (next.has(value)) next.delete(value);
else next.add(value);
return next.size > 0 ? [...next].sort().join(",") : null;
}
const TAB_KEY: Record<Tab, string> = {
agents: "tabAgents",
routing: "tabRouting",
@@ -115,17 +133,6 @@ export default function OrchestrationPageClient() {
[collapsed, setParams]
);
const closeDrawer = useCallback(() => setParams({ node: null }), [setParams]);
const onActionDone = useCallback(
(newNodeId?: string) => {
refetch();
if (newNodeId) setParams({ node: newNodeId });
},
[refetch, setParams]
);
const clearFilters = useCallback(
() => setParams({ q: null, state: null, source: null, provider: null }),
[setParams]
);
const selectedNode = nodeId ? (snapshot.nodes.find((n) => n.id === nodeId) ?? null) : null;
const onNodeClick = (id: string) =>
@@ -159,8 +166,6 @@ export default function OrchestrationPageClient() {
onToggleCompleted={setShowCompleted}
collapsed={collapsed}
onToggleCollapse={onToggleCollapse}
filter={filter}
onClearFilters={clearFilters}
/>
)}
{tab === "routing" && (
@@ -183,17 +188,8 @@ export default function OrchestrationPageClient() {
{tab === "history" && <HistoryTab />}
</div>
</div>
{/* A successful repeat hands back the CANVAS id of the task it created: refetch, then
focus it (`?node=`) so the operator lands on the new run instead of staring at the
finished one. No id (approve/cancel, or a creation response without one) keeps the
current selection. The History tab renders its own drawer and deliberately does NOT
navigate (HistoryTab.tsx) — its runs are not addressable in the live snapshot. */}
{tab !== "history" && (
<OrchestrationDrawer
node={selectedNode}
onClose={closeDrawer}
onActionDone={onActionDone}
/>
<OrchestrationDrawer node={selectedNode} onClose={closeDrawer} onActionDone={refetch} />
)}
</div>
);

View File

@@ -8,7 +8,6 @@
import { useEffect, useRef, useState } from "react";
import { useTranslations } from "next-intl";
import { isEmptyFilter } from "./model/filterSnapshot";
import { toggleCsv } from "./model/urlParams";
import type { OrchFilter } from "./model/filterSnapshot";
import { ORCH_STATES } from "./model/orchestrationTypes";
import type { OrchSource, OrchState } from "./model/orchestrationTypes";
@@ -31,6 +30,14 @@ const SOURCE_KEY: Record<(typeof SOURCES)[number], string> = {
const SEARCH_DEBOUNCE_MS = 300;
/** Toggle `value` in `current`, returning the next CSV (or `null` to drop the param). */
function toggleCsv<T extends string>(current: ReadonlySet<T>, value: T): string | null {
const next = new Set(current);
if (next.has(value)) next.delete(value);
else next.add(value);
return next.size > 0 ? [...next].sort().join(",") : null;
}
const chipClass = (active: boolean) =>
`text-[10px] px-2 py-0.5 rounded-full border whitespace-nowrap ${
active ? "border-primary bg-primary/10 text-primary" : "border-border text-muted"
@@ -92,31 +99,16 @@ export function OrchestrationToolbar({
[]
);
const cancelPendingSearch = () => {
if (timerRef.current) clearTimeout(timerRef.current);
timerRef.current = null;
};
const handleSearchChange = (v: string) => {
setText(v);
cancelPendingSearch();
if (timerRef.current) clearTimeout(timerRef.current);
timerRef.current = setTimeout(() => setParams({ q: v || null }), SEARCH_DEBOUNCE_MS);
};
/**
* Every chip write goes through here so a debounce timer armed by a keystroke moments
* earlier is dropped first. Left pending it would fire ~300ms later with a `setParams`
* closed over the PRE-chip search params and silently revert the chip the user just
* clicked (the writer rebuilds the whole query string from `params`).
*/
const patchParams = (patch: Record<string, string | null>) => {
cancelPendingSearch();
setParams(patch);
};
const handleClear = () => {
setText("");
patchParams({ q: null, state: null, source: null, provider: null });
if (timerRef.current) clearTimeout(timerRef.current);
setParams({ q: null, state: null, source: null, provider: null });
};
return (
@@ -126,7 +118,6 @@ export function OrchestrationToolbar({
value={text}
onChange={(e) => handleSearchChange(e.target.value)}
placeholder={t("searchPlaceholder")}
aria-label={t("searchPlaceholder")}
className="text-xs px-2 py-1 rounded border border-border bg-transparent min-w-[160px]"
/>
<ChipGroup
@@ -134,14 +125,14 @@ export function OrchestrationToolbar({
values={ORCH_STATES}
active={filter.states}
renderLabel={(s) => t(STATE_KEY[s])}
onToggle={(s) => patchParams({ state: toggleCsv(filter.states, s) })}
onToggle={(s) => setParams({ state: toggleCsv(filter.states, s) })}
/>
<ChipGroup
label={t("filterSources")}
values={SOURCES}
active={filter.sources}
renderLabel={(s) => t(SOURCE_KEY[s])}
onToggle={(s) => patchParams({ source: toggleCsv(filter.sources, s) })}
onToggle={(s) => setParams({ source: toggleCsv(filter.sources, s) })}
/>
{providerKeys.length > 0 && (
<ChipGroup
@@ -149,7 +140,7 @@ export function OrchestrationToolbar({
values={providerKeys}
active={filter.providers}
renderLabel={(p) => p}
onToggle={(p) => patchParams({ provider: toggleCsv(filter.providers, p) })}
onToggle={(p) => setParams({ provider: toggleCsv(filter.providers, p) })}
/>
)}
{!isEmptyFilter(filter) && (

View File

@@ -4,7 +4,7 @@ import { useTranslations } from "next-intl";
import { StatusDot } from "@/shared/components/flow/StatusDot";
import { orchStateColor, type OrchNode, type OrchState } from "../model/orchestrationTypes";
import { useDrawerDetail } from "./useDrawerDetail";
import type { DrawerError, RepeatOutcome } from "./useDrawerDetail";
import type { DrawerError } from "./useDrawerDetail";
import type { CloudAgentTask } from "@/lib/cloudAgent/types";
import type { A2ATask } from "@/lib/a2a/taskManager";
@@ -349,18 +349,15 @@ function RepeatButton({
}: {
canRepeat: boolean;
busy: boolean;
repeat: () => Promise<RepeatOutcome>;
onActionDone: (newNodeId?: string) => void;
repeat: () => Promise<boolean>;
onActionDone: () => void;
onToast: (text: string) => void;
t: Translate;
}) {
const { confirming, onClick } = useTwoClickConfirm(() => {
void (async () => {
// The created task's CANVAS id is handed to `onActionDone` so the page can focus it —
// `undefined` when the creation response carried no usable id (plain refetch, no jump).
const { ok, newNodeId } = await repeat();
if (ok) {
onActionDone(newNodeId ?? undefined);
if (await repeat()) {
onActionDone();
onToast(t("repeatDone"));
}
})();
@@ -398,8 +395,8 @@ function DrawerActions({
busy: boolean;
approve: () => Promise<boolean>;
cancel: () => Promise<boolean>;
repeat: () => Promise<RepeatOutcome>;
onActionDone: (newNodeId?: string) => void;
repeat: () => Promise<boolean>;
onActionDone: () => void;
onToast: (text: string) => void;
t: Translate;
}) {
@@ -491,7 +488,7 @@ export function OrchestrationDrawer({
}: {
node: OrchNode | null;
onClose: () => void;
onActionDone: (newNodeId?: string) => void;
onActionDone: () => void;
}) {
const t = useTranslations("orchestration");
const {

View File

@@ -143,14 +143,7 @@ function repeatReqForA2a(
};
}
/**
* conductor repeat builder — `POST /api/conductor/tasks` (D1 task-creation route).
* `cli`/`model` come from the hub's `requirements` (`ConductorTaskDetail`, hubProxy.ts) and are
* carried over so the repeat lands on the SAME runner profile/model the original task was
* pinned to. Both are `z.string().optional()` in the route's Zod: a `null` would 400, so a
* missing requirement must OMIT the field (`undefined`) rather than send `null` — and the two
* are independent (one may be set while the other is not).
*/
/** conductor repeat builder — `POST /api/conductor/tasks` (D1 task-creation route). */
function repeatReqForConductor(detail: unknown): { url: string; init: RequestInit } | null {
const d = detail as ConductorTaskDetail | null;
if (!d?.repo || !d?.prompt) return null;
@@ -161,8 +154,6 @@ function repeatReqForConductor(detail: unknown): { url: string; init: RequestIni
prompt: d.prompt,
baseRef: d.base_ref ?? undefined,
mode: d.mode,
cli: d.cli ?? undefined,
model: d.model ?? undefined,
}),
};
}
@@ -177,36 +168,6 @@ export function repeatReqFor(
return null;
}
/** `<prefix><id>` when `id` is a non-empty string, `null` otherwise (never a bare prefix). */
function prefixedNodeId(prefix: string, id: unknown): string | null {
return typeof id === "string" && id.length > 0 ? `${prefix}${id}` : null;
}
/**
* CANVAS node id of the task a successful repeat just created, from the creation response
* body — `null` whenever the body does not carry a usable id (the caller then simply refetches
* without focusing anything). The canvas addresses nodes by PREFIXED id
* (`mergeSnapshot.ts`), so the raw upstream id is never returned on its own. Response
* envelopes, verified against the live routes:
* - conductor (`POST /api/conductor/tasks`): `{ task_id }`.
* - cloud-agent (`POST /api/v1/agents/tasks`): `{ data: { id } }`.
* - a2a (`POST /a2a`, JSON-RPC `message/send`): `{ result: { task: { id } } }`.
*/
export function newNodeIdFrom(node: OrchNode, body: unknown): string | null {
if (!body || typeof body !== "object") return null;
const b = body as Record<string, unknown>;
if (node.id.startsWith("conductor:task:")) return prefixedNodeId("conductor:task:", b.task_id);
if (node.id.startsWith("cloud-agent:")) {
const data = b.data as { id?: unknown } | undefined;
return prefixedNodeId("cloud-agent:", data?.id);
}
if (node.id.startsWith("a2a:")) {
const result = b.result as { task?: { id?: unknown } } | undefined;
return prefixedNodeId("a2a:", result?.task?.id);
}
return null;
}
/**
* Unwraps a task-detail GET response to the actual task payload. Each source's
* route has its own envelope — verified against the live handlers, not assumed:
@@ -297,62 +258,38 @@ function useFetchDetail(
* that never happened, so the `/a2a` action also inspects the envelope. Only the numeric
* `error.code` is surfaced (`RPC <code>`) — never the upstream `error.message`.
*/
function jsonRpcErrorCode(body: unknown): number | undefined {
const b = body as { error?: { code?: unknown } } | undefined | null;
const code = b?.error?.code;
return typeof code === "number" ? code : b?.error ? -32603 : undefined;
}
/**
* Reads an action response body ONCE, tolerating a non-JSON/empty body. A body that cannot be
* parsed is not evidence of failure — the status already stood — so it yields `null` and the
* action stays successful (it just has no new-task id to focus).
*/
async function readJsonBody(res: { json?: () => Promise<unknown> }): Promise<unknown> {
async function jsonRpcErrorCode(res: {
json?: () => Promise<unknown>;
}): Promise<number | undefined> {
try {
return (await res.json?.()) ?? null;
const body = (await res.json?.()) as { error?: { code?: unknown } } | undefined;
const code = body?.error?.code;
return typeof code === "number" ? code : body?.error ? -32603 : undefined;
} catch {
return null;
// A non-JSON / already-consumed body is not evidence of failure — the status stands.
return undefined;
}
}
/** Outcome of an action POST: whether it succeeded, plus the parsed body on success. */
interface ActionOutcome {
ok: boolean;
body: unknown;
}
async function performAction(
req: { url: string; init: RequestInit } | null,
setActionError: (text: string) => void,
clearError: () => void
): Promise<ActionOutcome> {
if (!req) return { ok: false, body: null };
setActionError: (text: string) => void
): Promise<boolean> {
if (!req) return false;
try {
const res = await fetch(req.url, req.init);
if (!res.ok) throw new Error(`HTTP ${res.status}`);
const body = await readJsonBody(res);
if (req.url === "/a2a") {
const code = jsonRpcErrorCode(body);
const code = await jsonRpcErrorCode(res);
if (code !== undefined) throw new Error(`RPC ${code}`);
}
// The banner is not sticky: a retry (or any later action) that works clears whatever
// detail/action error was on screen, so the drawer never shows a failure the operator
// already recovered from.
clearError();
return { ok: true, body };
return true;
} catch (err) {
setActionError(toSafeErrorText(err));
return { ok: false, body: null };
return false;
}
}
/** Result of the drawer's repeat action: success plus the canvas id of the created task. */
export interface RepeatOutcome {
ok: boolean;
newNodeId: string | null;
}
export function useDrawerDetail(node: OrchNode | null) {
const [detail, setDetail] = useState<unknown | null>(null);
const [isLoading, setIsLoading] = useState(false);
@@ -369,19 +306,15 @@ export function useDrawerDetail(node: OrchNode | null) {
const { canApprove, canCancel } = deriveActionAvailability(route, node);
const repeatReq = node ? repeatReqFor(node, detail) : null;
const runAction = async (
req: { url: string; init: RequestInit } | null
): Promise<ActionOutcome> => {
if (busy) return { ok: false, body: null };
const runAction = async (req: { url: string; init: RequestInit } | null): Promise<boolean> => {
if (busy) return false;
setBusy(true);
try {
return await performAction(req, setActionError, () => setErrorState(null));
return await performAction(req, setActionError);
} finally {
setBusy(false);
}
};
const runBooleanAction = async (req: { url: string; init: RequestInit } | null) =>
(await runAction(req)).ok;
return {
detail,
@@ -392,13 +325,8 @@ export function useDrawerDetail(node: OrchNode | null) {
canApprove,
canCancel,
canRepeat: !!repeatReq && !busy,
approve: () => runBooleanAction(route?.approveReq ?? null),
cancel: () => runBooleanAction(route?.cancelReq ?? null),
// Only the repeat reports a new node id: approve/cancel act on the task already open, so
// there is nothing new to focus (and their responses can echo the SAME task's id back).
repeat: async (): Promise<RepeatOutcome> => {
const { ok, body } = await runAction(repeatReq);
return { ok, newNodeId: ok && node ? newNodeIdFrom(node, body) : null };
},
approve: () => runAction(route?.approveReq ?? null),
cancel: () => runAction(route?.cancelReq ?? null),
repeat: () => runAction(repeatReq),
};
}

View File

@@ -17,13 +17,6 @@ interface StatusEdgeData {
state?: OrchState;
active?: boolean;
mirror?: boolean;
/**
* Particle budget gate, set by `orchestrationToFlow` (see `PARTICLE_EDGE_CAP`). `false`
* means the canvas has too many simultaneously active edges to animate them all — the edge
* still renders its active stroke, just without the 3 SMIL particles. Absent/`true` keeps
* the animation (so a caller that never sets it behaves exactly as before).
*/
particles?: boolean;
}
/** Mesma precedência do edgeStyle da v1: failed > active > succeeded > idle. */
@@ -59,7 +52,6 @@ function StatusEdgeImpl(props: EdgeProps) {
style={{ ...s, ...(data.mirror ? { strokeDasharray: "6 4" } : {}) }}
/>
{data.active &&
data.particles !== false &&
Array.from({ length: PARTICLES }, (_, i) => (
<ellipse
key={i}

View File

@@ -13,7 +13,7 @@ import { fromCloudAgent } from "../model/fromCloudAgent";
import { fromA2A } from "../model/fromA2A";
import { fromConductor } from "../model/fromConductor";
import { mergeSnapshot } from "../model/mergeSnapshot";
import type { OrchSnapshot, OrchSource, SourceStatus } from "../model/orchestrationTypes";
import type { OrchSnapshot, SourceStatus } from "../model/orchestrationTypes";
export const POLL_MS = 5_000;
export const POLL_MS_WS_CONNECTED = 30_000;
@@ -59,27 +59,13 @@ export function snapshotContentKey(s: OrchSnapshot): string {
]);
}
/**
* Builds the 3-source status list from a `Promise.allSettled` triple.
*
* `prev` is the source-status list from the previous poll: a source that is failing NOW
* reuses the `staleSince` it already had when it was ALSO failing in `prev` (so the
* timestamp pins to the FIRST failure instead of advancing every tick — that advance both
* misreported "stale since" as the last poll and defeated `snapshotContentKey`'s stability,
* since it serializes `sources`). A source that recovers loses `staleSince`; a source
* failing for the first time (or failing again after recovering) is stamped with `nowIso`.
*/
/** Builds the 3-source status list from a `Promise.allSettled` triple. */
function buildSourceStatuses(
ca: PromiseSettledResult<{ data: CloudAgentTask[] }>,
a2a: PromiseSettledResult<{ tasks: A2ATask[] }>,
cond: PromiseSettledResult<FleetSnapshot>,
nowIso: string,
prev: SourceStatus[]
nowIso: string
): SourceStatus[] {
const staleSinceFor = (source: OrchSource): string => {
const prevStatus = prev.find((s) => s.source === source);
return prevStatus && !prevStatus.ok && prevStatus.staleSince ? prevStatus.staleSince : nowIso;
};
const next: SourceStatus[] = [];
if (ca.status === "fulfilled") next.push({ source: "cloud-agent", ok: true });
else
@@ -87,16 +73,10 @@ function buildSourceStatuses(
source: "cloud-agent",
ok: false,
error: String(ca.reason),
staleSince: staleSinceFor("cloud-agent"),
staleSince: nowIso,
});
if (a2a.status === "fulfilled") next.push({ source: "a2a", ok: true });
else
next.push({
source: "a2a",
ok: false,
error: String(a2a.reason),
staleSince: staleSinceFor("a2a"),
});
else next.push({ source: "a2a", ok: false, error: String(a2a.reason), staleSince: nowIso });
if (cond.status === "fulfilled") {
next.push({ source: "conductor", ok: true, offline: cond.value.offline });
} else
@@ -104,7 +84,7 @@ function buildSourceStatuses(
source: "conductor",
ok: false,
error: String(cond.reason),
staleSince: staleSinceFor("conductor"),
staleSince: nowIso,
});
return next;
}
@@ -144,6 +124,7 @@ export function useOrchestrationSnapshot() {
if (controller.signal.aborted) return;
const nowMs = Date.now();
const nowIso = new Date(nowMs).toISOString();
const next = buildSourceStatuses(ca, a2a, cond, nowIso);
// Failed sources keep the previously stored slice — only overwrite what
// actually resolved this round ("last good data" contract from the brief).
@@ -152,10 +133,7 @@ export function useOrchestrationSnapshot() {
a2a: a2a.status === "fulfilled" ? a2a.value.tasks : prev.a2a,
conductor: cond.status === "fulfilled" ? cond.value : prev.conductor,
}));
// Functional updater form: gives `buildSourceStatuses` the latest previous
// statuses (for the staleSince-pinning rule) without a stale closure over
// `statuses` and without adding a ref or an extra effect for it.
setStatuses((prevStatuses) => buildSourceStatuses(ca, a2a, cond, nowIso, prevStatuses));
setStatuses(next);
setPolledAt(nowMs);
setIsLoading(false);
};

View File

@@ -1,158 +0,0 @@
/**
* Pure model to compare two Orchestration Canvas history runs side by side (History tab,
* PR-A / Task A1). No React, no side effects.
*
* `normalizeRunSide` turns a `HistoryItem` plus its raw detail payload (the JSON body of
* `GET /api/a2a/tasks/[id]` for `source === "a2a"`, or the in-memory `CloudAgentTask` for
* `source === "cloud-agent"`) into a `RunSide` with a normalized `events` timeline and
* `memoryHits` list. `detail` is untrusted: for A2A it round-trips through
* `reconstituteHistoricalTask` (`src/app/api/a2a/tasks/[id]/route.ts`) plus client-supplied
* `metadata` with no Zod validation, so every field is read defensively — a malformed shape
* drops the offending item (or yields `[]`) instead of throwing. See Fase 2 lesson in
* `historyModel.ts` / global-constraints: `memoryHits: "boom"` used to crash the drawer.
*
* `buildComparison` derives signed deltas (`right - left`) for `durationMs`/`cost` only when
* both sides have a finite value, and always reports `eventCount`'s delta and whether the two
* sides share the same (source, identity) pair.
*/
import type { HistoryItem } from "./historyModel";
export interface RunEvent {
label: string;
timestamp: string | null;
}
export interface RunMemoryHit {
id: string;
key: string;
type: string;
snippet: string;
}
export interface RunSide {
item: HistoryItem;
events: RunEvent[];
memoryHits: RunMemoryHit[];
}
export interface RunDeltas {
durationMs: number | null;
cost: number | null;
eventCount: number;
sameIdentity: boolean;
}
export interface RunComparison {
left: RunSide;
right: RunSide;
deltas: RunDeltas;
}
function isRecord(value: unknown): value is Record<string, unknown> {
return typeof value === "object" && value !== null && !Array.isArray(value);
}
function isNonEmptyString(value: unknown): value is string {
return typeof value === "string" && value.length > 0;
}
/** One `reconstituteHistoricalTask` event (`{ timestamp: string; state: string; message?: string
* }`) → `RunEvent`, or `null` if the raw shape does not match. Split out of `a2aEventsFrom` only
* to keep that function's cognitive complexity under the ratchet — no behavior change; the same
* field checks run in the same order. */
function a2aEventFrom(raw: unknown): RunEvent | null {
if (!isRecord(raw)) return null;
const { state, message, timestamp } = raw;
if (!isNonEmptyString(state)) return null;
if (message !== undefined && typeof message !== "string") return null;
if (timestamp !== undefined && timestamp !== null && typeof timestamp !== "string") return null;
return {
label: typeof message === "string" ? message : state,
timestamp: typeof timestamp === "string" ? timestamp : null,
};
}
/** `{ timestamp: string; state: string; message?: string }` — reconstituteHistoricalTask's shape. */
function a2aEventsFrom(detail: Record<string, unknown>): RunEvent[] {
const events = detail.events;
if (!Array.isArray(events)) return [];
const out: RunEvent[] = [];
for (const raw of events) {
const event = a2aEventFrom(raw);
if (event) out.push(event);
}
return out;
}
/** `CloudAgentActivity[]` (`src/lib/cloudAgent/types.ts`) — each item has `type` and `content`. */
function cloudAgentEventsFrom(detail: Record<string, unknown>): RunEvent[] {
const activities = detail.activities;
if (!Array.isArray(activities)) return [];
const out: RunEvent[] = [];
for (const raw of activities) {
if (!isRecord(raw)) continue;
const { content, timestamp } = raw;
if (!isNonEmptyString(content)) continue;
if (timestamp !== undefined && timestamp !== null && typeof timestamp !== "string") continue;
out.push({
label: content,
timestamp: typeof timestamp === "string" ? timestamp : null,
});
}
return out;
}
function memoryHitsFrom(detail: Record<string, unknown>): RunMemoryHit[] {
const metadata = detail.metadata;
if (!isRecord(metadata)) return [];
const hits = metadata.memoryHits;
if (!Array.isArray(hits)) return [];
const out: RunMemoryHit[] = [];
for (const raw of hits) {
if (!isRecord(raw)) continue;
const { id, key, type, snippet } = raw;
if (
typeof id === "string" &&
typeof key === "string" &&
typeof type === "string" &&
typeof snippet === "string"
) {
out.push({ id, key, type, snippet });
}
}
return out;
}
/**
* Builds one side of a comparison. NEVER throws — `detail` is untrusted JSON (persisted rows
* for A2A history, client-supplied `metadata` with no Zod) so any shape mismatch just drops
* the offending item; a missing/non-object `detail` yields empty `events`/`memoryHits`.
*/
export function normalizeRunSide(item: HistoryItem, detail: unknown): RunSide {
if (!isRecord(detail)) {
return { item, events: [], memoryHits: [] };
}
const events = item.source === "a2a" ? a2aEventsFrom(detail) : cloudAgentEventsFrom(detail);
const memoryHits = item.source === "a2a" ? memoryHitsFrom(detail) : [];
return { item, events, memoryHits };
}
function signedDelta(left: number | null, right: number | null): number | null {
if (!Number.isFinite(left) || !Number.isFinite(right)) return null;
return (right as number) - (left as number);
}
/** Combines two normalized sides into a comparison with signed `right - left` deltas. */
export function buildComparison(left: RunSide, right: RunSide): RunComparison {
return {
left,
right,
deltas: {
durationMs: signedDelta(left.item.durationMs, right.item.durationMs),
cost: signedDelta(left.item.cost, right.item.cost),
eventCount: right.events.length - left.events.length,
sameIdentity:
left.item.source === right.item.source && left.item.identity === right.item.identity,
},
};
}

View File

@@ -7,7 +7,6 @@ import {
type OrchSnapshot,
type OrchSource,
type OrchState,
type SourceIssue,
type SourceStatus,
} from "./orchestrationTypes";
@@ -156,9 +155,8 @@ function capWorkNodesWithOverflow(
}
/**
* (1) root — link every present SourceNode, flag the failing/offline ones (materializing a
* placeholder only when the source has no node at all) so the UI can show them stale.
* Returns nodes with the root prepended.
* (1) root — link every present SourceNode, plus failed sources so the UI can show them
* stale. Returns nodes with the root prepended.
*/
function buildRootAndSourceEdges(
nodes: OrchNode[],
@@ -170,30 +168,22 @@ function buildRootAndSourceEdges(
const nextEdges = [...edges];
const sourceIds = new Set(nextNodes.filter((n) => n.kind === "source").map((n) => n.id));
for (const s of sources) {
// `!s.ok` covers hard failures; `s.offline` also flags a source that reported
// ok:true but offline:true (e.g. Conductor with no hub configured) — otherwise
// that source's "offline" sublabel can never render.
if ((s.ok && !s.offline) || s.source === "routing") continue;
const id = `source:${s.source}`;
const issue: SourceIssue = s.offline ? "offline" : "error";
const index = nextNodes.findIndex((n) => n.id === id && n.kind === "source");
if (index === -1) {
// `!s.ok` covers hard failures; `s.offline` also materializes a placeholder
// for a source that reported ok:true but offline:true (e.g. Conductor with
// no hub configured) — otherwise that source never gets a SourceNode at all
// and its "offline" sublabel can never render.
if ((!s.ok || s.offline) && !sourceIds.has(`source:${s.source}`) && s.source !== "routing") {
nextNodes.push({
id,
id: `source:${s.source}`,
kind: "source",
source: s.source,
label: s.source,
sublabel: issue,
sourceIssue: issue,
sublabel: s.offline ? "offline" : "error",
sourceIssue: s.offline ? "offline" : "error",
staleSince: s.staleSince,
});
sourceIds.add(id);
continue;
sourceIds.add(`source:${s.source}`);
}
// A source that HAD data and only now started failing keeps its nodes/counts —
// it just gains the issue flags. Copy rather than mutate: the original object is
// still referenced by `parts.<source>.nodes` and this function's contract is Pure.
nextNodes[index] = { ...nextNodes[index], sourceIssue: issue, staleSince: s.staleSince };
}
for (const id of sourceIds) {
nextEdges.push({

View File

@@ -11,13 +11,6 @@ const LAYER_Y: Record<OrchNodeKind, number> = {
};
const X_GAP = 260;
/**
* Maximum simultaneously ACTIVE edges that still get the StatusEdge particle stream. Each
* animated edge runs 3 SMIL `<animateMotion>` particles, so a busy canvas would otherwise pay
* hundreds of concurrent animations; past the cap every edge renders as a plain colored stroke.
*/
export const PARTICLE_EDGE_CAP = 40;
export interface OrchestrationToFlowOptions {
collapsed?: ReadonlySet<OrchSource>;
}
@@ -72,18 +65,12 @@ export function orchestrationToFlow(
>,
};
});
const particles = visibleEdges.filter((e) => e.active).length <= PARTICLE_EDGE_CAP;
const edges: Edge[] = visibleEdges.map((e) => ({
id: e.id,
source: e.from,
target: e.to,
type: "status",
data: {
state: stateOf.get(e.to),
active: e.active,
mirror: e.kind === "mirror",
particles,
},
data: { state: stateOf.get(e.to), active: e.active, mirror: e.kind === "mirror" },
}));
const workIdsKey = visibleNodes

View File

@@ -1,34 +0,0 @@
/**
* CSV query-param helpers shared by the Orchestration page client (which parses `?state=` /
* `?source=` / `?provider=` / `?collapsed=` out of the URL) and the toolbar (which toggles them
* back in). Both used to carry their own private copy of `toggleCsv`; a single definition keeps
* the round-trip (parse → toggle → parse) consistent. Pure — never mutates its inputs.
*/
/**
* CSV → Set. Each token is trimmed and empty tokens are dropped, so a hand-edited or
* shared URL like `?state=running, failed` parses the same as `?state=running,failed`.
* With `valid`, tokens outside that set are discarded (an unknown state chip in the URL
* must not survive into the filter).
*/
export function parseCsvSet<T extends string>(raw: string | null, valid?: ReadonlySet<T>): Set<T> {
const out = new Set<T>();
if (!raw) return out;
for (const token of raw.split(",")) {
const v = token.trim();
if (!v) continue;
if (!valid || valid.has(v as T)) out.add(v as T);
}
return out;
}
/**
* Toggle `value` in `current`, returning the next CSV — or `null` when the list becomes
* empty, so the caller's `setParams` drops the param from the URL instead of leaving `?state=`.
*/
export function toggleCsv<T extends string>(current: ReadonlySet<T>, value: T): string | null {
const next = new Set(current);
if (next.has(value)) next.delete(value);
else next.add(value);
return next.size > 0 ? [...next].sort().join(",") : null;
}

View File

@@ -5,8 +5,6 @@ import { useTranslations } from "next-intl";
import Link from "next/link";
import { FlowCanvas } from "@/shared/components/flow/FlowCanvas";
import { orchestrationToFlow } from "../model/orchestrationToFlow";
import { EMPTY_FILTER, isEmptyFilter } from "../model/filterSnapshot";
import type { OrchFilter } from "../model/filterSnapshot";
import type { OrchNode, OrchSnapshot, OrchSource } from "../model/orchestrationTypes";
import { OrchestratorNode } from "../nodes/OrchestratorNode";
import { SourceNode } from "../nodes/SourceNode";
@@ -36,8 +34,6 @@ export function AgentsTab({
onToggleCompleted,
collapsed = EMPTY_COLLAPSED,
onToggleCollapse,
filter = EMPTY_FILTER,
onClearFilters,
}: {
snapshot: OrchSnapshot;
onNodeClick: (orchNodeId: string) => void;
@@ -45,8 +41,6 @@ export function AgentsTab({
onToggleCompleted: (v: boolean) => void;
collapsed?: ReadonlySet<OrchSource>;
onToggleCollapse?: (s: OrchSource) => void;
filter?: OrchFilter;
onClearFilters?: () => void;
}) {
const t = useTranslations("orchestration");
const { nodes, edges, fitKey } = useMemo(
@@ -64,21 +58,6 @@ export function AgentsTab({
onNodeClick(node.id);
};
// An empty canvas means two very different things. With no filter active it is "nothing is
// running" and the setup CTAs are the right next step; under an ACTIVE filter the runs may
// well exist and simply not match, so pointing the operator at the setup pages would be wrong
// advice — offer to clear the filter instead.
if (!hasWork && !isEmptyFilter(filter)) {
return (
<div className="flex flex-col items-center justify-center h-full gap-3 text-muted">
<p className="text-sm">{t("noMatches")}</p>
<button type="button" className="text-xs underline" onClick={() => onClearFilters?.()}>
{t("clearFilters")}
</button>
</div>
);
}
if (!hasWork) {
return (
<div className="flex flex-col items-center justify-center h-full gap-3 text-muted">

Some files were not shown because too many files have changed in this diff Show More