mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-08-06 15:22:12 +03:00
* chore(release): open v3.8.38 development cycle
* fix(executors): strip client_metadata for cerebras and mistral (#4727)
Integrated into release/v3.8.38 (leva 5)
* fix(codebuddy): only send reasoning params when client requests reasoning (#5019)
Integrated into release/v3.8.38 (leva 5)
* fix(sse): keep streaming for forceStream providers when client requests JSON (#5021)
Integrated into release/v3.8.38 (leva 5)
* fix(sse): guard non-JSON SSE lines and duplicate [DONE] (#4937)
Integrated into release/v3.8.38 (leva 5)
* feat(blackbox): refresh provider model catalog (#4935)
Integrated into release/v3.8.38 (leva 5)
* fix(sse): dedupe case-variant Anthropic version/beta headers (#4846)
Integrated into release/v3.8.38 (leva 5)
* feat(sse): Kiro inline <thinking> stream splitter (#4911)
Integrated into release/v3.8.38 (leva 5)
* feat(cursor): parse Composer DeepSeek-style inline tool calls (#4912)
Integrated into release/v3.8.38 (leva 5)
* feat(proxy): auth-less host:port batch import (#4938)
Integrated into release/v3.8.38 (leva 5)
* fix(oauth): support Kiro IDC (organization) token import (#4944)
Integrated into release/v3.8.38 (leva 5)
* fix(translator): preserve cache_control for DashScope OpenAI-compat providers (port from 9router#2069) (#5013)
Integrated into release/v3.8.38 (leva 5)
* fix(tts): resolve Gemini TTS models from catalog (#4934)
Integrated into release/v3.8.38 (leva 5)
* fix(sse): don't cool down the connection on a self-inflicted upstream timeout (504) (#5064)
Integrated into release/v3.8.38 (leva 5)
* fix(sse): robust Anthropic /v1/messages streaming — real ping keepalive + client-disconnect guard (#5063)
Integrated into release/v3.8.38 (leva 5)
* feat(video): add Alibaba DashScope (wan2.7-t2v) provider (#5051)
Integrated into release/v3.8.38 (leva 5)
* fix: preserve model hidden flags (isHidden) across model sync (#5086)
Integrated into release/v3.8.38 (leva 5)
* fix(models): derive model discovery config from registry modelsUrl (#5087)
Integrated into release/v3.8.38 (leva 5)
* fix(compression): replace fileURLToPath(import.meta.url) with runtime anchors for standalone bundle (#5089)
Integrated into release/v3.8.38 (leva 5)
* feat(cc): add summarized thinking display toggle (#5055)
Integrated into release/v3.8.38 (leva 5)
* Harden selected API error responses (#5032)
Integrated into release/v3.8.38 (leva 5)
* chore(quality): rebaseline file-size for leva 5 PR batch drift
6 frozen files grew from merged leva-5 PRs (cursor #4912, kiro #4911,
videoGeneration #5051, default #4727, base #4846, chat #5064); all covered
by per-PR tests. See _rebaseline_2026_06_26_leva5 in the baseline.
* feat(compression): compression playground (Play + Compare tabs) in the studio (#5080)
Integrated into release/v3.8.38
* fix(combo): fail over on empty-content 502 instead of exhausting the provider (#5085) (#5104)
* fix(dashboard): surface detailed credential-validation error in add-connection modal (#5088) (#5106)
* feat(providers): allow local/private provider URLs by default with scoped metadata-safe guard (#5066) (#5107)
* fix(diagnostics): treat non-streaming Claude messages shape as valid output (#5108) (#5116)
* fix(db): translate pt-BR SQLite driver-fallback log lines to English (#5103) (#5115)
* fix(sse): repair release base-reds — malformed-response false positives + header casing + stale tests (#5117)
Repairs the release/v3.8.38 base-reds; unblocks #5078.
* chore(quality): rebaseline file-size for responseSanitizer (#5117) + AddApiKeyModal drift
* fix(translator): forward image tool_result blocks as image_url (#5100)
Base-reds fixed (#5117); image tool_result→image_url. Integrated into release/v3.8.38.
* fix(responses): default text.format for openai-compatible responses providers (#5101)
Base-reds fixed (#5117); default text.format + file-size rebaseline. Integrated into release/v3.8.38.
* feat(dashboard): expose Fusion judgeModel + fusionTuning in the combo editor (#5074)
Base-reds fixed (#5117); Fusion editor + file-size rebaseline. Integrated into release/v3.8.38.
* feat(quota): add opt-in Codex/Claude auto-ping keepalive (#5102)
Base-reds fixed (#5117); auto-ping keepalive + file-size rebaseline. Integrated into release/v3.8.38.
* test(release): relocate 2 orphan test files into the collected flat tests/unit dir (#5120)
Unblocks Lint (test-discovery) on #5078. Integrated into release/v3.8.38.
* fix(translator): preserve reasoning-replay reasoning_content + repair 3 release-green test reds (#5122)
Repairs 3 release-green test reds + test-masking; unblocks #5078.
* test(golden): redact live Node version from provider translate-path snapshot (#5125)
Final golden unblock for #5078.
* test(golden): redact OmniRoute app version from translate-path snapshot (#5126)
Coverage shard golden unblock for #5078.
* Ignore disconnect races during in-band stream error handling (#5007)
Integrated into release/v3.8.38
* Track final connection IDs in failover logs (#5016)
Integrated into release/v3.8.38
* fix(sse): convert Gemini body to OpenAI format in antigravity MITM handler (#4845)
Integrated into release/v3.8.38 (rebased on tip, CHANGELOG re-injected)
* feat(providers): add ZenMux Free session-cookie provider (#5105)
Integrated into release/v3.8.38 (rebased on tip, CHANGELOG re-injected)
* feat(dashboard): click-to-edit model alias in provider page (#5119)
Integrated into release/v3.8.38 (rebased on tip, i18n scope verified, CHANGELOG re-injected)
* feat(mcp): web-session robustness — cookie dedup (PR6) + browser-pool observability (PR7) (#3368) (#5121)
Integrated into release/v3.8.38 (rebased on tip; cookie-dedup branch extracted to findExistingCookieConnection helper → complexity-neutral; CHANGELOG added)
* fix(usage): dedupe request-usage logging and debounce stats (#4940)
Integrated into release/v3.8.38 (rebased on tip; DB-handle hang was stale-base artifact — resetDbInstance already closes the handle, test green 5/5; file-size drift consolidated at release; CHANGELOG re-injected)
* fix(dashboard): key model visibility toggle on canonical providerId (#5091)
Integrated into release/v3.8.38 (retargeted main→release; .tsx visibility-key test green 2/2)
* chore(deps): bump actions/cache from 5.0.5 to 6.0.0 (#5112)
Integrated into release/v3.8.38 (retargeted main→release; workflow-only actions/cache bump — unit failures were stale main base-reds)
* fix(streaming): harden long OpenAI-compatible SSE streams (#5124)
Integrated into release/v3.8.38 (rebased on tip; streamHandler conflict with #5007 disconnect-guard resolved — both coexist, stream-handler 22/22 green)
* feat: Add Grok Build (xAI) provider with OAuth import-token flow (#5020)
Integrated into release/v3.8.38 (rebased on tip; Hard Rule #11 fix — Grok public client_id now via resolvePublicCred(grok_id), 3 literals removed; grok-oauth 7/7 + check:public-creds green)
* feat(providers): add Factory (factory.ai) as a subscription gateway provider (#5065)
Integrated into release/v3.8.38 (rebased on tip; added factory registry test for PR Test Policy + fixed check:env-doc-sync phantom FACTORY_API_KEY; factory loads in PROVIDERS, no Zod issue — that flag was a false positive)
* chore(test): reconcile golden snapshot + apikey count for new providers
#5020 (grok-cli), #5065 (factory), #5105 (zenmux-free) added providers but did
not regenerate tests/snapshots/provider/translate-path.json (now +3 entries) nor
bump the APIKEY_PROVIDERS count (159->160 for the factory gateway). Test-only
reconciliation; no production change.
* fix(resilience): harden quota and model lockout edge cases (#5093)
Integrated into release/v3.8.38 (rebased on tip). TRUST-BUT-VERIFY: dropped the PR's 0dd7df641 'fix unit gates' commit which reverted #5122 reasoning-replay (preserveReasoningContent) + re-introduced #4849 O(n^2) growth, and restored 5 tests it had realigned. Kept only the 3 declared resilience fixes (quota cutoff guard, gemini MIME, model-lockout maxCooldownMs); 23/23 green.
* Hydrate quota cache and scope auto combo candidates (#5015)
Integrated into release/v3.8.38 (rebased on tip). Kept core quota-cache hydration + auto-combo candidate scoping + combos UI; dropped out-of-scope toolCloaking refactor (conflicted with #4813 stripEnumDescriptions — took tip) and the unrelated sse-auth test split. Added quota-cache-hydrate-5015 regression test (Rule #18); combo-account-allowlist 8/8 + hydration 2/2 green.
* chore(quality): reconcile complexity + file-size baselines for v3.8.38 owner-PR batch
complexity 1972->1978 (+6) and file-size providers.ts 1093->1107 / usageHistory.ts
934->983 — drift from the /review-prs merge batch (#4845/#5105/#5020/#4940/#5093/
#5015 + #5121 cookie-dedup helper extraction). check:complexity/check:file-size do
not run on the PR->release fast-path, so the branch accrued unmeasured; all legit
feature/fix growth, not regression. See per-key justifications in each baseline.
* fix(security): exact-host Anthropic baseUrl check (CodeQL js/incomplete-url-substring-sanitization #674) (#5130)
The anthropic-compatible Bearer-fallback gate decided whether a configured baseUrl
targeted the official api.anthropic.com host via a substring `.includes("api.anthropic.com")`.
A look-alike upstream such as `https://api.anthropic.com.evil.test` or
`https://evil.test/?x=api.anthropic.com` matched the substring and was wrongly treated as
official, suppressing the Bearer fallback meant for third-party gateways
(CodeQL #674, js/incomplete-url-substring-sanitization, high).
Replace the substring test with an exported `isOfficialAnthropicBaseUrl()` helper that
parses the URL and compares the hostname for exact equality. Empty baseUrl stays official;
scheme-less hosts are parsed with an assumed https://; an unparseable baseUrl falls back to
third-party (Bearer emitted) as the safer default. Behavior for legitimate official/third-party
baseUrls is unchanged.
Adds tests/unit/anthropic-official-baseurl-host.test.ts covering official, look-alike,
scheme-less, and unparseable inputs plus a static guard that the substring pattern is gone.
* fix(proxy): repair one-click Deno & Cloudflare relay deployments (#5128) (#5132)
* fix(services): embed WS proxy honours LIVE_WS_HOST; reject empty messages early (#5110) (#5133)
* fix(api): resolve /v1/models/{id} case-insensitively (#5082) (#5135)
* fix(providers): add MiniMax M3 & Nemotron 3 Ultra to Cline catalog (#3321) (#5136)
* fix(proxy): make SOCKS5 handshake timeout tunable via SOCKS_HANDSHAKE_TIMEOUT_MS (#5109) (#5137)
* feat(sidebar): add support for colored menu icons (#3812)
Integrated into release/v3.8.38 (recreated on tip — fork had unrelated history; added getSidebarIconAccent regression test, Rule #18). Clean 2-file UI feature.
* fix(providers): complete grok-cli OAuth wiring + zenmux-free web-session metadata
Base-red repair for #5020 (grok-cli) and #5105 (zenmux-free), surfaced by the
full CI on the release PR (#5078) — the PR->release fast-path does not run the
oauth-providers-config / web-session-credentials / provider-consistency gates.
- grok-cli: register in OAUTH_PROVIDERS (providers.ts canonical list, fixes
check:provider-consistency), add OAUTH_PROVIDER_IDS.GROK_CLI + GROK_CLI_CONFIG
in oauth constants (provider config now sourced there, not a local literal),
align oauth-providers-config.test.ts (EXPECTED_PROVIDER_KEYS + config map).
- zenmux-free: declare its web-session credential requirement (full Cookie header)
in WEB_SESSION_CREDENTIAL_REQUIREMENTS.
Local: oauth-providers-config 27/27, web-session-credentials 4/4, grok-cli-oauth
7/7, check:provider-consistency OK, +115 OAUTH_PROVIDERS tests green.
* Fix resilience settings page response mapping (#5139)
Integrated into release/v3.8.38. Thanks @rdself for the fix and the regression test.
* fix(kiro): retire claude-sonnet-4.5 from catalog + pin 400 model-unavailable test (#5140)
Extracted the real change from #5140 (the bot PR regenerated the entire
freeModelCatalog.data.ts + touched package-lock.json; only the targeted
edits are kept here):
- remove claude-sonnet-4.5 from the Kiro registry entry
- remove the matching kiro free-model catalog row
- pin Kiro's verbatim 400 "Invalid model..." to isModelUnavailableError
Closes #4484
* fix(sidebar): drop orphan `settings` accent color (typecheck:core red) (#5142)
SIDEBAR_ICON_ACCENTS is typed Partial<Record<HideableSidebarItemId, string>>,
but `settings` is not a hideable item id (only `settings-general`,
`settings-appearance`, … and `context-settings` exist; there is no item with
`id: "settings"`), so the accent was unreachable. It broke `typecheck:core`
on the release tip ("'settings' does not exist in type …", introduced by
#3812 colored menu icons). Removing the orphan key restores a clean
typecheck:core (rc=0).
* feat: salvage batch 2 — diagnostics null-guard (#5096) + observed quota reset windows (#5025) (#5141)
* fix(diagnostics): null-guard content blocks in detectMalformedNonStream
A null (or non-object) entry in a Claude-native `content` array made the
non-stream classifier throw `TypeError: Cannot read properties of null
(reading 'type')`, crashing the malformed-response detection path. Guard
before type-asserting each block: a null/non-object block is simply skipped.
Two regression tests added (null block among valid blocks → null; only-null
blocks → empty_choices).
Salvaged from closed PR #5096 (base-stale; only the defensive guard — the
Claude-shape recognition it also carried already landed via #5108).
Co-authored-by: herjarsa <herjarsa@users.noreply.github.com>
* feat(quota): persist observed provider quota reset windows
Adds `provider_quota_reset_events` (migration 108) + `db/quotaResetEvents.ts`
to record real upstream weekly-quota window transitions whenever a quota
refresh shows the reset rolling to a new cycle (different day, later resetAt).
`apiKeyUsageLimits` now prefers the observed window start over the inferred
`resetAt − 7d`, falling back to snapshot inference when no event is recorded
yet. `quotaCache.setQuotaCache` records the transition opportunistically.
`recordProviderQuotaResetEventIfChanged` only fires for the primary weekly
window (not daily/sonnet), is idempotent (INSERT OR IGNORE on the unique
window key), and no-ops when the reset didn't actually roll. 4 unit tests
(tests/unit/lib/quota-reset-events.test.ts).
Salvaged from closed PR #5025 (which bundled this with two unrelated
features + a colliding migration 104). Renumbered to 108; module re-exported
from localDb (Rule #2).
Co-authored-by: Witroch4 <175152067+Witroch4@users.noreply.github.com>
---------
Co-authored-by: herjarsa <herjarsa@users.noreply.github.com>
Co-authored-by: Witroch4 <175152067+Witroch4@users.noreply.github.com>
* docs(i18n): sync 3.8.38 CHANGELOG section to 41 mirrors (unblock docs-accuracy) (#5144)
The root CHANGELOG [3.8.38] section grew with this cycle's merged PRs, but the
docs/i18n/<lang>/CHANGELOG.md mirrors were not re-synced — drifting >25% in body
size and failing check:docs-sync (the "Docs accuracy" fast-gate step) for every
open PR against the release.
Ran scripts/release/sync-changelog-i18n.mjs 3.8.38 3.8.37 to copy the root
[3.8.38] section into all 41 mirrors. check:docs-all now passes (exit 0).
Sections are copied verbatim; the per-language translation pass runs at release
time via i18n:run — this only restores the size-sync the gate enforces.
* feat(compression): pure per-step fidelity checker (4 invariants, fail-open)
* feat(compression): fidelityGate config + rejected breakdown fields
* feat(compression): wire per-step fidelity gate into stacked pipeline (opt-in)
* feat(compression): preview route accepts fidelityGate flag (playground)
* feat(compression): playground fidelity-gate toggle + lane rejection display
* docs(compression): note fidelityGate advanced thresholds are intentionally API-omitted
* refactor(compression): extract fidelity-gate step helpers to shrink strategySelector (file-size gate)
bodyToText and gateAdvance moved to fidelityGateStep.ts; StackAccumulator exported.
strategySelector: 889->854 (-35). Residual +6 vs pre-Milestone-B frozen 848 is the
irreducible StackOptions.fidelityGate field + two stacked-loop dispatch reads + import.
Baseline updated to 854 with justification. No cycle introduced (import type only).
940 compression tests pass; typecheck clean.
* test(usage): wire usageHistoryDedup under unit runner brace-list (#5145)
Integrated into release/v3.8.38.
* feat: salvage batch from closed stale PRs (#5038, #5057, #5076) (#5138)
Integrated into release/v3.8.38.
* test(combo): deterministic routing-decision matrix for all 17 strategies (#5146)
Integrated into release/v3.8.38.
* feat(compression): fuzzy near-duplicate dedup (session-dedup 2nd pass + playground toggle) (#5143)
Integrated into release/v3.8.38.
* chore(quality): rebaseline file-size for sidebarVisibility.ts + chat.ts drift (#5147)
Mid-cycle drift on release/v3.8.38 from already-merged PRs that the fast-path
(PR->release skips check:file-size) let accumulate without a bump:
- src/shared/constants/sidebarVisibility.ts 1100->1198 (#3812 colored menu
icons, per-item accent map; #5142 dropped one orphan, net still above frozen)
- src/sse/handlers/chat.ts 1560->1575 (#5064 self-inflicted-timeout cooldown
skip + #5124 long OpenAI-compatible SSE hardening + #5110 embed-WS
LIVE_WS_HOST honour / early empty-message reject)
Each covered by its own PR tests; structural shrink of chat.ts tracked in #3501.
Unblocks the Fast Quality Gates for PRs targeting release/v3.8.38.
* chore(release): finalize v3.8.38 CHANGELOG + cycle reconciliation
- Reconcile [3.8.38]: +18 bullets (compression fidelity-gate/fuzzy-dedup #5143,
quota keepalive #5102, web-session robustness #5121, MiniMax/Nemotron #5136,
model-visibility #5091, failover logs #5016, disconnect races #5007, sidebar
orphan #5142, SRE playbooks salvage #5138, new Security #5130 + Maintenance roll-up)
- Credit salvaged-PR authors (@JxnLexn / @KooshaPari / @herjarsa / @Witroch4)
- Remove phantom bullet for CLOSED-not-merged #5092 (setup aggregator never landed)
- Fix isHidden bullet PR citation #4389 -> #5086 (@herjarsa)
- Back-fill forgotten v3.8.36 bullet: #5026 crypto.randomUUID ID-gen (@hamsa0x7)
- Sync 41 i18n CHANGELOG mirrors; README What's New -> v3.8.38
- Rebaseline cycle drift: eslint 3987->4002, cognitive 833->841, dead-exports
345->346, cyclomatic 1978->1980 (file-size handled by #5147)
* fix(i18n): add missing English UI labels (#5153)
Integrated into release/v3.8.38
* Preserve non-stream reasoning fields for compatible clients (#5155)
Integrated into release/v3.8.38
* feat(compression): ionizer engine — lossy JSON-array sampling reversible via CCR (#5148)
Integrated into release/v3.8.38
* test(combo): gated live smoke for combo strategies (in-process + VPS HTTP) (#5151)
Integrated into release/v3.8.38
* test: refresh release expectations to match current code (#5150)
Integrated into release/v3.8.38 (test-only base-red alignment extracted from #5150)
---------
Co-authored-by: Éder Costa <eder.almeida.costa@gmail.com>
Co-authored-by: José Victor Ferreira <root@josevictor.me>
Co-authored-by: Hernan Javier Ardila Sanchez <hjasgr@gmail.com>
Co-authored-by: fulorgnas <46461624+fulorgnas@users.noreply.github.com>
Co-authored-by: Randi <55005611+rdself@users.noreply.github.com>
Co-authored-by: Jan Leon <Jan.gaschler@gmail.com>
Co-authored-by: R. Beltran <rbeltran8000@gmail.com>
Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>
Co-authored-by: KooshaPari <42529354+KooshaPari@users.noreply.github.com>
Co-authored-by: Ramel Tecnologia - Rafa Martins <146174365+rafacpti23@users.noreply.github.com>
Co-authored-by: herjarsa <herjarsa@users.noreply.github.com>
Co-authored-by: Witroch4 <175152067+Witroch4@users.noreply.github.com>
944 lines
34 KiB
TypeScript
944 lines
34 KiB
TypeScript
import {
|
|
BaseExecutor,
|
|
mergeUpstreamExtraHeaders,
|
|
type ExecuteInput,
|
|
type ExecutorLog,
|
|
type ProviderCredentials,
|
|
} from "./base.ts";
|
|
import { PROVIDERS } from "../config/constants.ts";
|
|
import { v4 as uuidv4 } from "uuid";
|
|
import { refreshKiroToken } from "../services/tokenRefresh.ts";
|
|
import { splitInlineThinking, flushPendingThinking, type KiroThinkingState } from "./kiroThinking.ts";
|
|
|
|
type JsonRecord = Record<string, unknown>;
|
|
|
|
type UsageSummary = {
|
|
prompt_tokens: number;
|
|
completion_tokens: number;
|
|
total_tokens: number;
|
|
cache_read_input_tokens?: number;
|
|
cache_creation_input_tokens?: number;
|
|
};
|
|
|
|
type KiroStreamState = {
|
|
endDetected: boolean;
|
|
finishEmitted: boolean;
|
|
startEmitted: boolean;
|
|
stopSeen: boolean;
|
|
hasToolCalls: boolean;
|
|
toolCallIndex: number;
|
|
seenToolIds: Map<string, number>;
|
|
toolArgsEmitted: Map<string, string>;
|
|
toolArgsBuffered: Map<string, { toolIndex: number; canonical: string }>;
|
|
totalContentLength?: number;
|
|
contextUsagePercentage?: number;
|
|
hasContextUsage?: boolean;
|
|
hasMeteringEvent?: boolean;
|
|
usage?: UsageSummary;
|
|
hasReasoningContent?: boolean;
|
|
reasoningChunkCount?: number;
|
|
// Inline-thinking splitter state (populated only when thinkingExpected=true).
|
|
thinking?: KiroThinkingState;
|
|
};
|
|
|
|
type EventFrame = {
|
|
headers: Record<string, string>;
|
|
payload: JsonRecord | null;
|
|
};
|
|
|
|
class ByteQueue {
|
|
private chunks: Uint8Array[] = [];
|
|
private headOffset = 0;
|
|
length = 0;
|
|
|
|
push(chunk: Uint8Array) {
|
|
if (!(chunk instanceof Uint8Array) || chunk.length === 0) return;
|
|
this.chunks.push(chunk);
|
|
this.length += chunk.length;
|
|
}
|
|
|
|
peekUint32BE(offset = 0): number | null {
|
|
if (this.length < offset + 4) return null;
|
|
|
|
let value = 0;
|
|
for (let i = 0; i < 4; i++) {
|
|
value = (value << 8) | this.byteAt(offset + i);
|
|
}
|
|
return value >>> 0;
|
|
}
|
|
|
|
read(length: number): Uint8Array | null {
|
|
if (length < 0 || this.length < length) return null;
|
|
|
|
const output = new Uint8Array(length);
|
|
let written = 0;
|
|
|
|
while (written < length) {
|
|
const head = this.chunks[0];
|
|
const available = head.length - this.headOffset;
|
|
const take = Math.min(available, length - written);
|
|
output.set(head.subarray(this.headOffset, this.headOffset + take), written);
|
|
written += take;
|
|
this.headOffset += take;
|
|
this.length -= take;
|
|
|
|
if (this.headOffset >= head.length) {
|
|
this.chunks.shift();
|
|
this.headOffset = 0;
|
|
}
|
|
}
|
|
|
|
return output;
|
|
}
|
|
|
|
private byteAt(offset: number): number {
|
|
let remaining = offset;
|
|
for (let i = 0; i < this.chunks.length; i++) {
|
|
const chunk = this.chunks[i];
|
|
const start = i === 0 ? this.headOffset : 0;
|
|
const available = chunk.length - start;
|
|
if (remaining < available) {
|
|
return chunk[start + remaining];
|
|
}
|
|
remaining -= available;
|
|
}
|
|
return 0;
|
|
}
|
|
}
|
|
|
|
// ── CRC32 lookup table (IEEE polynomial, no dependency) ──
|
|
const CRC32_TABLE = new Uint32Array(256);
|
|
const TEXT_ENCODER = new TextEncoder();
|
|
const TEXT_DECODER = new TextDecoder();
|
|
for (let i = 0; i < 256; i++) {
|
|
let c = i;
|
|
for (let j = 0; j < 8; j++) {
|
|
c = c & 1 ? 0xedb88320 ^ (c >>> 1) : c >>> 1;
|
|
}
|
|
CRC32_TABLE[i] = c >>> 0;
|
|
}
|
|
|
|
// Full per-frame message-CRC validation is O(frame bytes) and runs for EVERY frame of
|
|
// every Kiro response on the main thread. The transport is TLS-protected and the 8-byte
|
|
// prelude CRC already guards framing, so the full-message CRC is redundant overhead that
|
|
// contributes to the CPU-runaway on large/long generations. Keep it opt-in for debugging.
|
|
const KIRO_VERIFY_FULL_CRC = process.env.KIRO_VERIFY_FULL_CRC === "true";
|
|
|
|
function crc32(buf: Uint8Array) {
|
|
let crc = 0xffffffff;
|
|
for (let i = 0; i < buf.length; i++) {
|
|
crc = CRC32_TABLE[(crc ^ buf[i]) & 0xff] ^ (crc >>> 8);
|
|
}
|
|
return (crc ^ 0xffffffff) >>> 0;
|
|
}
|
|
|
|
/**
|
|
* Flush buffered tool arguments at finish boundaries.
|
|
*
|
|
* Kiro/CodeWhisperer streams toolUseEvent.input as PARTIAL OBJECTS that grow over time
|
|
* (e.g. {command:"cat /home"} then {command:"cat /home/wxsys"}). Re-stringifying each one
|
|
* and emitting it as an OpenAI argument delta produces overlapping prefixes that
|
|
* concatenate into unparseable garbage downstream ("Unterminated string").
|
|
*
|
|
* Fix: defer object-form payloads into state.toolArgsBuffered keyed by toolCallId, keep
|
|
* only the latest canonical, and emit ONCE here as the complete arguments string (the
|
|
* final object is the source of truth — intermediate states are noise). String-form
|
|
* payloads are already concatenable deltas and are emitted incrementally.
|
|
*/
|
|
export function flushBufferedToolArgs(
|
|
state: Pick<KiroStreamState, "toolArgsBuffered" | "toolArgsEmitted">,
|
|
controller: { enqueue: (chunk: Uint8Array) => void },
|
|
ctx: { responseId: string; created: number; model: string }
|
|
): void {
|
|
if (!state.toolArgsBuffered || state.toolArgsBuffered.size === 0) return;
|
|
const { responseId, created, model } = ctx;
|
|
for (const [toolCallId, info] of state.toolArgsBuffered) {
|
|
const alreadyEmitted = state.toolArgsEmitted.get(toolCallId) || "";
|
|
if (info.canonical && info.canonical !== alreadyEmitted) {
|
|
const argsChunk: JsonRecord = {
|
|
id: responseId,
|
|
object: "chat.completion.chunk",
|
|
created,
|
|
model,
|
|
choices: [
|
|
{
|
|
index: 0,
|
|
delta: {
|
|
tool_calls: [
|
|
{
|
|
index: info.toolIndex,
|
|
function: { arguments: info.canonical },
|
|
},
|
|
],
|
|
},
|
|
finish_reason: null,
|
|
},
|
|
],
|
|
};
|
|
controller.enqueue(TEXT_ENCODER.encode(`data: ${JSON.stringify(argsChunk)}\n\n`));
|
|
state.toolArgsEmitted.set(toolCallId, info.canonical);
|
|
}
|
|
}
|
|
state.toolArgsBuffered.clear();
|
|
}
|
|
|
|
function buildKiroFinishChunk(
|
|
state: KiroStreamState,
|
|
responseId: string,
|
|
created: number,
|
|
model: string,
|
|
includeUsage: boolean
|
|
): JsonRecord {
|
|
const finishChunk: JsonRecord = {
|
|
id: responseId,
|
|
object: "chat.completion.chunk",
|
|
created,
|
|
model,
|
|
choices: [
|
|
{
|
|
index: 0,
|
|
delta: {},
|
|
finish_reason: state.hasToolCalls ? "tool_calls" : "stop",
|
|
},
|
|
],
|
|
};
|
|
|
|
if (includeUsage && state.usage) {
|
|
finishChunk.usage = state.usage;
|
|
}
|
|
|
|
return finishChunk;
|
|
}
|
|
|
|
function ensureKiroUsage(state: KiroStreamState) {
|
|
if (state.usage) return;
|
|
|
|
const estimatedOutputTokens =
|
|
state.totalContentLength && state.totalContentLength > 0
|
|
? Math.max(1, Math.floor(state.totalContentLength / 4))
|
|
: 0;
|
|
|
|
const estimatedInputTokens =
|
|
state.contextUsagePercentage && state.contextUsagePercentage > 0
|
|
? Math.floor((state.contextUsagePercentage * 200000) / 100)
|
|
: 0;
|
|
|
|
if (estimatedInputTokens <= 0 && estimatedOutputTokens <= 0) return;
|
|
|
|
state.usage = {
|
|
prompt_tokens: estimatedInputTokens,
|
|
completion_tokens: estimatedOutputTokens,
|
|
total_tokens: estimatedInputTokens + estimatedOutputTokens,
|
|
};
|
|
}
|
|
|
|
/**
|
|
* Resolve the AWS region for a Kiro/CodeWhisperer connection. Enterprise AWS IAM Identity
|
|
* Center accounts are region-bound: the access token, the Q Developer profile ARN and the
|
|
* runtime endpoint must all match the region the IdC instance lives in (e.g. eu-central-1).
|
|
* A request signed for one region is rejected by another ("bearer token is invalid"), and a
|
|
* regional profileArn sent to us-east-1 fails with "Improperly formed request". Falls back to
|
|
* the region embedded in the profileArn, then us-east-1 (the AWS Builder ID default).
|
|
*/
|
|
export function resolveKiroRegion(
|
|
credentials: { providerSpecificData?: unknown } | null | undefined
|
|
): string {
|
|
const psd = (credentials?.providerSpecificData || {}) as Record<string, unknown>;
|
|
const region = typeof psd.region === "string" ? psd.region.trim().toLowerCase() : "";
|
|
if (region) return region;
|
|
const arn = typeof psd.profileArn === "string" ? psd.profileArn.toLowerCase() : "";
|
|
const match = arn.match(/^arn:aws:codewhisperer:([a-z0-9-]+):/);
|
|
return match ? match[1] : "us-east-1";
|
|
}
|
|
|
|
/**
|
|
* CodeWhisperer/Amazon Q runtime host for a region. us-east-1 keeps the legacy
|
|
* codewhisperer.us-east-1 host (AWS Builder ID); other regions use the regional Amazon Q
|
|
* endpoint q.{region}.amazonaws.com — codewhisperer.{region}.amazonaws.com does not resolve
|
|
* for non-us-east-1 regions.
|
|
*/
|
|
export function kiroRuntimeHost(region: string): string {
|
|
return region === "us-east-1"
|
|
? "https://codewhisperer.us-east-1.amazonaws.com"
|
|
: `https://q.${region}.amazonaws.com`;
|
|
}
|
|
|
|
/**
|
|
* KiroExecutor - Executor for Kiro AI (AWS CodeWhisperer)
|
|
* Uses AWS CodeWhisperer streaming API with AWS EventStream binary format
|
|
*/
|
|
export class KiroExecutor extends BaseExecutor {
|
|
constructor(providerId = "kiro") {
|
|
super(providerId, PROVIDERS[providerId] || PROVIDERS.kiro);
|
|
}
|
|
|
|
buildHeaders(credentials: ProviderCredentials, stream = true) {
|
|
void stream;
|
|
const headers = {
|
|
...this.config.headers,
|
|
"Amz-Sdk-Request": "attempt=1; max=3",
|
|
"Amz-Sdk-Invocation-Id": uuidv4(),
|
|
"x-amzn-bedrock-cache-control": "enable",
|
|
"anthropic-beta": "prompt-caching-2024-07-31",
|
|
};
|
|
|
|
if (credentials.accessToken) {
|
|
headers["Authorization"] = `Bearer ${credentials.accessToken}`;
|
|
}
|
|
|
|
return headers;
|
|
}
|
|
|
|
transformRequest(model: string, body: unknown, stream: boolean, credentials: unknown): unknown {
|
|
void stream;
|
|
void credentials;
|
|
const b = body as Record<string, unknown>;
|
|
|
|
// Kiro API is strict and rejects any unknown top-level fields (like 'tools', 'stream', 'model', etc.)
|
|
// We only preserve the fields specifically built by the openai-to-kiro translator.
|
|
const kiroPayload: Record<string, unknown> = {};
|
|
if (b.conversationState !== undefined) kiroPayload.conversationState = b.conversationState;
|
|
if (b.profileArn !== undefined) kiroPayload.profileArn = b.profileArn;
|
|
if (b.inferenceConfig !== undefined) kiroPayload.inferenceConfig = b.inferenceConfig;
|
|
|
|
// Fallback: if somehow conversationState isn't there, return the rest without model
|
|
// (for backward compatibility if something else bypasses the translator)
|
|
if (!kiroPayload.conversationState) {
|
|
const { model: _model, ...rest } = b;
|
|
return rest;
|
|
}
|
|
|
|
return kiroPayload;
|
|
}
|
|
|
|
/**
|
|
* Custom execute for Kiro - handles AWS EventStream binary response
|
|
*/
|
|
async execute({
|
|
model,
|
|
body,
|
|
stream,
|
|
credentials,
|
|
signal,
|
|
log,
|
|
upstreamExtraHeaders,
|
|
}: ExecuteInput) {
|
|
// Route to the region-specific CodeWhisperer/Amazon Q endpoint. Enterprise IAM Identity
|
|
// Center accounts (e.g. eu-central-1) are rejected by the default us-east-1 host; only the
|
|
// regional endpoint accepts the region-bound token + profileArn.
|
|
const region = resolveKiroRegion(credentials);
|
|
const url = `${kiroRuntimeHost(region)}/generateAssistantResponse`;
|
|
const headers = this.buildHeaders(credentials, stream);
|
|
mergeUpstreamExtraHeaders(headers, upstreamExtraHeaders);
|
|
const transformedBody = await this.transformRequest(model, body, stream, credentials);
|
|
|
|
const response = await fetch(url, {
|
|
method: "POST",
|
|
headers,
|
|
body: JSON.stringify(transformedBody),
|
|
signal,
|
|
});
|
|
|
|
if (!response.ok) {
|
|
return { response, url, headers, transformedBody };
|
|
}
|
|
|
|
// For Kiro, we need to transform the binary EventStream to SSE.
|
|
// Create a TransformStream to convert binary to SSE text.
|
|
//
|
|
// When the user enabled thinking, Claude on Kiro streams its reasoning
|
|
// **inline** as `<thinking>…</thinking>` blocks inside
|
|
// `assistantResponseEvent.content` rather than as separate
|
|
// `reasoningContentEvent` frames. We pass a hint so the transform stream
|
|
// can split that inline reasoning into the OpenAI `delta.reasoning_content`
|
|
// channel.
|
|
const tb = transformedBody as Record<string, unknown>;
|
|
const userContent =
|
|
(
|
|
(
|
|
(
|
|
(tb?.conversationState as Record<string, unknown>)
|
|
?.currentMessage as Record<string, unknown>
|
|
)?.userInputMessage as Record<string, unknown>
|
|
)?.content as string
|
|
) || "";
|
|
const thinkingExpected = userContent.includes("<thinking_mode>enabled</thinking_mode>");
|
|
const transformedResponse = this.transformEventStreamToSSE(response, model, { thinkingExpected });
|
|
|
|
return { response: transformedResponse, url, headers, transformedBody };
|
|
}
|
|
|
|
/**
|
|
* Transform AWS EventStream binary response to SSE text stream.
|
|
* Using TransformStream instead of ReadableStream.pull() to avoid Workers timeout.
|
|
*
|
|
* @param response Upstream raw fetch response (binary EventStream).
|
|
* @param model Logical model id (kept in OpenAI chunks for clients).
|
|
* @param opts
|
|
* @param opts.thinkingExpected When true, scan inbound
|
|
* `assistantResponseEvent.content` for inline `<thinking>…</thinking>`
|
|
* blocks and split them into the OpenAI `delta.reasoning_content` channel.
|
|
* Required for Claude on Kiro when `<thinking_mode>enabled</thinking_mode>`
|
|
* is in the system prompt, because Kiro streams reasoning inline rather
|
|
* than as separate `reasoningContentEvent` frames.
|
|
*/
|
|
transformEventStreamToSSE(
|
|
response: Response,
|
|
model: string,
|
|
opts: { thinkingExpected?: boolean } = {}
|
|
) {
|
|
const thinkingExpected = !!opts.thinkingExpected;
|
|
const buffer = new ByteQueue();
|
|
let chunkIndex = 0;
|
|
const responseId = `chatcmpl-${Date.now()}`;
|
|
const created = Math.floor(Date.now() / 1000);
|
|
const state: KiroStreamState = {
|
|
endDetected: false,
|
|
finishEmitted: false,
|
|
startEmitted: false,
|
|
stopSeen: false,
|
|
hasToolCalls: false,
|
|
toolCallIndex: 0,
|
|
seenToolIds: new Map(),
|
|
toolArgsEmitted: new Map(),
|
|
toolArgsBuffered: new Map(),
|
|
hasReasoningContent: false,
|
|
reasoningChunkCount: 0,
|
|
thinking: thinkingExpected ? { thinkingMode: false, pendingTag: "" } : undefined,
|
|
};
|
|
|
|
const transformStream = new TransformStream(
|
|
{
|
|
async transform(chunk, controller) {
|
|
buffer.push(chunk);
|
|
|
|
// Parse events from buffer
|
|
let iterations = 0;
|
|
const maxIterations = 1000;
|
|
while (buffer.length >= 16 && iterations < maxIterations) {
|
|
iterations++;
|
|
const totalLength = buffer.peekUint32BE(0);
|
|
|
|
if (!totalLength || totalLength < 16 || totalLength > buffer.length) break;
|
|
|
|
const eventData = buffer.read(totalLength);
|
|
if (!eventData) break;
|
|
|
|
const event = parseEventFrame(eventData);
|
|
if (!event) continue;
|
|
|
|
// Emit a role-only start chunk on the FIRST successfully-parsed AWS
|
|
// EventStream frame. CodeWhisperer sends framing/metadata events before
|
|
// the first content token, and on large/agentic contexts the gap before
|
|
// that first `assistantResponseEvent` can be many seconds. The backend
|
|
// stream-readiness gate (ensureStreamReadiness) holds the ENTIRE response
|
|
// from the client until it observes a useful SSE frame, so without an
|
|
// early frame the client sees a frozen connection for that whole window
|
|
// (up to STREAM_READINESS_TIMEOUT_MS — 180s as configured by VibeProxy),
|
|
// then a burst — the "minutes instead of seconds, not streaming" symptom.
|
|
// A role-only `chat.completion.chunk` is a non-ping structured payload, so
|
|
// it satisfies hasStreamReadinessSignal and hands the stream off
|
|
// immediately. Mirrors the early lifecycle frame other executors already
|
|
// emit (Claude message_start / OpenAI response.created). The downstream
|
|
// idle timeout still guards genuine post-start stalls.
|
|
if (!state.startEmitted) {
|
|
state.startEmitted = true;
|
|
const startChunk: JsonRecord = {
|
|
id: responseId,
|
|
object: "chat.completion.chunk",
|
|
created,
|
|
model,
|
|
choices: [
|
|
{
|
|
index: 0,
|
|
delta: { role: "assistant" },
|
|
finish_reason: null,
|
|
},
|
|
],
|
|
};
|
|
chunkIndex++;
|
|
controller.enqueue(TEXT_ENCODER.encode(`data: ${JSON.stringify(startChunk)}\n\n`));
|
|
}
|
|
|
|
const eventType = event.headers[":event-type"] || "";
|
|
|
|
// Track total content length for token estimation
|
|
if (!state.totalContentLength) state.totalContentLength = 0;
|
|
if (!state.contextUsagePercentage) state.contextUsagePercentage = 0;
|
|
|
|
// Handle assistantResponseEvent
|
|
if (eventType === "assistantResponseEvent") {
|
|
const content =
|
|
typeof event.payload?.content === "string" ? event.payload.content : "";
|
|
if (!content) {
|
|
continue;
|
|
}
|
|
state.totalContentLength += content.length;
|
|
|
|
if (thinkingExpected && state.thinking) {
|
|
// Claude on Kiro emits reasoning inline as `<thinking>…</thinking>`
|
|
// when `<thinking_mode>enabled</thinking_mode>` is in the system prompt.
|
|
// Split it into the OpenAI `reasoning_content` channel so downstream
|
|
// consumers see the same shape they would get from a native reasoning model.
|
|
const thinkingState = state.thinking;
|
|
splitInlineThinking(
|
|
thinkingState,
|
|
content,
|
|
(text) => {
|
|
if (!text) return;
|
|
const chunk: JsonRecord = {
|
|
id: responseId,
|
|
object: "chat.completion.chunk",
|
|
created,
|
|
model,
|
|
choices: [
|
|
{
|
|
index: 0,
|
|
delta: chunkIndex === 0 ? { role: "assistant", content: text } : { content: text },
|
|
finish_reason: null,
|
|
},
|
|
],
|
|
};
|
|
chunkIndex++;
|
|
controller.enqueue(TEXT_ENCODER.encode(`data: ${JSON.stringify(chunk)}\n\n`));
|
|
},
|
|
(reasoning) => {
|
|
if (!reasoning) return;
|
|
state.hasReasoningContent = true;
|
|
const reasoningDelta: JsonRecord =
|
|
(state.reasoningChunkCount ?? 0) === 0 && chunkIndex === 0
|
|
? { role: "assistant", reasoning_content: reasoning }
|
|
: { reasoning_content: reasoning };
|
|
const chunk: JsonRecord = {
|
|
id: responseId,
|
|
object: "chat.completion.chunk",
|
|
created,
|
|
model,
|
|
choices: [
|
|
{
|
|
index: 0,
|
|
delta: reasoningDelta,
|
|
finish_reason: null,
|
|
},
|
|
],
|
|
};
|
|
chunkIndex++;
|
|
state.reasoningChunkCount = (state.reasoningChunkCount ?? 0) + 1;
|
|
controller.enqueue(TEXT_ENCODER.encode(`data: ${JSON.stringify(chunk)}\n\n`));
|
|
}
|
|
);
|
|
} else {
|
|
const chunk: JsonRecord = {
|
|
id: responseId,
|
|
object: "chat.completion.chunk",
|
|
created,
|
|
model,
|
|
choices: [
|
|
{
|
|
index: 0,
|
|
delta: chunkIndex === 0 ? { role: "assistant", content } : { content },
|
|
finish_reason: null,
|
|
},
|
|
],
|
|
};
|
|
chunkIndex++;
|
|
controller.enqueue(TEXT_ENCODER.encode(`data: ${JSON.stringify(chunk)}\n\n`));
|
|
}
|
|
}
|
|
|
|
// Handle codeEvent
|
|
if (eventType === "codeEvent" && event.payload?.content) {
|
|
const chunk: JsonRecord = {
|
|
id: responseId,
|
|
object: "chat.completion.chunk",
|
|
created,
|
|
model,
|
|
choices: [
|
|
{
|
|
index: 0,
|
|
delta: { content: event.payload.content },
|
|
finish_reason: null,
|
|
},
|
|
],
|
|
};
|
|
chunkIndex++;
|
|
controller.enqueue(TEXT_ENCODER.encode(`data: ${JSON.stringify(chunk)}\n\n`));
|
|
}
|
|
|
|
// Handle toolUseEvent
|
|
if (eventType === "toolUseEvent" && event.payload) {
|
|
state.hasToolCalls = true;
|
|
const toolUse = event.payload;
|
|
const toolUses = Array.isArray(toolUse) ? toolUse : [toolUse];
|
|
|
|
for (const singleToolUse of toolUses) {
|
|
const toolCallId = singleToolUse.toolUseId || `call_${Date.now()}`;
|
|
const toolName = singleToolUse.name || "";
|
|
const toolInput = singleToolUse.input;
|
|
|
|
let toolIndex;
|
|
const isNewTool = !state.seenToolIds.has(toolCallId);
|
|
|
|
if (isNewTool) {
|
|
toolIndex = state.toolCallIndex++;
|
|
state.seenToolIds.set(toolCallId, toolIndex);
|
|
|
|
const startChunk = {
|
|
id: responseId,
|
|
object: "chat.completion.chunk",
|
|
created,
|
|
model,
|
|
choices: [
|
|
{
|
|
index: 0,
|
|
delta: {
|
|
...(chunkIndex === 0 ? { role: "assistant" } : {}),
|
|
tool_calls: [
|
|
{
|
|
index: toolIndex,
|
|
id: toolCallId,
|
|
type: "function",
|
|
function: {
|
|
name: toolName,
|
|
arguments: "",
|
|
},
|
|
},
|
|
],
|
|
},
|
|
finish_reason: null,
|
|
},
|
|
],
|
|
};
|
|
chunkIndex++;
|
|
controller.enqueue(
|
|
TEXT_ENCODER.encode(`data: ${JSON.stringify(startChunk)}\n\n`)
|
|
);
|
|
} else {
|
|
toolIndex = state.seenToolIds.get(toolCallId);
|
|
}
|
|
|
|
if (toolInput !== undefined) {
|
|
if (typeof toolInput === "string") {
|
|
// String-form payloads are already concatenable incremental deltas —
|
|
// emit immediately and track what we've sent.
|
|
state.toolArgsEmitted.set(
|
|
toolCallId,
|
|
(state.toolArgsEmitted.get(toolCallId) || "") + toolInput
|
|
);
|
|
|
|
const argsChunk = {
|
|
id: responseId,
|
|
object: "chat.completion.chunk",
|
|
created,
|
|
model,
|
|
choices: [
|
|
{
|
|
index: 0,
|
|
delta: {
|
|
tool_calls: [
|
|
{
|
|
index: toolIndex,
|
|
function: {
|
|
arguments: toolInput,
|
|
},
|
|
},
|
|
],
|
|
},
|
|
finish_reason: null,
|
|
},
|
|
],
|
|
};
|
|
chunkIndex++;
|
|
controller.enqueue(
|
|
TEXT_ENCODER.encode(`data: ${JSON.stringify(argsChunk)}\n\n`)
|
|
);
|
|
} else if (typeof toolInput === "object" && toolInput !== null) {
|
|
// Object-form payloads are PARTIAL OBJECTS that grow over time. Buffer
|
|
// the latest canonical and flush once at a finish boundary, otherwise the
|
|
// overlapping JSON prefixes concatenate into unparseable garbage.
|
|
state.toolArgsBuffered.set(toolCallId, {
|
|
toolIndex,
|
|
canonical: JSON.stringify(toolInput),
|
|
});
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// Handle messageStopEvent
|
|
if (eventType === "messageStopEvent") {
|
|
flushBufferedToolArgs(state, controller, { responseId, created, model });
|
|
state.stopSeen = true;
|
|
}
|
|
|
|
// Handle contextUsageEvent to extract contextUsagePercentage
|
|
if (eventType === "contextUsageEvent") {
|
|
const contextUsage =
|
|
typeof event.payload?.contextUsagePercentage === "number"
|
|
? event.payload.contextUsagePercentage
|
|
: 0;
|
|
if (contextUsage <= 0) {
|
|
continue;
|
|
}
|
|
state.contextUsagePercentage = contextUsage;
|
|
// Mark that we received context usage event
|
|
state.hasContextUsage = true;
|
|
}
|
|
|
|
// Handle meteringEvent - mark that we received it
|
|
if (eventType === "meteringEvent") {
|
|
state.hasMeteringEvent = true;
|
|
}
|
|
|
|
// Handle metricsEvent for token usage
|
|
if (eventType === "metricsEvent") {
|
|
// Extract usage data from metricsEvent payload
|
|
const metrics = event.payload?.metricsEvent || event.payload;
|
|
if (metrics && typeof metrics === "object") {
|
|
const inputTokens =
|
|
typeof (metrics as JsonRecord).inputTokens === "number"
|
|
? ((metrics as JsonRecord).inputTokens as number)
|
|
: 0;
|
|
const outputTokens =
|
|
typeof (metrics as JsonRecord).outputTokens === "number"
|
|
? ((metrics as JsonRecord).outputTokens as number)
|
|
: 0;
|
|
|
|
const cacheReadTokens =
|
|
typeof (metrics as JsonRecord).cacheReadTokens === "number"
|
|
? ((metrics as JsonRecord).cacheReadTokens as number)
|
|
: 0;
|
|
|
|
const cacheCreationTokens =
|
|
typeof (metrics as JsonRecord).cacheCreationTokens === "number"
|
|
? ((metrics as JsonRecord).cacheCreationTokens as number)
|
|
: 0;
|
|
|
|
if (inputTokens > 0 || outputTokens > 0) {
|
|
state.usage = {
|
|
prompt_tokens: inputTokens,
|
|
completion_tokens: outputTokens,
|
|
total_tokens: inputTokens + outputTokens,
|
|
...(cacheReadTokens > 0 && { cache_read_input_tokens: cacheReadTokens }),
|
|
...(cacheCreationTokens > 0 && {
|
|
cache_creation_input_tokens: cacheCreationTokens,
|
|
}),
|
|
};
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
if (iterations >= maxIterations) {
|
|
console.warn("[Kiro] Max iterations reached in event parsing");
|
|
}
|
|
},
|
|
|
|
flush(controller) {
|
|
// Flush any buffered tool arguments (partial-object payloads) before finishing —
|
|
// idempotent against toolArgsEmitted if messageStopEvent already flushed them.
|
|
flushBufferedToolArgs(state, controller, { responseId, created, model });
|
|
|
|
// Drain any pending inline-thinking tag fragment so we don't drop
|
|
// trailing characters when the stream ends mid-tag (e.g. `<thi`).
|
|
if (thinkingExpected && state.thinking) {
|
|
const thinkingState = state.thinking;
|
|
flushPendingThinking(
|
|
thinkingState,
|
|
(text) => {
|
|
if (!text) return;
|
|
const chunk: JsonRecord = {
|
|
id: responseId,
|
|
object: "chat.completion.chunk",
|
|
created,
|
|
model,
|
|
choices: [{ index: 0, delta: { content: text }, finish_reason: null }],
|
|
};
|
|
chunkIndex++;
|
|
controller.enqueue(TEXT_ENCODER.encode(`data: ${JSON.stringify(chunk)}\n\n`));
|
|
},
|
|
(reasoning) => {
|
|
if (!reasoning) return;
|
|
const chunk: JsonRecord = {
|
|
id: responseId,
|
|
object: "chat.completion.chunk",
|
|
created,
|
|
model,
|
|
choices: [
|
|
{ index: 0, delta: { reasoning_content: reasoning }, finish_reason: null },
|
|
],
|
|
};
|
|
chunkIndex++;
|
|
controller.enqueue(TEXT_ENCODER.encode(`data: ${JSON.stringify(chunk)}\n\n`));
|
|
}
|
|
);
|
|
}
|
|
|
|
// Emit finish chunk if not already sent
|
|
if (!state.finishEmitted) {
|
|
state.finishEmitted = true;
|
|
ensureKiroUsage(state);
|
|
const finishChunk = buildKiroFinishChunk(state, responseId, created, model, true);
|
|
controller.enqueue(TEXT_ENCODER.encode(`data: ${JSON.stringify(finishChunk)}\n\n`));
|
|
}
|
|
|
|
// Send final done message
|
|
controller.enqueue(TEXT_ENCODER.encode("data: [DONE]\n\n"));
|
|
},
|
|
},
|
|
{ highWaterMark: 16384 },
|
|
{ highWaterMark: 16384 }
|
|
);
|
|
|
|
// Pipe response body through transform stream
|
|
const transformedStream = response.body.pipeThrough(transformStream);
|
|
|
|
return new Response(transformedStream, {
|
|
status: response.status,
|
|
statusText: response.statusText,
|
|
headers: {
|
|
"Content-Type": "text/event-stream",
|
|
"Cache-Control": "no-cache",
|
|
Connection: "keep-alive",
|
|
},
|
|
});
|
|
}
|
|
|
|
async refreshCredentials(credentials: ProviderCredentials, log?: ExecutorLog | null) {
|
|
if (!credentials.refreshToken) return null;
|
|
|
|
try {
|
|
// Use centralized refreshKiroToken function (handles both AWS SSO OIDC and Social Auth)
|
|
const result = await refreshKiroToken(
|
|
credentials.refreshToken,
|
|
credentials.providerSpecificData,
|
|
log
|
|
);
|
|
|
|
if (!result || result.error) return result;
|
|
|
|
// If client was re-registered (expired/invalid clientId/clientSecret after DB import,
|
|
// TTL expiry, or browser conflict), update providerSpecificData with new credentials (#2524).
|
|
if (result._newClientId) {
|
|
const updatedPsd = {
|
|
...(credentials.providerSpecificData || {}),
|
|
clientId: result._newClientId,
|
|
clientSecret: result._newClientSecret,
|
|
clientSecretExpiresAt: result._newClientSecretExpiresAt,
|
|
};
|
|
return {
|
|
accessToken: result.accessToken,
|
|
refreshToken: result.refreshToken,
|
|
expiresIn: result.expiresIn,
|
|
providerSpecificData: updatedPsd,
|
|
};
|
|
}
|
|
|
|
return result;
|
|
} catch (error) {
|
|
const err = error instanceof Error ? error : new Error(String(error));
|
|
log?.error?.("TOKEN", `Kiro refresh error: ${err.message}`);
|
|
return null;
|
|
}
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Parse AWS EventStream frame
|
|
*/
|
|
function parseEventFrame(data: Uint8Array): EventFrame | null {
|
|
try {
|
|
const view = new DataView(data.buffer, data.byteOffset);
|
|
const totalLength = view.getUint32(0, false);
|
|
const headersLength = view.getUint32(4, false);
|
|
|
|
// ── CRC32 validation ──
|
|
// Prelude CRC covers bytes [0..7] (totalLength + headersLength)
|
|
const preludeCRC = view.getUint32(8, false);
|
|
const computedPreludeCRC = crc32(data.slice(0, 8));
|
|
if (preludeCRC !== computedPreludeCRC) {
|
|
console.warn(
|
|
`[Kiro] Prelude CRC mismatch: expected ${preludeCRC}, got ${computedPreludeCRC} — skipping corrupted frame`
|
|
);
|
|
return null;
|
|
}
|
|
|
|
// Message CRC covers bytes [0..totalLength-5] (everything except the CRC itself).
|
|
// Skipped by default (O(frame bytes) per frame) — the prelude CRC above already
|
|
// validates framing and the stream is TLS-protected. Enable KIRO_VERIFY_FULL_CRC=true
|
|
// to restore full validation for debugging corrupted-stream issues.
|
|
if (KIRO_VERIFY_FULL_CRC) {
|
|
const messageCRC = view.getUint32(data.length - 4, false);
|
|
const computedMessageCRC = crc32(data.slice(0, data.length - 4));
|
|
if (messageCRC !== computedMessageCRC) {
|
|
console.warn(
|
|
`[Kiro] Message CRC mismatch: expected ${messageCRC}, got ${computedMessageCRC} — skipping corrupted frame`
|
|
);
|
|
return null;
|
|
}
|
|
}
|
|
// Parse headers
|
|
const headers: Record<string, string> = {};
|
|
let offset = 12; // After prelude
|
|
const headerEnd = 12 + headersLength;
|
|
|
|
while (offset < headerEnd && offset < data.length) {
|
|
const nameLen = data[offset];
|
|
offset++;
|
|
if (offset + nameLen > data.length) break;
|
|
|
|
const name = TEXT_DECODER.decode(data.subarray(offset, offset + nameLen));
|
|
offset += nameLen;
|
|
|
|
const headerType = data[offset];
|
|
offset++;
|
|
|
|
if (headerType === 7) {
|
|
// String type
|
|
const valueLen = (data[offset] << 8) | data[offset + 1];
|
|
offset += 2;
|
|
if (offset + valueLen > data.length) break;
|
|
|
|
const value = TEXT_DECODER.decode(data.subarray(offset, offset + valueLen));
|
|
offset += valueLen;
|
|
headers[name] = value;
|
|
} else {
|
|
break;
|
|
}
|
|
}
|
|
|
|
// Parse payload
|
|
const payloadStart = 12 + headersLength;
|
|
const payloadEnd = data.length - 4; // Exclude message CRC
|
|
|
|
let payload: JsonRecord | null = null;
|
|
if (payloadEnd > payloadStart) {
|
|
const payloadStr = TEXT_DECODER.decode(data.subarray(payloadStart, payloadEnd));
|
|
|
|
// Skip empty or whitespace-only payloads
|
|
if (!payloadStr || !payloadStr.trim()) {
|
|
return { headers, payload: null };
|
|
}
|
|
|
|
try {
|
|
payload = JSON.parse(payloadStr);
|
|
} catch (parseError) {
|
|
const err = parseError instanceof Error ? parseError : new Error(String(parseError));
|
|
// Log parse error for debugging
|
|
console.warn(
|
|
`[Kiro] Failed to parse payload: ${err.message} | payload: ${payloadStr.substring(0, 100)}`
|
|
);
|
|
payload = { raw: payloadStr };
|
|
}
|
|
}
|
|
|
|
return { headers, payload };
|
|
} catch (err) {
|
|
const error = err instanceof Error ? err : new Error(String(err));
|
|
console.warn(`[Kiro] Frame parse error: ${error.message}`);
|
|
return null;
|
|
}
|
|
}
|
|
|
|
export default KiroExecutor;
|