mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-07-31 20:32:20 +03:00
* chore(release): open v3.8.39 development cycle * docs(changelog): backfill 5 v3.8.38 bullets merged after release finalize These PRs squash-merged into release/v3.8.38 between the CHANGELOG finalize (ff57be32f) and the merge-to-main (ae6e2342d), so they shipped in the v3.8.38 tag but had no bullet: - feat(compression): Ionizer engine (lossy JSON-array sampling + CCR) (#5148) - fix(sse): preserve non-stream reasoning fields (#5155, @rdself) - fix(i18n): add missing English UI labels (#5153, @rdself) - test(combo): gated live smoke (#5151) + release-expectations refresh (#5150, @KooshaPari) (#5129 exact-host Anthropic baseUrl is already covered by the #5130 bullet — same CodeQL #674.) Synced 41 i18n CHANGELOG mirrors. * feat(compression): TOON best-of-N candidate encoder + encoder A/B table (#5163) Integrated into release/v3.8.39. TOON best-of-N candidate encoder (GCF default, fail-open). 17/17 unit tests pass on merge result; CI reds were base-stale + Quality Ratchet DRIFT. * fix(zenmux): normalize vendor-prefixed GLM system roles (#5158) Integrated into release/v3.8.39. ZenMux vendor-prefixed GLM system-role normalization; 12/12 role-normalizer tests pass on merge result. CI reds base-stale. * [codex] fix xAI OAuth test and reasoning effort (#5157) Integrated into release/v3.8.39. xAI reasoning-effort normalization (max/xhigh→high) + OAuth test config; 46/46 xai-translator tests pass on merge result. CI reds base-stale. * docs(i18n): add Traditional Chinese (zh-TW) README and update zh-CN to latest (#5162) Integrated into release/v3.8.39. Traditional Chinese (zh-TW) README + zh-CN refresh; docs-only. * test(security): guard PII redaction stays opt-in (default off) + Hard Rule #20 (#5159) Integrated into release/v3.8.39. PII opt-in regression guard + Hard Rule #20; rebased to strip base-drift (+81/-1). 5/5 guard tests pass; flip-proof verified. * test(combo): deterministic context-relay universal-handoff coverage (closes phase-2 TODO) (#5168) Integrated into release/v3.8.39. Deterministic context-relay universal-handoff coverage (3 tests); 3/3 pass on merge result. * docs(i18n): full sync zh-TW and zh-CN README with canonical English v3.8.39 (#5171) Integrated into release/v3.8.39. Full zh-TW docs tree + zh-CN sync with canonical English v3.8.39; docs-only. * fix(serve): honour HOSTNAME from .env instead of hardcoding 0.0.0.0 (#5134) (#5170) Integrated into release/v3.8.39. HOSTNAME env override in serve (#5134) + regression test (4/4, TDD flip-proof verified). * fix(sse): resolve nameless deepseek-web tool blocks via parameter-schema match (#5154) (#5173) Integrated into release/v3.8.39. Schema-based nameless deepseek-web tool-block resolution (#5154); 6/6 tests pass on merge result (incl. ambiguous/no-match negatives + named-tag no-regression). * fix(sse): normalize array user content for Command Code to avoid upstream 400 (#5166) (#5174) Integrated into release/v3.8.39. Normalize array user content for Command Code (#5166, user-array/400 symptom); 4/4 tests pass on merge result. * fix(sse): defer </think> close so it never leaks before tool_calls (#5123) (#5175) Integrated into release/v3.8.39. Defer </think> close so it never leaks before tool_calls (#5123); 4/4 tests pass (incl. #4633 no-regression). CHANGELOG synced to keep all 3 v3.8.39 fixes. * fix(dashboard): use amber for home update-step warning icon (#5176) Integrated into release/v3.8.39. Amber for home update-step warning icon; 1/1 UI test. * fix(api): LAN/Tailscale dashboard — host-aware CSP + GET-exempt version route + combo field errors (#5083) (#5177) Integrated into release/v3.8.39. Host-aware CSP (ReDoS/injection-safe host validation) + GET-exempt /api/system/version (POST/spawn stays LOCAL_ONLY, exact-match safe-methods-only) + COMBO_002 firstField. 44/44 tests + route-guard membership gate green. CHANGELOG synced to keep all 4 v3.8.39 fixes. * fix(api): replace #5083 global middleware CSP with declarative ws: scheme (#5083) Follow-up to PR #5177 (merged): that version implemented the LAN-CSP fix (Bug 1) with a new global `src/middleware.ts` + `src/server/csp.ts`, which contradicts the project's documented architecture — 'No global Next.js middleware — interception is route-specific' (CLAUDE.md / AGENTS.md) — and was merged unverified (middleware vs next.config header precedence was never confirmed in a real build). This replaces that approach with the minimal, declarative equivalent: • next.config.mjs: connect-src now permits the bare `ws:` scheme (symmetric with the bare `wss:` already allowed) so the dashboard can reach its own Live WS server from a LAN/Tailscale host. No middleware. • Removes src/middleware.ts, src/server/csp.ts, and tests/unit/csp-host-aware.test.ts. • Adds tests/unit/csp-lan-ws-5083.test.ts (incl. a guard asserting src/middleware.ts does NOT exist, so the global-middleware approach cannot silently return). Bugs 2 (GET-exempt /api/system/version) and 3 (COMBO_002 field surfacing) from #5177 are unaffected and remain in place. Co-authored-by: KooshaPari <KooshaPari@users.noreply.github.com> * test(combo): end-to-end quota-share DRR routing-decision coverage (matrix parity) (#5179) Integrated into release/v3.8.39. Quota-share DRR routing-decision coverage (matrix parity); 2/2 pass on merge result. * feat(agent-bridge): graceful cert-install fallback with manual guide for containers (#4546) (#5178) Integrated into release/v3.8.39. Agent-bridge graceful cert-install fallback + manual guide (#4546); 6/6 tests pass on merge result. * fix(antigravity): family-scoped quota lockout (gemini/claude buckets) (#5180) Integrated into release/v3.8.39 — family-scoped antigravity quota lockout. Rebased from v3.8.37 + validated (vitest 5/5, typecheck clean, full combo-matrix green, model-lockout 99/0). Same-model cross-account retry (chat.ts) deferred pending live antigravity VPS validation. * fix(cli): force NODE_ENV to match dev/start run mode in custom Next server (#5189) Integrated into release/v3.8.39. Force NODE_ENV to match dev/start run mode in custom Next server; 2/2 source-scan+ordering tests pass on merge result. * feat(compression): CCR ranged/grep/stats retrieval (ReDoS-safe, backward-compat) (#5187) Integrated into release/v3.8.39. CCR ranged/grep/stats retrieval (safe-regex ReDoS guard + length/match caps); 17/17 tests pass on merge result. * docs(combo): sync all combo/routing-strategy docs to current state + document test coverage (#5185) Integrated into release/v3.8.39. Combo/routing-strategy docs sync; docs-only. * fix(mcp): return 404 (not 400) for unknown Streamable HTTP session id (#5169) (#5191) * fix(api): respect blocked Auto (Zero-Config) provider in /v1/models catalog (#5192) (#5194) * test(combo): deterministic context-relay codex quota-handoff coverage (closes last gap) (#5195) * test(ci): wire antigravity-quota-family under test:vitest (fix test-discovery orphan) (#5196) * fix(oauth): antigravity login no longer hangs — fire-and-forget onboarding + bounded post-exchange (#5193) Antigravity OAuth hang fix (no-PKCE/no-openid + bounded post-exchange + exchange-500 fix). Includes #5200 (Koosha) revert + owner rebaseline to keep documented comments. Integrated into release/v3.8.39. * feat(oauth): remote Antigravity login via local helper + paste-credentials (#5203) Remote Antigravity login: local helper (omniroute login antigravity) + paste-credentials. Integrated into release/v3.8.39. * fix(translator): accept Claude Messages shape in non-stream malformed-200 guard (#5156) Integrated into release/v3.8.39 * fix(cli): default dev bundler to Turbopack (16.2.x panic no longer reproduces) (#5206) Integrated into release/v3.8.39 * fix(cli): auto-calibrate server V8 heap from physical RAM (#5172) (#5213) The server was spawned with a fixed --max-old-space-size=512 (omniroute serve) or no heap flag at all (Electron), so RAM-rich boxes still OOM-crashed under load (Ineffective mark-compacts near heap limit ~500MB) with many providers/ accounts and large model catalogs. New calibrateHeapFallbackMb(os.totalmem()) defaults the heap to ~35% of RAM clamped [512,4096], wired into serve.mjs and electron/main.js. Explicit OMNIROUTE_MEMORY_MB still wins (#2939 unchanged). Also addresses #5160 (same OOM root); #5152 (docker) benefits via the same knob. Closes #5172 * fix(proxy): coalesce fast-fail health probes (#5208) Integrated into release/v3.8.39 * fix(proxy): close dispatchers when clearing cache (#5202) Integrated into release/v3.8.39 * fix(cli): raise dev server Node heap limit to 8GB to prevent OOM (#5198) Integrated into release/v3.8.39 * fix(auth): allow synthetic no-auth fallback for mimocode (#5205) Integrated into release/v3.8.39 * fix(oauth): preserve Antigravity refresh_token on empty/omitted upstream response (#3850) (#5214) Google's OAuth refresh tokens are non-rotating: the refresh response usually omits refresh_token and occasionally returns it as an empty string. The Antigravity executor used `typeof tokens.refresh_token === "string" ? ... ` which accepts "" (typeof "" === "string") and overwrote the stored token with empty, nulling it on first refresh. Now treats non-string OR empty as absent and preserves credentials.refreshToken, matching refreshGoogleToken semantics. Closes #3850 * fix(responses): normalize non-array input (#5204) Integrated into release/v3.8.39 * fix(stream): normalize safety finish reasons via shared helper (#5197) Integrated into release/v3.8.39 * fix(request-logger): never render negative '(-100%)' compression badge (#5201) Integrated into release/v3.8.39 * fix(combo): reject empty responses api output (#5207) Integrated into release/v3.8.39 — combo failover now rejects empty Responses API output (validateQuality). Baseline rebaseline dropped (main-measured drift; maintainer rebaselines at release). * fix(pwa): prefer cached navigation before offline page (#5209) Integrated into release/v3.8.39 — PWA service worker prefers cached navigation before offline page (#5165). * chore(release): v3.8.39 — 2026-06-28 * chore(release): rebaseline openapi+i18n coverage ratchet drift for v3.8.39 --------- Co-authored-by: Arthur Bodera <abodera@gmail.com> Co-authored-by: Nguyen Minh <lop123thcs@gmail.com> Co-authored-by: lunkerchen <labanchen@gmail.com> Co-authored-by: Ankit <177378174+anki1kr@users.noreply.github.com> Co-authored-by: KooshaPari <KooshaPari@users.noreply.github.com> Co-authored-by: Ardem2025 <ardemb22@gmail.com> Co-authored-by: backryun <bakryun0718@proton.me> Co-authored-by: Anton <39598727+NomenAK@users.noreply.github.com> Co-authored-by: KooshaPari <42529354+KooshaPari@users.noreply.github.com> Co-authored-by: Wilson <pedbookmed@gmail.com> Co-authored-by: Randi <55005611+rdself@users.noreply.github.com>
128 lines
4.7 KiB
TypeScript
128 lines
4.7 KiB
TypeScript
/**
|
|
* #3089 — Convert a complete OpenAI-style chat-completion JSON body into an
|
|
* equivalent OpenAI SSE (`chat.completion.chunk`) stream.
|
|
*
|
|
* Some "reasoning" openai-compatible upstreams ignore a `stream: true` request
|
|
* and reply with a single `application/json` chat-completion body instead of an
|
|
* SSE stream. OmniRoute's streaming readiness check only recognizes SSE `data:`
|
|
* frames, so such a body produced a spurious `STREAM_EARLY_EOF` / HTTP 502 even
|
|
* though it carried valid `content` / `reasoning_content`. Synthesizing an SSE
|
|
* stream from that JSON lets the normal streaming pipeline (and the client) get
|
|
* a valid stream that preserves both `content` and `reasoning_content`.
|
|
*
|
|
* Returns "" when the text is not a parseable chat-completion object with at
|
|
* least one choice — callers then fall back to the original (error) handling.
|
|
*/
|
|
import { normalizeOpenAICompatibleFinishReasonString } from "./finishReason.ts";
|
|
import { getUnsupportedReasoningValue } from "./reasoningFields.ts";
|
|
|
|
type JsonRecord = Record<string, unknown>;
|
|
|
|
function isRecord(value: unknown): value is JsonRecord {
|
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
}
|
|
|
|
function nonEmptyString(value: unknown): string {
|
|
return typeof value === "string" && value.length > 0 ? value : "";
|
|
}
|
|
|
|
function addReadableReasoning(message: JsonRecord, delta: JsonRecord): boolean {
|
|
const reasoningContent = nonEmptyString(message.reasoning_content);
|
|
if (reasoningContent) {
|
|
delta.reasoning_content = reasoningContent;
|
|
return true;
|
|
}
|
|
|
|
const reasoning = nonEmptyString(message.reasoning);
|
|
if (reasoning) {
|
|
delta.reasoning = reasoning;
|
|
return true;
|
|
}
|
|
|
|
return false;
|
|
}
|
|
|
|
function addUnsupportedReasoning(message: JsonRecord, delta: JsonRecord) {
|
|
const reasoningContent = getUnsupportedReasoningValue(message);
|
|
if (reasoningContent) {
|
|
delta.reasoning_content = reasoningContent;
|
|
}
|
|
}
|
|
|
|
function buildReasoningDelta(message: JsonRecord): JsonRecord | null {
|
|
const delta: JsonRecord = {};
|
|
if (Array.isArray(message.reasoning_details)) {
|
|
delta.reasoning_details = message.reasoning_details;
|
|
}
|
|
|
|
if (!addReadableReasoning(message, delta)) {
|
|
addUnsupportedReasoning(message, delta);
|
|
}
|
|
|
|
return Object.keys(delta).length > 0 ? delta : null;
|
|
}
|
|
|
|
function sseEvent(payload: JsonRecord): string {
|
|
return `data: ${JSON.stringify(payload)}\n\n`;
|
|
}
|
|
|
|
export function synthesizeOpenAiSseFromJson(jsonText: string): string {
|
|
let parsed: unknown;
|
|
try {
|
|
parsed = JSON.parse(jsonText);
|
|
} catch {
|
|
return "";
|
|
}
|
|
if (!isRecord(parsed)) return "";
|
|
|
|
const choices = parsed.choices;
|
|
if (!Array.isArray(choices) || choices.length === 0) return "";
|
|
|
|
const id = typeof parsed.id === "string" && parsed.id ? parsed.id : "chatcmpl-omniroute-sse";
|
|
const created = typeof parsed.created === "number" ? parsed.created : 0;
|
|
const model = typeof parsed.model === "string" ? parsed.model : "";
|
|
const base = { id, object: "chat.completion.chunk", created, model };
|
|
|
|
let out = "";
|
|
let emittedAny = false;
|
|
|
|
choices.forEach((choice, fallbackIndex) => {
|
|
if (!isRecord(choice)) return;
|
|
const index = typeof choice.index === "number" ? choice.index : fallbackIndex;
|
|
const message = isRecord(choice.message) ? choice.message : {};
|
|
|
|
// Emit role, reasoning_content, content and tool_calls as SEPARATE sequential
|
|
// deltas — the same shape a real reasoning model streams (reasoning first,
|
|
// then content). Combining them in one delta caused the openai→openai
|
|
// translator to re-split and DUPLICATE reasoning_content across chunks
|
|
// (#3089 follow-up); separate deltas pass through cleanly with no duplication.
|
|
const role = typeof message.role === "string" ? message.role : "assistant";
|
|
const emitDelta = (delta: JsonRecord) => {
|
|
out += sseEvent({ ...base, choices: [{ index, delta, finish_reason: null }] });
|
|
};
|
|
|
|
emitDelta({ role });
|
|
const reasoningDelta = buildReasoningDelta(message);
|
|
if (reasoningDelta) {
|
|
emitDelta(reasoningDelta);
|
|
}
|
|
if (typeof message.content === "string" && message.content.length > 0) {
|
|
emitDelta({ content: message.content });
|
|
}
|
|
if (Array.isArray(message.tool_calls) && message.tool_calls.length > 0) {
|
|
emitDelta({ tool_calls: message.tool_calls });
|
|
}
|
|
|
|
const finishReason = normalizeOpenAICompatibleFinishReasonString(choice.finish_reason);
|
|
const finalChoice: JsonRecord = { index, delta: {}, finish_reason: finishReason };
|
|
const finalChunk: JsonRecord = { ...base, choices: [finalChoice] };
|
|
if (isRecord(parsed.usage)) finalChunk.usage = parsed.usage;
|
|
out += sseEvent(finalChunk);
|
|
emittedAny = true;
|
|
});
|
|
|
|
if (!emittedAny) return "";
|
|
out += "data: [DONE]\n\n";
|
|
return out;
|
|
}
|