mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-07-26 09:52:11 +03:00
fix: remove implicit API key request caps (#2289)
Conflicts resolved and integrated into release/v3.8.0. Thanks again!
This commit is contained in:
110
CHANGELOG.md
110
CHANGELOG.md
@@ -296,62 +296,62 @@
|
||||
|
||||
Thank you to all **55+ community contributors** who made v3.8.0 possible! 🎉
|
||||
|
||||
| Contributor | PRs | Contributions |
|
||||
| :--------------------------------------------------------- | :-: | :--------------------------------------------------------------------------------- |
|
||||
| [@NomenAK](https://github.com/NomenAK) | 12 | #2217, #2218, #2219, #2221, #2222, #2223, #2224, #2228, #2233, #2234, #2242, #2192 |
|
||||
| Contributor | PRs | Contributions |
|
||||
| :--------------------------------------------------------- | :-: | :----------------------------------------------------------------------------------------------- |
|
||||
| [@NomenAK](https://github.com/NomenAK) | 12 | #2217, #2218, #2219, #2221, #2222, #2223, #2224, #2228, #2233, #2234, #2242, #2192 |
|
||||
| [@oyi77](https://github.com/oyi77) | 14 | #2010, #2014, #2041, #2052, #2061, #2074, #2091, #2094, #2096, #2131, #2135, #2240, #2283, #2295 |
|
||||
| [@backryun](https://github.com/backryun) | 9 | #1992, #2033, #2088, #2123, #2138, #2141, #2150, #2177, #2279 |
|
||||
| [@Brkic-Nikola](https://github.com/Brkic-Nikola) | 6 | #2165, #2189, #2190, #2191, #2192, #2197 |
|
||||
| [@Gioxaa](https://github.com/Gioxaa) | 5 | #2105, #2149, #2153, #2154, #2159 |
|
||||
| [@dhaern](https://github.com/dhaern) | 4 | #2028, #2039, #2087, #2090 |
|
||||
| [@andrewmunsell](https://github.com/andrewmunsell) | 3 | #2169, #2176, #2238 |
|
||||
| [@ddarkr](https://github.com/ddarkr) | 4 | #2047, #2199, #2243, #2271 |
|
||||
| [@nickwizard](https://github.com/nickwizard) | 3 | #1991, #2196, #2227 |
|
||||
| [@herjarsa](https://github.com/herjarsa) | 3 | #2030, #2136, #2152 |
|
||||
| [@rafacpti23](https://github.com/rafacpti23) | 3 | #2086, #2146, #2201 |
|
||||
| [@Tentoxa](https://github.com/Tentoxa) | 2 | #2011, #2053 |
|
||||
| [@wauputr4](https://github.com/wauputr4) | 2 | #2009, #2046 |
|
||||
| [@hartmark](https://github.com/hartmark) | 4 | #2045, #2137, #2294, #2299 |
|
||||
| [@payne0420](https://github.com/payne0420) | 2 | #2082, #2128 |
|
||||
| [@bypanghu](https://github.com/bypanghu) | 2 | #2027, #2156 |
|
||||
| [@eleata](https://github.com/eleata) | 2 | #2116, #2133 |
|
||||
| [@Tr0sT](https://github.com/Tr0sT) | 1 | #2012 |
|
||||
| [@AveryanAlex](https://github.com/AveryanAlex) | 1 | #2008 |
|
||||
| [@rodrigogbbr-stack](https://github.com/rodrigogbbr-stack) | 1 | #1996 |
|
||||
| [@NekoMonci12](https://github.com/NekoMonci12) | 1 | #1999 |
|
||||
| [@congvc-dev](https://github.com/congvc-dev) | 1 | #2004 |
|
||||
| [@tatsster](https://github.com/tatsster) | 1 | #2007 |
|
||||
| [@xssdem](https://github.com/xssdem) | 1 | #2023 |
|
||||
| [@wucm667](https://github.com/wucm667) | 1 | #2031 |
|
||||
| [@tces1](https://github.com/tces1) | 1 | #2048 |
|
||||
| [@guanbear](https://github.com/guanbear) | 1 | #2054 |
|
||||
| [@Gi99lin](https://github.com/Gi99lin) | 1 | #2055 |
|
||||
| [@ivan-mezentsev](https://github.com/ivan-mezentsev) | 1 | #2063 |
|
||||
| [@JxnLexn](https://github.com/JxnLexn) | 1 | #2019 |
|
||||
| [@yoviarpauzi](https://github.com/yoviarpauzi) | 1 | #2092 |
|
||||
| [@gleber](https://github.com/gleber) | 1 | #2103 |
|
||||
| [@rilham97](https://github.com/rilham97) | 1 | #2104 |
|
||||
| [@boa-z](https://github.com/boa-z) | 1 | #2115 |
|
||||
| [@rdself](https://github.com/rdself) | 1 | #2118 |
|
||||
| [@clousky2020](https://github.com/clousky2020) | 1 | #2119 |
|
||||
| [@abhinavjnu](https://github.com/abhinavjnu) | 1 | #2122 |
|
||||
| [@HoaPham98](https://github.com/HoaPham98) | 1 | #2089 |
|
||||
| [@christlau](https://github.com/christlau) | 1 | #2129 |
|
||||
| [@flyingmongoose](https://github.com/flyingmongoose) | 1 | #2134 |
|
||||
| [@05dunski](https://github.com/05dunski) | 1 | #1978 (cherry-picked) |
|
||||
| [@DavyMassoneto](https://github.com/DavyMassoneto) | 1 | #2140 |
|
||||
| [@Zhaba1337228](https://github.com/Zhaba1337228) | 1 | #2168 |
|
||||
| [@faisalill](https://github.com/faisalill) | 1 | #2166 |
|
||||
| [@Yosee11](https://github.com/Yosee11) | 1 | #2164 |
|
||||
| [@hachimed](https://github.com/hachimed) | 1 | #2162 |
|
||||
| [@JohnDoe-oss](https://github.com/JohnDoe-oss) | 1 | #2161 |
|
||||
| [@brucevoin](https://github.com/brucevoin) | 1 | #2163 |
|
||||
| [@InkshadeWoods](https://github.com/InkshadeWoods) | 1 | #2202 |
|
||||
| [@kang-heewon](https://github.com/kang-heewon) | 1 | #2231 |
|
||||
| [@one-vs](https://github.com/one-vs) | 1 | #2236 |
|
||||
| [@thepigdestroyer](https://github.com/thepigdestroyer) | 2 | #2290, #2291 |
|
||||
| [@josephvoxone](https://github.com/josephvoxone) | 1 | #2289 |
|
||||
| [@mrmm](https://github.com/mrmm) | 2 | #2286, #2305 |
|
||||
| [@backryun](https://github.com/backryun) | 9 | #1992, #2033, #2088, #2123, #2138, #2141, #2150, #2177, #2279 |
|
||||
| [@Brkic-Nikola](https://github.com/Brkic-Nikola) | 6 | #2165, #2189, #2190, #2191, #2192, #2197 |
|
||||
| [@Gioxaa](https://github.com/Gioxaa) | 5 | #2105, #2149, #2153, #2154, #2159 |
|
||||
| [@dhaern](https://github.com/dhaern) | 4 | #2028, #2039, #2087, #2090 |
|
||||
| [@andrewmunsell](https://github.com/andrewmunsell) | 3 | #2169, #2176, #2238 |
|
||||
| [@ddarkr](https://github.com/ddarkr) | 4 | #2047, #2199, #2243, #2271 |
|
||||
| [@nickwizard](https://github.com/nickwizard) | 3 | #1991, #2196, #2227 |
|
||||
| [@herjarsa](https://github.com/herjarsa) | 3 | #2030, #2136, #2152 |
|
||||
| [@rafacpti23](https://github.com/rafacpti23) | 3 | #2086, #2146, #2201 |
|
||||
| [@Tentoxa](https://github.com/Tentoxa) | 2 | #2011, #2053 |
|
||||
| [@wauputr4](https://github.com/wauputr4) | 2 | #2009, #2046 |
|
||||
| [@hartmark](https://github.com/hartmark) | 4 | #2045, #2137, #2294, #2299 |
|
||||
| [@payne0420](https://github.com/payne0420) | 2 | #2082, #2128 |
|
||||
| [@bypanghu](https://github.com/bypanghu) | 2 | #2027, #2156 |
|
||||
| [@eleata](https://github.com/eleata) | 2 | #2116, #2133 |
|
||||
| [@Tr0sT](https://github.com/Tr0sT) | 1 | #2012 |
|
||||
| [@AveryanAlex](https://github.com/AveryanAlex) | 1 | #2008 |
|
||||
| [@rodrigogbbr-stack](https://github.com/rodrigogbbr-stack) | 1 | #1996 |
|
||||
| [@NekoMonci12](https://github.com/NekoMonci12) | 1 | #1999 |
|
||||
| [@congvc-dev](https://github.com/congvc-dev) | 1 | #2004 |
|
||||
| [@tatsster](https://github.com/tatsster) | 1 | #2007 |
|
||||
| [@xssdem](https://github.com/xssdem) | 1 | #2023 |
|
||||
| [@wucm667](https://github.com/wucm667) | 1 | #2031 |
|
||||
| [@tces1](https://github.com/tces1) | 1 | #2048 |
|
||||
| [@guanbear](https://github.com/guanbear) | 1 | #2054 |
|
||||
| [@Gi99lin](https://github.com/Gi99lin) | 1 | #2055 |
|
||||
| [@ivan-mezentsev](https://github.com/ivan-mezentsev) | 1 | #2063 |
|
||||
| [@JxnLexn](https://github.com/JxnLexn) | 1 | #2019 |
|
||||
| [@yoviarpauzi](https://github.com/yoviarpauzi) | 1 | #2092 |
|
||||
| [@gleber](https://github.com/gleber) | 1 | #2103 |
|
||||
| [@rilham97](https://github.com/rilham97) | 1 | #2104 |
|
||||
| [@boa-z](https://github.com/boa-z) | 1 | #2115 |
|
||||
| [@rdself](https://github.com/rdself) | 1 | #2118 |
|
||||
| [@clousky2020](https://github.com/clousky2020) | 1 | #2119 |
|
||||
| [@abhinavjnu](https://github.com/abhinavjnu) | 1 | #2122 |
|
||||
| [@HoaPham98](https://github.com/HoaPham98) | 1 | #2089 |
|
||||
| [@christlau](https://github.com/christlau) | 1 | #2129 |
|
||||
| [@flyingmongoose](https://github.com/flyingmongoose) | 1 | #2134 |
|
||||
| [@05dunski](https://github.com/05dunski) | 1 | #1978 (cherry-picked) |
|
||||
| [@DavyMassoneto](https://github.com/DavyMassoneto) | 1 | #2140 |
|
||||
| [@Zhaba1337228](https://github.com/Zhaba1337228) | 1 | #2168 |
|
||||
| [@faisalill](https://github.com/faisalill) | 1 | #2166 |
|
||||
| [@Yosee11](https://github.com/Yosee11) | 1 | #2164 |
|
||||
| [@hachimed](https://github.com/hachimed) | 1 | #2162 |
|
||||
| [@JohnDoe-oss](https://github.com/JohnDoe-oss) | 1 | #2161 |
|
||||
| [@brucevoin](https://github.com/brucevoin) | 1 | #2163 |
|
||||
| [@InkshadeWoods](https://github.com/InkshadeWoods) | 1 | #2202 |
|
||||
| [@kang-heewon](https://github.com/kang-heewon) | 1 | #2231 |
|
||||
| [@one-vs](https://github.com/one-vs) | 1 | #2236 |
|
||||
| [@thepigdestroyer](https://github.com/thepigdestroyer) | 2 | #2290, #2291 |
|
||||
| [@josephvoxone](https://github.com/josephvoxone) | 1 | #2289 |
|
||||
| [@mrmm](https://github.com/mrmm) | 2 | #2286, #2305 |
|
||||
|
||||
## [3.7.9] — 2026-05-03
|
||||
|
||||
|
||||
91
docs/specs/2026-05-16-adaptive-stream-readiness-design.md
Normal file
91
docs/specs/2026-05-16-adaptive-stream-readiness-design.md
Normal file
@@ -0,0 +1,91 @@
|
||||
# Adaptive Stream Readiness Timeout
|
||||
|
||||
## Problem
|
||||
|
||||
Long Codex continue sessions can produce very large Responses API payloads (hundreds of messages, 20 tools, and large cached input). OmniRoute currently uses a fixed `STREAM_READINESS_TIMEOUT_MS` default of 30 seconds for the first useful SSE event. That fixed threshold is too short for some large Codex requests, even when the upstream later completes successfully.
|
||||
|
||||
Recent local evidence showed:
|
||||
|
||||
- Small/medium Codex requests confirm readiness in roughly 0.8-2.5 seconds.
|
||||
- Large Codex requests can run 60+ seconds and still complete successfully.
|
||||
- A fixed 30 second readiness timeout can therefore create false failures: `Stream produced no useful content within 30000ms`.
|
||||
|
||||
The solution should avoid a blanket manual timeout increase, because that would slow fallback for genuinely dead streams.
|
||||
|
||||
## Goals
|
||||
|
||||
- Keep small requests fast to fail when the upstream stream is dead.
|
||||
- Give large/tool-heavy Codex Responses requests more time to produce first useful content.
|
||||
- Make timeout decisions visible in logs for future debugging.
|
||||
- Preserve the existing default behavior unless the request shape justifies extra budget.
|
||||
- Keep a hard upper bound so zombie streams cannot hang indefinitely.
|
||||
|
||||
## Non-Goals
|
||||
|
||||
- Do not change provider fallback ordering in this spec.
|
||||
- Do not alter account health/rate-limit policy.
|
||||
- Do not change post-readiness stream idle behavior.
|
||||
- Do not implement compression or summarization in this change.
|
||||
|
||||
## Design
|
||||
|
||||
Add a small policy helper that computes the readiness timeout from request shape:
|
||||
|
||||
`open-sse/utils/streamReadinessPolicy.ts`
|
||||
|
||||
The helper accepts:
|
||||
|
||||
- `baseTimeoutMs`
|
||||
- `provider`
|
||||
- `model`
|
||||
- `body`
|
||||
|
||||
It returns:
|
||||
|
||||
- `timeoutMs`
|
||||
- `reasons`
|
||||
|
||||
The initial heuristic is intentionally conservative:
|
||||
|
||||
- Start with the configured base timeout, usually 30 seconds.
|
||||
- Add budget for large input arrays or message arrays.
|
||||
- Add budget for tool-heavy requests.
|
||||
- Add budget for Codex GPT-5.5 Responses requests, because local evidence shows these can take longer on large sessions.
|
||||
- Cap the result at 120 seconds by default.
|
||||
|
||||
This is adaptive, not purely provider-based: Codex only receives the extra budget when the payload is large/tool-heavy enough to justify it.
|
||||
|
||||
## Integration
|
||||
|
||||
In `open-sse/handlers/chatCore.ts`, replace the direct use of `STREAM_READINESS_TIMEOUT_MS` in `ensureStreamReadiness` with the policy result.
|
||||
|
||||
Log the chosen timeout and reasons when it differs from the base timeout, for example:
|
||||
|
||||
```text
|
||||
[sse] stream readiness timeout=90000ms base=30000ms reason=codex,gpt-5.5,large_input,tool_heavy
|
||||
```
|
||||
|
||||
## Failure Behavior
|
||||
|
||||
If no useful stream content appears before the adaptive timeout, OmniRoute should continue using the existing failure path and return `STREAM_READINESS_TIMEOUT`. This change only changes the budget, not the fallback/error semantics.
|
||||
|
||||
## Testing
|
||||
|
||||
Add unit tests for the policy helper:
|
||||
|
||||
- Small request keeps the base timeout.
|
||||
- Large message array increases timeout.
|
||||
- Tool-heavy request increases timeout.
|
||||
- Codex GPT-5.5 large request receives a larger timeout.
|
||||
- Timeout is capped at the maximum.
|
||||
- Zero/disabled base timeout remains zero so readiness checks can still be disabled by config.
|
||||
|
||||
Add or update a handler-level test only if needed after unit coverage.
|
||||
|
||||
## Acceptance Criteria
|
||||
|
||||
- Large Codex continue sessions get an adaptive readiness timeout above 30 seconds.
|
||||
- Small requests still use 30 seconds by default.
|
||||
- The max adaptive timeout cannot exceed 120 seconds unless explicitly changed in code later.
|
||||
- Unit tests cover the policy and pass.
|
||||
- Existing pre-commit checks pass.
|
||||
@@ -316,11 +316,22 @@ export function orderHeaders(
|
||||
* Apply a CLI fingerprint to headers and body.
|
||||
* Returns { headers, bodyString } with the correct ordering.
|
||||
*/
|
||||
function stripInternalBodyFields(body: unknown): unknown {
|
||||
if (!body || typeof body !== "object" || Array.isArray(body)) return body;
|
||||
|
||||
const record = body as Record<string, unknown>;
|
||||
delete record._claudeCodeRequiresLowercaseToolNames;
|
||||
delete record._nativeCodexPassthrough;
|
||||
delete record._omnirouteResponsesStore;
|
||||
return body;
|
||||
}
|
||||
|
||||
export function applyFingerprint(
|
||||
provider: string,
|
||||
headers: Record<string, string>,
|
||||
body: unknown
|
||||
): { headers: Record<string, string>; bodyString: string } {
|
||||
body = stripInternalBodyFields(body);
|
||||
const normalizedProvider = normalizeCliCompatProviderId(provider || "");
|
||||
const fingerprintKey = isClaudeCodeCompatible(provider)
|
||||
? "claude-code-compatible"
|
||||
|
||||
@@ -398,7 +398,10 @@ function stripStoredItemReferences(body: Record<string, unknown>): void {
|
||||
.map((functionCall) => ({
|
||||
type: "function_call",
|
||||
call_id: functionCall.call_id,
|
||||
name: functionCall.name,
|
||||
name:
|
||||
typeof functionCall.name === "string"
|
||||
? functionCall.name.slice(0, 128)
|
||||
: functionCall.name,
|
||||
arguments: functionCall.arguments,
|
||||
}));
|
||||
|
||||
@@ -579,7 +582,7 @@ function normalizeCodexTools(body: Record<string, unknown>): void {
|
||||
for (const st of tool.tools as unknown[]) {
|
||||
if (st && typeof st === "object" && !Array.isArray(st)) {
|
||||
const subTool = st as Record<string, unknown>;
|
||||
const name = typeof subTool.name === "string" ? subTool.name.trim() : "";
|
||||
const name = typeof subTool.name === "string" ? subTool.name.trim().slice(0, 128) : "";
|
||||
if (name) validToolNames.add(name);
|
||||
}
|
||||
}
|
||||
@@ -644,7 +647,7 @@ function normalizeCodexTools(body: Record<string, unknown>): void {
|
||||
delete tool[key];
|
||||
}
|
||||
tool.type = "function";
|
||||
tool.name = name;
|
||||
tool.name = name.slice(0, 128);
|
||||
if (description) tool.description = description;
|
||||
tool.parameters = parameters;
|
||||
|
||||
|
||||
@@ -10,6 +10,7 @@ import {
|
||||
withBodyTimeout,
|
||||
} from "../utils/stream.ts";
|
||||
import { ensureStreamReadiness } from "../utils/streamReadiness.ts";
|
||||
import { resolveStreamReadinessTimeout } from "../utils/streamReadinessPolicy.ts";
|
||||
import { createStreamController, pipeWithDisconnect } from "../utils/streamHandler.ts";
|
||||
import { createSseHeartbeatTransform, shapeForClientFormat } from "../utils/sseHeartbeat.ts";
|
||||
import { addBufferToUsage, filterUsageForFormat, estimateUsage } from "../utils/usageTracking.ts";
|
||||
@@ -4377,8 +4378,21 @@ export async function handleChatCore({
|
||||
}
|
||||
|
||||
// Streaming response
|
||||
const streamReadinessPolicy = resolveStreamReadinessTimeout({
|
||||
baseTimeoutMs: STREAM_READINESS_TIMEOUT_MS,
|
||||
provider,
|
||||
model,
|
||||
body: (finalBody || translatedBody) as Record<string, unknown> | null | undefined,
|
||||
});
|
||||
if (streamReadinessPolicy.timeoutMs !== streamReadinessPolicy.baseTimeoutMs) {
|
||||
log?.debug?.(
|
||||
"STREAM",
|
||||
`adaptive readiness timeout=${streamReadinessPolicy.timeoutMs}ms base=${streamReadinessPolicy.baseTimeoutMs}ms reason=${streamReadinessPolicy.reasons.join(",")}`
|
||||
);
|
||||
}
|
||||
|
||||
const streamReadiness = await ensureStreamReadiness(providerResponse, {
|
||||
timeoutMs: STREAM_READINESS_TIMEOUT_MS,
|
||||
timeoutMs: streamReadinessPolicy.timeoutMs,
|
||||
provider,
|
||||
model,
|
||||
log,
|
||||
|
||||
@@ -21,6 +21,28 @@ export function isInternalAssistantMessage(record: JsonRecord): boolean {
|
||||
return INTERNAL_ASSISTANT_PHASES.has(phase);
|
||||
}
|
||||
|
||||
// OpenAI Responses API enforces two constraints on name fields in input items:
|
||||
// 1. Max 128 characters
|
||||
// 2. Must match ^[a-zA-Z0-9_-]+$
|
||||
// Sanitize after cloning so upstream never sees an invalid name.
|
||||
function sanitizeFunctionName(name: string): string {
|
||||
// Replace any character not in [a-zA-Z0-9_-] with underscore, then truncate.
|
||||
return name.replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 128);
|
||||
}
|
||||
|
||||
function truncateInputItemName(item: unknown): unknown {
|
||||
const record = toRecord(item);
|
||||
if (!record) return item;
|
||||
if (
|
||||
(record.type === "function_call" || record.type === "function_call_output") &&
|
||||
typeof record.name === "string" &&
|
||||
!/^[a-zA-Z0-9_-]{1,128}$/.test(record.name)
|
||||
) {
|
||||
return { ...record, name: sanitizeFunctionName(record.name) };
|
||||
}
|
||||
return item;
|
||||
}
|
||||
|
||||
export function sanitizeResponsesInputItems(items: readonly unknown[], clone = true): unknown[] {
|
||||
const sanitized: unknown[] = [];
|
||||
|
||||
@@ -30,7 +52,8 @@ export function sanitizeResponsesInputItems(items: readonly unknown[], clone = t
|
||||
continue;
|
||||
}
|
||||
|
||||
sanitized.push(clone ? structuredClone(item) : item);
|
||||
const cloned = clone ? structuredClone(item) : item;
|
||||
sanitized.push(truncateInputItemName(cloned));
|
||||
}
|
||||
|
||||
return sanitized;
|
||||
|
||||
93
open-sse/utils/streamReadinessPolicy.ts
Normal file
93
open-sse/utils/streamReadinessPolicy.ts
Normal file
@@ -0,0 +1,93 @@
|
||||
type StreamReadinessBody = Record<string, unknown> | null | undefined;
|
||||
|
||||
export type StreamReadinessPolicyInput = {
|
||||
baseTimeoutMs: number;
|
||||
provider?: string | null;
|
||||
model?: string | null;
|
||||
body?: StreamReadinessBody;
|
||||
maxTimeoutMs?: number;
|
||||
};
|
||||
|
||||
export type StreamReadinessPolicyResult = {
|
||||
timeoutMs: number;
|
||||
baseTimeoutMs: number;
|
||||
reasons: string[];
|
||||
};
|
||||
|
||||
const DEFAULT_MAX_TIMEOUT_MS = 120_000;
|
||||
const LARGE_ITEM_THRESHOLD = 150;
|
||||
const VERY_LARGE_ITEM_THRESHOLD = 400;
|
||||
const TOOL_HEAVY_THRESHOLD = 15;
|
||||
const LARGE_CHAR_THRESHOLD = 250_000;
|
||||
const VERY_LARGE_CHAR_THRESHOLD = 750_000;
|
||||
|
||||
function countArrayField(body: StreamReadinessBody, field: "input" | "messages" | "tools"): number {
|
||||
const value = body?.[field];
|
||||
return Array.isArray(value) ? value.length : 0;
|
||||
}
|
||||
|
||||
function estimateBodyChars(body: StreamReadinessBody): number {
|
||||
if (!body) return 0;
|
||||
try {
|
||||
return JSON.stringify(body).length;
|
||||
} catch {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
function isCodexGpt55(provider?: string | null, model?: string | null): boolean {
|
||||
const normalizedProvider = (provider || "").toLowerCase();
|
||||
const normalizedModel = (model || "").toLowerCase();
|
||||
return normalizedProvider === "codex" && normalizedModel.includes("gpt-5.5");
|
||||
}
|
||||
|
||||
export function resolveStreamReadinessTimeout(
|
||||
input: StreamReadinessPolicyInput
|
||||
): StreamReadinessPolicyResult {
|
||||
const baseTimeoutMs = Math.max(0, Math.floor(input.baseTimeoutMs || 0));
|
||||
if (baseTimeoutMs <= 0) {
|
||||
return { timeoutMs: baseTimeoutMs, baseTimeoutMs, reasons: ["disabled"] };
|
||||
}
|
||||
|
||||
const maxTimeoutMs = Math.max(baseTimeoutMs, input.maxTimeoutMs ?? DEFAULT_MAX_TIMEOUT_MS);
|
||||
const reasons: string[] = [];
|
||||
let timeoutMs = baseTimeoutMs;
|
||||
|
||||
const inputCount = countArrayField(input.body, "input");
|
||||
const messageCount = countArrayField(input.body, "messages");
|
||||
const itemCount = Math.max(inputCount, messageCount);
|
||||
const toolCount = countArrayField(input.body, "tools");
|
||||
const estimatedChars = estimateBodyChars(input.body);
|
||||
const codexGpt55 = isCodexGpt55(input.provider, input.model);
|
||||
|
||||
if (itemCount > VERY_LARGE_ITEM_THRESHOLD) {
|
||||
timeoutMs += 45_000;
|
||||
reasons.push("very_large_history");
|
||||
} else if (itemCount > LARGE_ITEM_THRESHOLD) {
|
||||
timeoutMs += 20_000;
|
||||
reasons.push("large_history");
|
||||
}
|
||||
|
||||
if (toolCount >= TOOL_HEAVY_THRESHOLD) {
|
||||
timeoutMs += 15_000;
|
||||
reasons.push("tool_heavy");
|
||||
}
|
||||
|
||||
if (estimatedChars > VERY_LARGE_CHAR_THRESHOLD) {
|
||||
timeoutMs += 45_000;
|
||||
reasons.push("very_large_payload");
|
||||
} else if (estimatedChars > LARGE_CHAR_THRESHOLD) {
|
||||
timeoutMs += 20_000;
|
||||
reasons.push("large_payload");
|
||||
}
|
||||
|
||||
if (codexGpt55 && (itemCount > LARGE_ITEM_THRESHOLD || toolCount >= TOOL_HEAVY_THRESHOLD)) {
|
||||
timeoutMs += 30_000;
|
||||
reasons.push("codex_gpt_5_5_large_responses");
|
||||
}
|
||||
|
||||
timeoutMs = Math.min(timeoutMs, maxTimeoutMs);
|
||||
if (timeoutMs === baseTimeoutMs) reasons.push("base");
|
||||
|
||||
return { timeoutMs, baseTimeoutMs, reasons };
|
||||
}
|
||||
@@ -1,5 +1,5 @@
|
||||
import { createHmac } from "node:crypto";
|
||||
// eslint-disable-next-line @typescript-eslint/no-require-imports
|
||||
|
||||
let machineIdSync: (original?: boolean) => string;
|
||||
try {
|
||||
// Use require() to bypass webpack static analysis that breaks the default export
|
||||
|
||||
@@ -465,7 +465,9 @@ test("enforceApiKeyPolicy does not rate-limit unrestricted keys by default", asy
|
||||
const unrestrictedKey = await createKeyWithPolicy({ allowedModels: ["openai/*"] });
|
||||
const policy = await loadPolicy("default-no-request-limit");
|
||||
|
||||
for (let i = 0; i < 1005; i += 1) {
|
||||
// 5 calls is enough to prove no rate-limit fires; 1005 DB-backed iterations
|
||||
// were flagged as unnecessary overhead in code review.
|
||||
for (let i = 0; i < 5; i += 1) {
|
||||
const result = await policy.enforceApiKeyPolicy(
|
||||
makePolicyRequest(unrestrictedKey.key),
|
||||
"openai/gpt-4.1"
|
||||
|
||||
@@ -79,6 +79,23 @@ test("CLI fingerprint toggles only expose implemented fingerprints and functiona
|
||||
assert.equal((CLI_COMPAT_TOGGLE_IDS as readonly string[]).includes("github"), false);
|
||||
});
|
||||
|
||||
test("CLI fingerprint strips OmniRoute internal body fields before upstream serialization", () => {
|
||||
const claude = applyFingerprint(
|
||||
"claude",
|
||||
{ Authorization: "Bearer token" },
|
||||
{
|
||||
model: "claude-sonnet-4-6",
|
||||
messages: [],
|
||||
stream: true,
|
||||
_claudeCodeRequiresLowercaseToolNames: true,
|
||||
}
|
||||
);
|
||||
|
||||
const body = JSON.parse(claude.bodyString);
|
||||
assert.equal(body._claudeCodeRequiresLowercaseToolNames, undefined);
|
||||
assert.deepEqual(Object.keys(body), ["model", "messages", "stream"]);
|
||||
});
|
||||
|
||||
test("CLI fingerprint preserves Codex executor User-Agent and maps legacy Copilot alias", () => {
|
||||
const codex = applyFingerprint(
|
||||
"codex",
|
||||
|
||||
49
tests/unit/responses-input-sanitizer-name.test.ts
Normal file
49
tests/unit/responses-input-sanitizer-name.test.ts
Normal file
@@ -0,0 +1,49 @@
|
||||
import test from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import { sanitizeResponsesInputItems } from "../../open-sse/services/responsesInputSanitizer.ts";
|
||||
|
||||
test("truncates function_call name longer than 128 chars", () => {
|
||||
const longName = "a".repeat(156);
|
||||
const items = [{ type: "function_call", call_id: "c1", name: longName, arguments: "{}" }];
|
||||
const result = sanitizeResponsesInputItems(items) as Array<Record<string, unknown>>;
|
||||
assert.ok((result[0].name as string).length <= 128);
|
||||
});
|
||||
|
||||
test("strips illegal characters from function_call name leaving only [a-zA-Z0-9_-]", () => {
|
||||
const items = [
|
||||
{ type: "function_call", call_id: "c2", name: "mcp__ns__get.issue item", arguments: "{}" },
|
||||
];
|
||||
const result = sanitizeResponsesInputItems(items) as Array<Record<string, unknown>>;
|
||||
assert.match(result[0].name as string, /^[a-zA-Z0-9_-]+$/);
|
||||
});
|
||||
|
||||
test("strips illegal characters from function_call_output name", () => {
|
||||
const items = [
|
||||
{ type: "function_call_output", call_id: "c3", name: "tool.with.dots", output: "ok" },
|
||||
];
|
||||
const result = sanitizeResponsesInputItems(items) as Array<Record<string, unknown>>;
|
||||
assert.match(result[0].name as string, /^[a-zA-Z0-9_-]+$/);
|
||||
});
|
||||
|
||||
test("leaves a valid name unchanged", () => {
|
||||
const items = [
|
||||
{ type: "function_call", call_id: "c4", name: "valid_tool-name", arguments: "{}" },
|
||||
];
|
||||
const result = sanitizeResponsesInputItems(items) as Array<Record<string, unknown>>;
|
||||
assert.equal(result[0].name, "valid_tool-name");
|
||||
});
|
||||
|
||||
test("does not modify message items", () => {
|
||||
const items = [{ type: "message", role: "user", content: "hello" }];
|
||||
const result = sanitizeResponsesInputItems(items) as Array<Record<string, unknown>>;
|
||||
assert.deepEqual(result[0], { type: "message", role: "user", content: "hello" });
|
||||
});
|
||||
|
||||
test("handles name that is both too long and has illegal chars", () => {
|
||||
const badName = "mcp__ns__get.issue.".repeat(10); // 190 chars with dots
|
||||
const items = [{ type: "function_call", call_id: "c5", name: badName, arguments: "{}" }];
|
||||
const result = sanitizeResponsesInputItems(items) as Array<Record<string, unknown>>;
|
||||
const name = result[0].name as string;
|
||||
assert.ok(name.length <= 128);
|
||||
assert.match(name, /^[a-zA-Z0-9_-]+$/);
|
||||
});
|
||||
90
tests/unit/stream-readiness-policy.test.ts
Normal file
90
tests/unit/stream-readiness-policy.test.ts
Normal file
@@ -0,0 +1,90 @@
|
||||
import test from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import { resolveStreamReadinessTimeout } from "../../open-sse/utils/streamReadinessPolicy.ts";
|
||||
|
||||
function items(count: number): Array<{ role: string; content: string }> {
|
||||
return Array.from({ length: count }, (_, index) => ({
|
||||
role: "user",
|
||||
content: `message ${index}`,
|
||||
}));
|
||||
}
|
||||
|
||||
function tools(count: number): Array<{ type: string; name: string }> {
|
||||
return Array.from({ length: count }, (_, index) => ({ type: "function", name: `tool_${index}` }));
|
||||
}
|
||||
|
||||
test("keeps the base timeout for small requests", () => {
|
||||
const result = resolveStreamReadinessTimeout({
|
||||
baseTimeoutMs: 30_000,
|
||||
provider: "codex",
|
||||
model: "gpt-5.5",
|
||||
body: { input: items(3), tools: tools(2) },
|
||||
});
|
||||
|
||||
assert.equal(result.timeoutMs, 30_000);
|
||||
assert.deepEqual(result.reasons, ["base"]);
|
||||
});
|
||||
|
||||
test("increases timeout for large conversation history", () => {
|
||||
const result = resolveStreamReadinessTimeout({
|
||||
baseTimeoutMs: 30_000,
|
||||
provider: "openai",
|
||||
model: "gpt-4.1",
|
||||
body: { input: items(181) },
|
||||
});
|
||||
|
||||
assert.equal(result.timeoutMs, 50_000);
|
||||
assert.ok(result.reasons.includes("large_history"));
|
||||
});
|
||||
|
||||
test("increases timeout for tool-heavy requests", () => {
|
||||
const result = resolveStreamReadinessTimeout({
|
||||
baseTimeoutMs: 30_000,
|
||||
provider: "openai",
|
||||
model: "gpt-4.1",
|
||||
body: { input: items(10), tools: tools(20) },
|
||||
});
|
||||
|
||||
assert.equal(result.timeoutMs, 45_000);
|
||||
assert.ok(result.reasons.includes("tool_heavy"));
|
||||
});
|
||||
|
||||
test("gives Codex GPT-5.5 large Responses requests extra readiness budget", () => {
|
||||
const result = resolveStreamReadinessTimeout({
|
||||
baseTimeoutMs: 30_000,
|
||||
provider: "codex",
|
||||
model: "gpt-5.5",
|
||||
body: { input: items(181), tools: tools(20) },
|
||||
});
|
||||
|
||||
assert.equal(result.timeoutMs, 95_000);
|
||||
assert.ok(result.reasons.includes("large_history"));
|
||||
assert.ok(result.reasons.includes("tool_heavy"));
|
||||
assert.ok(result.reasons.includes("codex_gpt_5_5_large_responses"));
|
||||
});
|
||||
|
||||
test("caps adaptive timeout at maxTimeoutMs", () => {
|
||||
const result = resolveStreamReadinessTimeout({
|
||||
baseTimeoutMs: 30_000,
|
||||
maxTimeoutMs: 120_000,
|
||||
provider: "codex",
|
||||
model: "gpt-5.5",
|
||||
body: { input: items(500), tools: tools(20), instructions: "x".repeat(800_000) },
|
||||
});
|
||||
|
||||
assert.equal(result.timeoutMs, 120_000);
|
||||
assert.ok(result.reasons.includes("very_large_history"));
|
||||
assert.ok(result.reasons.includes("very_large_payload"));
|
||||
});
|
||||
|
||||
test("preserves zero timeout so readiness checks can be disabled", () => {
|
||||
const result = resolveStreamReadinessTimeout({
|
||||
baseTimeoutMs: 0,
|
||||
provider: "codex",
|
||||
model: "gpt-5.5",
|
||||
body: { input: items(500), tools: tools(20) },
|
||||
});
|
||||
|
||||
assert.equal(result.timeoutMs, 0);
|
||||
assert.deepEqual(result.reasons, ["disabled"]);
|
||||
});
|
||||
Reference in New Issue
Block a user