mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-09-16 03:42:21 +03:00
* test(infra): retry recursive temp-dir removal on main (main twin of #11968)
`main` has been red since b342c1a361 on the vitest and integration gates:
✖ tests/unit/autoCombo/provider-family-combos.test.ts > auto/<family>
✖ chat pipeline applies Codex OAuth fingerprint and priority tier inside combos
Both call resetStorage() from beforeEach, which does an fs.rmSync(TEST_DATA_DIR,
{recursive: true, force: true}) with no retry, and intermittently loses the race
with a not-yet-released SQLite handle (ENOTEMPTY).
release/v3.8.51 fixed this in #11968 with a mechanical codemod adding
maxRetries/retryDelay to every recursive rm/rmSync/rmdirSync under tests/, but
that PR landed only on the release branch. Because main only receives work at
the release squash, it stayed broken for the whole cycle — and repo-wide gates
then turn every open PR into main red on checks unrelated to their diff.
This is the --base main twin: re-runs the same codemod that already shipped on
the release branch (scripts/ad-hoc/codemod-rm-maxretries.mjs), so the two
branches converge on identical test-teardown semantics. Test-only; no product
logic is touched.
The remaining three failures reported on #12133 (unit full suite exceeding its
4800s ceiling, package-artifact exceeding 1200s, and the boot-smoke that is
skipped as a consequence) are runner-contention timeouts, not code defects —
validate-release-green.mjs runs those heavy gates concurrently on one shared
hosted runner. There is no fix to port for those.
* chore(scripts): carry the rm-maxretries codemod onto main alongside its output
The codemod that generated the previous commit lives in the repo on
release/v3.8.51 (added by #11968) but was never on main. Bringing it over keeps
the tool next to the change it produced, so the transformation stays
reproducible and auditable from either branch.
130 lines
4.7 KiB
TypeScript
130 lines
4.7 KiB
TypeScript
/**
|
|
* #4165 — classify Bottleneck's execution expiration accurately.
|
|
*
|
|
* OmniRoute passes the legacy `requestQueue.maxWaitMs` value to Bottleneck as
|
|
* the job `expiration`. Bottleneck starts that timer only after a job leaves
|
|
* QUEUED, so it bounds limiter-managed execution and does not bound queue wait.
|
|
*
|
|
* The raw Bottleneck message (`This job timed out after <N> ms.`) still needs an
|
|
* OmniRoute-owned code and message so it cannot masquerade as an upstream-
|
|
* generated timeout. The original error remains available as `.cause`.
|
|
*/
|
|
import test from "node:test";
|
|
import assert from "node:assert/strict";
|
|
import fs from "node:fs";
|
|
import os from "node:os";
|
|
import path from "node:path";
|
|
|
|
const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-rl-execution-timeout-"));
|
|
process.env.DATA_DIR = TEST_DATA_DIR;
|
|
|
|
// Dynamic imports are required because DATA_DIR must be set before DB modules evaluate.
|
|
const core = await import("../../src/lib/db/core.ts");
|
|
const resilienceSettings = await import("../../src/lib/resilience/settings.ts");
|
|
const rateLimitManager = await import("../../open-sse/services/rateLimitManager.ts");
|
|
const { getClientSafeLocalRateLimitError, getTrustedLocalRateLimitError } =
|
|
await import("../../open-sse/services/rateLimitManager/errors.ts");
|
|
const { formatProviderError } = await import("../../open-sse/utils/error.ts");
|
|
|
|
// This contract test deliberately drives Bottleneck's real expiration timer.
|
|
function wait(ms: number) {
|
|
const { promise, resolve } = Promise.withResolvers<void>();
|
|
setTimeout(resolve, ms);
|
|
return promise;
|
|
}
|
|
|
|
test.afterEach(async () => {
|
|
await rateLimitManager.__resetRateLimitManagerForTests();
|
|
});
|
|
|
|
test.after(() => {
|
|
core.resetDbInstance();
|
|
fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 });
|
|
});
|
|
|
|
// Drive a real Bottleneck execution expiration with a function that outlives it.
|
|
async function triggerExecutionExpiration() {
|
|
await rateLimitManager.applyRequestQueueSettings({
|
|
...resilienceSettings.DEFAULT_RESILIENCE_SETTINGS.requestQueue,
|
|
autoEnableApiKeyProviders: false,
|
|
concurrentRequests: 1,
|
|
requestsPerMinute: 100000,
|
|
minTimeBetweenRequestsMs: 0,
|
|
maxWaitMs: 40,
|
|
});
|
|
rateLimitManager.enableRateLimitProtection("conn-execution-timeout");
|
|
|
|
return rateLimitManager.withRateLimit("openai", "conn-execution-timeout", "gpt-4o", async () => {
|
|
await wait(400); // > maxWaitMs (40ms) → Bottleneck fails the job
|
|
return "should-not-reach";
|
|
});
|
|
}
|
|
|
|
test("#4165 execution expiration is local and accurately named", async () => {
|
|
let caught: (Error & { code?: string; cause?: { message?: string } }) | undefined;
|
|
try {
|
|
await triggerExecutionExpiration();
|
|
assert.fail("expected the limiter-managed execution to expire");
|
|
} catch (err) {
|
|
caught = err as Error & { code?: string; cause?: { message?: string } };
|
|
}
|
|
assert.ok(caught, "an error should have been thrown");
|
|
|
|
assert.equal(
|
|
caught.code,
|
|
"RATE_LIMIT_EXECUTION_TIMEOUT",
|
|
"error must carry the local execution-expiration code"
|
|
);
|
|
|
|
assert.match(caught.message, /execution expiration/i);
|
|
assert.match(caught.message, /does not bound queue wait/i);
|
|
assert.match(
|
|
caught.message,
|
|
/not an upstream-generated timeout/i,
|
|
"message should explicitly disclaim an upstream-generated timeout"
|
|
);
|
|
assert.doesNotMatch(
|
|
caught.message,
|
|
/This job timed out/,
|
|
"raw Bottleneck/upstream-looking string must not leak into the surfaced message"
|
|
);
|
|
|
|
// The original Bottleneck error is preserved for debugging.
|
|
assert.ok(caught.cause, "original error should be preserved as cause");
|
|
assert.match(String(caught.cause?.message ?? ""), /This job timed out/);
|
|
|
|
assert.deepEqual(getTrustedLocalRateLimitError(caught), {
|
|
code: "RATE_LIMIT_EXECUTION_TIMEOUT",
|
|
status: 504,
|
|
});
|
|
const safeError = getClientSafeLocalRateLimitError(caught);
|
|
assert.ok(safeError);
|
|
const clientMessage = formatProviderError(safeError, "openai", "gpt-4o", 504);
|
|
assert.match(clientMessage, /execution expiration/i);
|
|
assert.doesNotMatch(
|
|
clientMessage,
|
|
/This job timed out/,
|
|
"client and call-log formatting must not append the retained Bottleneck cause"
|
|
);
|
|
});
|
|
|
|
test("#4165 a job that completes within the execution expiration is unaffected", async () => {
|
|
await rateLimitManager.applyRequestQueueSettings({
|
|
...resilienceSettings.DEFAULT_RESILIENCE_SETTINGS.requestQueue,
|
|
autoEnableApiKeyProviders: false,
|
|
concurrentRequests: 1,
|
|
requestsPerMinute: 100000,
|
|
minTimeBetweenRequestsMs: 0,
|
|
maxWaitMs: 5000,
|
|
});
|
|
rateLimitManager.enableRateLimitProtection("conn-fast");
|
|
|
|
const result = await rateLimitManager.withRateLimit(
|
|
"openai",
|
|
"conn-fast",
|
|
"gpt-4o",
|
|
async () => "ok"
|
|
);
|
|
assert.equal(result, "ok");
|
|
});
|