mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-08-01 21:02:12 +03:00
* fix(auth): restore configurable HEALTHCHECK_BATCH_SIZE dropped by #7719 (#7875) * docs(env): document HEALTHCHECK_BATCH_SIZE in .env.example (#7875) * docs(env): document HEALTHCHECK_BATCH_SIZE in .env.example + ENVIRONMENT.md (#7875)
This commit is contained in:
committed by
GitHub
parent
39ccfcf28c
commit
4eea1cd14f
@@ -1664,6 +1664,11 @@ APP_LOG_TO_FILE=true
|
||||
# HEALTHCHECK_JITTER_MIN_MS=500
|
||||
# HEALTHCHECK_JITTER_MAX_MS=5000
|
||||
|
||||
# Concurrent-check batch size for the startup token-healthcheck sweep. Larger
|
||||
# values check more connections in parallel; smaller values reduce burst load.
|
||||
# Used by: src/lib/tokenHealthCheck.ts. Default: 20 (Issue #7875, regression of #7719).
|
||||
# HEALTHCHECK_BATCH_SIZE=20
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
# 22. DEBUGGING
|
||||
# ═══════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
1
changelog.d/fixes/7875-healthcheck-jitter-config.md
Normal file
1
changelog.d/fixes/7875-healthcheck-jitter-config.md
Normal file
@@ -0,0 +1 @@
|
||||
- fix(auth): restore `HEALTHCHECK_BATCH_SIZE` env-var configurability of the OAuth token-health-check sweep batch size, dropped as a hardcoded constant by #7719 (#7875)
|
||||
@@ -864,6 +864,7 @@ Anthropic-compatible provider instead.
|
||||
| `HEALTHCHECK_STAGGER_MS` | `3000` | `src/lib/tokenHealthCheck.ts` | Stagger interval (ms) between provider token healthchecks at startup. |
|
||||
| `HEALTHCHECK_JITTER_MIN_MS` | `500` | `src/lib/tokenHealthCheck.ts` | Minimum randomized jitter (ms) added on top of `HEALTHCHECK_STAGGER_MS` between provider token healthchecks, to prevent bursting (Issue #1220). |
|
||||
| `HEALTHCHECK_JITTER_MAX_MS` | `5000` | `src/lib/tokenHealthCheck.ts` | Maximum randomized jitter (ms) added on top of `HEALTHCHECK_STAGGER_MS` between provider token healthchecks, to prevent bursting (Issue #1220). |
|
||||
| `HEALTHCHECK_BATCH_SIZE` | `20` | `src/lib/tokenHealthCheck.ts` | Concurrent-check batch size for the startup token-healthcheck sweep; larger values check more connections in parallel, smaller values reduce burst load (Issue #7875, regression of #7719). |
|
||||
| `REQUEST_RETRY` | `2` | `src/sse/services/cooldownAwareRetry.ts` | Number of automatic retries on model-scoped cooldown responses before returning error to client. |
|
||||
| `MAX_RETRY_INTERVAL_SEC` | `30` | `src/sse/services/cooldownAwareRetry.ts` | Max backoff interval (seconds) between cooldown retries. Capped by this value regardless of upstream `Retry-After`. |
|
||||
| `HEADROOM_URL` | `http://localhost:8787` | `src/lib/headroom/detect.ts` | Headroom token-saver proxy URL. The dashboard lifecycle (`api/headroom/*`) spawns a local `headroom-ai` CLI on loopback by default; override only to point at an external Docker sidecar proxy. |
|
||||
|
||||
@@ -31,7 +31,7 @@ import { refreshGithubCopilotSubTokenIfNeeded } from "@/lib/tokenHealthCheckCopi
|
||||
const LOG_PREFIX = "[HealthCheck]";
|
||||
const TRUE_ENV_VALUES = new Set(["1", "true", "yes", "on"]);
|
||||
const TICK_MS = 60 * 1000; // sweep interval: every 60 seconds (restored — #7719 dropped the const but kept two call sites)
|
||||
const BATCH_SIZE = 20;
|
||||
const DEFAULT_BATCH_SIZE = 20;
|
||||
const DEFAULT_HEALTH_CHECK_INTERVAL_MIN = 60; // default per-connection interval
|
||||
|
||||
function isBuildProcess(): boolean {
|
||||
@@ -161,6 +161,18 @@ export function clearRefreshCircuit(
|
||||
return next;
|
||||
}
|
||||
|
||||
/**
|
||||
* Concurrent-check batch size for the sweep, read per-call (not at module
|
||||
* load) so tests — and operators — can override it via HEALTHCHECK_BATCH_SIZE
|
||||
* without restarting the process. #7719 hardcoded this to a module-level
|
||||
* `const BATCH_SIZE = 20`, silently dropping the configurability restored
|
||||
* here (#7875). Falls back to DEFAULT_BATCH_SIZE on a missing/invalid value.
|
||||
*/
|
||||
function getConfiguredBatchSize(): number {
|
||||
const configured = parseInt(process.env.HEALTHCHECK_BATCH_SIZE || "", 10);
|
||||
return Number.isFinite(configured) && configured > 0 ? configured : DEFAULT_BATCH_SIZE;
|
||||
}
|
||||
|
||||
function isEnvFlagEnabled(name: string): boolean {
|
||||
const value = process.env[name];
|
||||
if (!value) return false;
|
||||
@@ -330,7 +342,7 @@ export async function sweep() {
|
||||
// batches. The inter-batch stagger preserves the original burst-
|
||||
// prevention intent (Issue #1220) while reducing total sweep time from
|
||||
// O(total × staggerMs) to O(total ÷ batchSize × staggerMs).
|
||||
const batchSize = Math.min(BATCH_SIZE, total);
|
||||
const batchSize = Math.min(getConfiguredBatchSize(), total);
|
||||
for (let offset = 0; offset < total; offset += batchSize) {
|
||||
const batchEnd = Math.min(offset + batchSize, total);
|
||||
const batch: Array<Promise<void>> = [];
|
||||
|
||||
129
tests/unit/tokenHealthCheck-batchSize.test.ts
Normal file
129
tests/unit/tokenHealthCheck-batchSize.test.ts
Normal file
@@ -0,0 +1,129 @@
|
||||
/**
|
||||
* Regression test for #7875 — PR #7719 (perf: batch concurrency for the
|
||||
* OAuth token-health-check sweep) replaced the configurable healthcheck
|
||||
* batch size with a hardcoded `const BATCH_SIZE = 20;` in
|
||||
* src/lib/tokenHealthCheck.ts, losing all configurability.
|
||||
*
|
||||
* Fix: `sweep()` must read HEALTHCHECK_BATCH_SIZE (default 20) per call,
|
||||
* the same pattern already used for HEALTHCHECK_STAGGER_MS /
|
||||
* HEALTHCHECK_JITTER_MIN_MS / HEALTHCHECK_JITTER_MAX_MS.
|
||||
*
|
||||
* This proves the batch size is configurable by shrinking it below the
|
||||
* connection count and counting how many inter-batch stagger delays
|
||||
* `sweep()` schedules via `setTimeout`. With 5 connections and
|
||||
* HEALTHCHECK_BATCH_SIZE=2, batches are [2, 2, 1] => 2 inter-batch gaps.
|
||||
* With the batch size hardcoded at 20, all 5 connections run in a single
|
||||
* batch => 0 inter-batch gaps. `global.setTimeout` is intercepted (and
|
||||
* fires immediately) so the assertion is deterministic and does not rely
|
||||
* on noisy wall-clock timing.
|
||||
*/
|
||||
import test from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import fs from "node:fs";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
|
||||
process.env.NODE_ENV = "test";
|
||||
|
||||
const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-batchsize-health-"));
|
||||
process.env.DATA_DIR = TEST_DATA_DIR;
|
||||
|
||||
const core = await import("../../src/lib/db/core.ts");
|
||||
const providersDb = await import("../../src/lib/db/providers.ts");
|
||||
|
||||
async function resetStorage() {
|
||||
core.resetDbInstance();
|
||||
for (let attempt = 0; attempt < 10; attempt++) {
|
||||
try {
|
||||
if (fs.existsSync(TEST_DATA_DIR)) {
|
||||
fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true });
|
||||
}
|
||||
break;
|
||||
} catch (error) {
|
||||
const code = (error as NodeJS.ErrnoException)?.code;
|
||||
if ((code === "EBUSY" || code === "EPERM") && attempt < 9) {
|
||||
await new Promise((resolve) => setTimeout(resolve, 50 * (attempt + 1)));
|
||||
} else {
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
}
|
||||
fs.mkdirSync(TEST_DATA_DIR, { recursive: true });
|
||||
}
|
||||
|
||||
test.after(() => {
|
||||
core.resetDbInstance();
|
||||
try {
|
||||
if (fs.existsSync(TEST_DATA_DIR)) {
|
||||
fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true });
|
||||
}
|
||||
} catch {
|
||||
/* best effort cleanup */
|
||||
}
|
||||
});
|
||||
|
||||
test("sweep() respects HEALTHCHECK_BATCH_SIZE instead of a hardcoded 20", async () => {
|
||||
await resetStorage();
|
||||
|
||||
// 5 connections, isActive=false so checkConnection() returns immediately
|
||||
// at the !conn.isActive guard without any OAuth network calls.
|
||||
for (let i = 1; i <= 5; i++) {
|
||||
await providersDb.createProviderConnection({
|
||||
provider: "openai",
|
||||
authType: "oauth",
|
||||
name: `BatchSize Test ${i}`,
|
||||
email: `bs${i}@example.com`,
|
||||
refreshToken: "test-rt",
|
||||
isActive: false,
|
||||
});
|
||||
}
|
||||
|
||||
const origSetting = process.env.HEALTHCHECK_SKIP_PROVIDERS;
|
||||
const origBatchSize = process.env.HEALTHCHECK_BATCH_SIZE;
|
||||
const origStagger = process.env.HEALTHCHECK_STAGGER_MS;
|
||||
process.env.HEALTHCHECK_BATCH_SIZE = "2";
|
||||
process.env.HEALTHCHECK_STAGGER_MS = "3000"; // matches the default so a real stagger delay is scheduled
|
||||
delete process.env.HEALTHCHECK_SKIP_PROVIDERS;
|
||||
delete process.env.HEALTHCHECK_JITTER_MIN_MS;
|
||||
delete process.env.HEALTHCHECK_JITTER_MAX_MS;
|
||||
|
||||
// Intercept setTimeout so the test runs instantly and deterministically —
|
||||
// we only care about the *delay values* sweep() schedules, not real elapsed time.
|
||||
const originalSetTimeout = global.setTimeout;
|
||||
const scheduledDelays: number[] = [];
|
||||
(global as unknown as { setTimeout: typeof setTimeout }).setTimeout = ((
|
||||
fn: (...args: unknown[]) => void,
|
||||
delay?: number,
|
||||
...args: unknown[]
|
||||
) => {
|
||||
scheduledDelays.push(delay ?? 0);
|
||||
return originalSetTimeout(fn, 0, ...args);
|
||||
}) as typeof setTimeout;
|
||||
|
||||
try {
|
||||
const { sweep } = await import("../../src/lib/tokenHealthCheck.ts");
|
||||
await sweep();
|
||||
|
||||
// Each inter-batch gap schedules a stagger-delay setTimeout (>= HEALTHCHECK_STAGGER_MS)
|
||||
// followed by a 0ms yield setTimeout. Count only the stagger-delay ones.
|
||||
const staggerCalls = scheduledDelays.filter((d) => d >= 3000);
|
||||
|
||||
// 5 connections / batch size 2 -> batches of [2, 2, 1] -> 2 inter-batch gaps.
|
||||
// With the batch size hardcoded at 20, all 5 connections would run in a
|
||||
// single batch -> 0 inter-batch gaps.
|
||||
assert.equal(
|
||||
staggerCalls.length,
|
||||
2,
|
||||
`expected 2 inter-batch stagger delays with HEALTHCHECK_BATCH_SIZE=2 and 5 connections ` +
|
||||
`(batches of [2,2,1]), got ${staggerCalls.length} — scheduled delays: ${JSON.stringify(scheduledDelays)}`
|
||||
);
|
||||
} finally {
|
||||
global.setTimeout = originalSetTimeout;
|
||||
if (origSetting !== undefined) process.env.HEALTHCHECK_SKIP_PROVIDERS = origSetting;
|
||||
else delete process.env.HEALTHCHECK_SKIP_PROVIDERS;
|
||||
if (origBatchSize !== undefined) process.env.HEALTHCHECK_BATCH_SIZE = origBatchSize;
|
||||
else delete process.env.HEALTHCHECK_BATCH_SIZE;
|
||||
if (origStagger !== undefined) process.env.HEALTHCHECK_STAGGER_MS = origStagger;
|
||||
else delete process.env.HEALTHCHECK_STAGGER_MS;
|
||||
}
|
||||
});
|
||||
Reference in New Issue
Block a user