From 4eea1cd14f5f8b767bb87349e38a0faed2d7b009 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza <8016841+diegosouzapw@users.noreply.github.com> Date: Tue, 21 Jul 2026 09:15:33 -0300 Subject: [PATCH] fix(auth): restore configurable HEALTHCHECK_BATCH_SIZE dropped by #7719 (#7875) (#7970) * fix(auth): restore configurable HEALTHCHECK_BATCH_SIZE dropped by #7719 (#7875) * docs(env): document HEALTHCHECK_BATCH_SIZE in .env.example (#7875) * docs(env): document HEALTHCHECK_BATCH_SIZE in .env.example + ENVIRONMENT.md (#7875) --- .env.example | 5 + .../fixes/7875-healthcheck-jitter-config.md | 1 + docs/reference/ENVIRONMENT.md | 1 + src/lib/tokenHealthCheck.ts | 16 ++- tests/unit/tokenHealthCheck-batchSize.test.ts | 129 ++++++++++++++++++ 5 files changed, 150 insertions(+), 2 deletions(-) create mode 100644 changelog.d/fixes/7875-healthcheck-jitter-config.md create mode 100644 tests/unit/tokenHealthCheck-batchSize.test.ts diff --git a/.env.example b/.env.example index 058bfc547d..5be7f65deb 100644 --- a/.env.example +++ b/.env.example @@ -1664,6 +1664,11 @@ APP_LOG_TO_FILE=true # HEALTHCHECK_JITTER_MIN_MS=500 # HEALTHCHECK_JITTER_MAX_MS=5000 +# Concurrent-check batch size for the startup token-healthcheck sweep. Larger +# values check more connections in parallel; smaller values reduce burst load. +# Used by: src/lib/tokenHealthCheck.ts. Default: 20 (Issue #7875, regression of #7719). +# HEALTHCHECK_BATCH_SIZE=20 + # ═══════════════════════════════════════════════════════════════════════════════ # 22. DEBUGGING # ═══════════════════════════════════════════════════════════════════════════════ diff --git a/changelog.d/fixes/7875-healthcheck-jitter-config.md b/changelog.d/fixes/7875-healthcheck-jitter-config.md new file mode 100644 index 0000000000..499820acef --- /dev/null +++ b/changelog.d/fixes/7875-healthcheck-jitter-config.md @@ -0,0 +1 @@ +- fix(auth): restore `HEALTHCHECK_BATCH_SIZE` env-var configurability of the OAuth token-health-check sweep batch size, dropped as a hardcoded constant by #7719 (#7875) diff --git a/docs/reference/ENVIRONMENT.md b/docs/reference/ENVIRONMENT.md index 4f30b75187..3e000b67da 100644 --- a/docs/reference/ENVIRONMENT.md +++ b/docs/reference/ENVIRONMENT.md @@ -864,6 +864,7 @@ Anthropic-compatible provider instead. | `HEALTHCHECK_STAGGER_MS` | `3000` | `src/lib/tokenHealthCheck.ts` | Stagger interval (ms) between provider token healthchecks at startup. | | `HEALTHCHECK_JITTER_MIN_MS` | `500` | `src/lib/tokenHealthCheck.ts` | Minimum randomized jitter (ms) added on top of `HEALTHCHECK_STAGGER_MS` between provider token healthchecks, to prevent bursting (Issue #1220). | | `HEALTHCHECK_JITTER_MAX_MS` | `5000` | `src/lib/tokenHealthCheck.ts` | Maximum randomized jitter (ms) added on top of `HEALTHCHECK_STAGGER_MS` between provider token healthchecks, to prevent bursting (Issue #1220). | +| `HEALTHCHECK_BATCH_SIZE` | `20` | `src/lib/tokenHealthCheck.ts` | Concurrent-check batch size for the startup token-healthcheck sweep; larger values check more connections in parallel, smaller values reduce burst load (Issue #7875, regression of #7719). | | `REQUEST_RETRY` | `2` | `src/sse/services/cooldownAwareRetry.ts` | Number of automatic retries on model-scoped cooldown responses before returning error to client. | | `MAX_RETRY_INTERVAL_SEC` | `30` | `src/sse/services/cooldownAwareRetry.ts` | Max backoff interval (seconds) between cooldown retries. Capped by this value regardless of upstream `Retry-After`. | | `HEADROOM_URL` | `http://localhost:8787` | `src/lib/headroom/detect.ts` | Headroom token-saver proxy URL. The dashboard lifecycle (`api/headroom/*`) spawns a local `headroom-ai` CLI on loopback by default; override only to point at an external Docker sidecar proxy. | diff --git a/src/lib/tokenHealthCheck.ts b/src/lib/tokenHealthCheck.ts index f57c012d30..4007a5d60d 100644 --- a/src/lib/tokenHealthCheck.ts +++ b/src/lib/tokenHealthCheck.ts @@ -31,7 +31,7 @@ import { refreshGithubCopilotSubTokenIfNeeded } from "@/lib/tokenHealthCheckCopi const LOG_PREFIX = "[HealthCheck]"; const TRUE_ENV_VALUES = new Set(["1", "true", "yes", "on"]); const TICK_MS = 60 * 1000; // sweep interval: every 60 seconds (restored — #7719 dropped the const but kept two call sites) -const BATCH_SIZE = 20; +const DEFAULT_BATCH_SIZE = 20; const DEFAULT_HEALTH_CHECK_INTERVAL_MIN = 60; // default per-connection interval function isBuildProcess(): boolean { @@ -161,6 +161,18 @@ export function clearRefreshCircuit( return next; } +/** + * Concurrent-check batch size for the sweep, read per-call (not at module + * load) so tests — and operators — can override it via HEALTHCHECK_BATCH_SIZE + * without restarting the process. #7719 hardcoded this to a module-level + * `const BATCH_SIZE = 20`, silently dropping the configurability restored + * here (#7875). Falls back to DEFAULT_BATCH_SIZE on a missing/invalid value. + */ +function getConfiguredBatchSize(): number { + const configured = parseInt(process.env.HEALTHCHECK_BATCH_SIZE || "", 10); + return Number.isFinite(configured) && configured > 0 ? configured : DEFAULT_BATCH_SIZE; +} + function isEnvFlagEnabled(name: string): boolean { const value = process.env[name]; if (!value) return false; @@ -330,7 +342,7 @@ export async function sweep() { // batches. The inter-batch stagger preserves the original burst- // prevention intent (Issue #1220) while reducing total sweep time from // O(total × staggerMs) to O(total ÷ batchSize × staggerMs). - const batchSize = Math.min(BATCH_SIZE, total); + const batchSize = Math.min(getConfiguredBatchSize(), total); for (let offset = 0; offset < total; offset += batchSize) { const batchEnd = Math.min(offset + batchSize, total); const batch: Array> = []; diff --git a/tests/unit/tokenHealthCheck-batchSize.test.ts b/tests/unit/tokenHealthCheck-batchSize.test.ts new file mode 100644 index 0000000000..c6a088b3f3 --- /dev/null +++ b/tests/unit/tokenHealthCheck-batchSize.test.ts @@ -0,0 +1,129 @@ +/** + * Regression test for #7875 — PR #7719 (perf: batch concurrency for the + * OAuth token-health-check sweep) replaced the configurable healthcheck + * batch size with a hardcoded `const BATCH_SIZE = 20;` in + * src/lib/tokenHealthCheck.ts, losing all configurability. + * + * Fix: `sweep()` must read HEALTHCHECK_BATCH_SIZE (default 20) per call, + * the same pattern already used for HEALTHCHECK_STAGGER_MS / + * HEALTHCHECK_JITTER_MIN_MS / HEALTHCHECK_JITTER_MAX_MS. + * + * This proves the batch size is configurable by shrinking it below the + * connection count and counting how many inter-batch stagger delays + * `sweep()` schedules via `setTimeout`. With 5 connections and + * HEALTHCHECK_BATCH_SIZE=2, batches are [2, 2, 1] => 2 inter-batch gaps. + * With the batch size hardcoded at 20, all 5 connections run in a single + * batch => 0 inter-batch gaps. `global.setTimeout` is intercepted (and + * fires immediately) so the assertion is deterministic and does not rely + * on noisy wall-clock timing. + */ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +process.env.NODE_ENV = "test"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-batchsize-health-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const providersDb = await import("../../src/lib/db/providers.ts"); + +async function resetStorage() { + core.resetDbInstance(); + for (let attempt = 0; attempt < 10; attempt++) { + try { + if (fs.existsSync(TEST_DATA_DIR)) { + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + } + break; + } catch (error) { + const code = (error as NodeJS.ErrnoException)?.code; + if ((code === "EBUSY" || code === "EPERM") && attempt < 9) { + await new Promise((resolve) => setTimeout(resolve, 50 * (attempt + 1))); + } else { + throw error; + } + } + } + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +test.after(() => { + core.resetDbInstance(); + try { + if (fs.existsSync(TEST_DATA_DIR)) { + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + } + } catch { + /* best effort cleanup */ + } +}); + +test("sweep() respects HEALTHCHECK_BATCH_SIZE instead of a hardcoded 20", async () => { + await resetStorage(); + + // 5 connections, isActive=false so checkConnection() returns immediately + // at the !conn.isActive guard without any OAuth network calls. + for (let i = 1; i <= 5; i++) { + await providersDb.createProviderConnection({ + provider: "openai", + authType: "oauth", + name: `BatchSize Test ${i}`, + email: `bs${i}@example.com`, + refreshToken: "test-rt", + isActive: false, + }); + } + + const origSetting = process.env.HEALTHCHECK_SKIP_PROVIDERS; + const origBatchSize = process.env.HEALTHCHECK_BATCH_SIZE; + const origStagger = process.env.HEALTHCHECK_STAGGER_MS; + process.env.HEALTHCHECK_BATCH_SIZE = "2"; + process.env.HEALTHCHECK_STAGGER_MS = "3000"; // matches the default so a real stagger delay is scheduled + delete process.env.HEALTHCHECK_SKIP_PROVIDERS; + delete process.env.HEALTHCHECK_JITTER_MIN_MS; + delete process.env.HEALTHCHECK_JITTER_MAX_MS; + + // Intercept setTimeout so the test runs instantly and deterministically — + // we only care about the *delay values* sweep() schedules, not real elapsed time. + const originalSetTimeout = global.setTimeout; + const scheduledDelays: number[] = []; + (global as unknown as { setTimeout: typeof setTimeout }).setTimeout = (( + fn: (...args: unknown[]) => void, + delay?: number, + ...args: unknown[] + ) => { + scheduledDelays.push(delay ?? 0); + return originalSetTimeout(fn, 0, ...args); + }) as typeof setTimeout; + + try { + const { sweep } = await import("../../src/lib/tokenHealthCheck.ts"); + await sweep(); + + // Each inter-batch gap schedules a stagger-delay setTimeout (>= HEALTHCHECK_STAGGER_MS) + // followed by a 0ms yield setTimeout. Count only the stagger-delay ones. + const staggerCalls = scheduledDelays.filter((d) => d >= 3000); + + // 5 connections / batch size 2 -> batches of [2, 2, 1] -> 2 inter-batch gaps. + // With the batch size hardcoded at 20, all 5 connections would run in a + // single batch -> 0 inter-batch gaps. + assert.equal( + staggerCalls.length, + 2, + `expected 2 inter-batch stagger delays with HEALTHCHECK_BATCH_SIZE=2 and 5 connections ` + + `(batches of [2,2,1]), got ${staggerCalls.length} — scheduled delays: ${JSON.stringify(scheduledDelays)}` + ); + } finally { + global.setTimeout = originalSetTimeout; + if (origSetting !== undefined) process.env.HEALTHCHECK_SKIP_PROVIDERS = origSetting; + else delete process.env.HEALTHCHECK_SKIP_PROVIDERS; + if (origBatchSize !== undefined) process.env.HEALTHCHECK_BATCH_SIZE = origBatchSize; + else delete process.env.HEALTHCHECK_BATCH_SIZE; + if (origStagger !== undefined) process.env.HEALTHCHECK_STAGGER_MS = origStagger; + else delete process.env.HEALTHCHECK_STAGGER_MS; + } +});