diff --git a/changelog.d/fixes/9147-catalog-eventloop-yield.md b/changelog.d/fixes/9147-catalog-eventloop-yield.md new file mode 100644 index 0000000000..1f27c92b33 --- /dev/null +++ b/changelog.d/fixes/9147-catalog-eventloop-yield.md @@ -0,0 +1 @@ +- fix(api): yield the event loop during catalog builds and bulk-load override/hidden-model tables (#9147) \ No newline at end of file diff --git a/src/app/api/v1/models/catalog.ts b/src/app/api/v1/models/catalog.ts index 0ba6430434..c40e7b5fb1 100644 --- a/src/app/api/v1/models/catalog.ts +++ b/src/app/api/v1/models/catalog.ts @@ -6,9 +6,9 @@ import { getAllCustomModels, getSettings, getCachedProviderNodes, - getModelIsHidden, getModelAliases, getDatabaseSettings, + getHiddenModelsByProvider, } from "@/lib/localDb"; import { createLazyConnectionView } from "@/lib/db/providers/lazyConnectionView"; import { extractAliasBackedModels } from "./aliasBackedModels"; @@ -132,7 +132,7 @@ export { } from "./catalogCache"; export type { CachedCatalog } from "./catalogCache"; -const BUILTIN_AUTO_YIELD_INTERVAL = 8; +const BUILTIN_AUTO_YIELD_INTERVAL = 2; function yieldCatalogBuildTurn(): Promise { return new Promise((resolve) => setImmediate(resolve)); @@ -156,6 +156,8 @@ export async function getUnifiedModelsResponse( try { settingsForAuth = await getSettings(); } catch {} + // #9147: yield before auth check to allow event loop tick + await yieldCatalogBuildTurn(); const authRejection = await getModelCatalogAuthRejection(request, settingsForAuth, { ...corsHeaders, ...diagnosticHeaders, @@ -226,6 +228,28 @@ async function buildUnifiedModelsResponseCore( corsHeaders: Record = {} ) { const diagnosticHeaders = getCatalogDiagnosticsHeaders({ request }); + // #9147: this builder walks connections + model registries at catalog scale with no + // event-loop yield, so a large deployment pins the single Node.js thread for the + // whole build (reporter: 183 connections / 2000+ models → 10.1s stall that blocks the + // dashboard WS heartbeat). Yield every `catYIELD_EVERY` items across the hot loops. + const catYIELD_EVERY = 20; + let catYieldCount = 0; + const maybeYieldCatalogBuild = async (): Promise => { + catYieldCount++; + if (catYieldCount % catYIELD_EVERY === 0) { + await yieldCatalogBuildTurn(); + } + }; + // #9147: `getModelIsHidden()` is a SQLite read per call (custom row + compat list) + // and the build consults it ~16× per entry. Bulk-load the hidden-model map once + // (one query — `getHiddenModelsByProvider`) and resolve from memory for the whole + // build. A provider absent from the map has no hidden models at all — `false`, + // no on-demand fallback (that would reintroduce the per-call SQLite reads). + const hiddenModelsByProvider = getHiddenModelsByProvider(); + const isModelHiddenBulk = (providerId: string, modelId: string): boolean => { + const hiddenSet = hiddenModelsByProvider.get(providerId); + return hiddenSet ? hiddenSet.has(modelId) : false; + }; try { let settings: Record = {}; try { @@ -237,6 +261,10 @@ async function buildUnifiedModelsResponseCore( ...diagnosticHeaders, }); if (authRejection) return authRejection; + + // #9147: yield after auth check before DB initialization prologue + await yieldCatalogBuildTurn(); + const { aliasToProviderId, providerIdToAlias } = buildAliasMaps(); const _qp = new URL(request.url).searchParams.get("prefix"); const prefixMode = @@ -321,6 +349,7 @@ async function buildUnifiedModelsResponseCore( // Get combos let combos = []; + await yieldCatalogBuildTurn(); try { combos = await getCombos(); } catch (e) { @@ -588,7 +617,7 @@ async function buildUnifiedModelsResponseCore( timestamp, (c) => buildComboCatalogMetadata(c, combos) ); - const quotaFinal = applyCatalogPostFilters(request, quotaModels, { + const quotaFinal = await applyCatalogPostFilters(request, quotaModels, { connections, prefixMode, aliasToProviderId, @@ -683,7 +712,7 @@ async function buildUnifiedModelsResponseCore( ) as ComboCatalogTarget[]; const visibleTargets = comboTargets.filter((target) => { const resolved = getComboTargetModelId(target); - return resolved ? !getModelIsHidden(resolved.providerId, resolved.modelId) : true; + return resolved ? !isModelHiddenBulk(resolved.providerId, resolved.modelId) : true; }); if (visibleTargets.length === 0) continue; @@ -700,11 +729,16 @@ async function buildUnifiedModelsResponseCore( parent: null, ...comboMetadata, }); + + // #9147: combos can number hundreds at catalog scale — yield periodically. + await maybeYieldCatalogBuild(); } let syncedModelsByProvider: Record = {}; try { + await yieldCatalogBuildTurn(); syncedModelsByProvider = await getAllActiveSyncedModels(); + await yieldCatalogBuildTurn(); } catch (e) { // DB unavailable — log and fall through; static models remain as defaults. console.log("[catalog] Could not fetch synced available models:", e); @@ -792,7 +826,7 @@ async function buildUnifiedModelsResponseCore( if (!isModelSelectable(canonicalProviderId, model.id)) continue; if (!providerSupportsModel(canonicalProviderId, model.id)) continue; const aliasId = `${alias}/${model.id}`; - if (getModelIsHidden(canonicalProviderId, model.id)) continue; + if (isModelHiddenBulk(canonicalProviderId, model.id)) continue; if (shouldHidePaid(canonicalProviderId, model.id, (model as { pricing?: unknown }).pricing)) continue; @@ -846,12 +880,15 @@ async function buildUnifiedModelsResponseCore( ...thinkingCapabilities, }); } + + // #9147: static model walk is the densest loop — yield periodically. + await maybeYieldCatalogBuild(); } } for (const modelId of CODEX_NATIVE_UNPREFIXED_MODELS) { if (!providerSupportsModel("codex", modelId)) continue; - if (getModelIsHidden("codex", modelId)) continue; + if (isModelHiddenBulk("codex", modelId)) continue; const alias = providerIdToAlias.codex || "cx"; const aliasId = `${alias}/${modelId}`; @@ -912,7 +949,7 @@ async function buildUnifiedModelsResponseCore( if (canonicalProviderId === "codex" && isCodexDiscoveryModelExcluded(sm)) { continue; } - if (getModelIsHidden(providerId, sm.id)) continue; + if (isModelHiddenBulk(providerId, sm.id)) continue; // #6457: some upstream discovery catalogs (e.g. HuggingFace's live // `/v1/models`) return image/diffusion models with no modality info, // so `endpoints` below would default to ["chat"] and misrepresent @@ -1024,6 +1061,9 @@ async function buildUnifiedModelsResponseCore( }); } } + + // #9147: synced-model union is usually the largest walk — yield periodically. + await maybeYieldCatalogBuild(); } } } catch (err) { @@ -1053,7 +1093,7 @@ async function buildUnifiedModelsResponseCore( if (hidePaid && !isFree) continue; // #9293: respect per-model hidden flags (e.g. operator hid google/chirp-3 // from the OpenRouter provider, so it should not appear in the live catalog). - if (getModelIsHidden("openrouter", openRouterModel.id)) continue; + if (isModelHiddenBulk("openrouter", openRouterModel.id)) continue; const supportedParameters = Array.isArray(openRouterModel.supported_parameters) ? openRouterModel.supported_parameters : []; @@ -1094,6 +1134,9 @@ async function buildUnifiedModelsResponseCore( ...(outputModalities.length > 0 ? { output_modalities: outputModalities } : {}), ...(Object.keys(capabilities).length > 0 ? { capabilities } : {}), }); + + // #9147: OpenRouter catalog can be large — yield periodically. + await maybeYieldCatalogBuild(); } } catch (err) { console.error("[catalog] Error loading OpenRouter catalog:", err); @@ -1138,7 +1181,7 @@ async function buildUnifiedModelsResponseCore( // Helper: strip the provider prefix from a specialty model ID to get the // provider-relative path (e.g. "openrouter/google/chirp-3" -> "google/chirp-3"). - // This is the correct key used by getModelIsHidden() — using .split("/").pop() + // This is the correct key used by the hidden-model lookup — using .split("/").pop() // here would discard all but the last segment and miss stored flags for // providers whose model IDs carry a sub-path (e.g. OpenRouter scoped models). const getSpecialtyModelRelativeId = (modelId: string, provider: string): string => @@ -1149,7 +1192,7 @@ async function buildUnifiedModelsResponseCore( if (!isProviderActive(embModel.provider)) continue; const rawModelId = getSpecialtyModelRelativeId(embModel.id, embModel.provider); if (!providerSupportsModel(embModel.provider, rawModelId)) continue; - if (getModelIsHidden(embModel.provider, rawModelId)) continue; + if (isModelHiddenBulk(embModel.provider, rawModelId)) continue; if (hasEquivalentSpecialtyModel(embModel.provider, rawModelId, "embedding", embModel.id)) { continue; } @@ -1169,7 +1212,7 @@ async function buildUnifiedModelsResponseCore( if (!isProviderActive(imgModel.provider)) continue; const rawModelId = getSpecialtyModelRelativeId(imgModel.id, imgModel.provider); if (!providerSupportsModel(imgModel.provider, rawModelId)) continue; - if (getModelIsHidden(imgModel.provider, rawModelId)) continue; + if (isModelHiddenBulk(imgModel.provider, rawModelId)) continue; models.push({ id: imgModel.id, object: "model", @@ -1189,7 +1232,7 @@ async function buildUnifiedModelsResponseCore( if (!isProviderActive(rerankModel.provider)) continue; const rawModelId = getSpecialtyModelRelativeId(rerankModel.id, rerankModel.provider); if (!providerSupportsModel(rerankModel.provider, rawModelId)) continue; - if (getModelIsHidden(rerankModel.provider, rawModelId)) continue; + if (isModelHiddenBulk(rerankModel.provider, rawModelId)) continue; if (hasEquivalentSpecialtyModel(rerankModel.provider, rawModelId, "rerank", rerankModel.id)) { continue; } @@ -1208,7 +1251,7 @@ async function buildUnifiedModelsResponseCore( if (!isProviderActive(audioModel.provider)) continue; const rawModelId = getSpecialtyModelRelativeId(audioModel.id, audioModel.provider); if (!providerSupportsModel(audioModel.provider, rawModelId)) continue; - if (getModelIsHidden(audioModel.provider, rawModelId)) continue; + if (isModelHiddenBulk(audioModel.provider, rawModelId)) continue; models.push({ id: audioModel.id, object: "model", @@ -1224,7 +1267,7 @@ async function buildUnifiedModelsResponseCore( if (!isProviderActive(modModel.provider)) continue; const rawModelId = getSpecialtyModelRelativeId(modModel.id, modModel.provider); if (!providerSupportsModel(modModel.provider, rawModelId)) continue; - if (getModelIsHidden(modModel.provider, rawModelId)) continue; + if (isModelHiddenBulk(modModel.provider, rawModelId)) continue; models.push({ id: modModel.id, object: "model", @@ -1239,7 +1282,7 @@ async function buildUnifiedModelsResponseCore( if (!isProviderActive(videoModel.provider)) continue; const rawModelId = getSpecialtyModelRelativeId(videoModel.id, videoModel.provider); if (!providerSupportsModel(videoModel.provider, rawModelId)) continue; - if (getModelIsHidden(videoModel.provider, rawModelId)) continue; + if (isModelHiddenBulk(videoModel.provider, rawModelId)) continue; models.push({ id: videoModel.id, object: "model", @@ -1260,7 +1303,7 @@ async function buildUnifiedModelsResponseCore( if (!isProviderActive(musicModel.provider)) continue; const rawModelId = getSpecialtyModelRelativeId(musicModel.id, musicModel.provider); if (!providerSupportsModel(musicModel.provider, rawModelId)) continue; - if (getModelIsHidden(musicModel.provider, rawModelId)) continue; + if (isModelHiddenBulk(musicModel.provider, rawModelId)) continue; models.push({ id: musicModel.id, object: "model", @@ -1306,7 +1349,7 @@ async function buildUnifiedModelsResponseCore( if (!isUnifiedChatSourceModelSelectable(canonicalProviderId, { ...model, id: modelId })) continue; if (model.isHidden === true) continue; - if (getModelIsHidden(canonicalProviderId, modelId)) continue; + if (isModelHiddenBulk(canonicalProviderId, modelId)) continue; // #6328: apply hidePaidModels to user-defined custom rows too. // Custom entries do not carry pricing, so shouldHidePaid() decides // via FREE_MODEL_IDS_BY_PROVIDER — matches synced/PROVIDER_MODELS. @@ -1441,6 +1484,9 @@ async function buildUnifiedModelsResponseCore( ...(providerVisionFields || {}), }); } + + // #9147: custom-model walk — yield periodically. + await maybeYieldCatalogBuild(); } } } catch (e) { @@ -1486,7 +1532,7 @@ async function buildUnifiedModelsResponseCore( continue; } - if (getModelIsHidden(canonicalProviderId, modelId)) continue; + if (isModelHiddenBulk(canonicalProviderId, modelId)) continue; // #6328: apply hidePaidModels to alias-backed rows too. Alias mappings // point at providerKey/modelId with no pricing, so shouldHidePaid() // decides via the FREE_MODEL_IDS_BY_PROVIDER catalog tier. @@ -1559,7 +1605,7 @@ async function buildUnifiedModelsResponseCore( for (const model of fallbackModels) { const modelId = typeof model.id === "string" ? model.id : null; if (!modelId) continue; - if (getModelIsHidden(canonicalProviderId, modelId)) continue; + if (isModelHiddenBulk(canonicalProviderId, modelId)) continue; // #6328: apply hidePaidModels to managed-fallback rows too. Compatible // provider fallbacks lack pricing; shouldHidePaid() decides via the // FREE_MODEL_IDS_BY_PROVIDER catalog tier. @@ -1586,6 +1632,9 @@ async function buildUnifiedModelsResponseCore( ...(contextLength ? { context_length: contextLength } : {}), ...(visionFields || {}), }); + + // #9147: per-connection fallback walk — yield periodically. + await maybeYieldCatalogBuild(); } } @@ -1631,7 +1680,7 @@ async function buildUnifiedModelsResponseCore( } } // ?configuredOnly — hide models that have no eligible DB connection. - finalModels = applyCatalogPostFilters(request, finalModels, { + finalModels = await applyCatalogPostFilters(request, finalModels, { connections, prefixMode, aliasToProviderId, diff --git a/src/app/api/v1/models/catalogResponse.ts b/src/app/api/v1/models/catalogResponse.ts index 7da55d5bca..198a5a6d30 100644 --- a/src/app/api/v1/models/catalogResponse.ts +++ b/src/app/api/v1/models/catalogResponse.ts @@ -32,6 +32,7 @@ import { enrichCatalogModelEntry, type CatalogEnrichmentSnapshot, } from "@/lib/modelMetadataRegistry"; +import { createModelCapabilityResolutionSnapshot } from "@/lib/modelCapabilityResolutionSnapshot"; import { isModelCatalogNamesEnabled } from "@/shared/utils/featureFlags"; import { extractApiKey } from "@/sse/services/auth"; import { maybeOmitCatalogModelName } from "./catalogHelpers"; @@ -45,7 +46,7 @@ import { isCodexModelCatalogClient } from "./catalogRequest"; * returns early, but it still owes the caller these steps — the discovery mirrors in * particular are what let Claude Code see a quota pool's models at all. */ -export function applyCatalogPostFilters( +export async function applyCatalogPostFilters( request: Request, models: Array>, ctx: { @@ -54,7 +55,8 @@ export function applyCatalogPostFilters( aliasToProviderId: Record; hideNoThinkVariants?: boolean; } -): Array> { +): Promise>> { + const yieldTurn = (): Promise => new Promise((resolve) => setImmediate(resolve)); let finalModels = models; // variants are only generated for surviving models. @@ -65,6 +67,11 @@ export function applyCatalogPostFilters( }); } + // #9147: the variant-append passes each walk the full model list (O(n) per pass), + // so a catalog-scale build must not run all of them in one synchronous stretch. + // Yield once between the expensive passes to let the event loop breathe. + await yieldTurn(); + // Advertise Claude reasoning-effort variants (claude/-{low,medium,high[,xhigh]}). // Derived from the already key-filtered list so a variant only appears when its real // model is permitted. Runs before the no-thinking pass: the gateway already routes these @@ -139,11 +146,15 @@ export function applyCatalogPostFilters( ); } + await yieldTurn(); + // #7694: advertise `/-` variants for synced models that // captured `reasoning.supported_efforts` at sync time (capabilities.effort_tiers). // Derived from the already key-filtered list; skips codex/kimi (own suffix mechanism). finalModels = appendSyncedEffortVariants(finalModels); + await yieldTurn(); + // #4424 follow-up — drop exact-duplicate ids that slip through the per-source push // guards (e.g. `codex/gpt-5.5`, `veo-free/seedance` listed twice). Keyed by listing // identity (id, type, subtype) so the intentional same-id audio transcription/speech @@ -207,25 +218,45 @@ export async function finalizeCatalogResponse( } const includeModelNames = isModelCatalogNamesEnabled(); - const enrichedModels = disambiguateCatalogModelNames( - finalModels.map((model) => { - if (model.owned_by === "combo") { - return maybeOmitCatalogModelName(model, includeModelNames); - } - const enriched = enrichCatalogModelEntry(model, undefined, enrichmentSnapshot); - const fallbackContextLength = getContextFallback(enriched); - const listedModel = fallbackContextLength - ? { ...enriched, context_length: fallbackContextLength } - : enriched; - return maybeOmitCatalogModelName(listedModel, includeModelNames); - }) - ); - // Canonical provider-grouped publication: one contiguous block per provider, - // combos pinned first. Stable — preserves combo sort_order, connection priority, - // and equal-id audio twins. Grouped by owned_by (canonical identity), not the - // routing alias prefix. Applied after enrichment/disambiguation so the final - // serialized order is what every consumer sees; cached as part of the body. + // #9147: enrichment is the most expensive single stage of the catalog build — + // per-entry provider/model resolution plus pricing + token/context override + // lookups. Two fixes so a large catalog cannot pin the Node.js thread here: + // (1) bulk-load the synced-capability + override tables ONCE into an in-memory + // snapshot (#9199 machinery) so per-entry enrichment never hits SQLite; + // (2) yield to the event loop every `YIELD_EVERY` entries so even the remaining + // per-entry work is interleaved with other callers / the dashboard WS. + const yieldTurn = (): Promise => new Promise((resolve) => setImmediate(resolve)); + await yieldTurn(); + const capabilityResolutionSnapshot = createModelCapabilityResolutionSnapshot(); + const enriched: Array> = []; + const catYIELD_EVERY = 5; + let catEnrichCount = 0; + for (const model of finalModels) { + let listedModel: Record; + if (model.owned_by === "combo") { + listedModel = maybeOmitCatalogModelName(model, includeModelNames); + } else { + const entry = enrichCatalogModelEntry(model, undefined, { + ...enrichmentSnapshot, + capabilityResolutionSnapshot, + }); + const fallbackContextLength = getContextFallback(entry); + listedModel = fallbackContextLength + ? { ...entry, context_length: fallbackContextLength } + : entry; + listedModel = maybeOmitCatalogModelName(listedModel, includeModelNames); + } + enriched.push(listedModel); + catEnrichCount++; + if (catEnrichCount % catYIELD_EVERY === 0) { + await yieldTurn(); + } + } + await yieldTurn(); + const enrichedModels = disambiguateCatalogModelNames(enriched); + await yieldTurn(); const orderedModels = sortCatalogModelsProviderGrouped(enrichedModels); + await yieldTurn(); // Codex CLI compatibility: its model-catalog refresh (codex_models_manager) does // GET /v1/models?client_version= and decodes a JSON object with a TOP-LEVEL // `models` array, so the OpenAI-standard `{object,data}` shape makes it fail with diff --git a/src/lib/modelCapabilities.ts b/src/lib/modelCapabilities.ts index a4416925b0..577ac9748d 100644 --- a/src/lib/modelCapabilities.ts +++ b/src/lib/modelCapabilities.ts @@ -616,9 +616,12 @@ function getContextOverride( /** * Resolve a persisted context override by canonical id, then by the exact raw * alias supplied by the caller. Neither lookup inherits to related models. + * + * `snapshot` is the #9147 build-local bulk load; when supplied the on-demand + * SQLite read is skipped and the preloaded nested map is used instead. */ -export function getResolvedModelContextOverride(input: CapabilityInput): number | null { - return getContextOverride(resolveCapabilityInput(input)); +export function getResolvedModelContextOverride(input: CapabilityInput, snapshot?: ModelCapabilityResolutionSnapshot | null): number | null { + return getContextOverride(resolveCapabilityInput(input), snapshot); } function getInputTokenCapabilityOverride(resolved: { diff --git a/src/lib/modelMetadataRegistry.ts b/src/lib/modelMetadataRegistry.ts index 667230839b..413836e5a6 100644 --- a/src/lib/modelMetadataRegistry.ts +++ b/src/lib/modelMetadataRegistry.ts @@ -28,6 +28,7 @@ import { CANONICAL_EFFORT_VALUES, extendCodexGpt56EffortValues, } from "@/shared/reasoning/effortStandardization"; +import type { ModelCapabilityResolutionSnapshot } from "@/lib/modelCapabilityResolutionSnapshot"; const MODEL_METADATA_SCHEMA_VERSION = "model-metadata-v1"; @@ -40,6 +41,9 @@ type JsonRecord = Record; export interface CatalogEnrichmentSnapshot { modelsDevPricing: PricingByProvider | null; providerNodeIdsByPrefix?: Readonly>; + /** #9147: build-local bulk load of synced capabilities + token/context overrides + * so per-entry enrichment never hits SQLite again (see catalogResponse.ts). */ + capabilityResolutionSnapshot?: ModelCapabilityResolutionSnapshot | null; } interface CatalogDiagnosticsOptions { @@ -200,20 +204,27 @@ export function getCatalogDiagnosticsHeaders( export function getCanonicalModelMetadata(input: { provider?: string | null; model?: string | null; + snapshot?: ModelCapabilityResolutionSnapshot | null; }): CanonicalModelMetadata | null { const modelId = asNonEmptyString(input.model); if (!modelId) return null; - const resolved = getResolvedModelCapabilities({ - provider: input.provider || null, - model: modelId, - }); + const resolved = getResolvedModelCapabilities( + { + provider: input.provider || null, + model: modelId, + }, + undefined, + input.snapshot || null + ); const provider = resolved.provider; const providerAlias = provider ? PROVIDER_ID_TO_ALIAS[provider] || provider : null; const registryModel = getRegistryModel(providerAlias || provider, resolved.model || modelId); const staticSpec = getModelSpec(resolved.model || modelId); const syncedCapability = - provider && resolved.model ? getSyncedCapability(provider, resolved.model) : null; + provider && resolved.model + ? getSyncedCapability(provider, resolved.model, input.snapshot?.synced ?? null) + : null; const canonicalStaticAlias = resolveStaticModelAlias(resolved.model || modelId); const modalities = buildModalities( resolved.modalitiesInput, @@ -420,7 +431,11 @@ export function enrichCatalogModelEntry( return id; })(); - const metadata = getCanonicalModelMetadata({ provider, model }); + const metadata = getCanonicalModelMetadata({ + provider, + model, + snapshot: snapshot?.capabilityResolutionSnapshot ?? null, + }); if (!metadata) return entry; const registryModel = getRegistryModel( metadata.providerAlias || metadata.provider, @@ -436,7 +451,11 @@ export function enrichCatalogModelEntry( getAuthoritativeContextWindow(metadata.model) ?? getAuthoritativeContextWindow(model); const specialtySurface = isNonChatCatalogSurface(entry.type); - const persistedContextWindow = getResolvedModelContextOverride({ provider, model }); + const capabilitySnapshot = snapshot?.capabilityResolutionSnapshot ?? null; + const persistedContextWindow = getResolvedModelContextOverride( + { provider, model }, + capabilitySnapshot + ); const capabilityFields = { ...(typeof metadata.capabilities.vision === "boolean" ? { vision: metadata.capabilities.vision } @@ -528,10 +547,15 @@ export function enrichCatalogModelEntry( } const persistedOutputLimit = - getModelCapabilityOverride(provider, model, "max_output_tokens") ?? - getModelCapabilityOverride(provider, model, "max_token") ?? - getModelCapabilityOverride(publicProvider, model, "max_output_tokens") ?? - getModelCapabilityOverride(publicProvider, model, "max_token"); + getModelCapabilityOverride(provider, model, "max_output_tokens", capabilitySnapshot?.maxTokenOverrides) ?? + getModelCapabilityOverride(provider, model, "max_token", capabilitySnapshot?.maxTokenOverrides) ?? + getModelCapabilityOverride( + publicProvider, + model, + "max_output_tokens", + capabilitySnapshot?.maxTokenOverrides + ) ?? + getModelCapabilityOverride(publicProvider, model, "max_token", capabilitySnapshot?.maxTokenOverrides); if (persistedOutputLimit !== null) { nextEntry.max_output_tokens = persistedOutputLimit; } else if ( diff --git a/tests/unit/9147-catalog-eventloop-yield.test.ts b/tests/unit/9147-catalog-eventloop-yield.test.ts new file mode 100644 index 0000000000..d915660896 --- /dev/null +++ b/tests/unit/9147-catalog-eventloop-yield.test.ts @@ -0,0 +1,88 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-9147-")); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.API_KEY_SECRET = process.env.API_KEY_SECRET || "catalog-9147-test-secret"; + +const core = await import("../../src/lib/db/core.ts"); +const apiKeysDb = await import("../../src/lib/db/apiKeys.ts"); +const v1ModelsCatalog = await import("../../src/app/api/v1/models/catalog.ts"); + +const CONNECTION_COUNT = 60; +const MODELS_PER_CONNECTION = 12; // ~720 synced models total + +async function resetStorage() { + core.resetDbInstance(); + apiKeysDb.resetApiKeyState(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); + v1ModelsCatalog.__resetCatalogBuilderRunsForTest(); +} + +async function seedCatalogScaleDataset() { + const db = core.getDbInstance(); + const now = new Date().toISOString(); + const insertConn = db.prepare( + `INSERT INTO provider_connections (id, provider, auth_type, name, priority, is_active, api_key, created_at, updated_at) + VALUES (?, 'openai-compatible', 'apikey', ?, ?, 1, ?, ?, ?)` + ); + const insertModels = db.prepare( + `INSERT INTO key_value (namespace, key, value) VALUES ('syncedAvailableModels', ?, ?)` + ); + const seedTx = db.transaction(() => { + for (let i = 0; i < CONNECTION_COUNT; i++) { + const id = `probe-conn-${i}`; + insertConn.run(id, `probe-connection-${i}`, i, `sk-probe-${i}`, now, now); + const models = Array.from({ length: MODELS_PER_CONNECTION }, (_, m) => ({ + id: `probe-model-${i}-${m}`, + name: `Probe Model ${i}-${m}`, + contextLength: 128000, + })); + insertModels.run(`openai-compatible:${id}`, JSON.stringify(models)); + } + }); + seedTx(); +} + +test.beforeEach(async () => { + await resetStorage(); +}); + +test.after(async () => { + core.resetDbInstance(); + apiKeysDb.resetApiKeyState(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +test("#9147 — catalog build at catalog-scale must not pin the event loop for a long stretch", async () => { + await seedCatalogScaleDataset(); + const req = new Request("http://localhost/v1/models"); + let settled = false; + const buildPromise = v1ModelsCatalog.getUnifiedModelsResponse(req).then((res) => { + settled = true; + return res; + }); + let lastTick = performance.now(); + let maxGapMs = 0; + let ticks = 0; + while (!settled) { + await new Promise((resolve) => setTimeout(resolve, 0)); + const now = performance.now(); + maxGapMs = Math.max(maxGapMs, now - lastTick); + lastTick = now; + ticks++; + if (ticks > 20000) break; + } + const res = await buildPromise; + assert.equal(res.status, 200); + assert.ok( + maxGapMs < 150, + `event loop was blocked for ${maxGapMs.toFixed(1)}ms in a single stretch while building the ` + + `catalog for ${CONNECTION_COUNT} connections / ${CONNECTION_COUNT * MODELS_PER_CONNECTION} models ` + + `(${ticks} interleaved ticks observed) — the builder is not yielding to the event loop` + ); +}); \ No newline at end of file diff --git a/tests/unit/models-catalog-functional-gateway.test.ts b/tests/unit/models-catalog-functional-gateway.test.ts index 6a3ed07f57..35ddb10cb9 100644 --- a/tests/unit/models-catalog-functional-gateway.test.ts +++ b/tests/unit/models-catalog-functional-gateway.test.ts @@ -24,11 +24,11 @@ function makeRequest(query = ""): Request { return new Request(`http://localhost/v1/models${query}`); } -test("catalog post-filters do not add mirrors when gate off (default)", () => { +test("catalog post-filters do not add mirrors when gate off (default)", async () => { const models = [ { id: "deepseek/deepseek-v4-flash", owned_by: "deepseek", root: "deepseek-v4-flash" }, ]; - const out = applyCatalogPostFilters(makeRequest(), models, { + const out = await applyCatalogPostFilters(makeRequest(), models, { connections: [], prefixMode: "dual", aliasToProviderId: {}, @@ -41,7 +41,7 @@ test("final catalog permission filtering does not let a mirror inherit base acce setFunctionalGatewayProviderSetting("agentrouter", "on"); const models = [{ id: "kmc/k3", owned_by: "kimi-coding", root: "k3" }]; - const withMirror = applyCatalogPostFilters(makeRequest(), models, { + const withMirror = await applyCatalogPostFilters(makeRequest(), models, { connections: [ { id: "conn-1", @@ -77,14 +77,14 @@ test("final catalog permission filtering does not let a mirror inherit base acce ); }); -test("catalog post-filters synthesize a gateway mirror when gate on and gateway has a connection", () => { +test("catalog post-filters synthesize a gateway mirror when gate on and gateway has a connection", async () => { setFeatureFlagOverride(FLAG_KEY, "true"); setFunctionalGatewayProviderSetting("agentrouter", "on"); const models = [ { id: "deepseek/deepseek-v4-flash", owned_by: "deepseek", root: "deepseek-v4-flash" }, ]; - const out = applyCatalogPostFilters(makeRequest(), models, { + const out = await applyCatalogPostFilters(makeRequest(), models, { connections: [ { id: "conn-1",