From 98296169ed55709c32fe0b37cdba53a1a55c4801 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza <8016841+diegosouzapw@users.noreply.github.com> Date: Wed, 24 Jun 2026 08:07:42 -0300 Subject: [PATCH] =?UTF-8?q?feat(quota):=20cap=20per-(key,model)=20?= =?UTF-8?q?=E2=80=94=20quota=5Fallocation=5Fmodel=5Fcaps=20[Fase=203=20#7]?= =?UTF-8?q?=20(#4927)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(quota): cap per-(key,model) com tabela quota_allocation_model_caps [Fase 3 #7] Fecha o buraco onde uma API key pode drenar o pool inteiro consumindo um único modelo. Tabela nova: quota_allocation_model_caps(pool_id, api_key_id, model, cap_value, cap_unit) PK composta (pool_id, api_key_id, model). cap_unit alinhado ao QuotaUnit existente. Comportamento: keyA acima do cap para modelo M → bloqueada somente em M; ainda permitida em qualquer outro modelo no mesmo pool. Cap <= EPSILON → ignorado (seed). Consumo por-(key,model) usa bucket segregado no quota_consumption existente (poolId mangled ':model:') com window fixa 'hourly'; nenhuma nova tabela ou método de store necessário. Módulo novo: src/lib/db/quotaModelCaps.ts (getModelCap/setModelCap/deleteModelCap/listModelCaps) enforce.ts ganha o pre-check em enforceQuotaShare + recording em recordConsumption. EnforceInput e RecordConsumptionInput ganham model?: string (backward-compatible). localDb.ts re-exporta os 4 helpers (Hard Rule #2). TDD: tests/unit/quota-per-key-model.test.ts — 4 cenários (bloqueia em M, permite em M2, sem cap → sem bloqueio, EPSILON → ignorado). Todos os gates de qualidade passam. * feat(quota): plumba model resolvido no hot path para ativar o per-(key,model) cap [Fase 3 #7] A tabela/enforce do commit anterior estavam INERTES: o hot path não passava `model` ao enforce nem ao record, então nenhum model-cap disparava em produção. Plumbagem (model resolvido = mesma var usada no log/roteamento, pós background-redirect/alias): - chatCore.ts: enforceQuotaShare ganha `model`; scheduleQuotaShareConsumption recebe `model`. - chatCore/quotaShareConsumption.ts: threade `model` no RecordConsumptionInput (non-streaming). - spendRecorder.ts: recordStreamingConsumption já recebia `model` — agora o coloca no RecordConsumptionInput (streaming accrue por-modelo). - embeddings.ts: enforce + record ganham `model`. Namespace do cap = id do modelo RESOLVIDO (o mesmo de modelForScope/pendingScope/getUnsupportedParams), não o requestedModel cru nem o finalModelToUpstream (sem prefixo de provider). Operador configura o cap contra esse id. `model || undefined` em todos os pontos: vazio/null → check pulado (fail-safe, zero latência — só um campo no objeto). Teste de integração novo (tests/unit/quota-per-key-model-hotpath.test.ts): prova end-to-end que N consumos via scheduleQuotaShareConsumption({model}) → enforceQuotaShare({model}) bloqueia, e que outro modelo no mesmo pool ainda passa; + guard de que enforce SEM model nunca dispara model-cap. --------- Co-authored-by: Diego Rodrigues de Sa e Souza --- open-sse/handlers/chatCore.ts | 4 + .../chatCore/quotaShareConsumption.ts | 3 + open-sse/handlers/embeddings.ts | 4 + .../106_quota_allocation_model_caps.sql | 32 +++ src/lib/db/quotaModelCaps.ts | 128 +++++++++++ src/lib/localDb.ts | 7 + src/lib/quota/enforce.ts | 89 +++++++- src/lib/quota/spendRecorder.ts | 2 + src/lib/quota/types.ts | 13 ++ .../unit/quota-per-key-model-hotpath.test.ts | 166 +++++++++++++++ tests/unit/quota-per-key-model.test.ts | 201 ++++++++++++++++++ 11 files changed, 646 insertions(+), 3 deletions(-) create mode 100644 src/lib/db/migrations/106_quota_allocation_model_caps.sql create mode 100644 src/lib/db/quotaModelCaps.ts create mode 100644 tests/unit/quota-per-key-model-hotpath.test.ts create mode 100644 tests/unit/quota-per-key-model.test.ts diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 75db44fa95..4cb72ec4c1 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -1927,6 +1927,9 @@ export async function handleChatCore({ apiKeyId: apiKeyInfo.id, connectionId: credentials.connectionId, provider: provider ?? "unknown", + // Resolved model id (post background-redirect / alias) — the same scope the + // router/log use. Operators configure per-(key,model) caps against THIS id. + model: model || undefined, estimatedCost: {}, }).catch((err: unknown) => { log?.warn?.( @@ -3616,6 +3619,7 @@ export async function handleChatCore({ apiKeyId: apiKeyInfo?.id, connectionId: credentials?.connectionId, provider, + model, usage, estimatedCost, log, diff --git a/open-sse/handlers/chatCore/quotaShareConsumption.ts b/open-sse/handlers/chatCore/quotaShareConsumption.ts index e38a413e8b..cb50ea2e22 100644 --- a/open-sse/handlers/chatCore/quotaShareConsumption.ts +++ b/open-sse/handlers/chatCore/quotaShareConsumption.ts @@ -14,6 +14,7 @@ export async function scheduleQuotaShareConsumption(args: { apiKeyId: string | null | undefined; connectionId: string | null | undefined; provider: string | null | undefined; + model?: string | null | undefined; usage: unknown; estimatedCost: number; log?: LoggerLike; @@ -28,6 +29,8 @@ export async function scheduleQuotaShareConsumption(args: { apiKeyId: args.apiKeyId, connectionId: args.connectionId, provider: args.provider ?? "unknown", + // Per-(key,model) cap accounting — same resolved model id used at enforce time. + model: args.model ?? undefined, cost: buildConsumptionCost(args.usage, args.estimatedCost), }, args.log diff --git a/open-sse/handlers/embeddings.ts b/open-sse/handlers/embeddings.ts index 05ecda2d9f..d88d13888d 100644 --- a/open-sse/handlers/embeddings.ts +++ b/open-sse/handlers/embeddings.ts @@ -209,6 +209,8 @@ export async function handleEmbedding({ apiKeyId, connectionId, provider, + // Per-(key,model) cap — resolved embedding model id (same scope used in logs/routing). + model: model || undefined, }); if (quotaDecision.kind === "block") { return { @@ -324,6 +326,8 @@ export async function handleEmbedding({ apiKeyId, connectionId, provider, + // Per-(key,model) cap accounting — same resolved model id used at enforce time. + model: model || undefined, cost: { tokens: data.usage?.prompt_tokens || data.usage?.total_tokens || 0, requests: 1, diff --git a/src/lib/db/migrations/106_quota_allocation_model_caps.sql b/src/lib/db/migrations/106_quota_allocation_model_caps.sql new file mode 100644 index 0000000000..6a96a04f8b --- /dev/null +++ b/src/lib/db/migrations/106_quota_allocation_model_caps.sql @@ -0,0 +1,32 @@ +-- Migration 106: quota_allocation_model_caps +-- +-- Adds per-(pool, api_key, model) budget caps so a single API key cannot drain +-- the shared quota pool by hammering one model (Group B hardening, Fase 3 #7). +-- +-- Schema design: +-- pool_id — references quota_pools.id (no FK cascade; orphan cleanup is +-- app-layer responsibility to avoid 3-table chain fragility in SQLite) +-- api_key_id — the API key being capped (same as in quota_allocations) +-- model — exact model identifier string (e.g. "kimi-k2", "gpt-4o") +-- cap_value — maximum allowed consumption in the given unit per hourly window +-- cap_unit — one of 'percent','requests','tokens','usd' (matches QuotaUnit enum) +-- +-- cap_value CHECK > 0: zero/negative caps are rejected at DB level. +-- The enforce layer additionally skips values ≤ Number.EPSILON (placeholder seeds). +-- +-- Primary key: (pool_id, api_key_id, model) — one cap row per triple. +-- Idempotent: safe to run more than once. +-- +-- Part of: Group B — Quota Sharing Engine, Fase 3 #7. + +CREATE TABLE IF NOT EXISTS quota_allocation_model_caps ( + pool_id TEXT NOT NULL, + api_key_id TEXT NOT NULL, + model TEXT NOT NULL, + cap_value REAL NOT NULL CHECK (cap_value > 0), + cap_unit TEXT NOT NULL CHECK (cap_unit IN ('percent','requests','tokens','usd')), + PRIMARY KEY (pool_id, api_key_id, model) +); + +CREATE INDEX IF NOT EXISTS idx_qamc_pool_key + ON quota_allocation_model_caps(pool_id, api_key_id); diff --git a/src/lib/db/quotaModelCaps.ts b/src/lib/db/quotaModelCaps.ts new file mode 100644 index 0000000000..86306122d2 --- /dev/null +++ b/src/lib/db/quotaModelCaps.ts @@ -0,0 +1,128 @@ +/** + * db/quotaModelCaps.ts — CRUD for quota_allocation_model_caps table. + * + * Per-(pool_id, api_key_id, model) budget caps for the Quota Share Engine. + * Closes the "one key drains the pool on a single model" attack (Fase 3 #7). + * + * cap_unit aligns with QuotaUnit: "requests" | "tokens" | "usd" | "percent". + * cap_value of ≤ Number.EPSILON is treated as a placeholder by the enforce + * layer (not enforced), consistent with the planRegistry EPSILON convention. + * + * All SQL goes through prepared statements — never raw string interpolation + * (Hard Rule #5). + */ + +import { getDbInstance } from "./core"; + +// --------------------------------------------------------------------------- +// Types +// --------------------------------------------------------------------------- + +type QuotaUnit = "percent" | "requests" | "tokens" | "usd"; + +export interface ModelCap { + poolId: string; + apiKeyId: string; + model: string; + capValue: number; + capUnit: QuotaUnit; +} + +interface ModelCapRow { + pool_id: string; + api_key_id: string; + model: string; + cap_value: number; + cap_unit: string; +} + +// --------------------------------------------------------------------------- +// Internal helpers +// --------------------------------------------------------------------------- + +interface StatementLike { + all: (...params: unknown[]) => TRow[]; + get: (...params: unknown[]) => TRow | undefined; + run: (...params: unknown[]) => { changes: number }; +} + +interface DbLike { + prepare: (sql: string) => StatementLike; +} + +function rowToModelCap(row: ModelCapRow): ModelCap { + return { + poolId: row.pool_id, + apiKeyId: row.api_key_id, + model: row.model, + capValue: row.cap_value, + capUnit: row.cap_unit as QuotaUnit, + }; +} + +function getDb(): DbLike { + return getDbInstance() as unknown as DbLike; +} + +// --------------------------------------------------------------------------- +// Public API +// --------------------------------------------------------------------------- + +/** + * Retrieve the cap for a specific (pool, key, model) triple. + * Returns null if no cap is configured. + */ +export function getModelCap(poolId: string, apiKeyId: string, model: string): ModelCap | null { + const row = getDb() + .prepare( + `SELECT pool_id, api_key_id, model, cap_value, cap_unit + FROM quota_allocation_model_caps + WHERE pool_id = ? AND api_key_id = ? AND model = ?` + ) + .get(poolId, apiKeyId, model); + return row ? rowToModelCap(row) : null; +} + +/** + * List all model caps for a given (pool, key) pair. + */ +export function listModelCaps(poolId: string, apiKeyId: string): ModelCap[] { + const rows = getDb() + .prepare( + `SELECT pool_id, api_key_id, model, cap_value, cap_unit + FROM quota_allocation_model_caps + WHERE pool_id = ? AND api_key_id = ?` + ) + .all(poolId, apiKeyId); + return rows.map(rowToModelCap); +} + +/** + * Insert or replace a model cap. + * cap_value must be > 0 (enforced by DB CHECK constraint). + */ +export function setModelCap(cap: ModelCap): void { + getDb() + .prepare( + `INSERT INTO quota_allocation_model_caps + (pool_id, api_key_id, model, cap_value, cap_unit) + VALUES (?, ?, ?, ?, ?) + ON CONFLICT(pool_id, api_key_id, model) DO UPDATE SET + cap_value = excluded.cap_value, + cap_unit = excluded.cap_unit` + ) + .run(cap.poolId, cap.apiKeyId, cap.model, cap.capValue, cap.capUnit); +} + +/** + * Remove the cap for a specific (pool, key, model) triple. + * No-op if it does not exist. + */ +export function deleteModelCap(poolId: string, apiKeyId: string, model: string): void { + getDb() + .prepare( + `DELETE FROM quota_allocation_model_caps + WHERE pool_id = ? AND api_key_id = ? AND model = ?` + ) + .run(poolId, apiKeyId, model); +} diff --git a/src/lib/localDb.ts b/src/lib/localDb.ts index 0979fa4138..2ff7601a6e 100755 --- a/src/lib/localDb.ts +++ b/src/lib/localDb.ts @@ -574,6 +574,13 @@ export { upsertAllocations, listAllocationsForApiKey, } from "./db/quotaPools"; +// Quota per-(key, model) caps — Group B Fase 3 #7 +export { + getModelCap, + listModelCaps, + setModelCap, + deleteModelCap, +} from "./db/quotaModelCaps"; export { // Quota Groups (B2) diff --git a/src/lib/quota/enforce.ts b/src/lib/quota/enforce.ts index 29a81acbd3..3d8d844f5a 100644 --- a/src/lib/quota/enforce.ts +++ b/src/lib/quota/enforce.ts @@ -21,6 +21,7 @@ import { resolvePlan } from "./planResolver"; import { getSaturation } from "./saturationSignals"; import { getQuotaStore } from "./QuotaStore"; import { listAllocationsForApiKey, getPool } from "@/lib/db/quotaPools"; +import { getModelCap } from "@/lib/db/quotaModelCaps"; // --------------------------------------------------------------------------- // Constants @@ -106,8 +107,64 @@ export async function enforceQuotaShare(input: EnforceInput): Promise EPSILON), peek the per-model consumption bucket and + // block ONLY this model if the cap is reached. Other models in the same pool + // remain unaffected — cap is per-model, not global. + // + // Consumption is stored in quota_consumption using a model-scoped dimension key: + // poolId = "${pool.id}:model:${model}" (distinct from pool-level rows) + // unit = cap.capUnit + // window = "hourly" (rate-limiting window; fixed for model caps) + // + // Fail-open per B16: any error reading the cap or peeking the store → skip check. + if (input.model) { + let modelCap: import("@/lib/db/quotaModelCaps").ModelCap | null = null; + try { + modelCap = getModelCap(pool.id, input.apiKeyId, input.model); + } catch { + // DB error — fail-open per B16 + } + if (modelCap && modelCap.capValue > Number.EPSILON) { + const modelBucketPoolId = `${pool.id}:model:${input.model}`; + const modelDimKey = { + poolId: modelBucketPoolId, + unit: modelCap.capUnit, + window: "hourly" as const, + }; + const modelConsumed = await store.peek(input.apiKeyId, modelDimKey).catch(() => 0); + if (modelConsumed >= modelCap.capValue) { + try { + const { notifyWebhookEvent } = await import("@/lib/webhookDispatcher"); + notifyWebhookEvent("quota.exceeded", { + apiKeyId: input.apiKeyId, + provider: input.provider, + connectionId: input.connectionId, + reason: "model-cap", + model: input.model, + }); + } catch { + // webhook is best-effort + } + return { + kind: "block", + reason: `Model cap reached for your API key on ${input.provider}/${input.model} [model-cap]`, + httpStatus: 429, + }; + } + } + } + if (!plan.dimensions.length) { // No dimensions configured → nothing to enforce return { kind: "allow" }; @@ -122,7 +179,6 @@ export async function enforceQuotaShare(input: EnforceInput): Promise Number.EPSILON) { + const cost = costForUnit(input.cost, modelCap.capUnit); + if (cost > 0) { + const modelBucketPoolId = `${poolId}:model:${input.model}`; + const modelDimKey = { + poolId: modelBucketPoolId, + unit: modelCap.capUnit, + window: "hourly" as const, + }; + await store.consume(input.apiKeyId, modelDimKey, cost).catch(() => { + // Fail-open per B29 + }); + } + } + } } // --------------------------------------------------------------------------- diff --git a/src/lib/quota/spendRecorder.ts b/src/lib/quota/spendRecorder.ts index b64c0d0b65..6222ec0920 100644 --- a/src/lib/quota/spendRecorder.ts +++ b/src/lib/quota/spendRecorder.ts @@ -122,6 +122,8 @@ export async function recordStreamingConsumption( apiKeyId, connectionId, provider: resolvedProvider, + // Per-(key,model) cap accounting on streaming traffic — same resolved model id. + model: model || undefined, cost: buildConsumptionCost(streamUsage, estimatedCost), }, deps.log diff --git a/src/lib/quota/types.ts b/src/lib/quota/types.ts index 98c119e2ee..a612549ab7 100644 --- a/src/lib/quota/types.ts +++ b/src/lib/quota/types.ts @@ -66,6 +66,13 @@ export interface EnforceInput { apiKeyId: string; connectionId: string; provider: string; + /** + * Optional model identifier. When present, `enforceQuotaShare` checks for a + * per-(key, model) cap row in `quota_allocation_model_caps` and blocks only + * this model if the cap is reached (Fase 3 #7). Fully backward-compatible: + * callers that do not pass `model` receive unchanged behaviour. + */ + model?: string; estimatedCost?: { tokens?: number; usd?: number; requests?: number }; } @@ -77,5 +84,11 @@ export interface RecordConsumptionInput { apiKeyId: string; connectionId: string; provider: string; + /** + * Optional model identifier. When present, `recordConsumption` also + * increments the per-(key, model) consumption bucket used by the model-cap + * pre-check in `enforceQuotaShare` (Fase 3 #7). Backward-compatible. + */ + model?: string; cost: { tokens?: number; usd?: number; requests?: number }; } diff --git a/tests/unit/quota-per-key-model-hotpath.test.ts b/tests/unit/quota-per-key-model-hotpath.test.ts new file mode 100644 index 0000000000..aa6f84dc33 --- /dev/null +++ b/tests/unit/quota-per-key-model-hotpath.test.ts @@ -0,0 +1,166 @@ +/** + * tests/unit/quota-per-key-model-hotpath.test.ts + * + * Integration test for the per-(key, model) cap END-TO-END through the actual + * hot-path hooks (Fase 3 #7 plumbing). Unlike quota-per-key-model.test.ts (which + * drives recordConsumption/enforceQuotaShare directly), this proves the `model` + * field actually flows through: + * + * scheduleQuotaShareConsumption(...) ← non-streaming POST-hook (chatCore) + * → scheduleRecordConsumption → recordConsumption (model-scoped bucket) + * enforceQuotaShare({ ..., model }) ← PRE-hook the chatCore enforce site uses + * + * Scenario: + * - Configure a cap of N requests for (keyA, modelM). + * - Drive N consumptions through scheduleQuotaShareConsumption({ model: modelM }). + * - enforceQuotaShare({ model: modelM }) → block (the hook plumbed `model`). + * - enforceQuotaShare({ model: modelM2 }) → allow (cap is per-model, other model free). + * + * If the hot-path hook ever drops `model` again (feature goes inert), the block + * assertion fails — guarding the plumbing this PR adds. + * + * Part of: Group B — Quota Sharing Engine, Fase 3 #7. + */ + +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +// ── Single isolated DATA_DIR (same reset pattern as db-quota-pools.test.ts) ─── +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-model-cap-hotpath-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const { createPool, upsertAllocations } = await import("../../src/lib/db/quotaPools.ts"); +const { setModelCap } = await import("../../src/lib/db/quotaModelCaps.ts"); +const { enforceQuotaShare } = await import("../../src/lib/quota/enforce.ts"); +const { resetQuotaStoreSingleton } = await import("../../src/lib/quota/storeFactory.ts"); +const { scheduleQuotaShareConsumption } = await import( + "../../open-sse/handlers/chatCore/quotaShareConsumption.ts" +); + +// ── Fixtures ────────────────────────────────────────────────────────────────── +const CONN_ID = "conn-model-cap-hotpath"; +const PROVIDER = "kimi"; // kimi has {unit:"requests", window:"hourly", limit:1500} in planRegistry +const KEY_A = "key-model-cap-hotpath-a"; +const MODEL_M = "kimi-k2"; +const MODEL_M2 = "kimi-k2-lite"; +const CAP_N = 3; // requests + +async function resetStorage() { + resetQuotaStoreSingleton(); + core.resetDbInstance(); + for (let attempt = 0; attempt < 10; attempt++) { + try { + if (fs.existsSync(TEST_DATA_DIR)) { + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + } + break; + } catch (err: unknown) { + const code = (err as NodeJS.ErrnoException)?.code; + if ((code === "EBUSY" || code === "EPERM") && attempt < 9) { + await new Promise((resolve) => setTimeout(resolve, 50 * (attempt + 1))); + } else { + throw err; + } + } + } + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +test.beforeEach(async () => { + await resetStorage(); +}); + +test.after(async () => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +function makePool() { + const pool = createPool({ connectionId: CONN_ID, name: "Model Cap Hotpath Pool" }); + upsertAllocations(pool.id, [{ apiKeyId: KEY_A, weight: 100, policy: "hard" }]); + return pool; +} + +/** + * Drive ONE consumption through the real non-streaming hot-path hook. + * scheduleQuotaShareConsumption → scheduleRecordConsumption (setImmediate) → + * recordConsumption. We await a macrotask tick so the setImmediate fires. + */ +async function consumeViaHotPath(model: string, requests: number) { + for (let i = 0; i < requests; i++) { + await scheduleQuotaShareConsumption({ + apiKeyId: KEY_A, + connectionId: CONN_ID, + provider: PROVIDER, + model, + // usage with prompt+completion tokens so buildConsumptionCost computes tokens; + // for a "requests" cap the requests:1 field is what matters. + usage: { prompt_tokens: 5, completion_tokens: 5 }, + estimatedCost: 0, + }); + // Let the setImmediate-scheduled recordConsumption run before the next iteration. + await new Promise((r) => setImmediate(r)); + await new Promise((r) => setTimeout(r, 5)); + } +} + +// --------------------------------------------------------------------------- +// End-to-end: cap blocks via the hot-path hook (proves `model` is plumbed) +// --------------------------------------------------------------------------- +test("hot-path: model cap blocks after N consumptions driven through scheduleQuotaShareConsumption", async () => { + const pool = makePool(); + setModelCap({ poolId: pool.id, apiKeyId: KEY_A, model: MODEL_M, capValue: CAP_N, capUnit: "requests" }); + + // Drive CAP_N consumptions through the REAL non-streaming hot-path hook. + await consumeViaHotPath(MODEL_M, CAP_N); + + // The enforce PRE-hook (with model, as chatCore now calls it) must block on model M. + const blocked = await enforceQuotaShare({ + apiKeyId: KEY_A, + connectionId: CONN_ID, + provider: PROVIDER, + model: MODEL_M, + estimatedCost: {}, + }); + assert.equal(blocked.kind, "block", "model M must be blocked after N hot-path consumptions"); + assert.ok( + "reason" in blocked && blocked.reason.includes("model-cap"), + `reason must mention model-cap; got: ${"reason" in blocked ? blocked.reason : "(no reason)"}`, + ); + + // A different model in the SAME pool (no cap) must still be allowed. + const allowedOther = await enforceQuotaShare({ + apiKeyId: KEY_A, + connectionId: CONN_ID, + provider: PROVIDER, + model: MODEL_M2, + estimatedCost: {}, + }); + assert.equal(allowedOther.kind, "allow", "model M2 (no cap) must still be allowed"); +}); + +// --------------------------------------------------------------------------- +// Regression guard: hot-path WITHOUT model on enforce → no model-cap block. +// (If a caller forgets to pass model, the cap simply does not fire — fail-open.) +// --------------------------------------------------------------------------- +test("hot-path: enforce WITHOUT model never triggers model-cap block (fail-safe)", async () => { + const pool = makePool(); + setModelCap({ poolId: pool.id, apiKeyId: KEY_A, model: MODEL_M, capValue: 1, capUnit: "requests" }); + + // Consume via hot path WITH model so the bucket fills. + await consumeViaHotPath(MODEL_M, 2); + + // Enforce WITHOUT model: the model-cap pre-check is skipped entirely. + // (Pool-level fair-share still runs; weight=100, well under fair-share → allow.) + const noModel = await enforceQuotaShare({ + apiKeyId: KEY_A, + connectionId: CONN_ID, + provider: PROVIDER, + estimatedCost: {}, + }); + assert.equal(noModel.kind, "allow", "enforce without model → no model-cap block"); +}); diff --git a/tests/unit/quota-per-key-model.test.ts b/tests/unit/quota-per-key-model.test.ts new file mode 100644 index 0000000000..0ce1970382 --- /dev/null +++ b/tests/unit/quota-per-key-model.test.ts @@ -0,0 +1,201 @@ +/** + * tests/unit/quota-per-key-model.test.ts + * + * TDD for per-(key, model) budget/rate-limit cap (Fase 3 #7). + * + * Scenarios: + * 1. keyA has a cap of N requests for model M; after N uses → enforce blocks keyA on model M. + * 2. keyA blocked on M still allowed on model M2 (no cap / cap not reached) in the same pool. + * 3. No cap configured → behaviour unchanged (no block). + * 4. Cap value ≤ EPSILON → ignored (placeholder skip, consistent with planRegistry pattern). + * + * Uses real SQLite (same single-dir reset pattern as db-quota-pools.test.ts). + * Live enforceQuotaShare + recordConsumption path ensures end-to-end correctness. + * + * Part of: Group B — Quota Sharing Engine, Fase 3 #7. + */ + +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +// ── Single isolated DATA_DIR (same pattern as db-quota-pools.test.ts) ──── +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-model-cap-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +// ── Module imports (after DATA_DIR is set) ──────────────────────────────── +const core = await import("../../src/lib/db/core.ts"); +const { createPool, upsertAllocations } = await import("../../src/lib/db/quotaPools.ts"); +const { setModelCap } = await import("../../src/lib/db/quotaModelCaps.ts"); +const { enforceQuotaShare, recordConsumption } = await import("../../src/lib/quota/enforce.ts"); +const { resetQuotaStoreSingleton } = await import("../../src/lib/quota/storeFactory.ts"); + +// ── Storage reset helper (same as db-quota-pools.test.ts) ──────────────── +async function resetStorage() { + resetQuotaStoreSingleton(); + core.resetDbInstance(); + for (let attempt = 0; attempt < 10; attempt++) { + try { + if (fs.existsSync(TEST_DATA_DIR)) { + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + } + break; + } catch (err: unknown) { + const code = (err as NodeJS.ErrnoException)?.code; + if ((code === "EBUSY" || code === "EPERM") && attempt < 9) { + await new Promise((resolve) => setTimeout(resolve, 50 * (attempt + 1))); + } else { + throw err; + } + } + } + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +// ── Test fixtures ───────────────────────────────────────────────────────── +const CONN_ID = "conn-model-cap-test"; +const PROVIDER = "kimi"; // kimi has {unit:"requests", window:"hourly", limit:1500} in planRegistry +const KEY_A = "key-model-cap-a"; +const MODEL_M = "kimi-k2"; +const MODEL_M2 = "kimi-k2-lite"; +const CAP_N = 3; // requests + +// ── Hooks ───────────────────────────────────────────────────────────────── +test.beforeEach(async () => { + await resetStorage(); +}); + +test.after(async () => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +// ── Helper: create pool with KEY_A allocation ───────────────────────────── +function makePool() { + const pool = createPool({ connectionId: CONN_ID, name: "Model Cap Test Pool" }); + upsertAllocations(pool.id, [{ apiKeyId: KEY_A, weight: 100, policy: "hard" }]); + return pool; +} + +// --------------------------------------------------------------------------- +// Scenario 1: cap N requests on model M → block after N uses +// --------------------------------------------------------------------------- +test("per-(key,model) cap — keyA blocked on model M after N requests", async () => { + const pool = makePool(); + setModelCap({ poolId: pool.id, apiKeyId: KEY_A, model: MODEL_M, capValue: CAP_N, capUnit: "requests" }); + + // Simulate CAP_N prior consumptions + for (let i = 0; i < CAP_N; i++) { + await recordConsumption({ + apiKeyId: KEY_A, + connectionId: CONN_ID, + provider: PROVIDER, + model: MODEL_M, + cost: { requests: 1 }, + }); + } + + const result = await enforceQuotaShare({ + apiKeyId: KEY_A, + connectionId: CONN_ID, + provider: PROVIDER, + model: MODEL_M, + estimatedCost: {}, + }); + + assert.equal(result.kind, "block", "must block when model cap is reached"); + assert.ok( + "reason" in result && result.reason.includes("model-cap"), + `reason must mention model-cap; got: ${"reason" in result ? result.reason : "(no reason)"}`, + ); + assert.equal("httpStatus" in result && result.httpStatus, 429, "must return 429"); +}); + +// --------------------------------------------------------------------------- +// Scenario 2: keyA blocked on M still allowed on M2 +// --------------------------------------------------------------------------- +test("per-(key,model) cap — keyA blocked on M, still allowed on M2 same pool", async () => { + const pool = makePool(); + setModelCap({ poolId: pool.id, apiKeyId: KEY_A, model: MODEL_M, capValue: 1, capUnit: "requests" }); + + // Consume the single request cap on model M + await recordConsumption({ + apiKeyId: KEY_A, + connectionId: CONN_ID, + provider: PROVIDER, + model: MODEL_M, + cost: { requests: 1 }, + }); + + // Model M must be blocked + const resultM = await enforceQuotaShare({ + apiKeyId: KEY_A, + connectionId: CONN_ID, + provider: PROVIDER, + model: MODEL_M, + estimatedCost: {}, + }); + assert.equal(resultM.kind, "block", "model M should be blocked"); + assert.ok( + "reason" in resultM && resultM.reason.includes("model-cap"), + `reason must mention model-cap; got: ${"reason" in resultM ? resultM.reason : "(no reason)"}`, + ); + + // Model M2 (no cap configured) must still be allowed + const resultM2 = await enforceQuotaShare({ + apiKeyId: KEY_A, + connectionId: CONN_ID, + provider: PROVIDER, + model: MODEL_M2, + estimatedCost: {}, + }); + assert.equal(resultM2.kind, "allow", "model M2 should still be allowed (no cap on M2)"); +}); + +// --------------------------------------------------------------------------- +// Scenario 3: no cap configured → behaviour unchanged (allow) +// --------------------------------------------------------------------------- +test("per-(key,model) cap — no cap configured → no block (unchanged behaviour)", async () => { + makePool(); + // No setModelCap call — cap table is empty + + const result = await enforceQuotaShare({ + apiKeyId: KEY_A, + connectionId: CONN_ID, + provider: PROVIDER, + model: MODEL_M, + estimatedCost: {}, + }); + + // With a pool but no model cap, the pool-level fair-share check runs. + // KEY_A has weight=100, 0 consumption → well within fair-share → allow. + assert.equal(result.kind, "allow", "no model cap → no block"); +}); + +// --------------------------------------------------------------------------- +// Scenario 4: cap ≤ EPSILON → ignored (placeholder skip) +// --------------------------------------------------------------------------- +test("per-(key,model) cap — EPSILON cap value → ignored, request allowed", async () => { + const pool = makePool(); + + // Insert a placeholder cap directly (Number.EPSILON > 0 passes DB CHECK constraint + // but enforce.ts skips it: !(capValue > Number.EPSILON) → true for EPSILON). + core.getDbInstance() + .prepare( + `INSERT INTO quota_allocation_model_caps (pool_id, api_key_id, model, cap_value, cap_unit) + VALUES (?, ?, ?, ?, ?)` + ) + .run(pool.id, KEY_A, MODEL_M, Number.EPSILON, "requests"); + + const result = await enforceQuotaShare({ + apiKeyId: KEY_A, + connectionId: CONN_ID, + provider: PROVIDER, + model: MODEL_M, + estimatedCost: {}, + }); + + assert.equal(result.kind, "allow", "EPSILON cap → placeholder → skip → allow"); +});