diff --git a/changelog.d/fixes/10769-cache-stats-real-cache.md b/changelog.d/fixes/10769-cache-stats-real-cache.md new file mode 100644 index 0000000000..baaacc660d --- /dev/null +++ b/changelog.d/fixes/10769-cache-stats-real-cache.md @@ -0,0 +1 @@ +- **fix(api):** `/api/cache/stats` reported the prompt-cache LRU, which no request path ever writes to — it answered `0 hit / 0 miss, size 0` while the semantic cache served real traffic, and the Health and Usage dashboards rendered that as fact. It now reports the semantic cache's in-memory entries, with the same response shape ([#PRNUM](https://github.com/diegosouzapw/OmniRoute/pull/10769)) — thanks @Poid-ZA, who first fixed this in #9446. diff --git a/src/app/api/cache/stats/route.ts b/src/app/api/cache/stats/route.ts index bdeedb7bec..33f3c447f9 100644 --- a/src/app/api/cache/stats/route.ts +++ b/src/app/api/cache/stats/route.ts @@ -1,5 +1,5 @@ import { NextRequest, NextResponse } from "next/server"; -import { getPromptCache } from "@/lib/cacheLayer"; +import { clearMemoryCache, getMemoryCacheStats } from "@/lib/semanticCache"; import { isAuthenticated } from "@/shared/utils/apiAuth"; import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/error.ts"; @@ -9,9 +9,7 @@ export async function GET(req: NextRequest) { } try { - const cache = getPromptCache(); - const stats = cache.getStats(); - return NextResponse.json(stats); + return NextResponse.json(getMemoryCacheStats()); } catch (error) { return NextResponse.json({ error: sanitizeErrorMessage(error) }, { status: 500 }); } @@ -23,8 +21,7 @@ export async function DELETE(req: NextRequest) { } try { - const cache = getPromptCache(); - cache.clear(); + clearMemoryCache(); return NextResponse.json({ success: true, message: "Cache cleared" }); } catch (error) { return NextResponse.json({ error: sanitizeErrorMessage(error) }, { status: 500 }); diff --git a/src/lib/semanticCache.ts b/src/lib/semanticCache.ts index b185b19963..87ae5c14e5 100644 --- a/src/lib/semanticCache.ts +++ b/src/lib/semanticCache.ts @@ -105,6 +105,22 @@ function getMemoryCache() { return memoryCache; } +/** + * In-memory LRU stats for the semantic cache. + * + * Exposed for `/api/cache/stats`, which used to report `getPromptCache()` — an + * LRU that nothing writes to, so it always answered 0 hit / 0 miss. Same shape + * as `LRUCache.getStats()`, so callers do not have to change. + */ +export function getMemoryCacheStats(): ReturnType { + return getMemoryCache().getStats(); +} + +/** Drop the in-memory LRU without touching the `semantic_cache` table. */ +export function clearMemoryCache(): void { + getMemoryCache().clear(); +} + // ─── Signature Generation ───────────────── /** diff --git a/tests/unit/cache-stats-reports-semantic-cache.test.ts b/tests/unit/cache-stats-reports-semantic-cache.test.ts new file mode 100644 index 0000000000..3c7bade142 --- /dev/null +++ b/tests/unit/cache-stats-reports-semantic-cache.test.ts @@ -0,0 +1,52 @@ +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; + +import { getPromptCache } from "../../src/lib/cacheLayer.ts"; +import { + clearMemoryCache, + getMemoryCacheStats, + setCachedResponse, +} from "../../src/lib/semanticCache.ts"; + +// Regression guard for /api/cache/stats. +// +// The route used to read getPromptCache() — an LRU that no request path writes +// to. It answered "0 hit / 0 miss, size 0" no matter how much traffic the +// semantic cache served, and two dashboard pages rendered that as fact. +// +// The first assertion fails against the old wiring: caching a response fills the +// semantic cache and leaves the prompt cache empty. +describe("cache stats report the cache that requests actually use", () => { + it("counts an entry written through the semantic cache", () => { + clearMemoryCache(); + getPromptCache().clear(); + + const before = getMemoryCacheStats(); + setCachedResponse("sig-cache-stats-guard", "gpt-4.1", { choices: [] }, 42); + const after = getMemoryCacheStats(); + + assert.equal(after.size, before.size + 1); + assert.equal(getPromptCache().getStats().size, 0); + }); + + it("keeps the shape the dashboards read, with a numeric hit rate", () => { + const stats = getMemoryCacheStats(); + + for (const key of ["size", "maxSize", "hits", "misses", "hitRate"]) { + assert.ok(key in stats, `missing ${key}`); + } + // Both dashboard pages call hitRate.toFixed(1); a string would throw there. + assert.equal(typeof stats.hitRate, "number"); + assert.equal(typeof stats.size, "number"); + assert.equal(typeof stats.maxSize, "number"); + }); + + it("clears the in-memory entries", () => { + setCachedResponse("sig-cache-stats-clear", "gpt-4.1", { choices: [] }, 1); + assert.ok(getMemoryCacheStats().size > 0); + + clearMemoryCache(); + + assert.equal(getMemoryCacheStats().size, 0); + }); +});