diff --git a/changelog.d/fixes/12885-semantic-cache-no-truncated-writes.md b/changelog.d/fixes/12885-semantic-cache-no-truncated-writes.md new file mode 100644 index 0000000000..ad643d50bd --- /dev/null +++ b/changelog.d/fixes/12885-semantic-cache-no-truncated-writes.md @@ -0,0 +1 @@ +- fix(cache): never write a truncated completion (`finish_reason: "length"`/`max_tokens`) into the semantic cache — a partial answer cached under a temperature:0 signature was served to every later identical request, permanently returning a mid-sentence reply that no retry cleared (#12885) diff --git a/open-sse/handlers/chatCore/semanticCacheStore.ts b/open-sse/handlers/chatCore/semanticCacheStore.ts index 4c5f44c622..00111eef88 100644 --- a/open-sse/handlers/chatCore/semanticCacheStore.ts +++ b/open-sse/handlers/chatCore/semanticCacheStore.ts @@ -12,6 +12,7 @@ import { generateSignature as defaultGenerateSignature, setCachedResponse as defaultSetCachedResponse, isCacheableForWrite as defaultIsCacheableForWrite, + isTruncatedCompletion as defaultIsTruncatedCompletion, } from "@/lib/semanticCache"; import { isSmallEnoughForSemanticCache as defaultIsSmallEnough } from "../../utils/estimateSize.ts"; import { getSemanticCacheManager } from "../../services/cache/semanticCacheManager.ts"; @@ -32,6 +33,8 @@ type UsageLike = { prompt_tokens?: number; completion_tokens?: number } | null | export interface SemanticCacheStoreDeps { isCacheableForWrite: typeof defaultIsCacheableForWrite; + /** Optional so pre-existing callers/tests with partial deps keep working. */ + isTruncatedCompletion?: typeof defaultIsTruncatedCompletion; isSmallEnoughForSemanticCache: typeof defaultIsSmallEnough; generateSignature: typeof defaultGenerateSignature; setCachedResponse: typeof defaultSetCachedResponse; @@ -39,6 +42,7 @@ export interface SemanticCacheStoreDeps { const DEFAULT_DEPS: SemanticCacheStoreDeps = { isCacheableForWrite: defaultIsCacheableForWrite, + isTruncatedCompletion: defaultIsTruncatedCompletion, isSmallEnoughForSemanticCache: defaultIsSmallEnough, generateSignature: defaultGenerateSignature, setCachedResponse: defaultSetCachedResponse, @@ -61,6 +65,7 @@ export function storeSemanticCacheResponse( if ( !args.enabled || !deps.isCacheableForWrite(args.body, args.headers) || + (deps.isTruncatedCompletion ?? defaultIsTruncatedCompletion)(args.translatedResponse) || !deps.isSmallEnoughForSemanticCache(args.translatedResponse) ) { return; diff --git a/open-sse/handlers/chatCore/streamingSemanticCacheStore.ts b/open-sse/handlers/chatCore/streamingSemanticCacheStore.ts index 58f6df5b96..26832ba7d6 100644 --- a/open-sse/handlers/chatCore/streamingSemanticCacheStore.ts +++ b/open-sse/handlers/chatCore/streamingSemanticCacheStore.ts @@ -13,6 +13,7 @@ import { generateSignature as defaultGenerateSignature, setCachedResponse as defaultSetCachedResponse, isCacheableForWrite as defaultIsCacheableForWrite, + isTruncatedCompletion as defaultIsTruncatedCompletion, } from "@/lib/semanticCache"; import { isSmallEnoughForSemanticCache as defaultIsSmallEnough } from "../../utils/estimateSize.ts"; import { getSemanticCacheManager } from "../../services/cache/semanticCacheManager.ts"; @@ -31,6 +32,8 @@ type CacheBody = { export interface StreamingSemanticCacheStoreDeps { isCacheableForWrite: typeof defaultIsCacheableForWrite; + /** Optional so pre-existing callers/tests with partial deps keep working. */ + isTruncatedCompletion?: typeof defaultIsTruncatedCompletion; isSmallEnoughForSemanticCache: typeof defaultIsSmallEnough; generateSignature: typeof defaultGenerateSignature; setCachedResponse: typeof defaultSetCachedResponse; @@ -38,6 +41,7 @@ export interface StreamingSemanticCacheStoreDeps { const DEFAULT_DEPS: StreamingSemanticCacheStoreDeps = { isCacheableForWrite: defaultIsCacheableForWrite, + isTruncatedCompletion: defaultIsTruncatedCompletion, isSmallEnoughForSemanticCache: defaultIsSmallEnough, generateSignature: defaultGenerateSignature, setCachedResponse: defaultSetCachedResponse, @@ -113,7 +117,8 @@ export function storeStreamingSemanticCacheResponse( !args.enabled || args.streamStatus !== 200 || !args.streamResponseBody || - !deps.isCacheableForWrite(args.body, args.headers) + !deps.isCacheableForWrite(args.body, args.headers) || + (deps.isTruncatedCompletion ?? defaultIsTruncatedCompletion)(args.streamResponseBody) ) { return; } diff --git a/src/lib/semanticCache.ts b/src/lib/semanticCache.ts index 03fe8938ed..3ff0d8e978 100644 --- a/src/lib/semanticCache.ts +++ b/src/lib/semanticCache.ts @@ -469,3 +469,56 @@ export function isCacheableForWrite(body, headers) { if (body.temperature !== 0) return false; return true; } + +/** + * A response cut short by the output-token ceiling is a partial answer, not a + * reusable one. Caching it under a temperature:0 signature pins the truncation + * for every later identical request — the caller sees a mid-sentence reply that + * no retry clears, because each retry is served the same poisoned entry. + * + * Only `length` (and its Claude-side spelling `max_tokens`) is treated as + * truncation. `stop`, `tool_calls`, and a missing/unknown reason are complete + * responses and stay cacheable, so this never narrows the cache beyond the bug. + */ +const TRUNCATED_FINISH_REASONS = new Set(["length", "max_tokens"]); + +export function isTruncatedCompletion(response: unknown): boolean { + if (!response || typeof response !== "object") return false; + const r = response as { + choices?: Array<{ finish_reason?: unknown }>; + stop_reason?: unknown; + }; + if (Array.isArray(r.choices)) { + for (const choice of r.choices) { + const reason = choice?.finish_reason; + if (typeof reason === "string" && TRUNCATED_FINISH_REASONS.has(reason)) return true; + } + } + // Claude-format responses carry the reason at the top level instead. + if (typeof r.stop_reason === "string" && TRUNCATED_FINISH_REASONS.has(r.stop_reason)) { + return true; + } + return false; +} + +/** + * Streaming variant: the assembled SSE body is scanned for a truncating + * finish_reason. Parsing is intentionally tolerant — an unparseable chunk is + * treated as "not known to be truncated" so a malformed frame never silently + * disables caching. + */ +export function isTruncatedStreamBody(streamBody: unknown): boolean { + if (typeof streamBody !== "string" || streamBody.length === 0) return false; + for (const line of streamBody.split("\n")) { + const trimmed = line.trim(); + if (!trimmed.startsWith("data:")) continue; + const payload = trimmed.slice(5).trim(); + if (!payload || payload === "[DONE]") continue; + try { + if (isTruncatedCompletion(JSON.parse(payload))) return true; + } catch { + // Non-JSON frame — ignore. + } + } + return false; +} diff --git a/tests/unit/semantic-cache-no-truncated-writes.test.ts b/tests/unit/semantic-cache-no-truncated-writes.test.ts new file mode 100644 index 0000000000..47750e4107 --- /dev/null +++ b/tests/unit/semantic-cache-no-truncated-writes.test.ts @@ -0,0 +1,101 @@ +// Regression: a truncated upstream response (finish_reason "length") must never be +// written to the semantic cache. A partial completion cached under a temperature:0 +// signature is served to every subsequent identical request, permanently returning +// a mid-sentence answer that no retry can clear (only a cache flush). +// Observed live on OmniRoute against github/gemini-3.5-flash: temperature:0 returned +// finish_reason "length" at 93 completion tokens on every call, while the same request +// with x-omniroute-no-cache:true returned a complete 239-token response. (#12885) +import { test } from "node:test"; +import assert from "node:assert/strict"; + +const { storeSemanticCacheResponse } = + await import("../../open-sse/handlers/chatCore/semanticCacheStore.ts"); +const { storeStreamingSemanticCacheResponse } = + await import("../../open-sse/handlers/chatCore/streamingSemanticCacheStore.ts"); +// Real truncation predicate — only the cache backend and the temperature/size +// gates are stubbed, so these tests exercise the actual detection logic. +const { isTruncatedCompletion } = await import("../../src/lib/semanticCache.ts"); + +function deps(stored: unknown[]) { + return { + isCacheableForWrite: () => true, + isTruncatedCompletion, + isSmallEnoughForSemanticCache: () => true, + generateSignature: () => "sig", + setCachedResponse: (_s: unknown, _m: string, r: unknown, t: number) => stored.push({ r, t }), + } as never; +} + +test("does not cache a non-streaming response truncated by max_tokens", () => { + const stored: unknown[] = []; + storeSemanticCacheResponse( + { + enabled: true, + body: { messages: [{ role: "user", content: "hi" }], temperature: 0 }, + headers: undefined, + translatedResponse: { + choices: [{ finish_reason: "length", message: { content: "partial answ" } }], + }, + model: "gemini-3.5-flash", + usage: { prompt_tokens: 10, completion_tokens: 93 }, + }, + deps(stored) + ); + assert.equal(stored.length, 0, "truncated response must not be cached"); +}); + +test("still caches a complete non-streaming response", () => { + const stored: unknown[] = []; + storeSemanticCacheResponse( + { + enabled: true, + body: { messages: [{ role: "user", content: "hi" }], temperature: 0 }, + headers: undefined, + translatedResponse: { + choices: [{ finish_reason: "stop", message: { content: "complete answer" } }], + }, + model: "gemini-3.5-flash", + usage: { prompt_tokens: 10, completion_tokens: 239 }, + }, + deps(stored) + ); + assert.equal(stored.length, 1, "complete response must still be cached"); +}); + +test("does not cache a streaming response truncated by max_tokens", () => { + const stored: unknown[] = []; + storeStreamingSemanticCacheResponse( + { + enabled: true, + streamStatus: 200, + streamResponseBody: { + choices: [{ finish_reason: "length", message: { content: "partial" } }], + }, + body: { messages: [{ role: "user", content: "hi" }], temperature: 0 }, + headers: undefined, + model: "gemini-3.5-flash", + streamUsage: { prompt_tokens: 10, completion_tokens: 93 }, + }, + deps(stored) + ); + assert.equal(stored.length, 0, "truncated streaming response must not be cached"); +}); + +test("still caches a complete streaming response", () => { + const stored: unknown[] = []; + storeStreamingSemanticCacheResponse( + { + enabled: true, + streamStatus: 200, + streamResponseBody: { + choices: [{ finish_reason: "stop", message: { content: "complete" } }], + }, + body: { messages: [{ role: "user", content: "hi" }], temperature: 0 }, + headers: undefined, + model: "gemini-3.5-flash", + streamUsage: { prompt_tokens: 10, completion_tokens: 239 }, + }, + deps(stored) + ); + assert.equal(stored.length, 1, "complete streaming response must still be cached"); +});