From 5de6c5108bff6a0e7ab8a2479f040c21d050ddfc Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E5=B0=8F=E5=A6=8D=E5=84=BF=20=E2=9C=A8?= Date: Fri, 18 Sep 2026 22:32:56 +0800 Subject: [PATCH] fix(usage): preserve nested prompt cache reads before cost calculation (#13760) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(usage): preserve nested prompt cache reads * docs(changelog): add nested cache-read cost fix --------- Co-authored-by: 千乘妍 (Xiaoyaner) Co-authored-by: diegosouzapw <8016841+diegosouzapw@users.noreply.github.com> --- .../fixes/13760-nested-prompt-cache-cost.md | 1 + open-sse/utils/usageTracking.ts | 12 +- tests/unit/cache-read-cost-13746.test.ts | 203 ++++++++++++++++++ tests/unit/usage-extractor.test.ts | 4 +- 4 files changed, 217 insertions(+), 3 deletions(-) create mode 100644 changelog.d/fixes/13760-nested-prompt-cache-cost.md create mode 100644 tests/unit/cache-read-cost-13746.test.ts diff --git a/changelog.d/fixes/13760-nested-prompt-cache-cost.md b/changelog.d/fixes/13760-nested-prompt-cache-cost.md new file mode 100644 index 0000000000..5d8519c643 --- /dev/null +++ b/changelog.d/fixes/13760-nested-prompt-cache-cost.md @@ -0,0 +1 @@ +- **fix(usage):** preserve nested prompt cache-read counters before cost calculation ([#13760](https://github.com/diegosouzapw/OmniRoute/pull/13760)) — fixes [#13746](https://github.com/diegosouzapw/OmniRoute/issues/13746) diff --git a/open-sse/utils/usageTracking.ts b/open-sse/utils/usageTracking.ts index 16339a95af..95cdd63657 100644 --- a/open-sse/utils/usageTracking.ts +++ b/open-sse/utils/usageTracking.ts @@ -479,6 +479,8 @@ function clearCachedTokenDetail(v if (!value || typeof value !== "object" || Array.isArray(value)) return value; const result = { ...value }; if (result.cached_tokens !== undefined) result.cached_tokens = 0; + if (result.cache_creation_tokens !== undefined) result.cache_creation_tokens = 0; + if (result.cache_write_tokens !== undefined) result.cache_write_tokens = 0; return result; } @@ -513,6 +515,8 @@ export function sanitizeProviderUsageForRequest( if (format === FORMATS.CLAUDE) { result.input_tokens = estimatedInput; + result.prompt_tokens_details = clearCachedTokenDetail(result.prompt_tokens_details); + result.input_tokens_details = clearCachedTokenDetail(result.input_tokens_details); result.cache_read_input_tokens = 0; result.cache_creation_input_tokens = 0; return result; @@ -531,6 +535,7 @@ export function sanitizeProviderUsageForRequest( if (format === FORMATS.OPENAI_RESPONSES) { result.input_tokens = estimatedInput; + result.prompt_tokens_details = clearCachedTokenDetail(result.prompt_tokens_details); result.input_tokens_details = clearCachedTokenDetail(result.input_tokens_details); result.cache_read_input_tokens = 0; result.cache_creation_input_tokens = 0; @@ -624,7 +629,12 @@ export function normalizeUsage(usage: UsageLike | null | undefined) { assignNumber("output_tokens", usage?.output_tokens); assignNumber("cache_read_input_tokens", usage?.cache_read_input_tokens); assignNumber("cache_creation_input_tokens", pickCacheCreationTokens(usage)); - assignNumber("cached_tokens", usage?.cached_tokens); + assignNumber( + "cached_tokens", + usage?.cached_tokens ?? + usage?.prompt_tokens_details?.cached_tokens ?? + usage?.input_tokens_details?.cached_tokens + ); assignNumber("no_cache_tokens", usage?.no_cache_tokens); assignNumber("reasoning_tokens", usage?.reasoning_tokens); // xAI's exact provider-reported cost (port of decolua/9router#2453, capability A — diff --git a/tests/unit/cache-read-cost-13746.test.ts b/tests/unit/cache-read-cost-13746.test.ts new file mode 100644 index 0000000000..2324235415 --- /dev/null +++ b/tests/unit/cache-read-cost-13746.test.ts @@ -0,0 +1,203 @@ +/** + * Regression coverage for #13746. + * + * Non-streaming Chat/Responses callers pass provider usage through normalizeUsage() + * before calculateCost(). Nested cache-read counters were dropped at that boundary, + * so the calculator charged the whole prompt at the normal input rate. + * + * Pricing and usage below are synthetic: input=$10/M, cache-read=$1/M, + * output=$50/M; 100,000 input tokens with 99,000 cached tokens must cost $0.109. + */ + +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-cache-read-13746-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const { normalizeUsage, sanitizeProviderUsageForRequest } = + await import("../../open-sse/utils/usageTracking.ts"); +const { createSSEStream } = await import("../../open-sse/utils/stream.ts"); +const { calculateCost } = await import("../../src/lib/usage/costCalculator.ts"); + +const PROVIDER = "synthetic-cache-read-13746"; +const MODEL = "synthetic-cache-read-model-13746"; +const PRICING = { input: 10, cached: 1, output: 50 }; + +function seedSyntheticPricing(): void { + core + .getDbInstance() + .prepare("INSERT OR REPLACE INTO key_value (namespace, key, value) VALUES (?, ?, ?)") + .run("pricing", PROVIDER, JSON.stringify({ [MODEL]: PRICING })); +} + +async function costOf(usage: Record): Promise { + return calculateCost(PROVIDER, MODEL, normalizeUsage(usage)); +} + +async function readSse(chunks: string[], options: Record = {}): Promise { + const source = new ReadableStream({ + start(controller) { + for (const chunk of chunks) controller.enqueue(new TextEncoder().encode(chunk)); + controller.close(); + }, + }); + return new Response( + source.pipeThrough( + createSSEStream({ + mode: "translate", + sourceFormat: "openai", + targetFormat: "openai-responses", + clientResponseFormat: "openai-responses", + provider: PROVIDER, + model: MODEL, + ...options, + }) + ) + ).text(); +} + +function assertCost(actual: number, expected: number): void { + assert.ok(Math.abs(actual - expected) < Number.EPSILON, `${actual} !== ${expected}`); +} + +test.before(() => { + seedSyntheticPricing(); +}); + +test.after(async () => { + await new Promise((resolve) => setTimeout(resolve, 50)); + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +test("implausible usage repair clears nested cache-read details before normalization", () => { + const claude = sanitizeProviderUsageForRequest( + { + input_tokens: 1_000, + cache_read_input_tokens: 336_000, + prompt_tokens_details: { cached_tokens: 336_000, cache_creation_tokens: 12 }, + }, + {}, + "claude" + ); + assert.ok((claude?.input_tokens ?? 0) > 0); + assert.equal(claude?.cache_read_input_tokens, 0); + assert.equal(claude?.prompt_tokens_details?.cached_tokens, 0); + assert.equal(claude?.prompt_tokens_details?.cache_creation_tokens, 0); + + const responses = sanitizeProviderUsageForRequest( + { + input_tokens: 336_000, + input_tokens_details: { cached_tokens: 336_000, cache_creation_tokens: 12 }, + }, + {}, + "openai-responses" + ); + assert.ok((responses?.input_tokens ?? 0) > 0); + assert.equal(responses?.input_tokens_details?.cached_tokens, 0); + assert.equal(responses?.input_tokens_details?.cache_creation_tokens, 0); +}); + +test("Responses SSE metadata uses normalized nested cache-read usage", async () => { + const previous = process.env.OMNIROUTE_SSE_COMMENTS; + process.env.OMNIROUTE_SSE_COMMENTS = "on"; + try { + const output = await readSse([ + `event: response.completed\ndata: ${JSON.stringify({ + type: "response.completed", + response: { + id: "resp-cache-read-13746", + output: [{ type: "message", content: [{ type: "output_text", text: "done" }] }], + usage: { + input_tokens: 100_000, + output_tokens: 0, + input_tokens_details: { cached_tokens: 99_000 }, + }, + }, + })}\n\n`, + ]); + + assert.match(output, /: x-omniroute-response-cost=0\.1090000000/); + } finally { + if (previous === undefined) delete process.env.OMNIROUTE_SSE_COMMENTS; + else process.env.OMNIROUTE_SSE_COMMENTS = previous; + } +}); + +test("Chat nested prompt_tokens_details.cached_tokens reaches cost calculation", async () => { + const usage = normalizeUsage({ + prompt_tokens: 100_000, + completion_tokens: 0, + prompt_tokens_details: { cached_tokens: 99_000 }, + }); + + assert.equal(usage?.cached_tokens, 99_000); + assertCost(await calculateCost(PROVIDER, MODEL, usage), 0.109); +}); + +test("Responses nested input_tokens_details.cached_tokens reaches cost calculation", async () => { + const usage = normalizeUsage({ + input_tokens: 100_000, + output_tokens: 0, + input_tokens_details: { cached_tokens: 99_000 }, + }); + + assert.equal(usage?.cached_tokens, 99_000); + assertCost(await calculateCost(PROVIDER, MODEL, usage), 0.109); +}); + +test("flat cache-read fields retain their existing cost semantics", async () => { + assertCost( + await costOf({ prompt_tokens: 100_000, cached_tokens: 99_000, completion_tokens: 0 }), + 0.109 + ); + assertCost( + await costOf({ input_tokens: 100_000, cache_read_input_tokens: 99_000, output_tokens: 0 }), + 0.109 + ); +}); + +test("an explicit zero cache-read count wins over a nested fallback", async () => { + const usage = normalizeUsage({ + prompt_tokens: 100_000, + prompt_tokens_details: { cached_tokens: 99_000 }, + cached_tokens: 0, + }); + + assert.equal(usage?.cached_tokens, 0); + assert.equal(await calculateCost(PROVIDER, MODEL, usage), 1); +}); + +test("missing or non-finite cache-read fields stay omitted", async () => { + const missing = normalizeUsage({ prompt_tokens: 100_000, completion_tokens: 0 }); + assert.equal(Object.hasOwn(missing ?? {}, "cached_tokens"), false); + assert.equal(Object.hasOwn(missing ?? {}, "cache_read_input_tokens"), false); + assert.equal(await calculateCost(PROVIDER, MODEL, missing), 1); + + const nonFinite = normalizeUsage({ + prompt_tokens: 100_000, + input_tokens_details: { cached_tokens: Number.NaN }, + }); + assert.equal(Object.hasOwn(nonFinite ?? {}, "cached_tokens"), false); + assert.equal(await calculateCost(PROVIDER, MODEL, nonFinite), 1); +}); + +test("nested cache-creation remains compatible with cache-read normalization", async () => { + const usage = normalizeUsage({ + input_tokens: 100_000, + output_tokens: 0, + input_tokens_details: { + cached_tokens: 99_000, + cache_creation_tokens: 5, + }, + }); + + assert.equal(usage?.cached_tokens, 99_000); + assert.equal(usage?.cache_creation_input_tokens, 5); + assertCost(await calculateCost(PROVIDER, MODEL, usage), 0.109); +}); diff --git a/tests/unit/usage-extractor.test.ts b/tests/unit/usage-extractor.test.ts index 58c94cf224..bfd40f854e 100644 --- a/tests/unit/usage-extractor.test.ts +++ b/tests/unit/usage-extractor.test.ts @@ -4,7 +4,7 @@ import assert from "node:assert/strict"; const { extractUsageFromResponse } = await import("../../open-sse/handlers/usageExtractor.ts"); const { extractUsage, normalizeUsage } = await import("../../open-sse/utils/usageTracking.ts"); -test("normalizeUsage keeps only finite numeric fields for stream cost calculation", () => { +test("normalizeUsage keeps finite nested cache-read fields for stream cost calculation", () => { assert.deepEqual( normalizeUsage({ input_tokens: "12", @@ -12,7 +12,7 @@ test("normalizeUsage keeps only finite numeric fields for stream cost calculatio total_tokens: Number.POSITIVE_INFINITY, input_tokens_details: { cached_tokens: 2 }, }), - { input_tokens: 12, output_tokens: 3 } + { input_tokens: 12, output_tokens: 3, cached_tokens: 2 } ); });