mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-09-15 19:32:20 +03:00
Landed with the design call resolved per the owner's pick — **option 1**: the synced store is now endpoint-agnostic (persistDiscoveredModels and managedModelImport no longer drop non-chat models at write time), and chat selectability moved to read time (auto-pool expansion in autoStrategy applies filterChatSelectableModels; the models-route projection already had its chatOnly filter). Your discovery test now passes end-to-end (3/3): /api/show capabilities persist per connection and image/embedding requests route through the advertising host. Reconciliation notes: conflicted areas merged onto the current tip (adobe discovery import, requestedModel preflight signature, resolvedProvider fast-path coexists with the synced-route override — explicit resolution wins); carried base-red drains (#10055 memoization, #11071 test variants) dropped as already-landed; the managed-model-import exclusion test was propagated to the new contract (image/video models persist; the read filter still hides them from chat pickers — pinned by a new assertion). Full battery: 205/206 focused (the one red is a confirmed periodic-timer timing flake on the loaded devbox — 20/20 isolated), autoCombo vitest 30/30, combo suites 46/46, gates + typecheck clean. Thank you @yourspraveen — the capability probe + routing design was right; it just needed the store contract opened up. Fixes #11087.
79 lines
2.3 KiB
TypeScript
79 lines
2.3 KiB
TypeScript
import test from "node:test";
|
|
import assert from "node:assert/strict";
|
|
import { extractUsageFromResponse } from "../../open-sse/handlers/usageExtractor.ts";
|
|
import { extractUsage } from "../../open-sse/utils/usageTracking.ts";
|
|
import { computeCostFromPricing } from "../../src/lib/usage/costCalculator.ts";
|
|
|
|
test("reasoning tokens are not billed twice when reasoning matches the output price", () => {
|
|
const cost = computeCostFromPricing(
|
|
{ input: 1, output: 10, reasoning: 10 },
|
|
{ prompt_tokens: 0, completion_tokens: 1_000, reasoning_tokens: 500 }
|
|
);
|
|
|
|
assert.equal(cost, 0.01);
|
|
});
|
|
|
|
test("reasoning tokens add only the declared premium above the output price", () => {
|
|
const cost = computeCostFromPricing(
|
|
{ input: 1, output: 10, reasoning: 22 },
|
|
{ prompt_tokens: 0, completion_tokens: 1_000, reasoning_tokens: 500 }
|
|
);
|
|
|
|
assert.equal(cost, 0.016);
|
|
});
|
|
|
|
test("reasoning tokens add no cost when the model declares no reasoning price", () => {
|
|
const cost = computeCostFromPricing(
|
|
{ input: 1, output: 10 },
|
|
{ prompt_tokens: 0, completion_tokens: 1_000, reasoning_tokens: 500 }
|
|
);
|
|
|
|
assert.equal(cost, 0.01);
|
|
});
|
|
|
|
test("streaming Gemini usage includes thoughts in completion tokens", () => {
|
|
const usage = extractUsage({
|
|
usageMetadata: {
|
|
promptTokenCount: 100,
|
|
candidatesTokenCount: 40,
|
|
thoughtsTokenCount: 10,
|
|
totalTokenCount: 150,
|
|
},
|
|
});
|
|
|
|
assert.equal(usage.completion_tokens, 50);
|
|
assert.equal(usage.reasoning_tokens, 10);
|
|
});
|
|
|
|
test("non-streaming Gemini usage includes thoughts in completion tokens", () => {
|
|
const usage = extractUsageFromResponse(
|
|
{
|
|
usageMetadata: {
|
|
promptTokenCount: 100,
|
|
candidatesTokenCount: 40,
|
|
thoughtsTokenCount: 10,
|
|
},
|
|
},
|
|
"gemini"
|
|
);
|
|
|
|
assert.equal(usage.completion_tokens, 50);
|
|
assert.equal(usage.reasoning_tokens, 10);
|
|
});
|
|
|
|
test("Gemini usage normalization preserves candidate and thought pricing", () => {
|
|
const usage = extractUsage({
|
|
usageMetadata: {
|
|
promptTokenCount: 100,
|
|
candidatesTokenCount: 40,
|
|
thoughtsTokenCount: 10,
|
|
totalTokenCount: 150,
|
|
},
|
|
});
|
|
|
|
const cost = computeCostFromPricing({ input: 1, output: 4, reasoning: 10 }, usage);
|
|
|
|
const expected = (100 * 1 + 40 * 4 + 10 * 10) / 1_000_000;
|
|
assert.ok(Math.abs(cost - expected) < 1e-12);
|
|
});
|