Files
OmniRoute/tests/unit/translator-responses-cache-usage.test.ts
Praveen K Palaniswamy 65e81158ab fix(ollama): route models by advertised capability (#11088)
Landed with the design call resolved per the owner's pick — **option 1**: the synced store is now endpoint-agnostic (persistDiscoveredModels and managedModelImport no longer drop non-chat models at write time), and chat selectability moved to read time (auto-pool expansion in autoStrategy applies filterChatSelectableModels; the models-route projection already had its chatOnly filter). Your discovery test now passes end-to-end (3/3): /api/show capabilities persist per connection and image/embedding requests route through the advertising host.

Reconciliation notes: conflicted areas merged onto the current tip (adobe discovery import, requestedModel preflight signature, resolvedProvider fast-path coexists with the synced-route override — explicit resolution wins); carried base-red drains (#10055 memoization, #11071 test variants) dropped as already-landed; the managed-model-import exclusion test was propagated to the new contract (image/video models persist; the read filter still hides them from chat pickers — pinned by a new assertion). Full battery: 205/206 focused (the one red is a confirmed periodic-timer timing flake on the loaded devbox — 20/20 isolated), autoCombo vitest 30/30, combo suites 46/46, gates + typecheck clean.

Thank you @yourspraveen — the capability probe + routing design was right; it just needed the store contract opened up. Fixes #11087.
2026-08-23 11:45:01 -03:00

87 lines
2.4 KiB
TypeScript

import test from "node:test";
import assert from "node:assert/strict";
const { openaiResponsesToOpenAIResponse } =
await import("../../open-sse/translator/response/openai-responses.ts");
const { openaiToClaudeResponse } =
await import("../../open-sse/translator/response/openai-to-claude.ts");
function createResponsesState() {
return {
started: false,
finishReasonSent: false,
completedOutputItems: [],
funcArgsBuf: {},
funcNames: {},
funcCallIds: {},
funcArgsDone: {},
funcItemAdded: {},
funcItemDone: {},
};
}
function createClaudeState() {
return {
toolCalls: new Map(),
_pendingXmlToolCalls: [],
_xmlInvokeBuffer: "",
};
}
test("Responses cache usage survives the OpenAI-to-Claude streaming chain", () => {
const openaiChunk = openaiResponsesToOpenAIResponse(
{
type: "response.completed",
response: {
id: "resp-cache-hit",
model: "gpt-5.6-sol",
output: [],
usage: {
input_tokens: 21_023,
input_tokens_details: { cached_tokens: 20_224 },
output_tokens: 5,
},
},
},
createResponsesState()
);
assert.equal(openaiChunk.usage.prompt_tokens, 21_023);
assert.equal(openaiChunk.usage.prompt_tokens_details.cached_tokens, 20_224);
assert.equal(openaiChunk.usage.total_tokens, 21_028);
const events = openaiToClaudeResponse(openaiChunk, createClaudeState());
const messageDelta = events.find((event) => event.type === "message_delta");
assert.deepEqual(messageDelta.usage, {
input_tokens: 799,
output_tokens: 5,
cache_read_input_tokens: 20_224,
});
});
test("Anthropic-style Responses usage still adds cache tokens to total prompt tokens", () => {
const openaiChunk = openaiResponsesToOpenAIResponse(
{
type: "response.completed",
response: {
id: "resp-anthropic-usage",
model: "claude-test",
output: [],
usage: {
input_tokens: 799,
cache_read_input_tokens: 20_224,
cache_creation_input_tokens: 100,
output_tokens: 5,
},
},
},
createResponsesState()
);
assert.equal(openaiChunk.usage.prompt_tokens, 21_123);
assert.equal(openaiChunk.usage.prompt_tokens_details.cached_tokens, 20_224);
assert.equal(openaiChunk.usage.prompt_tokens_details.cache_creation_tokens, 100);
assert.equal(openaiChunk.usage.total_tokens, 21_128);
});