mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-09-16 03:42:21 +03:00
Landed with the design call resolved per the owner's pick — **option 1**: the synced store is now endpoint-agnostic (persistDiscoveredModels and managedModelImport no longer drop non-chat models at write time), and chat selectability moved to read time (auto-pool expansion in autoStrategy applies filterChatSelectableModels; the models-route projection already had its chatOnly filter). Your discovery test now passes end-to-end (3/3): /api/show capabilities persist per connection and image/embedding requests route through the advertising host. Reconciliation notes: conflicted areas merged onto the current tip (adobe discovery import, requestedModel preflight signature, resolvedProvider fast-path coexists with the synced-route override — explicit resolution wins); carried base-red drains (#10055 memoization, #11071 test variants) dropped as already-landed; the managed-model-import exclusion test was propagated to the new contract (image/video models persist; the read filter still hides them from chat pickers — pinned by a new assertion). Full battery: 205/206 focused (the one red is a confirmed periodic-timer timing flake on the loaded devbox — 20/20 isolated), autoCombo vitest 30/30, combo suites 46/46, gates + typecheck clean. Thank you @yourspraveen — the capability probe + routing design was right; it just needed the store contract opened up. Fixes #11087.
94 lines
3.1 KiB
TypeScript
94 lines
3.1 KiB
TypeScript
import test from "node:test";
|
|
import assert from "node:assert/strict";
|
|
|
|
import { DefaultExecutor } from "../../open-sse/executors/default.ts";
|
|
|
|
// DefaultExecutor.ensureThinkingBudget — max_tokens floor for
|
|
// reasoning models (prevents empty content when the budget is undersized).
|
|
// Previously gated to clinepass only; now applies to all providers (#6912).
|
|
|
|
test("bumps undersized max_tokens to 4096 for a clinepass reasoning model", () => {
|
|
const executor = new DefaultExecutor("clinepass");
|
|
const body = {
|
|
model: "cline-pass/deepseek-v4-pro",
|
|
reasoning_effort: "high",
|
|
max_tokens: 512,
|
|
} as Record<string, unknown>;
|
|
|
|
executor.ensureThinkingBudget(body, "cline-pass/deepseek-v4-pro");
|
|
assert.equal(body.max_tokens, 4096);
|
|
});
|
|
|
|
test("sets max_tokens floor when absent for a reasoning model", () => {
|
|
const executor = new DefaultExecutor("clinepass");
|
|
const body = {
|
|
model: "cline-pass/deepseek-v4-flash",
|
|
reasoning_effort: "medium",
|
|
} as Record<string, unknown>;
|
|
|
|
executor.ensureThinkingBudget(body, "cline-pass/deepseek-v4-flash");
|
|
assert.equal(body.max_tokens, 4096);
|
|
});
|
|
|
|
test("leaves an already-sufficient budget untouched", () => {
|
|
const executor = new DefaultExecutor("clinepass");
|
|
const body = {
|
|
model: "cline-pass/deepseek-v4-pro",
|
|
reasoning_effort: "high",
|
|
max_tokens: 8000,
|
|
} as Record<string, unknown>;
|
|
|
|
executor.ensureThinkingBudget(body, "cline-pass/deepseek-v4-pro");
|
|
assert.equal(body.max_tokens, 8000);
|
|
});
|
|
|
|
test("no-op when reasoning is disabled", () => {
|
|
const executor = new DefaultExecutor("clinepass");
|
|
const body = {
|
|
model: "cline-pass/deepseek-v4-pro",
|
|
max_tokens: 100,
|
|
} as Record<string, unknown>;
|
|
|
|
executor.ensureThinkingBudget(body, "cline-pass/deepseek-v4-pro");
|
|
assert.equal(body.max_tokens, 100);
|
|
});
|
|
|
|
test("applies the floor to GLM-5.2 now that the official catalog marks it reasoning-capable", () => {
|
|
const executor = new DefaultExecutor("clinepass");
|
|
const body = {
|
|
model: "cline-pass/glm-5.2",
|
|
reasoning_effort: "high",
|
|
max_tokens: 100,
|
|
} as Record<string, unknown>;
|
|
|
|
executor.ensureThinkingBudget(body, "cline-pass/glm-5.2");
|
|
assert.equal(body.max_tokens, 4096);
|
|
});
|
|
|
|
test("no-op for an unknown model without reasoning metadata", () => {
|
|
const executor = new DefaultExecutor("clinepass");
|
|
const body = {
|
|
model: "cline-pass/unknown-model",
|
|
reasoning_effort: "high",
|
|
max_tokens: 100,
|
|
} as Record<string, unknown>;
|
|
|
|
executor.ensureThinkingBudget(body, "cline-pass/unknown-model");
|
|
assert.equal(body.max_tokens, 100);
|
|
});
|
|
|
|
test("bumps undersized max_tokens for a non-clinepass reasoning provider (gate removed, #6912)", () => {
|
|
// Issue #6912: ensureThinkingBudget was gated to clinepass only.
|
|
// Now it applies to all providers. Use nvidia (non-clinepass) which has
|
|
// Nemotron Nano with supportsReasoning in the NVIDIA registry.
|
|
const executor = new DefaultExecutor("nvidia");
|
|
const body = {
|
|
model: "nvidia/nvidia-nemotron-nano-9b-v2",
|
|
reasoning_effort: "high",
|
|
max_tokens: 100,
|
|
} as Record<string, unknown>;
|
|
|
|
executor.ensureThinkingBudget(body, "nvidia/nvidia-nemotron-nano-9b-v2");
|
|
assert.equal(body.max_tokens, 4096);
|
|
});
|