mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-09-14 02:42:24 +03:00
Landed with the design call resolved per the owner's pick — **option 1**: the synced store is now endpoint-agnostic (persistDiscoveredModels and managedModelImport no longer drop non-chat models at write time), and chat selectability moved to read time (auto-pool expansion in autoStrategy applies filterChatSelectableModels; the models-route projection already had its chatOnly filter). Your discovery test now passes end-to-end (3/3): /api/show capabilities persist per connection and image/embedding requests route through the advertising host. Reconciliation notes: conflicted areas merged onto the current tip (adobe discovery import, requestedModel preflight signature, resolvedProvider fast-path coexists with the synced-route override — explicit resolution wins); carried base-red drains (#10055 memoization, #11071 test variants) dropped as already-landed; the managed-model-import exclusion test was propagated to the new contract (image/video models persist; the read filter still hides them from chat pickers — pinned by a new assertion). Full battery: 205/206 focused (the one red is a confirmed periodic-timer timing flake on the loaded devbox — 20/20 isolated), autoCombo vitest 30/30, combo suites 46/46, gates + typecheck clean. Thank you @yourspraveen — the capability probe + routing design was right; it just needed the store contract opened up. Fixes #11087.
80 lines
2.3 KiB
TypeScript
80 lines
2.3 KiB
TypeScript
import assert from "node:assert/strict";
|
|
import test from "node:test";
|
|
|
|
const { isLocalQueueCapacityErrorBody, isRequestScopedUpstreamFailure, shouldSkipConnDisable } =
|
|
await import("../../open-sse/services/combo/comboPredicates.ts");
|
|
const { handleComboChat } = await import("../../open-sse/services/combo.ts");
|
|
|
|
function localQueueResponse() {
|
|
return new Response(
|
|
JSON.stringify({
|
|
error: {
|
|
message: "Local rate-limit queue wait exceeded",
|
|
code: "RATE_LIMIT_QUEUE_TIMEOUT",
|
|
type: "local_queue_capacity",
|
|
},
|
|
}),
|
|
{ status: 429, headers: { "content-type": "application/json" } }
|
|
);
|
|
}
|
|
|
|
function createLog() {
|
|
return { info: () => {}, warn: () => {}, error: () => {}, debug: () => {} };
|
|
}
|
|
|
|
test("local queue capacity is request-scoped and never disables a healthy connection", () => {
|
|
assert.equal(
|
|
isRequestScopedUpstreamFailure({
|
|
code: "RATE_LIMIT_QUEUE_TIMEOUT",
|
|
type: "local_queue_capacity",
|
|
}),
|
|
true
|
|
);
|
|
assert.equal(
|
|
shouldSkipConnDisable(
|
|
{
|
|
status: 429,
|
|
errorCode: "RATE_LIMIT_QUEUE_TIMEOUT",
|
|
errorType: "local_queue_capacity",
|
|
},
|
|
false,
|
|
false,
|
|
"nvidia"
|
|
),
|
|
true
|
|
);
|
|
assert.equal(
|
|
isLocalQueueCapacityErrorBody({
|
|
error: { code: "RATE_LIMIT_QUEUE_TIMEOUT", type: "local_queue_capacity" },
|
|
}),
|
|
true
|
|
);
|
|
});
|
|
|
|
test("combo returns a local queue capacity response without retrying or rotating targets", async () => {
|
|
let calls = 0;
|
|
const response = await handleComboChat({
|
|
body: { model: "openai/gpt-4" },
|
|
combo: {
|
|
name: `local-queue-capacity-${Math.random().toString(16).slice(2)}`,
|
|
strategy: "priority",
|
|
models: ["openai/gpt-4", "openai/gpt-4o-mini"],
|
|
config: { maxRetries: 2, maxSetRetries: 1, retryDelayMs: 0, fallbackDelayMs: 0 },
|
|
},
|
|
handleSingleModel: async () => {
|
|
calls += 1;
|
|
return localQueueResponse();
|
|
},
|
|
isModelAvailable: async () => true,
|
|
log: createLog() as never,
|
|
settings: {},
|
|
allCombos: null,
|
|
});
|
|
|
|
assert.equal(response.status, 429);
|
|
assert.equal(calls, 1, "a local queue limit must not amplify into retries or fallback calls");
|
|
const body = await response.json();
|
|
assert.equal(body.error.code, "RATE_LIMIT_QUEUE_TIMEOUT");
|
|
assert.equal(body.error.type, "local_queue_capacity");
|
|
});
|