From 00c80fd14ab5af1e0e018bcc5bd295a70e0cc482 Mon Sep 17 00:00:00 2001 From: Dizzle <112548150+maxmad64bis@users.noreply.github.com> Date: Sun, 23 Aug 2026 19:38:57 +0200 Subject: [PATCH] feat(models): surface learned reasoning_effort sets in catalog, variants, and dispatch (#11252) Validated on the combined 12-PR batch board + the resolved merge against the post-#11232 tip: focused suites 73/73 (learned-reasoning-effort-caps, synced-capabilities-learned-effort-override, synced-effort-suffix-learned-validation, effort-tiers-loop-catalog-e2e, reasoning-effort-clamp-and-retry, reasoning-effort-learned-capability) + opencode-plugin effort-tier-variants 4/4, typecheck:core clean, gates within baseline. The stacked-branch conflict after #11232 squash-landed was resolved by hand (the learned-caps module keeps both the Set API and the new model-scoped lookup). The effort_tiers loop is closed end-to-end: catalog advertises exactly what the upstream accepts, and - suffix variants resolve against the learned set. Thank you @maxmad64bis! --- @omniroute/opencode-plugin/package.json | 2 +- @omniroute/opencode-plugin/src/index.ts | 20 +++- .../tests/effort-tier-variants.test.ts | 62 ++++++++++ .../effort-tiers-loop-learned-sets.md | 1 + .../services/learnedReasoningEffortCaps.ts | 32 ++++- open-sse/utils/syncedEffortVariants.ts | 3 +- src/app/api/v1/models/catalog.ts | 14 ++- src/app/api/v1/models/syncedCapabilities.ts | 49 ++++++-- src/sse/services/model.ts | 33 +++++- .../effort-tiers-loop-catalog-e2e.test.ts | 112 ++++++++++++++++++ .../learned-reasoning-effort-caps.test.ts | 62 +++++++++- ...pabilities-learned-effort-override.test.ts | 98 +++++++++++++++ ...d-effort-suffix-learned-validation.test.ts | 80 +++++++++++++ 13 files changed, 541 insertions(+), 27 deletions(-) create mode 100644 @omniroute/opencode-plugin/tests/effort-tier-variants.test.ts create mode 100644 changelog.d/features/effort-tiers-loop-learned-sets.md create mode 100644 tests/unit/effort-tiers-loop-catalog-e2e.test.ts create mode 100644 tests/unit/synced-capabilities-learned-effort-override.test.ts create mode 100644 tests/unit/synced-effort-suffix-learned-validation.test.ts diff --git a/@omniroute/opencode-plugin/package.json b/@omniroute/opencode-plugin/package.json index f096226d79..fa373a83b6 100644 --- a/@omniroute/opencode-plugin/package.json +++ b/@omniroute/opencode-plugin/package.json @@ -23,7 +23,7 @@ "scripts": { "build": "tsup", "clean": "rm -rf dist", - "test": "node --import tsx/esm --test tests/scaffold.test.ts tests/auth.test.ts tests/options-schema.test.ts tests/multi-instance.test.ts tests/fetch-interceptor.test.ts tests/provider.test.ts tests/gemini-sanitize.test.ts tests/combos.test.ts tests/config-shim.test.ts tests/features.test.ts tests/feature-defaults.test.ts tests/usable-combo.test.ts tests/disk-snapshot-perms.test.ts tests/fork-features.test.ts tests/auto-combo-context.test.ts tests/provider-id-routing.test.ts tests/management-read-token.test.ts tests/auto-sync.test.ts tests/model-allowlist.test.ts tests/log-level.test.ts", + "test": "node --import tsx/esm --test tests/scaffold.test.ts tests/auth.test.ts tests/options-schema.test.ts tests/multi-instance.test.ts tests/fetch-interceptor.test.ts tests/provider.test.ts tests/gemini-sanitize.test.ts tests/combos.test.ts tests/config-shim.test.ts tests/features.test.ts tests/feature-defaults.test.ts tests/usable-combo.test.ts tests/disk-snapshot-perms.test.ts tests/fork-features.test.ts tests/auto-combo-context.test.ts tests/provider-id-routing.test.ts tests/management-read-token.test.ts tests/auto-sync.test.ts tests/model-allowlist.test.ts tests/log-level.test.ts tests/effort-tier-variants.test.ts", "prepublishOnly": "npm run clean && npm run build && npm test" }, "keywords": [ diff --git a/@omniroute/opencode-plugin/src/index.ts b/@omniroute/opencode-plugin/src/index.ts index be985361c9..50768e9351 100644 --- a/@omniroute/opencode-plugin/src/index.ts +++ b/@omniroute/opencode-plugin/src/index.ts @@ -1161,6 +1161,8 @@ export interface OmniRouteRawModelEntry { attachment?: boolean; structured_output?: boolean; temperature?: boolean; + /** Runtime-learned or synced reasoning tiers (server-gated, blind-mapped). */ + effort_tiers?: string[]; }; release_date?: string; last_updated?: string; @@ -1302,6 +1304,18 @@ export function mapRawModelToModelV2( ctx: { providerId: string; baseURL: string; apiFormat?: { anthropicPrefixes?: string[] } } ): ModelV2 { const caps = raw.capabilities ?? {}; + // effort_tiers loop: server-declared tiers become ModelV2 variants so the + // UI offers exactly the tiers OmniRoute vouches for (instead of opencode's + // invented [low, medium, high] fallback). Blind: filtering/exclusion rules + // live server-side. Absent/empty/malformed => key omitted ENTIRELY (an + // empty variants object would suppress opencode's fallback for this model). + const declaredTiers = Array.isArray(caps.effort_tiers) + ? caps.effort_tiers.filter((t): t is string => typeof t === "string" && t.length > 0) + : []; + const variants = + declaredTiers.length > 0 + ? Object.fromEntries(declaredTiers.map((tier) => [tier, { reasoningEffort: tier }])) + : undefined; const inMods = new Set(raw.input_modalities ?? ["text"]); const outMods = new Set(raw.output_modalities ?? ["text"]); @@ -1315,10 +1329,7 @@ export function mapRawModelToModelV2( // OpenCode looks up `-m /` as model id `` under // the plugin provider (#10345). Other bare ids still prefix with // `providerId` so credentials resolve as `(omniroute, model)`. - id: - raw.id.includes("/") || raw.owned_by === "combo" - ? raw.id - : `${ctx.providerId}/${raw.id}`, + id: raw.id.includes("/") || raw.owned_by === "combo" ? raw.id : `${ctx.providerId}/${raw.id}`, /** * Display name. Falls back to raw.id when no enrichment is available; * the caller (`createOmniRouteProviderHook`) overlays @@ -1357,6 +1368,7 @@ export function mapRawModelToModelV2( ...(typeof raw.max_input_tokens === "number" ? { input: raw.max_input_tokens } : {}), output: typeof raw.max_output_tokens === "number" ? raw.max_output_tokens : 0, }, + ...(variants ? { variants } : {}), status: "active", options: {}, headers: {}, diff --git a/@omniroute/opencode-plugin/tests/effort-tier-variants.test.ts b/@omniroute/opencode-plugin/tests/effort-tier-variants.test.ts new file mode 100644 index 0000000000..2127ea8109 --- /dev/null +++ b/@omniroute/opencode-plugin/tests/effort-tier-variants.test.ts @@ -0,0 +1,62 @@ +/** + * effort_tiers loop — plugin maps server-declared tiers to ModelV2 variants. + * Blind mapping (I3): no owned_by/provider knowledge here — the SERVER gates + * eligibility (shouldExposeSyncedEffortVariants). Absence semantics (M3): + * no tiers => NO variants key at all (an empty object would also kill + * opencode's own fallback for non-tiered models). + */ +import test from "node:test"; +import assert from "node:assert/strict"; +import { mapRawModelToModelV2, type OmniRouteRawModelEntry } from "../src/index.js"; + +const CTX = { providerId: "omniroute", baseURL: "http://127.0.0.1:20128" } as const; + +test("maps declared tiers to reasoningEffort variants", () => { + const raw: OmniRouteRawModelEntry = { + id: "oc/x-preview-f-free", + owned_by: "opencode", + capabilities: { reasoning: true, effort_tiers: ["low", "high", "max"] }, + }; + const model = mapRawModelToModelV2(raw, { ...CTX }); + const variants = (model as unknown as Record).variants as + Record> | undefined; + assert.ok(variants, "variants key present when tiers declared"); + assert.deepEqual(Object.keys(variants).sort(), ["high", "low", "max"]); + assert.deepEqual(variants.max, { reasoningEffort: "max" }); + assert.deepEqual(variants.low, { reasoningEffort: "low" }); +}); + +test("no tiers => NO variants key (not an empty object)", () => { + const raw: OmniRouteRawModelEntry = { + id: "plain-model", + capabilities: { reasoning: true }, + }; + const model = mapRawModelToModelV2(raw, { ...CTX }) as unknown as Record; + assert.equal("variants" in model, false); +}); + +test("empty or malformed tiers array => NO variants key", () => { + const empty = mapRawModelToModelV2( + { id: "m", capabilities: { effort_tiers: [] } }, + { ...CTX } + ) as unknown as Record; + assert.equal("variants" in empty, false); + + const junk = mapRawModelToModelV2( + { id: "m", capabilities: { effort_tiers: [42, null, "ok"] as unknown as string[] } }, + { ...CTX } + ) as unknown as Record; + const variants = junk.variants as Record> | undefined; + assert.deepEqual(Object.keys(variants ?? {}), ["ok"], "non-string tokens dropped"); +}); + +test("static registry entry WITH tiers also gets variants (N1 blast radius)", () => { + const raw: OmniRouteRawModelEntry = { + id: "some-static-model", + owned_by: "registry", + capabilities: { effort_tiers: ["minimal", "high"] }, + }; + const model = mapRawModelToModelV2(raw, { ...CTX }) as unknown as Record; + const variants = model.variants as Record> | undefined; + assert.deepEqual(Object.keys(variants ?? {}).sort(), ["high", "minimal"]); +}); diff --git a/changelog.d/features/effort-tiers-loop-learned-sets.md b/changelog.d/features/effort-tiers-loop-learned-sets.md new file mode 100644 index 0000000000..29b5b10ec6 --- /dev/null +++ b/changelog.d/features/effort-tiers-loop-learned-sets.md @@ -0,0 +1 @@ +- **feat(catalog):** surface runtime-learned `reasoning_effort` tiers in `/v1/models` `capabilities.effort_tiers` (learned set replaces synced metadata when present), map them to OpenCode `ModelV2.variants` in the OmniRoute plugin, and align dispatch `-` suffix validation to the effective (learned ?? synced) set — so the UI offers exactly the tiers the upstream accepts (e.g. `{low, high, max}` for `oc/x-preview-f-free`) and each advertised variant completes. Excludes codex/glm/kimi, which keep their own dedicated `-{effort}` suffix mechanism and never gain `effort_tiers` from this path (related to #7694, builds on #11232) diff --git a/open-sse/services/learnedReasoningEffortCaps.ts b/open-sse/services/learnedReasoningEffortCaps.ts index ef7e7b5f32..5ed852765c 100644 --- a/open-sse/services/learnedReasoningEffortCaps.ts +++ b/open-sse/services/learnedReasoningEffortCaps.ts @@ -63,6 +63,30 @@ export function getLearnedReasoningEffort( return v ? new Set(v) : null; } +/** + * Model-scoped lookup bridging the key-space gap between executors and the + * catalog: executors record under their CONNECTION id + * (`openai-compatible-chat-:`, cf. compatibleProviderId.ts), + * while the catalog loops on provider ids (`opencode`, …) — an exact + * `${provider}:${model}` lookup would always miss. Scans by model segment + * instead. Multiple connections teaching different sets for the same model + * name intersect (most restrictive proven set wins — conservative across + * connections sharing one catalog entry). + */ +export function getLearnedReasoningEffortForModel( + model: string | null | undefined +): Set | null { + const m = typeof model === "string" ? model.trim().toLowerCase() : ""; + if (!m || learnedCaps.size === 0) return null; + let result: Set | null = null; + for (const [key, value] of learnedCaps) { + const colon = key.indexOf(":"); + if (colon === -1 || key.slice(colon + 1) !== m) continue; + result = result ? new Set([...result].filter((v) => value.has(v))) : new Set(value); + } + return result && result.size > 0 ? result : null; +} + /** * Record that `acceptedValues` is the enum the upstream advertised for * provider+model, and store the accepted set. Returns the stored set, or null @@ -85,7 +109,13 @@ export function recordLearnedReasoningEffort( const lowered = typeof raw === "string" ? raw.trim().toLowerCase() : ""; if (lowered && REASONING_EFFORT_ORDER.includes(lowered)) newSet.add(lowered); } - if (newSet.size === 0) return null; + if (newSet.size === 0) { + // OBS2/M5: a 4xx advertised an enum we cannot map — say so, never learn silently. + console.warn( + `[learnedReasoningEffortCaps] unrecognized reasoning_effort enum for ${key}: ${acceptedValues.join(", ")} — nothing learned` + ); + return null; + } const existing = learnedCaps.get(key); if (existing !== undefined) { diff --git a/open-sse/utils/syncedEffortVariants.ts b/open-sse/utils/syncedEffortVariants.ts index 33c5ec8c56..213128cb38 100644 --- a/open-sse/utils/syncedEffortVariants.ts +++ b/open-sse/utils/syncedEffortVariants.ts @@ -33,7 +33,8 @@ export const SYNCED_EFFORT_SKIP_PROVIDERS = new Set(["codex", "glm", "glm-cn", " /** Provider-id prefixes covering that mechanism's multiple connection variants (kimi-coding, kimi-coding-apikey). */ const SYNCED_EFFORT_SKIP_PROVIDER_PREFIXES = ["kimi"]; -function isSkippedEffortProvider(ownedBy: string): boolean { +/** Whether `ownedBy` already owns its own `-{effort}` suffix mechanism (never synthesize/expose another). */ +export function isSkippedEffortProvider(ownedBy: string): boolean { return ( SYNCED_EFFORT_SKIP_PROVIDERS.has(ownedBy) || SYNCED_EFFORT_SKIP_PROVIDER_PREFIXES.some((prefix) => ownedBy.startsWith(prefix)) diff --git a/src/app/api/v1/models/catalog.ts b/src/app/api/v1/models/catalog.ts index c56a543d56..9a2252d9ee 100644 --- a/src/app/api/v1/models/catalog.ts +++ b/src/app/api/v1/models/catalog.ts @@ -1120,6 +1120,9 @@ async function buildUnifiedModelsResponseCore( else if (endpoints.includes("rerank")) modelType = "rerank"; else if (endpoints.includes("images")) modelType = "image"; else if (endpoints.includes("audio")) modelType = "audio"; + // Same owned_by the alias/canonical entries below will carry — computed once + // so the effort_tiers exclusion (codex/glm/kimi) and the entries agree. + const syncedOwnedBy = resolvePublicOwnerId(providerId, canonicalProviderId); const syncedFields = { ...(modelType ? { type: modelType } : {}), ...(apiFormat !== "chat-completions" ? { api_format: apiFormat } : {}), @@ -1133,12 +1136,19 @@ async function buildUnifiedModelsResponseCore( : {}), // #4264/#7694: vision + reasoning-effort-tier flags captured at sync time, // merged into a single capabilities object (see ./syncedCapabilities.ts). - ...(buildSyncedCapabilities(sm) ? { capabilities: buildSyncedCapabilities(sm) } : {}), + // ownedBy gates effort_tiers off for codex/glm/kimi (own suffix mechanism). + ...(buildSyncedCapabilities(sm, syncedOwnedBy) + ? { capabilities: buildSyncedCapabilities(sm, syncedOwnedBy) } + : {}), }; const existingAliasModel = models.find((model) => model.id === aliasId); if (existingAliasModel) { - const mergedCapabilities = mergeSyncedCapabilities(existingAliasModel.capabilities, sm); + const mergedCapabilities = mergeSyncedCapabilities( + existingAliasModel.capabilities, + sm, + syncedOwnedBy + ); Object.assign(existingAliasModel, syncedFields); if (mergedCapabilities) existingAliasModel.capabilities = mergedCapabilities; continue; diff --git a/src/app/api/v1/models/syncedCapabilities.ts b/src/app/api/v1/models/syncedCapabilities.ts index a42671b331..bf82f0f980 100644 --- a/src/app/api/v1/models/syncedCapabilities.ts +++ b/src/app/api/v1/models/syncedCapabilities.ts @@ -5,26 +5,54 @@ * to keep the vision (#4264) and reasoning-effort-tier (#7694) flags merged into a SINGLE * `capabilities` object rather than two separate spreads that would silently overwrite one * another via object-spread order. A model can be both vision- and reasoning-capable. + * + * effort_tiers loop (2026-08-23): a runtime-learned accepted set (#11232, + * learnedReasoningEffortCaps) REPLACES the synced `supportedThinkingEfforts` + * when one exists — the proven contract beats the advertised one. Lookup is + * model-scoped: executors record under connection ids while this module sees + * provider ids, so an exact provider:model key would always miss. + * + * Exclusion gate: `ownedBy` is REQUIRED and checked against + * `isSkippedEffortProvider` (codex/glm/kimi — providers that already own a + * conflicting `-{effort}` suffix mechanism, see syncedEffortVariants.ts, #7694). + * Without this, the blind opencode-plugin mapping (`capabilities.effort_tiers` + * -> ModelV2 `variants`) would double-handle those providers' native suffix + * ids. `shouldExposeSyncedEffortVariants` gates only the *synthetic* + * `-` catalog entries (open-sse/utils/syncedEffortVariants.ts) — it + * never runs over the base entry's `capabilities`, so it cannot substitute + * for this check. Required (not optional) so no call site can silently skip it. */ +// Use the same canonical alias as catalogModelPolicy.ts (l.1) — a relative path from +// src/app/api/v1/models/ to open-sse/ would need 5 `../` and silently breaks under +// refactors. (Confirmed convention: grep "from \"@omniroute/open-sse" src/app/api/v1/models/) +import { getLearnedReasoningEffortForModel } from "@omniroute/open-sse/services/learnedReasoningEffortCaps.ts"; +import { isSkippedEffortProvider } from "@omniroute/open-sse/utils/syncedEffortVariants.ts"; interface SyncedCapabilityFlags { + id?: string; supportsVision?: boolean; supportedThinkingEfforts?: string[]; } -function hasEffortTiers(sm: SyncedCapabilityFlags): boolean { - return Array.isArray(sm.supportedThinkingEfforts) && sm.supportedThinkingEfforts.length > 0; +function effectiveEffortTiers(sm: SyncedCapabilityFlags, ownedBy: string): string[] | undefined { + if (isSkippedEffortProvider(ownedBy)) return undefined; + const learned = sm.id ? getLearnedReasoningEffortForModel(sm.id) : null; + if (learned) return [...learned]; + return Array.isArray(sm.supportedThinkingEfforts) && sm.supportedThinkingEfforts.length > 0 + ? sm.supportedThinkingEfforts + : undefined; } /** Build the `capabilities` object for a fresh synced-model catalog entry, or `undefined` when neither flag applies. */ export function buildSyncedCapabilities( - sm: SyncedCapabilityFlags + sm: SyncedCapabilityFlags, + ownedBy: string ): Record | undefined { - const effortTiers = hasEffortTiers(sm); - if (!sm.supportsVision && !effortTiers) return undefined; + const tiers = effectiveEffortTiers(sm, ownedBy); + if (!sm.supportsVision && !tiers) return undefined; return { ...(sm.supportsVision ? { vision: true } : {}), - ...(effortTiers ? { effort_tiers: sm.supportedThinkingEfforts! } : {}), + ...(tiers ? { effort_tiers: tiers } : {}), }; } @@ -35,13 +63,14 @@ export function buildSyncedCapabilities( */ export function mergeSyncedCapabilities( existing: Record | undefined, - sm: SyncedCapabilityFlags + sm: SyncedCapabilityFlags, + ownedBy: string ): Record | undefined { - const effortTiers = hasEffortTiers(sm); - if (!sm.supportsVision && !effortTiers && !existing) return undefined; + const tiers = effectiveEffortTiers(sm, ownedBy); + if (!sm.supportsVision && !tiers && !existing) return undefined; return { ...(existing || {}), ...(sm.supportsVision ? { vision: true } : {}), - ...(effortTiers ? { effort_tiers: sm.supportedThinkingEfforts! } : {}), + ...(tiers ? { effort_tiers: tiers } : {}), }; } diff --git a/src/sse/services/model.ts b/src/sse/services/model.ts index 12179740e3..f39404df60 100644 --- a/src/sse/services/model.ts +++ b/src/sse/services/model.ts @@ -16,6 +16,7 @@ import { splitSyncedEffortSuffix, stripContextWindowSuffix, } from "@omniroute/open-sse/services/model.ts"; +import { getLearnedReasoningEffortForModel } from "@omniroute/open-sse/services/learnedReasoningEffortCaps.ts"; import { REGISTRY } from "@omniroute/open-sse/config/providerRegistry.ts"; import { getRegisteredProviderEffortBaseModelId } from "@omniroute/open-sse/utils/registeredEffortVariants.ts"; @@ -124,6 +125,20 @@ function isSyncedEffortSkippedProvider(providerId: string): boolean { return SYNCED_EFFORT_SKIP_PROVIDER_PREFIXES.some((prefix) => providerId.startsWith(prefix)); } +/** + * C1: effective tier set for suffix validation = learned ?? sync. The catalog + * advertises variants from the learned set; validating the suffix against raw + * synced metadata would strand learned-only tiers (dead-on-arrival ids). + */ +function effectiveKnownEfforts( + modelId: string, + syncedEfforts: readonly string[] | null | undefined +): string[] { + const learned = getLearnedReasoningEffortForModel(modelId); + if (learned) return [...learned]; + return Array.isArray(syncedEfforts) ? [...syncedEfforts] : []; +} + /** Resolve a suffix against an explicitly tiered static registry model. */ function resolveRegistryModelIdAndEffort( providerId: string, @@ -139,7 +154,10 @@ function resolveRegistryModelIdAndEffort( for (const candidate of registryModels) { if (!Array.isArray(candidate?.supportedThinkingEfforts)) continue; - const attempt = splitSyncedEffortSuffix(modelId, candidate.supportedThinkingEfforts); + const attempt = splitSyncedEffortSuffix( + modelId, + effectiveKnownEfforts(candidate.id, candidate.supportedThinkingEfforts) + ); if (attempt.effort && attempt.baseModel === candidate.id) { return { modelId: attempt.baseModel, effort: attempt.effort }; } @@ -183,7 +201,7 @@ function resolveSyncedModelIdAndEffort( } const attempt = splitSyncedEffortSuffix( modelId, - candidate.supportedThinkingEfforts as string[] + effectiveKnownEfforts(candidate.id, candidate.supportedThinkingEfforts as string[]) ); if (attempt.effort && attempt.baseModel === candidate.id) { return { modelId: attempt.baseModel, effort: attempt.effort }; @@ -232,7 +250,9 @@ function resolveRuntimeFormats( ): RuntimeModelMeta { const apiFormat = (typeof customMatch?.apiFormat === "string" ? customMatch.apiFormat : undefined) || - (typeof compatOverrideMatch?.apiFormat === "string" ? compatOverrideMatch.apiFormat : undefined) || + (typeof compatOverrideMatch?.apiFormat === "string" + ? compatOverrideMatch.apiFormat + : undefined) || (syncedMatch?.apiFormat === "responses" ? "responses" : undefined); const targetFormat = typeof customMatch?.targetFormat === "string" @@ -359,7 +379,12 @@ async function lookupModelMeta( const available = !liveCatalog.authoritative || Boolean(customMatch || syncedMatch || liveBackedEffortVariant); - const metadata = buildRuntimeModelMeta(customMatch, syncedMatch, registryMatch, compatOverrideMatch); + const metadata = buildRuntimeModelMeta( + customMatch, + syncedMatch, + registryMatch, + compatOverrideMatch + ); if (effort) metadata.resolvedThinkingEffort = effort; return { modelId: resolvedModelId, metadata, available }; diff --git a/tests/unit/effort-tiers-loop-catalog-e2e.test.ts b/tests/unit/effort-tiers-loop-catalog-e2e.test.ts new file mode 100644 index 0000000000..f2fe6f89c6 --- /dev/null +++ b/tests/unit/effort-tiers-loop-catalog-e2e.test.ts @@ -0,0 +1,112 @@ +/** + * effort_tiers loop — I1 end-to-end proof: a set recorded through the REAL + * record path (executor-style connection key) surfaces in the REAL catalog + * response (/api/v1/models), including the learned-only variant entry. + * Never "fix" this test by injecting the same string on both sides. + */ +import { test, after, beforeEach } from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-effort-loop-e2e-")); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.API_KEY_SECRET = process.env.API_KEY_SECRET || "loop-e2e-secret"; + +const core = await import("../../src/lib/db/core.ts"); +const providersDb = await import("../../src/lib/db/providers.ts"); +const modelsDb = await import("../../src/lib/db/models.ts"); +const v1ModelsCatalog = await import("../../src/app/api/v1/models/catalog.ts"); +const { recordLearnedReasoningEffort, __test_resetLearnedReasoningEffortCaps } = + await import("../../open-sse/services/learnedReasoningEffortCaps.ts"); + +async function resetStorage() { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +// Copied verbatim from sync-reasoning-supported-efforts-7694.test.ts +async function seedProviderConnection(provider: string) { + return providersDb.createProviderConnection({ + provider, + authType: "apikey", + name: `${provider}-${Math.random().toString(16).slice(2, 8)}`, + apiKey: `${provider}-key`, + isActive: true, + testStatus: "active", + }); +} + +test.beforeEach(async () => { + __test_resetLearnedReasoningEffortCaps(); + await resetStorage(); +}); + +test.after(async () => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +test("learned set flows end-to-end into /v1/models capabilities and variant entries", async () => { + // Non-namespaced id on purpose: mirrors the real incident model + // (x-preview-f-free) where sm.id === the executor-visible post-strip id. + const MODEL_ID = "loop-model-e2e"; + const connection = await seedProviderConnection("huggingface"); + await modelsDb.replaceSyncedAvailableModelsForConnection("huggingface", connection.id, [ + { + id: MODEL_ID, + name: "Loop Model E2E", + supportedThinkingEfforts: ["none", "low", "medium", "high"], + }, + ]); + + // Simulate the real 400 learning path (base.ts calls exactly this, with the + // executor's CONNECTION id as provider key): + recordLearnedReasoningEffort("openai-compatible-chat-eaff6869", MODEL_ID, ["low", "high", "max"]); + + const response = await v1ModelsCatalog.getUnifiedModelsResponse( + new Request("http://localhost/api/v1/models") + ); + assert.equal(response.status, 200); + const body = (await response.json()) as { + data: Array<{ id: string; capabilities?: { effort_tiers?: string[] } }>; + }; + + const baseEntry = body.data.find((m) => m.id.endsWith(MODEL_ID)); + assert.ok(baseEntry, "base entry present"); + assert.deepEqual(baseEntry!.capabilities?.effort_tiers, ["low", "high", "max"]); + + const maxVariant = body.data.find((m) => m.id === `${baseEntry!.id}-max`); + assert.ok(maxVariant, "learned-only tier synthesized as a variant entry"); +}); + +test("excluded provider (glm) never surfaces effort_tiers, learned or synced", async () => { + const MODEL_ID = "glm-4-flash"; + const connection = await seedProviderConnection("glm"); + await modelsDb.replaceSyncedAvailableModelsForConnection("glm", connection.id, [ + { + id: MODEL_ID, + name: "GLM 4 Flash", + supportedThinkingEfforts: ["none", "low", "medium", "high"], + }, + ]); + recordLearnedReasoningEffort("glm-connection-1", MODEL_ID, ["low", "high"]); + + const response = await v1ModelsCatalog.getUnifiedModelsResponse( + new Request("http://localhost/api/v1/models") + ); + assert.equal(response.status, 200); + const body = (await response.json()) as { + data: Array<{ id: string; capabilities?: { effort_tiers?: string[] } }>; + }; + + const baseEntry = body.data.find((m) => m.id.endsWith(MODEL_ID)); + assert.ok(baseEntry, "base entry present"); + assert.equal( + baseEntry!.capabilities?.effort_tiers, + undefined, + "glm owns its own -{effort} suffix mechanism — the catalog must not also expose effort_tiers" + ); +}); diff --git a/tests/unit/learned-reasoning-effort-caps.test.ts b/tests/unit/learned-reasoning-effort-caps.test.ts index c98eb8fad0..c057351814 100644 --- a/tests/unit/learned-reasoning-effort-caps.test.ts +++ b/tests/unit/learned-reasoning-effort-caps.test.ts @@ -91,7 +91,12 @@ test("records the highest recognized value from the accepted list", () => { ]) as unknown as Set; assert.ok(learned instanceof Set); assert.ok(learned.has("high")); - assert.equal((getLearnedReasoningEffort("ovh", "qwen3-coder-30b-a3b-instruct") as unknown as Set).has("high"), true); + assert.equal( + ( + getLearnedReasoningEffort("ovh", "qwen3-coder-30b-a3b-instruct") as unknown as Set + ).has("high"), + true + ); }); test("returns null and stores nothing when acceptedValues has no recognized token", () => { @@ -116,7 +121,10 @@ test("monotonic decrease: a later, higher accepted-list never ratchets the cap b test("a later, lower accepted-list does ratchet the cap down", () => { recordLearnedReasoningEffort("acme", "model-x", ["none", "low", "medium", "high"]); - const learned = recordLearnedReasoningEffort("acme", "model-x", ["none", "low"]) as unknown as Set; + const learned = recordLearnedReasoningEffort("acme", "model-x", [ + "none", + "low", + ]) as unknown as Set; assert.equal(learned.size, 2); assert.ok(learned.has("low")); assert.equal((getLearnedReasoningEffort("acme", "model-x") as unknown as Set).size, 2); @@ -177,8 +185,12 @@ test("getLearnedReasoningEffort returns null for unknown provider+model", () => test("getLearnedReasoningEffort is keyed case-insensitively on provider+model", () => { recordLearnedReasoningEffort("OVH", "Qwen3-Coder-30B", ["none", "high"]); - assert.ok((getLearnedReasoningEffort("ovh", "qwen3-coder-30b") as unknown as Set).has("high")); - assert.ok((getLearnedReasoningEffort("OVH", "QWEN3-CODER-30B") as unknown as Set).has("high")); + assert.ok( + (getLearnedReasoningEffort("ovh", "qwen3-coder-30b") as unknown as Set).has("high") + ); + assert.ok( + (getLearnedReasoningEffort("OVH", "QWEN3-CODER-30B") as unknown as Set).has("high") + ); }); test("different providers for the same model id have independent caps", () => { @@ -194,3 +206,45 @@ test("handles empty/null provider or model gracefully", () => { assert.equal(recordLearnedReasoningEffort("", "m", ["high"]), null); assert.equal(recordLearnedReasoningEffort("p", "", ["high"]), null); }); + +// ── getLearnedReasoningEffortForModel ──────────────────────────────────────── + +import { getLearnedReasoningEffortForModel } from "../../open-sse/services/learnedReasoningEffortCaps.ts"; + +test("getLearnedReasoningEffortForModel finds a set recorded under any provider key", () => { + recordLearnedReasoningEffort("openai-compatible-chat-eaff6869", "X-Preview-F-Free", [ + "low", + "high", + "max", + ]); + const set = getLearnedReasoningEffortForModel("x-preview-f-free"); + assert.ok(set); + assert.deepEqual([...set].sort(), ["high", "low", "max"]); +}); + +test("getLearnedReasoningEffortForModel intersects when multiple providers disagree", () => { + recordLearnedReasoningEffort("conn-a", "shared-model", ["low", "high", "max"]); + recordLearnedReasoningEffort("conn-b", "shared-model", ["low"]); + const set = getLearnedReasoningEffortForModel("shared-model"); + assert.ok(set); + assert.deepEqual([...set], ["low"]); +}); + +test("getLearnedReasoningEffortForModel returns null when nothing learned or empty model", () => { + assert.equal(getLearnedReasoningEffortForModel("never-learned"), null); + assert.equal(getLearnedReasoningEffortForModel(""), null); + assert.equal(getLearnedReasoningEffortForModel(undefined), null); +}); + +test("recordLearnedReasoningEffort warns when every token is unrecognized", () => { + const warnings: string[] = []; + const orig = console.warn; + console.warn = (msg: string) => warnings.push(msg); + try { + const result = recordLearnedReasoningEffort("p", "m", ["bogus-one", "bogus-two"]); + assert.equal(result, null); + assert.ok(warnings.some((w) => w.includes("reasoning_effort") && w.includes("bogus-one"))); + } finally { + console.warn = orig; + } +}); diff --git a/tests/unit/synced-capabilities-learned-effort-override.test.ts b/tests/unit/synced-capabilities-learned-effort-override.test.ts new file mode 100644 index 0000000000..f33d08dc56 --- /dev/null +++ b/tests/unit/synced-capabilities-learned-effort-override.test.ts @@ -0,0 +1,98 @@ +/** + * effort_tiers loop — learned set overrides synced metadata in catalog + * capabilities (design 2026-08-23, decisions: appris > sync, in-memory). + * Records go through the REAL record path (executor-style connection keys) + * then read back through the catalog builders — proves the key-space bridge, + * unlike a unit injection of the same string on both sides. + */ +import { test, after, beforeEach } from "node:test"; +import assert from "node:assert/strict"; +import { + recordLearnedReasoningEffort, + __test_resetLearnedReasoningEffortCaps, +} from "../../open-sse/services/learnedReasoningEffortCaps.ts"; +import { + buildSyncedCapabilities, + mergeSyncedCapabilities, +} from "../../src/app/api/v1/models/syncedCapabilities.ts"; + +beforeEach(() => __test_resetLearnedReasoningEffortCaps()); +after(() => __test_resetLearnedReasoningEffortCaps()); + +const SYNC_TIERS = ["none", "low", "medium", "high", "xhigh"]; + +test("learned set replaces synced effort_tiers", () => { + recordLearnedReasoningEffort("openai-compatible-chat-eaff6869", "x-preview-f-free", [ + "low", + "high", + "max", + ]); + const caps = buildSyncedCapabilities( + { id: "x-preview-f-free", supportedThinkingEfforts: SYNC_TIERS }, + "huggingface" + ); + assert.deepEqual(caps?.effort_tiers, ["low", "high", "max"]); +}); + +test("nothing learned keeps synced metadata untouched", () => { + const caps = buildSyncedCapabilities( + { id: "some-synced-model", supportedThinkingEfforts: SYNC_TIERS }, + "huggingface" + ); + assert.deepEqual(caps?.effort_tiers, SYNC_TIERS); +}); + +test("neither learned nor synced yields undefined", () => { + const caps = buildSyncedCapabilities({ id: "plain-model" }, "huggingface"); + assert.equal(caps, undefined); +}); + +test("merge path keeps vision AND applies the learned override", () => { + recordLearnedReasoningEffort("conn-a", "vision-model", ["low", "max"]); + const merged = mergeSyncedCapabilities( + { tool_calling: true }, + { id: "vision-model", supportsVision: true, supportedThinkingEfforts: SYNC_TIERS }, + "huggingface" + ); + assert.equal(merged?.vision, true); + assert.equal(merged?.tool_calling, true); + assert.deepEqual(merged?.effort_tiers, ["low", "max"]); +}); + +// Exclusion gate (#7694): codex/glm/kimi already own a conflicting +// `-{effort}` suffix mechanism — the blind opencode-plugin mapping must never +// see effort_tiers for them, learned or synced, or it double-handles the suffix. +for (const ownedBy of ["codex", "glm", "glm-cn", "glmt", "kimi", "kimi-coding-apikey"]) { + test(`build: excluded provider "${ownedBy}" never gets effort_tiers (synced)`, () => { + const caps = buildSyncedCapabilities( + { id: "excluded-model", supportedThinkingEfforts: SYNC_TIERS }, + ownedBy + ); + assert.equal(caps?.effort_tiers, undefined); + }); + + test(`build: excluded provider "${ownedBy}" never gets effort_tiers (learned)`, () => { + recordLearnedReasoningEffort(`conn-${ownedBy}`, "excluded-model", ["low", "max"]); + const caps = buildSyncedCapabilities( + { id: "excluded-model", supportedThinkingEfforts: SYNC_TIERS }, + ownedBy + ); + assert.equal(caps?.effort_tiers, undefined); + }); +} + +test("excluded provider still gets vision through buildSyncedCapabilities", () => { + const caps = buildSyncedCapabilities({ id: "codex-vision-model", supportsVision: true }, "codex"); + assert.deepEqual(caps, { vision: true }); +}); + +test("merge path also excludes codex/glm/kimi from effort_tiers", () => { + recordLearnedReasoningEffort("conn-glm", "glm-model", ["low", "max"]); + const merged = mergeSyncedCapabilities( + { tool_calling: true }, + { id: "glm-model", supportsVision: true, supportedThinkingEfforts: SYNC_TIERS }, + "glm" + ); + assert.equal(merged?.vision, true); + assert.equal(merged?.effort_tiers, undefined); +}); diff --git a/tests/unit/synced-effort-suffix-learned-validation.test.ts b/tests/unit/synced-effort-suffix-learned-validation.test.ts new file mode 100644 index 0000000000..a06ff2d479 --- /dev/null +++ b/tests/unit/synced-effort-suffix-learned-validation.test.ts @@ -0,0 +1,80 @@ +/** + * C1 — the `-` suffix resolver validates against the EFFECTIVE tier set + * (learned ?? sync), not raw synced metadata. Without this, the catalog + * advertises /-max (learned set) but dispatch refuses to strip + * `-max` because sync metadata lacks the tier — dead-on-arrival variant. + * Harness mirrors deepseek-thinking-efforts.test.ts (custom provider + + * persistDiscoveredModels + async getModelInfo). + */ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-c1-effort-dispatch-")); +process.env.DATA_DIR = TEST_DATA_DIR; +process.env.API_KEY_SECRET = process.env.API_KEY_SECRET || "c1-test-secret"; + +const core = await import("../../src/lib/db/core.ts"); +const providersDb = await import("../../src/lib/db/providers.ts"); +const modelDiscovery = await import("../../src/lib/providerModels/modelDiscovery.ts"); +const { getModelInfo } = await import("../../src/sse/services/model.ts"); +const { recordLearnedReasoningEffort, __test_resetLearnedReasoningEffortCaps } = + await import("@omniroute/open-sse/services/learnedReasoningEffortCaps.ts"); + +async function resetStorage() { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +const PROVIDER = "c1prov"; +const MODEL_ID = "c1-model"; + +async function seed() { + const connection = await providersDb.createProviderConnection({ + provider: PROVIDER, + authType: "apikey", + name: "c1-runtime-efforts", + apiKey: `${PROVIDER}-key`, + isActive: true, + testStatus: "active", + }); + // Sync tiers deliberately EXCLUDE max — only the learned set will vouch for it. + await modelDiscovery.persistDiscoveredModels(PROVIDER, connection.id, [ + { id: MODEL_ID, reasoning: { supported_efforts: ["none", "low", "medium", "high"] } }, + ]); +} + +test.beforeEach(async () => { + __test_resetLearnedReasoningEffortCaps(); + await resetStorage(); + await seed(); +}); + +test.after(async () => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +test("-max resolves once the learned set advertises it (sync metadata does not)", async () => { + // Real record path, executor-style CONNECTION key — NOT the provider alias. + recordLearnedReasoningEffort("openai-compatible-chat-eaff6869", MODEL_ID, ["low", "high", "max"]); + const info = await getModelInfo(`${PROVIDER}/${MODEL_ID}-max`); + assert.equal(info.provider, PROVIDER); + assert.equal(info.model, MODEL_ID); + assert.equal(info.resolvedThinkingEffort, "max"); +}); + +test("-medium still resolves via sync tiers even before anything is learned", async () => { + const info = await getModelInfo(`${PROVIDER}/${MODEL_ID}-medium`); + assert.equal(info.model, MODEL_ID); + assert.equal(info.resolvedThinkingEffort, "medium"); +}); + +test("a tier neither learned nor synced is left untouched (literal id)", async () => { + recordLearnedReasoningEffort("conn-a", MODEL_ID, ["low"]); + const info = await getModelInfo(`${PROVIDER}/${MODEL_ID}-ultra`); + assert.equal(info.resolvedThinkingEffort, undefined); +});