From ed0638c0f13b6cebd14a8a6e6ecd71136559eda3 Mon Sep 17 00:00:00 2001 From: NOXX - Commiter Date: Sun, 14 Jun 2026 16:28:48 +0300 Subject: [PATCH] feat(gemini/vertex): surface Veo video models in dynamic discovery (#3839) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Gemini / Vertex / Vertex AI Express already discover their catalog dynamically from v1beta/models, but video (Veo) models use predictLongRunning, which was not mapped — so they never surfaced. parseGeminiModelsList now recognizes predictLongRunning and exposes Veo video models alongside chat/image/embedding/audio. Integrated into release/v3.8.25. Co-authored-by: diegosouzapw --- file-size-baseline.json | 5 +- src/lib/providerModels/geminiModelsParser.ts | 61 ++++++++++++++------ src/shared/validation/schemas.ts | 1 + tests/unit/gemini-models-parser.test.ts | 26 +++++++++ 4 files changed, 72 insertions(+), 21 deletions(-) diff --git a/file-size-baseline.json b/file-size-baseline.json index a2b4da4fc3..687a10eaa8 100644 --- a/file-size-baseline.json +++ b/file-size-baseline.json @@ -101,7 +101,7 @@ "src/shared/constants/providers.ts": 3147, "src/shared/constants/sidebarVisibility.ts": 1006, "src/shared/services/cliRuntime.ts": 1090, - "src/shared/validation/schemas.ts": 2522, + "src/shared/validation/schemas.ts": 2523, "src/sse/handlers/chat.ts": 1425, "src/sse/services/auth.ts": 2207 }, @@ -126,5 +126,6 @@ "_rebaseline_2026_06_14_3825_combo_stickiness": "Re-baseline #3825 (sessionless combo stickiness + reasoning-aware readiness): combo.ts 5164→5198 (+34, pós-prettier). Crescimento = deriveComboSessionKey() + effectiveSessionId threading nos sites de read/write do pin server-side. streamReadinessPolicy.ts não-frozen (sob cap). Lógica coesa no handler de combo; não-extraível.", "_rebaseline_2026_06_14_r3_3835_kiro_pricing": "PR #3835 own growth: pricing.ts 1470→1508 (+38 = missing Kiro pricing rows, claude-sonnet-4.6 etc., pure data). Also carries inherited release/v3.8.25 drift not yet frozen by prior r2 merges: RequestLoggerV2.tsx 1276→1282 (#3820 resizable log table) and combo.ts 5198→5203 (#3811 round-robin replay-response fix). This PR does not touch RequestLoggerV2/combo.ts source; updating the frozen values restores Fast Quality Gates on the current release branch.", "_rebaseline_2026_06_14_r3_3849_transient_hide": "PR #3849 own growth: providerPageHelpers.ts 939→955 (+16 = expanded JSDoc on the auto-hide policy + transient-failure guard in evaluateTestAllEntry). Cohesive helper logic; not extractable.", - "_rebaseline_2026_06_14_r3_3838_opencode_quota": "PR #3838 own growth: usage.ts 3394→3408 (+14 = clearer OpenCode Go missing-quota-API messages with OMNIROUTE_OPENCODE_GO_QUOTA_URL override hint + upstream issue links). Message text only; no logic extractable." + "_rebaseline_2026_06_14_r3_3838_opencode_quota": "PR #3838 own growth: usage.ts 3394→3408 (+14 = clearer OpenCode Go missing-quota-API messages with OMNIROUTE_OPENCODE_GO_QUOTA_URL override hint + upstream issue links). Message text only; no logic extractable.", + "_rebaseline_2026_06_14_r3_3839_veo_video": "PR #3839 own growth: schemas.ts 2522→2523 (+1 = Veo video model (predictLongRunning) validation for Gemini/Vertex dynamic discovery)." } diff --git a/src/lib/providerModels/geminiModelsParser.ts b/src/lib/providerModels/geminiModelsParser.ts index cf40da47d2..e4c8063d0b 100644 --- a/src/lib/providerModels/geminiModelsParser.ts +++ b/src/lib/providerModels/geminiModelsParser.ts @@ -3,20 +3,30 @@ * * Each model's `supportedGenerationMethods` is mapped to OmniRoute endpoints: * - generateContent / generateAnswer → "chat" - * - predict / predictLongRunning → "images" (Imagen models) - * - embedContent → "embeddings" - * - bidiGenerateContent → "audio" + * - predict → "images" (Imagen image generation) + * - predictLongRunning → "video" (Veo video generation) + * - embedContent → "embeddings" + * - bidiGenerateContent → "audio" (Live real-time audio) * - * This is shared by the `gemini` discovery config and the `vertex` discovery branch: Vertex AI - * Express keys (and Service Account JSON via a minted OAuth token) list models from the same - * endpoint, so image models (imagen-*, gemini-*-image) surface dynamically instead of being - * limited to the small static registry list. + * Model-id heuristics refine the long-running bucket because Google exposes both + * Imagen and Veo via long-running methods on the same endpoint: + * - id contains "veo" → ensure "video" + * - id contains "imagen" → force "images" (never "video") + * + * Note: `gemini-*-image` models (e.g. gemini-3-pro-image) generate images via the + * regular `generateContent` path, so they stay "chat" (image output is a chat + * modality) and are intentionally NOT reclassified as "images". + * + * This is shared by the `gemini` discovery config and the `vertex` / + * `vertex-partner` (incl. Vertex AI Express key) discovery branches, so every + * model the account can access — chat, image, video, audio and embeddings — + * surfaces dynamically instead of being limited to the small static registry. */ const METHOD_TO_ENDPOINT: Record = { generateContent: "chat", embedContent: "embeddings", predict: "images", - predictLongRunning: "images", + predictLongRunning: "video", bidiGenerateContent: "audio", generateAnswer: "chat", }; @@ -45,20 +55,33 @@ export function parseGeminiModelsList(data: any): GeminiDiscoveryModel[] { const methods: string[] = Array.isArray(m.supportedGenerationMethods) ? (m.supportedGenerationMethods as string[]) : []; - const endpoints = [ - ...new Set( - methods - .filter((method) => !IGNORED_METHODS.has(method)) - .map((method) => METHOD_TO_ENDPOINT[method] || "chat") - ), - ]; - if (endpoints.length === 0) endpoints.push("chat"); + + const endpoints = new Set( + methods + .filter((method) => !IGNORED_METHODS.has(method)) + .map((method) => METHOD_TO_ENDPOINT[method] || "chat") + ); + + const id = ((m.name as string) || (m.id as string) || "").replace(/^models\//, ""); + const lowerId = id.toLowerCase(); + + // Google exposes Imagen (image) and Veo (video) via long-running methods; the + // method alone can't always distinguish them, so refine by model id. + if (lowerId.includes("veo")) { + endpoints.add("video"); + } + if (lowerId.includes("imagen")) { + endpoints.delete("video"); + endpoints.add("images"); + } + + if (endpoints.size === 0) endpoints.add("chat"); return { ...m, - id: ((m.name as string) || (m.id as string) || "").replace(/^models\//, ""), - name: (m.displayName as string) || ((m.name as string) || "").replace(/^models\//, ""), - supportedEndpoints: endpoints, + id, + name: (m.displayName as string) || id, + supportedEndpoints: [...endpoints], ...(typeof m.inputTokenLimit === "number" ? { inputTokenLimit: m.inputTokenLimit } : {}), ...(typeof m.outputTokenLimit === "number" ? { outputTokenLimit: m.outputTokenLimit } : {}), ...(typeof m.description === "string" ? { description: m.description } : {}), diff --git a/src/shared/validation/schemas.ts b/src/shared/validation/schemas.ts index 58a0a7bf70..dc2084a672 100644 --- a/src/shared/validation/schemas.ts +++ b/src/shared/validation/schemas.ts @@ -1056,6 +1056,7 @@ export const providerModelMutationSchema = z.object({ "rerank", "images", "audio", + "video", "audio-transcriptions", "audio-speech", "images-generations", diff --git a/tests/unit/gemini-models-parser.test.ts b/tests/unit/gemini-models-parser.test.ts index bdaac7f0fd..79b6ddf216 100644 --- a/tests/unit/gemini-models-parser.test.ts +++ b/tests/unit/gemini-models-parser.test.ts @@ -36,6 +36,18 @@ const SAMPLE = { displayName: "Gemini Live", supportedGenerationMethods: ["bidiGenerateContent"], }, + { + name: "models/veo-3.0-generate-001", + displayName: "Veo 3.0", + supportedGenerationMethods: ["predictLongRunning"], + }, + { + // Defensive: an Imagen model exposed via a long-running method must stay + // "images", never "video". + name: "models/imagen-future-preview", + displayName: "Imagen Future", + supportedGenerationMethods: ["predictLongRunning"], + }, ], }; @@ -77,6 +89,20 @@ test("parseGeminiModelsList maps embedContent and bidiGenerateContent", () => { ); }); +test("parseGeminiModelsList maps Veo predictLongRunning models to the video endpoint", () => { + const models = parseGeminiModelsList(SAMPLE); + const veo = models.find((m) => m.id === "veo-3.0-generate-001"); + assert.ok(veo, "veo-3.0-generate-001 should be present"); + assert.deepEqual(veo!.supportedEndpoints, ["video"]); +}); + +test("parseGeminiModelsList keeps Imagen as images even via a long-running method", () => { + const models = parseGeminiModelsList(SAMPLE); + const imagen = models.find((m) => m.id === "imagen-future-preview"); + assert.ok(imagen, "imagen-future-preview should be present"); + assert.deepEqual(imagen!.supportedEndpoints, ["images"]); +}); + test("parseGeminiModelsList defaults to chat and tolerates empty/missing input", () => { assert.deepEqual(parseGeminiModelsList({}), []); assert.deepEqual(parseGeminiModelsList(null), []);