From fa0b0effbe4cce685bf8b61ba2d95a5c099041b0 Mon Sep 17 00:00:00 2001 From: Diego Rodrigues de Sa e Souza Date: Fri, 14 Aug 2026 14:26:08 -0300 Subject: [PATCH] feat(providers): derive imageToText from the OCR registry + chutes dots.ocr seed (#10400) * feat(ocr): transformation layer on ocrRegistry (Mistral shape canonical) * feat(ocr): Azure Document Intelligence provider (prebuilt-read, analyze+poll) * feat(ocr): generic dispatch with per-provider transformation and DI poll loop * test(ocr): align sanitized-500 assert with HR#12 error sanitization The test's own title ("returns a sanitized 500") describes the new behavior mandated by HR#12 (never leak err.message in a response body). The old regex asserted the pre-sanitization leak (`OCR request failed: socket closed`) as expected output, which contradicted its own title and the sanitization this task intentionally introduced in open-sse/handlers/ocr.ts. Scoped to this single assertion only. * fix(ocr): fail fast on non-ok poll responses instead of misleading 504 pollOcrOperation now checks pollRes.ok and returns a sanitized 502 immediately (logging the upstream status via console.error) instead of looping until the 30-attempt cap and surfacing a misleading timeout for what was actually an auth/upstream error during polling. * feat(ocr): route/docs for multi-provider /v1/ocr - Route: map the connection's providerSpecificData.baseUrl onto credentials.baseUrl (resolveOcrCredentials) so azure-document-intelligence connections resolve their endpoint the same way every other custom-endpoint provider does (src/lib/providers/validation/*); previously handleOcr only saw a baseUrl when a caller set it directly, so the DB-backed Azure connection endpoint was never forwarded. - v1OcrSchema.model is already a free-form string, no schema change needed. - Docs: add the /v1/ocr provider table + example + Azure poll-flow note to API_REFERENCE.md, and describe the provider/model prefix + async poll behavior in openapi.yaml. - Test: tests/unit/ocr-route-contract.test.ts covers getAllOcrModels/ parseOcrModel for both providers and resolveOcrCredentials's mapping. * feat(providers): derive imageToText serviceKind from the OCR registry * feat(providers): chutes imageToText (dots.ocr seed) * chore(quality): rebaseline gateways.ts file-size for imageToText serviceKinds Same rebaseline as #10275 (frozen 1250 -> 1252): this branch adds the chutes serviceKinds declaration, the second of the two data lines. * chore(quality): rebaseline deadExports for the OCR/image-to-text series --------- Co-authored-by: Xiangzhe --- open-sse/config/mediaServiceKinds.ts | 14 +++++++++---- .../constants/providers/apikey/gateways.ts | 4 ++++ tests/unit/imagetotext-derivation.test.ts | 21 +++++++++++++++++++ 3 files changed, 35 insertions(+), 4 deletions(-) create mode 100644 tests/unit/imagetotext-derivation.test.ts diff --git a/open-sse/config/mediaServiceKinds.ts b/open-sse/config/mediaServiceKinds.ts index 22a692c971..77141c7757 100644 --- a/open-sse/config/mediaServiceKinds.ts +++ b/open-sse/config/mediaServiceKinds.ts @@ -13,9 +13,11 @@ * derives membership from here instead of duplicating it by hand, so adding a * provider to a registry automatically surfaces it — no second edit, no drift. * - * Kinds without a backing registry (imageToText, webSearch, webFetch, llm) are - * still declared explicitly via `serviceKinds` on the provider entry; callers - * union the two sources. + * `imageToText` is additionally derived from `OCR_PROVIDERS` (see + * `resolveProviderServiceKinds`): a provider registered in the OCR registry gets + * `imageToText` for free, no manual `serviceKinds` edit needed. Kinds without any + * backing registry (webSearch, webFetch, llm) are still declared explicitly via + * `serviceKinds` on the provider entry; callers union declared + derived sources. */ import { AUDIO_TRANSCRIPTION_PROVIDERS, AUDIO_SPEECH_PROVIDERS } from "./audioRegistry.ts"; import { VIDEO_PROVIDERS } from "./videoRegistry.ts"; @@ -58,7 +60,8 @@ export function getRegistryMediaKinds(providerId: string): RegistryMediaKind[] { /** * Full set of serviceKinds for a provider: the explicitly declared ones (llm, - * web*, imageToText) unioned with the media kinds derived from the registries. + * web*, imageToText) unioned with the media kinds derived from the registries, + * plus `imageToText` derived from the OCR registry when not already declared. */ export function resolveProviderServiceKinds( providerId: string, @@ -66,5 +69,8 @@ export function resolveProviderServiceKinds( ): string[] { const set = new Set(declared ?? []); for (const kind of getRegistryMediaKinds(providerId)) set.add(kind); + if (Object.prototype.hasOwnProperty.call(OCR_PROVIDERS, providerId)) { + set.add("imageToText"); + } return [...set]; } diff --git a/src/shared/constants/providers/apikey/gateways.ts b/src/shared/constants/providers/apikey/gateways.ts index 15b8fb3edc..cbfd651529 100644 --- a/src/shared/constants/providers/apikey/gateways.ts +++ b/src/shared/constants/providers/apikey/gateways.ts @@ -933,6 +933,10 @@ export const APIKEY_PROVIDERS_GATEWAYS = { "No free tier as of 2026 — Chutes moved to pay-as-you-go (free Early Access ended 2026-03).", authHint: "Bearer API key for the Chutes OpenAI-compatible gateway.", passthroughModels: true, + // dots.ocr (rednote-hilab/dots.ocr) is served via Chutes discovery — no static + // model entry needed (passthroughModels). Declare imageToText alongside llm + // (declaring serviceKinds means "llm" must be explicit too, see #10275). + serviceKinds: ["llm", "imageToText"], }, // Factory AI ("Factory Droids") subscription gateway — the same backend the // local `droid` CLI shells into, exposed here as an OpenAI-compatible HTTP diff --git a/tests/unit/imagetotext-derivation.test.ts b/tests/unit/imagetotext-derivation.test.ts new file mode 100644 index 0000000000..59cfad7db8 --- /dev/null +++ b/tests/unit/imagetotext-derivation.test.ts @@ -0,0 +1,21 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { resolveProviderServiceKinds } from "../../open-sse/config/mediaServiceKinds.ts"; +import { AI_PROVIDERS } from "../../src/shared/constants/providers.ts"; + +test("OCR-registry providers derive imageToText without manual declaration", () => { + assert.ok(resolveProviderServiceKinds("mistral", undefined).includes("imageToText")); + assert.ok( + resolveProviderServiceKinds("azure-document-intelligence", undefined).includes("imageToText") + ); +}); + +test("non-OCR providers do not gain imageToText implicitly", () => { + assert.ok(!resolveProviderServiceKinds("groq", undefined).includes("imageToText")); +}); + +test("chutes declares llm + imageToText (dots.ocr seed, served via passthrough discovery)", () => { + const kinds = resolveProviderServiceKinds("chutes", AI_PROVIDERS.chutes.serviceKinds); + assert.ok(kinds.includes("imageToText")); + assert.ok(kinds.includes("llm")); +});