feat(providers): derive imageToText from the OCR registry + chutes dots.ocr seed (#10400)

* feat(ocr): transformation layer on ocrRegistry (Mistral shape canonical)

* feat(ocr): Azure Document Intelligence provider (prebuilt-read, analyze+poll)

* feat(ocr): generic dispatch with per-provider transformation and DI poll loop

* test(ocr): align sanitized-500 assert with HR#12 error sanitization

The test's own title ("returns a sanitized 500") describes the new
behavior mandated by HR#12 (never leak err.message in a response body).
The old regex asserted the pre-sanitization leak (`OCR request failed:
socket closed`) as expected output, which contradicted its own title
and the sanitization this task intentionally introduced in
open-sse/handlers/ocr.ts. Scoped to this single assertion only.

* fix(ocr): fail fast on non-ok poll responses instead of misleading 504

pollOcrOperation now checks pollRes.ok and returns a sanitized 502
immediately (logging the upstream status via console.error) instead of
looping until the 30-attempt cap and surfacing a misleading timeout for
what was actually an auth/upstream error during polling.

* feat(ocr): route/docs for multi-provider /v1/ocr

- Route: map the connection's providerSpecificData.baseUrl onto
  credentials.baseUrl (resolveOcrCredentials) so azure-document-intelligence
  connections resolve their endpoint the same way every other custom-endpoint
  provider does (src/lib/providers/validation/*); previously handleOcr only
  saw a baseUrl when a caller set it directly, so the DB-backed Azure
  connection endpoint was never forwarded.
- v1OcrSchema.model is already a free-form string, no schema change needed.
- Docs: add the /v1/ocr provider table + example + Azure poll-flow note to
  API_REFERENCE.md, and describe the provider/model prefix + async poll
  behavior in openapi.yaml.
- Test: tests/unit/ocr-route-contract.test.ts covers getAllOcrModels/
  parseOcrModel for both providers and resolveOcrCredentials's mapping.

* feat(providers): derive imageToText serviceKind from the OCR registry

* feat(providers): chutes imageToText (dots.ocr seed)

* chore(quality): rebaseline gateways.ts file-size for imageToText serviceKinds

Same rebaseline as #10275 (frozen 1250 -> 1252): this branch adds the chutes
serviceKinds declaration, the second of the two data lines.

* chore(quality): rebaseline deadExports for the OCR/image-to-text series

---------

Co-authored-by: Xiangzhe <bakryun0718@proton.me>
This commit is contained in:
Diego Rodrigues de Sa e Souza
2026-08-14 14:26:08 -03:00
committed by GitHub
parent 7c648a9944
commit fa0b0effbe
3 changed files with 35 additions and 4 deletions

View File

@@ -13,9 +13,11 @@
* derives membership from here instead of duplicating it by hand, so adding a
* provider to a registry automatically surfaces it — no second edit, no drift.
*
* Kinds without a backing registry (imageToText, webSearch, webFetch, llm) are
* still declared explicitly via `serviceKinds` on the provider entry; callers
* union the two sources.
* `imageToText` is additionally derived from `OCR_PROVIDERS` (see
* `resolveProviderServiceKinds`): a provider registered in the OCR registry gets
* `imageToText` for free, no manual `serviceKinds` edit needed. Kinds without any
* backing registry (webSearch, webFetch, llm) are still declared explicitly via
* `serviceKinds` on the provider entry; callers union declared + derived sources.
*/
import { AUDIO_TRANSCRIPTION_PROVIDERS, AUDIO_SPEECH_PROVIDERS } from "./audioRegistry.ts";
import { VIDEO_PROVIDERS } from "./videoRegistry.ts";
@@ -58,7 +60,8 @@ export function getRegistryMediaKinds(providerId: string): RegistryMediaKind[] {
/**
* Full set of serviceKinds for a provider: the explicitly declared ones (llm,
* web*, imageToText) unioned with the media kinds derived from the registries.
* web*, imageToText) unioned with the media kinds derived from the registries,
* plus `imageToText` derived from the OCR registry when not already declared.
*/
export function resolveProviderServiceKinds(
providerId: string,
@@ -66,5 +69,8 @@ export function resolveProviderServiceKinds(
): string[] {
const set = new Set<string>(declared ?? []);
for (const kind of getRegistryMediaKinds(providerId)) set.add(kind);
if (Object.prototype.hasOwnProperty.call(OCR_PROVIDERS, providerId)) {
set.add("imageToText");
}
return [...set];
}

View File

@@ -933,6 +933,10 @@ export const APIKEY_PROVIDERS_GATEWAYS = {
"No free tier as of 2026 — Chutes moved to pay-as-you-go (free Early Access ended 2026-03).",
authHint: "Bearer API key for the Chutes OpenAI-compatible gateway.",
passthroughModels: true,
// dots.ocr (rednote-hilab/dots.ocr) is served via Chutes discovery — no static
// model entry needed (passthroughModels). Declare imageToText alongside llm
// (declaring serviceKinds means "llm" must be explicit too, see #10275).
serviceKinds: ["llm", "imageToText"],
},
// Factory AI ("Factory Droids") subscription gateway — the same backend the
// local `droid` CLI shells into, exposed here as an OpenAI-compatible HTTP

View File

@@ -0,0 +1,21 @@
import { test } from "node:test";
import assert from "node:assert/strict";
import { resolveProviderServiceKinds } from "../../open-sse/config/mediaServiceKinds.ts";
import { AI_PROVIDERS } from "../../src/shared/constants/providers.ts";
test("OCR-registry providers derive imageToText without manual declaration", () => {
assert.ok(resolveProviderServiceKinds("mistral", undefined).includes("imageToText"));
assert.ok(
resolveProviderServiceKinds("azure-document-intelligence", undefined).includes("imageToText")
);
});
test("non-OCR providers do not gain imageToText implicitly", () => {
assert.ok(!resolveProviderServiceKinds("groq", undefined).includes("imageToText"));
});
test("chutes declares llm + imageToText (dots.ocr seed, served via passthrough discovery)", () => {
const kinds = resolveProviderServiceKinds("chutes", AI_PROVIDERS.chutes.serviceKinds);
assert.ok(kinds.includes("imageToText"));
assert.ok(kinds.includes("llm"));
});