fix: map OpenAI-compat voice names to real ElevenLabs voice_ids (#10589) (#10748)

handleElevenLabsSpeech forwarded body.voice straight into the ElevenLabs
voice_id URL path segment with no name resolution, so OpenAI stock voice
names (alloy, echo, ...) and ElevenLabs display names (Rachel, ...) 404'd
upstream instead of resolving. Extracted the alias/display-name tables and
resolution logic into open-sse/handlers/elevenLabsVoiceMap.ts (kept
audioSpeech.ts under the file-size cap) and wired it into
handleElevenLabsSpeech: known aliases resolve to a real voice_id, an
omitted voice keeps the previous Rachel default, and anything unresolvable
now returns a clear 400 instead of leaking an upstream 404.

Co-authored-by: Markus Hartung <mail@hartmark.se>
This commit is contained in:
Diego Rodrigues de Sa e Souza
2026-08-19 11:08:45 -03:00
committed by GitHub
parent a8000bf1a2
commit a1a37bbe7f
4 changed files with 206 additions and 2 deletions

View File

@@ -0,0 +1 @@
- fix(sse): map OpenAI-compat voice names to real ElevenLabs voice_ids in direct TTS (#10589)

View File

@@ -25,6 +25,7 @@ import { handleAwsPollySpeech } from "../executors/awsPollyTts.ts";
import { handleEdgeTtsSpeech } from "../executors/edgeTts.ts";
import { GttsUpstreamError, normalizeGttsLang, synthesizeGtts } from "../executors/gtts.ts";
import { errorResponse } from "../utils/error.ts";
import { resolveElevenLabsVoiceId } from "./elevenLabsVoiceMap.ts";
import { audioStreamResponse, upstreamErrorResponse } from "../utils/audioResponse.ts";
import {
getKieCallbackUrl,
@@ -263,8 +264,20 @@ async function handleSonioxSpeech(providerConfig, body, modelId, token) {
* voice_id is mapped from the OpenAI `voice` parameter
*/
async function handleElevenLabsSpeech(providerConfig, body, modelId, token) {
// ElevenLabs uses voice_id in URL path; default to "21m00Tcm4TlvDq8ikWAM" (Rachel)
const voiceId = body.voice || "21m00Tcm4TlvDq8ikWAM";
// ElevenLabs uses voice_id in URL path. body.voice may be an OpenAI stock voice name
// (alloy, echo, ...), a known ElevenLabs display name (Rachel, ...), or a raw voice_id;
// resolve it to a real voice_id before it ever reaches the URL. Defaults to Rachel
// ("21m00Tcm4TlvDq8ikWAM") when omitted.
if (typeof body.voice === "string" && !isValidPathSegment(body.voice)) {
return errorResponse(400, "Invalid voice ID");
}
const voiceId = resolveElevenLabsVoiceId(body.voice);
if (!voiceId) {
return errorResponse(
400,
"Unknown ElevenLabs voice. Provide a real ElevenLabs voice_id, a supported OpenAI voice name (alloy, echo, fable, onyx, nova, shimmer), or a known ElevenLabs display name."
);
}
if (!isValidPathSegment(voiceId)) {
return errorResponse(400, "Invalid voice ID");
}

View File

@@ -0,0 +1,68 @@
/**
* OpenAI-compat `voice` name -> ElevenLabs `voice_id` resolution.
*
* ElevenLabs' TTS endpoint takes a real `voice_id` (a ~20-char alphanumeric token, e.g.
* `21m00Tcm4TlvDq8ikWAM`) as a URL path segment. OpenAI TTS stock voice names (`alloy`,
* `echo`, ...) and ElevenLabs human-readable display names (`Rachel`) are not valid
* `voice_id`s on their own — forwarding them unmapped 404s upstream. This module resolves
* a client-supplied `voice` value to a real `voice_id`, or reports that it cannot.
*
* See #10589.
*/
// OpenAI TTS stock voice names -> real ElevenLabs voice_id (premade voices, widely
// available across ElevenLabs accounts/plans).
const OPENAI_VOICE_TO_ELEVENLABS_ID: Record<string, string> = {
alloy: "21m00Tcm4TlvDq8ikWAM", // Rachel
echo: "pNInz6obpgDQGcFmaJgB", // Adam
fable: "nPczCjzI2devNBz1zQrb", // Brian
onyx: "ErXwobaYiN019PkySvjV", // Antoni
nova: "EXAVITQu4vr4xnSDxMaL", // Bella
shimmer: "ThT5KcBeYPX3keUQqHPh", // Dorothy
};
// A handful of well-known ElevenLabs display names -> voice_id, matched case-insensitively,
// so a request like `voice: "Rachel"` (a real display name but not a raw voice_id) resolves
// instead of 404-ing upstream.
const ELEVENLABS_DISPLAY_NAME_TO_ID: Record<string, string> = {
rachel: "21m00Tcm4TlvDq8ikWAM",
adam: "pNInz6obpgDQGcFmaJgB",
brian: "nPczCjzI2devNBz1zQrb",
antoni: "ErXwobaYiN019PkySvjV",
bella: "EXAVITQu4vr4xnSDxMaL",
dorothy: "ThT5KcBeYPX3keUQqHPh",
};
export const ELEVENLABS_DEFAULT_VOICE_ID = "21m00Tcm4TlvDq8ikWAM"; // Rachel
// A real ElevenLabs voice_id is a ~20-char alphanumeric token (e.g. 21m00Tcm4TlvDq8ikWAM).
const ELEVENLABS_VOICE_ID_PATTERN = /^[A-Za-z0-9]{16,32}$/;
/**
* Resolve an OpenAI-compat `voice` value (or ElevenLabs display name) to a real
* ElevenLabs voice_id. Returns null when the value is present but cannot be
* resolved to a known alias and does not itself look like a raw voice_id.
*/
export function resolveElevenLabsVoiceId(voice: unknown): string | null {
if (voice === undefined || voice === null || voice === "") {
return ELEVENLABS_DEFAULT_VOICE_ID;
}
if (typeof voice !== "string") {
return null;
}
const trimmed = voice.trim();
if (!trimmed) {
return ELEVENLABS_DEFAULT_VOICE_ID;
}
const lower = trimmed.toLowerCase();
if (OPENAI_VOICE_TO_ELEVENLABS_ID[lower]) {
return OPENAI_VOICE_TO_ELEVENLABS_ID[lower];
}
if (ELEVENLABS_DISPLAY_NAME_TO_ID[lower]) {
return ELEVENLABS_DISPLAY_NAME_TO_ID[lower];
}
if (ELEVENLABS_VOICE_ID_PATTERN.test(trimmed)) {
return trimmed;
}
return null;
}

View File

@@ -132,6 +132,128 @@ test("handleAudioSpeech rejects invalid ElevenLabs voice identifiers", async ()
}
});
test("handleAudioSpeech maps OpenAI stock voice name alloy to a real ElevenLabs voice_id", async () => {
const originalFetch = globalThis.fetch;
let capturedUrl;
globalThis.fetch = async (url) => {
capturedUrl = String(url);
return new Response(new Uint8Array([1, 2, 3]), {
status: 200,
headers: { "content-type": "audio/mpeg" },
});
};
try {
const response = await handleAudioSpeech({
body: {
model: "elevenlabs/eleven_multilingual_v2",
input: "hello",
voice: "alloy",
},
credentials: { apiKey: "xi-key" },
});
assert.equal(response.status, 200);
assert.equal(
capturedUrl,
"https://api.elevenlabs.io/v1/text-to-speech/21m00Tcm4TlvDq8ikWAM"
);
} finally {
globalThis.fetch = originalFetch;
}
});
test("handleAudioSpeech resolves ElevenLabs display name 'rachel' case-insensitively", async () => {
const originalFetch = globalThis.fetch;
let capturedUrl;
globalThis.fetch = async (url) => {
capturedUrl = String(url);
return new Response(new Uint8Array([1, 2, 3]), {
status: 200,
headers: { "content-type": "audio/mpeg" },
});
};
try {
const response = await handleAudioSpeech({
body: {
model: "elevenlabs/eleven_multilingual_v2",
input: "hello",
voice: "rachel",
},
credentials: { apiKey: "xi-key" },
});
assert.equal(response.status, 200);
assert.equal(
capturedUrl,
"https://api.elevenlabs.io/v1/text-to-speech/21m00Tcm4TlvDq8ikWAM"
);
} finally {
globalThis.fetch = originalFetch;
}
});
test("handleAudioSpeech defaults to Rachel's voice_id when voice is omitted", async () => {
const originalFetch = globalThis.fetch;
let capturedUrl;
globalThis.fetch = async (url) => {
capturedUrl = String(url);
return new Response(new Uint8Array([1, 2, 3]), {
status: 200,
headers: { "content-type": "audio/mpeg" },
});
};
try {
const response = await handleAudioSpeech({
body: {
model: "elevenlabs/eleven_multilingual_v2",
input: "hello",
},
credentials: { apiKey: "xi-key" },
});
assert.equal(response.status, 200);
assert.equal(
capturedUrl,
"https://api.elevenlabs.io/v1/text-to-speech/21m00Tcm4TlvDq8ikWAM"
);
} finally {
globalThis.fetch = originalFetch;
}
});
test("handleAudioSpeech returns 400 for an unresolvable ElevenLabs voice name instead of forwarding upstream", async () => {
const originalFetch = globalThis.fetch;
let called = false;
globalThis.fetch = async () => {
called = true;
throw new Error("should not fetch");
};
try {
const response = await handleAudioSpeech({
body: {
model: "elevenlabs/eleven_turbo_v2_5",
input: "unknown voice",
voice: "totally-not-a-voice",
},
credentials: { apiKey: "xi-key" },
});
const payload = (await response.json()) as { error: { message: string } };
assert.equal(response.status, 400);
assert.match(payload.error.message, /Unknown ElevenLabs voice/);
assert.equal(called, false);
} finally {
globalThis.fetch = originalFetch;
}
});
test("handleAudioSpeech maps Cartesia voice and wav output settings", async () => {
const originalFetch = globalThis.fetch;
let captured;