diff --git a/open-sse/handlers/audioSpeech.ts b/open-sse/handlers/audioSpeech.ts index 229e4eb14f..abf4817a7f 100644 --- a/open-sse/handlers/audioSpeech.ts +++ b/open-sse/handlers/audioSpeech.ts @@ -24,13 +24,28 @@ import { errorResponse } from "../utils/error.ts"; * Return a CORS error response from an upstream fetch failure */ function upstreamErrorResponse(res, errText) { - return new Response(errText, { - status: res.status, - headers: { - "Content-Type": "application/json", - "Access-Control-Allow-Origin": getCorsOrigin(), - }, - }); + // Always return JSON so the client can detect 401/credential errors reliably + let errorMessage: string; + try { + const parsed = JSON.parse(errText); + errorMessage = + parsed?.err_msg || + parsed?.error?.message || + parsed?.error || + parsed?.message || + parsed?.detail || + errText; + } catch { + errorMessage = errText || `Upstream error (${res.status})`; + } + + return Response.json( + { error: { message: errorMessage, code: res.status } }, + { + status: res.status, + headers: { "Access-Control-Allow-Origin": getCorsOrigin() }, + } + ); } /** diff --git a/open-sse/handlers/audioTranscription.ts b/open-sse/handlers/audioTranscription.ts index 54454b905d..c7148f1af5 100644 --- a/open-sse/handlers/audioTranscription.ts +++ b/open-sse/handlers/audioTranscription.ts @@ -26,13 +26,28 @@ type TranscriptionCredentials = { * Return a CORS error response from an upstream fetch failure */ function upstreamErrorResponse(res, errText) { - return new Response(errText, { - status: res.status, - headers: { - "Content-Type": "application/json", - "Access-Control-Allow-Origin": getCorsOrigin(), - }, - }); + // Always return JSON so the client can parse the error reliably + let errorMessage: string; + try { + const parsed = JSON.parse(errText); + errorMessage = + parsed?.err_msg || + parsed?.error?.message || + parsed?.error || + parsed?.message || + parsed?.detail || + errText; + } catch { + errorMessage = errText || `Upstream error (${res.status})`; + } + + return Response.json( + { error: { message: errorMessage, code: res.status } }, + { + status: res.status, + headers: { "Access-Control-Allow-Origin": getCorsOrigin() }, + } + ); } /** @@ -71,9 +86,14 @@ async function handleDeepgramTranscription(providerConfig, file, modelId, token) const data = await res.json(); // Transform Deepgram response to OpenAI Whisper format - const text = data.results?.channels?.[0]?.alternatives?.[0]?.transcript || ""; + const text = data.results?.channels?.[0]?.alternatives?.[0]?.transcript ?? null; - return Response.json({ text }, { headers: { "Access-Control-Allow-Origin": getCorsOrigin() } }); + // null means the audio had no recognizable speech (music, silence, etc.) + // Return it explicitly so the client can distinguish from a credentials error + return Response.json( + { text: text ?? "", noSpeechDetected: text === null || text === "" }, + { headers: { "Access-Control-Allow-Origin": getCorsOrigin() } } + ); } /** diff --git a/src/app/(dashboard)/dashboard/media/MediaPageClient.tsx b/src/app/(dashboard)/dashboard/media/MediaPageClient.tsx index 1b3aaa27b3..7a88e6a8b7 100644 --- a/src/app/(dashboard)/dashboard/media/MediaPageClient.tsx +++ b/src/app/(dashboard)/dashboard/media/MediaPageClient.tsx @@ -328,6 +328,7 @@ function getVoiceList(providerId: string) { function parseApiError(raw: any, statusCode: number): { message: string; isCredentials: boolean } { const msg = raw?.error?.message || + raw?.err_msg || raw?.error || raw?.message || raw?.detail || @@ -340,6 +341,7 @@ function parseApiError(raw: any, statusCode: number): { message: string; isCrede msg.toLowerCase().includes("invalid api key") || msg.toLowerCase().includes("unauthorized") || msg.toLowerCase().includes("authentication") || + msg.toLowerCase().includes("api key") || statusCode === 401 || statusCode === 403); @@ -519,12 +521,22 @@ export default function MediaPageClient() { throw new Error(message); } const data = await res.json(); - // Warn if text is empty (likely missing credentials that returned silently) + // Check for noSpeechDetected flag (music, silence, etc.) — NOT a credential error + if (data?.noSpeechDetected) { + setError( + `No speech detected in the audio file. If you uploaded music or a silent file, try an audio file with spoken words. Provider: "${selectedProvider}".` + ); + setIsCredentialsError(false); + setLoading(false); + return; + } + // Warn if text is empty without the noSpeechDetected flag (unexpected) if (data && typeof data.text === "string" && data.text.trim() === "") { setError( - `Transcription returned empty text. Make sure you have a valid API key for "${selectedProvider}" configured in /dashboard/providers.` + `Transcription returned empty text. The audio may contain no recognizable speech, or the "${selectedProvider}" API key may be invalid. Check Dashboard → Logs → Proxy for details.` ); - setIsCredentialsError(true); + // Only mark as credential error if we can confirm it from context + setIsCredentialsError(false); setLoading(false); return; }