From 9c29c412dfa8ff41b14c9597c13bf2949ae0d101 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 12 Sep 2026 13:46:53 +0000 Subject: [PATCH] Fix TTS voice being ignored/inconsistent in read-aloud and video export Three related bugs caused a chosen fal.ai voice (e.g. "Rex") to be silently replaced: - AI Assistant read-aloud and voice mode called useTextToSpeech() without a voice, so any fal.ai fallback used the hardcoded default voice instead of the one configured in Settings -> Speech. - The article-to-video renderer never forwarded the selected voice at all, so every export used the current model's first listed voice. - Article-to-video narration re-resolved the "auto" Deepgram/fal.ai provider choice per chunk, so a single transient Deepgram failure mid-render silently swapped just that chunk to fal.ai's voice, producing a video whose narrator changes partway through. The provider is now resolved once per render job and pinned for every chunk, so a failure now fails the render instead of mixing voices. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01R54PjuaaN6QnvxVCcxiuVs --- backend/app/routers/settings.py | 33 ++++++++++++++++++- backend/app/video/worker.py | 27 +++++++++++++-- .../src/components/AIConversationPanel.tsx | 8 +++-- frontend/src/hooks/useVoiceMode.ts | 8 ++++- 4 files changed, 70 insertions(+), 6 deletions(-) diff --git a/backend/app/routers/settings.py b/backend/app/routers/settings.py index bd81b4c..54db566 100644 --- a/backend/app/routers/settings.py +++ b/backend/app/routers/settings.py @@ -1671,6 +1671,29 @@ def load_speech_config(session: Session, user_id: str) -> Dict[str, Any]: } +def load_selected_voice(session: Session, user_id: str) -> Optional[str]: + """The user's globally selected fal.ai voice (e.g. "Rex"), read for callers + outside a request — the article-to-video renderer runs on a worker thread + with no request body to carry it in. + + Persisted as a generic per-user setting under the key "tts_model" (see the + frontend settings store's `updateAppSettings({ tts_model: voice })`) — + confusingly the same key name `load_speech_config` uses for the actual + fal.ai *model* id, but a separate row: this one lives at the top level of + UserSetting, that one inside the `_SPEECH_CONFIG` JSON blob. + """ + row = session.exec( + select(UserSetting).where(UserSetting.user_id == user_id, UserSetting.key == "tts_model") + ).first() + if not row or not row.value: + return None + try: + value = json.loads(row.value) + except (ValueError, TypeError): + return None + return value if isinstance(value, str) and value.strip() else None + + def get_voices_for_model(session: Session, model_id: str, custom_models: List[Dict[str, Any]]) -> List[str]: """Get available voices for a given TTS model.""" # Check curated models @@ -2060,6 +2083,7 @@ async def _fal_tts(session: Session, user_id: str, api_key: str, text: str, voic async def synthesize_tts_bytes( session: Session, user_id: str, text: str, *, voice: Optional[str] = None, + provider_override: Optional[str] = None, ) -> Tuple[bytes, str]: """Synthesise one piece of text with the account's TTS settings. @@ -2068,6 +2092,13 @@ async def synthesize_tts_bytes( `(audio_bytes, media_type)`. Callers outside a request (the render worker runs on its own thread) drive it with `asyncio.run`. + `provider_override` lets a caller that makes many calls for one artifact — + the renderer, one call per narration chunk — pin whichever provider "auto" + would have picked for the *first* chunk and reuse it for every chunk after. + Left as "auto" per call, a transient Deepgram hiccup partway through would + silently fall back to fal.ai for just that one chunk (see below), handing + back a video whose narrator audibly changes voice partway through. + Raises HTTPException on failure — the HTTP endpoint surfaces that directly, and the worker turns it into a job error. """ @@ -2078,7 +2109,7 @@ async def synthesize_tts_bytes( raise HTTPException(status_code=400, detail={"code": "text_too_long", "message": f"Text exceeds {_TTS_MAX_CHARS} characters"}) speech_cfg = load_speech_config(session, user_id) - tts_provider = speech_cfg["tts_provider"] + tts_provider = provider_override or speech_cfg["tts_provider"] deepgram_key = load_deepgram_api_key(session, user_id) fal_key = load_fal_api_key(session, user_id) diff --git a/backend/app/video/worker.py b/backend/app/video/worker.py index 86f2c87..9625654 100644 --- a/backend/app/video/worker.py +++ b/backend/app/video/worker.py @@ -68,11 +68,34 @@ def _tts_caller(user_id: str): bill for the narration only once. Each call opens its own session because this runs on a worker thread, not in a request. """ - from app.routers.settings import synthesize_tts_bytes + from app.routers.settings import ( + load_deepgram_api_key, load_selected_voice, load_speech_config, synthesize_tts_bytes, + ) + + with Session(engine) as session: + # Read-aloud gets the user's chosen voice from the request body; a + # render has no request, so without this the fal.ai path fell back to + # the selected model's first listed voice for every render, silently + # ignoring whatever voice (e.g. a non-default one like "Rex") the user + # picked in Settings -> Speech. + voice = load_selected_voice(session, user_id) + + # Resolve "auto" to a concrete provider once, up front, and pin it for + # every chunk of this render. Left as "auto" per chunk, a transient + # Deepgram hiccup partway through a narration would silently fall back + # to fal.ai — a different engine with a different voice — for just + # that one chunk, handing back a video that audibly changes voice + # mid-way. Pinned, the same hiccup fails the render instead, which is + # a render worth retrying rather than a video worth keeping. + tts_provider = load_speech_config(session, user_id)["tts_provider"] + if tts_provider == "auto": + tts_provider = "deepgram" if load_deepgram_api_key(session, user_id) else "fal" def call(text: str) -> bytes: with Session(engine) as session: - data, _media_type = asyncio.run(synthesize_tts_bytes(session, user_id, text)) + data, _media_type = asyncio.run( + synthesize_tts_bytes(session, user_id, text, voice=voice, provider_override=tts_provider) + ) return data return call diff --git a/frontend/src/components/AIConversationPanel.tsx b/frontend/src/components/AIConversationPanel.tsx index 5fec066..960c051 100644 --- a/frontend/src/components/AIConversationPanel.tsx +++ b/frontend/src/components/AIConversationPanel.tsx @@ -416,6 +416,7 @@ export default function AIConversationPanel({ const falKeyConfigured = useSettingsStore((s) => s.falKeyConfigured) const deepgramKeyConfigured = useSettingsStore((s) => s.deepgramKeyConfigured) const sttProvider = useSettingsStore((s) => s.sttProvider) + const ttsVoice = useSettingsStore((s) => s.voice) const categories = useCategoriesStore((s) => s.categories) // Conversation and session state (self-managed — not driven by props) @@ -536,8 +537,11 @@ export default function AIConversationPanel({ // Read-aloud for assistant responses. One player instance for the panel — only // one message speaks at a time (tracked by speakingMsgId). The TTS provider - // (Deepgram Flux or fal.ai fallback) is resolved server-side. - const readAloud = useTextToSpeech() + // (Deepgram Flux or fal.ai fallback) is resolved server-side, but the fal.ai + // fallback still needs the user's chosen voice passed through explicitly — + // omitting it here silently fell back to the hardcoded default voice + // instead of the one configured in Settings → Speech. + const readAloud = useTextToSpeech({ model: ttsVoice }) useEffect(() => { if (readAloud.status === 'idle' || readAloud.status === 'error') setSpeakingMsgId(null) }, [readAloud.status]) diff --git a/frontend/src/hooks/useVoiceMode.ts b/frontend/src/hooks/useVoiceMode.ts index 81bd95d..ebde2f7 100644 --- a/frontend/src/hooks/useVoiceMode.ts +++ b/frontend/src/hooks/useVoiceMode.ts @@ -1,6 +1,7 @@ import { useCallback, useEffect, useRef, useState } from 'react' import { connectFluxStream, type FluxStreamEvent, type FluxStreamHandle } from '@/api/fluxStream' import { useTextToSpeech } from '@/hooks/useTextToSpeech' +import { useSettingsStore } from '@/stores/settings' // The explicit conversational states the voice overlay renders. export type VoiceState = @@ -49,7 +50,12 @@ export function useVoiceMode(options: UseVoiceModeOptions): UseVoiceModeReturn { const [interimText, setInterimText] = useState('') const [errorMessage, setErrorMessage] = useState('') - const tts = useTextToSpeech() + // Voice mode speaks via Deepgram Flux, whose voice comes from the account's + // Deepgram config, not this hook's `model` option — but a Deepgram hiccup + // falls back to fal.ai (see synthesize_tts_bytes), and that fallback needs + // the configured fal.ai voice or it silently speaks the hardcoded default. + const ttsVoice = useSettingsStore((s) => s.voice) + const tts = useTextToSpeech({ model: ttsVoice }) // useTextToSpeech returns a fresh object every render, so hold it in a ref and // have the lifecycle callbacks below read tts.stop()/tts.play() through it. // Depending on the `tts` object directly would recreate teardown() every render,