Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 32 additions & 1 deletion backend/app/routers/settings.py
Original file line number Diff line number Diff line change
Expand Up @@ -1671,6 +1671,29 @@ def load_speech_config(session: Session, user_id: str) -> Dict[str, Any]:
}


def load_selected_voice(session: Session, user_id: str) -> Optional[str]:
"""The user's globally selected fal.ai voice (e.g. "Rex"), read for callers
outside a request — the article-to-video renderer runs on a worker thread
with no request body to carry it in.

Persisted as a generic per-user setting under the key "tts_model" (see the
frontend settings store's `updateAppSettings({ tts_model: voice })`) —
confusingly the same key name `load_speech_config` uses for the actual
fal.ai *model* id, but a separate row: this one lives at the top level of
UserSetting, that one inside the `_SPEECH_CONFIG` JSON blob.
"""
row = session.exec(
select(UserSetting).where(UserSetting.user_id == user_id, UserSetting.key == "tts_model")
).first()
if not row or not row.value:
return None
try:
value = json.loads(row.value)
except (ValueError, TypeError):
return None
return value if isinstance(value, str) and value.strip() else None


def get_voices_for_model(session: Session, model_id: str, custom_models: List[Dict[str, Any]]) -> List[str]:
"""Get available voices for a given TTS model."""
# Check curated models
Expand Down Expand Up @@ -2060,6 +2083,7 @@ async def _fal_tts(session: Session, user_id: str, api_key: str, text: str, voic

async def synthesize_tts_bytes(
session: Session, user_id: str, text: str, *, voice: Optional[str] = None,
provider_override: Optional[str] = None,
) -> Tuple[bytes, str]:
"""Synthesise one piece of text with the account's TTS settings.

Expand All @@ -2068,6 +2092,13 @@ async def synthesize_tts_bytes(
`(audio_bytes, media_type)`. Callers outside a request (the render worker
runs on its own thread) drive it with `asyncio.run`.

`provider_override` lets a caller that makes many calls for one artifact —
the renderer, one call per narration chunk — pin whichever provider "auto"
would have picked for the *first* chunk and reuse it for every chunk after.
Left as "auto" per call, a transient Deepgram hiccup partway through would
silently fall back to fal.ai for just that one chunk (see below), handing
back a video whose narrator audibly changes voice partway through.

Raises HTTPException on failure — the HTTP endpoint surfaces that directly,
and the worker turns it into a job error.
"""
Expand All @@ -2078,7 +2109,7 @@ async def synthesize_tts_bytes(
raise HTTPException(status_code=400, detail={"code": "text_too_long", "message": f"Text exceeds {_TTS_MAX_CHARS} characters"})

speech_cfg = load_speech_config(session, user_id)
tts_provider = speech_cfg["tts_provider"]
tts_provider = provider_override or speech_cfg["tts_provider"]
deepgram_key = load_deepgram_api_key(session, user_id)
fal_key = load_fal_api_key(session, user_id)

Expand Down
27 changes: 25 additions & 2 deletions backend/app/video/worker.py
Original file line number Diff line number Diff line change
Expand Up @@ -68,11 +68,34 @@ def _tts_caller(user_id: str):
bill for the narration only once. Each call opens its own session because
this runs on a worker thread, not in a request.
"""
from app.routers.settings import synthesize_tts_bytes
from app.routers.settings import (
load_deepgram_api_key, load_selected_voice, load_speech_config, synthesize_tts_bytes,
)

with Session(engine) as session:
# Read-aloud gets the user's chosen voice from the request body; a
# render has no request, so without this the fal.ai path fell back to
# the selected model's first listed voice for every render, silently
# ignoring whatever voice (e.g. a non-default one like "Rex") the user
# picked in Settings -> Speech.
voice = load_selected_voice(session, user_id)

# Resolve "auto" to a concrete provider once, up front, and pin it for
# every chunk of this render. Left as "auto" per chunk, a transient
# Deepgram hiccup partway through a narration would silently fall back
# to fal.ai — a different engine with a different voice — for just
# that one chunk, handing back a video that audibly changes voice
# mid-way. Pinned, the same hiccup fails the render instead, which is
# a render worth retrying rather than a video worth keeping.
tts_provider = load_speech_config(session, user_id)["tts_provider"]
if tts_provider == "auto":
tts_provider = "deepgram" if load_deepgram_api_key(session, user_id) else "fal"

def call(text: str) -> bytes:
with Session(engine) as session:
data, _media_type = asyncio.run(synthesize_tts_bytes(session, user_id, text))
data, _media_type = asyncio.run(
synthesize_tts_bytes(session, user_id, text, voice=voice, provider_override=tts_provider)
)
return data

return call
Expand Down
8 changes: 6 additions & 2 deletions frontend/src/components/AIConversationPanel.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -416,6 +416,7 @@ export default function AIConversationPanel({
const falKeyConfigured = useSettingsStore((s) => s.falKeyConfigured)
const deepgramKeyConfigured = useSettingsStore((s) => s.deepgramKeyConfigured)
const sttProvider = useSettingsStore((s) => s.sttProvider)
const ttsVoice = useSettingsStore((s) => s.voice)
const categories = useCategoriesStore((s) => s.categories)

// Conversation and session state (self-managed — not driven by props)
Expand Down Expand Up @@ -536,8 +537,11 @@ export default function AIConversationPanel({

// Read-aloud for assistant responses. One player instance for the panel — only
// one message speaks at a time (tracked by speakingMsgId). The TTS provider
// (Deepgram Flux or fal.ai fallback) is resolved server-side.
const readAloud = useTextToSpeech()
// (Deepgram Flux or fal.ai fallback) is resolved server-side, but the fal.ai
// fallback still needs the user's chosen voice passed through explicitly —
// omitting it here silently fell back to the hardcoded default voice
// instead of the one configured in Settings → Speech.
const readAloud = useTextToSpeech({ model: ttsVoice })
useEffect(() => {
if (readAloud.status === 'idle' || readAloud.status === 'error') setSpeakingMsgId(null)
}, [readAloud.status])
Expand Down
8 changes: 7 additions & 1 deletion frontend/src/hooks/useVoiceMode.ts
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
import { useCallback, useEffect, useRef, useState } from 'react'
import { connectFluxStream, type FluxStreamEvent, type FluxStreamHandle } from '@/api/fluxStream'
import { useTextToSpeech } from '@/hooks/useTextToSpeech'
import { useSettingsStore } from '@/stores/settings'

// The explicit conversational states the voice overlay renders.
export type VoiceState =
Expand Down Expand Up @@ -49,7 +50,12 @@ export function useVoiceMode(options: UseVoiceModeOptions): UseVoiceModeReturn {
const [interimText, setInterimText] = useState('')
const [errorMessage, setErrorMessage] = useState('')

const tts = useTextToSpeech()
// Voice mode speaks via Deepgram Flux, whose voice comes from the account's
// Deepgram config, not this hook's `model` option — but a Deepgram hiccup
// falls back to fal.ai (see synthesize_tts_bytes), and that fallback needs
// the configured fal.ai voice or it silently speaks the hardcoded default.
const ttsVoice = useSettingsStore((s) => s.voice)
const tts = useTextToSpeech({ model: ttsVoice })
// useTextToSpeech returns a fresh object every render, so hold it in a ref and
// have the lifecycle callbacks below read tts.stop()/tts.play() through it.
// Depending on the `tts` object directly would recreate teardown() every render,
Expand Down