diff --git a/README.md b/README.md index f367be4..7a936bd 100644 --- a/README.md +++ b/README.md @@ -51,6 +51,7 @@ A full-featured, self-hosted notes application with a block editor, an agentic A - **Transitions** between segments — a dip through black or white, or a blend (dissolve, wipe, slide, circle open). A dip is drawn inside each segment and costs nothing; a blend needs the finished video encoded a second time, and the dialog says so - **Motion on stills (Ken Burns)** — a slow zoom or pan over each image, with an adjustable travel distance and an option to include the title and chapter screens. Video clips are left alone, since the footage already moves. A drifting shot is rendered well above the output frame and scaled back down, which is what keeps the movement smooth instead of stepping a pixel at a time — it costs render time, so a shot that drifts is slower than one that doesn't. On a segment long enough that one sweep across it would be too slow to see, the motion cycles instead — drifting A to B, then B to A, in legs short enough to stay visible, rather than crawling once across the whole segment or holding still; a higher travel distance lengthens each leg. A per-segment encode timeout scales with the segment's own length rather than a flat cap, so a long section isn't cut off before it can finish - **Background music** — an uploaded track mixed under the narration, ducking beneath speech and coming back up in the gaps, with its own level and fade in/out. A short track loops and a long one is cut to the video; the picture is never re-encoded to add it +- **Intro and outro clips** — an uploaded video played before the title screen and another after the last section, each whole and with its own sound, fitted to the frame. Nothing of the render is drawn over them (no watermark, overlay or waveform), the background music plays only between them, and each gets its own chapter marker - **Quotes on screen** — a blockquote gets its own segment, with the words shown over the same picture while they are read, and a trailing "— name" line picked up as the attribution - **Title screens**, optional chapter screens, chapter markers embedded in the MP4, an automatic thumbnail, and **subtitles** as an `.srt` sidecar, a track inside the MP4, or burned into the picture - A chapter screen reads its own heading while the words are on screen; with chapter screens off, the heading is read inside the section it introduces @@ -58,7 +59,7 @@ A full-featured, self-hosted notes application with a block editor, an agentic A - An adjustable **pause at the end of every segment** — a paragraph, a section, a title or chapter screen — held after the last word before cutting to what's next, so a segment finishes rather than getting clipped by the cut - **Every text size is adjustable** — title screen, chapter screen, watermark icon and caption, and the fixed overlay — set as a percentage of the frame height, so one choice holds at every resolution and aspect ratio - Renders in the background with progress in the header and the browser tab, and can be cancelled mid-render — the finished video is attached to the note by the server, so it arrives even if you close the tab -- Choice of TTS voice and speaking rate per video, and options are remembered between renders — grouped into Format, Narration, Motion & audio, Branding and Structure tabs +- Choice of TTS voice and speaking rate per video, and options are remembered between renders — saved to your account, so they follow you to any browser, and grouped into Format, Narration, Motion & audio, Branding, Intro & outro and Structure tabs. Files the options point at (intro, outro, watermark icon, music) are kept in your media, and the unlinked-file sweep treats them as in use ### Sharing & export - Export to PDF, Word (.docx), Markdown, HTML, MP3, MP4 video, or clipboard diff --git a/backend/app/asset_utils.py b/backend/app/asset_utils.py index 3c99500..a4b5013 100644 --- a/backend/app/asset_utils.py +++ b/backend/app/asset_utils.py @@ -22,12 +22,12 @@ import os import uuid from dataclasses import dataclass -from typing import List, Optional, Tuple +from typing import Any, Iterator, List, Optional, Set, Tuple import json from sqlmodel import Session, col, or_, select -from app.models import Note, NoteAsset, Theme, TranscriptionJob, User, VideoRenderJob +from app.models import Note, NoteAsset, Theme, TranscriptionJob, User, UserSetting, VideoRenderJob from app.routers.media import MEDIA_DIR, categorize_extension logger = logging.getLogger(__name__) @@ -269,6 +269,46 @@ def sync_note_assets(session: Session, note) -> int: return 0 +def _strings(value: Any) -> Iterator[str]: + """Every string anywhere inside a decoded JSON value.""" + if isinstance(value, str): + yield value + elif isinstance(value, dict): + for item in value.values(): + yield from _strings(item) + elif isinstance(value, list): + for item in value: + yield from _strings(item) + + +def settings_media_filenames(session: Session, user_id: str) -> Set[str]: + """Filenames in this user's media dir that one of their settings points at. + + The video dialog's saved options hold the intro and outro clips, watermark, + music and background the user picked once and reuses on every render. Those + files belong to no note, so without this they would read as leaked and be + offered up for deletion by the unlinked-file sweep. + """ + names: Set[str] = set() + try: + values = session.exec(select(UserSetting.value).where(UserSetting.user_id == user_id)).all() + except Exception: + logger.exception("Could not read settings for user %s", user_id) + return names + for raw in values: + if not raw or MEDIA_URL_PREFIX not in raw: + continue + try: + decoded = json.loads(raw) + except (ValueError, TypeError): + continue + for text in _strings(decoded): + parsed = parse_media_url(text) + if parsed and parsed[0] == user_id: + names.add(parsed[1]) + return names + + def file_is_referenced( session: Session, user_id: str, @@ -338,6 +378,9 @@ def file_is_referenced( ).first(): return True + if filename in settings_media_filenames(session, user_id): + return True + return False diff --git a/backend/app/routers/assets.py b/backend/app/routers/assets.py index 401d6d4..6398fa2 100644 --- a/backend/app/routers/assets.py +++ b/backend/app/routers/assets.py @@ -22,6 +22,7 @@ register_asset, release_media_file, remove_media_file, + settings_media_filenames, sync_note_assets, ) from app.database import get_session @@ -327,6 +328,10 @@ def add_url(url: Optional[str]): ).all(): names.update(n for n in (result, subtitle, thumb) if n) + # Media a setting holds on to — the video dialog's intro and outro clips, + # watermark and music — is the user's own content, not a leak. + names.update(settings_media_filenames(session, user_id)) + return names diff --git a/backend/app/video/ffmpeg.py b/backend/app/video/ffmpeg.py index 0191c53..e38507e 100644 --- a/backend/app/video/ffmpeg.py +++ b/backend/app/video/ffmpeg.py @@ -659,6 +659,7 @@ def crossfade_total(durations: Sequence[float], overlap: float) -> float: def build_music_command( source: str, music: str, output: str, *, duration: float, spec: MusicSpec, duck: bool, + start: float = 0.0, end: Optional[float] = None, ) -> List[str]: """Mix a background bed under the finished video, without touching the picture. @@ -670,14 +671,28 @@ def build_music_command( `-stream_loop -1` covers a bed shorter than the video and the output `-t` truncates one that is longer. `normalize=0` stops `amix` halving the narration to make room, which is the default and never what anyone wants. + + `start`/`end` narrow the bed to a window of the video — the article between + an intro and an outro, which bring their own sound. The fades land at the + window's edges, and the bed is padded with silence past its end so the mix + and the ducking compressor run on to the end of the narration as before. """ - fade_out_at = max(0.0, duration - max(0.0, spec.fade_out)) + end = duration if end is None else max(0.0, min(end, duration)) + start = max(0.0, min(start, end)) + window = max(0.1, end - start) + fade_out_at = max(0.0, window - max(0.0, spec.fade_out)) bed = (f"[1:a]volume={max(0.0, min(1.0, spec.volume)):.3f}," f"aresample=48000,aformat=channel_layouts=stereo") if spec.fade_in > 0: bed += f",afade=t=in:st=0:d={spec.fade_in:.3f}" if spec.fade_out > 0: bed += f",afade=t=out:st={fade_out_at:.3f}:d={spec.fade_out:.3f}" + if start > 0 or end < duration: + bed += f",atrim=end={window:.3f}" + if start > 0: + delay = int(round(start * 1000)) + bed += f",adelay={delay}|{delay}" + bed += ",apad" chains = [f"{bed}[m]"] if duck: diff --git a/backend/app/video/options.py b/backend/app/video/options.py index 4e25228..c13cf6d 100644 --- a/backend/app/video/options.py +++ b/backend/app/video/options.py @@ -365,6 +365,23 @@ def _sane_scrim(cls, v: float) -> float: return max(0.0, min(1.0, float(v))) +class BumperSpec(BaseModel): + """A pre-made clip played before (intro) or after (outro) the article. + + It plays whole, with its own sound, fitted to the frame like any other clip + in a note — but it is the user's own branding, so nothing of the render's is + drawn over it: no watermark, no text overlay, no waveform, and the music bed + stops short of it. Transitions still apply, so it joins the video the same + way every other segment does. + """ + + enabled: bool = False + url: Optional[str] = None # /media/... video + # The uploaded file's own name. The file is stored under a UUID, so this is + # the only readable label the dialog has for it. + name: str = "" + + class RenderOptions(BaseModel): """The complete render configuration. Persisted as JSON on the job row.""" @@ -391,6 +408,11 @@ class RenderOptions(BaseModel): quotes: QuoteSpec = Field(default_factory=QuoteSpec) code: CodeSpec = Field(default_factory=CodeSpec) + # Clips bracketing the whole video: the intro plays before the title + # screen, the outro after the last section. See BumperSpec. + intro: BumperSpec = Field(default_factory=BumperSpec) + outro: BumperSpec = Field(default_factory=BumperSpec) + # Append the finished video to the note as a playable block. Done by the # worker rather than the browser so a render survives the tab being closed. insert_into_note: bool = True diff --git a/backend/app/video/renderer.py b/backend/app/video/renderer.py index 9dd6d47..592f2ae 100644 --- a/backend/app/video/renderer.py +++ b/backend/app/video/renderer.py @@ -287,12 +287,23 @@ def render( if write_srt(os.path.join(work_dir, name), narration.cues): shot_srt = name - if shot.chapter: + # An intro or outro's "chapter" names the clip, not a section of the + # article, so it must not become the line the overlay follows. + if shot.chapter and not shot.bumper: current_chapter = shot.chapter + # An intro or outro is the user's own branded clip: nothing of the + # render's — watermark, overlay text, waveform — is drawn over it. + shot_options = options + if shot.bumper and options.waveform.enabled: + shot_options = options.model_copy(deep=True) + shot_options.waveform.enabled = False + shot_layer = layer shot_overlay = overlay_name - if dynamic_overlay_text: + if shot.bumper: + shot_layer = shot_overlay = None + elif dynamic_overlay_text: # The intro stretch before any heading is marked with the note's # own title as its "chapter" (see segment()'s title card) — showing # it again underneath the title would just repeat it, so that @@ -349,7 +360,7 @@ def _encode(with_options: RenderOptions) -> None: # A segment encoded by an earlier attempt that died later on (usually # at the stitch) is reused rather than encoded again. key = shot_cache.shot_key( - _argv(options), + _argv(shot_options), [background, narration.path, shot_overlay, shot_srt], work_dir, ) @@ -358,14 +369,14 @@ def _encode(with_options: RenderOptions) -> None: reused += 1 else: try: - _encode(options) + _encode(shot_options) except F.FFmpegError as exc: # Losing a forty-minute render to one expensive segment is a bad # trade when the two most expensive things in it are also the two # least important. Retry once without them; the narration is # already synthesised and cached, so this costs no speech. logger.warning("Segment %d failed (%s) — retrying it plainer", index + 1, exc) - plain = options.model_copy(deep=True) + plain = shot_options.model_copy(deep=True) plain.ken_burns.effect = "none" plain.waveform.enabled = False _encode(plain) @@ -436,6 +447,14 @@ def _encode(with_options: RenderOptions) -> None: final = "stitched.mp4" + # Where the article itself sits on the finished timeline. An intro or + # outro brings its own picture and sound, so the music bed and the poster + # frame are both taken from between them. The intro's last `overlap` + # seconds are already blending into the article; its end is the moment + # the blend has finished. + intro_end = durations[0] if shots[0].bumper == "intro" else 0.0 + outro_length = durations[-1] if shots[-1].bumper == "outro" else 0.0 + # ── background music ────────────────────────────────────────────────── # A bed has to run continuously across shot boundaries, so it can only go # on once the shots are joined. Mixing here re-encodes the audio alone — @@ -452,6 +471,8 @@ def _encode(with_options: RenderOptions) -> None: F.build_music_command( final, music_path, "scored.mp4", duration=scored_length, spec=options.music, duck=duck, + start=max(0.0, intro_end - overlap), + end=scored_length - outro_length, ), cwd=work_dir, timeout=1800, ) @@ -498,7 +519,8 @@ def _encode(with_options: RenderOptions) -> None: F.build_poster_command( os.path.join(user_dir, video_filename), os.path.join(user_dir, thumbnail_filename), - at_seconds=min(1.0, max(0.0, timeline / 2)), + at_seconds=intro_end + min( + 1.0, max(0.0, (timeline - intro_end - outro_length) / 2)), ), timeout=120, ) diff --git a/backend/app/video/segmenter.py b/backend/app/video/segmenter.py index de8a715..a07542e 100644 --- a/backend/app/video/segmenter.py +++ b/backend/app/video/segmenter.py @@ -58,6 +58,9 @@ class Shot: # `narration` above only carries its text when narrate_code is on — see # segment()'s codeBlock branch. code_text: Optional[str] = None + # "intro" or "outro" for a clip bracketing the article (see _bumper_shot). + # The renderer draws nothing of its own over one of these. + bumper: Optional[str] = None # Set on the second half of a sounded-clip pair, purely for readable logs. label: str = "" @@ -561,4 +564,36 @@ def flush(next_shot: Optional[Shot]) -> None: continue trimmed.append(shot) result.shots = trimmed + + # The intro and outro bracket the article rather than stand in for it, so a + # note with nothing of its own to show still renders nothing. + if result.shots: + intro = _bumper_shot("intro", options, user_id, media_dir, result.warnings) + outro = _bumper_shot("outro", options, user_id, media_dir, result.warnings) + if intro is not None: + result.shots.insert(0, intro) + if outro is not None: + result.shots.append(outro) return result + + +def _bumper_shot( + which: str, options: RenderOptions, user_id: str, media_dir: str, warnings: List[str], +) -> Optional[Shot]: + """The intro or outro clip as a shot, or None when it is off or unusable. + + It is a sounded clip like any other in a note: played whole, with its own + audio when it has some and silence when it doesn't, which the renderer + probes for itself. + """ + spec = options.intro if which == "intro" else options.outro + if not spec.enabled or not spec.url: + return None + path = resolve_media_path(spec.url, user_id, media_dir) + if path is None or _media_kind(spec.url) != "video": + warnings.append(f"The {which} was skipped: that clip could not be read.") + return None + return Shot( + kind="video_sound", background=path, + chapter=which.capitalize(), bumper=which, label=which, + ) diff --git a/backend/tests/test_note_assets.py b/backend/tests/test_note_assets.py index 61edfb4..ab52f84 100644 --- a/backend/tests/test_note_assets.py +++ b/backend/tests/test_note_assets.py @@ -30,7 +30,7 @@ release_media_file, sync_note_assets, ) -from app.models import Note, NoteAsset, Theme, User +from app.models import Note, NoteAsset, Theme, User, UserSetting from app.routers.assets import _ai_eligible, _role_for USER = "user-1" @@ -351,6 +351,49 @@ def test_release_keeps_a_file_used_as_a_theme_background(session, media_dir): assert release_media_file(session, USER, url) is False +def _save_setting(session, key, value, user_id=USER): + session.add(UserSetting(user_id=user_id, key=key, value=json.dumps(value))) + session.commit() + + +def test_release_keeps_a_file_a_saved_setting_points_at(session, media_dir): + """The video dialog's intro clip belongs to no note, but it is still in use.""" + url = media_file(media_dir, USER, "intro.mp4") + _save_setting(session, "video_render_options", + {"intro": {"enabled": True, "url": url}, "music": {"url": None}}) + + assert release_media_file(session, USER, url) is False + assert (media_dir / USER / "intro.mp4").exists() + + +def test_settings_media_filenames_reads_nested_values_and_only_the_users_own(session): + _save_setting(session, "video_render_options", { + "intro": {"url": f"/media/{USER}/intro.mp4"}, + "outro": {"url": f"/media/{USER}/outro.mp4"}, + "watermark": {"url": f"/media/{OTHER_USER}/theirs.png"}, + "diagram_images": {"b1": f"/media/{USER}/diagram.png"}, + "list": [f"/media/{USER}/in-a-list.mp3", "not a url", 3, None], + }) + _save_setting(session, "video_render_options", {"intro": {"url": f"/media/{OTHER_USER}/x.mp4"}}, + user_id=OTHER_USER) + session.add(UserSetting(user_id=USER, key="broken", value="/media/{not json")) + session.commit() + + assert asset_utils.settings_media_filenames(session, USER) == { + "intro.mp4", "outro.mp4", "diagram.png", "in-a-list.mp3", + } + + +def test_the_unlinked_sweep_does_not_offer_up_a_settings_file(session, media_dir): + from app.routers.assets import _referenced_filenames + + url = media_file(media_dir, USER, "outro.mp4") + _save_setting(session, "video_render_options", {"outro": {"enabled": False, "url": url}}) + + # Switched off but still chosen: the clip is kept for when it's switched back on. + assert "outro.mp4" in _referenced_filenames(session, USER) + + def test_release_never_touches_another_users_file(session, media_dir): """Content pasted from a shared note points into the author's media dir, not ours.""" url = media_file(media_dir, OTHER_USER, "theirs.png") diff --git a/backend/tests/test_video_bumpers.py b/backend/tests/test_video_bumpers.py new file mode 100644 index 0000000..9200478 --- /dev/null +++ b/backend/tests/test_video_bumpers.py @@ -0,0 +1,205 @@ +"""Intro and outro clips, driven through the real `render()`. + +ffmpeg is faked the way test_video_render_resume.py fakes it: every command +"succeeds" by writing its output file, and the argv is kept so the tests can +read what each step was asked to do. Durations are probed from a table so the +timeline arithmetic — where the music starts and stops, where the poster frame +is taken — comes out at numbers worth asserting. +""" + +import json +import os + +from app.video import ffmpeg as F +from app.video.options import RenderOptions +from app.video.renderer import render + +INTRO_SECONDS = 3.0 +OUTRO_SECONDS = 4.0 +STITCHED_SECONDS = 20.0 + + +class FakeFFmpeg: + def __init__(self, monkeypatch): + self.calls = [] + self.chapters = "" + monkeypatch.setattr(F, "ffmpeg_available", lambda: True) + monkeypatch.setattr(F, "available_filters", lambda: frozenset( + {"xfade", "concat", "zoompan", "subtitles", "showwaves", "sidechaincompress"})) + monkeypatch.setattr(F, "probe_duration", self.probe_duration) + monkeypatch.setattr(F, "probe_has_audio", lambda path: "intro" in path) + monkeypatch.setattr(F, "run", self.run) + + @staticmethod + def probe_duration(path): + name = os.path.basename(path) + if name.startswith("intro"): + return INTRO_SECONDS + if name.startswith("outro"): + return OUTRO_SECONDS + if name == "stitched.mp4": + return STITCHED_SECONDS + return 2.0 + + def run(self, argv, *, cwd=None, timeout=0): + self.calls.append(list(argv)) + if "chapters.ffmeta" in argv: + with open(os.path.join(cwd or ".", "chapters.ffmeta"), encoding="utf-8") as f: + self.chapters = f.read() + open(os.path.join(cwd or ".", argv[-1]), "wb").write(b"out") + + def encode_of(self, clip): + return next(c for c in self.calls if os.path.basename(c[-1]).startswith("shot_") + and any(a.endswith(clip) for a in c)) + + def body_encodes(self): + return [c for c in self.calls if os.path.basename(c[-1]).startswith("shot_") + and not any(a.endswith(("intro.mp4", "outro.mp4")) for a in c)] + + def command_writing(self, name): + return next(c for c in self.calls if os.path.basename(c[-1]) == name) + + +def _graph(argv): + return argv[argv.index("-filter_complex") + 1] + + +def _setup(media): + user_dir = media / "u1" + user_dir.mkdir(exist_ok=True) + for name in ("intro.mp4", "outro.mp4", "a.png", "bed.mp3"): + (user_dir / name).write_bytes(b"x") + text = lambda t: [{"type": "text", "text": t}] + return json.dumps([ + {"id": "1", "type": "image", "props": {"url": "/media/u1/a.png"}}, + {"id": "2", "type": "paragraph", "content": text("The article itself.")}, + ]) + + +def _options(**overrides): + return RenderOptions(**{ + "title_card": False, + "intro": {"enabled": True, "url": "/media/u1/intro.mp4", "name": "My intro.mp4"}, + "outro": {"enabled": True, "url": "/media/u1/outro.mp4", "name": "My outro.mp4"}, + **overrides, + }) + + +def _render(media, options): + return render( + job_id="j1", user_id="u1", media_dir=str(media), note_content=_setup(media), + note_title="Note", author="", options=options, preview=True, + tts=lambda text: b"audio", progress=lambda *a: None, + max_shots=50, max_narration_chars=100_000, + ) + + +def test_the_intro_opens_the_video_and_the_outro_closes_it(tmp_path, monkeypatch): + ff = FakeFFmpeg(monkeypatch) + _render(tmp_path, _options()) + + playlist = ff.command_writing("stitched.mp4") + assert "shots.txt" in playlist + encodes = [c[-1] for c in ff.calls if os.path.basename(c[-1]).startswith("shot_")] + assert encodes == ["shot_0000.mp4", "shot_0001.mp4", "shot_0002.mp4"] + assert ff.encode_of("intro.mp4")[-1] == "shot_0000.mp4" + assert ff.encode_of("outro.mp4")[-1] == "shot_0002.mp4" + + +def test_a_bumper_plays_with_its_own_sound_or_with_silence(tmp_path, monkeypatch): + ff = FakeFFmpeg(monkeypatch) + _render(tmp_path, _options()) + + # The intro has an audio track and keeps it; the outro has none and gets + # silence in its place, exactly like a sounded clip in the note itself. + assert "[0:a]" in _graph(ff.encode_of("intro.mp4")) + assert "anullsrc" in " ".join(ff.encode_of("outro.mp4")) + + +def test_nothing_of_the_render_is_drawn_over_a_bumper(tmp_path, monkeypatch): + ff = FakeFFmpeg(monkeypatch) + _render(tmp_path, _options( + watermark={"enabled": True, "text": "by me"}, + overlay_text={"enabled": True, "mode": "fixed", "text": "Channel"}, + waveform={"enabled": True}, + )) + + for clip in ("intro.mp4", "outro.mp4"): + argv = ff.encode_of(clip) + assert not any(a.endswith(".png") for a in argv), f"{clip} got an overlay input" + assert "showwaves" not in _graph(argv) + # ...while the article in between still gets all three. + (body,) = ff.body_encodes() + assert any(a.endswith(".png") and "overlay" in a for a in body) + assert "showwaves" in _graph(body) + + +def test_the_title_chapter_overlay_never_follows_a_bumpers_chapter(tmp_path, monkeypatch): + """An intro's "Intro" mark must not become the chapter line under the title.""" + ff = FakeFFmpeg(monkeypatch) + drawn = [] + from app.video import compose + real = compose.overlay_layer + + def spy(*args, chapter_line=None, **kwargs): + drawn.append(chapter_line) + return real(*args, chapter_line=chapter_line, **kwargs) + + monkeypatch.setattr(compose, "overlay_layer", spy) + _render(tmp_path, _options(overlay_text={"enabled": True, "mode": "title_chapter"})) + + assert "Intro" not in drawn and "Outro" not in drawn + assert ff.body_encodes() + + +def test_the_music_bed_scores_only_the_article(tmp_path, monkeypatch): + ff = FakeFFmpeg(monkeypatch) + _render(tmp_path, _options( + music={"enabled": True, "url": "/media/u1/bed.mp3", "fade_in": 0, "fade_out": 0}, + )) + + graph = _graph(ff.command_writing("scored.mp4")) + # It comes in when the intro ends and is cut where the outro begins. + window = STITCHED_SECONDS - INTRO_SECONDS - OUTRO_SECONDS + delay = int(INTRO_SECONDS * 1000) + assert f"atrim=end={window:.3f},adelay={delay}|{delay},apad" in graph + + +def test_without_bumpers_the_music_bed_is_unchanged(tmp_path, monkeypatch): + ff = FakeFFmpeg(monkeypatch) + _render(tmp_path, RenderOptions( + title_card=False, music={"enabled": True, "url": "/media/u1/bed.mp3"}, + )) + + graph = _graph(ff.command_writing("scored.mp4")) + assert "atrim" not in graph and "adelay" not in graph + + +def test_a_crossfade_starts_the_music_as_the_intro_blends_away(tmp_path, monkeypatch): + ff = FakeFFmpeg(monkeypatch) + _render(tmp_path, _options( + transition={"style": "dissolve", "duration": 0.5}, + music={"enabled": True, "url": "/media/u1/bed.mp3", "fade_in": 0, "fade_out": 0}, + )) + + assert ff.command_writing("stitched.mp4")[-1] == "stitched.mp4" + graph = _graph(ff.command_writing("scored.mp4")) + assert "adelay=2500|2500" in graph + + +def test_the_intro_and_outro_are_marked_as_chapters(tmp_path, monkeypatch): + ff = FakeFFmpeg(monkeypatch) + _render(tmp_path, _options()) + + titles = [line.split("=", 1)[1] for line in ff.chapters.splitlines() + if line.startswith("title=")] + assert titles[0] == "Intro" and titles[-1] == "Outro" + + +def test_the_poster_frame_is_taken_from_the_article_not_the_intro(tmp_path, monkeypatch): + ff = FakeFFmpeg(monkeypatch) + result = _render(tmp_path, _options()) + + poster = ff.command_writing(result.thumbnail_filename) + seek = float(poster[poster.index("-ss") + 1]) + assert seek == INTRO_SECONDS + 1.0 diff --git a/backend/tests/test_video_ffmpeg.py b/backend/tests/test_video_ffmpeg.py index ad60d55..693a178 100644 --- a/backend/tests/test_video_ffmpeg.py +++ b/backend/tests/test_video_ffmpeg.py @@ -1342,3 +1342,42 @@ def test_emoji_are_stripped_from_short_chunks_too(): joined = "".join(c.text for c in chunks) assert "\U0001F697" not in joined assert joined == "Drive the car home." + + +def test_a_bed_without_a_window_is_not_trimmed_or_delayed(): + graph = _graph(F.build_music_command("in.mp4", "b.mp3", "o.mp4", + duration=40.0, spec=_music(), duck=False)) + assert "atrim" not in graph and "adelay" not in graph and "apad" not in graph + + +def test_a_bed_windowed_between_an_intro_and_an_outro(): + """It starts after the intro, stops before the outro, and fades at those edges.""" + graph = _graph(F.build_music_command( + "in.mp4", "b.mp3", "o.mp4", duration=40.0, + spec=_music(fade_in=2.0, fade_out=3.0), duck=False, start=5.0, end=32.0)) + assert "afade=t=in:st=0:d=2.000" in graph + # 27 seconds of bed, faded over its own last three. + assert "afade=t=out:st=24.000:d=3.000" in graph + assert "atrim=end=27.000,adelay=5000|5000,apad[m]" in graph + # The output still runs the whole video, intro and outro included. + argv = F.build_music_command("in.mp4", "b.mp3", "o.mp4", duration=40.0, + spec=_music(), duck=False, start=5.0, end=32.0) + assert argv[argv.index("-t") + 1] == "40.000" + + +def test_a_bed_stopping_before_an_outro_needs_no_delay(): + graph = _graph(F.build_music_command( + "in.mp4", "b.mp3", "o.mp4", duration=40.0, + spec=_music(fade_out=0.0), duck=True, end=30.0)) + assert "atrim=end=30.000,apad[m]" in graph + assert "adelay" not in graph + # Padded rather than ended, so the ducking compressor keeps running. + assert "[m][key]sidechaincompress=" in graph + + +def test_a_window_out_of_range_is_clamped_to_the_video(): + graph = _graph(F.build_music_command( + "in.mp4", "b.mp3", "o.mp4", duration=10.0, + spec=_music(fade_out=0.0), duck=False, start=12.0, end=50.0)) + # Nothing left to score: a sliver, never a negative trim. + assert "atrim=end=0.100" in graph diff --git a/backend/tests/test_video_segmenter.py b/backend/tests/test_video_segmenter.py index 7ef071b..e600681 100644 --- a/backend/tests/test_video_segmenter.py +++ b/backend/tests/test_video_segmenter.py @@ -789,3 +789,86 @@ def test_markers_are_not_counted_as_narration_characters(): marked = _run([{"type": "paragraph", "content": _text("Hello [pause:2s] world.")}], media_root=root) assert marked.narration_chars == plain.narration_chars + + +# ── intro and outro ────────────────────────────────────────────────────────── + +def _bracketed(**overrides): + return RenderOptions(**{ + "title_card": True, + "intro": {"enabled": True, "url": "/media/u1/intro.mp4"}, + "outro": {"enabled": True, "url": "/media/u1/outro.mp4"}, + **overrides, + }) + + +def test_intro_and_outro_bracket_the_whole_article(): + root = _media("intro.mp4", "outro.mp4", "a.png") + plan = _run([ + {"id": "1", "type": "image", "props": {"url": "/media/u1/a.png"}}, + {"id": "2", "type": "paragraph", "content": _text("The article.")}, + ], media_root=root, options=_bracketed(), title="Note") + + assert [s.bumper for s in plan.shots] == ["intro", None, None, "outro"] + # The intro plays before even the title screen. + assert plan.shots[1].card_kind == "title" + intro, outro = plan.shots[0], plan.shots[-1] + assert intro.kind == outro.kind == "video_sound" + assert intro.background.endswith("intro.mp4") + assert outro.background.endswith("outro.mp4") + # A clip is never narrated over, and each is its own chapter in the MP4. + assert intro.narration == outro.narration == "" + assert (intro.chapter, outro.chapter) == ("Intro", "Outro") + assert plan.warnings == [] + + +def test_a_bumper_is_played_whether_or_not_it_carries_sound(): + """The renderer probes for audio itself; the segmenter must not drop a silent clip.""" + root = _media("intro.mp4", "outro.mp4") + plan = _run([{"id": "1", "type": "paragraph", "content": _text("Words.")}], + media_root=root, options=_bracketed(title_card=False), loud=()) + + assert [s.bumper for s in plan.shots] == ["intro", None, "outro"] + + +def test_a_switched_off_bumper_is_left_out_even_with_a_clip_chosen(): + root = _media("intro.mp4", "outro.mp4") + options = _bracketed(title_card=False) + options.outro.enabled = False + plan = _run([{"id": "1", "type": "paragraph", "content": _text("Words.")}], + media_root=root, options=options) + + assert [s.bumper for s in plan.shots] == ["intro", None] + + +def test_a_missing_or_foreign_bumper_is_skipped_with_a_warning(): + root = _media("a.png") + options = RenderOptions( + title_card=False, + intro={"enabled": True, "url": "/media/u1/gone.mp4"}, + outro={"enabled": True, "url": "/media/u2/theirs.mp4"}, + ) + plan = _run([{"id": "1", "type": "paragraph", "content": _text("Words.")}], + media_root=root, options=options) + + assert [s.bumper for s in plan.shots] == [None] + assert any("intro was skipped" in w for w in plan.warnings) + assert any("outro was skipped" in w for w in plan.warnings) + + +def test_a_still_image_is_not_accepted_as_a_bumper(): + root = _media("logo.png") + options = RenderOptions(title_card=False, + intro={"enabled": True, "url": "/media/u1/logo.png"}) + plan = _run([{"id": "1", "type": "paragraph", "content": _text("Words.")}], + media_root=root, options=options) + + assert [s.bumper for s in plan.shots] == [None] + assert any("intro was skipped" in w for w in plan.warnings) + + +def test_bumpers_alone_do_not_make_an_empty_note_renderable(): + root = _media("intro.mp4", "outro.mp4") + plan = _run([], media_root=root, options=_bracketed(title_card=False)) + + assert plan.shots == [] diff --git a/frontend/src/api/videoGen.ts b/frontend/src/api/videoGen.ts index 1c145d5..8d72ae1 100644 --- a/frontend/src/api/videoGen.ts +++ b/frontend/src/api/videoGen.ts @@ -73,6 +73,16 @@ export interface CardTextSizes { subtitle_pct: number } +/** A pre-made clip played before (intro) or after (outro) the article — whole, + * with its own sound, and with nothing of the render's drawn over it. */ +export interface BumperClip { + enabled: boolean + /** /media/... video in the user's own media. */ + url: string | null + /** The uploaded file's name; the stored file is named by a UUID. */ + name: string +} + export interface RenderOptions { aspect: AspectRatio resolution: VideoResolution @@ -134,6 +144,8 @@ export interface RenderOptions { size_pct: number color: string; scrim: number } + intro: BumperClip + outro: BumperClip insert_into_note: boolean title_card: boolean chapter_screens: boolean @@ -176,6 +188,8 @@ export const DEFAULT_RENDER_OPTIONS: RenderOptions = { music: { enabled: false, url: null, volume: 0.18, duck: true, fade_in: 1.5, fade_out: 3.0 }, quotes: { enabled: false, position: 'center', size_pct: 4.2, color: '#ffffff', accent: '#818cf8', scrim: 0.55 }, code: { position: 'center', size_pct: 3.4, color: '#e2e8f0', scrim: 0.72 }, + intro: { enabled: false, url: null, name: '' }, + outro: { enabled: false, url: null, name: '' }, insert_into_note: true, title_card: true, chapter_screens: false, diff --git a/frontend/src/components/VideoGenModal.tsx b/frontend/src/components/VideoGenModal.tsx index 07f8c19..17116ac 100644 --- a/frontend/src/components/VideoGenModal.tsx +++ b/frontend/src/components/VideoGenModal.tsx @@ -1,17 +1,30 @@ -import { useEffect, useMemo, useState } from 'react' +import { useEffect, useMemo, useRef, useState } from 'react' import { createPortal } from 'react-dom' import { X, Clapperboard, Loader2, Upload } from 'lucide-react' import { mediaApi } from '@/api/media' import { settingsApi } from '@/api/settings' import { DEFAULT_RENDER_OPTIONS, isCrossfade, videoGenApi, - type AspectRatio, type FitMode, type KenBurnsEffect, type OverlayPosition, + type AspectRatio, type BumperClip, type FitMode, type KenBurnsEffect, type OverlayPosition, type OverlayTextMode, type QuotePosition, type RenderOptions, type SubtitleMode, type TransitionStyle, type VideoEstimate, type VideoResolution, type WaveMode, type WavePosition, } from '@/api/videoGen' +import { useSettingsStore } from '@/stores/settings' import { apiErrorMessage } from '@/utils/format' const OPTIONS_KEY = 'gecko-video-gen-options' +/** The account setting the options are saved under. Saved there rather than only + * in this browser so they follow the user to any device — and so the server + * knows the intro, outro, watermark and music files they point at are still in + * use and never offers them up as leaked. */ +const SETTINGS_KEY = 'video_render_options' +/** Long enough that dragging a slider saves once, at the end. */ +const SAVE_DELAY_MS = 800 + +type UploadKind = 'watermark' | 'music' | 'intro' | 'outro' +const UPLOAD_NOUN: Record = { + watermark: 'image', music: 'track', intro: 'clip', outro: 'clip', +} interface Props { noteId: string @@ -22,6 +35,19 @@ interface Props { onClose: () => void } +/** The last saved options: the account's copy, else this browser's older local + * one (which is all there was before they were saved to the account). */ +function readStoredOptions(): Partial | null { + const saved = useSettingsStore.getState().appSettings[SETTINGS_KEY] + if (saved && typeof saved === 'object' && !Array.isArray(saved)) return saved as Partial + try { + const raw = localStorage.getItem(OPTIONS_KEY) + return raw ? JSON.parse(raw) as Partial : null + } catch { + return null // unparseable or private mode — defaults are fine + } +} + /** Merge a stored blob over the defaults, field by field. * * Persisted options outlive the shape that wrote them, so anything missing or @@ -30,9 +56,8 @@ interface Props { function loadStoredOptions(): RenderOptions { const base: RenderOptions = structuredClone(DEFAULT_RENDER_OPTIONS) try { - const raw = localStorage.getItem(OPTIONS_KEY) - if (!raw) return base - const stored = JSON.parse(raw) as Partial + const stored = readStoredOptions() + if (!stored || typeof stored !== 'object') return base for (const key of Object.keys(base) as (keyof RenderOptions)[]) { const value = stored[key] if (value === undefined || value === null) continue @@ -71,19 +96,75 @@ function SizeSlider({ label, value, min, max, onChange }: { ) } +/** An intro or outro: switched on, a clip uploaded into the user's own media, + * and a player to check it with. */ +function ClipPicker({ label, hint, clip, uploading, busy, onUpload, onChange }: { + label: string + hint: string + clip: BumperClip + /** This picker's upload is in flight. */ + uploading: boolean + /** Any upload is in flight, so a second one can't start. */ + busy: boolean + onUpload: (file: File) => void + onChange: (changes: Partial) => void +}) { + return ( +
+ + {clip.enabled && ( +
+
+ + {clip.url && ( + <> + + {clip.name || 'Uploaded clip'} + + + + )} +
+ {clip.url && ( +
+ )} +
+ ) +} + const ASPECTS: { id: AspectRatio; label: string; hint: string }[] = [ { id: '16:9', label: '16:9', hint: 'YouTube' }, { id: '9:16', label: '9:16', hint: 'Shorts / TikTok' }, { id: '1:1', label: '1:1', hint: 'Instagram' }, ] -type TabId = 'format' | 'narration' | 'motion' | 'branding' | 'structure' +type TabId = 'format' | 'narration' | 'motion' | 'branding' | 'bumpers' | 'structure' const TABS: { id: TabId; label: string }[] = [ { id: 'format', label: 'Format' }, { id: 'narration', label: 'Narration' }, { id: 'motion', label: 'Motion & audio' }, { id: 'branding', label: 'Branding' }, + { id: 'bumpers', label: 'Intro & outro' }, { id: 'structure', label: 'Structure' }, ] @@ -111,7 +192,7 @@ export default function VideoGenModal({ noteId, noteTitle, diagramImages, onGene const [estimate, setEstimate] = useState(null) const [voices, setVoices] = useState([]) const [busy, setBusy] = useState<'preview' | 'full' | null>(null) - const [uploading, setUploading] = useState<'watermark' | 'music' | null>(null) + const [uploading, setUploading] = useState(null) const [error, setError] = useState(null) const payload = useMemo( @@ -119,13 +200,42 @@ export default function VideoGenModal({ noteId, noteTitle, diagramImages, onGene [options, diagramImages], ) - // Persist as the user goes, so the next video starts from the last one's setup. + // Persist as the user goes, so the next video starts from the last one's + // setup: to this browser at once, and to the account once the user pauses. + // Opening the dialog saves nothing — only a change does — so a stale local + // copy is never pushed over what another device saved. + const savedJson = useRef(null) + const pendingJson = useRef(null) + + function flushSave() { + const json = pendingJson.current + if (json === null) return + pendingJson.current = null + savedJson.current = json + useSettingsStore.getState() + .updateAppSettings({ [SETTINGS_KEY]: JSON.parse(json) }) + // This browser still has it, and the next change tries the account again. + .catch(() => { savedJson.current = '' }) + } + useEffect(() => { + const json = JSON.stringify({ ...options, diagram_images: {} }) try { - localStorage.setItem(OPTIONS_KEY, JSON.stringify({ ...options, diagram_images: {} })) + localStorage.setItem(OPTIONS_KEY, json) } catch { /* private mode */ } + if (savedJson.current === null) { + savedJson.current = json // what was just loaded + return + } + if (json === savedJson.current) return + pendingJson.current = json + const timer = setTimeout(flushSave, SAVE_DELAY_MS) + return () => clearTimeout(timer) }, [options]) + // Closing the dialog mid-pause still saves the last change. + useEffect(() => () => flushSave(), []) + useEffect(() => { let cancelled = false settingsApi.getSpeechSettings() @@ -153,6 +263,8 @@ export default function VideoGenModal({ noteId, noteTitle, diagramImages, onGene payload.shot_end_pause_ms, payload.heading_pause_ms, payload.paragraph_pause_ms, payload.transition.style, payload.transition.duration, + // The intro and outro are segments of their own and count toward the length. + payload.intro.enabled, payload.intro.url, payload.outro.enabled, payload.outro.url, ]) function patch(changes: Partial) { @@ -161,18 +273,25 @@ export default function VideoGenModal({ noteId, noteTitle, diagramImages, onGene type GroupKey = 'waveform' | 'watermark' | 'overlay_text' | 'fallback' | 'title_card_text' | 'chapter_card_text' | 'transition' | 'ken_burns' | 'music' | 'quotes' | 'code' + | 'intro' | 'outro' function patchGroup(group: K, changes: Partial) { setOptions((prev) => ({ ...prev, [group]: { ...prev[group], ...changes } })) } - async function upload(kind: 'watermark' | 'music', file: File) { + /** Uploads land in the user's own media, not in this note: they're reused for + * every video, and the saved options are what keep them. */ + async function upload(kind: UploadKind, file: File) { setUploading(kind) setError(null) try { const res = await mediaApi.upload(file) - patchGroup(kind, { url: res.data.url, enabled: true }) + if (kind === 'intro' || kind === 'outro') { + patchGroup(kind, { url: res.data.url, name: file.name, enabled: true }) + } else { + patchGroup(kind, { url: res.data.url, enabled: true }) + } } catch (e) { - setError(apiErrorMessage(e, `Could not upload that ${kind === 'music' ? 'track' : 'image'}`)) + setError(apiErrorMessage(e, `Could not upload that ${UPLOAD_NOUN[kind]}`)) } finally { setUploading(null) } @@ -663,6 +782,40 @@ export default function VideoGenModal({ noteId, noteTitle, diagramImages, onGene )} + {/* ── Intro & outro ────────────────────────────────────────────── */} + {tab === 'bumpers' && ( + <> + void upload('intro', f)} + onChange={(changes) => patchGroup('intro', changes)} + /> +
+ void upload('outro', f)} + onChange={(changes) => patchGroup('outro', changes)} + /> +
+

+ Each clip plays whole, with its own sound, fitted to the frame + like any other video. Nothing is drawn over it — no watermark, + text overlay or waveform — and background music stops short of + it. Transitions still join it to the rest of the video. Clips + are kept in your media and saved with these settings, so they're + used for every video until you change them. +

+ + )} + {/* ── Structure ────────────────────────────────────────────────── */} {tab === 'structure' && ( <>