Removing the narration_combined
This commit is contained in:
@@ -1,14 +1,14 @@
|
|||||||
{
|
{
|
||||||
"talking_head_S1": {
|
"talking_head_S1": {
|
||||||
"source_file": "talking_head_S1.mov",
|
"source_file": "talking_head_S1.mov",
|
||||||
"output_file": "talking_head_S1_processed.mov",
|
"output_file": "processed/talking_head_S1_processed.mov",
|
||||||
"cutout": "talkinghead",
|
"cutout": "talkinghead",
|
||||||
"always_visible": true,
|
"always_visible": true,
|
||||||
"filter": "talkinghead"
|
"filter": "talkinghead"
|
||||||
},
|
},
|
||||||
"talking_head_S3": {
|
"talking_head_S3": {
|
||||||
"source_file": "talking_head_S3.mov",
|
"source_file": "talking_head_S3.mov",
|
||||||
"output_file": "talking_head_S3_processed.mov",
|
"output_file": "processed/talking_head_S3_processed.mov",
|
||||||
"cutout": "talkinghead",
|
"cutout": "talkinghead",
|
||||||
"always_visible": true,
|
"always_visible": true,
|
||||||
"filter": "talkinghead"
|
"filter": "talkinghead"
|
||||||
|
|||||||
@@ -9,7 +9,7 @@
|
|||||||
"keynote_file": "media/example.key",
|
"keynote_file": "media/example.key",
|
||||||
"transcript": "media/videos/talking_head.transcript.json",
|
"transcript": "media/videos/talking_head.transcript.json",
|
||||||
"narration": "media/narration/narration.json",
|
"narration": "media/narration/narration.json",
|
||||||
"background": "shared_assets/solarpunk.png",
|
"background": "blackbackground",
|
||||||
"videos": "media/videos/videos.json",
|
"videos": "media/videos/videos.json",
|
||||||
"slides": "media/slides/Example/slides.json",
|
"slides": "media/slides/Example/slides.json",
|
||||||
"audio": "media/audio/audio.json",
|
"audio": "media/audio/audio.json",
|
||||||
|
|||||||
@@ -0,0 +1,13 @@
|
|||||||
|
# Tasks: example
|
||||||
|
_Generated: 2026-07-15_
|
||||||
|
|
||||||
|
## Slide Alignment Issues (7)
|
||||||
|
Slide markers that could not be matched to the spoken narration (likely adlibbed).
|
||||||
|
|
||||||
|
- [ ] `S6` — _"(repaired: voice continues after the video finished)"_
|
||||||
|
- [ ] `S7` — _"This is the first slide. It appears immediately."_
|
||||||
|
- [ ] `S8` — _"However, this is the second slide. It should appea"_
|
||||||
|
- [ ] `S9` — _"This is me talking alongside a video. The video is"_
|
||||||
|
- [ ] `S10` — _"I will continue to talk without pause, but in the"_
|
||||||
|
- [ ] `S11` — _"Notice how my voice continues after the video fini"_
|
||||||
|
- [ ] `S12` — _"(repaired: voice continues after the video finished)"_
|
||||||
+100
-118
@@ -4221,6 +4221,42 @@ def _chunked_render(
|
|||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
def _build_merged_transcription(project_path: Path, config, verbose: bool = False):
|
||||||
|
"""Deterministic merged transcript for slide alignment.
|
||||||
|
|
||||||
|
Builds a single word-level transcript from the per-segment transcripts +
|
||||||
|
the current narration.json skip/take, re-timed into the combined timeline
|
||||||
|
(see narration.build_narration_schedule). This keeps alignment in sync with
|
||||||
|
narration.json and avoids re-transcribing the combined file.
|
||||||
|
|
||||||
|
Returns None (caller falls back to the on-disk transcript) when there are no
|
||||||
|
narration segments or any segment is missing its per-segment transcript.
|
||||||
|
"""
|
||||||
|
from .parser import parse_narration, get_video_duration
|
||||||
|
from .narration import build_narration_schedule
|
||||||
|
|
||||||
|
try:
|
||||||
|
narration, narration_dir = parse_narration(project_path, config)
|
||||||
|
except GnommoError:
|
||||||
|
return None
|
||||||
|
if not narration:
|
||||||
|
return None
|
||||||
|
|
||||||
|
transcripts_dir = narration_dir / "transcripts"
|
||||||
|
missing = [sid for sid in narration if not (transcripts_dir / f"{sid}.json").exists()]
|
||||||
|
if missing:
|
||||||
|
if verbose:
|
||||||
|
print(f" Merged transcript unavailable (no per-segment transcript for: "
|
||||||
|
f"{', '.join(missing)}) — using on-disk transcript.")
|
||||||
|
return None
|
||||||
|
|
||||||
|
_segments, merged = build_narration_schedule(
|
||||||
|
narration, narration_dir, get_video_duration,
|
||||||
|
transcripts_dir=transcripts_dir, verbose=verbose,
|
||||||
|
)
|
||||||
|
return merged or None
|
||||||
|
|
||||||
|
|
||||||
def cmd_render(
|
def cmd_render(
|
||||||
project_path: Path,
|
project_path: Path,
|
||||||
verbose: bool,
|
verbose: bool,
|
||||||
@@ -4303,129 +4339,59 @@ def cmd_render(
|
|||||||
print(f" Using {res} dir: {videos_dir}")
|
print(f" Using {res} dir: {videos_dir}")
|
||||||
audio, audio_dir = parse_audio(project_path, config)
|
audio, audio_dir = parse_audio(project_path, config)
|
||||||
|
|
||||||
# Load whisper transcription JSON
|
# --- Narration: render-time concat of the processed segments ---
|
||||||
# Resolve the combined narration skeleton. The .mov file — not the videos.json
|
# narration.json is the single source of truth. The processed segments are
|
||||||
# entry — is the source of truth: stitch run without the external drive leaves
|
# concatenated directly in the render graph — there is no narration_combined
|
||||||
# the file in media/videos/ even when the videos.json entry is absent (e.g. a
|
# file anymore.
|
||||||
# metadata pull overwrote it). Legacy multi-segment projects are handled below.
|
from .narration import build_narration_schedule
|
||||||
combined_path = videos_dir / "narration_combined.mov"
|
from .parser import parse_narration as _parse_narr, get_video_duration
|
||||||
resolved_combined = _resolve_narration_combined(project_path, videos_dir, config)
|
|
||||||
narration_json = project_path / "media" / "narration" / "narration.json"
|
|
||||||
_narr_segments = _read_json(narration_json) if narration_json.exists() else {}
|
|
||||||
if resolved_combined and resolved_combined.exists():
|
|
||||||
# File is available (locally or via process cache). Ensure a videos.json
|
|
||||||
# entry exists — synthesizing one when stitch's entry was lost — then use it.
|
|
||||||
if "narration_combined" not in videos:
|
|
||||||
from .models import VideoSource
|
|
||||||
|
|
||||||
_first_seg = next(iter(_narr_segments.values()), {})
|
narration_map, narration_seg_dir = _parse_narr(project_path, config)
|
||||||
_seg_cutout = (
|
narration_schedule: list = []
|
||||||
_first_seg.get("cutout") if isinstance(_first_seg, dict) else None
|
narration_source = None
|
||||||
)
|
transcript_path = None
|
||||||
videos["narration_combined"] = VideoSource(
|
if narration_map:
|
||||||
source_file="narration_combined.mov",
|
narration_schedule, _ = build_narration_schedule(
|
||||||
cutout=_seg_cutout or "talkinghead",
|
narration_map, narration_seg_dir, get_video_duration
|
||||||
always_visible=True,
|
)
|
||||||
volume=1.0,
|
missing = [s.seg_id for s in narration_schedule if not s.source_path.exists()]
|
||||||
)
|
if missing:
|
||||||
if resolved_combined != combined_path:
|
|
||||||
# File lives on external disk — point the VideoSource at the absolute
|
|
||||||
# path so the renderer doesn't re-resolve it via the local videos_dir.
|
|
||||||
videos["narration_combined"].source_file = str(resolved_combined)
|
|
||||||
transcript_path = resolved_combined.with_suffix(".transcript.json")
|
|
||||||
config.main_video = "narration_combined"
|
|
||||||
if verbose:
|
|
||||||
print(
|
print(
|
||||||
f" Using combined narration: {resolved_combined.name} (volume={videos['narration_combined'].volume})"
|
f"Error: processed narration segment(s) not found: {', '.join(missing)}",
|
||||||
)
|
|
||||||
elif isinstance(config.main_video, list) and len(config.main_video) > 1:
|
|
||||||
# Legacy: Multi-segment narration with main_video array in project.json
|
|
||||||
resolved_combined, _ = resolve_with_cache(combined_path, project_path)
|
|
||||||
transcript_path = resolved_combined.with_suffix(".transcript.json")
|
|
||||||
|
|
||||||
if not resolved_combined.exists():
|
|
||||||
print(
|
|
||||||
f"Error: Combined narration not found: {combined_path}", file=sys.stderr
|
|
||||||
)
|
|
||||||
print(
|
|
||||||
"Run 'gnommo -p <project> concat' first to concatenate segments.",
|
|
||||||
file=sys.stderr,
|
file=sys.stderr,
|
||||||
)
|
)
|
||||||
|
print(f"Run 'gnommo -p {project_path.name} preprocess' first.", file=sys.stderr)
|
||||||
return 1
|
return 1
|
||||||
|
# Talking-head cutout/zoom/audio settings come from the first segment.
|
||||||
|
narration_source = narration_map[narration_schedule[0].seg_id]
|
||||||
|
|
||||||
# Create a synthetic video entry for the combined narration
|
# --- Transcript for slide alignment ---
|
||||||
# Inherit settings from the first segment
|
# Prefer the deterministic merged transcript (per-segment transcripts re-timed
|
||||||
first_segment_id = config.main_video[0]
|
# into the concatenated timeline). Otherwise fall back to an on-disk transcript.
|
||||||
if first_segment_id in videos:
|
transcription = _build_merged_transcription(project_path, config, verbose)
|
||||||
first_segment = videos[first_segment_id]
|
if transcription is not None:
|
||||||
from .models import VideoSource
|
|
||||||
|
|
||||||
combined_video = VideoSource(
|
|
||||||
source_file="narration_combined.mov",
|
|
||||||
filter=first_segment.filter,
|
|
||||||
output_file=None, # Already processed
|
|
||||||
cutout=first_segment.cutout,
|
|
||||||
always_visible=True,
|
|
||||||
skip=0.0, # Already trimmed during concatenation
|
|
||||||
take=None,
|
|
||||||
)
|
|
||||||
videos["_narration_combined"] = combined_video
|
|
||||||
config.main_video = "_narration_combined"
|
|
||||||
|
|
||||||
if verbose:
|
if verbose:
|
||||||
print(f" Using combined narration: {combined_path.name}")
|
print(f" Using merged per-segment transcript ({len(transcription)} words)")
|
||||||
elif _narr_segments:
|
|
||||||
# narration.json has segments, but the combined .mov could not be found.
|
|
||||||
# Distinguish "stitched, file unreachable" from "never stitched" so the
|
|
||||||
# hint is actionable instead of always blaming videos.json.
|
|
||||||
if "narration_combined" in videos:
|
|
||||||
print(
|
|
||||||
f"Error: narration_combined.mov could not be found.", file=sys.stderr
|
|
||||||
)
|
|
||||||
print(
|
|
||||||
f"videos.json references narration_combined, but the file is not on disk "
|
|
||||||
f"(checked local media/videos and the process cache).",
|
|
||||||
file=sys.stderr,
|
|
||||||
)
|
|
||||||
print(_narration_combined_hint(project_path, config), file=sys.stderr)
|
|
||||||
else:
|
|
||||||
print(
|
|
||||||
f"Error: narration_combined not found in videos.json", file=sys.stderr
|
|
||||||
)
|
|
||||||
print(
|
|
||||||
f"You have narration segments in narration.json but haven't stitched them.",
|
|
||||||
file=sys.stderr,
|
|
||||||
)
|
|
||||||
print(
|
|
||||||
f"Run 'gnommo -p {project_path.name} stitch' first.",
|
|
||||||
file=sys.stderr,
|
|
||||||
)
|
|
||||||
return 1
|
|
||||||
else:
|
else:
|
||||||
# Single video - look for .transcript.json next to the narration video
|
if config.transcript_path and (project_path / config.transcript_path).exists():
|
||||||
result = _find_narration_video(config, videos)
|
transcript_path = project_path / config.transcript_path
|
||||||
if result:
|
elif narration_map:
|
||||||
video_id, narration_source = result
|
# Legacy on-disk transcript that used to accompany narration_combined.
|
||||||
config.main_video = video_id # Ensure main_video is set to the found video
|
transcript_path = videos_dir / "narration_combined.transcript.json"
|
||||||
video_path = videos_dir / narration_source.source_file
|
|
||||||
transcript_path = video_path.with_suffix(".transcript.json")
|
|
||||||
else:
|
else:
|
||||||
transcript_path = project_path / "transcript.json"
|
result = _find_narration_video(config, videos)
|
||||||
|
if result:
|
||||||
# If project.json specifies a transcript path, prefer it (always local)
|
_vid, _src = result
|
||||||
if config.transcript_path:
|
config.main_video = _vid
|
||||||
local_transcript = project_path / config.transcript_path
|
transcript_path = (videos_dir / _src.source_file).with_suffix(".transcript.json")
|
||||||
if local_transcript.exists():
|
else:
|
||||||
transcript_path = local_transcript
|
transcript_path = project_path / "transcript.json"
|
||||||
|
transcript_path, _ = resolve_with_cache(transcript_path, project_path)
|
||||||
# Try cache fallback for transcript
|
if not transcript_path.exists():
|
||||||
transcript_path, _ = resolve_with_cache(transcript_path, project_path)
|
print(f"Error: Transcription not found: {transcript_path}", file=sys.stderr)
|
||||||
if not transcript_path.exists():
|
print(f"Run 'gnommo -p {project_path.name} transcribe' first.", file=sys.stderr)
|
||||||
print(f"Error: Transcription not found: {transcript_path}", file=sys.stderr)
|
return 1
|
||||||
print(f"Run 'gnommo -p {project_path.name} transcribe' first.", file=sys.stderr)
|
transcription = load_transcript(transcript_path, project_path)
|
||||||
return 1
|
|
||||||
|
|
||||||
transcription = load_transcript(transcript_path, project_path)
|
|
||||||
|
|
||||||
if verbose:
|
if verbose:
|
||||||
print(f" - Markers in manuscript: {len(markers)}")
|
print(f" - Markers in manuscript: {len(markers)}")
|
||||||
@@ -4455,9 +4421,13 @@ def cmd_render(
|
|||||||
audio,
|
audio,
|
||||||
audio_dir,
|
audio_dir,
|
||||||
slide_range=slide_range,
|
slide_range=slide_range,
|
||||||
|
narration_schedule=narration_schedule,
|
||||||
|
narration_source=narration_source,
|
||||||
)
|
)
|
||||||
if plan.time_offset > 0:
|
if plan.time_offset > 0:
|
||||||
print(f" Time offset: {plan.time_offset:.1f}s (partial render)")
|
print(f" Time offset: {plan.time_offset:.1f}s (partial render)")
|
||||||
|
if plan.narration_segments:
|
||||||
|
print(f" Narration concat: {len(plan.narration_segments)} segment(s) at render time")
|
||||||
|
|
||||||
# Print detailed render plan with alignment info
|
# Print detailed render plan with alignment info
|
||||||
_print_render_plan_details(plan, marker_timings, slides)
|
_print_render_plan_details(plan, marker_timings, slides)
|
||||||
@@ -4584,10 +4554,19 @@ def cmd_render(
|
|||||||
("manuscript.txt", project_path / "manuscript.txt", _state.HASH),
|
("manuscript.txt", project_path / "manuscript.txt", _state.HASH),
|
||||||
("project.json", project_path / "project.json", _state.HASH),
|
("project.json", project_path / "project.json", _state.HASH),
|
||||||
("slides.json", _slides_json, _state.HASH),
|
("slides.json", _slides_json, _state.HASH),
|
||||||
("transcript", transcript_path, _state.HASH),
|
# narration.json (skip/take) drives the concat timeline + alignment.
|
||||||
|
("narration.json", project_path / "media" / "narration" / "narration.json", _state.HASH),
|
||||||
]
|
]
|
||||||
if resolved_combined:
|
if transcript_path:
|
||||||
specs.append(("narration_combined", resolved_combined, _state.META))
|
specs.append(("transcript", transcript_path, _state.HASH))
|
||||||
|
# The concatenated narration segments (the render's main video input).
|
||||||
|
for _seg in plan.narration_segments:
|
||||||
|
specs.append((f"narr:{_seg.seg_id}", _seg.source_path, _state.META))
|
||||||
|
# Per-segment transcripts feed the merged transcript for alignment.
|
||||||
|
_tdir = project_path / "media" / "narration" / "transcripts"
|
||||||
|
if _tdir.is_dir():
|
||||||
|
for _tj in sorted(_tdir.glob("*.json")):
|
||||||
|
specs.append((f"transcript:{_tj.stem}", _tj, _state.HASH))
|
||||||
for _sid, _sdef in slides.items():
|
for _sid, _sdef in slides.items():
|
||||||
specs.append((f"slide:{_sid}", _slides_dir / _sdef.image, _state.META))
|
specs.append((f"slide:{_sid}", _slides_dir / _sdef.image, _state.META))
|
||||||
return specs
|
return specs
|
||||||
@@ -4606,7 +4585,10 @@ def cmd_render(
|
|||||||
if _render_current and output_path.exists():
|
if _render_current and output_path.exists():
|
||||||
try:
|
try:
|
||||||
_out_mtime = output_path.stat().st_mtime
|
_out_mtime = output_path.stat().st_mtime
|
||||||
for _dep in (resolved_combined, transcript_path):
|
_deps = [s.source_path for s in plan.narration_segments]
|
||||||
|
if transcript_path:
|
||||||
|
_deps.append(transcript_path)
|
||||||
|
for _dep in _deps:
|
||||||
if _dep and Path(_dep).exists() and Path(_dep).stat().st_mtime > _out_mtime:
|
if _dep and Path(_dep).exists() and Path(_dep).stat().st_mtime > _out_mtime:
|
||||||
print(f" {Path(_dep).name} is newer than the render — regenerating.")
|
print(f" {Path(_dep).name} is newer than the render — regenerating.")
|
||||||
_render_current = False
|
_render_current = False
|
||||||
|
|||||||
@@ -526,6 +526,11 @@ class RenderPlan:
|
|||||||
narration_pauses: list[NarrationPause] = field(
|
narration_pauses: list[NarrationPause] = field(
|
||||||
default_factory=list
|
default_factory=list
|
||||||
) # Gaps in narration for interstitial videos
|
) # Gaps in narration for interstitial videos
|
||||||
|
# Render-time narration concat: ordered segments (skip/take + offset) to
|
||||||
|
# concatenate directly at render time instead of using a single pre-stitched
|
||||||
|
# narration_combined input. Typed loosely (list of narration.NarrationSegment)
|
||||||
|
# to avoid a circular import between models and narration.
|
||||||
|
narration_segments: list = field(default_factory=list)
|
||||||
# Outro sequence (plays after narration ends)
|
# Outro sequence (plays after narration ends)
|
||||||
outro_events: list["OutroEvent"] = field(
|
outro_events: list["OutroEvent"] = field(
|
||||||
default_factory=list
|
default_factory=list
|
||||||
|
|||||||
+2
-1
@@ -58,7 +58,8 @@ def build_narration_schedule(
|
|||||||
|
|
||||||
Args:
|
Args:
|
||||||
narration: seg_id -> VideoSource (from parse_narration).
|
narration: seg_id -> VideoSource (from parse_narration).
|
||||||
narration_dir: base dir the processed files resolve against.
|
narration_dir: base dir the processed files resolve against
|
||||||
|
(media/narration; output_file is like processed/…mov).
|
||||||
get_duration: callable(Path) -> float (ffprobe duration), used only
|
get_duration: callable(Path) -> float (ffprobe duration), used only
|
||||||
when a segment has no explicit take.
|
when a segment has no explicit take.
|
||||||
transcripts_dir: where per-segment {seg_id}.json transcripts live
|
transcripts_dir: where per-segment {seg_id}.json transcripts live
|
||||||
|
|||||||
+97
-30
@@ -328,6 +328,36 @@ def _build_audio_channel_filter(use_audio_channels: str) -> str:
|
|||||||
return "" # "both" - no filter needed
|
return "" # "both" - no filter needed
|
||||||
|
|
||||||
|
|
||||||
|
def _build_narration_concat_prefilter(seg_input_idxs: list[int], config) -> tuple:
|
||||||
|
"""Normalize + concatenate narration segment inputs into one narration stream.
|
||||||
|
|
||||||
|
Each segment is scaled to the project resolution with alpha-safe transparent
|
||||||
|
padding and setsar=1 (and its audio resampled to a canonical format) so that
|
||||||
|
mismatched source formats can't crash the concat. Returns
|
||||||
|
(video_label, audio_label, filter_lines).
|
||||||
|
"""
|
||||||
|
width, height = config.resolution
|
||||||
|
fps = config.fps
|
||||||
|
lines: list[str] = []
|
||||||
|
pairs: list[tuple[str, str]] = []
|
||||||
|
for j, idx in enumerate(seg_input_idxs):
|
||||||
|
v, a = f"ncv{j}", f"nca{j}"
|
||||||
|
lines.append(
|
||||||
|
f"[{idx}:v]fps={fps},setpts=PTS-STARTPTS,format=yuva444p10le,"
|
||||||
|
f"scale={width}:{height}:force_original_aspect_ratio=decrease,"
|
||||||
|
f"pad={width}:{height}:(ow-iw)/2:(oh-ih)/2:color=0x00000000,setsar=1[{v}]"
|
||||||
|
)
|
||||||
|
lines.append(
|
||||||
|
f"[{idx}:a]aresample=async=1,"
|
||||||
|
f"aformat=sample_rates=48000:channel_layouts=stereo,"
|
||||||
|
f"asetpts=PTS-STARTPTS[{a}]"
|
||||||
|
)
|
||||||
|
pairs.append((v, a))
|
||||||
|
joins = "".join(f"[{v}][{a}]" for v, a in pairs)
|
||||||
|
lines.append(f"{joins}concat=n={len(pairs)}:v=1:a=1[narrsrc_v][narrsrc_a]")
|
||||||
|
return "narrsrc_v", "narrsrc_a", lines
|
||||||
|
|
||||||
|
|
||||||
def build_ffmpeg_command(plan: RenderPlan, output_path: Path) -> list[str]:
|
def build_ffmpeg_command(plan: RenderPlan, output_path: Path) -> list[str]:
|
||||||
"""Build the complete FFmpeg command as a list of arguments."""
|
"""Build the complete FFmpeg command as a list of arguments."""
|
||||||
cmd = ["ffmpeg", "-y"] # -y to overwrite output
|
cmd = ["ffmpeg", "-y"] # -y to overwrite output
|
||||||
@@ -351,25 +381,47 @@ def build_ffmpeg_command(plan: RenderPlan, output_path: Path) -> list[str]:
|
|||||||
# Track input indices
|
# Track input indices
|
||||||
input_idx = 0
|
input_idx = 0
|
||||||
|
|
||||||
# Input: always_visible videos (like talking head)
|
# Input: narration (talking head).
|
||||||
# Add -ss seek BEFORE -i for skip parameter and/or partial rendering
|
# Concat mode: when plan.narration_segments is set, add each processed segment
|
||||||
|
# as its own input (trimmed by skip/take) and concatenate them in-graph into a
|
||||||
|
# single normalized narration stream — replacing the pre-stitched
|
||||||
|
# narration_combined input. Otherwise, the single always-visible input path.
|
||||||
|
# Add -ss seek BEFORE -i for skip parameter and/or partial rendering.
|
||||||
always_visible_inputs: list[int] = []
|
always_visible_inputs: list[int] = []
|
||||||
for video_id, video_source, cutout in plan.narration_videos:
|
narration_concat = None # (video_label, audio_label) when concat mode is active
|
||||||
video_path = _resolve_video_path(
|
if plan.narration_segments:
|
||||||
videos_dir, video_source, shared_assets_dir, project_path
|
seg_input_idxs: list[int] = []
|
||||||
|
for seg in plan.narration_segments:
|
||||||
|
total_seek = (seg.skip or 0.0) + plan.input_seek_time
|
||||||
|
if total_seek > 0:
|
||||||
|
cmd.extend(["-ss", f"{total_seek:.3f}"])
|
||||||
|
if seg.take is not None:
|
||||||
|
cmd.extend(["-t", f"{seg.take:.3f}"])
|
||||||
|
cmd.extend(["-analyzeduration", "0", "-probesize", "1000"])
|
||||||
|
cmd.extend(["-i", str(seg.source_path)])
|
||||||
|
seg_input_idxs.append(input_idx)
|
||||||
|
input_idx += 1
|
||||||
|
_nv_label, _na_label, _narr_concat_lines = _build_narration_concat_prefilter(
|
||||||
|
seg_input_idxs, plan.config
|
||||||
)
|
)
|
||||||
# Combine video skip setting with partial render offset
|
narration_concat = (_nv_label, _na_label, _narr_concat_lines)
|
||||||
total_seek = video_source.skip + plan.input_seek_time
|
else:
|
||||||
if total_seek > 0:
|
for video_id, video_source, cutout in plan.narration_videos:
|
||||||
cmd.extend(["-ss", f"{total_seek:.3f}"])
|
video_path = _resolve_video_path(
|
||||||
# Skip stream analysis — codec params are in the container header, and
|
videos_dir, video_source, shared_assets_dir, project_path
|
||||||
# duration is already known by gnommo via ffprobe (plan.total_duration).
|
)
|
||||||
# Without this, FFmpeg reads 100MB+ of compressed data per input at 4K
|
# Combine video skip setting with partial render offset
|
||||||
# bitrates before encoding starts ("Estimating duration from bitrate").
|
total_seek = video_source.skip + plan.input_seek_time
|
||||||
cmd.extend(["-analyzeduration", "0", "-probesize", "1000"])
|
if total_seek > 0:
|
||||||
cmd.extend(["-i", str(video_path)])
|
cmd.extend(["-ss", f"{total_seek:.3f}"])
|
||||||
always_visible_inputs.append(input_idx)
|
# Skip stream analysis — codec params are in the container header, and
|
||||||
input_idx += 1
|
# duration is already known by gnommo via ffprobe (plan.total_duration).
|
||||||
|
# Without this, FFmpeg reads 100MB+ of compressed data per input at 4K
|
||||||
|
# bitrates before encoding starts ("Estimating duration from bitrate").
|
||||||
|
cmd.extend(["-analyzeduration", "0", "-probesize", "1000"])
|
||||||
|
cmd.extend(["-i", str(video_path)])
|
||||||
|
always_visible_inputs.append(input_idx)
|
||||||
|
input_idx += 1
|
||||||
|
|
||||||
from .cache import resolve_with_cache
|
from .cache import resolve_with_cache
|
||||||
|
|
||||||
@@ -563,6 +615,7 @@ def build_ffmpeg_command(plan: RenderPlan, output_path: Path) -> list[str]:
|
|||||||
video_events_with_audio,
|
video_events_with_audio,
|
||||||
outro_inputs,
|
outro_inputs,
|
||||||
outro_events_with_audio,
|
outro_events_with_audio,
|
||||||
|
narration_concat,
|
||||||
)
|
)
|
||||||
cmd.extend(["-filter_complex", filter_complex])
|
cmd.extend(["-filter_complex", filter_complex])
|
||||||
|
|
||||||
@@ -571,8 +624,9 @@ def build_ffmpeg_command(plan: RenderPlan, output_path: Path) -> list[str]:
|
|||||||
|
|
||||||
# Determine audio source
|
# Determine audio source
|
||||||
# Priority: [aout] from filter > triggered video > no audio
|
# Priority: [aout] from filter > triggered video > no audio
|
||||||
# Note: we always create [aout] when always_visible_inputs exists
|
# Note: [aout] is created whenever there's narration audio — either a single
|
||||||
if always_visible_inputs:
|
# always-visible input or the in-graph concatenated narration [narrsrc_a].
|
||||||
|
if always_visible_inputs or narration_concat:
|
||||||
cmd.extend(
|
cmd.extend(
|
||||||
["-map", "[aout]"]
|
["-map", "[aout]"]
|
||||||
) # Audio from filter (may be segmented or simple copy)
|
) # Audio from filter (may be segmented or simple copy)
|
||||||
@@ -869,6 +923,7 @@ def build_filter_complex(
|
|||||||
video_events_with_audio: set[int] = None,
|
video_events_with_audio: set[int] = None,
|
||||||
outro_inputs: dict[int, int] = None, # outro event_index -> input_idx
|
outro_inputs: dict[int, int] = None, # outro event_index -> input_idx
|
||||||
outro_events_with_audio: set[int] = None,
|
outro_events_with_audio: set[int] = None,
|
||||||
|
narration_concat: tuple = None, # (video_label, audio_label, filter_lines) or None
|
||||||
) -> str:
|
) -> str:
|
||||||
"""
|
"""
|
||||||
Build the filter_complex string for FFmpeg.
|
Build the filter_complex string for FFmpeg.
|
||||||
@@ -888,6 +943,14 @@ def build_filter_complex(
|
|||||||
width, height = plan.config.resolution
|
width, height = plan.config.resolution
|
||||||
filters: list[str] = []
|
filters: list[str] = []
|
||||||
|
|
||||||
|
# Concat mode: emit the narration segment normalize+concat prefilter up front.
|
||||||
|
# It produces [narrsrc_v]/[narrsrc_a], which the talking-head node and the
|
||||||
|
# main-audio path consume in place of a single narration input.
|
||||||
|
narr_v_label = narr_a_label = None
|
||||||
|
if narration_concat:
|
||||||
|
narr_v_label, narr_a_label, _concat_lines = narration_concat
|
||||||
|
filters.extend(_concat_lines)
|
||||||
|
|
||||||
# Create base layer (background)
|
# Create base layer (background)
|
||||||
if has_background:
|
if has_background:
|
||||||
if bg_is_image:
|
if bg_is_image:
|
||||||
@@ -946,7 +1009,9 @@ def build_filter_complex(
|
|||||||
|
|
||||||
# Layer 3: Talking head — above below-videos, but under slides so fullscreen slides cover it
|
# Layer 3: Talking head — above below-videos, but under slides so fullscreen slides cover it
|
||||||
for i, (video_id, video_source, cutout) in enumerate(plan.narration_videos):
|
for i, (video_id, video_source, cutout) in enumerate(plan.narration_videos):
|
||||||
input_idx = always_visible_inputs[i]
|
# Concat mode feeds the talking head from the in-graph concatenated
|
||||||
|
# narration [narrsrc_v]; otherwise from the single narration input pad.
|
||||||
|
narr_src = f"[{narr_v_label}]" if narr_v_label else f"[{always_visible_inputs[i]}:v]"
|
||||||
cut_x, cut_y, cut_width, cut_height = _calculate_cutout_position(
|
cut_x, cut_y, cut_width, cut_height = _calculate_cutout_position(
|
||||||
cutout, width, height
|
cutout, width, height
|
||||||
)
|
)
|
||||||
@@ -955,10 +1020,10 @@ def build_filter_complex(
|
|||||||
zoomed_width = int(cut_width * zoom)
|
zoomed_width = int(cut_width * zoom)
|
||||||
zoomed_height = int(cut_height * zoom)
|
zoomed_height = int(cut_height * zoom)
|
||||||
|
|
||||||
if not plan.narration_pauses:
|
if not plan.narration_pauses or narration_concat:
|
||||||
video_label = f"av{i}"
|
video_label = f"av{i}"
|
||||||
filters.append(
|
filters.append(
|
||||||
f"[{input_idx}:v]fps={plan.config.fps},setpts=PTS-STARTPTS,"
|
f"{narr_src}fps={plan.config.fps},setpts=PTS-STARTPTS,"
|
||||||
f"format=yuva444p10le,"
|
f"format=yuva444p10le,"
|
||||||
f"scale={zoomed_width}:{zoomed_height}:force_original_aspect_ratio=increase,"
|
f"scale={zoomed_width}:{zoomed_height}:force_original_aspect_ratio=increase,"
|
||||||
f"crop={cut_width}:{cut_height}:(iw-{cut_width})/2:(ih-{cut_height})/2,"
|
f"crop={cut_width}:{cut_height}:(iw-{cut_width})/2:(ih-{cut_height})/2,"
|
||||||
@@ -981,7 +1046,7 @@ def build_filter_complex(
|
|||||||
seg_label = f"av{i}_seg{seg_idx}"
|
seg_label = f"av{i}_seg{seg_idx}"
|
||||||
pts_offset = out_start
|
pts_offset = out_start
|
||||||
filters.append(
|
filters.append(
|
||||||
f"[{input_idx}:v]trim={src_start:.3f}:{src_end:.3f},"
|
f"{narr_src}trim={src_start:.3f}:{src_end:.3f},"
|
||||||
f"setpts=PTS-STARTPTS+{pts_offset:.3f}/TB,"
|
f"setpts=PTS-STARTPTS+{pts_offset:.3f}/TB,"
|
||||||
f"format=yuva444p10le,"
|
f"format=yuva444p10le,"
|
||||||
f"scale={zoomed_width}:{zoomed_height}:force_original_aspect_ratio=increase,"
|
f"scale={zoomed_width}:{zoomed_height}:force_original_aspect_ratio=increase,"
|
||||||
@@ -1174,8 +1239,10 @@ def build_filter_complex(
|
|||||||
filters.append(f"[{current_label}]copy[vout]")
|
filters.append(f"[{current_label}]copy[vout]")
|
||||||
|
|
||||||
# Audio mixing: combine main audio with sound effects
|
# Audio mixing: combine main audio with sound effects
|
||||||
if always_visible_inputs:
|
if always_visible_inputs or narration_concat:
|
||||||
main_audio_idx = always_visible_inputs[0]
|
# In concat mode the main narration audio is the in-graph [narrsrc_a];
|
||||||
|
# otherwise it's the first always-visible input's audio pad.
|
||||||
|
_main_aud = f"[{narr_a_label}]" if narr_a_label else f"[{always_visible_inputs[0]}:a]"
|
||||||
audio_labels_to_mix = []
|
audio_labels_to_mix = []
|
||||||
|
|
||||||
# Get audio channel setting and volume from first narration video
|
# Get audio channel setting and volume from first narration video
|
||||||
@@ -1202,7 +1269,7 @@ def build_filter_complex(
|
|||||||
plan.narration_end_time if plan.outro_events else plan.total_duration
|
plan.narration_end_time if plan.outro_events else plan.total_duration
|
||||||
)
|
)
|
||||||
|
|
||||||
if not plan.narration_pauses:
|
if not plan.narration_pauses or narration_concat:
|
||||||
# Simple case: trim main audio to end before outro (with optional channel and volume filters)
|
# Simple case: trim main audio to end before outro (with optional channel and volume filters)
|
||||||
filter_parts = []
|
filter_parts = []
|
||||||
if channel_filter:
|
if channel_filter:
|
||||||
@@ -1215,16 +1282,16 @@ def build_filter_complex(
|
|||||||
filter_parts.append(f"atrim=0:{audio_end_time:.3f}")
|
filter_parts.append(f"atrim=0:{audio_end_time:.3f}")
|
||||||
filter_parts.append("asetpts=PTS-STARTPTS")
|
filter_parts.append("asetpts=PTS-STARTPTS")
|
||||||
filters.append(
|
filters.append(
|
||||||
f"[{main_audio_idx}:a]{','.join(filter_parts)}[main_aud]"
|
f"{_main_aud}{','.join(filter_parts)}[main_aud]"
|
||||||
)
|
)
|
||||||
audio_labels_to_mix.append("[main_aud]")
|
audio_labels_to_mix.append("[main_aud]")
|
||||||
elif filter_parts:
|
elif filter_parts:
|
||||||
filters.append(
|
filters.append(
|
||||||
f"[{main_audio_idx}:a]{','.join(filter_parts)}[main_aud]"
|
f"{_main_aud}{','.join(filter_parts)}[main_aud]"
|
||||||
)
|
)
|
||||||
audio_labels_to_mix.append("[main_aud]")
|
audio_labels_to_mix.append("[main_aud]")
|
||||||
else:
|
else:
|
||||||
audio_labels_to_mix.append(f"[{main_audio_idx}:a]")
|
audio_labels_to_mix.append(f"{_main_aud}")
|
||||||
else:
|
else:
|
||||||
# Complex case: segment the narration audio for pauses
|
# Complex case: segment the narration audio for pauses
|
||||||
segments = _build_narration_segments(plan.narration_pauses, audio_end_time)
|
segments = _build_narration_segments(plan.narration_pauses, audio_end_time)
|
||||||
@@ -1244,7 +1311,7 @@ def build_filter_complex(
|
|||||||
if volume_filter:
|
if volume_filter:
|
||||||
filter_parts.append(volume_filter)
|
filter_parts.append(volume_filter)
|
||||||
filters.append(
|
filters.append(
|
||||||
f"[{main_audio_idx}:a]{','.join(filter_parts)}[{seg_label}]"
|
f"{_main_aud}{','.join(filter_parts)}[{seg_label}]"
|
||||||
)
|
)
|
||||||
audio_labels_to_mix.append(f"[{seg_label}]")
|
audio_labels_to_mix.append(f"[{seg_label}]")
|
||||||
|
|
||||||
|
|||||||
+47
-35
@@ -637,6 +637,8 @@ def build_render_plan(
|
|||||||
audio: Optional[dict[str, AudioDefinition]] = None,
|
audio: Optional[dict[str, AudioDefinition]] = None,
|
||||||
audio_dir: Optional[Path] = None,
|
audio_dir: Optional[Path] = None,
|
||||||
slide_range: Optional[tuple[str, Optional[str]]] = None,
|
slide_range: Optional[tuple[str, Optional[str]]] = None,
|
||||||
|
narration_schedule: Optional[list] = None,
|
||||||
|
narration_source: Optional[VideoSource] = None,
|
||||||
) -> tuple[RenderPlan, list[MarkerTiming]]:
|
) -> tuple[RenderPlan, list[MarkerTiming]]:
|
||||||
"""
|
"""
|
||||||
Build a complete render plan from manuscript and transcription.
|
Build a complete render plan from manuscript and transcription.
|
||||||
@@ -655,25 +657,52 @@ def build_render_plan(
|
|||||||
audio = audio or {}
|
audio = audio or {}
|
||||||
audio_dir = audio_dir or project_path
|
audio_dir = audio_dir or project_path
|
||||||
|
|
||||||
# Find the main narration video first (need skip value for timing adjustment)
|
|
||||||
narration_video_id = config.main_video
|
|
||||||
if isinstance(narration_video_id, list):
|
|
||||||
narration_video_id = narration_video_id[0] if narration_video_id else None
|
|
||||||
if not (narration_video_id and narration_video_id in videos):
|
|
||||||
raise ValueError(
|
|
||||||
f"Main video '{narration_video_id}' not specified or not found in videos. "
|
|
||||||
f"Available: {list(videos.keys())}"
|
|
||||||
)
|
|
||||||
narration_video = videos[narration_video_id]
|
|
||||||
|
|
||||||
# Align markers to transcription timestamps
|
# Align markers to transcription timestamps
|
||||||
marker_timings = align_markers_to_transcription(
|
marker_timings = align_markers_to_transcription(
|
||||||
manuscript_text, transcription, slides=slides, videos=videos, audio=audio
|
manuscript_text, transcription, slides=slides, videos=videos, audio=audio
|
||||||
)
|
)
|
||||||
|
|
||||||
# Apply skip offset: if narration video has skip, subtract it from all timestamps
|
# Find shared_assets directory
|
||||||
# This accounts for the fact that the video will start at skip seconds, not 0
|
shared_assets_dir = None
|
||||||
narration_skip = narration_video.skip
|
if (project_path / "shared_assets").exists():
|
||||||
|
shared_assets_dir = project_path / "shared_assets"
|
||||||
|
elif (project_path.parent / "shared_assets").exists():
|
||||||
|
shared_assets_dir = project_path.parent / "shared_assets"
|
||||||
|
|
||||||
|
# Track which files are loaded from external cache
|
||||||
|
cached_files: set[str] = set()
|
||||||
|
|
||||||
|
# --- Narration source ---
|
||||||
|
# Render-time concat: narration is the concatenation of the scheduled
|
||||||
|
# segments, so there is no single file to probe — the total duration is the
|
||||||
|
# sum of the segment durations and skip is already baked into each segment.
|
||||||
|
if narration_schedule:
|
||||||
|
narration_video_id = "narration"
|
||||||
|
narration_video = narration_source or VideoSource(
|
||||||
|
source_file="", cutout=config.default_slide_type, always_visible=True
|
||||||
|
)
|
||||||
|
narration_skip = 0.0
|
||||||
|
full_duration = sum(seg.duration for seg in narration_schedule)
|
||||||
|
else:
|
||||||
|
narration_video_id = config.main_video
|
||||||
|
if isinstance(narration_video_id, list):
|
||||||
|
narration_video_id = narration_video_id[0] if narration_video_id else None
|
||||||
|
if not (narration_video_id and narration_video_id in videos):
|
||||||
|
raise ValueError(
|
||||||
|
f"Main video '{narration_video_id}' not specified or not found in videos. "
|
||||||
|
f"Available: {list(videos.keys())}"
|
||||||
|
)
|
||||||
|
narration_video = videos[narration_video_id]
|
||||||
|
narration_skip = narration_video.skip
|
||||||
|
video_path, is_cached = _resolve_video_path(
|
||||||
|
videos_dir, narration_video, shared_assets_dir, project_path
|
||||||
|
)
|
||||||
|
if is_cached:
|
||||||
|
cached_files.add(narration_video_id)
|
||||||
|
full_duration = get_video_duration(video_path)
|
||||||
|
|
||||||
|
# Apply skip offset: if narration starts at `skip` seconds, subtract it from
|
||||||
|
# all marker timestamps so they line up with the trimmed timeline.
|
||||||
if narration_skip > 0:
|
if narration_skip > 0:
|
||||||
for timing in marker_timings:
|
for timing in marker_timings:
|
||||||
if timing.timestamp >= 0:
|
if timing.timestamp >= 0:
|
||||||
@@ -685,30 +714,12 @@ def build_render_plan(
|
|||||||
if timing.timestamp >= 0:
|
if timing.timestamp >= 0:
|
||||||
marker_times[timing.marker_id] = timing.timestamp
|
marker_times[timing.marker_id] = timing.timestamp
|
||||||
|
|
||||||
# Find shared_assets directory
|
|
||||||
shared_assets_dir = None
|
|
||||||
if (project_path / "shared_assets").exists():
|
|
||||||
shared_assets_dir = project_path / "shared_assets"
|
|
||||||
elif (project_path.parent / "shared_assets").exists():
|
|
||||||
shared_assets_dir = project_path.parent / "shared_assets"
|
|
||||||
|
|
||||||
narration_video = videos[narration_video_id]
|
|
||||||
cutout = config.cutouts[narration_video.cutout]
|
cutout = config.cutouts[narration_video.cutout]
|
||||||
|
|
||||||
# Track which files are loaded from external cache
|
|
||||||
cached_files: set[str] = set()
|
|
||||||
|
|
||||||
narration_videos: list[tuple[str, VideoSource, CutoutDefinition]] = []
|
|
||||||
video_path, is_cached = _resolve_video_path(
|
|
||||||
videos_dir, narration_video, shared_assets_dir, project_path
|
|
||||||
)
|
|
||||||
if is_cached:
|
|
||||||
cached_files.add(narration_video_id)
|
|
||||||
full_duration = get_video_duration(video_path)
|
|
||||||
# Adjust duration for skip (content starts at skip, so effective duration is less)
|
# Adjust duration for skip (content starts at skip, so effective duration is less)
|
||||||
effective_duration = full_duration - narration_skip
|
effective_duration = full_duration - narration_skip
|
||||||
# Get total duration from first always_visible video
|
narration_videos: list[tuple[str, VideoSource, CutoutDefinition]] = [
|
||||||
narration_videos.append((narration_video_id, narration_video, cutout))
|
(narration_video_id, narration_video, cutout)
|
||||||
|
]
|
||||||
# Resolve slide range to time range
|
# Resolve slide range to time range
|
||||||
time_offset = 0.0
|
time_offset = 0.0
|
||||||
render_end_time = effective_duration
|
render_end_time = effective_duration
|
||||||
@@ -925,6 +936,7 @@ def build_render_plan(
|
|||||||
input_seek_time=time_offset,
|
input_seek_time=time_offset,
|
||||||
shared_assets_dir=shared_assets_dir,
|
shared_assets_dir=shared_assets_dir,
|
||||||
narration_pauses=narration_pauses,
|
narration_pauses=narration_pauses,
|
||||||
|
narration_segments=narration_schedule or [],
|
||||||
outro_events=outro_events,
|
outro_events=outro_events,
|
||||||
narration_end_time=narration_end_time,
|
narration_end_time=narration_end_time,
|
||||||
cached_files=cached_files,
|
cached_files=cached_files,
|
||||||
|
|||||||
Reference in New Issue
Block a user