From f9ff847f6b10ec8f90f4b1197decbab4581a06a0 Mon Sep 17 00:00:00 2001 From: jenstandstad Date: Sun, 5 Jul 2026 12:16:35 +0200 Subject: [PATCH] Adding fixes to the stitcher --- all.sh | 2 +- example/media/videos/videos.json | 3 +- gnommo/cache.py | 144 +++++-- gnommo/cli.py | 713 +++++++++++++++++++++++++++---- gnommo/parser.py | 10 +- gnommo/pexels.py | 181 +++++++- gnommo/preprocessor.py | 7 +- gnommo/renderer.py | 67 ++- gnommo/transformer.py | 16 +- gnommo/validator.py | 34 +- renderforce.sh | 9 + 11 files changed, 1019 insertions(+), 167 deletions(-) create mode 100755 renderforce.sh diff --git a/all.sh b/all.sh index 38e81d9..cc88c70 100755 --- a/all.sh +++ b/all.sh @@ -4,6 +4,6 @@ ./gnommo.sh -p video2 all --force --prod ./gnommo.sh -p video3 all --force --prod ./gnommo.sh -p video4 all --force --prod -#./gnommo.sh -p video5 all --force +./gnommo.sh -p video5 all --force --prod #./gnommo.sh -p video6 all --force diff --git a/example/media/videos/videos.json b/example/media/videos/videos.json index 4e99af5..44f3270 100644 --- a/example/media/videos/videos.json +++ b/example/media/videos/videos.json @@ -29,11 +29,10 @@ "take": 25, "skip": 0 }, - "Zoomin_MontageZoom": { + "zoomin_montagezoom": { "description": "Montage zoom", "source_file": "MontageZoom.mp4", "output_file": "MontageZoom.mp4", - "pause_narration": 5, "cutout": "square", "is_shared": true, "filter": [] diff --git a/gnommo/cache.py b/gnommo/cache.py index 84e4d9b..849828f 100644 --- a/gnommo/cache.py +++ b/gnommo/cache.py @@ -16,6 +16,7 @@ from pathlib import Path from typing import Optional, Tuple _cache_config: Optional[dict] = None +_assets_config: Optional[dict] = None _perf_config: Optional[dict] = None @@ -102,51 +103,107 @@ def load_cache_config() -> Optional[Path]: return None +def load_assets_process_cache() -> Optional[Path]: + """Return the process-cache path on the [assets] disk, or None if not configured. + + Derived by replacing the last component of the [assets] path with + that name + "cache". E.g.: + [assets] path = /Volumes/LaCie Jens/Projects/gnommo + → process cache = /Volumes/LaCie Jens/Projects/gnommocache + + This mirrors the GnommoDisk convention where the asset root is + /Volumes/GnommoDisk/gnommo and the process cache is /Volumes/GnommoDisk/gnommocache. + """ + assets_path = load_assets_config() + if assets_path is None: + return None + return assets_path.parent / (assets_path.name + "cache") + + +def load_assets_config() -> Optional[Path]: + """Load gnommo.conf and return the [assets] path if configured. + + The assets path is a second external fallback (e.g. a LaCie drive) with + the same directory layout as the gnommo project root. Resolution order is: + local → cache ([cache] path) → assets ([assets] path). + + Example ~/.gnommo.conf: + [assets] + path = /Volumes/LaCie Jens/Projects/gnommo + """ + global _assets_config + if _assets_config is not None: + return _assets_config.get("path") + + config_path = Path.home() / ".gnommo.conf" + if not config_path.exists(): + _assets_config = {} + return None + + config = configparser.ConfigParser() + config.read(config_path) + + if config.has_option("assets", "path"): + assets_path = Path(config.get("assets", "path")) + _assets_config = {"path": assets_path} + return assets_path + + _assets_config = {} + return None + + +def _resolve_against_base( + local_path: Path, project_path: Path, base: Path +) -> Optional[Path]: + """Try to find local_path mirrored under base. + + Tries two mappings: + 1. project-relative: base / project_name / relative_to_project + 2. gnommo-root-relative: base / relative_to_project_parent (e.g. shared_assets/…) + """ + try: + relative = local_path.relative_to(project_path) + p = base / project_path.name / relative + if p.exists(): + return p + except ValueError: + pass + + try: + relative = local_path.relative_to(project_path.parent) + p = base / relative + if p.exists(): + return p + except ValueError: + pass + + return None + + def resolve_with_cache( local_path: Path, project_path: Path, ) -> Tuple[Path, bool]: - """ - Resolve a file path with cache fallback (read-only). + """Resolve a file path with external-disk fallback (read-only). - Checks the local path first. If not found and cache is configured, - checks the cache directory which mirrors the project structure. - - Args: - local_path: The expected local path to the file - project_path: The project root directory + Resolution order: + 1. local_path (always checked first) + 2. [cache] path — typically GnommoDisk + 3. [assets] path — optional second drive (e.g. LaCie) Returns: - Tuple of (resolved_path, is_cached) where is_cached=True if - the file was found in the external cache instead of locally. + Tuple of (resolved_path, is_from_external) where is_from_external=True + when the file was found on an external drive rather than locally. """ - # Check local path first if local_path.exists(): return local_path, False - # Check cache - cache_base = load_cache_config() - if cache_base is None: - return local_path, False # No cache configured - - # Try 1: path inside the project → cache_base / project_name / relative - try: - relative = local_path.relative_to(project_path) - cache_path = cache_base / project_path.name / relative - if cache_path.exists(): - return cache_path, True - except ValueError: - pass # local_path is not under project_path - - # Try 2: path relative to gnommo root (sibling dirs like shared_assets) - # e.g. shared_assets/pexels/file.mp4 → cache_base / shared_assets / pexels / file.mp4 - try: - relative = local_path.relative_to(project_path.parent) - cache_path = cache_base / relative - if cache_path.exists(): - return cache_path, True - except ValueError: - pass # local_path is not under project_path.parent either + for base in (load_cache_config(), load_assets_config()): + if base is None: + continue + resolved = _resolve_against_base(local_path, project_path, base) + if resolved is not None: + return resolved, True return local_path, False @@ -185,15 +242,16 @@ def load_server_config() -> Optional[dict]: def is_cache_configured() -> bool: - """Check if cache is configured (for status messages).""" - return load_cache_config() is not None + """Check if any external fallback is configured.""" + return load_cache_config() is not None or load_assets_config() is not None def get_cache_info() -> Optional[str]: - """Get a human-readable cache configuration string.""" - cache_path = load_cache_config() - if cache_path is None: - return None - if cache_path.exists(): - return f"{cache_path} (connected)" - return f"{cache_path} (not connected)" + """Get a human-readable string of all configured external paths.""" + parts = [] + for label, path in (("cache", load_cache_config()), ("assets", load_assets_config())): + if path is None: + continue + status = "connected" if path.exists() else "not connected" + parts.append(f"{path} ({status})") + return "; ".join(parts) if parts else None diff --git a/gnommo/cli.py b/gnommo/cli.py index 5a5f31b..fb08a42 100644 --- a/gnommo/cli.py +++ b/gnommo/cli.py @@ -76,8 +76,8 @@ Examples: "-p", "--project", type=str, - required=True, - help="Project directory", + default=None, + help="Project directory (required for all actions except 'pexels --search')", ) parser.add_argument( "action", @@ -193,6 +193,12 @@ Examples: default=-40.0, help="For trim: silence threshold in dB (default: -40). Raise (e.g. -25) to ignore clothing/room noise.", ) + parser.add_argument( + "--model", + type=str, + default="base", + help="For trim: Whisper model for transcription (tiny/base/small/medium/large, default: base)", + ) parser.add_argument( "--crf", type=int, @@ -216,17 +222,38 @@ Examples: dest="alpha_quality", help="For transcode --processed: HEVC alpha quality 0.0-1.0 (default: 0.75; lower=smaller file)", ) + parser.add_argument( + "--search", + type=str, + default=None, + metavar="QUERY", + help="For pexels: search Pexels for QUERY and download all results to the assets disk", + ) + parser.add_argument( + "--max", + type=int, + default=200, + dest="search_max", + help="For pexels --search: maximum number of videos to download (default: 200)", + ) args = parser.parse_args() - # Resolve project path - project_path = Path(args.project) - if not project_path.is_absolute(): - project_path = Path.cwd() / project_path + # Resolve project path (optional for 'pexels --search') + action = args.action + if args.project is None: + if action == "pexels" and args.search: + project_path = Path.cwd() # placeholder; cmd_pexels won't use it in search mode + else: + parser.error("argument -p/--project is required") + return 1 + else: + project_path = Path(args.project) + if not project_path.is_absolute(): + project_path = Path.cwd() / project_path try: # Handle actions - action = args.action if action == "import": return cmd_import(project_path, args.force, args.verbose) @@ -243,7 +270,7 @@ Examples: ) elif action == "trim": return cmd_trim( - project_path, args.verbose, args.force, args.threshold, args.res + project_path, args.verbose, args.force, args.threshold, args.res, args.model ) elif action == "transcode": return cmd_transcode( @@ -312,7 +339,7 @@ Examples: project_path, args.verbose, args.file, args.prod, args.res ) elif action == "pexels": - return cmd_pexels(project_path, args.verbose) + return cmd_pexels(project_path, args.verbose, args.search, args.search_max) except GnommoError as e: print(f"Error: {e}", file=sys.stderr) @@ -329,6 +356,213 @@ Examples: # ============================================================================= +def _prompter_break_line(text: str, max_words: int = 10) -> list[str]: + """Break a long line into short breath-sized chunks for teleprompter display. + + Priority order for break points: + 1. Sentence end (. ? !) — always break here + 2. Em/en-dash ( — / - ) — break if ≥ 4 words accumulated + 3. Semicolon (;) — break if ≥ 4 words accumulated + 4. Comma (,) — break if ≥ 5 words accumulated + 5. Word ceiling (max_words) — force break at word boundary + + Each output line is stripped and non-empty. + """ + import re + + words = text.split() + if len(words) <= max_words: + return [text.strip()] if text.strip() else [] + + lines: list[str] = [] + current: list[str] = [] + + # Dash detection: standalone dash token, or word that is entirely dashes/hyphens + _DASH_RE = re.compile(r"^[-–—]+$") + _SENTENCE_END_RE = re.compile(r'[.?!]["\']?$') + _SOFT_END_COMMA = re.compile(r",$") + _SOFT_END_SEMI = re.compile(r";$") + + def flush(force: bool = False) -> None: + """Flush current words to a line. + + If the accumulated chunk is very short (≤ 2 words) and this isn't a + sentence-end forced flush, hold it — it will merge into the next line. + This prevents orphaned 1-2-word lines from soft breaks. + """ + if not force and len(current) <= 2: + return # too short to stand alone — let more words join + chunk = " ".join(current).strip() + if chunk: + lines.append(chunk) + current.clear() + + i = 0 + while i < len(words): + word = words[i] + n = len(current) + + # Peek ahead: is the next token a standalone dash? + next_is_dash = (i + 1 < len(words)) and _DASH_RE.match(words[i + 1]) + + if _DASH_RE.match(word): + # Standalone dash: attach to current line then break. + current.append(word) + if n > 0: + flush() + elif _SENTENCE_END_RE.search(word): + current.append(word) + flush(force=True) # sentence end always flushes regardless of length + elif _SOFT_END_SEMI.search(word) and n >= 4: + current.append(word) + flush() + elif _SOFT_END_COMMA.search(word) and n >= 5: + current.append(word) + flush() + elif next_is_dash and n >= 5: + flush() + current.append(word) + elif n >= max_words: + flush(force=True) # word ceiling: must break even if line is short + current.append(word) + else: + current.append(word) + + i += 1 + + flush() + return lines + + +def _export_prompter_manuscript( + manuscript_path: Path, + verbose: bool, + prompter_wpm: int = 130, + max_words: int = 10, +) -> None: + """Generate manuscript_prompter.txt from manuscript.txt. + + Strips all bracketed markers except [S*] and [cue:*], auto-breaks long + lines into breath-sized chunks, and converts [pause:Xs] markers into blank + lines calibrated to the teleprompter scroll speed. + + Calibration: the scroll speed is assumed to match reading speed (prompter_wpm). + At that speed each line takes (60 * max_words / prompter_wpm) seconds, so a + [pause:Xs] marker becomes round(X * prompter_wpm / (60 * max_words)) blank lines, + preceded by a visible label line ("· · · 8s · · ·"). + """ + import re + + output_path = manuscript_path.parent / "manuscript_prompter.txt" + text = manuscript_path.read_text(encoding="utf-8") + + _MARKER_RE = re.compile(r"\[[^\]]+\]") + _KEEP_RE = re.compile(r"^\[S[^\]]*\]$|^\[cue:[^\]]*\]$") + _SEGMENT_RE = re.compile(r"^\[segment:[^\]]*\]$") + _PAUSE_RE = re.compile(r"^\[pause:(\d+(?:\.\d+)?)s?\]$") + _BLOCK_BREAK = "\x00BLOCK\x00" + + def _pause_lines(seconds: float) -> list[str]: + """Convert a pause duration into a label + blank lines for the teleprompter.""" + n_blanks = max(1, round(seconds * prompter_wpm / (60 * max_words))) + label = f"· · · {seconds:g}s · · ·" + return [label] + [""] * n_blanks + + # Step 1: strip non-display markers, collect cleaned lines + cleaned_lines: list[str] = [] + for line in text.splitlines(): + stripped = line.strip() + if _MARKER_RE.fullmatch(stripped): + if _SEGMENT_RE.match(stripped): + cleaned_lines.append(_BLOCK_BREAK) + elif _KEEP_RE.match(stripped): + cleaned_lines.append(stripped) + else: + m = _PAUSE_RE.match(stripped) + if m: + cleaned_lines.extend(_pause_lines(float(m.group(1)))) + # else: drop marker-only lines entirely + else: + def _replace(m: re.Match) -> str: + # Handle inline [pause:Xs] — convert to label inline + if _PAUSE_RE.match(m.group(0)): + sec = float(_PAUSE_RE.match(m.group(0)).group(1)) + return f"· · · {sec:g}s · · ·" + return m.group(0) if _KEEP_RE.match(m.group(0)) else "" + cleaned = _MARKER_RE.sub(_replace, line).strip() + if cleaned: + cleaned_lines.append(cleaned) + else: + cleaned_lines.append("") # preserve intentional blank lines + + # Step 2: auto-break long text lines; pass markers, blanks, and pause labels through + _PAUSE_LABEL_RE = re.compile(r"^· · · .+ · · ·$") + broken: list[str] = [] + for line in cleaned_lines: + stripped = line.strip() + if not stripped or stripped == _BLOCK_BREAK or _KEEP_RE.match(stripped) or _PAUSE_LABEL_RE.match(stripped): + broken.append(line) + else: + broken.extend(_prompter_break_line(stripped)) + + # Step 3: merge each slide marker onto the first word of its slide, + # then lay out the final lines. + # - [S*] markers are prepended inline to the next text line + # - Block-break sentinels expand to a double blank (Elgato block boundary) + # - Ordinary consecutive blank lines are collapsed to one + merged: list[str] = [] + pending_marker: str = "" + pending_breaks: int = 0 # block-breaks that arrived while a marker was pending + for line in broken: + stripped = line.strip() + if stripped and _KEEP_RE.match(stripped): + pending_marker = stripped # hold until the next text line + elif pending_marker: + if stripped and stripped != _BLOCK_BREAK: + # Flush any buffered block-breaks first, then attach marker to text + for _ in range(pending_breaks): + merged.append(_BLOCK_BREAK) + pending_breaks = 0 + merged.append(f"{pending_marker} {stripped}") + pending_marker = "" + elif stripped == _BLOCK_BREAK: + # Buffer the block-break — emit it after the marker attaches + pending_breaks += 1 + else: + # Ordinary blank: carry forward (swallow into the marker's gap) + pass + else: + merged.append(line) + if pending_marker: + for _ in range(pending_breaks): + merged.append(_BLOCK_BREAK) + merged.append(pending_marker) + + result: list[str] = [] + prev_blank_count = 0 + + for line in merged: + if line == _BLOCK_BREAK: + needed = 2 - prev_blank_count + result.extend([""] * max(needed, 0)) + prev_blank_count = 2 + continue + + is_blank = not line.strip() + + if is_blank: + if prev_blank_count >= 1: + continue # collapse ordinary consecutive blanks + result.append("") + prev_blank_count += 1 + else: + result.append(line) + prev_blank_count = 0 + + output_path.write_text("\n".join(result).strip() + "\n", encoding="utf-8") + print(f" Prompter manuscript → {output_path.name}") + + def cmd_import(project_path: Path, force: bool, verbose: bool) -> int: """Import assets and generate metadata JSON files.""" from .parser import parse_project_config, _read_json @@ -421,15 +655,29 @@ def cmd_import(project_path: Path, force: bool, verbose: bool) -> int: for marker in markers: for prefix in _SHORTHAND_PREFIXES: if marker.startswith(prefix): - vid_id = marker[len(prefix):] + vid_id = marker[len(prefix):].lower() if vid_id not in local_vids and vid_id not in seen_missing: + hint = ( + "run 'gnommo pexels' to download" + if vid_id.startswith("pexels/") + else "add it manually" + ) print( f" ⚠ [{marker}] video '{vid_id}' not found in " - f"videos.json or shared_assets — add it manually" + f"videos.json or shared_assets — {hint}" ) seen_missing.add(vid_id) break + # Generate teleprompter version of manuscript + if manuscript_path.exists(): + import configparser as _cp + _pcfg = _cp.ConfigParser() + _pcfg.read(Path.home() / ".gnommo.conf") + _wpm = int(_pcfg.get("prompter", "wpm", fallback="130")) + _mpw = int(_pcfg.get("prompter", "max_words_per_line", fallback="10")) + _export_prompter_manuscript(manuscript_path, verbose, prompter_wpm=_wpm, max_words=_mpw) + print("Import complete.") return 0 @@ -473,7 +721,7 @@ def _import_shared_audio( added = 0 for f in audio_files: - audio_id = f.stem + audio_id = f.stem.lower() if audio_id in existing: if verbose: print(f" Skipping {audio_id} (already in audio.json)") @@ -619,21 +867,27 @@ def _probe_video_metadata( ) # Mirror renderer._resolve_video_path: try output_file first, then source_file + # Use resolve_with_cache so files on external disks (LaCie, GnommoDisk) are found. + from .cache import resolve_with_cache + video_path = None output_file = video_data.get("output_file") if output_file: for candidate_dir in [base_dir, base_dir.parent]: candidate = candidate_dir / output_file + candidate, _ = resolve_with_cache(candidate, project_path) if candidate.exists(): video_path = candidate break mov_candidate = candidate.with_suffix(".mov") + mov_candidate, _ = resolve_with_cache(mov_candidate, project_path) if mov_candidate.exists(): video_path = mov_candidate break if video_path is None: source_candidate = base_dir / video_data["source_file"] + source_candidate, _ = resolve_with_cache(source_candidate, project_path) if source_candidate.exists(): video_path = source_candidate @@ -717,8 +971,13 @@ def _sync_shared_videos_to_local( elif verbose: print(f" Shared '{video_id}': already in local videos.json, skipping") continue - # New entry — copy from shared and mark it as shared - local_entry = dict(shared_entry) + # New entry — copy base metadata from shared and mark it as shared. + # Exclude presentation fields (cutout, layer) so _project_markers_to_videos + # can set them from the manuscript without stale shared values interfering. + _PRESENTATION_FIELDS = {"cutout", "layer", "pause_narration"} + local_entry = { + k: v for k, v in shared_entry.items() if k not in _PRESENTATION_FIELDS + } local_entry["is_shared"] = True local_videos[video_id] = local_entry added.append(video_id) @@ -769,15 +1028,15 @@ def _import_shared_assets(shared_assets_dir: Path, verbose: bool) -> None: video_extensions = {".mov", ".mp4", ".webm", ".avi", ".mkv", ".m4v"} # Find all video files in shared_assets (root level and subdirectories). - # Also scan the GnommoDisk cache mirror so files placed there are registered. - from .cache import load_cache_config + # Also scan external disk mirrors so files placed there are registered. + from .cache import load_assets_config, load_cache_config scan_roots: list[Path] = [shared_assets_dir] - cache_base = load_cache_config() - if cache_base: - cache_shared = cache_base / "shared_assets" - if cache_shared.exists() and cache_shared != shared_assets_dir: - scan_roots.append(cache_shared) + for external_base in (load_cache_config(), load_assets_config()): + if external_base: + ext_shared = external_base / "shared_assets" + if ext_shared.exists() and ext_shared != shared_assets_dir: + scan_roots.append(ext_shared) video_files: list[tuple[Path, Path]] = [] # (relative_path, absolute_path) seen_rel: set[str] = set() # deduplicate by relative path @@ -821,12 +1080,28 @@ def _import_shared_assets(shared_assets_dir: Path, verbose: bool) -> None: if videos_json_path.exists(): existing_videos = _read_json(videos_json_path) + # Remove entries whose source file no longer exists on any scan root + removed_count = 0 + stale_keys = [ + vid_id + for vid_id, vid_data in existing_videos.items() + if vid_data.get("source_file") + and not any((root / vid_data["source_file"]).exists() for root in scan_roots) + ] + for vid_id in stale_keys: + del existing_videos[vid_id] + removed_count += 1 + if verbose: + print(f" Removed stale shared entry: {vid_id}") + if removed_count > 0: + print(f" Removed {removed_count} stale shared asset(s) from videos.json") + # Add new videos (don't overwrite existing) added_count = 0 for rel_path, abs_path in sorted(video_files): - # Use path relative to shared_assets without extension as video_id - # e.g., "Logo" for root files, "pexels/6759604-hd" for subdirectory files - video_id = str(rel_path.with_suffix("")) + # Use path relative to shared_assets without extension as video_id (lowercase) + # e.g., "logo" for root files, "pexels/6759604-hd" for subdirectory files + video_id = str(rel_path.with_suffix("")).lower() if video_id in existing_videos: if verbose: @@ -840,11 +1115,12 @@ def _import_shared_assets(shared_assets_dir: Path, verbose: bool) -> None: if verbose: print(f" Added: {video_id}") - if added_count > 0: + if added_count > 0 or removed_count > 0: # Write updated videos.json with open(videos_json_path, "w", encoding="utf-8") as f: json.dump(existing_videos, f, indent=2) - print(f" Updated {videos_json_path} (+{added_count} shared assets)") + if added_count > 0: + print(f" Updated {videos_json_path} (+{added_count} shared assets)") else: print(f" No new shared assets to add") @@ -940,14 +1216,32 @@ def _import_videos(videos_dir: Path, config, verbose: bool) -> None: if videos_json_path.exists(): existing_videos = _read_json(videos_json_path) + # Remove entries whose source file no longer exists on disk. + # Skip is_shared entries — their source files live in shared_assets, not videos_dir. + removed_count = 0 + stale_keys = [ + vid_id + for vid_id, vid_data in existing_videos.items() + if not vid_data.get("is_shared") + and vid_data.get("source_file") + and not (videos_dir / vid_data["source_file"]).exists() + ] + for vid_id in stale_keys: + del existing_videos[vid_id] + removed_count += 1 + if verbose: + print(f" Removed stale entry: {vid_id}") + if removed_count > 0: + print(f" Removed {removed_count} stale video(s) from {videos_json_path.name}") + # Get available filter presets from config default_filters = config.default_filters if config else {} # Add new videos (don't overwrite existing) added_count = 0 for video_file in sorted(video_files): - # Use filename without extension as video_id - video_id = video_file.stem + # Use filename without extension as video_id (lowercase for new entries) + video_id = video_file.stem.lower() if video_id in existing_videos: if verbose: @@ -981,11 +1275,12 @@ def _import_videos(videos_dir: Path, config, verbose: bool) -> None: existing_videos[video_id] = video_entry added_count += 1 - if added_count > 0: + if added_count > 0 or removed_count > 0: # Write updated videos.json with open(videos_json_path, "w", encoding="utf-8") as f: json.dump(existing_videos, f, indent=2) - print(f" Updated {videos_json_path.name} (+{added_count} videos)") + if added_count > 0: + print(f" Updated {videos_json_path.name} (+{added_count} videos)") else: print(f" No new videos to add") @@ -1018,6 +1313,41 @@ def _import_narration_segments(narration_dir: Path, config, verbose: bool) -> No if narration_json_path.exists(): existing_narration = _read_json(narration_json_path) + # Remove stale entries: no raw_mov file and source_file also gone + _raw_video_exts_set = {".mov", ".mp4", ".avi", ".mkv", ".m4v"} + raw_stems: set[str] = set() + if raw_dir.exists(): + for _f in raw_dir.iterdir(): + if _f.is_file() and _f.suffix.lower() in _raw_video_exts_set: + raw_stems.add(_f.stem.lower()) + + stale_keys = [ + seg_id + for seg_id, seg_data in existing_narration.items() + if seg_id.lower() not in raw_stems + and not (narration_dir / seg_data.get("source_file", "")).exists() + ] + for seg_id in stale_keys: + del existing_narration[seg_id] + print(f" Removed stale narration segment: {seg_id}") + + # Normalise keys to lowercase, merging case-duplicates by keeping the entry + # with more data (skip/take values are the most important thing to preserve). + normalised: dict = {} + merged_count = 0 + for seg_id, seg_data in existing_narration.items(): + lower_id = seg_id.lower() + if lower_id in normalised: + existing_has_trim = "skip" in normalised[lower_id] or "take" in normalised[lower_id] + incoming_has_trim = "skip" in seg_data or "take" in seg_data + if incoming_has_trim and not existing_has_trim: + normalised[lower_id] = seg_data + merged_count += 1 + print(f" Merged duplicate narration segment: '{seg_id}' → '{lower_id}'") + else: + normalised[lower_id] = seg_data + existing_narration = normalised + default_filters = config.default_filters if config else {} added_count = 0 @@ -1041,6 +1371,7 @@ def _import_narration_segments(narration_dir: Path, config, verbose: bool) -> No # Strip _processed suffix for cleaner segment IDs if present if segment_id.endswith("_processed"): segment_id = segment_id[:-10] + segment_id = segment_id.lower() if segment_id in existing_narration: if verbose: @@ -1066,7 +1397,7 @@ def _import_narration_segments(narration_dir: Path, config, verbose: bool) -> No # 2. Scan raw/ — add entries for raw files not yet in narration.json for video_file in _scan(raw_dir): - segment_id = video_file.stem + segment_id = video_file.stem.lower() if segment_id in existing_narration: if verbose: @@ -1089,12 +1420,20 @@ def _import_narration_segments(narration_dir: Path, config, verbose: bool) -> No added_count += 1 print(f" Added narration segment: {segment_id} (from raw_mov)") - if added_count > 0 or not narration_json_path.exists(): + removed_count = len(stale_keys) + if added_count > 0 or removed_count > 0 or merged_count > 0 or not narration_json_path.exists(): with open(narration_json_path, "w", encoding="utf-8") as f: json.dump(existing_narration, f, indent=2) - if added_count > 0: - print(f" Updated narration.json (+{added_count} segments)") + if added_count > 0 or removed_count > 0 or merged_count > 0: + parts = [] + if added_count: + parts.append(f"+{added_count}") + if removed_count: + parts.append(f"-{removed_count}") + if merged_count: + parts.append(f"merged {merged_count} duplicate(s)") + print(f" Updated narration.json ({', '.join(parts)} segments)") else: if not existing_narration: print(f" narration.json created (empty — add files to processed/ or raw/)") @@ -1327,7 +1666,12 @@ def _write_tasks_file( # ============================================================================= -def cmd_pexels(project_path: Path, verbose: bool) -> int: +def cmd_pexels( + project_path: Path, + verbose: bool, + search_query: Optional[str] = None, + search_max: int = 200, +) -> int: """Download missing Pexels videos and enrich metadata for existing ones.""" from .parser import parse_manuscript, parse_project_config, parse_videos from .pexels import ( @@ -1336,6 +1680,7 @@ def cmd_pexels(project_path: Path, verbose: bool) -> int: download_video, update_videos_json, enrich_missing_descriptions, + search_and_download, ) api_key = get_pexels_api_key() @@ -1350,20 +1695,66 @@ def cmd_pexels(project_path: Path, verbose: bool) -> int: ) return 1 - config = parse_project_config(project_path) - _, markers, _, _ = parse_manuscript(project_path) - videos, _ = parse_videos(project_path, config) + # --- Search mode: download all results for a query to the assets disk --- + if search_query: + from .cache import load_assets_config + + # Find shared_assets: prefer assets disk (LaCie), fall back to local + assets_root = load_assets_config() + if assets_root and assets_root.exists(): + pexels_dir = assets_root / "shared_assets" / "pexels" + # Use the assets disk's shared_assets/videos.json as the registry + shared_videos_json = assets_root / "shared_assets" / "videos.json" + print(f"Downloading to: {pexels_dir}") + else: + # No assets disk — find local shared_assets from project_path or cwd + local_shared = _find_shared_assets(project_path) + if not local_shared: + print("Error: shared_assets directory not found.", file=sys.stderr) + return 1 + pexels_dir = local_shared / "pexels" + shared_videos_json = local_shared / "videos.json" + print(f"Assets disk not found — downloading to local: {pexels_dir}") + + dl, sk = search_and_download( + search_query, pexels_dir, shared_videos_json, api_key, max_results=search_max + ) + print(f"\nDone — {dl} downloaded, {sk} already present.") + return 0 shared_assets_dir = _find_shared_assets(project_path) if not shared_assets_dir: print("Error: shared_assets directory not found.", file=sys.stderr) return 1 - local_videos_json = project_path / config.videos_path shared_videos_json = shared_assets_dir / "videos.json" - # 1. Download missing files + # --- Normal mode: download missing files referenced in the manuscript --- + config = parse_project_config(project_path) + _, markers, _, _ = parse_manuscript(project_path) + videos, _ = parse_videos(project_path, config) + + local_videos_json = project_path / config.videos_path + + # 1. Collect missing files: manuscript markers + background missing = find_missing_pexels_videos(markers, videos, shared_assets_dir) + + # Also check background video from project.json + if config and config.background and config.background.startswith("pexels/"): + from .cache import resolve_with_cache + + bg_id = config.background + shared_vids_raw = _read_json(shared_videos_json) if shared_videos_json.exists() else {} + bg_entry = shared_vids_raw.get(bg_id) + if bg_entry: + bg_sf = bg_entry.get("source_file", "") + bg_path = shared_assets_dir / bg_sf + resolved, _ = resolve_with_cache(bg_path, shared_assets_dir) + if not resolved.exists(): + # Only add if not already in the manuscript-marker list + if not any(vid_id == bg_id for vid_id, _ in missing): + missing.append((bg_id, bg_sf)) + failed = 0 if missing: print(f"Downloading {len(missing)} missing Pexels video(s)...") @@ -1395,6 +1786,7 @@ def cmd_pexels(project_path: Path, verbose: bool) -> int: def cmd_validate(project_path: Path, verbose: bool) -> int: """Validate project configuration.""" from .parser import ( + parse_audio, parse_manuscript, parse_project_config, parse_slides, @@ -1413,15 +1805,17 @@ def cmd_validate(project_path: Path, verbose: bool) -> int: config = parse_project_config(project_path) slides = parse_slides(project_path, config) videos, videos_dir = parse_videos(project_path, config) + audio, _ = parse_audio(project_path, config) if verbose: print(f" - Markers in manuscript: {len(markers)}") print(f" - Slides defined: {len(slides)}") print(f" - Videos defined: {len(videos)}") + print(f" - Audio cues defined: {len(audio)}") # Validate warnings = validate_project( - project_path, markers, config, slides, videos, videos_dir, malformed + project_path, markers, config, slides, videos, videos_dir, malformed, audio ) for w in warnings: print(f" Warning: {w}") @@ -1440,13 +1834,32 @@ def cmd_validate(project_path: Path, verbose: bool) -> int: def _resolve_process_cache(project_path: Path, config) -> Optional[Path]: - """Return per-project cache dir on external disk, or None if not configured.""" - if not (config and config.process_cache): - return None - p = Path(config.process_cache) - if not p.is_absolute(): - p = (project_path / p).resolve() - return p / project_path.name + """Return per-project cache dir on an external disk, or None if none is accessible. + + Resolution order: + 1. config.process_cache (project.json) — e.g. /Volumes/GnommoDisk/gnommocache + Must already exist (disk must be mounted and dir pre-created). + 2. [assets] process cache (~/.gnommo.conf) — derived as cache + e.g. /Volumes/LaCie Jens/Projects/gnommocache. + Only requires the parent to exist (volume mounted); the cache dir + itself is created on first use by the caller. + """ + from .cache import load_assets_process_cache + + # Primary: explicitly configured in project.json + if config and config.process_cache: + p = Path(config.process_cache) + if not p.is_absolute(): + p = (project_path / p).resolve() + if p.exists(): + return p / project_path.name + + # Fallback: [assets] disk (need only the parent volume/dir to be present) + assets_cache = load_assets_process_cache() + if assets_cache and assets_cache.parent.exists(): + return assets_cache / project_path.name + + return None def _narration_combined_hint(project_path: Path, config) -> str: @@ -1799,9 +2212,104 @@ def cmd_preprocess( # ============================================================================= -# Trim Command — auto-detect silence bounds for narration segments +# Trim Command — transcript-based trimming for slide-range segments # ============================================================================= +# Words so common they're useless for matching slide boundaries. +_TRIM_STOP_WORDS = frozenset({ + "a", "an", "the", "and", "or", "but", "in", "on", "at", "to", "of", + "for", "is", "it", "its", "we", "our", "this", "that", "with", "be", + "are", "was", "has", "not", "so", "as", "by", "do", "if", "up", + "you", "i", "he", "she", "they", "them", "just", "now", "can", + "than", "then", "from", "about", "into", "out", "what", "there", + "when", "how", "who", "all", "very", "also", "more", "get", "have", +}) + + +def _parse_segment_slide_range(seg_id: str) -> "tuple[int, int | None] | None": + """Parse 'S11-24' → (11, 24), 'S1-end' → (1, None), else None.""" + m = re.match(r"^s(\d+)-(?:s?(\d+)|(end))$", seg_id.lower()) + if not m: + return None + start = int(m.group(1)) + end = int(m.group(2)) if m.group(2) else None + return (start, end) + + +def _extract_slide_texts(manuscript_path: Path) -> "dict[int, str]": + """Return {slide_num: text} for each [SN] marker in the manuscript.""" + text = manuscript_path.read_text(encoding="utf-8") + parts = re.split(r"\[S(\d+)\]", text) + result: dict[int, str] = {} + i = 1 + while i + 1 < len(parts): + slide_num = int(parts[i]) + content = re.sub(r"\[[^\]]+\]", " ", parts[i + 1]) + result[slide_num] = content.strip() + i += 2 + return result + + +def _trim_content_words(text: str) -> "list[str]": + """Lowercase words stripped of punctuation, excluding stop words.""" + words = re.findall(r"[a-zA-Z0-9']+", text.lower()) + return [w for w in words if w not in _TRIM_STOP_WORDS and len(w) > 2] + + +def _find_slide_end_in_transcript( + transcript_words: list, + slide_text: str, + verbose: bool = False, +) -> "float | None": + """ + Locate where slide_text ends in the transcript and return that word's + end timestamp. Searches the tail of the transcript for the last few + content words from slide_text using a sequential fuzzy match. + Returns None if no confident match is found. + """ + target = _trim_content_words(slide_text)[-12:] + if not target: + return None + + # Normalise transcript words (strip punctuation, lowercase) + norm = [re.sub(r"[^a-z0-9']", "", w.word.lower()) for w in transcript_words] + # Index into transcript of content words only + content_idxs = [ + i for i, w in enumerate(norm) + if w and w not in _TRIM_STOP_WORDS and len(w) > 2 + ] + if not content_idxs: + return None + + n = len(target) + threshold = max(1, int(n * 0.55)) + + # Scan backwards: find the latest window of n content words that matches + for end_ci in range(len(content_idxs) - 1, n - 2, -1): + window_idxs = content_idxs[max(0, end_ci - n + 1): end_ci + 1] + window = [norm[i] for i in window_idxs] + + # Sequential match: iterate target left-to-right, consume window matches + score = 0 + wi = 0 + for t_word in target: + while wi < len(window): + w = window[wi] + wi += 1 + if w == t_word or (len(w) >= 4 and len(t_word) >= 4 and w[:4] == t_word[:4]): + score += 1 + break + + if score >= threshold: + last_tw = transcript_words[content_idxs[end_ci]] + if verbose: + print(f"\n → matched slide end: '{last_tw.word}' at {last_tw.end:.2f}s") + return last_tw.end + + if verbose: + print(f"\n → could not match slide end (target tail: {target[-5:]})") + return None + def cmd_trim( project_path: Path, @@ -1809,25 +2317,24 @@ def cmd_trim( force: bool = False, threshold_db: float = -40.0, res: str = "full", + whisper_model: str = "base", ) -> int: """ - Auto-detect silence bounds for all narration segments and write skip/take - values into narration.json. + Trim narration segments and write skip/take values into narration.json. - For each segment: - skip = max(0, first_sound_time - 0.5) - take = last_sound_time + 3.0 - skip (capped at file duration) + For segments named S{N}-{M}.mov or S{N}-end.mov: + - Transcribes audio with Whisper to get word-level timestamps + - skip = max(0, first_word.start - 0.5) + - take = (end of last word on slide M) + 0.15 - skip + - S{N}-end.mov: only trims the start, no end trim + - Transcripts are cached in narration/transcripts/{seg_id}.json - Segments that already have explicit skip or take values are left unchanged - unless --force is passed. - - Use --threshold to adjust sensitivity, e.g. -25 to ignore clothing/room - noise that sits above -40 dB. + For other segments: falls back to silence detection. """ from .parser import parse_project_config, parse_narration from .preprocessor import detect_silence_bounds, get_video_duration - print(f"Auto-trimming narration: {project_path.name}") + print(f"Trimming narration: {project_path.name}") config = parse_project_config(project_path) narration, narration_dir = parse_narration(project_path, config) @@ -1837,12 +2344,18 @@ def cmd_trim( print(" Run 'gnommo -p import' first.") return 1 - # Build a lookup of raw source files by segment ID. Raw files give cleaner - # silence detection — loudnorm can introduce early peaks in processed audio. + # Load slide texts from manuscript for transcript-based trimming + manuscript_path = project_path / "manuscript.txt" + slide_texts: dict[int, str] = {} + if manuscript_path.exists(): + slide_texts = _extract_slide_texts(manuscript_path) + _video_exts = {".mov", ".mp4", ".avi", ".mkv", ".m4v"} raw_dir = narration_dir / "raw_mov" compressed_dir = narration_dir / "raw_mp4" + transcripts_dir = narration_dir / "transcripts" + # Build lookup of raw files keyed by lowercased stem raw_lookup: dict[str, Path] = {} for search_dir in (raw_dir, compressed_dir): if search_dir.exists(): @@ -1855,7 +2368,7 @@ def cmd_trim( stem = f.stem if stem.endswith("_compressed"): stem = stem[: -len("_compressed")] - raw_lookup[stem] = f + raw_lookup[stem.lower()] = f narration_json_path = narration_dir / "narration.json" raw_data: dict = _read_json(narration_json_path) @@ -1870,7 +2383,7 @@ def cmd_trim( print(f" {seg_id}: already trimmed, skipping (use --force to redo)") continue - # Prefer raw file; fall back to processed if raw not available. + # Prefer raw file; fall back to source_file from narration.json source_path = raw_lookup.get(seg_id) if source_path is None: source_path = narration_dir / seg.source_file @@ -1878,6 +2391,68 @@ def cmd_trim( print(f" {seg_id}: source file not found, skipping") continue + slide_range = _parse_segment_slide_range(seg_id) if slide_texts else None + + if slide_range is not None: + # --- Transcript-based trimming --- + from .transcriber import transcribe_video, save_transcript, load_transcript, TranscriptionError + + start_slide, end_slide = slide_range + transcripts_dir.mkdir(parents=True, exist_ok=True) + transcript_path = transcripts_dir / f"{seg_id}.json" + + try: + if transcript_path.exists() and not force: + words = load_transcript(transcript_path) + print(f" {seg_id}: loaded cached transcript ({len(words)} words)", end="", flush=True) + else: + print(f" {seg_id}: transcribing {source_path.parent.name}/{source_path.name} (model={whisper_model})...", end="", flush=True) + words = transcribe_video(source_path, model=whisper_model) + save_transcript(words, transcript_path) + print(f" {len(words)} words", end="", flush=True) + + if not words: + print(f" — no words found, falling back to silence detection") + slide_range = None + else: + total_dur = get_video_duration(source_path) + new_skip = max(0.0, round(words[0].start - 0.5, 3)) + + if end_slide is None: + # S{N}-end: only trim start + new_take = round(total_dur - new_skip, 3) + print( + f" first={words[0].start:.2f}s" + f" → skip={new_skip:.3f}s take={new_take:.3f}s (no end trim)" + ) + else: + last_slide_text = slide_texts.get(end_slide, "") + end_ts = None + if last_slide_text: + end_ts = _find_slide_end_in_transcript(words, last_slide_text, verbose) + if end_ts is None: + print(f"\n ⚠ could not locate S{end_slide} end in transcript — trimming start only") + new_take = round(total_dur - new_skip, 3) + else: + new_take = round(min(end_ts + 0.15 - new_skip, total_dur - new_skip), 3) + new_take = max(0.0, new_take) + print( + f" first={words[0].start:.2f}s S{end_slide}_end={end_ts:.2f}s" + f" → skip={new_skip:.3f}s take={new_take:.3f}s" + ) + + raw_data[seg_id]["skip"] = new_skip + raw_data[seg_id]["take"] = new_take + updated += 1 + continue + + except Exception as exc: + from .transcriber import TranscriptionError + label = "Whisper not installed" if "openai-whisper" in str(exc) else str(exc) + print(f"\n ⚠ transcription failed ({label}), falling back to silence detection") + slide_range = None # fall through + + # --- Silence-based fallback --- print( f" {seg_id}: analysing {source_path.parent.name}/{source_path.name}...", end="", @@ -2670,7 +3245,7 @@ def _project_markers_to_videos( for marker in markers: for prefix, implied in _SHORTHAND_PREFIXES.items(): if marker.startswith(prefix): - video_id = marker[len(prefix):] + video_id = marker[len(prefix):].lower() cutout, layer = implied[0], implied[1] projection[video_id] = { "cutout": cutout, @@ -2720,14 +3295,6 @@ def _project_markers_to_videos( if updated_local: print(f" Projected marker semantics → videos.json: {', '.join(updated_local)}") - # Also project into shared_assets/videos.json for pexels/library videos - shared_assets_dir = _find_shared_assets(project_path) if project_path else None - if shared_assets_dir: - shared_videos_json = shared_assets_dir / "videos.json" - updated_shared = _apply_projection(shared_videos_json) - if updated_shared: - print(f" Projected marker semantics → shared_assets/videos.json: {', '.join(updated_shared)}") - def _writeback_video_metadata(plan, project_path, config) -> None: """Write back cutout/layer derived from shorthand markers to videos.json. diff --git a/gnommo/parser.py b/gnommo/parser.py index 1951b5a..82f91b1 100644 --- a/gnommo/parser.py +++ b/gnommo/parser.py @@ -80,9 +80,11 @@ def parse_manuscript( # Strip [cite:...] markers from text so they don't pollute alignment text = re.sub(r"\[cite:[^\]]+\]", "", text) - # Strip [marker:...] and [cue:...] markers (personal recording cues, ignored by pipeline) + # Strip narrator cues (ignored by pipeline) text = re.sub(r"\[marker:[^\]]+\]", "", text) text = re.sub(r"\[cue:[^\]]+\]", "", text) + text = re.sub(r"\[pause\]", "", text) + text = re.sub(r"\[stop\]", "", text) # Extract all valid markers like [S1], [video:demo], [vf2m:pexels/clip-name], etc. # Include / and - to capture pexels/library video IDs; . to catch file extensions in markers. @@ -395,10 +397,14 @@ def parse_timestamp(value: str) -> float: Returns: Time in seconds as a float. """ - if not value: + if value is None: return 0.0 + if isinstance(value, (int, float)): + return float(value) value = value.strip() + if not value: + return 0.0 # Remove trailing 's' if present (e.g., "3.5s") if "h" in value: diff --git a/gnommo/pexels.py b/gnommo/pexels.py index f4eb02f..5653adf 100644 --- a/gnommo/pexels.py +++ b/gnommo/pexels.py @@ -237,14 +237,13 @@ def enrich_missing_descriptions( ] # Filter to those whose file exists on disk - project_root = shared_assets_dir.parent to_enrich = [] for vid_id, entry in candidates: sf = entry.get("source_file", "") if not sf: continue path = shared_assets_dir / sf - resolved, _ = resolve_with_cache(path, project_root) + resolved, _ = resolve_with_cache(path, shared_assets_dir) if resolved.exists(): pexels_id = extract_pexels_id(sf) if pexels_id: @@ -268,6 +267,173 @@ def enrich_missing_descriptions( return updated +def _search_videos( + query: str, api_key: str, per_page: int = 80, page: int = 1 +) -> Optional[dict]: + """Call the Pexels video search API and return the raw response.""" + import urllib.parse + + params = urllib.parse.urlencode({"query": query, "per_page": per_page, "page": page}) + url = f"https://api.pexels.com/videos/search?{params}" + req = urllib.request.Request( + url, + headers={"Authorization": api_key, "User-Agent": "Mozilla/5.0 gnommo/1.0"}, + ) + try: + with urllib.request.urlopen(req, timeout=15) as resp: + return json.loads(resp.read()) + except Exception as e: + print(f" Pexels search error: {e}", flush=True) + return None + + +def _pick_best_quality(video_files: list) -> Optional[dict]: + """Pick the highest-resolution MP4 from a search result's video_files list.""" + mp4s = [f for f in video_files if f.get("file_type") == "video/mp4"] + if not mp4s: + mp4s = video_files + if not mp4s: + return None + return max(mp4s, key=lambda f: f.get("width", 0) * f.get("height", 0)) + + +def _make_source_filename(pexels_id: str, video_file: dict) -> str: + """Build a canonical filename like 12345678_1920_1080_30fps.mp4.""" + w = video_file.get("width", 0) + h = video_file.get("height", 0) + fps = round(float(video_file.get("fps") or 0)) + return f"{pexels_id}_{w}_{h}_{fps}fps.mp4" + + +def _download_bytes(url: str, target_path: Path) -> bool: + """Stream-download url to target_path with a progress indicator. Returns True on success.""" + try: + req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0 gnommo/1.0"}) + with urllib.request.urlopen(req, timeout=300) as resp: + total = int(resp.headers.get("Content-Length") or 0) + done = 0 + chunks: list[bytes] = [] + while True: + chunk = resp.read(524288) # 512 KB + if not chunk: + break + chunks.append(chunk) + done += len(chunk) + if total: + pct = done * 100 // total + print( + f" {pct:3d}% {done/1048576:.1f}/{total/1048576:.1f} MB\r", + end="", + flush=True, + ) + print(f" Done — {done/1048576:.1f} MB ", flush=True) + target_path.write_bytes(b"".join(chunks)) + return True + except Exception as e: + print(f"\n Download failed: {e}", flush=True) + return False + + +def search_and_download( + query: str, + pexels_dir: Path, + shared_videos_json: Path, + api_key: str, + max_results: int = 200, +) -> tuple[int, int]: + """Search Pexels for *query* and download all results to pexels_dir. + + Each video is saved as ``pexels_dir/{pexels_id}_{w}_{h}_{fps}fps.mp4`` and + registered in *shared_videos_json* so the renderer can find it. + + Returns (downloaded_count, skipped_count). + """ + print(f"Searching Pexels for '{query}'...", flush=True) + pexels_dir.mkdir(parents=True, exist_ok=True) + + # Load existing registry so we can skip already-downloaded videos + existing: dict = {} + if shared_videos_json.exists(): + with open(shared_videos_json, "r", encoding="utf-8") as f: + existing = json.load(f) + + downloaded = 0 + skipped = 0 + page = 1 + + while downloaded + skipped < max_results: + per_page = min(80, max_results - downloaded - skipped) + result = _search_videos(query, api_key, per_page=per_page, page=page) + if not result: + break + + videos = result.get("videos", []) + if not videos: + break + + total_results = result.get("total_results", 0) + print( + f" Page {page}: {len(videos)} result(s) (Pexels total: {total_results})", + flush=True, + ) + + for video in videos: + pexels_id = str(video.get("id", "")) + video_files = video.get("video_files", []) + if not pexels_id or not video_files: + continue + + best = _pick_best_quality(video_files) + if not best: + continue + + filename = _make_source_filename(pexels_id, best) + video_id = f"pexels/{Path(filename).stem}" + target_path = pexels_dir / filename + source_file = f"pexels/{filename}" + + # Skip if already registered or file already on disk + if video_id in existing or target_path.exists(): + if video_id not in existing: + # File exists but not registered — register it + pass + else: + skipped += 1 + continue + + description = description_from_url(video.get("url", "")) + duration = float(video.get("duration") or 0) or None + w = best.get("width", "?") + h = best.get("height", "?") + fps = best.get("fps", "?") + q = best.get("quality", "?") + label = f'"{description}" — ' if description else "" + print(f" [{pexels_id}] {label}{q} {w}x{h} @ {fps}fps", flush=True) + print(f" → {target_path}", flush=True) + + if not target_path.exists(): + if not _download_bytes(best["link"], target_path): + continue + + # Register in shared videos.json + existing[video_id] = { + "source_file": source_file, + "description": description, + "duration": duration, + "has_audio": False, + } + with open(shared_videos_json, "w", encoding="utf-8") as f: + json.dump(existing, f, indent=2, ensure_ascii=False) + + downloaded += 1 + + if not result.get("next_page"): + break + page += 1 + + return downloaded, skipped + + def find_missing_pexels_videos( manuscript_markers: list[str], videos: dict, @@ -293,19 +459,20 @@ def find_missing_pexels_videos( prefix = next((p for p in _VIDEO_PREFIXES if marker.startswith(p)), None) if prefix is None: continue - video_id = marker[len(prefix):] + video_id = marker[len(prefix):].lower() if video_id in seen or not video_id.startswith("pexels/"): continue seen.add(video_id) source_file = videos.get(video_id, None) if source_file is None: - continue - sf = source_file.source_file if hasattr(source_file, "source_file") else source_file + # Not in videos.json yet — synthesize expected path from the ID + sf = video_id + ".mp4" + else: + sf = source_file.source_file if hasattr(source_file, "source_file") else source_file candidate = shared_assets_dir / sf - # resolve_with_cache needs a project_path — use shared_assets parent - resolved, _ = resolve_with_cache(candidate, shared_assets_dir.parent) + resolved, _ = resolve_with_cache(candidate, shared_assets_dir) if not resolved.exists(): missing.append((video_id, sf)) diff --git a/gnommo/preprocessor.py b/gnommo/preprocessor.py index 2303619..fb33f33 100644 --- a/gnommo/preprocessor.py +++ b/gnommo/preprocessor.py @@ -2262,12 +2262,7 @@ def stitch_narration_segments( Returns: Path to the stitched video file. """ - if len(segment_ids) == 1: - # Single segment - just return its processed path - video_source = videos[segment_ids[0]] - return get_preprocessed_path(videos_dir, video_source) - - print(f" Concatenating {len(segment_ids)} narration segments...") + print(f" Concatenating {len(segment_ids)} narration segment(s)...") # Create temp directory for trimmed segments temp_dir = output_path.parent / "concat_temp" diff --git a/gnommo/renderer.py b/gnommo/renderer.py index bc3c047..4ebd6d6 100644 --- a/gnommo/renderer.py +++ b/gnommo/renderer.py @@ -261,15 +261,21 @@ def _resolve_video_path( def _has_audio_stream(video_path: Path) -> bool: - """Check if a video file contains a non-empty audio stream. + """Check if a video file contains a non-empty, decodable audio stream. Uses -analyzeduration 0 to avoid the slow avformat_find_stream_info() scan that happens when an MP4 has a declared audio track with no actual frames — ffprobe would otherwise scan the entire file looking for audio packets. - Also checks nb_frames to reject ghost audio tracks (stream header exists in - the moov atom but no sample data in stsc/stsz). + Rejects: + - Ghost audio tracks (stream header exists but no sample data, nb_frames=0) + - Codecs FFmpeg cannot decode (e.g. 'apac' = Apple Packed Audio Codec used + on Apple Silicon Macs; causes "matches no streams" in the filtergraph) """ + # Codecs that are declared in the container but FFmpeg cannot decode. + # These appear as "Audio: none (apac / 0x63617061)" in FFmpeg output. + _UNSUPPORTED_CODECS = {"apac"} + result = subprocess.run( [ "ffprobe", @@ -282,7 +288,7 @@ def _has_audio_stream(video_path: Path) -> bool: "-select_streams", "a:0", "-show_entries", - "stream=index,nb_frames", + "stream=index,nb_frames,codec_name", "-of", "csv=p=0", str(video_path), @@ -293,8 +299,12 @@ def _has_audio_stream(video_path: Path) -> bool: output = result.stdout.strip() if not output: return False - # output is "index" or "index,nb_frames" + # output is "index,nb_frames,codec_name" (or subset if fields are missing) parts = output.split(",") + if len(parts) >= 3: + codec_name = parts[2].strip() + if codec_name in _UNSUPPORTED_CODECS: + return False # Codec declared but not decodable by FFmpeg if len(parts) >= 2: nb_frames = parts[1].strip() if nb_frames == "0": @@ -419,6 +429,17 @@ def build_ffmpeg_command(plan: RenderPlan, output_path: Path) -> list[str]: video_inputs: dict[int, int] = {} # event_index -> input_idx video_events_with_audio: set[int] = set() # event indices whose files have audio + # Live-probe cache: avoids probing the same file twice within one render call. + # Always probe rather than trusting cached has_audio — stale values (e.g. from when a + # file was local but is now only on an external disk with different content) cause + # "[N:a] matches no streams" errors that crash the render. + _audio_probe_cache: dict[Path, bool] = {} + + def _probe_has_audio(path: Path) -> bool: + if path not in _audio_probe_cache: + _audio_probe_cache[path] = _has_audio_stream(path) + return _audio_probe_cache[path] + for i, event in enumerate(plan.video_events): video_path = _resolve_video_path( videos_dir, event.video_source, shared_assets_dir, project_path @@ -437,11 +458,22 @@ def build_ffmpeg_command(plan: RenderPlan, output_path: Path) -> list[str]: remaining = event.video_source.duration - skip needs_loop = remaining < clip_duration - 0.1 # 0.1 s tolerance + # Always live-probe audio presence — cached has_audio can be stale (e.g. when + # files moved from local to external disk). Results are cached per path so each + # unique file is only probed once per render call. + has_audio = _probe_has_audio(video_path) + if has_audio: + video_events_with_audio.add(i) + if needs_loop: cmd.extend(["-stream_loop", "-1"]) if skip > 0: cmd.extend(["-ss", f"{skip:.3f}"]) - cmd.extend(["-analyzeduration", "0", "-probesize", "1000"]) + # Use 1 MB probesize for files with audio so FFmpeg can locate the audio stream + # in the moov atom (large UHD files may have moov atoms > 1 KB). Keep 1000 for + # video-only files to skip ghost-track scanning without a performance cost. + probesize = "1000000" if has_audio else "1000" + cmd.extend(["-analyzeduration", "0", "-probesize", probesize]) # Use pre-probed duration (or loop-limited duration) to tell FFmpeg exactly # how much to read, preventing scans of ghost audio tracks on empty streams. if needs_loop: @@ -453,14 +485,6 @@ def build_ffmpeg_command(plan: RenderPlan, output_path: Path) -> list[str]: cmd.extend(["-i", str(video_path)]) video_inputs[i] = input_idx input_idx += 1 - has_audio = event.video_source.has_audio - if has_audio is None: - print( - f" Warning: no cached metadata for '{event.video_source.source_file}' — run 'gnommo import' to avoid slow probing" - ) - has_audio = _has_audio_stream(video_path) - if has_audio: - video_events_with_audio.add(i) # Input: outro videos (play after narration ends) outro_inputs: dict[int, int] = {} # event_index -> input_idx @@ -471,9 +495,14 @@ def build_ffmpeg_command(plan: RenderPlan, output_path: Path) -> list[str]: videos_dir, event.video_source, shared_assets_dir, project_path ) skip = event.video_source.skip or 0.0 + has_audio = _probe_has_audio(video_path) + if has_audio: + outro_events_with_audio.add(i) + if skip > 0: cmd.extend(["-ss", f"{skip:.3f}"]) - cmd.extend(["-analyzeduration", "0", "-probesize", "1000"]) + probesize = "1000000" if has_audio else "1000" + cmd.extend(["-analyzeduration", "0", "-probesize", probesize]) if event.video_source.duration is not None: remaining = event.video_source.duration - skip if remaining > 0: @@ -481,14 +510,6 @@ def build_ffmpeg_command(plan: RenderPlan, output_path: Path) -> list[str]: cmd.extend(["-i", str(video_path)]) outro_inputs[i] = input_idx input_idx += 1 - has_audio = event.video_source.has_audio - if has_audio is None: - print( - f" Warning: no cached metadata for '{event.video_source.source_file}' — run 'gnommo import' to avoid slow probing" - ) - has_audio = _has_audio_stream(video_path) - if has_audio: - outro_events_with_audio.add(i) # Track where audio inputs start num_inputs_before_audio = input_idx diff --git a/gnommo/transformer.py b/gnommo/transformer.py index e519684..0293d6d 100644 --- a/gnommo/transformer.py +++ b/gnommo/transformer.py @@ -831,11 +831,15 @@ def build_render_plan( ) ) - # Offset all events that come AFTER this pause + # Offset all events that come AFTER this pause. + # Use >= so a slide that transitions at exactly narration_time is + # pushed past the pause (matching the >= already used for video events). + # Also extend the end_time of the current slide so it stays visible + # as the background behind the pause-video overlay, avoiding a gap. for slide_event in slide_events: - if slide_event.start_time > narration_time: + if slide_event.start_time >= narration_time: slide_event.start_time += pause_duration - if slide_event.end_time > narration_time: + if slide_event.end_time >= narration_time: slide_event.end_time += pause_duration for vid_event in video_events: @@ -1100,7 +1104,7 @@ def _extract_video_events( (p for p in _SHORTHAND_PREFIXES if mid.startswith(p)), None ) if shorthand_match: - video_id = mid[len(shorthand_match) :] + video_id = mid[len(shorthand_match) :].lower() if video_id not in videos: warnings.append( f"[{mid}] references unknown video '{video_id}' — skipped. " @@ -1121,7 +1125,7 @@ def _extract_video_events( # --- legacy [video:xxx] --- if mid.startswith("video:"): - video_id = mid[6:] + video_id = mid[6:].lower() if video_id not in videos: warnings.append( f"[video:{video_id}] references unknown video '{video_id}' — skipped." @@ -1138,7 +1142,7 @@ def _extract_video_events( # --- [narration:xxx] --- if mid.startswith("narration:"): - video_id = mid[10:] + video_id = mid[10:].lower() if video_id not in videos: warnings.append( f"[narration:{video_id}] references unknown video '{video_id}' — skipped." diff --git a/gnommo/validator.py b/gnommo/validator.py index 2d68fca..392732e 100644 --- a/gnommo/validator.py +++ b/gnommo/validator.py @@ -22,6 +22,7 @@ def validate_project( videos: dict[str, VideoSource], videos_dir: Path, malformed_markers: list[tuple[int, str]] = None, + audio: dict = None, ) -> list[ValidationIssue]: """ Validate all parsed project data. Raises ValidationError if any issues found. @@ -34,6 +35,7 @@ def validate_project( - Background video exists (if specified) - Slide types are valid - No malformed markers in manuscript + - All audio: markers in manuscript exist in audio.json """ issues: list[ValidationIssue] = [] warnings: list[ValidationIssue] = [] @@ -52,9 +54,9 @@ def validate_project( for marker in manuscript_markers: prefix = next((p for p in _VIDEO_PREFIXES if marker.startswith(p)), None) if prefix is not None: - referenced_video_ids.add(marker[_VIDEO_PREFIXES[prefix]:]) + referenced_video_ids.add(marker[_VIDEO_PREFIXES[prefix]:].lower()) elif marker.startswith("narration:"): - referenced_video_ids.add(marker[10:]) + referenced_video_ids.add(marker[10:].lower()) # Check for malformed markers first (these are likely typos) if malformed_markers: @@ -84,7 +86,7 @@ def validate_project( (p for p in _VIDEO_PREFIXES if marker.startswith(p)), None ) if matched_prefix is not None: - video_id = marker[_VIDEO_PREFIXES[matched_prefix] :] + video_id = marker[_VIDEO_PREFIXES[matched_prefix] :].lower() if video_id not in videos: hint = "" if "." in video_id: @@ -111,7 +113,7 @@ def validate_project( # Validate narration trigger markers (narration:xxx) - continuous videos if marker.startswith("narration:"): - video_id = marker[10:] # Remove 'narration:' prefix + video_id = marker[10:].lower() # Remove 'narration:' prefix if video_id not in videos: warnings.append( ValidationIssue( @@ -135,6 +137,10 @@ def validate_project( if marker.startswith("segment:"): continue + # Bare narrator cues (teleprompter hints, not pipeline markers) + if marker in ("pause", "stop"): + continue + # Unknown namespaced markers (e.g. [background:xxx]) — not supported, ignore with warning if ":" in marker: warnings.append( @@ -320,6 +326,26 @@ def validate_project( ) ) + # Check all audio: markers in manuscript exist in audio.json + if audio is not None: + seen_audio: set[str] = set() + manuscript_path = project_path / "manuscript.txt" + for marker in manuscript_markers: + audio_id = None + if marker.startswith("audio:"): + audio_id = marker[6:] + elif marker.startswith("A") and len(marker) > 1 and marker[1:].isalnum(): + audio_id = marker[1:] + if audio_id and audio_id not in seen_audio: + seen_audio.add(audio_id) + if audio_id not in audio: + warnings.append( + ValidationIssue( + f"Audio marker [{marker}] referenced in manuscript but '{audio_id}' not defined in audio.json — will be silent at render", + manuscript_path, + ) + ) + # If any issues, raise ValidationError if issues: raise ValidationError(issues) diff --git a/renderforce.sh b/renderforce.sh new file mode 100755 index 0000000..a56d845 --- /dev/null +++ b/renderforce.sh @@ -0,0 +1,9 @@ +#!/bin/sh + +./gnommo.sh -p video1 render --force --prod +./gnommo.sh -p video2 render --force --prod +./gnommo.sh -p video3 render --force --prod +./gnommo.sh -p video4 render --force --prod +./gnommo.sh -p video5 render --force --prod +#./gnommo.sh -p video6 render --force --prod +