From 83715e85e8128df99c68c12e208e8529e605f55d Mon Sep 17 00:00:00 2001 From: jenstandstad Date: Wed, 15 Jul 2026 19:47:31 +0200 Subject: [PATCH] The stitch cleanup and the gnommokey improvement --- .../transcripts/talking_head_S1.json | 497 ++++++++ .../transcripts/talking_head_S3.json | 497 ++++++++ example/tasks.md | 2 +- gnommo/cli.py | 1128 ++--------------- gnommo/models.py | 9 +- gnommo/narration.py | 4 +- gnommo/preprocessor.py | 220 +--- gnommo/renderer.py | 2 +- gnommo/state.py | 2 +- gnommo/transcriber.py | 85 -- gnommo/transfer.py | 1 - 11 files changed, 1162 insertions(+), 1285 deletions(-) create mode 100644 example/media/narration/transcripts/talking_head_S1.json create mode 100644 example/media/narration/transcripts/talking_head_S3.json diff --git a/example/media/narration/transcripts/talking_head_S1.json b/example/media/narration/transcripts/talking_head_S1.json new file mode 100644 index 0000000..518b134 --- /dev/null +++ b/example/media/narration/transcripts/talking_head_S1.json @@ -0,0 +1,497 @@ +[ + { + "word": "This", + "start": 10.74, + "end": 11.44 + }, + { + "word": "is", + "start": 11.44, + "end": 11.64 + }, + { + "word": "the", + "start": 11.64, + "end": 11.82 + }, + { + "word": "first", + "start": 11.82, + "end": 12.04 + }, + { + "word": "slide.", + "start": 12.04, + "end": 12.44 + }, + { + "word": "It", + "start": 12.92, + "end": 13.34 + }, + { + "word": "appears", + "start": 13.34, + "end": 13.7 + }, + { + "word": "immediate.", + "start": 13.7, + "end": 14.18 + }, + { + "word": "However,", + "start": 15.36, + "end": 16.06 + }, + { + "word": "this", + "start": 16.38, + "end": 16.48 + }, + { + "word": "is", + "start": 16.48, + "end": 16.62 + }, + { + "word": "the", + "start": 16.62, + "end": 16.8 + }, + { + "word": "second", + "start": 16.8, + "end": 17.08 + }, + { + "word": "slide.", + "start": 17.08, + "end": 17.42 + }, + { + "word": "It", + "start": 17.78, + "end": 18.02 + }, + { + "word": "should", + "start": 18.02, + "end": 18.24 + }, + { + "word": "appear", + "start": 18.24, + "end": 18.56 + }, + { + "word": "one", + "start": 18.56, + "end": 19.02 + }, + { + "word": "second", + "start": 19.02, + "end": 19.5 + }, + { + "word": "prior", + "start": 19.5, + "end": 19.92 + }, + { + "word": "to", + "start": 19.92, + "end": 20.16 + }, + { + "word": "the", + "start": 20.16, + "end": 20.26 + }, + { + "word": "word", + "start": 20.26, + "end": 20.54 + }, + { + "word": "when", + "start": 20.54, + "end": 21.24 + }, + { + "word": "I", + "start": 21.24, + "end": 21.32 + }, + { + "word": "say", + "start": 21.32, + "end": 21.5 + }, + { + "word": "whoever", + "start": 21.5, + "end": 21.86 + }, + { + "word": "first", + "start": 21.86, + "end": 22.44 + }, + { + "word": "time.", + "start": 22.44, + "end": 22.7 + }, + { + "word": "This", + "start": 24.3, + "end": 25.0 + }, + { + "word": "is", + "start": 25.0, + "end": 25.14 + }, + { + "word": "me", + "start": 25.14, + "end": 25.38 + }, + { + "word": "taking,", + "start": 25.38, + "end": 25.78 + }, + { + "word": "talking", + "start": 26.14, + "end": 27.18 + }, + { + "word": "alongside", + "start": 27.18, + "end": 27.66 + }, + { + "word": "a", + "start": 27.66, + "end": 27.92 + }, + { + "word": "video.", + "start": 27.92, + "end": 28.16 + }, + { + "word": "The", + "start": 28.68, + "end": 28.96 + }, + { + "word": "video", + "start": 28.96, + "end": 29.2 + }, + { + "word": "is", + "start": 29.2, + "end": 29.4 + }, + { + "word": "constrained", + "start": 29.4, + "end": 29.82 + }, + { + "word": "within", + "start": 29.82, + "end": 30.18 + }, + { + "word": "the", + "start": 30.18, + "end": 30.36 + }, + { + "word": "red", + "start": 30.36, + "end": 30.52 + }, + { + "word": "square.", + "start": 30.52, + "end": 30.94 + }, + { + "word": "Notice", + "start": 31.3, + "end": 31.48 + }, + { + "word": "how", + "start": 31.48, + "end": 31.78 + }, + { + "word": "the", + "start": 31.78, + "end": 31.96 + }, + { + "word": "video", + "start": 31.96, + "end": 32.16 + }, + { + "word": "stops", + "start": 32.16, + "end": 32.48 + }, + { + "word": "immediately", + "start": 32.48, + "end": 32.98 + }, + { + "word": "when", + "start": 32.98, + "end": 33.4 + }, + { + "word": "we", + "start": 33.4, + "end": 33.58 + }, + { + "word": "make", + "start": 33.58, + "end": 33.76 + }, + { + "word": "the", + "start": 33.76, + "end": 34.0 + }, + { + "word": "transition", + "start": 34.0, + "end": 34.42 + }, + { + "word": "to", + "start": 34.42, + "end": 34.72 + }, + { + "word": "the", + "start": 34.72, + "end": 34.84 + }, + { + "word": "next", + "start": 34.84, + "end": 35.06 + }, + { + "word": "slide.", + "start": 35.06, + "end": 35.48 + }, + { + "word": "I", + "start": 37.2, + "end": 37.76 + }, + { + "word": "will", + "start": 37.76, + "end": 37.82 + }, + { + "word": "continue", + "start": 37.82, + "end": 38.12 + }, + { + "word": "to", + "start": 38.12, + "end": 38.34 + }, + { + "word": "talk", + "start": 38.34, + "end": 38.58 + }, + { + "word": "without", + "start": 38.58, + "end": 38.92 + }, + { + "word": "pause,", + "start": 38.92, + "end": 39.26 + }, + { + "word": "but", + "start": 39.5, + "end": 39.6 + }, + { + "word": "in", + "start": 39.6, + "end": 39.72 + }, + { + "word": "the", + "start": 39.72, + "end": 39.8 + }, + { + "word": "finished", + "start": 39.8, + "end": 40.0 + }, + { + "word": "recording", + "start": 40.0, + "end": 40.48 + }, + { + "word": "there", + "start": 40.48, + "end": 41.22 + }, + { + "word": "will", + "start": 41.22, + "end": 41.38 + }, + { + "word": "be", + "start": 41.38, + "end": 41.58 + }, + { + "word": "a", + "start": 41.58, + "end": 41.68 + }, + { + "word": "pause", + "start": 41.68, + "end": 41.96 + }, + { + "word": "before", + "start": 41.96, + "end": 42.32 + }, + { + "word": "the", + "start": 42.32, + "end": 42.52 + }, + { + "word": "narration", + "start": 42.52, + "end": 43.06 + }, + { + "word": "continues.", + "start": 43.06, + "end": 43.66 + }, + { + "word": "Now", + "start": 44.44, + "end": 44.56 + }, + { + "word": "a", + "start": 44.56, + "end": 44.7 + }, + { + "word": "video", + "start": 44.7, + "end": 44.94 + }, + { + "word": "will", + "start": 44.94, + "end": 45.12 + }, + { + "word": "play", + "start": 45.12, + "end": 45.4 + }, + { + "word": "that", + "start": 45.4, + "end": 45.8 + }, + { + "word": "pauses", + "start": 45.8, + "end": 46.52 + }, + { + "word": "the", + "start": 46.52, + "end": 46.8 + }, + { + "word": "narration.", + "start": 46.8, + "end": 47.22 + }, + { + "word": "Notice", + "start": 48.66, + "end": 49.22 + }, + { + "word": "how", + "start": 49.22, + "end": 49.44 + }, + { + "word": "my", + "start": 49.44, + "end": 49.6 + }, + { + "word": "voice", + "start": 49.6, + "end": 49.84 + }, + { + "word": "continues", + "start": 49.84, + "end": 50.38 + }, + { + "word": "after", + "start": 50.38, + "end": 50.88 + }, + { + "word": "the", + "start": 50.88, + "end": 51.04 + }, + { + "word": "video", + "start": 51.04, + "end": 51.28 + }, + { + "word": "finished.", + "start": 51.28, + "end": 51.8 + } +] \ No newline at end of file diff --git a/example/media/narration/transcripts/talking_head_S3.json b/example/media/narration/transcripts/talking_head_S3.json new file mode 100644 index 0000000..30c9de6 --- /dev/null +++ b/example/media/narration/transcripts/talking_head_S3.json @@ -0,0 +1,497 @@ +[ + { + "word": "This", + "start": 10.632, + "end": 11.312 + }, + { + "word": "is", + "start": 11.312, + "end": 11.512 + }, + { + "word": "the", + "start": 11.512, + "end": 11.692 + }, + { + "word": "first", + "start": 11.692, + "end": 11.912 + }, + { + "word": "slide.", + "start": 11.912, + "end": 12.312 + }, + { + "word": "It", + "start": 12.852, + "end": 13.192 + }, + { + "word": "appears", + "start": 13.192, + "end": 13.552 + }, + { + "word": "immediate.", + "start": 13.552, + "end": 14.032 + }, + { + "word": "However,", + "start": 15.452, + "end": 15.932 + }, + { + "word": "this", + "start": 16.272, + "end": 16.352 + }, + { + "word": "is", + "start": 16.352, + "end": 16.492 + }, + { + "word": "the", + "start": 16.492, + "end": 16.652 + }, + { + "word": "second", + "start": 16.652, + "end": 16.952 + }, + { + "word": "slide.", + "start": 16.952, + "end": 17.292 + }, + { + "word": "It", + "start": 17.572, + "end": 17.872 + }, + { + "word": "should", + "start": 17.872, + "end": 18.112 + }, + { + "word": "appear", + "start": 18.112, + "end": 18.432 + }, + { + "word": "one", + "start": 18.432, + "end": 18.892 + }, + { + "word": "second", + "start": 18.892, + "end": 19.372 + }, + { + "word": "prior", + "start": 19.372, + "end": 19.792 + }, + { + "word": "to", + "start": 19.792, + "end": 20.032 + }, + { + "word": "the", + "start": 20.032, + "end": 20.152 + }, + { + "word": "word", + "start": 20.152, + "end": 20.412 + }, + { + "word": "when", + "start": 20.412, + "end": 21.112 + }, + { + "word": "I", + "start": 21.112, + "end": 21.192 + }, + { + "word": "say", + "start": 21.192, + "end": 21.352 + }, + { + "word": "whoever", + "start": 21.352, + "end": 21.732 + }, + { + "word": "first", + "start": 21.732, + "end": 22.312 + }, + { + "word": "time.", + "start": 22.312, + "end": 22.592 + }, + { + "word": "This", + "start": 24.532, + "end": 24.872 + }, + { + "word": "is", + "start": 24.872, + "end": 25.032 + }, + { + "word": "me", + "start": 25.032, + "end": 25.252 + }, + { + "word": "taking,", + "start": 25.252, + "end": 25.652 + }, + { + "word": "talking", + "start": 26.092, + "end": 27.052 + }, + { + "word": "alongside", + "start": 27.052, + "end": 27.532 + }, + { + "word": "a", + "start": 27.532, + "end": 27.792 + }, + { + "word": "video.", + "start": 27.792, + "end": 28.052 + }, + { + "word": "The", + "start": 28.652, + "end": 28.832 + }, + { + "word": "video", + "start": 28.832, + "end": 29.092 + }, + { + "word": "is", + "start": 29.092, + "end": 29.272 + }, + { + "word": "constrained", + "start": 29.272, + "end": 29.712 + }, + { + "word": "within", + "start": 29.712, + "end": 30.052 + }, + { + "word": "the", + "start": 30.052, + "end": 30.232 + }, + { + "word": "red", + "start": 30.232, + "end": 30.392 + }, + { + "word": "square.", + "start": 30.392, + "end": 30.792 + }, + { + "word": "Notice", + "start": 30.792, + "end": 31.352 + }, + { + "word": "how", + "start": 31.352, + "end": 31.652 + }, + { + "word": "the", + "start": 31.652, + "end": 31.832 + }, + { + "word": "video", + "start": 31.832, + "end": 32.032 + }, + { + "word": "stops", + "start": 32.032, + "end": 32.372 + }, + { + "word": "immediately", + "start": 32.372, + "end": 32.852 + }, + { + "word": "when", + "start": 32.852, + "end": 33.272 + }, + { + "word": "we", + "start": 33.272, + "end": 33.452 + }, + { + "word": "make", + "start": 33.452, + "end": 33.632 + }, + { + "word": "the", + "start": 33.632, + "end": 33.872 + }, + { + "word": "transition", + "start": 33.872, + "end": 34.292 + }, + { + "word": "to", + "start": 34.292, + "end": 34.592 + }, + { + "word": "the", + "start": 34.592, + "end": 34.712 + }, + { + "word": "next", + "start": 34.712, + "end": 34.932 + }, + { + "word": "slide.", + "start": 34.932, + "end": 35.392 + }, + { + "word": "I", + "start": 37.112, + "end": 37.632 + }, + { + "word": "will", + "start": 37.632, + "end": 37.692 + }, + { + "word": "continue", + "start": 37.692, + "end": 37.992 + }, + { + "word": "to", + "start": 37.992, + "end": 38.212 + }, + { + "word": "talk", + "start": 38.212, + "end": 38.452 + }, + { + "word": "without", + "start": 38.452, + "end": 38.792 + }, + { + "word": "pause,", + "start": 38.792, + "end": 39.132 + }, + { + "word": "but", + "start": 39.372, + "end": 39.472 + }, + { + "word": "in", + "start": 39.472, + "end": 39.592 + }, + { + "word": "the", + "start": 39.592, + "end": 39.652 + }, + { + "word": "finished", + "start": 39.652, + "end": 39.872 + }, + { + "word": "recording", + "start": 39.872, + "end": 40.352 + }, + { + "word": "there", + "start": 40.352, + "end": 41.092 + }, + { + "word": "will", + "start": 41.092, + "end": 41.252 + }, + { + "word": "be", + "start": 41.252, + "end": 41.452 + }, + { + "word": "a", + "start": 41.452, + "end": 41.552 + }, + { + "word": "pause", + "start": 41.552, + "end": 41.812 + }, + { + "word": "before", + "start": 41.812, + "end": 42.192 + }, + { + "word": "the", + "start": 42.192, + "end": 42.392 + }, + { + "word": "narration", + "start": 42.392, + "end": 42.932 + }, + { + "word": "continues.", + "start": 42.932, + "end": 43.552 + }, + { + "word": "Now", + "start": 44.232, + "end": 44.432 + }, + { + "word": "a", + "start": 44.432, + "end": 44.572 + }, + { + "word": "video", + "start": 44.572, + "end": 44.812 + }, + { + "word": "will", + "start": 44.812, + "end": 44.972 + }, + { + "word": "play", + "start": 44.972, + "end": 45.272 + }, + { + "word": "that", + "start": 45.272, + "end": 45.672 + }, + { + "word": "pauses", + "start": 45.672, + "end": 46.412 + }, + { + "word": "the", + "start": 46.412, + "end": 46.672 + }, + { + "word": "narration.", + "start": 46.672, + "end": 47.092 + }, + { + "word": "Notice", + "start": 48.352, + "end": 49.092 + }, + { + "word": "how", + "start": 49.092, + "end": 49.312 + }, + { + "word": "my", + "start": 49.312, + "end": 49.492 + }, + { + "word": "voice", + "start": 49.492, + "end": 49.752 + }, + { + "word": "continues", + "start": 49.752, + "end": 50.272 + }, + { + "word": "after", + "start": 50.272, + "end": 50.752 + }, + { + "word": "the", + "start": 50.752, + "end": 50.932 + }, + { + "word": "video", + "start": 50.932, + "end": 51.152 + }, + { + "word": "finished.", + "start": 51.152, + "end": 51.652 + } +] \ No newline at end of file diff --git a/example/tasks.md b/example/tasks.md index b335843..5d74b27 100644 --- a/example/tasks.md +++ b/example/tasks.md @@ -4,7 +4,7 @@ _Generated: 2026-07-15_ ## Slide Alignment Issues (7) Slide markers that could not be matched to the spoken narration (likely adlibbed). -- [ ] `S6` — _"(repaired: voice continues after the video finished)"_ +- [ ] `S6` — _"(end of: voice continues after the video finished)"_ - [ ] `S7` — _"This is the first slide. It appears immediately."_ - [ ] `S8` — _"However, this is the second slide. It should appea"_ - [ ] `S9` — _"This is me talking alongside a video. The video is"_ diff --git a/gnommo/cli.py b/gnommo/cli.py index 967edf1..7c26aea 100644 --- a/gnommo/cli.py +++ b/gnommo/cli.py @@ -41,7 +41,6 @@ Examples: gnommo -p video1 clear Delete preprocessed outputs so preprocess re-runs them gnommo -p video1 prune Remove unused entries from videos.json/audio.json/narration.json gnommo -p video1 prune --dry-run Preview which manifest entries would be removed - gnommo -p video1 stitch --res tiny -f Fast stitch with new begin/end values gnommo -p video1 trim Auto-detect silence and set skip/take in narration.json gnommo -p video1 trim --force Redo trim even for segments that already have skip/take gnommo -p video1 trim --threshold -25 Raise threshold to ignore clothing/room noise @@ -54,11 +53,12 @@ Examples: gnommo -p video1 transcode --processed --dry-run Preview what would be compressed gnommo -p video1 transcode --force Re-transcode even if output already exists gnommo -p video0 new Create a new project with standard folder structure - gnommo -p video1 all Full pipeline: import → preprocess → trim → stitch → render → push → handoff → up + gnommo -p video1 all Full pipeline: import → preprocess → trim → render → push → handoff → up gnommo -p video1 render --dry-run Show FFmpeg command without running + gnommo -p video1 grade Sample a few seconds of a raw_mov clip through the talkinghead filters for grading + gnommo -p video1 grade --ss 12 --dur 4 Seek 12s in, produce a 4s preview + gnommo -p video1 grade --file media/narration/raw_mov/clipA.mov Grade a specific raw clip gnommo -p video1 description Generate YouTube description file - gnommo -p video1 transcribe Narration file for timing of slides - gnommo -p video1 transcribe --final Transcribe outputted file and generate SRT for YouTube gnommo -p video1 archive Copy project to connected external drive gnommo -p video1 load Copy project from external drive to local gnommo -p video1 commit -m "msg" Record a commit to commits.log (required before up) @@ -75,10 +75,6 @@ Examples: gnommo -p video1 handoff Upload the rendered video to the local server gnommo -p video1 handoff --file X Upload a specific video file instead of out/ Note: 'push' sends metadata (script/slides/etc); 'handoff' uploads the actual video file. - gnommo -p video1 extract-audio --combined Extract audio from narration_combined.mov - gnommo -p video1 extract-audio --combined --channel left Extract left channel only - gnommo -p video1 extract-audio --segment seg01 Extract from a specific segment - gnommo -p video1 master Extract raw + processed audio for A/B comparison """, ) parser.add_argument( @@ -104,11 +100,10 @@ Examples: "validate", "preprocess", "pre", - "stitch", "trim", "render", + "grade", "all", - "transcribe", "align", "import", "description", @@ -117,8 +112,6 @@ Examples: "commit", "up", "down", - "extract-audio", - "master", "push", "pull", "handoff", @@ -175,28 +168,6 @@ Examples: default=1, help="Number of parallel workers for preprocessing (default: 1)", ) - parser.add_argument( - "--final", - action="store_true", - help="For transcribe: transcribe the final rendered video and generate SRT captions for YouTube", - ) - parser.add_argument( - "--segment", - type=str, - help="For extract-audio: specific segment ID to extract (default: all segments)", - ) - parser.add_argument( - "--channel", - type=str, - choices=["auto", "left", "right", "both"], - default="both", - help="For extract-audio: which audio channel(s) to extract (default: both)", - ) - parser.add_argument( - "--combined", - action="store_true", - help="For extract-audio: extract from narration_combined.mov instead of individual segments", - ) parser.add_argument( "--file", default=None, @@ -263,6 +234,20 @@ Examples: dest="search_max", help="For pexels --search: maximum number of videos to download (default: 200)", ) + parser.add_argument( + "--ss", + type=float, + default=None, + dest="grade_ss", + help="For grade: seconds to seek into the raw clip before sampling (default: 5)", + ) + parser.add_argument( + "--dur", + type=float, + default=3.0, + dest="grade_dur", + help="For grade: duration in seconds of the preview clip (default: 3)", + ) parser.add_argument( "--ffmpeg-log", type=str, @@ -332,13 +317,6 @@ Examples: args.processed, args.alpha_quality, ) - elif action in ("stitch"): - return cmd_stitch( - project_path, - args.verbose, - args.force, - args.res, - ) elif action == "render": return cmd_render( project_path, @@ -349,8 +327,14 @@ Examples: args.force, chunk_slides=args.chunk_slides, ) - elif action == "transcribe": - return cmd_transcribe(project_path, args.verbose, args.res, args.final) + elif action == "grade": + return cmd_grade( + project_path, + args.verbose, + file=args.file, + ss=args.grade_ss, + dur=args.grade_dur, + ) elif action == "align": return cmd_align(project_path, args.verbose) elif action == "all": @@ -375,12 +359,6 @@ Examples: elif action == "down": from .transfer import cmd_down return cmd_down(project_path, args.verbose, args.dry_run) - elif action == "extract-audio": - return cmd_extract_audio( - project_path, args.verbose, args.segment, args.channel, args.combined - ) - elif action == "master": - return cmd_master(project_path, args.verbose, args.channel) elif action == "push": from .push import cmd_push @@ -1247,7 +1225,6 @@ def _import_videos(videos_dir: Path, config, verbose: bool) -> None: and f.suffix.lower() in video_extensions and "_processed" not in f.stem # Exclude any _processed files and "_fixed" not in f.stem # Exclude any _fixed files - and not f.name.startswith("narration_combined") ] # Also exclude files in subdirectories (proxy/, intermediate/, etc.) @@ -1305,29 +1282,15 @@ def _import_videos(videos_dir: Path, config, verbose: bool) -> None: print(f" Skipping {video_id} (already exists)") continue - # Determine if this is a talking head segment - # Match patterns like: talkinghead, talkingheadS01, talkinghead_s01, etc. - is_narration_combined = "narration_combined" in video_file.stem.lower() # Build the video entry video_entry = { "source_file": video_file.name, + "output_file": video_file.name, + "cutout": "square", + "filter": [], } - - if is_narration_combined: - video_entry["output_file"] = None - video_entry["cutout"] = "talkinghead" - video_entry["always_visible"] = True - video_entry["skip"] = 0 - video_entry["filter"] = [] - print(f" Added talking head segment: {video_id}") - else: - # Regular video - - video_entry["output_file"] = video_file.name - video_entry["cutout"] = "square" - video_entry["filter"] = [] - if verbose: - print(f" Added: {video_id}") + if verbose: + print(f" Added: {video_id}") existing_videos[video_id] = video_entry added_count += 1 @@ -1925,53 +1888,6 @@ def _resolve_process_cache(project_path: Path, config) -> Optional[Path]: return None -def _narration_combined_hint(project_path: Path, config) -> str: - """Return a helpful hint when narration_combined.mov cannot be found. - - If external storage is configured but the volume isn't mounted, the stitch - command wouldn't help — the disk is just not connected. - """ - from .cache import load_cache_config - - missing_paths = [] - - cache_base = load_cache_config() - if cache_base is not None and not cache_base.exists(): - missing_paths.append(cache_base) - - if config and config.process_cache: - pc = Path(config.process_cache) - if not pc.is_absolute(): - pc = (project_path / pc).resolve() - if not pc.exists(): - missing_paths.append(pc) - - if missing_paths: - return ( - f"External disk not connected (expected at {missing_paths[0]}).\n" - "Connect the disk and try again." - ) - return "Run 'gnommo -p stitch' first." - - -def _resolve_narration_combined( - project_path: Path, videos_dir: Path, config -) -> Optional[Path]: - """Find narration_combined.mov: local → GnommoCache → process_cache.""" - local = videos_dir / "narration_combined.mov" - if local.exists(): - return local - resolved, _ = resolve_with_cache(local, project_path) - if resolved.exists(): - return resolved - pc_root = _resolve_process_cache(project_path, config) - if pc_root: - pc_path = pc_root / "media" / "videos" / "narration_combined.mov" - if pc_path.exists(): - return pc_path - return None - - def cmd_new(project_path: Path, verbose: bool) -> int: """Create a new gnommo project with standard folder structure and a project.json template.""" project_name = project_path.name @@ -2143,7 +2059,6 @@ Done. Here is what to do next: gnommo -p {project_name} import # extract slides from Keynote gnommo -p {project_name} pre # chroma key + audio normalise gnommo -p {project_name} trim # auto-detect skip/take per segment - gnommo -p {project_name} stitch # join narration into one file gnommo -p {project_name} render # produce the final video Or run everything in one go: @@ -2714,7 +2629,7 @@ def cmd_preprocess( print(f"\n Updated narration.json ({len(successfully_processed)} segment(s))") print( - f"\n Run 'gnommo -p stitch' to stitch narration segments into one full length narration file." + f"\n Narration segments are concatenated automatically at render time — run 'gnommo -p render'." ) # Also preprocess videos from videos.json (e.g. chroma key, color grade) @@ -3530,213 +3445,6 @@ def cmd_transcode( # ============================================================================= -def cmd_stitch( - project_path: Path, - verbose: bool, - force: bool = False, - res: str = "full", -) -> int: - """ - Stitch narration segments from narration.json. - - Reads segments from media/narration/narration.json, applies begin/end - trimming during concatenation, and writes output to media/videos/narration_combined.mov. - Also creates/updates an entry in videos.json with volume property. - """ - from .parser import parse_project_config, parse_narration, parse_videos - from .preprocessor import ( - stitch_narration_segments, - ensure_downscaled_files_exist, - RES_CONFIGS, - ) - - mode_str = f" ({res.upper()})" if res != "full" else "" - print(f"Stitching narration: {project_path.name}{mode_str}") - - config = parse_project_config(project_path) - narration, narration_dir = parse_narration(project_path, config) - - if not narration: - print(" No narration segments found in media/narration/narration.json") - print(" Run 'gnommo -p import' first to populate narration.json") - return 1 - - # narration.json (skip/take/order) drives stitch — capture its local path for - # fingerprinting before narration_dir is redirected to a cache/res subdir. - _local_narration_json = narration_dir / "narration.json" - - # Get videos_dir for output - if config and config.videos_path: - videos_json_path = project_path / config.videos_path - videos_dir = videos_json_path.parent - else: - videos_dir = project_path / "media" / "videos" - - # When process_cache is set, redirect processed segment reads and combined output. - # Mirror media/ structure so GnommoCache (resolve_with_cache) finds files during render. - cache_root = _resolve_process_cache(project_path, config) - if cache_root: - narration_dir = cache_root / "media" / "narration" - narration_dir.mkdir(parents=True, exist_ok=True) - videos_dir_out = cache_root / "media" / "videos" - videos_dir_out.mkdir(parents=True, exist_ok=True) - print(f" Using process cache: {cache_root}") - else: - videos_dir_out = videos_dir - - # Use downscaled dirs for non-full res - if res != "full": - cfg = RES_CONFIGS[res] - narration_dir = ensure_downscaled_files_exist( - narration_dir, res, force=False, verbose=verbose - ) - videos_dir_out = videos_dir_out / cfg[2] - videos_dir_out.mkdir(parents=True, exist_ok=True) - print(f" Using {res} dirs: {narration_dir}, {videos_dir_out}") - - # Get segment IDs in natural order (Segment2 before Segment10) - segment_ids = sorted(narration.keys(), key=lambda s: [int(t) if t.isdigit() else t.lower() for t in re.split(r'(\d+)', s)]) - - # Show what we're stitching, and — importantly — where each segment's trim - # points came from: the user-friendly begin/end aliases, an explicit - # skip/take (e.g. written by 'trim'), or the project-level defaults. This - # makes the start/end determination visible instead of a bare number. - default_begin = config.default_begin if config else 0.0 - default_end_trim = config.default_end_trim if config else 0.0 - try: - _raw_narr = _read_json(_local_narration_json) if _local_narration_json.exists() else {} - except (OSError, json.JSONDecodeError): - _raw_narr = {} - - print(f"\n Segments ({len(segment_ids)}):") - for segment_id in segment_ids: - seg = narration[segment_id] - entry = _raw_narr.get(segment_id) or _raw_narr.get(segment_id.lower()) or {} - - if entry.get("begin"): - skip_from = f"from begin={entry['begin']}" - elif entry.get("start"): - skip_from = f"from start={entry['start']}" - elif "skip" in entry: - skip_from = "explicit skip" - elif default_begin: - skip_from = f"from default_begin={default_begin:g}s" - else: - skip_from = None - - if entry.get("end"): - take_from = f"from end={entry['end']}" - elif "take" in entry: - take_from = "explicit take" - elif seg.take is not None and default_end_trim: - take_from = f"from default_end_trim={default_end_trim:g}s" - else: - take_from = None - - skip_disp = f"skip={seg.skip:.1f}s" + (f" ({skip_from})" if skip_from else "") - take_disp = f"take={seg.take:.1f}s" if seg.take is not None else "take=to end" - if take_from: - take_disp += f" ({take_from})" - print(f" - {segment_id}: {skip_disp} · {take_disp}") - - stitch_output = videos_dir_out / "narration_combined.mov" - - # Stage-level staleness: skip when narration.json and every processed segment - # are unchanged since the last successful stitch AND the output still exists. - # A changed input (re-trim, reprocessed segment) auto-triggers a regenerate. - from . import state as _state - - _stitch_key = f"stitch:{res}" - _stitch_inputs = _state.compute( - [("narration.json", _local_narration_json, _state.HASH)] - + [ - (f"seg:{sid}", narration_dir / narration[sid].source_file, _state.META) - for sid in segment_ids - ] - ) - _stitch_current = _state.is_current( - project_path, _stitch_key, _stitch_inputs, [stitch_output] - ) - - if stitch_output.exists() and not force and _stitch_current: - print(f"\n Combined narration up to date: {stitch_output.name}") - print(" (inputs unchanged since last stitch — use --force to regenerate)") - else: - if stitch_output.exists() and not force and not _stitch_current: - print("\n Inputs changed since last stitch — regenerating.") - # Extract loudnorm config from talkinghead filter so stitch uses - # per-project settings instead of hardcoded defaults. - _loudnorm_cfg = None - if config and config.default_filters: - for _f in config.default_filters.get("talkinghead") or []: - if isinstance(_f, dict) and _f.get("type") == "audio_normalize": - _loudnorm_cfg = _f - break - stitch_narration_segments( - narration_dir, - segment_ids, - narration, - stitch_output, - verbose=verbose, - default_end_trim=config.default_end_trim if config else 0.0, - loudnorm_config=_loudnorm_cfg, - ) - # Run import videos again to update duration metadata (skip when using cache - # since narration_combined.mov lives on the external disk, not in videos_dir). - if not cache_root: - _import_videos(videos_dir_out, config, verbose) - - # Record fingerprints (recompute output META now that it exists) so the - # next run can detect whether inputs changed. - _state.record( - project_path, - _stitch_key, - _state.compute( - [("narration.json", _local_narration_json, _state.HASH)] - + [ - (f"seg:{sid}", narration_dir / narration[sid].source_file, _state.META) - for sid in segment_ids - ] - ), - ) - - # Always update the MAIN videos.json (parent of subdir when using low/tiny res) - # Downscaled dirs only affect file paths, not JSON metadata updates - main_videos_dir = ( - videos_dir_out.parent if (res != "full" and not cache_root) else videos_dir - ) - videos_json_path = main_videos_dir / "videos.json" - if True: # Always update JSON regardless of proxy mode - existing_videos: dict = {} - if videos_json_path.exists(): - existing_videos = _read_json(videos_json_path) - - # Get cutout from first narration segment - first_seg = narration[segment_ids[0]] - cutout = first_seg.cutout or "talkinghead" - - # Create/update narration_combined entry - existing_videos["narration_combined"] = { - "source_file": "narration_combined.mov", - "cutout": cutout, - "always_visible": True, - "volume": 1.0, - } - - with open(videos_json_path, "w", encoding="utf-8") as f: - json.dump(existing_videos, f, indent=2) - print(f"\n Updated videos.json with narration_combined entry (volume=1.0)") - print(" Edit videos.json to adjust volume if needed.") - - print("\nConcatenation complete.") - - # Automatically transcribe to keep transcript in sync with narration - print("\n" + "=" * 60) - print("Auto-running transcribe to sync with new narration...") - print("=" * 60 + "\n") - return cmd_transcribe(project_path, verbose, res=res) - - # ============================================================================= # Render Command # ============================================================================= @@ -4340,9 +4048,8 @@ def cmd_render( audio, audio_dir = parse_audio(project_path, config) # --- Narration: render-time concat of the processed segments --- - # narration.json is the single source of truth. The processed segments are - # concatenated directly in the render graph — there is no narration_combined - # file anymore. + # narration.json is the single source of truth; the processed segments are + # concatenated directly in the render graph. from .narration import build_narration_schedule from .parser import parse_narration as _parse_narr, get_video_duration @@ -4376,8 +4083,16 @@ def cmd_render( if config.transcript_path and (project_path / config.transcript_path).exists(): transcript_path = project_path / config.transcript_path elif narration_map: - # Legacy on-disk transcript that used to accompany narration_combined. - transcript_path = videos_dir / "narration_combined.transcript.json" + # Narration project with no per-segment transcripts to merge. + print( + "Error: No per-segment transcripts found for the narration segments.", + file=sys.stderr, + ) + print( + f"Run 'gnommo -p {project_path.name} trim' first (it transcribes each segment).", + file=sys.stderr, + ) + return 1 else: result = _find_narration_video(config, videos) if result: @@ -4389,7 +4104,7 @@ def cmd_render( transcript_path, _ = resolve_with_cache(transcript_path, project_path) if not transcript_path.exists(): print(f"Error: Transcription not found: {transcript_path}", file=sys.stderr) - print(f"Run 'gnommo -p {project_path.name} transcribe' first.", file=sys.stderr) + print(f"Run 'gnommo -p {project_path.name} trim' first (it produces per-segment transcripts).", file=sys.stderr) return 1 transcription = load_transcript(transcript_path, project_path) @@ -4646,139 +4361,99 @@ def _find_narration_video(config, videos: dict) -> Optional[tuple[str, "VideoSou return None -def cmd_transcribe( - project_path: Path, verbose: bool, res: str = "full", final: bool = False +# ============================================================================= +# Grade Command +# ============================================================================= + + +def cmd_grade( + project_path: Path, + verbose: bool, + file: Optional[str] = None, + ss: Optional[float] = None, + dur: float = 3.0, ) -> int: - """Transcribe video audio using Whisper.""" - from .transcriber import transcribe_video, save_transcript, words_to_srt - from .parser import parse_project_config, parse_videos - from .preprocessor import ensure_downscaled_files_exist + """Sample a few seconds of a raw narration clip through the talkinghead + filter chain so you can iterate on gnommokey / color_grade settings without + running a full preprocess. + + Writes two files to the project root: + grade_preview.mov — the exact keyed ProRes 4444 output (alpha over black) + grade_preview.mp4 — the same result flattened over mid-gray, easy to view + in any player (best for judging spill and skin tone) + """ + from .parser import parse_project_config + from .preprocessor import _process_chunk_to_prores4444, get_video_duration config = parse_project_config(project_path) - # Handle --final mode: transcribe the rendered output for YouTube captions - if final: - path = project_path / "out" / f"{config.output_video}.mp4" - return _transcribe_final(path, verbose) - - mode_str = f" ({res.upper()})" if res != "full" else "" - print(f"Transcribing: {project_path.name}{mode_str}") - - videos, videos_dir = parse_videos(project_path, config) - if not videos: - print("Error: No videos defined in videos.json", file=sys.stderr) - return 1 - - # Non-full res: use downscaled video directory - if res != "full": - videos_dir = ensure_downscaled_files_exist( - videos_dir, res, force=False, verbose=verbose + talkinghead_filter = (config.default_filters or {}).get("talkinghead", []) + if not talkinghead_filter: + print( + " ERROR: No 'talkinghead' filter defined in project.json default_filters." ) + return 1 - # Check for multi-segment narration (concatenated file) - if isinstance(config.main_video, list) and len(config.main_video) > 1: - video_path = videos_dir / "narration_combined.mov" - if not video_path.exists(): - print(f"Error: Combined narration not found: {video_path}", file=sys.stderr) - print( - "Run 'gnommo -p pre' first to concatenate segments.", - file=sys.stderr, + # --- Resolve source clip --- + raw_dir = project_path / "media" / "narration" / "raw_mov" + _video_exts = {".mov", ".mp4", ".avi", ".mkv", ".m4v"} + + if file: + source = Path(file) + if not source.is_absolute(): + source = project_path / file + if not source.exists(): + alt = raw_dir / Path(file).name + if alt.exists(): + source = alt + if not source.exists(): + print(f" ERROR: source file not found: {file}") + return 1 + else: + candidates = ( + sorted( + f + for f in raw_dir.iterdir() + if f.is_file() + and f.suffix.lower() in _video_exts + and not f.name.startswith(".") ) + if raw_dir.exists() + else [] + ) + if not candidates: + print(f" ERROR: no raw clips found in {raw_dir}") + print(" Pass --file to point at a specific clip.") return 1 - print(f" Using combined narration: {video_path.name}") - else: - # Single video - find it using existing logic - result = _find_narration_video(config, videos) - if not result: - print("Error: No suitable video found for transcription", file=sys.stderr) - return 1 + source = candidates[0] - video_id, video_source = result - video_path = videos_dir / video_source.source_file + # --- Resolve seek / duration, clamped to the clip length --- + clip_len = get_video_duration(source) + if ss is None: + # Default: 5s in, or centred if the clip is short. + ss = 5.0 if clip_len > 8 else max(0.0, clip_len / 2 - dur / 2) + if ss >= clip_len: + ss = max(0.0, clip_len - dur) + take = min(dur, max(0.1, clip_len - ss)) - if ( - not video_path.exists() - and video_source.source_file == "narration_combined.mov" - ): - found = _resolve_narration_combined(project_path, videos_dir, config) - if found: - video_path = found - if not video_path.exists(): - video_path, _ = resolve_with_cache(video_path, project_path) - if not video_path.exists(): - print(f"Error: Video not found: {video_path}", file=sys.stderr) - return 1 + print(f"Grading preview: {project_path.name}") + print(f" Source: {source}") + print(f" Sample: {take:.1f}s starting at {ss:.1f}s (clip is {clip_len:.1f}s)") + print(f" Filters: {len(talkinghead_filter)} step(s)") - print(f" Video: {video_path.name}") + mov_out = project_path / "grade_preview.mov" + _process_chunk_to_prores4444( + source, + mov_out, + talkinghead_filter, + start_time=ss, + chunk_duration=take, + verbose=verbose, + take=take, + keep_audio=False, + ) - words = transcribe_video(video_path, model="base") - - # Save to project-local path if configured in project.json (keeps transcript off external drives) - if config.transcript_path: - output_path = project_path / config.transcript_path - output_path.parent.mkdir(parents=True, exist_ok=True) - else: - output_path = video_path.with_suffix(".transcript.json") - save_transcript(words, output_path) - - print(f" - Transcribed {len(words)} words") - print(f" - Duration: {words[-1].end:.1f}s" if words else " - No words found") - print(f" - Saved: {output_path}") - - if verbose and words: - preview = " ".join(w.word for w in words[:10]) - print(f" - Preview: {preview}...") - - return 0 - - -def _transcribe_final(final_video: Path, verbose: bool) -> int: - """ - Transcribe the final rendered video and generate SRT captions for YouTube. - - Looks and creates out filename.srt suitable for upload. - """ - from .transcriber import transcribe_video, save_transcript, words_to_srt - - print(f"Transcribing final output: {final_video}") - - if not final_video.exists(): - print(f"Error: Final video not found: {final_video}", file=sys.stderr) - print("Run 'gnommo render' first.", file=sys.stderr) - return 1 - - print(f" Video: {final_video.name}") - - # Transcribe with word-level timestamps - words = transcribe_video(final_video, model="base") - - if not words: - print("Error: No words transcribed from video", file=sys.stderr) - return 1 - - # Save JSON transcript - transcript_path = final_video.with_suffix(".transcript.json") - save_transcript(words, transcript_path) - - # Generate SRT captions - srt_path = final_video.with_suffix(".srt") - srt_content = words_to_srt(words) - srt_path.write_text(srt_content, encoding="utf-8") - - print(f" - Transcribed {len(words)} words") - print(f" - Duration: {words[-1].end:.1f}s") - print(f" - Transcript: {transcript_path}") - print(f" - Captions: {srt_path}") - - # Count caption segments - caption_count = srt_content.count("\n\n") + 1 - print(f" - Caption segments: {caption_count}") - - if verbose and words: - preview = " ".join(w.word for w in words[:15]) - print(f" - Preview: {preview}...") - - print("\nSRT file ready for YouTube upload.") + print(f"\n Done. Keyed preview (ProRes 4444, alpha): {mov_out}") return 0 @@ -4831,7 +4506,7 @@ def cmd_align(project_path: Path, verbose: bool) -> int: transcript_path, _ = resolve_with_cache(transcript_path, project_path) if not transcript_path.exists(): print(f"Error: Transcription not found: {transcript_path}", file=sys.stderr) - print(f"Run 'gnommo -p {project_path.name} transcribe' first.", file=sys.stderr) + print(f"Run 'gnommo -p {project_path.name} trim' first (it produces per-segment transcripts).", file=sys.stderr) return 1 print(f" Loading: {transcript_path.name}") @@ -4943,7 +4618,7 @@ def cmd_all( res: str = "full", force: bool = False, ) -> int: - """Run full pipeline: import → prune → preprocess → trim → stitch → render → push → handoff → up. + """Run full pipeline: import → prune → preprocess → trim → render → push → handoff → up. Cascade rule: if any stage produces output, all subsequent stages are forced to re-run (cascade_force=True), regardless of whether --force was passed. @@ -4958,7 +4633,7 @@ def cmd_all( # True so all downstream stages re-run unconditionally. cascade_force = force - print(">>> Step 1/9: Import\n") + print(">>> Step 1/8: Import\n") t0 = time.time() result = cmd_import(project_path, cascade_force, verbose) if result != 0: @@ -4968,14 +4643,14 @@ def cmd_all( ): cascade_force = True - print("\n>>> Step 2/9: Prune\n") + print("\n>>> Step 2/8: Prune\n") # Drop manifest entries left over from edits (e.g. a video split into two # projects). Only removes unused entries, so it never forces downstream re-runs. result = cmd_prune(project_path, verbose, dry_run) if result != 0: return result - print("\n>>> Step 3/9: Preprocess\n") + print("\n>>> Step 3/8: Preprocess\n") t0 = time.time() result = cmd_preprocess( project_path, verbose, dry_run, cascade_force, workers=1, res=res @@ -4987,7 +4662,7 @@ def cmd_all( ) or _files_modified_since(project_path, t0, "*_processed.webm"): cascade_force = True - print("\n>>> Step 4/9: Trim\n") + print("\n>>> Step 4/8: Trim\n") # Skip the (Whisper-heavy) trim stage when nothing upstream changed # (cache intact) and every segment is already resolved — i.e. it has a # cached transcript or explicit skip/take. A cascade_force from preprocess @@ -5003,30 +4678,22 @@ def cmd_all( if _files_modified_since(project_path, t0, "narration.json"): cascade_force = True - print("\n>>> Step 5/9: Stitch\n") - t0 = time.time() - result = cmd_stitch(project_path, verbose, cascade_force, res=res) - if result != 0: - return result - if _files_modified_since(project_path, t0, "narration_combined.mov"): - cascade_force = True - - print("\n>>> Step 6/9: Render\n") + print("\n>>> Step 5/8: Render\n") result = cmd_render(project_path, verbose, dry_run, res=res, force=cascade_force) if result != 0: return result - print("\n>>> Step 7/9: Push\n") + print("\n>>> Step 6/8: Push\n") result = cmd_push(project_path, verbose, force=False, prod=True) if result != 0: return result - print("\n>>> Step 8/9: Handoff\n") + print("\n>>> Step 7/8: Handoff\n") result = cmd_handoff(project_path, verbose, file_override=None, prod=True, res=res) if result != 0: return result - print("\n>>> Step 9/9: Upload\n") + print("\n>>> Step 8/8: Upload\n") from .transfer import cmd_up return cmd_up(project_path, verbose, dry_run) @@ -5079,7 +4746,7 @@ def cmd_description(project_path: Path, verbose: bool) -> int: else: print(f" Warning: No transcription found at {transcript_path}") print( - f" Run 'gnommo -p {project_path.name} transcribe' for better timestamps." + f" Run 'gnommo -p {project_path.name} trim' for better timestamps." ) # Align markers to get timings @@ -5153,7 +4820,6 @@ _RSYNC_EXCLUDES = [ "media/videos/intermediate/**", "media/narration/processed/", "media/narration/processed/**", - "media/videos/narration_combined.mov", # Low-res preview files (generated locally, not synced) "media/narration/low/", "media/narration/low/**", @@ -5333,523 +4999,5 @@ def cmd_load(project_path: Path, verbose: bool, dry_run: bool) -> int: -# ============================================================================= -# Extract Audio Command -# ============================================================================= - - -def _extract_audio_file( - source_path: Path, - output_dir: Path, - name: str, - channel: str, - verbose: bool, -) -> int: - """ - Extract audio from a single video file to WAV. - - Args: - source_path: Path to the source video file - output_dir: Directory to save the WAV file - name: Base name for the output file (without extension) - channel: "left", "right", or "both" - verbose: Print verbose output - - Returns: - 0 on success, 1 on error - """ - # Build output filename - if channel == "both": - output_name = f"{name}.wav" - else: - output_name = f"{name}_{channel}.wav" - output_path = output_dir / output_name - - print(f" Channel: {channel}") - print(f" Source: {source_path}") - print(f" Output: {output_path}") - - # Build ffmpeg command - cmd = [ - "ffmpeg", - "-y", # Overwrite - "-i", - str(source_path), - "-vn", # No video - ] - - # Channel selection - if channel == "left": - cmd.extend(["-af", "pan=mono|c0=c0"]) - elif channel == "right": - cmd.extend(["-af", "pan=mono|c0=c1"]) - # "both" keeps stereo, no filter needed - - # Output format: 48kHz 16-bit WAV (standard for audio editing) - cmd.extend( - [ - "-ar", - "48000", # 48kHz sample rate - "-acodec", - "pcm_s16le", # 16-bit PCM - str(output_path), - ] - ) - - if verbose: - print(f" Command: {' '.join(cmd)}") - - print(f" Extracting...", end=" ", flush=True) - result = subprocess.run(cmd, capture_output=True, text=True) - if result.returncode != 0: - print(f"Error!") - print(f" {result.stderr}", file=sys.stderr) - return 1 - - # Get duration info - duration_cmd = [ - "ffprobe", - "-v", - "error", - "-show_entries", - "format=duration", - "-of", - "default=noprint_wrappers=1:nokey=1", - str(output_path), - ] - duration_result = subprocess.run(duration_cmd, capture_output=True, text=True) - duration_str = "" - if duration_result.returncode == 0: - try: - duration = float(duration_result.stdout.strip()) - duration_str = f" ({duration:.1f}s)" - except ValueError: - pass - - print(f"Done{duration_str}") - - print(f"\n Open in Audition to experiment with:") - print(f" - Effect > Noise Reduction") - print(f" - Effect > Compressor") - print(f" - Effect > Filter Curve EQ") - print(f" - Effect > Loudness Normalization") - print( - f"\n Once you find good settings, update narration.json with matching filter config." - ) - - return 0 - - -def cmd_extract_audio( - project_path: Path, - verbose: bool, - segment: Optional[str] = None, - channel: str = "both", - combined: bool = False, -) -> int: - """ - Extract audio from narration segments to WAV files for editing in Audacity. - - This allows you to experiment with audio processing settings (EQ, compression, - noise reduction) in external software before applying them in the pipeline. - - Args: - project_path: Path to the project directory - verbose: Enable verbose output - segment: Specific segment ID to extract, or None for all segments - channel: Which channel(s) to extract: "left", "right", or "both" - combined: If True, extract from narration_combined.mov instead of segments - """ - from .parser import parse_project_config, parse_narration, parse_videos - - print(f"Extracting audio: {project_path.name}") - - config = parse_project_config(project_path) - - # Handle --combined mode: extract from narration_combined.mov - if combined: - videos, videos_dir = parse_videos(project_path, config) - combined_path = _resolve_narration_combined( - project_path, videos_dir, config - ) or (videos_dir / "narration_combined.mov") - - if not combined_path.exists(): - print( - f"Error: narration_combined.mov not found at {combined_path}", - file=sys.stderr, - ) - print(_narration_combined_hint(project_path, config), file=sys.stderr) - return 1 - - # Output to project out/ directory - audio_dir = project_path / "out" - audio_dir.mkdir(parents=True, exist_ok=True) - - return _extract_audio_file( - combined_path, audio_dir, "narration_combined", channel, verbose - ) - - # Normal mode: extract from individual segments - narration, narration_dir = parse_narration(project_path, config) - - if not narration: - print(" No narration segments found in media/narration/narration.json") - print(" Run 'gnommo -p import' first to populate narration.json") - return 1 - - # Create output directory - audio_dir = narration_dir / "audio" - audio_dir.mkdir(parents=True, exist_ok=True) - - # Determine which segments to process - if segment: - if segment not in narration: - print( - f"Error: Segment '{segment}' not found in narration.json", - file=sys.stderr, - ) - print( - f"Available segments: {', '.join(sorted(narration.keys()))}", - file=sys.stderr, - ) - return 1 - segments_to_process = [(segment, narration[segment])] - else: - segments_to_process = sorted(narration.items()) - - print(f" Channel: {channel}") - print(f" Output: {audio_dir}/") - print(f" Segments: {len(segments_to_process)}") - - # Process each segment - for segment_id, segment_source in segments_to_process: - source_path = narration_dir / segment_source.source_file - if not source_path.exists(): - print(f" Warning: Source not found: {source_path.name}, skipping") - continue - - # Build output filename - if channel == "both": - output_name = f"{segment_id}.wav" - else: - output_name = f"{segment_id}_{channel}.wav" - output_path = audio_dir / output_name - - print(f"\n {segment_id}:") - print(f" Source: {source_path.name}") - print(f" Output: {output_name}") - - # Build ffmpeg command - cmd = [ - "ffmpeg", - "-y", # Overwrite - "-i", - str(source_path), - "-vn", # No video - ] - - # Channel selection - if channel == "left": - cmd.extend(["-af", "pan=mono|c0=c0"]) - elif channel == "right": - cmd.extend(["-af", "pan=mono|c0=c1"]) - # "both" keeps stereo, no filter needed - - # Output format: 48kHz 16-bit WAV (standard for audio editing) - cmd.extend( - [ - "-ar", - "48000", # 48kHz sample rate - "-acodec", - "pcm_s16le", # 16-bit PCM - str(output_path), - ] - ) - - if verbose: - print(f" Command: {' '.join(cmd)}") - - result = subprocess.run(cmd, capture_output=True, text=True) - if result.returncode != 0: - print(f" Error: {result.stderr}", file=sys.stderr) - return 1 - - # Get duration info - duration_cmd = [ - "ffprobe", - "-v", - "error", - "-show_entries", - "format=duration", - "-of", - "default=noprint_wrappers=1:nokey=1", - str(output_path), - ] - duration_result = subprocess.run(duration_cmd, capture_output=True, text=True) - if duration_result.returncode == 0: - try: - duration = float(duration_result.stdout.strip()) - print(f" Duration: {duration:.1f}s") - except ValueError: - pass - - print(f" Done") - - print(f"\n Audio files saved to: {audio_dir}") - print(f"\n Open in Audacity to experiment with:") - print(f" - Effect > Noise Reduction") - print(f" - Effect > Compressor") - print(f" - Effect > Filter Curve EQ") - print(f" - Effect > Loudness Normalization") - print( - f"\n Once you find good settings, update narration.json with matching filter config." - ) - - return 0 - - -# ============================================================================= -# Master Command (A/B audio comparison) -# ============================================================================= - - -def cmd_master( - project_path: Path, - verbose: bool, - channel: str = "both", -) -> int: - """ - Extract raw and processed audio from narration_combined for A/B comparison. - - Outputs: - out/narration_combined.wav - Raw audio (no processing) - out/narration_combined_processed.wav - With audio filters applied - - This lets you compare the effect of your audio processing chain. - """ - from .parser import parse_project_config, parse_videos - from .preprocessor import parse_audio_normalize_config - - print(f"Audio mastering: {project_path.name}") - - config = parse_project_config(project_path) - videos, videos_dir = parse_videos(project_path, config) - - # Find narration_combined.mov - combined_path = _resolve_narration_combined(project_path, videos_dir, config) or ( - videos_dir / "narration_combined.mov" - ) - if not combined_path.exists(): - print( - f"Error: narration_combined.mov not found at {combined_path}", - file=sys.stderr, - ) - print(_narration_combined_hint(project_path, config), file=sys.stderr) - return 1 - - # Output directory - out_dir = project_path / "out" - out_dir.mkdir(parents=True, exist_ok=True) - - raw_output = out_dir / "narration_combined.wav" - processed_output = out_dir / "narration_combined_processed.wav" - - # Find audio_normalize config from default_filters - audio_config = None - if config.default_filters: - for preset_name, filters in config.default_filters.items(): - for f in filters: - if f.get("type") == "audio_normalize": - audio_config = f - print(f" Using audio config from: default_filters.{preset_name}") - break - if audio_config: - break - - if not audio_config: - print(" Warning: No audio_normalize filter found in default_filters") - print(" Will only extract raw audio.") - - # Build channel filter - channel_filter = "" - if channel == "left": - channel_filter = "pan=mono|c0=c0," - elif channel == "right": - channel_filter = "pan=mono|c0=c1," - - # Step 1: Extract raw audio - print(f"\n Extracting raw audio...") - raw_cmd = [ - "ffmpeg", - "-y", - "-i", - str(combined_path), - "-vn", - ] - if channel_filter: - raw_cmd.extend(["-af", channel_filter.rstrip(",")]) - raw_cmd.extend( - [ - "-ar", - "48000", - "-acodec", - "pcm_s16le", - str(raw_output), - ] - ) - - if verbose: - print(f" Command: {' '.join(raw_cmd)}") - - result = subprocess.run(raw_cmd, capture_output=True, text=True) - if result.returncode != 0: - print(f" Error extracting raw audio: {result.stderr}", file=sys.stderr) - return 1 - print(f" Saved: {raw_output.name}") - - # Step 2: Extract processed audio (if we have config) - if audio_config: - print(f"\n Applying audio filters...") - cfg = parse_audio_normalize_config(audio_config) - - # Build filter chain (same order as apply_audio_normalize) - audio_filters = [] - - # Channel mapping - if channel_filter: - audio_filters.append(channel_filter.rstrip(",")) - - # EQ bands - for band in cfg.eq_bands: - if band.type == "lowshelf": - audio_filters.append( - f"lowshelf=f={band.freq:.1f}:g={band.gain:.1f}:t=q:w={band.q:.2f}" - ) - elif band.type == "highshelf": - audio_filters.append( - f"highshelf=f={band.freq:.1f}:g={band.gain:.1f}:t=q:w={band.q:.2f}" - ) - else: - audio_filters.append( - f"equalizer=f={band.freq:.1f}:width_type=q:width={band.q:.2f}:g={band.gain:.1f}" - ) - - # High-pass - if cfg.highpass > 0: - audio_filters.append(f"highpass=f={cfg.highpass:.1f}") - - # Low-pass - if cfg.lowpass > 0: - audio_filters.append(f"lowpass=f={cfg.lowpass:.1f}") - - # Room EQ - if cfg.room_eq: - audio_filters.append( - f"equalizer=f={cfg.room_eq_freq:.1f}:width_type=q:width={cfg.room_eq_width:.2f}:g={cfg.room_eq_gain:.1f}" - ) - - # Denoise - if cfg.denoise: - audio_filters.append(f"afftdn=nf={cfg.noise_floor:.1f}") - - # Gate - if cfg.gate: - audio_filters.append( - f"agate=threshold={cfg.gate_threshold:.1f}dB" - f":range={cfg.gate_range:.1f}dB" - f":attack={cfg.gate_attack:.1f}" - f":release={cfg.gate_release:.1f}" - ) - - # Compressor - if cfg.compress: - audio_filters.append( - f"acompressor=threshold={cfg.threshold:.1f}dB" - f":ratio={cfg.ratio:.1f}" - f":attack={cfg.attack:.1f}" - f":release={cfg.release:.1f}" - f":makeup={cfg.makeup:.1f}dB" - ) - - # Loudness normalization - if cfg.normalize: - audio_filters.append( - f"loudnorm=I={cfg.target_lufs:.1f}" - f":LRA={cfg.target_lra:.1f}" - f":TP={cfg.target_tp:.1f}" - ) - - filter_chain = ",".join(audio_filters) - - if verbose: - print(f" Filter chain: {filter_chain}") - - # Print filter summary - print(f" Filters applied:") - if cfg.eq_bands: - print(f" - EQ: {len(cfg.eq_bands)} bands") - if cfg.highpass > 0: - print(f" - Highpass: {cfg.highpass}Hz") - if cfg.denoise: - print(f" - Denoise: floor={cfg.noise_floor}dB") - if cfg.gate: - print(f" - Gate: threshold={cfg.gate_threshold}dB") - if cfg.compress: - print(f" - Compressor: ratio={cfg.ratio}:1, attack={cfg.attack}ms") - if cfg.normalize: - print(f" - Loudnorm: target={cfg.target_lufs} LUFS") - - processed_cmd = [ - "ffmpeg", - "-y", - "-i", - str(combined_path), - "-vn", - "-af", - filter_chain, - "-ar", - "48000", - "-acodec", - "pcm_s16le", - str(processed_output), - ] - - if verbose: - print(f" Command: {' '.join(processed_cmd)}") - - result = subprocess.run(processed_cmd, capture_output=True, text=True) - if result.returncode != 0: - print(f" Error applying filters: {result.stderr}", file=sys.stderr) - return 1 - print(f" Saved: {processed_output.name}") - - # Get durations - def get_duration(path): - cmd = [ - "ffprobe", - "-v", - "error", - "-show_entries", - "format=duration", - "-of", - "default=noprint_wrappers=1:nokey=1", - str(path), - ] - r = subprocess.run(cmd, capture_output=True, text=True) - try: - return float(r.stdout.strip()) - except: - return 0 - - duration = get_duration(raw_output) - - print(f"\n Output files ({duration:.1f}s):") - print(f" {raw_output}") - print(f" {processed_output}") - print(f"\n Open both in Audition to A/B compare the processing.") - - return 0 - - if __name__ == "__main__": sys.exit(main()) diff --git a/gnommo/models.py b/gnommo/models.py index 34f39ed..5f11281 100644 --- a/gnommo/models.py +++ b/gnommo/models.py @@ -132,6 +132,13 @@ class GnommoKeyConfig: # How aggressively to apply despill (0-1) despill_strength: float = 0.5 + # Interior green-limiter (0.0-1.0, 0 = off). Suppresses green cast/spill + # across the whole frame even where green is NOT the dominant channel — the + # case the bias/edge despill misses (e.g. green bounce on skin/a bald head). + # Caps green at a reference blended between max(r,b) [0.0] and the r/b + # average [1.0]. 0.5-0.7 removes cast without pushing skin magenta. + spill_suppress: float = 0.0 + # Alpha bias: influences edge treatment (RGB) # Can help with edge color contamination alpha_bias: tuple[int, int, int] = None @@ -528,7 +535,7 @@ class RenderPlan: ) # Gaps in narration for interstitial videos # Render-time narration concat: ordered segments (skip/take + offset) to # concatenate directly at render time instead of using a single pre-stitched - # narration_combined input. Typed loosely (list of narration.NarrationSegment) + # single pre-stitched narration input. Typed loosely (list of narration.NarrationSegment) # to avoid a circular import between models and narration. narration_segments: list = field(default_factory=list) # Outro sequence (plays after narration ends) diff --git a/gnommo/narration.py b/gnommo/narration.py index 6f44fb4..f877716 100644 --- a/gnommo/narration.py +++ b/gnommo/narration.py @@ -1,6 +1,6 @@ """Deterministic narration scheduling for render-time segment stitching. -Instead of pre-stitching segments into narration_combined.mov, the render stage +Rather than pre-stitching segments into one file, the render stage concatenates the processed segments directly. From narration.json + the cached per-segment transcripts this module computes two things: @@ -8,7 +8,7 @@ per-segment transcripts this module computes two things: combined timeline) — this drives the ffmpeg concat at render time; and 2. the merged word-level transcript, with every word re-timed into the combined timeline — this drives slide alignment, exactly what - re-transcribing narration_combined.mov used to produce, but derived + re-transcribing a pre-stitched narration file used to produce, but derived deterministically (no re-transcription, no separate combined file). The processed files share framerate and format and are uncompressed, so the diff --git a/gnommo/preprocessor.py b/gnommo/preprocessor.py index 0e2e818..53888e3 100644 --- a/gnommo/preprocessor.py +++ b/gnommo/preprocessor.py @@ -1206,6 +1206,22 @@ def build_gnommokey_filter(config: dict) -> str: parts.append(f"geq=r='{new_r}':g='{new_g}':b='{new_b}':a='alpha(X,Y)'") + # Interior spill suppression: cap the spill channel across the WHOLE frame, + # including fully-opaque interior pixels the bias/edge despill can't reach + # (green bounce on skin, a bald head, etc.). Caps the channel at a reference + # blended between max(other two) [t=0] and their average [t=1]; green is only + # ever reduced, never boosted, so non-spilled pixels are untouched. + if cfg.spill_suppress > 0: + t = min(max(cfg.spill_suppress, 0.0), 1.0) + if is_green_screen: + ref = f"((1-{t:.3f})*max(r(X,Y),b(X,Y))+{t:.3f}*(r(X,Y)+b(X,Y))/2)" + new_g = f"min(g(X,Y),{ref})" + parts.append(f"geq=r='r(X,Y)':g='{new_g}':b='b(X,Y)':a='alpha(X,Y)'") + else: + ref = f"((1-{t:.3f})*max(r(X,Y),g(X,Y))+{t:.3f}*(r(X,Y)+g(X,Y))/2)" + new_b = f"min(b(X,Y),{ref})" + parts.append(f"geq=r='r(X,Y)':g='g(X,Y)':b='{new_b}':a='alpha(X,Y)'") + # Edge-aware despill: aggressively suppress green at semi-transparent edges # This targets the 2-4px green fringe that regular despill misses # edge_factor is high (1.0) at alpha=128, low (0) at alpha=0 or 255 @@ -1292,6 +1308,7 @@ def parse_gnommokey_config(config: dict) -> GnommoKeyConfig: clip_white=float(config.get("clip_white", 100.0)), despill_bias=despill_bias, despill_strength=float(config.get("despill_strength", 0.5)), + spill_suppress=float(config.get("spill_suppress", 0.0)), alpha_bias=alpha_bias, protect_luma=int(config.get("protect_luma", -1)), shadow_boost=float(config.get("shadow_boost", 0.0)), @@ -2314,206 +2331,3 @@ def needs_preprocessing(videos_dir: Path, video_source: VideoSource) -> bool: return True return True - - -def _build_loudnorm_filter(loudnorm_config: Optional[dict]) -> str: - _cfg = loudnorm_config or {} - _lufs = float(_cfg.get("target_lufs", -14)) - _lra = float(_cfg.get("target_lra", 11)) - _tp = float(_cfg.get("target_tp", -1.5)) - return f"loudnorm=I={_lufs:.1f}:LRA={_lra:.1f}:TP={_tp:.1f}" - - -def _build_reencode_args(source_path: Path) -> tuple[list[str], list[str]]: - """ - Return (video_args, audio_args) for re-encoding an inter-frame source to a - normalized intra-frame format. Does not include -avoid_negative_ts or the - output path — callers add those. - """ - has_alpha = _video_has_alpha(source_path) - if has_alpha: - video_args = [ - "-vf", "fps=30,format=yuva444p10le", - "-c:v", "prores_ks", "-profile:v", "4", "-pix_fmt", "yuva444p10le", - ] - audio_args = ["-c:a", "pcm_s16le"] - else: - video_args = [ - "-vf", "fps=30", - "-c:v", "libx264", "-preset", "fast", "-crf", "18", - "-movflags", "+faststart", - ] - audio_args = ["-c:a", "aac", "-b:a", "192k"] - return video_args, audio_args - - -def stitch_narration_segments( - videos_dir: Path, - segment_ids: list[str], - videos: dict[str, VideoSource], - output_path: Path, - verbose: bool = False, - default_end_trim: float = 0.0, - loudnorm_config: Optional[dict] = None, -) -> Path: - """ - Stitch multiple narration video segments into a single file. - - Each segment's skip and take values are applied to trim dead video at the - start/end of each recording. The segments are concatenated in the order - specified by segment_ids. - - Args: - videos_dir: Directory containing video files - segment_ids: Ordered list of video IDs from videos.json - videos: Dict of video ID -> VideoSource from videos.json - output_path: Path for the concatenated output file - verbose: Enable verbose output - default_end_trim: Seconds to trim from the end when no explicit end/take is set - - Returns: - Path to the stitched video file. - """ - print(f" Concatenating {len(segment_ids)} narration segment(s)...") - - needs_loudnorm = any( - videos[seg_id].defer_loudnorm for seg_id in segment_ids if seg_id in videos - ) - loudnorm_filter = _build_loudnorm_filter(loudnorm_config) if needs_loudnorm else None - - # ------------------------------------------------------------------ # - # Gather per-segment metadata # - # ------------------------------------------------------------------ # - segments: list[tuple[Path, float, Optional[float], float, str, bool]] = [] - for video_id in segment_ids: - if video_id not in videos: - raise PreprocessError( - f"Narration segment '{video_id}' not found in videos.json", - filter_type=None, - ) - video_source = videos[video_id] - source_path = get_preprocessed_path(videos_dir, video_source) - if not source_path.exists(): - raise PreprocessError( - f"Narration segment not found: {source_path}", - filter_type=None, - ) - full_duration = get_video_duration(source_path) - skip = video_source.skip or 0.0 - take = video_source.take - if take is None and default_end_trim > 0: - take = max(0.0, full_duration - skip - default_end_trim) - effective_duration = min(take, full_duration - skip) if take is not None else full_duration - skip - codec = _get_video_codec(source_path) - is_intra = codec in _INTRA_ONLY_CODECS - if verbose: - mode = "stream copy (fast)" if is_intra else "re-encode" - print(f" {video_id}: {source_path.name} [{codec or '?'}] skip={skip}s take={take or 'all'}s → {effective_duration:.1f}s ({mode})") - segments.append((source_path, skip, take, effective_duration, codec, is_intra)) - - # ------------------------------------------------------------------ # - # Single-segment fast path — no temp file, no concat round-trip # - # ------------------------------------------------------------------ # - if len(segments) == 1: - source_path, skip, take, effective_duration, codec, is_intra = segments[0] - print(f" Trimming → {output_path.name}" + (" + loudnorm" if needs_loudnorm else "") + "...") - cmd = ["ffmpeg", "-y"] - if skip > 0: - cmd.extend(["-ss", str(skip)]) - cmd.extend(["-i", str(source_path)]) - if take is not None: - cmd.extend(["-t", str(take)]) - if is_intra: - cmd.extend(["-c:v", "copy"]) - audio_args = ["-c:a", "copy"] - else: - video_args, audio_args = _build_reencode_args(source_path) - cmd.extend(video_args) - if needs_loudnorm: - cmd.extend(["-af", loudnorm_filter, "-c:a", "aac", "-b:a", "192k"]) - else: - cmd.extend(audio_args) - cmd.extend(["-avoid_negative_ts", "make_zero", str(output_path)]) - result = subprocess.run(cmd, capture_output=True, text=True) - if result.returncode != 0: - raise PreprocessError( - f"Failed to trim segment {segment_ids[0]}", - filter_type="concat", - command=" ".join(cmd), - stderr=result.stderr, - ) - total_duration = get_video_duration(output_path) - print(f" Stitched duration: {format_time(total_duration)}") - return output_path - - # ------------------------------------------------------------------ # - # Multi-segment path — trim each to temp, then concat + loudnorm # - # ------------------------------------------------------------------ # - temp_dir = output_path.parent / "concat_temp" - temp_dir.mkdir(parents=True, exist_ok=True) - trimmed_segments: list[Path] = [] - - for i, (source_path, skip, take, effective_duration, codec, is_intra) in enumerate(segments): - trimmed_path = temp_dir / f"segment_{i:03d}.mov" - cmd = ["ffmpeg", "-y"] - if skip > 0: - cmd.extend(["-ss", str(skip)]) - cmd.extend(["-i", str(source_path)]) - if take is not None: - cmd.extend(["-t", str(take)]) - if is_intra: - cmd.extend(["-c:v", "copy", "-c:a", "copy"]) - else: - video_args, audio_args = _build_reencode_args(source_path) - cmd.extend(video_args) - cmd.extend(audio_args) - cmd.extend(["-avoid_negative_ts", "make_zero", str(trimmed_path)]) - result = subprocess.run(cmd, capture_output=True, text=True) - if result.returncode != 0: - raise PreprocessError( - f"Failed to trim segment {segment_ids[i]}", - filter_type="concat", - command=" ".join(cmd), - stderr=result.stderr, - ) - trimmed_segments.append(trimmed_path) - - concat_list = temp_dir / "concat_list.txt" - with open(concat_list, "w", encoding="utf-8") as f: - for seg in trimmed_segments: - f.write(f"file '{seg.resolve()}'\n") - - print(f" Stitching {len(trimmed_segments)} segments → {output_path.name}" + (" + loudnorm" if needs_loudnorm else "") + "...") - cmd = [ - "ffmpeg", "-y", - "-f", "concat", "-safe", "0", "-i", str(concat_list), - "-c:v", "copy", - ] - if needs_loudnorm: - cmd.extend(["-af", loudnorm_filter, "-c:a", "aac", "-b:a", "192k"]) - else: - cmd.extend(["-c:a", "copy"]) - cmd.extend(["-movflags", "+faststart", str(output_path)]) - - result = subprocess.run(cmd, capture_output=True, text=True) - if result.returncode != 0: - raise PreprocessError( - "Segment concatenation failed", - filter_type="concat", - command=" ".join(cmd), - stderr=result.stderr, - ) - - # Clean up temp files - for segment in trimmed_segments: - if segment.exists(): - segment.unlink() - concat_list.unlink() - try: - temp_dir.rmdir() - except OSError: - pass - - total_duration = get_video_duration(output_path) - print(f" Stitched duration: {format_time(total_duration)}") - return output_path diff --git a/gnommo/renderer.py b/gnommo/renderer.py index 60176c2..e98e5ed 100644 --- a/gnommo/renderer.py +++ b/gnommo/renderer.py @@ -385,7 +385,7 @@ def build_ffmpeg_command(plan: RenderPlan, output_path: Path) -> list[str]: # Concat mode: when plan.narration_segments is set, add each processed segment # as its own input (trimmed by skip/take) and concatenate them in-graph into a # single normalized narration stream — replacing the pre-stitched - # narration_combined input. Otherwise, the single always-visible input path. + # single pre-stitched narration input. Otherwise, the single always-visible input path. # Add -ss seek BEFORE -i for skip parameter and/or partial rendering. always_visible_inputs: list[int] = [] narration_concat = None # (video_label, audio_label) when concat mode is active diff --git a/gnommo/state.py b/gnommo/state.py index acb28eb..b2d99cb 100644 --- a/gnommo/state.py +++ b/gnommo/state.py @@ -12,7 +12,7 @@ Fingerprinting is hybrid: project.json, slides.json, audio.json, transcripts) are hashed (sha256) so a ``touch`` or a git checkout that only rewrites mtimes doesn't force a needless rerun; - - large media (processed segments, narration_combined.mov, source videos and + - large media (processed narration segments, source videos and images) use mtime+size, which is cheap and good enough to detect real edits. The state file is purely an optimization: any read/parse/write failure degrades diff --git a/gnommo/transcriber.py b/gnommo/transcriber.py index a80fc86..7fb2fb7 100644 --- a/gnommo/transcriber.py +++ b/gnommo/transcriber.py @@ -102,88 +102,3 @@ def load_transcript( return [ TranscribedWord(word=w["word"], start=w["start"], end=w["end"]) for w in data ] - - -def _format_srt_timestamp(seconds: float) -> str: - """Format seconds as SRT timestamp: HH:MM:SS,mmm""" - hours = int(seconds // 3600) - minutes = int((seconds % 3600) // 60) - secs = int(seconds % 60) - millis = int((seconds % 1) * 1000) - return f"{hours:02d}:{minutes:02d}:{secs:02d},{millis:03d}" - - -def words_to_srt( - words: list[TranscribedWord], - max_words_per_line: int = 10, - max_duration: float = 5.0, - gap_threshold: float = 1.0, -) -> str: - """ - Convert word-level timestamps to SRT caption format. - - Groups words into readable caption segments based on: - - Maximum words per line (default: 10) - - Maximum segment duration (default: 5 seconds) - - Natural gaps between words (default: 1 second pause triggers new segment) - - Args: - words: List of TranscribedWord with timestamps - max_words_per_line: Maximum words before splitting to new segment - max_duration: Maximum duration of a single caption segment - gap_threshold: Pause duration that triggers a new segment - - Returns: - SRT formatted string ready for YouTube upload - """ - if not words: - return "" - - segments: list[tuple[float, float, str]] = [] # (start, end, text) - current_words: list[str] = [] - segment_start: float = words[0].start - segment_end: float = words[0].end - - for i, word in enumerate(words): - # Check if we should start a new segment - start_new_segment = False - - # Gap between words - if current_words and (word.start - segment_end) > gap_threshold: - start_new_segment = True - - # Too many words - if len(current_words) >= max_words_per_line: - start_new_segment = True - - # Segment too long - if current_words and (word.end - segment_start) > max_duration: - start_new_segment = True - - if start_new_segment and current_words: - # Save current segment - text = " ".join(current_words) - segments.append((segment_start, segment_end, text)) - # Start new segment - current_words = [] - segment_start = word.start - - current_words.append(word.word) - segment_end = word.end - - # Don't forget the last segment - if current_words: - text = " ".join(current_words) - segments.append((segment_start, segment_end, text)) - - # Format as SRT - srt_lines = [] - for idx, (start, end, text) in enumerate(segments, 1): - srt_lines.append(str(idx)) - srt_lines.append( - f"{_format_srt_timestamp(start)} --> {_format_srt_timestamp(end)}" - ) - srt_lines.append(text) - srt_lines.append("") # Blank line between entries - - return "\n".join(srt_lines) diff --git a/gnommo/transfer.py b/gnommo/transfer.py index 804b723..e1c15da 100644 --- a/gnommo/transfer.py +++ b/gnommo/transfer.py @@ -31,7 +31,6 @@ _DOWN_EXCLUDES = [ "media/videos/intermediate/", "media/narration/low/", "media/videos/low/", - "media/videos/narration_combined.mov", "**/chunks/", "*.tmp", ".*", # rsync in-progress temp files (.filename.XXXXXX) and .DS_Store