The stitch cleanup and the gnommokey improvement

This commit is contained in:
2026-07-15 19:47:31 +02:00
parent 56e1cd985e
commit 83715e85e8
11 changed files with 1162 additions and 1285 deletions
@@ -0,0 +1,497 @@
[
{
"word": "This",
"start": 10.74,
"end": 11.44
},
{
"word": "is",
"start": 11.44,
"end": 11.64
},
{
"word": "the",
"start": 11.64,
"end": 11.82
},
{
"word": "first",
"start": 11.82,
"end": 12.04
},
{
"word": "slide.",
"start": 12.04,
"end": 12.44
},
{
"word": "It",
"start": 12.92,
"end": 13.34
},
{
"word": "appears",
"start": 13.34,
"end": 13.7
},
{
"word": "immediate.",
"start": 13.7,
"end": 14.18
},
{
"word": "However,",
"start": 15.36,
"end": 16.06
},
{
"word": "this",
"start": 16.38,
"end": 16.48
},
{
"word": "is",
"start": 16.48,
"end": 16.62
},
{
"word": "the",
"start": 16.62,
"end": 16.8
},
{
"word": "second",
"start": 16.8,
"end": 17.08
},
{
"word": "slide.",
"start": 17.08,
"end": 17.42
},
{
"word": "It",
"start": 17.78,
"end": 18.02
},
{
"word": "should",
"start": 18.02,
"end": 18.24
},
{
"word": "appear",
"start": 18.24,
"end": 18.56
},
{
"word": "one",
"start": 18.56,
"end": 19.02
},
{
"word": "second",
"start": 19.02,
"end": 19.5
},
{
"word": "prior",
"start": 19.5,
"end": 19.92
},
{
"word": "to",
"start": 19.92,
"end": 20.16
},
{
"word": "the",
"start": 20.16,
"end": 20.26
},
{
"word": "word",
"start": 20.26,
"end": 20.54
},
{
"word": "when",
"start": 20.54,
"end": 21.24
},
{
"word": "I",
"start": 21.24,
"end": 21.32
},
{
"word": "say",
"start": 21.32,
"end": 21.5
},
{
"word": "whoever",
"start": 21.5,
"end": 21.86
},
{
"word": "first",
"start": 21.86,
"end": 22.44
},
{
"word": "time.",
"start": 22.44,
"end": 22.7
},
{
"word": "This",
"start": 24.3,
"end": 25.0
},
{
"word": "is",
"start": 25.0,
"end": 25.14
},
{
"word": "me",
"start": 25.14,
"end": 25.38
},
{
"word": "taking,",
"start": 25.38,
"end": 25.78
},
{
"word": "talking",
"start": 26.14,
"end": 27.18
},
{
"word": "alongside",
"start": 27.18,
"end": 27.66
},
{
"word": "a",
"start": 27.66,
"end": 27.92
},
{
"word": "video.",
"start": 27.92,
"end": 28.16
},
{
"word": "The",
"start": 28.68,
"end": 28.96
},
{
"word": "video",
"start": 28.96,
"end": 29.2
},
{
"word": "is",
"start": 29.2,
"end": 29.4
},
{
"word": "constrained",
"start": 29.4,
"end": 29.82
},
{
"word": "within",
"start": 29.82,
"end": 30.18
},
{
"word": "the",
"start": 30.18,
"end": 30.36
},
{
"word": "red",
"start": 30.36,
"end": 30.52
},
{
"word": "square.",
"start": 30.52,
"end": 30.94
},
{
"word": "Notice",
"start": 31.3,
"end": 31.48
},
{
"word": "how",
"start": 31.48,
"end": 31.78
},
{
"word": "the",
"start": 31.78,
"end": 31.96
},
{
"word": "video",
"start": 31.96,
"end": 32.16
},
{
"word": "stops",
"start": 32.16,
"end": 32.48
},
{
"word": "immediately",
"start": 32.48,
"end": 32.98
},
{
"word": "when",
"start": 32.98,
"end": 33.4
},
{
"word": "we",
"start": 33.4,
"end": 33.58
},
{
"word": "make",
"start": 33.58,
"end": 33.76
},
{
"word": "the",
"start": 33.76,
"end": 34.0
},
{
"word": "transition",
"start": 34.0,
"end": 34.42
},
{
"word": "to",
"start": 34.42,
"end": 34.72
},
{
"word": "the",
"start": 34.72,
"end": 34.84
},
{
"word": "next",
"start": 34.84,
"end": 35.06
},
{
"word": "slide.",
"start": 35.06,
"end": 35.48
},
{
"word": "I",
"start": 37.2,
"end": 37.76
},
{
"word": "will",
"start": 37.76,
"end": 37.82
},
{
"word": "continue",
"start": 37.82,
"end": 38.12
},
{
"word": "to",
"start": 38.12,
"end": 38.34
},
{
"word": "talk",
"start": 38.34,
"end": 38.58
},
{
"word": "without",
"start": 38.58,
"end": 38.92
},
{
"word": "pause,",
"start": 38.92,
"end": 39.26
},
{
"word": "but",
"start": 39.5,
"end": 39.6
},
{
"word": "in",
"start": 39.6,
"end": 39.72
},
{
"word": "the",
"start": 39.72,
"end": 39.8
},
{
"word": "finished",
"start": 39.8,
"end": 40.0
},
{
"word": "recording",
"start": 40.0,
"end": 40.48
},
{
"word": "there",
"start": 40.48,
"end": 41.22
},
{
"word": "will",
"start": 41.22,
"end": 41.38
},
{
"word": "be",
"start": 41.38,
"end": 41.58
},
{
"word": "a",
"start": 41.58,
"end": 41.68
},
{
"word": "pause",
"start": 41.68,
"end": 41.96
},
{
"word": "before",
"start": 41.96,
"end": 42.32
},
{
"word": "the",
"start": 42.32,
"end": 42.52
},
{
"word": "narration",
"start": 42.52,
"end": 43.06
},
{
"word": "continues.",
"start": 43.06,
"end": 43.66
},
{
"word": "Now",
"start": 44.44,
"end": 44.56
},
{
"word": "a",
"start": 44.56,
"end": 44.7
},
{
"word": "video",
"start": 44.7,
"end": 44.94
},
{
"word": "will",
"start": 44.94,
"end": 45.12
},
{
"word": "play",
"start": 45.12,
"end": 45.4
},
{
"word": "that",
"start": 45.4,
"end": 45.8
},
{
"word": "pauses",
"start": 45.8,
"end": 46.52
},
{
"word": "the",
"start": 46.52,
"end": 46.8
},
{
"word": "narration.",
"start": 46.8,
"end": 47.22
},
{
"word": "Notice",
"start": 48.66,
"end": 49.22
},
{
"word": "how",
"start": 49.22,
"end": 49.44
},
{
"word": "my",
"start": 49.44,
"end": 49.6
},
{
"word": "voice",
"start": 49.6,
"end": 49.84
},
{
"word": "continues",
"start": 49.84,
"end": 50.38
},
{
"word": "after",
"start": 50.38,
"end": 50.88
},
{
"word": "the",
"start": 50.88,
"end": 51.04
},
{
"word": "video",
"start": 51.04,
"end": 51.28
},
{
"word": "finished.",
"start": 51.28,
"end": 51.8
}
]
@@ -0,0 +1,497 @@
[
{
"word": "This",
"start": 10.632,
"end": 11.312
},
{
"word": "is",
"start": 11.312,
"end": 11.512
},
{
"word": "the",
"start": 11.512,
"end": 11.692
},
{
"word": "first",
"start": 11.692,
"end": 11.912
},
{
"word": "slide.",
"start": 11.912,
"end": 12.312
},
{
"word": "It",
"start": 12.852,
"end": 13.192
},
{
"word": "appears",
"start": 13.192,
"end": 13.552
},
{
"word": "immediate.",
"start": 13.552,
"end": 14.032
},
{
"word": "However,",
"start": 15.452,
"end": 15.932
},
{
"word": "this",
"start": 16.272,
"end": 16.352
},
{
"word": "is",
"start": 16.352,
"end": 16.492
},
{
"word": "the",
"start": 16.492,
"end": 16.652
},
{
"word": "second",
"start": 16.652,
"end": 16.952
},
{
"word": "slide.",
"start": 16.952,
"end": 17.292
},
{
"word": "It",
"start": 17.572,
"end": 17.872
},
{
"word": "should",
"start": 17.872,
"end": 18.112
},
{
"word": "appear",
"start": 18.112,
"end": 18.432
},
{
"word": "one",
"start": 18.432,
"end": 18.892
},
{
"word": "second",
"start": 18.892,
"end": 19.372
},
{
"word": "prior",
"start": 19.372,
"end": 19.792
},
{
"word": "to",
"start": 19.792,
"end": 20.032
},
{
"word": "the",
"start": 20.032,
"end": 20.152
},
{
"word": "word",
"start": 20.152,
"end": 20.412
},
{
"word": "when",
"start": 20.412,
"end": 21.112
},
{
"word": "I",
"start": 21.112,
"end": 21.192
},
{
"word": "say",
"start": 21.192,
"end": 21.352
},
{
"word": "whoever",
"start": 21.352,
"end": 21.732
},
{
"word": "first",
"start": 21.732,
"end": 22.312
},
{
"word": "time.",
"start": 22.312,
"end": 22.592
},
{
"word": "This",
"start": 24.532,
"end": 24.872
},
{
"word": "is",
"start": 24.872,
"end": 25.032
},
{
"word": "me",
"start": 25.032,
"end": 25.252
},
{
"word": "taking,",
"start": 25.252,
"end": 25.652
},
{
"word": "talking",
"start": 26.092,
"end": 27.052
},
{
"word": "alongside",
"start": 27.052,
"end": 27.532
},
{
"word": "a",
"start": 27.532,
"end": 27.792
},
{
"word": "video.",
"start": 27.792,
"end": 28.052
},
{
"word": "The",
"start": 28.652,
"end": 28.832
},
{
"word": "video",
"start": 28.832,
"end": 29.092
},
{
"word": "is",
"start": 29.092,
"end": 29.272
},
{
"word": "constrained",
"start": 29.272,
"end": 29.712
},
{
"word": "within",
"start": 29.712,
"end": 30.052
},
{
"word": "the",
"start": 30.052,
"end": 30.232
},
{
"word": "red",
"start": 30.232,
"end": 30.392
},
{
"word": "square.",
"start": 30.392,
"end": 30.792
},
{
"word": "Notice",
"start": 30.792,
"end": 31.352
},
{
"word": "how",
"start": 31.352,
"end": 31.652
},
{
"word": "the",
"start": 31.652,
"end": 31.832
},
{
"word": "video",
"start": 31.832,
"end": 32.032
},
{
"word": "stops",
"start": 32.032,
"end": 32.372
},
{
"word": "immediately",
"start": 32.372,
"end": 32.852
},
{
"word": "when",
"start": 32.852,
"end": 33.272
},
{
"word": "we",
"start": 33.272,
"end": 33.452
},
{
"word": "make",
"start": 33.452,
"end": 33.632
},
{
"word": "the",
"start": 33.632,
"end": 33.872
},
{
"word": "transition",
"start": 33.872,
"end": 34.292
},
{
"word": "to",
"start": 34.292,
"end": 34.592
},
{
"word": "the",
"start": 34.592,
"end": 34.712
},
{
"word": "next",
"start": 34.712,
"end": 34.932
},
{
"word": "slide.",
"start": 34.932,
"end": 35.392
},
{
"word": "I",
"start": 37.112,
"end": 37.632
},
{
"word": "will",
"start": 37.632,
"end": 37.692
},
{
"word": "continue",
"start": 37.692,
"end": 37.992
},
{
"word": "to",
"start": 37.992,
"end": 38.212
},
{
"word": "talk",
"start": 38.212,
"end": 38.452
},
{
"word": "without",
"start": 38.452,
"end": 38.792
},
{
"word": "pause,",
"start": 38.792,
"end": 39.132
},
{
"word": "but",
"start": 39.372,
"end": 39.472
},
{
"word": "in",
"start": 39.472,
"end": 39.592
},
{
"word": "the",
"start": 39.592,
"end": 39.652
},
{
"word": "finished",
"start": 39.652,
"end": 39.872
},
{
"word": "recording",
"start": 39.872,
"end": 40.352
},
{
"word": "there",
"start": 40.352,
"end": 41.092
},
{
"word": "will",
"start": 41.092,
"end": 41.252
},
{
"word": "be",
"start": 41.252,
"end": 41.452
},
{
"word": "a",
"start": 41.452,
"end": 41.552
},
{
"word": "pause",
"start": 41.552,
"end": 41.812
},
{
"word": "before",
"start": 41.812,
"end": 42.192
},
{
"word": "the",
"start": 42.192,
"end": 42.392
},
{
"word": "narration",
"start": 42.392,
"end": 42.932
},
{
"word": "continues.",
"start": 42.932,
"end": 43.552
},
{
"word": "Now",
"start": 44.232,
"end": 44.432
},
{
"word": "a",
"start": 44.432,
"end": 44.572
},
{
"word": "video",
"start": 44.572,
"end": 44.812
},
{
"word": "will",
"start": 44.812,
"end": 44.972
},
{
"word": "play",
"start": 44.972,
"end": 45.272
},
{
"word": "that",
"start": 45.272,
"end": 45.672
},
{
"word": "pauses",
"start": 45.672,
"end": 46.412
},
{
"word": "the",
"start": 46.412,
"end": 46.672
},
{
"word": "narration.",
"start": 46.672,
"end": 47.092
},
{
"word": "Notice",
"start": 48.352,
"end": 49.092
},
{
"word": "how",
"start": 49.092,
"end": 49.312
},
{
"word": "my",
"start": 49.312,
"end": 49.492
},
{
"word": "voice",
"start": 49.492,
"end": 49.752
},
{
"word": "continues",
"start": 49.752,
"end": 50.272
},
{
"word": "after",
"start": 50.272,
"end": 50.752
},
{
"word": "the",
"start": 50.752,
"end": 50.932
},
{
"word": "video",
"start": 50.932,
"end": 51.152
},
{
"word": "finished.",
"start": 51.152,
"end": 51.652
}
]
+1 -1
View File
@@ -4,7 +4,7 @@ _Generated: 2026-07-15_
## Slide Alignment Issues (7) ## Slide Alignment Issues (7)
Slide markers that could not be matched to the spoken narration (likely adlibbed). Slide markers that could not be matched to the spoken narration (likely adlibbed).
- [ ] `S6`_"(repaired: voice continues after the video finished)"_ - [ ] `S6`_"(end of: voice continues after the video finished)"_
- [ ] `S7`_"This is the first slide. It appears immediately."_ - [ ] `S7`_"This is the first slide. It appears immediately."_
- [ ] `S8`_"However, this is the second slide. It should appea"_ - [ ] `S8`_"However, this is the second slide. It should appea"_
- [ ] `S9`_"This is me talking alongside a video. The video is"_ - [ ] `S9`_"This is me talking alongside a video. The video is"_
+135 -987
View File
File diff suppressed because it is too large Load Diff
+8 -1
View File
@@ -132,6 +132,13 @@ class GnommoKeyConfig:
# How aggressively to apply despill (0-1) # How aggressively to apply despill (0-1)
despill_strength: float = 0.5 despill_strength: float = 0.5
# Interior green-limiter (0.0-1.0, 0 = off). Suppresses green cast/spill
# across the whole frame even where green is NOT the dominant channel — the
# case the bias/edge despill misses (e.g. green bounce on skin/a bald head).
# Caps green at a reference blended between max(r,b) [0.0] and the r/b
# average [1.0]. 0.5-0.7 removes cast without pushing skin magenta.
spill_suppress: float = 0.0
# Alpha bias: influences edge treatment (RGB) # Alpha bias: influences edge treatment (RGB)
# Can help with edge color contamination # Can help with edge color contamination
alpha_bias: tuple[int, int, int] = None alpha_bias: tuple[int, int, int] = None
@@ -528,7 +535,7 @@ class RenderPlan:
) # Gaps in narration for interstitial videos ) # Gaps in narration for interstitial videos
# Render-time narration concat: ordered segments (skip/take + offset) to # Render-time narration concat: ordered segments (skip/take + offset) to
# concatenate directly at render time instead of using a single pre-stitched # concatenate directly at render time instead of using a single pre-stitched
# narration_combined input. Typed loosely (list of narration.NarrationSegment) # single pre-stitched narration input. Typed loosely (list of narration.NarrationSegment)
# to avoid a circular import between models and narration. # to avoid a circular import between models and narration.
narration_segments: list = field(default_factory=list) narration_segments: list = field(default_factory=list)
# Outro sequence (plays after narration ends) # Outro sequence (plays after narration ends)
+2 -2
View File
@@ -1,6 +1,6 @@
"""Deterministic narration scheduling for render-time segment stitching. """Deterministic narration scheduling for render-time segment stitching.
Instead of pre-stitching segments into narration_combined.mov, the render stage Rather than pre-stitching segments into one file, the render stage
concatenates the processed segments directly. From narration.json + the cached concatenates the processed segments directly. From narration.json + the cached
per-segment transcripts this module computes two things: per-segment transcripts this module computes two things:
@@ -8,7 +8,7 @@ per-segment transcripts this module computes two things:
combined timeline) — this drives the ffmpeg concat at render time; and combined timeline) — this drives the ffmpeg concat at render time; and
2. the merged word-level transcript, with every word re-timed into the 2. the merged word-level transcript, with every word re-timed into the
combined timeline — this drives slide alignment, exactly what combined timeline — this drives slide alignment, exactly what
re-transcribing narration_combined.mov used to produce, but derived re-transcribing a pre-stitched narration file used to produce, but derived
deterministically (no re-transcription, no separate combined file). deterministically (no re-transcription, no separate combined file).
The processed files share framerate and format and are uncompressed, so the The processed files share framerate and format and are uncompressed, so the
+17 -203
View File
@@ -1206,6 +1206,22 @@ def build_gnommokey_filter(config: dict) -> str:
parts.append(f"geq=r='{new_r}':g='{new_g}':b='{new_b}':a='alpha(X,Y)'") parts.append(f"geq=r='{new_r}':g='{new_g}':b='{new_b}':a='alpha(X,Y)'")
# Interior spill suppression: cap the spill channel across the WHOLE frame,
# including fully-opaque interior pixels the bias/edge despill can't reach
# (green bounce on skin, a bald head, etc.). Caps the channel at a reference
# blended between max(other two) [t=0] and their average [t=1]; green is only
# ever reduced, never boosted, so non-spilled pixels are untouched.
if cfg.spill_suppress > 0:
t = min(max(cfg.spill_suppress, 0.0), 1.0)
if is_green_screen:
ref = f"((1-{t:.3f})*max(r(X,Y),b(X,Y))+{t:.3f}*(r(X,Y)+b(X,Y))/2)"
new_g = f"min(g(X,Y),{ref})"
parts.append(f"geq=r='r(X,Y)':g='{new_g}':b='b(X,Y)':a='alpha(X,Y)'")
else:
ref = f"((1-{t:.3f})*max(r(X,Y),g(X,Y))+{t:.3f}*(r(X,Y)+g(X,Y))/2)"
new_b = f"min(b(X,Y),{ref})"
parts.append(f"geq=r='r(X,Y)':g='g(X,Y)':b='{new_b}':a='alpha(X,Y)'")
# Edge-aware despill: aggressively suppress green at semi-transparent edges # Edge-aware despill: aggressively suppress green at semi-transparent edges
# This targets the 2-4px green fringe that regular despill misses # This targets the 2-4px green fringe that regular despill misses
# edge_factor is high (1.0) at alpha=128, low (0) at alpha=0 or 255 # edge_factor is high (1.0) at alpha=128, low (0) at alpha=0 or 255
@@ -1292,6 +1308,7 @@ def parse_gnommokey_config(config: dict) -> GnommoKeyConfig:
clip_white=float(config.get("clip_white", 100.0)), clip_white=float(config.get("clip_white", 100.0)),
despill_bias=despill_bias, despill_bias=despill_bias,
despill_strength=float(config.get("despill_strength", 0.5)), despill_strength=float(config.get("despill_strength", 0.5)),
spill_suppress=float(config.get("spill_suppress", 0.0)),
alpha_bias=alpha_bias, alpha_bias=alpha_bias,
protect_luma=int(config.get("protect_luma", -1)), protect_luma=int(config.get("protect_luma", -1)),
shadow_boost=float(config.get("shadow_boost", 0.0)), shadow_boost=float(config.get("shadow_boost", 0.0)),
@@ -2314,206 +2331,3 @@ def needs_preprocessing(videos_dir: Path, video_source: VideoSource) -> bool:
return True return True
return True return True
def _build_loudnorm_filter(loudnorm_config: Optional[dict]) -> str:
_cfg = loudnorm_config or {}
_lufs = float(_cfg.get("target_lufs", -14))
_lra = float(_cfg.get("target_lra", 11))
_tp = float(_cfg.get("target_tp", -1.5))
return f"loudnorm=I={_lufs:.1f}:LRA={_lra:.1f}:TP={_tp:.1f}"
def _build_reencode_args(source_path: Path) -> tuple[list[str], list[str]]:
"""
Return (video_args, audio_args) for re-encoding an inter-frame source to a
normalized intra-frame format. Does not include -avoid_negative_ts or the
output path — callers add those.
"""
has_alpha = _video_has_alpha(source_path)
if has_alpha:
video_args = [
"-vf", "fps=30,format=yuva444p10le",
"-c:v", "prores_ks", "-profile:v", "4", "-pix_fmt", "yuva444p10le",
]
audio_args = ["-c:a", "pcm_s16le"]
else:
video_args = [
"-vf", "fps=30",
"-c:v", "libx264", "-preset", "fast", "-crf", "18",
"-movflags", "+faststart",
]
audio_args = ["-c:a", "aac", "-b:a", "192k"]
return video_args, audio_args
def stitch_narration_segments(
videos_dir: Path,
segment_ids: list[str],
videos: dict[str, VideoSource],
output_path: Path,
verbose: bool = False,
default_end_trim: float = 0.0,
loudnorm_config: Optional[dict] = None,
) -> Path:
"""
Stitch multiple narration video segments into a single file.
Each segment's skip and take values are applied to trim dead video at the
start/end of each recording. The segments are concatenated in the order
specified by segment_ids.
Args:
videos_dir: Directory containing video files
segment_ids: Ordered list of video IDs from videos.json
videos: Dict of video ID -> VideoSource from videos.json
output_path: Path for the concatenated output file
verbose: Enable verbose output
default_end_trim: Seconds to trim from the end when no explicit end/take is set
Returns:
Path to the stitched video file.
"""
print(f" Concatenating {len(segment_ids)} narration segment(s)...")
needs_loudnorm = any(
videos[seg_id].defer_loudnorm for seg_id in segment_ids if seg_id in videos
)
loudnorm_filter = _build_loudnorm_filter(loudnorm_config) if needs_loudnorm else None
# ------------------------------------------------------------------ #
# Gather per-segment metadata #
# ------------------------------------------------------------------ #
segments: list[tuple[Path, float, Optional[float], float, str, bool]] = []
for video_id in segment_ids:
if video_id not in videos:
raise PreprocessError(
f"Narration segment '{video_id}' not found in videos.json",
filter_type=None,
)
video_source = videos[video_id]
source_path = get_preprocessed_path(videos_dir, video_source)
if not source_path.exists():
raise PreprocessError(
f"Narration segment not found: {source_path}",
filter_type=None,
)
full_duration = get_video_duration(source_path)
skip = video_source.skip or 0.0
take = video_source.take
if take is None and default_end_trim > 0:
take = max(0.0, full_duration - skip - default_end_trim)
effective_duration = min(take, full_duration - skip) if take is not None else full_duration - skip
codec = _get_video_codec(source_path)
is_intra = codec in _INTRA_ONLY_CODECS
if verbose:
mode = "stream copy (fast)" if is_intra else "re-encode"
print(f" {video_id}: {source_path.name} [{codec or '?'}] skip={skip}s take={take or 'all'}s → {effective_duration:.1f}s ({mode})")
segments.append((source_path, skip, take, effective_duration, codec, is_intra))
# ------------------------------------------------------------------ #
# Single-segment fast path — no temp file, no concat round-trip #
# ------------------------------------------------------------------ #
if len(segments) == 1:
source_path, skip, take, effective_duration, codec, is_intra = segments[0]
print(f" Trimming → {output_path.name}" + (" + loudnorm" if needs_loudnorm else "") + "...")
cmd = ["ffmpeg", "-y"]
if skip > 0:
cmd.extend(["-ss", str(skip)])
cmd.extend(["-i", str(source_path)])
if take is not None:
cmd.extend(["-t", str(take)])
if is_intra:
cmd.extend(["-c:v", "copy"])
audio_args = ["-c:a", "copy"]
else:
video_args, audio_args = _build_reencode_args(source_path)
cmd.extend(video_args)
if needs_loudnorm:
cmd.extend(["-af", loudnorm_filter, "-c:a", "aac", "-b:a", "192k"])
else:
cmd.extend(audio_args)
cmd.extend(["-avoid_negative_ts", "make_zero", str(output_path)])
result = subprocess.run(cmd, capture_output=True, text=True)
if result.returncode != 0:
raise PreprocessError(
f"Failed to trim segment {segment_ids[0]}",
filter_type="concat",
command=" ".join(cmd),
stderr=result.stderr,
)
total_duration = get_video_duration(output_path)
print(f" Stitched duration: {format_time(total_duration)}")
return output_path
# ------------------------------------------------------------------ #
# Multi-segment path — trim each to temp, then concat + loudnorm #
# ------------------------------------------------------------------ #
temp_dir = output_path.parent / "concat_temp"
temp_dir.mkdir(parents=True, exist_ok=True)
trimmed_segments: list[Path] = []
for i, (source_path, skip, take, effective_duration, codec, is_intra) in enumerate(segments):
trimmed_path = temp_dir / f"segment_{i:03d}.mov"
cmd = ["ffmpeg", "-y"]
if skip > 0:
cmd.extend(["-ss", str(skip)])
cmd.extend(["-i", str(source_path)])
if take is not None:
cmd.extend(["-t", str(take)])
if is_intra:
cmd.extend(["-c:v", "copy", "-c:a", "copy"])
else:
video_args, audio_args = _build_reencode_args(source_path)
cmd.extend(video_args)
cmd.extend(audio_args)
cmd.extend(["-avoid_negative_ts", "make_zero", str(trimmed_path)])
result = subprocess.run(cmd, capture_output=True, text=True)
if result.returncode != 0:
raise PreprocessError(
f"Failed to trim segment {segment_ids[i]}",
filter_type="concat",
command=" ".join(cmd),
stderr=result.stderr,
)
trimmed_segments.append(trimmed_path)
concat_list = temp_dir / "concat_list.txt"
with open(concat_list, "w", encoding="utf-8") as f:
for seg in trimmed_segments:
f.write(f"file '{seg.resolve()}'\n")
print(f" Stitching {len(trimmed_segments)} segments → {output_path.name}" + (" + loudnorm" if needs_loudnorm else "") + "...")
cmd = [
"ffmpeg", "-y",
"-f", "concat", "-safe", "0", "-i", str(concat_list),
"-c:v", "copy",
]
if needs_loudnorm:
cmd.extend(["-af", loudnorm_filter, "-c:a", "aac", "-b:a", "192k"])
else:
cmd.extend(["-c:a", "copy"])
cmd.extend(["-movflags", "+faststart", str(output_path)])
result = subprocess.run(cmd, capture_output=True, text=True)
if result.returncode != 0:
raise PreprocessError(
"Segment concatenation failed",
filter_type="concat",
command=" ".join(cmd),
stderr=result.stderr,
)
# Clean up temp files
for segment in trimmed_segments:
if segment.exists():
segment.unlink()
concat_list.unlink()
try:
temp_dir.rmdir()
except OSError:
pass
total_duration = get_video_duration(output_path)
print(f" Stitched duration: {format_time(total_duration)}")
return output_path
+1 -1
View File
@@ -385,7 +385,7 @@ def build_ffmpeg_command(plan: RenderPlan, output_path: Path) -> list[str]:
# Concat mode: when plan.narration_segments is set, add each processed segment # Concat mode: when plan.narration_segments is set, add each processed segment
# as its own input (trimmed by skip/take) and concatenate them in-graph into a # as its own input (trimmed by skip/take) and concatenate them in-graph into a
# single normalized narration stream — replacing the pre-stitched # single normalized narration stream — replacing the pre-stitched
# narration_combined input. Otherwise, the single always-visible input path. # single pre-stitched narration input. Otherwise, the single always-visible input path.
# Add -ss seek BEFORE -i for skip parameter and/or partial rendering. # Add -ss seek BEFORE -i for skip parameter and/or partial rendering.
always_visible_inputs: list[int] = [] always_visible_inputs: list[int] = []
narration_concat = None # (video_label, audio_label) when concat mode is active narration_concat = None # (video_label, audio_label) when concat mode is active
+1 -1
View File
@@ -12,7 +12,7 @@ Fingerprinting is hybrid:
project.json, slides.json, audio.json, transcripts) are hashed (sha256) so a project.json, slides.json, audio.json, transcripts) are hashed (sha256) so a
``touch`` or a git checkout that only rewrites mtimes doesn't force a ``touch`` or a git checkout that only rewrites mtimes doesn't force a
needless rerun; needless rerun;
- large media (processed segments, narration_combined.mov, source videos and - large media (processed narration segments, source videos and
images) use mtime+size, which is cheap and good enough to detect real edits. images) use mtime+size, which is cheap and good enough to detect real edits.
The state file is purely an optimization: any read/parse/write failure degrades The state file is purely an optimization: any read/parse/write failure degrades
-85
View File
@@ -102,88 +102,3 @@ def load_transcript(
return [ return [
TranscribedWord(word=w["word"], start=w["start"], end=w["end"]) for w in data TranscribedWord(word=w["word"], start=w["start"], end=w["end"]) for w in data
] ]
def _format_srt_timestamp(seconds: float) -> str:
"""Format seconds as SRT timestamp: HH:MM:SS,mmm"""
hours = int(seconds // 3600)
minutes = int((seconds % 3600) // 60)
secs = int(seconds % 60)
millis = int((seconds % 1) * 1000)
return f"{hours:02d}:{minutes:02d}:{secs:02d},{millis:03d}"
def words_to_srt(
words: list[TranscribedWord],
max_words_per_line: int = 10,
max_duration: float = 5.0,
gap_threshold: float = 1.0,
) -> str:
"""
Convert word-level timestamps to SRT caption format.
Groups words into readable caption segments based on:
- Maximum words per line (default: 10)
- Maximum segment duration (default: 5 seconds)
- Natural gaps between words (default: 1 second pause triggers new segment)
Args:
words: List of TranscribedWord with timestamps
max_words_per_line: Maximum words before splitting to new segment
max_duration: Maximum duration of a single caption segment
gap_threshold: Pause duration that triggers a new segment
Returns:
SRT formatted string ready for YouTube upload
"""
if not words:
return ""
segments: list[tuple[float, float, str]] = [] # (start, end, text)
current_words: list[str] = []
segment_start: float = words[0].start
segment_end: float = words[0].end
for i, word in enumerate(words):
# Check if we should start a new segment
start_new_segment = False
# Gap between words
if current_words and (word.start - segment_end) > gap_threshold:
start_new_segment = True
# Too many words
if len(current_words) >= max_words_per_line:
start_new_segment = True
# Segment too long
if current_words and (word.end - segment_start) > max_duration:
start_new_segment = True
if start_new_segment and current_words:
# Save current segment
text = " ".join(current_words)
segments.append((segment_start, segment_end, text))
# Start new segment
current_words = []
segment_start = word.start
current_words.append(word.word)
segment_end = word.end
# Don't forget the last segment
if current_words:
text = " ".join(current_words)
segments.append((segment_start, segment_end, text))
# Format as SRT
srt_lines = []
for idx, (start, end, text) in enumerate(segments, 1):
srt_lines.append(str(idx))
srt_lines.append(
f"{_format_srt_timestamp(start)} --> {_format_srt_timestamp(end)}"
)
srt_lines.append(text)
srt_lines.append("") # Blank line between entries
return "\n".join(srt_lines)
-1
View File
@@ -31,7 +31,6 @@ _DOWN_EXCLUDES = [
"media/videos/intermediate/", "media/videos/intermediate/",
"media/narration/low/", "media/narration/low/",
"media/videos/low/", "media/videos/low/",
"media/videos/narration_combined.mov",
"**/chunks/", "**/chunks/",
"*.tmp", "*.tmp",
".*", # rsync in-progress temp files (.filename.XXXXXX) and .DS_Store ".*", # rsync in-progress temp files (.filename.XXXXXX) and .DS_Store