Adding fixes to renderer

This commit is contained in:
2026-08-03 14:38:33 +02:00
parent a218e1cc9f
commit f852b36291
8 changed files with 229 additions and 14 deletions
+11
View File
@@ -334,6 +334,14 @@ class VideoSource:
cutout: Optional[ cutout: Optional[
str str
] = None # Name of cutout to place video in (from project.json cutouts) ] = None # Name of cutout to place video in (from project.json cutouts)
# CSS-like placement when the video's aspect ratio differs from the cutout:
# object_fit: "cover" (default) fills the cutout and crops the overflow
# (zoomed by `zoom`); "contain" shrinks the whole video to fit
# inside and pads the remainder transparently (no cropping).
# object_position: which edge to anchor to — "center" (default) | "top" |
# "bottom" | "left" | "right".
object_fit: str = "cover"
object_position: str = "center"
always_visible: bool = False # If True, video is always shown (like talking head) always_visible: bool = False # If True, video is always shown (like talking head)
is_shared: bool = False # If True, source_file is relative to shared_assets/ is_shared: bool = False # If True, source_file is relative to shared_assets/
pause_narration: float = ( pause_narration: float = (
@@ -434,6 +442,9 @@ class VideoEvent:
cutout: "CutoutDefinition" cutout: "CutoutDefinition"
cutout_name: str = "" # resolved cutout name (e.g. "fullscreen"), for display cutout_name: str = "" # resolved cutout name (e.g. "fullscreen"), for display
layer: str = "above" # "above" = on top of slides; "below" = behind slides layer: str = "above" # "above" = on top of slides; "below" = behind slides
# Resolved per-occurrence CSS-like cutout placement (see VideoSource).
object_fit: str = "cover"
object_position: str = "center"
# Effective audio volume for THIS occurrence: an events.json override if present, # Effective audio volume for THIS occurrence: an events.json override if present,
# else the videos.json default. The renderer reads this (not video_source.volume). # else the videos.json default. The renderer reads this (not video_source.volume).
volume: float = 1.0 volume: float = 1.0
+5 -1
View File
@@ -58,7 +58,9 @@ def _resolve_case_insensitive(path: Path) -> Path:
# presentation fields materialized onto events.json and resolved per-event # presentation fields materialized onto events.json and resolved per-event
# (transformer.resolve_video_presentation). Global params (skip/zoom/volume/…) are not # (transformer.resolve_video_presentation). Global params (skip/zoom/volume/…) are not
# yet overridable inline; unknown keys are ignored. # yet overridable inline; unknown keys are ignored.
_MARKER_OVERRIDE_KEYS = frozenset({"cutout", "layer", "end_on", "take", "volume"}) _MARKER_OVERRIDE_KEYS = frozenset(
{"cutout", "layer", "end_on", "take", "volume", "object-fit", "object-position"}
)
# Override keys that are numeric (coerced to float). # Override keys that are numeric (coerced to float).
_MARKER_NUMERIC_KEYS = frozenset({"take", "volume"}) _MARKER_NUMERIC_KEYS = frozenset({"take", "volume"})
@@ -619,6 +621,8 @@ def parse_videos(
skip=skip, skip=skip,
zoom=video_data.get("zoom", 1.0), zoom=video_data.get("zoom", 1.0),
cutout=video_data.get("cutout"), cutout=video_data.get("cutout"),
object_fit=video_data.get("object-fit", "cover"),
object_position=video_data.get("object-position", "center"),
always_visible=video_data.get("always_visible", False), always_visible=video_data.get("always_visible", False),
is_shared=video_data.get("is_shared", False), is_shared=video_data.get("is_shared", False),
pause_narration=float(video_data.get("pause_narration", 0)), pause_narration=float(video_data.get("pause_narration", 0)),
+38 -12
View File
@@ -756,6 +756,38 @@ def build_ffmpeg_command(plan: RenderPlan, output_path: Path) -> list[str]:
return cmd return cmd
def _fit_filter(
w: int, h: int, zoom: float, object_fit: str = "cover", object_position: str = "center"
) -> str:
"""Scale+crop/pad chain that places a source into a w×h cutout, CSS-style.
object_fit "cover" (default): fill the cutout (scaled by `zoom`) and crop the
overflow — object-fit: cover. "contain": shrink the whole video to fit inside and
pad the remainder transparently — object-fit: contain (`zoom` is not applied,
since nothing is cropped). object_position anchors the crop (cover) or the padded
video (contain): center (default) | top | bottom | left | right.
With the defaults (cover/center) this is byte-identical to the long-standing
`scale=…increase,crop=W:H:(iw-W)/2:(ih-H)/2` used everywhere, so callers that pass
defaults render exactly as before.
"""
pos = (object_position or "center").lower()
if (object_fit or "cover").lower() == "contain":
px = "0" if pos == "left" else (f"(ow-iw)" if pos == "right" else "(ow-iw)/2")
py = "0" if pos == "top" else (f"(oh-ih)" if pos == "bottom" else "(oh-ih)/2")
return (
f"scale={w}:{h}:force_original_aspect_ratio=decrease,"
f"pad={w}:{h}:{px}:{py}:color=0x00000000"
)
zw, zh = int(w * zoom), int(h * zoom)
cx = "0" if pos == "left" else (f"(iw-{w})" if pos == "right" else f"(iw-{w})/2")
cy = "0" if pos == "top" else (f"(ih-{h})" if pos == "bottom" else f"(ih-{h})/2")
return (
f"scale={zw}:{zh}:force_original_aspect_ratio=increase,"
f"crop={w}:{h}:{cx}:{cy}"
)
def _calculate_cutout_position( def _calculate_cutout_position(
cutout: CutoutDefinition, frame_width: int, frame_height: int cutout: CutoutDefinition, frame_width: int, frame_height: int
) -> tuple[int, int, int, int]: ) -> tuple[int, int, int, int]:
@@ -1103,8 +1135,7 @@ def build_filter_complex(
filters.append( filters.append(
f"[{video_idx}:v]format=yuva444p10le," f"[{video_idx}:v]format=yuva444p10le,"
f"setpts=PTS-STARTPTS+{start_pts:.3f}/TB," f"setpts=PTS-STARTPTS+{start_pts:.3f}/TB,"
f"scale={zoomed_width}:{zoomed_height}:force_original_aspect_ratio=increase," f"{_fit_filter(cut_width, cut_height, zoom, event.object_fit, event.object_position)},"
f"crop={cut_width}:{cut_height}:(iw-{cut_width})/2:(ih-{cut_height})/2,"
f"format=rgba[{video_label}]" f"format=rgba[{video_label}]"
) )
@@ -1135,8 +1166,7 @@ def build_filter_complex(
filters.append( filters.append(
f"{narr_src}fps={plan.config.fps},setpts=PTS-STARTPTS," f"{narr_src}fps={plan.config.fps},setpts=PTS-STARTPTS,"
f"format=yuva444p10le," f"format=yuva444p10le,"
f"scale={zoomed_width}:{zoomed_height}:force_original_aspect_ratio=increase," f"{_fit_filter(cut_width, cut_height, zoom)},"
f"crop={cut_width}:{cut_height}:(iw-{cut_width})/2:(ih-{cut_height})/2,"
f"format=rgba[{video_label}]" f"format=rgba[{video_label}]"
) )
@@ -1169,8 +1199,7 @@ def build_filter_complex(
f"[{split_labels[seg_idx]}]trim={src_start:.3f}:{src_end:.3f}," f"[{split_labels[seg_idx]}]trim={src_start:.3f}:{src_end:.3f},"
f"setpts=PTS-STARTPTS+{pts_offset:.3f}/TB," f"setpts=PTS-STARTPTS+{pts_offset:.3f}/TB,"
f"format=yuva444p10le," f"format=yuva444p10le,"
f"scale={zoomed_width}:{zoomed_height}:force_original_aspect_ratio=increase," f"{_fit_filter(cut_width, cut_height, zoom)},"
f"crop={cut_width}:{cut_height}:(iw-{cut_width})/2:(ih-{cut_height})/2,"
f"format=rgba[{seg_label}]" f"format=rgba[{seg_label}]"
) )
@@ -1206,8 +1235,7 @@ def build_filter_complex(
filters.append( filters.append(
f"[{video_idx}:v]format=yuva444p10le," f"[{video_idx}:v]format=yuva444p10le,"
f"setpts=PTS-STARTPTS+{start_pts:.3f}/TB," f"setpts=PTS-STARTPTS+{start_pts:.3f}/TB,"
f"scale={zoomed_width}:{zoomed_height}:force_original_aspect_ratio=increase," f"{_fit_filter(cut_width, cut_height, zoom, event.object_fit, event.object_position)},"
f"crop={cut_width}:{cut_height}:(iw-{cut_width})/2:(ih-{cut_height})/2,"
f"format=rgba[{video_label}]" f"format=rgba[{video_label}]"
) )
@@ -1263,8 +1291,7 @@ def build_filter_complex(
filters.append( filters.append(
f"[{video_idx}:v]format=rgba," f"[{video_idx}:v]format=rgba,"
f"setpts=PTS-STARTPTS+{start_pts:.3f}/TB," f"setpts=PTS-STARTPTS+{start_pts:.3f}/TB,"
f"scale={zoomed_width}:{zoomed_height}:force_original_aspect_ratio=increase," f"{_fit_filter(cut_width, cut_height, zoom, event.object_fit, event.object_position)}"
f"crop={cut_width}:{cut_height}:(iw-{cut_width})/2:(ih-{cut_height})/2"
f"[{video_label}]" f"[{video_label}]"
) )
@@ -1339,8 +1366,7 @@ def build_filter_complex(
filters.append( filters.append(
f"[{video_idx}:v]format=yuva444p10le," f"[{video_idx}:v]format=yuva444p10le,"
f"setpts=PTS-STARTPTS+{start_pts:.3f}/TB," f"setpts=PTS-STARTPTS+{start_pts:.3f}/TB,"
f"scale={zoomed_width}:{zoomed_height}:force_original_aspect_ratio=increase," f"{_fit_filter(cut_width, cut_height, zoom, event.object_fit, event.object_position)},"
f"crop={cut_width}:{cut_height}:(iw-{cut_width})/2:(ih-{cut_height})/2,"
f"format=rgba[{video_label}]" f"format=rgba[{video_label}]"
) )
+9 -1
View File
@@ -32,7 +32,7 @@ from .transformer import MarkerTiming, resolve_video_presentation
# Per-occurrence presentation fields ALWAYS materialized onto video events (atomic # Per-occurrence presentation fields ALWAYS materialized onto video events (atomic
# events.json, GUI-ready). Round-tripped as overrides so a stored value drives render. # events.json, GUI-ready). Round-tripped as overrides so a stored value drives render.
_PRESENTATION_KEYS = ("cutout", "layer", "end_on", "take") _PRESENTATION_KEYS = ("cutout", "layer", "end_on", "take", "object-fit", "object-position")
# `volume` is materialized SPARSELY — only when actually overridden (inline/GUI/manual), # `volume` is materialized SPARSELY — only when actually overridden (inline/GUI/manual),
# so the videos.json value keeps flowing as the default and only a real override pins it. # so the videos.json value keeps flowing as the default and only a real override pins it.
# Round-tripping still carries it whenever present in the event dict. # Round-tripping still carries it whenever present in the event dict.
@@ -94,6 +94,8 @@ def marker_type(marker_id: str, slides: dict, videos: dict, audio: dict) -> str:
return "audio" return "audio"
if _ci_contains(CAMERA_PRESETS, marker_id): if _ci_contains(CAMERA_PRESETS, marker_id):
return "camera" return "camera"
if marker_id.startswith("end:"):
return "end"
return "other" return "other"
@@ -192,6 +194,12 @@ def derive_events(
# videos.json default keeps flowing until someone pins it here. # videos.json default keeps flowing until someone pins it here.
if t.overrides and "volume" in t.overrides: if t.overrides and "volume" in t.overrides:
e["volume"] = pres["volume"] e["volume"] = pres["volume"]
# object-fit/position are sparse too: materialize only when non-default
# so plain center-cover videos keep a clean events.json.
if pres["object_fit"] != "cover":
e["object-fit"] = pres["object_fit"]
if pres["object_position"] != "center":
e["object-position"] = pres["object_position"]
pd = _pause_duration(t.marker_id, videos) pd = _pause_duration(t.marker_id, videos)
if pd: if pd:
e["pause_duration"] = pd e["pause_duration"] = pd
+47
View File
@@ -99,6 +99,12 @@ def resolve_video_presentation(
) )
# Volume defaults to the videos.json value; an events.json/inline override wins. # Volume defaults to the videos.json value; an events.json/inline override wins.
volume = overrides["volume"] if "volume" in overrides else video_source.volume volume = overrides["volume"] if "volume" in overrides else video_source.volume
# CSS-like cutout placement (hyphenated keys mirror CSS; inline/events override
# the videos.json default, which defaults to cover/center).
object_fit = overrides.get("object-fit") or video_source.object_fit or "cover"
object_position = (
overrides.get("object-position") or video_source.object_position or "center"
)
return { return {
"handle": handle, "handle": handle,
@@ -108,6 +114,8 @@ def resolve_video_presentation(
"take": take, "take": take,
"pause_narration": float(pause_narration or 0.0), "pause_narration": float(pause_narration or 0.0),
"volume": float(volume if volume is not None else 1.0), "volume": float(volume if volume is not None else 1.0),
"object_fit": object_fit,
"object_position": object_position,
} }
@@ -246,6 +254,11 @@ def _is_known_marker(
if audio_id in audio: if audio_id in audio:
return True return True
# Explicit end markers: [end:handle] stops a video started with end_on=end_marker.
# Known so it aligns to its spoken position (and isn't stripped as filler).
if marker_id.startswith("end:"):
return True
return False return False
@@ -1369,6 +1382,16 @@ def _extract_video_events(
# a clip when the next video begins, so videos never overlap. # a clip when the next video begins, so videos never overlap.
video_start_times = sorted(t for t, *_ in video_markers) video_start_times = sorted(t for t, *_ in video_markers)
# [end:handle] control markers: explicit end points for videos started with
# end_on=end_marker. Collected as {handle: sorted[timestamps]}. They are not a
# video prefix, so they never become video events themselves.
end_markers: dict[str, list[float]] = {}
for timing in marker_timings:
if timing.timestamp is not None and timing.timestamp >= 0 and timing.marker_id.startswith("end:"):
end_markers.setdefault(timing.marker_id[4:].lower(), []).append(timing.timestamp)
for _h in end_markers:
end_markers[_h].sort()
events: list[VideoEvent] = [] events: list[VideoEvent] = []
for start_time, marker_id, video_id, trigger_type, overrides in video_markers: for start_time, marker_id, video_id, trigger_type, overrides in video_markers:
video_source = videos[video_id] video_source = videos[video_id]
@@ -1389,6 +1412,8 @@ def _extract_video_events(
take = pres["take"] take = pres["take"]
pause_narration = pres["pause_narration"] pause_narration = pres["pause_narration"]
volume = pres["volume"] volume = pres["volume"]
object_fit = pres["object_fit"]
object_position = pres["object_position"]
if end_on == "take" and take is not None: if end_on == "take" and take is not None:
end_time = start_time + take end_time = start_time + take
@@ -1427,6 +1452,26 @@ def _extract_video_events(
# the narration is held for that long, so the overlay should match. # the narration is held for that long, so the overlay should match.
if pause_narration: if pause_narration:
end_time = max(end_time, start_time + pause_narration) end_time = max(end_time, start_time + pause_narration)
elif end_on == "end_marker":
# Explicit end: stop at the first [end:handle] placed after this clip
# starts (so the same handle can be reused in different sections).
ends = [t for t in end_markers.get(video_id, ()) if t > start_time]
if ends:
end_time = ends[0]
else:
# No matching [end:handle] — fall back to next_video and warn rather
# than silently running to the end of the render.
end_time = total_duration
for vt in video_start_times:
if vt > start_time:
end_time = vt
break
warnings.append(
f"[{marker_id}] end_on=end_marker but no [end:{video_id}] found "
f"after it — ending at the next video instead."
)
if pause_narration:
end_time = max(end_time, start_time + pause_narration)
else: else:
# end_on None ([narration:] with no explicit end) — runs to end. # end_on None ([narration:] with no explicit end) — runs to end.
end_time = total_duration end_time = total_duration
@@ -1462,6 +1507,8 @@ def _extract_video_events(
cutout_name=cutout_name, cutout_name=cutout_name,
layer=layer, layer=layer,
volume=volume, volume=volume,
object_fit=object_fit,
object_position=object_position,
skip_override=skip_override, skip_override=skip_override,
) )
) )
+19
View File
@@ -145,6 +145,18 @@ def validate_project(
if marker in ("pause", "stop"): if marker in ("pause", "stop"):
continue continue
# Explicit end markers: [end:handle] stops a video started with end_on=end_marker.
if marker.startswith("end:"):
handle = marker[4:].lower()
if handle not in videos:
warnings.append(
ValidationIssue(
f"[{marker}] ends a video, but '{handle}' isn't defined in videos.json.",
project_path / "manuscript.txt",
)
)
continue
# Unknown namespaced markers (e.g. [background:xxx]) — not supported, ignore with warning # Unknown namespaced markers (e.g. [background:xxx]) — not supported, ignore with warning
if ":" in marker: if ":" in marker:
warnings.append( warnings.append(
@@ -155,6 +167,13 @@ def validate_project(
) )
continue continue
# Only slide-shaped ids (S1, S54, …) are slide references. Other bare
# bracketed tokens are prose the author wrote, not markers — e.g. a vector
# "[1, 1, 1]" or "[2, 2, 2]" in the narration (the ", …" tail makes the regex
# read the leading number as a marker id). Don't flag those as missing slides.
if not (len(marker) > 1 and marker[0] in "Ss" and marker[1:].isdigit()):
continue
if marker not in slides: if marker not in slides:
issues.append( issues.append(
ValidationIssue( ValidationIssue(
+58
View File
@@ -0,0 +1,58 @@
"""[end:handle] explicit end markers: a video started with end_on=end_marker
stops at the first [end:handle] placed after it."""
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from gnommo.transformer import _extract_video_events, MarkerTiming
from gnommo.models import VideoSource, CutoutDefinition
def check(name, cond):
print(f" {'PASS' if cond else 'FAIL'} {name}")
assert cond, name
VIDEOS = {"fart": VideoSource(source_file="fart.mp4", cutout="fullscreen", layer="below")}
CUTOUTS = {"fullscreen": CutoutDefinition(x=0, y=0, height=1080, width=1920)}
def mt(mid, t, ov=None):
return MarkerTiming(mid, t, "text", 1.0, ov)
# 1. [vfb:fart, end_on=end_marker] @10 ; [end:fart] @25 → ends at 25
events, warns = _extract_video_events(
[mt("vfb:fart", 10.0, {"end_on": "end_marker"}), mt("end:fart", 25.0)],
VIDEOS, CUTOUTS, {}, 100.0,
)
check("[end:fart] is not itself a video event", len(events) == 1)
check("video starts at 10.0", abs(events[0].start_time - 10.0) < 1e-6)
check("video ends at the [end:fart] marker (25.0)", abs(events[0].end_time - 25.0) < 1e-6)
check("no warnings", not warns)
# 2. earliest [end:fart] AFTER the start wins; an earlier one is ignored (reuse handle)
events2, _ = _extract_video_events(
[
mt("end:fart", 5.0), # before start → ignored
mt("vfb:fart", 10.0, {"end_on": "end_marker"}),
mt("end:fart", 20.0), # first after start
mt("end:fart", 40.0),
],
VIDEOS, CUTOUTS, {}, 100.0,
)
check("uses first end marker after start (20.0)", abs(events2[0].end_time - 20.0) < 1e-6)
# 3. fallback: end_on=end_marker but no [end:fart] → next video + warning
videos3 = {**VIDEOS, "other": VideoSource(source_file="o.mp4", cutout="square")}
cutouts3 = {**CUTOUTS, "square": CutoutDefinition(x=0, y=0, height=864, width=864)}
events3, warns3 = _extract_video_events(
[mt("vfb:fart", 10.0, {"end_on": "end_marker"}), mt("vst:other", 30.0)],
videos3, cutouts3, {}, 100.0,
)
fart_ev = next(e for e in events3 if e.video_id == "fart")
check("fallback ends at next video (30.0)", abs(fart_ev.end_time - 30.0) < 1e-6)
check("warns about the missing [end:fart]", any("end_on=end_marker" in w for w in warns3))
print("\nAll [end:handle] tests passed.")
+42
View File
@@ -0,0 +1,42 @@
"""CSS-like cutout placement: object-fit (cover/contain) + object-position."""
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from gnommo.renderer import _fit_filter
def check(name, cond):
print(f" {'PASS' if cond else 'FAIL'} {name}")
assert cond, name
# Defaults reproduce the long-standing cover+center string exactly (no render churn).
check(
"cover/center == legacy scale+crop",
_fit_filter(864, 864, 1.0, "cover", "center")
== "scale=864:864:force_original_aspect_ratio=increase,crop=864:864:(iw-864)/2:(ih-864)/2",
)
check(
"cover applies zoom",
_fit_filter(864, 864, 1.5, "cover", "center").startswith("scale=1296:1296:"),
)
# cover anchors the crop by position.
check("cover/top crops from bottom (y=0)", ":(iw-864)/2:0" in _fit_filter(864, 864, 1.0, "cover", "top"))
check("cover/bottom (y=ih-H)", ":(iw-864)/2:(ih-864)" in _fit_filter(864, 864, 1.0, "cover", "bottom"))
check("cover/left (x=0)", "crop=864:864:0:(ih-864)/2" in _fit_filter(864, 864, 1.0, "cover", "left"))
check("cover/right (x=iw-W)", "crop=864:864:(iw-864):(ih-864)/2" in _fit_filter(864, 864, 1.0, "cover", "right"))
# contain shrinks to fit and pads; position places the padded video.
check(
"contain/top fits inside, pads to top",
_fit_filter(864, 864, 1.0, "contain", "top")
== "scale=864:864:force_original_aspect_ratio=decrease,pad=864:864:(ow-iw)/2:0:color=0x00000000",
)
check("contain ignores zoom", "scale=864:864:" in _fit_filter(864, 864, 2.0, "contain", "center"))
check("contain/bottom pads to bottom", ":(ow-iw)/2:(oh-ih):" in _fit_filter(864, 864, 1.0, "contain", "bottom"))
check("contain/left pads to left", "pad=864:864:0:(oh-ih)/2" in _fit_filter(864, 864, 1.0, "contain", "left"))
print("\nAll object-fit/object-position tests passed.")