"""
video_helper.main
=================
Implementation module for ``video_helper``: frame extraction across
three backends (VidGear / PyAV / ffmpeg-pipe), file-level conversion,
sparse / range / interval indexing, hwaccel routing, scale-fit-and-pad,
multi-destination yields (numpy / torch / pil), and small ffmpeg-shelled
helpers (chunk extraction, concatenation, subtitle burn-in, …).
Usage Example
-------------
>>> from video_helper.main import extract_frames, video_dimensions
>>> info = video_dimensions("clip.mp4")
>>> for frame in extract_frames("clip.mp4", frame_step=5):
... # frame.shape == (H, W, 3) — BGR uint8
... pass
The public surface is re-exported from ``video_helper`` (the package
``__init__``); prefer ``import video_helper as vh`` and call
``vh.extract_frames(...)`` instead of touching this module directly.
Author
------
Warith Harchaoui, Ph.D. — https://linkedin.com/in/warith-harchaoui/
"""
# ``from __future__ import annotations`` makes every annotation a string at
# runtime (PEP 563). It lets us reference forward / optional types (e.g.
# ``torch.Tensor``, ``av.VideoFrame``) in signatures without importing the
# heavy optional dependency at module import time — the annotation is only
# evaluated by tooling, never executed.
from __future__ import annotations
import os
import platform
import re
import shutil
import subprocess
from collections.abc import Iterator, Sequence
from typing import TYPE_CHECKING
import cv2
import ffmpeg
import numpy as np
import os_helper as osh
from vidgear.gears import VideoGear
# ``torch`` is an *optional* extra: import it only for type-checking so the
# ``torch.device`` / ``torch.Tensor`` annotations resolve for tooling, while
# runtime import stays lazy (inside the functions that need it).
if TYPE_CHECKING: # pragma: no cover — types only, never executed at runtime
import av
import torch
# Logging goes through os-helper (``osh.info/warning/error/debug``), the shared
# logging surface for the AI Helpers suite. It respects ``osh.verbosity(...)``
# so the embedding application controls output from one place — no bare
# ``print`` and no per-module stdlib logger (rule 6).
# File extensions we accept as "video". Kept as a plain list (not a set) so
# error messages can show a stable, human-ordered list; membership tests are
# case-normalized at the call site.
video_extensions: list[str] = [
"mp4",
"avi",
"mov",
"wmv",
"flv",
"mkv",
"webm",
"mpeg",
"mpg",
"m4v",
"3gp",
"ogv",
"mxf",
"ts",
"vob",
"m2ts",
"mts",
"rm",
"asf",
]
def _generate_css_from_colors(color_codes: set[str], css_file_path: str) -> None:
"""
Generate a CSS file with styles for each unique color code.
Parameters
----------
color_codes : Set[str]
A set of unique hex color codes.
css_file_path : str
Path to save the generated CSS file.
Usage
-----
>>> unique_colors = {'#FF0000', '#00FF00', '#0000FF'}
>>> css_file = "styles.css"
>>> _generate_css_from_colors(unique_colors, css_file)
"""
with open(css_file_path, "w", encoding="utf-8") as css_file:
for color in color_codes:
# Strip the '#' from the color code for the class name and use the hex code for styling
class_name = color.replace("#", "").lower()
css_file.write(f"::cue(.{class_name}) {{ color: {color}; }}\n")
osh.info(f"CSS file generated: {css_file_path}")
[docs]
def srt2vtt(srt_file_path: str, vtt_file_path: str = None, css_file_path: str = None) -> None:
"""
Convert an SRT subtitle file to WebVTT, preserving font colors and emitting a companion CSS file.
Any ``<font color="#RRGGBB">…</font>`` tag in the SRT is rewritten as a
WebVTT ``<c.<hex_lowercase>>…</c>`` cue class, and a stylesheet binding
each class to its color is written next to the VTT.
Parameters
----------
srt_file_path : str
Path to the input .srt file.
vtt_file_path : str, optional
Path to the output .vtt file. Defaults to ``<srt_stem>.vtt`` next
to the input.
css_file_path : str, optional
Path to the output .css file. Defaults to ``<srt_stem>.css`` next
to the input.
Examples
--------
>>> srt2vtt("subtitles.srt")
>>> srt2vtt("subtitles.srt", "out.vtt", "out.css")
"""
unique_colors = extract_unique_colors(srt_file_path)
folder, stem, _ = osh.folder_name_ext(srt_file_path)
if osh.emptystring(vtt_file_path):
vtt_file_path = osh.join([folder, stem + ".vtt"])
if osh.emptystring(css_file_path):
css_file_path = osh.join([folder, stem + ".css"])
_generate_css_from_colors(unique_colors, css_file_path)
def convert_color_tags(line: str) -> str:
"""Rewrite ``<font color="#RRGGBB">…</font>`` as ``<c.rrggbb>…</c>``."""
color_tag_pattern = re.compile(r'<font color="(#\w{6})">(.*?)<\/font>', re.IGNORECASE)
def replace_color(match: re.Match) -> str:
"""Turn one regex match of a ``<font color>`` tag into a VTT ``<c>``.
Parameters
----------
match : re.Match
Match with group 1 = ``#RRGGBB`` color, group 2 = inner text.
Returns
-------
str
``<c.rrggbb>text</c>`` — the WebVTT class-cue equivalent, with
the color used as a lowercase CSS class name (see the sidecar
CSS produced by ``_generate_css_from_colors``).
"""
# Normalize the hex to uppercase for the human-facing value, but
# derive the CSS class name in lowercase without the leading '#'.
color_code = match.group(1).upper()
text = match.group(2)
class_name = color_code.replace("#", "").lower()
return f"<c.{class_name}>{text}</c>"
return color_tag_pattern.sub(replace_color, line)
with open(srt_file_path, encoding="utf-8") as srt_file:
lines = srt_file.readlines()
with open(vtt_file_path, "w", encoding="utf-8") as vtt_file:
vtt_file.write("WEBVTT\n\n")
for line in lines:
line = convert_color_tags(line)
# WebVTT uses '.' for the fractional separator in timecodes; SRT uses ','.
if "-->" in line:
line = line.replace(",", ".")
vtt_file.write(line)
osh.info(f"WebVTT saved: {vtt_file_path}")
[docs]
def is_valid_video_file(video_file: str) -> bool:
"""
Check that ``video_file`` exists, has a known video extension, and contains a video stream.
Combines an extension check (against :data:`video_extensions`) with an
``ffprobe`` invocation so both a fake ``.mp4`` (no video stream) and a
real video renamed to ``.xyz`` are rejected.
HTTP / HTTPS URLs short-circuit to ``True``: the only way to truly
validate a remote URL is to spend bandwidth fetching part of the
stream, and ffmpeg will surface a clear error downstream if the URL
is bad. Callers passing a yt-dlp-resolved direct URL (with
``http_headers``) wouldn't even get past this gate otherwise.
Parameters
----------
video_file : str
Path to the input video file, or an HTTP / HTTPS URL.
Returns
-------
bool
True iff the file exists, ffprobe finds at least one video stream,
and the extension is in :data:`video_extensions` — OR
``video_file`` is an HTTP / HTTPS URL.
"""
# HTTP / HTTPS URLs are trusted: ffprobe / ffmpeg-on-pipe surface a
# clear error later if the URL is invalid, and there is no way to
# cheaply verify a remote stream's contents without downloading.
if video_file.startswith(("http://", "https://")):
return True
if not osh.file_exists(video_file):
osh.info(f"Video file not found: {video_file}")
return False
valid = False
try:
probe = ffmpeg.probe(video_file)
next(s for s in probe["streams"] if s["codec_type"] == "video")
valid = True
except Exception:
valid = False
_, _, ext = osh.folder_name_ext(video_file)
if ext.lower() not in video_extensions:
valid = False
osh.info(f"Video file {video_file} is {'valid' if valid else 'invalid'}")
return valid
[docs]
def video_dimensions(video_file: str, http_headers: dict | None = None) -> dict:
"""
Get the dimensions of a video file (or URL) using ``ffmpeg-python``.
Returned keys: ``width``, ``height``, ``duration``, ``frame_rate``,
``has_sound``.
Parameters
----------
video_file : str
Path to the input video file, OR an HTTP / HTTPS URL (e.g. a
yt-dlp-resolved direct media URL).
http_headers : dict[str, str], optional
HTTP headers (User-Agent, Referer, Cookie, …) forwarded to
ffprobe via ``-headers``. Required when ``video_file`` is a URL
that needs specific headers — e.g. yt-dlp-resolved YouTube live,
members-only, age-gated content. Ignored when ``video_file`` is
a local path.
Returns
-------
dict
``{"width": int, "height": int, "duration": float,
"frame_rate": float, "has_sound": bool}``.
Examples
--------
>>> d = video_dimensions("video.mp4")
>>> print(d)
{'width': 1920, 'height': 1080, 'duration': 10.0, 'frame_rate': 30.0, 'has_sound': True}
Notes
-----
Uses :func:`ffmpeg.probe` (a thin wrapper over ``ffprobe``) for
metadata extraction. ``http_headers`` are passed through ffprobe's
``-headers`` flag so URL-protected streams probe correctly.
"""
is_url = video_file.startswith(("http://", "https://"))
if not is_url:
osh.checkfile(video_file, msg=f"Video file not found: {video_file}")
# For URL inputs, splice the user's http_headers into ffprobe's
# input options. ffmpeg-python's ``probe`` accepts arbitrary kwargs
# which it converts into CLI flags.
probe_kwargs: dict = {}
if is_url and http_headers:
# ffmpeg expects a single CRLF-separated string before -i.
probe_kwargs["headers"] = "\r\n".join(f"{k}: {v}" for k, v in http_headers.items()) + "\r\n"
probe = ffmpeg.probe(video_file, **probe_kwargs)
video_info = next(s for s in probe["streams"] if s["codec_type"] == "video")
width = int(video_info["width"])
height = int(video_info["height"])
duration = float(video_info["duration"])
frame_rate = video_info["r_frame_rate"]
frame_rate = frame_rate.split("/")
frame_rate = float(frame_rate[0]) / float(frame_rate[1])
has_sound = any(s["codec_type"] == "audio" for s in probe["streams"])
d = {
"width": width,
"height": height,
"duration": duration,
"frame_rate": frame_rate,
"has_sound": has_sound,
}
return d
[docs]
def video_converter(
input_video: str,
output_video: str | None = None,
frame_rate: int | None = None,
width: int | None = None,
height: int | None = None,
without_sound: bool = False,
) -> None:
"""
Convert a video file to a new format with specified options.
Parameters
----------
input_video : str
Path to the input video file.
output_video : str
Path to the output video file.
frame_rate : int, optional
Frame rate of the output video file.
width : int, optional
Width of the output video file. If only width is specified, aspect ratio is maintained. If width is odd, it is reduced by 1 (ffmpeg reasons).
height : int, optional
Height of the output video file. If only height is specified, aspect ratio is maintained. If height is odd, it is reduced by 1 (ffmpeg reasons).
without_sound : bool, optional
Remove audio from the output video file.
Notes
-----
- The output video file will be in the same format as the input, unless an output file with a different format is specified.
Examples
--------
>>> video_converter("input.mp4", "output.mp4", frame_rate=30, width=640, height=480)
>>> video_converter("input.mp4", "output.mp4", without_sound=True)
"""
osh.info(f"Converting video file:\n\t{input_video}\ninto\n\t{output_video}")
# Check if the input video file exists and is valid
assert is_valid_video_file(input_video), f"Input video file not okay:\n\t{input_video}"
quiet = osh.verbosity() <= 0 # Determine verbosity level
# Extract folder and file extension details
fi, bi, input_ext = osh.folder_name_ext(input_video)
if osh.emptystring(output_video):
output_video = osh.join(fi, bi + "-converted" + "." + input_ext)
fo, bo, output_ext = osh.folder_name_ext(output_video)
# If no conversion is required, just copy the streams — but only when
# the container does not change. Cross-container stream copy (e.g.
# WebM/VP8 into .mp4) is rejected by ffmpeg because most codecs are
# not muxable in every container. In that case we transcode to
# H.264/AAC, which is the lingua franca every MP4-class container
# accepts.
if not frame_rate and not width and not height and not without_sound:
if input_ext.lower() == output_ext.lower():
ffmpeg.input(input_video).output(output_video, vcodec="copy", acodec="copy").run(
overwrite_output=True, quiet=quiet
)
else:
ffmpeg.input(input_video).output(
output_video,
vcodec="libx264",
acodec="aac",
pix_fmt="yuv420p",
).run(overwrite_output=True, quiet=quiet)
assert is_valid_video_file(output_video), f"Failed to convert video file:\n\t{output_video}"
osh.info(f"Video file converted successfully:\n{output_video}")
return
# Ensure width and height are even
if width and width % 2 != 0:
width -= 1
if height and height % 2 != 0:
height -= 1
# Use temporary files for conversion if needed
with (
osh.temporary_filename(suffix=".mp4", mode="wb") as temp_input,
osh.temporary_filename(suffix=".mp4", mode="wb") as temp_output,
):
# Normalize the input into an .mp4 container before the filter
# pass. Codec-copy only works when source codecs are already
# MP4-compatible (h264 + aac); for VP8/VP9/Opus/etc. we must
# transcode, otherwise ffmpeg refuses the cross-container mux.
if input_ext.lower() != "mp4":
transcode_kwargs = {"vcodec": "libx264", "pix_fmt": "yuv420p"}
if without_sound:
transcode_kwargs["an"] = None
else:
transcode_kwargs["acodec"] = "aac"
ffmpeg.input(input_video).output(temp_input, **transcode_kwargs).run(
overwrite_output=True, quiet=quiet
)
else:
osh.copyfile(input_video, temp_input) # Direct copy if already MP4
stream = ffmpeg.input(temp_input)
# Dictionary to store ffmpeg options
ffmpeg_options = {}
# Set the frame rate if specified
if frame_rate:
ffmpeg_options["r"] = frame_rate
# Apply video scaling and padding if needed
if width and height:
ffmpeg_options["vf"] = (
f"scale='min({width},iw*{height}/ih):min({height},ih*{width}/iw)',pad='{width}:{height}:(ow-iw)/2:(oh-ih)/2:black'"
)
elif width:
ffmpeg_options["vf"] = f"scale={width}:-1" # Maintain aspect ratio
elif height:
ffmpeg_options["vf"] = f"scale=-1:{height}" # Maintain aspect ratio
# Remove audio if specified
if without_sound:
ffmpeg_options["an"] = None # Disable audio stream
# Perform the conversion with the specified options
stream = stream.output(temp_output, **ffmpeg_options)
stream.run(overwrite_output=True, quiet=quiet)
# If the output format is not MP4, perform final conversion
if output_ext.lower() != "mp4":
ffmpeg.input(temp_output).output(output_video, vcodec="copy", acodec="copy").run(
overwrite_output=True, quiet=quiet
)
else:
osh.copyfile(temp_output, output_video) # Copy to final output if MP4
# Validate the final output video
assert is_valid_video_file(output_video), (
f"Failed to convert video file:\n\t{input_video}\ninto\n\t{output_video}"
)
# Retrieve video properties for validation
d = video_dimensions(output_video)
# Validate frame rate
if frame_rate:
error = round(100 * np.abs(d["frame_rate"] - frame_rate) / frame_rate)
assert error < 2, (
f"Failed to set frame rate for video file ({d['frame_rate']} vs {frame_rate}, error = {error}%):\n\t{output_video}"
)
# Validate width and height
if width:
assert d["width"] == width, f"Failed to set width for video file:\n\t{output_video}"
if height:
assert d["height"] == height, f"Failed to set height for video file:\n\t{output_video}"
# Validate sound removal
if without_sound:
assert not d["has_sound"], f"Failed to remove audio from video file:\n\t{output_video}"
osh.info(f"Video file converted successfully:\n\t{output_video}")
# ──────────────────────────────────────────────────────────────────────────
# Backend dispatcher for ``extract_frames``
#
# Three backends with distinct sweet spots, picked automatically (override
# via the ``backend=`` kwarg). See SPEED_ANALYSIS.md for the empirical
# evidence behind the routing rules.
#
# - ``vidgear`` → fastest path for **full sequential decode** on
# macOS (OpenCV+AVFoundation + worker thread); the
# only backend that supports ``stabilize=True``.
# Decodes from t=0 with no real seek, so it pays a
# large tax on windowed / sparse reads.
# - ``pyav`` → fastest for **windowed sequential** and **sparse**
# (frame_indices / frame_times) thanks to keyframe
# seek. Default for everything that isn't a full
# sequential read. Honors ``hwaccel``.
# - ``ffmpeg-pipe`` → ffmpeg subprocess fallback (true ``-ss``/``-to``
# seek). Used only when PyAV isn't installed.
#
# History: a decord backend was prototyped in v1.4 but removed before
# release — empirical benchmarks (see SPEED_ANALYSIS.md) showed PyAV
# beating decord by ~30% on its own sweet-spot (sparse access) while
# avoiding decord's ffmpeg@4 source-build install pain.
#
# All backends yield BGR uint8 ndarrays of shape (H, W, 3) for backward
# compatibility with the OpenCV / VidGear convention.
# ──────────────────────────────────────────────────────────────────────────
_BACKENDS = ("auto", "vidgear", "pyav", "ffmpeg-pipe")
def _resolve_hwaccel(hwaccel: str | None) -> str | None:
"""Translate ``hwaccel="auto"`` into a concrete value supported by the
locally-installed ffmpeg, or None when no acceleration is available.
A value of ``None`` (Python None) disables hwaccel — this is the
default for ``extract_frames`` because the SPEED_ANALYSIS benchmark
showed VideoToolbox hurting every scenario on a 426×426 H.264 clip
(the format-conversion overhead eats the decode win on small frames).
Worth re-evaluating on 4K HEVC content.
Any explicit string is returned as-is so the caller can opt into
something exotic without fighting the heuristic.
"""
if hwaccel != "auto":
return hwaccel
if shutil.which("ffmpeg") is None:
return None
try:
supported = subprocess.run(
["ffmpeg", "-hide_banner", "-hwaccels"],
capture_output=True,
text=True,
check=False,
).stdout
except Exception:
return None
system = platform.system().lower()
if system == "darwin" and "videotoolbox" in supported:
return "videotoolbox"
if "cuda" in supported and shutil.which("nvidia-smi") is not None:
return "cuda"
if "qsv" in supported:
return "qsv"
return None
def _have_pyav() -> bool:
"""Return whether the optional ``pyav`` extra is importable.
Returns
-------
bool
``True`` when ``import av`` succeeds, ``False`` otherwise. Used by the
backend dispatcher to decide whether the PyAV path is available.
"""
# Import-probe (same strategy as ``_have_torch`` / ``_have_pil``): a bare
# import is the most reliable availability signal and stays cheap.
try:
import av # noqa: F401
return True
except ImportError:
return False
def _choose_backend(
backend: str,
stabilize: bool,
sparse: bool,
full_sequential: bool,
) -> str:
"""Resolve ``backend="auto"`` against installed packages and call shape.
Routing rules (when ``backend="auto"``):
- ``stabilize=True`` → vidgear (forced; only one that supports it)
- sparse access (indices / times) → pyav if installed, else vidgear (range+filter fallback)
- full sequential (start=0, end=total)→ vidgear (4× faster than PyAV on macOS — see SPEED_ANALYSIS.md)
- windowed sequential → pyav if installed, else ffmpeg-pipe if ffmpeg on PATH, else vidgear
"""
if backend not in _BACKENDS:
raise ValueError(f"Unknown backend {backend!r}; expected one of {_BACKENDS}")
if stabilize:
if backend not in ("auto", "vidgear"):
raise ValueError(
f"stabilize=True requires the vidgear backend; got backend={backend!r}"
)
return "vidgear"
if backend != "auto":
return backend
if sparse:
# Sparse access: PyAV's keyframe-seek + PTS filter is the fastest
# option we ship (see SPEED_ANALYSIS.md). VidGear's no-seek loop
# is a last-resort fallback when PyAV is missing.
return "pyav" if _have_pyav() else "vidgear"
if full_sequential:
# Full sequential reads: VidGear (OpenCV+AVFoundation + worker
# thread) beats PyAV by 4× on macOS. No reason to give that up.
return "vidgear"
# Windowed sequential: PyAV's keyframe seek wins. ffmpeg-pipe is a
# backup when PyAV is missing; VidGear last because it has no real seek.
if _have_pyav():
return "pyav"
if shutil.which("ffmpeg") is not None:
return "ffmpeg-pipe"
return "vidgear"
def _resolve_indices(
duration: float,
frame_rate: float,
start_index: int | None,
end_index: int | None,
start_instant: float | None,
end_instant: float | None,
frame_step: int,
frame_interval: float | None,
frame_indices: Sequence[int] | None,
frame_times: Sequence[float] | None,
) -> tuple[list[int] | None, int, int, int, bool]:
"""Normalize the public range/sparse API into a single representation.
Returns
-------
indices : list[int] | None
Concrete indices when sparse access was requested, else None.
start_index, end_index : int
Inclusive bounds when sequential access was requested.
frame_step : int
Sampling stride (>=1).
sparse : bool
True iff the caller specified ``frame_indices`` / ``frame_times``.
"""
total = int(duration * frame_rate)
if frame_times is not None:
frame_indices = [int(round(t * frame_rate)) for t in frame_times]
if frame_indices is not None:
indices = sorted({int(i) for i in frame_indices if 0 <= int(i) < total})
return indices, 0, total - 1, 1, True
if start_instant is not None:
start_index = int(start_instant * frame_rate)
if end_instant is not None:
end_index = int(end_instant * frame_rate)
if start_index is None:
start_index = 0
if end_index is None:
end_index = total
if frame_interval is not None:
frame_step = max(1, int(frame_interval * frame_rate))
frame_step = max(1, int(frame_step))
assert 0 <= start_index <= end_index <= total, (
f"Invalid frame range:\n\t{start_index} "
f"({osh.time2str(1.0 * start_index / frame_rate)}) to {end_index} "
f"({osh.time2str(1.0 * end_index / frame_rate)}).\n"
f"It should be within 0 to {total} (for {osh.time2str(duration)} "
f"at {frame_rate} fps)"
)
return None, start_index, end_index, frame_step, False
def _extract_via_vidgear(
video_path: str,
start_index: int,
end_index: int,
frame_step: int,
stabilize: bool,
) -> Iterator[np.ndarray]:
"""Decode-everything-and-filter loop on top of VidGear / OpenCV.
Only path that supports software stabilization. Decodes from frame 0
even when ``start_index > 0`` (no real seek) — accept this cost when
stabilization is required or when faster backends aren't installed.
"""
stream = VideoGear(source=video_path, stabilize=stabilize).start()
current_index = 0
try:
while True:
frame = stream.read()
if frame is None:
break
if current_index < start_index:
current_index += 1
continue
if current_index <= end_index and (current_index - start_index) % frame_step == 0:
yield frame
if current_index > end_index:
break
current_index += 1
finally:
stream.stop()
def _join_http_headers(http_headers: dict | None) -> str | None:
"""ffmpeg / PyAV want a single CRLF-separated header string."""
if not http_headers:
return None
return "\r\n".join(f"{k}: {v}" for k, v in http_headers.items()) + "\r\n"
def _parse_pad_color(pad_color: str) -> tuple[int, int, int]:
"""Parse opaque ``pad_color`` into a BGR uint8 tuple.
Accepted values:
- common names: ``"black"`` / ``"white"`` / ``"red"`` / ``"green"`` / ``"blue"``
- hex: ``"#RRGGBB"``
- ``"transparent"`` → :class:`ValueError` (would require 4-channel
BGRA output, breaking the ``(H, W, 3)`` contract — planned for v1.6.0).
"""
name = pad_color.lower().strip()
if name == "transparent":
raise ValueError(
"pad_color='transparent' is not supported in v1.5.0 — it would "
"require 4-channel BGRA / RGBA output, breaking the (H, W, 3) "
"contract on every destination. Planned for v1.6.0."
)
named = {
"black": (0, 0, 0),
"white": (255, 255, 255),
"red": (0, 0, 255), # BGR
"green": (0, 255, 0),
"blue": (255, 0, 0), # BGR
"yellow": (0, 255, 255),
"cyan": (255, 255, 0),
"magenta": (255, 0, 255),
"gray": (128, 128, 128),
"grey": (128, 128, 128),
}
if name in named:
return named[name]
if name.startswith("#") and len(name) == 7:
try:
r = int(name[1:3], 16)
g = int(name[3:5], 16)
b = int(name[5:7], 16)
return (b, g, r)
except ValueError:
pass
raise ValueError(
f"Unknown pad_color {pad_color!r}; expected one of "
f"{sorted(named)} / '#RRGGBB' / 'transparent' (transparent → v1.6.0)"
)
def _apply_output_transform(
frame: np.ndarray,
output_width: int | None,
output_height: int | None,
pad_color_bgr: tuple[int, int, int],
) -> np.ndarray:
"""Resize a BGR frame to (``output_width``, ``output_height``).
Behavior:
- Both dimensions set → scale-fit (preserve aspect ratio) then pad
with ``pad_color_bgr`` so the output is exactly the requested size.
- Only one dimension set → scale to that dimension preserving aspect
ratio (the other dimension is derived; no pad).
- Neither set → return frame unchanged.
Uses ``cv2.INTER_AREA`` when downscaling (sharper for downsizing)
and ``cv2.INTER_LINEAR`` when upscaling (cheap, no ringing).
"""
if output_width is None and output_height is None:
return frame
h, w = frame.shape[:2]
if output_width is not None and output_height is not None:
scale = min(output_width / w, output_height / h)
new_w = max(1, int(round(w * scale)))
new_h = max(1, int(round(h * scale)))
interp = cv2.INTER_AREA if scale < 1.0 else cv2.INTER_LINEAR
scaled = cv2.resize(frame, (new_w, new_h), interpolation=interp)
pad_top = (output_height - new_h) // 2
pad_bottom = output_height - new_h - pad_top
pad_left = (output_width - new_w) // 2
pad_right = output_width - new_w - pad_left
return cv2.copyMakeBorder(
scaled,
pad_top,
pad_bottom,
pad_left,
pad_right,
cv2.BORDER_CONSTANT,
value=list(pad_color_bgr),
)
if output_width is not None:
scale = output_width / w
new_h = max(1, int(round(h * scale)))
interp = cv2.INTER_AREA if scale < 1.0 else cv2.INTER_LINEAR
return cv2.resize(frame, (output_width, new_h), interpolation=interp)
# output_height is not None
scale = output_height / h
new_w = max(1, int(round(w * scale)))
interp = cv2.INTER_AREA if scale < 1.0 else cv2.INTER_LINEAR
return cv2.resize(frame, (new_w, output_height), interpolation=interp)
def _extract_via_pyav(
video_path: str,
start_index: int,
end_index: int,
frame_step: int,
sparse_indices: Sequence[int] | None,
frame_rate: float,
hwaccel: str | None,
http_headers: dict | None = None,
) -> Iterator[np.ndarray]:
"""PyAV-based decode with keyframe seek and optional hardware accel.
Notes
-----
PyAV's ``container.seek`` accepts an offset in ``AV_TIME_BASE`` units
(microseconds) when no ``stream`` is passed. We rely on that and
recover the exact frame index from ``frame.pts * stream.time_base``,
so a coarse keyframe seek is fine — we just drop everything before the
requested index.
Hardware acceleration is wired through ``av.codec.hwaccel.HWAccel``
(not the format-context ``options=`` kwarg, which is silently ignored
for hwaccel — that bug existed in v1.4.0-dev and inflated all
``hwaccel="auto"`` cells in SPEED_ANALYSIS.md to be no-ops).
"""
import av # lazy
# HTTP headers (User-Agent / Referer / Cookie / Authorization) are
# fed to libavformat via the AVFormatContext options dict — that's
# the right place for them (unlike hwaccel, which the same kwarg
# silently ignores; see _extract_via_pyav notes above).
open_options: dict | None = None
headers_str = _join_http_headers(http_headers)
if headers_str:
open_options = {"headers": headers_str}
if hwaccel:
try:
hw = av.codec.hwaccel.HWAccel(device_type=hwaccel)
container = (
av.open(video_path, hwaccel=hw, options=open_options)
if open_options
else av.open(video_path, hwaccel=hw)
)
except (av.error.ValueError, ValueError) as exc:
osh.warning(
"PyAV hwaccel=%r unavailable (%s); falling back to software decode",
hwaccel,
exc,
)
container = (
av.open(video_path, options=open_options) if open_options else av.open(video_path)
)
else:
container = (
av.open(video_path, options=open_options) if open_options else av.open(video_path)
)
try:
stream = container.streams.video[0]
stream.thread_type = "AUTO"
def _seek_to_seconds(seconds: float) -> None:
"""Seek the container to the keyframe at-or-before ``seconds``.
Parameters
----------
seconds : float
Target position in seconds (clamped to >= 0).
"""
# AV_TIME_BASE is 1 µs; convert seconds → microseconds. ``backward``
# lands on the nearest preceding keyframe so decoding stays valid.
offset_us = max(0, int(seconds * 1_000_000))
container.seek(offset_us, any_frame=False, backward=True)
def _index_of(frame: av.VideoFrame) -> int:
"""Map a decoded frame's PTS to its integer frame index.
Parameters
----------
frame : av.VideoFrame
Decoded PyAV frame.
Returns
-------
int
Zero-based frame index, or ``-1`` when the frame carries no PTS
(rare, but possible on some malformed streams).
"""
# No presentation timestamp → we cannot place the frame; signal -1
# so the caller skips it rather than mis-indexing the stream.
if frame.pts is None:
return -1
# PTS is in stream time-base units; scale to seconds then to a
# frame index at the stream's frame rate.
return int(round(float(frame.pts * stream.time_base) * frame_rate))
if sparse_indices is not None and len(sparse_indices) > 0:
wanted = sorted(set(sparse_indices))
wanted_set = set(wanted)
_seek_to_seconds(wanted[0] / frame_rate)
for frame in container.decode(stream):
index = _index_of(frame)
if index in wanted_set:
yield frame.to_ndarray(format="bgr24")
wanted_set.discard(index)
if not wanted_set:
break
return
# Sequential range: seek to a keyframe at-or-before start_index,
# then PTS-filter to the exact bounds.
if start_index > 0:
_seek_to_seconds(start_index / frame_rate)
for frame in container.decode(stream):
index = _index_of(frame)
if index < start_index:
continue
if index > end_index:
break
if (index - start_index) % frame_step == 0:
yield frame.to_ndarray(format="bgr24")
finally:
container.close()
def _extract_via_ffmpeg_pipe(
video_path: str,
start_index: int,
end_index: int,
frame_step: int,
frame_rate: float,
width: int,
height: int,
hwaccel: str | None,
http_headers: dict | None = None,
) -> Iterator[np.ndarray]:
"""ffmpeg subprocess with -ss/-to true seek and raw bgr24 over a pipe.
Sequential only (no sparse). Useful when PyAV is unavailable but
ffmpeg is. Hwaccel is honored when supported by the local build.
HTTP headers (``http_headers``) are passed via ``-headers`` and
placed before ``-i`` so they reach the input demuxer.
"""
start_s = start_index / frame_rate
end_s = (end_index + 1) / frame_rate
cmd = ["ffmpeg", "-hide_banner", "-loglevel", "error", "-nostdin"]
if hwaccel:
cmd += ["-hwaccel", hwaccel]
headers_str = _join_http_headers(http_headers)
if headers_str:
cmd += ["-headers", headers_str]
cmd += ["-ss", f"{start_s:.6f}", "-to", f"{end_s:.6f}", "-i", video_path]
if frame_step > 1:
# Sample every Nth frame after the seek.
cmd += ["-vf", f"select=not(mod(n\\,{frame_step}))", "-vsync", "vfr"]
cmd += ["-f", "rawvideo", "-pix_fmt", "bgr24", "-"]
frame_size = width * height * 3
proc = subprocess.Popen(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
try:
while True:
raw = proc.stdout.read(frame_size)
if not raw or len(raw) < frame_size:
break
yield np.frombuffer(raw, dtype=np.uint8).reshape((height, width, 3)).copy()
finally:
if proc.poll() is None:
proc.terminate()
try:
proc.wait(timeout=2.0)
except subprocess.TimeoutExpired:
proc.kill()
proc.wait()
# Drain stderr for diagnostics.
err = proc.stderr.read().decode("utf-8", errors="replace").strip()
if err and proc.returncode not in (0, None):
osh.warning("ffmpeg stderr: %s", err)
# ──────────────────────────────────────────────────────────────────────────
# Destination converter — yields frames in the user's preferred form
# with the **conventional** colorspace and axis layout for that framework.
#
# Shapes & colorspaces (the cheat sheet)
# ----------------------------------------
# Notation: N = batch size, T = time (frames in a video), C = channels
# (always 3), H = height, W = width.
#
# destination="numpy" (OpenCV-compatible: BGR uint8, channels last)
# ┌──────────┬─────────────┬────────────────────────┐
# │ layout │ batch_size │ shape (BGR uint8) │
# ├──────────┼─────────────┼────────────────────────┤
# │ any │ None │ (H, W, 3) HWC │
# │ image │ N │ (N, H, W, 3) NHWC │
# │ video │ N │ (N, H, W, 3) THWC │ (same memory as NHWC; T == N)
# └──────────┴─────────────┴────────────────────────┘
#
# destination="torch" (PyTorch-conventional: RGB uint8, channels first)
# ┌──────────┬─────────────┬────────────────────────┐
# │ layout │ batch_size │ shape (RGB uint8) │
# ├──────────┼─────────────┼────────────────────────┤
# │ any │ None │ (3, H, W) CHW │
# │ image │ N │ (N, 3, H, W) NCHW │ (batch of independent images)
# │ video │ N │ (3, N, H, W) CTHW │ (single video clip, T == N)
# └──────────┴─────────────┴────────────────────────┘
#
# destination="pil" (PIL/Pillow convention: RGB, size = (W, H), no batch)
# Each yield is a single ``PIL.Image.Image`` with ``mode="RGB"`` and
# ``size = (W, H)``. ``batch_size`` is not supported (Pillow has no
# batched image type); ``layout`` is ignored. Pillow is imported lazily.
# Use this destination for code that calls PIL/Pillow methods (filters,
# paste, draw, …); otherwise numpy or torch are cheaper.
#
# Notes:
# - ``layout`` only matters when ``batch_size`` is set (for numpy/torch).
# Without batching, each yield is a single frame (HWC numpy / CHW
# torch / PIL image) regardless.
# - For numpy the layout choice is purely *semantic* (NHWC and THWC
# share the same memory). For torch it's a real permutation
# (NCHW vs CTHW differ in axis order).
# - The BGR→RGB conversion for torch happens via ``tensor.flip(-1)``
# on the channel axis (cheap, view-style until ``.contiguous()``).
# - When ``destination`` is ``"torch"`` / ``"pil"``, the corresponding
# library is imported lazily — video-helper itself does NOT take torch
# or Pillow as a dependency. Install via the ``[torch]`` / ``[pil]``
# extras (or bring your own).
# ──────────────────────────────────────────────────────────────────────────
def _have_torch() -> bool:
"""Return whether the optional ``torch`` extra is importable.
Returns
-------
bool
``True`` when ``import torch`` succeeds, ``False`` otherwise. Used to
gate ``destination="torch"`` without hard-depending on torch.
"""
# Probe by import: cheaper and more reliable than inspecting metadata,
# and it also catches broken installs (present but not importable).
try:
import torch # noqa: F401
return True
except ImportError:
return False
def _have_pil() -> bool:
"""Return whether the optional ``pillow`` extra is importable.
Returns
-------
bool
``True`` when ``import PIL.Image`` succeeds, ``False`` otherwise. Gates
``destination="pil"``.
"""
# Same import-probe strategy as ``_have_torch`` — keep the two in sync.
try:
import PIL.Image # noqa: F401
return True
except ImportError:
return False
def _resolve_torch_device(device: str) -> torch.device:
"""Translate ``device`` ("auto" / "cpu" / "mps" / "cuda") into a torch.device.
"auto" → cuda if available, else mps if available, else cpu.
Parameters
----------
device : str
Requested device string, or ``"auto"`` for best-available.
Returns
-------
torch.device
Concrete torch device object.
"""
import torch # lazy — only imported when a torch destination is requested
if device == "auto":
if torch.cuda.is_available():
return torch.device("cuda")
if hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
return torch.device("mps")
return torch.device("cpu")
return torch.device(device)
def _bgr_hwc_to_torch_chw_rgb(np_frame: np.ndarray, device: torch.device) -> torch.Tensor:
"""Convert one numpy HWC BGR uint8 frame to a torch CHW RGB uint8 tensor.
Parameters
----------
np_frame : numpy.ndarray
Single frame ``(H, W, 3)``, BGR uint8 (OpenCV convention).
device : torch.device
Destination device for the resulting tensor.
Returns
-------
torch.Tensor
Tensor ``(3, H, W)``, RGB uint8, on ``device``.
"""
import torch # lazy
# flip(-1): reverse the channel axis (BGR → RGB).
# permute(2, 0, 1): HWC → CHW.
# contiguous(): force a single copy so downstream ops don't trip on a
# non-contiguous tensor; the .to(device) call below would do its own
# copy anyway, so this is essentially free.
return torch.from_numpy(np_frame).flip(-1).permute(2, 0, 1).contiguous().to(device)
def _bgr_to_torch_image_batch(np_batch_nhwc: np.ndarray, device: torch.device) -> torch.Tensor:
"""Convert numpy NHWC BGR uint8 batch → torch NCHW RGB uint8 on device.
Parameters
----------
np_batch_nhwc : numpy.ndarray
Batch ``(N, H, W, 3)``, BGR uint8.
device : torch.device
Destination device.
Returns
-------
torch.Tensor
Tensor ``(N, 3, H, W)``, RGB uint8, on ``device``.
"""
import torch
return (
torch.from_numpy(np_batch_nhwc)
.flip(-1) # NHWC BGR → NHWC RGB
.permute(0, 3, 1, 2) # NHWC → NCHW
.contiguous()
.to(device)
)
def _bgr_to_torch_video_clip(np_clip_thwc: np.ndarray, device: torch.device) -> torch.Tensor:
"""Convert numpy THWC BGR uint8 clip → torch CTHW RGB uint8 on device.
Parameters
----------
np_clip_thwc : numpy.ndarray
Clip ``(T, H, W, 3)``, BGR uint8.
device : torch.device
Destination device.
Returns
-------
torch.Tensor
Tensor ``(3, T, H, W)`` (CTHW), RGB uint8, on ``device``.
"""
import torch
return (
torch.from_numpy(np_clip_thwc)
.flip(-1) # THWC BGR → THWC RGB
.permute(3, 0, 1, 2) # THWC → CTHW
.contiguous()
.to(device)
)
def _to_destination(
np_frames: Iterator[np.ndarray],
destination: str,
device: str,
batch_size: int | None,
layout: str,
) -> Iterator[object]:
"""Convert/batch the upstream numpy-frame iterator into the requested destination.
See the cheat-sheet at the top of this section for shapes / colorspaces.
Parameters
----------
np_frames : Iterator[numpy.ndarray]
Upstream HWC BGR uint8 frames.
destination : str
One of ``"numpy"``, ``"torch"``, ``"pil"``.
device : str
Torch device string (only used when ``destination == "torch"``).
batch_size : int or None
Batch size for stacked yields, or ``None`` for one item at a time.
layout : str
``"image"`` (per-frame / NCHW) or ``"video"`` (clip / CTHW) batching.
Yields
------
object
numpy arrays, torch tensors, or PIL images depending on
``destination`` — the concrete type is documented in the section
cheat-sheet above (kept as ``object`` here because the yield type is
genuinely destination-dependent).
"""
if destination not in ("numpy", "torch", "pil"):
raise ValueError(
f"Unknown destination {destination!r}; expected 'numpy', 'torch', or 'pil'"
)
if destination == "pil" and batch_size is not None:
raise ValueError(
"destination='pil' does not support batch_size — Pillow has no "
"batched image type. Iterate per-frame, or use destination='numpy' "
"/ 'torch' for batched tensors."
)
if layout not in ("image", "video"):
raise ValueError(f"Unknown layout {layout!r}; expected 'image' or 'video'")
if batch_size is not None and batch_size < 1:
raise ValueError(f"batch_size must be >= 1, got {batch_size}")
# ------------ destination="numpy" -----------------------------------
if destination == "numpy":
if batch_size is None:
# HWC BGR uint8 — pass-through, no copy.
yield from np_frames
return
# NHWC == THWC for numpy (only the semantic name differs); we stack
# along axis 0 regardless of layout.
batch: list = []
for frame in np_frames:
batch.append(frame)
if len(batch) == batch_size:
yield np.stack(batch, axis=0)
batch = []
if batch:
yield np.stack(batch, axis=0)
return
# ------------ destination="pil" -------------------------------------
if destination == "pil":
if not _have_pil():
raise ImportError(
"destination='pil' requires Pillow. Install with: "
"pip install 'video-helper[pil]' (or bring your own PIL)"
)
from PIL import Image # lazy
for frame in np_frames:
# Flip BGR → RGB before handing to PIL (which is RGB-native).
yield Image.fromarray(frame[:, :, ::-1])
return
# ------------ destination="torch" -----------------------------------
if not _have_torch():
raise ImportError(
"destination='torch' requires PyTorch. Install with: pip install 'video-helper[torch]' "
"(or bring your own torch)"
)
dev = _resolve_torch_device(device)
if batch_size is None:
# CHW RGB uint8 per yielded frame. layout is irrelevant here
# (each yield is a single frame, no time / batch axis).
for frame in np_frames:
yield _bgr_hwc_to_torch_chw_rgb(frame, dev)
return
# Batched: layout chooses the axis convention.
batch = []
for frame in np_frames:
batch.append(frame)
if len(batch) == batch_size:
stacked = np.stack(batch, axis=0) # NHWC == THWC, BGR
if layout == "image":
yield _bgr_to_torch_image_batch(stacked, dev) # NCHW RGB
else: # "video"
yield _bgr_to_torch_video_clip(stacked, dev) # CTHW RGB
batch = []
if batch:
stacked = np.stack(batch, axis=0)
if layout == "image":
yield _bgr_to_torch_image_batch(stacked, dev)
else:
yield _bgr_to_torch_video_clip(stacked, dev)
[docs]
def dump_frames(frames_list: list[np.ndarray], output_movie: str, fps: int = 30) -> None:
"""
Save frames to a video file.
Parameters
----------
frames_list : List[np.ndarray]
A list of frames as numpy arrays.
output_movie : str
Path to the output video file.
fps : int, optional
Frame rate of the output video file. Defaults to 30.
Notes
-----
The function uses VidGear to write the frames to a video file.
Usage
-----
>>> frames = [frame1, frame2, frame3]
>>> dump_frames(frames, "output.mp4")
"""
assert len(frames_list) > 0, "No frames to dump!"
height, width, channels = frames_list[0].shape
assert all(frame.shape == (height, width, channels) for frame in frames_list), (
"Frames do not have consistent dimensions!"
)
_, _, output_ext = osh.folder_name_ext(output_movie)
quiet = osh.verbosity() <= 0
with osh.temporary_folder() as temp_folder:
try:
frame_pattern = osh.join([temp_folder, "frame_%09d.png"])
for i, frame in enumerate(frames_list):
frame_path = frame_pattern % i
frame_bgr = cv2.cvtColor(frame, cv2.COLOR_RGB2BGR) # BGR convention of opencv
cv2.imwrite(frame_path, frame_bgr)
temp_movie = osh.join([temp_folder, "temp_movie.mp4"])
ffmpeg.input(frame_pattern, framerate=fps).output(temp_movie).run(
overwrite_output=True, quiet=quiet
)
if output_ext.lower() != "mp4":
ffmpeg.input(temp_movie).output(output_movie, vcodec="copy", acodec="copy").run(
overwrite_output=True, quiet=quiet
)
else:
osh.copyfile(temp_movie, output_movie)
except Exception as e:
# Chain the original exception (``from e``) so the traceback keeps
# the ffmpeg-level cause instead of masking it (ruff B904).
raise Exception(f"Error occurred while dumping frames to video: {e}") from e
osh.info(f"Video saved successfully: {output_movie}")
# ──────────────────────────────────────────────────────────────────────────
# Pipeline helpers — composition primitives shared by video editors that
# need to glue clips, generate stand-ins (solid color, looped still),
# overlay graphics with time-varying expressions, burn subtitles, mux a
# voice track, or read a duration without re-implementing ffprobe glue.
# ──────────────────────────────────────────────────────────────────────────
[docs]
def video_duration(input_video: str) -> float:
"""
Return the duration (seconds) of a video file.
Parameters
----------
input_video : str
Path to the input video file.
Returns
-------
float
Duration in seconds.
Notes
-----
Mirror of ``audio_helper.get_audio_duration`` for the video side.
Uses ``video_dimensions`` under the hood, which already calls
``ffmpeg.probe`` — kept as a top-level convenience so callers don't
have to remember the dict key.
Examples
--------
>>> video_duration("clip.mp4")
12.34
"""
osh.checkfile(input_video, msg=f"Video file not found: {input_video}")
return float(video_dimensions(input_video)["duration"])
[docs]
def black_video(
duration: float,
width: int,
height: int,
output_video: str,
frame_rate: int = 30,
) -> None:
"""
Generate a silent solid-black video of ``duration`` seconds.
Parameters
----------
duration : float
Output duration in seconds.
width : int
Output frame width in pixels (rounded down to even — H.264 yuv420p
requires even dimensions).
height : int
Output frame height in pixels (rounded down to even).
output_video : str
Path to the output video file (.mp4 recommended).
frame_rate : int, optional
Output frame rate (default 30).
Notes
-----
Useful as a "buffer" / breathing clip between two visuals in a montage,
or as a placeholder when an asset is missing. Encoded as H.264 yuv420p
with no audio track.
Examples
--------
>>> black_video(0.5, 1920, 1080, "buffer.mp4")
"""
assert duration > 0, f"black_video duration must be > 0, got {duration}"
assert width > 0 and height > 0, f"black_video needs positive dims, got {width}x{height}"
# H.264 with yuv420p requires even width/height; drop the last odd pixel
# rather than fail the encode on a user-supplied odd dimension.
if width % 2:
width -= 1
if height % 2:
height -= 1
quiet = osh.verbosity() <= 0
(
ffmpeg.input(f"color=c=black:s={width}x{height}:r={frame_rate}", f="lavfi", t=duration)
.output(output_video, vcodec="libx264", pix_fmt="yuv420p", preset="medium", crf=20, an=None)
.run(overwrite_output=True, quiet=quiet)
)
assert is_valid_video_file(output_video), f"Failed to write black_video:\n\t{output_video}"
[docs]
def image_loop_to_video(
image: str,
duration: float,
output_video: str,
frame_rate: int = 30,
width: int = None,
height: int = None,
) -> None:
"""
Loop a still image for ``duration`` seconds into a silent video.
Parameters
----------
image : str
Path to the input still (PNG, JPG, …).
duration : float
Output duration in seconds.
output_video : str
Path to the output video file (.mp4 recommended).
frame_rate : int, optional
Output frame rate (default 30).
width, height : int, optional
If both provided, the image is letterboxed (scale + pad with black)
to the target viewport. Width and height are rounded down to even.
Notes
-----
Common in title cards, screenshot scenes, slide-style montages.
Encoded as H.264 yuv420p with no audio track.
Examples
--------
>>> image_loop_to_video("title.png", 3.0, "title.mp4",
... width=1920, height=1080)
"""
osh.checkfile(image, msg=f"Image not found: {image}")
assert duration > 0, f"image_loop_to_video duration must be > 0, got {duration}"
quiet = osh.verbosity() <= 0
stream = ffmpeg.input(image, loop=1, framerate=frame_rate, t=duration)
out_kwargs = {
"vcodec": "libx264",
"pix_fmt": "yuv420p",
"preset": "medium",
"crf": 20,
"an": None,
"t": duration,
}
# H.264 with yuv420p needs even dimensions; nudge any odd target down by
# one pixel so the encoder never rejects the requested size.
if width and height:
if width % 2:
width -= 1
if height % 2:
height -= 1
out_kwargs["vf"] = (
f"scale='min({width},iw*{height}/ih):min({height},ih*{width}/iw)',"
f"pad='{width}:{height}:(ow-iw)/2:(oh-ih)/2:black',"
f"format=yuv420p,fps={frame_rate}"
)
elif width:
if width % 2:
width -= 1
out_kwargs["vf"] = f"scale={width}:-1,format=yuv420p,fps={frame_rate}"
elif height:
if height % 2:
height -= 1
out_kwargs["vf"] = f"scale=-1:{height},format=yuv420p,fps={frame_rate}"
else:
out_kwargs["vf"] = f"format=yuv420p,fps={frame_rate}"
stream.output(output_video, **out_kwargs).run(overwrite_output=True, quiet=quiet)
assert is_valid_video_file(output_video), (
f"Failed to write image_loop_to_video:\n\t{output_video}"
)
[docs]
def concat_videos(
input_videos: list[str],
output_video: str,
reencode: bool = True,
frame_rate: int = None,
) -> None:
"""
Concatenate ``input_videos`` end-to-end into ``output_video``.
Parameters
----------
input_videos : List[str]
Ordered list of input video paths.
output_video : str
Path to the output video file (.mp4 recommended).
reencode : bool, optional
Whether to re-encode (libx264). Default ``True`` — strongly
recommended when the inputs come from different sources, since
the concat demuxer's stream-copy path requires identical codec,
timebase, frame rate and resolution; mismatched inputs produce
audio/video drift or hard ffmpeg errors. Set ``False`` only when
the inputs are guaranteed bit-identical containers.
frame_rate : int, optional
Force this output frame rate (only used when ``reencode=True``).
Notes
-----
Uses the ffmpeg ``concat`` *demuxer* (text manifest) which is the
only correct way to concatenate variable-length clips end-to-end
without re-timing artefacts. The temporary manifest is written to a
process-temp file and removed automatically.
Examples
--------
>>> concat_videos(["intro.mp4", "body.mp4", "outro.mp4"], "final.mp4")
"""
assert len(input_videos) > 0, "concat_videos: empty input list"
for v in input_videos:
osh.checkfile(v, msg=f"Input video not found: {v}")
quiet = osh.verbosity() <= 0
with osh.temporary_filename(suffix=".txt", mode="w") as manifest:
with open(manifest, "w") as fh:
for v in input_videos:
# ffmpeg concat demuxer: single-quote the path, escape inner quotes
p = os.path.abspath(v).replace("'", r"'\''")
fh.write(f"file '{p}'\n")
stream = ffmpeg.input(manifest, format="concat", safe=0)
out_kwargs = {}
if reencode:
out_kwargs.update(
{
"vcodec": "libx264",
"pix_fmt": "yuv420p",
"preset": "medium",
"crf": 20,
"an": None,
}
)
if frame_rate:
out_kwargs["r"] = frame_rate
else:
out_kwargs.update({"vcodec": "copy", "acodec": "copy"})
stream.output(output_video, **out_kwargs).run(
overwrite_output=True,
quiet=quiet,
)
assert is_valid_video_file(output_video), f"Failed to write concat_videos:\n\t{output_video}"
[docs]
def overlay_image(
input_video: str,
image: str,
output_video: str,
x: str = "0",
y: str = "0",
scale_width: int = None,
) -> None:
"""
Overlay a still image (PNG with alpha works) on top of a video.
Parameters
----------
input_video : str
Path to the base video.
image : str
Path to the overlay image (PNG with alpha is the typical case —
cursors, watermarks, logos).
output_video : str
Path to the output video file (.mp4 recommended).
x, y : str, optional
Overlay positions. Plain integers (``"10"``) place the image
statically; ffmpeg overlay expressions (``"if(lt(t,1.0),0,100)"``,
``"W/2-w/2"``, …) move the overlay over time. Default ``"0"``,
``"0"`` (top-left).
scale_width : int, optional
If provided, scale the overlay to this width keeping aspect ratio
— useful for cursor PNGs that come at a different size than the
target frame.
Notes
-----
Time-varying expressions are evaluated per-frame
(``eval=frame``) so animations stay smooth at any framerate. The
underlying video stream is re-encoded (libx264) and the original
audio track, if any, is preserved.
Examples
--------
>>> overlay_image("clip.mp4", "cursor.png", "out.mp4",
... x="if(lt(t,2),100,400)", y="200",
... scale_width=24)
"""
osh.checkfile(input_video, msg=f"Input video not found: {input_video}")
osh.checkfile(image, msg=f"Overlay image not found: {image}")
quiet = osh.verbosity() <= 0
in_v = ffmpeg.input(input_video)
in_img = ffmpeg.input(image)
if scale_width:
in_img = in_img.filter("scale", scale_width, -1)
overlaid = ffmpeg.overlay(in_v.video, in_img, x=x, y=y, eval="frame", format="auto")
out_kwargs = {"vcodec": "libx264", "pix_fmt": "yuv420p", "preset": "medium", "crf": 20}
# Preserve the original audio stream if present (probe lazily).
has_audio = video_dimensions(input_video).get("has_sound", False)
if has_audio:
out = ffmpeg.output(overlaid, in_v.audio, output_video, acodec="copy", **out_kwargs)
else:
out = ffmpeg.output(overlaid, output_video, an=None, **out_kwargs)
out.run(overwrite_output=True, quiet=quiet)
assert is_valid_video_file(output_video), f"Failed to write overlay_image:\n\t{output_video}"
[docs]
def mux_audio_video(
input_video: str,
input_audio: str,
output_video: str,
audio_codec: str = "aac",
audio_bitrate: str = "192k",
shortest: bool = False,
) -> None:
"""
Mux a separate audio track onto a (typically silent) video.
Parameters
----------
input_video : str
Path to the video file. Any existing audio track is replaced.
input_audio : str
Path to the audio file (WAV, MP3, AAC, …).
output_video : str
Path to the output video file (.mp4 recommended).
audio_codec : str, optional
Audio codec for the output stream (default ``"aac"``). Use
``"copy"`` if the input audio is already in a container-compatible
codec.
audio_bitrate : str, optional
Audio bitrate when re-encoding (default ``"192k"``); ignored when
``audio_codec="copy"``.
shortest : bool, optional
If ``True``, the output stops when the shorter of the two streams
ends. If ``False`` (default), the output keeps the video length and
the audio is padded with silence (or truncated) by the muxer.
Notes
-----
Video stream is copied — no re-encoding — so the muxing is fast and
lossless on the video side. Use this after assembling a silent
``visuals.mp4`` and a separate ``narration.wav`` track.
Examples
--------
>>> mux_audio_video("silent.mp4", "voice.wav", "final.mp4")
"""
osh.checkfile(input_video, msg=f"Input video not found: {input_video}")
osh.checkfile(input_audio, msg=f"Input audio not found: {input_audio}")
quiet = osh.verbosity() <= 0
in_v = ffmpeg.input(input_video)
in_a = ffmpeg.input(input_audio)
out_kwargs = {"vcodec": "copy", "acodec": audio_codec}
if audio_codec != "copy":
out_kwargs["audio_bitrate"] = audio_bitrate
if shortest:
out_kwargs["shortest"] = None
ffmpeg.output(in_v.video, in_a.audio, output_video, **out_kwargs).run(
overwrite_output=True,
quiet=quiet,
)
assert is_valid_video_file(output_video), f"Failed to write mux_audio_video:\n\t{output_video}"
[docs]
def burn_subtitles(
input_video: str,
subtitles_file: str,
output_video: str,
force_style: str = None,
) -> None:
"""
Burn subtitles from an .srt / .vtt / .ass file into the video frames.
Parameters
----------
input_video : str
Path to the input video file.
subtitles_file : str
Path to a subtitles file in one of the formats libass understands:
* **.srt** — plain SubRip. Renders in the libass default style;
``<font color="…">`` tags are honored.
* **.vtt** — WebVTT. Cue-class colors (``<c.red>…</c>``) and any
inline ``::cue`` rules from a ``STYLE`` block are honored.
The companion ``srt2vtt()`` writes both pieces in one shot.
* **.ass** / **.ssa** — Advanced SubStation Alpha. All
per-cue formatting (font, color, outline, position) is
honored as authored.
output_video : str
Path to the output video file (.mp4 recommended).
force_style : str, optional
ASS-style override forwarded to the ``subtitles`` filter's
``force_style`` argument — e.g.
``"FontName=Helvetica,FontSize=24,PrimaryColour=&H00FFFFFF&"``.
Useful for SRT (which has no native styling) or to override a
global property of a VTT/ASS file without editing it. Per-cue
colors from VTT/ASS still win against ``force_style`` keys they
explicitly set.
Notes
-----
A single backend (libass through ffmpeg's ``subtitles`` filter)
handles all three formats, so there is no need for separate
``burn_srt`` / ``burn_vtt`` / ``burn_ass`` functions. The filter
mounts the file by path; we escape ``:`` to ``\\:`` and ``'`` to
``\\'`` so absolute paths on macOS / Windows behave. Video is
re-encoded (the filter rewrites every frame), audio is copied if
present.
Examples
--------
Plain SRT, default style:
>>> burn_subtitles("clip.mp4", "subs.srt", "captioned.mp4")
Colored WebVTT (cue classes carry their own colors):
>>> burn_subtitles("clip.mp4", "subs.vtt", "captioned.mp4")
Force a font + size on top of any source format:
>>> burn_subtitles("clip.mp4", "subs.vtt", "captioned.mp4",
... force_style="FontName=Helvetica,FontSize=28,Outline=2")
"""
osh.checkfile(input_video, msg=f"Input video not found: {input_video}")
osh.checkfile(subtitles_file, msg=f"Subtitles file not found: {subtitles_file}")
quiet = osh.verbosity() <= 0
# Check the obvious user error first (wrong file format) so callers on a
# libass-less ffmpeg still get a clear ValueError on a bogus path instead
# of the generic "libass missing" RuntimeError.
_, _, ext = osh.folder_name_ext(subtitles_file)
if ext.lower() not in {"srt", "vtt", "ass", "ssa"}:
raise ValueError(
f"burn_subtitles only accepts .srt / .vtt / .ass / .ssa "
f"(got .{ext}). For other formats, convert first — e.g. "
f"srt2vtt() in this same module."
)
# The `subtitles` filter is provided by libass — ffmpeg builds without
# `--enable-libass` (some Homebrew formulae, minimal docker images, …)
# silently lack it and produce a cryptic "Error parsing filterchain".
# Fail fast with an actionable error instead.
import subprocess as _sp
_filters = _sp.run(
["ffmpeg", "-hide_banner", "-filters"], capture_output=True, text=True, check=False
).stdout
if "subtitles" not in _filters:
raise RuntimeError(
"ffmpeg has no `subtitles` filter (libass missing). Rebuild "
"ffmpeg with `--enable-libass`, or on macOS: "
"`brew uninstall ffmpeg && brew install ffmpeg --HEAD` "
"(homebrew-core's bottle ships without libass on some archs)."
)
# Escape special chars for the subtitles filter — colons mainly, also
# backslashes and single quotes. Order matters: backslash first.
subs_abs = os.path.abspath(subtitles_file)
subs_esc = subs_abs.replace("\\", "\\\\").replace(":", r"\:").replace("'", r"\'")
vf = f"subtitles='{subs_esc}'"
if not osh.emptystring(force_style):
vf += f":force_style='{force_style}'"
in_stream = ffmpeg.input(input_video)
has_audio = video_dimensions(input_video).get("has_sound", False)
out_kwargs = {
"vcodec": "libx264",
"pix_fmt": "yuv420p",
"preset": "medium",
"crf": 20,
"vf": vf,
}
if has_audio:
out_kwargs["acodec"] = "copy"
else:
out_kwargs["an"] = None
in_stream.output(output_video, **out_kwargs).run(
overwrite_output=True,
quiet=quiet,
)
assert is_valid_video_file(output_video), f"Failed to write burn_subtitles:\n\t{output_video}"