403 lines
18 KiB
Python
403 lines
18 KiB
Python
from __future__ import annotations
|
|
|
|
import base64
|
|
import json
|
|
import math
|
|
import shutil
|
|
import subprocess
|
|
from bisect import bisect_right
|
|
from pathlib import Path
|
|
from typing import Any, Iterable, Iterator
|
|
|
|
from PIL import Image, ImageCms, ImageOps, UnidentifiedImageError
|
|
|
|
MAX_DURATION_SECONDS = 2 * 60 * 60
|
|
MAX_DIMENSION = 32768
|
|
MAX_PIXELS = 160_000_000
|
|
MAX_MERGED_FRAMES = 30_000
|
|
SAMPLE_MS = 50
|
|
OUTPUT_SIZE = 64
|
|
MIN_CENTER = -31.0
|
|
MAX_CENTER = 32.0
|
|
|
|
|
|
class MediaConversionError(ValueError):
|
|
pass
|
|
|
|
|
|
def rgb_scene(rgb: bytes) -> dict[str, Any]:
|
|
if len(rgb) != 64 * 64 * 3:
|
|
raise MediaConversionError("converted frame is not 64x64 RGB888")
|
|
return {
|
|
"version": 1, "width": 64, "height": 64,
|
|
"pixelRgb": base64.b64encode(rgb).decode("ascii"), "elements": [],
|
|
}
|
|
|
|
|
|
def _run(args: list[str], *, timeout: int = 120, stdout: Any = subprocess.PIPE) -> subprocess.CompletedProcess:
|
|
try:
|
|
result = subprocess.run(
|
|
args, stdin=subprocess.DEVNULL, stdout=stdout, stderr=subprocess.PIPE,
|
|
timeout=timeout, check=False,
|
|
)
|
|
except FileNotFoundError as exc:
|
|
raise MediaConversionError(f"required decoder is missing: {args[0]}") from exc
|
|
except subprocess.TimeoutExpired as exc:
|
|
raise MediaConversionError("media decoder made no progress before timeout") from exc
|
|
if result.returncode:
|
|
detail = result.stderr.decode("utf-8", "replace")[-1200:].strip()
|
|
raise MediaConversionError(detail or "media decoder failed")
|
|
return result
|
|
|
|
|
|
def _hex_color(value: Any, field: str) -> tuple[int, int, int]:
|
|
if not isinstance(value, str) or len(value) != 7 or value[0] != "#":
|
|
raise MediaConversionError(f"{field} must be #RRGGBB")
|
|
try:
|
|
return tuple(bytes.fromhex(value[1:])) # type: ignore[return-value]
|
|
except ValueError as exc:
|
|
raise MediaConversionError(f"{field} must be #RRGGBB") from exc
|
|
|
|
|
|
def validate_settings(value: Any) -> dict[str, Any]:
|
|
if not isinstance(value, dict):
|
|
raise MediaConversionError("settings must be an object")
|
|
fit_mode = value.get("fit_mode")
|
|
if fit_mode not in {"crop", "contain", "stretch"}:
|
|
raise MediaConversionError("fit_mode must be crop, contain, or stretch")
|
|
center_x, center_y, zoom = value.get("center_x"), value.get("center_y"), value.get("zoom")
|
|
for field, number in (("center_x", center_x), ("center_y", center_y)):
|
|
if (
|
|
isinstance(number, bool) or not isinstance(number, (int, float))
|
|
or not math.isfinite(number) or not MIN_CENTER <= number <= MAX_CENTER
|
|
):
|
|
raise MediaConversionError(f"{field} must be a finite free-framing coordinate")
|
|
if isinstance(zoom, bool) or not isinstance(zoom, (int, float)) or not 1 <= zoom <= 16:
|
|
raise MediaConversionError("zoom must be in 1..16")
|
|
_hex_color(value.get("transparency_color"), "transparency_color")
|
|
_hex_color(value.get("padding_color"), "padding_color")
|
|
return {
|
|
"fit_mode": fit_mode, "center_x": float(center_x), "center_y": float(center_y),
|
|
"zoom": float(zoom), "transparency_color": value["transparency_color"].upper(),
|
|
"padding_color": value["padding_color"].upper(),
|
|
}
|
|
|
|
|
|
def _orient_and_color(image: Image.Image) -> Image.Image:
|
|
image = ImageOps.exif_transpose(image)
|
|
rgba = image.convert("RGBA")
|
|
profile = image.info.get("icc_profile")
|
|
if profile:
|
|
alpha = rgba.getchannel("A")
|
|
try:
|
|
source = ImageCms.ImageCmsProfile(bytes(profile))
|
|
target = ImageCms.createProfile("sRGB")
|
|
rgb = ImageCms.profileToProfile(rgba.convert("RGB"), source, target, outputMode="RGB")
|
|
rgba = rgb.convert("RGBA")
|
|
rgba.putalpha(alpha)
|
|
except (ImageCms.PyCMSError, OSError, TypeError, ValueError) as exc:
|
|
raise MediaConversionError("image ICC profile cannot be converted to sRGB") from exc
|
|
return rgba
|
|
|
|
|
|
def _sample_aspect_ratio(value: Any) -> float:
|
|
if not isinstance(value, str) or ":" not in value:
|
|
return 1.0
|
|
numerator, denominator = value.split(":", 1)
|
|
try:
|
|
ratio = float(numerator) / float(denominator)
|
|
except (ValueError, ZeroDivisionError):
|
|
return 1.0
|
|
return ratio if math.isfinite(ratio) and ratio > 0 else 1.0
|
|
|
|
|
|
def _metadata_display_dimensions(metadata: dict[str, Any]) -> tuple[float, float]:
|
|
width, height = float(metadata["width"]), float(metadata["height"])
|
|
if metadata.get("decoder") == "ffmpeg":
|
|
width *= _sample_aspect_ratio(metadata.get("sample_aspect_ratio"))
|
|
return width, height
|
|
|
|
|
|
def _free_crop_geometry(
|
|
width: float, height: float, settings: dict[str, Any],
|
|
) -> tuple[int, int, int, int]:
|
|
scale = OUTPUT_SIZE * settings["zoom"] / max(width, height)
|
|
out_w = max(1, round(width * scale))
|
|
out_h = max(1, round(height * scale))
|
|
left = round(OUTPUT_SIZE / 2 - settings["center_x"] * out_w)
|
|
top = round(OUTPUT_SIZE / 2 - settings["center_y"] * out_h)
|
|
if left >= OUTPUT_SIZE or left + out_w <= 0 or top >= OUTPUT_SIZE or top + out_h <= 0:
|
|
raise MediaConversionError("free-framing position must leave at least one output pixel visible")
|
|
return out_w, out_h, left, top
|
|
|
|
|
|
def validate_settings_for_metadata(value: Any, metadata: dict[str, Any] | None) -> dict[str, Any]:
|
|
settings = validate_settings(value)
|
|
if settings["fit_mode"] == "crop" and metadata is not None:
|
|
_free_crop_geometry(*_metadata_display_dimensions(metadata), settings)
|
|
return settings
|
|
|
|
|
|
def _composite_free_crop(
|
|
image: Image.Image, settings: dict[str, Any], geometry: tuple[int, int, int, int],
|
|
) -> bytes:
|
|
out_w, out_h, left, top = geometry
|
|
transparency = _hex_color(settings["transparency_color"], "transparency_color")
|
|
padding = _hex_color(settings["padding_color"], "padding_color")
|
|
resized = image if image.size == (out_w, out_h) else image.resize((out_w, out_h), Image.Resampling.LANCZOS)
|
|
content = Image.alpha_composite(
|
|
Image.new("RGBA", resized.size, (*transparency, 255)), resized,
|
|
).convert("RGB")
|
|
output = Image.new("RGB", (OUTPUT_SIZE, OUTPUT_SIZE), padding)
|
|
output.paste(content, (left, top))
|
|
return output.tobytes()
|
|
|
|
|
|
def transform_frame(image: Image.Image, settings: dict[str, Any]) -> bytes:
|
|
settings = validate_settings(settings)
|
|
image = _orient_and_color(image)
|
|
width, height = image.size
|
|
if width < 1 or height < 1 or width > MAX_DIMENSION or height > MAX_DIMENSION or width * height > MAX_PIXELS:
|
|
raise MediaConversionError("media pixel dimensions exceed the safety limit")
|
|
transparency = _hex_color(settings["transparency_color"], "transparency_color")
|
|
padding = _hex_color(settings["padding_color"], "padding_color")
|
|
mode = settings["fit_mode"]
|
|
if mode == "crop":
|
|
return _composite_free_crop(image, settings, _free_crop_geometry(width, height, settings))
|
|
if mode == "stretch":
|
|
image = image.resize((64, 64), Image.Resampling.LANCZOS)
|
|
background = Image.new("RGBA", (64, 64), (*transparency, 255))
|
|
return Image.alpha_composite(background, image).convert("RGB").tobytes()
|
|
scale = min(64 / width, 64 / height)
|
|
size = (max(1, round(width * scale)), max(1, round(height * scale)))
|
|
image = image.resize(size, Image.Resampling.LANCZOS)
|
|
content = Image.alpha_composite(Image.new("RGBA", size, (*transparency, 255)), image).convert("RGB")
|
|
output = Image.new("RGB", (64, 64), padding)
|
|
output.paste(content, ((64 - size[0]) // 2, (64 - size[1]) // 2))
|
|
return output.tobytes()
|
|
|
|
|
|
def _check_dimensions(width: int, height: int) -> None:
|
|
if width < 1 or height < 1 or width > MAX_DIMENSION or height > MAX_DIMENSION or width * height > MAX_PIXELS:
|
|
raise MediaConversionError("media pixel dimensions exceed the safety limit")
|
|
|
|
|
|
def _pillow_probe(source: Path) -> tuple[dict[str, Any], list[Image.Image]]:
|
|
with Image.open(source) as image:
|
|
width, height = image.size
|
|
_check_dimensions(width, height)
|
|
count = int(getattr(image, "n_frames", 1))
|
|
durations, previews = [], []
|
|
preview_indexes = {
|
|
min(count - 1, round(i * (count - 1) / min(4, count - 1)))
|
|
for i in range(min(5, count))
|
|
} if count > 1 else {0}
|
|
for index in range(count):
|
|
image.seek(index)
|
|
duration = image.info.get("duration", 100 if count > 1 else 0)
|
|
if isinstance(duration, bool) or not isinstance(duration, (int, float)) or duration <= 0:
|
|
duration = 100
|
|
durations.append(int(round(duration)))
|
|
if index in preview_indexes:
|
|
previews.append(_orient_and_color(image.copy()))
|
|
total_ms = sum(durations) if count > 1 else 0
|
|
if total_ms > MAX_DURATION_SECONDS * 1000:
|
|
raise MediaConversionError("media duration exceeds 2 hours")
|
|
return {
|
|
"decoder": "pillow", "format": str(image.format or "image").lower(),
|
|
"width": previews[0].width, "height": previews[0].height, "duration_ms": total_ms,
|
|
"source_frame_count": count, "dynamic": count > 1,
|
|
"has_alpha": "A" in image.getbands() or "transparency" in image.info,
|
|
}, previews
|
|
|
|
|
|
def _ffprobe(source: Path) -> dict[str, Any]:
|
|
result = _run([
|
|
"ffprobe", "-v", "error", "-protocol_whitelist", "file,pipe",
|
|
"-show_streams", "-show_format", "-of", "json", str(source),
|
|
])
|
|
try:
|
|
value = json.loads(result.stdout)
|
|
except json.JSONDecodeError as exc:
|
|
raise MediaConversionError("ffprobe returned invalid metadata") from exc
|
|
streams = [
|
|
stream for stream in value.get("streams", [])
|
|
if stream.get("codec_type") == "video"
|
|
and not int((stream.get("disposition") or {}).get("attached_pic", 0))
|
|
]
|
|
if not streams:
|
|
raise MediaConversionError("input does not contain a supported video stream")
|
|
stream = streams[0]
|
|
width, height = int(stream.get("width") or 0), int(stream.get("height") or 0)
|
|
_check_dimensions(width, height)
|
|
duration = stream.get("duration") or (value.get("format") or {}).get("duration")
|
|
try:
|
|
duration_ms = round(float(duration) * 1000)
|
|
except (TypeError, ValueError):
|
|
raise MediaConversionError("video duration is unavailable")
|
|
if duration_ms <= 0 or duration_ms > MAX_DURATION_SECONDS * 1000:
|
|
raise MediaConversionError("media duration must be in 0..2 hours")
|
|
transfer = str(stream.get("color_transfer") or "").lower()
|
|
hdr = transfer in {"smpte2084", "arib-std-b67"}
|
|
if hdr:
|
|
filters = _run(["ffmpeg", "-hide_banner", "-filters"]).stdout.decode("utf-8", "replace")
|
|
if " zscale " not in filters or " tonemap " not in filters:
|
|
raise MediaConversionError("HDR input requires FFmpeg zscale and tonemap filters")
|
|
return {
|
|
"decoder": "ffmpeg", "format": str((value.get("format") or {}).get("format_name") or "video"),
|
|
"width": width, "height": height, "duration_ms": duration_ms,
|
|
"source_frame_count": int(stream.get("nb_frames") or 0), "dynamic": True,
|
|
"has_alpha": "a" in str(stream.get("pix_fmt") or ""),
|
|
"stream_index": int(stream.get("index") or 0), "hdr": hdr,
|
|
"sample_aspect_ratio": str(stream.get("sample_aspect_ratio") or "1:1"),
|
|
}
|
|
|
|
|
|
def _heif_probe(source: Path) -> tuple[dict[str, Any], list[Image.Image]]:
|
|
decoded = source.parent / "decoded-heif.png"
|
|
_run(["heif-convert", str(source), str(decoded)], timeout=120)
|
|
metadata, images = _pillow_probe(decoded)
|
|
metadata.update({"decoder": "heif", "format": "heif", "dynamic": False})
|
|
return metadata, images
|
|
|
|
|
|
def analyze(source: Path, preview_dir: Path) -> dict[str, Any]:
|
|
preview_dir.mkdir(parents=True, exist_ok=True)
|
|
try:
|
|
metadata, images = _pillow_probe(source)
|
|
for index, image in enumerate(images):
|
|
image.thumbnail((512, 512), Image.Resampling.LANCZOS)
|
|
image.save(preview_dir / f"{index}.png", "PNG")
|
|
except (UnidentifiedImageError, OSError):
|
|
try:
|
|
metadata, images = _heif_probe(source)
|
|
for index, image in enumerate(images):
|
|
image.thumbnail((512, 512), Image.Resampling.LANCZOS)
|
|
image.save(preview_dir / f"{index}.png", "PNG")
|
|
except MediaConversionError:
|
|
metadata = _ffprobe(source)
|
|
count = min(5, max(1, math.ceil(metadata["duration_ms"] / 1000)))
|
|
times = [metadata["duration_ms"] * i / max(1, count - 1) / 1000 for i in range(count)]
|
|
for index, when in enumerate(times):
|
|
_run([
|
|
"ffmpeg", "-nostdin", "-v", "error", "-protocol_whitelist", "file,pipe",
|
|
"-ss", f"{when:.3f}", "-i", str(source), "-map", f"0:{metadata['stream_index']}",
|
|
"-frames:v", "1", "-vf",
|
|
"scale=trunc(iw*sar+0.5):ih:flags=lanczos,setsar=1,"
|
|
"scale=512:512:force_original_aspect_ratio=decrease:flags=lanczos",
|
|
"-y", str(preview_dir / f"{index}.png"),
|
|
], timeout=120)
|
|
metadata["preview_count"] = len(list(preview_dir.glob("*.png")))
|
|
return metadata
|
|
|
|
|
|
def _pillow_frames(source: Path, settings: dict[str, Any]) -> Iterator[tuple[bytes, int]]:
|
|
with Image.open(source) as image:
|
|
count = int(getattr(image, "n_frames", 1))
|
|
if count == 1:
|
|
yield transform_frame(image.copy(), settings), 0
|
|
return
|
|
frames, ends, elapsed = [], [], 0
|
|
for index in range(count):
|
|
image.seek(index)
|
|
duration = image.info.get("duration", 100)
|
|
if not isinstance(duration, (int, float)) or isinstance(duration, bool) or duration <= 0:
|
|
duration = 100
|
|
elapsed += int(round(duration))
|
|
ends.append(elapsed)
|
|
frames.append(image.copy())
|
|
if elapsed > MAX_DURATION_SECONDS * 1000:
|
|
raise MediaConversionError("media duration exceeds 2 hours")
|
|
sample_times = range(0, elapsed, SAMPLE_MS)
|
|
for time_ms in sample_times:
|
|
index = min(len(frames) - 1, bisect_right(ends, time_ms))
|
|
duration = min(SAMPLE_MS, elapsed - time_ms)
|
|
yield transform_frame(frames[index], settings), duration
|
|
|
|
|
|
def _ffmpeg_frames(source: Path, metadata: dict[str, Any], settings: dict[str, Any]) -> Iterator[tuple[bytes, int]]:
|
|
width, height = _metadata_display_dimensions(metadata)
|
|
mode = settings["fit_mode"]
|
|
if mode == "contain":
|
|
scale = min(64 / width, 64 / height)
|
|
out_w, out_h = max(1, round(width * scale)), max(1, round(height * scale))
|
|
geometry = f"scale={out_w}:{out_h}:flags=lanczos"
|
|
elif mode == "stretch":
|
|
out_w = out_h = 64
|
|
geometry = "scale=64:64:flags=lanczos"
|
|
else:
|
|
out_w, out_h, left, top = _free_crop_geometry(width, height, settings)
|
|
geometry = f"scale={out_w}:{out_h}:flags=lanczos"
|
|
color_filter = ""
|
|
if metadata.get("hdr"):
|
|
color_filter = "zscale=t=linear:npl=100,format=gbrpf32le,tonemap=hable,zscale=p=bt709:t=bt709:m=bt709,"
|
|
filters = f"fps=20,{color_filter}{geometry},setsar=1,format=rgba"
|
|
try:
|
|
process = subprocess.Popen([
|
|
"ffmpeg", "-nostdin", "-v", "error", "-protocol_whitelist", "file,pipe",
|
|
"-i", str(source), "-map", f"0:{metadata['stream_index']}", "-an", "-sn", "-dn",
|
|
"-vf", filters, "-f", "rawvideo", "-pix_fmt", "rgba", "pipe:1",
|
|
], stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
|
except FileNotFoundError as exc:
|
|
raise MediaConversionError("required decoder is missing: ffmpeg") from exc
|
|
assert process.stdout is not None
|
|
frame_bytes = out_w * out_h * 4
|
|
count = 0
|
|
while True:
|
|
raw = process.stdout.read(frame_bytes)
|
|
if not raw:
|
|
break
|
|
if len(raw) != frame_bytes:
|
|
process.kill()
|
|
raise MediaConversionError("FFmpeg returned a truncated video frame")
|
|
rgba = Image.frombytes("RGBA", (out_w, out_h), raw)
|
|
if mode == "contain":
|
|
transparency = _hex_color(settings["transparency_color"], "transparency_color")
|
|
padding = _hex_color(settings["padding_color"], "padding_color")
|
|
content = Image.alpha_composite(Image.new("RGBA", rgba.size, (*transparency, 255)), rgba).convert("RGB")
|
|
output = Image.new("RGB", (64, 64), padding)
|
|
output.paste(content, ((64 - out_w) // 2, (64 - out_h) // 2))
|
|
rgb = output.tobytes()
|
|
elif mode == "crop":
|
|
rgb = _composite_free_crop(rgba, settings, (out_w, out_h, left, top))
|
|
else:
|
|
background = Image.new("RGBA", (64, 64), (*_hex_color(settings["transparency_color"], "transparency_color"), 255))
|
|
rgb = Image.alpha_composite(background, rgba).convert("RGB").tobytes()
|
|
count += 1
|
|
yield rgb, SAMPLE_MS
|
|
stderr = process.stderr.read().decode("utf-8", "replace") if process.stderr else ""
|
|
if process.wait(timeout=30):
|
|
raise MediaConversionError(stderr[-1200:].strip() or "FFmpeg conversion failed")
|
|
if count == 0:
|
|
raise MediaConversionError("video decoder produced no frames")
|
|
|
|
|
|
def converted_frames(source: Path, metadata: dict[str, Any], settings: dict[str, Any]) -> Iterator[tuple[bytes, int]]:
|
|
settings = validate_settings_for_metadata(settings, metadata)
|
|
raw_frames: Iterable[tuple[bytes, int]]
|
|
if metadata["decoder"] in {"pillow", "heif"}:
|
|
actual_source = source if metadata["decoder"] == "pillow" else source.parent / "decoded-heif.png"
|
|
raw_frames = _pillow_frames(actual_source, settings)
|
|
else:
|
|
raw_frames = _ffmpeg_frames(source, metadata, settings)
|
|
previous: bytes | None = None
|
|
duration = 0
|
|
merged = 0
|
|
for rgb, frame_duration in raw_frames:
|
|
if previous is None:
|
|
previous, duration = rgb, frame_duration
|
|
elif rgb == previous:
|
|
duration += frame_duration
|
|
else:
|
|
merged += 1
|
|
if merged > MAX_MERGED_FRAMES:
|
|
raise MediaConversionError("converted animation exceeds 30,000 merged frames")
|
|
yield previous, max(SAMPLE_MS, duration)
|
|
previous, duration = rgb, frame_duration
|
|
if previous is None:
|
|
raise MediaConversionError("decoder produced no frames")
|
|
merged += 1
|
|
if merged > MAX_MERGED_FRAMES:
|
|
raise MediaConversionError("converted animation exceeds 30,000 merged frames")
|
|
yield previous, max(SAMPLE_MS, duration) if metadata["dynamic"] else 0
|