from __future__ import annotations import base64 import json import math import shutil import subprocess from bisect import bisect_right from pathlib import Path from typing import Any, Iterable, Iterator from PIL import Image, ImageCms, ImageOps, UnidentifiedImageError MAX_DURATION_SECONDS = 2 * 60 * 60 MAX_DIMENSION = 32768 MAX_PIXELS = 160_000_000 MAX_MERGED_FRAMES = 30_000 SAMPLE_MS = 50 OUTPUT_SIZE = 64 MIN_CENTER = -31.0 MAX_CENTER = 32.0 class MediaConversionError(ValueError): pass def rgb_scene(rgb: bytes) -> dict[str, Any]: if len(rgb) != 64 * 64 * 3: raise MediaConversionError("converted frame is not 64x64 RGB888") return { "version": 1, "width": 64, "height": 64, "pixelRgb": base64.b64encode(rgb).decode("ascii"), "elements": [], } def _run(args: list[str], *, timeout: int = 120, stdout: Any = subprocess.PIPE) -> subprocess.CompletedProcess: try: result = subprocess.run( args, stdin=subprocess.DEVNULL, stdout=stdout, stderr=subprocess.PIPE, timeout=timeout, check=False, ) except FileNotFoundError as exc: raise MediaConversionError(f"required decoder is missing: {args[0]}") from exc except subprocess.TimeoutExpired as exc: raise MediaConversionError("media decoder made no progress before timeout") from exc if result.returncode: detail = result.stderr.decode("utf-8", "replace")[-1200:].strip() raise MediaConversionError(detail or "media decoder failed") return result def _hex_color(value: Any, field: str) -> tuple[int, int, int]: if not isinstance(value, str) or len(value) != 7 or value[0] != "#": raise MediaConversionError(f"{field} must be #RRGGBB") try: return tuple(bytes.fromhex(value[1:])) # type: ignore[return-value] except ValueError as exc: raise MediaConversionError(f"{field} must be #RRGGBB") from exc def validate_settings(value: Any) -> dict[str, Any]: if not isinstance(value, dict): raise MediaConversionError("settings must be an object") fit_mode = value.get("fit_mode") if fit_mode not in {"crop", "contain", "stretch"}: raise MediaConversionError("fit_mode must be crop, contain, or stretch") center_x, center_y, zoom = value.get("center_x"), value.get("center_y"), value.get("zoom") for field, number in (("center_x", center_x), ("center_y", center_y)): if ( isinstance(number, bool) or not isinstance(number, (int, float)) or not math.isfinite(number) or not MIN_CENTER <= number <= MAX_CENTER ): raise MediaConversionError(f"{field} must be a finite free-framing coordinate") if isinstance(zoom, bool) or not isinstance(zoom, (int, float)) or not 1 <= zoom <= 16: raise MediaConversionError("zoom must be in 1..16") _hex_color(value.get("transparency_color"), "transparency_color") _hex_color(value.get("padding_color"), "padding_color") return { "fit_mode": fit_mode, "center_x": float(center_x), "center_y": float(center_y), "zoom": float(zoom), "transparency_color": value["transparency_color"].upper(), "padding_color": value["padding_color"].upper(), } def _orient_and_color(image: Image.Image) -> Image.Image: image = ImageOps.exif_transpose(image) rgba = image.convert("RGBA") profile = image.info.get("icc_profile") if profile: alpha = rgba.getchannel("A") try: source = ImageCms.ImageCmsProfile(bytes(profile)) target = ImageCms.createProfile("sRGB") rgb = ImageCms.profileToProfile(rgba.convert("RGB"), source, target, outputMode="RGB") rgba = rgb.convert("RGBA") rgba.putalpha(alpha) except (ImageCms.PyCMSError, OSError, TypeError, ValueError) as exc: raise MediaConversionError("image ICC profile cannot be converted to sRGB") from exc return rgba def _sample_aspect_ratio(value: Any) -> float: if not isinstance(value, str) or ":" not in value: return 1.0 numerator, denominator = value.split(":", 1) try: ratio = float(numerator) / float(denominator) except (ValueError, ZeroDivisionError): return 1.0 return ratio if math.isfinite(ratio) and ratio > 0 else 1.0 def _metadata_display_dimensions(metadata: dict[str, Any]) -> tuple[float, float]: width, height = float(metadata["width"]), float(metadata["height"]) if metadata.get("decoder") == "ffmpeg": width *= _sample_aspect_ratio(metadata.get("sample_aspect_ratio")) return width, height def _free_crop_geometry( width: float, height: float, settings: dict[str, Any], ) -> tuple[int, int, int, int]: scale = OUTPUT_SIZE * settings["zoom"] / max(width, height) out_w = max(1, round(width * scale)) out_h = max(1, round(height * scale)) left = round(OUTPUT_SIZE / 2 - settings["center_x"] * out_w) top = round(OUTPUT_SIZE / 2 - settings["center_y"] * out_h) if left >= OUTPUT_SIZE or left + out_w <= 0 or top >= OUTPUT_SIZE or top + out_h <= 0: raise MediaConversionError("free-framing position must leave at least one output pixel visible") return out_w, out_h, left, top def validate_settings_for_metadata(value: Any, metadata: dict[str, Any] | None) -> dict[str, Any]: settings = validate_settings(value) if settings["fit_mode"] == "crop" and metadata is not None: _free_crop_geometry(*_metadata_display_dimensions(metadata), settings) return settings def _composite_free_crop( image: Image.Image, settings: dict[str, Any], geometry: tuple[int, int, int, int], ) -> bytes: out_w, out_h, left, top = geometry transparency = _hex_color(settings["transparency_color"], "transparency_color") padding = _hex_color(settings["padding_color"], "padding_color") resized = image if image.size == (out_w, out_h) else image.resize((out_w, out_h), Image.Resampling.LANCZOS) content = Image.alpha_composite( Image.new("RGBA", resized.size, (*transparency, 255)), resized, ).convert("RGB") output = Image.new("RGB", (OUTPUT_SIZE, OUTPUT_SIZE), padding) output.paste(content, (left, top)) return output.tobytes() def transform_frame(image: Image.Image, settings: dict[str, Any]) -> bytes: settings = validate_settings(settings) image = _orient_and_color(image) width, height = image.size if width < 1 or height < 1 or width > MAX_DIMENSION or height > MAX_DIMENSION or width * height > MAX_PIXELS: raise MediaConversionError("media pixel dimensions exceed the safety limit") transparency = _hex_color(settings["transparency_color"], "transparency_color") padding = _hex_color(settings["padding_color"], "padding_color") mode = settings["fit_mode"] if mode == "crop": return _composite_free_crop(image, settings, _free_crop_geometry(width, height, settings)) if mode == "stretch": image = image.resize((64, 64), Image.Resampling.LANCZOS) background = Image.new("RGBA", (64, 64), (*transparency, 255)) return Image.alpha_composite(background, image).convert("RGB").tobytes() scale = min(64 / width, 64 / height) size = (max(1, round(width * scale)), max(1, round(height * scale))) image = image.resize(size, Image.Resampling.LANCZOS) content = Image.alpha_composite(Image.new("RGBA", size, (*transparency, 255)), image).convert("RGB") output = Image.new("RGB", (64, 64), padding) output.paste(content, ((64 - size[0]) // 2, (64 - size[1]) // 2)) return output.tobytes() def _check_dimensions(width: int, height: int) -> None: if width < 1 or height < 1 or width > MAX_DIMENSION or height > MAX_DIMENSION or width * height > MAX_PIXELS: raise MediaConversionError("media pixel dimensions exceed the safety limit") def _pillow_probe(source: Path) -> tuple[dict[str, Any], list[Image.Image]]: with Image.open(source) as image: width, height = image.size _check_dimensions(width, height) count = int(getattr(image, "n_frames", 1)) durations, previews = [], [] preview_indexes = { min(count - 1, round(i * (count - 1) / min(4, count - 1))) for i in range(min(5, count)) } if count > 1 else {0} for index in range(count): image.seek(index) duration = image.info.get("duration", 100 if count > 1 else 0) if isinstance(duration, bool) or not isinstance(duration, (int, float)) or duration <= 0: duration = 100 durations.append(int(round(duration))) if index in preview_indexes: previews.append(_orient_and_color(image.copy())) total_ms = sum(durations) if count > 1 else 0 if total_ms > MAX_DURATION_SECONDS * 1000: raise MediaConversionError("media duration exceeds 2 hours") return { "decoder": "pillow", "format": str(image.format or "image").lower(), "width": previews[0].width, "height": previews[0].height, "duration_ms": total_ms, "source_frame_count": count, "dynamic": count > 1, "has_alpha": "A" in image.getbands() or "transparency" in image.info, }, previews def _ffprobe(source: Path) -> dict[str, Any]: result = _run([ "ffprobe", "-v", "error", "-protocol_whitelist", "file,pipe", "-show_streams", "-show_format", "-of", "json", str(source), ]) try: value = json.loads(result.stdout) except json.JSONDecodeError as exc: raise MediaConversionError("ffprobe returned invalid metadata") from exc streams = [ stream for stream in value.get("streams", []) if stream.get("codec_type") == "video" and not int((stream.get("disposition") or {}).get("attached_pic", 0)) ] if not streams: raise MediaConversionError("input does not contain a supported video stream") stream = streams[0] width, height = int(stream.get("width") or 0), int(stream.get("height") or 0) _check_dimensions(width, height) duration = stream.get("duration") or (value.get("format") or {}).get("duration") try: duration_ms = round(float(duration) * 1000) except (TypeError, ValueError): raise MediaConversionError("video duration is unavailable") if duration_ms <= 0 or duration_ms > MAX_DURATION_SECONDS * 1000: raise MediaConversionError("media duration must be in 0..2 hours") transfer = str(stream.get("color_transfer") or "").lower() hdr = transfer in {"smpte2084", "arib-std-b67"} if hdr: filters = _run(["ffmpeg", "-hide_banner", "-filters"]).stdout.decode("utf-8", "replace") if " zscale " not in filters or " tonemap " not in filters: raise MediaConversionError("HDR input requires FFmpeg zscale and tonemap filters") return { "decoder": "ffmpeg", "format": str((value.get("format") or {}).get("format_name") or "video"), "width": width, "height": height, "duration_ms": duration_ms, "source_frame_count": int(stream.get("nb_frames") or 0), "dynamic": True, "has_alpha": "a" in str(stream.get("pix_fmt") or ""), "stream_index": int(stream.get("index") or 0), "hdr": hdr, "sample_aspect_ratio": str(stream.get("sample_aspect_ratio") or "1:1"), } def _heif_probe(source: Path) -> tuple[dict[str, Any], list[Image.Image]]: decoded = source.parent / "decoded-heif.png" _run(["heif-convert", str(source), str(decoded)], timeout=120) metadata, images = _pillow_probe(decoded) metadata.update({"decoder": "heif", "format": "heif", "dynamic": False}) return metadata, images def analyze(source: Path, preview_dir: Path) -> dict[str, Any]: preview_dir.mkdir(parents=True, exist_ok=True) try: metadata, images = _pillow_probe(source) for index, image in enumerate(images): image.thumbnail((512, 512), Image.Resampling.LANCZOS) image.save(preview_dir / f"{index}.png", "PNG") except (UnidentifiedImageError, OSError): try: metadata, images = _heif_probe(source) for index, image in enumerate(images): image.thumbnail((512, 512), Image.Resampling.LANCZOS) image.save(preview_dir / f"{index}.png", "PNG") except MediaConversionError: metadata = _ffprobe(source) count = min(5, max(1, math.ceil(metadata["duration_ms"] / 1000))) times = [metadata["duration_ms"] * i / max(1, count - 1) / 1000 for i in range(count)] for index, when in enumerate(times): _run([ "ffmpeg", "-nostdin", "-v", "error", "-protocol_whitelist", "file,pipe", "-ss", f"{when:.3f}", "-i", str(source), "-map", f"0:{metadata['stream_index']}", "-frames:v", "1", "-vf", "scale=trunc(iw*sar+0.5):ih:flags=lanczos,setsar=1," "scale=512:512:force_original_aspect_ratio=decrease:flags=lanczos", "-y", str(preview_dir / f"{index}.png"), ], timeout=120) metadata["preview_count"] = len(list(preview_dir.glob("*.png"))) return metadata def _pillow_frames(source: Path, settings: dict[str, Any]) -> Iterator[tuple[bytes, int]]: with Image.open(source) as image: count = int(getattr(image, "n_frames", 1)) if count == 1: yield transform_frame(image.copy(), settings), 0 return frames, ends, elapsed = [], [], 0 for index in range(count): image.seek(index) duration = image.info.get("duration", 100) if not isinstance(duration, (int, float)) or isinstance(duration, bool) or duration <= 0: duration = 100 elapsed += int(round(duration)) ends.append(elapsed) frames.append(image.copy()) if elapsed > MAX_DURATION_SECONDS * 1000: raise MediaConversionError("media duration exceeds 2 hours") sample_times = range(0, elapsed, SAMPLE_MS) for time_ms in sample_times: index = min(len(frames) - 1, bisect_right(ends, time_ms)) duration = min(SAMPLE_MS, elapsed - time_ms) yield transform_frame(frames[index], settings), duration def _ffmpeg_frames(source: Path, metadata: dict[str, Any], settings: dict[str, Any]) -> Iterator[tuple[bytes, int]]: width, height = _metadata_display_dimensions(metadata) mode = settings["fit_mode"] if mode == "contain": scale = min(64 / width, 64 / height) out_w, out_h = max(1, round(width * scale)), max(1, round(height * scale)) geometry = f"scale={out_w}:{out_h}:flags=lanczos" elif mode == "stretch": out_w = out_h = 64 geometry = "scale=64:64:flags=lanczos" else: out_w, out_h, left, top = _free_crop_geometry(width, height, settings) geometry = f"scale={out_w}:{out_h}:flags=lanczos" color_filter = "" if metadata.get("hdr"): color_filter = "zscale=t=linear:npl=100,format=gbrpf32le,tonemap=hable,zscale=p=bt709:t=bt709:m=bt709," filters = f"fps=20,{color_filter}{geometry},setsar=1,format=rgba" try: process = subprocess.Popen([ "ffmpeg", "-nostdin", "-v", "error", "-protocol_whitelist", "file,pipe", "-i", str(source), "-map", f"0:{metadata['stream_index']}", "-an", "-sn", "-dn", "-vf", filters, "-f", "rawvideo", "-pix_fmt", "rgba", "pipe:1", ], stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE) except FileNotFoundError as exc: raise MediaConversionError("required decoder is missing: ffmpeg") from exc assert process.stdout is not None frame_bytes = out_w * out_h * 4 count = 0 while True: raw = process.stdout.read(frame_bytes) if not raw: break if len(raw) != frame_bytes: process.kill() raise MediaConversionError("FFmpeg returned a truncated video frame") rgba = Image.frombytes("RGBA", (out_w, out_h), raw) if mode == "contain": transparency = _hex_color(settings["transparency_color"], "transparency_color") padding = _hex_color(settings["padding_color"], "padding_color") content = Image.alpha_composite(Image.new("RGBA", rgba.size, (*transparency, 255)), rgba).convert("RGB") output = Image.new("RGB", (64, 64), padding) output.paste(content, ((64 - out_w) // 2, (64 - out_h) // 2)) rgb = output.tobytes() elif mode == "crop": rgb = _composite_free_crop(rgba, settings, (out_w, out_h, left, top)) else: background = Image.new("RGBA", (64, 64), (*_hex_color(settings["transparency_color"], "transparency_color"), 255)) rgb = Image.alpha_composite(background, rgba).convert("RGB").tobytes() count += 1 yield rgb, SAMPLE_MS stderr = process.stderr.read().decode("utf-8", "replace") if process.stderr else "" if process.wait(timeout=30): raise MediaConversionError(stderr[-1200:].strip() or "FFmpeg conversion failed") if count == 0: raise MediaConversionError("video decoder produced no frames") def converted_frames(source: Path, metadata: dict[str, Any], settings: dict[str, Any]) -> Iterator[tuple[bytes, int]]: settings = validate_settings_for_metadata(settings, metadata) raw_frames: Iterable[tuple[bytes, int]] if metadata["decoder"] in {"pillow", "heif"}: actual_source = source if metadata["decoder"] == "pillow" else source.parent / "decoded-heif.png" raw_frames = _pillow_frames(actual_source, settings) else: raw_frames = _ffmpeg_frames(source, metadata, settings) previous: bytes | None = None duration = 0 merged = 0 for rgb, frame_duration in raw_frames: if previous is None: previous, duration = rgb, frame_duration elif rgb == previous: duration += frame_duration else: merged += 1 if merged > MAX_MERGED_FRAMES: raise MediaConversionError("converted animation exceeds 30,000 merged frames") yield previous, max(SAMPLE_MS, duration) previous, duration = rgb, frame_duration if previous is None: raise MediaConversionError("decoder produced no frames") merged += 1 if merged > MAX_MERGED_FRAMES: raise MediaConversionError("converted animation exceeds 30,000 merged frames") yield previous, max(SAMPLE_MS, duration) if metadata["dynamic"] else 0