Files
matrix-screen-controller/核桃派软件源代码/app/media/converter.py
T

403 lines
18 KiB
Python

from __future__ import annotations
import base64
import json
import math
import shutil
import subprocess
from bisect import bisect_right
from pathlib import Path
from typing import Any, Iterable, Iterator
from PIL import Image, ImageCms, ImageOps, UnidentifiedImageError
MAX_DURATION_SECONDS = 2 * 60 * 60
MAX_DIMENSION = 32768
MAX_PIXELS = 160_000_000
MAX_MERGED_FRAMES = 30_000
SAMPLE_MS = 50
OUTPUT_SIZE = 64
MIN_CENTER = -31.0
MAX_CENTER = 32.0
class MediaConversionError(ValueError):
pass
def rgb_scene(rgb: bytes) -> dict[str, Any]:
if len(rgb) != 64 * 64 * 3:
raise MediaConversionError("converted frame is not 64x64 RGB888")
return {
"version": 1, "width": 64, "height": 64,
"pixelRgb": base64.b64encode(rgb).decode("ascii"), "elements": [],
}
def _run(args: list[str], *, timeout: int = 120, stdout: Any = subprocess.PIPE) -> subprocess.CompletedProcess:
try:
result = subprocess.run(
args, stdin=subprocess.DEVNULL, stdout=stdout, stderr=subprocess.PIPE,
timeout=timeout, check=False,
)
except FileNotFoundError as exc:
raise MediaConversionError(f"required decoder is missing: {args[0]}") from exc
except subprocess.TimeoutExpired as exc:
raise MediaConversionError("media decoder made no progress before timeout") from exc
if result.returncode:
detail = result.stderr.decode("utf-8", "replace")[-1200:].strip()
raise MediaConversionError(detail or "media decoder failed")
return result
def _hex_color(value: Any, field: str) -> tuple[int, int, int]:
if not isinstance(value, str) or len(value) != 7 or value[0] != "#":
raise MediaConversionError(f"{field} must be #RRGGBB")
try:
return tuple(bytes.fromhex(value[1:])) # type: ignore[return-value]
except ValueError as exc:
raise MediaConversionError(f"{field} must be #RRGGBB") from exc
def validate_settings(value: Any) -> dict[str, Any]:
if not isinstance(value, dict):
raise MediaConversionError("settings must be an object")
fit_mode = value.get("fit_mode")
if fit_mode not in {"crop", "contain", "stretch"}:
raise MediaConversionError("fit_mode must be crop, contain, or stretch")
center_x, center_y, zoom = value.get("center_x"), value.get("center_y"), value.get("zoom")
for field, number in (("center_x", center_x), ("center_y", center_y)):
if (
isinstance(number, bool) or not isinstance(number, (int, float))
or not math.isfinite(number) or not MIN_CENTER <= number <= MAX_CENTER
):
raise MediaConversionError(f"{field} must be a finite free-framing coordinate")
if isinstance(zoom, bool) or not isinstance(zoom, (int, float)) or not 1 <= zoom <= 16:
raise MediaConversionError("zoom must be in 1..16")
_hex_color(value.get("transparency_color"), "transparency_color")
_hex_color(value.get("padding_color"), "padding_color")
return {
"fit_mode": fit_mode, "center_x": float(center_x), "center_y": float(center_y),
"zoom": float(zoom), "transparency_color": value["transparency_color"].upper(),
"padding_color": value["padding_color"].upper(),
}
def _orient_and_color(image: Image.Image) -> Image.Image:
image = ImageOps.exif_transpose(image)
rgba = image.convert("RGBA")
profile = image.info.get("icc_profile")
if profile:
alpha = rgba.getchannel("A")
try:
source = ImageCms.ImageCmsProfile(bytes(profile))
target = ImageCms.createProfile("sRGB")
rgb = ImageCms.profileToProfile(rgba.convert("RGB"), source, target, outputMode="RGB")
rgba = rgb.convert("RGBA")
rgba.putalpha(alpha)
except (ImageCms.PyCMSError, OSError, TypeError, ValueError) as exc:
raise MediaConversionError("image ICC profile cannot be converted to sRGB") from exc
return rgba
def _sample_aspect_ratio(value: Any) -> float:
if not isinstance(value, str) or ":" not in value:
return 1.0
numerator, denominator = value.split(":", 1)
try:
ratio = float(numerator) / float(denominator)
except (ValueError, ZeroDivisionError):
return 1.0
return ratio if math.isfinite(ratio) and ratio > 0 else 1.0
def _metadata_display_dimensions(metadata: dict[str, Any]) -> tuple[float, float]:
width, height = float(metadata["width"]), float(metadata["height"])
if metadata.get("decoder") == "ffmpeg":
width *= _sample_aspect_ratio(metadata.get("sample_aspect_ratio"))
return width, height
def _free_crop_geometry(
width: float, height: float, settings: dict[str, Any],
) -> tuple[int, int, int, int]:
scale = OUTPUT_SIZE * settings["zoom"] / max(width, height)
out_w = max(1, round(width * scale))
out_h = max(1, round(height * scale))
left = round(OUTPUT_SIZE / 2 - settings["center_x"] * out_w)
top = round(OUTPUT_SIZE / 2 - settings["center_y"] * out_h)
if left >= OUTPUT_SIZE or left + out_w <= 0 or top >= OUTPUT_SIZE or top + out_h <= 0:
raise MediaConversionError("free-framing position must leave at least one output pixel visible")
return out_w, out_h, left, top
def validate_settings_for_metadata(value: Any, metadata: dict[str, Any] | None) -> dict[str, Any]:
settings = validate_settings(value)
if settings["fit_mode"] == "crop" and metadata is not None:
_free_crop_geometry(*_metadata_display_dimensions(metadata), settings)
return settings
def _composite_free_crop(
image: Image.Image, settings: dict[str, Any], geometry: tuple[int, int, int, int],
) -> bytes:
out_w, out_h, left, top = geometry
transparency = _hex_color(settings["transparency_color"], "transparency_color")
padding = _hex_color(settings["padding_color"], "padding_color")
resized = image if image.size == (out_w, out_h) else image.resize((out_w, out_h), Image.Resampling.LANCZOS)
content = Image.alpha_composite(
Image.new("RGBA", resized.size, (*transparency, 255)), resized,
).convert("RGB")
output = Image.new("RGB", (OUTPUT_SIZE, OUTPUT_SIZE), padding)
output.paste(content, (left, top))
return output.tobytes()
def transform_frame(image: Image.Image, settings: dict[str, Any]) -> bytes:
settings = validate_settings(settings)
image = _orient_and_color(image)
width, height = image.size
if width < 1 or height < 1 or width > MAX_DIMENSION or height > MAX_DIMENSION or width * height > MAX_PIXELS:
raise MediaConversionError("media pixel dimensions exceed the safety limit")
transparency = _hex_color(settings["transparency_color"], "transparency_color")
padding = _hex_color(settings["padding_color"], "padding_color")
mode = settings["fit_mode"]
if mode == "crop":
return _composite_free_crop(image, settings, _free_crop_geometry(width, height, settings))
if mode == "stretch":
image = image.resize((64, 64), Image.Resampling.LANCZOS)
background = Image.new("RGBA", (64, 64), (*transparency, 255))
return Image.alpha_composite(background, image).convert("RGB").tobytes()
scale = min(64 / width, 64 / height)
size = (max(1, round(width * scale)), max(1, round(height * scale)))
image = image.resize(size, Image.Resampling.LANCZOS)
content = Image.alpha_composite(Image.new("RGBA", size, (*transparency, 255)), image).convert("RGB")
output = Image.new("RGB", (64, 64), padding)
output.paste(content, ((64 - size[0]) // 2, (64 - size[1]) // 2))
return output.tobytes()
def _check_dimensions(width: int, height: int) -> None:
if width < 1 or height < 1 or width > MAX_DIMENSION or height > MAX_DIMENSION or width * height > MAX_PIXELS:
raise MediaConversionError("media pixel dimensions exceed the safety limit")
def _pillow_probe(source: Path) -> tuple[dict[str, Any], list[Image.Image]]:
with Image.open(source) as image:
width, height = image.size
_check_dimensions(width, height)
count = int(getattr(image, "n_frames", 1))
durations, previews = [], []
preview_indexes = {
min(count - 1, round(i * (count - 1) / min(4, count - 1)))
for i in range(min(5, count))
} if count > 1 else {0}
for index in range(count):
image.seek(index)
duration = image.info.get("duration", 100 if count > 1 else 0)
if isinstance(duration, bool) or not isinstance(duration, (int, float)) or duration <= 0:
duration = 100
durations.append(int(round(duration)))
if index in preview_indexes:
previews.append(_orient_and_color(image.copy()))
total_ms = sum(durations) if count > 1 else 0
if total_ms > MAX_DURATION_SECONDS * 1000:
raise MediaConversionError("media duration exceeds 2 hours")
return {
"decoder": "pillow", "format": str(image.format or "image").lower(),
"width": previews[0].width, "height": previews[0].height, "duration_ms": total_ms,
"source_frame_count": count, "dynamic": count > 1,
"has_alpha": "A" in image.getbands() or "transparency" in image.info,
}, previews
def _ffprobe(source: Path) -> dict[str, Any]:
result = _run([
"ffprobe", "-v", "error", "-protocol_whitelist", "file,pipe",
"-show_streams", "-show_format", "-of", "json", str(source),
])
try:
value = json.loads(result.stdout)
except json.JSONDecodeError as exc:
raise MediaConversionError("ffprobe returned invalid metadata") from exc
streams = [
stream for stream in value.get("streams", [])
if stream.get("codec_type") == "video"
and not int((stream.get("disposition") or {}).get("attached_pic", 0))
]
if not streams:
raise MediaConversionError("input does not contain a supported video stream")
stream = streams[0]
width, height = int(stream.get("width") or 0), int(stream.get("height") or 0)
_check_dimensions(width, height)
duration = stream.get("duration") or (value.get("format") or {}).get("duration")
try:
duration_ms = round(float(duration) * 1000)
except (TypeError, ValueError):
raise MediaConversionError("video duration is unavailable")
if duration_ms <= 0 or duration_ms > MAX_DURATION_SECONDS * 1000:
raise MediaConversionError("media duration must be in 0..2 hours")
transfer = str(stream.get("color_transfer") or "").lower()
hdr = transfer in {"smpte2084", "arib-std-b67"}
if hdr:
filters = _run(["ffmpeg", "-hide_banner", "-filters"]).stdout.decode("utf-8", "replace")
if " zscale " not in filters or " tonemap " not in filters:
raise MediaConversionError("HDR input requires FFmpeg zscale and tonemap filters")
return {
"decoder": "ffmpeg", "format": str((value.get("format") or {}).get("format_name") or "video"),
"width": width, "height": height, "duration_ms": duration_ms,
"source_frame_count": int(stream.get("nb_frames") or 0), "dynamic": True,
"has_alpha": "a" in str(stream.get("pix_fmt") or ""),
"stream_index": int(stream.get("index") or 0), "hdr": hdr,
"sample_aspect_ratio": str(stream.get("sample_aspect_ratio") or "1:1"),
}
def _heif_probe(source: Path) -> tuple[dict[str, Any], list[Image.Image]]:
decoded = source.parent / "decoded-heif.png"
_run(["heif-convert", str(source), str(decoded)], timeout=120)
metadata, images = _pillow_probe(decoded)
metadata.update({"decoder": "heif", "format": "heif", "dynamic": False})
return metadata, images
def analyze(source: Path, preview_dir: Path) -> dict[str, Any]:
preview_dir.mkdir(parents=True, exist_ok=True)
try:
metadata, images = _pillow_probe(source)
for index, image in enumerate(images):
image.thumbnail((512, 512), Image.Resampling.LANCZOS)
image.save(preview_dir / f"{index}.png", "PNG")
except (UnidentifiedImageError, OSError):
try:
metadata, images = _heif_probe(source)
for index, image in enumerate(images):
image.thumbnail((512, 512), Image.Resampling.LANCZOS)
image.save(preview_dir / f"{index}.png", "PNG")
except MediaConversionError:
metadata = _ffprobe(source)
count = min(5, max(1, math.ceil(metadata["duration_ms"] / 1000)))
times = [metadata["duration_ms"] * i / max(1, count - 1) / 1000 for i in range(count)]
for index, when in enumerate(times):
_run([
"ffmpeg", "-nostdin", "-v", "error", "-protocol_whitelist", "file,pipe",
"-ss", f"{when:.3f}", "-i", str(source), "-map", f"0:{metadata['stream_index']}",
"-frames:v", "1", "-vf",
"scale=trunc(iw*sar+0.5):ih:flags=lanczos,setsar=1,"
"scale=512:512:force_original_aspect_ratio=decrease:flags=lanczos",
"-y", str(preview_dir / f"{index}.png"),
], timeout=120)
metadata["preview_count"] = len(list(preview_dir.glob("*.png")))
return metadata
def _pillow_frames(source: Path, settings: dict[str, Any]) -> Iterator[tuple[bytes, int]]:
with Image.open(source) as image:
count = int(getattr(image, "n_frames", 1))
if count == 1:
yield transform_frame(image.copy(), settings), 0
return
frames, ends, elapsed = [], [], 0
for index in range(count):
image.seek(index)
duration = image.info.get("duration", 100)
if not isinstance(duration, (int, float)) or isinstance(duration, bool) or duration <= 0:
duration = 100
elapsed += int(round(duration))
ends.append(elapsed)
frames.append(image.copy())
if elapsed > MAX_DURATION_SECONDS * 1000:
raise MediaConversionError("media duration exceeds 2 hours")
sample_times = range(0, elapsed, SAMPLE_MS)
for time_ms in sample_times:
index = min(len(frames) - 1, bisect_right(ends, time_ms))
duration = min(SAMPLE_MS, elapsed - time_ms)
yield transform_frame(frames[index], settings), duration
def _ffmpeg_frames(source: Path, metadata: dict[str, Any], settings: dict[str, Any]) -> Iterator[tuple[bytes, int]]:
width, height = _metadata_display_dimensions(metadata)
mode = settings["fit_mode"]
if mode == "contain":
scale = min(64 / width, 64 / height)
out_w, out_h = max(1, round(width * scale)), max(1, round(height * scale))
geometry = f"scale={out_w}:{out_h}:flags=lanczos"
elif mode == "stretch":
out_w = out_h = 64
geometry = "scale=64:64:flags=lanczos"
else:
out_w, out_h, left, top = _free_crop_geometry(width, height, settings)
geometry = f"scale={out_w}:{out_h}:flags=lanczos"
color_filter = ""
if metadata.get("hdr"):
color_filter = "zscale=t=linear:npl=100,format=gbrpf32le,tonemap=hable,zscale=p=bt709:t=bt709:m=bt709,"
filters = f"fps=20,{color_filter}{geometry},setsar=1,format=rgba"
try:
process = subprocess.Popen([
"ffmpeg", "-nostdin", "-v", "error", "-protocol_whitelist", "file,pipe",
"-i", str(source), "-map", f"0:{metadata['stream_index']}", "-an", "-sn", "-dn",
"-vf", filters, "-f", "rawvideo", "-pix_fmt", "rgba", "pipe:1",
], stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
except FileNotFoundError as exc:
raise MediaConversionError("required decoder is missing: ffmpeg") from exc
assert process.stdout is not None
frame_bytes = out_w * out_h * 4
count = 0
while True:
raw = process.stdout.read(frame_bytes)
if not raw:
break
if len(raw) != frame_bytes:
process.kill()
raise MediaConversionError("FFmpeg returned a truncated video frame")
rgba = Image.frombytes("RGBA", (out_w, out_h), raw)
if mode == "contain":
transparency = _hex_color(settings["transparency_color"], "transparency_color")
padding = _hex_color(settings["padding_color"], "padding_color")
content = Image.alpha_composite(Image.new("RGBA", rgba.size, (*transparency, 255)), rgba).convert("RGB")
output = Image.new("RGB", (64, 64), padding)
output.paste(content, ((64 - out_w) // 2, (64 - out_h) // 2))
rgb = output.tobytes()
elif mode == "crop":
rgb = _composite_free_crop(rgba, settings, (out_w, out_h, left, top))
else:
background = Image.new("RGBA", (64, 64), (*_hex_color(settings["transparency_color"], "transparency_color"), 255))
rgb = Image.alpha_composite(background, rgba).convert("RGB").tobytes()
count += 1
yield rgb, SAMPLE_MS
stderr = process.stderr.read().decode("utf-8", "replace") if process.stderr else ""
if process.wait(timeout=30):
raise MediaConversionError(stderr[-1200:].strip() or "FFmpeg conversion failed")
if count == 0:
raise MediaConversionError("video decoder produced no frames")
def converted_frames(source: Path, metadata: dict[str, Any], settings: dict[str, Any]) -> Iterator[tuple[bytes, int]]:
settings = validate_settings_for_metadata(settings, metadata)
raw_frames: Iterable[tuple[bytes, int]]
if metadata["decoder"] in {"pillow", "heif"}:
actual_source = source if metadata["decoder"] == "pillow" else source.parent / "decoded-heif.png"
raw_frames = _pillow_frames(actual_source, settings)
else:
raw_frames = _ffmpeg_frames(source, metadata, settings)
previous: bytes | None = None
duration = 0
merged = 0
for rgb, frame_duration in raw_frames:
if previous is None:
previous, duration = rgb, frame_duration
elif rgb == previous:
duration += frame_duration
else:
merged += 1
if merged > MAX_MERGED_FRAMES:
raise MediaConversionError("converted animation exceeds 30,000 merged frames")
yield previous, max(SAMPLE_MS, duration)
previous, duration = rgb, frame_duration
if previous is None:
raise MediaConversionError("decoder produced no frames")
merged += 1
if merged > MAX_MERGED_FRAMES:
raise MediaConversionError("converted animation exceeds 30,000 merged frames")
yield previous, max(SAMPLE_MS, duration) if metadata["dynamic"] else 0