from __future__ import annotations import argparse import hashlib import json import os from pathlib import Path import subprocess import time from typing import Any from uuid import uuid4 CANDIDATE_RELEASE = "6.1.31-matrix-axp313a1" RECORD_NAME = "kernel-recovery.json" BOOT_HASHES = { "Image-matrix-axp313a1": "f8b6801cb9a300ee126635fd8834c5cdc6bd55857bb4438982db7f6fd78a8541", "sun50i-h616-walnutpi-1b-matrix-axp313a1.dtb": "10ef0eca3b9ec2063a8e4f47d0afc5e30b6e1a2ed8c4fcf9b0450630e2c11f66", "sun50i-h616-walnutpi-1b-emmc-matrix-axp313a1.dtb": "d0203cd03eb2c316cb36ed94016cbeb469d499ce06d866aaeef0494c45037c12", } MODULE_COUNT = 3035 MODULE_DIGEST = "8326558d841bc137c87d153d68903da574ebccaaee1bc6d6c129a8526af29346" class KernelRecovery: def __init__(self, data_root: Path, *, host_root: Path = Path("/")) -> None: self.data_root = Path(data_root) self.host_root = Path(host_root) self.record_path = self.data_root / RECORD_NAME def host(self, absolute: str) -> Path: return self.host_root / absolute.lstrip("/") @staticmethod def _digest(path: Path) -> bytes: digest = hashlib.sha256() with path.open("rb") as handle: for chunk in iter(lambda: handle.read(1024 * 1024), b""): digest.update(chunk) return digest.digest() def boot_id(self) -> str: return self.host("/proc/sys/kernel/random/boot_id").read_text(encoding="ascii").strip() def release(self) -> str: return self.host("/proc/sys/kernel/osrelease").read_text(encoding="ascii").strip() def capability(self) -> dict[str, Any]: policies = {} for policy in sorted(self.host("/sys/devices/system/cpu/cpufreq").glob("policy*")): if not policy.is_dir(): continue try: current = (policy / "scaling_governor").read_text(encoding="ascii").strip() available = (policy / "scaling_available_governors").read_text(encoding="ascii").split() except OSError: current, available = "", [] policies[policy.name] = {"current": current, "performance": "performance" in available} try: release = self.release() except OSError: release = "unavailable" return { "release": release, "policies": policies, "available": bool(policies) and all(value["current"] and value["performance"] for value in policies.values()), "marker": self.host("/boot/matrix-kernel-good").is_file(), } def verify_candidate(self) -> str | None: boot = self.host("/boot") for name, expected in BOOT_HASHES.items(): path = boot / name try: if self._digest(path).hex() != expected: return f"candidate file differs from registered artifact: {name}" except OSError: return f"candidate file is missing or unreadable: {name}" original = ( "Image", "sun50i-h616-walnutpi-1b.dtb", "sun50i-h616-walnutpi-1b-emmc.dtb", "boot.cmd.matrix-original", "boot.scr.matrix-original", "matrix-original.SHA256SUMS", ) if any(not (boot / name).is_file() for name in original): return "original kernel rollback files are incomplete" try: command = (boot / "boot.cmd").read_text(encoding="utf-8") script = (boot / "boot.scr").read_bytes() except (OSError, UnicodeError): return "managed boot script is unreadable" if command.count("matrix_kernel_candidate=1") != 1 or command.count("matrix-kernel-good") != 2: return "managed boot script is missing or ambiguous" if b"matrix_kernel_candidate=1" not in script or b"Image-matrix-axp313a1" not in script: return "compiled boot script does not select the candidate" health_unit = self.host("/etc/systemd/system/matrix-axp313a-health.service") health_script = self.host("/opt/matrix-screen-controller-system/axp313a_kernel_health.py") try: unit = health_unit.read_text(encoding="utf-8") except OSError: return "candidate health service is missing" if "ConditionKernelCommandLine=matrix_kernel_candidate=1" not in unit or not health_script.is_file(): return "candidate health service is incomplete" enabled = self.host("/etc/systemd/system/multi-user.target.wants/matrix-axp313a-health.service") if not enabled.exists(): return "candidate health service is not enabled" module_root = self.host(f"/lib/modules/{CANDIDATE_RELEASE}") if not (module_root / "modules.dep").is_file(): return "candidate module dependency index is missing" modules = sorted(module_root.rglob("*.ko"), key=lambda item: item.relative_to(module_root).as_posix()) if len(modules) != MODULE_COUNT: return "candidate module count differs from registered artifact" aggregate = hashlib.sha256() try: for module in modules: aggregate.update(module.relative_to(module_root).as_posix().encode("ascii")) aggregate.update(b"\0") aggregate.update(self._digest(module)) except (OSError, UnicodeError): return "candidate modules are unreadable" if aggregate.hexdigest() != MODULE_DIGEST: return "candidate modules differ from registered artifact" return None def read_record(self) -> dict[str, Any] | None: if not self.record_path.exists(): return None try: value = json.loads(self.record_path.read_text(encoding="utf-8")) except (OSError, UnicodeError, json.JSONDecodeError): return {"state": "failed", "reason": "kernel recovery record is invalid; manual inspection required"} if not isinstance(value, dict) or value.get("schema_version") != 1: return {"state": "failed", "reason": "kernel recovery record has an unsupported schema"} return value def write_record(self, state: str, reason: str, **fields: Any) -> None: self.data_root.mkdir(parents=True, exist_ok=True) document = {"schema_version": 1, "state": state, "reason": reason, "recorded_at": time.time(), **fields} temporary = self.record_path.with_name(f".{RECORD_NAME}.{uuid4().hex}.tmp") try: with temporary.open("x", encoding="utf-8", newline="\n") as handle: json.dump(document, handle, ensure_ascii=False, sort_keys=True) handle.write("\n") handle.flush() os.fsync(handle.fileno()) os.replace(temporary, self.record_path) finally: temporary.unlink(missing_ok=True) def status(self) -> dict[str, str]: record = self.read_record() if record: if record.get("state") == "pending" and record.get("attempts") == 0: recorded_at = record.get("recorded_at") if isinstance(recorded_at, (int, float)) and time.time() - recorded_at > 180: self.write_record("failed", "kernel recovery task did not start", attempts=0) return {"state": "failed", "reason": "kernel recovery task did not start"} if record.get("state") == "pending" and record.get("attempts") == 1: recorded_at = record.get("recorded_at") if (isinstance(recorded_at, (int, float)) and time.time() - recorded_at > 300 and self.boot_id() == record.get("source_boot_id")): self.write_record("failed", "candidate reboot did not occur", attempts=1) return {"state": "failed", "reason": "candidate reboot did not occur"} if record.get("state") == "succeeded": capability = self.capability() if capability["release"] != CANDIDATE_RELEASE: return {"state": "failed", "reason": "candidate kernel was lost after a successful recovery"} if not capability["available"] or not capability["marker"]: return {"state": "pending", "reason": "waiting for candidate boot health"} return {"state": str(record.get("state", "failed")), "reason": str(record.get("reason", ""))} capability = self.capability() if capability["release"] == CANDIDATE_RELEASE and capability["available"] and capability["marker"]: return {"state": "not_needed", "reason": ""} return {"state": "ready", "reason": "candidate kernel is not active"} def prepare_attempt(self, ota_result_id: str) -> bool: capability = self.capability() if capability["release"] == CANDIDATE_RELEASE and capability["available"] and capability["marker"]: return False if self.read_record() is not None: return False if capability["release"] != "6.1.31": self.write_record("failed", "automatic recovery only supports the original 6.1.31 kernel", ota_result_id=ota_result_id) return False reason = self.verify_candidate() if reason: self.write_record("requires_package", reason, ota_result_id=ota_result_id) return False self.write_record("pending", "waiting for one candidate boot", ota_result_id=ota_result_id, source_boot_id=self.boot_id(), attempts=0) return True def arm_and_reboot(self, *, reboot: bool = True) -> None: record = self.read_record() if not record or record.get("state") != "pending" or record.get("attempts") != 0: raise RuntimeError("no unused kernel recovery attempt is pending") recorded_at = record.get("recorded_at") if isinstance(recorded_at, (int, float)) and time.time() - recorded_at > 180: self.write_record("failed", "kernel recovery task started too late", attempts=0) return if self.boot_id() != record.get("source_boot_id"): self.write_record("failed", "boot changed before kernel recovery was armed", attempts=0) return if self.release() != "6.1.31": self.write_record("failed", "automatic recovery requires the original 6.1.31 kernel", attempts=0) return for _ in range(60): worker = subprocess.run(["systemctl", "is-active", "--quiet", "matrix-screen-controller-ota.service"], check=False, timeout=5) if worker.returncode != 0 and not (self.data_root / "ota/component-transaction").exists(): break time.sleep(1) else: self.write_record("failed", "OTA component transaction did not finish", attempts=1) return reason = self.verify_candidate() if reason: self.write_record("requires_package", reason, attempts=1) return marker = self.host("/boot/matrix-kernel-good") temporary = marker.with_name(".matrix-kernel-good.recovery") if marker.exists() or temporary.exists(): self.write_record("failed", "candidate boot marker already exists; manual inspection required", attempts=1) return try: with temporary.open("x", encoding="ascii", newline="\n") as handle: handle.write(f"{CANDIDATE_RELEASE}\n") handle.flush() os.fsync(handle.fileno()) os.replace(temporary, marker) if hasattr(os, "sync"): os.sync() self.write_record("pending", "candidate reboot requested", attempts=1, source_boot_id=record["source_boot_id"], ota_result_id=record.get("ota_result_id", "")) if reboot: subprocess.run(["systemctl", "reboot"], check=True, timeout=15) except Exception as exc: temporary.unlink(missing_ok=True) marker.unlink(missing_ok=True) self.write_record("failed", f"could not request candidate reboot: {exc}", attempts=1) raise def finalize_after_boot(self, *, wait_seconds: int = 100) -> None: record = self.read_record() if record and record.get("state") == "succeeded" and self.boot_id() != record.get("successful_boot_id"): for _ in range(wait_seconds): capability = self.capability() if capability["release"] != CANDIDATE_RELEASE: self.write_record("failed", "candidate kernel was lost after a successful recovery", attempts=1) return if capability["available"] and capability["marker"]: self.write_record("succeeded", "candidate kernel and cpufreq passed boot health", attempts=1, successful_boot_id=self.boot_id()) return time.sleep(1) self.write_record("failed", "candidate boot health did not complete", attempts=1) return if not record or record.get("state") != "pending" or record.get("attempts") != 1: return if self.boot_id() == record.get("source_boot_id"): return for _ in range(wait_seconds): capability = self.capability() if capability["release"] != CANDIDATE_RELEASE: self.write_record("failed", "candidate boot fell back to the original kernel", attempts=1) return if capability["available"] and capability["marker"]: self.write_record("succeeded", "candidate kernel and cpufreq passed boot health", attempts=1, successful_boot_id=self.boot_id()) return time.sleep(1) self.write_record("failed", "candidate boot health did not complete", attempts=1) def main() -> int: parser = argparse.ArgumentParser() parser.add_argument("command", choices=("arm-and-reboot", "status")) parser.add_argument("--data-root", type=Path, default=Path("/var/lib/matrix-screen-controller")) args = parser.parse_args() recovery = KernelRecovery(args.data_root) if args.command == "status": print(json.dumps(recovery.status(), ensure_ascii=False)) else: recovery.arm_and_reboot() return 0 if __name__ == "__main__": raise SystemExit(main())