from __future__ import annotations import argparse import hashlib import json import re from dataclasses import dataclass, field from html.parser import HTMLParser from pathlib import Path from typing import Any, Iterable SOURCE_ROOT = Path(__file__).resolve().parents[1] STATIC_ROOT = SOURCE_ROOT / "app" / "static" OUTPUT_PATH = STATIC_ROOT / "ui-copy.json" DYNAMIC_PATH = STATIC_ROOT / "ui-copy-dynamic.json" TEXT_ATTRIBUTES = ("aria-label", "placeholder", "data-unavailable-reason", "title") ANCHOR_TAGS = {"article", "section", "header", "main", "form", "fieldset", "div"} IGNORED_TAGS = {"script", "style", "svg", "path", "noscript"} PLACEHOLDER_PATTERN = re.compile(r"\{([a-zA-Z][a-zA-Z0-9_]*)\}") FUNCTIONAL_VALUE_PATTERN = re.compile(r"^[+\-]?(?:\d+(?:\.\d+)?|#[0-9a-fA-F]{3,8})(?:\s*(?:%|°|Hz|ms|px|V))?$") @dataclass class HtmlNode: tag: str attrs: dict[str, str] parent: "HtmlNode | None" = None children: list["HtmlNode | str"] = field(default_factory=list) class TreeParser(HTMLParser): def __init__(self) -> None: super().__init__(convert_charrefs=True) self.root = HtmlNode("document", {}) self.stack = [self.root] def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: node = HtmlNode(tag, {key: value or "" for key, value in attrs}, self.stack[-1]) self.stack[-1].children.append(node) if tag not in {"area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", "param", "source", "track", "wbr"}: self.stack.append(node) def handle_startendtag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: node = HtmlNode(tag, {key: value or "" for key, value in attrs}, self.stack[-1]) self.stack[-1].children.append(node) def handle_endtag(self, tag: str) -> None: for index in range(len(self.stack) - 1, 0, -1): if self.stack[index].tag == tag: del self.stack[index:] return def handle_data(self, data: str) -> None: self.stack[-1].children.append(data) def normalized_text(value: str) -> str: return re.sub(r"\s+", " ", value).strip() def stable_id(prefix: str, *parts: object) -> str: raw = "\0".join(str(part) for part in parts) return f"{prefix}.{hashlib.sha256(raw.encode('utf-8')).hexdigest()[:16]}" def iter_nodes(node: HtmlNode) -> Iterable[HtmlNode]: for child in node.children: if isinstance(child, HtmlNode): yield child yield from iter_nodes(child) def nearest_scope(node: HtmlNode) -> str: current: HtmlNode | None = node while current: identifier = current.attrs.get("id", "") if identifier.startswith("workspace-"): return identifier.removeprefix("workspace-") if current.tag == "dialog" and identifier: return f"dialog:{identifier}" current = current.parent return "global" def copy_ignored(node: HtmlNode) -> bool: current: HtmlNode | None = node while current: if "data-ui-copy-ignore" in current.attrs or "data-ui-copy-value" in current.attrs: return True current = current.parent return False def selector_for(node: HtmlNode) -> str: if identifier := node.attrs.get("id"): return f"#{identifier}" parts: list[str] = [] current: HtmlNode | None = node while current and current.tag != "document": if identifier := current.attrs.get("id"): parts.append(f"#{identifier}") break parent = current.parent if parent is None: parts.append(current.tag) break siblings = [child for child in parent.children if isinstance(child, HtmlNode) and child.tag == current.tag] position = next(index for index, sibling in enumerate(siblings, 1) if sibling is current) parts.append(f"{current.tag}:nth-of-type({position})") current = parent return " > ".join(reversed(parts)) def placeholders(text: str) -> list[str]: return PLACEHOLDER_PATTERN.findall(text) def add_item(items: dict[str, dict[str, Any]], item_id: str, *, text: str, scope: str, source: str, kind: str, render: dict[str, Any] | None = None) -> None: if not text or item_id in items: return entry: dict[str, Any] = { "text": text, "scope": scope, "source": source, "kind": kind, "placeholders": placeholders(text), } if render: entry["render"] = render items[item_id] = entry def collect_html(html_path: Path, items: dict[str, dict[str, Any]], anchors: dict[str, dict[str, str]]) -> None: parser = TreeParser() parser.feed(html_path.read_text(encoding="utf-8")) for node in iter_nodes(parser.root): if node.tag in IGNORED_TAGS or copy_ignored(node): continue selector = selector_for(node) scope = nearest_scope(node) if node.tag in ANCHOR_TAGS: anchor_id = stable_id("anchor", selector) anchors[anchor_id] = {"selector": selector, "scope": scope, "tag": node.tag} text_position = 0 default_hidden_text = normalized_text(node.attrs.get("data-ui-copy-default-text", "")) if default_hidden_text: item_id = stable_id("copy", "index.html", selector, "text", text_position) add_item( items, item_id, text=default_hidden_text, scope=scope, source=f"index.html::{selector}::text[{text_position}]", kind="html_text", render={ "type": "static_text", "selector": selector, "text_index": text_position, "default_hidden": True, }, ) text_position += 1 for child in node.children: if not isinstance(child, str): continue text = normalized_text(child) if not text or text == "-" or FUNCTIONAL_VALUE_PATTERN.fullmatch(text): continue item_id = stable_id("copy", "index.html", selector, "text", text_position) add_item( items, item_id, text=text, scope=scope, source=f"index.html::{selector}::text[{text_position}]", kind="html_text", render={"type": "static_text", "selector": selector, "text_index": text_position}, ) text_position += 1 for attribute in TEXT_ATTRIBUTES: text = normalized_text(node.attrs.get(attribute, "")) if not text: continue item_id = stable_id("copy", "index.html", selector, "attribute", attribute) add_item( items, item_id, text=text, scope=scope, source=f"index.html::{selector}::@{attribute}", kind="html_attribute", render={"type": "attribute", "selector": selector, "attribute": attribute}, ) def scan_javascript_literals(source: str) -> Iterable[tuple[int, str, str]]: index = 0 while index < len(source): quote = source[index] if quote not in {'"', "'", "`"}: index += 1 continue start = index index += 1 value: list[str] = [] expression_depth = 0 placeholder_number = 0 while index < len(source): char = source[index] if char == "\\": if index + 1 < len(source): value.extend((char, source[index + 1])) index += 2 continue if quote == "`" and char == "$" and index + 1 < len(source) and source[index + 1] == "{": placeholder_number += 1 value.append(f"{{value{placeholder_number}}}") index += 2 expression_depth = 1 inner_quote: str | None = None while index < len(source) and expression_depth: current = source[index] if inner_quote: if current == "\\": index += 2 continue if current == inner_quote: inner_quote = None elif current in {'"', "'", "`"}: inner_quote = current elif current == "{": expression_depth += 1 elif current == "}": expression_depth -= 1 index += 1 continue if char == quote: index += 1 raw = "".join(value) raw = raw.replace("\\n", "\n").replace("\\r", "\r").replace("\\t", "\t") raw = raw.replace(f"\\{quote}", quote).replace("\\\\", "\\") yield start, quote, raw break value.append(char) index += 1 def collect_dynamic(items: dict[str, dict[str, Any]]) -> None: raw = json.loads(DYNAMIC_PATH.read_text(encoding="utf-8")) if not isinstance(raw, dict): raise ValueError("ui-copy-dynamic.json must be an object") for item_id, value in raw.items(): if not isinstance(value, dict) or set(value) - {"text", "scope", "source", "default_hidden"}: raise ValueError(f"invalid dynamic UI copy item: {item_id}") if not {"text", "scope", "source"} <= set(value) or type(value.get("default_hidden", False)) is not bool: raise ValueError(f"invalid dynamic UI copy item: {item_id}") add_item( items, item_id, text=normalized_text(value["text"]), scope=value["scope"], source=value["source"], kind="dynamic_template", render={"type": "dynamic_template", **({"default_hidden": True} if value.get("default_hidden") else {})}, ) def build_catalog() -> dict[str, Any]: items: dict[str, dict[str, Any]] = {} anchors: dict[str, dict[str, str]] = {} collect_html(STATIC_ROOT / "index.html", items, anchors) collect_dynamic(items) return { "schema_version": 2, "items": dict(sorted(items.items())), "anchors": dict(sorted(anchors.items())), "insertions": [], } def canonical_bytes(catalog: dict[str, Any]) -> bytes: return (json.dumps(catalog, ensure_ascii=False, indent=2, sort_keys=True) + "\n").encode("utf-8") def main() -> int: parser = argparse.ArgumentParser(description="Build or verify the UI copy catalog") parser.add_argument("--check", action="store_true", help="fail when ui-copy.json is not current") args = parser.parse_args() content = canonical_bytes(build_catalog()) if args.check: if not OUTPUT_PATH.is_file() or OUTPUT_PATH.read_bytes() != content: raise SystemExit("ui-copy.json is out of date; run scripts/build_ui_copy_catalog.py") print("ui-copy.json is current") return 0 OUTPUT_PATH.write_bytes(content) print(f"wrote {OUTPUT_PATH} ({len(json.loads(content)['items'])} items)") return 0 if __name__ == "__main__": raise SystemExit(main())