← Files PresentonARCHIVED FILE

skills/presenton/scripts/validate_html.py

8.76 KB · Oct 2, 2026 · 00:32 UTC

↓ Download file

#!/usr/bin/env python3
"""Validate the structural contract required by Presenton's html-to-any exporter."""

from __future__ import annotations

import argparse
import re
import sys
from html.parser import HTMLParser
from pathlib import Path
from urllib.parse import parse_qs, unquote_plus, urlsplit


GENERIC_FONT_FAMILIES = {
    "cursive",
    "fantasy",
    "fangsong",
    "inherit",
    "initial",
    "math",
    "monospace",
    "revert",
    "sans-serif",
    "serif",
    "system-ui",
    "ui-monospace",
    "ui-rounded",
    "ui-sans-serif",
    "ui-serif",
    "unset",
}
LOCAL_SYSTEM_FONT_FAMILIES = {
    "arial",
    "arial black",
    "calibri",
    "cambria",
    "candara",
    "comic sans ms",
    "consolas",
    "courier",
    "courier new",
    "dejavu sans",
    "dejavu serif",
    "georgia",
    "helvetica",
    "impact",
    "liberation mono",
    "lucida console",
    "menlo",
    "monaco",
    "segoe ui",
    "sfmono-regular",
    "tahoma",
    "times",
    "times new roman",
    "trebuchet ms",
    "verdana",
}


class PresentationParser(HTMLParser):
    def __init__(self) -> None:
        super().__init__(convert_charrefs=True)
        self.stack: list[tuple[str, bool]] = []
        self.wrapper_count = 0
        self.wrapper_depth: int | None = None
        self.slide_count = 0
        self.direct_text = False

    def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
        attributes = dict(attrs)
        is_wrapper = attributes.get("id") == "presentation-slides-wrapper"
        if is_wrapper:
            self.wrapper_count += 1
            if self.wrapper_depth is None:
                self.wrapper_depth = len(self.stack)
        elif self.wrapper_depth is not None and len(self.stack) == self.wrapper_depth + 1:
            self.slide_count += 1
        self.stack.append((tag, is_wrapper))

    def handle_startendtag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
        self.handle_starttag(tag, attrs)
        self.handle_endtag(tag)

    def handle_endtag(self, tag: str) -> None:
        if not self.stack:
            return
        popped_tag, was_wrapper = self.stack.pop()
        if was_wrapper:
            self.wrapper_depth = None
        elif popped_tag != tag:
            # HTMLParser is permissive. The browser will repair malformed markup,
            # so leave detailed nesting diagnostics to a browser-based inspection.
            return

    def handle_data(self, data: str) -> None:
        if (
            self.wrapper_depth is not None
            and len(self.stack) == self.wrapper_depth + 1
            and data.strip()
        ):
            self.direct_text = True


class FontMarkupParser(HTMLParser):
    def __init__(self) -> None:
        super().__init__(convert_charrefs=True)
        self.arbitrary_families: list[str] = []

    def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
        classes = dict(attrs).get("class") or ""
        for class_name in classes.split():
            if class_name.startswith("font-[") and class_name.endswith("]"):
                family = class_name[6:-1].strip("'\"").replace("_", " ")
                if family:
                    self.arbitrary_families.append(family)


def split_font_stack(value: str) -> list[str]:
    return [part.strip().strip("'\"") for part in value.split(",") if part.strip()]


def normalize_font_name(name: str) -> str:
    return re.sub(r"\s+", " ", name.strip()).casefold()


def custom_font_names(values: list[str]) -> set[str]:
    names: set[str] = set()
    for value in values:
        for family in split_font_stack(value):
            normalized = normalize_font_name(family)
            if (
                normalized
                and normalized not in GENERIC_FONT_FAMILIES
                and normalized not in LOCAL_SYSTEM_FONT_FAMILIES
                and not normalized.startswith(("var(", "--"))
            ):
                names.add(normalized)
    return names


def validate_font_loading(html: str) -> list[str]:
    """Require custom fonts used in slide markup to be declared by head imports."""

    head_match = re.search(r"<head\b[^>]*>(.*?)</head\s*>", html, re.IGNORECASE | re.DOTALL)
    body_match = re.search(r"<body\b[^>]*>(.*?)</body\s*>", html, re.IGNORECASE | re.DOTALL)
    if not head_match or not body_match:
        return []

    head = head_match.group(1)
    body = body_match.group(1)
    used_values = re.findall(r"font-family\s*:\s*([^;}\n]+)", body, re.IGNORECASE)
    parser = FontMarkupParser()
    parser.feed(body)
    parser.close()
    used_names = custom_font_names(used_values + parser.arbitrary_families)
    if not used_names:
        return []

    imported_names: set[str] = set()
    for encoded_family in re.findall(r"[?&]family=([^&#'\"\s)]+)", head, re.IGNORECASE):
        imported_names.add(
            normalize_font_name(unquote_plus(encoded_family).split(":", 1)[0].replace("+", " "))
        )

    # Support an inline @font-face declaration if a future format relaxes the
    # no-style-block rule. A declaration is only valid when it has a remote or
    # HTTPS source, so a bare font-family name is never treated as a load.
    for block in re.findall(r"@font-face\s*\{(.*?)\}", head, re.IGNORECASE | re.DOTALL):
        if re.search(r"\bsrc\s*:\s*[^;}]*https://", block, re.IGNORECASE):
            family_match = re.search(r"font-family\s*:\s*([^;}\n]+)", block, re.IGNORECASE)
            if family_match:
                imported_names.update(custom_font_names([family_match.group(1)]))

    missing = sorted(name for name in used_names if name not in imported_names)
    return [
        "Custom font '{}' is used in slide markup but is not imported in <head>; "
        "add a matching absolute HTTPS stylesheet link.".format(name)
        for name in missing
    ]


def validate_html(html: str) -> list[str]:
    errors: list[str] = []
    lowered = html.lower()
    for required in ("<!doctype html", "<html", "<head", "<body"):
        if required not in lowered:
            errors.append(f"Missing required document marker: {required}.")

    if "https://cdn.tailwindcss.com" not in lowered:
        errors.append("Missing required Tailwind CDN script.")
    if re.search(r"\sstyle\s*=", html, re.I):
        errors.append("Use Tailwind classes instead of inline style attributes.")
    if re.search(r"<style(?:\s|>)", html, re.I):
        errors.append("Use Tailwind classes instead of embedded style blocks.")
    if "<canvas" in lowered and "https://cdn.jsdelivr.net/npm/chart.js" not in lowered:
        errors.append("Chart canvases require the Chart.js CDN script.")
    if re.search(r"data:[a-z]+/[a-z0-9.+-]+(?:;[^,'\"\s]+)*,", html, re.I):
        errors.append(
            "Do not embed data/base64 URLs; upload images first and use absolute HTTPS URLs."
        )

    parser = PresentationParser()
    try:
        parser.feed(html)
        parser.close()
    except Exception as exc:  # pragma: no cover - defensive parser boundary
        errors.append(f"HTML parsing failed: {exc}")

    if parser.wrapper_count != 1:
        errors.append(
            "Expected exactly one element with id='presentation-slides-wrapper'; "
            f"found {parser.wrapper_count}."
        )
    if parser.slide_count < 1:
        errors.append("The presentation wrapper must have at least one direct element child.")
    if parser.direct_text:
        errors.append("The presentation wrapper contains non-whitespace text outside slide elements.")

    width_signal = re.search(r"(?:width\s*:\s*1280px|w-\[1280px\])", html, re.I)
    height_signal = re.search(r"(?:height\s*:\s*720px|h-\[720px\])", html, re.I)
    if not width_signal or not height_signal:
        errors.append("Missing a recognizable 1280×720 px slide dimension rule.")

    local_sources = re.findall(
        r"(?:src|href)\s*=\s*['\"](?!https://|data:|#|mailto:|tel:)([^'\"]+)['\"]",
        html,
        re.I,
    )
    if local_sources:
        examples = ", ".join(local_sources[:3])
        errors.append(f"Use absolute HTTPS URLs instead of local/relative assets: {examples}")

    errors.extend(validate_font_loading(html))

    return errors


def main() -> int:
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("html", type=Path, help="HTML presentation file")
    args = parser.parse_args()

    try:
        source = args.html.read_text(encoding="utf-8")
    except OSError as exc:
        print(f"ERROR: Could not read {args.html}: {exc}", file=sys.stderr)
        return 2

    errors = validate_html(source)
    if errors:
        for error in errors:
            print(f"ERROR: {error}", file=sys.stderr)
        return 1

    parsed = PresentationParser()
    parsed.feed(source)
    print(f"OK: {args.html} contains {parsed.slide_count} slide(s).")
    return 0


if __name__ == "__main__":
    raise SystemExit(main())

SHA-256: f83f6b6d4b83a022a310070c7e41d504bc61adf4b1524626f7aea3b3e4b40609