From 898b52edcb47bcb3e9d6106e74ca73e74ea01e70 Mon Sep 17 00:00:00 2001 From: sillylaird Date: Thu, 3 Sep 2026 00:33:59 +0000 Subject: import live www.sillylaird.ca webroot --- tools/html_audit.py | 123 ++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 123 insertions(+) create mode 100644 tools/html_audit.py (limited to 'tools/html_audit.py') diff --git a/tools/html_audit.py b/tools/html_audit.py new file mode 100644 index 0000000..e4dad11 --- /dev/null +++ b/tools/html_audit.py @@ -0,0 +1,123 @@ +#!/usr/bin/env python3 +"""Lightweight HTML audit for common a11y/markup issues. + +Scans both *.html and *.php files. PHP blocks are stripped before HTML parsing +so they don't confuse html.parser. +""" + +from __future__ import annotations + +import re +from html.parser import HTMLParser +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] + +SKIP_DIRS = {"partials", ".git", ".agents", ".claude", "tools", + "pub", "docs", "locales", "admin", "api"} +SKIP_FILES = { + "test.html", + "test_jp.html", + "test_zh.html", + "startpage/test.html", + "vibe/admin_vibe.php", + "changelog/admin.php", +} + +# Replace and blocks with a placeholder before HTML +# parsing. Using a non-empty placeholder (rather than "") preserves attributes +# like title="" so they don't look empty to the audit. +RE_PHP = re.compile(r"<\?.*?\?>", re.S) +PHP_PLACEHOLDER = "phpval" + +# IDs that legitimately appear multiple times in a file because only one +# instance is rendered at runtime (e.g. branched by `if ($lang === 'en')`). +# Map of file -> {ids} to suppress the duplicate-ids check for. +DUP_ID_ALLOWLIST: dict[str, set[str]] = { + "index.php": {"guestbook-form-iframe", "guestbook-comments-iframe"}, +} + + +class AuditParser(HTMLParser): + def __init__(self) -> None: + super().__init__() + self.ids: dict[str, int] = {} + self.duplicate_ids: set[str] = set() + self.missing_alt: list[str] = [] + self.missing_iframe_title: list[str] = [] + self.blank_rel: list[str] = [] + + def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: + attr_map = {k.lower(): (v or "") for k, v in attrs} + + if "id" in attr_map: + ident = attr_map["id"] + if ident: + if ident in self.ids: + self.duplicate_ids.add(ident) + self.ids[ident] = self.ids.get(ident, 0) + 1 + + if tag == "img": + if "alt" not in attr_map: + src = attr_map.get("src", "") + self.missing_alt.append(src) + + if tag == "iframe": + if not attr_map.get("title", ""): + src = attr_map.get("src", "") + self.missing_iframe_title.append(src) + + if tag == "a": + if attr_map.get("target", "") == "_blank": + rel = attr_map.get("rel", "") + if "noopener" not in rel: + href = attr_map.get("href", "") + self.blank_rel.append(href) + + +def main() -> int: + issues = [] + + files: list[Path] = [] + for ext in ("*.html", "*.php"): + files.extend(ROOT.rglob(ext)) + + for src_file in files: + if any(part in SKIP_DIRS for part in src_file.parts): + continue + rel = src_file.relative_to(ROOT).as_posix() + if rel in SKIP_FILES: + continue + + text = src_file.read_text(encoding="utf-8", errors="ignore") + text = RE_PHP.sub(PHP_PLACEHOLDER, text) + + parser = AuditParser() + parser.feed(text) + + dup_ids = parser.duplicate_ids - DUP_ID_ALLOWLIST.get(rel, set()) + if dup_ids: + issues.append((rel, "duplicate-ids", sorted(dup_ids))) + if parser.missing_alt: + issues.append((rel, "img-missing-alt", parser.missing_alt)) + if parser.missing_iframe_title: + issues.append((rel, "iframe-missing-title", parser.missing_iframe_title)) + if parser.blank_rel: + issues.append((rel, "target-blank-missing-noopener", parser.blank_rel)) + + if not issues: + print("OK: no audit issues found") + return 0 + + print("HTML audit issues:") + for rel, kind, items in issues: + print(f"- {rel}: {kind}") + for item in items[:10]: + print(f" - {item}") + if len(items) > 10: + print(f" - ... ({len(items) - 10} more)") + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) -- cgit v1.2.3