aboutsummaryrefslogtreecommitdiffstats
path: root/tools/html_audit.py
diff options
context:
space:
mode:
Diffstat (limited to 'tools/html_audit.py')
-rw-r--r--tools/html_audit.py123
1 files changed, 123 insertions, 0 deletions
diff --git a/tools/html_audit.py b/tools/html_audit.py
new file mode 100644
index 0000000..e4dad11
--- /dev/null
+++ b/tools/html_audit.py
@@ -0,0 +1,123 @@
+#!/usr/bin/env python3
+"""Lightweight HTML audit for common a11y/markup issues.
+
+Scans both *.html and *.php files. PHP blocks are stripped before HTML parsing
+so they don't confuse html.parser.
+"""
+
+from __future__ import annotations
+
+import re
+from html.parser import HTMLParser
+from pathlib import Path
+
+ROOT = Path(__file__).resolve().parents[1]
+
+SKIP_DIRS = {"partials", ".git", ".agents", ".claude", "tools",
+ "pub", "docs", "locales", "admin", "api"}
+SKIP_FILES = {
+ "test.html",
+ "test_jp.html",
+ "test_zh.html",
+ "startpage/test.html",
+ "vibe/admin_vibe.php",
+ "changelog/admin.php",
+}
+
+# Replace <?php ... ?> and <?= ... ?> blocks with a placeholder before HTML
+# parsing. Using a non-empty placeholder (rather than "") preserves attributes
+# like title="<?= ... ?>" so they don't look empty to the audit.
+RE_PHP = re.compile(r"<\?.*?\?>", re.S)
+PHP_PLACEHOLDER = "phpval"
+
+# IDs that legitimately appear multiple times in a file because only one
+# instance is rendered at runtime (e.g. branched by `if ($lang === 'en')`).
+# Map of file -> {ids} to suppress the duplicate-ids check for.
+DUP_ID_ALLOWLIST: dict[str, set[str]] = {
+ "index.php": {"guestbook-form-iframe", "guestbook-comments-iframe"},
+}
+
+
+class AuditParser(HTMLParser):
+ def __init__(self) -> None:
+ super().__init__()
+ self.ids: dict[str, int] = {}
+ self.duplicate_ids: set[str] = set()
+ self.missing_alt: list[str] = []
+ self.missing_iframe_title: list[str] = []
+ self.blank_rel: list[str] = []
+
+ def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
+ attr_map = {k.lower(): (v or "") for k, v in attrs}
+
+ if "id" in attr_map:
+ ident = attr_map["id"]
+ if ident:
+ if ident in self.ids:
+ self.duplicate_ids.add(ident)
+ self.ids[ident] = self.ids.get(ident, 0) + 1
+
+ if tag == "img":
+ if "alt" not in attr_map:
+ src = attr_map.get("src", "")
+ self.missing_alt.append(src)
+
+ if tag == "iframe":
+ if not attr_map.get("title", ""):
+ src = attr_map.get("src", "")
+ self.missing_iframe_title.append(src)
+
+ if tag == "a":
+ if attr_map.get("target", "") == "_blank":
+ rel = attr_map.get("rel", "")
+ if "noopener" not in rel:
+ href = attr_map.get("href", "")
+ self.blank_rel.append(href)
+
+
+def main() -> int:
+ issues = []
+
+ files: list[Path] = []
+ for ext in ("*.html", "*.php"):
+ files.extend(ROOT.rglob(ext))
+
+ for src_file in files:
+ if any(part in SKIP_DIRS for part in src_file.parts):
+ continue
+ rel = src_file.relative_to(ROOT).as_posix()
+ if rel in SKIP_FILES:
+ continue
+
+ text = src_file.read_text(encoding="utf-8", errors="ignore")
+ text = RE_PHP.sub(PHP_PLACEHOLDER, text)
+
+ parser = AuditParser()
+ parser.feed(text)
+
+ dup_ids = parser.duplicate_ids - DUP_ID_ALLOWLIST.get(rel, set())
+ if dup_ids:
+ issues.append((rel, "duplicate-ids", sorted(dup_ids)))
+ if parser.missing_alt:
+ issues.append((rel, "img-missing-alt", parser.missing_alt))
+ if parser.missing_iframe_title:
+ issues.append((rel, "iframe-missing-title", parser.missing_iframe_title))
+ if parser.blank_rel:
+ issues.append((rel, "target-blank-missing-noopener", parser.blank_rel))
+
+ if not issues:
+ print("OK: no audit issues found")
+ return 0
+
+ print("HTML audit issues:")
+ for rel, kind, items in issues:
+ print(f"- {rel}: {kind}")
+ for item in items[:10]:
+ print(f" - {item}")
+ if len(items) > 10:
+ print(f" - ... ({len(items) - 10} more)")
+ return 1
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())