#!/usr/bin/env python3 """Lightweight HTML audit for common a11y/markup issues. Scans both *.html and *.php files. PHP blocks are stripped before HTML parsing so they don't confuse html.parser. """ from __future__ import annotations import re from html.parser import HTMLParser from pathlib import Path ROOT = Path(__file__).resolve().parents[1] SKIP_DIRS = {"partials", ".git", ".agents", ".claude", "tools", "pub", "docs", "locales", "admin", "api"} SKIP_FILES = { "test.html", "test_jp.html", "test_zh.html", "startpage/test.html", "vibe/admin_vibe.php", "changelog/admin.php", } # Replace and blocks with a placeholder before HTML # parsing. Using a non-empty placeholder (rather than "") preserves attributes # like title="" so they don't look empty to the audit. RE_PHP = re.compile(r"<\?.*?\?>", re.S) PHP_PLACEHOLDER = "phpval" # IDs that legitimately appear multiple times in a file because only one # instance is rendered at runtime (e.g. branched by `if ($lang === 'en')`). # Map of file -> {ids} to suppress the duplicate-ids check for. DUP_ID_ALLOWLIST: dict[str, set[str]] = { "index.php": {"guestbook-form-iframe", "guestbook-comments-iframe"}, } class AuditParser(HTMLParser): def __init__(self) -> None: super().__init__() self.ids: dict[str, int] = {} self.duplicate_ids: set[str] = set() self.missing_alt: list[str] = [] self.missing_iframe_title: list[str] = [] self.blank_rel: list[str] = [] def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: attr_map = {k.lower(): (v or "") for k, v in attrs} if "id" in attr_map: ident = attr_map["id"] if ident: if ident in self.ids: self.duplicate_ids.add(ident) self.ids[ident] = self.ids.get(ident, 0) + 1 if tag == "img": if "alt" not in attr_map: src = attr_map.get("src", "") self.missing_alt.append(src) if tag == "iframe": if not attr_map.get("title", ""): src = attr_map.get("src", "") self.missing_iframe_title.append(src) if tag == "a": if attr_map.get("target", "") == "_blank": rel = attr_map.get("rel", "") if "noopener" not in rel: href = attr_map.get("href", "") self.blank_rel.append(href) def main() -> int: issues = [] files: list[Path] = [] for ext in ("*.html", "*.php"): files.extend(ROOT.rglob(ext)) for src_file in files: if any(part in SKIP_DIRS for part in src_file.parts): continue rel = src_file.relative_to(ROOT).as_posix() if rel in SKIP_FILES: continue text = src_file.read_text(encoding="utf-8", errors="ignore") text = RE_PHP.sub(PHP_PLACEHOLDER, text) parser = AuditParser() parser.feed(text) dup_ids = parser.duplicate_ids - DUP_ID_ALLOWLIST.get(rel, set()) if dup_ids: issues.append((rel, "duplicate-ids", sorted(dup_ids))) if parser.missing_alt: issues.append((rel, "img-missing-alt", parser.missing_alt)) if parser.missing_iframe_title: issues.append((rel, "iframe-missing-title", parser.missing_iframe_title)) if parser.blank_rel: issues.append((rel, "target-blank-missing-noopener", parser.blank_rel)) if not issues: print("OK: no audit issues found") return 0 print("HTML audit issues:") for rel, kind, items in issues: print(f"- {rel}: {kind}") for item in items[:10]: print(f" - {item}") if len(items) > 10: print(f" - ... ({len(items) - 10} more)") return 1 if __name__ == "__main__": raise SystemExit(main())