diff options
Diffstat (limited to '')
| -rw-r--r-- | tools/README.md | 16 | ||||
| -rwxr-xr-x | tools/backup.sh | 13 | ||||
| -rw-r--r-- | tools/generate_sitemap.py | 160 | ||||
| -rw-r--r-- | tools/html_audit.py | 123 | ||||
| -rwxr-xr-x | tools/image_budget.py | 72 | ||||
| -rw-r--r-- | tools/link_check.py | 108 | ||||
| -rwxr-xr-x | tools/link_rot.py | 135 | ||||
| -rwxr-xr-x | tools/optimize_images.sh | 93 | ||||
| -rwxr-xr-x | tools/run_audits.sh | 52 | ||||
| -rwxr-xr-x | tools/security_check.sh | 62 | ||||
| -rw-r--r-- | tools/translate_pages.py | 258 | ||||
| -rwxr-xr-x | tools/update_openbsd_version.sh | 38 | ||||
| -rwxr-xr-x | tools/uptime_check.sh | 6 |
13 files changed, 1136 insertions, 0 deletions
diff --git a/tools/README.md b/tools/README.md new file mode 100644 index 0000000..4c13824 --- /dev/null +++ b/tools/README.md @@ -0,0 +1,16 @@ +# Tools + +Small maintenance helpers. No generator required. See also the root `README.md`. + +- `python3 tools/link_check.py` — internal href/src targets exist +- `python3 tools/html_audit.py` — quick a11y/markup audit +- `python3 tools/generate_sitemap.py` — rebuild `sitemap.xml` with lastmod + hreflang +- `python3 tools/translate_pages.py` — generate missing `*_zh` / `*_jp` page stubs +- `python3 tools/link_rot.py` — external link rot checker +- `python3 tools/image_budget.py` — flag oversized images (scans assets, startpage, gaming, …) +- `./tools/optimize_images.sh` — recompress/resize images (backs up first; `DRY_RUN=1` to preview) +- `./tools/run_audits.sh` — audits + sitemap together (cron-friendly; `STRICT=1` to fail on findings) +- `./tools/security_check.sh` — response headers + session cookie flags +- `./tools/backup.sh` — tar.gz backup (`BACKUP_DIR` override) +- `./tools/uptime_check.sh` — curl uptime check (`URL` override) +- `./tools/update_openbsd_version.sh` — scrape openbsd.org current release into `startpage/openbsd_version.txt` (cron; boot banner) diff --git a/tools/backup.sh b/tools/backup.sh new file mode 100755 index 0000000..36790bf --- /dev/null +++ b/tools/backup.sh @@ -0,0 +1,13 @@ +#!/usr/bin/env bash +set -euo pipefail + +ROOT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +BACKUP_DIR="${BACKUP_DIR:-/tmp/www.sillylaird.ca-backups}" + +mkdir -p "$BACKUP_DIR" +TS=$(date -u +"%Y%m%d-%H%M%S") +ARCHIVE="$BACKUP_DIR/www.sillylaird.ca-$TS.tar.gz" + +tar -czf "$ARCHIVE" -C "$ROOT_DIR" . + +echo "Backup written to $ARCHIVE" diff --git a/tools/generate_sitemap.py b/tools/generate_sitemap.py new file mode 100644 index 0000000..ab7e25d --- /dev/null +++ b/tools/generate_sitemap.py @@ -0,0 +1,160 @@ +#!/usr/bin/env python3 +"""Generate sitemap.xml from local PHP/HTML pages. + +The site is PHP-based with the convention that translated copies live next to +the canonical file as `foo_jp.php` and `foo_zh.php`. When a triple exists, all +three URLs go into the sitemap with an `xhtml:link` hreflang block so search +engines understand the relationship. + +Standalone pages get a plain `<url>` entry. +""" + +from __future__ import annotations + +from collections import defaultdict +from pathlib import Path +from datetime import datetime, timezone + +ROOT = Path(__file__).resolve().parents[1] +SITE = "https://www.sillylaird.ca" + +# Directories that never produce navigable URLs. +EXCLUDE_DIRS = { + ".git", ".agents", ".claude", + "partials", "admin", "api", "locales", + "tools", "pub", "docs", + # Legacy redirect-only path; canonical is /startpage/ + "mstartpage", +} + +# Specific files that are not user-navigable pages: error pages, proxies, +# iframe-only components, internal endpoints, config. +EXCLUDE_FILES = { + "404.php", "50x.php", + "gemini-proxy.php", "lastfm-proxy.php", + "guestbook-form.php", "guestbook-comments.php", + "submit.php", "bbs.php", "bbs.cgi", + "vibe.php", + "feed.xml.php", +} + +# Files matching these path suffixes are excluded (admin/internal endpoints). +EXCLUDE_PATH_SUFFIXES = { + "changelog/admin.php", + "changelog/api/latest.php", + "changelog/auth.php", + "changelog/db.php", + "changelog/latest.php", + "changelog/login.php", + "changelog/logout.php", + "vibe/admin.php", + "vibe/admin_vibe.php", + "gaming/collection/admin.php", + "gaming/collection/_lib.php", + "whatsleft/privacy/index.php", +} + + +def should_skip(path: Path) -> bool: + parts = set(path.relative_to(ROOT).parts) + if parts & EXCLUDE_DIRS: + return True + if path.name in EXCLUDE_FILES: + return True + rel = path.relative_to(ROOT).as_posix() + if rel in EXCLUDE_PATH_SUFFIXES: + return True + if path.name.endswith("~"): + return True + return False + + +def url_for(path: Path) -> str: + rel = path.relative_to(ROOT).as_posix() + if rel == "index.php" or rel == "index.html": + return SITE + "/" + if rel.endswith("/index.php"): + return SITE + "/" + rel[: -len("index.php")] + if rel.endswith("/index.html"): + return SITE + "/" + rel[: -len("index.html")] + return SITE + "/" + rel + + +def lastmod_for(path: Path) -> str: + ts = path.stat().st_mtime + return datetime.fromtimestamp(ts, tz=timezone.utc).strftime("%Y-%m-%d") + + +def split_lang(name: str) -> tuple[str, str]: + """Return (base, lang). For foo_jp.php returns ('foo.php', 'ja').""" + for suffix, lang in (("_jp.php", "ja"), ("_zh.php", "zh"), + ("_jp.html", "ja"), ("_zh.html", "zh")): + if name.endswith(suffix): + stem = name[: -len(suffix)] + ext = ".php" if suffix.endswith(".php") else ".html" + return stem + ext, lang + # also handle index_jp.php -> index.php family + return name, "en" + + +def main() -> int: + files: list[Path] = [] + for ext in ("*.php", "*.html"): + for p in ROOT.rglob(ext): + if should_skip(p): + continue + files.append(p) + + # Group by (parent_dir, canonical_name) so we can emit hreflang for triples. + families: dict[tuple[Path, str], dict[str, Path]] = defaultdict(dict) + for p in files: + canonical_name, lang = split_lang(p.name) + families[(p.parent, canonical_name)][lang] = p + + lines = [ + '<?xml version="1.0" encoding="UTF-8"?>', + '<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"', + ' xmlns:xhtml="http://www.w3.org/1999/xhtml">', + ] + + # Emit URLs sorted for stable diffs. + keyed: list[tuple[str, str, dict[str, Path]]] = [] + for (parent, canonical_name), variants in families.items(): + # Pick the URL of the canonical (en) entry if present, otherwise any. + anchor_path = variants.get("en") or next(iter(variants.values())) + anchor_url = url_for(anchor_path) + keyed.append((anchor_url, canonical_name, variants)) + keyed.sort(key=lambda x: x[0]) + + for _anchor_url, _canonical_name, variants in keyed: + is_triple = len(variants) > 1 + for lang in ("en", "ja", "zh"): + if lang not in variants: + continue + p = variants[lang] + loc = url_for(p) + lines.append(" <url>") + lines.append(f" <loc>{loc}</loc>") + lines.append(f" <lastmod>{lastmod_for(p)}</lastmod>") + if is_triple: + en_url = url_for(variants["en"]) if "en" in variants else loc + for hl in ("en", "ja", "zh"): + if hl in variants: + lines.append( + f' <xhtml:link rel="alternate" hreflang="{hl}" ' + f'href="{url_for(variants[hl])}"/>' + ) + lines.append( + f' <xhtml:link rel="alternate" hreflang="x-default" href="{en_url}"/>' + ) + lines.append(" </url>") + + lines.append("</urlset>") + + (ROOT / "sitemap.xml").write_text("\n".join(lines) + "\n", encoding="utf-8") + print(f"wrote sitemap.xml with {sum(len(v) for v in families.values())} URLs") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/html_audit.py b/tools/html_audit.py new file mode 100644 index 0000000..e4dad11 --- /dev/null +++ b/tools/html_audit.py @@ -0,0 +1,123 @@ +#!/usr/bin/env python3 +"""Lightweight HTML audit for common a11y/markup issues. + +Scans both *.html and *.php files. PHP blocks are stripped before HTML parsing +so they don't confuse html.parser. +""" + +from __future__ import annotations + +import re +from html.parser import HTMLParser +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] + +SKIP_DIRS = {"partials", ".git", ".agents", ".claude", "tools", + "pub", "docs", "locales", "admin", "api"} +SKIP_FILES = { + "test.html", + "test_jp.html", + "test_zh.html", + "startpage/test.html", + "vibe/admin_vibe.php", + "changelog/admin.php", +} + +# Replace <?php ... ?> and <?= ... ?> blocks with a placeholder before HTML +# parsing. Using a non-empty placeholder (rather than "") preserves attributes +# like title="<?= ... ?>" so they don't look empty to the audit. +RE_PHP = re.compile(r"<\?.*?\?>", re.S) +PHP_PLACEHOLDER = "phpval" + +# IDs that legitimately appear multiple times in a file because only one +# instance is rendered at runtime (e.g. branched by `if ($lang === 'en')`). +# Map of file -> {ids} to suppress the duplicate-ids check for. +DUP_ID_ALLOWLIST: dict[str, set[str]] = { + "index.php": {"guestbook-form-iframe", "guestbook-comments-iframe"}, +} + + +class AuditParser(HTMLParser): + def __init__(self) -> None: + super().__init__() + self.ids: dict[str, int] = {} + self.duplicate_ids: set[str] = set() + self.missing_alt: list[str] = [] + self.missing_iframe_title: list[str] = [] + self.blank_rel: list[str] = [] + + def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: + attr_map = {k.lower(): (v or "") for k, v in attrs} + + if "id" in attr_map: + ident = attr_map["id"] + if ident: + if ident in self.ids: + self.duplicate_ids.add(ident) + self.ids[ident] = self.ids.get(ident, 0) + 1 + + if tag == "img": + if "alt" not in attr_map: + src = attr_map.get("src", "") + self.missing_alt.append(src) + + if tag == "iframe": + if not attr_map.get("title", ""): + src = attr_map.get("src", "") + self.missing_iframe_title.append(src) + + if tag == "a": + if attr_map.get("target", "") == "_blank": + rel = attr_map.get("rel", "") + if "noopener" not in rel: + href = attr_map.get("href", "") + self.blank_rel.append(href) + + +def main() -> int: + issues = [] + + files: list[Path] = [] + for ext in ("*.html", "*.php"): + files.extend(ROOT.rglob(ext)) + + for src_file in files: + if any(part in SKIP_DIRS for part in src_file.parts): + continue + rel = src_file.relative_to(ROOT).as_posix() + if rel in SKIP_FILES: + continue + + text = src_file.read_text(encoding="utf-8", errors="ignore") + text = RE_PHP.sub(PHP_PLACEHOLDER, text) + + parser = AuditParser() + parser.feed(text) + + dup_ids = parser.duplicate_ids - DUP_ID_ALLOWLIST.get(rel, set()) + if dup_ids: + issues.append((rel, "duplicate-ids", sorted(dup_ids))) + if parser.missing_alt: + issues.append((rel, "img-missing-alt", parser.missing_alt)) + if parser.missing_iframe_title: + issues.append((rel, "iframe-missing-title", parser.missing_iframe_title)) + if parser.blank_rel: + issues.append((rel, "target-blank-missing-noopener", parser.blank_rel)) + + if not issues: + print("OK: no audit issues found") + return 0 + + print("HTML audit issues:") + for rel, kind, items in issues: + print(f"- {rel}: {kind}") + for item in items[:10]: + print(f" - {item}") + if len(items) > 10: + print(f" - ... ({len(items) - 10} more)") + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/image_budget.py b/tools/image_budget.py new file mode 100755 index 0000000..f8fc822 --- /dev/null +++ b/tools/image_budget.py @@ -0,0 +1,72 @@ +#!/usr/bin/env python3 +"""Image size budget check. + +Flags images that exceed a per-file byte budget so heavy assets don't creep +back in after an optimization pass. Offline and fast — safe for the default +audit run. + +Budgets (override on the CLI): + raster (png/jpg/jpeg/webp) : 250 KB + gif : 600 KB (animations are bigger) + +Usage: + python tools/image_budget.py [--max-kb 250] [--gif-max-kb 600] +""" + +from __future__ import annotations + +import argparse +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +# Include section asset dirs that hold page-local images (not only /assets/img). +SCAN_DIRS = ( + "assets/img", + "images", + "startpage", + "computers", + "gaming", + "vblog", + "mstartpage", +) +RASTER = {".png", ".jpg", ".jpeg", ".webp"} + + +def main() -> int: + ap = argparse.ArgumentParser(description="Image size budget check") + ap.add_argument("--max-kb", type=int, default=250) + ap.add_argument("--gif-max-kb", type=int, default=600) + args = ap.parse_args() + + over: list[tuple[int, str, int]] = [] # (size, path, budget) + for d in SCAN_DIRS: + base = ROOT / d + if not base.is_dir(): + continue + for f in base.rglob("*"): + if not f.is_file(): + continue + ext = f.suffix.lower() + if ext == ".gif": + budget = args.gif_max_kb * 1024 + elif ext in RASTER: + budget = args.max_kb * 1024 + else: + continue + sz = f.stat().st_size + if sz > budget: + over.append((sz, f.relative_to(ROOT).as_posix(), budget)) + + if over: + print("Images over budget:") + for sz, path, budget in sorted(over, reverse=True): + print(f"- {path}: {sz/1024:.0f} KB (budget {budget//1024} KB)") + print("\nTip: run ./tools/optimize_images.sh") + return 1 + + print("OK: all images within budget") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/link_check.py b/tools/link_check.py new file mode 100644 index 0000000..a0d9e27 --- /dev/null +++ b/tools/link_check.py @@ -0,0 +1,108 @@ +#!/usr/bin/env python3 +"""Very small internal link checker. + +Checks: +- href="/path" and src="/path" for local files +- Only checks local paths (starting with / or relative), skips http(s), mailto, xmpp, onion, etc. + +Scans both *.html and *.php files. PHP blocks are scanned as raw text — the +href/src regex doesn't care about surrounding PHP syntax. + +Usage: + python tools/link_check.py +""" + +from __future__ import annotations + +import re +from pathlib import Path + + +ROOT = Path(__file__).resolve().parents[1] + +SKIP_DIRS = {".git", ".agents", ".claude", "tools", "pub", "docs", + "admin", "api", "locales", "partials"} +SKIP_FILES = { + "test.html", + "test_jp.html", + "test_zh.html", + "startpage/test.html", +} + +RE_URL = re.compile(r"\b(?:href|src)=(['\"])(.*?)\1", re.I) +# Match URLs that look like dynamic PHP output: anything containing <?...?>. +RE_DYNAMIC = re.compile(r"<\?.*?\?>", re.S) + + +def is_external(u: str) -> bool: + u = u.strip() + return ( + u.startswith("http://") + or u.startswith("https://") + or u.startswith("mailto:") + or u.startswith("xmpp:") + or u.startswith("signal:") + or u.startswith("data:") + or u.startswith("javascript:") + or u.startswith("gopher:") + or u.startswith("gemini:") + or u.startswith("#") + or u.startswith("//") + or u.endswith(".onion/") + or ".onion" in u + ) + + +def normalize(p: Path, url: str) -> Path | None: + url = url.split("#", 1)[0].split("?", 1)[0].strip() + if not url: + return None + if is_external(url): + return None + + if url.startswith("/"): + return (ROOT / url.lstrip("/")).resolve() + + # relative + return (p.parent / url).resolve() + + +def main() -> int: + missing = [] + files: list[Path] = [] + for ext in ("*.html", "*.php"): + files.extend(ROOT.rglob(ext)) + for src_file in files: + if any(part in SKIP_DIRS for part in src_file.parts): + continue + rel = src_file.relative_to(ROOT).as_posix() + if rel in SKIP_FILES: + continue + text = src_file.read_text(encoding="utf-8", errors="ignore") + for m in RE_URL.finditer(text): + url = m.group(2) + # Skip URLs that contain PHP expressions; they're computed at runtime. + if RE_DYNAMIC.search(url): + continue + target = normalize(src_file, url) + if not target: + continue + # if it points to a directory, allow index.html or index.php + if target.is_dir(): + if (target / "index.html").exists() or (target / "index.php").exists(): + continue + if not target.exists(): + missing.append((str(src_file.relative_to(ROOT)), url)) + + if missing: + print("Missing local links:") + for src, url in missing: + print(f"- {src}: {url}") + return 1 + + print("OK: no missing local href/src found") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/link_rot.py b/tools/link_rot.py new file mode 100755 index 0000000..b3a7613 --- /dev/null +++ b/tools/link_rot.py @@ -0,0 +1,135 @@ +#!/usr/bin/env python3 +"""External link rot checker. + +Collects every external http(s) link from the site's *.php / *.html files, +dedupes them, and checks each one over the network. Reports links that are +dead (connection error, timeout) or return a 4xx/5xx status. + +Network-bound and slow-ish, so it is NOT part of the default audit run. Wire +it in with CHECK_LINKS=1 ./tools/run_audits.sh, or run it directly. + +Skips mailto/xmpp/onion/i2p and other non-web schemes. Treats 401/403/405 and +429 as "alive but gated" by default (many sites block bots) — pass --strict to +flag those too. + +Usage: + python tools/link_rot.py [--timeout 10] [--workers 16] [--strict] +""" + +from __future__ import annotations + +import argparse +import re +import sys +from concurrent.futures import ThreadPoolExecutor, as_completed +from pathlib import Path +from urllib.request import Request, urlopen +from urllib.error import HTTPError, URLError + +ROOT = Path(__file__).resolve().parents[1] + +SKIP_DIRS = {".git", ".agents", ".claude", "tools", "docs", "locales", "partials"} + +RE_URL = re.compile(r"\b(?:href|src)=(['\"])(https?://.*?)\1", re.I) +RE_DYNAMIC = re.compile(r"<\?.*?\?>", re.S) + +# Hosts that are part of this site / known-internal; no point rot-checking. +SKIP_HOST_SUBSTR = (".onion", ".i2p", "localhost", "127.0.0.1") + +# Statuses that mean "reachable but bot-gated" — not rot unless --strict. +SOFT_STATUSES = {401, 403, 405, 429} + +UA = "Mozilla/5.0 (compatible; sillylaird-linkrot/1.0)" + + +def collect() -> dict[str, set[str]]: + """Return {url: {source files}} for every static external link.""" + links: dict[str, set[str]] = {} + files: list[Path] = [] + for ext in ("*.html", "*.php"): + files.extend(ROOT.rglob(ext)) + for f in files: + if any(part in SKIP_DIRS for part in f.parts): + continue + text = f.read_text(encoding="utf-8", errors="ignore") + for m in RE_URL.finditer(text): + url = m.group(2).strip() + if RE_DYNAMIC.search(url): + continue + if any(s in url.lower() for s in SKIP_HOST_SUBSTR): + continue + links.setdefault(url, set()).add(f.relative_to(ROOT).as_posix()) + return links + + +def check(url: str, timeout: float) -> tuple[str, int | None, str]: + """Return (url, status_or_None, note). status None => connection failure.""" + # Try HEAD first, fall back to GET (many servers reject HEAD). + for method in ("HEAD", "GET"): + try: + req = Request(url, method=method, headers={"User-Agent": UA}) + with urlopen(req, timeout=timeout) as resp: + return url, resp.status, "" + except HTTPError as e: + if method == "HEAD" and e.code in (403, 405, 501): + continue # retry with GET + return url, e.code, "" + except (URLError, TimeoutError) as e: + reason = getattr(e, "reason", e) + if method == "HEAD": + continue # retry with GET before giving up + return url, None, str(reason) + except Exception as e: # noqa: BLE001 - report anything odd as dead + return url, None, str(e) + return url, None, "unreachable" + + +def main() -> int: + ap = argparse.ArgumentParser(description="External link rot checker") + ap.add_argument("--timeout", type=float, default=10.0) + ap.add_argument("--workers", type=int, default=16) + ap.add_argument("--strict", action="store_true", + help="also flag 401/403/405/429 (bot-gated) responses") + args = ap.parse_args() + + links = collect() + if not links: + print("OK: no external links found") + return 0 + + print(f"Checking {len(links)} external links…", file=sys.stderr) + dead: list[tuple[str, int | None, str]] = [] + soft: list[tuple[str, int | None, str]] = [] + + with ThreadPoolExecutor(max_workers=args.workers) as ex: + futs = {ex.submit(check, u, args.timeout): u for u in links} + for fut in as_completed(futs): + url, status, note = fut.result() + if status is None: + dead.append((url, status, note)) + elif status >= 400: + (soft if status in SOFT_STATUSES else dead).append((url, status, note)) + + def show(rows): + for url, status, note in sorted(rows, key=lambda r: r[0]): + tag = status if status is not None else "DEAD" + extra = f" ({note})" if note else "" + srcs = ", ".join(sorted(links[url])) + print(f"- [{tag}] {url}{extra}\n in: {srcs}") + + if not args.strict and soft: + print(f"\nbot-gated (not counted; use --strict): {len(soft)}") + show(soft) + + flagged = dead + (soft if args.strict else []) + if flagged: + print(f"\nBroken external links ({len(flagged)}):") + show(flagged) + return 1 + + print(f"OK: all {len(links)} external links reachable") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/optimize_images.sh b/tools/optimize_images.sh new file mode 100755 index 0000000..6538448 --- /dev/null +++ b/tools/optimize_images.sh @@ -0,0 +1,93 @@ +#!/usr/bin/env bash +# Optimize site images in place. Lossy but visually conservative. +# +# - JPEG : strip metadata, progressive, quality 82 +# - PNG : strip metadata, max zlib recompression (lossless) +# - GIF : gifsicle -O3 (lossless frame optimization); --lossy on big ones +# - A short list of oversized tile PNGs is downscaled to a sane max width, +# since the layout never displays them larger than ~320px. +# +# Originals are tarred to "$BACKUP_DIR" before anything is touched. +# Re-running is safe: already-small files are left alone. +# +# Env: +# BACKUP_DIR where to write the pre-run tarball (default /tmp) +# DRY_RUN=1 print what would change, touch nothing +set -u +cd "$(dirname "$0")/.." +ROOT="$(pwd)" +BACKUP_DIR="${BACKUP_DIR:-/tmp}" +DRY="${DRY_RUN:-0}" + +have() { command -v "$1" >/dev/null 2>&1; } +have convert || { echo "need ImageMagick (convert)"; exit 1; } + +# Tile PNGs the layout caps at ~320px wide; 640 keeps a 2x retina buffer. +# Section heroes are capped wider (display size rarely exceeds ~960px). +declare -A MAXW=( + [assets/img/runescape.png]=640 + [assets/img/soldierfront.png]=640 + [assets/img/multibox.png]=640 + [assets/img/stepmania.png]=640 + [startpage/videogame.jpg]=720 + [computers/laptop-compaq.jpg]=960 + [gaming/runescape/runescape.png]=640 +) + +fmt() { numfmt --to=iec "$1" 2>/dev/null || echo "$1"; } +size() { stat -c%s "$1"; } + +if [ "$DRY" != "1" ]; then + ts="$(date +%Y%m%d-%H%M%S)" + tar -czf "$BACKUP_DIR/images-pre-optimize-$ts.tar.gz" \ + assets/img images startpage computers gaming vblog mstartpage 2>/dev/null \ + && echo "backup: $BACKUP_DIR/images-pre-optimize-$ts.tar.gz" +fi + +total_before=0; total_after=0 +process() { + local f="$1" before after; before="$(size "$f")"; total_before=$((total_before+before)) + local tmp="$f.opt.$$" + local fl + fl="$(printf '%s' "$f" | tr '[:upper:]' '[:lower:]')" + case "$fl" in + *.jpg|*.jpeg) + convert "$f" -strip -interlace Plane -quality 82 "$tmp" 2>/dev/null ;; + *.png) + local w="${MAXW[$f]:-}" + [ -z "$w" ] && w="${MAXW[$fl]:-}" + if [ -n "$w" ]; then + convert "$f" -strip -resize "${w}x${w}>" -define png:compression-level=9 -define png:compression-filter=5 "$tmp" 2>/dev/null + else + convert "$f" -strip -define png:compression-level=9 -define png:compression-filter=5 "$tmp" 2>/dev/null + fi ;; + *.gif) + if ! have gifsicle; then total_after=$((total_after+before)); return; fi + if [ "$before" -gt 400000 ]; then + gifsicle -O3 --lossy=60 "$f" -o "$tmp" 2>/dev/null + else + gifsicle -O3 "$f" -o "$tmp" 2>/dev/null + fi ;; + *) total_after=$((total_after+before)); return ;; + esac + if [ ! -s "$tmp" ]; then rm -f "$tmp"; total_after=$((total_after+before)); return; fi + after="$(size "$tmp")" + if [ "$after" -lt "$before" ]; then + if [ "$DRY" = "1" ]; then + printf ' %-40s %8s -> %8s\n' "$f" "$(fmt "$before")" "$(fmt "$after")"; rm -f "$tmp" + else + mv "$tmp" "$f" + fi + total_after=$((total_after+after)) + else + rm -f "$tmp"; total_after=$((total_after+before)) + fi +} + +while IFS= read -r -d '' f; do process "$f"; done < <( + find assets/img images startpage computers gaming vblog mstartpage \ + -type f \( -iname '*.jpg' -o -iname '*.jpeg' \ + -o -iname '*.png' -o -iname '*.gif' \) -print0 2>/dev/null +) + +echo "total: $(fmt "$total_before") -> $(fmt "$total_after")" diff --git a/tools/run_audits.sh b/tools/run_audits.sh new file mode 100755 index 0000000..cf6c548 --- /dev/null +++ b/tools/run_audits.sh @@ -0,0 +1,52 @@ +#!/usr/bin/env bash +# Run all repo audits and regenerate the sitemap. +# Suitable for cron; exits non-zero if anything failed. +# +# Env: +# SILENT=1 suppress per-audit headers +# STRICT=1 also exit non-zero if audits report issues +# (default: only exit non-zero on script errors, not on findings — +# so the cron mail surfaces the report without flapping daily) +# CHECK_LINKS=1 also run the (slow, network-bound) external link rot check + +set -u +cd "$(dirname "$0")/.." + +ROOT="$(pwd)" +PY="${PYTHON:-python3}" + +heading() { [ "${SILENT:-0}" = "1" ] || printf '\n=== %s ===\n' "$1"; } +run_audit() { + local name="$1"; shift + heading "$name" + if "$@"; then + return 0 + else + local rc=$? + [ "${STRICT:-0}" = "1" ] && return $rc + return 0 + fi +} + +# Generate sitemap first (unconditional, idempotent). +heading "generate_sitemap" +"$PY" "$ROOT/tools/generate_sitemap.py" || exit $? + +# Audits — STRICT controls whether their findings fail the run. +run_audit "link_check" "$PY" "$ROOT/tools/link_check.py" +LC_RC=$? +run_audit "html_audit" "$PY" "$ROOT/tools/html_audit.py" +HA_RC=$? +run_audit "image_budget" "$PY" "$ROOT/tools/image_budget.py" +IB_RC=$? + +LR_RC=0 +if [ "${CHECK_LINKS:-0}" = "1" ]; then + run_audit "link_rot" "$PY" "$ROOT/tools/link_rot.py" + LR_RC=$? +fi + +if [ "${STRICT:-0}" = "1" ]; then + exit $(( LC_RC | HA_RC | IB_RC | LR_RC )) +fi +exit 0 diff --git a/tools/security_check.sh b/tools/security_check.sh new file mode 100755 index 0000000..e62fa57 --- /dev/null +++ b/tools/security_check.sh @@ -0,0 +1,62 @@ +#!/usr/bin/env bash +# Check important deployed security headers and PHP session cookie flags. + +set -u + +fail=0 + +check_contains() { + local name="$1" + local haystack="$2" + local needle="$3" + if printf '%s' "$haystack" | grep -Fqi "$needle"; then + printf 'OK: %s\n' "$name" + else + printf 'FAIL: %s missing %s\n' "$name" "$needle" >&2 + fail=1 + fi +} + +check_not_contains() { + local name="$1" + local haystack="$2" + local needle="$3" + if printf '%s' "$haystack" | grep -Fqi "$needle"; then + printf 'FAIL: %s contains %s\n' "$name" "$needle" >&2 + fail=1 + else + printf 'OK: %s\n' "$name" + fi +} + +fetch_headers() { + local host="$1" + local path="$2" + curl -skI --resolve "${host}:443:127.0.0.1" "https://${host}${path}" +} + +www_headers="$(fetch_headers www.sillylaird.ca /)" +# Session cookies are only set on pages that start a session (admin login, admin panel). +admin_headers="$(fetch_headers www.sillylaird.ca /admin/login.php)" +guestbook_headers="$(fetch_headers guestbook.sillylaird.ca /form.php)" + +check_contains "www cookie Secure" "$admin_headers" "set-cookie: PHPSESSID=" +check_contains "www cookie Secure" "$admin_headers" "secure" +check_contains "www cookie HttpOnly" "$admin_headers" "httponly" +check_contains "www cookie SameSite" "$admin_headers" "samesite=lax" +check_contains "www HSTS" "$www_headers" "strict-transport-security:" +check_contains "www nosniff" "$www_headers" "x-content-type-options: nosniff" +check_contains "www CSP" "$www_headers" "content-security-policy:" +check_contains "www CSP frame policy" "$www_headers" "frame-ancestors 'none'" +check_not_contains "www server version" "$www_headers" "nginx/" + +check_contains "guestbook cookie Secure" "$guestbook_headers" "set-cookie: PHPSESSID=" +check_contains "guestbook cookie Secure" "$guestbook_headers" "secure" +check_contains "guestbook cookie HttpOnly" "$guestbook_headers" "httponly" +check_contains "guestbook cookie SameSite" "$guestbook_headers" "samesite=lax" +check_contains "guestbook HSTS" "$guestbook_headers" "strict-transport-security:" +check_contains "guestbook nosniff" "$guestbook_headers" "x-content-type-options: nosniff" +check_contains "guestbook CSP frame policy" "$guestbook_headers" "frame-ancestors https://www.sillylaird.ca" +check_not_contains "guestbook server version" "$guestbook_headers" "nginx/" + +exit "$fail" diff --git a/tools/translate_pages.py b/tools/translate_pages.py new file mode 100644 index 0000000..3127d66 --- /dev/null +++ b/tools/translate_pages.py @@ -0,0 +1,258 @@ +#!/usr/bin/env python3 +"""Generate zh/jp copies of all HTML pages (except startpage/). + +This is a best-effort, offline translation helper. + +- It copies each *.html to *_zh.html and *_jp.html (same directory). +- It preserves all HTML structure, links, ids, classes. +- It translates only user-visible text nodes and some common attributes. +- It skips anything under "startpage/". + +Notes: +- This is not a static site generator. It only writes additional files. +- Translation quality depends on the dictionaries below. +""" + +from __future__ import annotations + +import os +import re +from pathlib import Path + + +ROOT = Path(__file__).resolve().parents[1] + + +SKIP_DIRS = { + "startpage", + "mstartpage", + "partials", +} + + +# Tags whose text content should not be translated. +SKIP_TAGS = { + "script", + "style", + "code", + "pre", + "kbd", + "samp", +} + + +# Very small phrase dictionaries (hand-tuned for this repo). +# For anything not in the dictionary, we leave the text as-is. +ZH = { + "Skip to content": "跳至内容", + "Menu": "菜单", + "Language": "语言", + "Home": "首页", + "StartPage": "StartPage", + "Blog": "博客", + "Guestbook": "留言板", + "Journal": "日志", + "Diary": "日记", + "Gaming": "游戏", + "Bookmarks": "书签", + "Accounts": "账户", + "Computers": "电脑设备", + "Contact": "联系", + "Welcome": "欢迎", + "My Current Vibe": "当前氛围", + "Music": "音乐", + "Current Blog": "当前博客", + "Changelog": "更新日志", + "Friends": "朋友们", + "Games": "游戏", + "Countries": "国家", + "Sponsors / VPNs / Buttons": "赞助商 / VPN / 按钮", + "Open guestbook": "打开留言板", + "Loading…": "加载中…", + "Loading...": "加载中…", + "Licensed under": "采用", + "site": "网站", + "Error": "错误", +} + + +JA = { + "Skip to content": "本文へ移動", + "Menu": "メニュー", + "Language": "言語", + "Home": "ホーム", + "StartPage": "StartPage", + "Blog": "ブログ", + "Guestbook": "ゲストブック", + "Journal": "ジャーナル", + "Diary": "日記", + "Gaming": "ゲーム", + "Bookmarks": "ブックマーク", + "Accounts": "アカウント", + "Computers": "コンピューター", + "Contact": "連絡先", + "Welcome": "ようこそ", + "My Current Vibe": "今の雰囲気", + "Music": "音楽", + "Current Blog": "現在のブログ", + "Changelog": "更新履歴", + "Friends": "友達", + "Games": "ゲーム", + "Countries": "国", + "Sponsors / VPNs / Buttons": "スポンサー / VPN / ボタン", + "Open guestbook": "ゲストブックを開く", + "Loading…": "読み込み中…", + "Loading...": "読み込み中…", + "Licensed under": "ライセンス:", + "Error": "エラー", +} + + +ATTR_TRANSLATE = { + "title", + "aria-label", + "aria-labelledby", # generally ids; don't translate + "alt", + "placeholder", +} + + +RE_TAG = re.compile(r"(<[^>]+>)") +RE_TEXT_NODE = re.compile(r"^(\s*)(.*?)(\s*)$", re.S) +RE_ATTR = re.compile(r'(\s)([a-zA-Z_:.-]+)=("[^"]*"|\'[\s\S]*?\')') + + +def should_skip_path(p: Path) -> bool: + rel = p.relative_to(ROOT) + parts = set(rel.parts) + return any(d in parts for d in SKIP_DIRS) + + +def translate_phrase(s: str, mapping: dict[str, str]) -> str: + # Exact match first + if s in mapping: + return mapping[s] + + # Replace common UI tokens inside longer strings (simple, conservative) + out = s + for k, v in mapping.items(): + if k and k in out: + out = out.replace(k, v) + return out + + +def translate_text_node(text: str, mapping: dict[str, str]) -> str: + m = RE_TEXT_NODE.match(text) + if not m: + return text + lead, core, tail = m.group(1), m.group(2), m.group(3) + + # Skip empty or purely whitespace + if not core.strip(): + return text + + # Skip if it's just punctuation/symbols + if not re.search(r"[A-Za-z]", core): + return text + + translated = translate_phrase(core, mapping) + return f"{lead}{translated}{tail}" + + +def tag_name(tag: str) -> str | None: + # tag is like <div ...> or </div> + t = tag.strip()[1:-1].strip() + if not t: + return None + if t.startswith("!") or t.startswith("?"): + return None + if t.startswith("/"): + t = t[1:].lstrip() + name = re.split(r"\s+", t, maxsplit=1)[0].lower() + return name + + +def translate_attrs(tag: str, mapping: dict[str, str]) -> str: + # Don't touch aria-labelledby since it's usually an id. + def repl(m: re.Match[str]) -> str: + space, key, val = m.group(1), m.group(2), m.group(3) + k = key.lower() + if k not in ATTR_TRANSLATE or k == "aria-labelledby": + return m.group(0) + quote = val[0] + inner = val[1:-1] + new_inner = translate_phrase(inner, mapping) + if new_inner == inner: + return m.group(0) + return f"{space}{key}={quote}{new_inner}{quote}" + + return RE_ATTR.sub(repl, tag) + + +def translate_html(src: str, mapping: dict[str, str]) -> str: + parts = RE_TAG.split(src) + out: list[str] = [] + + skip_depth = 0 + for part in parts: + if part.startswith("<") and part.endswith(">"): + name = tag_name(part) + + # track skip tags nesting + if name in SKIP_TAGS: + if part.lstrip().startswith("</"): + if skip_depth > 0: + skip_depth -= 1 + else: + skip_depth += 1 + + out.append(translate_attrs(part, mapping)) + else: + if skip_depth > 0: + out.append(part) + else: + out.append(translate_text_node(part, mapping)) + + return "".join(out) + + +def write_if_changed(path: Path, content: str) -> None: + old = path.read_text(encoding="utf-8", errors="ignore") if path.exists() else None + if old == content: + return + path.write_text(content, encoding="utf-8") + + +def main() -> int: + html_files = sorted(ROOT.rglob("*.html")) + for p in html_files: + if should_skip_path(p): + continue + + # Skip already translated files + if p.name.endswith("_zh.html") or p.name.endswith("_jp.html"): + continue + + # Only translate pages that look like they are part of the unified site + # (Keep legacy old HTML alone unless user explicitly wants all) + src = p.read_text(encoding="utf-8", errors="ignore") + + # Output names + zh_path = p.with_name(p.stem + "_zh.html") + jp_path = p.with_name(p.stem + "_jp.html") + + zh = translate_html(src, ZH) + jp = translate_html(src, JA) + + # Set lang attribute if present + zh = re.sub(r"<html\s+lang=\"[^\"]*\"", '<html lang="zh"', zh, count=1) + jp = re.sub(r"<html\s+lang=\"[^\"]*\"", '<html lang="ja"', jp, count=1) + + write_if_changed(zh_path, zh) + write_if_changed(jp_path, jp) + + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/update_openbsd_version.sh b/tools/update_openbsd_version.sh new file mode 100755 index 0000000..d2f7c8f --- /dev/null +++ b/tools/update_openbsd_version.sh @@ -0,0 +1,38 @@ +#!/usr/bin/env bash +# Fetch the current OpenBSD release from openbsd.org and cache it for the +# startpage boot banner. Safe to run from cron (idempotent; only rewrites +# the cache when the version string is valid and changed or missing). +set -euo pipefail + +OUT="${OUT:-/mnt/slab/www.sillylaird.ca/startpage/openbsd_version.txt}" +URL="${URL:-https://www.openbsd.org/}" +TMP="$(mktemp)" +trap 'rm -f "$TMP"' EXIT + +if ! curl -fsSL --max-time 25 -A 'sillylaird-openbsd-version/1.0' -o "$TMP" "$URL"; then + echo "update_openbsd_version: fetch failed: $URL" >&2 + exit 1 +fi + +# Prefer the puffy banner alt text: [OpenBSD 7.9] +ver="$(sed -n 's/.*\[OpenBSD \([0-9][0-9]*\.[0-9][0-9]*\)\].*/\1/p' "$TMP" | head -n1 || true)" + +# Fallback: first "OpenBSD X.Y" mention that is not a year-range copyright. +if [[ -z "$ver" ]]; then + ver="$(grep -oE 'OpenBSD [0-9]+\.[0-9]+' "$TMP" | head -n1 | awk '{print $2}' || true)" +fi + +if [[ ! "$ver" =~ ^[0-9]+\.[0-9]+$ ]]; then + echo "update_openbsd_version: could not parse version from $URL" >&2 + exit 1 +fi + +mkdir -p "$(dirname "$OUT")" +if [[ -f "$OUT" ]] && [[ "$(tr -d '[:space:]' < "$OUT")" == "$ver" ]]; then + echo "openbsd $ver (unchanged)" + exit 0 +fi + +printf '%s\n' "$ver" > "$OUT" +chmod 644 "$OUT" +echo "openbsd $ver -> $OUT" diff --git a/tools/uptime_check.sh b/tools/uptime_check.sh new file mode 100755 index 0000000..0bbc6e8 --- /dev/null +++ b/tools/uptime_check.sh @@ -0,0 +1,6 @@ +#!/usr/bin/env bash +set -euo pipefail + +URL="${URL:-https://www.sillylaird.ca/}" + +curl -fsS -o /dev/null -w "HTTP %{http_code} in %{time_total}s\n" "$URL" |
