aboutsummaryrefslogtreecommitdiffstats
path: root/tools/generate_sitemap.py
diff options
context:
space:
mode:
authorsillylaird <sillyfanboy@gmail.com>2026-09-03 00:33:59 +0000
committersillylaird <sillyfanboy@gmail.com>2026-09-03 00:33:59 +0000
commit898b52edcb47bcb3e9d6106e74ca73e74ea01e70 (patch)
tree85c6ee5ad58b860144551184d4cf86b560c62b91 /tools/generate_sitemap.py
downloadwww-898b52edcb47bcb3e9d6106e74ca73e74ea01e70.tar.gz
www-898b52edcb47bcb3e9d6106e74ca73e74ea01e70.zip
import live www.sillylaird.ca webrootHEADmain
Diffstat (limited to 'tools/generate_sitemap.py')
-rw-r--r--tools/generate_sitemap.py160
1 files changed, 160 insertions, 0 deletions
diff --git a/tools/generate_sitemap.py b/tools/generate_sitemap.py
new file mode 100644
index 0000000..ab7e25d
--- /dev/null
+++ b/tools/generate_sitemap.py
@@ -0,0 +1,160 @@
+#!/usr/bin/env python3
+"""Generate sitemap.xml from local PHP/HTML pages.
+
+The site is PHP-based with the convention that translated copies live next to
+the canonical file as `foo_jp.php` and `foo_zh.php`. When a triple exists, all
+three URLs go into the sitemap with an `xhtml:link` hreflang block so search
+engines understand the relationship.
+
+Standalone pages get a plain `<url>` entry.
+"""
+
+from __future__ import annotations
+
+from collections import defaultdict
+from pathlib import Path
+from datetime import datetime, timezone
+
+ROOT = Path(__file__).resolve().parents[1]
+SITE = "https://www.sillylaird.ca"
+
+# Directories that never produce navigable URLs.
+EXCLUDE_DIRS = {
+ ".git", ".agents", ".claude",
+ "partials", "admin", "api", "locales",
+ "tools", "pub", "docs",
+ # Legacy redirect-only path; canonical is /startpage/
+ "mstartpage",
+}
+
+# Specific files that are not user-navigable pages: error pages, proxies,
+# iframe-only components, internal endpoints, config.
+EXCLUDE_FILES = {
+ "404.php", "50x.php",
+ "gemini-proxy.php", "lastfm-proxy.php",
+ "guestbook-form.php", "guestbook-comments.php",
+ "submit.php", "bbs.php", "bbs.cgi",
+ "vibe.php",
+ "feed.xml.php",
+}
+
+# Files matching these path suffixes are excluded (admin/internal endpoints).
+EXCLUDE_PATH_SUFFIXES = {
+ "changelog/admin.php",
+ "changelog/api/latest.php",
+ "changelog/auth.php",
+ "changelog/db.php",
+ "changelog/latest.php",
+ "changelog/login.php",
+ "changelog/logout.php",
+ "vibe/admin.php",
+ "vibe/admin_vibe.php",
+ "gaming/collection/admin.php",
+ "gaming/collection/_lib.php",
+ "whatsleft/privacy/index.php",
+}
+
+
+def should_skip(path: Path) -> bool:
+ parts = set(path.relative_to(ROOT).parts)
+ if parts & EXCLUDE_DIRS:
+ return True
+ if path.name in EXCLUDE_FILES:
+ return True
+ rel = path.relative_to(ROOT).as_posix()
+ if rel in EXCLUDE_PATH_SUFFIXES:
+ return True
+ if path.name.endswith("~"):
+ return True
+ return False
+
+
+def url_for(path: Path) -> str:
+ rel = path.relative_to(ROOT).as_posix()
+ if rel == "index.php" or rel == "index.html":
+ return SITE + "/"
+ if rel.endswith("/index.php"):
+ return SITE + "/" + rel[: -len("index.php")]
+ if rel.endswith("/index.html"):
+ return SITE + "/" + rel[: -len("index.html")]
+ return SITE + "/" + rel
+
+
+def lastmod_for(path: Path) -> str:
+ ts = path.stat().st_mtime
+ return datetime.fromtimestamp(ts, tz=timezone.utc).strftime("%Y-%m-%d")
+
+
+def split_lang(name: str) -> tuple[str, str]:
+ """Return (base, lang). For foo_jp.php returns ('foo.php', 'ja')."""
+ for suffix, lang in (("_jp.php", "ja"), ("_zh.php", "zh"),
+ ("_jp.html", "ja"), ("_zh.html", "zh")):
+ if name.endswith(suffix):
+ stem = name[: -len(suffix)]
+ ext = ".php" if suffix.endswith(".php") else ".html"
+ return stem + ext, lang
+ # also handle index_jp.php -> index.php family
+ return name, "en"
+
+
+def main() -> int:
+ files: list[Path] = []
+ for ext in ("*.php", "*.html"):
+ for p in ROOT.rglob(ext):
+ if should_skip(p):
+ continue
+ files.append(p)
+
+ # Group by (parent_dir, canonical_name) so we can emit hreflang for triples.
+ families: dict[tuple[Path, str], dict[str, Path]] = defaultdict(dict)
+ for p in files:
+ canonical_name, lang = split_lang(p.name)
+ families[(p.parent, canonical_name)][lang] = p
+
+ lines = [
+ '<?xml version="1.0" encoding="UTF-8"?>',
+ '<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"',
+ ' xmlns:xhtml="http://www.w3.org/1999/xhtml">',
+ ]
+
+ # Emit URLs sorted for stable diffs.
+ keyed: list[tuple[str, str, dict[str, Path]]] = []
+ for (parent, canonical_name), variants in families.items():
+ # Pick the URL of the canonical (en) entry if present, otherwise any.
+ anchor_path = variants.get("en") or next(iter(variants.values()))
+ anchor_url = url_for(anchor_path)
+ keyed.append((anchor_url, canonical_name, variants))
+ keyed.sort(key=lambda x: x[0])
+
+ for _anchor_url, _canonical_name, variants in keyed:
+ is_triple = len(variants) > 1
+ for lang in ("en", "ja", "zh"):
+ if lang not in variants:
+ continue
+ p = variants[lang]
+ loc = url_for(p)
+ lines.append(" <url>")
+ lines.append(f" <loc>{loc}</loc>")
+ lines.append(f" <lastmod>{lastmod_for(p)}</lastmod>")
+ if is_triple:
+ en_url = url_for(variants["en"]) if "en" in variants else loc
+ for hl in ("en", "ja", "zh"):
+ if hl in variants:
+ lines.append(
+ f' <xhtml:link rel="alternate" hreflang="{hl}" '
+ f'href="{url_for(variants[hl])}"/>'
+ )
+ lines.append(
+ f' <xhtml:link rel="alternate" hreflang="x-default" href="{en_url}"/>'
+ )
+ lines.append(" </url>")
+
+ lines.append("</urlset>")
+
+ (ROOT / "sitemap.xml").write_text("\n".join(lines) + "\n", encoding="utf-8")
+ print(f"wrote sitemap.xml with {sum(len(v) for v in families.values())} URLs")
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())