aboutsummaryrefslogtreecommitdiffstats
path: root/tools/generate_sitemap.py
blob: ab7e25d4836a28fc4a56a0e079ac82720ea32e75 (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
#!/usr/bin/env python3
"""Generate sitemap.xml from local PHP/HTML pages.

The site is PHP-based with the convention that translated copies live next to
the canonical file as `foo_jp.php` and `foo_zh.php`. When a triple exists, all
three URLs go into the sitemap with an `xhtml:link` hreflang block so search
engines understand the relationship.

Standalone pages get a plain `<url>` entry.
"""

from __future__ import annotations

from collections import defaultdict
from pathlib import Path
from datetime import datetime, timezone

ROOT = Path(__file__).resolve().parents[1]
SITE = "https://www.sillylaird.ca"

# Directories that never produce navigable URLs.
EXCLUDE_DIRS = {
    ".git", ".agents", ".claude",
    "partials", "admin", "api", "locales",
    "tools", "pub", "docs",
    # Legacy redirect-only path; canonical is /startpage/
    "mstartpage",
}

# Specific files that are not user-navigable pages: error pages, proxies,
# iframe-only components, internal endpoints, config.
EXCLUDE_FILES = {
    "404.php", "50x.php",
    "gemini-proxy.php", "lastfm-proxy.php",
    "guestbook-form.php", "guestbook-comments.php",
    "submit.php", "bbs.php", "bbs.cgi",
    "vibe.php",
    "feed.xml.php",
}

# Files matching these path suffixes are excluded (admin/internal endpoints).
EXCLUDE_PATH_SUFFIXES = {
    "changelog/admin.php",
    "changelog/api/latest.php",
    "changelog/auth.php",
    "changelog/db.php",
    "changelog/latest.php",
    "changelog/login.php",
    "changelog/logout.php",
    "vibe/admin.php",
    "vibe/admin_vibe.php",
    "gaming/collection/admin.php",
    "gaming/collection/_lib.php",
    "whatsleft/privacy/index.php",
}


def should_skip(path: Path) -> bool:
    parts = set(path.relative_to(ROOT).parts)
    if parts & EXCLUDE_DIRS:
        return True
    if path.name in EXCLUDE_FILES:
        return True
    rel = path.relative_to(ROOT).as_posix()
    if rel in EXCLUDE_PATH_SUFFIXES:
        return True
    if path.name.endswith("~"):
        return True
    return False


def url_for(path: Path) -> str:
    rel = path.relative_to(ROOT).as_posix()
    if rel == "index.php" or rel == "index.html":
        return SITE + "/"
    if rel.endswith("/index.php"):
        return SITE + "/" + rel[: -len("index.php")]
    if rel.endswith("/index.html"):
        return SITE + "/" + rel[: -len("index.html")]
    return SITE + "/" + rel


def lastmod_for(path: Path) -> str:
    ts = path.stat().st_mtime
    return datetime.fromtimestamp(ts, tz=timezone.utc).strftime("%Y-%m-%d")


def split_lang(name: str) -> tuple[str, str]:
    """Return (base, lang). For foo_jp.php returns ('foo.php', 'ja')."""
    for suffix, lang in (("_jp.php", "ja"), ("_zh.php", "zh"),
                         ("_jp.html", "ja"), ("_zh.html", "zh")):
        if name.endswith(suffix):
            stem = name[: -len(suffix)]
            ext = ".php" if suffix.endswith(".php") else ".html"
            return stem + ext, lang
        # also handle index_jp.php -> index.php family
    return name, "en"


def main() -> int:
    files: list[Path] = []
    for ext in ("*.php", "*.html"):
        for p in ROOT.rglob(ext):
            if should_skip(p):
                continue
            files.append(p)

    # Group by (parent_dir, canonical_name) so we can emit hreflang for triples.
    families: dict[tuple[Path, str], dict[str, Path]] = defaultdict(dict)
    for p in files:
        canonical_name, lang = split_lang(p.name)
        families[(p.parent, canonical_name)][lang] = p

    lines = [
        '<?xml version="1.0" encoding="UTF-8"?>',
        '<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"',
        '        xmlns:xhtml="http://www.w3.org/1999/xhtml">',
    ]

    # Emit URLs sorted for stable diffs.
    keyed: list[tuple[str, str, dict[str, Path]]] = []
    for (parent, canonical_name), variants in families.items():
        # Pick the URL of the canonical (en) entry if present, otherwise any.
        anchor_path = variants.get("en") or next(iter(variants.values()))
        anchor_url = url_for(anchor_path)
        keyed.append((anchor_url, canonical_name, variants))
    keyed.sort(key=lambda x: x[0])

    for _anchor_url, _canonical_name, variants in keyed:
        is_triple = len(variants) > 1
        for lang in ("en", "ja", "zh"):
            if lang not in variants:
                continue
            p = variants[lang]
            loc = url_for(p)
            lines.append("  <url>")
            lines.append(f"    <loc>{loc}</loc>")
            lines.append(f"    <lastmod>{lastmod_for(p)}</lastmod>")
            if is_triple:
                en_url = url_for(variants["en"]) if "en" in variants else loc
                for hl in ("en", "ja", "zh"):
                    if hl in variants:
                        lines.append(
                            f'    <xhtml:link rel="alternate" hreflang="{hl}" '
                            f'href="{url_for(variants[hl])}"/>'
                        )
                lines.append(
                    f'    <xhtml:link rel="alternate" hreflang="x-default" href="{en_url}"/>'
                )
            lines.append("  </url>")

    lines.append("</urlset>")

    (ROOT / "sitemap.xml").write_text("\n".join(lines) + "\n", encoding="utf-8")
    print(f"wrote sitemap.xml with {sum(len(v) for v in families.values())} URLs")
    return 0


if __name__ == "__main__":
    raise SystemExit(main())