diff options
| author | Laird Lonergan <sillyfanboy@gmail.com> | 2026-09-03 01:40:19 +0000 |
|---|---|---|
| committer | Laird Lonergan <sillyfanboy@gmail.com> | 2026-09-03 01:40:19 +0000 |
| commit | 43ed72ae3cd7e5f17aeb3b51000fa69614495b1e (patch) | |
| tree | 2441b172d5ca92b87bbbb6a6810fc6f224484357 /blogmirror.c | |
| parent | initial (diff) | |
| download | www-43ed72ae3cd7e5f17aeb3b51000fa69614495b1e.tar.gz www-43ed72ae3cd7e5f17aeb3b51000fa69614495b1e.zip | |
Reads the blog's RSS feed (full post body lives in content:encoded) and
writes flat mothra-friendly HTML under html/ — an index plus one page per
post at YYYY/MM/DD/slug.html.
Plain ANSI C so it builds with APE pcc on 9front and cc elsewhere; fetches
with hget when present, curl otherwise. mirror.rc regenerates and pushes.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01YAgU4WFfVJKNK8CkpxhEcg
Diffstat (limited to '')
| -rw-r--r-- | blogmirror.c | 424 |
1 files changed, 424 insertions, 0 deletions
diff --git a/blogmirror.c b/blogmirror.c new file mode 100644 index 0000000..a479386 --- /dev/null +++ b/blogmirror.c @@ -0,0 +1,424 @@ +/* + * blogmirror - mirror blog.sillylaird.ca into flat mothra-friendly HTML. + * + * Reads the blog's RSS 2.0 feed (which carries the full post body in + * <content:encoded>) and writes a static tree: + * + * index.html list of posts + * YYYY/MM/DD/slug.html one page per post + * + * Plain ANSI C: builds with pcc on 9front (APE) and with cc elsewhere. + * + * blogmirror [-u url] [-f file] [-o outdir] [-n maxposts] + * + * With -f it parses a feed already on disk; otherwise it shells out to + * hget(1) (curl elsewhere) to fetch -u into a temporary file first. + */ + +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <sys/types.h> +#include <sys/stat.h> + +#define FEEDURL "https://blog.sillylaird.ca/feed.xml.php" +#define SITE "https://blog.sillylaird.ca/" +#define TMPFEED "/tmp/blogmirror.rss" + +static char *outdir = "html"; +static int maxposts = 0; /* 0 = all */ +static int nposts = 0; + +static void * +emalloc(size_t n) +{ + void *p = malloc(n); + if(p == NULL){ + fprintf(stderr, "blogmirror: out of memory\n"); + exit(1); + } + return p; +} + +static char * +estrdup(const char *s) +{ + char *p = emalloc(strlen(s) + 1); + strcpy(p, s); + return p; +} + +/* bounded concatenation; returns 0 if it would not fit */ +static int +cats(char *d, size_t n, const char *s) +{ + size_t have = strlen(d), want = strlen(s); + + if(have + want + 1 > n) + return 0; + memcpy(d + have, s, want + 1); + return 1; +} + +/* does this file exist? */ +static int +exists(const char *path) +{ + FILE *f = fopen(path, "rb"); + + if(f == NULL) + return 0; + fclose(f); + return 1; +} + +/* read a whole file, NUL terminated */ +static char * +slurp(const char *path) +{ + FILE *f; + char *b; + long n; + + if((f = fopen(path, "rb")) == NULL) + return NULL; + fseek(f, 0L, SEEK_END); + n = ftell(f); + rewind(f); + if(n < 0){ + fclose(f); + return NULL; + } + b = emalloc((size_t)n + 1); + if(fread(b, 1, (size_t)n, f) != (size_t)n){ + fclose(f); + free(b); + return NULL; + } + b[n] = '\0'; + fclose(f); + return b; +} + +/* text between open and close, starting at p, not past end. malloc'd. */ +static char * +between(char *p, char *end, const char *open, const char *close) +{ + char *a, *b, *s; + size_t n; + + if((a = strstr(p, open)) == NULL || a >= end) + return NULL; + a += strlen(open); + if((b = strstr(a, close)) == NULL || b > end) + return NULL; + n = (size_t)(b - a); + s = emalloc(n + 1); + memcpy(s, a, n); + s[n] = '\0'; + return s; +} + +/* one UTF-8 rune out of a code point */ +static int +putrune(char *d, long r) +{ + if(r < 0x80){ + d[0] = (char)r; + return 1; + } + if(r < 0x800){ + d[0] = (char)(0xC0 | (r >> 6)); + d[1] = (char)(0x80 | (r & 0x3F)); + return 2; + } + if(r < 0x10000){ + d[0] = (char)(0xE0 | (r >> 12)); + d[1] = (char)(0x80 | ((r >> 6) & 0x3F)); + d[2] = (char)(0x80 | (r & 0x3F)); + return 3; + } + d[0] = (char)(0xF0 | (r >> 18)); + d[1] = (char)(0x80 | ((r >> 12) & 0x3F)); + d[2] = (char)(0x80 | ((r >> 6) & 0x3F)); + d[3] = (char)(0x80 | (r & 0x3F)); + return 4; +} + +/* & < > " ' ' ’ -> text, in place */ +static char * +unxml(char *s) +{ + char *r, *w; + long v; + + if(s == NULL) + return NULL; + for(r = w = s; *r != '\0'; ){ + if(*r != '&'){ + *w++ = *r++; + continue; + } + if(strncmp(r, "&", 5) == 0){ *w++ = '&'; r += 5; continue; } + if(strncmp(r, "<", 4) == 0){ *w++ = '<'; r += 4; continue; } + if(strncmp(r, ">", 4) == 0){ *w++ = '>'; r += 4; continue; } + if(strncmp(r, """, 6) == 0){ *w++ = '"'; r += 6; continue; } + if(strncmp(r, "'", 6) == 0){ *w++ = '\''; r += 6; continue; } + if(r[1] == '#'){ + char *e; + if(r[2] == 'x' || r[2] == 'X') + v = strtol(r + 3, &e, 16); + else + v = strtol(r + 2, &e, 10); + if(*e == ';' && v > 0){ + w += putrune(w, v); + r = e + 1; + continue; + } + } + *w++ = *r++; + } + *w = '\0'; + return s; +} + +/* escape text for HTML output */ +static void +puthtml(FILE *f, const char *s) +{ + if(s == NULL) + return; + for(; *s != '\0'; s++) + switch(*s){ + case '&': fputs("&", f); break; + case '<': fputs("<", f); break; + case '>': fputs(">", f); break; + case '"': fputs(""", f); break; + default: fputc(*s, f); break; + } +} + +/* mkdir -p over the directory part of path */ +static void +mkpath(const char *path) +{ + char buf[1024]; + char *p; + + if(strlen(path) >= sizeof buf) + return; + strcpy(buf, path); + for(p = strchr(buf, '/'); p != NULL; p = strchr(p + 1, '/')){ + *p = '\0'; + if(buf[0] != '\0') + mkdir(buf, 0777); + *p = '/'; + } +} + +/* blog URL -> repo-relative path, e.g. 2026/05/29/hello-world.html */ +static const char * +urlpath(const char *link) +{ + const char *p = link; + + if(strncmp(p, SITE, strlen(SITE)) == 0) + p += strlen(SITE); + else if((p = strstr(link, "://")) != NULL){ + p += 3; + p = strchr(p, '/'); + if(p == NULL) + return ""; + p++; + } else + p = link; + while(*p == '/') + p++; + return p; +} + +static void +header(FILE *f, const char *title) +{ + fputs("<!DOCTYPE html>\n<html>\n\n<head>\n" + " <meta charset=\"utf-8\">\n <title>", f); + puthtml(f, title); + fputs(" \xe2\x80\x93 SillyLaird</title>\n</head>\n\n<body>\n", f); +} + +static void +footer(FILE *f, const char *up) +{ + fputs("\n <hr>\n\n <p>\n <a href=\"", f); + fputs(up, f); + fputs("index.html\">Blog index</a>\n </p>\n\n</body>\n\n</html>\n", f); +} + +static void +writepost(const char *path, const char *title, const char *date, + const char *body, const char *link) +{ + char file[1024]; + char up[64]; + const char *p; + FILE *f; + int depth; + + if(path[0] == '\0') + return; + file[0] = '\0'; + if(!cats(file, sizeof file, outdir) || !cats(file, sizeof file, "/") + || !cats(file, sizeof file, path)){ + fprintf(stderr, "blogmirror: path too long: %s\n", path); + return; + } + mkpath(file); + if((f = fopen(file, "wb")) == NULL){ + fprintf(stderr, "blogmirror: %s: cannot create\n", file); + return; + } + + up[0] = '\0'; + for(depth = 0, p = path; *p != '\0'; p++) + if(*p == '/') + depth++; + for(; depth > 0 && strlen(up) + 3 < sizeof up; depth--) + strcat(up, "../"); + + header(f, title); + fputs(" <h2>", f); + puthtml(f, title); + fputs("</h2>\n\n <p><font size=\"-1\">", f); + puthtml(f, date); + fputs("</font></p>\n\n", f); + fputs(body, f); /* already HTML from the blog */ + fputs("\n\n <p><font size=\"-1\">Mirror of <a href=\"", f); + puthtml(f, link); + fputs("\">", f); + puthtml(f, link); + fputs("</a></font></p>\n", f); + footer(f, up); + fclose(f); + fprintf(stderr, "%s\n", file); +} + +int +main(int argc, char *argv[]) +{ + char *url = FEEDURL; + char *feedfile = NULL; + char *feed, *p, *end, *item; + char cmd[1024]; + char file[1024]; + FILE *ix; + int i; + + for(i = 1; i < argc; i++){ + if(strcmp(argv[i], "-u") == 0 && i + 1 < argc) + url = argv[++i]; + else if(strcmp(argv[i], "-f") == 0 && i + 1 < argc) + feedfile = argv[++i]; + else if(strcmp(argv[i], "-o") == 0 && i + 1 < argc) + outdir = argv[++i]; + else if(strcmp(argv[i], "-n") == 0 && i + 1 < argc) + maxposts = atoi(argv[++i]); + else { + fprintf(stderr, + "usage: blogmirror [-u url] [-f file] [-o outdir] [-n maxposts]\n"); + exit(1); + } + } + + if(feedfile == NULL){ + cmd[0] = '\0'; + if(exists("/bin/hget")){ /* 9front */ + cats(cmd, sizeof cmd, "hget '"); + cats(cmd, sizeof cmd, url); + cats(cmd, sizeof cmd, "' >"); + cats(cmd, sizeof cmd, TMPFEED); + } else { + cats(cmd, sizeof cmd, "curl -fsS '"); + cats(cmd, sizeof cmd, url); + cats(cmd, sizeof cmd, "' -o "); + cats(cmd, sizeof cmd, TMPFEED); + } + if(system(cmd) != 0){ + fprintf(stderr, "blogmirror: fetch failed: %s\n", url); + exit(1); + } + feedfile = TMPFEED; + } + + if((feed = slurp(feedfile)) == NULL){ + fprintf(stderr, "blogmirror: %s: cannot read\n", feedfile); + exit(1); + } + + mkdir(outdir, 0777); + file[0] = '\0'; + cats(file, sizeof file, outdir); + cats(file, sizeof file, "/index.html"); + if((ix = fopen(file, "wb")) == NULL){ + fprintf(stderr, "blogmirror: %s: cannot create\n", file); + exit(1); + } + header(ix, "Blog"); + fputs(" <h2>Blog</h2>\n\n" + " <p>Mirror of <a href=\"" SITE "\">blog.sillylaird.ca</a>.</p>\n\n" + " <table cellspacing=\"0\" cellpadding=\"0\" border=\"0\">\n", ix); + + for(p = feed; (p = strstr(p, "<item>")) != NULL; p = end){ + char *title, *link, *date, *desc, *body; + const char *path; + + if((end = strstr(p, "</item>")) == NULL) + break; + item = p; + title = unxml(between(item, end, "<title>", "</title>")); + link = unxml(between(item, end, "<link>", "</link>")); + date = unxml(between(item, end, "<pubDate>", "</pubDate>")); + desc = unxml(between(item, end, "<description>", "</description>")); + body = between(item, end, "<content:encoded><![CDATA[", "]]></content:encoded>"); + + if(title == NULL || link == NULL){ + free(title); free(link); free(date); free(desc); free(body); + continue; + } + if(date == NULL) + date = estrdup(""); + if(body == NULL) + body = estrdup(desc != NULL ? desc : ""); + + path = urlpath(link); + writepost(path, title, date, body, link); + + fputs(" <tr>\n <td width=\"10\"></td>\n <td valign=\"top\">\n" + " <a href=\"", ix); + puthtml(ix, path); + fputs("\">", ix); + puthtml(ix, title); + fputs("</a><br>\n" + " <table cellspacing=\"0\" cellpadding=\"0\" border=\"0\">\n" + " <tr>\n <td width=\"10\"></td>\n <td>\n" + " <font size=\"-1\">\n ", ix); + puthtml(ix, date); + fputs("<br>\n ", ix); + puthtml(ix, desc != NULL ? desc : ""); + fputs("\n </font>\n </td>\n </tr>\n" + " </table>\n </td>\n </tr>\n", ix); + + free(title); free(link); free(date); free(desc); free(body); + nposts++; + if(maxposts > 0 && nposts >= maxposts) + break; + end += 7; /* past </item> */ + } + + fputs(" </table>\n", ix); + footer(ix, ""); + fclose(ix); + free(feed); + + fprintf(stderr, "%s: %d posts\n", file, nposts); + return 0; +} |
