/* * blogmirror - mirror blog.sillylaird.ca into flat mothra-friendly HTML. * * Reads the blog's RSS 2.0 feed (which carries the full post body in * ) and writes a static tree: * * index.html list of posts * YYYY/MM/DD/slug.html one page per post * * Plain ANSI C: builds with pcc on 9front (APE) and with cc elsewhere. * * blogmirror [-u url] [-f file] [-o outdir] [-n maxposts] * * With -f it parses a feed already on disk; otherwise it shells out to * hget(1) (curl elsewhere) to fetch -u into a temporary file first. */ #include #include #include #include #include #define FEEDURL "https://blog.sillylaird.ca/feed.xml.php" #define SITE "https://blog.sillylaird.ca/" #define TMPFEED "/tmp/blogmirror.rss" static char *outdir = "html"; static int maxposts = 0; /* 0 = all */ static int nposts = 0; static void * emalloc(size_t n) { void *p = malloc(n); if(p == NULL){ fprintf(stderr, "blogmirror: out of memory\n"); exit(1); } return p; } static char * estrdup(const char *s) { char *p = emalloc(strlen(s) + 1); strcpy(p, s); return p; } /* bounded concatenation; returns 0 if it would not fit */ static int cats(char *d, size_t n, const char *s) { size_t have = strlen(d), want = strlen(s); if(have + want + 1 > n) return 0; memcpy(d + have, s, want + 1); return 1; } /* does this file exist? */ static int exists(const char *path) { FILE *f = fopen(path, "rb"); if(f == NULL) return 0; fclose(f); return 1; } /* read a whole file, NUL terminated */ static char * slurp(const char *path) { FILE *f; char *b; long n; if((f = fopen(path, "rb")) == NULL) return NULL; fseek(f, 0L, SEEK_END); n = ftell(f); rewind(f); if(n < 0){ fclose(f); return NULL; } b = emalloc((size_t)n + 1); if(fread(b, 1, (size_t)n, f) != (size_t)n){ fclose(f); free(b); return NULL; } b[n] = '\0'; fclose(f); return b; } /* text between open and close, starting at p, not past end. malloc'd. */ static char * between(char *p, char *end, const char *open, const char *close) { char *a, *b, *s; size_t n; if((a = strstr(p, open)) == NULL || a >= end) return NULL; a += strlen(open); if((b = strstr(a, close)) == NULL || b > end) return NULL; n = (size_t)(b - a); s = emalloc(n + 1); memcpy(s, a, n); s[n] = '\0'; return s; } /* one UTF-8 rune out of a code point */ static int putrune(char *d, long r) { if(r < 0x80){ d[0] = (char)r; return 1; } if(r < 0x800){ d[0] = (char)(0xC0 | (r >> 6)); d[1] = (char)(0x80 | (r & 0x3F)); return 2; } if(r < 0x10000){ d[0] = (char)(0xE0 | (r >> 12)); d[1] = (char)(0x80 | ((r >> 6) & 0x3F)); d[2] = (char)(0x80 | (r & 0x3F)); return 3; } d[0] = (char)(0xF0 | (r >> 18)); d[1] = (char)(0x80 | ((r >> 12) & 0x3F)); d[2] = (char)(0x80 | ((r >> 6) & 0x3F)); d[3] = (char)(0x80 | (r & 0x3F)); return 4; } /* & < > " ' ' ’ -> text, in place */ static char * unxml(char *s) { char *r, *w; long v; if(s == NULL) return NULL; for(r = w = s; *r != '\0'; ){ if(*r != '&'){ *w++ = *r++; continue; } if(strncmp(r, "&", 5) == 0){ *w++ = '&'; r += 5; continue; } if(strncmp(r, "<", 4) == 0){ *w++ = '<'; r += 4; continue; } if(strncmp(r, ">", 4) == 0){ *w++ = '>'; r += 4; continue; } if(strncmp(r, """, 6) == 0){ *w++ = '"'; r += 6; continue; } if(strncmp(r, "'", 6) == 0){ *w++ = '\''; r += 6; continue; } if(r[1] == '#'){ char *e; if(r[2] == 'x' || r[2] == 'X') v = strtol(r + 3, &e, 16); else v = strtol(r + 2, &e, 10); if(*e == ';' && v > 0){ w += putrune(w, v); r = e + 1; continue; } } *w++ = *r++; } *w = '\0'; return s; } /* escape text for HTML output */ static void puthtml(FILE *f, const char *s) { if(s == NULL) return; for(; *s != '\0'; s++) switch(*s){ case '&': fputs("&", f); break; case '<': fputs("<", f); break; case '>': fputs(">", f); break; case '"': fputs(""", f); break; default: fputc(*s, f); break; } } /* mkdir -p over the directory part of path */ static void mkpath(const char *path) { char buf[1024]; char *p; if(strlen(path) >= sizeof buf) return; strcpy(buf, path); for(p = strchr(buf, '/'); p != NULL; p = strchr(p + 1, '/')){ *p = '\0'; if(buf[0] != '\0') mkdir(buf, 0777); *p = '/'; } } /* blog URL -> repo-relative path, e.g. 2026/05/29/hello-world.html */ static const char * urlpath(const char *link) { const char *p = link; if(strncmp(p, SITE, strlen(SITE)) == 0) p += strlen(SITE); else if((p = strstr(link, "://")) != NULL){ p += 3; p = strchr(p, '/'); if(p == NULL) return ""; p++; } else p = link; while(*p == '/') p++; return p; } static void header(FILE *f, const char *title) { fputs("\n\n\n\n" " \n ", f); puthtml(f, title); fputs(" \xe2\x80\x93 SillyLaird\n\n\n\n", f); } static void footer(FILE *f, const char *up) { fputs("\n
\n\n

\n Blog index\n

\n\n\n\n\n", f); } static void writepost(const char *path, const char *title, const char *date, const char *body, const char *link) { char file[1024]; char up[64]; const char *p; FILE *f; int depth; if(path[0] == '\0') return; file[0] = '\0'; if(!cats(file, sizeof file, outdir) || !cats(file, sizeof file, "/") || !cats(file, sizeof file, path)){ fprintf(stderr, "blogmirror: path too long: %s\n", path); return; } mkpath(file); if((f = fopen(file, "wb")) == NULL){ fprintf(stderr, "blogmirror: %s: cannot create\n", file); return; } up[0] = '\0'; for(depth = 0, p = path; *p != '\0'; p++) if(*p == '/') depth++; for(; depth > 0 && strlen(up) + 3 < sizeof up; depth--) strcat(up, "../"); header(f, title); fputs("

", f); puthtml(f, title); fputs("

\n\n

", f); puthtml(f, date); fputs("

\n\n", f); fputs(body, f); /* already HTML from the blog */ fputs("\n\n

Mirror of ", f); puthtml(f, link); fputs("

\n", f); footer(f, up); fclose(f); fprintf(stderr, "%s\n", file); } int main(int argc, char *argv[]) { char *url = FEEDURL; char *feedfile = NULL; char *feed, *p, *end, *item; char cmd[1024]; char file[1024]; FILE *ix; int i; for(i = 1; i < argc; i++){ if(strcmp(argv[i], "-u") == 0 && i + 1 < argc) url = argv[++i]; else if(strcmp(argv[i], "-f") == 0 && i + 1 < argc) feedfile = argv[++i]; else if(strcmp(argv[i], "-o") == 0 && i + 1 < argc) outdir = argv[++i]; else if(strcmp(argv[i], "-n") == 0 && i + 1 < argc) maxposts = atoi(argv[++i]); else { fprintf(stderr, "usage: blogmirror [-u url] [-f file] [-o outdir] [-n maxposts]\n"); exit(1); } } if(feedfile == NULL){ cmd[0] = '\0'; if(exists("/bin/hget")){ /* 9front */ cats(cmd, sizeof cmd, "hget '"); cats(cmd, sizeof cmd, url); cats(cmd, sizeof cmd, "' >"); cats(cmd, sizeof cmd, TMPFEED); } else { cats(cmd, sizeof cmd, "curl -fsS '"); cats(cmd, sizeof cmd, url); cats(cmd, sizeof cmd, "' -o "); cats(cmd, sizeof cmd, TMPFEED); } if(system(cmd) != 0){ fprintf(stderr, "blogmirror: fetch failed: %s\n", url); exit(1); } feedfile = TMPFEED; } if((feed = slurp(feedfile)) == NULL){ fprintf(stderr, "blogmirror: %s: cannot read\n", feedfile); exit(1); } mkdir(outdir, 0777); file[0] = '\0'; cats(file, sizeof file, outdir); cats(file, sizeof file, "/index.html"); if((ix = fopen(file, "wb")) == NULL){ fprintf(stderr, "blogmirror: %s: cannot create\n", file); exit(1); } header(ix, "Blog"); fputs("

Blog

\n\n" "

Mirror of blog.sillylaird.ca.

\n\n" " \n", ix); for(p = feed; (p = strstr(p, "")) != NULL; p = end){ char *title, *link, *date, *desc, *body; const char *path; if((end = strstr(p, "")) == NULL) break; item = p; title = unxml(between(item, end, "", "")); link = unxml(between(item, end, "", "")); date = unxml(between(item, end, "", "")); desc = unxml(between(item, end, "", "")); body = between(item, end, ""); if(title == NULL || link == NULL){ free(title); free(link); free(date); free(desc); free(body); continue; } if(date == NULL) date = estrdup(""); if(body == NULL) body = estrdup(desc != NULL ? desc : ""); path = urlpath(link); writepost(path, title, date, body, link); fputs(" \n \n \n \n", ix); free(title); free(link); free(date); free(desc); free(body); nposts++; if(maxposts > 0 && nposts >= maxposts) break; end += 7; /* past */ } fputs("
\n" " ", ix); puthtml(ix, title); fputs("
\n" " \n" " \n \n \n \n" "
\n" " \n ", ix); puthtml(ix, date); fputs("
\n ", ix); puthtml(ix, desc != NULL ? desc : ""); fputs("\n
\n
\n
\n", ix); footer(ix, ""); fclose(ix); free(feed); fprintf(stderr, "%s: %d posts\n", file, nposts); return 0; }