aboutsummaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
-rw-r--r--README.md34
-rw-r--r--blogmirror.c424
-rw-r--r--html/2026/05/29/hello-world.html26
-rw-r--r--html/index.html42
-rwxr-xr-xmirror.rc22
-rw-r--r--mkfile16
6 files changed, 564 insertions, 0 deletions
diff --git a/README.md b/README.md
new file mode 100644
index 0000000..132ab05
--- /dev/null
+++ b/README.md
@@ -0,0 +1,34 @@
+# blogmirror
+
+Mirrors <https://blog.sillylaird.ca> into flat, mothra-friendly HTML under
+`html/`, in the same plain style as the rest of this site.
+
+It reads the blog's RSS feed (`/feed.xml.php`), which carries the full post
+body in `<content:encoded>`, so no HTML scraping is involved. Output:
+
+ html/index.html post list
+ html/YYYY/MM/DD/slug.html one page per post
+
+## Build (9front)
+
+ mk
+
+`blogmirror.c` is plain ANSI C and builds with APE `pcc`; it also builds with
+`cc` on Linux/BSD for testing. Fetching uses `hget` when `/bin/hget` exists,
+`curl` otherwise.
+
+## Run
+
+ ./blogmirror mirror the live feed into html/
+ ./blogmirror -o /tmp/out somewhere else
+ ./blogmirror -n 10 only the newest 10 posts
+ ./blogmirror -f feed.rss parse a feed already on disk
+ ./blogmirror -u URL a different feed
+
+## Refresh and publish
+
+ ./mirror.rc
+
+which regenerates `html/`, commits the changed files, and runs `git/push`.
+Requires `git/fs`, a loaded ssh key in factotum, and a remote already set in
+`.git/config`.
diff --git a/blogmirror.c b/blogmirror.c
new file mode 100644
index 0000000..a479386
--- /dev/null
+++ b/blogmirror.c
@@ -0,0 +1,424 @@
+/*
+ * blogmirror - mirror blog.sillylaird.ca into flat mothra-friendly HTML.
+ *
+ * Reads the blog's RSS 2.0 feed (which carries the full post body in
+ * <content:encoded>) and writes a static tree:
+ *
+ * index.html list of posts
+ * YYYY/MM/DD/slug.html one page per post
+ *
+ * Plain ANSI C: builds with pcc on 9front (APE) and with cc elsewhere.
+ *
+ * blogmirror [-u url] [-f file] [-o outdir] [-n maxposts]
+ *
+ * With -f it parses a feed already on disk; otherwise it shells out to
+ * hget(1) (curl elsewhere) to fetch -u into a temporary file first.
+ */
+
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/types.h>
+#include <sys/stat.h>
+
+#define FEEDURL "https://blog.sillylaird.ca/feed.xml.php"
+#define SITE "https://blog.sillylaird.ca/"
+#define TMPFEED "/tmp/blogmirror.rss"
+
+static char *outdir = "html";
+static int maxposts = 0; /* 0 = all */
+static int nposts = 0;
+
+static void *
+emalloc(size_t n)
+{
+ void *p = malloc(n);
+ if(p == NULL){
+ fprintf(stderr, "blogmirror: out of memory\n");
+ exit(1);
+ }
+ return p;
+}
+
+static char *
+estrdup(const char *s)
+{
+ char *p = emalloc(strlen(s) + 1);
+ strcpy(p, s);
+ return p;
+}
+
+/* bounded concatenation; returns 0 if it would not fit */
+static int
+cats(char *d, size_t n, const char *s)
+{
+ size_t have = strlen(d), want = strlen(s);
+
+ if(have + want + 1 > n)
+ return 0;
+ memcpy(d + have, s, want + 1);
+ return 1;
+}
+
+/* does this file exist? */
+static int
+exists(const char *path)
+{
+ FILE *f = fopen(path, "rb");
+
+ if(f == NULL)
+ return 0;
+ fclose(f);
+ return 1;
+}
+
+/* read a whole file, NUL terminated */
+static char *
+slurp(const char *path)
+{
+ FILE *f;
+ char *b;
+ long n;
+
+ if((f = fopen(path, "rb")) == NULL)
+ return NULL;
+ fseek(f, 0L, SEEK_END);
+ n = ftell(f);
+ rewind(f);
+ if(n < 0){
+ fclose(f);
+ return NULL;
+ }
+ b = emalloc((size_t)n + 1);
+ if(fread(b, 1, (size_t)n, f) != (size_t)n){
+ fclose(f);
+ free(b);
+ return NULL;
+ }
+ b[n] = '\0';
+ fclose(f);
+ return b;
+}
+
+/* text between open and close, starting at p, not past end. malloc'd. */
+static char *
+between(char *p, char *end, const char *open, const char *close)
+{
+ char *a, *b, *s;
+ size_t n;
+
+ if((a = strstr(p, open)) == NULL || a >= end)
+ return NULL;
+ a += strlen(open);
+ if((b = strstr(a, close)) == NULL || b > end)
+ return NULL;
+ n = (size_t)(b - a);
+ s = emalloc(n + 1);
+ memcpy(s, a, n);
+ s[n] = '\0';
+ return s;
+}
+
+/* one UTF-8 rune out of a code point */
+static int
+putrune(char *d, long r)
+{
+ if(r < 0x80){
+ d[0] = (char)r;
+ return 1;
+ }
+ if(r < 0x800){
+ d[0] = (char)(0xC0 | (r >> 6));
+ d[1] = (char)(0x80 | (r & 0x3F));
+ return 2;
+ }
+ if(r < 0x10000){
+ d[0] = (char)(0xE0 | (r >> 12));
+ d[1] = (char)(0x80 | ((r >> 6) & 0x3F));
+ d[2] = (char)(0x80 | (r & 0x3F));
+ return 3;
+ }
+ d[0] = (char)(0xF0 | (r >> 18));
+ d[1] = (char)(0x80 | ((r >> 12) & 0x3F));
+ d[2] = (char)(0x80 | ((r >> 6) & 0x3F));
+ d[3] = (char)(0x80 | (r & 0x3F));
+ return 4;
+}
+
+/* &amp; &lt; &gt; &quot; &apos; &#39; &#x2019; -> text, in place */
+static char *
+unxml(char *s)
+{
+ char *r, *w;
+ long v;
+
+ if(s == NULL)
+ return NULL;
+ for(r = w = s; *r != '\0'; ){
+ if(*r != '&'){
+ *w++ = *r++;
+ continue;
+ }
+ if(strncmp(r, "&amp;", 5) == 0){ *w++ = '&'; r += 5; continue; }
+ if(strncmp(r, "&lt;", 4) == 0){ *w++ = '<'; r += 4; continue; }
+ if(strncmp(r, "&gt;", 4) == 0){ *w++ = '>'; r += 4; continue; }
+ if(strncmp(r, "&quot;", 6) == 0){ *w++ = '"'; r += 6; continue; }
+ if(strncmp(r, "&apos;", 6) == 0){ *w++ = '\''; r += 6; continue; }
+ if(r[1] == '#'){
+ char *e;
+ if(r[2] == 'x' || r[2] == 'X')
+ v = strtol(r + 3, &e, 16);
+ else
+ v = strtol(r + 2, &e, 10);
+ if(*e == ';' && v > 0){
+ w += putrune(w, v);
+ r = e + 1;
+ continue;
+ }
+ }
+ *w++ = *r++;
+ }
+ *w = '\0';
+ return s;
+}
+
+/* escape text for HTML output */
+static void
+puthtml(FILE *f, const char *s)
+{
+ if(s == NULL)
+ return;
+ for(; *s != '\0'; s++)
+ switch(*s){
+ case '&': fputs("&amp;", f); break;
+ case '<': fputs("&lt;", f); break;
+ case '>': fputs("&gt;", f); break;
+ case '"': fputs("&quot;", f); break;
+ default: fputc(*s, f); break;
+ }
+}
+
+/* mkdir -p over the directory part of path */
+static void
+mkpath(const char *path)
+{
+ char buf[1024];
+ char *p;
+
+ if(strlen(path) >= sizeof buf)
+ return;
+ strcpy(buf, path);
+ for(p = strchr(buf, '/'); p != NULL; p = strchr(p + 1, '/')){
+ *p = '\0';
+ if(buf[0] != '\0')
+ mkdir(buf, 0777);
+ *p = '/';
+ }
+}
+
+/* blog URL -> repo-relative path, e.g. 2026/05/29/hello-world.html */
+static const char *
+urlpath(const char *link)
+{
+ const char *p = link;
+
+ if(strncmp(p, SITE, strlen(SITE)) == 0)
+ p += strlen(SITE);
+ else if((p = strstr(link, "://")) != NULL){
+ p += 3;
+ p = strchr(p, '/');
+ if(p == NULL)
+ return "";
+ p++;
+ } else
+ p = link;
+ while(*p == '/')
+ p++;
+ return p;
+}
+
+static void
+header(FILE *f, const char *title)
+{
+ fputs("<!DOCTYPE html>\n<html>\n\n<head>\n"
+ " <meta charset=\"utf-8\">\n <title>", f);
+ puthtml(f, title);
+ fputs(" \xe2\x80\x93 SillyLaird</title>\n</head>\n\n<body>\n", f);
+}
+
+static void
+footer(FILE *f, const char *up)
+{
+ fputs("\n <hr>\n\n <p>\n <a href=\"", f);
+ fputs(up, f);
+ fputs("index.html\">Blog index</a>\n </p>\n\n</body>\n\n</html>\n", f);
+}
+
+static void
+writepost(const char *path, const char *title, const char *date,
+ const char *body, const char *link)
+{
+ char file[1024];
+ char up[64];
+ const char *p;
+ FILE *f;
+ int depth;
+
+ if(path[0] == '\0')
+ return;
+ file[0] = '\0';
+ if(!cats(file, sizeof file, outdir) || !cats(file, sizeof file, "/")
+ || !cats(file, sizeof file, path)){
+ fprintf(stderr, "blogmirror: path too long: %s\n", path);
+ return;
+ }
+ mkpath(file);
+ if((f = fopen(file, "wb")) == NULL){
+ fprintf(stderr, "blogmirror: %s: cannot create\n", file);
+ return;
+ }
+
+ up[0] = '\0';
+ for(depth = 0, p = path; *p != '\0'; p++)
+ if(*p == '/')
+ depth++;
+ for(; depth > 0 && strlen(up) + 3 < sizeof up; depth--)
+ strcat(up, "../");
+
+ header(f, title);
+ fputs(" <h2>", f);
+ puthtml(f, title);
+ fputs("</h2>\n\n <p><font size=\"-1\">", f);
+ puthtml(f, date);
+ fputs("</font></p>\n\n", f);
+ fputs(body, f); /* already HTML from the blog */
+ fputs("\n\n <p><font size=\"-1\">Mirror of <a href=\"", f);
+ puthtml(f, link);
+ fputs("\">", f);
+ puthtml(f, link);
+ fputs("</a></font></p>\n", f);
+ footer(f, up);
+ fclose(f);
+ fprintf(stderr, "%s\n", file);
+}
+
+int
+main(int argc, char *argv[])
+{
+ char *url = FEEDURL;
+ char *feedfile = NULL;
+ char *feed, *p, *end, *item;
+ char cmd[1024];
+ char file[1024];
+ FILE *ix;
+ int i;
+
+ for(i = 1; i < argc; i++){
+ if(strcmp(argv[i], "-u") == 0 && i + 1 < argc)
+ url = argv[++i];
+ else if(strcmp(argv[i], "-f") == 0 && i + 1 < argc)
+ feedfile = argv[++i];
+ else if(strcmp(argv[i], "-o") == 0 && i + 1 < argc)
+ outdir = argv[++i];
+ else if(strcmp(argv[i], "-n") == 0 && i + 1 < argc)
+ maxposts = atoi(argv[++i]);
+ else {
+ fprintf(stderr,
+ "usage: blogmirror [-u url] [-f file] [-o outdir] [-n maxposts]\n");
+ exit(1);
+ }
+ }
+
+ if(feedfile == NULL){
+ cmd[0] = '\0';
+ if(exists("/bin/hget")){ /* 9front */
+ cats(cmd, sizeof cmd, "hget '");
+ cats(cmd, sizeof cmd, url);
+ cats(cmd, sizeof cmd, "' >");
+ cats(cmd, sizeof cmd, TMPFEED);
+ } else {
+ cats(cmd, sizeof cmd, "curl -fsS '");
+ cats(cmd, sizeof cmd, url);
+ cats(cmd, sizeof cmd, "' -o ");
+ cats(cmd, sizeof cmd, TMPFEED);
+ }
+ if(system(cmd) != 0){
+ fprintf(stderr, "blogmirror: fetch failed: %s\n", url);
+ exit(1);
+ }
+ feedfile = TMPFEED;
+ }
+
+ if((feed = slurp(feedfile)) == NULL){
+ fprintf(stderr, "blogmirror: %s: cannot read\n", feedfile);
+ exit(1);
+ }
+
+ mkdir(outdir, 0777);
+ file[0] = '\0';
+ cats(file, sizeof file, outdir);
+ cats(file, sizeof file, "/index.html");
+ if((ix = fopen(file, "wb")) == NULL){
+ fprintf(stderr, "blogmirror: %s: cannot create\n", file);
+ exit(1);
+ }
+ header(ix, "Blog");
+ fputs(" <h2>Blog</h2>\n\n"
+ " <p>Mirror of <a href=\"" SITE "\">blog.sillylaird.ca</a>.</p>\n\n"
+ " <table cellspacing=\"0\" cellpadding=\"0\" border=\"0\">\n", ix);
+
+ for(p = feed; (p = strstr(p, "<item>")) != NULL; p = end){
+ char *title, *link, *date, *desc, *body;
+ const char *path;
+
+ if((end = strstr(p, "</item>")) == NULL)
+ break;
+ item = p;
+ title = unxml(between(item, end, "<title>", "</title>"));
+ link = unxml(between(item, end, "<link>", "</link>"));
+ date = unxml(between(item, end, "<pubDate>", "</pubDate>"));
+ desc = unxml(between(item, end, "<description>", "</description>"));
+ body = between(item, end, "<content:encoded><![CDATA[", "]]></content:encoded>");
+
+ if(title == NULL || link == NULL){
+ free(title); free(link); free(date); free(desc); free(body);
+ continue;
+ }
+ if(date == NULL)
+ date = estrdup("");
+ if(body == NULL)
+ body = estrdup(desc != NULL ? desc : "");
+
+ path = urlpath(link);
+ writepost(path, title, date, body, link);
+
+ fputs(" <tr>\n <td width=\"10\"></td>\n <td valign=\"top\">\n"
+ " <a href=\"", ix);
+ puthtml(ix, path);
+ fputs("\">", ix);
+ puthtml(ix, title);
+ fputs("</a><br>\n"
+ " <table cellspacing=\"0\" cellpadding=\"0\" border=\"0\">\n"
+ " <tr>\n <td width=\"10\"></td>\n <td>\n"
+ " <font size=\"-1\">\n ", ix);
+ puthtml(ix, date);
+ fputs("<br>\n ", ix);
+ puthtml(ix, desc != NULL ? desc : "");
+ fputs("\n </font>\n </td>\n </tr>\n"
+ " </table>\n </td>\n </tr>\n", ix);
+
+ free(title); free(link); free(date); free(desc); free(body);
+ nposts++;
+ if(maxposts > 0 && nposts >= maxposts)
+ break;
+ end += 7; /* past </item> */
+ }
+
+ fputs(" </table>\n", ix);
+ footer(ix, "");
+ fclose(ix);
+ free(feed);
+
+ fprintf(stderr, "%s: %d posts\n", file, nposts);
+ return 0;
+}
diff --git a/html/2026/05/29/hello-world.html b/html/2026/05/29/hello-world.html
new file mode 100644
index 0000000..2a63edc
--- /dev/null
+++ b/html/2026/05/29/hello-world.html
@@ -0,0 +1,26 @@
+<!DOCTYPE html>
+<html>
+
+<head>
+ <meta charset="utf-8">
+ <title>Hello World! – SillyLaird</title>
+</head>
+
+<body>
+ <h2>Hello World!</h2>
+
+ <p><font size="-1">Fri, 29 May 2026 13:50:50 +0000</font></p>
+
+this site has been updated pretty much to how i want it. now i just need to polish some ideas and write content for it. most of the tools are coded so i dont have to worry about another project. :)
+
+ <p><font size="-1">Mirror of <a href="https://blog.sillylaird.ca/2026/05/29/hello-world.html">https://blog.sillylaird.ca/2026/05/29/hello-world.html</a></font></p>
+
+ <hr>
+
+ <p>
+ <a href="../../../index.html">Blog index</a>
+ </p>
+
+</body>
+
+</html>
diff --git a/html/index.html b/html/index.html
new file mode 100644
index 0000000..eca7576
--- /dev/null
+++ b/html/index.html
@@ -0,0 +1,42 @@
+<!DOCTYPE html>
+<html>
+
+<head>
+ <meta charset="utf-8">
+ <title>Blog – SillyLaird</title>
+</head>
+
+<body>
+ <h2>Blog</h2>
+
+ <p>Mirror of <a href="https://blog.sillylaird.ca/">blog.sillylaird.ca</a>.</p>
+
+ <table cellspacing="0" cellpadding="0" border="0">
+ <tr>
+ <td width="10"></td>
+ <td valign="top">
+ <a href="2026/05/29/hello-world.html">Hello World!</a><br>
+ <table cellspacing="0" cellpadding="0" border="0">
+ <tr>
+ <td width="10"></td>
+ <td>
+ <font size="-1">
+ Fri, 29 May 2026 13:50:50 +0000<br>
+ this site has been updated pretty much to how i want it. now i just need to polish some ideas and write content for it. most of the tools are coded so i dont have to worry about another project. :)
+ </font>
+ </td>
+ </tr>
+ </table>
+ </td>
+ </tr>
+ </table>
+
+ <hr>
+
+ <p>
+ <a href="index.html">Blog index</a>
+ </p>
+
+</body>
+
+</html>
diff --git a/mirror.rc b/mirror.rc
new file mode 100755
index 0000000..8d1b106
--- /dev/null
+++ b/mirror.rc
@@ -0,0 +1,22 @@
+#!/bin/rc
+# Refresh the blog mirror in html/ and push it. Run from the repo root.
+rfork en
+
+if(! test -x ./blogmirror){
+ echo 'build first: mk' >[1=2]
+ exit 'nobinary'
+}
+
+git/fs >[2]/dev/null # harmless if already running
+
+./blogmirror -o html || exit 'mirror'
+
+files=`{walk -f html}
+if(~ $#files 0)
+ exit 'nothing mirrored'
+
+git/add $files
+if(git/commit -m 'mirror blog.sillylaird.ca' $files)
+ git/push
+if not
+ echo 'no changes' >[1=2]
diff --git a/mkfile b/mkfile
new file mode 100644
index 0000000..34ddb90
--- /dev/null
+++ b/mkfile
@@ -0,0 +1,16 @@
+# 9front: mk (uses APE cc, needs stdio + system(2))
+CC=pcc
+TARG=blogmirror
+
+$TARG: blogmirror.c
+ $CC -o $TARG blogmirror.c
+
+install:V: $TARG
+ mkdir -p $home/bin/$objtype
+ cp $TARG $home/bin/$objtype/$TARG
+
+test:V: $TARG
+ ./$TARG -o /tmp/blogmirror.out
+
+clean:V:
+ rm -f $TARG *.[0-9] [0-9].out