From 78a4d76b9d8e16ebf8b0e9dddc8a31005153ba7e Mon Sep 17 00:00:00 2001 From: "@permadeath.com" Date: Fri, 7 Aug 2026 18:49:51 -0400 Subject: [PATCH] feat(web): the blog has an Atom feed MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit There was no way to follow the blog except by visiting it. The collection's frontmatter already carries everything a feed wants, so this is an endpoint at /atom.xml and a in the layout — no dependency, and a document short enough to read. Atom rather than RSS: `published` and `updated` mean exactly what publishDate and updatedDate mean, and every entry is required to carry an id, so a reader can tell a retitled post from a new one. The feed is dated by its newest post rather than by the build, or a deploy would report news on a day nothing was written. push.sh gives it the content type the layout claims for it; the CLI would otherwise guess application/xml from the extension. --- README.md | 2 +- web/scripts/atom.test.mjs | 157 +++++++++++++++++++++++++++++++++ web/scripts/push.sh | 10 +++ web/src/layouts/Base.astro | 11 +++ web/src/pages/atom.xml.ts | 112 +++++++++++++++++++++++ web/src/pages/blog/index.astro | 8 ++ web/src/styles.css | 9 ++ 7 files changed, 308 insertions(+), 1 deletion(-) create mode 100644 web/scripts/atom.test.mjs create mode 100644 web/src/pages/atom.xml.ts diff --git a/README.md b/README.md index ae1a3a8..ea537ce 100644 --- a/README.md +++ b/README.md @@ -192,7 +192,7 @@ locally on an x86 machine. | launching a match against Princess | works, in production (M1). Session-gated; Fargate only — local dev answers 503 | | the WebSocket proxy in front of a match | works, in production. Lives inside the api binary, so an api deploy drops every live match (M5 splits it out) | | listing your matches | works. Stored status only, so a match that died still reads `ready` | -| the site's five destinations — Home, Camo, FAQ, Blog, About | works. Camo needs no account to design or download a camo. Blog is prerendered from markdown in `web/src/content/blog`; the posts are not ATProto records yet | +| the site's five destinations — Home, Camo, FAQ, Blog, About | works. Camo needs no account to design or download a camo. Blog is prerendered from markdown in `web/src/content/blog` and has an Atom feed at `/atom.xml`; the posts are not ATProto records yet | | writing anything to a player's PDS | camo only: save, rename and delete, written by the API with the player's grant. Never exercised against a real account yet. Match results are still M3 | | player camo | designing, saving and editing work; the collection is read straight from the player's PDS. What is missing is the other half of M4 — nothing yet puts a saved camo into a match manifest | | the AppView | not started | diff --git a/web/scripts/atom.test.mjs b/web/scripts/atom.test.mjs new file mode 100644 index 0000000..4733d04 --- /dev/null +++ b/web/scripts/atom.test.mjs @@ -0,0 +1,157 @@ +/** + * The blog's feed, read out of the build. + * + * A feed is the one page on the site nobody looks at. It is fetched by + * software, in a reader somewhere else, and every way it goes wrong is silent + * from here: an unescaped ampersand in a title ends the document at that byte, + * a relative href resolves against the reader rather than against the site, + * and a feed stamped with the build time reports news on every deploy. + * + * Read out of dist/ rather than out of the source, because the source is a + * function and the thing subscribers get is its output. `npm test` builds + * first. + * + * Run with `npm test`. + */ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { existsSync, readFileSync, readdirSync } from "node:fs"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; + +const web = fileURLToPath(new URL("../", import.meta.url)); +const dist = join(web, "dist"); + +const SITE = "https://lance.blue"; + +const feedPath = join(dist, "atom.xml"); +const feed = existsSync(feedPath) ? readFileSync(feedPath, "utf8") : null; + +/** Every occurrence of one element's text content, in document order. */ +function all(name) { + return [...feed.matchAll(new RegExp(`<${name}>([^<]*)`, "g"))].map( + (m) => m[1], + ); +} + +/** The entries, as their own strings, so a field can be read per entry. */ +function entries() { + return [...feed.matchAll(/([\s\S]*?)<\/entry>/g)].map((m) => m[1]); +} + +test("the build writes the feed the layout points at", () => { + assert.ok(feed, `no ${feedPath}. The endpoint did not build.`); + + const layout = readFileSync(join(web, "src/layouts/Base.astro"), "utf8"); + const link = + //.exec( + layout, + ); + assert.ok( + link, + "Base.astro no longer advertises a feed, so nothing finds it", + ); + assert.match( + link[0], + /href="\/atom\.xml"/, + "the layout points somewhere the build does not write", + ); +}); + +test("nothing in the document is text that XML would read as markup", () => { + // The failure this exists for: a title with an ampersand in it. Every reader + // stops at that byte, and the site looks fine. + const bare = [ + ...feed.matchAll(/&(?!(?:amp|lt|gt|quot|apos|#\d+|#x[0-9a-f]+);)/gi), + ]; + assert.equal( + bare.length, + 0, + `${bare.length} bare & in the feed; the first is at ${bare[0]?.index}`, + ); + + // Angle brackets are harder to catch by eye and the same class of bug, so + // pin the shape instead: every < in the document opens a tag this file wrote. + const tags = [...feed.matchAll(/<[^>]*>/g)].length; + const opens = [...feed.matchAll(/ { + for (const href of [...feed.matchAll(/href="([^"]*)"/g)].map((m) => m[1])) { + assert.ok(href.startsWith(SITE), `${href} is not absolute against ${SITE}`); + } + for (const id of all("id")) { + assert.ok(id.startsWith(SITE), `id ${id} is not absolute against ${SITE}`); + } +}); + +test("the feed says where it is, and where its human page is", () => { + assert.match( + feed, + new RegExp(`rel="self"[^>]*href="${SITE}/atom\\.xml"`), + "no rel=self, so a reader handed this document cannot refetch it", + ); + assert.match( + feed, + new RegExp(`rel="alternate" type="text/html" href="${SITE}/blog"`), + "no alternate, so there is nowhere to send a human", + ); +}); + +test("there is one entry per post the build published", () => { + const built = readdirSync(join(dist, "blog"), { withFileTypes: true }) + .filter((e) => e.isDirectory()) + .map((e) => e.name); + assert.ok(built.length, "the build wrote no posts at all"); + + const linked = entries().map((entry) => /([^<]*)<\/id>/.exec(entry)?.[1]); + assert.deepEqual( + [...linked].sort(), + built.map((slug) => `${SITE}/blog/${slug}`).sort(), + "the feed and the built pages disagree about what has been posted", + ); +}); + +test("the feed is dated by its newest post, not by the build", () => { + // A feed stamped at build time changes on every deploy, and a reader that + // trusts the field reports news on a day nothing was written. + const feedUpdated = all("updated")[0]; + const perEntry = entries().map( + (entry) => /([^<]*)<\/updated>/.exec(entry)?.[1], + ); + assert.ok(perEntry.length, "no entries to date the feed by"); + + const newest = perEntry.slice().sort().at(-1); + assert.equal( + feedUpdated, + newest, + "the feed's is not its newest entry's", + ); + + // The entries are newest first, which is the order a reader shows them in + // when it has nothing else to go on. + assert.deepEqual( + perEntry, + perEntry.slice().sort().reverse(), + "the entries are not newest first", + ); +}); + +test("a draft is not in the feed", () => { + const posts = join(web, "src/content/blog"); + const drafts = readdirSync(posts) + .filter((name) => name.endsWith(".md")) + .filter((name) => + /^draft:\s*true$/m.test(readFileSync(join(posts, name), "utf8")), + ) + .map((name) => name.replace(/\.md$/, "")); + + for (const slug of drafts) { + assert.doesNotMatch( + feed, + new RegExp(`${SITE}/blog/${slug}\\b`), + `${slug} is a draft and is in the feed`, + ); + } +}); diff --git a/web/scripts/push.sh b/web/scripts/push.sh index 9a0bc0f..0eac06f 100755 --- a/web/scripts/push.sh +++ b/web/scripts/push.sh @@ -50,6 +50,16 @@ while IFS= read -r page; do --cache-control "no-cache" done < <(find dist -mindepth 2 -name index.html) +# The feed does have an extension, and what the CLI guesses from it is +# application/xml. Readers take that, but it is not what the on every +# page says the document is, and the two disagreeing is the sort of thing that +# is only ever noticed by whichever reader is strict about it. +if [ -f dist/atom.xml ]; then + aws s3 cp dist/atom.xml "s3://$BUCKET/$prefix/atom.xml" \ + --content-type "application/atom+xml; charset=utf-8" \ + --cache-control "no-cache" +fi + # The same fix for the standard.site key, which has no extension either and # has to arrive as text rather than as a download. Absent until `sequoia init` # writes it, so this is a no-op until the publication record exists. diff --git a/web/src/layouts/Base.astro b/web/src/layouts/Base.astro index 6381aba..73c25f2 100644 --- a/web/src/layouts/Base.astro +++ b/web/src/layouts/Base.astro @@ -99,6 +99,17 @@ const cardSize = image === DEFAULT_CARD ? { width: 1200, height: 630 } : null; + { + /* Where a reader looks for the feed. On every page rather than only the + blog's: autodiscovery is how a subscribe button finds one at all, and + it is asked of whatever address the reader was handed. */ + } + diff --git a/web/src/pages/atom.xml.ts b/web/src/pages/atom.xml.ts new file mode 100644 index 0000000..b3811aa --- /dev/null +++ b/web/src/pages/atom.xml.ts @@ -0,0 +1,112 @@ +/** + * The blog as a feed. + * + * Hand-written rather than pulled from a package: the whole of Atom that this + * needs is one element per post, the frontmatter already carries every field + * it wants, and a feed is a thing you have to be able to read to trust. + * + * Atom rather than RSS because the collection's dates map onto it exactly — + * `published` never moves and `updated` is what `updatedDate` means — and + * because every entry is required to carry an id, so a reader can tell a + * retitled post from a new one. + * + * Every URL here is absolute against `site` in astro.config.mjs. A feed is + * read somewhere else by definition, and a relative href in one resolves + * against the reader, which is nowhere. + */ + +import type { APIContext } from "astro"; +import { getCollection } from "astro:content"; + +/** Where readers subscribe. Also spelled out in Base.astro's . */ +export const FEED_PATH = "/atom.xml"; + +/** + * The five characters that are not text in XML. + * + * Everything that goes into the document below goes through here, including + * the parts that look like they could not contain one: a post title is written + * by a human in a markdown file, and an ampersand in one would otherwise end + * the document at that byte for every reader at once. + */ +function xml(text: string): string { + return text + .replace(/&/g, "&") + .replace(//g, ">") + .replace(/"/g, """) + .replace(/'/g, "'"); +} + +export async function GET(context: APIContext): Promise { + const site = context.site; + if (!site) throw new Error("astro.config.mjs has no site; the feed needs it"); + + const url = (path: string): string => new URL(path, site).href; + + // The same rule the blog index uses, so a reader and a visitor never + // disagree about what has been posted. Drafts are readable while writing and + // are not built, and the check is on import.meta.env.DEV rather than on a + // flag so there is no way to ship one by forgetting to turn something back + // on. + const posts = ( + await getCollection( + "blog", + ({ data }) => import.meta.env.DEV || !data.draft, + ) + ).sort((a, b) => b.data.publishDate.getTime() - a.data.publishDate.getTime()); + + const stamp = (date: Date): string => date.toISOString(); + const changed = (post: (typeof posts)[number]): Date => + post.data.updatedDate ?? post.data.publishDate; + + // The feed's own timestamp is the newest post's, never the build's. A feed + // stamped at build time changes every deploy, and a reader that trusts the + // field reports news on a day nothing was written. + // + // With no posts at all there is nothing to date it by, and Atom requires the + // element. The epoch is the honest answer: nothing here has ever been + // updated. + const updated = posts.length ? changed(posts[0]!) : new Date(0); + + const entries = posts.map((post) => { + const href = url(`/blog/${post.id}`); + return [ + " ", + ` ${xml(post.data.title)}`, + ` `, + // The address is the id. It is stable for as long as the post is at it, + // which is what `trailingSlash: "never"` and the canonical link already + // promise everywhere else on the site. + ` ${xml(href)}`, + ` ${stamp(post.data.publishDate)}`, + ` ${stamp(changed(post))}`, + ` ${xml(post.data.description)}`, + ...post.data.tags.map((tag) => ` `), + " ", + ].join("\n"); + }); + + const document = [ + '', + '', + // The same string the blog index's carries, so the two do not + // drift into naming the same thing differently. No <subtitle>: there is + // nothing to say there that the site does not already say. + " <title>Blog · lance.blue", + ` ${xml(url(FEED_PATH))}`, + ` ${stamp(updated)}`, + // rel="self" is how a reader that was handed this document knows where to + // fetch it again; the alternate is the page a human should be sent to. + ` `, + ` `, + " lance.blue", + ...entries, + "", + "", + ].join("\n"); + + return new Response(document, { + headers: { "content-type": "application/atom+xml; charset=utf-8" }, + }); +} diff --git a/web/src/pages/blog/index.astro b/web/src/pages/blog/index.astro index a989314..47e4b2b 100644 --- a/web/src/pages/blog/index.astro +++ b/web/src/pages/blog/index.astro @@ -52,6 +52,14 @@ const day = new Intl.DateTimeFormat("en", { ) } + { + /* Autodiscovery is in the layout's head, and no browser has surfaced it + for years. This is how anyone actually subscribes. */ + } + +