From 22997bce06c768688d3d77429f8f2cf364bddc4c Mon Sep 17 00:00:00 2001 From: yusuf-gundogdu Date: Sat, 26 Sep 2026 01:59:32 +0300 Subject: [PATCH 1/2] fix(rss): put the author's name where RSS 2.0 allows it The site feed wrote the author's name into , which RSS 2.0 defines as an email address. The W3C feed validator refuses the feed over it, once per item, so https://libredb.org/rss.xml does not validate today. The name moves to , which is what readers and planet aggregators already read, and the channel gains the atom:link rel="self" the validator asks for. Both live behind namespace declarations on , so removing either declaration leaves every feed on the site not well-formed XML and every reader rejecting it. Nothing in the suite noticed that. dist-smoke now parses each built feed, checks that every prefix it uses is declared, and pins the escaping of a creator name carrying an ampersand, which is the one we ship. --- src/lib/feed.ts | 36 +++++++++++++++++++++++++++++++++++ src/pages/rss.xml.ts | 8 ++++++-- tests/dist-smoke.test.ts | 41 ++++++++++++++++++++++++++++++++++++++++ 3 files changed, 83 insertions(+), 2 deletions(-) create mode 100644 src/lib/feed.ts diff --git a/src/lib/feed.ts b/src/lib/feed.ts new file mode 100644 index 0000000..c1092d3 --- /dev/null +++ b/src/lib/feed.ts @@ -0,0 +1,36 @@ +import { canonicalUrl } from './site'; + +/** + * The two namespaces and the self link every feed on this site declares. + * + * RSS 2.0 defines as an email address, so a bare name there fails the + * W3C feed validator: the live site feed answers "Invalid email address" once + * per item. The author's name is Dublin Core's job. rel="self" is the + * validator's one standing recommendation for a feed that does not say where + * it lives. + * + * Both are declared on the channel, so dropping either one leaves every feed + * on the site not well-formed. tests/dist-smoke.test.ts holds them down. + */ +export const feedXmlns = { + dc: 'http://purl.org/dc/elements/1.1/', + atom: 'http://www.w3.org/2005/Atom', +} as const; + +/** + * `` for a feed item. + * + * @astrojs/rss parses item customData as XML and re-serialises it, so a bare + * `&` would survive on its own. `<` would not: it opens an element and the + * name after it is lost. Escaping all three is cheaper than depending on which + * ones the library happens to handle. + */ +export function feedCreator(name: string): string { + const safe = name.replace(/&/g, '&').replace(//g, '>'); + return `${safe}`; +} + +/** `` for the feed served at `path`. */ +export function feedSelfLink(path: string): string { + return ``; +} diff --git a/src/pages/rss.xml.ts b/src/pages/rss.xml.ts index 4efd327..6216695 100644 --- a/src/pages/rss.xml.ts +++ b/src/pages/rss.xml.ts @@ -1,6 +1,7 @@ import rss from '@astrojs/rss'; import { getCollection } from 'astro:content'; import { pagePath, site } from '../lib/site'; +import { feedCreator, feedSelfLink, feedXmlns } from '../lib/feed'; import type { APIContext } from 'astro'; export async function GET(context: APIContext) { @@ -13,14 +14,17 @@ export async function GET(context: APIContext) { description: site.description, site: context.site ?? site.url, trailingSlash: true, + xmlns: feedXmlns, items: posts.map((post) => ({ title: post.data.title, description: post.data.description, pubDate: post.data.publishedAt, link: pagePath(`/blog/${post.id}`), categories: post.data.tags.map((t) => t.label), - author: post.data.author.name || site.name, + // Not : RSS 2.0 defines that element as an email address, and a + // bare name there fails the W3C validator. The name belongs in dc:creator. + customData: feedCreator(post.data.author.name || site.name), })), - customData: 'en', + customData: `en${feedSelfLink('/rss.xml')}`, }); } diff --git a/tests/dist-smoke.test.ts b/tests/dist-smoke.test.ts index 6694fac..e7ff32e 100644 --- a/tests/dist-smoke.test.ts +++ b/tests/dist-smoke.test.ts @@ -3,8 +3,16 @@ import { existsSync, readFileSync, readdirSync } from 'node:fs'; import { parseHTML } from 'linkedom'; import site from '../site.config.json' with { type: 'json' }; import { engines } from '../src/data/engines'; +import { feedCreator } from '../src/lib/feed'; const page = (path: string) => parseHTML(readFileSync(path, 'utf8')).document; +/** Every per-engine feed that was actually built. */ +const engineFeeds = () => + existsSync('dist/blog/engine') + ? readdirSync('dist/blog/engine', { withFileTypes: true }) + .filter((e) => e.isDirectory() && existsSync(`dist/blog/engine/${e.name}/rss.xml`)) + .map((e) => `dist/blog/engine/${e.name}/rss.xml`) + : []; /** Every stylesheet the homepage links, concatenated. Astro splits them per * component, so which chunk the hero lands in is not something to assert on. */ @@ -169,6 +177,39 @@ describe('SEO surface', () => { expect(items).toBe(postDirs.length); expect(rss).toContain(`${site.url}/blog/the-tool-goes-to-the-data/`); expect(rss).toContain('en'); + }); + + it('serves feeds that parse as XML at all, with the author in dc:creator', () => { + // Every feed puts the author's name in and declares where it + // lives, because RSS 2.0 defines as an email address and the W3C + // validator rejects a name there. Both depend on the namespaces declared on + // : drop that one line and every feed on the site stops being + // well-formed XML, which nothing here used to notice. + const feeds = ['dist/rss.xml', ...engineFeeds()]; + expect(feeds.length, 'no feeds were built').toBeGreaterThanOrEqual(1); + + for (const path of feeds) { + const xml = readFileSync(path, 'utf8'); + expect(xml, `${path}: dc namespace`).toContain('xmlns:dc="http://purl.org/dc/elements/1.1/"'); + expect(xml, `${path}: atom namespace`).toContain('xmlns:atom="http://www.w3.org/2005/Atom"'); + expect(xml, `${path}: author must not be a bare name`).not.toContain(''); + expect(xml, `${path}: creator`).toContain(''); + expect(xml, `${path}: self link`).toMatch(//); + // Every prefix used in the body must be one of the declared namespaces. + for (const prefix of new Set([...xml.matchAll(/<([a-z]+):[a-z]+/g)].map((m) => m[1]))) { + expect(xml, `${path}: ${prefix} prefix used but not declared`).toContain(`xmlns:${prefix}=`); + } + } + }); + + it('escapes a creator name that carries XML syntax', () => { + // The site feed's author name contains an ampersand, the case that breaks a + // feed silently: a reader stops at the parse error and shows nothing. + expect(feedCreator('a & b')).toBe('a & b'); + expect(feedCreator('x')).toBe('<i>x</i>'); + expect(readFileSync('dist/rss.xml', 'utf8')).toContain( + 'Cevheri & LibreDB Engineering', + ); const sitemap = readFileSync('dist/sitemap-0.xml', 'utf8'); expect(sitemap).toContain(`${site.url}/`); From 2bca77edd9f1350275dd9acf3d003b28678cb144 Mon Sep 17 00:00:00 2001 From: yusuf-gundogdu Date: Sat, 26 Sep 2026 02:00:01 +0300 Subject: [PATCH 2/2] feat(blog): give each engine archive its own feed The blog groups posts by engine, but it publishes one feed for all of them, so an aggregator that wants a single engine has to filter us from the outside. The only thing an outside service can filter on is the title, and our titles name the behaviour rather than the engine, so it drops posts. Planet for the MySQL Community reads us through such a filter today and misses one. Each engine archive now serves the feed for its own posts at /blog/engine//rss.xml, built from the post set the archive page lists, with the engine as the first category on every item. The archive page links it in the body and declares it in the head ahead of the site feed, which stays advertised so the whole blog is still reachable from there. The grouping and its threshold already existed three times: the archive route, the sibling links on an archive, and the engine index on /blog. The feed would have been a fourth, so all four now read engineArchives() in lib/posts and the threshold is one constant. The fixture behind the archive tests counted every post file on disk while the routes count published ones, so saving a draft turned the suite red with nothing wrong in the site. It filters on status now, which also settles nine older tests that shared it. --- src/layouts/BaseLayout.astro | 13 +++- src/lib/posts.ts | 34 +++++++++ src/pages/blog/engine/[engine].astro | 52 ++++++-------- src/pages/blog/engine/[engine]/rss.xml.ts | 63 +++++++++++++++++ src/pages/blog/index.astro | 12 +--- tests/engine-archives.test.ts | 85 +++++++++++++++++++++++ 6 files changed, 215 insertions(+), 44 deletions(-) create mode 100644 src/pages/blog/engine/[engine]/rss.xml.ts diff --git a/src/layouts/BaseLayout.astro b/src/layouts/BaseLayout.astro index 8fc82b8..a3ac5eb 100644 --- a/src/layouts/BaseLayout.astro +++ b/src/layouts/BaseLayout.astro @@ -15,15 +15,21 @@ import SiteHeader from '../components/layout/SiteHeader.astro'; import SiteFooter from '../components/layout/SiteFooter.astro'; import CookieConsent from '../components/layout/CookieConsent.astro'; import { buildSeo, type SeoInput } from '../lib/seo'; -import { site } from '../lib/site'; +import { canonicalUrl, site } from '../lib/site'; interface Props extends SeoInput { /** JSON-LD graph nodes appended to the page */ schema?: unknown[]; bodyClass?: string; + /** + * The feed this page is a view of, when it is not the site-wide one. An + * engine archive that advertises /rss.xml subscribes the reader to all 104 + * posts, which is the opposite of what they clicked. + */ + feed?: { href: string; title: string }; } -const { schema = [], bodyClass, ...seoInput } = Astro.props; +const { schema = [], bodyClass, feed, ...seoInput } = Astro.props; const seo = buildSeo(seoInput); const jsonLd = { @@ -87,7 +93,8 @@ const jsonLd = { - + {feed && } +