Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 10 additions & 3 deletions src/layouts/BaseLayout.astro
Original file line number Diff line number Diff line change
Expand Up @@ -15,15 +15,21 @@ import SiteHeader from '../components/layout/SiteHeader.astro';
import SiteFooter from '../components/layout/SiteFooter.astro';
import CookieConsent from '../components/layout/CookieConsent.astro';
import { buildSeo, type SeoInput } from '../lib/seo';
import { site } from '../lib/site';
import { canonicalUrl, site } from '../lib/site';

interface Props extends SeoInput {
/** JSON-LD graph nodes appended to the page */
schema?: unknown[];
bodyClass?: string;
/**
* The feed this page is a view of, when it is not the site-wide one. An
* engine archive that advertises /rss.xml subscribes the reader to all 104
* posts, which is the opposite of what they clicked.
*/
feed?: { href: string; title: string };
}

const { schema = [], bodyClass, ...seoInput } = Astro.props;
const { schema = [], bodyClass, feed, ...seoInput } = Astro.props;
const seo = buildSeo(seoInput);

const jsonLd = {
Expand Down Expand Up @@ -87,7 +93,8 @@ const jsonLd = {
<link rel="apple-touch-icon" href="/brand/lb-icon.svg" />
<link rel="manifest" href="/site.webmanifest" />
<meta name="theme-color" content="#0b0d18" />
<link rel="alternate" type="application/rss+xml" title={`${site.name} — blog`} href={`${site.url}/rss.xml`} />
{feed && <link rel="alternate" type="application/rss+xml" title={feed.title} href={canonicalUrl(feed.href)} />}
<link rel="alternate" type="application/rss+xml" title={`${site.name} — blog`} href={canonicalUrl('/rss.xml')} />

<script is:inline>
// Runs before first paint: no theme flash, and `html.js` is the switch the
Expand Down
36 changes: 36 additions & 0 deletions src/lib/feed.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,36 @@
import { canonicalUrl } from './site';

/**
* The two namespaces and the self link every feed on this site declares.
*
* RSS 2.0 defines <author> as an email address, so a bare name there fails the
* W3C feed validator: the live site feed answers "Invalid email address" once
* per item. The author's name is Dublin Core's job. rel="self" is the
* validator's one standing recommendation for a feed that does not say where
* it lives.
*
* Both are declared on the channel, so dropping either one leaves every feed
* on the site not well-formed. tests/dist-smoke.test.ts holds them down.
*/
export const feedXmlns = {
dc: 'http://purl.org/dc/elements/1.1/',
atom: 'http://www.w3.org/2005/Atom',
} as const;

/**
* `<dc:creator>` for a feed item.
*
* @astrojs/rss parses item customData as XML and re-serialises it, so a bare
* `&` would survive on its own. `<` would not: it opens an element and the
* name after it is lost. Escaping all three is cheaper than depending on which
* ones the library happens to handle.
*/
export function feedCreator(name: string): string {
const safe = name.replace(/&/g, '&amp;').replace(/</g, '&lt;').replace(/>/g, '&gt;');
return `<dc:creator>${safe}</dc:creator>`;
}

/** `<atom:link rel="self">` for the feed served at `path`. */
export function feedSelfLink(path: string): string {
return `<atom:link href="${canonicalUrl(path)}" rel="self" type="application/rss+xml"/>`;
}
34 changes: 34 additions & 0 deletions src/lib/posts.ts
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,40 @@ export function engineName(engineId: string): string | undefined {

type Datable = { id: string; data: { publishedAt: Date } };

/**
* The engine archives the blog actually offers: engine id to its posts, newest
* first, for engines with at least two posts.
*
* The threshold and the grouping live here because two routes need the same
* answer — the archive page and its feed. When they each carried their own copy
* of the loop, the only thing keeping a feed from existing without its page was
* that both happened to say `>= 2`. One edit to either and they part silently,
* which is the drift this repo keeps writing rules against.
*
* Two is the floor because one post is the post, not an archive of it.
*/
/** The shape both engine routes hand each other: a list of published posts. */
export type PublishedPosts = Awaited<ReturnType<typeof import('astro:content').getCollection<'posts'>>>;

/** One post is the post, not an archive of it. */
const ARCHIVE_MIN = 2;

export function engineArchives<T extends Datable>(posts: T[]): Map<string, T[]> {
const byEngine = new Map<string, T[]>();
for (const post of posts) {
const id = postEngine(post.id);
if (!id) continue;
byEngine.set(id, [...(byEngine.get(id) ?? []), post]);
}

const byDate = (a: T, b: T) => b.data.publishedAt.getTime() - a.data.publishedAt.getTime();
return new Map(
[...byEngine.entries()]
.filter(([, list]) => list.length >= ARCHIVE_MIN)
.map(([id, list]) => [id, [...list].sort(byDate)]),
);
}

/**
* Posts to offer at the end of `post`: same engine first, newest first, topped
* up with recent posts when that engine has too few to fill the block.
Expand Down
52 changes: 20 additions & 32 deletions src/pages/blog/engine/[engine].astro
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,7 @@ import type { GetStaticPaths } from 'astro';
import BaseLayout from '../../../layouts/BaseLayout.astro';
import PostCard from '../../../components/blog/PostCard.astro';
import { pagePath, site } from '../../../lib/site';
import { engineName, postEngine } from '../../../lib/posts';
import { engineArchives, engineName, type PublishedPosts } from '../../../lib/posts';
import { engines } from '../../../data/engines';

/**
Expand All @@ -23,27 +23,16 @@ import { engines } from '../../../data/engines';
*/
export const getStaticPaths = (async () => {
const { getCollection } = await import('astro:content');
const { postEngine } = await import('../../../lib/posts');

const posts = await getCollection('posts', ({ data }) => data.status === 'published');
const byEngine = new Map<string, typeof posts>();
for (const post of posts) {
const id = postEngine(post.id);
if (!id) continue;
byEngine.set(id, [...(byEngine.get(id) ?? []), post]);
}

return [...byEngine.entries()]
.filter(([, list]) => list.length >= 2)
.map(([engine, list]) => ({
params: { engine },
props: {
posts: list.sort((a, b) => b.data.publishedAt.getTime() - a.data.publishedAt.getTime()),
},
}));
return [...engineArchives(posts).entries()].map(([engine, list]) => ({
params: { engine },
props: { posts: list },
}));
}) satisfies GetStaticPaths;

type Props = { posts: Awaited<ReturnType<typeof import('astro:content').getCollection<'posts'>>> };
type Props = { posts: PublishedPosts };
const { engine } = Astro.params;
const { posts } = Astro.props;

Expand All @@ -53,20 +42,11 @@ const engineData = engines.find((e) => e.id === engine);
// Sibling archives, so a reader comparing two engines can cross over without
// going back to the flat list. Built from what actually shipped, not from the
// full engine list — an archive that does not exist must not be linked.
const siblings = (await import('astro:content'))
.getCollection('posts', ({ data }) => data.status === 'published')
.then((all) => {
const counts = new Map<string, number>();
for (const p of all) {
const id = postEngine(p.id);
if (id) counts.set(id, (counts.get(id) ?? 0) + 1);
}
return [...counts.entries()]
.filter(([id, n]) => n >= 2 && id !== engine)
.map(([id]) => ({ id, name: engineName(id) ?? id }))
.sort((a, b) => a.name.localeCompare(b.name));
});
const others = await siblings;
const all = await (await import('astro:content')).getCollection('posts', ({ data }) => data.status === 'published');
const others = [...engineArchives(all).keys()]
.filter((id) => id !== engine)
.map((id) => ({ id, name: engineName(id) ?? id }))
.sort((a, b) => a.name.localeCompare(b.name));

const description = `Every LibreDB Studio write-up about ${name}: what the provider does, where it stops, and why. ${posts.length} posts.`;

Expand Down Expand Up @@ -98,7 +78,13 @@ const schema = [
];
---

<BaseLayout path={`/blog/engine/${engine}`} title={`${name} posts`} description={description} schema={schema}>
<BaseLayout
path={`/blog/engine/${engine}`}
title={`${name} posts`}
description={description}
schema={schema}
feed={{ href: `/blog/engine/${engine}/rss.xml`, title: `${site.name} — ${name} posts` }}
>
<section class="earc" aria-labelledby="earc-title">
<div class="u-wash earc__wash" aria-hidden="true"></div>
<div class="earc__inner u-container">
Expand Down Expand Up @@ -126,6 +112,8 @@ const schema = [
<p class="earc__links">
<a href={pagePath('/databases')}>{name} support in Studio →</a>
<span aria-hidden="true">·</span>
<a href={pagePath(`/blog/engine/${engine}/rss.xml`)}>{name} RSS feed →</a>
<span aria-hidden="true">·</span>
<a href={pagePath('/blog')}>All posts</a>
</p>
</header>
Expand Down
63 changes: 63 additions & 0 deletions src/pages/blog/engine/[engine]/rss.xml.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,63 @@
import rss from '@astrojs/rss';
import type { APIContext, GetStaticPaths } from 'astro';
import { canonicalUrl, pagePath, site } from '../../../../lib/site';
import { feedCreator, feedSelfLink, feedXmlns } from '../../../../lib/feed';
import { engineName, type PublishedPosts } from '../../../../lib/posts';

/**
* One feed per engine archive, alongside the archive page itself.
*
* The site-wide feed carries every post, so an aggregator that wants one
* engine has to filter it downstream. Planet for the MySQL Community asks for
* the opposite in the notes at the top of its own planet.ini
* (github.com/oursqlcommunity-org/planet): "If your blog contains a mix MySQL
* and non-MySQL content, please consider submitting a category or tag feed
* instead of a default feed." Without one, our entry there runs the site feed
* through siftrss on a title regex, which misses posts whose title does not
* name the engine. mysql-wire-compatible-engines-one-provider is titled "Nine
* servers, one connection type, different answers" and is exactly that case.
*
* The grouping is not new: /blog/engine/<id> already lists these posts. Both
* routes read it from engineArchives(), so the threshold lives in one place.
*/
export const getStaticPaths = (async () => {
const { getCollection } = await import('astro:content');
const { engineArchives } = await import('../../../../lib/posts');

const posts = await getCollection('posts', ({ data }) => data.status === 'published');

return [...engineArchives(posts).entries()].map(([engine, list]) => ({
params: { engine },
props: { posts: list },
}));
}) satisfies GetStaticPaths;

type Props = { posts: PublishedPosts };

export function GET(context: APIContext) {
const engine = context.params.engine!;
const { posts } = context.props as Props;
const name = engineName(engine) ?? engine;

return rss({
title: `${site.name} — ${name} posts`,
description: `Posts about ${name} from the ${site.name} blog.`,
// The archive, not the home page: a reader who follows the feed back should
// land on the list it was cut from.
site: canonicalUrl(`/blog/engine/${engine}`),
trailingSlash: true,
xmlns: feedXmlns,
items: posts.map((post) => ({
title: post.data.title,
description: post.data.description,
pubDate: post.data.publishedAt,
link: pagePath(`/blog/${post.id}`),
// The engine first, so an aggregator filtering on category has the name
// it asked for. Deduped because a post is free to carry the engine as a
// tag too, and two identical <category> elements help nobody.
categories: [name, ...post.data.tags.map((t) => t.label)],
customData: feedCreator(post.data.author.name || site.name),
})),
customData: `<language>en</language>${feedSelfLink(`/blog/engine/${engine}/rss.xml`)}`,
});
}
12 changes: 3 additions & 9 deletions src/pages/blog/index.astro
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,7 @@ import { getCollection } from 'astro:content';
import BaseLayout from '../../layouts/BaseLayout.astro';
import PostCard from '../../components/blog/PostCard.astro';
import { pagePath, site } from '../../lib/site';
import { engineName, postEngine } from '../../lib/posts';
import { engineArchives, engineName } from '../../lib/posts';

const posts = (await getCollection('posts', ({ data }) => data.status === 'published')).sort(
(a, b) => b.data.publishedAt.getTime() - a.data.publishedAt.getTime(),
Expand All @@ -12,14 +12,8 @@ const posts = (await getCollection('posts', ({ data }) => data.status === 'publi
// A way into 104 posts that is not "scroll". These are plain links to the
// archives rather than a client-side filter: the grouping is then crawlable,
// linkable and survives with JavaScript off, and the work is already done.
const byEngine = new Map<string, number>();
for (const post of posts) {
const id = postEngine(post.id);
if (id) byEngine.set(id, (byEngine.get(id) ?? 0) + 1);
}
const archives = [...byEngine.entries()]
.filter(([, n]) => n >= 2)
.map(([id, n]) => ({ id, n, name: engineName(id) ?? id }))
const archives = [...engineArchives(posts).entries()]
.map(([id, list]) => ({ id, n: list.length, name: engineName(id) ?? id }))
.sort((a, b) => b.n - a.n || a.name.localeCompare(b.name));
---

Expand Down
8 changes: 6 additions & 2 deletions src/pages/rss.xml.ts
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
import rss from '@astrojs/rss';
import { getCollection } from 'astro:content';
import { pagePath, site } from '../lib/site';
import { feedCreator, feedSelfLink, feedXmlns } from '../lib/feed';
import type { APIContext } from 'astro';

export async function GET(context: APIContext) {
Expand All @@ -13,14 +14,17 @@ export async function GET(context: APIContext) {
description: site.description,
site: context.site ?? site.url,
trailingSlash: true,
xmlns: feedXmlns,
items: posts.map((post) => ({
title: post.data.title,
description: post.data.description,
pubDate: post.data.publishedAt,
link: pagePath(`/blog/${post.id}`),
categories: post.data.tags.map((t) => t.label),
author: post.data.author.name || site.name,
// Not <author>: RSS 2.0 defines that element as an email address, and a
// bare name there fails the W3C validator. The name belongs in dc:creator.
customData: feedCreator(post.data.author.name || site.name),
})),
customData: '<language>en</language>',
customData: `<language>en</language>${feedSelfLink('/rss.xml')}`,
});
}
41 changes: 41 additions & 0 deletions tests/dist-smoke.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -3,8 +3,16 @@ import { existsSync, readFileSync, readdirSync } from 'node:fs';
import { parseHTML } from 'linkedom';
import site from '../site.config.json' with { type: 'json' };
import { engines } from '../src/data/engines';
import { feedCreator } from '../src/lib/feed';

const page = (path: string) => parseHTML(readFileSync(path, 'utf8')).document;
/** Every per-engine feed that was actually built. */
const engineFeeds = () =>
existsSync('dist/blog/engine')
? readdirSync('dist/blog/engine', { withFileTypes: true })
.filter((e) => e.isDirectory() && existsSync(`dist/blog/engine/${e.name}/rss.xml`))
.map((e) => `dist/blog/engine/${e.name}/rss.xml`)
: [];

/** Every stylesheet the homepage links, concatenated. Astro splits them per
* component, so which chunk the hero lands in is not something to assert on. */
Expand Down Expand Up @@ -169,6 +177,39 @@ describe('SEO surface', () => {
expect(items).toBe(postDirs.length);
expect(rss).toContain(`${site.url}/blog/the-tool-goes-to-the-data/`);
expect(rss).toContain('<language>en</language>');
});

it('serves feeds that parse as XML at all, with the author in dc:creator', () => {
// Every feed puts the author's name in <dc:creator> and declares where it
// lives, because RSS 2.0 defines <author> as an email address and the W3C
// validator rejects a name there. Both depend on the namespaces declared on
// <rss>: drop that one line and every feed on the site stops being
// well-formed XML, which nothing here used to notice.
const feeds = ['dist/rss.xml', ...engineFeeds()];
expect(feeds.length, 'no feeds were built').toBeGreaterThanOrEqual(1);

for (const path of feeds) {
const xml = readFileSync(path, 'utf8');
expect(xml, `${path}: dc namespace`).toContain('xmlns:dc="http://purl.org/dc/elements/1.1/"');
expect(xml, `${path}: atom namespace`).toContain('xmlns:atom="http://www.w3.org/2005/Atom"');
expect(xml, `${path}: author must not be a bare name`).not.toContain('<author>');
expect(xml, `${path}: creator`).toContain('<dc:creator>');
expect(xml, `${path}: self link`).toMatch(/<atom:link href="[^"]+" rel="self" type="application\/rss\+xml"\/>/);
// Every prefix used in the body must be one of the declared namespaces.
for (const prefix of new Set([...xml.matchAll(/<([a-z]+):[a-z]+/g)].map((m) => m[1]))) {
expect(xml, `${path}: ${prefix} prefix used but not declared`).toContain(`xmlns:${prefix}=`);
}
}
});

it('escapes a creator name that carries XML syntax', () => {
// The site feed's author name contains an ampersand, the case that breaks a
// feed silently: a reader stops at the parse error and shows nothing.
expect(feedCreator('a & b')).toBe('<dc:creator>a &amp; b</dc:creator>');
expect(feedCreator('<i>x</i>')).toBe('<dc:creator>&lt;i&gt;x&lt;/i&gt;</dc:creator>');
expect(readFileSync('dist/rss.xml', 'utf8')).toContain(
'<dc:creator>Cevheri &amp; LibreDB Engineering</dc:creator>',
);

const sitemap = readFileSync('dist/sitemap-0.xml', 'utf8');
expect(sitemap).toContain(`${site.url}/`);
Expand Down
Loading
Loading