From 3a39bb7697d756bab55bdd7e98b282bf9c196ecb Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Thu, 1 Oct 2026 10:57:00 -0700 Subject: [PATCH 1/2] chore(content): add check:library-content audit for blog, library, and customer posts Validates frontmatter against the strict ContentFrontmatterSchema, slug/folder parity, local ogImage existence, MDX compilation with remark-gfm, FAQ placement, and internal post links (including retired and moved slugs). Moves the library slug redirect maps into lib/library/retired-slugs.ts so next.config.ts and the audit share one source. Fixes two FAQ answers that rendered Markdown links as literal text. --- .../library/ai-agent-vs-chatbot/index.mdx | 4 +- apps/sim/lib/library/retired-slugs.ts | 29 ++ apps/sim/next.config.ts | 29 +- bun.lock | 1 + package.json | 2 + scripts/check-library-content.test.ts | 222 +++++++++ scripts/check-library-content.ts | 441 ++++++++++++++++++ 7 files changed, 698 insertions(+), 30 deletions(-) create mode 100644 apps/sim/lib/library/retired-slugs.ts create mode 100644 scripts/check-library-content.test.ts create mode 100644 scripts/check-library-content.ts diff --git a/apps/sim/content/library/ai-agent-vs-chatbot/index.mdx b/apps/sim/content/library/ai-agent-vs-chatbot/index.mdx index 208c5eccee1..2ad7b36c603 100644 --- a/apps/sim/content/library/ai-agent-vs-chatbot/index.mdx +++ b/apps/sim/content/library/ai-agent-vs-chatbot/index.mdx @@ -12,9 +12,9 @@ ogImage: /library/ai-agent-vs-chatbot/cover.jpg draft: false faq: - q: "Do AI agents always produce more accurate answers than chatbots?" - a: "Answer accuracy is the degree to which a system's response is correct and supported by the available evidence. In [Sim](https://www.sim.ai), you can inspect each workflow step and control which instructions, data, and tools the model uses. This visibility helps you find the source of an error and improve the workflow without assuming that an agent is inherently more accurate than a chatbot." + a: "Answer accuracy is the degree to which a system's response is correct and supported by the available evidence. In Sim, you can inspect each workflow step and control which instructions, data, and tools the model uses. This visibility helps you find the source of an error and improve the workflow without assuming that an agent is inherently more accurate than a chatbot." - q: "Can a chatbot and an AI agent work together?" - a: "A hybrid application uses a chatbot for conversation and an AI agent for actions that require tools or multiple steps. Sim connects our [Chat interface](https://docs.sim.ai/execution/chat) to agent workflows you build in a visual workspace with connected integrations. You can give users one conversational interface while the agent handles work across connected applications." + a: "A hybrid application uses a chatbot for conversation and an AI agent for actions that require tools or multiple steps. Sim connects our Chat interface to agent workflows you build in a visual workspace with connected integrations. You can give users one conversational interface while the agent handles work across connected applications." - q: "Do I need to know how to code to build an AI agent?" a: "A visual agent builder represents workflow logic as configurable blocks that connect models to tools, so you do not have to program every step. Sim provides a visual workspace for creating and testing these workflows. You can add code when needed without building the orchestration layer from scratch." - q: "Can AI agents run with local models?" diff --git a/apps/sim/lib/library/retired-slugs.ts b/apps/sim/lib/library/retired-slugs.ts new file mode 100644 index 00000000000..cfddc9b30d8 --- /dev/null +++ b/apps/sim/lib/library/retired-slugs.ts @@ -0,0 +1,29 @@ +/** + * AEO/GEO-style posts (listicles, comparisons, how-tos) split out of `/blog` + * into the dedicated `/library` section so `/blog` stays editorial-only. + * `next.config.ts` redirects `/blog/` to `/library/` for each. + */ +export const LIBRARY_MOVED_BLOG_SLUGS = [ + 'best-zapier-alternatives', + 'ai-agents-vs-rpa', + 'ai-agent-vs-chatbot', + 'openai-vs-n8n-vs-sim', + 'ai-agent-ideas', + 'how-to-create-an-ai-agent', +] as const + +/** + * Library articles retired by merging into a stronger article on the same + * search intent, keyed by retired slug. `next.config.ts` redirects each + * retired URL to the surviving article, and `check:library-content` rejects + * new links to a retired slug. + */ +export const LIBRARY_MERGED_SLUGS: Readonly> = { + 'automation-anywhere-alternative': 'ai-agents-vs-rpa', + 'ai-native-vs-traditional-workflow-automation': + 'ai-native-workflow-automation-vs-traditional-automation', + 'best-ai-workflow-builders-small-teams-2026': 'best-ai-workflow-builders', + 'best-ai-agent-builder-2026': 'best-ai-agent-platforms-2026', + 'best-ai-agent-builders-slack-crm-automation-2026': 'best-ai-agents-for-slack', + 'best-open-source-ai-agent-frameworks': 'open-source-ai-agent-platforms', +} diff --git a/apps/sim/next.config.ts b/apps/sim/next.config.ts index dd84738690f..88c2c65be93 100644 --- a/apps/sim/next.config.ts +++ b/apps/sim/next.config.ts @@ -9,34 +9,7 @@ import { getWorkflowExecutionCSPPolicy, } from './lib/core/security/csp' import { LANDING_ROUTES } from './lib/landing/routes' - -/** - * AEO/GEO-style posts (listicles, comparisons, how-tos) split out of `/blog` - * into the dedicated `/library` section so `/blog` stays editorial-only. - */ -const LIBRARY_MOVED_BLOG_SLUGS = [ - 'best-zapier-alternatives', - 'ai-agents-vs-rpa', - 'ai-agent-vs-chatbot', - 'openai-vs-n8n-vs-sim', - 'ai-agent-ideas', - 'how-to-create-an-ai-agent', -] as const - -/** - * Library articles retired by merging into a stronger article on the same - * search intent, keyed by retired slug. Keeps indexed URLs and inbound links - * pointing at the surviving article. - */ -const LIBRARY_MERGED_SLUGS: Record = { - 'automation-anywhere-alternative': 'ai-agents-vs-rpa', - 'ai-native-vs-traditional-workflow-automation': - 'ai-native-workflow-automation-vs-traditional-automation', - 'best-ai-workflow-builders-small-teams-2026': 'best-ai-workflow-builders', - 'best-ai-agent-builder-2026': 'best-ai-agent-platforms-2026', - 'best-ai-agent-builders-slack-crm-automation-2026': 'best-ai-agents-for-slack', - 'best-open-source-ai-agent-frameworks': 'open-source-ai-agent-platforms', -} +import { LIBRARY_MERGED_SLUGS, LIBRARY_MOVED_BLOG_SLUGS } from './lib/library/retired-slugs' const nextConfig: NextConfig = { devIndicators: false, diff --git a/bun.lock b/bun.lock index eb4bb7cd7c0..e2719eca5a2 100644 --- a/bun.lock +++ b/bun.lock @@ -7,6 +7,7 @@ "devDependencies": { "@babel/parser": "7.29.2", "@biomejs/biome": "2.0.6", + "@mdx-js/mdx": "3.1.1", "@octokit/rest": "^21.0.0", "@sim/utils": "workspace:*", "@types/opentype.js": "1.3.10", diff --git a/package.json b/package.json index 18fc1fc8916..aeaefe9f360 100644 --- a/package.json +++ b/package.json @@ -71,6 +71,7 @@ "check:native-typecheck": "bun run scripts/check-native-typecheck.ts", "check:source-text": "bun run scripts/check-source-text.ts", "check:site-urls": "bun run scripts/check-site-urls.ts", + "check:library-content": "bun run scripts/check-library-content.ts", "check:spec-example-ids": "bun run scripts/check-spec-example-ids.ts", "check:script-tests": "bun run scripts/check-script-test-coverage.ts", "check:test-patterns": "bun run scripts/check-test-patterns.ts", @@ -162,6 +163,7 @@ "devDependencies": { "@babel/parser": "7.29.2", "@biomejs/biome": "2.0.6", + "@mdx-js/mdx": "3.1.1", "@octokit/rest": "^21.0.0", "@sim/utils": "workspace:*", "@types/opentype.js": "1.3.10", diff --git a/scripts/check-library-content.test.ts b/scripts/check-library-content.test.ts new file mode 100644 index 00000000000..c6204719ecc --- /dev/null +++ b/scripts/check-library-content.test.ts @@ -0,0 +1,222 @@ +import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import path from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { + type ContentCheckConfig, + checkContent, + checkMergedSlugTargets, + indexPosts, + resolvePostArg, +} from './check-library-content' + +let root: string +let config: ContentCheckConfig + +const CLEAN_FRONTMATTER = { + title: 'A clean library post', + description: 'A description that is comfortably over twenty characters.', + date: '2026-09-01', + authors: '[sim]', +} + +function writePost( + section: string, + slug: string, + body: string, + frontmatter: Record = {} +) { + const fields = { + slug, + ...CLEAN_FRONTMATTER, + ogImage: `/${section}/${slug}/cover.jpg`, + ...frontmatter, + } + const yaml = Object.entries(fields) + .map(([key, value]) => (value.startsWith('\n') ? `${key}:${value}` : `${key}: ${value}`)) + .join('\n') + const dir = path.join(config.contentDir, section, slug) + mkdirSync(dir, { recursive: true }) + writeFileSync(path.join(dir, 'index.mdx'), `---\n${yaml}\n---\n\n${body}\n`) + const imageDir = path.join(config.publicDir, section, slug) + mkdirSync(imageDir, { recursive: true }) + writeFileSync(path.join(imageDir, 'cover.jpg'), '') +} + +async function findingsFor(section: string, slug: string) { + const { findings } = await checkContent(config, [{ section: section as 'library', slug }]) + return findings.map(({ line, rule, message }) => ({ line, rule, message })) +} + +beforeEach(() => { + root = mkdtempSync(path.join(tmpdir(), 'library-content-')) + config = { + contentDir: path.join(root, 'content'), + publicDir: path.join(root, 'public'), + reservedSegments: { blog: new Set(['tags']), library: new Set(['tags']), customers: new Set() }, + mergedSlugs: { 'old-guide': 'kept-guide' }, + movedBlogSlugs: ['moved-post'], + } + mkdirSync(path.join(config.contentDir, 'authors'), { recursive: true }) + writeFileSync(path.join(config.contentDir, 'authors', 'sim.json'), '{"id":"sim","name":"Sim"}') + writePost('library', 'kept-guide', 'The surviving guide.') +}) + +afterEach(() => { + rmSync(root, { recursive: true, force: true }) +}) + +describe('check-library-content', () => { + it('passes a clean post with valid internal links, assets, reserved routes, and code samples', async () => { + writePost( + 'library', + 'clean', + [ + '## Overview', + 'See [the guide](/library/kept-guide) and https://www.sim.ai/library/kept-guide#intro.', + '![diagram](/library/clean/diagram.png) and [tags](/library/tags) are not posts.', + 'An apex https://sim.ai/library/missing link belongs to check:site-urls.', + '| a | b |', + '| - | - |', + '| 1 | 2 |', + '```md', + '## FAQ', + '[old](/library/old-guide)', + '```', + ].join('\n') + ) + expect(await findingsFor('library', 'clean')).toEqual([]) + }) + + it('rejects an unknown frontmatter key such as canonical, at its line', async () => { + writePost('library', 'post', 'Body.', { canonical: 'https://www.sim.ai/library/post' }) + expect(await findingsFor('library', 'post')).toEqual([ + { line: 8, rule: 'frontmatter', message: 'Unknown frontmatter key "canonical".' }, + ]) + }) + + it('rejects frontmatter that fails the schema', async () => { + writePost('library', 'post', 'Body.', { title: 'Hi' }) + const findings = await findingsFor('library', 'post') + expect(findings).toHaveLength(1) + expect(findings[0]).toMatchObject({ line: 3, rule: 'frontmatter' }) + }) + + it('rejects invalid YAML', async () => { + writePost('library', 'post', 'Body.', { title: 'Broken: [unclosed' }) + expect((await findingsFor('library', 'post'))[0]).toMatchObject({ rule: 'frontmatter' }) + }) + + it('rejects an author with no profile', async () => { + writePost('library', 'post', 'Body.', { authors: '[ghost]' }) + expect(await findingsFor('library', 'post')).toEqual([ + { + line: 6, + rule: 'frontmatter', + message: 'Author "ghost" has no profile in apps/sim/content/authors.', + }, + ]) + }) + + it('rejects a slug that differs from the folder name', async () => { + writePost('library', 'post', 'Body.', { slug: 'other' }) + expect(await findingsFor('library', 'post')).toEqual([ + { line: 2, rule: 'slug', message: 'slug "other" does not match the folder name "post".' }, + ]) + }) + + it('rejects a missing or remote ogImage', async () => { + writePost('library', 'missing', 'Body.', { ogImage: '/library/missing/nope.jpg' }) + writePost('library', 'remote', 'Body.', { ogImage: 'https://example.com/a.jpg' }) + expect(await findingsFor('library', 'missing')).toEqual([ + { + line: 7, + rule: 'og-image', + message: 'ogImage "/library/missing/nope.jpg" does not exist under apps/sim/public.', + }, + ]) + expect((await findingsFor('library', 'remote'))[0]).toMatchObject({ rule: 'og-image' }) + }) + + it('reports an MDX compile error at its file line', async () => { + writePost('library', 'post', 'Intro.\n\nA tag
{ + writePost('library', 'post', 'Intro.\n\n## FAQs\n\nQ and A.') + expect(await findingsFor('library', 'post')).toEqual([ + { line: 12, rule: 'faq', message: 'Body has a "## FAQs" heading.' }, + ]) + }) + + it('rejects Markdown link syntax inside an FAQ answer, at the answer line', async () => { + writePost('library', 'post', 'Body.', { + faq: '\n - q: "Plain question?"\n a: "Plain answer."\n - q: "Linked?"\n a: "See [Sim](https://www.sim.ai)."', + }) + expect(await findingsFor('library', 'post')).toEqual([ + { + line: 12, + rule: 'faq', + message: 'faq[1].a contains Markdown link syntax, which renders as literal text.', + }, + ]) + }) + + it('rejects links to missing, retired, and moved posts, naming the replacement', async () => { + writePost( + 'library', + 'post', + [ + '[a](/library/missing)', + '[b](https://www.sim.ai/library/old-guide)', + 'c', + '[d](/blog/kept-guide)', + ].join('\n') + ) + expect(await findingsFor('library', 'post')).toEqual([ + { + line: 10, + rule: 'internal-link', + message: '/library/missing does not exist (no apps/sim/content/library/missing/index.mdx).', + }, + { + line: 11, + rule: 'internal-link', + message: '/library/old-guide is retired; it was merged into /library/kept-guide.', + }, + { + line: 12, + rule: 'internal-link', + message: '/blog/moved-post moved to /library/moved-post.', + }, + { + line: 13, + rule: 'internal-link', + message: '/blog/kept-guide does not exist (no apps/sim/content/blog/kept-guide/index.mdx).', + }, + ]) + }) + + it('rejects a retired slug whose replacement no longer exists', () => { + config.mergedSlugs = { 'old-guide': 'gone-guide' } + const findings = checkMergedSlugTargets(config, indexPosts(config.contentDir), 'map.ts') + expect(findings.map((finding) => finding.message)).toEqual([ + 'Retired slug "old-guide" redirects to /library/gone-guide, which does not exist.', + ]) + }) + + it('resolves a post argument by section/slug or bare slug, and rejects unknown ones', () => { + const posts = indexPosts(config.contentDir) + expect(resolvePostArg('library/kept-guide', posts)).toEqual([ + { section: 'library', slug: 'kept-guide' }, + ]) + expect(resolvePostArg('kept-guide', posts)).toEqual([ + { section: 'library', slug: 'kept-guide' }, + ]) + expect(typeof resolvePostArg('blog/kept-guide', posts)).toBe('string') + expect(typeof resolvePostArg('nope', posts)).toBe('string') + }) +}) diff --git a/scripts/check-library-content.ts b/scripts/check-library-content.ts new file mode 100644 index 00000000000..4da0e633e9d --- /dev/null +++ b/scripts/check-library-content.ts @@ -0,0 +1,441 @@ +#!/usr/bin/env bun +/** + * Validates every content post under `apps/sim/content/{blog,library,customers}//index.mdx`. + * + * Content is mostly written by agents and merged without a build, so a bad post only fails at + * `next build` (an invalid frontmatter key, an MDX syntax error) or never fails at all (a link to a + * retired or misspelled slug 301s or 404s in production). Each rule here is exact, so nothing + * needs an override: + * + * - `frontmatter`: gray-matter parses it and it passes the strict `ContentFrontmatterSchema`, so + * an unknown key such as `canonical` fails; every author id has a JSON file in `content/authors`. + * - `slug`: the `slug` field equals the post's folder name. + * - `og-image`: `ogImage` is a local path to a file under `apps/sim/public`. + * - `mdx`: the body compiles with the MDX compiler and `remark-gfm`, as the registry compiles it. + * - `faq`: the body has no FAQ heading (the FAQ lives in frontmatter, which renders it and emits + * its JSON-LD), and no FAQ question or answer contains Markdown link syntax (it renders as text). + * - `internal-link`: every `https://www.sim.ai/
/` link, and every relative + * `/
/` link target, names an existing post folder rather than a retired or moved + * slug. Apex `https://sim.ai` links belong to `check:site-urls`. + * + * Run one post with `--slug
/` or `--slug `. + */ +import { existsSync, readdirSync, readFileSync, statSync } from 'node:fs' +import path from 'node:path' +import { compile } from '@mdx-js/mdx' +import matter from 'gray-matter' +import remarkGfm from 'remark-gfm' +import { ContentFrontmatterSchema } from '../apps/sim/lib/content/schema' +import { + LIBRARY_MERGED_SLUGS, + LIBRARY_MOVED_BLOG_SLUGS, +} from '../apps/sim/lib/library/retired-slugs' + +export const SECTIONS = ['blog', 'library', 'customers'] as const +export type Section = (typeof SECTIONS)[number] + +export type Rule = 'frontmatter' | 'slug' | 'og-image' | 'mdx' | 'faq' | 'internal-link' + +export interface Finding { + file: string + line: number + rule: Rule + message: string + hint: string +} + +export interface ContentCheckConfig { + /** Holds one folder per section (`/
//index.mdx`) plus `authors/`. */ + contentDir: string + /** The app's `public/` directory, which local `ogImage` paths resolve against. */ + publicDir: string + /** Static route segments beside `[slug]` (e.g. `tags`, `rss.xml`) that are not posts. */ + reservedSegments: Readonly>> + /** Retired library slug to the slug that replaced it. */ + mergedSlugs: Readonly> + /** Blog slugs that now live under `/library`. */ + movedBlogSlugs: readonly string[] +} + +export interface PostRef { + section: Section + slug: string +} + +/** + * An absolute `www.sim.ai` URL anywhere, or a relative path in a Markdown link target or an + * `href` attribute. Group 1 is the section, group 2 the first segment, group 3 anything after it. + */ +const INTERNAL_LINK = + /(?:https?:\/\/www\.sim\.ai|(?<=\]\(\s*|href=\{?["'`]))\/(library|blog|customers)\/([^\s)"'`#?/<>\]]+)(\/[^\s)"'`#?<>\]]*)?/g +const MARKDOWN_LINK = /\[[^\]\n]*\]\([^)\n]*\)/ +const FAQ_HEADING = /^#{1,6}\s+FAQs?\s*:?\s*$/i +const CODE_FENCE = /^\s*(```|~~~)/ + +function isSection(value: string): value is Section { + return (SECTIONS as readonly string[]).includes(value) +} + +/** Post folders per section: every directory that holds an `index.mdx`. */ +export function indexPosts(contentDir: string): Record> { + const index = {} as Record> + for (const section of SECTIONS) { + const sectionDir = path.join(contentDir, section) + index[section] = new Set( + existsSync(sectionDir) + ? readdirSync(sectionDir, { withFileTypes: true }) + .filter( + (entry) => + entry.isDirectory() && existsSync(path.join(sectionDir, entry.name, 'index.mdx')) + ) + .map((entry) => entry.name) + : [] + ) + } + return index +} + +/** Static route segments beside each section's `[slug]` route, read from the app router tree. */ +export function readReservedSegments(sectionAppDir: (section: Section) => string) { + const reserved = {} as Record> + for (const section of SECTIONS) { + const dir = sectionAppDir(section) + reserved[section] = new Set( + existsSync(dir) + ? readdirSync(dir, { withFileTypes: true }) + .filter((entry) => entry.isDirectory() && !/^[[(_]/.test(entry.name)) + .map((entry) => entry.name) + : [] + ) + } + return reserved +} + +/** Parses `
/` or a bare ``, resolving the latter against the post index. */ +export function resolvePostArg( + arg: string, + posts: Record> +): PostRef[] | string { + const [first, second] = arg.replace(/\/+$/, '').split('/') + if (second !== undefined) { + if (!isSection(first)) return `Unknown section "${first}"; use one of ${SECTIONS.join(', ')}.` + return posts[first].has(second) ? [{ section: first, slug: second }] : `No post at ${arg}.` + } + const matches = SECTIONS.filter((section) => posts[section].has(first)).map((section) => ({ + section, + slug: first, + })) + if (matches.length === 0) return `No post named "${first}" in ${SECTIONS.join(', ')}.` + return matches +} + +function lineOfKey(frontmatterLines: string[], key: string): number { + const index = frontmatterLines.findIndex((line) => line.startsWith(`${key}:`)) + return index === -1 ? 1 : index + 2 +} + +/** Validates one post, returning every finding (empty when the post is clean). */ +export async function checkPost( + config: ContentCheckConfig, + posts: Record>, + authorIds: ReadonlySet, + { section, slug }: PostRef +): Promise { + const file = path.join(config.contentDir, section, slug, 'index.mdx') + const raw = readFileSync(file, 'utf-8') + const findings: Finding[] = [] + const report = (line: number, rule: Rule, message: string, hint: string) => + findings.push({ file, line, rule, message, hint }) + + let parsed: matter.GrayMatterFile + try { + // A fresh options object bypasses gray-matter's content-keyed cache. + parsed = matter(raw, {}) + } catch (error) { + const mark = (error as { mark?: { line?: number } }).mark + report( + typeof mark?.line === 'number' ? mark.line + 2 : 1, + 'frontmatter', + `Frontmatter is not valid YAML: ${(error as Error).message.split('\n')[0]}`, + 'Fix the YAML between the --- fences (quote values that contain a colon).' + ) + return findings + } + + const body = parsed.content + const bodyOffset = raw.endsWith(body) + ? raw.slice(0, raw.length - body.length).split('\n').length - 1 + : 0 + const frontmatterLines = raw.split('\n').slice(1, Math.max(bodyOffset - 1, 1)) + + const result = ContentFrontmatterSchema.safeParse(parsed.data) + if (!result.success) { + for (const issue of result.error.issues) { + if (issue.code === 'unrecognized_keys') { + for (const key of issue.keys) { + report( + lineOfKey(frontmatterLines, key), + 'frontmatter', + `Unknown frontmatter key "${key}".`, + key === 'canonical' + ? 'Remove it: the canonical URL is derived from the section and slug.' + : `Remove it, or add it to ContentFrontmatterSchema in apps/sim/lib/content/schema.ts.` + ) + } + continue + } + const key = String(issue.path[0] ?? '') + report( + key ? lineOfKey(frontmatterLines, key) : 1, + 'frontmatter', + `Frontmatter ${issue.path.join('.') || '(root)'}: ${issue.message}`, + 'Match ContentFrontmatterSchema in apps/sim/lib/content/schema.ts.' + ) + } + } + + const data = parsed.data as Record + + if (Array.isArray(data.authors)) { + for (const author of data.authors) { + if (typeof author === 'string' && !authorIds.has(author)) { + report( + lineOfKey(frontmatterLines, 'authors'), + 'frontmatter', + `Author "${author}" has no profile in apps/sim/content/authors.`, + `Use one of: ${[...authorIds].sort().join(', ')}.` + ) + } + } + } + + if (typeof data.slug === 'string' && data.slug !== slug) { + report( + lineOfKey(frontmatterLines, 'slug'), + 'slug', + `slug "${data.slug}" does not match the folder name "${slug}".`, + `Set slug: ${slug}, or rename the folder (and its public/ assets) to match.` + ) + } + + if (typeof data.ogImage === 'string') { + const line = lineOfKey(frontmatterLines, 'ogImage') + if (!data.ogImage.startsWith('/')) { + report( + line, + 'og-image', + `ogImage "${data.ogImage}" is not a local path.`, + `Point it at a file under apps/sim/public, e.g. /${section}/${slug}/cover.jpg.` + ) + } else { + const target = path.join(config.publicDir, data.ogImage) + if (!existsSync(target) || !statSync(target).isFile()) { + report( + line, + 'og-image', + `ogImage "${data.ogImage}" does not exist under apps/sim/public.`, + `Add apps/sim/public${data.ogImage}, or fix the path.` + ) + } + } + } + + if (Array.isArray(data.faq)) { + const questionLines: number[] = [] + const answerLines: number[] = [] + frontmatterLines.forEach((line, index) => { + if (/^\s*-?\s*q:/.test(line)) questionLines.push(index + 2) + if (/^\s*-?\s*a:/.test(line)) answerLines.push(index + 2) + }) + data.faq.forEach((entry: unknown, index: number) => { + if (!entry || typeof entry !== 'object') return + const { q, a } = entry as { q?: unknown; a?: unknown } + for (const [field, value, lines] of [ + ['q', q, questionLines], + ['a', a, answerLines], + ] as const) { + if (typeof value === 'string' && MARKDOWN_LINK.test(value)) { + report( + lines[index] ?? lineOfKey(frontmatterLines, 'faq'), + 'faq', + `faq[${index}].${field} contains Markdown link syntax, which renders as literal text.`, + 'Write the FAQ in plain text; put the link in the article body instead.' + ) + } + } + }) + } + + const bodyLines = body.split('\n') + let fence: string | null = null + bodyLines.forEach((text, index) => { + const fenceMatch = CODE_FENCE.exec(text) + if (fenceMatch) { + if (fence === null) fence = fenceMatch[1] + else if (fenceMatch[1] === fence) fence = null + return + } + if (fence !== null) return + const line = bodyOffset + index + 1 + if (FAQ_HEADING.test(text.trim())) { + report( + line, + 'faq', + `Body has a "${text.trim()}" heading.`, + 'Move the questions into the frontmatter `faq:` list (q/a pairs) and delete the section; the page renders it and emits FAQPage JSON-LD.' + ) + } + for (const match of text.matchAll(INTERNAL_LINK)) { + const linkSection = match[1] as Section + const target = match[2] + const rest = match[3] ?? '' + // A deeper path is a public asset (`/library//cover.jpg`) or a static sub-route. + if (rest !== '' && rest !== '/') continue + if (config.reservedSegments[linkSection].has(target)) continue + const link = `/${linkSection}/${target}` + if (linkSection === 'library' && target in config.mergedSlugs) { + const kept = config.mergedSlugs[target] + report( + line, + 'internal-link', + `${link} is retired; it was merged into /library/${kept}.`, + `Link /library/${kept} instead.` + ) + } else if (linkSection === 'blog' && config.movedBlogSlugs.includes(target)) { + report( + line, + 'internal-link', + `${link} moved to /library/${target}.`, + `Link /library/${target} instead.` + ) + } else if (!posts[linkSection].has(target)) { + const elsewhere = SECTIONS.filter((other) => posts[other].has(target)) + report( + line, + 'internal-link', + `${link} does not exist (no apps/sim/content/${linkSection}/${target}/index.mdx).`, + elsewhere.length > 0 + ? `Did you mean ${elsewhere.map((other) => `/${other}/${target}`).join(' or ')}?` + : 'Fix the slug or remove the link.' + ) + } + } + }) + + try { + await compile(body, { remarkPlugins: [remarkGfm], outputFormat: 'function-body' }) + } catch (error) { + const { line, reason, message } = error as { line?: number; reason?: string; message?: string } + report( + typeof line === 'number' ? bodyOffset + line : bodyOffset + 1, + 'mdx', + `MDX does not compile: ${reason ?? message}`, + 'Escape a literal `<` or `{` as `<` / `\\{`, and close every JSX tag.' + ) + } + + return findings +} + +/** Every retired library slug must map to a post that still exists. */ +export function checkMergedSlugTargets( + config: ContentCheckConfig, + posts: Record>, + file: string +): Finding[] { + return Object.entries(config.mergedSlugs) + .filter(([, kept]) => !posts.library.has(kept)) + .map(([retired, kept]) => ({ + file, + line: 1, + rule: 'internal-link' as const, + message: `Retired slug "${retired}" redirects to /library/${kept}, which does not exist.`, + hint: 'Point LIBRARY_MERGED_SLUGS at a surviving library post.', + })) +} + +export function readAuthorIds(contentDir: string): Set { + const dir = path.join(contentDir, 'authors') + if (!existsSync(dir)) return new Set() + return new Set( + readdirSync(dir) + .filter((name) => name.endsWith('.json')) + .map((name) => (JSON.parse(readFileSync(path.join(dir, name), 'utf-8')) as { id: string }).id) + ) +} + +export async function checkContent( + config: ContentCheckConfig, + only?: PostRef[] +): Promise<{ checked: number; findings: Finding[] }> { + const posts = indexPosts(config.contentDir) + const authorIds = readAuthorIds(config.contentDir) + const targets = + only ?? SECTIONS.flatMap((section) => [...posts[section]].map((slug) => ({ section, slug }))) + const results = await Promise.all( + targets.map((target) => checkPost(config, posts, authorIds, target)) + ) + return { checked: targets.length, findings: results.flat() } +} + +async function main() { + const root = path.resolve(import.meta.dir, '..') + const appDir = path.join(root, 'apps/sim') + const config: ContentCheckConfig = { + contentDir: path.join(appDir, 'content'), + publicDir: path.join(appDir, 'public'), + reservedSegments: readReservedSegments((section) => + path.join(appDir, 'app/(landing)', section) + ), + mergedSlugs: LIBRARY_MERGED_SLUGS, + movedBlogSlugs: LIBRARY_MOVED_BLOG_SLUGS, + } + + const args = process.argv.slice(2) + const flagIndex = args.indexOf('--slug') + const slugArg = flagIndex === -1 ? args.find((arg) => !arg.startsWith('-')) : args[flagIndex + 1] + if (flagIndex !== -1 && !slugArg) { + console.error('Usage: bun run check:library-content [--slug
/ | ]') + process.exit(1) + } + + let only: PostRef[] | undefined + const findings: Finding[] = [] + if (slugArg) { + const resolved = resolvePostArg(slugArg, indexPosts(config.contentDir)) + if (typeof resolved === 'string') { + console.error(resolved) + process.exit(1) + } + only = resolved + } else { + findings.push( + ...checkMergedSlugTargets( + config, + indexPosts(config.contentDir), + path.join(appDir, 'lib/library/retired-slugs.ts') + ) + ) + } + + const result = await checkContent(config, only) + findings.push(...result.findings) + + if (findings.length > 0) { + findings.sort((a, b) => a.file.localeCompare(b.file) || a.line - b.line) + console.error( + `Library content audit failed: ${findings.length} problem(s) in ${result.checked} post(s).\n\n` + + findings + .map( + (finding) => + ` ${path.relative(root, finding.file)}:${finding.line} [${finding.rule}] ${finding.message}\n fix: ${finding.hint}` + ) + .join('\n') + ) + process.exit(1) + } + + console.log(`Library content audit passed (${result.checked} posts).`) +} + +if (import.meta.main) await main() From c632938d927b5284d63cf44f486e160527d560e0 Mon Sep 17 00:00:00 2001 From: Waleed Latif Date: Thu, 1 Oct 2026 11:05:54 -0700 Subject: [PATCH 2/2] fix(content): tighten check:library-content link, author, and ogImage rules Treat draft blog/library posts and customer stories missing from CUSTOMER_STORIES as unserved link targets, reserve only sibling folders that define a page or route, validate author profiles with AuthorSchema, check moved blog slug destinations, and reject ogImage paths that resolve outside public. --- scripts/check-library-content.test.ts | 86 ++++++++++++- scripts/check-library-content.ts | 176 +++++++++++++++++++------- 2 files changed, 210 insertions(+), 52 deletions(-) diff --git a/scripts/check-library-content.test.ts b/scripts/check-library-content.test.ts index c6204719ecc..65db6936e0d 100644 --- a/scripts/check-library-content.test.ts +++ b/scripts/check-library-content.test.ts @@ -5,8 +5,9 @@ import { afterEach, beforeEach, describe, expect, it } from 'vitest' import { type ContentCheckConfig, checkContent, - checkMergedSlugTargets, + checkRedirectTargets, indexPosts, + readReservedSegments, resolvePostArg, } from './check-library-content' @@ -56,10 +57,12 @@ beforeEach(() => { reservedSegments: { blog: new Set(['tags']), library: new Set(['tags']), customers: new Set() }, mergedSlugs: { 'old-guide': 'kept-guide' }, movedBlogSlugs: ['moved-post'], + customerSlugs: ['acme'], } mkdirSync(path.join(config.contentDir, 'authors'), { recursive: true }) writeFileSync(path.join(config.contentDir, 'authors', 'sim.json'), '{"id":"sim","name":"Sim"}') writePost('library', 'kept-guide', 'The surviving guide.') + writePost('library', 'moved-post', 'Moved from the blog.') }) afterEach(() => { @@ -202,9 +205,86 @@ describe('check-library-content', () => { it('rejects a retired slug whose replacement no longer exists', () => { config.mergedSlugs = { 'old-guide': 'gone-guide' } - const findings = checkMergedSlugTargets(config, indexPosts(config.contentDir), 'map.ts') + const findings = checkRedirectTargets(config, indexPosts(config.contentDir), 'map.ts') expect(findings.map((finding) => finding.message)).toEqual([ - 'Retired slug "old-guide" redirects to /library/gone-guide, which does not exist.', + '/library/old-guide redirects to /library/gone-guide, which does not exist (no apps/sim/content/library/gone-guide/index.mdx).', + ]) + }) + + it('rejects a moved blog slug whose library destination is missing or a draft', () => { + config.movedBlogSlugs = ['moved-post', 'never-moved'] + writePost('library', 'moved-post', 'Unpublished.', { draft: 'true' }) + const findings = checkRedirectTargets(config, indexPosts(config.contentDir), 'map.ts') + expect(findings.map((finding) => finding.message)).toEqual([ + '/blog/moved-post redirects to /library/moved-post, which is a draft, so it 404s.', + '/blog/never-moved redirects to /library/never-moved, which does not exist (no apps/sim/content/library/never-moved/index.mdx).', + ]) + }) + + it('rejects a link to a draft blog or library post', async () => { + writePost('library', 'unpublished', 'Draft.', { draft: 'true' }) + writePost('library', 'post', '[a](/library/unpublished)') + expect(await findingsFor('library', 'post')).toEqual([ + { + line: 10, + rule: 'internal-link', + message: '/library/unpublished is a draft, so it 404s.', + }, + ]) + }) + + it('accepts a registered draft customer story but rejects an unregistered one', async () => { + writePost('customers', 'acme', 'Registered draft.', { draft: 'true' }) + writePost('customers', 'globex', 'Folder without a CUSTOMER_STORIES entry.') + writePost('library', 'post', '[a](/customers/acme)\n[b](/customers/globex)') + expect(await findingsFor('library', 'post')).toEqual([ + { + line: 11, + rule: 'internal-link', + message: + '/customers/globex is not in CUSTOMER_STORIES (apps/sim/lib/customers/data.ts), so it 404s.', + }, + ]) + }) + + it('reserves only sibling folders that define a page or route handler', () => { + const appDir = path.join(root, 'app') + for (const [dir, file] of [ + ['tags', 'page.tsx'], + ['rss.xml', 'route.ts'], + ['[slug]', 'page.tsx'], + ['components', 'card.tsx'], + ]) { + mkdirSync(path.join(appDir, dir), { recursive: true }) + writeFileSync(path.join(appDir, dir, file), '') + } + expect([...readReservedSegments(() => appDir).customers].sort()).toEqual(['rss.xml', 'tags']) + }) + + it('rejects an invalid author profile and does not accept its id', async () => { + writeFileSync(path.join(config.contentDir, 'authors', 'ghost.json'), '{"id":"ghost"}') + writeFileSync(path.join(config.contentDir, 'authors', 'broken.json'), '{not json') + writePost('library', 'post', 'Body.', { authors: '[ghost]' }) + const findings = await findingsFor('library', 'post') + expect(findings.map(({ rule, message }) => ({ rule, message }))).toEqual([ + { rule: 'frontmatter', message: expect.stringMatching(/^Author profile is invalid/) }, + { rule: 'frontmatter', message: expect.stringMatching(/^Author profile is invalid: name/) }, + { + rule: 'frontmatter', + message: 'Author "ghost" has no profile in apps/sim/content/authors.', + }, + ]) + }) + + it('rejects an ogImage that resolves outside public, even when the file exists', async () => { + writeFileSync(path.join(root, 'secret.jpg'), '') + writePost('library', 'post', 'Body.', { ogImage: '/../secret.jpg' }) + expect(await findingsFor('library', 'post')).toEqual([ + { + line: 7, + rule: 'og-image', + message: 'ogImage "/../secret.jpg" resolves outside apps/sim/public.', + }, ]) }) diff --git a/scripts/check-library-content.ts b/scripts/check-library-content.ts index 4da0e633e9d..796ebb36cdd 100644 --- a/scripts/check-library-content.ts +++ b/scripts/check-library-content.ts @@ -15,8 +15,10 @@ * - `faq`: the body has no FAQ heading (the FAQ lives in frontmatter, which renders it and emits * its JSON-LD), and no FAQ question or answer contains Markdown link syntax (it renders as text). * - `internal-link`: every `https://www.sim.ai/
/` link, and every relative - * `/
/` link target, names an existing post folder rather than a retired or moved - * slug. Apex `https://sim.ai` links belong to `check:site-urls`. + * `/
/` link target, names a page that serves: a published blog or library post, + * or a customer story registered in `CUSTOMER_STORIES` — never a retired or moved slug. Every + * retired or moved slug redirects to a published library post. Apex `https://sim.ai` links + * belong to `check:site-urls`. * * Run one post with `--slug
/` or `--slug `. */ @@ -25,7 +27,8 @@ import path from 'node:path' import { compile } from '@mdx-js/mdx' import matter from 'gray-matter' import remarkGfm from 'remark-gfm' -import { ContentFrontmatterSchema } from '../apps/sim/lib/content/schema' +import { AuthorSchema, ContentFrontmatterSchema } from '../apps/sim/lib/content/schema' +import { CUSTOMER_STORIES } from '../apps/sim/lib/customers/data' import { LIBRARY_MERGED_SLUGS, LIBRARY_MOVED_BLOG_SLUGS, @@ -55,8 +58,13 @@ export interface ContentCheckConfig { mergedSlugs: Readonly> /** Blog slugs that now live under `/library`. */ movedBlogSlugs: readonly string[] + /** Customer slugs the `/customers/[slug]` route serves (`CUSTOMER_STORIES`). */ + customerSlugs: readonly string[] } +/** Post folders per section, keyed by slug, with each post's `draft` flag. */ +export type PostIndex = Record> + export interface PostRef { section: Section slug: string @@ -71,31 +79,56 @@ const INTERNAL_LINK = const MARKDOWN_LINK = /\[[^\]\n]*\]\([^)\n]*\)/ const FAQ_HEADING = /^#{1,6}\s+FAQs?\s*:?\s*$/i const CODE_FENCE = /^\s*(```|~~~)/ +const ROUTE_FILES = ['page.tsx', 'page.ts', 'route.tsx', 'route.ts'] function isSection(value: string): value is Section { return (SECTIONS as readonly string[]).includes(value) } +/** A post whose frontmatter does not parse counts as published; its own check reports it. */ +function readDraft(file: string): boolean { + try { + return matter(readFileSync(file, 'utf-8'), {}).data.draft === true + } catch { + return false + } +} + /** Post folders per section: every directory that holds an `index.mdx`. */ -export function indexPosts(contentDir: string): Record> { - const index = {} as Record> +export function indexPosts(contentDir: string): PostIndex { + const index = {} as PostIndex for (const section of SECTIONS) { const sectionDir = path.join(contentDir, section) - index[section] = new Set( - existsSync(sectionDir) - ? readdirSync(sectionDir, { withFileTypes: true }) - .filter( - (entry) => - entry.isDirectory() && existsSync(path.join(sectionDir, entry.name, 'index.mdx')) - ) - .map((entry) => entry.name) - : [] - ) + index[section] = new Map() + if (!existsSync(sectionDir)) continue + for (const entry of readdirSync(sectionDir, { withFileTypes: true })) { + const file = path.join(sectionDir, entry.name, 'index.mdx') + if (entry.isDirectory() && existsSync(file)) { + index[section].set(entry.name, { draft: readDraft(file) }) + } + } } return index } -/** Static route segments beside each section's `[slug]` route, read from the app router tree. */ +/** Why `/
/` would not serve a page, or null when it does. */ +export function unservedReason( + posts: PostIndex, + customerSlugs: readonly string[], + section: Section, + slug: string +): string | null { + const post = posts[section].get(slug) + if (!post) return `does not exist (no apps/sim/content/${section}/${slug}/index.mdx)` + if (section === 'customers') { + return customerSlugs.includes(slug) + ? null + : 'is not in CUSTOMER_STORIES (apps/sim/lib/customers/data.ts), so it 404s' + } + return post.draft ? 'is a draft, so it 404s' : null +} + +/** Static routes beside each section's `[slug]` route: folders that define a page or route handler. */ export function readReservedSegments(sectionAppDir: (section: Section) => string) { const reserved = {} as Record> for (const section of SECTIONS) { @@ -103,7 +136,12 @@ export function readReservedSegments(sectionAppDir: (section: Section) => string reserved[section] = new Set( existsSync(dir) ? readdirSync(dir, { withFileTypes: true }) - .filter((entry) => entry.isDirectory() && !/^[[(_]/.test(entry.name)) + .filter( + (entry) => + entry.isDirectory() && + !/^[[(_]/.test(entry.name) && + ROUTE_FILES.some((name) => existsSync(path.join(dir, entry.name, name))) + ) .map((entry) => entry.name) : [] ) @@ -112,10 +150,7 @@ export function readReservedSegments(sectionAppDir: (section: Section) => string } /** Parses `
/` or a bare ``, resolving the latter against the post index. */ -export function resolvePostArg( - arg: string, - posts: Record> -): PostRef[] | string { +export function resolvePostArg(arg: string, posts: PostIndex): PostRef[] | string { const [first, second] = arg.replace(/\/+$/, '').split('/') if (second !== undefined) { if (!isSection(first)) return `Unknown section "${first}"; use one of ${SECTIONS.join(', ')}.` @@ -137,7 +172,7 @@ function lineOfKey(frontmatterLines: string[], key: string): number { /** Validates one post, returning every finding (empty when the post is clean). */ export async function checkPost( config: ContentCheckConfig, - posts: Record>, + posts: PostIndex, authorIds: ReadonlySet, { section, slug }: PostRef ): Promise { @@ -228,8 +263,16 @@ export async function checkPost( `Point it at a file under apps/sim/public, e.g. /${section}/${slug}/cover.jpg.` ) } else { - const target = path.join(config.publicDir, data.ogImage) - if (!existsSync(target) || !statSync(target).isFile()) { + const target = path.resolve(config.publicDir, `.${data.ogImage}`) + const relative = path.relative(config.publicDir, target) + if (relative.startsWith('..') || path.isAbsolute(relative)) { + report( + line, + 'og-image', + `ogImage "${data.ogImage}" resolves outside apps/sim/public.`, + `Point it at a file under apps/sim/public, e.g. /${section}/${slug}/cover.jpg.` + ) + } else if (!existsSync(target) || !statSync(target).isFile()) { report( line, 'og-image', @@ -308,15 +351,20 @@ export async function checkPost( `${link} moved to /library/${target}.`, `Link /library/${target} instead.` ) - } else if (!posts[linkSection].has(target)) { - const elsewhere = SECTIONS.filter((other) => posts[other].has(target)) + } else { + const reason = unservedReason(posts, config.customerSlugs, linkSection, target) + if (!reason) continue + const elsewhere = SECTIONS.filter( + (other) => + other !== linkSection && !unservedReason(posts, config.customerSlugs, other, target) + ) report( line, 'internal-link', - `${link} does not exist (no apps/sim/content/${linkSection}/${target}/index.mdx).`, + `${link} ${reason}.`, elsewhere.length > 0 ? `Did you mean ${elsewhere.map((other) => `/${other}/${target}`).join(' or ')}?` - : 'Fix the slug or remove the link.' + : 'Fix the slug, publish the target, or remove the link.' ) } } @@ -337,31 +385,59 @@ export async function checkPost( return findings } -/** Every retired library slug must map to a post that still exists. */ -export function checkMergedSlugTargets( +/** Every retired or moved slug must redirect to a library post that serves a page. */ +export function checkRedirectTargets( config: ContentCheckConfig, - posts: Record>, + posts: PostIndex, file: string ): Finding[] { - return Object.entries(config.mergedSlugs) - .filter(([, kept]) => !posts.library.has(kept)) - .map(([retired, kept]) => ({ + const redirects = [ + ...Object.entries(config.mergedSlugs).map(([from, to]) => [`/library/${from}`, to]), + ...config.movedBlogSlugs.map((slug) => [`/blog/${slug}`, slug]), + ] + return redirects.flatMap(([from, to]) => { + const reason = unservedReason(posts, config.customerSlugs, 'library', to) + if (!reason) return [] + return { file, line: 1, rule: 'internal-link' as const, - message: `Retired slug "${retired}" redirects to /library/${kept}, which does not exist.`, - hint: 'Point LIBRARY_MERGED_SLUGS at a surviving library post.', - })) + message: `${from} redirects to /library/${to}, which ${reason}.`, + hint: 'Point the redirect in retired-slugs.ts at a published library post.', + } + }) } -export function readAuthorIds(contentDir: string): Set { +/** Author ids from every author JSON that passes `AuthorSchema`; invalid profiles are findings. */ +export function readAuthors(contentDir: string): { ids: Set; findings: Finding[] } { const dir = path.join(contentDir, 'authors') - if (!existsSync(dir)) return new Set() - return new Set( - readdirSync(dir) - .filter((name) => name.endsWith('.json')) - .map((name) => (JSON.parse(readFileSync(path.join(dir, name), 'utf-8')) as { id: string }).id) - ) + const ids = new Set() + const findings: Finding[] = [] + if (!existsSync(dir)) return { ids, findings } + for (const name of readdirSync(dir) + .filter((entry) => entry.endsWith('.json')) + .sort()) { + const file = path.join(dir, name) + let json: unknown + try { + json = JSON.parse(readFileSync(file, 'utf-8')) + } catch (error) { + json = error + } + const result = AuthorSchema.safeParse(json) + if (result.success) { + ids.add(result.data.id) + continue + } + findings.push({ + file, + line: 1, + rule: 'frontmatter', + message: `Author profile is invalid: ${result.error.issues.map((issue) => `${issue.path.join('.') || '(root)'} ${issue.message}`).join('; ')}`, + hint: 'Match AuthorSchema in apps/sim/lib/content/schema.ts (valid JSON with id and name).', + }) + } + return { ids, findings } } export async function checkContent( @@ -369,13 +445,14 @@ export async function checkContent( only?: PostRef[] ): Promise<{ checked: number; findings: Finding[] }> { const posts = indexPosts(config.contentDir) - const authorIds = readAuthorIds(config.contentDir) + const authors = readAuthors(config.contentDir) const targets = - only ?? SECTIONS.flatMap((section) => [...posts[section]].map((slug) => ({ section, slug }))) + only ?? + SECTIONS.flatMap((section) => [...posts[section].keys()].map((slug) => ({ section, slug }))) const results = await Promise.all( - targets.map((target) => checkPost(config, posts, authorIds, target)) + targets.map((target) => checkPost(config, posts, authors.ids, target)) ) - return { checked: targets.length, findings: results.flat() } + return { checked: targets.length, findings: [...authors.findings, ...results.flat()] } } async function main() { @@ -389,6 +466,7 @@ async function main() { ), mergedSlugs: LIBRARY_MERGED_SLUGS, movedBlogSlugs: LIBRARY_MOVED_BLOG_SLUGS, + customerSlugs: CUSTOMER_STORIES.map((story) => story.slug), } const args = process.argv.slice(2) @@ -410,7 +488,7 @@ async function main() { only = resolved } else { findings.push( - ...checkMergedSlugTargets( + ...checkRedirectTargets( config, indexPosts(config.contentDir), path.join(appDir, 'lib/library/retired-slugs.ts')