diff --git a/__tests__/markers.test.ts b/__tests__/markers.test.ts index 2be96667..cc98421f 100644 --- a/__tests__/markers.test.ts +++ b/__tests__/markers.test.ts @@ -44,13 +44,34 @@ describe('locateSection', () => { expect(body(afterExample('\\'))).toBe('\nx'); }); - // A single pair is located wherever it sits. Generated text can hold an - // unclosed fence — a description whose fence the description updater - // flattened — and code detection must not let it hide the pairs after it. - it('locates a single pair even after an unclosed fence', () => { - expect( - body(`\n\`\`\`\n\n${afterExample()}`), - ).toBe('\nx'); + // A fence that nothing closes runs to the end of the document, and markers + // inside it still count, so it cannot hide the pairs after it. + it('locates pairs inside and after a fence that nothing closes', () => { + const source = `\n\`\`\`\n\n${afterExample()}`; + + expect(body(source, 'description')).toBe('\n```'); + expect(body(source)).toBe('\nx'); + }); + + describe('markers inside closed code', () => { + it('does not pair a start marker with an end marker in a later code example', () => { + const source = ['', 'prose', '```', '', '```'].join( + '\n', + ); + + expect(body(source)).toBe(''); + }); + + it.each([ + ['a fenced code block', ['```', '', '', '```']], + ['inline code', ['Add `x ` to your README.']], + [ + 'inline code opened by triple backticks', + ['Add ```x ``` to your README.'], + ], + ])('reports a pair that is only an example in %s as missing', (_label, example) => { + expect(body(example.join('\n'))).toBe(''); + }); }); // #691: a README that documents the markers repeats them. diff --git a/__tests__/readme-editor.test.ts b/__tests__/readme-editor.test.ts index 23e62d33..6fa2c234 100644 --- a/__tests__/readme-editor.test.ts +++ b/__tests__/readme-editor.test.ts @@ -270,6 +270,28 @@ describe('ReadmeEditor', () => { }, ); + // The only end marker is an example in closed code, so pairing it with + // the real start marker would replace the prose and code between them. + it('leaves the file unchanged when the only end marker is in a code example', async () => { + const original = [ + '', + '', + USER_PROSE, + '', + '```markdown', + '', + '```', + '', + ].join('\n'); + fs.writeFileSync(readmePath, original, 'utf8'); + const editor = new ReadmeEditor(readmePath); + editor.updateSection('inputs', UNALIGNED); + + await editor.dumpToFile(); + + expect(read()).toBe(original); + }); + it.each([ ['above', (fence: string) => [fence, '', ...pair]], ['below', (fence: string) => [...pair, '', fence]], diff --git a/__tests__/verify-readme-contract.test.ts b/__tests__/verify-readme-contract.test.ts index 36eacad3..a7b7004f 100644 --- a/__tests__/verify-readme-contract.test.ts +++ b/__tests__/verify-readme-contract.test.ts @@ -920,6 +920,28 @@ describe('README contract verifier regressions', () => { expect(output).toContain('content outside the section markers is byte-identical'); }); + // A start marker and an end marker inside a closed code example are not a + // pair, so a run that filled between them rewrote the user's text. + it('rejects a run that paired a start marker with an end marker in a code example', () => { + const original = [ + '', + '', + 'USER PROSE', + '', + '```markdown', + '', + '```', + '', + ].join('\n'); + const filled = ['', '', 'NEW', '', '```', ''].join( + '\n', + ); + + expect(annotations(() => verify(filled, ACTION, undefined, '.', original))).toContain( + 'rewrote content outside the section markers', + ); + }); + it('names an ambiguous section and accepts it left unchanged', () => { const original = withProse( [ diff --git a/docs/tool-contract.md b/docs/tool-contract.md index f7579187..73ab8213 100644 --- a/docs/tool-contract.md +++ b/docs/tool-contract.md @@ -47,15 +47,15 @@ bytes, and its spans are written as LF, because no single ending reproduces it. Which bytes are a span is a separate question, decided by `src/markers.ts` rather than by the formatter. A marker straight after a backtick or a backslash -is quoted, and is never a marker. A section with one start marker and one end -marker after it is filled wherever the pair sits. For any other shape, the -markers inside code — inline code, fenced or indented code blocks — are examples -and do not count. If the markers that still count are not a single pair, the -section is left unchanged with a warning naming the marker lines, because any -guess at the intended pair would replace text outside it. Code decides only -when the markers are not a single pair: generated text can hold an unclosed -fence, and a code check on every lookup would let that fence hide every pair -after it. +is quoted, and is never a marker. A marker inside closed code — inline code, an +indented code block, or a fenced code block that its closing fence ends — is an +example and does not count. A marker inside a fence that nothing closes does +count: that fence runs to the end of the document, and text generated from an +action's metadata can hold one, so treating it as code would hide every marker +after it. A section is filled only when the markers that count are one start +marker and one end marker after it. Any other shape leaves the section +unchanged, with a warning naming the marker lines, because any guess at the +intended pair would replace text outside it. The line break and indentation before an end marker that starts its line belong to the marker, not the span. A padded span always puts its end marker on its diff --git a/scripts/verify-readme-contract.mjs b/scripts/verify-readme-contract.mjs index 88337068..4af13396 100644 --- a/scripts/verify-readme-contract.mjs +++ b/scripts/verify-readme-contract.mjs @@ -118,13 +118,29 @@ const fail = (message) => { console.log(`::error::${message}`); }; -/** The half-open ranges Markdown renders as code, from prettier's parser. */ -const codeRanges = (source) => { +/** + * The half-open ranges whose markers are examples, from prettier's parser: + * inline code, indented code, and fenced code that its closing fence ends. A + * fence that nothing closes runs to the end of the document and is not one. + */ +const exampleRanges = (source) => { const shift = source.startsWith('') ? 1 : 0; const ranges = []; + // Only a fenced block can be unclosed. Its node starts at its fence; an + // indented block's node starts at its indentation, and inline code is its + // own node kind. + const closedFence = (text) => { + const lines = text.split('\n'); + const opener = /^(`{3,}|~{3,})/.exec(lines[0])?.[1]; + if (!opener) return true; + const closer = lines.at(-1).replace(/^[\t >]*/, '').trimEnd(); + return lines.length > 1 && closer.length >= opener.length && /^(`+|~+)$/.test(closer) && closer[0] === opener[0]; + }; const walk = (node) => { if ((node.type === 'code' || node.type === 'inlineCode') && node.position) { - ranges.push([node.position.start.offset + shift, node.position.end.offset + shift]); + const from = node.position.start.offset + shift; + const to = node.position.end.offset + shift; + if (node.type === 'inlineCode' || closedFence(source.slice(from, to))) ranges.push([from, to]); } (node.children ?? []).forEach(walk); }; @@ -139,8 +155,7 @@ const codeRanges = (source) => { * Written apart from `src/markers.ts` on purpose, so each implementation * checks the other rather than agreeing by construction. The rules are the * same: a marker straight after a backtick or backslash is quoted, the name is - * matched literally, and when the markers are not a single pair, the markers - * inside code are examples. + * matched literally, and a marker inside closed code is an example. */ const markersOf = (source, name) => { const literal = name.replaceAll(/[$()*+.?[\\\]^{|}]/g, '\\$&'); @@ -148,16 +163,12 @@ const markersOf = (source, name) => { [...source.matchAll(new RegExp(`(?`, 'g'))].map( (match) => ({ at: match.index, after: match.index + match[0].length }), ); - let starts = find('start'); - let ends = find('end'); - const pair = starts.length === 1 && ends.length === 1 && ends[0].at >= starts[0].after; - if (!pair) { - const code = codeRanges(source); - const live = ({ at }) => !code.some(([from, to]) => at >= from && at < to); - starts = starts.filter(live); - ends = ends.filter(live); - } - return { starts, ends }; + const starts = find('start'); + const ends = find('end'); + if (starts.length === 0 && ends.length === 0) return { starts, ends }; + const examples = exampleRanges(source); + const counts = ({ at }) => !examples.some(([from, to]) => at >= from && at < to); + return { starts: starts.filter(counts), ends: ends.filter(counts) }; }; /** diff --git a/src/markers.ts b/src/markers.ts index 45545a74..960f90ad 100644 --- a/src/markers.ts +++ b/src/markers.ts @@ -5,16 +5,17 @@ * A marker straight after a backtick or a backslash is quoted or escaped, and * is never a marker. The name is matched literally. * - * A section with one start marker and one end marker after it is located - * wherever the pair sits. Any other shape — a README that documents the - * markers repeats them — has the markers inside code set aside as examples: - * inline code, fenced or indented code blocks. Code decides only when the - * markers are not a single pair: text this tool generated can hold an unclosed - * fence, and a code check on every lookup would let that fence hide every pair - * after it. + * A marker inside closed code — inline code, an indented code block, or a + * fenced code block that its closing fence ends — is an example and does not + * count. A marker inside a fence that nothing closes does count: such a fence + * runs to the end of the document, and text generated from an action's + * metadata can hold one, so treating it as code would hide every marker after + * it. * - * Any other shape is reported rather than guessed at, because a wrong guess - * replaces text outside the pair, which is the user's. + * A section is located only when the markers that count are one start marker + * and one end marker after it. Any other shape is reported rather than guessed + * at, because a wrong guess replaces text outside the pair, which is the + * user's. */ import * as markdown from 'prettier/plugins/markdown'; @@ -105,21 +106,42 @@ function linesOf(source: string, offsets: number[]): number[] { } /** - * Whether the markers are exactly one start marker and one end marker after it. - * @param {RegExpExecArray[]} starts - The start markers. - * @param {RegExpExecArray[]} ends - The end markers. - * @returns {boolean} - Whether they form a single pair. + * Whether a fenced code block is one that nothing closes. + * @param {string} code - The source text of a fenced code block. + * @returns {boolean} - Whether it is an unclosed fence. */ -function isPair(starts: RegExpExecArray[], ends: RegExpExecArray[]): boolean { - const [start] = starts; - const [end] = ends; - return ( - starts.length === 1 && - ends.length === 1 && - start !== undefined && - end !== undefined && - end.index >= start.index + start[0].length - ); +function isUnclosedFence(code: string): boolean { + const lines = code.split('\n'); + // A fenced block's node starts at its fence, so the opener is always there. + const opener = /^(`{3,}|~{3,})/.exec(lines[0] ?? '')?.[1] ?? '```'; + const closer = (lines.at(-1) ?? '').replace(/^[\t >]*/, '').trimEnd(); + const closes = + lines.length > 1 && + closer.length >= opener.length && + closer === opener.charAt(0).repeat(closer.length); + return !closes; +} + +/** The last document `exampleRanges` parsed, and its result. */ +let parsed: { source: string; ranges: [number, number, boolean][] } | undefined; + +/** + * The half-open ranges of `source` whose markers are examples: every code + * range except a fenced code block that nothing closes. The last result is kept, because + * each section is located more than once in the same document. + * @param {string} source - The document. + * @returns {Array<[number, number, boolean]>} - The example ranges. + */ +function exampleRanges(source: string): [number, number, boolean][] { + if (parsed?.source !== source) { + parsed = { + source, + ranges: codeRanges(source).filter( + ([from, to, fenced]) => !fenced || !isUnclosedFence(source.slice(from, to)), + ), + }; + } + return parsed.ranges; } /** @@ -137,12 +159,12 @@ export function locateSection(source: string, name: string): SectionSpan { let starts = markers('start'); let ends = markers('end'); - if (!isPair(starts, ends)) { - const code = codeRanges(source); - const live = (match: RegExpExecArray): boolean => - !code.some(([from, to]) => match.index >= from && match.index < to); - starts = starts.filter(live); - ends = ends.filter(live); + if (starts.length > 0 || ends.length > 0) { + const examples = exampleRanges(source); + const counts = (match: RegExpExecArray): boolean => + !examples.some(([from, to]) => match.index >= from && match.index < to); + starts = starts.filter(counts); + ends = ends.filter(counts); } const lines = (): number[] => linesOf( @@ -206,25 +228,6 @@ function closestSection(name: string, sections: readonly string[]): string | und return best?.section; } -/** - * Whether a fenced code block is one that nothing closes. It runs to the end of - * the document, so the markers inside it are not examples: generated text can - * hold such a fence, and it would otherwise hide every marker after it. - * @param {string} code - The source text of a fenced code block. - * @returns {boolean} - Whether it is an unclosed fence. - */ -function isUnclosedFence(code: string): boolean { - const lines = code.split('\n'); - // A fenced block's node starts at its fence, so the opener is always there. - const opener = /^(`{3,}|~{3,})/.exec(lines[0] ?? '')?.[1] ?? '```'; - const closer = (lines.at(-1) ?? '').replace(/^[\t >]*/, '').trimEnd(); - const closes = - lines.length > 1 && - closer.length >= opener.length && - closer === opener.charAt(0).repeat(closer.length); - return !closes; -} - /** * Warnings about markers that stop a README from being generated as its * author intended, found before any section is written: @@ -234,9 +237,8 @@ function isUnclosedFence(code: string): boolean { * - a README with no marker for any section being generated, which the tool * otherwise leaves unchanged in silence. * - * A marker inside closed code is an example and is not reported. A marker - * inside a fence that nothing closes is reported, as `locateSection` would - * still find it. + * Markers are read as `locateSection` reads them: one inside closed code is + * an example and is not reported. * @param {string} source - The document. * @param {readonly string[]} sections - Every section name the tool knows. * @param {readonly string[]} requested - The sections being generated. @@ -247,9 +249,7 @@ export function diagnoseMarkers( sections: readonly string[], requested: readonly string[] = sections, ): string[] { - const examples = codeRanges(source).filter( - ([from, to, fenced]) => !fenced || !isUnclosedFence(source.slice(from, to)), - ); + const examples = exampleRanges(source); const warnings: string[] = []; for (const match of source.matchAll(/(?/g)) { const name = match[2] ?? '';