Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
35 changes: 28 additions & 7 deletions __tests__/markers.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -44,13 +44,34 @@ describe('locateSection', () => {
expect(body(afterExample('\\<!-- end inputs -->'))).toBe('\nx');
});

// A single pair is located wherever it sits. Generated text can hold an
// unclosed fence — a description whose fence the description updater
// flattened — and code detection must not let it hide the pairs after it.
it('locates a single pair even after an unclosed fence', () => {
expect(
body(`<!-- start description -->\n\`\`\`\n<!-- end description -->\n${afterExample()}`),
).toBe('\nx');
// A fence that nothing closes runs to the end of the document, and markers
// inside it still count, so it cannot hide the pairs after it.
it('locates pairs inside and after a fence that nothing closes', () => {
const source = `<!-- start description -->\n\`\`\`\n<!-- end description -->\n${afterExample()}`;

expect(body(source, 'description')).toBe('\n```');
expect(body(source)).toBe('\nx');
});

describe('markers inside closed code', () => {
it('does not pair a start marker with an end marker in a later code example', () => {
const source = ['<!-- start inputs -->', 'prose', '```', '<!-- end inputs -->', '```'].join(
'\n',
);

expect(body(source)).toBe('<unpaired>');
});

it.each([
['a fenced code block', ['```', '<!-- start inputs -->', '<!-- end inputs -->', '```']],
['inline code', ['Add `x <!-- start inputs --><!-- end inputs -->` to your README.']],
[
'inline code opened by triple backticks',
['Add ```x <!-- start inputs --><!-- end inputs -->``` to your README.'],
],
])('reports a pair that is only an example in %s as missing', (_label, example) => {
expect(body(example.join('\n'))).toBe('<missing>');
});
});

// #691: a README that documents the markers repeats them.
Expand Down
22 changes: 22 additions & 0 deletions __tests__/readme-editor.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -270,6 +270,28 @@ describe('ReadmeEditor', () => {
},
);

// The only end marker is an example in closed code, so pairing it with
// the real start marker would replace the prose and code between them.
it('leaves the file unchanged when the only end marker is in a code example', async () => {
const original = [
'<!-- start inputs -->',
'',
USER_PROSE,
'',
'```markdown',
'<!-- end inputs -->',
'```',
'',
].join('\n');
fs.writeFileSync(readmePath, original, 'utf8');
const editor = new ReadmeEditor(readmePath);
editor.updateSection('inputs', UNALIGNED);

await editor.dumpToFile();

expect(read()).toBe(original);
});

it.each([
['above', (fence: string) => [fence, '', ...pair]],
['below', (fence: string) => [...pair, '', fence]],
Expand Down
22 changes: 22 additions & 0 deletions __tests__/verify-readme-contract.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -920,6 +920,28 @@ describe('README contract verifier regressions', () => {
expect(output).toContain('content outside the section markers is byte-identical');
});

// A start marker and an end marker inside a closed code example are not a
// pair, so a run that filled between them rewrote the user's text.
it('rejects a run that paired a start marker with an end marker in a code example', () => {
const original = [
'<!-- start inputs -->',
'',
'USER PROSE',
'',
'```markdown',
'<!-- end inputs -->',
'```',
'',
].join('\n');
const filled = ['<!-- start inputs -->', '', 'NEW', '<!-- end inputs -->', '```', ''].join(
'\n',
);

expect(annotations(() => verify(filled, ACTION, undefined, '.', original))).toContain(
'rewrote content outside the section markers',
);
});

it('names an ambiguous section and accepts it left unchanged', () => {
const original = withProse(
[
Expand Down
18 changes: 9 additions & 9 deletions docs/tool-contract.md
Original file line number Diff line number Diff line change
Expand Up @@ -47,15 +47,15 @@ bytes, and its spans are written as LF, because no single ending reproduces it.

Which bytes are a span is a separate question, decided by `src/markers.ts`
rather than by the formatter. A marker straight after a backtick or a backslash
is quoted, and is never a marker. A section with one start marker and one end
marker after it is filled wherever the pair sits. For any other shape, the
markers inside code — inline code, fenced or indented code blocks — are examples
and do not count. If the markers that still count are not a single pair, the
section is left unchanged with a warning naming the marker lines, because any
guess at the intended pair would replace text outside it. Code decides only
when the markers are not a single pair: generated text can hold an unclosed
fence, and a code check on every lookup would let that fence hide every pair
after it.
is quoted, and is never a marker. A marker inside closed code — inline code, an
indented code block, or a fenced code block that its closing fence ends — is an
example and does not count. A marker inside a fence that nothing closes does
count: that fence runs to the end of the document, and text generated from an
action's metadata can hold one, so treating it as code would hide every marker
after it. A section is filled only when the markers that count are one start
marker and one end marker after it. Any other shape leaves the section
unchanged, with a warning naming the marker lines, because any guess at the
intended pair would replace text outside it.

The line break and indentation before an end marker that starts its line belong
to the marker, not the span. A padded span always puts its end marker on its
Expand Down
41 changes: 26 additions & 15 deletions scripts/verify-readme-contract.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -118,13 +118,29 @@ const fail = (message) => {
console.log(`::error::${message}`);
};

/** The half-open ranges Markdown renders as code, from prettier's parser. */
const codeRanges = (source) => {
/**
* The half-open ranges whose markers are examples, from prettier's parser:
* inline code, indented code, and fenced code that its closing fence ends. A
* fence that nothing closes runs to the end of the document and is not one.
*/
const exampleRanges = (source) => {
const shift = source.startsWith('') ? 1 : 0;
const ranges = [];
// Only a fenced block can be unclosed. Its node starts at its fence; an
// indented block's node starts at its indentation, and inline code is its
// own node kind.
const closedFence = (text) => {
const lines = text.split('\n');
const opener = /^(`{3,}|~{3,})/.exec(lines[0])?.[1];
if (!opener) return true;
const closer = lines.at(-1).replace(/^[\t >]*/, '').trimEnd();
return lines.length > 1 && closer.length >= opener.length && /^(`+|~+)$/.test(closer) && closer[0] === opener[0];
};
const walk = (node) => {
if ((node.type === 'code' || node.type === 'inlineCode') && node.position) {
ranges.push([node.position.start.offset + shift, node.position.end.offset + shift]);
const from = node.position.start.offset + shift;
const to = node.position.end.offset + shift;
if (node.type === 'inlineCode' || closedFence(source.slice(from, to))) ranges.push([from, to]);
}
(node.children ?? []).forEach(walk);
};
Expand All @@ -139,25 +155,20 @@ const codeRanges = (source) => {
* Written apart from `src/markers.ts` on purpose, so each implementation
* checks the other rather than agreeing by construction. The rules are the
* same: a marker straight after a backtick or backslash is quoted, the name is
* matched literally, and when the markers are not a single pair, the markers
* inside code are examples.
* matched literally, and a marker inside closed code is an example.
*/
const markersOf = (source, name) => {
const literal = name.replaceAll(/[$()*+.?[\\\]^{|}]/g, '\\$&');
const find = (kind) =>
[...source.matchAll(new RegExp(`(?<![\`\\\\])<!--\\s+${kind}\\s+${literal}\\s+-->`, 'g'))].map(
(match) => ({ at: match.index, after: match.index + match[0].length }),
);
let starts = find('start');
let ends = find('end');
const pair = starts.length === 1 && ends.length === 1 && ends[0].at >= starts[0].after;
if (!pair) {
const code = codeRanges(source);
const live = ({ at }) => !code.some(([from, to]) => at >= from && at < to);
starts = starts.filter(live);
ends = ends.filter(live);
}
return { starts, ends };
const starts = find('start');
const ends = find('end');
if (starts.length === 0 && ends.length === 0) return { starts, ends };
const examples = exampleRanges(source);
const counts = ({ at }) => !examples.some(([from, to]) => at >= from && at < to);
return { starts: starts.filter(counts), ends: ends.filter(counts) };
};

/**
Expand Down
108 changes: 54 additions & 54 deletions src/markers.ts
Original file line number Diff line number Diff line change
Expand Up @@ -5,16 +5,17 @@
* A marker straight after a backtick or a backslash is quoted or escaped, and
* is never a marker. The name is matched literally.
*
* A section with one start marker and one end marker after it is located
* wherever the pair sits. Any other shape — a README that documents the
* markers repeats them — has the markers inside code set aside as examples:
* inline code, fenced or indented code blocks. Code decides only when the
* markers are not a single pair: text this tool generated can hold an unclosed
* fence, and a code check on every lookup would let that fence hide every pair
* after it.
* A marker inside closed code — inline code, an indented code block, or a
* fenced code block that its closing fence ends — is an example and does not
* count. A marker inside a fence that nothing closes does count: such a fence
* runs to the end of the document, and text generated from an action's
* metadata can hold one, so treating it as code would hide every marker after
* it.
*
* Any other shape is reported rather than guessed at, because a wrong guess
* replaces text outside the pair, which is the user's.
* A section is located only when the markers that count are one start marker
* and one end marker after it. Any other shape is reported rather than guessed
* at, because a wrong guess replaces text outside the pair, which is the
* user's.
*/

import * as markdown from 'prettier/plugins/markdown';
Expand Down Expand Up @@ -105,21 +106,42 @@ function linesOf(source: string, offsets: number[]): number[] {
}

/**
* Whether the markers are exactly one start marker and one end marker after it.
* @param {RegExpExecArray[]} starts - The start markers.
* @param {RegExpExecArray[]} ends - The end markers.
* @returns {boolean} - Whether they form a single pair.
* Whether a fenced code block is one that nothing closes.
* @param {string} code - The source text of a fenced code block.
* @returns {boolean} - Whether it is an unclosed fence.
*/
function isPair(starts: RegExpExecArray[], ends: RegExpExecArray[]): boolean {
const [start] = starts;
const [end] = ends;
return (
starts.length === 1 &&
ends.length === 1 &&
start !== undefined &&
end !== undefined &&
end.index >= start.index + start[0].length
);
function isUnclosedFence(code: string): boolean {
const lines = code.split('\n');
// A fenced block's node starts at its fence, so the opener is always there.
const opener = /^(`{3,}|~{3,})/.exec(lines[0] ?? '')?.[1] ?? '```';
const closer = (lines.at(-1) ?? '').replace(/^[\t >]*/, '').trimEnd();
const closes =
lines.length > 1 &&
closer.length >= opener.length &&
closer === opener.charAt(0).repeat(closer.length);
return !closes;
}

/** The last document `exampleRanges` parsed, and its result. */
let parsed: { source: string; ranges: [number, number, boolean][] } | undefined;

/**
* The half-open ranges of `source` whose markers are examples: every code
* range except a fenced code block that nothing closes. The last result is kept, because
* each section is located more than once in the same document.
* @param {string} source - The document.
* @returns {Array<[number, number, boolean]>} - The example ranges.
*/
function exampleRanges(source: string): [number, number, boolean][] {
if (parsed?.source !== source) {
parsed = {
source,
ranges: codeRanges(source).filter(
([from, to, fenced]) => !fenced || !isUnclosedFence(source.slice(from, to)),
),
};
}
return parsed.ranges;
}

/**
Expand All @@ -137,12 +159,12 @@ export function locateSection(source: string, name: string): SectionSpan {

let starts = markers('start');
let ends = markers('end');
if (!isPair(starts, ends)) {
const code = codeRanges(source);
const live = (match: RegExpExecArray): boolean =>
!code.some(([from, to]) => match.index >= from && match.index < to);
starts = starts.filter(live);
ends = ends.filter(live);
if (starts.length > 0 || ends.length > 0) {
const examples = exampleRanges(source);
const counts = (match: RegExpExecArray): boolean =>
!examples.some(([from, to]) => match.index >= from && match.index < to);
starts = starts.filter(counts);
ends = ends.filter(counts);
}
const lines = (): number[] =>
linesOf(
Expand Down Expand Up @@ -206,25 +228,6 @@ function closestSection(name: string, sections: readonly string[]): string | und
return best?.section;
}

/**
* Whether a fenced code block is one that nothing closes. It runs to the end of
* the document, so the markers inside it are not examples: generated text can
* hold such a fence, and it would otherwise hide every marker after it.
* @param {string} code - The source text of a fenced code block.
* @returns {boolean} - Whether it is an unclosed fence.
*/
function isUnclosedFence(code: string): boolean {
const lines = code.split('\n');
// A fenced block's node starts at its fence, so the opener is always there.
const opener = /^(`{3,}|~{3,})/.exec(lines[0] ?? '')?.[1] ?? '```';
const closer = (lines.at(-1) ?? '').replace(/^[\t >]*/, '').trimEnd();
const closes =
lines.length > 1 &&
closer.length >= opener.length &&
closer === opener.charAt(0).repeat(closer.length);
return !closes;
}

/**
* Warnings about markers that stop a README from being generated as its
* author intended, found before any section is written:
Expand All @@ -234,9 +237,8 @@ function isUnclosedFence(code: string): boolean {
* - a README with no marker for any section being generated, which the tool
* otherwise leaves unchanged in silence.
*
* A marker inside closed code is an example and is not reported. A marker
* inside a fence that nothing closes is reported, as `locateSection` would
* still find it.
* Markers are read as `locateSection` reads them: one inside closed code is
* an example and is not reported.
* @param {string} source - The document.
* @param {readonly string[]} sections - Every section name the tool knows.
* @param {readonly string[]} requested - The sections being generated.
Expand All @@ -247,9 +249,7 @@ export function diagnoseMarkers(
sections: readonly string[],
requested: readonly string[] = sections,
): string[] {
const examples = codeRanges(source).filter(
([from, to, fenced]) => !fenced || !isUnclosedFence(source.slice(from, to)),
);
const examples = exampleRanges(source);
const warnings: string[] = [];
for (const match of source.matchAll(/(?<![`\\])<!--\s+(start|end)\s+(\S+)\s+-->/g)) {
const name = match[2] ?? '';
Expand Down
Loading