From 76595dcde4ad3e54a9720f964d655efd04c9d288 Mon Sep 17 00:00:00 2001 From: Yarchik Date: Wed, 24 Jun 2026 00:06:57 +0100 Subject: [PATCH] fix: do not line-end-normalize a CR character reference in content XML 1.0 section 2.11 end-of-line normalization applies only to literal line breaks in the input, not to characters produced by a character or entity reference (sections 4.6/4.1). A character reference resolving to a carriage return (` ` or ` `) in element content was being converted to a line feed because consumeContentReference routed the resolved text through addText's unconditional normalizeLineBreaks call. Add a `normalize` parameter to addText (default true) and pass false from consumeContentReference so resolved references preserve U+000D. The CDATA and CharData callers keep the default, since literal content is subject to section 2.11. The attribute path was already correct (issue #6). --- src/lib/Parser.ts | 13 ++++++++++--- tests/lib/Parser.test.js | 18 ++++++++++++++++++ 2 files changed, 28 insertions(+), 3 deletions(-) diff --git a/src/lib/Parser.ts b/src/lib/Parser.ts index 8081cb8..8f4c429 100644 --- a/src/lib/Parser.ts +++ b/src/lib/Parser.ts @@ -65,12 +65,19 @@ export class Parser { /** * Adds the given _text_ to the document, either by appending it to a * preceding `XmlText` node (if possible) or by creating a new `XmlText` node. + * + * When _normalize_ is `true` (the default), line breaks in _text_ are + * normalized per section 2.11 of the XML spec. This must be `false` for text + * that comes from a character or entity reference, since references aren't + * subject to line break normalization. */ - addText(text: string, charIndex: number) { + addText(text: string, charIndex: number, normalize = true) { let { children } = this.currentNode; let { length } = children; - text = normalizeLineBreaks(text); + if (normalize) { + text = normalizeLineBreaks(text); + } if (length > 0) { let prevNode = children[length - 1]; @@ -297,7 +304,7 @@ export class Parser { let ref = this.consumeReference(); return ref - ? this.addText(ref, startIndex) + ? this.addText(ref, startIndex, false) : false; } diff --git a/tests/lib/Parser.test.js b/tests/lib/Parser.test.js index 1eddd91..d7f5f6e 100644 --- a/tests/lib/Parser.test.js +++ b/tests/lib/Parser.test.js @@ -390,6 +390,24 @@ describe('Parser', () => { assert.strictEqual(root.attributes.d, " a z "); }); + // A character reference for a carriage return (` ` / ` `) in content + // must be preserved as a literal `\r`. Line break normalization only applies + // to literal line breaks in the input, not to characters that come from a + // character reference. + // https://www.w3.org/TR/2008/REC-xml-20081126/#sec-line-ends + // https://www.w3.org/TR/2008/REC-xml-20081126/#entproc + it("doesn't normalize a character reference for a carriage return in content", () => { + assert.strictEqual(parseXml(' ').root.children[0].text, '\r'); + assert.strictEqual(parseXml(' ').root.children[0].text, '\r'); + + // A literal `\r` is normalized to `\n`, but a ` ` reference following + // it must remain a `\r`. + assert.strictEqual(parseXml('x\r y').root.children[0].text, 'x\n\ry'); + + // ` ` must remain `\r\n` rather than collapsing to `\n` or `\n\n`. + assert.strictEqual(parseXml(' ').root.children[0].text, '\r\n'); + }); + it('handles many character references in a single attribute', () => { let { root } = parseXml(''); assert.strictEqual(root.attributes.b, "<".repeat(35));