From 5ddd18c8611d561e751d0b0665c2ab7bd9b67dd4 Mon Sep 17 00:00:00 2001 From: baraline Date: Fri, 2 Oct 2026 17:01:18 +0200 Subject: [PATCH 01/16] test(content): port easyvista-python-client 0.4.0's display oracle MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit easyvista-python-client 0.4.0 carries this package's converter plus 15 reader fixes. Porting those fixes starts with the oracle their tests are written against, kept identical but for the package's names: - inside one line (a cell, a heading, a link), the edge of a block or of a nested table's cell is a word boundary, as a browser shows it and as the converter writes it (a space); the 0.6.0 oracle joined the words; - an ordered list's start is read with isdecimal, as the converter will read it, where isdigit made the oracle itself raise on "²"; - one_line, text_words and loose are new: the display a table inside a link owes, the displayed words without formatting, and a comparison loose on percent-encoded URLs. The existing content tests pass unchanged against it. Co-Authored-By: Claude Opus 5.5 --- glpi_python_client/content/tests/display.py | 109 +++++++++++++++++++- 1 file changed, 106 insertions(+), 3 deletions(-) diff --git a/glpi_python_client/content/tests/display.py b/glpi_python_client/content/tests/display.py index ef7f2c6..73c3c9c 100644 --- a/glpi_python_client/content/tests/display.py +++ b/glpi_python_client/content/tests/display.py @@ -9,12 +9,29 @@ A table's rows are its own: a table nested in a cell is read as words in that cell, as a browser shows it inside the cell. A table with no row displays nothing. + +Kept identical, but for the package's names, to the oracle of +``easyvista-python-client`` 0.4.0, whose converter is this one. That +package added :func:`one_line`, :func:`text_words` and :func:`loose`, and +changed two rules of the 0.6.0 oracle, each marked where it is: + +* inside one line -- a cell, a heading, a link -- the edge of a block or of + a nested table's cell is a word boundary, as a browser shows it (a new + line, a new box) and as the converter writes it (a space); and +* an ordered list's ``start`` is read with ``isdecimal``, as the converter + reads it, where ``isdigit`` made the oracle itself raise on ``"²"``. + +What the oracle cannot see: underline and highlight (````, ````, +````), struck text, link and image titles, and line breaks inside a +heading. """ from __future__ import annotations -from bs4 import ( - BeautifulSoup, +import urllib.parse + +from bs4 import BeautifulSoup +from bs4.element import ( Comment, Declaration, Doctype, @@ -38,6 +55,13 @@ } _SKIPPED = (Comment, Doctype, Declaration, ProcessingInstruction) +#: Since 0.6.1: inside one line -- a cell, a heading, a link -- the +#: edge of a block or of a nested table's cell separates words. A browser +#: starts a new line or draws a new box there, so ``ab`` +#: or ``

a

b

`` inside a cell never displays ``ab``; the 0.6.0 +#: oracle joined them, which failed the converter for writing ``a b``. +_WORD_EDGES = _BLOCKS | {"td", "th", "tr", "caption"} + Format = tuple[bool, bool, bool, str | None] #: No formatting, and not in a link. @@ -111,6 +135,10 @@ def _inline(node: Tag, words: _Words, fmt: Format) -> None: words.token(_image(child, fmt)) elif child.name == "a" and child.get("href"): _inline(child, words, (fmt[0], fmt[1], fmt[2], str(child.get("href")))) + elif child.name in _WORD_EDGES: + words.boundary() + _inline(child, words, _with_format(fmt, child.name)) + words.boundary() else: _inline(child, words, _with_format(fmt, child.name)) @@ -126,7 +154,9 @@ def _list(node: Tag, fmt: Format) -> tuple[object, ...]: ordered = node.name == "ol" start = str(node.get("start") or "1") - first = int(start) if start.isdigit() else 1 + # Since 0.6.1: isdecimal, as the converter reads it. isdigit + # accepts "²", and int("²") raises. + first = int(start) if start.isdecimal() else 1 items: list[object] = [] for position, item in enumerate(node.find_all("li", recursive=False)): content = _display_blocks(item, fmt) @@ -218,3 +248,76 @@ def displayed(html: str) -> tuple[object, ...]: """Return what a browser displays of ``html``, as comparable data.""" return _display_blocks(BeautifulSoup(html, "html.parser")) + + +def one_line(html: str) -> tuple[object, ...]: + """Return ``html`` read as one paragraph of inline content. + + For the one shape whose display the converter changes on purpose: a + table inside a link, which a browser draws as a table and the converter + writes as the link's text (Markdown has no table inside a link). What it + owes there is this: every word, in order, each with its formatting and + its link. + """ + + words = _Words() + _inline(BeautifulSoup(html, "html.parser"), words, _PLAIN) + paragraph = words.finish() + return (("p", paragraph),) if paragraph else () + + +#: Where a browser starts a new word whatever the text says: a block's +#: edges, a line break, a cell and an image. +_BREAKING = frozenset( + """ + address article aside blockquote body br caption center dd details dir div + dl dt fieldset figcaption figure footer form h1 h2 h3 h4 h5 h6 header hr + html img li main menu nav ol p pre section summary table tbody td tfoot th + thead tr ul + """.split() +) + + +def text_words(html: str) -> list[str]: + """Return the words ``html`` displays, in order, without their formatting. + + A blunt check that does not depend on :func:`displayed`. A + word ends only where a browser breaks the text -- see ``_BREAKING`` -- + never at an inline tag's edge, so ``foobar`` is one word, as it + displays. The walk keeps its own stack, so a deep body costs no + recursion. + """ + + soup = BeautifulSoup(html, "html.parser") + pieces: list[str] = [] + stack: list[object] = [soup] + while stack: + node = stack.pop() + if isinstance(node, Tag): + if node.name in _HIDDEN: + continue + edge = " " if node.name in _BREAKING else "" + pieces.append(edge) + stack.append(edge) # the closing edge, taken after the children + stack.extend(reversed(node.contents)) + elif isinstance(node, _SKIPPED): + continue + else: # a text node, or the str closing edge pushed above + pieces.append(str(node)) + return "".join(pieces).split() + + +def loose(display: object) -> object: + """Return ``display`` with every string percent-decoded. + + cmark-gfm percent-encodes a link or image target it renders -- a space + becomes ``%20``, a backslash ``%5C`` -- and a browser follows either + spelling to the same place. Comparing loosely lets a test hold the + converter to the display without holding it to one URL spelling. + """ + + if isinstance(display, str): + return urllib.parse.unquote(display) + if isinstance(display, tuple): + return tuple(loose(value) for value in display) + return display From a4a906eff4fa2732aefdd321d3aa7817e3bc54a3 Mon Sep 17 00:00:00 2001 From: baraline Date: Fri, 2 Oct 2026 17:07:52 +0200 Subject: [PATCH 02/16] fix(content): keep alt-text escapes, ~~~ lines, "!" and "<" as text Ports fixes 1 to 5 of easyvista-python-client 0.4.0's reader, which is this converter plus 15 fixes: 1. An escape or character reference in an image's alt text is kept. markdown-it-py 3 leaves it a text_special token, which mdformat rendered as nothing: alt "a_b_c.png" read back as "abc.png". 2. A paragraph line opening with ~~~ is escaped; CommonMark read it as a code fence, which swallowed the rest of the body. 3. A "!" ending a text right before a link is escaped; "Attention!" then a link read back as an image. 4. An alt text opening with "^" is escaped; cmark-gfm never opens an image on "![^". 5. A "<" mdformat left bare after an escaped one is escaped; "<" read back as "\<" and an autolink. Tests first: test_fixes.py sections 1 to 5, ported with the package's names; 13 of their 17 tests failed before the change (the other 4 are controls), all pass after it. Co-Authored-By: Claude Opus 5.5 --- glpi_python_client/content/conversion.py | 29 +++ .../content/tests/test_fixes.py | 179 ++++++++++++++++++ 2 files changed, 208 insertions(+) create mode 100644 glpi_python_client/content/tests/test_fixes.py diff --git a/glpi_python_client/content/conversion.py b/glpi_python_client/content/conversion.py index 0bc1ff0..a3eaf53 100644 --- a/glpi_python_client/content/conversion.py +++ b/glpi_python_client/content/conversion.py @@ -44,6 +44,7 @@ RenderContext, RenderTreeNode, ) +from mdformat.renderer.typing import Postprocess from glpi_python_client._errors import GlpiContentError @@ -316,6 +317,16 @@ class _Converter(MarkdownConverter): def _inherited(self, name: str) -> Any: return getattr(super(), name) + def escape(self, text: str, parent_tags: set[str]) -> str: + """Escape a text's last ``!`` as well: before a link it makes an image. + + A ``!`` meets a ``[`` only there, markdownify escaping a ``[`` in text; + mdformat keeps that escape before a link and drops it anywhere else. + """ + + text = str(self._inherited("escape")(text, parent_tags)) + return text[:-1] + "\\!" if text.endswith("!") else text + def _markup( self, el: Tag, text: str, parent_tags: set[str], markers: str, tag: str ) -> str: @@ -415,6 +426,8 @@ def convert_img(self, el: Tag, text: str, parent_tags: set[str]) -> str: alt = " ".join(str(el.get("alt") or "").split()) alt = str(self._inherited("escape")(alt, parent_tags)) + if alt.startswith("^") and "_noformat" not in parent_tags: + alt = "\\" + alt # cmark-gfm reads "![^" as "!" and a link src = _destination(str(el.get("src") or "")) return f"![{alt}]({src}{_title(str(el.get('title') or ''))})" @@ -473,6 +486,10 @@ def _list_item(node: RenderTreeNode, context: RenderContext) -> str: #: character no backslash escapes. ``2k - 1`` read as the same ``k`` there. _NEEDLESS_DOUBLING = re.compile(r"(?`` became ``\<``, an autolink. +_SECOND_LESS_THAN = re.compile(r"(?<=\\<)<(?=[^ ]|$)") + def _text(node: RenderTreeNode, context: RenderContext) -> str: """Render text as mdformat does, less one backslash where it escapes nothing. @@ -481,9 +498,14 @@ def _text(node: RenderTreeNode, context: RenderContext) -> str: """ text = DEFAULT_RENDERERS["text"](node, context) + text = _SECOND_LESS_THAN.sub(r"\\<", text) return _NEEDLESS_DOUBLING.sub(lambda found: found.group(1)[:-1], text) +#: A paragraph line opening with ``~~~``, which CommonMark reads as a fence. +_TILDE_FENCE = re.compile("^~~~", re.MULTILINE) + + class _Lists: """An mdformat extension: linear list items, short text and rules.""" @@ -492,6 +514,13 @@ class _Lists: "text": _text, "hr": lambda node, context: "---", } + #: What mdformat drops: an escape or entity in an image's alt text, which + #: markdown-it-py 3 leaves a ``text_special`` token mdformat renders as + #: nothing, and the escape on a paragraph line opening with ``~~~``. + POSTPROCESSORS: Mapping[str, Postprocess] = { + "text_special": lambda text, node, context: node.markup, + "paragraph": lambda text, node, context: _TILDE_FENCE.sub(r"\\~~~", text), + } class _Renderer(MDRenderer): diff --git a/glpi_python_client/content/tests/test_fixes.py b/glpi_python_client/content/tests/test_fixes.py new file mode 100644 index 0000000..296e879 --- /dev/null +++ b/glpi_python_client/content/tests/test_fixes.py @@ -0,0 +1,179 @@ +"""One regression test per reader fix ported from easyvista-python-client 0.4.0. + +That package's converter is this package's 0.6.0 converter (commit +``917f030``) plus fifteen fixes, each verified there on 2026-10-02 against +synthetic generators and against its own earlier test corpus. 0.6.1 ports +them, so that the two readers give the same Markdown. Each section below +pins one, numbered as in both packages' changelogs: + +1. an image's alt keeps its escapes and entities; +2. ``~~~`` opening a line stays text, not a code fence; +3. a ``!`` ending a text right before a link does not make it an image; +4. an alt opening with ``^`` stays an image; +5. a second ``<`` after an escaped one does not open an autolink. + +Every word is invented and every URL is under ``example.org``. +""" + +from __future__ import annotations + +import pytest +from bs4 import BeautifulSoup + +from glpi_python_client.content.conversion import GlpiContentConverter +from glpi_python_client.content.tests.test_round_trip import assert_survives + +read = GlpiContentConverter.from_transport +render = GlpiContentConverter.to_transport + + +def _alts(html: str) -> list[str]: + """The alt of every image, as the HTML parser reads it.""" + + soup = BeautifulSoup(html, "html.parser") + return [str(image.get("alt")) for image in soup.find_all("img")] + + +# --------------------------------------------------------------------------- +# 1. An image's alt keeps its escapes and entities +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize( + "alt", + [ + pytest.param("a_b C:\\Temp *x* [y] & `z`", id="syntax"), + pytest.param("R&amp;D &copy;", id="literal-entities"), + pytest.param("\\_ et \\*", id="backslashes"), + pytest.param("x < y", id="less-than"), + ], +) +def test_1_an_image_alt_keeps_its_escapes_and_entities(alt: str) -> None: + """markdown-it-py 3 leaves an escape or an entity in alt text as a + ``text_special`` token, which mdformat rendered as nothing.""" + + html = f'

{alt}

' + + markdown = assert_survives(html) + + assert _alts(render(markdown)) == _alts(html) + + +# --------------------------------------------------------------------------- +# 2. '~~~' opening a line stays text +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize( + "html", + [ + pytest.param("

~~~

", id="paragraph"), + pytest.param("

intro
~~~
suite

", id="after-a-break"), + pytest.param("
  • a
    ~~~~ b
", id="in-a-list-item"), + pytest.param("

a
~~~ x

", id="quoted"), + ], +) +def test_2_a_tilde_run_opening_a_line_stays_text(html: str) -> None: + """CommonMark reads ``~~~`` at a line start as a fence, which swallowed + the rest of the body; mdformat has no guard for it.""" + + markdown = assert_survives(html) + + assert "\\~~~" in markdown + + +def test_2_a_tilde_run_inside_a_line_is_not_escaped() -> None: + """The control: only a run that opens a line is a fence, so only it is escaped.""" + + assert assert_survives("

a ~~~ b

") == "a ~~~ b" + + +# --------------------------------------------------------------------------- +# 3. A trailing '!' before a link +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize( + "html", + [ + pytest.param( + '

Attention!voir

', id="link" + ), + pytest.param( + '

a!' + 'x

', + id="linked-image", + ), + pytest.param( # the control: "!" was never an image + '

a!https://example.org/u

', + id="autolink", + ), + ], +) +def test_3_a_bang_before_a_link_does_not_make_an_image(html: str) -> None: + """``!`` then ``[text](url)`` is an image in Markdown.""" + + assert_survives(html) + + +def test_3_only_a_trailing_bang_is_escaped() -> None: + """A ``!`` meets a ``[`` only at a text's end, so no other is escaped. + + A pin on the output, not a guard on the fix's narrow form. mdformat + re-renders from the syntax tree and drops an escape that changes + nothing, so escaping every ``!`` gives this same Markdown: a review on + 2026-10-02 found no body, among 3,414 from this suite's generators, on + which the two forms differ. The narrow form was kept as the smaller + change. + """ + + assert assert_survives("

Hi! there! ok

") == "Hi! there! ok" + + +# --------------------------------------------------------------------------- +# 4. An alt opening with '^' +# --------------------------------------------------------------------------- + + +def test_4_an_alt_opening_with_a_caret_stays_an_image() -> None: + """cmark-gfm never opens an image on ``![^``, the footnote syntax.""" + + html = '

^x fin

' + + markdown = assert_survives(html) + + assert markdown.startswith("![\\^x]") + + +def test_4_in_code_the_caret_is_not_escaped() -> None: + """Inside code nothing is syntax, so a backslash there would show. + + This guards the fix's correction, not the fix. + """ + + markdown = read( + '

x^x

' + ) + + assert "\\^" not in markdown + + +# --------------------------------------------------------------------------- +# 5. A second '<' does not open an autolink +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize( + "html", + [ + pytest.param("

<<a@b.c> et <<b> c

", id="address"), + pytest.param("

<<https://example.org>

", id="url"), + ], +) +def test_5_a_second_less_than_stays_text(html: str) -> None: + """mdformat's escape pattern ate the character after an escaped ``<``, + so ``<`` became ``\\<``: an autolink.""" + + markdown = assert_survives(html) + + assert "\\<\\<" in markdown From c97a47ed2f901c5997f0fb7d5e58ca8872ee5d14 Mon Sep 17 00:00:00 2001 From: baraline Date: Fri, 2 Oct 2026 17:09:12 +0200 Subject: [PATCH 03/16] fix(content): read inline-code breaks, nested tables,
and Ports fixes 6 to 11 of easyvista-python-client 0.4.0's reader: 6. A
inside , or splits the span: a line break in a paragraph, a
in a cell, a space in a heading. It read back as a literal backslash inside one span. A "|" in or in a cell is escaped, as one in already was. 7. A table inside a heading or a link is written as its cells' text, as one inside a cell already was; its pipes showed as text, and inside a link the reading was not a fixed point. 8. A block inside such a flattened table stays on its holder's line. 9. A flattened cell's edge counts as a space when deciding whether "**" closes, so bold ending on punctuation there stays Markdown bold. 10.
is a block, as
is, except inside
; two 
blocks ran together. 11. , and are kept as raw tags; the formatting was dropped. Tests first: test_fixes.py sections 6 to 11; 18 of their 21 tests failed before the change (the other 3 are controls or guard a correction to a fix), all pass after it. Co-Authored-By: Claude Opus 5.5 --- glpi_python_client/content/conversion.py | 59 ++++-- .../content/tests/test_fixes.py | 200 +++++++++++++++++- 2 files changed, 246 insertions(+), 13 deletions(-) diff --git a/glpi_python_client/content/conversion.py b/glpi_python_client/content/conversion.py index a3eaf53..aa169c9 100644 --- a/glpi_python_client/content/conversion.py +++ b/glpi_python_client/content/conversion.py @@ -167,15 +167,19 @@ def _soup(html: str) -> BeautifulSoup: {"table", "thead", "tbody", "tfoot", "tr", "caption", "colgroup", "col"} ) +#: What Markdown writes on one line: a table inside one cannot be a table. +_ONE_LINE = frozenset({"td", "th", "a", "h1", "h2", "h3", "h4", "h5", "h6"}) + def _flatten_nested_tables(root: Tag) -> None: - """Write a table that sits inside a cell as its cells' text. + """Write a table that sits inside a cell, a heading or a link as its cells' text. A GFM cell holds one line, so a nested table -- the usual layout of an e-mail signature -- written as a table split the outer row and lost - every word. Inside a cell, a table's parts become ``span`` and its cells - ``glpi-cell``, which :class:`_Converter` writes as spaced inline text. - Renaming in one walk keeps it linear however deeply tables nest. + every word; in a heading or a link its pipes became text. Inside one, a + table's parts become ``span`` and its cells ``glpi-cell``, which + :class:`_Converter` writes as spaced inline text. Renaming in one walk + keeps it linear however deeply tables nest. """ cells = 0 @@ -192,8 +196,9 @@ def _flatten_nested_tables(root: Tag) -> None: node.name = "glpi-cell" if is_cell else "span" counted.append(False) continue - counted.append(is_cell) - cells += is_cell + holds = node.name in _ONE_LINE and (node.name != "a" or bool(node.get("href"))) + counted.append(holds) + cells += holds def _walk(root: Tag) -> Iterator[tuple[PageElement, bool]]: @@ -246,7 +251,8 @@ def _note_sides(root: Tag) -> None: One pass: ``last`` is the last character shown so far on the current line, and an element that has closed waits for the next one. A line edge - counts as a space, as it does for CommonMark. + counts as a space, as it does for CommonMark, and so does the edge of a + flattened cell, which :class:`_Converter` spaces. """ last = " " @@ -259,7 +265,7 @@ def _note_sides(root: Tag) -> None: else: waiting.append(node) continue - if node.name not in _BLOCKS and node.name != "br": + if node.name not in _BLOCKS and node.name not in ("br", "glpi-cell"): continue first = last = " " else: @@ -327,6 +333,11 @@ def escape(self, text: str, parent_tags: set[str]) -> str: text = str(self._inherited("escape")(text, parent_tags)) return text[:-1] + "\\!" if text.endswith("!") else text + def process_tag(self, node: Any, parent_tags: Any = None) -> str: + if node.name == "glpi-cell": # one line, as a cell is: its blocks are inline + parent_tags = {*(parent_tags or ()), "_inline"} + return str(self._inherited("process_tag")(node, parent_tags)) + def _markup( self, el: Tag, text: str, parent_tags: set[str], markers: str, tag: str ) -> str: @@ -364,26 +375,50 @@ def convert_s(self, el: Tag, text: str, parent_tags: set[str]) -> str: convert_del = convert_s convert_strike = convert_s + def convert_u(self, el: Tag, text: str, parent_tags: set[str]) -> str: + """Underlined or highlighted text: CommonMark has neither, so raw HTML.""" + + return self._markup(el, text, parent_tags, "", el.name) + + convert_mark = convert_ins = convert_u + + def convert_center(self, el: Tag, text: str, parent_tags: set[str]) -> str: + """``
``, obsolete, is a block, as ``
`` is.""" + + if "pre" in parent_tags: + return text + return str(self._inherited("convert_div")(el, text, parent_tags)) + def convert_glpi_cell(self, el: Tag, text: str, parent_tags: set[str]) -> str: """A cell of a table nested in a cell: its text, set apart by spaces.""" return f" {text.strip()} " def convert_br(self, el: Tag, text: str, parent_tags: set[str]) -> str: - if "pre" in parent_tags: - return "\n" + if "_noformat" in parent_tags: + return "\n" # in
, or in inline code, which convert_code splits
         if "td" in parent_tags or "th" in parent_tags:
             return "
" # a cell is one line of Markdown; cmark-gfm passes the tag return str(self._inherited("convert_br")(el, text, parent_tags)) def convert_code(self, el: Tag, text: str, parent_tags: set[str]) -> str: - code = str(self._inherited("convert_code")(el, text, parent_tags)) - if "td" in parent_tags or "th" in parent_tags: + """Inline code, a span per line: a code span cannot hold a line break.""" + + cell = "td" in parent_tags or "th" in parent_tags + lines = [text] if "_noformat" in parent_tags else text.split("\n") + join = "
" if cell else " " if "_inline" in parent_tags else "\\\n" + code = join.join( + str(self._inherited("convert_code")(el, line, parent_tags)) + for line in lines + ) + if cell: code = code.replace( "|", "\\|" ) # GFM splits a row on "|" before it reads code return code + convert_kbd = convert_samp = convert_code + def convert_list(self, el: Tag, text: str, parent_tags: set[str]) -> str: """A list; a nested one ends in a blank line. diff --git a/glpi_python_client/content/tests/test_fixes.py b/glpi_python_client/content/tests/test_fixes.py index 296e879..aed8226 100644 --- a/glpi_python_client/content/tests/test_fixes.py +++ b/glpi_python_client/content/tests/test_fixes.py @@ -10,7 +10,16 @@ 2. ``~~~`` opening a line stays text, not a code fence; 3. a ``!`` ending a text right before a link does not make it an image; 4. an alt opening with ``^`` stays an image; -5. a second ``<`` after an escaped one does not open an autolink. +5. a second ``<`` after an escaped one does not open an autolink; +6. a line break in inline code (``code``, ``kbd``, ``samp``) is kept in a + paragraph and a cell, and is a space in a heading, which is one line; + and a ``|`` in one of them in a cell stays in the cell; +7. a table inside a heading or a link is written as its cells' text, as + one inside a cell already was; +8. a block in such a table stays on its holder's line; +9. bold at a flattened cell's edge still closes; +10. ``
`` is a block, except inside ``
``;
+11. ````, ```` and ```` stay raw HTML.
 
 Every word is invented and every URL is under ``example.org``.
 """
@@ -21,6 +30,7 @@
 from bs4 import BeautifulSoup
 
 from glpi_python_client.content.conversion import GlpiContentConverter
+from glpi_python_client.content.tests.display import displayed, one_line
 from glpi_python_client.content.tests.test_round_trip import assert_survives
 
 read = GlpiContentConverter.from_transport
@@ -177,3 +187,191 @@ def test_5_a_second_less_than_stays_text(html: str) -> None:
     markdown = assert_survives(html)
 
     assert "\\<\\<" in markdown
+
+
+# ---------------------------------------------------------------------------
+# 6. A line break in inline code
+# ---------------------------------------------------------------------------
+
+
+@pytest.mark.parametrize(
+    ("html", "expected"),
+    [
+        pytest.param(
+            "

Sortie : ligne un
ligne deux
fin

", + "Sortie : `ligne un`\\\n`ligne deux` fin", + id="paragraph", + ), + pytest.param( + "

x ligne un
ligne deux
y

", + "x `ligne un`\\\n`ligne deux` y", + id="space-after-the-break", + ), + pytest.param( + "" + "
H
a un
deux
b
", + "| H |\n| -- |\n| a `un`
`deux` b |", + id="cell", + ), + ], +) +def test_6_a_line_break_in_inline_code_splits_the_span( + html: str, expected: str +) -> None: + """A code span cannot hold a line break, so the span ends and resumes.""" + + assert assert_survives(html) == expected + + +def test_6_a_line_break_in_inline_code_in_a_heading_is_a_space() -> None: + """A heading is one line in Markdown: the break becomes a space.""" + + markdown = read("

a un
deux
b

") + + assert markdown == "## a `un` `deux` b" + assert displayed(render(markdown)) == displayed( + "

a un deux b

" + ) + + +@pytest.mark.parametrize("tag", ["code", "kbd", "samp"]) +def test_6_a_pipe_in_inline_code_in_a_cell_stays_in_the_cell(tag: str) -> None: + """GFM splits a row on ``|`` before it reads code. ``code`` is the control: + 0.6.0 escaped it there already, but not in ``kbd`` or ``samp``.""" + + html = f"
H
<{tag}>a|b fin
" + + markdown = assert_survives(html) + + assert "`a\\|b` fin" in markdown + + +# --------------------------------------------------------------------------- +# 7. A table inside a heading or a link +# --------------------------------------------------------------------------- + + +def test_7_a_table_inside_a_heading_is_its_cells_text() -> None: + markdown = assert_survives( + "

Titre
ab

" + ) + + assert markdown == "## Titre a b" + + +def test_7_a_table_inside_a_link_is_the_links_text() -> None: + """A browser draws the table inside the link; Markdown cannot, so every + word is kept, in order, still linked.""" + + html = ( + '

avant ' + "
undeux
apres

" + ) + + markdown = read(html) + + assert markdown == "avant [un deux](https://example.org/u) apres" + assert displayed(render(markdown)) == one_line(html) + assert read(render(markdown)) == markdown + + +def test_7_a_table_inside_a_cell_is_its_cells_text() -> None: + """The control: 0.6.0 already flattened a table nested in a cell.""" + + markdown = assert_survives( + "
A
" + "
undeux
" + ) + + assert markdown == "| A |\n| -- |\n| un deux |" + + +# --------------------------------------------------------------------------- +# 8. A block in a flattened table stays on its holder's line +# --------------------------------------------------------------------------- + + +def test_8_blocks_in_a_table_inside_a_link_stay_on_the_links_line() -> None: + html = ( + '

avant

un

deux

' + "
z
apres

" + ) + + markdown = read(html) + + assert markdown == "avant [un deux z](https://example.org/u) apres" + assert displayed(render(markdown)) == one_line(html) + + +# --------------------------------------------------------------------------- +# 9. Bold at a flattened cell's edge still closes +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize( + "cells", + [ + pytest.param("xgras.y", id="ends-on-a-dot"), + pytest.param("(gras)y", id="parenthesised"), + ], +) +def test_9_bold_at_a_flattened_cells_edge_is_markdown_bold(cells: str) -> None: + """The converter spaces a flattened cell, so its edge counts as a space. + 0.6.0 counted the next cell's text and wrote raw ````, + which read back as ``**`` and was not a fixed point.""" + + markdown = assert_survives( + f"" + "
A
{cells}
" + ) + + assert "" not in markdown + + +# --------------------------------------------------------------------------- +# 10.
is a block, except inside
+# ---------------------------------------------------------------------------
+
+
+@pytest.mark.parametrize(
+    ("html", "expected"),
+    [
+        pytest.param("
Titre
suite", "Titre\n\nsuite", id="alone"), + pytest.param("

a

b
c

", "a\n\nb\n\nc", id="in-a-line"), + ], +) +def test_10_center_is_a_block(html: str, expected: str) -> None: + assert assert_survives(html) == expected + + +def test_10_center_inside_pre_is_left_alone() -> None: + """The correction to the fix, not the fix: as a block there it split the code.""" + + markdown = assert_survives("
un\n
deux
\ntrois
") + + assert markdown == "```\nun\ndeux\ntrois\n```" + + +# --------------------------------------------------------------------------- +# 11. , and stay raw HTML +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize("tag", ["u", "mark", "ins"]) +def test_11_underline_and_highlight_stay_raw_html(tag: str) -> None: + """CommonMark has neither; 0.6.0 dropped the formatting.""" + + markdown = assert_survives(f"

a <{tag}>trilo fin

") + + assert markdown == f"a <{tag}>trilo fin" + assert f"<{tag}>trilo" in render(markdown) + + +def test_11_the_edges_move_outside_the_tag() -> None: + assert read("

a trilo b

") == "a trilo b" + + +def test_11_underline_inside_a_link() -> None: + markdown = assert_survives('

lien

') + + assert markdown == "[lien](https://example.org/u)" From c2c1d246256dc0f43863ac136e6d01e9801262f7 Mon Sep 17 00:00:00 2001 From: baraline Date: Fri, 2 Oct 2026 17:25:16 +0200 Subject: [PATCH 04/16] fix(content): number long lists and split edges in linear time, degrade bad numbers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ports fixes 12 to 15 of easyvista-python-client 0.4.0's reader: 12. Ordered items are numbered in linear time: each item takes its number from the item before it, where markdownify counted every earlier sibling. start is read with isdecimal, so start="²" or "½" counts from 1 as a browser does instead of failing the body. 13. The pattern that splits emphasis and links into edges and content is greedy, so linear on runs of spaces or
; the lazy one was quadratic. 14. A colspan or start markdownify cannot read as a number ("²", or more digits than CPython converts) degrades the body to its text, as a body too deep for the stack does, instead of raising GlpiContentError. 15. Every "<" after a body's last ">" is read as text. CPython's html.parser before 3.11.14, 3.12.12 and 3.13.6 is quadratic on a run of unfinished tags there (CVE-2025-6069); patched releases dropped the unfinished tag, so "x --- glpi_python_client/content/conversion.py | 35 ++- .../content/tests/test_conversion.py | 47 +++- glpi_python_client/content/tests/test_cost.py | 202 ++++++++++++++++++ .../content/tests/test_fixes.py | 188 +++++++++++++++- .../content/tests/test_round_trip.py | 36 ---- 5 files changed, 466 insertions(+), 42 deletions(-) create mode 100644 glpi_python_client/content/tests/test_cost.py diff --git a/glpi_python_client/content/conversion.py b/glpi_python_client/content/conversion.py index aa169c9..f77530f 100644 --- a/glpi_python_client/content/conversion.py +++ b/glpi_python_client/content/conversion.py @@ -91,9 +91,14 @@ #: carry the characters displayed on either side of one (:func:`_note_sides`). _EMPHASIS = frozenset({"b", "strong", "em", "i"}) _BEFORE, _AFTER = "data-glpi-before", "data-glpi-after" +_NUMBER = "data-glpi-number" # an ordered item's number (_Converter.convert_li) #: Text split into leading line breaks and spaces, content, trailing ones. -_EDGES = re.compile(r"((?:\\\n|\s)*)(.*?)((?:\\\n|\s)*)", re.DOTALL) +#: The content ends on its last character that is neither, found greedily: +#: a lazy ``.*?`` rescanned the run after it at every step, quadratic. +_EDGES = re.compile( + r"((?:\\\n|\s)*)((?:.*(?:[^\\\s]|\\(?!\n)))?)((?:\\\n|\s)*)", re.DOTALL +) #: A URL that is its own CommonMark autolink and that markdown-it leaves as it #: is: printable ASCII it does not percent-encode. @@ -431,6 +436,27 @@ def convert_list(self, el: Tag, text: str, parent_tags: set[str]) -> str: convert_ul = convert_list convert_ol = convert_list + def convert_li(self, el: Tag, text: str, parent_tags: set[str]) -> str: + """An item; an ordered one numbered one past the item before it. + + markdownify counts every item before each one, quadratic in a long + list, so each item keeps its number for the next. ``isdecimal``, + where markdownify's ``isnumeric`` let ``int("²")`` raise. + """ + + if el.parent is None or el.parent.name != "ol": + return str(self._inherited("convert_li")(el, text, parent_tags)) + before = el.find_previous_sibling("li") + start = str(el.parent.get("start") or "") + first = int(start) if start.isdecimal() else 1 + number = int(str(before[_NUMBER])) + 1 if isinstance(before, Tag) else first + el[_NUMBER] = str(number) + if not text.strip(): + return "\n" + bullet = f"{number}. " + body = re.sub("^(?=.)", " " * len(bullet), text.strip(), flags=re.M) + return bullet + body[len(bullet) :] + "\n" + def convert_pre(self, el: Tag, text: str, parent_tags: set[str]) -> str: """A fenced block, its fence longer than any backtick run in the code.""" @@ -701,10 +727,15 @@ def from_transport(value: object, *, plain_text_is_markdown: bool = False) -> st return content if not _looks_like_html(content): content = _plain_text_html(content) + # No "<" after the last ">" can finish a tag; CPython's html.parser before + # 3.11.14/3.12.12/3.13.6 rescans to the end for each (quadratic). + head, end, tail = content.rpartition(">") + content = head + end + tail.replace("<", "<") try: try: return html_to_markdown(content) - except RecursionError: + except (RecursionError, ValueError): # ValueError: markdownify's + # int() of a colspan or start such as "²" or 5,000 digits return html_to_markdown(_plain_text_html(_text_of(content))) except ParserRejectedMarkup: # html.parser gives up on a few malformed declarations; the diff --git a/glpi_python_client/content/tests/test_conversion.py b/glpi_python_client/content/tests/test_conversion.py index 85cc58e..438841d 100644 --- a/glpi_python_client/content/tests/test_conversion.py +++ b/glpi_python_client/content/tests/test_conversion.py @@ -118,8 +118,12 @@ def rejecting(markup: str) -> BeautifulSoup: def test_an_inbound_fault_surfaces_as_a_glpi_error( monkeypatch: pytest.MonkeyPatch, ) -> None: + """The fault is a ``RuntimeError``: since fix 14 a ``ValueError`` goes to + the text fallback first (:func:`test_a_value_error_is_retried_as_text_first`), + and a ``RuntimeError`` reaches the error path directly.""" + def failing(html: str) -> str: - raise ValueError("converter fault") + raise RuntimeError("converter fault") monkeypatch.setattr(conversion, "html_to_markdown", failing) @@ -127,6 +131,46 @@ def failing(html: str) -> str: read("

x

") assert isinstance(caught.value, GlpiError) + assert isinstance(caught.value.__cause__, RuntimeError) + assert "Could not convert GLPI HTML content to Markdown" in str(caught.value) + + +def test_a_value_error_is_retried_as_text_first( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """Fix 14: a ``ValueError`` is answered like a ``RecursionError``, by the text. + + markdownify calls ``int()`` on a ``colspan`` or a ``start``, which raises + on ``"²"`` or on more digits than CPython converts. The body is read again + as its text; only if that fails too is the error reported. + """ + + calls: list[str] = [] + convert = conversion.html_to_markdown + + def failing_once(html: str) -> str: + calls.append(html) + if len(calls) == 1: + raise ValueError("invalid literal for int()") + return convert(html) + + monkeypatch.setattr(conversion, "html_to_markdown", failing_once) + + assert read("

un deux

") == "un deux" + assert len(calls) == 2 + + +def test_a_value_error_on_the_text_path_too_is_a_glpi_error( + monkeypatch: pytest.MonkeyPatch, +) -> None: + def failing(html: str) -> str: + raise ValueError("converter fault") + + monkeypatch.setattr(conversion, "html_to_markdown", failing) + + with pytest.raises(GlpiContentError) as caught: + read("

x

") + assert isinstance(caught.value.__cause__, ValueError) @@ -142,6 +186,7 @@ def failing(markdown: str) -> str: render("**x**") assert isinstance(caught.value.__cause__, RuntimeError) + assert "Could not render Markdown content as GLPI HTML" in str(caught.value) def test_deeply_nested_markdown_renders() -> None: diff --git a/glpi_python_client/content/tests/test_cost.py b/glpi_python_client/content/tests/test_cost.py new file mode 100644 index 0000000..0fb408f --- /dev/null +++ b/glpi_python_client/content/tests/test_cost.py @@ -0,0 +1,202 @@ +"""What a long or a deep body costs: growth, not seconds. + +A converter that is quadratic somewhere converts an ordinary body in +milliseconds and a pathological one in minutes, so the cost tests measure +how the time grows: converting a body four times as long should take about +four times as long. A quadratic pass takes about sixteen times as long. The +tests assert ``time(4n) / time(n) < 8``, halfway between the two on a log +scale. + +An absolute budget would be the wrong instrument. Up to 0.6.0 this +package's tests gave each long body 20 seconds. That is generous enough +for a shared CI runner, so it catches a quadratic pass only once a body is +long enough to cost 20 seconds, and the tests then spend most of their +time converting. A ratio needs far shorter bodies -- a few thousand items +where those tests had 20,000 -- and the runner's speed cancels out. + +Ported from easyvista-python-client 0.4.0, whose converter is this one. +The shapes are 0.6.0's four long bodies, in this form; three bodies dense +with syntax from that package's earlier tests; and one per fix that +removed a quadratic. Measured 2026-10-02, timed the way :func:`growth` +times, at these sizes, three readings per shape: this converter grew by +3.0 to 7.5 on CPython 3.12.3 and by 2.6 to 5.7 on 3.13.14. On 3.12.3, +0.6.0 grew by 10.3 and 10.4 on the ordered list, 14.2 and 15.5 on bold +holding line breaks and 30.9 and 34.4 on the unfinished-tag tail (two +readings each), and reverting fix 12, 13 or 15 alone grew by 10.4, 14.1 +and 33.2 on its shape (one reading each). + +A ratio between 8 and 10 is measured once more and the second reading +decides, so that one burst of load cannot fail a linear pass; a reading of +10 or more fails at once. +""" + +from __future__ import annotations + +import gc +import math +import time +from collections.abc import Callable + +import pytest + +from glpi_python_client.content.conversion import GlpiContentConverter + +read = GlpiContentConverter.from_transport +render = GlpiContentConverter.to_transport + +#: ``time(4n) / time(n)`` at or above this fails: linear reads about 4, +#: quadratic about 16. +_LIMIT = 8 + +#: The shortest reading worth trusting: a body that converts faster is read +#: again until the clock has run this long, and the time is the average. +_SHORTEST = 0.02 + + +def _timed(html: str) -> float: + """Seconds to read ``html`` once, with the cyclic collector out of the way. + + A collection starting mid-conversion is charged to whichever body + happens to be converting, so the heap is collected before the clock + starts and the collector stays off until it stops. A body read in a few + milliseconds is read again until :data:`_SHORTEST` has passed, so that + the scheduler's tick does not decide the ratio. + """ + + gc.collect() + gc.disable() + try: + reads = 0 + started = time.perf_counter() + while True: + read(html) + reads += 1 + took = time.perf_counter() - started + if took >= _SHORTEST: + return took / reads + finally: + gc.enable() + + +def growth(make: Callable[[int], str], n: int, rounds: int = 3) -> float: + """Return ``time(make(4n)) / time(make(n))``, each the best of ``rounds``. + + The two sizes alternate, so a burst of load on a shared machine slows + both rather than one; the best of several runs drops the runs it hit. + """ + + small, large = make(n), make(4 * n) + best_small = best_large = math.inf + for _ in range(rounds): + best_small = min(best_small, _timed(small)) + best_large = min(best_large, _timed(large)) + return best_large / best_small + + +def assert_linear(make: Callable[[int], str], n: int) -> None: + ratio = growth(make, n) + if _LIMIT <= ratio < 1.25 * _LIMIT: # near the line: see the module docstring + ratio = growth(make, n) + assert ratio < _LIMIT, ( + f"converting a body four times as long took {ratio:.1f} times as long" + ) + + +@pytest.mark.parametrize( + ("make", "n"), + [ + pytest.param( + lambda n: "

" + "ligne
" * n + "

", 1500, id="line-breaks" + ), + pytest.param( + lambda n: "
    " + "
  • x
  • " * n + "
", 1000, id="list-items" + ), + pytest.param( + lambda n: "
    " + "
  • " * n + "
", 2000, id="empty-items" + ), + pytest.param( + lambda n: "

" + "gras mot " * n + "

", 750, id="emphasis" + ), + ], +) +def test_a_long_body_converts_in_linear_time( + make: Callable[[int], str], n: int +) -> None: + """0.6.0's long bodies: each pass mdformat makes is linear.""" + + assert_linear(make, n) + + +@pytest.mark.parametrize( + ("make", "n"), + [ + pytest.param( + lambda n: "

" + "[a " * n + "

", 2500, id="unclosed-brackets" + ), + pytest.param( + lambda n: "

" + "_a " * n + "

", 2500, id="underscores-opening-words" + ), + pytest.param(lambda n: "

" + "``` " * n + "

", 2000, id="backtick-runs"), + ], +) +def test_a_body_dense_with_syntax_converts_in_linear_time( + make: Callable[[int], str], n: int +) -> None: + """One paragraph packed with what CommonMark scans for, each escaped. + + From easyvista-python-client's earlier tests, where 20,000 of each + took from 8 to 117 seconds before the passes of that package's + converter of the day were made linear. This converter is not quite + linear on them either: that package measured the ratio at about 4.3 at + these sizes and about 7 at eight times them (2026-10-02, CPython + 3.12.11), a superlinear term it traced to markdown-it-py 3.0 joining + the text of one long escape-dense line. So the sizes are kept small: + the test catches a pass that turns quadratic, not that term. + """ + + assert_linear(make, n) + + +def test_a_long_ordered_list_converts_in_linear_time() -> None: + """Fix 12: each item takes its number from the item before it. + + markdownify counted every sibling before each item, which was quadratic. + On 5,000 items 0.6.0 took 3.9 seconds against 0.8 with the fix + (measured 2026-10-02, CPython 3.12.3, best of three). The list is + written one item per line, as an editor writes it: each newline is one + more sibling to count, which makes the quadratic easier to see. + """ + + assert_linear(lambda n: "
    \n" + "
  1. x
  2. \n" * n + "
", 1250) + + +def test_bold_holding_a_long_run_of_line_breaks_converts_in_linear_time() -> None: + """Fix 13: the regex that moves an element's edge breaks outside it is greedy. + + The lazy one rescanned the run after it at every step. + """ + + assert_linear(lambda n: "

a" + "
\n" * n + "b

", 2000) + + +def test_a_tail_of_unfinished_tags_converts_in_linear_time() -> None: + """Fix 15: no ``<`` after the last ``>`` reaches the parser as a ``<``. + + CPython's ``html.parser`` before 3.11.14, 3.12.12 and 3.13.6 rescanned + to the end of the input for each one (CVE-2025-6069). On a patched + interpreter the parser is linear anyway, so there this test cannot see + the guard go; on an unpatched one it fails within seconds without it. + """ + + assert_linear(lambda n: "

r

" + "x
None: + """markdownify recurses per nesting level; past the stack, the text is kept.""" + + html = "
" * 3000 + "

__init__ au fond

" + "
" * 3000 + + markdown = read(html) + + assert "init" in markdown + assert "au fond" in render(markdown) diff --git a/glpi_python_client/content/tests/test_fixes.py b/glpi_python_client/content/tests/test_fixes.py index aed8226..0e711c6 100644 --- a/glpi_python_client/content/tests/test_fixes.py +++ b/glpi_python_client/content/tests/test_fixes.py @@ -19,18 +19,37 @@ 8. a block in such a table stays on its holder's line; 9. bold at a flattened cell's edge still closes; 10. ``
`` is a block, except inside ``
``;
-11. ````, ```` and ```` stay raw HTML.
-
+11. ````, ```` and ```` stay raw HTML;
+12. an ordered list is numbered in linear time, and a ``start`` that is
+    not a decimal number counts from 1;
+13. the regex splitting a text's edges is greedy, so linear;
+14. a number markdownify cannot read sends the body to the text fallback;
+15. an unfinished tag at the very end is read as text (the CVE-2025-6069
+    guard).
+
+The cost side of 12, 13 and 15 is tested in :mod:`.test_cost`. Run against
+0.6.0 on 2026-10-02 (CPython 3.12.3), 36 of these 54 tests failed. Of the
+18 that passed there, 17 are controls, guards on a correction to a fix,
+pins of output a fix leaves as it was, or behaviour a cost fix had to keep,
+and each says which. The other is test 15's ``attribute`` case, which
+passed only because 3.12.3 predates CPython's own fix; that test says why.
 Every word is invented and every URL is under ``example.org``.
 """
 
 from __future__ import annotations
 
+import sys
+
 import pytest
 from bs4 import BeautifulSoup
 
+from glpi_python_client.content import conversion
 from glpi_python_client.content.conversion import GlpiContentConverter
-from glpi_python_client.content.tests.display import displayed, one_line
+from glpi_python_client.content.tests.display import (
+    displayed,
+    one_line,
+    text_words,
+)
 from glpi_python_client.content.tests.test_round_trip import assert_survives
 
 read = GlpiContentConverter.from_transport
@@ -375,3 +394,166 @@ def test_11_underline_inside_a_link() -> None:
     markdown = assert_survives('

lien

') assert markdown == "[lien](https://example.org/u)" + + +# --------------------------------------------------------------------------- +# 12. Ordered-list numbering +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize( + ("html", "expected"), + [ + pytest.param( + '
  1. a
  2. b
  3. c
    1. d
', + "3. a\n4. b\n5. c\n 1. d", + id="start-and-a-nested-list", + ), + pytest.param( + '
  1. a
  2. b
', "0. a\n1. b", id="start-zero" + ), + pytest.param('
  1. a
  2. b
', "1. a\n2. b", id="½"), + pytest.param('
  1. a
  2. b
', "1. a\n2. b", id="²"), + ], +) +def test_12_an_ordered_item_is_numbered_one_past_the_item_before( + html: str, expected: str +) -> None: + """Each item keeps its number for the next, where markdownify counted + every item before each one. ``isdecimal``, where markdownify's + ``isnumeric`` let ``int("½")`` raise -- a browser counts such a list + from 1, and so does the converter now. + + With a decimal ``start``, 0.6.0 numbered the same: what the fix + changed there is the cost (:mod:`.test_cost`), and the first two cases + pin that the rewrite numbers as markdownify did. + """ + + assert assert_survives(html) == expected + + +def test_12_an_empty_ordered_item_still_counts_for_the_next() -> None: + """An empty item takes a number and writes nothing. + + Markdown has no empty ordered item, so the item cannot survive: the + browser shows ``3. a`` and ``5. c``, and the Markdown, renumbered by + mdformat, shows ``3. a`` and ``4. c``. The words and the fixed point + are kept. 0.6.0 wrote the same; this pins that the rewrite still + does, and no other test reaches its branch for an empty ordered item. + """ + + html = '
  1. a
  2. c
' + + markdown = read(html) + + assert markdown == "3. a\n4. c" + assert text_words(render(markdown)) == text_words(html) + assert read(render(markdown)) == markdown + + +# --------------------------------------------------------------------------- +# 13. The edge regex is greedy +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize( + ("text", "groups"), + [ + pytest.param(" a b ", (" ", "a b", " "), id="spaces"), + pytest.param("\\\n a\\\n", ("\\\n ", "a", "\\\n"), id="hard-breaks"), + pytest.param("a\\b", ("", "a\\b", ""), id="inner-backslash"), + pytest.param("x\\", ("", "x\\", ""), id="trailing-backslash"), + pytest.param(" \n ", (" \n ", "", ""), id="nothing-inside"), + pytest.param("", ("", "", ""), id="empty"), + ], +) +def test_13_the_edge_regex_splits_leading_content_and_trailing( + text: str, groups: tuple[str, str, str] +) -> None: + """Leading breaks and spaces, the content, trailing ones. The greedy + pattern finds the content's last character from the end, where the + lazy one rescanned the run after it at every step. The lazy one split + the same way, only slower: these pin that the rewrite still does, and + :mod:`.test_cost` pins the cost.""" + + edges = conversion._EDGES.fullmatch(text) + + assert edges is not None + assert edges.groups() == groups + + +# --------------------------------------------------------------------------- +# 14. A number markdownify cannot read +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize( + "html", + [ + pytest.param( + '
trelmvosk
', + id="colspan-²", + ), + pytest.param( + '
  1. trelm
  2. vosk
', + id="start-of-5000-digits", + ), + ], +) +def test_14_a_number_markdownify_cannot_read_falls_back_to_text(html: str) -> None: + """markdownify's ``int()`` raised ``ValueError`` -- on ``"²"``, or on more + digits than CPython converts -- and the whole body failed. Now the body + is read as its text: every word, one line per row or item. + + The digit limit is pinned to CPython's default of 4,300: the + interpreter takes another from ``PYTHONINTMAXSTRDIGITS``, and ``0`` + lifts it, under which 5,000 digits are a number and the case tests + nothing. + """ + + limit = sys.get_int_max_str_digits() + sys.set_int_max_str_digits(4300) + try: + markdown = read(html) + finally: + sys.set_int_max_str_digits(limit) + + assert text_words(render(markdown)) == ["trelm", "vosk"] + assert read(render(markdown)) == markdown + + +# --------------------------------------------------------------------------- +# 15. An unfinished tag at the very end +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize( + ("html", "expected"), + [ + pytest.param("

r

x r

x <r

fin <", "r\n\nfin \\<", id="bare" + ), + ], +) +def test_15_an_unfinished_tag_at_the_end_reads_as_text( + html: str, expected: str +) -> None: + """No ``<`` after the last ``>`` can finish a tag, so each is text. + + CPython's ``html.parser`` before 3.11.14, 3.12.12 and 3.13.6 rescanned + to the end for each such ``<`` -- quadratic, CVE-2025-6069 -- and the + releases that fixed it drop the unfinished tag instead: on 3.13.14, + 0.6.0 read ``x
" + "ligne
" * 20_000 + "

", id="line-breaks"), - pytest.param("
    " + "
  • x
  • " * 20_000 + "
", id="list-items"), - pytest.param("
    " + "
  • " * 20_000 + "
", id="empty-items"), - pytest.param("

" + "gras mot " * 20_000 + "

", id="emphasis"), - ], -) -def test_a_long_body_converts_in_linear_time(html: str) -> None: - """The budget is generous on purpose: this guards the complexity, not the speed.""" - - started = time.perf_counter() - read(html) - - assert time.perf_counter() - started < 20 - - -def test_a_body_too_deep_to_convert_keeps_its_text() -> None: - """markdownify recurses per nesting level; past the stack, the text is kept.""" - - html = "
" * 3000 + "

__init__ au fond

" + "
" * 3000 - - markdown = read(html) - - assert "init" in markdown - assert "au fond" in render(markdown) From 96847dff550f6d4477a3a7ae1f5a1e408fa46680 Mon Sep 17 00:00:00 2001 From: baraline Date: Fri, 2 Oct 2026 17:27:15 +0200 Subject: [PATCH 05/16] test(content): port easyvista-python-client 0.4.0's property and round-trip tests - test_properties.py: the seeded generators that package verified the 15 reader fixes with, each body held to three things: it displays what the HTML displayed, it is a fixed point, and it keeps its words in order. Families: image alt and link text, "~~~" at a line start, "!" before a link or image, inline code holding line breaks, tables nested in a cell, a heading or a link, raw // in every holder, and whole generated documents. On 0.6.0 all 16 tests failed, 39 to 100 percent of each test's bodies (CPython 3.12.3); none fails now. - test_round_trip.py: that package's earlier regression bodies (HTML only, held to display plus fixed point), an e-mail notification template laid out as nested tables, a CR LF and an entity in plain text, and a pin on what the write models' validator reads as HTML. - test_conversion.py: the converter is not a sanitiser, pinned exactly in both directions, as the user guide will say. Every body is synthetic and every URL is under example.org. Reverting any one of the 15 fixes alone, in a disposable copy, fails at least one test of this suite (CPython 3.12.3; fix 15's functional case needs a patched interpreter and was confirmed on 3.13.14). The content suite runs in about 21 seconds. Co-Authored-By: Claude Opus 5.5 --- .../content/tests/test_conversion.py | 47 ++ .../content/tests/test_properties.py | 786 ++++++++++++++++++ .../content/tests/test_round_trip.py | 320 ++++++- 3 files changed, 1143 insertions(+), 10 deletions(-) create mode 100644 glpi_python_client/content/tests/test_properties.py diff --git a/glpi_python_client/content/tests/test_conversion.py b/glpi_python_client/content/tests/test_conversion.py index 438841d..6c70668 100644 --- a/glpi_python_client/content/tests/test_conversion.py +++ b/glpi_python_client/content/tests/test_conversion.py @@ -195,3 +195,50 @@ def test_deeply_nested_markdown_renders() -> None: markdown = "".join(" " * (2 * level) + "- x\n" for level in range(600)) assert render(markdown).count("", + "
  • code
", + "
    • code
    ", + '
    • i
      code
    • ' + "
    ", + ], + "struck": [ + "

    ancien nouveau

    ", + "

    ancien nouveau

    ", + "

    ancien nouveau

    ", + ], + "hidden": [ + "

    Bonjour

    ", + "RE: Imprimante

    Bonjour

    " + "", + ], + "misreading": [ + "

    # pas un titre

    ", + "

    #4521 est un doublon

    ", + "

    2026. Une annee

    ", + ], + "upper-case": [ + "

    Bonjour

    ", + "
    Bonjour", + "
    Bonjour
    ", + ], +} + + +@pytest.mark.parametrize( + "html", + [ + pytest.param(html, id=f"{group}-{index}") + for group, bodies in EARLIER_REGRESSIONS.items() + for index, html in enumerate(bodies) + ], +) +def test_an_earlier_regression_body_survives(html: str) -> None: + assert_survives(html) + + @pytest.mark.parametrize( ("html", "expected"), [ @@ -189,7 +378,10 @@ def test_ordinary_prose_reads_back_as_typed(html: str, expected: str) -> None: def test_a_table_nested_in_a_cell_keeps_its_text() -> None: - """A signature laid out as a table inside a table keeps every word.""" + """A table inside a table keeps every word. + + E-mail signatures and notification templates are often laid out so. + """ html = ( "
    Signature
    " @@ -226,6 +418,65 @@ def test_a_fence_keeps_its_language() -> None: assert read(render(markdown)) == markdown +# --------------------------------------------------------------------------- +# A notification template: a layout table around a table of label/value cells +# --------------------------------------------------------------------------- +# +# From easyvista-python-client 0.4.0. E-mail templates are commonly laid out +# as an outer one-cell table around an inner one-row table: each label in +# bold over its value, spacer cells between them, and here one label in a +# cell of its own beside its value's cell. That shape exercises the +# flattening of a table nested in a cell, fix 9 of test_fixes.py (bold that +# ends on punctuation at a flattened cell's edge still closes -- 0.6.0 lost +# the fixed point there) and
    in a cell, all at once. Every word below +# is invented. + +_LABELS = ["Zorvane", "Plimet", "Quastor", "Brindel"] +_VALUES = ["trelm 4471", "oskar-vint", "maludi pref", "kobra 12/09"] + + +def _notification_template() -> str: + cells = [] + for label, value in zip(_LABELS, _VALUES, strict=True): + cells.append("") # a spacer cell + cells.append( + f"" + ) + cells.append("") + inner = "
    Jean Dupont
    {label}


    {value}

    Velusk :sanquo
    " + "".join(cells) + "
    " + footer = ( + 'Grovak le suivi ' + 'Trelune' + ) + return f"
    {inner}
    {footer}
    " + + +def test_a_notification_template_keeps_every_word_in_order() -> None: + """Every word, in order, each label still bold and the link still a link. + + The layout does not survive: GFM has no nested table, so the inner row + becomes one line of text in the outer cell, and a GFM table always has a + header row, so the outer table gains an empty one. Both are limits of + the format. Apart from that header row, the display is the same. + """ + + html = _notification_template() + + markdown = read(html) + rendered = render(markdown) + + assert text_words(rendered) == text_words(html) + name, rows = cast(tuple[str, tuple[object, ...]], displayed(html)[0]) + empty_header = ((True, ()),) # the header row GFM requires: one empty cell + assert displayed(rendered) == ((name, (empty_header, *rows)),) + for label in [*_LABELS, "Velusk :"]: + assert f"**{label}**" in markdown + assert "](https://example.org/suivi?ref=PX-7)" in markdown + assert "![Trelune](https://example.org/marque.png)" in markdown + assert "glpi-cell" not in markdown and "data-glpi-" not in markdown + assert read(rendered) == markdown + + # --------------------------------------------------------------------------- # Plain text and the write path # --------------------------------------------------------------------------- @@ -240,21 +491,38 @@ def test_a_fence_keeps_its_language() -> None: r"\\serveur\partage", "if x0", "Bonjour,\n\nMerci.", + pytest.param("Velk ondra\r\nPrastin", id="crlf"), ], ) def test_a_plain_text_body_reads_as_the_text_it_is(text: str) -> None: - """A body with no HTML element is literal text, its lines lines.""" + """A body with no HTML element is literal text, its lines lines. + + The ``crlf`` case: a CR LF separates lines as LF does. An HTML reading + of the same characters would show a space instead. + """ markdown = read(text) shown = displayed(render(markdown)) - expected = displayed( - "

    " + text.replace("<", "<").replace("\n", "
    ") + "

    " - ) + lines = text.replace("<", "<").splitlines() + expected = displayed("

    " + "
    ".join(lines) + "

    ") assert shown == expected assert read(render(markdown)) == markdown +def test_an_entity_in_a_plain_text_body_is_read_literally() -> None: + """With no element, `` `` is six characters of text, not a space. + + Pinned because it is the consequence of reading a body with no HTML + element as text, and the one a reviewer would most likely expect the + other way. + """ + + markdown = read("Trasvel ok") + + assert displayed(render(markdown)) == displayed("

    Trasvel&nbsp;ok

    ") + + @pytest.mark.parametrize( "markdown", [ @@ -276,6 +544,38 @@ def test_the_write_path_still_converts_an_html_document() -> None: ) +@pytest.mark.parametrize( + ("markdown", "expected"), + [ + pytest.param( + " puis
    suite **gras**", + "puis\\\nsuite \\*\\*gras\\*\\*", + id="autolink-first", + ), + pytest.param( + " puis Ctrl **gras**", + "puis `Ctrl` \\*\\*gras\\*\\*", + id="angle-text-first", + ), + ], +) +def test_markdown_opening_with_angle_brackets_and_holding_html_is_read_as_html( + markdown: str, expected: str +) -> None: + """A limit, pinned so that the documentation stays true. + + ``plain_text_is_markdown=True`` passes a value through unless it starts + with ``<`` and holds a real HTML element *anywhere*. So Markdown that + opens with an autolink or with angle-bracketed text, and carries inline + HTML further on, is read as HTML: the autolink, an unknown element to + the HTML parser, is dropped, and the Markdown syntax is escaped as text. + The write models' validator passes ``plain_text_is_markdown=True``, so + ``PostTicket(content=...)`` reads such a value the same way. + """ + + assert read(markdown, plain_text_is_markdown=True) == expected + + # --------------------------------------------------------------------------- # Markdown first # --------------------------------------------------------------------------- From 25a014ab62db13311d7a6d2b8d9822a9b3002ef7 Mon Sep 17 00:00:00 2001 From: baraline Date: Fri, 2 Oct 2026 17:28:04 +0200 Subject: [PATCH 06/16] build: adopt easyvista-python-client 0.4.0's bounds for the reader's libraries The ported reader fixes rely on private surfaces of markdownify and mdformat, and the two packages' readers should resolve to the same libraries when installed side by side. The bounds are now those of easyvista-python-client[content] 0.4.0, each explained in pyproject.toml: - markdown-it-py>=3.0,<4: the alt-text fix relies on 3.x's text_special tokens; 4.x ends a ragged table early. - markdownify>=1.2.3,<1.3: the overrides use process_tag, escape and the _inline/_noformat pseudo-tags; 1.2.3 is the release the fixes were measured with, and the cap is a precaution. - mdformat>=0.7.22,<0.8: unchanged; mdformat-tables 1.0 requires it too. - mdformat-tables>=1.0,<1.1: the cap is a precaution. Co-Authored-By: Claude Opus 5.5 --- pyproject.toml | 24 ++++++++++++++++++++---- 1 file changed, 20 insertions(+), 4 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 62f048e..9903d08 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -64,11 +64,27 @@ dependencies = [ "cmarkgfm>=2025.10", "httpx>=0.28", "lxml>=4.9", - "markdown-it-py>=3.0", - "markdownify>=1.2", - # The reader extends mdformat's renderer, an API that may move at 0.8. + # The four bounds below are easyvista-python-client[content] 0.4.0's, + # whose reader is this one: the two packages install side by side and + # read the same HTML into the same Markdown. The reader overrides + # private surfaces of markdownify and mdformat, so raise a bound only + # with the content tests re-run against the new release. + # + # Keeping an image's alt-text escapes relies on markdown-it-py 3's + # text_special tokens. 4.x caps the cells it fills in, which ends a + # ragged table early (measured by easyvista-python-client, 2026-10-02). + "markdown-it-py>=3.0,<4", + # The reader overrides markdownify's process_tag, escape and convert_* + # methods and relies on its _inline and _noformat pseudo-tags. 1.2.3 is + # the release the reader's fixes were measured with; 1.3 did not exist + # on 2026-10-02, so the cap is a precaution. + "markdownify>=1.2.3,<1.3", + # The reader extends mdformat's renderer, an API that may move at 0.8, + # and mdformat-tables 1.0 requires mdformat<0.8 too. "mdformat>=0.7.22,<0.8", - "mdformat-tables>=1.0", + # GFM tables for mdformat; 1.1 did not exist on 2026-10-02, so the cap + # is a precaution. + "mdformat-tables>=1.0,<1.1", "pydantic>=2.8", # Never imported by this package, and deliberately so. httpcore decides # whether it is running under asyncio or trio by probing for sniffio on From 8296ead9fd114be98dba31a71802669c3bd49447 Mon Sep 17 00:00:00 2001 From: baraline Date: Fri, 2 Oct 2026 19:05:25 +0200 Subject: [PATCH 07/16] build: give checkable reasons for the reader's dependency bounds 25a014a's comments stated what this package cannot check from its own repository: that markdownify 1.2.3 is the release the fixes were measured with, and that markdownify 1.3 and mdformat-tables 1.1 did not exist on 2026-10-02. They now say what can be checked: - mdformat 0.7.22 itself requires markdown-it-py<4 (its metadata), and easyvista-python-client measured markdown-it-py 4 ending a ragged table early; - 1.2.3 is easyvista-python-client's markdownify floor, and this package's content tests pass on 1.2.2 (CPython 3.12.3) and 1.2.3 (3.13.14); - the markdownify and mdformat-tables caps are precautions. The bounds themselves are unchanged. Co-Authored-By: Claude Opus 5.5 --- pyproject.toml | 20 ++++++++++---------- 1 file changed, 10 insertions(+), 10 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 9903d08..9c42076 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -65,25 +65,25 @@ dependencies = [ "httpx>=0.28", "lxml>=4.9", # The four bounds below are easyvista-python-client[content] 0.4.0's, - # whose reader is this one: the two packages install side by side and - # read the same HTML into the same Markdown. The reader overrides - # private surfaces of markdownify and mdformat, so raise a bound only - # with the content tests re-run against the new release. + # whose reader is this one, so the two packages resolve to the same + # libraries when installed side by side. The reader overrides private + # surfaces of markdownify and mdformat, so raise a bound only with the + # content tests re-run against the new release. # # Keeping an image's alt-text escapes relies on markdown-it-py 3's - # text_special tokens. 4.x caps the cells it fills in, which ends a - # ragged table early (measured by easyvista-python-client, 2026-10-02). + # text_special tokens. mdformat 0.7.22 itself requires markdown-it-py<4, + # and easyvista-python-client measured 4.x ending a ragged table early. "markdown-it-py>=3.0,<4", # The reader overrides markdownify's process_tag, escape and convert_* # methods and relies on its _inline and _noformat pseudo-tags. 1.2.3 is - # the release the reader's fixes were measured with; 1.3 did not exist - # on 2026-10-02, so the cap is a precaution. + # easyvista-python-client's floor; this package's content tests pass on + # 1.2.2 and 1.2.3. The cap is a precaution against a minor release that + # moves those surfaces. "markdownify>=1.2.3,<1.3", # The reader extends mdformat's renderer, an API that may move at 0.8, # and mdformat-tables 1.0 requires mdformat<0.8 too. "mdformat>=0.7.22,<0.8", - # GFM tables for mdformat; 1.1 did not exist on 2026-10-02, so the cap - # is a precaution. + # GFM tables for mdformat. The cap is a precaution, as for markdownify. "mdformat-tables>=1.0,<1.1", "pydantic>=2.8", # Never imported by this package, and deliberately so. httpcore decides From 46a90c082bd6ec95be77bb54ef0e86c4f37db9f6 Mon Sep 17 00:00:00 2001 From: baraline Date: Fri, 2 Oct 2026 19:08:46 +0200 Subject: [PATCH 08/16] test(content): re-measure test_cost's figures on an idle machine 96847df's docstring figures were read while other runs shared the machine. Re-measured 2026-10-02 with growth() itself, three readings per shape, nothing else running: - this converter: 3.8 to 5.4 on CPython 3.12.3, 2.9 to 4.5 on 3.13.14 (the limit is 8); - 0.6.0 on 3.12.3: 10.9 to 11.3 (ordered list), 13.0 to 17.1 (bold holding line breaks), 28.5 to 31.1 (unfinished-tag tail); - reverting fix 12, 13 or 15 alone: 10.4 to 12.0, 16.8 to 17.4 and 31.9 to 33.0 on its shape; - 5,000 ordered items, best of three: 3.1 s on 0.6.0, 0.6 s now. All within the ranges the docstring gave; only the figures change. Co-Authored-By: Claude Opus 5.5 --- glpi_python_client/content/tests/test_cost.py | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/glpi_python_client/content/tests/test_cost.py b/glpi_python_client/content/tests/test_cost.py index 0fb408f..a93dccd 100644 --- a/glpi_python_client/content/tests/test_cost.py +++ b/glpi_python_client/content/tests/test_cost.py @@ -17,13 +17,13 @@ Ported from easyvista-python-client 0.4.0, whose converter is this one. The shapes are 0.6.0's four long bodies, in this form; three bodies dense with syntax from that package's earlier tests; and one per fix that -removed a quadratic. Measured 2026-10-02, timed the way :func:`growth` -times, at these sizes, three readings per shape: this converter grew by -3.0 to 7.5 on CPython 3.12.3 and by 2.6 to 5.7 on 3.13.14. On 3.12.3, -0.6.0 grew by 10.3 and 10.4 on the ordered list, 14.2 and 15.5 on bold -holding line breaks and 30.9 and 34.4 on the unfinished-tag tail (two -readings each), and reverting fix 12, 13 or 15 alone grew by 10.4, 14.1 -and 33.2 on its shape (one reading each). +removed a quadratic. Measured 2026-10-02 on an otherwise idle machine, +timed by :func:`growth` at these sizes, three readings per shape: this +converter grew by 3.8 to 5.4 on CPython 3.12.3 and by 2.9 to 4.5 on +3.13.14. On 3.12.3, 0.6.0 grew by 10.9 to 11.3 on the ordered list, 13.0 +to 17.1 on bold holding line breaks and 28.5 to 31.1 on the unfinished-tag +tail, and reverting fix 12, 13 or 15 alone grew by 10.4 to 12.0, 16.8 to +17.4 and 31.9 to 33.0 on its shape. A ratio between 8 and 10 is measured once more and the second reading decides, so that one burst of load cannot fail a linear pass; a reading of @@ -161,7 +161,7 @@ def test_a_long_ordered_list_converts_in_linear_time() -> None: """Fix 12: each item takes its number from the item before it. markdownify counted every sibling before each item, which was quadratic. - On 5,000 items 0.6.0 took 3.9 seconds against 0.8 with the fix + On 5,000 items 0.6.0 took 3.1 seconds against 0.6 with the fix (measured 2026-10-02, CPython 3.12.3, best of three). The list is written one item per line, as an editor writes it: each newline is one more sibling to count, which makes the quadratic easier to see. From 7f596b97bc3de7cd5cb53891bce221628c306107 Mon Sep 17 00:00:00 2001 From: baraline Date: Fri, 2 Oct 2026 19:10:38 +0200 Subject: [PATCH 09/16] docs(content): document the ported reader; the fixed point is an aim The user guide, the API reference, the module docstring and three skills now describe the reader as 0.6.1 has it. Every behaviour stated was reproduced on this branch on 2026-10-02. - User guide, rich-text section: the escape table, what each element becomes (raw /// included), that neither direction sanitises, what survives a round trip, the structures Markdown cannot spell, the known holes, and the CVE-2025-6069 guard. The statement that a read always displays the same and reads back as itself, which 0.6.0 made unqualified, is now the aim it is, there, in the API reference and in the module docstring. - The deep-nesting note gives measured depths: about 326
    levels from a shallow stack (0.6.0: about 490; the reader spends one more frame per level) and 194
    , on CPython 3.12.3 and 3.13.14. It no longer says a read never raises: started within about 30 frames of the recursion limit, it raises GlpiContentError, then a bare RecursionError (measured: from about 975 and 994 frames). - The write-model rule is stated exactly: Markdown passes through unless it starts with "<" and holds an HTML element anywhere, so an opening autolink followed by inline HTML is lost. - from_transport documents that RecursionError; GlpiContentError's docstring named python-markdown as the writer, which is cmark-gfm. - The installation page names the converter's libraries. development.md pointed at a module docstring that no longer holds the depth derivation; the 0.5.0 changelog entry does. - Skills glpi-ticket-timeline, glpi-ticket-workflow and glpi-knowledge-base: raw tags, the write-model rule, the sanitiser caveat and the new depth. Found while checking the guide: the claim, which easyvista-python-client's docs also make, that a second write-and-read cycle of Markdown changes nothing more failed on 23 of 4,500 generated bodies -- adjacent lists with different bullets, and a fence whose info string holds a character reference. The guide lists both. The conversion module still differs from easyvista-python-client 0.4.0's only in names, messages, docstrings and its import guard (AST comparison). Co-Authored-By: Claude Opus 5.5 --- docs/api_reference.rst | 16 +- docs/development.md | 9 +- docs/installation.rst | 7 +- docs/user_guide.rst | 228 +++++++++++++++++++++-- glpi_python_client/_errors.py | 4 +- glpi_python_client/content/conversion.py | 35 +++- skills/glpi-knowledge-base/SKILL.md | 2 +- skills/glpi-ticket-timeline/SKILL.md | 4 +- skills/glpi-ticket-workflow/SKILL.md | 4 +- 9 files changed, 266 insertions(+), 43 deletions(-) diff --git a/docs/api_reference.rst b/docs/api_reference.rst index 0d259cd..db84c91 100644 --- a/docs/api_reference.rst +++ b/docs/api_reference.rst @@ -115,17 +115,21 @@ its page being read. The Markdown is CommonMark with GFM tables, rendered by cmark-gfm. It spells a body's text as literal text -- ``\_\_init\_\_``, ``\\\serveur``, -``\# pas un titre`` -- so rendering it displays what GLPI displayed and -reading that back gives the same Markdown. A write model keeps the caller's -Markdown verbatim unless it starts with an HTML tag. See -:ref:`content-conversion`. +``\# pas un titre`` -- so that rendering it displays what GLPI displayed and +reading that back gives the same Markdown. That is the aim; the user guide +lists the shapes where it falls short. Underline, highlight and struck text +stay as raw ````, ````, ```` and ````. Neither direction +sanitises: raw HTML and ``javascript:`` links pass through. A write model +keeps the caller's Markdown verbatim unless it starts with ``<`` and holds an +HTML element. See :ref:`content-conversion`. Very deeply nested HTML is the case worth knowing about. ``markdownify`` walks the document recursively and runs out of stack at a few hundred levels of nesting. The converter attempts the conversion and, if the walk does not fit, reads the body's text instead, a line per -block: the words survive, the structure does not. Anything else that goes -wrong in either direction raises :class:`GlpiContentError`. +block: the words survive, the structure does not. A ``colspan`` or ``start`` +that markdownify cannot read as a number degrades the same way. Anything +else that goes wrong in either direction raises :class:`GlpiContentError`. Because the conversion is cached on first read, a read model should be treated as immutable afterwards: assigning to ``content_html``, or diff --git a/docs/development.md b/docs/development.md index c8e5cfe..36e839d 100644 --- a/docs/development.md +++ b/docs/development.md @@ -86,13 +86,16 @@ python -m pytest articles (`content` and `description`) and article revisions. It is wired into the models by `models/api_schema/_content.py`, eagerly on the write models and through a cached property on the read ones. - Inbound HTML too deep for `markdownify` to walk is tag-stripped + Inbound HTML too deep for `markdownify` to walk, or holding a + `colspan` or `start` it cannot read as a number, is tag-stripped rather than parsed. The depth is not predicted: the conversion is attempted and the `RecursionError` answered, because the budget is the caller's remaining stack and no bound computed in advance can know it. - The module docstring carries the derivation and the rejected + The 0.5.0 changelog entry carries the derivation and the rejected alternatives (`sys.setrecursionlimit`, and a thread with a larger - stack, which needs the same global). The prohibition is enforced by + stack, which needs the same global). The reader is kept identical, but + for names, messages, docstrings and an import guard, to + `easyvista-python-client`'s; change the two together. The prohibition is enforced by `testing/tests/test_raise_site_audit.py`, not just written down. - `glpi_python_client.testing` exposes `make_client` and `make_async_client` factories that produce in-memory clients with no diff --git a/docs/installation.rst b/docs/installation.rst index c10cd41..660e109 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -6,7 +6,12 @@ Requirements ``glpi-python-client`` supports Python 3.11 and newer. Runtime dependencies are installed from the package metadata and include ``httpx``, ``tenacity``, -``beautifulsoup4``, ``lxml``, and ``pydantic``. +``beautifulsoup4``, ``lxml``, and ``pydantic``, plus the libraries the +rich-text converter is built on: ``markdownify``, ``mdformat``, +``mdformat-tables`` and ``markdown-it-py`` to read, and ``cmarkgfm`` to +write. The reader overrides private surfaces of ``markdownify`` and +``mdformat``, so those four carry upper bounds; ``pyproject.toml`` gives the +reason for each. Install from PyPI ----------------- diff --git a/docs/user_guide.rst b/docs/user_guide.rst index 585df9c..c95b3db 100644 --- a/docs/user_guide.rst +++ b/docs/user_guide.rst @@ -1758,11 +1758,13 @@ The whole page is built in one pass, so a single unconvertible record used to make its page-mates unreadable too. The failure is now scoped to the record whose body you actually read. -The Markdown is CommonMark with GFM tables. Rendering it -- as the package -does on the way back to GLPI, with cmark-gfm -- displays what GLPI -displayed, and reading that rendering back gives the same Markdown. Text in -a body is literal: Markdown has one spelling for ``__init__`` typed by a -user and for bold ``init``, so text that would read as syntax is escaped: +The Markdown is CommonMark with GFM tables. The aim is that rendering it -- +as the package does on the way back to GLPI, with cmark-gfm -- displays what +GLPI displayed, and that reading that rendering back gives the same +Markdown. It is an aim, not a guarantee for every body: `What survives a +round trip`_ lists where it falls short. Text in a body is literal: Markdown +has one spelling for ``__init__`` typed by a user and for bold ``init``, so +text that would read as syntax is escaped: .. code-block:: python @@ -1781,28 +1783,95 @@ back as ``\`` and a newline, a nested list is indented by its bullet's width, and a table comes back unpadded. A body with no HTML element is plain text and is read as GLPI shows it, its lines as lines. +More of what GLPI displays, and the Markdown it reads as: + +.. code-block:: text + + GLPI displays Markdown + ---------------------------- ---------------------------- + __init__ \_\_init\_\_ + *important* \*important\* + # 4521 (at a line start) \# 4521 + 1. pas une liste 1\. pas une liste + - pas une liste \- pas une liste + > pas une citation \> pas une citation + [lien](x) \[lien\](x) + \ + ~~~ (at a line start) \~~~ + Attention! then a link Attention\![lien](https://example.org) + +Text GLPI displays as markup is read back as that text, so ``<b>`` +reads as ``\``, never as a live ````. What each kind of element +becomes: + +* **Bold and italic** become ``**`` and ``*``, or raw ```` and + ```` where CommonMark would not close the markers, for example bold + that ends in punctuation directly followed by a letter: + ``prix:(10)euros`` reads as ``prix:(10)euros``. +* **Underline, highlight and inserted text** (````, ````, + ````) stay as those raw tags, and **struck text** (````, + ````, ````) as a raw ````. CommonMark has no spelling for + any of them, and writing passes the tags through, so the formatting + survives. The cost is raw HTML in the Markdown. +* **Links** become ``[text](https://... "title")``, and a link whose text is + its own URL, a pasted link, becomes the autolink ````. No + link target is filtered, ``javascript:`` included (see `It is not a + sanitiser`_). An **image** becomes ``![alt](src "title")``, its alt text + escaped like any other text. +* **Lists** nest and keep their numbers, including an ``
      ``; a + ``start`` that is not a decimal number, such as ``²``, counts from 1, as + a browser does. **Tables** become GFM tables, and a ``
      `` inside a cell + stays a raw ``
      ``, since a GFM cell is one line. A table inside a cell, + a heading or a link is written as its cells' text, spaced apart. +* **Preformatted blocks** become fences that keep the ``language-`` class + cmark-gfm writes, with a fence longer than any run of backticks in the + code. **Inline code** holding a line break becomes one code span per line. +* ``
      `` and ``
      `` are blocks. ````, ```` goes out as a live ``
  • ") == 600 + + +# --------------------------------------------------------------------------- +# It is not a sanitiser +# --------------------------------------------------------------------------- +# +# What the user guide ("Rich-text content", docs/user_guide.rst) says, +# pinned exactly: a cmark-gfm option or release that changed any of it would +# make the documentation wrong in one direction or the other. The writing +# direction neutralises nothing; the reading direction keeps a body's link +# targets, and keeps text a body displays as text. easyvista-python-client +# 0.4.0 pins the same. + + +@pytest.mark.parametrize( + ("markdown", "html"), + [ + pytest.param( + "", "", id="raw-html" + ), + pytest.param( + "[x](javascript:alert(1))", + '

    x

    ', + id="link-target", + ), + pytest.param( # python-markdown, before 0.6.0, left this as raw markup + "", + '

    javascript:alert(1)

    ', + id="autolink", + ), + ], +) +def test_the_write_path_renders_what_it_is_given_live(markdown: str, html: str) -> None: + assert render(markdown) == html + + +def test_the_read_path_keeps_a_bodys_javascript_link() -> None: + assert read('

    x

    ') == ( + "[x]()" + ) + + +def test_markup_a_body_displays_as_text_stays_text_both_ways() -> None: + markdown = read("

    <script>

    ") + + assert markdown == "\\
    code