From 3f7bf1ba82f53d97e5d3d88553e74337857e0abf Mon Sep 17 00:00:00 2001 From: Abhishek Date: Wed, 30 Sep 2026 11:33:03 +0530 Subject: [PATCH] test(pipelines): cover the hugo_ingest functions the v4 ingest path uses utils/hugo_ingest.py had no tests. #249's test plan names tests/test_hugo_ingest.py, but the file was never committed (#259). Only the production surface is covered. canonical_rag_ingest imports parse_frontmatter and process_html_table; clean_hugo_markdown has no caller, so testing it would add coverage that exercises nothing. Expectations are taken from the module's docstrings and inline comments, from how parse_canonical_document consumes the results, and from docs/RAG_V4_ARCHITECTURE.md, rather than from whatever the code currently returns: - parse_frontmatter: YAML and TOML give the title/description/weight that parse_canonical_document reads; a byte-order mark does not hide the frontmatter; missing keys stay absent; empty input is returned as-is; malformed frontmatter falls back to ({}, content) instead of aborting the run. - process_html_table: tables become GFM tables; rowspan values repeat into every spanned row (release tables); colspan keeps rows the same width; an escaped pipe does not add a column; text around and between tables stays off the table rows, as canonical_rag_ingest groups rows; the markdownify fallback still produces a table; text with no table is returned unchanged. The ingest dependencies were only in pipelines/requirements.txt and Dockerfile.pipeline, so the tests would have skipped in PR Safety. They are added to requirements-test.txt at the same pinned versions. Signed-off-by: Abhishek --- requirements-test.txt | 7 ++ tests/test_hugo_ingest.py | 215 ++++++++++++++++++++++++++++++++++++++ 2 files changed, 222 insertions(+) create mode 100644 tests/test_hugo_ingest.py diff --git a/requirements-test.txt b/requirements-test.txt index ee80e13..7e6689b 100644 --- a/requirements-test.txt +++ b/requirements-test.txt @@ -5,3 +5,10 @@ fastmcp==3.4.2 langchain-text-splitters==0.3.8 pyyaml==6.0.2 PyJWT[crypto]==2.10.1 +# docs ingest (hugo_ingest / canonical_rag_ingest); pinned to match +# docs-agent-mcp/pipelines/requirements.txt and Dockerfile.pipeline +beautifulsoup4==4.15.0 +python-frontmatter==1.1.0 +markdownify==1.2.0 +html-table-rescuer==0.3.1 +toml==0.10.2 diff --git a/tests/test_hugo_ingest.py b/tests/test_hugo_ingest.py new file mode 100644 index 0000000..6491e3d --- /dev/null +++ b/tests/test_hugo_ingest.py @@ -0,0 +1,215 @@ +"""Tests for utils/hugo_ingest.py. + +Only the production surface is covered: canonical_rag_ingest imports +parse_frontmatter and process_html_table. clean_hugo_markdown has no caller. + +Expectations come from the module's docstrings and inline comments, the way +parse_canonical_document consumes the results, and docs/RAG_V4_ARCHITECTURE.md, +which names Hugo frontmatter and release tables as inputs v4 must handle. +""" + +import sys +from pathlib import Path + +import pytest + +UTILS_DIR = Path(__file__).parent.parent / "docs-agent-mcp" / "pipelines" / "utils" +sys.path.insert(0, str(UTILS_DIR)) + +for _dependency in ("frontmatter", "html_table_rescuer", "markdownify", "bs4", "toml"): + pytest.importorskip(_dependency, reason="hugo_ingest tests need the docs ingest dependencies") + +import hugo_ingest # noqa: E402 +from hugo_ingest import parse_frontmatter, process_html_table # noqa: E402 + + +def gfm_tables(text): + """Group consecutive `|` lines into tables, as canonical_rag_ingest does.""" + tables, current = [], [] + for line in text.splitlines(): + if line.lstrip().startswith("|"): + current.append(line.strip()) + elif current: + tables.append(current) + current = [] + if current: + tables.append(current) + return tables + + +def cells(row): + """Split a GFM row into cells, honouring escaped pipes.""" + inner = row.strip()[1:-1] + return [cell.strip().replace("\0", "|") for cell in inner.replace("\\|", "\0").split("|")] + + +def rows(table): + return [cells(line) for line in table] + + +# --- parse_frontmatter ------------------------------------------------------- + +YAML_PAGE = """--- +title: Katib +description: Hyperparameter tuning +weight: 20 +--- +# Overview + +Body text. +""" + +TOML_PAGE = """+++ +title = "Katib" +description = "Hyperparameter tuning" +weight = 20 ++++ +# Overview + +Body text. +""" + + +@pytest.mark.parametrize("page", [YAML_PAGE, TOML_PAGE], ids=["yaml", "toml"]) +def test_frontmatter_yields_the_fields_parse_canonical_document_reads(page): + meta, body = parse_frontmatter(page) + + assert meta["title"] == "Katib" + assert meta["description"] == "Hyperparameter tuning" + # parse_canonical_document does int(meta.get("weight") or 0). + assert int(meta.get("weight") or 0) == 20 + assert body.startswith("# Overview") + assert "title" not in body and "+++" not in body and "---" not in body + + +def test_yaml_and_toml_frontmatter_parse_to_the_same_metadata(): + assert parse_frontmatter(YAML_PAGE) == parse_frontmatter(TOML_PAGE) + + +def test_page_without_frontmatter_keeps_its_body(): + page = "# Just markdown\n\nSome text.\n" + meta, body = parse_frontmatter(page) + + assert meta == {} + assert body.strip() == page.strip() + + +def test_byte_order_mark_does_not_hide_frontmatter(): + meta, body = parse_frontmatter("\ufeff" + YAML_PAGE) + + assert meta["title"] == "Katib" + assert body.startswith("# Overview") + + +def test_missing_keys_are_absent_so_callers_fall_back_to_defaults(): + meta, _ = parse_frontmatter("---\ntitle: Only a title\n---\nbody\n") + + assert meta == {"title": "Only a title"} + assert int(meta.get("weight") or 0) == 0 + + +@pytest.mark.parametrize("content", ["", None], ids=["empty", "none"]) +def test_empty_input_returns_empty_metadata_unchanged(content): + assert parse_frontmatter(content) == ({}, content) + + +@pytest.mark.parametrize( + "page", + [ + "---\ntitle: [unclosed\n---\nbody\n", + "+++\ntitle = \n+++\nbody\n", + ], + ids=["yaml", "toml"], +) +def test_malformed_frontmatter_falls_back_instead_of_failing_the_run(page): + # One bad page must not abort ingest of the whole docs tree. + assert parse_frontmatter(page) == ({}, page) + + +# --- process_html_table ------------------------------------------------------ + + +def test_html_without_a_table_is_returned_unchanged(): + # Round-tripping through BeautifulSoup would rewrite this as + # "Katib & Trainer
v1.9". + html = "Katib & Trainer
v1.9, with bold and no tables." + assert process_html_table(html) == html + + +def test_table_becomes_a_gfm_table_with_header_separator_and_rows(): + html = "
ComponentVersion
Katibv0.17
" + out = process_html_table(html) + + assert "ReleaseComponentVersion" + '1.9Katibv0.17' + "Trainerv1.8" + ) + [table] = gfm_tables(process_html_table(html)) + + assert rows(table)[2:] == [["1.9", "Katib", "v0.17"], ["1.9", "Trainer", "v1.8"]] + + +def test_colspan_keeps_every_row_the_same_width(): + html = '
AB
merged
' + [table] = gfm_tables(process_html_table(html)) + + widths = {len(row) for row in rows(table)} + assert widths == {2} + assert rows(table)[2][0] == "merged" + + +def test_pipe_inside_a_cell_does_not_add_a_column(): + html = "
FlagValues
--modea|b
" + [table] = gfm_tables(process_html_table(html)) + + assert rows(table)[2] == ["--mode", "a|b"] + + +def test_text_around_a_table_is_kept_off_the_table_rows(): + html = "Intro para.\n
H
v
\nOutro para." + out = process_html_table(html) + + assert "Intro para." in out and "Outro para." in out + [table] = gfm_tables(out) + assert not any("para" in line for line in table) + + +def test_text_between_two_tables_separates_them(): + html = ( + "
X
1
" + "mid" + "
Y
2
" + ) + tables = gfm_tables(process_html_table(html)) + + assert [rows(t)[0] for t in tables] == [["X"], ["Y"]] + assert [rows(t)[2] for t in tables] == [["1"], ["2"]] + + +def test_falls_back_to_markdownify_when_the_table_parser_returns_nothing(monkeypatch): + class EmptyParser: + def __init__(self, *args, **kwargs): + pass + + def parse(self): + return [] + + monkeypatch.setattr(hugo_ingest, "TableParser", EmptyParser) + html = "
Component
Katib
" + out = process_html_table(html) + + assert "