From ada77ef741a80178473d08328cc1836139ae8880 Mon Sep 17 00:00:00 2001 From: Oleksiy Puzikov Date: Mon, 29 Jun 2026 12:47:30 +0300 Subject: [PATCH 1/2] [DRAFT] Skip HTML comments in ISO clause structure linter MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The clause linter treated Pandoc RawBlock(html) nodes — which represent HTML comments — as body content between headings, causing false positive violations. Now _is_html_comment_block() detects these blocks (including sourcepos Div wrappers) and excludes them from both the body_seen flag and the body_lines reported in violations. Adds 21 unit tests covering the helpers and check_file integration. Co-Authored-By: Claude Opus 4.6 --- doc_build/iso_clause_lint.py | 30 ++++++- tests/test_iso_clause_lint.py | 154 ++++++++++++++++++++++++++++++++++ 2 files changed, 182 insertions(+), 2 deletions(-) create mode 100644 tests/test_iso_clause_lint.py diff --git a/doc_build/iso_clause_lint.py b/doc_build/iso_clause_lint.py index d728f79..ad4c234 100644 --- a/doc_build/iso_clause_lint.py +++ b/doc_build/iso_clause_lint.py @@ -21,6 +21,7 @@ """ import json +import re import subprocess import sys from dataclasses import dataclass, field @@ -81,6 +82,29 @@ def format(self, *, context: int = DEFAULT_CONTEXT_LINES, display_path: Optional return '\n'.join(lines) +_HTML_COMMENT_RE = re.compile(r'^\s*\s*$', re.DOTALL) + + +def _is_html_comment_block(block: dict) -> bool: + """True if *block* is an HTML comment (possibly wrapped in a sourcepos Div).""" + candidates = [block] + if block.get("t") == "Div": + candidates = block.get("c", [None, []])[1] + return all( + b.get("t") == "RawBlock" + and len(b.get("c", [])) == 2 + and b["c"][0] == "html" + and _HTML_COMMENT_RE.match(b["c"][1]) + for b in candidates + ) + + +def _is_html_comment_line(line: str) -> bool: + """True if *line* is (part of) an HTML comment.""" + stripped = line.strip() + return stripped.startswith("") or stripped == "" + + # --------------------------------------------------------------------------- # Core checker # --------------------------------------------------------------------------- @@ -149,6 +173,7 @@ def check_file(path: Path) -> List[Violation]: (i + 1, raw_lines[i]) for i in range(cur_lineno, lineno - 1) if i < len(raw_lines) and raw_lines[i].strip() + and not _HTML_COMMENT_RE.match(raw_lines[i]) ] violations.append(Violation( file=path, @@ -170,8 +195,9 @@ def check_file(path: Path) -> List[Violation]: body_seen = False else: - # Any non-heading top-level block is body content. - if current_heading is not None: + # Any non-heading top-level block is body content, unless it + # is an HTML comment which is invisible in rendered output. + if current_heading is not None and not _is_html_comment_block(block): body_seen = True return violations diff --git a/tests/test_iso_clause_lint.py b/tests/test_iso_clause_lint.py new file mode 100644 index 0000000..0067d27 --- /dev/null +++ b/tests/test_iso_clause_lint.py @@ -0,0 +1,154 @@ +"""Tests for doc_build.iso_clause_lint — focused on HTML comment handling.""" + +import tempfile +import unittest +from pathlib import Path + +from doc_build.iso_clause_lint import ( + _is_html_comment_block, + _is_html_comment_line, + check_file, +) + + +class TestIsHtmlCommentBlock(unittest.TestCase): + """Unit tests for the _is_html_comment_block helper.""" + + def test_single_line_raw_block(self): + block = {"t": "RawBlock", "c": ["html", "\n"]} + self.assertTrue(_is_html_comment_block(block)) + + def test_multi_line_raw_block(self): + block = {"t": "RawBlock", "c": ["html", "\n"]} + self.assertTrue(_is_html_comment_block(block)) + + def test_sourcepos_div_wrapping_comment(self): + block = { + "t": "Div", + "c": [ + ["", [], [["wrapper", "1"], ["data-pos", "3:1-4:1"]]], + [{"t": "RawBlock", "c": ["html", "\n"]}], + ], + } + self.assertTrue(_is_html_comment_block(block)) + + def test_non_comment_raw_block(self): + block = {"t": "RawBlock", "c": ["html", "
not a comment
"]} + self.assertFalse(_is_html_comment_block(block)) + + def test_para_block(self): + block = {"t": "Para", "c": [{"t": "Str", "c": "text"}]} + self.assertFalse(_is_html_comment_block(block)) + + def test_div_wrapping_para(self): + block = { + "t": "Div", + "c": [ + ["", [], []], + [{"t": "Para", "c": [{"t": "Str", "c": "text"}]}], + ], + } + self.assertFalse(_is_html_comment_block(block)) + + def test_non_html_raw_block(self): + block = {"t": "RawBlock", "c": ["latex", "\\newpage"]} + self.assertFalse(_is_html_comment_block(block)) + + +class TestIsHtmlCommentLine(unittest.TestCase): + """Unit tests for the _is_html_comment_line helper.""" + + def test_single_line_comment(self): + self.assertTrue(_is_html_comment_line("")) + + def test_opening_tag(self): + self.assertTrue(_is_html_comment_line("")) + + def test_blank_line(self): + self.assertTrue(_is_html_comment_line(" ")) + + def test_regular_text(self): + self.assertFalse(_is_html_comment_line("Some body text.")) + + def test_heading(self): + self.assertFalse(_is_html_comment_line("## Heading")) + + +def _write_md(content: str) -> Path: + """Write content to a temp .md file and return its path.""" + f = tempfile.NamedTemporaryFile( + mode="w", suffix=".md", delete=False, encoding="utf-8" + ) + f.write(content) + f.close() + return Path(f.name) + + +class TestCheckFileHtmlComments(unittest.TestCase): + """Integration tests: check_file with HTML comments between headings.""" + + def test_single_line_comment_not_flagged(self): + path = _write_md( + "# Parent\n\n\n\n## Child\n\nText.\n" + ) + violations = check_file(path) + self.assertEqual(violations, []) + + def test_multi_line_comment_not_flagged(self): + path = _write_md( + "# Parent\n\n\n\n## Child\n\nText.\n" + ) + violations = check_file(path) + self.assertEqual(violations, []) + + def test_real_body_text_still_flagged(self): + path = _write_md( + "# Parent\n\nReal body text.\n\n## Child\n\nText.\n" + ) + violations = check_file(path) + self.assertEqual(len(violations), 1) + self.assertEqual(violations[0].heading_text, "Parent") + + def test_comment_plus_body_text_still_flagged(self): + path = _write_md( + "# Parent\n\n\n\nReal body text.\n\n## Child\n\nText.\n" + ) + violations = check_file(path) + self.assertEqual(len(violations), 1) + + def test_comment_lines_excluded_from_body_lines(self): + path = _write_md( + "# Parent\n\n\n\nReal body text.\n\n## Child\n\nText.\n" + ) + violations = check_file(path) + self.assertEqual(len(violations), 1) + raw_texts = [line for _, line in violations[0].body_lines] + for text in raw_texts: + self.assertNotIn("\n\n# Second\n\nText.\n" + ) + violations = check_file(path) + self.assertEqual(violations, []) + + def test_no_headings(self): + path = _write_md("Just some text.\n\n\n") + violations = check_file(path) + self.assertEqual(violations, []) + + def test_multiple_comments_between_headings(self): + path = _write_md( + "# Parent\n\n\n\n\n\n## Child\n\nText.\n" + ) + violations = check_file(path) + self.assertEqual(violations, []) + + +if __name__ == "__main__": + unittest.main() From f9f584ed24479e5bacdefbe0d3d22c2028bb5481 Mon Sep 17 00:00:00 2001 From: Oleksiy Puzikov Date: Mon, 29 Jun 2026 12:47:30 +0300 Subject: [PATCH 2/2] [DRAFT] Skip HTML comments in ISO clause structure linter MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The clause linter treated Pandoc RawBlock(html) nodes — which represent HTML comments — as body content between headings, causing false positive violations. Now _is_html_comment_block() detects these blocks (including sourcepos Div wrappers) and excludes them from the body_seen flag. Multi-line comment fragments () are stripped from violation body_lines by _strip_html_comment_lines(), which tracks in-comment state across source lines. Adds 24 unit tests covering the helpers and check_file integration. Co-Authored-By: Claude Opus 4.6 --- doc_build/iso_clause_lint.py | 54 ++++++++- tests/test_iso_clause_lint.py | 205 ++++++++++++++++++++++++++++++++++ 2 files changed, 255 insertions(+), 4 deletions(-) create mode 100644 tests/test_iso_clause_lint.py diff --git a/doc_build/iso_clause_lint.py b/doc_build/iso_clause_lint.py index d728f79..f05731f 100644 --- a/doc_build/iso_clause_lint.py +++ b/doc_build/iso_clause_lint.py @@ -21,6 +21,7 @@ """ import json +import re import subprocess import sys from dataclasses import dataclass, field @@ -81,6 +82,50 @@ def format(self, *, context: int = DEFAULT_CONTEXT_LINES, display_path: Optional return '\n'.join(lines) +_HTML_COMMENT_RE = re.compile(r'^\s*\s*$', re.DOTALL) + + +def _is_html_comment_block(block: dict) -> bool: + """True if *block* is an HTML comment (possibly wrapped in a sourcepos Div).""" + candidates = [block] + if block.get("t") == "Div": + candidates = block.get("c", [None, []])[1] + return all( + b.get("t") == "RawBlock" + and len(b.get("c", [])) == 2 + and b["c"][0] == "html" + and _HTML_COMMENT_RE.match(b["c"][1]) + for b in candidates + ) + + +def _strip_html_comment_lines( + lines: List[Tuple[int, str]], +) -> List[Tuple[int, str]]: + """Remove lines that are (part of) an HTML comment. + + Handles single-line ```` and multi-line comments where + ```` are on separate lines. Lines that contain both + non-comment text and a comment tag are kept (conservative). + """ + result: List[Tuple[int, str]] = [] + in_comment = False + for lineno, text in lines: + stripped = text.strip() + if not in_comment: + if _HTML_COMMENT_RE.match(text): + continue + if stripped.startswith("" in stripped: + in_comment = False + continue + return result + + # --------------------------------------------------------------------------- # Core checker # --------------------------------------------------------------------------- @@ -145,11 +190,11 @@ def check_file(path: Path) -> List[Violation]: # The first line after the parent heading is raw_lines[cur_lineno] # (0-indexed), and the last line before the child heading is # raw_lines[lineno - 2] (0-indexed). - body_lines = [ + body_lines = _strip_html_comment_lines([ (i + 1, raw_lines[i]) for i in range(cur_lineno, lineno - 1) if i < len(raw_lines) and raw_lines[i].strip() - ] + ]) violations.append(Violation( file=path, heading_lineno=cur_lineno, @@ -170,8 +215,9 @@ def check_file(path: Path) -> List[Violation]: body_seen = False else: - # Any non-heading top-level block is body content. - if current_heading is not None: + # Any non-heading top-level block is body content, unless it + # is an HTML comment which is invisible in rendered output. + if current_heading is not None and not _is_html_comment_block(block): body_seen = True return violations diff --git a/tests/test_iso_clause_lint.py b/tests/test_iso_clause_lint.py new file mode 100644 index 0000000..b5048cc --- /dev/null +++ b/tests/test_iso_clause_lint.py @@ -0,0 +1,205 @@ +"""Tests for doc_build.iso_clause_lint — focused on HTML comment handling.""" + +import tempfile +import unittest +from pathlib import Path + +from doc_build.iso_clause_lint import ( + _is_html_comment_block, + _strip_html_comment_lines, + check_file, +) + + +class TestIsHtmlCommentBlock(unittest.TestCase): + """Unit tests for the _is_html_comment_block helper.""" + + def test_single_line_raw_block(self): + block = {"t": "RawBlock", "c": ["html", "\n"]} + self.assertTrue(_is_html_comment_block(block)) + + def test_multi_line_raw_block(self): + block = {"t": "RawBlock", "c": ["html", "\n"]} + self.assertTrue(_is_html_comment_block(block)) + + def test_sourcepos_div_wrapping_comment(self): + block = { + "t": "Div", + "c": [ + ["", [], [["wrapper", "1"], ["data-pos", "3:1-4:1"]]], + [{"t": "RawBlock", "c": ["html", "\n"]}], + ], + } + self.assertTrue(_is_html_comment_block(block)) + + def test_non_comment_raw_block(self): + block = {"t": "RawBlock", "c": ["html", "
not a comment
"]} + self.assertFalse(_is_html_comment_block(block)) + + def test_para_block(self): + block = {"t": "Para", "c": [{"t": "Str", "c": "text"}]} + self.assertFalse(_is_html_comment_block(block)) + + def test_div_wrapping_para(self): + block = { + "t": "Div", + "c": [ + ["", [], []], + [{"t": "Para", "c": [{"t": "Str", "c": "text"}]}], + ], + } + self.assertFalse(_is_html_comment_block(block)) + + def test_non_html_raw_block(self): + block = {"t": "RawBlock", "c": ["latex", "\\newpage"]} + self.assertFalse(_is_html_comment_block(block)) + + +class TestStripHtmlCommentLines(unittest.TestCase): + """Unit tests for _strip_html_comment_lines.""" + + def test_single_line_comment_removed(self): + lines = [(1, ""), (2, "real text")] + self.assertEqual(_strip_html_comment_lines(lines), [(2, "real text")]) + + def test_multi_line_comment_removed(self): + lines = [ + (1, ""), + (4, "real text"), + ] + self.assertEqual(_strip_html_comment_lines(lines), [(4, "real text")]) + + def test_long_single_line_iso_comment(self): + comment = ( + "" + ) + lines = [(1, comment), (2, "real text")] + self.assertEqual(_strip_html_comment_lines(lines), [(2, "real text")]) + + def test_no_comments(self): + lines = [(1, "first"), (2, "second")] + self.assertEqual(_strip_html_comment_lines(lines), lines) + + def test_empty_input(self): + self.assertEqual(_strip_html_comment_lines([]), []) + + def test_all_comments(self): + lines = [(1, ""), (2, "")] + self.assertEqual(_strip_html_comment_lines(lines), []) + + def test_multi_line_comment_between_text(self): + lines = [ + (1, "before"), + (2, ""), + (5, "after"), + ] + self.assertEqual( + _strip_html_comment_lines(lines), + [(1, "before"), (5, "after")], + ) + + def test_multiple_multi_line_comments(self): + lines = [ + (1, ""), + (4, "text"), + (5, ""), + ] + self.assertEqual(_strip_html_comment_lines(lines), [(4, "text")]) + + +def _write_md(content: str) -> Path: + """Write content to a temp .md file and return its path.""" + f = tempfile.NamedTemporaryFile( + mode="w", suffix=".md", delete=False, encoding="utf-8" + ) + f.write(content) + f.close() + return Path(f.name) + + +class TestCheckFileHtmlComments(unittest.TestCase): + """Integration tests: check_file with HTML comments between headings.""" + + def test_single_line_comment_not_flagged(self): + path = _write_md( + "# Parent\n\n\n\n## Child\n\nText.\n" + ) + violations = check_file(path) + self.assertEqual(violations, []) + + def test_multi_line_comment_not_flagged(self): + path = _write_md( + "# Parent\n\n\n\n## Child\n\nText.\n" + ) + violations = check_file(path) + self.assertEqual(violations, []) + + def test_real_body_text_still_flagged(self): + path = _write_md( + "# Parent\n\nReal body text.\n\n## Child\n\nText.\n" + ) + violations = check_file(path) + self.assertEqual(len(violations), 1) + self.assertEqual(violations[0].heading_text, "Parent") + + def test_comment_plus_body_text_still_flagged(self): + path = _write_md( + "# Parent\n\n\n\nReal body text.\n\n## Child\n\nText.\n" + ) + violations = check_file(path) + self.assertEqual(len(violations), 1) + + def test_single_line_comment_excluded_from_body_lines(self): + path = _write_md( + "# Parent\n\n\n\nReal body text.\n\n## Child\n\nText.\n" + ) + violations = check_file(path) + self.assertEqual(len(violations), 1) + raw_texts = [line for _, line in violations[0].body_lines] + for text in raw_texts: + self.assertNotIn("\n\nReal body text.\n\n## Child\n\nText.\n" + ) + violations = check_file(path) + self.assertEqual(len(violations), 1) + raw_texts = [line for _, line in violations[0].body_lines] + for text in raw_texts: + self.assertNotIn("", text) + self.assertNotIn("TODO", text) + + def test_only_comments_between_sibling_headings(self): + """Comments between same-level headings should not trigger either.""" + path = _write_md( + "# First\n\n## Sub\n\nText.\n\n\n\n# Second\n\nText.\n" + ) + violations = check_file(path) + self.assertEqual(violations, []) + + def test_no_headings(self): + path = _write_md("Just some text.\n\n\n") + violations = check_file(path) + self.assertEqual(violations, []) + + def test_multiple_comments_between_headings(self): + path = _write_md( + "# Parent\n\n\n\n\n\n## Child\n\nText.\n" + ) + violations = check_file(path) + self.assertEqual(violations, []) + + +if __name__ == "__main__": + unittest.main()