From 6f188b5331368abe132703def35863cdda453017 Mon Sep 17 00:00:00 2001 From: linxi123-A <3951574582@qq.com> Date: Sun, 13 Sep 2026 13:04:03 +0800 Subject: [PATCH 1/3] fix: stop heading pass from splitting last character off heading lines The "ensure blank line after heading" regex matched (heading)([^\n]), whose second group can never be the newline that follows the heading line, so backtracking consumed the heading's own last character instead: "# Title" became "# Titl\n\ne", render_html emitted
e
, and re-running enhance() kept shredding the tail (non-idempotent). Anchor the pass to true line starts ((?m)^), require the literal trailing newline, and insert a newline only when the next line is non-empty (lookahead), which makes the pass idempotent and purely insertive. Adds 6 heading-integrity regression tests (all failed on the previous implementation, all pass now). --- tests/test_markdown_formatter.py | 48 ++++++++++++++++++++++++++++++-- utils/markdown_formatter.py | 5 +++- 2 files changed, 50 insertions(+), 3 deletions(-) diff --git a/tests/test_markdown_formatter.py b/tests/test_markdown_formatter.py index 77e5bf69..be4e3a2c 100644 --- a/tests/test_markdown_formatter.py +++ b/tests/test_markdown_formatter.py @@ -8,8 +8,9 @@ The assertions pin behavior verified against the current implementation that is also implied by the docstrings. The heading-spacing pass in ``enhance`` -has known quirks on isolated heading lines; that path is deliberately not -pinned here so a future fix to it stays green. +previously corrupted heading text (``"# Title"`` became ``"# Titl\n\ne"``); +it is fixed and covered by the dedicated "heading integrity" regression +tests at the end of this file. """ import pytest @@ -230,3 +231,46 @@ def test_render_html_escapes_angle_brackets_inside_code_blocks(): "if (a < b && c > d) {}\n" "" ) + + +# --------------------------------------------------------------------------- +# heading integrity (regression: the "blank line after heading" pass used to +# split the last character off every heading line, e.g. "# Title" became +# "# Titl\n\ne", and re-running enhance() kept shredding the tail further) +# --------------------------------------------------------------------------- + +def test_enhance_preserves_isolated_heading_text(): + # Before the fix: "# Titl\n\ne" + assert MarkdownFormatter.enhance("# Title") == "# Title" + + +def test_enhance_inserts_blank_line_between_heading_and_paragraph(): + # The pass may insert whitespace, but must never move existing characters. + # Before the fix: "# Titl\n\ne\nBody text" + assert MarkdownFormatter.enhance("# Title\nBody text") == "# Title\n\nBody text" + + +def test_enhance_heading_spacing_is_idempotent(): + # State-consistency check: enhancing an already-enhanced document must be + # a no-op. Before the fix the tail kept degrading on every pass + # ("Title" -> "Titl" + "e" -> "Tit" + "l" + "e" ...). + once = MarkdownFormatter.enhance("# Title\nBody text") + assert MarkdownFormatter.enhance(once) == once + + +def test_enhance_leaves_existing_blank_line_after_heading_alone(): + text = "# Title\n\nBody text" + assert MarkdownFormatter.enhance(text) == text + + +def test_enhance_leaves_heading_with_trailing_newline_alone(): + text = "# Title\n" + assert MarkdownFormatter.enhance(text) == text + + +def test_render_html_keeps_heading_text_intact(): + # User-visible symptom: the corrupted heading leaked into rendered HTML. + # Before the fix: "e
" + html = MarkdownFormatter.render_html("# Title") + assert "e
" not in html diff --git a/utils/markdown_formatter.py b/utils/markdown_formatter.py index f9eb0ad2..1bd1d452 100644 --- a/utils/markdown_formatter.py +++ b/utils/markdown_formatter.py @@ -51,7 +51,10 @@ def enhance(text: str, default_lang: str = 'cpp') -> str: # 确保标题前后有空行 text = re.sub(r'([^\n])(#{1,6}\s)', r'\1\n\n\2', text) - text = re.sub(r'(#{1,6}[^\n]+)([^\n])', r'\1\n\n\2', text) + # 仅在标题行末(字面换行处)插入空行;lookahead 确保下一行非空, + # 避免重复插入(幂等)。不得用 [^\n] 匹配行尾之后的内容—— + # 那会回溯吞掉标题行内的字符(历史缺陷:标题末字符被拆走)。 + text = re.sub(r'(?m)^(#{1,6}[^\n]+\n)(?=[^\n])', r'\1\n', text) # 确保代码块格式正确 text = MarkdownFormatter._fix_code_blocks(text, default_lang) From 5312434b4d40ad9adb75316d1eab9d17f99f4339 Mon Sep 17 00:00:00 2001 From: linxi123-A <3951574582@qq.com> Date: Mon, 14 Sep 2026 22:41:48 +0800 Subject: [PATCH 2/3] test: pin inline-hash non-heading input to heading fix Follow-up to review on #47: the unanchored pre-fix heading pass also split an inline "#" line ("a #1 fan\nnext" -> "a #1 fa\n\nn\nnext"), the same data-loss mechanism as the standalone-heading symptom. Lock the (?m)^ line-start anchor behavior with a regression test so an inline hash stays byte-for-byte untouched. --- tests/test_markdown_formatter.py | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/tests/test_markdown_formatter.py b/tests/test_markdown_formatter.py index be4e3a2c..00f3560d 100644 --- a/tests/test_markdown_formatter.py +++ b/tests/test_markdown_formatter.py @@ -274,3 +274,12 @@ def test_render_html_keeps_heading_text_intact(): html = MarkdownFormatter.render_html("# Title") assert "e
" not in html + + +def test_enhance_does_not_treat_inline_hash_as_heading(): + # A "#" in the middle of a line is not a heading. The old unanchored + # heading pass matched it anyway and split the tail off (data loss): + # "a #1 fan\nnext" -> "a #1 fa\n\nn\nnext". The (?m)^ anchor of the + # fixed pass leaves this input byte-for-byte untouched. + text = "a #1 fan\nnext" + assert MarkdownFormatter.enhance(text) == text From b0d8a8d09dfdef399be630fce039c5360b823169 Mon Sep 17 00:00:00 2001 From: linxi123-A <3951574582@qq.com> Date: Thu, 17 Sep 2026 23:38:19 +0800 Subject: [PATCH 3/3] fix: anchor heading-spacing passes to line start for H2-H6 --- tests/test_markdown_formatter.py | 47 ++++++++++++++++++++++++++++++-- utils/markdown_formatter.py | 7 +++-- 2 files changed, 49 insertions(+), 5 deletions(-) diff --git a/tests/test_markdown_formatter.py b/tests/test_markdown_formatter.py index 00f3560d..a633ea00 100644 --- a/tests/test_markdown_formatter.py +++ b/tests/test_markdown_formatter.py @@ -234,9 +234,13 @@ def test_render_html_escapes_angle_brackets_inside_code_blocks(): # --------------------------------------------------------------------------- -# heading integrity (regression: the "blank line after heading" pass used to -# split the last character off every heading line, e.g. "# Title" became -# "# Titl\n\ne", and re-running enhance() kept shredding the tail further) +# heading integrity (two regressions in enhance()'s heading passes): +# 1) the "blank line AFTER heading" pass split the last character off every +# heading line, e.g. "# Title" became "# Titl\n\ne", and re-running +# enhance() kept shredding the tail further; +# 2) the "blank line BEFORE heading" pass matched inside multi-mark headings: +# "## Title" became an empty H1 plus an H1 ("#\n\n# Title"), and H3-H6 +# kept degrading on repeated runs. # --------------------------------------------------------------------------- def test_enhance_preserves_isolated_heading_text(): @@ -283,3 +287,40 @@ def test_enhance_does_not_treat_inline_hash_as_heading(): # fixed pass leaves this input byte-for-byte untouched. text = "a #1 fan\nnext" assert MarkdownFormatter.enhance(text) == text + + +@pytest.mark.parametrize("marks", ["##", "###", "####", "#####", "######"]) +def test_enhance_preserves_heading_level_and_text_h2_h6(marks): + # The "blank line BEFORE heading" pass used to match inside the heading + # itself: its first group ate the first "#" as preceding text and the + # remaining marks became the heading, so "## Title" degraded to an empty + # H1 plus an H1 ("#\n\n# Title"). Level and text must survive; the only + # allowed change is inserting one blank line after the heading. + text = f"{marks} Title\nBody text" + assert MarkdownFormatter.enhance(text) == f"{marks} Title\n\nBody text" + + +@pytest.mark.parametrize("marks", ["##", "###", "####", "#####", "######"]) +def test_enhance_h2_h6_idempotent(marks): + # State-consistency: H3-H6 used to keep degrading on repeated runs + # (e.g. "### Title" -> "#\n\n## Title" -> "#\n\n#\n\n# Title"). + once = MarkdownFormatter.enhance(f"{marks} Title\nBody text") + assert MarkdownFormatter.enhance(once) == once + + +def test_enhance_inserts_blank_line_between_paragraph_and_heading(): + # The legitimate job of the "blank line before heading" pass: a physical + # line-start heading preceded directly by a text line gets one blank + # line inserted, with no character moved or re-interpreted. + text = "Intro line\n## Title\nBody text" + assert MarkdownFormatter.enhance(text) == ( + "Intro line\n\n## Title\n\nBody text" + ) + + +def test_render_html_keeps_h2_heading_level_and_text(): + # User-visible symptom of the split: an H2 rendered as an empty H1 plus + # an H1 ("\n