From 931240c1b0d13950ac28c414efa0bfb129f10b65 Mon Sep 17 00:00:00 2001 From: simpleqt <89645338+simpleqt@users.noreply.github.com> Date: Mon, 28 Sep 2026 11:16:50 +0800 Subject: [PATCH] fix(chunkers): keep fenced code blocks out of markdown header repair MarkdownChunker counted '#'-comment lines inside fenced code blocks as level-1 markdown headers. Enough comments (five in a commented-out python snippet) tripped the malformed-hierarchy detector, and _fix_header_hierarchy then rewrote the comments into '## ...' lines inside the code block, corrupting embedded code in the produced chunks. Track CommonMark-style fences (backtick and tilde) in both the detector and the fixer and skip masked lines; code outside fences and real headers are handled exactly as before. Red/green verified: three new tests fail against the old implementation and pass with the fix; the pre-existing header-repair behavior is pinned by two further tests. --- src/memos/chunkers/markdown_chunker.py | 39 +++++++++- .../test_markdown_chunker_code_fence.py | 76 +++++++++++++++++++ 2 files changed, 113 insertions(+), 2 deletions(-) create mode 100644 tests/chunkers/test_markdown_chunker_code_fence.py diff --git a/src/memos/chunkers/markdown_chunker.py b/src/memos/chunkers/markdown_chunker.py index a37370200..bb2d30f25 100644 --- a/src/memos/chunkers/markdown_chunker.py +++ b/src/memos/chunkers/markdown_chunker.py @@ -75,12 +75,41 @@ def chunk(self, text: str, **kwargs) -> list[str] | list[Chunk]: logger.debug(f"Generated {len(chunks)} chunks from input text") return chunks + _FENCE_RE = re.compile(r"^\s{0,3}(`{3,}|~{3,})") + + @classmethod + def _code_block_mask(cls, lines: list[str]) -> list[bool]: + """Mark which lines sit inside a fenced code block. + + Fenced code (``` or ~~~, CommonMark-style) is tracked so that ``#`` + comments inside embedded code are never mistaken for markdown + headers. The fence lines themselves are masked as well. + """ + mask = [] + fence_char = None + for line in lines: + fence_match = cls._FENCE_RE.match(line) + if fence_match: + char = fence_match.group(1)[0] + if fence_char is None: + fence_char = char + elif char == fence_char: + fence_char = None + mask.append(True) + else: + mask.append(fence_char is not None) + return mask + def _detect_malformed_headers(self, text: str) -> bool: """Detect if markdown has improper header hierarchy usage.""" # Extract all valid markdown header lines header_levels = [] pattern = re.compile(r"^#{1,6}\s+.+") - for line in text.split("\n"): + lines = text.split("\n") + code_mask = self._code_block_mask(lines) + for line, in_code in zip(lines, code_mask, strict=True): + if in_code: + continue stripped_line = line.strip() if pattern.match(stripped_line): hash_match = re.match(r"^(#+)", stripped_line) @@ -123,10 +152,16 @@ def _fix_header_hierarchy(self, text: str) -> str: """ header_pattern = re.compile(r"^(#{1,6})\s+(.+)$") lines = text.split("\n") + code_mask = self._code_block_mask(lines) fixed_lines = [] first_valid_header = False - for line in lines: + for line, in_code in zip(lines, code_mask, strict=True): + if in_code: + # Fenced code must pass through untouched: `#` comments in + # embedded code are not headers. + fixed_lines.append(line) + continue stripped_line = line.strip() # Match valid header lines (invalid # lines kept as-is) header_match = header_pattern.match(stripped_line) diff --git a/tests/chunkers/test_markdown_chunker_code_fence.py b/tests/chunkers/test_markdown_chunker_code_fence.py new file mode 100644 index 000000000..85e0547d2 --- /dev/null +++ b/tests/chunkers/test_markdown_chunker_code_fence.py @@ -0,0 +1,76 @@ +import unittest + +from unittest.mock import patch + +from memos.chunkers.markdown_chunker import MarkdownChunker + + +class TestMarkdownChunkerCodeFence(unittest.TestCase): + """Fenced code blocks must never be treated as markdown headers. + + ``#``-comment lines inside embedded code (```python blocks, shell + scripts, ...) used to be counted as level-1 headers by + ``_detect_malformed_headers``; enough of them triggered the + "malformed hierarchy" repair, and ``_fix_header_hierarchy`` rewrote the + comments into ``## ...`` lines inside the code block, corrupting it. + """ + + def _chunker(self) -> MarkdownChunker: + with patch("langchain_text_splitters.MarkdownHeaderTextSplitter"): + return MarkdownChunker(config=None, auto_fix_headers=True) + + def test_code_block_only_has_no_headers(self): + text = "```python\n# one\n# two\n# three\n# four\n# five\nx = 1\n```\n" + chunker = self._chunker() + + self.assertFalse(chunker._detect_malformed_headers(text)) + + def test_fix_leaves_code_block_intact(self): + text = ( + "# Title\n\n" + "Intro.\n\n" + "```python\n" + "# comment one\n" + "# comment two\n" + "# comment three\n" + "# comment four\n" + "# comment five\n" + "x = 1\n" + "```\n" + ) + chunker = self._chunker() + + # the fixer must leave fenced code byte-for-byte intact + self.assertEqual(chunker._fix_header_hierarchy(text), text) + + def test_real_malformed_headers_still_fixed(self): + text = "# A\n# B\n# C\nbody\n" + chunker = self._chunker() + + self.assertTrue(chunker._detect_malformed_headers(text)) + fixed = chunker._fix_header_hierarchy(text) + self.assertIn("# A\n", fixed) + self.assertIn("## B\n", fixed) + self.assertIn("## C\n", fixed) + + def test_tilde_fence_ignored_and_real_headers_fixed(self): + text = "~~~\n# not a header\n~~~\n\n# Real One\n# Real Two\n" + chunker = self._chunker() + + # only the two real headers are counted, which is malformed + self.assertTrue(chunker._detect_malformed_headers(text)) + fixed = chunker._fix_header_hierarchy(text) + self.assertIn("~~~\n# not a header\n~~~", fixed) + self.assertIn("# Real One\n", fixed) + self.assertIn("## Real Two\n", fixed) + + def test_unclosed_fence_holds_to_end(self): + text = "```python\n# one\n# two\n# three\n# four\n# five\n" + chunker = self._chunker() + + self.assertFalse(chunker._detect_malformed_headers(text)) + self.assertEqual(chunker._fix_header_hierarchy(text), text) + + +if __name__ == "__main__": + unittest.main()