From e26db125fa1fce0e29d6c578ab2e3503c127e015 Mon Sep 17 00:00:00 2001 From: Taranum01 Date: Mon, 28 Sep 2026 04:22:23 +0530 Subject: [PATCH] fix: correctly handle headings after closing code fences in MarkdownHeaderSplitter (#12954) --- .../preprocessors/markdown_header_splitter.py | 76 ++++++++++++++----- ...rkdownHeaderSplitter-94316fc2a373c50a.yaml | 10 +++ .../test_markdown_header_splitter.py | 34 +++++++++ 3 files changed, 99 insertions(+), 21 deletions(-) create mode 100644 releasenotes/notes/treat-closing-fence-longer-than-opener-as-code-block-in-MarkdownHeaderSplitter-94316fc2a373c50a.yaml diff --git a/haystack/components/preprocessors/markdown_header_splitter.py b/haystack/components/preprocessors/markdown_header_splitter.py index e6ffc38eb1..c526187fa0 100644 --- a/haystack/components/preprocessors/markdown_header_splitter.py +++ b/haystack/components/preprocessors/markdown_header_splitter.py @@ -78,25 +78,13 @@ def __init__( self.header_split_levels = header_split_levels self._header_split_levels_set = set(header_split_levels) self._header_pattern = re.compile(r"(?m)^(#{1,6}) (.+)$") # ATX-style .md-headers - - # Matches fenced code blocks delimited by triple backticks (```) or triple tildes (~~~). - # Broken down: - # ^ - fence must start at the beginning of a line (MULTILINE) - # (?P`{3,}|~{3,}) - # - named capture group "fence": three or more backticks OR three or - # more tildes. Capturing it allows the closing fence to be matched - # with a backreference, so ```-opened blocks must close with ``` - # and ~~~-opened blocks must close with ~~~. - # [^\n]* - optional language identifier (e.g. "python") and any other text - # on the opening fence line, up to the newline - # \n - newline ending the opening fence line - # .*? - the code block body, matched lazily (DOTALL so . matches newlines) - # ^(?P=fence) - closing fence: must be identical to the opening fence (backreference), - # and must start at the beginning of a line - # \s*$ - optional trailing whitespace after the closing fence - self._code_block_pattern = re.compile( - r"^(?P`{3,}|~{3,})[^\n]*\n.*?^(?P=fence)\s*$", re.MULTILINE | re.DOTALL - ) + # Per-line patterns used by `_code_block_spans` to recognise fenced code blocks. The + # closing-fence length rule (≥ opening-fence length, same character) is enforced in + # `_code_block_spans` rather than via a regex backreference, since CommonMark allows + # the closing fence to be longer than the opening one and a single regex cannot express + # that constraint: https://spec.commonmark.org/0.31.2/#fenced-code-blocks . + self._opening_fence_pattern = re.compile(r"(`{3,}|~{3,}).*") + self._closing_fence_pattern = re.compile(r"(`{3,}|~{3,})\s*$") self._is_warmed_up = False @@ -118,8 +106,54 @@ def warm_up(self) -> None: self._is_warmed_up = True def _code_block_spans(self, text: str) -> list[tuple[int, int]]: - """Return the (start, end) character spans of all fenced code blocks in text.""" - return [(m.start(), m.end()) for m in self._code_block_pattern.finditer(text)] + """ + Return the (start, end) character spans of all fenced code blocks in text. + + A fenced code block begins with a line of three or more backticks or tildes (optionally + followed by an info string) and ends with a line that consists entirely of the same + character with at least as many repetitions as the opening fence, followed only by + whitespace. The spans cover the opening fence line through the end of the closing + fence line (exclusive of its trailing newline) so that header matches inside the body + fall within ``[start, end)``. + """ + spans: list[tuple[int, int]] = [] + # Absolute offset of each line's first character; the last entry sits past the final + # newline so we can compute end offsets without special-casing the tail. + line_starts = [0] + for idx, ch in enumerate(text): + if ch == "\n": + line_starts.append(idx + 1) + + i = 0 + while i < len(line_starts): + line_start = line_starts[i] + line_end = line_starts[i + 1] - 1 if i + 1 < len(line_starts) else len(text) + opening = self._opening_fence_pattern.fullmatch(text[line_start:line_end]) + if opening is None: + i += 1 + continue + fence_char = text[line_start] + fence_len = 0 + while line_start + fence_len < line_end and text[line_start + fence_len] == fence_char: + fence_len += 1 + j = i + 1 + while j < len(line_starts): + close_start = line_starts[j] + close_end = line_starts[j + 1] - 1 if j + 1 < len(line_starts) else len(text) + closing = self._closing_fence_pattern.fullmatch(text[close_start:close_end]) + if closing is not None and text[close_start] == fence_char: + close_len = 0 + while close_start + close_len < close_end and text[close_start + close_len] == fence_char: + close_len += 1 + if close_len >= fence_len: + spans.append((line_start, close_end)) + i = j + 1 + break + j += 1 + else: + i += 1 + + return spans def _split_text_by_markdown_headers(self, text: str, doc_id: str) -> list[dict]: """ diff --git a/releasenotes/notes/treat-closing-fence-longer-than-opener-as-code-block-in-MarkdownHeaderSplitter-94316fc2a373c50a.yaml b/releasenotes/notes/treat-closing-fence-longer-than-opener-as-code-block-in-MarkdownHeaderSplitter-94316fc2a373c50a.yaml new file mode 100644 index 0000000000..747ccf5f20 --- /dev/null +++ b/releasenotes/notes/treat-closing-fence-longer-than-opener-as-code-block-in-MarkdownHeaderSplitter-94316fc2a373c50a.yaml @@ -0,0 +1,10 @@ +--- +fixes: + - | + ``MarkdownHeaderSplitter`` now treats fenced code blocks whose closing fence is + longer than the opening fence as code blocks (the CommonMark rule, where the + closing fence must be of the same character with at least as many repetitions + as the opener). Previously the closing fence had to match the opener exactly, + so a four-backtick opener with a five-backtick closer was not recognised as + a code block and any ``#`` lines inside it were incorrectly treated as + Markdown headers, splitting the code block apart. \ No newline at end of file diff --git a/test/components/preprocessors/test_markdown_header_splitter.py b/test/components/preprocessors/test_markdown_header_splitter.py index 237431ee8d..2959a2dfc7 100644 --- a/test/components/preprocessors/test_markdown_header_splitter.py +++ b/test/components/preprocessors/test_markdown_header_splitter.py @@ -641,6 +641,40 @@ def test_code_block_with_no_real_headers(self): assert docs[0].content == text assert "header" not in docs[0].meta + def test_closing_fence_longer_than_opening(self): + """CommonMark allows the closing fence to be longer than the opening fence (same character, ≥ length). + + Regression for #12954: a header-like line inside such a block was being treated as a real header, + splitting the code block in half. + """ + splitter = MarkdownHeaderSplitter() + + backtick_text = "# Intro\nBody\n````python\n# fake\n`````\n## Real\nEnd\n" + docs = splitter.run(documents=[Document(content=backtick_text)])["documents"] + assert [doc.meta["header"] for doc in docs] == ["Intro", "Real"] + assert "fake" not in [doc.meta["header"] for doc in docs] + # the chunk slice still preserves the code block byte-exactly under the real header + assert docs[0].content == "# Intro\nBody\n````python\n# fake\n`````\n" + assert docs[1].content == "## Real\nEnd\n" + + tilde_text = "# Intro\nBody\n~~~~python\n# fake\n~~~~~\n## Real\nEnd\n" + docs = splitter.run(documents=[Document(content=tilde_text)])["documents"] + assert [doc.meta["header"] for doc in docs] == ["Intro", "Real"] + assert "fake" not in [doc.meta["header"] for doc in docs] + assert docs[0].content == "# Intro\nBody\n~~~~python\n# fake\n~~~~~\n" + assert docs[1].content == "## Real\nEnd\n" + + def test_closing_fence_longer_when_header_precedes_fence(self): + """A header that immediately precedes a code fence still splits; the fence scan starts on the opener line.""" + text = "# Header\n```python\n# not a header\n`````\n## Real\nEnd\n" + splitter = MarkdownHeaderSplitter() + docs = splitter.run(documents=[Document(content=text)])["documents"] + + assert [doc.meta["header"] for doc in docs] == ["Header", "Real"] + assert "not a header" not in [doc.meta["header"] for doc in docs] + assert docs[0].content == "# Header\n```python\n# not a header\n`````\n" + assert docs[1].content == "## Real\nEnd\n" + def test_invalid_secondary_split_at_init(): """Test that an invalid secondary split type raises an error at initialization time."""