From b069ae1bb7d70c235c5d122d48f7f4f5b523378b Mon Sep 17 00:00:00 2001 From: Manohar Paturi <186662190+ManoharPaturi@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:07:48 +0530 Subject: [PATCH 1/2] fix: keep whitespace inside fenced code blocks when normalizing results MarkItDown._convert applied its whitespace cleanup (rstrip every line, collapse 3+ newlines to 2) to the entire converted document, which corrupted fenced code blocks: blank-line runs inside a fence collapsed and trailing spaces were stripped, changing code content. Track code fences while normalizing and leave their contents untouched. Signed-off-by: Manohar Paturi <186662190+ManoharPaturi@users.noreply.github.com> --- .../markitdown/src/markitdown/_markitdown.py | 73 ++++++++++++++++++- .../tests/test_code_fence_whitespace.py | 47 ++++++++++++ 2 files changed, 117 insertions(+), 3 deletions(-) create mode 100644 packages/markitdown/tests/test_code_fence_whitespace.py diff --git a/packages/markitdown/src/markitdown/_markitdown.py b/packages/markitdown/src/markitdown/_markitdown.py index 888d1e6260..f258630e0a 100644 --- a/packages/markitdown/src/markitdown/_markitdown.py +++ b/packages/markitdown/src/markitdown/_markitdown.py @@ -53,6 +53,74 @@ ) +# A fenced code block delimiter: three or more backticks or tildes, optionally +# indented by up to three spaces (CommonMark fence syntax). +_CODE_FENCE_RE = re.compile(r"^ {0,3}(`{3,}|~{3,})") + + +def _is_code_fence(line: str) -> Optional[re.Match]: + """Match a code fence delimiter at the start of a line, if any.""" + return _CODE_FENCE_RE.match(line) + + +def _closes_fence(line: str, fence_char: str, fence_len: int) -> bool: + """Whether a line inside a fenced code block closes the fence.""" + match = _is_code_fence(line) + if match is None: + return False + marker = match.group(1) + # The closing fence uses the same character, is at least as long, and + # carries no info string. + return marker[0] == fence_char and len(marker) >= fence_len and not line[match.end() :].strip() + + +def _normalize_whitespace_outside_code_fences(text: str) -> str: + """Normalize whitespace while leaving fenced code blocks untouched. + + Outside code fences, the previous normalization is applied: trailing + whitespace is stripped from each line and runs of three or more newlines + are collapsed to two. Content inside a fenced code block is + whitespace-significant, so it is passed through verbatim apart from + CRLF newline normalization. + """ + lines = re.split(r"\r?\n", text) + out: list[str] = [] + segment: list[str] = [] + fence_char = "" + fence_len = 0 + + def flush_segment() -> None: + if segment: + normalized = re.sub( + r"\n{3,}", "\n\n", "\n".join(line.rstrip() for line in segment) + ) + out.append(normalized) + segment.clear() + + for line in lines: + if fence_char: + # Inside a code fence: keep the line exactly as it is. + out.append(line) + if _closes_fence(line, fence_char, fence_len): + fence_char = "" + continue + + match = _is_code_fence(line) + if match is not None: + flush_segment() + # An opening fence may carry an info string (e.g. ```python). + marker = match.group(1) + fence_char = marker[0] + fence_len = len(marker) + out.append(line.rstrip()) + continue + + segment.append(line) + + flush_segment() + return "\n".join(out) + + def _get_content_disposition_filename(content_disposition: str) -> Optional[str]: message = Message() message["content-disposition"] = content_disposition @@ -683,10 +751,9 @@ def _convert( if res is not None: # Normalize the content - res.text_content = "\n".join( - [line.rstrip() for line in re.split(r"\r?\n", res.text_content)] + res.text_content = _normalize_whitespace_outside_code_fences( + res.text_content ) - res.text_content = re.sub(r"\n{3,}", "\n\n", res.text_content) return res # If we got this far without success, report any exceptions diff --git a/packages/markitdown/tests/test_code_fence_whitespace.py b/packages/markitdown/tests/test_code_fence_whitespace.py new file mode 100644 index 0000000000..0e6f8c3d6c --- /dev/null +++ b/packages/markitdown/tests/test_code_fence_whitespace.py @@ -0,0 +1,47 @@ +"""Regression test: whitespace inside fenced code blocks must survive conversion. + +MarkItDown._convert used to rstrip every line and collapse runs of 3+ newlines +across the whole result, which corrupts code blocks: two consecutive blank +lines inside a fence collapsed to one, and trailing spaces were stripped. +""" + +import io + +from markitdown import MarkItDown, StreamInfo + + +def _convert_markdown(content: bytes) -> str: + return ( + MarkItDown() + .convert( + io.BytesIO(content), + stream_info=StreamInfo(mimetype="text/markdown", extension=".md"), + ) + .markdown + ) + + +def test_blank_line_runs_inside_code_fence_are_preserved() -> None: + content = b"# Title\n\n```python\ndef a():\n pass\n\n\ndef b():\n pass\n```\n\nEnd\n" + result = _convert_markdown(content) + assert "pass\n\n\ndef b" in result + + +def test_trailing_spaces_inside_code_fence_are_preserved() -> None: + content = b"```text\nline with trailing spaces \n```\n" + result = _convert_markdown(content) + assert "line with trailing spaces \n" in result + + +def test_normalization_still_applies_outside_code_fence() -> None: + content = b"para \n\n\n\nother\n" + result = _convert_markdown(content) + assert result == "para\n\nother\n" + + +def test_tilde_fence_content_is_preserved() -> None: + # An info string ("text") keeps the sample away from charset misdetection + # of bare "~~~" lines (charset_normalizer reads them as shift_jis_2004). + content = b"~~~text\n\n\nkept\n~~~\n" + result = _convert_markdown(content) + assert "~~~text\n\n\nkept\n~~~" in result From 5c8e16d7e8589460f7a113e65bf2d328acbff5b3 Mon Sep 17 00:00:00 2001 From: Manohar Paturi <186662190+ManoharPaturi@users.noreply.github.com> Date: Fri, 18 Sep 2026 17:45:44 +0530 Subject: [PATCH 2/2] fix(docx): clamp Word Heading 7-9 to markdown h6 mammoth's default style map stops at Heading 6, so Word's deeper heading styles came through as plain paragraphs and their structure was dropped. Add style-map entries mapping Heading 7/8/9 to h6, the deepest level markdown supports. Signed-off-by: Manohar Paturi <186662190+ManoharPaturi@users.noreply.github.com> --- .../markitdown/converters/_docx_converter.py | 8 +++++++ .../tests/test_docx_heading_levels.py | 22 +++++++++++++++++++ 2 files changed, 30 insertions(+) create mode 100644 packages/markitdown/tests/test_docx_heading_levels.py diff --git a/packages/markitdown/src/markitdown/converters/_docx_converter.py b/packages/markitdown/src/markitdown/converters/_docx_converter.py index 8e5a736b8c..bc64864e54 100644 --- a/packages/markitdown/src/markitdown/converters/_docx_converter.py +++ b/packages/markitdown/src/markitdown/converters/_docx_converter.py @@ -29,6 +29,13 @@ _UNDERLINE_STYLE_MAP = "u => u" +# Word defines Heading 1-9; mammoth's default style map stops at Heading 6, so +# deeper headings came through as plain paragraphs and lost their structure. +# Markdown has no level past 6, so clamp 7-9 to h6. +_DEEP_HEADING_STYLE_MAP = "\n".join( + f"p[style-name='Heading {level}'] => h6:fresh" for level in (7, 8, 9) +) + def _read_embedded_style_map(file_stream: BinaryIO) -> Optional[str]: """Read the style map embedded in a .docx, if it has one.""" @@ -98,6 +105,7 @@ def convert( caller_style_map, embedded_style_map, _UNDERLINE_STYLE_MAP, + _DEEP_HEADING_STYLE_MAP, ) if part ) diff --git a/packages/markitdown/tests/test_docx_heading_levels.py b/packages/markitdown/tests/test_docx_heading_levels.py new file mode 100644 index 0000000000..3e91836f8f --- /dev/null +++ b/packages/markitdown/tests/test_docx_heading_levels.py @@ -0,0 +1,22 @@ +"""Word Heading 7-9 must clamp to markdown h6, not drop structure.""" +import io + +from docx import Document + +from markitdown import MarkItDown, StreamInfo + + +def test_docx_heading_levels_1_to_9(): + doc = Document() + for level in range(1, 10): + doc.add_heading(f"H{level}", level=level) + buf = io.BytesIO() + doc.save(buf) + buf.seek(0) + result = MarkItDown().convert_stream(buf, stream_info=StreamInfo(extension=".docx")) + lines = [line for line in result.markdown.splitlines() if line.strip()] + assert lines[0] == "# H1" + assert lines[5] == "###### H6" + # 7-9 clamp to h6 instead of losing heading structure entirely + assert lines[6] == "###### H7" + assert lines[8] == "###### H9"