Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
73 changes: 70 additions & 3 deletions packages/markitdown/src/markitdown/_markitdown.py
Original file line number Diff line number Diff line change
Expand Up @@ -53,6 +53,74 @@
)


# A fenced code block delimiter: three or more backticks or tildes, optionally
# indented by up to three spaces (CommonMark fence syntax).
_CODE_FENCE_RE = re.compile(r"^ {0,3}(`{3,}|~{3,})")


def _is_code_fence(line: str) -> Optional[re.Match]:
"""Match a code fence delimiter at the start of a line, if any."""
return _CODE_FENCE_RE.match(line)


def _closes_fence(line: str, fence_char: str, fence_len: int) -> bool:
"""Whether a line inside a fenced code block closes the fence."""
match = _is_code_fence(line)
if match is None:
return False
marker = match.group(1)
# The closing fence uses the same character, is at least as long, and
# carries no info string.
return marker[0] == fence_char and len(marker) >= fence_len and not line[match.end() :].strip()


def _normalize_whitespace_outside_code_fences(text: str) -> str:
"""Normalize whitespace while leaving fenced code blocks untouched.

Outside code fences, the previous normalization is applied: trailing
whitespace is stripped from each line and runs of three or more newlines
are collapsed to two. Content inside a fenced code block is
whitespace-significant, so it is passed through verbatim apart from
CRLF newline normalization.
"""
lines = re.split(r"\r?\n", text)
out: list[str] = []
segment: list[str] = []
fence_char = ""
fence_len = 0

def flush_segment() -> None:
if segment:
normalized = re.sub(
r"\n{3,}", "\n\n", "\n".join(line.rstrip() for line in segment)
)
out.append(normalized)
segment.clear()

for line in lines:
if fence_char:
# Inside a code fence: keep the line exactly as it is.
out.append(line)
if _closes_fence(line, fence_char, fence_len):
fence_char = ""
continue

match = _is_code_fence(line)
if match is not None:
flush_segment()
# An opening fence may carry an info string (e.g. ```python).
marker = match.group(1)
fence_char = marker[0]
fence_len = len(marker)
out.append(line.rstrip())
continue

segment.append(line)

flush_segment()
return "\n".join(out)


def _get_content_disposition_filename(content_disposition: str) -> Optional[str]:
message = Message()
message["content-disposition"] = content_disposition
Expand Down Expand Up @@ -683,10 +751,9 @@ def _convert(

if res is not None:
# Normalize the content
res.text_content = "\n".join(
[line.rstrip() for line in re.split(r"\r?\n", res.text_content)]
res.text_content = _normalize_whitespace_outside_code_fences(
res.text_content
)
res.text_content = re.sub(r"\n{3,}", "\n\n", res.text_content)
return res

# If we got this far without success, report any exceptions
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,13 @@

_UNDERLINE_STYLE_MAP = "u => u"

# Word defines Heading 1-9; mammoth's default style map stops at Heading 6, so
# deeper headings came through as plain paragraphs and lost their structure.
# Markdown has no level past 6, so clamp 7-9 to h6.
_DEEP_HEADING_STYLE_MAP = "\n".join(
f"p[style-name='Heading {level}'] => h6:fresh" for level in (7, 8, 9)
)


def _read_embedded_style_map(file_stream: BinaryIO) -> Optional[str]:
"""Read the style map embedded in a .docx, if it has one."""
Expand Down Expand Up @@ -98,6 +105,7 @@ def convert(
caller_style_map,
embedded_style_map,
_UNDERLINE_STYLE_MAP,
_DEEP_HEADING_STYLE_MAP,
)
if part
)
Expand Down
47 changes: 47 additions & 0 deletions packages/markitdown/tests/test_code_fence_whitespace.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,47 @@
"""Regression test: whitespace inside fenced code blocks must survive conversion.

MarkItDown._convert used to rstrip every line and collapse runs of 3+ newlines
across the whole result, which corrupts code blocks: two consecutive blank
lines inside a fence collapsed to one, and trailing spaces were stripped.
"""

import io

from markitdown import MarkItDown, StreamInfo


def _convert_markdown(content: bytes) -> str:
return (
MarkItDown()
.convert(
io.BytesIO(content),
stream_info=StreamInfo(mimetype="text/markdown", extension=".md"),
)
.markdown
)


def test_blank_line_runs_inside_code_fence_are_preserved() -> None:
content = b"# Title\n\n```python\ndef a():\n pass\n\n\ndef b():\n pass\n```\n\nEnd\n"
result = _convert_markdown(content)
assert "pass\n\n\ndef b" in result


def test_trailing_spaces_inside_code_fence_are_preserved() -> None:
content = b"```text\nline with trailing spaces \n```\n"
result = _convert_markdown(content)
assert "line with trailing spaces \n" in result


def test_normalization_still_applies_outside_code_fence() -> None:
content = b"para \n\n\n\nother\n"
result = _convert_markdown(content)
assert result == "para\n\nother\n"


def test_tilde_fence_content_is_preserved() -> None:
# An info string ("text") keeps the sample away from charset misdetection
# of bare "~~~" lines (charset_normalizer reads them as shift_jis_2004).
content = b"~~~text\n\n\nkept\n~~~\n"
result = _convert_markdown(content)
assert "~~~text\n\n\nkept\n~~~" in result
22 changes: 22 additions & 0 deletions packages/markitdown/tests/test_docx_heading_levels.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,22 @@
"""Word Heading 7-9 must clamp to markdown h6, not drop structure."""
import io

from docx import Document

from markitdown import MarkItDown, StreamInfo


def test_docx_heading_levels_1_to_9():
doc = Document()
for level in range(1, 10):
doc.add_heading(f"H{level}", level=level)
buf = io.BytesIO()
doc.save(buf)
buf.seek(0)
result = MarkItDown().convert_stream(buf, stream_info=StreamInfo(extension=".docx"))
lines = [line for line in result.markdown.splitlines() if line.strip()]
assert lines[0] == "# H1"
assert lines[5] == "###### H6"
# 7-9 clamp to h6 instead of losing heading structure entirely
assert lines[6] == "###### H7"
assert lines[8] == "###### H9"