diff --git a/packages/markitdown/src/markitdown/converters/_docx_converter.py b/packages/markitdown/src/markitdown/converters/_docx_converter.py
index 8e5a736b8c..1e5d682e00 100644
--- a/packages/markitdown/src/markitdown/converters/_docx_converter.py
+++ b/packages/markitdown/src/markitdown/converters/_docx_converter.py
@@ -29,6 +29,19 @@
_UNDERLINE_STYLE_MAP = "u => u"
+# Mammoth's default style map covers Word's Heading 1-6, by style id and by
+# style name. Word goes up to Heading 9, and Markdown has no level past 6, so
+# clamp the rest rather than let them fall through as plain paragraphs. This is
+# lossy on purpose: Heading 6-9 all come out as h6. A caller's style_map or the
+# document's embedded one is applied first, so either can map them differently.
+_DEEP_HEADING_STYLE_MAP = """\
+p.Heading7 => h6:fresh
+p.Heading8 => h6:fresh
+p.Heading9 => h6:fresh
+p[style-name='Heading 7'] => h6:fresh
+p[style-name='Heading 8'] => h6:fresh
+p[style-name='Heading 9'] => h6:fresh"""
+
def _read_embedded_style_map(file_stream: BinaryIO) -> Optional[str]:
"""Read the style map embedded in a .docx, if it has one."""
@@ -98,6 +111,7 @@ def convert(
caller_style_map,
embedded_style_map,
_UNDERLINE_STYLE_MAP,
+ _DEEP_HEADING_STYLE_MAP,
)
if part
)
diff --git a/packages/markitdown/tests/test_docx_headings.py b/packages/markitdown/tests/test_docx_headings.py
new file mode 100644
index 0000000000..c3cca0c1e6
--- /dev/null
+++ b/packages/markitdown/tests/test_docx_headings.py
@@ -0,0 +1,113 @@
+"""Word headings deeper than level 6 clamp to h6 rather than losing structure."""
+
+import io
+import zipfile
+
+import pytest
+
+from markitdown import MarkItDown, StreamInfo
+
+
+WORD_NAMESPACE = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
+STYLES_RELATIONSHIP_TYPE = (
+ "http://schemas.openxmlformats.org/officeDocument/2006/relationships/styles"
+)
+LEVELS = range(1, 10)
+
+
+def _headings_docx(style_id_format: str, *, stylesheet: bool) -> io.BytesIO:
+ """Build a minimal .docx holding one paragraph per Word heading level."""
+ style_ids = {level: style_id_format.format(level=level) for level in LEVELS}
+ body = "".join(
+ f''
+ f"Level {level}"
+ for level, style_id in style_ids.items()
+ )
+ styles_relationship = (
+ f''
+ if stylesheet
+ else ""
+ )
+
+ stream = io.BytesIO()
+ with zipfile.ZipFile(stream, "w") as archive:
+ archive.writestr(
+ "[Content_Types].xml",
+ """
+
+
+
+
+""",
+ )
+ archive.writestr(
+ "_rels/.rels",
+ """
+
+
+""",
+ )
+ archive.writestr(
+ "word/_rels/document.xml.rels",
+ f"""
+
+ {styles_relationship}
+""",
+ )
+ archive.writestr(
+ "word/document.xml",
+ f"""
+
+ {body}
+""",
+ )
+ if stylesheet:
+ styles = "".join(
+ f''
+ f''
+ for level, style_id in style_ids.items()
+ )
+ archive.writestr(
+ "word/styles.xml",
+ f"""
+{styles}""",
+ )
+
+ stream.seek(0)
+ return stream
+
+
+@pytest.mark.parametrize(
+ ("style_id_format", "stylesheet"),
+ [
+ # The style ids Word itself writes, matched without a stylesheet ...
+ ("Heading{level}", False),
+ # ... and the ids a localized Word writes, matched by their style name.
+ ("Titre{level}", True),
+ ],
+)
+def test_docx_headings_past_level_six_are_clamped(
+ style_id_format: str, stylesheet: bool
+) -> None:
+ docx_stream = _headings_docx(style_id_format, stylesheet=stylesheet)
+
+ result = MarkItDown().convert_stream(
+ docx_stream, stream_info=StreamInfo(extension=".docx")
+ )
+
+ assert result.markdown.split("\n\n") == [
+ f"{'#' * min(level, 6)} Level {level}" for level in LEVELS
+ ]
+
+
+def test_docx_caller_style_map_overrides_deep_heading_clamp() -> None:
+ # A caller-supplied style map still outranks the Heading 7-9 clamp.
+ docx_stream = _headings_docx("Heading{level}", stylesheet=False)
+
+ result = MarkItDown(style_map="p.Heading7 => p:fresh").convert_stream(
+ docx_stream, stream_info=StreamInfo(extension=".docx")
+ )
+
+ assert "\n\nLevel 7\n\n" in result.markdown
+ assert "###### Level 7" not in result.markdown