diff --git a/packages/markitdown/src/markitdown/converters/_docx_converter.py b/packages/markitdown/src/markitdown/converters/_docx_converter.py index 8e5a736b8c..1e5d682e00 100644 --- a/packages/markitdown/src/markitdown/converters/_docx_converter.py +++ b/packages/markitdown/src/markitdown/converters/_docx_converter.py @@ -29,6 +29,19 @@ _UNDERLINE_STYLE_MAP = "u => u" +# Mammoth's default style map covers Word's Heading 1-6, by style id and by +# style name. Word goes up to Heading 9, and Markdown has no level past 6, so +# clamp the rest rather than let them fall through as plain paragraphs. This is +# lossy on purpose: Heading 6-9 all come out as h6. A caller's style_map or the +# document's embedded one is applied first, so either can map them differently. +_DEEP_HEADING_STYLE_MAP = """\ +p.Heading7 => h6:fresh +p.Heading8 => h6:fresh +p.Heading9 => h6:fresh +p[style-name='Heading 7'] => h6:fresh +p[style-name='Heading 8'] => h6:fresh +p[style-name='Heading 9'] => h6:fresh""" + def _read_embedded_style_map(file_stream: BinaryIO) -> Optional[str]: """Read the style map embedded in a .docx, if it has one.""" @@ -98,6 +111,7 @@ def convert( caller_style_map, embedded_style_map, _UNDERLINE_STYLE_MAP, + _DEEP_HEADING_STYLE_MAP, ) if part ) diff --git a/packages/markitdown/tests/test_docx_headings.py b/packages/markitdown/tests/test_docx_headings.py new file mode 100644 index 0000000000..c3cca0c1e6 --- /dev/null +++ b/packages/markitdown/tests/test_docx_headings.py @@ -0,0 +1,113 @@ +"""Word headings deeper than level 6 clamp to h6 rather than losing structure.""" + +import io +import zipfile + +import pytest + +from markitdown import MarkItDown, StreamInfo + + +WORD_NAMESPACE = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" +STYLES_RELATIONSHIP_TYPE = ( + "http://schemas.openxmlformats.org/officeDocument/2006/relationships/styles" +) +LEVELS = range(1, 10) + + +def _headings_docx(style_id_format: str, *, stylesheet: bool) -> io.BytesIO: + """Build a minimal .docx holding one paragraph per Word heading level.""" + style_ids = {level: style_id_format.format(level=level) for level in LEVELS} + body = "".join( + f'' + f"Level {level}" + for level, style_id in style_ids.items() + ) + styles_relationship = ( + f'' + if stylesheet + else "" + ) + + stream = io.BytesIO() + with zipfile.ZipFile(stream, "w") as archive: + archive.writestr( + "[Content_Types].xml", + """ + + + + +""", + ) + archive.writestr( + "_rels/.rels", + """ + + +""", + ) + archive.writestr( + "word/_rels/document.xml.rels", + f""" + + {styles_relationship} +""", + ) + archive.writestr( + "word/document.xml", + f""" + + {body} +""", + ) + if stylesheet: + styles = "".join( + f'' + f'' + for level, style_id in style_ids.items() + ) + archive.writestr( + "word/styles.xml", + f""" +{styles}""", + ) + + stream.seek(0) + return stream + + +@pytest.mark.parametrize( + ("style_id_format", "stylesheet"), + [ + # The style ids Word itself writes, matched without a stylesheet ... + ("Heading{level}", False), + # ... and the ids a localized Word writes, matched by their style name. + ("Titre{level}", True), + ], +) +def test_docx_headings_past_level_six_are_clamped( + style_id_format: str, stylesheet: bool +) -> None: + docx_stream = _headings_docx(style_id_format, stylesheet=stylesheet) + + result = MarkItDown().convert_stream( + docx_stream, stream_info=StreamInfo(extension=".docx") + ) + + assert result.markdown.split("\n\n") == [ + f"{'#' * min(level, 6)} Level {level}" for level in LEVELS + ] + + +def test_docx_caller_style_map_overrides_deep_heading_clamp() -> None: + # A caller-supplied style map still outranks the Heading 7-9 clamp. + docx_stream = _headings_docx("Heading{level}", stylesheet=False) + + result = MarkItDown(style_map="p.Heading7 => p:fresh").convert_stream( + docx_stream, stream_info=StreamInfo(extension=".docx") + ) + + assert "\n\nLevel 7\n\n" in result.markdown + assert "###### Level 7" not in result.markdown