diff --git a/packages/markitdown/src/markitdown/converters/_pdf_converter.py b/packages/markitdown/src/markitdown/converters/_pdf_converter.py index ffbcbd990..80e82db34 100644 --- a/packages/markitdown/src/markitdown/converters/_pdf_converter.py +++ b/packages/markitdown/src/markitdown/converters/_pdf_converter.py @@ -10,6 +10,63 @@ # Pattern for MasterFormat-style partial numbering (e.g., ".1", ".2", ".10") PARTIAL_NUMBERING_PATTERN = re.compile(r"^\.\d+$") +# Characters that legitimately end a sentence/clause. A paragraph break is +# only treated as a genuine paragraph boundary if the preceding text ends +# with one of these. +_SENTENCE_END_CHARS = ".!?:;\"'”’" + +# Short lettered/numbered list markers (e.g. "a)", "(i)", "1.") that must +# stay on their own paragraph even though they start with a lowercase letter. +# No trailing whitespace is required after the marker, since PDF extraction +# sometimes drops the space between a list marker and its text (e.g. "a)text"). +_LIST_MARKER_PATTERN = re.compile(r"^\(?[A-Za-z0-9]{1,4}[.)]") + + +def _is_lowercase_continuation(para_stripped: str) -> bool: + """ + True if `para_stripped` reads as a continuation of the previous sentence: + it starts (optionally after a single opening parenthesis) with a + lowercase letter, and isn't itself a list marker like "a)" or "(i)". + """ + if not para_stripped or _LIST_MARKER_PATTERN.match(para_stripped): + return False + + first_char = para_stripped[1:2] if para_stripped[0] == "(" else para_stripped[:1] + return first_char.islower() + + +def _merge_wrapped_paragraph_breaks(text: str) -> str: + """ + Post-process extracted text to merge paragraph breaks that PDF layout + analysis (pdfminer/pdfplumber) sometimes inserts in the middle of a + sentence because of small, spurious variations in line spacing. + + A genuine paragraph never starts with a lowercase letter, so when a + blank-line-separated block starts with a lowercase word AND the previous + block doesn't end with sentence-ending punctuation, the "paragraph + break" is actually a wrapped line from the same sentence -- join the two + with a single space instead of a blank line. + """ + paragraphs = text.split("\n\n") + if len(paragraphs) < 2: + return text + + merged = [paragraphs[0]] + for para in paragraphs[1:]: + prev_stripped = merged[-1].rstrip() + para_stripped = para.lstrip() + + if ( + prev_stripped + and prev_stripped[-1] not in _SENTENCE_END_CHARS + and _is_lowercase_continuation(para_stripped) + ): + merged[-1] = f"{prev_stripped} {para_stripped}" + else: + merged.append(para) + + return "\n\n".join(merged) + def _merge_partial_numbering_lines(text: str) -> str: """ @@ -586,4 +643,7 @@ def convert( # Post-process to merge MasterFormat-style partial numbering with following text markdown = _merge_partial_numbering_lines(markdown) + # Post-process to merge paragraph breaks spuriously inserted mid-sentence + markdown = _merge_wrapped_paragraph_breaks(markdown) + return DocumentConverterResult(markdown=markdown) diff --git a/packages/markitdown/tests/test_pdf_wrapped_paragraph_breaks.py b/packages/markitdown/tests/test_pdf_wrapped_paragraph_breaks.py new file mode 100644 index 000000000..c3e5a2c40 --- /dev/null +++ b/packages/markitdown/tests/test_pdf_wrapped_paragraph_breaks.py @@ -0,0 +1,76 @@ +#!/usr/bin/env python3 -m pytest +"""Tests for merging paragraph breaks spuriously inserted mid-sentence by PDF +layout analysis (see: https://github.com/microsoft/markitdown/issues/2370).""" + +from markitdown.converters._pdf_converter import _merge_wrapped_paragraph_breaks + + +class TestMergeWrappedParagraphBreaks: + def test_merges_break_before_lowercase_continuation(self): + """A blank-line break followed by a lowercase word is a wrapped line, + not a real paragraph boundary, and should be joined with a space.""" + text = ( + "– Federal Decree-Law No. 47 of 2022 on the Taxation of " + "Corporations and Businesses, and\n\nits amendments," + ) + assert _merge_wrapped_paragraph_breaks(text) == ( + "– Federal Decree-Law No. 47 of 2022 on the Taxation of " + "Corporations and Businesses, and its amendments," + ) + + def test_merges_break_before_parenthetical_continuation(self): + """A break before a lowercase parenthetical aside is also a wrap.""" + text = ( + "distribution, warehousing, logistics or inventory management " + "functions constitutes 51%\n\n(fifty one percent) or more of " + "their Revenue for the relevant Tax Period." + ) + assert _merge_wrapped_paragraph_breaks(text) == ( + "distribution, warehousing, logistics or inventory management " + "functions constitutes 51% (fifty one percent) or more of " + "their Revenue for the relevant Tax Period." + ) + + def test_does_not_merge_real_paragraph_break(self): + """A real paragraph boundary -- previous text ends with terminal + punctuation and/or the next block starts with an uppercase letter -- + must be left untouched.""" + text = ( + "This is the end of a sentence.\n\n" + "This is a new paragraph that starts with a capital letter." + ) + assert _merge_wrapped_paragraph_breaks(text) == text + + def test_does_not_merge_when_next_block_is_uppercase(self): + """Even without terminal punctuation, an uppercase-starting next + block (e.g. a heading or list item) is left alone, since only a + lowercase start is an unambiguous continuation signal.""" + text = "Some heading fragment\n\nNext Heading" + assert _merge_wrapped_paragraph_breaks(text) == text + + def test_single_paragraph_is_unchanged(self): + text = "Just one paragraph with no breaks." + assert _merge_wrapped_paragraph_breaks(text) == text + + def test_does_not_merge_lettered_list_items(self): + """Lettered list markers (e.g. "a)", "b)") start with a lowercase + letter but must stay separate paragraphs, not get glued together.""" + text = "a) first clause\n\nb) second clause" + assert _merge_wrapped_paragraph_breaks(text) == text + + def test_does_not_merge_parenthetical_list_items(self): + text = "some heading\n\n(i) first item\n\n(ii) second item" + assert _merge_wrapped_paragraph_breaks(text) == text + + def test_does_not_merge_numeric_paragraph_into_next(self): + """A paragraph that is just a number must not glue onto a following + paragraph merely because that paragraph's first letter is lowercase.""" + text = "42\n\n7 apples were purchased." + assert _merge_wrapped_paragraph_breaks(text) == text + + def test_does_not_merge_list_markers_with_no_space_after_marker(self): + """PDF extraction sometimes drops the space after a list marker + (e.g. "a)text" instead of "a) text"); the marker must still be + recognized and kept on its own paragraph.""" + text = "some heading\n\na)first clause\n\nb)second clause" + assert _merge_wrapped_paragraph_breaks(text) == text