diff --git a/changelog.d/wzgu-breadcrumb.fixed.md b/changelog.d/wzgu-breadcrumb.fixed.md new file mode 100644 index 000000000..f3fe4b311 --- /dev/null +++ b/changelog.d/wzgu-breadcrumb.fixed.md @@ -0,0 +1 @@ +**ExxonMobil's 10-K Item 7 opened on a "Table of Contents" breadcrumb again in 5.59.0.** Its page header is a one-row table, and once section tables rendered cell by cell the two labels arrived on one line, which the breadcrumb stripper did not recognise. A line whose cells are all navigation labels or page numbers is now stripped; Item 7 opens on its MD&A heading. (bead edgartools-wzgu) diff --git a/edgar/documents/extractors/toc_section_extractor.py b/edgar/documents/extractors/toc_section_extractor.py index fa756d7fc..31f167b7a 100644 --- a/edgar/documents/extractors/toc_section_extractor.py +++ b/edgar/documents/extractors/toc_section_extractor.py @@ -435,6 +435,22 @@ def _find_supplement_start(self, positions: Dict[str, int], lo: int, hi: int): # re-attributed body heading, plus the page number that often trails it. _NAV_TEXT_RE = re.compile(r'^(?:table of contents|financial table of contents)$', re.IGNORECASE) _NAV_NUM_RE = re.compile(r'^\d{1,4}$') + # A rendered table row separates its cells with two or more spaces, so a + # breadcrumb laid out as a one-row table arrives as a single line. + _TABLE_CELL_SEP_RE = re.compile(r'\s{2,}') + + def _is_nav_line(self, stripped: str) -> bool: + """True for a breadcrumb line: navigation labels, optionally with page numbers. + + ExxonMobil's page header is a one-row table, "Table of Contents | Financial + Table of Contents". Once tables rendered cell by cell (edgartools-wzgu) it + arrived as one line, which the one-label-per-line match missed, so the + breadcrumb opened Item 7 again. At least one cell must be a label, so a + row of bare numbers is never mistaken for navigation. + """ + cells = self._TABLE_CELL_SEP_RE.split(stripped) + return (any(self._NAV_TEXT_RE.match(c) for c in cells) + and all(self._NAV_TEXT_RE.match(c) or self._NAV_NUM_RE.match(c) for c in cells)) def _strip_leading_nav(self, text: Optional[str]) -> Optional[str]: """Drop leading blank / navigation-breadcrumb lines from a section's text. @@ -457,7 +473,7 @@ def _strip_leading_nav(self, text: Optional[str]) -> Optional[str]: if not stripped: i += 1 continue # blank lines don't reset breadcrumb adjacency - if self._NAV_TEXT_RE.match(stripped): + if self._is_nav_line(stripped): prev_was_breadcrumb = True i += 1 continue diff --git a/tests/issues/regression/test_issue_826.py b/tests/issues/regression/test_issue_826.py index 1504d7d2e..7dca917bd 100644 --- a/tests/issues/regression/test_issue_826.py +++ b/tests/issues/regression/test_issue_826.py @@ -63,7 +63,7 @@ def test_section_tables_no_duplication_aapl_10k(): assert len({id(t) for t in tables}) == len(tables) # The fix only removes redundant serialization — it must not drop content. - assert len(section.text()) == 60874 + assert len(section.text()) == 62223 # tables rendered cell by cell (edgartools-wzgu); same letters and digits @pytest.mark.network diff --git a/tests/issues/regression/test_wzgu_breadcrumb_table_row.py b/tests/issues/regression/test_wzgu_breadcrumb_table_row.py new file mode 100644 index 000000000..2388ab39a --- /dev/null +++ b/tests/issues/regression/test_wzgu_breadcrumb_table_row.py @@ -0,0 +1,45 @@ +"""A page-header breadcrumb laid out as a one-row table is stripped from a section's start. + +Bead: edgartools-wzgu (follow-up). Found by the network test +tests/issues/regression/test_issue_rv86_incorporated_mda.py after 5.59.0. + +ExxonMobil's 10-K puts "Table of Contents" and "Financial Table of Contents" in +one table row at the top of each page. ``_strip_leading_nav`` removes that +breadcrumb when a re-attributed anchor lands on it, matching one label per line. +Before the table-rendering fix those cells arrived on separate lines; once tables +rendered cell by cell they arrived as one line, "Table of Contents Financial +Table of Contents", which matched nothing, so XOM's Item 7 opened on the +breadcrumb again. Offline. +""" +import pytest + +from edgar.documents.extractors.toc_section_extractor import SECSectionExtractor + +pytestmark = pytest.mark.fast + +BODY = "MANAGEMENT'S DISCUSSION AND ANALYSIS OF FINANCIAL CONDITION\n\nForward-looking statements" + + +def _strip(text): + extractor = SECSectionExtractor.__new__(SECSectionExtractor) + return extractor._strip_leading_nav(text) + + +@pytest.mark.parametrize("header", [ + "Table of Contents Financial Table of Contents", # XOM's one-row table + "Table of Contents Financial Table of Contents 58", # with a page number cell + "Table of Contents\nFinancial Table of Contents", # the old line-per-cell form + "Table of Contents\n58", # label then page number +]) +def test_breadcrumb_is_stripped(header): + assert _strip(f"{header}\n\n{BODY}") == BODY + + +@pytest.mark.parametrize("first_line", [ + "2024 2023", # a row of bare numbers is data, not navigation + "2024", # a standalone number with no breadcrumb is kept + "Table of Contents Revenue by segment", # a label sharing a row with real text +]) +def test_content_that_only_resembles_a_breadcrumb_is_kept(first_line): + text = f"{first_line}\n\n{BODY}" + assert _strip(text) == text