Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions changelog.d/wzgu-breadcrumb.fixed.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
**ExxonMobil's 10-K Item 7 opened on a "Table of Contents" breadcrumb again in 5.59.0.** Its page header is a one-row table, and once section tables rendered cell by cell the two labels arrived on one line, which the breadcrumb stripper did not recognise. A line whose cells are all navigation labels or page numbers is now stripped; Item 7 opens on its MD&A heading. (bead edgartools-wzgu)
18 changes: 17 additions & 1 deletion edgar/documents/extractors/toc_section_extractor.py
Original file line number Diff line number Diff line change
Expand Up @@ -435,6 +435,22 @@ def _find_supplement_start(self, positions: Dict[str, int], lo: int, hi: int):
# re-attributed body heading, plus the page number that often trails it.
_NAV_TEXT_RE = re.compile(r'^(?:table of contents|financial table of contents)$', re.IGNORECASE)
_NAV_NUM_RE = re.compile(r'^\d{1,4}$')
# A rendered table row separates its cells with two or more spaces, so a
# breadcrumb laid out as a one-row table arrives as a single line.
_TABLE_CELL_SEP_RE = re.compile(r'\s{2,}')

def _is_nav_line(self, stripped: str) -> bool:
"""True for a breadcrumb line: navigation labels, optionally with page numbers.

ExxonMobil's page header is a one-row table, "Table of Contents | Financial
Table of Contents". Once tables rendered cell by cell (edgartools-wzgu) it
arrived as one line, which the one-label-per-line match missed, so the
breadcrumb opened Item 7 again. At least one cell must be a label, so a
row of bare numbers is never mistaken for navigation.
"""
cells = self._TABLE_CELL_SEP_RE.split(stripped)
return (any(self._NAV_TEXT_RE.match(c) for c in cells)
and all(self._NAV_TEXT_RE.match(c) or self._NAV_NUM_RE.match(c) for c in cells))

def _strip_leading_nav(self, text: Optional[str]) -> Optional[str]:
"""Drop leading blank / navigation-breadcrumb lines from a section's text.
Expand All @@ -457,7 +473,7 @@ def _strip_leading_nav(self, text: Optional[str]) -> Optional[str]:
if not stripped:
i += 1
continue # blank lines don't reset breadcrumb adjacency
if self._NAV_TEXT_RE.match(stripped):
if self._is_nav_line(stripped):
prev_was_breadcrumb = True
i += 1
continue
Expand Down
2 changes: 1 addition & 1 deletion tests/issues/regression/test_issue_826.py
Original file line number Diff line number Diff line change
Expand Up @@ -63,7 +63,7 @@ def test_section_tables_no_duplication_aapl_10k():
assert len({id(t) for t in tables}) == len(tables)

# The fix only removes redundant serialization — it must not drop content.
assert len(section.text()) == 60874
assert len(section.text()) == 62223 # tables rendered cell by cell (edgartools-wzgu); same letters and digits


@pytest.mark.network
Expand Down
45 changes: 45 additions & 0 deletions tests/issues/regression/test_wzgu_breadcrumb_table_row.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,45 @@
"""A page-header breadcrumb laid out as a one-row table is stripped from a section's start.

Bead: edgartools-wzgu (follow-up). Found by the network test
tests/issues/regression/test_issue_rv86_incorporated_mda.py after 5.59.0.

ExxonMobil's 10-K puts "Table of Contents" and "Financial Table of Contents" in
one table row at the top of each page. ``_strip_leading_nav`` removes that
breadcrumb when a re-attributed anchor lands on it, matching one label per line.
Before the table-rendering fix those cells arrived on separate lines; once tables
rendered cell by cell they arrived as one line, "Table of Contents Financial
Table of Contents", which matched nothing, so XOM's Item 7 opened on the
breadcrumb again. Offline.
"""
import pytest

from edgar.documents.extractors.toc_section_extractor import SECSectionExtractor

pytestmark = pytest.mark.fast

BODY = "MANAGEMENT'S DISCUSSION AND ANALYSIS OF FINANCIAL CONDITION\n\nForward-looking statements"


def _strip(text):
extractor = SECSectionExtractor.__new__(SECSectionExtractor)
return extractor._strip_leading_nav(text)


@pytest.mark.parametrize("header", [
"Table of Contents Financial Table of Contents", # XOM's one-row table
"Table of Contents Financial Table of Contents 58", # with a page number cell
"Table of Contents\nFinancial Table of Contents", # the old line-per-cell form
"Table of Contents\n58", # label then page number
])
def test_breadcrumb_is_stripped(header):
assert _strip(f"{header}\n\n{BODY}") == BODY


@pytest.mark.parametrize("first_line", [
"2024 2023", # a row of bare numbers is data, not navigation
"2024", # a standalone number with no breadcrumb is kept
"Table of Contents Revenue by segment", # a label sharing a row with real text
])
def test_content_that_only_resembles_a_breadcrumb_is_kept(first_line):
text = f"{first_line}\n\n{BODY}"
assert _strip(text) == text
Loading