Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
62 changes: 56 additions & 6 deletions wikify/engine/loader/cleanup.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,7 @@
import re
import unicodedata
from collections import Counter
from itertools import groupby
from itertools import groupby, pairwise

from wikify.engine.loader.table_stitch import (
html_tables_to_markdown,
Expand Down Expand Up @@ -51,7 +51,7 @@
_SIGNOFF = ("prepared by", "issued by", "approved by", "reviewed by", "authorized by")
_SIGNOFF_LABEL = re.compile(r"(?i)\b(prepared|issued|approved|reviewed|authori[sz]ed) by\s*[:\-\u2013]")
_SIGNOFF_LINE = re.compile(r"(?i)^[*_\s]*(prepared|issued|approved|reviewed|authori[sz]ed) by\s*[:\-\u2013]")
_PAGE_NUMBER = re.compile(r"^\s*\d{1,4}\s*$")
_PAGE_NUMBER = re.compile(r"(?i)^\s*(?:(?:page|pg\.?)\s*)?\d{1,4}\s*$")
_SENTENCE_START = re.compile(r"^[a-z]")
# A sentence cut by a page break ran to the page edge, so its last line is never a short label.
_MIN_BROKEN_LINE_LENGTH = 40
Expand All @@ -70,6 +70,10 @@
MIN_TRANSCRIBED_ROWS = 4
TYPED_TABLE_WORD_SHARE = 0.6
_TABLE_WORD = re.compile(r"[^\W_]{3,}")
_HEADING_RE = re.compile(r"^(#{1,6})\s+(.*\S)\s*$")
_BREADCRUMB_SPLIT = re.compile(r"\s+>\s+")
_NOT_BREADCRUMB_START = ("#", "|", ">", "-", "*", "+")
BREADCRUMB_MIN_PAGES = 2


def _plain(line: str) -> str:
Expand Down Expand Up @@ -265,6 +269,50 @@ def strip_boilerplate(pages: list[tuple[int, str]], boilerplate: set[str]) -> li
return out


def markdown_heading(level: int, title: str) -> str:
return f"{'#' * min(6, level)} {title}"


def page_breadcrumb(lines: list[str]) -> tuple[int, int, str, list[str]] | None:
content = [index for index, line in enumerate(lines) if line.strip()]
for index, next_index in pairwise(content[:EDGE_LINES]):
text = lines[index].strip()
if text.startswith(_NOT_BREADCRUMB_START) or _STRUCTURAL_LINE.match(text):
continue
parts = [part.strip() for part in _BREADCRUMB_SPLIT.split(_plain(text))]
heading = _HEADING_RE.match(lines[next_index])
if len(parts) > 1 and all(parts) and heading and _norm(heading.group(2)) == _norm(parts[-1]):
return index, next_index, heading.group(2), parts
return None


def breadcrumbs_to_headings(pages: list[tuple[int, str]]) -> list[tuple[int, str]]:
page_lines = [md.splitlines() for _, md in pages]
crumbs = [page_breadcrumb(lines) for lines in page_lines]
if sum(1 for crumb in crumbs if crumb) < BREADCRUMB_MIN_PAGES:
return pages
previous: list[str] = []
for lines, crumb in zip(page_lines, crumbs, strict=True):
if not crumb:
continue
crumb_index, heading_index, title, parts = crumb
trail = [_norm(part) for part in parts]
opened = 0
for old, new in zip(previous, trail[:-1], strict=False):
if old != new:
break
opened += 1
lines[heading_index] = markdown_heading(len(parts), title)
lines[crumb_index] = "\n\n".join(
markdown_heading(depth + 1, parts[depth]) for depth in range(opened, len(parts) - 1)
)
previous = trail
return [
(page_no, "\n".join(lines).strip() if crumb else md)
for (page_no, md), lines, crumb in zip(pages, page_lines, crumbs, strict=True)
]


def _ends_mid_sentence(line: str) -> bool:
text = line.strip()
return (
Expand Down Expand Up @@ -449,10 +497,12 @@ def clean_pages(
) -> list[tuple[int, str]]:
pages = [(page_no, html_tables_to_markdown(md)) for page_no, md in pages]
page_texts = page_texts or {}
stripped = [
(page_no, contents_lines_to_table(drop_transcribed_figures(md, page_texts.get(page_no, ""))))
for page_no, md in strip_boilerplate(pages, find_boilerplate(pages))
]
stripped = breadcrumbs_to_headings(
[
(page_no, contents_lines_to_table(drop_transcribed_figures(md, page_texts.get(page_no, ""))))
for page_no, md in strip_boilerplate(pages, find_boilerplate(pages))
]
)
stitched = stitch_cross_page_tables(stripped)
joined = join_page_breaks([(page_no, merge_continuation_rows(md)) for page_no, md in stitched])
return reindent_list_continuations(joined)
Expand Down
11 changes: 11 additions & 0 deletions wikify/engine/loader/page_plan.py
Original file line number Diff line number Diff line change
Expand Up @@ -128,6 +128,8 @@ def enforce_invariants(
pages.add(index)
elif index in chosen and parent in pages:
pages.add(index)
if not pages:
pages = pdf_page_openers(sections, parents)
pages = add_numbered_siblings(parents, numbers, pages)
pages, parents = split_long_pages(sections, parents, numbers, pages)
pages, parents = keep_sibling_order(sections, parents, numbers, pages)
Expand All @@ -143,6 +145,15 @@ def enforce_invariants(
return pages or {0}, parents


def pdf_page_openers(sections: list[Section], parents: list[int | None]) -> set[int]:
openers = {
index
for index, parent in enumerate(parents)
if index and parent is None and (index == 1 or opens_pdf_page(sections, index))
}
return openers if len(openers) > 1 else set()


def add_numbered_siblings(
parents: list[int | None], numbers: list[tuple[int, ...] | None], pages: set[int]
) -> set[int]:
Expand Down
27 changes: 22 additions & 5 deletions wikify/engine/loader/sectionizer.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@
from dataclasses import dataclass
from itertools import pairwise

from wikify.engine.loader.cleanup import _SEP_ONLY
from wikify.engine.loader.cleanup import _HEADING_RE, _SEP_ONLY
from wikify.engine.loader.toc import correct_level

MAX_TITLE_LENGTH = 140
Expand All @@ -15,7 +15,6 @@
RUNNING_HEADER_MIN_PAGES = 3
TOP_OF_PAGE_LINES = 3

_HEADING_RE = re.compile(r"^(#{1,6})\s+(.*\S)\s*$")
_NUM_RE = re.compile(r"^(\d+(?:\.\d+)*)\.?\s+\S")
_LEADING_NUM = re.compile(r"^(\d+)\b")
_DOUBLE_NUM = re.compile(r"^\d+\.\s+\d")
Expand All @@ -24,6 +23,8 @@
_LINK = re.compile(r"\[([^\]]*)\]\([^)]*\)")
_HTML_TAG = re.compile(r"</?[A-Za-z][^>]*>")
_LETTERED_NUM = re.compile(r"^(\d+(?:\.\d+)+)\.[a-z](?![a-z])\s*")
_CONTINUED = re.compile(r"(?i)^(.*?)\s*\((?:continued|cont'd|cont\.?)\)$")
_CONTINUED_FROM = re.compile(r"(?i)^continued(?: from (?:the )?previous page)?$")
_TOC_NUMBERED = re.compile(r"^(\d+(?:\.\d+)*)\.?\s+(.*?)(?:\s*\.{3,})?(?:\s+\d{1,4})?$")
_CONTENTS_TITLE = re.compile(r"(?i)^(?:table of )?contents$|^index$")
_TOC_ENTRY = re.compile(
Expand Down Expand Up @@ -94,6 +95,16 @@ def _repeats_open_section(title: str, stack: list[tuple[int, str, bool]]) -> boo
)


def continues_open_section(title: str, stack: list[tuple[int, str, bool]]) -> bool:
if _CONTINUED_FROM.match(title):
return bool(stack)
continued = _CONTINUED.match(title)
if not continued:
return False
words = _title_words(continued.group(1))
return any(words == _title_words(open_title) for _, open_title, _ in stack)


def extends_number(number: tuple[int, ...] | None, prefix: tuple[int, ...]) -> bool:
return bool(number) and len(number) > len(prefix) and number[: len(prefix)] == prefix

Expand Down Expand Up @@ -374,6 +385,7 @@ def sectionize(pages: list[tuple[int, str]], level_map: dict[str, int] | None =
max_chapter = 0
capital_chapters = True
last_list_item = (0, 0)
last_step_number = (0, 0)
last_number: tuple[int, ...] = ()
running_headers = running_header_titles(pages)
opened_headers: set[str] = set()
Expand Down Expand Up @@ -422,10 +434,15 @@ def flush():
m = None
elif m and _is_label(title, level_map):
line, m = f"**{title}**", None
step = title if m else line.strip()
if step.isdigit():
last_step_number = (page_no, int(step))
if m:
title = correct_chapter_typo(title, last_number)
if _repeats_open_section(title, stack) or repeats_running_header(
title, stack, recurring_titles
if (
_repeats_open_section(title, stack)
or repeats_running_header(title, stack, recurring_titles)
or continues_open_section(title, stack)
):
if current is not None:
current.page_end = page_no
Expand Down Expand Up @@ -466,7 +483,7 @@ def flush():
and not opens_next_chapter(cnum, last_number, next_dotted, title)
)
or (max_chapter and capital_chapters and not title.isupper())
or (page_no, cnum - 1) == last_list_item
or (page_no, cnum - 1) in (last_list_item, last_step_number)
):
level = 2
demoted = True
Expand Down
10 changes: 10 additions & 0 deletions wikify/tests/test_page_plan.py
Original file line number Diff line number Diff line change
Expand Up @@ -126,6 +126,16 @@ def test_a_short_memo_of_unnumbered_headings_keeps_its_text(self):
for text in ("Memo to all renal staff.", "## Clinic closure", "Call extension 4455."):
self.assertIn(text, pages[0].markdown)

def test_short_unnumbered_sections_each_opening_a_pdf_page_stay_pages(self):
topics = ["Five rights", "IV administration", "Covert medication", "Self-administration"]
sections = [_sec(["Preamble"], 1, markdown="Repeat Prescribing Protocol")]
sections += [
_sec([topic], 2 * n + 1, 2 * n + 2, markdown=_words(20)) for n, topic in enumerate(topics)
]
pages = plan_pages(sections)
self.assertEqual(_titles(pages), topics)
self.assertIn("Repeat Prescribing Protocol", pages[0].markdown)

def test_hierarchy_of_the_remaining_pages_is_unchanged(self):
pages = plan_pages(_outline())
self.assertEqual(pages[-1].hierarchy_path, ["6.1 Protocols", "6.1.2 Lupus nephritis"])
Expand Down
74 changes: 74 additions & 0 deletions wikify/tests/test_sectionize.py
Original file line number Diff line number Diff line change
Expand Up @@ -919,6 +919,80 @@ def test_clean_pages_lays_contents_lines_out_as_the_contents_table_they_continue
)
self.assertEqual(cleaned[3], "")

def _breadcrumb_pages(self):
topics = [
("Administering Medicines", "Five rights"),
("Administering Medicines", "IV administration"),
("Controlled Drugs", "Storage and keys"),
("Controlled Drugs", "Register entries"),
]
pages = []
for index, (chapter, topic) in enumerate(topics):
opening = 2 * index + 1
pages.append(
(
opening,
f"Repeat Prescribing Protocol\n\nPage {opening}\n\n{chapter} > {topic}\n\n"
f"# {topic}\n\nPractice for {topic.lower()}.",
)
)
pages.append((opening + 1, f"# {topic} (continued)\n\nMore on {topic.lower()}."))
return pages

def test_clean_pages_strips_a_page_label_at_the_page_edge(self):
cleaned = dict(clean_pages(self._breadcrumb_pages()))
self.assertNotIn("Page 1", cleaned[1])
self.assertNotIn("Page 5", cleaned[5])

def test_a_breadcrumb_running_header_nests_the_section_under_its_chapter(self):
secs = sectionize(clean_pages(self._breadcrumb_pages()))
paths = {s.title: s.hierarchy_path for s in secs}
self.assertEqual(paths["Five rights"], ["Administering Medicines", "Five rights"])
self.assertEqual(paths["IV administration"], ["Administering Medicines", "IV administration"])
self.assertEqual(paths["Register entries"], ["Controlled Drugs", "Register entries"])
self.assertEqual([s.title for s in secs].count("Administering Medicines"), 1)

def test_a_breadcrumb_needs_a_matching_heading_below_it(self):
pages = [
(1, "Open Settings > Users to add staff.\n\n# Users\n\nbody"),
(2, "Open Settings > Roles to add roles.\n\n# Roles\n\nbody"),
]
cleaned = dict(clean_pages(pages))
self.assertIn("Open Settings > Users to add staff.", cleaned[1])

def test_a_continued_heading_folds_into_the_open_section(self):
secs = sectionize(clean_pages(self._breadcrumb_pages()))
titles = [s.title for s in secs]
self.assertFalse(any("continued" in title for title in titles))
five_rights = next(s for s in secs if s.title == "Five rights")
self.assertIn("More on five rights.", five_rights.markdown)
self.assertEqual((five_rights.page_start, five_rights.page_end), (1, 2))

def test_a_continued_from_previous_page_heading_is_not_a_section(self):
pages = [
(1, "# Abbreviations\n\nIV: intravenous"),
(2, "# Continued from the previous page.\n\nPO: oral"),
]
secs = sectionize(pages)
self.assertEqual([s.title for s in secs], ["Abbreviations"])
self.assertIn("PO: oral", secs[0].markdown)

def test_a_numbered_heading_after_bare_step_numbers_is_a_step_not_a_chapter(self):
method = (
"# The Method\n\nIntro.\n\n### 1\n\n## Build a fresh bench\n\nNever upgrade in place.\n\n"
"### 2\n\n## Restore a backup\n\nAlways restore fresh.\n\n## 3 Run bench migrate\n\n"
"Expect failures.\n\n4\n\n## Read the whole log\n\nA clean exit is not enough."
)
pages = [
(2, method),
(3, "# Preparing the New Bench\n\nSet up the environment.\n\n## The Apps\n\nList the apps."),
]
secs = sectionize(pages)
paths = {s.title: s.hierarchy_path for s in secs}
self.assertEqual(paths["3 Run bench migrate"], ["The Method", "3 Run bench migrate"])
self.assertEqual(paths["Preparing the New Bench"], ["Preparing the New Bench"])
self.assertEqual(paths["The Apps"], ["Preparing the New Bench", "The Apps"])


class TestEmptySectionsAreFlagged(FrappeTestCase):
def test_section_without_markdown_is_chunked_as_title_only(self):
Expand Down
Loading