Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 7 additions & 3 deletions haystack/components/preprocessors/recursive_splitter.py
Original file line number Diff line number Diff line change
Expand Up @@ -464,16 +464,16 @@ def _run_one(self, doc: Document) -> list[Document]:
self._add_overlap_info(current_position, new_doc, new_docs)

# count page breaks in the chunk
current_page += chunk.count("\f")
chunk_page = current_page + chunk.count("\f")

# if there are consecutive page breaks at the end with no more text, adjust the page number
# e.g: "text\f\f\f" -> 3 page breaks, but current_page should be 1
consecutive_page_breaks = len(chunk) - len(chunk.rstrip("\f"))

if consecutive_page_breaks > 0:
new_doc.meta["page_number"] = current_page - consecutive_page_breaks
new_doc.meta["page_number"] = chunk_page - consecutive_page_breaks
else:
new_doc.meta["page_number"] = current_page
new_doc.meta["page_number"] = chunk_page

# keep the new chunk doc and update the current position
new_docs.append(new_doc)
Expand All @@ -485,6 +485,10 @@ def _run_one(self, doc: Document) -> list[Document]:
overlap_char_len = len(overlap_str)
else:
overlap_char_len = 0
# Advance the page counter only over the characters the next chunk does not repeat: the
# overlapping tail reappears at the start of the next chunk, so a page break inside it would
# otherwise be counted once per chunk and shift every following page number upwards.
current_page += chunk.count("\f", 0, len(chunk) - overlap_char_len)
current_position += len(chunk) - overlap_char_len

return new_docs
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
---
fixes:
- |
Fixed ``RecursiveDocumentSplitter`` counting a page break character (``\f``) once per chunk it appears in when
``split_overlap`` is used. A page break inside an overlapping region is repeated in consecutive chunks, so the
``page_number`` metadata of every following chunk was shifted upwards and could exceed the number of pages in the
document. The page counter now advances only over the part of a chunk that the next chunk does not repeat.
30 changes: 30 additions & 0 deletions test/components/preprocessors/test_recursive_splitter.py
Original file line number Diff line number Diff line change
Expand Up @@ -365,6 +365,36 @@ def test_run_split_by_sentence_count_page_breaks_split_unit_char() -> None:
assert chunks_docs[6].meta["split_idx_start"] == text.index(chunks_docs[6].content)


def test_run_count_page_breaks_once_with_overlap_split_unit_char() -> None:
# A page break that falls inside an overlapping region is repeated in consecutive chunks. It must only
# advance the page counter once, otherwise every chunk after it gets a page number that is too high.
splitter = RecursiveDocumentSplitter(split_length=14, split_overlap=5, separators=[" "], split_unit="char")

text = "This is page one.\fThis is page two, it is longer."

documents = splitter.run(documents=[Document(content=text)])
chunks_docs = documents["documents"]
assert len(chunks_docs) == 5

assert chunks_docs[0].content == "This is page "
assert chunks_docs[0].meta["page_number"] == 1

# the page break is part of this chunk and of the overlapping tail repeated in the next chunk
assert chunks_docs[1].content == "page one.\fThis"
assert chunks_docs[1].meta["page_number"] == 2

# the repeated page break was previously counted a second time, pushing these chunks to page 3
# even though the document only has two pages
assert chunks_docs[2].content == "\fThis is page "
assert chunks_docs[2].meta["page_number"] == 2

assert chunks_docs[3].content == "page two, it i"
assert chunks_docs[3].meta["page_number"] == 2

assert chunks_docs[4].content == " it is longer."
assert chunks_docs[4].meta["page_number"] == 2


def test_run_split_document_with_overlap_character_unit():
splitter = RecursiveDocumentSplitter(split_length=20, split_overlap=10, separators=["."], split_unit="char")
text = """A simple sentence1. A bright sentence2. A clever sentence3"""
Expand Down
Loading