Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 10 additions & 1 deletion haystack/components/preprocessors/recursive_splitter.py
Original file line number Diff line number Diff line change
Expand Up @@ -313,7 +313,16 @@ def _chunk_text(self, text: str) -> list[str]:
if curr_separator == "sentence":
# re. ignore: correct SentenceSplitter initialization is checked at the initialization of the component
sentence_with_spans = self.nltk_tokenizer.split_sentences(text) # type: ignore
splits = [sentence["sentence"] for sentence in sentence_with_spans]
if self.sentence_splitter_params.get("keep_white_spaces", False):
# Slice the original text between sentence starts instead of using the tokenizer's sentence
# strings: even with keep_white_spaces=True the tokenizer drops whitespace at the end of the text
# (e.g. a trailing "\f" or "\n\n"), which silently removes page breaks and shifts
# page_number and split_idx_start for every following chunk.
starts = [0] + [sentence["start"] for sentence in sentence_with_spans[1:]]
ends = starts[1:] + [len(text)]
splits = [text[start:end] for start, end in zip(starts, ends, strict=True)]
else:
splits = [sentence["sentence"] for sentence in sentence_with_spans]
else:
# add escape "\" to the separator and wrapped it in a group so that it's included in the splits as well
escaped_separator = re.escape(curr_separator)
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
---
fixes:
- |
``RecursiveDocumentSplitter`` no longer drops whitespace at the end of a text piece when it splits by
``"sentence"`` with ``keep_white_spaces=True`` (the default). Before, a trailing page break or newline
(for example ``"...sentence.\f\n\n"``) was removed from the chunk, so the chunks no longer added up to the
original text, ``page_number`` stayed on the first page and ``split_idx_start`` drifted for every
following chunk.
20 changes: 20 additions & 0 deletions test/components/preprocessors/test_recursive_splitter.py
Original file line number Diff line number Diff line change
Expand Up @@ -1308,3 +1308,23 @@ def test_fallback_word_unit_no_trailing_whitespace_only_chunk():
for doc in result:
assert doc.content is not None
assert doc.content.strip()


def test_run_sentence_separator_keeps_trailing_page_breaks_and_whitespace() -> None:
splitter = RecursiveDocumentSplitter(split_length=6, split_overlap=0, split_unit="word")
splitter.warm_up()
text = (
"Intro para one. It has two sentences.\f\n\nSecond page starts. Also two sentences here.\f\n\nThird page. End."
)

chunks = splitter.run([Document(content=text)])["documents"]

assert "".join(chunk.content for chunk in chunks) == text
for chunk in chunks:
start = chunk.meta["split_idx_start"]
assert text[start : start + len(chunk.content)] == chunk.content
assert chunks[1].content == "It has two sentences.\f\n\n"
assert chunks[2].content == "Second page starts. "
assert chunks[2].meta["page_number"] == 2
assert chunks[4].content == "Third page. End."
assert chunks[4].meta["page_number"] == 3