diff --git a/haystack/components/preprocessors/recursive_splitter.py b/haystack/components/preprocessors/recursive_splitter.py index 0917d7e799..2f2c247960 100644 --- a/haystack/components/preprocessors/recursive_splitter.py +++ b/haystack/components/preprocessors/recursive_splitter.py @@ -313,7 +313,16 @@ def _chunk_text(self, text: str) -> list[str]: if curr_separator == "sentence": # re. ignore: correct SentenceSplitter initialization is checked at the initialization of the component sentence_with_spans = self.nltk_tokenizer.split_sentences(text) # type: ignore - splits = [sentence["sentence"] for sentence in sentence_with_spans] + if self.sentence_splitter_params.get("keep_white_spaces", False): + # Slice the original text between sentence starts instead of using the tokenizer's sentence + # strings: even with keep_white_spaces=True the tokenizer drops whitespace at the end of the text + # (e.g. a trailing "\f" or "\n\n"), which silently removes page breaks and shifts + # page_number and split_idx_start for every following chunk. + starts = [0] + [sentence["start"] for sentence in sentence_with_spans[1:]] + ends = starts[1:] + [len(text)] + splits = [text[start:end] for start, end in zip(starts, ends, strict=True)] + else: + splits = [sentence["sentence"] for sentence in sentence_with_spans] else: # add escape "\" to the separator and wrapped it in a group so that it's included in the splits as well escaped_separator = re.escape(curr_separator) diff --git a/releasenotes/notes/fix-recursive-splitter-sentence-trailing-whitespace-3f9c2a1d7e4b5c60.yaml b/releasenotes/notes/fix-recursive-splitter-sentence-trailing-whitespace-3f9c2a1d7e4b5c60.yaml new file mode 100644 index 0000000000..ccb6365f5f --- /dev/null +++ b/releasenotes/notes/fix-recursive-splitter-sentence-trailing-whitespace-3f9c2a1d7e4b5c60.yaml @@ -0,0 +1,8 @@ +--- +fixes: + - | + ``RecursiveDocumentSplitter`` no longer drops whitespace at the end of a text piece when it splits by + ``"sentence"`` with ``keep_white_spaces=True`` (the default). Before, a trailing page break or newline + (for example ``"...sentence.\f\n\n"``) was removed from the chunk, so the chunks no longer added up to the + original text, ``page_number`` stayed on the first page and ``split_idx_start`` drifted for every + following chunk. diff --git a/test/components/preprocessors/test_recursive_splitter.py b/test/components/preprocessors/test_recursive_splitter.py index 6790082f9e..4b412cb132 100644 --- a/test/components/preprocessors/test_recursive_splitter.py +++ b/test/components/preprocessors/test_recursive_splitter.py @@ -1308,3 +1308,23 @@ def test_fallback_word_unit_no_trailing_whitespace_only_chunk(): for doc in result: assert doc.content is not None assert doc.content.strip() + + +def test_run_sentence_separator_keeps_trailing_page_breaks_and_whitespace() -> None: + splitter = RecursiveDocumentSplitter(split_length=6, split_overlap=0, split_unit="word") + splitter.warm_up() + text = ( + "Intro para one. It has two sentences.\f\n\nSecond page starts. Also two sentences here.\f\n\nThird page. End." + ) + + chunks = splitter.run([Document(content=text)])["documents"] + + assert "".join(chunk.content for chunk in chunks) == text + for chunk in chunks: + start = chunk.meta["split_idx_start"] + assert text[start : start + len(chunk.content)] == chunk.content + assert chunks[1].content == "It has two sentences.\f\n\n" + assert chunks[2].content == "Second page starts. " + assert chunks[2].meta["page_number"] == 2 + assert chunks[4].content == "Third page. End." + assert chunks[4].meta["page_number"] == 3