From 93263bf2127c31c6b22b3e659672c8aebd4646ec Mon Sep 17 00:00:00 2001 From: LindseyZ1205 Date: Sat, 26 Sep 2026 16:18:16 -0400 Subject: [PATCH 1/5] fix: sort SentenceWindowRetriever context before merging text The context documents were merged into `context_windows` before being sorted by `split_id`. When chunks have no `split_idx_start` (for example chunks from MarkdownHeaderSplitter), `merge_documents_text` concatenates them in the order returned by the Document Store, so the context text could be scrambled while `context_documents` was correctly sorted. Sort first and merge the sorted list, in both run and run_async. Co-Authored-By: Claude Opus 5.5 --- .../components/retrievers/sentence_window_retriever.py | 4 ++-- ...e-window-retriever-context-order-a1267b171dea93bb.yaml | 8 ++++++++ .../retrievers/test_sentence_window_retriever.py | 4 ++++ .../retrievers/test_sentence_window_retriever_async.py | 4 ++++ 4 files changed, 18 insertions(+), 2 deletions(-) create mode 100644 releasenotes/notes/fix-sentence-window-retriever-context-order-a1267b171dea93bb.yaml diff --git a/haystack/components/retrievers/sentence_window_retriever.py b/haystack/components/retrievers/sentence_window_retriever.py index 0eecfb12971..7c491d69c0b 100644 --- a/haystack/components/retrievers/sentence_window_retriever.py +++ b/haystack/components/retrievers/sentence_window_retriever.py @@ -279,8 +279,8 @@ def _retrieve_context_for_document(self, doc: Document, window_size: int) -> tup assert split_id is not None filter_conditions = self._build_filter_conditions(split_id, window_size, source_ids) context_docs = self.document_store.filter_documents(filter_conditions) - context_text = self.merge_documents_text(context_docs) context_docs_sorted = sorted(context_docs, key=lambda doc: doc.meta[self.split_id_meta_field]) + context_text = self.merge_documents_text(context_docs_sorted) return context_text, context_docs_sorted @@ -302,8 +302,8 @@ async def _retrieve_context_for_document_async(self, doc: Document, window_size: filter_conditions = self._build_filter_conditions(split_id, window_size, source_ids) # Ignoring type error because DocumentStore protocol doesn't define filter_documents_async context_docs = await self.document_store.filter_documents_async(filter_conditions) # type: ignore[attr-defined] - context_text = self.merge_documents_text(context_docs) context_docs_sorted = sorted(context_docs, key=lambda doc: doc.meta[self.split_id_meta_field]) + context_text = self.merge_documents_text(context_docs_sorted) return context_text, context_docs_sorted diff --git a/releasenotes/notes/fix-sentence-window-retriever-context-order-a1267b171dea93bb.yaml b/releasenotes/notes/fix-sentence-window-retriever-context-order-a1267b171dea93bb.yaml new file mode 100644 index 00000000000..6d04fbcdd22 --- /dev/null +++ b/releasenotes/notes/fix-sentence-window-retriever-context-order-a1267b171dea93bb.yaml @@ -0,0 +1,8 @@ +--- +fixes: + - | + Fixed ``SentenceWindowRetriever`` returning ``context_windows`` with the chunks in the wrong order when the + documents have no ``split_idx_start`` meta field, for example chunks produced by ``MarkdownHeaderSplitter``. + The chunks were concatenated in the order returned by the Document Store instead of by ``split_id``, so the + context passed to an LLM could be scrambled while ``context_documents`` was correctly sorted. The chunks are now + sorted by ``split_id`` before being merged, in both ``run`` and ``run_async``. diff --git a/test/components/retrievers/test_sentence_window_retriever.py b/test/components/retrievers/test_sentence_window_retriever.py index 7f5f3f0b3d8..d4f82d00f62 100644 --- a/test/components/retrievers/test_sentence_window_retriever.py +++ b/test/components/retrievers/test_sentence_window_retriever.py @@ -283,6 +283,10 @@ def test_run_custom_fields(self, in_memory_doc_store): # run the retriever with a document whose content = "Sentence 4." result = retriever.run(retrieved_documents=[doc for doc in docs if doc.content == "Sentence 4."]) assert len(result["context_documents"]) == 7 + assert [doc.meta["split_id_test"] for doc in result["context_documents"]] == [1, 2, 3, 4, 5, 6, 7] + assert result["context_windows"] == [ + "Sentence 1.Sentence 2.Sentence 3.Sentence 4.Sentence 5.Sentence 6.Sentence 7." + ] def test_run_with_multiple_source_ids(self, in_memory_doc_store): docs = [ diff --git a/test/components/retrievers/test_sentence_window_retriever_async.py b/test/components/retrievers/test_sentence_window_retriever_async.py index de85b08e60c..710bd6ac216 100644 --- a/test/components/retrievers/test_sentence_window_retriever_async.py +++ b/test/components/retrievers/test_sentence_window_retriever_async.py @@ -176,6 +176,10 @@ async def test_run_async_custom_fields(self, in_memory_doc_store): # run the retriever with a document whose content = "Sentence 4." result = await retriever.run_async(retrieved_documents=[doc for doc in docs if doc.content == "Sentence 4."]) assert len(result["context_documents"]) == 7 + assert [doc.meta["split_id_test"] for doc in result["context_documents"]] == [1, 2, 3, 4, 5, 6, 7] + assert result["context_windows"] == [ + "Sentence 1.Sentence 2.Sentence 3.Sentence 4.Sentence 5.Sentence 6.Sentence 7." + ] @pytest.mark.asyncio async def test_run_async_with_multiple_source_ids(self, in_memory_doc_store): From c5e5ef8f2769c9f21e09b79b0e07001a57129e20 Mon Sep 17 00:00:00 2001 From: bogdankostic Date: Fri, 2 Oct 2026 12:45:17 +0200 Subject: [PATCH 2/5] docs: describe actual context_documents order in SentenceWindowRetriever The run/run_async docstrings said context_documents are sorted by split_idx_start. They are sorted by split_id_meta_field within each window, grouped in the order of retrieved_documents, and documents shared by overlapping windows appear once per window. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../retrievers/sentence_window_retriever.py | 16 ++++++++++------ 1 file changed, 10 insertions(+), 6 deletions(-) diff --git a/haystack/components/retrievers/sentence_window_retriever.py b/haystack/components/retrievers/sentence_window_retriever.py index cb00d68327e..b5a89fa3dfc 100644 --- a/haystack/components/retrievers/sentence_window_retriever.py +++ b/haystack/components/retrievers/sentence_window_retriever.py @@ -192,9 +192,11 @@ def run(self, retrieved_documents: list[Document], window_size: int | None = Non A dictionary with the following keys: - `context_windows`: A list of strings, where each string represents the concatenated text from the context window of the corresponding document in `retrieved_documents`. - - `context_documents`: A list `Document` objects, containing the retrieved documents plus the context - document surrounding them. The documents are sorted by the `split_idx_start` - meta field. + - `context_documents`: A list of `Document` objects, containing the retrieved documents plus the + context documents surrounding them, grouped by window in the order of + `retrieved_documents`. Within each window, the documents are sorted by the + meta field set in `split_id_meta_field`. Documents shared by overlapping + windows appear once per window. """ window_size = self.window_size if window_size is None else window_size @@ -223,9 +225,11 @@ async def run_async(self, retrieved_documents: list[Document], window_size: int A dictionary with the following keys: - `context_windows`: A list of strings, where each string represents the concatenated text from the context window of the corresponding document in `retrieved_documents`. - - `context_documents`: A list `Document` objects, containing the retrieved documents plus the context - document surrounding them. The documents are sorted by the `split_idx_start` - meta field. + - `context_documents`: A list of `Document` objects, containing the retrieved documents plus the + context documents surrounding them, grouped by window in the order of + `retrieved_documents`. Within each window, the documents are sorted by the + meta field set in `split_id_meta_field`. Documents shared by overlapping + windows appear once per window. """ window_size = self.window_size if window_size is None else window_size From a554424db6faaea96c396e5ef6536ab250be3cde Mon Sep 17 00:00:00 2001 From: bogdankostic Date: Fri, 2 Oct 2026 12:53:22 +0200 Subject: [PATCH 3/5] docs: correct SentenceWindowRetriever output order on docs site The Output variables row said context_documents are ordered by split_idx_start. They are grouped by retrieved document and ordered by split_id within each window. Updated docs/ and the current stable version-3.3 page. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../pipeline-components/retrievers/sentencewindowretriever.mdx | 2 +- .../pipeline-components/retrievers/sentencewindowretriever.mdx | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/docs-website/docs/pipeline-components/retrievers/sentencewindowretriever.mdx b/docs-website/docs/pipeline-components/retrievers/sentencewindowretriever.mdx index e9c5de9fc97..d0d5ccbe8b8 100644 --- a/docs-website/docs/pipeline-components/retrievers/sentencewindowretriever.mdx +++ b/docs-website/docs/pipeline-components/retrievers/sentencewindowretriever.mdx @@ -16,7 +16,7 @@ Use this component to retrieve neighboring sentences around relevant sentences t | **Most common position in a pipeline** | Used after the main Retriever component, like the `InMemoryEmbeddingRetriever` or any other Retriever. | | **Mandatory init variables** | `document_store`: An instance of a Document Store | | **Mandatory run variables** | `retrieved_documents`: A list of already retrieved documents for which you want to get a context window | -| **Output variables** | `context_windows`: A list of strings

`context_documents`: A list of documents ordered by `split_idx_start` | +| **Output variables** | `context_windows`: A list of strings, one per retrieved document

`context_documents`: A list of documents, grouped by retrieved document and ordered by `split_id` within each window | | **API reference** | [Retrievers](/reference/retrievers-api) | | **GitHub link** | https://github.com/deepset-ai/haystack/blob/main/haystack/components/retrievers/sentence_window_retriever.py | | **Package name** | `haystack-ai` | diff --git a/docs-website/versioned_docs/version-3.3/pipeline-components/retrievers/sentencewindowretriever.mdx b/docs-website/versioned_docs/version-3.3/pipeline-components/retrievers/sentencewindowretriever.mdx index e9c5de9fc97..d0d5ccbe8b8 100644 --- a/docs-website/versioned_docs/version-3.3/pipeline-components/retrievers/sentencewindowretriever.mdx +++ b/docs-website/versioned_docs/version-3.3/pipeline-components/retrievers/sentencewindowretriever.mdx @@ -16,7 +16,7 @@ Use this component to retrieve neighboring sentences around relevant sentences t | **Most common position in a pipeline** | Used after the main Retriever component, like the `InMemoryEmbeddingRetriever` or any other Retriever. | | **Mandatory init variables** | `document_store`: An instance of a Document Store | | **Mandatory run variables** | `retrieved_documents`: A list of already retrieved documents for which you want to get a context window | -| **Output variables** | `context_windows`: A list of strings

`context_documents`: A list of documents ordered by `split_idx_start` | +| **Output variables** | `context_windows`: A list of strings, one per retrieved document

`context_documents`: A list of documents, grouped by retrieved document and ordered by `split_id` within each window | | **API reference** | [Retrievers](/reference/retrievers-api) | | **GitHub link** | https://github.com/deepset-ai/haystack/blob/main/haystack/components/retrievers/sentence_window_retriever.py | | **Package name** | `haystack-ai` | From 4f55e4dc080cef7facdf77e11178655026935adc Mon Sep 17 00:00:00 2001 From: bogdankostic Date: Fri, 2 Oct 2026 13:03:54 +0200 Subject: [PATCH 4/5] docs: expand SentenceWindowRetriever overview for next version Document the metadata the component relies on (source_id, split_id, optional split_idx_start), which splitters produce it, what happens when it is missing, and what the two outputs contain. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../retrievers/sentencewindowretriever.mdx | 19 +++++++++++++++---- 1 file changed, 15 insertions(+), 4 deletions(-) diff --git a/docs-website/docs/pipeline-components/retrievers/sentencewindowretriever.mdx b/docs-website/docs/pipeline-components/retrievers/sentencewindowretriever.mdx index d0d5ccbe8b8..cbce80d54b8 100644 --- a/docs-website/docs/pipeline-components/retrievers/sentencewindowretriever.mdx +++ b/docs-website/docs/pipeline-components/retrievers/sentencewindowretriever.mdx @@ -25,13 +25,24 @@ Use this component to retrieve neighboring sentences around relevant sentences t ## Overview -The "sentence window" is a retrieval technique that allows for the retrieval of the context around relevant sentences. +Small chunks work well for retrieval because they match a query precisely, but they're often too short for an LLM to answer from. `SentenceWindowRetriever` solves this by fetching the chunks around each retrieved document from the same source document, so the LLM gets the surrounding context. Despite its name, it works with chunks of any size, not only sentences. -During indexing, documents are broken into smaller chunks or sentences and indexed. During retrieval, the sentences most relevant to a given query, based on a certain similarity metric, are retrieved. +Use it after another Retriever, such as `InMemoryEmbeddingRetriever` or `InMemoryBM25Retriever`. For each retrieved document, it fetches up to `window_size` chunks before and after it (3 by default). You can override `window_size` for a single call by passing it to `run()`. -Once we have the relevant sentences, we can retrieve neighboring sentences to provide full context. The number of neighboring sentences to retrieve is defined by a fixed number of sentences before and after the relevant sentence. +### Required metadata -This component is meant to be used with other Retrievers, such as the `InMemoryEmbeddingRetriever`. These Retrievers find relevant sentences by comparing a query against indexed sentences using a similarity metric. Then, the `SentenceWindowRetriever` component retrieves neighboring sentences around the relevant ones by leveraging metadata stored in the `Document` object. +The component uses these `meta` fields to find neighboring chunks: + +- `source_id`: identifies the original document a chunk comes from. To match on several fields, pass a list to `source_id_meta_field`. +- `split_id`: the chunk's position within its source document. To use a different field, set `split_id_meta_field`. +- `split_idx_start` (optional): the chunk's start position in the source text. If every chunk in a window has it, text that overlaps between chunks is removed when they're merged. Otherwise, the chunks are joined in `split_id` order without removing any overlap. + +[`DocumentSplitter`](../preprocessors/documentsplitter.mdx) and [`RecursiveDocumentSplitter`](../preprocessors/recursivesplitter.mdx) add all three fields. If a retrieved document is missing `source_id` or `split_id`, the component raises a `ValueError`. To pass such documents through unchanged instead, set `raise_on_missing_meta_fields=False`. + +### Outputs + +- `context_windows`: one string per retrieved document, in the same order, containing the merged text of its window. +- `context_documents`: the documents in each window, grouped by retrieved document and ordered by `split_id` within each window. If two retrieved documents are close together, their windows overlap and the shared chunks appear in both. ## Usage From 6037a8e92fdc39b737a84125b29ef606dd3f4f35 Mon Sep 17 00:00:00 2001 From: bogdankostic Date: Fri, 2 Oct 2026 13:09:14 +0200 Subject: [PATCH 5/5] docs: restore sentence-window technique description in overview Bring back the technique name and the indexing/retrieval workflow that the previous overview rewrite dropped. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../retrievers/sentencewindowretriever.mdx | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs-website/docs/pipeline-components/retrievers/sentencewindowretriever.mdx b/docs-website/docs/pipeline-components/retrievers/sentencewindowretriever.mdx index cbce80d54b8..e961cf24ca2 100644 --- a/docs-website/docs/pipeline-components/retrievers/sentencewindowretriever.mdx +++ b/docs-website/docs/pipeline-components/retrievers/sentencewindowretriever.mdx @@ -25,9 +25,9 @@ Use this component to retrieve neighboring sentences around relevant sentences t ## Overview -Small chunks work well for retrieval because they match a query precisely, but they're often too short for an LLM to answer from. `SentenceWindowRetriever` solves this by fetching the chunks around each retrieved document from the same source document, so the LLM gets the surrounding context. Despite its name, it works with chunks of any size, not only sentences. +Sentence-window retrieval is a technique for retrieving the context around relevant text. During indexing, documents are split into small chunks, such as sentences, and written to a Document Store. During retrieval, a Retriever such as `InMemoryEmbeddingRetriever` or `InMemoryBM25Retriever` finds the chunks most relevant to the query. `SentenceWindowRetriever` then fetches up to `window_size` chunks before and after each of them (3 by default) from the same source document. -Use it after another Retriever, such as `InMemoryEmbeddingRetriever` or `InMemoryBM25Retriever`. For each retrieved document, it fetches up to `window_size` chunks before and after it (3 by default). You can override `window_size` for a single call by passing it to `run()`. +This combines the strengths of both chunk sizes: small chunks match a query precisely, and the surrounding window gives the LLM enough context to answer. Despite its name, the component works with chunks of any size, not only sentences. You can override `window_size` for a single call by passing it to `run()`. ### Required metadata