Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 3 additions & 2 deletions haystack/components/retrievers/sentence_window_retriever.py
Original file line number Diff line number Diff line change
Expand Up @@ -322,8 +322,9 @@ def _assemble_context(
for fetched in fetched_documents
if self._is_in_window(fetched, source_ids, split_id - window_size, split_id + window_size)
]
context_text.append(self.merge_documents_text(context_docs))
context_documents.extend(sorted(context_docs, key=lambda d: d.meta[self.split_id_meta_field]))
context_docs_sorted = sorted(context_docs, key=lambda d: d.meta[self.split_id_meta_field])
context_text.append(self.merge_documents_text(context_docs_sorted))
context_documents.extend(context_docs_sorted)

return {"context_windows": context_text, "context_documents": context_documents}

Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
---
fixes:
- |
Fixed ``SentenceWindowRetriever`` returning ``context_windows`` with the chunks in the wrong order when the
documents have no ``split_idx_start`` meta field, for example chunks produced by ``MarkdownHeaderSplitter``.
The chunks were concatenated in the order returned by the Document Store instead of by ``split_id``, so the
context passed to an LLM could be scrambled while ``context_documents`` was correctly sorted. The chunks are now
sorted by ``split_id`` before being merged, in both ``run`` and ``run_async``.
Original file line number Diff line number Diff line change
Expand Up @@ -284,6 +284,10 @@ def test_run_custom_fields(self, in_memory_doc_store):
# run the retriever with a document whose content = "Sentence 4."
result = retriever.run(retrieved_documents=[doc for doc in docs if doc.content == "Sentence 4."])
assert len(result["context_documents"]) == 7
assert [doc.meta["split_id_test"] for doc in result["context_documents"]] == [1, 2, 3, 4, 5, 6, 7]
assert result["context_windows"] == [
"Sentence 1.Sentence 2.Sentence 3.Sentence 4.Sentence 5.Sentence 6.Sentence 7."
]

def test_run_with_multiple_source_ids(self, in_memory_doc_store):
docs = [
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -176,6 +176,10 @@ async def test_run_async_custom_fields(self, in_memory_doc_store):
# run the retriever with a document whose content = "Sentence 4."
result = await retriever.run_async(retrieved_documents=[doc for doc in docs if doc.content == "Sentence 4."])
assert len(result["context_documents"]) == 7
assert [doc.meta["split_id_test"] for doc in result["context_documents"]] == [1, 2, 3, 4, 5, 6, 7]
assert result["context_windows"] == [
"Sentence 1.Sentence 2.Sentence 3.Sentence 4.Sentence 5.Sentence 6.Sentence 7."
]

@pytest.mark.asyncio
async def test_run_async_with_multiple_source_ids(self, in_memory_doc_store):
Expand Down