From 2090b05f1027f23ce828fa902597921465949827 Mon Sep 17 00:00:00 2001 From: Tyagiquamar Date: Sun, 27 Sep 2026 15:20:48 +0530 Subject: [PATCH] fix(CSVDocumentSplitter): use positional column index instead of column label read_csv_kwargs allows overriding how pandas reads the CSV, e.g. setting header=0 so the first row becomes the column labels. The splitter then crashed at int(split_df.columns[0]) because the label is a string, and sorted sub-tables by label instead of position. Resolve the first column of each sub-table to its positional index in the original DataFrame for both the sort and the col_idx_start metadata. Fixes #12965 Signed-off-by: Tyagiquamar --- .../preprocessors/csv_document_splitter.py | 8 +++++--- .../preprocessors/test_csv_document_splitter.py | 16 ++++++++++++++++ 2 files changed, 21 insertions(+), 3 deletions(-) diff --git a/haystack/components/preprocessors/csv_document_splitter.py b/haystack/components/preprocessors/csv_document_splitter.py index ea1dfa78c51..10d279da9b1 100644 --- a/haystack/components/preprocessors/csv_document_splitter.py +++ b/haystack/components/preprocessors/csv_document_splitter.py @@ -145,8 +145,10 @@ def run(self, documents: list[Document]) -> dict[str, list[Document]]: ) continue - # Sort split_dfs first by row index, then by column index - split_dfs.sort(key=lambda dataframe: (dataframe.index[0], dataframe.columns[0])) + # Sort split_dfs first by row index, then by column index. Column + # labels can be non-integer strings when read_csv_kwargs sets + # header, so sort by positional index instead of the label. + split_dfs.sort(key=lambda dataframe: (dataframe.index[0], df.columns.get_loc(dataframe.columns[0]))) for split_id, split_df in enumerate(split_dfs): split_documents.append( @@ -156,7 +158,7 @@ def run(self, documents: list[Document]) -> dict[str, list[Document]]: **deepcopy(document.meta), "source_id": document.id, "row_idx_start": int(split_df.index[0]), - "col_idx_start": int(split_df.columns[0]), + "col_idx_start": int(df.columns.get_loc(split_df.columns[0])), "split_id": split_id, }, ) diff --git a/test/components/preprocessors/test_csv_document_splitter.py b/test/components/preprocessors/test_csv_document_splitter.py index e56c4c6ca64..3d0b3d37b90 100644 --- a/test/components/preprocessors/test_csv_document_splitter.py +++ b/test/components/preprocessors/test_csv_document_splitter.py @@ -398,3 +398,19 @@ def test_split_by_row_with_empty_rows(self, caplog: LogCaptureFixture) -> None: def test_incorrect_split_mode(self) -> None: with pytest.raises(ValueError, match="not recognized"): CSVDocumentSplitter(split_mode="incorrect_mode") # type: ignore[arg-type] + + def test_read_csv_kwargs_header_zero(self) -> None: + splitter = CSVDocumentSplitter( + row_split_threshold=1, + column_split_threshold=None, + read_csv_kwargs={"header": 0}, + ) + doc = Document(content="name,score\nAda,9\n\nBob,8") + result = splitter.run([doc])["documents"] + assert len(result) == 2 + for split_doc in result: + assert split_doc.meta["col_idx_start"] == 0 + assert result[0].meta["row_idx_start"] == 0 + assert result[1].meta["row_idx_start"] == 2 + assert "Ada,9" in result[0].content + assert "Bob,8" in result[1].content