From 58d3455a2ef08fe32f543f0079199508fb5bb690 Mon Sep 17 00:00:00 2001 From: sclfcz Date: Fri, 25 Sep 2026 23:39:54 +0800 Subject: [PATCH 1/6] fix(MetaFieldRanker): apply missing_meta when no document has the field When every document lacks the ranking field, run() returned the input documents unchanged, bypassing the configured missing_meta policy: a batch with one rated document dropped the unrated ones, while an entirely unrated batch passed through intact even with missing_meta="drop". Apply the policy in that branch as well (drop -> empty list, top/bottom keep the documents, as before) and keep the warning that explains the situation. --- haystack/components/rankers/meta_field.py | 4 +++- test/components/rankers/test_metafield.py | 19 +++++++++++++++++++ 2 files changed, 22 insertions(+), 1 deletion(-) diff --git a/haystack/components/rankers/meta_field.py b/haystack/components/rankers/meta_field.py index b5ce892394..cde5d5efba 100644 --- a/haystack/components/rankers/meta_field.py +++ b/haystack/components/rankers/meta_field.py @@ -264,10 +264,12 @@ def run( "The parameter is currently set to '{meta_field}', but none of the provided " "Documents with IDs {document_ids} have this meta key.\n" "Set to the name of a field that is present within the provided Documents.\n" - "Returning the of the original Documents since there are no values to rank.", + "Applying the configured policy instead of ranking.", meta_field=self.meta_field, document_ids=",".join([doc.id for doc in deduplicated_documents]), ) + if self.missing_meta == "drop": + return {"documents": []} return {"documents": deduplicated_documents[:top_k]} if len(docs_missing_meta_field) > 0: diff --git a/test/components/rankers/test_metafield.py b/test/components/rankers/test_metafield.py index a998ab4cb5..e65dffb700 100644 --- a/test/components/rankers/test_metafield.py +++ b/test/components/rankers/test_metafield.py @@ -338,3 +338,22 @@ def test_none_meta_value_is_handled_as_missing_meta(self, missing_meta, expected ] output = ranker.run(documents=docs_before) assert [doc.id for doc in output["documents"]] == expected_ids + + +def test_missing_meta_drop_when_no_document_has_the_field(): + """The missing_meta policy applies even when no document has the field. + + The ranker returned every input document in that case, so + missing_meta="drop" silently stopped filtering: a batch with no rated + document kept the unrated ones. + """ + docs = [Document(id="a", content="a"), Document(id="b", content="b")] + + assert ( + MetaFieldRanker(meta_field="rating", missing_meta="drop").run(docs)["documents"] + == [] + ) + + for policy in ("bottom", "top"): + ranker = MetaFieldRanker(meta_field="rating", missing_meta=policy) + assert [d.id for d in ranker.run(docs)["documents"]] == ["a", "b"] From 1d9ad449ba1e1aeeb70ecf85a70143f3edf9fde7 Mon Sep 17 00:00:00 2001 From: sclfcz Date: Sat, 26 Sep 2026 09:51:26 +0800 Subject: [PATCH 2/6] fix(CSVDocumentSplitter): report column positions when read_csv_kwargs sets a header The splitter overrides header to None so columns are integer positions and col_idx_start is read straight from columns[0]. A caller passing read_csv_kwargs={"header": 0} (or "infer") gets the first row as string labels, and int(columns[0]) then raised ValueError for every sub-table, so the document could not be split at all. Map the labels back to their position in the original frame. --- .../preprocessors/csv_document_splitter.py | 8 +++++- ...est_csv_document_splitter_header_kwargs.py | 27 +++++++++++++++++++ 2 files changed, 34 insertions(+), 1 deletion(-) create mode 100644 test/components/preprocessors/test_csv_document_splitter_header_kwargs.py diff --git a/haystack/components/preprocessors/csv_document_splitter.py b/haystack/components/preprocessors/csv_document_splitter.py index ea1dfa78c5..7d44fa4dee 100644 --- a/haystack/components/preprocessors/csv_document_splitter.py +++ b/haystack/components/preprocessors/csv_document_splitter.py @@ -148,6 +148,12 @@ def run(self, documents: list[Document]) -> dict[str, list[Document]]: # Sort split_dfs first by row index, then by column index split_dfs.sort(key=lambda dataframe: (dataframe.index[0], dataframe.columns[0])) + # Columns are only positional when ``header=None``. A caller passing + # ``read_csv_kwargs={"header": 0}`` gets the first row as string labels, + # and ``int(label)`` then failed for every sub-table. Map labels back to + # their position in the original frame instead. + column_positions = {label: position for position, label in enumerate(df.columns)} + for split_id, split_df in enumerate(split_dfs): split_documents.append( Document( @@ -156,7 +162,7 @@ def run(self, documents: list[Document]) -> dict[str, list[Document]]: **deepcopy(document.meta), "source_id": document.id, "row_idx_start": int(split_df.index[0]), - "col_idx_start": int(split_df.columns[0]), + "col_idx_start": column_positions[split_df.columns[0]], "split_id": split_id, }, ) diff --git a/test/components/preprocessors/test_csv_document_splitter_header_kwargs.py b/test/components/preprocessors/test_csv_document_splitter_header_kwargs.py new file mode 100644 index 0000000000..26cdadee6f --- /dev/null +++ b/test/components/preprocessors/test_csv_document_splitter_header_kwargs.py @@ -0,0 +1,27 @@ +"""read_csv_kwargs that sets a header must not break CSV splitting.""" + +from haystack import Document +from haystack.components.preprocessors.csv_document_splitter import CSVDocumentSplitter + +CSV = "name,score\nAda,9\n\nBob,8" + + +def test_default_header_none_keeps_positional_columns(): + splitter = CSVDocumentSplitter(row_split_threshold=1, column_split_threshold=None) + result = splitter.run([Document(content=CSV)]) + assert [ + (d.meta["row_idx_start"], d.meta["col_idx_start"]) for d in result["documents"] + ] == [(0, 0), (3, 0)] + + +def test_read_csv_kwargs_header_reports_column_positions(): + """With a caller-supplied header the columns are labels, not positions.""" + splitter = CSVDocumentSplitter( + row_split_threshold=1, + column_split_threshold=None, + read_csv_kwargs={"header": 0}, + ) + result = splitter.run([Document(content=CSV)]) + assert [ + (d.meta["row_idx_start"], d.meta["col_idx_start"]) for d in result["documents"] + ] == [(0, 0), (2, 0)] From 3476437f845107cadef73c8b7f95f0a0d78b028c Mon Sep 17 00:00:00 2001 From: sclfcz Date: Sat, 26 Sep 2026 23:13:06 +0800 Subject: [PATCH 3/6] fix(CSVDocumentSplitter): order sub-tables by column position, not by header label The same label-vs-position mixup: the sort key used columns[0], which is an integer position only while header=None. With a caller-supplied header the sub-tables were ordered alphabetically, so a header like "z,_,a" returned the column at position 2 before the one at position 0 and swapped their split_ids. Sort on the mapped position instead; the default path is unchanged. --- .../preprocessors/csv_document_splitter.py | 16 +++++++++----- ...est_csv_document_splitter_header_kwargs.py | 21 ++++++++++++++++--- 2 files changed, 29 insertions(+), 8 deletions(-) diff --git a/haystack/components/preprocessors/csv_document_splitter.py b/haystack/components/preprocessors/csv_document_splitter.py index 7d44fa4dee..4f9f8ffb3d 100644 --- a/haystack/components/preprocessors/csv_document_splitter.py +++ b/haystack/components/preprocessors/csv_document_splitter.py @@ -145,15 +145,21 @@ def run(self, documents: list[Document]) -> dict[str, list[Document]]: ) continue - # Sort split_dfs first by row index, then by column index - split_dfs.sort(key=lambda dataframe: (dataframe.index[0], dataframe.columns[0])) - # Columns are only positional when ``header=None``. A caller passing # ``read_csv_kwargs={"header": 0}`` gets the first row as string labels, - # and ``int(label)`` then failed for every sub-table. Map labels back to - # their position in the original frame instead. + # so ``int(label)`` failed for every sub-table and sorting on the label + # ordered the sub-tables alphabetically. Map labels back to their + # position in the original frame instead. column_positions = {label: position for position, label in enumerate(df.columns)} + # Sort split_dfs first by row index, then by column position + split_dfs.sort( + key=lambda dataframe: ( + dataframe.index[0], + column_positions[dataframe.columns[0]], + ) + ) + for split_id, split_df in enumerate(split_dfs): split_documents.append( Document( diff --git a/test/components/preprocessors/test_csv_document_splitter_header_kwargs.py b/test/components/preprocessors/test_csv_document_splitter_header_kwargs.py index 26cdadee6f..cc72dc5887 100644 --- a/test/components/preprocessors/test_csv_document_splitter_header_kwargs.py +++ b/test/components/preprocessors/test_csv_document_splitter_header_kwargs.py @@ -3,12 +3,13 @@ from haystack import Document from haystack.components.preprocessors.csv_document_splitter import CSVDocumentSplitter -CSV = "name,score\nAda,9\n\nBob,8" +ROWS = "name,score\nAda,9\n\nBob,8" +COLUMNS = "z,_,a\n1,,2\n3,,4\n" def test_default_header_none_keeps_positional_columns(): splitter = CSVDocumentSplitter(row_split_threshold=1, column_split_threshold=None) - result = splitter.run([Document(content=CSV)]) + result = splitter.run([Document(content=ROWS)]) assert [ (d.meta["row_idx_start"], d.meta["col_idx_start"]) for d in result["documents"] ] == [(0, 0), (3, 0)] @@ -21,7 +22,21 @@ def test_read_csv_kwargs_header_reports_column_positions(): column_split_threshold=None, read_csv_kwargs={"header": 0}, ) - result = splitter.run([Document(content=CSV)]) + result = splitter.run([Document(content=ROWS)]) assert [ (d.meta["row_idx_start"], d.meta["col_idx_start"]) for d in result["documents"] ] == [(0, 0), (2, 0)] + + +def test_header_labels_do_not_reorder_sub_tables(): + """Sub-tables keep the original column order, not the alphabetical one.""" + splitter = CSVDocumentSplitter( + row_split_threshold=None, + column_split_threshold=1, + read_csv_kwargs={"header": 0}, + ) + result = splitter.run([Document(content=COLUMNS)]) + assert [ + (d.meta["col_idx_start"], d.meta["split_id"]) for d in result["documents"] + ] == [(0, 0), (2, 1)] + assert [d.content.strip().replace("\n", "/") for d in result["documents"]] == ["1/3", "2/4"] From a58312970c8abd9b5959fd4b850b80e03302230d Mon Sep 17 00:00:00 2001 From: sclfcz Date: Tue, 29 Sep 2026 00:22:51 +0800 Subject: [PATCH 4/6] chore: keep this branch to the CSV splitter fix meta_field.py and its test already landed through the merged MetaFieldRanker PR, so take main's version and leave only the CSV splitter change here. --- haystack/components/rankers/meta_field.py | 3 +-- test/components/rankers/test_metafield.py | 32 +++++++++++------------ 2 files changed, 17 insertions(+), 18 deletions(-) diff --git a/haystack/components/rankers/meta_field.py b/haystack/components/rankers/meta_field.py index cde5d5efba..c3b3475560 100644 --- a/haystack/components/rankers/meta_field.py +++ b/haystack/components/rankers/meta_field.py @@ -258,7 +258,6 @@ def run( docs_with_meta_field = [doc for doc in deduplicated_documents if doc.meta.get(self.meta_field) is not None] docs_missing_meta_field = [doc for doc in deduplicated_documents if doc.meta.get(self.meta_field) is None] - # If all docs are missing self.meta_field return original documents if len(docs_with_meta_field) == 0: logger.warning( "The parameter is currently set to '{meta_field}', but none of the provided " @@ -268,7 +267,7 @@ def run( meta_field=self.meta_field, document_ids=",".join([doc.id for doc in deduplicated_documents]), ) - if self.missing_meta == "drop": + if missing_meta == "drop": return {"documents": []} return {"documents": deduplicated_documents[:top_k]} diff --git a/test/components/rankers/test_metafield.py b/test/components/rankers/test_metafield.py index e65dffb700..6e25f871c8 100644 --- a/test/components/rankers/test_metafield.py +++ b/test/components/rankers/test_metafield.py @@ -339,21 +339,21 @@ def test_none_meta_value_is_handled_as_missing_meta(self, missing_meta, expected output = ranker.run(documents=docs_before) assert [doc.id for doc in output["documents"]] == expected_ids - -def test_missing_meta_drop_when_no_document_has_the_field(): - """The missing_meta policy applies even when no document has the field. - - The ranker returned every input document in that case, so - missing_meta="drop" silently stopped filtering: a batch with no rated - document kept the unrated ones. - """ - docs = [Document(id="a", content="a"), Document(id="b", content="b")] - - assert ( - MetaFieldRanker(meta_field="rating", missing_meta="drop").run(docs)["documents"] - == [] + @pytest.mark.parametrize( + ("meta", "init_missing_meta", "run_missing_meta", "expected_ids"), + [ + ({}, "drop", None, []), + ({}, "top", None, ["a", "b"]), + ({}, "bottom", None, ["a", "b"]), + ({}, "bottom", "drop", []), + ({}, "drop", "top", ["a", "b"]), + ({"rating": None}, "drop", None, []), + ], ) + def test_missing_meta_when_all_values_are_missing(self, meta, init_missing_meta, run_missing_meta, expected_ids): + ranker = MetaFieldRanker(meta_field="rating", missing_meta=init_missing_meta) + docs = [Document(id="a", content="a", meta=meta), Document(id="b", content="b", meta=meta)] + + output = ranker.run(documents=docs, missing_meta=run_missing_meta) - for policy in ("bottom", "top"): - ranker = MetaFieldRanker(meta_field="rating", missing_meta=policy) - assert [d.id for d in ranker.run(docs)["documents"]] == ["a", "b"] + assert [doc.id for doc in output["documents"]] == expected_ids From cc40db27476a4356a191e832cc5b11d8e2c8ee39 Mon Sep 17 00:00:00 2001 From: sclfcz Date: Tue, 29 Sep 2026 08:51:31 +0800 Subject: [PATCH 5/6] style+notes: format the CSV splitter change and add the release note CI runs ruff format (2 files needed it) and reno (one note per PR); both are satisfied now. --- .../preprocessors/csv_document_splitter.py | 7 +----- ...litter-header-kwargs-fc8ce7f31997395f.yaml | 9 +++++++ ...est_csv_document_splitter_header_kwargs.py | 24 ++++--------------- 3 files changed, 15 insertions(+), 25 deletions(-) create mode 100644 releasenotes/notes/csv-splitter-header-kwargs-fc8ce7f31997395f.yaml diff --git a/haystack/components/preprocessors/csv_document_splitter.py b/haystack/components/preprocessors/csv_document_splitter.py index 4f9f8ffb3d..fd7795898c 100644 --- a/haystack/components/preprocessors/csv_document_splitter.py +++ b/haystack/components/preprocessors/csv_document_splitter.py @@ -153,12 +153,7 @@ def run(self, documents: list[Document]) -> dict[str, list[Document]]: column_positions = {label: position for position, label in enumerate(df.columns)} # Sort split_dfs first by row index, then by column position - split_dfs.sort( - key=lambda dataframe: ( - dataframe.index[0], - column_positions[dataframe.columns[0]], - ) - ) + split_dfs.sort(key=lambda dataframe: (dataframe.index[0], column_positions[dataframe.columns[0]])) for split_id, split_df in enumerate(split_dfs): split_documents.append( diff --git a/releasenotes/notes/csv-splitter-header-kwargs-fc8ce7f31997395f.yaml b/releasenotes/notes/csv-splitter-header-kwargs-fc8ce7f31997395f.yaml new file mode 100644 index 0000000000..5045e559c4 --- /dev/null +++ b/releasenotes/notes/csv-splitter-header-kwargs-fc8ce7f31997395f.yaml @@ -0,0 +1,9 @@ +--- +fixes: + - | + ``CSVDocumentSplitter`` now reports the column position of each sub-table and orders + sub-tables by column position instead of by column label. Passing + ``read_csv_kwargs={"header": 0}`` used to raise ``ValueError: invalid literal for int() + with base 10`` for every sub-table, because ``header=None`` was the only case in which + the labels were positional; with a caller-supplied header the labels are strings, and + sorting on them reordered the sub-tables alphabetically. diff --git a/test/components/preprocessors/test_csv_document_splitter_header_kwargs.py b/test/components/preprocessors/test_csv_document_splitter_header_kwargs.py index cc72dc5887..0e27ff35d1 100644 --- a/test/components/preprocessors/test_csv_document_splitter_header_kwargs.py +++ b/test/components/preprocessors/test_csv_document_splitter_header_kwargs.py @@ -10,33 +10,19 @@ def test_default_header_none_keeps_positional_columns(): splitter = CSVDocumentSplitter(row_split_threshold=1, column_split_threshold=None) result = splitter.run([Document(content=ROWS)]) - assert [ - (d.meta["row_idx_start"], d.meta["col_idx_start"]) for d in result["documents"] - ] == [(0, 0), (3, 0)] + assert [(d.meta["row_idx_start"], d.meta["col_idx_start"]) for d in result["documents"]] == [(0, 0), (3, 0)] def test_read_csv_kwargs_header_reports_column_positions(): """With a caller-supplied header the columns are labels, not positions.""" - splitter = CSVDocumentSplitter( - row_split_threshold=1, - column_split_threshold=None, - read_csv_kwargs={"header": 0}, - ) + splitter = CSVDocumentSplitter(row_split_threshold=1, column_split_threshold=None, read_csv_kwargs={"header": 0}) result = splitter.run([Document(content=ROWS)]) - assert [ - (d.meta["row_idx_start"], d.meta["col_idx_start"]) for d in result["documents"] - ] == [(0, 0), (2, 0)] + assert [(d.meta["row_idx_start"], d.meta["col_idx_start"]) for d in result["documents"]] == [(0, 0), (2, 0)] def test_header_labels_do_not_reorder_sub_tables(): """Sub-tables keep the original column order, not the alphabetical one.""" - splitter = CSVDocumentSplitter( - row_split_threshold=None, - column_split_threshold=1, - read_csv_kwargs={"header": 0}, - ) + splitter = CSVDocumentSplitter(row_split_threshold=None, column_split_threshold=1, read_csv_kwargs={"header": 0}) result = splitter.run([Document(content=COLUMNS)]) - assert [ - (d.meta["col_idx_start"], d.meta["split_id"]) for d in result["documents"] - ] == [(0, 0), (2, 1)] + assert [(d.meta["col_idx_start"], d.meta["split_id"]) for d in result["documents"]] == [(0, 0), (2, 1)] assert [d.content.strip().replace("\n", "/") for d in result["documents"]] == ["1/3", "2/4"] From c8fba1a896f55ae8c9d344cb4a4c1ef7685577b3 Mon Sep 17 00:00:00 2001 From: sclfcz Date: Wed, 30 Sep 2026 09:18:21 +0800 Subject: [PATCH 6/6] style: sort the imports in the CSV splitter change CI's format job reported I001 (import block is un-sorted). --- haystack/components/preprocessors/csv_document_splitter.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/haystack/components/preprocessors/csv_document_splitter.py b/haystack/components/preprocessors/csv_document_splitter.py index fd7795898c..3850e4ba82 100644 --- a/haystack/components/preprocessors/csv_document_splitter.py +++ b/haystack/components/preprocessors/csv_document_splitter.py @@ -6,9 +6,10 @@ from io import StringIO from typing import Any, Literal, get_args -from haystack import Document, component, logging from haystack.lazy_imports import LazyImport +from haystack import Document, component, logging + with LazyImport("Run 'pip install pandas'") as pandas_import: import pandas as pd