From 58d3455a2ef08fe32f543f0079199508fb5bb690 Mon Sep 17 00:00:00 2001 From: sclfcz Date: Fri, 25 Sep 2026 23:39:54 +0800 Subject: [PATCH 1/2] fix(MetaFieldRanker): apply missing_meta when no document has the field When every document lacks the ranking field, run() returned the input documents unchanged, bypassing the configured missing_meta policy: a batch with one rated document dropped the unrated ones, while an entirely unrated batch passed through intact even with missing_meta="drop". Apply the policy in that branch as well (drop -> empty list, top/bottom keep the documents, as before) and keep the warning that explains the situation. --- haystack/components/rankers/meta_field.py | 4 +++- test/components/rankers/test_metafield.py | 19 +++++++++++++++++++ 2 files changed, 22 insertions(+), 1 deletion(-) diff --git a/haystack/components/rankers/meta_field.py b/haystack/components/rankers/meta_field.py index b5ce892394..cde5d5efba 100644 --- a/haystack/components/rankers/meta_field.py +++ b/haystack/components/rankers/meta_field.py @@ -264,10 +264,12 @@ def run( "The parameter is currently set to '{meta_field}', but none of the provided " "Documents with IDs {document_ids} have this meta key.\n" "Set to the name of a field that is present within the provided Documents.\n" - "Returning the of the original Documents since there are no values to rank.", + "Applying the configured policy instead of ranking.", meta_field=self.meta_field, document_ids=",".join([doc.id for doc in deduplicated_documents]), ) + if self.missing_meta == "drop": + return {"documents": []} return {"documents": deduplicated_documents[:top_k]} if len(docs_missing_meta_field) > 0: diff --git a/test/components/rankers/test_metafield.py b/test/components/rankers/test_metafield.py index a998ab4cb5..e65dffb700 100644 --- a/test/components/rankers/test_metafield.py +++ b/test/components/rankers/test_metafield.py @@ -338,3 +338,22 @@ def test_none_meta_value_is_handled_as_missing_meta(self, missing_meta, expected ] output = ranker.run(documents=docs_before) assert [doc.id for doc in output["documents"]] == expected_ids + + +def test_missing_meta_drop_when_no_document_has_the_field(): + """The missing_meta policy applies even when no document has the field. + + The ranker returned every input document in that case, so + missing_meta="drop" silently stopped filtering: a batch with no rated + document kept the unrated ones. + """ + docs = [Document(id="a", content="a"), Document(id="b", content="b")] + + assert ( + MetaFieldRanker(meta_field="rating", missing_meta="drop").run(docs)["documents"] + == [] + ) + + for policy in ("bottom", "top"): + ranker = MetaFieldRanker(meta_field="rating", missing_meta=policy) + assert [d.id for d in ranker.run(docs)["documents"]] == ["a", "b"] From 1d9ad449ba1e1aeeb70ecf85a70143f3edf9fde7 Mon Sep 17 00:00:00 2001 From: sclfcz Date: Sat, 26 Sep 2026 09:51:26 +0800 Subject: [PATCH 2/2] fix(CSVDocumentSplitter): report column positions when read_csv_kwargs sets a header The splitter overrides header to None so columns are integer positions and col_idx_start is read straight from columns[0]. A caller passing read_csv_kwargs={"header": 0} (or "infer") gets the first row as string labels, and int(columns[0]) then raised ValueError for every sub-table, so the document could not be split at all. Map the labels back to their position in the original frame. --- .../preprocessors/csv_document_splitter.py | 8 +++++- ...est_csv_document_splitter_header_kwargs.py | 27 +++++++++++++++++++ 2 files changed, 34 insertions(+), 1 deletion(-) create mode 100644 test/components/preprocessors/test_csv_document_splitter_header_kwargs.py diff --git a/haystack/components/preprocessors/csv_document_splitter.py b/haystack/components/preprocessors/csv_document_splitter.py index ea1dfa78c5..7d44fa4dee 100644 --- a/haystack/components/preprocessors/csv_document_splitter.py +++ b/haystack/components/preprocessors/csv_document_splitter.py @@ -148,6 +148,12 @@ def run(self, documents: list[Document]) -> dict[str, list[Document]]: # Sort split_dfs first by row index, then by column index split_dfs.sort(key=lambda dataframe: (dataframe.index[0], dataframe.columns[0])) + # Columns are only positional when ``header=None``. A caller passing + # ``read_csv_kwargs={"header": 0}`` gets the first row as string labels, + # and ``int(label)`` then failed for every sub-table. Map labels back to + # their position in the original frame instead. + column_positions = {label: position for position, label in enumerate(df.columns)} + for split_id, split_df in enumerate(split_dfs): split_documents.append( Document( @@ -156,7 +162,7 @@ def run(self, documents: list[Document]) -> dict[str, list[Document]]: **deepcopy(document.meta), "source_id": document.id, "row_idx_start": int(split_df.index[0]), - "col_idx_start": int(split_df.columns[0]), + "col_idx_start": column_positions[split_df.columns[0]], "split_id": split_id, }, ) diff --git a/test/components/preprocessors/test_csv_document_splitter_header_kwargs.py b/test/components/preprocessors/test_csv_document_splitter_header_kwargs.py new file mode 100644 index 0000000000..26cdadee6f --- /dev/null +++ b/test/components/preprocessors/test_csv_document_splitter_header_kwargs.py @@ -0,0 +1,27 @@ +"""read_csv_kwargs that sets a header must not break CSV splitting.""" + +from haystack import Document +from haystack.components.preprocessors.csv_document_splitter import CSVDocumentSplitter + +CSV = "name,score\nAda,9\n\nBob,8" + + +def test_default_header_none_keeps_positional_columns(): + splitter = CSVDocumentSplitter(row_split_threshold=1, column_split_threshold=None) + result = splitter.run([Document(content=CSV)]) + assert [ + (d.meta["row_idx_start"], d.meta["col_idx_start"]) for d in result["documents"] + ] == [(0, 0), (3, 0)] + + +def test_read_csv_kwargs_header_reports_column_positions(): + """With a caller-supplied header the columns are labels, not positions.""" + splitter = CSVDocumentSplitter( + row_split_threshold=1, + column_split_threshold=None, + read_csv_kwargs={"header": 0}, + ) + result = splitter.run([Document(content=CSV)]) + assert [ + (d.meta["row_idx_start"], d.meta["col_idx_start"]) for d in result["documents"] + ] == [(0, 0), (2, 0)]