Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -148,6 +148,12 @@ def run(self, documents: list[Document]) -> dict[str, list[Document]]:
# Sort split_dfs first by row index, then by column index
split_dfs.sort(key=lambda dataframe: (dataframe.index[0], dataframe.columns[0]))

# Columns are only positional when ``header=None``. A caller passing
# ``read_csv_kwargs={"header": 0}`` gets the first row as string labels,
# and ``int(label)`` then failed for every sub-table. Map labels back to
# their position in the original frame instead.
column_positions = {label: position for position, label in enumerate(df.columns)}

for split_id, split_df in enumerate(split_dfs):
split_documents.append(
Document(
Expand All @@ -156,7 +162,7 @@ def run(self, documents: list[Document]) -> dict[str, list[Document]]:
**deepcopy(document.meta),
"source_id": document.id,
"row_idx_start": int(split_df.index[0]),
"col_idx_start": int(split_df.columns[0]),
"col_idx_start": column_positions[split_df.columns[0]],
"split_id": split_id,
},
)
Expand Down
4 changes: 3 additions & 1 deletion haystack/components/rankers/meta_field.py
Original file line number Diff line number Diff line change
Expand Up @@ -264,10 +264,12 @@ def run(
"The parameter <meta_field> is currently set to '{meta_field}', but none of the provided "
"Documents with IDs {document_ids} have this meta key.\n"
"Set <meta_field> to the name of a field that is present within the provided Documents.\n"
"Returning the <top_k> of the original Documents since there are no values to rank.",
"Applying the configured <missing_meta> policy instead of ranking.",
meta_field=self.meta_field,
document_ids=",".join([doc.id for doc in deduplicated_documents]),
)
if self.missing_meta == "drop":
return {"documents": []}
return {"documents": deduplicated_documents[:top_k]}

if len(docs_missing_meta_field) > 0:
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,27 @@
"""read_csv_kwargs that sets a header must not break CSV splitting."""

from haystack import Document
from haystack.components.preprocessors.csv_document_splitter import CSVDocumentSplitter

CSV = "name,score\nAda,9\n\nBob,8"


def test_default_header_none_keeps_positional_columns():
splitter = CSVDocumentSplitter(row_split_threshold=1, column_split_threshold=None)
result = splitter.run([Document(content=CSV)])
assert [
(d.meta["row_idx_start"], d.meta["col_idx_start"]) for d in result["documents"]
] == [(0, 0), (3, 0)]


def test_read_csv_kwargs_header_reports_column_positions():
"""With a caller-supplied header the columns are labels, not positions."""
splitter = CSVDocumentSplitter(
row_split_threshold=1,
column_split_threshold=None,
read_csv_kwargs={"header": 0},
)
result = splitter.run([Document(content=CSV)])
assert [
(d.meta["row_idx_start"], d.meta["col_idx_start"]) for d in result["documents"]
] == [(0, 0), (2, 0)]
19 changes: 19 additions & 0 deletions test/components/rankers/test_metafield.py
Original file line number Diff line number Diff line change
Expand Up @@ -338,3 +338,22 @@ def test_none_meta_value_is_handled_as_missing_meta(self, missing_meta, expected
]
output = ranker.run(documents=docs_before)
assert [doc.id for doc in output["documents"]] == expected_ids


def test_missing_meta_drop_when_no_document_has_the_field():
"""The missing_meta policy applies even when no document has the field.

The ranker returned every input document in that case, so
missing_meta="drop" silently stopped filtering: a batch with no rated
document kept the unrated ones.
"""
docs = [Document(id="a", content="a"), Document(id="b", content="b")]

assert (
MetaFieldRanker(meta_field="rating", missing_meta="drop").run(docs)["documents"]
== []
)

for policy in ("bottom", "top"):
ranker = MetaFieldRanker(meta_field="rating", missing_meta=policy)
assert [d.id for d in ranker.run(docs)["documents"]] == ["a", "b"]