diff --git a/haystack/components/converters/xlsx.py b/haystack/components/converters/xlsx.py index 69fdb1b16c..86aeaa0d76 100644 --- a/haystack/components/converters/xlsx.py +++ b/haystack/components/converters/xlsx.py @@ -234,6 +234,15 @@ def _extract_tables(self, bytestream: ByteStream) -> tuple[list[str], list[dict] "missingval": "", **self.table_format_kwargs, } + if resolved_kwargs["tablefmt"] == "pipe": + value = value.replace( + { + r"\r\n|\r|\n": " ", # keep in-cell line breaks from creating extra Markdown rows + r"(\\*)\|": r"\1\1\\|", # escape pipes but preserve any preceding literal backslashes + }, + regex=True, + ) + # to_markdown uses tabulate, whose missingval only covers None: a NaN # reaches the formatter as a number and is written out as "nan". Replace # the empty cells with None so an empty cell reads as empty, the way diff --git a/releasenotes/notes/xlsx-markdown-escape-25b936b365fd48f1.yaml b/releasenotes/notes/xlsx-markdown-escape-25b936b365fd48f1.yaml new file mode 100644 index 0000000000..561316176a --- /dev/null +++ b/releasenotes/notes/xlsx-markdown-escape-25b936b365fd48f1.yaml @@ -0,0 +1,6 @@ +--- +fixes: + - | + ``XLSXToDocument`` now escapes pipe characters and replaces in-cell line breaks + with spaces in the default Markdown pipe table output. This keeps cell content + from being interpreted as additional table columns or rows. diff --git a/test/components/converters/test_xlsx_to_document.py b/test/components/converters/test_xlsx_to_document.py index fa2da6a29f..458d564568 100644 --- a/test/components/converters/test_xlsx_to_document.py +++ b/test/components/converters/test_xlsx_to_document.py @@ -108,6 +108,31 @@ def test_run_markdown_missing_value(self, test_files_path: Path) -> None: == "| | A | B |\n|---:|:------|:------|\n| 1 | col_c | col_d |\n| 2 | True | N/A |" ) + @pytest.mark.parametrize( + ("cell_value", "expected_row"), + [ + pytest.param("a|b", "| 1 | a\\|b |", id="pipe"), + pytest.param(r"a\|b", r"| 1 | a\\\|b |", id="backslash-before-pipe"), + pytest.param("first\r\nsecond", "| 1 | first second |", id="windows-line-break"), + pytest.param("first line\nsecond line", "| 1 | first line second line |", id="line-break"), + ], + ) + def test_run_markdown_escapes_cell_content(self, tmp_path: Path, cell_value: str, expected_row: str) -> None: + """A pipe would be read as a column separator, and a line break would end the row in the middle.""" + workbook = Workbook() + sheet = workbook.active + assert sheet is not None + sheet["A1"] = cell_value + path = tmp_path / "cell.xlsx" + workbook.save(path) + + content = XLSXToDocument(table_format="markdown").run(sources=[path])["documents"][0].content + assert content is not None + rows = content.split("\n") + + assert len(rows) == 3 + assert rows[2] == expected_row + @pytest.mark.parametrize( "sheet_name, expected_sheet_name, expected_content", [