From 067f02681fe701b4ce7d727dfa201f8b0561e85b Mon Sep 17 00:00:00 2001 From: chrikrah <48338417+chrikrah@users.noreply.github.com> Date: Sat, 26 Sep 2026 13:49:56 +0000 Subject: [PATCH 1/4] fix: keep an XLSX cell from breaking the markdown table XLSXToDocument wrote cell values straight into a pipe table with no escaping, so a cell containing a pipe gained a column and lost its tail to any Markdown reader, and a cell containing an Alt+Enter line break was split across two table rows. DOCXToDocument already solves this. _escape_markdown_cell in haystack/components/converters/docx.py was added for the same two patterns in PR #12779, merged two days ago. This ports it to the XLSX converter. The class docstring says the content is the table saved in CSV or Markdown format, so the two formats have to describe the same table. With B2 set to 'blue | red' the CSV path emits it intact and the Markdown path loses '| red'. The escape runs via DataFrame.map before the existing NaN substitution, with an astype(object) in between: pandas re-infers StringDtype after map, and the later where(notna(), None) can then no longer store None, which would undo #12776. --- haystack/components/converters/xlsx.py | 29 ++++++++++++++++++- ...xlsx-markdown-escape-a1b2c3d4e5f60718.yaml | 7 +++++ .../converters/test_xlsx_to_document.py | 22 ++++++++++++++ 3 files changed, 57 insertions(+), 1 deletion(-) create mode 100644 releasenotes/notes/xlsx-markdown-escape-a1b2c3d4e5f60718.yaml diff --git a/haystack/components/converters/xlsx.py b/haystack/components/converters/xlsx.py index 69fdb1b16cb..ddc4fd6325c 100644 --- a/haystack/components/converters/xlsx.py +++ b/haystack/components/converters/xlsx.py @@ -4,6 +4,7 @@ import io import os +import re from pathlib import Path from typing import Any, Literal @@ -14,6 +15,9 @@ logger = logging.getLogger(__name__) +_MARKDOWN_CELL_BREAK_PATTERN = re.compile(r"\s*(?:\r\n|\r|\n)\s*") +_MARKDOWN_CELL_PIPE_PATTERN = re.compile(r"(? Any: + """ + Makes a cell's value safe to put between the pipes of a Markdown table row. + + A cell holding an in-cell line break (Alt+Enter in Excel) would end the row in the middle, + and a pipe typed in a cell would be read as a column separator. A non-string value is + returned unchanged so `tabulate` keeps formatting numbers and `missingval` keeps working. + + :param value: The cell value. + :returns: The value with line breaks collapsed and pipes escaped, if it is a string. + """ + if not isinstance(value, str): + return value + value = _MARKDOWN_CELL_BREAK_PATTERN.sub(" ", value) + # The backslash run in front of the pipe is doubled first, so a backslash the + # cell already contains cannot consume the escape. + return _MARKDOWN_CELL_PIPE_PATTERN.sub(lambda match: match.group(1) * 2 + r"\|", value) + @staticmethod def _generate_excel_column_names(n_cols: int) -> list[str]: result = [] @@ -238,7 +261,11 @@ def _extract_tables(self, bytestream: ByteStream) -> tuple[list[str], list[dict] # reaches the formatter as a number and is written out as "nan". Replace # the empty cells with None so an empty cell reads as empty, the way # to_csv already writes it, and so missingval keeps working. - filled = value.astype(object).where(value.notna(), None) + # `.map()` re-infers the column dtype, so restore `object` before the NaN + # substitution below can store `None` in it. + escaped = value.astype(object).map(self._escape_markdown_cell).astype(object) + filled = escaped.where(value.notna(), None) + resolved_kwargs["headers"] = [self._escape_markdown_cell(h) for h in resolved_kwargs["headers"]] tables.append(filled.to_markdown(**resolved_kwargs)) # add sheet_name to metadata metadata.append({"xlsx": {"sheet_name": key}}) diff --git a/releasenotes/notes/xlsx-markdown-escape-a1b2c3d4e5f60718.yaml b/releasenotes/notes/xlsx-markdown-escape-a1b2c3d4e5f60718.yaml new file mode 100644 index 00000000000..a5cb6bd1d6e --- /dev/null +++ b/releasenotes/notes/xlsx-markdown-escape-a1b2c3d4e5f60718.yaml @@ -0,0 +1,7 @@ +--- +fixes: + - | + `XLSXToDocument` now escapes pipe characters and line breaks in cell values when + `table_format="markdown"`. A cell containing `|` previously gained a table column and lost + its tail to any Markdown reader, and a cell containing an Alt+Enter line break was split + across two table rows. This matches the escaping `DOCXToDocument` already applies. diff --git a/test/components/converters/test_xlsx_to_document.py b/test/components/converters/test_xlsx_to_document.py index fa2da6a29f3..e5979ebd44f 100644 --- a/test/components/converters/test_xlsx_to_document.py +++ b/test/components/converters/test_xlsx_to_document.py @@ -108,6 +108,28 @@ def test_run_markdown_missing_value(self, test_files_path: Path) -> None: == "| | A | B |\n|---:|:------|:------|\n| 1 | col_c | col_d |\n| 2 | True | N/A |" ) + @pytest.mark.parametrize( + ("cell_value", "expected_row"), + [ + pytest.param("a|b", "| 1 | a\\|b |", id="pipe"), + pytest.param("first line\nsecond line", "| 1 | first line second line |", id="line-break"), + ], + ) + def test_run_markdown_escapes_cell_content(self, tmp_path: Path, cell_value: str, expected_row: str) -> None: + """A pipe would be read as a column separator, and a line break would end the row in the middle.""" + workbook = Workbook() + workbook.active["A1"] = cell_value + path = tmp_path / "cell.xlsx" + workbook.save(path) + + content = XLSXToDocument(table_format="markdown").run(sources=[path])["documents"][0].content + rows = content.split("\n") + + assert len(rows) == 3 + assert rows[2] == expected_row + # Every row describes the same number of columns as the header. + assert all(row.count("|") - row.count("\\|") == 3 for row in rows) + @pytest.mark.parametrize( "sheet_name, expected_sheet_name, expected_content", [ From 0b81c4029b684fef4a99ebf0259c690edebfb29e Mon Sep 17 00:00:00 2001 From: anakin87 Date: Mon, 28 Sep 2026 17:55:01 +0200 Subject: [PATCH 2/4] simplify --- haystack/components/converters/xlsx.py | 39 ++++++------------- ...xlsx-markdown-escape-a1b2c3d4e5f60718.yaml | 7 ++-- .../converters/test_xlsx_to_document.py | 9 +++-- 3 files changed, 20 insertions(+), 35 deletions(-) diff --git a/haystack/components/converters/xlsx.py b/haystack/components/converters/xlsx.py index ddc4fd6325c..541dea7bff5 100644 --- a/haystack/components/converters/xlsx.py +++ b/haystack/components/converters/xlsx.py @@ -4,7 +4,6 @@ import io import os -import re from pathlib import Path from typing import Any, Literal @@ -15,9 +14,6 @@ logger = logging.getLogger(__name__) -_MARKDOWN_CELL_BREAK_PATTERN = re.compile(r"\s*(?:\r\n|\r|\n)\s*") -_MARKDOWN_CELL_PIPE_PATTERN = re.compile(r"(? Any: - """ - Makes a cell's value safe to put between the pipes of a Markdown table row. - - A cell holding an in-cell line break (Alt+Enter in Excel) would end the row in the middle, - and a pipe typed in a cell would be read as a column separator. A non-string value is - returned unchanged so `tabulate` keeps formatting numbers and `missingval` keeps working. - - :param value: The cell value. - :returns: The value with line breaks collapsed and pipes escaped, if it is a string. - """ - if not isinstance(value, str): - return value - value = _MARKDOWN_CELL_BREAK_PATTERN.sub(" ", value) - # The backslash run in front of the pipe is doubled first, so a backslash the - # cell already contains cannot consume the escape. - return _MARKDOWN_CELL_PIPE_PATTERN.sub(lambda match: match.group(1) * 2 + r"\|", value) - @staticmethod def _generate_excel_column_names(n_cols: int) -> list[str]: result = [] @@ -257,15 +234,21 @@ def _extract_tables(self, bytestream: ByteStream) -> tuple[list[str], list[dict] "missingval": "", **self.table_format_kwargs, } + formatted = value + if resolved_kwargs["tablefmt"] == "pipe": + formatted = formatted.replace( + { + r"\r\n|\r|\n": " ", # keep in-cell line breaks from creating extra Markdown rows + r"(\\*)\|": r"\1\1\\|", # escape pipes but preserve any preceding literal backslashes + }, + regex=True, + ) + # to_markdown uses tabulate, whose missingval only covers None: a NaN # reaches the formatter as a number and is written out as "nan". Replace # the empty cells with None so an empty cell reads as empty, the way # to_csv already writes it, and so missingval keeps working. - # `.map()` re-infers the column dtype, so restore `object` before the NaN - # substitution below can store `None` in it. - escaped = value.astype(object).map(self._escape_markdown_cell).astype(object) - filled = escaped.where(value.notna(), None) - resolved_kwargs["headers"] = [self._escape_markdown_cell(h) for h in resolved_kwargs["headers"]] + filled = formatted.astype(object).where(value.notna(), None) tables.append(filled.to_markdown(**resolved_kwargs)) # add sheet_name to metadata metadata.append({"xlsx": {"sheet_name": key}}) diff --git a/releasenotes/notes/xlsx-markdown-escape-a1b2c3d4e5f60718.yaml b/releasenotes/notes/xlsx-markdown-escape-a1b2c3d4e5f60718.yaml index a5cb6bd1d6e..561316176a3 100644 --- a/releasenotes/notes/xlsx-markdown-escape-a1b2c3d4e5f60718.yaml +++ b/releasenotes/notes/xlsx-markdown-escape-a1b2c3d4e5f60718.yaml @@ -1,7 +1,6 @@ --- fixes: - | - `XLSXToDocument` now escapes pipe characters and line breaks in cell values when - `table_format="markdown"`. A cell containing `|` previously gained a table column and lost - its tail to any Markdown reader, and a cell containing an Alt+Enter line break was split - across two table rows. This matches the escaping `DOCXToDocument` already applies. + ``XLSXToDocument`` now escapes pipe characters and replaces in-cell line breaks + with spaces in the default Markdown pipe table output. This keeps cell content + from being interpreted as additional table columns or rows. diff --git a/test/components/converters/test_xlsx_to_document.py b/test/components/converters/test_xlsx_to_document.py index e5979ebd44f..458d564568d 100644 --- a/test/components/converters/test_xlsx_to_document.py +++ b/test/components/converters/test_xlsx_to_document.py @@ -112,23 +112,26 @@ def test_run_markdown_missing_value(self, test_files_path: Path) -> None: ("cell_value", "expected_row"), [ pytest.param("a|b", "| 1 | a\\|b |", id="pipe"), + pytest.param(r"a\|b", r"| 1 | a\\\|b |", id="backslash-before-pipe"), + pytest.param("first\r\nsecond", "| 1 | first second |", id="windows-line-break"), pytest.param("first line\nsecond line", "| 1 | first line second line |", id="line-break"), ], ) def test_run_markdown_escapes_cell_content(self, tmp_path: Path, cell_value: str, expected_row: str) -> None: """A pipe would be read as a column separator, and a line break would end the row in the middle.""" workbook = Workbook() - workbook.active["A1"] = cell_value + sheet = workbook.active + assert sheet is not None + sheet["A1"] = cell_value path = tmp_path / "cell.xlsx" workbook.save(path) content = XLSXToDocument(table_format="markdown").run(sources=[path])["documents"][0].content + assert content is not None rows = content.split("\n") assert len(rows) == 3 assert rows[2] == expected_row - # Every row describes the same number of columns as the header. - assert all(row.count("|") - row.count("\\|") == 3 for row in rows) @pytest.mark.parametrize( "sheet_name, expected_sheet_name, expected_content", From e237f578af1b27387c7589654e6aa4109a3d6964 Mon Sep 17 00:00:00 2001 From: anakin87 Date: Mon, 28 Sep 2026 17:57:51 +0200 Subject: [PATCH 3/4] more --- haystack/components/converters/xlsx.py | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/haystack/components/converters/xlsx.py b/haystack/components/converters/xlsx.py index 541dea7bff5..86aeaa0d769 100644 --- a/haystack/components/converters/xlsx.py +++ b/haystack/components/converters/xlsx.py @@ -234,9 +234,8 @@ def _extract_tables(self, bytestream: ByteStream) -> tuple[list[str], list[dict] "missingval": "", **self.table_format_kwargs, } - formatted = value if resolved_kwargs["tablefmt"] == "pipe": - formatted = formatted.replace( + value = value.replace( { r"\r\n|\r|\n": " ", # keep in-cell line breaks from creating extra Markdown rows r"(\\*)\|": r"\1\1\\|", # escape pipes but preserve any preceding literal backslashes @@ -248,7 +247,7 @@ def _extract_tables(self, bytestream: ByteStream) -> tuple[list[str], list[dict] # reaches the formatter as a number and is written out as "nan". Replace # the empty cells with None so an empty cell reads as empty, the way # to_csv already writes it, and so missingval keeps working. - filled = formatted.astype(object).where(value.notna(), None) + filled = value.astype(object).where(value.notna(), None) tables.append(filled.to_markdown(**resolved_kwargs)) # add sheet_name to metadata metadata.append({"xlsx": {"sheet_name": key}}) From 26f59a6d89dfb880c4d828f7983fcb963c19f0c6 Mon Sep 17 00:00:00 2001 From: anakin87 Date: Mon, 28 Sep 2026 18:00:18 +0200 Subject: [PATCH 4/4] fix relnote filename --- ...d4e5f60718.yaml => xlsx-markdown-escape-25b936b365fd48f1.yaml} | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename releasenotes/notes/{xlsx-markdown-escape-a1b2c3d4e5f60718.yaml => xlsx-markdown-escape-25b936b365fd48f1.yaml} (100%) diff --git a/releasenotes/notes/xlsx-markdown-escape-a1b2c3d4e5f60718.yaml b/releasenotes/notes/xlsx-markdown-escape-25b936b365fd48f1.yaml similarity index 100% rename from releasenotes/notes/xlsx-markdown-escape-a1b2c3d4e5f60718.yaml rename to releasenotes/notes/xlsx-markdown-escape-25b936b365fd48f1.yaml