From 6c6359311b8f2bcf2331bafefbfbd3b1106bacc2 Mon Sep 17 00:00:00 2001 From: Jayanth Date: Sun, 27 Sep 2026 15:47:56 +0000 Subject: [PATCH] fix: keep hyperlink addresses in DOCXToDocument table cells Table cells were serialized from cell.text, which holds only the display text of a hyperlink, so link_format="markdown" or "plain" kept addresses in body paragraphs but dropped them in tables. Cell paragraphs now go through the same link handling as body paragraphs, in both the markdown and csv table formats. link_format="none" output is unchanged. Fixes #12977 Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_016PQv3cLuiTKPtLNj9i1cma Signed-off-by: Jayanth --- haystack/components/converters/docx.py | 17 +++++++--- ...ocx-table-cell-links-c4f34cc2a07e8163.yaml | 9 ++++++ .../converters/test_docx_file_to_document.py | 31 +++++++++++++++++++ 3 files changed, 53 insertions(+), 4 deletions(-) create mode 100644 releasenotes/notes/docx-table-cell-links-c4f34cc2a07e8163.yaml diff --git a/haystack/components/converters/docx.py b/haystack/components/converters/docx.py index 41e2c761e5..6d2fc906b8 100644 --- a/haystack/components/converters/docx.py +++ b/haystack/components/converters/docx.py @@ -27,7 +27,7 @@ with LazyImport("Run 'pip install python-docx'") as docx_import: import docx from docx.document import Document as DocxDocument - from docx.table import Table + from docx.table import Table, _Cell from docx.text.hyperlink import Hyperlink from docx.text.paragraph import Paragraph from docx.text.run import Run @@ -327,6 +327,15 @@ def _escape_markdown_cell(text: str) -> str: # cell already contains cannot consume the escape. return _MARKDOWN_CELL_PIPE_PATTERN.sub(lambda match: match.group(1) * 2 + r"\|", text) + def _cell_text(self, cell: "_Cell") -> str: + """ + Returns a table cell's text with links formatted like links in body paragraphs. + + :param cell: The DOCX table cell. + :returns: The cell's paragraphs joined by newlines, as `cell.text` joins them. + """ + return "\n".join(self._process_links_in_paragraph(paragraph) for paragraph in cell.paragraphs) + def _table_to_markdown(self, table: "Table") -> str: """ Converts a DOCX table to a Markdown string. @@ -340,7 +349,7 @@ def _table_to_markdown(self, table: "Table") -> str: # Calculate max width for each column, on the escaped text that is written out for row in table.rows: for i, cell in enumerate(row.cells): - cell_text = self._escape_markdown_cell(cell.text.strip()) + cell_text = self._escape_markdown_cell(self._cell_text(cell).strip()) if i >= len(max_col_widths): max_col_widths.append(len(cell_text)) else: @@ -349,7 +358,7 @@ def _table_to_markdown(self, table: "Table") -> str: # Process rows for i, row in enumerate(table.rows): md_row = [ - self._escape_markdown_cell(cell.text.strip()).ljust(max_col_widths[j]) + self._escape_markdown_cell(self._cell_text(cell).strip()).ljust(max_col_widths[j]) for j, cell in enumerate(row.cells) ] markdown.append("| " + " | ".join(md_row) + " |") @@ -373,7 +382,7 @@ def _table_to_csv(self, table: "Table") -> str: # Process rows for row in table.rows: - csv_row = [cell.text.strip() for cell in row.cells] + csv_row = [self._cell_text(cell).strip() for cell in row.cells] csv_writer.writerow(csv_row) # Get the CSV as a string and strip any trailing newlines diff --git a/releasenotes/notes/docx-table-cell-links-c4f34cc2a07e8163.yaml b/releasenotes/notes/docx-table-cell-links-c4f34cc2a07e8163.yaml new file mode 100644 index 0000000000..12971611e2 --- /dev/null +++ b/releasenotes/notes/docx-table-cell-links-c4f34cc2a07e8163.yaml @@ -0,0 +1,9 @@ +--- +fixes: + - | + Fixed ``DOCXToDocument`` dropping hyperlink addresses inside tables. With + ``link_format`` set to ``markdown`` or ``plain``, links in table cells were + written as their display text only, while links in body paragraphs kept their + address. Links in table cells are now formatted the same way, in both the + ``markdown`` and ``csv`` table formats. The default ``link_format="none"`` + output is unchanged. diff --git a/test/components/converters/test_docx_file_to_document.py b/test/components/converters/test_docx_file_to_document.py index 688770ae29..d7509fe7e9 100644 --- a/test/components/converters/test_docx_file_to_document.py +++ b/test/components/converters/test_docx_file_to_document.py @@ -11,6 +11,9 @@ import docx import pytest +from docx.opc.constants import RELATIONSHIP_TYPE +from docx.oxml import OxmlElement +from docx.oxml.ns import qn from haystack import Document, Pipeline from haystack.components.converters.docx import DOCXLinkFormat, DOCXMetadata, DOCXTableFormat, DOCXToDocument @@ -491,3 +494,31 @@ def test_no_link_extraction(self, test_files_path): assert "[PDF](https://en.wikipedia.org/wiki/PDF)" not in content assert "PDF (https://en.wikipedia.org/wiki/PDF)" not in content + + @pytest.mark.parametrize("table_format", ["markdown", "csv"]) + @pytest.mark.parametrize( + ("link_format", "expected_link"), + [("markdown", "[docs](https://example.com/reference)"), ("plain", "docs (https://example.com/reference)")], + ) + def test_link_extraction_in_table(self, tmp_path, table_format, link_format, expected_link): + """A link in a table cell keeps its address, the same as a link in a body paragraph.""" + doc = docx.Document() + paragraph = doc.add_table(rows=1, cols=1).cell(0, 0).paragraphs[0] + relationship_id = paragraph.part.relate_to( + "https://example.com/reference", RELATIONSHIP_TYPE.HYPERLINK, is_external=True + ) + hyperlink = OxmlElement("w:hyperlink") + hyperlink.set(qn("r:id"), relationship_id) + run = OxmlElement("w:r") + text = OxmlElement("w:t") + text.text = "docs" + run.append(text) + hyperlink.append(run) + paragraph._p.append(hyperlink) + path = tmp_path / "table_with_link.docx" + doc.save(str(path)) + + converter = DOCXToDocument(table_format=table_format, link_format=link_format) + content = converter.run(sources=[path])["documents"][0].content + + assert expected_link in content