diff --git a/haystack/components/converters/docx.py b/haystack/components/converters/docx.py index 41e2c761e5..6d2fc906b8 100644 --- a/haystack/components/converters/docx.py +++ b/haystack/components/converters/docx.py @@ -27,7 +27,7 @@ with LazyImport("Run 'pip install python-docx'") as docx_import: import docx from docx.document import Document as DocxDocument - from docx.table import Table + from docx.table import Table, _Cell from docx.text.hyperlink import Hyperlink from docx.text.paragraph import Paragraph from docx.text.run import Run @@ -327,6 +327,15 @@ def _escape_markdown_cell(text: str) -> str: # cell already contains cannot consume the escape. return _MARKDOWN_CELL_PIPE_PATTERN.sub(lambda match: match.group(1) * 2 + r"\|", text) + def _cell_text(self, cell: "_Cell") -> str: + """ + Returns a table cell's text with links formatted like links in body paragraphs. + + :param cell: The DOCX table cell. + :returns: The cell's paragraphs joined by newlines, as `cell.text` joins them. + """ + return "\n".join(self._process_links_in_paragraph(paragraph) for paragraph in cell.paragraphs) + def _table_to_markdown(self, table: "Table") -> str: """ Converts a DOCX table to a Markdown string. @@ -340,7 +349,7 @@ def _table_to_markdown(self, table: "Table") -> str: # Calculate max width for each column, on the escaped text that is written out for row in table.rows: for i, cell in enumerate(row.cells): - cell_text = self._escape_markdown_cell(cell.text.strip()) + cell_text = self._escape_markdown_cell(self._cell_text(cell).strip()) if i >= len(max_col_widths): max_col_widths.append(len(cell_text)) else: @@ -349,7 +358,7 @@ def _table_to_markdown(self, table: "Table") -> str: # Process rows for i, row in enumerate(table.rows): md_row = [ - self._escape_markdown_cell(cell.text.strip()).ljust(max_col_widths[j]) + self._escape_markdown_cell(self._cell_text(cell).strip()).ljust(max_col_widths[j]) for j, cell in enumerate(row.cells) ] markdown.append("| " + " | ".join(md_row) + " |") @@ -373,7 +382,7 @@ def _table_to_csv(self, table: "Table") -> str: # Process rows for row in table.rows: - csv_row = [cell.text.strip() for cell in row.cells] + csv_row = [self._cell_text(cell).strip() for cell in row.cells] csv_writer.writerow(csv_row) # Get the CSV as a string and strip any trailing newlines diff --git a/releasenotes/notes/docx-table-cell-links-c4f34cc2a07e8163.yaml b/releasenotes/notes/docx-table-cell-links-c4f34cc2a07e8163.yaml new file mode 100644 index 0000000000..12971611e2 --- /dev/null +++ b/releasenotes/notes/docx-table-cell-links-c4f34cc2a07e8163.yaml @@ -0,0 +1,9 @@ +--- +fixes: + - | + Fixed ``DOCXToDocument`` dropping hyperlink addresses inside tables. With + ``link_format`` set to ``markdown`` or ``plain``, links in table cells were + written as their display text only, while links in body paragraphs kept their + address. Links in table cells are now formatted the same way, in both the + ``markdown`` and ``csv`` table formats. The default ``link_format="none"`` + output is unchanged. diff --git a/test/components/converters/test_docx_file_to_document.py b/test/components/converters/test_docx_file_to_document.py index 688770ae29..d7509fe7e9 100644 --- a/test/components/converters/test_docx_file_to_document.py +++ b/test/components/converters/test_docx_file_to_document.py @@ -11,6 +11,9 @@ import docx import pytest +from docx.opc.constants import RELATIONSHIP_TYPE +from docx.oxml import OxmlElement +from docx.oxml.ns import qn from haystack import Document, Pipeline from haystack.components.converters.docx import DOCXLinkFormat, DOCXMetadata, DOCXTableFormat, DOCXToDocument @@ -491,3 +494,31 @@ def test_no_link_extraction(self, test_files_path): assert "[PDF](https://en.wikipedia.org/wiki/PDF)" not in content assert "PDF (https://en.wikipedia.org/wiki/PDF)" not in content + + @pytest.mark.parametrize("table_format", ["markdown", "csv"]) + @pytest.mark.parametrize( + ("link_format", "expected_link"), + [("markdown", "[docs](https://example.com/reference)"), ("plain", "docs (https://example.com/reference)")], + ) + def test_link_extraction_in_table(self, tmp_path, table_format, link_format, expected_link): + """A link in a table cell keeps its address, the same as a link in a body paragraph.""" + doc = docx.Document() + paragraph = doc.add_table(rows=1, cols=1).cell(0, 0).paragraphs[0] + relationship_id = paragraph.part.relate_to( + "https://example.com/reference", RELATIONSHIP_TYPE.HYPERLINK, is_external=True + ) + hyperlink = OxmlElement("w:hyperlink") + hyperlink.set(qn("r:id"), relationship_id) + run = OxmlElement("w:r") + text = OxmlElement("w:t") + text.text = "docs" + run.append(text) + hyperlink.append(run) + paragraph._p.append(hyperlink) + path = tmp_path / "table_with_link.docx" + doc.save(str(path)) + + converter = DOCXToDocument(table_format=table_format, link_format=link_format) + content = converter.run(sources=[path])["documents"][0].content + + assert expected_link in content