Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 13 additions & 4 deletions haystack/components/converters/docx.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,7 +27,7 @@
with LazyImport("Run 'pip install python-docx'") as docx_import:
import docx
from docx.document import Document as DocxDocument
from docx.table import Table
from docx.table import Table, _Cell
from docx.text.hyperlink import Hyperlink
from docx.text.paragraph import Paragraph
from docx.text.run import Run
Expand Down Expand Up @@ -327,6 +327,15 @@ def _escape_markdown_cell(text: str) -> str:
# cell already contains cannot consume the escape.
return _MARKDOWN_CELL_PIPE_PATTERN.sub(lambda match: match.group(1) * 2 + r"\|", text)

def _cell_text(self, cell: "_Cell") -> str:
"""
Returns a table cell's text with links formatted like links in body paragraphs.

:param cell: The DOCX table cell.
:returns: The cell's paragraphs joined by newlines, as `cell.text` joins them.
"""
return "\n".join(self._process_links_in_paragraph(paragraph) for paragraph in cell.paragraphs)

def _table_to_markdown(self, table: "Table") -> str:
"""
Converts a DOCX table to a Markdown string.
Expand All @@ -340,7 +349,7 @@ def _table_to_markdown(self, table: "Table") -> str:
# Calculate max width for each column, on the escaped text that is written out
for row in table.rows:
for i, cell in enumerate(row.cells):
cell_text = self._escape_markdown_cell(cell.text.strip())
cell_text = self._escape_markdown_cell(self._cell_text(cell).strip())
if i >= len(max_col_widths):
max_col_widths.append(len(cell_text))
else:
Expand All @@ -349,7 +358,7 @@ def _table_to_markdown(self, table: "Table") -> str:
# Process rows
for i, row in enumerate(table.rows):
md_row = [
self._escape_markdown_cell(cell.text.strip()).ljust(max_col_widths[j])
self._escape_markdown_cell(self._cell_text(cell).strip()).ljust(max_col_widths[j])
for j, cell in enumerate(row.cells)
]
markdown.append("| " + " | ".join(md_row) + " |")
Expand All @@ -373,7 +382,7 @@ def _table_to_csv(self, table: "Table") -> str:

# Process rows
for row in table.rows:
csv_row = [cell.text.strip() for cell in row.cells]
csv_row = [self._cell_text(cell).strip() for cell in row.cells]
csv_writer.writerow(csv_row)

# Get the CSV as a string and strip any trailing newlines
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,9 @@
---
fixes:
- |
Fixed ``DOCXToDocument`` dropping hyperlink addresses inside tables. With
``link_format`` set to ``markdown`` or ``plain``, links in table cells were
written as their display text only, while links in body paragraphs kept their
address. Links in table cells are now formatted the same way, in both the
``markdown`` and ``csv`` table formats. The default ``link_format="none"``
output is unchanged.
31 changes: 31 additions & 0 deletions test/components/converters/test_docx_file_to_document.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,9 @@

import docx
import pytest
from docx.opc.constants import RELATIONSHIP_TYPE
from docx.oxml import OxmlElement
from docx.oxml.ns import qn

from haystack import Document, Pipeline
from haystack.components.converters.docx import DOCXLinkFormat, DOCXMetadata, DOCXTableFormat, DOCXToDocument
Expand Down Expand Up @@ -491,3 +494,31 @@ def test_no_link_extraction(self, test_files_path):

assert "[PDF](https://en.wikipedia.org/wiki/PDF)" not in content
assert "PDF (https://en.wikipedia.org/wiki/PDF)" not in content

@pytest.mark.parametrize("table_format", ["markdown", "csv"])
@pytest.mark.parametrize(
("link_format", "expected_link"),
[("markdown", "[docs](https://example.com/reference)"), ("plain", "docs (https://example.com/reference)")],
)
def test_link_extraction_in_table(self, tmp_path, table_format, link_format, expected_link):
"""A link in a table cell keeps its address, the same as a link in a body paragraph."""
doc = docx.Document()
paragraph = doc.add_table(rows=1, cols=1).cell(0, 0).paragraphs[0]
relationship_id = paragraph.part.relate_to(
"https://example.com/reference", RELATIONSHIP_TYPE.HYPERLINK, is_external=True
)
hyperlink = OxmlElement("w:hyperlink")
hyperlink.set(qn("r:id"), relationship_id)
run = OxmlElement("w:r")
text = OxmlElement("w:t")
text.text = "docs"
run.append(text)
hyperlink.append(run)
paragraph._p.append(hyperlink)
path = tmp_path / "table_with_link.docx"
doc.save(str(path))

converter = DOCXToDocument(table_format=table_format, link_format=link_format)
content = converter.run(sources=[path])["documents"][0].content

assert expected_link in content
Loading