Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
60 changes: 30 additions & 30 deletions packages/markitdown-ocr/tests/test_xlsx_converter.py
Original file line number Diff line number Diff line change
Expand Up @@ -95,24 +95,24 @@ def test_xlsx_image_middle(svc: MockOCRService) -> None:
"## Revenue\n"
"| Q1 Report | Unnamed: 1 |\n"
"| --- | --- |\n"
"| NaN | NaN |\n"
"| | |\n"
"| Revenue | $50,000 |\n"
"| NaN | NaN |\n"
"| NaN | NaN |\n"
"| NaN | NaN |\n"
"| NaN | NaN |\n"
"| | |\n"
"| | |\n"
"| | |\n"
"| | |\n"
"| Profit Margin | 40% |\n\n"
"### Images in this sheet:\n\n"
f"{_OCR_BLOCK}\n\n"
"## Expenses\n"
"| Expense Breakdown | Unnamed: 1 |\n"
"| --- | --- |\n"
"| NaN | NaN |\n"
"| | |\n"
"| Expenses | $30,000 |\n"
"| NaN | NaN |\n"
"| NaN | NaN |\n"
"| NaN | NaN |\n"
"| NaN | NaN |\n"
"| | |\n"
"| | |\n"
"| | |\n"
"| | |\n"
"| Savings | $5,000 |\n\n"
"### Images in this sheet:\n\n"
f"{_OCR_BLOCK}"
Expand All @@ -133,12 +133,12 @@ def test_xlsx_image_end(svc: MockOCRService) -> None:
"| Total Revenue | $500,000 |\n"
"| Total Expenses | $300,000 |\n"
"| Net Profit | $200,000 |\n"
"| NaN | NaN |\n"
"| NaN | NaN |\n"
"| NaN | NaN |\n"
"| NaN | NaN |\n"
"| NaN | NaN |\n"
"| Signature: | NaN |\n\n"
"| | |\n"
"| | |\n"
"| | |\n"
"| | |\n"
"| | |\n"
"| Signature: | |\n\n"
"### Images in this sheet:\n\n"
f"{_OCR_BLOCK}\n\n"
"## Budget\n"
Expand All @@ -147,12 +147,12 @@ def test_xlsx_image_end(svc: MockOCRService) -> None:
"| Marketing | $100,000 |\n"
"| R&D | $150,000 |\n"
"| Operations | $50,000 |\n"
"| NaN | NaN |\n"
"| NaN | NaN |\n"
"| NaN | NaN |\n"
"| NaN | NaN |\n"
"| NaN | NaN |\n"
"| Approved: | NaN |\n\n"
"| | |\n"
"| | |\n"
"| | |\n"
"| | |\n"
"| | |\n"
"| Approved: | |\n\n"
"### Images in this sheet:\n\n"
f"{_OCR_BLOCK}"
)
Expand All @@ -170,10 +170,10 @@ def test_xlsx_multiple_images(svc: MockOCRService) -> None:
"| Dashboard |\n"
"| --- |\n"
"| Status: Active |\n"
"| NaN |\n"
"| NaN |\n"
"| NaN |\n"
"| NaN |\n"
"| |\n"
"| |\n"
"| |\n"
"| |\n"
"| Performance Summary |\n\n"
"### Images in this sheet:\n\n"
f"{_OCR_BLOCK}\n\n"
Expand Down Expand Up @@ -204,27 +204,27 @@ def test_xlsx_complex_layout(svc: MockOCRService) -> None:
"## Complex Report\n"
"| Annual Report 2024 | Unnamed: 1 |\n"
"| --- | --- |\n"
"| NaN | NaN |\n"
"| | |\n"
"| Month | Sales |\n"
"| Jan | 1000 |\n"
"| Feb | 1200 |\n"
"| NaN | NaN |\n"
"| | |\n"
"| Total | 2200 |\n\n"
"### Images in this sheet:\n\n"
f"{_OCR_BLOCK}\n\n"
f"{_OCR_BLOCK}\n\n"
"## Customers\n"
"| Customer Metrics | Unnamed: 1 |\n"
"| --- | --- |\n"
"| NaN | NaN |\n"
"| | |\n"
"| New Customers | 250 |\n"
"| Retention Rate | 92% |\n\n"
"### Images in this sheet:\n\n"
f"{_OCR_BLOCK}\n\n"
"## Regions\n"
"| Regional Breakdown | Unnamed: 1 |\n"
"| --- | --- |\n"
"| NaN | NaN |\n"
"| | |\n"
"| Region | Revenue |\n"
"| North | $800K |\n"
"| South | $600K |\n\n"
Expand Down
27 changes: 26 additions & 1 deletion packages/markitdown-ocr/tests/test_xlsx_inheritance.py
Original file line number Diff line number Diff line change
Expand Up @@ -288,10 +288,35 @@ def fixed_read(*args: Any, **kwargs: Any):
assert "Future shared fix" in result and "<native>" not in result
assert result.count("[Image OCR]") == 2
repair.assert_called_once()
assert calls == [{"sheet_name": None, "engine": "openpyxl"}] * 2
assert calls == [{"sheet_name": None, "engine": "openpyxl", "dtype": object}] * 2
service.extract_text.assert_called_once()


def test_blank_cell_stays_blank_and_keeps_its_column_native() -> None:
workbook = openpyxl.Workbook()
sheet = workbook.active
sheet.title = "Cells"
sheet.append(["Units", "Shipped"])
sheet.append([12, True])
sheet.append([None, None])
sheet.append([7, False])
sheet.add_image(SheetImage(io.BytesIO(_RED)), "D1")
stream = io.BytesIO()
workbook.save(stream)
workbook.close()

assert _convert(XlsxConverterWithOCR(_service()), stream.getvalue()) == (
"## Cells\n"
"| Units | Shipped |\n"
"| --- | --- |\n"
"| 12 | True |\n"
"| | |\n"
"| 7 | False |\n\n"
"### Images in this sheet:\n\n"
"*[Image OCR] \nrecognized \n[End OCR]*"
)


def test_reported_ocr_error_warns_once_and_keeps_native_output() -> None:
service = Mock(
extract_text=Mock(
Expand Down
55 changes: 50 additions & 5 deletions packages/markitdown/src/markitdown/converters/_xlsx_converter.py
Original file line number Diff line number Diff line change
@@ -1,3 +1,4 @@
import datetime
import io
import re
import sys
Expand Down Expand Up @@ -52,18 +53,60 @@ def _read_xlsx_sheets(
repaired_stream = None
try:
try:
sheets = pd.read_excel(file_stream, sheet_name=None, engine="openpyxl")
sheets = pd.read_excel(
file_stream, sheet_name=None, engine="openpyxl", dtype=object
)
except TypeError as exc:
if "showZeroes" not in str(exc):
raise
repaired_stream = _repair_sheetview_show_zeroes(file_stream, start_pos)
sheets = pd.read_excel(repaired_stream, sheet_name=None, engine="openpyxl")
sheets = pd.read_excel(
repaired_stream, sheet_name=None, engine="openpyxl", dtype=object
)
yield sheets, repaired_stream if repaired_stream is not None else file_stream
finally:
if repaired_stream is not None:
repaired_stream.close()


def _format_cell(value: Any) -> str:
"""Render one cell the way the spreadsheet holds it.

openpyxl hands back a ``datetime`` for a date cell, and ``str()`` on it
appends a midnight time that is not in the spreadsheet.
"""
if isinstance(value, datetime.datetime):
if value.time() == datetime.time.min:
return value.date().isoformat()
return value.isoformat(sep=" ")
return str(value)


def _format_float(value: float) -> str:
"""Render a fractional number with the 15 significant digits Excel shows."""
return format(value, ".15g")


def sheet_to_html(sheet: Any) -> str:
"""Render one sheet as an HTML table.

``na_rep=""`` keeps an empty cell empty: the default writes the string
``NaN`` into it, which reads as a value rather than as a blank. The
formatters are what keep the remaining cells rendered as themselves --
pandas applies ``na_rep`` to the blanks, ``float_format`` to fractional
numbers, and the formatter to everything else, without re-inferring a
dtype for the column. A float needs ``float_format`` because with
``index=False`` pandas renders it with its own display precision and
never reaches ``formatters``.
"""
return sheet.to_html(
index=False,
na_rep="",
formatters=[_format_cell] * len(sheet.columns),
float_format=_format_float,
)


def _rename_show_zeroes_attribute(data: bytes) -> bytes:
return _SHEET_VIEW_START_TAG.sub(
lambda match: _SHOW_ZEROES_ATTRIBUTE.sub(rb"showZeros\1", match.group(0)),
Expand Down Expand Up @@ -150,7 +193,7 @@ def convert(

for s in sheets:
md_content += f"## {s}\n"
html_content = sheets[s].to_html(index=False)
html_content = sheet_to_html(sheets[s])
md_content += (
self._html_converter.convert_string(
html_content, **kwargs
Expand Down Expand Up @@ -237,11 +280,13 @@ def convert(
_xls_dependency_exc_info[2]
)

sheets = pd.read_excel(file_stream, sheet_name=None, engine="xlrd")
sheets = pd.read_excel(
file_stream, sheet_name=None, engine="xlrd", dtype=object
)
md_content = ""
for s in sheets:
md_content += f"## {s}\n"
html_content = sheets[s].to_html(index=False)
html_content = sheet_to_html(sheets[s])
md_content += (
self._html_converter.convert_string(
html_content, **kwargs
Expand Down
Binary file not shown.
Loading