From 1fb26a7ca9aa9d305ff932d4226d8ac3a1c7bd01 Mon Sep 17 00:00:00 2001 From: L4XB Date: Wed, 16 Sep 2026 14:26:06 +0200 Subject: [PATCH 1/4] fix(xlsx): a blank cell should not rewrite the rest of its column `DataFrame.to_html` writes the string `NaN` into an empty cell, and pandas upcasts any column that holds one. A spreadsheet with a single blank cell was therefore converted with `NaN` where the blank is, `12.0` where the sheet says `12`, `1.0` where it says `TRUE`, and `NaT` where a date is missing. Read the sheets with `dtype=object` so a blank no longer changes the type of its neighbours, render the blanks with `na_rep=""`, and pass a per-cell formatter so the remaining values are rendered as themselves - including a date cell, which openpyxl returns as a `datetime` whose `str()` would append a midnight time the spreadsheet does not have. `.xls` keeps pandas' own inference, since xlrd stores every number as a double and there is no integer to preserve; it shares the renderer, so its blanks stop reading as `NaN` too. The same two lines in `markitdown-ocr`'s XLSX converter now go through that renderer as well. --- .../_xlsx_converter_with_ocr.py | 11 +- .../markitdown/converters/_xlsx_converter.py | 44 +++++- .../markitdown/tests/test_xlsx_blank_cells.py | 148 ++++++++++++++++++ 3 files changed, 195 insertions(+), 8 deletions(-) create mode 100644 packages/markitdown/tests/test_xlsx_blank_cells.py diff --git a/packages/markitdown-ocr/src/markitdown_ocr/_xlsx_converter_with_ocr.py b/packages/markitdown-ocr/src/markitdown_ocr/_xlsx_converter_with_ocr.py index 481e071953..421b20646f 100644 --- a/packages/markitdown-ocr/src/markitdown_ocr/_xlsx_converter_with_ocr.py +++ b/packages/markitdown-ocr/src/markitdown_ocr/_xlsx_converter_with_ocr.py @@ -8,6 +8,7 @@ from typing import Any, BinaryIO, Optional from markitdown.converters import HtmlConverter +from markitdown.converters._xlsx_converter import sheet_to_html from markitdown import DocumentConverter, DocumentConverterResult, StreamInfo from markitdown._exceptions import ( MissingDependencyException, @@ -90,12 +91,14 @@ def _convert_standard( ) -> DocumentConverterResult: """Standard conversion without OCR.""" file_stream.seek(0) - sheets = pd.read_excel(file_stream, sheet_name=None, engine="openpyxl") + sheets = pd.read_excel( + file_stream, sheet_name=None, engine="openpyxl", dtype=object + ) md_content = "" for sheet_name in sheets: md_content += f"## {sheet_name}\n" - html_content = sheets[sheet_name].to_html(index=False) + html_content = sheet_to_html(sheets[sheet_name]) md_content += ( self._html_converter.convert_string( html_content, **kwargs @@ -122,9 +125,9 @@ def _convert_with_ocr( file_stream.seek(0) try: df = pd.read_excel( - file_stream, sheet_name=sheet_name, engine="openpyxl" + file_stream, sheet_name=sheet_name, engine="openpyxl", dtype=object ) - html_content = df.to_html(index=False) + html_content = sheet_to_html(df) md_content += ( self._html_converter.convert_string( html_content, **kwargs diff --git a/packages/markitdown/src/markitdown/converters/_xlsx_converter.py b/packages/markitdown/src/markitdown/converters/_xlsx_converter.py index 355dd8f8d7..bc65343859 100644 --- a/packages/markitdown/src/markitdown/converters/_xlsx_converter.py +++ b/packages/markitdown/src/markitdown/converters/_xlsx_converter.py @@ -1,3 +1,4 @@ +import datetime import io import re import sys @@ -46,13 +47,46 @@ def _read_xlsx_sheets(file_stream: BinaryIO) -> dict[str, Any]: start_pos = file_stream.tell() try: - return pd.read_excel(file_stream, sheet_name=None, engine="openpyxl") + return pd.read_excel( + file_stream, sheet_name=None, engine="openpyxl", dtype=object + ) except TypeError as exc: if "showZeroes" not in str(exc): raise repaired_stream = _repair_sheetview_show_zeroes(file_stream, start_pos) - return pd.read_excel(repaired_stream, sheet_name=None, engine="openpyxl") + return pd.read_excel( + repaired_stream, sheet_name=None, engine="openpyxl", dtype=object + ) + + +def _format_cell(value: Any) -> str: + """Render one cell the way the spreadsheet holds it. + + openpyxl hands back a ``datetime`` for a date cell, and ``str()`` on it + appends a midnight time that is not in the spreadsheet. + """ + if isinstance(value, datetime.datetime): + if value.time() == datetime.time.min: + return value.date().isoformat() + return value.isoformat(sep=" ") + return str(value) + + +def sheet_to_html(sheet: Any) -> str: + """Render one sheet as an HTML table. + + ``na_rep=""`` keeps an empty cell empty: the default writes the string + ``NaN`` into it, which reads as a value rather than as a blank. The + formatters are what keep the remaining cells rendered as themselves -- + pandas applies ``na_rep`` to the blanks and the formatter to everything + else, without re-inferring a dtype for the column. + """ + return sheet.to_html( + index=False, + na_rep="", + formatters=[_format_cell] * len(sheet.columns), + ) def _rename_show_zeroes_attribute(data: bytes) -> bytes: @@ -135,7 +169,7 @@ def convert( md_content = "" for s in sheets: md_content += f"## {s}\n" - html_content = sheets[s].to_html(index=False) + html_content = sheet_to_html(sheets[s]) md_content += ( self._html_converter.convert_string( html_content, **kwargs @@ -193,11 +227,13 @@ def convert( _xls_dependency_exc_info[2] ) + # xlrd stores every number as a double, so there is no integer to keep + # here and the sheets are read with pandas' own inference. sheets = pd.read_excel(file_stream, sheet_name=None, engine="xlrd") md_content = "" for s in sheets: md_content += f"## {s}\n" - html_content = sheets[s].to_html(index=False) + html_content = sheet_to_html(sheets[s]) md_content += ( self._html_converter.convert_string( html_content, **kwargs diff --git a/packages/markitdown/tests/test_xlsx_blank_cells.py b/packages/markitdown/tests/test_xlsx_blank_cells.py new file mode 100644 index 0000000000..7afdc30924 --- /dev/null +++ b/packages/markitdown/tests/test_xlsx_blank_cells.py @@ -0,0 +1,148 @@ +"""A blank cell must not rewrite the rest of its column. + +`DataFrame.to_html` writes the string ``NaN`` into an empty cell by default, and +pandas upcasts a column that holds one, so a single empty cell turned `12` into +`12.0` and `True` into `1.0` for every other row of that column. +""" + +import datetime +import io + +import pytest + +from markitdown import MarkItDown, StreamInfo + +openpyxl = pytest.importorskip("openpyxl") +pytest.importorskip("pandas") + + +def _xlsx(rows: list[list[object]]) -> io.BytesIO: + workbook = openpyxl.Workbook() + sheet = workbook.active + sheet.title = "Sheet1" + for row in rows: + sheet.append(row) + stream = io.BytesIO() + workbook.save(stream) + stream.seek(0) + return stream + + +def _convert(rows: list[list[object]]) -> str: + return ( + MarkItDown(enable_plugins=False) + .convert_stream( + _xlsx(rows), + stream_info=StreamInfo(extension=".xlsx"), + ) + .markdown + ) + + +def test_blank_cell_stays_blank() -> None: + markdown = _convert( + [ + ["Product", "Notes"], + ["Widget", None], + ["Gadget", "backordered"], + ] + ) + + assert markdown == ( + "## Sheet1\n" + "| Product | Notes |\n" + "| --- | --- |\n" + "| Widget | |\n" + "| Gadget | backordered |" + ) + assert "NaN" not in markdown + + +def test_a_blank_cell_does_not_turn_the_column_into_floats() -> None: + markdown = _convert( + [ + ["Units", "Year"], + [12, 2026], + [None, 2025], + ] + ) + + assert markdown == ( + "## Sheet1\n" + "| Units | Year |\n" + "| --- | --- |\n" + "| 12 | 2026 |\n" + "| | 2025 |" + ) + + +def test_a_blank_cell_does_not_turn_booleans_into_numbers() -> None: + markdown = _convert( + [ + ["Shipped", "Id"], + [True, 1], + [None, 2], + [False, 3], + ] + ) + + assert markdown == ( + "## Sheet1\n" + "| Shipped | Id |\n" + "| --- | --- |\n" + "| True | 1 |\n" + "| | 2 |\n" + "| False | 3 |" + ) + + +def test_a_date_cell_carries_no_time_and_a_blank_one_is_empty() -> None: + """A date cell has no time to show; a datetime cell keeps the one it has.""" + markdown = _convert( + [ + ["Due", "Id"], + [datetime.date(2026, 1, 5), 1], + [None, 2], + [datetime.datetime(2026, 2, 9, 14, 30), 3], + ] + ) + + assert markdown == ( + "## Sheet1\n" + "| Due | Id |\n" + "| --- | --- |\n" + "| 2026-01-05 | 1 |\n" + "| | 2 |\n" + "| 2026-02-09 14:30:00 | 3 |" + ) + + +def test_a_sheet_without_blanks_is_unchanged() -> None: + """Guard: the common case must render exactly as it did before.""" + markdown = _convert( + [ + ["Alpha", "Beta"], + [89, 82], + [76, 89], + ] + ) + + assert markdown == ( + "## Sheet1\n" + "| Alpha | Beta |\n" + "| --- | --- |\n" + "| 89 | 82 |\n" + "| 76 | 89 |" + ) + + +def test_a_cell_holding_markup_is_unchanged() -> None: + """Guard: the HTML round trip through the table renderer is untouched.""" + markdown = _convert( + [ + ["Note"], + ["bold & italic"], + ] + ) + + assert markdown == ("## Sheet1\n| Note |\n| --- |\n| bold & italic |") From c8db514ec45aa9176c644445fc75a6cd464bba4e Mon Sep 17 00:00:00 2001 From: L4XB Date: Thu, 17 Sep 2026 18:40:14 +0200 Subject: [PATCH 2/4] test(ocr): expect blank cells, not NaN, in the XLSX output The OCR converter renders its tables through `XlsxConverter` now, so the blank-cell fix applies to it. Its snapshots pinned the old `NaN` cells, and the inheritance test pinned the exact `read_excel` arguments. Both now expect the fixed behaviour. A new test converts a sheet with an image and a blank row between an int and a bool column. It checks that the OCR output keeps `12` and `True` and leaves the blank cells empty. --- .../tests/test_xlsx_converter.py | 60 +++++++++---------- .../tests/test_xlsx_inheritance.py | 27 ++++++++- 2 files changed, 56 insertions(+), 31 deletions(-) diff --git a/packages/markitdown-ocr/tests/test_xlsx_converter.py b/packages/markitdown-ocr/tests/test_xlsx_converter.py index 113960dddb..2deb42c2cf 100644 --- a/packages/markitdown-ocr/tests/test_xlsx_converter.py +++ b/packages/markitdown-ocr/tests/test_xlsx_converter.py @@ -95,24 +95,24 @@ def test_xlsx_image_middle(svc: MockOCRService) -> None: "## Revenue\n" "| Q1 Report | Unnamed: 1 |\n" "| --- | --- |\n" - "| NaN | NaN |\n" + "| | |\n" "| Revenue | $50,000 |\n" - "| NaN | NaN |\n" - "| NaN | NaN |\n" - "| NaN | NaN |\n" - "| NaN | NaN |\n" + "| | |\n" + "| | |\n" + "| | |\n" + "| | |\n" "| Profit Margin | 40% |\n\n" "### Images in this sheet:\n\n" f"{_OCR_BLOCK}\n\n" "## Expenses\n" "| Expense Breakdown | Unnamed: 1 |\n" "| --- | --- |\n" - "| NaN | NaN |\n" + "| | |\n" "| Expenses | $30,000 |\n" - "| NaN | NaN |\n" - "| NaN | NaN |\n" - "| NaN | NaN |\n" - "| NaN | NaN |\n" + "| | |\n" + "| | |\n" + "| | |\n" + "| | |\n" "| Savings | $5,000 |\n\n" "### Images in this sheet:\n\n" f"{_OCR_BLOCK}" @@ -133,12 +133,12 @@ def test_xlsx_image_end(svc: MockOCRService) -> None: "| Total Revenue | $500,000 |\n" "| Total Expenses | $300,000 |\n" "| Net Profit | $200,000 |\n" - "| NaN | NaN |\n" - "| NaN | NaN |\n" - "| NaN | NaN |\n" - "| NaN | NaN |\n" - "| NaN | NaN |\n" - "| Signature: | NaN |\n\n" + "| | |\n" + "| | |\n" + "| | |\n" + "| | |\n" + "| | |\n" + "| Signature: | |\n\n" "### Images in this sheet:\n\n" f"{_OCR_BLOCK}\n\n" "## Budget\n" @@ -147,12 +147,12 @@ def test_xlsx_image_end(svc: MockOCRService) -> None: "| Marketing | $100,000 |\n" "| R&D | $150,000 |\n" "| Operations | $50,000 |\n" - "| NaN | NaN |\n" - "| NaN | NaN |\n" - "| NaN | NaN |\n" - "| NaN | NaN |\n" - "| NaN | NaN |\n" - "| Approved: | NaN |\n\n" + "| | |\n" + "| | |\n" + "| | |\n" + "| | |\n" + "| | |\n" + "| Approved: | |\n\n" "### Images in this sheet:\n\n" f"{_OCR_BLOCK}" ) @@ -170,10 +170,10 @@ def test_xlsx_multiple_images(svc: MockOCRService) -> None: "| Dashboard |\n" "| --- |\n" "| Status: Active |\n" - "| NaN |\n" - "| NaN |\n" - "| NaN |\n" - "| NaN |\n" + "| |\n" + "| |\n" + "| |\n" + "| |\n" "| Performance Summary |\n\n" "### Images in this sheet:\n\n" f"{_OCR_BLOCK}\n\n" @@ -204,11 +204,11 @@ def test_xlsx_complex_layout(svc: MockOCRService) -> None: "## Complex Report\n" "| Annual Report 2024 | Unnamed: 1 |\n" "| --- | --- |\n" - "| NaN | NaN |\n" + "| | |\n" "| Month | Sales |\n" "| Jan | 1000 |\n" "| Feb | 1200 |\n" - "| NaN | NaN |\n" + "| | |\n" "| Total | 2200 |\n\n" "### Images in this sheet:\n\n" f"{_OCR_BLOCK}\n\n" @@ -216,7 +216,7 @@ def test_xlsx_complex_layout(svc: MockOCRService) -> None: "## Customers\n" "| Customer Metrics | Unnamed: 1 |\n" "| --- | --- |\n" - "| NaN | NaN |\n" + "| | |\n" "| New Customers | 250 |\n" "| Retention Rate | 92% |\n\n" "### Images in this sheet:\n\n" @@ -224,7 +224,7 @@ def test_xlsx_complex_layout(svc: MockOCRService) -> None: "## Regions\n" "| Regional Breakdown | Unnamed: 1 |\n" "| --- | --- |\n" - "| NaN | NaN |\n" + "| | |\n" "| Region | Revenue |\n" "| North | $800K |\n" "| South | $600K |\n\n" diff --git a/packages/markitdown-ocr/tests/test_xlsx_inheritance.py b/packages/markitdown-ocr/tests/test_xlsx_inheritance.py index 98a104b84d..f2566e8552 100644 --- a/packages/markitdown-ocr/tests/test_xlsx_inheritance.py +++ b/packages/markitdown-ocr/tests/test_xlsx_inheritance.py @@ -288,10 +288,35 @@ def fixed_read(*args: Any, **kwargs: Any): assert "Future shared fix" in result and "" not in result assert result.count("[Image OCR]") == 2 repair.assert_called_once() - assert calls == [{"sheet_name": None, "engine": "openpyxl"}] * 2 + assert calls == [{"sheet_name": None, "engine": "openpyxl", "dtype": object}] * 2 service.extract_text.assert_called_once() +def test_blank_cell_stays_blank_and_keeps_its_column_native() -> None: + workbook = openpyxl.Workbook() + sheet = workbook.active + sheet.title = "Cells" + sheet.append(["Units", "Shipped"]) + sheet.append([12, True]) + sheet.append([None, None]) + sheet.append([7, False]) + sheet.add_image(SheetImage(io.BytesIO(_RED)), "D1") + stream = io.BytesIO() + workbook.save(stream) + workbook.close() + + assert _convert(XlsxConverterWithOCR(_service()), stream.getvalue()) == ( + "## Cells\n" + "| Units | Shipped |\n" + "| --- | --- |\n" + "| 12 | True |\n" + "| | |\n" + "| 7 | False |\n\n" + "### Images in this sheet:\n\n" + "*[Image OCR] \nrecognized \n[End OCR]*" + ) + + def test_reported_ocr_error_warns_once_and_keeps_native_output() -> None: service = Mock( extract_text=Mock( From da24555f17fa2f42992ef5a2a9ddf67b5bbdcfba Mon Sep 17 00:00:00 2001 From: L4XB Date: Thu, 17 Sep 2026 18:52:18 +0200 Subject: [PATCH 3/4] fix(xls): read .xls sheets as object too `.xls` kept pandas' own type inference on the grounds that xlrd stores every number as a double. But pandas turns a whole-number double back into an int, so a blank cell still re-typed its column (`12.0`, `1.0`). It also turned a date column into `datetime64`, and the shared date formatter raised on its `NaT`: "NaTType does not support time". An `.xls` file with a blank date cell failed to convert at all. Read `.xls` with `dtype=object`, like `.xlsx`. `test.xls` renders the same as before. --- .../markitdown/converters/_xlsx_converter.py | 6 ++-- .../tests/test_files/test_blank_cells.xls | Bin 0 -> 5632 bytes .../markitdown/tests/test_xlsx_blank_cells.py | 26 ++++++++++++++++++ 3 files changed, 29 insertions(+), 3 deletions(-) create mode 100644 packages/markitdown/tests/test_files/test_blank_cells.xls diff --git a/packages/markitdown/src/markitdown/converters/_xlsx_converter.py b/packages/markitdown/src/markitdown/converters/_xlsx_converter.py index 90b9f9939a..d188fe2381 100644 --- a/packages/markitdown/src/markitdown/converters/_xlsx_converter.py +++ b/packages/markitdown/src/markitdown/converters/_xlsx_converter.py @@ -271,9 +271,9 @@ def convert( _xls_dependency_exc_info[2] ) - # xlrd stores every number as a double, so there is no integer to keep - # here and the sheets are read with pandas' own inference. - sheets = pd.read_excel(file_stream, sheet_name=None, engine="xlrd") + sheets = pd.read_excel( + file_stream, sheet_name=None, engine="xlrd", dtype=object + ) md_content = "" for s in sheets: md_content += f"## {s}\n" diff --git a/packages/markitdown/tests/test_files/test_blank_cells.xls b/packages/markitdown/tests/test_files/test_blank_cells.xls new file mode 100644 index 0000000000000000000000000000000000000000..dff9e75605f6c79b33e88314169ab7e6da55e00d GIT binary patch literal 5632 zcmeHLO-NKx6#m|OGdd&wy-|Y>w}R8YGXfsuI$7+@ju{CHOMCSPhP^`3`` z2fkOtz(>9(FpT!0gcDgu*sn447!pGxv4Pv<6l1eS6;ga!ahc-#s`*@TvvRH}ZbS%k z#&JwX+iUmG7HFy4tWsGqv4O`rW>z^3sK9ILq%K{T#ZIXo`;z@VgBwK<_7!3htaPiB zaFRWPVP6UEpa3{VsE@_!TUx@t3fu)(#UECrt!-~CHivR9k4bz_MSyRT9)Xt;kH5Zo zFc1NLh))4G|M{%TO8fw;yAnUh5@+*om_A?>sZ+Y$#<|(L$&e$Vv(Irx8WvCuVaV$T zx)S{(XSk9&`}&86owx8h;GQd`%g&V|-EW1&ag=D^)?HVM02c0jeHZKP zh@Mo=C^-VM39BV6q;OeAaB#5*Gzb^h>WCeRuZRw|Xunc%IYVVoH60a8or7blb5*4^ z^s`%GCb~7Fzm5IzFNdFm$3I24vp>~3f4+R`eRKj<`~{5Dhdg7bGwO_6HwB(X)2+c1 zM^3E|irw=n^P~!x+*>*W{qf`^ORy+gkYFozDT9>E-=$1aV%4QAQcO6+mgxAH9|hXF zXcj2X#xAn_W`LAy>jfh-XeET5(_&A~dP{AHt5C3>?7_pAkfG01389oLkdU+XTOn&Wx<3Tp<0u yP;`&ddlCca)i?%n8NN)1|FwSmmpGOQS0mg%VXJB6DqFMv%=zc)WEH)?^8X7oC+J83 literal 0 HcmV?d00001 diff --git a/packages/markitdown/tests/test_xlsx_blank_cells.py b/packages/markitdown/tests/test_xlsx_blank_cells.py index 7afdc30924..f2c492e9f6 100644 --- a/packages/markitdown/tests/test_xlsx_blank_cells.py +++ b/packages/markitdown/tests/test_xlsx_blank_cells.py @@ -7,6 +7,7 @@ import datetime import io +from pathlib import Path import pytest @@ -117,6 +118,31 @@ def test_a_date_cell_carries_no_time_and_a_blank_one_is_empty() -> None: ) +def test_an_xls_blank_cell_is_read_the_same_way() -> None: + """`.xls` is read with the same options as `.xlsx`. + + With pandas' own inference, a blank turned a date column into + ``datetime64``, and the date formatter cannot render its ``NaT``. + """ + pytest.importorskip("xlrd") + path = Path(__file__).parent / "test_files" / "test_blank_cells.xls" + with path.open("rb") as stream: + markdown = ( + MarkItDown(enable_plugins=False) + .convert_stream(stream, stream_info=StreamInfo(extension=".xls")) + .markdown + ) + + assert markdown == ( + "## Sheet1\n" + "| Units | Shipped | Due | Id |\n" + "| --- | --- | --- | --- |\n" + "| 12 | True | 2026-01-05 | 1 |\n" + "| | | | 2 |\n" + "| 7 | False | 2026-02-09 14:30:00 | 3 |" + ) + + def test_a_sheet_without_blanks_is_unchanged() -> None: """Guard: the common case must render exactly as it did before.""" markdown = _convert( From 06ded9e881b7a125b4e9783e9937034586736c89 Mon Sep 17 00:00:00 2001 From: L4XB Date: Thu, 17 Sep 2026 20:21:15 +0200 Subject: [PATCH 4/4] fix(xlsx): keep a fractional number at full precision With `index=False`, pandas renders a float in an object column through its own `float_format` and never reaches `formatters`, so `1e-07` came out as `0.0` once the sheets were read with `dtype=object`. On main, which reads with inference, the same cell rendered as `1.000000e-07`. The table now passes `float_format`, which formats with the 15 significant digits Excel shows: `1e-07`, `3.14159265358979`, and `0.1 + 0.2` as `0.3`. Both the .xlsx and the .xls path render through `sheet_to_html`, so both are covered. --- .../markitdown/converters/_xlsx_converter.py | 13 +++++++++++-- .../markitdown/tests/test_xlsx_blank_cells.py | 18 ++++++++++++++++++ 2 files changed, 29 insertions(+), 2 deletions(-) diff --git a/packages/markitdown/src/markitdown/converters/_xlsx_converter.py b/packages/markitdown/src/markitdown/converters/_xlsx_converter.py index d188fe2381..49ce5c0b1e 100644 --- a/packages/markitdown/src/markitdown/converters/_xlsx_converter.py +++ b/packages/markitdown/src/markitdown/converters/_xlsx_converter.py @@ -82,19 +82,28 @@ def _format_cell(value: Any) -> str: return str(value) +def _format_float(value: float) -> str: + """Render a fractional number with the 15 significant digits Excel shows.""" + return format(value, ".15g") + + def sheet_to_html(sheet: Any) -> str: """Render one sheet as an HTML table. ``na_rep=""`` keeps an empty cell empty: the default writes the string ``NaN`` into it, which reads as a value rather than as a blank. The formatters are what keep the remaining cells rendered as themselves -- - pandas applies ``na_rep`` to the blanks and the formatter to everything - else, without re-inferring a dtype for the column. + pandas applies ``na_rep`` to the blanks, ``float_format`` to fractional + numbers, and the formatter to everything else, without re-inferring a + dtype for the column. A float needs ``float_format`` because with + ``index=False`` pandas renders it with its own display precision and + never reaches ``formatters``. """ return sheet.to_html( index=False, na_rep="", formatters=[_format_cell] * len(sheet.columns), + float_format=_format_float, ) diff --git a/packages/markitdown/tests/test_xlsx_blank_cells.py b/packages/markitdown/tests/test_xlsx_blank_cells.py index f2c492e9f6..3440dff6a0 100644 --- a/packages/markitdown/tests/test_xlsx_blank_cells.py +++ b/packages/markitdown/tests/test_xlsx_blank_cells.py @@ -77,6 +77,24 @@ def test_a_blank_cell_does_not_turn_the_column_into_floats() -> None: ) +def test_a_fractional_number_keeps_its_value() -> None: + markdown = _convert( + [ + ["Small", "Pi"], + [1e-7, 3.14159265358979], + [None, 2], + ] + ) + + assert markdown == ( + "## Sheet1\n" + "| Small | Pi |\n" + "| --- | --- |\n" + "| 1e-07 | 3.14159265358979 |\n" + "| | 2 |" + ) + + def test_a_blank_cell_does_not_turn_booleans_into_numbers() -> None: markdown = _convert( [