diff --git a/packages/markitdown/src/markitdown/converters/_xlsx_converter.py b/packages/markitdown/src/markitdown/converters/_xlsx_converter.py index 9f794a3b77..706645517f 100644 --- a/packages/markitdown/src/markitdown/converters/_xlsx_converter.py +++ b/packages/markitdown/src/markitdown/converters/_xlsx_converter.py @@ -150,7 +150,14 @@ def convert( for s in sheets: md_content += f"## {s}\n" - html_content = sheets[s].to_html(index=False) + html_content = sheets[s].to_html( + index=False, + # pandas defaults large/small floats to 6-significant-digit + # scientific notation, silently dropping digits the user + # entered (123456789.123 -> 1.234568e+08). repr gives the + # shortest string that round-trips the exact stored value. + float_format=lambda value: repr(float(value)), + ) md_content += ( self._html_converter.convert_string( html_content, **kwargs @@ -241,7 +248,14 @@ def convert( md_content = "" for s in sheets: md_content += f"## {s}\n" - html_content = sheets[s].to_html(index=False) + html_content = sheets[s].to_html( + index=False, + # pandas defaults large/small floats to 6-significant-digit + # scientific notation, silently dropping digits the user + # entered (123456789.123 -> 1.234568e+08). repr gives the + # shortest string that round-trips the exact stored value. + float_format=lambda value: repr(float(value)), + ) md_content += ( self._html_converter.convert_string( html_content, **kwargs diff --git a/packages/markitdown/tests/test_files/float_precision.txt b/packages/markitdown/tests/test_files/float_precision.txt new file mode 100644 index 0000000000..8c5e46edf9 --- /dev/null +++ b/packages/markitdown/tests/test_files/float_precision.txt @@ -0,0 +1,3 @@ +Synthetic BIFF8 workbook generated with xlwt 1.3.0. +One sheet named Precision; A1 = value, A2 = 123456789.123, A3 = 0.1, A4 = 1e-10. +No real user data. The committed fixture avoids adding an XLS writer to test or runtime dependencies. diff --git a/packages/markitdown/tests/test_files/float_precision.xls b/packages/markitdown/tests/test_files/float_precision.xls new file mode 100644 index 0000000000..58e98b3fa6 Binary files /dev/null and b/packages/markitdown/tests/test_files/float_precision.xls differ diff --git a/packages/markitdown/tests/test_xlsx_float_precision.py b/packages/markitdown/tests/test_xlsx_float_precision.py new file mode 100644 index 0000000000..7ad8dcfc44 --- /dev/null +++ b/packages/markitdown/tests/test_xlsx_float_precision.py @@ -0,0 +1,55 @@ +"""Spreadsheet float values must survive conversion without pandas' 6-digit scientific notation.""" +import io +from pathlib import Path + +import numpy as np +import pandas as pd +from openpyxl import Workbook + +from markitdown import MarkItDown, StreamInfo + + +def test_xlsx_float_precision_preserved(): + wb = Workbook() + ws = wb.active + ws["A1"] = "money" + ws["A2"] = 123456789.123 + ws["A3"] = 0.1 + ws["A4"] = 1e-10 + buf = io.BytesIO() + wb.save(buf) + buf.seek(0) + result = MarkItDown().convert_stream(buf, stream_info=StreamInfo(extension=".xlsx")) + assert "123456789.123" in result.markdown + assert "1.234568e+08" not in result.markdown + assert "| 0.1 |" in result.markdown + assert "| 1e-10 |" in result.markdown + + +def test_xls_float_precision_preserved(): + fixture = Path(__file__).parent / "test_files" / "float_precision.xls" + with fixture.open("rb") as stream: + result = MarkItDown().convert_stream( + stream, stream_info=StreamInfo(extension=".xls") + ) + cells = [ + line.strip("| ") + for line in result.markdown.splitlines() + if line.startswith("|") + ] + assert "123456789.123" in cells + assert "0.1" in cells + assert "1e-10" in cells + + +def test_float_format_handles_numpy_scalars(): + # pandas hands float_format numpy scalars, and with numpy >= 2 a bare + # repr(value) would render "np.float64(0.1)" in the table. The float() + # cast inside the converter's lambda is what prevents that. + df = pd.DataFrame({"money": [np.float64(0.1)]}) + buf = io.BytesIO() + df.to_excel(buf, index=False) + buf.seek(0) + result = MarkItDown().convert_stream(buf, stream_info=StreamInfo(extension=".xlsx")) + assert "np.float64" not in result.markdown + assert "| 0.1 |" in result.markdown