From 383fe6d3013cde316c42b63d3fa92ed15547b41c Mon Sep 17 00:00:00 2001 From: Manohar Paturi <186662190+ManoharPaturi@users.noreply.github.com> Date: Fri, 18 Sep 2026 17:52:02 +0530 Subject: [PATCH 1/3] fix(xlsx): render float values faithfully instead of 6-digit scientific pandas to_html's default float format silently drops entered digits (123456789.123 -> 1.234568e+08). Pass float_format so the shortest round-tripping representation of the stored value is used. Applies to both the .xlsx and .xls paths. Signed-off-by: Manohar Paturi <186662190+ManoharPaturi@users.noreply.github.com> --- .../markitdown/converters/_xlsx_converter.py | 18 +++++++++++++-- .../tests/test_xlsx_float_precision.py | 22 +++++++++++++++++++ 2 files changed, 38 insertions(+), 2 deletions(-) create mode 100644 packages/markitdown/tests/test_xlsx_float_precision.py diff --git a/packages/markitdown/src/markitdown/converters/_xlsx_converter.py b/packages/markitdown/src/markitdown/converters/_xlsx_converter.py index 9f794a3b77..d0945c050b 100644 --- a/packages/markitdown/src/markitdown/converters/_xlsx_converter.py +++ b/packages/markitdown/src/markitdown/converters/_xlsx_converter.py @@ -150,7 +150,14 @@ def convert( for s in sheets: md_content += f"## {s}\n" - html_content = sheets[s].to_html(index=False) + html_content = sheets[s].to_html( + index=False, + # pandas defaults large/small floats to 6-significant-digit + # scientific notation, silently dropping digits the user + # entered (123456789.123 -> 1.234568e+08). repr gives the + # shortest string that round-trips the exact stored value. + float_format=lambda value: repr(float(value)), + ) md_content += ( self._html_converter.convert_string( html_content, **kwargs @@ -241,7 +248,14 @@ def convert( md_content = "" for s in sheets: md_content += f"## {s}\n" - html_content = sheets[s].to_html(index=False) + html_content = sheets[s].to_html( + index=False, + # pandas defaults large/small floats to 6-significant-digit + # scientific notation, silently dropping digits the user + # entered (123456789.123 -> 1.234568e+08). repr gives the + # shortest string that round-trips the exact stored value. + float_format=lambda value: repr(float(value)), + ) md_content += ( self._html_converter.convert_string( html_content, **kwargs diff --git a/packages/markitdown/tests/test_xlsx_float_precision.py b/packages/markitdown/tests/test_xlsx_float_precision.py new file mode 100644 index 0000000000..b8ebc80eb6 --- /dev/null +++ b/packages/markitdown/tests/test_xlsx_float_precision.py @@ -0,0 +1,22 @@ +"""Spreadsheet float values must survive conversion without pandas' 6-digit scientific notation.""" +import io + +from openpyxl import Workbook + +from markitdown import MarkItDown, StreamInfo + + +def test_xlsx_float_precision_preserved(): + wb = Workbook() + ws = wb.active + ws["A1"] = "money" + ws["A2"] = 123456789.123 + ws["A3"] = 0.1 + ws["A4"] = 1e-10 + buf = io.BytesIO() + wb.save(buf) + buf.seek(0) + result = MarkItDown().convert_stream(buf, stream_info=StreamInfo(extension=".xlsx")) + assert "123456789.123" in result.markdown + assert "1.234568e+08" not in result.markdown + assert "| 0.1 |" in result.markdown From b8880255cc096f89ba882076b8e36c07c750e833 Mon Sep 17 00:00:00 2001 From: cyurekli Date: Sun, 27 Sep 2026 00:43:25 +0200 Subject: [PATCH 2/3] Cover XLS and small floating-point values in precision regressions --- .../tests/test_files/float_precision.txt | 3 +++ .../tests/test_files/float_precision.xls | Bin 0 -> 5632 bytes .../tests/test_xlsx_float_precision.py | 18 ++++++++++++++++++ 3 files changed, 21 insertions(+) create mode 100644 packages/markitdown/tests/test_files/float_precision.txt create mode 100644 packages/markitdown/tests/test_files/float_precision.xls diff --git a/packages/markitdown/tests/test_files/float_precision.txt b/packages/markitdown/tests/test_files/float_precision.txt new file mode 100644 index 0000000000..8c5e46edf9 --- /dev/null +++ b/packages/markitdown/tests/test_files/float_precision.txt @@ -0,0 +1,3 @@ +Synthetic BIFF8 workbook generated with xlwt 1.3.0. +One sheet named Precision; A1 = value, A2 = 123456789.123, A3 = 0.1, A4 = 1e-10. +No real user data. The committed fixture avoids adding an XLS writer to test or runtime dependencies. diff --git a/packages/markitdown/tests/test_files/float_precision.xls b/packages/markitdown/tests/test_files/float_precision.xls new file mode 100644 index 0000000000000000000000000000000000000000..58e98b3fa6c7aa4f9101d193fe1c78a66feddbe9 GIT binary patch literal 5632 zcmeHLO=wd=5dQWhX_L~Xc`+5NP(s00+g?NuUeebJ+LI-UpopMpn?qsTnPqpSK z2%Z!@6ngPe^-?QU(3=OLqKBd)CiU2psNkXXIWwqxsLhLqwM62uj5lO&8y5>-_AsPGNK?}_DO;R7PMCOm;C zO6mv}tl^$r)EB6y!$zm*!o(FGN}IPL*^V|mCmZ|Hb)M{`4Pda4NoD^|MTfr`)1bT! zrE}>*DqBK~5;%>ob{zP^17La@*Yf(}pb!VXu}>xk|3=Eodif2*4@|&5E)%>66LUUp z5kj(%9?hK0R6ZA*S+u6mNrN9FAN)+BPxYx=H<=z;ZmDQR z^RDn-xU(O4cY>fnJDS8Ue@--rR;+?C0|Cs$77nJrEPOO>mYJCSeBt2E{GCCzVn2$o z5u$!B8ciZI(Muw>;Cy*4xF438n18bP=EM1SiGCxBS?&k}%+|<>$9>bKMnEH=5zq)| z1T+E~0gZr0pgsujnU@czd}ie{KYe4!9RObj@V(9|{kYWhen> LtCK2rf8_rM3gXE| literal 0 HcmV?d00001 diff --git a/packages/markitdown/tests/test_xlsx_float_precision.py b/packages/markitdown/tests/test_xlsx_float_precision.py index b8ebc80eb6..b4b1980d59 100644 --- a/packages/markitdown/tests/test_xlsx_float_precision.py +++ b/packages/markitdown/tests/test_xlsx_float_precision.py @@ -1,5 +1,6 @@ """Spreadsheet float values must survive conversion without pandas' 6-digit scientific notation.""" import io +from pathlib import Path from openpyxl import Workbook @@ -20,3 +21,20 @@ def test_xlsx_float_precision_preserved(): assert "123456789.123" in result.markdown assert "1.234568e+08" not in result.markdown assert "| 0.1 |" in result.markdown + assert "| 1e-10 |" in result.markdown + + +def test_xls_float_precision_preserved(): + fixture = Path(__file__).parent / "test_files" / "float_precision.xls" + with fixture.open("rb") as stream: + result = MarkItDown().convert_stream( + stream, stream_info=StreamInfo(extension=".xls") + ) + cells = [ + line.strip("| ") + for line in result.markdown.splitlines() + if line.startswith("|") + ] + assert "123456789.123" in cells + assert "0.1" in cells + assert "1e-10" in cells From d362f08177a99fecdbf530bbb0f40b04285044b6 Mon Sep 17 00:00:00 2001 From: Manohar Paturi Date: Thu, 1 Oct 2026 23:28:50 +0530 Subject: [PATCH 3/3] test(xlsx): cover numpy scalars and fix call-site formatting Add a regression proving the float() cast inside float_format is load-bearing: pandas hands the formatter numpy scalars, and a bare repr(value) renders 'np.float64(0.1)' in the table on numpy >= 2. Also reformat the xls to_html call site; the committed indentation failed the repo's black 23.7.0 pre-commit hook. --- .../src/markitdown/converters/_xlsx_converter.py | 14 +++++++------- .../markitdown/tests/test_xlsx_float_precision.py | 15 +++++++++++++++ 2 files changed, 22 insertions(+), 7 deletions(-) diff --git a/packages/markitdown/src/markitdown/converters/_xlsx_converter.py b/packages/markitdown/src/markitdown/converters/_xlsx_converter.py index d0945c050b..706645517f 100644 --- a/packages/markitdown/src/markitdown/converters/_xlsx_converter.py +++ b/packages/markitdown/src/markitdown/converters/_xlsx_converter.py @@ -249,13 +249,13 @@ def convert( for s in sheets: md_content += f"## {s}\n" html_content = sheets[s].to_html( - index=False, - # pandas defaults large/small floats to 6-significant-digit - # scientific notation, silently dropping digits the user - # entered (123456789.123 -> 1.234568e+08). repr gives the - # shortest string that round-trips the exact stored value. - float_format=lambda value: repr(float(value)), - ) + index=False, + # pandas defaults large/small floats to 6-significant-digit + # scientific notation, silently dropping digits the user + # entered (123456789.123 -> 1.234568e+08). repr gives the + # shortest string that round-trips the exact stored value. + float_format=lambda value: repr(float(value)), + ) md_content += ( self._html_converter.convert_string( html_content, **kwargs diff --git a/packages/markitdown/tests/test_xlsx_float_precision.py b/packages/markitdown/tests/test_xlsx_float_precision.py index b4b1980d59..7ad8dcfc44 100644 --- a/packages/markitdown/tests/test_xlsx_float_precision.py +++ b/packages/markitdown/tests/test_xlsx_float_precision.py @@ -2,6 +2,8 @@ import io from pathlib import Path +import numpy as np +import pandas as pd from openpyxl import Workbook from markitdown import MarkItDown, StreamInfo @@ -38,3 +40,16 @@ def test_xls_float_precision_preserved(): assert "123456789.123" in cells assert "0.1" in cells assert "1e-10" in cells + + +def test_float_format_handles_numpy_scalars(): + # pandas hands float_format numpy scalars, and with numpy >= 2 a bare + # repr(value) would render "np.float64(0.1)" in the table. The float() + # cast inside the converter's lambda is what prevents that. + df = pd.DataFrame({"money": [np.float64(0.1)]}) + buf = io.BytesIO() + df.to_excel(buf, index=False) + buf.seek(0) + result = MarkItDown().convert_stream(buf, stream_info=StreamInfo(extension=".xlsx")) + assert "np.float64" not in result.markdown + assert "| 0.1 |" in result.markdown