Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 16 additions & 2 deletions packages/markitdown/src/markitdown/converters/_xlsx_converter.py
Original file line number Diff line number Diff line change
Expand Up @@ -150,7 +150,14 @@ def convert(

for s in sheets:
md_content += f"## {s}\n"
html_content = sheets[s].to_html(index=False)
html_content = sheets[s].to_html(
index=False,
# pandas defaults large/small floats to 6-significant-digit
# scientific notation, silently dropping digits the user
# entered (123456789.123 -> 1.234568e+08). repr gives the
# shortest string that round-trips the exact stored value.
float_format=lambda value: repr(float(value)),
)
md_content += (
self._html_converter.convert_string(
html_content, **kwargs
Expand Down Expand Up @@ -241,7 +248,14 @@ def convert(
md_content = ""
for s in sheets:
md_content += f"## {s}\n"
html_content = sheets[s].to_html(index=False)
html_content = sheets[s].to_html(
index=False,
# pandas defaults large/small floats to 6-significant-digit
# scientific notation, silently dropping digits the user
# entered (123456789.123 -> 1.234568e+08). repr gives the
# shortest string that round-trips the exact stored value.
float_format=lambda value: repr(float(value)),
)
md_content += (
self._html_converter.convert_string(
html_content, **kwargs
Expand Down
3 changes: 3 additions & 0 deletions packages/markitdown/tests/test_files/float_precision.txt
Original file line number Diff line number Diff line change
@@ -0,0 +1,3 @@
Synthetic BIFF8 workbook generated with xlwt 1.3.0.
One sheet named Precision; A1 = value, A2 = 123456789.123, A3 = 0.1, A4 = 1e-10.
No real user data. The committed fixture avoids adding an XLS writer to test or runtime dependencies.
Binary file not shown.
55 changes: 55 additions & 0 deletions packages/markitdown/tests/test_xlsx_float_precision.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,55 @@
"""Spreadsheet float values must survive conversion without pandas' 6-digit scientific notation."""
import io
from pathlib import Path

import numpy as np
import pandas as pd
from openpyxl import Workbook

from markitdown import MarkItDown, StreamInfo


def test_xlsx_float_precision_preserved():
wb = Workbook()
ws = wb.active
ws["A1"] = "money"
ws["A2"] = 123456789.123
ws["A3"] = 0.1
ws["A4"] = 1e-10
buf = io.BytesIO()
wb.save(buf)
buf.seek(0)
result = MarkItDown().convert_stream(buf, stream_info=StreamInfo(extension=".xlsx"))
assert "123456789.123" in result.markdown
assert "1.234568e+08" not in result.markdown
assert "| 0.1 |" in result.markdown
assert "| 1e-10 |" in result.markdown


def test_xls_float_precision_preserved():
fixture = Path(__file__).parent / "test_files" / "float_precision.xls"
with fixture.open("rb") as stream:
result = MarkItDown().convert_stream(
stream, stream_info=StreamInfo(extension=".xls")
)
cells = [
line.strip("| ")
for line in result.markdown.splitlines()
if line.startswith("|")
]
assert "123456789.123" in cells
assert "0.1" in cells
assert "1e-10" in cells


def test_float_format_handles_numpy_scalars():
# pandas hands float_format numpy scalars, and with numpy >= 2 a bare
# repr(value) would render "np.float64(0.1)" in the table. The float()
# cast inside the converter's lambda is what prevents that.
df = pd.DataFrame({"money": [np.float64(0.1)]})
buf = io.BytesIO()
df.to_excel(buf, index=False)
buf.seek(0)
result = MarkItDown().convert_stream(buf, stream_info=StreamInfo(extension=".xlsx"))
assert "np.float64" not in result.markdown
assert "| 0.1 |" in result.markdown