diff --git a/packages/markitdown/src/markitdown/converters/_xlsx_converter.py b/packages/markitdown/src/markitdown/converters/_xlsx_converter.py index 9f794a3b77..e7157fb4bb 100644 --- a/packages/markitdown/src/markitdown/converters/_xlsx_converter.py +++ b/packages/markitdown/src/markitdown/converters/_xlsx_converter.py @@ -149,23 +149,35 @@ def convert( images = _XlsxImages(workbook_stream) for s in sheets: - md_content += f"## {s}\n" - html_content = sheets[s].to_html(index=False) - md_content += ( - self._html_converter.convert_string( - html_content, **kwargs - ).markdown.strip() - + "\n\n" + image_content = ( + images.to_html(s, self._image_to_html, kwargs) + if images is not None + else None ) - if images is not None: - image_content = images.to_html(s, self._image_to_html, kwargs) - if image_content: - md_content += ( - self._html_converter.convert_string( - image_content, **kwargs - ).markdown.strip() - + "\n\n" - ) + # A completely empty sheet has no columns. pandas then emits a + # column-less HTML table that markdownify turns into broken syntax + # ("|\n| |"). Skip the table, and the sheet when it has no images + # either. A header-only sheet still has columns and already renders + # as a well-formed empty table. + has_table = not sheets[s].columns.empty + if not has_table and not image_content: + continue + md_content += f"## {s}\n" + if has_table: + html_content = sheets[s].to_html(index=False) + md_content += ( + self._html_converter.convert_string( + html_content, **kwargs + ).markdown.strip() + + "\n\n" + ) + if image_content: + md_content += ( + self._html_converter.convert_string( + image_content, **kwargs + ).markdown.strip() + + "\n\n" + ) return DocumentConverterResult(markdown=md_content.strip()) @@ -240,6 +252,10 @@ def convert( sheets = pd.read_excel(file_stream, sheet_name=None, engine="xlrd") md_content = "" for s in sheets: + # Same empty-sheet guard as XlsxConverter: no columns means pandas + # would emit a column-less table that is not valid Markdown. + if sheets[s].columns.empty: + continue md_content += f"## {s}\n" html_content = sheets[s].to_html(index=False) md_content += ( diff --git a/packages/markitdown/tests/test_xlsx_empty_sheet.py b/packages/markitdown/tests/test_xlsx_empty_sheet.py new file mode 100644 index 0000000000..70df82c56d --- /dev/null +++ b/packages/markitdown/tests/test_xlsx_empty_sheet.py @@ -0,0 +1,97 @@ +import io + +from openpyxl import Workbook + +from markitdown import MarkItDown, StreamInfo + + +def _xlsx_bytes(workbook: Workbook) -> bytes: + buf = io.BytesIO() + workbook.save(buf) + return buf.getvalue() + + +def test_completely_empty_sheet_is_skipped() -> None: + """A sheet with no columns must not emit malformed table syntax.""" + workbook = Workbook() + workbook.active.title = "HasData" + workbook.active["A1"] = "col" + workbook.active["A2"] = "v" + workbook.create_sheet("CompletelyEmpty") + + result = MarkItDown().convert_stream( + io.BytesIO(_xlsx_bytes(workbook)), + stream_info=StreamInfo(extension=".xlsx"), + ) + + assert "## HasData" in result.markdown + assert "| col |" in result.markdown + assert "| v |" in result.markdown + assert "CompletelyEmpty" not in result.markdown + assert "|\n| |" not in result.markdown + + +def test_header_only_sheet_keeps_empty_table() -> None: + """A sheet with a header row but no data already produces a valid table.""" + workbook = Workbook() + workbook.active.title = "HeaderOnly" + workbook.active["A1"] = "col" + + result = MarkItDown().convert_stream( + io.BytesIO(_xlsx_bytes(workbook)), + stream_info=StreamInfo(extension=".xlsx"), + ) + + assert "## HeaderOnly" in result.markdown + assert "| col |" in result.markdown + assert "| --- |" in result.markdown + + +def test_single_completely_empty_workbook_emits_nothing() -> None: + workbook = Workbook() + workbook.active.title = "Empty" + + result = MarkItDown().convert_stream( + io.BytesIO(_xlsx_bytes(workbook)), + stream_info=StreamInfo(extension=".xlsx"), + ) + + assert result.markdown.strip() == "" + assert "|" not in result.markdown + + +def test_empty_sheet_with_image_keeps_the_image() -> None: + """Skipping an empty sheet's table must not drop images anchored on it.""" + from typing import Any, BinaryIO, Optional + + from openpyxl.drawing.image import Image as SheetImage + from PIL import Image + + from markitdown.converters import XlsxConverter + + class ImageConverter(XlsxConverter): + def _image_to_html( + self, image_stream: BinaryIO, stream_info: StreamInfo, **kwargs: Any + ) -> Optional[str]: + return "

sheet image

" + + png = io.BytesIO() + Image.new("RGB", (2, 2), "red").save(png, "PNG") + workbook = Workbook() + workbook.active.title = "HasData" + workbook.active["A1"] = "col" + workbook.create_sheet("OnlyAnImage").add_image( + SheetImage(io.BytesIO(png.getvalue())), "B2" + ) + workbook.create_sheet("CompletelyEmpty") + + markdown = ( + ImageConverter() + .convert(io.BytesIO(_xlsx_bytes(workbook)), StreamInfo(extension=".xlsx")) + .markdown + ) + + assert "## OnlyAnImage" in markdown + assert "sheet image" in markdown.split("## OnlyAnImage", 1)[1] + assert "## CompletelyEmpty" not in markdown + assert "| |" not in markdown