From 4e128061f46511e035ebdaa9b0aba03796f71097 Mon Sep 17 00:00:00 2001 From: V S S L Deepak Janapa <125086435+deepakjanapa@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:08:11 +0530 Subject: [PATCH 1/4] feat: add DOCX document loader --- CHANGELOG.md | 4 ++ ragframework/document/__init__.py | 3 +- ragframework/document/loaders.py | 64 ++++++++++++++++++++++++++++- tests/test_document/test_loaders.py | 64 ++++++++++++++++++++++++++++- 4 files changed, 131 insertions(+), 4 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a9290d3..7ce6559 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Added + +- `DocxLoader` with support for per-paragraph and whole-file modes (closes #2) + ## [0.2.0] - 2026-09-19 ### Added diff --git a/ragframework/document/__init__.py b/ragframework/document/__init__.py index 56f931a..68f35b9 100644 --- a/ragframework/document/__init__.py +++ b/ragframework/document/__init__.py @@ -2,12 +2,13 @@ from ragframework.document.chunkers import FixedSizeChunker, RecursiveChunker, SentenceChunker -from .loaders import MarkdownLoader, PDFLoader, TextFileLoader +from .loaders import DocxLoader, MarkdownLoader, PDFLoader, TextFileLoader __all__ = [ "TextFileLoader", "MarkdownLoader", "PDFLoader", + "DocxLoader", "FixedSizeChunker", "RecursiveChunker", "SentenceChunker", diff --git a/ragframework/document/loaders.py b/ragframework/document/loaders.py index 9f3ee55..9c43df7 100644 --- a/ragframework/document/loaders.py +++ b/ragframework/document/loaders.py @@ -1,7 +1,7 @@ """Built-in document loaders. -These loaders handle plain text and Markdown files, PDF with no extra dependencies. -For DOCX, HTML, and other formats see the open issues in +These loaders handle plain text, Markdown, PDF, and DOCX files. +For HTML and other formats see the open issues in `.github/GOOD_FIRST_ISSUES.md`. """ @@ -13,6 +13,11 @@ from ragframework.base import Document, DocumentLoader from ragframework.exceptions import LoaderError +try: + from docx import Document as DocxDocument +except ImportError: + DocxDocument = None + def _make_id(source: str) -> str: return hashlib.md5(source.encode()).hexdigest()[:12] @@ -73,6 +78,61 @@ def load(self, source: str) -> list[Document]: ] +class DocxLoader(DocumentLoader): + """Load a DOCX file into one or more :class:`Document` objects.""" + + def __init__(self, split_paragraphs: bool = True) -> None: + self.split_paragraphs = split_paragraphs + + def load(self, source: str) -> list[Document]: + path = Path(source) + + if not path.exists(): + raise LoaderError(f"File not found: {source}") + + if not path.is_file(): + raise LoaderError(f"Not a file: {source}") + + if DocxDocument is None: + raise LoaderError( + "DOCX support requires 'ragframework[docx]'. " + "Install it with: pip install ragframework[docx]" + ) + + try: + docx = DocxDocument(str(path)) + except Exception as exc: + raise LoaderError(f"Could not read DOCX file {source}: {exc}") from exc + paragraphs = [paragraph.text for paragraph in docx.paragraphs if paragraph.text.strip()] + if self.split_paragraphs: + return [ + Document( + id=_make_id(f"{source}_paragraph{i}"), + content=paragraph, + metadata={ + "source": source, + "filename": path.name, + "format": "docx", + "paragraph_number": i, + }, + ) + for i, paragraph in enumerate(paragraphs, start=1) + ] + + return [ + Document( + id=_make_id(source), + content="\n\n".join(paragraphs), + metadata={ + "source": source, + "filename": path.name, + "format": "docx", + "split_paragraphs": False, + }, + ) + ] + + class PDFLoader(DocumentLoader): """Load a PDF file into one or more :class:`Document` objects. By default, each page is loaded as a separate document with metadata indicating the page number. diff --git a/tests/test_document/test_loaders.py b/tests/test_document/test_loaders.py index a30aebb..6c175e8 100644 --- a/tests/test_document/test_loaders.py +++ b/tests/test_document/test_loaders.py @@ -6,7 +6,12 @@ import pytest -from ragframework.document.loaders import MarkdownLoader, PDFLoader, TextFileLoader +from ragframework.document.loaders import ( + DocxLoader, + MarkdownLoader, + PDFLoader, + TextFileLoader, +) from ragframework.exceptions import LoaderError @@ -49,6 +54,63 @@ def test_missing_file_raises(self): loader.load("/no/such/file.md") +class TestDocxLoader: + def test_loads_paragraphs(self, tmp_path): + from docx import Document as DocxDocument + + docx_path = tmp_path / "doc.docx" + + doc = DocxDocument() + doc.add_paragraph("First paragraph") + doc.add_paragraph("") + doc.add_paragraph("Second paragraph") + doc.save(docx_path) + + loader = DocxLoader() + docs = loader.load(str(docx_path)) + + assert len(docs) == 2 + assert docs[0].content == "First paragraph" + assert docs[1].content == "Second paragraph" + assert docs[0].metadata["format"] == "docx" + assert docs[0].metadata["paragraph_number"] == 1 + assert docs[1].metadata["paragraph_number"] == 2 + + def test_loads_whole_file(self, tmp_path): + from docx import Document as DocxDocument + + docx_path = tmp_path / "doc.docx" + + doc = DocxDocument() + doc.add_paragraph("First paragraph") + doc.add_paragraph("Second paragraph") + doc.save(docx_path) + + loader = DocxLoader(split_paragraphs=False) + docs = loader.load(str(docx_path)) + + assert len(docs) == 1 + assert "First paragraph" in docs[0].content + assert "Second paragraph" in docs[0].content + assert docs[0].metadata["format"] == "docx" + assert docs[0].metadata["split_paragraphs"] is False + + def test_missing_file_raises(self): + loader = DocxLoader() + + with pytest.raises(LoaderError, match="File not found"): + loader.load("/nonexistent/path/file.docx") + + def test_invalid_docx_raises(self, tmp_path): + docx_path = tmp_path / "invalid.docx" + docx_path.write_text("This is not a valid DOCX file.", encoding="utf-8") + + loader = DocxLoader() + + with pytest.raises(LoaderError, match="Could not read DOCX file"): + loader.load(str(docx_path)) + + def test_pdf_loader_requires_pypdf(tmp_path, monkeypatch): pdf_path = tmp_path / "doc.pdf" pdf_path.write_bytes(b"%PDF-1.4\n%EOF\n") From c9e2f6b07d33cf988fb327dc48a1c98cff6643fe Mon Sep 17 00:00:00 2001 From: V S S L Deepak Janapa <125086435+deepakjanapa@users.noreply.github.com> Date: Thu, 24 Sep 2026 01:17:33 +0530 Subject: [PATCH 2/4] merge upstream main into add-docx-loader --- CHANGELOG.md | 1 + 1 file changed, 1 insertion(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 1238052..85069f9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -22,6 +22,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Fixed - Enforce LF line endings with `.gitattributes` across platforms while keeping PNG files binary (closes #48). +- `InMemoryRetriever.retrieve()` now returns an empty list for non-positive `top_k` values and raises `RetrieverError` for non-integer or boolean `top_k` values (closes #23). - `InMemoryRetriever` now raises `RetrieverError` for invalid vectors and dimension mismatches, validates complete batches before updating stored data, and treats empty batches as a no-op. Vector validation and normalization are shared with `FAISSRetriever` (closes #26). ### Added From 48b12d83a325c7dcdccffd5e3684b6ec09488aca Mon Sep 17 00:00:00 2001 From: V S S L Deepak Janapa <125086435+deepakjanapa@users.noreply.github.com> Date: Fri, 25 Sep 2026 23:12:47 +0530 Subject: [PATCH 3/4] fix: address maintainer feedback for DOCX loader --- .github/workflows/ci.yml | 2 +- CHANGELOG.md | 2 ++ ragframework/document/__init__.py | 9 ++++++++- tests/test_document/test_loaders.py | 15 +++++++++++++++ 4 files changed, 26 insertions(+), 2 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 7634e95..ffe6fbb 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -25,7 +25,7 @@ jobs: cache: pip - name: Install dependencies - run: pip install -e ".[dev,faiss,chromadb]" + run: pip install -e ".[dev,faiss,chromadb,docx]" - name: Lint with ruff run: ruff check ragframework/ diff --git a/CHANGELOG.md b/CHANGELOG.md index 85069f9..9af38c7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,6 +25,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - `InMemoryRetriever.retrieve()` now returns an empty list for non-positive `top_k` values and raises `RetrieverError` for non-integer or boolean `top_k` values (closes #23). - `InMemoryRetriever` now raises `RetrieverError` for invalid vectors and dimension mismatches, validates complete batches before updating stored data, and treats empty batches as a no-op. Vector validation and normalization are shared with `FAISSRetriever` (closes #26). +## [0.2.0] - 2026-09-19 + ### Added - Optional reranking stage with `Reranker`, `CrossEncoderReranker`, `NoOpReranker`, and configurable pre-rerank retrieval depth (closes #38) diff --git a/ragframework/document/__init__.py b/ragframework/document/__init__.py index 83e99c2..dec3766 100644 --- a/ragframework/document/__init__.py +++ b/ragframework/document/__init__.py @@ -1,3 +1,7 @@ +"""Document loading and chunking utilities.""" + +from ragframework.document.chunkers import FixedSizeChunker, RecursiveChunker, SentenceChunker + from .html import HTMLLoader from .loaders import DocxLoader, MarkdownLoader, PDFLoader, TextFileLoader from .tabular import CSVLoader, JSONLLoader @@ -10,4 +14,7 @@ "CSVLoader", "JSONLLoader", "HTMLLoader", -] \ No newline at end of file + "FixedSizeChunker", + "RecursiveChunker", + "SentenceChunker", +] diff --git a/tests/test_document/test_loaders.py b/tests/test_document/test_loaders.py index 6c175e8..5022db7 100644 --- a/tests/test_document/test_loaders.py +++ b/tests/test_document/test_loaders.py @@ -6,6 +6,7 @@ import pytest +import ragframework.document.loaders as loaders from ragframework.document.loaders import ( DocxLoader, MarkdownLoader, @@ -100,6 +101,20 @@ def test_missing_file_raises(self): with pytest.raises(LoaderError, match="File not found"): loader.load("/nonexistent/path/file.docx") + + def test_missing_dependency_raises(self, tmp_path, monkeypatch): + docx_path = tmp_path / "doc.docx" + docx_path.write_bytes(b"fake docx content") + + monkeypatch.setattr(loaders, "DocxDocument", None) + + loader = DocxLoader() + + with pytest.raises( + LoaderError, + match=r"DOCX support requires 'ragframework\[docx\]'", + ): + loader.load(str(docx_path)) def test_invalid_docx_raises(self, tmp_path): docx_path = tmp_path / "invalid.docx" From 8adcbae48a664759f5f18dbffceacf22f73d0c98 Mon Sep 17 00:00:00 2001 From: V S S L Deepak Janapa <125086435+deepakjanapa@users.noreply.github.com> Date: Sun, 27 Sep 2026 15:03:55 +0530 Subject: [PATCH 4/4] fix: address DOCX review feedback --- .github/workflows/ci.yml | 2 +- ragframework/document/loaders.py | 11 ++++------- tests/test_document/test_loaders.py | 12 +++++++++--- 3 files changed, 14 insertions(+), 11 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index ffe6fbb..c6dfdd4 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -52,7 +52,7 @@ jobs: cache: pip - name: Install type-check dependencies - run: pip install -c .github/constraints-typecheck.txt -e ".[faiss]" "mypy>=1.10" + run: pip install -c .github/constraints-typecheck.txt -e ".[faiss,docx]" "mypy>=1.10" - name: Type check with mypy run: mypy ragframework/ diff --git a/ragframework/document/loaders.py b/ragframework/document/loaders.py index 9c43df7..057a94c 100644 --- a/ragframework/document/loaders.py +++ b/ragframework/document/loaders.py @@ -13,11 +13,6 @@ from ragframework.base import Document, DocumentLoader from ragframework.exceptions import LoaderError -try: - from docx import Document as DocxDocument -except ImportError: - DocxDocument = None - def _make_id(source: str) -> str: return hashlib.md5(source.encode()).hexdigest()[:12] @@ -93,11 +88,13 @@ def load(self, source: str) -> list[Document]: if not path.is_file(): raise LoaderError(f"Not a file: {source}") - if DocxDocument is None: + try: + from docx import Document as DocxDocument + except ImportError as exc: raise LoaderError( "DOCX support requires 'ragframework[docx]'. " "Install it with: pip install ragframework[docx]" - ) + ) from exc try: docx = DocxDocument(str(path)) diff --git a/tests/test_document/test_loaders.py b/tests/test_document/test_loaders.py index 5022db7..9ad6388 100644 --- a/tests/test_document/test_loaders.py +++ b/tests/test_document/test_loaders.py @@ -6,7 +6,6 @@ import pytest -import ragframework.document.loaders as loaders from ragframework.document.loaders import ( DocxLoader, MarkdownLoader, @@ -101,12 +100,19 @@ def test_missing_file_raises(self): with pytest.raises(LoaderError, match="File not found"): loader.load("/nonexistent/path/file.docx") - + def test_missing_dependency_raises(self, tmp_path, monkeypatch): docx_path = tmp_path / "doc.docx" docx_path.write_bytes(b"fake docx content") - monkeypatch.setattr(loaders, "DocxDocument", None) + original_import = builtins.__import__ + + def mock_import(name, *args, **kwargs): + if name == "docx": + raise ImportError("No module named 'docx'") + return original_import(name, *args, **kwargs) + + monkeypatch.setattr(builtins, "__import__", mock_import) loader = DocxLoader()