diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 7634e95..c6dfdd4 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -25,7 +25,7 @@ jobs: cache: pip - name: Install dependencies - run: pip install -e ".[dev,faiss,chromadb]" + run: pip install -e ".[dev,faiss,chromadb,docx]" - name: Lint with ruff run: ruff check ragframework/ @@ -52,7 +52,7 @@ jobs: cache: pip - name: Install type-check dependencies - run: pip install -c .github/constraints-typecheck.txt -e ".[faiss]" "mypy>=1.10" + run: pip install -c .github/constraints-typecheck.txt -e ".[faiss,docx]" "mypy>=1.10" - name: Type check with mypy run: mypy ragframework/ diff --git a/CHANGELOG.md b/CHANGELOG.md index e6be596..5bb1087 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,6 +10,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [0.3.0] - 2026-09-25 ### Added + +- `DocxLoader` with support for per-paragraph and whole-file modes (closes #2) - PEP 561 `py.typed` marker in source distributions and wheels so downstream type checkers can use the package's annotations (closes #47). - Add an optional `max_chars` limit to `SentenceChunker`. - `RAGConfig.embed_batch_size` so `RAGPipeline.ingest()` embeds chunks in bounded batches (closes #42) @@ -19,17 +21,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - `CSVLoader` and `JSONLLoader` for loading selected record fields into documents with configurable IDs, metadata, encoding, and row/line-specific errors, using only the standard library (closes #36). - `HTMLLoader` for extracting readable text and title metadata from local HTML files and HTTP(S) URLs with no optional dependencies (closes #35). -### Changed -- README roadmap updated to reflect shipped components. - ### Fixed -- `InMemoryRetriever.retrieve()` now returns an empty list for non-positive `top_k` values and raises `RetrieverError` for non-integer or boolean `top_k` values (closes #23). + - Enforce LF line endings with `.gitattributes` across platforms while keeping PNG files binary (closes #48). +- `InMemoryRetriever.retrieve()` now returns an empty list for non-positive `top_k` values and raises `RetrieverError` for non-integer or boolean `top_k` values (closes #23). - `InMemoryRetriever` now raises `RetrieverError` for invalid vectors and dimension mismatches, validates complete batches before updating stored data, and treats empty batches as a no-op. Vector validation and normalization are shared with `FAISSRetriever` (closes #26). ## [0.2.0] - 2026-09-19 ### Added + - Optional reranking stage with `Reranker`, `CrossEncoderReranker`, `NoOpReranker`, and configurable pre-rerank retrieval depth (closes #38) - `AnthropicGenerator` for grounded answers via the Anthropic Messages API (closes #20) - `ChromaRetriever` for ephemeral and persistent ChromaDB-backed vector retrieval (closes #5) @@ -40,17 +41,21 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - `AsyncRAGPipeline` for asynchronous RAG ingestion and querying using `asyncio.to_thread()` (closes #9) ### Fixed + - Validate configured embedding dimensions during ingestion and querying (closes #29) ### Changed + - Make `embedding_dim` validation opt-in and add `RAGPipeline.from_config()` for chunk settings - License metadata now uses an SPDX expression (`license = "MIT"`) in `pyproject.toml` - Releases are published to PyPI via GitHub Actions trusted publishing (see `RELEASING.md`) ====== + ## [0.1.0] - 2026-03-24 ### Added + - Initial project scaffold with modular architecture - Abstract base classes: `DocumentLoader`, `TextChunker`, `Embedder`, `Retriever`, `Generator` - Core dataclasses: `Document`, `Chunk`, `RAGResponse` diff --git a/ragframework/document/__init__.py b/ragframework/document/__init__.py index 930840e..dec3766 100644 --- a/ragframework/document/__init__.py +++ b/ragframework/document/__init__.py @@ -3,13 +3,14 @@ from ragframework.document.chunkers import FixedSizeChunker, RecursiveChunker, SentenceChunker from .html import HTMLLoader -from .loaders import MarkdownLoader, PDFLoader, TextFileLoader +from .loaders import DocxLoader, MarkdownLoader, PDFLoader, TextFileLoader from .tabular import CSVLoader, JSONLLoader __all__ = [ "TextFileLoader", "MarkdownLoader", "PDFLoader", + "DocxLoader", "CSVLoader", "JSONLLoader", "HTMLLoader", diff --git a/ragframework/document/loaders.py b/ragframework/document/loaders.py index 9f3ee55..057a94c 100644 --- a/ragframework/document/loaders.py +++ b/ragframework/document/loaders.py @@ -1,7 +1,7 @@ """Built-in document loaders. -These loaders handle plain text and Markdown files, PDF with no extra dependencies. -For DOCX, HTML, and other formats see the open issues in +These loaders handle plain text, Markdown, PDF, and DOCX files. +For HTML and other formats see the open issues in `.github/GOOD_FIRST_ISSUES.md`. """ @@ -73,6 +73,63 @@ def load(self, source: str) -> list[Document]: ] +class DocxLoader(DocumentLoader): + """Load a DOCX file into one or more :class:`Document` objects.""" + + def __init__(self, split_paragraphs: bool = True) -> None: + self.split_paragraphs = split_paragraphs + + def load(self, source: str) -> list[Document]: + path = Path(source) + + if not path.exists(): + raise LoaderError(f"File not found: {source}") + + if not path.is_file(): + raise LoaderError(f"Not a file: {source}") + + try: + from docx import Document as DocxDocument + except ImportError as exc: + raise LoaderError( + "DOCX support requires 'ragframework[docx]'. " + "Install it with: pip install ragframework[docx]" + ) from exc + + try: + docx = DocxDocument(str(path)) + except Exception as exc: + raise LoaderError(f"Could not read DOCX file {source}: {exc}") from exc + paragraphs = [paragraph.text for paragraph in docx.paragraphs if paragraph.text.strip()] + if self.split_paragraphs: + return [ + Document( + id=_make_id(f"{source}_paragraph{i}"), + content=paragraph, + metadata={ + "source": source, + "filename": path.name, + "format": "docx", + "paragraph_number": i, + }, + ) + for i, paragraph in enumerate(paragraphs, start=1) + ] + + return [ + Document( + id=_make_id(source), + content="\n\n".join(paragraphs), + metadata={ + "source": source, + "filename": path.name, + "format": "docx", + "split_paragraphs": False, + }, + ) + ] + + class PDFLoader(DocumentLoader): """Load a PDF file into one or more :class:`Document` objects. By default, each page is loaded as a separate document with metadata indicating the page number. diff --git a/tests/test_document/test_loaders.py b/tests/test_document/test_loaders.py index a30aebb..9ad6388 100644 --- a/tests/test_document/test_loaders.py +++ b/tests/test_document/test_loaders.py @@ -6,7 +6,12 @@ import pytest -from ragframework.document.loaders import MarkdownLoader, PDFLoader, TextFileLoader +from ragframework.document.loaders import ( + DocxLoader, + MarkdownLoader, + PDFLoader, + TextFileLoader, +) from ragframework.exceptions import LoaderError @@ -49,6 +54,84 @@ def test_missing_file_raises(self): loader.load("/no/such/file.md") +class TestDocxLoader: + def test_loads_paragraphs(self, tmp_path): + from docx import Document as DocxDocument + + docx_path = tmp_path / "doc.docx" + + doc = DocxDocument() + doc.add_paragraph("First paragraph") + doc.add_paragraph("") + doc.add_paragraph("Second paragraph") + doc.save(docx_path) + + loader = DocxLoader() + docs = loader.load(str(docx_path)) + + assert len(docs) == 2 + assert docs[0].content == "First paragraph" + assert docs[1].content == "Second paragraph" + assert docs[0].metadata["format"] == "docx" + assert docs[0].metadata["paragraph_number"] == 1 + assert docs[1].metadata["paragraph_number"] == 2 + + def test_loads_whole_file(self, tmp_path): + from docx import Document as DocxDocument + + docx_path = tmp_path / "doc.docx" + + doc = DocxDocument() + doc.add_paragraph("First paragraph") + doc.add_paragraph("Second paragraph") + doc.save(docx_path) + + loader = DocxLoader(split_paragraphs=False) + docs = loader.load(str(docx_path)) + + assert len(docs) == 1 + assert "First paragraph" in docs[0].content + assert "Second paragraph" in docs[0].content + assert docs[0].metadata["format"] == "docx" + assert docs[0].metadata["split_paragraphs"] is False + + def test_missing_file_raises(self): + loader = DocxLoader() + + with pytest.raises(LoaderError, match="File not found"): + loader.load("/nonexistent/path/file.docx") + + def test_missing_dependency_raises(self, tmp_path, monkeypatch): + docx_path = tmp_path / "doc.docx" + docx_path.write_bytes(b"fake docx content") + + original_import = builtins.__import__ + + def mock_import(name, *args, **kwargs): + if name == "docx": + raise ImportError("No module named 'docx'") + return original_import(name, *args, **kwargs) + + monkeypatch.setattr(builtins, "__import__", mock_import) + + loader = DocxLoader() + + with pytest.raises( + LoaderError, + match=r"DOCX support requires 'ragframework\[docx\]'", + ): + loader.load(str(docx_path)) + + def test_invalid_docx_raises(self, tmp_path): + docx_path = tmp_path / "invalid.docx" + docx_path.write_text("This is not a valid DOCX file.", encoding="utf-8") + + loader = DocxLoader() + + with pytest.raises(LoaderError, match="Could not read DOCX file"): + loader.load(str(docx_path)) + + def test_pdf_loader_requires_pypdf(tmp_path, monkeypatch): pdf_path = tmp_path / "doc.pdf" pdf_path.write_bytes(b"%PDF-1.4\n%EOF\n")