diff --git a/haystack/components/routers/document_type_router.py b/haystack/components/routers/document_type_router.py index 24b9723b91..99008c2b3a 100644 --- a/haystack/components/routers/document_type_router.py +++ b/haystack/components/routers/document_type_router.py @@ -134,9 +134,15 @@ def run(self, documents: list[Document]) -> dict[str, list[Document]]: mime_type = doc.meta.get(self.mime_type_meta_field) if self.mime_type_meta_field else None file_path = doc.meta.get(self.file_path_meta_field) if self.file_path_meta_field else None + if not isinstance(mime_type, str): + mime_type = None + if mime_type is None and file_path: # if mime_type is not provided, try to guess it from the file path - mime_type = _guess_mime_type(Path(file_path)) + try: + mime_type = _guess_mime_type(Path(file_path)) + except (TypeError, ValueError): + mime_type = None matched = False if mime_type: diff --git a/releasenotes/notes/fix-document-type-router-non-str-mime-e4f8b29a1c706d53.yaml b/releasenotes/notes/fix-document-type-router-non-str-mime-e4f8b29a1c706d53.yaml new file mode 100644 index 0000000000..97943a00e8 --- /dev/null +++ b/releasenotes/notes/fix-document-type-router-non-str-mime-e4f8b29a1c706d53.yaml @@ -0,0 +1,5 @@ +--- +fixes: + - | + ``DocumentTypeRouter`` now routes documents whose MIME metadata is not a string to ``unclassified``, + instead of raising ``TypeError`` inside ``re.fullmatch``. diff --git a/test/components/routers/test_document_type_router.py b/test/components/routers/test_document_type_router.py index 38620a0bf6..951494d2ad 100644 --- a/test/components/routers/test_document_type_router.py +++ b/test/components/routers/test_document_type_router.py @@ -191,6 +191,35 @@ def test_run_with_missing_metadata(self): assert len(result["unclassified"]) == 3 assert "text/plain" not in result + def test_run_with_non_string_mime_type_metadata(self): + docs = [ + Document(content="List mime", meta={"mime_type": ["text/plain"]}), + Document(content="Int mime", meta={"mime_type": 123}), + Document(content="None mime", meta={"mime_type": None}), + Document(content="Valid mime", meta={"mime_type": "text/plain"}), + ] + + router = DocumentTypeRouter(mime_type_meta_field="mime_type", mime_types=["text/plain"]) + result = router.run(documents=docs) + + assert len(result["text/plain"]) == 1 + assert result["text/plain"][0].content == "Valid mime" + assert len(result["unclassified"]) == 3 + + def test_run_with_non_string_file_path_metadata(self): + docs = [ + Document(content="List path", meta={"file_path": ["example.txt"]}), + Document(content="Int path", meta={"file_path": 123}), + Document(content="Valid path", meta={"file_path": "example.txt"}), + ] + + router = DocumentTypeRouter(file_path_meta_field="file_path", mime_types=["text/plain"]) + result = router.run(documents=docs) + + assert len(result["text/plain"]) == 1 + assert result["text/plain"][0].content == "Valid path" + assert len(result["unclassified"]) == 2 + def test_run_with_regex_patterns(self): docs = [ Document(content="Plain text", meta={"mime_type": "text/plain"}),