From 76c9dea5761e0a44bd0e214442ceccc30980d60d Mon Sep 17 00:00:00 2001 From: Mohammad Hijjawi Date: Thu, 17 Sep 2026 10:59:21 +0000 Subject: [PATCH 1/2] fix: skip non-string MIME metadata in DocumentTypeRouter re.fullmatch requires a string. List or integer mime_type metadata raised TypeError. --- .../components/routers/document_type_router.py | 8 +++++++- ...type-router-non-str-mime-e4f8b29a1c706d53.yaml | 5 +++++ .../routers/test_document_type_router.py | 15 +++++++++++++++ 3 files changed, 27 insertions(+), 1 deletion(-) create mode 100644 releasenotes/notes/fix-document-type-router-non-str-mime-e4f8b29a1c706d53.yaml diff --git a/haystack/components/routers/document_type_router.py b/haystack/components/routers/document_type_router.py index 24b9723b918..99008c2b3aa 100644 --- a/haystack/components/routers/document_type_router.py +++ b/haystack/components/routers/document_type_router.py @@ -134,9 +134,15 @@ def run(self, documents: list[Document]) -> dict[str, list[Document]]: mime_type = doc.meta.get(self.mime_type_meta_field) if self.mime_type_meta_field else None file_path = doc.meta.get(self.file_path_meta_field) if self.file_path_meta_field else None + if not isinstance(mime_type, str): + mime_type = None + if mime_type is None and file_path: # if mime_type is not provided, try to guess it from the file path - mime_type = _guess_mime_type(Path(file_path)) + try: + mime_type = _guess_mime_type(Path(file_path)) + except (TypeError, ValueError): + mime_type = None matched = False if mime_type: diff --git a/releasenotes/notes/fix-document-type-router-non-str-mime-e4f8b29a1c706d53.yaml b/releasenotes/notes/fix-document-type-router-non-str-mime-e4f8b29a1c706d53.yaml new file mode 100644 index 00000000000..97943a00e80 --- /dev/null +++ b/releasenotes/notes/fix-document-type-router-non-str-mime-e4f8b29a1c706d53.yaml @@ -0,0 +1,5 @@ +--- +fixes: + - | + ``DocumentTypeRouter`` now routes documents whose MIME metadata is not a string to ``unclassified``, + instead of raising ``TypeError`` inside ``re.fullmatch``. diff --git a/test/components/routers/test_document_type_router.py b/test/components/routers/test_document_type_router.py index 38620a0bf6b..e106cd38828 100644 --- a/test/components/routers/test_document_type_router.py +++ b/test/components/routers/test_document_type_router.py @@ -191,6 +191,21 @@ def test_run_with_missing_metadata(self): assert len(result["unclassified"]) == 3 assert "text/plain" not in result + def test_run_with_non_string_mime_type_metadata(self): + docs = [ + Document(content="List mime", meta={"mime_type": ["text/plain"]}), + Document(content="Int mime", meta={"mime_type": 123}), + Document(content="None mime", meta={"mime_type": None}), + Document(content="Valid mime", meta={"mime_type": "text/plain"}), + ] + + router = DocumentTypeRouter(mime_type_meta_field="mime_type", mime_types=["text/plain"]) + result = router.run(documents=docs) + + assert len(result["text/plain"]) == 1 + assert result["text/plain"][0].content == "Valid mime" + assert len(result["unclassified"]) == 3 + def test_run_with_regex_patterns(self): docs = [ Document(content="Plain text", meta={"mime_type": "text/plain"}), From 87199867ce8c94d3ca5c4feb6d7df6571334d562 Mon Sep 17 00:00:00 2001 From: Mohammad Hijjawi Date: Thu, 1 Oct 2026 14:56:38 +0100 Subject: [PATCH 2/2] test: cover DocumentTypeRouter MIME guess errors List or integer file_path metadata makes Path() raise TypeError. Check that those documents go to unclassified. --- .../routers/test_document_type_router.py | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/test/components/routers/test_document_type_router.py b/test/components/routers/test_document_type_router.py index e106cd38828..951494d2ad8 100644 --- a/test/components/routers/test_document_type_router.py +++ b/test/components/routers/test_document_type_router.py @@ -206,6 +206,20 @@ def test_run_with_non_string_mime_type_metadata(self): assert result["text/plain"][0].content == "Valid mime" assert len(result["unclassified"]) == 3 + def test_run_with_non_string_file_path_metadata(self): + docs = [ + Document(content="List path", meta={"file_path": ["example.txt"]}), + Document(content="Int path", meta={"file_path": 123}), + Document(content="Valid path", meta={"file_path": "example.txt"}), + ] + + router = DocumentTypeRouter(file_path_meta_field="file_path", mime_types=["text/plain"]) + result = router.run(documents=docs) + + assert len(result["text/plain"]) == 1 + assert result["text/plain"][0].content == "Valid path" + assert len(result["unclassified"]) == 2 + def test_run_with_regex_patterns(self): docs = [ Document(content="Plain text", meta={"mime_type": "text/plain"}),