From 2c5528f5e76b95664e2c704472b405b8fd6549ba Mon Sep 17 00:00:00 2001 From: davidsbatista <7937824+davidsbatista@users.noreply.github.com> Date: Tue, 22 Sep 2026 09:21:19 +0000 Subject: [PATCH] Sync Core Integrations API reference (opendataloader_pdf) on Docusaurus --- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- .../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++---- 17 files changed, 340 insertions(+), 68 deletions(-) diff --git a/docs-website/reference/integrations-api/opendataloader_pdf.md b/docs-website/reference/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-2.18/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.18/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-2.18/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-2.18/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-2.19/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.19/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-2.19/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-2.19/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-2.20/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.20/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-2.20/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-2.20/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-2.21/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.21/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-2.21/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-2.21/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-2.22/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.22/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-2.22/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-2.22/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-2.23/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.23/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-2.23/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-2.23/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-2.24/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.24/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-2.24/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-2.24/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-2.25/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.25/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-2.25/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-2.25/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-2.26/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.26/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-2.26/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-2.26/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-2.27/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.27/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-2.27/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-2.27/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-2.28/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.28/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-2.28/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-2.28/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-2.29/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.29/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-2.29/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-2.29/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-2.30/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.30/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-2.30/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-2.30/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-2.31/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.31/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-2.31/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-2.31/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-3.0/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-3.0/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-3.0/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-3.0/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field. diff --git a/docs-website/reference_versioned_docs/version-3.1/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-3.1/integrations-api/opendataloader_pdf.md index ce3d17e5b83..9afa304b62b 100644 --- a/docs-website/reference_versioned_docs/version-3.1/integrations-api/opendataloader_pdf.md +++ b/docs-website/reference_versioned_docs/version-3.1/integrations-api/opendataloader_pdf.md @@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf" OpenDataLoader PDF converter component. The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and -returns Haystack Document objects. +returns Haystack Document objects. It can also extract images to a persistent directory and return one image +Document per extracted file. Java 11 or newer must be installed and available on PATH. @@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH. ```python from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter -converter = OpenDataLoaderConverter(output_format="markdown") +converter = OpenDataLoaderConverter( + output_format="markdown", extract_images=True, image_output_dir="extracted_images" +) result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"}) documents = result["documents"] +image_documents = result["image_documents"] print(documents[0].content) +print(documents[0].meta["file_path"]) ``` #### __init__ @@ -35,7 +40,9 @@ print(documents[0].content) __init__( *, output_format: OutputFormat = "markdown", - convert_kwargs: dict[str, Any] | None = None + convert_kwargs: dict[str, Any] | None = None, + extract_images: bool = False, + image_output_dir: str | Path | None = None ) -> None ``` @@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter. - **output_format** (OutputFormat) – Format OpenDataLoader should produce. - **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the [OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options). + The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored. +- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output. +- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images + in a unique subdirectory of this directory. Required when `extract_images` is `True`. + +**Raises:** + +- ValueError – If image extraction is enabled without an output directory. #### to_dict @@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents. **Returns:** -- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents. +- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the + persistent extracted image path in its `file_path` metadata field.