From 2c5528f5e76b95664e2c704472b405b8fd6549ba Mon Sep 17 00:00:00 2001
From: davidsbatista <7937824+davidsbatista@users.noreply.github.com>
Date: Tue, 22 Sep 2026 09:21:19 +0000
Subject: [PATCH] Sync Core Integrations API reference (opendataloader_pdf) on
Docusaurus
---
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
.../integrations-api/opendataloader_pdf.md | 24 +++++++++++++++----
17 files changed, 340 insertions(+), 68 deletions(-)
diff --git a/docs-website/reference/integrations-api/opendataloader_pdf.md b/docs-website/reference/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-2.18/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.18/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-2.18/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-2.18/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-2.19/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.19/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-2.19/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-2.19/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-2.20/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.20/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-2.20/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-2.20/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-2.21/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.21/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-2.21/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-2.21/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-2.22/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.22/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-2.22/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-2.22/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-2.23/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.23/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-2.23/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-2.23/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-2.24/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.24/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-2.24/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-2.24/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-2.25/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.25/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-2.25/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-2.25/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-2.26/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.26/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-2.26/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-2.26/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-2.27/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.27/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-2.27/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-2.27/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-2.28/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.28/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-2.28/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-2.28/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-2.29/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.29/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-2.29/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-2.29/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-2.30/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.30/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-2.30/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-2.30/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-2.31/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-2.31/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-2.31/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-2.31/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-3.0/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-3.0/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-3.0/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-3.0/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.
diff --git a/docs-website/reference_versioned_docs/version-3.1/integrations-api/opendataloader_pdf.md b/docs-website/reference_versioned_docs/version-3.1/integrations-api/opendataloader_pdf.md
index ce3d17e5b83..9afa304b62b 100644
--- a/docs-website/reference_versioned_docs/version-3.1/integrations-api/opendataloader_pdf.md
+++ b/docs-website/reference_versioned_docs/version-3.1/integrations-api/opendataloader_pdf.md
@@ -13,7 +13,8 @@ slug: "/integrations-opendataloader-pdf"
OpenDataLoader PDF converter component.
The component accepts PDF file paths and Haystack ByteStream objects, runs OpenDataLoader PDF extraction, and
-returns Haystack Document objects.
+returns Haystack Document objects. It can also extract images to a persistent directory and return one image
+Document per extracted file.
Java 11 or newer must be installed and available on PATH.
@@ -22,11 +23,15 @@ Java 11 or newer must be installed and available on PATH.
```python
from haystack_integrations.components.converters.opendataloader_pdf import OpenDataLoaderConverter
-converter = OpenDataLoaderConverter(output_format="markdown")
+converter = OpenDataLoaderConverter(
+ output_format="markdown", extract_images=True, image_output_dir="extracted_images"
+)
result = converter.run(sources=["report.pdf"], meta={"source": "annual-report"})
documents = result["documents"]
+image_documents = result["image_documents"]
print(documents[0].content)
+print(documents[0].meta["file_path"])
```
#### __init__
@@ -35,7 +40,9 @@ print(documents[0].content)
__init__(
*,
output_format: OutputFormat = "markdown",
- convert_kwargs: dict[str, Any] | None = None
+ convert_kwargs: dict[str, Any] | None = None,
+ extract_images: bool = False,
+ image_output_dir: str | Path | None = None
) -> None
```
@@ -46,6 +53,14 @@ Initialize the OpenDataLoader converter.
- **output_format** (OutputFormat) – Format OpenDataLoader should produce.
- **convert_kwargs** (dict\[str, Any\] | None) – Additional arguments passed to `opendataloader_pdf.convert`. See the
[OpenDataLoader PDF Python options](https://opendataloader.org/docs/quick-start-python#convert-options).
+ The `image_output` and `image_dir` arguments are managed by this component; supplied values are ignored.
+- **extract_images** (bool) – Whether to extract images and return them through the `image_documents` output.
+- **image_output_dir** (str | Path | None) – Persistent root directory for extracted image files. Each `run()` stores its images
+ in a unique subdirectory of this directory. Required when `extract_images` is `True`.
+
+**Raises:**
+
+- ValueError – If image extraction is enabled without an output directory.
#### to_dict
@@ -94,4 +109,5 @@ Convert PDF sources into Haystack Documents.
**Returns:**
-- dict\[str, list\[Document\]\] – Dictionary containing the converted Documents.
+- dict\[str, list\[Document\]\] – Dictionary containing the converted text Documents and image Documents. Each image Document has the
+ persistent extracted image path in its `file_path` metadata field.