Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 4 additions & 2 deletions render/src/pixelrag_render/backends/pdf.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,7 @@ def render_pdf(
output_dir: Directory to write the tile subdirectory into.
dpi: Resolution for rendering (default 200 gives ~1650×2200px for A4).
pages: 1-based list of page numbers to render. ``None`` renders all pages.
Tile filenames and indices retain their zero-based source page index.
quality: JPEG quality 1-100 (default 85).
stem: Override for the tile directory name. Defaults to the PDF filename
stem. The pipeline passes the article_id here so directory names
Expand Down Expand Up @@ -82,10 +83,11 @@ def render_pdf(

saved_tiles: list[str] = []
chunks_info: list[dict] = []
for idx, img in enumerate(images):
first_index = convert_kwargs.get("first_page", 1) - 1
for idx, img in enumerate(images, start=first_index):
# If caller provided a sparse page list, skip pages not in the list
if pages is not None:
page_num = min(pages) + idx
page_num = idx + 1
if page_num not in pages:
continue

Expand Down
43 changes: 43 additions & 0 deletions tests/test_pdf_page_indices.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,43 @@
"""Selected PDF pages retain the same zero-based IDs as a full render."""

import json
import sys
from types import ModuleType

import pytest
from PIL import Image
from pixelrag_render import render_pdf


@pytest.mark.parametrize("pages", [None, [1], [3], [4], [2, 3], [1, 3], [2, 4], [4, 2]])
def test_pdf_page_indices_match_full_render(tmp_path, monkeypatch, pages):
source = tmp_path / "document.pdf"
source.touch()
colors = ["red", "green", "blue", "yellow"]
converter = ModuleType("pdf2image")

def convert_from_path(**kwargs):
first = kwargs.get("first_page", 1)
last = kwargs.get("last_page", 4)
return [
Image.new("RGB", (16, 16), colors[i - 1]) for i in range(first, last + 1)
]

converter.convert_from_path = convert_from_path
monkeypatch.setitem(sys.modules, "pdf2image", converter)
full_dir = render_pdf(source, tmp_path / "full")[0]
selected_dir = render_pdf(source, tmp_path / "selected", pages=pages)[0]
expected_ids = [0, 1, 2, 3] if pages is None else sorted(p - 1 for p in pages)
expected_files = [f"tile_{i:04d}.jpg" for i in expected_ids]
manifest = json.loads((selected_dir / "tiles.json").read_text())
chunks = json.loads((selected_dir / "chunks.json").read_text())["chunks"]

assert manifest["tiles"] == expected_files
assert [c["tile_index"] for c in chunks] == expected_ids
assert [c["file"] for c in chunks] == expected_files
assert [c["tile"] for c in chunks] == expected_files
assert sorted(p.name for p in selected_dir.glob("tile_*.jpg")) == expected_files
for filename in expected_files:
assert (selected_dir / filename).read_bytes() == (
full_dir / filename
).read_bytes()