diff --git a/.github/dependabot.yml b/.github/dependabot.yml new file mode 100644 index 0000000..8615d4e --- /dev/null +++ b/.github/dependabot.yml @@ -0,0 +1,19 @@ +# Keeps the three pinned surfaces current: the uv lockfile, the SHA-pinned +# actions in ci.yml, and the digest-pinned images in the Dockerfile. +version: 2 +updates: + - package-ecosystem: uv + directory: / + schedule: + interval: weekly + groups: + dev-tooling: + patterns: [ruff, pyright, pytest, pre-commit] + - package-ecosystem: github-actions + directory: / + schedule: + interval: weekly + - package-ecosystem: docker + directory: / + schedule: + interval: weekly diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index d96ba79..df88bbe 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -5,16 +5,26 @@ on: branches: [main] pull_request: +# Least privilege: nothing here writes to the repository. +permissions: + contents: read + +# A newer push to the same branch or PR supersedes the run in flight. +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + jobs: quality: runs-on: ubuntu-latest + timeout-minutes: 15 steps: - - uses: actions/checkout@v4 - - uses: astral-sh/setup-uv@v6 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 with: enable-cache: true - name: Sync dependencies - run: uv sync --frozen + run: uv sync --locked - name: Lint run: uv run ruff check . - name: Format check @@ -24,16 +34,17 @@ jobs: - name: Tests run: uv run pytest - name: Validate decision register - run: python3 docs/scripts/validate_decisions.py docs/DECISIONS.md + run: uv run python docs/scripts/validate_decisions.py docs/DECISIONS.md docker: runs-on: ubuntu-latest + timeout-minutes: 20 steps: - - uses: actions/checkout@v4 - - uses: astral-sh/setup-uv@v6 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 with: enable-cache: true - name: Sync dependencies - run: uv sync --frozen + run: uv sync --locked - name: Container contract tests (builds the image) run: uv run pytest -m container diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index edec9bc..72bbcba 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -1,6 +1,9 @@ +# The Python tools run through uv so that pre-commit, CI and a developer's +# shell all use the single version pinned in uv.lock. Only the generic +# hygiene hooks come from a remote repo. repos: - repo: https://github.com/pre-commit/pre-commit-hooks - rev: v5.0.0 + rev: v6.0.0 hooks: - id: check-yaml - id: check-toml @@ -8,14 +11,23 @@ repos: - id: end-of-file-fixer - id: check-added-large-files - - repo: https://github.com/astral-sh/ruff-pre-commit - rev: v0.14.9 + - repo: local hooks: - id: ruff-check - args: ["--fix"] + name: ruff check + entry: uv run --locked ruff check --fix --force-exclude + language: system + types_or: [python, pyi] + require_serial: true - id: ruff-format - - - repo: https://github.com/RobertCraigie/pyright-python - rev: v1.1.408 - hooks: + name: ruff format + entry: uv run --locked ruff format --force-exclude + language: system + types_or: [python, pyi] + require_serial: true - id: pyright + name: pyright + entry: uv run --locked pyright + language: system + types_or: [python, pyi] + pass_filenames: false diff --git a/Dockerfile b/Dockerfile index 1de0489..cb5eb7e 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,16 +1,40 @@ -# Deliberately simple starter image. Planned hardening: pinned-by-digest base, -# multi-stage uv build (no pip/build tooling in the runtime layer), verified -# read-only-rootfs posture. -FROM python:3.14-slim +# Build stage: uv resolves the locked dependency set into a self-contained +# virtualenv. Nothing from this stage's tooling (uv, caches) reaches the +# runtime image. +FROM python:3.14-slim@sha256:cad9a2c871761c413caa6fdd6441c783451e740a48aaeba60ae62a8b53525ef6 AS build +COPY --from=ghcr.io/astral-sh/uv:0.10@sha256:72ab0aeb448090480ccabb99fb5f52b0dc3c71923bffb5e2e26517a1c27b7fec /uv /bin/uv +ENV UV_COMPILE_BYTECODE=1 \ + UV_LINK_MODE=copy \ + UV_PYTHON_DOWNLOADS=never WORKDIR /app -COPY pyproject.toml README.md LICENSE ./ + +# Dependencies first: this layer only rebuilds when the lockfile changes. +RUN --mount=type=cache,target=/root/.cache/uv \ + --mount=type=bind,source=uv.lock,target=uv.lock \ + --mount=type=bind,source=pyproject.toml,target=pyproject.toml \ + uv sync --frozen --no-install-project --no-dev + +COPY pyproject.toml README.md LICENSE uv.lock ./ COPY src ./src -RUN pip install --no-cache-dir . +RUN --mount=type=cache,target=/root/.cache/uv \ + uv sync --frozen --no-dev --no-editable + +# Runtime: the digest-pinned interpreter, the built virtualenv, and nothing +# else. No package installer, no build tooling, a fixed non-root UID, and a +# root filesystem that works read-only (all writes go to the output mount). +FROM python:3.14-slim@sha256:cad9a2c871761c413caa6fdd6441c783451e740a48aaeba60ae62a8b53525ef6 + +RUN /usr/local/bin/python -m pip uninstall --yes pip \ + && rm -r /usr/local/lib/python3.14/ensurepip \ + && useradd --uid 10001 --create-home appuser + +COPY --from=build /app/.venv /app/.venv +ENV PATH="/app/.venv/bin:$PATH" \ + PYTHONDONTWRITEBYTECODE=1 -# Non-root from day one; fixed UID so volume-permission guidance stays concrete. -RUN useradd --uid 10001 --create-home appuser USER 10001 +WORKDIR /app # All behavior comes from PDFOPS_* environment variables - no arguments, no CMD. ENTRYPOINT ["python", "-m", "pdf_ops"] diff --git a/README.md b/README.md index 0182b88..ba10704 100644 --- a/README.md +++ b/README.md @@ -8,10 +8,12 @@ operation per container run - **merge** multiple PDFs into one, or **extract** t attachments embedded in a PDF - configured entirely through environment variables. The design - architecture, library tradeoffs, security posture, limitations - is -summarized in [`docs/DESIGN.md`](docs/DESIGN.md); the working notes behind it are -[`docs/DESIGN_NOTES.md`](docs/DESIGN_NOTES.md), and individual choices, with their -alternatives and status, live in the decision register at -[`docs/DECISIONS.md`](docs/DECISIONS.md). +summarized in [`docs/DESIGN.md`](docs/DESIGN.md) and drawn, view by view, in +[`docs/ARCHITECTURE.md`](docs/ARCHITECTURE.md); the runtime behavior contract - mounts, +passwords, output policy, log events, error codes - is [`docs/OPERATIONS.md`](docs/OPERATIONS.md); +the working notes behind the design are [`docs/DESIGN_NOTES.md`](docs/DESIGN_NOTES.md), +and individual choices, with their alternatives and status, live in the decision +register at [`docs/DECISIONS.md`](docs/DECISIONS.md). ## Quick start @@ -19,7 +21,27 @@ alternatives and status, live in the decision register at docker build -t pdf-ops . # The container runs as UID 10001 - the output dir must be writable by it -mkdir -p in out && chmod 777 out +mkdir -p in out secret && chmod 777 out + +# No PDFs at hand? Conjure the inputs the examples below use: +uv sync && uv run python - <<'PY' +from pypdf import PdfWriter +for name, pages in (("a.pdf", 1), ("b.pdf", 2)): + w = PdfWriter() + for _ in range(pages): + w.add_blank_page(width=200, height=300) + with open(f"in/{name}", "wb") as h: + w.write(h) +w = PdfWriter(); w.add_blank_page(width=200, height=300) +w.add_attachment("data.csv", b"x,y\n1,2\n") +with open("in/report.pdf", "wb") as h: + w.write(h) +w = PdfWriter(); w.add_blank_page(width=200, height=300) +w.encrypt(user_password="s3cret-pw", algorithm="AES-256") +with open("in/locked.pdf", "wb") as h: + w.write(h) +open("secret/pw", "w").write("s3cret-pw\n") +PY # Merge two PDFs from a mounted input dir into a mounted output dir docker run --rm \ @@ -41,7 +63,7 @@ docker run --rm \ docker run --rm \ -v "$PWD/in:/in:ro" -v "$PWD/out:/out" -v "$PWD/secret:/secret:ro" \ -e PDFOPS_OPERATION=merge \ - -e PDFOPS_INPUTS=/in/locked.pdf:/in/plain.pdf \ + -e PDFOPS_INPUTS=/in/locked.pdf:/in/a.pdf \ -e PDFOPS_OUTPUT=/out/merged.pdf \ -e PDFOPS_PASSWORD_FILE=/secret/pw \ -e PDFOPS_OUTPUT_ENCRYPTION=inherit \ @@ -66,10 +88,10 @@ from `PDFOPS_*` variables, and the mounted volumes provide inputs and receive ou | `PDFOPS_FAIL_ON_NO_ATTACHMENTS` | extract | no | `true`, `false` (case-insensitive) - fail (exit 3) when the PDF has no attachments | `false` | | `PDFOPS_PASSWORD_FILE` | both | no | path to a mounted secret file holding the password (preferred channel; one trailing newline stripped) | - | | `PDFOPS_PASSWORD` | both | no | the password itself - discouraged: env values leak via `kubectl describe`, `/proc//environ`, crash tooling | - | -| `PDFOPS_OUTPUT_ENCRYPTION` | merge | no | `never`, `inherit`, `always` (case-insensitive) - see below | `never` | +| `PDFOPS_OUTPUT_ENCRYPTION` | merge | no | `never`, `inherit`, `always` (case-insensitive) - see [Output encryption](docs/OPERATIONS.md#output-encryption) | `never` | | `PDFOPS_OUTPUT_PASSWORD_FILE` | merge | no | secret file holding the password for the merged output | - | | `PDFOPS_OUTPUT_PASSWORD` | merge | no | output password as a direct value (same caveats as `PDFOPS_PASSWORD`) | - | -| `PDFOPS_ON_EXISTS` | both | no | `fail`, `overwrite`, `skip` (case-insensitive) - see Retries | `fail` | +| `PDFOPS_ON_EXISTS` | both | no | `fail`, `overwrite`, `skip` (case-insensitive) - see [Existing outputs](docs/OPERATIONS.md#atomic-writes-and-existing-outputs) | `fail` | | `PDFOPS_LOG_LEVEL` | - | no | `debug`, `info`, `warning`, `error` (case-insensitive) | `info` | Strictness rules, all exit 2: any other `PDFOPS_*` variable is rejected as a probable @@ -80,7 +102,8 @@ repeated path is almost always a templating bug that would silently duplicate co ## Path conventions and output behavior The full behavior contract - mounts and permissions, password semantics, output -encryption, atomic writes, the existing-output policy, and attachment-name safety - +encryption, atomic writes, the existing-output policy, attachment-name safety, and +resource sizing - lives in [`docs/OPERATIONS.md`](docs/OPERATIONS.md). The short version: everything is mounted volumes with absolute in-container paths, the container runs as non-root UID 10001, outputs are written atomically (a complete file or nothing), and @@ -98,11 +121,14 @@ mounted volumes with absolute in-container paths, the container runs as non-root | 5 | password required/wrong/unsupported | | 6 | output conflict or output location unusable | +The finer-grained `error_code` carried by every `operation_failed` event is listed +per exit code in [`docs/OPERATIONS.md`](docs/OPERATIONS.md#error-codes). + ## Logging Output is JSON lines on stdout - one event per line, stderr stays empty. Lifecycle -events narrate progress and respect `PDFOPS_LOG_LEVEL`; passwords are echoed as -presence only (`unset` / `set(env)` / `set(file)`), never as values. Every run ends +events narrate progress and respect `PDFOPS_LOG_LEVEL`; the `config_loaded` event +echoes passwords as presence only (`unset` / `set(env)` / `set(file)`), never as values. Every run ends with exactly one terminal event - `operation_complete` or `operation_failed` (with a machine-readable `error_code`) - which no log level suppresses, so a workflow engine can always branch on the last line: @@ -129,24 +155,32 @@ code: | 5 | no | the password will still be wrong | | 6 | usually no | output conflict/location - `DISK_FULL` is the judgment call (space may free up) | -Argo example - retry only on unexpected errors, with `skip` making any retry after a -lost-but-successful pod a free no-op: +Argo example - retry unexpected errors (exit 1) and pod-level errors, which never +produce an exit code (a lost pod reports "-1"); with `skip` in the environment, a +retry after a lost-but-successful pod is a free no-op: ```yaml retryStrategy: limit: "3" - expression: "asInt(lastRetry.exitCode) == 1" + expression: >- + lastRetry.status == "Error" or asInt(lastRetry.exitCode) == 1 # and in the container env: # PDFOPS_ON_EXISTS: skip ``` +A complete `WorkflowTemplate` - security context, secret-mounted password, +retry strategy, measured resource sizing - lives in +[`deploy/argo-example.yaml`](deploy/argo-example.yaml). + ## Development ```sh uv sync # deps + venv +uv run pre-commit install # once: the hooks run the same tools through uv uv run pytest # unit + integration tests uv run pytest -m container # container-contract tests (needs Docker) uv run ruff check . # lint +uv run ruff format --check . # formatting (CI enforces it) uv run pyright # strict type check uv run pre-commit run -a # full hook chain ``` diff --git a/SECURITY.md b/SECURITY.md new file mode 100644 index 0000000..270e0fb --- /dev/null +++ b/SECURITY.md @@ -0,0 +1,23 @@ +# Security policy + +pdf-ops parses untrusted PDFs and writes their attachment names to a mounted +filesystem, so input handling is security-relevant by design. The posture - +attachment-name sanitization, the layered no-leak guarantee for passwords, +atomic outputs, a hardened read-only container - is described in +`docs/DESIGN.md`. + +## Reporting a vulnerability + +Please do not open a public issue. Use GitHub's private vulnerability reporting +on this repository (Security tab, "Report a vulnerability"); if that is not +enabled, contact the maintainer directly. Include a way to reproduce: the PDF +or how to build one, the `PDFOPS_*` configuration, and the log lines. + +## In scope + +- An attachment name that writes outside `PDFOPS_OUTPUT_DIR`. +- Password material appearing in any output: stdout, stderr, files, tracebacks. +- A partial or mixed output surviving a failed or killed run. +- A deterministic bad input classified as retryable (exit 1). +- A hostile PDF structure that escapes the error taxonomy (exit 1) instead of + classifying as a data problem. diff --git a/deploy/argo-example.yaml b/deploy/argo-example.yaml new file mode 100644 index 0000000..5666ce4 --- /dev/null +++ b/deploy/argo-example.yaml @@ -0,0 +1,67 @@ +# Argo Workflows example: pdf-ops as a single workflow step. +# +# What this template encodes: +# - the full security posture the image is tested against (read-only root +# filesystem, non-root UID 10001, no capabilities, no privilege escalation); +# - fsGroup so the output volume is writable by the container UID; +# - the password arriving as a mounted secret file, never as an env value; +# - retryStrategy that retries exit 1 (the one nondeterministic class; see +# the exit-code table in the README) and pod-level errors, which never +# produce an exit code (Argo reports "-1") - the lost-pod case. Paired +# with PDFOPS_ON_EXISTS=skip, a retry after a lost-but-successful pod is +# a free no-op; +# - memory sized by the measured rule of thumb: total expected input size +# plus 128 MB (see docs/OPERATIONS.md, "Resource sizing"). +apiVersion: argoproj.io/v1alpha1 +kind: WorkflowTemplate +metadata: + name: pdf-ops-merge +spec: + entrypoint: merge + securityContext: + fsGroup: 10001 + volumes: + - name: documents + persistentVolumeClaim: + claimName: documents # the shared volume carrying inputs and outputs + - name: pdf-password + secret: + secretName: pdf-password # key "password" -> file /secrets/password + templates: + - name: merge + retryStrategy: + limit: "3" + expression: >- + lastRetry.status == "Error" or asInt(lastRetry.exitCode) == 1 + container: + image: pdf-ops:latest # replace with your registry reference + env: + - name: PDFOPS_OPERATION + value: merge + - name: PDFOPS_INPUTS + value: /data/in/a.pdf:/data/in/b.pdf + - name: PDFOPS_OUTPUT + value: /data/out/merged.pdf + - name: PDFOPS_ON_EXISTS + value: skip + - name: PDFOPS_PASSWORD_FILE + value: /secrets/password + resources: + requests: + memory: 640Mi # ~500 MB of inputs + 128 MB headroom + cpu: 500m + limits: + memory: 640Mi + securityContext: + runAsNonRoot: true + runAsUser: 10001 + readOnlyRootFilesystem: true + allowPrivilegeEscalation: false + capabilities: + drop: [ALL] + volumeMounts: + - name: documents + mountPath: /data + - name: pdf-password + mountPath: /secrets + readOnly: true diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md new file mode 100644 index 0000000..39ab709 --- /dev/null +++ b/docs/ARCHITECTURE.md @@ -0,0 +1,262 @@ +# Architecture in diagrams + +Eight views of the same system, drawn so they render on GitHub, in pull requests +and in most editors. Each says why the shape is what it is and links into the +interactive deep-dive at [`diagrams/index.html`](diagrams/index.html), where nodes +link to the source lines they describe. Prose lives in [`DESIGN.md`](DESIGN.md), the +operator contract in [`OPERATIONS.md`](OPERATIONS.md), and every choice in the +decision register [`DECISIONS.md`](DECISIONS.md). + +A test (`tests/unit/test_docs.py`) checks that every module under `src/pdf_ops` is +named on this page, so no module can go missing from this view. + +## 1. Where it runs + +One container run is one workflow step. Everything the process needs arrives as +environment variables and mounted volumes; everything it reports leaves as JSON +lines on stdout and an exit code. For merge, the password file is read only after +the existing-output check says work remains, which is why a merge retry after a +lost-but-successful pod succeeds even when the secret mount is already gone; +extract resolves it up front, before touching the carrier. + +```mermaid +flowchart LR + engine["Workflow engine
Argo step with retryStrategy"] + subgraph pod["Pod: one container run"] + env["PDFOPS_* environment"] + proc["python -m pdf_ops
UID 10001, read-only rootfs"] + data[("/data volume
inputs read, outputs written atomically")] + secret[("/secrets volume
password file, read-only")] + end + engine -- "env, volumes, memory limit" --> env + env --> proc + data <--> proc + secret -. "merge reads it late; extract up front" .-> proc + proc -- "JSON lines on stdout" --> engine + proc -- "exit code 0-6" --> engine +``` + +Deep-dive: the deployment posture this is tested under is +[`../deploy/argo-example.yaml`](../deploy/argo-example.yaml). + +## 2. Modules and the direction of dependency + +Every arrow is a real `import`; nothing points upward. `errors.py` and +`logging_setup.py` import nothing from the package, so anything may use them. +`engine_pikepdf.py` is the only module that imports pikepdf, reached through +`engine.py`'s `get_engine()` with a lazy import, so the seam knows its +implementation while the rest of the package never does. + +```mermaid +flowchart TB + subgraph entry["Entry and orchestration"] + n_entry["__main__.py
process boundary: os.environ in, exit code out"] + n_main["main.py
run(env): the one error boundary"] + end + subgraph ops["Operations"] + n_merge["merge.py
validate everything, then write once"] + n_extract["extract.py
untrusted names in, atomic files out"] + end + subgraph seam["Engine seam"] + n_engine["engine.py
PdfEngine protocol, OpenedInput"] + n_pike["engine_pikepdf.py
the only module importing pikepdf"] + end + subgraph found["Foundations"] + n_config["config.py
pure parse of the env contract"] + n_inputs["inputs.py
up-front input validation"] + n_output["output.py
atomic writes, existing-output policy"] + n_secrets["secrets.py
Secret wrapper, refs, late resolution"] + n_logging["logging_setup.py
JSON lines, scrubbing, terminal events"] + n_errors["errors.py
ExitCode, ErrorCode, the error classes"] + end + n_entry --> n_main + n_entry --> n_config + n_main --> n_config + n_main --> n_merge + n_main --> n_extract + n_main --> n_secrets + n_main --> n_logging + n_main --> n_errors + n_merge --> n_config + n_merge --> n_engine + n_merge --> n_inputs + n_merge --> n_output + n_merge --> n_secrets + n_merge --> n_errors + n_extract --> n_config + n_extract --> n_engine + n_extract --> n_inputs + n_extract --> n_output + n_extract --> n_secrets + n_extract --> n_errors + n_engine -. "lazy, inside get_engine()" .-> n_pike + n_engine --> n_secrets + n_pike --> n_engine + n_pike --> n_secrets + n_pike --> n_errors + n_config --> n_secrets + n_config --> n_errors + n_output --> n_config + n_output --> n_errors + n_inputs --> n_errors + n_secrets --> n_logging + n_secrets --> n_errors +``` + +Deep-dive: [`diagrams/index.html#architecture`](diagrams/index.html#architecture). + +## 3. One run, end to end + +`run(env)` is the only place that catches exceptions and the only place that emits a +terminal event. Configuration is parsed completely before any file is touched; +merge resolves secrets only after the existing-output policy has decided the run +proceeds, extract at dispatch; output goes to a temp file in the destination +directory and is renamed in one step. + +```mermaid +sequenceDiagram + autonumber + participant E as __main__ + participant R as main.run + participant C as config + participant O as merge / extract + participant F as inputs / output + participant S as secrets + participant P as engine_pikepdf + participant L as logging_setup + E->>R: run(env) + R->>L: setup_logging() + R->>C: parse_config(env) + C-->>R: MergeConfig or ExtractConfig, else ConfigError + R->>L: config_loaded (passwords as presence only) + R->>O: dispatch by config type + O->>F: check_output_path / validate_inputs + O->>S: resolve_and_register (merge: after the skip decision) + O->>P: open_input(path, password) + P-->>O: OpenedInput: pages, encryption facts, repair warnings + O->>F: atomic_output(path): temp file beside the target + O->>P: merge_to / list_attachments + F-->>O: fsync and rename, or cleanup on failure + O-->>R: result fields + R->>L: emit_terminal(operation_complete or operation_failed) + R-->>E: exit code +``` + +## 4. Failures: classes, codes, exit codes + +The exit code is the external API and stays small. Each error class maps to exactly +one code; the finer `error_code` travels in the terminal event. Anything that is not +a `PdfOpsError` is a bug and exits 1, the only class an engine should retry. The +complete code table is in [`OPERATIONS.md#error-codes`](OPERATIONS.md#error-codes). + +```mermaid +flowchart LR + any["A failure inside run(env)"] --> pred{"PdfOpsError?"} + pred -- "no" --> c1["exit 1 UNEXPECTED
UNEXPECTED_ERROR, exc_type, traceback"] + pred -- "yes" --> cls{"which class?"} + cls --> c2["ConfigError, exit 2
UNKNOWN_VAR, MISSING_VAR, INVALID_*, DUPLICATE_INPUTS,
CONFLICTING_PASSWORD_SOURCES, MISSING_OUTPUT_PASSWORD,
PASSWORD_FILE_UNREADABLE, EMPTY_PASSWORD, ..."] + cls --> c3["InputError, exit 3
INPUT_MISSING, INPUT_IS_DIRECTORY,
INPUT_UNREADABLE, NO_ATTACHMENTS"] + cls --> c4["InvalidPdfError, exit 4
NOT_A_PDF, CORRUPT_PDF, UNSUPPORTED_PDF_FEATURE"] + cls --> c5["PasswordError, exit 5
PASSWORD_REQUIRED, WRONG_PASSWORD, UNSUPPORTED_ENCRYPTION"] + cls --> c6["OutputError, exit 6
OUTPUT_DIR_MISSING, OUTPUT_IS_DIRECTORY, OUTPUT_EXISTS,
OUTPUT_NOT_WRITABLE, DISK_FULL"] +``` + +## 5. Extract: the boundary an attachment name crosses + +An attachment name is attacker-controlled text that ends up as a filename on a +mounted volume. Every name passes through a pure, table-tested sanitizer, then a +casefolded collision check (the volume may be case-insensitive), then the +existing-output policy, then a containment re-check that backs the sanitizer up, +and only then an atomic write. + +```mermaid +flowchart LR + pdf["Untrusted PDF
/Names/EmbeddedFiles name tree"] --> raw["Raw name
separators, traversal, control characters,
any length, duplicates, or not even a string"] + raw --> san["sanitize_attachment_name
basename, strip C0/C1, cap at 200 bytes,
deterministic attachment_n fallback"] + san --> dedupe["_dedupe
casefolded collisions get -1, -2 suffixes"] + dedupe --> policy{"target exists?"} + policy -- "fail (default)" --> refuse["OUTPUT_EXISTS, exit 6
nothing written"] + policy -- "skip" --> keep["completed prior work
left as is"] + policy -- "absent, or overwrite" --> check["containment re-check
parent of target is PDFOPS_OUTPUT_DIR"] + check --> write["atomic_output
temp file in the output dir, then rename"] + write --> fs[("PDFOPS_OUTPUT_DIR")] +``` + +Deep-dive: [`diagrams/index.html#extract`](diagrams/index.html#extract). + +## 6. Retries: the existing-output policy as a state machine + +Workflow engines retry at least once, and a pod can vanish after its work +succeeded. `PDFOPS_ON_EXISTS` decides what the next attempt does with an output +that already exists; stale temp debris from a killed write is removed before the +first write of the next attempt. + +```mermaid +stateDiagram-v2 + state "Config parsed" as Parsed + state "Output exists?" as Exists + state "Temp write" as Writing + state "Refused: OUTPUT_EXISTS, exit 6" as Refused + state "Skipped: exit 0, skipped true" as Skipped + state "Complete: exit 0" as Complete + state "Killed mid-write" as Killed + [*] --> Invoked + Invoked --> Parsed : parse_config + Invoked --> [*] : ConfigError, exit 2 + Parsed --> Exists : check_output_path + Exists --> Writing : absent + Exists --> Writing : present and overwrite + Exists --> Refused : present and fail (default) + Exists --> Skipped : present and skip + Writing --> Complete : fsync, rename, terminal event + Writing --> Killed : pod lost + Killed --> Parsed : retry, stale temp removed before the first write + Refused --> [*] + Skipped --> [*] + Complete --> [*] + note right of Skipped + merge: whole-run no-op, inputs and password file never read + extract: per file, only the missing attachments are written + end note +``` + +Deep-dive: [`diagrams/index.html#lifecycle`](diagrams/index.html#lifecycle). + +## 7. Tests and the cross-library oracle + +pypdf is a dev-only dependency with one job: it builds every fixture and re-reads +every output, so each test is a check between two independent PDF libraries. +Fixtures are generated, never checked in. Three tiers run at three costs. + +```mermaid +flowchart LR + subgraph oracle["pypdf: the test oracle"] + build["conftest factories build fixtures
plain, encrypted, damaged, dangling refs,
hostile name trees, raw attachments"] + verify["pypdf re-reads the outputs
page counts, encryption, attachment bytes"] + end + subgraph sut["pdf_ops with the pikepdf engine"] + unit["unit: pure functions
config, sanitizer, errors, secrets, logging, docs sync"] + integ["integration: run(env) in-process
invariants on every run: empty stderr,
JSON lines only, exactly one terminal event, last"] + cont["container: docker build and run
golden merge and extract, mounted secret,
hardened posture, no package installer"] + end + build --> integ + build --> cont + integ --> verify + cont --> verify +``` + +## 8. From commit to image + +One toolchain, one version of each tool, pinned in `uv.lock`: the hooks, CI and a +developer's shell all run ruff and pyright through uv. The image is built from the +same lockfile and ships nothing that could change it. + +```mermaid +flowchart LR + commit["commit"] --> hooks["pre-commit through uv
hygiene, ruff check and format, pyright"] + hooks --> push["push, pull request"] + push --> quality["CI quality job
uv sync --locked, ruff, format, pyright,
pytest, decision-register validator"] + push --> docker["CI docker job
build the image, container contract tests"] + docker --> image["image: builder stage uv sync --locked, then runtime
digest-pinned base, no pip or ensurepip, UID 10001"] + bot["Dependabot
uv.lock, actions, base image"] -.-> push +``` diff --git a/docs/DECISIONS.md b/docs/DECISIONS.md index 6c22da3..b23d0e2 100644 --- a/docs/DECISIONS.md +++ b/docs/DECISIONS.md @@ -2,7 +2,7 @@ > **Project:** Containerized PDF operations (merge, extract attachments) for workflow systems > **Started:** 2026-08-31 -> **Last updated:** 2026-09-02 +> **Last updated:** 2026-09-04 This document is the authoritative register of all architectural decisions for this project. New decisions are appended with the next available `D-###`. See [DECISION_TRACKING_STANDARD.md](DECISION_TRACKING_STANDARD.md) for format, vocabularies, and workflow. CI validation: [`scripts/validate_decisions.py`](scripts/validate_decisions.py). @@ -18,9 +18,9 @@ This document is the authoritative register of all architectural decisions for t | [`D-004`](#D-004) | 🟢 | config | Env contract: PDFOPS_ prefix, fail-fast pure parsing, unknown-var rejection | 2026-08-31 | All config from PDFOPS_* env vars parsed by a pure function over Mapping[str,str] before any file is touched; os.environ only in __main__. Unknown PDFOPS_* vars are rejected as probable typos (exit 2 UNKNOWN_VAR); empty equals missing. | [`DESIGN_NOTES.md section 3`](DESIGN_NOTES.md) | - | | [`D-005`](#D-005) | 🟢 | observability | Observability: JSON-lines on stdout via stdlib logging | 2026-08-31 | One JSON object per line on stdout (workflow engine captures step logs); stable event tokens with structured fields; exactly one terminal event per run from the single error boundary. Stdlib logging with a small formatter - no structlog/OTel dependency. | [`DESIGN_NOTES.md section 4`](DESIGN_NOTES.md) | - | | [`D-006`](#D-006) | 🔵 | error-handling | Transient exit-code band (10+) for retryable failures | 2026-08-31 | Deferred (2026-08-31): the 0-6 map stands as-is with exit 1 the only maybe-retryable code. A dedicated 10+ transient band (e.g. transient I/O) would let Argo retryStrategy expressions retry precisely; revisit when retry semantics become load-bearing. | [`DESIGN_NOTES.md section 2`](DESIGN_NOTES.md) | - | -| [`D-007`](#D-007) | ⏸ | config | PDFOPS_INPUTS list separator | 2026-08-31 | Deferred (2026-08-31): the merge implementation ships the recommended default - os.pathsep (colon) with explicit ordered paths, no globs - as provisional. Alternatives (comma, newline, JSON array) parked; revisit if colon-in-path or workflow-templating friction appears. | [`DESIGN_NOTES.md section 3`](DESIGN_NOTES.md) | - | -| [`D-008`](#D-008) | ⏸ | config | Operation value case strictness | 2026-08-31 | Deferred (2026-08-31): strict lowercase merge/extract (whitespace tolerated) stands as implemented. Case-insensitive acceptance parked; revisit at README/contract freeze or on operator feedback. | [`DESIGN_NOTES.md section 3`](DESIGN_NOTES.md) | - | -| [`D-009`](#D-009) | ⏸ | config | Unknown PDFOPS_* variable hard rejection | 2026-08-31 | Deferred (2026-08-31): hard rejection (exit 2 UNKNOWN_VAR) stands as implemented. Softening to a warning parked; revisit if a platform legitimately injects foreign PDFOPS_* vars (e.g. when the deployment example is written). | [`DESIGN_NOTES.md section 3`](DESIGN_NOTES.md) | - | +| [`D-007`](#D-007) | 🟢 | config | PDFOPS_INPUTS list separator | 2026-09-03 | Settled (2026-09-03, deferred since 2026-08-31): os.pathsep (colon) with explicit ordered paths and no globs stands. It survived the container suite, the deployment example, and every walkthrough; colon-in-path never materialized and is pathological for mounted paths anyway (documented limitation). Alternatives (comma, newline, JSON array) rejected as churn without a driving case. | [`DESIGN_NOTES.md section 3`](DESIGN_NOTES.md) | - | +| [`D-008`](#D-008) | 🟢 | config | Operation value case strictness | 2026-09-03 | Settled (2026-09-03, deferred since 2026-08-31): strict lowercase merge/extract (whitespace tolerated) stands. Operation values come from workflow templates, not humans typing - case tolerance would add contract surface without value, and the error message already names the accepted values. | [`DESIGN_NOTES.md section 3`](DESIGN_NOTES.md) | - | +| [`D-009`](#D-009) | 🟢 | config | Unknown PDFOPS_* variable hard rejection | 2026-09-03 | Settled (2026-09-03, deferred since 2026-08-31): hard rejection (exit 2 UNKNOWN_VAR) stands. The revisit trigger - a platform injecting foreign PDFOPS_* variables - was tested by writing the deployment example, which injects none; typo protection keeps outweighing a hypothetical soft mode. | [`DESIGN_NOTES.md section 3`](DESIGN_NOTES.md) | - | | [`D-010`](#D-010) | 🟢 | reliability | Atomic output writes: temp file in destination dir + os.replace | 2026-08-31 | All output goes to a temp file in the destination directory (same filesystem), is fsynced, and is renamed over the final path in one step; the final path holds a complete PDF or nothing. Existing output refused (OUTPUT_EXISTS) until the overwrite/skip policy lands; missing output dir refused, never auto-created. | [`DESIGN_NOTES.md section 6`](DESIGN_NOTES.md) | - | | [`D-011`](#D-011) | 🟢 | pdf-engine | Merge is pages-only; input attachments/bookmarks/metadata not carried | 2026-08-31 | The merged output carries pages only: bookmarks, form fields, metadata, and embedded attachments of the inputs are not copied (no mainstream Python library copies /Names/EmbeddedFiles on merge - attachments would drop silently). Documented limitation; detect-and-warn vs fail-loud vs qpdf --copy-attachments-from evaluated when attachment handling is built out. | [`DESIGN_NOTES.md section 5`](DESIGN_NOTES.md) | - | | [`D-012`](#D-012) | 🟢 | error-handling | Merge input validation: collect-all, first problem sets exit class | 2026-08-31 | Every input is checked up front (exists, is a file, readable, %PDF- magic) before any write; all problems are reported in one failure event with the full list in context.problems, and the exit class follows the first problem in input order. Duplicate inputs are a hard config error; a single input is a valid merge. | [`DESIGN_NOTES.md section 6`](DESIGN_NOTES.md) | - | @@ -34,8 +34,15 @@ This document is the authoritative register of all architectural decisions for t | [`D-020`](#D-020) | 🟢 | error-handling | Retryability contract: exit codes stay 0-6, retry guidance is documentation | 2026-09-01 | No 10+ transient exit-code band: the 0-6 map is a published API and nearly every failure class is deterministic, so retryability lives in the README - a per-code table (exit 1 the only default-retryable, DISK_FULL the judgment call) and an Argo retryStrategy.expression snippet, paired with PDFOPS_ON_EXISTS=skip for at-least-once safety. | [`DESIGN_NOTES.md section 9`](DESIGN_NOTES.md) | - | | [`D-021`](#D-021) | 🟢 | reliability | PDFOPS_ON_EXISTS tri-state: merge whole-run skip, extract per-file completion | 2026-09-01 | fail (default) refuses; overwrite replaces atomically; skip treats existing output as completed prior work - merge short-circuits without reading inputs, extract writes only missing attachments (sound because every written file is atomic and therefore whole). Stale temp debris matching the run's own targets is removed at startup under a documented single-writer-per-output assumption. | [`DESIGN_NOTES.md section 9`](DESIGN_NOTES.md) | - | | [`D-022`](#D-022) | 🟢 | reliability | Output files honor the process umask, not mkstemp's 0600 | 2026-09-02 | atomic_output re-chmods its temp file to 0666 & ~umask at creation: mkstemp's private 0600 would ride through os.replace onto the published output, leaving it unreadable by a downstream step running as a different UID on a shared volume. Invisible under Docker Desktop's ownership-mapping mounts, real on native Linux bind mounts - caught by CI on the first Linux-host run. | [`DESIGN_NOTES.md section 6`](DESIGN_NOTES.md) | - | +| [`D-023`](#D-023) | 🟢 | pdf-engine | Engine swapped to pikepdf; pypdf demoted to dev-dependency test oracle | 2026-09-02 | engine_pikepdf.py (qpdf-backed) replaces engine_pypdf.py as the runtime engine, executing the D-002 plan: better large-file and corrupt-input behavior, native AES-256 (R=6 pinned). password_type user/owner/empty reporting survives via qpdf's password-matched flags; qpdf repairs light damage pypdf refused (pinned as behavior; warnings surface as events - at open via OpenedInput.warnings, after lazy reads/writes via collect_warnings); duplicate attachment names preserved by walking /Names/EmbeddedFiles directly with cycle and type guards (hostile shapes degrade or classify as data problems, never exit 1); the merged output is saved through the atomic layer's open temp file, keeping the single-temp cleanup contract; failed-open algorithm labels come from a best-effort raw /Encrypt scan. pypdf stays in the dev group building fixtures and verifying outputs - every test is a cross-library check. | [`DESIGN_NOTES.md section 1`](DESIGN_NOTES.md) | - | +| [`D-024`](#D-024) | 🟢 | security | Secrets stay stdlib, consolidated into one module | 2026-09-03 | All secret handling (Secret wrapper, EnvSecret/FileSecret source refs with resolve()/describe(), Secrets bundle, scrub registration) consolidated into secrets.py; config.py only parses which source is configured. Shelf options evaluated and rejected: pydantic SecretStr (compiled dependency for one masked-repr class), pydantic-settings (config-layer rewrite; secrets_dir expects field-named files in a fixed directory - a different contract from PDFOPS_PASSWORD_FILE=; ValidationError would need retranslation into the error_code taxonomy), scanner-style log redactors (pattern heuristics, weaker than the exact-value field-restricted scrub that avoids the password-oracle problem). | [`DESIGN_NOTES.md section 8`](DESIGN_NOTES.md) | - | +| [`D-025`](#D-025) | 🟢 | container | Hardened runtime image: digest-pinned multi-stage build, read-only rootfs | 2026-09-03 | The image is a two-stage build: a digest-pinned uv stage resolves the lockfile into a self-contained virtualenv with compiled bytecode; the runtime stage (digest-pinned python:3.14-slim) carries only that venv, uninstalls the base image's pip, removes stdlib ensurepip (its bundled wheel would restore pip in one command), and runs as fixed non-root UID 10001. Read-only root filesystem is proven by a container test running the golden merge under --read-only --cap-drop ALL --security-opt no-new-privileges (all writes land in the output mount by design). deploy/argo-example.yaml ships the full posture incl. fsGroup, secret-mounted password, a retry expression covering exit 1 plus pod-level Error nodes (which carry no exit code), and memory sized by the measured input+128MB rule. Distroless bases considered and not taken: pinned slim minus pip reaches most of the value while staying debuggable. | [`DESIGN_NOTES.md section 11`](DESIGN_NOTES.md) | - | +| [`D-026`](#D-026) | 🟢 | infra | One uv-locked toolchain with SHA-pinned, Dependabot-watched CI | 2026-09-04 | pre-commit runs ruff and pyright as local hooks through uv run --locked, so hooks, CI and a developer's shell all resolve the single version pinned in uv.lock (the remote-hook revs had drifted behind the lock). CI runs with a read-only token, per-ref concurrency, job timeouts and actions pinned to full commit SHAs; uv sync --locked replaces --frozen. scripts/ and docs/scripts/ join the ruff and pyright gates - the bare 'scripts' exclude had silently covered both, leaving the CI-run decision validator unlinted; the widened rule set (SIM, PTH, PIE, RET, PERF, FURB, N; ASYNC dropped, no async code exists) surfaced nine findings, fixed in place, and pyright now reports ignore comments that suppress nothing. Dependabot watches the three pinned surfaces weekly: the uv lockfile, the actions, the Docker digests. | [`DESIGN_NOTES.md section 12`](DESIGN_NOTES.md) | - | +| [`D-027`](#D-027) | 🟢 | error-handling | Input validation extracted into its own module | 2026-09-04 | validate_inputs, the magic-bytes probe and the input-problem classification set moved byte-for-byte from merge.py into inputs.py; extract.py imports from inputs instead of reaching into merge. This removes the only import edge between the two operation modules, which the design doc presents as parallel peers, and gives the input-problem vocabulary a single home. Pure code motion: error codes, the one-failure-reports-all contract (D-012) and the context.problems log shape are unchanged. | [`DESIGN_NOTES.md section 6`](DESIGN_NOTES.md) | - | +| [`D-028`](#D-028) | 🟢 | error-handling | Error codes typed as a StrEnum with a drift-tested documentation table | 2026-09-04 | The 32 error_code string literals scattered across src/ become a single ErrorCode StrEnum in errors.py, and PdfOpsError takes error_code: ErrorCode - a typo in a code is now a pyright error instead of a silent new vocabulary entry. StrEnum serializes identically to the raw strings, so the JSON log contract is byte-identical; the untouched test assertions that parse log output and compare raw strings pin that independently. docs/OPERATIONS.md gains the complete code table grouped by exit class, and a unit test fails when the enum and the table drift in either direction, so a new code cannot ship undocumented. The remaining small string vocabularies (password_type, password source, output action) get pyright-checked Literal aliases. | [`DESIGN_NOTES.md section 2`](DESIGN_NOTES.md) | - | +| [`D-029`](#D-029) | 🟢 | project | Security policy and package metadata; future-import dropped on 3.14 | 2026-09-04 | SECURITY.md documents private vulnerability reporting with an in-scope list that maps one-to-one onto the test-pinned guarantees (attachment-name containment, the password no-leak layers, atomic outputs, no taxonomy escapes). pyproject gains project.urls and trove classifiers, including Private :: Do Not Upload so an accidental publish is refused by the index. from __future__ import annotations dropped across the tree: the project pins Python 3.14, where deferred annotation evaluation is the default, so the import was pure noise; there are no TYPE_CHECKING guards anywhere that depended on it. README's dev commands now include the format check CI enforces and the one-time pre-commit install. | [`DESIGN_NOTES.md section 13`](DESIGN_NOTES.md) | - | -**Counts:** 22 total decisions - 17 🟢 decided, 0 🟡 pending, 3 ⏸ deferred, 2 🔵 superseded. +**Counts:** 29 total decisions - 27 🟢 decided, 0 🟡 pending, 0 ⏸ deferred, 2 🔵 superseded. ### Index by area @@ -43,22 +50,22 @@ This document is the authoritative register of all architectural decisions for t | Area | Count | IDs | |---|---|---| -| error-handling | 5 | D-003, D-006, D-012, D-015, D-020 | +| error-handling | 7 | D-003, D-006, D-012, D-015, D-020, D-027, D-028 | | config | 4 | D-004, D-007, D-008, D-009 | -| pdf-engine | 4 | D-002, D-011, D-016, D-018 | -| security | 4 | D-013, D-014, D-017, D-019 | +| pdf-engine | 5 | D-002, D-011, D-016, D-018, D-023 | +| security | 5 | D-013, D-014, D-017, D-019, D-024 | | reliability | 3 | D-010, D-021, D-022 | | observability | 1 | D-005 | -| project | 1 | D-001 | -| **Total** | **22** | | +| container | 1 | D-025 | +| project | 2 | D-001, D-029 | +| infra | 1 | D-026 | +| **Total** | **29** | | ### Open decisions (🟡 Pending + ⏸ Deferred) | ID | Status | Owner / Trigger | |---|---|---| -| [`D-007`](#D-007) | ⏸ Deferred | Post-merge review, or `:`-in-path / templating friction | -| [`D-008`](#D-008) | ⏸ Deferred | Contract freeze, or mixed-case values seen in practice | -| [`D-009`](#D-009) | ⏸ Deferred | Platform injecting foreign `PDFOPS_*` vars (deployment-example watch) | +| _(none)_ | | | ### Decisions scheduled for revisit @@ -146,37 +153,34 @@ Per-decision details: status, decided date, rationale, related decisions, and th ### D-007 - **Title:** PDFOPS_INPUTS list separator -- **Status:** ⏸ Deferred +- **Status:** 🟢 Decided - **Area:** config -- **Decided on:** 2026-08-31 -- **Summary:** Deferred (2026-08-31): the merge implementation ships the recommended default - os.pathsep (colon) with explicit ordered paths, no globs - as provisional. Alternatives (comma, newline, JSON array) parked; revisit if colon-in-path or workflow-templating friction appears. +- **Decided on:** 2026-09-03 +- **Summary:** Settled (2026-09-03, deferred since 2026-08-31): os.pathsep (colon) with explicit ordered paths and no globs stands. It survived the container suite, the deployment example, and every walkthrough; colon-in-path never materialized and is pathological for mounted paths anyway (documented limitation). Alternatives (comma, newline, JSON array) rejected as churn without a driving case. - **Risk:** low - **Reversibility:** expensive -- **Phase trigger:** Post-merge review, or the first path containing `:` / workflow-templating friction with the provisional os.pathsep separator. - **Where:** [`DESIGN_NOTES.md section 3`](DESIGN_NOTES.md) ### D-008 - **Title:** Operation value case strictness -- **Status:** ⏸ Deferred +- **Status:** 🟢 Decided - **Area:** config -- **Decided on:** 2026-08-31 -- **Summary:** Deferred (2026-08-31): strict lowercase merge/extract (whitespace tolerated) stands as implemented. Case-insensitive acceptance parked; revisit at README/contract freeze or on operator feedback. +- **Decided on:** 2026-09-03 +- **Summary:** Settled (2026-09-03, deferred since 2026-08-31): strict lowercase merge/extract (whitespace tolerated) stands. Operation values come from workflow templates, not humans typing - case tolerance would add contract surface without value, and the error message already names the accepted values. - **Risk:** low - **Reversibility:** cheap -- **Phase trigger:** README/contract freeze, or operator feedback that mixed-case operation values occur in practice. - **Where:** [`DESIGN_NOTES.md section 3`](DESIGN_NOTES.md) ### D-009 - **Title:** Unknown PDFOPS_* variable hard rejection -- **Status:** ⏸ Deferred +- **Status:** 🟢 Decided - **Area:** config -- **Decided on:** 2026-08-31 -- **Summary:** Deferred (2026-08-31): hard rejection (exit 2 UNKNOWN_VAR) stands as implemented. Softening to a warning parked; revisit if a platform legitimately injects foreign PDFOPS_* vars (e.g. when the deployment example is written). +- **Decided on:** 2026-09-03 +- **Summary:** Settled (2026-09-03, deferred since 2026-08-31): hard rejection (exit 2 UNKNOWN_VAR) stands. The revisit trigger - a platform injecting foreign PDFOPS_* variables - was tested by writing the deployment example, which injects none; typo protection keeps outweighing a hypothetical soft mode. - **Risk:** low - **Reversibility:** cheap -- **Phase trigger:** A platform legitimately injecting foreign `PDFOPS_*` variables (watch when writing the deployment example). - **Where:** [`DESIGN_NOTES.md section 3`](DESIGN_NOTES.md) @@ -329,6 +333,83 @@ Per-decision details: status, decided date, rationale, related decisions, and th - **Amends:** [D-010](#D-010) - **Where:** [`DESIGN_NOTES.md section 6`](DESIGN_NOTES.md) + +### D-023 +- **Title:** Engine swapped to pikepdf; pypdf demoted to dev-dependency test oracle +- **Status:** 🟢 Decided +- **Area:** pdf-engine +- **Decided on:** 2026-09-02 +- **Summary:** engine_pikepdf.py (qpdf-backed) replaces engine_pypdf.py as the runtime engine, executing the D-002 plan: better large-file and corrupt-input behavior, native AES-256 (R=6 pinned). password_type user/owner/empty reporting survives via qpdf's password-matched flags; qpdf repairs light damage pypdf refused (pinned as behavior; warnings surface as events - at open via OpenedInput.warnings, after lazy reads/writes via collect_warnings); duplicate attachment names preserved by walking /Names/EmbeddedFiles directly with cycle and type guards (hostile shapes degrade or classify as data problems, never exit 1); the merged output is saved through the atomic layer's open temp file, keeping the single-temp cleanup contract; failed-open algorithm labels come from a best-effort raw /Encrypt scan. pypdf stays in the dev group building fixtures and verifying outputs - every test is a cross-library check. +- **Risk:** medium +- **Reversibility:** cheap +- **Where:** [`DESIGN_NOTES.md section 1`](DESIGN_NOTES.md) + + +### D-024 +- **Title:** Secrets stay stdlib, consolidated into one module +- **Status:** 🟢 Decided +- **Area:** security +- **Decided on:** 2026-09-03 +- **Summary:** All secret handling (Secret wrapper, EnvSecret/FileSecret source refs with resolve()/describe(), Secrets bundle, scrub registration) consolidated into secrets.py; config.py only parses which source is configured. Shelf options evaluated and rejected: pydantic SecretStr (compiled dependency for one masked-repr class), pydantic-settings (config-layer rewrite; secrets_dir expects field-named files in a fixed directory - a different contract from PDFOPS_PASSWORD_FILE=; ValidationError would need retranslation into the error_code taxonomy), scanner-style log redactors (pattern heuristics, weaker than the exact-value field-restricted scrub that avoids the password-oracle problem). +- **Risk:** low +- **Reversibility:** cheap +- **Where:** [`DESIGN_NOTES.md section 8`](DESIGN_NOTES.md) + + +### D-025 +- **Title:** Hardened runtime image: digest-pinned multi-stage build, read-only rootfs +- **Status:** 🟢 Decided +- **Area:** container +- **Decided on:** 2026-09-03 +- **Summary:** The image is a two-stage build: a digest-pinned uv stage resolves the lockfile into a self-contained virtualenv with compiled bytecode; the runtime stage (digest-pinned python:3.14-slim) carries only that venv, uninstalls the base image's pip, removes stdlib ensurepip (its bundled wheel would restore pip in one command), and runs as fixed non-root UID 10001. Read-only root filesystem is proven by a container test running the golden merge under --read-only --cap-drop ALL --security-opt no-new-privileges (all writes land in the output mount by design). deploy/argo-example.yaml ships the full posture incl. fsGroup, secret-mounted password, a retry expression covering exit 1 plus pod-level Error nodes (which carry no exit code), and memory sized by the measured input+128MB rule. Distroless bases considered and not taken: pinned slim minus pip reaches most of the value while staying debuggable. +- **Risk:** low +- **Reversibility:** cheap +- **Where:** [`DESIGN_NOTES.md section 11`](DESIGN_NOTES.md) + + +### D-026 +- **Title:** One uv-locked toolchain with SHA-pinned, Dependabot-watched CI +- **Status:** 🟢 Decided +- **Area:** infra +- **Decided on:** 2026-09-04 +- **Summary:** pre-commit runs ruff and pyright as local hooks through uv run --locked, so hooks, CI and a developer's shell all resolve the single version pinned in uv.lock (the remote-hook revs had drifted behind the lock). CI runs with a read-only token, per-ref concurrency, job timeouts and actions pinned to full commit SHAs; uv sync --locked replaces --frozen. scripts/ and docs/scripts/ join the ruff and pyright gates - the bare 'scripts' exclude had silently covered both, leaving the CI-run decision validator unlinted; the widened rule set (SIM, PTH, PIE, RET, PERF, FURB, N; ASYNC dropped, no async code exists) surfaced nine findings, fixed in place, and pyright now reports ignore comments that suppress nothing. Dependabot watches the three pinned surfaces weekly: the uv lockfile, the actions, the Docker digests. +- **Risk:** low +- **Reversibility:** cheap +- **Where:** [`DESIGN_NOTES.md section 12`](DESIGN_NOTES.md) + + +### D-027 +- **Title:** Input validation extracted into its own module +- **Status:** 🟢 Decided +- **Area:** error-handling +- **Decided on:** 2026-09-04 +- **Summary:** validate_inputs, the magic-bytes probe and the input-problem classification set moved byte-for-byte from merge.py into inputs.py; extract.py imports from inputs instead of reaching into merge. This removes the only import edge between the two operation modules, which the design doc presents as parallel peers, and gives the input-problem vocabulary a single home. Pure code motion: error codes, the one-failure-reports-all contract (D-012) and the context.problems log shape are unchanged. +- **Risk:** low +- **Reversibility:** cheap +- **Where:** [`DESIGN_NOTES.md section 6`](DESIGN_NOTES.md) + + +### D-028 +- **Title:** Error codes typed as a StrEnum with a drift-tested documentation table +- **Status:** 🟢 Decided +- **Area:** error-handling +- **Decided on:** 2026-09-04 +- **Summary:** The 32 error_code string literals scattered across src/ become a single ErrorCode StrEnum in errors.py, and PdfOpsError takes error_code: ErrorCode - a typo in a code is now a pyright error instead of a silent new vocabulary entry. StrEnum serializes identically to the raw strings, so the JSON log contract is byte-identical; the untouched test assertions that parse log output and compare raw strings pin that independently. docs/OPERATIONS.md gains the complete code table grouped by exit class, and a unit test fails when the enum and the table drift in either direction, so a new code cannot ship undocumented. The remaining small string vocabularies (password_type, password source, output action) get pyright-checked Literal aliases. +- **Risk:** low +- **Reversibility:** cheap +- **Where:** [`DESIGN_NOTES.md section 2`](DESIGN_NOTES.md) + + +### D-029 +- **Title:** Security policy and package metadata; future-import dropped on 3.14 +- **Status:** 🟢 Decided +- **Area:** project +- **Decided on:** 2026-09-04 +- **Summary:** SECURITY.md documents private vulnerability reporting with an in-scope list that maps one-to-one onto the test-pinned guarantees (attachment-name containment, the password no-leak layers, atomic outputs, no taxonomy escapes). pyproject gains project.urls and trove classifiers, including Private :: Do Not Upload so an accidental publish is refused by the index. from __future__ import annotations dropped across the tree: the project pins Python 3.14, where deferred annotation evaluation is the default, so the import was pure noise; there are no TYPE_CHECKING guards anywhere that depended on it. README's dev commands now include the format check CI enforces and the one-time pre-commit install. +- **Risk:** low +- **Reversibility:** cheap +- **Where:** [`DESIGN_NOTES.md section 13`](DESIGN_NOTES.md) + --- ## Architectural Decision Records (Full Analysis) diff --git a/docs/DESIGN.md b/docs/DESIGN.md index 820223e..f10bf85 100644 --- a/docs/DESIGN.md +++ b/docs/DESIGN.md @@ -33,10 +33,20 @@ Small modules with one-way dependencies: | `errors.py` | the exit-code taxonomy and `error_code` vocabulary | | `main.py` | `run(env) -> int` - the single error boundary; emits the one terminal event | | `engine.py` | the `PdfEngine` Protocol - the library swap seam | -| `engine_pypdf.py` | the **only** module importing pypdf; translates its failure modes into the taxonomy | +| `engine_pikepdf.py` | the **only** module importing pikepdf; translates qpdf's failure modes into the taxonomy | +| `inputs.py` | up-front input validation shared by both operations; every bad input reported in one failure | | `merge.py` / `extract.py` | orchestration: validate everything, then write | | `output.py` | atomic writes, existing-output policy, stale-temp cleanup | -| `logging_setup.py` / `secret.py` | JSON formatter, secret scrubbing, third-party log routing; the `Secret` wrapper | +| `secrets.py` | the whole secret lifecycle: `Secret` wrapper, source refs, resolution, scrub registration | +| `logging_setup.py` | JSON formatter, secret scrubbing, third-party log routing | + +An interactive map of these modules - each node linking to the source lines it +describes - lives at [`diagrams/index.html`](diagrams/index.html#architecture), +alongside diagrams of the retry lifecycle, the extract trust boundary, and the +password flow, all navigable from one page. The same structure drawn to render on +GitHub - deployment context, module graph, run sequence, failure taxonomy, trust +boundary, retry machine, test oracle, delivery pipeline - is +[`ARCHITECTURE.md`](ARCHITECTURE.md). Cross-cutting rules: unknown or operation-inapplicable `PDFOPS_*` variables are hard errors - a silently ignored misspelling becomes a confusing downstream failure @@ -47,23 +57,25 @@ one), fsynced, then renamed: the final path holds a complete file or nothing. ## PDF library -**pypdf today, pikepdf next, behind the seam ([D-002](DECISIONS.md#D-002)).** pypdf -(BSD-3) is pure Python with the richest attachments API - list-valued, faithful to -duplicate names - and reports which password matched; its weaknesses are large-file -memory behavior and no repair path for damaged files. pikepdf (MPL-2.0, qpdf-backed, -self-contained wheels) is the stronger production engine on exactly those axes, but its -attachments mapping collapses duplicate names, so the swap walks the -`/Names/EmbeddedFiles` name tree directly. PyMuPDF was rejected on licensing, not -capability: shipping an AGPL container image as a workflow step is an exposure this -project does not accept. Because all library exceptions are translated in one module, -the swap is a single-module change, and pypdf then stays as a dev-dependency -cross-checking oracle in the tests. - -Three pypdf traps are handled explicitly: it leaks *builtin* exceptions on pathological -files (classified as `CORRUPT_PDF`, not an internal error, so deterministic bad inputs -don't look retryable); `decrypt()` reports failure through its return value, not an -exception; and its write-side encryption defaults to legacy RC4 - AES-256 is passed -explicitly, so output encryption never downgrades. +**pikepdf in production, pypdf as the test oracle ([D-002](DECISIONS.md#D-002), +[D-023](DECISIONS.md#D-023)).** The first iterations ran on pypdf (BSD-3, pure Python - +the fastest start); once the seam and its tests hardened, the engine was swapped to +pikepdf (MPL-2.0, C++ qpdf backend, self-contained wheels), which is stronger on +exactly the production axes: large-file memory behavior, corrupt-input robustness, and +native AES-256. The swap touched one module - the point of the seam - and the whole +suite passed against the new engine with only the corruption fixtures adapted, because +qpdf *repairs* light damage (truncation, a mangled xref) that pypdf rejected; those +repairs surface as warning events rather than being silently absorbed. pypdf remains a +dev-dependency building test fixtures and independently verifying outputs, so every +green test is implicitly a two-library cross-check. PyMuPDF was rejected on licensing, +not capability: shipping an AGPL container image as a workflow step is an exposure +this project does not accept. + +qpdf specifics handled explicitly: its attachments mapping collapses duplicate names, +so extraction walks the `/Names/EmbeddedFiles` name tree directly (cycle-guarded - +hostile trees can self-reference); parse warnings arrive through qpdf's own channel, +not Python logging, and are carried per input into the JSON event stream; and output +encryption pins `R=6` so the result is always AES-256, never a legacy scheme. ## Operations and attachment security @@ -139,9 +151,13 @@ candidate fixes. (2) One password serves all merge inputs; a per-input map is th extension. (3) The extracted *set* is not transactional - files are individually atomic, so a staging-directory handoff is the fix if a consumer needs all-or-nothing. (4) `skip` trusts an existing file as completed prior output; a checksum-verified skip -is future work. (5) Resource behavior is unmeasured - the pikepdf swap comes with -generated large-file benchmarks feeding real K8s requests/limits guidance. (6) An -OOM-kill is a SIGKILL no in-process error boundary can catch; the workflow engine -reports it itself - documented rather than handled. Next in line beyond the -swap: container hardening (digest-pinned base, verified read-only rootfs, a shipped -Argo `WorkflowTemplate` example) and merge bookmark/metadata carry-over. +is future work. (5) Memory scales linearly with total input size: measured in-container +(`scripts/benchmark.py`), peak process RSS is about total input bytes plus ~40 MB of +fixed overhead, independent of file count - a 500 MB merge peaks near 530 MB, so +multi-gigabyte merges need matching memory; sizing guidance is in +[`OPERATIONS.md`](OPERATIONS.md). (6) An OOM-kill is a SIGKILL no in-process error +boundary can catch; the workflow engine reports it itself - documented rather than +handled. The runtime image is a digest-pinned multi-stage build with no package +installer, verified by a container test to run with a read-only root filesystem and +no capabilities; `deploy/argo-example.yaml` ships the full posture. Next in line: +merge bookmark/metadata carry-over. diff --git a/docs/DESIGN_NOTES.md b/docs/DESIGN_NOTES.md index 78df0b9..49f5870 100644 --- a/docs/DESIGN_NOTES.md +++ b/docs/DESIGN_NOTES.md @@ -34,7 +34,32 @@ for the resource-behavior and corrupt-input quality dimensions. Keeping the engi one seam (`engine.py`) makes the swap a single-module change and forces library exceptions to be translated into the application taxonomy in exactly one place. -## 2. Exit-code taxonomy (per [D-003](DECISIONS.md#D-003)) +**Swap executed ([D-023](DECISIONS.md#D-023)):** `engine_pikepdf.py` replaced +`engine_pypdf.py`; pypdf moved to the dev group, where it still builds every test +fixture and independently verifies outputs - each green test is a two-library +cross-check. Realities found at swap time: qpdf exposes which document password a +supplied string matched (`user_password_matched`/`owner_password_matched`), so the +`user`/`owner`/`empty` reporting survived intact; qpdf *repairs* light damage +(truncation, mangled xref) that pypdf refused, so the corrupt-input fixtures had to +become genuinely unrecoverable and the repair behavior is pinned as its own test with +warnings surfaced as `pdf_library_message` events (qpdf reports through its own +channel, not Python logging: open-time warnings ride on `OpenedInput.warnings`, and +because qpdf reads stream data lazily, repairs discovered during the write or during +attachment reads are drained afterwards via the engine's `collect_warnings`); the +merged output is saved through the atomic-write layer's already-open temp file rather +than by path - handed a path to an existing file, pikepdf would route through its own +second hidden temp, whose debris after a kill no cleanup would ever match; qpdf hands +malformed structures over as native Python values, so the walks guard node types and +keep a builtin-exception net (a hostile name tree must classify as a data problem, +never as a retryable internal error, and an integer name-tree key must not become an +attacker-sized `bytes(n)` allocation); when an open fails +outright there is no handle to read encryption facts from, so the algorithm label on +`PASSWORD_REQUIRED`/`WRONG_PASSWORD` errors comes from a best-effort raw scan of the +plaintext `/Encrypt` dictionary; and duplicate-name fidelity required walking +`/Names/EmbeddedFiles` directly (cycle-guarded) because `Pdf.attachments` is a +Mapping, exactly as predicted in section 1. + +## 2. Exit-code taxonomy (per [D-003](DECISIONS.md#D-003), [D-028](DECISIONS.md#D-028)) The process exit code is the application's external API toward the workflow engine. Classes, not fine-grained codes - workflow engines branch on codes, and codes are a scarce, stable @@ -60,6 +85,14 @@ dedicated `10+` transient band was resolved with the retry-semantics work deterministic, and retryability is better expressed as documentation the operator composes (README's per-code table + retryStrategy expression) than as more codes. +**Typed vocabulary ([D-028](DECISIONS.md#D-028)):** the fine-grained `error_code` +tokens started as string literals at each raise site. They are now a single +`ErrorCode` StrEnum, so the complete vocabulary is readable in one place and a +typo is a type error rather than a silent new code. StrEnum members serialize +exactly like the raw strings, so nothing changes on the wire; the log-parsing +tests that compare raw strings pin that independently. The full table, grouped +by exit class, lives in OPERATIONS.md with a drift test against the enum. + ## 3. Environment-variable contract (per [D-004](DECISIONS.md#D-004)) Conventions: @@ -78,7 +111,7 @@ Conventions: - Missing and empty values are treated identically (`MISSING_VAR`) - an empty value almost always means a broken template substitution upstream. - `PDFOPS_INPUTS` is an **explicit ordered list** (`os.pathsep`-separated - the `PATH` - convention; see deferred [D-007](DECISIONS.md#D-007)). No glob support, deliberately: + convention; per [D-007](DECISIONS.md#D-007)). No glob support, deliberately: merge order must be explicit, not lexicographic luck, or retries and re-runs can produce different documents. Duplicates are rejected (`DUPLICATE_INPUTS`). A single input is allowed - workflows fan in variable-length lists that can be of length one. @@ -147,7 +180,7 @@ copies `/Names/EmbeddedFiles` on a page-level merge, so attachments in merge inp silently dropped - options (detect-and-warn, fail-loud flag, qpdf's `--copy-attachments-from`) are evaluated when attachment handling is built out. -## 6. Input validation and atomic output (per [D-010](DECISIONS.md#D-010), [D-012](DECISIONS.md#D-012)) +## 6. Input validation and atomic output (per [D-010](DECISIONS.md#D-010), [D-012](DECISIONS.md#D-012), [D-027](DECISIONS.md#D-027)) **Collect-all validation ([D-012](DECISIONS.md#D-012)):** every input is checked up front (exists, is a file, readable, starts with `%PDF-`) and *all* problems are reported in one @@ -155,6 +188,13 @@ failure event - an operator fixing a broken workflow learns about every bad inpu single run, not one per retry. The exit class follows the first problem in input order (deterministic); the full list travels in `context.problems`. +**One home for the check ([D-027](DECISIONS.md#D-027)):** the validation originally +lived in `merge.py` with `extract.py` importing it from there - the only import edge +between two modules the design doc presents as parallel peers. It moved byte-for-byte +into `inputs.py`, so both operations depend on the shared module instead of one +depending on the other. Pure code motion: error codes, the collect-all contract and +the `context.problems` shape are unchanged. + **Atomic output ([D-010](DECISIONS.md#D-010)):** all output is written to a temp file created *in the destination directory* - same filesystem, because `os.replace` is only atomic within one - then fsynced and renamed over the final path in a single step (plus a @@ -215,6 +255,15 @@ rather than half-supported. ## 8. Passwords and output encryption (per [D-017](DECISIONS.md#D-017), [D-018](DECISIONS.md#D-018), [D-019](DECISIONS.md#D-019)) +**Where the code lives ([D-024](DECISIONS.md#D-024)):** the whole lifecycle - wrapper, +source refs, resolution, scrub registration - sits in one module (`secrets.py`) after +an evaluation of the shelf options: pydantic's `SecretStr` buys one masked-repr class +for a compiled dependency; pydantic-settings would rewrite the config layer, and its +`secrets_dir` convention (field-named files in a fixed directory) is a different +contract from `PDFOPS_PASSWORD_FILE=`; scanner-style log redactors +are pattern-heuristic - strictly weaker than the exact-value, field-restricted scrub +below, which exists precisely because naive scrubbing becomes a password oracle. + **Channels ([D-017](DECISIONS.md#D-017)):** one password, two mutually exclusive sources - `PDFOPS_PASSWORD_FILE` (a mounted secret, the preferred channel: it never appears in pod specs, `kubectl describe`, or `/proc//environ`) and `PDFOPS_PASSWORD` (kept for local @@ -323,3 +372,125 @@ another target's temps - and a run's own planned outputs are structurally exclud cleanup, which happens entirely before the first write. Merge's `skip` short-circuit reads nothing at all: not the inputs, and not the mounted password file (secrets resolve lazily, after the skip decision). + +## 10. Resource behavior (measured 2026-09-03) + +Method: `scripts/benchmark.py` generates large fixtures (incompressible random +bytes in uncompressed streams, so sizes are honest; never committed) and runs +each scenario through the real container image under a small wrapper that +reports both the operation process's peak RSS (`ru_maxrss`) and the cgroup's +`memory.peak` - the number Kubernetes actually meters, which additionally +counts the reclaimable page cache the run touched. Environment: Docker Desktop +on an Apple Silicon host; durations are indicative, the memory profile is +structural. + +| Scenario | Duration | Peak process RSS | Peak cgroup memory | +|---|---|---|---| +| merge 2 x 5 MB (baseline) | 0.05 s | ~36 MB | ~51 MB | +| merge 2 x 250 MB | 1.8 s | ~529 MB | ~1519 MB | +| merge 20 x 25 MB | 1.2 s | ~531 MB | ~1515 MB | +| merge 250 MB AES-256 in, re-encrypted out | 2.8 s | ~281 MB | ~779 MB | +| extract 10 x 25 MB attachments | 0.5 s | ~333 MB | ~579 MB | + +Findings, in order of consequence: + +- **Peak process memory is linear in total input bytes** - roughly total input + size plus ~40 MB of fixed interpreter/library overhead - and indifferent to + how those bytes are split across files (2 x 250 MB and 20 x 25 MB profile + identically). The writer holds the copied stream data until `save()` + completes, so a merge effectively buffers one output's worth of content. + Multi-gigabyte merges therefore need matching memory; a streaming rewrite is + future work, noted rather than planned. +- **Crypto costs time, not memory**: decrypt + AES-256 re-encrypt of 250 MB + added about a second and peaked at input size like any other run. +- **The cgroup peak runs ~2-3x the process RSS** purely from page cache + (inputs read plus output written). Cache is reclaimable under memory + pressure, so limits do not need to cover it - the sizing rule in + `OPERATIONS.md` (total expected input + 128 MB) is set against RSS. +- OOM behavior is unchanged from section 2's taxonomy stance: exceeding the + limit is a SIGKILL, no terminal event, the workflow engine reports it. + +## 11. Container hardening (2026-09-03) + +The image became production-shaped in one pass: + +- **Multi-stage build**: a uv build stage (pinned by digest) resolves the + lockfile into a self-contained virtualenv with compiled bytecode; the + runtime stage copies only that venv onto a digest-pinned `python:3.14-slim` + base. No uv, no caches, and no package installer ship - the base image's + own pip is uninstalled AND stdlib `ensurepip` is removed (its bundled + wheel would restore pip in one command), with a container test probing + both interpreters (the venv python cannot see base site-packages, so a + venv-only probe would be vacuous). +- **Read-only root filesystem, proven not promised**: all writes go to the + output mount by design (temp files live in the destination directory for + rename atomicity, and pikepdf saves through our open handle), so the image + runs under `--read-only --cap-drop ALL --security-opt no-new-privileges` + unmodified - a container test executes the golden merge under exactly that + posture. +- **`deploy/argo-example.yaml`**: a WorkflowTemplate wiring together + everything the docs describe - the security context (non-root 10001, + read-only rootfs, no capabilities, fsGroup for output writability), the + password as a mounted secret file, a retry expression covering exit 1 and + pod-level errors (an Error-phase node never produces an exit code - Argo + substitutes "-1" - so exit-code-only expressions silently skip the + lost-pod case they are usually written for) paired with + `PDFOPS_ON_EXISTS=skip`, and memory sized by the measured rule from + section 10. + +Considered and not taken: distroless/static bases (the pinned slim base with +pip removed reaches most of the value while keeping a debuggable Python +layout); image signing/SBOM (registry- and org-specific, noted as release +engineering rather than image structure). + +## 12. Toolchain and CI pinning (per [D-026](DECISIONS.md#D-026)) + +The repo had three places that could each pick their own tool versions: the +pre-commit hook pins, uv.lock, and whatever a developer's shell resolved. +They had already drifted - the hooks pinned older ruff and pyright releases +than the lockfile. The fix is structural rather than a version bump: the +Python tools run as local pre-commit hooks through `uv run --locked`, so +hooks, CI and a developer's shell all execute the single version pinned in +uv.lock. Only the generic hygiene hooks (whitespace, YAML/TOML syntax, +large files) still come from a remote hook repo. + +Two related gaps closed in the same pass: + +- **The gates now cover every script.** The bare `scripts` entry in the + ruff exclude list silently matched both `scripts/` and `docs/scripts/`, + leaving the decision-register validator - which CI executes on every + push - unlinted and untyped. Both directories joined the ruff and pyright + gates; the widened rule set (SIM, PTH, PIE, RET, PERF, FURB, N added; + ASYNC dropped because no async code exists) surfaced nine findings, all + fixed with behavior-preserving edits. pyright's + `reportUnnecessaryTypeIgnoreComment` keeps ignore comments honest; four + that suppressed nothing were removed. +- **CI itself is pinned and least-privilege.** Actions are pinned to full + commit SHAs (a tag can be moved; a SHA cannot), the workflow token is + read-only, runs on the same ref supersede each other, and jobs carry + timeouts. `uv sync --locked` verifies the lockfile matches pyproject + instead of silently trusting it. + +Pinning everything creates a staleness problem, so Dependabot watches the +three pinned surfaces weekly: the uv lockfile (dev tooling grouped into one +PR), the action SHAs, and the Docker base-image digests. + +## 13. Repository housekeeping (per [D-029](DECISIONS.md#D-029)) + +Three small decisions that make the repository read correctly from the +outside: + +- **SECURITY.md**: private vulnerability reporting, with an in-scope list + that is a one-to-one summary of the guarantees the test suite pins - + attachment-name containment, the password no-leak layers, whole-or-absent + outputs, and the rule that a hostile input must classify as a data + problem, never as a retryable internal error. +- **Package metadata**: `project.urls` and trove classifiers in pyproject, + led by `Private :: Do Not Upload` - the package is a container payload, + not a library, and the index refuses that classifier if anyone ever runs + a publish by mistake. +- **No `from __future__ import annotations`**: the project pins Python + 3.14, where deferred annotation evaluation (PEP 649) is the default. The + import survived from habit in every module; dropping it removes a + visible signal of code written for older interpreters. No TYPE_CHECKING + guard anywhere depended on it. diff --git a/docs/OPERATIONS.md b/docs/OPERATIONS.md index 934a111..4f7fc31 100644 --- a/docs/OPERATIONS.md +++ b/docs/OPERATIONS.md @@ -15,6 +15,10 @@ output policies, and what the log stream carries. The short reference tables - Outputs get regular `open()`-style permissions (the container umask, `0644` by default), so a later step running as a different user can read them from a shared volume. +- The image needs no writable root filesystem: every write lands in the output + mount, so `readOnlyRootFilesystem: true` (plus dropped capabilities and no + privilege escalation) works unmodified - a container test runs the golden merge + under exactly that posture. ## Passwords @@ -23,15 +27,18 @@ output policies, and what the log stream carries. The short reference tables the spec-standard empty-password try, exactly like every PDF viewer. Wrong password -> exit 5 naming the failing input. - The password itself never appears in any output: the in-process `Secret` wrapper - renders as `***`, the logging layer scrubs registered secret values from every - event including tracebacks, and the process scrubs `PDFOPS_PASSWORD` from its own - environment on startup. + renders as `***`, the logging layer scrubs registered secret values (four + characters or longer; a shorter one draws a `redaction_degraded` warning) from the + free-text fields of every event including tracebacks, and the process scrubs + `PDFOPS_PASSWORD` and `PDFOPS_OUTPUT_PASSWORD` from its own environment on startup. - Each `input_opened` event reports the encryption algorithm (read from the PDF's plaintext `/Encrypt` dictionary) and how the file opened (`user`/`owner`/`empty`). - A permissions-locked input among user-locked ones never fails just because a - password was supplied: the empty try still applies per input. + password was supplied: the empty try still applies per input (the exact call + sequence is drawn in [`diagrams/index.html#passwords`](diagrams/index.html#passwords)). - Passwords containing control characters are rejected (exit 2) as encoding - accidents. + accidents, at the moment the password is resolved - a merge that short-circuits + on `skip` never reads it. - Note the env channel's inherent limit: the initial environment block stays visible to `docker inspect` and `/proc//environ` - the file channel is the one that keeps the value out of the process's environment entirely. @@ -45,7 +52,11 @@ this step; `always` encrypts unconditionally. The output password comes from `PDFOPS_OUTPUT_PASSWORD_FILE`/`PDFOPS_OUTPUT_PASSWORD`, falling back to the *explicitly supplied* input password (never the empty auto-try). Output encryption is always AES-256, whatever the inputs used. Supplying an output password while the mode -is `never` is a hard configuration error. +is `never` is a hard configuration error. With no password available anywhere +(`always` at config parse, or `inherit` when the only encrypted inputs opened via the +empty try) the run fails with `MISSING_OUTPUT_PASSWORD`, exit 2, before anything is +written. `security_downgrade` is a warning-level event, so `PDFOPS_LOG_LEVEL=error` +hides it; the terminal event's `output_encrypted: false` is the level-proof signal. ## Atomic writes and existing outputs @@ -62,9 +73,13 @@ is `never` is a hard configuration error. are written (`attachments_skipped` reports the rest), so a crashed run's partial set gets finished by the retry - sound because every file this tool writes is atomic and therefore whole. Both `skip` modes trust that an existing file is a - completed prior output. + completed prior output. A directory at the output path, or at any extraction + target name, is refused under every policy (`OUTPUT_IS_DIRECTORY`, exit 6). - Temp debris from a crashed prior run (`.name.*.tmp` matching this run's own - targets) is removed at startup with a `stale_temp_removed` event. One writer per + targets) is removed before the first write, with a `stale_temp_removed` event; a + run refused or failed before that point leaves it for the next attempt. The whole + run/retry state machine is drawn in + [`diagrams/index.html#lifecycle`](diagrams/index.html#lifecycle). One writer per output path at a time is assumed - which a workflow engine guarantees per step. - All inputs are validated (existence, readability, PDF header) **before anything is written**, and every bad input is reported in a single failure event. @@ -74,27 +89,107 @@ is `never` is a hard configuration error. - **Attachment names are treated as untrusted input**: extraction reduces every name to a sanitized basename (path separators, traversal segments, and control characters removed; deterministic `attachment_` fallback), so a hostile PDF can - never write outside `PDFOPS_OUTPUT_DIR`. Duplicate names get deterministic - `-1`/`-2` suffixes; the original name is logged whenever sanitization changed it. + never write outside `PDFOPS_OUTPUT_DIR`. Names longer than 200 bytes are + truncated. Duplicate names get deterministic `-1`/`-2` suffixes, with collisions + detected case-insensitively (the volume may be), so `Report.txt` and `report.txt` + become `Report.txt` and `report-1.txt` on every filesystem; the original name is + logged whenever sanitization changed it. +- The path every untrusted name travels is drawn in + [`diagrams/index.html#extract`](diagrams/index.html#extract). - Extraction order is the PDF's name-tree order - deterministic across runs. Each file is written atomically; under the default `fail` policy any pre-existing file (or symlink) at a target name refuses the whole run **before** anything is written (exit 6) - see `PDFOPS_ON_EXISTS` above for the retry-friendly modes. A directory at an output path is refused under every policy (`OUTPUT_IS_DIRECTORY`). A PDF with zero attachments is a success with `attachments_extracted=0` unless - `PDFOPS_FAIL_ON_NO_ATTACHMENTS` is set. + `PDFOPS_FAIL_ON_NO_ATTACHMENTS=true` (exit 3, `NO_ATTACHMENTS`). + +## Resource sizing + +Measured in-container (`scripts/benchmark.py`, Docker Desktop on an Apple Silicon +host - durations are indicative, the memory profile is structural): + +| Scenario | Duration | Peak process RSS | Peak cgroup memory | +|---|---|---|---| +| merge 2 x 5 MB (baseline) | 0.05 s | ~36 MB | ~51 MB | +| merge 2 x 250 MB | 1.8 s | ~529 MB | ~1519 MB | +| merge 20 x 25 MB | 1.2 s | ~531 MB | ~1515 MB | +| merge 250 MB AES-256 in, re-encrypted out | 2.8 s | ~281 MB | ~779 MB | +| extract 10 x 25 MB attachments | 0.5 s | ~333 MB | ~579 MB | + +The rule of thumb: **peak process memory is roughly the total input size plus +~40 MB of fixed overhead**, and it depends on total bytes, not file count. +Decrypt/re-encrypt adds CPU time (about a second per 250 MB here), not memory. +The cgroup number runs higher because it also counts the page cache the run +touched; that cache is reclaimable, so it does not need to fit under a memory +limit. Practical Kubernetes guidance: set `requests.memory` and +`limits.memory` to about **total expected input size + 128 MB**. A run that +exceeds the limit is OOM-killed (SIGKILL) - no terminal event is emitted, and +the workflow engine reports the kill itself. ## Log events -One JSON object per line on stdout; stderr stays empty. Lifecycle events narrate -progress at their log levels (`config_loaded`, `input_opened`, `merge_written`, -`attachment_extracted`, `stale_temp_removed`, `pdf_library_message` for damage the -PDF engine repaired, `security_downgrade`, `password_unused`, ...). The terminal -event is never suppressed by `PDFOPS_LOG_LEVEL`: - -- `operation_complete` - merge: `pages`, `bytes_written`, `output_path`, - `output_encrypted`; extract: `attachments_extracted`, `bytes_written`, plus - `attachments_skipped` under `skip`. Always: `exit_code`, `duration_s`. -- `operation_failed` - `error_code` (machine-readable, finer-grained than the exit - code), `error_message`, `exit_code`, `context` (e.g. the failing input), and a - `traceback` for unexpected errors. +One JSON object per line on stdout; stderr stays empty. Every event carries `ts`, +`level` and `event`; the other fields depend on the event. Lifecycle events respect +`PDFOPS_LOG_LEVEL`; the two terminal events never do. This is the complete +vocabulary: a test checks it against every event the code emits. + +| Event | Level | When, and what it carries | +|---|---|---| +| `config_loaded` | info | the parsed configuration: `operation`, `log_level`, `on_exists`, and `password` as presence only (`unset` / `set(env)` / `set(file)`), never values; merge runs also carry `output_encryption` and `output_password` (presence only) | +| `operation_started` | info | dispatch into merge or extract | +| `input_opened` | info | per input: `input`, `pages`, `encrypted`, `algorithm`, `password_type` (`user` / `owner` / `empty`) | +| `pdf_library_message` | warning, or the library's own higher level | `detail` and `source`: damage the engine repaired (`source: qpdf`), or anything the PDF library or a Python warning routes through logging - those records bypass `PDFOPS_LOG_LEVEL` | +| `password_unused` | warning | a password was supplied but no input needed it | +| `security_downgrade` | warning | encrypted inputs merged into a plaintext output under `never`: `encrypted_inputs` | +| `redaction_degraded` | warning | a supplied secret is shorter than four characters and is not scrubbed from free text | +| `stale_temp_removed` | warning | a prior run's temp debris for this target was removed: `temp_file` | +| `output_skipped` | info | merge under `skip`: the existing output is accepted as completed work, `output_path` | +| `output_overwritten` | info | after the write: `output_path` for merge, `replaced` and `count` for extract | +| `output_encrypted` | info | the merged output was encrypted: `algorithm`, `password_source` (`output` / `input-fallback`) | +| `merge_written` | info | `output_path`, `pages_per_input`, `output_encrypted` | +| `attachments_skipped` | info | extract under `skip`: `skipped` (names), `count` | +| `attachment_extracted` | info | per file: `attachment`, `bytes`, and `original_name` - the raw document name (capped at 200 characters) when sanitization or dedupe renamed the file, `null` otherwise | +| `operation_complete` | info | terminal, exit 0. Always `operation`, `exit_code`, `duration_s`. Merge adds `inputs_merged`, `pages`, `bytes_written`, `output_path`, `output_encrypted`, or under `skip` only `skipped: true` and `output_path`. Extract adds `attachments_extracted`, `bytes_written`, and `attachments_skipped` when any file was skipped | +| `operation_failed` | error | terminal, exit 1-6. Always `error_code`, `exit_code`, `duration_s`. Predictable failures add `error_message` and `context`; an unexpected error (exit 1) adds `exc_type` and `traceback` instead | + +### Error codes + +`error_code` values, grouped by the exit code they travel with. This is the +complete vocabulary: a test checks the table against the `ErrorCode` enum in +`errors.py`, so a new code cannot ship undocumented. + +| Code | Exit | Meaning | +|---|---|---| +| `UNEXPECTED_ERROR` | 1 | an internal error; the event carries `exc_type` and `traceback` | +| `UNKNOWN_VAR` | 2 | a `PDFOPS_*` variable the application does not understand (probable typo) | +| `INAPPLICABLE_VAR` | 2 | a variable that belongs to the other operation | +| `MISSING_VAR` | 2 | a required variable is unset or empty | +| `INVALID_OPERATION` | 2 | `PDFOPS_OPERATION` is not `merge` or `extract` | +| `INVALID_LOG_LEVEL` | 2 | `PDFOPS_LOG_LEVEL` is not one of the accepted levels | +| `INVALID_INPUTS` | 2 | `PDFOPS_INPUTS` has an empty path component | +| `DUPLICATE_INPUTS` | 2 | `PDFOPS_INPUTS` lists the same path more than once | +| `INVALID_FLAG` | 2 | a boolean variable is not `true` or `false` | +| `INVALID_ON_EXISTS` | 2 | `PDFOPS_ON_EXISTS` is not `fail`, `overwrite` or `skip` | +| `INVALID_OUTPUT_ENCRYPTION` | 2 | `PDFOPS_OUTPUT_ENCRYPTION` is not `never`, `inherit` or `always` | +| `CONFLICTING_PASSWORD_SOURCES` | 2 | both the value and the file channel of one password are set | +| `OUTPUT_PASSWORD_WITHOUT_ENCRYPTION` | 2 | an output password is supplied while output encryption is `never` | +| `MISSING_OUTPUT_PASSWORD` | 2 | output encryption is required but no explicit password is available | +| `PASSWORD_FILE_UNREADABLE` | 2 | the password file cannot be read or is not UTF-8 text | +| `EMPTY_PASSWORD` | 2 | the password file is empty | +| `PASSWORD_UNSUPPORTED_CHARACTERS` | 2 | the password contains control characters | +| `INPUT_MISSING` | 3 | an input path does not exist or is not a regular file | +| `INPUT_IS_DIRECTORY` | 3 | an input path is a directory | +| `INPUT_UNREADABLE` | 3 | an input file cannot be opened for reading | +| `NO_ATTACHMENTS` | 3 | the PDF has no attachments and `PDFOPS_FAIL_ON_NO_ATTACHMENTS=true` | +| `NOT_A_PDF` | 4 | an input does not start with the `%PDF-` header | +| `CORRUPT_PDF` | 4 | the PDF engine cannot parse or fully read the file | +| `UNSUPPORTED_PDF_FEATURE` | 4 | the file uses a stream filter this build cannot decode | +| `PASSWORD_REQUIRED` | 5 | the input is encrypted and no password was supplied | +| `WRONG_PASSWORD` | 5 | the supplied password does not open the input | +| `UNSUPPORTED_ENCRYPTION` | 5 | the input uses an encryption scheme this build cannot process | +| `OUTPUT_DIR_MISSING` | 6 | the output directory does not exist | +| `OUTPUT_IS_DIRECTORY` | 6 | an output path is a directory | +| `OUTPUT_EXISTS` | 6 | an output exists and `PDFOPS_ON_EXISTS` is `fail` | +| `OUTPUT_NOT_WRITABLE` | 6 | the output location is not writable | +| `DISK_FULL` | 6 | no space left on device while writing | diff --git a/docs/diagrams/extract-trust.dataflow.json b/docs/diagrams/extract-trust.dataflow.json new file mode 100644 index 0000000..772ea4d --- /dev/null +++ b/docs/diagrams/extract-trust.dataflow.json @@ -0,0 +1,216 @@ +{ + "schema_version": 1, + "diagram_type": "dataflow", + "meta": { + "title": "Extract: Untrusted Names to Safe Files", + "quality_profile": "showcase", + "views": [ + { + "id": "name-path", + "label": "The name gauntlet", + "focus": [ + "carrier", + "treewalk", + "sanitizer", + "planner", + "containment" + ], + "note": "Every attachment name is attacker-controlled until the sanitizer and plan are done with it." + }, + { + "id": "payload-path", + "label": "The payload path", + "focus": [ + "treewalk", + "writer" + ], + "note": "Payload bytes are opaque content - they are stored, never interpreted as paths." + }, + { + "id": "landing", + "label": "Atomic landing", + "focus": [ + "containment", + "writer" + ], + "note": "Only verified targets reach the atomic write into the mounted directory." + } + ], + "viewBox": [ + 1080, + 500 + ] + }, + "stages": [ + { + "label": "Untrusted source" + }, + { + "label": "Engine walk" + }, + { + "label": "Sanitize" + }, + { + "label": "Plan" + }, + { + "label": "Write" + } + ], + "nodes": [ + { + "id": "carrier", + "type": "external", + "label": "Carrier PDF", + "sublabel": "attacker-controlled", + "stage": 0, + "row": 1, + "tag": "hostile" + }, + { + "id": "treewalk", + "type": "backend", + "label": "Name-tree walk", + "sublabel": "duplicates preserved", + "stage": 1, + "row": 1, + "tag": "cycle-guarded" + }, + { + "id": "sanitizer", + "type": "security", + "label": "Name sanitizer", + "sublabel": "pure function", + "stage": 2, + "row": 0, + "tag": "basename only" + }, + { + "id": "planner", + "type": "backend", + "label": "Target plan", + "sublabel": "casefolded dedupe", + "stage": 3, + "row": 0, + "tag": "-1/-2 suffixes" + }, + { + "id": "containment", + "type": "security", + "label": "Containment recheck", + "sublabel": "parent must be the mount", + "stage": 3, + "row": 2, + "tag": "last line" + }, + { + "id": "writer", + "type": "database", + "label": "Atomic write", + "sublabel": "into PDFOPS_OUTPUT_DIR", + "stage": 4, + "row": 1, + "tag": "temp + rename" + } + ], + "flows": [ + { + "id": "raw-tree", + "from": "carrier", + "to": "treewalk", + "label": "raw name tree", + "classification": "untrusted structure", + "variant": "security" + }, + { + "id": "raw-names", + "from": "treewalk", + "to": "sanitizer", + "label": "attachment names", + "classification": "untrusted strings", + "variant": "security" + }, + { + "id": "payload-bytes", + "from": "treewalk", + "to": "writer", + "label": "payload bytes", + "classification": "opaque content", + "variant": "dashed", + "fromSide": "bottom", + "toSide": "bottom", + "via": [ + [ + 315, + 430 + ], + [ + 960, + 430 + ] + ] + }, + { + "id": "safe-names", + "from": "sanitizer", + "to": "planner", + "label": "safe basenames", + "classification": "fallback for garbage", + "variant": "emphasis", + "labelAt": [ + 638, + 200 + ] + }, + { + "id": "planned-targets", + "from": "planner", + "to": "containment", + "label": "planned targets", + "classification": "per-volume unique", + "variant": "default", + "labelAt": [ + 745, + 200 + ] + }, + { + "id": "verified-paths", + "from": "containment", + "to": "writer", + "label": "verified paths", + "classification": "inside the mount", + "variant": "emphasis" + } + ], + "cards": [ + { + "dot": "rose", + "title": "Trust boundary", + "items": [ + "A name like ../../evil.txt would be a write-anywhere primitive", + "Sanitizing always beats rejecting: renaming is lossless and logged", + "The containment recheck guards against sanitizer regressions" + ] + }, + { + "dot": "emerald", + "title": "Determinism", + "items": [ + "Extraction order is the PDF name-tree order", + "Duplicate names get deterministic -1/-2 suffixes", + "Collisions are detected on casefolded names for case-insensitive volumes" + ] + }, + { + "dot": "cyan", + "title": "Payloads", + "items": [ + "Payload bytes never pass through path handling", + "Each file is written atomically: whole or absent", + "A non-string name key still extracts under a fallback name" + ] + } + ] +} diff --git a/docs/diagrams/extract-trust.html b/docs/diagrams/extract-trust.html new file mode 100644 index 0000000..d0acd71 --- /dev/null +++ b/docs/diagrams/extract-trust.html @@ -0,0 +1,14817 @@ + + + + + + + Extract: Untrusted Names to Safe Files Diagram + + + + + + + + + + + +
+ +
+
+
+

Extract: Untrusted Names to Safe Files

+
+
+ + + + + + + +
+ + Extract: Untrusted Names to Safe Files + A data-flow diagram generated by Archify. + + + + + + + + + + + + + + + + + + + + + + + + + 01 / Untrusted source + + + 02 / Engine walk + + + 03 / Sanitize + + + 04 / Plan + + + 05 / Write + + + + + + + + + + + + Carrier PDF · attacker-controlled · 01 / Untrusted source · hostile + + + + Carrier PDF + attacker-controlled + hostile + + + + Name-tree walk · duplicates preserved · 02 / Engine walk · cycle-guarded + + + + Name-tree walk + duplicates preserved + cycle-guarded + + + + Name sanitizer · pure function · 03 / Sanitize · basename only + + + + Name sanitizer + pure function + basename only + + + + Target plan · casefolded dedupe · 04 / Plan · -1/-2 suffixes + + + + Target plan + casefolded dedupe + -1/-2 suffixes + + + + Containment recheck · parent must be the mount · 04 / Plan · last line + + + + Containment recheck + parent must be the mount + last line + + + + Atomic write · into PDFOPS_OUTPUT_DIR · 05 / Write · temp + rename + + + + Atomic write + into PDFOPS_OUTPUT_DIR + temp + rename + + + + + + raw name tree + untrusted structure + + + + attachment names + untrusted strings + + + + payload bytes + opaque content + + + + safe basenames + fallback for garbage + + + + planned targets + per-volume unique + + + + verified paths + inside the mount + + + + + Legend + + + primary data + + + + policy / PII + + + + async batch + + + + data store + + + + data flow + + + +

+ + + + + + + + + +
+ + +
+
+
+
+

Trust boundary

+
+
    +
  • • A name like ../../evil.txt would be a write-anywhere primitive
  • +
  • • Sanitizing always beats rejecting: renaming is lossless and logged
  • +
  • • The containment recheck guards against sanitizer regressions
  • +
+
+ +
+
+
+

Determinism

+
+
    +
  • • Extraction order is the PDF name-tree order
  • +
  • • Duplicate names get deterministic -1/-2 suffixes
  • +
  • • Collisions are detected on casefolded names for case-insensitive volumes
  • +
+
+ +
+
+
+

Payloads

+
+
    +
  • • Payload bytes never pass through path handling
  • +
  • • Each file is written atomically: whole or absent
  • +
  • • A non-string name key still extracts under a fallback name
  • +
+
+
+ +
+ + + + diff --git a/docs/diagrams/index.html b/docs/diagrams/index.html new file mode 100644 index 0000000..7a84166 --- /dev/null +++ b/docs/diagrams/index.html @@ -0,0 +1,102 @@ + + + + + +pdf-ops System Diagrams + + + +
+

pdf-ops diagrams

+ + keys 1-4 switch +
+
+ + + diff --git a/docs/diagrams/password-flow.html b/docs/diagrams/password-flow.html new file mode 100644 index 0000000..1695d54 --- /dev/null +++ b/docs/diagrams/password-flow.html @@ -0,0 +1,14861 @@ + + + + + + + Password Flow: Supplied Try, Empty Fallback Diagram + + + + + + + + + + + +
+ +
+
+
+

Password Flow: Supplied Try, Empty Fallback

+
+
+ + + + + + + +
+ + Password Flow: Supplied Try, Empty Fallback + A sequence diagram generated by Archify. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + resolve_and_register() + + after the skip check + + + + + + + Secret (renders ***) + + + + + + + + open_input(path, password) + + + + + + + + open with supplied password + + + + + + + + PasswordError + + this input never needed it + + + + + + + open with empty password + + what every viewer does + + + + + + + opened + matched flags + + + + + + + + OpenedInput(password_type=empty) + + + + + + + + input_opened + algorithm + + + + + + + Lazy resolution + + + + Supplied password + + + + Empty fallback + + + + + run_merge · orchestration · Sequence participant + + + + run_merge + orchestration + + + + secrets · lazy resolution · Sequence participant + + + + secrets + lazy resolution + + + + engine · open_input · Sequence participant + + + + engine + open_input + + + + qpdf · via pikepdf · Sequence participant + + + + qpdf + via pikepdf + + + + log stream · JSON on stdout · Sequence participant + + + + log stream + JSON on stdout + + + + + Legend + + + request + + + + return + + + + security + + + + async trace + + + + default message + + + +

+ + + + + + + + + +
+ + +
+
+
+
+

No-leak layers

+
+
    +
  • • The Secret wrapper renders *** in every repr and f-string
  • +
  • • reveal() happens only inside the engine
  • +
  • • A wrong password names the input, never the password
  • +
+
+ +
+
+
+

Empty fallback

+
+
    +
  • • The empty try runs per input, even after a supplied password fails
  • +
  • • Permissions-locked files open exactly like every PDF viewer opens them
  • +
  • • A mixed merge needs only the one real password
  • +
+
+ +
+
+
+

Observability

+
+
    +
  • • input_opened reports algorithm and user/owner/empty
  • +
  • • The algorithm label survives even failed opens
  • +
  • • Failures keep the password exit class (exit 5)
  • +
+
+
+ +
+ + + + diff --git a/docs/diagrams/password-flow.sequence.json b/docs/diagrams/password-flow.sequence.json new file mode 100644 index 0000000..ca1b215 --- /dev/null +++ b/docs/diagrams/password-flow.sequence.json @@ -0,0 +1,226 @@ +{ + "schema_version": 1, + "diagram_type": "sequence", + "meta": { + "title": "Password Flow: Supplied Try, Empty Fallback", + "quality_profile": "showcase", + "column_fit": "spread", + "views": [ + { + "id": "lazy-secrets", + "label": "Lazy resolution", + "focus": [ + "mergeop", + "secretsmod" + ], + "note": "The password file is read only after the skip decision says work is needed." + }, + { + "id": "supplied-try", + "label": "Supplied password", + "focus": [ + "engine", + "qpdflib" + ], + "note": "qpdf tries exactly the string it is given - a mismatch raises." + }, + { + "id": "empty-fallback", + "label": "Empty fallback", + "focus": [ + "engine", + "qpdflib", + "events" + ], + "note": "The spec-standard empty try still applies, so permissions-locked inputs never fail a mixed merge." + } + ] + }, + "participants": [ + { + "id": "mergeop", + "type": "backend", + "label": "run_merge", + "sublabel": "orchestration" + }, + { + "id": "secretsmod", + "type": "security", + "label": "secrets", + "sublabel": "lazy resolution" + }, + { + "id": "engine", + "type": "backend", + "label": "engine", + "sublabel": "open_input" + }, + { + "id": "qpdflib", + "type": "external", + "label": "qpdf", + "sublabel": "via pikepdf" + }, + { + "id": "events", + "type": "messagebus", + "label": "log stream", + "sublabel": "JSON on stdout" + } + ], + "segments": [ + { + "from": 160, + "to": 280, + "label": "Lazy resolution" + }, + { + "from": 300, + "to": 460, + "label": "Supplied password" + }, + { + "from": 480, + "to": 700, + "label": "Empty fallback" + } + ], + "messages": [ + { + "id": "resolve-call", + "from": "mergeop", + "to": "secretsmod", + "y": 195, + "label": "resolve_and_register()", + "variant": "security", + "note": "after the skip check" + }, + { + "id": "secret-back", + "from": "secretsmod", + "to": "mergeop", + "y": 250, + "label": "Secret (renders ***)", + "variant": "return" + }, + { + "id": "open-call", + "from": "mergeop", + "to": "engine", + "y": 330, + "label": "open_input(path, password)", + "variant": "emphasis" + }, + { + "id": "supplied-open", + "from": "engine", + "to": "qpdflib", + "y": 380, + "label": "open with supplied password", + "variant": "default" + }, + { + "id": "password-error", + "from": "qpdflib", + "to": "engine", + "y": 430, + "label": "PasswordError", + "variant": "security", + "note": "this input never needed it" + }, + { + "id": "empty-open", + "from": "engine", + "to": "qpdflib", + "y": 510, + "label": "open with empty password", + "variant": "default", + "note": "what every viewer does" + }, + { + "id": "opened-ok", + "from": "qpdflib", + "to": "engine", + "y": 560, + "label": "opened + matched flags", + "variant": "return" + }, + { + "id": "opened-input", + "from": "engine", + "to": "mergeop", + "y": 615, + "label": "OpenedInput(password_type=empty)", + "variant": "return" + }, + { + "id": "log-opened", + "from": "mergeop", + "to": "events", + "y": 665, + "label": "input_opened + algorithm", + "variant": "dashed" + } + ], + "activations": [ + { + "participant": "mergeop", + "from": 185, + "to": 675, + "type": "backend" + }, + { + "participant": "secretsmod", + "from": 190, + "to": 256, + "type": "security" + }, + { + "participant": "engine", + "from": 325, + "to": 620, + "type": "backend" + }, + { + "participant": "qpdflib", + "from": 375, + "to": 566, + "type": "external" + }, + { + "participant": "events", + "from": 660, + "to": 690, + "type": "messagebus" + } + ], + "cards": [ + { + "dot": "rose", + "title": "No-leak layers", + "items": [ + "The Secret wrapper renders *** in every repr and f-string", + "reveal() happens only inside the engine", + "A wrong password names the input, never the password" + ] + }, + { + "dot": "emerald", + "title": "Empty fallback", + "items": [ + "The empty try runs per input, even after a supplied password fails", + "Permissions-locked files open exactly like every PDF viewer opens them", + "A mixed merge needs only the one real password" + ] + }, + { + "dot": "cyan", + "title": "Observability", + "items": [ + "input_opened reports algorithm and user/owner/empty", + "The algorithm label survives even failed opens", + "Failures keep the password exit class (exit 5)" + ] + } + ] +} diff --git a/docs/diagrams/pdf-ops-architecture.html b/docs/diagrams/pdf-ops-architecture.html new file mode 100644 index 0000000..d90271b --- /dev/null +++ b/docs/diagrams/pdf-ops-architecture.html @@ -0,0 +1,14894 @@ + + + + + + + pdf-ops Module Architecture Diagram + + + + + + + + + + + +
+ +
+
+
+

pdf-ops Module Architecture

+
+
+ + + + + + + +
+ + pdf-ops Module Architecture + An architecture diagram generated by Archify. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + Environment · PDFOPS_* variables · Architecture component + + + + Environment + PDFOPS_* variables + + + + Config parser · pure, filesystem-free · Architecture component · exit 2 fast + + + + Config parser + pure, filesystem-free + exit 2 fast + + + + Error boundary · run(env) -> exit code · Architecture component · one terminal event + + + + Error boundary + run(env) -> exit code + one terminal event + + + + Secrets · wrapper + resolution · Architecture component · renders *** + + + + Secrets + wrapper + resolution + renders *** + + + + Merge · validate all, then write · Architecture component + + + + Merge + validate all, then write + + + + Extract · names are untrusted · Architecture component + + + + Extract + names are untrusted + + + + Input validation · shared, collect-all · Architecture component · one failure, all problems + + + + Input validation + shared, collect-all + one failure, all problems + + + + PdfEngine Protocol · the swap seam · sole PDF-library importer · one importer + + + + PdfEngine Protocol + the swap seam + one importer + + + + pikepdf engine · taxonomy translation · sole PDF-library importer · qpdf-backed + + + + pikepdf engine + taxonomy translation + qpdf-backed + + + + Atomic output · temp + fsync + rename · Architecture component · whole or absent + + + + Atomic output + temp + fsync + rename + whole or absent + + + + JSON log stream · stdout, stderr empty · Architecture component + + + + JSON log stream + stdout, stderr empty + + + + Workflow engine · Argo step · Architecture component · at-least-once + + + + Workflow engine + Argo step + at-least-once + + + + + + 13 variables + + + + typed config + + + + + + + + + + get_engine() + + + + saves into open temp + + + + resolve lazily + + + + reveal() at decrypt + + + + + terminal event + exit code + + + + + + sole PDF-library importer + + + + + Legend + + + Backend + + + + Database + + + + Security + + + + Message bus + + + + External + + + +

+ + + + + + + + + +
+ + +
+
+
+
+

Contract

+
+
    +
  • • Configuration is parsed completely before any file is touched
  • +
  • • Exactly one terminal JSON event per run; the exit code is the API
  • +
  • • Exit classes 0-6 with machine-readable error_code strings
  • +
+
+ +
+
+
+

Seam

+
+
    +
  • • One module imports the PDF library and translates its failures
  • +
  • • The pypdf-to-pikepdf swap touched exactly that module
  • +
  • • pypdf remains as the cross-checking test oracle
  • +
+
+ +
+
+
+

No-leak layers

+
+
    +
  • • Secrets resolve lazily - a skipped merge never reads the password file
  • +
  • • The raw value is reachable only inside the engine
  • +
  • • Log scrubbing is field-restricted to avoid a password oracle
  • +
+
+
+ +
+ + + + diff --git a/docs/diagrams/pdf-ops.architecture.json b/docs/diagrams/pdf-ops.architecture.json new file mode 100644 index 0000000..f8cab60 --- /dev/null +++ b/docs/diagrams/pdf-ops.architecture.json @@ -0,0 +1,445 @@ +{ + "schema_version": 1, + "diagram_type": "architecture", + "meta": { + "title": "pdf-ops Module Architecture", + "quality_profile": "showcase", + "repository": { + "url": "https://github.com/Radko-D/python-pdfops", + "revision": "7327d5388dcfbb1e2e5b7cf7b1cf38f68d5b7782" + }, + "views": [ + { + "id": "operation-path", + "label": "Operation path", + "focus": [ + "envvars", + "configparse", + "runboundary", + "mergeop", + "extractop", + "inputvalidation", + "engineseam", + "qpdfengine", + "atomicwrite" + ], + "note": "Follow one run from environment variables to the atomically published output." + }, + { + "id": "noleak-path", + "label": "Secret handling", + "focus": [ + "runboundary", + "secretstore", + "qpdfengine" + ], + "note": "Secrets resolve lazily and reveal only inside the engine." + }, + { + "id": "operator-interface", + "label": "Operator interface", + "focus": [ + "runboundary", + "eventstream", + "workflowengine" + ], + "note": "JSON events and the exit code are the only outward surfaces." + } + ] + }, + "components": [ + { + "id": "envvars", + "type": "external", + "label": "Environment", + "sublabel": "PDFOPS_* variables", + "pos": [ + 40, + 300 + ], + "size": [ + 140, + 60 + ] + }, + { + "id": "configparse", + "type": "backend", + "label": "Config parser", + "sublabel": "pure, filesystem-free", + "pos": [ + 250, + 300 + ], + "size": [ + 150, + 60 + ], + "tag": "exit 2 fast", + "sources": [ + { + "path": "src/pdf_ops/config.py", + "line": 136, + "end_line": 171, + "label": "parse_config" + } + ] + }, + { + "id": "runboundary", + "type": "backend", + "label": "Error boundary", + "sublabel": "run(env) -> exit code", + "pos": [ + 470, + 300 + ], + "size": [ + 150, + 60 + ], + "tag": "one terminal event", + "sources": [ + { + "path": "src/pdf_ops/main.py", + "line": 16, + "end_line": 79, + "label": "run" + } + ] + }, + { + "id": "secretstore", + "type": "security", + "label": "Secrets", + "sublabel": "wrapper + resolution", + "pos": [ + 470, + 130 + ], + "size": [ + 150, + 64 + ], + "tag": "renders ***", + "sources": [ + { + "path": "src/pdf_ops/secrets.py", + "line": 29, + "end_line": 51, + "label": "Secret" + }, + { + "path": "src/pdf_ops/secrets.py", + "line": 129, + "end_line": 153, + "label": "resolve_and_register" + } + ] + }, + { + "id": "mergeop", + "type": "backend", + "label": "Merge", + "sublabel": "validate all, then write", + "pos": [ + 690, + 210 + ], + "size": [ + 140, + 60 + ], + "sources": [ + { + "path": "src/pdf_ops/merge.py", + "line": 18, + "end_line": 97, + "label": "run_merge" + } + ] + }, + { + "id": "extractop", + "type": "backend", + "label": "Extract", + "sublabel": "names are untrusted", + "pos": [ + 690, + 390 + ], + "size": [ + 140, + 60 + ], + "sources": [ + { + "path": "src/pdf_ops/extract.py", + "line": 27, + "end_line": 159, + "label": "run_extract" + }, + { + "path": "src/pdf_ops/extract.py", + "line": 185, + "end_line": 224, + "label": "sanitizer" + } + ] + }, + { + "id": "inputvalidation", + "type": "backend", + "label": "Input validation", + "sublabel": "shared, collect-all", + "pos": [ + 690, + 300 + ], + "size": [ + 140, + 60 + ], + "tag": "one failure, all problems", + "sources": [ + { + "path": "src/pdf_ops/inputs.py", + "line": 21, + "end_line": 51, + "label": "validate_inputs" + } + ] + }, + { + "id": "engineseam", + "type": "backend", + "label": "PdfEngine Protocol", + "sublabel": "the swap seam", + "pos": [ + 900, + 300 + ], + "size": [ + 160, + 60 + ], + "tag": "one importer", + "sources": [ + { + "path": "src/pdf_ops/engine.py", + "line": 72, + "end_line": 107, + "label": "Protocol" + } + ] + }, + { + "id": "qpdfengine", + "type": "backend", + "label": "pikepdf engine", + "sublabel": "taxonomy translation", + "pos": [ + 1130, + 300 + ], + "size": [ + 150, + 60 + ], + "tag": "qpdf-backed", + "sources": [ + { + "path": "src/pdf_ops/engine_pikepdf.py", + "line": 38, + "end_line": 184, + "label": "PikepdfEngine" + } + ] + }, + { + "id": "atomicwrite", + "type": "database", + "label": "Atomic output", + "sublabel": "temp + fsync + rename", + "pos": [ + 1130, + 470 + ], + "size": [ + 150, + 60 + ], + "tag": "whole or absent", + "sources": [ + { + "path": "src/pdf_ops/output.py", + "line": 107, + "end_line": 140, + "label": "atomic_output" + } + ] + }, + { + "id": "eventstream", + "type": "messagebus", + "label": "JSON log stream", + "sublabel": "stdout, stderr empty", + "pos": [ + 470, + 470 + ], + "size": [ + 150, + 60 + ], + "sources": [ + { + "path": "src/pdf_ops/logging_setup.py", + "line": 144, + "end_line": 173, + "label": "setup_logging" + } + ] + }, + { + "id": "workflowengine", + "type": "external", + "label": "Workflow engine", + "sublabel": "Argo step", + "pos": [ + 130, + 470 + ], + "size": [ + 150, + 60 + ], + "tag": "at-least-once" + } + ], + "boundaries": [ + { + "kind": "security-group", + "label": "sole PDF-library importer", + "wraps": [ + "engineseam", + "qpdfengine" + ] + } + ], + "connections": [ + { + "id": "env-parse", + "from": "envvars", + "to": "configparse", + "label": "13 variables", + "variant": "emphasis" + }, + { + "id": "parse-run", + "from": "configparse", + "to": "runboundary", + "label": "typed config" + }, + { + "id": "run-merge", + "from": "runboundary", + "to": "mergeop" + }, + { + "id": "run-extract", + "from": "runboundary", + "to": "extractop" + }, + { + "id": "merge-validate", + "from": "mergeop", + "to": "inputvalidation" + }, + { + "id": "extract-validate", + "from": "extractop", + "to": "inputvalidation" + }, + { + "id": "merge-seam", + "from": "mergeop", + "to": "engineseam" + }, + { + "id": "extract-seam", + "from": "extractop", + "to": "engineseam" + }, + { + "id": "seam-engine", + "from": "engineseam", + "to": "qpdfengine", + "label": "get_engine()", + "labelDy": 22 + }, + { + "id": "engine-write", + "from": "qpdfengine", + "to": "atomicwrite", + "label": "saves into open temp", + "labelDy": 45 + }, + { + "id": "run-secrets", + "from": "runboundary", + "to": "secretstore", + "label": "resolve lazily", + "variant": "security" + }, + { + "id": "secrets-engine", + "from": "secretstore", + "to": "qpdfengine", + "label": "reveal() at decrypt", + "variant": "security" + }, + { + "id": "run-events", + "from": "runboundary", + "to": "eventstream", + "variant": "dashed" + }, + { + "id": "events-argo", + "from": "eventstream", + "to": "workflowengine", + "label": "terminal event + exit code", + "variant": "emphasis" + } + ], + "cards": [ + { + "dot": "cyan", + "title": "Contract", + "items": [ + "Configuration is parsed completely before any file is touched", + "Exactly one terminal JSON event per run; the exit code is the API", + "Exit classes 0-6 with machine-readable error_code strings" + ] + }, + { + "dot": "emerald", + "title": "Seam", + "items": [ + "One module imports the PDF library and translates its failures", + "The pypdf-to-pikepdf swap touched exactly that module", + "pypdf remains as the cross-checking test oracle" + ] + }, + { + "dot": "rose", + "title": "No-leak layers", + "items": [ + "Secrets resolve lazily - a skipped merge never reads the password file", + "The raw value is reachable only inside the engine", + "Log scrubbing is field-restricted to avoid a password oracle" + ] + } + ] +} diff --git a/docs/diagrams/retry-machine.html b/docs/diagrams/retry-machine.html new file mode 100644 index 0000000..be77690 --- /dev/null +++ b/docs/diagrams/retry-machine.html @@ -0,0 +1,14827 @@ + + + + + + + Run Lifecycle and Retry Semantics Diagram + + + + + + + + + + + +
+ +
+
+
+

Run Lifecycle and Retry Semantics

+
+
+ + + + + + + +
+ + Run Lifecycle and Retry Semantics + A lifecycle diagram generated by Archify. + + + + + + + + + + + + + + + + + + + + + + + + + 01 / Run phases + + 02 / Interruptions + + 03 / Policy outcomes + + + + + + + + + + + + + Invoked · one operation per run · Run phases · entry + + + + 01 + Invoked + one operation per run + entry + + + + Config parsed · before any file I/O · Run phases · pure + + + + 02 + Config parsed + before any file I/O + pure + + + + Output exists? · PDFOPS_ON_EXISTS · Run phases · policy + + + + 03 + Output exists? + PDFOPS_ON_EXISTS + policy + + + + Temp write · fsync, then one rename · Run phases · atomic + + + + 04 + Temp write + fsync, then one rename + atomic + + + + Complete · exit 0, terminal event · Run phases · done + + + + 05 + Complete + exit 0, terminal event + done + + + + Skipped · exit 0, reads nothing · Policy outcomes · skip + + + + Skipped + exit 0, reads nothing + skip + + + + Killed mid-write · temp debris only · Interruptions · retryable + + + + Killed mid-write + temp debris only + retryable + + + + OUTPUT_EXISTS · exit 6, no clobber · Policy outcomes · fail + + + + OUTPUT_EXISTS + exit 6, no clobber + fail + + + + + + skip: completed prior work + + + + fail (default) + + + + pod lost + + + + retry: stale temp removed + + + + + Legend + + + start + + + + active state + + + + decision + + + + terminal success + + + + failure / exit + + + +

+ + + + + + + + + +
+ + +
+
+
+
+

Atomicity

+
+
    +
  • • The final path holds a complete file or nothing
  • +
  • • Work happens in a temp file inside the destination directory
  • +
  • • One rename publishes; a crash leaves only namable debris
  • +
+
+ +
+
+
+

Idempotent retries

+
+
    +
  • • skip treats an existing output as finished work and reads nothing
  • +
  • • extract completes a partial set file by file under skip
  • +
  • • overwrite replaces atomically for reprocessing pipelines
  • +
+
+ +
+
+
+

Deterministic failures

+
+
    +
  • • Exit classes 2-6 are permanent; retrying them is waste
  • +
  • • Only exit 1 and pod-level errors are worth a retry
  • +
  • • The terminal JSON event carries the machine-readable error_code
  • +
+
+
+ +
+ + + + diff --git a/docs/diagrams/retry-machine.lifecycle.json b/docs/diagrams/retry-machine.lifecycle.json new file mode 100644 index 0000000..7b965eb --- /dev/null +++ b/docs/diagrams/retry-machine.lifecycle.json @@ -0,0 +1,255 @@ +{ + "schema_version": 1, + "diagram_type": "lifecycle", + "meta": { + "title": "Run Lifecycle and Retry Semantics", + "quality_profile": "showcase", + "views": [ + { + "id": "happy-run", + "label": "One clean run", + "focus": [ + "invoked", + "parsed", + "existscheck", + "publishing", + "complete" + ], + "note": "Follow a run from invocation to the atomically published output." + }, + { + "id": "policy-branches", + "label": "Existing-output policy", + "focus": [ + "existscheck", + "skipped", + "refused" + ], + "note": "PDFOPS_ON_EXISTS decides what an existing output means." + }, + { + "id": "crash-retry", + "label": "Crash and retry", + "focus": [ + "publishing", + "killed", + "parsed" + ], + "note": "A killed run leaves debris the next run removes before writing." + } + ] + }, + "lanes": [ + { + "id": "main", + "label": "Run phases" + }, + { + "id": "recovery", + "label": "Interruptions" + }, + { + "id": "terminal", + "label": "Policy outcomes" + } + ], + "states": [ + { + "id": "invoked", + "type": "start", + "label": "Invoked", + "sublabel": "one operation per run", + "lane": "main", + "col": 0, + "step": "01", + "tag": "entry" + }, + { + "id": "parsed", + "type": "active", + "label": "Config parsed", + "sublabel": "before any file I/O", + "lane": "main", + "col": 1, + "step": "02", + "tag": "pure" + }, + { + "id": "existscheck", + "type": "decision", + "label": "Output exists?", + "sublabel": "PDFOPS_ON_EXISTS", + "lane": "main", + "col": 2, + "step": "03", + "tag": "policy" + }, + { + "id": "publishing", + "type": "active", + "label": "Temp write", + "sublabel": "fsync, then one rename", + "lane": "main", + "col": 3, + "step": "04", + "tag": "atomic" + }, + { + "id": "complete", + "type": "success", + "label": "Complete", + "sublabel": "exit 0, terminal event", + "lane": "main", + "col": 4, + "step": "05", + "tag": "done" + }, + { + "id": "skipped", + "type": "success", + "label": "Skipped", + "sublabel": "exit 0, reads nothing", + "lane": "terminal", + "col": 0, + "tag": "skip" + }, + { + "id": "killed", + "type": "failure", + "label": "Killed mid-write", + "sublabel": "temp debris only", + "lane": "recovery", + "col": 2, + "tag": "retryable" + }, + { + "id": "refused", + "type": "failure", + "label": "OUTPUT_EXISTS", + "sublabel": "exit 6, no clobber", + "lane": "terminal", + "col": 1, + "tag": "fail" + } + ], + "transitions": [ + { + "id": "policy-skip", + "from": "existscheck", + "to": "skipped", + "label": "skip: completed prior work", + "variant": "emphasis", + "route": "drop" + }, + { + "id": "policy-fail", + "from": "existscheck", + "to": "refused", + "label": "fail (default)", + "variant": "security", + "fromSide": "right", + "toSide": "top", + "via": [ + [ + 479, + 157 + ], + [ + 479, + 212 + ], + [ + 556, + 212 + ] + ], + "labelAt": [ + 517, + 200 + ] + }, + { + "id": "write-killed", + "from": "publishing", + "to": "killed", + "label": "pod lost", + "variant": "security", + "labelAt": [ + 672, + 246 + ], + "fromSide": "right", + "toSide": "top", + "via": [ + [ + 633, + 157 + ], + [ + 633, + 258 + ], + [ + 710, + 258 + ] + ] + }, + { + "id": "retry-cleanup", + "from": "killed", + "to": "parsed", + "label": "retry: stale temp removed", + "variant": "emphasis", + "fromSide": "bottom", + "toSide": "top", + "via": [ + [ + 710, + 560 + ], + [ + 16, + 560 + ], + [ + 16, + 60 + ], + [ + 248, + 60 + ] + ] + } + ], + "cards": [ + { + "dot": "emerald", + "title": "Atomicity", + "items": [ + "The final path holds a complete file or nothing", + "Work happens in a temp file inside the destination directory", + "One rename publishes; a crash leaves only namable debris" + ] + }, + { + "dot": "amber", + "title": "Idempotent retries", + "items": [ + "skip treats an existing output as finished work and reads nothing", + "extract completes a partial set file by file under skip", + "overwrite replaces atomically for reprocessing pipelines" + ] + }, + { + "dot": "rose", + "title": "Deterministic failures", + "items": [ + "Exit classes 2-6 are permanent; retrying them is waste", + "Only exit 1 and pod-level errors are worth a retry", + "The terminal JSON event carries the machine-readable error_code" + ] + } + ] +} diff --git a/docs/scripts/_standard_parser.py b/docs/scripts/_standard_parser.py index 2ff7f9e..845ab69 100755 --- a/docs/scripts/_standard_parser.py +++ b/docs/scripts/_standard_parser.py @@ -12,8 +12,6 @@ controlled-vocabularies section to change vocabularies - every script picks up the change. """ -from __future__ import annotations - import re import sys from pathlib import Path @@ -58,22 +56,21 @@ # Vocabulary parsing # ----------------------------------------------------------------------------- -# Standard path resolution (v0.11.27): prefer --standard-path CLI arg if -# provided, fall back to relative-to-script default. The default resolves -# correctly when this script lives at docs/scripts/ in a scaffolded project -# (parent.parent -> docs/). When invoked from skill-side -# (~/.claude/skills/init-decisions/templates/), the default fails because -# the standard isn't co-located - pass --standard-path explicitly. The arg -# unblocks verify-framework Ability's documented CLI invocation against -# legacy projects whose docs/scripts/ doesn't have a copy of this script. + +# Standard path resolution: prefer the --standard-path CLI arg when given, +# fall back to the relative-to-script default. The default resolves +# correctly when this script lives at docs/scripts/ next to the standard +# (parent.parent -> docs/); a copy running from anywhere else must pass +# --standard-path explicitly because the standard is not co-located. def _resolve_standard_path() -> Path: import sys + if "--standard-path" in sys.argv: idx = sys.argv.index("--standard-path") if idx + 1 < len(sys.argv): path = Path(sys.argv[idx + 1]).resolve() # Pop the flag + value so consuming scripts' argparse doesn't reject them. - del sys.argv[idx:idx + 2] + del sys.argv[idx : idx + 2] return path return Path(__file__).resolve().parent.parent / "DECISION_TRACKING_STANDARD.md" @@ -134,8 +131,8 @@ def _abort(msg: str) -> None: def _load() -> dict[str, frozenset[str]]: if not STANDARD_PATH.exists(): _abort( - f"DECISION_TRACKING_STANDARD.md not found - required for vocabulary parsing. " - f"Run /init-decisions to scaffold it." + "DECISION_TRACKING_STANDARD.md not found - required for vocabulary parsing. " + "It belongs next to DECISIONS.md in docs/." ) text = STANDARD_PATH.read_text(encoding="utf-8") sections = _parse_vocab_sections(text) diff --git a/docs/scripts/validate_decisions.py b/docs/scripts/validate_decisions.py index d4169e0..1553937 100755 --- a/docs/scripts/validate_decisions.py +++ b/docs/scripts/validate_decisions.py @@ -30,12 +30,11 @@ Exit code 0 on clean run, 1 on any violation. Prints one line per issue. """ -from __future__ import annotations - import re import sys from collections import Counter, defaultdict from dataclasses import dataclass, field +from itertools import pairwise from pathlib import Path from _standard_parser import ( @@ -67,9 +66,7 @@ ) GREEN_STATUS_ICON = "🟢" -SUMMARY_PLACEHOLDERS: frozenset[str] = frozenset( - {"tbd", "todo", "tba", "-", " - ", "n/a"} -) +SUMMARY_PLACEHOLDERS: frozenset[str] = frozenset({"tbd", "todo", "tba", "-", " - ", "n/a"}) @dataclass @@ -126,11 +123,7 @@ def parse_master_register( if stripped == "## Master Register": in_master = True continue - if ( - in_master - and (stripped.startswith("## ") or stripped.startswith("### ")) - and stripped != "## Master Register" - ): + if in_master and stripped.startswith(("## ", "### ")) and stripped != "## Master Register": in_master = False if not in_master: continue @@ -165,9 +158,7 @@ def parse_master_register( return decisions, total_claimed -def parse_anchor_pages( - lines: list[str], decisions: dict[str, Decision], report: Report -) -> None: +def parse_anchor_pages(lines: list[str], decisions: dict[str, Decision], report: Report) -> None: """Populate anchor_fields on each Decision by reading its anchor page.""" i = 0 n = len(lines) @@ -203,11 +194,7 @@ def parse_anchor_pages( j = i + 1 while j < n: sub = lines[j].rstrip("\n") - if ( - ANCHOR_HEADING_RE.match(sub) - or sub.startswith("## ") - or sub.startswith("---") - ): + if ANCHOR_HEADING_RE.match(sub) or sub.startswith(("## ", "---")): break fm = ANCHOR_FIELD_RE.match(sub) if fm: @@ -233,11 +220,11 @@ def check_sequence(decisions: dict[str, Decision], report: Report) -> None: return if nums[0] != 1: report.err(f"ID sequence starts at D-{nums[0]:03d}, expected D-001") - for prev, curr in zip(nums, nums[1:]): + for prev, curr in pairwise(nums): if curr != prev + 1: report.err( f"ID sequence gap: D-{prev:03d} -> D-{curr:03d} " - f"(missing D-{prev+1:03d}..D-{curr-1:03d})" + f"(missing D-{prev + 1:03d}..D-{curr - 1:03d})" ) @@ -258,14 +245,11 @@ def check_areas(decisions: dict[str, Decision], report: Report) -> None: for did, d in sorted(decisions.items()): if d.area not in AREAS: report.err( - f"{did}: area '{d.area}' not in standard's taxonomy " - f"({', '.join(sorted(AREAS))})" + f"{did}: area '{d.area}' not in standard's taxonomy ({', '.join(sorted(AREAS))})" ) anchor_area = d.anchor_fields.get("Area", "").strip() if anchor_area and anchor_area != d.area: - report.err( - f"{did}: area mismatch - master='{d.area}' anchor='{anchor_area}'" - ) + report.err(f"{did}: area mismatch - master='{d.area}' anchor='{anchor_area}'") def check_vocabularies(decisions: dict[str, Decision], report: Report) -> None: @@ -311,9 +295,7 @@ def check_vocabularies(decisions: dict[str, Decision], report: Report) -> None: ) -def check_index_counts( - lines: list[str], decisions: dict[str, Decision], report: Report -) -> None: +def check_index_counts(lines: list[str], decisions: dict[str, Decision], report: Report) -> None: """Verify the 'Index by area' table agrees with the master register.""" actual_counts: Counter[str] = Counter(d.area for d in decisions.values()) actual_ids: dict[str, list[str]] = defaultdict(list) @@ -353,20 +335,18 @@ def check_index_counts( if count != len(decisions): report.err( f"Index by area: Total count {count} != actual {len(decisions)} " - f"(line {i+1})" + f"(line {i + 1})" ) else: claimed_areas.add(area) if area not in AREAS: - report.err( - f"Index by area: unknown area '{area}' at line {i+1}" - ) + report.err(f"Index by area: unknown area '{area}' at line {i + 1}") i += 1 continue if count != actual_counts[area]: report.err( f"Index by area: area '{area}' count {count} != actual " - f"{actual_counts[area]} at line {i+1}" + f"{actual_counts[area]} at line {i + 1}" ) listed = sorted(set(ID_RE.findall(ids_cell))) actual_set = sorted({i.split("-")[1] for i in actual_ids[area]}) @@ -375,11 +355,11 @@ def check_index_counts( extra = sorted(set(listed) - set(actual_set)) parts = [] if missing: - parts.append(f"missing {', '.join('D-'+m for m in missing)}") + parts.append(f"missing {', '.join('D-' + m for m in missing)}") if extra: - parts.append(f"extra {', '.join('D-'+e for e in extra)}") + parts.append(f"extra {', '.join('D-' + e for e in extra)}") report.err( - f"Index by area: area '{area}' ID list drift at line {i+1} - " + f"Index by area: area '{area}' ID list drift at line {i + 1} - " + "; ".join(parts) ) i += 1 @@ -390,8 +370,7 @@ def check_index_counts( for area in AREAS: if actual_counts[area] and area not in claimed_areas: report.err( - f"Index by area: area '{area}' has {actual_counts[area]} " - f"decisions but no row" + f"Index by area: area '{area}' has {actual_counts[area]} decisions but no row" ) @@ -410,9 +389,7 @@ def check_related_ids(decisions: dict[str, Decision], report: Report) -> None: ) -def check_doc_links( - decisions: dict[str, Decision], docs_root: Path, report: Report -) -> None: +def check_doc_links(decisions: dict[str, Decision], docs_root: Path, report: Report) -> None: """Every `discussion` / `Where` link target (relative path) must exist in docs/.""" for did, d in sorted(decisions.items()): candidates: list[tuple[str, str]] = [] @@ -431,10 +408,7 @@ def check_doc_links( continue resolved = (docs_root / path_part).resolve() if not resolved.exists(): - report.err( - f"{did}: {label} link '{path_part}' -> {resolved} " - f"does not exist" - ) + report.err(f"{did}: {label} link '{path_part}' -> {resolved} does not exist") def check_green_summaries(decisions: dict[str, Decision], report: Report) -> None: @@ -457,9 +431,7 @@ def check_green_summaries(decisions: dict[str, Decision], report: Report) -> Non ) -def check_adr_references( - lines: list[str], decisions: dict[str, Decision], report: Report -) -> None: +def check_adr_references(lines: list[str], decisions: dict[str, Decision], report: Report) -> None: """Every ``ADR-###`` link in the master register's "ADR" column must resolve to an ``## ADR-###:`` heading in the same document.""" existing_adrs: set[str] = set() @@ -497,7 +469,8 @@ def check_decide_under_assumption_cross_links( DECISION_TRACKING_STANDARD.md for the rationale. """ assumed = [ - d for d in decisions.values() + d + for d in decisions.values() if "assumed" in d.anchor_fields.get("Status", d.status).lower() ] if not assumed: @@ -557,13 +530,9 @@ def main(argv: list[str]) -> int: check_decide_under_assumption_cross_links(decisions, docs_root, report) if total_claimed is not None and total_claimed != len(decisions): - report.err( - f"Counts line claims {total_claimed} total decisions, actual = {len(decisions)}" - ) + report.err(f"Counts line claims {total_claimed} total decisions, actual = {len(decisions)}") - rel_path = ( - path.relative_to(Path.cwd()) if path.is_relative_to(Path.cwd()) else path - ) + rel_path = path.relative_to(Path.cwd()) if path.is_relative_to(Path.cwd()) else path print(f"Parsed {len(decisions)} decisions from {rel_path}") if report.warnings: print(f"\n{len(report.warnings)} warning(s):") diff --git a/pyproject.toml b/pyproject.toml index b31e5e9..a86bde0 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -9,20 +9,38 @@ authors = [ { email = "29573973+Radko-D@users.noreply.github.com" } ] requires-python = ">=3.14" +classifiers = [ + # Guard against an accidental `uv publish`: PyPI refuses this classifier. + # Remove it when the package is meant to be published. + "Private :: Do Not Upload", + "Environment :: No Input/Output (Daemon)", + "Intended Audience :: System Administrators", + "Operating System :: POSIX :: Linux", + "Programming Language :: Python :: 3.14", + "Topic :: Office/Business", + "Typing :: Typed", +] dependencies = [ - "pypdf[crypto]>=6.16.2", + "pikepdf>=10.12.0", ] +[project.urls] +Repository = "https://github.com/Radko-D/python-pdfops" +Design = "https://github.com/Radko-D/python-pdfops/blob/main/docs/DESIGN.md" + [project.scripts] pdf-ops = "pdf_ops.__main__:main" [tool.ruff] line-length = 100 target-version = "py314" -exclude = [".venv", "__pycache__", "build", "dist", "scripts"] +exclude = [".venv", "__pycache__", "build", "dist"] [tool.ruff.lint] -select = ["E", "W", "F", "I", "UP", "B", "C4", "ARG", "ASYNC", "RUF"] +select = [ + "E", "W", "F", "I", "UP", "B", "C4", "ARG", "RUF", + "SIM", "PTH", "PIE", "RET", "PERF", "FURB", "N", +] ignore = ["E501"] [tool.ruff.lint.per-file-ignores] @@ -39,7 +57,7 @@ known-first-party = ["pdf_ops"] pythonVersion = "3.14" venvPath = "." venv = ".venv" -include = ["src", "tests"] +include = ["src", "tests", "scripts", "docs/scripts"] exclude = [".venv", "__pycache__", "build", "dist"] typeCheckingMode = "strict" reportMissingTypeStubs = false @@ -49,6 +67,8 @@ reportUnknownVariableType = false reportUnknownParameterType = false reportUnusedImport = false reportPrivateUsage = false +# An ignore comment that suppresses nothing is drift - keep them honest. +reportUnnecessaryTypeIgnoreComment = true [tool.pytest.ini_options] testpaths = ["tests"] @@ -67,6 +87,7 @@ build-backend = "uv_build" [dependency-groups] dev = [ "pre-commit>=4.6.2", + "pypdf[crypto]>=6.16.2", "pyright>=1.1.411", "pytest>=9.1.1", "ruff>=0.16.5", diff --git a/scripts/benchmark.py b/scripts/benchmark.py new file mode 100644 index 0000000..c9680b2 --- /dev/null +++ b/scripts/benchmark.py @@ -0,0 +1,215 @@ +"""Resource benchmarks: peak memory and duration per operation, in-container. + +Generates large fixtures (never committed), runs each scenario through the +real container image, and reports two memory numbers per run: + +- process peak RSS (ru_maxrss of the pdf_ops process) - what the + application itself needs; +- cgroup memory.peak - what Kubernetes meters against limits, which also + counts the (reclaimable) page cache the run touched. + +Usage: python3 scripts/benchmark.py [--image pdf-ops:bench] [--workdir DIR] +The workdir defaults to /tmp/pdfops-bench (a path Docker Desktop shares). +""" + +import argparse +import json +import os +import shutil +import subprocess +import time +from pathlib import Path + +import pikepdf + +MB = 1024 * 1024 + +# One page of PDF structure per megabyte of incompressible payload: random +# bytes in uncompressed content streams, so file sizes are honest and the +# engine cannot shrink them away. +PAGE_STREAM_BYTES = 1 * MB + +# Runs the operation under a tiny wrapper so ru_maxrss of the child (the +# real entrypoint) and the cgroup peak are both reported on stderr. +WRAPPER = ( + "import resource, subprocess, sys\n" + "proc = subprocess.run([sys.executable, '-m', 'pdf_ops'])\n" + "rss_kb = resource.getrusage(resource.RUSAGE_CHILDREN).ru_maxrss\n" + "print(f'PEAK_RSS_KB={rss_kb}', file=sys.stderr)\n" + "try:\n" + " peak = open('/sys/fs/cgroup/memory.peak').read().strip()\n" + " print(f'CGROUP_PEAK={peak}', file=sys.stderr)\n" + "except OSError:\n" + " pass\n" + "sys.exit(proc.returncode)\n" +) + + +def make_pdf(path: Path, size_mb: int) -> None: + if path.exists(): + return + pdf = pikepdf.Pdf.new() + for _ in range(size_mb): + page = pdf.make_indirect( + pikepdf.Dictionary( + Type=pikepdf.Name.Page, + MediaBox=[0, 0, 612, 792], + Contents=pdf.make_stream(os.urandom(PAGE_STREAM_BYTES)), + ) + ) + pdf.pages.append(pikepdf.Page(page)) + pdf.save(path) + + +def make_encrypted_pdf(path: Path, size_mb: int, password: str) -> None: + if path.exists(): + return + plain = path.with_suffix(".plain.tmp") + make_pdf(plain, size_mb) + with pikepdf.open(plain) as pdf: + pdf.save(path, encryption=pikepdf.Encryption(user=password, owner=password, R=6)) + plain.unlink() + + +def make_attachment_carrier(path: Path, count: int, each_mb: int) -> None: + if path.exists(): + return + pdf = pikepdf.Pdf.new() + pdf.add_blank_page(page_size=(612, 792)) + for index in range(count): + name = f"payload-{index:02d}.bin" + spec = pikepdf.AttachedFileSpec( + pdf, + os.urandom(each_mb * MB), + description="benchmark payload", + filename=name, + mime_type="application/octet-stream", + creation_date="", + mod_date="", + ) + pdf.attachments[name] = spec + pdf.save(path) + + +def run_container(image: str, env: dict[str, str], volumes: dict[Path, str]) -> dict[str, object]: + cmd = ["docker", "run", "--rm", "--entrypoint", "python"] + for key, value in env.items(): + cmd += ["-e", f"{key}={value}"] + for host, spec in volumes.items(): + cmd += ["-v", f"{host}:{spec}"] + cmd += [image, "-c", WRAPPER] + + started = time.monotonic() + proc = subprocess.run(cmd, capture_output=True, text=True, timeout=1800) + wall = time.monotonic() - started + + events = [json.loads(line) for line in proc.stdout.strip().splitlines() if line] + terminal = events[-1] if events else {} + rss_kb = cgroup_peak = None + for line in proc.stderr.splitlines(): + if line.startswith("PEAK_RSS_KB="): + rss_kb = int(line.split("=", 1)[1]) + elif line.startswith("CGROUP_PEAK="): + cgroup_peak = int(line.split("=", 1)[1]) + return { + "exit": proc.returncode, + "wall_s": round(wall, 1), + "duration_s": terminal.get("duration_s"), + "rss_mb": round(rss_kb / 1024) if rss_kb else None, + "cgroup_mb": round(cgroup_peak / MB) if cgroup_peak else None, + "event": terminal.get("event"), + } + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--image", default="pdf-ops:bench") + parser.add_argument("--workdir", default="/tmp/pdfops-bench") + args = parser.parse_args() + + base = Path(args.workdir) + fixtures = base / "fixtures" + fixtures.mkdir(parents=True, exist_ok=True) + + print("generating fixtures (cached across runs) ...") + make_pdf(fixtures / "small-a.pdf", 5) + make_pdf(fixtures / "small-b.pdf", 5) + make_pdf(fixtures / "big-a.pdf", 250) + make_pdf(fixtures / "big-b.pdf", 250) + for index in range(20): + make_pdf(fixtures / f"mid-{index:02d}.pdf", 25) + make_encrypted_pdf(fixtures / "big-locked.pdf", 250, "bench-password") + make_attachment_carrier(fixtures / "carrier.pdf", 10, 25) + (fixtures / "pw").write_text("bench-password\n") + + scenarios: list[tuple[str, dict[str, str]]] = [ + ( + "merge 2 x 5 MB (baseline)", + { + "PDFOPS_OPERATION": "merge", + "PDFOPS_INPUTS": "/in/small-a.pdf:/in/small-b.pdf", + "PDFOPS_OUTPUT": "/out/baseline.pdf", + }, + ), + ( + "merge 2 x 250 MB", + { + "PDFOPS_OPERATION": "merge", + "PDFOPS_INPUTS": "/in/big-a.pdf:/in/big-b.pdf", + "PDFOPS_OUTPUT": "/out/big.pdf", + }, + ), + ( + "merge 20 x 25 MB", + { + "PDFOPS_OPERATION": "merge", + "PDFOPS_INPUTS": ":".join(f"/in/mid-{i:02d}.pdf" for i in range(20)), + "PDFOPS_OUTPUT": "/out/many.pdf", + }, + ), + ( + "merge 250 MB AES-256 in, re-encrypted out", + { + "PDFOPS_OPERATION": "merge", + "PDFOPS_INPUTS": "/in/big-locked.pdf", + "PDFOPS_OUTPUT": "/out/relocked.pdf", + "PDFOPS_PASSWORD_FILE": "/in/pw", + "PDFOPS_OUTPUT_ENCRYPTION": "inherit", + }, + ), + ( + "extract 10 x 25 MB attachments", + { + "PDFOPS_OPERATION": "extract", + "PDFOPS_INPUT": "/in/carrier.pdf", + "PDFOPS_OUTPUT_DIR": "/out", + }, + ), + ] + + print(f"{'scenario':44} {'exit':>4} {'op_s':>7} {'rss_mb':>7} {'cgroup_mb':>9}") + results: list[tuple[str, dict[str, object]]] = [] + for name, env in scenarios: + out_dir = base / "out" + shutil.rmtree(out_dir, ignore_errors=True) + out_dir.mkdir() + out_dir.chmod(0o777) + result = run_container(args.image, env, {fixtures: "/in:ro", out_dir: "/out"}) + results.append((name, result)) + print( + f"{name:44} {result['exit']:>4} {result['duration_s']!s:>7} " + f"{result['rss_mb']!s:>7} {result['cgroup_mb']!s:>9}" + ) + + print("\nmarkdown:") + print("| Scenario | Duration | Peak process RSS | Peak cgroup memory |") + print("|---|---|---|---|") + for name, result in results: + print( + f"| {name} | {result['duration_s']} s | ~{result['rss_mb']} MB " + f"| ~{result['cgroup_mb']} MB |" + ) + + +if __name__ == "__main__": + main() diff --git a/src/pdf_ops/__main__.py b/src/pdf_ops/__main__.py index 63a373a..ae08eb3 100644 --- a/src/pdf_ops/__main__.py +++ b/src/pdf_ops/__main__.py @@ -4,8 +4,6 @@ everything else operates on plain mappings and return values. """ -from __future__ import annotations - import os import sys diff --git a/src/pdf_ops/config.py b/src/pdf_ops/config.py index f585fa0..bea88dc 100644 --- a/src/pdf_ops/config.py +++ b/src/pdf_ops/config.py @@ -7,8 +7,6 @@ readability of paths are operation-stage concerns, not configuration ones. """ -from __future__ import annotations - import logging import os from collections.abc import Mapping @@ -17,8 +15,8 @@ from pathlib import Path from typing import ClassVar, Literal -from pdf_ops.errors import ConfigError -from pdf_ops.secret import Secret +from pdf_ops.errors import ConfigError, ErrorCode +from pdf_ops.secrets import EnvSecret, FileSecret, Secret, SecretRef ENV_PREFIX = "PDFOPS_" @@ -109,24 +107,6 @@ class OutputEncryption(StrEnum): ALWAYS = "always" -@dataclass(frozen=True, slots=True) -class EnvSecret: - """A secret supplied directly in an environment variable.""" - - value: Secret - - -@dataclass(frozen=True, slots=True) -class FileSecret: - """A secret to be read from a mounted file (resolved after parsing - - the parser itself stays filesystem-free).""" - - path: Path - - -type SecretRef = EnvSecret | FileSecret - - @dataclass(frozen=True, slots=True) class MergeConfig: operation: ClassVar[Literal[Operation.MERGE]] = Operation.MERGE @@ -196,7 +176,7 @@ def _reject_unknown_vars(env: Mapping[str, str]) -> None: raise ConfigError( f"unknown environment variable(s): {', '.join(unknown)}; " f"accepted: {', '.join(sorted(KNOWN_VARS))}", - error_code="UNKNOWN_VAR", + error_code=ErrorCode.UNKNOWN_VAR, context={"unknown_vars": unknown}, ) @@ -213,7 +193,7 @@ def _reject_inapplicable_vars( if present: raise ConfigError( f"variable(s) not applicable to operation '{operation.value}': {', '.join(present)}", - error_code="INAPPLICABLE_VAR", + error_code=ErrorCode.INAPPLICABLE_VAR, context={"operation": operation.value, "inapplicable_vars": present}, ) @@ -223,7 +203,7 @@ def _parse_operation(env: Mapping[str, str]) -> Operation: if not raw: raise ConfigError( f"{VAR_OPERATION} is required (accepted values: merge, extract)", - error_code="MISSING_VAR", + error_code=ErrorCode.MISSING_VAR, context={"var": VAR_OPERATION}, ) try: @@ -231,7 +211,7 @@ def _parse_operation(env: Mapping[str, str]) -> Operation: except ValueError: raise ConfigError( f"{VAR_OPERATION} has invalid value {raw!r} (accepted values: merge, extract)", - error_code="INVALID_OPERATION", + error_code=ErrorCode.INVALID_OPERATION, context={"var": VAR_OPERATION, "value": raw}, ) from None @@ -245,7 +225,7 @@ def _parse_log_level(env: Mapping[str, str]) -> int: raise ConfigError( f"{VAR_LOG_LEVEL} has invalid value {raw!r} " f"(accepted values: {', '.join(_LOG_LEVELS).lower()}, case-insensitive)", - error_code="INVALID_LOG_LEVEL", + error_code=ErrorCode.INVALID_LOG_LEVEL, context={"var": VAR_LOG_LEVEL, "value": raw}, ) return level @@ -257,7 +237,7 @@ def _parse_inputs(env: Mapping[str, str]) -> tuple[Path, ...]: raise ConfigError( f"{VAR_INPUTS} is required for merge " f"(ordered file paths separated by {INPUTS_SEPARATOR!r})", - error_code="MISSING_VAR", + error_code=ErrorCode.MISSING_VAR, context={"var": VAR_INPUTS}, ) parts = [part.strip() for part in raw.split(INPUTS_SEPARATOR)] @@ -265,7 +245,7 @@ def _parse_inputs(env: Mapping[str, str]) -> tuple[Path, ...]: raise ConfigError( f"{VAR_INPUTS} contains an empty path component " f"(check for stray {INPUTS_SEPARATOR!r} separators)", - error_code="INVALID_INPUTS", + error_code=ErrorCode.INVALID_INPUTS, context={"var": VAR_INPUTS, "value": raw}, ) paths = [Path(part) for part in parts] @@ -280,7 +260,7 @@ def _parse_inputs(env: Mapping[str, str]) -> tuple[Path, ...]: duplicates = sorted({part for part in parts if Path(part) in duplicated_paths}) raise ConfigError( f"{VAR_INPUTS} lists the same path more than once: {', '.join(duplicates)}", - error_code="DUPLICATE_INPUTS", + error_code=ErrorCode.DUPLICATE_INPUTS, context={"var": VAR_INPUTS, "duplicates": duplicates}, ) return tuple(paths) @@ -291,7 +271,7 @@ def _parse_output(env: Mapping[str, str]) -> Path: if not raw: raise ConfigError( f"{VAR_OUTPUT} is required for merge (path of the output PDF)", - error_code="MISSING_VAR", + error_code=ErrorCode.MISSING_VAR, context={"var": VAR_OUTPUT}, ) return Path(raw) @@ -302,7 +282,7 @@ def _parse_single_path(env: Mapping[str, str], var: str, purpose: str) -> Path: if not raw: raise ConfigError( f"{var} is required for extract ({purpose})", - error_code="MISSING_VAR", + error_code=ErrorCode.MISSING_VAR, context={"var": var}, ) return Path(raw) @@ -317,7 +297,7 @@ def _parse_flag(env: Mapping[str, str], var: str, *, default: bool = False) -> b return normalized == "true" raise ConfigError( f"{var} has invalid value {raw!r} (accepted values: true, false, case-insensitive)", - error_code="INVALID_FLAG", + error_code=ErrorCode.INVALID_FLAG, context={"var": var, "value": raw}, ) @@ -334,7 +314,7 @@ def _parse_secret_pair(env: Mapping[str, str], value_var: str, file_var: str) -> if raw_value and raw_file: raise ConfigError( f"{value_var} and {file_var} are mutually exclusive - supply one", - error_code="CONFLICTING_PASSWORD_SOURCES", + error_code=ErrorCode.CONFLICTING_PASSWORD_SOURCES, context={"vars": [value_var, file_var]}, ) if raw_value: @@ -354,7 +334,7 @@ def _parse_on_exists(env: Mapping[str, str]) -> OnExists: raise ConfigError( f"{VAR_ON_EXISTS} has invalid value {raw!r} " "(accepted values: fail, overwrite, skip, case-insensitive)", - error_code="INVALID_ON_EXISTS", + error_code=ErrorCode.INVALID_ON_EXISTS, context={"var": VAR_ON_EXISTS, "value": raw}, ) from None @@ -369,7 +349,7 @@ def _parse_output_encryption(env: Mapping[str, str]) -> OutputEncryption: raise ConfigError( f"{VAR_OUTPUT_ENCRYPTION} has invalid value {raw!r} " "(accepted values: never, inherit, always, case-insensitive)", - error_code="INVALID_OUTPUT_ENCRYPTION", + error_code=ErrorCode.INVALID_OUTPUT_ENCRYPTION, context={"var": VAR_OUTPUT_ENCRYPTION, "value": raw}, ) from None @@ -383,94 +363,14 @@ def _parse_output_password(env: Mapping[str, str], password: SecretRef | None) - raise ConfigError( f"an output password is supplied but {VAR_OUTPUT_ENCRYPTION} is 'never' " "(set it to 'inherit' or 'always', or remove the output password)", - error_code="OUTPUT_PASSWORD_WITHOUT_ENCRYPTION", + error_code=ErrorCode.OUTPUT_PASSWORD_WITHOUT_ENCRYPTION, context={"var": VAR_OUTPUT_ENCRYPTION}, ) if mode is OutputEncryption.ALWAYS and output_password is None and password is None: raise ConfigError( f"{VAR_OUTPUT_ENCRYPTION}=always requires an output password " f"({VAR_OUTPUT_PASSWORD_FILE}/{VAR_OUTPUT_PASSWORD}) or an input password to fall back to", - error_code="MISSING_OUTPUT_PASSWORD", + error_code=ErrorCode.MISSING_OUTPUT_PASSWORD, context={"var": VAR_OUTPUT_ENCRYPTION}, ) return output_password - - -def resolve_secret(ref: SecretRef | None) -> Secret | None: - """Materialize a secret reference; the one place secret file I/O happens. - - A single trailing newline is stripped (hand-created secret files usually - have one; the password itself is otherwise taken byte-for-byte). - """ - match ref: - case None: - return None - case EnvSecret(value=value): - _reject_control_characters(value.reveal(), "the environment") - return value - case FileSecret(path=path): - try: - raw = path.read_text(encoding="utf-8") - except OSError as err: - raise ConfigError( - f"cannot read password file {path}: {err.strerror or err}", - error_code="PASSWORD_FILE_UNREADABLE", - context={"path": str(path)}, - ) from err - except UnicodeDecodeError as err: - # Deliberately no decode detail: it would name a byte of the - # secret and its position. - raise ConfigError( - f"password file {path} is not valid UTF-8 text", - error_code="PASSWORD_FILE_UNREADABLE", - context={"path": str(path)}, - ) from err - raw = raw.removesuffix("\n").removesuffix("\r") - if not raw: - raise ConfigError( - f"password file {path} is empty", - error_code="EMPTY_PASSWORD", - context={"path": str(path)}, - ) - _reject_control_characters(raw, str(path)) - return Secret(raw) - - -def _reject_control_characters(value: str, origin: str) -> None: - """A password containing control characters is almost certainly an - encoding or copy-paste accident - and downstream cryptographic - normalization (SASLprep) would warn about it, naming the codepoint.""" - if any(ord(ch) < 32 or 0x7F <= ord(ch) <= 0x9F for ch in value): - raise ConfigError( - f"the password from {origin} contains control characters " - "(check for encoding or copy-paste issues)", - error_code="PASSWORD_UNSUPPORTED_CHARACTERS", - context={"source": origin}, - ) - - -@dataclass(frozen=True, slots=True) -class Secrets: - """Materialized secrets for one run, resolved from the config's refs.""" - - password: Secret | None - output_password: Secret | None - - -def resolve_secrets(config: Config) -> Secrets: - """Read any file-based secrets; raises ConfigError on unreadable/empty files.""" - output_password = ( - resolve_secret(config.output_password) if isinstance(config, MergeConfig) else None - ) - return Secrets(password=resolve_secret(config.password), output_password=output_password) - - -def describe_secret(ref: SecretRef | None) -> str: - """Presence-only description for the config-echo log event.""" - match ref: - case None: - return "unset" - case EnvSecret(): - return "set(env)" - case FileSecret(): - return "set(file)" diff --git a/src/pdf_ops/engine.py b/src/pdf_ops/engine.py index 2a3b473..9bea013 100644 --- a/src/pdf_ops/engine.py +++ b/src/pdf_ops/engine.py @@ -10,14 +10,16 @@ before any output work starts. """ -from __future__ import annotations - from collections.abc import Sequence from dataclasses import dataclass from pathlib import Path -from typing import Protocol +from typing import Literal, Protocol + +from pdf_ops.secrets import Secret -from pdf_ops.secret import Secret +# How an encrypted input opened: with the user or owner password supplied, +# or through the spec-standard empty-password try. +type PasswordKind = Literal["user", "owner", "empty"] @dataclass(frozen=True, slots=True) @@ -37,7 +39,21 @@ class OpenedInput: pages: int encrypted: bool algorithm: str | None - password_type: str | None + password_type: PasswordKind | None + # Recoverable-damage messages the library reported while parsing + # ("repairing", xref rebuilt, ...). The operation layer surfaces them as + # events; anything unrecoverable raises instead. + warnings: tuple[str, ...] = () + + def event_fields(self) -> dict[str, str | int | bool | None]: + """The ``input_opened`` payload - one schema for every operation.""" + return { + "input": str(self.path), + "pages": self.pages, + "encrypted": self.encrypted, + "algorithm": self.algorithm, + "password_type": self.password_type, + } @dataclass(frozen=True, slots=True) @@ -69,11 +85,13 @@ def merge_to( inputs: Sequence[OpenedInput], destination: Path, output_password: Secret | None, - ) -> None: + ) -> list[str]: """Merge ``inputs`` (in order) into a PDF at ``destination``, AES-256-encrypted with ``output_password`` when given. ``destination`` is a temp path provided by the atomic-write layer. + Returns library warnings raised during the write (sources may be + read lazily, so repairs can surface here rather than at open time). """ ... @@ -82,9 +100,14 @@ def list_attachments(self, opened: OpenedInput) -> list[Attachment]: (deterministic across runs). Duplicate names are preserved.""" ... + def collect_warnings(self, opened: OpenedInput) -> list[str]: + """Library warnings accumulated on ``opened`` since the last harvest + (lazy readers keep discovering repairs after open time).""" + ... + def get_engine() -> PdfEngine: """The single swap point for the PDF library backing the operations.""" - from pdf_ops.engine_pypdf import PypdfEngine + from pdf_ops.engine_pikepdf import PikepdfEngine - return PypdfEngine() + return PikepdfEngine() diff --git a/src/pdf_ops/engine_pikepdf.py b/src/pdf_ops/engine_pikepdf.py new file mode 100644 index 0000000..c844888 --- /dev/null +++ b/src/pdf_ops/engine_pikepdf.py @@ -0,0 +1,355 @@ +"""pikepdf-backed engine implementation. + +The only module that imports pikepdf. Translates qpdf's failure modes into +the application taxonomy so callers never see library-specific errors, and is +the only code that calls ``Secret.reveal()``. +""" + +import re +import warnings +from collections.abc import Generator, Sequence +from contextlib import contextmanager +from pathlib import Path +from typing import Any, cast + +import pikepdf + +from pdf_ops.engine import Attachment, OpenedInput, PasswordKind +from pdf_ops.errors import ErrorCode, InvalidPdfError, PasswordError +from pdf_ops.secrets import Secret + +# pikepdf converts PDF integers/reals/booleans/null to native Python values, +# so type-blind access on attacker-controlled structures (a name tree whose +# node is an integer, a filespec whose /EF is a number, a dangling reference +# resolving to None) raises builtin exceptions, not PdfError. These catches +# wrap ONLY walks over document structure, so a builtin exception there means +# a file shape the engine cannot process, never a bug in our code. OSError +# deliberately excluded: I/O failures must keep their own classification. +_STRUCTURE_FAILURES = ( + AttributeError, + KeyError, + IndexError, + TypeError, + ValueError, + RecursionError, +) + + +class PikepdfEngine: + def open_input(self, path: Path, password: Secret | None) -> OpenedInput: + supplied = password.reveal() if password is not None else "" + used = supplied + try: + pdf = _open_quietly(path, supplied) + except pikepdf.PasswordError: + # No handle exists on this path, so the algorithm label comes + # from a raw scan of the plaintext /Encrypt dictionary. + algorithm = _describe_encryption_raw(path) + if password is None: + raise PasswordError( + f"{path} is encrypted ({algorithm}) and requires a password", + error_code=ErrorCode.PASSWORD_REQUIRED, + context={"input": str(path), "algorithm": algorithm}, + ) from None + # The supplied password failed - but this input may not need it + # at all (a permissions-locked file among user-locked ones in a + # merge). The spec-standard empty try still applies; qpdf only + # tries the string it was given. + try: + pdf = _open_quietly(path, "") + used = "" + except pikepdf.PasswordError: + raise PasswordError( + f"the supplied password does not open {path} ({algorithm})", + error_code=ErrorCode.WRONG_PASSWORD, + context={"input": str(path), "algorithm": algorithm}, + ) from None + except pikepdf.PdfError as err: + raise _corrupt(path, err) from err + except pikepdf.PdfError as err: + if _mentions_encryption(path) and "encrypt" in str(err).lower(): + # Certificate security handlers, exotic revisions: an + # encryption problem, not a malformed file - the operator + # remedy lives in the password class. + raise PasswordError( + f"{path} uses an encryption scheme this build cannot process", + error_code=ErrorCode.UNSUPPORTED_ENCRYPTION, + context={"input": str(path)}, + ) from err + raise _corrupt(path, err) from err + + encrypted = bool(pdf.is_encrypted) + algorithm = _describe_encryption(pdf) if encrypted else None + password_type: PasswordKind | None = None + if encrypted: + if not used: + password_type = "empty" + else: + # qpdf records which of the two document passwords the + # supplied string matched. + password_type = "owner" if pdf.owner_password_matched else "user" + + with _translating(path): + pages = len(pdf.pages) + + return OpenedInput( + path=path, + handle=pdf, + pages=pages, + encrypted=encrypted, + algorithm=algorithm, + password_type=password_type, + # qpdf reports recoverable damage ("repairing", xref rebuilds) + # through its warning channel, not Python logging; carried here + # so the operation layer can surface them as JSON events. + warnings=tuple(str(message) for message in pdf.get_warnings()), + ) + + def merge_to( + self, + inputs: Sequence[OpenedInput], + destination: Path, + output_password: Secret | None, + ) -> list[str]: + with pikepdf.Pdf.new() as merged: + for opened in inputs: + source = cast(pikepdf.Pdf, opened.handle) + with _translating(opened.path): + merged.pages.extend(source.pages) + + encryption = None + if output_password is not None: + raw = output_password.reveal() + # R=6 pinned explicitly: AES-256, never a legacy scheme. + encryption = pikepdf.Encryption(user=raw, owner=raw, R=6) + try: + # Saved through the already-open temp file, NOT by path: given + # a path to an existing file, pikepdf routes through its own + # atomic-overwrite (a second hidden temp in the destination + # directory that a kill would orphan outside the stale-temp + # naming scheme). Writing into the atomic-write layer's temp + # keeps one temp file, one rename, and the mode it set. + with destination.open("wb") as handle: + if encryption is not None: + merged.save(handle, encryption=encryption) + else: + merged.save(handle) + except pikepdf.PdfError as err: + # qpdf copies source streams lazily at save, so a source that + # turns out unreadable surfaces here without attribution. + raise InvalidPdfError( + f"a merge input could not be fully read while writing: {err}", + error_code=ErrorCode.CORRUPT_PDF, + context={"inputs": [str(one.path) for one in inputs]}, + ) from err + # Repairs discovered during the lazy copy accumulate on the + # sources and the writer after open-time harvesting. + late: list[str] = [str(m) for m in merged.get_warnings()] + for opened in inputs: + late.extend(self.collect_warnings(opened)) + return late + + def list_attachments(self, opened: OpenedInput) -> list[Attachment]: + pdf = cast(pikepdf.Pdf, opened.handle) + with _translating(opened.path): + entries = _embedded_file_entries(pdf) + + attachments: list[Attachment] = [] + for raw_name, spec in entries: + with _translating(opened.path): + embedded: Any = spec.get("/EF") if isinstance(spec, pikepdf.Dictionary) else None + if embedded is None or not isinstance(embedded, pikepdf.Dictionary): + # A bare file reference (no embedded stream) or a + # malformed /EF: nothing extractable in this entry. + continue + stream: Any = embedded.get("/F") + if stream is None: + stream = embedded.get("/UF") + if stream is None: + continue + data = bytes(stream.read_bytes()) + # A name-tree key must be a PDF string; qpdf hands anything else + # over as a native Python value (an integer key would make a + # bytes()/str() conversion attacker-sized). A non-string key gets + # the deterministic fallback name downstream - the payload is + # still extracted. str() of a pikepdf.String decodes + # PDFDocEncoding and UTF-16 per spec. + name = str(raw_name) if isinstance(raw_name, pikepdf.String) else "" + attachments.append(Attachment(name=name, data=data)) + return attachments + + def collect_warnings(self, opened: OpenedInput) -> list[str]: + """Warnings qpdf accumulated on the handle since the last harvest.""" + return [str(m) for m in cast(pikepdf.Pdf, opened.handle).get_warnings()] + + +def _open_quietly(path: Path, password: str) -> pikepdf.Pdf: + """Open, muting pikepdf's password-was-not-needed UserWarning: the + operation layer already reports that case as a ``password_unused`` event + with more context, and a duplicate through the warnings channel would + only add noise.""" + with warnings.catch_warnings(): + warnings.filterwarnings("ignore", message="A password was provided") + return pikepdf.open(path, password=password) + + +def _embedded_file_entries(pdf: pikepdf.Pdf) -> list[tuple[Any, Any]]: + """(name, filespec) pairs from /Names/EmbeddedFiles, in tree order. + + Walked directly rather than through ``pdf.attachments``: that Mapping + collapses duplicate attachment names, which real PDFs contain and the + extraction contract preserves. + """ + root: Any = pdf.Root + names_dict: Any = root.get("/Names") + if not isinstance(names_dict, pikepdf.Dictionary): + return [] + tree: Any = names_dict.get("/EmbeddedFiles") + if tree is None: + return [] + entries: list[tuple[Any, Any]] = [] + _walk_name_tree(tree, entries, set()) + return entries + + +def _walk_name_tree(node: Any, out: list[tuple[Any, Any]], seen: set[tuple[int, int]]) -> None: + if not isinstance(node, pikepdf.Dictionary): + # A node that is not a dictionary (integer, null from a dangling + # reference, ...) cannot carry entries; hostile trees do this. + return + objgen = cast("tuple[int, int]", tuple(node.objgen)) + if objgen != (0, 0): # (0, 0) means a direct object - not a stable identity + if objgen in seen: # hostile trees can contain reference cycles + return + seen.add(objgen) + kids: Any = node.get("/Kids") + if isinstance(kids, pikepdf.Array): + for kid in kids: + _walk_name_tree(kid, out, seen) + names: Any = node.get("/Names") + if isinstance(names, pikepdf.Array): + out.extend((names[index], names[index + 1]) for index in range(0, len(names) - 1, 2)) + + +def _describe_encryption(pdf: pikepdf.Pdf) -> str: + """Human label for the encryption of a successfully opened file.""" + try: + info = pdf.encryption + version = int(info.V) + bits = int(info.bits) + if version == 5: + return "AES-256" + if version == 4: + return "AES-128" if "aes" in str(info.stream_method).lower() else f"RC4-{bits}" + if version in (1, 2): + return f"RC4-{bits}" + return f"V{version}" + except Exception: + return "unknown" + + +# Bounded digit runs with a lookahead cutoff: these tokens are read from raw, +# attacker-controlled bytes, and an unbounded \d+ capture would both match +# decoy digit walls and feed int() something conversion-limited. +_ENCRYPT_REF = re.compile(rb"/Encrypt\s+(\d{1,9})(?!\d)\s+(\d{1,5})(?!\d)\s+R") +_VERSION_TOKEN = re.compile(rb"/V\s+(\d{1,3})(?!\d)") +_LENGTH_TOKEN = re.compile(rb"/Length\s+(\d{1,5})(?!\d)") + +# Reading cap for the failed-open scan: files larger than this only get their +# tail examined, degrading the label to "unknown" when the /Encrypt object +# sits earlier - acceptable for an error-context label. +_RAW_SCAN_LIMIT = 32 * 1024 * 1024 + + +def _describe_encryption_raw(path: Path) -> str: + """Best-effort label when the file could not be opened at all. + + The /Encrypt dictionary is plaintext, but without a handle the only way + to it is a raw scan: the trailer names the object (``/Encrypt N G R``) + and the tokens are read from that object's own body, so a stray ``/V`` + elsewhere (a form-field value, a decoy) is not mistaken for the + encryption version. Anything unresolvable degrades to "unknown" - this + only ever feeds error-context observability. + """ + try: + size = path.stat().st_size + with path.open("rb") as handle: + if size > _RAW_SCAN_LIMIT: + handle.seek(size - _RAW_SCAN_LIMIT) + data = handle.read(_RAW_SCAN_LIMIT) + except OSError: + return "unknown" + + reference = None + for match in _ENCRYPT_REF.finditer(data): + reference = match # the trailer (last occurrence) is authoritative + if reference is None: + return "unknown" + header = b"%d %d obj" % (int(reference.group(1)), int(reference.group(2))) + at = data.rfind(header) + if at < 0: + return "unknown" + body = data[at : at + 2048] + + version_match = _VERSION_TOKEN.search(body) + if version_match is None: + return "unknown" + version = int(version_match.group(1)) + length_match = _LENGTH_TOKEN.search(body) + length = int(length_match.group(1)) if length_match else 40 + if version == 5: + return "AES-256" + if version == 4: + return "AES-128" if b"/AESV2" in body else f"RC4-{length}" + if version == 2: + return f"RC4-{length}" + if version == 1: + return "RC4-40" + return f"V{version}" + + +def _mentions_encryption(path: Path) -> bool: + """Best-effort check whether a file qpdf refused to open carries an + /Encrypt dictionary (it lives near the trailer).""" + try: + with path.open("rb") as handle: + handle.seek(max(0, path.stat().st_size - 8192)) + return b"/Encrypt" in handle.read() + except OSError: + return False + + +@contextmanager +def _translating(path: Path) -> Generator[None]: + """Translate pikepdf's failure modes while walking ``path``'s structure. + + Wraps ONLY walks over document structure (see ``_STRUCTURE_FAILURES``): + a builtin exception inside means a file shape the engine cannot process, + never a bug in our code. The open and save paths keep their finer handling. + """ + try: + yield + except pikepdf.PdfError as err: + raise _translated_data_error(path, err) from err + except _STRUCTURE_FAILURES as err: + raise _corrupt(path, err) from err + + +def _translated_data_error(path: Path, err: Exception) -> InvalidPdfError: + if "unfilterable" in str(err): + # A stream filter qpdf cannot decode: a permanent, data-dependent + # condition distinct from structural corruption. + return InvalidPdfError( + f"{path} uses a PDF feature this build cannot process: {err}", + error_code=ErrorCode.UNSUPPORTED_PDF_FEATURE, + context={"input": str(path)}, + ) + return _corrupt(path, err) + + +def _corrupt(path: Path, err: Exception) -> InvalidPdfError: + return InvalidPdfError( + f"cannot parse {path} as a PDF: {err}", + error_code=ErrorCode.CORRUPT_PDF, + context={"input": str(path)}, + ) diff --git a/src/pdf_ops/engine_pypdf.py b/src/pdf_ops/engine_pypdf.py deleted file mode 100644 index 1845c5e..0000000 --- a/src/pdf_ops/engine_pypdf.py +++ /dev/null @@ -1,250 +0,0 @@ -"""pypdf-backed engine implementation. - -The only module that imports pypdf. Translates pypdf's failure modes into the -application taxonomy so callers never see library-specific errors, and is the -only code that calls ``Secret.reveal()``. -""" - -from __future__ import annotations - -from collections.abc import Sequence -from pathlib import Path -from typing import Any, cast - -from pypdf import PasswordType, PdfReader, PdfWriter -from pypdf.errors import DependencyError, PyPdfError - -from pdf_ops.engine import Attachment, OpenedInput -from pdf_ops.errors import InvalidPdfError, PasswordError -from pdf_ops.secret import Secret - -# pypdf leaks builtin exceptions (AttributeError, KeyError, ...) on some -# pathological files whose header and xref are valid - e.g. a catalog with no -# /Pages raises AttributeError deep in page-tree flattening. These catches -# wrap ONLY pypdf calls, so a builtin exception here means a file pypdf cannot -# process, not a bug in our code. OSError deliberately excluded: I/O failures -# must keep their own classification. -_PARSE_FAILURES = ( - PyPdfError, - AttributeError, - KeyError, - IndexError, - TypeError, - ValueError, - RecursionError, -) - -# Raised while decoding streams whose filter pypdf does not implement -# (NotImplementedError, e.g. an unknown /Filter) or needs an unavailable -# external capability for (DependencyError, e.g. jbig2dec). Permanent, -# data-dependent conditions - never internal errors. -_UNSUPPORTED_FEATURES = (DependencyError, NotImplementedError) - - -class PypdfEngine: - def open_input(self, path: Path, password: Secret | None) -> OpenedInput: - try: - reader = PdfReader(path) - except DependencyError as err: - # At construction time this means auto-decryption needed a - # cryptography capability this build does not ship. - raise PasswordError( - f"{path} uses encryption this build cannot process", - error_code="UNSUPPORTED_ENCRYPTION", - context={"input": str(path)}, - ) from err - except NotImplementedError as err: - if _mentions_encryption(path): - # Certificate security handlers, exotic /V values, etc.: an - # encryption problem, not a malformed file - the operator - # remedy lives in the password class. - raise PasswordError( - f"{path} uses an encryption scheme this build cannot process", - error_code="UNSUPPORTED_ENCRYPTION", - context={"input": str(path)}, - ) from err - raise _unsupported(path, err) from err - except _PARSE_FAILURES as err: - raise _corrupt(path, err) from err - - encrypted = bool(reader.is_encrypted) - algorithm: str | None = None - password_type: str | None = None - if encrypted: - # The /Encrypt dictionary is plaintext: the algorithm is known - # before any password attempt. - algorithm = _describe_encryption(reader) - password_type = _decrypt(reader, path, password, algorithm) - - try: - # Force xref/page-tree resolution now; pypdf parses lazily and - # would otherwise surface corruption mid-write. - pages = len(reader.pages) - except _UNSUPPORTED_FEATURES as err: - raise _unsupported(path, err) from err - except _PARSE_FAILURES as err: - raise _corrupt(path, err) from err - - return OpenedInput( - path=path, - handle=reader, - pages=pages, - encrypted=encrypted, - algorithm=algorithm, - password_type=password_type, - ) - - def merge_to( - self, - inputs: Sequence[OpenedInput], - destination: Path, - output_password: Secret | None, - ) -> None: - writer = PdfWriter() - for opened in inputs: - reader = cast(PdfReader, opened.handle) - try: - writer.append(reader) - except _UNSUPPORTED_FEATURES as err: - raise _unsupported(opened.path, err) from err - except _PARSE_FAILURES as err: - raise _corrupt(opened.path, err) from err - - if output_password is not None: - # algorithm passed explicitly: pypdf's default is legacy RC4 for - # backwards compatibility - never acceptable for new output. - writer.encrypt(user_password=output_password.reveal(), algorithm="AES-256") - - with destination.open("wb") as handle: - writer.write(handle) - - def list_attachments(self, opened: OpenedInput) -> list[Attachment]: - reader = cast(PdfReader, opened.handle) - try: - # attachment_list walks the document-level /Names/EmbeddedFiles - # tree in name order and preserves duplicate names. - items = list(reader.attachment_list) - except _UNSUPPORTED_FEATURES as err: - raise _unsupported(opened.path, err) from err - except _PARSE_FAILURES as err: - raise _corrupt(opened.path, err) from err - - attachments: list[Attachment] = [] - for item in items: - try: - # pypdf annotates name as str, but raw name trees can yield - # byte strings at runtime - keep the defensive type. - raw_name = cast("str | bytes", item.name) - content = item.content - except _UNSUPPORTED_FEATURES as err: - raise _unsupported(opened.path, err) from err - except _PARSE_FAILURES as err: - raise _corrupt(opened.path, err) from err - # Spec-legal name trees can carry non-UTF-8 byte strings; the - # sanitizer downstream expects str, so decode lossily rather than - # letting one odd entry abort the run. - name = ( - raw_name - if isinstance(raw_name, str) - else bytes(raw_name).decode("utf-8", errors="replace") - ) - attachments.append(Attachment(name=name, data=bytes(content))) - return attachments - - -def _decrypt(reader: PdfReader, path: Path, password: Secret | None, algorithm: str | None) -> str: - """Decrypt with the supplied password, or the spec-standard empty try. - - Returns how the file opened: ``user``/``owner``/``empty``. pypdf's - ``decrypt()`` reports failure through its return value, not an exception - - the check must be explicit. - """ - supplied = password.reveal() if password is not None else "" - try: - result = reader.decrypt(supplied) - except DependencyError as err: - raise PasswordError( - f"{path} uses encryption this build cannot process ({algorithm})", - error_code="UNSUPPORTED_ENCRYPTION", - context={"input": str(path), "algorithm": algorithm}, - ) from err - except _PARSE_FAILURES as err: - raise _corrupt(path, err) from err - - if result == PasswordType.NOT_DECRYPTED: - if password is None: - raise PasswordError( - f"{path} is encrypted ({algorithm}) and requires a password", - error_code="PASSWORD_REQUIRED", - context={"input": str(path), "algorithm": algorithm}, - ) - # The supplied password failed - but this input may not need it at - # all (a permissions-locked file among user-locked ones in a merge). - # The spec-standard empty try still applies before giving up. - try: - empty_result = reader.decrypt("") - except _PARSE_FAILURES as err: - raise _corrupt(path, err) from err - if empty_result != PasswordType.NOT_DECRYPTED: - return "empty" - raise PasswordError( - f"the supplied password does not open {path} ({algorithm})", - error_code="WRONG_PASSWORD", - context={"input": str(path), "algorithm": algorithm}, - ) - if not supplied: - return "empty" - return "owner" if result == PasswordType.OWNER_PASSWORD else "user" - - -def _describe_encryption(reader: PdfReader) -> str: - """Best-effort human label for the /Encrypt dictionary (plaintext - metadata - readable before any password attempt).""" - try: - encrypt_obj: Any = reader.trailer.get("/Encrypt") - if encrypt_obj is None: - return "unknown" - encrypt: Any = encrypt_obj.get_object() - version = int(encrypt.get("/V", 0)) - length = int(encrypt.get("/Length", 40)) - if version == 5: - return "AES-256" - if version == 4: - crypt_filters: Any = encrypt.get("/CF") - if crypt_filters is not None and "/AESV2" in str(crypt_filters): - return "AES-128" - return f"RC4-{length}" - if version == 2: - return f"RC4-{length}" - if version == 1: - return "RC4-40" - return f"V{version}" - except Exception: - return "unknown" - - -def _mentions_encryption(path: Path) -> bool: - """Best-effort check whether a file that pypdf refused to construct a - reader for carries an /Encrypt dictionary (it lives near the trailer).""" - try: - with path.open("rb") as handle: - handle.seek(max(0, path.stat().st_size - 8192)) - return b"/Encrypt" in handle.read() - except OSError: - return False - - -def _corrupt(path: Path, err: Exception) -> InvalidPdfError: - return InvalidPdfError( - f"cannot parse {path} as a PDF: {err}", - error_code="CORRUPT_PDF", - context={"input": str(path)}, - ) - - -def _unsupported(path: Path, err: Exception) -> InvalidPdfError: - return InvalidPdfError( - f"{path} uses a PDF feature this build cannot process: {err}", - error_code="UNSUPPORTED_PDF_FEATURE", - context={"input": str(path)}, - ) diff --git a/src/pdf_ops/errors.py b/src/pdf_ops/errors.py index 4990744..6b96a4e 100644 --- a/src/pdf_ops/errors.py +++ b/src/pdf_ops/errors.py @@ -7,9 +7,7 @@ carried by every raised error and emitted in the terminal log event. """ -from __future__ import annotations - -from enum import IntEnum +from enum import IntEnum, StrEnum from typing import Any @@ -25,10 +23,58 @@ class ExitCode(IntEnum): OUTPUT = 6 +class ErrorCode(StrEnum): + """The machine-readable vocabulary carried by every ``operation_failed`` event. + + Grouped by the exit class each code travels with. The complete table with + meanings lives in docs/OPERATIONS.md; a test keeps the two in sync, so a + new code cannot ship undocumented. + """ + + # exit 1 - unexpected + UNEXPECTED_ERROR = "UNEXPECTED_ERROR" + # exit 2 - configuration + UNKNOWN_VAR = "UNKNOWN_VAR" + INAPPLICABLE_VAR = "INAPPLICABLE_VAR" + MISSING_VAR = "MISSING_VAR" + INVALID_OPERATION = "INVALID_OPERATION" + INVALID_LOG_LEVEL = "INVALID_LOG_LEVEL" + INVALID_INPUTS = "INVALID_INPUTS" + DUPLICATE_INPUTS = "DUPLICATE_INPUTS" + INVALID_FLAG = "INVALID_FLAG" + INVALID_ON_EXISTS = "INVALID_ON_EXISTS" + INVALID_OUTPUT_ENCRYPTION = "INVALID_OUTPUT_ENCRYPTION" + CONFLICTING_PASSWORD_SOURCES = "CONFLICTING_PASSWORD_SOURCES" + OUTPUT_PASSWORD_WITHOUT_ENCRYPTION = "OUTPUT_PASSWORD_WITHOUT_ENCRYPTION" + MISSING_OUTPUT_PASSWORD = "MISSING_OUTPUT_PASSWORD" + PASSWORD_FILE_UNREADABLE = "PASSWORD_FILE_UNREADABLE" + EMPTY_PASSWORD = "EMPTY_PASSWORD" + PASSWORD_UNSUPPORTED_CHARACTERS = "PASSWORD_UNSUPPORTED_CHARACTERS" + # exit 3 - input + INPUT_MISSING = "INPUT_MISSING" + INPUT_IS_DIRECTORY = "INPUT_IS_DIRECTORY" + INPUT_UNREADABLE = "INPUT_UNREADABLE" + NO_ATTACHMENTS = "NO_ATTACHMENTS" + # exit 4 - invalid PDF + NOT_A_PDF = "NOT_A_PDF" + CORRUPT_PDF = "CORRUPT_PDF" + UNSUPPORTED_PDF_FEATURE = "UNSUPPORTED_PDF_FEATURE" + # exit 5 - password + PASSWORD_REQUIRED = "PASSWORD_REQUIRED" + WRONG_PASSWORD = "WRONG_PASSWORD" + UNSUPPORTED_ENCRYPTION = "UNSUPPORTED_ENCRYPTION" + # exit 6 - output + OUTPUT_DIR_MISSING = "OUTPUT_DIR_MISSING" + OUTPUT_IS_DIRECTORY = "OUTPUT_IS_DIRECTORY" + OUTPUT_EXISTS = "OUTPUT_EXISTS" + OUTPUT_NOT_WRITABLE = "OUTPUT_NOT_WRITABLE" + DISK_FULL = "DISK_FULL" + + class PdfOpsError(Exception): """Base class for every predictable failure. - ``error_code`` is a stable machine-readable token (e.g. ``MISSING_VAR``); + ``error_code`` is a stable machine-readable token from ``ErrorCode``; ``context`` holds structured detail for the failure log event. Neither may ever contain secret material - messages carry paths and names, not values. """ @@ -39,7 +85,7 @@ def __init__( self, message: str, *, - error_code: str, + error_code: ErrorCode, context: dict[str, Any] | None = None, ) -> None: super().__init__(message) diff --git a/src/pdf_ops/extract.py b/src/pdf_ops/extract.py index 2377eb2..4f0f169 100644 --- a/src/pdf_ops/extract.py +++ b/src/pdf_ops/extract.py @@ -6,17 +6,16 @@ write is verified to stay inside the output directory. """ -from __future__ import annotations - import logging from dataclasses import dataclass from typing import Any -from pdf_ops.config import ExtractConfig, OnExists, Secrets +from pdf_ops.config import ExtractConfig, OnExists from pdf_ops.engine import Attachment, get_engine -from pdf_ops.errors import InputError, OutputError -from pdf_ops.merge import validate_inputs +from pdf_ops.errors import ErrorCode, InputError, OutputError +from pdf_ops.inputs import validate_inputs from pdf_ops.output import atomic_output, check_output_dir, clean_stale_temps +from pdf_ops.secrets import Secrets # Filesystem NAME_MAX is 255 bytes on the relevant filesystems; leave room # for collision suffixes and the atomic-write temp prefix. @@ -31,16 +30,9 @@ def run_extract(config: ExtractConfig, secrets: Secrets, logger: logging.Logger) engine = get_engine() opened = engine.open_input(config.input, secrets.password) - logger.info( - "input_opened", - extra={ - "input": str(config.input), - "pages": opened.pages, - "encrypted": opened.encrypted, - "algorithm": opened.algorithm, - "password_type": opened.password_type, - }, - ) + logger.info("input_opened", extra=opened.event_fields()) + for message in opened.warnings: + logger.warning("pdf_library_message", extra={"detail": message, "source": "qpdf"}) if secrets.password is not None and not opened.encrypted: logger.warning( "password_unused", @@ -48,12 +40,15 @@ def run_extract(config: ExtractConfig, secrets: Secrets, logger: logging.Logger) ) attachments = engine.list_attachments(opened) + for message in engine.collect_warnings(opened): + # attachment streams are read lazily, so repairs can surface here + logger.warning("pdf_library_message", extra={"detail": message, "source": "qpdf"}) if not attachments: if config.fail_on_no_attachments: raise InputError( f"{config.input} contains no embedded attachments " "(failing because PDFOPS_FAIL_ON_NO_ATTACHMENTS=true)", - error_code="NO_ATTACHMENTS", + error_code=ErrorCode.NO_ATTACHMENTS, context={"input": str(config.input)}, ) return {"attachments_extracted": 0, "bytes_written": 0} @@ -71,7 +66,7 @@ def run_extract(config: ExtractConfig, secrets: Secrets, logger: logging.Logger) if directories: raise OutputError( f"target name(s) are directories in {config.output_dir}: {', '.join(directories)}", - error_code="OUTPUT_IS_DIRECTORY", + error_code=ErrorCode.OUTPUT_IS_DIRECTORY, context={"output_dir": str(config.output_dir), "directories": directories}, ) @@ -93,7 +88,7 @@ def run_extract(config: ExtractConfig, secrets: Secrets, logger: logging.Logger) f"{len(conflicts)} file(s) already exist in {config.output_dir}: " f"{', '.join(conflicts)} (refusing to overwrite; " "set PDFOPS_ON_EXISTS to overwrite or skip for retry semantics)", - error_code="OUTPUT_EXISTS", + error_code=ErrorCode.OUTPUT_EXISTS, context={"output_dir": str(config.output_dir), "conflicts": conflicts}, ) case OnExists.SKIP: @@ -141,7 +136,9 @@ def run_extract(config: ExtractConfig, secrets: Secrets, logger: logging.Logger) "attachment_extracted", extra={ "attachment": item.name, - "original_name": item.original if item.original != item.name else None, + # capped: the original is attacker-controlled and can be + # arbitrarily long - it must not balloon the log stream + "original_name": (item.original[:200] if item.original != item.name else None), "bytes": len(item.data), }, ) diff --git a/src/pdf_ops/inputs.py b/src/pdf_ops/inputs.py new file mode 100644 index 0000000..6845241 --- /dev/null +++ b/src/pdf_ops/inputs.py @@ -0,0 +1,64 @@ +"""Up-front input validation shared by every operation. + +Both operations check their inputs before anything is written; the probe +and the problem classification live here so merge and extract cannot +drift apart. +""" + +from collections.abc import Sequence +from pathlib import Path + +from pdf_ops.errors import ErrorCode, InputError, InvalidPdfError, PdfOpsError + +PDF_MAGIC = b"%PDF-" + +# Problem kinds found during input validation, in exit-code class order. +_INPUT_PROBLEMS = frozenset( + {ErrorCode.INPUT_MISSING, ErrorCode.INPUT_IS_DIRECTORY, ErrorCode.INPUT_UNREADABLE} +) + + +def validate_inputs(inputs: Sequence[Path]) -> None: + """Check every input up front and report all problems in one failure. + + An operator fixing a broken workflow should learn about every bad input + from a single run, not one per retry. The raised error's class (and thus + the exit code) follows the first problem in input order; the full list + travels in ``context``. + """ + problems: list[dict[str, str]] = [] + first_code: ErrorCode | None = None + for path in inputs: + code = _check_one(path) + if code is None: + continue + if first_code is None: + first_code = code + problems.append({"input": str(path), "error_code": code}) + if first_code is None: + return + + error_class: type[PdfOpsError] = ( + InputError if first_code in _INPUT_PROBLEMS else InvalidPdfError + ) + raise error_class( + f"{len(problems)} of {len(inputs)} input(s) unusable; " + f"first: {problems[0]['input']} ({first_code})", + error_code=first_code, + context={"problems": problems}, + ) + + +def _check_one(path: Path) -> ErrorCode | None: + if path.is_dir(): + return ErrorCode.INPUT_IS_DIRECTORY + if not path.is_file(): + return ErrorCode.INPUT_MISSING + try: + with path.open("rb") as handle: + head = handle.read(len(PDF_MAGIC)) + except OSError: + return ErrorCode.INPUT_UNREADABLE + if not head.startswith(PDF_MAGIC): + return ErrorCode.NOT_A_PDF + return None diff --git a/src/pdf_ops/logging_setup.py b/src/pdf_ops/logging_setup.py index 6e9090a..9840062 100644 --- a/src/pdf_ops/logging_setup.py +++ b/src/pdf_ops/logging_setup.py @@ -5,8 +5,6 @@ detail is passed via ``extra`` and merged into the payload. """ -from __future__ import annotations - import json import logging import re @@ -70,9 +68,9 @@ def _scrub(value: Any) -> Any: value = value.replace(secret, "***") return value if isinstance(value, dict): - return {key: _scrub(item) for key, item in value.items()} # pyright: ignore[reportUnknownVariableType] + return {key: _scrub(item) for key, item in value.items()} if isinstance(value, (list, tuple)): - return [_scrub(item) for item in value] # pyright: ignore[reportUnknownVariableType] + return [_scrub(item) for item in value] return value @@ -135,10 +133,12 @@ def filter(self, record: logging.LogRecord) -> bool: # Loggers whose records must reach stdout as JSON instead of falling through -# to logging.lastResort on stderr: the PDF library's recoverable-corruption -# warnings, and Python warnings (via logging.captureWarnings below). Anything -# on stderr would break the JSON-only/empty-stderr operator contract. -_THIRD_PARTY_LOGGERS = ("pypdf", "py.warnings") +# to logging.lastResort on stderr: anything the PDF library routes through +# Python logging, and Python warnings (via logging.captureWarnings below). +# Anything on stderr would break the JSON-only/empty-stderr operator +# contract. (qpdf's own parse warnings don't pass through here - the engine +# collects them per input and the operation layer emits them as events.) +_THIRD_PARTY_LOGGERS = ("pikepdf", "py.warnings") def setup_logging(level: int = logging.INFO) -> logging.Logger: diff --git a/src/pdf_ops/main.py b/src/pdf_ops/main.py index 0b7ad62..e5bdce2 100644 --- a/src/pdf_ops/main.py +++ b/src/pdf_ops/main.py @@ -1,25 +1,16 @@ """Top-level orchestration: the single error boundary and operation dispatch.""" -from __future__ import annotations - import logging import time from collections.abc import Callable, Mapping from typing import Any -from pdf_ops.config import ( - Config, - ExtractConfig, - MergeConfig, - Secrets, - describe_secret, - parse_config, - resolve_secrets, -) -from pdf_ops.errors import ExitCode, PdfOpsError +from pdf_ops.config import Config, ExtractConfig, MergeConfig, parse_config +from pdf_ops.errors import ErrorCode, ExitCode, PdfOpsError from pdf_ops.extract import run_extract -from pdf_ops.logging_setup import emit_terminal, register_secret_value, setup_logging +from pdf_ops.logging_setup import emit_terminal, setup_logging from pdf_ops.merge import run_merge +from pdf_ops.secrets import Secrets, describe_secret, resolve_and_register def run(env: Mapping[str, str]) -> int: @@ -41,17 +32,8 @@ def get_secrets() -> Secrets: # Resolved lazily: merge's skip short-circuit must succeed even # when the mounted password file is already gone - a retry after # success reads nothing at all. - secrets = resolve_secrets(config) - for secret in (secrets.password, secrets.output_password): - if secret is not None and not register_secret_value(secret.reveal()): - logger.warning( - "redaction_degraded", - extra={ - "detail": "a supplied secret is too short for defense-in-depth " - "log scrubbing; the structural no-leak layers still apply" - }, - ) - return secrets + output_ref = config.output_password if isinstance(config, MergeConfig) else None + return resolve_and_register(config.password, output_ref, logger) logger.info("config_loaded", extra=_config_echo(config)) result = _dispatch(config, get_secrets, logger) @@ -87,7 +69,7 @@ def get_secrets() -> Secrets: logging.ERROR, "operation_failed", { - "error_code": "UNEXPECTED_ERROR", + "error_code": ErrorCode.UNEXPECTED_ERROR, "exit_code": int(ExitCode.UNEXPECTED), "duration_s": round(time.monotonic() - started, 3), }, diff --git a/src/pdf_ops/merge.py b/src/pdf_ops/merge.py index b7ab6bd..b244d8a 100644 --- a/src/pdf_ops/merge.py +++ b/src/pdf_ops/merge.py @@ -1,22 +1,18 @@ """The merge operation: validate everything, then write once, atomically.""" -from __future__ import annotations - import logging -from collections.abc import Callable, Sequence -from pathlib import Path -from typing import Any +from collections.abc import Callable +from typing import Any, Literal -from pdf_ops.config import MergeConfig, OutputEncryption, Secrets +from pdf_ops.config import MergeConfig, OutputEncryption from pdf_ops.engine import OpenedInput, get_engine -from pdf_ops.errors import ConfigError, InputError, InvalidPdfError, PdfOpsError +from pdf_ops.errors import ConfigError, ErrorCode +from pdf_ops.inputs import validate_inputs from pdf_ops.output import atomic_output, check_output_path, clean_stale_temps -from pdf_ops.secret import Secret - -PDF_MAGIC = b"%PDF-" +from pdf_ops.secrets import Secret, Secrets -# Problem kinds found during input validation, in exit-code class order. -_INPUT_PROBLEMS = frozenset({"INPUT_MISSING", "INPUT_IS_DIRECTORY", "INPUT_UNREADABLE"}) +# Where the output password came from, for the output_encrypted event. +type PasswordSource = Literal["output", "input-fallback"] def run_merge( @@ -41,16 +37,9 @@ def run_merge( for path in config.inputs: one = engine.open_input(path, secrets.password) opened.append(one) - logger.info( - "input_opened", - extra={ - "input": str(path), - "pages": one.pages, - "encrypted": one.encrypted, - "algorithm": one.algorithm, - "password_type": one.password_type, - }, - ) + logger.info("input_opened", extra=one.event_fields()) + for message in one.warnings: + logger.warning("pdf_library_message", extra={"detail": message, "source": "qpdf"}) encrypted_count = sum(1 for one in opened if one.encrypted) output_password, password_source = _choose_output_password(config, secrets, encrypted_count) @@ -78,7 +67,10 @@ def run_merge( ) with atomic_output(config.output) as tmp_path: - engine.merge_to(opened, tmp_path, output_password) + late_warnings = engine.merge_to(opened, tmp_path, output_password) + for message in late_warnings: + # sources are read lazily, so repairs can surface at write time + logger.warning("pdf_library_message", extra={"detail": message, "source": "qpdf"}) if action == "overwrite": logger.info("output_overwritten", extra={"output_path": str(config.output)}) @@ -106,7 +98,7 @@ def run_merge( def _choose_output_password( config: MergeConfig, secrets: Secrets, encrypted_count: int -) -> tuple[Secret | None, str | None]: +) -> tuple[Secret | None, PasswordSource | None]: """Apply the output-encryption policy; returns (password, source-label). The fallback to the input password uses only an *explicitly supplied* @@ -127,49 +119,6 @@ def _choose_output_password( f"output encryption is required (PDFOPS_OUTPUT_ENCRYPTION={mode.value}) but no " "explicit password is available - the encrypted input(s) opened with the empty " "password; supply PDFOPS_OUTPUT_PASSWORD_FILE or PDFOPS_OUTPUT_PASSWORD", - error_code="MISSING_OUTPUT_PASSWORD", + error_code=ErrorCode.MISSING_OUTPUT_PASSWORD, context={"output_encryption": mode.value}, ) - - -def validate_inputs(inputs: Sequence[Path]) -> None: - """Check every input up front and report all problems in one failure. - - An operator fixing a broken workflow should learn about every bad input - from a single run, not one per retry. The raised error's class (and thus - the exit code) follows the first problem in input order; the full list - travels in ``context``. - """ - problems: list[dict[str, str]] = [] - for path in inputs: - code = _check_one(path) - if code is not None: - problems.append({"input": str(path), "error_code": code}) - if not problems: - return - - first = problems[0] - error_class: type[PdfOpsError] = ( - InputError if first["error_code"] in _INPUT_PROBLEMS else InvalidPdfError - ) - raise error_class( - f"{len(problems)} of {len(inputs)} input(s) unusable; " - f"first: {first['input']} ({first['error_code']})", - error_code=first["error_code"], - context={"problems": problems}, - ) - - -def _check_one(path: Path) -> str | None: - if path.is_dir(): - return "INPUT_IS_DIRECTORY" - if not path.is_file(): - return "INPUT_MISSING" - try: - with path.open("rb") as handle: - head = handle.read(len(PDF_MAGIC)) - except OSError: - return "INPUT_UNREADABLE" - if not head.startswith(PDF_MAGIC): - return "NOT_A_PDF" - return None diff --git a/src/pdf_ops/output.py b/src/pdf_ops/output.py index 195b902..b15d3ac 100644 --- a/src/pdf_ops/output.py +++ b/src/pdf_ops/output.py @@ -7,18 +7,19 @@ PDF where a downstream workflow step could read it. """ -from __future__ import annotations - import errno import os import tempfile from collections.abc import Generator -from contextlib import contextmanager +from contextlib import contextmanager, suppress from pathlib import Path -from typing import NoReturn +from typing import Literal, NoReturn from pdf_ops.config import OnExists -from pdf_ops.errors import OutputError +from pdf_ops.errors import ErrorCode, OutputError + +# What check_output_path decided for an existing-output policy. +type OutputAction = Literal["proceed", "skip", "overwrite"] def check_output_dir(directory: Path) -> None: @@ -27,12 +28,12 @@ def check_output_dir(directory: Path) -> None: raise OutputError( f"output directory {directory} does not exist " "(output locations are mounted; a missing directory is a workflow bug)", - error_code="OUTPUT_DIR_MISSING", + error_code=ErrorCode.OUTPUT_DIR_MISSING, context={"output_dir": str(directory)}, ) -def check_output_path(path: Path, on_exists: OnExists) -> str: +def check_output_path(path: Path, on_exists: OnExists) -> OutputAction: """Fail fast on unusable output locations, before any work is done. Returns the resolved action: ``"proceed"`` (no conflict), ``"skip"`` @@ -44,7 +45,7 @@ def check_output_path(path: Path, on_exists: OnExists) -> str: raise OutputError( f"output directory {parent} does not exist " "(output locations are mounted; a missing directory is a workflow bug)", - error_code="OUTPUT_DIR_MISSING", + error_code=ErrorCode.OUTPUT_DIR_MISSING, context={"output": str(path)}, ) if path.is_dir() and not path.is_symlink(): @@ -52,7 +53,7 @@ def check_output_path(path: Path, on_exists: OnExists) -> str: # work and os.replace cannot atomically replace a directory. raise OutputError( f"output path {path} is a directory", - error_code="OUTPUT_IS_DIRECTORY", + error_code=ErrorCode.OUTPUT_IS_DIRECTORY, context={"output": str(path)}, ) if not (path.is_symlink() or path.exists()): @@ -68,7 +69,7 @@ def check_output_path(path: Path, on_exists: OnExists) -> str: raise OutputError( f"output {path} already exists (refusing to overwrite; " "set PDFOPS_ON_EXISTS to overwrite or skip for retry semantics)", - error_code="OUTPUT_EXISTS", + error_code=ErrorCode.OUTPUT_EXISTS, context={"output": str(path)}, ) @@ -128,7 +129,7 @@ def atomic_output(path: Path) -> Generator[Path]: yield tmp_path with tmp_path.open("rb") as handle: os.fsync(handle.fileno()) - os.replace(tmp_path, path) + tmp_path.replace(path) _fsync_dir(path.parent) except OSError as err: _cleanup(tmp_path) @@ -154,23 +155,21 @@ def _translate_os_error(err: OSError, path: Path) -> Exception: if err.errno == errno.ENOSPC: return OutputError( f"no space left on device while writing {path}", - error_code="DISK_FULL", + error_code=ErrorCode.DISK_FULL, context={"output": str(path)}, ) if err.errno in (errno.EACCES, errno.EPERM, errno.EROFS): return OutputError( f"output location {path} is not writable", - error_code="OUTPUT_NOT_WRITABLE", + error_code=ErrorCode.OUTPUT_NOT_WRITABLE, context={"output": str(path)}, ) return err def _cleanup(tmp_path: Path) -> None: - try: + with suppress(OSError): # best effort - never mask the original failure tmp_path.unlink(missing_ok=True) - except OSError: # best effort - never mask the original failure - pass def _fsync_dir(directory: Path) -> None: diff --git a/src/pdf_ops/secret.py b/src/pdf_ops/secret.py deleted file mode 100644 index a3f0e6d..0000000 --- a/src/pdf_ops/secret.py +++ /dev/null @@ -1,28 +0,0 @@ -"""A string wrapper that cannot leak through repr, str, or f-strings. - -Non-leakage is structural, not a convention: the raw value is reachable only -through the explicit ``reveal()`` accessor, which is called in exactly one -place - the engine's decrypt/encrypt calls. The logging layer additionally -scrubs registered secret values from every record as defense in depth. -""" - -from __future__ import annotations - - -class Secret: - __slots__ = ("_value",) - - def __init__(self, value: str) -> None: - self._value = value - - def reveal(self) -> str: - return self._value - - def __repr__(self) -> str: - return "***" - - def __str__(self) -> str: - return "***" - - def __bool__(self) -> bool: - return bool(self._value) diff --git a/src/pdf_ops/secrets.py b/src/pdf_ops/secrets.py new file mode 100644 index 0000000..011495d --- /dev/null +++ b/src/pdf_ops/secrets.py @@ -0,0 +1,165 @@ +"""The whole secret lifecycle in one place. + +Four layers, one module: + +1. ``Secret`` - a wrapper that cannot leak through repr, str, or f-strings; + the raw value is reachable only through ``reveal()``. Its callers are + this module (validation and scrub registration) and the engine's + decrypt/encrypt calls - nothing else. +2. ``EnvSecret``/``FileSecret`` - parse-time references to where a secret + comes from. The config parser stays filesystem-free by capturing the + source, not the value. +3. ``resolve_secret``/``resolve_and_register`` - the one place secret file + I/O happens, deliberately late: merge's skip short-circuit never resolves + at all, so a retry after success works even when the mounted password + file is already gone. +4. Registration - resolved values are handed to the logging layer's + defense-in-depth scrubber (the structural wrapper is the primary + guarantee; scrubbing catches library residue such as exception messages). +""" + +import logging +from dataclasses import dataclass +from pathlib import Path + +from pdf_ops.errors import ConfigError, ErrorCode +from pdf_ops.logging_setup import register_secret_value + + +class Secret: + """A string that renders as ``***`` everywhere; ``reveal()`` is the only + way to the value, which keeps accidental leakage a type error rather + than a code-review hope.""" + + __slots__ = ("_value",) + + def __init__(self, value: str) -> None: + self._value = value + + def reveal(self) -> str: + return self._value + + def __repr__(self) -> str: + return "***" + + def __str__(self) -> str: + return "***" + + def __bool__(self) -> bool: + return bool(self._value) + + +@dataclass(frozen=True, slots=True) +class EnvSecret: + """A secret supplied directly in an environment variable.""" + + value: Secret + + def describe(self) -> str: + return "set(env)" + + def resolve(self) -> Secret: + _reject_control_characters(self.value.reveal(), "the environment") + return self.value + + +@dataclass(frozen=True, slots=True) +class FileSecret: + """A secret to be read from a mounted file (resolved after parsing - + the parser itself stays filesystem-free).""" + + path: Path + + def describe(self) -> str: + return "set(file)" + + def resolve(self) -> Secret: + """Read the file; a single trailing newline is stripped (hand-created + secret files usually have one; the password itself is otherwise taken + byte-for-byte).""" + try: + raw = self.path.read_text(encoding="utf-8") + except OSError as err: + raise ConfigError( + f"cannot read password file {self.path}: {err.strerror or err}", + error_code=ErrorCode.PASSWORD_FILE_UNREADABLE, + context={"path": str(self.path)}, + ) from err + except UnicodeDecodeError as err: + # Deliberately no decode detail: it would name a byte of the + # secret and its position. + raise ConfigError( + f"password file {self.path} is not valid UTF-8 text", + error_code=ErrorCode.PASSWORD_FILE_UNREADABLE, + context={"path": str(self.path)}, + ) from err + raw = raw.removesuffix("\n").removesuffix("\r") + if not raw: + raise ConfigError( + f"password file {self.path} is empty", + error_code=ErrorCode.EMPTY_PASSWORD, + context={"path": str(self.path)}, + ) + _reject_control_characters(raw, str(self.path)) + return Secret(raw) + + +type SecretRef = EnvSecret | FileSecret + + +def describe_secret(ref: SecretRef | None) -> str: + """Presence-only description for the config-echo log event.""" + return "unset" if ref is None else ref.describe() + + +def resolve_secret(ref: SecretRef | None) -> Secret | None: + """Materialize a secret reference; raises ConfigError on unreadable, + non-UTF-8, or empty files and on control characters.""" + return None if ref is None else ref.resolve() + + +@dataclass(frozen=True, slots=True) +class Secrets: + """Materialized secrets for one run, resolved from the config's refs.""" + + password: Secret | None + output_password: Secret | None + + +def resolve_and_register( + password: SecretRef | None, + output_password: SecretRef | None, + logger: logging.Logger, +) -> Secrets: + """Materialize both secrets and wire them into log scrubbing. + + The input password resolves first, deliberately: when both sources are + bad, the failure event names the primary secret's problem. + """ + secrets = Secrets( + password=resolve_secret(password), + output_password=resolve_secret(output_password), + ) + for secret in (secrets.password, secrets.output_password): + if secret is not None and not register_secret_value(secret.reveal()): + logger.warning( + "redaction_degraded", + extra={ + "detail": "a supplied secret is too short for defense-in-depth " + "log scrubbing; the structural no-leak layers still apply" + }, + ) + return secrets + + +def _reject_control_characters(value: str, origin: str) -> None: + """A password containing control characters is almost certainly an + encoding or copy-paste accident - and downstream cryptographic + normalization (SASLprep) would warn about it, naming the codepoint.""" + if any(ord(ch) < 32 or 0x7F <= ord(ch) <= 0x9F for ch in value): + raise ConfigError( + f"the password from {origin} contains control characters " + "(check for encoding or copy-paste issues)", + error_code=ErrorCode.PASSWORD_UNSUPPORTED_CHARACTERS, + context={"source": origin}, + ) diff --git a/tests/conftest.py b/tests/conftest.py index c8e0dd4..caba690 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -5,8 +5,6 @@ from the library version. """ -from __future__ import annotations - import json from collections.abc import Callable from pathlib import Path @@ -16,11 +14,10 @@ from pypdf import PdfWriter from pdf_ops.main import run +from tests.helpers import RunApp, build_raw_pdf TERMINAL_EVENTS = {"operation_complete", "operation_failed"} -RunApp = Callable[[dict[str, str]], tuple[int, list[dict[str, Any]]]] - @pytest.fixture def run_app(capsys: pytest.CaptureFixture[str]) -> RunApp: @@ -77,19 +74,21 @@ def _make(name: str = "fake.pdf", content: bytes = b"plain text, not a pdf\n") - @pytest.fixture def make_corrupt_pdf(make_pdf: Callable[..., Path], tmp_path: Path) -> Callable[..., Path]: - """A file with a valid ``%PDF-`` header that fails to parse. + """A file with a valid ``%PDF-`` header that no parser can recover. - ``truncate`` cuts the file in half (destroys xref + EOF marker); - ``mangle-xref`` corrupts the cross-reference section in place. + qpdf-backed engines repair light damage (truncation, a mangled xref) by + reconstructing the cross-reference table, so unrecoverable corruption has + to destroy the object structure itself: ``garbage-body`` has no objects + or trailer at all; ``no-objects`` keeps the file's shape but breaks every + object keyword, leaving reconstruction nothing to find. """ - def _make(name: str = "corrupt.pdf", mode: str = "truncate") -> Path: - source = make_pdf(name=f"pristine-{name}", pages=2) - data = source.read_bytes() - if mode == "truncate": - data = data[: len(data) // 2] - elif mode == "mangle-xref": - data = data.replace(b"xref", b"xrfx", 1) + def _make(name: str = "corrupt.pdf", mode: str = "garbage-body") -> Path: + if mode == "garbage-body": + data = b"%PDF-1.7\n" + b"\x89\x00garbage" * 40 + elif mode == "no-objects": + source = make_pdf(name=f"pristine-{name}", pages=2) + data = source.read_bytes().replace(b" obj", b" obX") else: # pragma: no cover - guard against typos in tests raise ValueError(f"unknown corruption mode: {mode}") path = tmp_path / name @@ -99,25 +98,18 @@ def _make(name: str = "corrupt.pdf", mode: str = "truncate") -> Path: return _make -def _build_raw_pdf(objects: list[str | bytes]) -> bytes: - """Minimal hand-assembled PDF with a correct xref - for structural cases - the writer API refuses to produce (dangling references, missing /Pages, - raw name-tree bytes).""" - out = bytearray(b"%PDF-1.4\n") - offsets: list[int] = [] - for number, body in enumerate(objects, start=1): - offsets.append(len(out)) - body_bytes = body if isinstance(body, bytes) else body.encode() - out += f"{number} 0 obj\n".encode() + body_bytes + b"\nendobj\n" - xref_pos = len(out) - out += f"xref\n0 {len(objects) + 1}\n".encode() - out += b"0000000000 65535 f \n" - for offset in offsets: - out += f"{offset:010d} 00000 n \n".encode() - out += ( - f"trailer\n<< /Size {len(objects) + 1} /Root 1 0 R >>\nstartxref\n{xref_pos}\n%%EOF\n" - ).encode() - return bytes(out) +@pytest.fixture +def make_damaged_pdf(make_pdf: Callable[..., Path], tmp_path: Path) -> Callable[..., Path]: + """A parseable-after-repair file: real content with a mangled xref. The + engine reconstructs the cross-reference table and reports warnings.""" + + def _make(name: str = "damaged.pdf") -> Path: + source = make_pdf(name=f"pristine-{name}", pages=2) + path = tmp_path / name + path.write_bytes(source.read_bytes().replace(b"xref", b"xrfx", 1)) + return path + + return _make @pytest.fixture @@ -154,12 +146,12 @@ def _make( @pytest.fixture def make_dangling_ref_pdf(tmp_path: Path) -> Callable[..., Path]: """A parseable PDF whose page /Contents points at a missing object - - pypdf merges it successfully but logs a recoverable-corruption warning.""" + the engine repairs it and reports a recoverable-corruption warning.""" def _make(name: str = "dangling.pdf") -> Path: path = tmp_path / name path.write_bytes( - _build_raw_pdf( + build_raw_pdf( [ "<< /Type /Catalog /Pages 2 0 R >>", "<< /Type /Pages /Kids [3 0 R] /Count 1 >>", @@ -174,12 +166,12 @@ def _make(name: str = "dangling.pdf") -> Path: @pytest.fixture def make_pathological_pdf(tmp_path: Path) -> Callable[..., Path]: - """Valid header and xref, but a catalog with no /Pages - pypdf raises a - builtin AttributeError instead of its own exception type.""" + """Valid header and xref, but a catalog with no /Pages - a structural + hole that must classify as corrupt, not as an internal error.""" def _make(name: str = "pathological.pdf") -> Path: path = tmp_path / name - path.write_bytes(_build_raw_pdf(["<< /Type /Catalog >>"])) + path.write_bytes(build_raw_pdf(["<< /Type /Catalog >>"])) return path return _make @@ -224,7 +216,7 @@ def _make( filter_part = (b" /Filter " + filter_entry) if filter_entry else b"" path = tmp_path / name path.write_bytes( - _build_raw_pdf( + build_raw_pdf( [ b"<< /Type /Catalog /Pages 2 0 R /Names << /EmbeddedFiles " b"<< /Names [ " + name_literal + b" 4 0 R ] >> >> >>", diff --git a/tests/container/test_image.py b/tests/container/test_image.py index be17038..2ce6398 100644 --- a/tests/container/test_image.py +++ b/tests/container/test_image.py @@ -8,8 +8,6 @@ host paths, and pytest's default tmp dir is not among them. """ -from __future__ import annotations - import json import os import shutil @@ -56,8 +54,9 @@ def docker_run( *, volumes: dict[Path, str] | None = None, entrypoint: list[str] | None = None, + extra_args: list[str] | None = None, ) -> subprocess.CompletedProcess[str]: - cmd = ["docker", "run", "--rm"] + cmd = ["docker", "run", "--rm", *(extra_args or [])] for key, value in env.items(): cmd += ["-e", f"{key}={value}"] for host, spec in (volumes or {}).items(): @@ -69,6 +68,11 @@ def docker_run( return subprocess.run(cmd, capture_output=True, text=True, timeout=120) +# The security posture the deploy example promises: read-only root +# filesystem, no capabilities, no privilege escalation. +HARDENED = ["--read-only", "--cap-drop", "ALL", "--security-opt", "no-new-privileges"] + + def test_invalid_config_exits_2_with_json_only_stdout(image: str) -> None: result = docker_run(image, {"PDFOPS_OPERATION": "bogus"}) assert result.returncode == 2 @@ -203,3 +207,52 @@ def test_golden_encrypted_merge_with_mounted_password_file(image: str, mount_dir encrypt_dict = cast("Any", reader.trailer["/Encrypt"]).get_object() assert int(encrypt_dict["/V"]) == 5 # AES-256, not legacy RC4 + + +def test_hardened_run_read_only_rootfs(image: str, mount_dir: Path) -> None: + # The image must function under the deploy example's full security + # posture: nothing writable but the mounted output directory (work files + # live there by design, so a read-only root filesystem costs nothing). + in_dir = mount_dir / "in" + out_dir = mount_dir / "out" + in_dir.mkdir() + out_dir.mkdir() + out_dir.chmod(0o777) + + writer = PdfWriter() + writer.add_blank_page(width=200, height=300) + with (in_dir / "a.pdf").open("wb") as handle: + writer.write(handle) + + result = docker_run( + image, + { + "PDFOPS_OPERATION": "merge", + "PDFOPS_INPUTS": "/in/a.pdf", + "PDFOPS_OUTPUT": "/out/merged.pdf", + }, + volumes={in_dir: "/in:ro", out_dir: "/out"}, + extra_args=HARDENED, + ) + + assert result.returncode == 0, result.stdout + result.stderr + assert result.stderr == "" + events = [json.loads(line) for line in result.stdout.strip().splitlines()] + assert events[-1]["event"] == "operation_complete" + assert (out_dir / "merged.pdf").exists() + + +def test_runtime_image_has_no_package_installer(image: str) -> None: + # The runtime stage carries the locked virtualenv and nothing to mutate + # it with. Two probes, because the venv python structurally cannot see + # the base interpreter's site-packages: pip must be gone from the venv + # AND from the base install, and stdlib ensurepip (whose bundled wheel + # would restore pip in one command) must be gone with it. + probe = ( + "import importlib.util, sys; " + "sys.exit(1 if importlib.util.find_spec('pip') " + "or importlib.util.find_spec('ensurepip') else 0)" + ) + for interpreter in ("python", "/usr/local/bin/python3.14"): + result = docker_run(image, {}, entrypoint=[interpreter, "-c", probe]) + assert result.returncode == 0, f"{interpreter}: " + result.stdout + result.stderr diff --git a/tests/helpers.py b/tests/helpers.py new file mode 100644 index 0000000..b5eaa77 --- /dev/null +++ b/tests/helpers.py @@ -0,0 +1,49 @@ +"""Shared test helpers. + +A plain module rather than conftest.py: conftest is a pytest plugin, and +importing names from it couples tests to how pytest loaded it. +""" + +import logging +from collections.abc import Callable +from typing import Any + +RunApp = Callable[[dict[str, str]], tuple[int, list[dict[str, Any]]]] + + +def make_record( + msg: str = "some_event", + *, + name: str = "pdf_ops", + level: int = logging.INFO, + exc_info: Any = None, + **extra: Any, +) -> logging.LogRecord: + """A LogRecord as the formatter sees it, with ``extra`` fields attached.""" + record = logging.LogRecord( + name=name, level=level, pathname=__file__, lineno=1, msg=msg, args=None, exc_info=exc_info + ) + for key, value in extra.items(): + setattr(record, key, value) + return record + + +def build_raw_pdf(objects: list[str | bytes]) -> bytes: + """Minimal hand-assembled PDF with a correct xref - for structural cases + the writer API refuses to produce (dangling references, missing /Pages, + raw name-tree bytes).""" + out = bytearray(b"%PDF-1.4\n") + offsets: list[int] = [] + for number, body in enumerate(objects, start=1): + offsets.append(len(out)) + body_bytes = body if isinstance(body, bytes) else body.encode() + out += f"{number} 0 obj\n".encode() + body_bytes + b"\nendobj\n" + xref_pos = len(out) + out += f"xref\n0 {len(objects) + 1}\n".encode() + out += b"0000000000 65535 f \n" + for offset in offsets: + out += f"{offset:010d} 00000 n \n".encode() + out += ( + f"trailer\n<< /Size {len(objects) + 1} /Root 1 0 R >>\nstartxref\n{xref_pos}\n%%EOF\n" + ).encode() + return bytes(out) diff --git a/tests/integration/test_entrypoint.py b/tests/integration/test_entrypoint.py index 067c984..8489d9e 100644 --- a/tests/integration/test_entrypoint.py +++ b/tests/integration/test_entrypoint.py @@ -5,8 +5,6 @@ and stdout/stderr split match the documented contract. """ -from __future__ import annotations - import json import subprocess import sys diff --git a/tests/integration/test_extract.py b/tests/integration/test_extract.py index 8814d4c..dace7be 100644 --- a/tests/integration/test_extract.py +++ b/tests/integration/test_extract.py @@ -5,15 +5,13 @@ PDFOPS_OUTPUT_DIR. """ -from __future__ import annotations - from collections.abc import Callable from pathlib import Path import pytest import pdf_ops.extract -from tests.conftest import RunApp +from tests.helpers import RunApp, build_raw_pdf pytestmark = pytest.mark.integration @@ -291,6 +289,23 @@ def test_utf16_traversal_name_stays_contained( assert sorted(p.name for p in out_dir.iterdir()) == ["sneak.txt"] assert not (tmp_path / "sneak.txt").exists() + def test_non_ascii_names_survive_with_correct_text( + self, + make_pdf_with_attachments: Callable[..., Path], + out_dir: Path, + run_app: RunApp, + ) -> None: + # Name-tree strings are PDFDocEncoded or UTF-16, not UTF-8 - the + # extracted filename must carry the actual characters, not mojibake. + resume = "r\u00e9sum\u00e9.pdf" + report = "\u043e\u0442\u0447\u0451\u0442.txt" + carrier = make_pdf_with_attachments([(resume, b"cv"), (report, b"report")]) + code, events = run_app(extract_env(carrier, out_dir)) + assert code == 0 + assert events[-1]["attachments_extracted"] == 2 + assert (out_dir / resume).read_bytes() == b"cv" + assert (out_dir / report).read_bytes() == b"report" + def test_non_utf8_name_bytes_extract_inside_output_dir( self, make_raw_attachment_pdf: Callable[..., Path], @@ -311,14 +326,164 @@ def test_unknown_stream_filter_is_unprocessable_not_internal( out_dir: Path, run_app: RunApp, ) -> None: - # pypdf raises NotImplementedError for a filter it doesn't implement; - # that must classify as a data problem (exit 4), never exit 1. + # A stream filter the engine cannot decode must classify as a data + # problem (exit 4), never exit 1. carrier = make_raw_attachment_pdf(b"(ok.txt)", filter_entry=b"/FooBar") code, events = run_app(extract_env(carrier, out_dir)) assert code == 4 assert events[-1]["error_code"] == "UNSUPPORTED_PDF_FEATURE" assert list(out_dir.iterdir()) == [] + def test_bare_file_reference_without_embedded_stream_is_skipped( + self, tmp_path: Path, out_dir: Path, run_app: RunApp + ) -> None: + # A filespec with no /EF is a mere pointer to an external file - + # there is nothing embedded to extract, and it must not abort the + # entries that do carry data. + carrier = tmp_path / "bare-ref.pdf" + carrier.write_bytes( + build_raw_pdf( + [ + b"<< /Type /Catalog /Pages 2 0 R /Names << /EmbeddedFiles " + b"<< /Names [ (external.txt) 4 0 R (real.txt) 5 0 R ] >> >> >>", + "<< /Type /Pages /Kids [3 0 R] /Count 1 >>", + "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 300] >>", + b"<< /Type /Filespec /F (external.txt) >>", + b"<< /Type /Filespec /F (real.txt) /EF << /F 6 0 R >> >>", + b"<< /Length 7 >>\nstream\npayload\nendstream", + ] + ) + ) + code, events = run_app(extract_env(carrier, out_dir)) + assert code == 0 + assert events[-1]["attachments_extracted"] == 1 + assert (out_dir / "real.txt").read_bytes() == b"payload" + + def test_uf_only_filespec_extracts( + self, tmp_path: Path, out_dir: Path, run_app: RunApp + ) -> None: + # /EF may carry the stream under /UF (unicode name) with no /F. + carrier = tmp_path / "uf-only.pdf" + carrier.write_bytes( + build_raw_pdf( + [ + b"<< /Type /Catalog /Pages 2 0 R /Names << /EmbeddedFiles " + b"<< /Names [ (u.txt) 4 0 R ] >> >> >>", + "<< /Type /Pages /Kids [3 0 R] /Count 1 >>", + "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 300] >>", + b"<< /Type /Filespec /UF (u.txt) /EF << /UF 5 0 R >> >>", + b"<< /Length 7 >>\nstream\ncontent\nendstream", + ] + ) + ) + code, events = run_app(extract_env(carrier, out_dir)) + assert code == 0 + assert events[-1]["attachments_extracted"] == 1 + assert (out_dir / "u.txt").read_bytes() == b"content" + + @pytest.mark.parametrize( + ("tree_object", "expected_count"), + [ + # a /Kids entry whose reference dangles (resolves to null) + (b"<< /Kids [ 9 0 R ] >>", 0), + # /Names holding an integer instead of an array + (b"<< /Names 42 >>", 0), + # an integer where a kid node belongs + (b"<< /Kids [ 999 ] >>", 0), + ], + ) + def test_malformed_tree_shapes_degrade_never_crash( + self, + tmp_path: Path, + out_dir: Path, + run_app: RunApp, + tree_object: bytes, + expected_count: int, + ) -> None: + # Name trees are attacker-controlled: structurally malformed nodes + # must degrade to skipped entries (or classify as a data problem), + # never escape as exit 1 - the only class a workflow engine retries. + carrier = tmp_path / "malformed.pdf" + carrier.write_bytes( + build_raw_pdf( + [ + b"<< /Type /Catalog /Pages 2 0 R /Names << /EmbeddedFiles 4 0 R >> >>", + "<< /Type /Pages /Kids [3 0 R] /Count 1 >>", + "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 300] >>", + tree_object, + ] + ) + ) + code, events = run_app(extract_env(carrier, out_dir)) + assert code == 0 + assert events[-1]["attachments_extracted"] == expected_count + + def test_malformed_filespec_ef_is_skipped_sibling_survives( + self, tmp_path: Path, out_dir: Path, run_app: RunApp + ) -> None: + carrier = tmp_path / "bad-ef.pdf" + carrier.write_bytes( + build_raw_pdf( + [ + b"<< /Type /Catalog /Pages 2 0 R /Names << /EmbeddedFiles " + b"<< /Names [ (bad.txt) 4 0 R (good.txt) 5 0 R ] >> >> >>", + "<< /Type /Pages /Kids [3 0 R] /Count 1 >>", + "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 300] >>", + b"<< /Type /Filespec /F (bad.txt) /EF 42 >>", + b"<< /Type /Filespec /F (good.txt) /EF << /F 6 0 R >> >>", + b"<< /Length 7 >>\nstream\npayload\nendstream", + ] + ) + ) + code, events = run_app(extract_env(carrier, out_dir)) + assert code == 0 + assert events[-1]["attachments_extracted"] == 1 + assert (out_dir / "good.txt").read_bytes() == b"payload" + + def test_integer_name_key_gets_fallback_name_not_a_giant_allocation( + self, tmp_path: Path, out_dir: Path, run_app: RunApp + ) -> None: + # bytes(int) constructs that many zero bytes - a ~20-byte hostile + # entry must not become an attacker-sized allocation. A non-string + # key gets the deterministic fallback name; the payload survives. + carrier = tmp_path / "int-key.pdf" + carrier.write_bytes( + build_raw_pdf( + [ + b"<< /Type /Catalog /Pages 2 0 R /Names << /EmbeddedFiles " + b"<< /Names [ 999999999 4 0 R ] >> >> >>", + "<< /Type /Pages /Kids [3 0 R] /Count 1 >>", + "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 300] >>", + b"<< /Type /Filespec /F (x) /EF << /F 5 0 R >> >>", + b"<< /Length 7 >>\nstream\npayload\nendstream", + ] + ) + ) + code, events = run_app(extract_env(carrier, out_dir)) + assert code == 0 + assert events[-1]["attachments_extracted"] == 1 + assert (out_dir / "attachment_0").read_bytes() == b"payload" + + def test_cyclic_name_tree_terminates( + self, tmp_path: Path, out_dir: Path, run_app: RunApp + ) -> None: + # A hostile name tree can reference itself through /Kids; the walk + # must terminate instead of hanging the container. + carrier = tmp_path / "cyclic.pdf" + carrier.write_bytes( + build_raw_pdf( + [ + b"<< /Type /Catalog /Pages 2 0 R /Names << /EmbeddedFiles 4 0 R >> >>", + "<< /Type /Pages /Kids [3 0 R] /Count 1 >>", + "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 300] >>", + b"<< /Kids [ 4 0 R ] >>", + ] + ) + ) + code, events = run_app(extract_env(carrier, out_dir)) + assert code == 0 + assert events[-1]["attachments_extracted"] == 0 + class TestInvariants: def test_containment_recheck_refuses_a_broken_sanitizer( diff --git a/tests/integration/test_merge.py b/tests/integration/test_merge.py index ede5efb..040f2cf 100644 --- a/tests/integration/test_merge.py +++ b/tests/integration/test_merge.py @@ -1,7 +1,5 @@ """End-to-end merge runs through run(env): the merge operator contract.""" -from __future__ import annotations - import errno import os from collections.abc import Callable @@ -12,8 +10,8 @@ import pdf_ops.merge from pdf_ops.engine import OpenedInput -from pdf_ops.errors import InvalidPdfError -from tests.conftest import RunApp +from pdf_ops.errors import ErrorCode, InvalidPdfError +from tests.helpers import RunApp, build_raw_pdf pytestmark = pytest.mark.integration @@ -153,7 +151,7 @@ def test_empty_file_is_not_a_pdf( assert code == 4 assert events[-1]["error_code"] == "NOT_A_PDF" - @pytest.mark.parametrize("mode", ["truncate", "mangle-xref"]) + @pytest.mark.parametrize("mode", ["garbage-body", "no-objects"]) def test_corrupt_pdf_rejected( self, make_corrupt_pdf: Callable[..., Path], @@ -238,11 +236,11 @@ def test_unwritable_output_dir_refused( class TestPathologicalInputs: - def test_pypdf_builtin_exception_maps_to_corrupt( + def test_catalog_without_pages_maps_to_corrupt( self, make_pathological_pdf: Callable[..., Path], out_dir: Path, run_app: RunApp ) -> None: - # Valid header + xref but no /Pages: pypdf raises AttributeError, not - # its own exception type - it must still classify as exit 4, not 1. + # Valid header + xref but no /Pages: a structural hole the engine + # must classify as exit 4, never as an internal error. bad = make_pathological_pdf() code, events = run_app(merge_env([bad], out_dir / "m.pdf")) assert code == 4 @@ -251,17 +249,58 @@ def test_pypdf_builtin_exception_maps_to_corrupt( def test_library_warning_goes_to_stdout_json_not_stderr( self, make_dangling_ref_pdf: Callable[..., Path], out_dir: Path, run_app: RunApp ) -> None: - # A repairable input: pypdf warns ("Object 9 0 not defined") but the - # merge succeeds. run_app's invariants assert stderr stays empty and - # stdout stays JSON-only; the warning must surface as a JSON event. + # A repairable input: qpdf fixes it up but reports the damage. The + # run_app invariants assert stderr stays empty and stdout stays + # JSON-only; the warning must surface as a JSON event. repairable = make_dangling_ref_pdf() code, events = run_app(merge_env([repairable], out_dir / "m.pdf")) assert code == 0 warnings = [e for e in events if e["event"] == "pdf_library_message"] - assert warnings, "the pypdf warning must appear as a structured event" + assert warnings, "the library warning must appear as a structured event" assert warnings[0]["level"] == "warning" - assert "9 0" in warnings[0]["detail"] - assert warnings[0]["source"].startswith("pypdf") + assert "repairing" in warnings[0]["detail"] + assert warnings[0]["source"] == "qpdf" + + def test_repair_during_lazy_write_still_surfaces_a_warning( + self, tmp_path: Path, out_dir: Path, run_app: RunApp + ) -> None: + # qpdf reads stream data lazily: damage discovered only while the + # writer copies (a wrong stream /Length) repairs at write time and + # must still surface as an event, not be silently absorbed. + source = tmp_path / "late.pdf" + source.write_bytes( + build_raw_pdf( + [ + "<< /Type /Catalog /Pages 2 0 R >>", + "<< /Type /Pages /Kids [3 0 R] /Count 1 >>", + "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 300] /Contents 4 0 R >>", + b"<< /Length 999 >>\nstream\n0 0 m 10 10 l S\nendstream", + ] + ) + ) + code, events = run_app(merge_env([source], out_dir / "m.pdf")) + assert code == 0 + warnings = [e for e in events if e["event"] == "pdf_library_message"] + assert warnings, "a write-time repair must produce a structured event" + + def test_light_damage_is_repaired_with_warnings( + self, + make_damaged_pdf: Callable[..., Path], + make_pdf: Callable[..., Path], + out_dir: Path, + run_app: RunApp, + ) -> None: + # A mangled xref is recoverable: the engine reconstructs the table, + # the merge succeeds with every page, and the damage is visible in + # the log rather than silently absorbed. + damaged = make_damaged_pdf() + plain = make_pdf(name="plain.pdf") + output = out_dir / "m.pdf" + code, events = run_app(merge_env([damaged, plain], output)) + assert code == 0 + assert events[-1]["pages"] == 3 + assert any(e["event"] == "pdf_library_message" for e in events) + assert len(PdfReader(output).pages) == 3 class FakeEngine: @@ -276,7 +315,7 @@ def open_input(self, path: Path, password: object) -> OpenedInput: path=path, handle=None, pages=1, encrypted=False, algorithm=None, password_type=None ) - def merge_to(self, inputs: object, destination: Path, output_password: object) -> None: + def merge_to(self, inputs: object, destination: Path, output_password: object) -> list[str]: destination.write_bytes(b"%PDF- partial garbage") raise self.error @@ -310,7 +349,7 @@ def test_app_error_after_partial_write_cleans_temp( out_dir: Path, run_app: RunApp, ) -> None: - fake_engine(InvalidPdfError("boom mid-write", error_code="CORRUPT_PDF", context={})) + fake_engine(InvalidPdfError("boom mid-write", error_code=ErrorCode.CORRUPT_PDF, context={})) source = make_pdf() code, _ = run_app(merge_env([source], out_dir / "m.pdf")) assert code == 4 diff --git a/tests/integration/test_passwords.py b/tests/integration/test_passwords.py index 4e1c9dc..ecf1819 100644 --- a/tests/integration/test_passwords.py +++ b/tests/integration/test_passwords.py @@ -6,8 +6,6 @@ or crash. """ -from __future__ import annotations - from collections.abc import Callable from pathlib import Path @@ -17,8 +15,8 @@ import pdf_ops.merge from pdf_ops.engine import OpenedInput from pdf_ops.main import run -from pdf_ops.secret import Secret -from tests.conftest import RunApp +from pdf_ops.secrets import Secret +from tests.helpers import RunApp, build_raw_pdf from tests.integration.test_extract import extract_env from tests.integration.test_merge import merge_env @@ -103,7 +101,9 @@ def test_wrong_password_names_input_not_password( terminal = events[-1] assert terminal["error_code"] == "WRONG_PASSWORD" assert terminal["context"]["input"] == str(locked) - assert terminal["context"]["algorithm"].startswith("RC4") + # exact label: the failed-open scan must find the real /Encrypt + # object, not the first /Length token some stream happens to carry + assert terminal["context"]["algorithm"] == "RC4-128" assert not output.exists() def test_no_password_on_user_locked_input( @@ -114,6 +114,39 @@ def test_no_password_on_user_locked_input( assert code == 5 assert events[-1]["error_code"] == "PASSWORD_REQUIRED" + def test_certificate_encryption_reports_unsupported( + self, tmp_path: Path, out_dir: Path, run_app: RunApp + ) -> None: + # Certificate security handlers are out of scope; the operator remedy + # (a certificate and key) is neither a password nor a repair, so the + # classification must stay in the password class - never corrupt, and + # never a retryable internal error. + locked = tmp_path / "cert.pdf" + raw = build_raw_pdf( + [ + "<< /Type /Catalog /Pages 2 0 R >>", + "<< /Type /Pages /Kids [3 0 R] /Count 1 >>", + "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 300] >>", + b"<< /Filter /Adobe.PubSec /SubFilter /adbe.pkcs7.s5 /V 5 /R 6 >>", + ] + ) + locked.write_bytes(raw.replace(b"/Root 1 0 R", b"/Root 1 0 R /Encrypt 4 0 R")) + code, events = run_app(merge_env([locked], out_dir / "m.pdf")) + assert code == 5 + assert events[-1]["error_code"] == "UNSUPPORTED_ENCRYPTION" + + def test_password_required_still_reports_aes256_algorithm( + self, make_encrypted_pdf: Callable[..., Path], out_dir: Path, run_app: RunApp + ) -> None: + # When the file cannot be opened at all, the algorithm label comes + # from a raw scan of the plaintext /Encrypt dictionary - the + # observability must not disappear exactly when the operator needs it. + locked = make_encrypted_pdf(password=PW, algorithm="AES-256") + code, events = run_app(merge_env([locked], out_dir / "m.pdf")) + assert code == 5 + assert events[-1]["error_code"] == "PASSWORD_REQUIRED" + assert events[-1]["context"]["algorithm"] == "AES-256" + def test_owner_only_file_opens_without_password( self, make_encrypted_pdf: Callable[..., Path], out_dir: Path, run_app: RunApp ) -> None: @@ -337,7 +370,9 @@ class LeakyEngine: def open_input(self, path: Path, password: Secret | None) -> OpenedInput: raise RuntimeError(f"login failed for {password.reveal() if password else ''}") - def merge_to(self, inputs: object, destination: Path, output_password: object) -> None: + def merge_to( + self, inputs: object, destination: Path, output_password: object + ) -> list[str]: raise AssertionError("unreachable") def list_attachments(self, opened: object) -> list[object]: @@ -457,6 +492,31 @@ def test_owner_only_under_never_still_warns_downgrade( assert code == 0 assert any(e["event"] == "security_downgrade" for e in events) + def test_both_secrets_bad_reports_the_input_password_first( + self, + make_pdf: Callable[..., Path], + tmp_path: Path, + out_dir: Path, + run_app: RunApp, + ) -> None: + # The input password resolves before the output password, so a run + # where both sources are broken names the primary secret's problem. + empty_pw = tmp_path / "empty-pw" + empty_pw.write_text("") + missing_out_pw = tmp_path / "gone-pw" + plain = make_pdf() + env = merge_env( + [plain], + out_dir / "m.pdf", + PDFOPS_PASSWORD_FILE=str(empty_pw), + PDFOPS_OUTPUT_ENCRYPTION="always", + PDFOPS_OUTPUT_PASSWORD_FILE=str(missing_out_pw), + ) + code, events = run_app(env) + assert code == 2 + assert events[-1]["error_code"] == "EMPTY_PASSWORD" + assert events[-1]["context"]["path"] == str(empty_pw) + def test_short_password_degrades_redaction_with_warning( self, make_encrypted_pdf: Callable[..., Path], out_dir: Path, run_app: RunApp ) -> None: diff --git a/tests/integration/test_retries.py b/tests/integration/test_retries.py index a2cf584..528e67b 100644 --- a/tests/integration/test_retries.py +++ b/tests/integration/test_retries.py @@ -5,8 +5,6 @@ "the step ran before - what does running it again do?" """ -from __future__ import annotations - from collections.abc import Callable from pathlib import Path @@ -14,8 +12,8 @@ from pypdf import PdfReader import pdf_ops.merge -from pdf_ops.errors import InvalidPdfError -from tests.conftest import RunApp +from pdf_ops.errors import ErrorCode, InvalidPdfError +from tests.helpers import RunApp from tests.integration.test_extract import extract_env from tests.integration.test_merge import FakeEngine, merge_env @@ -304,7 +302,9 @@ def test_failed_overwrite_preserves_the_original( assert run_app(merge_env([source], output))[0] == 0 original = output.read_bytes() - fake_engine(InvalidPdfError("boom mid-rewrite", error_code="CORRUPT_PDF", context={})) + fake_engine( + InvalidPdfError("boom mid-rewrite", error_code=ErrorCode.CORRUPT_PDF, context={}) + ) code, _ = run_app(merge_env([source], output, PDFOPS_ON_EXISTS="overwrite")) assert code == 4 diff --git a/tests/integration/test_run.py b/tests/integration/test_run.py index 4e9df41..9e44383 100644 --- a/tests/integration/test_run.py +++ b/tests/integration/test_run.py @@ -6,15 +6,13 @@ one terminal event, emitted last. """ -from __future__ import annotations - import logging from typing import Any import pytest import pdf_ops.main -from tests.conftest import RunApp +from tests.helpers import RunApp pytestmark = pytest.mark.integration diff --git a/tests/unit/test_config.py b/tests/unit/test_config.py index 8fae44c..ac878ce 100644 --- a/tests/unit/test_config.py +++ b/tests/unit/test_config.py @@ -1,7 +1,5 @@ """Table-driven tests for the env-var configuration contract.""" -from __future__ import annotations - import logging import os from pathlib import Path @@ -9,18 +7,15 @@ import pytest from pdf_ops.config import ( - EnvSecret, ExtractConfig, - FileSecret, MergeConfig, OnExists, Operation, OutputEncryption, parse_config, - resolve_secret, ) from pdf_ops.errors import ConfigError -from pdf_ops.secret import Secret +from pdf_ops.secrets import EnvSecret, FileSecret, Secret, resolve_secret pytestmark = pytest.mark.unit diff --git a/tests/unit/test_docs.py b/tests/unit/test_docs.py new file mode 100644 index 0000000..1a5c2ba --- /dev/null +++ b/tests/unit/test_docs.py @@ -0,0 +1,37 @@ +"""Documentation that doubles as a contract stays in step with the code.""" + +import re +from pathlib import Path + +import pytest + +pytestmark = pytest.mark.unit + +ROOT = Path(__file__).resolve().parents[2] +SRC = ROOT / "src" / "pdf_ops" + +# Every way the package names a log event: a logger call, the terminal emitter, +# and the third-party filter that rewrites foreign records. +_EVENT_PATTERNS = ( + re.compile(r'logger\.(?:debug|info|warning|error)\(\s*"([a-z_]+)"'), + re.compile(r'emit_terminal\([^)]*?"([a-z_]+)"', re.DOTALL), + re.compile(r'record\.msg = "([a-z_]+)"'), +) + + +def test_every_module_is_in_the_architecture_doc() -> None: + doc = (ROOT / "docs" / "ARCHITECTURE.md").read_text() + modules = sorted(p.name for p in SRC.glob("*.py") if p.name != "__init__.py") + missing = [name for name in modules if name not in doc] + assert modules and not missing, missing + + +def test_every_log_event_is_documented() -> None: + emitted: set[str] = set() + for path in SRC.glob("*.py"): + text = path.read_text() + for pattern in _EVENT_PATTERNS: + emitted.update(pattern.findall(text)) + guide = (ROOT / "docs" / "OPERATIONS.md").read_text() + missing = sorted(name for name in emitted if f"`{name}`" not in guide) + assert emitted and not missing, missing diff --git a/tests/unit/test_encryption_label.py b/tests/unit/test_encryption_label.py new file mode 100644 index 0000000..af15bae --- /dev/null +++ b/tests/unit/test_encryption_label.py @@ -0,0 +1,44 @@ +"""The failed-open encryption label: a raw scan of an attacker-controlled +file that must stay correct on real layouts and harmless on hostile ones.""" + +from collections.abc import Callable +from pathlib import Path + +import pytest + +from pdf_ops.engine_pikepdf import ( + _describe_encryption_raw, +) + +pytestmark = pytest.mark.unit + + +class TestDescribeEncryptionRaw: + def test_aes256_labelled(self, make_encrypted_pdf: Callable[..., Path]) -> None: + locked = make_encrypted_pdf(password="pw", algorithm="AES-256") + assert _describe_encryption_raw(locked) == "AES-256" + + def test_rc4_128_exact_not_a_stream_length( + self, make_encrypted_pdf: Callable[..., Path] + ) -> None: + # The label must come from the /Encrypt object's own body - the + # first /Length token in the file belongs to some content stream. + locked = make_encrypted_pdf(password="pw") + assert _describe_encryption_raw(locked) == "RC4-128" + + def test_appended_digit_wall_neither_crashes_nor_mislabels( + self, make_encrypted_pdf: Callable[..., Path], tmp_path: Path + ) -> None: + # int() on an unbounded attacker-controlled digit run raises under + # the integer-conversion limit; the bounded token match must neither + # crash nor let a decoy /V win over the real /Encrypt object. + locked = make_encrypted_pdf(password="pw", algorithm="AES-256") + decoy = tmp_path / "decoy.pdf" + decoy.write_bytes(locked.read_bytes() + b"\n% /V " + b"5" * 6000 + b"\n") + assert _describe_encryption_raw(decoy) == "AES-256" + + def test_unencrypted_or_garbage_is_unknown(self, tmp_path: Path) -> None: + blob = tmp_path / "blob.pdf" + blob.write_bytes(b"%PDF-1.7\nno encryption anywhere\n%%EOF\n") + assert _describe_encryption_raw(blob) == "unknown" + assert _describe_encryption_raw(tmp_path / "missing.pdf") == "unknown" diff --git a/tests/unit/test_engine_translation.py b/tests/unit/test_engine_translation.py new file mode 100644 index 0000000..bbc7090 --- /dev/null +++ b/tests/unit/test_engine_translation.py @@ -0,0 +1,24 @@ +"""The structure-walk translation net: a builtin exception raised while +walking a document's structure is a data problem, never an internal error.""" + +from pathlib import Path + +import pytest + +from pdf_ops.engine_pikepdf import ( + _STRUCTURE_FAILURES, + _translating, +) +from pdf_ops.errors import ErrorCode, ExitCode, InvalidPdfError + +pytestmark = pytest.mark.unit + + +@pytest.mark.parametrize("exc_type", _STRUCTURE_FAILURES) +def test_builtin_failure_inside_a_walk_classifies_as_corrupt( + exc_type: type[Exception], +) -> None: + with pytest.raises(InvalidPdfError) as caught, _translating(Path("hostile.pdf")): + raise exc_type("hostile shape") + assert caught.value.error_code == ErrorCode.CORRUPT_PDF + assert caught.value.exit_code == ExitCode.INVALID_PDF diff --git a/tests/unit/test_errors.py b/tests/unit/test_errors.py index 0d94dbd..2a436b5 100644 --- a/tests/unit/test_errors.py +++ b/tests/unit/test_errors.py @@ -1,11 +1,13 @@ """The exception-to-exit-code mapping is the external API - pin it.""" -from __future__ import annotations +import re +from pathlib import Path import pytest from pdf_ops.errors import ( ConfigError, + ErrorCode, ExitCode, InputError, InvalidPdfError, @@ -28,7 +30,7 @@ ], ) def test_error_class_maps_to_exit_code(error_class: type[PdfOpsError], expected_code: int) -> None: - err = error_class("boom", error_code="SOME_CODE") + err = error_class("boom", error_code=ErrorCode.CORRUPT_PDF) assert int(err.exit_code) == expected_code @@ -37,7 +39,7 @@ def test_exit_code_values_are_stable() -> None: def test_error_carries_code_message_and_context() -> None: - err = ConfigError("bad value", error_code="INVALID_OPERATION", context={"var": "X"}) + err = ConfigError("bad value", error_code=ErrorCode.INVALID_OPERATION, context={"var": "X"}) assert err.message == "bad value" assert err.error_code == "INVALID_OPERATION" assert err.context == {"var": "X"} @@ -45,5 +47,15 @@ def test_error_carries_code_message_and_context() -> None: def test_context_defaults_to_empty_dict() -> None: - err = InputError("missing", error_code="INPUT_MISSING") + err = InputError("missing", error_code=ErrorCode.INPUT_MISSING) assert err.context == {} + + +def test_every_error_code_is_documented() -> None: + # The error-code table in docs/OPERATIONS.md is the operator-facing + # vocabulary; the enum is the code-facing one. Neither may drift. + guide = Path(__file__).resolve().parents[2] / "docs" / "OPERATIONS.md" + documented = set( + re.findall(r"^\| `([A-Z_]+)` \| [1-6] \|", guide.read_text(), flags=re.MULTILINE) + ) + assert documented == {code.value for code in ErrorCode} diff --git a/tests/unit/test_logging.py b/tests/unit/test_logging.py index 1aa2291..ea123ff 100644 --- a/tests/unit/test_logging.py +++ b/tests/unit/test_logging.py @@ -1,39 +1,22 @@ """The JSON log line shape is an operator interface - pin it.""" -from __future__ import annotations - import json import logging +import sys import pytest from pdf_ops.logging_setup import ( JsonFormatter, - _ThirdPartyEventFilter, # pyright: ignore[reportPrivateUsage] + _ThirdPartyEventFilter, emit_terminal, setup_logging, ) +from tests.helpers import make_record pytestmark = pytest.mark.unit -def make_record( - msg: str = "some_event", extra: dict[str, object] | None = None -) -> logging.LogRecord: - record = logging.LogRecord( - name="pdf_ops", - level=logging.INFO, - pathname=__file__, - lineno=1, - msg=msg, - args=None, - exc_info=None, - ) - for key, value in (extra or {}).items(): - setattr(record, key, value) - return record - - class TestJsonFormatter: def test_emits_valid_json_with_required_fields(self) -> None: payload = json.loads(JsonFormatter().format(make_record())) @@ -43,7 +26,7 @@ def test_emits_valid_json_with_required_fields(self) -> None: assert payload["ts"].endswith("+00:00") def test_extra_fields_are_merged_into_payload(self) -> None: - record = make_record(extra={"operation": "merge", "exit_code": 2, "context": {"a": 1}}) + record = make_record(operation="merge", exit_code=2, context={"a": 1}) payload = json.loads(JsonFormatter().format(record)) assert payload["operation"] == "merge" assert payload["exit_code"] == 2 @@ -53,14 +36,13 @@ def test_exception_info_is_included(self) -> None: try: raise ValueError("boom") except ValueError: - record = make_record() - record.exc_info = __import__("sys").exc_info() + record = make_record(exc_info=sys.exc_info()) payload = json.loads(JsonFormatter().format(record)) assert payload["exc_type"] == "ValueError" assert "boom" in payload["traceback"] def test_unserializable_values_fall_back_to_str(self) -> None: - record = make_record(extra={"path": object()}) + record = make_record(path=object()) payload = json.loads(JsonFormatter().format(record)) assert isinstance(payload["path"], str) @@ -126,15 +108,7 @@ def test_includes_exception_info_when_requested( class TestThirdPartyEventFilter: def pypdf_record(self, message: str) -> logging.LogRecord: - return logging.LogRecord( - name="pypdf", - level=logging.WARNING, - pathname=__file__, - lineno=1, - msg=message, - args=None, - exc_info=None, - ) + return make_record(message, name="pypdf", level=logging.WARNING) def test_saslprep_codepoints_are_masked(self) -> None: # pypdf's SASLprep warning names the exact codepoint of a password diff --git a/tests/unit/test_sanitize.py b/tests/unit/test_sanitize.py index 6e05ef9..9bed841 100644 --- a/tests/unit/test_sanitize.py +++ b/tests/unit/test_sanitize.py @@ -4,8 +4,6 @@ filesystem - this table is the security contract for that boundary. """ -from __future__ import annotations - import pytest from pdf_ops.engine import Attachment diff --git a/tests/unit/test_secret.py b/tests/unit/test_secrets.py similarity index 56% rename from tests/unit/test_secret.py rename to tests/unit/test_secrets.py index f241bf9..76e42c5 100644 --- a/tests/unit/test_secret.py +++ b/tests/unit/test_secrets.py @@ -1,10 +1,10 @@ """The Secret wrapper and the log-redaction layer - the no-leak machinery.""" -from __future__ import annotations - import json import logging +import sys from collections.abc import Generator +from typing import Any import pytest @@ -13,11 +13,24 @@ clear_registered_secrets, register_secret_value, ) -from pdf_ops.secret import Secret +from pdf_ops.secrets import Secret +from tests.helpers import make_record pytestmark = pytest.mark.unit +@pytest.fixture(autouse=True) +def clean_registry() -> Generator[None]: + # The redaction registry is process-global; every test starts and ends empty. + clear_registered_secrets() + yield + clear_registered_secrets() + + +def format_record(msg: str, **extra: Any) -> dict[str, object]: + return json.loads(JsonFormatter().format(make_record(msg, level=logging.ERROR, **extra))) + + class TestSecret: def test_never_leaks_through_repr_str_or_format(self) -> None: secret = Secret("hunter2") @@ -32,35 +45,15 @@ def test_bool_reflects_emptiness(self) -> None: class TestRedactionFilter: - @pytest.fixture(autouse=True) - def _clean_registry(self) -> Generator[None]: - clear_registered_secrets() - yield - clear_registered_secrets() - - def format_record(self, **extra: object) -> dict[str, object]: - record = logging.LogRecord( - name="pdf_ops", - level=logging.ERROR, - pathname=__file__, - lineno=1, - msg="an_event", - args=None, - exc_info=None, - ) - for key, value in extra.items(): - setattr(record, key, value) - return json.loads(JsonFormatter().format(record)) - def test_registered_value_scrubbed_from_string_fields(self) -> None: register_secret_value("hunter2") - payload = self.format_record(detail="failed with password hunter2 somewhere") + payload = format_record("an_event", detail="failed with password hunter2 somewhere") assert "hunter2" not in json.dumps(payload) assert payload["detail"] == "failed with password *** somewhere" def test_scrub_recurses_into_context_dicts_and_lists(self) -> None: register_secret_value("hunter2") - payload = self.format_record(context={"inputs": ["a hunter2 b"], "note": "hunter2"}) + payload = format_record("an_event", context={"inputs": ["a hunter2 b"], "note": "hunter2"}) serialized = json.dumps(payload) assert "hunter2" not in serialized assert "***" in serialized @@ -70,72 +63,41 @@ def test_traceback_payloads_are_scrubbed(self) -> None: try: raise RuntimeError("boom hunter2") except RuntimeError: - import sys - - record = logging.LogRecord( - name="pdf_ops", - level=logging.ERROR, - pathname=__file__, - lineno=1, - msg="operation_failed", - args=None, - exc_info=sys.exc_info(), - ) - payload = json.loads(JsonFormatter().format(record)) + payload = format_record("operation_failed", exc_info=sys.exc_info()) assert "hunter2" not in json.dumps(payload) assert "boom ***" in str(payload["traceback"]) class TestScrubIntegrity: - @pytest.fixture(autouse=True) - def _clean(self) -> Generator[None]: - clear_registered_secrets() - yield - clear_registered_secrets() - - def make_payload(self, **extra: object) -> dict[str, object]: - record = logging.LogRecord( - name="pdf_ops", - level=logging.INFO, - pathname=__file__, - lineno=1, - msg="merge_written", - args=None, - exc_info=None, - ) - for key, value in extra.items(): - setattr(record, key, value) - return json.loads(JsonFormatter().format(record)) - def test_token_fields_never_scrubbed(self) -> None: # A password equal to a known token ("merge") must not rewrite # code-controlled fields: doing so both breaks workflow-engine # matching and acts as a password oracle. register_secret_value("merge") - payload = self.make_payload(operation="merge", error_code="MERGE_FAILED") + payload = format_record("merge_written", operation="merge", error_code="MERGE_FAILED") assert payload["event"] == "merge_written" assert payload["operation"] == "merge" assert payload["error_code"] == "MERGE_FAILED" def test_free_text_fields_are_scrubbed(self) -> None: register_secret_value("merge") - payload = self.make_payload(detail="library said merge is wrong") + payload = format_record("merge_written", detail="library said merge is wrong") assert payload["detail"] == "library said *** is wrong" def test_overlapping_secrets_scrub_longest_first(self) -> None: register_secret_value("Spring2026") register_secret_value("Spring2026!x9") - payload = self.make_payload(detail="bad key 'Spring2026!x9' rejected") + payload = format_record("merge_written", detail="bad key 'Spring2026!x9' rejected") assert payload["detail"] == "bad key '***' rejected" assert "!x9" not in str(payload["detail"]) def test_repr_escaped_variant_also_scrubbed(self) -> None: register_secret_value("back\\slash-pw") # a library embedding the value via %r doubles the backslash - payload = self.make_payload(detail="rejected 'back\\\\slash-pw' here") + payload = format_record("merge_written", detail="rejected 'back\\\\slash-pw' here") assert "slash-pw" not in str(payload["detail"]) def test_too_short_secrets_are_not_registered(self) -> None: assert register_secret_value("abc") is False - payload = self.make_payload(detail="abc appears here") + payload = format_record("merge_written", detail="abc appears here") assert payload["detail"] == "abc appears here" diff --git a/uv.lock b/uv.lock index 27d90eb..df479a1 100644 --- a/uv.lock +++ b/uv.lock @@ -165,6 +165,80 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/cb/b1/3846dd7f199d53cb17f49cba7e651e9ce294d8497c8c150530ed11865bb8/iniconfig-2.3.0-py3-none-any.whl", hash = "sha256:f631c04d2c48c52b84d0d0549c99ff3859c98df65b3101406327ecc7d53fbf12", size = 7484, upload-time = "2025-10-18T21:55:41.639Z" }, ] +[[package]] +name = "lxml" +version = "6.1.3" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/23/ad/28ecd7cb894d172f3c9c80a075eeeb2017ac62e3632cee05a5f9493547eb/lxml-6.1.3.tar.gz", hash = "sha256:45222d94ddd511536f3b2f7d9deae3b2339b4ce0f075f1ca25703b07cad9dd21", size = 4211198, upload-time = "2026-09-02T14:48:02.287Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/0c/15/fc75a70b0af6021d0ea16811f1fc71cc42cd06ce90fe10f007a69b2eed84/lxml-6.1.3-cp314-cp314-macosx_10_15_universal2.whl", hash = "sha256:2bec13085dc8ef48a3fe62f7dfcacfeda2c785cdf19cc8eeda2bb9ed081da165", size = 8609725, upload-time = "2026-09-02T14:49:00.156Z" }, + { url = "https://files.pythonhosted.org/packages/84/ef/398fcf9018f881ec9aeaafae1ddd6586dfb13314a35d35e899de373dcae0/lxml-6.1.3-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:4f4db7c7e954d289d71878938348b3d91b904a3e8210a11939359fb758a58e7d", size = 4639629, upload-time = "2026-09-02T14:49:02.81Z" }, + { url = "https://files.pythonhosted.org/packages/a7/2d/49b6a6ad7ce8f64b07b9fe852ff0c6d3fcbb26db61bee4f63d4120180a1c/lxml-6.1.3-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:2cae5d5c90a62d9139c512a0cb1aad1d182b022b5740daea2617eb5bf7fc658e", size = 4965074, upload-time = "2026-09-02T14:49:05.133Z" }, + { url = "https://files.pythonhosted.org/packages/66/bc/6230cf80e4331c33383b0b6b73dc31a393dd76edd4cb73d761de5123034d/lxml-6.1.3-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:c6c0c13128a32eb04a51357e56a094e13aa8e6d3d1884de2e9ae923f6915e1a8", size = 5099355, upload-time = "2026-09-02T14:49:07.343Z" }, + { url = "https://files.pythonhosted.org/packages/ac/cf/d1143d9b7717e07a82f158a1fc9ce6e581fdad1226734950af869e3ffde4/lxml-6.1.3-cp314-cp314-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:2221e88679d1351e9a40aaee54bc65679b9795bbd0160bc3d5e36b163344eb75", size = 5036795, upload-time = "2026-09-02T14:49:09.65Z" }, + { url = "https://files.pythonhosted.org/packages/31/6f/194bb00ffb89712c30f5a7e1b8e685590e140fad6c8261fec172c09a3dc0/lxml-6.1.3-cp314-cp314-manylinux_2_26_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:cfb398886a7eb4c719161c3efcff2a1248febc53a4d8e5072d2d8a87fed84ac9", size = 5658740, upload-time = "2026-09-02T14:49:11.9Z" }, + { url = "https://files.pythonhosted.org/packages/e9/44/27e3cee3dcdb3b7bc09727b642bdbfcd098490ea77df04611db9060d7722/lxml-6.1.3-cp314-cp314-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:a7eb78ba28b187e1e9203a55c60fcf70df2d22cb205fe6d51b9383d6097419f0", size = 5245991, upload-time = "2026-09-02T14:49:14.154Z" }, + { url = "https://files.pythonhosted.org/packages/ca/e9/8312560579fc980bbd2233a8a673cc46f7d613d3633f2bf08a21e8f4ad13/lxml-6.1.3-cp314-cp314-manylinux_2_28_i686.whl", hash = "sha256:ea6b1e9105b4b24a34c722432d9fb578f9ed83af21fa1abda639011e0f22bbb6", size = 5354136, upload-time = "2026-09-02T14:49:16.459Z" }, + { url = "https://files.pythonhosted.org/packages/74/d8/eda60f4f73a9c780b5d6e1175484f66e6c81a2c93346e2906a1fec9c7a02/lxml-6.1.3-cp314-cp314-manylinux_2_31_armv7l.whl", hash = "sha256:e8b17e23df3e827a69d25af70990ca2420e92668aaffaeeb3cd2351d7916a023", size = 4704379, upload-time = "2026-09-02T14:49:19.032Z" }, + { url = "https://files.pythonhosted.org/packages/ba/c8/c9cc60057be78ac34bd2b842e45e6e88edbfe5e532e82c3b82381b7aab49/lxml-6.1.3-cp314-cp314-manylinux_2_38_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:1b7c37339d7e75cab9a123a04248e243cefefb302ad6db566ea0c77cbcde421e", size = 5258676, upload-time = "2026-09-02T14:49:21.306Z" }, + { url = "https://files.pythonhosted.org/packages/41/7b/66894008fee8d1785b8db129747ae963fd427b68f456918df7f2f24a8b98/lxml-6.1.3-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:83e3a51e7933db700a0da0db31849db3a24022d9970da9bb73001e1d0326fd92", size = 5090069, upload-time = "2026-09-02T14:49:23.562Z" }, + { url = "https://files.pythonhosted.org/packages/8b/31/c1b60404859f4c3cd1f41f29c65a24e25cea78fde822d9574a21f66810be/lxml-6.1.3-cp314-cp314-musllinux_1_2_armv7l.whl", hash = "sha256:9bde9ae026a55b9a192078dfa6e27dd0ca4a050171ab6272e92f97b757dfdf48", size = 4741958, upload-time = "2026-09-02T14:49:26.037Z" }, + { url = "https://files.pythonhosted.org/packages/23/b8/6285f0cf546f14da2554cabdeaf7c2c2ff3190c74807f0de2e8810a786f9/lxml-6.1.3-cp314-cp314-musllinux_1_2_ppc64le.whl", hash = "sha256:1a635e837b50a1819bebfedaac5916498ea024120969da8790500148fb0a894d", size = 5683245, upload-time = "2026-09-02T14:49:28.438Z" }, + { url = "https://files.pythonhosted.org/packages/d3/f6/2168cab44336dcb15fed0f0b78577225b83297cdf0dee349c95420c3dcb0/lxml-6.1.3-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:d0c5c362bc94f1929dc7e96e715bbe7bd17037f802e6d8f0d1545df9133c0559", size = 5246087, upload-time = "2026-09-02T14:49:30.955Z" }, + { url = "https://files.pythonhosted.org/packages/f5/89/32f5de69a0a31f30e6164981851f87b37ecb2c4ee838e504b88d49d4818e/lxml-6.1.3-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:c59e4265608da6a041f54646ecc0c9ecdbb19aaf14c4c684bb6c2114998cc415", size = 5269352, upload-time = "2026-09-02T14:49:33.502Z" }, + { url = "https://files.pythonhosted.org/packages/a2/a1/741d952ed3a7ef7a50055c6415aec3f067015e97f72f4389ce77b09657ba/lxml-6.1.3-cp314-cp314-win32.whl", hash = "sha256:2e62c569ec7531b679b184cbfe335c501c1d13c4b363560013019962eb630e6d", size = 3662783, upload-time = "2026-09-02T14:50:23.751Z" }, + { url = "https://files.pythonhosted.org/packages/0f/bc/5811cc73cac05e324e05ba9b0924e1a163a317a167ede8a9c748b11db30a/lxml-6.1.3-cp314-cp314-win_amd64.whl", hash = "sha256:66299564c046bc7e0cc5de5106601eae907e9fa5904cd68a323380a8502f7861", size = 4073951, upload-time = "2026-09-02T14:50:26.348Z" }, + { url = "https://files.pythonhosted.org/packages/92/18/3768c8b01ac3a9bed1914715e6011711b00e2a11628ffa6f7fa37f8e0269/lxml-6.1.3-cp314-cp314-win_arm64.whl", hash = "sha256:ebd054ad1737a68fb7c5c073d405cef2b88bb824e294de3b4a4e995b47f0e376", size = 3749279, upload-time = "2026-09-02T14:50:28.749Z" }, + { url = "https://files.pythonhosted.org/packages/72/38/84684784738d9451db2b330de2483f496690c3a5c642071df24135739b37/lxml-6.1.3-cp314-cp314t-macosx_10_15_universal2.whl", hash = "sha256:5a143e6207579de8baeded4eaac9134413200359f1969d636f0bfb98ee8c3c8f", size = 8860296, upload-time = "2026-09-02T14:49:36.346Z" }, + { url = "https://files.pythonhosted.org/packages/24/b7/fc4c50bb1b38e864010ea396046cabe85129bf9e65b11edcfbc37d356241/lxml-6.1.3-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:a1cec0f99b9b914d39176347a93b7610dc09324491aee1cbc57cd291a41a1d55", size = 4755190, upload-time = "2026-09-02T14:49:39.872Z" }, + { url = "https://files.pythonhosted.org/packages/94/e2/ee9aa6ed2b666b2db1f6f7fd48964ff9da39ebe827ef5eac0ab881f639d9/lxml-6.1.3-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:f6b9d2aad499c769ee8287609ab0e6de99d8bcea99c6e6c2e64945259fd52fb2", size = 4979517, upload-time = "2026-09-02T14:49:42.153Z" }, + { url = "https://files.pythonhosted.org/packages/29/e3/e7763d1661b283ddd4fa36f91b9a497db6b8d2aff55028b16c7f642e0755/lxml-6.1.3-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:28a23fefdb345b2d4d0ff2860571b5ff9a89a28b6a120f720e8fb0324d346626", size = 5115270, upload-time = "2026-09-02T14:49:44.493Z" }, + { url = "https://files.pythonhosted.org/packages/2d/cd/22205d5b4d177e3f4156f780412426ee7c7f8107809f119f0dcc40fa51e3/lxml-6.1.3-cp314-cp314t-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:545ccc14fb05485f48b4439ec35beb16d5b5280eb6c81c658bd4707a2a119414", size = 5032449, upload-time = "2026-09-02T14:49:46.841Z" }, + { url = "https://files.pythonhosted.org/packages/da/43/06a4626c3bb79ef8c501b674afab8100d64e798665bb2a97d1c960636a49/lxml-6.1.3-cp314-cp314t-manylinux_2_26_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:93476b6514b373fc6ca67d26c442784f7807c86f00635bfe79f935c3eab2af17", size = 5603325, upload-time = "2026-09-02T14:49:49.664Z" }, + { url = "https://files.pythonhosted.org/packages/d0/9c/733682a0c2de9f5779ba207bbb3f3f6be8c6bda863fc01739b186b38783a/lxml-6.1.3-cp314-cp314t-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:8db38ff3fb7aee7d6a82ae4da2eef1178656fe1216841fbd24870062a9d60473", size = 5229023, upload-time = "2026-09-02T14:49:52.447Z" }, + { url = "https://files.pythonhosted.org/packages/c6/8a/e69cdaca3fd33a647942925664f01b20908d41a6968c182305be9c38fb11/lxml-6.1.3-cp314-cp314t-manylinux_2_28_i686.whl", hash = "sha256:25f4118c438f96bb466e83108506d03d5c31b1bd2387e83e5b070bda6ded9c37", size = 5317811, upload-time = "2026-09-02T14:49:55.25Z" }, + { url = "https://files.pythonhosted.org/packages/2e/b2/0c397588174403c2ab68fc464abf97e03e7324f9c6cb6a99023104707195/lxml-6.1.3-cp314-cp314t-manylinux_2_31_armv7l.whl", hash = "sha256:1beb0f9909b26cee938df9ba56b15252a84429b1fc30ce6fca161390b9789a70", size = 4646516, upload-time = "2026-09-02T14:49:57.761Z" }, + { url = "https://files.pythonhosted.org/packages/56/7e/cfea25afafbe49db8b225764f7f74bb37c2a7f5e717d917d3d4a5e098ed4/lxml-6.1.3-cp314-cp314t-manylinux_2_38_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:3a27ac6c780c8b8a1cd231b58407634cafc1c4cc28cd6c7141362df0f36351e7", size = 5240626, upload-time = "2026-09-02T14:50:00.279Z" }, + { url = "https://files.pythonhosted.org/packages/a1/75/7a587771bb52ebb0e2c57b6dbe9fd96a70fbb54d72ddd97d54c5f8ec18d5/lxml-6.1.3-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:a1932d7ce78a561367512c594fe66eac2b2ec9b9264cfd9b5f950622f4a116e2", size = 5086619, upload-time = "2026-09-02T14:50:03.245Z" }, + { url = "https://files.pythonhosted.org/packages/1e/01/94c0ebe6d831861542d251e038052e52bf6d33f1d18f1cfffdc82851065a/lxml-6.1.3-cp314-cp314t-musllinux_1_2_armv7l.whl", hash = "sha256:7d0f5976aa2701996f759b30172925829867547bb073af0ae67d1307a0f0262c", size = 4758828, upload-time = "2026-09-02T14:50:05.873Z" }, + { url = "https://files.pythonhosted.org/packages/1f/f1/938d67bd0e5b1fdfa52be28aefdffbad57e1f6b8e921c2aab88542c75f40/lxml-6.1.3-cp314-cp314t-musllinux_1_2_ppc64le.whl", hash = "sha256:c5e7ce578aa8a80910a72a8ca0bbea3baae10100827249001999726a788456d8", size = 5627083, upload-time = "2026-09-02T14:50:08.555Z" }, + { url = "https://files.pythonhosted.org/packages/d8/65/4e51522f6c214650db0abb7b16ccd11b1238b8a05a8d59aa4ebed59c9f67/lxml-6.1.3-cp314-cp314t-musllinux_1_2_riscv64.whl", hash = "sha256:d97c5227621af74b111882a290b10f371780a38eef9d9e730408fba2259b52fb", size = 5235170, upload-time = "2026-09-02T14:50:11.255Z" }, + { url = "https://files.pythonhosted.org/packages/92/c2/e73d19365665f6b16ef84df21199befc3b06e4c539046ad2d9595f6fb9ea/lxml-6.1.3-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:da707f14ea3c35ee463d50acd596d6488e4b2b4ae7cf77a5bf93f55c023d63e8", size = 5252273, upload-time = "2026-09-02T14:50:13.782Z" }, + { url = "https://files.pythonhosted.org/packages/48/a9/7f386c84c9fe2854e1ca6e231c285e1c8f392971ac353c6865e6ec49faff/lxml-6.1.3-cp314-cp314t-win32.whl", hash = "sha256:9efe56a68179f3adc4de41861c9358931db03837c48dd5e1c78077b84dd07f3a", size = 3902712, upload-time = "2026-09-02T14:50:16.171Z" }, + { url = "https://files.pythonhosted.org/packages/82/a6/8a3eb793f7900ef01c7f99e6f5fcbcfbdff35251cfaef66b32a4c16352d6/lxml-6.1.3-cp314-cp314t-win_amd64.whl", hash = "sha256:c9389b3784b56c58d933b5e0aecdf28f901b073ff385358d8a7d40907f6e14b2", size = 4400979, upload-time = "2026-09-02T14:50:18.621Z" }, + { url = "https://files.pythonhosted.org/packages/cc/c4/3807bea283b4fe9e9d9f5dde46a73df91178472b335d2778e10b2a37aa22/lxml-6.1.3-cp314-cp314t-win_arm64.whl", hash = "sha256:32a409be3190b088f960ac92bfedfbef2f86c49ff940765e1548177592d20026", size = 3823401, upload-time = "2026-09-02T14:50:21.119Z" }, + { url = "https://files.pythonhosted.org/packages/e1/8e/4614fcd65496054cfb7172662f3576a59200278739506433b8c241ea422a/lxml-6.1.3-cp315-cp315-macosx_10_15_universal2.whl", hash = "sha256:6ea2f13dce778ca072ccee598bca46a092ce192e8fd907b6c1f0e52c800529a0", size = 8609378, upload-time = "2026-09-02T14:50:31.772Z" }, + { url = "https://files.pythonhosted.org/packages/f2/51/2cdce3c65fa99a6195dd8fbd512d33407c1000ad99f63e0a285b63d7a8eb/lxml-6.1.3-cp315-cp315-macosx_10_15_x86_64.whl", hash = "sha256:c581b1d68b3845fb86c6b2983e755b29bf001461c59fa411d2c26a911b6559a9", size = 4640022, upload-time = "2026-09-02T14:50:34.41Z" }, + { url = "https://files.pythonhosted.org/packages/52/09/0b30084e9eb1c546a4be3d9c56df70058d116b1a320400a59b0f7da87bf0/lxml-6.1.3-cp315-cp315-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:2e01125896585139453cab8cb235893644d8815d7509520da95ae3ee8d1c1f79", size = 5037928, upload-time = "2026-09-02T14:50:37.007Z" }, + { url = "https://files.pythonhosted.org/packages/b8/0e/5c37275a3e361f6138dc06db748ea565c1fe8a5f4ee5e2ddd80047c81a89/lxml-6.1.3-cp315-cp315-manylinux_2_26_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:290f66b97ede0e552e1cb44a0fd8a74f9753ee635b50830a0b122fb72788d015", size = 5661932, upload-time = "2026-09-02T14:50:39.777Z" }, + { url = "https://files.pythonhosted.org/packages/70/c5/b71ffb289b15e2642e2a3cf6d468c44da39ea119061a99e5b05e3d10f217/lxml-6.1.3-cp315-cp315-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:73fc05988ed20809450474ba760a87c8ad4e455fc09783c02195e56ec634b41a", size = 5249209, upload-time = "2026-09-02T14:50:42.141Z" }, + { url = "https://files.pythonhosted.org/packages/81/ea/9910da149a23932f9301652e57661cd9e42b0df18f12be21159b7255f92b/lxml-6.1.3-cp315-cp315-manylinux_2_31_armv7l.whl", hash = "sha256:dc3a44689eea43eab836e5c98a8ab015dc2419987d1ea6eafc7c590cdff86bed", size = 4704543, upload-time = "2026-09-02T14:50:44.634Z" }, + { url = "https://files.pythonhosted.org/packages/76/07/9290329cd188c62e22021f79df04ee0cc33d9a93b0d38bd65ccd452ad9d0/lxml-6.1.3-cp315-cp315-manylinux_2_38_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:209c3ccbfe35a04ac6d24f0611f9d1cbf8025d49991b14acd935236234d6c156", size = 5261298, upload-time = "2026-09-02T14:50:47.301Z" }, + { url = "https://files.pythonhosted.org/packages/c9/0c/aba78bd3401cd99b73a0aed8e2b9b43e14be94fab3603d4bbc8a62365f2a/lxml-6.1.3-cp315-cp315-musllinux_1_2_aarch64.whl", hash = "sha256:2f5b2a2b9811b853b39bfa41367c6d78747b8e3e80e07fc5a24aae295c1a4d7d", size = 5090453, upload-time = "2026-09-02T14:50:49.952Z" }, + { url = "https://files.pythonhosted.org/packages/8d/dc/fa4426c3355aa0216cbeb3911495b5f65a26e0df85859a89928fe28f0396/lxml-6.1.3-cp315-cp315-musllinux_1_2_armv7l.whl", hash = "sha256:6a406d0b3cb207b0fa460ed4dc93e866f44f105da0169361cb18ff998a44c7f0", size = 4744709, upload-time = "2026-09-02T14:50:52.394Z" }, + { url = "https://files.pythonhosted.org/packages/be/2b/224fe7918658ab7c532ac2412f3c1eb28f71e6364fb07566262d0cc6a7b6/lxml-6.1.3-cp315-cp315-musllinux_1_2_ppc64le.whl", hash = "sha256:53258656846f5c48996b882fb4b135885e088a3ad3d96b4bc0530f95124d1f69", size = 5685802, upload-time = "2026-09-02T14:50:55.043Z" }, + { url = "https://files.pythonhosted.org/packages/21/44/7d480819b9adcae5f84dd8ac529132c6b7a578544398225cd20321adcd91/lxml-6.1.3-cp315-cp315-musllinux_1_2_riscv64.whl", hash = "sha256:aa633613ff907ea91b9b0489a1f0da1b8725d8c6ccec6b77e8a1c9c235044bb0", size = 5249019, upload-time = "2026-09-02T14:50:57.985Z" }, + { url = "https://files.pythonhosted.org/packages/72/83/385a267ea1b6b283f2249dd827ef360a295e9db14e13ef4665a120c60d64/lxml-6.1.3-cp315-cp315-musllinux_1_2_x86_64.whl", hash = "sha256:90f709b9accab6b2e4d14f5c8718203877a0486bcb3afd74d8b539ecd1e961d4", size = 5271886, upload-time = "2026-09-02T14:51:01.667Z" }, + { url = "https://files.pythonhosted.org/packages/d8/0d/f967b0eb172ae876855a402d6d9b11fa86e3e0c89ca9bbfeadf7ffbfa719/lxml-6.1.3-cp315-cp315-win32.whl", hash = "sha256:b4fc6b03b9d9d90557274f571ab30e7fbbfc527955536935d96f98b6817a86e4", size = 3662894, upload-time = "2026-09-02T14:51:45.173Z" }, + { url = "https://files.pythonhosted.org/packages/f4/48/d8a8c4160a29e663109ad520bac2deb37fcd014756d024561e8bc3e611ec/lxml-6.1.3-cp315-cp315-win_amd64.whl", hash = "sha256:33cadd956b667997e4de1635fce9541f2e8ede2038fcde8cf55aa14d571d1bad", size = 4074626, upload-time = "2026-09-02T14:51:47.77Z" }, + { url = "https://files.pythonhosted.org/packages/25/20/3e1395d34d19f9254625d0b567b81cf70d37d3417be074f4d63b94a2be3c/lxml-6.1.3-cp315-cp315-win_arm64.whl", hash = "sha256:8a330c0ee5fa318c7b5cbbaad882baeca3f570357e7eb25ab34bf31008150758", size = 3749495, upload-time = "2026-09-02T14:51:50.663Z" }, + { url = "https://files.pythonhosted.org/packages/8f/c6/7465ffd9c43883526a382df6fa4846c9d8d419214f7effbf65270e795471/lxml-6.1.3-cp315-cp315t-macosx_10_15_universal2.whl", hash = "sha256:0bf5a3e397df2ec4258eb5eea4c1ac6cf013ca1abd04a176903bff20a70021fe", size = 8857677, upload-time = "2026-09-02T14:51:05.109Z" }, + { url = "https://files.pythonhosted.org/packages/ed/eb/1f3a917e299df43c8162c3e6f64fc2cea3bcf277910f35bff5b8e5d39901/lxml-6.1.3-cp315-cp315t-macosx_10_15_x86_64.whl", hash = "sha256:13d22c0d57355366b393936acf6b98a5e0edeadddd3fccbc6a846c50a76b8741", size = 4754522, upload-time = "2026-09-02T14:51:08.137Z" }, + { url = "https://files.pythonhosted.org/packages/d7/f9/f81b4bdb6efb7a596be29603d8758154d00a5f545db9f3cef9d9041c8f64/lxml-6.1.3-cp315-cp315t-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:cad7617727a96d189bd6f979d0fadf765198c7934e85f4edaba9bf3ad919a300", size = 5033744, upload-time = "2026-09-02T14:51:10.633Z" }, + { url = "https://files.pythonhosted.org/packages/c8/0f/26d9bfaacb319c86e0eca8a1a0bf1130d36a7afbd318883e23caea63763d/lxml-6.1.3-cp315-cp315t-manylinux_2_26_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:cae82b5ca24b0c2beedb269f6e2a96f466acd926879ab00ae19f1a65cbf9ffb0", size = 5615269, upload-time = "2026-09-02T14:51:13.357Z" }, + { url = "https://files.pythonhosted.org/packages/5d/90/73675f3f4141350ed65d6fec533b107d4e802c5caa340cf111771edd86e0/lxml-6.1.3-cp315-cp315t-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:69cafd61aea04ebb3502c93c2aaa568b12931ca0802231e0b5de76bf8b6e74bd", size = 5236280, upload-time = "2026-09-02T14:51:16.051Z" }, + { url = "https://files.pythonhosted.org/packages/fd/be/ed260767e7977de463a0f91f3f4fffcab85c0a2a024a21ffe1fa442c2c79/lxml-6.1.3-cp315-cp315t-manylinux_2_31_armv7l.whl", hash = "sha256:dc205732d593118cf701d986f40e9de7801bb2e371cb189ddbda9b7348f4d97e", size = 4650718, upload-time = "2026-09-02T14:51:19.102Z" }, + { url = "https://files.pythonhosted.org/packages/d0/fd/e9839d03b1e767f2725cf7d7d81b80d5f3f9fdc10ad8827e2479311b046e/lxml-6.1.3-cp315-cp315t-manylinux_2_38_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:88e719b9437f148f7e1465df845c758dd1598618cbea3a2fd1e61a715542f2b2", size = 5243376, upload-time = "2026-09-02T14:51:21.606Z" }, + { url = "https://files.pythonhosted.org/packages/34/a5/4606e347e2788c301f677004aa83e28d24da9fe663a24380122af57be6fc/lxml-6.1.3-cp315-cp315t-musllinux_1_2_aarch64.whl", hash = "sha256:40983eabefd13da003e68170928c7acc011f0d095eefce5871a3c71c9385fb9a", size = 5092340, upload-time = "2026-09-02T14:51:24.21Z" }, + { url = "https://files.pythonhosted.org/packages/ea/99/3314a8661cdf30f493c55a87db283961dfaae08451976a2ca418958e1804/lxml-6.1.3-cp315-cp315t-musllinux_1_2_armv7l.whl", hash = "sha256:fad67b12ffe0f71e02b4932b04883cbc76a9072bbd30731409d3523cf058b011", size = 4758768, upload-time = "2026-09-02T14:51:26.813Z" }, + { url = "https://files.pythonhosted.org/packages/30/58/3bdc577f78ea8b7d72d39a84506f7001d5b28728f43e5b84891e3b7d9a4a/lxml-6.1.3-cp315-cp315t-musllinux_1_2_ppc64le.whl", hash = "sha256:6cd11e7550d89e551a87dcec30f04b1fca32e86b68708aa01a4daa455d8605e5", size = 5649546, upload-time = "2026-09-02T14:51:29.453Z" }, + { url = "https://files.pythonhosted.org/packages/6a/e4/652633de1a2395949ebb7a8fc7d089aba12a2b45f0fefbc9d29e3e3ab3cf/lxml-6.1.3-cp315-cp315t-musllinux_1_2_riscv64.whl", hash = "sha256:ca0ec532ad2f5ba1e5ec120ac157769c57f01855b3d8bf37213f5d88abd9ba0a", size = 5234874, upload-time = "2026-09-02T14:51:32.262Z" }, + { url = "https://files.pythonhosted.org/packages/65/a6/c4581d171de30449304b4859bbd3607e9b40da13c0f88b68e6097c8d785e/lxml-6.1.3-cp315-cp315t-musllinux_1_2_x86_64.whl", hash = "sha256:e99e09ab7741f1281e2677f4c0058c7f5267d182530b09c87e4f6aa26adf3887", size = 5260043, upload-time = "2026-09-02T14:51:34.841Z" }, + { url = "https://files.pythonhosted.org/packages/b8/d7/ed6ee6186a89e69ca4ea9658b2a278f46a5efe8b5d4db56c7197f18653fe/lxml-6.1.3-cp315-cp315t-win32.whl", hash = "sha256:ace1d2c83b2bd24db5940600541140e87a325e119cb32d5fa9ad720d7e76648e", size = 3901093, upload-time = "2026-09-02T14:51:37.234Z" }, + { url = "https://files.pythonhosted.org/packages/67/9d/11d10257a4a048d04195d638bb61f0246ce2448eb05f682bcbab25a257a8/lxml-6.1.3-cp315-cp315t-win_amd64.whl", hash = "sha256:b49638355ea3bebba70da783ccbc630fd72afa16bc46c54474bfa1f9a915bbc6", size = 4395446, upload-time = "2026-09-02T14:51:39.884Z" }, + { url = "https://files.pythonhosted.org/packages/f8/b7/44edd7de434181c582892e68d1ffe6775ca403ce14aea07cb5a218a936cf/lxml-6.1.3-cp315-cp315t-win_arm64.whl", hash = "sha256:5a721a98c649855963811b59b55755b30566e7f7fc40bdc9803d66dee9f811cf", size = 3822836, upload-time = "2026-09-02T14:51:42.471Z" }, +] + [[package]] name = "nodeenv" version = "1.10.0" @@ -188,28 +262,107 @@ name = "pdf-ops" version = "0.1.0" source = { editable = "." } dependencies = [ - { name = "pypdf", extra = ["crypto"] }, + { name = "pikepdf" }, ] [package.dev-dependencies] dev = [ { name = "pre-commit" }, + { name = "pypdf", extra = ["crypto"] }, { name = "pyright" }, { name = "pytest" }, { name = "ruff" }, ] [package.metadata] -requires-dist = [{ name = "pypdf", extras = ["crypto"], specifier = ">=6.16.2" }] +requires-dist = [{ name = "pikepdf", specifier = ">=10.12.0" }] [package.metadata.requires-dev] dev = [ { name = "pre-commit", specifier = ">=4.6.2" }, + { name = "pypdf", extras = ["crypto"], specifier = ">=6.16.2" }, { name = "pyright", specifier = ">=1.1.411" }, { name = "pytest", specifier = ">=9.1.1" }, { name = "ruff", specifier = ">=0.16.5" }, ] +[[package]] +name = "pikepdf" +version = "10.12.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "lxml" }, + { name = "packaging" }, + { name = "pillow" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/1e/d4/f4383bb3ac90cb322cb340cd4253bfc19f80819a97d61a49077ab3a0581e/pikepdf-10.12.0.tar.gz", hash = "sha256:cbc790243a333a2c87bb4c1a69e3d7036b4a7f43c7fafc8ec7cee06985b48ae9", size = 4950459, upload-time = "2026-08-17T22:35:00.878Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/09/9a/8d0477481ab9d6a6e931a1d8c92d0fbb83cbec7a42c267509534c64a8cd0/pikepdf-10.12.0-cp314-abi3-macosx_14_0_arm64.whl", hash = "sha256:52ae9d8fc4650515ab36c7f7a5ea40fdee745bf1f478ea4d761f830905295969", size = 1815496, upload-time = "2026-08-17T22:34:36.422Z" }, + { url = "https://files.pythonhosted.org/packages/4a/6c/cd910cb5292ad1c2d2d0e5edd4f66dfb29004e724a0ad6494b00e1effcb0/pikepdf-10.12.0-cp314-abi3-macosx_15_0_x86_64.whl", hash = "sha256:9a200b2f2fc288e1f225dc601da09b3f83314bc30dada06fdd591169d518ecc3", size = 1908458, upload-time = "2026-08-17T22:34:37.918Z" }, + { url = "https://files.pythonhosted.org/packages/54/f3/1a66c7d2ac251d0ab53c5ff3b1bd79809fd50ed01b50e28f8c0a7740e454/pikepdf-10.12.0-cp314-abi3-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c7797fcc1762640c11caf00407b8fc898ed3b4cc89174109ac261cf17cc0eea1", size = 2068830, upload-time = "2026-08-17T22:34:39.423Z" }, + { url = "https://files.pythonhosted.org/packages/40/ca/1be2aef95a80f1047b67b8ee3480e266125cc1cbf6e09f14cfd222687665/pikepdf-10.12.0-cp314-abi3-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:84c3dcf8aefcc12d4d0642bbbceaf198acfb4a29a4630bbc134423378f824cd8", size = 2262093, upload-time = "2026-08-17T22:34:40.888Z" }, + { url = "https://files.pythonhosted.org/packages/da/b0/8c0c0c4aa681167dd2357b9d864455feb8a36dc09212e8822c603f0f05ef/pikepdf-10.12.0-cp314-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:51fbdbe71d30f97510de4be5b7c4169a3f2364cbd6497aaef9f5a44873ed043c", size = 3707018, upload-time = "2026-08-17T22:34:42.647Z" }, + { url = "https://files.pythonhosted.org/packages/e4/ad/ea403f4b964066f0d5db25ff69e388cfb80bf5727ffaf7e2d36ec5aefddd/pikepdf-10.12.0-cp314-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:1f30d4f83ec2893ed0e41aac2ba6ae4ff4df4e473ad1c0dd0659b81ee52d0958", size = 3904965, upload-time = "2026-08-17T22:34:44.422Z" }, + { url = "https://files.pythonhosted.org/packages/19/a2/872ce36f9e7ea503ebc8e27f0a168b9782768ac7cc14aa127dd098e4d266/pikepdf-10.12.0-cp314-abi3-win_amd64.whl", hash = "sha256:e255b57a58f5e4d7e1e4501d085233a2c074af9771c4f00b2c53d7e4ba6d7611", size = 3447660, upload-time = "2026-08-17T22:34:46.324Z" }, + { url = "https://files.pythonhosted.org/packages/eb/25/5593cf659a925299f67ccd3400feb59ccf11da0657b794f291fc686424b5/pikepdf-10.12.0-cp314-cp314t-macosx_14_0_arm64.whl", hash = "sha256:ae136ed20068b53d46c5c85496c2e214c525104a0aba2b909c5b851a5bd32b79", size = 1821912, upload-time = "2026-08-17T22:34:48.489Z" }, + { url = "https://files.pythonhosted.org/packages/d9/29/fcb71cc848c19766e7b6173da7bc5e795aeb6b5a6d63cb1b13e16c2f8ad4/pikepdf-10.12.0-cp314-cp314t-macosx_15_0_x86_64.whl", hash = "sha256:cdb3ad4d1a3670bcf74fc1c45e80d1b4581873902e5630221585c677d66c3230", size = 1914624, upload-time = "2026-08-17T22:34:50.204Z" }, + { url = "https://files.pythonhosted.org/packages/7f/bf/491e9c6cdc2ec85b8564b5263d1094a926484ab727fe095a1c7ab2fe4ec7/pikepdf-10.12.0-cp314-cp314t-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:7ab1aba5704ea66634c8d7071bd1a25c0c414aae13212707a8936e34ddc9902f", size = 2075370, upload-time = "2026-08-17T22:34:51.957Z" }, + { url = "https://files.pythonhosted.org/packages/52/bd/744278d477cad6caa0568a210f72a8ed482562ca73f01e23361946fb254b/pikepdf-10.12.0-cp314-cp314t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:ee8df5af9a9f540bc94403e2ced2b35696374a6c064c15e3ec6e0b143d44f420", size = 2267388, upload-time = "2026-08-17T22:34:53.742Z" }, + { url = "https://files.pythonhosted.org/packages/25/8f/0e64466bf683d68be081c6e9b9fa03f2aef4f36c72e2c698b80e3c5977fa/pikepdf-10.12.0-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:323e6f6f2411c85af3adab1cd9cdf8e5420ab21786a2ffa780c29d80c2f48e31", size = 3712912, upload-time = "2026-08-17T22:34:55.376Z" }, + { url = "https://files.pythonhosted.org/packages/b2/e4/c1af394d36f7e9198fa0eeecddfdc74079b723f4544d82b0d3b71ab87d64/pikepdf-10.12.0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:88c811eb3d77cdba37c56b4ad2ca521d636501fbcbb4040b0d179f6d290fea22", size = 3912679, upload-time = "2026-08-17T22:34:56.943Z" }, + { url = "https://files.pythonhosted.org/packages/ac/6b/a14cc1a0cfcd07b07841e6861d4dfc650f4f2ff54879bd8c2455d81aeac8/pikepdf-10.12.0-cp314-cp314t-win_amd64.whl", hash = "sha256:2b819d52b63768fd4c33dea64492dca558f6c08048a98617c108b9ca608ed4cb", size = 3480192, upload-time = "2026-08-17T22:34:58.963Z" }, +] + +[[package]] +name = "pillow" +version = "12.3.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/1c/3d/bb7fca845737cf9d7dbde16ed1843984665ff2e0a518f5db43e77ec540b9/pillow-12.3.0.tar.gz", hash = "sha256:3b8182a766685eaa002637e28b4ec8d6b18819a0c71f579bf0dbaa5830297cce", size = 47025035, upload-time = "2026-07-01T11:56:38.965Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/dc/01/001f65b68192f0228cc1dbbc8d2530ab5d58b61037ba0587f946fea607cd/pillow-12.3.0-cp314-cp314-ios_13_0_arm64_iphoneos.whl", hash = "sha256:9cf95fe4d0f84c82d282745d9bb08ad9f926efa00be4697e767b814ce40d4330", size = 4161736, upload-time = "2026-07-01T11:54:51.156Z" }, + { url = "https://files.pythonhosted.org/packages/1a/d2/0219746d0fd16fc8a84498e79452375be3797d3ce4044596ce565164b84f/pillow-12.3.0-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:8728f216dcdb6e6d555cf971cb34076139ad74b31fc2c14da4fafc741c5f6217", size = 4255435, upload-time = "2026-07-01T11:54:53.414Z" }, + { url = "https://files.pythonhosted.org/packages/c8/02/8d0bc62ef0302318c46ff2a512822d2610e81c7aa46c9b3abe6cbaca5ad0/pillow-12.3.0-cp314-cp314-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:a45650e8ce7fafffd731db8550230db6b0d306d181a90b67d3e6bca2f1990930", size = 3696262, upload-time = "2026-07-01T11:54:55.739Z" }, + { url = "https://files.pythonhosted.org/packages/85/e2/73c77d218410b14f5f2d565e8a998d5317b7b9c75368d29985139f7a46f0/pillow-12.3.0-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:ba54cfebe86920a559a7c4d6b9050791c20513650a1952ebe3368c7dc70306f8", size = 5350344, upload-time = "2026-07-01T11:54:57.657Z" }, + { url = "https://files.pythonhosted.org/packages/c7/da/32c752228ae345f489e3a42499d817b6c3996da7e8a3bc7a04fc806b243b/pillow-12.3.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:e158cb00350dc278f3b91551101aa7d12415a66ebf2c91d8d5ac14e56ddd3ad0", size = 4780131, upload-time = "2026-07-01T11:54:59.713Z" }, + { url = "https://files.pythonhosted.org/packages/b1/9d/8b2c807dbef61a5197c047afe99823787eb66f63daf9fb2432f91d6f0462/pillow-12.3.0-cp314-cp314-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:e9aeb04d6aef139de265b29683e119b638208f88cf73cdd1658aa07221165321", size = 6263757, upload-time = "2026-07-01T11:55:01.778Z" }, + { url = "https://files.pythonhosted.org/packages/5c/44/c85361f65dbe00eea8576ee467c768d25129989efb76e94f205e9ca9bb46/pillow-12.3.0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:251bf95b67017e27b13d82f5b326234ca62d70f9cf4c2b9032de2358a3b12c7b", size = 6936962, upload-time = "2026-07-01T11:55:03.93Z" }, + { url = "https://files.pythonhosted.org/packages/18/7e/e483414b35800b86b6f08dbbc7803fb5cd52c4d6f897f47d53ea2c7e6f65/pillow-12.3.0-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:fe3cca2e4e8a592be0f269a1ca4835c25199d9f3ce815c8491048f785b0a0198", size = 6339171, upload-time = "2026-07-01T11:55:05.989Z" }, + { url = "https://files.pythonhosted.org/packages/f0/f4/68c491844841ede6bed70189546b3ee9731cf9f2cbad396faff5e1ccba45/pillow-12.3.0-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:23aceaa007d6172b02c277f0cd359c79492bbb14f7072b4ede9fbcaf20648130", size = 7048116, upload-time = "2026-07-01T11:55:08.131Z" }, + { url = "https://files.pythonhosted.org/packages/a3/34/77f3f793fed8efc7d243f21b33c5a3f0d1c97ee70346d3db855587e155ff/pillow-12.3.0-cp314-cp314-win32.whl", hash = "sha256:af8d94b0db561cf68b88a267c5c44b49e134f525d0dc2cb7ed413a66bc23559a", size = 6467209, upload-time = "2026-07-01T11:55:10.408Z" }, + { url = "https://files.pythonhosted.org/packages/f1/e0/492879f69d94f91f60fc8cd05ba03650e9520afebb2fb7aa12777d7c7f38/pillow-12.3.0-cp314-cp314-win_amd64.whl", hash = "sha256:fdafc9cce40277e0f7a0feabce0ee50dd2fa1800f3b38015e51296b5e814048d", size = 7237707, upload-time = "2026-07-01T11:55:12.745Z" }, + { url = "https://files.pythonhosted.org/packages/c9/ac/6b11f2875f1c2ac040d84e1bbf9cf22a88038f901ca1037898b280b38365/pillow-12.3.0-cp314-cp314-win_arm64.whl", hash = "sha256:e91206ee562682b51b98ef4b26a6ef48fd84e15fd4c4bc5ec768eb641d206838", size = 2565995, upload-time = "2026-07-01T11:55:14.736Z" }, + { url = "https://files.pythonhosted.org/packages/52/69/c2208e56af9bfc1913afb24020297a691eb1d4ef688474c8a04913f65e04/pillow-12.3.0-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:164b31cd1a0490ab6efae01aa5df49da7061be0af1b30e035b6e9a1bfe34ee6e", size = 5352503, upload-time = "2026-07-01T11:55:17.076Z" }, + { url = "https://files.pythonhosted.org/packages/07/70/e5686d753e898a45d778ff1718dba8516ead6ab6b95d85fc8c4b70650cf2/pillow-12.3.0-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:5afb51d599ea772b8365ae807ae557f18bccfe46ab261fd1c2a9ed700fc6eb17", size = 4782956, upload-time = "2026-07-01T11:55:19.448Z" }, + { url = "https://files.pythonhosted.org/packages/d5/37/25c6692f06927ee973ff18c8d9ee98ad0b4d84ee67a09610c2dd1447958e/pillow-12.3.0-cp314-cp314t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:3edce1d53195db527e0191f84b71d02022de0540bf43a16ed734ed7537b07385", size = 6322855, upload-time = "2026-07-01T11:55:21.613Z" }, + { url = "https://files.pythonhosted.org/packages/cc/91/420637fcb8f1bc11029e403b4538e6694744428d8246118e45719f944556/pillow-12.3.0-cp314-cp314t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:bf16ba1b4d0b6b7c8e534936632270cf70eb00dbe09005bc345b2677b726855c", size = 6989642, upload-time = "2026-07-01T11:55:24.006Z" }, + { url = "https://files.pythonhosted.org/packages/10/08/b94d7811281ccf0d143a1cf768d1c49e1e54af63e7b708ab2ee3eb87face/pillow-12.3.0-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:24870b09b224f7ae3c39ed07d10e819d06f8720bc551847b1d623832b5b0e28d", size = 6391281, upload-time = "2026-07-01T11:55:26.252Z" }, + { url = "https://files.pythonhosted.org/packages/d2/87/24233f785f55474dc02ce3e739c5528a77e3a862e9333d1dd7a25cc31f70/pillow-12.3.0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:30f2aa603c41533cc25c05acd0da21636e84a315768feb631c937177db558931", size = 7096716, upload-time = "2026-07-01T11:55:28.318Z" }, + { url = "https://files.pythonhosted.org/packages/23/26/fcb2f6e37175b04f53570b59937867e2b80ee1685e744023153028fc14f9/pillow-12.3.0-cp314-cp314t-win32.whl", hash = "sha256:4b0a7fe987b14c31ebda6083f74f22b561fd3739bc0ac51e019622e3d72668c7", size = 6474125, upload-time = "2026-07-01T11:55:30.956Z" }, + { url = "https://files.pythonhosted.org/packages/90/de/3634abee5f1c9e13c56787b7d5517b0ba8d6de51700b95578cf338349c9f/pillow-12.3.0-cp314-cp314t-win_amd64.whl", hash = "sha256:962864dc93511324d51ddbb5b9f8731bf71675b93ca612a07441896f4688fb8c", size = 7242939, upload-time = "2026-07-01T11:55:34.044Z" }, + { url = "https://files.pythonhosted.org/packages/ce/2a/fd13f8eb24de5714a6eb444a3d67e2842c6c576e159a43793adf23051351/pillow-12.3.0-cp314-cp314t-win_arm64.whl", hash = "sha256:0740a512dc522224c77d9aa5a8d70d8b7d73fb91f2c21125d8d025d3b8990e45", size = 2567506, upload-time = "2026-07-01T11:55:35.988Z" }, + { url = "https://files.pythonhosted.org/packages/5d/dc/8fdce34ec725a33c81c6ba122b904d6b9024e50ea9ac7bede62fab54506c/pillow-12.3.0-cp315-cp315-ios_13_0_arm64_iphoneos.whl", hash = "sha256:0feb2e9d6ad6c9e3c06effe9d00f3f1e618a6643273576b016f591e9315a7139", size = 4162063, upload-time = "2026-07-01T11:55:37.941Z" }, + { url = "https://files.pythonhosted.org/packages/76/66/2044b9a63d3b84ff048228dfcb7cd9bf0df983e8470971bf7d4c57b693de/pillow-12.3.0-cp315-cp315-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:9e881fca225083806662a5c43d627d215f258ff43c890f831966c7d7ba9c7402", size = 4255549, upload-time = "2026-07-01T11:55:40.022Z" }, + { url = "https://files.pythonhosted.org/packages/52/7e/1f67e6f4ece6b582ee4b539decbcc9f848dc245a93ed8cd7338bafef72f1/pillow-12.3.0-cp315-cp315-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:4998562bf62a445225f22e07c896bb04b35b1b1f2eb6d760584c9c51d7a5f78c", size = 3696331, upload-time = "2026-07-01T11:55:41.98Z" }, + { url = "https://files.pythonhosted.org/packages/12/40/d306fc2c8e4d45d7f175c77edca7063be7b86fe7fe6e68f4353bf71d808c/pillow-12.3.0-cp315-cp315-macosx_10_15_x86_64.whl", hash = "sha256:dc624f6bc473dacdf7ef7eb8678d0d08edf15cd94fad6ae5c7d6cc67a4e4902f", size = 5350370, upload-time = "2026-07-01T11:55:44.028Z" }, + { url = "https://files.pythonhosted.org/packages/dd/44/668fb1437e8ce420f62d6106eb66e44a5971602a4d794615bdf79315d82d/pillow-12.3.0-cp315-cp315-macosx_11_0_arm64.whl", hash = "sha256:71d6097b330eea8fd15097780c8e89cb1a8ce7838669f48c5bacd6f663dd4701", size = 4780147, upload-time = "2026-07-01T11:55:46.073Z" }, + { url = "https://files.pythonhosted.org/packages/0c/08/93fa2e70e30a2d81547e481b6ee2bb9522117221fb1e0ce4b5df70967677/pillow-12.3.0-cp315-cp315-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:28ce87c5ab450a9dd970b52e5aca5fe63ed432d18a2eaddd1979a00a1ba24ace", size = 6273659, upload-time = "2026-07-01T11:55:48.264Z" }, + { url = "https://files.pythonhosted.org/packages/f8/6d/043e96ff814fc31a33077e4cba86082167db520c93632afdf2042febbb0c/pillow-12.3.0-cp315-cp315-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:6b02afb9b97f65fbca5f31db6a2a3ba21aa93030225f150fa3f249717e938fb4", size = 6947439, upload-time = "2026-07-01T11:55:50.503Z" }, + { url = "https://files.pythonhosted.org/packages/af/92/ba71d2ee2ac0edf3fa33bd9d5ee9ee080da70b1766f3ca3934f9938ddac9/pillow-12.3.0-cp315-cp315-musllinux_1_2_aarch64.whl", hash = "sha256:1182d52bc2d5e5d7d0949503aa7e36d12f42205dc287e4883f407b1988820d39", size = 6353577, upload-time = "2026-07-01T11:55:52.697Z" }, + { url = "https://files.pythonhosted.org/packages/0f/ce/e63064e2122923ff687c8ad792d0d736a7b3920a56a46982e81a7fdd25d6/pillow-12.3.0-cp315-cp315-musllinux_1_2_x86_64.whl", hash = "sha256:e795b7eb908249c4e43c7c99fac7c2c75dab0c43566e37db472a355f63693d71", size = 7060394, upload-time = "2026-07-01T11:55:55.149Z" }, + { url = "https://files.pythonhosted.org/packages/54/76/a09cc3ccc8d773a7283d34c38bec1708f9e3cc932093cbc4c5e71ac4060b/pillow-12.3.0-cp315-cp315-win32.whl", hash = "sha256:57b3d78c95ba9059768b10e28b813002261d3f3dfc55cc48b0c988f625175827", size = 6467375, upload-time = "2026-07-01T11:55:57.769Z" }, + { url = "https://files.pythonhosted.org/packages/3e/03/1846c49ba3b1d5550392a4bbd06d6fb4578e1cd91a803198b5c90f5f7d53/pillow-12.3.0-cp315-cp315-win_amd64.whl", hash = "sha256:fa4ecea169a355be7a3ade2c783e2ed12f0e40d2c5621cda8b3297faf7fbb9f5", size = 7237048, upload-time = "2026-07-01T11:55:59.975Z" }, + { url = "https://files.pythonhosted.org/packages/fb/bb/89f35dcc79610423f9f195504d7def7f0d1416a711541b42867e25fe3412/pillow-12.3.0-cp315-cp315-win_arm64.whl", hash = "sha256:877c3f311ff35410f690861c4409e7ccbf0cd2f878e50628a28e5a0bb689e658", size = 2566006, upload-time = "2026-07-01T11:56:02.143Z" }, + { url = "https://files.pythonhosted.org/packages/30/88/707027ba09942dfa2c28759b5c222d769290a41c6d20ea60ec250801941f/pillow-12.3.0-cp315-cp315t-macosx_10_15_x86_64.whl", hash = "sha256:e9871b1ffbfa9656b60aeee92ed5136a5742696006fa322b29ea3d8da0ecc9cf", size = 5352509, upload-time = "2026-07-01T11:56:04.2Z" }, + { url = "https://files.pythonhosted.org/packages/b0/6d/00352fa25332c2569cd387851f568cc5a4b75a9adbfb37ac4fbce4c02eec/pillow-12.3.0-cp315-cp315t-macosx_11_0_arm64.whl", hash = "sha256:53aa02d20d10c3d814d536aa4e5ac9b84ca0ff5a88377963b085ad6822f93e64", size = 4783167, upload-time = "2026-07-01T11:56:06.631Z" }, + { url = "https://files.pythonhosted.org/packages/13/4f/9e049dfa21af7c22427275720e2490267ba8138120add5c4c574deb69782/pillow-12.3.0-cp315-cp315t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:446c34dcc4324b084a53b705127dc15717b22c5e140ae0a3c38349d4efec071e", size = 6329237, upload-time = "2026-07-01T11:56:08.868Z" }, + { url = "https://files.pythonhosted.org/packages/36/16/cf6eeaae8d0fce8dd390a33437cf68c5d5bd73834a2bc6e2f14efda0ab45/pillow-12.3.0-cp315-cp315t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:cf1845d02ad822a369a49f2bb9345b1614744267682e7a03527dc3bf6eea1777", size = 6997047, upload-time = "2026-07-01T11:56:11.379Z" }, + { url = "https://files.pythonhosted.org/packages/1e/69/dbf769bdd55f48bf5733cac28edc6364ffaa072ec9ba336266e4fe66be55/pillow-12.3.0-cp315-cp315t-musllinux_1_2_aarch64.whl", hash = "sha256:186941b6aef820ad110fb01fb06eb925374dc3a21b17e37ec9a53b250c6fe2d1", size = 6400440, upload-time = "2026-07-01T11:56:13.908Z" }, + { url = "https://files.pythonhosted.org/packages/a0/e1/ffc9cfc2eea0d178da8018e18e959301ad9d6bc9f3edb7181e748a474b97/pillow-12.3.0-cp315-cp315t-musllinux_1_2_x86_64.whl", hash = "sha256:f13c32a3abd6079a66d9526e18dad9b6d280384d49d7c54040cd57b6424041d9", size = 7105895, upload-time = "2026-07-01T11:56:16.575Z" }, + { url = "https://files.pythonhosted.org/packages/18/f0/a5595c1e8c3ae44b9828cb2f0fa8155e5095ef04d6327b8f61cf44a3df85/pillow-12.3.0-cp315-cp315t-win32.whl", hash = "sha256:1657923d2d45afb66526e5b933e5b3052e6bdea196c90d3abb2424e18c77dae8", size = 6474384, upload-time = "2026-07-01T11:56:18.855Z" }, + { url = "https://files.pythonhosted.org/packages/e4/04/62bcd9f844984c5938d3b05264a61d797a29d3e0812341a8204af70bbdee/pillow-12.3.0-cp315-cp315t-win_amd64.whl", hash = "sha256:8cd2f7bdda092d99c9fc2fb7391354f306d01443d22785d0cbfafa2e2c8bb418", size = 7243537, upload-time = "2026-07-01T11:56:21.214Z" }, + { url = "https://files.pythonhosted.org/packages/3d/68/1f3066acedf37673694a7141381d8f811ae97f30d34413d236abe7d489f1/pillow-12.3.0-cp315-cp315t-win_arm64.whl", hash = "sha256:06ff022112bc9cbf83b60f8e028d94ad87b60621706487e65f673de61610ab59", size = 2567491, upload-time = "2026-07-01T11:56:23.506Z" }, +] + [[package]] name = "platformdirs" version = "4.11.5"