From ccd8a0f9ec3974c1f24e0b66ab401572d75649f6 Mon Sep 17 00:00:00 2001 From: briacSck Date: Sat, 20 Jun 2026 12:52:12 +0200 Subject: [PATCH] docs: rework repo for the end user (web UI) and trim developer docs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Prepares the handoff to a non-technical macOS user who runs only the web UI. User-facing: - Rewrite README.md as a plain-language, web-UI, macOS walkthrough: one-time setup (install uv, make a Mistral key, fill .env), each grading run (start script → browser → upload zip → download), and privacy cleanup when done. - Rewrite docs/data-policy.md in plain terms (what stays local, what is sent, check your Mistral retention setting, erase when done). - Ship a ready-to-use config.yaml (un-ignored; holds no secrets) with the validated masking regexes, auto_orient, and min_native_dpi=140, so she never edits YAML; course is set per-batch in the upload form. Foolproof setup: - Load .env in-app via python-dotenv (web app + CLI), so the API key and the web login come from a file instead of shell env vars. Robust to special characters like '&' in the password (verified: .env-only login returns 200). - Add .env.example, and double-click macOS scripts start.command / cleanup.command (executable bit + .gitattributes eol=lf so they run on a Mac). Cleanup: - Consolidate ARCHITECTURE + OPERATIONS + CHANGELOG into one concise MAINTAINERS.md. - Delete deployment docs (docs/deploy.md, Dockerfile), the 586-line PRD (ocr-grading-tool-plan.md), and fold docs/runbook.md + docs/mistral-setup.md into the README. All recoverable from git history. Tests: 59 passing (unchanged). No student data or trial artifacts are tracked. Co-Authored-By: Claude Opus 4.8 --- .env.example | 12 + .gitattributes | 3 + .gitignore | 1 - ARCHITECTURE.md | 107 ------- CHANGELOG.md | 84 ------ Dockerfile | 27 -- MAINTAINERS.md | 115 ++++++++ OPERATIONS.md | 108 -------- README.md | 172 ++++++++---- cleanup.command | 13 + config.yaml | 51 ++++ docs/data-policy.md | 98 +++---- docs/deploy.md | 96 ------- docs/mistral-setup.md | 61 ---- docs/runbook.md | 78 ------ ocr-grading-tool-plan.md | 586 --------------------------------------- pyproject.toml | 1 + src/ocr_grade/cli.py | 4 + src/ocr_grade/web/app.py | 5 + start.command | 26 ++ uv.lock | 2 + 21 files changed, 397 insertions(+), 1253 deletions(-) create mode 100644 .env.example create mode 100644 .gitattributes delete mode 100644 ARCHITECTURE.md delete mode 100644 CHANGELOG.md delete mode 100644 Dockerfile create mode 100644 MAINTAINERS.md delete mode 100644 OPERATIONS.md create mode 100755 cleanup.command create mode 100644 config.yaml delete mode 100644 docs/deploy.md delete mode 100644 docs/mistral-setup.md delete mode 100644 docs/runbook.md delete mode 100644 ocr-grading-tool-plan.md create mode 100755 start.command diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..6a9246f --- /dev/null +++ b/.env.example @@ -0,0 +1,12 @@ +# Copy this file to a new file named ".env" (same folder) and fill in the three +# values below. The app reads them automatically when it starts. Keep .env +# private — it is never shared or uploaded, and is ignored by git. + +# Your Mistral API key. Create one (free to start) at https://console.mistral.ai/ +# under "API Keys". Paste it after the = sign, no quotes, no spaces. +MISTRAL_API_KEY= + +# The login for the web page. Pick any username and a strong password — you'll +# type these in the browser each time you open the app. +OCR_GRADE_WEB_USER=crystal +OCR_GRADE_WEB_PASSWORD=change-me-to-something-strong diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..02436eb --- /dev/null +++ b/.gitattributes @@ -0,0 +1,3 @@ +# macOS shell scripts must keep Unix (LF) line endings or they won't run. +*.command text eol=lf +*.sh text eol=lf diff --git a/.gitignore b/.gitignore index db17491..67344df 100644 --- a/.gitignore +++ b/.gitignore @@ -10,7 +10,6 @@ __pycache__/ .pytest_cache/ .cache/ web-work/ -config.yaml scans/ out/ *.pdf diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md deleted file mode 100644 index e8ad784..0000000 --- a/ARCHITECTURE.md +++ /dev/null @@ -1,107 +0,0 @@ -# Architecture - -`pipeline.run_batch` orchestrates the stages below sequentially (one page at a -time — no thread pool); `cli.py` is a thin Rich UI on top. The OCR backend and -the header-OCR seam are dependency-injected, so the whole flow runs offline in -tests. - -``` - scans/*.pdf - │ - ▼ - ┌──────────────┐ discover + validate (sha256, page/size limits) → manifest.json - │ Ingestion │ - └──────┬───────┘ - ▼ - ┌──────────────┐ rasterize (PyMuPDF) → page PNG; deskew / denoise / contrast - │ Preprocess │ - └──────┬───────┘ - ▼ - ┌──────────────┐ black out header_box + regex hits; identity → LOCAL sidecar - │ Redaction │ (a cheap header-only OCR pass finds the identity to mask) - └──────┬───────┘ - │ masked page PNG ─────────────────────────────────────┐ - ▼ │ original - ┌──────────────┐ content-addressed cache → Mistral OCR │ (unmasked) - │ Mistral OCR │ → markdown (never re-billed for same bytes) │ scan page - │ + Cache │ │ - └──────┬───────┘ │ - ▼ │ - ┌──────────────┐ normalize markdown; split printed prompt │ - │ Postprocess │ from handwritten answer │ - └──────┬───────┘ │ - ▼ ▼ - ┌────────────────────────────────────────────────────────────────┐ - │ PDF Assembler — interleave scan + transcript (markdown-it → │ - │ HTML → fitz.Story), compress (pikepdf), 95 MB size guard │ - └───────────────────────────────┬────────────────────────────────┘ - ▼ - out/{course}_{exam}_{sid}.pdf + out/run_report.md -``` - -## Module responsibilities - -| Module | Responsibility | -| --- | --- | -| `cli.py` | Typer entrypoint (`run`, `dry-run`, `purge`, `version`); loads `Settings`, drives the Rich progress bar + live cost. No orchestration logic. | -| `pipeline.py` | The orchestrator: `run_batch`, `estimate` (dry-run), `purge_batch`. Wires every stage; dependency-injectable backend + header OCR. | -| `config.py` | Typed `Settings` (pydantic-settings) from `config.yaml` + `OCR_GRADE__*` env overrides; `MISTRAL_API_KEY` read only from its own env var. | -| `ingestion.py` | Discover + validate input PDFs (corrupt, low native DPI, Mistral page/size limits); write `manifest.json`. | -| `preprocess.py` | Rasterize pages to PNG (PyMuPDF, no Poppler) + toggleable deskew/denoise/contrast. | -| `redaction.py` | Local identity masking (header box + regex), via a cheap stubbable `HeaderOCR` seam; writes a local-only identity sidecar. | -| `ocr/base.py` | `OCRBackend` Protocol + `OCRResult`/`OCRBlock`/`PageMeta` types. | -| `ocr/mistral.py` | The only concrete backend: masked page → Mistral `/v1/ocr`, retries 429/5xx, prices each call. | -| `ocr/cache.py` | Content-addressed OCR cache keyed by masked-image bytes + backend + fingerprint; accumulates cost/hit/miss. | -| `postprocess.py` | Normalize markdown; split printed prompt from handwritten answer via token line-maps. | -| `pdf_assembler.py` | Interleave original scan + transcript page (markdown-it → HTML → `fitz.Story`), compress, enforce the 95 MB ceiling. | -| `reporting.py` | Write `out/run_report.md` (model, pages, failures, cost, wall time). Called by `pipeline.run_batch`. | -| `web/` | Optional single-user FastAPI UI (`app`, `batches`, `settings`): authenticated zip upload → `pipeline.run_batch` in a background task → status page + downloads. Thin wrapper; no pipeline logic. | -| `utils.py` | Small shared helpers. | - -## Repo layout - -``` -ocr-grade/ -├── pyproject.toml -├── README.md / ARCHITECTURE.md / OPERATIONS.md -├── config.example.yaml -├── src/ocr_grade/ -│ ├── cli.py # typer entrypoint: run, dry-run, purge, version -│ ├── pipeline.py # batch orchestration (run_batch / estimate / purge_batch) -│ ├── config.py # pydantic-settings: yaml + env config -│ ├── ingestion.py # discover + validate input PDFs -│ ├── preprocess.py # rasterize (PyMuPDF) + deskew/denoise/contrast -│ ├── redaction.py # local header masking + identity sidecar -│ ├── ocr/ -│ │ ├── base.py # OCRBackend Protocol + OCRResult types -│ │ ├── mistral.py # the only concrete OCR backend -│ │ └── cache.py # content-addressed OCR result cache -│ ├── postprocess.py # markdown cleanup + prompt/answer split -│ ├── pdf_assembler.py # interleave scan+transcript, compress, size-guard -│ ├── reporting.py # out/run_report.md writer -│ ├── web/ # optional single-user FastAPI upload UI -│ │ ├── app.py # routes + HTTP Basic auth -│ │ ├── batches.py # in-memory batch registry + background job -│ │ └── settings.py # WebSettings (env-only) -│ └── utils.py -├── tests/ -│ ├── fixtures/synthetic/ # synthetic only — no real student data, ever -│ └── test_*.py -└── docs/ - ├── mistral-setup.md - ├── data-policy.md - └── runbook.md -``` - -## Extension points - -- **Swap in a second OCR backend:** implement the `ocr.base.OCRBackend` Protocol - (`name`, a read-only `cache_fingerprint` property, and `transcribe(image_path, - page_meta) -> OCRResult`) and return it from `pipeline._build_backend`. Nothing - in ingestion/preprocess/redaction/postprocess/pdf_assembler needs to change — - the cache keys on `cache_fingerprint`, so a new backend's results stay - separate. -- **New masking rule:** add a regex to `redaction.regex_patterns`, or set a fixed - `redaction.header_box` per course preset — no code change for the common case. - For richer logic (e.g. a learned detector), extend `redaction.mask` and/or the - `HeaderOCR` seam. diff --git a/CHANGELOG.md b/CHANGELOG.md deleted file mode 100644 index 4de1279..0000000 --- a/CHANGELOG.md +++ /dev/null @@ -1,84 +0,0 @@ -# Changelog - -All notable changes to this project will be documented in this file. - -The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/). - -## [Unreleased] - -### Added -- Single-user web UI (`ocr_grade.web`, FastAPI): an HTTP Basic-authenticated - upload page where `POST /batches` accepts a folder zip of scanned exam PDFs, - runs the same `pipeline.run_batch` in a background task, and exposes a - per-batch status page plus download links (per-PDF and a "download all" zip) - for the interleaved transcripts. Reuses the same `MISTRAL_API_KEY`; batch state - is in-memory and nothing persists beyond the per-batch working directory (no - database, no multi-tenant features). Added a `Dockerfile` and `docs/deploy.md` - (Render / Fly.io / VPS, and why serverless platforms like Vercel don't fit). - New deps: `fastapi`, `uvicorn[standard]`, `python-multipart`. -- Project skeleton bootstrapped: `src/ocr_grade/` package layout, Typer CLI - (`run`, `dry-run`, `purge`, `version` — all no-ops except `version`), ruff + - mypy config, pre-commit hooks (ruff, real-data/secret guards), CI workflow, - doc stubs. -- Phase 0 Mistral OCR sample sanity-check script (`scripts/sample_sanity_check.py`), - validated against 80 anonymized sample pages. -- Typed `Settings` model (`ocr_grade.config`) loaded from `config.yaml` with - env var overrides (`OCR_GRADE____`); `MISTRAL_API_KEY` - is always read from its own plain env var and never from yaml. -- Ingestion + preprocess pipeline front-end: `ingestion.discover()` validates - input PDFs (corrupt, low native DPI, Mistral page/size limits) and writes a - `manifest.json`; `preprocess.rasterize()` renders pages to cached PNGs via - PyMuPDF (no Poppler needed) and `preprocess.clean()` does toggleable - deskew/denoise/CLAHE-contrast. Replaced `pdf2image` with `pymupdf`. -- Local identity masking (`redaction.mask()`): blacks out a configured header - box and, via a cheap header-only OCR pass (the stubbable `HeaderOCR` seam), - masks the header band when configured identity regexes match. Extracted - identity strings and masked regions are written to a LOCAL-ONLY sidecar under - the cache dir and never re-sent to Mistral. Added `truststore` dependency. -- OCR backend seam + cache (`ocr.base`, `ocr.cache`): `OCRBackend` Protocol with - `OCRResult`/`OCRBlock`/`PageMeta` types, and a content-addressed `OCRCache` - keyed by image bytes + backend name + model/params fingerprint that stores - results as JSON and accumulates per-batch cost/hit/miss stats so identical - pages are never re-billed. -- Mistral OCR backend (`ocr.mistral.MistralOCRBackend`): sends a masked page - inline as a base64 `image_url` data URI, parses the returned markdown into - heading/paragraph/list blocks (markdown-it-py), retries 429/5xx with tenacity - (max 5 attempts, honoring `Retry-After`), and prices each call from - `mistral_price_per_page`. Filled in `docs/mistral-setup.md` and - `docs/data-policy.md`. -- Markdown post-processing (`postprocess`): `clean_text()` normalizes whitespace - and blank runs while preserving markdown structure, and - `split_prompt_and_answer()` re-parses the cleaned markdown (markdown-it-py) to - split each page into the printed prompt (a leading h1/h2/h3 heading or a - printed-looking paragraph) and the handwritten answer, slicing losslessly via - token line-maps. -- Final PDF assembler (`pdf_assembler.build_interleaved`): interleaves each - original (unmasked) scan page with a transcript page rendered from the cleaned - markdown (markdown-it-py → HTML → `fitz.Story`), names the file - `{course}_{exam}_{student_id}.pdf` from the real student ID in the identity - sidecar, and enforces a 95 MB ceiling with a pikepdf compression pass that - falls back to 150-DPI scan downsampling and then a part-1/part-2 split. -- End-to-end pipeline + CLI (`pipeline`, `cli`): `run` wires - ingestion → preprocess → mask → Mistral OCR (cached) → postprocess → assembler - sequentially per page with a Rich progress bar and a live cost total, writing - one interleaved PDF per exam plus `out/run_report.md` (pages, failures, total - cost, wall time, model). `dry-run` transcribes page 1 of the first exam and - projects batch cost/time; `purge --batch ` deletes an exam's cached OCR - entries and intermediate artifacts. The OCR backend and header-OCR seam are - dependency-injected so the whole flow is tested offline. - -### Documentation -- Handoff documentation pass: rewrote `README.md` (30-second quickstart, sample - command, env-var table), `OPERATIONS.md` (install/configure/run/troubleshoot/ - purge/rotate-key), and `ARCHITECTURE.md` (refreshed diagram, per-module - responsibility table, extension points incl. `pipeline.py`); added - `docs/runbook.md` with the exact per-grading-cycle checklist. - -### Changed -- Moved the `run_report.md` writer into `reporting.py` (`write_run_report`) so the - module matches its documented responsibility; `pipeline.run_batch` imports it. -- `OCRBackend.cache_fingerprint` is now a read-only Protocol property (it is a - computed `@property` on the Mistral backend). -- Render transcript pages with PyMuPDF's self-contained `fitz.Story` engine - instead of WeasyPrint, which requires GTK/Pango/cairo native libraries that - are not installable on Windows. Dropped the `weasyprint` dependency. diff --git a/Dockerfile b/Dockerfile deleted file mode 100644 index 6df7a05..0000000 --- a/Dockerfile +++ /dev/null @@ -1,27 +0,0 @@ -# Container image for the single-user web UI (ocr_grade.web.app:app). -# Render and Fly.io build this for you — see docs/deploy.md. You do not need to -# run Docker locally to deploy. -FROM python:3.11-slim - -# OpenCV (opencv-python) needs these system libraries at runtime; PyMuPDF does -# not need anything extra (it bundles its own renderer — no Poppler/GTK). -RUN apt-get update \ - && apt-get install -y --no-install-recommends libgl1 libglib2.0-0 \ - && rm -rf /var/lib/apt/lists/* - -# uv for fast, locked installs. -COPY --from=ghcr.io/astral-sh/uv:latest /uv /usr/local/bin/uv - -WORKDIR /app - -# Install dependencies from the lockfile first (better layer caching). -COPY pyproject.toml uv.lock README.md ./ -COPY src ./src -RUN uv sync --frozen --no-dev - -ENV PATH="/app/.venv/bin:$PATH" - -# config.yaml is NOT baked in — provide it at deploy time (mount or commit your -# own), along with MISTRAL_API_KEY / OCR_GRADE_WEB_USER / OCR_GRADE_WEB_PASSWORD. -EXPOSE 8000 -CMD ["uvicorn", "ocr_grade.web.app:app", "--host", "0.0.0.0", "--port", "8000"] diff --git a/MAINTAINERS.md b/MAINTAINERS.md new file mode 100644 index 0000000..3a79caf --- /dev/null +++ b/MAINTAINERS.md @@ -0,0 +1,115 @@ +# Maintainer notes + +Developer/architecture reference for `ocr-grade`. End users never need this file — +they use `README.md`. Keep this concise; it replaces the old ARCHITECTURE / +OPERATIONS / CHANGELOG docs. + +## What it is + +A local-first tool that turns scanned handwritten exam PDFs into Gradescope-ready +**interleaved** PDFs: each scan page followed by its Mistral-OCR transcription +(printed prompt split from handwritten answer). Student identity is masked +**locally** before any page reaches Mistral. There are two front ends over the +same pipeline: a Typer **CLI** and a single-user **web UI** (FastAPI). + +## Architecture + +`pipeline.run_batch` runs the stages below sequentially, one page at a time. The +OCR backend and the header-OCR seam are dependency-injected, so the whole flow +runs offline in tests. + +``` +scans/*.pdf + → Ingestion discover + validate (sha256, DPI floor, Mistral page/size limits) + → Preprocess rasterize (PyMuPDF) → PNG; auto-orient; deskew/denoise/contrast + → Redaction detect rotation, black out the identity header band; + identity strings → LOCAL-only sidecar (never re-sent) + → Mistral OCR masked page → /v1/ocr → markdown (content-addressed cache) + → Postprocess normalize markdown; split printed prompt vs handwritten answer + → PDF Assembler interleave original scan + transcript; compress; 95 MB guard + → out/{course}_{exam}_{sid}.pdf + out/run_report.md +``` + +| Module | Responsibility | +| --- | --- | +| `cli.py` | Typer entrypoint (`run`, `dry-run`, `purge`, `version`); Rich progress + live cost. Loads `.env`. | +| `pipeline.py` | Orchestrator: `run_batch`, `estimate`, `purge_batch`. Injectable backend + header OCR. | +| `config.py` | Typed `Settings` (pydantic-settings) from `config.yaml` + `OCR_GRADE__*` env. `MISTRAL_API_KEY` read from its own env var. Includes `min_native_dpi` and `redaction.auto_orient`. | +| `ingestion.py` | Discover + validate PDFs (corrupt, low DPI, Mistral limits); write `manifest.json`. | +| `preprocess.py` | Rasterize (PyMuPDF, no Poppler); `rotate_image` + `text_is_horizontal` helpers; deskew/denoise/contrast. | +| `redaction.py` | `detect_orientation` (geometric axis + identity-OCR flip) and `mask` (header-band masking + local identity sidecar) behind the stubbable `HeaderOCR` seam. | +| `ocr/base.py`,`ocr/mistral.py`,`ocr/cache.py` | Backend Protocol + types; the Mistral `/v1/ocr` backend (retries 429/5xx); content-addressed cache keyed on masked-image bytes + model fingerprint. | +| `postprocess.py` | Markdown cleanup; split printed prompt from handwritten answer. | +| `pdf_assembler.py` | Interleave scan + transcript (`markdown-it` → HTML → `fitz.Story`), pikepdf compress, 95 MB size guard, `scan_rotation` baked into embedded scans. | +| `reporting.py` | Write `out/run_report.md`. | +| `web/` | Single-user FastAPI UI (`app`, `batches`, `settings`): authenticated zip upload → `run_batch` in a background task → status page + downloads. Loads `.env`. Thin wrapper, no pipeline logic. | + +All text I/O is pinned to `encoding="utf-8"` (Windows defaults to cp1252 and +crashes on math glyphs otherwise). + +## Develop + +```bash +uv sync --dev # Python 3.11 pinned via .python-version; uv fetches it +uv run pytest # full suite (one test makes a real Mistral call if a key is set) +uv run pytest -k "not integration" # offline suite +uv run ruff check src tests +``` + +Behind an intercepting corporate proxy, prefix network `uv` commands with +`UV_SYSTEM_CERTS=true`. + +## Run + +```bash +# CLI +export MISTRAL_API_KEY=... # or put it in .env +uv run ocr-grade dry-run --config config.yaml # OCR page 1, estimate cost/time +uv run ocr-grade run --config config.yaml # full batch → out/ +uv run ocr-grade purge --batch --config config.yaml + +# Web UI (what ships to the end user; .env supplies key + login) +uv run uvicorn ocr_grade.web.app:app --port 8000 +``` + +Config reference: every option is documented inline in `config.example.yaml`. +The shipped `config.yaml` is a ready-to-use copy (course set per-batch in the UI). + +## Troubleshoot + +| Symptom | Cause / fix | +| --- | --- | +| `No valid exams found` | `input_dir` wrong/empty, or all PDFs rejected (corrupt, below `min_native_dpi`, or over Mistral's 50 MB / 1000-page limits). | +| `invalid peer certificate: UnknownIssuer` during `uv …` | Intercepting corporate CA — re-run with `UV_SYSTEM_CERTS=true`. | +| 401 from Mistral | `MISTRAL_API_KEY` not set in the env / `.env`, or revoked. | +| Identity leaked into a transcript | Masking config doesn't match the template, or `auto_orient` mis-detected. Fix `redaction.*`, **purge** the batch, re-run. | +| Re-running re-bills done pages | Expected only if the model or masking changed (new cache key). | + +## OCR quality + +Transcription accuracy is bounded by handwriting legibility and scan resolution, +**not** by Mistral plan tier (tiers gate rate/quota, not model quality). Biggest +lever: scan at ~300 DPI. The interleaved original scan is intentionally kept next +to the transcript so the grader cross-checks rather than trusting OCR blindly. + +## Privacy invariant + +Only the **masked** page (identity blacked out) and one cheap header crop reach +Mistral. Extracted identity strings live only in the gitignored cache sidecars. +Never commit student data; `tests/fixtures/` is synthetic only. See +`docs/data-policy.md`. + +## Extension points + +- **Second OCR backend:** implement `ocr.base.OCRBackend` (`name`, + `cache_fingerprint`, `transcribe`) and return it from `pipeline._build_backend`. + Cache keys on the fingerprint, so results stay separate. No other stage changes. +- **New masking rule:** add a regex to `redaction.regex_patterns` or set a fixed + `redaction.header_box`; extend `redaction.mask` / the `HeaderOCR` seam for richer logic. + +## Version notes + +- **v1 (current):** full pipeline (ingestion → preprocess → redaction → Mistral OCR + + cache → postprocess → interleaved assembler), Typer CLI, single-user web UI. + Pre-ship hardening: configurable `min_native_dpi`, UTF-8 text I/O, dynamic page + auto-orientation before masking, `.env`-based config + macOS launch scripts. diff --git a/OPERATIONS.md b/OPERATIONS.md deleted file mode 100644 index 7074280..0000000 --- a/OPERATIONS.md +++ /dev/null @@ -1,108 +0,0 @@ -# Operations - -How a second operator installs, configures, runs, and maintains `ocr-grade`. -For the exact per-grading-cycle checklist, see `docs/runbook.md`. - -## Install - -```bash -uv sync --dev # Python 3.11 is pinned via .python-version; uv fetches it -``` - -No system packages are required. Rasterization and PDF layout both use -**PyMuPDF**, which bundles its own renderer — there is **no Poppler, GTK, Pango, -or cairo dependency** to install. (An earlier design used WeasyPrint for the -transcript pages; it was dropped precisely because those native libs are not -installable on Windows.) - -**Behind a corporate proxy that intercepts HTTPS:** prefix network `uv` commands -with `UV_SYSTEM_CERTS=true` (or export it once), so `uv` trusts the OS cert -store: - -```bash -UV_SYSTEM_CERTS=true uv sync --dev -``` - -## Configure - -1. Copy the example and edit it: - - ```bash - cp config.example.yaml config.yaml - ``` - -2. Set the key paths and the course preset: - - `input_dir` — folder of scanned exam PDFs (one PDF per student). - - `output_dir` — where interleaved PDFs + `run_report.md` are written. - - `cache_dir` — local scratch (originals, page PNGs, masked pages, identity - sidecars, OCR cache). Gitignored; keep it on controlled storage. - - `course_preset` — used in output filenames (e.g. `PE101`). - - `redaction.header_box` / `redaction.regex_patterns` — what gets masked - locally before OCR. **Tune these for your exam template and spot-check.** - - `mistral.model` — pin a dated revision for reproducible runs - (see `docs/mistral-setup.md`). - -3. Provide the API key via the environment (never in `config.yaml`): - - ```bash - export MISTRAL_API_KEY="…" - ``` - - See `docs/mistral-setup.md` for account + key creation and rotation. - -Any field can be overridden per run with an env var -`OCR_GRADE____` or with `run` flags (`--input`, `--output`, -`--course`). - -## Run a batch - -Always estimate first: - -```bash -uv run ocr-grade dry-run --config config.yaml -``` - -This transcribes **page 1 of the first exam only** and prints total pages, -estimated cost, and projected wall time. If that looks right: - -```bash -uv run ocr-grade run --config config.yaml -``` - -A Rich progress bar shows pages done and a live running cost. On completion it -prints a summary and writes `output_dir/run_report.md` (model, pages processed, -failures table, total Mistral cost, wall time, output files). `run` exits with a -non-zero status if any page or exam failed, so it is safe to chain in scripts. - -## Troubleshoot - -| Symptom | Likely cause / fix | -| --- | --- | -| `No valid exams found in ` | `input_dir` wrong/empty, or every PDF was rejected (corrupt, too low native DPI, or over Mistral's page/size limits). Check the per-exam status; re-scan offending files. | -| `invalid peer certificate: UnknownIssuer` during `uv …` | Intercepting corporate CA — re-run with `UV_SYSTEM_CERTS=true`. | -| Auth/401 from Mistral | `MISTRAL_API_KEY` not set/exported in this shell, or revoked. Re-export; rotate if needed. | -| Identity not masked / leaked into transcript | `redaction.header_box` / `regex_patterns` don't match this template. Fix config, **purge the batch** (cached OCR was run on the unmasked-as-intended image), and re-run. | -| Output PDF unexpectedly large | Drop `dpi` (300 → 250). The assembler already enforces a 95 MB ceiling by compressing, then downsampling scans to 150 DPI, then splitting into `_part1`/`_part2`. | -| Re-running re-bills pages you already did | It shouldn't — the OCR cache is content-addressed on the masked image. If you changed the model or masking, that's a new cache key and a legitimate re-bill. | -| Failures listed in `run_report.md` | Each row names the exam/page and reason; the batch continues past failures so good exams still produce PDFs. Fix the cause and re-run (cached pages are reused). | - -## Purge - -To delete the cached OCR results and intermediate artifacts for one exam (e.g. -after fixing masking, or to remove sensitive scratch when grading is done): - -```bash -uv run ocr-grade purge --batch --config config.yaml -``` - -`` is the exam's sha256 (the subdirectory name under `cache_dir`). This -removes `cache_dir//` and the content-addressed `cache_dir/ocr/.json` -entries that those pages produced. It does **not** delete already-written output -PDFs. - -## Rotate API key - -1. Generate a new key in the Mistral console; export it as `MISTRAL_API_KEY`. -2. Revoke the old key in the console. -3. Do this on a quarterly cadence, and **immediately** if a key is ever leaked or - committed. Full procedure: `docs/mistral-setup.md`. diff --git a/README.md b/README.md index be9c8e2..53b1f98 100644 --- a/README.md +++ b/README.md @@ -1,70 +1,140 @@ -# ocr-grade +# Exam Transcriber -A local-first CLI that turns scanned handwritten exam PDFs into Gradescope-ready -interleaved PDFs (each scan page followed by its **Mistral OCR** transcription, -with the printed prompt and the handwritten answer split apart). +Turn scanned, handwritten exam PDFs into clean, typed transcripts you can read and +grade quickly. For each page you get the **original scan with a readable +transcription right next to it**, and each student's **name and ID are +automatically blacked out** before anything leaves your computer. -Student identity is masked **locally** before any page reaches Mistral; the -extracted names/SIDs stay in a gitignored cache and are never re-sent. See -`docs/data-policy.md`. +You run it on your own Mac through a simple web page in your browser. No coding. -## Quickstart (≈30 seconds) +--- -```bash -uv sync --dev # install (Python 3.11, pinned via .python-version) -export MISTRAL_API_KEY="…" # your key — see docs/mistral-setup.md -cp config.example.yaml config.yaml # then edit input/output/course/redaction -uv run ocr-grade dry-run --config config.yaml # OCR page 1 only; prints cost + time estimate -uv run ocr-grade run --config config.yaml # full batch → out/*.pdf + out/run_report.md +## What you need + +- A Mac. +- The **`ocr-grade` folder** on your computer (the one containing this README). +- A free **Mistral** account for the transcription service (you'll make one below). +- About **10 minutes**, once, for first-time setup. + +--- + +## One-time setup (do this once) + +### 1. Install the helper tool (`uv`) + +Open the **Terminal** app (press `Cmd`+`Space`, type "Terminal", press Return). +Copy the line below, paste it into Terminal, and press Return: + +``` +curl -LsSf https://astral.sh/uv/install.sh | sh ``` -> On a corporate network whose root CA intercepts HTTPS, prefix `uv` network -> commands with `UV_SYSTEM_CERTS=true` (e.g. `UV_SYSTEM_CERTS=true uv sync --dev`). +When it finishes, **close the Terminal window**. (This installs a small tool the +app needs. You only do this once.) + +### 2. Get your Mistral API key + +1. Go to **https://console.mistral.ai/** and create an account (it's free to start). +2. Click **API Keys**, then **Create new key**. +3. **Copy the key now** — Mistral only shows it once. Paste it somewhere safe for a moment. -## Sample command +### 3. Enter your key and choose a password -```bash -# Override config fields from the command line: -uv run ocr-grade run \ - --config config.yaml \ - --input ./scans \ - --output ./out \ - --course PE101 +In the `ocr-grade` folder there's a file called **`.env.example`**. We'll make your +own copy of it. In Terminal, paste these two lines one at a time: + +``` +cd ~/Downloads/ocr-grade # change this if the folder is somewhere else +cp .env.example .env && open -e .env ``` -Outputs land in `output_dir` as `{course}_{exam}_{student_id}.pdf` (real student -ID from the local identity sidecar), plus `run_report.md` (pages, failures, total -Mistral cost, wall time, model). +A text editor opens your new **`.env`** file with three lines. Fill them in: -## Commands +- `MISTRAL_API_KEY=` — paste the key you copied (right after the `=`, no spaces). +- `OCR_GRADE_WEB_USER=` — a username you'll use to log into the app (e.g. your name). +- `OCR_GRADE_WEB_PASSWORD=` — a password you choose for the app. -| Command | What it does | -| --- | --- | -| `ocr-grade run` | Full pipeline over every valid exam in `input_dir`. | -| `ocr-grade dry-run` | Transcribe page 1 of the first exam; print estimated batch cost + projected time. | -| `ocr-grade purge --batch ` | Delete cached OCR results + intermediate artifacts for one exam. | -| `ocr-grade version` | Print the installed version. | +**Save** (`Cmd`+`S`) and close the editor. That's it — setup is done. -## Environment variables +--- -| Variable | Purpose | -| --- | --- | -| `MISTRAL_API_KEY` | **Required.** API key, read only from the env (never `config.yaml`). | -| `OCR_GRADE____` | Override any config field, e.g. `OCR_GRADE__MISTRAL__MODEL=mistral-ocr-2512`, `OCR_GRADE__DPI=250`. | -| `UV_SYSTEM_CERTS=true` | Trust the OS cert store (needed behind an intercepting corporate proxy). | +## Each time you grade exams -## Web UI (optional) +### 1. Start the app -A single-user, password-protected upload page is also available for running -batches from a browser instead of the terminal: +Double-click **`start.command`** in the `ocr-grade` folder. -```bash -OCR_GRADE_WEB_USER=me OCR_GRADE_WEB_PASSWORD=secret \ - uv run uvicorn ocr_grade.web.app:app --port 8000 -``` +- A black Terminal window opens and stays open — **that's normal, leave it open.** +- Your browser opens the app a few seconds later. If it doesn't, go to + **http://localhost:8000** yourself. +- The first time, your Mac may say the file is from an "unidentified developer." + If so: **right-click `start.command` → Open → Open**. You only do that once. + +### 2. Log in + +Type the **username and password** you chose in your `.env` file. + +### 3. Prepare your scans as a `.zip` -It reuses the same `config.yaml` and `MISTRAL_API_KEY`. See `docs/deploy.md` for -hosting it (Render / Fly.io / a small VPS). +Put **one PDF per student** in a folder. In Finder, select all those PDFs, +right-click, and choose **Compress** — that makes a `.zip` file. -See `ARCHITECTURE.md` for how it fits together, `OPERATIONS.md` for running a real -batch, and `docs/runbook.md` for the per-cycle grading checklist. +### 4. Upload and run + +On the app page: + +1. (Optional) Type your **course code** in the "Course override" box (e.g. `PE101`). + It just labels the output files. +2. Click **Choose File** and pick your `.zip`. +3. Click **Start batch**. + +The page shows progress and a running cost (transcription is a few tenths of a cent +per page). When it's done, **download** each transcript, or **Download all** as a +single `.zip`. The files save to your **Downloads** folder. + +### 5. When you're finished — clean up + +For your students' privacy, erase the scans and transcripts from the app when you're +done: + +1. Make sure you've **downloaded** everything you want to keep. +2. **Close the black Terminal window** (this stops the app). +3. Double-click **`cleanup.command`**. It deletes the uploaded scans and generated + transcripts from the computer. Anything already in your Downloads folder is kept. + +--- + +## Privacy + +Each student's name and ID are blacked out on your Mac **before** any page is sent +for transcription, and the names/IDs themselves are never sent. Only the +blacked-out page image goes to Mistral for reading. For the full details — and a +note to check your Mistral account's data settings before grading real exams — see +[docs/data-policy.md](docs/data-policy.md). + +--- + +## Tips for the best transcripts + +- **Scan clearly.** Scanning at about **300 DPI** gives noticeably better results. + The current scans (~144 DPI) work, but higher resolution helps a lot. +- **Expect to glance at the original.** Messy handwriting, math symbols, and + diagrams are hard for any transcription tool, so it can occasionally guess wrong. + That's exactly why the original scan sits right next to the transcript — skim both. +- Paying for Mistral does **not** improve accuracy (it only raises how fast/how much + you can run). Clearer, higher-resolution scans are what improve the transcript. + +--- + +## If something goes wrong + +| What you see | What to do | +| --- | --- | +| Double-clicking `start.command` does nothing | Right-click it → **Open** → **Open**. If still nothing, open Terminal, type `cd ` then drag the folder in, press Return, then run `bash start.command`. | +| "Web auth is not configured" | Your `.env` is missing the username/password. Re-open `.env` (step 3 of setup) and fill them in. | +| Login is rejected | The username/password must match your `.env` exactly. | +| "Unauthorized" / 401 error during a batch | Your Mistral key is missing or wrong in `.env`. Re-check it at https://console.mistral.ai/. | +| A transcript looks wrong | Compare it to the scan next to it; for messy handwriting this is expected. Rescanning that exam at higher resolution usually helps. | + +Questions or a key that stopped working? Make a new key at +https://console.mistral.ai/ and paste it into your `.env`. diff --git a/cleanup.command b/cleanup.command new file mode 100755 index 0000000..d622c7c --- /dev/null +++ b/cleanup.command @@ -0,0 +1,13 @@ +#!/bin/bash +# Double-click this file when you have finished grading and downloaded your +# transcripts. It erases the scanned exams and generated transcripts from this +# computer (the privacy "clean up when done" step). It does NOT touch anything +# you already saved to your Downloads folder. + +cd "$(dirname "$0")" || exit 1 + +rm -rf web-work .cache out + +echo "Done — all uploaded scans and generated transcripts have been erased from this computer." +echo "Anything you already downloaded is untouched." +echo "You can close this window." diff --git a/config.yaml b/config.yaml new file mode 100644 index 0000000..fb52093 --- /dev/null +++ b/config.yaml @@ -0,0 +1,51 @@ +# ocr-grade configuration. +# +# This file is ready to use as-is — you normally do NOT need to edit it. +# Your Mistral API key and the web login live in the .env file instead, never here. +# +# When you run the web page, the course code is set per batch in the upload form, +# and the input/output/cache folders below are ignored (the web app uses its own +# per-batch working folder). Those folders only matter for the command-line tool. + +input_dir: ./scans +output_dir: ./out +cache_dir: ./.cache + +# How sharply pages are rendered before transcription. Higher is crisper but +# larger; 300 is a good default. +dpi: 300 + +# Reject scans whose detected resolution is below this. The current scanner +# exports ~144 DPI, so this is set a little under 150 to accept them. +min_native_dpi: 140 + +# Default label stamped on output filenames when you don't type a course in the +# upload form. +course_preset: EXAM + +redaction: + # Automatically detect each exam's page rotation and turn it upright before + # blacking out identity, so the name/ID is found no matter how the scanner + # oriented the page. Leave this on. + auto_orient: true + # No fixed box — identity is found by the patterns below and the whole header + # band is masked when one matches. + header_box: null + regex_patterns: + - 'SID[:\s]*\d{7,10}' + - 'Name[:\s]*[A-Z][a-z]+ [A-Z][a-z]+' + +mistral: + # The API key is read from the .env file (MISTRAL_API_KEY), never from here. + model: mistral-ocr-latest + base_url: null + timeout_s: 60 + include_image_base64: false + +# Used only to estimate cost in the report; does not cap spending. +mistral_price_per_page: 0.001 + +preprocess_steps: + deskew: true + denoise: true + contrast: true diff --git a/docs/data-policy.md b/docs/data-policy.md index 180cacd..70b0862 100644 --- a/docs/data-policy.md +++ b/docs/data-policy.md @@ -1,54 +1,44 @@ -# Data policy - -This tool processes scanned **student exams**, which are sensitive education -records. This page describes what leaves your machine and what does not. - -## What stays local (never sent to Mistral) - -- **Original scanned PDFs** and the full-resolution rasterized page PNGs under - the cache directory. -- **Extracted student identity** (names, SIDs). The redaction pass - (`ocr_grade.redaction`) writes matched identity strings and masked regions to a - local-only sidecar (`.identity.json`) under the gitignored cache dir. - These strings are **never** sent back to Mistral. - -## What is sent to Mistral - -- **The masked page image only.** Before any full page is transcribed, the - header band (configured `header_box` and/or any region where an identity regex - matched) is blacked out with a solid rectangle. The masked PNG is sent inline - as a base64 `image_url` data URI to the `/v1/ocr` endpoint — no upload to - Mistral's `/files` store and no public URL hosting. -- **One cheap header-only crop**, once per page, for the identity-detection OCR - pass. This crop is what lets us find and mask the identity; it is sent only for - that read and the resulting identity strings are kept local (see above). - -The masked page → `OCRResult` (markdown + heading/paragraph/list blocks) is the -only transcription output; results are cached locally and keyed by image content -so the same page is never re-sent or re-billed. - -## Mistral's handling of submitted data - -Data you send to the Mistral API is governed by Mistral's current terms and -privacy policy, and by your account's data-processing settings: - -- Privacy Policy: -- Terms of Use / Data Processing: -- OCR API docs: - -Retention windows and whether submitted content may be used to improve models -depend on your plan and account configuration, and these terms change over time. -**Before processing real student data, verify the current policy and confirm your -account's training/retention settings** (e.g. opt out of training-on-your-data if -your plan offers it, and prefer a plan with a zero/short retention commitment). -When in doubt, treat anything sent to the API as leaving your custody — which is -exactly why identity is masked locally first. - -## Operator checklist - -- Confirm masking config (`redaction.header_box`, `redaction.regex_patterns`) is - correct for the exam template **before** a batch run; spot-check masked pages. -- Keep the cache dir (originals, PNGs, identity sidecars) on local/controlled - storage; it is gitignored and must not be committed or synced to shared drives. -- Delete cache artifacts when a grading run is complete and they are no longer - needed. +# Privacy & your students' data + +This tool handles scanned **student exams**, which are sensitive. Here's exactly +what stays on your Mac and what gets sent out, in plain terms. + +## What never leaves your computer + +- **The original scans** and the full-resolution page images. +- **Student names and IDs.** Before any page is transcribed, the app finds the + name/ID area and **blacks it out**. The actual names and IDs are saved only in a + private file on your Mac (used to name the output files) and are **never sent + anywhere**. + +## What gets sent for transcription + +- **Only the blacked-out page image.** The version of the page that goes to the + Mistral transcription service has the identity already covered with a solid black + box. It's sent directly to the service for reading — not uploaded to any public + link or shared store. +- **One small slice of the top of the page**, briefly, so the app can locate the + name/ID in order to black it out. That slice is used only for that check. + +That's it. The typed transcript that comes back is the only result, and it's kept +on your Mac. + +## One thing to check before grading real exams + +The transcription is done by **Mistral** (https://mistral.ai). Whether they keep +submitted images, and for how long, depends on your Mistral account settings and +their current terms, which can change. Before processing real student work: + +- Review Mistral's privacy terms: https://mistral.ai/terms/#privacy-policy +- In your Mistral account, prefer settings with **no training on your data** and the + **shortest retention** your plan offers. + +When in doubt, treat anything sent to the service as leaving your custody — which is +why identity is blacked out on your computer first. + +## Clean up when you're done + +When you've finished grading and downloaded your transcripts, double-click +**`cleanup.command`** (see the README). It erases the uploaded scans and generated +transcripts from the computer. Anything you've already saved to your Downloads +folder is untouched. diff --git a/docs/deploy.md b/docs/deploy.md deleted file mode 100644 index ff6af3b..0000000 --- a/docs/deploy.md +++ /dev/null @@ -1,96 +0,0 @@ -# Deploying the web UI - -The web UI (`ocr_grade.web.app:app`) is a small **single-user** FastAPI server: -upload a zip of scanned exam PDFs, it runs the same pipeline as the CLI in a -background task, then offers the interleaved PDFs for download. State lives in -memory + a working directory; there is no database. - -## Why a persistent server (not Vercel / serverless) - -This app is **stateful and long-running**, so serverless platforms (Vercel, -plain Lambda, etc.) are the wrong fit: - -- An OCR batch runs for **minutes** — past serverless function time limits. -- Batch status is held **in one long-lived process**; serverless instances are - cold/ephemeral and don't share that state across requests. -- It writes scans + output PDFs to a **working directory**; serverless gives only - an ephemeral `/tmp` that's wiped between invocations (download links would 404). -- PyMuPDF + OpenCV are heavy native deps that strain serverless bundle limits. - -Deploy it as an always-on container instead. The `Dockerfile` at the repo root is -the package format Render and Fly build for you — **you don't run Docker -yourself**. - -## Prerequisites (all targets) - -1. **A real `config.yaml`** — course preset, redaction/masking config, and - `mistral.model`. Masking is read from this file and is never guessed, so get it - right for your exam template before going live. Either commit a `config.yaml` - or mount one at deploy time. -2. **Secrets / env vars:** - - `MISTRAL_API_KEY` — same key the CLI uses. - - `OCR_GRADE_WEB_USER`, `OCR_GRADE_WEB_PASSWORD` — the login. - - Optional: `OCR_GRADE_WEB_WORKDIR` (default `web-work`), - `OCR_GRADE_WEB_BASE_CONFIG` (default `config.yaml`), - `OCR_GRADE_WEB_MAX_UPLOAD_MB` (default `200`). - -> **Single user.** This is intentionally one login (Prof. Chang) — no TA or -> multi-user accounts. Keep the `OCR_GRADE_WEB_USER` / `OCR_GRADE_WEB_PASSWORD` -> private; treat the password like the API key. - -> **Always serve over HTTPS.** HTTP Basic credentials are sent on every request. -> Render and Fly terminate TLS for you; on a VPS put Caddy or nginx in front. - -## Render (recommended) - -Easiest path for a non-DevOps owner; gives an HTTPS URL and a secrets UI. - -1. Push this repo to GitHub. -2. Render → **New → Web Service** → connect the repo. -3. **Runtime: Docker** (Render auto-detects the `Dockerfile`). -4. **Environment** → add `MISTRAL_API_KEY`, `OCR_GRADE_WEB_USER`, - `OCR_GRADE_WEB_PASSWORD` (and provide a `config.yaml` — commit one, or add it as - a Render Secret File mounted at `/app/config.yaml`). -5. **Health check path:** `/healthz`. -6. Deploy. Render builds the image and serves it at an `https://…onrender.com` - URL. That URL + the single login is all Prof. Chang needs. - -Note: Render's filesystem is ephemeral — finished batches don't survive a -restart/redeploy. That matches the "no persistence" design; just re-download -before a redeploy. Add a Render Disk if you want batches to persist. - -## Fly.io - -1. `fly launch` in the repo (it detects the `Dockerfile`; decline to deploy yet). -2. Set secrets: - ```bash - fly secrets set MISTRAL_API_KEY=… OCR_GRADE_WEB_USER=… OCR_GRADE_WEB_PASSWORD=… - ``` -3. Provide `config.yaml` (commit it, or mount via a Fly volume). -4. Set the health check to `GET /healthz` in `fly.toml`, then `fly deploy`. - -Fly's machine filesystem is ephemeral too; attach a volume mounted at the -`OCR_GRADE_WEB_WORKDIR` path only if you want batches to outlive a restart. - -## Small VPS - -```bash -docker build -t ocr-grade-web . -docker run -d --restart unless-stopped -p 8000:8000 \ - -e MISTRAL_API_KEY=… \ - -e OCR_GRADE_WEB_USER=… \ - -e OCR_GRADE_WEB_PASSWORD=… \ - -v "$PWD/config.yaml:/app/config.yaml:ro" \ - ocr-grade-web -``` - -Put **Caddy or nginx** in front for TLS (a `caddy reverse-proxy --to :8000` with a -real domain is enough). Without HTTPS the Basic-auth login is exposed in transit. - -## Local smoke test - -```bash -OCR_GRADE_WEB_USER=me OCR_GRADE_WEB_PASSWORD=secret \ - uv run uvicorn ocr_grade.web.app:app --port 8000 -# GET /healthz -> 200; open http://localhost:8000/ and sign in. -``` diff --git a/docs/mistral-setup.md b/docs/mistral-setup.md deleted file mode 100644 index b29caa9..0000000 --- a/docs/mistral-setup.md +++ /dev/null @@ -1,61 +0,0 @@ -# Mistral setup - -`ocr-grade` transcribes pages through Mistral's OCR API. You need a Mistral -account and an API key. - -## 1. Create an account and API key - -1. Sign up at . -2. Open **API Keys** in the console and generate a new key. -3. Copy it once — Mistral only shows the secret at creation time. - -## 2. Provide the key via `MISTRAL_API_KEY` (never commit it) - -The key is read **only** from the `MISTRAL_API_KEY` environment variable — never -from `config.yaml`, and never hardcoded. Export it from your shell profile: - -```bash -# ~/.bashrc / ~/.zshrc -export MISTRAL_API_KEY="…" -``` - -```powershell -# PowerShell profile ($PROFILE) -$env:MISTRAL_API_KEY = "…" -``` - -For local development you may instead place it in a **gitignored** `.env` file at -the repo root (`.env` is already in `.gitignore`). Never paste the key into -tracked files, commit messages, or `config.yaml` — the pre-commit secret guard -will block obvious leaks, but treat the key as you would a password. - -## 3. Pin the model in `config.yaml` - -```yaml -mistral: - model: mistral-ocr-latest # alias -> mistral-ocr-2512 -``` - -`mistral-ocr-latest` is an alias that currently resolves to the dated revision -`mistral-ocr-2512`. The alias can shift under you when Mistral ships a new -revision. For reproducible grading runs, pin the **dated** model id and bump it -deliberately after you re-validate OCR quality on a sample: - -```yaml -mistral: - model: mistral-ocr-2512 -``` - -You can also override per run without editing the file: -`OCR_GRADE__MISTRAL__MODEL=mistral-ocr-2512`. - -## 4. Key hygiene - -- **Rotate** the key on a quarterly schedule. -- **Revoke immediately** in the console if a key is ever leaked or committed, and - issue a replacement. -- Use a dedicated key for this tool so it can be revoked without disrupting other - integrations. - -See also `docs/data-policy.md` for what leaves your machine and our masking -guarantee. diff --git a/docs/runbook.md b/docs/runbook.md deleted file mode 100644 index b802c64..0000000 --- a/docs/runbook.md +++ /dev/null @@ -1,78 +0,0 @@ -# Grading-cycle runbook - -The exact steps to run for each exam you grade. Assumes the one-time setup in -`OPERATIONS.md` (Install + Configure) and `docs/mistral-setup.md` is done and a -working `config.yaml` exists. Run every command from the repo root. - -## Before you start (once per machine / session) - -1. Confirm the key is set in this shell: - - ```bash - echo "$MISTRAL_API_KEY" # should print your key, not blank - ``` - - If blank: `export MISTRAL_API_KEY="…"` (PowerShell: `$env:MISTRAL_API_KEY = "…"`). -2. Confirm `config.yaml` points at the right `input_dir`, `output_dir`, and - `course_preset` for this exam. - -## Each grading cycle - -1. **Drop the scans in.** Put the scanned exam PDFs (one PDF per student) in the - `input_dir` from `config.yaml`. Nothing else should be in that folder. - -2. **Check the masking config matches this exam template.** Open `config.yaml` - and confirm `redaction.header_box` / `redaction.regex_patterns` cover where the - student name and SID appear on *this* exam. If the template changed since last - time, update them first. - -3. **Dry-run to estimate cost and time:** - - ```bash - uv run ocr-grade dry-run --config config.yaml - ``` - - Read the estimated cost and projected wall time. If they look wrong (e.g. far - more pages than expected), stop and check `input_dir`. - -4. **Run the batch:** - - ```bash - uv run ocr-grade run --config config.yaml - ``` - - Watch the progress bar and the live running cost. It writes one - `{course}_{exam}_{student_id}.pdf` per student into `output_dir`, plus - `run_report.md`. - -5. **Read `out/run_report.md`.** Confirm pages processed matches expectations and - the **Failures** table is empty. If there are failures, each row names the exam - and reason — fix the cause (see `OPERATIONS.md` → Troubleshoot) and re-run - (already-done pages are reused from cache, not re-billed). - -6. **Spot-check 2–3 output PDFs.** Open them and verify: - - the student name / SID is **blacked out** on every scan page (privacy), and - - the transcript text matches the handwriting and the prompt/answer split looks - right. - - If identity is **not** masked, do not distribute. Fix `redaction.*` in - `config.yaml`, purge the affected exam (step 8), and re-run. - -7. **Upload to Gradescope.** Use the interleaved PDFs from `output_dir`. - -8. **Clean up when done.** Once grading is complete and you no longer need the - scratch (originals, page PNGs, identity sidecars, OCR cache), purge each exam - by its sha (the subfolder name under `cache_dir`): - - ```bash - uv run ocr-grade purge --batch --config config.yaml - ``` - - This removes cached OCR + intermediate artifacts but **keeps** the finished - output PDFs. See `docs/data-policy.md` for what was stored where. - -## Quarterly - -- Rotate the Mistral API key (`docs/mistral-setup.md` → Key hygiene). -- Re-confirm Mistral's data/retention settings still match policy - (`docs/data-policy.md`). diff --git a/ocr-grading-tool-plan.md b/ocr-grading-tool-plan.md deleted file mode 100644 index 9577622..0000000 --- a/ocr-grading-tool-plan.md +++ /dev/null @@ -1,586 +0,0 @@ -# OCR Grading Tool — PRD, Development Plan & Claude Code Prompts - -**Project owner:** Briac -**Client / primary user:** Prof. Crystal Chang (UC Berkeley — PE101, P156, third smaller course) -**Document version:** v0.4 — June 2026 (minimal / Textract-only / single AWS account) -**Status:** Pre-development; awaiting ~15 anonymized sample scripts. - -**Scope sizing:** ~400 papers/year × ~20 pages = ~8,000 pages, run in ~3 batches/semester, 2–3 operators. This plan is intentionally scoped to that workload — no second backend, no async path, no parallelism, no cost cap. Add complexity only when reality demands it. - ---- - -## 0. Executive Summary - -A local-first CLI that ingests scanned handwritten exam PDFs, transcribes each page using **AWS Textract** (handwriting + `LAYOUT` blocks for prompt/answer structure), and produces an **interleaved PDF** — scan page followed by its transcription — compatible with Gradescope's section-assignment workflow. Identifying information is stripped locally before any Textract call. - -Phase 1 success = one working CLI that turns Crystal's scanned PDF batch into Gradescope-ready interleaved PDFs (<100 MB each) within an evening run on real samples. - ---- - -# PART A — Product Requirements Document (PRD) - -## A.1 Problem Statement - -Crystal grades essay-based handwritten exams (~230 students/semester across 3 courses; ~16–22 pages/student). Roughly **1 in 5 scripts is hard to read**, slowing grading inside Gradescope. A prior GSI (Ray) built a basic Python OCR pipeline producing interleaved manuscript+transcription PDFs that proved useful but was not maintained. We need to rebuild and harden that workflow with AWS Textract, basic privacy controls, and clear operability. - -## A.2 Goals & Non-Goals - -### Goals (Phase 1 — MVP) -1. Accept a scanned exam PDF (one student or a batch) as input. -2. Produce an **interleaved PDF**: page 1 = scan, page 2 = transcription of page 1, etc. -3. Preserve exam structure (prompt at top, answer below) so Gradescope section assignment still works. -4. Keep output PDFs **under 100 MB** (Gradescope limit). -5. Strip names/IDs from images before any Textract call (so handwritten text isn't sent to a cloud OCR); reattach on the final local PDF using the original student ID. -6. Run unattended in an evening on a single laptop. -7. Ship documentation a second engineer could pick up cold. - -### Non-Goals -- Automated grading or rubric scoring. -- Second OCR backend / A/B harness. -- Async Textract path with S3 staging. -- Multi-threaded batch execution. -- Cost cap enforcement (negligible at this scale). -- Citation verification, multi-tenant web app, LMS integrations beyond Gradescope-compatible PDFs. - -### Stretch (Phase 2+) -- Lightweight private web upload UI (single-user auth) hosted on Crystal's AWS account. -- Confidence highlighting (low-confidence tokens flagged in red on the transcription page). -- Async S3 path — only if real scans actually exceed Textract sync limits. - -## A.3 Users & Use Cases - -| User | Need | Frequency | -|---|---|---| -| Crystal (primary) | Convert scans -> interleaved PDFs before grading window | 3x per semester per course | -| Briac (dev/operator) | Run, monitor, debug pipeline | As needed | - -**Primary use case:** Crystal finishes scanning -> drops PDFs into an input folder -> runs one command -> receives interleaved PDFs ready to upload to Gradescope. - -## A.4 Functional Requirements - -### F1. Ingestion -- F1.1 Accept multi-page PDFs (one PDF per student OR one PDF containing many students with a separator convention). -- F1.2 Accept folder of PDFs as batch input. -- F1.3 Validate: not corrupt, reasonable DPI (>=200), file size, page count. -- F1.4 Reject pages exceeding Textract sync limits (10 MB image, 5000x5000 px) with a clear error pointing the operator to lower the DPI in config. We will not implement the async path in Phase 1. - -### F2. Preprocessing -- F2.1 Rasterize each PDF page to image (configurable DPI, default 300; lower to 250 if file size becomes a problem). -- F2.2 Deskew, denoise, contrast-normalize (all individually toggleable). -- F2.3 Detect & redact identifying regions (student name, SID) via a fixed header bounding box from a course template OR a regex pass after a first cheap header-only OCR. Redacted copies go to Textract; originals stay local. -- F2.4 Persist a per-page manifest (page_id <-> original_image <-> redacted_image <-> identity_metadata). - -### F3. OCR (AWS Textract) -- F3.1 **Single backend: AWS Textract** via `boto3`, called from the operator's own AWS account. -- F3.2 Per page, call `analyze_document` with `FeatureTypes=["LAYOUT"]` (no `FORMS` — we don't need key-value pairs for essays). -- F3.3 Sync path only. Each call sends the redacted image as raw `Bytes`. -- F3.4 Retry with tenacity exponential backoff (max 5 attempts) on `ThrottlingException`, `ProvisionedThroughputExceededException`, `InternalServerError`. -- F3.5 Parse the BLOCK tree (PAGE -> LAYOUT blocks -> LINE -> WORD) into a structured result with per-line bbox, per-word confidence, and LAYOUT block types (TITLE, SECTION_HEADER, TEXT) preserved. -- F3.6 Caching: identical (image hash, feature flags) -> cached result, no re-call. This is what makes re-runs cheap. -- F3.7 Per-page cost & latency logged from a configurable price-per-page constant. - -### F4. Postprocessing -- F4.1 Light cleanup: collapse spurious line breaks, normalize whitespace, preserve paragraph structure. -- F4.2 Use Textract LAYOUT block types to detect printed prompt (`TITLE`/`SECTION_HEADER`) vs. handwritten answer (`TEXT`); keep prompt at top of the transcription page. -- F4.3 Optional confidence highlighting (Phase 2). - -### F5. PDF Assembly -- F5.1 Build interleaved PDF: [scan_page_1, transcript_page_1, scan_page_2, transcript_page_2, ...]. -- F5.2 Reattach the original student name/SID on the cover page of the final local PDF (the redacted version was only for Textract; the final local PDF is for Crystal's own use in Gradescope). -- F5.3 Compress: target <=95 MB; if exceeded, downsample scan images and split into part-1 / part-2 PDFs with consistent naming. -- F5.4 Output filename convention: `{course}_{exam}_{student_id}.pdf` using the real student ID from the identity sidecar. Student IDs are roster data already shared with Gradescope and Canvas, so no anonymization is needed in the filename. - -### F6. CLI & Config -- F6.1 Single command: `ocr-grade run --input ./scans --output ./out --course PE101`. -- F6.2 Config file (`config.yaml`) for AWS profile/region, DPI, redaction template, course presets, Textract price constants. -- F6.3 Dry-run mode (process page 1 of first exam, show estimated total cost and time for the batch). - -### F7. Observability -- F7.1 Per-batch run log: pages processed, failures, total Textract cost, wall time. -- F7.2 Per-page artifact directory retained for debugging (toggleable). -- F7.3 Summary report (`out/run_report.md`) generated next to the output PDFs. - -## A.5 Non-Functional Requirements - -| Area | Requirement | -|---|---| -| Performance | A semester batch (~2,500–4,600 pages) finishes overnight on a single laptop running sequential Textract calls. | -| Cost | Tracked per page; expected total ~$120–$400/year all-in. No hard cap. | -| Privacy | No name/SID leaves the local machine. All Textract calls receive redacted images only. AWS region pinned to a US region. | -| Reliability | Resumable: re-running on the same input skips already-processed pages via cache. Textract retries on throttling/5xx. | -| Portability | macOS and Linux; Python 3.11+; single `uv sync` or `pip install -e .`. | -| Documentation | README + ARCHITECTURE.md + OPERATIONS.md + docs/aws-setup.md + docs/data-policy.md + CHANGELOG. | - -## A.6 Privacy, Compliance, Data Handling - -- **Local-first**: scans never leave the operator's machine in unredacted form. -- **Redaction**: header-region masking + name/SID regex pass before any Textract call. -- **Vendor**: AWS Textract under standard AWS service terms (no customer-input training when called from the operator's own account). Region pinned to a US region. Policy excerpt + link captured in `docs/data-policy.md`. -- **AWS account model (Phase 1):** everything runs in Briac's personal AWS account. This keeps the dev loop fast and avoids needing scheduled co-working time with Crystal. If/when the pilot succeeds, Briac sets up an identical IAM user + policy in Crystal's AWS account so she can run it independently — same code, same config schema, just different `AWS_PROFILE`. IAM user is least-privilege: `textract:AnalyzeDocument` + `textract:DetectDocumentText`. -- **No S3 bucket needed** in Phase 1 (sync path only) — eliminates a whole class of data-residency concerns. -- **At rest (local)**: all artifacts in a single working directory the operator controls; no implicit cloud sync. -- **Deletion**: `ocr-grade purge --batch ` removes all intermediate artifacts; final PDFs are the only retained output. - -## A.7 Success Metrics (Pilot) - -- >=90% of pages produce a transcription a human can read alongside the scan without losing time vs. reading the scan alone. -- On the ~20% "hard handwriting" bucket, transcription is rated "helpful" by Crystal on >=60% of pages. -- Zero leakage incidents (no name/SID transmitted to Textract). -- Crystal can run a batch end-to-end with the documentation alone. - -## A.8 Risks & Mitigations - -| Risk | Mitigation | -|---|---| -| Textract handwriting quality on cursive / messy scripts | Manual review of 15 samples in Phase 0 before committing to Textract; tune preprocessing (denoise, contrast). If unacceptable, add a second backend later — keep the adapter interface in place. | -| Redaction misses an ID written elsewhere | Combine fixed-region masking + regex + manual review of first batch. | -| Gradescope upload limit (100 MB) | Compression + automatic split + filename convention. | -| Page exceeds Textract sync limits | Clear error message + config hint to lower DPI; add async path in Phase 2 only if it actually happens. | -| AWS Textract pricing/policy changes | Price constants in config, not code; policy snapshot in `docs/data-policy.md`. | -| Briac unavailable mid-semester | Documentation + clean repo + scheduled progress updates. AWS setup is documented as a self-contained checklist so Crystal can replicate it in her own account without Briac. | - -## A.9 Open Questions (carried from call notes) - -1. Final scan DPI and file characteristics — answered after sample receipt. -2. Can name/SID always be auto-stripped? — confirm after redaction pass on samples. -3. Final form factor: CLI only, or CLI + small private web UI? — decide after MVP. -4. When to migrate to Crystal's AWS account — after the MVP works on Briac's account and the pilot is green. - ---- - -# PART B — Development Plan - -## B.1 Phasing - -### Phase 0 — Discovery (Days 0–3, blocked on samples) -- Receive 15 anonymized samples from Crystal. -- Catalogue: page count distribution, handwriting difficulty buckets, header layout. -- Run Textract on all 15 samples manually (one-off script) and eyeball results with Crystal. **Decision gate:** is Textract good enough? If yes, proceed. If no, revisit backend choice before writing more code. -- **Exit:** sample report committed to repo. - -### Phase 1 — MVP CLI (Weeks 1–3) -- Implement ingestion, preprocessing, redaction, Textract backend (sync only), postprocessing with LAYOUT-aware split, PDF assembly, CLI, caching, docs. -- **Exit:** Crystal runs the CLI on her own machine on a real (non-sample) past exam batch. - -### Phase 2 — Hardening (Weeks 4–5) -- Confidence highlighting using Textract per-word confidence. -- First end-to-end pilot on a current semester exam. -- **Exit:** pilot retrospective with Crystal; go/no-go on web UI. - -### Phase 2.5 — AWS account migration (after pilot, before scaling to all 3 courses) -- Briac walks Crystal through `docs/aws-setup.md` in her own account (IAM user, policy, region pin, credentials). -- Briac switches `AWS_PROFILE` and runs the same pipeline against the same scans to verify identical output. -- **Exit:** Crystal owns the AWS account; Briac is an IAM operator there. - -### Phase 3 (optional) — Private Web UI (Weeks 6–8) -- Single-user authenticated upload page; backend reuses the same pipeline. -- Hosted on Crystal's AWS account. -- **Exit:** Crystal uses the web UI for a full grading cycle. - -## B.2 Architecture (MVP) - -``` - ┌──────────────┐ - scans/*.pdf │ Ingestion │ - ──────────────▶│ + Validate │ - └──────┬───────┘ - ▼ - ┌──────────────┐ per-page images - │ Preprocess │───────────────────┐ - │ (deskew, │ │ - │ denoise, │ │ - │ redact ID) │ │ - └──────┬───────┘ │ - ▼ │ - ┌──────────────┐ │ - │ Textract │ sync only │ - │ + Cache │ LAYOUT feature │ - └──────┬───────┘ │ - ▼ │ - ┌──────────────┐ │ - │ Postprocess │ uses Textract │ - │ (cleanup, │ LAYOUT blocks for │ - │ layout) │ prompt/answer │ - └──────┬───────┘ │ - ▼ ▼ - ┌──────────────────────────────────────┐ - │ PDF Assembler (interleave + compress)│ - └──────────────┬───────────────────────┘ - ▼ - out/*.pdf + run_report.md -``` - -**Repo layout:** -``` -ocr-grade/ -├── pyproject.toml -├── README.md -├── ARCHITECTURE.md -├── OPERATIONS.md -├── config.example.yaml -├── src/ocr_grade/ -│ ├── __init__.py -│ ├── cli.py -│ ├── config.py -│ ├── ingestion.py -│ ├── preprocess.py -│ ├── redaction.py -│ ├── ocr/ -│ │ ├── base.py # OCRBackend Protocol (future-proofs cheaply) -│ │ ├── textract.py # only concrete backend -│ │ └── cache.py -│ ├── postprocess.py -│ ├── pdf_assembler.py -│ ├── reporting.py -│ └── utils.py -├── tests/ -│ ├── fixtures/ # synthetic only — no real student data -│ └── test_*.py -└── docs/ - ├── aws-setup.md - ├── data-policy.md - └── runbook.md -``` - -## B.3 Tech Stack - -- **Language:** Python 3.11+ -- **Package mgr:** `uv` -- **PDF I/O:** `pypdf`, `pdf2image` (Poppler), `pikepdf` for compression -- **Imaging:** `Pillow`, `opencv-python` -- **OCR:** **AWS Textract via `boto3`** (sole backend) -- **CLI:** `typer` -- **Config:** `pydantic-settings` + YAML -- **Caching:** content-addressed local store (`.cache/`) -- **Testing:** `pytest`, `pytest-snapshot`, `moto[textract]` for AWS mocking -- **Lint/format:** `ruff`, `mypy` -- **CI:** GitHub Actions (lint, tests on PR) - -## B.4 Milestones & Deliverables - -| Milestone | Deliverable | Target | -|---|---|---| -| M0 — Samples received + Textract sanity-check | Sample report in `docs/samples.md` with verdict | After Crystal sends | -| M1 — Skeleton repo | Repo + CI + CLI scaffold | Week 1 | -| M2 — Textract happy path | One PDF in -> interleaved PDF out | Week 2 | -| M3 — Redaction + cache + docs | Privacy guarantees met; resumable runs; docs | Week 3 | -| M4 — Real batch run | Crystal runs on a past exam batch | Week 4 | -| M5 — Pilot run | Live semester pilot complete | End of Week 5 | -| M6 — Decision gate | Web UI go/no-go | Week 6 | - -## B.5 Working Agreements - -- Weekly written update to Crystal (Friday) — scope, progress, blockers, AWS costs. -- Every PR has a description that another engineer could read cold. -- No real student data committed to git, ever (pre-commit hook enforces this). -- AWS access keys never committed; only profiles or env vars; rotated quarterly. -- Phase 1 runs on Briac's personal AWS account; switch to Crystal's account only after the pilot is green. - ---- - -# PART C — Claude Code Prompt Set - -Use these prompts in order inside Claude Code at the repo root. Each is self-contained; paste it, let the agent work, review the diff, commit, then move on. - -> **Conventions:** small reviewable PRs; add or update tests; never commit real student data; update `README.md` and `CHANGELOG.md` as part of the change. - -### Prompt 0 — Phase 0 sample sanity-check (one-off, before building anything) -``` -Write `scripts/sample_sanity_check.py` — a one-off Phase 0 script (no package -structure yet). It: - -- Takes a folder of PDF samples and an AWS profile name. -- For each page of each PDF: rasterizes at 300 DPI (pdf2image), then calls - Textract `analyze_document` with FeatureTypes=["LAYOUT"] via boto3. -- Writes `samples_report.md` with, per page: a thumbnail of the scan, the - reconstructed Textract text, the detected LAYOUT block types, and the average - per-word confidence. -- Prints total Textract cost using a configurable price-per-page constant. - -Goal: Crystal and Briac eyeball this report together and decide whether Textract -quality is good enough on real handwriting before building the full app. - -Commit as `chore(phase0): textract sanity-check script`. -``` - -### Prompt 1 — Bootstrap the repository -``` -Set up a new Python project called `ocr-grade`: - -1. Initialize a `uv`-managed Python 3.11 project with `pyproject.toml`. -2. Dependencies: typer, pydantic, pydantic-settings, pyyaml, pillow, opencv-python, - pdf2image, pypdf, pikepdf, rich, tenacity, boto3. - Dev deps: pytest, pytest-cov, pytest-snapshot, ruff, mypy, pre-commit, - moto[textract]. -3. Package layout under `src/ocr_grade/`: cli, config, ingestion, preprocess, - redaction, ocr/{base,cache,textract}, postprocess, pdf_assembler, reporting, - utils. -4. Configure ruff + mypy. Add `.pre-commit-config.yaml` with: - - ruff - - hook rejecting any staged file under `tests/fixtures/real/` or any *.pdf - outside `tests/fixtures/synthetic/` - - hook scanning staged files for AWS access key patterns - (AKIA[0-9A-Z]{16}) and blocking the commit. -5. GitHub Actions workflow `.github/workflows/ci.yml` running lint + tests on PR. -6. Files: README.md, ARCHITECTURE.md (copy the architecture diagram and repo - layout from the project plan), OPERATIONS.md (stub), CHANGELOG.md, - config.example.yaml, docs/aws-setup.md (stub), docs/data-policy.md (stub). -7. Typer CLI entrypoint `ocr-grade --help` with subcommands `run`, `dry-run`, - `purge`, `version` — all no-ops printing TODO. -8. Verify with `uv run pytest` and `uv run ocr-grade --help`. - -Commit as `chore: bootstrap project skeleton`. -``` - -### Prompt 2 — Config & input model -``` -Implement `src/ocr_grade/config.py`: - -- Pydantic `Settings` model loaded from `config.yaml` + env vars. -- Fields: - input_dir, output_dir, cache_dir, - dpi (default 300), - course_preset (str), - redaction: { header_box: [x,y,w,h] | None, regex_patterns: list[str] }, - aws: { profile: str | None, - region: str = "us-east-1", - access_key_id: SecretStr | None, - secret_access_key: SecretStr | None }, - textract_price_per_page (float, configurable), - preprocess_steps: { deskew: bool, denoise: bool, contrast: bool }. -- A `load_settings(path)` helper. -- Update `config.example.yaml` with documented defaults and a commented AWS block. -- Tests: load example file, override via env var - (e.g. OCR_GRADE__AWS__REGION), validation errors on bad values. - -Commit as `feat(config): typed settings with yaml + env loading`. -``` - -### Prompt 3 — Ingestion & preprocessing -``` -Implement `ingestion.py` and `preprocess.py`: - -- `ingestion.discover(input_dir) -> list[ExamFile]` returning records with path, - page_count, sha256, detected course (from filename pattern), validation status. -- Reject corrupt PDFs and PDFs whose rasterized DPI looks below ~150. -- Reject pages that, after rasterization, would exceed Textract sync limits - (10 MB image or 5000x5000 px) with a clear error pointing at the `dpi` config. -- `preprocess.rasterize(exam_file, dpi) -> list[PageImage]` using pdf2image. -- `preprocess.clean(image) -> image` applying deskew (Hough), denoise, adaptive - contrast. Each step toggleable via settings.preprocess_steps. -- Write artifacts to `cache_dir//page_.png`. -- Tests with synthetic PDFs generated on the fly (reportlab) — no real student data. - -Commit as `feat(ingest): pdf -> cleaned page images with manifest`. -``` - -### Prompt 4 — Redaction -``` -Implement `redaction.py`: - -- `redact(page_image, settings) -> RedactedPage`: - 1. Mask the configured header bounding box (if any) with a solid rectangle. - 2. Run a cheap header-only OCR pass (Textract `detect_document_text` on a small - crop is fine; abstract behind a `HeaderOCR` interface stubbable in tests), - then apply configured regex patterns - (e.g. r"SID[:\s]*\d{7,10}", r"Name[:\s]*[A-Z][a-z]+ [A-Z][a-z]+") - and mask matching regions. - 3. Return: redacted image + sidecar JSON with masked regions + extracted - identity strings — kept LOCAL ONLY, never sent in full to Textract. -- Tests: synthetic page with a fake "Name: John Doe / SID: 12345678" header gets - the region masked and identity captured in the sidecar. - -Commit as `feat(redaction): local header masking + identity sidecar`. -``` - -### Prompt 5 — OCR adapter interface + cache -``` -Implement `ocr/base.py` and `ocr/cache.py`: - -- `class OCRBackend(Protocol)`: - name: str - def transcribe(self, image_path: Path, page_meta: PageMeta) -> OCRResult -- `OCRResult`: text, lines (each with text + bbox + per-word confidence + LAYOUT - block type), raw_response, cost_usd, latency_ms. -- `cache.py`: content-addressed cache keyed by sha256(image_bytes) + backend.name - + serialized feature flags. Stores OCRResult as JSON. -- A `get_or_call(backend, image_path, meta)` helper. -- Tests with a fake in-memory backend covering cache hits/misses and cost - accumulation. - -(We keep the Protocol even though Textract is currently the only backend — -30 lines that make a future swap trivial.) - -Commit as `feat(ocr): backend protocol + content-addressed cache`. -``` - -### Prompt 6 — AWS Textract backend (sole backend, sync only) -``` -Implement `ocr/textract.py`: - -- `boto3` Textract client. Region + credentials from settings (AWS profile OR - explicit access key + secret key as SecretStr). Never hardcoded. -- Single sync path: `analyze_document` with FeatureTypes=["LAYOUT"], passing the - redacted image as raw `Bytes`. No FORMS, no async, no S3. -- Wrap calls with tenacity exponential backoff (max 5 attempts) on - ThrottlingException, ProvisionedThroughputExceededException, - InternalServerError. -- Parse the BLOCK tree (PAGE -> LAYOUT_* -> LINE -> WORD) into OCRResult with - per-line bbox, per-word confidence, and LAYOUT block types (TITLE, - SECTION_HEADER, TEXT) preserved on each line. -- Per-call cost = settings.textract_price_per_page (do not hardcode). -- Unit tests use `moto` to mock Textract (happy path, throttling retry, - permanent failure). Integration test gated behind env var AWS_PROFILE — skipped - otherwise. -- Fill in `docs/aws-setup.md`: - * Create an IAM user `ocr-grade-operator`. - * Attach this least-privilege policy JSON verbatim: - { - "Version": "2012-10-17", - "Statement": [ - { "Effect": "Allow", - "Action": [ - "textract:AnalyzeDocument", - "textract:DetectDocumentText" - ], - "Resource": "*" } - ] - } - * Configure local `~/.aws/credentials` or env vars; pin the region. - * No S3 bucket needed in Phase 1. -- Fill in `docs/data-policy.md` with the current AWS service terms excerpt and - link stating Textract input is not used to train AWS models, plus the chosen - US region and the IAM policy summary. - -Commit as `feat(ocr): AWS Textract sync backend`. -``` - -### Prompt 7 — Postprocessing -``` -Implement `postprocess.py`: - -- `clean_text(ocr_result) -> CleanedPage`: collapse spurious line breaks, - normalize whitespace, preserve paragraph boundaries inferred from line bbox - gaps. -- `split_prompt_and_answer(cleaned) -> (prompt_block, answer_block)`: - use Textract LAYOUT block types — TITLE / SECTION_HEADER lines at the top of - the page form the prompt; TEXT lines below form the answer. If the page has no - TITLE/SECTION_HEADER blocks, return whole page as answer with prompt_block=None. -- Tests on synthetic OCRResults. - -Commit as `feat(postprocess): cleanup + LAYOUT-aware prompt/answer split`. -``` - -### Prompt 8 — PDF assembler -``` -Implement `pdf_assembler.py`: - -- `build_interleaved(exam, transcripts, out_path)`: alternates scan page (from - the original PDF, NOT redacted) and a generated transcript page. Transcript - page layout: course/exam header, page number, prompt block (if any) at top in - a smaller font, then the answer in monospace-ish body font. -- Compression pass with pikepdf; if final size > 95 MB, downsample images and - retry; if still > 95 MB, split into part-1 / part-2 with consistent filename - suffixes. -- Filename: `{course}_{exam}_{student_id}.pdf` using the real student ID from the - identity sidecar (no hashing — student IDs are roster data already shared with - Gradescope/Canvas). -- Tests: golden-file comparison on a tiny synthetic exam. - -Commit as `feat(pdf): interleaved manuscript+transcript assembler with size guard`. -``` - -### Prompt 9 — Wire the CLI end-to-end -``` -Wire `cli.py run` to execute the full pipeline: - -ingestion -> preprocess -> redact -> textract (cached) -> postprocess --> pdf_assembler. - -- Sequential per page (no ThreadPoolExecutor — overkill at our scale). -- Rich progress bar over total pages. -- Print a running cost total as we go. -- `dry-run`: process page 1 of the first exam only, print estimated total cost - and projected wall time for the full batch. -- `purge --batch `: delete cache entries and intermediate artifacts for - one exam. -- Emit `out/run_report.md`: pages processed, failures, total Textract cost, - wall time. - -Commit as `feat(cli): end-to-end run with dry-run and purge`. -``` - -### Prompt 10 — Documentation polish -``` -Update docs for handoff: - -- README.md: 30-second quickstart + sample command. -- OPERATIONS.md: install, configure, AWS setup pointer, run a batch, troubleshoot, - purge, rotate AWS keys. -- ARCHITECTURE.md: refresh diagram, list module responsibilities, list extension - points (swap in a second backend by implementing OCRBackend, new redaction rule). -- docs/aws-setup.md: finalize IAM policy (textract:AnalyzeDocument + - textract:DetectDocumentText only), region pin, credential storage, key - rotation cadence. -- docs/data-policy.md: AWS Textract data-policy excerpt + link, region, - redaction guarantees, deletion procedure. -- docs/runbook.md: exact steps Crystal runs each grading cycle. - -Commit as `docs: handoff-grade documentation`. -``` - -### Prompt 11 — (Optional Phase 2) Confidence highlighting -``` -Highlight low-confidence tokens in red on the transcript page, driven by Textract -per-word confidence (configurable threshold, default 0.6). Add a `--no-highlight` -CLI flag. Update golden tests. - -Commit as `feat(pdf): low-confidence token highlighting`. -``` - -### Prompt 12 — (Optional Phase 3) Minimal private web UI -``` -Add `web/` subpackage: - -- FastAPI app with one authenticated upload page (single user via HTTP basic auth - or a signed magic link from settings). -- Endpoint POST /batches accepts a folder zip, runs the same pipeline in a - background task, exposes a status page and a download link for the interleaved - PDFs. -- Dockerfile + docs/deploy-aws.md describing deployment on a single EC2 or - Fargate task inside Crystal's AWS account, reusing the same IAM role that - already has Textract permissions. -- No multi-tenant features. No persistence beyond the working directory. - -Commit as `feat(web): single-user private upload UI`. -``` - ---- - -## Appendix — Quickstart for Briac - -```bash -# Phase 0 (before building the app) -python scripts/sample_sanity_check.py ./samples --aws-profile ocr-grade-dev -# eyeball samples_report.md with Crystal; decide go/no-go on Textract. - -# Then, repo setup -git init ocr-grade && cd ocr-grade -# paste Prompt 1 into Claude Code; review; commit. Then Prompt 2, etc. - -# AWS one-time (per docs/aws-setup.md): -# - create IAM user ocr-grade-operator with the 2-action least-privilege policy -# - configure AWS_PROFILE locally -# - no S3 bucket needed - -# Typical run, once built -ocr-grade dry-run --input ./scans -ocr-grade run --input ./scans --output ./out --course PE101 -``` - -## Appendix — Items to confirm with Crystal before Phase 1 close - -1. AWS Textract data-policy excerpt reviewed and accepted (Briac's account; same policy when we migrate to hers). -2. Header bounding box for redaction (per course template). -3. Form factor after MVP: CLI only vs. CLI + private web UI. -4. Timing for the Phase 2.5 AWS migration to Crystal's account. diff --git a/pyproject.toml b/pyproject.toml index edf5da2..cf3314e 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -22,6 +22,7 @@ dependencies = [ "fastapi>=0.115", "uvicorn[standard]>=0.30", "python-multipart>=0.0.9", + "python-dotenv>=1", ] [project.scripts] diff --git a/src/ocr_grade/cli.py b/src/ocr_grade/cli.py index a916735..43b545a 100644 --- a/src/ocr_grade/cli.py +++ b/src/ocr_grade/cli.py @@ -11,6 +11,7 @@ from pathlib import Path import typer +from dotenv import load_dotenv from rich.console import Console from rich.progress import ( BarColumn, @@ -24,6 +25,9 @@ from ocr_grade.config import Settings, load_settings from ocr_grade.ingestion import ValidationStatus, discover +# Pick up MISTRAL_API_KEY (and any overrides) from a local .env if present. +load_dotenv() + app = typer.Typer( help="Turn scanned handwritten exam PDFs into Gradescope-ready interleaved transcripts." ) diff --git a/src/ocr_grade/web/app.py b/src/ocr_grade/web/app.py index b187463..4c2aad6 100644 --- a/src/ocr_grade/web/app.py +++ b/src/ocr_grade/web/app.py @@ -16,6 +16,7 @@ from pathlib import Path from typing import Annotated +from dotenv import load_dotenv from fastapi import BackgroundTasks, Depends, FastAPI, Form, HTTPException, UploadFile, status from fastapi.responses import HTMLResponse, JSONResponse, RedirectResponse, Response from fastapi.security import HTTPBasic, HTTPBasicCredentials @@ -24,6 +25,10 @@ from .batches import Batch, UploadError from .settings import WebSettings, get_web_settings +# Load the API key + login from a local .env file (next to the project) so the +# operator never has to export env vars by hand. Real env vars win (override=False). +load_dotenv() + app = FastAPI(title="ocr-grade") security = HTTPBasic(auto_error=False) diff --git a/start.command b/start.command new file mode 100755 index 0000000..ea78578 --- /dev/null +++ b/start.command @@ -0,0 +1,26 @@ +#!/bin/bash +# Double-click this file to start the exam transcriber. +# +# A Terminal window will open and stay open while the app is running — that is +# normal, just leave it open. Your web browser will open to the app a few +# seconds later. When you are finished, close this Terminal window to stop. + +cd "$(dirname "$0")" || exit 1 + +# Make sure the 'uv' tool is findable no matter how it was installed. +export PATH="$HOME/.local/bin:$HOME/.cargo/bin:/opt/homebrew/bin:/usr/local/bin:$PATH" + +if ! command -v uv >/dev/null 2>&1; then + echo "Could not find 'uv'. Please finish the one-time setup in the README first." + echo "Press any key to close." ; read -n 1 -s -r ; exit 1 +fi + +echo "Starting up (the first run may take a minute to install things)…" +uv sync --quiet + +# Open the browser shortly after the server is ready. +( sleep 4 ; open "http://localhost:8000" ) & + +echo "The app is running. Your browser should open automatically." +echo "Leave this window open while you work; close it when you are done." +uv run uvicorn ocr_grade.web.app:app --host 127.0.0.1 --port 8000 diff --git a/uv.lock b/uv.lock index a7f3fa9..85aa9ef 100644 --- a/uv.lock +++ b/uv.lock @@ -859,6 +859,7 @@ dependencies = [ { name = "pydantic-settings" }, { name = "pymupdf" }, { name = "pypdf" }, + { name = "python-dotenv" }, { name = "python-multipart" }, { name = "pyyaml" }, { name = "rich" }, @@ -892,6 +893,7 @@ requires-dist = [ { name = "pydantic-settings", specifier = ">=2" }, { name = "pymupdf", specifier = ">=1.27.2.3" }, { name = "pypdf", specifier = ">=4" }, + { name = "python-dotenv", specifier = ">=1" }, { name = "python-multipart", specifier = ">=0.0.9" }, { name = "pyyaml", specifier = ">=6" }, { name = "rich", specifier = ">=13" },