diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 00000000..1d9b1b42 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,121 @@ +name: CI + +# Triggers: push to main, and PRs targeting main or any feature branch. +on: + push: + branches: [main] + pull_request: + branches: [main, "feature/**"] + +# Cancel superseded runs on the same ref to save runner minutes. +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + +# Default to read-only token; no write actions performed. +permissions: + contents: read + +jobs: + backend-tests: + name: Backend tests (Python ${{ matrix.python-version }}) + runs-on: ubuntu-latest + timeout-minutes: 20 + strategy: + fail-fast: false + matrix: + python-version: ["3.12"] + + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: ${{ matrix.python-version }} + + # WHY uv instead of pip: pip's resolver backtracks for minutes on the + # transitive cohere / logfire / opentelemetry-* deps pulled in by + # arize-phoenix. uv resolves the same set in seconds and ships its own + # caching that's faster than setup-python's pip cache for this graph. + - name: Install uv + uses: astral-sh/setup-uv@v6 + with: + enable-cache: true + cache-dependency-glob: requirements.txt + + - name: Install dependencies + run: | + uv pip install --system -r requirements.txt + uv pip install --system pytest-cov + + # WHY cache HF models: BgeEmbedder (~120MB) and the ms-marco-MiniLM cross-encoder + # (~80MB) are downloaded on first use. Without this cache, every CI run re-downloads + # them, wasting ~30s and bandwidth. Cache key is invalidated when requirements.txt + # changes (which is when transformer versions might shift). + - name: Cache Hugging Face models + uses: actions/cache@v4 + with: + path: | + ~/.cache/huggingface + ~/.cache/torch + key: hf-${{ runner.os }}-${{ hashFiles('requirements.txt') }} + restore-keys: | + hf-${{ runner.os }}- + + # WHY dummy API keys + CI_LLM_MOCK: openai.OpenAI() raises at construction + # when OPENAI_API_KEY is unset; any non-empty string lets the client construct. + # CI_LLM_MOCK=true triggers the conftest fixture that monkeypatches + # openai.OpenAI to an in-process stub, so tests that exercise the real + # backend.query path don't actually hit the OpenAI API. Local dev is + # unaffected since CI_LLM_MOCK is unset there. + # OTLP traces still try to flush to localhost:6006 and log connection-refused + # warnings — expected and harmless in CI. + - name: Run pytest + env: + PYTHONDONTWRITEBYTECODE: "1" + CI_LLM_MOCK: "true" + OPENAI_API_KEY: dummy_for_ci + ANTHROPIC_API_KEY: dummy_for_ci + GLM_API_KEY: dummy_for_ci + run: | + python -m pytest tests/ -q \ + --cov=src --cov-report=term-missing --cov-report=xml \ + --maxfail=5 --durations=20 + + - name: Upload coverage report + if: always() + uses: actions/upload-artifact@v4 + with: + name: coverage-py${{ matrix.python-version }} + path: coverage.xml + if-no-files-found: ignore + + frontend-build: + name: Frontend build + lint + runs-on: ubuntu-latest + timeout-minutes: 10 + defaults: + run: + working-directory: frontend + + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Set up Node.js + uses: actions/setup-node@v4 + with: + node-version: "20" + cache: npm + cache-dependency-path: frontend/package-lock.json + + - name: Install dependencies + run: npm ci + + - name: Lint + run: npm run lint + + - name: Build + run: npm run build diff --git a/.gitignore b/.gitignore index 3bba130c..8935a6c4 100644 --- a/.gitignore +++ b/.gitignore @@ -10,3 +10,6 @@ books/ .worktrees/ data/rag.db data/chroma/ + +# Eval harness run outputs +eval_runs/ diff --git a/Architecture.md b/Architecture.md index f2ceca24..2f4a58ac 100644 --- a/Architecture.md +++ b/Architecture.md @@ -443,6 +443,106 @@ Run: `python -m pytest tests/ -v` --- +## Evaluation Harness + +The `src/eval/` package provides a reproducible evaluation system over labeled gold sets, separate from the user-facing chat path. + +### Layers + +| Module | Responsibility | +|--------|----------------| +| `src/eval/schemas.py` | Pydantic contracts: `EvalQuestion`, `EvalResult`, `AggregatedMetric`, `RunMetadata`, `MetricDelta`, `CompareResult`. | +| `src/eval/pricing.py` | Hard-coded model price table + `cost_usd()` helper. | +| `src/eval/statistics.py` | `bootstrap_ci()` and `paired_permutation_test()` for run-level confidence intervals and two-run significance testing. | +| `src/eval/metrics/retrieval.py` | Recall@k, MRR@k, nDCG@k over `(gold_chunk_ids, retrieved_chunk_ids)`. | +| `src/eval/metrics/operational.py` | Per-stage latency p50/p95/p99, cost, token aggregation. | +| `src/eval/metrics/refusal.py` | Regex + LLM-judge refusal correctness for unanswerable questions. | +| `src/eval/metrics/generation.py` | Wraps existing `src/evaluation.py` LLM-as-judge functions; adds `answer_correctness` (cosine + judge mean) and `context_recall`. | +| `src/eval/datasets/squad_v2.py` | Seeded sample + frozen 200-row JSONL artifact from HuggingFace `squad_v2`. | +| `src/eval/datasets/ml_papers.py` | Hand-labeled dev set loader + manifest SHA-256 verification. | +| `src/eval/config.py` | YAML-loaded `EvalConfig`. | +| `src/eval/storage.py` | Run-directory CRUD over `eval_runs//`. | +| `src/eval/pipeline_factory.py` + `src/eval/_telemetry.py` | Builds an isolated RAG pipeline per (config, dataset) using ephemeral Chroma. | +| `src/eval/aggregator.py` | Per-dataset + combined `AggregatedMetric` rows from per-question results. | +| `src/eval/runner.py` | Orchestrates `git_sha`, ingest, query+score loop, aggregation, persistence. | +| `src/eval/compare.py` | Two-run diff with paired permutation tests + per-question regressions/wins. | +| `src/eval/report.py` + `templates/eval/*.html.j2` | Standalone jinja2 HTML reports. | +| `src/eval/cli.py` | `run`/`list`/`show`/`compare` argparse subcommands. | + +### API + UI + +`src/api/routes/eval.py` exposes: + +| Method | Path | Description | +|--------|------|-------------| +| `GET` | `/api/eval/configs` | List available eval configs | +| `POST` | `/api/eval/run` | Start a new eval run (dispatched via `BackgroundTasks`) | +| `GET` | `/api/eval/runs` | List all eval runs | +| `GET` | `/api/eval/runs/{id}` | Get run metadata | +| `GET` | `/api/eval/runs/{id}/results` | Per-question results | +| `GET` | `/api/eval/runs/{id}/status` | Live status for in-progress runs | +| `GET` | `/api/eval/compare` | Two-run diff with significance tests | + +Long-running runs dispatch via FastAPI `BackgroundTasks` and report progress through an in-process `RunRegistry` (`src/api/services/eval_runs.py`). + +React route `/eval/*` mounts three views: +- **`RunsList`** — sortable/filterable table with multi-select compare +- **`RunDetail`** — metric chart + per-question table with lazy expand +- **`CompareView`** — side-by-side bars + Top Wins / Top Regressions cards + +Charts use `recharts` with CI whiskers. + +### Eval Run Directory + +Each run produces `eval_runs//` with: +- `metadata.json` — run ID, git SHA, config name, timestamps +- `questions.jsonl` — per-question scores and retrieved chunks +- `metrics.json` — aggregated metric values with bootstrap CIs +- `cost.json` — token counts and USD costs per model +- `config.yaml` — snapshot of the config used + +The `eval_runs/` directory is gitignored; the labeled dev sets in `eval_data/` are checked in. + +--- + +## Observability + +The system exports per-stage spans for every chat query via OpenTelemetry to [Arize Phoenix](https://github.com/Arize-ai/phoenix) on `localhost:6006`. + +### Spans + +`RAGBackend.query_with_telemetry` and `RAGBackend.stream_query` open spans: + +| Span | Attributes | +|------|------------| +| `rag.retrieve` | `top_k`, `chunk_count` | +| `rag.generate` | `model`, `prompt_tokens`, `completion_tokens`, `cost_usd` | + +### Telemetry Payload + +The same numbers are returned to the client as a `StageTelemetry` Pydantic model (`src/api/schemas/telemetry.py`): + +- REST `POST /api/query` — includes a `telemetry` field in the response JSON. +- WebSocket `/api/chat` — emits a final `{"type": "telemetry", "content": {...}}` event after the existing `done` event. + +The frontend renders these as a muted footer line under each assistant chat bubble: + +> *Retrieve 142ms · Generate 2.1s · 4,217 tok · $0.0083* + +with a hover tooltip showing the prompt/completion token split. + +### Running with Traces + +Phoenix is profile-gated in `docker-compose.yml`; bare `docker compose up` does not start it. + +```bash +docker compose --profile observability up +``` + +`init_observability()` (`src/observability.py`) is called during the FastAPI lifespan startup. It is idempotent and fail-quiet — if Phoenix is unreachable, spans become no-ops and the chat continues to work normally. + +--- + ## Key Design Decisions | Decision | Choice | Rationale | diff --git a/README.md b/README.md index 97381d13..846b49e9 100644 --- a/README.md +++ b/README.md @@ -176,6 +176,17 @@ docker compose up --build | API | [localhost:8001](http://localhost:8001) | | Swagger Docs | [localhost:8001/docs](http://localhost:8001/docs) | +### Running with traces (Phoenix) + +```bash +docker compose --profile observability up +``` + +Phoenix UI is available at http://localhost:6006. The FastAPI backend +will export per-stage spans (`rag.retrieve`, `rag.generate`) with token +counts and cost as span attributes. If Phoenix isn't running, the app +works normally — span export silently fails. + ### Manual Setup ```bash diff --git a/configs/eval/README.md b/configs/eval/README.md new file mode 100644 index 00000000..370bc75d --- /dev/null +++ b/configs/eval/README.md @@ -0,0 +1,7 @@ +# Eval configs + +YAML files in this directory define complete pipeline + eval-set configurations. Each file is loaded by `src.eval.config.load_config(path)` into an `EvalConfig` Pydantic model and consumed by `EvalRunner.run()`. + +To author a new config, copy `baseline.yaml`, change the fields you want to vary, and rename. Run with `python -m src.eval.cli run --config configs/eval/.yaml`. + +See `docs/superpowers/specs/2026-04-26-rag-eval-harness-phase-1-design.md` §7 for the full schema. diff --git a/configs/eval/baseline.yaml b/configs/eval/baseline.yaml new file mode 100644 index 00000000..dbd597b9 --- /dev/null +++ b/configs/eval/baseline.yaml @@ -0,0 +1,18 @@ +name: "baseline" +description: "Current production defaults: recursive chunker, gpt-5-mini answer model, gpt-4.1-nano reasoning model." +pipeline: + chunker: + strategy: "recursive" + chunk_size: 512 + chunk_overlap: 64 + retriever: + top_k: 5 + generator: + model: "gpt-5-mini" + reasoning_model: "gpt-4.1-nano" +eval: + datasets: ["squad_v2_dev_200", "ml_papers_v1"] + judge_model: "gpt-4.1-mini" + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 diff --git a/configs/eval/baseline_squad_only.yaml b/configs/eval/baseline_squad_only.yaml new file mode 100644 index 00000000..5e2f5548 --- /dev/null +++ b/configs/eval/baseline_squad_only.yaml @@ -0,0 +1,18 @@ +name: "baseline_squad_only" +description: "Baseline pipeline against squad_v2_dev_200 only (ml_papers_v1 not yet labeled)." +pipeline: + chunker: + strategy: "recursive" + chunk_size: 512 + chunk_overlap: 64 + retriever: + top_k: 5 + generator: + model: "gpt-5-mini" + reasoning_model: "gpt-4.1-nano" +eval: + datasets: ["squad_v2_dev_200"] + judge_model: "gpt-4.1-mini" + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 diff --git a/configs/eval/phase2/phase2_baseline.yaml b/configs/eval/phase2/phase2_baseline.yaml new file mode 100644 index 00000000..b39a39a1 --- /dev/null +++ b/configs/eval/phase2/phase2_baseline.yaml @@ -0,0 +1,13 @@ +name: "phase2_baseline" +description: "Baseline anchor for the Phase 2 matrix — identical to baseline_squad_only." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + retriever: {top_k: 5} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 diff --git a/configs/eval/phase2/phase2b_embedder.yaml b/configs/eval/phase2/phase2b_embedder.yaml new file mode 100644 index 00000000..e737d04b --- /dev/null +++ b/configs/eval/phase2/phase2b_embedder.yaml @@ -0,0 +1,14 @@ +name: "phase2b_embedder" +description: "Phase 2 tier 2b — swap default embedder for BAAI/bge-small-en-v1.5." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + embedder: {name: bge_small_en_v1_5} + retriever: {top_k: 5} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 diff --git a/configs/eval/phase2/phase2c_hybrid.yaml b/configs/eval/phase2/phase2c_hybrid.yaml new file mode 100644 index 00000000..7f298669 --- /dev/null +++ b/configs/eval/phase2/phase2c_hybrid.yaml @@ -0,0 +1,15 @@ +name: "phase2c_hybrid" +description: "Phase 2 tier 2c — add BM25 hybrid on top of BGE." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + embedder: {name: bge_small_en_v1_5} + retriever: {top_k: 5} + hybrid: {enabled: true, bm25_top_k: 20, dense_top_k: 20, rrf_k: 60} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 diff --git a/configs/eval/phase2/phase2d_rerank.yaml b/configs/eval/phase2/phase2d_rerank.yaml new file mode 100644 index 00000000..cb0aceef --- /dev/null +++ b/configs/eval/phase2/phase2d_rerank.yaml @@ -0,0 +1,16 @@ +name: "phase2d_rerank" +description: "Phase 2 tier 2d — add cross-encoder rerank on top of hybrid." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + embedder: {name: bge_small_en_v1_5} + retriever: {top_k: 5} + hybrid: {enabled: true, bm25_top_k: 20, dense_top_k: 20, rrf_k: 60} + reranker: {model: ms_marco_minilm_l6_v2, rerank_top_n: 20, final_top_k: 5} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 diff --git a/configs/eval/phase2/phase2e_rewrite.yaml b/configs/eval/phase2/phase2e_rewrite.yaml new file mode 100644 index 00000000..b202a5b5 --- /dev/null +++ b/configs/eval/phase2/phase2e_rewrite.yaml @@ -0,0 +1,17 @@ +name: "phase2e_rewrite" +description: "Phase 2 tier 2e — add LLM query rewriting on top of rerank." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + embedder: {name: bge_small_en_v1_5} + retriever: {top_k: 5} + hybrid: {enabled: true, bm25_top_k: 20, dense_top_k: 20, rrf_k: 60} + reranker: {model: ms_marco_minilm_l6_v2, rerank_top_n: 20, final_top_k: 5} + query_rewriter: {model: gpt-4.1-nano, max_expansions: 3} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 diff --git a/configs/eval/phase2/phase2f_models_gpt41mini.yaml b/configs/eval/phase2/phase2f_models_gpt41mini.yaml new file mode 100644 index 00000000..35e088fb --- /dev/null +++ b/configs/eval/phase2/phase2f_models_gpt41mini.yaml @@ -0,0 +1,18 @@ +name: "phase2f_models_gpt41mini" +description: "Phase 2 tier 2f — answer-model comparison: gpt-4.1-mini variant." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + embedder: {name: bge_small_en_v1_5} + retriever: {top_k: 5} + hybrid: {enabled: true, bm25_top_k: 20, dense_top_k: 20, rrf_k: 60} + reranker: {model: ms_marco_minilm_l6_v2, rerank_top_n: 20, final_top_k: 5} + query_rewriter: {model: gpt-4.1-nano, max_expansions: 3} + generator: {model: gpt-4.1-mini, reasoning_model: gpt-4.1-nano} + refusal_handler: {enabled: true, similarity_threshold: 0.35, no_answer_text: "I don't have enough information to answer that."} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 diff --git a/configs/eval/phase2/phase2f_models_gpt5mini.yaml b/configs/eval/phase2/phase2f_models_gpt5mini.yaml new file mode 100644 index 00000000..b8563386 --- /dev/null +++ b/configs/eval/phase2/phase2f_models_gpt5mini.yaml @@ -0,0 +1,19 @@ +name: "phase2f_models_gpt5mini" +description: "Phase 2 tier 2f — answer-model comparison on the full 2g stack: gpt-5-mini variant. + This config is identical to phase2g_refusal.yaml; PR-B re-uses the 2g artifact." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + embedder: {name: bge_small_en_v1_5} + retriever: {top_k: 5} + hybrid: {enabled: true, bm25_top_k: 20, dense_top_k: 20, rrf_k: 60} + reranker: {model: ms_marco_minilm_l6_v2, rerank_top_n: 20, final_top_k: 5} + query_rewriter: {model: gpt-4.1-nano, max_expansions: 3} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} + refusal_handler: {enabled: true, similarity_threshold: 0.35, no_answer_text: "I don't have enough information to answer that."} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 diff --git a/configs/eval/phase2/phase2f_models_haiku.yaml b/configs/eval/phase2/phase2f_models_haiku.yaml new file mode 100644 index 00000000..45cd0363 --- /dev/null +++ b/configs/eval/phase2/phase2f_models_haiku.yaml @@ -0,0 +1,18 @@ +name: "phase2f_models_haiku" +description: "Phase 2 tier 2f — answer-model comparison: claude-haiku-4-5 variant." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + embedder: {name: bge_small_en_v1_5} + retriever: {top_k: 5} + hybrid: {enabled: true, bm25_top_k: 20, dense_top_k: 20, rrf_k: 60} + reranker: {model: ms_marco_minilm_l6_v2, rerank_top_n: 20, final_top_k: 5} + query_rewriter: {model: gpt-4.1-nano, max_expansions: 3} + generator: {model: claude-haiku-4-5, reasoning_model: null} + refusal_handler: {enabled: true, similarity_threshold: 0.35, no_answer_text: "I don't have enough information to answer that."} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 diff --git a/configs/eval/phase2/phase2g_refusal.yaml b/configs/eval/phase2/phase2g_refusal.yaml new file mode 100644 index 00000000..8a85c70c --- /dev/null +++ b/configs/eval/phase2/phase2g_refusal.yaml @@ -0,0 +1,18 @@ +name: "phase2g_refusal" +description: "Phase 2 tier 2g — full stack with refusal handler." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + embedder: {name: bge_small_en_v1_5} + retriever: {top_k: 5} + hybrid: {enabled: true, bm25_top_k: 20, dense_top_k: 20, rrf_k: 60} + reranker: {model: ms_marco_minilm_l6_v2, rerank_top_n: 20, final_top_k: 5} + query_rewriter: {model: gpt-4.1-nano, max_expansions: 3} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} + refusal_handler: {enabled: true, similarity_threshold: 0.35, no_answer_text: "I don't have enough information to answer that."} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 diff --git a/docker-compose.yml b/docker-compose.yml index adbbca74..17f3ab86 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -22,6 +22,11 @@ services: - ./tests:/app/tests - ./data:/app/data - ./books:/app/books + # WHY: The eval harness reads YAML configs from configs/eval/ and writes + # run directories to eval_runs/. Mounting both lets the dev loop + # (host CLI runs ↔ in-container API) share artifacts with no copy. + - ./configs:/app/configs + - ./eval_runs:/app/eval_runs # WHY: Cache the ChromaDB ONNX embedding model (79MB) so it doesn't # re-download on every container restart. First startup is slow, # subsequent starts are instant. @@ -67,5 +72,20 @@ services: depends_on: api: condition: service_healthy + + # PATTERN: Profile-gating — Phoenix only starts when explicitly requested + # with `docker compose --profile observability up`. A bare + # `docker compose up` starts only api + frontend, keeping the dev + # experience lightweight. + # WHY Phoenix: Arize Phoenix is an open-source, zero-cloud LLM observability + # UI that accepts OTLP spans and renders per-stage trace timelines + # with token counts and cost breakdowns. + phoenix: + image: arizephoenix/phoenix:latest + ports: + - "6006:6006" + profiles: ["observability"] + restart: unless-stopped + volumes: chroma-model-cache: diff --git a/docs/superpowers/plans/2026-04-27-phase2-rag-quality-matrix.md b/docs/superpowers/plans/2026-04-27-phase2-rag-quality-matrix.md new file mode 100644 index 00000000..e8ce7e34 --- /dev/null +++ b/docs/superpowers/plans/2026-04-27-phase2-rag-quality-matrix.md @@ -0,0 +1,2866 @@ +# Phase 2 RAG Quality Matrix — Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Land a layered RAG ablation matrix that measures the lift each major architectural lever buys on top of the Phase 1 SQuAD-200 baseline, then ship a portfolio writeup attributing the lift to mechanism. + +**Architecture:** Two PRs stacked on `feature/eval-harness-1d`. PR-A adds 5 new pipeline modules (`BgeEmbedder`, `BM25HybridRetriever`, `CrossEncoderReranker`, `QueryRewriter`, `RefusalHandler`), extends `EvalConfig.pipeline` with backward-compatible sub-configs, refactors the cost ledger to cover all three LLM call sites (generator + judge + rewriter), and wires everything into `src/eval/pipeline_factory.py::build_pipeline`. PR-B executes 8 distinct evals against SQuAD-200, archives the small artifacts to `docs/phase2/runs/`, and ships `docs/PHASE2_RESULTS.md`. + +**Tech Stack:** Python 3.12 + Pydantic v2 + ChromaDB EphemeralClient + sentence-transformers (BGE-small + ms-marco-MiniLM cross-encoder) + rank-bm25 + pytest + the existing eval harness from Phase 1. Frontend untouched. + +**Spec source of truth:** [`docs/superpowers/specs/2026-04-27-phase2-rag-quality-matrix-design.md`](../specs/2026-04-27-phase2-rag-quality-matrix-design.md) + +--- + +## File Structure + +### New files (PR-A) + +| Path | Responsibility | +|------|----------------| +| `src/eval/embedders/__init__.py` | Re-export `BgeEmbedder`. | +| `src/eval/embedders/bge_small.py` | Chroma `EmbeddingFunction` adapter loading `BAAI/bge-small-en-v1.5` via `sentence-transformers`. | +| `src/eval/retrievers/__init__.py` | Re-export `BM25HybridRetriever`, `CrossEncoderReranker`. | +| `src/eval/retrievers/bm25_hybrid.py` | BM25 + dense retriever with Reciprocal Rank Fusion. Wraps a `ChromaVectorStore`. | +| `src/eval/retrievers/reranker.py` | `cross-encoder/ms-marco-MiniLM-L-6-v2` post-retrieval reranker. | +| `src/eval/transforms/__init__.py` | Re-export `QueryRewriter`, `RefusalHandler`. | +| `src/eval/transforms/query_rewriter.py` | LLM-based query expansion; returns answer text + token usage. | +| `src/eval/transforms/refusal_handler.py` | Pure-logic similarity gate; returns refusal text on low confidence. | +| `tests/test_eval_embedder_bge.py` | Unit + Chroma-integration tests for `BgeEmbedder`. | +| `tests/test_eval_retriever_bm25_hybrid.py` | RRF unit test + retrieval integration. | +| `tests/test_eval_retriever_reranker.py` | Cross-encoder rerank order. | +| `tests/test_eval_transform_rewriter.py` | Pass-through + stubbed-LLM expansion. | +| `tests/test_eval_transform_refusal.py` | Threshold gate + empty-candidates edge. | +| `tests/test_eval_cost_ledger.py` | Aggregator sums generator + judge + rewriter spend. | +| `tests/test_eval_cli_archive.py` | `cli archive` copies the four small artifacts only. | +| `tests/test_eval_pipeline_factory_phase2.py` | Each Phase 2 YAML produces a pipeline whose attributes match the tier toggles. | +| `tests/fixtures/phase2_corpus/*.txt` | Three tiny text docs for smoke + integration tests. | +| `configs/eval/phase2/phase2_baseline.yaml` | Baseline re-run anchor. | +| `configs/eval/phase2/phase2b_embedder.yaml` | + BGE embedder. | +| `configs/eval/phase2/phase2c_hybrid.yaml` | + BM25 hybrid. | +| `configs/eval/phase2/phase2d_rerank.yaml` | + cross-encoder rerank. | +| `configs/eval/phase2/phase2e_rewrite.yaml` | + query rewriting. | +| `configs/eval/phase2/phase2g_refusal.yaml` | + refusal handler. | +| `configs/eval/phase2/phase2f_models_gpt5mini.yaml` | 2g stack with `gpt-5-mini` (re-uses 2g artifact). | +| `configs/eval/phase2/phase2f_models_gpt41mini.yaml` | 2g stack with `gpt-4.1-mini`. | +| `configs/eval/phase2/phase2f_models_haiku.yaml` | 2g stack with `claude-haiku-4-5`. | + +### Modified files (PR-A) + +| Path | Change | +|------|--------| +| `src/eval/config.py` | Add 5 new sub-config models + `EvalCfg.spend_ceiling_usd`; extend `PipelineCfg`. | +| `src/eval/schemas.py` | Add `EvalResult.cost_breakdown: dict[str, float]`. | +| `src/eval/pricing.py` | Add `gpt-4.1-nano`, `claude-haiku-4-5` to `MODEL_PRICES` (verify `gpt-4.1-mini` present). | +| `src/eval/metrics/generation.py` | Judge functions return `(score, details, prompt_tokens, completion_tokens, cost_usd)`. | +| `src/eval/runner.py` | `_score_question` collects judge usage; `_query_one` collects rewriter usage; spend-ceiling enforcement. | +| `src/eval/pipeline_factory.py` | `build_pipeline` builds embedder/hybrid/reranker/rewriter/refusal; `EvalPipeline.query` invokes them. | +| `src/llm_handler.py` | Add `generate_with_usage(prompt, system_prompt) -> tuple[str, int, int]`. | +| `src/eval/cli.py` | Add `archive` subcommand. | +| `src/eval/aggregator.py` | Sum `cost_breakdown` into `cost.json` totals. | +| `requirements.txt` | + `rank-bm25`. | + +### New files (PR-B) + +| Path | Responsibility | +|------|----------------| +| `docs/phase2/runs/_phase2_*/{metrics,cost,metadata,config}.json` | Archived small artifacts. | +| `docs/phase2/compare/*.html` | Pairwise compare reports (5 chain + 2 model). | +| `docs/PHASE2_RESULTS.md` | Methodology, chart, significance table, findings. | + +### Modified files (PR-B) + +| Path | Change | +|------|--------| +| `README.md` | Link to `docs/PHASE2_RESULTS.md`. | + +--- + +## Branching + +```bash +git checkout feature/eval-harness-1d +git pull origin feature/eval-harness-1d +git checkout -b feature/phase2-pipeline-extensions +``` + +PR-A targets `feature/eval-harness-1d`. After Phase 1 merges, retarget PR-A to `main`. PR-B branches from PR-A's head once PR-A is reviewable. + +--- + +# PR-A — Pipeline Extensions + +## Task 1: Extend `PipelineCfg` schema with Phase 2 sub-configs + +**Files:** +- Modify: `src/eval/config.py` +- Modify: `tests/test_eval_config.py` (existing test file) + +- [ ] **Step 1: Write the failing tests** + +Append to `tests/test_eval_config.py`: + +```python +# --- Phase 2 schema additions ---------------------------------------------- + +import pytest +from pydantic import ValidationError +from src.eval.config import ( + EvalConfig, EmbedderCfg, HybridCfg, RerankerCfg, + QueryRewriterCfg, RefusalHandlerCfg, +) + +def test_phase2_subconfigs_default_to_off(tmp_path): + """Loading an existing baseline-shape YAML must produce all-default Phase 2 blocks.""" + yaml_text = """ +name: legacy_baseline +description: existing config without phase 2 blocks +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + retriever: {top_k: 5} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} +eval: + datasets: [squad_v2_dev_200] +""" + p = tmp_path / "legacy.yaml" + p.write_text(yaml_text) + from src.eval.config import load_config + cfg = load_config(p) + # All new blocks present with defaults + assert cfg.pipeline.embedder.name == "chroma_default" + assert cfg.pipeline.hybrid.enabled is False + assert cfg.pipeline.reranker.model is None + assert cfg.pipeline.query_rewriter.model is None + assert cfg.pipeline.refusal_handler.enabled is False + assert cfg.eval.spend_ceiling_usd is None + +def test_phase2_subconfig_typed_values(tmp_path): + """Phase 2 fields validate to the right types.""" + yaml_text = """ +name: phase2g +description: refusal handler enabled +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + retriever: {top_k: 5} + embedder: {name: bge_small_en_v1_5} + hybrid: {enabled: true, bm25_top_k: 20, dense_top_k: 20, rrf_k: 60} + reranker: {model: ms_marco_minilm_l6_v2, rerank_top_n: 20, final_top_k: 5} + query_rewriter: {model: gpt-4.1-nano, max_expansions: 3} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} + refusal_handler: {enabled: true, similarity_threshold: 0.35} +eval: + datasets: [squad_v2_dev_200] + spend_ceiling_usd: 1.5 +""" + p = tmp_path / "phase2g.yaml" + p.write_text(yaml_text) + from src.eval.config import load_config + cfg = load_config(p) + assert cfg.pipeline.embedder.name == "bge_small_en_v1_5" + assert cfg.pipeline.hybrid.enabled is True + assert cfg.pipeline.reranker.model == "ms_marco_minilm_l6_v2" + assert cfg.pipeline.query_rewriter.model == "gpt-4.1-nano" + assert cfg.pipeline.refusal_handler.similarity_threshold == 0.35 + assert cfg.eval.spend_ceiling_usd == 1.5 + +def test_phase2_unknown_field_rejected(tmp_path): + """extra='forbid' must reject unknown keys at load time.""" + yaml_text = """ +name: bad +description: typo in field name +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + retriever: {top_k: 5} + hybrid: {enabld: true} # typo on purpose + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} +eval: + datasets: [squad_v2_dev_200] +""" + p = tmp_path / "bad.yaml" + p.write_text(yaml_text) + from src.eval.config import load_config + with pytest.raises(ValidationError): + load_config(p) +``` + +- [ ] **Step 2: Run tests to verify they fail** + +```bash +pytest tests/test_eval_config.py::test_phase2_subconfigs_default_to_off \ + tests/test_eval_config.py::test_phase2_subconfig_typed_values \ + tests/test_eval_config.py::test_phase2_unknown_field_rejected -v +``` +Expected: FAIL with `ImportError` for `EmbedderCfg`/etc. + +- [ ] **Step 3: Implement the schema extension** + +Insert into `src/eval/config.py` immediately before the `class PipelineCfg` definition: + +```python +class EmbedderCfg(BaseModel): + """Embedder selection. None/default = ChromaDB built-in ONNX (current behavior). + + Phase 2 lever 2b: swap ChromaDB's default MiniLM (384-dim) for BAAI/bge-small-en-v1.5 + (also 384-dim) which is domain-tuned for retrieval. The factory wires this as a + Chroma EmbeddingFunction at collection-creation time, so ChromaVectorStore.upsert/query + auto-embed without per-call code changes. + """ + model_config = ConfigDict(extra="forbid") + name: Literal["chroma_default", "bge_small_en_v1_5"] = "chroma_default" + + +class HybridCfg(BaseModel): + """BM25 + dense retrieval with Reciprocal Rank Fusion. + + Phase 2 lever 2c: combine sparse (BM25) and dense (vector) signal. Disabled by default; + when enabled, the retriever fetches top-N candidates from each side and fuses them + with RRF: score(d) = sum over r in {dense, sparse} of 1 / (rrf_k + rank_r(d)). + """ + model_config = ConfigDict(extra="forbid") + enabled: bool = False + bm25_top_k: int = 20 + dense_top_k: int = 20 + rrf_k: int = 60 + + +class RerankerCfg(BaseModel): + """Cross-encoder rerank top-N → final-K. None = no rerank (current behavior). + + Phase 2 lever 2d: improve precision by re-scoring the top-N retrieved candidates + with a dedicated relevance model (ms-marco-MiniLM-L-6-v2). Adds latency but no + LLM cost. + """ + model_config = ConfigDict(extra="forbid") + model: Literal["ms_marco_minilm_l6_v2"] | None = None + rerank_top_n: int = 20 + final_top_k: int = 5 + + +class QueryRewriterCfg(BaseModel): + """LLM-based query expansion. None = no rewrite (current behavior). + + Phase 2 lever 2e: ask an LLM to produce up to N alternative phrasings of the user + query, retrieve against each, then deduplicate. Costs one LLM call per question; + captured in the cost ledger under the 'rewriter' bucket. + """ + model_config = ConfigDict(extra="forbid") + model: str | None = None + max_expansions: int = 3 + + +class RefusalHandlerCfg(BaseModel): + """Answerability gate. enabled=False = current behavior. + + Phase 2 lever 2g: when the top-1 retrieval similarity falls below `similarity_threshold`, + short-circuit to `no_answer_text` instead of calling the generator. This trades + answer_correctness on borderline-answerable questions for refusal_correctness on + truly unanswerable ones — exactly the trade-off the SQuAD v2 dev set surfaces. + """ + model_config = ConfigDict(extra="forbid") + enabled: bool = False + similarity_threshold: float = 0.35 + no_answer_text: str = "I don't have enough information to answer that." +``` + +Replace the existing `class PipelineCfg` block with: + +```python +class PipelineCfg(BaseModel): + """Aggregates all pipeline-level sub-configs into one validated structure. + + Phase 2 additions are all default-off so existing baseline configs keep validating + unchanged. Each new block is a documented lever; see configs/eval/phase2/*.yaml + for tier-by-tier toggles. + """ + chunker: ChunkerCfg + embedder: EmbedderCfg = Field(default_factory=EmbedderCfg) + retriever: RetrieverCfg + hybrid: HybridCfg = Field(default_factory=HybridCfg) + reranker: RerankerCfg = Field(default_factory=RerankerCfg) + query_rewriter: QueryRewriterCfg = Field(default_factory=QueryRewriterCfg) + generator: GeneratorCfg + refusal_handler: RefusalHandlerCfg = Field(default_factory=RefusalHandlerCfg) +``` + +In the same file, locate `class EvalCfg(BaseModel):` and append `spend_ceiling_usd: float | None = None` after the existing `seed: int = 42` line: + +```python + seed: int = 42 + # Phase 2: hard per-run spend ceiling. EvalRunner aborts when the cumulative + # generator + judge + rewriter cost crosses this. None disables the guard. + spend_ceiling_usd: float | None = None +``` + +- [ ] **Step 4: Run all tests to verify they pass** + +```bash +pytest tests/test_eval_config.py -v +``` +Expected: all tests PASS, including the existing baseline tests. + +- [ ] **Step 5: Commit** + +```bash +git add src/eval/config.py tests/test_eval_config.py +git commit -m "feat(eval): extend PipelineCfg with Phase 2 sub-configs and spend ceiling" +``` + +--- + +## Task 2: Cost ledger covers generator + judge + rewriter + +**Files:** +- Modify: `src/eval/schemas.py` +- Modify: `src/eval/metrics/generation.py` +- Modify: `src/eval/runner.py` +- Modify: `src/eval/aggregator.py` +- Modify: `src/llm_handler.py` +- Modify: `src/eval/pricing.py` +- Create: `tests/test_eval_cost_ledger.py` +- Modify: `tests/test_eval_runner.py` (existing — extend, don't break) + +- [ ] **Step 1: Write failing tests for the new cost ledger surface** + +Create `tests/test_eval_cost_ledger.py`: + +```python +"""Tests for the Phase 2 cost ledger covering generator + judge + rewriter spend.""" + +from __future__ import annotations + +from src.eval.schemas import EvalResult + + +def test_eval_result_has_cost_breakdown_field(): + """EvalResult must carry a cost_breakdown dict with per-bucket spend.""" + r = EvalResult( + question_id="q1", + dataset="squad_v2_dev_200", + retrieved_chunk_ids=[], + retrieved_chunks=[], + generated_answer="", + metrics={}, + timings_ms={}, + tokens={}, + cost_usd=0.0, + cost_breakdown={"generator": 0.0, "judge": 0.0, "rewriter": 0.0}, + ) + assert r.cost_breakdown["generator"] == 0.0 + assert r.cost_breakdown["judge"] == 0.0 + assert r.cost_breakdown["rewriter"] == 0.0 + + +def test_eval_result_cost_breakdown_defaults(): + """cost_breakdown must default to a generator-only dict when omitted.""" + r = EvalResult( + question_id="q1", + dataset="squad_v2_dev_200", + retrieved_chunk_ids=[], + retrieved_chunks=[], + generated_answer="", + metrics={}, + timings_ms={}, + tokens={}, + cost_usd=0.05, + ) + # back-compat default: existing records read as generator-only + assert r.cost_breakdown == {"generator": 0.05, "judge": 0.0, "rewriter": 0.0} + + +def test_aggregator_sums_cost_breakdown_into_totals(): + """aggregate_costs must surface per-bucket totals alongside total_usd.""" + from src.eval.metrics.operational import aggregate_costs + + results = [ + EvalResult( + question_id="q1", dataset="d", retrieved_chunk_ids=[], retrieved_chunks=[], + generated_answer="", metrics={}, timings_ms={}, tokens={}, + cost_usd=0.10, + cost_breakdown={"generator": 0.04, "judge": 0.05, "rewriter": 0.01}, + ), + EvalResult( + question_id="q2", dataset="d", retrieved_chunk_ids=[], retrieved_chunks=[], + generated_answer="", metrics={}, timings_ms={}, tokens={}, + cost_usd=0.20, + cost_breakdown={"generator": 0.08, "judge": 0.10, "rewriter": 0.02}, + ), + ] + summary = aggregate_costs(results) + assert round(summary["total_usd"], 4) == 0.30 + assert round(summary["generator_total_usd"], 4) == 0.12 + assert round(summary["judge_total_usd"], 4) == 0.15 + assert round(summary["rewriter_total_usd"], 4) == 0.03 + + +def test_llm_handler_generate_with_usage_returns_tokens(): + """LLMHandler.generate_with_usage must return (text, prompt_tokens, completion_tokens).""" + from src.llm_handler import LLMHandler + + # Use the dummy fallback path — no API key needed. + handler = LLMHandler("__dummy__") + text, prompt_tokens, completion_tokens = handler.generate_with_usage( + prompt="What is RAG?", system_prompt="Be brief." + ) + assert isinstance(text, str) + assert isinstance(prompt_tokens, int) and prompt_tokens > 0 + assert isinstance(completion_tokens, int) and completion_tokens >= 0 +``` + +- [ ] **Step 2: Run the tests to verify they fail** + +```bash +pytest tests/test_eval_cost_ledger.py -v +``` +Expected: FAIL — `cost_breakdown` field missing, `aggregate_costs` missing the new keys, `generate_with_usage` undefined. + +- [ ] **Step 3: Add `cost_breakdown` to `EvalResult`** + +In `src/eval/schemas.py`, locate `class EvalResult` and add after the `cost_usd: float` line: + +```python + cost_usd: float + # Phase 2: per-bucket breakdown. Defaults to generator-only when absent so + # Phase 1 records continue to round-trip through model_validate. + cost_breakdown: dict[str, float] = Field(default_factory=dict) + error: str | None = None + + @model_validator(mode="after") + def _backfill_cost_breakdown(self) -> "EvalResult": + if not self.cost_breakdown: + object.__setattr__(self, "cost_breakdown", { + "generator": self.cost_usd, + "judge": 0.0, + "rewriter": 0.0, + }) + return self +``` + +Add `from pydantic import model_validator` to the imports at the top of the file if not already present. + +- [ ] **Step 4: Add `generate_with_usage` to `LLMHandler`** + +In `src/llm_handler.py`, immediately after the existing `def generate` method (around line 162), insert: + +```python + def generate_with_usage( + self, + prompt: str, + system_prompt: str | None = None, + ) -> tuple[str, int, int]: + """Generate a response and return text plus prompt/completion token counts. + + WHY a separate method: the existing `generate()` returns only `str` and is + called in many places that don't need usage. Phase 2's cost ledger needs + token counts on every LLM call; rather than break callers, we add a parallel + method that uses the existing tokenizer to estimate counts client-side. + + Args: + prompt: User message. + system_prompt: Optional system instructions. + + Returns: + (response_text, prompt_tokens, completion_tokens). + """ + from src.eval._telemetry import count_tokens + + text = self.generate(prompt, system_prompt=system_prompt) + full_prompt = (system_prompt + "\n" + prompt) if system_prompt else prompt + prompt_tokens = count_tokens(full_prompt, self._model) + completion_tokens = count_tokens(text, self._model) + return text, prompt_tokens, completion_tokens +``` + +(Verify the attribute is `self._model` by reading the class init; if it's `self.model`, adjust accordingly.) + +- [ ] **Step 5: Extend `aggregate_costs` in `src/eval/metrics/operational.py`** + +Locate the existing `def aggregate_costs(results: list[EvalResult])`. Add the per-bucket totals before returning the summary dict: + +```python +def aggregate_costs(results: list[EvalResult]) -> dict[str, float]: + """Sum cost_usd and per-bucket breakdown across all results.""" + total = sum(r.cost_usd for r in results) + n = len(results) if results else 1 + generator_total = sum(r.cost_breakdown.get("generator", 0.0) for r in results) + judge_total = sum(r.cost_breakdown.get("judge", 0.0) for r in results) + rewriter_total = sum(r.cost_breakdown.get("rewriter", 0.0) for r in results) + return { + "total_usd": total, + "mean_usd_per_query": total / n, + "generator_total_usd": generator_total, + "judge_total_usd": judge_total, + "rewriter_total_usd": rewriter_total, + } +``` + +(Preserve the existing `total_prompt`/`total_completion` keys if `aggregate_tokens` lives in the same function; if separate, keep them untouched.) + +- [ ] **Step 6: Pricing — register Phase 2 models** + +In `src/eval/pricing.py`, locate the `MODEL_PRICES` dict and add (verify which are missing first): + +```python + # Phase 2 additions + "claude-haiku-4-5": ModelPrice(prompt_per_1m=1.0, completion_per_1m=5.0), + "gpt-4.1-nano": ModelPrice(prompt_per_1m=0.10, completion_per_1m=0.40), + # gpt-4.1-mini and gpt-5-mini should already be present +``` + +If `gpt-4.1-mini` or `gpt-5-mini` are missing, add them too with current published rates (consult OpenAI's pricing page; document the source in a comment). + +- [ ] **Step 7: Wire judge cost capture into `_score_question`** + +In `src/eval/metrics/generation.py`, change the judge functions to return cost. Take `_judge_factual_match` as the model: + +```python +def _judge_factual_match( + generated: str, + gold: str, + llm, +) -> tuple[float, str, int, int]: + """Score factual agreement and return (score, reasoning, prompt_tokens, completion_tokens).""" + # ... existing prompt construction unchanged ... + raw, prompt_tokens, completion_tokens = llm.generate_with_usage( + user_prompt, system_prompt=system_prompt, + ) + # ... existing JSON parsing unchanged, returning (score, reasoning, prompt_tokens, completion_tokens) + try: + parsed = json.loads(stripped) + score = max(0.0, min(1.0, float(parsed["factual_match"]))) + reasoning = str(parsed.get("reasoning", "")) + return score, reasoning, prompt_tokens, completion_tokens + except (json.JSONDecodeError, KeyError, ValueError): + return 0.0, "Judge returned malformed JSON; defaulting to 0.0", prompt_tokens, completion_tokens +``` + +Apply the same pattern to `judge_faithfulness`, `judge_answer_relevancy`, `judge_context_precision`, and the LLM call inside `answer_correctness`. Each returns `(score, details, prompt_tokens, completion_tokens)` where `details` is the existing dict. + +In `src/eval/runner.py`, in `_score_question`, accumulate judge tokens/cost into a local variable and write into `metric_details["judge_cost_usd"]` and `metric_details["judge_tokens"]`. Return them so `_query_one` can write them into `EvalResult.cost_breakdown`. Sketch: + +```python +def _score_question(question, chunks, answer, judge_llm) -> tuple[dict, dict, dict]: + """Returns (metrics, metric_details, judge_usage) where judge_usage = + {'cost_usd': float, 'prompt_tokens': int, 'completion_tokens': int}.""" + # ... existing scoring loop, accumulating tokens/cost from each judge call ... + judge_prompt_tokens = 0 + judge_completion_tokens = 0 + judge_cost = 0.0 + judge_model = ... # read from runner config + # for each judge call: + score, details, p_t, c_t = judge_function(...) + judge_prompt_tokens += p_t + judge_completion_tokens += c_t + judge_cost += pricing.cost_usd(judge_model, p_t, c_t) + # ... + return metrics, metric_details, { + "cost_usd": judge_cost, + "prompt_tokens": judge_prompt_tokens, + "completion_tokens": judge_completion_tokens, + } +``` + +In `_query_one`, after the existing telemetry assembly, add judge usage to the breakdown: + +```python +metrics, metric_details, judge_usage = _score_question(question, chunks, answer, judge_llm) +generator_cost = telemetry["cost_usd"] +return EvalResult( + ..., + cost_usd=generator_cost + judge_usage["cost_usd"], + cost_breakdown={ + "generator": generator_cost, + "judge": judge_usage["cost_usd"], + "rewriter": telemetry.get("rewriter_cost_usd", 0.0), + }, + ..., +) +``` + +(`rewriter_cost_usd` defaults to 0.0 here; Task 6 wires it in when the rewriter is enabled.) + +- [ ] **Step 8: Spend-ceiling enforcement** + +In `src/eval/runner.py`, inside `EvalRunner.run`, immediately after appending each `EvalResult`, add: + +```python +ceiling = config.eval.spend_ceiling_usd +if ceiling is not None: + cumulative = sum(r.cost_usd for r in all_results) + if cumulative > ceiling: + raise RuntimeError( + f"Spend ceiling exceeded: ${cumulative:.4f} > ${ceiling:.4f} " + f"after {len(all_results)} questions. Aborting run." + ) +``` + +- [ ] **Step 9: Run all eval tests and verify they pass** + +```bash +pytest tests/test_eval_cost_ledger.py tests/test_eval_runner.py \ + tests/test_eval_metrics_generation.py tests/test_eval_aggregator.py -v +``` +Expected: all PASS. Any pre-existing test that asserted exact dict shape on `aggregate_costs` may need a one-line update to allow the new keys. + +- [ ] **Step 10: Commit** + +```bash +git add src/eval/schemas.py src/eval/runner.py src/eval/metrics/generation.py \ + src/eval/metrics/operational.py src/eval/pricing.py src/llm_handler.py \ + tests/test_eval_cost_ledger.py tests/test_eval_runner.py \ + tests/test_eval_metrics_generation.py +git commit -m "feat(eval): cost ledger covers generator + judge + rewriter spend" +``` + +--- + +## Task 3: `BgeEmbedder` — Chroma EmbeddingFunction adapter + +**Files:** +- Create: `src/eval/embedders/__init__.py` +- Create: `src/eval/embedders/bge_small.py` +- Create: `tests/test_eval_embedder_bge.py` + +- [ ] **Step 1: Write the failing tests** + +Create `tests/test_eval_embedder_bge.py`: + +```python +"""Tests for BgeEmbedder — a Chroma EmbeddingFunction adapter for BAAI/bge-small-en-v1.5.""" + +from __future__ import annotations + +import pytest + + +@pytest.fixture(scope="module") +def embedder(): + """Module-scoped to amortize the model-load cost across tests.""" + from src.eval.embedders import BgeEmbedder + return BgeEmbedder() + + +def test_returns_384_dim_vectors(embedder): + out = embedder(["hello world"]) + assert len(out) == 1 + assert len(out[0]) == 384 + assert all(isinstance(x, float) for x in out[0]) + + +def test_synonyms_closer_than_unrelated(embedder): + """Sanity check that the right model is loaded — not a stub.""" + import numpy as np + a, b, c = embedder(["cat", "feline", "airplane"]) + a, b, c = np.array(a), np.array(b), np.array(c) + cos = lambda u, v: float(u @ v / (np.linalg.norm(u) * np.linalg.norm(v))) + assert cos(a, b) > cos(a, c), "BGE should rank cat~feline > cat~airplane" + + +def test_chroma_collection_uses_embedder(embedder): + """End-to-end: a Chroma collection created with BgeEmbedder retrieves the right doc.""" + import chromadb + client = chromadb.EphemeralClient() + coll = client.get_or_create_collection( + name="test_bge_e2e", + embedding_function=embedder, + metadata={"hnsw:space": "cosine"}, + ) + coll.upsert( + ids=["d1", "d2", "d3"], + documents=[ + "Cats are small carnivorous mammals often kept as pets.", + "Airplanes are powered flying vehicles with fixed wings.", + "Dogs are domesticated descendants of wolves.", + ], + ) + res = coll.query(query_texts=["What is a feline?"], n_results=1) + assert res["ids"][0][0] == "d1" +``` + +- [ ] **Step 2: Run tests to verify they fail** + +```bash +pytest tests/test_eval_embedder_bge.py -v +``` +Expected: FAIL — `ImportError: cannot import name 'BgeEmbedder'`. + +- [ ] **Step 3: Implement `BgeEmbedder`** + +Create `src/eval/embedders/bge_small.py`: + +```python +"""BgeEmbedder — Chroma EmbeddingFunction adapter for BAAI/bge-small-en-v1.5. + +Pipeline position: + Document → Chunks → [BgeEmbedder] → Vectors (384-dim) → ChromaDB + +Phase 2 lever 2b. The factory installs this on the Chroma collection at +creation time; ChromaVectorStore.upsert/query then auto-embeds via this +function with no per-call code change. + +Why bge-small-en-v1.5: + - 384 dim — same as ChromaDB's default ONNX MiniLM, so dimension-comparable. + - Strong on MTEB retrieval benchmarks (top-tier 33M-param model). + - Loadable via `sentence-transformers`, which is already a project dep. +""" + +from __future__ import annotations + +from typing import Sequence + +from chromadb.api.types import Documents, EmbeddingFunction, Embeddings + + +class BgeEmbedder(EmbeddingFunction[Documents]): + """Chroma-compatible embedding function backed by sentence-transformers. + + Caches the SentenceTransformer model on the instance to avoid re-loading + on every call. Each instance is safe to share across one collection. + """ + + MODEL_NAME = "BAAI/bge-small-en-v1.5" + + def __init__(self) -> None: + # WHY lazy import: sentence-transformers is heavy. Only import when an + # instance is created so module import remains cheap for tests that + # never construct one. + from sentence_transformers import SentenceTransformer + + self._model = SentenceTransformer(self.MODEL_NAME) + + def __call__(self, input: Documents) -> Embeddings: + """Encode a batch of documents into 384-dim vectors. + + Args: + input: List of strings to embed. + + Returns: + List of 384-element float lists, one per input document. + """ + # WHY tolist(): sentence-transformers returns a numpy array; Chroma + # expects a plain list[list[float]] for serialization. + vectors = self._model.encode(list(input), normalize_embeddings=True) + return vectors.tolist() + + @staticmethod + def name() -> str: + """Required by Chroma >= 0.4.x for embedding-function identification.""" + return "bge_small_en_v1_5" +``` + +Create `src/eval/embedders/__init__.py`: + +```python +"""Phase 2 embedder package — pluggable Chroma EmbeddingFunction adapters.""" + +from src.eval.embedders.bge_small import BgeEmbedder + +__all__ = ["BgeEmbedder"] +``` + +- [ ] **Step 4: Run tests to verify they pass** + +```bash +pytest tests/test_eval_embedder_bge.py -v +``` +Expected: all PASS. First run downloads ~120MB of model weights; subsequent runs use the local HF cache. + +- [ ] **Step 5: Commit** + +```bash +git add src/eval/embedders/__init__.py src/eval/embedders/bge_small.py \ + tests/test_eval_embedder_bge.py +git commit -m "feat(eval): add BgeEmbedder as Chroma EmbeddingFunction adapter" +``` + +--- + +## Task 4: `BM25HybridRetriever` — RRF fusion of sparse + dense + +**Files:** +- Create: `src/eval/retrievers/__init__.py` +- Create: `src/eval/retrievers/bm25_hybrid.py` +- Create: `tests/test_eval_retriever_bm25_hybrid.py` +- Modify: `requirements.txt` (+ `rank-bm25`) + +- [ ] **Step 1: Add the dep** + +Append to `requirements.txt`: + +``` +rank-bm25==0.2.2 +``` + +Install locally: + +```bash +pip install rank-bm25==0.2.2 +``` + +- [ ] **Step 2: Write the failing tests** + +Create `tests/test_eval_retriever_bm25_hybrid.py`: + +```python +"""Tests for BM25HybridRetriever — RRF fusion of BM25 (sparse) + dense (Chroma).""" + +from __future__ import annotations + +from src.vector_store import SearchResult + + +def _sr(chunk_id: str, content: str, score: float) -> SearchResult: + return SearchResult(chunk_id=chunk_id, content=content, score=score, metadata={}) + + +def test_rrf_fusion_asymmetric_inputs(): + """RRF on A=[a,b,c,d], B=[d,a] with rrf_k=60 yields fused order a, d, b, c.""" + from src.eval.retrievers.bm25_hybrid import reciprocal_rank_fusion + A = ["a", "b", "c", "d"] + B = ["d", "a"] + fused = reciprocal_rank_fusion([A, B], rrf_k=60) + assert fused == ["a", "d", "b", "c"] + + +def test_hybrid_retrieve_returns_top_k(): + """End-to-end: hybrid retriever combines BM25 and Chroma results into top-K.""" + import chromadb + from src.eval.retrievers.bm25_hybrid import BM25HybridRetriever + from src.vector_store import ChromaVectorStore + + client = chromadb.EphemeralClient() + coll = client.get_or_create_collection( + name="test_hybrid", metadata={"hnsw:space": "cosine"}, + ) + coll.upsert( + ids=["d1", "d2", "d3", "d4"], + documents=[ + "Cats are small carnivorous mammals often kept as pets.", + "Reciprocal rank fusion is a standard sparse-dense combination.", + "Hybrid search blends BM25 and dense retrieval signals.", + "Airplanes have fixed wings.", + ], + ) + vs = ChromaVectorStore(collection=coll) + retriever = BM25HybridRetriever( + vector_store=vs, + documents={"d1": coll.get(ids=["d1"])["documents"][0], + "d2": coll.get(ids=["d2"])["documents"][0], + "d3": coll.get(ids=["d3"])["documents"][0], + "d4": coll.get(ids=["d4"])["documents"][0]}, + bm25_top_k=3, + dense_top_k=3, + rrf_k=60, + ) + out = retriever.retrieve("hybrid sparse dense fusion", top_k=2) + assert len(out) == 2 + assert all(isinstance(r, SearchResult) for r in out) + # Top result should be one of d2 or d3 (both directly relevant). + assert out[0].chunk_id in {"d2", "d3"} +``` + +- [ ] **Step 3: Run tests to verify they fail** + +```bash +pytest tests/test_eval_retriever_bm25_hybrid.py -v +``` +Expected: FAIL — `ImportError`. + +- [ ] **Step 4: Implement `BM25HybridRetriever`** + +Create `src/eval/retrievers/bm25_hybrid.py`: + +```python +"""BM25HybridRetriever — Reciprocal Rank Fusion of sparse (BM25) + dense (Chroma) retrieval. + +Pipeline position: + query → [BM25 + Dense → RRF] → top-K SearchResult → Reranker / Generator + +Phase 2 lever 2c. The retriever keeps two parallel ranked lists (BM25 over +documents, dense over Chroma vectors), then fuses them with RRF: + + score(d) = sum over r in {dense, sparse} of 1 / (rrf_k + rank_r(d)) + +Why RRF over weighted-sum: RRF is parameter-light (one constant), robust to +score-scale differences across the two retrievers, and the literature shows +it consistently matches or beats tuned weighted-sum on benchmarks like BEIR. +""" + +from __future__ import annotations + +from typing import Sequence + +from rank_bm25 import BM25Okapi + +from src.vector_store import ChromaVectorStore, SearchResult + + +def reciprocal_rank_fusion( + rankings: Sequence[Sequence[str]], + rrf_k: int = 60, +) -> list[str]: + """Fuse multiple ranked ID lists into one via Reciprocal Rank Fusion. + + Args: + rankings: Iterable of ranked ID sequences. Each sequence is one + retriever's ranking, most-relevant first. + rrf_k: RRF constant (60 is the textbook default; smaller emphasizes + top-rank items more, larger flattens contributions). + + Returns: + Fused ranking, IDs ordered by descending fused score. + """ + scores: dict[str, float] = {} + for ranking in rankings: + for rank, item_id in enumerate(ranking, start=1): + scores[item_id] = scores.get(item_id, 0.0) + 1.0 / (rrf_k + rank) + return sorted(scores.keys(), key=lambda i: scores[i], reverse=True) + + +class BM25HybridRetriever: + """Retriever that fuses BM25 and dense Chroma rankings. + + The BM25 index is built once at construction time over a `documents` mapping. + Each retrieve() call queries both BM25 and the vector store, then RRF-fuses + the two rankings before truncating to the requested top-K. + """ + + def __init__( + self, + vector_store: ChromaVectorStore, + documents: dict[str, str], + bm25_top_k: int = 20, + dense_top_k: int = 20, + rrf_k: int = 60, + ) -> None: + """Build the BM25 index and store retrieval parameters. + + Args: + vector_store: Dense retriever (Chroma collection wrapper). + documents: Mapping of chunk_id → raw document text. BM25 needs + tokenized text; this dict is the authoritative corpus. + bm25_top_k: Number of candidates BM25 returns per query. + dense_top_k: Number of candidates the dense retriever returns. + rrf_k: RRF fusion constant. + """ + self._vector_store = vector_store + self._chunk_ids = list(documents.keys()) + # WHY simple split: rank-bm25 expects pre-tokenized inputs. A whitespace + # split is good enough for English RAG corpora; nltk stems/stopwords + # would help marginally but add a runtime dep we don't want here. + tokenized = [documents[i].lower().split() for i in self._chunk_ids] + self._bm25 = BM25Okapi(tokenized) + self._documents = documents + self._bm25_top_k = bm25_top_k + self._dense_top_k = dense_top_k + self._rrf_k = rrf_k + + def retrieve(self, query: str, top_k: int = 5) -> list[SearchResult]: + """Run BM25 + dense in parallel, RRF-fuse, return top-K SearchResults. + + Args: + query: Natural-language query. + top_k: Number of fused results to return. + + Returns: + Top-K SearchResult ordered by fused score descending. Score on each + result is the dense similarity (BM25 ranks aren't directly comparable; + keeping dense score lets downstream rerankers/refusal-handlers reuse + it as a confidence proxy). + """ + # --- Sparse side ------------------------------------------------------- + sparse_scores = self._bm25.get_scores(query.lower().split()) + sparse_ranked = sorted( + range(len(self._chunk_ids)), + key=lambda i: sparse_scores[i], + reverse=True, + )[: self._bm25_top_k] + sparse_ids = [self._chunk_ids[i] for i in sparse_ranked] + + # --- Dense side -------------------------------------------------------- + dense_results = self._vector_store.query( + query_text=query, top_k=self._dense_top_k, + ) + dense_ids = [r.chunk_id for r in dense_results] + dense_score_by_id = {r.chunk_id: r.score for r in dense_results} + + # --- Fusion ------------------------------------------------------------ + fused_ids = reciprocal_rank_fusion( + [sparse_ids, dense_ids], rrf_k=self._rrf_k, + )[:top_k] + + return [ + SearchResult( + chunk_id=cid, + content=self._documents[cid], + score=dense_score_by_id.get(cid, 0.0), + metadata={}, + ) + for cid in fused_ids + ] +``` + +Create `src/eval/retrievers/__init__.py`: + +```python +"""Phase 2 retriever package — hybrid sparse/dense retrieval and reranking.""" + +from src.eval.retrievers.bm25_hybrid import BM25HybridRetriever + +__all__ = ["BM25HybridRetriever"] +``` + +- [ ] **Step 5: Run tests to verify they pass** + +```bash +pytest tests/test_eval_retriever_bm25_hybrid.py -v +``` +Expected: PASS. + +- [ ] **Step 6: Commit** + +```bash +git add requirements.txt src/eval/retrievers/__init__.py \ + src/eval/retrievers/bm25_hybrid.py tests/test_eval_retriever_bm25_hybrid.py +git commit -m "feat(eval): add BM25HybridRetriever with RRF fusion" +``` + +--- + +## Task 5: `CrossEncoderReranker` — ms-marco-MiniLM rerank top-N → top-K + +**Files:** +- Create: `src/eval/retrievers/reranker.py` +- Modify: `src/eval/retrievers/__init__.py` +- Create: `tests/test_eval_retriever_reranker.py` + +- [ ] **Step 1: Write the failing test** + +Create `tests/test_eval_retriever_reranker.py`: + +```python +"""Tests for CrossEncoderReranker — re-scores candidates with a cross-encoder model.""" + +from __future__ import annotations + +import pytest + +from src.vector_store import SearchResult + + +@pytest.fixture(scope="module") +def reranker(): + from src.eval.retrievers.reranker import CrossEncoderReranker + return CrossEncoderReranker() + + +def test_obvious_match_ranks_first(reranker): + """Given five candidates with one obviously-relevant doc, it ranks first after rerank.""" + candidates = [ + SearchResult(chunk_id="d1", content="Pyramids of Giza were built around 2500 BC.", + score=0.5, metadata={}), + SearchResult(chunk_id="d2", content="Cats are small carnivorous mammals.", + score=0.6, metadata={}), + SearchResult(chunk_id="d3", content="What is the capital of France? Paris is the capital.", + score=0.4, metadata={}), + SearchResult(chunk_id="d4", content="Airplanes have fixed wings.", + score=0.3, metadata={}), + SearchResult(chunk_id="d5", content="Dogs are domesticated.", score=0.2, metadata={}), + ] + out = reranker.rerank("What is the capital of France?", candidates, final_top_k=3) + assert len(out) == 3 + assert out[0].chunk_id == "d3" + + +def test_rerank_preserves_search_result_shape(reranker): + candidates = [ + SearchResult(chunk_id="d1", content="hello", score=0.5, metadata={"k": "v"}), + SearchResult(chunk_id="d2", content="world", score=0.4, metadata={}), + ] + out = reranker.rerank("greeting", candidates, final_top_k=2) + assert all(isinstance(r, SearchResult) for r in out) + # Original score and metadata should round-trip. + found = {r.chunk_id: r for r in out} + assert found["d1"].metadata == {"k": "v"} +``` + +- [ ] **Step 2: Run tests to verify they fail** + +```bash +pytest tests/test_eval_retriever_reranker.py -v +``` +Expected: FAIL — `ImportError`. + +- [ ] **Step 3: Implement `CrossEncoderReranker`** + +Create `src/eval/retrievers/reranker.py`: + +```python +"""CrossEncoderReranker — re-scores retrieval candidates with a cross-encoder model. + +Pipeline position: + Retriever top-N → [CrossEncoderReranker] → top-K → Refusal / Generator + +Phase 2 lever 2d. Cross-encoders (single-tower models that consume both +the query and a candidate together) typically outperform bi-encoder retrieval +in precision at the cost of latency. We use ms-marco-MiniLM-L-6-v2 — small +enough to run on CPU in milliseconds per pair, trained on MS MARCO so the +ranking signal transfers well to general-domain QA. +""" + +from __future__ import annotations + +from src.vector_store import SearchResult + + +class CrossEncoderReranker: + """Wraps sentence-transformers CrossEncoder to re-score retrieval candidates.""" + + MODEL_NAME = "cross-encoder/ms-marco-MiniLM-L-6-v2" + + def __init__(self) -> None: + from sentence_transformers import CrossEncoder + + self._model = CrossEncoder(self.MODEL_NAME) + + def rerank( + self, + query: str, + candidates: list[SearchResult], + final_top_k: int, + ) -> list[SearchResult]: + """Re-score candidates against the query and return top-K reranked. + + Args: + query: Original user query. + candidates: Pre-retrieved chunks (typically top-N from a base retriever). + final_top_k: How many to keep after reranking. + + Returns: + Top-K SearchResult ordered by descending cross-encoder score. The + original `score` field is *replaced* with the cross-encoder score so + downstream consumers reading `result.score` get the more precise signal. + """ + if not candidates: + return [] + pairs = [(query, c.content) for c in candidates] + scores = self._model.predict(pairs) + # Pair each candidate with its new score, sort, truncate. + scored = sorted( + zip(candidates, scores), key=lambda t: t[1], reverse=True, + )[:final_top_k] + return [ + SearchResult( + chunk_id=c.chunk_id, + content=c.content, + score=float(s), + metadata=c.metadata, + ) + for c, s in scored + ] +``` + +Update `src/eval/retrievers/__init__.py`: + +```python +"""Phase 2 retriever package — hybrid sparse/dense retrieval and reranking.""" + +from src.eval.retrievers.bm25_hybrid import BM25HybridRetriever +from src.eval.retrievers.reranker import CrossEncoderReranker + +__all__ = ["BM25HybridRetriever", "CrossEncoderReranker"] +``` + +- [ ] **Step 4: Run tests to verify they pass** + +```bash +pytest tests/test_eval_retriever_reranker.py -v +``` +Expected: PASS. First call downloads ~80MB of model weights. + +- [ ] **Step 5: Commit** + +```bash +git add src/eval/retrievers/reranker.py src/eval/retrievers/__init__.py \ + tests/test_eval_retriever_reranker.py +git commit -m "feat(eval): add CrossEncoderReranker (ms-marco-MiniLM)" +``` + +--- + +## Task 6: `QueryRewriter` — LLM-based query expansion (with cost capture) + +**Files:** +- Create: `src/eval/transforms/__init__.py` +- Create: `src/eval/transforms/query_rewriter.py` +- Create: `tests/test_eval_transform_rewriter.py` + +- [ ] **Step 1: Write the failing tests** + +Create `tests/test_eval_transform_rewriter.py`: + +```python +"""Tests for QueryRewriter — LLM query expansion with cost capture.""" + +from __future__ import annotations + +from typing import Any + + +class _StubLLM: + """Records calls and returns canned responses + token counts.""" + + def __init__(self, response: str, prompt_tokens: int = 50, completion_tokens: int = 30): + self._response = response + self._prompt_tokens = prompt_tokens + self._completion_tokens = completion_tokens + self.calls: list[tuple[str, str | None]] = [] + + def generate_with_usage(self, prompt: str, system_prompt: str | None = None + ) -> tuple[str, int, int]: + self.calls.append((prompt, system_prompt)) + return self._response, self._prompt_tokens, self._completion_tokens + + +def test_no_model_passthrough(): + """When model is None, expand returns [query] unchanged with zero cost.""" + from src.eval.transforms import QueryRewriter + rw = QueryRewriter(model=None, max_expansions=3, llm=None) + queries, cost, p_t, c_t = rw.expand("What is RAG?") + assert queries == ["What is RAG?"] + assert cost == 0.0 + assert p_t == 0 + assert c_t == 0 + + +def test_expansion_returns_dedup_list_and_cost(): + """With a real model name and stub LLM, expand returns deduped expansions + cost.""" + from src.eval.transforms import QueryRewriter + stub = _StubLLM( + response='["What does RAG stand for?", "Define retrieval augmented generation", ' + '"What is RAG?"]', + prompt_tokens=80, completion_tokens=40, + ) + rw = QueryRewriter(model="gpt-4.1-nano", max_expansions=3, llm=stub) + queries, cost, p_t, c_t = rw.expand("What is RAG?") + # Original query is always first; duplicate dropped; max_expansions=3 cap respected. + assert queries[0] == "What is RAG?" + assert "What does RAG stand for?" in queries + assert "Define retrieval augmented generation" in queries + assert len(queries) == len(set(queries)) # no duplicates + assert len(queries) <= 4 # original + at most max_expansions + # Cost was computed from the stub's token counts at gpt-4.1-nano price. + assert cost > 0.0 + assert p_t == 80 + assert c_t == 40 + + +def test_malformed_llm_response_falls_back_to_passthrough(): + """If the LLM returns non-JSON, expand returns [query] and logs a warning.""" + from src.eval.transforms import QueryRewriter + stub = _StubLLM(response="not json at all", prompt_tokens=50, completion_tokens=10) + rw = QueryRewriter(model="gpt-4.1-nano", max_expansions=3, llm=stub) + queries, cost, _, _ = rw.expand("What is RAG?") + assert queries == ["What is RAG?"] + # Cost is still charged because the call did happen. + assert cost > 0.0 +``` + +- [ ] **Step 2: Run tests to verify they fail** + +```bash +pytest tests/test_eval_transform_rewriter.py -v +``` +Expected: FAIL — `ImportError`. + +- [ ] **Step 3: Implement `QueryRewriter`** + +Create `src/eval/transforms/query_rewriter.py`: + +```python +"""QueryRewriter — LLM-based query expansion with token/cost capture. + +Pipeline position: + user query → [QueryRewriter] → {q, q', q''} → Retriever → ... + +Phase 2 lever 2e. Expansion gives the retriever multiple lexical/semantic +formulations of the same intent, which raises recall on questions where the +original phrasing diverges from the corpus phrasing. We use a tiny model +(gpt-4.1-nano) because the task is cheap and we don't want this lever to +dominate the cost ledger. +""" + +from __future__ import annotations + +import json +import logging +import re +from typing import Protocol + +from src.eval import pricing + +logger = logging.getLogger(__name__) + + +class _LLMHandler(Protocol): + """Structural type for any object exposing generate_with_usage.""" + def generate_with_usage( + self, prompt: str, system_prompt: str | None = None, + ) -> tuple[str, int, int]: ... + + +class QueryRewriter: + """Expands one user query into up to N alternative phrasings via an LLM.""" + + SYSTEM_PROMPT = ( + "You rewrite user search queries into alternative phrasings that preserve " + "the original intent but vary surface form. Respond ONLY with a JSON " + "array of strings — no prose, no code fences." + ) + + def __init__( + self, + model: str | None, + max_expansions: int, + llm: _LLMHandler | None, + ) -> None: + """Configure the rewriter. + + Args: + model: LLM model name. None disables rewriting (pass-through). + max_expansions: Cap on the number of alternative phrasings to return. + llm: Object exposing generate_with_usage(prompt, system_prompt). Required + if model is not None. + """ + self._model = model + self._max_expansions = max_expansions + self._llm = llm + + def expand(self, query: str) -> tuple[list[str], float, int, int]: + """Expand `query` into up to N+1 unique phrasings. + + Returns: + (queries, cost_usd, prompt_tokens, completion_tokens). The original + query is always the first element. When `model is None`, returns + ([query], 0.0, 0, 0) and skips the LLM call. + """ + if self._model is None: + return [query], 0.0, 0, 0 + if self._llm is None: + raise ValueError("QueryRewriter has model set but no llm handler provided.") + + user_prompt = ( + f'Original query: "{query}"\n\n' + f"Return a JSON array of up to {self._max_expansions} alternative " + f"phrasings of this query. Do NOT include the original." + ) + raw, p_t, c_t = self._llm.generate_with_usage( + user_prompt, system_prompt=self.SYSTEM_PROMPT, + ) + cost = pricing.cost_usd(self._model, p_t, c_t) + + expansions = self._parse_expansions(raw) + # Always lead with original; dedupe; cap at original + max_expansions. + ordered: list[str] = [query] + for alt in expansions: + if alt and alt not in ordered: + ordered.append(alt) + if len(ordered) >= self._max_expansions + 1: + break + return ordered, cost, p_t, c_t + + @staticmethod + def _parse_expansions(raw: str) -> list[str]: + """Strip code fences and parse the JSON array; return [] on failure.""" + stripped = re.sub(r"^```(?:json)?\s*", "", raw.strip()) + stripped = re.sub(r"\s*```$", "", stripped).strip() + try: + parsed = json.loads(stripped) + except json.JSONDecodeError: + logger.warning("QueryRewriter got non-JSON response — falling back to [query] only.") + return [] + if not isinstance(parsed, list): + return [] + return [str(item) for item in parsed if isinstance(item, str)] +``` + +Create `src/eval/transforms/__init__.py`: + +```python +"""Phase 2 transforms — pre/post pipeline hooks (rewriter, refusal handler).""" + +from src.eval.transforms.query_rewriter import QueryRewriter + +__all__ = ["QueryRewriter"] +``` + +- [ ] **Step 4: Run tests to verify they pass** + +```bash +pytest tests/test_eval_transform_rewriter.py -v +``` +Expected: PASS. + +- [ ] **Step 5: Commit** + +```bash +git add src/eval/transforms/__init__.py src/eval/transforms/query_rewriter.py \ + tests/test_eval_transform_rewriter.py +git commit -m "feat(eval): add QueryRewriter for LLM-based expansion with cost capture" +``` + +--- + +## Task 7: `RefusalHandler` — similarity gate for unanswerable questions + +**Files:** +- Create: `src/eval/transforms/refusal_handler.py` +- Modify: `src/eval/transforms/__init__.py` +- Create: `tests/test_eval_transform_refusal.py` + +- [ ] **Step 1: Write the failing tests** + +Create `tests/test_eval_transform_refusal.py`: + +```python +"""Tests for RefusalHandler — pure-logic similarity gate.""" + +from __future__ import annotations + +from src.vector_store import SearchResult + + +def _sr(score: float, chunk_id: str = "d1") -> SearchResult: + return SearchResult(chunk_id=chunk_id, content="x", score=score, metadata={}) + + +def test_refuses_when_top1_below_threshold(): + from src.eval.transforms import RefusalHandler + h = RefusalHandler(enabled=True, similarity_threshold=0.35, + no_answer_text="I don't know.") + assert h.should_refuse([_sr(0.20), _sr(0.10)]) is True + + +def test_does_not_refuse_when_top1_above_threshold(): + from src.eval.transforms import RefusalHandler + h = RefusalHandler(enabled=True, similarity_threshold=0.35, + no_answer_text="I don't know.") + assert h.should_refuse([_sr(0.50), _sr(0.10)]) is False + + +def test_refuses_on_empty_candidates(): + from src.eval.transforms import RefusalHandler + h = RefusalHandler(enabled=True, similarity_threshold=0.35, + no_answer_text="I don't know.") + assert h.should_refuse([]) is True + + +def test_disabled_handler_never_refuses(): + from src.eval.transforms import RefusalHandler + h = RefusalHandler(enabled=False, similarity_threshold=0.35, + no_answer_text="I don't know.") + assert h.should_refuse([_sr(0.0)]) is False + assert h.should_refuse([]) is False + + +def test_refuse_response_returns_text_and_no_chunks(): + from src.eval.transforms import RefusalHandler + h = RefusalHandler(enabled=True, similarity_threshold=0.35, + no_answer_text="I cannot answer.") + chunks, answer = h.refuse_response() + assert chunks == [] + assert answer == "I cannot answer." +``` + +- [ ] **Step 2: Run tests to verify they fail** + +```bash +pytest tests/test_eval_transform_refusal.py -v +``` +Expected: FAIL — `ImportError`. + +- [ ] **Step 3: Implement `RefusalHandler`** + +Create `src/eval/transforms/refusal_handler.py`: + +```python +"""RefusalHandler — answerability gate based on top-1 retrieval similarity. + +Pipeline position: + Retriever (post-rerank) candidates → [RefusalHandler] → answer or refusal text + +Phase 2 lever 2g. SQuAD v2 includes 'unanswerable' questions whose gold +answer is the empty string. Phase 1's pipeline always tries to answer, +which means it scores poorly on `refusal_correctness`. RefusalHandler is +a deterministic short-circuit: when no candidate clears the similarity +threshold, return a fixed no-answer text instead of calling the LLM. +""" + +from __future__ import annotations + +from src.vector_store import SearchResult + + +class RefusalHandler: + """Deterministic answerability gate driven by top-1 similarity score.""" + + def __init__( + self, + enabled: bool, + similarity_threshold: float, + no_answer_text: str, + ) -> None: + """Configure the gate. + + Args: + enabled: When False, should_refuse always returns False. + similarity_threshold: Top-1 score must be >= this to NOT refuse. + no_answer_text: Text returned in place of an LLM answer on refusal. + """ + self._enabled = enabled + self._threshold = similarity_threshold + self._no_answer_text = no_answer_text + + def should_refuse(self, candidates: list[SearchResult]) -> bool: + """Return True if the pipeline should short-circuit to no-answer text.""" + if not self._enabled: + return False + if not candidates: + return True + return candidates[0].score < self._threshold + + def refuse_response(self) -> tuple[list[SearchResult], str]: + """Return ([], no_answer_text) — used when should_refuse is True.""" + return [], self._no_answer_text +``` + +Update `src/eval/transforms/__init__.py`: + +```python +"""Phase 2 transforms — pre/post pipeline hooks (rewriter, refusal handler).""" + +from src.eval.transforms.query_rewriter import QueryRewriter +from src.eval.transforms.refusal_handler import RefusalHandler + +__all__ = ["QueryRewriter", "RefusalHandler"] +``` + +- [ ] **Step 4: Run tests to verify they pass** + +```bash +pytest tests/test_eval_transform_refusal.py -v +``` +Expected: PASS. + +- [ ] **Step 5: Commit** + +```bash +git add src/eval/transforms/refusal_handler.py src/eval/transforms/__init__.py \ + tests/test_eval_transform_refusal.py +git commit -m "feat(eval): add RefusalHandler with similarity gate" +``` + +--- + +## Task 8: Wire all Phase 2 levers into `build_pipeline` + `EvalPipeline.query` + +**Files:** +- Modify: `src/eval/pipeline_factory.py` +- Create: `tests/test_eval_pipeline_factory_phase2.py` +- Modify: `tests/test_eval_pipeline_factory.py` (existing — verify still passes) +- Modify: `tests/test_eval_smoke.py` (existing — verify still passes) +- Create: `tests/fixtures/phase2_corpus/d1.txt`, `d2.txt`, `d3.txt` + +- [ ] **Step 1: Add fixture corpus** + +Create `tests/fixtures/phase2_corpus/d1.txt`: + +``` +Reciprocal rank fusion combines two ranked lists by summing 1/(k+rank) for each item. +``` + +Create `tests/fixtures/phase2_corpus/d2.txt`: + +``` +Cross-encoders score query-document pairs jointly and improve retrieval precision. +``` + +Create `tests/fixtures/phase2_corpus/d3.txt`: + +``` +Airplanes have fixed wings and powered engines. +``` + +- [ ] **Step 2: Write the failing factory tests** + +Create `tests/test_eval_pipeline_factory_phase2.py`: + +```python +"""Phase 2 factory tests — every tier YAML produces a pipeline whose attributes match.""" + +from __future__ import annotations + +from pathlib import Path + +import pytest + +from src.eval.config import load_config +from src.eval.pipeline_factory import build_pipeline + + +PHASE2_DIR = Path("configs/eval/phase2") + + +@pytest.fixture +def stub_llm(): + class _S: + def generate(self, prompt, system_prompt=None): + return "stub answer" + def generate_with_usage(self, prompt, system_prompt=None): + return "stub answer", 10, 5 + return _S() + + +@pytest.mark.parametrize("yaml_name,expects", [ + ("phase2_baseline.yaml", {"rewriter": False, "reranker": False, "refusal": False, "hybrid": False, "embedder": "chroma_default"}), + ("phase2b_embedder.yaml", {"rewriter": False, "reranker": False, "refusal": False, "hybrid": False, "embedder": "bge_small_en_v1_5"}), + ("phase2c_hybrid.yaml", {"rewriter": False, "reranker": False, "refusal": False, "hybrid": True, "embedder": "bge_small_en_v1_5"}), + ("phase2d_rerank.yaml", {"rewriter": False, "reranker": True, "refusal": False, "hybrid": True, "embedder": "bge_small_en_v1_5"}), + ("phase2e_rewrite.yaml", {"rewriter": True, "reranker": True, "refusal": False, "hybrid": True, "embedder": "bge_small_en_v1_5"}), + ("phase2g_refusal.yaml", {"rewriter": True, "reranker": True, "refusal": True, "hybrid": True, "embedder": "bge_small_en_v1_5"}), +]) +def test_phase2_yaml_builds_pipeline_with_expected_attrs(yaml_name, expects, stub_llm): + cfg = load_config(PHASE2_DIR / yaml_name) + pipeline = build_pipeline( + cfg, dataset_name="squad_v2_dev_200", + llm_override=stub_llm, judge_llm_override=stub_llm, + ) + try: + assert (pipeline.rewriter is not None) == expects["rewriter"] + assert (pipeline.reranker is not None) == expects["reranker"] + assert (pipeline.refusal_handler is not None) == expects["refusal"] + assert (pipeline.hybrid_retriever is not None) == expects["hybrid"] + assert cfg.pipeline.embedder.name == expects["embedder"] + finally: + pipeline.teardown() + + +def test_phase2_query_with_refusal_short_circuits(stub_llm, tmp_path): + """End-to-end smoke: refusal handler short-circuits when top-1 < threshold.""" + cfg = load_config(PHASE2_DIR / "phase2g_refusal.yaml") + pipeline = build_pipeline( + cfg, dataset_name="squad_v2_dev_200", + llm_override=stub_llm, judge_llm_override=stub_llm, + ) + try: + # Empty index → top-1 score is 0 → handler refuses. + chunks, answer, telemetry = pipeline.query("what is x?") + assert chunks == [] + assert answer == cfg.pipeline.refusal_handler.no_answer_text + assert "refusal_check" in telemetry["timings_ms"] + finally: + pipeline.teardown() +``` + +(Note: this test references YAMLs created in Task 9. Run the parametrized test only after Task 9 lands, OR write the YAMLs as part of this task. Per the spec, configs land in commit 10; for TDD we'll author them inline here as fixtures and move them to `configs/eval/phase2/` in commit 10. To keep the commit graph clean, this task creates the YAMLs in `tests/fixtures/phase2_configs/` and Task 9 moves them.) + +Adjust `PHASE2_DIR` in the test to `Path("tests/fixtures/phase2_configs")` and create those YAMLs as fixtures here. Task 9 then moves them. + +Create `tests/fixtures/phase2_configs/phase2_baseline.yaml`: + +```yaml +name: "phase2_baseline" +description: "Baseline anchor for the Phase 2 matrix — identical to baseline_squad_only." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + retriever: {top_k: 5} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 +``` + +Create `tests/fixtures/phase2_configs/phase2b_embedder.yaml`: + +```yaml +name: "phase2b_embedder" +description: "Phase 2 tier 2b — swap default embedder for BAAI/bge-small-en-v1.5." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + embedder: {name: bge_small_en_v1_5} + retriever: {top_k: 5} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 +``` + +Create `tests/fixtures/phase2_configs/phase2c_hybrid.yaml`: + +```yaml +name: "phase2c_hybrid" +description: "Phase 2 tier 2c — add BM25 hybrid on top of BGE." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + embedder: {name: bge_small_en_v1_5} + retriever: {top_k: 5} + hybrid: {enabled: true, bm25_top_k: 20, dense_top_k: 20, rrf_k: 60} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 +``` + +Create `tests/fixtures/phase2_configs/phase2d_rerank.yaml`: + +```yaml +name: "phase2d_rerank" +description: "Phase 2 tier 2d — add cross-encoder rerank on top of hybrid." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + embedder: {name: bge_small_en_v1_5} + retriever: {top_k: 5} + hybrid: {enabled: true, bm25_top_k: 20, dense_top_k: 20, rrf_k: 60} + reranker: {model: ms_marco_minilm_l6_v2, rerank_top_n: 20, final_top_k: 5} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 +``` + +Create `tests/fixtures/phase2_configs/phase2e_rewrite.yaml`: + +```yaml +name: "phase2e_rewrite" +description: "Phase 2 tier 2e — add LLM query rewriting on top of rerank." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + embedder: {name: bge_small_en_v1_5} + retriever: {top_k: 5} + hybrid: {enabled: true, bm25_top_k: 20, dense_top_k: 20, rrf_k: 60} + reranker: {model: ms_marco_minilm_l6_v2, rerank_top_n: 20, final_top_k: 5} + query_rewriter: {model: gpt-4.1-nano, max_expansions: 3} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 +``` + +Create `tests/fixtures/phase2_configs/phase2g_refusal.yaml`: + +```yaml +name: "phase2g_refusal" +description: "Phase 2 tier 2g — add refusal handler on top of the full upgraded retrieval stack." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + embedder: {name: bge_small_en_v1_5} + retriever: {top_k: 5} + hybrid: {enabled: true, bm25_top_k: 20, dense_top_k: 20, rrf_k: 60} + reranker: {model: ms_marco_minilm_l6_v2, rerank_top_n: 20, final_top_k: 5} + query_rewriter: {model: gpt-4.1-nano, max_expansions: 3} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} + refusal_handler: {enabled: true, similarity_threshold: 0.35, + no_answer_text: "I don't have enough information to answer that."} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 +``` + +- [ ] **Step 3: Run tests to verify they fail** + +```bash +pytest tests/test_eval_pipeline_factory_phase2.py -v +``` +Expected: FAIL — `pipeline.rewriter` etc. don't exist; `EvalPipeline` doesn't accept the new fields. + +- [ ] **Step 4: Extend `EvalPipeline`** + +In `src/eval/pipeline_factory.py`, modify the `@dataclass class EvalPipeline` to add the new optional fields after `judge_llm`: + +```python +@dataclass +class EvalPipeline: + # ... existing fields unchanged ... + chunker: TextChunker + vector_store: ChromaVectorStore + llm: object + judge_llm: object + config: EvalConfig + dataset_name: str + # Phase 2 additions — None when the corresponding lever is off. + hybrid_retriever: object | None = None # BM25HybridRetriever or None + reranker: object | None = None # CrossEncoderReranker or None + rewriter: object | None = None # QueryRewriter or None + refusal_handler: object | None = None # RefusalHandler or None + _client: object = field(repr=False, default=None) + _collection_name: str = field(repr=False, default="") +``` + +- [ ] **Step 5: Extend `EvalPipeline.query`** + +Replace the existing `EvalPipeline.query` body with the layered version. Keep the dummy/null cases as no-ops so Phase 1 baselines route through identical control flow. + +```python + def query(self, question: str) -> tuple[list[SearchResult], str, dict]: + """Run the configured pipeline; return (chunks, answer, telemetry). + + Telemetry keys: + timings_ms: subset of {"rewrite", "retrieve", "rerank", "refusal_check", "generate"} + tokens: {"prompt": int, "completion": int} for the generator call + cost_usd: generator-side cost (judge + rewriter accounted separately) + rewriter_cost_usd: rewriter spend (0.0 when disabled) + """ + from src.eval import pricing + timings: dict[str, float] = {} + rewriter_cost = 0.0 + + # ---- Rewrite (lever 2e) ----------------------------------------------- + t = time.perf_counter() + if self.rewriter is not None: + queries, rewriter_cost, _, _ = self.rewriter.expand(question) + else: + queries = [question] + timings["rewrite"] = (time.perf_counter() - t) * 1000.0 + + # ---- Retrieve --------------------------------------------------------- + top_k_initial = ( + self.config.pipeline.reranker.rerank_top_n + if self.reranker is not None else self.config.pipeline.retriever.top_k + ) + t = time.perf_counter() + if self.hybrid_retriever is not None: + # Hybrid: retrieve top-N for each rewritten query, dedup by chunk_id. + seen: dict[str, SearchResult] = {} + for q in queries: + for r in self.hybrid_retriever.retrieve(q, top_k=top_k_initial): + if r.chunk_id not in seen: + seen[r.chunk_id] = r + results = list(seen.values()) + else: + seen = {} + for q in queries: + for r in self.vector_store.query(query_text=q, top_k=top_k_initial): + if r.chunk_id not in seen: + seen[r.chunk_id] = r + results = list(seen.values()) + timings["retrieve"] = (time.perf_counter() - t) * 1000.0 + + # ---- Rerank (lever 2d) ------------------------------------------------ + t = time.perf_counter() + if self.reranker is not None: + results = self.reranker.rerank( + question, results, + final_top_k=self.config.pipeline.reranker.final_top_k, + ) + else: + results = results[: self.config.pipeline.retriever.top_k] + timings["rerank"] = (time.perf_counter() - t) * 1000.0 + + # ---- Refusal gate (lever 2g) ------------------------------------------ + t = time.perf_counter() + if self.refusal_handler is not None and self.refusal_handler.should_refuse(results): + chunks, answer = self.refusal_handler.refuse_response() + timings["refusal_check"] = (time.perf_counter() - t) * 1000.0 + telemetry = { + "timings_ms": timings, + "tokens": {"prompt": 0, "completion": 0}, + "cost_usd": 0.0, + "rewriter_cost_usd": rewriter_cost, + } + return chunks, answer, telemetry + timings["refusal_check"] = (time.perf_counter() - t) * 1000.0 + + # ---- Generate --------------------------------------------------------- + context = "\n\n".join(r.content for r in results) + system_prompt = ( + "You are a helpful assistant. Answer the question based solely on the " + "provided context. If the context does not contain enough information, " + "say so clearly." + ) + user_prompt = f"Context:\n{context}\n\nQuestion: {question}\n\nAnswer:" + full_prompt_text = system_prompt + "\n" + user_prompt + model = self.config.pipeline.generator.model + + t = time.perf_counter() + answer = self.llm.generate(user_prompt, system_prompt=system_prompt) + timings["generate"] = (time.perf_counter() - t) * 1000.0 + + prompt_tokens = count_tokens(full_prompt_text, model) + completion_tokens = count_tokens(answer, model) + cost = pricing.cost_usd(model, prompt_tokens, completion_tokens) + + telemetry = { + "timings_ms": timings, + "tokens": {"prompt": prompt_tokens, "completion": completion_tokens}, + "cost_usd": cost, + "rewriter_cost_usd": rewriter_cost, + } + return results, answer, telemetry +``` + +- [ ] **Step 6: Extend `build_pipeline`** + +After the existing chunker block in `build_pipeline`, replace the embedder/collection setup with: + +```python + # ---- Embedder (lever 2b) ------------------------------------------------- + embedder_cfg = config.pipeline.embedder + embedding_function = _build_embedding_function(embedder_cfg) + + collection_name = f"eval_{config.name}_{dataset_name}_{uuid.uuid4().hex[:6]}" + client = chromadb.EphemeralClient() + collection = client.get_or_create_collection( + name=collection_name, + embedding_function=embedding_function, + metadata={"hnsw:space": "cosine"}, + ) + vector_store = ChromaVectorStore(collection=collection) +``` + +Add the helper builders below `build_pipeline`: + +```python +def _build_embedding_function(cfg) -> object: + if cfg.name == "chroma_default": + from chromadb.utils import embedding_functions + return embedding_functions.DefaultEmbeddingFunction() + if cfg.name == "bge_small_en_v1_5": + from src.eval.embedders import BgeEmbedder + return BgeEmbedder() + raise ValueError(f"Unknown embedder name: {cfg.name}") + + +def _build_hybrid_retriever(cfg, vector_store, documents): + if not cfg.enabled: + return None + from src.eval.retrievers import BM25HybridRetriever + return BM25HybridRetriever( + vector_store=vector_store, documents=documents, + bm25_top_k=cfg.bm25_top_k, dense_top_k=cfg.dense_top_k, rrf_k=cfg.rrf_k, + ) + + +def _build_reranker(cfg): + if cfg.model is None: + return None + from src.eval.retrievers import CrossEncoderReranker + return CrossEncoderReranker() + + +def _build_rewriter(cfg, llm): + if cfg.model is None: + return None + from src.eval.transforms import QueryRewriter + return QueryRewriter(model=cfg.model, max_expansions=cfg.max_expansions, llm=llm) + + +def _build_refusal(cfg): + if not cfg.enabled: + return None + from src.eval.transforms import RefusalHandler + return RefusalHandler( + enabled=True, similarity_threshold=cfg.similarity_threshold, + no_answer_text=cfg.no_answer_text, + ) +``` + +Update the bottom of `build_pipeline` to instantiate the new fields: + +```python + # ---- LLM handlers (existing) --------------------------------------------- + llm = llm_override if llm_override is not None else LLMHandler(config.pipeline.generator.model) + judge_llm = ( + judge_llm_override if judge_llm_override is not None + else LLMHandler(config.eval.judge_model) + ) + # NOTE: hybrid_retriever is built lazily — it needs the {chunk_id: text} map + # which is only available after ingest(). Set hybrid_cfg here; build_pipeline + # caller wires the hybrid retriever inside ingest() once chunks are available. + + return EvalPipeline( + chunker=chunker, + vector_store=vector_store, + llm=llm, + judge_llm=judge_llm, + config=config, + dataset_name=dataset_name, + hybrid_retriever=None, # populated post-ingest in ingest() or by caller + reranker=_build_reranker(config.pipeline.reranker), + rewriter=_build_rewriter(config.pipeline.query_rewriter, llm=llm), + refusal_handler=_build_refusal(config.pipeline.refusal_handler), + _client=client, + _collection_name=collection_name, + ) +``` + +In `EvalPipeline.ingest`, at the end of `_ingest_squad` (and `_ingest_ml_papers`), add: + +```python + # Phase 2: build the hybrid retriever now that chunks/contexts are upserted. + if self.config.pipeline.hybrid.enabled: + from src.eval.pipeline_factory import _build_hybrid_retriever + documents_map = dict(zip(ids, documents)) + self.hybrid_retriever = _build_hybrid_retriever( + self.config.pipeline.hybrid, self.vector_store, documents_map, + ) +``` + +- [ ] **Step 7: Run tests to verify they pass** + +```bash +pytest tests/test_eval_pipeline_factory_phase2.py \ + tests/test_eval_pipeline_factory.py \ + tests/test_eval_smoke.py -v +``` +Expected: PASS. Existing Phase 1 factory tests must still pass (the new fields are all optional and None by default). + +- [ ] **Step 8: Commit** + +```bash +git add src/eval/pipeline_factory.py tests/test_eval_pipeline_factory_phase2.py \ + tests/fixtures/phase2_configs/ tests/fixtures/phase2_corpus/ +git commit -m "feat(eval): wire Phase 2 levers into build_pipeline + EvalPipeline.query" +``` + +--- + +## Task 9: Add `archive` subcommand to `cli` + +**Files:** +- Modify: `src/eval/cli.py` +- Create: `tests/test_eval_cli_archive.py` + +- [ ] **Step 1: Write the failing test** + +Create `tests/test_eval_cli_archive.py`: + +```python +"""Tests for `python -m src.eval.cli archive` — copies small artifacts only.""" + +from __future__ import annotations + +import json +from pathlib import Path + +from src.eval.cli import _cmd_archive + + +def test_archive_copies_four_artifacts(tmp_path): + # Build a fake run dir with the expected artifacts plus an irrelevant large file. + src = tmp_path / "eval_runs" / "fake_run" + src.mkdir(parents=True) + (src / "metrics.json").write_text("[]") + (src / "cost.json").write_text("{}") + (src / "metadata.json").write_text('{"run_id": "fake_run"}') + (src / "config.yaml").write_text("name: fake") + (src / "questions.jsonl").write_text("\n".join(["{}"] * 200)) # large + + dst = tmp_path / "docs" / "phase2" / "runs" / "fake_run" + rc = _cmd_archive(_FakeArgs(run_id="fake_run", to=str(dst), runs_root=str(tmp_path / "eval_runs"))) + assert rc == 0 + assert (dst / "metrics.json").exists() + assert (dst / "cost.json").exists() + assert (dst / "metadata.json").exists() + assert (dst / "config.yaml").exists() + # questions.jsonl is NOT copied (large), but its SHA should be in metadata.json + assert not (dst / "questions.jsonl").exists() + md = json.loads((dst / "metadata.json").read_text()) + assert "questions_jsonl_sha256" in md + + +class _FakeArgs: + def __init__(self, **kwargs): + for k, v in kwargs.items(): + setattr(self, k, v) +``` + +- [ ] **Step 2: Run test to verify it fails** + +```bash +pytest tests/test_eval_cli_archive.py -v +``` +Expected: FAIL — `_cmd_archive` doesn't exist. + +- [ ] **Step 3: Implement `_cmd_archive` in `src/eval/cli.py`** + +Append to `src/eval/cli.py` (after the existing command functions, before `main`): + +```python +def _cmd_archive(args: argparse.Namespace) -> int: + """Copy small artifacts of a run from eval_runs/ to a tracked location. + + Files copied: metrics.json, cost.json, metadata.json, config.yaml. + NOT copied: questions.jsonl (large). Its SHA-256 is recorded in metadata.json + under `questions_jsonl_sha256` so reviewers can verify against a re-run. + """ + import hashlib + import json + import shutil + from pathlib import Path + + runs_root = Path(getattr(args, "runs_root", None) or "eval_runs") + src = runs_root / args.run_id + if not src.exists(): + print(f"Run not found: {src}") + return 1 + + dst = Path(args.to) + dst.mkdir(parents=True, exist_ok=True) + + for name in ("metrics.json", "cost.json", "config.yaml"): + if (src / name).exists(): + shutil.copy2(src / name, dst / name) + + # Record questions.jsonl SHA in metadata.json before copying it + metadata = json.loads((src / "metadata.json").read_text()) + questions_path = src / "questions.jsonl" + if questions_path.exists(): + h = hashlib.sha256() + h.update(questions_path.read_bytes()) + metadata["questions_jsonl_sha256"] = h.hexdigest() + (dst / "metadata.json").write_text(json.dumps(metadata, indent=2)) + + print(f"Archived run {args.run_id} → {dst}") + return 0 +``` + +In `main`, register the subparser before `args = parser.parse_args(argv)`: + +```python + p_archive = subparsers.add_parser( + "archive", + help="Copy small run artifacts (metrics/cost/metadata/config) to a tracked path.", + ) + p_archive.add_argument("run_id", help="Run id to archive (must exist under eval_runs/).") + p_archive.add_argument("--to", required=True, help="Destination directory.") + p_archive.add_argument( + "--runs-root", default="eval_runs", + help="Root directory holding run subdirectories (default: eval_runs).", + ) + p_archive.set_defaults(func=_cmd_archive) +``` + +- [ ] **Step 4: Run test to verify it passes** + +```bash +pytest tests/test_eval_cli_archive.py -v +``` +Expected: PASS. + +- [ ] **Step 5: Commit** + +```bash +git add src/eval/cli.py tests/test_eval_cli_archive.py +git commit -m "feat(eval): add cli archive subcommand to copy small run artifacts to a tracked path" +``` + +--- + +## Task 10: Promote Phase 2 tier configs to `configs/eval/phase2/` + +**Files:** +- Move test fixtures into the canonical config directory. +- Add three answer-model variants for tier 2f. + +- [ ] **Step 1: Move tier YAMLs from fixtures to canonical location** + +```bash +mkdir -p configs/eval/phase2 +git mv tests/fixtures/phase2_configs/phase2_baseline.yaml configs/eval/phase2/phase2_baseline.yaml +git mv tests/fixtures/phase2_configs/phase2b_embedder.yaml configs/eval/phase2/phase2b_embedder.yaml +git mv tests/fixtures/phase2_configs/phase2c_hybrid.yaml configs/eval/phase2/phase2c_hybrid.yaml +git mv tests/fixtures/phase2_configs/phase2d_rerank.yaml configs/eval/phase2/phase2d_rerank.yaml +git mv tests/fixtures/phase2_configs/phase2e_rewrite.yaml configs/eval/phase2/phase2e_rewrite.yaml +git mv tests/fixtures/phase2_configs/phase2g_refusal.yaml configs/eval/phase2/phase2g_refusal.yaml +``` + +- [ ] **Step 2: Update the factory test path** + +In `tests/test_eval_pipeline_factory_phase2.py`, change `PHASE2_DIR = Path("tests/fixtures/phase2_configs")` to: + +```python +PHASE2_DIR = Path("configs/eval/phase2") +``` + +- [ ] **Step 3: Author the three 2f answer-model variants** + +Create `configs/eval/phase2/phase2f_models_gpt5mini.yaml`: + +```yaml +name: "phase2f_models_gpt5mini" +description: "Phase 2 tier 2f — answer-model comparison on the full 2g stack: gpt-5-mini variant. + This config is identical to phase2g_refusal.yaml; PR-B re-uses the 2g artifact rather + than re-running." +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + embedder: {name: bge_small_en_v1_5} + retriever: {top_k: 5} + hybrid: {enabled: true, bm25_top_k: 20, dense_top_k: 20, rrf_k: 60} + reranker: {model: ms_marco_minilm_l6_v2, rerank_top_n: 20, final_top_k: 5} + query_rewriter: {model: gpt-4.1-nano, max_expansions: 3} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} + refusal_handler: {enabled: true, similarity_threshold: 0.35, + no_answer_text: "I don't have enough information to answer that."} +eval: + datasets: [squad_v2_dev_200] + judge_model: gpt-4.1-mini + bootstrap_n: 1000 + permutation_n: 10000 + seed: 42 + spend_ceiling_usd: 1.5 +``` + +Create `configs/eval/phase2/phase2f_models_gpt41mini.yaml`: same as above with `generator.model: gpt-4.1-mini` and `name: "phase2f_models_gpt41mini"`. + +Create `configs/eval/phase2/phase2f_models_haiku.yaml`: same with `generator.model: claude-haiku-4-5` and `name: "phase2f_models_haiku"`. + +- [ ] **Step 4: Add a smoke test loading every Phase 2 YAML** + +Append to `tests/test_eval_pipeline_factory_phase2.py`: + +```python +def test_every_phase2_yaml_loads(): + """Every YAML under configs/eval/phase2/ must validate against EvalConfig.""" + for path in sorted(PHASE2_DIR.glob("*.yaml")): + cfg = load_config(path) + assert cfg.name == path.stem +``` + +- [ ] **Step 5: Run the test to verify all 9 YAMLs validate** + +```bash +pytest tests/test_eval_pipeline_factory_phase2.py -v +``` +Expected: PASS. + +- [ ] **Step 6: Commit** + +```bash +git add configs/eval/phase2/ tests/test_eval_pipeline_factory_phase2.py +git commit -m "chore(eval): add Phase 2 tier configs under configs/eval/phase2/" +``` + +--- + +## End of PR-A — Push and open PR + +- [ ] **Run the full test suite once before pushing** + +```bash +pytest -x -q +``` +Expected: zero failures. + +- [ ] **Push the branch and open the PR** + +```bash +git push -u origin feature/phase2-pipeline-extensions +gh pr create \ + --base feature/eval-harness-1d \ + --head feature/phase2-pipeline-extensions \ + --title "feat(eval): pipeline extensions for Phase 2 RAG quality matrix" \ + --body "$(cat <<'EOF' +## Summary + +PR-A of Phase 2: pipeline extensions only. PR-B follows with the experiment runs and writeup. + +- 5 new pipeline modules: BgeEmbedder, BM25HybridRetriever, CrossEncoderReranker, QueryRewriter, RefusalHandler. +- Schema extension to PipelineCfg with 5 sub-configs + EvalCfg.spend_ceiling_usd; backward compatible (existing baseline configs unchanged). +- Cost ledger refactor: EvalResult.cost_breakdown covers generator + judge + rewriter; aggregator surfaces per-bucket totals. +- New CLI: python -m src.eval.cli archive --to . +- 9 Phase 2 tier YAMLs under configs/eval/phase2/. +- New dep: rank-bm25. + +Stacked on #4 (Phase 1 PR-D). Will retarget to main once Phase 1 merges. + +## Test plan + +- [x] pytest -x -q passes locally +- [x] Existing Phase 1 baseline configs still load and produce the same pipeline shape +- [x] Each Phase 2 YAML builds a pipeline with the expected lever activations +- [ ] Reviewer: pip install -r requirements.txt adds exactly one entry (rank-bm25) +- [ ] Reviewer: python -m src.eval.cli run --config configs/eval/phase2/phase2_baseline.yaml succeeds with stubbed-LLM mode +EOF +)" +``` + +--- + +# PR-B — Experiments + Writeup + +PR-B is data-driven. Tasks below assume PR-A has merged or is at least in a state where its CLI works end-to-end against the real OpenAI API. + +> **Branching for PR-B:** +> ```bash +> git checkout feature/phase2-pipeline-extensions +> git pull +> git checkout -b feature/phase2-experiments-writeup +> ``` + +## Task 11: Run the baseline anchor + archive + +**Files:** +- Create: `docs/phase2/runs/_phase2_baseline/{metrics,cost,metadata,config}.json` + +- [ ] **Step 1: Run the baseline** + +```bash +set -a; source .env; set +a +conda run -n rag-qa --no-capture-output \ + python -m src.eval.cli run --config configs/eval/phase2/phase2_baseline.yaml +``` +Expected: end-of-output line `Run complete: phase2_baseline n=200 errors=0 `. Capture ``. + +- [ ] **Step 2: Archive small artifacts to `docs/phase2/runs/`** + +```bash +RUN_ID="" +SHORT_ID="${RUN_ID:0:23}" # date+config prefix +python -m src.eval.cli archive "$RUN_ID" \ + --to "docs/phase2/runs/${SHORT_ID}_phase2_baseline" +``` + +- [ ] **Step 3: Verify the four artifacts landed** + +```bash +ls "docs/phase2/runs/${SHORT_ID}_phase2_baseline" +``` +Expected: exactly `metrics.json`, `cost.json`, `metadata.json`, `config.yaml`. No `questions.jsonl`. + +- [ ] **Step 4: Commit** + +```bash +git add docs/phase2/runs/ +git commit -m "chore(eval): execute Phase 2 baseline run" +``` + +--- + +## Task 12: Run tier 2b (BGE embedder) + archive + +- [ ] **Step 1: Run** + +```bash +set -a; source .env; set +a +conda run -n rag-qa --no-capture-output \ + python -m src.eval.cli run --config configs/eval/phase2/phase2b_embedder.yaml +``` + +- [ ] **Step 2: Archive** + +```bash +RUN_ID="" +SHORT_ID="${RUN_ID:0:23}" +python -m src.eval.cli archive "$RUN_ID" \ + --to "docs/phase2/runs/${SHORT_ID}_phase2b_embedder" +``` + +- [ ] **Step 3: Spot-check that recall@5 is reported in metrics.json** + +```bash +python3 -c "import json; m=json.load(open('docs/phase2/runs/${SHORT_ID}_phase2b_embedder/metrics.json')); \ +[print(r) for r in m if r['metric_name']=='recall_at_5']" +``` + +- [ ] **Step 4: Commit** + +```bash +git add docs/phase2/runs/ +git commit -m "chore(eval): execute Phase 2 tier 2b — BGE embedder" +``` + +--- + +## Task 13: Run tier 2c (BM25 hybrid) + archive + +- [ ] **Step 1: Run** + +```bash +conda run -n rag-qa --no-capture-output \ + python -m src.eval.cli run --config configs/eval/phase2/phase2c_hybrid.yaml +``` + +- [ ] **Step 2: Archive + commit** + +```bash +RUN_ID="" +SHORT_ID="${RUN_ID:0:23}" +python -m src.eval.cli archive "$RUN_ID" \ + --to "docs/phase2/runs/${SHORT_ID}_phase2c_hybrid" +git add docs/phase2/runs/ +git commit -m "chore(eval): execute Phase 2 tier 2c — BM25 hybrid retrieval" +``` + +--- + +## Task 14: Run tier 2d (cross-encoder rerank) + archive + +- [ ] **Step 1–3: Same pattern as Task 13 with `phase2d_rerank.yaml`** + +```bash +conda run -n rag-qa --no-capture-output \ + python -m src.eval.cli run --config configs/eval/phase2/phase2d_rerank.yaml + +RUN_ID="" +SHORT_ID="${RUN_ID:0:23}" +python -m src.eval.cli archive "$RUN_ID" \ + --to "docs/phase2/runs/${SHORT_ID}_phase2d_rerank" +git add docs/phase2/runs/ +git commit -m "chore(eval): execute Phase 2 tier 2d — cross-encoder rerank" +``` + +--- + +## Task 15: Run tier 2e (query rewriting) + archive + +```bash +conda run -n rag-qa --no-capture-output \ + python -m src.eval.cli run --config configs/eval/phase2/phase2e_rewrite.yaml + +RUN_ID="" +SHORT_ID="${RUN_ID:0:23}" +python -m src.eval.cli archive "$RUN_ID" \ + --to "docs/phase2/runs/${SHORT_ID}_phase2e_rewrite" +git add docs/phase2/runs/ +git commit -m "chore(eval): execute Phase 2 tier 2e — query rewriting" +``` + +--- + +## Task 16: Run tier 2g (refusal handler) + archive + +```bash +conda run -n rag-qa --no-capture-output \ + python -m src.eval.cli run --config configs/eval/phase2/phase2g_refusal.yaml + +RUN_ID="" +SHORT_ID="${RUN_ID:0:23}" +python -m src.eval.cli archive "$RUN_ID" \ + --to "docs/phase2/runs/${SHORT_ID}_phase2g_refusal" +git add docs/phase2/runs/ +git commit -m "chore(eval): execute Phase 2 tier 2g — refusal handler" +``` + +--- + +## Task 17: Run tier 2f (answer-model comparison: gpt-4.1-mini and claude-haiku-4-5) + +The gpt-5-mini variant is identical to tier 2g — re-use that artifact. Only run the two non-baseline answer models. + +- [ ] **Step 1: Run gpt-4.1-mini variant** + +```bash +conda run -n rag-qa --no-capture-output \ + python -m src.eval.cli run --config configs/eval/phase2/phase2f_models_gpt41mini.yaml + +RUN_ID="" +SHORT_ID="${RUN_ID:0:23}" +python -m src.eval.cli archive "$RUN_ID" \ + --to "docs/phase2/runs/${SHORT_ID}_phase2f_models_gpt41mini" +``` + +- [ ] **Step 2: Run claude-haiku-4-5 variant** + +```bash +conda run -n rag-qa --no-capture-output \ + python -m src.eval.cli run --config configs/eval/phase2/phase2f_models_haiku.yaml + +RUN_ID="" +SHORT_ID="${RUN_ID:0:23}" +python -m src.eval.cli archive "$RUN_ID" \ + --to "docs/phase2/runs/${SHORT_ID}_phase2f_models_haiku" +``` + +- [ ] **Step 3: Symlink the gpt-5-mini artifact for completeness** + +```bash +G2G_DIR=$(ls -d docs/phase2/runs/*_phase2g_refusal | head -1) +G2G_BASENAME=$(basename "$G2G_DIR") +ln -s "../$G2G_BASENAME" "docs/phase2/runs/${G2G_BASENAME%_phase2g_refusal}_phase2f_models_gpt5mini" +``` + +- [ ] **Step 4: Commit** + +```bash +git add docs/phase2/runs/ +git commit -m "chore(eval): execute Phase 2 tier 2f — answer-model comparison (gpt-4.1-mini, claude-haiku-4-5)" +``` + +--- + +## Task 18: Generate pairwise compare reports + +5 chain comparisons + 2 cross-model comparisons = 7 reports under `docs/phase2/compare/`. + +- [ ] **Step 1: Helper script** + +Create `scripts/phase2_compare.sh`: + +```bash +#!/usr/bin/env bash +# Run pairwise compare for all Phase 2 chain + model comparisons. +set -euo pipefail + +mkdir -p docs/phase2/compare + +# Resolve the run_id for each tier from the archived metadata. +get_run_id() { + local tier_suffix="$1" + local meta=$(ls docs/phase2/runs/*_${tier_suffix}/metadata.json | head -1) + python3 -c "import json,sys; print(json.load(open(sys.argv[1]))['run_id'])" "$meta" +} + +BASELINE=$(get_run_id phase2_baseline) +B2B=$(get_run_id phase2b_embedder) +B2C=$(get_run_id phase2c_hybrid) +B2D=$(get_run_id phase2d_rerank) +B2E=$(get_run_id phase2e_rewrite) +B2G=$(get_run_id phase2g_refusal) +M_GPT5=$B2G +M_GPT41=$(get_run_id phase2f_models_gpt41mini) +M_HAIKU=$(get_run_id phase2f_models_haiku) + +# Chain comparisons +python -m src.eval.cli compare "$BASELINE" "$B2B" --html docs/phase2/compare/2b_vs_baseline.html +python -m src.eval.cli compare "$B2B" "$B2C" --html docs/phase2/compare/2c_vs_2b.html +python -m src.eval.cli compare "$B2C" "$B2D" --html docs/phase2/compare/2d_vs_2c.html +python -m src.eval.cli compare "$B2D" "$B2E" --html docs/phase2/compare/2e_vs_2d.html +python -m src.eval.cli compare "$B2E" "$B2G" --html docs/phase2/compare/2g_vs_2e.html + +# Cross-model comparisons (all on the 2g stack) +python -m src.eval.cli compare "$M_GPT5" "$M_GPT41" --html docs/phase2/compare/gpt41mini_vs_gpt5mini.html +python -m src.eval.cli compare "$M_GPT5" "$M_HAIKU" --html docs/phase2/compare/haiku_vs_gpt5mini.html + +echo "Wrote 7 compare reports to docs/phase2/compare/" +``` + +```bash +chmod +x scripts/phase2_compare.sh +``` + +- [ ] **Step 2: Run it** + +```bash +./scripts/phase2_compare.sh +``` + +- [ ] **Step 3: Verify 7 HTML files** + +```bash +ls docs/phase2/compare/*.html | wc -l +``` +Expected: `7`. + +- [ ] **Step 4: Commit** + +```bash +git add scripts/phase2_compare.sh docs/phase2/compare/ +git commit -m "feat(eval): pairwise compare reports for the Phase 2 matrix" +``` + +--- + +## Task 19: Write `docs/PHASE2_RESULTS.md` + +**Files:** +- Create: `docs/PHASE2_RESULTS.md` +- Create: `docs/phase2/chart.png` (rendered from compare data) + +- [ ] **Step 1: Generate the per-tier metric chart** + +Create `scripts/phase2_chart.py`: + +```python +"""Render per-tier metric chart from the archived Phase 2 run dirs. + +Usage: + python scripts/phase2_chart.py --out docs/phase2/chart.png +""" +from __future__ import annotations + +import argparse +import json +from pathlib import Path + +import matplotlib.pyplot as plt + +TIER_ORDER = [ + "phase2_baseline", "phase2b_embedder", "phase2c_hybrid", + "phase2d_rerank", "phase2e_rewrite", "phase2g_refusal", +] +METRICS = [ + "answer_correctness", "judge_faithfulness", + "recall_at_5", "refusal_correctness", +] + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--runs-dir", default="docs/phase2/runs") + parser.add_argument("--out", required=True) + args = parser.parse_args() + + runs_dir = Path(args.runs_dir) + tier_to_metrics: dict[str, dict[str, tuple[float, float, float]]] = {} + for tier in TIER_ORDER: + match = list(runs_dir.glob(f"*_{tier}/metrics.json")) + if not match: + print(f"missing: {tier}") + continue + rows = json.loads(match[0].read_text()) + tier_to_metrics[tier] = { + r["metric_name"]: (r["mean"], r["ci_low"], r["ci_high"]) + for r in rows if r["dataset"] == "(all)" and r["metric_name"] in METRICS + } + + fig, axes = plt.subplots(1, len(METRICS), figsize=(16, 4), sharey=True) + for ax, metric in zip(axes, METRICS): + means, lows, highs = [], [], [] + for tier in TIER_ORDER: + m = tier_to_metrics.get(tier, {}).get(metric, (0, 0, 0)) + means.append(m[0]) + lows.append(m[0] - m[1]) + highs.append(m[2] - m[0]) + ax.bar(range(len(TIER_ORDER)), means, yerr=[lows, highs], capsize=4) + ax.set_title(metric) + ax.set_xticks(range(len(TIER_ORDER))) + ax.set_xticklabels([t.replace("phase2", "") for t in TIER_ORDER], rotation=30, ha="right") + ax.set_ylim(0, 1) + fig.suptitle("Phase 2 — RAG Quality Matrix (per-tier means with 95% bootstrap CI)") + fig.tight_layout() + fig.savefig(args.out, dpi=140, bbox_inches="tight") + print(f"Wrote {args.out}") + + +if __name__ == "__main__": + main() +``` + +```bash +python scripts/phase2_chart.py --out docs/phase2/chart.png +``` + +- [ ] **Step 2: Author the writeup** + +Create `docs/PHASE2_RESULTS.md`: + +```markdown +# Phase 2 — RAG Quality Matrix Results + +> Layered ablation of seven RAG architectural levers on the Phase 1 SQuAD-200 baseline. +> Spec: [`docs/superpowers/specs/2026-04-27-phase2-rag-quality-matrix-design.md`](superpowers/specs/2026-04-27-phase2-rag-quality-matrix-design.md) +> Plan: [`docs/superpowers/plans/2026-04-27-phase2-rag-quality-matrix.md`](superpowers/plans/2026-04-27-phase2-rag-quality-matrix.md) + +## Methodology + +Phase 1 shipped an end-to-end evaluation harness with paired permutation tests and bootstrap CIs (PR #1–#4). Phase 2 uses that harness to measure five architectural levers in a layered chain — each tier inheriting from the previous tier and toggling exactly one variable — plus a parallel three-way answer-model comparison on top of the final 2g stack. The eval set is 200 SQuAD v2 dev questions; ML papers are deferred to Phase 3. + +Comparisons reported in this document use: +- **Bootstrap percentile CI** (n=1000) for per-tier means. +- **Paired permutation test** (n=10000) for between-tier deltas. +- A delta is "significant" at p < 0.05. + +## The matrix at a glance + +![Per-tier metric chart](phase2/chart.png) + +## Per-tier results + +| Tier | Lever toggled | answer_correctness Δ | judge_faithfulness Δ | refusal_correctness Δ | recall@5 Δ | p (paired) | +|------|----------------|------:|------:|------:|------:|------:| +| baseline | (none) | — | — | — | — | — | +| 2b | BGE embedder | | | | | | +| 2c | BM25 hybrid | | | | | | +| 2d | cross-encoder rerank | | | | | | +| 2e | query rewriting | | | | | | +| 2g | refusal handler | | | | | | + +## Answer-model comparison (on the 2g stack) + +| Generator model | answer_correctness | judge_faithfulness | refusal_correctness | total cost | +|-----------------|-------------------:|-------------------:|--------------------:|----------:| +| gpt-5-mini | | | | | +| gpt-4.1-mini | | | | | +| claude-haiku-4-5| | | | | + +## Findings + +For each lever, fill in 2-3 sentences referencing the paired-significance result. Examples: + +- **2b BGE embedder.** +- **2c BM25 hybrid.** +- **2d Cross-encoder rerank.** +- **2e Query rewriting.** +- **2g Refusal handler.** +- **2f Answer-model comparison.** + +## Winning stack + +.yaml> + +## Cost ledger + +| Run | Generator | Judge | Rewriter | Total | +|-----|----------:|------:|---------:|------:| +| baseline | | | | | +| 2b | | | | | +| 2c | | | | | +| 2d | | | | | +| 2e | | | | | +| 2g | | | | | +| 2f gpt-4.1-mini | | | | | +| 2f claude-haiku-4-5 | | | | | +| **Total** | | | | <≤ $5> | + +## Reproducibility + +```bash +# install (Phase 1 + Phase 2) +pip install -r requirements.txt + +# run any tier +python -m src.eval.cli run --config configs/eval/phase2/.yaml + +# regenerate compare reports +./scripts/phase2_compare.sh + +# regenerate chart +python scripts/phase2_chart.py --out docs/phase2/chart.png +``` + +## Implementation references + +- Schema: `src/eval/config.py` (PipelineCfg, EmbedderCfg, HybridCfg, RerankerCfg, QueryRewriterCfg, RefusalHandlerCfg) +- Pipeline modules: `src/eval/embedders/`, `src/eval/retrievers/`, `src/eval/transforms/` +- Factory: `src/eval/pipeline_factory.py::build_pipeline` +- Cost ledger: `EvalResult.cost_breakdown`, `aggregate_costs` in `src/eval/metrics/operational.py` +``` + +- [ ] **Step 3: Fill in the placeholders from the actual run data** + +Open each archived `metrics.json` and `cost.json`, copy real numbers into the tables. Open each `docs/phase2/compare/*.html` and copy the paired-permutation p-values into the per-tier table. Replace every `` block with concrete prose grounded in the data. + +- [ ] **Step 4: Commit** + +```bash +git add docs/PHASE2_RESULTS.md docs/phase2/chart.png scripts/phase2_chart.py +git commit -m "docs(eval): Phase 2 results — methodology, chart, findings" +``` + +--- + +## Task 20: Link Phase 2 results from the main README + +**Files:** +- Modify: `README.md` + +- [ ] **Step 1: Add a one-liner near the existing eval section** + +Find the existing "Evaluation" section in `README.md`. Append: + +```markdown +### Phase 2 — Quality matrix + +Phase 2 measured 7 architectural levers (BGE embedder, BM25 hybrid, cross-encoder rerank, LLM query rewriting, answer-model sweep, refusal handler) layered on top of the Phase 1 baseline against the same 200-question SQuAD v2 dev set. See [`docs/PHASE2_RESULTS.md`](docs/PHASE2_RESULTS.md) for the chart, paired-significance results, winning stack, and cost ledger. +``` + +- [ ] **Step 2: Commit** + +```bash +git add README.md +git commit -m "docs(readme): link Phase 2 results from main README" +``` + +--- + +## End of PR-B — Push and open PR + +- [ ] **Push and open the PR** + +```bash +git push -u origin feature/phase2-experiments-writeup +gh pr create \ + --base feature/phase2-pipeline-extensions \ + --head feature/phase2-experiments-writeup \ + --title "docs(eval): Phase 2 RAG quality matrix — results + findings" \ + --body "$(cat <<'EOF' +## Summary + +PR-B of Phase 2: experiment runs and writeup. Stacked on PR-A (#). + +- 8 distinct eval runs against SQuAD-200, archived under docs/phase2/runs/. +- 7 pairwise compare reports under docs/phase2/compare/ (5 chain + 2 model). +- docs/PHASE2_RESULTS.md: methodology, chart, paired-significance table, winning stack, per-lever findings, cost ledger. +- README updated with a link to the writeup. + +Live eval_runs/ directories stay local (gitignored). Each archived run dir contains only metrics.json, cost.json, metadata.json, config.yaml. The questions.jsonl SHA is recorded in metadata.json. + +## Test plan + +- [x] All 8 runs landed with errors=0 +- [x] All 7 compare HTML reports render and show non-empty significance numbers +- [x] Total cost across all runs is ≤ $5 (see ledger in writeup) +EOF +)" +``` + +--- + +# Self-Review + +The plan author runs this checklist after writing the plan: + +1. **Spec coverage:** Each spec section maps to at least one task — ✅ + - §1 goal + scope → Tasks 11–20 (data) + 1–10 (extensions) + - §2 architecture → Tasks 3–8 + - §3.1 schema → Task 1 + - §3.2 factory → Task 8 + - §3.3 matrix → Tasks 9, 10–17 + - §4.1 unit tests → embedded in Tasks 3–7 + - §4.2 factory tests → Task 8 + - §4.3 smoke → Task 8 (`test_phase2_query_with_refusal_short_circuits`) + - §4.6 cost ledger → Task 2 + - §5.1 PR-A commit list → Tasks 1–10 + - §5.2 PR-B commit list → Tasks 11–20 +2. **Placeholder scan:** No `TBD`, `TODO`, `fill in details`, or steps that describe without showing. The writeup task (19) intentionally leaves `` markers because it requires reading the live data; the steps explicitly require replacing them. ✅ +3. **Type consistency:** + - `EmbedderCfg.name` is `Literal["chroma_default", "bge_small_en_v1_5"]` everywhere. + - `RerankerCfg.model` is `Literal["ms_marco_minilm_l6_v2"] | None` everywhere. + - `QueryRewriter.expand` returns `tuple[list[str], float, int, int]` in the impl and the tests. + - `RefusalHandler.refuse_response` returns `tuple[list[SearchResult], str]` in impl and tests. + - `LLMHandler.generate_with_usage` returns `tuple[str, int, int]` in impl, tests, and the rewriter contract. ✅ + +--- + +## Execution Handoff + +**Plan complete and saved to `docs/superpowers/plans/2026-04-27-phase2-rag-quality-matrix.md`. Two execution options:** + +1. **Subagent-Driven (recommended)** — I dispatch a fresh subagent per task (Tasks 1–10 are TDD code tasks; Tasks 11–20 are run-and-archive). Review between tasks, fast iteration. + +2. **Inline Execution** — Execute tasks in this session using `superpowers:executing-plans`, batch execution with checkpoints for review. + +**Which approach?** diff --git a/docs/superpowers/specs/2026-04-27-phase2-rag-quality-matrix-design.md b/docs/superpowers/specs/2026-04-27-phase2-rag-quality-matrix-design.md new file mode 100644 index 00000000..fd3dff11 --- /dev/null +++ b/docs/superpowers/specs/2026-04-27-phase2-rag-quality-matrix-design.md @@ -0,0 +1,421 @@ +# Phase 2 — RAG Quality Matrix (Design Spec) + +> **Status:** revised 2026-04-27 after spec review. Pending implementation plan via `superpowers:writing-plans`. +> **Author:** Mohamed Elkholy +> **Predecessor:** [`2026-04-26-rag-eval-harness-phase-1-design.md`](./2026-04-26-rag-eval-harness-phase-1-design.md) (Phase 1, eval harness) +> **Eval baseline being targeted:** `eval_runs/2026-04-27_191733_baseline_squad_only_c925492` — 200 SQuAD v2 dev questions, real OpenAI API. + +--- + +## 1. Goal + Scope + +Phase 2 produces a layered ablation matrix that measures the lift each major RAG architectural lever buys on top of the Phase 1 baseline, then ships a portfolio-grade results document attributing the lift to mechanism. + +### 1.1 Goal + +Run 9 evals (1 baseline re-run + 5 single-lever tiers + 3 answer-model comparison variants) through the existing Phase 1 harness, with each tier in the layered chain inheriting from the previous tier so adjacent comparisons attribute lift to one mechanism at a time. Write findings. + +> **Why not 10:** Tier 2a (semantic chunking) is **deferred to Phase 3**. The current SQuAD ingest path (`src/eval/pipeline_factory.py::_ingest_squad`) stores each question's gold context as a single Chroma document and never invokes the chunker, so toggling `chunker.strategy` cannot produce measurable lift on the SQuAD-only matrix. The chunker only fires on `ml_papers_v1` ingest (`_ingest_ml_papers`), so semantic-vs-recursive lift is a Phase 3 question. + +### 1.2 In scope + +- 10 eval YAML configs under `configs/eval/phase2/`, each runnable through `python -m src.eval.cli run`. +- 5 new pipeline modules wired into the `EvalPipeline` factory (not into prod `RAGBackend`): + `BgeEmbedder`, `BM25HybridRetriever`, `CrossEncoderReranker`, `QueryRewriter`, `RefusalHandler`. +- Schema extension to `EvalConfig.pipeline` with 5 new sub-config blocks, all backward-compatible (defaults match current behavior). +- `docs/PHASE2_RESULTS.md`: methodology, per-tier metric chart, paired-significance table, "winning stack" recipe, ≥ 1 sentence finding per lever. + +### 1.3 Out of scope + +- Wiring any new module into the production `RAGBackend` / live `/chat` path. Phase 2 is measurement only. +- Adding `ml_papers_v1` to the matrix (deferred to Phase 3 once labeled). +- Auto-promotion of the winning stack to runtime defaults (Phase 2.5). + +### 1.4 Success criteria + +- All 8 distinct runs land on disk, visible in `/eval`, pairwise compare reports render. +- `PHASE2_RESULTS.md` ships with the chart + significance table + per-tier finding. +- Total API spend ≤ $5 (hard ceiling, no rollback). **Spend ledger covers all three LLM call sites** — generator, judge, and (if enabled) query-rewriter (see §4.6). + +### 1.5 Branching + +Phase 2 PR-A is stacked on `feature/eval-harness-1d` (current Phase 1 PR #4 tip). Once Phase 1 merges, PR-A retargets to `main`. + +--- + +## 2. Architecture + +The existing `EvalPipeline` factory at `src/eval/pipeline_factory.py` (`build_pipeline(config: EvalConfig, dataset_name: str, ...)` at line 269) already takes an `EvalConfig` and a dataset name and assembles a runtime pipeline. `EvalRunner` in `src/eval/runner.py:49` calls it directly. Phase 2 extends the existing function's switch statements and `EvalPipeline.__init__` signature; it does not introduce a new orchestrator and does not move the file. + +### 2.1 Module map (5 new files, all under `src/`) + +``` +src/ +├── eval/ +│ ├── pipeline_factory.py # existing — build_pipeline() extended with new branches +│ ├── runner.py # existing — _score_question extended with cost capture +│ ├── retrievers/ # NEW package +│ │ ├── __init__.py +│ │ ├── bm25_hybrid.py # BM25HybridRetriever (RRF fusion) +│ │ └── reranker.py # CrossEncoderReranker (ms-marco-MiniLM) +│ ├── embedders/ # NEW package +│ │ ├── __init__.py +│ │ └── bge_small.py # BgeEmbedder — Chroma EmbeddingFunction adapter +│ └── transforms/ # NEW package +│ ├── __init__.py +│ ├── query_rewriter.py # LLM-based query expansion +│ └── refusal_handler.py # answerability gate + no-answer prompt +``` + +Every new module is single-responsibility and budgeted under the 250-line CLAUDE.md ceiling. `pipeline_factory.py` is the only existing file at risk; if it crosses 250 lines, split into `pipeline_factory.py` (orchestration) + `factory_levers.py` (per-lever construction helpers). + +### 2.2 Lever-by-lever wiring + +| Tier | Module | Plug-in point in `EvalPipeline` | Notes | +|------|--------|---------------------------------|-------| +| ~~2a `chunking_semantic`~~ | — | — | **Deferred to Phase 3** (no-op on SQuAD ingest path). | +| 2b `embedder_bge` | `eval/embedders/bge_small.py` | passed to Chroma collection as its `embedding_function` at collection-creation time inside `build_pipeline`; `ChromaVectorStore.upsert/query` then auto-embeds via the collection function (see §3.2). | Uses `sentence-transformers` `BAAI/bge-small-en-v1.5` (33M params, 384-dim, on-device). | +| 2c `hybrid_bm25` | `eval/retrievers/bm25_hybrid.py` | retriever switch | Wraps `rank_bm25` + `ChromaVectorStore`; RRF fusion on top-20 from each side. | +| 2d `rerank_crossenc` | `eval/retrievers/reranker.py` | post-retrieval hook | `cross-encoder/ms-marco-MiniLM-L-6-v2`, top-20 → top-5. | +| 2e `query_rewrite` | `eval/transforms/query_rewriter.py` | pre-retrieval hook | One LLM call per query; 1–3 expansions, dedup-merged. Uses `gpt-4.1-nano`. | +| 2g `refusal_handler` | `eval/transforms/refusal_handler.py` | post-retrieval gate | Threshold on top-1 similarity; if low, short-circuit to no-answer text. | +| 2f `answer_model` (parallel comparison) | no new module | generator switch (existing) | YAML-only — three runs sweep `gpt-5-mini`, `gpt-4.1-mini`, `claude-haiku-4-5`, all on the **full 2g stack**. Reported as a side-by-side comparison, not a layered tier (see §3.3.2). | + +### 2.3 Data flow per query (final tier 2g) + +``` +query + └─► QueryRewriter ─► {q, q', q''} (2e) + └─► HybridRetriever (BM25 + BGE) (2b, 2c) + └─► CrossEncoderReranker top-5 (2d) + └─► RefusalHandler (2g) + ├─► low conf ─► "I don't know" + └─► high conf ─► Generator (gpt-5-mini) + └─► answer + telemetry +``` + +Each tier's config disables levers introduced after it (e.g., tier 2c sets `reranker.model: null`, `query_rewriter.model: null`). Tier 2f re-uses the 2g pipeline three times with three different `generator.model` values; it is a comparison, not a chain link. + +### 2.4 Dependency additions + +| Tier needing it | Dep | Size | +|------|-----|------| +| 2b | `sentence-transformers` already in deps (used by judge embedder) | 0 | +| 2c | `rank-bm25` | tiny, pure-Python | +| 2d | `sentence-transformers` cross-encoder model — pulled at runtime | model ~80MB | +| 2e | none — uses existing `LLMHandler` | 0 | +| 2f | none | 0 | +| 2g | none — pure logic | 0 | + +Net new third-party deps: **just `rank-bm25`**. + +--- + +## 3. Config Schema + Factory Contract + +### 3.1 Schema extensions (additive, backward-compatible) + +All new fields default to "off / current behavior" so `baseline.yaml` and `baseline_squad_only.yaml` keep validating unchanged. + +```python +# src/eval/config.py — extending PipelineCfg + +class EmbedderCfg(BaseModel): + """NEW — embedder selection. None = ChromaDB default ONNX (current behavior).""" + model_config = ConfigDict(extra="forbid") + name: Literal["chroma_default", "bge_small_en_v1_5"] = "chroma_default" + +class HybridCfg(BaseModel): + """NEW — BM25 + dense fusion. enabled=False is the default (pure dense).""" + model_config = ConfigDict(extra="forbid") + enabled: bool = False + bm25_top_k: int = 20 # candidates from sparse side + dense_top_k: int = 20 # candidates from dense side + rrf_k: int = 60 # standard RRF constant + +class RerankerCfg(BaseModel): + """NEW — cross-encoder rerank top-N → top-K. None = no rerank.""" + model_config = ConfigDict(extra="forbid") + model: Literal["ms_marco_minilm_l6_v2"] | None = None + rerank_top_n: int = 20 + final_top_k: int = 5 + +class QueryRewriterCfg(BaseModel): + """NEW — LLM query expansion. None = no rewrite.""" + model_config = ConfigDict(extra="forbid") + model: str | None = None # e.g. "gpt-4.1-nano" + max_expansions: int = 3 + +class RefusalHandlerCfg(BaseModel): + """NEW — answerability gate. enabled=False = current behavior.""" + model_config = ConfigDict(extra="forbid") + enabled: bool = False + similarity_threshold: float = 0.35 + no_answer_text: str = "I don't have enough information to answer that." + +# Extends PipelineCfg +class PipelineCfg(BaseModel): + chunker: ChunkerCfg + embedder: EmbedderCfg = Field(default_factory=EmbedderCfg) # NEW + retriever: RetrieverCfg + hybrid: HybridCfg = Field(default_factory=HybridCfg) # NEW + reranker: RerankerCfg = Field(default_factory=RerankerCfg) # NEW + query_rewriter: QueryRewriterCfg = Field(default_factory=QueryRewriterCfg) # NEW + generator: GeneratorCfg + refusal_handler: RefusalHandlerCfg = Field(default_factory=RefusalHandlerCfg) # NEW +``` + +`extra="forbid"` rejects typos at YAML load time; backward-compat is enforced by a test that loads existing baseline configs without modification. + +### 3.2 Factory contract + +`build_pipeline(config: EvalConfig, dataset_name: str, ...)` in `src/eval/pipeline_factory.py:269` is extended in place. The embedder is selected first because the Chroma collection must be created with the right embedding function: + +```python +# src/eval/pipeline_factory.py — extended + +def build_pipeline( + config: EvalConfig, dataset_name: str, ..., +) -> EvalPipeline: + embedding_function = _build_embedding_function(config.pipeline.embedder) # NEW + collection = _make_collection(name=..., embedding_function=embedding_function) + vector_store = ChromaVectorStore(collection=collection) + chunker = _build_chunker(config.pipeline.chunker) + + base_retriever = _build_retriever(config.pipeline, vector_store) # extended for hybrid + return EvalPipeline( + dataset_name=dataset_name, + chunker=chunker, + vector_store=vector_store, + retriever=base_retriever, + rewriter=_build_rewriter(config.pipeline.query_rewriter), # NEW (None when off) + reranker=_build_reranker(config.pipeline.reranker), # NEW (None when off) + refusal_handler=_build_refusal(config.pipeline.refusal_handler), # NEW (None when off) + generator=_build_generator(config.pipeline.generator), + ) +``` + +Inside `EvalPipeline.query`: + +```python +def query(self, q: str) -> EvalQueryResult: + queries = self.rewriter.expand(q) if self.rewriter else [q] + candidates = self.retriever.retrieve_many(queries) # dedup inside + if self.reranker: + candidates = self.reranker.rerank(q, candidates) + if self.refusal_handler and self.refusal_handler.should_refuse(candidates): + return self.refusal_handler.refuse_response(timings=...) + return self.generator.answer(q, candidates, timings=...) +``` + +Each lever is a no-op pass-through when its config is default; existing baselines route through identical control flow as today. + +#### 3.2.1 BGE embedder wiring (resolves spec-review H3) + +`BgeEmbedder` is **a Chroma `EmbeddingFunction` adapter**, not a separate vector encoder the pipeline calls explicitly. `_build_embedding_function(EmbedderCfg)` returns one of: + +- `EmbedderCfg.name == "chroma_default"` → `chromadb.utils.embedding_functions.DefaultEmbeddingFunction()` (current behavior; ChromaDB's built-in ONNX MiniLM, 384-dim). +- `EmbedderCfg.name == "bge_small_en_v1_5"` → `BgeEmbedder()`, a `chromadb.api.types.EmbeddingFunction` subclass that loads `BAAI/bge-small-en-v1.5` via `sentence-transformers` (384-dim) and exposes `__call__(input: list[str]) -> list[list[float]]`. + +The embedding function is set on the Chroma `Collection` at creation time. `ChromaVectorStore.upsert(documents=..., metadatas=...)` and `ChromaVectorStore.query(query_text=..., top_k=...)` then auto-embed via the collection's function (`src/vector_store.py:109-145`, `:154-175`); no per-call code change. + +Because `EvalRunner` recreates the collection on every run (existing pattern: each run is named after the run_id and is destroyed afterwards), there is **no risk of dimension mixing** between the default embedder and BGE, and **no migration work** between tiers. + +Tests for `BgeEmbedder`: +1. Conforms to `EmbeddingFunction` ABC (returns `list[list[float]]` of 384-dim vectors). +2. Works through `ChromaVectorStore` end-to-end on a 3-doc fixture (upsert → query → top-1 ID is the expected doc). +3. Cosine distance on the synonym pair `("cat", "feline")` is smaller than the unrelated pair `("cat", "airplane")` — proves the model is loading the right weights, not a stub. + +### 3.3 The 9-run matrix + +All YAMLs under `configs/eval/phase2/`. The matrix has two parts: + +#### 3.3.1 Layered chain (6 runs) + +Each tier inherits the previous tier's settings and toggles **one** field. `generator.model` stays at `gpt-5-mini` throughout the chain so the model variable is held constant; the answer-model question is answered separately by §3.3.2. + +| File | Lever toggled | Key field deltas (cumulative on previous tier) | +|------|----------------|----------------| +| `phase2_baseline.yaml` | (none — baseline re-run) | identical to `baseline_squad_only.yaml`; serves as the chain's anchor and lets the runner re-attribute spend to Phase 2 | +| `phase2b_embedder.yaml` | BGE embedder | + `embedder.name: bge_small_en_v1_5` | +| `phase2c_hybrid.yaml` | BM25 + dense | + `hybrid.enabled: true` | +| `phase2d_rerank.yaml` | cross-encoder rerank | + `reranker.model: ms_marco_minilm_l6_v2`, `rerank_top_n: 20`, `final_top_k: 5` | +| `phase2e_rewrite.yaml` | query rewriting | + `query_rewriter.model: gpt-4.1-nano`, `max_expansions: 3` | +| `phase2g_refusal.yaml` | refusal handler | + `refusal_handler.enabled: true`, `similarity_threshold: 0.35`; **`generator.model: gpt-5-mini`** (held constant) | + +#### 3.3.2 Answer-model comparison (3 runs, parallel to the chain) + +Tier 2f is **not** a chain link; it is three side-by-side runs that share the **full 2g pipeline** and vary **only** the `generator.model`. This decouples answer-model choice from the layered question and lets the writeup report a clean model comparison on a fixed retrieval/refusal stack. Each YAML inherits everything from `phase2g_refusal.yaml` and overrides `generator.model`: + +| File | `generator.model` | +|------|-------------------| +| `phase2f_models_gpt5mini.yaml` | `gpt-5-mini` (this run is identical to `phase2g_refusal.yaml`; PR-B re-uses that artifact rather than re-running it) | +| `phase2f_models_gpt41mini.yaml` | `gpt-4.1-mini` | +| `phase2f_models_haiku.yaml` | `claude-haiku-4-5` | + +**Total runs: 9** (6 chain + 3 model-comparison; the gpt-5-mini variant of 2f is the same artifact as `phase2g_refusal.yaml`, so PR-B authors 9 distinct YAMLs but only 8 runs need to execute). + +#### 3.3.3 Significance comparisons reported + +The writeup reports paired permutation tests on: +- 5 chain comparisons: 2b–baseline, 2c–2b, 2d–2c, 2e–2d, 2g–2e. +- 2 cross-model comparisons: gpt-4.1-mini–gpt-5-mini, claude-haiku-4-5–gpt-5-mini (each holding the rest of the 2g stack constant). + +**Conservative cost ceiling:** $0.10/run × 8 distinct runs = $0.80, well under the $5 budget. + +--- + +## 4. Testing Strategy + +Each new module ships with tests that lock its contract independently of the eval pipeline. Tests are tiered by speed: unit tests run on every CI invocation; integration tests run on demand. + +### 4.1 Unit tests (per new module, fast, no LLM calls) + +| Module | Assertions | +|--------|-----------| +| `BgeEmbedder` | `embed_documents([s])` returns a 384-dim vector. Cosine sim of `"cat"` and `"feline"` > `"cat"` and `"airplane"`. | +| `BM25HybridRetriever` | RRF fusion on asymmetric inputs (avoids tied scores): `A=[a,b,c,d]`, `B=[d,a]`, `rrf_k=60` → fused order `a, d, b, c`. (`a` wins from joint coverage at 1/61+1/62; `d` is second from rank-1 in B at 1/61+1/64; then `b` at 1/62; then `c` at 1/63.) | +| `CrossEncoderReranker` | Given a query and 5 candidates with one obvious match, the match ranks first after `rerank()`. ≤ 10s. | +| `QueryRewriter` | `model=None` → `expand(q)` returns `[q]` unchanged. With stubbed LLMHandler returning fixed expansions, `expand(q)` returns deduped `[q, q', q'']`. | +| `RefusalHandler` | `should_refuse(candidates)` returns `True` when top-1 similarity < threshold; `False` otherwise. Empty candidates → refuse. | +| `SemanticChunker` (existing) | Re-chunking the same text returns identical chunks (regression test for determinism). | +| `PipelineCfg` schema | Loading `baseline.yaml` (no Phase-2 fields) validates with all defaults. Loading `phase2g_refusal.yaml` validates with `enabled: true`. Unknown field raises `ValidationError`. | + +### 4.2 Factory tests (composition, no LLM calls) + +`tests/eval/test_pipeline_factory_phase2.py`: +- For each of the 9 Phase 2 YAML configs, `build_pipeline(config, dataset_name="squad_v2_dev_200")` returns an `EvalPipeline` whose attributes match expectations: `pipeline.rewriter is None` for tiers ≤ 2d; `pipeline.reranker is not None` for tiers ≥ 2d; `pipeline.refusal_handler is not None` only for tier 2g and the three 2f variants. +- Defaults round-trip: a config with no `hybrid`/`reranker` blocks builds a pipeline equivalent to baseline. + +### 4.3 End-to-end smoke (with stubbed LLM, no API spend) + +`tests/eval/test_phase2_smoke.py`: +- Builds tier 2g pipeline against a tiny fixture corpus (3 docs). +- Sends one answerable question → non-refusal answer. Sends one nonsense question → refusal text. +- Asserts the right modules fired by reading `EvalPipeline.timings_ms` keys (`rewrite`, `retrieve`, `rerank`, `refusal_check`, `generate`). + +### 4.4 Eval-data sanity tests (no API spend) + +1. `tests/eval/test_phase2_configs_load.py` — every YAML under `configs/eval/phase2/` validates and produces a buildable pipeline. +2. `tests/eval/test_phase2_squad_dataset.py` — the SQuAD200 dataset loader still returns 200 questions with the same `questions.jsonl` SHA after Phase-2 changes (catches accidental dataset corruption). + +### 4.5 Integration runs (real API, gated) + +The 8 distinct eval runs. Not in `tests/`; live in `eval_runs/`. Triggered manually: + +```bash +make phase2-matrix # or shell loop: +for cfg in configs/eval/phase2/*.yaml; do + python -m src.eval.cli run --config "$cfg" +done +``` + +The harness is the test. Each run produces a comparable artifact; pairwise compare runs after the matrix completes. + +### 4.6 Cost ledger (resolves spec-review M4) + +Phase 1's `cost.json` aggregates only generator-side cost because that's the only call site `EvalPipeline.query` knew about. Phase 2 has three LLM call sites (generator, judge, query-rewriter) and the hard $5 ceiling has to be enforceable across all of them. The fix is local: + +1. **Per-question record extension.** `EvalResult.cost_usd` becomes `EvalResult.cost_breakdown: dict[str, float]` (`{"generator": ..., "rewriter": ..., "judge": ...}`) plus `cost_usd` (sum). Schema migration is back-compatible — older records read with `cost_breakdown` defaulting to `{"generator": cost_usd}`. +2. **Generator cost.** Already captured in `EvalPipeline.query` telemetry. No change. +3. **Rewriter cost.** New: `QueryRewriter.expand` returns `(queries, cost_usd, prompt_tokens, completion_tokens)`. `EvalPipeline.query` adds the rewriter cost into the per-question breakdown. +4. **Judge cost.** `_score_question` in `src/eval/runner.py:312` already calls each judge through `LLMHandler`. Extend the judge wrappers (`src/eval/metrics/judge_*.py`) to return `(score, cost_usd, prompt_tokens, completion_tokens)`; sum into the per-question record. +5. **Aggregator.** `cost.json` totals `cost_breakdown` instead of `cost_usd`. Adds three new lines (`generator_total`, `rewriter_total`, `judge_total`) plus the existing `total_usd`. +6. **Guardrail.** `EvalRunner` reads `eval.spend_ceiling_usd: float | None` from the config (NEW field, defaults to None). When set, the runner aborts with a clear error if the running cumulative cost exceeds it. Phase 2 configs set `spend_ceiling_usd: 1.50` per run; the matrix-level ceiling stays in the writeup ledger. + +This addition is part of PR-A (commit 7 — factory wiring) so PR-B can rely on the new schema when it lands. + +--- + +## 5. Delivery Sequence + Risks + +### 5.1 PR-A — pipeline extensions (~1700 lines, code-only) + +Branched off `feature/eval-harness-1d`. Title: `feat(eval): pipeline extensions for Phase 2 RAG quality matrix`. + +| # | Commit | Files | +|---|--------|-------| +| 1 | `feat(eval): extend PipelineCfg with Phase 2 sub-configs and spend ceiling` | `src/eval/config.py`, `tests/eval/test_config.py` | +| 2 | `feat(eval): extend cost ledger to capture generator + judge + rewriter spend` | `src/eval/schemas.py` (`EvalResult.cost_breakdown`), `src/eval/runner.py` (`_score_question`), judge wrappers under `src/eval/metrics/`, aggregator, tests | +| 3 | `feat(eval): add BgeEmbedder as Chroma EmbeddingFunction adapter` | `src/eval/embedders/bge_small.py`, tests (incl. end-to-end through `ChromaVectorStore`) | +| 4 | `feat(eval): add BM25HybridRetriever with RRF fusion` | `src/eval/retrievers/bm25_hybrid.py`, tests, `requirements.txt` (+ `rank-bm25`) | +| 5 | `feat(eval): add CrossEncoderReranker (ms-marco-MiniLM)` | `src/eval/retrievers/reranker.py`, tests | +| 6 | `feat(eval): add QueryRewriter for LLM-based expansion (with cost capture)` | `src/eval/transforms/query_rewriter.py`, tests | +| 7 | `feat(eval): add RefusalHandler with similarity gate` | `src/eval/transforms/refusal_handler.py`, tests | +| 8 | `feat(eval): wire Phase 2 levers into build_pipeline + spend guardrail` | `src/eval/pipeline_factory.py`, factory tests, smoke test, runner spend-ceiling enforcement | +| 9 | `feat(eval): add `archive` subcommand to copy small run artifacts to a tracked tree` | `src/eval/cli.py`, tests | +| 10 | `chore(eval): add Phase 2 tier configs under configs/eval/phase2/` | 9 YAML files | + +**Acceptance for PR-A merge:** +- All unit + factory + smoke tests green. +- `pip install -r requirements.txt` adds exactly one entry (`rank-bm25`). +- Existing baseline configs (`baseline.yaml`, `baseline_squad_only.yaml`) still load and produce the same pipeline shape as before Phase 2 (tested in commit 1). +- `python -m src.eval.cli run --config configs/eval/phase2/phase2_baseline.yaml` succeeds end-to-end with stubbed-LLM mode. +- `python -m src.eval.cli archive --to /tmp/test_archive/` produces a folder containing exactly `metrics.json`, `cost.json`, `metadata.json`, `config.yaml`. + +### 5.2 PR-B — experiments + writeup (8 distinct runs + 1 doc, data-only) + +Branched off whatever PR-A merges into. Title: `docs(eval): Phase 2 RAG quality matrix — results + findings`. + +**Artifact location (resolves spec-review M6).** `eval_runs/` is gitignored at `.gitignore:15` and stays that way (run directories can be hundreds of MB and contain large `questions.jsonl` files). PR-B does **not** commit the live `eval_runs//` directories. Instead, PR-A adds a small CLI helper: + +```bash +python -m src.eval.cli archive --to docs/phase2/runs// +``` + +which copies only the small, reviewable artifacts (`metrics.json`, `cost.json`, `metadata.json`, `config.yaml`) into the tracked location. `questions.jsonl` is excluded from the archive (large) but its SHA goes into `metadata.json` so reviewers can verify the writeup against a re-run. + +| # | Commit | Content | +|---|--------|---------| +| 1 | `chore(eval): execute Phase 2 baseline run` | `docs/phase2/runs/_phase2_baseline/{metrics,cost,metadata,config}.json` (run dir stays local in `eval_runs/`) | +| 2–6 | one commit per tier 2b, 2c, 2d, 2e, 2g | archived artifacts per tier | +| 7 | `chore(eval): execute Phase 2 tier 2f — answer-model comparison (gpt-4.1-mini, claude-haiku-4-5)` | 2 archived run dirs (gpt-5-mini variant re-uses `phase2g_refusal` artifact) | +| 8 | `feat(eval): pairwise compare reports for the Phase 2 matrix` | HTML reports under `docs/phase2/compare/` | +| 9 | `docs(eval): Phase 2 results — methodology, chart, findings` | `docs/PHASE2_RESULTS.md` | +| 10 | `docs(readme): link Phase 2 results from main README` | one-liner + link in `README.md` | + +**Acceptance for PR-B merge:** +- 8 archived run dirs land under `docs/phase2/runs/`. The full `eval_runs//` continues to render in the local `/eval` UI but isn't part of the PR diff. +- `docs/PHASE2_RESULTS.md` contains: methodology, per-tier metric chart, paired-significance table for the 5 chain comparisons + 2 cross-model comparisons, "winning stack" recipe, ≥ 1 finding per lever. +- API spend ledger from `cost.json` totals (now including judge + rewriter, see §4.6) documented in the writeup. Target ≤ $5 actual. + +### 5.3 Sequencing relative to Phase 1 + +PR-A stacks on Phase 1 PR #4 (`feature/eval-harness-1d`). When #4 merges, PR-A retargets to `main`. Same pattern as Phase 1 stack. + +### 5.4 Risks + mitigations + +| # | Risk | Mitigation | +|---|------|------------| +| R1 | A tier's metric drops sharply (e.g., hybrid hurts on SQuAD because BM25 dominates and dense gets averaged down) | Hard-budget rule: keep the data, write the negative finding. Don't tune to fit. | +| R2 | Cross-encoder model download fails offline | Cache in `~/.cache/huggingface`; tier 2d test asserts model loads from cache after first pull. | +| R3 | `rank-bm25` adds tokenization cost on long ML-papers chunks | SQuAD-only matrix sidesteps this. Phase 3 gets a pre-tokenized corpus index. | +| R4 | LLM-based query rewriting adds 200 extra LLM calls (one per query) → cost spike | Use `gpt-4.1-nano` for rewriting. Estimated marginal: $0.02/run. | +| R5 | Refusal handler over-refuses → answer_correctness regresses | Threshold 0.35 starts conservative. Tier 2g exists to *measure* this trade-off; both metrics reported. | +| R6 | Tier 2c (hybrid) requires re-indexing Chroma with BGE embeddings; tier 2b's index isn't reusable | EvalPipeline factory rebuilds the index per run anyway (existing pattern). No new infra. | +| R7 | `eval_runs/` is gitignored, so committing live run dirs would either fail or require force-add | Resolved by §5.2: `cli archive` copies the small artifacts (`metrics.json`, `cost.json`, `metadata.json`, `config.yaml`) into the tracked `docs/phase2/runs/` tree. `questions.jsonl` SHA is recorded in `metadata.json` for reviewer-side verification. | +| R8 | Hand-running 8 distinct evals takes hours | Add a `make phase2-matrix` target that runs the full sweep with one command. Total wall time estimate: 5–7 hours unattended. | + +--- + +## 6. Decisions Locked in This Spec + +For the implementation plan to refer back to: + +1. **Lane:** end-to-end overhaul (lane 3 of brainstorm Q2). +2. **Structure:** layered stack with retrieval-quality-first ordering (option A of brainstorm Q5). +3. **Eval set:** SQuAD-only (deferring `ml_papers_v1` to Phase 3). +4. **Budget rule:** hard $5 ceiling, no tier rollback (option 1 of brainstorm Q6). Enforced per-run via `eval.spend_ceiling_usd` (§4.6) covering generator + judge + rewriter call sites. +5. **Deliverable:** data + portfolio writeup (option 2 of brainstorm Q7). +6. **Approach:** 2 PRs — PR-A pipeline extensions, PR-B experiments + writeup (option 3 of brainstorm Q-final). +7. **Branching:** off `feature/eval-harness-1d`; retargets to `main` after Phase 1 merges. + +### 6.1 Revisions from spec review (2026-04-27) + +- **H1.** Tier 2a (semantic chunking) deferred to Phase 3 — chunker is bypassed on SQuAD ingest path. Matrix shrinks to 9 YAMLs / 8 distinct runs. +- **H2.** Factory references updated to `src/eval/pipeline_factory.py::build_pipeline(config, dataset_name, ...)`. `EvalRunner` lives in `src/eval/runner.py`. +- **H3.** `BgeEmbedder` is a Chroma `EmbeddingFunction` adapter, installed on the collection at creation time. No per-call code change in `EvalPipeline`. Factory creates a fresh collection per run, eliminating dimension-mixing risk. +- **M4.** Cost ledger covers all three LLM call sites (generator, judge, rewriter) via `EvalResult.cost_breakdown`; spend ceiling enforced at runner level (§4.6). +- **M5.** Tier 2f reframed as a parallel three-way answer-model comparison on top of the full 2g stack. Tier 2g pins `generator.model: gpt-5-mini` to keep the chain's model variable constant. +- **M6.** Run artifacts archived to `docs/phase2/runs/` (tracked) via a new `cli archive` subcommand. Live `eval_runs/` stays gitignored. +- **L7.** RRF unit test rewritten with asymmetric inputs (`A=[a,b,c,d]`, `B=[d,a]`) so all four scores are distinct. diff --git a/eval_data/ml_papers_v1/LABELING_GUIDE.md b/eval_data/ml_papers_v1/LABELING_GUIDE.md new file mode 100644 index 00000000..1e8f3e7c --- /dev/null +++ b/eval_data/ml_papers_v1/LABELING_GUIDE.md @@ -0,0 +1,102 @@ +# ML Papers v1 — Labeling Guide + +This document is the authoritative rubric for hand-labeling the 50-question +ML Papers v1 dev set. Every question added to `questions.jsonl` MUST +satisfy these rules. The file itself is part of the eval contract — if +the rubric changes, the dev set version (`v1`) must change. + +## Corpus + +The corpus is 5–10 ML/AI papers listed in `corpus_manifest.json`. Each +entry pins the exact PDF by SHA-256 so the corpus is byte-stable. + +## Question schema + +Each row in `questions.jsonl` is a single `EvalQuestion` (Pydantic): + +```json +{ + "id": "", + "question": "Natural-language question.", + "gold_answer": "Concise reference answer (or null if unanswerable).", + "gold_chunk_ids": ["chunk_id_1", "chunk_id_2"], + "is_unanswerable": false, + "metadata": { + "source_paper": "attention-is-all-you-need", + "section": "3.2 Multi-Head Attention", + "difficulty": "definition" | "reasoning" | "multi_hop", + "labeler": "" + } +} +``` + +`chunk_id` is the SHA-256-prefix ID assigned by the chunker when the +PDF is ingested. To find the right `chunk_id`, ingest the corpus first +(`python -m src.eval.datasets.ml_papers --ingest`) and inspect the +collection. + +## What makes a good question + +A good question: + +- **Has exactly one defensible answer** in natural language. If two + knowledgeable readers would write different answers, rephrase. +- **Cannot be answered from the question text alone.** "What is + attention?" is bad — too vague. "In Vaswani et al. 2017, what is the + scaling factor applied to QK^T before softmax?" is good. +- **Has gold_chunk_ids that *causally* support the answer.** A chunk + that merely mentions the topic doesn't qualify; the chunk must + contain the information needed to derive the answer. +- **Has at most 3 gold_chunk_ids.** If you need more, the question is + too broad — split it. + +## Difficulty buckets (target distribution: ~20 / ~20 / ~10) + +- **definition** (20 questions) — single-fact lookup. *"What activation + function is used in the standard Transformer FFN?"* +- **reasoning** (20 questions) — requires understanding within one + passage. *"Why does scaled dot-product attention divide by √d_k?"* +- **multi_hop** (10 questions) — requires synthesizing across two or + more chunks (possibly across papers). *"How does ColBERT's late + interaction differ from DPR's bi-encoder design?"* + +## Unanswerable questions (target: 10 of 50) + +These must be: +- Plausibly on-topic for the corpus (e.g. an ML question about a model + the corpus does not cover). +- **Not** trivially off-topic (e.g. "what's the capital of France?" + doesn't test refusal usefully). +- Marked with `is_unanswerable: true`, `gold_answer: null`, + `gold_chunk_ids: []`. + +Examples: +- *"What batch size does BGE-M3 use during pre-training?"* — BGE-M3 + is not in the v1 corpus → unanswerable. +- *"What learning rate does the LoRA paper recommend for ResNet-50?"* + — LoRA paper does not cover ResNet → unanswerable. + +## Anti-patterns (DO NOT label these) + +- Questions whose answer is in the question. ("What is Section 3 + about?" if the chunk header is "Section 3: Attention".) +- Questions requiring outside knowledge not in the corpus. +- Questions with ambiguous wording ("Is X better?" — better at what?). +- Questions where the gold chunk is the question's own paraphrase. + +## Workflow + +1. Read one paper at a time. Take notes on questions that arise naturally. +2. For each question, find the supporting chunk in the ingested + collection (the runner exposes a helper). +3. Write the question, gold answer, and gold_chunk_ids to a draft file. +4. Self-review against this guide. +5. Append to `questions.jsonl` (one JSON object per line). +6. After each batch, run `python -m src.eval.datasets.ml_papers --validate` + to check schema and SHA-stability. + +## Versioning + +Any change to this guide that affects what counts as a valid question +requires bumping the version (`v1` → `v2`) and creating a new directory. +Do not edit `v1` after first publication. diff --git a/eval_data/ml_papers_v1/corpus_manifest.json b/eval_data/ml_papers_v1/corpus_manifest.json new file mode 100644 index 00000000..cbf2e36e --- /dev/null +++ b/eval_data/ml_papers_v1/corpus_manifest.json @@ -0,0 +1,6 @@ +{ + "version": "v1", + "description": "ML/AI literature corpus for the ml_papers_v1 dev set.", + "papers": [], + "comment_for_labelers": "Populate `papers` during corpus assembly. Each entry: {\"id\": \"slug\", \"title\": \"...\", \"source_url\": \"...\", \"local_path\": \"books/.pdf\", \"sha256\": \"\"}." +} diff --git a/eval_data/ml_papers_v1/questions.jsonl b/eval_data/ml_papers_v1/questions.jsonl new file mode 100644 index 00000000..e69de29b diff --git a/eval_data/squad_v2_dev_200/questions.jsonl b/eval_data/squad_v2_dev_200/questions.jsonl new file mode 100644 index 00000000..8fcee99d --- /dev/null +++ b/eval_data/squad_v2_dev_200/questions.jsonl @@ -0,0 +1,200 @@ +{"id":"10fba1a08c14e09d","question":"What is the height of the Swiss canton?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Rhine","context":"Near Tamins-Reichenau the Anterior Rhine and the Posterior Rhine join and form the Rhine. The river makes a distinctive turn to the north near Chur. This section is nearly 86 km long, and descends from a height of 599 m to 396 m. It flows through a wide glacial alpine valley known as the Rhine Valley (German: Rheintal). Near Sargans a natural dam, only a few metres high, prevents it from flowing into the open Seeztal valley and then through Lake Walen and Lake Zurich into the river Aare. The Alpine Rhine begins in the most western part of the Swiss canton of Graubünden, and later forms the border between Switzerland to the West and Liechtenstein and later Austria to the East."}} +{"id":"d599373aaad63b7a","question":"What US war caused a high amount of civil disobedience?","gold_answer":"Vietnam War","gold_chunk_ids":["d599373aaad63b7a"],"is_unanswerable":false,"metadata":{"title":"Civil_disobedience","context":"Courts have distinguished between two types of civil disobedience: \"Indirect civil disobedience involves violating a law which is not, itself, the object of protest, whereas direct civil disobedience involves protesting the existence of a particular law by breaking that law.\" During the Vietnam War, courts typically refused to excuse the perpetrators of illegal protests from punishment on the basis of their challenging the legality of the Vietnam War; the courts ruled it was a political question. The necessity defense has sometimes been used as a shadow defense by civil disobedients to deny guilt without denouncing their politically motivated acts, and to present their political beliefs in the courtroom. However, court cases such as U.S. v. Schoon have greatly curtailed the availability of the political necessity defense. Likewise, when Carter Wentworth was charged for his role in the Clamshell Alliance's 1977 illegal occupation of the Seabrook Station Nuclear Power Plant, the judge instructed the jury to disregard his competing harms defense, and he was found guilty. Fully Informed Jury Association activists have sometimes handed out educational leaflets inside courthouses despite admonitions not to; according to FIJA, many of them have escaped prosecution because \"prosecutors have reasoned (correctly) that if they arrest fully informed jury leafleters, the leaflets will have to be given to the leafleter's own jury as evidence.\""}} +{"id":"cc485fde8c8ca635","question":"Who found that there was a developed culture of Commissioner's who lacked responsibility?","gold_answer":"Committee of Independent Experts","gold_chunk_ids":["cc485fde8c8ca635"],"is_unanswerable":false,"metadata":{"title":"European_Union_law","context":"Commissioners have various privileges, such as being exempt from member state taxes (but not EU taxes), and having immunity from prosecution for doing official acts. Commissioners have sometimes been found to have abused their offices, particularly since the Santer Commission was censured by Parliament in 1999, and it eventually resigned due to corruption allegations. This resulted in one main case, Commission v Edith Cresson where the European Court of Justice held that a Commissioner giving her dentist a job, for which he was clearly unqualified, did in fact not break any law. By contrast to the ECJ's relaxed approach, a Committee of Independent Experts found that a culture had developed where few Commissioners had ‘even the slightest sense of responsibility’. This led to the creation of the European Anti-fraud Office. In 2012 it investigated the Maltese Commissioner for Health, John Dalli, who quickly resigned after allegations that he received a €60m bribe in connection with a Tobacco Products Directive. Beyond the Commission, the European Central Bank has relative executive autonomy in its conduct of monetary policy for the purpose of managing the euro. It has a six-person board appointed by the European Council, on the Council's recommendation. The President of the Council and a Commissioner can sit in on ECB meetings, but do not have voting rights."}} +{"id":"be631c29d01bbf3a","question":"What type of topological systems are found in numbers in Victoria?","gold_answer":"river systems","gold_chunk_ids":["be631c29d01bbf3a"],"is_unanswerable":false,"metadata":{"title":"Victoria_(Australia)","context":"Victoria contains many topographically, geologically and climatically diverse areas, ranging from the wet, temperate climate of Gippsland in the southeast to the snow-covered Victorian alpine areas which rise to almost 2,000 m (6,600 ft), with Mount Bogong the highest peak at 1,986 m (6,516 ft). There are extensive semi-arid plains to the west and northwest. There is an extensive series of river systems in Victoria. Most notable is the Murray River system. Other rivers include: Ovens River, Goulburn River, Patterson River, King River, Campaspe River, Loddon River, Wimmera River, Elgin River, Barwon River, Thomson River, Snowy River, Latrobe River, Yarra River, Maribyrnong River, Mitta River, Hopkins River, Merri River and Kiewa River. The state symbols include the pink heath (state flower), Leadbeater's possum (state animal) and the helmeted honeyeater (state bird)."}} +{"id":"7d4c6cb1df1747a1","question":"In U.S. states, what happens to the life expectancy in more economically equal ones?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Economic_inequality","context":"Effects of inequality researchers have found include higher rates of health and social problems, and lower rates of social goods, a lower level of economic utility in society from resources devoted on high-end consumption, and even a lower level of economic growth when human capital is neglected for high-end consumption. For the top 21 industrialised countries, counting each person equally, life expectancy is lower in more unequal countries (r = -.907). A similar relationship exists among US states (r = -.620)."}} +{"id":"f63fce7dedda0e5d","question":"What is the solubility water in oxygen dependent on?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Oxygen","context":"Oxygen is more soluble in water than nitrogen is. Water in equilibrium with air contains approximately 1 molecule of dissolved O\n2 for every 2 molecules of N\n2, compared to an atmospheric ratio of approximately 1:4. The solubility of oxygen in water is temperature-dependent, and about twice as much (14.6 mg·L−1) dissolves at 0 °C than at 20 °C (7.6 mg·L−1). At 25 °C and 1 standard atmosphere (101.3 kPa) of air, freshwater contains about 6.04 milliliters (mL) of oxygen per liter, whereas seawater contains about 4.95 mL per liter. At 5 °C the solubility increases to 9.0 mL (50% more than at 25 °C) per liter for water and 7.2 mL (45% more) per liter for sea water."}} +{"id":"5b1a5844958c84a4","question":"In what year did Harvard Stadium become the first ever concrete reinforced stadium in the country?","gold_answer":"1903","gold_chunk_ids":["5b1a5844958c84a4"],"is_unanswerable":false,"metadata":{"title":"Harvard_University","context":"Harvard's athletic rivalry with Yale is intense in every sport in which they meet, coming to a climax each fall in the annual football meeting, which dates back to 1875 and is usually called simply \"The Game\". While Harvard's football team is no longer one of the country's best as it often was a century ago during football's early days (it won the Rose Bowl in 1920), both it and Yale have influenced the way the game is played. In 1903, Harvard Stadium introduced a new era into football with the first-ever permanent reinforced concrete stadium of its kind in the country. The stadium's structure actually played a role in the evolution of the college game. Seeking to reduce the alarming number of deaths and serious injuries in the sport, Walter Camp (former captain of the Yale football team), suggested widening the field to open up the game. But the stadium was too narrow to accommodate a wider playing surface. So, other steps had to be taken. Camp would instead support revolutionary new rules for the 1906 season. These included legalizing the forward pass, perhaps the most significant rule change in the sport's history."}} +{"id":"81301cb3804c8bf3","question":"What doesn't change from being at rest to movement at a constant velocity?","gold_answer":"laws of physics","gold_chunk_ids":["81301cb3804c8bf3"],"is_unanswerable":false,"metadata":{"title":"Force","context":"For instance, while traveling in a moving vehicle at a constant velocity, the laws of physics do not change from being at rest. A person can throw a ball straight up in the air and catch it as it falls down without worrying about applying a force in the direction the vehicle is moving. This is true even though another person who is observing the moving vehicle pass by also observes the ball follow a curving parabolic path in the same direction as the motion of the vehicle. It is the inertia of the ball associated with its constant velocity in the direction of the vehicle's motion that ensures the ball continues to move forward even as it is thrown up and falls back down. From the perspective of the person in the car, the vehicle and everything inside of it is at rest: It is the outside world that is moving with a constant speed in the opposite direction. Since there is no experiment that can distinguish whether it is the vehicle that is at rest or the outside world that is at rest, the two situations are considered to be physically indistinguishable. Inertia therefore applies equally well to constant velocity motion as it does to rest."}} +{"id":"ad548d4b70004852","question":"What early Huguenot Church was established in England?","gold_answer":"The French Protestant Church of London","gold_chunk_ids":["ad548d4b70004852"],"is_unanswerable":false,"metadata":{"title":"Huguenot","context":"The French Protestant Church of London was established by Royal Charter in 1550. It is now located at Soho Square. Huguenot refugees flocked to Shoreditch, London. They established a major weaving industry in and around Spitalfields (see Petticoat Lane and the Tenterground) in East London. In Wandsworth, their gardening skills benefited the Battersea market gardens. The Old Truman Brewery, then known as the Black Eagle Brewery, was founded in 1724. The flight of Huguenot refugees from Tours, France drew off most of the workers of its great silk mills which they had built.[citation needed] Some of these immigrants moved to Norwich, which had accommodated an earlier settlement of Walloon weavers. The French added to the existing immigrant population, then comprising about a third of the population of the city."}} +{"id":"cc03a9a500b15400","question":"Who ruined Alexius Komnenos plans for an independent state?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Normans","context":"One of the first Norman mercenaries to serve as a Byzantine general was Hervé in the 1050s. By then however, there were already Norman mercenaries serving as far away as Trebizond and Georgia. They were based at Malatya and Edessa, under the Byzantine duke of Antioch, Isaac Komnenos. In the 1060s, Robert Crispin led the Normans of Edessa against the Turks. Roussel de Bailleul even tried to carve out an independent state in Asia Minor with support from the local population, but he was stopped by the Byzantine general Alexius Komnenos."}} +{"id":"f07b4833d539da96","question":"What are two anti-inflammatory molecules that peak during awake hours?","gold_answer":"cortisol and catecholamines","gold_chunk_ids":["f07b4833d539da96"],"is_unanswerable":false,"metadata":{"title":"Immune_system","context":"In contrast, during wake periods differentiated effector cells, such as cytotoxic natural killer cells and CTLs (cytotoxic T lymphocytes), peak in order to elicit an effective response against any intruding pathogens. As well during awake active times, anti-inflammatory molecules, such as cortisol and catecholamines, peak. There are two theories as to why the pro-inflammatory state is reserved for sleep time. First, inflammation would cause serious cognitive and physical impairments if it were to occur during wake times. Second, inflammation may occur during sleep times due to the presence of melatonin. Inflammation causes a great deal of oxidative stress and the presence of melatonin during sleep times could actively counteract free radical production during this time."}} +{"id":"0514a90ffe3d5818","question":"Of what form is the infinite amount of primes that comprise the special cases of Schinzel's hypothesis?","gold_answer":"n2 + 1","gold_chunk_ids":["0514a90ffe3d5818"],"is_unanswerable":false,"metadata":{"title":"Prime_number","context":"A third type of conjectures concerns aspects of the distribution of primes. It is conjectured that there are infinitely many twin primes, pairs of primes with difference 2 (twin prime conjecture). Polignac's conjecture is a strengthening of that conjecture, it states that for every positive integer n, there are infinitely many pairs of consecutive primes that differ by 2n. It is conjectured there are infinitely many primes of the form n2 + 1. These conjectures are special cases of the broad Schinzel's hypothesis H. Brocard's conjecture says that there are always at least four primes between the squares of consecutive primes greater than 2. Legendre's conjecture states that there is a prime number between n2 and (n + 1)2 for every positive integer n. It is implied by the stronger Cramér's conjecture."}} +{"id":"c2f4b0efca0a877b","question":"What involves mass production of similar items without designated planning?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Construction","context":"Construction is the process of constructing a building or infrastructure. Construction differs from manufacturing in that manufacturing typically involves mass production of similar items without a designated purchaser, while construction typically takes place on location for a known client. Construction as an industry comprises six to nine percent of the gross domestic product of developed countries. Construction starts with planning,[citation needed] design, and financing and continues until the project is built and ready for use."}} +{"id":"9962dcf56bf1fc5f","question":"When was the Miller-Urey experiment conducted?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"University_of_Chicago","context":"The University of Chicago has been the site of some important experiments and academic movements. In economics, the university has played an important role in shaping ideas about the free market and is the namesake of the Chicago school of economics, the school of economic thought supported by Milton Friedman and other economists. The university's sociology department was the first independent sociology department in the United States and gave birth to the Chicago school of sociology. In physics, the university was the site of the Chicago Pile-1 (the first self-sustained man-made nuclear reaction, part of the Manhattan Project), of Robert Millikan's oil-drop experiment that calculated the charge of the electron, and of the development of radiocarbon dating by Willard F. Libby in 1947. The chemical experiment that tested how life originated on early Earth, the Miller–Urey experiment, was conducted at the university. REM sleep was discovered at the university in 1953 by Nathaniel Kleitman and Eugene Aserinsky."}} +{"id":"bfc265123570fd06","question":"What immune response is not antigen-specific?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Immune_system","context":"The adaptive immune system evolved in early vertebrates and allows for a stronger immune response as well as immunological memory, where each pathogen is \"remembered\" by a signature antigen. The adaptive immune response is antigen-specific and requires the recognition of specific \"non-self\" antigens during a process called antigen presentation. Antigen specificity allows for the generation of responses that are tailored to specific pathogens or pathogen-infected cells. The ability to mount these tailored responses is maintained in the body by \"memory cells\". Should a pathogen infect the body more than once, these specific memory cells are used to quickly eliminate it."}} +{"id":"a60acb978e1c96be","question":"What type of climate does Jacksonville have?","gold_answer":"subtropical","gold_chunk_ids":["a60acb978e1c96be"],"is_unanswerable":false,"metadata":{"title":"Jacksonville,_Florida","context":"Like much of the south Atlantic region of the United States, Jacksonville has a humid subtropical climate (Köppen Cfa), with mild weather during winters and hot and humid weather during summers. Seasonal rainfall is concentrated in the warmest months from May through September, while the driest months are from November through April. Due to Jacksonville's low latitude and coastal location, the city sees very little cold weather, and winters are typically mild and sunny. Summers can be hot and wet, and summer thunderstorms with torrential but brief downpours are common."}} +{"id":"a0209a8108abdac2","question":"What county are Los Angeles, Orange, San Diego, San Bernardino, and Riverside located in?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Southern_California","context":"Its counties of Los Angeles, Orange, San Diego, San Bernardino, and Riverside are the five most populous in the state and all are in the top 15 most populous counties in the United States."}} +{"id":"9388247353847b7d","question":"What government set standards do only select schools have to meet?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Victoria_(Australia)","context":"Victorian schools are either publicly or privately funded. Public schools, also known as state or government schools, are funded and run directly by the Victoria Department of Education . Students do not pay tuition fees, but some extra costs are levied. Private fee-paying schools include parish schools run by the Roman Catholic Church and independent schools similar to British public schools. Independent schools are usually affiliated with Protestant churches. Victoria also has several private Jewish and Islamic primary and secondary schools. Private schools also receive some public funding. All schools must comply with government-set curriculum standards. In addition, Victoria has four government selective schools, Melbourne High School for boys, MacRobertson Girls' High School for girls, the coeducational schools John Monash Science School, Nossal High School and Suzanne Cory High School, and The Victorian College of the Arts Secondary School. Students at these schools are exclusively admitted on the basis of an academic selective entry test."}} +{"id":"2b2256ceabf10610","question":"What law staes that forces are interactions between bodies?","gold_answer":"Newton's Third","gold_chunk_ids":["2b2256ceabf10610"],"is_unanswerable":false,"metadata":{"title":"Force","context":"Newton's Third Law is a result of applying symmetry to situations where forces can be attributed to the presence of different objects. The third law means that all forces are interactions between different bodies,[Note 3] and thus that there is no such thing as a unidirectional force or a force that acts on only one body. Whenever a first body exerts a force F on a second body, the second body exerts a force −F on the first body. F and −F are equal in magnitude and opposite in direction. This law is sometimes referred to as the action-reaction law, with F called the \"action\" and −F the \"reaction\". The action and the reaction are simultaneous:"}} +{"id":"ee3f2fa33bac1254","question":"When was the Ottoman Caliphate abolished?","gold_answer":"1924","gold_chunk_ids":["ee3f2fa33bac1254"],"is_unanswerable":false,"metadata":{"title":"Islamism","context":"In its focus on the Caliphate, the party takes a different view of Muslim history than some other Islamists such as Muhammad Qutb. HT sees Islam's pivotal turning point as occurring not with the death of Ali, or one of the other four rightly guided Caliphs in the 7th century, but with the abolition of the Ottoman Caliphate in 1924. This is believed to have ended the true Islamic system, something for which it blames \"the disbelieving (Kafir) colonial powers\" working through Turkish modernist Mustafa Kemal Atatürk."}} +{"id":"962dd05c6249477f","question":"What religion was Hugues Capet?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Huguenot","context":"Some disagree with such double or triple non-French linguistic origins, arguing that for the word to have spread into common use in France, it must have originated in the French language. The \"Hugues hypothesis\" argues that the name was derived by association with Hugues Capet, king of France, who reigned long before the Reformation. He was regarded by the Gallicans and Protestants as a noble man who respected people's dignity and lives. Janet Gray and other supporters of the hypothesis suggest that the name huguenote would be roughly equivalent to little Hugos, or those who want Hugo."}} +{"id":"c2f70a4d73769317","question":"Why are normal body cells attacked by NK cells?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Immune_system","context":"Natural killer cells, or NK cells, are a component of the innate immune system which does not directly attack invading microbes. Rather, NK cells destroy compromised host cells, such as tumor cells or virus-infected cells, recognizing such cells by a condition known as \"missing self.\" This term describes cells with low levels of a cell-surface marker called MHC I (major histocompatibility complex) – a situation that can arise in viral infections of host cells. They were named \"natural killer\" because of the initial notion that they do not require activation in order to kill cells that are \"missing self.\" For many years it was unclear how NK cells recognize tumor cells and infected cells. It is now known that the MHC makeup on the surface of those cells is altered and the NK cells become activated through recognition of \"missing self\". Normal body cells are not recognized and attacked by NK cells because they express intact self MHC antigens. Those MHC antigens are recognized by killer cell immunoglobulin receptors (KIR) which essentially put the brakes on NK cells."}} +{"id":"61fdd38587ee9c66","question":"When did British begin to build fort under William Trent?","gold_answer":"early months of 1754","gold_chunk_ids":["61fdd38587ee9c66"],"is_unanswerable":false,"metadata":{"title":"French_and_Indian_War","context":"Even before Washington returned, Dinwiddie had sent a company of 40 men under William Trent to that point, where in the early months of 1754 they began construction of a small stockaded fort. Governor Duquesne sent additional French forces under Claude-Pierre Pecaudy de Contrecœur to relieve Saint-Pierre during the same period, and Contrecœur led 500 men south from Fort Venango on April 5, 1754. When these forces arrived at the fort on April 16, Contrecœur generously allowed Trent's small company to withdraw. He purchased their construction tools to continue building what became Fort Duquesne."}} +{"id":"8da4ccffd7bb5fa1","question":"Since Thoreau was not a well known writer what happened when he was arrested?","gold_answer":"was not covered in any newspapers","gold_chunk_ids":["8da4ccffd7bb5fa1"],"is_unanswerable":false,"metadata":{"title":"Civil_disobedience","context":"The earliest recorded incidents of collective civil disobedience took place during the Roman Empire[citation needed]. Unarmed Jews gathered in the streets to prevent the installation of pagan images in the Temple in Jerusalem.[citation needed][original research?] In modern times, some activists who commit civil disobedience as a group collectively refuse to sign bail until certain demands are met, such as favorable bail conditions, or the release of all the activists. This is a form of jail solidarity.[page needed] There have also been many instances of solitary civil disobedience, such as that committed by Thoreau, but these sometimes go unnoticed. Thoreau, at the time of his arrest, was not yet a well-known author, and his arrest was not covered in any newspapers in the days, weeks and months after it happened. The tax collector who arrested him rose to higher political office, and Thoreau's essay was not published until after the end of the Mexican War."}} +{"id":"9011d237027b0d6f","question":"What does an increase in the income share of the bottom 20 percent of people of a society result in?","gold_answer":"higher GDP growth","gold_chunk_ids":["9011d237027b0d6f"],"is_unanswerable":false,"metadata":{"title":"Economic_inequality","context":"According to International Monetary Fund economists, inequality in wealth and income is negatively correlated with the duration of economic growth spells (not the rate of growth). High levels of inequality prevent not just economic prosperity, but also the quality of a country's institutions and high levels of education. According to IMF staff economists, \"if the income share of the top 20 percent (the rich) increases, then GDP growth actually declines over the medium term, suggesting that the benefits do not trickle down. In contrast, an increase in the income share of the bottom 20 percent (the poor) is associated with higher GDP growth. The poor and the middle class matter the most for growth via a number of interrelated economic, social, and political channels.\""}} +{"id":"9040d4c214616f9f","question":"How many people lived in Oslo at the start of the plague outbreak in 1654?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Black_Death","context":"In 1466, perhaps 40,000 people died of the plague in Paris. During the 16th and 17th centuries, the plague was present in Paris around 30 per cent of the time. The Black Death ravaged Europe for three years before it continued on into Russia, where the disease was present somewhere in the country 25 times between 1350 to 1490. Plague epidemics ravaged London in 1563, 1593, 1603, 1625, 1636, and 1665, reducing its population by 10 to 30% during those years. Over 10% of Amsterdam's population died in 1623–25, and again in 1635–36, 1655, and 1664. Plague occurred in Venice 22 times between 1361 and 1528. The plague of 1576–77 killed 50,000 in Venice, almost a third of the population. Late outbreaks in central Europe included the Italian Plague of 1629–1631, which is associated with troop movements during the Thirty Years' War, and the Great Plague of Vienna in 1679. Over 60% of Norway's population died in 1348–50. The last plague outbreak ravaged Oslo in 1654."}} +{"id":"6f49340aaf086321","question":"What is the least critical resource measured in assessing the determination of a Turing machine's ability to solve any given set of problems?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Computational_complexity_theory","context":"For a precise definition of what it means to solve a problem using a given amount of time and space, a computational model such as the deterministic Turing machine is used. The time required by a deterministic Turing machine M on input x is the total number of state transitions, or steps, the machine makes before it halts and outputs the answer (\"yes\" or \"no\"). A Turing machine M is said to operate within time f(n), if the time required by M on each input of length n is at most f(n). A decision problem A can be solved in time f(n) if there exists a Turing machine operating in time f(n) that solves the problem. Since complexity theory is interested in classifying problems based on their difficulty, one defines sets of problems based on some criteria. For instance, the set of problems solvable within time f(n) on a deterministic Turing machine is then denoted by DTIME(f(n))."}} +{"id":"37226a6e9fb23bab","question":"What pope as a native of Poland?","gold_answer":"John Paul II","gold_chunk_ids":["37226a6e9fb23bab"],"is_unanswerable":false,"metadata":{"title":"Warsaw","context":"John Paul II's visits to his native country in 1979 and 1983 brought support to the budding solidarity movement and encouraged the growing anti-communist fervor there. In 1979, less than a year after becoming pope, John Paul celebrated Mass in Victory Square in Warsaw and ended his sermon with a call to \"renew the face\" of Poland: Let Thy Spirit descend! Let Thy Spirit descend and renew the face of the land! This land! These words were very meaningful for the Polish citizens who understood them as the incentive for the democratic changes."}} +{"id":"0389899376762ec9","question":"Besides the study of prime numbers, what general theory was considered the official example of the military?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Prime_number","context":"For a long time, number theory in general, and the study of prime numbers in particular, was seen as the canonical example of pure mathematics, with no applications outside of the self-interest of studying the topic with the exception of use of prime numbered gear teeth to distribute wear evenly. In particular, number theorists such as British mathematician G. H. Hardy prided themselves on doing work that had absolutely no military significance. However, this vision was shattered in the 1970s, when it was publicly announced that prime numbers could be used as the basis for the creation of public key cryptography algorithms. Prime numbers are also used for hash tables and pseudorandom number generators."}} +{"id":"7c1f10c16cb90c74","question":"What did articles 1 to 4 not generally require of workers?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"European_Union_law","context":"The Free Movement of Workers Regulation articles 1 to 7 set out the main provisions on equal treatment of workers. First, articles 1 to 4 generally require that workers can take up employment, conclude contracts, and not suffer discrimination compared to nationals of the member state. In a famous case, the Belgian Football Association v Bosman, a Belgian footballer named Jean-Marc Bosman claimed that he should be able to transfer from R.F.C. de Liège to USL Dunkerque when his contract finished, regardless of whether Dunkerque could afford to pay Liège the habitual transfer fees. The Court of Justice held \"the transfer rules constitute[d] an obstacle to free movement\" and were unlawful unless they could be justified in the public interest, but this was unlikely. In Groener v Minister for Education the Court of Justice accepted that a requirement to speak Gaelic to teach in a Dublin design college could be justified as part of the public policy of promoting the Irish language, but only if the measure was not disproportionate. By contrast in Angonese v Cassa di Risparmio di Bolzano SpA a bank in Bolzano, Italy, was not allowed to require Mr Angonese to have a bilingual certificate that could only be obtained in Bolzano. The Court of Justice, giving \"horizontal\" direct effect to TFEU article 45, reasoned that people from other countries would have little chance of acquiring the certificate, and because it was \"impossible to submit proof of the required linguistic knowledge by any other means\", the measure was disproportionate. Second, article 7(2) requires equal treatment in respect of tax. In Finanzamt Köln Altstadt v Schumacker the Court of Justice held that it contravened TFEU art 45 to deny tax benefits (e.g. for married couples, and social insurance expense deductions) to a man who worked in Germany, but was resident in Belgium when other German residents got the benefits. By contrast in Weigel v Finanzlandesdirektion für Vorarlberg the Court of Justice rejected Mr Weigel's claim that a re-registration charge upon bringing his car to Austria violated his right to free movement. Although the tax was \"likely to have a negative bearing on the decision of migrant workers to exercise their right to freedom of movement\", because the charge applied equally to Austrians, in absence of EU legislation on the matter it had to be regarded as justified. Third, people must receive equal treatment regarding \"social advantages\", although the Court has approved residential qualifying periods. In Hendrix v Employee Insurance Institute the Court of Justice held that a Dutch national was not entitled to continue receiving incapacity benefits when he moved to Belgium, because the benefit was \"closely linked to the socio-economic situation\" of the Netherlands. Conversely, in Geven v Land Nordrhein-Westfalen the Court of Justice held that a Dutch woman living in the Netherlands, but working between 3 and 14 hours a week in Germany, did not have a right to receive German child benefits, even though the wife of a man who worked full-time in Germany but was resident in Austria could. The general justifications for limiting free movement in TFEU article 45(3) are \"public policy, public security or public health\", and there is also a general exception in article 45(4) for \"employment in the public service\"."}} +{"id":"c9f17e2f6e2f4de7","question":"Was wasn't the plan for Langlades mission?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"French_and_Indian_War","context":"On March 17, 1752, the Governor-General of New France, Marquis de la Jonquière, died and was temporarily replaced by Charles le Moyne de Longueuil. His permanent replacement, the Marquis Duquesne, did not arrive in New France until 1752 to take over the post. The continuing British activity in the Ohio territories prompted Longueuil to dispatch another expedition to the area under the command of Charles Michel de Langlade, an officer in the Troupes de la Marine. Langlade was given 300 men, including French-Canadians and warriors of the Ottawa. His objective was to punish the Miami people of Pickawillany for not following Céloron's orders to cease trading with the British. On June 21, the French war party attacked the trading centre at Pickawillany, capturing three traders and killing 14 people of the Miami nation, including Old Briton. He was reportedly ritually cannibalized by some aboriginal members of the expedition."}} +{"id":"3ed617c3595c1aa2","question":"When was Otto von Bismarck born?","gold_answer":"1862","gold_chunk_ids":["3ed617c3595c1aa2"],"is_unanswerable":false,"metadata":{"title":"Imperialism","context":"Not a maritime power, and not a nation-state, as it would eventually become, Germany’s participation in Western imperialism was negligible until the late 19th century. The participation of Austria was primarily as a result of Habsburg control of the First Empire, the Spanish throne, and other royal houses.[further explanation needed] After the defeat of Napoleon, who caused the dissolution of that Holy Roman Empire, Prussia and the German states continued to stand aloof from imperialism, preferring to manipulate the European system through the Concert of Europe. After Prussia unified the other states into the second German Empire after the Franco-German War, its long-time Chancellor, Otto von Bismarck (1862–90), long opposed colonial acquisitions, arguing that the burden of obtaining, maintaining, and defending such possessions would outweigh any potential benefits. He felt that colonies did not pay for themselves, that the German bureaucratic system would not work well in the tropics and the diplomatic disputes over colonies would distract Germany from its central interest, Europe itself."}} +{"id":"e66641ddfb218079","question":"What does the average temperatures exceed in the summer?","gold_answer":"32 °C","gold_chunk_ids":["e66641ddfb218079"],"is_unanswerable":false,"metadata":{"title":"Victoria_(Australia)","context":"The Mallee and upper Wimmera are Victoria's warmest regions with hot winds blowing from nearby semi-deserts. Average temperatures exceed 32 °C (90 °F) during summer and 15 °C (59 °F) in winter. Except at cool mountain elevations, the inland monthly temperatures are 2–7 °C (4–13 °F) warmer than around Melbourne (see chart). Victoria's highest maximum temperature since World War II, of 48.8 °C (119.8 °F) was recorded in Hopetoun on 7 February 2009, during the 2009 southeastern Australia heat wave."}} +{"id":"ff587fb384a78b3f","question":"Who was one of the scientists for Public Health England in 2014?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Black_Death","context":"A variety of alternatives to the Y. pestis have been put forward. Twigg suggested that the cause was a form of anthrax, and Norman Cantor (2001) thought it may have been a combination of anthrax and other pandemics. Scott and Duncan have argued that the pandemic was a form of infectious disease that characterise as hemorrhagic plague similar to Ebola. Archaeologist Barney Sloane has argued that there is insufficient evidence of the extinction of a large number of rats in the archaeological record of the medieval waterfront in London and that the plague spread too quickly to support the thesis that the Y. pestis was spread from fleas on rats; he argues that transmission must have been person to person. However, no single alternative solution has achieved widespread acceptance. Many scholars arguing for the Y. pestis as the major agent of the pandemic suggest that its extent and symptoms can be explained by a combination of bubonic plague with other diseases, including typhus, smallpox and respiratory infections. In addition to the bubonic infection, others point to additional septicemic (a type of \"blood poisoning\") and pneumonic (an airborne plague that attacks the lungs before the rest of the body) forms of the plague, which lengthen the duration of outbreaks throughout the seasons and help account for its high mortality rate and additional recorded symptoms. In 2014, scientists with Public Health England announced the results of an examination of 25 bodies exhumed from the Clerkenwell area of London, as well as of wills registered in London during the period, which supported the pneumonic hypothesis."}} +{"id":"9ed717e6c5d01d1d","question":"What used to be described by the Schrodinger equation?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Force","context":"The notion \"force\" keeps its meaning in quantum mechanics, though one is now dealing with operators instead of classical variables and though the physics is now described by the Schrödinger equation instead of Newtonian equations. This has the consequence that the results of a measurement are now sometimes \"quantized\", i.e. they appear in discrete portions. This is, of course, difficult to imagine in the context of \"forces\". However, the potentials V(x,y,z) or fields, from which the forces generally can be derived, are treated similar to classical position variables, i.e., ."}} +{"id":"4f424102f55a7192","question":"Did Baran develop this \"only\" for use by the Air Force?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Packet_switching","context":"Baran developed the concept of distributed adaptive message block switching during his research at the RAND Corporation for the US Air Force into survivable communications networks, first presented to the Air Force in the summer of 1961 as briefing B-265, later published as RAND report P-2626 in 1962, and finally in report RM 3420 in 1964. Report P-2626 described a general architecture for a large-scale, distributed, survivable communications network. The work focuses on three key ideas: use of a decentralized network with multiple paths between any two points, dividing user messages into message blocks, later called packets, and delivery of these messages by store and forward switching."}} +{"id":"9981b7986312abe8","question":"Who conceptualized the aeolipile?","gold_answer":"Hero of Alexandria","gold_chunk_ids":["9981b7986312abe8"],"is_unanswerable":false,"metadata":{"title":"Steam_engine","context":"The history of the steam engine stretches back as far as the first century AD; the first recorded rudimentary steam engine being the aeolipile described by Greek mathematician Hero of Alexandria. In the following centuries, the few steam-powered \"engines\" known were, like the aeolipile, essentially experimental devices used by inventors to demonstrate the properties of steam. A rudimentary steam turbine device was described by Taqi al-Din in 1551 and by Giovanni Branca in 1629. Jerónimo de Ayanz y Beaumont received patents in 1606 for fifty steam powered inventions, including a water pump for draining inundated mines. Denis Papin, a Huguenot refugee, did some useful work on the steam digester in 1679, and first used a piston to raise weights in 1690."}} +{"id":"a77961bdc308304a","question":"How much of the Noord River flows into the North Sea?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Rhine","context":"The other third of the water flows through the Pannerdens Kanaal and redistributes in the IJssel and Nederrijn. The IJssel branch carries one ninth of the water flow of the Rhine north into the IJsselmeer (a former bay), while the Nederrijn carries approximately two ninths of the flow west along a route parallel to the Waal. However, at Wijk bij Duurstede, the Nederrijn changes its name and becomes the Lek. It flows farther west, to rejoin the Noord River into the Nieuwe Maas and to the North Sea."}} +{"id":"c42765f14e10a21b","question":"Where is it unlikely that the first multicomponent, adaptive immune system arose?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Immune_system","context":"It is likely that a multicomponent, adaptive immune system arose with the first vertebrates, as invertebrates do not generate lymphocytes or an antibody-based humoral response. Many species, however, utilize mechanisms that appear to be precursors of these aspects of vertebrate immunity. Immune systems appear even in the structurally most simple forms of life, with bacteria using a unique defense mechanism, called the restriction modification system to protect themselves from viral pathogens, called bacteriophages. Prokaryotes also possess acquired immunity, through a system that uses CRISPR sequences to retain fragments of the genomes of phage that they have come into contact with in the past, which allows them to block virus replication through a form of RNA interference. Offensive elements of the immune systems are also present in unicellular eukaryotes, but studies of their roles in defense are few."}} +{"id":"ade80dd23e56a014","question":"How many seats does Australia have in the Senate?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Victoria_(Australia)","context":"Politically, Victoria has 37 seats in the Australian House of Representatives and 12 seats in the Australian Senate. At state level, the Parliament of Victoria consists of the Legislative Assembly (the lower house) and the Legislative Council (the upper house). Victoria is currently governed by the Labor Party, with Daniel Andrews the current Premier. The personal representative of the Queen of Australia in the state is the Governor of Victoria, currently Linda Dessau. Local government is concentrated in 79 municipal districts, including 33 cities, although a number of unincorporated areas still exist, which are administered directly by the state."}} +{"id":"0eba7d89247d1b1b","question":"What medical treatment is completely different from acquired immunity?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Immune_system","context":"Pathogens can rapidly evolve and adapt, and thereby avoid detection and neutralization by the immune system; however, multiple defense mechanisms have also evolved to recognize and neutralize pathogens. Even simple unicellular organisms such as bacteria possess a rudimentary immune system, in the form of enzymes that protect against bacteriophage infections. Other basic immune mechanisms evolved in ancient eukaryotes and remain in their modern descendants, such as plants and invertebrates. These mechanisms include phagocytosis, antimicrobial peptides called defensins, and the complement system. Jawed vertebrates, including humans, have even more sophisticated defense mechanisms, including the ability to adapt over time to recognize specific pathogens more efficiently. Adaptive (or acquired) immunity creates immunological memory after an initial response to a specific pathogen, leading to an enhanced response to subsequent encounters with that same pathogen. This process of acquired immunity is the basis of vaccination."}} +{"id":"8d33e7430ef3900a","question":"What is higher in countries with more inequality for the top 21 industrialized countries?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Economic_inequality","context":"Effects of inequality researchers have found include higher rates of health and social problems, and lower rates of social goods, a lower level of economic utility in society from resources devoted on high-end consumption, and even a lower level of economic growth when human capital is neglected for high-end consumption. For the top 21 industrialised countries, counting each person equally, life expectancy is lower in more unequal countries (r = -.907). A similar relationship exists among US states (r = -.620)."}} +{"id":"d87d8c53209a7363","question":"What does the mayor's council divide itself into?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Warsaw","context":"Legislative power in Warsaw is vested in a unicameral Warsaw City Council (Rada Miasta), which comprises 60 members. Council members are elected directly every four years. Like most legislative bodies, the City Council divides itself into committees which have the oversight of various functions of the city government. Bills passed by a simple majority are sent to the mayor (the President of Warsaw), who may sign them into law. If the mayor vetoes a bill, the Council has 30 days to override the veto by a two-thirds majority vote."}} +{"id":"f476bc0362795b7d","question":"What is the temperature in the valley of the mountain range in winter?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Victoria_(Australia)","context":"The Victorian Alps in the northeast are the coldest part of Victoria. The Alps are part of the Great Dividing Range mountain system extending east-west through the centre of Victoria. Average temperatures are less than 9 °C (48 °F) in winter and below 0 °C (32 °F) in the highest parts of the ranges. The state's lowest minimum temperature of −11.7 °C (10.9 °F) was recorded at Omeo on 13 June 1965, and again at Falls Creek on 3 July 1970. Temperature extremes for the state are listed in the table below:"}} +{"id":"b46ddf9b1b0c7175","question":"Why did natural sedimentation by the Rhine compensate the transgression bby the sea?","gold_answer":"Rates of sea-level rise","gold_chunk_ids":["b46ddf9b1b0c7175"],"is_unanswerable":false,"metadata":{"title":"Rhine","context":"Since 7500 yr ago, a situation with tides and currents, very similar to present has existed. Rates of sea-level rise had dropped so far, that natural sedimentation by the Rhine and coastal processes together, could compensate the transgression by the sea; in the last 7000 years, the coast line was roughly at the same location. In the southern North Sea, due to ongoing tectonic subsidence, the sea level is still rising, at the rate of about 1–3 cm (0.39–1.18 in) per century (1 metre or 39 inches in last 3000 years)."}} +{"id":"8484bcf614e16160","question":"What covers most of the Amazon basin of Central America?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Amazon_rainforest","context":"The Amazon rainforest (Portuguese: Floresta Amazônica or Amazônia; Spanish: Selva Amazónica, Amazonía or usually Amazonia; French: Forêt amazonienne; Dutch: Amazoneregenwoud), also known in English as Amazonia or the Amazon Jungle, is a moist broadleaf forest that covers most of the Amazon basin of South America. This basin encompasses 7,000,000 square kilometres (2,700,000 sq mi), of which 5,500,000 square kilometres (2,100,000 sq mi) are covered by the rainforest. This region includes territory belonging to nine nations. The majority of the forest is contained within Brazil, with 60% of the rainforest, followed by Peru with 13%, Colombia with 10%, and with minor amounts in Venezuela, Ecuador, Bolivia, Guyana, Suriname and French Guiana. States or departments in four nations contain \"Amazonas\" in their names. The Amazon represents over half of the planet's remaining rainforests, and comprises the largest and most biodiverse tract of tropical rainforest in the world, with an estimated 390 billion individual trees divided into 16,000 species."}} +{"id":"33727025f1a652f9","question":"What medical treatment is used to increase oxygen uptake in a patient?","gold_answer":"oxygen supplementation","gold_chunk_ids":["33727025f1a652f9"],"is_unanswerable":false,"metadata":{"title":"Oxygen","context":"Uptake of O\n2 from the air is the essential purpose of respiration, so oxygen supplementation is used in medicine. Treatment not only increases oxygen levels in the patient's blood, but has the secondary effect of decreasing resistance to blood flow in many types of diseased lungs, easing work load on the heart. Oxygen therapy is used to treat emphysema, pneumonia, some heart disorders (congestive heart failure), some disorders that cause increased pulmonary artery pressure, and any disease that impairs the body's ability to take up and use gaseous oxygen."}} +{"id":"e9bbfc13a236ccbe","question":"What do the Waal and the Nederrijn-Lek discharge throguh?","gold_answer":"Meuse estuary","gold_chunk_ids":["e9bbfc13a236ccbe"],"is_unanswerable":false,"metadata":{"title":"Rhine","context":"At present, the branches Waal and Nederrijn-Lek discharge to the North Sea, through the former Meuse estuary, near Rotterdam. The river IJssel branch flows to the north and enters the IJsselmeer, formerly the Zuider Zee brackish lagoon; however, since 1932, a freshwater lake. The discharge of the Rhine is divided among three branches: the River Waal (6/9 of total discharge), the River Nederrijn – Lek (2/9 of total discharge) and the River IJssel (1/9 of total discharge). This discharge distribution has been maintained since 1709, by river engineering works, including the digging of the Pannerdens canal and since the 20th century, with the help of weirs in the Nederrijn river."}} +{"id":"79b7fbe70b31a6d6","question":"How can you find the absolute age of sedimentary rock units which do not contain radioactive isotopes?","gold_answer":"Dating of lava and volcanic ash layers found within a stratigraphic sequence","gold_chunk_ids":["79b7fbe70b31a6d6"],"is_unanswerable":false,"metadata":{"title":"Geology","context":"For many geologic applications, isotope ratios of radioactive elements are measured in minerals that give the amount of time that has passed since a rock passed through its particular closure temperature, the point at which different radiometric isotopes stop diffusing into and out of the crystal lattice. These are used in geochronologic and thermochronologic studies. Common methods include uranium-lead dating, potassium-argon dating, argon-argon dating and uranium-thorium dating. These methods are used for a variety of applications. Dating of lava and volcanic ash layers found within a stratigraphic sequence can provide absolute age data for sedimentary rock units which do not contain radioactive isotopes and calibrate relative dating techniques. These methods can also be used to determine ages of pluton emplacement. Thermochemical techniques can be used to determine temperature profiles within the crust, the uplift of mountain ranges, and paleotopography."}} +{"id":"c6aba3189b3db153","question":"Where does manufacturing take place?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Construction","context":"Construction is the process of constructing a building or infrastructure. Construction differs from manufacturing in that manufacturing typically involves mass production of similar items without a designated purchaser, while construction typically takes place on location for a known client. Construction as an industry comprises six to nine percent of the gross domestic product of developed countries. Construction starts with planning,[citation needed] design, and financing and continues until the project is built and ready for use."}} +{"id":"e2aea685a0f36743","question":"How many Africans were brought into the United States during the slave trade?","gold_answer":"12 to 15 million","gold_chunk_ids":["e2aea685a0f36743"],"is_unanswerable":false,"metadata":{"title":"Imperialism","context":"Some have described the internal strife between various people groups as a form of imperialism or colonialism. This internal form is distinct from informal U.S. imperialism in the form of political and financial hegemony. This internal form of imperialism is also distinct from the United States' formation of \"colonies\" abroad. Through the treatment of its indigenous peoples during westward expansion, the United States took on the form of an imperial power prior to any attempts at external imperialism. This internal form of empire has been referred to as \"internal colonialism\". Participation in the African slave trade and the subsequent treatment of its 12 to 15 million Africans is viewed by some to be a more modern extension of America's \"internal colonialism\". However, this internal colonialism faced resistance, as external colonialism did, but the anti-colonial presence was far less prominent due to the nearly complete dominance that the United States was able to assert over both indigenous peoples and African-Americans. In his lecture on April 16, 2003, Edward Said made a bold statement on modern imperialism in the United States, whom he described as using aggressive means of attack towards the contemporary Orient, \"due to their backward living, lack of democracy and the violation of women’s rights. The western world forgets during this process of converting the other that enlightenment and democracy are concepts that not all will agree upon\"."}} +{"id":"ee1eae622f85134f","question":"What makes up 0.9% of the Earth's crust by mass?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Oxygen","context":"Oxygen is the most abundant chemical element by mass in the Earth's biosphere, air, sea and land. Oxygen is the third most abundant chemical element in the universe, after hydrogen and helium. About 0.9% of the Sun's mass is oxygen. Oxygen constitutes 49.2% of the Earth's crust by mass and is the major component of the world's oceans (88.8% by mass). Oxygen gas is the second most common component of the Earth's atmosphere, taking up 20.8% of its volume and 23.1% of its mass (some 1015 tonnes).[d] Earth is unusual among the planets of the Solar System in having such a high concentration of oxygen gas in its atmosphere: Mars (with 0.1% O\n2 by volume) and Venus have far lower concentrations. The O\n2 surrounding these other planets is produced solely by ultraviolet radiation impacting oxygen-containing molecules such as carbon dioxide."}} +{"id":"20ce6a4ba0d133b2","question":"When did Khan formally declare the Yuan dynasty?","gold_answer":"1271","gold_chunk_ids":["20ce6a4ba0d133b2"],"is_unanswerable":false,"metadata":{"title":"Yuan_dynasty","context":"The Yuan dynasty (Chinese: 元朝; pinyin: Yuán Cháo), officially the Great Yuan (Chinese: 大元; pinyin: Dà Yuán; Mongolian: Yehe Yuan Ulus[a]), was the empire or ruling dynasty of China established by Kublai Khan, leader of the Mongolian Borjigin clan. Although the Mongols had ruled territories including today's North China for decades, it was not until 1271 that Kublai Khan officially proclaimed the dynasty in the traditional Chinese style. His realm was, by this point, isolated from the other khanates and controlled most of present-day China and its surrounding areas, including modern Mongolia and Korea. It was the first foreign dynasty to rule all of China and lasted until 1368, after which its Genghisid rulers returned to their Mongolian homeland and continued to rule the Northern Yuan dynasty. Some of the Mongolian Emperors of the Yuan mastered the Chinese language, while others only used their native language (i.e. Mongolian) and the 'Phags-pa script."}} +{"id":"c36c4b34b423be8c","question":"When did Great Britain sell Australia?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Victoria_(Australia)","context":"Prior to European settlement, the area now constituting Victoria was inhabited by a large number of Aboriginal peoples, collectively known as the Koori. With Great Britain having claimed the entire Australian continent east of the 135th meridian east in 1788, Victoria was included in the wider colony of New South Wales. The first settlement in the area occurred in 1803 at Sullivan Bay, and much of what is now Victoria was included in the Port Phillip District in 1836, an administrative division of New South Wales. Victoria was officially created a separate colony in 1851, and achieved self-government in 1855. The Victorian gold rush in the 1850s and 1860s significantly increased both the population and wealth of the colony, and by the Federation of Australia in 1901, Melbourne had become the largest city and leading financial centre in Australasia. Melbourne also served as capital of Australia until the construction of Canberra in 1927, with the Federal Parliament meeting in Melbourne's Parliament House and all principal offices of the federal government being based in Melbourne."}} +{"id":"152e264869216aa4","question":"How many people attended the 2003 IPCC meeting?","gold_answer":"350","gold_chunk_ids":["152e264869216aa4"],"is_unanswerable":false,"metadata":{"title":"Intergovernmental_Panel_on_Climate_Change","context":"The IPCC Panel is composed of representatives appointed by governments and organizations. Participation of delegates with appropriate expertise is encouraged. Plenary sessions of the IPCC and IPCC Working groups are held at the level of government representatives. Non Governmental and Intergovernmental Organizations may be allowed to attend as observers. Sessions of the IPCC Bureau, workshops, expert and lead authors meetings are by invitation only. Attendance at the 2003 meeting included 350 government officials and climate change experts. After the opening ceremonies, closed plenary sessions were held. The meeting report states there were 322 persons in attendance at Sessions with about seven-eighths of participants being from governmental organizations."}} +{"id":"b6b5b98d5d68d6e4","question":"What does continuous motion along dike swarms create for sediment to be deposited?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Geology","context":"The addition of new rock units, both depositionally and intrusively, often occurs during deformation. Faulting and other deformational processes result in the creation of topographic gradients, causing material on the rock unit that is increasing in elevation to be eroded by hillslopes and channels. These sediments are deposited on the rock unit that is going down. Continual motion along the fault maintains the topographic gradient in spite of the movement of sediment, and continues to create accommodation space for the material to deposit. Deformational events are often also associated with volcanism and igneous activity. Volcanic ashes and lavas accumulate on the surface, and igneous intrusions enter from below. Dikes, long, planar igneous intrusions, enter along cracks, and therefore often form in large numbers in areas that are being actively deformed. This can result in the emplacement of dike swarms, such as those that are observable across the Canadian shield, or rings of dikes around the lava tube of a volcano."}} +{"id":"c053e842ab751f40","question":"What is the largest Mansion in the west coast?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Fresno,_California","context":"Fresno has three large public parks, two in the city limits and one in county land to the southwest. Woodward Park, which features the Shinzen Japanese Gardens, numerous picnic areas and several miles of trails, is in North Fresno and is adjacent to the San Joaquin River Parkway. Roeding Park, near Downtown Fresno, is home to the Fresno Chaffee Zoo, and Rotary Storyland and Playland. Kearney Park is the largest of the Fresno region's park system and is home to historic Kearney Mansion and plays host to the annual Civil War Revisited, the largest reenactment of the Civil War in the west coast of the U.S."}} +{"id":"35f48af61823616a","question":"Who was subjected to a qualified minority vote of the Council for approval?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"European_Union_law","context":"The European Commission is the main executive body of the European Union. Article 17(1) of the Treaty on European Union states the Commission should \"promote the general interest of the Union\" while Article 17(3) adds that Commissioners should be \"completely independent\" and not \"take instructions from any Government\". Under article 17(2), \"Union legislative acts may only be adopted on the basis of a Commission proposal, except where the Treaties provide otherwise.\" This means that the Commission has a monopoly on initiating the legislative procedure, although the Council is the \"de facto catalyst of many legislative initiatives\". The Parliament can also formally request the Commission to submit a legislative proposal but the Commission can reject such a suggestion, giving reasons. The Commission's President (currently an ex-Luxembourg Prime Minister, Jean-Claude Juncker) sets the agenda for the EU's work. Decisions are taken by a simple majority vote, usually through a \"written procedure\" of circulating the proposals and adopting if there are no objections.[citation needed] Since Ireland refused to consent to changes in the Treaty of Lisbon 2007, there remains one Commissioner for each of the 28 member states, including the President and the High Representative for Foreign and Security Policy (currently Federica Mogherini). The Commissioners (and most importantly, the portfolios they will hold) are bargained over intensively by the member states. The Commissioners, as a block, are then subject to a qualified majority vote of the Council to approve, and majority approval of the Parliament. The proposal to make the Commissioners be drawn from the elected Parliament, was not adopted in the Treaty of Lisbon. This means Commissioners are, through the appointment process, the unelected subordinates of member state governments."}} +{"id":"2ed377b9e9bfe8a9","question":"What else was avoided by pharmas?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Pharmacy","context":"The word pharmacy is derived from its root word pharma which was a term used since the 15th–17th centuries. However, the original Greek roots from pharmakos imply sorcery or even poison. In addition to pharma responsibilities, the pharma offered general medical advice and a range of services that are now performed solely by other specialist practitioners, such as surgery and midwifery. The pharma (as it was referred to) often operated through a retail shop which, in addition to ingredients for medicines, sold tobacco and patent medicines. Often the place that did this was called an apothecary and several languages have this as the dominant term, though their practices are more akin to a modern pharmacy, in English the term apothecary would today be seen as outdated or only approproriate if herbal remedies were on offer to a large extent. The pharmas also used many other herbs not listed. The Greek word Pharmakeia (Greek: φαρμακεία) derives from pharmakon (φάρμακον), meaning \"drug\", \"medicine\" (or \"poison\").[n 1]"}} +{"id":"6e33b5d2aa72e468","question":"What is a type of Dodge built compact truck?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"1973_oil_crisis","context":"Compact trucks were introduced, such as the Toyota Hilux and the Datsun Truck, followed by the Mazda Truck (sold as the Ford Courier), and the Isuzu-built Chevrolet LUV. Mitsubishi rebranded its Forte as the Dodge D-50 a few years after the oil crisis. Mazda, Mitsubishi and Isuzu had joint partnerships with Ford, Chrysler, and GM, respectively. Later the American makers introduced their domestic replacements (Ford Ranger, Dodge Dakota and the Chevrolet S10/GMC S-15), ending their captive import policy."}} +{"id":"dd86127a478da6e6","question":"Who wrote later papers studying problems solvable by Turning machines?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Computational_complexity_theory","context":"Earlier papers studying problems solvable by Turing machines with specific bounded resources include John Myhill's definition of linear bounded automata (Myhill 1960), Raymond Smullyan's study of rudimentary sets (1961), as well as Hisao Yamada's paper on real-time computations (1962). Somewhat earlier, Boris Trakhtenbrot (1956), a pioneer in the field from the USSR, studied another specific complexity measure. As he remembers:"}} +{"id":"6629065e479f1e5d","question":"What has the tendency to increase wages in a field or job position?","gold_answer":"Competition amongst workers","gold_chunk_ids":["6629065e479f1e5d"],"is_unanswerable":false,"metadata":{"title":"Economic_inequality","context":"A job where there are many workers willing to work a large amount of time (high supply) competing for a job that few require (low demand) will result in a low wage for that job. This is because competition between workers drives down the wage. An example of this would be jobs such as dish-washing or customer service. Competition amongst workers tends to drive down wages due to the expendable nature of the worker in relation to his or her particular job. A job where there are few able or willing workers (low supply), but a large need for the positions (high demand), will result in high wages for that job. This is because competition between employers for employees will drive up the wage. Examples of this would include jobs that require highly developed skills, rare abilities, or a high level of risk. Competition amongst employers tends to drive up wages due to the nature of the job, since there is a relative shortage of workers for the particular position. Professional and labor organizations may limit the supply of workers which results in higher demand and greater incomes for members. Members may also receive higher wages through collective bargaining, political influence, or corruption."}} +{"id":"4faa93271c00fd22","question":"An MSP may introduce a bill as what?","gold_answer":"a private member","gold_chunk_ids":["4faa93271c00fd22"],"is_unanswerable":false,"metadata":{"title":"Scottish_Parliament","context":"Bills can be introduced to Parliament in a number of ways; the Scottish Government can introduce new laws or amendments to existing laws as a bill; a committee of the Parliament can present a bill in one of the areas under its remit; a member of the Scottish Parliament can introduce a bill as a private member; or a private bill can be submitted to Parliament by an outside proposer. Most draft laws are government bills introduced by ministers in the governing party. Bills pass through Parliament in a number of stages:"}} +{"id":"5339b0f75f6a651e","question":"What is the conventional measurement of the Rhine? ","gold_answer":"Rhine-kilometers\"","gold_chunk_ids":["5339b0f75f6a651e"],"is_unanswerable":false,"metadata":{"title":"Rhine","context":"The length of the Rhine is conventionally measured in \"Rhine-kilometers\" (Rheinkilometer), a scale introduced in 1939 which runs from the Old Rhine Bridge at Constance (0 km) to Hoek van Holland (1036.20 km). The river length is significantly shortened from the river's natural course due to number of canalisation projects completed in the 19th and 20th century.[note 7] The \"total length of the Rhine\", to the inclusion of Lake Constance and the Alpine Rhine is more difficult to measure objectively; it was cited as 1,232 kilometres (766 miles) by the Dutch Rijkswaterstaat in 2010.[note 1]"}} +{"id":"5fb2f8d8b19a46f1","question":"How do viruses overcome physical barriers?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Immune_system","context":"The success of any pathogen depends on its ability to elude host immune responses. Therefore, pathogens evolved several methods that allow them to successfully infect a host, while evading detection or destruction by the immune system. Bacteria often overcome physical barriers by secreting enzymes that digest the barrier, for example, by using a type II secretion system. Alternatively, using a type III secretion system, they may insert a hollow tube into the host cell, providing a direct route for proteins to move from the pathogen to the host. These proteins are often used to shut down host defenses."}} +{"id":"e2153f2d6f064b82","question":"What proclamation gave Huguenots special privileges in Brandenburg?","gold_answer":"Edict of Potsdam","gold_chunk_ids":["e2153f2d6f064b82"],"is_unanswerable":false,"metadata":{"title":"Huguenot","context":"Around 1685, Huguenot refugees found a safe haven in the Lutheran and Reformed states in Germany and Scandinavia. Nearly 50,000 Huguenots established themselves in Germany, 20,000 of whom were welcomed in Brandenburg-Prussia, where they were granted special privileges (Edict of Potsdam) and churches in which to worship (such as the Church of St. Peter and St. Paul, Angermünde) by Frederick William, Elector of Brandenburg and Duke of Prussia. The Huguenots furnished two new regiments of his army: the Altpreußische Infantry Regiments No. 13 (Regiment on foot Varenne) and 15 (Regiment on foot Wylich). Another 4,000 Huguenots settled in the German territories of Baden, Franconia (Principality of Bayreuth, Principality of Ansbach), Landgraviate of Hesse-Kassel, Duchy of Württemberg, in the Wetterau Association of Imperial Counts, in the Palatinate and Palatinate-Zweibrücken, in the Rhine-Main-Area (Frankfurt), in modern-day Saarland; and 1,500 found refuge in Hamburg, Bremen and Lower Saxony. Three hundred refugees were granted asylum at the court of George William, Duke of Brunswick-Lüneburg in Celle."}} +{"id":"988b6e6db7142656","question":"How long had John Paul II been the pope in 1983?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Warsaw","context":"John Paul II's visits to his native country in 1979 and 1983 brought support to the budding solidarity movement and encouraged the growing anti-communist fervor there. In 1979, less than a year after becoming pope, John Paul celebrated Mass in Victory Square in Warsaw and ended his sermon with a call to \"renew the face\" of Poland: Let Thy Spirit descend! Let Thy Spirit descend and renew the face of the land! This land! These words were very meaningful for the Polish citizens who understood them as the incentive for the democratic changes."}} +{"id":"afdefc7f08dd4c86","question":"What type of wages do people unable to afford an education receive?","gold_answer":"lower","gold_chunk_ids":["afdefc7f08dd4c86"],"is_unanswerable":false,"metadata":{"title":"Economic_inequality","context":"An important factor in the creation of inequality is variation in individuals' access to education. Education, especially in an area where there is a high demand for workers, creates high wages for those with this education, however, increases in education first increase and then decrease growth as well as income inequality. As a result, those who are unable to afford an education, or choose not to pursue optional education, generally receive much lower wages. The justification for this is that a lack of education leads directly to lower incomes, and thus lower aggregate savings and investment. Conversely, education raises incomes and promotes growth because it helps to unleash the productive potential of the poor."}} +{"id":"0bf1d4988dfac7b5","question":"Who was New France's governor?","gold_answer":"Marquis de Vaudreuil.","gold_chunk_ids":["0bf1d4988dfac7b5"],"is_unanswerable":false,"metadata":{"title":"French_and_Indian_War","context":"Johnson's expedition was better organized than Shirley's, which was noticed by New France's governor, the Marquis de Vaudreuil. He had primarily been concerned about the extended supply line to the forts on the Ohio, and had sent Baron Dieskau to lead the defenses at Frontenac against Shirley's expected attack. When Johnson was seen as the larger threat, Vaudreuil sent Dieskau to Fort St. Frédéric to meet that threat. Dieskau planned to attack the British encampment at Fort Edward at the upper end of navigation on the Hudson River, but Johnson had strongly fortified it, and Dieskau's Indian support was reluctant to attack. The two forces finally met in the bloody Battle of Lake George between Fort Edward and Fort William Henry. The battle ended inconclusively, with both sides withdrawing from the field. Johnson's advance stopped at Fort William Henry, and the French withdrew to Ticonderoga Point, where they began the construction of Fort Carillon (later renamed Fort Ticonderoga after British capture in 1759)."}} +{"id":"be23e46d6ea5fb7d","question":"What are two small car models that didn't recover in 1974?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"1973_oil_crisis","context":"An increase in imported cars into North America forced General Motors, Ford and Chrysler to introduce smaller and fuel-efficient models for domestic sales. The Dodge Omni / Plymouth Horizon from Chrysler, the Ford Fiesta and the Chevrolet Chevette all had four-cylinder engines and room for at least four passengers by the late 1970s. By 1985, the average American vehicle moved 17.4 miles per gallon, compared to 13.5 in 1970. The improvements stayed even though the price of a barrel of oil remained constant at $12 from 1974 to 1979. Sales of large sedans for most makes (except Chrysler products) recovered within two model years of the 1973 crisis. The Cadillac DeVille and Fleetwood, Buick Electra, Oldsmobile 98, Lincoln Continental, Mercury Marquis, and various other luxury oriented sedans became popular again in the mid-1970s. The only full-size models that did not recover were lower price models such as the Chevrolet Bel Air, and Ford Galaxie 500. Slightly smaller, mid-size models such as the Oldsmobile Cutlass, Chevrolet Monte Carlo, Ford Thunderbird and various other models sold well."}} +{"id":"73d4d4134510faa5","question":"In what year did Louis XIV start to bribe Protestants to convert to Catholicism?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Huguenot","context":"Louis XIV gained the throne in 1643 and acted increasingly aggressively to force the Huguenots to convert. At first he sent missionaries, backed by a fund to financially reward converts to Catholicism. Then he imposed penalties, closed Huguenot schools and excluded them from favored professions. Escalating, he instituted dragonnades, which included the occupation and looting of Huguenot homes by military troops, in an effort to forcibly convert them. In 1685, he issued the Edict of Fontainebleau, revoking the Edict of Nantes and declaring Protestantism illegal.[citation needed]"}} +{"id":"7ec9710b30575a7c","question":"In what year did medical knowledge begin to stagnate during the Middle Ages?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Black_Death","context":"Medical knowledge had stagnated during the Middle Ages. The most authoritative account at the time came from the medical faculty in Paris in a report to the king of France that blamed the heavens, in the form of a conjunction of three planets in 1345 that caused a \"great pestilence in the air\". This report became the first and most widely circulated of a series of plague tracts that sought to give advice to sufferers. That the plague was caused by bad air became the most widely accepted theory. Today, this is known as the Miasma theory. The word 'plague' had no special significance at this time, and only the recurrence of outbreaks during the Middle Ages gave it the name that has become the medical term."}} +{"id":"1d40f3eda19846c5","question":"How wide is the glacial alpine valley known as the Rhine Valley?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Rhine","context":"Near Tamins-Reichenau the Anterior Rhine and the Posterior Rhine join and form the Rhine. The river makes a distinctive turn to the north near Chur. This section is nearly 86 km long, and descends from a height of 599 m to 396 m. It flows through a wide glacial alpine valley known as the Rhine Valley (German: Rheintal). Near Sargans a natural dam, only a few metres high, prevents it from flowing into the open Seeztal valley and then through Lake Walen and Lake Zurich into the river Aare. The Alpine Rhine begins in the most western part of the Swiss canton of Graubünden, and later forms the border between Switzerland to the West and Liechtenstein and later Austria to the East."}} +{"id":"962531515c3bcd9f","question":"What percentage of electrical power in the United States is made by generators?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Steam_engine","context":"The final major evolution of the steam engine design was the use of steam turbines starting in the late part of the 19th century. Steam turbines are generally more efficient than reciprocating piston type steam engines (for outputs above several hundred horsepower), have fewer moving parts, and provide rotary power directly instead of through a connecting rod system or similar means. Steam turbines virtually replaced reciprocating engines in electricity generating stations early in the 20th century, where their efficiency, higher speed appropriate to generator service, and smooth rotation were advantages. Today most electric power is provided by steam turbines. In the United States 90% of the electric power is produced in this way using a variety of heat sources. Steam turbines were extensively applied for propulsion of large ships throughout most of the 20th century."}} +{"id":"a40611d8ea9956d5","question":"Who was bound to apply an EU law where a national rule conflicted?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"European_Union_law","context":"While constitutional law concerns the European Union's governance structure, administrative law binds EU institutions and member states to follow the law. Both member states and the Commission have a general legal right or \"standing\" (locus standi) to bring claims against EU institutions and other member states for breach of the treaties. From the EU's foundation, the Court of Justice also held that the Treaties allowed citizens or corporations to bring claims against EU and member state institutions for violation of the Treaties and Regulations, if they were properly interpreted as creating rights and obligations. However, under Directives, citizens or corporations were said in 1986 to not be allowed to bring claims against other non-state parties. This meant courts of member states were not bound to apply an EU law where a national rule conflicted, even though the member state government could be sued, if it would impose an obligation on another citizen or corporation. These rules on \"direct effect\" limit the extent to which member state courts are bound to administer EU law. All actions by EU institutions can be subject to judicial review, and judged by standards of proportionality, particularly where general principles of law, or fundamental rights are engaged. The remedy for a claimant where there has been a breach of the law is often monetary damages, but courts can also require specific performance or will grant an injunction, in order to ensure the law is effective as possible."}} +{"id":"c17826a5a47bd3e6","question":"How many subtypes of B cells exist?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Immune_system","context":"Both B cells and T cells carry receptor molecules that recognize specific targets. T cells recognize a \"non-self\" target, such as a pathogen, only after antigens (small fragments of the pathogen) have been processed and presented in combination with a \"self\" receptor called a major histocompatibility complex (MHC) molecule. There are two major subtypes of T cells: the killer T cell and the helper T cell. In addition there are regulatory T cells which have a role in modulating immune response. Killer T cells only recognize antigens coupled to Class I MHC molecules, while helper T cells and regulatory T cells only recognize antigens coupled to Class II MHC molecules. These two mechanisms of antigen presentation reflect the different roles of the two types of T cell. A third, minor subtype are the γδ T cells that recognize intact antigens that are not bound to MHC receptors."}} +{"id":"4af420e4f237c0a1","question":"After the Peterloo massacre what poet wrote The Massacre of Anarchy?","gold_answer":"Percy Shelley","gold_chunk_ids":["4af420e4f237c0a1"],"is_unanswerable":false,"metadata":{"title":"Civil_disobedience","context":"Following the Peterloo massacre of 1819, poet Percy Shelley wrote the political poem The Mask of Anarchy later that year, that begins with the images of what he thought to be the unjust forms of authority of his time—and then imagines the stirrings of a new form of social action. It is perhaps the first modern[vague] statement of the principle of nonviolent protest. A version was taken up by the author Henry David Thoreau in his essay Civil Disobedience, and later by Gandhi in his doctrine of Satyagraha. Gandhi's Satyagraha was partially influenced and inspired by Shelley's nonviolence in protest and political action. In particular, it is known that Gandhi would often quote Shelley's Masque of Anarchy to vast audiences during the campaign for a free India."}} +{"id":"cda7860cbf449680","question":"What percentage of Australia's veal is comes from Victoria?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Victoria_(Australia)","context":"Victoria is the centre of dairy farming in Australia. It is home to 60% of Australia's 3 million dairy cattle and produces nearly two-thirds of the nation's milk, almost 6.4 billion litres. The state also has 2.4 million beef cattle, with more than 2.2 million cattle and calves slaughtered each year. In 2003–04, Victorian commercial fishing crews and aquaculture industry produced 11,634 tonnes of seafood valued at nearly A$109 million. Blacklipped abalone is the mainstay of the catch, bringing in A$46 million, followed by southern rock lobster worth A$13.7 million. Most abalone and rock lobster is exported to Asia."}} +{"id":"9b8299c41c4c25ef","question":"In what office has Barack Obama recently served his last term?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Harvard_University","context":"Politics: U.N. Secretary General Ban Ki-moon; American political leaders John Hancock, John Adams, John Quincy Adams, Rutherford B. Hayes, Theodore Roosevelt, Franklin D. Roosevelt, John F. Kennedy, Al Gore, George W. Bush and Barack Obama; Chilean President Sebastián Piñera; Colombian President Juan Manuel Santos; Costa Rican President José María Figueres; Mexican Presidents Felipe Calderón, Carlos Salinas de Gortari and Miguel de la Madrid; Mongolian President Tsakhiagiin Elbegdorj; Peruvian President Alejandro Toledo; Taiwanese President Ma Ying-jeou; Canadian Governor General David Lloyd Johnston; Indian Member of Parliament Jayant Sinha; Albanian Prime Minister Fan S. Noli; Canadian Prime Ministers Mackenzie King and Pierre Trudeau; Greek Prime Minister Antonis Samaras; Israeli Prime Minister Benjamin Netanyahu; former Pakistani Prime Minister Benazir Bhutto; U. S. Secretary of Housing and Urban Development Shaun Donovan; Canadian political leader Michael Ignatieff; Pakistani Members of Provincial Assembly Murtaza Bhutto and Sanam Bhutto; Bangladesh Minister of Finance Abul Maal Abdul Muhith; President of Puntland Abdiweli Mohamed Ali; U.S. Ambassador to the European Union Anthony Luzzatto Gardner."}} +{"id":"60759dd2302a4824","question":"What can residential and non-residential also be broken into?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Construction","context":"In general, there are three sectors of construction: buildings, infrastructure and industrial. Building construction is usually further divided into residential and non-residential (commercial/institutional). Infrastructure is often called heavy/highway, heavy civil or heavy engineering. It includes large public works, dams, bridges, highways, water/wastewater and utility distribution. Industrial includes refineries, process chemical, power generation, mills and manufacturing plants. There are other ways to break the industry into sectors or markets."}} +{"id":"cda68354dfece40b","question":"Where didn't Iroquois Confederation control?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"French_and_Indian_War","context":"In between the French and the British, large areas were dominated by native tribes. To the north, the Mi'kmaq and the Abenaki were engaged in Father Le Loutre's War and still held sway in parts of Nova Scotia, Acadia, and the eastern portions of the province of Canada, as well as much of present-day Maine. The Iroquois Confederation dominated much of present-day Upstate New York and the Ohio Country, although the latter also included Algonquian-speaking populations of Delaware and Shawnee, as well as Iroquoian-speaking Mingo. These tribes were formally under Iroquois rule, and were limited by them in authority to make agreements."}} +{"id":"918063c6fed26694","question":"Who had Kublai wanted to succeed him?","gold_answer":"his eldest son, Zhenjin","gold_chunk_ids":["918063c6fed26694"],"is_unanswerable":false,"metadata":{"title":"Yuan_dynasty","context":"Following the conquest of Dali in 1253, the former ruling Duan dynasty were appointed as governors-general, recognized as imperial officials by the Yuan, Ming, and Qing-era governments, principally in the province of Yunnan. Succession for the Yuan dynasty, however, was an intractable problem, later causing much strife and internal struggle. This emerged as early as the end of Kublai's reign. Kublai originally named his eldest son, Zhenjin, as the Crown Prince, but he died before Kublai in 1285. Thus, Zhenjin's third son, with the support of his mother Kökejin and the minister Bayan, succeeded the throne and ruled as Temür Khan, or Emperor Chengzong, from 1294 to 1307. Temür Khan decided to maintain and continue much of the work begun by his grandfather. He also made peace with the western Mongol khanates as well as neighboring countries such as Vietnam, which recognized his nominal suzerainty and paid tributes for a few decades. However, the corruption in the Yuan dynasty began during the reign of Temür Khan."}} +{"id":"2f3d5a6343f53294","question":"The College grants Bachelor of Science/Arts degrees in 50 minors and how many majors?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"University_of_Chicago","context":"The College of the University of Chicago grants Bachelor of Arts and Bachelor of Science degrees in 50 academic majors and 28 minors. The college's academics are divided into five divisions: the Biological Sciences Collegiate Division, the Physical Sciences Collegiate Division, the Social Sciences Collegiate Division, the Humanities Collegiate Division, and the New Collegiate Division. The first four are sections within their corresponding graduate divisions, while the New Collegiate Division administers interdisciplinary majors and studies which do not fit in one of the other four divisions."}} +{"id":"e286e0a1f8061d17","question":"How is course content provided to a private school?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Private_school","context":"Funding for private schools is generally provided through student tuition, endowments, scholarship/voucher funds, and donations and grants from religious organizations or private individuals. Government funding for religious schools is either subject to restrictions or possibly forbidden, according to the courts' interpretation of the Establishment Clause of the First Amendment or individual state Blaine Amendments. Non-religious private schools theoretically could qualify for such funding without hassle, preferring the advantages of independent control of their student admissions and course content instead of the public funding they could get with charter status."}} +{"id":"b682a9732dd32d35","question":"What was Abu Hamaz al-Masri charged with when he was arrested?","gold_answer":"incitement to terrorism","gold_chunk_ids":["b682a9732dd32d35"],"is_unanswerable":false,"metadata":{"title":"Islamism","context":"Greater London has over 900,000 Muslims, (most of South Asian origins and concentrated in the East London boroughs of Newham, Tower Hamlets and Waltham Forest), and among them are some with a strong Islamist outlook. Their presence, combined with a perceived British policy of allowing them free rein, heightened by exposés such as the 2007 Channel 4 documentary programme Undercover Mosque, has given rise to the term Londonistan. Following the 9/11 attacks, however, Abu Hamza al-Masri, the imam of the Finsbury Park Mosque, was arrested and charged with incitement to terrorism which has caused many Islamists to leave the UK to avoid internment.[citation needed]"}} +{"id":"dd61735c7942505a","question":"If three identical fermions have a symmetric spin, the spatial variables must be what?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Force","context":"However, already in quantum mechanics there is one \"caveat\", namely the particles acting onto each other do not only possess the spatial variable, but also a discrete intrinsic angular momentum-like variable called the \"spin\", and there is the Pauli principle relating the space and the spin variables. Depending on the value of the spin, identical particles split into two different classes, fermions and bosons. If two identical fermions (e.g. electrons) have a symmetric spin function (e.g. parallel spins) the spatial variables must be antisymmetric (i.e. they exclude each other from their places much as if there was a repulsive force), and vice versa, i.e. for antiparallel spins the position variables must be symmetric (i.e. the apparent force must be attractive). Thus in the case of two fermions there is a strictly negative correlation between spatial and spin variables, whereas for two bosons (e.g. quanta of electromagnetic waves, photons) the correlation is strictly positive."}} +{"id":"2a26e838d4bc467e","question":"Messiaen says that composition with prime numbers was inspired by what?","gold_answer":"the movements of nature","gold_chunk_ids":["2a26e838d4bc467e"],"is_unanswerable":false,"metadata":{"title":"Prime_number","context":"Prime numbers have influenced many artists and writers. The French composer Olivier Messiaen used prime numbers to create ametrical music through \"natural phenomena\". In works such as La Nativité du Seigneur (1935) and Quatre études de rythme (1949–50), he simultaneously employs motifs with lengths given by different prime numbers to create unpredictable rhythms: the primes 41, 43, 47 and 53 appear in the third étude, \"Neumes rythmiques\". According to Messiaen this way of composing was \"inspired by the movements of nature, movements of free and unequal durations\"."}} +{"id":"a2f971bdc171bf03","question":"What area has become attractive for restaurants?","gold_answer":"Tower District","gold_chunk_ids":["a2f971bdc171bf03"],"is_unanswerable":false,"metadata":{"title":"Fresno,_California","context":"The neighborhood features restaurants, live theater and nightclubs, as well as several independent shops and bookstores, currently operating on or near Olive Avenue, and all within a few hundred feet of each other. Since renewal, the Tower District has become an attractive area for restaurant and other local businesses. Today, the Tower District is also known as the center of Fresno's LGBT and hipster Communities.; Additionally, Tower District is also known as the center of Fresno's local punk/goth/deathrock and heavy metal community.[citation needed]"}} +{"id":"db7005be2b0ac365","question":"What percent of the global assets in 2000 were owned by just 1% of adults?","gold_answer":"40%","gold_chunk_ids":["db7005be2b0ac365"],"is_unanswerable":false,"metadata":{"title":"Economic_inequality","context":"A study by the World Institute for Development Economics Research at United Nations University reports that the richest 1% of adults alone owned 40% of global assets in the year 2000. The three richest people in the world possess more financial assets than the lowest 48 nations combined. The combined wealth of the \"10 million dollar millionaires\" grew to nearly $41 trillion in 2008. A January 2014 report by Oxfam claims that the 85 wealthiest individuals in the world have a combined wealth equal to that of the bottom 50% of the world's population, or about 3.5 billion people. According to a Los Angeles Times analysis of the report, the wealthiest 1% owns 46% of the world's wealth; the 85 richest people, a small part of the wealthiest 1%, own about 0.7% of the human population's wealth, which is the same as the bottom half of the population. More recently, in January 2015, Oxfam reported that the wealthiest 1 percent will own more than half of the global wealth by 2016. An October 2014 study by Credit Suisse also claims that the top 1% now own nearly half of the world's wealth and that the accelerating disparity could trigger a recession. In October 2015, Credit Suisse published a study which shows global inequality continues to increase, and that half of the world's wealth is now in the hands of those in the top percentile, whose assets each exceed $759,900. A 2016 report by Oxfam claims that the 62 wealthiest individuals own as much wealth as the poorer half of the global population combined. Oxfam's claims have however been questioned on the basis of the methodology used: by using net wealth (adding up assets and subtracting debts), the Oxfam report, for instance, finds that there are more poor people in the United States and Western Europe than in China (due to a greater tendency to take on debts).[unreliable source?][unreliable source?] Anthony Shorrocks, the lead author of the Credit Suisse report which is one of the sources of Oxfam's data, considers the criticism about debt to be a \"silly argument\" and \"a non-issue . . . a diversion.\""}} +{"id":"c614efbf8bf0bff6","question":"What are other signs of insecurity in Mecca?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"1973_oil_crisis","context":"The USSR's invasion of Afghanistan was only one sign of insecurity in the region, also marked by increased American weapons sales, technology, and outright military presence. Saudi Arabia and Iran became increasingly dependent on American security assurances to manage both external and internal threats, including increased military competition between them over increased oil revenues. Both states were competing for preeminence in the Persian Gulf and using increased revenues to fund expanded militaries. By 1979, Saudi arms purchases from the US exceeded five times Israel's. Another motive for the large scale purchase of arms from the US by Saudi Arabia was the failure of the Shah during January 1979 to maintain control of Iran, a non-Arabic but largely Shiite Muslim nation, which fell to a theocratic Islamist government under the Ayatollah Ruhollah Khomeini in the wake of the 1979 Iranian Revolution. Saudi Arabia, on the other hand, is an Arab, largely Sunni Muslim nation headed by a near absolutist monarchy. In the wake of the Iranian revolution the Saudis were forced to deal with the prospect of internal destabilization via the radicalism of Islamism, a reality which would quickly be revealed in the seizure of the Grand Mosque in Mecca by Wahhabi extremists during November 1979 and a Shiite revolt in the oil rich Al-Hasa region of Saudi Arabia in December of the same year. In November 2010, Wikileaks leaked confidential diplomatic cables pertaining to the United States and its allies which revealed that the late Saudi King Abdullah urged the United States to attack Iran in order to destroy its potential nuclear weapons program, describing Iran as \"a snake whose head should be cut off without any procrastination.\""}} +{"id":"b1c5c07b03befc8c","question":"Along with private individuals and organizations, what groups sometimes runs ergänzungsschulen?","gold_answer":"religious","gold_chunk_ids":["b1c5c07b03befc8c"],"is_unanswerable":false,"metadata":{"title":"Private_school","context":"Ergänzungsschulen are secondary or post-secondary (non-tertiary) schools, which are run by private individuals, private organizations or rarely, religious groups and offer a type of education which is not available at public schools. Most of these schools are vocational schools. However, these vocational schools are not part of the German dual education system. Ergänzungsschulen have the freedom to operate outside of government regulation and are funded in whole by charging their students tuition fees."}} +{"id":"26ddbbf8ed939a31","question":"What are many complexity classes not defined by?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Computational_complexity_theory","context":"Many complexity classes are defined using the concept of a reduction. A reduction is a transformation of one problem into another problem. It captures the informal notion of a problem being at least as difficult as another problem. For instance, if a problem X can be solved using an algorithm for Y, X is no more difficult than Y, and we say that X reduces to Y. There are many different types of reductions, based on the method of reduction, such as Cook reductions, Karp reductions and Levin reductions, and the bound on the complexity of reductions, such as polynomial-time reductions or log-space reductions."}} +{"id":"52e67537509084e5","question":"What were steam engines used as a source of?","gold_answer":"power","gold_chunk_ids":["52e67537509084e5"],"is_unanswerable":false,"metadata":{"title":"Steam_engine","context":"Around 1800 Richard Trevithick and, separately, Oliver Evans in 1801 introduced engines using high-pressure steam; Trevithick obtained his high-pressure engine patent in 1802. These were much more powerful for a given cylinder size than previous engines and could be made small enough for transport applications. Thereafter, technological developments and improvements in manufacturing techniques (partly brought about by the adoption of the steam engine as a power source) resulted in the design of more efficient engines that could be smaller, faster, or more powerful, depending on the intended application."}} +{"id":"1555f84a1cdff836","question":"When was it discovered that prime numbers could be applied to the creation of public key military algorithms?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Prime_number","context":"For a long time, number theory in general, and the study of prime numbers in particular, was seen as the canonical example of pure mathematics, with no applications outside of the self-interest of studying the topic with the exception of use of prime numbered gear teeth to distribute wear evenly. In particular, number theorists such as British mathematician G. H. Hardy prided themselves on doing work that had absolutely no military significance. However, this vision was shattered in the 1970s, when it was publicly announced that prime numbers could be used as the basis for the creation of public key cryptography algorithms. Prime numbers are also used for hash tables and pseudorandom number generators."}} +{"id":"b200da298f8e4c87","question":"Which French composer wrote metrical music using prime numbers?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Prime_number","context":"Prime numbers have influenced many artists and writers. The French composer Olivier Messiaen used prime numbers to create ametrical music through \"natural phenomena\". In works such as La Nativité du Seigneur (1935) and Quatre études de rythme (1949–50), he simultaneously employs motifs with lengths given by different prime numbers to create unpredictable rhythms: the primes 41, 43, 47 and 53 appear in the third étude, \"Neumes rythmiques\". According to Messiaen this way of composing was \"inspired by the movements of nature, movements of free and unequal durations\"."}} +{"id":"d2cb20a041009047","question":"Which group has a primary role scrutinizing witnesses?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Scottish_Parliament","context":"Much of the work of the Scottish Parliament is done in committee. The role of committees is stronger in the Scottish Parliament than in other parliamentary systems, partly as a means of strengthening the role of backbenchers in their scrutiny of the government and partly to compensate for the fact that there is no revising chamber. The principal role of committees in the Scottish Parliament is to take evidence from witnesses, conduct inquiries and scrutinise legislation. Committee meetings take place on Tuesday, Wednesday and Thursday morning when Parliament is sitting. Committees can also meet at other locations throughout Scotland."}} +{"id":"0307ad2a0a282202","question":"The loss of Edinburgh Pentlands really disappointed whom the most?","gold_answer":"the Conservatives","gold_chunk_ids":["0307ad2a0a282202"],"is_unanswerable":false,"metadata":{"title":"Scottish_Parliament","context":"For the Conservatives, the main disappointment was the loss of Edinburgh Pentlands, the seat of former party leader David McLetchie, to the SNP. McLetchie was elected on the Lothian regional list and the Conservatives suffered a net loss of five seats, with leader Annabel Goldie claiming that their support had held firm. Nevertheless, she too announced she would step down as leader of the party. Cameron congratulated the SNP on their victory but vowed to campaign for the Union in the independence referendum."}} +{"id":"e090cd14772bc53e","question":"What was Zia-ul-Haq's official state ideology?","gold_answer":"Islamism","gold_chunk_ids":["e090cd14772bc53e"],"is_unanswerable":false,"metadata":{"title":"Islamism","context":"In July 1977, General Zia-ul-Haq overthrew Prime Minister Zulfiqar Ali Bhutto's regime in Pakistan. Ali Bhutto, a leftist in democratic competition with Islamists, had announced banning alcohol and nightclubs within six months, shortly before he was overthrown. Zia-ul-Haq was much more committed to Islamism, and \"Islamization\" or implementation of Islamic law, became a cornerstone of his eleven-year military dictatorship and Islamism became his \"official state ideology\". Zia ul Haq was an admirer of Mawdudi and Mawdudi's party Jamaat-e-Islami became the \"regime's ideological and political arm\". In Pakistan this Islamization from above was \"probably\" more complete \"than under any other regime except those in Iran and Sudan,\" but Zia-ul-Haq was also criticized by many Islamists for imposing \"symbols\" rather than substance, and using Islamization to legitimize his means of seizing power. Unlike neighboring Iran, Zia-ul-Haq's policies were intended to \"avoid revolutionary excess\", and not to strain relations with his American and Persian Gulf state allies. Zia-ul-Haq was killed in 1988 but Islamization remains an important element in Pakistani society."}} +{"id":"0eced8ccaabed9d8","question":"What kind of diploma is given when graduating from a Sonderungsverbot?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Private_school","context":"Ersatzschulen are ordinary primary or secondary schools, which are run by private individuals, private organizations or religious groups. These schools offer the same types of diplomas as public schools. Ersatzschulen lack the freedom to operate completely outside of government regulation. Teachers at Ersatzschulen must have at least the same education and at least the same wages as teachers at public schools, an Ersatzschule must have at least the same academic standards as a public school and Article 7, Paragraph 4 of the Grundgesetz, also forbids segregation of pupils according to the means of their parents (the so-called Sonderungsverbot). Therefore, most Ersatzschulen have very low tuition fees and/or offer scholarships, compared to most other Western European countries. However, it is not possible to finance these schools with such low tuition fees, which is why all German Ersatzschulen are additionally financed with public funds. The percentages of public money could reach 100% of the personnel expenditures. Nevertheless, Private Schools became insolvent in the past in Germany."}} +{"id":"38a37c310db16636","question":"What is the English translation of tawhid?","gold_answer":"unity of God","gold_chunk_ids":["38a37c310db16636"],"is_unanswerable":false,"metadata":{"title":"Islamism","context":"Maududi also believed that Muslim society could not be Islamic without Sharia, and Islam required the establishment of an Islamic state. This state should be a \"theo-democracy,\" based on the principles of: tawhid (unity of God), risala (prophethood) and khilafa (caliphate). Although Maududi talked about Islamic revolution, by \"revolution\" he meant not the violence or populist policies of the Iranian Revolution, but the gradual changing the hearts and minds of individuals from the top of society downward through an educational process or da'wah."}} +{"id":"ec235495c9aa41e9","question":"What fields have lost influence over pharmacy in the United States?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Pharmacy","context":"This shift has already commenced in some countries; for instance, pharmacists in Australia receive remuneration from the Australian Government for conducting comprehensive Home Medicines Reviews. In Canada, pharmacists in certain provinces have limited prescribing rights (as in Alberta and British Columbia) or are remunerated by their provincial government for expanded services such as medications reviews (Medschecks in Ontario). In the United Kingdom, pharmacists who undertake additional training are obtaining prescribing rights and this is because of pharmacy education. They are also being paid for by the government for medicine use reviews. In Scotland the pharmacist can write prescriptions for Scottish registered patients of their regular medications, for the majority of drugs, except for controlled drugs, when the patient is unable to see their doctor, as could happen if they are away from home or the doctor is unavailable. In the United States, pharmaceutical care or clinical pharmacy has had an evolving influence on the practice of pharmacy. Moreover, the Doctor of Pharmacy (Pharm. D.) degree is now required before entering practice and some pharmacists now complete one or two years of residency or fellowship training following graduation. In addition, consultant pharmacists, who traditionally operated primarily in nursing homes are now expanding into direct consultation with patients, under the banner of \"senior care pharmacy.\""}} +{"id":"c3f25dd837ddf4e6","question":"What do rules about conflict of interest involving doctors diagnosing patients not resemble?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Pharmacy","context":"The reason for the majority rule is the high risk of a conflict of interest and/or the avoidance of absolute powers. Otherwise, the physician has a financial self-interest in \"diagnosing\" as many conditions as possible, and in exaggerating their seriousness, because he or she can then sell more medications to the patient. Such self-interest directly conflicts with the patient's interest in obtaining cost-effective medication and avoiding the unnecessary use of medication that may have side-effects. This system reflects much similarity to the checks and balances system of the U.S. and many other governments.[citation needed]"}} +{"id":"edfca14912890636","question":"When was virgin media rebranded from NTL Telewest?","gold_answer":"2007","gold_chunk_ids":["edfca14912890636"],"is_unanswerable":false,"metadata":{"title":"Sky_(United_Kingdom)","context":"Virgin Media (re-branded in 2007 from NTL:Telewest) started to offer a high-definition television (HDTV) capable set top box, although from 30 November 2006 until 30 July 2009 it only carried one linear HD channel, BBC HD, after the conclusion of the ITV HD trial. Virgin Media has claimed that other HD channels were \"locked up\" or otherwise withheld from their platform, although Virgin Media did in fact have an option to carry Channel 4 HD in the future. Nonetheless, the linear channels were not offered, Virgin Media instead concentrating on its Video On Demand service to carry a modest selection of HD content. Virgin Media has nevertheless made a number of statements over the years, suggesting that more linear HD channels are on the way."}} +{"id":"e7945c331e4683c3","question":"How are packets normally forwarded","gold_answer":"by intermediate network nodes asynchronously using first-in, first-out buffering, but may be forwarded according to some scheduling discipline for fair queuing","gold_chunk_ids":["e7945c331e4683c3"],"is_unanswerable":false,"metadata":{"title":"Packet_switching","context":"Packet mode communication may be implemented with or without intermediate forwarding nodes (packet switches or routers). Packets are normally forwarded by intermediate network nodes asynchronously using first-in, first-out buffering, but may be forwarded according to some scheduling discipline for fair queuing, traffic shaping, or for differentiated or guaranteed quality of service, such as weighted fair queuing or leaky bucket. In case of a shared physical medium (such as radio or 10BASE5), the packets may be delivered according to a multiple access scheme."}} +{"id":"b92b9f2c093ec57c","question":"What rules does the IPCC have to follow?","gold_answer":"the Financial Regulations and Rules of the WMO","gold_chunk_ids":["b92b9f2c093ec57c"],"is_unanswerable":false,"metadata":{"title":"Intergovernmental_Panel_on_Climate_Change","context":"The IPCC receives funding through the IPCC Trust Fund, established in 1989 by the United Nations Environment Programme (UNEP) and the World Meteorological Organization (WMO), Costs of the Secretary and of housing the secretariat are provided by the WMO, while UNEP meets the cost of the Depute Secretary. Annual cash contributions to the Trust Fund are made by the WMO, by UNEP, and by IPCC Members; the scale of payments is determined by the IPCC Panel, which is also responsible for considering and adopting by consensus the annual budget. The organisation is required to comply with the Financial Regulations and Rules of the WMO."}} +{"id":"0509f461ed4f5f9b","question":"What is France a region of?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Normans","context":"The Normans (Norman: Nourmands; French: Normands; Latin: Normanni) were the people who in the 10th and 11th centuries gave their name to Normandy, a region in France. They were descended from Norse (\"Norman\" comes from \"Norseman\") raiders and pirates from Denmark, Iceland and Norway who, under their leader Rollo, agreed to swear fealty to King Charles III of West Francia. Through generations of assimilation and mixing with the native Frankish and Roman-Gaulish populations, their descendants would gradually merge with the Carolingian-based cultures of West Francia. The distinct cultural and ethnic identity of the Normans emerged initially in the first half of the 10th century, and it continued to evolve over the succeeding centuries."}} +{"id":"73a28087b2bc1b9f","question":"Who ranked Warsaw as the 32nd most liveable city in the world?","gold_answer":"Economist Intelligence Unit","gold_chunk_ids":["73a28087b2bc1b9f"],"is_unanswerable":false,"metadata":{"title":"Warsaw","context":"In 2012 the Economist Intelligence Unit ranked Warsaw as the 32nd most liveable city in the world. It was also ranked as one of the most liveable cities in Central Europe. Today Warsaw is considered an \"Alpha–\" global city, a major international tourist destination and a significant cultural, political and economic hub. Warsaw's economy, by a wide variety of industries, is characterised by FMCG manufacturing, metal processing, steel and electronic manufacturing and food processing. The city is a significant centre of research and development, BPO, ITO, as well as of the Polish media industry. The Warsaw Stock Exchange is one of the largest and most important in Central and Eastern Europe. Frontex, the European Union agency for external border security, has its headquarters in Warsaw. It has been said that Warsaw, together with Frankfurt, London, Paris and Barcelona is one of the cities with the highest number of skyscrapers in the European Union. Warsaw has also been called \"Eastern Europe’s chic cultural capital with thriving art and club scenes and serious restaurants\"."}} +{"id":"b2dd53069a631000","question":"What space-time path is seen as a curved line in space?","gold_answer":"ballistic trajectory","gold_chunk_ids":["b2dd53069a631000"],"is_unanswerable":false,"metadata":{"title":"Force","context":"Since then, and so far, general relativity has been acknowledged as the theory that best explains gravity. In GR, gravitation is not viewed as a force, but rather, objects moving freely in gravitational fields travel under their own inertia in straight lines through curved space-time – defined as the shortest space-time path between two space-time events. From the perspective of the object, all motion occurs as if there were no gravitation whatsoever. It is only when observing the motion in a global sense that the curvature of space-time can be observed and the force is inferred from the object's curved path. Thus, the straight line path in space-time is seen as a curved line in space, and it is called the ballistic trajectory of the object. For example, a basketball thrown from the ground moves in a parabola, as it is in a uniform gravitational field. Its space-time trajectory (when the extra ct dimension is added) is almost a straight line, slightly curved (with the radius of curvature of the order of few light-years). The time derivative of the changing momentum of the object is what we label as \"gravitational force\"."}} +{"id":"358c9a29a17ce1f8","question":"In what theory is the idea of a number exchanged with that of Noetherian arithmetic?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Prime_number","context":"In ring theory, the notion of number is generally replaced with that of ideal. Prime ideals, which generalize prime elements in the sense that the principal ideal generated by a prime element is a prime ideal, are an important tool and object of study in commutative algebra, algebraic number theory and algebraic geometry. The prime ideals of the ring of integers are the ideals (0), (2), (3), (5), (7), (11), … The fundamental theorem of arithmetic generalizes to the Lasker–Noether theorem, which expresses every ideal in a Noetherian commutative ring as an intersection of primary ideals, which are the appropriate generalizations of prime powers."}} +{"id":"30a9448b1566eb89","question":"What did the Treaties not seek to do since its foundation?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"European_Union_law","context":"Since its foundation, the Treaties sought to enable people to pursue their life goals in any country through free movement. Reflecting the economic nature of the project, the European Community originally focused upon free movement of workers: as a \"factor of production\". However, from the 1970s, this focus shifted towards developing a more \"social\" Europe. Free movement was increasingly based on \"citizenship\", so that people had rights to empower them to become economically and socially active, rather than economic activity being a precondition for rights. This means the basic \"worker\" rights in TFEU article 45 function as a specific expression of the general rights of citizens in TFEU articles 18 to 21. According to the Court of Justice, a \"worker\" is anybody who is economically active, which includes everyone in an employment relationship, \"under the direction of another person\" for \"remuneration\". A job, however, need not be paid in money for someone to be protected as a worker. For example, in Steymann v Staatssecretaris van Justitie, a German man claimed the right to residence in the Netherlands, while he volunteered plumbing and household duties in the Bhagwan community, which provided for everyone's material needs irrespective of their contributions. The Court of Justice held that Mr Steymann was entitled to stay, so long as there was at least an \"indirect quid pro quo\" for the work he did. Having \"worker\" status means protection against all forms of discrimination by governments, and employers, in access to employment, tax, and social security rights. By contrast a citizen, who is \"any person having the nationality of a Member State\" (TFEU article 20(1)), has rights to seek work, vote in local and European elections, but more restricted rights to claim social security. In practice, free movement has become politically contentious as nationalist political parties have manipulated fears about immigrants taking away people's jobs and benefits (paradoxically at the same time). Nevertheless, practically \"all available research finds little impact\" of \"labour mobility on wages and employment of local workers\"."}} +{"id":"bcae98a56ae60a06","question":"What are more complex animals labeled when they have a jelly-like layer?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Ctenophora","context":"Like sponges and cnidarians, ctenophores have two main layers of cells that sandwich a middle layer of jelly-like material, which is called the mesoglea in cnidarians and ctenophores; more complex animals have three main cell layers and no intermediate jelly-like layer. Hence ctenophores and cnidarians have traditionally been labelled diploblastic, along with sponges. Both ctenophores and cnidarians have a type of muscle that, in more complex animals, arises from the middle cell layer, and as a result some recent text books classify ctenophores as triploblastic, while others still regard them as diploblastic."}} +{"id":"58a1c8fdc595fa0a","question":"What machines are not equally powerful in principle?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Computational_complexity_theory","context":"Many types of Turing machines are used to define complexity classes, such as deterministic Turing machines, probabilistic Turing machines, non-deterministic Turing machines, quantum Turing machines, symmetric Turing machines and alternating Turing machines. They are all equally powerful in principle, but when resources (such as time or space) are bounded, some of these may be more powerful than others."}} +{"id":"cefcee0fa7f2a17b","question":"Who designed the golf course located at the Sunnyside Country Club?","gold_answer":"William P. Bell","gold_chunk_ids":["cefcee0fa7f2a17b"],"is_unanswerable":false,"metadata":{"title":"Fresno,_California","context":"The neighborhood of Sunnyside is on Fresno's far southeast side, bounded by Chestnut Avenue to the West. Its major thoroughfares are Kings Canyon Avenue and Clovis Avenue. Although parts of Sunnyside are within the City of Fresno, much of the neighborhood is a \"county island\" within Fresno County. Largely developed in the 1950s through the 1970s, it has recently experienced a surge in new home construction. It is also the home of the Sunnyside Country Club, which maintains a golf course designed by William P. Bell."}} +{"id":"331d82d20e7152a8","question":"The words There shall be a Scottish Parliament is inscribed around the foot of the what?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Scottish_Parliament","context":"In front of the Presiding Officers' desk is the parliamentary mace, which is made from silver and inlaid with gold panned from Scottish rivers and inscribed with the words: Wisdom, Compassion, Justice and Integrity. The words There shall be a Scottish Parliament, which are the first words of the Scotland Act, are inscribed around the head of the mace, which has a formal ceremonial role in the meetings of Parliament, reinforcing the authority of the Parliament in its ability to make laws. Presented to the Scottish Parliament by the Queen upon its official opening in July 1999, the mace is displayed in a glass case suspended from the lid. At the beginning of each sitting in the chamber, the lid of the case is rotated so that the mace is above the glass, to symbolise that a full meeting of the Parliament is taking place."}} +{"id":"cd3546740c4afd0e","question":"Who was the vice president in 1962?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"University_of_Chicago","context":"The university experienced its share of student unrest during the 1960s, beginning in 1962, when students occupied President George Beadle's office in a protest over the university's off-campus rental policies. After continued turmoil, a university committee in 1967 issued what became known as the Kalven Report. The report, a two-page statement of the university's policy in \"social and political action,\" declared that \"To perform its mission in the society, a university must sustain an extraordinary environment of freedom of inquiry and maintain an independence from political fashions, passions, and pressures.\" The report has since been used to justify decisions such as the university's refusal to divest from South Africa in the 1980s and Darfur in the late 2000s."}} +{"id":"d513506da6d3c6a3","question":" What was the name of Theodore Roosevelt’s policy of non-imperialism?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Imperialism","context":"The early United States expressed its opposition to Imperialism, at least in a form distinct from its own Manifest Destiny, through policies such as the Monroe Doctrine. However, beginning in the late 19th and early 20th century, policies such as Theodore Roosevelt’s interventionism in Central America and Woodrow Wilson’s mission to \"make the world safe for democracy\" changed all this. They were often backed by military force, but were more often affected from behind the scenes. This is consistent with the general notion of hegemony and imperium of historical empires. In 1898, Americans who opposed imperialism created the Anti-Imperialist League to oppose the US annexation of the Philippines and Cuba. One year later, a war erupted in the Philippines causing business, labor and government leaders in the US to condemn America's occupation in the Philippines as they also denounced them for causing the deaths of many Filipinos. American foreign policy was denounced as a \"racket\" by Smedley Butler, an American general. He said, \"Looking back on it, I might have given Al Capone a few hints. The best he could do was to operate his racket in three districts. I operated on three continents\"."}} +{"id":"9dfb74fb52080433","question":"On-line betting was supported by what network frame?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Packet_switching","context":"AUSTPAC was an Australian public X.25 network operated by Telstra. Started by Telecom Australia in the early 1980s, AUSTPAC was Australia's first public packet-switched data network, supporting applications such as on-line betting, financial applications — the Australian Tax Office made use of AUSTPAC — and remote terminal access to academic institutions, who maintained their connections to AUSTPAC up until the mid-late 1990s in some cases. Access can be via a dial-up terminal to a PAD, or, by linking a permanent X.25 node to the network.[citation needed]"}} +{"id":"4ed94ca10f68a071","question":"What kind of climate does southern California maintain?","gold_answer":"Mediterranean","gold_chunk_ids":["4ed94ca10f68a071"],"is_unanswerable":false,"metadata":{"title":"Southern_California","context":"Southern California contains a Mediterranean climate, with infrequent rain and many sunny days. Summers are hot and dry, while winters are a bit warm or mild and wet. Serious rain can occur unusually. In the summers, temperature ranges are 90-60's while as winters are 70-50's, usually all of Southern California have Mediterranean climate. But snow is very rare in the Southwest of the state, it occurs on the Southeast of the state."}} +{"id":"ccc415d5bd186695","question":"What gang entered the West Side in 2008?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Fresno,_California","context":"The neighborhood includes Kearney Boulevard, named after early 20th century entrepreneur and millionaire M. Theo Kearney, which extends from Fresno Street in Southwest Fresno about 20 mi (32 km) west to Kerman, California. A small, two-lane rural road for most of its length, Kearney Boulevard is lined with tall palm trees. The roughly half-mile stretch of Kearney Boulevard between Fresno Street and Thorne Ave was at one time the preferred neighborhood for Fresno's elite African-American families. Another section, Brookhaven, on the southern edge of the West Side south of Jensen and west of Elm, was given the name by the Fresno City Council in an effort to revitalize the neighborhood's image. The isolated subdivision was for years known as the \"Dogg Pound\" in reference to a local gang, and as of late 2008 was still known for high levels of violent crime."}} +{"id":"a3f0a1ded69ed8fa","question":"What does Black's Law Dictionary say that rebellion doesn't have to be?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Civil_disobedience","context":"There have been debates as to whether civil disobedience must necessarily be non-violent. Black's Law Dictionary includes non-violence in its definition of civil disobedience. Christian Bay's encyclopedia article states that civil disobedience requires \"carefully chosen and legitimate means,\" but holds that they do not have to be non-violent. It has been argued that, while both civil disobedience and civil rebellion are justified by appeal to constitutional defects, rebellion is much more destructive; therefore, the defects justifying rebellion must be much more serious than those justifying disobedience, and if one cannot justify civil rebellion, then one cannot justify a civil disobedients' use of force and violence and refusal to submit to arrest. Civil disobedients' refraining from violence is also said to help preserve society's tolerance of civil disobedience."}} +{"id":"fccae80c65f4b529","question":"What are considered less accurately to be \"fundamental interactions\"?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Force","context":"In modern particle physics, forces and the acceleration of particles are explained as a mathematical by-product of exchange of momentum-carrying gauge bosons. With the development of quantum field theory and general relativity, it was realized that force is a redundant concept arising from conservation of momentum (4-momentum in relativity and momentum of virtual particles in quantum electrodynamics). The conservation of momentum can be directly derived from the homogeneity or symmetry of space and so is usually considered more fundamental than the concept of a force. Thus the currently known fundamental forces are considered more accurately to be \"fundamental interactions\".:199–128 When particle A emits (creates) or absorbs (annihilates) virtual particle B, a momentum conservation results in recoil of particle A making impression of repulsion or attraction between particles A A' exchanging by B. This description applies to all forces arising from fundamental interactions. While sophisticated mathematical descriptions are needed to predict, in full detail, the accurate result of such interactions, there is a conceptually simple way to describe such interactions through the use of Feynman diagrams. In a Feynman diagram, each matter particle is represented as a straight line (see world line) traveling through time, which normally increases up or to the right in the diagram. Matter and anti-matter particles are identical except for their direction of propagation through the Feynman diagram. World lines of particles intersect at interaction vertices, and the Feynman diagram represents any force arising from an interaction as occurring at the vertex with an associated instantaneous change in the direction of the particle world lines. Gauge bosons are emitted away from the vertex as wavy lines and, in the case of virtual particle exchange, are absorbed at an adjacent vertex."}} +{"id":"02a709181e611562","question":"What did most people not require regarding social advantages?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"European_Union_law","context":"The Free Movement of Workers Regulation articles 1 to 7 set out the main provisions on equal treatment of workers. First, articles 1 to 4 generally require that workers can take up employment, conclude contracts, and not suffer discrimination compared to nationals of the member state. In a famous case, the Belgian Football Association v Bosman, a Belgian footballer named Jean-Marc Bosman claimed that he should be able to transfer from R.F.C. de Liège to USL Dunkerque when his contract finished, regardless of whether Dunkerque could afford to pay Liège the habitual transfer fees. The Court of Justice held \"the transfer rules constitute[d] an obstacle to free movement\" and were unlawful unless they could be justified in the public interest, but this was unlikely. In Groener v Minister for Education the Court of Justice accepted that a requirement to speak Gaelic to teach in a Dublin design college could be justified as part of the public policy of promoting the Irish language, but only if the measure was not disproportionate. By contrast in Angonese v Cassa di Risparmio di Bolzano SpA a bank in Bolzano, Italy, was not allowed to require Mr Angonese to have a bilingual certificate that could only be obtained in Bolzano. The Court of Justice, giving \"horizontal\" direct effect to TFEU article 45, reasoned that people from other countries would have little chance of acquiring the certificate, and because it was \"impossible to submit proof of the required linguistic knowledge by any other means\", the measure was disproportionate. Second, article 7(2) requires equal treatment in respect of tax. In Finanzamt Köln Altstadt v Schumacker the Court of Justice held that it contravened TFEU art 45 to deny tax benefits (e.g. for married couples, and social insurance expense deductions) to a man who worked in Germany, but was resident in Belgium when other German residents got the benefits. By contrast in Weigel v Finanzlandesdirektion für Vorarlberg the Court of Justice rejected Mr Weigel's claim that a re-registration charge upon bringing his car to Austria violated his right to free movement. Although the tax was \"likely to have a negative bearing on the decision of migrant workers to exercise their right to freedom of movement\", because the charge applied equally to Austrians, in absence of EU legislation on the matter it had to be regarded as justified. Third, people must receive equal treatment regarding \"social advantages\", although the Court has approved residential qualifying periods. In Hendrix v Employee Insurance Institute the Court of Justice held that a Dutch national was not entitled to continue receiving incapacity benefits when he moved to Belgium, because the benefit was \"closely linked to the socio-economic situation\" of the Netherlands. Conversely, in Geven v Land Nordrhein-Westfalen the Court of Justice held that a Dutch woman living in the Netherlands, but working between 3 and 14 hours a week in Germany, did not have a right to receive German child benefits, even though the wife of a man who worked full-time in Germany but was resident in Austria could. The general justifications for limiting free movement in TFEU article 45(3) are \"public policy, public security or public health\", and there is also a general exception in article 45(4) for \"employment in the public service\"."}} +{"id":"641503a2ec9339ce","question":"What is the duration of Harvard Academic year?","gold_answer":"beginning in early September and ending in mid-May","gold_chunk_ids":["641503a2ec9339ce"],"is_unanswerable":false,"metadata":{"title":"Harvard_University","context":"Harvard's academic programs operate on a semester calendar beginning in early September and ending in mid-May. Undergraduates typically take four half-courses per term and must maintain a four-course rate average to be considered full-time. In many concentrations, students can elect to pursue a basic program or an honors-eligible program requiring a senior thesis and/or advanced course work. Students graduating in the top 4–5% of the class are awarded degrees summa cum laude, students in the next 15% of the class are awarded magna cum laude, and the next 30% of the class are awarded cum laude. Harvard has chapters of academic honor societies such as Phi Beta Kappa and various committees and departments also award several hundred named prizes annually. Harvard, along with other universities, has been accused of grade inflation, although there is evidence that the quality of the student body and its motivation have also increased. Harvard College reduced the number of students who receive Latin honors from 90% in 2004 to 60% in 2005. Moreover, the honors of \"John Harvard Scholar\" and \"Harvard College Scholar\" will now be given only to the top 5 percent and the next 5 percent of each class."}} +{"id":"3c616e583398793e","question":"The Maroons are apart of what association?","gold_answer":"the University Athletic Association","gold_chunk_ids":["3c616e583398793e"],"is_unanswerable":false,"metadata":{"title":"University_of_Chicago","context":"The Maroons compete in the NCAA's Division III as members of the University Athletic Association (UAA). The university was a founding member of the Big Ten Conference and participated in the NCAA Division I Men's Basketball and Football and was a regular participant in the Men's Basketball tournament. In 1935, the University of Chicago reached the Sweet Sixteen. In 1935, Chicago Maroons football player Jay Berwanger became the first winner of the Heisman Trophy. However, the university chose to withdraw from the conference in 1946 after University President Robert Maynard Hutchins de-emphasized varsity athletics in 1939 and dropped football. (In 1969, Chicago reinstated football as a Division III team, resuming playing its home games at the new Stagg Field.)"}} +{"id":"ed4c3bfa03dba48b","question":"What is Thomas B. Edsall's profession?","gold_answer":"journalist","gold_chunk_ids":["ed4c3bfa03dba48b"],"is_unanswerable":false,"metadata":{"title":"Economic_inequality","context":"Conservative researchers have argued that income inequality is not significant because consumption, rather than income should be the measure of inequality, and inequality of consumption is less extreme than inequality of income in the US. Will Wilkinson of the libertarian Cato Institute states that \"the weight of the evidence shows that the run-up in consumption inequality has been considerably less dramatic than the rise in income inequality,\" and consumption is more important than income. According to Johnson, Smeeding, and Tory, consumption inequality was actually lower in 2001 than it was in 1986. The debate is summarized in \"The Hidden Prosperity of the Poor\" by journalist Thomas B. Edsall. Other studies have not found consumption inequality less dramatic than household income inequality, and the CBO's study found consumption data not \"adequately\" capturing \"consumption by high-income households\" as it does their income, though it did agree that household consumption numbers show more equal distribution than household income."}} +{"id":"69c6c81a614c4cd9","question":"What did the agreement not aim to do regarding Germany?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"European_Union_law","context":"EU Competition law has its origins in the European Coal and Steel Community (ECSC) agreement between France, Italy, Belgium, the Netherlands, Luxembourg and Germany in 1951 following the second World War. The agreement aimed to prevent Germany from re-establishing dominance in the production of coal and steel as members felt that its dominance had contributed to the outbreak of the war. Article 65 of the agreement banned cartels and article 66 made provisions for concentrations, or mergers, and the abuse of a dominant position by companies. This was the first time that competition law principles were included in a plurilateral regional agreement and established the trans-European model of competition law. In 1957 competition rules were included in the Treaty of Rome, also known as the EC Treaty, which established the European Economic Community (EEC). The Treaty of Rome established the enactment of competition law as one of the main aims of the EEC through the \"institution of a system ensuring that competition in the common market is not distorted\". The two central provisions on EU competition law on companies were established in article 85, which prohibited anti-competitive agreements, subject to some exemptions, and article 86 prohibiting the abuse of dominant position. The treaty also established principles on competition law for member states, with article 90 covering public undertakings, and article 92 making provisions on state aid. Regulations on mergers were not included as member states could not establish consensus on the issue at the time."}} +{"id":"8516c1f65b09b864","question":"What is PPP?","gold_answer":"Public-Private Partnering","gold_chunk_ids":["8516c1f65b09b864"],"is_unanswerable":false,"metadata":{"title":"Construction","context":"There is also a growing number of new forms of procurement that involve relationship contracting where the emphasis is on a co-operative relationship between the principal and contractor and other stakeholders within a construction project. New forms include partnering such as Public-Private Partnering (PPPs) aka private finance initiatives (PFIs) and alliances such as \"pure\" or \"project\" alliances and \"impure\" or \"strategic\" alliances. The focus on co-operation is to ameliorate the many problems that arise from the often highly competitive and adversarial practices within the construction industry."}} +{"id":"68df146ab5e257a0","question":"How much did students pay in total to go to Harvard in 2007?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Harvard_University","context":"For the 2012–13 school year annual tuition was $38,000, with a total cost of attendance of $57,000. Beginning 2007, families with incomes below $60,000 pay nothing for their children to attend, including room and board. Families with incomes between $60,000 to $80,000 pay only a few thousand dollars per year, and families earning between $120,000 and $180,000 pay no more than 10% of their annual incomes. In 2009, Harvard offered grants totaling $414 million across all eleven divisions;[further explanation needed] $340 million came from institutional funds, $35 million from federal support, and $39 million from other outside support. Grants total 88% of Harvard's aid for undergraduate students, with aid also provided by loans (8%) and work-study (4%)."}} +{"id":"95998c0251d23812","question":"How long was the Southern Pacific line?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Fresno,_California","context":"In 1872, the Central Pacific Railroad established a station near Easterby's—by now a hugely productive wheat farm—for its new Southern Pacific line. Soon there was a store around the station and the store grew the town of Fresno Station, later called Fresno. Many Millerton residents, drawn by the convenience of the railroad and worried about flooding, moved to the new community. Fresno became an incorporated city in 1885. By 1931 the Fresno Traction Company operated 47 streetcars over 49 miles of track."}} +{"id":"5002df3b069bf8cf","question":"What plans of the British did this attach on Oneida Carry set back?","gold_answer":"hopes for campaigns on Lake Ontario, and endangered the Oswego garrison","gold_chunk_ids":["5002df3b069bf8cf"],"is_unanswerable":false,"metadata":{"title":"French_and_Indian_War","context":"Governor Vaudreuil, who harboured ambitions to become the French commander in chief (in addition to his role as governor), acted during the winter of 1756 before those reinforcements arrived. Scouts had reported the weakness of the British supply chain, so he ordered an attack against the forts Shirley had erected at the Oneida Carry. In the March Battle of Fort Bull, French forces destroyed the fort and large quantities of supplies, including 45,000 pounds of gunpowder. They set back any British hopes for campaigns on Lake Ontario, and endangered the Oswego garrison, already short on supplies. French forces in the Ohio valley also continued to intrigue with Indians throughout the area, encouraging them to raid frontier settlements. This led to ongoing alarms along the western frontiers, with streams of refugees returning east to get away from the action."}} +{"id":"69acc9511efe070c","question":"The concept of inertia can explain the tendency of people to continue in what?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Force","context":"The concept of inertia can be further generalized to explain the tendency of objects to continue in many different forms of constant motion, even those that are not strictly constant velocity. The rotational inertia of planet Earth is what fixes the constancy of the length of a day and the length of a year. Albert Einstein extended the principle of inertia further when he explained that reference frames subject to constant acceleration, such as those free-falling toward a gravitating object, were physically equivalent to inertial reference frames. This is why, for example, astronauts experience weightlessness when in free-fall orbit around the Earth, and why Newton's Laws of Motion are more easily discernible in such environments. If an astronaut places an object with mass in mid-air next to himself, it will remain stationary with respect to the astronaut due to its inertia. This is the same thing that would occur if the astronaut and the object were in intergalactic space with no net force of gravity acting on their shared reference frame. This principle of equivalence was one of the foundational underpinnings for the development of the general theory of relativity."}} +{"id":"07a33f6a20cb9b60","question":"What kind of contract is given when the contractor is given a performance specification and must undertake the project from design to construction, while adhering to the performance specifications?","gold_answer":"\"design build\" contract","gold_chunk_ids":["07a33f6a20cb9b60"],"is_unanswerable":false,"metadata":{"title":"Construction","context":"The modern trend in design is toward integration of previously separated specialties, especially among large firms. In the past, architects, interior designers, engineers, developers, construction managers, and general contractors were more likely to be entirely separate companies, even in the larger firms. Presently, a firm that is nominally an \"architecture\" or \"construction management\" firm may have experts from all related fields as employees, or to have an associated company that provides each necessary skill. Thus, each such firm may offer itself as \"one-stop shopping\" for a construction project, from beginning to end. This is designated as a \"design build\" contract where the contractor is given a performance specification and must undertake the project from design to construction, while adhering to the performance specifications."}} +{"id":"1ea1e62eeb88f8e5","question":"In an ideal moral society, what would no citizens be free from?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Economic_inequality","context":"Robert Nozick argued that government redistributes wealth by force (usually in the form of taxation), and that the ideal moral society would be one where all individuals are free from force. However, Nozick recognized that some modern economic inequalities were the result of forceful taking of property, and a certain amount of redistribution would be justified to compensate for this force but not because of the inequalities themselves. John Rawls argued in A Theory of Justice that inequalities in the distribution of wealth are only justified when they improve society as a whole, including the poorest members. Rawls does not discuss the full implications of his theory of justice. Some see Rawls's argument as a justification for capitalism since even the poorest members of society theoretically benefit from increased innovations under capitalism; others believe only a strong welfare state can satisfy Rawls's theory of justice."}} +{"id":"38aa499248809ae5","question":"What is the nature of the relationship between T-cells and vitamin D?","gold_answer":"symbiotic relationship","gold_chunk_ids":["38aa499248809ae5"],"is_unanswerable":false,"metadata":{"title":"Immune_system","context":"When a T-cell encounters a foreign pathogen, it extends a vitamin D receptor. This is essentially a signaling device that allows the T-cell to bind to the active form of vitamin D, the steroid hormone calcitriol. T-cells have a symbiotic relationship with vitamin D. Not only does the T-cell extend a vitamin D receptor, in essence asking to bind to the steroid hormone version of vitamin D, calcitriol, but the T-cell expresses the gene CYP27B1, which is the gene responsible for converting the pre-hormone version of vitamin D, calcidiol into the steroid hormone version, calcitriol. Only after binding to calcitriol can T-cells perform their intended function. Other immune system cells that are known to express CYP27B1 and thus activate vitamin D calcidiol, are dendritic cells, keratinocytes and macrophages."}} +{"id":"eb2f38cd61b3b0b0","question":"Who first wrote about the Rhine's discovery and border?","gold_answer":"Maurus Servius Honoratus","gold_chunk_ids":["eb2f38cd61b3b0b0"],"is_unanswerable":false,"metadata":{"title":"Rhine","context":"The Rhine was not known to Herodotus and first enters the historical period in the 1st century BC in Roman-era geography. At that time, it formed the boundary between Gaul and Germania. The Upper Rhine had been part of the areal of the late Hallstatt culture since the 6th century BC, and by the 1st century BC, the areal of the La Tène culture covered almost its entire length, forming a contact zone with the Jastorf culture, i.e. the locus of early Celtic-Germanic cultural contact. In Roman geography, the Rhine formed the boundary between Gallia and Germania by definition; e.g. Maurus Servius Honoratus, Commentary on the Aeneid of Vergil (8.727) (Rhenus) fluvius Galliae, qui Germanos a Gallia dividit \"(The Rhine is a) river of Gaul, which divides the Germanic people from Gaul.\""}} +{"id":"dab14acdfa09a0e8","question":"What is the name of the property that designates a number as being prime or not?","gold_answer":"primality","gold_chunk_ids":["dab14acdfa09a0e8"],"is_unanswerable":false,"metadata":{"title":"Prime_number","context":"The property of being prime (or not) is called primality. A simple but slow method of verifying the primality of a given number n is known as trial division. It consists of testing whether n is a multiple of any integer between 2 and . Algorithms much more efficient than trial division have been devised to test the primality of large numbers. These include the Miller–Rabin primality test, which is fast but has a small probability of error, and the AKS primality test, which always produces the correct answer in polynomial time but is too slow to be practical. Particularly fast methods are available for numbers of special forms, such as Mersenne numbers. As of January 2016[update], the largest known prime number has 22,338,618 decimal digits."}} +{"id":"d73f05d08d230708","question":"What was extent of Celeron's expedition?","gold_answer":"about 3,000 miles (4,800 km) between June and November 1749.","gold_chunk_ids":["d73f05d08d230708"],"is_unanswerable":false,"metadata":{"title":"French_and_Indian_War","context":"Céloron's expedition force consisted of about 200 Troupes de la marine and 30 Indians. The expedition covered about 3,000 miles (4,800 km) between June and November 1749. It went up the St. Lawrence, continued along the northern shore of Lake Ontario, crossed the portage at Niagara, and followed the southern shore of Lake Erie. At the Chautauqua Portage (near present-day Barcelona, New York), the expedition moved inland to the Allegheny River, which it followed to the site of present-day Pittsburgh. There Céloron buried lead plates engraved with the French claim to the Ohio Country. Whenever he encountered British merchants or fur-traders, Céloron informed them of the French claims on the territory and told them to leave."}} +{"id":"b54eb1f2c05e2017","question":"By how many kilometers are shear waves separated when measuring the crust?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Geology","context":"Seismologists can use the arrival times of seismic waves in reverse to image the interior of the Earth. Early advances in this field showed the existence of a liquid outer core (where shear waves were not able to propagate) and a dense solid inner core. These advances led to the development of a layered model of the Earth, with a crust and lithosphere on top, the mantle below (separated within itself by seismic discontinuities at 410 and 660 kilometers), and the outer core and inner core below that. More recently, seismologists have been able to create detailed images of wave speeds inside the earth in the same way a doctor images a body in a CT scan. These images have led to a much more detailed view of the interior of the Earth, and have replaced the simplified layered model with a much more dynamic model."}} +{"id":"47b5b95aee737389","question":"What other topics can Civil disobedience pertain to?","gold_answer":"cultural traditions, social customs, religious beliefs","gold_chunk_ids":["47b5b95aee737389"],"is_unanswerable":false,"metadata":{"title":"Civil_disobedience","context":"Non-revolutionary civil disobedience is a simple disobedience of laws on the grounds that they are judged \"wrong\" by an individual conscience, or as part of an effort to render certain laws ineffective, to cause their repeal, or to exert pressure to get one's political wishes on some other issue. Revolutionary civil disobedience is more of an active attempt to overthrow a government (or to change cultural traditions, social customs, religious beliefs, etc...revolution doesn't have to be political, i.e. \"cultural revolution\", it simply implies sweeping and widespread change to a section of the social fabric). Gandhi's acts have been described as revolutionary civil disobedience. It has been claimed that the Hungarians under Ferenc Deák directed revolutionary civil disobedience against the Austrian government. Thoreau also wrote of civil disobedience accomplishing \"peaceable revolution.\" Howard Zinn, Harvey Wheeler, and others have identified the right espoused in The Declaration of Independence to \"alter or abolish\" an unjust government to be a principle of civil disobedience. "}} +{"id":"925cc7904d2b30e7","question":"Dial up or dedicated async connections connected who? ","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Packet_switching","context":"Tymnet was an international data communications network headquartered in San Jose, CA that utilized virtual call packet switched technology and used X.25, SNA/SDLC, BSC and ASCII interfaces to connect host computers (servers)at thousands of large companies, educational institutions, and government agencies. Users typically connected via dial-up connections or dedicated async connections. The business consisted of a large public network that supported dial-up users and a private network business that allowed government agencies and large companies (mostly banks and airlines) to build their own dedicated networks. The private networks were often connected via gateways to the public network to reach locations not on the private network. Tymnet was also connected to dozens of other public networks in the U.S. and internationally via X.25/X.75 gateways. (Interesting note: Tymnet was not named after Mr. Tyme. Another employee suggested the name.) "}} +{"id":"98ef9b787218f00a","question":"What did Kllbrandon's report in 1973 recommend establishing?","gold_answer":"directly elected Scottish Assembly","gold_chunk_ids":["98ef9b787218f00a"],"is_unanswerable":false,"metadata":{"title":"Scottish_Parliament","context":"For the next three hundred years, Scotland was directly governed by the Parliament of Great Britain and the subsequent Parliament of the United Kingdom, both seated at Westminster, and the lack of a Parliament of Scotland remained an important element in Scottish national identity. Suggestions for a 'devolved' Parliament were made before 1914, but were shelved due to the outbreak of the First World War. A sharp rise in nationalism in Scotland during the late 1960s fuelled demands for some form of home rule or complete independence, and in 1969 prompted the incumbent Labour government of Harold Wilson to set up the Kilbrandon Commission to consider the British constitution. One of the principal objectives of the commission was to examine ways of enabling more self-government for Scotland, within the unitary state of the United Kingdom. Kilbrandon published his report in 1973 recommending the establishment of a directly elected Scottish Assembly to legislate for the majority of domestic Scottish affairs."}} +{"id":"461afc946735ec59","question":"What is the term of office for each house member?","gold_answer":"four years","gold_chunk_ids":["461afc946735ec59"],"is_unanswerable":false,"metadata":{"title":"Victoria_(Australia)","context":"In November 2006, the Victorian Legislative Council elections were held under a new multi-member proportional representation system. The State of Victoria was divided into eight electorates with each electorate represented by five representatives elected by Single Transferable Vote. The total number of upper house members was reduced from 44 to 40 and their term of office is now the same as the lower house members—four years. Elections for the Victorian Parliament are now fixed and occur in November every four years. Prior to the 2006 election, the Legislative Council consisted of 44 members elected to eight-year terms from 22 two-member electorates."}} +{"id":"675cd86ab5405272","question":"By what process can active immunity be generated in an artificial manner?","gold_answer":"vaccination","gold_chunk_ids":["675cd86ab5405272"],"is_unanswerable":false,"metadata":{"title":"Immune_system","context":"Long-term active memory is acquired following infection by activation of B and T cells. Active immunity can also be generated artificially, through vaccination. The principle behind vaccination (also called immunization) is to introduce an antigen from a pathogen in order to stimulate the immune system and develop specific immunity against that particular pathogen without causing disease associated with that organism. This deliberate induction of an immune response is successful because it exploits the natural specificity of the immune system, as well as its inducibility. With infectious disease remaining one of the leading causes of death in the human population, vaccination represents the most effective manipulation of the immune system mankind has developed."}} +{"id":"a81e9b0af2d3e59a","question":"Why is there a need for a dedicated path?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Packet_switching","context":"In connectionless mode each packet includes complete addressing information. The packets are routed individually, sometimes resulting in different paths and out-of-order delivery. Each packet is labeled with a destination address, source address, and port numbers. It may also be labeled with the sequence number of the packet. This precludes the need for a dedicated path to help the packet find its way to its destination, but means that much more information is needed in the packet header, which is therefore larger, and this information needs to be looked up in power-hungry content-addressable memory. Each packet is dispatched and may go via different routes; potentially, the system has to do as much work for every packet as the connection-oriented system has to do in connection set-up, but with less information as to the application's requirements. At the destination, the original message/data is reassembled in the correct order, based on the packet sequence number. Thus a virtual connection, also known as a virtual circuit or byte stream is provided to the end-user by a transport layer protocol, although intermediate network nodes only provides a connectionless network layer service."}} +{"id":"9d37c1ebfc72c44b","question":"What has replaced lower skilled workers in the United States?","gold_answer":"machine labor","gold_chunk_ids":["9d37c1ebfc72c44b"],"is_unanswerable":false,"metadata":{"title":"Economic_inequality","context":"Trade liberalization may shift economic inequality from a global to a domestic scale. When rich countries trade with poor countries, the low-skilled workers in the rich countries may see reduced wages as a result of the competition, while low-skilled workers in the poor countries may see increased wages. Trade economist Paul Krugman estimates that trade liberalisation has had a measurable effect on the rising inequality in the United States. He attributes this trend to increased trade with poor countries and the fragmentation of the means of production, resulting in low skilled jobs becoming more tradeable. However, he concedes that the effect of trade on inequality in America is minor when compared to other causes, such as technological innovation, a view shared by other experts. Empirical economists Max Roser and Jesus Crespo-Cuaresma find support in the data that international trade is increasing income inequality. They empirically confirm the predictions of the Stolper–Samuelson theorem regarding the effects of international trade on the distribution of incomes. Lawrence Katz estimates that trade has only accounted for 5-15% of rising income inequality. Robert Lawrence argues that technological innovation and automation has meant that low-skilled jobs have been replaced by machine labor in wealthier nations, and that wealthier countries no longer have significant numbers of low-skilled manufacturing workers that could be affected by competition from poor countries."}} +{"id":"dd59fdf05c6802c4","question":"What is the most critical resource in the analysis of computational problems associated with non-deterministic Turing machines?","gold_answer":"time","gold_chunk_ids":["dd59fdf05c6802c4"],"is_unanswerable":false,"metadata":{"title":"Computational_complexity_theory","context":"However, some computational problems are easier to analyze in terms of more unusual resources. For example, a non-deterministic Turing machine is a computational model that is allowed to branch out to check many different possibilities at once. The non-deterministic Turing machine has very little to do with how we physically want to compute algorithms, but its branching exactly captures many of the mathematical models we want to analyze, so that non-deterministic time is a very important resource in analyzing computational problems."}} +{"id":"8cb74401c81ef526","question":"What types of other schools is the city of Harris well known for?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"University_of_Chicago","context":"The University of Chicago (UChicago, Chicago, or U of C) is a private research university in Chicago. The university, established in 1890, consists of The College, various graduate programs, interdisciplinary committees organized into four academic research divisions and seven professional schools. Beyond the arts and sciences, Chicago is also well known for its professional schools, which include the Pritzker School of Medicine, the University of Chicago Booth School of Business, the Law School, the School of Social Service Administration, the Harris School of Public Policy Studies, the Graham School of Continuing Liberal and Professional Studies and the Divinity School. The university currently enrolls approximately 5,000 students in the College and around 15,000 students overall."}} +{"id":"fab8024494aec3a4","question":"What do the top 400 richest Americans have more of than half of all Americans combined?","gold_answer":"wealth","gold_chunk_ids":["fab8024494aec3a4"],"is_unanswerable":false,"metadata":{"title":"Economic_inequality","context":"According to PolitiFact the top 400 richest Americans \"have more wealth than half of all Americans combined.\" According to the New York Times on July 22, 2014, the \"richest 1 percent in the United States now own more wealth than the bottom 90 percent\". Inherited wealth may help explain why many Americans who have become rich may have had a \"substantial head start\". In September 2012, according to the Institute for Policy Studies, \"over 60 percent\" of the Forbes richest 400 Americans \"grew up in substantial privilege\"."}} +{"id":"c1bbef069335dbe4","question":"Aristotle believed that objects in motion on Earth would stay that way if what?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Force","context":"Aristotle provided a philosophical discussion of the concept of a force as an integral part of Aristotelian cosmology. In Aristotle's view, the terrestrial sphere contained four elements that come to rest at different \"natural places\" therein. Aristotle believed that motionless objects on Earth, those composed mostly of the elements earth and water, to be in their natural place on the ground and that they will stay that way if left alone. He distinguished between the innate tendency of objects to find their \"natural place\" (e.g., for heavy bodies to fall), which led to \"natural motion\", and unnatural or forced motion, which required continued application of a force. This theory, based on the everyday experience of how objects move, such as the constant application of a force needed to keep a cart moving, had conceptual trouble accounting for the behavior of projectiles, such as the flight of arrows. The place where the archer moves the projectile was at the start of the flight, and while the projectile sailed through the air, no discernible efficient cause acts on it. Aristotle was aware of this problem and proposed that the air displaced through the projectile's path carries the projectile to its target. This explanation demands a continuum like air for change of place in general."}} +{"id":"311cffb4597c07fe","question":"Which building was vacated three times to allow for the meeting of the Church's General Assembly?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Scottish_Parliament","context":"Whilst the permanent building at Holyrood was being constructed, the Parliament's temporary home was the General Assembly Hall of the Church of Scotland on the Royal Mile in Edinburgh. Official photographs and television interviews were held in the courtyard adjoining the Assembly Hall, which is part of the School of Divinity of the University of Edinburgh. This building was vacated twice to allow for the meeting of the Church's General Assembly. In May 2000, the Parliament was temporarily relocated to the former Strathclyde Regional Council debating chamber in Glasgow, and to the University of Aberdeen in May 2002."}} +{"id":"c3d2d91d77d82581","question":"What stipend do students enrolled in priority courses receive?","gold_answer":"Tuition Fee Supplement","gold_chunk_ids":["c3d2d91d77d82581"],"is_unanswerable":false,"metadata":{"title":"Private_school","context":"The Education Service Contracting scheme of the government provides financial assistance for tuition and other school fees of students turned away from public high schools because of enrollment overflows. The Tuition Fee Supplement is geared to students enrolled in priority courses in post-secondary and non-degree programmes, including vocational and technical courses. The Private Education Student Financial Assistance is made available to underprivileged, but deserving high school graduates, who wish to pursue college/technical education in private colleges and universities."}} +{"id":"882319efd2e22dc5","question":"Who did the Mongols bring to Japan as administrators?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Yuan_dynasty","context":"At the same time the Mongols imported Central Asian Muslims to serve as administrators in China, the Mongols also sent Han Chinese and Khitans from China to serve as administrators over the Muslim population in Bukhara in Central Asia, using foreigners to curtail the power of the local peoples of both lands. Han Chinese were moved to Central Asian areas like Besh Baliq, Almaliq, and Samarqand by the Mongols where they worked as artisans and farmers. Alans were recruited into the Mongol forces with one unit called \"Right Alan Guard\" which was combined with \"recently surrendered\" soldiers, Mongols, and Chinese soldiers stationed in the area of the former Kingdom of Qocho and in Besh Balikh the Mongols established a Chinese military colony led by Chinese general Qi Kongzhi (Ch'i Kung-chih). After the Mongol conquest of Central Asia by Genghis Khan, foreigners were chosen as administrators and co-management with Chinese and Qara-Khitays (Khitans) of gardens and fields in Samarqand was put upon the Muslims as a requirement since Muslims were not allowed to manage without them. The Mongol appointed Governor of Samarqand was a Qara-Khitay (Khitan), held the title Taishi, familiar with Chinese culture his name was Ahai"}} +{"id":"b10a4de3f00df087","question":"What park covers an area of 76 ha.?","gold_answer":"Łazienki","gold_chunk_ids":["b10a4de3f00df087"],"is_unanswerable":false,"metadata":{"title":"Warsaw","context":"The Saxon Garden, covering the area of 15.5 ha, was formally a royal garden. There are over 100 different species of trees and the avenues are a place to sit and relax. At the east end of the park, the Tomb of the Unknown Soldier is situated. In the 19th century the Krasiński Palace Garden was remodelled by Franciszek Szanior. Within the central area of the park one can still find old trees dating from that period: maidenhair tree, black walnut, Turkish hazel and Caucasian wingnut trees. With its benches, flower carpets, a pond with ducks on and a playground for kids, the Krasiński Palace Garden is a popular strolling destination for the Varsovians. The Monument of the Warsaw Ghetto Uprising is also situated here. The Łazienki Park covers the area of 76 ha. The unique character and history of the park is reflected in its landscape architecture (pavilions, sculptures, bridges, cascades, ponds) and vegetation (domestic and foreign species of trees and bushes). What makes this park different from other green spaces in Warsaw is the presence of peacocks and pheasants, which can be seen here walking around freely, and royal carps in the pond. The Wilanów Palace Park, dates back to the second half of the 17th century. It covers the area of 43 ha. Its central French-styled area corresponds to the ancient, baroque forms of the palace. The eastern section of the park, closest to the Palace, is the two-level garden with a terrace facing the pond. The park around the Królikarnia Palace is situated on the old escarpment of the Vistula. The park has lanes running on a few levels deep into the ravines on both sides of the palace."}} +{"id":"9a6ffee44fcfb169","question":"What could inflammation do during sleep periods?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Immune_system","context":"In contrast, during wake periods differentiated effector cells, such as cytotoxic natural killer cells and CTLs (cytotoxic T lymphocytes), peak in order to elicit an effective response against any intruding pathogens. As well during awake active times, anti-inflammatory molecules, such as cortisol and catecholamines, peak. There are two theories as to why the pro-inflammatory state is reserved for sleep time. First, inflammation would cause serious cognitive and physical impairments if it were to occur during wake times. Second, inflammation may occur during sleep times due to the presence of melatonin. Inflammation causes a great deal of oxidative stress and the presence of melatonin during sleep times could actively counteract free radical production during this time."}} +{"id":"903940b4ef3beb0f","question":" bassett doesn't focus on what to illustrate his idea?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Imperialism","context":"To better illustrate this idea, Bassett focuses his analysis of the role of nineteenth-century maps during the \"scramble for Africa\". He states that maps \"contributed to empire by promoting, assisting, and legitimizing the extension of French and British power into West Africa\". During his analysis of nineteenth-century cartographic techniques, he highlights the use of blank space to denote unknown or unexplored territory. This provided incentives for imperial and colonial powers to obtain \"information to fill in blank spaces on contemporary maps\"."}} +{"id":"30e35b0865cf2a04","question":"Parliament elects two MSPs to serve as what officers?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Scottish_Parliament","context":"After each election to the Scottish Parliament, at the beginning of each parliamentary session, Parliament elects one MSP to serve as Presiding Officer, the equivalent of the speaker (currently Tricia Marwick), and two MSPs to serve as deputies (currently Elaine Smith and John Scott). The Presiding Officer and deputies are elected by a secret ballot of the 129 MSPs, which is the only secret ballot conducted in the Scottish Parliament. Principally, the role of the Presiding Officer is to chair chamber proceedings and the Scottish Parliamentary Corporate Body. When chairing meetings of the Parliament, the Presiding Officer and his/her deputies must be politically impartial. During debates, the Presiding Officer (or the deputy) is assisted by the parliamentary clerks, who give advice on how to interpret the standing orders that govern the proceedings of meetings. A vote clerk sits in front of the Presiding Officer and operates the electronic voting equipment and chamber clocks."}} +{"id":"0e2d93cac17ec6fd","question":"What was normal British defense?","gold_answer":"mustered local militia companies, generally ill trained and available only for short periods, to deal with native threats, but did not have any standing forces.","gold_chunk_ids":["0e2d93cac17ec6fd"],"is_unanswerable":false,"metadata":{"title":"French_and_Indian_War","context":"At the start of the war, no French regular army troops were stationed in North America, and few British troops. New France was defended by about 3,000 troupes de la marine, companies of colonial regulars (some of whom had significant woodland combat experience). The colonial government recruited militia support when needed. Most British colonies mustered local militia companies, generally ill trained and available only for short periods, to deal with native threats, but did not have any standing forces."}} +{"id":"f025a464e2b3c241","question":"In what city or the Arabs the twelve largest ethnic group?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Jacksonville,_Florida","context":"Jacksonville is the most populous city in Florida, and the twelfth most populous city in the United States. As of 2010[update], there were 821,784 people and 366,273 households in the city. Jacksonville has the country's tenth-largest Arab population, with a total population of 5,751 according to the 2000 United States Census. Jacksonville has Florida's largest Filipino American community, with 25,033 in the metropolitan area as of the 2010 Census. Much of Jacksonville's Filipino community served in or has ties to the United States Navy."}} +{"id":"f580ea5842675b6c","question":"What can orthogonal forces be when there are three components with two at right angles to each other?","gold_answer":"three-dimensional","gold_chunk_ids":["f580ea5842675b6c"],"is_unanswerable":false,"metadata":{"title":"Force","context":"As well as being added, forces can also be resolved into independent components at right angles to each other. A horizontal force pointing northeast can therefore be split into two forces, one pointing north, and one pointing east. Summing these component forces using vector addition yields the original force. Resolving force vectors into components of a set of basis vectors is often a more mathematically clean way to describe forces than using magnitudes and directions. This is because, for orthogonal components, the components of the vector sum are uniquely determined by the scalar addition of the components of the individual vectors. Orthogonal components are independent of each other because forces acting at ninety degrees to each other have no effect on the magnitude or direction of the other. Choosing a set of orthogonal basis vectors is often done by considering what set of basis vectors will make the mathematics most convenient. Choosing a basis vector that is in the same direction as one of the forces is desirable, since that force would then have only one non-zero component. Orthogonal force vectors can be three-dimensional with the third component being at right-angles to the other two."}} +{"id":"293e02dcb2b3e577","question":"How were X.75, ASCII, and other interfaces used?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Packet_switching","context":"Tymnet was an international data communications network headquartered in San Jose, CA that utilized virtual call packet switched technology and used X.25, SNA/SDLC, BSC and ASCII interfaces to connect host computers (servers)at thousands of large companies, educational institutions, and government agencies. Users typically connected via dial-up connections or dedicated async connections. The business consisted of a large public network that supported dial-up users and a private network business that allowed government agencies and large companies (mostly banks and airlines) to build their own dedicated networks. The private networks were often connected via gateways to the public network to reach locations not on the private network. Tymnet was also connected to dozens of other public networks in the U.S. and internationally via X.25/X.75 gateways. (Interesting note: Tymnet was not named after Mr. Tyme. Another employee suggested the name.) "}} +{"id":"0142e125d109838b","question":"When did OPEC issue a joint communique? ","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"1973_oil_crisis","context":"On August 15, 1971, the United States unilaterally pulled out of the Bretton Woods Accord. The US abandoned the Gold Exchange Standard whereby the value of the dollar had been pegged to the price of gold and all other currencies were pegged to the dollar, whose value was left to \"float\" (rise and fall according to market demand). Shortly thereafter, Britain followed, floating the pound sterling. The other industrialized nations followed suit with their respective currencies. Anticipating that currency values would fluctuate unpredictably for a time, the industrialized nations increased their reserves (by expanding their money supplies) in amounts far greater than before. The result was a depreciation of the dollar and other industrialized nations' currencies. Because oil was priced in dollars, oil producers' real income decreased. In September 1971, OPEC issued a joint communiqué stating that, from then on, they would price oil in terms of a fixed amount of gold."}} +{"id":"f9ee7ad28de0df08","question":"What is the area called near the Rhine Gorge with castles from the middle ages?","gold_answer":"the Romantic Rhine","gold_chunk_ids":["f9ee7ad28de0df08"],"is_unanswerable":false,"metadata":{"title":"Rhine","context":"Between Bingen and Bonn, the Middle Rhine flows through the Rhine Gorge, a formation which was created by erosion. The rate of erosion equaled the uplift in the region, such that the river was left at about its original level while the surrounding lands raised. The gorge is quite deep and is the stretch of the river which is known for its many castles and vineyards. It is a UNESCO World Heritage Site (2002) and known as \"the Romantic Rhine\", with more than 40 castles and fortresses from the Middle Ages and many quaint and lovely country villages."}} +{"id":"5ed7bd5a9ead2a92","question":"What uses a flexible set of rules to determine its future actions?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Computational_complexity_theory","context":"A deterministic Turing machine is the most basic Turing machine, which uses a fixed set of rules to determine its future actions. A probabilistic Turing machine is a deterministic Turing machine with an extra supply of random bits. The ability to make probabilistic decisions often helps algorithms solve problems more efficiently. Algorithms that use random bits are called randomized algorithms. A non-deterministic Turing machine is a deterministic Turing machine with an added feature of non-determinism, which allows a Turing machine to have multiple possible future actions from a given state. One way to view non-determinism is that the Turing machine branches into many possible computational paths at each step, and if it solves the problem in any of these branches, it is said to have solved the problem. Clearly, this model is not meant to be a physically realizable model, it is just a theoretically interesting abstract machine that gives rise to particularly interesting complexity classes. For examples, see non-deterministic algorithm."}} +{"id":"1d300c27653291bb","question":"Who claimed that the name Black Death first appeared in 1631?","gold_answer":"Gasquet","gold_chunk_ids":["1d300c27653291bb"],"is_unanswerable":false,"metadata":{"title":"Black_Death","context":"Gasquet (1908) claimed that the Latin name atra mors (Black Death) for the 14th-century epidemic first appeared in modern times in 1631 in a book on Danish history by J.I. Pontanus: \"Vulgo & ab effectu atram mortem vocatibant. (\"Commonly and from its effects, they called it the black death\"). The name spread through Scandinavia and then Germany, gradually becoming attached to the mid 14th-century epidemic as a proper name. In England, it was not until 1823 that the medieval epidemic was first called the Black Death."}} +{"id":"73a4dfa3521d86fd","question":"What army was pushing deep into Polish territory to pursue the Germans in 1944?","gold_answer":"the Red Army","gold_chunk_ids":["73a4dfa3521d86fd"],"is_unanswerable":false,"metadata":{"title":"Warsaw","context":"By July 1944, the Red Army was deep into Polish territory and pursuing the Germans toward Warsaw. Knowing that Stalin was hostile to the idea of an independent Poland, the Polish government-in-exile in London gave orders to the underground Home Army (AK) to try to seize control of Warsaw from the Germans before the Red Army arrived. Thus, on 1 August 1944, as the Red Army was nearing the city, the Warsaw Uprising began. The armed struggle, planned to last 48 hours, was partially successful, however it went on for 63 days. Eventually the Home Army fighters and civilians assisting them were forced to capitulate. They were transported to PoW camps in Germany, while the entire civilian population was expelled. Polish civilian deaths are estimated at between 150,000 and 200,000."}} +{"id":"493c53c268bc365e","question":"What state in Australia invented dairy farming?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Victoria_(Australia)","context":"Victoria is the centre of dairy farming in Australia. It is home to 60% of Australia's 3 million dairy cattle and produces nearly two-thirds of the nation's milk, almost 6.4 billion litres. The state also has 2.4 million beef cattle, with more than 2.2 million cattle and calves slaughtered each year. In 2003–04, Victorian commercial fishing crews and aquaculture industry produced 11,634 tonnes of seafood valued at nearly A$109 million. Blacklipped abalone is the mainstay of the catch, bringing in A$46 million, followed by southern rock lobster worth A$13.7 million. Most abalone and rock lobster is exported to Asia."}} +{"id":"8e1b9a45d45c4de7","question":"What other business district does Orange County envelop outside of Downtown Santa Ana and Newport Center?","gold_answer":"South Coast Metro","gold_chunk_ids":["8e1b9a45d45c4de7"],"is_unanswerable":false,"metadata":{"title":"Southern_California","context":"Orange County is a rapidly developing business center that includes Downtown Santa Ana, the South Coast Metro and Newport Center districts; as well as the Irvine business centers of The Irvine Spectrum, West Irvine, and international corporations headquartered at the University of California, Irvine. West Irvine includes the Irvine Tech Center and Jamboree Business Parks."}} +{"id":"92cb50001527b5b6","question":"What is the Upati Garden in Polish?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Warsaw","context":"Nearby, in Ogród Saski (the Saxon Garden), the Summer Theatre was in operation from 1870 to 1939, and in the inter-war period, the theatre complex also included Momus, Warsaw's first literary cabaret, and Leon Schiller's musical theatre Melodram. The Wojciech Bogusławski Theatre (1922–26), was the best example of \"Polish monumental theatre\". From the mid-1930s, the Great Theatre building housed the Upati Institute of Dramatic Arts – the first state-run academy of dramatic art, with an acting department and a stage directing department."}} +{"id":"838eb6e5a47c79ad","question":"What town was actually granted to the Huguenots on arrival?","gold_answer":"Manakin Town","gold_chunk_ids":["838eb6e5a47c79ad"],"is_unanswerable":false,"metadata":{"title":"Huguenot","context":"In 1700 several hundred French Huguenots migrated from England to the colony of Virginia, where the English Crown had promised them land grants in Lower Norfolk County. When they arrived, colonial authorities offered them instead land 20 miles above the falls of the James River, at the abandoned Monacan village known as Manakin Town, now in Powhatan County. Some settlers landed in present-day Chesterfield County. On 12 May 1705, the Virginia General Assembly passed an act to naturalise the 148 Huguenots still resident at Manakintown. Of the original 390 settlers in the isolated settlement, many had died; others lived outside town on farms in the English style; and others moved to different areas. Gradually they intermarried with their English neighbors. Through the 18th and 19th centuries, descendants of the French migrated west into the Piedmont, and across the Appalachian Mountains into the West of what became Kentucky, Tennessee, Missouri, and other states. In the Manakintown area, the Huguenot Memorial Bridge across the James River and Huguenot Road were named in their honor, as were many local features, including several schools, including Huguenot High School."}} +{"id":"e1769ec88bbcf92d","question":" What declined when Arab nationalism suffered?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Islamism","context":"The quick and decisive defeat of the Arab troops during the Six-Day War by Israeli troops constituted a pivotal event in the Arab Muslim world. The defeat along with economic stagnation in the defeated countries, was blamed on the secular Arab nationalism of the ruling regimes. A steep and steady decline in the popularity and credibility of secular, socialist and nationalist politics ensued. Ba'athism, Arab socialism, and Arab nationalism suffered, and different democratic and anti-democratic Islamist movements inspired by Maududi and Sayyid Qutb gained ground."}} +{"id":"d0126cc69ce384ad","question":"Who besides the Russians are often left out of the colonialism debat?","gold_answer":"Ottoman","gold_chunk_ids":["d0126cc69ce384ad"],"is_unanswerable":false,"metadata":{"title":"Imperialism","context":"The term \"imperialism\" is often conflated with \"colonialism\", however many scholars have argued that each have their own distinct definition. Imperialism and colonialism have been used in order to describe one's superiority, domination and influence upon a person or group of people. Robert Young writes that while imperialism operates from the center, is a state policy and is developed for ideological as well as financial reasons, colonialism is simply the development for settlement or commercial intentions. Colonialism in modern usage also tends to imply a degree of geographic separation between the colony and the imperial power. Particularly, Edward Said distinguishes the difference between imperialism and colonialism by stating; \"imperialism involved 'the practice, the theory and the attitudes of a dominating metropolitan center ruling a distant territory', while colonialism refers to the 'implanting of settlements on a distant territory.' Contiguous land empires such as the Russian or Ottoman are generally excluded from discussions of colonialism.:116 Thus it can be said that imperialism includes some form of colonialism, but colonialism itself does not automatically imply imperialism, as it lacks a political focus.[further explanation needed]"}} +{"id":"a3f3777ee2887a53","question":"Minister Robert Dinwiddie had an investment in what significant company?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"French_and_Indian_War","context":"Governor Robert Dinwiddie of Virginia was an investor in the Ohio Company, which stood to lose money if the French held their claim. To counter the French military presence in Ohio, in October 1753 Dinwiddie ordered the 21-year-old Major George Washington (whose brother was another Ohio Company investor) of the Virginia Regiment to warn the French to leave Virginia territory. Washington left with a small party, picking up along the way Jacob Van Braam as an interpreter; Christopher Gist, a company surveyor working in the area; and a few Mingo led by Tanaghrisson. On December 12, Washington and his men reached Fort Le Boeuf."}} +{"id":"2c3dd04e1b3406ba","question":"What was the capital of the Ottoman empire?","gold_answer":"Istanbul","gold_chunk_ids":["2c3dd04e1b3406ba"],"is_unanswerable":false,"metadata":{"title":"Imperialism","context":"With Istanbul as its capital and control of lands around the Mediterranean basin, the Ottoman Empire was at the center of interactions between the Eastern and Western worlds for six centuries. Following a long period of military setbacks against European powers, the Ottoman Empire gradually declined into the late nineteenth century. The empire allied with Germany in the early 20th century, with the imperial ambition of recovering its lost territories, but it dissolved in the aftermath of World War I, leading to the emergence of the new state of Turkey in the Ottoman Anatolian heartland, as well as the creation of modern Balkan and Middle Eastern states, thus ending Turkish colonial ambitions."}} +{"id":"188f874e1a442361","question":"Who were otachi?","gold_answer":"non-Mongol physicians","gold_chunk_ids":["188f874e1a442361"],"is_unanswerable":false,"metadata":{"title":"Yuan_dynasty","context":"The physicians of the Yuan court came from diverse cultures. Healers were divided into non-Mongol physicians called otachi and traditional Mongol shamans. The Mongols characterized otachi doctors by their use of herbal remedies, which was distinguished from the spiritual cures of Mongol shamanism. Physicians received official support from the Yuan government and were given special legal privileges. Kublai created the Imperial Academy of Medicine to manage medical treatises and the education of new doctors. Confucian scholars were attracted to the medical profession because it ensured a high income and medical ethics were compatible with Confucian virtues."}} +{"id":"6c6f34ab4d56e5c8","question":"What army did Warsaw successfully defend itself against?","gold_answer":"the Red Army","gold_chunk_ids":["6c6f34ab4d56e5c8"],"is_unanswerable":false,"metadata":{"title":"Warsaw","context":"Warsaw was occupied by Germany from 4 August 1915 until November 1918. The Allied Armistice terms required in Article 12 that Germany withdraw from areas controlled by Russia in 1914, which included Warsaw. Germany did so, and underground leader Piłsudski returned to Warsaw on 11 November and set up what became the Second Polish Republic, with Warsaw the capital. In the course of the Polish-Bolshevik War of 1920, the huge Battle of Warsaw was fought on the eastern outskirts of the city in which the capital was successfully defended and the Red Army defeated. Poland stopped by itself the full brunt of the Red Army and defeated an idea of the \"export of the revolution\"."}} +{"id":"eb3127019adbacb4","question":"What problem was proposed by Clay Mathematics Institute at the Alpha Prize Problems?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Computational_complexity_theory","context":"The question of whether P equals NP is one of the most important open questions in theoretical computer science because of the wide implications of a solution. If the answer is yes, many important problems can be shown to have more efficient solutions. These include various types of integer programming problems in operations research, many problems in logistics, protein structure prediction in biology, and the ability to find formal proofs of pure mathematics theorems. The P versus NP problem is one of the Millennium Prize Problems proposed by the Clay Mathematics Institute. There is a US$1,000,000 prize for resolving the problem."}} +{"id":"47f5396c304ef173","question":"Between 96,660 and 128,843 vertebrate species live where?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Amazon_rainforest","context":"The region is home to about 2.5 million insect species, tens of thousands of plants, and some 2,000 birds and mammals. To date, at least 40,000 plant species, 2,200 fishes, 1,294 birds, 427 mammals, 428 amphibians, and 378 reptiles have been scientifically classified in the region. One in five of all the bird species in the world live in the rainforests of the Amazon, and one in five of the fish species live in Amazonian rivers and streams. Scientists have described between 96,660 and 128,843 invertebrate species in Brazil alone."}} +{"id":"1548f727057669b2","question":"The European Court of Justice cannot uphold measures that are incompatible with what?","gold_answer":"fundamental rights recognised and protected in the constitutions of member states","gold_chunk_ids":["1548f727057669b2"],"is_unanswerable":false,"metadata":{"title":"European_Union_law","context":"Fundamental rights, as in human rights, were first recognised by the European Court of Justice in the late 60s and fundamental rights are now regarded as integral part of the general principles of European Union law. As such the European Court of Justice is bound to draw inspiration from the constitutional traditions common to the member states. Therefore, the European Court of Justice cannot uphold measures which are incompatible with fundamental rights recognised and protected in the constitutions of member states. The European Court of Justice also found that \"international treaties for the protection of human rights on which the member states have collaborated or of which they are signatories, can supply guidelines which should be followed within the framework of Community law.\""}} +{"id":"b46612c45e8efeb2","question":"What do four of the fraternities form?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"University_of_Chicago","context":"There are fifteen fraternities and seven sororities at the University of Chicago, as well as one co-ed community service fraternity, Alpha Phi Omega. Four of the sororities are members of the National Panhellenic Conference, and ten of the fraternities form the University of Chicago Interfraternity Council. In 2002, the Associate Director of Student Activities estimated that 8–10 percent of undergraduates were members of fraternities or sororities. The student activities office has used similar figures, stating that one in ten undergraduates participate in Greek life."}} +{"id":"e36248749bc33290","question":"Where can complexity classes RPP, BPP, PPP, BQP, MA, and PH be located?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Computational_complexity_theory","context":"Many known complexity classes are suspected to be unequal, but this has not been proved. For instance P ⊆ NP ⊆ PP ⊆ PSPACE, but it is possible that P = PSPACE. If P is not equal to NP, then P is not equal to PSPACE either. Since there are many known complexity classes between P and PSPACE, such as RP, BPP, PP, BQP, MA, PH, etc., it is possible that all these complexity classes collapse to one class. Proving that any of these classes are unequal would be a major breakthrough in complexity theory."}} +{"id":"414a9b0d7a787387","question":"How long was the Leon Theatre in operation?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Warsaw","context":"Nearby, in Ogród Saski (the Saxon Garden), the Summer Theatre was in operation from 1870 to 1939, and in the inter-war period, the theatre complex also included Momus, Warsaw's first literary cabaret, and Leon Schiller's musical theatre Melodram. The Wojciech Bogusławski Theatre (1922–26), was the best example of \"Polish monumental theatre\". From the mid-1930s, the Great Theatre building housed the Upati Institute of Dramatic Arts – the first state-run academy of dramatic art, with an acting department and a stage directing department."}} +{"id":"adeb5ae63d681b5c","question":"What is the longest river in Germany?","gold_answer":"Rhine","gold_chunk_ids":["adeb5ae63d681b5c"],"is_unanswerable":false,"metadata":{"title":"Rhine","context":"The Rhine is the longest river in Germany. It is here that the Rhine encounters some more of its main tributaries, such as the Neckar, the Main and, later, the Moselle, which contributes an average discharge of more than 300 m3/s (11,000 cu ft/s). Northeastern France drains to the Rhine via the Moselle; smaller rivers drain the Vosges and Jura Mountains uplands. Most of Luxembourg and a very small part of Belgium also drain to the Rhine via the Moselle. As it approaches the Dutch border, the Rhine has an annual mean discharge of 2,290 m3/s (81,000 cu ft/s) and an average width of 400 m (1,300 ft)."}} +{"id":"4447dd9fb109721f","question":"Some species of beroe have a pair of strips of adhesive cells on the stomach wall. What does it do?","gold_answer":"zip\" the mouth shut when the animal is not feeding,","gold_chunk_ids":["4447dd9fb109721f"],"is_unanswerable":false,"metadata":{"title":"Ctenophora","context":"The Beroida, also known as Nuda, have no feeding appendages, but their large pharynx, just inside the large mouth and filling most of the saclike body, bears \"macrocilia\" at the oral end. These fused bundles of several thousand large cilia are able to \"bite\" off pieces of prey that are too large to swallow whole – almost always other ctenophores. In front of the field of macrocilia, on the mouth \"lips\" in some species of Beroe, is a pair of narrow strips of adhesive epithelial cells on the stomach wall that \"zip\" the mouth shut when the animal is not feeding, by forming intercellular connections with the opposite adhesive strip. This tight closure streamlines the front of the animal when it is pursuing prey."}} +{"id":"2bbf0ce01cfc77be","question":"What could the Supplemental Nutrition Assistance Program purchase?","gold_answer":"essentials","gold_chunk_ids":["2bbf0ce01cfc77be"],"is_unanswerable":false,"metadata":{"title":"Sky_(United_Kingdom)","context":"The Daily Mail newspaper reported in 2012 that the UK government's benefits agency was checking claimants' \"Sky TV bills to establish if a woman in receipt of benefits as a single mother is wrongly claiming to be living alone\" – as, it claimed, subscription to sports channels would betray a man's presence in the household. In December, the UK’s parliament heard a claim that a subscription to BSkyB was ‘often damaging’, along with alcohol, tobacco and gambling. Conservative MP Alec Shelbrooke was proposing the payments of benefits and tax credits on a \"Welfare Cash Card\", in the style of the Supplemental Nutrition Assistance Program, that could be used to buy only \"essentials\"."}} +{"id":"fdb8082730f73f42","question":"When was the prime number theorem proven?","gold_answer":"at the end of the 19th century","gold_chunk_ids":["fdb8082730f73f42"],"is_unanswerable":false,"metadata":{"title":"Prime_number","context":"There are infinitely many primes, as demonstrated by Euclid around 300 BC. There is no known simple formula that separates prime numbers from composite numbers. However, the distribution of primes, that is to say, the statistical behaviour of primes in the large, can be modelled. The first result in that direction is the prime number theorem, proven at the end of the 19th century, which says that the probability that a given, randomly chosen number n is prime is inversely proportional to its number of digits, or to the logarithm of n."}} +{"id":"c3a342b070f9d9ae","question":"NSF was a major milestone for what?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Packet_switching","context":"The Computer Science Network (CSNET) was a computer network funded by the U.S. National Science Foundation (NSF) that began operation in 1981. Its purpose was to extend networking benefits, for computer science departments at academic and research institutions that could not be directly connected to ARPANET, due to funding or authorization limitations. It played a significant role in spreading awareness of, and access to, national networking and was a major milestone on the path to development of the global Internet."}} +{"id":"605131a802b3eb52","question":"What is the last source of European Union law?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"European_Union_law","context":"European Union law is a body of treaties and legislation, such as Regulations and Directives, which have direct effect or indirect effect on the laws of European Union member states. The three sources of European Union law are primary law, secondary law and supplementary law. The main sources of primary law are the Treaties establishing the European Union. Secondary sources include regulations and directives which are based on the Treaties. The legislature of the European Union is principally composed of the European Parliament and the Council of the European Union, which under the Treaties may establish secondary law to pursue the objective set out in the Treaties."}} +{"id":"b52db43aa6adfd3c","question":"Where were French North Americans unsettled?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"French_and_Indian_War","context":"The French population numbered about 75,000 and was heavily concentrated along the St. Lawrence River valley, with some also in Acadia (present-day New Brunswick and parts of Nova Scotia, including Île Royale (present-day Cape Breton Island)). Fewer lived in New Orleans, Biloxi, Mississippi, Mobile, Alabama and small settlements in the Illinois Country, hugging the east side of the Mississippi River and its tributaries. French fur traders and trappers traveled throughout the St. Lawrence and Mississippi watersheds, did business with local tribes, and often married Indian women. Traders married daughters of chiefs, creating high-ranking unions."}} +{"id":"68fd927926f2e68b","question":"Who governed the Central Region in the Yuan?","gold_answer":"the Central Secretariat","gold_chunk_ids":["68fd927926f2e68b"],"is_unanswerable":false,"metadata":{"title":"Yuan_dynasty","context":"The Central Region, consisting of present-day Hebei, Shandong, Shanxi, the south-eastern part of present-day Inner Mongolia and the Henan areas to the north of the Yellow River, was considered the most important region of the dynasty and directly governed by the Central Secretariat (or Zhongshu Sheng) at Khanbaliq (modern Beijing); similarly, another top-level administrative department called the Bureau of Buddhist and Tibetan Affairs (or Xuanzheng Yuan) held administrative rule over the whole of modern-day Tibet and a part of Sichuan, Qinghai and Kashmir."}} +{"id":"b5af52b84ba371dd","question":"When did Great Britain claim Australia? ","gold_answer":"1788","gold_chunk_ids":["b5af52b84ba371dd"],"is_unanswerable":false,"metadata":{"title":"Victoria_(Australia)","context":"Prior to European settlement, the area now constituting Victoria was inhabited by a large number of Aboriginal peoples, collectively known as the Koori. With Great Britain having claimed the entire Australian continent east of the 135th meridian east in 1788, Victoria was included in the wider colony of New South Wales. The first settlement in the area occurred in 1803 at Sullivan Bay, and much of what is now Victoria was included in the Port Phillip District in 1836, an administrative division of New South Wales. Victoria was officially created a separate colony in 1851, and achieved self-government in 1855. The Victorian gold rush in the 1850s and 1860s significantly increased both the population and wealth of the colony, and by the Federation of Australia in 1901, Melbourne had become the largest city and leading financial centre in Australasia. Melbourne also served as capital of Australia until the construction of Canberra in 1927, with the Federal Parliament meeting in Melbourne's Parliament House and all principal offices of the federal government being based in Melbourne."}} +{"id":"6542814daeb31e1d","question":"What skin-related symptom appears from the pneumonic plague?","gold_answer":"purple skin patches","gold_chunk_ids":["6542814daeb31e1d"],"is_unanswerable":false,"metadata":{"title":"Black_Death","context":"Other forms of plague have been implicated by modern scientists. The modern bubonic plague has a mortality rate of 30–75% and symptoms including fever of 38–41 °C (100–106 °F), headaches, painful aching joints, nausea and vomiting, and a general feeling of malaise. Left untreated, of those that contract the bubonic plague, 80 percent die within eight days. Pneumonic plague has a mortality rate of 90 to 95 percent. Symptoms include fever, cough, and blood-tinged sputum. As the disease progresses, sputum becomes free flowing and bright red. Septicemic plague is the least common of the three forms, with a mortality rate near 100%. Symptoms are high fevers and purple skin patches (purpura due to disseminated intravascular coagulation). In cases of pneumonic and particularly septicemic plague, the progress of the disease is so rapid that there would often be no time for the development of the enlarged lymph nodes that were noted as buboes."}} +{"id":"ec41e7cb64ed2d97","question":"What does the Nederrijn change it's name to?","gold_answer":"Lek","gold_chunk_ids":["ec41e7cb64ed2d97"],"is_unanswerable":false,"metadata":{"title":"Rhine","context":"The other third of the water flows through the Pannerdens Kanaal and redistributes in the IJssel and Nederrijn. The IJssel branch carries one ninth of the water flow of the Rhine north into the IJsselmeer (a former bay), while the Nederrijn carries approximately two ninths of the flow west along a route parallel to the Waal. However, at Wijk bij Duurstede, the Nederrijn changes its name and becomes the Lek. It flows farther west, to rejoin the Noord River into the Nieuwe Maas and to the North Sea."}} +{"id":"419b9831e76ef92c","question":"What is income inequality attributed to?","gold_answer":"differences in value added by labor, capital and land","gold_chunk_ids":["419b9831e76ef92c"],"is_unanswerable":false,"metadata":{"title":"Economic_inequality","context":"Neoclassical economics views inequalities in the distribution of income as arising from differences in value added by labor, capital and land. Within labor income distribution is due to differences in value added by different classifications of workers. In this perspective, wages and profits are determined by the marginal value added of each economic actor (worker, capitalist/business owner, landlord). Thus, in a market economy, inequality is a reflection of the productivity gap between highly-paid professions and lower-paid professions."}} +{"id":"52f05fc640ef7800","question":"What is one issue that takes away the complexity of a pharmacist's job?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Pharmacy","context":"Because of the complexity of medications including specific indications, effectiveness of treatment regimens, safety of medications (i.e., drug interactions) and patient compliance issues (in the hospital and at home) many pharmacists practicing in hospitals gain more education and training after pharmacy school through a pharmacy practice residency and sometimes followed by another residency in a specific area. Those pharmacists are often referred to as clinical pharmacists and they often specialize in various disciplines of pharmacy. For example, there are pharmacists who specialize in hematology/oncology, HIV/AIDS, infectious disease, critical care, emergency medicine, toxicology, nuclear pharmacy, pain management, psychiatry, anti-coagulation clinics, herbal medicine, neurology/epilepsy management, pediatrics, neonatal pharmacists and more."}} +{"id":"e5b10010679938c5","question":"Where is Biraben from?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Black_Death","context":"The plague repeatedly returned to haunt Europe and the Mediterranean throughout the 14th to 17th centuries. According to Biraben, the plague was present somewhere in Europe in every year between 1346 and 1671. The Second Pandemic was particularly widespread in the following years: 1360–63; 1374; 1400; 1438–39; 1456–57; 1464–66; 1481–85; 1500–03; 1518–31; 1544–48; 1563–66; 1573–88; 1596–99; 1602–11; 1623–40; 1644–54; and 1664–67. Subsequent outbreaks, though severe, marked the retreat from most of Europe (18th century) and northern Africa (19th century). According to Geoffrey Parker, \"France alone lost almost a million people to the plague in the epidemic of 1628–31.\""}} +{"id":"fdd3fcb616a8803a","question":"If the tops of the rock units within the folds remain pointing upwards, they are called what? ","gold_answer":"anticlines and synclines","gold_chunk_ids":["fdd3fcb616a8803a"],"is_unanswerable":false,"metadata":{"title":"Geology","context":"When rock units are placed under horizontal compression, they shorten and become thicker. Because rock units, other than muds, do not significantly change in volume, this is accomplished in two primary ways: through faulting and folding. In the shallow crust, where brittle deformation can occur, thrust faults form, which cause deeper rock to move on top of shallower rock. Because deeper rock is often older, as noted by the principle of superposition, this can result in older rocks moving on top of younger ones. Movement along faults can result in folding, either because the faults are not planar or because rock layers are dragged along, forming drag folds as slip occurs along the fault. Deeper in the Earth, rocks behave plastically, and fold instead of faulting. These folds can either be those where the material in the center of the fold buckles upwards, creating \"antiforms\", or where it buckles downwards, creating \"synforms\". If the tops of the rock units within the folds remain pointing upwards, they are called anticlines and synclines, respectively. If some of the units in the fold are facing downward, the structure is called an overturned anticline or syncline, and if all of the rock units are overturned or the correct up-direction is unknown, they are simply called by the most general terms, antiforms and synforms."}} +{"id":"2dd275fab44d976e","question":"During what period was southern Europe discovered?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Rhine","context":"In southern Europe, the stage was set in the Triassic Period of the Mesozoic Era, with the opening of the Tethys Ocean, between the Eurasian and African tectonic plates, between about 240 MBP and 220 MBP (million years before present). The present Mediterranean Sea descends from this somewhat larger Tethys sea. At about 180 MBP, in the Jurassic Period, the two plates reversed direction and began to compress the Tethys floor, causing it to be subducted under Eurasia and pushing up the edge of the latter plate in the Alpine Orogeny of the Oligocene and Miocene Periods. Several microplates were caught in the squeeze and rotated or were pushed laterally, generating the individual features of Mediterranean geography: Iberia pushed up the Pyrenees; Italy, the Alps, and Anatolia, moving west, the mountains of Greece and the islands. The compression and orogeny continue today, as shown by the ongoing raising of the mountains a small amount each year and the active volcanoes."}} +{"id":"4eb6a2b64a0d0f99","question":"How is the dioxygen covalent bond explained?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Oxygen","context":"In this dioxygen, the two oxygen atoms are chemically bonded to each other. The bond can be variously described based on level of theory, but is reasonably and simply described as a covalent double bond that results from the filling of molecular orbitals formed from the atomic orbitals of the individual oxygen atoms, the filling of which results in a bond order of two. More specifically, the double bond is the result of sequential, low-to-high energy, or Aufbau, filling of orbitals, and the resulting cancellation of contributions from the 2s electrons, after sequential filling of the low σ and σ* orbitals; σ overlap of the two atomic 2p orbitals that lie along the O-O molecular axis and π overlap of two pairs of atomic 2p orbitals perpendicular to the O-O molecular axis, and then cancellation of contributions from the remaining two of the six 2p electrons after their partial filling of the lowest π and π* orbitals."}} +{"id":"3c231dfa769b9efd","question":"What complexity class is commonly characterized by unknown algorithms to enhance solvability?","gold_answer":"NP","gold_chunk_ids":["3c231dfa769b9efd"],"is_unanswerable":false,"metadata":{"title":"Computational_complexity_theory","context":"The complexity class P is often seen as a mathematical abstraction modeling those computational tasks that admit an efficient algorithm. This hypothesis is called the Cobham–Edmonds thesis. The complexity class NP, on the other hand, contains many problems that people would like to solve efficiently, but for which no efficient algorithm is known, such as the Boolean satisfiability problem, the Hamiltonian path problem and the vertex cover problem. Since deterministic Turing machines are special non-deterministic Turing machines, it is easily observed that each problem in P is also member of the class NP."}} +{"id":"8dedc4281256b5f1","question":"What was sponsored by Francis Heisler in August 1957?","gold_answer":null,"gold_chunk_ids":[],"is_unanswerable":true,"metadata":{"title":"Civil_disobedience","context":"When the Committee for Non-Violent Action sponsored a protest in August 1957, at the Camp Mercury nuclear test site near Las Vegas, Nevada, 13 of the protesters attempted to enter the test site knowing that they faced arrest. At a pre-arranged announced time, one at a time they stepped across the \"line\" and were immediately arrested. They were put on a bus and taken to the Nye County seat of Tonopah, Nevada, and arraigned for trial before the local Justice of the Peace, that afternoon. A well known civil rights attorney, Francis Heisler, had volunteered to defend the arrested persons, advising them to plead \"nolo contendere\", as an alternative to pleading either guilty or not-guilty. The arrested persons were found \"guilty,\" nevertheless, and given suspended sentences, conditional on their not reentering the test site grounds.[citation needed]"}} diff --git a/eval_data/squad_v2_dev_200/seed.txt b/eval_data/squad_v2_dev_200/seed.txt new file mode 100644 index 00000000..bd41cba7 --- /dev/null +++ b/eval_data/squad_v2_dev_200/seed.txt @@ -0,0 +1 @@ +12345 \ No newline at end of file diff --git a/frontend/package-lock.json b/frontend/package-lock.json index 92af5cbe..f7cf0903 100644 --- a/frontend/package-lock.json +++ b/frontend/package-lock.json @@ -22,6 +22,7 @@ "react-markdown": "^10.1.0", "react-router": "^7.14.1", "react-syntax-highlighter": "^16.1.1", + "recharts": "^3.8.1", "remark-gfm": "^4.0.1", "shadcn": "^4.2.0", "sonner": "^2.0.7", @@ -1711,6 +1712,42 @@ "integrity": "sha512-U69T3ItWHvLwGg5eJ0n3I62nWuE6ilHlmz7zM0npLBRvPRd7e6NYmg54vvRtP5mZG7kZqZCFVdsTWo7BPtBujg==", "license": "MIT" }, + "node_modules/@reduxjs/toolkit": { + "version": "2.11.2", + "resolved": "https://registry.npmjs.org/@reduxjs/toolkit/-/toolkit-2.11.2.tgz", + "integrity": "sha512-Kd6kAHTA6/nUpp8mySPqj3en3dm0tdMIgbttnQ1xFMVpufoj+ADi8pXLBsd4xzTRHQa7t/Jv8W5UnCuW4kuWMQ==", + "license": "MIT", + "dependencies": { + "@standard-schema/spec": "^1.0.0", + "@standard-schema/utils": "^0.3.0", + "immer": "^11.0.0", + "redux": "^5.0.1", + "redux-thunk": "^3.1.0", + "reselect": "^5.1.0" + }, + "peerDependencies": { + "react": "^16.9.0 || ^17.0.0 || ^18 || ^19", + "react-redux": "^7.2.1 || ^8.1.3 || ^9.0.0" + }, + "peerDependenciesMeta": { + "react": { + "optional": true + }, + "react-redux": { + "optional": true + } + } + }, + "node_modules/@reduxjs/toolkit/node_modules/immer": { + "version": "11.1.4", + "resolved": "https://registry.npmjs.org/immer/-/immer-11.1.4.tgz", + "integrity": "sha512-XREFCPo6ksxVzP4E0ekD5aMdf8WMwmdNaz6vuvxgI40UaEiu6q3p8X52aU6GdyvLY3XXX/8R7JOTXStz/nBbRw==", + "license": "MIT", + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/immer" + } + }, "node_modules/@rolldown/pluginutils": { "version": "1.0.0-rc.3", "resolved": "https://registry.npmjs.org/@rolldown/pluginutils/-/pluginutils-1.0.0-rc.3.tgz", @@ -2061,6 +2098,18 @@ "url": "https://github.com/sponsors/sindresorhus" } }, + "node_modules/@standard-schema/spec": { + "version": "1.1.0", + "resolved": "https://registry.npmjs.org/@standard-schema/spec/-/spec-1.1.0.tgz", + "integrity": "sha512-l2aFy5jALhniG5HgqrD6jXLi/rUWrKvqN/qJx6yoJsgKhblVd+iqqU4RCXavm/jPityDo5TCvKMnpjKnOriy0w==", + "license": "MIT" + }, + "node_modules/@standard-schema/utils": { + "version": "0.3.0", + "resolved": "https://registry.npmjs.org/@standard-schema/utils/-/utils-0.3.0.tgz", + "integrity": "sha512-e7Mew686owMaPJVNNLs55PUvgz371nKgwsc4vxE49zsODpJEnxgxRo2y/OKrqueavXgZNMDVj3DdHFlaSAeU8g==", + "license": "MIT" + }, "node_modules/@tailwindcss/node": { "version": "4.2.2", "resolved": "https://registry.npmjs.org/@tailwindcss/node/-/node-4.2.2.tgz", @@ -2436,6 +2485,69 @@ "@babel/types": "^7.28.2" } }, + "node_modules/@types/d3-array": { + "version": "3.2.2", + "resolved": "https://registry.npmjs.org/@types/d3-array/-/d3-array-3.2.2.tgz", + "integrity": "sha512-hOLWVbm7uRza0BYXpIIW5pxfrKe0W+D5lrFiAEYR+pb6w3N2SwSMaJbXdUfSEv+dT4MfHBLtn5js0LAWaO6otw==", + "license": "MIT" + }, + "node_modules/@types/d3-color": { + "version": "3.1.3", + "resolved": "https://registry.npmjs.org/@types/d3-color/-/d3-color-3.1.3.tgz", + "integrity": "sha512-iO90scth9WAbmgv7ogoq57O9YpKmFBbmoEoCHDB2xMBY0+/KVrqAaCDyCE16dUspeOvIxFFRI+0sEtqDqy2b4A==", + "license": "MIT" + }, + "node_modules/@types/d3-ease": { + "version": "3.0.2", + "resolved": "https://registry.npmjs.org/@types/d3-ease/-/d3-ease-3.0.2.tgz", + "integrity": "sha512-NcV1JjO5oDzoK26oMzbILE6HW7uVXOHLQvHshBUW4UMdZGfiY6v5BeQwh9a9tCzv+CeefZQHJt5SRgK154RtiA==", + "license": "MIT" + }, + "node_modules/@types/d3-interpolate": { + "version": "3.0.4", + "resolved": "https://registry.npmjs.org/@types/d3-interpolate/-/d3-interpolate-3.0.4.tgz", + "integrity": "sha512-mgLPETlrpVV1YRJIglr4Ez47g7Yxjl1lj7YKsiMCb27VJH9W8NVM6Bb9d8kkpG/uAQS5AmbA48q2IAolKKo1MA==", + "license": "MIT", + "dependencies": { + "@types/d3-color": "*" + } + }, + "node_modules/@types/d3-path": { + "version": "3.1.1", + "resolved": "https://registry.npmjs.org/@types/d3-path/-/d3-path-3.1.1.tgz", + "integrity": "sha512-VMZBYyQvbGmWyWVea0EHs/BwLgxc+MKi1zLDCONksozI4YJMcTt8ZEuIR4Sb1MMTE8MMW49v0IwI5+b7RmfWlg==", + "license": "MIT" + }, + "node_modules/@types/d3-scale": { + "version": "4.0.9", + "resolved": "https://registry.npmjs.org/@types/d3-scale/-/d3-scale-4.0.9.tgz", + "integrity": "sha512-dLmtwB8zkAeO/juAMfnV+sItKjlsw2lKdZVVy6LRr0cBmegxSABiLEpGVmSJJ8O08i4+sGR6qQtb6WtuwJdvVw==", + "license": "MIT", + "dependencies": { + "@types/d3-time": "*" + } + }, + "node_modules/@types/d3-shape": { + "version": "3.1.8", + "resolved": "https://registry.npmjs.org/@types/d3-shape/-/d3-shape-3.1.8.tgz", + "integrity": "sha512-lae0iWfcDeR7qt7rA88BNiqdvPS5pFVPpo5OfjElwNaT2yyekbM0C9vK+yqBqEmHr6lDkRnYNoTBYlAgJa7a4w==", + "license": "MIT", + "dependencies": { + "@types/d3-path": "*" + } + }, + "node_modules/@types/d3-time": { + "version": "3.0.4", + "resolved": "https://registry.npmjs.org/@types/d3-time/-/d3-time-3.0.4.tgz", + "integrity": "sha512-yuzZug1nkAAaBlBBikKZTgzCeA+k1uy4ZFwWANOfKw5z5LRhV0gNA7gNkKm7HoK+HRN0wX3EkxGk0fpbWhmB7g==", + "license": "MIT" + }, + "node_modules/@types/d3-timer": { + "version": "3.0.2", + "resolved": "https://registry.npmjs.org/@types/d3-timer/-/d3-timer-3.0.2.tgz", + "integrity": "sha512-Ps3T8E8dZDam6fUyNiMkekK3XUsaUEik+idO9/YjPtfj2qruF8tFBXS7XhtE4iIXBLxhmLjP3SXpLhVf21I9Lw==", + "license": "MIT" + }, "node_modules/@types/debug": { "version": "4.1.13", "resolved": "https://registry.npmjs.org/@types/debug/-/debug-4.1.13.tgz", @@ -2549,6 +2661,12 @@ "integrity": "sha512-ko/gIFJRv177XgZsZcBwnqJN5x/Gien8qNOn0D5bQU/zAzVf9Zt3BlcUiLqhV9y4ARk0GbT3tnUiPNgnTXzc/Q==", "license": "MIT" }, + "node_modules/@types/use-sync-external-store": { + "version": "0.0.6", + "resolved": "https://registry.npmjs.org/@types/use-sync-external-store/-/use-sync-external-store-0.0.6.tgz", + "integrity": "sha512-zFDAD+tlpf2r4asuHEj0XH6pY6i0g5NeAHPn+15wk3BV6JA69eERFXC1gyGThDkVa1zCyKr5jox1+2LbV/AMLg==", + "license": "MIT" + }, "node_modules/@types/validate-npm-package-name": { "version": "4.0.2", "resolved": "https://registry.npmjs.org/@types/validate-npm-package-name/-/validate-npm-package-name-4.0.2.tgz", @@ -3584,6 +3702,127 @@ "integrity": "sha512-z1HGKcYy2xA8AGQfwrn0PAy+PB7X/GSj3UVJW9qKyn43xWa+gl5nXmU4qqLMRzWVLFC8KusUX8T/0kCiOYpAIQ==", "license": "MIT" }, + "node_modules/d3-array": { + "version": "3.2.4", + "resolved": "https://registry.npmjs.org/d3-array/-/d3-array-3.2.4.tgz", + "integrity": "sha512-tdQAmyA18i4J7wprpYq8ClcxZy3SC31QMeByyCFyRt7BVHdREQZ5lpzoe5mFEYZUWe+oq8HBvk9JjpibyEV4Jg==", + "license": "ISC", + "dependencies": { + "internmap": "1 - 2" + }, + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-color": { + "version": "3.1.0", + "resolved": "https://registry.npmjs.org/d3-color/-/d3-color-3.1.0.tgz", + "integrity": "sha512-zg/chbXyeBtMQ1LbD/WSoW2DpC3I0mpmPdW+ynRTj/x2DAWYrIY7qeZIHidozwV24m4iavr15lNwIwLxRmOxhA==", + "license": "ISC", + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-ease": { + "version": "3.0.1", + "resolved": "https://registry.npmjs.org/d3-ease/-/d3-ease-3.0.1.tgz", + "integrity": "sha512-wR/XK3D3XcLIZwpbvQwQ5fK+8Ykds1ip7A2Txe0yxncXSdq1L9skcG7blcedkOX+ZcgxGAmLX1FrRGbADwzi0w==", + "license": "BSD-3-Clause", + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-format": { + "version": "3.1.2", + "resolved": "https://registry.npmjs.org/d3-format/-/d3-format-3.1.2.tgz", + "integrity": "sha512-AJDdYOdnyRDV5b6ArilzCPPwc1ejkHcoyFarqlPqT7zRYjhavcT3uSrqcMvsgh2CgoPbK3RCwyHaVyxYcP2Arg==", + "license": "ISC", + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-interpolate": { + "version": "3.0.1", + "resolved": "https://registry.npmjs.org/d3-interpolate/-/d3-interpolate-3.0.1.tgz", + "integrity": "sha512-3bYs1rOD33uo8aqJfKP3JWPAibgw8Zm2+L9vBKEHJ2Rg+viTR7o5Mmv5mZcieN+FRYaAOWX5SJATX6k1PWz72g==", + "license": "ISC", + "dependencies": { + "d3-color": "1 - 3" + }, + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-path": { + "version": "3.1.0", + "resolved": "https://registry.npmjs.org/d3-path/-/d3-path-3.1.0.tgz", + "integrity": "sha512-p3KP5HCf/bvjBSSKuXid6Zqijx7wIfNW+J/maPs+iwR35at5JCbLUT0LzF1cnjbCHWhqzQTIN2Jpe8pRebIEFQ==", + "license": "ISC", + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-scale": { + "version": "4.0.2", + "resolved": "https://registry.npmjs.org/d3-scale/-/d3-scale-4.0.2.tgz", + "integrity": "sha512-GZW464g1SH7ag3Y7hXjf8RoUuAFIqklOAq3MRl4OaWabTFJY9PN/E1YklhXLh+OQ3fM9yS2nOkCoS+WLZ6kvxQ==", + "license": "ISC", + "dependencies": { + "d3-array": "2.10.0 - 3", + "d3-format": "1 - 3", + "d3-interpolate": "1.2.0 - 3", + "d3-time": "2.1.1 - 3", + "d3-time-format": "2 - 4" + }, + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-shape": { + "version": "3.2.0", + "resolved": "https://registry.npmjs.org/d3-shape/-/d3-shape-3.2.0.tgz", + "integrity": "sha512-SaLBuwGm3MOViRq2ABk3eLoxwZELpH6zhl3FbAoJ7Vm1gofKx6El1Ib5z23NUEhF9AsGl7y+dzLe5Cw2AArGTA==", + "license": "ISC", + "dependencies": { + "d3-path": "^3.1.0" + }, + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-time": { + "version": "3.1.0", + "resolved": "https://registry.npmjs.org/d3-time/-/d3-time-3.1.0.tgz", + "integrity": "sha512-VqKjzBLejbSMT4IgbmVgDjpkYrNWUYJnbCGo874u7MMKIWsILRX+OpX/gTk8MqjpT1A/c6HY2dCA77ZN0lkQ2Q==", + "license": "ISC", + "dependencies": { + "d3-array": "2 - 3" + }, + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-time-format": { + "version": "4.1.0", + "resolved": "https://registry.npmjs.org/d3-time-format/-/d3-time-format-4.1.0.tgz", + "integrity": "sha512-dJxPBlzC7NugB2PDLwo9Q8JiTR3M3e4/XANkreKSUxF8vvXKqm1Yfq4Q5dl8budlunRVlUUaDUgFt7eA8D6NLg==", + "license": "ISC", + "dependencies": { + "d3-time": "1 - 3" + }, + "engines": { + "node": ">=12" + } + }, + "node_modules/d3-timer": { + "version": "3.0.1", + "resolved": "https://registry.npmjs.org/d3-timer/-/d3-timer-3.0.1.tgz", + "integrity": "sha512-ndfJ/JxxMd3nw31uyKoY2naivF+r29V+Lc0svZxe1JvvIRmi8hUsrMvdOwgS1o6uBHmiz91geQ0ylPP0aj1VUA==", + "license": "ISC", + "engines": { + "node": ">=12" + } + }, "node_modules/data-uri-to-buffer": { "version": "4.0.1", "resolved": "https://registry.npmjs.org/data-uri-to-buffer/-/data-uri-to-buffer-4.0.1.tgz", @@ -3621,6 +3860,12 @@ } } }, + "node_modules/decimal.js-light": { + "version": "2.5.1", + "resolved": "https://registry.npmjs.org/decimal.js-light/-/decimal.js-light-2.5.1.tgz", + "integrity": "sha512-qIMFpTMZmny+MMIitAB6D7iVPEorVw6YQRWkvarTkT4tBeSLLiHzcwj6q0MmYSFCiVpiqPJTJEYIrpcPzVEIvg==", + "license": "MIT" + }, "node_modules/decode-named-character-reference": { "version": "1.3.0", "resolved": "https://registry.npmjs.org/decode-named-character-reference/-/decode-named-character-reference-1.3.0.tgz", @@ -3884,6 +4129,16 @@ "node": ">= 0.4" } }, + "node_modules/es-toolkit": { + "version": "1.46.0", + "resolved": "https://registry.npmjs.org/es-toolkit/-/es-toolkit-1.46.0.tgz", + "integrity": "sha512-IToJ6ct9OLl5zz6WsC/1vZEwfSZ7Myil+ygl5Tf30Xjn9AEkzNB4kqp2G7VUJKF1DtTx/ra5M5KLlXvzOg51BA==", + "license": "MIT", + "workspaces": [ + "docs", + "benchmarks" + ] + }, "node_modules/esbuild": { "version": "0.27.7", "resolved": "https://registry.npmjs.org/esbuild/-/esbuild-0.27.7.tgz", @@ -4170,6 +4425,12 @@ "node": ">= 0.6" } }, + "node_modules/eventemitter3": { + "version": "5.0.4", + "resolved": "https://registry.npmjs.org/eventemitter3/-/eventemitter3-5.0.4.tgz", + "integrity": "sha512-mlsTRyGaPBjPedk6Bvw+aqbsXDtoAyAzm5MO7JgU+yVRyMQ5O8bD4Kcci7BS85f93veegeCPkL8R4GLClnjLFw==", + "license": "MIT" + }, "node_modules/eventsource": { "version": "3.0.7", "resolved": "https://registry.npmjs.org/eventsource/-/eventsource-3.0.7.tgz", @@ -4968,6 +5229,16 @@ "node": ">= 4" } }, + "node_modules/immer": { + "version": "10.2.0", + "resolved": "https://registry.npmjs.org/immer/-/immer-10.2.0.tgz", + "integrity": "sha512-d/+XTN3zfODyjr89gM3mPq1WNX2B8pYsu7eORitdwyA2sBubnTl3laYlBk4sXY5FUa5qTZGBDPJICVbvqzjlbw==", + "license": "MIT", + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/immer" + } + }, "node_modules/import-fresh": { "version": "3.3.1", "resolved": "https://registry.npmjs.org/import-fresh/-/import-fresh-3.3.1.tgz", @@ -5006,6 +5277,15 @@ "integrity": "sha512-Nb2ctOyNR8DqQoR0OwRG95uNWIC0C1lCgf5Naz5H6Ji72KZ8OcFZLz2P5sNgwlyoJ8Yif11oMuYs5pBQa86csA==", "license": "MIT" }, + "node_modules/internmap": { + "version": "2.0.3", + "resolved": "https://registry.npmjs.org/internmap/-/internmap-2.0.3.tgz", + "integrity": "sha512-5Hh7Y1wQbvY5ooGgPbDaL5iYLAPzMTUrjMulskHLH6wnv/A+1q5rgEaiuqEjB+oxGXIVZs1FF+R/KPN3ZSQYYg==", + "license": "ISC", + "engines": { + "node": ">=12" + } + }, "node_modules/ip-address": { "version": "10.1.0", "resolved": "https://registry.npmjs.org/ip-address/-/ip-address-10.1.0.tgz", @@ -7582,6 +7862,13 @@ "react": "^19.2.5" } }, + "node_modules/react-is": { + "version": "19.2.5", + "resolved": "https://registry.npmjs.org/react-is/-/react-is-19.2.5.tgz", + "integrity": "sha512-Dn0t8IQhCmeIT3wu+Apm1/YVsJXsGWi6k4sPdnBIdqMVtHtv0IGi6dcpNpNkNac0zB2uUAqNX3MHzN8c+z2rwQ==", + "license": "MIT", + "peer": true + }, "node_modules/react-markdown": { "version": "10.1.0", "resolved": "https://registry.npmjs.org/react-markdown/-/react-markdown-10.1.0.tgz", @@ -7609,6 +7896,30 @@ "react": ">=18" } }, + "node_modules/react-redux": { + "version": "9.2.0", + "resolved": "https://registry.npmjs.org/react-redux/-/react-redux-9.2.0.tgz", + "integrity": "sha512-ROY9fvHhwOD9ySfrF0wmvu//bKCQ6AeZZq1nJNtbDC+kk5DuSuNX/n6YWYF/SYy7bSba4D4FSz8DJeKY/S/r+g==", + "license": "MIT", + "peer": true, + "dependencies": { + "@types/use-sync-external-store": "^0.0.6", + "use-sync-external-store": "^1.4.0" + }, + "peerDependencies": { + "@types/react": "^18.2.25 || ^19", + "react": "^18.0 || ^19", + "redux": "^5.0.0" + }, + "peerDependenciesMeta": { + "@types/react": { + "optional": true + }, + "redux": { + "optional": true + } + } + }, "node_modules/react-refresh": { "version": "0.18.0", "resolved": "https://registry.npmjs.org/react-refresh/-/react-refresh-0.18.0.tgz", @@ -7690,6 +8001,52 @@ "node": ">= 4" } }, + "node_modules/recharts": { + "version": "3.8.1", + "resolved": "https://registry.npmjs.org/recharts/-/recharts-3.8.1.tgz", + "integrity": "sha512-mwzmO1s9sFL0TduUpwndxCUNoXsBw3u3E/0+A+cLcrSfQitSG62L32N69GhqUrrT5qKcAE3pCGVINC6pqkBBQg==", + "license": "MIT", + "workspaces": [ + "www" + ], + "dependencies": { + "@reduxjs/toolkit": "^1.9.0 || 2.x.x", + "clsx": "^2.1.1", + "decimal.js-light": "^2.5.1", + "es-toolkit": "^1.39.3", + "eventemitter3": "^5.0.1", + "immer": "^10.1.1", + "react-redux": "8.x.x || 9.x.x", + "reselect": "5.1.1", + "tiny-invariant": "^1.3.3", + "use-sync-external-store": "^1.2.2", + "victory-vendor": "^37.0.2" + }, + "engines": { + "node": ">=18" + }, + "peerDependencies": { + "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0", + "react-dom": "^16.0.0 || ^17.0.0 || ^18.0.0 || ^19.0.0", + "react-is": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0" + } + }, + "node_modules/redux": { + "version": "5.0.1", + "resolved": "https://registry.npmjs.org/redux/-/redux-5.0.1.tgz", + "integrity": "sha512-M9/ELqF6fy8FwmkpnF0S3YKOqMyoWJ4+CS5Efg2ct3oY9daQvd/Pc71FpGZsVsbl3Cpb+IIcjBDUnnyBdQbq4w==", + "license": "MIT", + "peer": true + }, + "node_modules/redux-thunk": { + "version": "3.1.0", + "resolved": "https://registry.npmjs.org/redux-thunk/-/redux-thunk-3.1.0.tgz", + "integrity": "sha512-NW2r5T6ksUKXCabzhL9z+h206HQw/NJkcLm1GPImRQ8IzfXwRGqjVhKJGauHirT0DAuyy6hjdnMZaRoAcy0Klw==", + "license": "MIT", + "peerDependencies": { + "redux": "^5.0.0" + } + }, "node_modules/refractor": { "version": "5.0.0", "resolved": "https://registry.npmjs.org/refractor/-/refractor-5.0.0.tgz", @@ -8879,6 +9236,28 @@ "url": "https://opencollective.com/unified" } }, + "node_modules/victory-vendor": { + "version": "37.3.6", + "resolved": "https://registry.npmjs.org/victory-vendor/-/victory-vendor-37.3.6.tgz", + "integrity": "sha512-SbPDPdDBYp+5MJHhBCAyI7wKM3d5ivekigc2Dk2s7pgbZ9wIgIBYGVw4zGHBml/qTFbexrofXW6Gu4noGxrOwQ==", + "license": "MIT AND ISC", + "dependencies": { + "@types/d3-array": "^3.0.3", + "@types/d3-ease": "^3.0.0", + "@types/d3-interpolate": "^3.0.1", + "@types/d3-scale": "^4.0.2", + "@types/d3-shape": "^3.1.0", + "@types/d3-time": "^3.0.0", + "@types/d3-timer": "^3.0.0", + "d3-array": "^3.1.6", + "d3-ease": "^3.0.1", + "d3-interpolate": "^3.0.1", + "d3-scale": "^4.0.2", + "d3-shape": "^3.1.0", + "d3-time": "^3.0.0", + "d3-timer": "^3.0.1" + } + }, "node_modules/vite": { "version": "7.3.2", "resolved": "https://registry.npmjs.org/vite/-/vite-7.3.2.tgz", diff --git a/frontend/package.json b/frontend/package.json index f91a3f5c..03a7de49 100644 --- a/frontend/package.json +++ b/frontend/package.json @@ -26,6 +26,7 @@ "react-markdown": "^10.1.0", "react-router": "^7.14.1", "react-syntax-highlighter": "^16.1.1", + "recharts": "^3.8.1", "remark-gfm": "^4.0.1", "shadcn": "^4.2.0", "sonner": "^2.0.7", diff --git a/frontend/src/App.tsx b/frontend/src/App.tsx index 842f603a..a2982c74 100644 --- a/frontend/src/App.tsx +++ b/frontend/src/App.tsx @@ -7,6 +7,7 @@ import ChatPage from "@/pages/chat"; import UploadPage from "@/pages/upload"; import DocumentsPage from "@/pages/documents"; import SharedPage from "@/pages/shared"; +import { EvalPage } from "@/pages/eval-page"; const queryClient = new QueryClient({ defaultOptions: { @@ -22,6 +23,7 @@ const router = createBrowserRouter([ { path: "chat/:conversationId?", Component: ChatPage }, { path: "upload", Component: UploadPage }, { path: "documents", Component: DocumentsPage }, + { path: "eval/*", element: }, ], }, { path: "shared/:token", Component: SharedPage }, diff --git a/frontend/src/api/eval.ts b/frontend/src/api/eval.ts new file mode 100644 index 00000000..3bd0a78e --- /dev/null +++ b/frontend/src/api/eval.ts @@ -0,0 +1,261 @@ +/** + * Eval API client + TanStack Query hooks. + * + * API Layer Position: + * React components → [hooks] → fetch → /api/eval/* + * + * Design notes: + * - Field names match the backend's snake_case JSON wire format, consistent + * with the convention in types.ts (e.g. doc_id, chunk_id, created_at). + * - Uses the same `request` helper pattern as client.ts — a thin wrapper + * around fetch that throws on non-2xx with the backend's `detail` message. + * - useRunStatus polls every 1s by default; pass `refetchInterval: false` + * once a run hits "completed" / "failed" to stop polling. + * - Submitting a run invalidates the runs list so the new run shows + * up immediately. + */ + +import { useMutation, useQuery, useQueryClient } from "@tanstack/react-query"; + +// --------------------------------------------------------------------------- +// Types — mirror the backend DTOs from src/api/schemas/eval.py and +// src/eval/schemas.py. Field names stay snake_case to match JSON wire format, +// consistent with how types.ts handles doc_id, chunk_id, created_at, etc. +// --------------------------------------------------------------------------- + +export interface RunMetadata { + run_id: string; + config_name: string; + config_path: string; + git_sha: string; + started_at: string; // ISO timestamp + finished_at: string; + env_hash: string; + eval_set_versions: Record; + n_questions: number; + n_errors: number; + warnings: string[]; +} + +export interface RunSummary { + run_id: string; + config_name: string; + started_at: string; + finished_at: string; + n_questions: number; + n_errors: number; + headline_metric: number | null; +} + +export interface AggregatedMetric { + metric_name: string; + dataset: string | null; + mean: number; + ci_low: number; + ci_high: number; + n: number; +} + +export interface RunDetail { + metadata: RunMetadata; + aggregated: AggregatedMetric[]; + cost: Record; + n_results: number; +} + +export interface EvalResultRow { + question_id: string; + dataset: string; + generated_answer: string; + metrics: Record; + error: string | null; +} + +export interface PageResults { + items: EvalResultRow[]; + page: number; + page_size: number; + total: number; +} + +export interface RunStatus { + run_id: string; + status: "queued" | "running" | "completed" | "failed"; + progress: number; + n_completed: number; + n_total: number; + error_message: string | null; +} + +export interface MetricDelta { + metric_name: string; + dataset: string | null; + a_mean: number; + a_ci: [number, number]; + b_mean: number; + b_ci: [number, number]; + delta: number; + p_value: number; + significant: boolean; +} + +export interface CompareResult { + run_a: RunMetadata; + run_b: RunMetadata; + deltas: MetricDelta[]; + per_question_diff: Array>; +} + +// --------------------------------------------------------------------------- +// Fetch helpers — mirror the request helper from client.ts: same base URL +// env var, same error contract (throws Error with backend's detail message). +// WHY: a separate request helper here avoids coupling this module to client.ts +// while staying consistent — easier to read in isolation. +// --------------------------------------------------------------------------- + +const BASE_URL = import.meta.env.VITE_API_URL ?? ""; + +async function request(path: string, init?: RequestInit): Promise { + const res = await fetch(`${BASE_URL}${path}`, { + ...init, + headers: { ...init?.headers }, + }); + if (!res.ok) { + const body = await res.json().catch(() => ({})); + throw new Error(body.detail ?? `Request failed: ${res.status}`); + } + return res.json(); +} + +// --------------------------------------------------------------------------- +// API functions — one per backend endpoint. +// --------------------------------------------------------------------------- + +export async function listConfigs(): Promise { + return request("/api/eval/configs"); +} + +export async function listRuns(): Promise { + return request("/api/eval/runs"); +} + +export async function getRun(runId: string): Promise { + return request(`/api/eval/runs/${encodeURIComponent(runId)}`); +} + +export async function getRunResults( + runId: string, + page = 1, + pageSize = 50, +): Promise { + const qs = new URLSearchParams({ + page: String(page), + page_size: String(pageSize), + }); + return request( + `/api/eval/runs/${encodeURIComponent(runId)}/results?${qs}`, + ); +} + +export async function getRunResult( + runId: string, + questionId: string, +): Promise { + return request( + `/api/eval/runs/${encodeURIComponent(runId)}/results/${encodeURIComponent(questionId)}`, + ); +} + +export async function getRunStatus(runId: string): Promise { + return request( + `/api/eval/runs/${encodeURIComponent(runId)}/status`, + ); +} + +export async function submitRun( + configName: string, +): Promise<{ run_id: string; status: RunStatus["status"] }> { + return request("/api/eval/run", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ config_name: configName }), + }); +} + +export async function compareRuns( + idA: string, + idB: string, +): Promise { + const qs = new URLSearchParams({ a: idA, b: idB }); + return request(`/api/eval/compare?${qs}`); +} + +// --------------------------------------------------------------------------- +// TanStack Query hooks — one per query/mutation, matching the style in +// use-documents.ts and use-conversations.ts (named exports, no default). +// Query keys follow the existing pattern: ["noun"] or ["noun", param, ...]. +// --------------------------------------------------------------------------- + +export function useConfigs() { + return useQuery({ queryKey: ["eval-configs"], queryFn: listConfigs }); +} + +export function useRunsList() { + return useQuery({ queryKey: ["eval-runs"], queryFn: listRuns }); +} + +export function useRun(runId: string | undefined) { + return useQuery({ + queryKey: ["eval-run", runId], + queryFn: () => getRun(runId!), + enabled: !!runId, + }); +} + +export function useRunResults( + runId: string | undefined, + page = 1, + pageSize = 50, +) { + return useQuery({ + queryKey: ["eval-run-results", runId, page, pageSize], + queryFn: () => getRunResults(runId!, page, pageSize), + enabled: !!runId, + }); +} + +// PATTERN: refetchInterval drives live progress updates for in-flight runs. +// The caller should pass `refetchInterval: false` once status reaches +// "completed" or "failed" to stop polling and save network calls. +export function useRunStatus( + runId: string | undefined, + opts?: { refetchInterval?: number | false }, +) { + return useQuery({ + queryKey: ["eval-run-status", runId], + queryFn: () => getRunStatus(runId!), + enabled: !!runId, + refetchInterval: opts?.refetchInterval ?? 1000, + }); +} + +export function useCompareRuns( + idA: string | undefined, + idB: string | undefined, +) { + return useQuery({ + queryKey: ["eval-compare", idA, idB], + queryFn: () => compareRuns(idA!, idB!), + enabled: !!idA && !!idB, + }); +} + +export function useSubmitRun() { + const qc = useQueryClient(); + return useMutation({ + mutationFn: (configName: string) => submitRun(configName), + // WHY: invalidate the runs list so the newly queued run appears + // immediately in the UI without a manual page refresh. + onSuccess: () => qc.invalidateQueries({ queryKey: ["eval-runs"] }), + }); +} diff --git a/frontend/src/api/types.ts b/frontend/src/api/types.ts index c7a5a994..8c96be4d 100644 --- a/frontend/src/api/types.ts +++ b/frontend/src/api/types.ts @@ -76,6 +76,26 @@ export interface WsErrorMessage { content: string; } +/** + * Per-request timing and token-cost summary emitted by the backend after + * the "done" event. Mirrors the backend `StageTelemetry` dataclass. + * + * WHY a dedicated interface: keeps the telemetry shape self-documenting and + * lets the UI consume it without casting or runtime duck-typing. + */ +export interface TelemetryPayload { + retrieve_ms: number; + generate_ms: number; + prompt_tokens: number; + completion_tokens: number; + cost_usd: number; +} + +export interface WsTelemetryMessage { + type: "telemetry"; + content: TelemetryPayload; +} + export interface EvaluationScore { metric: string; score: number; @@ -96,7 +116,8 @@ export type WsMessage = | WsStatusMessage | WsDoneMessage | WsErrorMessage - | WsEvaluationMessage; + | WsEvaluationMessage + | WsTelemetryMessage; export interface ChatMessage { id: string; @@ -115,6 +136,8 @@ export interface ChatMessage { streamDone?: boolean; /** LLM-as-judge evaluation scores (faithfulness, relevancy, precision). */ evaluation?: EvaluationScore[]; + /** Per-request timing and cost from the backend telemetry event. */ + telemetry?: TelemetryPayload; } export interface ConversationSummary { diff --git a/frontend/src/components/chat/chat-message.tsx b/frontend/src/components/chat/chat-message.tsx index 43c68c58..2292a208 100644 --- a/frontend/src/components/chat/chat-message.tsx +++ b/frontend/src/components/chat/chat-message.tsx @@ -5,6 +5,7 @@ import { cn } from "@/lib/utils"; import { ThinkingPanel } from "./thinking-panel"; import { MarkdownRenderer } from "./markdown-renderer"; import { EvaluationBadge } from "./evaluation-badge"; +import { TelemetryFooter } from "./telemetry-footer"; // WHY removed TypingIndicator: The bouncing-dots placeholder used to show // for the brief window between sending a query and the first server event. @@ -134,6 +135,14 @@ export function ChatMessage({ message, onEvaluate }: ChatMessageProps) { )} + {/* TelemetryFooter — muted one-liner: Retrieve · Generate · Tokens · Cost. + Only renders once telemetry arrives (after the "done" WebSocket event) + and only on assistant messages. Purely additive; does not affect layout + of surrounding elements. */} + {!isUser && message.telemetry ? ( + + ) : null} + {/* PATTERN: EvaluationBadge is shown only after streaming completes (streamDone=true) so it doesn't flash up mid-response. The badge triggers the LLM-as-judge evaluation call on first render and diff --git a/frontend/src/components/chat/telemetry-footer.tsx b/frontend/src/components/chat/telemetry-footer.tsx new file mode 100644 index 00000000..5650bc1a --- /dev/null +++ b/frontend/src/components/chat/telemetry-footer.tsx @@ -0,0 +1,192 @@ +/** + * TelemetryFooter — per-message timing and cost summary bar. + * + * RAG Pipeline Position: + * Query → Retrieval → Generation → Response → [TELEMETRY FOOTER] + * ^^^ + * This component sits at the DISPLAY step, after a full assistant turn + * completes. It surfaces backend timing (retrieve_ms, generate_ms), + * token counts (prompt + completion), and cost so the developer can spot + * slow retrievals, verbose generations, or unexpectedly expensive calls + * without leaving the chat window. + * + * WHY a dedicated component: + * The telemetry data arrives as a separate WebSocket event ("telemetry") + * after the "done" event. Keeping its rendering isolated means the parent + * (ChatMessage) never needs to know about formatting logic — it just + * passes `message.telemetry` through. + * + * PATTERN: Tooltip-wrapped trigger — the one-liner summary is always visible; + * a hover tooltip surfaces the full prompt/completion split without + * cluttering the message layout. + */ + +import { Fragment } from "react"; + +import type { TelemetryPayload } from "@/api/types"; +import { + Tooltip, + TooltipContent, + TooltipTrigger, +} from "@/components/ui/tooltip"; +import { cn } from "@/lib/utils"; + +// --------------------------------------------------------------------------- +// Public API +// --------------------------------------------------------------------------- + +export interface TelemetryFooterProps { + telemetry: TelemetryPayload; + className?: string; +} + +// --------------------------------------------------------------------------- +// Formatting helpers +// --------------------------------------------------------------------------- + +/** + * Format retrieval latency as an integer millisecond value. + * Retrieval is fast enough that sub-second display is always the right unit. + */ +function formatRetrieve(ms: number): string { + return `Retrieve ${Math.round(ms)}ms`; +} + +/** + * Format generation latency: seconds for ≥1 s, milliseconds otherwise. + * + * WHY dual-unit: generation times span 100 ms (fast local model) to 30 s + * (slow hosted model). Showing "30,000ms" is less readable than "30.0s". + */ +function formatGenerate(ms: number): string { + if (ms >= 1000) { + return `Generate ${(ms / 1000).toFixed(1)}s`; + } + return `Generate ${Math.round(ms)}ms`; +} + +/** + * Format total token count with locale-aware thousands separators. + * e.g. 4217 → "4,217 tok" + */ +function formatTokens(prompt: number, completion: number): string { + const total = prompt + completion; + return `${total.toLocaleString()} tok`; +} + +/** + * Format cost in USD to 4 decimal places. + * Always renders, even when cost is exactly 0 ($0.0000). + * + * WHY always render zero: hiding a zero cost would make users think the + * telemetry event didn't arrive, rather than that the call was free (e.g. + * when a local Ollama model is used). + */ +function formatCost(cost: number): string { + return `$${cost.toFixed(4)}`; +} + +// --------------------------------------------------------------------------- +// Tooltip content — token breakdown table +// --------------------------------------------------------------------------- + +/** + * TooltipBreakdown — rendered inside the shadcn Tooltip popup. + * + * WHY tabular: the four rows (Prompt, Completion, Total, Cost) need alignment + * to be scannable. A two-column grid achieves this without a element. + */ +function TooltipBreakdown({ telemetry }: { telemetry: TelemetryPayload }) { + const { prompt_tokens, completion_tokens, cost_usd } = telemetry; + const total = prompt_tokens + completion_tokens; + + const rows: [string, string][] = [ + ["Prompt", `${prompt_tokens.toLocaleString()} tokens`], + ["Completion", `${completion_tokens.toLocaleString()} tokens`], + ["Total", `${total.toLocaleString()} tokens`], + ["Cost USD", cost_usd.toFixed(4)], + ]; + + return ( +
+ {rows.map(([label, value]) => ( + // WHY Fragment with key: short-form <>... can't carry a key, and + // React warns about missing keys on array children. + + {label}: + {value} + + ))} +
+ ); +} + +// --------------------------------------------------------------------------- +// TelemetryFooter — public export +// --------------------------------------------------------------------------- + +/** + * TelemetryFooter renders a muted one-liner beneath an assistant message + * showing retrieve latency, generate latency, token count, and cost. + * + * Usage: + * {message.telemetry ? : null} + * + * @param telemetry - Backend timing and token-cost payload from the + * "telemetry" WebSocket event. See TelemetryPayload in api/types.ts. + * @param className - Optional extra Tailwind classes for the container. + */ +export function TelemetryFooter({ + telemetry, + className, +}: TelemetryFooterProps): React.JSX.Element { + const segments = [ + formatRetrieve(telemetry.retrieve_ms), + formatGenerate(telemetry.generate_ms), + formatTokens(telemetry.prompt_tokens, telemetry.completion_tokens), + formatCost(telemetry.cost_usd), + ]; + + // Dot separator — aria-hidden so screen readers don't announce bullet chars. + const Dot = () => ( + + ); + + return ( + // PATTERN: TooltipProvider is already mounted at the App root, so we only + // need Tooltip > TooltipTrigger > TooltipContent here. + + {/* + * WHY render prop: @base-ui/react TooltipTrigger renders a
+ + + + + + + + + {rows.map((row) => ( + + + + + + ))} + +
Question IDDataset + A → B (Δ) +
+ {shortId(row.question_id)} + + {row.dataset} + + {/* + * PATTERN: Show raw scores for transparency so readers know + * whether a +0.20 delta is 0.60→0.80 or 0.10→0.30 — the + * absolute values matter for interpretation. + */} + {row.a_score.toFixed(2)} → {row.b_score.toFixed(2)}{" "} + = 0 ? "text-green-600" : "text-destructive" + } + > + ({fmtDelta(row.delta)}) + +
+ + + ); +} + +// --------------------------------------------------------------------------- +// Main component +// --------------------------------------------------------------------------- + +/** + * CompareView renders a side-by-side comparison of two eval runs. + * + * URL contract: `/eval/compare?a=&b=`. + * Both params must be present; missing either shows a friendly prompt. + * + * Data flow: + * URL params a, b + * → useCompareRuns(a, b) (single API call) + * → CompareResult { run_a, run_b, deltas, per_question_diff } + * → mapAggregatedFromDeltas (derives AggregatedMetric[] for each side) + * → MetricBars (comparison mode) + * → DiffCard × 2 (wins / regressions) + */ +export function CompareView(): React.JSX.Element { + const [searchParams] = useSearchParams(); + const a = searchParams.get("a") ?? undefined; + const b = searchParams.get("b") ?? undefined; + + // ------------------------------------------------------------------------- + // Missing-params guard — shown before the hook fires + // ------------------------------------------------------------------------- + + if (!a || !b) { + return ( +
+

+ Select two runs from the runs list to compare. +

+ + Go to runs list + +
+ ); + } + + // ------------------------------------------------------------------------- + // Data fetch — useCompareRuns is enabled only when both IDs are defined + // ------------------------------------------------------------------------- + + // IMPORTANT: hooks must be called unconditionally in React. The early return + // above only fires when a or b is undefined, so by this point both are + // strings — the hook's `enabled: !!idA && !!idB` will be true. + // + // WHY not call the hook before the guard: it would still be enabled=false and + // return { data: undefined, isLoading: false }, so this ordering is safe. + // Hooks are called on every render regardless; the guard just short-circuits + // the JSX, not the hook call itself. + + return ; +} + +/** + * CompareViewInner — rendered only when both run IDs are present. + * + * WHY split into an inner component: + * React's rules of hooks prohibit calling hooks conditionally. The outer + * CompareView must return early for the missing-params case. Moving the hook + * call here lets the guard live at the top level without violating the rules. + * + * PATTERN: This is a standard React "wrapper with guard → inner with hook" + * split. RunDetail uses the same pattern for runId from useParams. + */ +function CompareViewInner({ a, b }: { a: string; b: string }) { + const { data, isLoading, isError, error } = useCompareRuns(a, b); + + // ------------------------------------------------------------------------- + // Loading state + // ------------------------------------------------------------------------- + + if (isLoading) return ; + + // ------------------------------------------------------------------------- + // Error state — 409 (version mismatch) gets a distinct banner + // ------------------------------------------------------------------------- + + if (isError) { + const msg = + error instanceof Error ? error.message : "Unknown error"; + + // WHY check "mismatch": the backend detail string is + // "eval set version mismatch between runs" (compare.py, line 140). + // Matching "mismatch" is narrower than "version" and won't false-positive + // on other errors that might mention "version". + const isVersionMismatch = + msg.toLowerCase().includes("mismatch"); + + if (isVersionMismatch) { + return ( +
+
+ These runs cannot be compared because they used different eval sets. +
+ + ← Back to runs list + +
+ ); + } + + // Generic error (404, 500, network failure) + return ( +
+
+ Failed to load comparison: {msg} +
+ + ← Back to runs list + +
+ ); + } + + if (!data) return ; + + // ------------------------------------------------------------------------- + // Derived state + // ------------------------------------------------------------------------- + + const { run_a, run_b, deltas, per_question_diff } = data; + + // Cast per_question_diff to our typed interface once. The backend schema + // guarantees this shape (see src/eval/compare.py), but the wire type is + // Array> to avoid coupling the TS types to the + // Python DTO in full detail. + // WHY double cast (unknown → PerQuestionDiff[]): TypeScript won't allow a + // direct cast from Record[] to a named interface because the + // two types don't share an overlap that TS can verify statically. Casting + // through unknown is the idiomatic escape hatch when the shape is guaranteed + // by a runtime contract we trust (the backend schema). + const diffs = per_question_diff as unknown as PerQuestionDiff[]; + + // Wins: positive deltas, largest first, max 5. + const wins = diffs + .filter((d) => d.delta > 0) + .sort((x, y) => y.delta - x.delta) + .slice(0, 5); + + // Regressions: negative deltas, largest magnitude first (most negative → first), max 5. + const regressions = diffs + .filter((d) => d.delta < 0) + .sort((x, y) => x.delta - y.delta) + .slice(0, 5); + + // Rebuild AggregatedMetric arrays for MetricBars comparison mode. + const metricsA = mapAggregatedFromDeltas(deltas, "a"); + const metricsB = mapAggregatedFromDeltas(deltas, "b"); + + // ------------------------------------------------------------------------- + // Render — happy path + // ------------------------------------------------------------------------- + + return ( +
+ {/* Back link */} + + ← All runs + + + {/* Section 1: Run summaries — two-column grid */} +
+ + +
+ + {/* Section 2: Metric comparison chart */} + + + Metric Comparison + + + + {/* + * WHY a separate caption here despite MetricBars rendering its own + * significance list: MetricBars' line names *which* metrics are + * significant. This caption explains the test method so readers know + * what "significant" means without digging into the source. + * + * TRADE-OFF: mild duplication of context. The alternative — removing + * this caption — leaves the ★ symbol unexplained at the chart level. + */} +

+ ★ marks differences with p < 0.05 (paired permutation test, n=10 000). +

+
+
+ + {/* Sections 3 & 4: Top wins and top regressions cards */} + {(wins.length > 0 || regressions.length > 0) && ( +
+ {wins.length > 0 && ( + + )} + {regressions.length > 0 && ( + + )} +
+ )} +
+ ); +} diff --git a/frontend/src/components/eval/metric-bars.tsx b/frontend/src/components/eval/metric-bars.tsx new file mode 100644 index 00000000..53010cc9 --- /dev/null +++ b/frontend/src/components/eval/metric-bars.tsx @@ -0,0 +1,319 @@ +/** + * MetricBars — recharts BarChart for AggregatedMetric arrays with CI whiskers. + * + * RAG Pipeline Position (Evaluation Layer): + * Run → AggregatedMetric[] → [METRIC-BARS] → Visual bar chart + * ^^^ + * This component sits at the DISPLAY step of evaluation: it takes + * pre-aggregated per-metric statistics (mean, CI bounds) and renders + * them as a bar chart so engineers can compare quality at a glance. + * + * WHY recharts over a custom SVG: + * recharts gives us ErrorBar (CI whiskers), ResponsiveContainer, and + * animated bars without any D3 boilerplate. The trade-off is a heavier + * bundle, but recharts is already installed for this project. + * + * Two modes: + * - Single-run: one bar per (metric_name, dataset) row + ErrorBar whiskers. + * - Comparison: two bars per row (Run A / Run B). Statistically significant + * deltas are called out below the chart in a ★-prefixed list. + * + * TRADE-OFF: The ★ significant-delta annotation is rendered as text below the + * chart rather than as an overlay on the bars. Overlaying labels on grouped + * bars in recharts requires a custom renderer that re-computes + * bar x/y positions — fragile across bar widths. The text list is simpler, + * equally informative, and survives recharts version changes. + */ + +import { + Bar, + BarChart, + CartesianGrid, + ErrorBar, + Legend, + ResponsiveContainer, + Tooltip, + XAxis, + YAxis, +} from "recharts"; + +import type { AggregatedMetric, MetricDelta } from "@/api/eval"; + +// --------------------------------------------------------------------------- +// Public types +// --------------------------------------------------------------------------- + +export interface MetricBarsProps { + metrics: AggregatedMetric[]; + comparison?: { + b: AggregatedMetric[]; + deltas: MetricDelta[]; + }; + /** Chart height in px. Default 300. */ + height?: number; + className?: string; +} + +// --------------------------------------------------------------------------- +// Internal row shape consumed by recharts +// --------------------------------------------------------------------------- + +interface ChartRow { + /** X-axis label: "{metric_name} ({dataset || 'all'})" */ + label: string; + /** Run A mean */ + a_mean: number; + /** + * ErrorBar offset pair: [distance below mean, distance above mean]. + * recharts ErrorBar interprets a 2-element array as [below, above]. + */ + a_err: [number, number]; + /** Run B mean (comparison mode only) */ + b_mean?: number; + b_err?: [number, number]; + /** True when the matching MetricDelta.significant is true */ + significant?: boolean; +} + +// --------------------------------------------------------------------------- +// buildRows — transforms AggregatedMetric arrays into chart-ready rows +// --------------------------------------------------------------------------- + +/** + * Build ChartRow array from one or two sets of AggregatedMetric. + * + * Single-run: every metric in `a` becomes a row. + * Compare: only metrics present in BOTH `a` and `b` are included — partial + * matches would render an empty bar for the missing side, which is confusing. + * + * WHY offset-based ErrorBar: + * recharts ErrorBar for a single expects the dataKey to point at a + * 2-element array [belowOffset, aboveOffset], NOT absolute CI values. + * So we convert: below = mean - ci_low, above = ci_high - mean. + */ +function buildRows( + a: AggregatedMetric[], + b?: AggregatedMetric[], + deltas?: MetricDelta[], +): ChartRow[] { + // Build a lookup for Run B keyed by "metric_name|dataset" for O(1) access. + const bByKey = new Map(); + if (b) { + for (const m of b) { + bByKey.set(`${m.metric_name}|${m.dataset ?? ""}`, m); + } + } + + // Build a significance lookup keyed the same way. + const sigByKey = new Map(); + if (deltas) { + for (const d of deltas) { + sigByKey.set(`${d.metric_name}|${d.dataset ?? ""}`, d.significant); + } + } + + const rows: ChartRow[] = []; + + for (const am of a) { + const key = `${am.metric_name}|${am.dataset ?? ""}`; + const dataset = am.dataset ?? "all"; + const label = `${am.metric_name} (${dataset})`; + + const a_err: [number, number] = [ + am.mean - am.ci_low, + am.ci_high - am.mean, + ]; + + if (!b) { + // Single-run mode — no Run B needed. + rows.push({ label, a_mean: am.mean, a_err }); + continue; + } + + // Comparison mode — skip metrics missing from Run B. + const bm = bByKey.get(key); + if (!bm) continue; + + const b_err: [number, number] = [ + bm.mean - bm.ci_low, + bm.ci_high - bm.mean, + ]; + + rows.push({ + label, + a_mean: am.mean, + a_err, + b_mean: bm.mean, + b_err, + significant: sigByKey.get(key) ?? false, + }); + } + + return rows; +} + +// --------------------------------------------------------------------------- +// Custom tooltip +// --------------------------------------------------------------------------- + +interface TooltipPayloadEntry { + name: string; + value: number; + payload: ChartRow; + dataKey: string; +} + +interface CustomTooltipProps { + active?: boolean; + payload?: TooltipPayloadEntry[]; + label?: string; +} + +/** + * CustomTooltip — shows "mean (± half-CI-width)" formatted to 4 decimals. + * + * WHY custom tooltip instead of recharts default: + * The default tooltip would show raw a_err/b_err arrays, which are offsets + * not intuitive values. We convert back to ± half-CI for readability. + */ +function CustomTooltip({ active, payload, label }: CustomTooltipProps) { + if (!active || !payload || payload.length === 0) return null; + + return ( +
+

{label}

+ {payload.map((entry) => { + // entry.dataKey is "a_mean" or "b_mean"; the matching err key is "a_err" or "b_err". + const errKey = entry.dataKey === "a_mean" ? "a_err" : "b_err"; + const err = entry.payload[errKey as "a_err" | "b_err"]; + // Half-CI-width is the average of [below, above] offsets. + const halfCI = err ? ((err[0] + err[1]) / 2).toFixed(4) : "—"; + return ( +

+ {entry.name}: {entry.value.toFixed(4)} (±{halfCI}) +

+ ); + })} +
+ ); +} + +// --------------------------------------------------------------------------- +// MetricBars — public export +// --------------------------------------------------------------------------- + +/** + * MetricBars renders AggregatedMetric arrays as a bar chart. + * + * @example Single-run + * + * + * @example Comparison + * + */ +export function MetricBars({ + metrics, + comparison, + height = 300, + className, +}: MetricBarsProps) { + // Empty state — return a muted paragraph so the parent doesn't render + // dead whitespace or an empty chart frame. + if (metrics.length === 0) { + return ( +

No metrics yet.

+ ); + } + + const rows = buildRows(metrics, comparison?.b, comparison?.deltas); + + // Y-axis: use [0, 1] for standard 0-1 metrics. If any mean exceeds 1, + // let recharts auto-scale (pass undefined to let it decide). + // WHY: Most RAG metrics (recall@k, faithfulness, NDCG) live in [0, 1]. + // Auto-scaling for edge cases (e.g., raw token counts) prevents clipping. + const allMeans = rows.flatMap((r) => + r.b_mean !== undefined ? [r.a_mean, r.b_mean] : [r.a_mean], + ); + const maxMean = Math.max(...allMeans); + const yDomain: [number | string, number | string] = + maxMean <= 1.0 ? [0, 1] : [0, "auto"]; + + // Collect labels of significant deltas for the annotation list. + const significantLabels = rows + .filter((r) => r.significant) + .map((r) => r.label); + + return ( +
+ + + + {/* + * XAxis angle -30 with textAnchor="end" and extra height prevents + * long labels from overlapping. This mirrors the pattern used in + * evaluation dashboards where metric names can be verbose. + */} + + + } /> + + + {/* Run A bar — always rendered */} + + {/* + * PATTERN: ErrorBar dataKey points at the [below, above] offset + * array on each row. recharts uses these to draw whiskers relative + * to the bar top, not absolute SVG coordinates. + */} + + + + {/* Run B bar — only in comparison mode */} + {comparison ? ( + + + + ) : null} + + + + {/* + * Significant delta annotations — rendered as text below the chart. + * WHY text list over bar overlay: overlaying labels on grouped bars + * requires re-computing bar x/y positions via recharts internals, + * which is fragile. The list is simpler and equally informative. + */} + {comparison && significantLabels.length > 0 && ( +
+ ★ Statistically significant (p < 0.05):{" "} + {significantLabels.join(", ")} +
+ )} + + {comparison && significantLabels.length === 0 && ( +

+ No statistically significant differences between runs. +

+ )} +
+ ); +} diff --git a/frontend/src/components/eval/new-eval-run-dialog.tsx b/frontend/src/components/eval/new-eval-run-dialog.tsx new file mode 100644 index 00000000..efeeb054 --- /dev/null +++ b/frontend/src/components/eval/new-eval-run-dialog.tsx @@ -0,0 +1,365 @@ +/** + * NewEvalRunDialog — config picker + run submit + status-poll toast. + * + * Frontend Position: + * RunsList "New Run" button → [Dialog] → POST /api/eval/run + * → toast → poll /status → "View Run" + * + * WHY this approach (Approach B — self-contained toast): + * The toast lifecycle is independent of the dialog: after submit the dialog + * closes but the toast must live on until the run finishes. We manage both + * in one file by keeping `activeRunId` state here. The toast renders as a + * fixed-position card outside the dialog markup, so it persists even after + * the dialog closes. + * + * PATTERN: Controlled dialog — the `open` / `onOpenChange` props come from + * RunsList, which renders at a stable JSX position (not + * inside a conditional branch). This guarantees React never remounts the + * component (and loses `activeRunId`) when the runs list transitions from + * "empty" to "loaded" after the first run is submitted. + */ + +import React, { useState } from "react"; +import { useNavigate } from "react-router"; +import { useQuery } from "@tanstack/react-query"; +import { + Dialog, + DialogContent, + DialogFooter, + DialogHeader, + DialogTitle, +} from "@/components/ui/dialog"; +import { Button } from "@/components/ui/button"; +import { Skeleton } from "@/components/ui/skeleton"; +import { Progress } from "@/components/ui/progress"; +import { getRunStatus, useConfigs, useSubmitRun } from "@/api/eval"; + +// --------------------------------------------------------------------------- +// Public interface +// --------------------------------------------------------------------------- + +export interface NewEvalRunDialogProps { + open: boolean; + /** Called by Base UI Dialog when it wants to open/close (dismiss on Escape, + * backdrop click). We also call it explicitly after a successful submit. */ + onOpenChange: (open: boolean) => void; +} + +// --------------------------------------------------------------------------- +// Main exported component +// --------------------------------------------------------------------------- + +/** + * NewEvalRunDialog renders two co-located UIs: + * 1. A modal dialog for picking a config and submitting a run. + * 2. A fixed-position toast card that appears after submit and polls the + * run status until completion or failure. + * + * WHY co-located: only one component tracks `activeRunId`, keeping the state + * in one place and avoiding cross-component event buses. + */ +export function NewEvalRunDialog({ + open, + onOpenChange, +}: NewEvalRunDialogProps): React.JSX.Element { + const [selectedConfig, setSelectedConfig] = useState(""); + // activeRunId is set on submit success and cleared when the toast is dismissed. + const [activeRunId, setActiveRunId] = useState(null); + + const { data: configs, isLoading: configsLoading } = useConfigs(); + const submitRun = useSubmitRun(); + + // --------------------------------------------------------------------------- + // Handlers + // --------------------------------------------------------------------------- + + function handleSubmit() { + if (!selectedConfig) return; + + submitRun.mutate(selectedConfig, { + onSuccess(result) { + // Close the dialog and start tracking the new run. + onOpenChange(false); + setSelectedConfig(""); + setActiveRunId(result.run_id); + }, + }); + } + + // Base UI's onOpenChange passes (open, eventDetails) — we only need `open`. + // When the dialog closes (Escape or backdrop), reset form state but keep the + // active toast if one is running. + function handleOpenChange(next: boolean) { + if (!next) { + // Reset form when dialog closes (cancel or dismiss). + setSelectedConfig(""); + submitRun.reset(); + } + onOpenChange(next); + } + + // --------------------------------------------------------------------------- + // Render + // --------------------------------------------------------------------------- + + return ( + <> + {/* ------------------------------------------------------------------ */} + {/* Modal dialog */} + {/* ------------------------------------------------------------------ */} + + + + New Eval Run + + +
+ + + {configsLoading ? ( + /* Loading skeleton — shown while /api/eval/configs is in flight */ + + ) : configs && configs.length > 0 ? ( + /* + * WHY native is accessible, keyboard-navigable, and renders + * identically in all browsers — the right default here. + */ + + ) : ( + /* + * Empty-state hint — no configs to choose from. We render plain + * text (not a disabled : pills make all options visible + * at once and are faster to toggle for 2-4 datasets — the typical + * range for this eval harness. + */} + {availableDatasets.length > 0 && ( +
+ {["all", ...availableDatasets].map((ds) => ( + + ))} +
+ )} + + + + + + {/* Section 3: Cost summary */} + + + {/* Section 4: Per-question table */} + + + + Per-question Results + + ({n_results} total) + + + + + {resultsLoading ? ( +
+ {Array.from({ length: 5 }).map((_, i) => ( + + ))} +
+ ) : ( + + + + {/* Expand toggle — no label, just space */} + + Question ID + Dataset + Error + {metricCols.map((col) => ( + + {col} + + ))} + + + + {items.map((row) => ( + <> + {/* + * PATTERN: key on question_id so React doesn't reuse DOM + * nodes across rows when the page changes. Using array index + * as key would cause stale expanded state after pagination. + */} + toggleRow(row.question_id)} + data-state={ + expandedId === row.question_id ? "selected" : undefined + } + > + + {expandedId === row.question_id ? ( + + ) : ( + + )} + + + {shortId(row.question_id)} + + {row.dataset} + + {row.error ?? ""} + + {metricCols.map((col) => ( + + {row.metrics[col] !== undefined + ? row.metrics[col].toFixed(3) + : "—"} + + ))} + + + {/* + * PATTERN: Render ExpandedDetail only when this row is + * expanded. Mounting triggers the lazy getRunResult query; + * unmounting on collapse keeps the result cached via + * TanStack Query — re-expanding is instant. + * + * WHY a separate : we need the expanded content to + * span all table columns cleanly. A colSpan cell inside + * its own row is the only way to achieve this in HTML tables. + */} + {expandedId === row.question_id && runId && ( + + + + + + )} + + ))} + + {items.length === 0 && ( + + + No results on this page. + + + )} + +
+ )} +
+
+ + {/* Section 5: Pagination controls */} + {totalPages > 1 && ( +
+ + + Page {page} of {totalPages} + + +
+ )} +
+ ); +} diff --git a/frontend/src/components/eval/runs-list.tsx b/frontend/src/components/eval/runs-list.tsx new file mode 100644 index 00000000..4999dbe3 --- /dev/null +++ b/frontend/src/components/eval/runs-list.tsx @@ -0,0 +1,527 @@ +/** + * RunsList — table of eval runs with sort, filter, and multi-select compare. + * + * Frontend Position: + * /eval (index route) → [RunsList] → row click → /eval/runs/:runId + * → "Compare" → /eval/compare?a=&b= + * → "New Run" → NewEvalRunDialog (Task 10) + * + * WHY this file exists: + * The eval dashboard entry point. Engineers need to see all past runs at a + * glance, filter by config, sort by any column, and quickly navigate to + * detailed views or side-by-side comparisons. + * + * PATTERN: Filter → sort → render is a classic derived-state pipeline. + * Both transformations are memoised with useMemo so they don't re-run on + * unrelated parent re-renders. + */ + +import React, { useMemo, useState } from "react"; +import { useNavigate } from "react-router"; +import { useDebounce } from "@/hooks/use-debounce"; +import { useRunsList } from "@/api/eval"; +import type { RunSummary } from "@/api/eval"; +import { Button } from "@/components/ui/button"; +import { Input } from "@/components/ui/input"; +import { Skeleton } from "@/components/ui/skeleton"; +import { + Table, + TableBody, + TableCell, + TableHead, + TableHeader, + TableRow, +} from "@/components/ui/table"; +import { NewEvalRunDialog } from "@/components/eval/new-eval-run-dialog"; + +// --------------------------------------------------------------------------- +// Types +// --------------------------------------------------------------------------- + +type SortKey = keyof Pick< + RunSummary, + "config_name" | "started_at" | "n_questions" | "n_errors" | "headline_metric" +>; + +type SortDir = "asc" | "desc"; + +// --------------------------------------------------------------------------- +// Helpers +// --------------------------------------------------------------------------- + +/** + * Format an ISO timestamp as a short locale date + time string. + * + * WHY Intl.DateTimeFormat over date-fns: + * No extra dependency; the browser-native formatter handles locale/timezone + * automatically and is already used in doc-table for date-only formatting. + */ +const dateFmt = new Intl.DateTimeFormat(undefined, { + dateStyle: "short", + timeStyle: "short", +}); + +function formatDate(iso: string): string { + try { + return dateFmt.format(new Date(iso)); + } catch { + return iso; + } +} + +/** + * Format headline_metric to 3 decimal places, or "—" if null. + * + * WHY 3 decimals: standard precision for recall/NDCG metrics reported to + * engineers — enough signal without false precision. + */ +function formatMetric(value: number | null): string { + return value === null ? "—" : value.toFixed(3); +} + +/** + * Truncate a run_id UUID to the first 8 chars for display. + * + * WHY: Full UUIDs overflow narrow columns and are unreadable at a glance. + * 8 chars is enough to disambiguate within a typical run list. + */ +function shortId(id: string): string { + return id.length > 8 ? id.slice(0, 8) + "…" : id; +} + +/** + * Generic comparator that handles null values, always sorting them last + * regardless of sort direction. + * + * WHY nulls-last: a null headline_metric means the run has no results yet, + * which is less informative than a real value. Keeping nulls at the bottom + * regardless of direction avoids them dominating the first rows on asc. + */ +function compare(a: RunSummary, b: RunSummary, key: SortKey, dir: SortDir): number { + const va = a[key]; + const vb = b[key]; + + // Nulls always sink to the bottom. + if (va === null && vb === null) return 0; + if (va === null) return 1; + if (vb === null) return -1; + + // ISO strings sort correctly as strings (lexicographic == chronological). + // Numbers sort numerically. Both comparisons are uniform here. + const cmp = va < vb ? -1 : va > vb ? 1 : 0; + return dir === "asc" ? cmp : -cmp; +} + +// --------------------------------------------------------------------------- +// Sort header button +// --------------------------------------------------------------------------- + +/** Renders a clickable column header with an ↑/↓ glyph when active. */ +function SortHead({ + label, + sortKey, + currentKey, + currentDir, + onSort, + className, +}: { + label: string; + sortKey: SortKey; + currentKey: SortKey; + currentDir: SortDir; + onSort: (key: SortKey) => void; + className?: string; +}) { + const isActive = currentKey === sortKey; + return ( + + + + ); +} + +// --------------------------------------------------------------------------- +// Loading skeleton +// --------------------------------------------------------------------------- + +/** Renders 5 placeholder rows while the API request is in-flight. */ +function RunsListSkeleton() { + return ( +
+
+ +
+ + +
+
+ + + + {["w-6", "w-24", "w-40", "w-36", "w-20", "w-16", "w-28"].map( + (w, i) => ( + + + + ), + )} + + + + {Array.from({ length: 5 }).map((_, i) => ( + + {Array.from({ length: 7 }).map((__, j) => ( + + + + ))} + + ))} + +
+
+ ); +} + +// --------------------------------------------------------------------------- +// Main component +// --------------------------------------------------------------------------- + +/** + * RunsList renders the full list of eval runs. + * + * State machines: + * - Loading → skeleton + * - Error → error banner + * - Empty → "No eval runs yet" callout + * - Loaded → sorted, filtered, paginated table + * + * Multi-select behaviour: + * Selected IDs are kept in a Set. Filtering/sorting does NOT clear + * the selection — selected IDs that scroll off the visible (filtered) list + * remain selected and still count toward the "Compare Selected" threshold. + * This avoids surprising selection resets while the user is typing a filter. + * + * PATTERN: is rendered OUTSIDE the conditional body branches + * so React never unmounts it when the list transitions from "empty" to + * "loaded" after the first run is submitted. If the dialog were inside a + * conditional early-return, its `activeRunId` state (and therefore the + * in-progress toast) would be lost the moment the list refreshes. + */ +export function RunsList() { + const navigate = useNavigate(); + const { data, isLoading, isError, error } = useRunsList(); + + // Dialog open state lives here so the dialog is rendered at a stable + // JSX position (the fragment root below) regardless of which body branch + // is currently active. + const [dialogOpen, setDialogOpen] = useState(false); + + // Search input (raw) and debounced query fed into useMemo. + const [searchRaw, setSearchRaw] = useState(""); + const query = useDebounce(searchRaw, 300); + + // Sort state — default: started_at descending (most recent first). + const [sortKey, setSortKey] = useState("started_at"); + const [sortDir, setSortDir] = useState("desc"); + + // Multi-select: a Set of run_ids that are checked. + // WHY Set: O(1) has/toggle, serialises cheaply to an array for the URL. + const [selected, setSelected] = useState>(new Set()); + + // ------------------------------------------------------------------------- + // Derived state: filter then sort + // ------------------------------------------------------------------------- + + const filtered = useMemo(() => { + if (!data) return []; + if (!query) return data; + const q = query.toLowerCase(); + return data.filter((r) => r.config_name.toLowerCase().includes(q)); + }, [data, query]); + + const sorted = useMemo(() => { + return [...filtered].sort((a, b) => compare(a, b, sortKey, sortDir)); + }, [filtered, sortKey, sortDir]); + + // ------------------------------------------------------------------------- + // Handlers + // ------------------------------------------------------------------------- + + function handleSortClick(key: SortKey) { + if (key === sortKey) { + // Same column → toggle direction. + setSortDir((d) => (d === "asc" ? "desc" : "asc")); + } else { + // New column → default to descending so largest/latest shows first. + setSortKey(key); + setSortDir("desc"); + } + } + + function toggleSelect(id: string) { + setSelected((prev) => { + const next = new Set(prev); + if (next.has(id)) { + next.delete(id); + } else { + next.add(id); + } + return next; + }); + } + + function handleCompare() { + // Guard: only proceed when exactly 2 are selected (belt-and-suspenders + // since the button is also disabled otherwise). + const ids = Array.from(selected); + if (ids.length !== 2) return; + navigate(`/eval/compare?a=${encodeURIComponent(ids[0])}&b=${encodeURIComponent(ids[1])}`); + } + + function handleRowClick(runId: string) { + navigate(`/eval/runs/${runId}`); + } + + // ------------------------------------------------------------------------- + // Body: choose the appropriate inner content for the current data state. + // Using a variable (not early returns) keeps at a + // stable position in the returned JSX tree. + // ------------------------------------------------------------------------- + + let body: React.JSX.Element; + + if (isLoading) { + body = ; + } else if (isError) { + body = ( +
+ Failed to load eval runs:{" "} + {error instanceof Error ? error.message : "Unknown error"} +
+ ); + } else if (!data || data.length === 0) { + body = ( +
+
+ +
+
+

No eval runs yet.

+

+ Click "New Run" to kick off your first evaluation. +

+
+
+ ); + } else { + const sortHeadProps = { + currentKey: sortKey, + currentDir: sortDir, + onSort: handleSortClick, + }; + + body = ( +
+ {/* Toolbar: search left, action buttons right */} +
+ setSearchRaw(e.target.value)} + className="max-w-xs" + /> + +
+ {/* + * WHY exactly-2 gate: comparing 1 or 3+ runs has no defined meaning + * in the current CompareView (it accepts exactly two run IDs). + */} + + + +
+
+ + + + + {/* Checkbox header — no sort, just labels the column */} + + + {/* Run ID — not sortable; truncated for display only */} + Run ID + + + + + + + + + + + {sorted.map((run) => ( + toggleSelect(run.run_id)} + onRowClick={() => handleRowClick(run.run_id)} + /> + ))} + + {/* Filter produced no results, but runs exist */} + {sorted.length === 0 && ( + + + No runs match your filter. + + + )} + +
+
+ ); + } + + // ------------------------------------------------------------------------- + // Render + // ------------------------------------------------------------------------- + + return ( + <> + {/* + * WHY p-6 wrapper: matches CompareView/RunDetail's outer padding so the + * eval routes share consistent breathing room and the toolbar doesn't + * sit flush against the top of
. + */} +
{body}
+ {/* + * PATTERN: Dialog rendered at a stable position in the JSX tree — + * outside the conditional `body` branches — so React never unmounts + * it when data state changes. This preserves the toast's `activeRunId` + * state across the empty→loaded transition that happens right after + * the first run is submitted. + */} + + + ); +} + +// --------------------------------------------------------------------------- +// Row sub-component — keeps the main render readable +// --------------------------------------------------------------------------- + +function RunRow({ + run, + isSelected, + onToggle, + onRowClick, +}: { + run: RunSummary; + isSelected: boolean; + onToggle: () => void; + onRowClick: () => void; +}) { + return ( + + {/* + * PATTERN: stopPropagation on the checkbox cell, not just the input, + * so the row click does not fire when the user is toggling selection. + * Mirrors the doc-table pattern for delete/expand buttons. + */} + { + e.stopPropagation(); + onToggle(); + }} + className="pr-0" + > + e.stopPropagation()} + /> + + + + {shortId(run.run_id)} + + + {run.config_name} + + {formatDate(run.started_at)} + + {run.n_questions} + + + {/* Non-zero errors are highlighted so they stand out at a glance. */} + 0 ? "text-destructive font-medium" : undefined}> + {run.n_errors} + + + + + {formatMetric(run.headline_metric)} + + + ); +} + diff --git a/frontend/src/components/layout/sidebar.tsx b/frontend/src/components/layout/sidebar.tsx index cf2c3f7e..90a2af0a 100644 --- a/frontend/src/components/layout/sidebar.tsx +++ b/frontend/src/components/layout/sidebar.tsx @@ -24,6 +24,7 @@ import { NavLink, useLocation, useNavigate } from "react-router"; import { Upload, FolderOpen, + BarChart3, PanelLeftClose, PanelLeftOpen, FileText, @@ -79,6 +80,7 @@ import type { ConversationSummary } from "@/api/types"; const NAV_ITEMS = [ { to: "/upload", label: "Upload", icon: Upload }, { to: "/documents", label: "Documents", icon: FolderOpen }, + { to: "/eval", label: "Evaluation", icon: BarChart3 }, ] as const; // --- Date grouping helper --- diff --git a/frontend/src/hooks/use-chat.ts b/frontend/src/hooks/use-chat.ts index 3cbe9c22..f5e9f3b4 100644 --- a/frontend/src/hooks/use-chat.ts +++ b/frontend/src/hooks/use-chat.ts @@ -1,7 +1,7 @@ import { useCallback, useRef, useState } from "react"; import { useQueryClient } from "@tanstack/react-query"; import { wsUrl } from "@/api/client"; -import type { ChatMessage, EvaluationScore, SourceInfo, WsMessage } from "@/api/types"; +import type { ChatMessage, EvaluationScore, SourceInfo, TelemetryPayload, WsMessage } from "@/api/types"; let msgCounter = 0; function nextId() { @@ -127,6 +127,27 @@ export function useChat() { setTimeout(() => { if (ws.readyState === WebSocket.OPEN) ws.close(); }, 30_000); + } else if (data.type === "telemetry") { + // WHY: The backend emits a "telemetry" event immediately after "done" + // (before evaluation) with per-request timing and token cost. + // We attach it to the assistant message so the UI can render a + // TelemetryFooter without polling or a separate API call. + // + // PATTERN: Same setMessages updater as the other event handlers — + // merge the payload into the target message by ID. + // + // TEST TODO: A React Testing Library test should feed the sequence + // status → reasoning → token → done → telemetry → evaluation + // and assert that the last assistant message has both a non-empty + // `content` string and a `telemetry` object with numeric fields + // (retrieve_ms, generate_ms, prompt_tokens, completion_tokens, cost_usd). + setMessages((prev) => + prev.map((m) => + m.id === assistantIdRef.current + ? { ...m, telemetry: data.content as TelemetryPayload } + : m + ) + ); } else if (data.type === "evaluation") { // WHY: The backend fires a separate WebSocket event after the "done" // event with real-time faithfulness scores. We attach it to the diff --git a/frontend/src/hooks/use-debounce.ts b/frontend/src/hooks/use-debounce.ts new file mode 100644 index 00000000..ee0a6644 --- /dev/null +++ b/frontend/src/hooks/use-debounce.ts @@ -0,0 +1,30 @@ +/** + * useDebounce — delays propagating a value until it has been stable for + * `delay` ms. + * + * WHY a separate hook: + * Debouncing is used in multiple components (search inputs, live filters). + * Extracting to a shared hook avoids duplicate useEffect+setTimeout logic + * and keeps each component file under the 250-line guideline. + * + * PATTERN: useEffect + setTimeout is the canonical lightweight debounce. + * For heavy computation debouncing, a library hook (e.g., use-debounce) is + * worth the dependency — for simple UI filters this is sufficient. + */ + +import { useEffect, useRef, useState } from "react"; + +export function useDebounce(value: T, delay: number): T { + const [debounced, setDebounced] = useState(value); + const timerRef = useRef | null>(null); + + useEffect(() => { + if (timerRef.current !== null) clearTimeout(timerRef.current); + timerRef.current = setTimeout(() => setDebounced(value), delay); + return () => { + if (timerRef.current !== null) clearTimeout(timerRef.current); + }; + }, [value, delay]); + + return debounced; +} diff --git a/frontend/src/pages/eval-page.tsx b/frontend/src/pages/eval-page.tsx new file mode 100644 index 00000000..f64ba5b1 --- /dev/null +++ b/frontend/src/pages/eval-page.tsx @@ -0,0 +1,41 @@ +/** + * EvalPage — sub-route dispatcher for /eval/*. + * + * RAG Pipeline Position: + * This is an EVALUATION layer component — it sits outside the ingestion/ + * retrieval/generation pipeline and provides tooling to measure how well + * that pipeline performs. + * + * /eval → RunsList (browse & manage evaluation runs) + * /eval/runs/:runId → RunDetail (per-question breakdown + metrics) + * /eval/compare?a=&b= → CompareView (side-by-side metric comparison) + * + * WHY nested : + * The parent router registers /eval/* (wildcard). React Router strips the + * matched prefix (/eval) and passes the remainder to this component's own + * , which resolves the sub-path. This avoids co-locating eval + * routing logic in the root router and keeps eval concerns self-contained. + * + * PATTERN: Page components are thin route dispatchers. Business logic lives + * inside the individual feature components (RunsList, RunDetail, + * CompareView), not here. + */ + +import { Routes, Route } from "react-router"; + +import { RunsList } from "@/components/eval/runs-list"; +import { RunDetail } from "@/components/eval/run-detail"; +import { CompareView } from "@/components/eval/compare-view"; + +export function EvalPage() { + return ( + + {/* /eval → full list of evaluation runs */} + } /> + {/* /eval/runs/:runId → per-question detail for one run */} + } /> + {/* /eval/compare?a=&b= → side-by-side metric comparison */} + } /> + + ); +} diff --git a/requirements.txt b/requirements.txt index ce88782b..72dfd2e9 100644 --- a/requirements.txt +++ b/requirements.txt @@ -4,12 +4,20 @@ uvicorn>=0.23 pydantic>=2.0 pytest>=7.0 httpx>=0.25 +jinja2>=3 pypdf>=4.0 python-docx>=1.0 beautifulsoup4>=4.12 requests>=2.31 python-multipart>=0.0.6 +pyyaml>=6 openai>=1.0 anthropic>=0.40 chromadb>=1.0.0 +datasets>=2.16,<3 +sentence-transformers>=2.7 sqlmodel>=0.0.24 +arize-phoenix>=4.0 +opentelemetry-exporter-otlp-proto-http>=1.25 +opentelemetry-sdk>=1.25 +rank-bm25==0.2.2 diff --git a/src/api/main.py b/src/api/main.py index 6a3f35ba..c3d58e9f 100644 --- a/src/api/main.py +++ b/src/api/main.py @@ -41,6 +41,9 @@ query_router, upload_router, ) +from src.api.routes.eval import router as eval_router +from src.api.services.eval_runs import RunRegistry +from src.observability import init_observability @asynccontextmanager @@ -80,6 +83,21 @@ async def lifespan(app: FastAPI): app.state.engine = engine app.state.backend = RAGBackend(engine=engine, collection=collection) + # STEP 4: Create the eval run registry (in-memory, thread-safe). + # WHY: The registry tracks in-flight eval runs across requests. It must + # be a singleton on app.state so POST /api/eval/run and GET /api/eval/runs/{id}/status + # share the same instance — otherwise status polls would see an empty registry. + app.state.run_registry = RunRegistry() + + # STEP 5: Initialise OpenTelemetry tracing toward Phoenix (optional). + # WHY: Called here — after core resources are ready — so span export never + # blocks startup. init_observability is fail-quiet: if Phoenix is + # unreachable it logs a warning and traces become no-ops. Passing None + # (env var absent) uses the function's built-in default endpoint. + # TRADE-OFF: We don't gate on env var presence. The function handles None + # correctly and doing the check here would duplicate its logic. + init_observability(otlp_endpoint=os.getenv("OTLP_ENDPOINT")) + yield # Shutdown: no explicit cleanup needed — SQLite and ChromaDB handle @@ -119,6 +137,9 @@ async def lifespan(app: FastAPI): app.include_router(documents_router) app.include_router(conversations_router) app.include_router(evaluation_router) +# PATTERN: eval router is separate from the existing evaluation_router (which handles +# per-message evaluation). This router manages the full eval harness (runs, configs, compare). +app.include_router(eval_router) @app.get("/health") diff --git a/src/api/models.py b/src/api/models.py index c64abe64..a9a04f14 100644 --- a/src/api/models.py +++ b/src/api/models.py @@ -9,6 +9,7 @@ from pydantic import BaseModel, Field +from src.api.schemas.telemetry import StageTelemetry from src.config import DEFAULT_MODEL @@ -50,7 +51,11 @@ class SourceInfo(BaseModel): class QueryResponse(BaseModel): - """Response body for the query endpoint.""" + """Response body for the query endpoint. + + ADDITIVE: `telemetry` was added in Task 5 (Sub-plan 1D). All other fields + are unchanged. Clients that ignore unknown fields are unaffected. + """ answer: str = Field(description="Generated answer.") sources: List[SourceInfo] = Field(default_factory=list, description="Retrieved source chunks.") @@ -58,6 +63,13 @@ class QueryResponse(BaseModel): ge=0.0, le=1.0, description="Estimated answer confidence (0–1)." ) latency_ms: float = Field(description="Total request latency in milliseconds.") + # WHY: StageTelemetry carries per-stage timing and token-cost numbers. + # Optional with None default so existing callers that construct + # QueryResponse without telemetry (e.g., older tests) still validate. + telemetry: Optional[StageTelemetry] = Field( + default=None, + description="Per-stage timing, token counts, and cost for this request.", + ) class UploadResponse(BaseModel): diff --git a/src/api/routes/eval.py b/src/api/routes/eval.py new file mode 100644 index 00000000..d8465835 --- /dev/null +++ b/src/api/routes/eval.py @@ -0,0 +1,478 @@ +""" +Eval API routes. + +API Layer Position: + Frontend (/eval pages) → [routes] → eval package + storage + registry + +Design decisions: + - Long-running runs dispatched via FastAPI's BackgroundTasks; the + POST returns 202 immediately with a run_id the client can poll. + - RunRegistry tracks in-flight progress for the status endpoint; + persisted runs live on disk via storage.save_run. + - Pagination on results so a 200-question run doesn't ship 200KB+ + of generated text per page load. + - 409 on eval-set version mismatch in compare so the UI can surface + a clear "these runs aren't comparable" banner. +""" + +from __future__ import annotations + +import os +import subprocess +from datetime import datetime, timezone +from pathlib import Path + +from fastapi import APIRouter, BackgroundTasks, HTTPException, Query, Request, status + +from src.api.schemas.eval import ( + AggregatedMetricDTO, + EvalResultDTO, + RunDetailDTO, + RunStatusDTO, + RunSubmitRequest, + RunSubmitResponse, + RunSummaryDTO, +) +from src.api.services.eval_runs import RunRegistry +from src.eval.compare import compare_runs as _compare_runs_impl +from src.eval.config import load_config +from src.eval.runner import EvalRunner +from src.eval.schemas import CompareResult, EvalResult +from src.eval.storage import compute_run_id, list_runs, load_run + +router = APIRouter(prefix="/api/eval", tags=["eval"]) + +# WHY: module-level attribute so monkeypatch.setattr("src.api.routes.eval.CONFIGS_DIR", ...) +# works in tests. Route handlers read this name from module globals at call time. +CONFIGS_DIR = Path("configs/eval") + + +# --------------------------------------------------------------------------- # +# Dependency helper # +# --------------------------------------------------------------------------- # + +def _get_registry(request: Request) -> RunRegistry: + """Extract the shared RunRegistry from app.state. + + PATTERN: Thin helper matching the get_backend pattern in dependencies.py. + Routes call _get_registry(request) to stay testable and explicit. + + WHY lazy init: Starlette's TestClient does not run the lifespan when used + outside a context manager (as in the test fixture). Lazy init guarantees + a registry exists even when the lifespan startup hook hasn't fired — + the registry is still correct because it's a plain in-memory dict. + """ + if not hasattr(request.app.state, "run_registry"): + # PATTERN: thread-safe because attribute assignment on a single object is + # atomic in CPython; worst case two threads both create a registry and one + # overwrites the other — acceptable for test scenarios. + request.app.state.run_registry = RunRegistry() + return request.app.state.run_registry + + +# --------------------------------------------------------------------------- # +# Background worker # +# --------------------------------------------------------------------------- # + +def _run_eval_in_background( + config_name: str, + run_id: str, + registry: RunRegistry, +) -> None: + """Synchronous worker invoked via BackgroundTasks. + + WHY sync (not async): EvalRunner is CPU/IO-mixed and calls blocking LLM + APIs. Sync BackgroundTasks workers are run in a threadpool by Starlette, + keeping the event loop free. An async worker would block the loop. + + Pipeline position: DISPATCH — called once per POST /api/eval/run, + runs the full EvalRunner lifecycle, then marks the run done/failed + in the registry. + """ + # WHY live import of storage: the tmp_eval_runs fixture reloads + # src.eval.storage after setting EVAL_RUNS_DIR. Importing at call time + # ensures we see the reloaded module attribute value. + import src.eval.storage as _storage + + cfg_path = CONFIGS_DIR / f"{config_name}.yaml" + cfg = load_config(cfg_path) + + # PATTERN: respect EVAL_LLM_OVERRIDE_DUMMY=1 — same logic as cli._cmd_run. + # This makes the test harness fast (no real LLM calls). + llm_override = None + judge_llm_override = None + if os.getenv("EVAL_LLM_OVERRIDE_DUMMY") == "1": + from src.eval.cli import _DummyLLM + dummy = _DummyLLM() + llm_override = dummy + judge_llm_override = dummy + + runner = EvalRunner( + cfg, + config_path=cfg_path, + llm_override=llm_override, + judge_llm_override=judge_llm_override, + on_progress=lambda done, total: registry.update_progress(run_id, done), + # WHY run_id_override: we pre-computed the run_id at submit time so the + # registry could be populated before the run starts. Passing it here + # ensures EvalRunner saves to the same directory the status endpoint expects. + run_id_override=run_id, + ) + + try: + runner.run() + registry.mark_completed(run_id) + except Exception as exc: + registry.mark_failed(run_id, str(exc)) + + +# --------------------------------------------------------------------------- # +# GET /api/eval/configs # +# --------------------------------------------------------------------------- # + +@router.get( + "/configs", + response_model=list[str], + summary="List available eval config names", +) +def list_configs() -> list[str]: + """Return the names (without .yaml extension) of all configs in configs/eval/. + + The client uses this to populate the "New eval run" dropdown — no need + to expose the full YAML content at this level. + """ + # WHY: CONFIGS_DIR is read as a module global so monkeypatching works. + if not CONFIGS_DIR.exists(): + return [] + return [p.stem for p in sorted(CONFIGS_DIR.glob("*.yaml"))] + + +# --------------------------------------------------------------------------- # +# POST /api/eval/run # +# --------------------------------------------------------------------------- # + +@router.post( + "/run", + response_model=RunSubmitResponse, + status_code=status.HTTP_202_ACCEPTED, + summary="Submit an eval run (non-blocking)", +) +def submit_run( + body: RunSubmitRequest, + background_tasks: BackgroundTasks, + request: Request, +) -> RunSubmitResponse: + """Queue an eval run and return immediately with a run_id to poll. + + PATTERN: async job — the route validates the config exists, pre-computes + the run_id (same algorithm as EvalRunner so they agree on the directory + name), registers the run in the registry as "queued", dispatches via + BackgroundTasks, then returns 202. The client polls /runs/{run_id}/status. + + WHY pre-compute run_id: the registry must track the run BEFORE it + starts, so the status endpoint can return "queued" immediately after + submission. EvalRunner accepts run_id_override to use the same id. + """ + config_name = body.config_name + cfg_path = CONFIGS_DIR / f"{config_name}.yaml" + + # 404 if the config file doesn't exist. + if not cfg_path.exists(): + raise HTTPException( + status_code=status.HTTP_404_NOT_FOUND, + detail=f"Config '{config_name}' not found in {CONFIGS_DIR}.", + ) + + # Pre-compute run_id using the same algorithm as EvalRunner.run(). + started_at = datetime.now(timezone.utc) + try: + git_sha = subprocess.check_output( + ["git", "rev-parse", "HEAD"], text=True + ).strip() + except Exception: + git_sha = "unknown" + + run_id = compute_run_id(config_name, started_at, git_sha) + + # Register before dispatch so status can return "queued" immediately. + registry = _get_registry(request) + # WHY n_total=0: we don't know question count until the runner loads datasets. + # update_progress transitions the entry to "running" on first call. + registry.register(run_id, n_total=0) + + # Dispatch the synchronous worker via BackgroundTasks (runs in threadpool). + background_tasks.add_task(_run_eval_in_background, config_name, run_id, registry) + + return RunSubmitResponse(run_id=run_id, status="queued") + + +# --------------------------------------------------------------------------- # +# GET /api/eval/runs # +# --------------------------------------------------------------------------- # + +@router.get( + "/runs", + response_model=list[RunSummaryDTO], + summary="List all completed eval runs", +) +def list_eval_runs() -> list[RunSummaryDTO]: + """Return all persisted runs sorted by started_at descending. + + WHY disk-only: completed runs are on disk; in-flight runs lack aggregated + metrics so they aren't useful in the list view. The status endpoint covers + in-flight monitoring. + """ + # WHY live import of list_runs: called via the function (which reads + # EVAL_RUNS_DIR from the module global at call time), so the reloaded + # module attribute is always used correctly. + runs = list_runs() + result: list[RunSummaryDTO] = [] + for meta in runs: + # Compute headline metric: recall_at_5 mean if available, else None. + headline: float | None = None + try: + run_data = load_run(meta.run_id) + aggregated = run_data["aggregated"] + for agg in aggregated: + if agg.metric_name == "recall_at_5" and agg.dataset is not None: + headline = agg.mean + break + except Exception: + # TRADE-OFF: if a run is partially written, skip its headline metric. + pass + result.append( + RunSummaryDTO( + run_id=meta.run_id, + config_name=meta.config_name, + started_at=meta.started_at, + finished_at=meta.finished_at, + n_questions=meta.n_questions, + n_errors=meta.n_errors, + headline_metric=headline, + ) + ) + return result + + +# --------------------------------------------------------------------------- # +# GET /api/eval/runs/{run_id} # +# --------------------------------------------------------------------------- # + +@router.get( + "/runs/{run_id}", + response_model=RunDetailDTO, + summary="Get full detail for one eval run", +) +def get_run(run_id: str) -> RunDetailDTO: + """Return metadata + aggregated metrics + cost for a single run. + + Results are paginated separately — this response is bounded regardless + of how many questions the run evaluated. + """ + try: + run_data = load_run(run_id) + except FileNotFoundError: + raise HTTPException( + status_code=status.HTTP_404_NOT_FOUND, + detail=f"Run '{run_id}' not found.", + ) + + meta = run_data["metadata"] + aggregated = run_data["aggregated"] + cost = run_data["cost"] + results = run_data["results"] + + agg_dtos = [ + AggregatedMetricDTO( + metric_name=a.metric_name, + dataset=a.dataset, + mean=a.mean, + ci_low=a.ci_low, + ci_high=a.ci_high, + n=a.n, + ) + for a in aggregated + ] + + return RunDetailDTO( + metadata=meta, + aggregated=agg_dtos, + cost=cost, + n_results=len(results), + ) + + +# --------------------------------------------------------------------------- # +# GET /api/eval/runs/{run_id}/results # +# --------------------------------------------------------------------------- # + +@router.get( + "/runs/{run_id}/results", + summary="Paginated per-question results for a run", +) +def get_run_results( + run_id: str, + page: int = Query(default=1, ge=1, description="1-indexed page number"), + page_size: int = Query(default=50, ge=1, le=200, description="Items per page (max 200)"), +) -> dict: + """Return a paginated slice of per-question results. + + WHY paginated: a 200-question run has ~200KB of generated text + metrics. + Streaming the whole payload on every page load is wasteful; pagination + lets the UI load one screenful at a time. + + Returns: + Dict with keys: items, page, page_size, total. + """ + try: + run_data = load_run(run_id) + except FileNotFoundError: + raise HTTPException( + status_code=status.HTTP_404_NOT_FOUND, + detail=f"Run '{run_id}' not found.", + ) + + results: list[EvalResult] = run_data["results"] + total = len(results) + + # WHY 0-indexed slice: page=1 → items[0:page_size], page=2 → items[page_size:2*page_size]. + start = (page - 1) * page_size + end = start + page_size + page_items = results[start:end] + + dtos = [ + EvalResultDTO( + question_id=r.question_id, + dataset=r.dataset, + generated_answer=r.generated_answer, + metrics=r.metrics, + error=r.error, + ) + for r in page_items + ] + + return {"items": [d.model_dump() for d in dtos], "page": page, "page_size": page_size, "total": total} + + +# --------------------------------------------------------------------------- # +# GET /api/eval/runs/{run_id}/results/{question_id} # +# --------------------------------------------------------------------------- # + +@router.get( + "/runs/{run_id}/results/{question_id}", + response_model=EvalResult, + summary="Get full EvalResult for one question", +) +def get_question_result(run_id: str, question_id: str) -> EvalResult: + """Return the full EvalResult (including retrieved_chunks, metric_details). + + WHY separate endpoint: the list view uses EvalResultDTO (slim), which omits + the large fields. This endpoint returns the full schema for the detail drawer. + """ + try: + run_data = load_run(run_id) + except FileNotFoundError: + raise HTTPException( + status_code=status.HTTP_404_NOT_FOUND, + detail=f"Run '{run_id}' not found.", + ) + + results: list[EvalResult] = run_data["results"] + for r in results: + if r.question_id == question_id: + return r + + raise HTTPException( + status_code=status.HTTP_404_NOT_FOUND, + detail=f"Question '{question_id}' not found in run '{run_id}'.", + ) + + +# --------------------------------------------------------------------------- # +# GET /api/eval/runs/{run_id}/status # +# --------------------------------------------------------------------------- # + +@router.get( + "/runs/{run_id}/status", + response_model=RunStatusDTO, + summary="Poll the status of an in-progress or completed eval run", +) +def get_run_status(run_id: str, request: Request) -> RunStatusDTO: + """Return the current status and progress for a run. + + WHY two-phase lookup: + 1. Check in-memory registry (covers queued/running/recently-completed). + 2. If not in registry, check disk (covers runs from previous server + restarts that the registry has evicted). + 3. 404 if neither source has the run. + """ + registry = _get_registry(request) + entry = registry.get(run_id) + + if entry is not None: + progress = ( + (entry.n_completed / entry.n_total) + if entry.n_total > 0 and entry.status == "completed" + else (1.0 if entry.status == "completed" else 0.0) + ) + return RunStatusDTO( + run_id=run_id, + status=entry.status, + progress=progress, + n_completed=entry.n_completed, + n_total=entry.n_total, + error_message=entry.error_message, + ) + + # Fallback: check disk for runs not in the registry (e.g. post-restart). + try: + run_data = load_run(run_id) + meta = run_data["metadata"] + return RunStatusDTO( + run_id=run_id, + status="completed", + progress=1.0, + n_completed=meta.n_questions, + n_total=meta.n_questions, + error_message=None, + ) + except FileNotFoundError: + raise HTTPException( + status_code=status.HTTP_404_NOT_FOUND, + detail=f"Run '{run_id}' not found.", + ) + + +# --------------------------------------------------------------------------- # +# GET /api/eval/compare # +# --------------------------------------------------------------------------- # + +@router.get( + "/compare", + response_model=CompareResult, + summary="Compare two eval runs (delta metrics + per-question diff)", +) +def compare_runs( + a: str = Query(..., description="run_id of run A"), + b: str = Query(..., description="run_id of run B"), +) -> CompareResult: + """Compute metric deltas and per-question diffs between two runs. + + TRADE-OFF: 409 on eval-set version mismatch rather than silently computing + a misleading comparison. Different eval-set versions mean different questions, + so per-question deltas are meaningless. Surfaces this clearly to the UI. + """ + try: + return _compare_runs_impl(a, b) + except FileNotFoundError as exc: + raise HTTPException( + status_code=status.HTTP_404_NOT_FOUND, + detail=str(exc), + ) + except ValueError as exc: + # WHY 409 (Conflict): the comparison is a logical conflict — the two + # runs are not comparable because they evaluated different question sets. + raise HTTPException( + status_code=status.HTTP_409_CONFLICT, + detail=str(exc), + ) diff --git a/src/api/routes/query.py b/src/api/routes/query.py index bcf20470..0d61f6ee 100644 --- a/src/api/routes/query.py +++ b/src/api/routes/query.py @@ -37,7 +37,10 @@ async def query(request_body: QueryRequest, request: Request) -> QueryResponse: start = time.perf_counter() backend = request.app.state.backend - result = backend.query( + # WHY query_with_telemetry: replaces the plain query() call so we get + # per-stage timing and token-cost numbers in the response. The + # result_dict has the same shape as before — only telemetry is new. + result, telemetry = backend.query_with_telemetry( request_body.query, top_k=request_body.top_k, model=request_body.model, @@ -61,6 +64,8 @@ async def query(request_body: QueryRequest, request: Request) -> QueryResponse: sources=sources, confidence=result.get("confidence", 0.0), latency_ms=latency_ms, + # ADDITIVE: telemetry field added in Task 5 (Sub-plan 1D). + telemetry=telemetry, ) @@ -177,6 +182,11 @@ async def chat_websocket(websocket: WebSocket) -> None: "message_id": data.get("message_id"), "conversation_id": data.get("conversation_id"), }) + elif event_type == "telemetry": + # WHY: stream_query yields ("telemetry", StageTelemetry.model_dump()) + # after the done event. Forward it verbatim so the frontend + # can render per-stage timing and cost without polling. + await websocket.send_json({"type": "telemetry", "content": data}) except Exception as exc: logger.error("Streaming error: %s", exc) await websocket.send_json( diff --git a/src/api/schemas/__init__.py b/src/api/schemas/__init__.py new file mode 100644 index 00000000..a54a30dc --- /dev/null +++ b/src/api/schemas/__init__.py @@ -0,0 +1,3 @@ +"""API schemas package — request/response DTOs for all route modules.""" + +from __future__ import annotations diff --git a/src/api/schemas/eval.py b/src/api/schemas/eval.py new file mode 100644 index 00000000..3dccf65f --- /dev/null +++ b/src/api/schemas/eval.py @@ -0,0 +1,155 @@ +""" +API DTOs for the eval routes. + +API Layer Position: + src/api/routes/eval.py → [DTOs] → JSON response + +Design decisions: + - DTOs separate from internal eval.schemas: API layer can evolve + independently of the storage/runner layer. + - Reuse RunMetadata (internal schema) inside RunDetailDTO instead of + cloning fields — RunMetadata is JSON-serialisable already. + - EvalResultDTO is a SLIM projection: omits retrieved_chunks (long + text) and metric_details (LLM judge JSON blobs) so list views are + fast. The dedicated /runs/{id}/results/{qid} endpoint returns the + full EvalResult. +""" + +from __future__ import annotations + +from datetime import datetime +from typing import Literal + +from pydantic import BaseModel + +from src.eval.schemas import RunMetadata + + +class RunSummaryDTO(BaseModel): + """Lightweight run summary for list views. + + Teaches: projection pattern — expose only the fields the UI needs + for a list row, not the full run record. Keeps list endpoints fast. + + Pipeline role: Response body for GET /api/eval/runs (list endpoint). + """ + + run_id: str + config_name: str + started_at: datetime + finished_at: datetime + n_questions: int + n_errors: int + # WHY: headline_metric is optional because a run may have failed + # before any metrics were computed. + headline_metric: float | None # recall_at_5 mean if present, else None + + +class AggregatedMetricDTO(BaseModel): + """One aggregated metric for API responses. + + Teaches: mirror-DTO pattern — mirrors AggregatedMetric (internal) + but lives in the API layer so the two can diverge independently. + + Pipeline role: Element of RunDetailDTO.aggregated; also usable as + a standalone response for per-metric endpoints. + """ + + metric_name: str + # WHY: dataset=None means the metric is combined across all datasets. + dataset: str | None + mean: float + ci_low: float + ci_high: float + n: int + + +class RunDetailDTO(BaseModel): + """Full run detail — metadata + aggregated metrics + cost summary. + + Teaches: composition over duplication — embeds RunMetadata directly + rather than copying its 10+ fields. The internal schema is already + JSON-serialisable via Pydantic, so nesting is zero-cost. + + Pipeline role: Response body for GET /api/eval/runs/{run_id}. + Results are paginated separately to keep this response bounded. + """ + + # PATTERN: Reuse internal RunMetadata directly — avoids field drift + # between the storage layer and the API layer. + metadata: RunMetadata + aggregated: list[AggregatedMetricDTO] + cost: dict[str, float] + # WHY: n_results tells the UI how many pages to expect without + # requiring it to load all results upfront. + n_results: int # results paginated separately + + +class EvalResultDTO(BaseModel): + """Slim per-question result for UI table rows. + + Teaches: projection pattern — retrieved_chunks and metric_details + are excluded here (they can be MBs per run) and are only returned + by the dedicated /runs/{id}/results/{qid} endpoint. + + Pipeline role: Element of the paginated results list response for + GET /api/eval/runs/{run_id}/results. + """ + + question_id: str + dataset: str + generated_answer: str + metrics: dict[str, float] + # WHY: error is None for successful evaluations; non-None means the + # pipeline raised an exception for this question. + error: str | None + + +class RunSubmitRequest(BaseModel): + """Request body for POST /api/eval/runs. + + Teaches: thin request model — only the config name is needed; all + other run parameters come from the config file itself. + + Pipeline role: Validated by FastAPI before reaching the route handler. + """ + + # WHY: config_name must match a file in configs/eval/ — validation + # of that constraint happens in the route handler, not here. + config_name: str # must match a file in configs/eval/ + + +class RunSubmitResponse(BaseModel): + """Response body for POST /api/eval/runs. + + Teaches: async job pattern — the run is queued immediately and the + caller polls /runs/{run_id}/status for progress. + + Pipeline role: Returned synchronously by the submit endpoint; the + run_id is the handle for all subsequent status and result queries. + """ + + run_id: str + status: Literal["queued", "running", "completed", "failed"] + + +class RunStatusDTO(BaseModel): + """Polling response for GET /api/eval/runs/{run_id}/status. + + Teaches: progress reporting pattern — progress (0.0–1.0) and + n_completed/n_total give the UI enough information to render a + progress bar without polling the full result list. + + Pipeline role: Returned by the status endpoint; polled by the UI + until status is "completed" or "failed". + """ + + run_id: str + status: Literal["queued", "running", "completed", "failed"] + # WHY: float 0.0–1.0 maps directly to a CSS/recharts progress bar. + progress: float # 0.0 - 1.0 + n_completed: int + n_total: int + # WHY: error_message is None unless status == "failed"; surfacing it + # here avoids a separate error endpoint for the common case. + error_message: str | None diff --git a/src/api/schemas/telemetry.py b/src/api/schemas/telemetry.py new file mode 100644 index 00000000..b7b57ba7 --- /dev/null +++ b/src/api/schemas/telemetry.py @@ -0,0 +1,26 @@ +""" +StageTelemetry — per-stage timing, token, and cost numbers returned to the +chat client alongside the answer. + +API Layer Position: + RAGBackend.query → returns (answer, sources, StageTelemetry) + /api/query response includes StageTelemetry as `telemetry` field + WebSocket emits a final `telemetry` event with this payload after `done` + +Frontend renders these as a muted footer under each assistant chat bubble: + "Retrieve 142ms · Generate 2.1s · 4,217 tok · $0.0083" +""" + +from __future__ import annotations + +from pydantic import BaseModel, Field + + +class StageTelemetry(BaseModel): + """Per-stage observability numbers for one chat turn.""" + + retrieve_ms: float = Field(ge=0.0) + generate_ms: float = Field(ge=0.0) + prompt_tokens: int = Field(ge=0) + completion_tokens: int = Field(ge=0) + cost_usd: float = Field(ge=0.0) diff --git a/src/api/services/__init__.py b/src/api/services/__init__.py new file mode 100644 index 00000000..38f1d63f --- /dev/null +++ b/src/api/services/__init__.py @@ -0,0 +1 @@ +"""Service layer for the RAG-QA API — stateful helpers used across route handlers.""" diff --git a/src/api/services/eval_runs.py b/src/api/services/eval_runs.py new file mode 100644 index 00000000..13b2de71 --- /dev/null +++ b/src/api/services/eval_runs.py @@ -0,0 +1,210 @@ +""" +In-process registry of in-flight eval runs. + +API Layer Position: + EvalRunner.run() (async dispatch) ──┐ + │ + POST /api/eval/run ──────────┼──→ [RunRegistry] ←─ GET /api/eval/runs/{id}/status + │ + on_progress callback ──────────┘ + +Design decisions: + - In-memory + threading.Lock — FastAPI's threadpool for sync route + handlers means concurrent reads/writes are real. Locked dict is + simple and sufficient for portfolio-scale traffic. + - Auto-evict completed runs after TTL so the registry doesn't grow + unbounded; runs are persisted to disk separately by storage.save_run + so eviction loses no information. + - Active runs (no completed_at) are NEVER evicted regardless of age — + a stuck run should be visible, not silently disappear. +""" + +from __future__ import annotations + +import threading +from dataclasses import dataclass, field +from datetime import datetime, timedelta, timezone +from typing import Literal + + +@dataclass +class RunStatus: + """Snapshot of a single eval run's lifecycle state. + + Concept taught: + Mutable dataclass as a simple value object — no ORM, no Pydantic, + just a plain struct the registry owns and callers can read. + + Pipeline position: + Lives inside RunRegistry._runs dict, keyed by run_id. Created at + register(), mutated by update_progress / mark_completed / mark_failed. + """ + + run_id: str + status: Literal["queued", "running", "completed", "failed"] + n_completed: int + n_total: int + error_message: str | None = None + # WHY: completed_at drives TTL-based eviction. Active runs keep this None + # so evict_old() can distinguish "still running" from "done long ago". + completed_at: datetime | None = None + + +class RunRegistry: + """Thread-safe in-process registry of eval run states. + + Concept taught: + The simplest possible shared-state solution for an async web server + that uses a threadpool for sync handlers. A single threading.Lock + serialises every read and write — no race conditions, no external + dependency, trivially correct at portfolio scale. + + Why threading.Lock over asyncio.Lock: + FastAPI runs sync route handlers in a threadpool (not the event loop), + so asyncio primitives don't protect against thread-level data races. + threading.Lock is the right tool here. + + Pipeline position: + Instantiated once at app startup (singleton on app.state). Shared by + POST /api/eval/run (write) and GET /api/eval/runs/{id}/status (read). + """ + + def __init__(self) -> None: + # PATTERN: Single lock guards the entire dict — simple and sufficient. + self._lock = threading.Lock() + self._runs: dict[str, RunStatus] = {} + + # ------------------------------------------------------------------ + # Write operations + # ------------------------------------------------------------------ + + def register(self, run_id: str, n_total: int) -> None: + """Create a new run entry in the queued state. + + Args: + run_id: Unique identifier for the run (caller's responsibility). + n_total: Total number of evaluation items to process. + """ + with self._lock: + self._runs[run_id] = RunStatus( + run_id=run_id, + status="queued", + n_completed=0, + n_total=n_total, + ) + + def update_progress(self, run_id: str, n_completed: int) -> None: + """Record incremental progress; transitions queued→running on first call. + + Only valid when the run is in queued or running state. Silently ignores + unknown run IDs to keep progress callbacks fire-and-forget safe. + + Args: + run_id: The run to update. + n_completed: Number of items completed so far. + """ + with self._lock: + entry = self._runs.get(run_id) + if entry is None or entry.status not in ("queued", "running"): + return + # WHY: First progress call transitions queued→running so callers + # can distinguish "not started" from "in progress". + entry.status = "running" + entry.n_completed = n_completed + + def mark_completed(self, run_id: str) -> None: + """Finalise a run as successfully completed. + + Sets n_completed = n_total and stamps completed_at with current UTC + time, enabling TTL-based eviction. + + Args: + run_id: The run to finalise. + """ + with self._lock: + entry = self._runs.get(run_id) + if entry is None: + return + entry.status = "completed" + entry.n_completed = entry.n_total + entry.completed_at = datetime.now(timezone.utc) + + def mark_failed(self, run_id: str, error: str) -> None: + """Record a run as failed with an error message. + + Also stamps completed_at so the TTL eviction logic treats failed runs + the same as completed ones — they don't need to live forever either. + + Args: + run_id: The run that failed. + error: Human-readable error description. + """ + with self._lock: + entry = self._runs.get(run_id) + if entry is None: + return + entry.status = "failed" + entry.error_message = error + entry.completed_at = datetime.now(timezone.utc) + + # ------------------------------------------------------------------ + # Read operations + # ------------------------------------------------------------------ + + def get(self, run_id: str) -> RunStatus | None: + """Return the RunStatus for a run, or None if not found. + + Args: + run_id: The run to look up. + + Returns: + The RunStatus object (mutable — callers may read fields directly) + or None if the run_id is unknown or was evicted. + """ + with self._lock: + return self._runs.get(run_id) + + def list_active(self) -> list[RunStatus]: + """Return all runs currently in queued or running state. + + Returns: + List of RunStatus objects for active runs; order is not guaranteed. + """ + with self._lock: + # WHY: snapshot under lock so the list is consistent even if + # another thread marks a run completed concurrently. + return [ + s for s in self._runs.values() + if s.status in ("queued", "running") + ] + + # ------------------------------------------------------------------ + # Maintenance + # ------------------------------------------------------------------ + + def evict_old(self, ttl_seconds: float = 3600.0) -> int: + """Remove completed/failed runs whose completed_at is older than ttl. + + Active runs (completed_at is None) are NEVER evicted, regardless of + how long they have been running. + + Args: + ttl_seconds: Maximum age in seconds for a completed/failed entry. + + Returns: + Number of entries removed from the registry. + """ + cutoff = datetime.now(timezone.utc) - timedelta(seconds=ttl_seconds) + to_evict: list[str] = [] + + with self._lock: + for run_id, entry in self._runs.items(): + # TRADE-OFF: We only evict entries that have a completed_at + # timestamp. A stuck/hung active run will never be evicted — + # it stays visible so operators can notice and investigate. + if entry.completed_at is not None and entry.completed_at < cutoff: + to_evict.append(run_id) + for run_id in to_evict: + del self._runs[run_id] + + return len(to_evict) diff --git a/src/backend.py b/src/backend.py index 604398f1..a6748e4d 100644 --- a/src/backend.py +++ b/src/backend.py @@ -32,6 +32,7 @@ import hashlib import logging import tempfile +import time import uuid from datetime import datetime, timezone from pathlib import Path @@ -40,6 +41,7 @@ from sqlalchemy import Engine from sqlmodel import Session, select +from .api.schemas.telemetry import StageTelemetry from .config import ( DEFAULT_MODEL, EVAL_MODEL, @@ -49,6 +51,8 @@ TOP_K_RESULTS, ) from .document_loader import DocumentLoader, TextChunker +from .eval._telemetry import count_tokens +from .eval.pricing import cost_usd from .evaluation import ( evaluate_answer_relevancy, evaluate_context_precision, @@ -59,6 +63,7 @@ from .models.document import DocumentRecord from .models.evaluation import MessageEvaluation from .models.message import Message, MessageSource +from .observability import get_tracer from .vector_store import ChromaVectorStore logger = logging.getLogger(__name__) @@ -298,6 +303,9 @@ def query( ) -> dict[str, Any]: """Run a full RAG query: retrieve chunks -> build context -> generate answer. + Thin wrapper over query_with_telemetry() that discards the StageTelemetry + so existing callers (tests, route layer) see no behaviour change. + Args: question: Natural language question from the user. top_k: Number of chunks to retrieve (default from config). @@ -306,19 +314,71 @@ def query( Returns: Dict with answer (str), sources (list[dict]), confidence (float). """ - k = top_k or TOP_K_RESULTS + result, _ = self.query_with_telemetry(question, top_k=top_k, model=model) + return result + + def query_with_telemetry( + self, + question: str, + top_k: int | None = None, + model: str | None = None, + ) -> tuple[dict[str, Any], StageTelemetry]: + """Run a full RAG query and return per-stage observability data. + + Identical to query() in output, but also returns a StageTelemetry + object with retrieve_ms, generate_ms, prompt_tokens, completion_tokens, + and cost_usd. The route layer (Task 5) calls this method so the REST + response can include telemetry without changing the query() contract. + + WHY a sibling instead of modifying query(): + Existing tests assert on result["answer"] and result["sources"] from + query(). Changing query() to return a tuple would break them silently + at dict-access time. The sibling keeps the tested contract intact. + + RAG Pipeline Position: + Question -> [RETRIEVE (traced)] -> [GENERATE (traced)] -> (answer + telemetry) - # WHY query_text: ChromaDB auto-embeds the question using the same - # embedding function that was used to embed the chunks. This - # ensures query and document embeddings live in the same space. - results = self.vector_store.query(query_text=question, top_k=k) + Args: + question: Natural language question from the user. + top_k: Number of chunks to retrieve (default from config). + model: LLM model override (creates a new handler if different). + + Returns: + Tuple of (result_dict, StageTelemetry). result_dict has the same + shape as query(): {answer, sources, confidence}. + """ + k = top_k or TOP_K_RESULTS + tracer = get_tracer() + + # ---- PHASE 1: Retrieval (timed + traced) ---------------------------- + # WHY: We open a span here rather than using @traced_stage because the + # retrieval call is a one-liner on self.vector_store — refactoring + # it to return (payload, attrs) would require an indirection wrapper + # that adds more lines than the inline approach. + t_retrieve_start = time.perf_counter() + with tracer.start_as_current_span("rag.retrieve") as retrieve_span: + retrieve_span.set_attribute("top_k", k) + retrieve_span.set_attribute("question_len", len(question)) + results = self.vector_store.query(query_text=question, top_k=k) + retrieve_span.set_attribute("results_count", len(results)) + retrieve_ms = (time.perf_counter() - t_retrieve_start) * 1000 if not results: - return { - "answer": "No documents indexed yet. Please upload documents first.", - "sources": [], - "confidence": 0.0, - } + # PATTERN: Early return with zero telemetry — no LLM call was made. + return ( + { + "answer": "No documents indexed yet. Please upload documents first.", + "sources": [], + "confidence": 0.0, + }, + StageTelemetry( + retrieve_ms=round(retrieve_ms, 2), + generate_ms=0.0, + prompt_tokens=0, + completion_tokens=0, + cost_usd=0.0, + ), + ) # Build context string from retrieved chunks context = "\n\n".join( @@ -331,7 +391,22 @@ def query( if model and model != self.llm.model: handler = LLMHandler(model=model) - answer = handler.generate_with_context(question, context) + # WHY reconstruct prompt strings here: generate_with_context() builds + # these internally before calling self.generate(). We reproduce them to + # count exactly the tokens that were billed, not a proxy string. + answer_system_prompt = ( + "You are a helpful assistant. Answer the user's question based solely on the " + "provided context. If the context does not contain enough information, say so." + ) + answer_user_prompt = f"Context:\n{context}\n\nQuestion: {question}\n\nAnswer:" + + # ---- PHASE 2: Generation (timed + traced) --------------------------- + t_generate_start = time.perf_counter() + with tracer.start_as_current_span("rag.generate") as generate_span: + generate_span.set_attribute("model", handler.model) + answer = handler.generate_with_context(question, context) + generate_span.set_attribute("answer_len", len(answer)) + generate_ms = (time.perf_counter() - t_generate_start) * 1000 sources = [ { @@ -346,16 +421,33 @@ def query( ] # PATTERN: Confidence = clamped average of top-3 similarity scores. - # This gives a rough signal of retrieval quality without - # requiring a separate calibration model. top_scores = [r.score for r in results[: min(3, len(results))]] confidence = max(0.0, min(1.0, sum(top_scores) / len(top_scores))) - return { - "answer": answer, - "sources": sources, - "confidence": round(confidence, 4), - } + # ---- Telemetry assembly --------------------------------------------- + # WHY count system_prompt + user_prompt together: together they form + # the full input that the provider billed as prompt tokens. + prompt_text = answer_system_prompt + "\n" + answer_user_prompt + prompt_tokens = count_tokens(prompt_text, handler.model) + completion_tokens = count_tokens(answer, handler.model) + total_cost = cost_usd(handler.model, prompt_tokens, completion_tokens) + + telemetry = StageTelemetry( + retrieve_ms=round(retrieve_ms, 2), + generate_ms=round(generate_ms, 2), + prompt_tokens=prompt_tokens, + completion_tokens=completion_tokens, + cost_usd=total_cost, + ) + + return ( + { + "answer": answer, + "sources": sources, + "confidence": round(confidence, 4), + }, + telemetry, + ) def stream_query( self, @@ -367,10 +459,15 @@ def stream_query( """Retrieve context and stream reasoning + answer with chain-of-thought events. Event stream shape (in order): - ("status", str) — programmatic retrieval milestones - ("reasoning", str) — LLM chain-of-thought tokens (brief pre-answer pass) - ("token", str) — final answer tokens - ("done", dict) — sources + persistence metadata + ("status", str) — programmatic retrieval milestones + ("reasoning", str) — LLM chain-of-thought tokens (brief pre-answer pass) + ("token", str) — final answer tokens + ("done", dict) — sources + persistence metadata [UNCHANGED SHAPE] + ("telemetry", dict) — StageTelemetry dict (NEW — additive, after done) + + WHY telemetry is last: Old consumers that handle status/reasoning/token/done + and ignore unknown event types continue to work. The ("telemetry", ...) event + is purely additive — no existing event shape is modified. WHY two LLM calls: The reasoning pass uses a focused prompt that asks the model to think out loud about the retrieved context BEFORE giving @@ -381,19 +478,41 @@ def stream_query( for UX, not a durable artifact of the conversation. Persisting it would double storage and confuse the sliding-window context. + WHY telemetry covers the answer pass only (not reasoning): + The reasoning pass uses a separate model (REASONING_MODEL) with its own + cost. Telemetry here tracks the user-visible answer generation; mixing + two model costs into one StageTelemetry would confuse the "per-query + cost" display. Reasoning cost is a separate concern. + Yields: Tuples as described above. """ k = top_k or TOP_K_RESULTS + tracer = get_tracer() - # ---- PHASE 0: Retrieval ------------------------------------------------ + # ---- PHASE 0: Retrieval (timed + traced) -------------------------------- yield ("status", "Searching indexed documents...") - results = self.vector_store.query(query_text=question, top_k=k) + t_retrieve_start = time.perf_counter() + with tracer.start_as_current_span("rag.retrieve") as retrieve_span: + retrieve_span.set_attribute("top_k", k) + retrieve_span.set_attribute("question_len", len(question)) + results = self.vector_store.query(query_text=question, top_k=k) + retrieve_span.set_attribute("results_count", len(results)) + retrieve_ms = (time.perf_counter() - t_retrieve_start) * 1000 if not results: yield ("status", "No indexed documents — nothing to retrieve.") yield ("token", "No documents indexed yet. Please upload documents first.") yield ("done", {"sources": []}) + # PATTERN: Emit zero telemetry even on early return so the route + # layer always gets a telemetry event it can forward. + yield ("telemetry", StageTelemetry( + retrieve_ms=round(retrieve_ms, 2), + generate_ms=0.0, + prompt_tokens=0, + completion_tokens=0, + cost_usd=0.0, + ).model_dump()) return # WHY: Summarise retrieval in one status line so the user can see which @@ -477,7 +596,7 @@ def stream_query( logger.warning("Reasoning pass failed: %s", exc) yield ("status", "Reasoning unavailable — skipping to answer.") - # ---- PHASE 2: Answer pass ---------------------------------------------- + # ---- PHASE 2: Answer pass (timed + traced) ----------------------------- yield ("status", "Composing answer...") system_prompt = ( @@ -497,8 +616,10 @@ def stream_query( ) user_prompt = f"Context:\n{context}\n\nQuestion: {question}\n\nAnswer:" - # Accumulate full response for persistence - full_response = [] + # Accumulate full response for persistence and token counting + full_response: list[str] = [] + + t_generate_start = time.perf_counter() if conversation_id: # PHASE 1: Save user message BEFORE streaming @@ -515,9 +636,15 @@ def stream_query( messages.append({"role": "user", "content": user_prompt}) # PHASE 4: Stream via messages API (multi-turn aware) - for token in handler.stream_messages(messages): - full_response.append(token) - yield ("token", token) + with tracer.start_as_current_span("rag.generate") as generate_span: + generate_span.set_attribute("model", handler.model) + generate_span.set_attribute("has_conversation", True) + for token in handler.stream_messages(messages): + full_response.append(token) + yield ("token", token) + generate_span.set_attribute("answer_len", sum(len(t) for t in full_response)) + + generate_ms = (time.perf_counter() - t_generate_start) * 1000 # PHASE 5: Save assistant message + sources assistant_content = "".join(full_response) @@ -538,14 +665,44 @@ def stream_query( "conversation_id": conversation_id, }) + # WHY count full messages list for prompt: the multi-turn path sends + # the sliding window + system prompt + user turn as one request. + # We join all message content to estimate the billed input tokens. + prompt_text = "\n".join(m["content"] for m in messages) + else: # No conversation — simple single-turn streaming - for token in handler.stream_response(user_prompt, system_prompt=system_prompt): - full_response.append(token) - yield ("token", token) + with tracer.start_as_current_span("rag.generate") as generate_span: + generate_span.set_attribute("model", handler.model) + generate_span.set_attribute("has_conversation", False) + for token in handler.stream_response(user_prompt, system_prompt=system_prompt): + full_response.append(token) + yield ("token", token) + generate_span.set_attribute("answer_len", sum(len(t) for t in full_response)) + + generate_ms = (time.perf_counter() - t_generate_start) * 1000 yield ("done", {"sources": sources}) + prompt_text = system_prompt + "\n" + user_prompt + + # ---- Telemetry assembly (after done, additive) ----------------------- + # WHY after done: the done event is what the client waits for to show + # sources. Telemetry is a secondary signal — emit it last so done's + # latency is not affected by token-counting arithmetic. + answer_text = "".join(full_response) + prompt_tokens = count_tokens(prompt_text, handler.model) + completion_tokens = count_tokens(answer_text, handler.model) + total_cost = cost_usd(handler.model, prompt_tokens, completion_tokens) + + yield ("telemetry", StageTelemetry( + retrieve_ms=round(retrieve_ms, 2), + generate_ms=round(generate_ms, 2), + prompt_tokens=prompt_tokens, + completion_tokens=completion_tokens, + cost_usd=total_cost, + ).model_dump()) + # ------------------------------------------------------------------ # # Document management # # ------------------------------------------------------------------ # diff --git a/src/eval/__init__.py b/src/eval/__init__.py new file mode 100644 index 00000000..ba197058 --- /dev/null +++ b/src/eval/__init__.py @@ -0,0 +1,47 @@ +"""RAG eval harness — schemas, metrics, statistics, datasets, runner, storage, compare. + +See docs/superpowers/specs/2026-04-26-rag-eval-harness-phase-1-design.md +for the full design. This package exposes the pure-Python foundation +plus the runner/storage/compare layer; the API and frontend live in +separate modules introduced by Sub-plan 1C. +""" + +from __future__ import annotations + +from src.eval.compare import compare_runs +from src.eval.config import EvalConfig, load_config +from src.eval.pricing import MODEL_PRICES, ModelPrice, cost_usd +from src.eval.runner import EvalRunner +from src.eval.schemas import ( + AggregatedMetric, + CompareResult, + EvalQuestion, + EvalResult, + MetricDelta, + RunMetadata, +) +from src.eval.statistics import bootstrap_ci, paired_permutation_test +from src.eval.storage import list_runs, load_run, save_run + +__all__ = [ + # 1A — schemas, pricing, statistics + "AggregatedMetric", + "CompareResult", + "EvalQuestion", + "EvalResult", + "MetricDelta", + "MODEL_PRICES", + "ModelPrice", + "RunMetadata", + "bootstrap_ci", + "cost_usd", + "paired_permutation_test", + # 1B — config, runner, storage, compare + "EvalConfig", + "EvalRunner", + "compare_runs", + "list_runs", + "load_config", + "load_run", + "save_run", +] diff --git a/src/eval/_telemetry.py b/src/eval/_telemetry.py new file mode 100644 index 00000000..cde5f6f3 --- /dev/null +++ b/src/eval/_telemetry.py @@ -0,0 +1,60 @@ +"""Internal telemetry helpers for the eval harness. + +This module provides token counting and timing utilities used by +EvalPipeline.query() to track cost and latency. It's kept separate from +pipeline_factory.py to keep that file under the project's 250-line ceiling. + +Token Counting: + Counting tokens is essential for eval cost tracking. This module tries + tiktoken first (exact model-aware tokenization) and falls back to a + word-count heuristic if tiktoken doesn't know the model. The fallback + ensures eval doesn't hard-fail on new model releases. +""" + +from __future__ import annotations + +import logging + +logger = logging.getLogger(__name__) + +# WHY: one-time warning flag for tiktoken fallback — we don't want the +# warning to spam on every token-count call throughout an eval run. +_tiktoken_warned = False + +try: + import tiktoken as _tiktoken # type: ignore +except ImportError: + _tiktoken = None # type: ignore + + +def count_tokens(text: str, model: str) -> int: + """Count tokens in text, falling back to word-count * 1.3 if tiktoken fails. + + WHY the fallback: tiktoken doesn't know every model (new OpenAI releases + ship before tiktoken is updated). Eval should not hard-fail on a missing + tokenizer — a ±30% estimate is fine for cost/latency tracking. + + Args: + text: The text to count tokens for. + model: Model name used to select the tiktoken encoding. + + Returns: + Estimated token count (int). + """ + global _tiktoken_warned + if _tiktoken is not None: + try: + enc = _tiktoken.encoding_for_model(model) + return len(enc.encode(text)) + except Exception: + # Unknown model for tiktoken — fall through to word estimate + pass + + if not _tiktoken_warned: + logger.warning( + "tiktoken not installed or model %r unknown — " + "using word-count × 1.3 for token estimates.", + model, + ) + _tiktoken_warned = True + return int(len(text.split()) * 1.3) diff --git a/src/eval/aggregator.py b/src/eval/aggregator.py new file mode 100644 index 00000000..0d89e491 --- /dev/null +++ b/src/eval/aggregator.py @@ -0,0 +1,118 @@ +""" +Metric aggregator — turns per-question EvalResult rows into AggregatedMetric +rows with bootstrap confidence intervals. + +Eval Harness Position: + list[EvalResult] → [AGGREGATOR] → list[AggregatedMetric] (per-dataset + combined) + → list[warnings] (skipped low-N combos) + +Design decisions: + - Per-dataset rows let the UI compare squad_v2 vs ml_papers performance. + - Combined rows give a single headline number across all datasets. + - Skip combos with <3 samples — bootstrap CIs are meaningless on n<3. + - Use config.eval.bootstrap_n and config.eval.seed so all runs with the + same config yield reproducible CIs. +""" + +from __future__ import annotations + +from collections import defaultdict +from typing import Any + +from src.eval.config import EvalConfig +from src.eval.schemas import AggregatedMetric, EvalResult +from src.eval.statistics import bootstrap_ci + +MIN_SAMPLES = 3 + + +def aggregate( + results: list[EvalResult], + config: EvalConfig, +) -> tuple[list[AggregatedMetric], list[str]]: + """For every (metric_name, dataset) combo present in results, compute + bootstrap CIs and return list[AggregatedMetric] + list[warnings]. + + Behavior: + - Per-dataset rows: emit one AggregatedMetric per (metric, dataset). + - Combined row: emit one AggregatedMetric per metric with dataset=None. + - Skip metric/dataset combos with <3 non-NaN samples; append a + warning string (e.g. "Skipped recall_at_5 on ml_papers_v1: only 2 samples") + to the warnings list. + - NaN values are dropped per metric per question by bootstrap_ci. + - Errored results (r.error is not None) are excluded. + + Args: + results: Per-question evaluation results from the runner. + config: Eval run configuration (provides bootstrap_n and seed). + + Returns: + A two-tuple of (aggregated_metrics, warnings) where aggregated_metrics + is a list of AggregatedMetric (per-dataset and combined) and warnings + is a list of strings describing skipped low-N combos. + """ + # WHY: Filter errored results first so downstream grouping never sees them. + # An errored result has undefined metric values — including it would + # silently skew aggregates if the metrics dict happens to be non-empty. + valid = [r for r in results if r.error is None] + + # PATTERN: Collect raw score lists keyed by (metric_name, dataset). + # defaultdict(list) avoids repeated "if key not in" guards. + per_dataset: dict[tuple[str, str], list[float]] = defaultdict(list) + + for r in valid: + for metric_name, score in r.metrics.items(): + per_dataset[(metric_name, r.dataset)].append(score) + + # Build the combined (across all datasets) view keyed by metric_name alone. + # WHY: Combine after grouping so we don't double-count results — iterate + # the already-filtered per_dataset dict rather than valid again. + combined: dict[str, list[float]] = defaultdict(list) + for (metric_name, _dataset), scores in per_dataset.items(): + combined[metric_name].extend(scores) + + aggregated: list[AggregatedMetric] = [] + warnings: list[str] = [] + + bootstrap_n = config.eval.bootstrap_n + seed = config.eval.seed + + # --- Per-dataset rows --- + for (metric_name, dataset), scores in per_dataset.items(): + n = len(scores) + if n < MIN_SAMPLES: + warnings.append( + f"Skipped {metric_name} on {dataset}: only {n} samples" + ) + continue + + mean, ci_low, ci_high = bootstrap_ci(scores, n_resamples=bootstrap_n, seed=seed) + aggregated.append(AggregatedMetric( + metric_name=metric_name, + dataset=dataset, + mean=mean, + ci_low=ci_low, + ci_high=ci_high, + n=n, + )) + + # --- Combined rows (dataset=None) --- + for metric_name, scores in combined.items(): + n = len(scores) + if n < MIN_SAMPLES: + warnings.append( + f"Skipped {metric_name} combined: only {n} samples" + ) + continue + + mean, ci_low, ci_high = bootstrap_ci(scores, n_resamples=bootstrap_n, seed=seed) + aggregated.append(AggregatedMetric( + metric_name=metric_name, + dataset=None, + mean=mean, + ci_low=ci_low, + ci_high=ci_high, + n=n, + )) + + return aggregated, warnings diff --git a/src/eval/cli.py b/src/eval/cli.py new file mode 100644 index 00000000..adf56185 --- /dev/null +++ b/src/eval/cli.py @@ -0,0 +1,304 @@ +""" +CLI for the RAG eval harness. + +Usage: + python -m src.eval.cli run --config configs/eval/baseline.yaml + python -m src.eval.cli list + python -m src.eval.cli show [--html] + python -m src.eval.cli compare [--html] + +Environment variables honored: + EVAL_RUNS_DIR — override default runs directory. + EVAL_LLM_OVERRIDE_DUMMY — if "1", inject a dummy LLM (test path). + EVAL_SQUAD_PATH — override the SQuAD frozen-set path (test path). + +Design decisions: + - argparse over click — no extra dep, sufficient for 4 subcommands. + - Each subcommand returns an int exit code; main() returns it for + SystemExit. Makes subprocess testing trivial (assert returncode == 0). +""" + +from __future__ import annotations + +import argparse +import logging +import os +from pathlib import Path +from typing import Any + +logger = logging.getLogger(__name__) + + +# --------------------------------------------------------------------------- # +# DummyLLM — test-only, gated behind EVAL_LLM_OVERRIDE_DUMMY=1 # +# --------------------------------------------------------------------------- # + +class _DummyLLM: + """Returns canned data for any prompt — used only when EVAL_LLM_OVERRIDE_DUMMY=1.""" + def generate(self, prompt: str, system_prompt: str | None = None) -> str: + if "JSON" in (system_prompt or "") or '"score"' in prompt: + return ('{"score": 1.0, "claims": [], "chunks": [], ' + '"factual_match": 1.0, "is_refusal": false, "reasoning": "ok"}') + return "" + + +# --------------------------------------------------------------------------- # +# Subcommand handlers # +# --------------------------------------------------------------------------- # + +def _cmd_run(args: argparse.Namespace) -> int: + """Load config, run EvalRunner, print run_id and one-line summary.""" + # PATTERN: Patch module attribute before constructing EvalRunner so that + # runner._load_questions picks up the override via the live attr read. + if env_squad := os.getenv("EVAL_SQUAD_PATH"): + from src.eval.datasets import squad_v2 as squad_ds + squad_ds.DEFAULT_OUTPUT_PATH = Path(env_squad) + import src.eval.storage as _storage + _storage.EVAL_RUNS_DIR = Path(os.getenv("EVAL_RUNS_DIR", "eval_runs")) + + from src.eval.config import load_config + from src.eval.runner import EvalRunner + try: + config = load_config(args.config) + except (FileNotFoundError, Exception) as exc: + print(f"Error loading config: {exc}") + return 1 + llm_override = None + judge_llm_override = None + if os.getenv("EVAL_LLM_OVERRIDE_DUMMY") == "1": + dummy = _DummyLLM() + llm_override = dummy + judge_llm_override = dummy + runner = EvalRunner( + config, + config_path=args.config, + llm_override=llm_override, + judge_llm_override=judge_llm_override, + ) + + try: + metadata = runner.run() + except Exception as exc: + print(f"Run failed: {exc}") + logger.exception("Run failed") + return 1 + + # WHY last-token placement: test extracts run_id as the last whitespace-token + # on the line containing both "cli-test" and "_". The bare run_id must be + # the final token — no "key=value" wrapper around it. + print(f"Run complete: {metadata.config_name} n={metadata.n_questions}" + f" errors={metadata.n_errors} {metadata.run_id}") + return 0 + + +def _cmd_list(args: argparse.Namespace) -> int: + """Print a table of all eval runs.""" + import src.eval.storage as _storage + _storage.EVAL_RUNS_DIR = Path(os.getenv("EVAL_RUNS_DIR", "eval_runs")) + + runs = _storage.list_runs() + + if not runs: + print("No runs found.") + return 0 + + # Table header + hdr = f"{'run_id':<45} {'config':<15} {'started':<20} {'n_q':>5} {'err':>5}" + print(hdr) + print("-" * len(hdr)) + + for meta in runs: + ts = meta.started_at.strftime("%Y-%m-%d %H:%M:%S") + print( + f"{meta.run_id:<45} {meta.config_name:<15} {ts:<20}" + f" {meta.n_questions:>5} {meta.n_errors:>5}" + ) + + return 0 + + +def _cmd_show(args: argparse.Namespace) -> int: + """Print aggregated metrics; optionally write report.html.""" + import src.eval.storage as _storage + _storage.EVAL_RUNS_DIR = Path(os.getenv("EVAL_RUNS_DIR", "eval_runs")) + + try: + run = _storage.load_run(args.run_id) + except FileNotFoundError as exc: + print(f"Error: {exc}") + return 1 + + meta = run["metadata"] + print(f"Run: {meta.run_id}") + print(f"Config: {meta.config_name}") + print(f"Started: {meta.started_at.strftime('%Y-%m-%d %H:%M:%S')}") + print(f"N questions: {meta.n_questions} errors: {meta.n_errors}") + print() + + aggregated = run["aggregated"] + if aggregated: + print(f"{'metric':<35} {'dataset':<20} {'mean':>8} {'ci_low':>8} {'ci_high':>8}") + print("-" * 82) + for agg in sorted(aggregated, key=lambda a: (a.metric_name, a.dataset or "")): + ds = agg.dataset or "(all)" + print( + f"{agg.metric_name:<35} {ds:<20}" + f" {agg.mean:>8.4f} {agg.ci_low:>8.4f} {agg.ci_high:>8.4f}" + ) + else: + print("No aggregated metrics found.") + + if args.html: + from src.eval.report import render_run_html + html = render_run_html(run) + html_path = _storage.EVAL_RUNS_DIR / args.run_id / "report.html" + html_path.write_text(html) + print(f"\nHTML report written to: {html_path}") + + return 0 + + +def _cmd_compare(args: argparse.Namespace) -> int: + """Print delta table for two runs; optionally write compare HTML.""" + import src.eval.storage as _storage + _storage.EVAL_RUNS_DIR = Path(os.getenv("EVAL_RUNS_DIR", "eval_runs")) + + from src.eval.compare import compare_runs + + try: + result = compare_runs(args.id_a, args.id_b) + except FileNotFoundError as exc: + print(f"Error: {exc}") + return 1 + except ValueError as exc: + # TRADE-OFF: mismatch in eval_set_versions is surfaced as a clear error, + # not silently ignored — comparing different question pools is meaningless. + print(f"Error: {exc}") + return 1 + + print(f"Comparing A: {args.id_a}") + print(f" vs B: {args.id_b}") + print() + + if result.deltas: + hdr = f"{'metric':<35} {'dataset':<20} {'a_mean':>8} {'b_mean':>8} {'delta':>8} {'p':>8}" + print(hdr) + print("-" * len(hdr)) + for d in sorted(result.deltas, key=lambda x: (x.metric_name, x.dataset or "")): + ds = d.dataset or "(all)" + sig = "*" if d.significant else " " + print( + f"{d.metric_name:<35} {ds:<20}" + f" {d.a_mean:>8.4f} {d.b_mean:>8.4f} {d.delta:>+8.4f} {d.p_value:>8.4f}{sig}" + ) + else: + # WHY still print something: compare test checks for "recall" OR "delta" + # in stdout. With <3 paired questions the permutation test is skipped. + print("No metric deltas computed (insufficient paired questions).") + # Print the metrics that were attempted so the output is informative. + try: + run_a = _storage.load_run(args.id_a) + agg_names = {a.metric_name for a in run_a["aggregated"]} + if agg_names: + print("Metrics in run A: " + ", ".join(sorted(agg_names))) + except Exception: + pass + + if args.html: + from src.eval.report import render_compare_html + html = render_compare_html(result) + html_path = _storage.EVAL_RUNS_DIR / f"compare_{args.id_a}_{args.id_b}.html" + html_path.write_text(html) + print(f"\nHTML comparison written to: {html_path}") + + return 0 + + +def _cmd_archive(args: argparse.Namespace) -> int: + """Copy small artifacts of a run from eval_runs/ to a tracked location. + + Files copied: metrics.json, cost.json, metadata.json, config.yaml. + NOT copied: questions.jsonl (large). Its SHA-256 is recorded in metadata.json + under `questions_jsonl_sha256` so reviewers can verify against a re-run. + + Args: + args: Namespace with run_id, to (destination directory), runs_root. + + Returns: + 0 on success, 1 if the run directory is not found. + """ + import hashlib + import json + import shutil + + runs_root = Path(getattr(args, "runs_root", None) or "eval_runs") + src = runs_root / args.run_id + if not src.exists(): + print(f"Run not found: {src}") + return 1 + + dst = Path(args.to) + dst.mkdir(parents=True, exist_ok=True) + + for name in ("metrics.json", "cost.json", "config.yaml"): + if (src / name).exists(): + shutil.copy2(src / name, dst / name) + + # Record questions.jsonl SHA in metadata.json before copying it. + metadata = json.loads((src / "metadata.json").read_text()) + questions_path = src / "questions.jsonl" + if questions_path.exists(): + h = hashlib.sha256() + h.update(questions_path.read_bytes()) + metadata["questions_jsonl_sha256"] = h.hexdigest() + (dst / "metadata.json").write_text(json.dumps(metadata, indent=2)) + + print(f"Archived run {args.run_id} → {dst}") + return 0 + + +# --------------------------------------------------------------------------- # +# Entry point # +# --------------------------------------------------------------------------- # + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(prog="src.eval.cli") + sub = parser.add_subparsers(dest="cmd", required=True) + + p_run = sub.add_parser("run", help="Run an eval from a YAML config") + p_run.add_argument("--config", required=True, type=Path) + + sub.add_parser("list", help="List all eval runs") + + p_show = sub.add_parser("show", help="Show aggregated metrics for a run") + p_show.add_argument("run_id") + p_show.add_argument("--html", action="store_true") + + p_compare = sub.add_parser("compare", help="Compare two runs") + p_compare.add_argument("id_a") + p_compare.add_argument("id_b") + p_compare.add_argument("--html", action="store_true") + + p_archive = sub.add_parser( + "archive", + help="Copy small run artifacts (metrics/cost/metadata/config) to a tracked path.", + ) + p_archive.add_argument("run_id", help="Run id to archive (must exist under runs_root).") + p_archive.add_argument("--to", required=True, help="Destination directory.") + p_archive.add_argument( + "--runs-root", default="eval_runs", dest="runs_root", + help="Root directory holding run subdirectories (default: eval_runs).", + ) + + args = parser.parse_args(argv) + return { + "run": _cmd_run, + "list": _cmd_list, + "show": _cmd_show, + "compare": _cmd_compare, + "archive": _cmd_archive, + }[args.cmd](args) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/eval/compare.py b/src/eval/compare.py new file mode 100644 index 00000000..a93e2bb2 --- /dev/null +++ b/src/eval/compare.py @@ -0,0 +1,237 @@ +""" +Two-run comparison — diff aggregated metrics between two eval runs with +paired significance tests, plus per-question regressions/wins. + +Eval Harness Position: + load_run(A) ─┐ + ├─→ [COMPARE] → CompareResult (deltas + per_question_diff) + load_run(B) ─┘ + +Design decisions: + - Validate eval_set_versions match before comparing — different versions + mean different questions, and a delta is meaningless across them. + - Paired permutation test (vs unpaired t-test) because the same question + appears in both runs; the pairing reduces noise from question variance. + - Headline metric chosen as recall_at_5 if present, else alphabetic first + — predictable and reproducible per_question_diff selection. +""" + +from __future__ import annotations + +import logging +from collections import defaultdict +from typing import Any + +from src.eval.schemas import ( + AggregatedMetric, + CompareResult, + EvalResult, + MetricDelta, + RunMetadata, +) +from src.eval.statistics import paired_permutation_test +from src.eval.storage import load_run + +logger = logging.getLogger(__name__) + + +def _score_index( + results: list[EvalResult], +) -> dict[tuple[str, str, str], float]: + """Build a lookup table from (question_id, dataset, metric) → score. + + WHY: Flat dict keyed by 3-tuple lets us O(1)-look up any specific + (question, dataset, metric) combo when pairing across runs A and B. + + Args: + results: Per-question eval outputs for one run. + + Returns: + Dict mapping (question_id, dataset, metric_name) to float score. + """ + index: dict[tuple[str, str, str], float] = {} + for r in results: + for metric, score in r.metrics.items(): + index[(r.question_id, r.dataset, metric)] = score + return index + + +def _agg_lookup( + aggregated: list[AggregatedMetric], +) -> dict[tuple[str, str | None], AggregatedMetric]: + """Index aggregated metrics by (metric_name, dataset). + + Args: + aggregated: Aggregated metric list from a run. + + Returns: + Dict for O(1) lookup by (metric_name, dataset). + """ + return {(a.metric_name, a.dataset): a for a in aggregated} + + +def _paired_values( + scores_a: dict[tuple[str, str, str], float], + scores_b: dict[tuple[str, str, str], float], + metric: str, + dataset: str, +) -> tuple[list[float], list[float], list[str]]: + """Extract paired per-question scores for a given (metric, dataset). + + Only includes question IDs that appear in both runs with non-NaN scores. + + Args: + scores_a: Score index for run A. + scores_b: Score index for run B. + metric: Metric name to filter on. + dataset: Dataset name to filter on. + + Returns: + Three-tuple (a_values, b_values, question_ids) where all three lists + are aligned by position. + """ + # Gather question_ids that have a score in run A for this (metric, dataset). + candidates = { + qid + for (qid, ds, m) in scores_a + if ds == dataset and m == metric + } + a_vals: list[float] = [] + b_vals: list[float] = [] + qids: list[str] = [] + for qid in sorted(candidates): # sorted for determinism + a_score = scores_a.get((qid, dataset, metric)) + b_score = scores_b.get((qid, dataset, metric)) + if a_score is None or b_score is None: + continue + import math + if math.isnan(a_score) or math.isnan(b_score): + continue + a_vals.append(a_score) + b_vals.append(b_score) + qids.append(qid) + return a_vals, b_vals, qids + + +def compare_runs(id_a: str, id_b: str) -> CompareResult: + """Load two runs, compute per-metric deltas with paired permutation + significance tests, and pick the top per-question diffs. + + Raises: + FileNotFoundError: If either run is missing. + ValueError: If the two runs used different eval_set_versions. + """ + # WHY: We import load_run at call time (not module level) so that + # monkeypatched EVAL_RUNS_DIR is resolved after the test fixture reloads + # src.eval.storage. The import at module level is fine because the function + # itself references EVAL_RUNS_DIR only at call time inside storage.py. + import src.eval.storage as _storage + + run_a = _storage.load_run(id_a) + run_b = _storage.load_run(id_b) + + meta_a: RunMetadata = run_a["metadata"] + meta_b: RunMetadata = run_b["metadata"] + + # Step 2: Validate eval_set_versions match. + # TRADE-OFF: Strict equality — even a subset mismatch means the question + # pools differ, making per-question deltas misleading. + if meta_a.eval_set_versions != meta_b.eval_set_versions: + raise ValueError("eval set version mismatch between runs") + + results_a: list[EvalResult] = run_a["results"] + results_b: list[EvalResult] = run_b["results"] + agg_a: list[AggregatedMetric] = run_a["aggregated"] + agg_b: list[AggregatedMetric] = run_b["aggregated"] + + # Step 3: Build per-question score indexes. + scores_a = _score_index(results_a) + scores_b = _score_index(results_b) + + # Step 4: Index aggregated metrics and find combos present in both runs. + lookup_a = _agg_lookup(agg_a) + lookup_b = _agg_lookup(agg_b) + shared_combos = set(lookup_a.keys()) & set(lookup_b.keys()) + + deltas: list[MetricDelta] = [] + + # WHY: sort key coerces None dataset to "" so None and str are comparable. + for metric_name, dataset in sorted(shared_combos, key=lambda t: (t[0], t[1] or "")): + # dataset=None entries represent cross-dataset rollups; per-question + # scores always carry a real dataset name so we can't pair them. + # Skip None-dataset combos: there are no EvalResult rows with dataset=None. + if dataset is None: + logger.debug( + "Skipping dataset=None combo for metric '%s' (no per-question pairing possible)", + metric_name, + ) + continue + + a_vals, b_vals, qids = _paired_values(scores_a, scores_b, metric_name, dataset) + + if len(a_vals) < 3: + logger.debug( + "Skipping (%s, %s): only %d paired questions (need ≥ 3)", + metric_name, + dataset, + len(a_vals), + ) + continue + + delta, p_value = paired_permutation_test(a_vals, b_vals, n_resamples=10000, seed=42) + significant = p_value < 0.05 + + agg_entry_a = lookup_a[(metric_name, dataset)] + agg_entry_b = lookup_b[(metric_name, dataset)] + + deltas.append( + MetricDelta( + metric_name=metric_name, + dataset=dataset, + a_mean=agg_entry_a.mean, + a_ci=(agg_entry_a.ci_low, agg_entry_a.ci_high), + b_mean=agg_entry_b.mean, + b_ci=(agg_entry_b.ci_low, agg_entry_b.ci_high), + delta=delta, + p_value=p_value, + significant=significant, + ) + ) + + # Step 5: Pick headline metric for per-question diff. + # PATTERN: Prefer recall_at_5 for consistency across evals; + # fall back to alphabetically-first metric for reproducibility. + all_metrics = sorted({m for (m, _) in shared_combos}) + headline = "recall_at_5" if "recall_at_5" in all_metrics else (all_metrics[0] if all_metrics else None) + + # Step 6: Compute per-question diffs for the headline metric. + # Collect across all real datasets (exclude None) where headline metric appears. + per_question_rows: list[dict[str, Any]] = [] + + if headline is not None: + # Gather all (dataset) combos that use the headline metric (excluding None). + headline_datasets = sorted( + {ds for (m, ds) in shared_combos if m == headline and ds is not None} + ) + for ds in headline_datasets: + a_vals, b_vals, qids = _paired_values(scores_a, scores_b, headline, ds) + for qid, a_score, b_score in zip(qids, a_vals, b_vals): + raw_delta = b_score - a_score + per_question_rows.append({ + "question_id": qid, + "dataset": ds, + "a_score": a_score, + "b_score": b_score, + "delta": raw_delta, + }) + + # Sort by absolute delta descending, then cap at top 10. + per_question_rows.sort(key=lambda row: abs(row["delta"]), reverse=True) + top10 = per_question_rows[:10] + + return CompareResult( + run_a=meta_a, + run_b=meta_b, + deltas=deltas, + per_question_diff=top10, + ) diff --git a/src/eval/config.py b/src/eval/config.py new file mode 100644 index 00000000..e9ebd0f0 --- /dev/null +++ b/src/eval/config.py @@ -0,0 +1,203 @@ +""" +EvalConfig — typed pipeline-and-eval configuration loaded from YAML. + +Eval Harness Position: + configs/eval/*.yaml → load_config() → EvalConfig → EvalRunner.run() + +Design decisions: + - Pydantic v2 with nested models (ChunkerCfg, RetrieverCfg, etc.) so + each subsystem owns its own validation surface. + - Literal["..."] on dataset names and chunker strategy gives schema- + level rejection of typos before the runner spins up. + - YAML over JSON for human authorability — eval configs are written + by hand, not generated. +""" + +from __future__ import annotations + +from pathlib import Path +from typing import Literal + +import yaml +from pydantic import BaseModel, ConfigDict, Field + + +class ChunkerCfg(BaseModel): + """Chunking strategy configuration. + + Teaches: how chunking parameters propagate from config to the pipeline. + Why Literal: rejects typos ("recusive") at validation time, not at runtime. + Pipeline position: INDEXING step — Document → [Chunker] → Embeddings. + """ + + strategy: Literal["fixed", "recursive", "semantic"] = "recursive" + chunk_size: int = 512 + chunk_overlap: int = 64 + + +class RetrieverCfg(BaseModel): + """Retriever configuration. + + Teaches: separating retrieval hyperparameters from the retriever implementation. + Pipeline position: QUERYING step — Embeddings → [Retriever] → Top-K chunks. + """ + + top_k: int = 5 + + +class GeneratorCfg(BaseModel): + """LLM generator configuration. + + Teaches: optional reasoning model pattern — some providers support a + lightweight "thinking" model before the final answer model. + Pipeline position: QUERYING step — Chunks → [Generator] → Answer. + """ + + model: str = "gpt-5-mini" + # WHY: reasoning_model is optional — None disables the CoT pre-pass. + reasoning_model: str | None = "gpt-4.1-nano" + + +class EmbedderCfg(BaseModel): + """Embedder selection. None/default = ChromaDB built-in ONNX (current behavior). + + Phase 2 lever 2b: swap ChromaDB's default MiniLM (384-dim) for BAAI/bge-small-en-v1.5 + (also 384-dim) which is domain-tuned for retrieval. The factory wires this as a + Chroma EmbeddingFunction at collection-creation time, so ChromaVectorStore.upsert/query + auto-embed without per-call code changes. + """ + + model_config = ConfigDict(extra="forbid") + name: Literal["chroma_default", "bge_small_en_v1_5"] = "chroma_default" + + +class HybridCfg(BaseModel): + """BM25 + dense retrieval with Reciprocal Rank Fusion. + + Phase 2 lever 2c: combine sparse (BM25) and dense (vector) signal. Disabled by default; + when enabled, the retriever fetches top-N candidates from each side and fuses them + with RRF: score(d) = sum over r in {dense, sparse} of 1 / (rrf_k + rank_r(d)). + """ + + model_config = ConfigDict(extra="forbid") + enabled: bool = False + bm25_top_k: int = 20 + dense_top_k: int = 20 + rrf_k: int = 60 + + +class RerankerCfg(BaseModel): + """Cross-encoder rerank top-N → final-K. None = no rerank (current behavior). + + Phase 2 lever 2d: improve precision by re-scoring the top-N retrieved candidates + with a dedicated relevance model (ms-marco-MiniLM-L-6-v2). Adds latency but no + LLM cost. + """ + + model_config = ConfigDict(extra="forbid") + model: Literal["ms_marco_minilm_l6_v2"] | None = None + rerank_top_n: int = 20 + final_top_k: int = 5 + + +class QueryRewriterCfg(BaseModel): + """LLM-based query expansion. None = no rewrite (current behavior). + + Phase 2 lever 2e: ask an LLM to produce up to N alternative phrasings of the user + query, retrieve against each, then deduplicate. Costs one LLM call per question; + captured in the cost ledger under the 'rewriter' bucket. + """ + + model_config = ConfigDict(extra="forbid") + model: str | None = None + max_expansions: int = 3 + + +class RefusalHandlerCfg(BaseModel): + """Answerability gate. enabled=False = current behavior. + + Phase 2 lever 2g: when the top-1 retrieval similarity falls below `similarity_threshold`, + short-circuit to `no_answer_text` instead of calling the generator. This trades + answer_correctness on borderline-answerable questions for refusal_correctness on + truly unanswerable ones — exactly the trade-off the SQuAD v2 dev set surfaces. + """ + + model_config = ConfigDict(extra="forbid") + enabled: bool = False + similarity_threshold: float = 0.35 + no_answer_text: str = "I don't have enough information to answer that." + + +class PipelineCfg(BaseModel): + """Aggregates all pipeline-level sub-configs into one validated structure. + + Phase 2 additions are all default-off so existing baseline configs keep validating + unchanged. Each new block is a documented lever; see configs/eval/phase2/*.yaml + for tier-by-tier toggles. + """ + + chunker: ChunkerCfg + embedder: EmbedderCfg = Field(default_factory=EmbedderCfg) + retriever: RetrieverCfg + hybrid: HybridCfg = Field(default_factory=HybridCfg) + reranker: RerankerCfg = Field(default_factory=RerankerCfg) + query_rewriter: QueryRewriterCfg = Field(default_factory=QueryRewriterCfg) + generator: GeneratorCfg + refusal_handler: RefusalHandlerCfg = Field(default_factory=RefusalHandlerCfg) + + +class EvalCfg(BaseModel): + """Evaluation harness parameters. + + Teaches: statistical evaluation design — bootstrap CI and permutation + tests are the two workhorses for comparing RAG pipeline variants. + Why seed: reproducibility across runs and machines. + """ + + datasets: list[Literal["squad_v2_dev_200", "ml_papers_v1"]] + judge_model: str = "gpt-4.1-mini" + bootstrap_n: int = 1000 + permutation_n: int = 10000 + seed: int = 42 + # Phase 2: hard per-run spend ceiling. EvalRunner aborts when the cumulative + # generator + judge + rewriter cost crosses this. None disables the guard. + spend_ceiling_usd: float | None = None + + +class EvalConfig(BaseModel): + """Root configuration model for one named evaluation run. + + Each YAML file in configs/eval/ maps 1:1 to one EvalConfig instance. + The name field doubles as a human-readable label in eval reports. + """ + + name: str + description: str = "" + pipeline: PipelineCfg + # TRADE-OFF: `eval` shadows the Python builtin, but as a field name on a + # Pydantic model it is unambiguous — accessed as cfg.eval.datasets, never + # called as a function. Kept for schema clarity over renaming. + eval: EvalCfg + + +def load_config(path: Path) -> EvalConfig: + """Load a YAML config file, validate it, and return an EvalConfig. + + Args: + path: Filesystem path to a YAML config file. + + Returns: + Validated EvalConfig instance. + + Raises: + FileNotFoundError: If path does not exist. + pydantic.ValidationError: If the YAML content fails schema validation. + """ + # WHY: Raise FileNotFoundError explicitly rather than letting yaml.safe_load + # raise a less informative OSError. Callers can distinguish "wrong path" + # from "bad content" without inspecting exception types. + if not path.exists(): + raise FileNotFoundError(f"Eval config not found: {path}") + + raw = yaml.safe_load(path.read_text()) + return EvalConfig.model_validate(raw) diff --git a/src/eval/datasets/__init__.py b/src/eval/datasets/__init__.py new file mode 100644 index 00000000..39d8778a --- /dev/null +++ b/src/eval/datasets/__init__.py @@ -0,0 +1,7 @@ +"""Eval dataset loaders. + +Each loader returns an iterable of EvalQuestion objects and freezes +its sample to eval_data//questions.jsonl for reproducibility. +""" + +from __future__ import annotations diff --git a/src/eval/datasets/ml_papers.py b/src/eval/datasets/ml_papers.py new file mode 100644 index 00000000..c79d6e7e --- /dev/null +++ b/src/eval/datasets/ml_papers.py @@ -0,0 +1,103 @@ +""" +ML Papers v1 dev-set loader. + +Eval Harness Position: + hand-labeled questions.jsonl + corpus_manifest.json + ↓ + load_questions / verify_corpus_manifest + ↓ + runner reads questions, ingests pinned PDFs into Chroma + +Design decisions: + - Corpus manifest pins each PDF by SHA-256 so the eval corpus is + BYTE-STABLE across machines. If a PDF on disk is replaced or + corrupted, the loader raises rather than silently producing + different chunks downstream. + - Empty papers list and empty questions.jsonl are both valid: the + skeleton ships with both empty so the test suite passes before any + labeling has happened. Labeling fills them in incrementally. + - load_questions returns a plain list[EvalQuestion]; the runner is + responsible for the ingest step. +""" + +from __future__ import annotations + +import hashlib +import json +import logging +from pathlib import Path +from typing import Any + +from src.eval.schemas import EvalQuestion + +logger = logging.getLogger(__name__) + +DEFAULT_QUESTIONS_PATH = Path("eval_data/ml_papers_v1/questions.jsonl") +DEFAULT_MANIFEST_PATH = Path("eval_data/ml_papers_v1/corpus_manifest.json") + + +class ManifestVerificationError(Exception): + """Raised when a corpus PDF is missing or has a SHA mismatch.""" + + +def load_questions(path: Path = DEFAULT_QUESTIONS_PATH) -> list[EvalQuestion]: + """Read the hand-labeled questions JSONL. + + An empty file returns an empty list (the skeleton state before any + labeling). A missing file raises. + """ + if not path.exists(): + raise FileNotFoundError(f"ML Papers questions file not found: {path}") + out: list[EvalQuestion] = [] + with path.open() as f: + for line in f: + line = line.strip() + if not line: + continue + out.append(EvalQuestion.model_validate_json(line)) + return out + + +def _sha256_of(path: Path) -> str: + """Compute the SHA-256 hex digest of a file in 64KB chunks.""" + h = hashlib.sha256() + with path.open("rb") as f: + for chunk in iter(lambda: f.read(65536), b""): + h.update(chunk) + return h.hexdigest() + + +def verify_corpus_manifest( + manifest_path: Path = DEFAULT_MANIFEST_PATH, +) -> list[dict[str, Any]]: + """Verify every paper listed in the manifest matches its pinned SHA-256. + + Args: + manifest_path: Path to corpus_manifest.json. + + Returns: + The ``papers`` list from the manifest, after successful verification. + May be empty if no papers have been added yet. + + Raises: + ManifestVerificationError: If any pinned PDF is missing on disk + or its SHA-256 does not match the recorded hash. + """ + with manifest_path.open() as f: + manifest = json.load(f) + papers = manifest.get("papers", []) + for paper in papers: + local_path = Path(paper["local_path"]) + if not local_path.exists(): + raise ManifestVerificationError( + f"Paper {paper['id']!r} not found at {local_path}" + ) + actual_sha = _sha256_of(local_path) + expected_sha = paper["sha256"] + if actual_sha != expected_sha: + raise ManifestVerificationError( + f"Paper {paper['id']!r} sha256 mismatch: " + f"expected {expected_sha}, got {actual_sha}" + ) + logger.info("Verified corpus manifest: %d paper(s)", len(papers)) + return papers diff --git a/src/eval/datasets/squad_v2.py b/src/eval/datasets/squad_v2.py new file mode 100644 index 00000000..360ae70b --- /dev/null +++ b/src/eval/datasets/squad_v2.py @@ -0,0 +1,126 @@ +""" +SQuAD v2 dev-set sampling and freezing for the eval harness. + +Eval Harness Position: + HuggingFace `squad_v2` → seeded sample of 200 → frozen JSONL artifact + ↓ + runner reads via load_frozen + +Design decisions: + - Sample is FROZEN to JSONL on disk (checked into git) so the dev + set is byte-identical across machines and runs. The seed is recorded + separately so anyone can regenerate. + - Each sampled context becomes ONE chunk_id == question_id. This + keeps the corpus unit and the gold-chunk unit aligned for SQuAD, + where every question has exactly one supporting context. The runner + is responsible for ingesting these contexts into a separate Chroma + collection (covered in sub-plan 1B). + - Unanswerable rows preserve the empty-answers / no-gold-chunk shape + so the refusal metric works downstream. +""" + +from __future__ import annotations + +import hashlib +import logging +from pathlib import Path + +from src.eval.schemas import EvalQuestion + +logger = logging.getLogger(__name__) + +DEFAULT_SAMPLE_SIZE = 200 +DEFAULT_SEED = 12345 +DEFAULT_OUTPUT_PATH = Path("eval_data/squad_v2_dev_200/questions.jsonl") + + +def _stable_id(question_text: str, context: str) -> str: + """Stable hash of (question, context) — used as both question_id and chunk_id.""" + h = hashlib.sha256() + h.update(question_text.encode("utf-8")) + h.update(b"\x00") + h.update(context.encode("utf-8")) + return h.hexdigest()[:16] + + +def sample_and_freeze( + output_path: Path = DEFAULT_OUTPUT_PATH, + sample_size: int = DEFAULT_SAMPLE_SIZE, + seed: int = DEFAULT_SEED, +) -> list[EvalQuestion]: + """Sample ``sample_size`` rows from squad_v2 dev split and freeze to JSONL. + + Args: + output_path: Destination JSONL file. + sample_size: Number of (question, context, answers) tuples. + seed: PRNG seed for the sample. + + Returns: + The sampled :class:`EvalQuestion` list (also written to disk). + """ + # Local import: `datasets` pulls heavy transitive deps; keep + # `import src.eval.datasets.squad_v2` cheap. + from datasets import load_dataset + + logger.info("Loading squad_v2 validation split (caches in ~/.cache/huggingface)...") + ds = load_dataset("squad_v2", split="validation") + shuffled = ds.shuffle(seed=seed) + sample = shuffled.select(range(sample_size)) + + questions: list[EvalQuestion] = [] + for row in sample: + question_text: str = row["question"] + context: str = row["context"] + answer_texts: list[str] = row["answers"]["text"] + is_unanswerable = len(answer_texts) == 0 + + chunk_id = _stable_id(question_text, context) + + questions.append( + EvalQuestion( + id=chunk_id, + question=question_text, + gold_answer=None if is_unanswerable else answer_texts[0], + gold_chunk_ids=[] if is_unanswerable else [chunk_id], + is_unanswerable=is_unanswerable, + metadata={ + "title": row.get("title", ""), + # context kept in metadata so the ingestion step can + # find the text without re-querying HF. + "context": context, + }, + ) + ) + + output_path.parent.mkdir(parents=True, exist_ok=True) + with output_path.open("w") as f: + for q in questions: + f.write(q.model_dump_json() + "\n") + + logger.info( + "Froze %d SQuAD v2 questions to %s (seed=%d)", + len(questions), + output_path, + seed, + ) + return questions + + +def load_frozen(path: Path = DEFAULT_OUTPUT_PATH) -> list[EvalQuestion]: + """Load a previously-frozen JSONL of EvalQuestion rows. + + Raises: + FileNotFoundError: If ``path`` does not exist. + """ + if not path.exists(): + raise FileNotFoundError( + f"Frozen SQuAD set not found at {path}. " + f"Run `python -m src.eval.datasets.squad_v2` to generate it." + ) + with path.open() as f: + return [EvalQuestion.model_validate_json(line) for line in f if line.strip()] + + +if __name__ == "__main__": + logging.basicConfig(level=logging.INFO) + sample_and_freeze() diff --git a/src/eval/embedders/__init__.py b/src/eval/embedders/__init__.py new file mode 100644 index 00000000..49980853 --- /dev/null +++ b/src/eval/embedders/__init__.py @@ -0,0 +1,5 @@ +"""Phase 2 embedder package — pluggable Chroma EmbeddingFunction adapters.""" + +from src.eval.embedders.bge_small import BgeEmbedder + +__all__ = ["BgeEmbedder"] diff --git a/src/eval/embedders/bge_small.py b/src/eval/embedders/bge_small.py new file mode 100644 index 00000000..5bc9c105 --- /dev/null +++ b/src/eval/embedders/bge_small.py @@ -0,0 +1,55 @@ +"""BgeEmbedder — Chroma EmbeddingFunction adapter for BAAI/bge-small-en-v1.5. + +Pipeline position: + Document → Chunks → [BgeEmbedder] → Vectors (384-dim) → ChromaDB + +Phase 2 lever 2b. The factory installs this on the Chroma collection at +creation time; ChromaVectorStore.upsert/query then auto-embeds via this +function with no per-call code change. + +Why bge-small-en-v1.5: + - 384 dim — same as ChromaDB's default ONNX MiniLM, so dimension-comparable. + - Strong on MTEB retrieval benchmarks (top-tier 33M-param model). + - Loadable via `sentence-transformers`, which is already a project dep. +""" + +from __future__ import annotations + +from chromadb.api.types import Documents, EmbeddingFunction, Embeddings + + +class BgeEmbedder(EmbeddingFunction[Documents]): + """Chroma-compatible embedding function backed by sentence-transformers. + + Caches the SentenceTransformer model on the instance to avoid re-loading + on every call. Each instance is safe to share across one collection. + """ + + MODEL_NAME = "BAAI/bge-small-en-v1.5" + + def __init__(self) -> None: + # WHY lazy import: sentence-transformers is heavy. Only import when an + # instance is created so module import remains cheap for tests that + # never construct one. + from sentence_transformers import SentenceTransformer + + self._model = SentenceTransformer(self.MODEL_NAME) + + def __call__(self, input: Documents) -> Embeddings: + """Encode a batch of documents into 384-dim vectors. + + Args: + input: List of strings to embed. + + Returns: + List of 384-element float lists, one per input document. + """ + # WHY tolist(): sentence-transformers returns a numpy array; Chroma + # expects a plain list[list[float]] for serialization. + vectors = self._model.encode(list(input), normalize_embeddings=True) + return vectors.tolist() + + @staticmethod + def name() -> str: + """Required by Chroma >= 0.4.x for embedding-function identification.""" + return "bge_small_en_v1_5" diff --git a/src/eval/metrics/__init__.py b/src/eval/metrics/__init__.py new file mode 100644 index 00000000..e0a3c518 --- /dev/null +++ b/src/eval/metrics/__init__.py @@ -0,0 +1,7 @@ +"""Metric modules for the RAG eval harness. + +Each metric module is a small set of pure functions over typed inputs. +Metrics do not perform I/O — they take values and return numbers. +""" + +from __future__ import annotations diff --git a/src/eval/metrics/generation.py b/src/eval/metrics/generation.py new file mode 100644 index 00000000..192344d0 --- /dev/null +++ b/src/eval/metrics/generation.py @@ -0,0 +1,249 @@ +""" +Generation metrics — answer correctness, context recall, and thin +wrappers around src.evaluation.* LLM-as-judge functions for batch use. + +Eval Harness Position: + Generator → answer ─┐ + Retriever → context ─┼─→ [GENERATION METRICS] → scores + details + GoldSet → gold_* ─┘ (per-question metrics) + +Design decisions: + - REUSE the existing src/evaluation.py judges rather than duplicate + their prompts; this module exposes them in a metric-shaped signature + that returns (score, details_dict) for the runner to log. + - Answer correctness is the MEAN of two sub-scores: embedding cosine + and LLM-judge factual match. Either alone is unreliable — cosine + rewards lexical overlap, judge rewards meaning. Combining is more + robust without being statistically fancy. + - Embedder is the same all-MiniLM-L6-v2 ChromaDB ships with — zero + new dependency. + - context_recall is computed without an LLM call: it's a set-overlap + ratio over chunk IDs. Cheap, deterministic, defendable. +""" + +from __future__ import annotations + +import json +import re +from typing import Any + +import numpy as np +from sentence_transformers import SentenceTransformer + +from src.evaluation import ( + evaluate_answer_relevancy, + evaluate_context_precision, + evaluate_faithfulness, +) + +# PATTERN: Lazy global — the SentenceTransformer model is ~80 MB on disk and +# takes ~0.5 s to load. We defer loading until the first call so importing +# this module doesn't pay the cost when only other metrics are needed. +_EMBEDDER: SentenceTransformer | None = None + + +def _embed(text: str) -> np.ndarray: + """Encode a string to a 384-dim vector using all-MiniLM-L6-v2. + + Args: + text: Input string to encode. + + Returns: + Float64 numpy array of shape (384,). + """ + global _EMBEDDER + if _EMBEDDER is None: + # WHY: ChromaDB ships this model, so it's already a transitive dep. + _EMBEDDER = SentenceTransformer("all-MiniLM-L6-v2") + return np.asarray(_EMBEDDER.encode(text), dtype=float) + + +def _cosine(u: np.ndarray, v: np.ndarray) -> float: + """Compute cosine similarity between two vectors, returning 0.0 for zero-norm inputs. + + Args: + u: First vector. + v: Second vector. + + Returns: + Cosine similarity in [-1, 1], or 0.0 if either vector has zero norm. + """ + nu = float(np.linalg.norm(u)) + nv = float(np.linalg.norm(v)) + if nu == 0 or nv == 0: # DEFENSIVE: zero-norm → treat as orthogonal + return 0.0 + return float(np.dot(u, v) / (nu * nv)) + + +def context_recall( + gold_chunk_ids: list[str], + retrieved_chunk_ids: list[str], +) -> float: + """Compute recall of gold chunk IDs against retrieved chunk IDs. + + Recall = |gold ∩ retrieved| / |gold|. Returns NaN when gold is empty + because recall is undefined with no positive examples. + + WHY set-overlap instead of LLM: recall over known IDs is deterministic + and cheap. It answers "did we find what we needed?" without LLM cost. + + Args: + gold_chunk_ids: Ground-truth chunk IDs that should be retrieved. + retrieved_chunk_ids: Chunk IDs returned by the retriever (full set, + not truncated by top-k before calling this function). + + Returns: + Recall in [0.0, 1.0], or float('nan') if gold_chunk_ids is empty. + """ + if not gold_chunk_ids: + return float("nan") + gold_set = set(gold_chunk_ids) + retrieved_set = set(retrieved_chunk_ids) + return len(gold_set & retrieved_set) / len(gold_set) + + +def _judge_factual_match( + generated: str, + gold: str, + llm: Any, +) -> tuple[float, str]: + """Ask the LLM to score factual agreement between generated and gold answers. + + Instructs the LLM to return strict JSON: {"factual_match": float, "reasoning": str}. + Score is clamped to [0, 1]. On JSON parse failure, returns (0.0, default message). + + Args: + generated: The answer produced by the RAG pipeline. + gold: The reference (ground-truth) answer. + llm: LLM handler with generate(prompt, system_prompt=) method. + + Returns: + Tuple of (factual_match_score, reasoning_string). + """ + system_prompt = ( + "You are a factual accuracy evaluator. " + "Compare the generated answer to the gold answer and score factual agreement. " + "Use 1.0 for full factual match, 0.5 for partial match, 0.0 for no match. " + "Respond ONLY with valid JSON — no prose, no code fences." + ) + user_prompt = ( + f"Generated answer: {generated}\n\n" + f"Gold answer: {gold}\n\n" + 'Return JSON with exactly: {"factual_match": , "reasoning": ""}' + ) + + raw = llm.generate(user_prompt, system_prompt=system_prompt) + + # DEFENSIVE: strip markdown fences before parsing, identical to evaluation.py + stripped = re.sub(r"^```(?:json)?\s*", "", raw.strip()) + stripped = re.sub(r"\s*```$", "", stripped).strip() + + try: + parsed = json.loads(stripped) + score = max(0.0, min(1.0, float(parsed["factual_match"]))) + reasoning = str(parsed.get("reasoning", "")) + return score, reasoning + except (json.JSONDecodeError, KeyError, ValueError): + return (0.0, "Judge returned malformed JSON; defaulting to 0.0") + + +def answer_correctness( + generated: str, + gold: str, + llm: Any, +) -> tuple[float, dict]: + """Score answer correctness as the mean of embedding cosine and LLM factual match. + + TRADE-OFF: cosine similarity captures lexical/semantic overlap but can be + fooled by paraphrasing or topic drift. The LLM judge captures factual + agreement but is noisy. Averaging both sub-scores is more robust than + either alone without adding statistical complexity. + + Args: + generated: The answer produced by the RAG pipeline. + gold: The reference (ground-truth) answer. + llm: LLM handler with generate(prompt, system_prompt=) method. + + Returns: + Tuple of (combined_score, details_dict). combined_score is in [0, 1]. + details_dict contains "cosine", "judge_factual_match", "judge_reasoning". + """ + cos = _cosine(_embed(generated), _embed(gold)) + judge_score, judge_reasoning = _judge_factual_match(generated, gold, llm) + combined = (cos + judge_score) / 2.0 + details: dict = { + "cosine": cos, + "judge_factual_match": judge_score, + "judge_reasoning": judge_reasoning, + } + return combined, details + + +def judge_faithfulness( + answer: str, + contexts: list[str], + llm: Any, +) -> tuple[float, dict]: + """Wrap evaluate_faithfulness for batch-mode use in the eval harness. + + Returns (score, details_dict) instead of the raw (score, reasoning, details_json) + tuple so the runner can log a uniform structure across all metrics. + + Args: + answer: The answer generated by the RAG pipeline. + contexts: Retrieved text chunks used to generate the answer. + llm: LLM handler with generate(prompt, system_prompt=) method. + + Returns: + Tuple of (faithfulness_score, details_dict). + details_dict is parsed from the judge's JSON output, plus "reasoning". + """ + score, reasoning, details_json = evaluate_faithfulness(answer, contexts, llm) + details: dict = json.loads(details_json) if details_json else {} + details["reasoning"] = reasoning + return score, details + + +def judge_answer_relevancy( + question: str, + answer: str, + llm: Any, +) -> tuple[float, dict]: + """Wrap evaluate_answer_relevancy for batch-mode use in the eval harness. + + Args: + question: The original user question. + answer: The answer generated by the RAG pipeline. + llm: LLM handler with generate(prompt, system_prompt=) method. + + Returns: + Tuple of (relevancy_score, details_dict). + details_dict contains "reasoning". + """ + # WHY only two-tuple: evaluate_answer_relevancy has no details_json — it + # returns just (score, reasoning). Wrap into a dict for uniform structure. + score, reasoning = evaluate_answer_relevancy(question, answer, llm) + details: dict = {"reasoning": reasoning} + return score, details + + +def judge_context_precision( + question: str, + contexts: list[str], + llm: Any, +) -> tuple[float, dict]: + """Wrap evaluate_context_precision for batch-mode use in the eval harness. + + Args: + question: The original user question. + contexts: Retrieved text chunks (ordered by retrieval rank). + llm: LLM handler with generate(prompt, system_prompt=) method. + + Returns: + Tuple of (precision_score, details_dict). + details_dict is parsed from the judge's JSON output, plus "reasoning". + """ + score, reasoning, details_json = evaluate_context_precision(question, contexts, llm) + details: dict = json.loads(details_json) if details_json else {} + details["reasoning"] = reasoning + return score, details diff --git a/src/eval/metrics/operational.py b/src/eval/metrics/operational.py new file mode 100644 index 00000000..d5ce10d9 --- /dev/null +++ b/src/eval/metrics/operational.py @@ -0,0 +1,139 @@ +""" +Operational aggregators — latency, cost, token totals over a run. + +Eval Harness Position: + list[EvalResult] → [OPERATIONAL] → run-level summary in metrics.json + ^^^^^^^^^^^^ + Pure aggregations. Errored questions are skipped (their timings/ + costs are not representative of healthy pipeline behavior). + +Design decisions: + - p50/p95/p99 via numpy.percentile with default 'linear' interpolation + — the standard convention; matches how tools like Datadog report + percentiles. Don't switch to 'nearest' without good reason. + - Errored results excluded from all aggregates: a 30-second timeout + on a broken question would skew p99 misleadingly. The error count + lives in RunMetadata.n_errors. +""" + +from __future__ import annotations + +from typing import Iterable + +import numpy as np + +from src.eval.schemas import EvalResult + + +def _healthy(results: Iterable[EvalResult]) -> list[EvalResult]: + """Return only results where the pipeline did not raise.""" + return [r for r in results if r.error is None] + + +def aggregate_timings(results: Iterable[EvalResult]) -> dict[str, dict[str, float]]: + """Compute p50/p95/p99 latency percentiles per pipeline stage. + + Stage names are auto-discovered from the healthy results so the + function works with any timings_ms schema without configuration. + Results where a given stage is absent are skipped for that stage + (not treated as zero — a missing stage means the step didn't run). + + Args: + results: Iterable of EvalResult instances from a run. + + Returns: + Mapping of stage_name → {"p50": ms, "p95": ms, "p99": ms}. + Empty dict if there are no healthy results. + """ + healthy = _healthy(results) + if not healthy: + return {} + + # WHY: Discover stages dynamically — avoids hardcoding stage names + # and naturally handles runs with different pipeline configurations. + stage_names: set[str] = {stage for r in healthy for stage in r.timings_ms} + + output: dict[str, dict[str, float]] = {} + for stage in stage_names: + # Collect values only where the stage is present in that result. + values = [r.timings_ms[stage] for r in healthy if stage in r.timings_ms] + if not values: + continue + arr = np.array(values, dtype=float) + output[stage] = { + "p50": float(np.percentile(arr, 50)), + "p95": float(np.percentile(arr, 95)), + "p99": float(np.percentile(arr, 99)), + } + return output + + +def aggregate_costs(results: Iterable[EvalResult]) -> dict[str, float]: + """Compute total and mean cost in USD across healthy results, with per-bucket breakdown. + + Args: + results: Iterable of EvalResult instances from a run. + + Returns: + Dict with keys "total_usd", "mean_usd_per_query", "generator_total_usd", + "judge_total_usd", "rewriter_total_usd". All are 0.0 when there are no + healthy results. + """ + healthy = _healthy(results) + if not healthy: + return { + "total_usd": 0.0, + "mean_usd_per_query": 0.0, + "generator_total_usd": 0.0, + "judge_total_usd": 0.0, + "rewriter_total_usd": 0.0, + } + + costs = [r.cost_usd for r in healthy] + total = float(sum(costs)) + mean = total / len(costs) + generator_total = sum(r.cost_breakdown.get("generator", 0.0) for r in healthy) + judge_total = sum(r.cost_breakdown.get("judge", 0.0) for r in healthy) + rewriter_total = sum(r.cost_breakdown.get("rewriter", 0.0) for r in healthy) + return { + "total_usd": total, + "mean_usd_per_query": mean, + "generator_total_usd": generator_total, + "judge_total_usd": judge_total, + "rewriter_total_usd": rewriter_total, + } + + +def aggregate_tokens(results: Iterable[EvalResult]) -> dict[str, float | int]: + """Compute total and mean prompt/completion token counts. + + Args: + results: Iterable of EvalResult instances from a run. + + Returns: + Dict with keys "total_prompt", "total_completion", + "mean_prompt", "mean_completion". All are 0 when there are + no healthy results. + """ + healthy = _healthy(results) + if not healthy: + return { + "total_prompt": 0, + "total_completion": 0, + "mean_prompt": 0.0, + "mean_completion": 0.0, + } + + prompt_counts = [r.tokens.get("prompt", 0) for r in healthy] + completion_counts = [r.tokens.get("completion", 0) for r in healthy] + + total_prompt = sum(prompt_counts) + total_completion = sum(completion_counts) + n = len(healthy) + + return { + "total_prompt": total_prompt, + "total_completion": total_completion, + "mean_prompt": total_prompt / n, + "mean_completion": total_completion / n, + } diff --git a/src/eval/metrics/refusal.py b/src/eval/metrics/refusal.py new file mode 100644 index 00000000..c87c076f --- /dev/null +++ b/src/eval/metrics/refusal.py @@ -0,0 +1,133 @@ +""" +Refusal correctness — does the answer correctly refuse on unanswerable +questions and correctly attempt on answerable ones? + +Eval Harness Position: + Pipeline → answer + is_unanswerable → [REFUSAL] → 1.0 / 0.0 + ^^^^^^^ + Two-stage detection: + 1. Regex fast path on canonical refusal phrases. + 2. LLM-as-judge fallback when the regex doesn't trigger but + refusal is ambiguous. + +Design decisions: + - Binary score (0.0 / 1.0). A partial refusal isn't really a thing + in this dataset; either the model bailed out or it didn't. + - Regex first because LLM-judge calls cost real money on a 200-row + dev set, and the canonical phrasings catch ~80% of cases. + - Fallback uses the same JSON parsing convention as + src.evaluation.* for consistency. +""" + +from __future__ import annotations + +import json +import logging +import re + +logger = logging.getLogger(__name__) + + +# Canonical refusal patterns. Case-insensitive; match anywhere in answer. +_REFUSAL_PATTERNS = [ + re.compile(r"\bcannot answer\b", re.IGNORECASE), + re.compile(r"\bcan't answer\b", re.IGNORECASE), + re.compile(r"\bunable to answer\b", re.IGNORECASE), + re.compile(r"\bdo(es)? not contain\b", re.IGNORECASE), + re.compile(r"\bdoesn't contain\b", re.IGNORECASE), + re.compile(r"\bnot (mentioned|stated|provided|specified)\b", re.IGNORECASE), + re.compile(r"\bno information\b", re.IGNORECASE), + re.compile(r"\bcontext does not (address|cover|contain|mention)\b", re.IGNORECASE), + re.compile(r"\bI don'?t know\b", re.IGNORECASE), + re.compile(r"\bcannot be answered\b", re.IGNORECASE), +] + + +def is_refusal(answer: str) -> bool: + """Heuristic check: does the answer use canonical refusal phrasing? + + Args: + answer: The generated answer text to classify. + + Returns: + True if any canonical refusal pattern matches; False otherwise. + """ + return any(p.search(answer) for p in _REFUSAL_PATTERNS) + + +def _judge_is_refusal(answer: str, llm) -> bool: + """LLM-as-judge fallback for ambiguous refusal detection. + + WHY: When the regex fast path doesn't match, we delegate to an LLM + that can understand hedged, indirect, or colloquial refusals that + don't use canonical phrasing (e.g. "I'd rather not speculate"). + + PATTERN: Strict JSON output — {"is_refusal": bool} — avoids + parsing freeform text and keeps the judge deterministic. + + Args: + answer: The generated answer text to classify. + llm: Any object with a generate(prompt, system_prompt=) method. + + Returns: + True if the LLM judges the answer as a refusal; False otherwise + (also False on JSON parse failure, logged as warning). + """ + system_prompt = ( + "You are a classification assistant. " + "Respond ONLY with a JSON object in this exact format: " + '{"is_refusal": true} or {"is_refusal": false}. ' + "No other text." + ) + user_prompt = ( + "Does the following answer refuse to answer the question " + "(e.g. claims it cannot answer, lacks context, or doesn't know)?\n\n" + f"Answer: {answer}" + ) + + raw = llm.generate(user_prompt, system_prompt=system_prompt) + + # PATTERN: Strip markdown code fences before parsing JSON — LLMs often + # wrap JSON in ```json ... ``` even when instructed not to. + cleaned = re.sub(r"^```(?:json)?\s*|\s*```$", "", raw.strip()) + + try: + parsed = json.loads(cleaned) + return bool(parsed.get("is_refusal", False)) + except (json.JSONDecodeError, AttributeError) as exc: + logger.warning("LLM judge returned non-JSON response: %r (%s)", raw, exc) + return False + + +def refusal_correctness(answer: str, is_unanswerable: bool, llm) -> float: + """Score whether the answer correctly handles an answerable/unanswerable question. + + Scoring logic: + - refused AND is_unanswerable → 1.0 (correct refusal) + - not refused AND answerable → 1.0 (correct attempt) + - refused AND answerable → 0.0 (false refusal) + - not refused AND unanswerable → 0.0 (missed refusal) + + TRADE-OFF: Two-stage detection keeps LLM costs low. The regex fast + path handles canonical phrasings without any API call. Only truly + ambiguous answers pay the LLM-judge cost. + + Args: + answer: The generated answer text. + is_unanswerable: Ground-truth flag — True if the question has + no answer in the retrieved context. + llm: Any object with a generate(prompt, system_prompt=) method. + Only called when the regex fast path doesn't match. + + Returns: + 1.0 if refusal classification matches is_unanswerable, else 0.0. + """ + # PATTERN: Regex fast path — no LLM call if the answer uses + # canonical refusal phrasing. Covers ~80% of cases. + if is_refusal(answer): + refused = True + else: + # Fallback: delegate to LLM judge for ambiguous cases. + refused = _judge_is_refusal(answer, llm) + + return 1.0 if (refused == is_unanswerable) else 0.0 diff --git a/src/eval/metrics/retrieval.py b/src/eval/metrics/retrieval.py new file mode 100644 index 00000000..e11b1c5b --- /dev/null +++ b/src/eval/metrics/retrieval.py @@ -0,0 +1,127 @@ +""" +Retrieval metrics — Recall@k, MRR@k, nDCG@k. + +Eval Harness Position: + Retriever → retrieved_chunk_ids → [METRICS] ← gold_chunk_ids + ^^^^^^^ + Pure functions over (gold_chunk_ids, retrieved_chunk_ids). No I/O, + no LLM calls. Fast, deterministic, unit-testable in isolation. + +Design decisions: + - Operate on opaque string IDs, not chunk objects, so the same + metrics work over Chroma chunk IDs, BM25 doc IDs, or any other + identifier scheme. + - Empty gold returns NaN (not 0.0) because the metric is undefined, + not zero. The aggregator drops NaN values per-metric per-question. + - nDCG uses binary relevance (gold or not gold) — graded relevance + would require richer labels, deferred to a future phase. +""" + +from __future__ import annotations + +import math +from typing import Sequence + +# Sentinel returned for undefined metrics (empty gold set). +# Callers should check math.isnan() and skip these in aggregation. +NAN = float("nan") + + +def recall_at_k( + gold_chunk_ids: Sequence[str], + retrieved_chunk_ids: Sequence[str], + k: int, +) -> float: + """Fraction of gold chunks found in the top-k retrieved results. + + Args: + gold_chunk_ids: Ground-truth relevant chunk IDs. + retrieved_chunk_ids: Ranked list of retrieved chunk IDs (best first). + k: Cutoff — only the first k entries of retrieved_chunk_ids count. + + Returns: + |gold ∩ retrieved[:k]| / |gold|, or NaN if gold is empty. + """ + if not gold_chunk_ids: + return NAN + + gold_set = set(gold_chunk_ids) + top_k = set(retrieved_chunk_ids[:k]) + return len(gold_set & top_k) / len(gold_set) + + +def mrr_at_k( + gold_chunk_ids: Sequence[str], + retrieved_chunk_ids: Sequence[str], + k: int, +) -> float: + """Reciprocal rank of the first relevant chunk in the top-k results. + + MRR captures how early the first relevant result appears. It rewards + systems that surface at least one correct answer near the top of the list. + + Args: + gold_chunk_ids: Ground-truth relevant chunk IDs. + retrieved_chunk_ids: Ranked list of retrieved chunk IDs (best first). + k: Cutoff — only the first k entries of retrieved_chunk_ids count. + + Returns: + 1/rank of the first hit (1-indexed), 0.0 if no hit in top-k, + or NaN if gold is empty. + """ + if not gold_chunk_ids: + return NAN + + gold_set = set(gold_chunk_ids) + for rank, chunk_id in enumerate(retrieved_chunk_ids[:k], start=1): + if chunk_id in gold_set: + return 1.0 / rank + return 0.0 + + +def ndcg_at_k( + gold_chunk_ids: Sequence[str], + retrieved_chunk_ids: Sequence[str], + k: int, +) -> float: + """Normalized Discounted Cumulative Gain at rank k (binary relevance). + + nDCG measures both the presence and the rank of relevant results. + Higher-ranked hits contribute more than lower-ranked ones (logarithmic + discount). Normalizing by the ideal DCG makes the score comparable + across queries with different numbers of gold chunks. + + Args: + gold_chunk_ids: Ground-truth relevant chunk IDs. + retrieved_chunk_ids: Ranked list of retrieved chunk IDs (best first). + k: Cutoff — only the first k entries of retrieved_chunk_ids count. + + Returns: + DCG / IDCG in [0.0, 1.0], 0.0 if IDCG is 0, or NaN if gold is empty. + """ + if not gold_chunk_ids: + return NAN + + gold_set = set(gold_chunk_ids) + + # Compute actual DCG: sum 1/log2(rank+1) for each hit in retrieved[:k]. + # WHY rank+1 inside log2: standard DCG convention so rank-1 gives + # log2(2)=1 (a gain of 1.0, not infinity). Without +1, log2(1)=0 → div/0. + dcg = sum( + 1.0 / math.log2(rank + 1) + for rank, chunk_id in enumerate(retrieved_chunk_ids[:k], start=1) + if chunk_id in gold_set + ) + + # Compute ideal DCG: assume all gold chunks are retrieved at the top ranks. + # We only credit up to min(|gold|, k) ideal positions. + ideal_hits = min(len(gold_chunk_ids), k) + idcg = sum(1.0 / math.log2(rank + 1) for rank in range(1, ideal_hits + 1)) + + # TRADE-OFF: returning 0.0 when IDCG=0 instead of NaN because an + # empty-gold case is already handled above; IDCG=0 here means k=0, + # which is a caller error, and 0.0 is a safe neutral value. + if idcg == 0.0: + return 0.0 + + return dcg / idcg diff --git a/src/eval/pipeline_factory.py b/src/eval/pipeline_factory.py new file mode 100644 index 00000000..1ab889a4 --- /dev/null +++ b/src/eval/pipeline_factory.py @@ -0,0 +1,511 @@ +""" +EvalPipeline factory — builds a fresh, isolated RAG pipeline from an +EvalConfig + dataset name. + +Eval Harness Position: + EvalConfig + dataset → [PIPELINE FACTORY] → EvalPipeline + ↓ + ingest → query → teardown + (used by EvalRunner) + +Design decisions: + - Ephemeral Chroma collection per (config, dataset) so two concurrent + runs cannot pollute each other's vectors. Random suffix on the + collection name guards against collisions. + - Per-stage timings via time.perf_counter() so the runner can record + p50/p95/p99 latency at aggregation time. + - Token counting: tiktoken if available, word-count×1.3 fallback — + eval should not hard-fail because a tokenizer for a new model + isn't installed. + - Test doubles (DummyLLM) inject via *_override params; production + uses LLMHandler(model_name). + +Return type of query(): + - Returns list[SearchResult] (from vector_store.SearchResult), not + list[Chunk]. SearchResult carries chunk_id + score which the runner + needs for retrieval metrics. The test asserts only isinstance(chunks, list) + so this type is compatible. +""" + +from __future__ import annotations + +import logging +import time +import uuid +from dataclasses import dataclass, field +from typing import Any + +import chromadb + +from src.document_loader import TextChunker +from src.eval._telemetry import count_tokens +from src.eval.config import EvalConfig +from src.eval.schemas import EvalQuestion +from src.llm_handler import LLMHandler +from src.vector_store import ChromaVectorStore, SearchResult + +logger = logging.getLogger(__name__) + + +# --------------------------------------------------------------------------- # +# EvalPipeline # +# --------------------------------------------------------------------------- # + +@dataclass +class EvalPipeline: + """An isolated, ephemeral RAG pipeline for one (config, dataset) eval run. + + Teaches: the factory pattern — consumers never call __init__ directly; + they use build_pipeline() which sets up all components and wires them + into a coherent, ready-to-use pipeline. + + Why ephemeral Chroma: each pipeline gets its own collection with a + random suffix, so parallel eval runs cannot pollute each other's vectors. + teardown() deletes the collection after the run completes. + """ + + chunker: TextChunker + vector_store: ChromaVectorStore + llm: Any # LLMHandler or a test double with .generate(prompt, system_prompt) + judge_llm: Any # Same contract as llm + config: EvalConfig + dataset_name: str + + # Phase 2 additions — None when the corresponding lever is off. + hybrid_retriever: object | None = None # BM25HybridRetriever or None + reranker: object | None = None # CrossEncoderReranker or None + rewriter: object | None = None # QueryRewriter or None + refusal_handler: object | None = None # RefusalHandler or None + + # Private: needed for teardown() — ChromaVectorStore doesn't own the client. + # WHY not reach into vector_store._collection._client: that would couple us + # to ChromaDB internals that could change. Own the client reference here. + _client: chromadb.ClientAPI = field(repr=False, default=None) # type: ignore[assignment] + _collection_name: str = field(repr=False, default="") + + def ingest(self, questions: list[EvalQuestion]) -> None: + """Upsert question contexts into the vector store. + + For squad_v2_dev_200: + Each question carries its context in metadata["context"]. The context + IS the chunk — SQuAD is designed so each question has exactly one + supporting passage. question.id serves as the Chroma document ID, + which makes gold_chunk_id lookup trivial in the retrieval metric. + + For ml_papers_v1: + Loads corpus_manifest.json, reads each pinned PDF via DocumentLoader, + chunks it with this pipeline's TextChunker, and upserts the chunks. + If the manifest is missing or empty, ingest is a no-op (logged). + + Args: + questions: Gold-labeled questions from the dataset loader. + """ + if self.dataset_name == "squad_v2_dev_200": + self._ingest_squad(questions) + elif self.dataset_name == "ml_papers_v1": + self._ingest_ml_papers() + else: + logger.warning( + "Unknown dataset %r — ingest is a no-op.", self.dataset_name + ) + + def _ingest_squad(self, questions: list[EvalQuestion]) -> None: + """Upsert each question's context as one Chroma document. + + PATTERN: question.id == chunk_id == gold_chunk_id. This alignment + means the retrieval metric can check retrieved IDs directly against + EvalQuestion.gold_chunk_ids without any translation layer. + """ + ids: list[str] = [] + documents: list[str] = [] + metadatas: list[dict[str, Any]] = [] + + for q in questions: + ctx = q.metadata.get("context", "") + if not ctx: + logger.warning("SQuAD question %s has no context — skipping.", q.id) + continue + ids.append(q.id) + documents.append(ctx) + # WHY include doc_id matching chunk_id: ChromaVectorStore.delete_by_doc_id + # filters on metadata["doc_id"]. We don't use deletion in eval, but + # keeping the schema consistent with production reduces surprises. + metadatas.append({"doc_id": q.id, "question_id": q.id}) + + if ids: + self.vector_store.upsert(ids=ids, documents=documents, metadatas=metadatas) + logger.info("Ingested %d SQuAD contexts into '%s'.", len(ids), self._collection_name) + + # Phase 2: build the hybrid retriever now that chunks are upserted. + # WHY here (lazy): BM25HybridRetriever needs the full chunk corpus at + # construction time. build_pipeline() runs before ingest, so we defer. + if self.config.pipeline.hybrid.enabled: + documents_map = dict(zip(ids, documents)) + self.hybrid_retriever = _build_hybrid_retriever( + self.config.pipeline.hybrid, self.vector_store, documents_map, + ) + + def _ingest_ml_papers(self) -> None: + """Load, chunk, and upsert PDFs listed in corpus_manifest.json. + + WHY graceful no-op on missing manifest: the ML Papers dataset ships + with an empty manifest skeleton so tests pass before any PDFs are + labeled. A missing manifest is not an error for the eval harness — + it means no papers have been added yet. + """ + import json + from pathlib import Path + + from src.document_loader import DocumentLoader + + manifest_path = Path("eval_data/ml_papers_v1/corpus_manifest.json") + if not manifest_path.exists(): + logger.info("ML Papers manifest not found at %s — ingest is a no-op.", manifest_path) + return + + with manifest_path.open() as f: + manifest = json.load(f) + + papers = manifest.get("papers", []) + if not papers: + logger.info("ML Papers manifest has no papers — ingest is a no-op.") + return + + loader = DocumentLoader() + for paper in papers: + local_path = Path(paper["local_path"]) + try: + doc = loader.load(local_path) + except (FileNotFoundError, ValueError) as exc: + logger.warning("Skipping paper %s: %s", paper.get("id"), exc) + continue + + chunks = self.chunker.chunk(doc) + if not chunks: + continue + + self.vector_store.upsert( + ids=[c.chunk_id for c in chunks], + documents=[c.content for c in chunks], + metadatas=[{"doc_id": c.doc_id, "paper_id": paper.get("id", "")} for c in chunks], + ) + logger.info( + "Ingested paper %s: %d chunks.", paper.get("id"), len(chunks) + ) + + # Phase 2: build hybrid retriever over all upserted chunks. + # WHY after the loop: we need the complete corpus before building BM25. + if self.config.pipeline.hybrid.enabled: + all_ids = self.vector_store._collection.get()["ids"] + all_docs = self.vector_store._collection.get()["documents"] + if all_ids: + documents_map = dict(zip(all_ids, all_docs)) + self.hybrid_retriever = _build_hybrid_retriever( + self.config.pipeline.hybrid, self.vector_store, documents_map, + ) + + def query(self, question: str) -> tuple[list[SearchResult], str, dict]: + """Retrieve relevant chunks and generate an answer with timing + cost telemetry. + + Phase 2 pipeline steps: rewrite → retrieve (hybrid or dense) → rerank → + refusal gate → generate. Each step is a no-op when the corresponding + config lever is off, preserving backward compatibility with Phase 1 callers. + + Args: + question: Natural language question from the eval set. + + Returns: + Tuple of (chunks, answer, telemetry). telemetry keys: + timings_ms: dict of stage→ms for rewrite, retrieve, rerank, + refusal_check, generate + tokens: {"prompt": int, "completion": int} + cost_usd: float (generator side) + rewriter_cost_usd: float (rewriter side, 0.0 when disabled) + """ + from src.eval import pricing + + timings: dict[str, float] = {} + rewriter_cost = 0.0 + + # ---- Rewrite (lever 2e) ----------------------------------------------- + t = time.perf_counter() + if self.rewriter is not None: + queries, rewriter_cost, _, _ = self.rewriter.expand(question) + else: + queries = [question] + timings["rewrite"] = (time.perf_counter() - t) * 1000.0 + + # ---- Retrieve --------------------------------------------------------- + # WHY use rerank_top_n for initial fetch when a reranker is active: + # the reranker needs a wider candidate pool to re-score before final_top_k. + top_k_initial = ( + self.config.pipeline.reranker.rerank_top_n + if self.reranker is not None else self.config.pipeline.retriever.top_k + ) + t = time.perf_counter() + if self.hybrid_retriever is not None: + seen: dict[str, SearchResult] = {} + for q in queries: + for r in self.hybrid_retriever.retrieve(q, top_k=top_k_initial): + if r.chunk_id not in seen: + seen[r.chunk_id] = r + results = list(seen.values()) + else: + seen = {} + for q in queries: + for r in self.vector_store.query(query_text=q, top_k=top_k_initial): + if r.chunk_id not in seen: + seen[r.chunk_id] = r + results = list(seen.values()) + timings["retrieve"] = (time.perf_counter() - t) * 1000.0 + + # ---- Rerank (lever 2d) ------------------------------------------------ + t = time.perf_counter() + if self.reranker is not None: + results = self.reranker.rerank( + question, results, + final_top_k=self.config.pipeline.reranker.final_top_k, + ) + else: + results = results[: self.config.pipeline.retriever.top_k] + timings["rerank"] = (time.perf_counter() - t) * 1000.0 + + # ---- Refusal gate (lever 2g) ------------------------------------------ + t = time.perf_counter() + if self.refusal_handler is not None and self.refusal_handler.should_refuse(results): + chunks, answer = self.refusal_handler.refuse_response() + timings["refusal_check"] = (time.perf_counter() - t) * 1000.0 + return chunks, answer, { + "timings_ms": timings, + "tokens": {"prompt": 0, "completion": 0}, + "cost_usd": 0.0, + "rewriter_cost_usd": rewriter_cost, + } + timings["refusal_check"] = (time.perf_counter() - t) * 1000.0 + + # ---- Generate --------------------------------------------------------- + context = "\n\n".join(r.content for r in results) + system_prompt = ( + "You are a helpful assistant. Answer the question based solely on the " + "provided context. If the context does not contain enough information, " + "say so clearly." + ) + user_prompt = f"Context:\n{context}\n\nQuestion: {question}\n\nAnswer:" + # WHY count both: the LLM sees system_prompt + user_prompt as prompt tokens. + full_prompt_text = system_prompt + "\n" + user_prompt + model = self.config.pipeline.generator.model + + t = time.perf_counter() + answer = self.llm.generate(user_prompt, system_prompt=system_prompt) + timings["generate"] = (time.perf_counter() - t) * 1000.0 + + # ---- Token counting + cost estimation --------------------------------- + prompt_tokens = count_tokens(full_prompt_text, model) + completion_tokens = count_tokens(answer, model) + cost = pricing.cost_usd(model, prompt_tokens, completion_tokens) + + return results, answer, { + "timings_ms": timings, + "tokens": {"prompt": prompt_tokens, "completion": completion_tokens}, + "cost_usd": cost, + "rewriter_cost_usd": rewriter_cost, + } + + def teardown(self) -> None: + """Delete the ephemeral Chroma collection and release the client reference. + + WHY idempotent: the test suite may call teardown() in a finally block + after the collection was already deleted explicitly. A double-call must + be a silent no-op, not an exception. + """ + if self._client is None or not self._collection_name: + return # already torn down + + try: + self._client.delete_collection(self._collection_name) + logger.debug("Deleted Chroma collection '%s'.", self._collection_name) + except Exception as exc: + # Not-found is the common idempotency case; log and continue. + logger.debug("teardown: delete_collection raised (already gone?): %s", exc) + finally: + # Null out so a subsequent call is a no-op (idempotency guarantee). + self._client = None # type: ignore[assignment] + self._collection_name = "" + + +# --------------------------------------------------------------------------- # +# Factory # +# --------------------------------------------------------------------------- # + +def build_pipeline( + config: EvalConfig, + dataset_name: str, + llm_override: object | None = None, + judge_llm_override: object | None = None, +) -> EvalPipeline: + """Construct a fresh, isolated EvalPipeline for a (config, dataset) pair. + + Teaches: the factory function pattern — all wiring happens here so callers + receive a ready-to-use object with no boilerplate. Each call produces an + independent Chroma collection so concurrent runs don't share state. + + Args: + config: Validated EvalConfig specifying chunker, retriever, generator, + and eval parameters. + dataset_name: Name key identifying which dataset the pipeline handles + (e.g. "squad_v2_dev_200", "ml_papers_v1"). + llm_override: If provided, use this object as the answer LLM instead + of building an LLMHandler. Primarily for test doubles. + judge_llm_override: If provided, use this object as the judge LLM. + + Returns: + A configured EvalPipeline ready for ingest() → query() → teardown(). + """ + # ---- Chunker --------------------------------------------------------------- + chunker_cfg = config.pipeline.chunker + chunker = TextChunker( + chunk_size=chunker_cfg.chunk_size, + chunk_overlap=chunker_cfg.chunk_overlap, + strategy=chunker_cfg.strategy, + ) + + # ---- Embedder (lever 2b) --------------------------------------------------- + # WHY build before the collection: Chroma binds an EmbeddingFunction to the + # collection at creation time and uses it for all subsequent upserts/queries. + embedder_cfg = config.pipeline.embedder + embedding_function = _build_embedding_function(embedder_cfg) + + # ---- Chroma collection (ephemeral — lives only for this pipeline) ---------- + # WHY EphemeralClient: no disk I/O, no port, no cleanup needed beyond + # client.delete_collection(). Perfectly isolated per build_pipeline() call. + # WHY random suffix: prevents name collisions if two pipelines with the + # same config+dataset names are built in the same process. + # NOTE: First call auto-downloads all-MiniLM-L6-v2 ONNX (~80MB) if not cached. + collection_name = f"eval_{config.name}_{dataset_name}_{uuid.uuid4().hex[:6]}" + client = chromadb.EphemeralClient() + collection = client.get_or_create_collection( + name=collection_name, + embedding_function=embedding_function, + # WHY cosine: ChromaVectorStore converts distance→similarity via + # score = max(0, 1 - distance). This only makes sense in cosine space + # where distance ∈ [0, 2] and identical vectors have distance 0. + metadata={"hnsw:space": "cosine"}, + ) + vector_store = ChromaVectorStore(collection=collection) + + # ---- LLM handlers ---------------------------------------------------------- + llm = llm_override if llm_override is not None else LLMHandler(config.pipeline.generator.model) + judge_llm = ( + judge_llm_override if judge_llm_override is not None + else LLMHandler(config.eval.judge_model) + ) + + return EvalPipeline( + chunker=chunker, + vector_store=vector_store, + llm=llm, + judge_llm=judge_llm, + config=config, + dataset_name=dataset_name, + hybrid_retriever=None, # built lazily in _ingest_squad / _ingest_ml_papers + reranker=_build_reranker(config.pipeline.reranker), + rewriter=_build_rewriter(config.pipeline.query_rewriter, llm=llm), + refusal_handler=_build_refusal(config.pipeline.refusal_handler), + _client=client, + _collection_name=collection_name, + ) + + +# --------------------------------------------------------------------------- # +# Phase 2 component builders # +# --------------------------------------------------------------------------- # + +def _build_embedding_function(cfg) -> object: + """Build the Chroma EmbeddingFunction for the given embedder config. + + Args: + cfg: EmbedderCfg with a `name` field identifying the embedder variant. + + Returns: + A Chroma-compatible EmbeddingFunction instance. + + Raises: + ValueError: If cfg.name is not a known embedder key. + """ + if cfg.name == "chroma_default": + from chromadb.utils import embedding_functions + return embedding_functions.DefaultEmbeddingFunction() + if cfg.name == "bge_small_en_v1_5": + from src.eval.embedders import BgeEmbedder + return BgeEmbedder() + raise ValueError(f"Unknown embedder name: {cfg.name}") + + +def _build_hybrid_retriever(cfg, vector_store, documents: dict[str, str]): + """Build BM25HybridRetriever if hybrid is enabled, else return None. + + Args: + cfg: HybridCfg specifying enabled flag and RRF/top-k parameters. + vector_store: Dense retriever (Chroma collection wrapper). + documents: Mapping of chunk_id → raw document text for BM25 indexing. + + Returns: + BM25HybridRetriever when cfg.enabled, else None. + """ + if not cfg.enabled: + return None + from src.eval.retrievers import BM25HybridRetriever + return BM25HybridRetriever( + vector_store=vector_store, documents=documents, + bm25_top_k=cfg.bm25_top_k, dense_top_k=cfg.dense_top_k, rrf_k=cfg.rrf_k, + ) + + +def _build_reranker(cfg): + """Build CrossEncoderReranker if a reranker model is configured, else None. + + Args: + cfg: RerankerCfg. model=None means reranking is disabled. + + Returns: + CrossEncoderReranker when cfg.model is set, else None. + """ + if cfg.model is None: + return None + from src.eval.retrievers import CrossEncoderReranker + return CrossEncoderReranker() + + +def _build_rewriter(cfg, llm): + """Build QueryRewriter if a rewriter model is configured, else None. + + Args: + cfg: QueryRewriterCfg. model=None disables query expansion. + llm: LLM handler (same object as the pipeline answer LLM). QueryRewriter + will call llm.generate_with_usage() to produce expansions. + + Returns: + QueryRewriter when cfg.model is set, else None. + """ + if cfg.model is None: + return None + from src.eval.transforms import QueryRewriter + return QueryRewriter(model=cfg.model, max_expansions=cfg.max_expansions, llm=llm) + + +def _build_refusal(cfg): + """Build RefusalHandler if enabled, else None. + + Args: + cfg: RefusalHandlerCfg. enabled=False means the gate is disabled. + + Returns: + RefusalHandler when cfg.enabled, else None. + """ + if not cfg.enabled: + return None + from src.eval.transforms import RefusalHandler + return RefusalHandler( + enabled=True, similarity_threshold=cfg.similarity_threshold, + no_answer_text=cfg.no_answer_text, + ) diff --git a/src/eval/pricing.py b/src/eval/pricing.py new file mode 100644 index 00000000..1327a362 --- /dev/null +++ b/src/eval/pricing.py @@ -0,0 +1,90 @@ +""" +Model pricing table and cost arithmetic. + +Eval Harness Position: + Pipeline → tokens → [PRICING] → cost_usd → EvalResult.cost_usd + ^^^^^^^^ + Pure data + a 4-line function. Hard-coded prices are fine for Phase 1; + if prices change frequently we move to a JSON file in a later phase. + +Design decisions: + - Hard-coded table, not env-driven, so price changes are visible in + git diffs and reviewable in PRs. + - Unknown model returns 0.0 (with a logged warning) rather than + raising — eval should not crash because of an outdated price table; + it should surface the gap in logs. + - Prices in USD per 1M tokens (industry-standard quoting unit). + - Prices intentionally MAY be slightly stale; this is a portfolio + project, not a billing system. Update them when convenient. +""" + +from __future__ import annotations + +import logging +from dataclasses import dataclass + +logger = logging.getLogger(__name__) + + +@dataclass(frozen=True) +class ModelPrice: + """USD per 1M tokens, separate prompt and completion rates.""" + + prompt_per_1m: float + completion_per_1m: float + + +# WHY hard-coded: see module docstring. Update as needed; rare event. +# Source for rates: each provider's public pricing page as of 2026-04. +MODEL_PRICES: dict[str, ModelPrice] = { + # OpenAI + "gpt-5-mini": ModelPrice(prompt_per_1m=0.25, completion_per_1m=2.00), + "gpt-5-nano": ModelPrice(prompt_per_1m=0.05, completion_per_1m=0.40), + "gpt-4.1-mini": ModelPrice(prompt_per_1m=0.40, completion_per_1m=1.60), + "gpt-4.1-nano": ModelPrice(prompt_per_1m=0.10, completion_per_1m=0.40), + "gpt-4o-mini": ModelPrice(prompt_per_1m=0.15, completion_per_1m=0.60), + # Anthropic + "claude-haiku-4-5": ModelPrice(prompt_per_1m=1.00, completion_per_1m=5.00), + "claude-sonnet-4-6": ModelPrice(prompt_per_1m=3.00, completion_per_1m=15.00), + # GLM (Zhipu) + "glm-5.1": ModelPrice(prompt_per_1m=0.50, completion_per_1m=1.50), +} + + +def cost_usd(model: str, prompt_tokens: int, completion_tokens: int) -> float: + """Compute total cost in USD for one inference call. + + Args: + model: Model id matching a key in :data:`MODEL_PRICES`. + prompt_tokens: Input/prompt token count, ``>= 0``. + completion_tokens: Output/completion token count, ``>= 0``. + + Returns: + Total cost in USD. ``0.0`` if the model is unknown (a warning + is logged) — the eval should not crash on missing prices. + + Raises: + ValueError: If either token count is negative. + """ + # PATTERN: validate at the boundary. Internal callers shouldn't + # send negatives, but this function is reachable from the + # API too (via the runner) so we fail loud on bad input. + if prompt_tokens < 0 or completion_tokens < 0: + raise ValueError( + f"Token counts must be non-negative: " + f"prompt={prompt_tokens}, completion={completion_tokens}" + ) + + price = MODEL_PRICES.get(model) + if price is None: + logger.warning( + "Unknown model %r in cost_usd — returning 0.0. " + "Add it to MODEL_PRICES to record real cost.", + model, + ) + return 0.0 + + # Prices are quoted per 1M tokens. + return (prompt_tokens / 1_000_000) * price.prompt_per_1m + ( + completion_tokens / 1_000_000 + ) * price.completion_per_1m diff --git a/src/eval/report.py b/src/eval/report.py new file mode 100644 index 00000000..48e1d5e1 --- /dev/null +++ b/src/eval/report.py @@ -0,0 +1,53 @@ +""" +HTML report renderer — single-run and two-run reports via jinja2. + +Eval Harness Position: + storage.load_run() / compare.compare_runs() → [REPORT] → standalone HTML + +Design decisions: + - Self-contained HTML (inline CSS, no external assets) so a report is + portable — copyable to a gist, attachable to an email, openable from + a checked-out repo. + - jinja2 templates live in templates/eval/ so designers can edit them + without touching Python. + - Templates KEEP IT SIMPLE: tables, basic CSS, no JS. The React UI + (Sub-plan 1C) is the rich interactive surface. +""" + +from __future__ import annotations + +from pathlib import Path +from typing import Any + +from jinja2 import Environment, FileSystemLoader, select_autoescape + +from src.eval.schemas import CompareResult + +TEMPLATES_DIR = Path(__file__).parent.parent.parent / "templates" / "eval" + +_env = Environment( + loader=FileSystemLoader(str(TEMPLATES_DIR)), + autoescape=select_autoescape(["html"]), +) + + +def render_run_html(run: dict[str, Any]) -> str: + """Render a single eval run dict to HTML.""" + template = _env.get_template("run_report.html.j2") + return template.render( + metadata=run["metadata"], + aggregated=run["aggregated"], + results=run["results"], + cost=run["cost"], + ) + + +def render_compare_html(compare: CompareResult) -> str: + """Render a CompareResult to HTML.""" + template = _env.get_template("compare_report.html.j2") + return template.render( + run_a=compare.run_a, + run_b=compare.run_b, + deltas=compare.deltas, + per_question_diff=compare.per_question_diff, + ) diff --git a/src/eval/retrievers/__init__.py b/src/eval/retrievers/__init__.py new file mode 100644 index 00000000..9417a9e5 --- /dev/null +++ b/src/eval/retrievers/__init__.py @@ -0,0 +1,6 @@ +"""Phase 2 retriever package — hybrid sparse/dense retrieval and reranking.""" + +from src.eval.retrievers.bm25_hybrid import BM25HybridRetriever +from src.eval.retrievers.reranker import CrossEncoderReranker + +__all__ = ["BM25HybridRetriever", "CrossEncoderReranker"] diff --git a/src/eval/retrievers/bm25_hybrid.py b/src/eval/retrievers/bm25_hybrid.py new file mode 100644 index 00000000..cba3718e --- /dev/null +++ b/src/eval/retrievers/bm25_hybrid.py @@ -0,0 +1,128 @@ +"""BM25HybridRetriever — Reciprocal Rank Fusion of sparse (BM25) + dense (Chroma) retrieval. + +Pipeline position: + query → [BM25 + Dense → RRF] → top-K SearchResult → Reranker / Generator + +Phase 2 lever 2c. The retriever keeps two parallel ranked lists (BM25 over +documents, dense over Chroma vectors), then fuses them with RRF: + + score(d) = sum over r in {dense, sparse} of 1 / (rrf_k + rank_r(d)) + +Why RRF over weighted-sum: RRF is parameter-light (one constant), robust to +score-scale differences across the two retrievers, and the literature shows +it consistently matches or beats tuned weighted-sum on benchmarks like BEIR. +""" + +from __future__ import annotations + +from typing import Sequence + +from rank_bm25 import BM25Okapi + +from src.vector_store import ChromaVectorStore, SearchResult + + +def reciprocal_rank_fusion( + rankings: Sequence[Sequence[str]], + rrf_k: int = 60, +) -> list[str]: + """Fuse multiple ranked ID lists into one via Reciprocal Rank Fusion. + + Args: + rankings: Iterable of ranked ID sequences. Each sequence is one + retriever's ranking, most-relevant first. + rrf_k: RRF constant (60 is the textbook default; smaller emphasizes + top-rank items more, larger flattens contributions). + + Returns: + Fused ranking, IDs ordered by descending fused score. + """ + scores: dict[str, float] = {} + for ranking in rankings: + for rank, item_id in enumerate(ranking, start=1): + scores[item_id] = scores.get(item_id, 0.0) + 1.0 / (rrf_k + rank) + return sorted(scores.keys(), key=lambda i: scores[i], reverse=True) + + +class BM25HybridRetriever: + """Retriever that fuses BM25 and dense Chroma rankings. + + The BM25 index is built once at construction time over a `documents` mapping. + Each retrieve() call queries both BM25 and the vector store, then RRF-fuses + the two rankings before truncating to the requested top-K. + """ + + def __init__( + self, + vector_store: ChromaVectorStore, + documents: dict[str, str], + bm25_top_k: int = 20, + dense_top_k: int = 20, + rrf_k: int = 60, + ) -> None: + """Build the BM25 index and store retrieval parameters. + + Args: + vector_store: Dense retriever (Chroma collection wrapper). + documents: Mapping of chunk_id → raw document text. BM25 needs + tokenized text; this dict is the authoritative corpus. + bm25_top_k: Number of candidates BM25 returns per query. + dense_top_k: Number of candidates the dense retriever returns. + rrf_k: RRF fusion constant. + """ + self._vector_store = vector_store + self._chunk_ids = list(documents.keys()) + # WHY simple split: rank-bm25 expects pre-tokenized inputs. A whitespace + # split is good enough for English RAG corpora; nltk stems/stopwords + # would help marginally but add a runtime dep we don't want here. + tokenized = [documents[i].lower().split() for i in self._chunk_ids] + self._bm25 = BM25Okapi(tokenized) + self._documents = documents + self._bm25_top_k = bm25_top_k + self._dense_top_k = dense_top_k + self._rrf_k = rrf_k + + def retrieve(self, query: str, top_k: int = 5) -> list[SearchResult]: + """Run BM25 + dense in parallel, RRF-fuse, return top-K SearchResults. + + Args: + query: Natural-language query. + top_k: Number of fused results to return. + + Returns: + Top-K SearchResult ordered by fused score descending. Score on each + result is the dense similarity (BM25 ranks aren't directly comparable; + keeping dense score lets downstream rerankers/refusal-handlers reuse + it as a confidence proxy). + """ + # --- Sparse side ------------------------------------------------------- + sparse_scores = self._bm25.get_scores(query.lower().split()) + sparse_ranked = sorted( + range(len(self._chunk_ids)), + key=lambda i: sparse_scores[i], + reverse=True, + )[: self._bm25_top_k] + sparse_ids = [self._chunk_ids[i] for i in sparse_ranked] + + # --- Dense side -------------------------------------------------------- + dense_results = self._vector_store.query( + query_text=query, top_k=self._dense_top_k, + ) + dense_ids = [r.chunk_id for r in dense_results] + dense_score_by_id = {r.chunk_id: r.score for r in dense_results} + + # --- Fusion ------------------------------------------------------------ + fused_ids = reciprocal_rank_fusion( + [sparse_ids, dense_ids], rrf_k=self._rrf_k, + )[:top_k] + + return [ + SearchResult( + chunk_id=cid, + content=self._documents[cid], + score=dense_score_by_id.get(cid, 0.0), + metadata={}, + doc_id="", + ) + for cid in fused_ids + ] diff --git a/src/eval/retrievers/reranker.py b/src/eval/retrievers/reranker.py new file mode 100644 index 00000000..a79d53da --- /dev/null +++ b/src/eval/retrievers/reranker.py @@ -0,0 +1,62 @@ +"""CrossEncoderReranker — re-scores retrieval candidates with a cross-encoder model. + +Pipeline position: + Retriever top-N → [CrossEncoderReranker] → top-K → Refusal / Generator + +Phase 2 lever 2d. Cross-encoders (single-tower models that consume both +the query and a candidate together) typically outperform bi-encoder retrieval +in precision at the cost of latency. We use ms-marco-MiniLM-L-6-v2 — small +enough to run on CPU in milliseconds per pair, trained on MS MARCO so the +ranking signal transfers well to general-domain QA. +""" + +from __future__ import annotations + +from src.vector_store import SearchResult + + +class CrossEncoderReranker: + """Wraps sentence-transformers CrossEncoder to re-score retrieval candidates.""" + + MODEL_NAME = "cross-encoder/ms-marco-MiniLM-L-6-v2" + + def __init__(self) -> None: + from sentence_transformers import CrossEncoder + + self._model = CrossEncoder(self.MODEL_NAME) + + def rerank( + self, + query: str, + candidates: list[SearchResult], + final_top_k: int, + ) -> list[SearchResult]: + """Re-score candidates against the query and return top-K reranked. + + Args: + query: Original user query. + candidates: Pre-retrieved chunks (typically top-N from a base retriever). + final_top_k: How many to keep after reranking. + + Returns: + Top-K SearchResult ordered by descending cross-encoder score. The + original `score` field is *replaced* with the cross-encoder score so + downstream consumers reading `result.score` get the more precise signal. + """ + if not candidates: + return [] + pairs = [(query, c.content) for c in candidates] + scores = self._model.predict(pairs) + scored = sorted( + zip(candidates, scores), key=lambda t: t[1], reverse=True, + )[:final_top_k] + return [ + SearchResult( + doc_id=c.doc_id, + chunk_id=c.chunk_id, + content=c.content, + score=float(s), + metadata=c.metadata, + ) + for c, s in scored + ] diff --git a/src/eval/runner.py b/src/eval/runner.py new file mode 100644 index 00000000..008511a0 --- /dev/null +++ b/src/eval/runner.py @@ -0,0 +1,350 @@ +""" +EvalRunner — orchestrates a full evaluation run end-to-end. + +Eval Harness Position: + EvalConfig → [EvalRunner] → eval_runs//{metadata,questions,metrics,cost,config} + +Lifecycle: + 1. Resolve git SHA, env hash, run_id. + 2. For each dataset: load questions → build pipeline → query+score each → + teardown. + 3. Aggregate metrics with bootstrap CIs. + 4. Persist via storage.save_run. + +Design decisions: + - Per-question try/except so one broken question doesn't kill the run. + Failures land in EvalResult.error and count toward RunMetadata.n_errors. + - llm_override / judge_llm_override let tests inject DummyLLM without + touching real API calls. + - on_progress callback for the API's status polling endpoint + (added in Sub-plan 1C). +""" + +from __future__ import annotations + +import hashlib +import logging +import subprocess +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Callable + +import yaml + +from src.eval import storage as _storage +from src.eval.aggregator import aggregate +from src.eval.config import EvalConfig +from src.eval.datasets import ml_papers as ml_papers_ds +from src.eval.datasets import squad_v2 as squad_ds +from src.eval.metrics.generation import ( + answer_correctness, + context_recall, + judge_answer_relevancy, + judge_context_precision, + judge_faithfulness, +) +from src.eval.metrics.operational import aggregate_costs, aggregate_tokens +from src.eval.metrics.refusal import refusal_correctness +from src.eval.metrics.retrieval import mrr_at_k, ndcg_at_k, recall_at_k +from src.eval.pipeline_factory import build_pipeline +from src.eval.schemas import EvalQuestion, EvalResult, RunMetadata +from src.eval.storage import compute_run_id, save_run + +logger = logging.getLogger(__name__) + +_K_VALUES = [1, 3, 5, 10] + + +def _sha256_of_file(path: Path) -> str: + """Compute SHA-256 hex digest of a file's bytes. Returns 'unknown' if missing.""" + try: + h = hashlib.sha256() + with path.open("rb") as f: + for chunk in iter(lambda: f.read(65536), b""): + h.update(chunk) + return h.hexdigest() + except OSError: + return "unknown" + + +def _score_question( + question: EvalQuestion, + chunks: list, + answer: str, + judge_llm: Any, +) -> tuple[dict[str, float], dict[str, Any]]: + """Compute all applicable metrics for one question. + + WHY extracted: keeps the per-question try/except in run() slim and + makes metric logic independently testable. + + Args: + question: The gold-labeled question. + chunks: SearchResult list returned by pipeline.query(). + answer: Generated answer string. + judge_llm: LLM judge (real or test double). + + Returns: + Tuple of (metrics dict, metric_details dict). + """ + retrieved_ids = [r.chunk_id for r in chunks] + retrieved_texts = [r.content for r in chunks] + + metrics: dict[str, float] = {} + details: dict[str, Any] = {} + + # --- Always: refusal correctness --- + metrics["refusal_correctness"] = refusal_correctness( + answer, question.is_unanswerable, judge_llm + ) + + # --- Retrieval metrics (only when gold chunk IDs are known) --- + if question.gold_chunk_ids: + for k in _K_VALUES: + metrics[f"recall_at_{k}"] = recall_at_k(question.gold_chunk_ids, retrieved_ids, k) + metrics[f"mrr_at_{k}"] = mrr_at_k(question.gold_chunk_ids, retrieved_ids, k) + metrics[f"ndcg_at_{k}"] = ndcg_at_k(question.gold_chunk_ids, retrieved_ids, k) + + metrics["context_recall"] = context_recall(question.gold_chunk_ids, retrieved_ids) + + # LLM-judge generation metrics + faith_score, faith_details = judge_faithfulness(answer, retrieved_texts, judge_llm) + metrics["judge_faithfulness"] = faith_score + details["judge_faithfulness"] = faith_details + + cp_score, cp_details = judge_context_precision(question.question, retrieved_texts, judge_llm) + metrics["judge_context_precision"] = cp_score + details["judge_context_precision"] = cp_details + + ar_score, ar_details = judge_answer_relevancy(question.question, answer, judge_llm) + metrics["judge_answer_relevancy"] = ar_score + details["judge_answer_relevancy"] = ar_details + + # --- Answer correctness (only when gold answer is known) --- + if question.gold_answer: + ac_score, ac_details = answer_correctness(answer, question.gold_answer, judge_llm) + metrics["answer_correctness"] = ac_score + details["answer_correctness"] = ac_details + + return metrics, details + + +class EvalRunner: + """Orchestrates a full end-to-end evaluation run from config to disk. + + Teaches: the orchestrator pattern — EvalRunner owns the lifecycle + (load → ingest → query → score → aggregate → persist) but delegates + each step to specialized components. This keeps each component + independently testable and swappable. + + Pipeline position: TOP-LEVEL — receives EvalConfig, produces a + run directory under EVAL_RUNS_DIR with five well-known files. + """ + + def __init__( + self, + config: EvalConfig, + *, + config_path: Path | None = None, + llm_override: object | None = None, + judge_llm_override: object | None = None, + on_progress: Callable[[int, int], None] | None = None, + run_id_override: str | None = None, + ) -> None: + self._config = config + self._config_path = str(config_path) if config_path else f"" + self._llm_override = llm_override + self._judge_llm_override = judge_llm_override + self._on_progress = on_progress + # WHY run_id_override: the API pre-computes the run_id so it can register + # the run in RunRegistry BEFORE the runner starts (enabling status polling). + # When set, we use this id instead of computing one from timestamp+sha. + self._run_id_override = run_id_override + + def run(self) -> RunMetadata: + """Execute the full eval lifecycle and return run provenance. + + Returns: + RunMetadata with run_id, timing, error counts, and warnings. + """ + config = self._config + started_at = datetime.now(timezone.utc) + + # --- Git SHA --- + # WHY try/except: the harness may run outside a git repo (CI containers, + # zip-extracted deployments). Fall back to 'unknown' rather than crashing. + try: + git_sha = subprocess.check_output( + ["git", "rev-parse", "HEAD"], text=True + ).strip() + except Exception: + git_sha = "unknown" + + # --- Env hash (requirements.txt fingerprint) --- + env_hash = _sha256_of_file(Path("requirements.txt"))[:16] + + # --- Run ID and directory --- + # WHY: If run_id_override is set (from the API route), use it directly. + # This ensures the registered registry run_id matches the saved directory. + run_id = self._run_id_override or compute_run_id(config.name, started_at, git_sha) + # WHY _storage.EVAL_RUNS_DIR at call time: the fixture reloads storage + # after setting EVAL_RUNS_DIR env var, but runner's top-level import + # already bound the old value. Reading from the live module attribute + # ensures we pick up the reloaded (test-patched) path. + run_dir = _storage.EVAL_RUNS_DIR / run_id + + # --- Eval-set version fingerprints --- + # WHY live attribute read: squad_5 fixture patches DEFAULT_OUTPUT_PATH + # after import; reading the module attribute picks up the patched value. + eval_set_versions: dict[str, str] = {} + for dataset_name in config.eval.datasets: + if dataset_name == "squad_v2_dev_200": + fp = squad_ds.DEFAULT_OUTPUT_PATH + elif dataset_name == "ml_papers_v1": + fp = ml_papers_ds.DEFAULT_QUESTIONS_PATH + else: + fp = None + eval_set_versions[dataset_name] = ( + _sha256_of_file(fp)[:16] if fp is not None else "unknown" + ) + + # --- Load all datasets up front so we know total_questions --- + # WHY pre-load: the progress callback needs total before the first + # on_progress(1, total) call. Eager load also surfaces missing files + # before any pipeline work starts. + dataset_questions: dict[str, list[EvalQuestion]] = {} + for dataset_name in config.eval.datasets: + qs = self._load_questions(dataset_name) + dataset_questions[dataset_name] = qs + + total_questions = sum(len(qs) for qs in dataset_questions.values()) + + # --- Per-dataset pipeline loop --- + all_results: list[EvalResult] = [] + + for dataset_name, questions in dataset_questions.items(): + pipeline = build_pipeline( + config, + dataset_name, + llm_override=self._llm_override, + judge_llm_override=self._judge_llm_override, + ) + try: + pipeline.ingest(questions) + for question in questions: + result = self._run_question( + question, dataset_name, pipeline, pipeline.judge_llm + ) + all_results.append(result) + if self._on_progress is not None: + self._on_progress(len(all_results), total_questions) + # Phase 2: abort if spend ceiling is exceeded. + ceiling = config.eval.spend_ceiling_usd + if ceiling is not None: + cumulative = sum(r.cost_usd for r in all_results) + if cumulative > ceiling: + raise RuntimeError( + f"Spend ceiling exceeded: ${cumulative:.4f} > ${ceiling:.4f} " + f"after {len(all_results)} questions. Aborting run." + ) + finally: + # WHY finally: ensures teardown even if a question raises + # an unhandled exception outside the per-question try block. + pipeline.teardown() + + # --- Aggregate and persist --- + aggregated, warnings = aggregate(all_results, config) + cost_summary = {**aggregate_costs(all_results), **aggregate_tokens(all_results)} + finished_at = datetime.now(timezone.utc) + + metadata = RunMetadata( + run_id=run_id, + config_name=config.name, + config_path=self._config_path, + git_sha=git_sha, + started_at=started_at, + finished_at=finished_at, + env_hash=env_hash, + eval_set_versions=eval_set_versions, + n_questions=len(all_results), + n_errors=sum(1 for r in all_results if r.error), + warnings=warnings, + ) + + config_yaml_text = yaml.safe_dump(config.model_dump()) + save_run(run_dir, metadata, all_results, aggregated, cost_summary, config_yaml_text) + + return metadata + + def _load_questions(self, dataset_name: str) -> list[EvalQuestion]: + """Load questions for one dataset, with graceful fallback for empty sets. + + WHY live attribute reads (e.g. squad_ds.DEFAULT_OUTPUT_PATH): test + fixtures patch the module attribute after import; calling + load_frozen(squad_ds.DEFAULT_OUTPUT_PATH) reads the patched value, + whereas load_frozen() would use the default arg bound at def-time. + """ + if dataset_name == "squad_v2_dev_200": + try: + return squad_ds.load_frozen(squad_ds.DEFAULT_OUTPUT_PATH) + except FileNotFoundError as exc: + logger.warning("SQuAD dataset not found: %s — skipping.", exc) + return [] + + if dataset_name == "ml_papers_v1": + try: + questions = ml_papers_ds.load_questions(ml_papers_ds.DEFAULT_QUESTIONS_PATH) + except FileNotFoundError as exc: + logger.warning("ML Papers questions not found: %s — skipping.", exc) + return [] + if not questions: + logger.warning("ML Papers dataset is empty (skeleton state) — skipping.") + return questions + + logger.warning("Unknown dataset %r — skipping.", dataset_name) + return [] + + def _run_question( + self, + question: EvalQuestion, + dataset_name: str, + pipeline: Any, + judge_llm: Any, + ) -> EvalResult: + """Query the pipeline and score one question; capture any exception as error. + + WHY outer try/except: a single malformed question or judge response + must not abort the entire run. The error lands in EvalResult.error + and increments RunMetadata.n_errors; the run continues. + """ + try: + chunks, answer, telemetry = pipeline.query(question.question) + metrics, metric_details = _score_question(question, chunks, answer, judge_llm) + return EvalResult( + question_id=question.id, + dataset=dataset_name, + retrieved_chunk_ids=[r.chunk_id for r in chunks], + retrieved_chunks=[r.content for r in chunks], + generated_answer=answer, + metrics=metrics, + metric_details=metric_details, + timings_ms=telemetry["timings_ms"], + tokens=telemetry["tokens"], + cost_usd=telemetry["cost_usd"], + error=None, + ) + except Exception as exc: + logger.exception("Error on question %s: %s", question.id, exc) + return EvalResult( + question_id=question.id, + dataset=dataset_name, + retrieved_chunk_ids=[], + retrieved_chunks=[], + generated_answer="", + metrics={}, + metric_details={}, + timings_ms={}, + tokens={}, + cost_usd=0.0, + error=str(exc), + ) diff --git a/src/eval/schemas.py b/src/eval/schemas.py new file mode 100644 index 00000000..ca28a760 --- /dev/null +++ b/src/eval/schemas.py @@ -0,0 +1,163 @@ +""" +Pydantic schemas for the RAG eval harness. + +Eval Harness Position: + EvalRunner → Pipeline → Metrics → AggregatedMetric → Storage + ^^^ ^^^ ^^^^^^^^^^^^^^^^^^^^ + These types are the contracts that tie the layers together. Every + layer reads/writes one of these models — no untyped dicts cross + module boundaries. + +Design decisions: + - Pydantic v2 over plain dataclasses: free JSON round-trip (we + persist results as JSON Lines) and field validation at boundaries. + - frozen=True on EvalQuestion to prevent accidental mutation after + a labeled gold set is loaded — ground truth is immutable. + - Modern T | None syntax (Python 3.10+). +""" + +from __future__ import annotations + +from datetime import datetime +from typing import Any + +from pydantic import BaseModel, ConfigDict, Field, model_validator + + +class EvalQuestion(BaseModel): + """A single labeled question from a gold eval set. + + Teaches: immutable value objects via frozen=True. The gold set is + ground truth — no runtime code should mutate it after load. + + Pipeline role: INPUT to EvalRunner; defines the question + expected + answer that all metrics are computed against. + """ + + model_config = ConfigDict(frozen=True) + + id: str + question: str + gold_answer: str | None = None + gold_chunk_ids: list[str] = Field(default_factory=list) + is_unanswerable: bool = False + metadata: dict[str, Any] = Field(default_factory=dict) + + +class EvalResult(BaseModel): + """Output record for a single question evaluation run. + + Teaches: JSON Lines persistence pattern — one record per question, + serialized with model_dump_json() and stored line-by-line so large + eval runs can be streamed without loading the full set into memory. + + Pipeline role: OUTPUT of EvalRunner per question; INPUT to the + aggregation layer that computes AggregatedMetric summaries. + """ + + question_id: str + dataset: str + retrieved_chunk_ids: list[str] + retrieved_chunks: list[str] + generated_answer: str + metrics: dict[str, float] + metric_details: dict[str, Any] = Field(default_factory=dict) + timings_ms: dict[str, float] + tokens: dict[str, int] + cost_usd: float + # Phase 2: per-bucket breakdown. Defaults to generator-only when absent so + # Phase 1 records continue to round-trip through model_validate. + cost_breakdown: dict[str, float] = Field(default_factory=dict) + error: str | None = None + + @model_validator(mode="after") + def _backfill_cost_breakdown(self) -> "EvalResult": + if not self.cost_breakdown: + self.cost_breakdown = { + "generator": self.cost_usd, + "judge": 0.0, + "rewriter": 0.0, + } + return self + + +class AggregatedMetric(BaseModel): + """Aggregate statistics for one metric across a dataset (or all datasets). + + Teaches: confidence intervals as first-class citizens — never report + a mean without CI bounds. n is explicit so downstream callers can + detect low-sample results and weight them appropriately. + + Pipeline role: OUTPUT of the aggregation layer; INPUT to the + reporting and comparison layers. + """ + + metric_name: str + dataset: str | None = Field(default=None, description="None = combined across datasets") + mean: float + ci_low: float + ci_high: float + n: int + + +class RunMetadata(BaseModel): + """Provenance record for a complete eval run. + + Teaches: reproducibility by design — every run captures the git SHA, + config path, and env hash so results can be traced back to exact + code + config + environment. This is the audit trail. + + Pipeline role: Written once per run; attached to CompareResult so + A/B comparisons always carry full provenance for both runs. + """ + + run_id: str + config_name: str + config_path: str + git_sha: str + started_at: datetime + finished_at: datetime + env_hash: str + eval_set_versions: dict[str, str] + n_questions: int + n_errors: int + warnings: list[str] = Field(default_factory=list) + + +class MetricDelta(BaseModel): + """Statistical comparison of one metric between two runs. + + Teaches: significance testing as a schema concern — delta alone is + misleading without a p-value and CI bounds. Consumers should gate + decisions on significant=True, not raw delta magnitude. + + Pipeline role: Element of CompareResult.deltas; one per + (metric, dataset) pair being compared. + """ + + metric_name: str + dataset: str | None = None + a_mean: float + a_ci: tuple[float, float] + b_mean: float + b_ci: tuple[float, float] + delta: float + p_value: float + significant: bool + + +class CompareResult(BaseModel): + """Full A/B comparison between two eval runs. + + Teaches: structured comparison output — embedding run provenance + (RunMetadata) directly in the result means the comparison is + self-contained and doesn't require external lookups to interpret. + + Pipeline role: Terminal output of the comparison layer; consumed + by reporting tools and the CI gate that blocks regressions. + """ + + run_a: RunMetadata + run_b: RunMetadata + deltas: list[MetricDelta] + per_question_diff: list[dict[str, Any]] = Field(default_factory=list) diff --git a/src/eval/statistics.py b/src/eval/statistics.py new file mode 100644 index 00000000..a99f3e7e --- /dev/null +++ b/src/eval/statistics.py @@ -0,0 +1,149 @@ +""" +Statistical wrappers — bootstrap confidence intervals and paired +permutation tests for eval-run comparison. + +Eval Harness Position: + per-question scores → [STATISTICS] → AggregatedMetric (with CIs) + → MetricDelta (with p-values) + +Design decisions: + - Bootstrap percentile method (not BCa) — simpler, well-understood, + sufficient for the precision we report. With n=200 samples and + n_resamples=1000 the CI is stable to ~1%. + - NaN values are dropped per-metric per-call: a question that didn't + have a defined metric (e.g. recall_at_k with empty gold) doesn't + poison the aggregate. + - All randomness is seeded; identical inputs produce identical CIs. +""" + +from __future__ import annotations + +import numpy as np + + +def bootstrap_ci( + values: list[float], + n_resamples: int = 1000, + seed: int = 12345, +) -> tuple[float, float, float]: + """Compute the sample mean and a 95% bootstrap percentile confidence interval. + + Bootstrap resampling draws ``n_resamples`` samples of size ``n`` with + replacement from ``arr``, computes the mean of each resample, and then + takes the 2.5th and 97.5th percentiles of those means as the CI bounds. + + Args: + values: Raw per-question metric scores. May contain NaN — they are + dropped before resampling. + n_resamples: Number of bootstrap resamples. Higher values reduce + Monte-Carlo noise in the CI bounds but cost more compute. + 1000 is stable to ~1% for n ≥ 50. + seed: RNG seed for reproducibility. Identical seed + values produce + identical output. + + Returns: + A three-tuple ``(mean, ci_low, ci_high)`` where: + - ``mean`` is the sample mean of the non-NaN values, + - ``ci_low`` is the 2.5th percentile of resampled means (lower CI), + - ``ci_high`` is the 97.5th percentile of resampled means (upper CI). + + Raises: + ValueError: If all values are NaN and no valid observations remain. + """ + # WHY: Convert to float64 array first so NaN detection via np.isnan works + # uniformly regardless of the input list's original dtype. + arr = np.array(values, dtype=np.float64) + + # Drop NaN entries: a missing metric on one question shouldn't skew the + # aggregate or cause all-NaN resamples. + arr = arr[~np.isnan(arr)] + + if arr.size == 0: + raise ValueError("Cannot compute bootstrap CI on all-NaN input") + + # PATTERN: Short-circuit for single values — resampling a length-1 array + # always produces the same mean, so skip the loop. + if arr.size == 1: + v = float(arr[0]) + return (v, v, v) + + n = arr.size + + # WHY: default_rng (PCG64) is the recommended NumPy RNG since 1.17. + # It is faster and statistically superior to np.random.seed(). + rng = np.random.default_rng(seed) + + # Generate all resample index matrices in one shot: shape (n_resamples, n). + # This vectorised form avoids a Python-level loop and is ~10–50× faster + # than calling rng.choice() inside a loop. + indices = rng.integers(0, n, size=(n_resamples, n)) + + # TRADE-OFF: arr[indices] materialises an (n_resamples, n) float64 array + # in memory. For n=200 and n_resamples=1000 that's 200 KB — negligible. + # For n=100k you would want chunked computation; not needed here. + resampled_means = arr[indices].mean(axis=1) + + mean = float(arr.mean()) + ci_low = float(np.percentile(resampled_means, 2.5)) + ci_high = float(np.percentile(resampled_means, 97.5)) + + return (mean, ci_low, ci_high) + + +def paired_permutation_test( + a: list[float], + b: list[float], + n_resamples: int = 10000, + seed: int = 12345, +) -> tuple[float, float]: + """Two-sided paired permutation test on the mean difference. + + The null hypothesis is that the labels "a" and "b" are exchangeable + on a per-pair basis (i.e. swapping ``a[i]`` and ``b[i]`` for any + subset of indices doesn't change the joint distribution). We + estimate the p-value by randomly sign-flipping each paired + difference and counting how often the resampled mean difference + is at least as extreme as the observed one. + + Args: + a: First sample (e.g. per-question metric for run A). + b: Second sample, paired one-to-one with ``a``. + n_resamples: Number of permutation draws. + seed: PRNG seed for reproducibility. + + Returns: + ``(observed_delta, p_value)`` where ``observed_delta`` is the + mean of ``b[i] - a[i]`` over surviving pairs and ``p_value`` + is two-sided. + + Raises: + ValueError: If ``a`` and ``b`` have different lengths or no + pairs survive NaN-drop. + """ + if len(a) != len(b): + raise ValueError( + f"Paired samples must have equal length: len(a)={len(a)}, len(b)={len(b)}" + ) + arr_a = np.asarray(a, dtype=float) + arr_b = np.asarray(b, dtype=float) + # Drop pairs where either side is NaN. + mask = ~(np.isnan(arr_a) | np.isnan(arr_b)) + diffs = arr_b[mask] - arr_a[mask] + if diffs.size == 0: + raise ValueError("No surviving pairs after NaN-drop") + + observed = float(diffs.mean()) + + # WHY sign-flips: under the exchangeability null, swapping a[i] and + # b[i] negates the i-th difference. Random sign-flips draw from + # the exact null distribution of the mean difference. + rng = np.random.default_rng(seed) + signs = rng.choice([-1.0, 1.0], size=(n_resamples, diffs.size)) + resampled_means = (signs * diffs).mean(axis=1) + + # Two-sided p-value with Phipson & Smyth (2010) correction (+1/+1) + # to prevent meaningless p == 0.0. + extreme = np.sum(np.abs(resampled_means) >= abs(observed)) + p_value = float((extreme + 1) / (n_resamples + 1)) + + return observed, p_value diff --git a/src/eval/storage.py b/src/eval/storage.py new file mode 100644 index 00000000..dc247b5e --- /dev/null +++ b/src/eval/storage.py @@ -0,0 +1,217 @@ +""" +Storage layer — read/write eval run directories. + +Eval Harness Position: + EvalRunner → save_run() → eval_runs//{metadata,questions,metrics,cost,config} + CLI/API → list_runs() / load_run() → render + +Design decisions: + - One directory per run, with five well-known files. Plain JSON / JSONL + so any tool (jq, pandas, the eye) can inspect a run. + - EVAL_RUNS_DIR is env-overridable so tests use tmp dirs without + touching the user's real eval_runs/. + - delete_run refuses path traversal — destructive operations get a + safety check at the boundary. +""" + +from __future__ import annotations + +import json +import os +import shutil +from datetime import datetime +from pathlib import Path +from typing import Any + +from src.eval.schemas import AggregatedMetric, EvalResult, RunMetadata + +# WHY: env-overridable so pytest can point at a temp dir without touching real data. +EVAL_RUNS_DIR = Path(os.getenv("EVAL_RUNS_DIR", "eval_runs")) + + +def compute_run_id(config_name: str, started_at: datetime, git_sha: str) -> str: + """Build a human-readable, sortable run ID. + + Format: YYYY-MM-DD_HHMMSS__ + + Args: + config_name: Name of the eval config (e.g. "baseline"). + started_at: UTC datetime the run began. + git_sha: Full or partial git SHA of the current commit. + + Returns: + Deterministic string suitable for use as a directory name. + + Teaches: + Sortable directory names — lexicographic order == chronological + order because the timestamp is the leading component. This makes + `ls -1 eval_runs/` an implicit run history without any index file. + """ + ts = started_at.strftime("%Y-%m-%d_%H%M%S") + sha7 = git_sha[:7] + return f"{ts}_{config_name}_{sha7}" + + +def save_run( + run_dir: Path, + metadata: RunMetadata, + results: list[EvalResult], + aggregated: list[AggregatedMetric], + cost: dict[str, Any], + config_yaml_text: str, +) -> None: + """Persist all artifacts for one eval run to disk. + + Writes five files into run_dir (creating it and parents if needed): + - metadata.json — RunMetadata as pretty JSON + - questions.jsonl — one EvalResult per line (JSON Lines) + - metrics.json — list of AggregatedMetric as pretty JSON + - cost.json — cost summary dict as pretty JSON + - config.yaml — raw YAML text of the config that drove this run + + Args: + run_dir: Destination directory (created if absent). + metadata: Provenance record for the run. + results: Per-question evaluation outputs. + aggregated: Metric aggregates across the run. + cost: Cost summary (total_usd, mean_usd_per_query, etc.). + config_yaml_text: Raw YAML string of the EvalConfig used. + + Teaches: + JSON Lines (JSONL) for streaming — questions.jsonl can be read + one line at a time for arbitrarily large eval runs, unlike a + monolithic JSON array that must be fully parsed before any record + is accessible. + """ + # PATTERN: mkdir(parents=True, exist_ok=True) is the idiomatic way to + # ensure a directory exists without racing on creation. + run_dir.mkdir(parents=True, exist_ok=True) + + # metadata.json — Pydantic's model_dump_json handles datetime serialization. + (run_dir / "metadata.json").write_text(metadata.model_dump_json(indent=2)) + + # questions.jsonl — one record per line; empty file for zero results. + with (run_dir / "questions.jsonl").open("w") as fh: + for result in results: + fh.write(result.model_dump_json() + "\n") + + # metrics.json — list of AggregatedMetric dicts; default=str handles any + # non-JSON-native types (e.g. numpy floats) gracefully. + metrics_data = [am.model_dump() for am in aggregated] + (run_dir / "metrics.json").write_text( + json.dumps(metrics_data, indent=2, default=str) + ) + + # cost.json — plain dict; default=str for safety. + (run_dir / "cost.json").write_text(json.dumps(cost, indent=2, default=str)) + + # config.yaml — raw text, no parsing needed at write time. + (run_dir / "config.yaml").write_text(config_yaml_text) + + +def load_run(run_id: str) -> dict: + """Load all artifacts for a run from disk. + + Args: + run_id: Directory name under EVAL_RUNS_DIR. + + Returns: + Dict with keys: + - "metadata" → RunMetadata + - "results" → list[EvalResult] + - "aggregated" → list[AggregatedMetric] + - "cost" → dict + + Raises: + FileNotFoundError: If EVAL_RUNS_DIR / run_id does not exist. + + Teaches: + model_validate_json vs model_validate — use model_validate_json + when reading raw JSON strings (avoids an intermediate parse step), + model_validate when you already have a Python dict/list. + """ + run_dir = EVAL_RUNS_DIR / run_id + if not run_dir.exists(): + raise FileNotFoundError(f"Run {run_id} not found at {run_dir}") + + metadata = RunMetadata.model_validate_json( + (run_dir / "metadata.json").read_text() + ) + + # JSONL: skip blank lines to handle trailing newlines robustly. + results = [ + EvalResult.model_validate_json(line) + for line in (run_dir / "questions.jsonl").read_text().splitlines() + if line.strip() + ] + + aggregated = [ + AggregatedMetric.model_validate(d) + for d in json.loads((run_dir / "metrics.json").read_text()) + ] + + cost = json.loads((run_dir / "cost.json").read_text()) + + return { + "metadata": metadata, + "results": results, + "aggregated": aggregated, + "cost": cost, + } + + +def list_runs() -> list[RunMetadata]: + """Enumerate all valid eval runs in EVAL_RUNS_DIR. + + A valid run is a subdirectory containing metadata.json. Directories + without metadata.json (e.g. incomplete or interrupted runs) are silently + skipped. + + Returns: + RunMetadata instances sorted by started_at descending (newest first). + + Teaches: + Convention over configuration — no index file is needed because + the filesystem *is* the index. Any directory with metadata.json + is a valid run; the rest are ignored. + """ + if not EVAL_RUNS_DIR.exists(): + return [] + + runs: list[RunMetadata] = [] + for entry in EVAL_RUNS_DIR.iterdir(): + if not entry.is_dir(): + continue + metadata_file = entry / "metadata.json" + if not metadata_file.exists(): + # TRADE-OFF: silently skip incomplete runs rather than raising. + # A corrupted run shouldn't block listing all other runs. + continue + runs.append(RunMetadata.model_validate_json(metadata_file.read_text())) + + # Descending by started_at so the most recent run appears first. + runs.sort(key=lambda r: r.started_at, reverse=True) + return runs + + +def delete_run(run_id: str) -> None: + """Permanently delete a run directory. + + Args: + run_id: Directory name under EVAL_RUNS_DIR. + + Raises: + ValueError: If run_id contains path traversal characters ('..' or '/'). + FileNotFoundError: If the run directory does not exist (from shutil.rmtree). + + Teaches: + SECURITY: validate before act — check for traversal *before* any + filesystem call. An attacker supplying run_id="../../../etc" must + be rejected at the boundary, not after the path is constructed. + """ + # SECURITY: reject path traversal before touching the filesystem. + if ".." in run_id or "/" in run_id or "\\" in run_id: + raise ValueError(f"Invalid run_id: {run_id}") + + run_dir = EVAL_RUNS_DIR / run_id + shutil.rmtree(run_dir) diff --git a/src/eval/transforms/__init__.py b/src/eval/transforms/__init__.py new file mode 100644 index 00000000..1dc285e7 --- /dev/null +++ b/src/eval/transforms/__init__.py @@ -0,0 +1,6 @@ +"""Phase 2 transforms — pre/post pipeline hooks (rewriter, refusal handler).""" + +from src.eval.transforms.query_rewriter import QueryRewriter +from src.eval.transforms.refusal_handler import RefusalHandler + +__all__ = ["QueryRewriter", "RefusalHandler"] diff --git a/src/eval/transforms/query_rewriter.py b/src/eval/transforms/query_rewriter.py new file mode 100644 index 00000000..cc5eb2d9 --- /dev/null +++ b/src/eval/transforms/query_rewriter.py @@ -0,0 +1,107 @@ +"""QueryRewriter — LLM-based query expansion with token/cost capture. + +Pipeline position: + user query → [QueryRewriter] → {q, q', q''} → Retriever → ... + +Phase 2 lever 2e. Expansion gives the retriever multiple lexical/semantic +formulations of the same intent, which raises recall on questions where the +original phrasing diverges from the corpus phrasing. We use a tiny model +(gpt-4.1-nano) because the task is cheap and we don't want this lever to +dominate the cost ledger. +""" + +from __future__ import annotations + +import json +import logging +import re +from typing import Protocol + +from src.eval import pricing + +logger = logging.getLogger(__name__) + + +class _LLMHandler(Protocol): + """Structural type for any object exposing generate_with_usage.""" + + def generate_with_usage( + self, prompt: str, system_prompt: str | None = None, + ) -> tuple[str, int, int]: ... + + +class QueryRewriter: + """Expands one user query into up to N alternative phrasings via an LLM.""" + + SYSTEM_PROMPT = ( + "You rewrite user search queries into alternative phrasings that preserve " + "the original intent but vary surface form. Respond ONLY with a JSON " + "array of strings — no prose, no code fences." + ) + + def __init__( + self, + model: str | None, + max_expansions: int, + llm: _LLMHandler | None, + ) -> None: + """Configure the rewriter. + + Args: + model: LLM model name. None disables rewriting (pass-through). + max_expansions: Cap on the number of alternative phrasings to return. + llm: Object exposing generate_with_usage(prompt, system_prompt). Required + if model is not None. + """ + self._model = model + self._max_expansions = max_expansions + self._llm = llm + + def expand(self, query: str) -> tuple[list[str], float, int, int]: + """Expand `query` into up to N+1 unique phrasings. + + Returns: + (queries, cost_usd, prompt_tokens, completion_tokens). The original + query is always the first element. When `model is None`, returns + ([query], 0.0, 0, 0) and skips the LLM call. + """ + if self._model is None: + return [query], 0.0, 0, 0 + if self._llm is None: + raise ValueError("QueryRewriter has model set but no llm handler provided.") + + user_prompt = ( + f'Original query: "{query}"\n\n' + f"Return a JSON array of up to {self._max_expansions} alternative " + f"phrasings of this query. Do NOT include the original." + ) + raw, p_t, c_t = self._llm.generate_with_usage( + user_prompt, system_prompt=self.SYSTEM_PROMPT, + ) + cost = pricing.cost_usd(self._model, p_t, c_t) + + expansions = self._parse_expansions(raw) + # Always lead with original; dedupe; cap at original + max_expansions. + ordered: list[str] = [query] + for alt in expansions: + if alt and alt not in ordered: + ordered.append(alt) + if len(ordered) >= self._max_expansions + 1: + break + return ordered, cost, p_t, c_t + + @staticmethod + def _parse_expansions(raw: str) -> list[str]: + """Strip code fences and parse the JSON array; return [] on failure.""" + stripped = re.sub(r"^```(?:json)?\s*", "", raw.strip()) + stripped = re.sub(r"\s*```$", "", stripped).strip() + try: + parsed = json.loads(stripped) + except json.JSONDecodeError: + logger.warning( + "QueryRewriter got non-JSON response — falling back to [query] only." + ) + return [] + if not isinstance(parsed, list): + return [] + return [str(item) for item in parsed if isinstance(item, str)] diff --git a/src/eval/transforms/refusal_handler.py b/src/eval/transforms/refusal_handler.py new file mode 100644 index 00000000..5e2bb1be --- /dev/null +++ b/src/eval/transforms/refusal_handler.py @@ -0,0 +1,61 @@ +"""RefusalHandler — answerability gate based on top-1 retrieval similarity. + +Pipeline position: + Retriever (post-rerank) candidates → [RefusalHandler] → answer or refusal text + +Phase 2 lever 2g. SQuAD v2 includes 'unanswerable' questions whose gold +answer is the empty string. Phase 1's pipeline always tries to answer, +which means it scores poorly on `refusal_correctness`. RefusalHandler is +a deterministic short-circuit: when no candidate clears the similarity +threshold, return a fixed no-answer text instead of calling the LLM. +""" + +from __future__ import annotations + +from src.vector_store import SearchResult + + +class RefusalHandler: + """Deterministic answerability gate driven by top-1 similarity score.""" + + def __init__( + self, + enabled: bool, + similarity_threshold: float, + no_answer_text: str, + ) -> None: + """Configure the gate. + + Args: + enabled: When False, should_refuse always returns False. + similarity_threshold: Top-1 score must be >= this to NOT refuse. + no_answer_text: Text returned in place of an LLM answer on refusal. + """ + self._enabled = enabled + self._threshold = similarity_threshold + self._no_answer_text = no_answer_text + + def should_refuse(self, candidates: list[SearchResult]) -> bool: + """Return True if the pipeline should short-circuit to no-answer text. + + Args: + candidates: Retrieved chunks ordered by descending similarity. + May be empty. + + Returns: + True when the handler is enabled and the top-1 score is below + the threshold (or candidates is empty); False otherwise. + """ + if not self._enabled: + return False + if not candidates: + return True + return candidates[0].score < self._threshold + + def refuse_response(self) -> tuple[list[SearchResult], str]: + """Return ([], no_answer_text) — used when should_refuse is True. + + Returns: + A 2-tuple of (empty chunk list, configured no-answer text). + """ + return [], self._no_answer_text diff --git a/src/llm_handler.py b/src/llm_handler.py index 9589c53b..52864342 100644 --- a/src/llm_handler.py +++ b/src/llm_handler.py @@ -178,6 +178,33 @@ def generate_with_context(self, query: str, context: str) -> str: user_prompt = f"Context:\n{context}\n\nQuestion: {query}\n\nAnswer:" return self.generate(user_prompt, system_prompt=system_prompt) + def generate_with_usage( + self, + prompt: str, + system_prompt: str | None = None, + ) -> tuple[str, int, int]: + """Generate a response and return text plus prompt/completion token counts. + + WHY a separate method: the existing `generate()` returns only `str` and is + called in many places that don't need usage. Phase 2's cost ledger needs + token counts on every LLM call; rather than break callers, we add a parallel + method that uses the existing tokenizer to estimate counts client-side. + + Args: + prompt: User message. + system_prompt: Optional system instructions. + + Returns: + (response_text, prompt_tokens, completion_tokens). + """ + from src.eval._telemetry import count_tokens + + text = self.generate(prompt, system_prompt=system_prompt) + full_prompt = (system_prompt + "\n" + prompt) if system_prompt else prompt + prompt_tokens = count_tokens(full_prompt, self.model) + completion_tokens = count_tokens(text, self.model) + return text, prompt_tokens, completion_tokens + def stream_response( self, prompt: str, diff --git a/src/observability.py b/src/observability.py new file mode 100644 index 00000000..c6e3dda2 --- /dev/null +++ b/src/observability.py @@ -0,0 +1,189 @@ +""" +Observability — OpenTelemetry tracer initialization and per-stage +span decorator. + +API Layer Position: + RAGBackend.query → @traced_stage("rag.retrieve") → span exported via OTLP + → Phoenix UI at :6006 + +Design decisions: + - Idempotent init via a module-level flag — multiple imports / lifespan + callbacks won't double-install processors. + - Fail QUIETLY on exporter / endpoint errors. Eval and chat must work + when Phoenix is down; observability is opt-in. + - traced_stage uses the (payload, attrs) return convention so existing + return shapes don't change at call sites; the decorator strips attrs. + - Attribute coercion: OTel restricts attribute types. Non-primitives get + str()-coerced rather than dropped (visibility > strictness). +""" + +from __future__ import annotations + +import logging +import os +from functools import wraps +from typing import Any, Callable, TypeVar + +logger = logging.getLogger(__name__) + +TRACER_NAME = "rag-qa" +_INITIALIZED = False + +# OTel primitive types that span.set_attribute accepts as scalars. +_OTEL_SCALARS = (str, int, float, bool) + + +def init_observability(otlp_endpoint: str | None = None) -> None: + """Initialize global TracerProvider + OTLP exporter. + + Idempotent — safe to call multiple times. + ``otlp_endpoint`` defaults to the ``OTLP_ENDPOINT`` env var, or + ``http://localhost:6006/v1/traces`` when neither is set. + + On import errors or connection failures, fails QUIETLY (logs warning, + spans become no-ops). The system never crashes due to OTel. + + Args: + otlp_endpoint: OTLP HTTP traces endpoint URL. Pass ``None`` to + use the default derived from the environment variable. + """ + # PATTERN: module-level flag makes this idempotent — calling from + # multiple lifespan callbacks or test setups is safe. + global _INITIALIZED + if _INITIALIZED: + return + _INITIALIZED = True + + endpoint = otlp_endpoint or os.getenv( + "OTLP_ENDPOINT", "http://localhost:6006/v1/traces" + ) + + try: + from opentelemetry.sdk.trace import TracerProvider + from opentelemetry.sdk.trace.export import BatchSpanProcessor + from opentelemetry.exporter.otlp.proto.http.trace_exporter import ( + OTLPSpanExporter, + ) + import opentelemetry.trace as otel_trace + + # WHY: OTLPSpanExporter is lazy — it won't attempt a connection + # until the first batch is flushed, so construction never raises + # even if the endpoint is unreachable. + exporter = OTLPSpanExporter(endpoint=endpoint) + provider = TracerProvider() + provider.add_span_processor(BatchSpanProcessor(exporter)) + otel_trace.set_tracer_provider(provider) + + except Exception as exc: # noqa: BLE001 + # TRADE-OFF: We catch broadly here because we never want Phoenix + # being unavailable to crash the RAG service. A warning is enough. + logger.warning( + "Observability init failed — spans will be no-ops. Reason: %s", exc + ) + + +def get_tracer(): + """Return the rag-qa tracer (always works; no-op if init never called). + + Returns: + An ``opentelemetry.trace.Tracer`` instance bound to TRACER_NAME. + When no provider has been installed this returns the global no-op + tracer, so call sites never need to guard for None. + """ + import opentelemetry.trace as otel_trace + + # WHY: get_tracer() delegates to whatever TracerProvider is currently + # registered globally. If init_observability was never called that's + # the SDK default (no-op), which is perfectly fine. + return otel_trace.get_tracer(TRACER_NAME) + + +def _coerce_attr(value: Any) -> Any: + """Coerce a value to an OTel-legal span attribute type. + + OTel accepts: str, int, float, bool, or homogeneous Sequence thereof. + Everything else is str()-converted so we retain visibility over dropping. + + Args: + value: Raw attribute value from the decorated function. + + Returns: + A value safe to pass to ``span.set_attribute``. + """ + if isinstance(value, _OTEL_SCALARS): + return value + + # Sequences: pass through only when every element is a scalar of the + # same OTel-legal type. Mixed or complex elements fall back to str(). + if isinstance(value, (list, tuple)): + if value and all(isinstance(el, _OTEL_SCALARS) for el in value): + # OTel SDK coerces list → tuple internally; list is fine here. + return value + # Empty or mixed-type list — convert whole thing. + return str(value) + + # Dicts, None, and anything else: str-coerce for visibility. + return str(value) + + +def traced_stage(name: str): + """Decorator that opens a span around the wrapped function. + + The wrapped function MUST return ``(payload, attrs_dict)``. The + decorator opens a span named ``name``, calls the function, records + each ``attrs_dict`` entry as a span attribute, then returns only + ``payload`` — callers see the same shape as before instrumentation. + + On exception the span is marked ERROR and the exception is re-raised. + + Args: + name: Span name (e.g. ``"rag.retrieve"``). + + Returns: + A decorator that wraps ``f(*args, **kwargs) -> tuple[Any, dict]`` + and exposes only the payload to callers. + + Example:: + + @traced_stage("rag.retrieve") + def retrieve(query: str): + chunks = _do_search(query) + return chunks, {"chunk_count": len(chunks)} + + result = retrieve("what is RAG?") # returns chunks directly + """ + def decorator(f: Callable) -> Callable: + @wraps(f) + def wrapper(*args: Any, **kwargs: Any) -> Any: + tracer = get_tracer() + with tracer.start_as_current_span(name) as span: + try: + payload, attrs = f(*args, **kwargs) + except Exception as exc: + # PATTERN: Record exception details on the span so + # Phoenix shows the stack trace, then propagate. + from opentelemetry.trace import Status, StatusCode + span.record_exception(exc) + span.set_status(Status(StatusCode.ERROR)) + raise + + # Set attributes after successful return. + for key, raw_val in attrs.items(): + coerced = _coerce_attr(raw_val) + try: + span.set_attribute(key, coerced) + except Exception: # noqa: BLE001 + logger.warning( + "traced_stage: could not set attribute %r=%r, skipping", + key, + coerced, + ) + + from opentelemetry.trace import Status, StatusCode + span.set_status(Status(StatusCode.OK)) + + return payload + + return wrapper + + return decorator diff --git a/templates/eval/.gitkeep b/templates/eval/.gitkeep new file mode 100644 index 00000000..e69de29b diff --git a/templates/eval/compare_report.html.j2 b/templates/eval/compare_report.html.j2 new file mode 100644 index 00000000..35974eb4 --- /dev/null +++ b/templates/eval/compare_report.html.j2 @@ -0,0 +1,73 @@ + + + + +Eval Compare: {{ run_a.run_id }} vs {{ run_b.run_id }} + + + +

Eval Comparison

+

A: {{ run_a.run_id }} ({{ run_a.config_name }})   + B: {{ run_b.run_id }} ({{ run_b.config_name }})

+ +

Metric Deltas

+ + + + + + + + {% for d in deltas %} + + + + + + + + + + {% endfor %} + +
MetricDatasetA meanB meanΔ (B − A)p
{{ d.metric_name }}{{ d.dataset or "" }}{{ "%.4f"|format(d.a_mean) }}{{ "%.4f"|format(d.b_mean) }} + {{ "%+.4f"|format(d.delta) }} + {{ "%.4f"|format(d.p_value) }}{% if d.significant %}★{% endif %}
+

★ = p < 0.05 (paired permutation test, n=10000)

+ +

Top Per-Question Differences

+ + + + + + + {% for d in per_question_diff %} + + + + + + + + {% endfor %} + +
Question IDDatasetA scoreB scoreΔ
{{ d.question_id }}{{ d.dataset }}{{ "%.4f"|format(d.a_score) }}{{ "%.4f"|format(d.b_score) }} + {{ "%+.4f"|format(d.delta) }} +
+ + + diff --git a/templates/eval/run_report.html.j2 b/templates/eval/run_report.html.j2 new file mode 100644 index 00000000..04910eb3 --- /dev/null +++ b/templates/eval/run_report.html.j2 @@ -0,0 +1,92 @@ + + + + +Eval Run: {{ metadata.run_id }} + + + +

Eval Run {{ metadata.run_id }}

+ +
+
Config
{{ metadata.config_name }}
+
Git SHA
{{ metadata.git_sha }}
+
Started
{{ metadata.started_at }}
+
Finished
{{ metadata.finished_at }}
+
Questions
{{ metadata.n_questions }}
+
Errors
{{ metadata.n_errors }}
+
Datasets
{{ metadata.eval_set_versions|map('string')|join(', ') }}
+
+ +{% if metadata.warnings %} +
+ Warnings: +
    {% for w in metadata.warnings %}
  • {{ w }}
  • {% endfor %}
+
+{% endif %} + +

Aggregated Metrics

+ + + + + + + {% for m in aggregated %} + + + + + + + + {% endfor %} + +
MetricDatasetMean95% CIn
{{ m.metric_name }}{{ m.dataset or "" }}{{ "%.4f"|format(m.mean) }}[{{ "%.4f"|format(m.ci_low) }}, {{ "%.4f"|format(m.ci_high) }}]{{ m.n }}
+ +

Cost Summary

+
+
Total USD
${{ "%.4f"|format(cost.total_usd|default(0.0)) }}
+
Mean USD per query
${{ "%.4f"|format(cost.mean_usd_per_query|default(0.0)) }}
+
Total prompt tokens
{{ cost.total_prompt|default(0) }}
+
Total completion tokens
{{ cost.total_completion|default(0) }}
+
+ +

Per-Question Results ({{ results|length }})

+ + + + + + + {% for r in results[:200] %} + + + + + + + {% endfor %} + +
Question IDDatasetCost USDError
{{ r.question_id }}{{ r.dataset }}${{ "%.4f"|format(r.cost_usd) }}{{ r.error or "" }}
+{% if results|length > 200 %}

(showing first 200 of {{ results|length }} results)

{% endif %} + + + diff --git a/tests/conftest.py b/tests/conftest.py index 5adfec8d..4a106195 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -26,6 +26,86 @@ from src.vector_store import ChromaVectorStore +# --------------------------------------------------------------------------- # +# CI mode: stub LLM provider calls # +# --------------------------------------------------------------------------- # +# +# WHY: A handful of tests (test_query_after_ingest, test_*telemetry) construct +# a real RAGBackend whose query path hits openai's HTTP API. Locally that +# works because the developer has OPENAI_API_KEY set. In CI no real API key +# is available, so we stub openai.OpenAI() to avoid the AuthenticationError +# that would otherwise abort these tests. +# +# Activated by the CI_LLM_MOCK env var so local dev behavior is unchanged. +# --------------------------------------------------------------------------- # + + +@pytest.fixture(scope="session", autouse=True) +def _stub_openai_in_ci(): + """Replace openai.OpenAI() with an in-process stub when CI_LLM_MOCK is truthy. + + The stub mimics the chat.completions.create() shape used by LLMHandler, + returning a canned response with a content attribute and an id. No network + call is made. + """ + if os.getenv("CI_LLM_MOCK", "").lower() not in ("1", "true", "yes"): + yield + return + + try: + import openai + except ImportError: + yield + return + + from types import SimpleNamespace + + STUB_TEXT = "Stubbed answer for CI." + + def _make_stub_response(): + choice = SimpleNamespace( + message=SimpleNamespace(content=STUB_TEXT, role="assistant"), + finish_reason="stop", + index=0, + ) + usage = SimpleNamespace(prompt_tokens=10, completion_tokens=8, total_tokens=18) + return SimpleNamespace( + id="chatcmpl-stub", choices=[choice], usage=usage, + model="stub", created=0, object="chat.completion", + ) + + def _make_stub_stream_chunks(): + # Two-token stream so consumers see at least one yield then a finish. + for piece in (STUB_TEXT, ""): + delta = SimpleNamespace(content=piece, role="assistant") + choice = SimpleNamespace(delta=delta, finish_reason=None, index=0) + yield SimpleNamespace( + id="chatcmpl-stub", choices=[choice], model="stub", + created=0, object="chat.completion.chunk", + ) + + class _StubCompletions: + def create(self, **kwargs): + if kwargs.get("stream"): + return _make_stub_stream_chunks() + return _make_stub_response() + + class _StubChat: + def __init__(self): + self.completions = _StubCompletions() + + class _StubOpenAIClient: + def __init__(self, *args, **kwargs): + self.chat = _StubChat() + + original = openai.OpenAI + openai.OpenAI = _StubOpenAIClient + try: + yield + finally: + openai.OpenAI = original + + # --------------------------------------------------------------------------- # # Constants # # --------------------------------------------------------------------------- # diff --git a/tests/fixtures/phase2_corpus/d1.txt b/tests/fixtures/phase2_corpus/d1.txt new file mode 100644 index 00000000..f6e6a5e8 --- /dev/null +++ b/tests/fixtures/phase2_corpus/d1.txt @@ -0,0 +1 @@ +Reciprocal rank fusion combines two ranked lists by summing 1/(k+rank) for each item. diff --git a/tests/fixtures/phase2_corpus/d2.txt b/tests/fixtures/phase2_corpus/d2.txt new file mode 100644 index 00000000..6a5c52ff --- /dev/null +++ b/tests/fixtures/phase2_corpus/d2.txt @@ -0,0 +1 @@ +Cross-encoders score query-document pairs jointly and improve retrieval precision. diff --git a/tests/fixtures/phase2_corpus/d3.txt b/tests/fixtures/phase2_corpus/d3.txt new file mode 100644 index 00000000..2a59ea81 --- /dev/null +++ b/tests/fixtures/phase2_corpus/d3.txt @@ -0,0 +1 @@ +Airplanes have fixed wings and powered engines. diff --git a/tests/test_api_eval_routes.py b/tests/test_api_eval_routes.py new file mode 100644 index 00000000..8edab01c --- /dev/null +++ b/tests/test_api_eval_routes.py @@ -0,0 +1,177 @@ +"""Tests for src.api.routes.eval.""" + +from __future__ import annotations + +import json +import os +import time +from pathlib import Path + +import pytest +from fastapi.testclient import TestClient + + +PROJECT_ROOT = Path(__file__).resolve().parent.parent + + +@pytest.fixture +def synthetic_squad(monkeypatch, tmp_path): + from src.eval.schemas import EvalQuestion + questions = [ + EvalQuestion( + id=f"q{i}", question=f"What is fact {i}?", + gold_answer=f"Fact {i}.", gold_chunk_ids=[f"q{i}"], + metadata={"context": f"Fact {i} is important.", "title": "t"}, + ) + for i in range(3) + ] + path = tmp_path / "squad.jsonl" + with path.open("w") as f: + for q in questions: + f.write(q.model_dump_json() + "\n") + monkeypatch.setattr("src.eval.datasets.squad_v2.DEFAULT_OUTPUT_PATH", path) + return path + + +@pytest.fixture +def configs_dir(tmp_path, monkeypatch): + """Create configs/eval/test.yaml in a temp dir; monkeypatch CONFIGS_DIR.""" + cfg_dir = tmp_path / "configs" / "eval" + cfg_dir.mkdir(parents=True) + cfg_path = cfg_dir / "test.yaml" + cfg_path.write_text(""" +name: api-test +description: "" +pipeline: + chunker: {strategy: "recursive", chunk_size: 256, chunk_overlap: 32} + retriever: {top_k: 3} + generator: {model: "gpt-4.1-nano", reasoning_model: null} +eval: + datasets: ["squad_v2_dev_200"] + judge_model: "gpt-4.1-nano" + bootstrap_n: 50 + permutation_n: 50 + seed: 1 +""") + monkeypatch.setattr("src.api.routes.eval.CONFIGS_DIR", cfg_dir) + return cfg_dir + + +@pytest.fixture +def tmp_eval_runs(tmp_path, monkeypatch): + runs = tmp_path / "eval_runs" + runs.mkdir() + monkeypatch.setenv("EVAL_RUNS_DIR", str(runs)) + import importlib + import src.eval.storage + importlib.reload(src.eval.storage) + yield runs + monkeypatch.delenv("EVAL_RUNS_DIR", raising=False) + importlib.reload(src.eval.storage) + + +@pytest.fixture +def client_with_dummy_llm(monkeypatch): + """TestClient where the eval route uses a dummy LLM via env override.""" + monkeypatch.setenv("EVAL_LLM_OVERRIDE_DUMMY", "1") + from src.api.main import app + yield TestClient(app) + + +class TestConfigsEndpoint: + def test_lists_configs(self, configs_dir, client_with_dummy_llm): + r = client_with_dummy_llm.get("/api/eval/configs") + assert r.status_code == 200 + assert "test" in r.json() + + +class TestRunSubmitAndStatus: + def test_submit_and_complete(self, configs_dir, tmp_eval_runs, + synthetic_squad, client_with_dummy_llm): + r = client_with_dummy_llm.post( + "/api/eval/run", json={"config_name": "test"} + ) + assert r.status_code == 202, r.text + body = r.json() + run_id = body["run_id"] + assert body["status"] == "queued" + + # Poll status until completed (or fail after 30s) + for _ in range(60): + sr = client_with_dummy_llm.get(f"/api/eval/runs/{run_id}/status") + assert sr.status_code == 200 + if sr.json()["status"] in ("completed", "failed"): + break + time.sleep(0.5) + assert sr.json()["status"] == "completed", sr.json() + + def test_unknown_config_returns_404(self, configs_dir, client_with_dummy_llm): + r = client_with_dummy_llm.post( + "/api/eval/run", json={"config_name": "nope"} + ) + assert r.status_code == 404 + + +class TestRunsList: + def test_lists_completed_runs(self, configs_dir, tmp_eval_runs, + synthetic_squad, client_with_dummy_llm): + # Submit + wait + r = client_with_dummy_llm.post( + "/api/eval/run", json={"config_name": "test"} + ) + run_id = r.json()["run_id"] + for _ in range(60): + sr = client_with_dummy_llm.get(f"/api/eval/runs/{run_id}/status") + if sr.json()["status"] == "completed": + break + time.sleep(0.5) + + # List + list_r = client_with_dummy_llm.get("/api/eval/runs") + assert list_r.status_code == 200 + runs = list_r.json() + assert any(r["run_id"] == run_id for r in runs) + + +class TestRunDetailAndResults: + def test_get_run_detail(self, configs_dir, tmp_eval_runs, + synthetic_squad, client_with_dummy_llm): + r = client_with_dummy_llm.post( + "/api/eval/run", json={"config_name": "test"} + ) + run_id = r.json()["run_id"] + for _ in range(60): + sr = client_with_dummy_llm.get(f"/api/eval/runs/{run_id}/status") + if sr.json()["status"] == "completed": + break + time.sleep(0.5) + + dr = client_with_dummy_llm.get(f"/api/eval/runs/{run_id}") + assert dr.status_code == 200 + d = dr.json() + assert d["metadata"]["run_id"] == run_id + assert d["n_results"] == 3 + + def test_get_run_results_paginated(self, configs_dir, tmp_eval_runs, + synthetic_squad, client_with_dummy_llm): + r = client_with_dummy_llm.post( + "/api/eval/run", json={"config_name": "test"} + ) + run_id = r.json()["run_id"] + for _ in range(60): + sr = client_with_dummy_llm.get(f"/api/eval/runs/{run_id}/status") + if sr.json()["status"] == "completed": + break + time.sleep(0.5) + + rr = client_with_dummy_llm.get( + f"/api/eval/runs/{run_id}/results?page=1&page_size=2" + ) + assert rr.status_code == 200 + body = rr.json() + assert len(body["items"]) == 2 + assert body["total"] == 3 + + def test_missing_run_returns_404(self, configs_dir, client_with_dummy_llm): + r = client_with_dummy_llm.get("/api/eval/runs/nonexistent") + assert r.status_code == 404 diff --git a/tests/test_api_eval_runs_service.py b/tests/test_api_eval_runs_service.py new file mode 100644 index 00000000..766276ab --- /dev/null +++ b/tests/test_api_eval_runs_service.py @@ -0,0 +1,110 @@ +"""Tests for src.api.services.eval_runs.""" + +from __future__ import annotations + +import time +from concurrent.futures import ThreadPoolExecutor +from datetime import datetime, timedelta, timezone + +import pytest + +from src.api.services.eval_runs import RunRegistry, RunStatus + + +class TestBasicLifecycle: + def test_register_then_get(self): + reg = RunRegistry() + reg.register("r1", n_total=10) + s = reg.get("r1") + assert s is not None + assert s.run_id == "r1" + assert s.status == "queued" + assert s.n_completed == 0 + assert s.n_total == 10 + + def test_update_progress_transitions_to_running(self): + reg = RunRegistry() + reg.register("r1", n_total=10) + reg.update_progress("r1", 3) + s = reg.get("r1") + assert s.status == "running" + assert s.n_completed == 3 + + def test_mark_completed(self): + reg = RunRegistry() + reg.register("r1", n_total=10) + reg.mark_completed("r1") + s = reg.get("r1") + assert s.status == "completed" + assert s.n_completed == 10 + assert s.completed_at is not None + + def test_mark_failed(self): + reg = RunRegistry() + reg.register("r1", n_total=10) + reg.mark_failed("r1", "boom") + s = reg.get("r1") + assert s.status == "failed" + assert s.error_message == "boom" + assert s.completed_at is not None + + def test_get_unknown_returns_none(self): + assert RunRegistry().get("nope") is None + + +class TestListActive: + def test_only_returns_queued_or_running(self): + reg = RunRegistry() + reg.register("queued", 10) + reg.register("running", 10) + reg.update_progress("running", 1) + reg.register("done", 10) + reg.mark_completed("done") + reg.register("err", 10) + reg.mark_failed("err", "x") + + active_ids = {s.run_id for s in reg.list_active()} + assert active_ids == {"queued", "running"} + + +class TestConcurrentUpdates: + def test_concurrent_progress_updates_remain_consistent(self): + reg = RunRegistry() + reg.register("r1", n_total=100) + + def bump(i: int): + reg.update_progress("r1", i) + + with ThreadPoolExecutor(max_workers=10) as ex: + list(ex.map(bump, range(100))) + + s = reg.get("r1") + # The final n_completed should be one of the values written; + # the important invariant is no exceptions and not None. + assert s is not None + assert 0 <= s.n_completed <= 100 + + +class TestEviction: + def test_evicts_completed_after_ttl(self): + reg = RunRegistry() + reg.register("old", 10) + reg.mark_completed("old") + # Forge an older completed_at to simulate elapsed time. + s = reg.get("old") + s.completed_at = datetime.now(timezone.utc) - timedelta(seconds=7200) + + reg.register("new", 10) + reg.mark_completed("new") + + evicted = reg.evict_old(ttl_seconds=3600.0) + assert evicted == 1 + assert reg.get("old") is None + assert reg.get("new") is not None + + def test_does_not_evict_active(self): + reg = RunRegistry() + reg.register("active", 10) + evicted = reg.evict_old(ttl_seconds=0.0) + assert evicted == 0 + assert reg.get("active") is not None diff --git a/tests/test_api_query_telemetry.py b/tests/test_api_query_telemetry.py new file mode 100644 index 00000000..fe5aef06 --- /dev/null +++ b/tests/test_api_query_telemetry.py @@ -0,0 +1,264 @@ +"""Tests for telemetry surfacing in the REST and WebSocket query routes. + +API Layer Position: + RAGBackend.query_with_telemetry → (result_dict, StageTelemetry) + POST /api/query → QueryResponse with `telemetry` field [REST] + GET /api/chat → WebSocket stream ends with telemetry event [WS] + +What concept it teaches: + Route-layer serialization testing: mock the backend dependency, assert the + route converts the returned StageTelemetry into the right JSON shape. + This isolates the serialization concern from real LLM / ChromaDB I/O. + +Why mock approach instead of full-ingest: + - RAGBackend telemetry behaviour is already tested in test_backend_telemetry.py. + - Task 5 only changes the route layer (query.py + models.py). + - Mocking the backend keeps tests fast (<1 s) and dependency-free. +""" + +from __future__ import annotations + +import json +from typing import Iterator +from unittest.mock import MagicMock, patch + +import pytest +from fastapi.testclient import TestClient + + +# --------------------------------------------------------------------------- +# Shared helpers +# --------------------------------------------------------------------------- + +_FAKE_TELEMETRY = { + "retrieve_ms": 42.5, + "generate_ms": 310.0, + "prompt_tokens": 512, + "completion_tokens": 128, + "cost_usd": 0.0034, +} + +_FAKE_RESULT = { + "answer": "RAG combines retrieval with generation.", + "sources": [ + { + "doc_id": "doc-1", + "chunk_id": "chunk-1", + "filename": "rag.txt", + "score": 0.92, + "excerpt": "RAG stands for Retrieval-Augmented Generation.", + } + ], + "confidence": 0.85, +} + + +def _make_telemetry_model(): + """Return a StageTelemetry instance with the fake values.""" + from src.api.schemas.telemetry import StageTelemetry + return StageTelemetry(**_FAKE_TELEMETRY) + + +# --------------------------------------------------------------------------- +# REST endpoint — POST /api/query +# --------------------------------------------------------------------------- + + +class TestRestTelemetry: + """POST /api/query response must include a well-formed `telemetry` field.""" + + @pytest.fixture + def client(self): + """TestClient with a mocked backend attached to app.state. + + WHY: TestClient runs the FastAPI lifespan, which creates a real + RAGBackend on app.state. We replace it with a MagicMock after + startup so query_with_telemetry() returns predictable values + without touching ChromaDB or an LLM. + """ + from src.api.main import app + with TestClient(app) as c: + mock_backend = MagicMock() + mock_backend.query_with_telemetry.return_value = ( + _FAKE_RESULT, _make_telemetry_model() + ) + # evaluate_faithfulness_realtime must not raise during WS teardown + mock_backend.evaluate_faithfulness_realtime.return_value = {} + app.state.backend = mock_backend + yield c + + def test_response_includes_telemetry_field(self, client): + """The telemetry object appears at the top level of the JSON response.""" + r = client.post("/api/query", json={"query": "What is RAG?"}) + assert r.status_code == 200, r.text + body = r.json() + assert "telemetry" in body, f"'telemetry' key missing from response: {body}" + + def test_telemetry_has_all_five_fields(self, client): + """All five StageTelemetry fields are present in the response body.""" + r = client.post("/api/query", json={"query": "What is RAG?"}) + assert r.status_code == 200 + t = r.json()["telemetry"] + for field in ("retrieve_ms", "generate_ms", "prompt_tokens", + "completion_tokens", "cost_usd"): + assert field in t, f"Missing telemetry field: {field}" + assert t[field] >= 0, f"Telemetry field {field} must be >= 0" + + def test_telemetry_values_match_backend(self, client): + """Telemetry values round-trip correctly from backend to JSON response.""" + r = client.post("/api/query", json={"query": "What is RAG?"}) + t = r.json()["telemetry"] + assert t["retrieve_ms"] == _FAKE_TELEMETRY["retrieve_ms"] + assert t["generate_ms"] == _FAKE_TELEMETRY["generate_ms"] + assert t["prompt_tokens"] == _FAKE_TELEMETRY["prompt_tokens"] + assert t["completion_tokens"] == _FAKE_TELEMETRY["completion_tokens"] + assert t["cost_usd"] == pytest.approx(_FAKE_TELEMETRY["cost_usd"]) + + def test_existing_response_fields_unchanged(self, client): + """Adding telemetry must not drop or alter existing response fields.""" + r = client.post("/api/query", json={"query": "What is RAG?"}) + body = r.json() + assert "answer" in body + assert "sources" in body + assert "confidence" in body + assert "latency_ms" in body + + def test_query_with_telemetry_is_called_not_query(self, client): + """The route must call query_with_telemetry, not the plain query(). + + WHY: If someone reverts to backend.query() the telemetry field would + silently become None (the Optional default). This assertion catches it. + """ + from src.api.main import app + client.post("/api/query", json={"query": "test"}) + app.state.backend.query_with_telemetry.assert_called_once() + app.state.backend.query.assert_not_called() + + +# --------------------------------------------------------------------------- +# WebSocket endpoint — /api/chat +# --------------------------------------------------------------------------- + + +class TestWebSocketTelemetry: + """WebSocket /api/chat must forward the telemetry event from stream_query.""" + + def _make_stream(self) -> list[tuple[str, object]]: + """Minimal event stream: status → done → telemetry.""" + return [ + ("status", "Searching indexed documents..."), + ("token", "RAG combines retrieval with generation."), + ("done", { + "sources": [ + { + "doc_id": "doc-1", + "chunk_id": "chunk-1", + "filename": "rag.txt", + "score": 0.92, + "excerpt": "RAG stands for Retrieval-Augmented Generation.", + } + ], + "message_id": "msg-abc", + "conversation_id": "conv-xyz", + }), + ("telemetry", _FAKE_TELEMETRY), + ] + + @pytest.fixture + def ws_client(self): + """TestClient with a mocked streaming backend.""" + from src.api.main import app + with TestClient(app) as c: + mock_backend = MagicMock() + + def _fake_stream(*args, **kwargs) -> Iterator: + yield from self._make_stream() + + mock_backend.stream_query.side_effect = _fake_stream + mock_backend.evaluate_faithfulness_realtime.return_value = {} + app.state.backend = mock_backend + yield c + + def test_stream_emits_telemetry_event(self, ws_client): + """The WebSocket stream must include a telemetry event.""" + from src.api.main import app + with ws_client.websocket_connect("/api/chat") as ws: + ws.send_json({"query": "What is RAG?", "top_k": 3}) + events = [] + # Drain all events until we see telemetry or hit 20 messages. + # WHY cap: if telemetry is never emitted, we stop instead of hanging. + for _ in range(20): + try: + msg = ws.receive_json() + events.append(msg) + if msg.get("type") == "telemetry": + break + except Exception: + break + + assert any(e.get("type") == "telemetry" for e in events), ( + f"No telemetry event received. Got event types: {[e.get('type') for e in events]}" + ) + + def test_telemetry_event_has_content_key(self, ws_client): + """The telemetry event must be shaped: {type: 'telemetry', content: {...}}.""" + with ws_client.websocket_connect("/api/chat") as ws: + ws.send_json({"query": "What is RAG?", "top_k": 3}) + tele_event = None + for _ in range(20): + try: + msg = ws.receive_json() + if msg.get("type") == "telemetry": + tele_event = msg + break + except Exception: + break + + assert tele_event is not None, "No telemetry event received." + assert "content" in tele_event, f"telemetry event missing 'content' key: {tele_event}" + + def test_telemetry_content_has_all_five_fields(self, ws_client): + """All five StageTelemetry fields must appear inside the content dict.""" + with ws_client.websocket_connect("/api/chat") as ws: + ws.send_json({"query": "What is RAG?", "top_k": 3}) + tele_event = None + for _ in range(20): + try: + msg = ws.receive_json() + if msg.get("type") == "telemetry": + tele_event = msg + break + except Exception: + break + + content = tele_event["content"] + for field in ("retrieve_ms", "generate_ms", "prompt_tokens", + "completion_tokens", "cost_usd"): + assert field in content, f"Missing telemetry field: {field}" + assert content[field] >= 0, f"Telemetry field {field} must be >= 0" + + def test_done_event_shape_unchanged(self, ws_client): + """Adding telemetry must not modify the done event shape. + + STRICT: done must still carry sources, message_id, conversation_id. + This guards against accidentally merging telemetry into done. + """ + with ws_client.websocket_connect("/api/chat") as ws: + ws.send_json({"query": "What is RAG?", "top_k": 3}) + done_event = None + for _ in range(20): + try: + msg = ws.receive_json() + if msg.get("type") == "done": + done_event = msg + if msg.get("type") == "telemetry": + break + except Exception: + break + + assert done_event is not None, "No done event received." + assert "sources" in done_event, f"done event missing 'sources': {done_event}" + assert "message_id" in done_event + assert "conversation_id" in done_event + # telemetry must NOT be embedded inside done + assert "telemetry" not in done_event diff --git a/tests/test_api_schemas_eval.py b/tests/test_api_schemas_eval.py new file mode 100644 index 00000000..399fb6d6 --- /dev/null +++ b/tests/test_api_schemas_eval.py @@ -0,0 +1,120 @@ +"""Tests for src.api.schemas.eval DTOs.""" + +from __future__ import annotations + +from datetime import datetime, timezone + +import pytest +from pydantic import ValidationError + +from src.api.schemas.eval import ( + AggregatedMetricDTO, + EvalResultDTO, + RunDetailDTO, + RunStatusDTO, + RunSubmitRequest, + RunSubmitResponse, + RunSummaryDTO, +) +from src.eval.schemas import RunMetadata + + +def _meta() -> RunMetadata: + now = datetime.now(timezone.utc) + return RunMetadata( + run_id="r1", config_name="baseline", config_path="x.yaml", + git_sha="abc1234", started_at=now, finished_at=now, + env_hash="h", eval_set_versions={"squad_v2_dev_200": "v1"}, + n_questions=10, n_errors=0, + ) + + +class TestRunSummaryDTO: + def test_construction(self): + now = datetime.now(timezone.utc) + d = RunSummaryDTO( + run_id="r1", config_name="baseline", + started_at=now, finished_at=now, + n_questions=10, n_errors=0, headline_metric=0.84, + ) + assert d.headline_metric == 0.84 + + def test_headline_metric_optional(self): + now = datetime.now(timezone.utc) + d = RunSummaryDTO( + run_id="r1", config_name="baseline", + started_at=now, finished_at=now, + n_questions=10, n_errors=0, headline_metric=None, + ) + assert d.headline_metric is None + + +class TestAggregatedMetricDTO: + def test_construction(self): + d = AggregatedMetricDTO( + metric_name="recall_at_5", dataset="squad_v2_dev_200", + mean=0.84, ci_low=0.81, ci_high=0.87, n=200, + ) + assert d.dataset == "squad_v2_dev_200" + + def test_dataset_can_be_none(self): + d = AggregatedMetricDTO( + metric_name="recall_at_5", dataset=None, + mean=0.84, ci_low=0.81, ci_high=0.87, n=200, + ) + assert d.dataset is None + + +class TestRunDetailDTO: + def test_construction(self): + d = RunDetailDTO( + metadata=_meta(), + aggregated=[AggregatedMetricDTO( + metric_name="x", dataset=None, mean=0.5, + ci_low=0.4, ci_high=0.6, n=10, + )], + cost={"total_usd": 0.01, "mean_usd_per_query": 0.001}, + n_results=10, + ) + assert d.metadata.run_id == "r1" + assert d.n_results == 10 + + +class TestEvalResultDTO: + def test_construction(self): + d = EvalResultDTO( + question_id="q1", dataset="squad_v2_dev_200", + generated_answer="ans", metrics={"recall_at_5": 1.0}, + error=None, + ) + assert d.error is None + + +class TestRunSubmitRequest: + def test_construction(self): + r = RunSubmitRequest(config_name="baseline") + assert r.config_name == "baseline" + + def test_missing_config_name_raises(self): + with pytest.raises(ValidationError): + RunSubmitRequest() + + +class TestRunSubmitResponse: + def test_valid_status(self): + r = RunSubmitResponse(run_id="r1", status="queued") + assert r.status == "queued" + + def test_invalid_status_raises(self): + with pytest.raises(ValidationError): + RunSubmitResponse(run_id="r1", status="invalid") + + +class TestRunStatusDTO: + def test_construction(self): + s = RunStatusDTO( + run_id="r1", status="running", + progress=0.5, n_completed=5, n_total=10, + error_message=None, + ) + assert s.progress == 0.5 diff --git a/tests/test_backend_telemetry.py b/tests/test_backend_telemetry.py new file mode 100644 index 00000000..31a90102 --- /dev/null +++ b/tests/test_backend_telemetry.py @@ -0,0 +1,233 @@ +"""Tests that RAGBackend instrumentation produces a StageTelemetry payload. + +RAG Pipeline Position: + Question -> Retrieve -> Generate -> (answer, StageTelemetry) + ^^^ + These tests verify that query_with_telemetry() and stream_query() both emit + well-formed StageTelemetry with non-negative numeric fields. + +What concept it teaches: + Testing the observability layer in isolation from real LLM providers. + LLMHandler falls back to a dummy response when no API keys are configured, + so these tests run without any network access and with zero cost. + +Why this fixture pattern: + Mirrors test_backend.py exactly — EphemeralClient for ChromaDB (no disk I/O, + isolated per test) and sqlite:// for SQLite (in-memory, vanishes on teardown). + Both stores are fully isolated; no shared mutable state between tests. +""" + +from __future__ import annotations + +import uuid +from pathlib import Path + +import chromadb +import pytest + +from src.backend import RAGBackend +from src.api.schemas.telemetry import StageTelemetry +from src.database import create_db_and_tables, get_engine + + +# --------------------------------------------------------------------------- # +# Fixtures (mirrored from test_backend.py) # +# --------------------------------------------------------------------------- # + +@pytest.fixture +def tmp_sqlite_engine(): + """In-memory SQLite engine with all tables created.""" + engine = get_engine("sqlite://") + create_db_and_tables(engine) + return engine + + +@pytest.fixture +def chroma_backend_collection(): + """Ephemeral ChromaDB collection with auto-embedding enabled. + + WHY unique name: EphemeralClient shares an in-process store. + A UUID suffix ensures complete isolation between test runs. + """ + client = chromadb.EphemeralClient() + return client.get_or_create_collection( + name=f"test_backend_telemetry_{uuid.uuid4().hex}", + metadata={"hnsw:space": "cosine"}, + ) + + +@pytest.fixture +def backend(tmp_sqlite_engine, chroma_backend_collection): + """Fully-wired RAGBackend with ephemeral ChromaDB + in-memory SQLite.""" + return RAGBackend(engine=tmp_sqlite_engine, collection=chroma_backend_collection) + + +@pytest.fixture +def ingested_backend(backend: RAGBackend, tmp_path: Path) -> RAGBackend: + """A backend that has already ingested a small text document. + + WHY pre-ingest: query_with_telemetry requires at least one indexed chunk + to take the non-empty retrieval path and call the LLM. Without a document, + both retrieve and generate phases return zeros — not useful to test. + """ + content = ( + "Retrieval-Augmented Generation (RAG) combines document retrieval " + "with large language model generation to produce grounded answers. " + "The retrieval step finds relevant chunks via vector similarity search. " + "The generation step builds a prompt from those chunks and queries the LLM." + ) + doc = tmp_path / "rag_intro.txt" + doc.write_text(content, encoding="utf-8") + backend.ingest_file(doc) + return backend + + +# --------------------------------------------------------------------------- # +# Tests # +# --------------------------------------------------------------------------- # + +class TestQueryWithTelemetry: + """Tests for RAGBackend.query_with_telemetry().""" + + def test_telemetry_fields_are_non_negative_after_ingest( + self, ingested_backend: RAGBackend + ): + """query_with_telemetry returns a StageTelemetry with all non-negative fields. + + PATTERN: With no real LLM configured, LLMHandler falls back to a dummy + response. The dummy response still produces valid answer text, which + means token counting and cost computation run on real strings — the + telemetry shape is exercised even without a real LLM call. + """ + result, telemetry = ingested_backend.query_with_telemetry("What is RAG?") + + # Result dict has the same shape as query() + assert "answer" in result + assert "sources" in result + assert len(result["sources"]) > 0 + + # Telemetry is the right type + assert isinstance(telemetry, StageTelemetry) + + # All five numeric fields are non-negative + assert telemetry.retrieve_ms >= 0.0 + assert telemetry.generate_ms >= 0.0 + assert telemetry.prompt_tokens >= 0 + assert telemetry.completion_tokens >= 0 + assert telemetry.cost_usd >= 0.0 + + def test_telemetry_zeros_when_no_documents(self, backend: RAGBackend): + """When no documents are indexed, StageTelemetry has zero generate/token fields. + + WHY: The early-return branch skips the LLM call entirely. + retrieve_ms records real elapsed time (even for an empty search); + generate_ms, prompt_tokens, completion_tokens, and cost_usd are all 0. + """ + result, telemetry = backend.query_with_telemetry("What is RAG?") + + assert result["answer"].startswith("No documents indexed") + assert telemetry.generate_ms == 0.0 + assert telemetry.prompt_tokens == 0 + assert telemetry.completion_tokens == 0 + assert telemetry.cost_usd == 0.0 + # retrieve_ms is still measured (we did make the call, it just returned empty) + assert telemetry.retrieve_ms >= 0.0 + + def test_query_unchanged_after_sibling_added(self, ingested_backend: RAGBackend): + """query() still returns the plain dict — the new sibling has no side effects. + + WHY: This is the regression guard for the Option A design choice. + Existing tests use result["answer"] / result["sources"] on the dict + returned by query(). If we accidentally broke the dict shape, this + test catches it immediately. + """ + result = ingested_backend.query("What is RAG?") + + assert isinstance(result, dict) + assert "answer" in result + assert "sources" in result + assert "confidence" in result + + +class TestStreamQueryTelemetry: + """Tests for the telemetry event emitted by stream_query().""" + + def test_stream_query_emits_telemetry_event_last( + self, ingested_backend: RAGBackend + ): + """stream_query yields a ("telemetry", dict) as the final event after ("done", ...). + + WHY last: The done event is what the client waits for to display sources. + Telemetry is a secondary signal. Emitting it last ensures done latency + is not delayed by token-counting arithmetic. + """ + events = list(ingested_backend.stream_query("What is RAG?")) + + # The last event must be the telemetry event + last_type, last_data = events[-1] + assert last_type == "telemetry", ( + f"Expected last event type 'telemetry', got {last_type!r}. " + f"All event types: {[e[0] for e in events]}" + ) + + # The telemetry payload must be a dict (model_dump() output) + assert isinstance(last_data, dict) + + # All five fields must be present and non-negative + for field in ("retrieve_ms", "generate_ms", "prompt_tokens", "completion_tokens", "cost_usd"): + assert field in last_data, f"Missing telemetry field: {field}" + assert last_data[field] >= 0, f"Telemetry field {field} is negative: {last_data[field]}" + + def test_stream_query_done_event_unchanged(self, ingested_backend: RAGBackend): + """The ("done", ...) event shape is not modified by the telemetry addition. + + STRICT: The done event must still carry 'sources' (and nothing else + unexpected). This test guards against accidental mutation of the done + event dict. + """ + events = list(ingested_backend.stream_query("What is RAG?")) + + done_events = [(t, d) for t, d in events if t == "done"] + assert len(done_events) == 1, f"Expected exactly one done event, got {len(done_events)}" + + _, done_data = done_events[0] + assert "sources" in done_data + + def test_stream_query_emits_telemetry_on_empty_store(self, backend: RAGBackend): + """The early-return (no documents) path still emits a telemetry event. + + WHY: The route layer always expects a telemetry event. If the early-return + path omitted it, the frontend would never receive telemetry data for + failed queries — a silent gap that's hard to debug. + """ + events = list(backend.stream_query("What is RAG?")) + + event_types = [e[0] for e in events] + assert "telemetry" in event_types + + last_type, last_data = events[-1] + assert last_type == "telemetry" + assert last_data["generate_ms"] == 0.0 + assert last_data["prompt_tokens"] == 0 + + def test_stream_query_existing_events_order_preserved( + self, ingested_backend: RAGBackend + ): + """Existing event types appear in the expected order before telemetry. + + The protocol guarantees: status* → reasoning* → status → token* → done → telemetry + (where * = zero or more). This test checks that all mandatory events + are still present and that telemetry is appended AFTER done. + """ + events = list(ingested_backend.stream_query("What is RAG?")) + event_types = [e[0] for e in events] + + assert "status" in event_types + assert "done" in event_types + assert "telemetry" in event_types + + done_idx = next(i for i, (t, _) in enumerate(events) if t == "done") + telemetry_idx = next(i for i, (t, _) in enumerate(events) if t == "telemetry") + assert telemetry_idx > done_idx, ( + "telemetry event must come after done event" + ) diff --git a/tests/test_eval_aggregator.py b/tests/test_eval_aggregator.py new file mode 100644 index 00000000..8c78c46d --- /dev/null +++ b/tests/test_eval_aggregator.py @@ -0,0 +1,97 @@ +"""Tests for src.eval.aggregator.""" + +from __future__ import annotations + +import pytest + +from src.eval.aggregator import aggregate +from src.eval.config import EvalConfig +from src.eval.schemas import AggregatedMetric, EvalResult + + +def _baseline_config() -> EvalConfig: + return EvalConfig.model_validate({ + "name": "test", "description": "", + "pipeline": { + "chunker": {"strategy": "recursive", "chunk_size": 256, "chunk_overlap": 32}, + "retriever": {"top_k": 3}, + "generator": {"model": "gpt-4.1-nano", "reasoning_model": None}, + }, + "eval": { + "datasets": ["squad_v2_dev_200", "ml_papers_v1"], + "judge_model": "gpt-4.1-nano", + "bootstrap_n": 200, "permutation_n": 100, "seed": 42, + }, + }) + + +def _r(qid: str, dataset: str, metrics: dict[str, float], error: str | None = None) -> EvalResult: + return EvalResult( + question_id=qid, dataset=dataset, + retrieved_chunk_ids=[], retrieved_chunks=[], + generated_answer="", metrics=metrics, metric_details={}, + timings_ms={}, tokens={"prompt": 0, "completion": 0}, cost_usd=0.0, + error=error, + ) + + +class TestAggregate: + def test_per_dataset_and_combined(self): + cfg = _baseline_config() + results = [] + # 5 squad results with recall_at_5 = 1.0 + results += [_r(f"s{i}", "squad_v2_dev_200", {"recall_at_5": 1.0}) for i in range(5)] + # 5 ml_papers results with recall_at_5 = 0.0 + results += [_r(f"m{i}", "ml_papers_v1", {"recall_at_5": 0.0}) for i in range(5)] + + aggregated, warnings = aggregate(results, cfg) + assert warnings == [] + + by_key = {(a.metric_name, a.dataset): a for a in aggregated} + # Three rows: squad-only, ml_papers-only, combined + assert by_key[("recall_at_5", "squad_v2_dev_200")].mean == pytest.approx(1.0) + assert by_key[("recall_at_5", "ml_papers_v1")].mean == pytest.approx(0.0) + assert by_key[("recall_at_5", None)].mean == pytest.approx(0.5) + + def test_low_n_skipped_and_warned(self): + cfg = _baseline_config() + # Only 2 squad results — below the 3-sample minimum. + results = [ + _r("s1", "squad_v2_dev_200", {"recall_at_5": 1.0}), + _r("s2", "squad_v2_dev_200", {"recall_at_5": 0.5}), + ] + aggregated, warnings = aggregate(results, cfg) + # Per-dataset row skipped; combined also <3 samples → also skipped. + assert all(a.metric_name != "recall_at_5" or a.dataset is None for a in aggregated) or aggregated == [] + # At least one warning mentions the skipped metric. + assert any("recall_at_5" in w for w in warnings) + + def test_excludes_errored_results(self): + cfg = _baseline_config() + results = [_r(f"s{i}", "squad_v2_dev_200", {"recall_at_5": 1.0}) for i in range(5)] + results.append(_r("err", "squad_v2_dev_200", {"recall_at_5": 0.0}, error="boom")) + aggregated, _ = aggregate(results, cfg) + # Errored row excluded → mean still 1.0, n=5. + squad = next(a for a in aggregated if a.metric_name == "recall_at_5" and a.dataset == "squad_v2_dev_200") + assert squad.mean == pytest.approx(1.0) + assert squad.n == 5 + + def test_multiple_metrics(self): + cfg = _baseline_config() + results = [ + _r(f"s{i}", "squad_v2_dev_200", + {"recall_at_5": 1.0, "faithfulness": 0.9}) + for i in range(5) + ] + aggregated, _ = aggregate(results, cfg) + names = {a.metric_name for a in aggregated} + assert names == {"recall_at_5", "faithfulness"} + + def test_seed_propagated(self): + """Two runs with the same config should yield identical CIs.""" + cfg = _baseline_config() + results = [_r(f"s{i}", "squad_v2_dev_200", {"r": float(i % 2)}) for i in range(20)] + a1, _ = aggregate(results, cfg) + a2, _ = aggregate(results, cfg) + assert {(a.metric_name, a.dataset, a.mean, a.ci_low, a.ci_high) for a in a1} == \ + {(a.metric_name, a.dataset, a.mean, a.ci_low, a.ci_high) for a in a2} diff --git a/tests/test_eval_cli.py b/tests/test_eval_cli.py new file mode 100644 index 00000000..cc7da316 --- /dev/null +++ b/tests/test_eval_cli.py @@ -0,0 +1,175 @@ +"""End-to-end CLI tests for src.eval.cli.""" + +from __future__ import annotations + +import os +import subprocess +import sys +from pathlib import Path + +import pytest + + +PROJECT_ROOT = Path(__file__).resolve().parent.parent + + +def _run_cli(args: list[str], env_overrides: dict[str, str] | None = None) -> subprocess.CompletedProcess: + env = os.environ.copy() + if env_overrides: + env.update(env_overrides) + return subprocess.run( + [sys.executable, "-m", "src.eval.cli", *args], + cwd=PROJECT_ROOT, + env=env, + capture_output=True, + text=True, + check=False, + ) + + +@pytest.fixture +def tmp_eval_runs(tmp_path: Path) -> Path: + runs = tmp_path / "eval_runs" + runs.mkdir() + return runs + + +@pytest.fixture +def synthetic_squad(tmp_path: Path, monkeypatch) -> Path: + """Write a tiny 3-question synthetic squad set and override the loader's path.""" + from src.eval.schemas import EvalQuestion + questions = [ + EvalQuestion( + id=f"q{i}", question=f"What is fact {i}?", + gold_answer=f"Fact {i}.", gold_chunk_ids=[f"q{i}"], + metadata={"context": f"Fact {i} is important.", "title": "t"}, + ) + for i in range(3) + ] + path = tmp_path / "squad.jsonl" + with path.open("w") as f: + for q in questions: + f.write(q.model_dump_json() + "\n") + return path + + +@pytest.fixture +def cli_config(tmp_path: Path, synthetic_squad: Path) -> Path: + """Write a baseline-shaped YAML config; runner picks it up.""" + import yaml + config_data = { + "name": "cli-test", "description": "", + "pipeline": { + "chunker": {"strategy": "recursive", "chunk_size": 256, "chunk_overlap": 32}, + "retriever": {"top_k": 3}, + "generator": {"model": "gpt-4.1-nano", "reasoning_model": None}, + }, + "eval": { + "datasets": ["squad_v2_dev_200"], + "judge_model": "gpt-4.1-nano", + "bootstrap_n": 100, "permutation_n": 100, "seed": 42, + }, + } + config_path = tmp_path / "test_config.yaml" + config_path.write_text(yaml.safe_dump(config_data)) + return config_path + + +def _patch_squad_path(env: dict, squad_path: Path) -> dict: + """Inject a python -c snippet to set DEFAULT_OUTPUT_PATH before runner runs. + + Easier: set EVAL_SQUAD_PATH and have the CLI honor it. We extend the CLI + so EVAL_SQUAD_PATH overrides the default frozen path. + """ + env["EVAL_SQUAD_PATH"] = str(squad_path) + return env + + +class TestCliRun: + def test_run_completes_and_prints_run_id(self, tmp_eval_runs, cli_config, synthetic_squad): + env = { + "EVAL_RUNS_DIR": str(tmp_eval_runs), + "EVAL_LLM_OVERRIDE_DUMMY": "1", + "EVAL_SQUAD_PATH": str(synthetic_squad), + } + result = _run_cli(["run", "--config", str(cli_config)], env_overrides=env) + assert result.returncode == 0, f"stderr: {result.stderr}" + assert "cli-test" in result.stdout + # A run dir was created + run_dirs = list(tmp_eval_runs.iterdir()) + assert len(run_dirs) == 1 + + +class TestCliList: + def test_list_shows_runs(self, tmp_eval_runs, cli_config, synthetic_squad): + # First create a run + env = { + "EVAL_RUNS_DIR": str(tmp_eval_runs), + "EVAL_LLM_OVERRIDE_DUMMY": "1", + "EVAL_SQUAD_PATH": str(synthetic_squad), + } + _run_cli(["run", "--config", str(cli_config)], env_overrides=env) + + result = _run_cli(["list"], env_overrides=env) + assert result.returncode == 0 + assert "cli-test" in result.stdout + + def test_list_empty(self, tmp_eval_runs): + env = {"EVAL_RUNS_DIR": str(tmp_eval_runs)} + result = _run_cli(["list"], env_overrides=env) + assert result.returncode == 0 + # Should print something (header or "No runs"); just confirm no crash + assert result.stdout != "" or result.stderr == "" + + +class TestCliShow: + def test_show_prints_metrics(self, tmp_eval_runs, cli_config, synthetic_squad): + env = { + "EVAL_RUNS_DIR": str(tmp_eval_runs), + "EVAL_LLM_OVERRIDE_DUMMY": "1", + "EVAL_SQUAD_PATH": str(synthetic_squad), + } + run_result = _run_cli(["run", "--config", str(cli_config)], env_overrides=env) + # Extract run_id from stdout (it's printed somewhere) + run_id = next(line for line in run_result.stdout.split("\n") + if "cli-test" in line and "_" in line).strip().split()[-1] + + result = _run_cli(["show", run_id], env_overrides=env) + assert result.returncode == 0 + # Show should print at least the run_id and at least one metric name + assert run_id in result.stdout or "metric" in result.stdout.lower() + + def test_show_with_html_writes_file(self, tmp_eval_runs, cli_config, synthetic_squad): + env = { + "EVAL_RUNS_DIR": str(tmp_eval_runs), + "EVAL_LLM_OVERRIDE_DUMMY": "1", + "EVAL_SQUAD_PATH": str(synthetic_squad), + } + run_result = _run_cli(["run", "--config", str(cli_config)], env_overrides=env) + run_id = next(line for line in run_result.stdout.split("\n") + if "cli-test" in line and "_" in line).strip().split()[-1] + + result = _run_cli(["show", run_id, "--html"], env_overrides=env) + assert result.returncode == 0 + html_path = tmp_eval_runs / run_id / "report.html" + assert html_path.exists() + assert " RunMetadata: + now = datetime.now(timezone.utc) + return RunMetadata( + run_id=run_id, config_name=run_id, config_path=f"{run_id}.yaml", + git_sha="x" * 7, started_at=now, finished_at=now, env_hash="h", + eval_set_versions=versions or {"squad_v2_dev_200": "v1"}, + n_questions=10, n_errors=0, + ) + + +def _r(qid: str, dataset: str, score: float) -> EvalResult: + return EvalResult( + question_id=qid, dataset=dataset, + retrieved_chunk_ids=[], retrieved_chunks=[], + generated_answer="", metrics={"recall_at_5": score}, + timings_ms={}, tokens={"prompt": 0, "completion": 0}, cost_usd=0.0, + ) + + +def _agg(metric_name: str, dataset: str | None, mean: float, n: int = 10) -> AggregatedMetric: + return AggregatedMetric( + metric_name=metric_name, dataset=dataset, + mean=mean, ci_low=mean - 0.05, ci_high=mean + 0.05, n=n, + ) + + +@pytest.fixture +def tmp_eval_runs(tmp_path, monkeypatch): + runs = tmp_path / "eval_runs" + runs.mkdir() + monkeypatch.setenv("EVAL_RUNS_DIR", str(runs)) + import importlib + import src.eval.storage + importlib.reload(src.eval.storage) + yield src.eval.storage + monkeypatch.delenv("EVAL_RUNS_DIR", raising=False) + importlib.reload(src.eval.storage) + + +def _save_synthetic_run(storage, run_id: str, results: list[EvalResult], + aggregated: list[AggregatedMetric], + versions: dict[str, str] | None = None) -> None: + meta = _make_metadata(run_id, versions=versions) + storage.save_run( + storage.EVAL_RUNS_DIR / run_id, meta, results, aggregated, + {"total_usd": 0.0, "mean_usd_per_query": 0.0}, + f"name: {run_id}\n", + ) + + +class TestCompareRuns: + def test_constant_shift_significant(self, tmp_eval_runs): + # Run A: scores 0.5 for all 10 questions; Run B: scores 0.6 for all. + results_a = [_r(f"q{i}", "squad_v2_dev_200", 0.5) for i in range(10)] + results_b = [_r(f"q{i}", "squad_v2_dev_200", 0.6) for i in range(10)] + agg_a = [_agg("recall_at_5", "squad_v2_dev_200", 0.5), + _agg("recall_at_5", None, 0.5)] + agg_b = [_agg("recall_at_5", "squad_v2_dev_200", 0.6), + _agg("recall_at_5", None, 0.6)] + _save_synthetic_run(tmp_eval_runs, "A", results_a, agg_a) + _save_synthetic_run(tmp_eval_runs, "B", results_b, agg_b) + + result = compare_runs("A", "B") + assert result.run_a.run_id == "A" + assert result.run_b.run_id == "B" + # All deltas ≈ +0.1 + for d in result.deltas: + assert d.delta == pytest.approx(0.1, abs=0.01) + assert d.significant is True + + def test_version_mismatch_raises(self, tmp_eval_runs): + results_a = [_r("q1", "squad_v2_dev_200", 0.5)] + results_b = [_r("q1", "squad_v2_dev_200", 0.6)] + agg = [_agg("recall_at_5", "squad_v2_dev_200", 0.5)] + _save_synthetic_run(tmp_eval_runs, "A", results_a, agg, + versions={"squad_v2_dev_200": "v1"}) + _save_synthetic_run(tmp_eval_runs, "B", results_b, agg, + versions={"squad_v2_dev_200": "v2"}) + + with pytest.raises(ValueError, match="eval set version mismatch"): + compare_runs("A", "B") + + def test_per_question_diff_sorted_and_capped(self, tmp_eval_runs): + # 12 questions; 5 with delta=+0.5, 5 with delta=-0.5, 2 with delta=0 + results_a = [] + results_b = [] + for i in range(5): + results_a.append(_r(f"big_pos_{i}", "squad_v2_dev_200", 0.0)) + results_b.append(_r(f"big_pos_{i}", "squad_v2_dev_200", 0.5)) + for i in range(5): + results_a.append(_r(f"big_neg_{i}", "squad_v2_dev_200", 0.5)) + results_b.append(_r(f"big_neg_{i}", "squad_v2_dev_200", 0.0)) + for i in range(2): + results_a.append(_r(f"flat_{i}", "squad_v2_dev_200", 0.5)) + results_b.append(_r(f"flat_{i}", "squad_v2_dev_200", 0.5)) + agg = [_agg("recall_at_5", "squad_v2_dev_200", 0.4), + _agg("recall_at_5", None, 0.4)] + _save_synthetic_run(tmp_eval_runs, "A", results_a, agg) + _save_synthetic_run(tmp_eval_runs, "B", results_b, agg) + + result = compare_runs("A", "B") + # At most 10 entries + assert len(result.per_question_diff) <= 10 + # Sorted by absolute delta descending + deltas = [abs(d["delta"]) for d in result.per_question_diff] + assert deltas == sorted(deltas, reverse=True) + # Flat questions should be excluded (or at least not at the top) + flat_in_top = sum(1 for d in result.per_question_diff if d["question_id"].startswith("flat_")) + assert flat_in_top == 0 diff --git a/tests/test_eval_config.py b/tests/test_eval_config.py new file mode 100644 index 00000000..78073810 --- /dev/null +++ b/tests/test_eval_config.py @@ -0,0 +1,175 @@ +"""Tests for src.eval.config.""" + +from __future__ import annotations + +from pathlib import Path + +import pytest +import yaml +from pydantic import ValidationError + +from src.eval.config import ( + EvalCfg, + EvalConfig, + GeneratorCfg, + PipelineCfg, + RetrieverCfg, + ChunkerCfg, + load_config, +) + + +def _baseline_dict() -> dict: + return { + "name": "baseline", + "description": "test", + "pipeline": { + "chunker": {"strategy": "recursive", "chunk_size": 512, "chunk_overlap": 64}, + "retriever": {"top_k": 5}, + "generator": {"model": "gpt-5-mini", "reasoning_model": "gpt-4.1-nano"}, + }, + "eval": { + "datasets": ["squad_v2_dev_200"], + "judge_model": "gpt-4.1-mini", + "bootstrap_n": 1000, + "permutation_n": 10000, + "seed": 42, + }, + } + + +class TestEvalConfigConstruction: + def test_full_construction(self): + cfg = EvalConfig.model_validate(_baseline_dict()) + assert cfg.name == "baseline" + assert cfg.pipeline.chunker.strategy == "recursive" + assert cfg.eval.datasets == ["squad_v2_dev_200"] + + def test_defaults_applied_when_omitted(self): + d = _baseline_dict() + del d["pipeline"]["chunker"]["chunk_size"] + del d["eval"]["bootstrap_n"] + cfg = EvalConfig.model_validate(d) + assert cfg.pipeline.chunker.chunk_size == 512 # default + assert cfg.eval.bootstrap_n == 1000 # default + + def test_missing_required_field_raises(self): + d = _baseline_dict() + del d["name"] + with pytest.raises(ValidationError): + EvalConfig.model_validate(d) + + def test_unknown_dataset_raises(self): + d = _baseline_dict() + d["eval"]["datasets"] = ["unknown_dataset"] + with pytest.raises(ValidationError): + EvalConfig.model_validate(d) + + def test_invalid_chunker_strategy_raises(self): + d = _baseline_dict() + d["pipeline"]["chunker"]["strategy"] = "wrong" + with pytest.raises(ValidationError): + EvalConfig.model_validate(d) + + +class TestLoadConfig: + def test_loads_yaml_file(self, tmp_path: Path): + path = tmp_path / "test.yaml" + path.write_text(yaml.safe_dump(_baseline_dict())) + cfg = load_config(path) + assert cfg.name == "baseline" + assert cfg.pipeline.retriever.top_k == 5 + + def test_round_trip(self, tmp_path: Path): + original = EvalConfig.model_validate(_baseline_dict()) + path = tmp_path / "rt.yaml" + path.write_text(yaml.safe_dump(original.model_dump())) + loaded = load_config(path) + assert loaded == original + + def test_missing_file_raises(self, tmp_path: Path): + with pytest.raises(FileNotFoundError): + load_config(tmp_path / "missing.yaml") + + def test_baseline_yaml_loads(self): + """The shipped baseline config must always validate.""" + cfg = load_config(Path("configs/eval/baseline.yaml")) + assert cfg.name == "baseline" + assert "squad_v2_dev_200" in cfg.eval.datasets + + +# --- Phase 2 schema additions ---------------------------------------------- + + +def test_phase2_subconfigs_default_to_off(tmp_path): + """Loading an existing baseline-shape YAML must produce all-default Phase 2 blocks.""" + yaml_text = """ +name: legacy_baseline +description: existing config without phase 2 blocks +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + retriever: {top_k: 5} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} +eval: + datasets: [squad_v2_dev_200] +""" + p = tmp_path / "legacy.yaml" + p.write_text(yaml_text) + from src.eval.config import load_config + cfg = load_config(p) + assert cfg.pipeline.embedder.name == "chroma_default" + assert cfg.pipeline.hybrid.enabled is False + assert cfg.pipeline.reranker.model is None + assert cfg.pipeline.query_rewriter.model is None + assert cfg.pipeline.refusal_handler.enabled is False + assert cfg.eval.spend_ceiling_usd is None + + +def test_phase2_subconfig_typed_values(tmp_path): + """Phase 2 fields validate to the right types.""" + yaml_text = """ +name: phase2g +description: refusal handler enabled +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + retriever: {top_k: 5} + embedder: {name: bge_small_en_v1_5} + hybrid: {enabled: true, bm25_top_k: 20, dense_top_k: 20, rrf_k: 60} + reranker: {model: ms_marco_minilm_l6_v2, rerank_top_n: 20, final_top_k: 5} + query_rewriter: {model: gpt-4.1-nano, max_expansions: 3} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} + refusal_handler: {enabled: true, similarity_threshold: 0.35} +eval: + datasets: [squad_v2_dev_200] + spend_ceiling_usd: 1.5 +""" + p = tmp_path / "phase2g.yaml" + p.write_text(yaml_text) + from src.eval.config import load_config + cfg = load_config(p) + assert cfg.pipeline.embedder.name == "bge_small_en_v1_5" + assert cfg.pipeline.hybrid.enabled is True + assert cfg.pipeline.reranker.model == "ms_marco_minilm_l6_v2" + assert cfg.pipeline.query_rewriter.model == "gpt-4.1-nano" + assert cfg.pipeline.refusal_handler.similarity_threshold == 0.35 + assert cfg.eval.spend_ceiling_usd == 1.5 + + +def test_phase2_unknown_field_rejected(tmp_path): + """extra='forbid' must reject unknown keys at load time.""" + yaml_text = """ +name: bad +description: typo in field name +pipeline: + chunker: {strategy: recursive, chunk_size: 512, chunk_overlap: 64} + retriever: {top_k: 5} + hybrid: {enabld: true} + generator: {model: gpt-5-mini, reasoning_model: gpt-4.1-nano} +eval: + datasets: [squad_v2_dev_200] +""" + p = tmp_path / "bad.yaml" + p.write_text(yaml_text) + from src.eval.config import load_config + with pytest.raises(ValidationError): + load_config(p) diff --git a/tests/test_eval_cost_ledger.py b/tests/test_eval_cost_ledger.py new file mode 100644 index 00000000..779862a8 --- /dev/null +++ b/tests/test_eval_cost_ledger.py @@ -0,0 +1,80 @@ +"""Tests for the Phase 2 cost ledger covering generator + judge + rewriter spend.""" + +from __future__ import annotations + +from src.eval.schemas import EvalResult + + +def test_eval_result_has_cost_breakdown_field(): + """EvalResult must carry a cost_breakdown dict with per-bucket spend.""" + r = EvalResult( + question_id="q1", + dataset="squad_v2_dev_200", + retrieved_chunk_ids=[], + retrieved_chunks=[], + generated_answer="", + metrics={}, + timings_ms={}, + tokens={}, + cost_usd=0.0, + cost_breakdown={"generator": 0.0, "judge": 0.0, "rewriter": 0.0}, + ) + assert r.cost_breakdown["generator"] == 0.0 + assert r.cost_breakdown["judge"] == 0.0 + assert r.cost_breakdown["rewriter"] == 0.0 + + +def test_eval_result_cost_breakdown_defaults(): + """cost_breakdown must default to a generator-only dict when omitted.""" + r = EvalResult( + question_id="q1", + dataset="squad_v2_dev_200", + retrieved_chunk_ids=[], + retrieved_chunks=[], + generated_answer="", + metrics={}, + timings_ms={}, + tokens={}, + cost_usd=0.05, + ) + # back-compat default: existing records read as generator-only + assert r.cost_breakdown == {"generator": 0.05, "judge": 0.0, "rewriter": 0.0} + + +def test_aggregator_sums_cost_breakdown_into_totals(): + """aggregate_costs must surface per-bucket totals alongside total_usd.""" + from src.eval.metrics.operational import aggregate_costs + + results = [ + EvalResult( + question_id="q1", dataset="d", retrieved_chunk_ids=[], retrieved_chunks=[], + generated_answer="", metrics={}, timings_ms={}, tokens={}, + cost_usd=0.10, + cost_breakdown={"generator": 0.04, "judge": 0.05, "rewriter": 0.01}, + ), + EvalResult( + question_id="q2", dataset="d", retrieved_chunk_ids=[], retrieved_chunks=[], + generated_answer="", metrics={}, timings_ms={}, tokens={}, + cost_usd=0.20, + cost_breakdown={"generator": 0.08, "judge": 0.10, "rewriter": 0.02}, + ), + ] + summary = aggregate_costs(results) + assert round(summary["total_usd"], 4) == 0.30 + assert round(summary["generator_total_usd"], 4) == 0.12 + assert round(summary["judge_total_usd"], 4) == 0.15 + assert round(summary["rewriter_total_usd"], 4) == 0.03 + + +def test_llm_handler_generate_with_usage_returns_tokens(): + """LLMHandler.generate_with_usage must return (text, prompt_tokens, completion_tokens).""" + from src.llm_handler import LLMHandler + + # Use the dummy fallback path — no API key needed. + handler = LLMHandler("__dummy__") + text, prompt_tokens, completion_tokens = handler.generate_with_usage( + prompt="What is RAG?", system_prompt="Be brief." + ) + assert isinstance(text, str) + assert isinstance(prompt_tokens, int) and prompt_tokens > 0 + assert isinstance(completion_tokens, int) and completion_tokens >= 0 diff --git a/tests/test_eval_datasets_ml_papers.py b/tests/test_eval_datasets_ml_papers.py new file mode 100644 index 00000000..bcd8bf14 --- /dev/null +++ b/tests/test_eval_datasets_ml_papers.py @@ -0,0 +1,119 @@ +"""Tests for src.eval.datasets.ml_papers.""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +import pytest + +from src.eval.datasets.ml_papers import ( + DEFAULT_QUESTIONS_PATH, + DEFAULT_MANIFEST_PATH, + ManifestVerificationError, + load_questions, + verify_corpus_manifest, +) +from src.eval.schemas import EvalQuestion + + +@pytest.fixture +def temp_data_dir(tmp_path: Path) -> Path: + return tmp_path + + +def _write_questions(path: Path, questions: list[EvalQuestion]) -> None: + with path.open("w") as f: + for q in questions: + f.write(q.model_dump_json() + "\n") + + +class TestLoadQuestions: + def test_loads_existing_jsonl(self, temp_data_dir: Path): + path = temp_data_dir / "questions.jsonl" + sample = [ + EvalQuestion( + id="q1", question="What is attention?", + gold_answer="A weighted sum.", gold_chunk_ids=["c1"], + ), + ] + _write_questions(path, sample) + loaded = load_questions(path) + assert loaded == sample + + def test_empty_file_returns_empty_list(self, temp_data_dir: Path): + path = temp_data_dir / "questions.jsonl" + path.write_text("") + assert load_questions(path) == [] + + def test_missing_file_raises(self, temp_data_dir: Path): + with pytest.raises(FileNotFoundError): + load_questions(temp_data_dir / "missing.jsonl") + + +class TestVerifyCorpusManifest: + def test_valid_manifest_returns_papers(self, temp_data_dir: Path): + pdf_path = temp_data_dir / "fake.pdf" + pdf_path.write_bytes(b"hello world") + sha = hashlib.sha256(b"hello world").hexdigest() + + manifest_path = temp_data_dir / "manifest.json" + manifest_path.write_text(json.dumps({ + "version": "v1", + "description": "test", + "papers": [{ + "id": "fake", + "title": "Fake Paper", + "source_url": "https://example.com", + "local_path": str(pdf_path), + "sha256": sha, + }], + })) + papers = verify_corpus_manifest(manifest_path) + assert len(papers) == 1 + assert papers[0]["id"] == "fake" + + def test_tampered_sha_raises(self, temp_data_dir: Path): + pdf_path = temp_data_dir / "fake.pdf" + pdf_path.write_bytes(b"hello world") + bad_sha = "0" * 64 + + manifest_path = temp_data_dir / "manifest.json" + manifest_path.write_text(json.dumps({ + "version": "v1", + "description": "test", + "papers": [{ + "id": "fake", "title": "Fake", "source_url": "https://x", + "local_path": str(pdf_path), "sha256": bad_sha, + }], + })) + with pytest.raises(ManifestVerificationError, match="sha256 mismatch"): + verify_corpus_manifest(manifest_path) + + def test_missing_pdf_raises(self, temp_data_dir: Path): + manifest_path = temp_data_dir / "manifest.json" + manifest_path.write_text(json.dumps({ + "version": "v1", + "description": "test", + "papers": [{ + "id": "missing", "title": "Missing", "source_url": "https://x", + "local_path": str(temp_data_dir / "absent.pdf"), + "sha256": "0" * 64, + }], + })) + with pytest.raises(ManifestVerificationError, match="not found"): + verify_corpus_manifest(manifest_path) + + def test_empty_papers_list_is_ok(self, temp_data_dir: Path): + manifest_path = temp_data_dir / "manifest.json" + manifest_path.write_text(json.dumps({ + "version": "v1", "description": "skeleton", "papers": [], + })) + assert verify_corpus_manifest(manifest_path) == [] + + +class TestDefaults: + def test_default_paths_point_at_v1(self): + assert "ml_papers_v1" in str(DEFAULT_QUESTIONS_PATH) + assert "ml_papers_v1" in str(DEFAULT_MANIFEST_PATH) diff --git a/tests/test_eval_datasets_squad.py b/tests/test_eval_datasets_squad.py new file mode 100644 index 00000000..06a0c347 --- /dev/null +++ b/tests/test_eval_datasets_squad.py @@ -0,0 +1,96 @@ +"""Tests for src.eval.datasets.squad_v2.""" + +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from src.eval.datasets.squad_v2 import ( + DEFAULT_OUTPUT_PATH, + DEFAULT_SAMPLE_SIZE, + DEFAULT_SEED, + load_frozen, + sample_and_freeze, +) +from src.eval.schemas import EvalQuestion + + +@pytest.fixture +def temp_freeze_path(tmp_path: Path) -> Path: + return tmp_path / "questions.jsonl" + + +class TestSampleAndFreeze: + def test_sample_size_matches(self, temp_freeze_path: Path): + result = sample_and_freeze( + output_path=temp_freeze_path, sample_size=10, seed=42 + ) + assert len(result) == 10 + assert temp_freeze_path.exists() + + def test_seed_reproducibility(self, tmp_path: Path): + a = sample_and_freeze( + output_path=tmp_path / "a.jsonl", sample_size=5, seed=99 + ) + b = sample_and_freeze( + output_path=tmp_path / "b.jsonl", sample_size=5, seed=99 + ) + assert [q.id for q in a] == [q.id for q in b] + + def test_includes_unanswerable_rows(self, temp_freeze_path: Path): + result = sample_and_freeze( + output_path=temp_freeze_path, sample_size=50, seed=7 + ) + n_unanswerable = sum(1 for q in result if q.is_unanswerable) + n_answerable = sum(1 for q in result if not q.is_unanswerable) + assert n_unanswerable > 0 + assert n_answerable > 0 + + def test_each_row_has_required_fields(self, temp_freeze_path: Path): + result = sample_and_freeze( + output_path=temp_freeze_path, sample_size=5, seed=1 + ) + for q in result: + assert isinstance(q, EvalQuestion) + assert q.id + assert q.question + if not q.is_unanswerable: + assert q.gold_answer + assert len(q.gold_chunk_ids) >= 1 + else: + assert q.gold_answer is None + assert q.gold_chunk_ids == [] + + +class TestLoadFrozen: + def test_round_trip(self, temp_freeze_path: Path): + original = sample_and_freeze( + output_path=temp_freeze_path, sample_size=5, seed=5 + ) + loaded = load_frozen(temp_freeze_path) + assert loaded == original + + def test_loads_from_jsonl_file(self, tmp_path: Path): + path = tmp_path / "manual.jsonl" + manual = [ + EvalQuestion(id="q1", question="Why?", gold_answer="A", gold_chunk_ids=["c"]), + EvalQuestion(id="q2", question="How?", is_unanswerable=True), + ] + with path.open("w") as f: + for q in manual: + f.write(q.model_dump_json() + "\n") + loaded = load_frozen(path) + assert loaded == manual + + def test_missing_file_raises(self, tmp_path: Path): + with pytest.raises(FileNotFoundError): + load_frozen(tmp_path / "missing.jsonl") + + +class TestDefaults: + def test_default_paths_and_constants_exposed(self): + assert DEFAULT_SAMPLE_SIZE == 200 + assert DEFAULT_SEED == 12345 + assert "squad_v2_dev_200" in str(DEFAULT_OUTPUT_PATH) diff --git a/tests/test_eval_embedder_bge.py b/tests/test_eval_embedder_bge.py new file mode 100644 index 00000000..6eee0691 --- /dev/null +++ b/tests/test_eval_embedder_bge.py @@ -0,0 +1,53 @@ +"""Tests for BgeEmbedder — a Chroma EmbeddingFunction adapter for BAAI/bge-small-en-v1.5.""" + +from __future__ import annotations + +import pytest + + +@pytest.fixture(scope="module") +def embedder(): + """Module-scoped to amortize the model-load cost across tests.""" + from src.eval.embedders import BgeEmbedder + return BgeEmbedder() + + +def test_returns_384_dim_vectors(embedder): + import numpy as np + out = embedder(["hello world"]) + assert len(out) == 1 + assert len(out[0]) == 384 + # WHY (float, np.floating): chromadb 1.5.8's EmbeddingFunction.__init_subclass__ + # wraps __call__ with normalize_embeddings(), which always converts scalars to + # numpy.float32 regardless of what the adapter returns. Python float alone fails. + assert all(isinstance(x, (float, np.floating)) for x in out[0]) + + +def test_synonyms_closer_than_unrelated(embedder): + """Sanity check that the right model is loaded — not a stub.""" + import numpy as np + a, b, c = embedder(["cat", "feline", "airplane"]) + a, b, c = np.array(a), np.array(b), np.array(c) + cos = lambda u, v: float(u @ v / (np.linalg.norm(u) * np.linalg.norm(v))) + assert cos(a, b) > cos(a, c), "BGE should rank cat~feline > cat~airplane" + + +def test_chroma_collection_uses_embedder(embedder): + """End-to-end: a Chroma collection created with BgeEmbedder retrieves the right doc.""" + import chromadb + client = chromadb.EphemeralClient() + coll = client.get_or_create_collection( + name="test_bge_e2e", + embedding_function=embedder, + metadata={"hnsw:space": "cosine"}, + ) + coll.upsert( + ids=["d1", "d2", "d3"], + documents=[ + "Cats are small carnivorous mammals often kept as pets.", + "Airplanes are powered flying vehicles with fixed wings.", + "Dogs are domesticated descendants of wolves.", + ], + ) + res = coll.query(query_texts=["What is a feline?"], n_results=1) + assert res["ids"][0][0] == "d1" diff --git a/tests/test_eval_integration.py b/tests/test_eval_integration.py new file mode 100644 index 00000000..c23c591b --- /dev/null +++ b/tests/test_eval_integration.py @@ -0,0 +1,110 @@ +"""End-to-end integration test for Sub-plan 1B: full eval lifecycle. + +Runs two evals back-to-back with different top_k values, then compares +them. Uses DummyLLM + 5-question synthetic SQuAD slice so no network +calls are made. +""" + +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from src.eval import EvalConfig, EvalRunner, compare_runs, list_runs, load_run +from src.eval.schemas import EvalQuestion + + +class DummyLLM: + """Returns canned data — JSON for judges, plain for generation.""" + def generate(self, prompt: str, system_prompt: str | None = None) -> str: + if "JSON" in (system_prompt or "") or '"score"' in prompt or '"is_refusal"' in prompt: + return json.dumps({ + "score": 1.0, "claims": [], "chunks": [], + "factual_match": 1.0, "is_refusal": False, "reasoning": "ok", + }) + return "" + + +def _make_config(name: str, top_k: int) -> EvalConfig: + return EvalConfig.model_validate({ + "name": name, "description": f"top_k={top_k}", + "pipeline": { + "chunker": {"strategy": "recursive", "chunk_size": 256, "chunk_overlap": 32}, + "retriever": {"top_k": top_k}, + "generator": {"model": "gpt-4.1-nano", "reasoning_model": None}, + }, + "eval": { + "datasets": ["squad_v2_dev_200"], + "judge_model": "gpt-4.1-nano", + "bootstrap_n": 100, "permutation_n": 100, "seed": 42, + }, + }) + + +@pytest.fixture +def tmp_eval_runs(tmp_path, monkeypatch): + runs = tmp_path / "eval_runs" + runs.mkdir() + monkeypatch.setenv("EVAL_RUNS_DIR", str(runs)) + import importlib + import src.eval.storage + importlib.reload(src.eval.storage) + yield src.eval.storage + monkeypatch.delenv("EVAL_RUNS_DIR", raising=False) + importlib.reload(src.eval.storage) + + +@pytest.fixture +def synthetic_squad(monkeypatch, tmp_path): + questions = [ + EvalQuestion( + id=f"q{i}", question=f"What is fact {i}?", + gold_answer=f"Fact {i}.", gold_chunk_ids=[f"q{i}"], + metadata={"context": f"Fact {i} is important.", "title": "t"}, + ) + for i in range(5) + ] + path = tmp_path / "squad.jsonl" + with path.open("w") as f: + for q in questions: + f.write(q.model_dump_json() + "\n") + monkeypatch.setattr("src.eval.datasets.squad_v2.DEFAULT_OUTPUT_PATH", path) + return path + + +class TestFullLifecycle: + def test_two_runs_then_compare(self, tmp_eval_runs, synthetic_squad): + # Run 1: top_k=3 + cfg_a = _make_config("topk-3", top_k=3) + runner_a = EvalRunner( + cfg_a, llm_override=DummyLLM(), judge_llm_override=DummyLLM(), + ) + meta_a = runner_a.run() + assert meta_a.n_questions == 5 + assert meta_a.n_errors == 0 + + # Run 2: top_k=1 + cfg_b = _make_config("topk-1", top_k=1) + runner_b = EvalRunner( + cfg_b, llm_override=DummyLLM(), judge_llm_override=DummyLLM(), + ) + meta_b = runner_b.run() + assert meta_b.n_questions == 5 + + # list_runs sees both + runs = list_runs() + assert len(runs) == 2 + + # load_run round-trips both + loaded_a = load_run(meta_a.run_id) + loaded_b = load_run(meta_b.run_id) + assert loaded_a["metadata"].config_name == "topk-3" + assert loaded_b["metadata"].config_name == "topk-1" + + # compare_runs produces a CompareResult (eval_set_versions match + # since both runs used the same synthetic SQuAD) + cmp = compare_runs(meta_a.run_id, meta_b.run_id) + assert cmp.run_a.run_id == meta_a.run_id + assert cmp.run_b.run_id == meta_b.run_id diff --git a/tests/test_eval_metrics_generation.py b/tests/test_eval_metrics_generation.py new file mode 100644 index 00000000..be93c19b --- /dev/null +++ b/tests/test_eval_metrics_generation.py @@ -0,0 +1,115 @@ +"""Tests for src.eval.metrics.generation.""" + +from __future__ import annotations + +import json +import math +from unittest.mock import patch + +import numpy as np +import pytest + +from src.eval.metrics.generation import ( + answer_correctness, + context_recall, + judge_answer_relevancy, + judge_context_precision, + judge_faithfulness, +) + + +class FakeLLM: + """Returns queued JSON payloads in order from generate().""" + + def __init__(self, payloads: list[dict]): + self.payloads = list(payloads) + self.calls: list[str] = [] + + def generate(self, prompt: str, system_prompt: str | None = None) -> str: + self.calls.append(prompt) + if not self.payloads: + raise RuntimeError("FakeLLM out of payloads") + return json.dumps(self.payloads.pop(0)) + + +class TestContextRecall: + def test_perfect_recall(self): + gold = ["c1", "c2"] + retrieved = ["c1", "c2", "c3"] + assert context_recall(gold, retrieved) == 1.0 + + def test_partial_recall(self): + gold = ["c1", "c2", "c3", "c4"] + retrieved = ["c1", "c2", "x"] + assert context_recall(gold, retrieved) == pytest.approx(0.5) + + def test_zero_recall(self): + assert context_recall(["c1"], ["x", "y"]) == 0.0 + + def test_empty_gold_returns_nan(self): + assert math.isnan(context_recall([], ["x"])) + + +class TestAnswerCorrectness: + def test_high_similarity_and_judge_match(self): + with patch( + "src.eval.metrics.generation._embed", + side_effect=lambda text: np.array([1.0, 0.0, 0.0]), + ): + llm = FakeLLM([{"factual_match": 1.0, "reasoning": "Same answer."}]) + score, details = answer_correctness( + generated="Paris is the capital of France.", + gold="The capital of France is Paris.", + llm=llm, + ) + assert score == pytest.approx(1.0) + assert details["cosine"] == pytest.approx(1.0) + assert details["judge_factual_match"] == pytest.approx(1.0) + + def test_low_similarity_and_judge_mismatch(self): + with patch( + "src.eval.metrics.generation._embed", + side_effect=[ + np.array([1.0, 0.0]), + np.array([0.0, 1.0]), + ], + ): + llm = FakeLLM([{"factual_match": 0.0, "reasoning": "Different."}]) + score, details = answer_correctness( + generated="The Eiffel Tower is in Paris.", + gold="The Statue of Liberty is in New York.", + llm=llm, + ) + assert score == pytest.approx(0.0) + + def test_partial_match(self): + with patch( + "src.eval.metrics.generation._embed", + side_effect=[np.array([1.0, 0.0]), np.array([0.5, 0.5])], + ): + llm = FakeLLM([{"factual_match": 0.5, "reasoning": "Partly."}]) + score, _ = answer_correctness( + generated="Mostly right.", gold="The answer.", llm=llm + ) + # cosine = 0.5/sqrt(0.5) ≈ 0.7071; judge = 0.5; mean ≈ 0.6036 + assert score == pytest.approx((1 / np.sqrt(2) + 0.5) / 2, abs=1e-3) + + +class TestJudgeWrappers: + def test_judge_faithfulness_returns_score(self): + llm = FakeLLM([{"claims": [], "score": 0.9, "reasoning": "ok"}]) + score, details = judge_faithfulness("ans", ["ctx1"], llm) + assert score == pytest.approx(0.9) + assert "claims" in details + + def test_judge_answer_relevancy_returns_score(self): + llm = FakeLLM([{"score": 0.8, "reasoning": "ok"}]) + score, details = judge_answer_relevancy("Q?", "A.", llm) + assert score == pytest.approx(0.8) + assert details["reasoning"] == "ok" + + def test_judge_context_precision_returns_score(self): + llm = FakeLLM([{"chunks": [], "score": 0.6, "reasoning": "ok"}]) + score, details = judge_context_precision("Q?", ["c1", "c2"], llm) + assert score == pytest.approx(0.6) + assert "chunks" in details diff --git a/tests/test_eval_metrics_operational.py b/tests/test_eval_metrics_operational.py new file mode 100644 index 00000000..daa41c39 --- /dev/null +++ b/tests/test_eval_metrics_operational.py @@ -0,0 +1,84 @@ +"""Tests for src.eval.metrics.operational.""" + +from __future__ import annotations + +import pytest + +from src.eval.metrics.operational import ( + aggregate_costs, + aggregate_timings, + aggregate_tokens, +) +from src.eval.schemas import EvalResult + + +def _make_result( + *, + retrieve_ms: float = 100.0, + generate_ms: float = 1000.0, + prompt_tokens: int = 100, + completion_tokens: int = 50, + cost: float = 0.001, + error: str | None = None, +) -> EvalResult: + return EvalResult( + question_id="q", + dataset="d", + retrieved_chunk_ids=[], + retrieved_chunks=[], + generated_answer="", + metrics={}, + timings_ms={"retrieve": retrieve_ms, "generate": generate_ms}, + tokens={"prompt": prompt_tokens, "completion": completion_tokens}, + cost_usd=cost, + error=error, + ) + + +class TestAggregateTimings: + def test_basic_percentiles(self): + results = [_make_result(retrieve_ms=float(i)) for i in range(1, 11)] + agg = aggregate_timings(results) + assert agg["retrieve"]["p50"] == pytest.approx(5.5) + assert agg["retrieve"]["p95"] == pytest.approx(9.55) + assert agg["retrieve"]["p99"] == pytest.approx(9.91) + + def test_skips_errored_results(self): + results = [ + _make_result(retrieve_ms=10.0), + _make_result(retrieve_ms=99999.0, error="boom"), + ] + agg = aggregate_timings(results) + assert agg["retrieve"]["p50"] == pytest.approx(10.0) + + def test_empty_input_returns_empty_dict(self): + assert aggregate_timings([]) == {} + + +class TestAggregateCosts: + def test_total_and_mean(self): + results = [_make_result(cost=0.01), _make_result(cost=0.03)] + agg = aggregate_costs(results) + assert agg["total_usd"] == pytest.approx(0.04) + assert agg["mean_usd_per_query"] == pytest.approx(0.02) + + def test_skips_errored(self): + results = [ + _make_result(cost=0.01), + _make_result(cost=999.0, error="boom"), + ] + agg = aggregate_costs(results) + assert agg["total_usd"] == pytest.approx(0.01) + + +class TestAggregateTokens: + def test_totals_and_means(self): + results = [ + _make_result(prompt_tokens=100, completion_tokens=50), + _make_result(prompt_tokens=200, completion_tokens=100), + ] + agg = aggregate_tokens(results) + assert agg["total_prompt"] == 300 + assert agg["total_completion"] == 150 + assert agg["mean_prompt"] == pytest.approx(150.0) + assert agg["mean_completion"] == pytest.approx(75.0) diff --git a/tests/test_eval_metrics_refusal.py b/tests/test_eval_metrics_refusal.py new file mode 100644 index 00000000..3ec60047 --- /dev/null +++ b/tests/test_eval_metrics_refusal.py @@ -0,0 +1,97 @@ +"""Tests for src.eval.metrics.refusal.""" + +from __future__ import annotations + +import json + +import pytest + +from src.eval.metrics.refusal import is_refusal, refusal_correctness + + +class FakeLLM: + """Returns a fixed JSON string from generate(); records calls.""" + + def __init__(self, payload: dict): + self.payload = payload + self.calls: list[tuple[str, str | None]] = [] + + def generate(self, prompt: str, system_prompt: str | None = None) -> str: + self.calls.append((prompt, system_prompt)) + return json.dumps(self.payload) + + +class TestIsRefusal: + @pytest.mark.parametrize( + "answer", + [ + "I cannot answer this based on the provided context.", + "The context does not contain information about that.", + "I don't know based on the documents I was given.", + "This question cannot be answered from the retrieved sources.", + "There is no information in the context to answer this question.", + "The provided context does not address this question.", + ], + ) + def test_clear_refusals_match(self, answer: str): + assert is_refusal(answer) is True + + @pytest.mark.parametrize( + "answer", + [ + "The Eiffel Tower is in Paris.", + "According to the context, X equals Y.", + "Yes — the documents state that ...", + ], + ) + def test_clear_answers_dont_match(self, answer: str): + assert is_refusal(answer) is False + + +class TestRefusalCorrectness: + def test_correct_refusal_on_unanswerable(self): + llm = FakeLLM({"is_refusal": True}) + score = refusal_correctness( + answer="I cannot answer this based on the provided context.", + is_unanswerable=True, + llm=llm, + ) + assert score == 1.0 + assert llm.calls == [] + + def test_incorrect_attempt_on_unanswerable(self): + llm = FakeLLM({"is_refusal": False}) + score = refusal_correctness( + answer="The capital of France is Paris.", + is_unanswerable=True, + llm=llm, + ) + assert score == 0.0 + + def test_correct_attempt_on_answerable(self): + llm = FakeLLM({"is_refusal": False}) + score = refusal_correctness( + answer="The capital of France is Paris.", + is_unanswerable=False, + llm=llm, + ) + assert score == 1.0 + + def test_incorrect_refusal_on_answerable(self): + llm = FakeLLM({"is_refusal": True}) + score = refusal_correctness( + answer="I cannot answer this based on the provided context.", + is_unanswerable=False, + llm=llm, + ) + assert score == 0.0 + + def test_llm_judge_fallback_on_ambiguous(self): + llm = FakeLLM({"is_refusal": True}) + score = refusal_correctness( + answer="That's an interesting question, but I'd rather not speculate.", + is_unanswerable=True, + llm=llm, + ) + assert score == 1.0 + assert len(llm.calls) == 1 diff --git a/tests/test_eval_metrics_retrieval.py b/tests/test_eval_metrics_retrieval.py new file mode 100644 index 00000000..00c51a5c --- /dev/null +++ b/tests/test_eval_metrics_retrieval.py @@ -0,0 +1,102 @@ +"""Tests for src.eval.metrics.retrieval — Recall@k, MRR@k, nDCG@k.""" + +from __future__ import annotations + +import math + +import pytest + +from src.eval.metrics.retrieval import mrr_at_k, ndcg_at_k, recall_at_k + + +# --------------------------------------------------------------------------- +# Recall@k +# --------------------------------------------------------------------------- + + +class TestRecallAtK: + def test_perfect_recall(self): + assert recall_at_k(["a", "b", "c"], ["a", "b", "c", "x", "y"], 5) == pytest.approx(1.0) + + def test_partial_recall(self): + assert recall_at_k(["a", "b", "c"], ["a", "x", "y", "z", "w"], 5) == pytest.approx(1 / 3) + + def test_zero_recall(self): + assert recall_at_k(["a", "b", "c"], ["x", "y", "z"], 3) == pytest.approx(0.0) + + def test_truncation_misses(self): + # k=2 cuts before a and b appear + assert recall_at_k(["a", "b"], ["x", "y", "a", "b"], 2) == pytest.approx(0.0) + + def test_truncation_hits(self): + # k=4 includes both + assert recall_at_k(["a", "b"], ["x", "y", "a", "b"], 4) == pytest.approx(1.0) + + def test_empty_gold_is_nan(self): + assert math.isnan(recall_at_k([], ["a", "b"], 5)) + + def test_k_larger_than_retrieved(self): + assert recall_at_k(["a"], ["a"], 10) == pytest.approx(1.0) + + +# --------------------------------------------------------------------------- +# MRR@k +# --------------------------------------------------------------------------- + + +class TestMrrAtK: + def test_rank_1(self): + assert mrr_at_k(["a"], ["a", "b", "c"], 5) == pytest.approx(1.0) + + def test_rank_2(self): + assert mrr_at_k(["a"], ["x", "a", "b"], 5) == pytest.approx(0.5) + + def test_no_hit(self): + assert mrr_at_k(["a"], ["x", "y", "z"], 3) == pytest.approx(0.0) + + def test_multi_gold_first_hit(self): + # b is at rank 2, a is at rank 3 — MRR uses FIRST hit (b, rank 2 → 1/2) + assert mrr_at_k(["a", "b"], ["x", "b", "a"], 5) == pytest.approx(0.5) + + def test_truncation_misses(self): + # a is at position 4 (0-indexed 3), k=3 cuts it off + assert mrr_at_k(["a"], ["x", "y", "z", "a"], 3) == pytest.approx(0.0) + + def test_truncation_hits(self): + # k=4 includes a at rank 4 → 1/4 + assert mrr_at_k(["a"], ["x", "y", "z", "a"], 4) == pytest.approx(0.25) + + def test_empty_gold_is_nan(self): + assert math.isnan(mrr_at_k([], ["a"], 5)) + + +# --------------------------------------------------------------------------- +# nDCG@k +# --------------------------------------------------------------------------- + + +class TestNdcgAtK: + def test_perfect_ndcg(self): + assert ndcg_at_k(["a", "b", "c"], ["a", "b", "c"], 3) == pytest.approx(1.0) + + def test_zero_ndcg(self): + assert ndcg_at_k(["a"], ["x", "y", "z"], 3) == pytest.approx(0.0) + + def test_single_hit_rank_1(self): + # DCG = 1/log2(2) = 1, IDCG = 1 → nDCG = 1.0 + assert ndcg_at_k(["a"], ["a", "x"], 2) == pytest.approx(1.0) + + def test_single_hit_rank_2(self): + # DCG = 1/log2(3), IDCG = 1/log2(2) = 1 → nDCG = 1/log2(3) + assert ndcg_at_k(["a"], ["x", "a"], 2) == pytest.approx(1.0 / math.log2(3)) + + def test_partial_hit(self): + # gold=["a","b"], retrieved=["x","a"], k=2 + # DCG = 1/log2(3) (a at rank 2, b not present) + # IDCG = 1/log2(2) + 1/log2(3) (ideal: both hits at ranks 1 and 2) + dcg = 1.0 / math.log2(3) + idcg = 1.0 / math.log2(2) + 1.0 / math.log2(3) + assert ndcg_at_k(["a", "b"], ["x", "a"], 2) == pytest.approx(dcg / idcg) + + def test_empty_gold_is_nan(self): + assert math.isnan(ndcg_at_k([], ["a"], 5)) diff --git a/tests/test_eval_pipeline_factory.py b/tests/test_eval_pipeline_factory.py new file mode 100644 index 00000000..6e8783d7 --- /dev/null +++ b/tests/test_eval_pipeline_factory.py @@ -0,0 +1,96 @@ +"""Tests for src.eval.pipeline_factory.""" + +from __future__ import annotations + +import json + +import pytest + +from src.eval.config import EvalConfig +from src.eval.pipeline_factory import EvalPipeline, build_pipeline +from src.eval.schemas import EvalQuestion + + +class DummyLLM: + """Returns a fixed answer; tracks calls.""" + def __init__(self, answer: str = ""): + self.answer = answer + self.calls: list[tuple[str, str | None]] = [] + + def generate(self, prompt: str, system_prompt: str | None = None) -> str: + self.calls.append((prompt, system_prompt)) + return self.answer + + +def _baseline_config() -> EvalConfig: + return EvalConfig.model_validate({ + "name": "test", + "description": "", + "pipeline": { + "chunker": {"strategy": "recursive", "chunk_size": 256, "chunk_overlap": 32}, + "retriever": {"top_k": 3}, + "generator": {"model": "gpt-4.1-nano", "reasoning_model": None}, + }, + "eval": { + "datasets": ["squad_v2_dev_200"], + "judge_model": "gpt-4.1-nano", + "bootstrap_n": 100, "permutation_n": 100, "seed": 7, + }, + }) + + +def _squad_question(qid: str, ctx: str, q: str = "Q?") -> EvalQuestion: + return EvalQuestion( + id=qid, question=q, gold_answer="A", gold_chunk_ids=[qid], + metadata={"context": ctx, "title": "t"}, + ) + + +class TestBuildPipeline: + def test_returns_pipeline_with_components(self): + cfg = _baseline_config() + p = build_pipeline(cfg, "squad_v2_dev_200", + llm_override=DummyLLM("answer"), + judge_llm_override=DummyLLM("{}")) + assert isinstance(p, EvalPipeline) + assert p.config is cfg + assert p.dataset_name == "squad_v2_dev_200" + p.teardown() + + +class TestIngestAndQuery: + def test_squad_ingest_then_query(self): + cfg = _baseline_config() + p = build_pipeline(cfg, "squad_v2_dev_200", + llm_override=DummyLLM("Paris"), + judge_llm_override=DummyLLM("{}")) + try: + qs = [ + _squad_question("q1", "Paris is the capital of France.", "What is the capital of France?"), + _squad_question("q2", "The Eiffel Tower is in Paris.", "Where is the Eiffel Tower?"), + ] + p.ingest(qs) + + chunks, answer, telemetry = p.query("What is the capital of France?") + assert isinstance(chunks, list) + assert len(chunks) >= 1 + assert answer == "Paris" + assert "timings_ms" in telemetry + assert "tokens" in telemetry + assert "cost_usd" in telemetry + assert telemetry["timings_ms"]["retrieve"] >= 0.0 + assert telemetry["timings_ms"]["generate"] >= 0.0 + assert telemetry["tokens"]["prompt"] > 0 + assert telemetry["tokens"]["completion"] >= 0 + assert telemetry["cost_usd"] >= 0.0 + finally: + p.teardown() + + +class TestTeardown: + def test_teardown_does_not_raise(self): + cfg = _baseline_config() + p = build_pipeline(cfg, "squad_v2_dev_200", + llm_override=DummyLLM(), + judge_llm_override=DummyLLM()) + p.teardown() # should not raise diff --git a/tests/test_eval_pipeline_factory_phase2.py b/tests/test_eval_pipeline_factory_phase2.py new file mode 100644 index 00000000..e4f00d63 --- /dev/null +++ b/tests/test_eval_pipeline_factory_phase2.py @@ -0,0 +1,83 @@ +"""Phase 2 factory tests — every tier YAML produces a pipeline with expected lever activations. + +Design note on hybrid_retriever: + The hybrid retriever is built lazily during ingest() (after the chunk corpus is + available), NOT at build_pipeline() time. So the parametrized test checks the + *config flag* (cfg.pipeline.hybrid.enabled) to confirm YAML routing, not the + runtime field (pipeline.hybrid_retriever). The smoke test exercises the runtime + field after an ingest call. +""" + +from __future__ import annotations + +from pathlib import Path + +import pytest + +from src.eval.config import load_config +from src.eval.pipeline_factory import build_pipeline + + +PHASE2_DIR = Path("configs/eval/phase2") + + +@pytest.fixture +def stub_llm(): + class _S: + def generate(self, prompt, system_prompt=None): + return "stub answer" + def generate_with_usage(self, prompt, system_prompt=None): + return "stub answer", 10, 5 + return _S() + + +@pytest.mark.parametrize("yaml_name,expects", [ + ("phase2_baseline.yaml", {"rewriter": False, "reranker": False, "refusal": False, "hybrid": False, "embedder": "chroma_default"}), + ("phase2b_embedder.yaml", {"rewriter": False, "reranker": False, "refusal": False, "hybrid": False, "embedder": "bge_small_en_v1_5"}), + ("phase2c_hybrid.yaml", {"rewriter": False, "reranker": False, "refusal": False, "hybrid": True, "embedder": "bge_small_en_v1_5"}), + ("phase2d_rerank.yaml", {"rewriter": False, "reranker": True, "refusal": False, "hybrid": True, "embedder": "bge_small_en_v1_5"}), + ("phase2e_rewrite.yaml", {"rewriter": True, "reranker": True, "refusal": False, "hybrid": True, "embedder": "bge_small_en_v1_5"}), + ("phase2g_refusal.yaml", {"rewriter": True, "reranker": True, "refusal": True, "hybrid": True, "embedder": "bge_small_en_v1_5"}), +]) +def test_phase2_yaml_builds_pipeline_with_expected_attrs(yaml_name, expects, stub_llm): + cfg = load_config(PHASE2_DIR / yaml_name) + pipeline = build_pipeline( + cfg, dataset_name="squad_v2_dev_200", + llm_override=stub_llm, judge_llm_override=stub_llm, + ) + try: + assert (pipeline.rewriter is not None) == expects["rewriter"] + assert (pipeline.reranker is not None) == expects["reranker"] + assert (pipeline.refusal_handler is not None) == expects["refusal"] + # WHY cfg.pipeline.hybrid.enabled (not pipeline.hybrid_retriever is not None): + # hybrid_retriever is built lazily in _ingest_squad() once the chunk corpus + # is available. build_pipeline() always sets it to None; the config flag is + # the authoritative signal that the lever was parsed and routed correctly. + assert cfg.pipeline.hybrid.enabled == expects["hybrid"] + assert cfg.pipeline.embedder.name == expects["embedder"] + finally: + pipeline.teardown() + + +def test_phase2_query_with_refusal_short_circuits(stub_llm): + """End-to-end smoke: refusal handler short-circuits when top-1 < threshold (empty index).""" + cfg = load_config(PHASE2_DIR / "phase2g_refusal.yaml") + pipeline = build_pipeline( + cfg, dataset_name="squad_v2_dev_200", + llm_override=stub_llm, judge_llm_override=stub_llm, + ) + try: + # Empty index → retrieval returns [] → handler refuses. + chunks, answer, telemetry = pipeline.query("what is x?") + assert chunks == [] + assert answer == cfg.pipeline.refusal_handler.no_answer_text + assert "refusal_check" in telemetry["timings_ms"] + finally: + pipeline.teardown() + + +def test_every_phase2_yaml_loads(): + """Every YAML under configs/eval/phase2/ must validate against EvalConfig.""" + for path in sorted(PHASE2_DIR.glob("*.yaml")): + cfg = load_config(path) + assert cfg.name == path.stem, f"{path.name}: cfg.name={cfg.name!r} != {path.stem!r}" diff --git a/tests/test_eval_pricing.py b/tests/test_eval_pricing.py new file mode 100644 index 00000000..51c0b670 --- /dev/null +++ b/tests/test_eval_pricing.py @@ -0,0 +1,51 @@ +"""Tests for src.eval.pricing.""" + +from __future__ import annotations + +import logging + +import pytest + +from src.eval.pricing import MODEL_PRICES, ModelPrice, cost_usd + + +class TestModelPriceTable: + def test_known_models_present(self): + for model_id in ("gpt-5-mini", "gpt-4.1-mini", "gpt-4.1-nano"): + assert model_id in MODEL_PRICES + assert isinstance(MODEL_PRICES[model_id], ModelPrice) + + def test_prices_positive(self): + for price in MODEL_PRICES.values(): + assert price.prompt_per_1m > 0 + assert price.completion_per_1m > 0 + + +class TestCostUsd: + def test_known_model_basic(self): + price = MODEL_PRICES["gpt-4.1-mini"] + result = cost_usd("gpt-4.1-mini", prompt_tokens=1_000_000, completion_tokens=0) + assert result == pytest.approx(price.prompt_per_1m) + + def test_combined_cost(self): + price = MODEL_PRICES["gpt-4.1-mini"] + result = cost_usd( + "gpt-4.1-mini", prompt_tokens=500_000, completion_tokens=500_000 + ) + expected = 0.5 * price.prompt_per_1m + 0.5 * price.completion_per_1m + assert result == pytest.approx(expected) + + def test_zero_tokens(self): + assert cost_usd("gpt-4.1-mini", 0, 0) == 0.0 + + def test_unknown_model_returns_zero_with_warning(self, caplog): + with caplog.at_level(logging.WARNING): + result = cost_usd("nonexistent-model", 100, 100) + assert result == 0.0 + assert any("unknown model" in rec.message.lower() for rec in caplog.records) + + def test_negative_tokens_raises(self): + with pytest.raises(ValueError): + cost_usd("gpt-4.1-mini", -1, 0) + with pytest.raises(ValueError): + cost_usd("gpt-4.1-mini", 0, -1) diff --git a/tests/test_eval_report.py b/tests/test_eval_report.py new file mode 100644 index 00000000..67719567 --- /dev/null +++ b/tests/test_eval_report.py @@ -0,0 +1,86 @@ +"""Tests for src.eval.report.""" + +from __future__ import annotations + +from datetime import datetime, timezone + +from src.eval.report import render_compare_html, render_run_html +from src.eval.schemas import ( + AggregatedMetric, + CompareResult, + EvalResult, + MetricDelta, + RunMetadata, +) + + +def _meta(run_id: str = "test") -> RunMetadata: + now = datetime.now(timezone.utc) + return RunMetadata( + run_id=run_id, config_name="baseline", + config_path="x.yaml", git_sha="abc1234", + started_at=now, finished_at=now, env_hash="h", + eval_set_versions={"squad_v2_dev_200": "v1"}, + n_questions=2, n_errors=0, + ) + + +def _result(qid: str) -> EvalResult: + return EvalResult( + question_id=qid, dataset="squad_v2_dev_200", + retrieved_chunk_ids=[], retrieved_chunks=[], + generated_answer="ans", metrics={"recall_at_5": 1.0}, + timings_ms={}, tokens={"prompt": 10, "completion": 5}, cost_usd=0.0001, + ) + + +class TestRenderRunHtml: + def test_basic_render(self): + run = { + "metadata": _meta("run-1"), + "results": [_result("q1"), _result("q2")], + "aggregated": [ + AggregatedMetric(metric_name="recall_at_5", mean=1.0, + ci_low=1.0, ci_high=1.0, n=2), + ], + "cost": {"total_usd": 0.0002, "mean_usd_per_query": 0.0001, + "total_prompt": 20, "total_completion": 10}, + } + html = render_run_html(run) + assert " SearchResult: + return SearchResult(chunk_id=chunk_id, content=content, score=score, metadata={}, doc_id="") + + +def test_rrf_fusion_asymmetric_inputs(): + """RRF on A=[a,b,c,d], B=[d,a] with rrf_k=60 yields fused order a, d, b, c.""" + from src.eval.retrievers.bm25_hybrid import reciprocal_rank_fusion + A = ["a", "b", "c", "d"] + B = ["d", "a"] + fused = reciprocal_rank_fusion([A, B], rrf_k=60) + assert fused == ["a", "d", "b", "c"] + + +def test_hybrid_retrieve_returns_top_k(): + """End-to-end: hybrid retriever combines BM25 and Chroma results into top-K.""" + import chromadb + from src.eval.retrievers.bm25_hybrid import BM25HybridRetriever + from src.vector_store import ChromaVectorStore + + client = chromadb.EphemeralClient() + coll = client.get_or_create_collection( + name="test_hybrid", metadata={"hnsw:space": "cosine"}, + ) + coll.upsert( + ids=["d1", "d2", "d3", "d4"], + documents=[ + "Cats are small carnivorous mammals often kept as pets.", + "Reciprocal rank fusion is a standard sparse-dense combination.", + "Hybrid search blends BM25 and dense retrieval signals.", + "Airplanes have fixed wings.", + ], + ) + vs = ChromaVectorStore(collection=coll) + retriever = BM25HybridRetriever( + vector_store=vs, + documents={"d1": coll.get(ids=["d1"])["documents"][0], + "d2": coll.get(ids=["d2"])["documents"][0], + "d3": coll.get(ids=["d3"])["documents"][0], + "d4": coll.get(ids=["d4"])["documents"][0]}, + bm25_top_k=3, + dense_top_k=3, + rrf_k=60, + ) + out = retriever.retrieve("hybrid sparse dense fusion", top_k=2) + assert len(out) == 2 + assert all(isinstance(r, SearchResult) for r in out) + # Top result should be one of d2 or d3 (both directly relevant). + assert out[0].chunk_id in {"d2", "d3"} diff --git a/tests/test_eval_retriever_reranker.py b/tests/test_eval_retriever_reranker.py new file mode 100644 index 00000000..135ff083 --- /dev/null +++ b/tests/test_eval_retriever_reranker.py @@ -0,0 +1,43 @@ +"""Tests for CrossEncoderReranker — re-scores candidates with a cross-encoder model.""" + +from __future__ import annotations + +import pytest + +from src.vector_store import SearchResult + + +def _sr(chunk_id: str, content: str, score: float, metadata: dict | None = None) -> SearchResult: + return SearchResult(doc_id="", chunk_id=chunk_id, content=content, + score=score, metadata=metadata or {}) + + +@pytest.fixture(scope="module") +def reranker(): + from src.eval.retrievers.reranker import CrossEncoderReranker + return CrossEncoderReranker() + + +def test_obvious_match_ranks_first(reranker): + """Given five candidates with one obviously-relevant doc, it ranks first after rerank.""" + candidates = [ + _sr("d1", "Pyramids of Giza were built around 2500 BC.", 0.5), + _sr("d2", "Cats are small carnivorous mammals.", 0.6), + _sr("d3", "What is the capital of France? Paris is the capital.", 0.4), + _sr("d4", "Airplanes have fixed wings.", 0.3), + _sr("d5", "Dogs are domesticated.", 0.2), + ] + out = reranker.rerank("What is the capital of France?", candidates, final_top_k=3) + assert len(out) == 3 + assert out[0].chunk_id == "d3" + + +def test_rerank_preserves_search_result_shape(reranker): + candidates = [ + _sr("d1", "hello", 0.5, metadata={"k": "v"}), + _sr("d2", "world", 0.4), + ] + out = reranker.rerank("greeting", candidates, final_top_k=2) + assert all(isinstance(r, SearchResult) for r in out) + found = {r.chunk_id: r for r in out} + assert found["d1"].metadata == {"k": "v"} diff --git a/tests/test_eval_runner.py b/tests/test_eval_runner.py new file mode 100644 index 00000000..16c4bc3c --- /dev/null +++ b/tests/test_eval_runner.py @@ -0,0 +1,119 @@ +"""Tests for src.eval.runner.""" + +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from src.eval.config import EvalConfig +from src.eval.runner import EvalRunner +from src.eval.schemas import EvalQuestion + + +class DummyLLM: + """Returns canned answers / canned JSON for any prompt.""" + def __init__(self, answer: str = "", judge_payload: dict | None = None): + self.answer = answer + self.judge_payload = judge_payload or { + "score": 1.0, "claims": [], "chunks": [], "factual_match": 1.0, + "is_refusal": False, "reasoning": "ok", + } + self.calls: list[str] = [] + def generate(self, prompt: str, system_prompt: str | None = None) -> str: + self.calls.append(prompt) + # Heuristic: judge prompts request JSON; answer prompts don't. + if "JSON" in (system_prompt or "") or '"score"' in prompt or '"claims"' in prompt or 'JSON' in prompt: + return json.dumps(self.judge_payload) + return self.answer + + +def _baseline_config() -> EvalConfig: + return EvalConfig.model_validate({ + "name": "test", "description": "", + "pipeline": { + "chunker": {"strategy": "recursive", "chunk_size": 256, "chunk_overlap": 32}, + "retriever": {"top_k": 3}, + "generator": {"model": "gpt-4.1-nano", "reasoning_model": None}, + }, + "eval": { + "datasets": ["squad_v2_dev_200"], + "judge_model": "gpt-4.1-nano", + "bootstrap_n": 100, "permutation_n": 100, "seed": 42, + }, + }) + + +@pytest.fixture +def tmp_eval_runs(tmp_path, monkeypatch): + runs = tmp_path / "eval_runs" + runs.mkdir() + monkeypatch.setenv("EVAL_RUNS_DIR", str(runs)) + import importlib + import src.eval.storage + importlib.reload(src.eval.storage) + yield src.eval.storage + monkeypatch.delenv("EVAL_RUNS_DIR", raising=False) + importlib.reload(src.eval.storage) + + +@pytest.fixture +def squad_5(monkeypatch, tmp_path): + """Override the SQuAD frozen path with a tiny 5-question synthetic set.""" + questions = [ + EvalQuestion( + id=f"q{i}", question=f"What is fact {i}?", + gold_answer=f"Fact {i}.", + gold_chunk_ids=[f"q{i}"], + metadata={"context": f"Fact {i} is important.", "title": "t"}, + ) + for i in range(5) + ] + path = tmp_path / "squad.jsonl" + with path.open("w") as f: + for q in questions: + f.write(q.model_dump_json() + "\n") + monkeypatch.setattr("src.eval.datasets.squad_v2.DEFAULT_OUTPUT_PATH", path) + return path + + +class TestEvalRunner: + def test_end_to_end_squad(self, tmp_eval_runs, squad_5): + cfg = _baseline_config() + runner = EvalRunner( + cfg, + llm_override=DummyLLM("Fact 0."), + judge_llm_override=DummyLLM(judge_payload={ + "score": 1.0, "claims": [], "chunks": [], "factual_match": 1.0, + "is_refusal": False, "reasoning": "ok", + }), + ) + meta = runner.run() + assert meta.n_questions == 5 + assert meta.n_errors == 0 + assert meta.config_name == "test" + + # Verify run dir contains all expected files + run_dir = tmp_eval_runs.EVAL_RUNS_DIR / meta.run_id + for f in ["metadata.json", "questions.jsonl", "metrics.json", + "cost.json", "config.yaml"]: + assert (run_dir / f).exists() + + # Reload via storage + loaded = tmp_eval_runs.load_run(meta.run_id) + assert len(loaded["results"]) == 5 + assert loaded["aggregated"], "aggregated metrics should be non-empty" + + def test_progress_callback(self, tmp_eval_runs, squad_5): + cfg = _baseline_config() + progress_calls = [] + runner = EvalRunner( + cfg, + llm_override=DummyLLM("answer"), + judge_llm_override=DummyLLM(), + on_progress=lambda done, total: progress_calls.append((done, total)), + ) + runner.run() + assert len(progress_calls) == 5 + assert progress_calls[-1] == (5, 5) diff --git a/tests/test_eval_schemas.py b/tests/test_eval_schemas.py new file mode 100644 index 00000000..1bd20e50 --- /dev/null +++ b/tests/test_eval_schemas.py @@ -0,0 +1,145 @@ +"""Tests for the eval harness Pydantic schemas.""" + +from __future__ import annotations + +from datetime import datetime, timezone + +import pytest +from pydantic import ValidationError + +from src.eval.schemas import ( + AggregatedMetric, + CompareResult, + EvalQuestion, + EvalResult, + MetricDelta, + RunMetadata, +) + + +class TestEvalQuestion: + def test_minimal_construction(self): + q = EvalQuestion(id="abc", question="Why?") + assert q.gold_answer is None + assert q.gold_chunk_ids == [] + assert q.is_unanswerable is False + assert q.metadata == {} + + def test_full_construction(self): + q = EvalQuestion( + id="abc", + question="Why?", + gold_answer="Because.", + gold_chunk_ids=["c1", "c2"], + is_unanswerable=False, + metadata={"difficulty": "hard"}, + ) + assert q.gold_answer == "Because." + assert q.metadata["difficulty"] == "hard" + + def test_frozen_blocks_mutation(self): + q = EvalQuestion(id="abc", question="Why?") + with pytest.raises(ValidationError): + q.question = "What?" # type: ignore[misc] + + +class TestEvalResult: + def test_round_trip_json(self): + r = EvalResult( + question_id="abc", + dataset="squad_v2_dev_200", + retrieved_chunk_ids=["c1"], + retrieved_chunks=["text"], + generated_answer="ans", + metrics={"recall_at_5": 1.0}, + timings_ms={"retrieve": 12.0, "generate": 100.0}, + tokens={"prompt": 100, "completion": 50}, + cost_usd=0.001, + ) + encoded = r.model_dump_json() + decoded = EvalResult.model_validate_json(encoded) + assert decoded == r + + def test_error_default_none(self): + r = EvalResult( + question_id="abc", + dataset="ml_papers_v1", + retrieved_chunk_ids=[], + retrieved_chunks=[], + generated_answer="", + metrics={}, + timings_ms={}, + tokens={"prompt": 0, "completion": 0}, + cost_usd=0.0, + ) + assert r.error is None + + +class TestAggregatedMetric: + def test_construction(self): + m = AggregatedMetric( + metric_name="recall_at_5", + mean=0.84, + ci_low=0.81, + ci_high=0.87, + n=200, + ) + assert m.dataset is None + + def test_with_dataset(self): + m = AggregatedMetric( + metric_name="faithfulness", + dataset="ml_papers_v1", + mean=0.91, + ci_low=0.85, + ci_high=0.96, + n=50, + ) + assert m.dataset == "ml_papers_v1" + + +class TestRunMetadata: + def test_construction(self): + now = datetime.now(timezone.utc) + meta = RunMetadata( + run_id="2026-04-26_143022_baseline_a3f9c1", + config_name="baseline", + config_path="configs/eval/baseline.yaml", + git_sha="a3f9c1", + started_at=now, + finished_at=now, + env_hash="deadbeef", + eval_set_versions={"squad_v2_dev_200": "abc123"}, + n_questions=200, + n_errors=0, + ) + assert meta.warnings == [] + + +class TestMetricDelta: + def test_construction(self): + d = MetricDelta( + metric_name="recall_at_5", + a_mean=0.80, + a_ci=(0.77, 0.83), + b_mean=0.85, + b_ci=(0.82, 0.88), + delta=0.05, + p_value=0.001, + significant=True, + ) + assert d.significant is True + assert d.delta == pytest.approx(0.05) + + +class TestCompareResult: + def test_construction(self): + now = datetime.now(timezone.utc) + meta_a = RunMetadata( + run_id="A", config_name="a", config_path="a.yaml", git_sha="x", + started_at=now, finished_at=now, env_hash="h", + eval_set_versions={}, n_questions=10, n_errors=0, + ) + meta_b = meta_a.model_copy(update={"run_id": "B"}) + result = CompareResult(run_a=meta_a, run_b=meta_b, deltas=[]) + assert result.per_question_diff == [] diff --git a/tests/test_eval_smoke.py b/tests/test_eval_smoke.py new file mode 100644 index 00000000..da3d5c44 --- /dev/null +++ b/tests/test_eval_smoke.py @@ -0,0 +1,52 @@ +"""End-of-sub-plan-1A smoke test: verifies the package surface is +importable and the metric pieces compose end-to-end without a runner.""" + +from __future__ import annotations + +import math + + +def test_top_level_imports(): + from src.eval import ( + AggregatedMetric, + CompareResult, + EvalQuestion, + EvalResult, + MetricDelta, + MODEL_PRICES, + RunMetadata, + bootstrap_ci, + cost_usd, + paired_permutation_test, + ) + + assert callable(bootstrap_ci) + assert callable(paired_permutation_test) + assert callable(cost_usd) + assert MODEL_PRICES + for cls in ( + AggregatedMetric, CompareResult, EvalQuestion, + EvalResult, MetricDelta, RunMetadata, + ): + assert isinstance(cls, type) + + +def test_compose_retrieval_then_aggregate(): + """Running retrieval metrics across a synthetic dev-set composes correctly.""" + from src.eval import bootstrap_ci, EvalQuestion + from src.eval.metrics.retrieval import recall_at_k + + questions = [ + EvalQuestion(id=str(i), question="Q?", gold_chunk_ids=["c1"]) + for i in range(50) + ] + retrieved_per_q = [["c1", "x"] if i % 5 != 0 else ["x", "y"] for i in range(50)] + recalls = [ + recall_at_k(q.gold_chunk_ids, ret, k=5) + for q, ret in zip(questions, retrieved_per_q) + ] + assert sum(recalls) / len(recalls) == 0.8 + + mean, low, high = bootstrap_ci(recalls, n_resamples=200, seed=42) + assert mean == 0.8 + assert low < 0.8 < high or math.isclose(low, 0.8) or math.isclose(high, 0.8) diff --git a/tests/test_eval_statistics.py b/tests/test_eval_statistics.py new file mode 100644 index 00000000..0a088606 --- /dev/null +++ b/tests/test_eval_statistics.py @@ -0,0 +1,85 @@ +"""Tests for src.eval.statistics.""" + +from __future__ import annotations + +import math + +import numpy as np +import pytest + +from src.eval.statistics import bootstrap_ci, paired_permutation_test + + +SEED = 12345 + + +class TestBootstrapCI: + def test_known_distribution_brackets_true_mean(self): + """The 95% CI on a normal sample should usually contain the true mean.""" + rng = np.random.default_rng(SEED) + true_mean = 5.0 + sample = rng.normal(loc=true_mean, scale=1.0, size=200) + + mean, low, high = bootstrap_ci(sample.tolist(), n_resamples=1000, seed=SEED) + assert low < true_mean < high + assert mean == pytest.approx(float(np.mean(sample))) + + def test_seed_reproducibility(self): + values = [1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0, 10.0] + a = bootstrap_ci(values, n_resamples=500, seed=SEED) + b = bootstrap_ci(values, n_resamples=500, seed=SEED) + assert a == b + + def test_drops_nan_values(self): + values = [1.0, 2.0, float("nan"), 3.0, float("nan"), 4.0] + mean, _, _ = bootstrap_ci(values, n_resamples=200, seed=SEED) + assert mean == pytest.approx(2.5) + + def test_all_nan_raises(self): + with pytest.raises(ValueError): + bootstrap_ci([float("nan"), float("nan")], n_resamples=100, seed=SEED) + + def test_single_value_yields_zero_width_ci(self): + mean, low, high = bootstrap_ci([7.0], n_resamples=100, seed=SEED) + assert mean == 7.0 + assert low == 7.0 + assert high == 7.0 + + +class TestPairedPermutationTest: + def test_identical_distributions_high_p(self): + rng = np.random.default_rng(SEED) + sample = rng.normal(0, 1, size=100).tolist() + delta, p = paired_permutation_test( + sample, sample, n_resamples=2000, seed=SEED + ) + assert delta == pytest.approx(0.0) + assert p > 0.5 + + def test_clear_effect_low_p(self): + rng = np.random.default_rng(SEED) + a = rng.normal(0.0, 1.0, size=100) + b = a + 1.0 + delta, p = paired_permutation_test( + a.tolist(), b.tolist(), n_resamples=2000, seed=SEED + ) + assert delta == pytest.approx(1.0, abs=0.01) + assert p < 0.01 + + def test_seed_reproducibility(self): + a = [1.0, 2.0, 3.0, 4.0, 5.0] + b = [1.5, 2.1, 2.9, 4.3, 5.0] + d1, p1 = paired_permutation_test(a, b, n_resamples=500, seed=SEED) + d2, p2 = paired_permutation_test(a, b, n_resamples=500, seed=SEED) + assert d1 == d2 + assert p1 == p2 + + def test_unequal_lengths_raises(self): + with pytest.raises(ValueError): + paired_permutation_test([1.0, 2.0], [1.0, 2.0, 3.0], seed=SEED) + + def test_drops_paired_nans(self): + a = [1.0, float("nan"), 3.0, 4.0] + b = [1.5, 2.5, float("nan"), 4.5] + delta, _ = paired_permutation_test(a, b, n_resamples=200, seed=SEED) + assert delta == pytest.approx(0.5) diff --git a/tests/test_eval_storage.py b/tests/test_eval_storage.py new file mode 100644 index 00000000..2ff1dd97 --- /dev/null +++ b/tests/test_eval_storage.py @@ -0,0 +1,148 @@ +"""Tests for src.eval.storage.""" + +from __future__ import annotations + +import json +import os +from datetime import datetime, timezone +from pathlib import Path + +import pytest + +from src.eval.schemas import ( + AggregatedMetric, + EvalResult, + RunMetadata, +) + + +@pytest.fixture +def tmp_eval_runs(tmp_path: Path, monkeypatch): + """Set EVAL_RUNS_DIR to a temp dir and re-import storage to pick it up.""" + runs_dir = tmp_path / "eval_runs" + runs_dir.mkdir() + monkeypatch.setenv("EVAL_RUNS_DIR", str(runs_dir)) + # Force re-evaluation of EVAL_RUNS_DIR by re-importing. + import importlib + import src.eval.storage + importlib.reload(src.eval.storage) + yield src.eval.storage + # cleanup: reload back to default for other tests + monkeypatch.delenv("EVAL_RUNS_DIR", raising=False) + importlib.reload(src.eval.storage) + + +def _make_metadata(run_id: str = "test-run") -> RunMetadata: + now = datetime.now(timezone.utc) + return RunMetadata( + run_id=run_id, + config_name="baseline", + config_path="configs/eval/baseline.yaml", + git_sha="abc1234", + started_at=now, + finished_at=now, + env_hash="deadbeef", + eval_set_versions={"squad_v2_dev_200": "v1"}, + n_questions=2, + n_errors=0, + ) + + +def _make_result(qid: str = "q1") -> EvalResult: + return EvalResult( + question_id=qid, dataset="squad_v2_dev_200", + retrieved_chunk_ids=["c1"], retrieved_chunks=["text"], + generated_answer="ans", metrics={"recall_at_5": 1.0}, + timings_ms={"retrieve": 12.0, "generate": 100.0}, + tokens={"prompt": 50, "completion": 25}, cost_usd=0.001, + ) + + +class TestComputeRunId: + def test_format(self, tmp_eval_runs): + ts = datetime(2026, 4, 26, 14, 30, 22, tzinfo=timezone.utc) + rid = tmp_eval_runs.compute_run_id("baseline", ts, "a3f9c1abcdef") + assert rid == "2026-04-26_143022_baseline_a3f9c1a" + + def test_deterministic(self, tmp_eval_runs): + ts = datetime(2026, 4, 26, 14, 30, 22, tzinfo=timezone.utc) + rid1 = tmp_eval_runs.compute_run_id("x", ts, "abc1234567") + rid2 = tmp_eval_runs.compute_run_id("x", ts, "abc1234567") + assert rid1 == rid2 + + +class TestSaveAndLoadRun: + def test_round_trip(self, tmp_eval_runs): + meta = _make_metadata("test-run-1") + results = [_make_result("q1"), _make_result("q2")] + aggregated = [ + AggregatedMetric( + metric_name="recall_at_5", mean=1.0, + ci_low=1.0, ci_high=1.0, n=2, + ) + ] + cost = {"total_usd": 0.002, "mean_usd_per_query": 0.001} + run_dir = tmp_eval_runs.EVAL_RUNS_DIR / meta.run_id + tmp_eval_runs.save_run( + run_dir, meta, results, aggregated, cost, "name: test\n" + ) + + loaded = tmp_eval_runs.load_run(meta.run_id) + assert loaded["metadata"] == meta + assert loaded["results"] == results + assert loaded["aggregated"] == aggregated + assert loaded["cost"] == cost + + def test_files_created(self, tmp_eval_runs): + meta = _make_metadata("test-run-2") + run_dir = tmp_eval_runs.EVAL_RUNS_DIR / meta.run_id + tmp_eval_runs.save_run(run_dir, meta, [], [], {}, "name: test\n") + for f in ["metadata.json", "questions.jsonl", "metrics.json", "cost.json", "config.yaml"]: + assert (run_dir / f).exists(), f"Missing {f}" + + def test_load_missing_raises(self, tmp_eval_runs): + with pytest.raises(FileNotFoundError): + tmp_eval_runs.load_run("does-not-exist") + + +class TestListRuns: + def test_lists_completed_runs_descending(self, tmp_eval_runs): + # Create two runs with distinct timestamps. + meta_old = _make_metadata("old-run") + meta_old = meta_old.model_copy(update={ + "started_at": datetime(2026, 1, 1, tzinfo=timezone.utc), + "finished_at": datetime(2026, 1, 1, tzinfo=timezone.utc), + }) + meta_new = _make_metadata("new-run") + meta_new = meta_new.model_copy(update={ + "started_at": datetime(2026, 4, 1, tzinfo=timezone.utc), + "finished_at": datetime(2026, 4, 1, tzinfo=timezone.utc), + }) + for m in (meta_old, meta_new): + run_dir = tmp_eval_runs.EVAL_RUNS_DIR / m.run_id + tmp_eval_runs.save_run(run_dir, m, [], [], {}, "x: y\n") + runs = tmp_eval_runs.list_runs() + assert [r.run_id for r in runs] == ["new-run", "old-run"] + + def test_ignores_dirs_without_metadata(self, tmp_eval_runs): + (tmp_eval_runs.EVAL_RUNS_DIR / "incomplete-run").mkdir() + assert tmp_eval_runs.list_runs() == [] + + def test_empty_dir_returns_empty(self, tmp_eval_runs): + assert tmp_eval_runs.list_runs() == [] + + +class TestDeleteRun: + def test_removes_run_dir(self, tmp_eval_runs): + meta = _make_metadata("doomed-run") + run_dir = tmp_eval_runs.EVAL_RUNS_DIR / meta.run_id + tmp_eval_runs.save_run(run_dir, meta, [], [], {}, "x: y\n") + assert run_dir.exists() + tmp_eval_runs.delete_run(meta.run_id) + assert not run_dir.exists() + + def test_refuses_path_traversal(self, tmp_eval_runs): + with pytest.raises(ValueError): + tmp_eval_runs.delete_run("../etc") + with pytest.raises(ValueError): + tmp_eval_runs.delete_run("a/b") diff --git a/tests/test_eval_transform_refusal.py b/tests/test_eval_transform_refusal.py new file mode 100644 index 00000000..9cf1d6f6 --- /dev/null +++ b/tests/test_eval_transform_refusal.py @@ -0,0 +1,48 @@ +"""Tests for RefusalHandler — pure-logic similarity gate.""" + +from __future__ import annotations + +from src.vector_store import SearchResult + + +def _sr(score: float, chunk_id: str = "d1") -> SearchResult: + return SearchResult(doc_id="", chunk_id=chunk_id, content="x", + score=score, metadata={}) + + +def test_refuses_when_top1_below_threshold(): + from src.eval.transforms import RefusalHandler + h = RefusalHandler(enabled=True, similarity_threshold=0.35, + no_answer_text="I don't know.") + assert h.should_refuse([_sr(0.20), _sr(0.10)]) is True + + +def test_does_not_refuse_when_top1_above_threshold(): + from src.eval.transforms import RefusalHandler + h = RefusalHandler(enabled=True, similarity_threshold=0.35, + no_answer_text="I don't know.") + assert h.should_refuse([_sr(0.50), _sr(0.10)]) is False + + +def test_refuses_on_empty_candidates(): + from src.eval.transforms import RefusalHandler + h = RefusalHandler(enabled=True, similarity_threshold=0.35, + no_answer_text="I don't know.") + assert h.should_refuse([]) is True + + +def test_disabled_handler_never_refuses(): + from src.eval.transforms import RefusalHandler + h = RefusalHandler(enabled=False, similarity_threshold=0.35, + no_answer_text="I don't know.") + assert h.should_refuse([_sr(0.0)]) is False + assert h.should_refuse([]) is False + + +def test_refuse_response_returns_text_and_no_chunks(): + from src.eval.transforms import RefusalHandler + h = RefusalHandler(enabled=True, similarity_threshold=0.35, + no_answer_text="I cannot answer.") + chunks, answer = h.refuse_response() + assert chunks == [] + assert answer == "I cannot answer." diff --git a/tests/test_eval_transform_rewriter.py b/tests/test_eval_transform_rewriter.py new file mode 100644 index 00000000..48dc2707 --- /dev/null +++ b/tests/test_eval_transform_rewriter.py @@ -0,0 +1,62 @@ +"""Tests for QueryRewriter — LLM query expansion with cost capture.""" + +from __future__ import annotations + + +class _StubLLM: + """Records calls and returns canned responses + token counts.""" + + def __init__(self, response: str, prompt_tokens: int = 50, completion_tokens: int = 30): + self._response = response + self._prompt_tokens = prompt_tokens + self._completion_tokens = completion_tokens + self.calls: list[tuple[str, str | None]] = [] + + def generate_with_usage(self, prompt: str, system_prompt: str | None = None + ) -> tuple[str, int, int]: + self.calls.append((prompt, system_prompt)) + return self._response, self._prompt_tokens, self._completion_tokens + + +def test_no_model_passthrough(): + """When model is None, expand returns [query] unchanged with zero cost.""" + from src.eval.transforms import QueryRewriter + rw = QueryRewriter(model=None, max_expansions=3, llm=None) + queries, cost, p_t, c_t = rw.expand("What is RAG?") + assert queries == ["What is RAG?"] + assert cost == 0.0 + assert p_t == 0 + assert c_t == 0 + + +def test_expansion_returns_dedup_list_and_cost(): + """With a real model name and stub LLM, expand returns deduped expansions + cost.""" + from src.eval.transforms import QueryRewriter + stub = _StubLLM( + response='["What does RAG stand for?", "Define retrieval augmented generation", ' + '"What is RAG?"]', + prompt_tokens=80, completion_tokens=40, + ) + rw = QueryRewriter(model="gpt-4.1-nano", max_expansions=3, llm=stub) + queries, cost, p_t, c_t = rw.expand("What is RAG?") + # Original query is always first; duplicate dropped; max_expansions=3 cap respected. + assert queries[0] == "What is RAG?" + assert "What does RAG stand for?" in queries + assert "Define retrieval augmented generation" in queries + assert len(queries) == len(set(queries)) # no duplicates + assert len(queries) <= 4 # original + at most max_expansions + # Cost was computed from the stub's token counts at gpt-4.1-nano price. + assert cost > 0.0 + assert p_t == 80 + assert c_t == 40 + + +def test_malformed_llm_response_falls_back_to_passthrough(): + """If the LLM returns non-JSON, expand returns [query] and logs a warning.""" + from src.eval.transforms import QueryRewriter + stub = _StubLLM(response="not json at all", prompt_tokens=50, completion_tokens=10) + rw = QueryRewriter(model="gpt-4.1-nano", max_expansions=3, llm=stub) + queries, cost, _, _ = rw.expand("What is RAG?") + assert queries == ["What is RAG?"] + # Cost is still charged because the call did happen. + assert cost > 0.0 diff --git a/tests/test_observability.py b/tests/test_observability.py new file mode 100644 index 00000000..8e93f929 --- /dev/null +++ b/tests/test_observability.py @@ -0,0 +1,99 @@ +"""Tests for src.observability.""" + +from __future__ import annotations + +import pytest +from opentelemetry import trace +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import SimpleSpanProcessor +from opentelemetry.sdk.trace.export.in_memory_span_exporter import ( + InMemorySpanExporter, +) + +from src.observability import ( + TRACER_NAME, + get_tracer, + init_observability, + traced_stage, +) + + +@pytest.fixture +def in_memory_exporter(): + """Install an InMemorySpanExporter on a fresh TracerProvider for the test.""" + exporter = InMemorySpanExporter() + provider = TracerProvider() + provider.add_span_processor(SimpleSpanProcessor(exporter)) + # Force-install our test provider, overriding any prior init. + trace._TRACER_PROVIDER = provider # type: ignore[attr-defined] + yield exporter + exporter.clear() + + +class TestInitObservability: + def test_idempotent(self): + """Calling init twice doesn't crash and doesn't double-install.""" + init_observability() + init_observability() + # No assertion — just doesn't raise. + + def test_unreachable_endpoint_does_not_raise(self): + """A bad endpoint is logged but doesn't crash.""" + # Reset the idempotency flag so this call actually attempts init. + import src.observability as obs + obs._INITIALIZED = False # type: ignore[attr-defined] + init_observability(otlp_endpoint="http://127.0.0.1:1/v1/traces") + # Subsequent traced spans must still work (as no-ops or local). + @traced_stage("test.bad-endpoint") + def f(): + return "x", {"k": 1} + assert f() == "x" + + +class TestGetTracer: + def test_returns_tracer(self): + t = get_tracer() + assert t is not None + + +class TestTracedStage: + def test_records_span_with_attrs(self, in_memory_exporter): + @traced_stage("rag.retrieve") + def retrieve(query: str): + return ["chunk1", "chunk2"], {"top_k": 5, "chunk_count": 2} + + result = retrieve("test query") + assert result == ["chunk1", "chunk2"] + + spans = in_memory_exporter.get_finished_spans() + assert len(spans) == 1 + span = spans[0] + assert span.name == "rag.retrieve" + assert span.attributes["top_k"] == 5 + assert span.attributes["chunk_count"] == 2 + + def test_propagates_exception_and_records_error(self, in_memory_exporter): + @traced_stage("rag.fail") + def boom(): + raise RuntimeError("kaboom") + + with pytest.raises(RuntimeError, match="kaboom"): + boom() + + spans = in_memory_exporter.get_finished_spans() + assert len(spans) == 1 + # Span recorded ERROR status + from opentelemetry.trace import StatusCode + assert spans[0].status.status_code == StatusCode.ERROR + + def test_attribute_coercion(self, in_memory_exporter): + """Non-primitive attribute values are coerced to str.""" + @traced_stage("rag.coerce") + def f(): + return None, {"a_dict": {"x": 1}, "a_list_of_str": ["a", "b"]} + f() + spans = in_memory_exporter.get_finished_spans() + # a_dict gets coerced to str representation; a_list_of_str passes through. + attrs = spans[0].attributes + assert "a_dict" in attrs + assert tuple(attrs["a_list_of_str"]) == ("a", "b")