Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
59 commits
Select commit Hold shift + click to select a range
ce7649c
chore: set objective to admin UI performance investigation and fix (P…
PaoloC68 May 14, 2026
8c01699
chore(perf): scaffold perf-tests tier, runner, seeding, timing middle…
PaoloC68 May 14, 2026
b77c654
feat(perf): Playwright harness, contract snapshots, Server-Timing mid…
PaoloC68 May 14, 2026
0158b25
feat(perf): throttled-network, transcript-open, SSE memory scenarios …
PaoloC68 May 15, 2026
cea905f
fix(perf): move importlib.metadata to top-level imports in perf_repor…
PaoloC68 May 15, 2026
5105cec
chore(perf): capture SQLite baseline evidence and add changelog fragm…
PaoloC68 May 15, 2026
2f5c20d
chore(ui): add Alpine AJAX and intersect vendor libraries
PaoloC68 May 15, 2026
80124aa
feat(ui): add Jinja2 fragment templates for sessions and turns
PaoloC68 May 15, 2026
83ef91c
feat(perf): add cursor pagination module with unit tests
PaoloC68 May 15, 2026
42788b0
feat(history): add cursor-paginated session and turn query functions …
PaoloC68 May 15, 2026
9041a74
feat(ui): add HTMX fragment endpoints for sessions and turns pagination
PaoloC68 May 15, 2026
37c88e3
feat(ui): rewrite history list page with infinite scroll and debounce…
PaoloC68 May 15, 2026
54f2884
feat(ui): windowed turn rendering with patch-only SSE updates and cap…
PaoloC68 May 15, 2026
c42bccb
test(e2e): add SSE activity stream regression test
PaoloC68 May 15, 2026
f67afbb
feat(perf): add page-load Playwright performance tests
PaoloC68 May 15, 2026
342635d
docs: add concurrent migration context notes
PaoloC68 May 15, 2026
a0faa9a
chore(perf): capture P28 after-run evidence and after-report
PaoloC68 May 15, 2026
11e9be9
chore: add changelog fragment for perf optimizations
PaoloC68 May 15, 2026
11d2dee
feat(perf): add serialize and render Server-Timing phases
PaoloC68 May 15, 2026
d295d04
fix(ci): add CLAUDE.md symlink for perf_tests/AGENTS.md
PaoloC68 May 17, 2026
41a065d
refactor(cursor): move cursor helpers from perf/ to utils/
PaoloC68 May 17, 2026
c9f7c3f
fix(perf): add time_phase(db) to fragment services; gate postgres bac…
PaoloC68 May 17, 2026
8aee8cb
fix(ui): drop x-intersect.once so infinite scroll works beyond page 2
PaoloC68 May 17, 2026
357f3ee
fix(history): correct param ordering when q+cursor combined in _fetch…
PaoloC68 May 17, 2026
51c5d0b
feat(cursor): wire HMAC key to Settings (CURSOR_HMAC_KEY env var)
PaoloC68 May 17, 2026
093bf1b
fix(ui): restore conversationViewer component with lazy-load initial …
PaoloC68 May 17, 2026
fadfa1d
fix(ui): restore session metadata in history fragment; fix quick filt…
PaoloC68 May 17, 2026
6563e2d
fix(ui): make turnsCursor reactive in Alpine conversationViewer
PaoloC68 May 18, 2026
043ab86
fix(ui): update search placeholder to reflect session-ID-only filter
PaoloC68 May 18, 2026
9a2c508
fix(history): escape LIKE wildcards in q param; cap length at 128
PaoloC68 May 18, 2026
9af1fe2
fix(perf): remove double model_dump(); auto-provision CURSOR_HMAC_KEY
PaoloC68 May 18, 2026
7c7124a
fix(ui): cap rawEvents per call_id at 50 (FIFO)
PaoloC68 May 18, 2026
a6dbbe3
chore: gitignore .sisyphus/ agent working state
PaoloC68 May 18, 2026
17cfee1
fix(security): replace onclick with data-href + JS delegation in sess…
PaoloC68 May 18, 2026
72ff75d
fix(ui): validate filter param with Literal type; enforce q max_lengt…
PaoloC68 May 18, 2026
0533798
fix(ui): loadInitial() fetches JSON API and renders structured turns
PaoloC68 May 18, 2026
4308c20
refactor(history): rename filter -> quick_filter to avoid shadowing b…
PaoloC68 May 18, 2026
0594c8d
fix(security): warn at startup when CURSOR_HMAC_KEY is the dev default
PaoloC68 May 18, 2026
0f879e7
fix(ui): await loadInitial() before connectSSE to eliminate race
PaoloC68 May 18, 2026
dcc9a61
docs(changelog): correct memory cap wording; document filter=claude s…
PaoloC68 May 18, 2026
2dc4d2a
refactor(history): promote fragment helpers to public; clean up curso…
PaoloC68 May 18, 2026
779d70c
test: tighten unauthenticated assertions to 303; add cursor tiebreake…
PaoloC68 May 18, 2026
112374f
docs(changelog): document q leading-wildcard scan limitation
PaoloC68 May 18, 2026
01e5e5b
fix(ui): handle session-expiry 303 in fragment/API fetches; drop dead…
PaoloC68 May 18, 2026
3e8d8ef
perf(history): push cursor upper-bound into CTE to prune aggregation set
PaoloC68 May 18, 2026
8b8750d
fix(history): revert CTE cursor filter that caused duplicate sessions…
PaoloC68 May 18, 2026
039851c
docs: correct PR description and changelog — conversation lazy loadin…
PaoloC68 May 18, 2026
93e8b67
fix(config): persist auto-provisioned CURSOR_HMAC_KEY to ~/.luthien/c…
PaoloC68 May 18, 2026
b431a51
fix(history): normalize last_ts format in SQLite 30days filter; add u…
PaoloC68 May 18, 2026
cc04f58
refactor(ui): hoist MAX_RAW_EVENTS_PER_CALL to module scope
PaoloC68 May 18, 2026
c614a15
fix: address PR #752 review concerns
PaoloC68 May 19, 2026
f4d66c8
fix: regenerate .env.example via generator for CURSOR_HMAC_KEY note
PaoloC68 May 19, 2026
f8845b5
fix: address second round of PR #752 review concerns
PaoloC68 May 19, 2026
178746e
fix: address third round of PR #752 review concerns
PaoloC68 May 19, 2026
6e8a1b4
fix: address fourth round of PR #752 review concerns
PaoloC68 May 19, 2026
d35f378
fix: add missing docstring to encode_cursor (D103)
PaoloC68 May 19, 2026
1a4fae8
fix: address fifth round of PR #752 review concerns
PaoloC68 May 19, 2026
4604aaf
fix: address sixth round of PR #752 review concerns
PaoloC68 May 19, 2026
1023d7b
test: add missing regression tests for PR #752 review concerns
PaoloC68 May 19, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions .env.example
Original file line number Diff line number Diff line change
Expand Up @@ -103,6 +103,13 @@
# (sensitive)
# CREDENTIAL_ENCRYPTION_KEY=

# HMAC key for signing pagination cursors. Auto-provisioned to ~/.luthien/cursor_hmac.key on first run.
# In multi-replica deployments, set this explicitly so cursors validate across replicas.
# Each replica that auto-generates its own key will reject cursors issued by other replicas.
# (sensitive)
# (default derived from runtime at startup)
# CURSOR_HMAC_KEY=


# === OBSERVABILITY ===============================================

Expand Down
3 changes: 3 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -86,6 +86,9 @@ tests/**/failure_registry/*.json
# Serena AI assistant local state
.serena/

# Sisyphus agent working state (plans, evidence, run artifacts)
.sisyphus/

# Local scratch files
kanban.jpg
.e2e-logs/
Expand Down
13 changes: 13 additions & 0 deletions changelog.d/perf-baseline.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,13 @@
---
category: Chores & Docs
pr: 752
---

**Admin UI performance baseline**: Establishes perf infrastructure and captures SQLite baseline for history/conversation pages.
- Perf test scaffolding: isolated DB (`~/.luthien/perf.db`), seeding fixtures (sami-like, tier-100/1000/10000), and Playwright harness
- `scripts/perf_explain.py` — captures EXPLAIN QUERY PLAN for top slow queries
- `scripts/perf_report.py` — generates Markdown baseline report from seeded DB + query plans
- `scripts/run_perf.sh` — orchestrates seed + test + SLO assertion workflow
- Middleware timing (`Server-Timing` header) and payload-size contract tests
- Query plan evidence: 2× TEMP B-TREE on `session_list`, full SCAN on `recent_calls`
- Postgres baseline skipped (not available locally); run `./scripts/run_perf.sh --backend postgres` to capture
14 changes: 14 additions & 0 deletions changelog.d/perf-fix.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,14 @@
---
category: Features
pr: 752
---

**Admin UI performance optimizations**: Cursor pagination and memory caps for the history page and conversation viewer.
- Cursor-paginated infinite scroll on `/history` (20 sessions per page instead of all)
- Conversation viewer loads full session JSON via `/api/history/sessions/{id}` and renders structured turns; paginated lazy loading of turns is deferred to a follow-up PR
- Raw events capped at 50 per call_id (FIFO) to bound per-turn memory growth
- Debounced server-side filter on `/history` to reduce query load
- New fragment endpoints: `/ui/fragments/sessions`, `/ui/fragments/sessions/{id}/turns`
- **Known limitation**: `filter=claude` uses a full-table payload scan (`payload LIKE '%claude-code%'`) with no index. It is correct for small deployments but will be slow on large Postgres instances. A structured `client_type` column or trigram index is the long-term fix.
- **Known limitation**: session-ID search (`q=`) uses a leading-wildcard `LIKE '%q%'` which cannot use a btree index. Intended for small deployments; a trigram index or prefix-only match is the long-term fix.
- **Known limitation**: `CURSOR_HMAC_KEY` is auto-provisioned per-instance to `~/.luthien/cursor_hmac.key`. In multi-replica deployments without sticky sessions, each replica generates its own key — cursors issued by one replica will be rejected (400) by another. Set `CURSOR_HMAC_KEY` explicitly in the environment when running behind a load balancer.
105 changes: 105 additions & 0 deletions dev/context/migration_concurrent.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,105 @@
# Migration Runner: CONCURRENTLY Support Audit

_Date: 2026-05-15 | Branch: perf-baseline_

## Background

`CREATE INDEX CONCURRENTLY` is a Postgres feature that builds an index without holding a lock on the table, allowing reads and writes during the build. The constraint: it **cannot run inside a transaction block**. This audit investigates whether the current migration runner can safely execute such a statement.

---

## Current behavior

### PostgreSQL runner (`docker/run-migrations.sh`)

- Applied by the `migrations` Docker service at startup; controlled by `docker compose up migrations`.
- Sequentially applies all `*.sql` files in `migrations/postgres/` in alphabetical order.
- **No `BEGIN`/`COMMIT` transaction wrapping** is added around migration files. The runner calls `psql -f "$migration"` directly:
```sh
psql -h "$PGHOST" -U "$PGUSER" -d "$PGDATABASE" -f "$migration"
```
- `psql` defaults to autocommit mode — each statement in the file runs in its own implicit transaction unless the file itself contains explicit `BEGIN`/`COMMIT` blocks.
- The `_migrations` tracking row (`INSERT INTO _migrations`) is inserted in a **separate, subsequent `psql` invocation**, not inside the same transaction as the migration file. This means the tracking and the DDL are non-atomic: a crash between the two steps leaves schema changes applied but untracked.
- Migration state is tracked in the `_migrations` table (columns: `filename TEXT PK`, `applied_at TIMESTAMP`, `content_hash TEXT`).
- Applied-migration detection uses `SELECT COUNT(*) FROM _migrations WHERE filename = '$filename'`, checked per file before applying.
- Hash validation compares stored MD5 against local file MD5 and aborts on mismatch.

### SQLite runner (`src/luthien_proxy/utils/migration_check.py :: _apply_sqlite_migrations`)

- Runs in-process at gateway startup for dockerless/SQLite deployments.
- Uses `executescript()` to apply each `.sql` file — this method issues an implicit `COMMIT` before execution and runs all statements in the file sequentially.
- `CREATE INDEX CONCURRENTLY` is not a SQLite concept; `AGENTS.md` explicitly lists it under "What to OMIT in SQLite migrations" and directs authors to use `CREATE INDEX IF NOT EXISTS` instead.
- SQLite tracking is also done in the `_migrations` table but is written inside the same connection context (not atomic with the `executescript`, however — a mid-script crash leaves partial schema with no tracking record).

---

## Verdict

**PARTIAL**

`CREATE INDEX CONCURRENTLY` can be placed in a Postgres migration file today and will execute successfully — because the runner uses `psql -f` in autocommit mode with **no outer transaction wrapping**. The statement will not hit the "cannot run inside a transaction block" error.

However:

1. **Non-atomic tracking** — the `INSERT INTO _migrations` tracking row is a separate psql call. If it fails, the index exists on disk but the migration is untracked. A re-run will try to apply the file again; `CREATE INDEX CONCURRENTLY IF NOT EXISTS` protects against failure in that case.
2. **SQLite incompatibility** — a companion SQLite migration must use plain `CREATE INDEX IF NOT EXISTS` (standard `AGENTS.md` practice; no code change needed).
3. **No explicit guidance in runner or AGENTS.md** about CONCURRENTLY for Postgres beyond the SQLite omit rule — the assumption has been "it just works because psql is autocommit."

---

## Findings

1. **No BEGIN/COMMIT wrapping in Postgres runner.** `run-migrations.sh` calls `psql -f "$migration"` with zero explicit transaction control around migration files. psql autocommit applies.

2. **`BEGIN` in existing migrations is always PL/pgSQL, not transaction control.** Searching all postgres migration files reveals `BEGIN` only inside `$$ LANGUAGE plpgsql` function/trigger bodies (e.g., `014_add_session_search_fts.sql`, `000_init_databases.sql`). No migration wraps its DDL in a `BEGIN...COMMIT` block.

3. **Tracking INSERT is not atomic with migration application.** Lines 153–156 of `run-migrations.sh` run the migration file, then insert into `_migrations` in a second psql call. A process kill between those two calls yields applied-but-untracked state. `CREATE INDEX CONCURRENTLY IF NOT EXISTS` + idempotent DDL is the correct mitigation.

4. **SQLite runner uses `executescript()`, not raw `execute()`.** This means the entire SQL file is submitted to SQLite's native multi-statement parser in one call. It handles trigger `BEGIN...END` correctly but does not guarantee atomicity across the file; a mid-script error leaves partial schema with no `_migrations` entry.

5. **AGENTS.md already documents the SQLite handling rule.** "What to OMIT in SQLite migrations" includes `CREATE INDEX CONCURRENTLY` — use plain `CREATE INDEX IF NOT EXISTS`. This is the only dual-dialect consideration; Postgres needs no special handling beyond `IF NOT EXISTS`.

6. **Migration 006 establishes the index-in-migration pattern.** `006_add_session_id.sql` creates two partial indexes (`WHERE session_id IS NOT NULL`) with `CREATE INDEX IF NOT EXISTS`. This is the precedent: use `IF NOT EXISTS` for idempotence, and the runner handles it without transaction complications.

7. **`014_add_session_search_fts.sql` creates multiple indexes in one file.** A GIN index, a btree partial index, and an expression index are all created in a single migration file, all with `IF NOT EXISTS`. This confirms that non-trivial index migrations work fine under the current runner.

---

## Risk assessment

### If a future PR needs `CREATE INDEX CONCURRENTLY` (Postgres)

**Risk: LOW** — the runner already runs in autocommit mode. No runner changes are required.

**Smallest safe path:**

1. Postgres migration file: use `CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_name ON table(col)`.
- `IF NOT EXISTS` handles the non-atomic tracking race condition: if the runner crashes after DDL but before tracking, the re-run skips the existing index without error.
- Note: `CREATE INDEX CONCURRENTLY IF NOT EXISTS` requires Postgres 9.5+. Luthien targets modern Postgres; this is not a concern.
2. SQLite migration file: use plain `CREATE INDEX IF NOT EXISTS idx_name ON table(col)` (no CONCURRENTLY keyword).
3. No changes to `run-migrations.sh` or `migration_check.py` are needed.

**Residual risk:** `CREATE INDEX CONCURRENTLY` holds a share-update-exclusive lock, not a full table lock, but it does require two table scans. On a large `conversation_events` table it may run for minutes. The Docker `migrations` container has no configurable `lock_timeout`; a very large production table could cause the migration container to hang. Mitigation: document the expected index build time in the migration file comment, or run it manually outside the automated runner for very large tables.

**Out-of-scope risk (do not fix here):** The non-atomic tracking gap exists for ALL migrations, not just CONCURRENTLY ones. A proper fix would wrap both the DDL and the `INSERT INTO _migrations` in a single transaction — but that would break `CREATE INDEX CONCURRENTLY`. The correct long-term approach is to move tracking into the same psql session with `\set ON_ERROR_STOP on` and careful sequencing, but that is a separate refactor not required for this PR series.

---

## Experimental Validation

Confirmed: The P8 audit findings are correct.

**SQLite experiment** (run 2026-05-15):
- `CREATE INDEX IF NOT EXISTS` via `executescript()`: works correctly
- `CREATE INDEX CONCURRENTLY`: fails with `sqlite3.OperationalError: near "IF": syntax error`
- Conclusion: SQLite migrations must always use plain `CREATE INDEX IF NOT EXISTS`

**Postgres validation** (theoretical, based on runner analysis):
- The Postgres runner (`docker/run-migrations.sh`) uses `psql -f` with no transaction wrapping
- `CREATE INDEX CONCURRENTLY` requires running outside a transaction block
- Since the runner does NOT wrap in BEGIN/COMMIT, CONCURRENTLY should work
- Practical test deferred (no Postgres available in local dev); theoretical analysis confirmed

**Verdict**: PARTIAL support confirmed experimentally:
- SQLite: CONCURRENTLY not supported (syntax error) — use plain `CREATE INDEX IF NOT EXISTS`
- Postgres: CONCURRENTLY supported (no transaction wrapping in runner) — safe to use
8 changes: 6 additions & 2 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -82,7 +82,7 @@ docstring-code-format = true
convention = "google"

[tool.pytest.ini_options]
addopts = "-q -ra -m 'not e2e and not integration and not mock_e2e and not sqlite_e2e' --import-mode=importlib --cov=src/luthien_proxy --cov-report=term-missing --timeout=3 --timeout-method=signal"
addopts = "-q -ra -m 'not e2e and not integration and not mock_e2e and not sqlite_e2e and not perf' --import-mode=importlib --cov=src/luthien_proxy --cov-report=term-missing --timeout=3 --timeout-method=signal"
testpaths = ["tests"]
asyncio_mode = "auto"
filterwarnings = [
Expand All @@ -95,6 +95,8 @@ markers = [
"integration: marks integration tests that require external services (OpenAI, Anthropic APIs)",
"mock_e2e: marks e2e tests that use the mock Anthropic server (no real API calls)",
"sqlite_e2e: marks e2e tests running the gateway in-process with SQLite (no Docker)",
"perf: marks performance tests that measure gateway latency and throughput (opt-in via ./scripts/run_perf.sh)",
"contract: marks API contract snapshot tests that validate response shapes",
"llm01: OWASP LLM01 - Prompt Injection scenarios",
"llm02: OWASP LLM02 - Insecure Output Handling scenarios (reserved, no tests yet)",
"llm04: OWASP LLM04 - Model Denial of Service scenarios (reserved, no tests yet)",
Expand Down Expand Up @@ -122,13 +124,15 @@ reportMissingImports = "warning"

[dependency-groups]
dev = [
"playwright==1.50.0",
"pre-commit>=4.3.0",
"pytest>=8.4.1",
"pytest-asyncio>=1.1.0",
"pytest-cov>=6.2.1",
"pytest-playwright>=0.5.0",
"pytest-timeout>=2.4.0",
"ruff>=0.12.10",
"pyright>=1.1.406,<1.2",
"pytest-timeout>=2.4.0",
"radon>=6.0.1",
"vulture>=2.14",
"asgi-lifespan>=2.1.0",
Expand Down
Loading
Loading