diff --git a/.env.example b/.env.example
index 8eb4fce..ccfcc3c 100644
--- a/.env.example
+++ b/.env.example
@@ -12,16 +12,23 @@
# LLM / Pipeline Configuration
# ------------------------------------------------------------------------------
-# LLM backend: openai-compat (aliases: ferry, http) | litellm
+# LLM backend: openai-compat (aliases: ferry, http) | litellm | claude-cli | kiro-cli
+# Read case-insensitively; any other value is a configuration error (exit 2).
PRXREF_LLM_BACKEND=openai-compat
-# Base URL for any OpenAI-compatible /chat/completions endpoint. REQUIRED.
+# Base URL for any OpenAI-compatible /chat/completions endpoint. REQUIRED for
+# openai-compat/ferry/http. Not used by litellm (it resolves each model's own
+# provider endpoint), claude-cli or kiro-cli: a set value is ignored there with
+# one INFO line. A LiteLLM proxy is OpenAI-compatible, so point openai-compat
+# at it.
PRXREF_LLM_BASE_URL=https://openrouter.ai/api/v1
-# API key for that endpoint. Leave empty for a local no-auth server.
+# API key for that endpoint (openai-compat only). Leave empty for a local
+# no-auth server.
PRXREF_LLM_API_KEY=
-# Comma-separated fallback chain, cheapest first. REQUIRED.
+# Fallback chain, cheapest first, comma- or whitespace-separated. REQUIRED by
+# every backend.
PRXREF_LLM_MODELS=z-ai/glm-5.3-flash
# Reasoning effort for models that cannot disable reasoning (e.g. low|high|max
@@ -52,10 +59,21 @@ PRXREF_LLM_REASONING_EFFORT=
# Optional integer sampling seed, sent as top-level "seed" to
# OpenAI-compatible backends. Must be >= 0 (0 is a valid seed). Empty or
-# unset omits it from the request entirely, leaving the provider's own
-# seed behaviour in place.
+# unset does not omit it: one random seed is derived per process, sent on
+# every call of the run, and reported as sampling.seed in the run record.
# PRXREF_LLM_SEED=
+# claude-cli / kiro-cli only: the CLI binary to run (~ is expanded). Empty =
+# "claude" or "kiro-cli" found on PATH. The CLI must already be installed and
+# logged in (your own subscription, your own machine: see docs/llm.md). A path
+# that cannot be found is a configuration error (exit 2).
+# PRXREF_LLM_CLI_PATH=
+
+# claude-cli / kiro-cli only: how many CLI processes one client may run at
+# once. Each call is a full CLI process, and subscription limits are per
+# account. Must be > 0.
+# PRXREF_LLM_CLI_CONCURRENCY=2
+
# Findings below this confidence floor are dropped (default 0.6).
# A probability: must be within [0.0, 1.0] inclusive.
PRXREF_CONFIDENCE_FLOOR=0.6
@@ -128,6 +146,92 @@ PRXREF_MAX_CHUNKS=8
# total-failure notice always names its status regardless.
# PRXREF_POST_VERDICT=1
+# Fallback price table, used only when the backend reports no dollar cost
+# (OpenRouter's usage.cost, a LiteLLM gateway's x-litellm-response-cost
+# header, litellm's response_cost, claude-cli's total_cost_usd). Inline JSON
+# (starting with "{") or a path to a JSON file; USD per MILLION tokens, keyed
+# on the exact model name prxref reports (model=). A run priced from it is
+# flagged cost_estimated. A malformed table is a configuration error (exit 2).
+# Local or free models behind a gateway that reports nothing: give them a zero
+# entry, or the run reads "cost unknown" (null), never $0.
+# PRXREF_PRICE_TABLE={"openai/gpt-4o-mini": {"input": 0.15, "output": 0.60}}
+# PRXREF_PRICE_TABLE=./prxref-prices.json
+
+# Set to exactly 1 to append the run's cost to the posted summary attribution
+# line ("… · 3.1s · $0.0007"; "~$0.0007 (est.)" when estimated, else
+# "$0.0007 (API-equivalent)" when every reported cost came from claude-cli;
+# "cost unknown"). Default off; the cost is always in the run record and the JSON.
+# PRXREF_POST_COST=0
+
+# Advisory-only threshold on lines changed (added + removed, excluding
+# lock/generated files): above it, the summary gets one non-blocking heads-up
+# line at the top. Unset (default) disables it. Must be >= 0; 0 is a legal,
+# extreme value, distinct from unset, not the "off" spelling. Never affects
+# the verdict or the exit code.
+# PRXREF_SIZE_WARN_LINES=
+
+# Same contract as PRXREF_SIZE_WARN_LINES, thresholding files instead.
+# PRXREF_SIZE_WARN_FILES=
+
+# Extra glob patterns (fnmatch, case-sensitive, matched against the full diff
+# path; * crosses /) excluded from both size counts above, ADDED to the
+# built-in lock/generated-file detection. Comma- or whitespace-separated, so
+# a literal space in a glob is written ?. Empty (default) adds nothing.
+# PRXREF_SIZE_IGNORE_GLOBS=
+
+# Spec/ticket sources to review against: web URLs, local file or directory
+# paths, or Jira ticket URLs. Comma- or whitespace-separated when set here;
+# repeatable `--spec` flags replace this list entirely (no merge). In CI, a
+# local path inside the PR's checkout is content the PR itself controls.
+# PRXREF_SPEC_SOURCES=
+
+# Raw fetched characters kept per spec source before pruning. Must be > 0.
+# Truncation at the cap is announced in the fetched text, never silent.
+# PRXREF_SPEC_MAX_CHARS=120000
+
+# Token budget for the spec digest injected into worker prompts. Must be > 0.
+# PRXREF_SPEC_DIGEST_TOKENS=3000
+
+# Team review-rules file (Markdown, optional front matter with a severity:
+# map) added to every review prompt. A missing, unreadable or malformed file
+# is a configuration error (exit 2). `--rules-file PATH` wins for one run;
+# `--rules-file ""` turns it off. Read it from a trusted checkout: in CI the
+# workspace is usually the PR's own code, so a rules file inside it lets the
+# PR rewrite its own review rules. See docs/review-rules.md.
+# PRXREF_REVIEW_RULES=
+
+# Characters of the rules body (after the front matter) kept in the prompt;
+# a longer body is truncated with a warning. Must be > 0.
+# PRXREF_REVIEW_RULES_MAX_CHARS=12000
+
+# Plain-text or Markdown file holding the ticket this PR implements. When set,
+# every finding is marked in, out of, or of unknown ticket scope. An empty
+# file means "this PR has no ticket". A missing or non-UTF-8 file is a
+# configuration error (exit 2). Ignored by `prxref serve`.
+# `--context-file PATH` wins for one run; `--context-file ""` turns it off.
+# PRXREF_TICKET_CONTEXT_FILE=
+
+# Characters of ticket text kept in the prompt; longer text is truncated with
+# a visible marker. Must be > 0.
+# PRXREF_TICKET_CONTEXT_MAX_CHARS=6000
+
+# Jira base URL (scheme://host plus any context path) that ticket fetches are
+# looked up on, overriding a ticket URL's own base (a self-hosted board often
+# sits behind a different REST host than its browse URL). Jira credentials
+# are only ever sent here. Empty uses the ticket URL's own base, anonymously.
+# PRXREF_JIRA_BASE_URL=
+
+# Jira account email for HTTP basic auth on ticket fetches, used only together
+# with PRXREF_JIRA_BASE_URL: set without it, the fetch stays anonymous and a
+# warning is logged. Leave empty for anonymous access to public boards.
+# Missing credentials are a fetch failure (the review proceeds un-grounded
+# with a note), never a configuration error.
+# PRXREF_JIRA_EMAIL=
+
+# Jira API token paired with PRXREF_JIRA_EMAIL for HTTP basic auth, sent only
+# to PRXREF_JIRA_BASE_URL.
+# PRXREF_JIRA_API_TOKEN=
+
# ------------------------------------------------------------------------------
# Per-Forge Authentication Tokens
# ------------------------------------------------------------------------------
@@ -159,6 +263,12 @@ PRXREF_MAX_CHUNKS=8
# GitLab token (Personal, project, or group access token with API access)
# PRXREF_GITLAB_TOKEN=
+# Azure DevOps personal access token (Code (Read) to review; Code (Read & write)
+# to post). Sent as Basic ":PAT". When empty, the Pipelines SYSTEM_ACCESSTOKEN
+# is used as a Bearer token; when both are empty, requests are anonymous
+# (public projects, read-only).
+# PRXREF_AZURE_DEVOPS_TOKEN=
+
# ------------------------------------------------------------------------------
# Webhooks Configuration
# ------------------------------------------------------------------------------
@@ -172,6 +282,11 @@ PRXREF_MAX_CHUNKS=8
# Secret token / HMAC secret for GitLab webhook payloads (X-Gitlab-Token)
# PRXREF_GITLAB_WEBHOOK_SECRET=
+# Azure DevOps service-hook secret: compared in constant time with the
+# PASSWORD of the hook's Basic auth (the user name is ignored). Empty rejects
+# Azure DevOps webhooks with 401 unless PRXREF_ALLOW_UNSIGNED=1.
+# PRXREF_AZURE_DEVOPS_WEBHOOK_SECRET=
+
# Set to exactly 1 to accept unsigned webhooks (default off; insecure).
# Only the literal "1" works — true/yes/on are intentionally NOT accepted.
PRXREF_ALLOW_UNSIGNED=0
diff --git a/.gitignore b/.gitignore
index 746c138..165d860 100644
--- a/.gitignore
+++ b/.gitignore
@@ -11,3 +11,4 @@ dist/
docs/superpowers/
docs/configurability/
docs/release-hardening/
+docs/issues/
diff --git a/CHANGELOG.md b/CHANGELOG.md
index c9714df..453e21c 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -5,6 +5,373 @@ All notable changes to this project are documented here.
The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and
this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
+Issue numbers in entries before 0.14.0 refer to the project's previous issue
+tracker.
+
+## [0.14.0] — 2026-09-24
+
+The inputs release. A review can now be grounded in the spec a PR implements,
+follow a team's own review rules, and judge each finding against the ticket the
+PR is for. A replay mode reviews pinned commits for evaluation, Azure DevOps
+becomes the fifth forge, two backends run a review on your own Claude Code or
+Kiro CLI login, and every run records its dollar cost and can flag an oversized
+PR. Each new input is off until you configure it.
+
+### Added
+
+- **Spec-grounded review (`--spec`, `PRXREF_SPEC_SOURCES`).** Name web pages,
+ local spec files or directories, and Jira ticket URLs with a repeatable
+ `--spec URL_OR_PATH` or the list-valued `PRXREF_SPEC_SOURCES`. prxref fetches
+ them, prunes them to a digest of the constraints relevant to this diff
+ (`PRXREF_SPEC_MAX_CHARS` caps each fetched text, `PRXREF_SPEC_DIGEST_TOKENS`
+ the digest), and puts the digest into every chunk worker's and the whole-PR
+ sweep's prompt. A diff that breaks a quoted constraint draws a finding of the
+ new 🔍 `spec` severity, ranked below `warning`. Spec findings are advisory:
+ they never change the verdict or count toward the error cap, and
+ `PRXREF_FAIL_ON=any` is the opt-in gate. A source that fails never blocks the
+ review; the summary's grounding note lists it as `source N (kind): reason`.
+ The digest keeps hard-wrapped statements whole, splits a long or multi-rule
+ block into one MUST/SHOULD/MAY unit per sentence, files each constraint under
+ its own section heading (Markdown, setext or HTML `
`–`
`), turns a
+ version pin on a line of its own into a MUST, and ranks constraints by the
+ diff's content words while ignoring normative ones such as `must` or
+ `required`. Each URL or Jira source gets a 15 s socket timeout, a 30 s
+ wall-clock budget and one retry with no backoff (`Retry-After` is ignored),
+ and a page served without a charset is decoded by its `` tag, then as
+ UTF-8, then as cp1252. See the README's "Review Against a Spec or Ticket",
+ `docs/quality.md` "Spec grounding" and `docs/deploy.md` "Spec Sources in CI
+ and on the Daemon".
+- **A `spec` finding has to be earned.** A run counts as spec-grounded only when
+ at least one constraint reached the prompts. When every source failed or none
+ held a constraint, nothing is injected, the prompts say that no specs were
+ provided, and a `spec` finding the model returns anyway is posted as a
+ `warning` (logged at INFO and counted by a `specs relabel` trace event). On a
+ grounded run the hedge gate skips the text that a finding's `Spec: "…"` quote
+ copies verbatim from the digest, compared case-insensitively and in a finding
+ of any severity, so a condition that belongs to the spec ("If a session
+ already exists, the server MUST …") does not drop the finding as hedged. A
+ quote the digest does not hold exempts nothing.
+- **Jira tickets as spec sources.** A Jira issue URL (`/browse/KEY-1` or a REST
+ issue URL, either one under a context path of up to two segments, a Cloud
+ team-managed issue view, or a board URL carrying `selectedIssue`) is read from
+ Jira REST, and the ticket's summary, type, labels and description lines rank
+ ahead of every other constraint. `PRXREF_JIRA_BASE_URL`, `PRXREF_JIRA_EMAIL`
+ and `PRXREF_JIRA_API_TOKEN` configure access (see Security for where the
+ credentials go). A 401, a 403 or an anonymous 404 comes with a credentials
+ hint, and a 200 that is not a JSON issue, such as an SSO login page, fails
+ that source with a clean message.
+- **Spec grounding in the log, the run record and the trace.** Each failed
+ source logs one WARNING, `spec source N/T (kind, origin) failed
+ (best-effort): reason`, in every run mode, with a URL origin cut to
+ `scheme://host[:port]/path`. Every run with spec sources logs one INFO line,
+ `spec grounding: ok/T source(s) ok, N constraint(s) injected`. The run record
+ and `--format json` carry `spec_grounding` (`sources`, `ok`, `failed`,
+ `constraints`, `digest_sha256`), and the trace's `specs` event is `ok`, or
+ `fail` with the `reasons` when no source was fetched or the stage crashed.
+- **Azure DevOps Repos forge (#2).** `prxref review` and `prxref serve` handle
+ Azure DevOps Services (`dev.azure.com`, `*.visualstudio.com`) and Azure DevOps
+ Server (any host, with the collection in the URL). Inline findings post as
+ active threads, the summary is one closed PR-level thread that later runs
+ update in place, and stale inline comments of prxref's own are pruned. Azure
+ DevOps has no unified-diff endpoint, so the diff is rebuilt from the Diffs API
+ (merge-base semantics) plus blob contents: pure renames and known-binary files
+ are never downloaded, and a file past the per-blob, file-count or byte budget
+ keeps its header without hunks. Authentication is `PRXREF_AZURE_DEVOPS_TOKEN`
+ (a PAT), else `SYSTEM_ACCESSTOKEN` inside Azure Pipelines, else anonymous
+ reads of a public project. The webhook server accepts the
+ `git.pullrequest.created` and `git.pullrequest.updated` service hooks for an
+ active PR, checking the HTTP Basic password against
+ `PRXREF_AZURE_DEVOPS_WEBHOOK_SECRET` in constant time (unset, it answers 401
+ unless `PRXREF_ALLOW_UNSIGNED=1`). The CLI help, the unrecognized-URL hint and
+ the package description name Azure DevOps; setup is in `docs/forges.md`
+ section 5 and `docs/deploy.md`.
+- **Team review rules (#3).** `--rules-file PATH` or `PRXREF_REVIEW_RULES` names
+ a Markdown or plain-text checklist. Its body goes into the system prompt of
+ every chunk worker and the sweep, under a `## Team review rules` heading and
+ inside `` tags, and `PRXREF_REVIEW_RULES_MAX_CHARS` (default
+ 12000) caps it, with a truncation line and one WARNING. An optional `severity:`
+ front-matter block maps team words onto `error`, `warning` or `outofscope`
+ (`blocker: error`), and a mapped word the model writes is rewritten before
+ every quality pass instead of being dropped as an invalid severity; `spec` is
+ not a target, and prxref's own severities cannot be remapped. Other
+ front-matter keys are ignored, so a skill file works unmodified. The run
+ record's `review_rules` holds the path, the raw file's SHA-256, the lengths,
+ the truncation flag and the severity map, never the text; `-v` prints a
+ `rules:` line, and the trace gains `rules ok` and `rules remap` events.
+ `--rules-file ""` turns an environment-configured file off for one run, an
+ unusable file exits 2 before any network call, and the webhook server
+ re-reads the file for every review. See `docs/review-rules.md`.
+- **Ticket context and a per-finding scope (#4).** `--context-file PATH` or
+ `PRXREF_TICKET_CONTEXT_FILE` names the ticket a PR implements, and every
+ finding is judged `in`, `out` or `unknown` against it. The ticket is quoted to
+ the model as fenced, untrusted data, capped by `PRXREF_TICKET_CONTEXT_MAX_CHARS`
+ (default 6000). An out-of-ticket finding is marked 🟦 in front of its severity
+ glyph, listed last in the summary under **🟦 Outside the ticket (N)**,
+ labelled `OUTSIDE TICKET` in its inline comment, yields inline slots to
+ in-ticket findings of the same severity, and ends in `[scope: out]` in CLI
+ text. Scope never feeds dedup, the error cap, the verdict or `PRXREF_FAIL_ON`.
+ An empty file means "this PR has no ticket" and the summary says so, and a
+ ticket without acceptance criteria (an `Acceptance criteria` or `Definition of
+ done` heading or label, a task-list item, or a Gherkin `Given` … `Then`) gets
+ a note that scope was judged from its description alone. Every finding in
+ `--format json` carries `scope`, which stays `unknown` without a ticket; the
+ record's `ticket_context` holds metadata only. The webhook server never reads
+ a ticket file, and warns once at startup if the variable is set.
+- **Replay mode for evaluation (#5).** `--base-sha` and `--head-sha` review a
+ pinned commit range (the merge-base diff, with file context read at the
+ pinned head), `--no-threads` hides the PR's existing discussion, and
+ `--diff-file` reviews a diff file, with `--pr-url` or with no forge at all
+ (a `git format-patch` mail's subject, body and author become the PR's title,
+ description and author). A replay never posts, even for a library caller that
+ passes `post=True`; a blank replay diff is an `Error` run rather than an
+ `Approved` one; and the run record gains a `replay` stamp. Every built-in
+ forge implements the new optional `Forge.get_compare_diff(ref, *, base_sha,
+ head_sha)`, and `docs/forges.md` documents each forge's endpoint and caveats.
+- **Subscription CLI backends: `claude-cli` and `kiro-cli` (#6).**
+ `PRXREF_LLM_BACKEND=claude-cli` or `kiro-cli` reviews with your own installed,
+ logged-in Claude Code or Kiro CLI instead of an HTTP endpoint: one process per
+ model attempt, started from a fresh temporary directory, with the diff on
+ stdin. `claude-cli` runs `claude -p` with no built-in tools, settings files,
+ MCP servers or saved session; it removes eight credential-routing variables
+ (`ANTHROPIC_API_KEY`, `ANTHROPIC_BASE_URL`, `CLAUDE_CODE_USE_BEDROCK` and the
+ like) from the child's environment so the call stays on your subscription
+ login, maps `PRXREF_LLM_REASONING_EFFORT` to `--effort`, and warns if the CLI
+ reports an API-key source, loads tools anyway, or reports a rate-limit status
+ other than `allowed`. `kiro-cli` runs `kiro-cli chat --no-interactive` on the
+ v2 agent engine, with a per-call agent file that carries the system prompt and
+ the model and allows no tools, MCP servers or resources. `PRXREF_LLM_MODELS`
+ is walked as a fallback chain, `PRXREF_LLM_CLI_PATH` overrides the binary,
+ `PRXREF_LLM_CLI_CONCURRENCY` (default 2) caps the processes running at once, a
+ deadline miss kills the whole process group, and a missing CLI exits 2 before
+ any forge or model call. See the "Subscription CLI backends" section of
+ `docs/llm.md`.
+- **Dollar cost in the run record (#7).** `cost_usd` is the total of the
+ review's calls, the chunk workers plus the sweep, including truncated or
+ unparseable responses that were billed, and `0.0` when no model call went
+ out. A figure the backend reported always wins: the response body's
+ `usage.cost` (OpenRouter), the `x-litellm-response-cost` header (a LiteLLM
+ gateway or llm-ferry), litellm's `response_cost`, or the Claude Code CLI's
+ `total_cost_usd`, an API-equivalent at list price rather than what a
+ subscription is billed. Otherwise the cost is estimated from
+ `PRXREF_PRICE_TABLE` (inline JSON or a JSON file path, USD per million tokens,
+ keyed by exact model name, validated at load so a malformed table exits 2),
+ and `cost_estimated` is set. Otherwise it is `null`, never `0` and never a
+ partial sum, and one INFO line names the unpriced models. `PRXREF_POST_COST=1`
+ appends the cost to the posted attribution line, `-v` prints it (`$…`,
+ `$… (API-equivalent)`, `~$… (est.)` or `cost unknown`), and the `chunk ok`,
+ `sweep ok` and `run ok` trace events, each unit's `.meta.json`
+ (`cost_usd`, `cost_source`) and the openai-compat attempt log line (`cost=`)
+ carry it. A run whose every reported figure came from the Claude Code CLI
+ shows its cost as `$0.0202 (API-equivalent)` on the `-v` line and on the
+ posted attribution, because `total_cost_usd` is the API list price, not a
+ subscription bill; an estimated run keeps `~$… (est.)`. The run record marks
+ such a run with `cost_api_equivalent: true`, a key no other run carries, and
+ `--format json` gains no key, because each unit's `cost_source` already says
+ `claude-cli`. prxref never asks a provider to add usage to a response. See
+ `docs/llm.md` "Cost accounting".
+- **PR size advisory (#8).** Set `PRXREF_SIZE_WARN_LINES` (lines added plus
+ removed) and/or `PRXREF_SIZE_WARN_FILES` (files changed), and a PR strictly
+ above either one gets a line at the top of its summary: "This PR changes N
+ lines in M files, above the team guideline of … Consider splitting it." Both
+ are off by default, and `0` is a real threshold. The counts come from the
+ parsed diff and skip lockfiles from the common ecosystems, generated files
+ (`*.snap`, `__snapshots__/`, `*.min.js`, `*.map`, `*.generated.*`,
+ `*.auto.*`) and any path matching `PRXREF_SIZE_IGNORE_GLOBS`. The advisory
+ never changes the verdict or the exit code; it is also reported as
+ `size_advisory` in the run record and `--format json`, and as a
+ `size advisory:` line in the CLI output.
+- **New run-record and `--format json` keys.** `cost_usd`, `cost_estimated`,
+ `review_rules`, `ticket_context`, `spec_grounding` and `size_advisory` are
+ present on every exit, error and empty-diff exits included, and `null` when
+ their feature is off. In `--format json` they follow the existing keys in a
+ fixed, documented order, and a replay run adds `replay`. The CLI text output
+ gains a `replay:` line on a replay, and `-v` adds `rules:`, `ticket:` (with the
+ in/out/unknown counts) and `spec:` lines. The README's "CLI Flags" section now
+ documents every `review` flag and lists the JSON keys in payload order.
+
+### Changed
+
+- **Minor findings render ⬜, and 🟦 now means "outside the ticket".**
+ `outofscope` findings and unrecognised severities show a grey square in the
+ summary counts, the findings list, inline comment headers and the library
+ formatter, where 0.13.0 showed 🟦; the JSON value and the `OUTOFSCOPE` label
+ are unchanged. The summary counts line also gains a `🔍 N spec` count on
+ every run. Every glyph now comes from one table, `prxref.markers`.
+- **The review prompts changed for every run.** The worker and sweep prompts
+ define the `spec` severity and its rules, and carry a `### Spec constraints`
+ block that reads `(no specs provided for this review)` when none are
+ configured, so a review of the same diff can differ from 0.13.0's even with
+ none of the new inputs set.
+- **An unrecognized `PRXREF_LLM_BACKEND` is a configuration error (exit 2)**
+ that names the variable and lists the six accepted values, checked before any
+ other LLM setting. In 0.13.0 it was a failed review that exited 0.
+- **`PRXREF_LLM_MODELS` splits on whitespace as well as commas**, like the new
+ list-valued `PRXREF_SPEC_SOURCES` and `PRXREF_SIZE_IGNORE_GLOBS`. 0.13.0 split
+ it on commas only.
+- **GitLab MR diffs are requested without `access_raw_diffs`.** Every earlier
+ release sent it to `/merge_requests/:iid/diffs`, which ignores it: gitlab.com
+ returned byte-identical diffs with and without it, and only the deprecated
+ `/changes` endpoint reads it. The reviewed diff is unchanged.
+
+### Fixed
+
+- **The `litellm` backend no longer requires `PRXREF_LLM_BASE_URL` (#1).** Only
+ `openai-compat`, `ferry` and `http` need it. Set on any other backend, it is
+ ignored with one INFO line and never forwarded, so a deployment that set a
+ placeholder URL to get past the old check keeps working unchanged. To use a
+ LiteLLM proxy, choose `openai-compat`.
+- **GitLab merge requests with more than 20 files are reviewed in full.** The
+ adapter read only the first page of GitLab's MR diff list, 20 files by
+ default, and dropped every file after that without saying so. It now reads
+ every page, and a page that cannot be read fails the review with an error
+ naming it instead of reviewing part of the MR. A warning names each file
+ GitLab sends without hunks (`too_large` or `collapsed`).
+- **GitHub and GitHub Enterprise API calls time out**: 10 s to connect and 30 s
+ per read, the same as the other forges. A stalled GitHub connection could
+ hang the review, and the webhook worker running it, forever. A write that
+ times out is not retried, so it cannot post a duplicate comment.
+- **A `{placeholder}` in PR text is shown literally.** A PR title or
+ description containing `{diff}` had the diff pasted into the prompt at that
+ spot, and a PR or finding title containing `{findings}` or `{attribution}` did
+ the same to the posted summary. Prompts and the summary are now filled in a
+ single pass.
+- **`load_config` no longer shares list defaults between calls.** Appending to
+ one loaded config's `llm_models` changed the default that every later load
+ started from.
+- **Under `PRXREF_FAIL_ON=error` or `any`, a review that ends with verdict
+ `Error` exits 1**, as documented since 0.4.0. Before, only a crash did: a
+ forge that could not be read, a diff that could not be parsed or chunked, and
+ a review in which every chunk failed all return an `Error` result rather than
+ raising, so a gating lane read those broken runs as green. The default
+ `never` is unchanged.
+- **A total LLM failure no longer counts a successful sweep as failed.** When
+ every chunk worker failed but the whole-PR sweep answered, the run correctly
+ ended with verdict `Error` but reported `chunks_reviewed` 0 and every review
+ unit failed. It now counts the sweep as reviewed: `chunks_reviewed` is 1,
+ `chunks_failed` is the number of chunks, and the two still add up to
+ `chunk_count`. The text output therefore reads `coverage: 1/2 chunks
+ reviewed` on a one-chunk PR. The verdict, the posted error notice and the
+ `PRXREF_FAIL_ON` exit code are unchanged.
+- **The `forge.get_diff` trace span counts bytes.** Its `bytes` field counted
+ characters, so a diff with non-ASCII text or a byte-order mark read short
+ (15,647 against 15,651 on one live Azure DevOps pull request).
+- **Documentation corrections.** `docs/deploy.md` no longer says there is no
+ `PRXREF_FAIL_ON` (there is: `never`, `error` or `any`, default `never`), its
+ exit-code table gains the `1` row, and its webhook table lists the Bitbucket
+ Cloud events prxref accepts (`pullrequest:created`, `pullrequest:updated`) and
+ gains Bitbucket Server / Data Center and Azure DevOps rows. `.env.example` and
+ `docs/env-vars.md` no longer say that an unset `PRXREF_LLM_SEED` leaves the
+ seed out of the request: prxref sends one random seed per process and reports
+ it as `sampling.seed`. `docs/llm.md` names the backends that apply the seed,
+ the reasoning effort and the max-tokens settings. `docs/forges.md` documents
+ GitLab's paged diff listing, and that reading a merge request's threads on
+ gitlab.com needs `PRXREF_GITLAB_TOKEN` even for a public project: gitlab.com
+ answers anonymous `/notes` and `/discussions` requests with HTTP 401, so a
+ tokenless review logs `discussion feed read was incomplete` and dedups against
+ no threads.
+
+### Security
+
+- **Earlier releases were withdrawn and the history rewritten.** The published
+ sdists of 0.10.1 through 0.13.0 and the wheels of 0.12.0 through 0.13.0 were
+ removed from PyPI, so 0.12.0 through 0.13.0 can no longer be installed from
+ it. The git history before 0.14.0 was rewritten to drop internal planning
+ notes: every tag and nearly every commit before 0.14.0 has a new SHA, so
+ re-clone an existing clone rather than pulling into it.
+- **Files named by path stay inside the working directory.** A rules file, a
+ ticket-context file or a local spec source that sits under the working
+ directory, such as a file committed in a PR checkout, must still resolve
+ under it once its symlinks are followed. So a committed
+ `docs/SPEC.md -> ~/.ssh/id_rsa` is refused without revealing the link target,
+ and every symlinked entry inside a spec directory is skipped. An absolute path
+ outside the working directory is the operator's own choice and is read as
+ given. These files are read as strict UTF-8 in bounded memory, and the rules
+ and ticket loaders refuse URLs.
+- **Credentials and paths stay out of what prxref sends.** Jira credentials go
+ only to `PRXREF_JIRA_BASE_URL`: set without it, ticket fetches are anonymous
+ and a WARNING names the variable, and a plain-`http` base URL is used with a
+ WARNING. The digest names a spec source to the model only by its last path
+ segment or bare host, with no query, userinfo, port or full local path (a
+ credential that is itself the last path segment still gets through), and the
+ spec failure reasons the summary posts carry no local path. The run record
+ and the trace hold a rules or ticket file's metadata, never its text.
+
+### Known limitations
+
+- **Unpunctuated spec lines merge.** Consecutive keyword lines in one paragraph
+ that end without punctuation (a list with no bullets or full stops) become a
+ single constraint labelled with the strongest keyword among them, and a run
+ of them longer than 400 characters is cut at 400, losing the rest. End each
+ rule with a full stop, or make it a list item.
+- **A `Spec: "…"` quote that never closes is barely exempt.** The hedge gate
+ exempts a quote only up to a closing quote mark. A quote with no closing
+ quote, or one that departs from the digest's wording before it closes, is
+ exempt only up to its last inner quote mark, and not at all without one, so a
+ condition inside it can still drop the finding as hedged.
+- **A spec directory is read shallowly.** A directory source reads at most the
+ first 20 `.md`, `.markdown`, `.txt` or `.adoc` files directly inside it, by
+ name, and says nothing about the rest. Files it skips as unreadable or
+ symlinked are named in a WARNING log line only, not in the posted note, the
+ run record or the trace.
+- **A spec directory's constraints are not tagged per file.** They carry the
+ directory's name and a line number counted through its files joined together,
+ each file under a `## ` heading, rather than the file's own name
+ and line.
+- **A Jira ticket can crowd out the spec.** Ticket lines rank ahead of every
+ other constraint and may fill up to 6,000 characters of the digest, a fixed
+ share (half the default `PRXREF_SPEC_DIGEST_TOKENS` budget) that does not
+ shrink with `PRXREF_SPEC_MAX_CHARS`. With a smaller digest budget, a long
+ ticket can leave no room for anything else.
+- **A `spec` finding's quote is not checked against the digest.** On a
+ grounded run, a `spec` finding keeps its severity even when the constraint it
+ quotes is not in the digest; only the hedge-gate exemption requires a match.
+- **A Jira ticket passed with `--spec` is not ticket context.** It grounds
+ `spec` findings but sets no finding's scope; to judge scope, pass the
+ ticket's text with `--context-file`.
+- **Spec grounding can crowd out a generic finding.** In the bundled eval,
+ case-002's one expected non-spec finding was missed in all 4 grounded runs,
+ across two models, and found in 3 of the 4 runs without the spec. That is two
+ runs per model in each arm, and the cause, grounding displacing generic
+ review, is inferred, not proven.
+- **`kiro-cli` runs always read "cost unknown".** Kiro meters credits, not
+ dollars, and reports no token counts, so a price table cannot estimate it
+ either; each call's INFO line carries the credits instead.
+- **`kiro-cli` isolation and reporting are partial.** Whether Kiro adds
+ user-level configuration, such as `~/.kiro/steering/`, to prxref's per-call
+ agent has not been verified, and the agent file cannot turn it off. Kiro does
+ not report which model ran, so the attribution names the model you
+ configured, and it keeps every chat, the diff included, under
+ `~/.kiro/sessions/cli/` (`docs/llm.md` shows how to delete them).
+- **GitLab files without hunks are not reviewed.** A file whose diff GitLab
+ withholds as `too_large` or `collapsed` arrives without hunks, so it is listed
+ header-only and not reviewed; a warning names each such file.
+- **GitLab merge requests past 5,000 files fail.** An MR whose diff listing runs
+ past 50 pages of 100 files fails the review rather than reviewing part of it.
+- **GitHub pull requests past 20,000 diff lines are not reviewed.** GitHub
+ refuses the unified diff of such a pull request with HTTP `406`
+ (`too_large`), so the review ends with verdict `Error` and posts the error
+ notice. There is no fallback to the paged file listing yet. This limit
+ applies to every earlier release too.
+- **A forge-less replay sees only the diff.** A `--diff-file` run without
+ `--pr-url` gives the workers no library versions and no out-of-hunk
+ definitions, gives the manifest claim check no full-file lines, and has no
+ existing discussion; its title and description come from a `git format-patch`
+ header when there is one, else the file name. With `--pr-url`, a replay keeps
+ the PR's current title and description, pinned SHAs without `--no-threads`
+ still show the current threads, and a `--diff-file` without `--head-sha` reads
+ file context at the PR's current head; the last two each log a warning.
+- **Azure DevOps is verified live for anonymous reads only.** On a public Azure
+ DevOps Services project, the forge reads, the dry-run output shape, the
+ pinned-range compare diff (including a PR whose target branch had moved on)
+ and a pinned-range replay were checked live. Posting the summary and inline
+ threads, pruning, PAT and `SYSTEM_ACCESSTOKEN` authentication and the
+ service-hook payload are tested against recorded API shapes only, and Azure
+ DevOps Server (a release that accepts REST `api-version=7.1`) is untested.
+- **A mistyped or unrecognized `--pr-url` exits 0 even under
+ `PRXREF_FAIL_ON=error` or `any`.** prxref prints the unrecognized-URL hint to
+ stderr and exits 0, because nothing was reviewed and there is no outcome to
+ gate on, so a gating lane with a malformed URL stays green.
+
## [0.13.0] — 2026-09-17
### Added
@@ -820,7 +1187,10 @@ Development baseline. Never published to PyPI and never tagged; superseded by
- Diff content is sent to whichever OpenAI-compatible endpoint you configure.
- Requires Python 3.12+. Tested on 3.12 and 3.13.
-[Unreleased]: https://github.com/sblattj/prxref/compare/v0.12.1...HEAD
+[Unreleased]: https://github.com/sblattj/prxref/compare/v0.14.0...HEAD
+[0.14.0]: https://github.com/sblattj/prxref/releases/tag/v0.14.0
+[0.13.0]: https://github.com/sblattj/prxref/releases/tag/v0.13.0
+[0.12.2]: https://github.com/sblattj/prxref/releases/tag/v0.12.2
[0.12.1]: https://github.com/sblattj/prxref/releases/tag/v0.12.1
[0.12.0]: https://github.com/sblattj/prxref/releases/tag/v0.12.0
[0.11.1]: https://github.com/sblattj/prxref/releases/tag/v0.11.1
diff --git a/CLAUDE.md b/CLAUDE.md
index ef67fbd..d89bf97 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -1,10 +1,10 @@
# prxref
-Fast automated AI code review for Bitbucket, GitLab, and GitHub — Cloud and self-hosted.
+Fast automated AI code review for Bitbucket, GitLab, GitHub, and Azure DevOps — Cloud and self-hosted.
## What This Is
-A Python CLI + webhook service that reviews PRs/MRs on any of the three major
+A Python CLI + webhook service that reviews PRs/MRs on any of the four major
forges by: parsing one unified diff, chunking it, running parallel single-shot
LLM worker reviews with a fallback model chain, gating findings through
deterministic quality passes, and posting inline comments + a summary.
@@ -15,7 +15,7 @@ auto-detects the forge.
## Tech
- Python 3.12+, `uv` for env/lock, hatchling packaging
-- One `Forge` Protocol (src/prxref/forges/base.py), four adapters
+- One `Forge` Protocol (src/prxref/forges/base.py), five adapters
- Bitbucket needs two of them: Cloud speaks `/2.0` on `bitbucket.org` only,
Server / Data Center speaks `/rest/api/1.0` on any host, so the adapter is
picked from the URL. `detect_forge` asks Cloud first, but that order is
@@ -26,6 +26,9 @@ auto-detects the forge.
first means a later loosening degrades into a shadowed forge rather than a
silently mis-routed one. GitHub and GitLab stay one adapter each, because
their self-hosted products differ only in base URL.
+- Azure DevOps is one adapter for Services and Server, asked last by
+ `detect_forge`; it has no unified-diff endpoint, so it rebuilds the diff
+ locally from the Diffs API change list plus blob contents.
- LLM access via a fallback chain (llm-ferry preferred, litellm optional,
plain-HTTP client as zero-dependency default) — provider-agnostic, no
Anthropic key by design
diff --git a/HANDOFF.md b/HANDOFF.md
index f9f6e86..c1b35f9 100644
--- a/HANDOFF.md
+++ b/HANDOFF.md
@@ -1,175 +1,297 @@
-# HANDOFF — v0.5.0 shipped: the Bitbucket Server / Data Center forge
+# HANDOFF — v0.14.0 shipped: the inputs release
-**Repo:** `sblattj/prxref` (public) · **Released:** 2026-08-28 · **Supersedes** the
-"cut v0.5.0" handoff written the same day.
+**Repo:** `sblattj/prxref` (public) · **Released:** 2026-09-24 · **Supersedes** the
+v0.5.0 handoff.
-The Bitbucket Server / Data Center forge is released. It had been finished and
-proven in real use but lived only on `origin/feat/bitbucket-server-forge`
-(`cc58915`), so every deployment that needed it ran a hand-maintained overlay of
-`bitbucket_server.py` on top of a tagged release. **Delete that overlay** — v0.5.0
-carries the forge natively.
+0.14.0 lets a review read more than the diff: the spec a PR implements, a team's
+own review rules, and the ticket the PR is for. It also adds a replay mode for
+evaluation, Azure DevOps as the fifth forge, two backends that run on a Claude
+Code or Kiro CLI login, a dollar cost for every run, and a PR size advisory. The
+user-facing account is the `[0.14.0]` section of `CHANGELOG.md`. This file is for
+whoever cuts the next release.
+
+This is the first release on the rewritten history. The repository was recreated
+on 2026-09-24. Every tag and nearly every commit before 0.14.0 has a new SHA, so
+re-clone rather than pull, and treat any SHA quoted in an older note as dead.
+Issue numbers restarted too: the eight 0.14.0 issues are #1 to #8 on the new
+tracker.
## What landed
-- `src/prxref/forges/bitbucket_server.py` — project and personal (`~slug`)
- repositories, deployment context paths, anchored inline comments, `start`/`limit`
- paging, and the `version` field Data Center requires when updating a comment.
-- Registration in both places that matter: the `detect_forge` module tuple in
- `forges/base.py` and the `impls` dict in `config.py`'s `make_forge`.
-- Three env vars: `PRXREF_BITBUCKET_SERVER_TOKEN` (falls back to
- `PRXREF_BITBUCKET_TOKEN`), `PRXREF_BITBUCKET_SERVER_USER` and
- `PRXREF_BITBUCKET_SERVER_PASSWORD`.
-- **A real bug fix, not just the new forge:** Bitbucket webhooks were broken for
- *both* products. The receiver accepted only `pr:opened` / `pr:modified` —
- Bitbucket **Server** event names — while reading the PR URL from
- `pullrequest.links.html.href`, which is Bitbucket **Cloud**'s payload shape. A
- genuine Cloud webhook was rejected as not reviewable; a genuine Server webhook
- produced no URL. Both dialects now work.
-- Docs, README, `CLAUDE.md` and `.env.example` updated, and every "Bitbucket is
- Cloud only" / "Server is not supported" claim removed.
-
-## Four things the previous handoff got wrong
-
-Recorded because each one would have cost the next person real time.
-
-1. **The Cloud-before-Server ordering rationale was false.** The old handoff said
- Cloud's parser is the more specific of the two and that Server-first would make
- Cloud URLs match Server. Tested by reversing the tuple and running six URLs
- through `detect_forge`: **every case resolved identically.** The parsers are
- disjoint — Cloud pins `^https?://bitbucket\.org/` plus a bare
- `owner/repo/pull-requests/N`; Server requires a `/projects|users/KEY/repos/REPO/`
- prefix. No URL matches both, including the adversarial `bitbucket.org` host with
- a Server-shaped path, which only Server matches under either order. The Cloud-first
- order is kept as **defence in depth** — if either parser is later loosened, the
- failure degrades into a shadowed forge rather than a mis-routed one — but it is
- not load-bearing, and no document should claim it is.
-
-2. **Both registration line numbers pointed at the wrong line.** `forges/base.py:93`
- and `config.py:378` are the `def` lines; the literals that actually need editing
- were the tuple at `base.py:97` and the dict at `config.py:386-390`. Cite the
- construct, not the function.
-
-3. **`uv run pytest` works again — the invocation is now the bare one.** It used
- to die with `Failed to spawn: pytest / No such file or directory (os error 2)`,
- because pytest lived in `[project.optional-dependencies] dev` and `uv run` never
- installs a project *extra*; the error reads like a broken venv rather than a
- missing flag. The dev tools now live in `[dependency-groups] dev`, which uv
- installs by default, so the `--extra dev` form is gone from every surface:
-
- ```bash
- uv run pytest
- ```
-
-4. **The diff direction that reads "what the branch changed" is backwards here.**
- `cc58915` was cut from v0.2.0, so `git diff main...cc58915` renders main's own
- v0.3/v0.4 work as *additions* — applying it literally reverts the entire config-surface
- release (17 config rows and 4 sections in `docs/env-vars.md` alone). Most hunks on
- that branch are regressions, not features. Diff the branch's **own** delta instead:
-
- ```bash
- git diff a7abbf1 cc58915 --
- ```
+- **Spec-grounded review.** `--spec` / `PRXREF_SPEC_SOURCES` fetch web pages,
+ local files or directories, and Jira tickets. `specs.build_spec_digest` prunes
+ them to the constraints relevant to the diff, and the digest goes into every
+ worker and sweep prompt. A diff that breaks a quoted constraint draws an
+ advisory `spec` finding. The finding has to be earned. When no constraint
+ reached the prompts, `quality.apply_spec_grounding` relabels any `spec` finding
+ as a `warning`. On a grounded run, `quality.apply_hedge_gate(...,
+ spec_digest=...)` skips the text a finding quotes verbatim from the digest.
+ The run record carries `spec_grounding`.
+- **#1 litellm without a base URL.** Only `openai-compat`, `ferry` and `http`
+ need `PRXREF_LLM_BASE_URL`. The other backends ignore it and log one INFO line.
+- **#2 Azure DevOps.** `forges/azure_devops.py` covers Services and Server, and
+ `detect_forge` asks it last. Azure DevOps has no unified-diff endpoint, so the
+ adapter rebuilds the diff from the Diffs API change list plus blob contents.
+ Webhooks arrive as service hooks, checked against a Basic-auth secret.
+- **#3 Team review rules.** `rules.py` reads `--rules-file` /
+ `PRXREF_REVIEW_RULES` into the system prompt. Optional `severity:` front matter
+ maps team words onto prxref severities, and `quality.apply_severity_map`
+ rewrites them before every other quality pass.
+- **#4 Ticket context and scope.** `ticket.py` reads `--context-file` /
+ `PRXREF_TICKET_CONTEXT_FILE`, and every finding is judged `in`, `out` or
+ `unknown` against it (`triage.normalize_scope`). 🟦 now marks an
+ out-of-ticket finding, so minor findings moved to ⬜.
+- **#5 Replay.** `--base-sha` / `--head-sha`, `--no-threads` and `--diff-file`
+ review a pinned range or a diff file. Every forge implements
+ `Forge.get_compare_diff`, `forges/replay.py` holds `LocalDiffForge` and
+ `ReplayForge`, and a replay never posts. `tests/evals/test_eval_replay.py`
+ replays each eval case with one offline CLI call.
+- **#6 Subscription CLI backends.** `llm_cli_backends.py` adds `claude-cli` and
+ `kiro-cli`. Each model attempt is one process started in a fresh temporary
+ directory, the credential-routing variables are stripped from its
+ environment, and a missed deadline kills the whole process group.
+- **#7 Dollar cost.** In `costs.py`, a figure the backend reported always wins.
+ Otherwise `PRXREF_PRICE_TABLE` gives an estimate, and without one the cost is
+ `null`, never `0` and never a partial sum.
+- **#8 Size advisory.** `PRXREF_SIZE_WARN_LINES` / `PRXREF_SIZE_WARN_FILES` flag
+ an oversized PR. `triage.count_size_relevant_changes` counts the parsed diff,
+ skipping lockfiles, generated files and `PRXREF_SIZE_IGNORE_GLOBS` matches.
+- **Fixes found on the way.** GitLab's MR diff listing now reads every page and
+ fails on a short read instead of stopping at 20 files. GitHub calls time out.
+ Prompts and the summary fill in a single pass, so a `{diff}` in PR text stays
+ literal. `load_config` no longer shares list defaults between calls. An
+ unrecognized `PRXREF_LLM_BACKEND` exits 2. Under `PRXREF_FAIL_ON=error` or
+ `any`, a review that ends with verdict `Error` exits 1. A total LLM failure
+ counts the sweep that answered, and the `forge.get_diff` trace span counts
+ bytes, not characters.
+- **Config went from 36 to 55 keys.** The 19 new keys are the CLI path and
+ concurrency, cost, size, spec, rules and ticket keys, the Jira credentials, and
+ the Azure DevOps token and webhook secret.
+
+## What this release taught
+
+Written down because each one cost a seat real time.
+
+1. **Lay the seams first, and make every stub fail closed.** A foundation stage
+ landed every shared surface with a placeholder body before any feature seat
+ started: config keys, run-record and JSON keys, prompt slots, trace events,
+ the glyph table and the new `Forge` method. That let the eight issue seats run
+ in parallel on disjoint files. A stub that fails closed cannot ship as a
+ silent no-op. The cost is that placeholder prose outlives the placeholder.
+ Docstrings saying the loaders "fail closed in this build" and the cost hooks
+ are "inert in this build" survived after the real bodies landed, and the
+ release had to sweep them. Grep for `in this build` before cutting.
+2. **Cite the symbol, not the line.** Several seats found the `file:line` pins in
+ their briefs stale against the base they had been given, and every one of
+ them still resolved by symbol name. Pin a SHA if you must give a line, and
+ prefer `module.function`.
+3. **Keep one table per cross-cutting literal.** Every severity and scope glyph
+ comes from `prxref.markers`, and `tests/test_markers.py`
+ (`TestGlyphsLiveInOnePlace`) fails when a glyph literal turns up anywhere else
+ in the package. So a glyph change is made in one table, and the test finds
+ any stray copy.
+4. **`Tracer.event(node, phase, **meta)` reserves two keyword names.** A dict
+ splatted into it must not carry a `node` or `phase` key. Such a call raises
+ `TypeError` at the call site, before tracing's never-raise guard can catch it.
+5. **A mutation check needs `PYTHONDONTWRITEBYTECODE=1`.** To prove a test can
+ fail, revert a line, watch the test go red, then restore the line and `cmp` it.
+ Without the variable, the mutant's bytecode gets cached. A same-size restore
+ within the same second can then run the mutant again.
+6. **Two contract rules over one field need a tiebreak.** One rule said every
+ new JSON key is always present and `null` when its feature is off. Another
+ said `replay` is absent on a normal run. The code followed the second, and a
+ seam test now pins that. The next contract should say which rule wins before
+ any seat starts.
## The coupling that will catch the next person adding a config key
-`tests/test_docs_consistency.py` compiles `docs/env-vars.md` and `.env.example`
-against `config._DEFAULTS` **in both directions**, and asserts two hard-coded
-integers built as `f"**{len(_DEFAULTS)}** configuration keys"` and
+`tests/test_docs_consistency.py` checks `docs/env-vars.md` and `.env.example`
+against `config._DEFAULTS` **in both directions**. It also asserts two hard-coded
+integers, built as `f"**{len(_DEFAULTS)}** configuration keys"` and
`f"for {len(_DEFAULTS)+len(_LEGACY_ENV_ALIASES)} accepted variable names"`.
-So a new config key is not a source change — it is an atomic four-surface change:
-`_DEFAULTS`, the `config.py` docstring, `.env.example`, and `docs/env-vars.md`
-including its counts and its `Per-Forge Auth (N)` section heading. Adding three keys
-here failed five tests until all four surfaces moved together. Current values: **33**
-keys, **1** legacy alias, **34** accepted names.
+So a new config key is not a source change. It is an atomic change across four
+surfaces: `_DEFAULTS` plus the `_INT_KEYS` / `_FLOAT_KEYS` / `_RANGES` /
+`_CHOICE_KEYS` tables, the `config.py` docstring, `.env.example`, and
+`docs/env-vars.md`, including its counts and its per-section headings. 0.14.0
+added 19 keys this way. Current values: **55** keys, **1** legacy alias, **56**
+accepted names.
## Release shape (follow this next time)
+How 0.14.0 was built:
+
+1. **Foundation.** Seats lay the shared seams every feature needs, each with a
+ placeholder body that fails closed. Nothing user-visible lands here.
+2. **Wave 1.** One seat per issue runs in parallel, each in its own worktree
+ with a disjoint file list. A seat fills a placeholder; it does not add a seam.
+3. **Wave 2.** Next come the pieces that needed two wave-1 bodies in place (the
+ replay CLI, replay over Azure DevOps, the second CLI backend) and the forge
+ fixes wave 1 surfaced.
+4. **One integration gate per merge.** Each seat branch merges into
+ `release/X.Y.Z` on its own, and a merge stays only if the full
+ `uv run pytest` and `uv run ruff check src tests` pass on the merged tree.
+5. **REL.** Parallel seats sweep stale docs, add cross-seat seam tests, and
+ write the version bump, the CHANGELOG and this file. Read-only live checks
+ follow against public PRs and real CLIs (see "Live checks" below).
+
+Cutting the release:
+
```bash
-uv build # produces BOTH sdist and wheel
-gh release upload vX.Y.Z dist/prxref-X.Y.Z.tar.gz dist/prxref-X.Y.Z-py3-none-any.whl
+# bump pyproject.toml and src/prxref/__init__.py, then:
+uv lock # uv.lock carries the version too
+git tag vX.Y.Z && git push origin vX.Y.Z
```
-Both assets matter: the v0.4.0 release ships both, and at least one consumer updates
-itself with `gh release download --pattern '*.tar.gz'`, which does **not** match
-GitHub's auto-generated source archive. A release without the attached sdist silently
-breaks those consumers.
+Pushing a `v*` tag runs `.github/workflows/release.yml`. Its `release` job runs
+`uv build` and creates the GitHub release with the wheel **and** the sdist
+attached. The job lists both by explicit pattern, not `dist/*`, which once
+shipped a stray `.gitignore` as an asset. Its `publish` job builds again and
+publishes to PyPI by OIDC trusted publishing, so no token is stored anywhere. The
+two jobs build separately, which makes the PyPI files and the release assets two
+builds of the same tag. Keep the attached sdist: at least one consumer updates
+itself with `gh release download --pattern '*.tar.gz'`, and that pattern does not
+match GitHub's auto-generated source archive.
+
+The repository was recreated, so check three things before its first tag push:
+Actions is enabled, the `pypi` environment exists, and the PyPI trusted
+publisher still names owner `sblattj`, repository `prxref`, workflow
+`release.yml` and environment `pypi`.
## Verified at release
```
-840 passed uv run pytest
-All checks passed! uv run ruff check src/ tests/
-0.5.0 uv run prxref --version
-bitbucket-server .../projects/PROJ/repos/app/pull-requests/42
-bitbucket https://bitbucket.org/ws/app/pull-requests/7
-github https://github.com/o/r/pull/3
-gitlab https://gitlab.com/o/r/-/merge_requests/9
-make_forge(ref) -> prxref.forges.bitbucket_server.ForgeImpl, name 'bitbucket-server'
+4222 passed uv run pytest -q
+All checks passed! uv run ruff check src tests
+0.14.0 uv run prxref --version
```
-The counts are the ones the v0.5.0 release run produced; the commands are written
-in today's form. That run spelled them `uv run --extra dev …`, which was correct
-before the dev tools moved to `[dependency-groups]` and does not work now.
+These counts come from the release tip, after every feature, fix and test
+branch had merged.
+
+### Live checks
+
+All ran on 2026-09-23 and were read-only: reviews ran with `--no-post` or with
+the forge's write methods captured, and a guard blocked forge writes. The
+targets were psf/requests#6963 on GitHub, a PR in a public Azure DevOps
+Services project (read anonymously), merge requests in the gitlab-org group on
+gitlab.com, and the bundled eval cases.
+
+- **#1 litellm without `PRXREF_LLM_BASE_URL`.** It answered with a cost and no
+ configuration error. With a URL set, it logged the "set but not used" INFO
+ line.
+- **#2 Azure DevOps.** The forge-level check passed 43 of 43. The dry-run JSON
+ key set equalled the GitHub control's in 3 of 3 runs. The first two lost
+ chunks to LLM timeouts and a malformed model reply, and the third reviewed
+ every chunk. The compare range matched `get_diff` byte for byte, including
+ two PRs one commit behind their target. A pinned-range replay passed 25 of
+ 25.
+- **#3 Team review rules.** 19 of 19 checks passed: the prompts, the run
+ record's hash and severity map, truncation, and exit 2 on a bad path.
+- **#4 Ticket context and scope.** Passed after the scope fix, which shows a
+ `"scope"` key in the prompt's JSON example while a ticket is active. On the
+ case-002 eval, gpt-4.1-mini went from 0 of 10 findings labelled to 12 of 12,
+ and prompts without a ticket stayed byte-identical. With an off-ticket file
+ added, 5 of 5 off-ticket findings came back `out`. The example's value does
+ not anchor the answer: with the example set to `out`, 12 of 12 on-ticket
+ findings still came back `in`.
+- **#5 Replay.** The mechanics passed: the stamp, the pinned diff, the hidden
+ threads, a forge-less `--diff-file` run with no token, and exit 2 on 6 of 6
+ bad flag sets. A replay pinned to a different PR's range is model-fragile:
+ the PR's current title and description are kept, as documented, and the
+ model returned non-JSON in 3 of 3 tries, while the matching control was
+ approved in 2 of 2.
+- **#6 Subscription CLI backends.** `claude-cli` ran on the subscription login,
+ with no `ANTHROPIC_API_KEY` in the child's environment and `--effort low`
+ passed through. `kiro-cli` read "cost unknown", logged its credits, and
+ honoured the model in its agent file. The `(API-equivalent)` cost label
+ landed after this check. It is covered by `tests/test_issue_67_cost.py` and
+ was not re-run live.
+- **#7 Dollar cost.** A reported cost matched the sum of its units, a
+ cost-stripping relay gave `null`, a price table's estimate matched the
+ formula, and a malformed table exited 2. The `-v` line showed `$…`,
+ `~$… (est.)` and `cost unknown`.
+- **#8 Size advisory.** 4 of 4 checks passed, and the parsed diff's counts
+ equalled the GitHub API's.
+- **Spec grounding.** On the bundled eval, grounded against ungrounded runs
+ surfaced 8 of 10 against 3 of 10 planted spec violations on gpt-4.1-mini,
+ and 10 of 10 against 4 of 10 on claude-haiku-4.5.
+- **GitLab paging.** A 130-file MR was read across 2 pages, and `too_large`
+ files were listed header-only with a warning.
+- **Spend.** About $0.31 of LLM calls in total. That is an upper bound,
+ because it counts calls that timed out before reporting a cost at their
+ largest possible cost. The `claude-cli` runs
+ (about $0.09 API-equivalent) and the `kiro-cli` runs (about 0.09 credits)
+ used subscriptions and are not included.
## Still open — not part of this release
-- **Observability / tracing** — prompt+response tracing and a `PRXREF_TRACE_DIR`
- are **done** — they landed as the `review --trace-dir` flag and the
- `PRXREF_TRACE_DIR` env var. Still open from this item: per-finding drop reasons
- and a machine-readable run report.
-- ~~**`review --timeout SECONDS`** — the per-run counterpart to `PRXREF_LLM_TIMEOUT`,
- which shipped natively in v0.4.0.~~ **Done** — the flag has landed.
-
-The next three came out of the v0.5.0 release review, which raised them against
-`bitbucket_server.py`. Each one is real, and each one is **repo-wide, not a porting
-defect**: the new adapter does what its siblings already do, so all three were left
-alone rather than fixed in one adapter and creating a four-way inconsistency. Whoever
-takes one on should change all four adapters in the same commit.
-
-- **Retries re-send non-idempotent writes.** All four retry sessions list `POST` in
- `allowed_methods` against `status_forcelist [429, 500, 502, 503, 504]` with
- `total=3` — `bitbucket_server.py:62-65`, `bitbucket.py:33-36`, `gitlab.py:33-36`,
- `github.py:27-29` (which adds `PATCH`). If a comment POST commits server-side and
- the response is lost to a 502/504 or a read timeout, urllib3 re-sends it and the
- comment is duplicated. The fix is to drop the write verbs from `allowed_methods`
- and let the caller decide, but it changes retry behaviour for every forge.
-- **Comment listings are capped and the cap is silent.** Bitbucket Server reads
- 5 x 100 activities (`bitbucket_server.py:17-18,267`) and Bitbucket Cloud 5 x 100
- comments (`bitbucket.py:216-220`); GitLab reads a single page of 50
- (`gitlab.py:232,321`) and GitHub a single unparameterised page, so 30
- (`github.py:124,167`). Past the cap `list_threads` under-reports and `post_summary`
- can miss its own `` and post a second summary. Server is the
- *most* thorough of the four here, not the least. A shared paging helper with an
- explicit "truncated" signal would fix all four at once.
-- **A failed comment-listing read is treated as "no summary exists".** In
- `post_summary`, a listing that errors leaves the existing-summary handle unset and
- control falls through to the create-a-new-comment POST:
- `bitbucket_server.py:310-311` then `:324`, and `gitlab.py:241-242` then `:254`.
- GitHub has no `try` at all — a transport error propagates — but a non-`ok`
- response takes the same fall-through (`github.py:125` -> `:139`). Bitbucket Cloud
- is furthest from correct: `post_summary` (`bitbucket.py:165`) never looks for an
- existing summary, so it posts a duplicate on *every* re-review. Distinguishing
- "read failed" from "nothing found" and skipping the post is the fix, and it is the
- same three-line change in each adapter.
-- ~~**Move the dev tools to a dependency group.**~~ **Done** — the dev tools moved
- from `[project.optional-dependencies] dev` to `[dependency-groups] dev`, so the bare
- `uv run pytest` and `uv run ruff check` work on a cold checkout and the `--extra dev`
- form is gone from CI and every doc. The behaviour change was accepted deliberately:
- dependency groups are not published in package metadata, so `pip install prxref[dev]`
- no longer resolves the tools. See the `Unreleased` section of `CHANGELOG.md`.
-
-- **`origin/feat/bitbucket-server-forge` can be deleted** once you are satisfied with
- v0.5.0. Everything worth keeping from it is on `main`; the rest is v0.2.0-era text.
-- **`CONTRIBUTING.md` has no inbound link any more.** Deleting the
- `### Bitbucket Server / Data Center (unsupported)` section from `docs/forges.md`
- removed the docs' only pointer to it. The file still exists and GitHub surfaces it
- natively, so nothing is broken — but nothing points at it either.
+The known limitations, in full in the CHANGELOG:
+
+- **Spec digest.**
+ - Unpunctuated keyword lines in one paragraph merge into one constraint, and a
+ run of them past 400 characters is cut.
+ - A spec directory reads only its first 20 files. It names the files it skips
+ in a log line only, and its constraints carry the directory's name rather
+ than each file's.
+ - A Jira ticket's fixed 6,000-character share can crowd out the rest of the
+ spec.
+ - A `spec` finding's quote is not checked against the digest.
+ - A `Spec: "…"` quote that never closes is barely exempt from the hedge gate.
+ - A Jira ticket passed with `--spec` sets no finding's scope.
+ - Grounding can crowd out a generic finding. In the bundled eval, case-002's
+ one expected non-spec finding was missed in 4 of 4 grounded runs across
+ two models and found in 3 of 4 ungrounded runs. That is two runs per model
+ in each arm, and the cause is inferred, not proven.
+- **`kiro-cli`.** It always reports "cost unknown". Whether user-level Kiro
+ configuration such as `~/.kiro/steering/` reaches prxref's per-call agent has
+ not been verified. The model that actually ran is not reported, and every chat
+ is kept under `~/.kiro/sessions/cli/`.
+- **GitLab.** Files GitLab withholds as `too_large` or `collapsed` are listed
+ header-only, and an MR past 5,000 files fails. Reading an MR's threads on
+ gitlab.com needs `PRXREF_GITLAB_TOKEN` even for a public project; without one,
+ thread dedup runs against no threads.
+- **GitHub.** A pull request whose diff runs past 20,000 lines gets HTTP `406`
+ `too_large` from the diff endpoint and ends as an `Error` run. The 0.14.0
+ release PR itself, at about 32,600 changed lines, hit this in CI. The fix is
+ a fallback that rebuilds the diff from the paged `/pulls/{number}/files`
+ listing, as the GitLab adapter does, with files GitHub sends without a
+ `patch` listed header-only.
+- **Replay.** A `--diff-file` run without `--pr-url` sees only the diff.
+- **Azure DevOps.** Only anonymous reads are verified live: the forge reads, the
+ dry-run output shape, the pinned-range compare diff and a pinned-range replay.
+ Posting, pruning, PAT and `SYSTEM_ACCESSTOKEN` authentication and service hooks
+ are tested against recorded API shapes only. Azure DevOps Server is untested.
+- **Gating.** A mistyped or unrecognized `--pr-url` exits 0 even under
+ `PRXREF_FAIL_ON=error` or `any`, because nothing was reviewed.
+
+Follow-ups a seat reported that did not land:
+
+- **Eval scoring is manual.** `tests/evals/test_eval_replay.py` replays every
+ case offline, but scoring (section 7.2 of `docs/spec-grounded-review.md`)
+ is not built: judging findings against a case's `expected.json` needs a live
+ model and stays manual.
+- **One seam is tested only in halves on two paths.** The size advisory and
+ the cost label are tested together on the main summary post
+ (`tests/test_release_seams.py`), but on the inline-accounting refresh post
+ and on the summary-only run each is tested alone.
+- **Scope labelling is measured on one fixture shape.** The live check added an
+ off-ticket file in its own directory to the case-002 eval, and ran three
+ times for each example value on one model, plus once more on a second model.
+ An off-ticket change inside an on-ticket file is unmeasured.
+
+The v0.5.0 handoff left three forge-wide items open. All three are **fixed**:
+
+- Every retry session allows only `GET`, `HEAD` and `OPTIONS`.
+- A comment listing that fails or comes back short raises `FeedReadError`
+ instead of passing for "no summary exists".
+- Every forge, Bitbucket Cloud included, finds its own summary by
+ `SUMMARY_MARKER` and updates it in place.
| Item | Value |
|---|---|
-| Released version | `0.5.0` (minor — new forge plus a webhook fix, nothing breaking) |
-| Registration points | the tuple in `forges/base.py`, the `impls` dict in `config.py` |
+| Released version | `0.14.0` (minor: new inputs, a forge and two backends; an unrecognized `PRXREF_LLM_BACKEND` now exits 2) |
+| Registration points | forges: the tuple in `forges/base.py` (`detect_forge`) and the `impls` dict in `config.py` (`make_forge`); LLM backends: `llm_backends.BACKENDS`; glyphs: `prxref.markers` |
| Version strings | `pyproject.toml`, `src/prxref/__init__.py`, and `uv.lock` |
| Test command | `uv run pytest` (dev tools are a `[dependency-groups]` group, not an extra) |
-| Release assets | sdist **and** wheel, both attached |
+| Release assets | wheel **and** sdist attached by `release.yml`; PyPI by OIDC trusted publishing |
diff --git a/README.md b/README.md
index ba02865..a660fc0 100644
--- a/README.md
+++ b/README.md
@@ -1,8 +1,8 @@
# prxref
-Fast automated AI code review for Bitbucket, GitLab, and GitHub — Cloud and self-hosted.
+Fast automated AI code review for Bitbucket, GitLab, GitHub, and Azure DevOps — Cloud and self-hosted.
-prxref inspects pull and merge requests across the three major code hosting forges in sub-minute review cycles. It parses unified diffs, partitions changes into risk-ranked chunks, gives each worker the dependency pins and out-of-hunk definitions its chunk references when the forge can serve file content, fans out parallel single-shot LLM reviews across a cheap-first model fallback chain, filters findings through deterministic quality gates, and publishes inline comments alongside an executive summary.
+prxref reviews pull and merge requests on Bitbucket, GitHub, GitLab, and Azure DevOps in sub-minute review cycles. It parses unified diffs, partitions changes into risk-ranked chunks, gives each worker the dependency pins and out-of-hunk definitions its chunk references when the forge can serve file content, fans out parallel single-shot LLM reviews across a cheap-first model fallback chain, filters findings through deterministic quality gates, and publishes inline comments alongside an executive summary. Give it the spec or ticket a change implements with `--spec` (a web page, a local file or directory, or a Jira ticket URL) and the review also checks the diff against that spec.
```
┌──────────────────────┐
@@ -17,7 +17,9 @@ prxref inspects pull and merge requests across the three major code hosting forg
▼
┌──────────────────────┐
│ Forge Adapter │
- │ (BB / GitHub / GL) │
+ │ (GitHub / GitLab / │
+ │ Bitbucket Cloud / │
+ │ BB Server / ADO) │
└──────────┬───────────┘
│
▼
@@ -54,7 +56,9 @@ prxref inspects pull and merge requests across the three major code hosting forg
Not every finding comes from a model, and no finding posts unfiltered. prxref
computes one class of finding directly from the parsed diff — the
release-shaped-PR check — and then runs every finding, model-authored or not,
-through eleven deterministic passes: location validation, `package.json` claim
+through the team severity map (only when the review rules declare one) and
+spec grounding, two passes that relabel a severity and drop nothing, and then
+through eleven more deterministic passes: location validation, `package.json` claim
checks, line alignment, thread dedup, settled-thread suppression, severity
consistency, the removal-claim check, the hedge gate, the quality gate, sweep
dedup, and the containment note. A filtered finding is never discarded
@@ -64,6 +68,26 @@ silently — it is kept with a `drop_reason` for the run log, and visible in a
The passes, the checks, every `drop_reason` string, and which of them have a
knob: [docs/quality.md](docs/quality.md).
+## PR Size Advisory
+
+A team that keeps PRs small can set `PRXREF_SIZE_WARN_LINES` (lines added plus
+removed) and/or `PRXREF_SIZE_WARN_FILES` (files changed). A PR above either
+threshold gets one line at the top of its summary, such as `This PR changes 812
+lines in 24 files, above the team guideline of 500 lines and 20 files. Consider
+splitting it.` The line names only the limits that were exceeded. Both
+thresholds are unset by default, which turns the advisory off; `0` is a real
+threshold that flags any change at all. The counts come from the parsed diff and
+skip the common ecosystems' lockfiles (`package-lock.json`, `uv.lock`,
+`Cargo.lock`, `go.sum`, …), generated files (`*.snap`, `__snapshots__/`, `*.min.js`, `*.map`,
+`*.generated.*`, `*.auto.*`), and any path matching `PRXREF_SIZE_IGNORE_GLOBS`,
+which adds to those built-ins and never replaces them. A binary file counts as one
+file and zero lines, so the line count is a lower bound when a forge omits a
+file's hunks. The advisory is not a finding: it never changes the verdict or the
+exit code, and with `--no-post` or `PRXREF_POST_MODE=inline` it appears only in
+the run record, under `--format json` as `size_advisory`, and as a
+`size advisory:` line in the CLI output. See
+[docs/env-vars.md](docs/env-vars.md) for the glob syntax.
+
## Quickstart
Run reviews instantly without local installation using `uvx`, or install the CLI globally:
@@ -80,6 +104,8 @@ prxref review --pr-url https://github.com/org/repo/pull/123
uv tool install git+https://github.com/sblattj/prxref
```
+To work on prxref itself (development setup, tests, and lint), see [CONTRIBUTING.md](CONTRIBUTING.md).
+
### Review Any Forge
Pass any PR or MR URL directly. Forge type, repository namespace, and pull request ID are detected automatically:
@@ -99,13 +125,20 @@ prxref review --pr-url https://github.com/owner/repository/pull/108
# GitLab & Self-Hosted GitLab (including nested subgroups)
prxref review --pr-url https://gitlab.com/group/subgroup/project/-/merge_requests/15
+
+# Azure DevOps Services (dev.azure.com or the legacy *.visualstudio.com host)
+prxref review --pr-url https://dev.azure.com/organization/project/_git/repository/pullrequest/42
+prxref review --pr-url https://organization.visualstudio.com/project/_git/repository/pullrequest/42
+
+# Azure DevOps Server (on-prem; the URL names the collection and the project)
+prxref review --pr-url https://ado.corp.example/tfs/DefaultCollection/project/_git/repository/pullrequest/42
```
-**Supported hosts.** Every forge is supported on any host. GitHub Enterprise Server and self-hosted GitLab share one adapter each with their SaaS products, which speak the same REST API at a different base URL. Bitbucket does not: Server / Data Center speaks `/rest/api/1.0` against different resource shapes, so it is a separate adapter selected automatically from the URL — `PRXREF_BITBUCKET_SERVER_TOKEN` for Data Center, `PRXREF_BITBUCKET_TOKEN` for Cloud. See [docs/forges.md](docs/forges.md).
+**Supported hosts.** Every forge is supported on any host. GitHub Enterprise Server and self-hosted GitLab share one adapter each with their SaaS products, which speak the same REST API at a different base URL. Bitbucket does not: Server / Data Center speaks `/rest/api/1.0` against different resource shapes, so it is a separate adapter selected automatically from the URL — `PRXREF_BITBUCKET_SERVER_TOKEN` for Data Center, `PRXREF_BITBUCKET_TOKEN` for Cloud. Azure DevOps Services and Server share one adapter. It has no diff endpoint to call, so it rebuilds the PR's diff from the changed files; a public project can be reviewed with no token at all. Posting to Azure DevOps is not yet verified against a live server, and Azure DevOps Server is untested. See [docs/forges.md](docs/forges.md).
## LLM Configuration
-prxref operates without direct cloud provider SDK keys (no Anthropic API keys). It ships with **no default endpoint and no default model chain**: point it at any OpenAI-compatible `/chat/completions` server (OpenRouter, Together, Groq, vLLM, Ollama, a self-hosted gateway), or install the optional `litellm` extra. `PRXREF_LLM_BASE_URL` and `PRXREF_LLM_MODELS` are required — leaving either unset exits `2` with an error naming the variable.
+prxref operates without direct cloud provider SDK keys (no Anthropic API keys). It ships with **no default endpoint and no default model chain**: point it at any OpenAI-compatible `/chat/completions` server (OpenRouter, Together, Groq, vLLM, Ollama, a self-hosted gateway), install the optional `litellm` extra, or run the `claude` or `kiro-cli` CLI you are already logged in to. `PRXREF_LLM_MODELS` is required on every backend and `PRXREF_LLM_BASE_URL` on `openai-compat`; leaving a required one unset exits `2` with an error naming the variable.
```bash
# Default backend: plain HTTP to any OpenAI-compatible endpoint
@@ -120,11 +153,18 @@ export PRXREF_LLM_MAX_TOKENS=4096 # raise this if you raise the eff
# pip install 'prxref[litellm]'
export PRXREF_LLM_BACKEND=litellm
export PRXREF_LLM_MODELS="openrouter/meta-llama/llama-3.3-70b-instruct,bedrock/anthropic.claude-3-7-sonnet-20250219-v1:0"
+
+# Optional: your own logged-in Claude Code CLI, on your own machine
+export PRXREF_LLM_BACKEND=claude-cli
+export PRXREF_LLM_MODELS="sonnet"
+export PRXREF_LLM_TIMEOUT=120 # each call includes CLI start-up
```
+`claude-cli` and `kiro-cli` run the CLI already installed and logged in on your machine, on your subscription and for your own use only. Do not use them for a team, a shared webhook, or CI; use an API key through `openai-compat` or `litellm` there. See [Subscription CLI backends](docs/llm.md#subscription-cli-backends-claude-cli-and-kiro-cli).
+
On a reasoning model the hidden reasoning trace draws from the **same** completion budget as the answer, so turning `PRXREF_LLM_REASONING_EFFORT` up makes truncation *more* likely. A truncated chunk is counted as failed and the posted summary names the reason and the variable to raise; see [Reasoning models and the token budget](docs/env-vars.md#reasoning-models-and-the-token-budget).
-Temperature `0.0` and a sampling `seed` are sent on every call — `PRXREF_LLM_SEED` when set, else one random seed per process shared by the whole run (issue #56) — but neither makes a review bit-reproducible — provider fingerprints, load-balanced backends, and gateways that ignore `seed` all still vary the model's output. Everything downstream of the model is deterministic: findings are ordered by `(file, line, title)` and the caps break ties by content, and the run record's `sampling` field reports which knobs were in force. See [Determinism](docs/llm.md#determinism-what-is-pinned-and-what-still-varies).
+On `openai-compat` and `litellm`, temperature `0.0` and a sampling `seed` are sent on every call — `PRXREF_LLM_SEED` when set, else one random seed per process shared by the whole run (issue #56) — but neither makes a review bit-reproducible — provider fingerprints, load-balanced backends, and gateways that ignore `seed` all still vary the model's output. The CLI backends send neither, and the run record's `sampling` field shows both as `null`. Everything downstream of the model is deterministic: findings are ordered by `(file, line, title)` and the caps break ties by content, and the run record's `sampling` field reports which knobs were in force. See [Determinism](docs/llm.md#determinism-what-is-pinned-and-what-still-varies).
See [docs/llm.md](docs/llm.md) for architecture, failover behavior, and backend setup, and [docs/env-vars.md](docs/env-vars.md#tuning-for-your-team) for tuning the confidence floor and finding caps to your team.
@@ -141,29 +181,172 @@ Configure the authentication token matching your forge:
| **GitHub** | `PRXREF_GITHUB_TOKEN` | Personal Access Token (PAT) or GitHub App token |
| **GitHub Enterprise** | `PRXREF_GITHUB_ENTERPRISE_TOKEN` | Used when host is not `github.com` (falls back to `PRXREF_GITHUB_TOKEN`) |
| **GitLab** | `PRXREF_GITLAB_TOKEN` | Personal, project, or group access token (`PRIVATE-TOKEN`) |
+| **Azure DevOps** | `PRXREF_AZURE_DEVOPS_TOKEN` | Personal access token: Code (Read) to review, Code (Read & write) to post |
+| **Azure DevOps (Pipelines)** | `SYSTEM_ACCESSTOKEN` | The job token, used when no PAT is set; map it into the step with `env: SYSTEM_ACCESSTOKEN: $(System.AccessToken)`. With neither set, public projects are read anonymously |
See [docs/env-vars.md](docs/env-vars.md) for the full configuration reference, [docs/forges.md](docs/forges.md) for forge specifics, [docs/quality.md](docs/quality.md) for the deterministic checks and every drop reason, and [docs/systemic-sweep.md](docs/systemic-sweep.md) for the whole-PR sweep's digest classes.
## Webhook Server
-Run prxref as a persistent daemon to handle webhook events from GitHub, Bitbucket, and GitLab:
+Run prxref as a persistent daemon to handle webhook events from GitHub, Bitbucket, GitLab, and Azure DevOps:
```bash
prxref serve --port 8080 --host 0.0.0.0
```
The service exposes:
-- `POST /webhook` — verifies HMAC or token signatures per forge, enqueues incoming PR events, and responds immediately with `202 Accepted`. A background worker processes reviews serially.
+- `POST /webhook` — verifies HMAC or token signatures per forge (for Azure DevOps service hooks, the Basic-auth password against `PRXREF_AZURE_DEVOPS_WEBHOOK_SECRET`), enqueues incoming PR events, and responds immediately with `202 Accepted`. A background worker processes reviews serially. Registering each forge's webhook: [docs/deploy.md](docs/deploy.md#2-webhook-registration).
- `GET /health` — liveness probe returning `{"ok": true}`.
+## Review Against a Spec or Ticket
+
+Give prxref the spec or ticket a change implements, and the review also checks the diff against it. A source is a public web page, a local file, a local directory (up to 20 `.md`, `.markdown`, `.txt` or `.adoc` files directly inside it), or a Jira ticket URL:
+
+```bash
+prxref review --pr-url https://github.com/org/repo/pull/123 \
+ --spec https://jira.example.com/browse/PROJ-42 \
+ --spec https://spec.example.com/client-guidelines.html \
+ --spec /etc/prxref/specs/
+```
+
+prxref fetches each source and keeps its RFC 2119 statements (MUST, SHOULD, MAY), version pins and naming rules; a Jira ticket's summary and description are kept line by line, up to 6,000 characters, and ranked first. It ranks the rest against the diff and adds that bounded digest (`PRXREF_SPEC_DIGEST_TOKENS`) to every chunk worker's prompt and to the whole-PR sweep's, with no extra model call. A finding whose only basis is one of those constraints is a 🔍 `spec` finding, and it quotes the constraint as `Spec: "…"`. `PRXREF_SPEC_SOURCES` sets the sources for every run, the webhook daemon included; `--spec` replaces that list for one run.
+
+- **Jira.** A public ticket needs no configuration. For a private one set `PRXREF_JIRA_BASE_URL`, `PRXREF_JIRA_EMAIL` and `PRXREF_JIRA_API_TOKEN`: credentials only go to `PRXREF_JIRA_BASE_URL`, and every other fetch is anonymous.
+- **Advisory.** 🔍 spec findings are advisory: they never change the verdict, and `PRXREF_FAIL_ON=error` ignores them. `PRXREF_FAIL_ON=any` is the opt-in gate.
+- **Best-effort.** A source that cannot be fetched never fails the review. The summary gains a grounding note that counts the constraints injected and names each failed source by its position and kind (`source 2 (url)`), never by its path or URL. When the digest ends up with no constraint at all, the review runs as if no spec had been given, and a `spec` finding the model emits anyway is relabelled `warning`.
+
+Every setting: [docs/env-vars.md](docs/env-vars.md). How grounding meets the quality passes: [docs/quality.md](docs/quality.md#spec-grounding). Spec sources in CI and on the daemon, fetch time bounds, and what the logs record: [docs/deploy.md](docs/deploy.md#7-spec-sources-in-ci-and-on-the-daemon).
+
+## Ticket Context and Scope
+
+Give prxref the ticket a PR is meant to implement, and every finding is marked `in`, `out`, or `unknown` against that ticket's scope:
+
+```bash
+prxref review --pr-url https://github.com/owner/repository/pull/108 --context-file ticket.md
+
+# or for every run
+export PRXREF_TICKET_CONTEXT_FILE=ticket.md
+```
+
+The file is plain text or Markdown: the ticket's title, description, and acceptance criteria, fetched from your tracker by a CI step. It must be a local file, and `prxref review` reads it before any network call. A URL, a missing file, a directory or other non-regular file, an unreadable file, or a file that is not UTF-8 is a configuration error: the run exits `2`, and the message names `--context-file` or `PRXREF_TICKET_CONTEXT_FILE`, whichever supplied the path. A path under the working directory that symlinks out of it is refused the same way. `--context-file PATH` wins over the variable for one run, and `--context-file ""` turns it off. Read the file from a trusted checkout or your CI, never from the PR under review, or the PR's author writes the ticket their change is judged against.
+
+A configured ticket is in one of three states:
+
+| File | Prompts | Summary note |
+|---|---|---|
+| Empty or whitespace only: "this PR has no ticket" | unchanged | `No ticket context for this PR — findings were not checked against a ticket's scope.` |
+| Text without acceptance criteria | ticket and scope ask added | `The ticket context has no acceptance criteria — scope was judged from its description alone.` |
+| Text with acceptance criteria | ticket and scope ask added | none |
+
+Acceptance criteria are recognized by any one of these: a heading or label standing alone on its line (`Acceptance criteria`, `Acceptance test(s)`, or `Definition of done` in any case, or `AC` in capitals, optionally as a `#` heading, in bold, or with a trailing colon), a Markdown task-list item (`- [ ] …` or `- [x] …`), or a Gherkin `Given` line followed later by a `Then` line.
+
+**What each finding's `scope` means.** `in`: the finding concerns what the ticket asks for, including code that visibly contradicts one of its acceptance criteria. `out`: it concerns a change the ticket does not ask for, such as an unrelated refactor or a drive-by edit. `unknown`: the ticket and the diff do not let the model tell. Without a ticket, or with an empty one, every finding is `unknown`, and so is any answer from the model other than exactly one of those three words. Scope is advisory only: it never changes a finding's severity or confidence, the verdict, the error cap, or `PRXREF_FAIL_ON`. The `outofscope` severity is unrelated and only means minor. How a scope shows on a posted comment is covered in [Finding Markers](#finding-markers).
+
+**How the ticket reaches the model.** The ticket text goes into the user prompt of every chunk worker and of the whole-PR sweep, under a `### Ticket context` heading. It sits inside a code fence it cannot close, with a line telling the model that it is data, not instructions. The request to add a `scope` to every finding is prxref's own policy, so it goes into the system prompt instead. While that request is in the prompt, the example finding under `## Output Format` in the worker and sweep prompts also carries `"scope": "in"`, because a model that copies the example rather than following the instruction would otherwise never label scope; without a ticket, or with an empty one, the prompts are unchanged. `PRXREF_TICKET_CONTEXT_MAX_CHARS` (default `6000`) caps the text. A longer ticket is cut, and a line after the fence says how many of its characters are shown. Criteria past the cap are not in view, so they do not count toward the state above.
+
+**What is recorded.** The `ticket_context` key of `--format json` and the run's `ticket ok` trace event hold the path, the SHA-256 of the file's raw bytes, its length in characters, the cap, whether it was truncated, whether it has acceptance criteria, and whether it was empty. The `-v` line shows the path, the start of the SHA-256, the length, and the active findings' `in`/`out`/`unknown` counts. None of these ever holds the ticket text. The text does appear in the prompt files that `--trace-dir` writes (`chunk0.user.md`, `sweep.user.md`), so treat that directory like the ticket itself.
+
+**The webhook daemon ignores the file.** One file cannot describe every PR a daemon sees, so `prxref serve` never reads `PRXREF_TICKET_CONTEXT_FILE` and logs a warning once at startup when it is set.
+
+## Team Review Rules
+
+Give prxref your team's review checklist and every chunk worker and the whole-PR sweep review against it:
+
+```bash
+prxref review --pr-url https://github.com/acme/widget/pull/42 --rules-file "$RUNNER_TEMP/prxref-rules.md"
+```
+
+- `--rules-file PATH`, or `PRXREF_REVIEW_RULES` for every run, names a Markdown or plain-text file. Its body is added to the **system** prompt of every review unit under a `## Team review rules` heading. The chunk workers check their chunk against it, and the sweep applies only the whole-PR and cross-file rules. Unset, nothing changes.
+- Optional front matter can map your team's severity words onto prxref's tiers in a `severity:` block (`blocker: error`, `major: warning`, `nit: outofscope`). A mapped word the model writes anyway is rewritten before every quality pass, so it is never dropped as an invalid severity. Other front-matter keys are ignored, so a skill file works unmodified.
+- `PRXREF_REVIEW_RULES_MAX_CHARS` (default `12000`) caps the body, with a warning when it truncates. The run record's `review_rules` carries the file's `sha256`, its character count, and the parsed map, never the rules text. It appears in `--format json`, the `-v` output, and the JSONL trace.
+- A missing, unreadable, or malformed file exits `2` before any network call, naming `--rules-file` or `PRXREF_REVIEW_RULES`.
+
+**Read the rules from a trusted checkout, never from the PR under review.** In CI the workspace is usually the PR's own code, so a rules file inside it lets the PR rewrite its own review rules. Copy the file from the target branch or keep it outside the repository. See [docs/review-rules.md](docs/review-rules.md) for the grammar, the CI recipes, and the daemon.
+
+## Finding Markers
+
+Each severity has one glyph. It is the same in the summary's counts line, the summary's findings list, and the header of every inline comment:
+
+| Marker | Severity | Meaning |
+|---|---|---|
+| 🟥 | `error` | The change breaks at runtime or is a real bug. |
+| 🟧 | `warning` | A risk or smell the diff introduces or worsens. |
+| 🔍 | `spec` | The diff contradicts a constraint quoted from a spec source. See [Review Against a Spec or Ticket](#review-against-a-spec-or-ticket). |
+| ⬜ | `outofscope` | Minor: misleading naming, a TODO without context, dead code the diff adds. An unrecognised severity also renders ⬜. |
+
+🟦 is not a severity. It marks a finding that the [ticket context](#ticket-context-and-scope) puts outside the ticket (`scope` is `out`), and it goes in front of the severity glyph, never in place of it:
+
+- **Summary:** those findings are listed after the others, under their own heading, for example `**🟦 Outside the ticket (2)**` followed by ``- 🟦 🟧 `src/app.py:12` — …``. If every finding is outside the ticket, the first list reads `No in-ticket findings.`.
+- **Inline comments:** the header reads, for example, `🤖 🟦 🟧 **[WARNING · OUTSIDE TICKET] …**`.
+- **CLI text output** (`--no-post` or `-v`): the finding line ends in ` [scope: out]`, or ` [scope: in]` for a finding inside the ticket.
+
+Findings inside the ticket (`in`) and findings the reviewer could not place (`unknown`) carry no scope marker, so a run without a ticket context renders exactly the severity glyphs. Scope never changes a finding's severity. It is not counted separately either: the counts line counts every active finding by severity, and the verdict, the error cap, and `PRXREF_FAIL_ON` ignore scope.
+
+Before 0.14.0, `outofscope` findings rendered 🟦. They now render ⬜ on every run, and 🟦 means only "outside the ticket".
+
## CLI Flags
-- `--pr-url URL` — full web URL of the PR or MR (required for `review`).
+`prxref review` takes:
+
+- `--pr-url URL` — full web URL of the PR or MR on Bitbucket, GitHub, GitLab, or Azure DevOps. Required unless `--diff-file` is given.
- `--no-post` — dry run; run review analysis and quality passes without writing comments to the forge. In text mode this also prints every active finding's location, title, and body, and every dropped finding with its drop reason.
- `--max-chunks N` — override maximum diff chunks evaluated (default `8`).
- `--timeout SECONDS` — override the per-model request deadline (default `45.0`, or `PRXREF_LLM_TIMEOUT` when set); the flag wins for the current invocation only.
-- `-v, --verbose` — output run timing, token counts, and finding breakdowns to stdout; in text mode this also prints finding bodies and dropped findings, same as `--no-post`.
-- `--format {text,json}` — output format for `review` (default `text`). `json` prints exactly one JSON object to stdout — `verdict`, `findings` (active first, then dropped, each with `file`, `line`, `severity`, `confidence`, `title`, `body`, `drop_reason`), `chunk_count`, `chunks_reviewed`, `chunks_failed`, `elapsed_ms`, `input_tokens`, `output_tokens`, `posted`, and `sampling` (the `temperature`, `seed`, and `models` the run had in force — every review result carries it).
+- `--spec URL_OR_PATH` — a spec or ticket to review the PR against: a public web URL, a local file or directory, or a Jira ticket URL. Repeatable. When given, the flags replace `PRXREF_SPEC_SOURCES` entirely rather than adding to it. See [Review Against a Spec or Ticket](#review-against-a-spec-or-ticket).
+- `--rules-file PATH` — your team's review rules (Markdown or text, with optional front matter carrying a `severity:` map), added to every review prompt. Overrides `PRXREF_REVIEW_RULES` for this run, and `--rules-file ""` turns an environment-configured file off. Read it from a trusted checkout, never from the PR under review. See [Team Review Rules](#team-review-rules).
+- `--context-file PATH` — the ticket the PR is meant to implement (plain text or Markdown). Every finding is then marked in, out of, or of unknown ticket scope, and an empty file means "this PR has no ticket". Overrides `PRXREF_TICKET_CONTEXT_FILE` for this run, and `--context-file ""` turns it off. See [Ticket Context and Scope](#ticket-context-and-scope).
+- `--trace-dir DIR` — write each review unit's exact prompt halves, raw model response, and metadata to `DIR` (`chunk0.system.md`, `chunk0.user.md`, `chunk0.response.json`, `chunk0.meta.json`, and so on for each chunk and for the whole-PR `sweep`). `PRXREF_TRACE_DIR` does the same for every run; the flag wins when both are set.
+- `-v, --verbose` — output run timing, token counts, cost, and finding breakdowns to stdout, plus one line each for the rules file, the ticket context (with the active findings' scope counts), and the spec sources when they are configured. In text mode this also prints finding bodies and dropped findings, same as `--no-post`.
+- `--format {text,json}` — output format for `review` (default `text`). `json` prints exactly one JSON object to stdout, with these keys in this order:
+ - `verdict`;
+ - `findings`: active first, then dropped, each with `file`, `line`, `severity`, `confidence`, `scope` (`in`, `out`, or `unknown` against the ticket context; always `unknown` without one), `title`, `body`, `drop_reason`;
+ - `chunk_count`, `chunks_reviewed`, `chunks_failed`, `elapsed_ms`, `input_tokens`, `output_tokens`;
+ - `cost_usd`: the run's cost in USD, `null` when no source could price it (never `0` for an unknown cost), and `cost_estimated`: `true` when any part of it came from `PRXREF_PRICE_TABLE`. See [Cost accounting](docs/llm.md#cost-accounting);
+ - `posted`;
+ - `review_rules` (`path`, `sha256`, `chars`, `max_chars`, `truncated`, `severity_map`), `ticket_context` (`path`, `sha256`, `chars`, `max_chars`, `truncated`, `has_acceptance_criteria`, `empty`; never the ticket text), `spec_grounding` (`sources`, `ok`, `failed`, `constraints`, `digest_sha256`), and `size_advisory` (`changed_lines`, `changed_files`, `lines_limit`, `files_limit`, `triggered`, `message`). These four are always present and `null` when their feature is off;
+ - `sampling`: the `temperature`, `seed`, and `models` the run had in force (every review result carries it);
+ - `replay`: the replay stamp (`base_sha`, `head_sha`, `threads`, `diff_file`), on replay runs only.
+
+Replay flags, for evaluation (see [Replay Mode (Evaluation)](#replay-mode-evaluation)). Any of them turns posting off for the run:
+
+- `--base-sha SHA` / `--head-sha SHA` — review the pinned range `BASE...HEAD` of the `--pr-url` repository (the merge-base diff, as the PR's own diff is), with file context read at `HEAD`. The two come as a pair, must be full 40- or 64-character hex commit SHAs, must differ, and need `--pr-url`.
+- `--no-threads` — hide the PR's existing threads from the prompt and from the thread-dedup passes.
+- `--diff-file PATH` — review this unified diff (`git diff` or `git format-patch` output) instead of fetching one; `--pr-url` becomes optional.
+
+The other subcommands: `prxref serve [--port N] [--host H]` runs the [webhook server](#webhook-server) (default port `8080`, default host `0.0.0.0`); `prxref trace render FILE [-o OUT]` renders a JSONL run trace (`PRXREF_TRACE_FILE`) to a standalone HTML pipeline view, written next to the trace unless `-o`/`--out` names the output; and `prxref --version` prints the version.
+
+## Replay Mode (Evaluation)
+
+A replay reviews a pinned, reproducible input instead of a PR as it stands, so one change can be reviewed again later, by another model or another prxref build, and compared. Three invocations cover it:
+
+```bash
+# A blind replay of a PR at two pinned commits, without its existing discussion
+prxref review --pr-url https://github.com/acme/widgets/pull/42 \
+ --base-sha 0123456789abcdef0123456789abcdef01234567 \
+ --head-sha 89abcdef0123456789abcdef0123456789abcdef \
+ --no-threads --format json
+
+# A diff on disk with its ticket and spec corpus; no PR and no forge at all
+prxref review --diff-file change.diff --context-file TICKET.md --spec docs/specs --format json
+
+# One eval case (see tests/evals/README.md)
+prxref review --diff-file tests/evals//diff.patch --context-file tests/evals//ticket.md \
+ --spec tests/evals//docs --no-post --format json
+```
+
+- **A replay never posts.** Any replay flag turns posting off for the run, with or without `--no-post`; when nothing else had already turned it off, the run logs `replay run: posting to the forge is disabled`. The replay forges also refuse every write, and a replay never prunes older comments.
+- **Pinned SHAs:** `--base-sha` and `--head-sha` come as a pair, must be full 40- or 64-character hex commit SHAs (resolve a short one with `git rev-parse`), are lowercased, must name two different commits, and need `--pr-url`. The review reads the merge-base diff `BASE...HEAD` and file context at `HEAD`; the endpoint each forge uses is under "Pinned Commit Range (Replay)" in [docs/forges.md](docs/forges.md).
+- **`--diff-file PATH`** reviews that file (`git diff` or `git format-patch` output) instead of fetching a diff. Without `--pr-url` nothing is contacted: there are no threads and no file context, and a `git format-patch` file supplies the title, description and author. With `--pr-url` the file replaces the PR's diff, and without `--head-sha` a warning says that file context is still read at the PR's current head.
+- **What a pinned replay does not pin.** The PR's *current* title and description still reach the prompt, and so do its current threads unless you add `--no-threads`; a replay at pinned SHAs without `--no-threads` logs a warning saying so.
+- **The record.** A replay's JSON record gains a `replay` stamp, always with all four keys, and the text summary prints it as a `replay:` line. A normal run's record has no `replay` key.
+
+ ```json
+ "replay": {"base_sha": null, "head_sha": null, "threads": "hidden", "diff_file": "change.diff"}
+ ```
+
+ `threads` is `"hidden"` under `--no-threads` or with no `--pr-url`, else `"shown"`; `diff_file` is the path as you typed it.
+- **Exit codes.** A bad set of replay flags exits `2` naming the flag, and it is checked before the PR URL is parsed. A review error inside a replay — an empty pinned range (a head already merged into the base) or a blank diff file — ends the run as an `Error` run: it exits `0` under the default `PRXREF_FAIL_ON=never`, and `1` under `error` or `any`, like any review that does not complete. See [Exit Codes](#exit-codes).
+- The replay flags have no environment variable, on purpose, and the [webhook server](#webhook-server) never replays.
## Exit Codes
@@ -171,8 +354,8 @@ The service exposes:
| Code | Meaning |
|---|---|
-| `0` | The run finished — **including every review error**: an empty diff, a network failure, an LLM timeout, bad forge credentials, an unrecognized URL, or a review in which every chunk failed. Diagnostics go to stderr; the pipeline step stays green. With `PRXREF_FAIL_ON` set (see below) a finding or a failed review can turn this into `1`. |
-| `1` | **Gated review outcome** — only when `PRXREF_FAIL_ON` is set: `error` exits `1` when the completed review carries an active error-severity finding, `any` exits `1` on any active finding, and under either value a review that fails to complete also exits `1`. The reason is printed to stderr. |
+| `0` | The run finished — **including every review error**: a network failure, an LLM timeout, bad forge credentials, an unrecognized URL, or a review in which every chunk failed. Diagnostics go to stderr; the pipeline step stays green. With `PRXREF_FAIL_ON` set to `error` or `any` (see below), only two outcomes turn this into `1`: a completed review whose active findings trip the policy, and a review that does not complete — it crashes, or it ends with verdict `Error` (the forge could not be read, the diff could not be parsed or chunked, or every chunk review failed). An empty PR diff is not a failure (verdict `Approved`, exit `0`), and an unrecognized URL stays `0` because nothing was reviewed. |
+| `1` | **Gated review outcome** — only when `PRXREF_FAIL_ON` is set: `error` exits `1` when the completed review carries an active error-severity finding, `any` exits `1` on any active finding, and under either value a review that does not complete also exits `1` — it crashes, or it ends with verdict `Error` (the forge could not be read, the diff could not be parsed or chunked, or every chunk review failed). An empty PR diff is not a failure (verdict `Approved`, exit `0`). The reason is printed to stderr. |
| `2` | **Usage or configuration error** — no subcommand, invalid command-line arguments, or a required value missing, malformed, outside its valid range, or outside its key's allowed vocabulary (`PRXREF_FAIL_ON` accepts only `never`, `error`, `any`). The message names the source that supplied it: the environment variable, or the CLI flag when a flag is what you typed. |
```
@@ -180,4 +363,6 @@ $ prxref review --pr-url https://github.com/org/repo/pull/1 --max-chunks 0
configuration error: --max-chunks: must be a finite number greater than 0, got 0
```
+[Replay mode](#replay-mode-evaluation) keeps the same split. A bad set of replay flags exits `2` naming the flag: neither `--pr-url` nor `--diff-file`, a lone `--base-sha` or `--head-sha`, a SHA that is not full 40- or 64-character hex, two equal SHAs, SHAs without `--pr-url`, or an unreadable `--diff-file`. These are checked before the PR URL is parsed, so they exit `2` even next to an unrecognized URL; pinned SHAs on a forge that cannot fetch a commit range also exit `2`, once the forge is known. An empty pinned range or a blank diff file is a review error, unlike an empty PR diff: the run ends as an `Error` run, which exits `0` under the default `PRXREF_FAIL_ON=never` and `1` under `error` or `any`.
+
`PRXREF_FAIL_ON` is the one opt-out of the advisory contract, and its default `never` is the doctrine above, unchanged. Setting it to `error` or `any` turns the reviewer into a merge gate — failing a build on a finding turns a probabilistic reviewer into a gate, and the first false positive teaches a team to bypass the gate, so think hard before you set it. Read the verdict from the posted summary, which also carries a partial-review banner when some chunks did not make it. Do not build a security control on the exit code. The webhook daemon has no exit code and is unaffected.
diff --git a/docs/deploy.md b/docs/deploy.md
index 86c82b2..c4dcc03 100644
--- a/docs/deploy.md
+++ b/docs/deploy.md
@@ -48,13 +48,36 @@ Configure secret tokens and match the events accepted by `prxref`:
| Forge | Secret Env Var | Reviewable Events | Notes |
|---|---|---|---|
| **GitHub** | `PRXREF_GITHUB_WEBHOOK_SECRET` | `Pull request` (actions: `opened`, `synchronize`) | HMAC-SHA256 in `X-Hub-Signature-256` |
-| **Bitbucket Cloud** | `PRXREF_BITBUCKET_WEBHOOK_SECRET` | `Pull Request: Created` (`pr:opened`), `Pull Request: Updated` (`pr:modified`) | HMAC-SHA256 in `X-Hub-Signature` |
+| **Bitbucket Cloud** | `PRXREF_BITBUCKET_WEBHOOK_SECRET` | `Pull Request: Created` (`pullrequest:created`), `Pull Request: Updated` (`pullrequest:updated`) | HMAC-SHA256 in `X-Hub-Signature` |
+| **Bitbucket Server / Data Center** | `PRXREF_BITBUCKET_WEBHOOK_SECRET` (the same secret as Cloud) | `pr:opened`, `pr:modified`, `pr:from_ref_updated` (the source branch moved) | HMAC-SHA256 in `X-Hub-Signature`; the same `X-Event-Key` header as Cloud, told apart by event name and payload shape |
| **GitLab** | `PRXREF_GITLAB_WEBHOOK_SECRET` | `Merge request events` (actions: `open`, `update`) | Secret token in `X-Gitlab-Token` header |
+| **Azure DevOps** | `PRXREF_AZURE_DEVOPS_WEBHOOK_SECRET` | `git.pullrequest.created`, `git.pullrequest.updated`, on a PR whose status is `active` | No event header and no signature: recognized by `publisherId: "tfs"` in the body, authenticated by the HTTP Basic auth password (the user name is ignored). Subscriptions: see [Azure DevOps service hooks](#azure-devops-service-hooks) |
*Note on Insecure Development Bypass:* Setting `PRXREF_ALLOW_UNSIGNED=1` allows unsigned payloads for local testing. Do not use in production.
*First deployment:* set `PRXREF_DRY_RUN=1` before pointing webhooks at a busy repository. The daemon then runs every review in full — fetch, chunk, LLM calls, quality gate — and writes nothing back to the forge, so you can read the logs and confirm the review is sane before it starts commenting. Unset it when you are satisfied. This is the only way to observe the daemon against real traffic: `--no-post` covers a single CLI invocation, and `serve` takes only `--host`/`--port`, so the daemon has no flag-based equivalent.
+### Azure DevOps service hooks
+
+Azure DevOps sends no event header and no signature. prxref recognizes its service hooks by the JSON body (`publisherId: "tfs"`) and authenticates them with HTTP Basic auth: the password must equal `PRXREF_AZURE_DEVOPS_WEBHOOK_SECRET` (compared in constant time), and the user name is ignored. With the secret unset, every Azure DevOps webhook gets `401` unless `PRXREF_ALLOW_UNSIGNED=1`.
+
+In **Project settings → Service hooks**, create two **Web Hooks** subscriptions:
+
+| Subscription | Trigger | Filters |
+|---|---|---|
+| Pull request created (`git.pullrequest.created`) | a PR is opened | repository and target branch, as you like |
+| Pull request updated (`git.pullrequest.updated`) | a PR changes | **Change: Source branch updated** |
+
+Set the **Change** filter on the updated subscription. Without it, every reviewer vote, status change and description edit triggers a full re-review: the receiver cannot tell those updates from a push, so it relies on the subscription to filter them.
+
+On the **Action** page of each subscription:
+- **URL:** `https:///webhook`, with TLS in front of `prxref serve`. Basic auth carries the secret itself, as GitLab's token header does, rather than a signature of the body, so over plain HTTP anyone on the path can read it.
+- **Basic authentication username:** anything, e.g. `prxref`.
+- **Basic authentication password:** the value of `PRXREF_AZURE_DEVOPS_WEBHOOK_SECRET`.
+- **Resource details to send:** **All**. prxref builds the PR URL from `resource.repository` and `resource.pullRequestId`, and reads `resource.status`; a smaller setting can leave them out, and the PR is then not reviewed.
+
+prxref reviews only a PR whose status is `active`. An update that completes or abandons a PR, and every other event type, is acknowledged with `202` and not reviewed. The daemon posts with `PRXREF_AZURE_DEVOPS_TOKEN`: a PAT with **Code (Read & write)**. Posting to Azure DevOps and the service-hook payload are verified against recorded shapes only, not a live server; see [Azure DevOps](forges.md#5-azure-devops-services--server).
+
---
## 3. Non-Docker Deployment (Systemd / Bare Metal)
@@ -128,20 +151,21 @@ The container includes a built-in curl-free health check using Python standard l
| Code | Meaning | Pipeline effect |
|---|---|---|
-| `0` | The run finished. This **includes every review error**: an empty diff, a network failure, an LLM timeout, bad forge credentials, an unrecognized PR URL, or a review in which every chunk failed. Diagnostics are printed to stderr. | Step stays green. |
-| `2` | A **usage or configuration error**: no subcommand, invalid arguments, or a required value missing, malformed, or out of range. The message names the source that supplied it — the environment variable, or the CLI flag when a flag was what the operator typed. | Step fails. This is the intended failure: it means prxref was invoked wrong or is misconfigured, not that your code is bad. |
+| `0` | The run finished. This **includes every review error**: a network failure, an LLM timeout, bad forge credentials, an unrecognized PR URL, or a review in which every chunk failed. Diagnostics are printed to stderr. With `PRXREF_FAIL_ON` set to `error` or `any`, only a completed review whose active findings trip that policy, or a review that does not complete — it crashes, or it ends with verdict `Error` (the forge could not be read, the diff could not be parsed or chunked, or every chunk review failed) — exits `1` instead (next row). An empty PR diff is not a failure (verdict `Approved`, exit `0`). | Step stays green. |
+| `1` | A **gated review outcome**, only when `PRXREF_FAIL_ON` is set: `error` exits `1` when the completed review carries an active error-severity finding, `any` exits `1` on any active finding, and under either value a review that does not complete also exits `1` — it crashes, or it ends with verdict `Error` (the forge could not be read, the diff could not be parsed or chunked, or every chunk review failed). An empty PR diff is not a failure (verdict `Approved`, exit `0`). The reason is printed to stderr. An unrecognized PR URL still exits `0`. | Step fails, because the lane opted in. |
+| `2` | A **usage or configuration error**: no subcommand, invalid arguments, or a required value missing, malformed, out of range, or outside its key's allowed vocabulary (`PRXREF_FAIL_ON` accepts only `never`, `error`, `any`). The message names the source that supplied it — the environment variable, or the CLI flag when a flag was what the operator typed. | Step fails. This is the intended failure: it means prxref was invoked wrong or is misconfigured, not that your code is bad. |
```bash
# A URL prxref cannot review — still exit 0
-$ prxref review --pr-url https://bitbucket.example.com/projects/P/repos/r/pull-requests/42
-unrecognized PR URL '...' — expected bitbucket.org, github.com, or gitlab.com PR/MR link
+$ prxref review --pr-url https://github.com/org/repo/issues/42
+unrecognized PR URL 'https://github.com/org/repo/issues/42' — expected a Bitbucket pull-requests, GitHub pull, or GitLab merge_requests link (bitbucket.org, github.com, gitlab.com, or a self-hosted Bitbucket Data Center, GitHub Enterprise Server, or GitLab host), or an Azure DevOps pullrequest link (dev.azure.com, *.visualstudio.com, or an Azure DevOps Server host); the URL must keep the forge's own path shape.
$ echo $?
0
-# Every chunk failed — still exit 0, and the forge gets an error notice
+# Both chunk workers failed; the sweep answered, but a sweep alone is not a review, so the verdict is Error — still exit 0 under the default PRXREF_FAIL_ON=never, and the forge gets an error notice
$ prxref review --pr-url https://github.com/org/repo/pull/1
verdict: Error
-coverage: 0/3 chunks reviewed
+coverage: 1/3 chunks reviewed
$ echo $?
0
@@ -155,5 +179,77 @@ $ echo $?
Practical consequences for a pipeline:
- **Do not add `continue-on-error` to hide review failures.** They already exit `0`. Suppressing errors instead hides the `2` that tells you the deployment is misconfigured — and a review step that can never fail is a review step nobody notices has stopped running.
-- **Do not gate a merge on the exit code.** There is deliberately no `PRXREF_FAIL_ON`. A probabilistic reviewer used as a gate is worse than no gate: the first false positive teaches the team to bypass it. Read the verdict from the posted summary comment instead.
+- **Do not gate a merge on the exit code.** By default it never gates: `PRXREF_FAIL_ON` defaults to `never`, under which no finding moves the exit code, and `PRXREF_FAIL_ON=error` or `PRXREF_FAIL_ON=any` is the explicit opt-in for a lane that wants the gate (the `1` row above). Think hard before you set it. A probabilistic reviewer used as a gate is worse than no gate: the first false positive teaches the team to bypass it. Read the verdict from the posted summary comment instead.
- **Watch for the partial-review banner.** A run where some chunks failed still exits `0` and still posts a summary; the banner in that summary (and the `coverage: N/M chunks reviewed` line on stdout) is the only signal that the review was incomplete. The most common cause is a starved completion budget — see [Reasoning models and the token budget](env-vars.md#reasoning-models-and-the-token-budget).
+
+---
+
+## 6. CLI Model Backends in Docker and CI
+
+The `claude-cli` and `kiro-cli` backends run the Claude Code CLI or the Kiro CLI that a developer has installed and logged in to on their own machine, on that developer's own subscription. They are not for the Docker image, CI, or the webhook daemon:
+
+- **The Docker image ships neither CLI.** It is `python:3.12-slim` plus the prxref wheel, so `claude` and `kiro-cli` are not on its `PATH`.
+- **Do not install or log in to one in CI or on the daemon.** A pipeline or a shared webhook server reviews for a whole team, and a personal subscription login is for your own use; Anthropic's terms do not allow a third-party product to offer claude.ai login or subscription rate limits without approval. Use an API key through `openai-compat` or `litellm` there, or Workload Identity Federation or Bedrock/Vertex/Foundry. The policy and everything else about these backends is in [Subscription CLI backends](llm.md#subscription-cli-backends-claude-cli-and-kiro-cli).
+
+A CLI backend whose binary cannot be found is a configuration error, so a lane that selects one by mistake fails loudly with exit `2` before any network call instead of posting nothing:
+
+```
+$ PRXREF_LLM_BACKEND=claude-cli PRXREF_LLM_MODELS=sonnet prxref review --pr-url https://github.com/org/repo/pull/1
+configuration error: PRXREF_LLM_BACKEND: claude-cli needs the 'claude' CLI, which was not found on PATH; install it and log in, or set PRXREF_LLM_CLI_PATH to its absolute path
+$ echo $?
+2
+```
+
+---
+
+## 7. Spec Sources in CI and on the Daemon
+
+Spec grounding fetches the sources the operator names in `--spec` or `PRXREF_SPEC_SOURCES` on every review; see the README's [Review Against a Spec or Ticket](../README.md#review-against-a-spec-or-ticket). A pull request cannot add a source. The CLI reads sources only from its flags and environment, and the daemon only from its own environment, never from a webhook payload or a PR description.
+
+### Trust the path, not only the flag
+
+A local spec path inside the PR's own checkout is content the PR controls: the PR can rewrite the constraints it is reviewed against. In CI, point `--spec` at a path outside the workspace, or at a URL.
+
+A relative source is confined to the working directory. A path under the working directory that resolves outside it once its symlinks are followed fails its source (`resolves outside the working directory`), and a directory source skips every symlinked entry in it. An absolute path outside the working directory is the operator's own choice and is read as given.
+
+### On the webhook daemon
+
+- **One list grounds every repository.** `PRXREF_SPEC_SOURCES` applies to every review the daemon runs, and the prompts tell the model to quote a violated constraint verbatim. Private spec text can therefore appear in a comment on any repository the daemon serves. Give the daemon only sources that every one of those repositories may see.
+- **Run it from a directory no PR can change.** Relative sources resolve against the daemon's working directory, so never start it inside a checkout. The systemd unit above uses `WorkingDirectory=/opt/prxref`.
+- **A slow spec host delays the queue.** The daemon reviews one PR at a time, so every source's fetch time is added to every review, and to every review queued behind it. The bounds below cap that cost per source.
+
+### Fetch time bounds
+
+Two module constants in `prxref.specs` bound a fetch. They are not configuration keys, and neither depends on `--timeout` or `PRXREF_LLM_TIMEOUT`:
+
+| Constant | Value | Bounds |
+|---|---|---|
+| `SPEC_FETCH_TIMEOUT_S` | `15` | Each HTTP attempt's connect timeout and its timeout per read. |
+| `SPEC_FETCH_BUDGET_S` | `30` | One source's body, in wall-clock seconds counted from before the request. |
+
+A spec fetch is retried once, with no backoff sleep, and `Retry-After` is ignored: a host that is down or asks for time is skipped, not waited for. The body is read one socket read at a time, so a host that trickles bytes cannot outlast the budget by more than one read timeout. Worst cases per source:
+
+- A host that accepts the connection and never answers costs about 30 s: two attempts of 15 s each.
+- A host that answers and then trickles its body costs about 45 s: the 30 s budget plus one 15 s read timeout.
+
+A web page longer than its byte cap, `4 × PRXREF_SPEC_MAX_CHARS + 4`, is cut with a `[source truncated at N chars]` marker. A Jira response over that cap fails its source.
+
+### What the logs, the run record and the trace say
+
+A failed source never fails the review. The posted grounding note names a failed source only by its position and kind. The operator gets more detail:
+
+- **One WARNING per failed source:** `spec source N/T (kind, origin) failed (best-effort): reason`. `kind` is `file`, `dir`, `url` or `jira`, or `unknown` when the source failed before its kind was known. The reason is redacted the same way as in the posted note. The origin is made safe to log:
+ - a local path is logged verbatim, since it names what to fix;
+ - a URL is cut to `scheme://host[:port]/path`, with no userinfo, query, fragment or `;params`;
+ - a URL that cannot be parsed is logged as `[unparseable origin]`.
+- **One INFO line per run with spec sources:** `spec grounding: OK/T source(s) ok, N constraint(s) injected`. N is 0 when the digest held no constraint and was not injected. If the spec stage itself crashes, one ERROR line replaces it (`spec grounding failed (best-effort): …`), and the run proceeds ungrounded.
+- **The run record's `spec_grounding` key,** also printed by `--format json`: `{sources, ok, failed, constraints, digest_sha256}`.
+ - `failed` lists `source N (kind): `, or `source N: …` when the kind is unknown. A crashed spec stage records `spec stage crashed: ` instead.
+ - `digest_sha256` is the SHA-256 of the injected digest, or `null` when nothing was injected.
+ - The whole key is `null` on a run without spec sources.
+ - `prxref review -v` prints it as `spec: OK/T source(s) ok, N constraint(s)`.
+- **The run trace (`PRXREF_TRACE_FILE`):**
+ - one `specs ok` event (at least one source fetched) or `specs fail` event (none did), with `sources`, `ok` and `constraints`; a `fail` event also carries the raw, unredacted `reasons`;
+ - a `specs relabel` event with `findings` when ungrounded `spec` findings were relabelled `warning` (see [docs/quality.md](quality.md#spec-grounding)).
+
+**Known limitation.** A file skipped inside a spec directory, whether a symlink or a file that cannot be read or decoded, is only logged at WARNING (`spec directory : skipped …`) while the other files are read. It does not appear in the posted note, the run record or the trace. A directory source fails, and shows up everywhere, only when none of its files could be read.
diff --git a/docs/env-vars.md b/docs/env-vars.md
index 3eda76d..d585f41 100644
--- a/docs/env-vars.md
+++ b/docs/env-vars.md
@@ -10,15 +10,17 @@ Configuration is loaded from built-in defaults, overridden by environment variab
| Variable | Default | Purpose |
|---|---|---|
-| `PRXREF_LLM_BACKEND` | `openai-compat` | LLM backend selector: `openai-compat`, `ferry`, or `http` (aliases for plain-HTTP OpenAI-compatible endpoint), or `litellm` (in-process LiteLLM router). |
-| `PRXREF_LLM_BASE_URL` | *(none — required)* | Base URL for the OpenAI-compatible endpoint (e.g. `https://openrouter.ai/api/v1`). Unset raises `ConfigError` and `prxref review` exits `2`. |
-| `PRXREF_LLM_API_KEY` | *(empty)* | API key / Bearer token sent to the OpenAI-compatible endpoint. Optional: leave empty for a local no-auth server. |
-| `PRXREF_LLM_MODELS` | *(none — required)* | Comma-separated model fallback chain evaluated in order, cheapest first. First model that answers successfully wins. Unset raises `ConfigError` and `prxref review` exits `2`. |
-| `PRXREF_LLM_REASONING_EFFORT` | *(empty)* | Reasoning effort for models that cannot disable reasoning (e.g. `low`\|`high`\|`max` for GLM-5.3-Flash). Empty omits the parameter entirely from the request. Provider-specific vocabulary; not validated client-side. Raising it makes truncation more likely — see [Reasoning models and the token budget](#reasoning-models-and-the-token-budget). |
-| `PRXREF_LLM_MAX_TOKENS` | `4096` | Completion-token budget for each worker's review call. Must be **greater than 0**. Too small and the model runs out of budget mid-JSON: that chunk is counted as failed and the summary says so. |
+| `PRXREF_LLM_BACKEND` | `openai-compat` | LLM backend selector: `openai-compat`, `ferry`, or `http` (aliases for the plain-HTTP OpenAI-compatible client), `litellm` (in-process LiteLLM router), or `claude-cli` / `kiro-cli` (an already-installed, already-logged-in coding CLI run as a subprocess — see [docs/llm.md](llm.md#subscription-cli-backends-claude-cli-and-kiro-cli)). Read case-insensitively; any other value raises `ConfigError` and `prxref review` exits `2`. |
+| `PRXREF_LLM_BASE_URL` | *(none)* | Base URL for the OpenAI-compatible endpoint (e.g. `https://openrouter.ai/api/v1`). **Required** for `openai-compat`/`ferry`/`http`: unset raises `ConfigError` and `prxref review` exits `2`. **Not used** by `litellm` (it resolves each model's own provider endpoint), `claude-cli` or `kiro-cli`; a value set there is ignored with one INFO line. A LiteLLM proxy is OpenAI-compatible, so point `openai-compat` at it. |
+| `PRXREF_LLM_API_KEY` | *(empty)* | API key / Bearer token sent to the OpenAI-compatible endpoint (`openai-compat` only). Optional: leave empty for a local no-auth server. |
+| `PRXREF_LLM_MODELS` | *(none — required)* | Model fallback chain evaluated in order, cheapest first, comma- **or** whitespace-separated. First model that answers successfully wins. Required by every backend: unset raises `ConfigError` and `prxref review` exits `2`. |
+| `PRXREF_LLM_REASONING_EFFORT` | *(empty)* | Reasoning effort for models that cannot disable reasoning (e.g. `low`\|`high`\|`max` for GLM-5.3-Flash). Sent as `reasoning_effort` in the request by `openai-compat`/`ferry`/`http` and as `--effort ` by `claude-cli`; `litellm` does not apply it, and `kiro-cli` drops it with one INFO line (see [docs/llm.md](llm.md#what-is-not-applied)). Empty omits the parameter entirely from the request. Provider-specific vocabulary; not validated client-side. Raising it makes truncation more likely — see [Reasoning models and the token budget](#reasoning-models-and-the-token-budget). |
+| `PRXREF_LLM_MAX_TOKENS` | `4096` | Completion-token budget (`max_tokens`) for each worker's review call on `openai-compat`/`ferry`/`http` and `litellm`; `claude-cli` and `kiro-cli` accept it and do not apply it (see [docs/llm.md](llm.md#what-is-not-applied)). Must be **greater than 0**. Too small and the model runs out of budget mid-JSON: that chunk is counted as failed and the summary says so. |
| `PRXREF_LLM_TIMEOUT` | `45.0` | Wall-clock deadline in seconds for one model's review call. Must be **greater than 0**. A model that runs past it is abandoned for the next one in the chain, so this is a per-model deadline, not a per-review one: a three-model chain can spend three times this value before the chunk is given up on. Under the default `openai-compat`/`ferry`/`http` backend the deadline is enforced client-side against elapsed time, including the response body — an endpoint that trickles bytes cannot outlast it. Under `litellm` the value is handed to that library, and its own timeout semantics apply. Overridable per run with `--timeout SECONDS`; the flag wins for that invocation only, and a bad value is reported as `--timeout`. |
-| `PRXREF_LLM_TEMPERATURE` | `0.0` (sent) | Sampling temperature, e.g. `0.2`. Must be **finite and >= 0**; there is no upper bound, because the maximum is provider-specific. Unset or empty sends the default `0.0` rather than omitting the parameter, so an identical diff reviews identically by default; a set value wins and restores provider-default sampling. |
-| `PRXREF_LLM_SEED` | *(empty — omitted)* | Optional integer sampling seed, sent as top-level `seed` to OpenAI-compatible backends (under `litellm` too). Must be **>= 0**; `0` is a valid seed. Empty or unset omits the parameter entirely, leaving the provider's own seed behaviour in place. Together with the default `PRXREF_LLM_TEMPERATURE=0` this is the strongest reproducibility lever the API offers. |
+| `PRXREF_LLM_TEMPERATURE` | `0.0` (sent) | Sampling temperature, e.g. `0.2`. Must be **finite and >= 0**; there is no upper bound, because the maximum is provider-specific. Unset or empty sends the default `0.0` rather than omitting the parameter, so an identical diff reviews identically by default; a set value wins and restores provider-default sampling. That holds on `openai-compat`/`ferry`/`http` and `litellm`; `claude-cli` and `kiro-cli` send no temperature, log one WARNING when it is set, and report `sampling.temperature` as `null`. |
+| `PRXREF_LLM_SEED` | *(auto-derived)* | Optional integer sampling seed, sent as top-level `seed` to OpenAI-compatible backends (under `litellm` too). Must be **>= 0**; `0` is a valid seed. Empty or unset does not omit the parameter: one random seed is derived per process and sent on every call of the run, and the run record's `sampling.seed` reports it. `claude-cli` and `kiro-cli` send no seed, log one WARNING when it is set, and report `sampling.seed` as `null`. Together with the default `PRXREF_LLM_TEMPERATURE=0` this is the strongest reproducibility lever the API offers. |
+| `PRXREF_LLM_CLI_PATH` | *(empty — found on `PATH`)* | `claude-cli` / `kiro-cli` only: the CLI binary to run (`~` is expanded). Empty looks up `claude` or `kiro-cli` on `PATH`. A path that does not resolve to an executable raises `ConfigError` and `prxref review` exits `2`. Ignored by the other backends. |
+| `PRXREF_LLM_CLI_CONCURRENCY` | `2` | `claude-cli` / `kiro-cli` only: how many CLI processes one client may run at once. Must be **greater than 0**. Each call is a full CLI process, and subscription limits are per account. |
| `PRXREF_CONFIDENCE_FLOOR` | `0.6` | Minimum confidence score. Must be **within `[0.0, 1.0]` inclusive** — it is a probability everywhere in the pipeline. Findings below the floor are dropped. |
| `PRXREF_MAX_ERROR_FINDINGS` | `10` | Maximum number of error-severity findings reported per review. Excess errors are dropped lowest-confidence-first. Must be **>= 0**; `0` is legal and caps every error. (Legacy alias: `PRXREF_MAX_ERRORS`.) |
| `PRXREF_MAX_CHUNKS` | `8` | Maximum number of diff chunks reviewed per PR. Must be **greater than 0**. Overridable per run with `--max-chunks`. |
@@ -28,11 +30,25 @@ Configuration is loaded from built-in defaults, overridden by environment variab
| `PRXREF_MAX_WORKERS` | `4` | Parallel chunk-review workers. Must be **greater than 0**. The cap that matters is usually the endpoint's rate limit, not the machine. |
| `PRXREF_MAX_INLINE_COMMENTS` | `15` | Maximum inline comments posted per review, applied **after** the quality gate. Must be **greater than 0**. Findings past the cap are still listed in the summary comment; only the inline posting is trimmed. |
| `PRXREF_TRACE_FILE` | *(empty — off)* | Path to append a JSONL run trace to. Empty disables tracing and the tracer becomes a no-op, so there is no cost when unset. One event per line (`run`, `forge.get_pr`, `forge.get_diff`, `parse_diff`, `build_chunks`, `chunk`, `heartbeat`, `post`), flushed as it happens. Each carries a phase: `start`, then `ok` or `fail`; `post` also uses `skip`, so a stage nobody asked to run is distinguishable from one the run never reached — a run still in flight, or one that was killed mid-hang, is as readable as a completed one. Render it to a standalone HTML pipeline view with `prxref trace render `. |
-| `PRXREF_TRACE_DIR` | *(empty — off)* | Directory for per-unit prompt/response traces. Each review unit (`chunk0`, `chunk1`, … and the whole-PR `sweep`) writes four files there: `.system.md` and `.user.md` (the exact rendered prompt halves), `.response.json` (the raw model text, JSON-encoded), and `.meta.json` (`unit`, `model`, token counts, `elapsed_ms`, `error`). Unset disables the dump entirely — no directory is created and there is no cost. Writes are best-effort: a failure is a logged warning, never a review failure, and a timeout retry overwrites the unit's files so the trace shows the attempt whose result was used. `--trace-dir DIR` on `prxref review` is the per-run equivalent and wins when both are set. |
+| `PRXREF_TRACE_DIR` | *(empty — off)* | Directory for per-unit prompt/response traces. Each review unit (`chunk0`, `chunk1`, … and the whole-PR `sweep`) writes four files there: `.system.md` and `.user.md` (the exact rendered prompt halves), `.response.json` (the raw model text, JSON-encoded), and `.meta.json` (`unit`, `model`, token counts, `elapsed_ms`, `error`, `cost_usd` — the dollar figure the backend reported for the call, `null` when it reported none — and `cost_source`, where that figure came from, `""` when `cost_usd` is `null`). Unset disables the dump entirely — no directory is created and there is no cost. Writes are best-effort: a failure is a logged warning, never a review failure, and a timeout retry overwrites the unit's files so the trace shows the attempt whose result was used. `--trace-dir DIR` on `prxref review` is the per-run equivalent and wins when both are set. |
| `PRXREF_DRY_RUN` | `False` | Set to the literal `1` to run the full review and write nothing to the forge — no summary, no inline comments. Applies to the webhook daemon as well as the CLI, which is the only way to watch the daemon against a real repository before letting it comment. `--no-post` is the per-invocation equivalent and still wins when the environment says nothing. Only the literal `1` enables it. |
-| `PRXREF_FAIL_ON` | `never` | Exit-code policy for `prxref review`. `never` (the default) keeps the advisory contract — the exit code never reflects findings. `error` exits `1` when the completed review carries an active error-severity finding; `any` exits `1` on any active finding. Under either value a review that fails to complete also exits `1`, so a gating lane cannot read a broken run as green. The webhook daemon has no exit code and is unaffected. See [Bad Configuration Is the Only Thing That Fails a Build](#bad-configuration-is-the-only-thing-that-fails-a-build). |
+| `PRXREF_FAIL_ON` | `never` | Exit-code policy for `prxref review`. `never` (the default) keeps the advisory contract — the exit code never reflects findings. `error` exits `1` when the completed review carries an active error-severity finding; `any` exits `1` on any active finding. Under either value a review that does not complete also exits `1` — it crashes, or it ends with verdict `Error` (the forge could not be read, the diff could not be parsed or chunked, or every chunk review failed) — so a gating lane cannot read a broken run as green. An empty PR diff is not a failure (verdict `Approved`, exit `0`). The webhook daemon has no exit code and is unaffected. See [Bad Configuration Is the Only Thing That Fails a Build](#bad-configuration-is-the-only-thing-that-fails-a-build). |
| `PRXREF_POST_MODE` | `summary+inline` | What gets posted to the forge: `summary+inline` (the summary comment, then inline comments only if the summary landed), `summary` (the summary comment only — inline comments are never posted), or `inline` (inline comments only — no summary is posted on any path, including the error notice). Any other value raises `ConfigError` and `prxref review` exits `2`. A dry run posts nothing in any mode. |
| `PRXREF_POST_VERDICT` | `True` | Set to the literal `1` to keep the verdict stamp in the posted summary; any other value renders the summary without it (no `Approved` / `Request-Changes` heading), keeping the findings, counts, and attribution. The computed verdict printed to stdout and the total-failure notice are unaffected. |
+| `PRXREF_PRICE_TABLE` | *(empty — no estimates)* | Fallback price table, used only when the backend reports no dollar cost of its own. Inline JSON (starting with `{`) or a path to a JSON file, e.g. `{"openai/gpt-4o-mini": {"input": 0.15, "output": 0.60}}`: USD per **million** tokens, keyed on the exact model name the run reports (`model=` in the attribution). Every entry is exactly `{"input": n, "output": n}` with `n` a finite number `>= 0`. A malformed table raises `ConfigError` and `prxref review` exits `2`. A run priced from the table is flagged `cost_estimated`; a model with neither a reported cost nor an entry reads "cost unknown", never `$0` — give free or local models a zero entry. See [Cost accounting](llm.md#cost-accounting). |
+| `PRXREF_POST_COST` | `False` | Set to the literal `1` to append the run's cost as the last field of the posted attribution line (`… · 3.1s · $0.0007`, `~$0.0007 (est.)` when estimated, else `$0.0007 (API-equivalent)` when every reported cost came from `claude-cli`, `cost unknown` when neither source priced it). Off by default, which keeps the attribution line byte-identical; the cost is always in the run record and the JSON output. See [Cost accounting](llm.md#cost-accounting). |
+| `PRXREF_SIZE_WARN_LINES` | *(empty — off)* | Advisory-only threshold on lines changed (added plus removed, excluding lock and generated files): a PR strictly above it gets one non-blocking heads-up line at the top of the summary. Must be **>= 0**; `0` is a legal, extreme threshold (any change at all), distinct from unset, which disables the check. Never affects the verdict or the exit code. |
+| `PRXREF_SIZE_WARN_FILES` | *(empty — off)* | Same contract as `PRXREF_SIZE_WARN_LINES`, counting changed files instead of lines. |
+| `PRXREF_SIZE_IGNORE_GLOBS` | *(empty)* | Extra glob patterns excluded from both size counts, **added** to the built-in lock/generated-file detection. Matched case-sensitively (`fnmatch`) against the full diff path, where `*` crosses `/`. Comma- or whitespace-separated, so a literal space in a glob is written `?`. |
+| `PRXREF_SPEC_SOURCES` | *(empty)* | Spec/ticket sources to review against: public web URLs, local file or directory paths, or Jira ticket URLs. Comma- **or** whitespace-separated when set through the environment. The repeatable `--spec` flag replaces this list entirely when given — there is no merge. In CI, a local path inside the PR's own checkout is content the PR controls; point it outside the workspace when the spec must be trusted. |
+| `PRXREF_SPEC_MAX_CHARS` | `120000` | Raw fetched characters kept per spec source (after decoding), before pruning. Must be **greater than 0**. Truncation at the cap is announced in the fetched text, never silent. |
+| `PRXREF_SPEC_DIGEST_TOKENS` | `3000` | Token budget for the final spec-constraint digest injected into worker prompts (estimated at 4 characters per token, like the systemic digest). Must be **greater than 0**. |
+| `PRXREF_REVIEW_RULES` | *(empty — off)* | Path to a team review-rules file (Markdown, with optional front matter carrying a `severity:` map) added to every review prompt. A missing, unreadable or malformed file raises `ConfigError` naming its source and `prxref review` exits `2`. `--rules-file PATH` wins for one run, and `--rules-file ""` turns the file off. Read it from a trusted checkout: in CI the workspace is usually the PR's own code, so a rules file inside it lets the PR rewrite its own review rules. See [docs/review-rules.md](review-rules.md). |
+| `PRXREF_REVIEW_RULES_MAX_CHARS` | `12000` | Characters of the rules body (after the front matter) kept in the prompt; a longer body is truncated with a warning. Must be **greater than 0**. |
+| `PRXREF_TICKET_CONTEXT_FILE` | *(empty — off)* | Plain-text or Markdown file holding the ticket this PR implements. When set, every finding is marked in, out of, or of unknown ticket scope. An empty or whitespace-only file means "this PR has no ticket". A missing or non-UTF-8 file raises `ConfigError` and `prxref review` exits `2`. The webhook daemon ignores it (and says so once). `--context-file PATH` wins for one run, and `--context-file ""` turns it off. |
+| `PRXREF_TICKET_CONTEXT_MAX_CHARS` | `6000` | Characters of ticket text kept in the prompt; longer text is truncated with a visible marker. Must be **greater than 0**. |
+
+The replay flags of `prxref review` (`--base-sha`, `--head-sha`, `--no-threads`, `--diff-file`) deliberately have no environment variable: set in the environment, a replay pin would silently pin every run, the webhook daemon's included.
### Per-Forge Authentication
@@ -47,6 +63,15 @@ Configuration is loaded from built-in defaults, overridden by environment variab
| `PRXREF_GITHUB_TOKEN` | *(empty)* | GitHub Personal Access Token or GitHub App token for `github.com`. |
| `PRXREF_GITHUB_ENTERPRISE_TOKEN` | *(empty)* | GitHub Enterprise token for custom/self-hosted GitHub Enterprise Server domains. Falls back to `PRXREF_GITHUB_TOKEN` if unset. |
| `PRXREF_GITLAB_TOKEN` | *(empty)* | GitLab Personal, Project, or Group Access Token (sent via `PRIVATE-TOKEN` header) for `gitlab.com` or self-hosted GitLab. |
+| `PRXREF_AZURE_DEVOPS_TOKEN` | *(empty)* | Azure DevOps personal access token, sent as Basic `:PAT`: **Code (Read)** to review, **Code (Read & write)** to post. When empty, the Pipelines `SYSTEM_ACCESSTOKEN` is sent as a Bearer token; when both are empty, requests are anonymous (public projects, read-only). |
+
+### Spec Sources / Jira
+
+| Variable | Default | Purpose |
+|---|---|---|
+| `PRXREF_JIRA_BASE_URL` | *(empty)* | Jira base URL — `scheme://host` plus any context path (e.g. `https://jira.example.com/jira`) — that ticket fetches are looked up on, overriding a ticket URL's own base; a self-hosted board often sits behind a different REST host than its browse URL. **Jira credentials are only ever sent here.** Empty resolves the ticket URL's own base, anonymously. An `http://` base with credentials is allowed but logs a warning. |
+| `PRXREF_JIRA_EMAIL` | *(empty)* | Jira account email for HTTP basic authentication when fetching a ticket, used together with `PRXREF_JIRA_API_TOKEN` and **only** together with `PRXREF_JIRA_BASE_URL`: credentials set without a base URL are never sent — the fetch stays anonymous and a warning is logged. When either credential is empty the fetch is anonymous, which public boards accept. |
+| `PRXREF_JIRA_API_TOKEN` | *(empty)* | Jira API token for HTTP basic authentication, paired with `PRXREF_JIRA_EMAIL` and sent only to `PRXREF_JIRA_BASE_URL`. Missing credentials are a fetch failure (the review proceeds un-grounded with a note naming these variables), not a configuration error. |
### Webhook Receiver
@@ -55,11 +80,12 @@ Configuration is loaded from built-in defaults, overridden by environment variab
| `PRXREF_BITBUCKET_WEBHOOK_SECRET` | *(empty)* | HMAC secret for Bitbucket webhooks, Cloud and Server alike (verified against `X-Hub-Signature` via HMAC-SHA256). |
| `PRXREF_GITHUB_WEBHOOK_SECRET` | *(empty)* | HMAC secret for GitHub webhooks (verified against `X-Hub-Signature-256` via HMAC-SHA256). |
| `PRXREF_GITLAB_WEBHOOK_SECRET` | *(empty)* | Secret token for GitLab webhooks (verified against `X-Gitlab-Token`). |
+| `PRXREF_AZURE_DEVOPS_WEBHOOK_SECRET` | *(empty)* | Secret for Azure DevOps service hooks, compared in constant time with the **password** of the hook's Basic authentication (the user name is ignored). Empty rejects Azure DevOps webhooks with `401` unless `PRXREF_ALLOW_UNSIGNED` is `1`. |
| `PRXREF_ALLOW_UNSIGNED` | `False` | Accepts webhooks without valid HMAC/token signatures (dev/testing only; logs a warning). Must be the literal string `1` — `true`/`yes`/`on` deliberately do **not** enable the bypass, so it cannot be switched on by a stray truthy value. |
## Bad Configuration Is the Only Thing That Fails a Build
-`prxref review` exits **0** on every review error — an empty diff, a network failure, an LLM timeout, bad forge credentials, even a review in which every chunk failed. prxref is an advisor, not a merge gate.
+Under the default `PRXREF_FAIL_ON=never`, `prxref review` exits **0** on every review error — a network failure, an LLM timeout, bad forge credentials, even a review in which every chunk failed — and on an empty PR diff, which is not an error at all. prxref is an advisor, not a merge gate.
It exits **2** on exactly one class of problem: a **configuration error**. That is a required value missing, a value that will not parse, a value outside its valid range, or one outside its key's allowed vocabulary (`PRXREF_FAIL_ON` accepts only `never`, `error`, or `any`). The check runs after the environment *and* any programmatic override, so no path into the config can smuggle a degenerate value through to the wire.
@@ -74,7 +100,7 @@ configuration error: --max-chunks: must be a finite number greater than 0, got 0
The second form exists because naming the environment variable unconditionally sent operators hunting for a `PRXREF_MAX_CHUNKS` they had never set.
-One knob can move the exit code beyond that: `PRXREF_FAIL_ON`. Its default `never` is everything above, unchanged. Setting it to `error` exits **1** when the completed review carries an active error-severity finding; `any` exits **1** on any active finding; and under either value a review that fails to complete also exits **1** — a gate that silently passes on a broken run is worse than none. An unrecognized PR URL still exits **0** under every value: nothing was reviewed, so there is no outcome to gate on. The webhook daemon has no exit code and is unaffected.
+One knob can move the exit code beyond that: `PRXREF_FAIL_ON`. Its default `never` is everything above, unchanged. Setting it to `error` exits **1** when the completed review carries an active error-severity finding; `any` exits **1** on any active finding; and under either value a review that does not complete also exits **1** — it crashes, or it ends with verdict `Error` (the forge could not be read, the diff could not be parsed or chunked, or every chunk review failed). A gate that silently passes on a broken run is worse than none. An empty PR diff is not a failure (verdict `Approved`, exit **0**). An unrecognized PR URL still exits **0** under every value: nothing was reviewed, so there is no outcome to gate on. The webhook daemon has no exit code and is unaffected.
Think hard before reaching for it. Failing a build on a finding turns a probabilistic reviewer into a merge gate, and the first false positive teaches the team to bypass the gate. Read the verdict from the posted summary instead — and do not build a security control on the exit code.
@@ -122,14 +148,15 @@ Neither knob affects the exit code — `PRXREF_FAIL_ON` is the only one that can
## Quality Passes and Drop Reasons
-The two knobs above are the only configuration that touches the filtering. The eleven deterministic passes themselves, the release-shaped-PR check, and every `drop_reason` string they emit are documented in one place: **[docs/quality.md](quality.md)**. Everything on that page other than the confidence floor and the error cap is a correctness check against the diff itself, not a noise lever, and has no environment variable.
+The two knobs above are the only configuration that touches the filtering. The eleven deterministic passes themselves, the team severity map and spec grounding that run before them, the release-shaped-PR check, and every `drop_reason` string they emit are documented in one place: **[docs/quality.md](quality.md)**. Everything on that page other than the confidence floor and the error cap is a correctness check against the diff itself, not a noise lever, and has no environment variable.
## Environment Cross-Check & Defaults
-The tables above define all **36** configuration keys in `src/prxref/config.py` (`_DEFAULTS`), and every one of them appears in `.env.example`:
+The tables above define all **55** configuration keys in `src/prxref/config.py` (`_DEFAULTS`), and every one of them appears in `.env.example`:
-- **LLM / Pipeline (23):** `PRXREF_LLM_BACKEND`, `PRXREF_LLM_BASE_URL`, `PRXREF_LLM_API_KEY`, `PRXREF_LLM_MODELS`, `PRXREF_LLM_REASONING_EFFORT`, `PRXREF_LLM_MAX_TOKENS`, `PRXREF_LLM_TIMEOUT`, `PRXREF_LLM_TEMPERATURE`, `PRXREF_LLM_SEED`, `PRXREF_CONFIDENCE_FLOOR`, `PRXREF_MAX_ERROR_FINDINGS`, `PRXREF_MAX_CHUNKS`, `PRXREF_CHUNK_TOKEN_BUDGET`, `PRXREF_CHUNK_MAX_FILES`, `PRXREF_CHUNK_CONTEXT_LINES`, `PRXREF_MAX_WORKERS`, `PRXREF_MAX_INLINE_COMMENTS`, `PRXREF_TRACE_FILE`, `PRXREF_TRACE_DIR`, `PRXREF_DRY_RUN`, `PRXREF_FAIL_ON`, `PRXREF_POST_MODE`, `PRXREF_POST_VERDICT`
-- **Per-Forge Auth (9):** `PRXREF_BITBUCKET_TOKEN`, `PRXREF_BITBUCKET_USER`, `PRXREF_BITBUCKET_APP_PASSWORD`, `PRXREF_BITBUCKET_SERVER_TOKEN`, `PRXREF_BITBUCKET_SERVER_USER`, `PRXREF_BITBUCKET_SERVER_PASSWORD`, `PRXREF_GITHUB_TOKEN`, `PRXREF_GITHUB_ENTERPRISE_TOKEN`, `PRXREF_GITLAB_TOKEN`
-- **Webhooks (4):** `PRXREF_BITBUCKET_WEBHOOK_SECRET`, `PRXREF_GITHUB_WEBHOOK_SECRET`, `PRXREF_GITLAB_WEBHOOK_SECRET`, `PRXREF_ALLOW_UNSIGNED`
+- **LLM / Pipeline (37):** `PRXREF_LLM_BACKEND`, `PRXREF_LLM_BASE_URL`, `PRXREF_LLM_API_KEY`, `PRXREF_LLM_MODELS`, `PRXREF_LLM_REASONING_EFFORT`, `PRXREF_LLM_MAX_TOKENS`, `PRXREF_LLM_TIMEOUT`, `PRXREF_LLM_TEMPERATURE`, `PRXREF_LLM_SEED`, `PRXREF_LLM_CLI_PATH`, `PRXREF_LLM_CLI_CONCURRENCY`, `PRXREF_CONFIDENCE_FLOOR`, `PRXREF_MAX_ERROR_FINDINGS`, `PRXREF_MAX_CHUNKS`, `PRXREF_CHUNK_TOKEN_BUDGET`, `PRXREF_CHUNK_MAX_FILES`, `PRXREF_CHUNK_CONTEXT_LINES`, `PRXREF_MAX_WORKERS`, `PRXREF_MAX_INLINE_COMMENTS`, `PRXREF_FAIL_ON`, `PRXREF_DRY_RUN`, `PRXREF_TRACE_FILE`, `PRXREF_TRACE_DIR`, `PRXREF_POST_MODE`, `PRXREF_POST_VERDICT`, `PRXREF_PRICE_TABLE`, `PRXREF_POST_COST`, `PRXREF_SIZE_WARN_LINES`, `PRXREF_SIZE_WARN_FILES`, `PRXREF_SIZE_IGNORE_GLOBS`, `PRXREF_SPEC_SOURCES`, `PRXREF_SPEC_MAX_CHARS`, `PRXREF_SPEC_DIGEST_TOKENS`, `PRXREF_REVIEW_RULES`, `PRXREF_REVIEW_RULES_MAX_CHARS`, `PRXREF_TICKET_CONTEXT_FILE`, `PRXREF_TICKET_CONTEXT_MAX_CHARS`
+- **Per-Forge Auth (10):** `PRXREF_BITBUCKET_TOKEN`, `PRXREF_BITBUCKET_USER`, `PRXREF_BITBUCKET_APP_PASSWORD`, `PRXREF_BITBUCKET_SERVER_TOKEN`, `PRXREF_BITBUCKET_SERVER_USER`, `PRXREF_BITBUCKET_SERVER_PASSWORD`, `PRXREF_GITHUB_TOKEN`, `PRXREF_GITHUB_ENTERPRISE_TOKEN`, `PRXREF_GITLAB_TOKEN`, `PRXREF_AZURE_DEVOPS_TOKEN`
+- **Spec Sources / Jira (3):** `PRXREF_JIRA_BASE_URL`, `PRXREF_JIRA_EMAIL`, `PRXREF_JIRA_API_TOKEN`
+- **Webhooks (5):** `PRXREF_BITBUCKET_WEBHOOK_SECRET`, `PRXREF_GITHUB_WEBHOOK_SECRET`, `PRXREF_GITLAB_WEBHOOK_SECRET`, `PRXREF_AZURE_DEVOPS_WEBHOOK_SECRET`, `PRXREF_ALLOW_UNSIGNED`
-*(36 configuration keys, plus one deprecated alias — `PRXREF_MAX_ERRORS` for `PRXREF_MAX_ERROR_FINDINGS` — for 37 accepted variable names.)*
+*(55 configuration keys, plus one deprecated alias — `PRXREF_MAX_ERRORS` for `PRXREF_MAX_ERROR_FINDINGS` — for 56 accepted variable names.)*
diff --git a/docs/forges.md b/docs/forges.md
index caf62d1..efed466 100644
--- a/docs/forges.md
+++ b/docs/forges.md
@@ -1,6 +1,6 @@
# Forge Integrations & Webhooks
-`prxref` provides unified pull/merge request reviews across Bitbucket (Cloud and Server / Data Center), GitHub (Cloud and Enterprise Server), and GitLab (SaaS and self-hosted).
+`prxref` provides unified pull/merge request reviews across Bitbucket (Cloud and Server / Data Center), GitHub (Cloud and Enterprise Server), GitLab (SaaS and self-hosted), and Azure DevOps (Services and Server).
## Supported Hosts
@@ -9,8 +9,9 @@
| **Bitbucket** | `bitbucket.org` | Supported — Bitbucket Server / Data Center, any host, including a deployment context path |
| **GitHub** | `github.com` | Supported — GitHub Enterprise Server, any host |
| **GitLab** | `gitlab.com` | Supported — any host, including nested subgroups |
+| **Azure DevOps** | `dev.azure.com`, `*.visualstudio.com` | Supported, untested live — Azure DevOps Server, any host; the URL must include the collection and the project |
-Every host is covered, but not by the same means. GitHub and GitLab are host-agnostic within one adapter each, because their self-hosted products speak the same REST API as their SaaS ones, differing only in base URL (`/api/v3` for GHES, `/api/v4` for every GitLab). Bitbucket is not: Server / Data Center exposes a different API surface (`/rest/api/1.0`) with different resource shapes, so it is a fourth adapter rather than a base-URL setting, selected automatically from the URL. See [Bitbucket Server / Data Center](#4-bitbucket-server--data-center).
+Every host is covered, but not by the same means. GitHub and GitLab are host-agnostic within one adapter each, because their self-hosted products speak the same REST API as their SaaS ones, differing only in base URL (`/api/v3` for GHES, `/api/v4` for every GitLab). Bitbucket is not: Server / Data Center exposes a different API surface (`/rest/api/1.0`) with different resource shapes, so it is a fourth adapter rather than a base-URL setting, selected automatically from the URL. See [Bitbucket Server / Data Center](#4-bitbucket-server--data-center). Azure DevOps is the fifth adapter, and like GitHub and GitLab it serves both products: Services and Server speak the same REST API and differ only in where the collection sits in the URL. See [Azure DevOps Services & Server](#5-azure-devops-services--server).
---
@@ -36,6 +37,7 @@ Every host is covered, but not by the same means. GitHub and GitLab are host-agn
- **Accepted Events:** `pullrequest:created`, `pullrequest:updated`
- **Payload:** PR URL read from `pullrequest.links.html.href`.
- **Signature Header:** `X-Hub-Signature` (HMAC-SHA256) validated against `PRXREF_BITBUCKET_WEBHOOK_SECRET`.
+- **Pinned Commit Range (Replay):** `GET /2.0/repositories/{owner}/{repo}/diff/{head_sha}..{base_sha}?topic=true` with `Accept: text/plain`, returning the changes on the head side of the merge-base. Bitbucket spells a range SOURCE..DEST, the reverse of git, so the head SHA comes first; the other order names the reverse range and returns a different diff that still parses. `topic=true` is the merge-base ("three-dot") form and is sent explicitly rather than left to the default, because `topic=false` diffs the two commits directly and so also shows whatever landed on the base after the fork. The text is returned unmodified, an empty range returns empty text, and an HTTP or transport error raises.
---
@@ -51,7 +53,7 @@ Every host is covered, but not by the same means. GitHub and GitLab are host-agn
- **API Endpoints & Behavior:**
- **Base URL:** `https://api.github.com` for `github.com`, or `https://{host}/api/v3` for GHES.
- **Metadata:** `GET /repos/{owner}/{repo}/pulls/{number}`
- - **Diffs:** `GET /repos/{owner}/{repo}/pulls/{number}` with `Accept: application/vnd.github.v3.diff, application/vnd.diff`.
+ - **Diffs:** `GET /repos/{owner}/{repo}/pulls/{number}` with `Accept: application/vnd.github.v3.diff, application/vnd.diff`. GitHub refuses this diff for a pull request whose diff runs past 20,000 lines (HTTP `406`, error code `too_large`), so such a PR ends as an `Error` run and gets the error notice. There is no fallback to the paged `/pulls/{number}/files` listing yet.
- **Summary Comments:** Managed on the issue comments endpoint (`/repos/{owner}/{repo}/issues/{number}/comments`). Summary deduplication is handled via the embedded hidden HTML marker ``. If an existing review comment contains this marker, it is updated via `PATCH /repos/{owner}/{repo}/issues/comments/{comment_id}` instead of creating a duplicate comment.
- **Inline Comments:** `POST /repos/{owner}/{repo}/pulls/{number}/comments` with `body`, `path`, `line`, and `side` (`RIGHT`). HTTP 422 errors (e.g. comment line not part of diff hunk) are gracefully skipped.
- **Thread List:** `GET /repos/{owner}/{repo}/pulls/{number}/comments`.
@@ -60,6 +62,7 @@ Every host is covered, but not by the same means. GitHub and GitLab are host-agn
- **Event Header:** `X-GitHub-Event` (must equal `pull_request`)
- **Accepted Actions:** `opened`, `synchronize`
- **Signature Header:** `X-Hub-Signature-256` (HMAC-SHA256) validated against `PRXREF_GITHUB_WEBHOOK_SECRET`.
+- **Pinned Commit Range (Replay):** `GET /repos/{owner}/{repo}/compare/{base_sha}...{head_sha}` on the same base URL (GHES included), with `Accept: application/vnd.github.diff`. The three dots are the merge-base form and are required, because the two-dot spelling returns 404. Without the diff media type the endpoint returns its JSON comparison object rather than a diff. The text is returned unmodified, an empty range (a head already merged into the base) returns empty text, and an HTTP or transport error raises.
---
@@ -74,15 +77,16 @@ Every host is covered, but not by the same means. GitHub and GitLab are host-agn
- **API Endpoints & Behavior:**
- **Base URL:** `https://{host}/api/v4/projects/{url_encoded_project_path}`
- **Metadata:** `GET /merge_requests/{number}`. Target SHA is resolved from `diff_refs.base_sha` or fallback branch lookup.
- - **Diffs:** `GET /merge_requests/{number}/diffs?access_raw_diffs=true`. Reconstructs a full unified multi-file diff string from GitLab's structured diff items (including new, deleted, renamed, and modified file headers).
+ - **Diffs:** `GET /merge_requests/{number}/diffs`, paged with `per_page=100` and read to the last page. Reconstructs a full unified multi-file diff string from GitLab's structured diff items (including new, deleted, renamed, and modified file headers). A page that cannot be read (a transport error, a non-OK status, or a body that is not a JSON list), or a listing longer than 50 pages (5,000 files), fails the review with `FeedReadError` rather than reviewing part of the MR. An MR with no file entries at all is an error too. An entry flagged `too_large` or `collapsed` carries no inline diff: it is logged as a warning and reviewed as a header-only file.
- **Summary Comments:** Managed via `GET/POST/PUT /merge_requests/{number}/notes`. Searches for `` and updates existing note via `PUT` if found.
- **Inline Comments:** Posted as discussions via `POST /merge_requests/{number}/discussions` with text position references (`base_sha`, `start_sha`, `head_sha`, `new_path`, `new_line`). If position anchoring fails with HTTP 400 (e.g. line outside diff or obsolete context), it automatically falls back to posting a plain note via `POST /merge_requests/{number}/notes` formatted with `file: {path}\n\n{body}`.
- - **Thread List:** `GET /merge_requests/{number}/discussions`.
+ - **Thread List:** `GET /merge_requests/{number}/discussions`. Thread dedup needs `PRXREF_GITLAB_TOKEN` even on a public gitlab.com project: gitlab.com serves the MR and its diffs anonymously but answers anonymous `/notes` and `/discussions` requests with HTTP 401, so a tokenless review logs `discussion feed read was incomplete` and dedups against no threads.
- **File Content:** `GET /repository/files/{url_encoded_path}/raw?ref={sha}` (path percent-encoded including slashes), best-effort, read with the same `PRIVATE-TOKEN` as everything else above (no extra scope). A non-2xx, oversize, or binary body returns `None` and is logged at debug, never a hard error.
- **Webhook Integration:**
- **Event Header:** `X-Gitlab-Event` (normalized to `MergeRequestHook`)
- **Accepted Actions:** `open`, `update`
- **Signature Header:** `X-Gitlab-Token` (plain secret token) validated against `PRXREF_GITLAB_WEBHOOK_SECRET`.
+- **Pinned Commit Range (Replay):** `GET /repository/compare?from={base_sha}&to={head_sha}&straight=false`. `straight=false` is the merge-base form; `straight=true` would diff the two commits directly. `unidiff` is deliberately not requested, so each entry's `diff` holds only its hunks, and the entries are rendered by the same header reconstruction as **Diffs** above. A response with `compare_timeout: true` raises rather than reviewing an incomplete file list. An entry flagged `too_large` or `collapsed` carries no inline diff: it is logged as a warning and reviewed as a header-only file. An empty range returns empty text, and an HTTP or transport error raises.
---
@@ -160,3 +164,180 @@ paging rather than `page`/`pagelen`. It therefore gets its own adapter.
capital `R`, and the list, both of which differ from Cloud.
- **Signature Header:** `X-Hub-Signature` (HMAC-SHA256) validated against
`PRXREF_BITBUCKET_WEBHOOK_SECRET`, the same secret Cloud uses.
+- **Pinned Commit Range (Replay):** two requests under
+ `{scheme}://{host}{context}/rest/api/1.0/projects/{key}/repos/{slug}`. First
+ `GET …/commits/{head_sha}/merge-base?otherCommitId={base_sha}`, whose `id` is the fork point;
+ then `GET …/diff?since={merge_base}&until={head_sha}` with `Accept: text/plain`, the raw diff,
+ returned unmodified. That raw diff runs from whatever `since` names, with no merge-base step
+ of its own, so the lookup is what makes it a three-dot diff. The spec lists the raw diff only
+ as `text/plain; qs=0.1`, so the request names that type. If the merge-base lookup fails (an
+ HTTP or transport error, or a response naming no commit), a warning is logged and the diff
+ runs from `since={base_sha}`. That is still right whenever the base SHA is already the fork
+ point, as a PR's recorded target commit usually is. An empty range returns empty text, and a
+ failed diff request raises. Both endpoints come from the Data Center 9.4 REST reference and
+ have **not been probed against a live Data Center**. No minimum version is claimed, but one
+ Atlassian knowledge-base article reports that the path-less `/diff` returns 400 on some older
+ versions.
+
+---
+
+## 5. Azure DevOps Services & Server
+
+Azure DevOps has no endpoint that returns a unified diff, so this is the one
+adapter that builds its diff instead of downloading it. One adapter covers
+Azure DevOps Services and Azure DevOps Server (on-prem): both speak REST
+`api-version=7.1`, and they differ only in where the collection sits in the URL.
+
+- **Forge Identifier:** `azure-devops`
+- **Supported URL Shapes:**
+ - `https://dev.azure.com/{organization}/{project}/_git/{repo}/pullrequest/{number}`
+ - `https://dev.azure.com/{organization}/_git/{repo}/pullrequest/{number}` (short form,
+ for a project named like its repository)
+ - `https://{organization}.visualstudio.com/{project}/_git/{repo}/pullrequest/{number}`,
+ with or without a `DefaultCollection` segment after the host, plus the same short form
+ - `http(s)://{host}/{collection path}/{project}/_git/{repo}/pullrequest/{number}` for
+ Azure DevOps Server, e.g. `https://{host}/tfs/DefaultCollection/{project}/_git/...`.
+ A Server URL must name both the collection and the project. With a single segment
+ before `_git` there is no telling which one it is, so the URL is rejected.
+ - Percent-encoded names (`Web%20Platform`) are decoded. A query string or fragment
+ (`?_a=files`) and a trailing route are ignored. The URL is normalized to the
+ explicit-project form.
+- **Scheme Note:** as on Bitbucket Server, the scheme of the URL you pass is kept, so
+ an on-prem server on plain HTTP works.
+- **Detection Order:** `detect_forge` asks this parser last. No other forge's pattern
+ accepts the `/_git/{repo}/pullrequest/{number}` shape, so the position is defensive.
+- **Authentication:** the first of these that is set wins.
+ 1. `PRXREF_AZURE_DEVOPS_TOKEN`: a personal access token, sent as Basic `:PAT` (empty
+ user name). **Code (Read)** to review; **Code (Read & write)** to post.
+ 2. `SYSTEM_ACCESSTOKEN`: the Azure Pipelines job token, sent as `Bearer`. Pipelines
+ does not hand it to scripts unless the step maps it (below).
+ 3. Neither: anonymous. A public project can be reviewed with no token at all.
+ Posting always needs one.
+
+ Every request sends `X-TFS-FedAuthRedirect: Suppress`, so an unauthenticated
+ request gets a plain `401` rather than a sign-in page. A `203` or a non-JSON body is
+ refused with an error that names `PRXREF_AZURE_DEVOPS_TOKEN`.
+- **Azure Pipelines:** map the job token into the step, and grant the project's
+ **Build Service** identity **Contribute to pull requests** on the repository so it
+ can post. `System.CollectionUri` ends with a `/` and covers both Services and Server:
+
+ ```yaml
+ steps:
+ - script: >-
+ uvx prxref review --pr-url
+ "$(System.CollectionUri)$(System.TeamProject)/_git/$(Build.Repository.Name)/pullrequest/$(System.PullRequest.PullRequestId)"
+ env:
+ SYSTEM_ACCESSTOKEN: $(System.AccessToken)
+ PRXREF_LLM_BASE_URL: $(PRXREF_LLM_BASE_URL)
+ PRXREF_LLM_MODELS: $(PRXREF_LLM_MODELS)
+ PRXREF_LLM_API_KEY: $(PRXREF_LLM_API_KEY)
+ ```
+
+ Run it as a build-validation pipeline (a branch policy), which is what sets the
+ `System.PullRequest.*` variables.
+- **API Endpoints & Behavior:**
+ - **Base URL:** `{scheme}://{host}{collection}/{project}/_apis/git/repositories/{repo}`,
+ always project-scoped (the organization-level routes refuse anonymous reads), with
+ `api-version=7.1` on every request.
+ - **Metadata:** `GET {base}/pullrequests/{number}`. Branches come from
+ `sourceRefName`/`targetRefName` without `refs/heads/`, and SHAs from
+ `lastMergeSourceCommit`/`lastMergeTargetCommit`. The author is
+ `createdBy.displayName`, because the unique name is null on anonymous reads.
+ - **Diffs:** rebuilt locally.
+ `GET {base}/diffs/commits?baseVersion={target sha}&targetVersion={source sha}&diffCommonCommit=true`
+ lists the changed files against the merge base, which is the PR's own view (three
+ dots). It is paged 1000 entries at a time until `allChangesIncluded`, and more
+ than 50 pages is an error, not a partial review. Contents come from
+ `GET {base}/blobs/{objectId}?$format=octetstream`, eight at a time, and `difflib`
+ renders a git-style unified diff, including `\ No newline at end of file`, that
+ `git apply` accepts. Azure DevOps detects renames itself, and a pure rename
+ fetches nothing. A binary file (by extension, or a NUL byte in its first 8000
+ bytes) renders as `Binary files … differ`. Content is capped at 512 KiB per blob,
+ 300 files and 16 MiB per diff; a file past a cap is listed without hunks, with a
+ warning. A blob that returns `404` or `410` is listed without hunks too, but any other
+ failed blob fetch fails the review rather than silently emptying a file. The
+ change list comes from the Diffs API rather than the PR's iterations because the
+ iterations list is not readable anonymously, even on a public project.
+ - **Summary Comments:** a PR-level thread with status `closed`, created with
+ `POST {base}/pullrequests/{number}/threads`. A re-review finds its earlier summary
+ by the `` marker and edits it with
+ `PATCH {base}/pullrequests/{number}/threads/{thread}/comments/{comment}`. When the
+ thread list cannot be read, nothing is posted, so a failed lookup never produces
+ a second summary.
+ - **Inline Comments:** one thread per finding, with status `active` and a
+ `threadContext` carrying the `/`-prefixed `filePath` and the line in
+ `rightFileStart`/`rightFileEnd`. When the PR's iterations are readable,
+ `pullRequestThreadContext` pins the thread to the latest iteration's
+ `changeTrackingId` for that file; otherwise it is left out. A 4xx on one comment
+ is skipped with a warning.
+ - **Thread List:** `GET {base}/pullrequests/{number}/threads` returns every thread in
+ one response. System threads (votes, pushes, status changes) and deleted threads
+ are skipped. A thread counts as resolved when its status is `fixed`, `wontFix`,
+ `closed` or `byDesign`.
+ - **Prune:** a stale prxref inline thread is removed by deleting its root comment,
+ `DELETE {base}/pullrequests/{number}/threads/{thread}/comments/{comment}`, matched
+ by the attribution marker. The summary thread and human replies are never touched.
+ - **File Content:** `GET {base}/items?path=/{path}&versionDescriptor.version={sha}&versionDescriptor.versionType=commit&download=true`,
+ best-effort, read with the same token as everything else above. A `404`, an
+ oversize (512 KiB) body, or a binary body returns `None` and is never a hard error.
+- **Thread statuses and "Check for comment resolution":** inline threads are posted
+ `active`, like an unresolved inline comment on every other forge. So a branch policy
+ that requires comment resolution holds the PR until someone resolves prxref's
+ threads, which is what GitHub's "require conversation resolution" rule already does.
+ The summary is posted `closed`, so it never blocks a merge.
+- **Webhook Integration:** Azure DevOps service hooks.
+ - **Detection:** there is no event header. The request is recognized by its JSON
+ body (`publisherId` is `tfs`), and only when none of the other forges' event
+ headers is present.
+ - **Accepted Events:** `git.pullrequest.created` and `git.pullrequest.updated`, and
+ only while `resource.status` is `active`. A completed or abandoned PR, like any
+ other event, is acknowledged with `202` and not reviewed.
+ - **Payload:** the PR URL is `resource.repository.webUrl` (else `remoteUrl` without
+ its `user@` prefix) plus `/pullrequest/{resource.pullRequestId}`.
+ - **Authentication:** HTTP Basic. The password is compared in constant time with
+ `PRXREF_AZURE_DEVOPS_WEBHOOK_SECRET`, and the user name is ignored. An unset secret
+ rejects every Azure DevOps webhook with `401` unless `PRXREF_ALLOW_UNSIGNED=1`.
+ Setup: [Azure DevOps service hooks](deploy.md#azure-devops-service-hooks).
+- **Known Limitations:**
+ - CRLF files lose their `\r` in the rendered diff. The diff parser reads lines
+ without their terminators on every forge.
+ - A path containing a tab or a newline cannot be written in a unified diff, so such
+ a file is skipped with a warning.
+ - Inline-comment line numbers can drift in a file that uses a form feed or a Unicode
+ line or paragraph separator inside a line.
+ - `difflib` does not promise a minimal diff on pathological files. Its output is
+ still self-consistent, and `git apply` accepts it.
+ - Azure DevOps exposes no file modes here, so every file is mode `100644`.
+- **What is tested live:** reviewing a public Azure DevOps Services project with no
+ token. The write paths (summary, inline threads, prune), PAT and `SYSTEM_ACCESSTOKEN`
+ authentication, and the service-hook payload are covered by unit tests against
+ recorded API shapes, but **have not been exercised against a live server**. Azure
+ DevOps Server is parsed and authenticated the same way but is **untested**. It needs
+ a release that accepts REST `api-version=7.1` (2022.1 or later, going by Microsoft's
+ API version table).
+- **Pinned Commit Range (Replay):** the **Diffs** path above with both ends pinned as
+ commits:
+ `GET {base}/diffs/commits?baseVersion={base_sha}&baseVersionType=commit&targetVersion={head_sha}&targetVersionType=commit&diffCommonCommit=true`,
+ paged with `$top=1000`/`$skip`, then `GET {base}/blobs/{objectId}?$format=octetstream`
+ for the file contents. `diffCommonCommit=true` is the merge-base ("three-dot")
+ form: the list runs from the merge base of the two commits to `head_sha`, and each
+ file's old side is its blob at that merge base. `false` would diff the two commits
+ directly and so also list whatever changed on the base after the fork. Given a PR's
+ own target and source commits, the result is the PR's diff. There is no raw text to
+ return unmodified: the diff is rebuilt as for **Diffs**, so it has no `index` lines or
+ function names after `@@`, every mode is `100644`, and a `similarity index` line
+ appears only for a pure rename. The same budgets apply: 512 KiB per blob, 300 files
+ and 16 MiB per diff, and a file past a cap, or whose blob is gone (`404` or `410`), is
+ listed without hunks, with a warning. An empty range returns empty text, which replay
+ reports as an error run. An HTTP or transport error raises, and so do a non-JSON
+ listing, any other failed blob fetch, and a listing longer than 50 pages. While the
+ adapter was designed, a prototype of this method returned a PR's own diff from a
+ public Azure DevOps Services project, and a probe there saw `true` leave out files
+ that `false` listed. On 2026-09-23 the shipped method was run live, read-only and with
+ no token, against a public Azure DevOps Services project. Given a PR's own target and
+ source commits, it returned the adapter's diff of that PR byte for byte, also for two
+ PRs whose target had gained a commit since they forked: for those, its file list
+ matched git's three-dot `target...source` diff, and the `false` listing named extra
+ files that it left out. A range whose two ends are the same commit returned empty
+ text. Beyond that run, its tests use recorded response shapes. Azure DevOps Server is
+ untested, as above.
diff --git a/docs/live-instance-verification/followup-tasks-real-forge-fixtures.md b/docs/live-instance-verification/followup-tasks-real-forge-fixtures.md
index 711a67c..76d221b 100644
--- a/docs/live-instance-verification/followup-tasks-real-forge-fixtures.md
+++ b/docs/live-instance-verification/followup-tasks-real-forge-fixtures.md
@@ -15,11 +15,10 @@ was transcribed from.
## Origin
-Produced by the `sharpen` retro of session
-`974bbdda-6362-4426-9971-fec77fabc1d9` (2026-08-30/31). That session stood up
-Bitbucket Data Center 10.4.2 in Docker with an unattended timebomb-license
-setup, seeded a repo with three planted bugs, opened a real PR, and ran
-`prxref review` against it with posting enabled.
+Produced by a retrospective of the 2026-08-30/31 live-instance session, which
+stood up Bitbucket Data Center 10.4.2 in Docker with an unattended
+timebomb-license setup, seeded a repo with three planted bugs, opened a real
+PR, and ran `prxref review` against it with posting enabled.
**Two genuine bugs fell out of one live run, neither of which any hand-written
fixture had ever caught, for the life of the project:**
diff --git a/docs/llm.md b/docs/llm.md
index 92e400d..67225a5 100644
--- a/docs/llm.md
+++ b/docs/llm.md
@@ -1,10 +1,10 @@
# LLM Backends & Failover Architecture
-`prxref` connects to LLM inference endpoints using two interchangeable backends: a lightweight OpenAI-compatible plain-HTTP client or an optional in-process `litellm` wrapper. There is no default endpoint and no default model chain — `PRXREF_LLM_BASE_URL` and `PRXREF_LLM_MODELS` are required, and leaving either unset raises `ConfigError` (`prxref review` exits `2`).
+`prxref` reaches a model through one of four backends: a lightweight OpenAI-compatible plain-HTTP client, an optional in-process `litellm` wrapper, or one of two subscription CLI backends, `claude-cli` and `kiro-cli`, that run your own logged-in CLI (see [Subscription CLI backends](#subscription-cli-backends-claude-cli-and-kiro-cli)). There is no default model chain on any backend — `PRXREF_LLM_MODELS` is required, and leaving it unset raises `ConfigError` (`prxref review` exits `2`). There is no default endpoint either: `PRXREF_LLM_BASE_URL` is required by the `openai-compat` backend (and its `ferry`/`http` aliases), with the same exit `2` when unset, and is not used by any other backend (see [Optional Backend: litellm](#optional-backend-litellm)).
## Key Architectural Principles
-1. **No Provider Credentials in `prxref`:** `prxref` reads no third-party cloud provider credentials (no AWS IAM keys, no OpenAI keys, no Google Cloud keys, and no Anthropic API keys). All provider credentials, quota pools, and upstream authentication live securely behind the inference proxy endpoint.
+1. **No Provider Credentials in `prxref`:** `prxref` never looks up, stores, or uses a provider credential (no AWS IAM keys, no OpenAI keys, no Google Cloud keys, and no Anthropic API keys), and its own settings are provider-neutral `PRXREF_*` names. The one key it sends is `PRXREF_LLM_API_KEY`, as a bearer token to the `openai-compat` endpoint you configured. Every provider key lives behind that endpoint, in the provider SDK's own environment (`litellm`), or inside your own logged-in CLI (`claude-cli`, `kiro-cli`). A CLI backend hands its CLI the environment `prxref` was started with; `claude-cli` first removes a fixed list of credential-routing variable *names*, so the CLI falls back to its subscription login, and never reads their values.
2. **Fast Caller-Side Failover:** Fallback is implemented as a fast sequential loop over the model list. If a model encounters HTTP 429 (rate limit), HTTP >= 500 (server/upstream error), connection failures, or timeouts, the client immediately advances to the next model in the chain without same-model retries.
---
@@ -14,7 +14,7 @@
The primary and default backend communicates via plain HTTP requests with any OpenAI-compatible `/v1/chat/completions` server — a hosted router (OpenRouter, Together, Groq), a self-hosted gateway such as `llm-ferry`, or a local runtime such as vLLM or Ollama.
- **Default Backend Alias:** `PRXREF_LLM_BACKEND=openai-compat` (aliases: `ferry`, `http`).
-- **Endpoint URL:** `PRXREF_LLM_BASE_URL=https://llm.example.com/v1`. Required; there is no default.
+- **Endpoint URL:** `PRXREF_LLM_BASE_URL=https://llm.example.com/v1`. Required for this backend; there is no default.
- **API Key:** `PRXREF_LLM_API_KEY` (sent as `Authorization: Bearer `). Optional — leave empty for a local no-auth server.
- **Models:** Model names are whatever the endpoint accepts, listed cheapest first. Required; there is no default.
@@ -24,7 +24,7 @@ Four variables shape the request itself. All are optional, and a bad value exits
| Variable | Default | Effect on the request |
|---|---|---|
-| `PRXREF_LLM_MAX_TOKENS` | `4096` | `max_tokens` on every worker call. Must be > 0. This is a per-call budget threaded config → orchestrator → reviewer → `invoke`; the client never reads it. |
+| `PRXREF_LLM_MAX_TOKENS` | `4096` | `max_tokens` on every worker call on `openai-compat` and `litellm`; the CLI backends accept it and do not apply it (see [What is not applied](#what-is-not-applied)). Must be > 0. This is a per-call budget threaded config → orchestrator → reviewer → `invoke`; the client never reads it. |
| `PRXREF_LLM_TIMEOUT` | `45.0` | The client's default request timeout, in seconds. Must be > 0. It is a **per-model** deadline: a model that exceeds it is abandoned and the next in the chain is tried immediately, so a chain of three can take up to three timeouts. |
| `PRXREF_LLM_TEMPERATURE` | `0.0` (sent) | `temperature` in the payload. Must be finite and >= 0; no upper bound, since the maximum is provider-specific. Unset or empty sends the default `0.0` rather than omitting the field, so an identical diff reviews identically by default; a set value wins. `PRXREF_LLM_REASONING_EFFORT` keeps its own pass-through-unvalidated rule. |
| `PRXREF_LLM_SEED` | *(auto-derived)* | Top-level `seed` in the payload, OpenAI-compatible backends and `litellm` alike. Must be an integer >= 0 (`0` is a valid seed). Unset derives one random seed per process, shared by every client the run builds, so all LLM calls in a run pin the same sampling state; the run record's `sampling.seed` reports it. |
@@ -68,7 +68,8 @@ For environments running without a centralized inference gateway, `prxref` suppo
- **Backend Setting:** `PRXREF_LLM_BACKEND=litellm`
- **Installation:** `pip install 'prxref[litellm]'`
-- **Shared settings:** `PRXREF_LLM_MAX_TOKENS`, `PRXREF_LLM_TIMEOUT`, `PRXREF_LLM_TEMPERATURE`, and `PRXREF_LLM_SEED` apply here too — temperature resolves to the same `0.0` default when unset, and a configured seed is passed as `seed=` to `litellm.completion`. `PRXREF_LLM_REASONING_EFFORT` is openai-compat only.
+- **Endpoint URL: not used.** litellm resolves each model's own provider endpoint and reads that provider's credential (for example `OPENROUTER_API_KEY`) from its own environment, so `PRXREF_LLM_BASE_URL` is not required here and neither it nor `PRXREF_LLM_API_KEY` is ever passed to litellm. A set `PRXREF_LLM_BASE_URL` is ignored with one INFO line (`PRXREF_LLM_BASE_URL is set but not used by the litellm backend; ignoring it`), so a deployment that set a placeholder URL to get past the check older releases applied to every backend keeps working unchanged. To route through a LiteLLM **proxy**, which speaks the OpenAI API, use the `openai-compat` backend with `PRXREF_LLM_BASE_URL` pointing at the proxy.
+- **Shared settings:** `PRXREF_LLM_MAX_TOKENS`, `PRXREF_LLM_TIMEOUT`, `PRXREF_LLM_TEMPERATURE`, and `PRXREF_LLM_SEED` apply here too — temperature resolves to the same `0.0` default when unset, and the seed, configured or else auto-derived, is passed as `seed=` to `litellm.completion`. `PRXREF_LLM_REASONING_EFFORT` is not applied by `litellm`: it reaches only `openai-compat` (as `reasoning_effort` in the payload) and `claude-cli` (as `--effort`), and `kiro-cli` ignores it too.
### Configuration Example
@@ -85,15 +86,138 @@ PRXREF_LLM_MODELS=bedrock/anthropic.claude-3-7-sonnet-20250219-v1:0,vertex_ai/ge
---
+## Subscription CLI backends: `claude-cli` and `kiro-cli`
+
+These two backends review with the Claude Code CLI or the Kiro CLI that is already installed and logged in on your machine, so the review runs on your own subscription instead of an API key. Every model attempt starts one CLI process. `PRXREF_LLM_MODELS` is walked as a failover chain exactly as on the other backends, and every posted comment still names the model.
+
+> **Policy.** These backends only run *your own* locally installed, logged-in CLI, for *your own* use. prxref never ships, stores, or brokers subscription credentials. Anthropic's Agent SDK terms do not allow third-party products to offer claude.ai login or subscription rate limits without approval, so do not use `claude-cli` for a team, a shared webhook, or CI: use an API key, Workload Identity Federation, or Bedrock/Vertex/Foundry through `openai-compat` or `litellm`. prxref's guidance for `kiro-cli` is the same: a developer's own machine, not CI or the webhook daemon. The Docker image ships neither CLI; see [CLI Model Backends in Docker and CI](deploy.md#6-cli-model-backends-in-docker-and-ci).
+
+### Requirements
+
+- **The CLI is installed.** `claude` or `kiro-cli` must be on `PATH`, or `PRXREF_LLM_CLI_PATH` must name it (`~` is expanded, and a bare name is looked up on `PATH`). The binary is resolved when the LLM client is built, before any network call, and a missing or non-executable one exits `2`:
+
+ ```
+ configuration error: PRXREF_LLM_BACKEND: kiro-cli needs the 'kiro-cli' CLI, which was not found on PATH; install it and log in, or set PRXREF_LLM_CLI_PATH to its absolute path
+ ```
+
+- **The CLI is logged in.** For `claude`, run `claude` once and `/login`; in a headless shell, create a token with `claude setup-token` and export it as `CLAUDE_CODE_OAUTH_TOKEN`, which is passed through to the CLI. For `kiro-cli`, the browser login is enough, and `kiro-cli whoami` shows the account it uses. A `KIRO_API_KEY` in the environment is passed through unchanged.
+- A logged-out CLI is not a configuration error. Every model fails, the review fails the way an unreachable endpoint does, and `prxref review` exits `0` — `1` under `PRXREF_FAIL_ON=error` or `any`, like any review that does not complete (see [Troubleshooting](#troubleshooting)).
+
+### Configuration example
+
+```bash
+PRXREF_LLM_BACKEND=claude-cli
+PRXREF_LLM_MODELS=sonnet,opus
+PRXREF_LLM_TIMEOUT=120
+```
+
+```bash
+PRXREF_LLM_BACKEND=kiro-cli
+PRXREF_LLM_MODELS=claude-haiku-4.5,claude-sonnet-4.5
+PRXREF_LLM_TIMEOUT=120
+```
+
+`PRXREF_LLM_CLI_PATH` and `PRXREF_LLM_CLI_CONCURRENCY` are the two settings only these backends read; see [env-vars.md](env-vars.md).
+
+### What runs
+
+`claude-cli`, one process per model attempt:
+
+```
+claude -p --model --output-format stream-json --verbose --tools "" --setting-sources "" --strict-mcp-config --no-session-persistence --max-turns 1 --system-prompt-file
+```
+
+`--effort ` is appended when `PRXREF_LLM_REASONING_EFFORT` is set. The CLI runs with no built-in tools, no settings files, no MCP servers, no saved session, and a single turn.
+
+`kiro-cli`, one process per model attempt:
+
+```
+kiro-cli chat --no-interactive --agent prxref-review --output-format stream-json --trust-tools= --agent-engine v2
+```
+
+prxref asks for the v2 agent engine because the v1 engine does not emit `stream-json`, and v2 does not apply a `--model` flag, so each attempt writes the agent file `.kiro/agents/prxref-review.json` into its working directory. That file carries the system prompt and the model, and allows no tools, no MCP servers and no resources (`"tools": []`, `"allowedTools": []`, `"mcpServers": {}`, `"includeMcpJson": false`, `"resources": []`); `--trust-tools=` trusts none either. Whether Kiro still adds user-level configuration, such as `~/.kiro/steering/`, to a working-directory agent has not been verified, and the agent file cannot turn it off.
+
+For both CLIs:
+
+- The process is started from an argument list, never through a shell, and the user message (the diff) goes on stdin, never into the arguments.
+- Each attempt runs in a fresh temporary working directory that is removed afterwards, whether the call answered, failed or timed out. For `claude-cli` it is empty and the system prompt file sits beside it; for `kiro-cli` it holds only the agent file.
+- `json_mode` calls append one fixed "respond with exactly one JSON object" instruction to the system prompt. Any code fence the model still adds is stripped by the reviewer's lenient parse.
+
+### Environment
+
+- **`claude-cli`** hands the CLI `prxref`'s environment minus eight names: `ANTHROPIC_API_KEY`, `ANTHROPIC_AUTH_TOKEN`, `ANTHROPIC_BASE_URL`, `ANTHROPIC_PROFILE`, `CLAUDE_CODE_USE_BEDROCK`, `CLAUDE_CODE_USE_VERTEX`, `CLAUDE_CODE_USE_FOUNDRY` and `CLAUDE_CODE_SIMPLE`. Each would move the call off your subscription login: an API key always wins in `-p` mode, a gateway token or base URL re-points the CLI, the three `USE_*` flags route it to a cloud provider, a profile selects an organization identity, and bare mode ignores the OAuth login. Only the names are removed; their values are never read. `CLAUDE_CODE_OAUTH_TOKEN`, `HOME` and `CLAUDE_CONFIG_DIR` are kept, because they are how the CLI finds your login.
+- **Managed settings are out of prxref's reach.** An organization's managed `apiKeyHelper` or forced gateway still loads under `--setting-sources ""`. The CLI reports which credential it used: the INFO line of every answered `claude-cli` call ends `auth=`, and a value other than `none` logs one WARNING, `claude-cli: the CLI reports apiKeySource=…, so this call is NOT on your subscription login (check managed settings / apiKeyHelper)`.
+- **`kiro-cli`** hands the CLI the environment unchanged.
+
+### What is not applied
+
+| Setting | `claude-cli` | `kiro-cli` |
+|---|---|---|
+| `PRXREF_LLM_BASE_URL` | Ignored, with one INFO line when set | Ignored, with one INFO line when set |
+| `PRXREF_LLM_API_KEY` | Ignored | Ignored |
+| `PRXREF_LLM_MAX_TOKENS` | Not applied | Not applied |
+| `PRXREF_LLM_TEMPERATURE`, `PRXREF_LLM_SEED` | Not applied, with one WARNING when set | Not applied, with one WARNING when set |
+| `PRXREF_LLM_REASONING_EFFORT` | `--effort `, passed unvalidated | Not applied, with one INFO line when set |
+
+Neither CLI takes a per-call output budget. `PRXREF_LLM_MAX_TOKENS` is deliberately not mapped to `CLAUDE_CODE_MAX_OUTPUT_TOKENS`: hitting that cap makes the CLI spend extra recovery turns and still end in an error. A `CLAUDE_CODE_MAX_OUTPUT_TOKENS` you export yourself reaches the CLI unchanged.
+
+Neither CLI takes a temperature or a seed either, so a review on a CLI backend is less reproducible than one on `openai-compat` or `litellm` (see [Determinism](#determinism-what-is-pinned-and-what-still-varies)), and the run record's `sampling` field shows `"temperature": null` and `"seed": null`.
+
+### Models
+
+- **`claude-cli`** takes whatever `claude --model` takes: an alias such as `sonnet`, `opus` or `haiku`, or a full model id. The attribution and the logs name the model the CLI reports it ran, so an alias shows up as its full id. A model the CLI rejects as unknown (a 404, or its unrecognized-model marker on stderr) is skipped for the rest of the run, with one WARNING.
+- **`kiro-cli`** takes the ids `kiro-cli chat --list-models` prints, such as `claude-haiku-4.5`. The model goes into the agent file because the v2 engine ignores `--model`: it warns `failed to set model … Method not found` and runs its `auto` model. Kiro does not report which model ran, so the attribution names the model you configured. An unknown id fails that model as `: prompt error: Internal error (possibly an unknown model; check kiro-cli chat --list-models)`. Because Kiro's error does not name the model, prxref does not skip it for the rest of the run: every call tries it again before moving on down the chain.
+
+### Concurrency and timeouts
+
+- `PRXREF_LLM_CLI_CONCURRENCY` (default `2`) caps the CLI processes one client runs at once. The review's workers queue for a free slot, and the wait does not count against the timeout. A subscription's rate window belongs to your account, so a higher cap spends it faster.
+- `PRXREF_LLM_TIMEOUT` is each model's wall-clock deadline, and it includes the CLI's own start-up: about 2.3 s for `claude` on a tiny prompt whose model time was 1.4 s, and 3.6–6.7 s for `kiro-cli`, as observed while designing these backends. Live `kiro-cli` calls on 2026-09-23 were much slower, 24–27 s for a one-line prompt and 28–37 s for a review call, so for `kiro-cli` `120` is the floor, not a comfortable margin. The 45 s default is sized for HTTP; use `120` or more. A model that misses its deadline has its whole process group killed, the chain moves on, and the review's zero-context retry applies exactly as it does over HTTP.
+
+### Tokens, cost and credits
+
+- **`claude-cli`** counts cache-creation and cache-read tokens as input tokens, because that is prompt the model read.
+- **`kiro-cli`** reports no token counts, so every Kiro call counts `0` tokens and the attribution reads `0 tok`. Kiro meters credits, not dollars. The INFO line of every answered call ends with the credits Kiro metered and its session id:
+
+ ```
+ INFO llm attempt 1/1 ok: backend=kiro-cli model=claude-haiku-4.5 3570ms in=0 out=0 finish=end_turn credits=0.0060 session=
+ ```
+
+What either backend contributes to `cost_usd` is set out in [Cost accounting](#cost-accounting).
+
+### Privacy
+
+- **`kiro-cli`** saves every chat under `~/.kiro/sessions/cli/.json` and `.jsonl`, prompt included, so every diff it reviewed is kept there. prxref does not delete these files, because deleting a session takes Kiro about 11 seconds. Use the `session=` id from the INFO line with `kiro-cli chat --delete-session `, or prune the directory yourself.
+- **`claude-cli`** runs with `--no-session-persistence`, so the CLI saves no transcript of a review.
+
+### Troubleshooting
+
+| What you see | What it means |
+|---|---|
+| Exit `2`, `configuration error: PRXREF_LLM_BACKEND: … needs the '…' CLI, which was not found on PATH` | The CLI is not installed, or not on the `PATH` prxref runs with. Install it, or set `PRXREF_LLM_CLI_PATH`. |
+| Exit `2`, `configuration error: PRXREF_LLM_CLI_PATH: '…' is not an executable file` | The override names a missing file or one without execute permission. |
+| Every model fails, and each reason quotes the CLI's own login or authentication error | The CLI is logged out. Log in again (for headless `claude`, refresh `CLAUDE_CODE_OAUTH_TOKEN`). |
+| `: prompt error: Internal error (possibly an unknown model; …)` | Kiro does not know that model id. Check `kiro-cli chat --list-models`. |
+| `: engine error: …` | Kiro failed before the model ran, for example because the installed `kiro-cli` cannot start the v2 agent engine. Update `kiro-cli`. |
+| WARNING `claude-cli: subscription rate limit status=… type=… utilization=…` | Your subscription window is close to its limit. A `rejected` status fails the call as `rate limited`. |
+| WARNING `claude-cli: the CLI reports apiKeySource=…` | Managed settings or an `apiKeyHelper` put the call on an API key, not your subscription. |
+| WARNING `claude-cli: the CLI loaded tools/MCP servers despite --tools '' --strict-mcp-config` | The CLI no longer honours the isolation flags; its options may have changed. |
+| `: timeout (TimeoutExpired after 45s)` | Start-up plus the answer took longer than `PRXREF_LLM_TIMEOUT`. Raise it to `120` or more. |
+
+---
+
## Determinism: what is pinned, and what still varies
- `PRXREF_LLM_TEMPERATURE` defaults to `0.0`, and `0.0` is **sent** on the wire
- rather than omitted.
-- `PRXREF_LLM_SEED` is sent on every call, on both backends: the configured
+ by the two API backends, `openai-compat` and `litellm`, rather than omitted.
+- `PRXREF_LLM_SEED` is sent on every call by the two API backends,
+ `openai-compat` and `litellm`: the configured
value when set, else one random seed derived per process and shared by every
client the run builds — temperature 0 alone cannot pin hosted inference
(issue #56), so an unseeded run still varies call to call. The `sampling`
- field reports which seed was in force.
+ field reports which seed was in force. The CLI backends, `claude-cli` and
+ `kiro-cli`, send no seed and no temperature: setting either logs one WARNING
+ that it is not applied, and `sampling` reports both as `null` (see
+ [What is not applied](#what-is-not-applied)).
- **Neither makes a review bit-reproducible.** Providers vary by system
fingerprint, load-balanced backends serve the same model from different
hardware, MoE routing shifts with batch composition, and many gateways accept
@@ -108,6 +232,75 @@ PRXREF_LLM_MODELS=bedrock/anthropic.claude-3-7-sonnet-20250219-v1:0,vertex_ai/ge
title), never by the order the workers happened to return in. The same
findings in any arrival order therefore produce the same review.
+## Cost accounting
+
+Every run record carries `cost_usd` (USD) and `cost_estimated` (bool), on every exit, and `--format json` prints both. A run's cost is in exactly one of three states:
+
+- **Reported.** The backend returned a dollar figure for each call. A reported figure always wins, even over a price-table entry for the same model.
+- **Estimated.** No figure came back, but `PRXREF_PRICE_TABLE` prices the model. `cost_estimated` is `true`.
+- **Unknown.** Neither. `cost_usd` is `null`: never `0`, and never a partial sum of the calls that were priced.
+
+### Where a reported figure comes from
+
+| Backend | Source | `cost_source` |
+|---|---|---|
+| `openai-compat` (`ferry`, `http`) | The response body's `usage.cost` (OpenRouter returns it on every completion without being asked), else the `x-litellm-response-cost` response header that a LiteLLM gateway or `llm-ferry` sets. The body value must be a JSON number. LiteLLM omits the header when it cannot price the call **and** when the cost is zero, so a free model behind a gateway reports nothing. | `usage.cost` / `x-litellm-response-cost` |
+| `litellm` | `response_cost`, which litellm computes from its own price map. prxref never calls `litellm.completion_cost()`. | `litellm` |
+| `claude-cli` | The CLI's `total_cost_usd`. On a subscription this is the **API-equivalent cost at list price, not your subscription bill**, so the `-v` line and the posted attribution label it `(API-equivalent)` (see [Where the cost shows](#where-the-cost-shows)). | `claude-cli` |
+| `kiro-cli` | None. Kiro reports credits, not dollars, and no token counts, so a price-table entry cannot estimate it either: a run on `kiro-cli` always reads "cost unknown". | — |
+
+prxref sends nothing extra to get a figure: the request never carries `usage: {"include": true}`. A figure that is not a finite number `>= 0` (a negative, `NaN`, a string in the body, an empty or `None` header) counts as no figure.
+
+### The price table
+
+`PRXREF_PRICE_TABLE` is inline JSON (the first non-space character is `{`) or a path to a JSON file. It maps a model name to USD per **million** tokens:
+
+```bash
+PRXREF_PRICE_TABLE='{"openai/gpt-4o-mini": {"input": 0.15, "output": 0.60}}'
+PRXREF_PRICE_TABLE=./prxref-prices.json
+```
+
+- The lookup is on the **exact** model name the call reported, which is the name shown as `model=` in the attribution. It can differ from the name in `PRXREF_LLM_MODELS`, because the endpoint's answer names the model. There is no prefix or pattern matching.
+- The table is only consulted for a call with no reported figure, and only when that call counted input tokens. Zero input tokens means the backend reported no usage, and an estimate would be a fake `$0`.
+- Give a free or local model a zero entry (`{"input": 0, "output": 0}`). Without one, a run on it reads "cost unknown", never `$0`.
+- An estimate prices every input token at the list rate, so it ignores prompt-cache discounts that a provider's own figure reflects. That is one more reason a reported figure always wins.
+- The schema is strict. Invalid JSON, an unreadable file, a missing or unknown field (`"ouput"`), a duplicate model, or a price that is not a finite number `>= 0` raises `ConfigError` naming `PRXREF_PRICE_TABLE`, and `prxref review` exits `2`.
+
+When a run ends up unknown because some model had neither a reported figure nor a usable table entry, prxref logs one INFO line naming the model(s), in the exact spelling to key the table on:
+
+```
+cost unknown: no reported cost and no usable PRXREF_PRICE_TABLE estimate for model(s) 'openai/gpt-4o-mini'
+```
+
+### Which calls count
+
+A review is its chunk workers plus the systemic sweep, and the total covers the same calls as the run's token counts:
+
+- A call whose response **arrived** is counted, including one that was then truncated or failed to parse. It was billed.
+- A call that raised (a timeout, a connection error, an HTTP error) returned nothing and adds nothing. A provider that bills abandoned generations may charge more than `cost_usd` says.
+- A run that sent requests and got no response back at all is unknown (`null`).
+- A run that made no LLM request (an empty diff, or a forge or diff error before the review) costs a known `0.0`.
+- Inside one `openai-compat` call, truncated completions that the fallback chain moved past were billed too, so they are added to that call's figure. Its token counts still cover only the answering model. If any of those completions came back without a figure, the call's figure is unknown.
+- The timeout retry (the one re-run with `context_lines=0`) replaces the first attempt's result, cost included, exactly as it replaces its tokens.
+- If the total cannot be computed at all, for example because a library caller passed a malformed table object, the run logs a WARNING and its cost is unknown. Cost accounting never fails a review.
+
+### Where the cost shows
+
+- The run record and `--format json`: `cost_usd` and `cost_estimated`.
+- `prxref review -v`: `cost: $0.0007`, `$0.0007 (API-equivalent)`, `~$0.0007 (est.)` or `cost unknown` after the token count.
+- The JSONL trace (`PRXREF_TRACE_FILE`): the `run ok` and `run fail` events carry `cost_usd` and `cost_estimated`. Each `chunk ok` and `sweep ok` event carries that unit's reported `cost_usd`; estimates are computed for the run only, so a unit priced from the table shows `null` there.
+- The per-unit trace files (`PRXREF_TRACE_DIR`): each `.meta.json` carries `cost_usd` and `cost_source`.
+- The posted comment, only with `PRXREF_POST_COST=1`. The cost is appended as the **last** field of the summary's attribution line and of the error notice's:
+
+ ```
+ Reviewed by prxref · model=openai/gpt-4o-mini · 4619 tok · 3.1s · $0.0007
+ Reviewed by prxref · model=openai/gpt-4o-mini · 4619 tok · 3.1s · ~$0.0007 (est.)
+ Reviewed by prxref · model=openai/gpt-4o-mini · 4619 tok · 3.1s · cost unknown
+ Reviewed by prxref · model=claude-sonnet-5 · 7564 tok · 13.4s · $0.0202 (API-equivalent)
+ ```
+
+ `(API-equivalent)` appears on the `-v` line and in the attribution when every reported figure in the run came from `claude-cli`; an estimated run keeps `~… (est.)`, and `--format json` adds no label (each unit's `cost_source` in the `PRXREF_TRACE_DIR` meta files names the source). A notice posted before any LLM request says `$0.00`, and a cost below $0.0001 reads `<$0.0001`, never `$0.00`. Inline comments never carry a cost. With the flag off, which is the default, the attribution line is byte-identical to a build without cost accounting.
+
## Worker Prompt Context
Each worker sees one chunk's unified diff, trimmed to `PRXREF_CHUNK_CONTEXT_LINES` lines around every change. Two optional blocks are appended after the diff to answer the questions the diff alone cannot.
diff --git a/docs/quality.md b/docs/quality.md
index b469f04..65afd40 100644
--- a/docs/quality.md
+++ b/docs/quality.md
@@ -44,7 +44,7 @@ it (noted in the table).
| 5 | `apply_settled_thread_suppression` | Drops a finding that re-litigates a subject a thread already argued out. Line-independent by design. A thread with no path — a general, unanchored PR comment — is ignored by this pass, since it cannot be "same path" as any finding. |
| 6 | `apply_severity_consistency` | Rewrites only: findings sharing a normalized title are all raised to the group's maximum severity. |
| 7 | `apply_removal_claim_check` | Drops a claim that a **named** path was removed when the post-image still carries it. The removal verb must **govern** that path (`removed src/app.py`, `src/app.py was removed`); a bare "removed" elsewhere in the body is not a removal claim. |
-| 8 | `apply_hedge_gate` | Drops a finding whose own text conditions the defect on a precondition never established from the diff. |
+| 8 | `apply_hedge_gate` | Drops a finding whose own text conditions the defect on a precondition never established from the diff. One part of the body is not read, in any finding whatever its severity: after a `Spec:` marker (that exact spelling; the opening quote is optional), the text the finding copies verbatim from the injected spec digest, compared case-insensitively, up to a closing quote. A condition inside a real constraint belongs to the spec, not the model. Everything else is read: text the digest does not hold, so a made-up `Spec: "…"` hides nothing; every quote when no digest was injected (no spec sources, or an ungrounded run); and the title. Known limitation: a quote with no closing quote after its verbatim text, or one that departs from the digest before its closing quote, is exempt only up to the last quote mark inside its verbatim part (an apostrophe counts), and not at all when there is none. |
| 9 | `apply_quality_gate` | Severity vocabulary, confidence floor, per-review error cap. Returns its findings in content order. |
| 10 | `apply_sweep_dedup` | Drops a sweep finding that restates a chunk finding which **survived** the gate. |
| 11 | `apply_containment_note` | Decoration only: suffixes a throw/panic/crash finding that never named its containment boundary. |
@@ -53,6 +53,116 @@ Threads are fetched once per review, **before** the workers run and **after**
the stale-inline-comment prune — reading threads first would let a run suppress
its own findings against prxref's own stale comments and then delete them.
+## Severity map from team review rules
+
+When the team review-rules file declares a severity map
+(`PRXREF_REVIEW_RULES` / `--rules-file`; see
+[docs/review-rules.md](review-rules.md)), `apply_severity_map` runs **before
+pass 1**. It rewrites a team severity word the model wrote (`blocker`) to the
+prxref tier the map gives it (`error`), matching case-insensitively and with
+runs of whitespace collapsed. It runs first because every later pass reads the
+severity: consistency groups by it, the sweep boundary is re-derived from it,
+and the quality gate would drop `blocker` as `invalid severity: 'blocker'`.
+
+- It **drops nothing**, so it has no row in the drop-reason table below. A
+ word the map does not name passes through and still dies at the gate as
+ `invalid severity`.
+- It never rewrites a finding that already carries one of prxref's own
+ severities or a `drop_reason`. It keeps every other field, `scope`
+ included, and the list's length and order.
+- Without rules, or with rules that map nothing, the pass is not called.
+- When it rewrites any finding, prxref logs `severity map: rewrote N
+ finding(s) from team severity words` at INFO and the JSONL trace gets a
+ `rules remap` event with `findings=N`.
+
+The map never targets `spec`, so the pass never mints a spec finding.
+
+## Spec grounding
+
+A run is **grounded** when the spec digest it built holds at least one
+constraint line (`specs.constraint_count` above 0). Only a grounded digest is
+injected into the prompts. A digest with no constraint line is not injected at
+all: no spec sources, every source failed, nothing extracted, or a
+`PRXREF_SPEC_DIGEST_TOKENS` budget too small for one line. Every review unit
+then sees the no-specs text `(no specs provided for this review)`, under which
+the prompts make `spec` an illegal severity.
+
+`apply_spec_grounding` runs right after the severity map and before
+`apply_location_validation` (pass 1 above), over chunk and sweep findings
+alike:
+
+- On an ungrounded run it relabels every `spec` finding as `warning`. The
+ severity is compared trimmed and lower-cased, so `SPEC` counts. It never
+ drops a finding and never raises one to `spec`, and because it runs before
+ `apply_severity_consistency`, an ungrounded `spec` finding can never lift a
+ same-title sibling to `spec`.
+- When it relabels anything, one INFO line gives the count (`spec grounding:
+ relabelled N spec finding(s) as warning (no spec constraint was
+ injected)`), and the run trace gets one `specs relabel` event with
+ `findings: N`. This can happen on a run with no spec sources at all, when a
+ model emits `spec` unasked.
+- On a grounded run it changes nothing.
+
+A relabel is not a drop, so it has no `drop_reason`. After this pass a `spec`
+finding is filtered like any other. `apply_severity_consistency` ranks
+`error` > `warning` > `spec` > `outofscope`, so a same-title `warning` or
+`error` raises it. The confidence floor applies to it. It never counts toward
+`PRXREF_MAX_ERROR_FINDINGS` and never moves the verdict. The hedge gate's
+`Spec: "…"` exemption (pass 8) reads the injected digest only, so an
+ungrounded run exempts nothing.
+
+Since 0.14.0 the worker and sweep prompts carry spec text on every run, with
+spec sources or without. Their system half carries the `spec` severity and the
+spec-grounded rules. Their user half carries a `### Spec constraints` block
+that reads `(no specs provided for this review)` when nothing is injected.
+
+## Ticket scope
+
+With a ticket context configured (`--context-file` / `PRXREF_TICKET_CONTEXT_FILE`,
+see [Ticket Context and Scope](../README.md#ticket-context-and-scope)), every
+finding carries a `scope` of `in`, `out`, or `unknown` relative to that ticket.
+Scope is **orthogonal to every pass on this page**: no pass reads it, it never
+changes a severity or a confidence, and it never feeds the verdict, the
+confidence floor, the error cap or its tie-break, or `PRXREF_FAIL_ON`. It is
+not the `outofscope` severity either, which only means minor. A finding never
+gains a `drop_reason` for its scope.
+
+- **Only an active ticket can set it.** The model is asked for a scope only
+ when the ticket has text. Raw chunk and sweep findings go through
+ `_enforce_scope` before the first pass: without a ticket, or with an empty
+ one, every finding is `unknown` whatever the model returned.
+- **The vocabulary is strict.** `triage.normalize_scope` keeps a value only
+ when it is exactly `in`, `out`, or `unknown` after trimming and case-folding.
+ `"In scope"`, `"yes"`, a boolean, or a missing key is `unknown`. There is no
+ synonym table, because a lenient mapping would turn a malformed answer into
+ a confident one.
+- **Sweep dedup ignores it.** `apply_sweep_dedup` matches on file and
+ normalized title, so a sweep finding that restates a surviving chunk finding
+ is still dropped when the two copies disagree on scope. The identity used to
+ re-derive the chunk/sweep boundary across the gate includes `scope`, so
+ neither copy's scope ends up on the other.
+- **It orders the inline batch within a severity.** When
+ `PRXREF_MAX_INLINE_COMMENTS` leaves room for only some findings, severity
+ decides first. Within one severity, an `out` finding yields its inline slot
+ to `in` and `unknown` ones, and confidence and content break the rest of the
+ ties. With no active ticket every scope is `unknown`, so the order is exactly
+ the severity-only one.
+- **Truncation can change the state.** Only the first
+ `PRXREF_TICKET_CONTEXT_MAX_CHARS` characters reach the model, and acceptance
+ criteria are detected on that kept text. A long ticket whose criteria come
+ after the cap therefore reads as a ticket without criteria, and the summary
+ says scope was judged from its description alone.
+
+## Replay runs and the thread passes
+
+A `--no-threads` replay gives passes 4 and 5 (`apply_thread_dedup` and
+`apply_settled_thread_suppression`) an empty thread list, so they drop nothing,
+and a `--diff-file` replay with no `--pr-url` has no threads to start with. A
+replay at pinned SHAs WITHOUT `--no-threads` still dedups against the PR's
+*current* threads, which may postdate the pinned head; the CLI logs a warning
+saying so. The stale-inline-comment prune never runs on a replay, because a
+replay never posts. See the README's "Replay Mode (Evaluation)".
+
## Drop reasons
| `drop_reason` | Pass | Meaning |
@@ -64,7 +174,7 @@ its own findings against prxref's own stale comments and then delete them.
| `settled in thread: ` | `apply_settled_thread_suppression` | A thread on the same path already argued this subject out. A **resolved** thread still settles it — resolution is a decision, not an expiry. |
| `claims removal of a path present in the post-image: ` | `apply_removal_claim_check` | A removal verb governs this path, and every path the claim names is still present after the PR lands. |
| `hedged: ""` | `apply_hedge_gate` | The finding's own text conditions the defect on something the model never established. |
-| `invalid severity: ''` | `apply_quality_gate` | Severity outside {`error`, `warning`, `outofscope`}. |
+| `invalid severity: ''` | `apply_quality_gate` | Severity outside {`error`, `warning`, `spec`, `outofscope`}. |
| `confidence below floor ` | `apply_quality_gate` | Below `PRXREF_CONFIDENCE_FLOOR`. |
| `error cap exceeded (max )` | `apply_quality_gate` | Beyond `PRXREF_MAX_ERROR_FINDINGS`. Ties break on finding content, not arrival order, so the cap is reproducible. |
| `duplicate of chunk finding` | `apply_sweep_dedup` | A whole-diff sweep finding restates a chunk finding that already survived the gate. |
diff --git a/docs/review-rules.md b/docs/review-rules.md
new file mode 100644
index 0000000..de41736
--- /dev/null
+++ b/docs/review-rules.md
@@ -0,0 +1,285 @@
+# Team Review Rules
+
+Most teams already keep a written review checklist: "every network call has a
+timeout", "no migration without a rollback", "a TODO names its ticket". Point
+prxref at that file and every review unit reads it as part of its
+instructions.
+
+```bash
+prxref review --pr-url https://github.com/acme/widget/pull/42 --rules-file .prxref/rules.md
+# or, for every run of this process (the flag wins when both are set):
+export PRXREF_REVIEW_RULES=/etc/prxref/rules.md
+```
+
+- The file is Markdown or plain text, UTF-8, with optional front matter.
+- `PRXREF_REVIEW_RULES` names it for every run. `--rules-file PATH` names it
+ for one run and wins over the variable. `--rules-file ""` turns an
+ environment-configured file off for one run.
+- Unset (the default), nothing changes: the prompts, the trace files, the
+ JSONL trace and the exit code are what they would be without the feature,
+ and the run record carries `review_rules: null`.
+
+**Read the rules from a trusted checkout, never from the pull request under
+review.** See [CI safety](#ci-safety-read-the-rules-from-something-the-pr-cannot-change).
+
+## Where the rules go
+
+The rules are **operator policy**, so they go in the **system** half of each
+prompt, after prxref's own instructions, under a `## Team review rules`
+heading. They never touch the user half, which holds the PR's own data (the
+title, the description, the ticket context, the spec constraints, the diff).
+Every chunk worker gets them, and so does the whole-PR systemic sweep, each
+with its own framing:
+
+- **A chunk worker** is told to check its chunk against the rules as well, and
+ that everything above still binds: a finding must cite a line of the diff,
+ follow the Confidence and No Speculation rules, and use only the Severity
+ Vocabulary. A rule the diff cannot show evidence for — a test run, a linked
+ ticket, a sign-off — produces no finding.
+- **The sweep** is told to apply only the rules about a whole-PR or cross-file
+ property the digest can show ("every new migration ships a rollback"). Rules
+ about individual lines belong to the chunk workers; repeating them in the
+ sweep would only duplicate their findings.
+
+The body sits inside `` … `` tags, so the file's own
+`##` headings never read as siblings of prxref's sections. The block for a
+chunk worker looks like this:
+
+```text
+## Team review rules
+
+The team that owns this repository reviews changes against the rules below. Check this chunk against them as well. […]
+
+Team severity words map onto that vocabulary: `blocker` → `error`; `major` → `warning`; `must fix` → `warning`; `nit` → `outofscope`. Classify a problem by the team's definition, then write the mapped word in `severity`.
+
+
+# Team rules
+
+- blocker: any network call without an explicit timeout.
+- major: a function longer than 80 lines.
+- nit: a TODO without a ticket id.
+
+```
+
+That is the block for the example file in
+[Front matter and the severity map](#front-matter-and-the-severity-map).
+
+- The severity paragraph appears only when the file maps severity words (see
+ below).
+- The `` section appears only when the body has text.
+- A file with neither a body nor a map adds nothing at all, not even the
+ heading.
+- The rules are added verbatim. Braces such as `{diff}`, or a line reading
+ `## Review Context`, render literally and cannot move anything else in the
+ prompt.
+- The timeout retry of a chunk keeps the rules. The retry trims bulk context,
+ and the rules are policy, not context.
+
+To see exactly what each unit received, run with `--trace-dir DIR`: the rules
+block is in `DIR/chunk0.system.md` … and `DIR/sweep.system.md`, and in no
+`*.user.md`.
+
+## Front matter and the severity map
+
+A team usually has its own severity words. Map them onto prxref's tiers in a
+`severity:` block of the file's front matter:
+
+```markdown
+---
+name: team-review
+description: |
+ The checklist every reviewer on this team uses.
+severity:
+ blocker: error # a merge blocker
+ major: warning
+ "Must Fix": warning
+ nit: outofscope
+---
+# Team rules
+
+- blocker: any network call without an explicit timeout.
+- major: a function longer than 80 lines.
+- nit: a TODO without a ticket id.
+```
+
+The model is shown the map and asked to write prxref's tier. If it writes the
+team's word anyway, a deterministic pass rewrites it before every other
+quality pass, so `blocker` reaches the quality gate as `error` instead of
+being dropped as `invalid severity: 'blocker'`. See
+[docs/quality.md](quality.md#severity-map-from-team-review-rules).
+
+**The grammar, exactly:**
+
+- **The fence.** Front matter exists only when the file's first line is `---`
+ and a later line is also `---`. The first such later line closes it, and
+ the body starts after it. A `---` anywhere else is an ordinary Markdown rule.
+ A first-line `---` that never closes is logged as a warning, and the whole
+ file is then rules text.
+- **Comments and blank lines.** Inside the fence, `#` at the start of a line
+ or after a space or tab starts a comment, and blank lines are skipped.
+- **Only `severity:` is read.** Every other top-level key (`name:`,
+ `description:`, …) is ignored and named once in an INFO log line, and any
+ indented lines under it are skipped. So a Claude-style skill file,
+ multi-line `description: |` included, works unmodified.
+- **The map.** `severity:` has nothing after the colon, and each indented line
+ under it is `: `. Either side may be quoted, and a word may
+ contain spaces (`Must Fix`). Words are case-insensitive, and runs of
+ whitespace inside them count as one space.
+- **The tiers** are `error`, `warning` and `outofscope`. `spec` is reserved
+ for findings grounded in a quoted spec constraint (`PRXREF_SPEC_SOURCES` /
+ `--spec`), so it is never a legal target: a team word mapped onto it would
+ mint spec findings on runs with no spec at all.
+- **prxref's own words** (`error`, `warning`, `spec`, `outofscope`) cannot be
+ remapped. The identity `error: error` is allowed and ignored.
+- An empty `severity:` block is legal and maps nothing.
+
+Each of these is a configuration error (exit `2`), reported as
+`: :: `:
+
+- an inline value (`severity: {blocker: error}`, `severity: error`);
+- a second `severity:` key;
+- an entry that is not `: `: a YAML list item (`- blocker`), or a
+ nested block;
+- an unknown tier (`blocker: eror`), or `spec`;
+- a remap of one of prxref's own words (`warning: error`);
+- one word mapped to two different tiers.
+
+For example:
+
+```text
+configuration error: --rules-file: .prxref/rules.md:7: unknown severity 'eror' for 'blocker'; expected one of error, outofscope, warning
+```
+
+The map is nested under `severity:`, rather than written as flat top-level
+`word: tier` lines, so the file can carry other front matter (a skill file's
+`name:` and `description:`). It also keeps the map strict: a typo in a tier
+is an error rather than something silently skipped.
+
+## The cap, the hash and the run record
+
+- **The cap.** `PRXREF_REVIEW_RULES_MAX_CHARS` (default `12000`, must be
+ greater than 0) caps the **body**: the text after the front matter, with
+ surrounding whitespace stripped. A longer body is cut at the cap, and the
+ block then ends with
+ `[team rules truncated: only the first 12000 of 18344 characters are shown]`.
+ prxref also logs one WARNING per run that names
+ `PRXREF_REVIEW_RULES_MAX_CHARS`.
+- **The hash.** `sha256` covers the raw bytes of the whole file, front matter
+ included, before decoding and capping. It equals `shasum -a 256 FILE`, it
+ does not change when you change the cap, and it changes when any byte of
+ the file does. It is how you tell which version of the rules reviewed which
+ PR.
+- **Strict text.** The file must be UTF-8 (a leading BOM is dropped, CRLF and
+ CR become LF) with no NUL bytes.
+
+Every review result carries a `review_rules` record, `null` when no rules are
+configured:
+
+```json
+{"path": ".prxref/rules.md", "sha256": "<64 hex>", "chars": 18344, "max_chars": 12000,
+ "truncated": true,
+ "severity_map": {"blocker": "error", "major": "warning", "must fix": "warning", "nit": "outofscope"}}
+```
+
+- `path` is the path as configured, not resolved.
+- `chars` and `truncated` describe the body after the front matter; `sha256`
+ covers the whole file.
+- The record never carries the rules text.
+
+It appears in these places:
+
+| Where | What |
+|---|---|
+| `--format json` | the `review_rules` key, always present, `null` when off |
+| `-v` text output | `rules: .prxref/rules.md sha256= chars=18344 (truncated at 12000)` |
+| JSONL trace (`PRXREF_TRACE_FILE`) | one `rules ok` event whose meta is the record, right after `run start`; a `rules remap` event with `findings=` when the map rewrote any finding |
+| `--trace-dir` | the rules block itself, in every `.system.md` |
+
+Nothing about the rules is added to the posted comments.
+
+## Cost
+
+The rules ride **every** review unit: each chunk and the sweep. So they add
+about `chars / 4 × (chunks + 1)` input tokens per run. At the 12000-character
+default that is roughly 3000 tokens per unit. They also count toward the
+prefill share of `PRXREF_LLM_TIMEOUT`, and the timeout retry keeps them.
+Keep the file to rules a reviewer can check from a diff.
+
+## Errors
+
+| Situation | Outcome |
+|---|---|
+| unset, `""`, whitespace, or `--rules-file ""` | no rules; `review_rules: null` |
+| the path is a URL (`https://…`) | configuration error: rules must be a local file path |
+| missing file, a directory, a FIFO or device, permission denied | configuration error: `cannot read rules file '': ` |
+| a path inside the working directory that symlinks out of it | configuration error: `… resolves outside the working directory` |
+| invalid UTF-8, or NUL bytes | configuration error |
+| malformed `severity:` block | configuration error with `:` |
+| `PRXREF_REVIEW_RULES_MAX_CHARS` 0, negative or not an integer | configuration error naming the variable |
+| a `---` first line that never closes | warning; the whole file is rules text |
+| body longer than the cap | warning; `truncated: true`; a truncation line in the block |
+| empty body and no map | warning; the record is still present; no block |
+| the model writes a mapped team word | rewritten to its tier before every quality pass |
+| the model writes a word that is not mapped | unchanged behaviour: dropped as `invalid severity` |
+
+Every configuration error names the input that supplied the path:
+`--rules-file` when the flag was given, else `PRXREF_REVIEW_RULES`. The file
+is read before any network call, so `prxref review` exits `2` without touching
+the forge or the model. This holds under `PRXREF_FAIL_ON` as well.
+
+## CI safety: read the rules from something the PR cannot change
+
+In CI, the workspace is usually the pull request's own code: a GitHub
+Actions `pull_request` workflow checks out the PR's merge commit, and a GitLab
+merge-request pipeline runs on the MR's source branch. So
+`--rules-file .prxref/rules.md` reads **the PR's copy** of the rules, and a PR
+could rewrite its own review rules. Read them from somewhere the PR cannot
+reach instead:
+
+- **The target branch, with plain git.** Fetch the branch the PR merges into
+ and copy the file out of it:
+
+ ```bash
+ git fetch --depth=1 origin "$TARGET_BRANCH"
+ git show FETCH_HEAD:.prxref/rules.md > "$RUNNER_TEMP/prxref-rules.md"
+ prxref review --pr-url "$PR_URL" --rules-file "$RUNNER_TEMP/prxref-rules.md"
+ ```
+
+ The target branch is `${{ github.base_ref }}` on GitHub Actions,
+ `$CI_MERGE_REQUEST_TARGET_BRANCH_NAME` on GitLab CI and
+ `$BITBUCKET_PR_DESTINATION_BRANCH` on Bitbucket Pipelines. `$RUNNER_TEMP` is
+ GitHub's per-job temporary directory; elsewhere, use any directory outside
+ the checkout, such as one from `mktemp -d`.
+- **Outside the repository.** A GitLab CI/CD variable of type *File* named
+ `PRXREF_REVIEW_RULES` works directly: the runner writes the value to a
+ temporary file and puts that file's path in the variable. A file baked into
+ the runner image, or one from your CI's secure-files store, works too.
+
+An absolute path outside the working directory, like the two above, is read
+as given. A path **inside** the working directory must still resolve inside
+it once its symlinks are followed, so a PR that commits
+`.prxref/rules.md -> /some/other/file` gets a configuration error rather than
+a read of that file.
+
+**The residual risk.** A PR that can edit the pipeline definition itself can
+change anything the pipeline does, the rules included. Protect the pipeline
+files with required review (for example `CODEOWNERS`), or run prxref as the
+webhook daemon, whose configuration no PR can reach.
+
+## The webhook daemon
+
+`prxref serve` reads `PRXREF_REVIEW_RULES` from its own environment and
+re-reads the file for every review. Editing the file therefore takes effect on
+the next webhook with no restart, and the recorded `sha256` says which version
+reviewed which PR. A webhook only ever delivers a PR URL, and the daemon has
+no checkout of the PR, so nothing a PR contains can reach the rules loader. A
+bad rules file fails each review with the configuration error in the daemon's
+log.
+
+If the daemon is started with a **relative** path, the path is resolved
+against the daemon's working directory. Start it from a directory whose
+contents no PR can change.
+
+The loader keeps that guarantee by design. It refuses URLs, it never reads the
+rules through the forge (for example, from the PR's head commit), and nothing
+builds the path from PR data.
diff --git a/docs/spec-grounded-review.md b/docs/spec-grounded-review.md
new file mode 100644
index 0000000..68b06df
--- /dev/null
+++ b/docs/spec-grounded-review.md
@@ -0,0 +1,942 @@
+# Spec-Grounded Review — Design
+
+> **This is a design record, not the manual.** It is the pre-implementation
+> design, written against the tree at 0.11.x; the feature shipped in 0.14.0.
+> Its `file:line` citations, line ranges and version numbers are historical
+> and no longer match the source, so resolve a citation by the symbol it
+> names. Where what shipped differs materially from the design, an
+> **As built (0.14.0)** note says so in place, and the shipped behaviour wins.
+>
+> - The dataset contract for `tests/evals/` is
+> [tests/evals/README.md](../tests/evals/README.md), not §7.
+> - 🟦 in this document is the pre-0.14 `outofscope` glyph. `outofscope` now
+> renders ⬜, and 🟦 marks a finding outside the ticket's scope. The glyphs
+> live in one table, `prxref.markers`.
+> - To use the feature, read the README's
+> [Review Against a Spec or Ticket](../README.md#review-against-a-spec-or-ticket),
+> [docs/quality.md](quality.md#spec-grounding) and
+> [docs/deploy.md](deploy.md#7-spec-sources-in-ci-and-on-the-daemon).
+
+Status: design record (v1), implemented in 0.14.0. Owner: prxref.
+
+## 0. What this feature is
+
+Today every prxref seat reviews the diff against generic bug classes only
+(`prompts/worker.md:1-23`). Spec-grounded review adds a second axis: the
+operator supplies scope/intent — a ticket plus a docs/spec corpus — and the
+reviewer must additionally catch **violations of that spec**, emitted as a new
+`spec` severity (🔍), distinct from `error` 🟥 / `warning` 🟧 / `outofscope` 🟦.
+
+Motivating case: a repo adapting to the MCP spec version `2026-07-28` is
+reviewed against the spec's client/server best-practices docs, so "the diff
+sends the protocol-version header the spec forbids" surfaces as a spec finding
+rather than nothing.
+
+### Locked decisions (user-approved, not to be re-litigated)
+
+1. **Input**: repeatable `--spec ` flag + `PRXREF_SPEC_SOURCES`
+ (comma/space separated). Accepts public web URLs, local file/dir paths, and
+ Jira ticket URLs.
+2. **Docs handling**: fetch + prune. Extract headings/constraints, prune to
+ diff-relevant slices, inject the result. Never post full docs raw. No
+ persistent graph index in v1 (§10).
+3. **Jira auth**: REST + HTTP basic auth via `PRXREF_JIRA_BASE_URL` /
+ `PRXREF_JIRA_EMAIL` / `PRXREF_JIRA_API_TOKEN`, plus anonymous access for
+ public boards. (MCP noted as an alternative fetch path, not designed here.)
+4. **Severity**: new `spec` severity, symbol 🔍, own ordering and fallback
+ rules (§5).
+
+### Non-negotiable posture carried over
+
+- stdlib + `requests` only in core; the spec fetcher uses the same zero-extra
+ dependency stack as the forges (`forges/github.py:10`).
+- Non-blocking: any spec-fetch failure degrades to "reviewed un-grounded, with
+ a note in the summary". Exit 2 remains reserved for config errors
+ (`cli.py:8-15`).
+- Every posted comment keeps model attribution (`forges/base.py:72`).
+
+---
+
+## 1. Input plumbing
+
+### 1.1 CLI flag
+
+Add to the `review` subparser in `_build_parser` (`cli.py:61-86`), beside
+`--max-chunks`:
+
+```python
+rev.add_argument(
+ "--spec",
+ action="append",
+ default=None,
+ metavar="URL_OR_PATH",
+ help="spec/ticket source to review against; repeatable (PRXREF_SPEC_SOURCES otherwise)",
+)
+```
+
+`_cmd_review` passes `spec_sources=args.spec` into `_run_review`, which
+forwards it as a `load_config` override exactly the way `max_chunks` does
+today (`cli.py:175-180`), with
+`source_labels={"spec_sources": "--spec"}` so a malformed value is reported as
+the flag the operator typed. This keeps `--spec` on the identical
+range-checked path as the env var and preserves the existing exit-2 contract
+(`cli.py:271-284`).
+
+Precedence: **`--spec` replaces `PRXREF_SPEC_SOURCES` when given; there is no merge.**
+This matches the documented precedence "built-in defaults < environment <
+overrides" (`config.py:104`).
+
+The webhook daemon inherits the feature for free: `_webhook_handler` calls the
+same `_run_review` (`cli.py:220-231`), so `PRXREF_SPEC_SOURCES` in the daemon's
+environment grounds every webhook-triggered review.
+
+### 1.2 Config keys
+
+Six new keys, all in `config._DEFAULTS` (`config.py:136-180`):
+
+| key | env var | type | default |
+|---|---|---|---|
+| `spec_sources` | `PRXREF_SPEC_SOURCES` | list (comma **and** whitespace separated) | `[]` |
+| `spec_max_chars` | `PRXREF_SPEC_MAX_CHARS` | int | `120000` |
+| `spec_digest_tokens` | `PRXREF_SPEC_DIGEST_TOKENS` | int | `3000` |
+| `jira_base_url` | `PRXREF_JIRA_BASE_URL` | str | `""` |
+| `jira_email` | `PRXREF_JIRA_EMAIL` | str | `""` |
+| `jira_api_token` | `PRXREF_JIRA_API_TOKEN` | str | `""` |
+
+Table placement:
+
+- `_LIST_KEYS` (`config.py:189`) += `spec_sources`. The existing list
+ coercion (`_coerce_env`, `config.py:295-296`) splits on commas only; widen it
+ to `re.split(r"[,\s]+")` so `PRXREF_SPEC_SOURCES` accepts space-separated values as
+ locked. The only other list key is `llm_models`, whose entries can never
+ contain spaces, so the widened split is a no-op for it.
+- `_INT_KEYS` (`config.py:182-186`) += `spec_max_chars`, `spec_digest_tokens`;
+ `_RANGES` (`config.py:252-264`) += `_Range(0)` for both (positive,
+ unbounded above, like every other size knob — `config.py:206-244`).
+- `jira_*` are plain string keys: no int/float/choice table entry, same as
+ every token key today.
+
+Semantics of the two ints: `spec_max_chars` bounds **raw fetched bytes per
+source** (post-decode); `spec_digest_tokens` bounds the **final digest text**
+injected into prompts, converted at the same 4-chars-per-token estimate the
+systemic digest uses (`systemic.py:70`).
+
+### 1.3 What is deliberately NOT a config error
+
+Missing/partial Jira credentials are **not** `ConfigError`. `load_config`
+validates values, not combinations; a ticket URL with no credentials takes the
+fetch-failure path (§2.5) and the review proceeds un-grounded with a note. The
+one thing the CLI will refuse is a malformed `--spec`/`PRXREF_SPEC_SOURCES` value —
+but since the value is an opaque string list, there is nothing to range-check;
+v1 adds no URL validation at config time. (Judgment call J2, §11.)
+
+---
+
+## 2. Fetch layer: new module `src/prxref/specs.py`
+
+One module, mirroring the shape of `systemic.py`: deterministic helpers, data
+in / text out, no LLM in the loop, docstrings on public API, no inline
+comments.
+
+### 2.1 Data shape
+
+```python
+@dataclass
+class SpecSource:
+ origin: str # the string the operator supplied, verbatim
+ kind: str # "file" | "dir" | "url" | "jira"
+ text: str # extracted plain text; "" when failed
+ error: str # "" on success, human-readable reason otherwise
+```
+
+```python
+@dataclass
+class TicketRef:
+ base_url: str # resolved REST base (env override or URL host)
+ key: str # e.g. "PROJ-123"
+ url: str # original ticket URL
+```
+
+### 2.2 Dispatch
+
+`parse_ticket_url(url: str) -> TicketRef | None` recognizes:
+
+- `{host}/browse/{KEY}-{n}` (Jira Cloud + Server classic),
+- `{host}/rest/api/{2|3}/issue/{KEY}-{n}` (raw REST links),
+- `{host}/jira/software/c/projects/{KEY}/issues/{KEY}-{n}` (Cloud new UI).
+
+Resolution: `PRXREF_JIRA_BASE_URL` wins when set (self-hosted boards often sit
+behind a different REST host than the browse URL); otherwise the ticket URL's
+own `scheme://host`.
+
+`fetch_specs(sources, *, max_chars, jira_base_url, jira_email, jira_api_token, session=None) -> list[SpecSource]`
+dispatches per source, in the order given:
+
+- `http://`/`https://` prefix + ticket match → Jira REST (§2.4).
+- `http(s)` otherwise → plain GET (§2.3).
+- existing filesystem path → file or directory (§2.3).
+- anything else → `SpecSource(error="not a URL or path")` — a per-source
+ failure, never an abort. The reason carries no path, because failure
+ reasons are posted (§6.4).
+
+> **As built (0.14.0):** `parse_ticket_url` also accepts a context path of up
+> to two segments before `/browse/` and `/rest/api/{2|3}/issue/` (Jira Server
+> under `/jira`), the Cloud team-managed issue view without `/c/`, and a
+> Cloud board URL on a `/jira/` path carrying `selectedIssue`. The
+> context-path bound keeps a Bitbucket Server `/projects/P/repos/R/browse/…`
+> URL from matching. A local path goes through
+> `text_inputs.confine_to_cwd` before anything stats or reads it (§2.3).
+
+### 2.3 Web + local fetching
+
+- **HTTP**: one `requests.Session` built like the forge sessions —
+ `LoggingRetry(total=3, backoff_factor=1, status_forcelist=[429,500,502,503,504],
+ allowed_methods=frozenset({"GET","HEAD","OPTIONS"}))` — reusing the read-only
+ retry policy verbatim from `forges/github.py:57-86` (same duplicate-POST
+ reasoning; the spec fetcher only ever GETs). Timeout
+ `SPEC_FETCH_TIMEOUT_S = 15` (module constant, not a config key in v1 —
+ §10). Content must arrive as `text/*`, `application/json`, or a text-like
+ markdown/html type; other content types fail that source.
+- **Size cap**: stream via `iter_content`, decode incrementally, stop at
+ `max_chars` and append an explicit `[source truncated at N chars]` marker —
+ truncation is announced, never silent (the same doctrine as
+ `systemic.TRUNCATION_MARKER`, `systemic.py:72-74`).
+- **HTML**: strip tags with `html.parser` (stdlib) to plain text before
+ extraction — spec pages are HTML more often than markdown.
+- **Local path**: read UTF-8 (errors → per-source failure), same
+ `max_chars` cap. A directory reads `*.md`, `*.markdown`, `*.txt`, `*.adoc`
+ in sorted filename order, capped at the first 20 files
+ (`SPEC_DIR_MAX_FILES = 20`), each through the same cap.
+
+> **As built (0.14.0):** the spec session does not reuse the forge retry
+> policy. `specs._create_default_session` retries **once**
+> (`LoggingRetry(total=1, …)`), with no backoff sleep before that retry, and
+> ignores `Retry-After` (`respect_retry_after_header=False`). The daemon
+> reviews one PR at a time, so a spec host that is down or asks for time is
+> skipped, not waited for. Two module constants bound a source, and neither
+> depends on `--timeout`:
+>
+> - `SPEC_FETCH_TIMEOUT_S = 15`: each attempt's connect timeout and its
+> timeout per read;
+> - `SPEC_FETCH_BUDGET_S = 30`: the wall clock for the body, counted from
+> before the request.
+>
+> `specs._read_stream` reads the body one socket read at a time
+> (`raw.read1(8192, decode_content=True)`, or `iter_content(chunk_size=1)`
+> on urllib3 1.x) and checks a monotonic deadline before each read, so a
+> trickling host costs at most the budget plus one read timeout, about 45 s.
+> A host that accepts the connection and never answers costs about 30 s:
+> two attempts of 15 s. The byte
+> cap is `4 * max_chars + 4`. The charset comes from the header, then an HTML
+> `` in the first 1024 bytes, then strict UTF-8 without a BOM, then
+> cp1252 with replacement. HTML is stripped after the cut, and the
+> truncation marker is appended after the stripping.
+>
+> Local files are read in bounded memory as strict UTF-8 without a BOM
+> (`text_inputs.read_capped_file`). The marker is appended only when a file
+> is longer than `max_chars`. A path under the working directory must still
+> resolve under it once its symlinks are followed, while an absolute path
+> outside it is read as given. A directory is read with `os.scandir`: it
+> skips every symlinked entry, and it skips a file it cannot read or decode
+> without failing the others. Skipped names are logged at WARNING only, and
+> they become the source's error only when no file was read. No reason
+> carries a local path: an `OSError` is reported by its class and
+> `strerror`.
+
+### 2.4 Jira REST (primary path)
+
+```
+GET {base}/rest/api/2/issue/{key}?fields=summary,description,issuetype,labels
+Authorization: Basic base64(email:api_token)
+```
+
+- Credentials present (both `jira_email` and `jira_api_token` non-empty) →
+ basic auth; absent → anonymous request (public boards work with no config).
+- Render the ticket as plain text: `Summary: …`, `Type: …`, `Labels: …`,
+ then the description body verbatim (Atlassian wiki-markup or ADF-plain —
+ v1 passes the text through; it is prose, and the extractor §3.2 reads prose
+ fine).
+- HTTP 401/403 without credentials → error text that names the three
+ `PRXREF_JIRA_*` variables, because that is the fix an operator can act on.
+ The reason may end up posted (§6.4), so it is written to survive
+ `redact_for_post` (`orchestrator.py:202-235`): variable NAMES only, never
+ values, host kept (a host is not a credential and `_URL_RE` will strip the
+ original ticket URL if it reappears).
+- MCP as an alternative fetch path is noted here as a future option only
+ (§10); REST basic-auth is the designed primary per the locked decisions.
+
+> **As built (0.14.0):** credentials only ever go to `PRXREF_JIRA_BASE_URL`.
+> Basic auth is sent only when that variable, `PRXREF_JIRA_EMAIL` and
+> `PRXREF_JIRA_API_TOKEN` are all set; every other ticket fetch is
+> anonymous. Credentials set without a base URL are withheld, with a WARNING
+> naming `PRXREF_JIRA_BASE_URL`. A plain-`http://` base URL is used, with a
+> WARNING. The variable-naming hint also fires on an anonymous 404, because
+> Jira Cloud hides a private issue from anonymous readers as 404. The
+> response streams through the same budgeted reader as a web page (§2.3): a
+> body over the byte cap, or a 200 that is not a JSON issue, fails the
+> source cleanly. A `Type:` or `Labels:` line whose value is empty is left
+> out, because every non-blank ticket line becomes a digest constraint.
+
+### 2.5 Failure doctrine
+
+`fetch_specs` never raises. Every exception inside a source becomes that
+source's `error` string. A run where **all** sources failed is not an error
+run: the pipeline behaves exactly like a run with no specs, plus a summary
+note listing the failures (§6.4). This is the same shape as `list_threads`
+best-effort failure (`orchestrator.py:447-451`).
+
+---
+
+## 3. Prune/digest: constraint extraction and diff-relevance pruning
+
+### 3.1 Goal
+
+Each fetched source becomes a compact **spec constraint digest**: a bounded,
+deterministic text the worker prompts can hold, containing the constraints
+that could plausibly be violated by *this* diff. Deterministic and model-free,
+like `systemic.build_digest` (`systemic.py:326-414`): same input → same text,
+so evals and traces stay stable.
+
+### 3.2 Extraction (regex pass, no model)
+
+Per source, in document order, collect:
+
+- **Headings**: `^#{1,6} ` (markdown), `^\n[A-Z][^\n]{0,80}\n[-=]{3,}$`
+ (setext/asciidoc) — kept as `[heading]` scoping lines so a constraint stays
+ attached to its section (e.g. "Client BEST PRACTICES" vs "Server").
+- **Normative statements**: sentences carrying RFC-2119-strength keywords,
+ matched inside blocks rather than physical lines (see *As built* below) —
+ `MUST`, `MUST NOT`, `SHALL`, `SHALL NOT`, `REQUIRED`, `SHALL NOT`,
+ `FORBIDDEN`, `MUST NEVER` (strength 3); `SHOULD`, `SHOULD NOT`,
+ `RECOMMENDED`, `recommended to`, `forbidden to` (strength 2); `MAY`,
+ `can`, `discouraged` (strength 1, lowest keep-priority). Case-sensitive for
+ the RFC-2119 all-caps forms, case-insensitive for the prose forms.
+- **Version pins**: `\b\d{4}-\d{2}-\d{2}\b` (spec revisions like
+ `2026-07-28`), `\b(?:v?\d+\.\d+(?:\.\d+)?)\b` adjacent to the words
+ `version|protocol|revision|draft`. A pinned version inside a kept constraint
+ is quoted verbatim; a version pin on its own line is kept as its own
+ constraint.
+- **Naming/shape rules**: sentences matching
+ `(?:MUST|SHOULD|SHALL)[^.]{0,120}(?:named|name|prefix|suffix|header|field|snake_case|camelCase|lowercase|uppercase)`
+ — the "tools MUST be named `mcp__`"-class constraints.
+- **Ticket text** (`kind == "jira"`): the summary always, and the description
+ kept in full up to a per-source sub-budget (`min(spec_max_chars // 4, 6000)`
+ chars) — the ticket is the *scope/intent*, small and high-value; it is not
+ pruned to keywords.
+
+Each kept unit renders as one line:
+
+```
+[spec:{origin-short}#{anchor-or-line-N}] (MUST)
+```
+
+`origin-short` is the source's basename or URL path tail; the Jira ticket's
+tag is `[ticket:{KEY}]`.
+
+> **As built (0.14.0):** matching runs on blocks and sentences, not physical
+> lines (`specs._spec_units`).
+>
+> - A paragraph or list item joins its wrapped and indented continuation
+> lines. A blank line, heading, setext underline, code fence, table row or
+> new list item ends a block, and a table row or fenced line is a unit of
+> its own.
+> - Each block is split into sentences (`e.g.`, `i.e.`, common abbreviations
+> and code spans never end one), and each sentence is matched on its own. A
+> block over 400 characters, or one with two or more matching sentences,
+> yields one unit per matching sentence. Otherwise the block is kept whole,
+> unless its only match is a prose `can`/`discouraged` or a bare version
+> pin, which keeps just that sentence.
+> - Known limitation: only `.`, `;`, `!` or `?` ends a sentence, so keyword
+> lines with no such punctuation, one after another in a paragraph with no
+> blank line between them, form one sentence and so one unit, labelled with
+> the highest strength any of them carries (`specs._split_sentences`,
+> `specs._strength`): `Clients MAY cache tokens` directly above
+> `Servers MUST reject expired tokens` is a single `(MUST)` unit.
+> - A unit ending in `:` carries the list that follows it, up to the cap. A
+> version pin counts only on a line that is nothing but the pin. Every unit
+> is anchored `L{n}` on its block's first line.
+> - Headings render as `[spec:{short}#{slug}] (heading) text`.
+> - A ticket unit renders `[ticket:{KEY}] statement` with no strength label.
+> Every non-blank ticket line is kept, up to a fixed 6000 characters
+> (`TICKET_DESC_BUDGET_CHARS`); the `min(spec_max_chars // 4, 6000)`
+> sub-budget is deferred.
+> - `origin-short` never carries a URL's query, fragment, userinfo or port,
+> but a credential that *is* the last path segment survives.
+
+### 3.3 Diff-relevance pruning
+
+Rank kept constraints, then keep until budget:
+
+1. **Ticket constraints** (from `jira` sources): always kept first — scope
+ beats relevance.
+2. **Relevant spec constraints**: score = overlap between the constraint's
+ content tokens and the diff's token set (file path segments + changed-line
+ text, both compound-split). Tokenization mirrors the existing evidence
+ vocabulary — 4-char floor, stopword-filtered, snake/camel parts split —
+ which `quality.py:443-475` (`_tokens`, `_evidence_tokens`) already
+ implements; v1 imports that logic (promote to a small shared helper or
+ duplicate narrowly; implementer's choice, pinned by a parity test).
+ Constraints with score ≥ 1 are "relevant".
+3. **Unmatched MUSTs** (strength 3, score 0): kept after relevant ones,
+ ordered by source order — a MUST the diff doesn't obviously touch is still
+ the cheapest place a sweep can find a violation of the "absence is
+ evidence" kind, exactly the migration-DDL argument in
+ `systemic.py:17-22`.
+4. SHOULDs (2) then MAYs (1) with score 0 are dropped first, then trailing
+ unmatched SHOULDs, as budget runs out.
+
+> **As built (0.14.0):** the score is the constraint's content tokens shared
+> with the diff, minus the normative keywords themselves
+> (`specs._NORMATIVE_TOKENS`: `must`, `shall`, `should`, `never`,
+> `required`, …), so a diff line that merely says "must" matches nothing.
+> Relevant constraints sort by score, then source order, then document
+> order. The unmatched tail runs MUST, then SHOULD, then MAY, and the budget
+> walk cuts it from the end. The tokenizer is `quality._tokens` /
+> `_evidence_tokens`, imported, not duplicated.
+
+### 3.4 Budget
+
+`build_spec_digest(sources, files, token_budget) -> str` enforces
+`token_budget × 4` chars (`PRXREF_SPEC_DIGEST_TOKENS`, default 3000 → ~12k
+chars), walking the ranked list and stopping with a final
+`[spec digest truncated: budget reached]` line. Every source that contributed
+at least one line gets an origin tag so the model can cite *which* spec a
+constraint came from; a source that contributed nothing after pruning gets a
+one-line `[spec:{origin}: nothing diff-relevant kept]` so silence is
+explained, not inferred away.
+
+The digest is built once per review, after `parse_unified_diff` (files are the
+pruning input) and before the worker fan-out.
+
+> **As built (0.14.0):**
+>
+> - `build_spec_digest` returns `""` when sources were given but no unit was
+> extracted from any of them, so the prompts show their no-specs text.
+> - A failed source gets no line in the digest; the grounding note (§6.4)
+> reports it. The not-contributed line uses the short origin:
+> `[spec:{short}: nothing diff-relevant kept]`.
+> - Ranking interleaves sections, so a constraint's heading line is
+> re-emitted whenever the open section changes, and a unit with no heading
+> after one that had one is preceded by `[spec:{short}] (heading) (no
+> section)`.
+> - `specs.constraint_count` counts only the constraint lines. The intro,
+> heading lines, truncation markers and bookkeeping lines never count, and
+> a digest with no constraint line is not injected at all (§4.1).
+
+---
+
+## 4. Prompt integration
+
+### 4.1 Where the digest enters
+
+Zero extra LLM calls in v1: the digest rides the **existing** per-chunk calls
+and the **existing** systemic sweep. (A dedicated spec-sweep call is §10 /
+judgment call J6.)
+
+Plumbing:
+
+- `reviewer.review_chunk` (`reviewer.py:285-337`) and
+ `reviewer.review_systemic` (`reviewer.py:340-371`) gain a
+ `spec_digest: str = ""` keyword.
+- `_render_prompt` (`reviewer.py:117-139`) and `_render_systemic_prompt`
+ (`reviewer.py:142-163`) add `.replace("{spec_digest}", …)` next to the
+ existing placeholder fills; empty digest renders the literal
+ `(no specs provided for this review)`.
+- The prompt templates gain a `### Spec constraints` block inside the
+ `## Review Context` section (between `{pr_description}` and the diff/digest
+ block), containing `{spec_digest}`.
+- `orchestrate_review` (`orchestrator.py:262-280`) gains
+ `spec_sources: Sequence[str] = ()`; it calls `specs.fetch_specs` + builds
+ the digest inside the same never-raise fence as every other stage
+ (`orchestrator.py:67-71`), then threads the digest into `_run_workers` →
+ `_run_worker` → `_invoke_chunk` → `review_chunk` and into `_run_sweep` →
+ `review_systemic`. Trace: one `tracer.event("specs", …)` recording
+ sources/ok/fail counts and final digest chars, between `build_chunks` and
+ the worker span.
+
+> **As built (0.14.0):**
+>
+> - The digest travels in `reviewer.PromptContext.spec_digest`, alongside
+> the other context blocks, rather than as a separate keyword on each call.
+> - Only a grounded digest is injected: `grounded =
+> specs.constraint_count(digest) > 0`. A digest with no constraint line (no
+> sources, every source failed, nothing extracted, or a budget too small
+> for one line) is not injected, so every prompt shows `(no specs provided
+> for this review)`.
+> - The trace records `specs ok|fail` with `{sources, ok, constraints}`. A
+> `fail` event (no source fetched) also carries the raw, unredacted
+> `reasons`, and a crashed spec stage emits `specs fail` with one
+> `spec stage crashed: …` reason. Digest chars are not recorded. The event
+> is emitted after the thread listing and before the worker fan-out. A
+> later `specs relabel {findings}` event marks ungrounded `spec` findings
+> relabelled `warning` (§5.2).
+> - The operator-facing WARNING and INFO lines and the run record's
+> `spec_grounding` key are documented in
+> [docs/deploy.md](deploy.md#what-the-logs-the-run-record-and-the-trace-say).
+
+### 4.2 What the prompts say
+
+`prompts/worker.md` — extend `## Severity Vocabulary` (`worker.md:7-11`) with:
+
+> - `spec` — the diff violates a constraint quoted in the Spec constraints
+> block below: a MUST/SHALL/required behaviour not implemented, a
+> forbidden behaviour implemented, a version pin or naming rule broken.
+> Only when specs were provided. Quote the violated constraint verbatim in
+> the body, prefixed `Spec: "`.
+
+and a new `## Spec-grounded rules` section: emit `spec` **only** for a
+conflict between the diff and a quoted constraint — never for a generic best
+practice not present in the block; when the block reads `(no specs provided
+for this review)`, `spec` is not a legal severity. Cite the diff line that
+violates it (same `file`/`line` contract as every finding, `worker.md:62`).
+
+`prompts/systemic.md` — same vocabulary bullet, plus one mission line: with
+the whole-diff digest plus any spec constraints in view, the sweep is the
+natural seat for cross-file spec classes (naming rules, version pins, "no
+component may do X" rules), while per-chunk seats catch line-local
+violations. Nothing else in the sweep's class list changes
+(`systemic.md:5-15`).
+
+`prompts/summary.md` — see §6.1 (counts line only).
+
+Severity wording matters for eval scoring: the constraint quote convention
+(`Spec: "…"`) gives expected.json a machine-checkable field and gives the
+severity-consistency pass distinctive titles (§5.3).
+
+> **As built (0.14.0):**
+>
+> - Each template splits at `## Review Context`. The severity bullet and the
+> `## Spec-grounded rules` section sit in the system half; the
+> `### Spec constraints` block with `{spec_digest}` sits in the user half,
+> before the diff or digest. So every prompt changed in 0.14.0, and the
+> worker and sweep prompts carry spec text on every run, with sources or
+> without.
+> - Both rule sections add one override sentence: when the only basis for a
+> finding is a constraint quoted in the block, its severity is `spec`. The
+> sweep's rules add that its own built-in classes (RLS, secrets, …) are
+> never spec constraints.
+> - The expected.json field this paragraph anticipates was not built. The
+> shipped dataset matches findings by `must_match` (see §7).
+> - A `Spec: "…"` quote also exempts its verbatim digest text from the hedge
+> gate (§5.2).
+
+---
+
+## 5. Quality-gate integration: the `spec` severity
+
+### 5.1 Vocabulary and ordering
+
+`quality.SEVERITIES` (`quality.py:52`) becomes
+`{"error", "warning", "spec", "outofscope"}`.
+
+**Ordering** — `spec` sits below `warning`, above `outofscope`:
+
+```
+error(0) > warning(1) > spec(2) > outofscope(3)
+```
+
+Rationale: a spec violation is an operator-requested contract breach — always
+worth reporting — but it is not claimed to break at runtime, so it does not
+outrank a generic warning. Updated everywhere the rank table is restated:
+
+- `quality._SEVERITY_RANK` (`quality.py:556`) — drives severity-consistency
+ max-raise; `spec: 2`, `outofscope: 3`.
+- `orchestrator._SEVERITY_RANK` (`orchestrator.py:129`) — inline-comment
+ priority; `.get(f.severity, 3)` at `orchestrator.py:522` keeps unknown
+ severities last without further edit.
+- `formatter._SEVERITY_ORDER` (`formatter.py:25`) — summary table order.
+
+**Unknown-severity fallback**: unchanged in kind. The gate drops any severity
+outside `SEVERITIES` with `drop_reason="invalid severity: …"`
+(`quality.py:890-895`), so unknown strings never post; the rendering-layer
+fallbacks (`formatter._norm_severity` → `"outofscope"`, `formatter.py:44-49`;
+`orchestrator._SEVERITY_MARKERS.get(…, "🟦")`, `orchestrator.py:971,:1070`)
+continue to map unknown → outofscope 🟦 for anything that reaches them. A
+model that misspells `spec` therefore loses the finding loudly (drop, audit
+trail) rather than silently mis-rendering it.
+
+> **As built (0.14.0):** `orchestrator._SEVERITY_MARKERS` no longer exists.
+> Every glyph comes from one table, `prxref.markers.SEVERITY_MARKERS`
+> (`error` 🟥, `warning` 🟧, `spec` 🔍, `outofscope` ⬜), and the orchestrator
+> renders through `markers.severity_marker`. An unknown severity renders
+> `markers.FALLBACK_MARKER`, which is ⬜, the `outofscope` glyph; 🟦 is now
+> `markers.OUT_OF_TICKET_MARKER`, a scope prefix, never a severity.
+> `formatter._norm_severity` still maps an unknown severity to `outofscope`.
+> The three rank tables above shipped as designed.
+
+### 5.2 Gate mechanics per pass
+
+- **Location validation** (`quality.py:70-93`): severity-agnostic — a spec
+ finding must name a diff path like any other.
+- **Line align** (`quality.py:349-411`): severity-agnostic — body-citation
+ precedence, `snap_line` tolerance 5 (`quality.py:62`), blank-anchor guard
+ all apply unchanged. The `Spec: "…"` quote in the body must not be mistaken
+ for a location; it is prose, and the citation regexes
+ (`quality.py:258-265`) only match `path:line` / `line N` shapes.
+- **Thread dedup** (`quality.py:527-553`): unchanged; a spec finding
+ duplicating an existing human thread is suppressed like any other.
+- **Severity consistency** (`quality.py:718-838`): `spec` joins the rank map,
+ so a title-group containing both `spec` and `warning` members raises to
+ `warning`, `spec`+`error` raises to `error`, etc. — current max-raise
+ semantics preserved (judgment call J5).
+- **Quality gate** (`quality.py:866-928`): confidence floor applies to spec
+ findings unchanged. The **error cap ignores `spec`** — it counts
+ `severity == "error"` only (`quality.py:913`) — so a spec-heavy review is
+ not crowded out by, nor crowding out, `PRXREF_MAX_ERROR_FINDINGS`. If spec
+ findings ever need their own cap, that is a future `PRXREF_MAX_SPEC_FINDINGS`
+ (not in v1).
+- **Sweep dedup** (`quality.py:577-615`): unchanged; a sweep-emitted spec
+ finding restating a surviving chunk finding still dedups on
+ `(file, normalize_title)`.
+
+> **As built (0.14.0):** two behaviours this list did not plan.
+>
+> - **A new pass, `quality.apply_spec_grounding`,** runs after the team
+> severity map and before location validation, over chunk and sweep
+> findings alike. On an ungrounded run it relabels every `spec` finding as
+> `warning` (compared trimmed and lower-cased); it never drops a finding and
+> never raises one to `spec`, and on a grounded run it changes nothing.
+> Because it runs ahead of severity consistency, an ungrounded `spec` can
+> never lift a same-title sibling. A relabel logs one INFO line and one
+> `specs relabel {findings}` trace event.
+> - **The hedge gate exempts a verbatim spec quote.** After a `Spec:` marker,
+> the text a finding copies verbatim from the injected digest, compared
+> case-insensitively, up to a closing quote, is not read for hedges, in a
+> finding of any severity. With no digest injected, nothing is exempt.
+>
+> The user-facing description, including the hedge exemption's known
+> limitation, is [docs/quality.md](quality.md#spec-grounding).
+
+### 5.3 Verdict and exit codes
+
+- **Verdict unchanged**: `Request-Changes` iff an active `error` survives
+ (`orchestrator.py:480-484`). `spec` findings do **not** move the verdict in
+ v1 (judgment call J3). `PRXREF_FAIL_ON=any` already gates on spec findings
+ for lanes that want a hard signal (`cli.py:250-260`); `FAIL_ON=error`
+ ignores them, exactly as designed.
+
+---
+
+## 6. Output
+
+### 6.1 Summary counts line
+
+`prompts/summary.md:5` and the fallback template
+(`orchestrator.py:237-243`) become:
+
+```
+🟥 {error_count} error · 🟧 {warning_count} warning · 🔍 {spec_count} spec · 🟦 {outofscope_count} outofscope
+```
+
+`_render_summary` (`orchestrator.py:965-993`): initialize
+`counts = {"error": 0, "warning": 0, "spec": 0, "outofscope": 0}` and add the
+`.replace("{spec_count}", …)` to the chain. The findings-bullet marker lookup
+(`orchestrator.py:971`) and inline renderer (`orchestrator.py:1069-1076`)
+need `"spec": "🔍"` in `_SEVERITY_MARKERS` (`orchestrator.py:124`); the
+inline header renders `[SPEC]`.
+
+`formatter.py` mirrors all of this (`:20-25`, counts at `:172-180`) so the
+forge-neutral renderer and the orchestrator renderer cannot drift; the
+existing orchestrator-template pin tests
+(`tests/test_orchestrator.py:30`, `tests/test_formatter.py:103,:115`) are
+updated to the new counts line in the same change.
+
+> **As built (0.14.0):** the shipped counts line ends `⬜ {outofscope_count}
+> outofscope`, not 🟦 (see §5.1), and `🔍 {spec_count} spec` appears on every
+> run, with spec sources or without. The line after it in
+> `prompts/summary.md` is `{spec_note}{ticket_note}`. A parity test holds
+> the summary templates' glyph literals to `prxref.markers`.
+
+### 6.2 Inline rendering
+
+Spec findings post as ordinary inline comments (`orchestrator.py:519-538`)
+with the 🔍 marker and `[SPEC]` label, anchored like any finding. They count
+against `PRXREF_MAX_INLINE_COMMENTS` by confidence+severity rank; at rank 2
+they yield the anchor to errors/warnings first, which is the intended
+posture.
+
+### 6.3 Attribution
+
+Unchanged: `_format_finding`'s trailing
+`Reviewed by prxref · model=…` (`orchestrator.py:1075`) and the summary's
+`{attribution}` line carry model attribution as required by convention.
+
+### 6.4 Grounding note in the summary
+
+When `spec_sources` was non-empty, the summary gains one blockquote line after
+the counts (via a new `{spec_note}` placeholder that renders `""` when no
+specs were requested, keeping today's output byte-identical otherwise):
+
+```
+> 🔍 Spec-grounded: 3 source(s) · 41 constraint(s) injected
+> ⚠️ Spec fetch failed for 1 source(s): ticket PROJ-9 (HTTP 401 — set PRXREF_JIRA_EMAIL/PRXREF_JIRA_API_TOKEN)
+```
+
+Failure reasons pass through `redact_for_post` before the post
+(`orchestrator.py:202-235` doctrine: everything interpolated into a posted
+comment is redacted; URLs are stripped by `_URL_RE`). A total fetch failure
+renders only the failure line, and the review reads as un-grounded — which it
+was.
+
+> **As built (0.14.0):** a failed source is labelled by its 1-based position
+> in the configured list and its kind, `source 2 (url)`, or `source 2` when
+> the kind was never determined; it is never named by its origin. The
+> example above is therefore stale: the shipped failure line reads
+> `> ⚠️ Spec fetch failed for 1 source(s): source 1 (jira): …`. The
+> `Spec-grounded` line counts every configured source and the constraint
+> lines actually injected, so it can read `0 constraint(s) injected` when
+> sources fetched but none held a constraint. Because the note reaches only
+> a posted summary, the same facts also go to the log and the run record on
+> every run with spec sources, `--no-post` included (see
+> [docs/deploy.md](deploy.md#what-the-logs-the-run-record-and-the-trace-say)).
+
+---
+
+## 7. Golden eval dataset: `tests/evals/`
+
+> **As built (0.14.0):** the dataset shipped, and so did an offline replay
+> run of every case (§7.1); the designed runner and the scoring did not.
+> **[`tests/evals/README.md`](../tests/evals/README.md) is the dataset
+> contract**, and it replaces the layout and `expected.json` schema this
+> section first proposed. In short:
+>
+> - Three cases ship, each a `tests/evals/case-NNN-/` directory holding
+> `ticket.md`, `docs/`, `diff.patch`, `expected.json` and `meta.json`.
+> - `expected.json` is a flat JSON array of must-find entries, each with
+> exactly `id`, `file`, `line_hint`, `severity`, `must_match` (a substring,
+> or a regex when prefixed `re:`) and `source` (`spec` for a planted
+> violation, `generic` for a plain bug). There is no `title_hint`, no
+> `constraint_ref` and no `nonfindings` list.
+> - `meta.json` carries a planted-violation manifest that maps one-to-one
+> onto the `source: "spec"` entries. It carries no score floor.
+> - `tests/evals/test_evals.py` is a structural scorer only. It proves every
+> case is well-formed and self-consistent and runs no LLM:
+> `uv run pytest tests/evals -q`.
+>
+> `source` keeps the meaning proposed here: `generic` marks an ordinary bug
+> the unguided reviewer should also catch, so a later scoring pass can check
+> that grounding costs no generic recall.
+
+### 7.1 Runner (design not built; a replay run shipped instead)
+
+> **Planned, not built.** Nothing in §7.1 or §7.2 exists in 0.14.0: there is
+> no `harness.py`, no plumbing or recorded stub-LLM mode, and no P/R/F1
+> scoring. What did ship, outside this design, is one pipeline run per case:
+> [`tests/evals/test_eval_replay.py`](../tests/evals/test_eval_replay.py)
+> reviews each case with one replay-mode `prxref review` call against a stub
+> LLM that finds nothing, which proves the wiring, not the review. Scoring the
+> findings against `expected.json` is still a manual, offline step
+> ([`tests/evals/README.md`](../tests/evals/README.md), "Running a case"). The
+> design below is kept for a later pass that automates that scoring.
+
+`tests/evals/harness.py` (data-local, not shipped) + a thin wrapper in
+`tests/evals/test_evals.py` (the file that holds today's structural scorer)
+so `uv run pytest tests/evals -q` runs it in CI:
+
+1. Load the case; build sources as `["/ticket.md", "/docs"]`
+ (file sources — the fetch layer is exercised by `test_specs.py` with
+ mocked sessions, not here).
+2. Drive the real pipeline: `parse_unified_diff(diff.patch)` →
+ `specs.build_spec_digest` → `reviewer.render prompt` paths → **stub LLM**
+ → the full quality pass chain (`apply_location_validation` →
+ `apply_line_align` → … → `apply_sweep_dedup`, `orchestrator.py:453-475`)
+ → active findings.
+3. **Stub LLM modes**:
+ - *plumbing mode* — per-case `stub_response.json` hand-written findings
+ exercising gate/anchor/dedup edges (drifted lines, sub-floor confidence,
+ invalid severities, sweep-vs-chunk duplicates);
+ - *recorded mode* — a real model's captured response for the case,
+ replayed verbatim (regression mode for model drift).
+ The stub satisfies the `LLMClient.invoke` shape (`reviewer.py:219-224`),
+ is injected the way `test_reviewer.py` doubles are today.
+4. Score **post-gate, post-alignment** output against `expected.json` —
+ never raw model output, so the eval measures what a PR author receives.
+
+### 7.2 Scoring metric (planned, not built)
+
+> **Against the shipped dataset:** `expected.json` has no `line`,
+> `title_hint`, `constraint_ref` or `nonfindings`. A scorer would match on
+> `file`, on `line_hint` within the line tolerance, and on `must_match`
+> against the finding body; the specificity check has no list to read; and
+> a per-case floor would live in the scorer, since `meta.json` holds none.
+> The line self-check below did ship, tighter than designed:
+> `test_line_hints_anchor_added_lines` requires every `line_hint` to be an
+> added line of the case diff, and `test_expected_json_schema` requires it
+> to be at least 1, so there is no file-level `0`.
+
+Matching rule, in order: same `file` (exact) **and**
+`|pred.line − exp.line| ≤ quality.DEFAULT_LINE_TOLERANCE` (5,
+`quality.py:62`) **and** token overlap between title+body and
+`title_hint + constraint_ref` using `quality.normalize_title` /
+`_tokens` (`quality.py:443-475, :561`). Expected `line: 0` (file-level)
+matches any line in the same file.
+
+- **Spec recall** = matched expected `source:"spec"` entries ÷ total spec
+ entries.
+- **Spec precision** = predicted `severity=="spec"` findings that match a spec
+ entry ÷ all predicted spec findings.
+- **Class-miss** (counted, not folded in): a prediction that matches an
+ expected spec entry but carries a non-spec severity. Binary good/bad hides
+ this third outcome — a "recall hit" that arrives as a generic warning is a
+ grounding failure the F1 must not launder.
+- **Specificity**: any predicted spec finding matching a `nonfindings` entry
+ is a counted false-positive of the worst kind.
+- **Anchor check**: predicted lines are snapped through the production
+ `apply_line_align` before scoring, so eval anchor tolerance can never
+ diverge from shipped behavior. Self-check: each expected line must be an
+ added line of the case diff (or 0) — validated when the case is loaded, so
+ a stale `expected.json` fails loudly, not as a mysterious 0-recall.
+- Aggregate: per-case P/R/F1 table plus means; the pytest wrapper asserts a
+ per-case floor (F1 ≥ 0.8, class-miss ≤ 1, zero specificity hits) from
+ `meta.json`.
+
+---
+
+## 8. Config surfaces checklist + docs updates
+
+The four-surface rule is enforced, not aspirational:
+`tests/test_docs_consistency.py:58-118` fails the build when a `_DEFAULTS` key
+misses any surface, and when `docs/env-vars.md`'s **stated counts** go stale.
+For six new keys:
+
+- [x] `src/prxref/config.py`: `_DEFAULTS` + six keys; `_LIST_KEYS` +=
+ `spec_sources`; `_INT_KEYS` += `spec_max_chars`, `spec_digest_tokens`;
+ `_RANGES` += both; module docstring env table (lines 5-83) += six
+ entries.
+- [x] `.env.example`: six commented entries with defaults (file pattern
+ `.env.example:11-60`).
+- [x] `docs/env-vars.md`: six table rows (LLM & Pipeline section for the
+ three spec keys; a new "Spec Sources / Jira" subsection for the three
+ `PRXREF_JIRA_*` keys); update the stated totals —
+ `**35** configuration keys` → `**41**`, and
+ `for 36 accepted variable names` → `for 42` (`docs/env-vars.md:122-128`;
+ the test asserts these strings, `test_docs_consistency.py:103-118`).
+- [x] `README.md`: short "Review against a spec or ticket" section with one
+ copy-paste example.
+- [x] `src/prxref/cli.py`: `--spec` flag + plumbing (§1.1).
+- [x] Prompt templates: `worker.md`, `systemic.md` (§4.2), `summary.md` (§6.1).
+- [x] No new `_CHOICE_KEYS` entry is needed (no enum-valued key in this
+ feature).
+
+> **As built (0.14.0):** every item shipped. The 35→41 / 36→42 totals above
+> are history: other keys landed between this design and the spec keys, and
+> when the spec keys landed the table held 55 configuration keys and 56
+> accepted names (the 55 plus the one deprecated alias). The test does not
+> hard-code either number. It computes them from `len(config._DEFAULTS)` and
+> `config._LEGACY_ENV_ALIASES` and checks the totals `docs/env-vars.md`
+> states.
+
+---
+
+## 9. Testing plan and rollout
+
+### 9.1 Unit tests
+
+- `tests/test_specs.py` (new): dispatch (file / dir / URL / ticket-URL
+ shapes / garbage); Jira auth present vs anonymous vs 401-message content;
+ size caps + truncation markers; HTML stripping; dir file cap + sort order;
+ `fetch_specs` never raises; digest determinism (same inputs twice →
+ identical bytes); pruning rank order (ticket > relevant > unmatched-MUST >
+ SHOULD); budget truncation marker; relevance scoring parity with
+ `quality._tokens`.
+- `tests/test_quality.py`: `spec` passes the gate; rank order
+ error > warning > spec > outofscope in `apply_severity_consistency`; unknown
+ severity still dropped; error cap ignores spec findings.
+- `tests/test_orchestrator.py`: `{spec_count}` + `{spec_note}` in the summary;
+ 🔍 marker in bullets and inline cards; verdict NOT moved by spec-only
+ findings; all-sources-failed run completes with the failure note; digest
+ reaches the worker prompt (captured via stub LLM) and the sweep prompt;
+ `redact_for_post` applied to fetch-failure notes.
+- `tests/test_reviewer.py`: `{spec_digest}` replacement; `(no specs provided…)`
+ default; `spec` severity passes through unfiltered (reviewer never gates,
+ `reviewer.py:299-300`).
+- `tests/test_cli.py`: `--spec` repeatable; override replaces env;
+ `--spec` values survive into `orchestrate_review` kwargs.
+- `tests/test_docs_consistency.py`: must stay green untouched — it is the
+ checklist enforcer (§8).
+
+### 9.2 Rollout
+
+- Version: `0.11.1` → **`0.12.0`** (`pyproject.toml:3`). New user-facing
+ flag + new severity = feature minor; no breaking change (the severity
+ vocabulary grows, but unknown-severity handling was already
+ drop-with-audit, so older consumers of run records degrade safely).
+- `CHANGELOG.md`: one feature entry in the style of the severity-rename entry
+ (`CHANGELOG.md:173-184`): what `--spec`/`PRXREF_SPEC_SOURCES` accept, the 🔍
+ `spec` severity and its ordering, Jira env vars, fetch-failure
+ non-blocking behavior, and the eval harness. Explicit note that verdict and
+ exit codes are unchanged (advisory doctrine preserved), and that
+ `PRXREF_FAIL_ON=any` is the opt-in gate for spec findings.
+
+> **As built:** the feature shipped in **0.14.0**, not 0.12.0. The eval work
+> that shipped is the dataset plus its structural scorer (§7), not a
+> harness.
+
+---
+
+## 10. Open questions / explicit non-goals (v1)
+
+- **No persistent graph index** (locked): specs are fetched, pruned, and
+ discarded per run. A future `PRXREF_SPEC_INDEX` (cache extracted constraint
+ sets keyed by URL hash with TTL) is the natural v2 — the extraction pass is
+ already deterministic and pure.
+- **No auto-discovery of specs from the PR body** (non-goal): if a PR
+ description links a ticket, prxref does not fetch it. Operators are
+ explicit.
+- **No multi-ticket traversal** (non-goal): multiple `--spec` sources are
+ allowed, but each is fetched independently; no epic→story expansion, no
+ linked-issue walks.
+- **No dedicated spec-sweep LLM call** (judgment call J6): v1 injects the
+ digest into existing chunk + sweep prompts. If recorded-mode evals show the
+ sweep prompt too loaded to catch spec classes, v2 adds a third single-shot
+ unit (mirroring `_run_sweep`, `orchestrator.py:847-918`) — one more
+ `chunk_count` unit, same failure shape. *As built (0.14.0): recorded-mode
+ evals were not built (§7.1), so this trigger cannot fire yet.*
+- **No MCP ticket fetch** in v1 (locked: REST basic-auth is primary; MCP noted
+ as an alternative path for a future backend).
+- **No per-source timeout/Retry config knobs** in v1: module constants
+ (`SPEC_FETCH_TIMEOUT_S`, `SPEC_DIR_MAX_FILES`); promote to env vars only if
+ real usage demands it. *As built (0.14.0): a third constant,
+ `SPEC_FETCH_BUDGET_S = 30`, bounds one source's body in wall-clock
+ seconds (§2.3).*
+- **Jira comments not fetched** (v1 keeps summary+description): comment
+ threads are noisy and frequently carry the debate the review is supposed to
+ settle.
+
+---
+
+## 11. Judgment calls for the user to confirm
+
+> **As built (0.14.0):** all eight calls shipped as proposed. The notes on J4
+> and J6 record what changed around them.
+
+- **J1 — `--spec` replaces `PRXREF_SPEC_SOURCES`; no merge.** Matches load_config
+ override precedence (`config.py:104`). Alternative: flag values append to
+ env values.
+- **J2 — missing Jira credentials are a fetch failure, not exit 2.** The
+ review proceeds un-grounded with a note naming the three env vars.
+ Alternative: a Jira ticket URL with zero `PRXREF_JIRA_*` config could be a
+ config error (exit 2) on the "required value missing" theory.
+- **J3 — `spec` findings do not move the verdict.** Verdict stays
+ error-only (`orchestrator.py:480-484`); `PRXREF_FAIL_ON=any` is the hard
+ gate. Alternative: any active `spec` finding also yields
+ `Request-Changes`.
+- **J4 — ordering `error > warning > spec > outofscope`; unknown still falls
+ back to outofscope/🟦 at the render layer and is dropped by the gate.**
+ *As built (0.14.0): the fallback glyph is now ⬜, `markers.FALLBACK_MARKER`,
+ because `outofscope` itself renders ⬜ (§5.1).*
+- **J5 — severity-consistency can rewrite a `spec` finding to `warning`/
+ `error` on a title collision** (current max-raise semantics with `spec` at
+ rank 2). Alternative: make `spec` sticky (exempt from raises), at the cost
+ of the pass's group-coherence guarantee.
+- **J6 — no extra LLM call in v1**: the digest rides chunk prompts + the
+ systemic sweep. Alternative: a dedicated third sweep unit for spec classes.
+ *As built (0.14.0): held; the recorded-mode evals that would test it were
+ not built (§7.1, §10).*
+- **J7 — list coercion widened to comma-or-whitespace for ALL list keys**
+ (touches `llm_models`' coercion too; behavior-identical for it).
+ Alternative: a `spec_sources`-only split rule.
+- **J8 — digest budget default 3000 tokens (~12k chars)**, on top of a
+ 25k-token chunk budget (`triage.py:17`). If prompts grow too large on the
+ smallest configured budgets, the digest could be charged a fixed share of
+ `PRXREF_CHUNK_TOKEN_BUDGET` instead of being an independent knob.
diff --git a/docs/systemic-sweep.md b/docs/systemic-sweep.md
index 0f0197a..3cc915b 100644
--- a/docs/systemic-sweep.md
+++ b/docs/systemic-sweep.md
@@ -48,7 +48,10 @@ raises it anyway.
The fetch is best-effort: a forge that cannot list threads still gets a full
review, with an empty discussion block. It is deliberately ordered AFTER the
stale-inline-comment prune, or the run would suppress its own findings against
-comments it is about to delete.
+comments it is about to delete. A `--no-threads` replay (and a `--diff-file`
+replay with no `--pr-url`) lists no threads, so no `### Existing discussion`
+block is printed and the two thread gates, `apply_thread_dedup` and
+`apply_settled_thread_suppression`, see no threads either.
## Drop reasons
diff --git a/pyproject.toml b/pyproject.toml
index e08960d..304c150 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -1,7 +1,7 @@
[project]
name = "prxref"
-version = "0.13.0"
-description = "Fast automated AI code review for Bitbucket, GitLab, and GitHub"
+version = "0.14.0"
+description = "Fast automated AI code review for Bitbucket, GitLab, GitHub, and Azure DevOps"
readme = "README.md"
requires-python = ">=3.12"
license = "MIT"
@@ -69,6 +69,7 @@ exclude = [
"docs/superpowers",
"docs/configurability",
"docs/release-hardening",
+ "docs/issues",
]
[tool.ruff]
diff --git a/src/prxref/__init__.py b/src/prxref/__init__.py
index 9f04c8c..f0f5d91 100644
--- a/src/prxref/__init__.py
+++ b/src/prxref/__init__.py
@@ -1,3 +1,3 @@
-"""prxref — fast automated AI code review for Bitbucket, GitLab, and GitHub."""
+"""prxref — fast automated AI code review for Bitbucket, GitLab, GitHub, and Azure DevOps."""
-__version__ = "0.13.0"
+__version__ = "0.14.0"
diff --git a/src/prxref/cli.py b/src/prxref/cli.py
index bc974f7..9e67a14 100644
--- a/src/prxref/cli.py
+++ b/src/prxref/cli.py
@@ -1,10 +1,31 @@
"""prxref command-line interface.
Provides three subcommands:
- * ``review --pr-url URL`` — one-shot PR/MR review from a forge URL.
+ * ``review --pr-url URL`` — one-shot PR/MR review from a Bitbucket, GitHub,
+ GitLab, or Azure DevOps URL (Cloud or self-hosted).
* ``serve [--port N] [--host H]`` — webhook listener daemon.
* ``trace render FILE`` — a JSONL run trace to a standalone HTML view.
+``review`` takes three optional inputs besides the PR itself, and they
+compose: ``--spec URL_OR_PATH`` (repeatable) grounds the review against specs
+or tickets and replaces ``PRXREF_SPEC_SOURCES``; ``--rules-file PATH`` adds a
+team review-rules file (``PRXREF_REVIEW_RULES``); ``--context-file PATH``
+names the ticket the PR implements (``PRXREF_TICKET_CONTEXT_FILE``), so each
+finding is marked in, out of, or of unknown ticket scope. Each flag wins over
+its variable, and ``--rules-file ""`` / ``--context-file ""`` turn the
+variable off for one run. Both files are read before any network call, so an
+unusable one is a configuration error. The webhook daemon reads the rules
+file from its own environment and never reads a ticket-context file.
+
+The replay flags review a pinned, reproducible input for evaluation:
+``--base-sha`` / ``--head-sha`` a commit range in the ``--pr-url``
+repository, ``--diff-file PATH`` a diff on disk (``--pr-url`` is then
+optional, and no forge is contacted without it), and ``--no-threads`` hides
+the PR's existing threads. Any of them makes the run a replay: it never
+posts, and its run record gains a ``replay`` stamp. They are validated
+before the URL is parsed, and a bad set exits 2 naming the flag. The webhook
+daemon never replays.
+
Non-blocking doctrine: ``review`` exits 0 on all review errors (empty diffs,
network failures, LLM timeouts, bad credentials), printing diagnostic notes to
stderr so a pipeline step never fails the build over an advisor's error. The one
@@ -17,11 +38,14 @@
``PRXREF_FAIL_ON`` is the one opt-out of that doctrine. The default ``never``
is the doctrine itself: findings never move the exit code. ``error`` exits 1
when the completed review carries an active error-severity finding; ``any``
-exits 1 on any active finding; and under either value a review that fails to
-complete also exits 1, because a gate that silently passes on a broken run is
-worse than none. An unrecognized PR URL still exits 0 under every value —
-nothing was reviewed, so there is no outcome to gate on. The webhook daemon
-has no exit code and is unaffected by the knob.
+exits 1 on any active finding; and under either value a review that does not
+complete also exits 1 — it crashes, or it ends with verdict ``Error`` (the
+forge could not be read, the diff could not be parsed or chunked, or every
+chunk review failed) — because a gate that silently passes on a broken run is
+worse than none. An empty PR diff is not a failure: it is reviewed as
+``Approved`` and exits 0. An unrecognized PR URL still exits 0 under every
+value — nothing was reviewed, so there is no outcome to gate on. The webhook
+daemon has no exit code and is unaffected by the knob.
``PRXREF_DRY_RUN=1`` suppresses every write to the forge on both paths — the
one-shot review and the webhook daemon — and ``--no-post`` does the same for a
@@ -38,15 +62,23 @@
import importlib
import json
import logging
+import os
+import re
import sys
import time
+from dataclasses import dataclass, replace
from pathlib import Path
from typing import Any
import prxref
from prxref.config import load_config, make_forge
+from prxref.costs import cost_label
from prxref.forges.base import detect_forge
+from prxref.forges.replay import LocalDiffForge, ReplayForge
from prxref.llm import ConfigError
+from prxref.rules import load_review_rules
+from prxref.ticket import load_ticket_context
+from prxref.triage import SCOPE_IN, SCOPE_OUT, normalize_scope
from prxref.viz import render_file
logger = logging.getLogger("prxref")
@@ -55,7 +87,7 @@
def _build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(
prog="prxref",
- description="Fast automated AI code review for Bitbucket, GitLab, and GitHub.",
+ description="Fast automated AI code review for Bitbucket, GitLab, GitHub, and Azure DevOps.",
)
parser.add_argument(
"--version",
@@ -67,8 +99,11 @@ def _build_parser() -> argparse.ArgumentParser:
rev = sub.add_parser("review", help="review one PR/MR from its web URL")
rev.add_argument(
"--pr-url",
- required=True,
- help="full URL of the PR or MR on Bitbucket, GitHub, or GitLab",
+ default=None,
+ help=(
+ "full URL of the PR or MR on Bitbucket, GitHub, GitLab, or Azure "
+ "DevOps (required unless --diff-file is given)"
+ ),
)
rev.add_argument(
"--no-post",
@@ -90,6 +125,74 @@ def _build_parser() -> argparse.ArgumentParser:
default=None,
help="override the per-model request deadline in seconds",
)
+ rev.add_argument(
+ "--spec",
+ action="append",
+ default=None,
+ metavar="URL_OR_PATH",
+ help=(
+ "spec/ticket source to review against; repeatable "
+ "(PRXREF_SPEC_SOURCES otherwise)"
+ ),
+ )
+ rev.add_argument(
+ "--rules-file",
+ default=None,
+ metavar="PATH",
+ help=(
+ "team review rules (Markdown/text) added to every review prompt; "
+ "overrides PRXREF_REVIEW_RULES, and '' turns it off for this run; "
+ "read it from a trusted checkout, never from the PR under review"
+ ),
+ )
+ rev.add_argument(
+ "--context-file",
+ default=None,
+ metavar="PATH",
+ help=(
+ "ticket context (plain text/Markdown) the PR is meant to "
+ "implement; findings get a scope of in/out/unknown against it; "
+ "overrides PRXREF_TICKET_CONTEXT_FILE, and '' turns it off for "
+ "this run"
+ ),
+ )
+ rev.add_argument(
+ "--base-sha",
+ default=None,
+ metavar="SHA",
+ help=(
+ "replay: review the range BASE...HEAD (merge-base diff, like the "
+ "PR's own) in the --pr-url repository; needs --head-sha; implies "
+ "no posting"
+ ),
+ )
+ rev.add_argument(
+ "--head-sha",
+ default=None,
+ metavar="SHA",
+ help=(
+ "replay: head commit of the pinned range; file context is read at "
+ "this commit; needs --base-sha"
+ ),
+ )
+ rev.add_argument(
+ "--no-threads",
+ action="store_true",
+ help=(
+ "replay: hide the PR's existing threads from the prompt and the "
+ "thread-dedup passes; implies no posting"
+ ),
+ )
+ rev.add_argument(
+ "--diff-file",
+ default=None,
+ metavar="PATH",
+ help=(
+ "replay: review this unified diff (git diff or git format-patch "
+ "output) instead of fetching one; --pr-url becomes optional; "
+ "implies no posting"
+ ),
+ )
rev.add_argument(
"-v",
"--verbose",
@@ -166,6 +269,40 @@ def _fmt_tokens(result: Any) -> str:
return f"{inp}+{out}"
+def _fmt_cost(result: Any) -> str:
+ """Render the run's cost for the ``-v`` line.
+
+ ``costs.cost_label`` of the record's ``cost_usd``, ``cost_estimated`` and
+ ``cost_api_equivalent`` (``$0.0007``, ``$0.0007 (API-equivalent)`` for a
+ claude-cli-priced run, ``~$0.0007 (est.)``, or ``cost unknown`` for
+ ``None``), and ``-`` when the result carries no ``cost_usd`` key at all.
+ An absent key means nothing measured the cost; ``None`` means it was
+ measured and no source could price it. The two are different claims, so
+ they print differently.
+ """
+ if not isinstance(result, dict) or "cost_usd" not in result:
+ return "-"
+ return cost_label(
+ result.get("cost_usd"), result.get("cost_estimated") is True,
+ api_equivalent=result.get("cost_api_equivalent") is True,
+ )
+
+
+def _dash(value: Any, width: int | None = None) -> str:
+ if value is None or value == "":
+ return "-"
+ text = str(value)
+ return text[:width] if width else text
+
+
+def _scope_counts(result: dict) -> tuple[int, int, int]:
+ active = result.get("findings_active")
+ scopes = [normalize_scope(getattr(f, "scope", None)) for f in active] if isinstance(active, list) else []
+ n_in = scopes.count(SCOPE_IN)
+ n_out = scopes.count(SCOPE_OUT)
+ return n_in, n_out, len(scopes) - n_in - n_out
+
+
def _print_summary(
result: Any,
elapsed_s: float,
@@ -173,28 +310,82 @@ def _print_summary(
verbose: bool,
out=None,
) -> None:
+ """Print the text-mode summary of one review.
+
+ Always printed: ``verdict:``; ``coverage:`` when a chunk failed;
+ ``size advisory:`` when the PR-size advisory fired; and ``replay:`` when
+ the run was a replay, so a replay can never be read as a live review.
+ Under ``-v`` it adds the finding counts, the ``elapsed/tokens/cost`` line,
+ and one line for each configured input: ``rules:``, ``ticket:`` (with the
+ active findings' scope counts), and ``spec:``. ``result`` may be partial,
+ or not a dict at all; a missing or ``None`` record prints nothing.
+ """
target = sys.stdout if out is None else out
+ record = result if isinstance(result, dict) else {}
verdict = result.get("verdict") if isinstance(result, dict) else result
print(f"verdict: {verdict if verdict is not None else 'done'}", file=target)
- failed = result.get("chunks_failed", 0) if isinstance(result, dict) else 0
+ failed = record.get("chunks_failed", 0)
if failed:
- reviewed = result.get("chunks_reviewed", 0)
+ reviewed = record.get("chunks_reviewed", 0)
print(f"coverage: {reviewed}/{reviewed + failed} chunks reviewed", file=target)
+ size = record.get("size_advisory")
+ if isinstance(size, dict) and size.get("message"):
+ print(f"size advisory: {size['message']}", file=target)
+ replay = record.get("replay")
+ if isinstance(replay, dict):
+ print(
+ f"replay: base={_dash(replay.get('base_sha'), 12)} head={_dash(replay.get('head_sha'), 12)} "
+ f"threads={_dash(replay.get('threads'))} diff_file={_dash(replay.get('diff_file'))}",
+ file=target,
+ )
if not verbose:
return
- dropped = result.get("findings_dropped", []) if isinstance(result, dict) else []
+ dropped = record.get("findings_dropped", [])
dropped = len(dropped) if isinstance(dropped, list) else 0
print(f"counts: {_fmt_counts(result)} (dropped: {dropped})", file=target)
- print(f"elapsed: {elapsed_s:.1f}s tokens: {_fmt_tokens(result)}", file=target)
+ print(f"elapsed: {elapsed_s:.1f}s tokens: {_fmt_tokens(result)} cost: {_fmt_cost(result)}", file=target)
+ rules = record.get("review_rules")
+ if isinstance(rules, dict):
+ truncated = f" (truncated at {_dash(rules.get('max_chars'))})" if rules.get("truncated") else ""
+ print(
+ f"rules: {_dash(rules.get('path'))} sha256={_dash(rules.get('sha256'), 12)} "
+ f"chars={_dash(rules.get('chars'))}{truncated}",
+ file=target,
+ )
+ ticket = record.get("ticket_context")
+ if isinstance(ticket, dict):
+ truncated = " truncated" if ticket.get("truncated") else ""
+ n_in, n_out, n_unknown = _scope_counts(record)
+ print(
+ f"ticket: {_dash(ticket.get('path'))} sha256={_dash(ticket.get('sha256'), 12)} "
+ f"chars={_dash(ticket.get('chars'))}{truncated} in={n_in} out={n_out} unknown={n_unknown}",
+ file=target,
+ )
+ spec = record.get("spec_grounding")
+ if isinstance(spec, dict):
+ print(
+ f"spec: {_dash(spec.get('ok'))}/{_dash(spec.get('sources'))} source(s) ok, "
+ f"{_dash(spec.get('constraints'))} constraint(s)",
+ file=target,
+ )
def _fmt_finding_line(f: Any) -> str:
- """Render one active finding as ``: (confidence 0.NN)``."""
+ """Render one active finding as ``: (confidence 0.NN)``.
+
+ A finding the ticket judged gains `` [scope: in]`` or `` [scope: out]``
+ after the frozen prefix; ``unknown`` (always the case without a ticket)
+ adds nothing.
+ """
severity = getattr(f, "severity", None) or ""
location = f"{getattr(f, 'file', '')}:{getattr(f, 'line', 0)}"
title = getattr(f, "title", None) or ""
confidence = getattr(f, "confidence", None) or 0.0
- return f"{severity} {location} {title} (confidence {confidence:.2f})"
+ line = f"{severity} {location} {title} (confidence {confidence:.2f})"
+ scope = getattr(f, "scope", None)
+ if scope in (SCOPE_IN, SCOPE_OUT):
+ line = f"{line} [scope: {scope}]"
+ return line
def _fmt_indented_body(body: str) -> str:
@@ -234,12 +425,18 @@ def _print_findings(result: Any, *, out=None) -> None:
def _finding_json(f: Any, *, drop_reason: str | None) -> dict:
"""Build one JSON finding row explicitly (``Finding`` is a dataclass, not
- JSON-serializable by default)."""
+ JSON-serializable by default).
+
+ ``scope`` is the finding's position relative to the ticket context
+ (``in``, ``out`` or ``unknown``); a finding object without the attribute
+ reports ``unknown``.
+ """
return {
"file": f.file,
"line": f.line,
"severity": f.severity,
"confidence": f.confidence,
+ "scope": getattr(f, "scope", "unknown"),
"title": f.title,
"body": f.body,
"drop_reason": drop_reason,
@@ -249,11 +446,21 @@ def _finding_json(f: Any, *, drop_reason: str | None) -> dict:
def _build_json_result(result: Any) -> dict:
"""Build the single JSON payload for ``--format json``.
+ Key order: ``verdict``, ``findings``, ``chunk_count``, ``chunks_reviewed``,
+ ``chunks_failed``, ``elapsed_ms``, ``input_tokens``, ``output_tokens``,
+ ``cost_usd``, ``cost_estimated``, ``posted``, ``review_rules``,
+ ``ticket_context``, ``spec_grounding``, ``size_advisory``, then
+ ``sampling`` and ``replay`` when present.
+
Tolerates an error-shaped or partial result (a dict missing keys, as an
- incomplete or failed run may return): every key defaults to ``None`` and
- ``findings`` defaults to ``[]`` rather than raising. ``sampling`` is
- forwarded only when the result already carries it — a sibling feature's
- key, not one this CLI invents.
+ incomplete or failed run may return): every always-present key defaults
+ to ``None`` and ``findings`` defaults to ``[]`` rather than raising. The
+ run-record keys new in 0.14 (``cost_usd`` through ``size_advisory``) are
+ always emitted and are ``null`` when their feature is off; ``cost_usd`` is
+ also ``null`` when no source could price the run, never ``0``.
+ ``sampling`` and ``replay`` are forwarded only when the result already
+ carries them. ``replay`` is on replay runs only, so a normal run's
+ payload has no ``replay`` key at all.
"""
if not isinstance(result, dict):
result = {}
@@ -273,37 +480,250 @@ def _build_json_result(result: Any) -> dict:
"elapsed_ms": result.get("elapsed_ms"),
"input_tokens": result.get("input_tokens"),
"output_tokens": result.get("output_tokens"),
+ "cost_usd": result.get("cost_usd"),
+ "cost_estimated": result.get("cost_estimated"),
"posted": result.get("posted"),
+ "review_rules": result.get("review_rules"),
+ "ticket_context": result.get("ticket_context"),
+ "spec_grounding": result.get("spec_grounding"),
+ "size_advisory": result.get("size_advisory"),
}
if "sampling" in result:
payload["sampling"] = result["sampling"]
+ if "replay" in result:
+ payload["replay"] = result["replay"]
return payload
+_FULL_SHA_RE = re.compile(r"[0-9a-fA-F]{40}(?:[0-9a-fA-F]{24})?")
+
+
+@dataclass(frozen=True)
+class _ReplayRequest:
+ """The validated replay flags of one ``review`` run (issue #65).
+
+ ``base_sha`` / ``head_sha`` are both full, lowercased SHAs or both
+ ``None``. ``diff_file`` is the path exactly as the operator typed it, and
+ ``diff_text`` is that file's text once ``_run_review`` has read it
+ (``None`` until then, and without ``--diff-file``).
+ """
+
+ base_sha: str | None = None
+ head_sha: str | None = None
+ no_threads: bool = False
+ diff_file: str | None = None
+ diff_text: str | None = None
+
+ def stamp(self, *, has_forge: bool) -> dict[str, Any]:
+ """The run record's ``replay`` stamp: four keys, in a fixed order, all present.
+
+ ``threads`` is ``"hidden"`` whenever the PR's threads were not
+ consulted: under ``--no-threads``, or with no forge at all
+ (``--diff-file`` without ``--pr-url``).
+ """
+ return {
+ "base_sha": self.base_sha,
+ "head_sha": self.head_sha,
+ "threads": "hidden" if self.no_threads or not has_forge else "shown",
+ "diff_file": self.diff_file,
+ }
+
+
+def _resolve_replay(
+ url: str | None,
+ *,
+ base_sha: str | None = None,
+ head_sha: str | None = None,
+ no_threads: bool = False,
+ diff_file: str | None = None,
+) -> _ReplayRequest | None:
+ """Validate the replay flags; ``None`` means a normal, non-replay run.
+
+ Pure: it reads nothing and calls nothing. ``_run_review`` calls it first,
+ before ``detect_forge``, so a bad set of replay flags exits 2 even next
+ to an unrecognised URL. A flag counts as given whenever it is not
+ ``None``, so an empty value is validated rather than ignored.
+
+ The checks run in this order, each a ``ConfigError`` naming its flag:
+ no ``--pr-url`` and no ``--diff-file``; only one of ``--base-sha`` /
+ ``--head-sha``; either one not a full 40- or 64-character hex SHA; the
+ two naming the same commit (compared lowercased); and a range without
+ ``--pr-url`` to resolve it in. Whether the forge can fetch the range is
+ only known once it exists, so ``_run_review`` checks that.
+ """
+ if url is None and diff_file is None:
+ raise ConfigError("--pr-url: required unless --diff-file is given")
+ if (base_sha is None) != (head_sha is None):
+ only = "--base-sha" if head_sha is None else "--head-sha"
+ raise ConfigError(f"--base-sha/--head-sha: must be given together (got only {only})")
+ if base_sha is not None and head_sha is not None:
+ for flag, value in (("--base-sha", base_sha), ("--head-sha", head_sha)):
+ if not _FULL_SHA_RE.fullmatch(value):
+ raise ConfigError(
+ f"{flag}: must be a full 40- or 64-character hex commit SHA, "
+ f"got {value!r} (resolve it with git rev-parse)"
+ )
+ base_sha, head_sha = base_sha.lower(), head_sha.lower()
+ if base_sha == head_sha:
+ raise ConfigError("--base-sha/--head-sha: must name two different commits")
+ if url is None:
+ raise ConfigError(
+ "--base-sha/--head-sha: need --pr-url (the range is resolved in "
+ "that PR's repository)"
+ )
+ if head_sha is None and not no_threads and diff_file is None:
+ return None
+ return _ReplayRequest(
+ base_sha=base_sha, head_sha=head_sha, no_threads=bool(no_threads),
+ diff_file=diff_file,
+ )
+
+
+def _read_diff_file(path: str) -> str:
+ """Read the ``--diff-file`` text; a file that cannot be read is a ``ConfigError``.
+
+ The message is ``--diff-file: cannot read '': ``, which
+ covers a missing file and a directory alike. Undecodable bytes are
+ replaced, not refused, because the diff is review input rather than
+ configuration. A blank file is not a configuration error either: the
+ replay forge raises on it, and the run ends as an ``Error`` run (exit 0,
+ or 1 under ``PRXREF_FAIL_ON=error`` or ``any``).
+ """
+ try:
+ return Path(path).read_text(encoding="utf-8", errors="replace")
+ except OSError as exc:
+ raise ConfigError(
+ f"--diff-file: cannot read {path!r}: {exc.strerror or exc}"
+ ) from exc
+
+
+def _replay_forge(forge: Any, ref: Any, replay: _ReplayRequest) -> ReplayForge:
+ """Wrap the ``--pr-url`` forge in a :class:`ReplayForge` for this replay.
+
+ A pinned range that has to be fetched (no ``--diff-file``) needs the
+ forge's optional ``get_compare_diff``; without it this raises the
+ ``ConfigError`` naming ``--base-sha/--head-sha`` (exit 2) before any
+ network call. Two combinations are allowed but logged as a WARNING,
+ because each leaks the PR's present into a replay: pinned SHAs without
+ ``--no-threads`` still show the PR's current threads, and a
+ ``--diff-file`` without ``--head-sha`` reads file context at the PR's
+ current head.
+ """
+ if (
+ replay.head_sha is not None
+ and replay.diff_text is None
+ and getattr(forge, "get_compare_diff", None) is None
+ ):
+ raise ConfigError(
+ f"--base-sha/--head-sha: the {ref.forge} forge cannot fetch a "
+ "pinned commit range"
+ )
+ if replay.head_sha is not None and not replay.no_threads:
+ logger.warning(
+ "replay at pinned SHAs still shows the PR's CURRENT threads to the "
+ "prompt and the dedup passes; add --no-threads for a blind replay"
+ )
+ if replay.diff_file is not None and replay.head_sha is None:
+ logger.warning(
+ "--diff-file with --pr-url and no --head-sha: file context is read "
+ "at the PR's current head, which may not match the file"
+ )
+ return ReplayForge(
+ forge, base_sha=replay.base_sha, head_sha=replay.head_sha,
+ hide_threads=replay.no_threads, diff_text=replay.diff_text,
+ )
+
+
+def _load_text_input(loader: Any, path: str, *, max_chars: int, source: str) -> Any:
+ """Run the rules or ticket-context ``loader``, fencing every failure into a ``ConfigError``.
+
+ The loaders raise ``ConfigError`` naming ``source`` themselves; an
+ ``OSError`` or ``ValueError`` that escapes one is re-raised as a
+ ``ConfigError`` naming it too. So an unusable file always exits 2 before
+ any network call, and nothing a loader raises can reach the orchestrator,
+ which reads the loaded object unfenced.
+ """
+ try:
+ return loader(path, max_chars=max_chars, source=source)
+ except ConfigError:
+ raise
+ except (OSError, ValueError) as exc:
+ raise ConfigError(f"{source}: cannot load {path!r}: {exc}") from exc
+
+
def _run_review(
- url: str,
+ url: str | None,
*,
post: bool = True,
max_chunks: int | None = None,
timeout: float | None = None,
trace_dir: str | None = None,
+ spec_sources: list[str] | None = None,
+ rules_file: str | None = None,
+ context_file: str | None = None,
+ base_sha: str | None = None,
+ head_sha: str | None = None,
+ no_threads: bool = False,
+ diff_file: str | None = None,
) -> Any:
- ref = detect_forge(url)
- if ref is None:
- return None
- # --max-chunks and --timeout arrive as load_config overrides (None is
- # ignored), so each flag is range-checked on exactly the same path as its
- # environment variable and its precedence is derived once, here. There is
+ replay = _resolve_replay(
+ url, base_sha=base_sha, head_sha=head_sha, no_threads=no_threads,
+ diff_file=diff_file,
+ )
+ # The diff file is read with the flags, before the URL is parsed, so an
+ # unreadable one exits 2 whatever the URL. Without --pr-url it is the
+ # whole input: a synthetic "local" ref, and no forge is ever built.
+ if replay is not None and replay.diff_file is not None:
+ replay = replace(replay, diff_text=_read_diff_file(replay.diff_file))
+ if url is None:
+ ref = LocalDiffForge.ref_for(replay.diff_file)
+ else:
+ ref = detect_forge(url)
+ if ref is None:
+ return None
+ # --max-chunks, --timeout, --spec, --rules-file and --context-file arrive
+ # as load_config overrides (None is ignored, "" is not), so each flag rides
+ # exactly the path its environment variable does: --max-chunks and
+ # --timeout are range-checked on the same pass as PRXREF_MAX_CHUNKS and
+ # PRXREF_LLM_TIMEOUT, --spec replaces PRXREF_SPEC_SOURCES wholesale rather
+ # than merging with it, and --rules-file "" / --context-file "" blank
+ # their variable for one run. Precedence is derived once, here. There is
# deliberately no way to inject a pre-built config dict: that would bypass
# _check_ranges and make every range guarantee conditional on nobody using
- # the bypass.
+ # the bypass. --timeout only ever feeds llm_timeout, for the LLM client:
+ # orchestrate_review has no timeout parameter.
cfg = load_config(
max_chunks=max_chunks,
llm_timeout=timeout,
trace_dir=trace_dir,
+ spec_sources=spec_sources,
+ review_rules=rules_file,
+ ticket_context_file=context_file,
# The operator typed a flag, so a rejection has to name the flag. Only
# the CLI knows that spelling; config takes the label and reports it.
- source_labels={"max_chunks": "--max-chunks", "llm_timeout": "--timeout"},
+ source_labels={
+ "max_chunks": "--max-chunks",
+ "llm_timeout": "--timeout",
+ "spec_sources": "--spec",
+ "review_rules": "--rules-file",
+ "ticket_context_file": "--context-file",
+ },
+ )
+ # Both files are read here, after config and before make_forge and the LLM
+ # client, so an unusable one exits 2 before any network I/O. load_config
+ # stays I/O-free. Each is reported under the input that supplied its
+ # path: the flag whenever it was given, else the variable.
+ rules = _load_text_input(
+ load_review_rules, cfg["review_rules"],
+ max_chars=cfg["review_rules_max_chars"],
+ source="--rules-file" if rules_file is not None else "PRXREF_REVIEW_RULES",
+ )
+ ticket = _load_text_input(
+ load_ticket_context, cfg["ticket_context_file"],
+ max_chars=cfg["ticket_context_max_chars"],
+ source=(
+ "--context-file" if context_file is not None else "PRXREF_TICKET_CONTEXT_FILE"
+ ),
)
# PRXREF_DRY_RUN is the standing "never write to the forge" switch and
# --no-post is the per-invocation one; either alone suppresses posting, so
@@ -314,7 +734,18 @@ def _run_review(
if post and cfg["dry_run"]:
logger.info("PRXREF_DRY_RUN=1: reviewing %s without posting to the forge", ref.url)
post = False
- forge = make_forge(ref)
+ # A replay reviews a pinned input for evaluation, never the live PR as it
+ # stands, so it must never write: any replay flag turns posting off, with
+ # or without --no-post. The replay forges also refuse every write.
+ if replay is not None and post:
+ logger.info("replay run: posting to the forge is disabled")
+ post = False
+ if url is None:
+ forge = LocalDiffForge(replay.diff_text, path=replay.diff_file)
+ else:
+ forge = make_forge(ref)
+ if replay is not None:
+ forge = _replay_forge(forge, ref, replay)
llm = importlib.import_module("prxref.llm_backends").create_llm_client(cfg)
orchestrate = importlib.import_module("prxref.orchestrator").orchestrate_review
return orchestrate(
@@ -342,6 +773,21 @@ def _run_review(
post_verdict=cfg["post_verdict"],
trace_file=cfg["trace_file"],
trace_dir=cfg["trace_dir"],
+ spec_sources=cfg["spec_sources"],
+ spec_max_chars=cfg["spec_max_chars"],
+ spec_digest_tokens=cfg["spec_digest_tokens"],
+ jira_base_url=cfg["jira_base_url"],
+ jira_email=cfg["jira_email"],
+ jira_api_token=cfg["jira_api_token"],
+ rules=rules,
+ ticket=ticket,
+ # Already parsed by load_config into {model: costs.ModelPrice}.
+ price_table=cfg["price_table"],
+ post_cost=cfg["post_cost"],
+ size_warn_lines=cfg["size_warn_lines"],
+ size_warn_files=cfg["size_warn_files"],
+ size_ignore_globs=cfg["size_ignore_globs"],
+ replay=replay.stamp(has_forge=url is not None) if replay is not None else None,
)
@@ -351,27 +797,45 @@ def _webhook_handler(url: str) -> None:
``post=True`` is the daemon's intent, not its last word: ``_run_review``
downgrades it when the configured dry run says so, which is the only way to
observe the daemon against a real repo without writing to it.
+
+ ``context_file=""`` blanks ``PRXREF_TICKET_CONTEXT_FILE`` for every
+ webhook: one static ticket file cannot describe every PR the daemon sees,
+ so its findings always carry scope ``unknown``. The team rules file still
+ comes from the daemon's environment, re-read on every webhook. The daemon
+ passes no replay flag, so it never replays.
"""
try:
- _run_review(url, post=True)
+ _run_review(url, post=True, context_file="")
except Exception:
logger.exception("webhook review failed for %s", url)
def _fail_on_exit(result: Any, fail_on: str) -> tuple[int, str | None]:
- """The exit code a completed review earns under the ``fail_on`` policy.
+ """The exit code a returned review result earns under the ``fail_on`` policy.
- Severity is compared exactly as the verdict is built in the orchestrator
- (``Request-Changes`` iff an active finding has severity ``error``), so the
- gate and the posted verdict can never disagree about what counts. A result
- without parseable findings is tolerated the way ``_fmt_counts`` tolerates
- one: nothing countable means nothing to gate on.
+ ``never`` is always 0. Under ``error`` and ``any``, a result with verdict
+ ``Error`` exits 1 whatever its findings: the orchestrator returns one
+ instead of raising when the forge could not be read, the diff could not be
+ parsed or chunked, or every chunk review failed, so it is a review that did
+ not complete — the same outcome as the crash ``_cmd_review`` gates, and one
+ a gating lane must not read as green.
+
+ Otherwise severity is compared exactly as the verdict is built in the
+ orchestrator (``Request-Changes`` iff an active finding has severity
+ ``error``), so the gate and the posted verdict can never disagree about
+ what counts. A result without parseable findings is tolerated the way
+ ``_fmt_counts`` tolerates one: nothing countable means nothing to gate on.
Returns the exit code and, when the gate fires, the stderr line that says
why — silence would read as a crash rather than a decision.
"""
if fail_on == "never":
return 0, None
+ if isinstance(result, dict) and result.get("verdict") == "Error":
+ return 1, (
+ f"PRXREF_FAIL_ON={fail_on}: review did not complete "
+ "(verdict Error); exiting 1"
+ )
findings = result.get("findings_active") if isinstance(result, dict) else None
if not isinstance(findings, list):
return 0, None
@@ -408,6 +872,13 @@ def _cmd_review(args: argparse.Namespace) -> int:
max_chunks=args.max_chunks,
timeout=args.timeout,
trace_dir=args.trace_dir,
+ spec_sources=args.spec,
+ rules_file=args.rules_file,
+ context_file=args.context_file,
+ base_sha=args.base_sha,
+ head_sha=args.head_sha,
+ no_threads=args.no_threads,
+ diff_file=args.diff_file,
)
except ConfigError as exc:
print(f"configuration error: {exc}", file=sys.stderr)
@@ -430,7 +901,9 @@ def _cmd_review(args: argparse.Namespace) -> int:
"pull-requests, GitHub pull, or GitLab merge_requests link "
"(bitbucket.org, github.com, gitlab.com, or a self-hosted "
"Bitbucket Data Center, GitHub Enterprise Server, or GitLab "
- "host); the URL must keep the forge's own path shape.",
+ "host), or an Azure DevOps pullrequest link (dev.azure.com, "
+ "*.visualstudio.com, or an Azure DevOps Server host); the URL "
+ "must keep the forge's own path shape.",
file=sys.stderr,
)
return 0
@@ -449,6 +922,14 @@ def _cmd_review(args: argparse.Namespace) -> int:
def _cmd_serve(args: argparse.Namespace) -> int:
+ # Said once at startup rather than per webhook: _webhook_handler blanks
+ # the variable on every review, and an operator who set it should learn
+ # that before the first PR arrives, not infer it from unscoped findings.
+ if os.environ.get("PRXREF_TICKET_CONTEXT_FILE", "").strip():
+ logger.warning(
+ "PRXREF_TICKET_CONTEXT_FILE is ignored by prxref serve: one file "
+ "cannot describe every PR"
+ )
serve_fn = importlib.import_module("prxref.webhooks").serve
serve_fn(port=args.port, host=args.host, handler=_webhook_handler)
return 0
diff --git a/src/prxref/config.py b/src/prxref/config.py
index 62e798b..9ced00e 100644
--- a/src/prxref/config.py
+++ b/src/prxref/config.py
@@ -3,11 +3,20 @@
Canonical environment-variable table (every name prefixed PRXREF_):
LLM / pipeline:
- PRXREF_LLM_BACKEND LLM backend: openai-compat | ferry | http (aliases) | litellm
- PRXREF_LLM_BASE_URL Base URL for the chosen backend (optional)
- PRXREF_LLM_API_KEY API key for the chosen backend (optional)
- PRXREF_LLM_MODELS Comma-separated model fallback chain, first
- that answers wins; empty = backend default
+ PRXREF_LLM_BACKEND LLM backend: openai-compat | ferry | http
+ (aliases) | litellm | claude-cli | kiro-cli,
+ read case-insensitively; any other value is
+ a configuration error
+ PRXREF_LLM_BASE_URL Base URL of the OpenAI-compatible endpoint;
+ required for openai-compat/ferry/http, not
+ used by litellm, claude-cli or kiro-cli (a
+ set value is ignored there with one INFO
+ line)
+ PRXREF_LLM_API_KEY API key for the openai-compat endpoint
+ (optional; empty for a local no-auth server)
+ PRXREF_LLM_MODELS Comma- or whitespace-separated model fallback
+ chain, first that answers wins; required by
+ every backend
PRXREF_LLM_REASONING_EFFORT Reasoning effort for models that cannot
disable reasoning; provider-specific string,
passed through unvalidated; empty = omit
@@ -28,6 +37,13 @@
"seed" in the request; >= 0 (0 is a valid
seed); empty or unset falls back to the
factory's once-per-process seed
+ PRXREF_LLM_CLI_PATH claude-cli / kiro-cli only: path to the CLI
+ binary, ``~`` expanded; empty = "claude" or
+ "kiro-cli" on PATH. Not found = configuration
+ error
+ PRXREF_LLM_CLI_CONCURRENCY claude-cli / kiro-cli only: max CLI
+ processes one client runs at once; positive
+ int (default 2)
PRXREF_CONFIDENCE_FLOOR Findings below this confidence are dropped;
a probability in [0.0, 1.0] (default 0.6)
PRXREF_MAX_ERROR_FINDINGS Max error-severity findings reported per
@@ -81,9 +97,14 @@
completed review carries an active
error-severity finding; "any" exits 1 on
any active finding. Under "error" and
- "any", a review that fails to complete
- also exits 1. The webhook daemon has no
- exit code and is unaffected.
+ "any", a review that does not complete
+ also exits 1: it crashes, or it ends with
+ verdict "Error" (the forge could not be
+ read, the diff could not be parsed or
+ chunked, or every chunk review failed).
+ An empty PR diff is not a failure
+ (verdict "Approved", exit 0). The webhook
+ daemon has no exit code and is unaffected.
PRXREF_POST_MODE What gets posted to the forge:
"summary+inline" (default) | "summary" |
"inline". Any other value is a
@@ -91,9 +112,99 @@
PRXREF_DRY_RUN / ``--no-post``, which post
nothing in any mode.
PRXREF_POST_VERDICT literal "1" keeps the verdict stamp in the
- posted summary; any other value renders the
- summary without it (default on). The
- total-failure notice always names its status.
+ posted summary; any other value renders the
+ summary without it (default on). The
+ total-failure notice always names its status.
+ PRXREF_PRICE_TABLE Fallback price table for runs whose backend
+ reports no dollar cost: inline JSON (first
+ non-space character "{") or a path to a JSON
+ file, mapping model name -> {"input": USD,
+ "output": USD} per million tokens, keyed on
+ the model name the run reports. A reported
+ cost always wins; a figure from this table
+ marks the run cost_estimated. A malformed
+ table is a configuration error. Empty (the
+ default) estimates nothing. After loading,
+ the key holds the parsed table (a dict).
+ PRXREF_POST_COST literal "1" appends the run's dollar cost to
+ the posted summary's attribution line
+ (default off). The cost is always in the run
+ record, --format json and the traces.
+ PRXREF_SIZE_WARN_LINES Advisory-only threshold on lines changed
+ (added + removed, from the parsed diff,
+ excluding lock and generated files); one
+ non-blocking line tops the summary when the
+ count is above it. Unset (default) disables
+ it; >= 0, where 0 is a legal threshold
+ distinct from unset. Never affects the
+ verdict or the exit code.
+ PRXREF_SIZE_WARN_FILES Same contract as PRXREF_SIZE_WARN_LINES,
+ thresholding files changed instead.
+ PRXREF_SIZE_IGNORE_GLOBS Extra fnmatch globs (case-sensitive, matched
+ against the full diff path, ``*`` crosses
+ ``/``) excluded from both size counts, ADDED
+ to the built-in lock-file and generated-file
+ detection, never replacing it. Empty
+ (default) adds nothing.
+ PRXREF_SPEC_SOURCES Spec/ticket sources to review against, as
+ comma- or whitespace-separated web URLs and
+ local file/dir paths; the repeatable
+ ``--spec`` flag replaces (never merges) this
+ list. Jira ticket URLs are routed to the
+ Jira REST fetcher below.
+ PRXREF_SPEC_MAX_CHARS Raw fetched characters kept per spec source
+ before pruning; positive int (default
+ 120000)
+ PRXREF_SPEC_DIGEST_TOKENS Token budget for the spec digest injected
+ into worker prompts; positive int (default
+ 3000)
+ PRXREF_REVIEW_RULES Path to a team review-rules file (Markdown,
+ optional front matter with a ``severity:``
+ map) added to every review prompt, by
+ ``prxref review`` and the webhook daemon
+ alike. A missing, unreadable or malformed
+ file is a configuration error. Read it from a checkout
+ the PR cannot change. ``--rules-file PATH``
+ wins; ``--rules-file ""`` turns it off for
+ one run. Empty (the default) = no rules.
+ PRXREF_REVIEW_RULES_MAX_CHARS Characters of the rules body (after the
+ front matter) kept in the prompt; longer is
+ truncated with a warning; positive int
+ (default 12000)
+ PRXREF_TICKET_CONTEXT_FILE Path to a text file holding the ticket this
+ PR implements; each finding is then marked
+ in, out of, or of unknown ticket scope. An
+ empty (or whitespace-only) file means "this
+ PR has no ticket". A missing, unreadable or
+ non-UTF-8 file is a configuration error.
+ Ignored by ``prxref serve``.
+ ``--context-file PATH`` wins;
+ ``--context-file ""`` turns it off for one
+ run. Empty (the default) = no ticket.
+ PRXREF_TICKET_CONTEXT_MAX_CHARS
+ Characters of ticket text kept in the
+ prompt; longer is truncated with a visible
+ marker; positive int (default 6000)
+
+Spec sources / Jira:
+ PRXREF_JIRA_BASE_URL Jira base URL (scheme://host plus any
+ context path) that ticket fetches are
+ looked up on, overriding a ticket URL's own
+ base (a self-hosted board often sits behind
+ a different REST host than its browse URL).
+ Jira credentials are only ever sent here;
+ empty = the ticket URL's own base, fetched
+ anonymously
+ PRXREF_JIRA_EMAIL Jira account email for HTTP basic auth,
+ used only together with
+ PRXREF_JIRA_BASE_URL; without it the fetch
+ is anonymous and a warning is logged.
+ Missing credentials are a fetch failure
+ (the review proceeds un-grounded), never a
+ configuration error.
+ PRXREF_JIRA_API_TOKEN Jira API token paired with
+ PRXREF_JIRA_EMAIL for HTTP basic auth, sent
+ only to PRXREF_JIRA_BASE_URL
Per-forge auth:
PRXREF_BITBUCKET_TOKEN Bitbucket Cloud bearer token
@@ -106,14 +217,28 @@
PRXREF_GITHUB_TOKEN GitHub token (github.com)
PRXREF_GITHUB_ENTERPRISE_TOKEN GitHub Enterprise token (GHES hosts)
PRXREF_GITLAB_TOKEN GitLab token
+ PRXREF_AZURE_DEVOPS_TOKEN Azure DevOps personal access token (Code
+ Read to review, Read & write to post); empty
+ falls back to SYSTEM_ACCESSTOKEN, then to
+ anonymous access (public projects only)
Webhooks:
PRXREF_BITBUCKET_WEBHOOK_SECRET HMAC secret for Bitbucket webhook payloads
PRXREF_GITHUB_WEBHOOK_SECRET HMAC secret for GitHub webhook payloads
PRXREF_GITLAB_WEBHOOK_SECRET HMAC secret for GitLab webhook payloads
+ PRXREF_AZURE_DEVOPS_WEBHOOK_SECRET
+ Basic-auth password of the Azure DevOps
+ service hook (the user name is ignored);
+ empty rejects Azure DevOps webhooks with
+ 401 unless PRXREF_ALLOW_UNSIGNED is "1"
PRXREF_ALLOW_UNSIGNED literal "1" accepts unsigned
webhooks (default off; insecure)
+List-valued keys (PRXREF_LLM_MODELS, PRXREF_SPEC_SOURCES and
+PRXREF_SIZE_IGNORE_GLOBS) split on any run of commas and/or whitespace, so no
+item can contain either; a glob that must match a literal space writes it as
+``?``.
+
Precedence: built-in defaults < environment < ``overrides`` kwargs.
An error names the source that actually supplied the offending value — the
environment variable that was read (including a legacy alias), or the caller's
@@ -132,10 +257,12 @@
import math
import os
+import re
from typing import NamedTuple
from prxref.forges.base import Forge, PRRef
+from . import costs
from .llm import ConfigError
from .quality import DEFAULT_CONFIDENCE_FLOOR, DEFAULT_MAX_ERRORS
from .triage import (
@@ -161,6 +288,8 @@
# the seed is a first-class int key — coerced and range-checked here —
# because "no seed" is representable in its own type.
"llm_seed": None,
+ "llm_cli_path": "",
+ "llm_cli_concurrency": 2,
"confidence_floor": DEFAULT_CONFIDENCE_FLOOR,
"max_error_findings": DEFAULT_MAX_ERRORS,
"max_chunks": 8,
@@ -179,6 +308,26 @@
"trace_dir": "",
"post_mode": "summary+inline",
"post_verdict": True,
+ # A str on the way in (inline JSON or a file path); _check_price_table
+ # replaces it with the parsed dict, so a loaded config never holds the raw
+ # text and every consumer sees one type.
+ "price_table": "",
+ "post_cost": False,
+ # ``None`` = the advisory is off, the second "None means off" class next
+ # to ``llm_seed``: 0 is a legal threshold, so it cannot spell "unset".
+ "size_warn_lines": None,
+ "size_warn_files": None,
+ "size_ignore_globs": [],
+ "spec_sources": [],
+ "spec_max_chars": 120000,
+ "spec_digest_tokens": 3000,
+ "review_rules": "",
+ "review_rules_max_chars": 12000,
+ "ticket_context_file": "",
+ "ticket_context_max_chars": 6000,
+ "jira_base_url": "",
+ "jira_email": "",
+ "jira_api_token": "",
"bitbucket_token": "",
"bitbucket_user": "",
"bitbucket_app_password": "",
@@ -188,9 +337,11 @@
"github_token": "",
"github_enterprise_token": "",
"gitlab_token": "",
+ "azure_devops_token": "",
"bitbucket_webhook_secret": "",
"github_webhook_secret": "",
"gitlab_webhook_secret": "",
+ "azure_devops_webhook_secret": "",
"allow_unsigned": False,
}
@@ -198,10 +349,13 @@
"max_error_findings", "max_chunks", "llm_max_tokens", "llm_seed",
"chunk_token_budget", "chunk_max_files", "chunk_context_lines",
"max_workers", "max_inline_comments",
+ "spec_max_chars", "spec_digest_tokens",
+ "llm_cli_concurrency", "review_rules_max_chars",
+ "ticket_context_max_chars", "size_warn_lines", "size_warn_files",
})
_FLOAT_KEYS = frozenset({"confidence_floor", "llm_timeout"})
-_BOOL_KEYS = frozenset({"allow_unsigned", "dry_run", "post_verdict"})
-_LIST_KEYS = frozenset({"llm_models"})
+_BOOL_KEYS = frozenset({"allow_unsigned", "dry_run", "post_verdict", "post_cost"})
+_LIST_KEYS = frozenset({"llm_models", "spec_sources", "size_ignore_globs"})
# An enum-valued key has no numeric interval to check, so its legal vocabulary
# is declared here instead and enforced on the same pass as the ranges. A
@@ -234,8 +388,9 @@ class _Range(NamedTuple):
worker count (``ThreadPoolExecutor`` rejects it) and for a chunk count
(``build_chunks`` raises on the overflow branch). Zero IS meaningful for the
error cap, where it means "report no errors", for the context-line count,
- where it means "emit the changed lines only", and for the sampling seed,
- where 0 is a perfectly valid seed.
+ where it means "emit the changed lines only", for the sampling seed,
+ where 0 is a perfectly valid seed, and for the PR-size thresholds, where
+ 0 flags any change at all.
"""
low: float
@@ -275,6 +430,13 @@ def describe(self) -> str:
"chunk_context_lines": _Range(0, low_inclusive=True),
"max_error_findings": _Range(0, low_inclusive=True),
"llm_seed": _Range(0, low_inclusive=True),
+ "spec_max_chars": _Range(0),
+ "spec_digest_tokens": _Range(0),
+ "llm_cli_concurrency": _Range(0),
+ "review_rules_max_chars": _Range(0),
+ "ticket_context_max_chars": _Range(0),
+ "size_warn_lines": _Range(0, low_inclusive=True),
+ "size_warn_files": _Range(0, low_inclusive=True),
"confidence_floor": _Range(0.0, 1.0, low_inclusive=True),
}
@@ -308,7 +470,9 @@ def _coerce_env(key: str, raw: str, source: str) -> object:
if key in _BOOL_KEYS:
return _truthy(raw)
if key in _LIST_KEYS:
- return [part.strip() for part in raw.split(",") if part.strip()]
+ return [
+ part.strip() for part in re.split(r"[,\s]+", raw) if part.strip()
+ ]
return raw
except ValueError as exc:
raise ConfigError(f"{source}: {exc}") from exc
@@ -331,7 +495,10 @@ def _check_ranges(cfg: dict[str, object], sources: dict[str, str]) -> None:
lives in the factory (``llm_seed``: unset falls back to the
once-per-process seed resolved in ``llm_backends.create_llm_client``):
there is no number to range-check, and
- "not configured" is not a violation. Every value that is not ``None`` —
+ "not configured" is not a violation. The PR-size thresholds
+ (``size_warn_lines``, ``size_warn_files``) are the second such class:
+ ``None`` means the advisory is off, because 0 is a legal threshold and
+ cannot double as "unset". Every value that is not ``None`` —
including one smuggled in through an override — is still checked.
"""
for key, rng in sorted(_RANGES.items()):
@@ -378,15 +545,32 @@ def _check_post_mode(cfg: dict[str, object], sources: dict[str, str]) -> None:
)
+def _check_price_table(cfg: dict[str, object], sources: dict[str, str]) -> None:
+ """Parse ``price_table`` in place, rejecting a malformed table.
+
+ Same doctrine as :func:`_check_post_mode`: it runs after environment AND
+ overrides, and a failure is a ``ConfigError`` naming whichever input
+ supplied the value, so a typo'd price is exit 2 before anything is
+ reviewed, never a silently wrong estimate. The value may be inline JSON,
+ a path to a JSON file, or a mapping from a library caller; all three are
+ validated by :func:`prxref.costs.parse_price_table`. Afterwards the key
+ holds a ``dict[str, costs.ModelPrice]``, ``{}`` when no table is set.
+ """
+ cfg["price_table"] = costs.parse_price_table(
+ cfg["price_table"], source=sources["price_table"]
+ )
+
+
def load_config(
*, source_labels: dict[str, str] | None = None, **overrides: object
) -> dict:
"""Build the runtime config dict from defaults, environment, then overrides.
Keys mirror the env table above (lowercase, no prefix). Env values are
- type-coerced per key (int / float / bool / comma-list / str); an empty or
- whitespace-only value reads as unset. A malformed value, one out of its
- numeric range, or one outside its key's allowed vocabulary raises
+ type-coerced per key (int / float / bool / comma-or-whitespace list /
+ str); an empty or whitespace-only value reads as unset. ``price_table``
+ comes back parsed, as a dict. A malformed value, one out of its numeric
+ range, or one outside its key's allowed vocabulary raises
:class:`~prxref.llm.ConfigError` naming the input that supplied it, which
the CLI turns into exit 2.
@@ -397,7 +581,10 @@ def load_config(
Config itself knows no flag names: the caller that owns the surface names
it.
"""
- cfg: dict[str, object] = dict(_DEFAULTS)
+ cfg: dict[str, object] = {
+ key: list(value) if isinstance(value, list) else value
+ for key, value in _DEFAULTS.items()
+ }
# What supplied each value, for error messages. Defaults start out attributed
# to their environment variable: that is the name an operator would set to
# change one, and a built-in default is never out of range anyway.
@@ -425,6 +612,7 @@ def load_config(
_check_ranges(cfg, sources)
_check_choices(cfg, sources)
_check_post_mode(cfg, sources)
+ _check_price_table(cfg, sources)
return cfg
@@ -434,13 +622,14 @@ def make_forge(ref: PRRef, session=None) -> Forge:
``session`` optionally injects a custom ``requests.Session`` (tests,
shared connection pools). Unknown forge names raise ``ValueError``.
"""
- from prxref.forges import bitbucket, bitbucket_server, github, gitlab
+ from prxref.forges import azure_devops, bitbucket, bitbucket_server, github, gitlab
impls = {
"bitbucket": bitbucket.ForgeImpl,
"bitbucket-server": bitbucket_server.ForgeImpl,
"github": github.ForgeImpl,
"gitlab": gitlab.ForgeImpl,
+ "azure-devops": azure_devops.ForgeImpl,
}
impl = impls.get(ref.forge)
if impl is None:
diff --git a/src/prxref/costs.py b/src/prxref/costs.py
new file mode 100644
index 0000000..ec0cfcb
--- /dev/null
+++ b/src/prxref/costs.py
@@ -0,0 +1,407 @@
+"""Dollar cost of a review run: reported, estimated, or unknown.
+
+A run's cost is always in exactly one of three states, and they are never
+blended into a number that looks more certain than it is:
+
+- **Reported.** The backend returned a dollar figure for the call:
+ OpenRouter's body ``usage.cost``, a LiteLLM gateway's (or llm-ferry's)
+ ``x-litellm-response-cost`` header, the litellm SDK's ``response_cost``, or
+ claude-cli's ``total_cost_usd``. A reported figure always wins.
+- **Estimated.** No figure came back, but ``PRXREF_PRICE_TABLE`` prices the
+ exact model name the call reported. The run is flagged ``cost_estimated``.
+- **Unknown.** Neither. The run's ``cost_usd`` is ``None``: never ``0`` and
+ never a partial sum of the units that were priced.
+
+Backends only report (:func:`valid_usd`, :func:`combine_reported`); nothing
+below the orchestrator ever sees the price table. Estimation happens once,
+over the finished run, in :func:`run_cost`. A run that made no LLM request at
+all never calls it: its cost is a known ``0.0``.
+
+claude-cli's figure is what the call would cost at API list price, not what
+a subscription is invoiced, so a run whose every reported figure came from
+it (:func:`api_equivalent_run`) is labelled ``(API-equivalent)`` wherever a
+person reads the cost (:func:`cost_label`).
+"""
+from __future__ import annotations
+
+import json
+import math
+from collections.abc import Iterable, Mapping, Sequence
+from pathlib import Path
+from typing import Any, NamedTuple
+
+from .llm import ConfigError
+
+_TOKENS_PER_PRICE_UNIT = 1_000_000
+_PRICE_FIELDS = ("input", "output")
+_TABLE_SHAPE = 'a JSON object mapping model name to {"input": USD, "output": USD} per million tokens'
+_ENTRY_EXAMPLE = '{"input": 0.15, "output": 0.60}'
+_UNKNOWN_MODEL = ""
+_SMALLEST_SHOWN = 0.0001
+_API_EQUIVALENT_SOURCES = frozenset({"claude-cli"})
+
+
+class ModelPrice(NamedTuple):
+ """List price of one model in USD per MILLION tokens."""
+
+ input: float
+ output: float
+
+
+class _DuplicateKeyError(ValueError):
+ def __init__(self, key: str) -> None:
+ super().__init__(key)
+ self.key = key
+
+
+def parse_price_table(
+ raw: str | Mapping[str, Any] | None, source: str = "PRXREF_PRICE_TABLE"
+) -> dict[str, ModelPrice]:
+ """Parse and strictly validate a price table.
+
+ ``raw`` is one of:
+
+ - ``None``, ``""`` or whitespace: no table, so ``{}``.
+ - a string whose first non-space character is ``{``: inline JSON.
+ - any other string: a path to a JSON file (``~`` expanded, read as UTF-8).
+ - a mapping: an already-decoded table from a library caller, validated
+ exactly like decoded JSON.
+
+ The table maps a model name to ``{"input": n, "output": n}``, each ``n`` a
+ finite number >= 0 in USD per million tokens. Model names are stripped
+ and must be non-empty and unique. Invalid JSON, an unreadable file, a
+ non-object at any level, a missing or unknown field (``"ouput"``), a
+ duplicate key, a bool, a numeric string, ``NaN``, ``Infinity`` or a
+ negative price each raise ``ConfigError`` whose message starts with
+ ``source``, so the CLI names the env var or flag that supplied it. Zero
+ prices are legal, for local or free models. Returns a new dict.
+ """
+ if raw is None:
+ return {}
+ if isinstance(raw, str):
+ text = raw.strip()
+ if not text:
+ return {}
+ data = _load_json(text, source, "") if text.startswith("{") else _load_json_file(text, source)
+ elif isinstance(raw, Mapping):
+ data = raw
+ else:
+ raise ConfigError(f"{source}: must be inline JSON or a path to a JSON file, got {type(raw).__name__}")
+ return _validate_table(data, source)
+
+
+def estimate_usd(
+ table: Mapping[str, ModelPrice], model: str, input_tokens: int, output_tokens: int
+) -> float | None:
+ """Price one call from ``table``, or ``None`` when the model has no entry.
+
+ The lookup is on the exact model name, never a prefix or a pattern:
+ a wrong estimate is worse than an unknown one.
+ """
+ price = table.get(model) if table and isinstance(model, str) else None
+ if price is None:
+ return None
+ return (
+ _count(input_tokens) * price.input / _TOKENS_PER_PRICE_UNIT
+ + _count(output_tokens) * price.output / _TOKENS_PER_PRICE_UNIT
+ )
+
+
+def valid_usd(value: object) -> float | None:
+ """Return ``value`` as a reported dollar amount, or ``None`` if it is not one.
+
+ Accepted: a finite real number >= 0 (not a bool), or a string that
+ parses to one, because response headers are strings. Everything else,
+ including ``NaN``, ``inf``, negatives and ``""``, is ``None``: an
+ unusable figure is no figure.
+ """
+ if isinstance(value, bool):
+ return None
+ if isinstance(value, str):
+ try:
+ number = float(value.strip())
+ except ValueError:
+ return None
+ elif isinstance(value, int | float):
+ try:
+ number = float(value)
+ except OverflowError:
+ return None
+ else:
+ return None
+ if not math.isfinite(number) or number < 0:
+ return None
+ return 0.0 if number == 0 else number
+
+
+def combine_reported(parts: Sequence[tuple[float | None, str]]) -> tuple[float | None, str]:
+ """Fold the reported costs of every attempt received inside ONE invoke.
+
+ Each part is ``(cost_usd, cost_source)``. Every attempt that came back
+ was billed, including truncated ones a fallback chain moved past, so the
+ figures are summed and the last part's source is kept. No parts, or any
+ part without a figure, gives ``(None, "")``: one unpriced attempt makes
+ the whole call's cost unknown.
+ """
+ parts = list(parts)
+ if not parts:
+ return None, ""
+ costs: list[float] = []
+ for cost, _source in parts:
+ if cost is None:
+ return None, ""
+ costs.append(cost)
+ return math.fsum(costs), parts[-1][1]
+
+
+def unit_cost(unit: Mapping[str, Any]) -> tuple[bool, float | None, str]:
+ """Read ``(received, cost_usd, cost_source)`` off one review unit.
+
+ ``unit`` is an orchestrator worker or sweep result, or a reviewer meta
+ dict. ``received`` is true when a completion came back: the unit names a
+ model or counts any tokens. A unit whose request raised has model ``""``
+ and zero tokens, so it was never received. The cost goes through
+ :func:`valid_usd`; a unit without cost keys (an older stub) reads as
+ ``(…, None, "")``. The source is ``""`` whenever the cost is ``None``.
+ """
+ received = (
+ bool(unit.get("model"))
+ or _count(unit.get("input_tokens")) > 0
+ or _count(unit.get("output_tokens")) > 0
+ )
+ cost = valid_usd(unit.get("cost_usd"))
+ source = str(unit.get("cost_source") or "") if cost is not None else ""
+ return received, cost, source
+
+
+def run_cost(
+ units: Iterable[Mapping[str, Any]], table: Mapping[str, ModelPrice] | None
+) -> tuple[float | None, bool, list[str]]:
+ """Total a run's cost: ``(cost_usd, cost_estimated, unpriced_models)``.
+
+ ``units`` are the orchestrator's chunk results plus the sweep; ``table``
+ is the parsed ``PRXREF_PRICE_TABLE`` (``None`` or ``{}`` estimates
+ nothing). Rules:
+
+ 1. A unit that was never received is skipped: its request raised, so no
+ completion came back to read a cost from.
+ 2. A received unit's reported cost is used whenever there is one, even if
+ the table also prices its model.
+ 3. Otherwise the table prices it, if the model has an exact entry and the
+ unit counted input tokens (every prompt has a system prompt, so zero
+ input tokens means the backend reported no usage and an estimate would
+ be a fake ``0``). Such a unit makes the run ``cost_estimated``.
+ 4. Otherwise the unit is unknown, and its model is named in
+ ``unpriced_models``.
+
+ No unit received gives ``(None, False, [])``: requests went out and
+ nothing came back to price. Any unknown unit gives ``(None, False,
+ sorted_models)``: unknown is ``None``, never ``0`` and never a partial
+ sum. Otherwise the total is rounded to 10 decimals, which removes float
+ noise and nothing else.
+
+ Boundaries: tokens and cost cover the same units, so a unit that failed
+ after its response arrived (a parse failure, a truncation) was billed and
+ is counted. A request abandoned at the deadline returned nothing and adds
+ nothing, so a provider that bills abandoned generations may charge more
+ than this reports. An estimate prices every input token at the list rate,
+ so it ignores prompt-cache discounts a provider's own figure reflects.
+ """
+ parts: list[float] = []
+ unpriced: set[str] = set()
+ estimated = False
+ received_any = False
+ for unit in units:
+ received, reported, _source = unit_cost(unit)
+ if not received:
+ continue
+ received_any = True
+ if reported is not None:
+ parts.append(reported)
+ continue
+ model = unit.get("model")
+ model = model if isinstance(model, str) else ""
+ input_tokens = _count(unit.get("input_tokens"))
+ estimate = None
+ if table and model and input_tokens > 0:
+ estimate = estimate_usd(table, model, input_tokens, _count(unit.get("output_tokens")))
+ if estimate is None:
+ unpriced.add(model or _UNKNOWN_MODEL)
+ continue
+ parts.append(estimate)
+ estimated = True
+ if not received_any:
+ return None, False, []
+ if unpriced:
+ return None, False, sorted(unpriced)
+ return round(math.fsum(parts), 10), estimated, []
+
+
+def api_equivalent_run(units: Iterable[Mapping[str, Any]]) -> bool:
+ """Whether a run's reported cost is claude-cli's API-equivalent figure.
+
+ ``units`` are the same review units :func:`run_cost` totals. True when at
+ least one unit was received with a reported cost and EVERY such unit's
+ ``cost_source`` is ``"claude-cli"``; one reported unit from any other
+ source makes it false, because the total is then not an API-equivalent
+ figure. Only reported costs vote (read through :func:`unit_cost`): a unit
+ that raised, or one with no figure, has no source, so a claude-cli unit
+ whose cost is ``None`` does not count. An estimated run keeps its
+ ``(est.)`` label whatever this returns (:func:`cost_label`).
+ """
+ sources = [
+ source
+ for received, reported, source in (unit_cost(unit) for unit in units)
+ if received and reported is not None
+ ]
+ return bool(sources) and all(source in _API_EQUIVALENT_SOURCES for source in sources)
+
+
+def format_usd(value: float) -> str:
+ """Render a dollar amount for people.
+
+ ``0`` is ``$0.00``; anything above zero but below $0.0001 is
+ ``<$0.0001``; below $1 shows four decimals (``$0.0007``); $1 and up shows
+ two (``$1.23``). A nonzero cost never renders as ``$0.00``. A value that
+ is not a finite number >= 0 raises ``ValueError``.
+ """
+ if isinstance(value, bool) or not isinstance(value, int | float):
+ raise ValueError(f"not a dollar amount: {value!r}")
+ try:
+ number = float(value)
+ except OverflowError as e:
+ raise ValueError(f"not a dollar amount: {value!r}") from e
+ if not math.isfinite(number) or number < 0:
+ raise ValueError(f"not a dollar amount: {value!r}")
+ if number == 0:
+ return "$0.00"
+ if number < _SMALLEST_SHOWN:
+ return "<$0.0001"
+ if number < 1:
+ four = f"{number:.4f}"
+ if four != "1.0000":
+ return f"${four}"
+ return f"${number:.2f}"
+
+
+def cost_label(cost_usd: float | None, estimated: bool, *, api_equivalent: bool = False) -> str:
+ """Return the label a run's cost is shown with.
+
+ Four forms:
+
+ - ``"cost unknown"`` when there is no usable figure;
+ - ``"~$0.0007 (est.)"`` when it was estimated, whatever ``api_equivalent``
+ says, because an estimate is the price table's figure, not the CLI's;
+ - ``"$0.0007 (API-equivalent)"`` when it was reported and
+ ``api_equivalent`` is true (:func:`api_equivalent_run`): claude-cli's
+ list-price figure, not a subscription bill;
+ - ``"$0.0007"`` when it was reported by any other source (or is a known
+ zero).
+
+ The same label goes on the attribution line and the CLI's ``-v`` line.
+ """
+ value = valid_usd(cost_usd)
+ if value is None:
+ return "cost unknown"
+ text = format_usd(value)
+ if estimated:
+ return f"~{text} (est.)"
+ return f"{text} (API-equivalent)" if api_equivalent else text
+
+
+def _count(value: object) -> int:
+ if isinstance(value, bool) or not isinstance(value, int):
+ return 0
+ return max(value, 0)
+
+
+def _load_json(text: str, source: str, where: str) -> Any:
+ try:
+ return json.loads(text, object_pairs_hook=_reject_duplicate_keys)
+ except json.JSONDecodeError as e:
+ raise ConfigError(f"{source}: not valid JSON ({e.msg} at line {e.lineno} column {e.colno}){where}") from e
+ except _DuplicateKeyError as e:
+ raise ConfigError(
+ f"{source}: duplicate key {e.key!r}{where}; each model name, and each field of an entry, may appear once"
+ ) from e
+
+
+def _load_json_file(text: str, source: str) -> Any:
+ path = Path(text).expanduser()
+ try:
+ data = path.read_bytes()
+ except OSError as e:
+ raise ConfigError(
+ f"{source}: cannot read price table file '{path}': {e.strerror or e} (inline JSON must start with '{{')"
+ ) from e
+ try:
+ body = data.decode("utf-8-sig")
+ except UnicodeDecodeError as e:
+ raise ConfigError(f"{source}: price table file '{path}' is not valid UTF-8") from e
+ return _load_json(body, source, f" in '{path}'")
+
+
+def _reject_duplicate_keys(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
+ result: dict[str, Any] = {}
+ for key, value in pairs:
+ if key in result:
+ raise _DuplicateKeyError(key)
+ result[key] = value
+ return result
+
+
+def _validate_table(data: object, source: str) -> dict[str, ModelPrice]:
+ if not isinstance(data, Mapping):
+ raise ConfigError(f"{source}: must be {_TABLE_SHAPE}, got {_json_kind(data)}")
+ table: dict[str, ModelPrice] = {}
+ for name, entry in data.items():
+ if not isinstance(name, str) or not name.strip():
+ raise ConfigError(f"{source}: every model name must be a non-empty string, got {name!r}")
+ model = name.strip()
+ if model in table:
+ raise ConfigError(f"{source}: duplicate model name {model!r}")
+ table[model] = _validate_entry(model, entry, source)
+ return table
+
+
+def _validate_entry(model: str, entry: object, source: str) -> ModelPrice:
+ if not isinstance(entry, Mapping):
+ raise ConfigError(f"{source}: the entry for {model!r} must be an object like {_ENTRY_EXAMPLE}, "
+ f"got {_json_kind(entry)}")
+ unknown = sorted(repr(key) for key in entry if key not in _PRICE_FIELDS)
+ if unknown:
+ raise ConfigError(f"{source}: the entry for {model!r} has unknown key(s) {', '.join(unknown)}; "
+ "only 'input' and 'output' are allowed")
+ missing = [f"'{field}'" for field in _PRICE_FIELDS if field not in entry]
+ if missing:
+ raise ConfigError(f"{source}: the entry for {model!r} is missing {' and '.join(missing)}")
+ return ModelPrice(*(_price(model, field, entry[field], source) for field in _PRICE_FIELDS))
+
+
+def _price(model: str, field: str, value: object, source: str) -> float:
+ problem = f"{source}: {model!r} {field} must be a finite number >= 0 (USD per million tokens), got "
+ if isinstance(value, bool) or not isinstance(value, int | float):
+ raise ConfigError(problem + _json_kind(value))
+ try:
+ number = float(value)
+ except OverflowError as e:
+ raise ConfigError(problem + "a number too large to represent") from e
+ if not math.isfinite(number) or number < 0:
+ raise ConfigError(problem + repr(value))
+ return number
+
+
+def _json_kind(value: object) -> str:
+ if value is None:
+ return "null"
+ if isinstance(value, bool):
+ return f"the boolean {str(value).lower()}"
+ if isinstance(value, int | float):
+ return f"the number {value!r}"
+ if isinstance(value, str):
+ return f"the string {value!r}"
+ if isinstance(value, Mapping):
+ return "an object"
+ if isinstance(value, list | tuple):
+ return "an array"
+ return type(value).__name__
diff --git a/src/prxref/forges/azure_devops.py b/src/prxref/forges/azure_devops.py
new file mode 100644
index 0000000..626dd09
--- /dev/null
+++ b/src/prxref/forges/azure_devops.py
@@ -0,0 +1,883 @@
+"""Azure DevOps Services / Server REST API forge implementation.
+
+Covers Azure DevOps Services (``dev.azure.com`` and the legacy
+``*.visualstudio.com`` hosts) and Azure DevOps Server (on-prem, any host, URL
+carrying the collection and the project). Every request speaks REST
+``api-version=7.1`` against a project-scoped route, because the org-level
+routes refuse anonymous callers even on public projects.
+
+Azure DevOps has no unified-diff endpoint, so ``get_diff`` rebuilds one: the
+Diffs API (``diffs/commits`` with ``diffCommonCommit=true``) lists the changed
+files from the merge base to the source head, each side's content comes from
+the blobs API by object id, and ``difflib`` renders git-apply-faithful hunks,
+``\\ No newline at end of file`` included. The same path serves
+``get_compare_diff`` for a pinned commit range.
+"""
+from __future__ import annotations
+
+import base64
+import concurrent.futures
+import difflib
+import functools
+import logging
+import os
+import re
+from collections.abc import Sequence
+from dataclasses import dataclass
+from typing import Any, NamedTuple
+from urllib.parse import quote, unquote, urlsplit
+
+import requests
+from requests.adapters import HTTPAdapter
+
+from prxref.forges.base import (
+ ATTRIBUTION_MARKER,
+ SUMMARY_MARKER,
+ FeedReadError,
+ InlineComment,
+ PRData,
+ PRRef,
+ Thread,
+ with_summary_marker,
+)
+from prxref.retry_logging import LoggingRetry
+
+logger = logging.getLogger(__name__)
+
+_API_VERSION = "7.1"
+_REQUEST_TIMEOUT = (10.0, 30.0)
+# A rejection body is operator-only diagnostics, never posted to the forge, but
+# it is still bounded so a long validation error cannot bury the log line.
+_ERROR_DETAIL_CHARS = 400
+# get_file_content is best-effort context, not the review itself: a body past
+# this size (or one that looks binary) is worth skipping rather than shipping
+# hundreds of KB into a worker prompt.
+_MAX_FILE_CONTENT_BYTES = 512 * 1024
+# The Diffs API pages with $top/$skip, and only the last page carries
+# allChangesIncluded. Running out of pages RAISES: an incomplete file list is a
+# wrong review, not a smaller one.
+_DIFF_PAGE_SIZE = 1000
+_MAX_PAGES = 50
+# Content budget for the rebuilt diff. Blob sizes are unknown until download,
+# so the file and byte budgets are checked between fetch batches; a file past
+# any of them keeps its header (the reviewer still sees it changed) and loses
+# its hunks, with one warning naming the count.
+_MAX_BLOB_BYTES = 512 * 1024
+_MAX_CONTENT_FILES = 300
+_MAX_TOTAL_BYTES = 16 * 1024 * 1024
+_FETCH_WORKERS = 8
+# git's own heuristic: a NUL in the first 8000 bytes of either side is binary.
+_BINARY_SNIFF_BYTES = 8000
+# Known-binary extensions are never downloaded at all.
+_BINARY_EXTENSIONS = frozenset({
+ ".png", ".jpg", ".jpeg", ".gif", ".ico", ".bmp", ".webp", ".pdf", ".zip", ".gz", ".7z", ".jar",
+ ".dll", ".exe", ".so", ".dylib", ".pdb", ".mov", ".mp4", ".mp3", ".wav", ".woff", ".woff2",
+ ".ttf", ".otf", ".eot", ".bacpac", ".dacpac", ".pfx", ".snk", ".nupkg",
+})
+_TOKEN_ENV = "PRXREF_AZURE_DEVOPS_TOKEN"
+_PIPELINE_TOKEN_ENV = "SYSTEM_ACCESSTOKEN"
+# Inline findings open as "active", like an unresolved review comment on every
+# other forge. The summary is posted "closed": it is the equivalent of a
+# GitHub issue comment, which nothing can require to be resolved, so it must
+# never hold a PR behind a "Check for comment resolution" policy.
+_INLINE_THREAD_STATUS = "active"
+_SUMMARY_THREAD_STATUS = "closed"
+_COMMENT_TYPE_TEXT = 1
+_RESOLVED_STATUSES = frozenset({"fixed", "wontfix", "closed", "bydesign"})
+
+_ADO_URL_RE = re.compile(
+ r"^(?Phttps?)://(?P[^/?#]+)"
+ r"(?P(?:/[^/?#]+)*?)"
+ r"/_git/(?P[^/?#]+)"
+ r"/pullrequest/(?P\d+)(?:[/?#].*)?$",
+ re.IGNORECASE,
+)
+
+
+class _Location(NamedTuple):
+ """Where a pull request lives: everything a request URL is rebuilt from."""
+
+ scheme: str
+ host: str
+ collection: str
+ project: str
+ repo: str
+ number: int
+
+
+def _locate(url: str) -> _Location | None:
+ """Split an Azure DevOps pull request URL into its location, or None.
+
+ The path segments in front of ``/_git/{repo}`` mean different things per
+ host. On ``dev.azure.com`` the first is the organization and an optional
+ second is the project. On ``*.visualstudio.com`` the organization is the
+ subdomain, and an optional leading ``DefaultCollection`` precedes an
+ optional project. Anywhere else is Azure DevOps Server, where the last
+ segment is the project and everything before it is the collection path;
+ fewer than two segments there is ambiguous and refused. A missing project
+ is the short form, whose project is named like the repository.
+ """
+ match = _ADO_URL_RE.match(url.strip())
+ if not match:
+ return None
+ scheme, host = match.group("scheme").lower(), match.group("host")
+ try:
+ hostname = (urlsplit(f"{scheme}://{host}").hostname or "").lower()
+ except ValueError:
+ return None
+ segments = [s for s in (match.group("prefix") or "").split("/") if s]
+ if hostname == "dev.azure.com":
+ if len(segments) not in (1, 2):
+ return None
+ collection, rest = "/" + segments[0], segments[1:]
+ elif hostname.endswith(".visualstudio.com"):
+ if segments and segments[0].lower() == "defaultcollection":
+ collection, rest = "/" + segments[0], segments[1:]
+ else:
+ collection, rest = "", segments
+ if len(rest) > 1:
+ return None
+ else:
+ if len(segments) < 2:
+ return None
+ collection, rest = "/" + "/".join(segments[:-1]), segments[-1:]
+ repo = unquote(match.group("repo"))
+ project = unquote(rest[0]) if rest else repo
+ return _Location(scheme, host, collection, project, repo, int(match.group("number")))
+
+
+def _response_detail(resp: requests.Response) -> str:
+ """Return a bounded, single-line rendering of an error response body."""
+ try:
+ body = resp.text or ""
+ except Exception: # noqa: BLE001 - a body that will not decode is not a failure
+ return ""
+ collapsed = " ".join(body.split())
+ if len(collapsed) > _ERROR_DETAIL_CHARS:
+ return collapsed[:_ERROR_DETAIL_CHARS] + "…"
+ return collapsed or ""
+
+
+def _make_retry_session() -> requests.Session:
+ """Build a requests.Session with bounded retries for transient failures.
+
+ Read verbs only, for the reason spelled out in bitbucket_server.py: a write
+ that commits server-side and then loses its response would be re-sent
+ whole by urllib3, and the PR would carry a duplicate comment.
+ """
+ session = requests.Session()
+ retry = LoggingRetry(
+ total=3,
+ connect=3,
+ read=3,
+ status=3,
+ backoff_factor=1.0,
+ status_forcelist=[429, 500, 502, 503, 504],
+ allowed_methods=frozenset(["GET", "HEAD", "OPTIONS"]),
+ respect_retry_after_header=True,
+ raise_on_status=False,
+ )
+ adapter = HTTPAdapter(max_retries=retry)
+ session.mount("https://", adapter)
+ session.mount("http://", adapter)
+ return session
+
+
+_DEFAULT_SESSION = _make_retry_session()
+
+
+def _strip_heads(ref_name: str) -> str:
+ """Return a branch ref name without its ``refs/heads/`` prefix."""
+ prefix = "refs/heads/"
+ return ref_name[len(prefix):] if ref_name.startswith(prefix) else ref_name
+
+
+def _extension(path: str | None) -> str:
+ """Return the lowercased extension of the last path segment, or ``""``."""
+ lowered = (path or "").lower()
+ dot = lowered.rfind(".")
+ return lowered[dot:] if dot > lowered.rfind("/") else ""
+
+
+@dataclass
+class _Change:
+ """One blob change from the Diffs API, classified for rendering."""
+
+ status: str
+ old: str | None
+ new: str | None
+ old_oid: str | None
+ new_oid: str | None
+ pure_rename: bool = False
+ binary: bool = False
+ header_only: bool = False
+
+ @property
+ def label(self) -> str:
+ """The path a log line names."""
+ return self.new or self.old or ""
+
+
+def _classify(changes: Sequence[dict]) -> list[_Change]:
+ """Turn raw Diffs API entries into the blob changes a diff renders.
+
+ Tree entries (folders) and commit entries (submodules) drop out. A rename
+ arrives as two entries, the ``rename`` itself plus a ``delete,
+ sourceRename`` for the old path; the second half is dropped, or every
+ rename would also show up as a deletion. The rename's source is in
+ ``sourceServerItem``.
+ """
+ out: list[_Change] = []
+ for change in changes:
+ item = change.get("item") or {}
+ kind = item.get("gitObjectType")
+ if kind != "blob":
+ if kind == "commit":
+ logger.debug("skipping submodule entry %s", item.get("path"))
+ continue
+ tokens = {t.strip().lower() for t in (change.get("changeType") or "").split(",")}
+ if "sourcerename" in tokens:
+ continue
+ path = (item.get("path") or "").lstrip("/")
+ if tokens & {"add", "undelete", "branch"}:
+ entry = _Change("added", None, path, None, item.get("objectId"))
+ elif "delete" in tokens:
+ entry = _Change("removed", path, None, item.get("originalObjectId"), None)
+ elif "rename" in tokens:
+ source = (change.get("sourceServerItem") or change.get("originalPath") or "").lstrip("/")
+ entry = _Change("renamed", source, path, item.get("originalObjectId"), item.get("objectId"))
+ else:
+ entry = _Change("modified", path, path, item.get("originalObjectId"), item.get("objectId"))
+ if any(ch in (entry.old or "") + (entry.new or "") for ch in "\t\n"):
+ logger.warning("skipping %r: a path with a tab or newline cannot be expressed in a diff", entry.label)
+ continue
+ entry.pure_rename = bool(
+ entry.status == "renamed"
+ and entry.old_oid
+ and entry.new_oid
+ and entry.old_oid.lower() == entry.new_oid.lower()
+ )
+ entry.binary = _extension(entry.new or entry.old) in _BINARY_EXTENSIONS
+ out.append(entry)
+ return out
+
+
+def _render_hunks(old: bytes, new: bytes) -> list[str]:
+ """Render the ``@@`` hunks between two text blobs, git-apply-faithfully.
+
+ Both sides split with ``splitlines(keepends=True)``: the diff parser splits
+ the whole diff with ``str.splitlines()``, so splitting content any other
+ way would leave a boundary (``\\x0c``, ``\\x85``, …) inside an emitted line
+ for the parser to split a second time, and the hunk bodies would stop
+ matching their ``@@`` counts. Keeping the terminators also makes a
+ trailing-newline-only change a hunk, as it is in git, and a last line
+ without a terminator gets git's ``\\ No newline at end of file`` marker.
+ """
+ old_lines = old.decode("utf-8", errors="replace").splitlines(keepends=True)
+ new_lines = new.decode("utf-8", errors="replace").splitlines(keepends=True)
+ out: list[str] = []
+ for raw in list(difflib.unified_diff(old_lines, new_lines, n=3, lineterm=""))[2:]:
+ if raw.startswith("@@"):
+ out.append(raw)
+ continue
+ body = raw[1:]
+ parts = body.splitlines()
+ content = parts[0] if parts else ""
+ out.append(raw[0] + content)
+ if content == body:
+ out.append("\\ No newline at end of file")
+ return out
+
+
+class ForgeImpl:
+ """Azure DevOps Services / Server Forge adapter."""
+
+ name: str = "azure-devops"
+
+ def __init__(self, session: requests.Session | None = None) -> None:
+ """Initialize with an optional custom requests Session."""
+ self._session = session if session is not None else _DEFAULT_SESSION
+
+ @staticmethod
+ def parse_pr_url(url: str) -> PRRef | None:
+ """Return a PRRef if this forge recognizes the URL, else None.
+
+ Accepts ``https://dev.azure.com/{org}/{project}/_git/{repo}/pullrequest/{n}``,
+ the same on ``{org}.visualstudio.com`` (with or without
+ ``DefaultCollection``), the short form without a project (the project
+ is then named like the repository), and Azure DevOps Server URLs on
+ any host whose path carries the collection and the project. ``owner``
+ holds the project and ``repo`` the repository, both percent-decoded.
+ The collection lives only in ``url``, which is normalized: the query
+ and fragment are dropped, the short form gains its explicit project,
+ and the scheme is lowercased but kept, because an on-prem Server can
+ serve plain HTTP.
+ """
+ loc = _locate(url)
+ if loc is None:
+ return None
+ normalized = (
+ f"{loc.scheme}://{loc.host}{loc.collection}/{quote(loc.project, safe='')}"
+ f"/_git/{quote(loc.repo, safe='')}/pullrequest/{loc.number}"
+ )
+ return PRRef(
+ forge="azure-devops",
+ host=loc.host,
+ owner=loc.project,
+ repo=loc.repo,
+ number=loc.number,
+ url=normalized,
+ )
+
+ def _api_base(self, ref: PRRef) -> str:
+ """Return the project-scoped repository API root, rebuilt from ``ref.url``."""
+ loc = _locate(ref.url)
+ if loc is None:
+ raise ValueError(f"not an Azure DevOps pull request URL: {ref.url}")
+ return (
+ f"{loc.scheme}://{loc.host}{loc.collection}/{quote(loc.project, safe='')}"
+ f"/_apis/git/repositories/{quote(loc.repo, safe='')}"
+ )
+
+ def _pr_api(self, ref: PRRef, suffix: str = "") -> str:
+ """Return the pull request's API URL, plus ``suffix``."""
+ return f"{self._api_base(ref)}/pullrequests/{ref.number}{suffix}"
+
+ def _headers(self, accept: str = "application/json") -> dict[str, str]:
+ """Build request headers, reading credentials from the environment at call time.
+
+ ``PRXREF_AZURE_DEVOPS_TOKEN`` (a PAT, sent as Basic auth with an empty
+ user name) wins; inside Azure Pipelines ``SYSTEM_ACCESSTOKEN`` (a
+ bearer token) is the fallback; with neither, requests go out
+ anonymously, which reads public projects. ``X-TFS-FedAuthRedirect:
+ Suppress`` is always sent, so an unauthenticated call gets a clean 401
+ instead of a redirect to a sign-in page.
+ """
+ headers = {"Accept": accept, "X-TFS-FedAuthRedirect": "Suppress"}
+ pat = os.environ.get(_TOKEN_ENV, "").strip()
+ pipeline = os.environ.get(_PIPELINE_TOKEN_ENV, "").strip()
+ if pat:
+ headers["Authorization"] = "Basic " + base64.b64encode(f":{pat}".encode()).decode()
+ elif pipeline:
+ headers["Authorization"] = f"Bearer {pipeline}"
+ return headers
+
+ def _get(
+ self,
+ url: str,
+ params: dict[str, Any] | None = None,
+ *,
+ accept: str = "application/json",
+ stream: bool = False,
+ ) -> requests.Response:
+ """GET ``url`` with the API version, the auth headers and the timeout."""
+ return self._session.get(
+ url,
+ params={"api-version": _API_VERSION, **(params or {})},
+ headers=self._headers(accept),
+ timeout=_REQUEST_TIMEOUT,
+ stream=stream,
+ )
+
+ @staticmethod
+ def _json(resp: requests.Response, what: str) -> dict:
+ """Return a response's JSON object, refusing anything that is not one.
+
+ A server that ignores the Suppress header answers an unauthenticated
+ call with a 2xx sign-in page (203 text/html). Parsing that as data
+ would read as success with garbage in it, so it raises instead, with
+ a hint to set the token.
+ """
+ resp.raise_for_status()
+ content_type = (resp.headers.get("Content-Type") or "").lower()
+ if resp.status_code == 203 or "json" not in content_type:
+ raise ValueError(
+ f"Azure DevOps returned a non-JSON {what} (HTTP {resp.status_code}); the request was "
+ f"probably not authenticated — set {_TOKEN_ENV}"
+ )
+ data = resp.json()
+ if not isinstance(data, dict):
+ raise ValueError(f"Azure DevOps returned a {type(data).__name__} {what}, not an object")
+ return data
+
+ def _pr_json(self, ref: PRRef) -> dict:
+ """Fetch the pull request's JSON."""
+ return self._json(self._get(self._pr_api(ref)), "pull request")
+
+ def get_pr(self, ref: PRRef) -> PRData:
+ """Fetch normalized PR metadata.
+
+ The author is the display name first: ``uniqueName`` is null for an
+ anonymous caller and an e-mail address for an authenticated one.
+ """
+ pr = self._pr_json(ref)
+ who = pr.get("createdBy") or {}
+ return PRData(
+ title=pr.get("title") or "",
+ description=pr.get("description") or "",
+ author=who.get("displayName") or who.get("uniqueName") or "",
+ source_branch=_strip_heads(pr.get("sourceRefName") or ""),
+ target_branch=_strip_heads(pr.get("targetRefName") or ""),
+ source_sha=(pr.get("lastMergeSourceCommit") or {}).get("commitId") or "",
+ target_sha=(pr.get("lastMergeTargetCommit") or {}).get("commitId") or "",
+ raw=pr,
+ )
+
+ def get_diff(self, ref: PRRef) -> str:
+ """Fetch the unified diff of the PR (all files), rebuilt from the Diffs API.
+
+ The range is the PR's own: the merge base of the last merged target
+ commit (or, lacking one, the target branch) up to the source head.
+ An empty result raises, as on every other forge.
+ """
+ pr = self._pr_json(ref)
+ head = (pr.get("lastMergeSourceCommit") or {}).get("commitId")
+ if not head:
+ raise ValueError(f"Azure DevOps PR {ref.number} has no source commit")
+ base = (pr.get("lastMergeTargetCommit") or {}).get("commitId")
+ if base:
+ text = self._diff_between(ref, base, "commit", head)
+ else:
+ text = self._diff_between(ref, _strip_heads(pr.get("targetRefName") or ""), "branch", head)
+ if not text:
+ raise ValueError(f"empty diff for Azure DevOps PR {ref.number}")
+ return text
+
+ def get_compare_diff(self, ref: PRRef, *, base_sha: str, head_sha: str) -> str:
+ """Return the unified diff of ``head_sha`` against its merge-base with ``base_sha``.
+
+ The same Diffs API reconstruction as ``get_diff``, with both ends given
+ as commits; ``diffCommonCommit=true`` supplies the three-dot semantics.
+ Raises on transport or HTTP failure; returns ``""`` for an empty range.
+ """
+ return self._diff_between(ref, base_sha, "commit", head_sha)
+
+ def _list_changes(self, ref: PRRef, base: str, base_type: str, head: str) -> list[dict]:
+ """Page through the Diffs API change list from the merge base to ``head``."""
+ changes: list[dict] = []
+ skip = 0
+ for _ in range(_MAX_PAGES):
+ page_json = self._json(
+ self._get(
+ f"{self._api_base(ref)}/diffs/commits",
+ {
+ "baseVersion": base,
+ "baseVersionType": base_type,
+ "targetVersion": head,
+ "targetVersionType": "commit",
+ "diffCommonCommit": "true",
+ "$top": _DIFF_PAGE_SIZE,
+ "$skip": skip,
+ },
+ ),
+ "diff listing",
+ )
+ page = page_json.get("changes") or []
+ changes.extend(page)
+ if page_json.get("allChangesIncluded") or not page:
+ return changes
+ skip += len(page)
+ raise ValueError(
+ f"Azure DevOps diff listing exceeded {_MAX_PAGES} pages; refusing an incomplete diff"
+ )
+
+ def _fetch_blob(self, ref: PRRef, oid: str) -> bytes | None:
+ """Download one blob by object id; ``None`` when it is gone or over the cap.
+
+ Any other failure raises: a diff where a 401 or a 5xx silently emptied
+ every file would be a review of nothing that still says "Approved".
+ """
+ resp = self._get(
+ f"{self._api_base(ref)}/blobs/{oid}",
+ {"$format": "octetstream"},
+ accept="application/octet-stream",
+ stream=True,
+ )
+ try:
+ if resp.status_code in (404, 410):
+ logger.debug("Azure DevOps blob %s not found (HTTP %s)", oid, resp.status_code)
+ return None
+ resp.raise_for_status()
+ buf = bytearray()
+ for chunk in resp.iter_content(64 * 1024):
+ buf += chunk
+ if len(buf) > _MAX_BLOB_BYTES:
+ return None
+ return bytes(buf)
+ finally:
+ resp.close()
+
+ def _fetch_contents(self, ref: PRRef, entries: list[_Change]) -> dict[str, bytes | None]:
+ """Fetch the blobs the diff needs, within the content budget.
+
+ Binary-by-extension files and pure renames need no content. The rest
+ are fetched in API order, one batch of ``_FETCH_WORKERS`` at a time,
+ and the budget is checked between batches; every file after it runs
+ out becomes header-only.
+ """
+ pending: list[_Change] = []
+ for entry in entries:
+ if entry.binary or entry.pure_rename:
+ continue
+ if (entry.status != "removed" and not entry.new_oid) or (
+ entry.status != "added" and not entry.old_oid
+ ):
+ logger.warning("Azure DevOps change for %s lacks an object id; header only", entry.label)
+ entry.header_only = True
+ continue
+ pending.append(entry)
+
+ blobs: dict[str, bytes | None] = {}
+ fetched = total = skipped = 0
+ fetch = functools.partial(self._fetch_blob, ref)
+ with concurrent.futures.ThreadPoolExecutor(max_workers=_FETCH_WORKERS) as pool:
+ for start in range(0, len(pending), _FETCH_WORKERS):
+ batch = pending[start:start + _FETCH_WORKERS]
+ if fetched >= _MAX_CONTENT_FILES or total > _MAX_TOTAL_BYTES:
+ for entry in batch:
+ entry.header_only = True
+ skipped += len(batch)
+ continue
+ oids = list(dict.fromkeys(
+ oid for entry in batch for oid in (entry.old_oid, entry.new_oid) if oid and oid not in blobs
+ ))
+ for oid, data in zip(oids, pool.map(fetch, oids), strict=True):
+ blobs[oid] = data
+ total += len(data or b"")
+ fetched += len(batch)
+ if skipped:
+ logger.warning("Azure DevOps diff: %d file(s) past the content budget are header-only", skipped)
+ return blobs
+
+ def _diff_between(self, ref: PRRef, base: str, base_type: str, head: str) -> str:
+ """Rebuild the unified diff from ``base``'s merge base with ``head`` up to ``head``.
+
+ One block per blob change, in API order. Paths are written unquoted,
+ and the ``---``/``+++`` lines are kept even for binary files (where git
+ omits them): the parser takes exact paths from them, which matters for
+ a path containing `` b/`` that the ``diff --git`` line alone would
+ split wrongly. Mode lines are always ``100644``; the Diffs API exposes
+ no file modes. Returns ``""`` when no blob changed.
+ """
+ entries = _classify(self._list_changes(ref, base, base_type, head))
+ blobs = self._fetch_contents(ref, entries)
+ out: list[str] = []
+ for entry in entries:
+ a_path = f"a/{entry.old or entry.new}"
+ b_path = f"b/{entry.new or entry.old}"
+ block = [f"diff --git {a_path} {b_path}"]
+ if entry.status == "added":
+ block.append("new file mode 100644")
+ elif entry.status == "removed":
+ block.append("deleted file mode 100644")
+ elif entry.status == "renamed":
+ if entry.pure_rename:
+ block.append("similarity index 100%")
+ block += [f"rename from {entry.old}", f"rename to {entry.new}"]
+ if entry.pure_rename:
+ out += block
+ continue
+ old_label = a_path if entry.status != "added" else "/dev/null"
+ new_label = b_path if entry.status != "removed" else "/dev/null"
+ block += [f"--- {old_label}", f"+++ {new_label}"]
+ if entry.binary:
+ out += block + [f"Binary files {old_label} and {new_label} differ"]
+ continue
+ if entry.header_only:
+ out += block
+ continue
+ old = blobs.get(entry.old_oid, b"") if entry.old_oid else b""
+ new = blobs.get(entry.new_oid, b"") if entry.new_oid else b""
+ if old is None or new is None:
+ logger.warning(
+ "Azure DevOps content for %s is over %d bytes or missing; header only",
+ entry.label, _MAX_BLOB_BYTES,
+ )
+ out += block
+ continue
+ if b"\x00" in old[:_BINARY_SNIFF_BYTES] or b"\x00" in new[:_BINARY_SNIFF_BYTES]:
+ out += block + [f"Binary files {old_label} and {new_label} differ"]
+ continue
+ out += block + _render_hunks(old, new)
+ return ("\n".join(out) + "\n") if out else ""
+
+ def get_file_content(self, ref: PRRef, path: str, *, sha: str) -> str | None:
+ """Return the text of ``path`` at commit ``sha``, best-effort.
+
+ Reads the ``items`` endpoint as raw bytes, so a UTF-8 BOM survives as
+ it does in the diff. Returns ``None`` on any failure, on a body over
+ 512 KiB, and on one that looks binary. Never raises.
+ """
+ if not sha:
+ return None
+ try:
+ resp = self._get(
+ f"{self._api_base(ref)}/items",
+ {
+ "path": "/" + path.lstrip("/"),
+ "versionDescriptor.version": sha,
+ "versionDescriptor.versionType": "commit",
+ "download": "true",
+ },
+ accept="application/octet-stream",
+ stream=True,
+ )
+ try:
+ if not resp.ok:
+ logger.debug("get_file_content got HTTP %s for %s@%s", resp.status_code, path, sha)
+ return None
+ buf = bytearray()
+ for chunk in resp.iter_content(64 * 1024):
+ buf += chunk
+ if len(buf) > _MAX_FILE_CONTENT_BYTES:
+ logger.debug("get_file_content body over 512 KiB for %s@%s", path, sha)
+ return None
+ finally:
+ resp.close()
+ except (requests.RequestException, ValueError) as e:
+ logger.debug("get_file_content failed for %s@%s: %s", path, sha, e)
+ return None
+ if b"\x00" in buf[:_BINARY_SNIFF_BYTES]:
+ logger.debug("get_file_content body looked binary for %s@%s", path, sha)
+ return None
+ return bytes(buf).decode("utf-8", errors="replace")
+
+ def _read_threads(self, ref: PRRef) -> list[dict]:
+ """Read every thread on the PR (one response; the API does not page them).
+
+ Raises ``FeedReadError`` on anything short of a complete list, so
+ ``post_summary`` can never mistake a failed read for "no summary".
+ """
+ try:
+ resp = self._get(self._pr_api(ref, "/threads"))
+ except requests.RequestException as e:
+ raise FeedReadError(f"Azure DevOps threads for PR {ref.number} could not be read: {e}") from e
+ if not resp.ok:
+ raise FeedReadError(
+ f"Azure DevOps threads for PR {ref.number} returned HTTP {resp.status_code}: "
+ f"{_response_detail(resp)}"
+ )
+ try:
+ value = self._json(resp, "thread list").get("value")
+ except ValueError as e:
+ raise FeedReadError(str(e)) from e
+ if not isinstance(value, list):
+ raise FeedReadError(f"Azure DevOps thread list for PR {ref.number} has no 'value' array")
+ return [t for t in value if isinstance(t, dict)]
+
+ @staticmethod
+ def _usable(thread: dict) -> bool:
+ """True for a live thread with comments that is not a system notice."""
+ comments = thread.get("comments") or []
+ return (
+ bool(comments)
+ and not thread.get("isDeleted")
+ and isinstance(comments[0], dict)
+ and comments[0].get("commentType") != "system"
+ )
+
+ @staticmethod
+ def _root(thread: dict) -> dict:
+ """Return the thread's first comment that is not deleted, or ``{}``."""
+ return next(
+ (c for c in thread.get("comments") or [] if isinstance(c, dict) and not c.get("isDeleted")),
+ {},
+ )
+
+ def list_threads(self, ref: PRRef) -> list[Thread]:
+ """List existing discussion threads on the PR.
+
+ System notices (votes, pushes, status changes) are skipped. A thread
+ counts as resolved when its status is fixed, won't-fix, closed or
+ by-design. A feed that cannot be read is logged and yields what was
+ read, because these threads only feed best-effort dedup.
+ """
+ threads: list[Thread] = []
+ try:
+ for thread in self._read_threads(ref):
+ if not self._usable(thread):
+ continue
+ context = thread.get("threadContext") or {}
+ root = self._root(thread)
+ who = root.get("author") or {}
+ threads.append(
+ Thread(
+ path=(context.get("filePath") or "").lstrip("/") or None,
+ line=(context.get("rightFileStart") or {}).get("line"),
+ resolved=str(thread.get("status") or "").lower() in _RESOLVED_STATUSES,
+ author=who.get("displayName") or who.get("uniqueName") or "",
+ body_snippet=(root.get("content") or "")[:200],
+ )
+ )
+ except FeedReadError as e:
+ logger.warning(
+ "thread read was incomplete for %s/%s#%s; thread dedup is working from the %d "
+ "threads that were read: %s",
+ ref.owner, ref.repo, ref.number, len(threads), e,
+ )
+ return threads
+
+ def post_summary(self, ref: PRRef, body: str) -> None:
+ """Post (or update) the top-level review summary comment.
+
+ The summary is a PR-level thread (no ``threadContext``) whose root
+ comment carries ``SUMMARY_MARKER``; an existing one has its root
+ comment PATCHed, and otherwise a new thread is created, closed. A
+ ``FeedReadError`` from the lookup propagates: a failed lookup must not
+ post a second summary.
+ """
+ body = with_summary_marker(body)
+ for thread in self._read_threads(ref):
+ if not self._usable(thread) or thread.get("threadContext"):
+ continue
+ root = self._root(thread)
+ if SUMMARY_MARKER not in (root.get("content") or ""):
+ continue
+ if thread.get("id") is None or root.get("id") is None:
+ continue
+ resp = self._session.patch(
+ self._pr_api(ref, f"/threads/{thread['id']}/comments/{root['id']}"),
+ params={"api-version": _API_VERSION},
+ json={"content": body},
+ headers=self._headers(),
+ timeout=_REQUEST_TIMEOUT,
+ )
+ resp.raise_for_status()
+ return
+ resp = self._session.post(
+ self._pr_api(ref, "/threads"),
+ params={"api-version": _API_VERSION},
+ json={
+ "comments": [{"parentCommentId": 0, "content": body, "commentType": _COMMENT_TYPE_TEXT}],
+ "status": _SUMMARY_THREAD_STATUS,
+ },
+ headers=self._headers(),
+ timeout=_REQUEST_TIMEOUT,
+ )
+ resp.raise_for_status()
+
+ def _change_tracking(self, ref: PRRef) -> dict[str, tuple[int, int]]:
+ """Map each changed path to its ``(changeTrackingId, iteration)`` in the latest iteration.
+
+ Best-effort enrichment for inline threads: iterations need credentials
+ even on a public project, and a thread posts without this context, so
+ any failure logs at debug and returns ``{}``.
+ """
+ try:
+ iterations = self._json(self._get(self._pr_api(ref, "/iterations")), "iteration list").get("value")
+ latest = max(int(i["id"]) for i in iterations or [])
+ tracking: dict[str, tuple[int, int]] = {}
+ skip = 0
+ for _ in range(_MAX_PAGES):
+ page = self._json(
+ self._get(self._pr_api(ref, f"/iterations/{latest}/changes"), {"$top": 2000, "$skip": skip}),
+ "iteration changes",
+ )
+ for entry in page.get("changeEntries") or []:
+ key = ((entry.get("item") or {}).get("path") or entry.get("originalPath") or "").lstrip("/")
+ if key and entry.get("changeTrackingId") is not None:
+ tracking[key] = (int(entry["changeTrackingId"]), latest)
+ next_skip = page.get("nextSkip")
+ if not next_skip:
+ break
+ skip = int(next_skip)
+ return tracking
+ except (requests.RequestException, ValueError, KeyError, TypeError, AttributeError) as e:
+ logger.debug("Azure DevOps iteration context unavailable for PR %s: %s", ref.number, e)
+ return {}
+
+ def post_inline_comments(self, ref: PRRef, comments: Sequence[InlineComment]) -> int:
+ """Post inline comments; returns the number actually posted.
+
+ Each finding becomes its own thread anchored to one line of the new
+ file, opened active. The latest iteration's ``changeTrackingId`` is
+ attached when it can be read. A 4xx (a line outside the diff, say)
+ is logged and skipped; a 5xx or transport failure is skipped too.
+ """
+ if not comments:
+ return 0
+ tracking = self._change_tracking(ref)
+ posted = 0
+ for comment in comments:
+ path = comment.path.lstrip("/")
+ payload: dict[str, Any] = {
+ "comments": [{"parentCommentId": 0, "content": comment.body, "commentType": _COMMENT_TYPE_TEXT}],
+ "status": _INLINE_THREAD_STATUS,
+ "threadContext": {
+ "filePath": "/" + path,
+ "rightFileStart": {"line": comment.line, "offset": 1},
+ "rightFileEnd": {"line": comment.line, "offset": 1},
+ },
+ }
+ if path in tracking:
+ tracking_id, iteration = tracking[path]
+ payload["pullRequestThreadContext"] = {
+ "changeTrackingId": tracking_id,
+ "iterationContext": {
+ "firstComparingIteration": iteration,
+ "secondComparingIteration": iteration,
+ },
+ }
+ try:
+ resp = self._session.post(
+ self._pr_api(ref, "/threads"),
+ params={"api-version": _API_VERSION},
+ json=payload,
+ headers=self._headers(),
+ timeout=_REQUEST_TIMEOUT,
+ )
+ if 200 <= resp.status_code < 300:
+ posted += 1
+ elif 400 <= resp.status_code < 500:
+ logger.warning(
+ "inline comment on %s:%s rejected (HTTP %s): %s",
+ comment.path, comment.line, resp.status_code, _response_detail(resp),
+ )
+ else:
+ resp.raise_for_status()
+ except requests.RequestException as e:
+ logger.warning("inline comment on %s:%s failed: %s", comment.path, comment.line, e)
+ return posted
+
+ def prune_inline_comments(self, ref: PRRef) -> int:
+ """Delete prxref-attributed inline comments; returns the count removed.
+
+ Only the root comment of a file-anchored thread whose body carries the
+ attribution marker is deleted, so the summary (a PR-level thread) and
+ every human comment are left alone; a human reply keeps its thread.
+ A delete the token may not perform (403 on another identity's comment)
+ is logged and skipped, and an unreadable feed ends the pass:
+ best-effort, because a cleanup must never abort the review.
+ """
+ try:
+ threads = self._read_threads(ref)
+ except FeedReadError as e:
+ logger.warning("prune of inline comments on %s/%s#%s skipped: %s", ref.owner, ref.repo, ref.number, e)
+ return 0
+ removed = 0
+ for thread in threads:
+ if not (self._usable(thread) and thread.get("threadContext")):
+ continue
+ root = self._root(thread)
+ if ATTRIBUTION_MARKER not in (root.get("content") or ""):
+ continue
+ if thread.get("id") is None or root.get("id") is None:
+ continue
+ try:
+ resp = self._session.delete(
+ self._pr_api(ref, f"/threads/{thread['id']}/comments/{root['id']}"),
+ params={"api-version": _API_VERSION},
+ headers=self._headers(),
+ timeout=_REQUEST_TIMEOUT,
+ )
+ except requests.RequestException as e:
+ logger.warning("could not prune inline thread %s: %s", thread.get("id"), e)
+ continue
+ if 200 <= resp.status_code < 300:
+ removed += 1
+ else:
+ logger.warning(
+ "could not prune inline thread %s on %s/%s#%s (HTTP %s): %s",
+ thread.get("id"), ref.owner, ref.repo, ref.number, resp.status_code, _response_detail(resp),
+ )
+ return removed
diff --git a/src/prxref/forges/base.py b/src/prxref/forges/base.py
index e2663cb..34c76e8 100644
--- a/src/prxref/forges/base.py
+++ b/src/prxref/forges/base.py
@@ -1,7 +1,8 @@
-"""Forge contract: one Protocol, four implementations.
+"""Forge contract: one Protocol, five implementations.
The implementations are bitbucket (Cloud), bitbucket_server (Server / Data
-Center), github (Cloud and Enterprise Server) and gitlab (SaaS and self-hosted).
+Center), github (Cloud and Enterprise Server), gitlab (SaaS and self-hosted)
+and azure_devops (Azure DevOps Services and Azure DevOps Server).
Every value that flows through the pipeline is forge-agnostic past this module.
Diff handling is deliberately unified: each forge returns ONE raw unified diff
@@ -19,7 +20,7 @@
class PRRef:
"""A pull/merge request identity, normalized across forges."""
- forge: str # "bitbucket" | "bitbucket-server" | "github" | "gitlab"
+ forge: str # "bitbucket" | "bitbucket-server" | "github" | "gitlab" | "azure-devops" | "local"
host: str # e.g. "bitbucket.org", "github.com", "gitlab.com", or self-hosted host
owner: str # workspace / org / group
repo: str
@@ -89,6 +90,10 @@ class FeedReadError(RuntimeError):
empty list for any exception, so returning the pages that WERE read beats
throwing them away. It logs a warning instead, so the under-read is
visible rather than silent.
+
+ It is not only about comments: GitLab's ``get_diff`` raises it when the
+ paged MR diff listing cannot be read to the end, because the files that
+ did arrive would otherwise be reviewed as if they were the whole MR.
"""
@@ -147,6 +152,18 @@ def get_file_content(self, ref: PRRef, path: str, *, sha: str) -> str | None:
"""
...
+ def get_compare_diff(self, ref: PRRef, *, base_sha: str, head_sha: str) -> str:
+ """Return the unified diff of ``head_sha`` against its merge-base with ``base_sha``.
+
+ Optional: callers resolve it with ``getattr(forge, "get_compare_diff", None)``,
+ so a Forge without it is still valid (replay then refuses pinned SHAs with a
+ configuration error). Three-dot semantics — exactly what the PR's own diff
+ shows when ``base_sha``/``head_sha`` are the PR's target/source commits.
+ Raises on transport or HTTP failure like ``get_diff``; returns ``""`` for an
+ empty range and leaves the judgement to the caller.
+ """
+ ...
+
def detect_forge(url: str) -> PRRef | None:
"""Try each registered forge's URL parser in order.
@@ -159,10 +176,14 @@ def detect_forge(url: str) -> PRRef | None:
any result today. It is kept deliberately anyway: whichever parser is
narrower should be asked first, so that loosening one later degrades into a
shadowed forge rather than a silently mis-routed one.
+
+ Azure DevOps is asked last for the same defensive reason: its URLs carry
+ ``/_git/{repo}/pullrequest/{n}``, which no other pattern accepts, so it
+ cannot shadow or be shadowed by the forges ahead of it.
"""
- from . import bitbucket, bitbucket_server, github, gitlab
+ from . import azure_devops, bitbucket, bitbucket_server, github, gitlab
- for forge in (bitbucket, bitbucket_server, github, gitlab):
+ for forge in (bitbucket, bitbucket_server, github, gitlab, azure_devops):
ref = forge.ForgeImpl.parse_pr_url(url)
if ref is not None:
return ref
diff --git a/src/prxref/forges/bitbucket.py b/src/prxref/forges/bitbucket.py
index 287dcd5..3dcd3fe 100644
--- a/src/prxref/forges/bitbucket.py
+++ b/src/prxref/forges/bitbucket.py
@@ -216,6 +216,30 @@ def get_diff(self, ref: PRRef) -> str:
return diff_text
+ def get_compare_diff(self, ref: PRRef, *, base_sha: str, head_sha: str) -> str:
+ """Return the unified diff of ``head_sha`` against its merge-base with ``base_sha``.
+
+ Bitbucket's diff spec is SOURCE..DEST, the reverse of git's order, so
+ the range is spelled ``{head}..{base}``; swapping the two yields a
+ different diff that still looks valid. ``topic=true`` selects the
+ merge-base (three-dot) diff the PR itself shows. It is the default
+ today, and it is sent explicitly because the result depends on it.
+ An empty range comes back as ``""``. Raises on an HTTP or transport
+ failure.
+ """
+ headers, auth = self._get_auth()
+ headers["Accept"] = "text/plain"
+ url = f"{_API_BASE}/repositories/{ref.owner}/{ref.repo}/diff/{head_sha}..{base_sha}"
+ resp = self._session.get(
+ url,
+ headers=headers,
+ auth=auth,
+ params={"topic": "true"},
+ timeout=_REQUEST_TIMEOUT,
+ )
+ resp.raise_for_status()
+ return resp.text
+
def _iter_comment_pages(self, ref: PRRef) -> Iterator[list[dict]]:
"""Yield the PR's comments one page at a time, following ``next``.
diff --git a/src/prxref/forges/bitbucket_server.py b/src/prxref/forges/bitbucket_server.py
index 2003b53..e64599b 100644
--- a/src/prxref/forges/bitbucket_server.py
+++ b/src/prxref/forges/bitbucket_server.py
@@ -246,15 +246,25 @@ def _scheme(self, ref: PRRef) -> str:
return match.group("scheme").lower()
return "https"
- def _pr_url(self, ref: PRRef, suffix: str = "") -> str:
- """Construct the Data Center API endpoint URL for a given PR."""
+ def _repo_url(self, ref: PRRef, suffix: str = "") -> str:
+ """Construct a Data Center API endpoint URL under the PR's repository.
+
+ The scheme and the deployment context path are the ones recovered from
+ ``ref.url``, and ``ref.owner`` is already the API project key (the
+ ``~slug`` form for a personal repository), so every repository-level
+ and PR-level request is built on this one prefix.
+ """
context = self._context_path(ref)
base = (
f"{self._scheme(ref)}://{ref.host}{context}/rest/api/1.0"
- f"/projects/{ref.owner}/repos/{ref.repo}/pull-requests/{ref.number}"
+ f"/projects/{ref.owner}/repos/{ref.repo}"
)
return f"{base}{suffix}"
+ def _pr_url(self, ref: PRRef, suffix: str = "") -> str:
+ """Construct the Data Center API endpoint URL for a given PR."""
+ return self._repo_url(ref, f"/pull-requests/{ref.number}{suffix}")
+
def get_pr(self, ref: PRRef) -> PRData:
"""Fetch normalized PR metadata."""
headers, auth = self._get_auth()
@@ -307,6 +317,58 @@ def get_diff(self, ref: PRRef) -> str:
return diff_text
+ def get_compare_diff(self, ref: PRRef, *, base_sha: str, head_sha: str) -> str:
+ """Return the unified diff of ``head_sha`` against its merge-base with ``base_sha``.
+
+ The raw diff endpoint (``/diff?since=&until=``) diffs from whatever
+ ``since`` names, with no merge-base step of its own, so passing the base
+ commit straight through would also show everything that landed on the
+ base after the fork. The merge-base is therefore resolved first
+ (``/commits/{head}/merge-base?otherCommitId={base}``) and used as
+ ``since``. When that lookup fails or names no commit, a WARNING is
+ logged and ``base_sha`` itself is used, which is still right whenever
+ it already is the fork point, as a PR's recorded target commit usually
+ is.
+
+ The raw diff is served only as ``text/plain``, at a low quality factor,
+ so the Accept header is explicit. An empty range comes back as ``""``.
+ Raises on an HTTP or transport failure of the diff request.
+ """
+ headers, auth = self._get_auth()
+ since = base_sha
+ try:
+ mb = self._session.get(
+ self._repo_url(ref, f"/commits/{head_sha}/merge-base"),
+ headers=headers,
+ auth=auth,
+ params={"otherCommitId": base_sha},
+ timeout=_REQUEST_TIMEOUT,
+ )
+ mb.raise_for_status()
+ commit = mb.json()
+ merge_base = commit.get("id") if isinstance(commit, dict) else None
+ if merge_base:
+ since = merge_base
+ else:
+ logger.warning(
+ "merge-base(%s, %s) named no commit; diffing from base_sha directly",
+ head_sha, base_sha,
+ )
+ except (requests.RequestException, ValueError) as e:
+ logger.warning(
+ "merge-base(%s, %s) failed (%s); diffing from base_sha directly",
+ head_sha, base_sha, e,
+ )
+ resp = self._session.get(
+ self._repo_url(ref, "/diff"),
+ headers={**headers, "Accept": "text/plain"},
+ auth=auth,
+ params={"since": since, "until": head_sha},
+ timeout=_REQUEST_TIMEOUT,
+ )
+ resp.raise_for_status()
+ return resp.text
+
def _iter_activity_pages(self, ref: PRRef) -> Iterator[list[dict]]:
"""Yield the COMMENTED activity entries one page at a time.
@@ -512,17 +574,13 @@ def get_file_content(self, ref: PRRef, path: str, *, sha: str) -> str | None:
Hits the repository-level ``/raw`` endpoint directly rather than a
pull-request-scoped one — this is a commit-addressed file read, not a
PR resource. ``ref.owner`` already carries the ``~slug`` form for a
- personal repository, so this builds the same project path
+ personal repository, so this builds on the same ``_repo_url`` prefix
``_pr_url`` does. Never raises.
"""
if not sha:
return None
headers, auth = self._get_auth()
- context = self._context_path(ref)
- url = (
- f"{self._scheme(ref)}://{ref.host}{context}/rest/api/1.0"
- f"/projects/{ref.owner}/repos/{ref.repo}/raw/{quote(path, safe='/')}"
- )
+ url = self._repo_url(ref, f"/raw/{quote(path, safe='/')}")
try:
resp = self._session.get(
url, headers=headers, auth=auth, params={"at": sha},
diff --git a/src/prxref/forges/github.py b/src/prxref/forges/github.py
index 4dc8cf3..98f6b5b 100644
--- a/src/prxref/forges/github.py
+++ b/src/prxref/forges/github.py
@@ -30,6 +30,10 @@
r"^https?://([^/]+)/([^/]+)/([^/]+)/pull/(\d+)(?:[/#?].*)?$",
re.IGNORECASE,
)
+# Connect and read deadlines, the same pair every other adapter passes. Without
+# one a stalled connection blocks the call forever, and with it the review and
+# the webhook worker running it.
+_REQUEST_TIMEOUT = (10.0, 30.0)
# Both comment reads used to go out unparameterised, which is GitHub's default
# page of 30 and no second page: a summary or a thread past the 30th comment
# did not exist as far as this adapter was concerned. 100 is the API maximum;
@@ -77,9 +81,11 @@ def _create_default_session() -> requests.Session:
# (which the server states it did not process) while holding it back
# on 502. Writes are therefore left to the caller, which already logs
# a failed post and carries on; a duplicated comment needs a human to
- # delete it. The other write verbs go with POST: no adapter issues a
- # DELETE, and the summary update (PUT, or PATCH on GitHub) is at best
- # a no-op on replay and at worst a version conflict. Connection
+ # delete it. The other write verbs go with POST: DELETE (the prune
+ # pass) is held back with them rather than special-cased for the
+ # idempotency a replayed delete would enjoy, and the summary update
+ # (PUT, or PATCH on GitHub) is at best a no-op on replay and at worst
+ # a version conflict. Connection
# errors are still retried for every verb: urllib3 gates only its
# read-error path on the method, and a connection that was never
# established carried no write to duplicate.
@@ -142,7 +148,9 @@ def _headers(self, host: str, extra: dict[str, str] | None = None) -> dict[str,
def get_pr(self, ref: PRRef) -> PRData:
"""Fetch normalized PR metadata."""
url = f"{self._api_base(ref)}/repos/{ref.owner}/{ref.repo}/pulls/{ref.number}"
- resp = self.session.get(url, headers=self._headers(ref.host))
+ resp = self.session.get(
+ url, headers=self._headers(ref.host), timeout=_REQUEST_TIMEOUT
+ )
resp.raise_for_status()
data: dict[str, Any] = resp.json()
@@ -168,7 +176,26 @@ def get_diff(self, ref: PRRef) -> str:
ref.host,
{"Accept": "application/vnd.github.v3.diff, application/vnd.diff"},
)
- resp = self.session.get(url, headers=headers)
+ resp = self.session.get(url, headers=headers, timeout=_REQUEST_TIMEOUT)
+ resp.raise_for_status()
+ return resp.text
+
+ def get_compare_diff(self, ref: PRRef, *, base_sha: str, head_sha: str) -> str:
+ """Return the unified diff of ``head_sha`` against its merge-base with ``base_sha``.
+
+ Uses the compare endpoint with three dots, ``{base}...{head}``, which
+ diffs from the merge-base exactly as the PR's own diff does; GitHub
+ answers the two-dot spelling with a 404. The diff media type makes the
+ body the raw diff text rather than the JSON comparison. An empty range
+ (``head_sha`` already merged into ``base_sha``) comes back as ``""``,
+ returned unmodified. Raises on an HTTP or transport failure.
+ """
+ url = (
+ f"{self._api_base(ref)}/repos/{ref.owner}/{ref.repo}"
+ f"/compare/{base_sha}...{head_sha}"
+ )
+ headers = self._headers(ref.host, {"Accept": "application/vnd.github.diff"})
+ resp = self.session.get(url, headers=headers, timeout=_REQUEST_TIMEOUT)
resp.raise_for_status()
return resp.text
@@ -195,6 +222,7 @@ def _iter_comment_pages(
url,
headers=headers,
params={"per_page": _PAGE_SIZE, "page": page_number},
+ timeout=_REQUEST_TIMEOUT,
)
except requests.RequestException as e:
raise FeedReadError(
@@ -252,10 +280,16 @@ def post_summary(self, ref: PRRef, body: str) -> None:
if existing_comment_id is not None:
patch_url = f"{self._api_base(ref)}/repos/{ref.owner}/{ref.repo}/issues/comments/{existing_comment_id}"
- resp = self.session.patch(patch_url, json={"body": body}, headers=headers)
+ resp = self.session.patch(
+ patch_url, json={"body": body}, headers=headers,
+ timeout=_REQUEST_TIMEOUT,
+ )
resp.raise_for_status()
else:
- resp = self.session.post(list_url, json={"body": body}, headers=headers)
+ resp = self.session.post(
+ list_url, json={"body": body}, headers=headers,
+ timeout=_REQUEST_TIMEOUT,
+ )
resp.raise_for_status()
def post_inline_comments(self, ref: PRRef, comments: Sequence[InlineComment]) -> int:
@@ -283,7 +317,9 @@ def post_inline_comments(self, ref: PRRef, comments: Sequence[InlineComment]) ->
"side": comment.side or "RIGHT",
"commit_id": commit_id,
}
- resp = self.session.post(url, json=payload, headers=headers)
+ resp = self.session.post(
+ url, json=payload, headers=headers, timeout=_REQUEST_TIMEOUT
+ )
if resp.status_code == 422:
# A line outside the diff is the expected 422 and skipping it
# is correct, but the same status covers a malformed payload
@@ -357,7 +393,9 @@ def get_file_content(self, ref: PRRef, path: str, *, sha: str) -> str | None:
)
headers = self._headers(ref.host, {"Accept": "application/vnd.github.raw+json"})
try:
- resp = self.session.get(url, headers=headers, params={"ref": sha})
+ resp = self.session.get(
+ url, headers=headers, params={"ref": sha}, timeout=_REQUEST_TIMEOUT
+ )
except requests.RequestException as e:
logger.debug("get_file_content failed for %s@%s: %s", path, sha, e)
return None
@@ -417,7 +455,9 @@ def prune_inline_comments(self, ref: PRRef) -> int:
f"{self._api_base(ref)}/repos/{ref.owner}/{ref.repo}"
f"/pulls/comments/{comment_id}"
)
- resp = self.session.delete(delete_url, headers=headers)
+ resp = self.session.delete(
+ delete_url, headers=headers, timeout=_REQUEST_TIMEOUT
+ )
if resp.ok:
removed += 1
else:
diff --git a/src/prxref/forges/gitlab.py b/src/prxref/forges/gitlab.py
index 1fd237b..ace1462 100644
--- a/src/prxref/forges/gitlab.py
+++ b/src/prxref/forges/gitlab.py
@@ -106,6 +106,57 @@ def _make_retry_session() -> requests.Session:
_DEFAULT_SESSION = _make_retry_session()
+def _render_diff_entries(diffs: list[dict]) -> str:
+ """Render GitLab's structured diff entries as one git-style unified diff.
+
+ GitLab serves no raw diff for a merge request or a compare, only a list of
+ per-file entries whose ``diff`` holds the hunks without their headers. This
+ rebuilds each file's ``diff --git`` header from the entry's flags, then
+ appends the hunks, so the parser downstream sees the same text shape every
+ other forge returns.
+ """
+ diff_parts: list[str] = []
+ for d in diffs:
+ old_path = d.get("old_path") or ""
+ new_path = d.get("new_path") or ""
+ new_file = d.get("new_file", False)
+ deleted_file = d.get("deleted_file", False)
+ renamed_file = d.get("renamed_file", False)
+ raw_diff = d.get("diff") or ""
+
+ header_lines = [f"diff --git a/{old_path} b/{new_path}"]
+ if new_file:
+ header_lines.append("new file mode 100644")
+ header_lines.append("--- /dev/null")
+ header_lines.append(f"+++ b/{new_path}")
+ elif deleted_file:
+ header_lines.append("deleted file mode 100644")
+ header_lines.append(f"--- a/{old_path}")
+ header_lines.append("+++ /dev/null")
+ elif renamed_file:
+ header_lines.append(f"rename from {old_path}")
+ header_lines.append(f"rename to {new_path}")
+ header_lines.append(f"--- a/{old_path}")
+ header_lines.append(f"+++ b/{new_path}")
+ else:
+ header_lines.append(f"--- a/{old_path}")
+ header_lines.append(f"+++ b/{new_path}")
+
+ file_unified = "\n".join(header_lines)
+ if raw_diff:
+ if not raw_diff.startswith("\n"):
+ file_unified += "\n"
+ file_unified += raw_diff
+ if not file_unified.endswith("\n"):
+ file_unified += "\n"
+ else:
+ file_unified += "\n"
+
+ diff_parts.append(file_unified)
+
+ return "".join(diff_parts)
+
+
class ForgeImpl:
"""GitLab Forge adapter."""
@@ -222,61 +273,89 @@ def get_pr(self, ref: PRRef) -> PRData:
)
def get_diff(self, ref: PRRef) -> str:
- """Fetch the raw unified diff of the PR (all files)."""
+ """Fetch the raw unified diff of the PR (all files).
+
+ The ``/diffs`` listing is paginated, 20 entries a page by default, so
+ a single unparameterised request reviewed the first 20 files of a
+ larger MR and silently dropped the rest. The listing is walked with
+ ``_iter_pages`` like every other GitLab collection here: a page that
+ cannot be read, or a listing that outruns the page budget, raises
+ ``FeedReadError`` rather than handing back the files that happened to
+ arrive. An entry GitLab marks ``collapsed`` or ``too_large`` carries no
+ hunks; it is rendered as a header-only file and logged at WARNING, as
+ ``get_compare_diff`` does. Raises ``ValueError`` for
+ an MR with no file entries at all.
+
+ ``access_raw_diffs`` is not sent: ``/diffs`` returns the same bodies
+ with or without it, and only the deprecated ``/changes`` endpoint
+ reads it.
+ """
headers = self._get_auth_headers()
base = self._api_base(ref)
url = f"{base}/merge_requests/{ref.number}/diffs"
- params = {"access_raw_diffs": "true"}
- resp = self._session.get(url, headers=headers, params=params, timeout=_REQUEST_TIMEOUT)
- resp.raise_for_status()
- diffs = resp.json()
+ diffs = [
+ entry
+ for page in self._iter_pages(ref, url, headers, what="MR diff list")
+ for entry in page
+ ]
if not diffs:
raise ValueError(
f"Empty diff received from GitLab for {ref.owner}/{ref.repo}#{ref.number}"
)
- diff_parts: list[str] = []
for d in diffs:
- old_path = d.get("old_path") or ""
- new_path = d.get("new_path") or ""
- new_file = d.get("new_file", False)
- deleted_file = d.get("deleted_file", False)
- renamed_file = d.get("renamed_file", False)
- raw_diff = d.get("diff") or ""
-
- header_lines = [f"diff --git a/{old_path} b/{new_path}"]
- if new_file:
- header_lines.append("new file mode 100644")
- header_lines.append("--- /dev/null")
- header_lines.append(f"+++ b/{new_path}")
- elif deleted_file:
- header_lines.append("deleted file mode 100644")
- header_lines.append(f"--- a/{old_path}")
- header_lines.append("+++ /dev/null")
- elif renamed_file:
- header_lines.append(f"rename from {old_path}")
- header_lines.append(f"rename to {new_path}")
- header_lines.append(f"--- a/{old_path}")
- header_lines.append(f"+++ b/{new_path}")
- else:
- header_lines.append(f"--- a/{old_path}")
- header_lines.append(f"+++ b/{new_path}")
-
- file_unified = "\n".join(header_lines)
- if raw_diff:
- if not raw_diff.startswith("\n"):
- file_unified += "\n"
- file_unified += raw_diff
- if not file_unified.endswith("\n"):
- file_unified += "\n"
- else:
- file_unified += "\n"
-
- diff_parts.append(file_unified)
-
- return "".join(diff_parts)
+ if d.get("too_large") or d.get("collapsed"):
+ logger.warning(
+ "GitLab MR diff: %s has no inline diff (too_large/collapsed); "
+ "it is reviewed as header-only",
+ d.get("new_path") or d.get("old_path"),
+ )
+ return _render_diff_entries(diffs)
+
+ def get_compare_diff(self, ref: PRRef, *, base_sha: str, head_sha: str) -> str:
+ """Return the unified diff of ``head_sha`` against its merge-base with ``base_sha``.
+
+ Uses ``repository/compare`` with ``straight=false``, the merge-base
+ form, and renders its ``diffs`` entries with the same header logic as
+ ``get_diff``. ``unidiff`` is deliberately not requested: it puts the
+ ``---``/``+++`` lines inside each entry, and the renderer would then
+ write them twice.
+
+ A ``compare_timeout`` means GitLab cut the file list short, so this
+ raises rather than hand back part of the range. An entry marked
+ ``too_large`` or ``collapsed`` carries no hunks; it is kept as a
+ header-only file and logged at WARNING. An empty range comes back as
+ ``""``. Raises on an HTTP or transport failure.
+ """
+ resp = self._session.get(
+ f"{self._api_base(ref)}/repository/compare",
+ headers=self._get_auth_headers(),
+ params={"from": base_sha, "to": head_sha, "straight": "false"},
+ timeout=_REQUEST_TIMEOUT,
+ )
+ resp.raise_for_status()
+ body = resp.json()
+ if not isinstance(body, dict):
+ raise ValueError(
+ f"GitLab compare {base_sha}...{head_sha} returned "
+ f"{type(body).__name__}, not a comparison object"
+ )
+ if body.get("compare_timeout"):
+ raise ValueError(
+ f"GitLab compare {base_sha}...{head_sha} timed out; its diff list "
+ "would be incomplete"
+ )
+ diffs = body.get("diffs") or []
+ for d in diffs:
+ if d.get("too_large") or d.get("collapsed"):
+ logger.warning(
+ "GitLab compare: %s has no inline diff (too_large/collapsed); "
+ "it is reviewed as header-only",
+ d.get("new_path") or d.get("old_path"),
+ )
+ return _render_diff_entries(diffs)
def _iter_pages(
self,
diff --git a/src/prxref/forges/replay.py b/src/prxref/forges/replay.py
new file mode 100644
index 0000000..2f16974
--- /dev/null
+++ b/src/prxref/forges/replay.py
@@ -0,0 +1,255 @@
+"""Read-only forges for evaluation replays (issue #65).
+
+A replay reviews a fixed, reproducible input instead of whatever the PR looks
+like now, and it never writes to a forge. Two forges serve it:
+
+- :class:`LocalDiffForge` serves a diff file on disk (``--diff-file`` with no
+ ``--pr-url``): no network, no threads, no file reads.
+- :class:`ReplayForge` wraps a real forge and pins what the orchestrator sees:
+ the diff of a commit range (``base_sha``/``head_sha``, through the inner
+ forge's optional ``get_compare_diff``) or a diff text, file reads at the
+ pinned head, and optionally no existing threads.
+
+Both raise on every write method, as defence in depth: the CLI already forces
+``post=False`` on a replay. Neither is registered in ``detect_forge`` or
+``make_forge``, because neither is ever produced from a URL, and neither
+implements ``get_compare_diff``: pinning is resolved against the inner forge.
+"""
+from __future__ import annotations
+
+import dataclasses
+import email
+import email.message
+import email.policy
+import email.utils
+import re
+from collections.abc import Sequence
+from pathlib import Path
+
+from .base import Forge, InlineComment, PRData, PRRef, Thread
+
+NEVER_POSTS = "replay runs never write to a forge"
+
+_DIFF_START_RE = re.compile(r"^diff --git ", re.MULTILINE)
+_PLAIN_DIFF_START_RE = re.compile(r"^--- ", re.MULTILINE)
+_SUBJECT_HEADER_RE = re.compile(r"^Subject:", re.MULTILINE | re.IGNORECASE)
+_PATCH_PREFIX_RE = re.compile(r"^\s*\[[^\]]*\bPATCH\b[^\]]*\]\s*", re.IGNORECASE)
+_ENCODED_TRANSFERS = ("quoted-printable", "base64")
+
+
+def _patch_metadata(text: str, path: str) -> tuple[str, str, str]:
+ """Return ``(title, description, author)`` for a diff file's text.
+
+ A ``git format-patch`` mail (the preamble before the first ``diff --git``
+ line starts with ``From `` and has a ``Subject:`` header) gives its
+ subject without the ``[PATCH …]`` tag (``[RFC PATCH v2 1/3]`` included;
+ a bracket without ``PATCH`` is part of the title), its body up to git's ``---``
+ diffstat separator, and the author's display name (else the address).
+ Headers are unfolded and RFC 2047-decoded by ``email.policy.default``.
+ A series yields the metadata of its first patch. Anything else is a plain
+ diff, titled ``Local diff `` with no description or author;
+ so is a mail that cannot be parsed, because metadata is never worth a
+ failed review.
+ """
+ plain = (f"Local diff {Path(path).name}", "", "")
+ start = _DIFF_START_RE.search(text) or _PLAIN_DIFF_START_RE.search(text)
+ preamble = text[: start.start()] if start else text
+ if not preamble.startswith("From ") or not _SUBJECT_HEADER_RE.search(preamble):
+ return plain
+ rest = preamble.split("\n", 1)[1] if "\n" in preamble else ""
+ try:
+ msg = email.message_from_string(rest, policy=email.policy.default)
+ title = _PATCH_PREFIX_RE.sub("", str(msg.get("Subject", ""))).strip()
+ name, address = email.utils.parseaddr(str(msg.get("From", "")))
+ body = _mail_body(msg)
+ except Exception: # noqa: BLE001 - metadata is never worth a failed review
+ return plain
+ kept: list[str] = []
+ for line in body.split("\n"):
+ if line.rstrip("\r") == "---":
+ break
+ kept.append(line)
+ return title or plain[0], "\n".join(kept).strip(), name or address
+
+
+def _mail_body(msg: email.message.Message) -> str:
+ """The text body of a parsed patch mail, decoded, or ``""`` if multipart.
+
+ ``get_content()`` is not used: on a message parsed from ``str`` it
+ re-decodes an 8bit body and mangles every non-ASCII character, while
+ ``get_payload()`` returns it as written. Only a quoted-printable or base64
+ body needs decoding, with the declared charset.
+ """
+ cte = str(msg.get("Content-Transfer-Encoding", "")).strip().lower()
+ if cte in _ENCODED_TRANSFERS:
+ raw = msg.get_payload(decode=True) or b""
+ return raw.decode(msg.get_content_charset() or "utf-8", errors="replace")
+ payload = msg.get_payload()
+ return payload if isinstance(payload, str) else ""
+
+
+class LocalDiffForge:
+ """A read-only Forge over a diff file: no network, no threads, no file reads, never posts.
+
+ It deliberately has no ``get_file_content``, and its PR has no head sha,
+ so the orchestrator skips context injection. ``get_diff`` raises on a
+ blank diff, which the orchestrator turns into an ``Error`` run: an empty
+ replay input almost always means the wrong file, never a clean PR.
+ """
+
+ name = "local"
+
+ def __init__(self, diff_text: str, *, path: str):
+ self._diff_text = diff_text
+ self._path = path
+
+ @staticmethod
+ def parse_pr_url(url: str) -> PRRef | None:
+ """Never recognizes a URL: a local diff is never produced from one."""
+ return None
+
+ @staticmethod
+ def ref_for(path: str) -> PRRef:
+ """The synthetic ``PRRef`` of a diff-only run: forge ``"local"``, the file's URI."""
+ return PRRef(
+ forge="local", host="", owner="", repo="", number=0,
+ url=Path(path).resolve().as_uri(),
+ )
+
+ def get_pr(self, ref: PRRef) -> PRData:
+ """PR metadata from the patch mail headers, or a title from the file name.
+
+ Both shas are empty and ``raw`` is ``{"diff_file": path}``, with the
+ path as the caller gave it.
+ """
+ title, description, author = _patch_metadata(self._diff_text, self._path)
+ return PRData(
+ title=title, description=description, author=author,
+ source_branch="", target_branch="", source_sha="", target_sha="",
+ raw={"diff_file": self._path},
+ )
+
+ def get_diff(self, ref: PRRef) -> str:
+ """The diff text, unmodified; raises ``ValueError`` when it is blank."""
+ if not self._diff_text.strip():
+ raise ValueError("replay diff from the --diff-file is empty")
+ return self._diff_text
+
+ def list_threads(self, ref: PRRef) -> list[Thread]:
+ """Always empty: a diff file has no discussion."""
+ return []
+
+ def post_summary(self, ref: PRRef, body: str) -> None:
+ """Always raises ``RuntimeError``: a replay never writes to a forge."""
+ raise RuntimeError(NEVER_POSTS)
+
+ def post_inline_comments(self, ref: PRRef, comments: Sequence[InlineComment]) -> int:
+ """Always raises ``RuntimeError``: a replay never writes to a forge."""
+ raise RuntimeError(NEVER_POSTS)
+
+ def prune_inline_comments(self, *args: object, **kwargs: object) -> int:
+ """Always raises ``RuntimeError``: a replay never writes to a forge."""
+ raise RuntimeError(NEVER_POSTS)
+
+
+class ReplayForge:
+ """Wrap a Forge so a review sees one pinned, reproducible input and can never post.
+
+ ``base_sha``/``head_sha`` (full, lowercased, given together) pin the
+ range: ``get_diff`` returns the inner forge's ``get_compare_diff`` of it,
+ and ``get_pr`` reports ``head_sha``/``base_sha`` as the PR's
+ ``source_sha``/``target_sha``, so every file read happens at the pinned
+ head. The PR's title and description stay the current ones. ``diff_text``,
+ when given, is the diff instead of any fetched one. A blank pinned or
+ ``diff_text`` diff raises ``ValueError``, which the orchestrator turns into
+ an ``Error`` run. With neither, ``get_diff`` is the inner forge's own, so
+ that run differs from a normal one only in its threads. ``hide_threads``
+ makes ``list_threads`` return ``[]`` without asking the inner forge.
+
+ Raises ``ValueError`` when only one of the two shas is given, or when a
+ pinned range must be fetched from a forge with no ``get_compare_diff``.
+ """
+
+ name = "replay"
+
+ def __init__(
+ self,
+ inner: Forge,
+ *,
+ base_sha: str | None = None,
+ head_sha: str | None = None,
+ hide_threads: bool = False,
+ diff_text: str | None = None,
+ ):
+ if bool(base_sha) != bool(head_sha):
+ raise ValueError("ReplayForge: base_sha and head_sha must be given together")
+ if head_sha and diff_text is None and getattr(inner, "get_compare_diff", None) is None:
+ forge_name = getattr(inner, "name", type(inner).__name__)
+ raise ValueError(
+ f"ReplayForge: the {forge_name} forge cannot fetch a pinned commit range"
+ )
+ self._inner = inner
+ self._base_sha = base_sha or None
+ self._head_sha = head_sha or None
+ self._hide_threads = hide_threads
+ self._diff_text = diff_text
+
+ @staticmethod
+ def parse_pr_url(url: str) -> PRRef | None:
+ """Never recognizes a URL: a replay always wraps an already-built forge."""
+ return None
+
+ def get_pr(self, ref: PRRef) -> PRData:
+ """The inner PR, with its shas replaced by the pinned range when one is set."""
+ pr = self._inner.get_pr(ref)
+ if self._head_sha:
+ pr = dataclasses.replace(
+ pr, source_sha=self._head_sha, target_sha=self._base_sha,
+ )
+ return pr
+
+ def get_diff(self, ref: PRRef) -> str:
+ """The replay's diff: ``diff_text``, else the pinned range, else the live PR diff."""
+ if self._diff_text is not None:
+ if not self._diff_text.strip():
+ raise ValueError("replay diff from the --diff-file is empty")
+ return self._diff_text
+ if self._head_sha:
+ text = self._inner.get_compare_diff(
+ ref, base_sha=self._base_sha, head_sha=self._head_sha,
+ )
+ if not text.strip():
+ raise ValueError(
+ f"replay diff from {self._base_sha[:12]}...{self._head_sha[:12]} "
+ "is empty (is the head already contained in the base?)"
+ )
+ return text
+ return self._inner.get_diff(ref)
+
+ def list_threads(self, ref: PRRef) -> list[Thread]:
+ """``[]`` when threads are hidden (the inner forge is not asked), else the inner's."""
+ if self._hide_threads:
+ return []
+ return self._inner.list_threads(ref)
+
+ def get_file_content(self, ref: PRRef, path: str, *, sha: str) -> str | None:
+ """The inner forge's file read; ``None`` when it has none or it raises."""
+ reader = getattr(self._inner, "get_file_content", None)
+ if reader is None:
+ return None
+ try:
+ return reader(ref, path, sha=sha)
+ except Exception: # noqa: BLE001 - the Protocol says this never raises
+ return None
+
+ def post_summary(self, ref: PRRef, body: str) -> None:
+ """Always raises ``RuntimeError``: a replay never writes to a forge."""
+ raise RuntimeError(NEVER_POSTS)
+
+ def post_inline_comments(self, ref: PRRef, comments: Sequence[InlineComment]) -> int:
+ """Always raises ``RuntimeError``: a replay never writes to a forge."""
+ raise RuntimeError(NEVER_POSTS)
+
+ def prune_inline_comments(self, *args: object, **kwargs: object) -> int:
+ """Always raises ``RuntimeError``: a replay never writes to a forge."""
+ raise RuntimeError(NEVER_POSTS)
diff --git a/src/prxref/formatter.py b/src/prxref/formatter.py
index 9d043e7..6b071a4 100644
--- a/src/prxref/formatter.py
+++ b/src/prxref/formatter.py
@@ -9,6 +9,8 @@
from collections import Counter
from pathlib import Path
+from .markers import SCOPE_LABELS, marker_for
+from .markers import SEVERITY_MARKERS as _SEVERITY_MARKERS
from .triage import Finding
try:
@@ -17,28 +19,23 @@
_reviewer_load_prompt = None
-_SEVERITY_MARKERS: dict[str, str] = {
- "error": "🟥",
- "warning": "🟧",
- "outofscope": "🟦",
+_SEVERITY_ORDER: dict[str, int] = {
+ "error": 0, "warning": 1, "spec": 2, "outofscope": 3,
}
-_SEVERITY_ORDER: dict[str, int] = {"error": 0, "warning": 1, "outofscope": 2}
-_DEFAULT_SUMMARY_TEMPLATE = """\
-## {verdict_banner}
-
-**Findings:** 🟥 {error_count} error · 🟧 {warning_count} warning · 🟦 {outofscope_count} outofscope
-
-{active_count} active of {total_count} raw
-
-{findings_table}
-{dropped_section}
----
-
-*chunks {chunk_count} · {input_tokens} in / {output_tokens} out tokens · {elapsed_s}s · model {model}*
-
-*{attribution}*
-"""
+_DEFAULT_SUMMARY_TEMPLATE = (
+ "## {verdict_banner}\n\n"
+ "**Findings:** 🟥 {error_count} error · 🟧 {warning_count} warning · "
+ "🔍 {spec_count} spec · ⬜ {outofscope_count} outofscope\n"
+ "{spec_note}\n"
+ "{active_count} active of {total_count} raw\n\n"
+ "{findings_table}\n"
+ "{dropped_section}\n"
+ "---\n\n"
+ "*chunks {chunk_count} · {input_tokens} in / {output_tokens} out tokens"
+ " · {elapsed_s}s · model {model}*\n\n"
+ "*{attribution}*\n"
+)
def _norm_severity(severity: str) -> str:
@@ -66,7 +63,11 @@ def _escape_cell(text: str) -> str:
def _findings_table(findings: list[Finding]) -> str:
- """Render the ``| severity | file:line | title |`` table, error-first."""
+ """Render the ``| severity | file:line | title |`` table, error-first.
+
+ The severity cell is :func:`markers.marker_for`, so a finding outside the
+ ticket shows the out-of-ticket marker in front of its severity glyph.
+ """
if not findings:
return "No findings survived the quality passes."
ordered = sorted(
@@ -84,7 +85,7 @@ def _findings_table(findings: list[Finding]) -> str:
"| --- | --- | --- |",
]
rows.extend(
- f"| {_SEVERITY_MARKERS[_norm_severity(f.severity)]} "
+ f"| {marker_for(_norm_severity(f.severity), f.scope)} "
f"| {_escape_cell(_fmt_location(f))} "
f"| {_escape_cell(f.title)} |"
for f in ordered
@@ -123,10 +124,11 @@ def _load_summary_template() -> str:
"""Load ``prompts/summary.md`` via the shared loader, else inline default.
Placeholder contract for the template owner: ``verdict_banner``,
- ``error_count``, ``warning_count``, ``outofscope_count``, ``active_count``,
- ``total_count``, ``findings_table``, ``dropped_section``,
- ``chunk_count``, ``input_tokens``, ``output_tokens``, ``elapsed_s``,
- ``model``, ``attribution``.
+ ``error_count``, ``warning_count``, ``spec_count``, ``spec_note``,
+ ``outofscope_count``, ``active_count``, ``total_count``,
+ ``findings_table``, ``dropped_section``, ``chunk_count``,
+ ``input_tokens``, ``output_tokens``, ``elapsed_s``, ``model``,
+ ``attribution``.
"""
if _reviewer_load_prompt is not None:
try:
@@ -148,9 +150,17 @@ def build_attribution(model: str, elapsed_ms: int, tokens: int) -> str:
def format_inline_comment(f: Finding, attribution: str) -> str:
- """Render one finding as a forge-neutral inline-comment body."""
- marker = _SEVERITY_MARKERS[_norm_severity(f.severity)]
- return f"{marker} **{f.title}**\n\n{f.body}\n\n*{attribution}*"
+ """Render one finding as a forge-neutral inline-comment body.
+
+ A finding outside the ticket (scope ``"out"``) gets the
+ :func:`markers.marker_for` prefix and its :data:`markers.SCOPE_LABELS`
+ entry, as `` **[OUTSIDE TICKET] **``; scope
+ ``"in"`` and ``"unknown"`` render exactly the severity-only body.
+ """
+ marker = marker_for(_norm_severity(f.severity), f.scope)
+ scope_label = SCOPE_LABELS.get(f.scope)
+ title = f"[{scope_label}] {f.title}" if scope_label else f.title
+ return f"{marker} **{title}**\n\n{f.body}\n\n*{attribution}*"
def format_summary(
@@ -177,9 +187,13 @@ def format_summary(
"warning_count": sum(
1 for f in findings_active if _norm_severity(f.severity) == "warning"
),
+ "spec_count": sum(
+ 1 for f in findings_active if _norm_severity(f.severity) == "spec"
+ ),
"outofscope_count": sum(
1 for f in findings_active if _norm_severity(f.severity) == "outofscope"
),
+ "spec_note": "",
"active_count": len(findings_active),
"total_count": len(findings_active) + len(findings_dropped),
"findings_table": _findings_table(findings_active),
diff --git a/src/prxref/heuristics.py b/src/prxref/heuristics.py
index d899fff..009c7cf 100644
--- a/src/prxref/heuristics.py
+++ b/src/prxref/heuristics.py
@@ -51,6 +51,11 @@
"composer.lock",
})
+# The same set, public, for callers outside this module: the PR-size advisory
+# hands it to ``triage.count_size_relevant_changes``, because triage must never
+# import heuristics. This module's own checks keep using the private name.
+LOCKFILE_BASENAMES: frozenset[str] = _LOCKFILE_BASENAMES
+
# Case-insensitive basename prefixes: CHANGELOG.md, changelog.rst,
# HISTORY.txt, RELEASE_NOTES.md all match regardless of extension or case.
_PREFIX_BASENAMES = ("changelog", "history", "release_notes")
diff --git a/src/prxref/llm.py b/src/prxref/llm.py
index 610f941..0dca840 100644
--- a/src/prxref/llm.py
+++ b/src/prxref/llm.py
@@ -1,10 +1,12 @@
"""LLM contract: protocol + fallback-chain factory.
-Backends live in llm_backends.py (ferry / litellm / http). This module freezes
-the interface the pipeline codes against. NO provider-specific keys are read
-here; backend selection is PRXREF_LLM_BACKEND=ferry|litellm|http and the model
-fallback chain is PRXREF_LLM_MODELS="model1,model2,..." (first that answers
-within timeout wins; failures fail over fast).
+Backends live in llm_backends.py (openai-compat with its ferry / http aliases,
+and litellm) and llm_cli_backends.py (claude-cli, kiro-cli). This module
+freezes the interface the pipeline codes against. NO provider-specific keys
+are read here; backend selection is
+PRXREF_LLM_BACKEND=openai-compat|ferry|http|litellm|claude-cli|kiro-cli and
+the model fallback chain is PRXREF_LLM_MODELS="model1,model2,..." (first that
+answers within timeout wins; failures fail over fast).
"""
from __future__ import annotations
@@ -33,6 +35,16 @@ class InvokeResult:
reviewer can tell an operator to raise ``PRXREF_LLM_MAX_TOKENS`` instead of
handing them a bare ``JSONDecodeError``. A backend that does not report one
leaves it ``""`` — absent, never guessed.
+
+ ``cost_usd`` is the dollar amount the backend REPORTED for this call. When
+ the backend's fallback chain moved past truncated completions, those were
+ billed too and are included. ``None`` means no figure was reported: it
+ never means free, and no backend may default it to ``0.0``. ``cost_source``
+ names where the figure came from (``"usage.cost"``,
+ ``"x-litellm-response-cost"``, ``"litellm"`` or ``"claude-cli"``), and is
+ ``""`` whenever ``cost_usd`` is ``None``. A price-table estimate never
+ appears here: backends only report, and :mod:`prxref.costs` estimates
+ once, over the whole run.
"""
text: str
@@ -42,6 +54,8 @@ class InvokeResult:
backend: str = ""
elapsed_ms: int = 0
finish_reason: str = ""
+ cost_usd: float | None = None
+ cost_source: str = ""
class LLMClient(Protocol):
diff --git a/src/prxref/llm_backends.py b/src/prxref/llm_backends.py
index fca7caf..5fbed7a 100644
--- a/src/prxref/llm_backends.py
+++ b/src/prxref/llm_backends.py
@@ -1,9 +1,16 @@
-"""LLM backends: an OpenAI-compatible plain-HTTP client and an optional litellm wrapper.
+"""LLM backends: OpenAI-compatible HTTP, optional litellm, and subscription CLIs (``llm_cli_backends``).
The primary backend speaks plain HTTP to any OpenAI-compatible
-``/chat/completions`` endpoint. There is no default endpoint and no default
-model chain: ``PRXREF_LLM_BASE_URL`` and ``PRXREF_LLM_MODELS`` are required,
-and an unset one raises ``ConfigError`` rather than guessing a host.
+``/chat/completions`` endpoint. There is no default model chain on any
+backend: ``PRXREF_LLM_MODELS`` is required, and an unset one raises
+``ConfigError``. There is no default endpoint either:
+``PRXREF_LLM_BASE_URL`` is required by the openai-compat backend (and its
+``ferry``/``http`` aliases) and raises ``ConfigError`` when unset rather than
+guessing a host. The other backends do not use it: litellm resolves each
+model's own provider endpoint, and the CLI backends talk to their CLI. A set
+value is ignored there with one INFO line, never forwarded, so a deployment
+that set a placeholder URL to get past the old unconditional check keeps its
+routing on upgrade.
Fallback is a caller-side loop over the model chain: a model that answers
with HTTP >= 500, HTTP 429, a connection error, a timeout, a malformed
@@ -17,9 +24,28 @@
mechanism (which advances on errors only — a truncated litellm answer
still returns as success).
-Tenet: no provider credential is ever read and no env name is
-provider-specific — provider keys live behind the configured endpoint,
-never here.
+The ``claude-cli`` and ``kiro-cli`` backends (``llm_cli_backends``) run the
+user's own installed, logged-in CLI as a subprocess, one process per call,
+with the same caller-side model chain. They take the model chain, the
+timeout, a process-count cap (``PRXREF_LLM_CLI_CONCURRENCY``) and an
+optional binary path (``PRXREF_LLM_CLI_PATH``); the base URL, the API key,
+``max_tokens``, temperature and seed are not applied. The factory imports
+that module lazily, so the HTTP backends never load it.
+
+Cost: a backend reports the dollar figure its provider returned and never
+estimates one. The openai-compat client reads the body's ``usage.cost``
+first, then a LiteLLM-based gateway's ``x-litellm-response-cost`` response
+header; the litellm client reads ``_hidden_params["response_cost"]``. No
+figure leaves ``InvokeResult.cost_usd`` as ``None``, never ``0.0``. The
+price-table estimate is made once per run, in :mod:`prxref.costs`.
+
+Tenet: prxref never reads, stores, or forwards a provider credential, and
+its own settings are provider-neutral ``PRXREF_*`` names. A provider key
+lives behind the configured endpoint (openai-compat), in the provider SDK's
+own environment (litellm), or inside the user's own logged-in CLI
+(claude-cli, kiro-cli). The CLI backends remove a fixed list of
+credential-routing variable NAMES from the child process environment so the
+CLI falls back to its subscription login; the values are never read.
"""
from __future__ import annotations
@@ -32,6 +58,7 @@
import requests
+from . import costs
from .llm import ConfigError, InvokeResult, LLMClient
DEFAULT_BASE_URL = ""
@@ -43,6 +70,10 @@
# actually reaches the wire. Resolved by create_llm_client when the operator
# left PRXREF_LLM_TEMPERATURE unset or empty.
DEFAULT_TEMPERATURE = 0.0
+DEFAULT_CLI_CONCURRENCY = 2
+OPENAI_COMPAT_BACKENDS = ("openai-compat", "ferry", "http")
+CLI_BACKENDS = ("claude-cli", "kiro-cli")
+BACKENDS = (*OPENAI_COMPAT_BACKENDS, "litellm", *CLI_BACKENDS)
logger = logging.getLogger(__name__)
# Connecting is not generating: a reachable endpoint answers the TCP/TLS
# handshake in well under this, so a separate, much smaller connect budget
@@ -102,6 +133,54 @@ def _openai_error_message(resp: requests.Response) -> str:
return getattr(resp, "text", "") or ""
+def _header(headers: object, name: str) -> object:
+ """Case-insensitive lookup of one response header; ``None`` when absent.
+
+ A real response carries a ``requests.structures.CaseInsensitiveDict``,
+ whose ``get`` already ignores case. A plain mapping (a test double, or
+ anything else a session hands back) gets an exact ``get`` first and then
+ a casefolded scan of its items, so ``X-LiteLLM-Response-Cost`` and
+ ``x-litellm-response-cost`` read the same everywhere.
+ """
+ if not headers:
+ return None
+ getter = getattr(headers, "get", None)
+ if callable(getter):
+ value = getter(name)
+ if value is not None:
+ return value
+ items = getattr(headers, "items", None)
+ if not callable(items):
+ return None
+ wanted = name.casefold()
+ for key, value in items():
+ if isinstance(key, str) and key.casefold() == wanted:
+ return value
+ return None
+
+
+def _reported_cost(usage: object, resp: object) -> tuple[float | None, str]:
+ """The dollar figure the provider reported for one completion, and its source.
+
+ The body's ``usage.cost`` (OpenRouter sends it unasked) wins; it must be
+ a JSON number, so a string there is no figure. Otherwise the
+ ``x-litellm-response-cost`` header that a LiteLLM-based gateway sets, a
+ string by nature. Anything :func:`prxref.costs.valid_usd` rejects
+ (negative, ``NaN``, ``""``, ``"None"``) is no figure, and no figure is
+ ``(None, "")``: never ``0.0``.
+ """
+ if isinstance(usage, dict):
+ body_cost = usage.get("cost")
+ if not isinstance(body_cost, str):
+ cost = costs.valid_usd(body_cost)
+ if cost is not None:
+ return cost, "usage.cost"
+ cost = costs.valid_usd(_header(getattr(resp, "headers", None), "x-litellm-response-cost"))
+ if cost is not None:
+ return cost, "x-litellm-response-cost"
+ return None, ""
+
+
def _mark_unavailable(model: str, unavailable: set[str], lock: threading.Lock) -> bool:
"""Add ``model`` to ``unavailable`` under ``lock``; ``True`` only for the adding thread.
@@ -120,8 +199,6 @@ def _mark_unavailable(model: str, unavailable: set[str], lock: threading.Lock) -
class OpenAICompatClient(LLMClient):
"""Plain-HTTP client for an OpenAI-compatible endpoint.
- Tries each model in ``models`` order (cheap first for speed). A model
- fails on HTTP >= 500, HTTP 429, any other HTTP error, a connection
Tries each model in ``models`` order (cheap first for speed). A model
fails on HTTP >= 500, HTTP 429, any other HTTP error, a connection
error, a timeout, a malformed body, or a truncated completion
@@ -251,6 +328,7 @@ def invoke(
failures: list[str] = []
last_truncated: InvokeResult | None = None
+ received: list[tuple[float | None, str]] = []
for attempt, model in enumerate(self.models, start=1):
if model in self._unavailable:
failures.append(f"{model}: skipped (unavailable)")
@@ -326,6 +404,11 @@ def invoke(
)
failures.append(f"{model}: malformed response ({exc.__class__.__name__})")
continue
+ # Every completion that came back was billed, a truncated one the
+ # chain moves past included, so the call's figure sums them all.
+ attempt_cost, attempt_source = _reported_cost(usage, resp)
+ received.append((attempt_cost, attempt_source))
+ cost_usd, cost_source = costs.combine_reported(received)
if finish_reason.strip().lower() in _TRUNCATION_FINISH_REASONS:
# A truncated completion is HTTP 200, so without this branch
# it returned as success and PRXREF_LLM_MODELS never advanced.
@@ -343,13 +426,15 @@ def invoke(
backend="openai-compat",
elapsed_ms=elapsed_ms,
finish_reason=finish_reason,
+ cost_usd=cost_usd,
+ cost_source=cost_source,
)
continue
logger.info(
- "llm attempt %d/%d ok: model=%s %dms in=%s out=%s finish=%s",
+ "llm attempt %d/%d ok: model=%s %dms in=%s out=%s finish=%s cost=%s",
attempt, len(self.models), resp_model, elapsed_ms,
usage.get("prompt_tokens") or 0, usage.get("completion_tokens") or 0,
- finish_reason or "-",
+ finish_reason or "-", "-" if attempt_cost is None else attempt_cost,
)
return InvokeResult(
text=text,
@@ -359,6 +444,8 @@ def invoke(
backend="openai-compat",
elapsed_ms=elapsed_ms,
finish_reason=finish_reason,
+ cost_usd=cost_usd,
+ cost_source=cost_source,
)
# Exhausting the chain on truncation alone is a last resort, not a
# failure: the best answer anyone managed is still handed back, with
@@ -462,6 +549,12 @@ def invoke(
choice = response.choices[0]
text = choice.message.content or ""
usage = getattr(response, "usage", None)
+ # litellm prices the call from its own bundled map and leaves the
+ # figure here; completion_cost() is never called, because it raises on
+ # a model the map does not know. A string is not a figure.
+ hidden = getattr(response, "_hidden_params", None)
+ raw_cost = hidden.get("response_cost") if isinstance(hidden, dict) else getattr(hidden, "response_cost", None)
+ cost_usd = None if isinstance(raw_cost, str) else costs.valid_usd(raw_cost)
return InvokeResult(
text=text,
input_tokens=getattr(usage, "prompt_tokens", 0) if usage else 0,
@@ -471,6 +564,8 @@ def invoke(
elapsed_ms=elapsed_ms,
# Absent on a provider that does not report one; never guessed.
finish_reason=str(getattr(choice, "finish_reason", "") or ""),
+ cost_usd=cost_usd,
+ cost_source="litellm" if cost_usd is not None else "",
)
def _maybe_mark_unavailable(
@@ -592,16 +687,28 @@ def create_llm_client(
"""Build the configured client from ``cfg`` overrides then PRXREF_LLM_* env.
``cfg`` keys (LLM_BACKEND, LLM_BASE_URL, LLM_API_KEY, LLM_MODELS,
- LLM_REASONING_EFFORT) win over env; env never includes provider
- credentials. PRXREF_LLM_BACKEND selects ``openai-compat`` (default)
- with ``ferry`` as an alias, or ``litellm``. PRXREF_LLM_BASE_URL and
- PRXREF_LLM_MODELS are required and have no defaults — an unset one
- raises :class:`~prxref.llm.ConfigError`. PRXREF_LLM_API_KEY is
- optional and may be empty for a local no-auth server.
- PRXREF_LLM_MODELS (comma list, cheap first) feeds litellm too.
+ LLM_REASONING_EFFORT, LLM_TIMEOUT, LLM_TEMPERATURE, LLM_SEED,
+ LLM_CLI_PATH, LLM_CLI_CONCURRENCY, in either case) win over env; env
+ never includes provider credentials. PRXREF_LLM_BACKEND is read
+ case-insensitively and selects ``openai-compat`` (the default, with
+ ``ferry`` and ``http`` as aliases), ``litellm``, ``claude-cli`` or
+ ``kiro-cli``. Any other value raises :class:`~prxref.llm.ConfigError`
+ naming PRXREF_LLM_BACKEND, before any other setting is looked at, so a
+ typo is reported as itself (exit 2) rather than as a missing endpoint or
+ a failed review.
+ PRXREF_LLM_MODELS (comma list, cheap first) is required by every
+ backend and has no default. PRXREF_LLM_BASE_URL has no default and is
+ required by the openai-compat family only; it is checked before the
+ models, so a run with both unset still names the endpoint first. The
+ other backends do not use it: when it is set anyway it is ignored with
+ one INFO line and never forwarded (litellm resolves each model's own
+ provider endpoint; a LiteLLM proxy is OpenAI-compatible and belongs on
+ ``openai-compat``). PRXREF_LLM_API_KEY is openai-compat only, optional,
+ and may be empty for a local no-auth server.
PRXREF_LLM_REASONING_EFFORT is passed through unvalidated to the
openai-compat client for models that cannot disable reasoning
- (e.g. GLM-5.3-Flash's ``low``/``high``/``max``); empty omits it.
+ (e.g. GLM-5.3-Flash's ``low``/``high``/``max``) and to claude-cli as its
+ effort setting; empty omits it, and litellm and kiro-cli ignore it.
PRXREF_LLM_TIMEOUT (seconds, default 45.0, must be > 0) becomes the
client's ``default_timeout``. PRXREF_LLM_TEMPERATURE is parsed to a
float (finite, >= 0 — no upper bound, since the maximum is
@@ -609,7 +716,8 @@ def create_llm_client(
``DEFAULT_TEMPERATURE`` (0.0), which IS sent — temperature 0 keeps
reviews reproducible by default, and an operator-set value wins.
PRXREF_LLM_SEED (integer >= 0, where 0 is a valid seed) is passed to
- both backends as a top-level ``seed`` and always wins when set. Unset
+ the openai-compat and litellm backends as a top-level ``seed`` and
+ always wins when set. Unset
or empty does NOT omit the field: temperature 0 alone cannot pin hosted
inference, so the factory derives ONE random seed per process
(:func:`_auto_run_seed`) and stamps it on every client it builds —
@@ -621,6 +729,19 @@ def create_llm_client(
``PRXREF_LLM_MAX_TOKENS`` is deliberately NOT read here: it is a
per-call budget threaded cfg -> orchestrator -> reviewer -> ``invoke``,
so a client-level copy could never win and would be dead config.
+
+ The CLI backends (``claude-cli``, ``kiro-cli``) are built by
+ :func:`prxref.llm_cli_backends.build_cli_client`, imported lazily.
+ PRXREF_LLM_CLI_PATH overrides the binary (empty = ``claude`` or
+ ``kiro-cli`` on ``PATH``; one that cannot be found is a ConfigError
+ naming it). PRXREF_LLM_CLI_CONCURRENCY caps the CLI processes one client
+ runs at once (integer >= 1, default ``DEFAULT_CLI_CONCURRENCY``), and is
+ re-checked here for callers that bypass ``config.load_config``. Neither
+ CLI has a temperature or seed option, so both are still parsed (a
+ malformed value still exits 2) but are not applied, and an explicitly
+ set one logs one WARNING saying so; the client's ``temperature`` and
+ ``seed`` attributes are ``None``, which the run record's ``sampling``
+ reports truthfully.
"""
cfg = cfg or {}
@@ -634,10 +755,15 @@ def _get(key: str, env: str, default: str | None = None) -> str | None:
return os.environ.get(env, default)
backend = (_get("LLM_BACKEND", "PRXREF_LLM_BACKEND", "openai-compat") or "").strip().lower() or "openai-compat"
+ if backend not in BACKENDS:
+ raise ConfigError(
+ f"PRXREF_LLM_BACKEND: must be one of {', '.join(BACKENDS)} "
+ f"(case-insensitive), got {backend!r}"
+ )
raw_models = _get("LLM_MODELS", "PRXREF_LLM_MODELS", DEFAULT_MODELS) or ""
models = [m.strip() for m in raw_models.split(",") if m.strip()]
base_url = _get("LLM_BASE_URL", "PRXREF_LLM_BASE_URL", DEFAULT_BASE_URL) or ""
- if not base_url.strip():
+ if backend in OPENAI_COMPAT_BACKENDS and not base_url.strip():
raise ConfigError(
"no LLM endpoint configured. Set PRXREF_LLM_BASE_URL to an "
"OpenAI-compatible /chat/completions endpoint "
@@ -669,7 +795,7 @@ def _get(key: str, env: str, default: str | None = None) -> str | None:
)
if seed is None:
seed = _auto_run_seed()
- if backend in ("openai-compat", "ferry", "http"):
+ if backend in OPENAI_COMPAT_BACKENDS:
return OpenAICompatClient(
base_url=base_url,
api_key=_get("LLM_API_KEY", "PRXREF_LLM_API_KEY", DEFAULT_API_KEY) or DEFAULT_API_KEY,
@@ -680,8 +806,42 @@ def _get(key: str, env: str, default: str | None = None) -> str | None:
temperature=temperature,
seed=seed,
)
+ if base_url.strip():
+ logger.info(
+ "PRXREF_LLM_BASE_URL is set but not used by the %s backend; ignoring it",
+ backend,
+ )
if backend == "litellm":
return LiteLLMClient(
models=models, default_timeout=timeout, temperature=temperature, seed=seed
)
- raise LLMError(f"unknown PRXREF_LLM_BACKEND {backend!r}; expected openai-compat|ferry|http|litellm")
+ unapplied = [
+ env
+ for key, env in (
+ ("LLM_TEMPERATURE", "PRXREF_LLM_TEMPERATURE"),
+ ("LLM_SEED", "PRXREF_LLM_SEED"),
+ )
+ if (_get(key, env) or "").strip()
+ ]
+ if unapplied:
+ logger.warning(
+ "%s %s not applied by %s (the CLI has no such option)",
+ " / ".join(unapplied),
+ "is" if len(unapplied) == 1 else "are",
+ backend,
+ )
+ concurrency = _int_setting(
+ _get("LLM_CLI_CONCURRENCY", "PRXREF_LLM_CLI_CONCURRENCY"),
+ "PRXREF_LLM_CLI_CONCURRENCY",
+ minimum=1,
+ )
+ from .llm_cli_backends import build_cli_client
+
+ return build_cli_client(
+ backend,
+ models=models,
+ default_timeout=timeout,
+ reasoning_effort=_get("LLM_REASONING_EFFORT", "PRXREF_LLM_REASONING_EFFORT") or None,
+ cli_path=_get("LLM_CLI_PATH", "PRXREF_LLM_CLI_PATH") or "",
+ concurrency=DEFAULT_CLI_CONCURRENCY if concurrency is None else concurrency,
+ )
diff --git a/src/prxref/llm_cli_backends.py b/src/prxref/llm_cli_backends.py
new file mode 100644
index 0000000..28ef6ca
--- /dev/null
+++ b/src/prxref/llm_cli_backends.py
@@ -0,0 +1,827 @@
+"""Subscription CLI backends: ``claude-cli`` and ``kiro-cli``.
+
+These backends run the user's own installed, logged-in CLI as a subprocess
+and walk ``PRXREF_LLM_MODELS`` as a caller-side chain, exactly like the
+openai-compat backend: a model that fails, times out or truncates is advanced
+past at once, a model the CLI names as unknown is skipped for the rest of the
+run, and exhausting the chain raises
+:class:`~prxref.llm_backends.LLMError` with per-model reasons. Every call is
+one process, launched from an argv list (never a shell) in a fresh, empty
+temporary working directory that is removed afterwards, with the user message
+on stdin. A deadline miss kills the whole process group and is spelled
+``: timeout (...)``, so the orchestrator's zero-context retry fires
+exactly as it does for HTTP. A per-client semaphore
+(``PRXREF_LLM_CLI_CONCURRENCY``) caps the processes running at once; the
+wait for a slot does not count against the deadline.
+
+``claude-cli`` runs ``claude -p`` with every built-in tool, settings source,
+MCP server and session file turned off and a single turn allowed, reads the
+``stream-json`` event stream, and loads the system prompt from a file beside
+(not inside) the working directory. The child environment is the parent's
+minus :data:`CLAUDE_ENV_DENYLIST`, a fixed list of credential-routing
+variable NAMES whose values are never read, so the CLI falls back to its own
+subscription login. ``PRXREF_LLM_REASONING_EFFORT`` becomes ``--effort``. The
+call's ``total_cost_usd`` is reported as ``InvokeResult.cost_usd`` with
+``cost_source`` ``"claude-cli"``: it is the CLI's API-equivalent figure at
+list price, not what a subscription is invoiced.
+
+Neither CLI has a temperature, seed or per-call output-token option, so none
+is applied: ``max_tokens`` is accepted and ignored, and each client's
+``temperature`` and ``seed`` attributes are ``None``, which the run record's
+``sampling`` reports truthfully.
+
+``kiro-cli`` runs ``kiro-cli chat --no-interactive`` on the v2 agent engine,
+because the v1 engine does not emit ``stream-json``. That engine does not
+apply a ``--model`` flag, so every attempt writes a working-directory-local agent file,
+``.kiro/agents/prxref-review.json``, that carries the system prompt and the
+chain model and allows no tools, MCP servers or resources. The environment is
+passed through unchanged, and ``PRXREF_LLM_REASONING_EFFORT`` is not applied.
+Kiro reports no token counts and meters credits rather than dollars, so a
+kiro answer counts zero tokens and its ``cost_usd`` is ``None``; the credits
+and the Kiro session id go to the INFO ok line instead.
+
+The module is stdlib-only and is imported lazily by
+:func:`prxref.llm_backends.create_llm_client`, so the HTTP backends never load
+it. The backend names live in ``prxref.llm_backends.CLI_BACKENDS``.
+"""
+from __future__ import annotations
+
+import dataclasses
+import json
+import logging
+import math
+import os
+import shutil
+import signal
+import subprocess
+import tempfile
+import threading
+import time
+from collections.abc import Mapping, Sequence
+from typing import NamedTuple
+
+from .costs import combine_reported, valid_usd
+from .llm import ConfigError, InvokeResult, LLMClient
+from .llm_backends import (
+ _TRUNCATION_FINISH_REASONS,
+ CLI_BACKENDS,
+ DEFAULT_CLI_CONCURRENCY,
+ LLMError,
+ _looks_permanently_unavailable,
+ _mark_unavailable,
+)
+
+logger = logging.getLogger(__name__)
+
+DEFAULT_BINARIES: Mapping[str, str] = {"claude-cli": "claude", "kiro-cli": "kiro-cli"}
+JSON_ONLY_INSTRUCTION = (
+ "\n\nRespond with exactly one JSON object and nothing else: no prose "
+ "before or after it and no markdown code fences."
+)
+CLAUDE_ENV_DENYLIST: tuple[str, ...] = (
+ "ANTHROPIC_API_KEY",
+ "ANTHROPIC_AUTH_TOKEN",
+ "ANTHROPIC_BASE_URL",
+ "ANTHROPIC_PROFILE",
+ "CLAUDE_CODE_USE_BEDROCK",
+ "CLAUDE_CODE_USE_VERTEX",
+ "CLAUDE_CODE_USE_FOUNDRY",
+ "CLAUDE_CODE_SIMPLE",
+)
+# After a SIGKILL the pipes still have to be drained and the child reaped; a
+# process that survives even that is killed directly and abandoned.
+_REAP_TIMEOUT_S = 5.0
+_DETAIL_CHARS = 200
+_UNRECOGNIZED_MODEL_MARKER = "[claude-code:unrecognized_model]"
+_CLAUDE_INPUT_TOKEN_FIELDS = ("input_tokens", "cache_creation_input_tokens", "cache_read_input_tokens")
+KIRO_AGENT_NAME = "prxref-review"
+_KIRO_LIST_MODELS_HINT = " (possibly an unknown model; check kiro-cli chat --list-models)"
+
+
+def _not_a_cli_backend(backend: str) -> ConfigError:
+ return ConfigError(
+ f"PRXREF_LLM_BACKEND: {backend!r} is not a CLI backend; expected one of {', '.join(CLI_BACKENDS)}"
+ )
+
+
+class _Attempt(NamedTuple):
+ """One CLI process's outcome, as the chain loop in :meth:`_CLIClient.invoke` consumes it.
+
+ ``result`` is the parsed answer, or ``None`` when the model failed, in
+ which case ``failure`` is the ``": ..."`` reason. ``unavailable``
+ marks the model as gone for the rest of the run. ``reported`` is the
+ ``(cost_usd, cost_source)`` of a response that came back -- billed even
+ when it is an error -- or ``None`` when nothing was received or the
+ backend never reports a dollar figure (kiro). ``log_extra`` is appended
+ to the INFO ok line.
+ """
+
+ result: InvokeResult | None
+ failure: str = ""
+ unavailable: bool = False
+ reported: tuple[float | None, str] | None = None
+ log_extra: str = ""
+
+
+class _ProcessFailed(Exception):
+ """An ``OSError`` raised while a launched CLI process ran; the process has been killed.
+
+ It keeps a failure after launch apart from a failure to launch, which
+ :meth:`_CLIClient._attempt` reports differently. ``error`` is the original.
+ """
+
+ def __init__(self, error: OSError):
+ super().__init__(str(error))
+ self.error = error
+
+
+def _run_with_deadline(
+ runner, argv: list[str], stdin_text: str, cwd: str, env: dict[str, str], deadline: float
+) -> tuple[int | None, str, str, bool]:
+ """Run one CLI process to completion or for at most ``deadline`` seconds of wall clock.
+
+ Returns ``(returncode, stdout, stderr, timed_out)``. On POSIX the process
+ leads its own session, so a deadline miss kills the whole group rather
+ than just the direct child (a wrapper script would otherwise leave the
+ model process behind). Any other exception while the process runs also
+ kills it before propagating; an ``OSError`` propagates as
+ :class:`_ProcessFailed`, so it is not mistaken for a failed launch.
+ """
+ posix = os.name == "posix"
+ proc = runner(
+ argv,
+ stdin=subprocess.PIPE,
+ stdout=subprocess.PIPE,
+ stderr=subprocess.PIPE,
+ cwd=cwd,
+ env=env,
+ text=True,
+ encoding="utf-8",
+ errors="replace",
+ start_new_session=posix,
+ )
+ try:
+ out, err = proc.communicate(input=stdin_text, timeout=deadline)
+ except subprocess.TimeoutExpired:
+ _kill_tree(proc, posix)
+ try:
+ proc.communicate(timeout=_REAP_TIMEOUT_S)
+ except subprocess.TimeoutExpired:
+ proc.kill()
+ return proc.returncode, "", "", True
+ except OSError as exc:
+ _kill_tree(proc, posix)
+ raise _ProcessFailed(exc) from exc
+ except BaseException:
+ _kill_tree(proc, posix)
+ raise
+ return proc.returncode, out or "", err or "", False
+
+
+def _kill_tree(proc, posix: bool) -> None:
+ """SIGKILL ``proc``'s process group on POSIX; otherwise, or if that fails, the process alone."""
+ if posix:
+ try:
+ os.killpg(proc.pid, signal.SIGKILL)
+ return
+ except (ProcessLookupError, PermissionError):
+ pass
+ try:
+ proc.kill()
+ except OSError:
+ pass
+
+
+def _count(value: object) -> int:
+ """A token count read from CLI JSON: a non-negative int, else 0 (bools and junk included)."""
+ if isinstance(value, bool) or not isinstance(value, int):
+ return 0
+ return max(value, 0)
+
+
+def _one_line(text: str) -> str:
+ return " ".join(text.split())
+
+
+def _with_cost(result: InvokeResult, received: Sequence[tuple[float | None, str]]) -> InvokeResult:
+ """``result`` carrying the folded reported cost of every response one invoke received."""
+ cost_usd, cost_source = combine_reported(received)
+ return dataclasses.replace(result, cost_usd=cost_usd, cost_source=cost_source)
+
+
+class _CLIClient(LLMClient):
+ """The model chain, process, deadline and concurrency handling shared by the CLI backends.
+
+ A subclass sets ``backend_name`` and implements three hooks:
+ :meth:`_prepare` writes the files its CLI reads under the per-attempt
+ temporary root and returns the working directory, :meth:`_argv` builds
+ the argv list, and :meth:`_parse` turns one finished process into an
+ :class:`_Attempt`. It may override :meth:`_child_env`, which passes the
+ parent environment through unchanged by default. The chain, the
+ unavailable-model memory, truncation, the deadline and process-group
+ kill, the concurrency cap and the cost fold all live here, so every CLI
+ behaves identically in the chain.
+ """
+
+ backend_name = ""
+
+ def __init__(
+ self,
+ *,
+ binary: str,
+ models: Sequence[str],
+ default_timeout: float,
+ concurrency: int = DEFAULT_CLI_CONCURRENCY,
+ reasoning_effort: str | None = None,
+ runner=subprocess.Popen,
+ ):
+ if not models:
+ raise ValueError("models must be a non-empty list")
+ if isinstance(concurrency, bool) or not isinstance(concurrency, int) or concurrency < 1:
+ raise ValueError(f"concurrency must be an integer >= 1, got {concurrency!r}")
+ self.binary = binary
+ self.models = list(models)
+ self.default_timeout = default_timeout
+ self.reasoning_effort = reasoning_effort or None
+ self.temperature: float | None = None
+ self.seed: int | None = None
+ self._runner = runner
+ self._slots = threading.BoundedSemaphore(concurrency)
+ self._unavailable: set[str] = set()
+ self._unavailable_lock = threading.Lock()
+ self._warned: set[str] = set()
+ self._warned_lock = threading.Lock()
+
+ def invoke(
+ self,
+ system: str,
+ user: str,
+ *,
+ max_tokens: int = 4096,
+ json_mode: bool = False,
+ timeout_s: float | None = None,
+ ) -> InvokeResult:
+ """Run the CLI once per model until one answers untruncated; fast-fail the rest.
+
+ ``max_tokens`` is accepted for protocol conformance and deliberately
+ not applied: neither CLI takes a per-call output budget, and capping
+ claude through its environment makes it spend recovery turns and end
+ in an error instead. ``json_mode`` appends
+ :data:`JSON_ONLY_INSTRUCTION` to the system prompt; any code fence
+ the model still adds is left for the reviewer's lenient parse.
+ ``timeout_s`` (else ``default_timeout``) is each model's wall-clock
+ deadline. If every model truncates, the last truncated answer is
+ returned rather than raised, as on openai-compat. The result's
+ ``cost_usd`` folds the reported cost of every response received in
+ this call, failed and truncated attempts included, because each was
+ billed; one received response without a figure makes it ``None``.
+ """
+ deadline = self.default_timeout if timeout_s is None else timeout_s
+ sys_text = system + (JSON_ONLY_INSTRUCTION if json_mode else "")
+ failures: list[str] = []
+ received: list[tuple[float | None, str]] = []
+ last_truncated: InvokeResult | None = None
+ total = len(self.models)
+ for attempt, model in enumerate(self.models, start=1):
+ if model in self._unavailable:
+ failures.append(f"{model}: skipped (unavailable)")
+ continue
+ logger.info(
+ "llm attempt %d/%d: backend=%s model=%s deadline=%.0fs",
+ attempt, total, self.backend_name, model, deadline,
+ )
+ outcome = self._attempt(model, sys_text, user, deadline)
+ if outcome.reported is not None:
+ received.append(outcome.reported)
+ result = outcome.result
+ if result is None:
+ logger.warning(
+ "llm attempt %d/%d failed: backend=%s %s",
+ attempt, total, self.backend_name, outcome.failure,
+ )
+ failures.append(outcome.failure)
+ if outcome.unavailable and _mark_unavailable(model, self._unavailable, self._unavailable_lock):
+ logger.warning(
+ "model=%s marked unavailable (%s), skipping for the rest of the run",
+ model, outcome.failure,
+ )
+ continue
+ if result.finish_reason.strip().lower() in _TRUNCATION_FINISH_REASONS:
+ logger.warning(
+ "llm attempt %d/%d truncated: backend=%s model=%s finish_reason=%s after %dms out=%s",
+ attempt, total, self.backend_name, result.model, result.finish_reason,
+ result.elapsed_ms, result.output_tokens,
+ )
+ failures.append(f"{model}: truncated (finish_reason={result.finish_reason})")
+ last_truncated = result
+ continue
+ logger.info(
+ "llm attempt %d/%d ok: backend=%s model=%s %dms in=%s out=%s finish=%s%s",
+ attempt, total, self.backend_name, result.model, result.elapsed_ms,
+ result.input_tokens, result.output_tokens, result.finish_reason or "-", outcome.log_extra,
+ )
+ return _with_cost(result, received)
+ if last_truncated is not None:
+ return _with_cost(last_truncated, received)
+ raise LLMError("all models failed: " + "; ".join(failures))
+
+ def _attempt(self, model: str, sys_text: str, user: str, deadline: float) -> _Attempt:
+ """Run one model's CLI process inside a concurrency slot and a throwaway directory."""
+ waiting_since = time.perf_counter()
+ with self._slots:
+ logger.debug(
+ "%s: waited %dms for a CLI slot",
+ self.backend_name, int((time.perf_counter() - waiting_since) * 1000),
+ )
+ root: str | None = None
+ try:
+ root = tempfile.mkdtemp(prefix=f"prxref-{self.backend_name}-")
+ cwd = self._prepare(root, model, sys_text)
+ argv = self._argv(root, model)
+ t0 = time.perf_counter()
+ rc, out, err, timed_out = _run_with_deadline(
+ self._runner, argv, user, cwd, self._child_env(), deadline
+ )
+ elapsed_ms = int((time.perf_counter() - t0) * 1000)
+ except _ProcessFailed as exc:
+ error = exc.error
+ return _Attempt(None, f"{model}: process failed ({type(error).__name__}: {error})")
+ except OSError as exc:
+ return _Attempt(None, f"{model}: launch failed ({type(exc).__name__}: {exc})")
+ finally:
+ if root is not None:
+ shutil.rmtree(root, ignore_errors=True)
+ if timed_out:
+ return _Attempt(None, f"{model}: timeout (TimeoutExpired after {deadline:.0f}s)")
+ return self._parse(model, rc, out, err, elapsed_ms)
+
+ def _prepare(self, root: str, model: str, sys_text: str) -> str:
+ """Write the files this CLI reads under ``root``; return the working directory."""
+ raise NotImplementedError
+
+ def _argv(self, root: str, model: str) -> list[str]:
+ """The argv list for one call to ``model``; ``argv[0]`` is :attr:`binary`."""
+ raise NotImplementedError
+
+ def _parse(self, model: str, rc: int | None, out: str, err: str, elapsed_ms: int) -> _Attempt:
+ """Turn one finished (not timed-out) process into an :class:`_Attempt`."""
+ raise NotImplementedError
+
+ def _child_env(self, environ: Mapping[str, str] | None = None) -> dict[str, str]:
+ """The child process environment: ``environ`` (default ``os.environ``) unchanged."""
+ return dict(os.environ if environ is None else environ)
+
+ def _warn_once(self, key: str, message: str, *args: object) -> None:
+ """Log ``message`` at WARNING the first time this client sees ``key``."""
+ with self._warned_lock:
+ if key in self._warned:
+ return
+ self._warned.add(key)
+ logger.warning(message, *args)
+
+
+class ClaudeCLIClient(_CLIClient):
+ """``claude-cli``: the user's own logged-in Claude Code CLI, one print-mode process per call.
+
+ The argv is ``claude -p --model --output-format stream-json --verbose
+ --tools "" --setting-sources "" --strict-mcp-config
+ --no-session-persistence --max-turns 1 --system-prompt-file ``,
+ plus ``--effort `` when a reasoning effort is set. The working
+ directory is an empty temporary directory, and the prompt file sits
+ beside it, not in it. The child environment drops
+ :data:`CLAUDE_ENV_DENYLIST` by name.
+
+ The answer is the stream's ``result`` event. ``is_error`` is
+ authoritative (the CLI reports ``subtype "success"`` on some failed
+ calls), and a non-zero exit, a missing result or a non-string result
+ also fail the model. Input tokens include cache creation and cache
+ reads, the model is the one the CLI reports it ran, and the cost is
+ ``total_cost_usd``. A 404, a 4xx naming the model as gone, or the CLI's
+ unrecognized-model marker marks the model unavailable for the run. The
+ ``system/init`` and ``rate_limit_event`` events feed one-time WARNINGs:
+ an ``apiKeySource`` other than none (the call is not on the subscription
+ login), tools or MCP servers loaded despite the flags, and a rate-limit
+ status other than ``allowed``.
+ """
+
+ backend_name = "claude-cli"
+
+ def _prepare(self, root: str, model: str, sys_text: str) -> str:
+ with open(os.path.join(root, "system.md"), "w", encoding="utf-8") as fh:
+ fh.write(sys_text)
+ cwd = os.path.join(root, "cwd")
+ os.mkdir(cwd)
+ return cwd
+
+ def _argv(self, root: str, model: str) -> list[str]:
+ argv = [
+ self.binary, "-p",
+ "--model", model,
+ "--output-format", "stream-json", "--verbose",
+ "--tools", "",
+ "--setting-sources", "",
+ "--strict-mcp-config",
+ "--no-session-persistence",
+ "--max-turns", "1",
+ "--system-prompt-file", os.path.join(root, "system.md"),
+ ]
+ if self.reasoning_effort:
+ argv += ["--effort", self.reasoning_effort]
+ return argv
+
+ def _child_env(self, environ: Mapping[str, str] | None = None) -> dict[str, str]:
+ """The parent environment minus :data:`CLAUDE_ENV_DENYLIST`; a denylisted value is never read."""
+ source = os.environ if environ is None else environ
+ return {key: source[key] for key in source if key not in CLAUDE_ENV_DENYLIST}
+
+ def _parse(self, model: str, rc: int | None, out: str, err: str, elapsed_ms: int) -> _Attempt:
+ init: dict | None = None
+ rate_limit: dict | None = None
+ res: dict | None = None
+ lines = [line for line in (raw.strip() for raw in out.splitlines()) if line]
+ skipped = 0
+ for line in lines:
+ try:
+ event = json.loads(line)
+ except ValueError:
+ event = None
+ if not isinstance(event, dict):
+ skipped += 1
+ continue
+ kind = event.get("type")
+ if kind == "system" and event.get("subtype") == "init":
+ if init is None:
+ init = event
+ elif kind == "rate_limit_event":
+ info = event.get("rate_limit_info")
+ if isinstance(info, dict):
+ rate_limit = info
+ elif kind == "result":
+ res = event
+ if skipped:
+ logger.debug("claude-cli: skipped %d non-event stdout line(s) for model=%s", skipped, model)
+ self._check_init(init)
+ self._check_rate_limit(rate_limit)
+
+ reported: tuple[float | None, str] | None = None
+ if res is not None:
+ cost = valid_usd(res.get("total_cost_usd"))
+ reported = (cost, self.backend_name if cost is not None else "")
+ text = res.get("result") if res is not None else None
+ if rc != 0 or res is None or res.get("is_error") is True or not isinstance(text, str):
+ kind = self._failure_kind(rc, res, rate_limit, unparseable=bool(lines) and skipped == len(lines))
+ if isinstance(text, str) and text.strip():
+ detail = _one_line(text)[:_DETAIL_CHARS]
+ elif err.strip():
+ detail = _one_line(err)[-_DETAIL_CHARS:]
+ else:
+ detail = "(no output)"
+ return _Attempt(
+ None,
+ f"{model}: {kind}: {detail}",
+ unavailable=self._names_model_unavailable(res, err),
+ reported=reported,
+ )
+
+ usage = res.get("usage")
+ if not isinstance(usage, dict):
+ usage = {}
+ finish_reason = res.get("stop_reason")
+ auth = init.get("apiKeySource") if init is not None else None
+ return _Attempt(
+ InvokeResult(
+ text=text,
+ input_tokens=sum(_count(usage.get(field)) for field in _CLAUDE_INPUT_TOKEN_FIELDS),
+ output_tokens=_count(usage.get("output_tokens")),
+ model=self._resolved_model(model, init, res),
+ backend=self.backend_name,
+ elapsed_ms=elapsed_ms,
+ finish_reason=finish_reason if isinstance(finish_reason, str) else "",
+ ),
+ reported=reported,
+ log_extra=f" auth={'-' if auth is None else auth}",
+ )
+
+ @staticmethod
+ def _failure_kind(rc: int | None, res: dict | None, rate_limit: dict | None, *, unparseable: bool) -> str:
+ """Name why a finished claude process did not answer, most specific first."""
+ if unparseable:
+ return "unparseable output"
+ if res is not None:
+ reason = res.get("terminal_reason")
+ if isinstance(reason, str) and reason and reason != "completed":
+ return reason
+ if rate_limit is not None and rate_limit.get("status") == "rejected":
+ return "rate limited"
+ if rc != 0:
+ return f"exit {rc}"
+ if res is None:
+ return "no result event"
+ return "error" if res.get("is_error") is True else "no result text"
+
+ @staticmethod
+ def _names_model_unavailable(res: dict | None, err: str) -> bool:
+ """True when the CLI says the model itself is gone (404, 4xx naming it, or its stderr marker)."""
+ if _UNRECOGNIZED_MODEL_MARKER in err:
+ return True
+ if res is None:
+ return False
+ status = res.get("api_error_status")
+ if isinstance(status, bool) or not isinstance(status, int):
+ return False
+ if status == 404:
+ return True
+ text = res.get("result")
+ return 400 <= status < 500 and isinstance(text, str) and _looks_permanently_unavailable(text)
+
+ @staticmethod
+ def _resolved_model(requested: str, init: dict | None, res: dict) -> str:
+ """The model the CLI ran: its ``modelUsage`` key (most output wins), else init's, else the alias."""
+ model_usage = res.get("modelUsage")
+ if isinstance(model_usage, dict) and model_usage:
+
+ def output_tokens(name: str) -> int:
+ entry = model_usage[name]
+ return _count(entry.get("outputTokens")) if isinstance(entry, dict) else 0
+
+ return str(max(model_usage, key=output_tokens))
+ reported = init.get("model") if init is not None else None
+ return reported if isinstance(reported, str) and reported else requested
+
+ def _check_init(self, init: dict | None) -> None:
+ if init is None:
+ return
+ source = init.get("apiKeySource")
+ if source not in (None, "none"):
+ self._warn_once(
+ "apiKeySource",
+ "claude-cli: the CLI reports apiKeySource=%s, so this call is NOT on your subscription "
+ "login (check managed settings / apiKeyHelper)",
+ source,
+ )
+ tools, servers = init.get("tools"), init.get("mcp_servers")
+ loaded = (
+ len(tools) if isinstance(tools, list) else 0,
+ len(servers) if isinstance(servers, list) else 0,
+ )
+ if any(loaded):
+ self._warn_once(
+ "tools",
+ "claude-cli: the CLI loaded tools/MCP servers despite --tools '' --strict-mcp-config "
+ "(%d/%d); the CLI's flags may have changed",
+ *loaded,
+ )
+
+ def _check_rate_limit(self, rate_limit: dict | None) -> None:
+ if rate_limit is None:
+ return
+ status = rate_limit.get("status")
+ if status in (None, "allowed"):
+ return
+ self._warn_once(
+ f"rate_limit:{status}",
+ "claude-cli: subscription rate limit status=%s type=%s utilization=%s",
+ status, rate_limit.get("rateLimitType"), rate_limit.get("utilization"),
+ )
+
+
+class KiroCLIClient(_CLIClient):
+ """``kiro-cli``: the user's own logged-in Kiro CLI, one headless chat process per call.
+
+ The argv is ``kiro-cli chat --no-interactive --agent prxref-review
+ --output-format stream-json --trust-tools= --agent-engine v2``. The v1
+ engine does not emit ``stream-json``, and v2 does not apply a ``--model``
+ flag, so each attempt writes ``.kiro/agents/prxref-review.json`` into
+ its temporary working directory: the system prompt, the chain model, and
+ no tools, allowed tools, MCP servers or resources. The environment is the
+ parent's, unchanged, and a reasoning effort is not applied.
+
+ The answer is the ``runFinished`` event's ``finalText``, or the joined
+ ``agent_message_chunk`` texts when Kiro marks that text truncated.
+ Success needs exit 0, no ``runError``, status ``success`` and a
+ non-empty answer. Kiro reports no tokens and meters credits, not
+ dollars, so the result counts zero tokens, names the requested model
+ (Kiro does not echo it) and has ``cost_usd`` ``None``; the summed
+ ``credit`` metering and the session id go to the INFO ok line. A
+ ``runError`` fails the model as `` error: ``, with a
+ ``--list-models`` hint at the ``prompt`` stage, where an unknown model
+ fails. No kiro failure marks a model unavailable, because none names the
+ model.
+ """
+
+ backend_name = "kiro-cli"
+
+ def _prepare(self, root: str, model: str, sys_text: str) -> str:
+ agents = os.path.join(root, ".kiro", "agents")
+ os.makedirs(agents)
+ agent = {
+ "name": KIRO_AGENT_NAME,
+ "description": "prxref single-shot reviewer: no tools, no MCP, no resources",
+ "prompt": sys_text,
+ "tools": [],
+ "allowedTools": [],
+ "mcpServers": {},
+ "includeMcpJson": False,
+ "resources": [],
+ "model": model,
+ }
+ with open(os.path.join(agents, f"{KIRO_AGENT_NAME}.json"), "w", encoding="utf-8") as fh:
+ json.dump(agent, fh)
+ return root
+
+ def _argv(self, root: str, model: str) -> list[str]:
+ return [
+ self.binary, "chat", "--no-interactive",
+ "--agent", KIRO_AGENT_NAME,
+ "--output-format", "stream-json",
+ "--trust-tools=",
+ "--agent-engine", "v2",
+ ]
+
+ def _parse(self, model: str, rc: int | None, out: str, err: str, elapsed_ms: int) -> _Attempt:
+ chunks: list[str] = []
+ finished: dict | None = None
+ run_error: dict | None = None
+ session = ""
+ credits: float | None = None
+ lines = [line for line in (raw.strip() for raw in out.splitlines()) if line]
+ skipped = 0
+ for line in lines:
+ try:
+ event = json.loads(line)
+ except ValueError:
+ event = None
+ if not isinstance(event, dict):
+ skipped += 1
+ continue
+ data = event.get("data")
+ if not isinstance(data, dict):
+ continue
+ if not session and isinstance(data.get("sessionId"), str):
+ session = data["sessionId"]
+ kind = event.get("type")
+ if kind == "metadata":
+ metered = _kiro_credits(data.get("meteringUsage"))
+ if metered is not None:
+ credits = metered if credits is None else credits + metered
+ elif kind == "sessionUpdate":
+ update = data.get("update")
+ if isinstance(update, dict) and update.get("sessionUpdate") == "agent_message_chunk":
+ content = update.get("content")
+ if isinstance(content, dict) and isinstance(content.get("text"), str):
+ chunks.append(content["text"])
+ elif kind == "runFinished":
+ finished = data
+ elif kind == "runError":
+ run_error = data
+ if skipped:
+ logger.debug("kiro-cli: skipped %d non-event stdout line(s) for model=%s", skipped, model)
+
+ final = finished.get("finalText") if finished is not None else None
+ if isinstance(final, str) and final.strip() and finished.get("finalTextTruncated") is not True:
+ text = final
+ else:
+ text = "".join(chunks)
+ status = finished.get("status") if finished is not None else None
+ if rc != 0 or run_error is not None or status != "success" or not text.strip():
+ if run_error is not None:
+ return _Attempt(None, self._run_error_reason(model, run_error))
+ unparseable = bool(lines) and skipped == len(lines)
+ if unparseable:
+ kind = "unparseable output"
+ elif rc != 0:
+ kind = f"exit {rc}"
+ elif finished is None:
+ kind = "no runFinished event"
+ elif status != "success":
+ kind = f"run status {status!r}"
+ else:
+ kind = "empty answer"
+ detail = _one_line(err)[-_DETAIL_CHARS:] if err.strip() else "(no output)"
+ return _Attempt(None, f"{model}: {kind}: {detail}")
+
+ stop_reason = finished.get("stopReason")
+ return _Attempt(
+ InvokeResult(
+ text=text,
+ input_tokens=0,
+ output_tokens=0,
+ model=model,
+ backend=self.backend_name,
+ elapsed_ms=elapsed_ms,
+ finish_reason=stop_reason if isinstance(stop_reason, str) else "",
+ ),
+ log_extra=(
+ f" credits={'-' if credits is None else format(credits, '.4f')}"
+ f" session={session or '-'}"
+ ),
+ )
+
+ @staticmethod
+ def _run_error_reason(model: str, run_error: dict) -> str:
+ """``: error: ``, plus the list-models hint at the prompt stage."""
+ stage = run_error.get("stage")
+ stage = stage if isinstance(stage, str) and stage.strip() else "run"
+ message = run_error.get("message")
+ detail = _one_line(message)[:_DETAIL_CHARS] if isinstance(message, str) and message.strip() else "(no message)"
+ hint = _KIRO_LIST_MODELS_HINT if stage == "prompt" else ""
+ return f"{model}: {stage} error: {detail}{hint}"
+
+
+def _kiro_credits(metering: object) -> float | None:
+ """The summed ``credit`` values of one Kiro ``meteringUsage`` list, or ``None`` if it has none."""
+ if not isinstance(metering, list):
+ return None
+ total: float | None = None
+ for entry in metering:
+ if not isinstance(entry, dict) or entry.get("unit") != "credit":
+ continue
+ value = entry.get("value")
+ if isinstance(value, bool) or not isinstance(value, (int, float)) or not math.isfinite(value):
+ continue
+ total = value if total is None else total + value
+ return total
+
+
+def resolve_cli_binary(backend: str, cli_path: str, *, which=shutil.which) -> str:
+ """Resolve the CLI binary for ``backend`` to an absolute executable path.
+
+ ``cli_path`` is ``PRXREF_LLM_CLI_PATH``: empty (or whitespace) means the
+ backend's default binary name (:data:`DEFAULT_BINARIES`) on ``PATH``;
+ otherwise it is ``~``-expanded and looked up through ``which``, so a bare
+ name searches ``PATH`` and a path must be an executable file. The result
+ is made absolute, because every call runs in a temporary working
+ directory. A binary that cannot be found raises
+ :class:`~prxref.llm.ConfigError` naming ``PRXREF_LLM_CLI_PATH`` when an
+ override was given and ``PRXREF_LLM_BACKEND`` otherwise, so a missing CLI
+ exits 2 before any network call. An unknown ``backend`` raises one naming
+ ``PRXREF_LLM_BACKEND``.
+ """
+ if backend not in CLI_BACKENDS:
+ raise _not_a_cli_backend(backend)
+ default = DEFAULT_BINARIES[backend]
+ override = (cli_path or "").strip()
+ name = os.path.expanduser(override) if override else default
+ found = which(name)
+ if not found:
+ if override:
+ raise ConfigError(
+ f"PRXREF_LLM_CLI_PATH: {name!r} is not an executable file (PRXREF_LLM_BACKEND={backend})"
+ )
+ raise ConfigError(
+ f"PRXREF_LLM_BACKEND: {backend} needs the {default!r} CLI, which was not found on PATH; "
+ f"install it and log in, or set PRXREF_LLM_CLI_PATH to its absolute path"
+ )
+ return os.path.abspath(found)
+
+
+def build_cli_client(
+ backend: str,
+ *,
+ models: Sequence[str],
+ default_timeout: float,
+ reasoning_effort: str | None,
+ cli_path: str,
+ concurrency: int,
+ which=shutil.which,
+ runner=subprocess.Popen,
+) -> LLMClient:
+ """Build the client for ``backend`` (one of ``llm_backends.CLI_BACKENDS``).
+
+ ``models`` is the chain walked in order; ``default_timeout`` is the
+ per-model deadline in seconds; ``reasoning_effort`` feeds claude's
+ ``--effort`` and is ignored by kiro; ``cli_path`` is passed to
+ :func:`resolve_cli_binary`, which runs first, so a missing CLI is a
+ :class:`~prxref.llm.ConfigError` before any client exists; ``concurrency``
+ caps the CLI processes this client runs at once and must be an integer
+ >= 1 (a ``ConfigError`` naming ``PRXREF_LLM_CLI_CONCURRENCY`` otherwise).
+ ``which`` and ``runner`` are the binary lookup and the process launcher,
+ injectable for tests. Building never starts a process.
+
+ ``claude-cli`` returns a :class:`ClaudeCLIClient` and ``kiro-cli`` a
+ :class:`KiroCLIClient`; a reasoning effort set for ``kiro-cli`` is
+ dropped with one INFO line saying so.
+ """
+ if backend not in CLI_BACKENDS:
+ raise _not_a_cli_backend(backend)
+ if isinstance(concurrency, bool) or not isinstance(concurrency, int) or concurrency < 1:
+ raise ConfigError(f"PRXREF_LLM_CLI_CONCURRENCY: must be an integer at least 1, got {concurrency!r}")
+ binary = resolve_cli_binary(backend, cli_path, which=which)
+ if backend == "kiro-cli":
+ if reasoning_effort:
+ logger.info("PRXREF_LLM_REASONING_EFFORT is not applied by kiro-cli")
+ return KiroCLIClient(
+ binary=binary,
+ models=models,
+ default_timeout=default_timeout,
+ concurrency=concurrency,
+ runner=runner,
+ )
+ return ClaudeCLIClient(
+ binary=binary,
+ models=models,
+ default_timeout=default_timeout,
+ concurrency=concurrency,
+ reasoning_effort=reasoning_effort,
+ runner=runner,
+ )
diff --git a/src/prxref/markers.py b/src/prxref/markers.py
new file mode 100644
index 0000000..8778f8e
--- /dev/null
+++ b/src/prxref/markers.py
@@ -0,0 +1,72 @@
+"""The one table of finding glyphs every rendering surface draws from.
+
+A severity maps to exactly one glyph, and that glyph is the same in the
+summary counts line, the summary findings list, an inline comment header and
+the CLI. A finding outside the ticket's scope keeps its severity glyph and
+gains the out-of-ticket prefix in front of it; that prefix is never a severity
+glyph itself, so scope and severity stay readable independently.
+
+The summary templates (``prompts/summary.md`` and the two fallback templates)
+keep their glyphs as literals because they are the readable source of the
+layout; a parity test holds each of them to this table. Changing a glyph or
+adding a severity is one edit here.
+"""
+from __future__ import annotations
+
+from collections.abc import Mapping
+from types import MappingProxyType
+
+from .triage import SCOPE_OUT, Finding
+
+SEVERITY_MARKERS: Mapping[str, str] = MappingProxyType({
+ "error": "🟥",
+ "warning": "🟧",
+ "spec": "🔍",
+ "outofscope": "⬜",
+})
+
+# An unrecognised severity renders as the minor class, matching the way the
+# library formatter folds an unknown severity into ``outofscope``.
+FALLBACK_MARKER: str = SEVERITY_MARKERS["outofscope"]
+
+OUT_OF_TICKET_MARKER: str = "🟦"
+
+SCOPE_LABELS: Mapping[str, str] = MappingProxyType({SCOPE_OUT: "OUTSIDE TICKET"})
+
+
+def severity_marker(severity: str) -> str:
+ """Return the glyph for ``severity``, or :data:`FALLBACK_MARKER` if unknown.
+
+ The lookup is exact: severities are normalised by the quality gate before
+ anything renders them.
+ """
+ return SEVERITY_MARKERS.get(severity, FALLBACK_MARKER)
+
+
+def marker_for(severity: str, scope: str) -> str:
+ """Return the full marker for a finding: its severity glyph, prefixed by
+ :data:`OUT_OF_TICKET_MARKER` and a space when ``scope`` is ``"out"``.
+
+ Scope ``"in"`` and ``"unknown"`` add nothing, so a run without a ticket
+ renders exactly the severity glyph.
+ """
+ marker = severity_marker(severity)
+ return f"{OUT_OF_TICKET_MARKER} {marker}" if scope == SCOPE_OUT else marker
+
+
+def inline_header(f: Finding) -> str:
+ """Return the first line of the pipeline's inline comment for ``f``.
+
+ The shape is ``🤖 **[] ** (`:`)``,
+ with the marker from :func:`marker_for` and ``:`` omitted for a
+ file-level finding. A finding outside the ticket also carries its
+ :data:`SCOPE_LABELS` entry inside the brackets, as
+ ``[WARNING · OUTSIDE TICKET]``; scope ``"in"`` and ``"unknown"`` render
+ exactly the severity header.
+ """
+ label = f.severity.upper()
+ scope_label = SCOPE_LABELS.get(f.scope)
+ if scope_label:
+ label = f"{label} · {scope_label}"
+ loc = f"{f.file}:{f.line}" if f.line > 0 else f.file
+ return f"🤖 {marker_for(f.severity, f.scope)} **[{label}] {f.title}** (`{loc}`)"
diff --git a/src/prxref/orchestrator.py b/src/prxref/orchestrator.py
index ba3454b..bb9b0ed 100644
--- a/src/prxref/orchestrator.py
+++ b/src/prxref/orchestrator.py
@@ -22,7 +22,16 @@
``context_lines=0`` rendering — a strictly smaller prompt attacks the
prefill-side share of the wall clock, and a truncated completion (the
response-side budget) is not a timeout and never reaches this retry.
-4. Systemic sweep: after the chunk workers, ONE more worker-style
+4. Spec grounding (best-effort, only when ``spec_sources`` is non-empty):
+ ``specs.fetch_specs`` + ``specs.build_spec_digest`` run inside the same
+ never-raise fence as every other stage, and the digest rides the
+ existing chunk calls and the systemic sweep — no extra LLM unit. The
+ run is grounded only when the digest holds at least one constraint
+ (``specs.constraint_count`` above 0); otherwise no digest is injected
+ and the prompts show their no-specs text. A run whose every source
+ failed behaves exactly like a run with no specs, plus a grounding note
+ in the summary.
+5. Systemic sweep: after the chunk workers, ONE more worker-style
single-shot call over the whole-PR digest built by
``systemic.build_digest`` (every file with hunk headers; short files and
migrations render their full added content, the rest only the
@@ -31,8 +40,9 @@
chunk results, and counts as one more review unit: ``chunk_count`` is
``len(chunks) + 1`` whenever the sweep ran, and a sweep failure is one
failed chunk in the partial-review banner.
-5. Deterministic checks and quality passes, in exactly this order — the
- raw chunk + sweep findings first gain
+6. Deterministic checks and quality passes, in exactly this order — the
+ raw chunk + sweep findings have their ``scope`` held to ``unknown``
+ unless a ticket is active (``_enforce_scope``), then gain
``heuristics.release_shape_findings(files)`` (a pure, no-LLM finding
about a PR that is ≥80% release machinery yet also touches source),
folded in BEFORE the passes so it is filtered like any other finding,
@@ -41,8 +51,14 @@
``apply_sweep_dedup`` and can never be dropped as a duplicate of a
chunk worker's own restatement:
- ``apply_location_validation`` (a ``file`` naming no path of the parsed
- diff is dropped, not rendered) → ``apply_manifest_claim_check`` (a
+ ``apply_severity_map`` (only when the team review rules declare a
+ severity map: a team word such as ``blocker`` becomes the prxref tier it
+ maps to; drops nothing) → ``apply_spec_grounding`` (on an ungrounded
+ run every ``spec`` finding, the sweep's included, is relabelled
+ ``warning``, counted by a ``specs relabel`` trace event; drops nothing)
+ → ``apply_location_validation`` (a ``file``
+ naming no path of the parsed diff is dropped, not rendered) →
+ ``apply_manifest_claim_check`` (a
``package.json`` claim whose dependency is not the key on the anchored
line, or whose asserted section disagrees with the actual one; it must
precede line align, which is what makes it read the model's RAW
@@ -56,7 +72,8 @@
title counts toward its group) → ``apply_removal_claim_check`` (a
claim that a NAMED path was removed when the post-image still carries
it) → ``apply_hedge_gate`` (a finding whose own text conditions the
- defect on something the worker never established) →
+ defect on something the worker never established; a ``Spec:`` quote
+ of the injected digest is not read as the finding's own text) →
``apply_quality_gate(confidence_floor=, max_errors=)``, which returns
its findings in content order, so the chunk/sweep boundary is
re-derived here from finding identity rather than carried across the
@@ -73,8 +90,12 @@
``drop_reason`` set, never silently discarded, and both lists come out
sorted by ``finding_sort_key``. Every result — including an error or
summary-only exit — carries a ``sampling`` record naming the
- temperature, seed, and model chain actually in force.
-6. Verdict: ``"Error"`` when every CHUNK review failed (a sweep success
+ temperature, seed, and model chain actually in force, and the run-record
+ keys that :func:`_run_record` stamps on every exit (``cost_usd``,
+ ``cost_estimated``, ``review_rules``, ``ticket_context``,
+ ``spec_grounding``, ``size_advisory``; ``replay`` on replays only, and
+ ``cost_api_equivalent`` on claude-cli-priced runs only).
+7. Verdict: ``"Error"`` when every CHUNK review failed (a sweep success
on a dead worker pool cannot carry the run); ``"Request-Changes"``
iff any active error-severity finding survives;
else ``"Approved"``. A partial failure keeps the verdict but the summary
@@ -83,9 +104,10 @@
blockquote) — a partial review reads as a successful one, so a failure
left only in the logs reaches nobody, and a file list left out of it
leaves the operator guessing which files went unreviewed.
-7. Post: summary rendered from ``reviewer.load_prompt("summary")`` with
+8. Post: summary rendered from ``reviewer.load_prompt("summary")`` with
placeholders ``{verdict} {title} {file_count} {error_count}
- {warning_count} {outofscope_count} {findings} {attribution}`` filled, plus
+ {warning_count} {spec_count} {spec_note} {ticket_note}
+ {outofscope_count} {findings} {attribution}`` filled, plus
inline comments for up to ``max_inline_comments`` active findings.
``post_mode`` narrows what is written: ``"summary+inline"`` (default) is
that full behaviour, ``"summary"`` skips the inline batch, ``"inline"``
@@ -105,16 +127,19 @@
"""
from __future__ import annotations
+import hashlib
import logging
import re
import threading
import time
from collections import Counter
-from collections.abc import Sequence
+from collections.abc import Mapping, Sequence
from concurrent.futures import ThreadPoolExecutor
+from dataclasses import replace
from typing import Any
+from urllib.parse import urlparse
-from . import chunk_context, heuristics, reviewer, systemic
+from . import chunk_context, costs, heuristics, reviewer, specs, systemic
from .forges.base import (
ATTRIBUTION_MARKER,
Forge,
@@ -124,6 +149,7 @@
Thread,
)
from .llm import LLMClient
+from .markers import OUT_OF_TICKET_MARKER, SEVERITY_MARKERS, inline_header, marker_for
from .quality import (
active,
apply_containment_note,
@@ -135,19 +161,27 @@
apply_removal_claim_check,
apply_settled_thread_suppression,
apply_severity_consistency,
+ apply_severity_map,
+ apply_spec_grounding,
apply_sweep_dedup,
apply_thread_dedup,
finding_rank_key,
finding_sort_key,
)
+from .reviewer import NO_PROMPT_CONTEXT, PromptContext, fill_template
from .trace import Tracer, get_tracer
from .triage import (
DEFAULT_CONTEXT_LINES,
DEFAULT_MAX_FILES_PER_CHUNK,
DEFAULT_TOKEN_BUDGET,
+ SCOPE_IN,
+ SCOPE_OUT,
+ SCOPE_UNKNOWN,
Finding,
added_lines_by_file,
build_chunks,
+ count_size_relevant_changes,
+ normalize_scope,
parse_unified_diff,
)
@@ -169,12 +203,17 @@
# letting a pathological run bury the findings under its own diagnostics.
MAX_REPORTED_REASONS = 3
-_SEVERITY_MARKERS = {"error": "🟥", "warning": "🟧", "outofscope": "🟦"}
-
# Inline-comment priority: the most severe findings get the anchor first, so
# a cap or a rejected anchor costs the run its least-important comments
-# rather than whatever happened to sit at the tail of chunk order.
-_SEVERITY_RANK = {"error": 0, "warning": 1, "outofscope": 2}
+# rather than whatever happened to sit at the tail of chunk order. spec sits
+# below warning (a spec violation is an operator-requested contract breach,
+# but not claimed to break at runtime) and above outofscope.
+_SEVERITY_RANK = {"error": 0, "warning": 1, "spec": 2, "outofscope": 3}
+
+# The tie-break after severity: within one severity, a finding outside the
+# ticket yields the inline slots to in-ticket and unjudged ones. With no active
+# ticket every scope is unknown, so the ordering is exactly the severity one.
+_SCOPE_RANK = {SCOPE_IN: 0, SCOPE_UNKNOWN: 0, SCOPE_OUT: 1}
_REDACTED = "[redacted]"
@@ -286,7 +325,9 @@ def redact_for_post(reason: str) -> str:
"🤖 **prxref review — {verdict}**\n\n"
"PR: {title}\n\n"
"Files reviewed: {file_count} · 🟥 {error_count} error · "
- "🟧 {warning_count} warning · 🟦 {outofscope_count} outofscope\n\n"
+ "🟧 {warning_count} warning · 🔍 {spec_count} spec · "
+ "⬜ {outofscope_count} outofscope\n"
+ "{spec_note}{ticket_note}\n"
"{findings}\n\n{attribution}"
)
@@ -326,12 +367,39 @@ def orchestrate_review(
post_verdict: bool = True,
trace_file: str | None = None,
trace_dir: str | None = None,
+ spec_sources: Sequence[str] = (),
+ spec_max_chars: int = 120000,
+ spec_digest_tokens: int = 3000,
+ jira_base_url: str = "",
+ jira_email: str = "",
+ jira_api_token: str = "",
+ rules: Any = None,
+ ticket: Any = None,
+ price_table: Mapping[str, Any] | None = None,
+ post_cost: bool = False,
+ size_warn_lines: int | None = None,
+ size_warn_files: int | None = None,
+ size_ignore_globs: Sequence[str] = (),
+ replay: Mapping[str, Any] | None = None,
) -> dict:
"""Run one full review pass over a PR and optionally post results.
Returns ``{verdict, findings_active, findings_dropped, chunk_count,
chunks_reviewed, chunks_failed, elapsed_ms, input_tokens, output_tokens,
- posted}``. Never raises on ANY stage failure — forge, diff parsing,
+ posted, sampling, cost_usd, cost_estimated, review_rules, ticket_context,
+ spec_grounding, size_advisory}``, plus ``replay`` on a replay run only.
+ Every exit, error and empty-diff exits included, goes through
+ :func:`_run_record`, so the last six keys are always present and are
+ ``None`` (``cost_usd``: ``0.0`` before any LLM request; ``cost_estimated``:
+ ``False``) when their feature is off or the run never reached it.
+ ``cost_usd`` is ``None`` when the cost is unknown, never ``0``.
+ ``cost_api_equivalent`` (always ``True``) is added only when every
+ reported unit cost came from claude-cli
+ (:func:`prxref.costs.api_equivalent_run`), so the CLI's ``-v`` line can
+ label the figure; ``--format json`` never emits it, because each unit's
+ ``cost_source`` (in the ``PRXREF_TRACE_DIR`` meta files) is already the
+ machine-readable label.
+ Never raises on ANY stage failure — forge, diff parsing,
chunking, or LLM — the run degrades to verdict ``"Error"`` with a posted
notice when ``post`` is true. Degenerate arguments are part of that: a
caller passing ``max_chunks=0`` gets an error run, not a ``ValueError``.
@@ -370,37 +438,129 @@ def orchestrate_review(
(model, token counts, elapsed, error) under that directory, labelled
``chunk0`` … ``chunkN-1`` and ``sweep``. Empty (the default) traces
nothing; a write failure is a logged warning, never a review failure.
+
+ ``spec_sources`` grounds the review against written specs: each entry is
+ fetched by :func:`prxref.specs.fetch_specs` and the pruned constraint
+ digest (:func:`prxref.specs.build_spec_digest`) is injected into every
+ worker prompt and the sweep prompt — no extra LLM unit. The fetch never
+ raises and never fails the run: sources that fail become a grounding
+ note in the summary (failure reasons pass through
+ :func:`redact_for_post` before posting), and a run whose every source
+ failed is exactly a run with no specs plus that note. A digest holding
+ no constraint (:func:`prxref.specs.constraint_count` is 0) is not
+ injected, and on such a run, with or without sources, every
+ model-emitted ``spec`` finding is relabelled ``warning``
+ (:func:`quality.apply_spec_grounding`); when any is, one INFO line and
+ one ``specs relabel`` trace event count them. The remaining
+ spec keywords mirror the config keys of the same names
+ (``spec_max_chars``, ``spec_digest_tokens``, ``jira_base_url``,
+ ``jira_email``, ``jira_api_token``); the defaults restate
+ ``config._DEFAULTS`` the way ``MAX_WORKERS`` does. ``spec_sources`` is
+ deliberately absent from the returned dict, like every other request
+ knob.
+
+ ``rules`` and ``ticket`` are the loaded review-rules and ticket-context
+ objects (``rules.ReviewRules`` / ``ticket.TicketContext``), duck-typed so
+ this module never imports theirs; ``None`` turns each off, and with both
+ off the prompts, posts, record and trace are exactly a run without them.
+ Their ``record()`` fills the ``review_rules`` / ``ticket_context`` keys on
+ every exit, and is the meta of one ``rules ok`` / ``ticket ok`` trace
+ event. The rules' ``prompt_block("worker")`` / ``("sweep")`` reach every
+ chunk and the sweep through one :class:`reviewer.PromptContext`, and
+ their ``severity_map`` goes to :func:`quality.apply_severity_map` ahead of
+ every quality pass (a ``rules remap`` event counts the rewrites). An
+ ACTIVE ticket (``ticket.active``) adds its ``scope_block()`` and
+ ``prompt_block()`` to every unit, which is what lets a finding carry a
+ ``scope`` of ``in`` or ``out``; otherwise every scope is forced to
+ ``unknown`` (:func:`_enforce_scope`). A configured ticket's ``note()``
+ rides the summary after the spec note, on the main and summary-only
+ posts but never the error notice, and an active ticket's scope counts
+ ride the ``run ok`` event.
+
+ ``price_table`` is the parsed ``PRXREF_PRICE_TABLE``
+ (:func:`prxref.costs.parse_price_table`); ``None`` or ``{}`` estimates
+ nothing. It is not read from the environment here, because parsing can
+ raise ``ConfigError`` and this function must not raise. ``post_cost``
+ appends the run's cost label (:func:`prxref.costs.cost_label`) as the
+ last field of the summary and error-notice attribution; off, both are
+ byte-identical to a run without it.
+
+ ``size_warn_lines`` / ``size_warn_files`` are the PR-size advisory
+ thresholds (``None`` = off; ``0`` is a legal threshold) and
+ ``size_ignore_globs`` the operator's extra ignore patterns. When either
+ threshold is set the advisory's stats ride the result under
+ ``size_advisory``, and a triggered advisory is prepended to every posted
+ summary. It never touches the verdict.
+
+ ``replay`` is the evaluation-replay stamp built by the CLI
+ (``{base_sha, head_sha, threads, diff_file}``). When given it is copied
+ into the returned dict under ``replay`` and into the ``run start`` trace
+ event, the one request knob that is echoed back, so a replay can never
+ be read as a live review. It changes nothing about how the review runs:
+ pinning and thread hiding live in the forge the caller passes.
"""
t0 = time.perf_counter()
tracer = get_tracer(trace_file)
sampling = _sampling(llm)
+ # The per-run record every exit is stamped with (_run_record). Built here
+ # with every always-present key at its "off / not reached" value; each
+ # stage assigns its own key as the run proceeds, so the value a return
+ # carries is the one in force at that exit.
+ run_inputs: dict[str, Any] = {
+ "cost_usd": 0.0,
+ "cost_estimated": False,
+ "cost_api_equivalent": False,
+ "review_rules": None,
+ "ticket_context": None,
+ "spec_grounding": None,
+ "size_advisory": None,
+ "replay": dict(replay) if replay is not None else None,
+ }
+ # Resolved once, before the first exit, so every exit records them and the
+ # empty-diff summary gets the ticket note. An inactive (empty) ticket is
+ # still recorded and still noted; it just asks the model for no scope.
+ if rules is not None:
+ run_inputs["review_rules"] = rules.record()
+ if ticket is not None:
+ run_inputs["ticket_context"] = ticket.record()
+ ticket_active = ticket is not None and bool(ticket.active)
+ ticket_note = ticket.note() if ticket is not None else ""
+ if ticket_note and not ticket_note.endswith("\n"):
+ ticket_note += "\n"
tracer.event(
"run", "start", forge=ref.forge, url=ref.url, number=ref.number,
sampling=sampling,
+ **({"replay": dict(replay)} if replay is not None else {}),
)
+ if run_inputs["review_rules"] is not None:
+ tracer.event("rules", "ok", **run_inputs["review_rules"])
+ if run_inputs["ticket_context"] is not None:
+ tracer.event("ticket", "ok", **run_inputs["ticket_context"])
try:
with tracer.span("forge.get_pr"):
pr = forge.get_pr(ref)
except Exception as e: # noqa: BLE001
logger.error("get_pr failed: %s", e)
- tracer.event("run", "fail")
- return _error_run(
+ tracer.event("run", "fail", **_cost_meta(run_inputs))
+ return _run_record(_error_run(
forge, ref, post, 0, f"get_pr failed: {e}", t0,
post_mode=post_mode, tracer=tracer, sampling=sampling,
- )
+ cost_label=_cost_label(run_inputs, post_cost),
+ ), run_inputs)
try:
with tracer.span("forge.get_diff") as sp:
raw = forge.get_diff(ref)
- sp["bytes"] = len(raw)
+ sp["bytes"] = len(raw.encode("utf-8"))
except Exception as e: # noqa: BLE001
logger.error("get_diff failed: %s", e)
- tracer.event("run", "fail")
- return _error_run(
+ tracer.event("run", "fail", **_cost_meta(run_inputs))
+ return _run_record(_error_run(
forge, ref, post, 0, f"get_diff failed: {e}", t0,
post_mode=post_mode, tracer=tracer, sampling=sampling,
- )
+ cost_label=_cost_label(run_inputs, post_cost),
+ ), run_inputs)
# Wrapped like every neighbouring stage. These two were the only ones that
# could raise out of orchestrate_review, which made the never-raise contract
@@ -414,11 +574,26 @@ def orchestrate_review(
sp["files"] = len(files)
except Exception as e: # noqa: BLE001
logger.error("parse_unified_diff failed: %s", e)
- tracer.event("run", "fail")
- return _error_run(
+ tracer.event("run", "fail", **_cost_meta(run_inputs))
+ return _run_record(_error_run(
forge, ref, post, 0, f"parse_unified_diff failed: {e}", t0,
post_mode=post_mode, tracer=tracer, sampling=sampling,
+ cost_label=_cost_label(run_inputs, post_cost),
+ ), run_inputs)
+
+ # Sized once, from the parsed files (never the raw diff), so every later
+ # exit carries the same stats and the size line can reach all three
+ # summary renders. Advisory only: a failure here is logged and the review
+ # goes on without it.
+ try:
+ run_inputs["size_advisory"] = _size_advisory(
+ files, lines_limit=size_warn_lines, files_limit=size_warn_files,
+ ignore_globs=size_ignore_globs,
)
+ except Exception as e: # noqa: BLE001
+ logger.warning("size advisory failed (continuing without it): %s", e)
+ run_inputs["size_advisory"] = None
+ size_advisory_line = _size_advisory_line(run_inputs["size_advisory"])
try:
with tracer.span("build_chunks") as sp:
@@ -429,11 +604,12 @@ def orchestrate_review(
sp["chunks"] = len(chunks)
except Exception as e: # noqa: BLE001
logger.error("build_chunks failed: %s", e)
- tracer.event("run", "fail")
- return _error_run(
+ tracer.event("run", "fail", **_cost_meta(run_inputs))
+ return _run_record(_error_run(
forge, ref, post, 0, f"build_chunks failed: {e}", t0,
post_mode=post_mode, tracer=tracer, sampling=sampling,
- )
+ cost_label=_cost_label(run_inputs, post_cost),
+ ), run_inputs)
if not chunks:
# No chunk survived build_chunks — an empty diff, or every file
@@ -442,13 +618,20 @@ def orchestrate_review(
# non-machinery file is binary still gets the deterministic finding
# instead of a silent Approved (issue #29 residual, concern #2).
release_shape = heuristics.release_shape_findings(files)
- tracer.event("run", "ok", chunks_reviewed=0, findings=len(release_shape))
- return _summary_only_run(
+ tracer.event(
+ "run", "ok", chunks_reviewed=0, findings=len(release_shape),
+ **_cost_meta(run_inputs),
+ **(_scope_counts(release_shape) if ticket_active else {}),
+ )
+ return _run_record(_summary_only_run(
forge, ref, pr, files, post, t0,
post_mode=post_mode, post_verdict=post_verdict, tracer=tracer,
sampling=sampling, release_shape_findings=release_shape,
confidence_floor=confidence_floor, max_errors=max_errors,
- )
+ ticket_note=ticket_note,
+ cost_label=_cost_label(run_inputs, post_cost),
+ size_advisory_line=size_advisory_line,
+ ), run_inputs)
# Pruned BEFORE the threads are listed, and both before the review units
# run. The prune-then-list order is load-bearing: reading threads first
@@ -469,11 +652,106 @@ def orchestrate_review(
logger.warning("list_threads failed (best-effort): %s", e)
threads = []
+ # Best-effort, like the thread listing: a spec-fetch failure is data for
+ # the grounding note, never a failed review. The digest is built once,
+ # after parse_unified_diff (the files are the pruning input) and before
+ # the worker fan-out, then rides the existing chunk + sweep calls.
+ spec_digest = ""
+ spec_note = ""
+ fetched: list[specs.SpecSource] | None = None
+ if spec_sources:
+ try:
+ fetched = specs.fetch_specs(
+ list(spec_sources),
+ max_chars=spec_max_chars,
+ jira_base_url=jira_base_url,
+ jira_email=jira_email,
+ jira_api_token=jira_api_token,
+ )
+ # The note only reaches a POSTED summary, so a --no-post, dry-run,
+ # or inline-only run would otherwise learn nothing about grounding.
+ # Logged before the digest is built, so a crash there keeps them.
+ for i, s in enumerate(fetched, start=1):
+ if s.error:
+ logger.warning(
+ "spec source %d/%d (%s, %s) failed (best-effort): %s",
+ i, len(fetched), s.kind or "unknown",
+ _log_safe_origin(s.origin), redact_for_post(s.error),
+ )
+ # Committed together at the end, so a crash leaves the run
+ # ungrounded, which is what its record says.
+ digest = specs.build_spec_digest(fetched, files, spec_digest_tokens)
+ note = _spec_note(fetched, digest)
+ spec_digest, spec_note = digest, note
+ except Exception as e: # noqa: BLE001
+ logger.error("spec grounding failed (best-effort): %s", e)
+ fetched = None
+ run_inputs["spec_grounding"] = {
+ "sources": len(spec_sources),
+ "ok": 0,
+ "failed": [f"spec stage crashed: {e.__class__.__name__}"],
+ "constraints": 0,
+ "digest_sha256": None,
+ }
+ tracer.event(
+ "specs", "fail",
+ sources=len(spec_sources), ok=0, constraints=0,
+ reasons=[f"spec stage crashed: {e.__class__.__name__}: {e}"],
+ )
+
+ # Grounded means at least one constraint line reached the digest. A digest
+ # without one (no sources, every source failed, nothing kept, or a budget
+ # too small for any unit) is not injected, so every prompt shows its
+ # no-specs text and forbids `spec`; apply_spec_grounding below relabels
+ # any `spec` the model emits anyway.
+ grounded = specs.constraint_count(spec_digest) > 0
+ injected = spec_digest if grounded else ""
+
+ # Recorded before the fan-out, so the total-failure exit carries it too.
+ # The record mirrors the posted note (labels, redacted reasons); the
+ # trace is operator-only and keeps the raw reasons. The hash encodes with
+ # surrogatepass because a Jira body's JSON escapes can decode to a lone
+ # surrogate, and this block sits outside the never-raise fence.
+ if fetched is not None:
+ ok = sum(1 for s in fetched if not s.error)
+ constraints = specs.constraint_count(injected)
+ failed = [
+ (f"source {i}{f' ({s.kind})' if s.kind else ''}", s.error)
+ for i, s in enumerate(fetched, start=1) if s.error
+ ]
+ run_inputs["spec_grounding"] = {
+ "sources": len(fetched),
+ "ok": ok,
+ "failed": [f"{label}: {redact_for_post(error)}" for label, error in failed],
+ "constraints": constraints,
+ "digest_sha256": (
+ hashlib.sha256(injected.encode("utf-8", "surrogatepass")).hexdigest()
+ if injected else None
+ ),
+ }
+ logger.info(
+ "spec grounding: %d/%d source(s) ok, %d constraint(s) injected",
+ ok, len(fetched), constraints,
+ )
+ tracer.event(
+ "specs", "ok" if ok else "fail",
+ sources=len(fetched), ok=ok, constraints=constraints,
+ **({} if ok else {"reasons": [f"{label}: {error}" for label, error in failed]}),
+ )
+
+ prompt_context = PromptContext(
+ rules_worker=rules.prompt_block("worker") if rules is not None else "",
+ rules_sweep=rules.prompt_block("sweep") if rules is not None else "",
+ ticket_scope=ticket.scope_block() if ticket_active else "",
+ ticket_context=ticket.prompt_block() if ticket_active else "",
+ spec_digest=injected,
+ )
reader = _make_file_reader(forge, ref, pr)
results = _run_workers(
llm, chunks, pr, max_tokens=max_tokens, max_workers=max_workers,
context_lines=context_lines, tracer=tracer,
reader=reader, all_files=files, trace_dir=trace_dir,
+ prompt_context=prompt_context,
)
# One more worker-style unit, not inside the pool: the sweep digests the
@@ -486,9 +764,25 @@ def orchestrate_review(
llm, files, pr, max_tokens=max_tokens,
token_budget=token_budget, tracer=tracer, threads=threads,
trace_dir=trace_dir,
+ prompt_context=prompt_context,
)
)
+ # Priced once every review unit is final, and BEFORE the total-failure
+ # exit below: requests went out, so that exit's record must say what they
+ # cost rather than the pre-request 0.0. Cost accounting never fails a
+ # review; a crash here leaves the cost unknown.
+ try:
+ _stamp_run_cost(
+ run_inputs, results, {} if price_table is None else price_table,
+ )
+ except Exception as e: # noqa: BLE001
+ logger.warning("cost accounting failed (continuing): %s", e)
+ run_inputs["cost_usd"] = None
+ run_inputs["cost_estimated"] = False
+ run_inputs["cost_api_equivalent"] = False
+ cost_label = _cost_label(run_inputs, post_cost)
+
input_tokens = sum(r["input_tokens"] for r in results)
output_tokens = sum(r["output_tokens"] for r in results)
model = next((r["model"] for r in results if r["model"]), "unknown")
@@ -500,12 +794,13 @@ def orchestrate_review(
if all(r["error"] for r in results[:-1]):
reason = f"all {len(chunks)} worker reviews failed ({results[0]['error']})"
logger.error("Total LLM failure: %s", reason)
- tracer.event("run", "fail")
- return _error_run(
+ tracer.event("run", "fail", **_cost_meta(run_inputs))
+ return _run_record(_error_run(
forge, ref, post, len(chunks) + 1, reason, t0, tracer=tracer,
model=model, input_tokens=input_tokens, output_tokens=output_tokens,
- post_mode=post_mode, sampling=sampling,
- )
+ post_mode=post_mode, sampling=sampling, cost_label=cost_label,
+ chunks_reviewed=sum(1 for r in results if not r["error"]),
+ ), run_inputs)
chunks_failed = sum(1 for r in results if r["error"])
chunks_reviewed = len(results) - chunks_failed
@@ -517,6 +812,7 @@ def orchestrate_review(
len(r["findings"]) for r in results[:-1] if not r["error"]
)
findings = [f for r in results if not r["error"] for f in r["findings"]]
+ findings = _enforce_scope(findings, ticket_active)
# Futures were submitted in chunk order, so results[i] is chunk[i]'s
# outcome for i < len(chunks): the zip pairs each failed review with the
@@ -546,6 +842,44 @@ def orchestrate_review(
findings = findings[:sweep_start] + release_shape + findings[sweep_start:]
sweep_start += len(release_shape)
+ # FIRST among the passes: a team word the map knows ("blocker") would
+ # otherwise die at the gate as an invalid severity, and consistency and
+ # _origin_key both read the severity. 1:1 and order-preserving, so
+ # sweep_start still marks the boundary.
+ if rules is not None and rules.severity_map:
+ mapped = apply_severity_map(findings, rules.severity_map)
+ remapped = sum(
+ 1
+ for before, after in zip(findings, mapped, strict=True)
+ if before.severity != after.severity
+ )
+ if remapped:
+ logger.info(
+ "severity map: rewrote %d finding(s) from team severity words",
+ remapped,
+ )
+ tracer.event("rules", "remap", findings=remapped)
+ findings = mapped
+
+ # Right after the map (whose tiers never include `spec`) and ahead of
+ # consistency, so an ungrounded `spec` can never raise a same-title
+ # sibling to spec. Covers the sweep's findings too; 1:1 and
+ # order-preserving, so sweep_start still marks the boundary.
+ graded = apply_spec_grounding(findings, grounded=grounded)
+ relabelled = sum(
+ 1
+ for before, after in zip(findings, graded, strict=True)
+ if before.severity != after.severity
+ )
+ if relabelled:
+ logger.info(
+ "spec grounding: relabelled %d spec finding(s) as warning "
+ "(no spec constraint was injected)",
+ relabelled,
+ )
+ tracer.event("specs", "relabel", findings=relabelled)
+ findings = graded
+
findings = apply_location_validation(findings, [f.path for f in files])
# BEFORE apply_line_align, deliberately: the manifest check compares the
# model's raw anchor against the key and section it claims, and realignment
@@ -569,7 +903,7 @@ def orchestrate_review(
)
findings = consistent
findings = apply_removal_claim_check(findings, files)
- findings = apply_hedge_gate(findings)
+ findings = apply_hedge_gate(findings, spec_digest=injected)
# The sweep boundary is positional, and the gate now returns its findings
# in content order, so the boundary is re-derived from the identity of the
# sweep's own findings rather than carried across the gate as an index.
@@ -635,6 +969,10 @@ def orchestrate_review(
# "findings may be incomplete" without which-files acts on nothing.
failed_chunks=failed_chunks,
include_verdict=post_verdict,
+ spec_note=spec_note,
+ ticket_note=ticket_note,
+ cost_label=cost_label,
+ size_advisory_line=size_advisory_line,
)
try:
forge.post_summary(ref, summary)
@@ -648,7 +986,11 @@ def orchestrate_review(
if post_inline_wanted and findings_active and (posted or not post_summary_wanted):
ordered = sorted(
findings_active,
- key=lambda f: (_SEVERITY_RANK.get(f.severity, 3), *finding_rank_key(f)),
+ key=lambda f: (
+ _SEVERITY_RANK.get(f.severity, 3),
+ _SCOPE_RANK.get(f.scope, 0),
+ *finding_rank_key(f),
+ ),
)
comments = [
InlineComment(
@@ -681,6 +1023,10 @@ def orchestrate_review(
chunks_reviewed=chunks_reviewed, chunks_failed=chunks_failed,
failed_chunks=failed_chunks,
include_verdict=post_verdict,
+ spec_note=spec_note,
+ ticket_note=ticket_note,
+ cost_label=cost_label,
+ size_advisory_line=size_advisory_line,
inline_accounting=_inline_accounting(
len(findings_active), inline_attempted, inline_posted,
failed=inline_failed, cap=max_inline_comments,
@@ -701,8 +1047,10 @@ def orchestrate_review(
"run", "ok", verdict=verdict,
chunks_reviewed=chunks_reviewed, chunks_failed=chunks_failed,
findings=len(findings_active),
+ **_cost_meta(run_inputs),
+ **(_scope_counts(findings_active) if ticket_active else {}),
)
- return {
+ return _run_record({
"verdict": verdict,
"findings_active": findings_active,
"findings_dropped": findings_dropped,
@@ -714,7 +1062,7 @@ def orchestrate_review(
"output_tokens": output_tokens,
"posted": posted,
"sampling": _sampling(llm),
- }
+ }, run_inputs)
def _origin_key(finding: Finding) -> tuple:
@@ -724,7 +1072,9 @@ def _origin_key(finding: Finding) -> tuple:
finding and a sweep finding that agree on file, line, title, and body
collide, ``finding_sort_key`` ties them, and the Counter walk hands the
first survivor to the sweep side — dropping the higher-confidence chunk
- copy as a "duplicate of chunk finding".
+ copy as a "duplicate of chunk finding". ``scope`` is in it for the same
+ reason: with a ticket active the two copies can disagree on it, and a
+ swap would put the sweep copy's scope in the chunk copy's slot.
"""
return (
finding.file,
@@ -733,9 +1083,38 @@ def _origin_key(finding: Finding) -> tuple:
finding.body,
finding.severity,
finding.confidence,
+ finding.scope,
)
+def _enforce_scope(findings: Sequence[Finding], active: bool) -> list[Finding]:
+ """Hold every finding's ``scope`` to what the run asked the model for.
+
+ With no active ticket the prompts never asked for a scope, so any value
+ other than ``unknown`` — from a test double, a library reviewer, or a
+ future backend that bypasses the reviewer's own gate — is reset to
+ ``unknown``. With one active, the value is normalized
+ (:func:`triage.normalize_scope`), so an unrecognised one is ``unknown``
+ too. Returns a new list in the same order; only a finding whose scope
+ changes is replaced, with :func:`dataclasses.replace`.
+ """
+ out: list[Finding] = []
+ for f in findings:
+ scope = normalize_scope(f.scope) if active else SCOPE_UNKNOWN
+ out.append(f if scope == f.scope else replace(f, scope=scope))
+ return out
+
+
+def _scope_counts(findings: Sequence[Finding]) -> dict[str, int]:
+ """The ``run ok`` event's ``scope_in`` / ``scope_out`` / ``scope_unknown``."""
+ counts = Counter(f.scope for f in findings)
+ return {
+ "scope_in": counts[SCOPE_IN],
+ "scope_out": counts[SCOPE_OUT],
+ "scope_unknown": counts[SCOPE_UNKNOWN],
+ }
+
+
def _sampling(llm: object) -> dict:
"""Report the sampling knobs a client had in force, duck-typed.
@@ -753,8 +1132,103 @@ def _elapsed_ms(t0: float) -> int:
return int((time.perf_counter() - t0) * 1000)
-def _attribution(model: str, tokens: int, elapsed_ms: int) -> str:
- return f"{ATTRIBUTION_MARKER} · model={model} · {tokens} tok · {elapsed_ms / 1000:.1f}s"
+def _run_record(result: dict, run_inputs: Mapping[str, Any]) -> dict:
+ """Stamp one exit's result with the per-run record; the single choke point.
+
+ Every return of :func:`orchestrate_review` goes through here, so a
+ run-record key is added once instead of at each exit, and no exit can be
+ missed. Each key of ``run_inputs`` is copied in with ``setdefault``
+ semantics — a key the exit's own dict already carries wins — except
+ ``replay``, which is written only when it is not ``None``: a normal run's
+ record has no ``replay`` key at all, and a replay's is a copy of the
+ stamp, never the caller's mapping. ``cost_api_equivalent`` is written
+ only when it is ``True``, so a run not priced by claude-cli has the same
+ record it had before the label existed. Returns ``result`` itself.
+ """
+ for key, value in run_inputs.items():
+ if key == "replay":
+ if value is not None:
+ result.setdefault(key, dict(value))
+ elif key == "cost_api_equivalent":
+ if value is True:
+ result.setdefault(key, True)
+ else:
+ result.setdefault(key, value)
+ return result
+
+
+def _cost_meta(run_inputs: Mapping[str, Any]) -> dict[str, Any]:
+ """The cost keys every ``run ok`` / ``run fail`` trace event carries."""
+ return {
+ "cost_usd": run_inputs.get("cost_usd"),
+ "cost_estimated": run_inputs.get("cost_estimated") is True,
+ }
+
+
+def _cost_label(run_inputs: Mapping[str, Any], post_cost: bool) -> str:
+ """The attribution's cost field, or ``""`` when ``post_cost`` is off.
+
+ ``""`` keeps every attribution byte-identical to a run without cost
+ posting; otherwise it is :func:`prxref.costs.cost_label` of the cost in
+ force at this exit (``$0.00`` before any LLM request, ``cost unknown``
+ when the run's cost could not be established, and ``$0.0007
+ (API-equivalent)`` when ``cost_api_equivalent`` is set).
+ """
+ if not post_cost:
+ return ""
+ return costs.cost_label(
+ run_inputs.get("cost_usd"), run_inputs.get("cost_estimated") is True,
+ api_equivalent=run_inputs.get("cost_api_equivalent") is True,
+ )
+
+
+def _stamp_run_cost(
+ run_inputs: dict,
+ units: Sequence[Mapping[str, Any]],
+ price_table: Mapping[str, Any],
+) -> None:
+ """Set ``run_inputs["cost_usd"]``, ``["cost_estimated"]`` and ``["cost_api_equivalent"]``.
+
+ Called once, after the sweep, with every review unit's result (the chunk
+ workers plus the sweep) and the parsed price table (``{}`` when unset).
+ The total is :func:`prxref.costs.run_cost`: each received unit's reported
+ cost, else a price-table estimate for its exact model name when the unit
+ counted input tokens (a unit reporting 0, as every kiro-cli unit does, is
+ never estimated), else the whole run is unknown (``None``, never ``0``
+ and never a partial sum). A run
+ left unknown by models with neither figure logs one INFO line naming
+ them, so a table keyed on the wrong model name diagnoses itself. A table
+ that is not a valid parsed table raises, and the caller records the cost
+ as unknown. ``cost_api_equivalent`` is
+ :func:`prxref.costs.api_equivalent_run` over the same units, derived here
+ once so the attribution and the CLI's ``-v`` line cannot disagree.
+ """
+ cost_usd, cost_estimated, unpriced = costs.run_cost(units, price_table)
+ if unpriced:
+ logger.info(
+ "cost unknown: no reported cost and no usable PRXREF_PRICE_TABLE "
+ "estimate for model(s) %s",
+ ", ".join(repr(m) for m in unpriced),
+ )
+ run_inputs["cost_usd"] = cost_usd
+ run_inputs["cost_estimated"] = cost_estimated
+ run_inputs["cost_api_equivalent"] = costs.api_equivalent_run(units)
+
+
+def _attribution(
+ model: str, tokens: int, elapsed_ms: int, *, cost_label: str = "",
+) -> str:
+ """The attribution line every posted comment carries.
+
+ ``cost_label`` (``"$0.0007"``, ``"$0.0007 (API-equivalent)"``,
+ ``"~$0.0007 (est.)"``, ``"cost unknown"``)
+ is appended as the LAST field, and only when non-empty: the existing
+ fields keep their order, so a consumer that parses ``model=`` or the
+ token count, and the prune pass that matches ``ATTRIBUTION_MARKER`` as a
+ prefix, see the same line whether or not cost is posted.
+ """
+ line = f"{ATTRIBUTION_MARKER} · model={model} · {tokens} tok · {elapsed_ms / 1000:.1f}s"
+ return f"{line} · {cost_label}" if cost_label else line
def _prune_stale_inline_comments(forge: Forge, ref: PRRef) -> None:
@@ -866,6 +1340,7 @@ def _run_workers(
max_workers: int = MAX_WORKERS, context_lines: int | None = None,
tracer: Tracer | None = None, reader=None, all_files=None,
trace_dir: str | None = None,
+ prompt_context: PromptContext = NO_PROMPT_CONTEXT,
) -> list[dict]:
# Never below 1: ThreadPoolExecutor rejects a zero-width pool, and a
# library caller is not gated by config's range check.
@@ -901,6 +1376,7 @@ def _heartbeat() -> None:
_run_worker, i + 1, len(chunks), llm, chunk, pr,
max_tokens, context_lines, tracer, reader, all_files,
trace_label=f"chunk{i}", trace_dir=trace_dir,
+ prompt_context=prompt_context,
)
for i, chunk in enumerate(chunks)
]
@@ -916,6 +1392,7 @@ def _heartbeat() -> None:
"findings": [], "error": f"worker crashed: {e}",
"input_tokens": 0, "output_tokens": 0,
"model": "", "elapsed_ms": 0,
+ "cost_usd": None, "cost_source": "",
})
return results
finally:
@@ -947,6 +1424,7 @@ def _invoke_chunk(
max_tokens: int | None, context_lines: int | None,
reader=None, *, include_definitions: bool = True, all_files=None,
trace_label: str = "", trace_dir: str | None = None,
+ prompt_context: PromptContext = NO_PROMPT_CONTEXT,
) -> dict:
"""One normalized :func:`reviewer.review_chunk` call; never raises.
@@ -962,7 +1440,15 @@ def _invoke_chunk(
retry, whose whole purpose is a smaller prompt. ``all_files`` is the PR's
full parsed file list; the reviewer reduces it to the bounded sibling
summary, which survives the retry because refuting evidence is not
- bulk context.
+ bulk context. ``prompt_context`` (rules, ticket, spec digest) is passed
+ unchanged on both attempts: it is intent, not bulk context, and a
+ dict-shaped finding keeps its ``scope`` only when
+ :attr:`reviewer.PromptContext.scope_active`.
+
+ The shape carries the reviewer's reported ``cost_usd`` and
+ ``cost_source`` beside the token counts; a call that raised, or a stub
+ whose meta lacks them, gives ``None`` and ``""``. Pricing is left to
+ :func:`_stamp_run_cost`, over the whole run.
"""
blocks = _context_blocks(chunk, reader, include_definitions=include_definitions)
try:
@@ -971,12 +1457,13 @@ def _invoke_chunk(
max_tokens=max_tokens, context_lines=context_lines,
context_blocks=blocks, sibling_files=all_files or (),
trace_label=trace_label, trace_dir=trace_dir or "",
+ prompt_context=prompt_context,
)
except Exception as e: # noqa: BLE001
return {
"findings": [], "error": str(e),
"input_tokens": 0, "output_tokens": 0, "model": "",
- "elapsed_ms": 0,
+ "elapsed_ms": 0, "cost_usd": None, "cost_source": "",
}
# reviewer returns (findings, meta); legacy dict stubs still accepted.
@@ -989,11 +1476,13 @@ def _invoke_chunk(
"model": meta.get("model", ""),
"elapsed_ms": meta.get("elapsed_ms", 0),
"error": meta.get("error", ""),
+ "cost_usd": meta.get("cost_usd"),
+ "cost_source": meta.get("cost_source", ""),
}
findings = []
for item in res.get("findings") or []:
- finding = _coerce_finding(item)
+ finding = _coerce_finding(item, accept_scope=prompt_context.scope_active)
if finding is not None:
findings.append(finding)
@@ -1004,6 +1493,8 @@ def _invoke_chunk(
"output_tokens": res.get("output_tokens", 0),
"model": res.get("model", ""),
"elapsed_ms": res.get("elapsed_ms", 0),
+ "cost_usd": res.get("cost_usd"),
+ "cost_source": res.get("cost_source", ""),
}
@@ -1012,6 +1503,7 @@ def _run_worker(
max_tokens: int | None = None, context_lines: int | None = None,
tracer: Tracer | None = None, reader=None, all_files=None,
trace_label: str = "", trace_dir: str | None = None,
+ *, prompt_context: PromptContext = NO_PROMPT_CONTEXT,
) -> dict:
tracer = tracer if tracer is not None else get_tracer()
t0 = time.perf_counter()
@@ -1028,7 +1520,7 @@ def _run_worker(
)
res = _invoke_chunk(
llm, chunk, pr, max_tokens, context_lines, reader, all_files=all_files,
- trace_label=trace_label, trace_dir=trace_dir,
+ trace_label=trace_label, trace_dir=trace_dir, prompt_context=prompt_context,
)
if (
res["error"]
@@ -1054,6 +1546,7 @@ def _run_worker(
llm, chunk, pr, max_tokens, _TIMEOUT_RETRY_CONTEXT_LINES, reader,
include_definitions=False, all_files=all_files,
trace_label=trace_label, trace_dir=trace_dir,
+ prompt_context=prompt_context,
)
error = res["error"]
@@ -1074,6 +1567,7 @@ def _run_worker(
model=res["model"],
input_tokens=res["input_tokens"],
output_tokens=res["output_tokens"],
+ cost_usd=res["cost_usd"],
)
return {
"findings": res["findings"],
@@ -1082,6 +1576,8 @@ def _run_worker(
"output_tokens": res["output_tokens"],
"model": res["model"],
"elapsed_ms": _elapsed_ms(t0),
+ "cost_usd": res["cost_usd"],
+ "cost_source": res["cost_source"],
}
@@ -1092,6 +1588,7 @@ def _run_sweep(
tracer: Tracer | None = None,
threads: Sequence[Thread] = (),
trace_dir: str | None = None,
+ prompt_context: PromptContext = NO_PROMPT_CONTEXT,
) -> dict:
"""Run the whole-PR systemic sweep as one worker-style review unit.
@@ -1099,7 +1596,11 @@ def _run_sweep(
``token_budget``), makes ONE single-shot call through
:func:`reviewer.review_systemic` — so ``PRXREF_LLM_MAX_TOKENS``, the
timeout, and the model fallback chain all apply as to any chunk — and
- returns the same result shape a chunk worker does. A failure is that
+ returns the same result shape a chunk worker does. ``prompt_context``
+ rides along into the sweep prompt (sweep rules and ticket scope in the
+ system half, ticket context and the spec digest in the user half), and a
+ dict-shaped finding keeps its ``scope`` only when
+ :attr:`reviewer.PromptContext.scope_active`. A failure is that
shape with ``error`` set prefixed ``systemic sweep:``, so the
partial-review banner names the unit that failed; it counts as one
failed chunk in the caller's coverage accounting.
@@ -1122,6 +1623,7 @@ def _run_sweep(
llm, digest, pr_title=pr.title, pr_description=pr.description,
max_tokens=max_tokens, threads=discussion,
trace_label="sweep", trace_dir=trace_dir or "",
+ prompt_context=prompt_context,
)
except Exception as e: # noqa: BLE001
logger.error("[sweep] raised: %s", e)
@@ -1133,11 +1635,12 @@ def _run_sweep(
"findings": [], "error": f"systemic sweep: {e}",
"input_tokens": 0, "output_tokens": 0, "model": "",
"elapsed_ms": _elapsed_ms(t0),
+ "cost_usd": None, "cost_source": "",
}
findings = []
for item in findings_raw:
- finding = _coerce_finding(item)
+ finding = _coerce_finding(item, accept_scope=prompt_context.scope_active)
if finding is not None:
findings.append(finding)
@@ -1157,6 +1660,7 @@ def _run_sweep(
model=meta.get("model", ""),
input_tokens=meta.get("input_tokens", 0),
output_tokens=meta.get("output_tokens", 0),
+ cost_usd=meta.get("cost_usd"),
)
return {
"findings": findings,
@@ -1165,10 +1669,12 @@ def _run_sweep(
"output_tokens": meta.get("output_tokens", 0),
"model": meta.get("model", ""),
"elapsed_ms": _elapsed_ms(t0),
+ "cost_usd": meta.get("cost_usd"),
+ "cost_source": meta.get("cost_source", ""),
}
-def _coerce_finding(item) -> Finding | None:
+def _coerce_finding(item, *, accept_scope: bool = False) -> Finding | None:
if isinstance(item, Finding):
return item
if isinstance(item, dict):
@@ -1180,6 +1686,10 @@ def _coerce_finding(item) -> Finding | None:
confidence=float(item.get("confidence") or 0.0),
title=str(item.get("title") or ""),
body=str(item.get("body") or ""),
+ scope=(
+ normalize_scope(item.get("scope")) if accept_scope
+ else SCOPE_UNKNOWN
+ ),
)
except (KeyError, TypeError, ValueError) as e:
logger.warning("dropping malformed finding %r: %s", item, e)
@@ -1203,7 +1713,30 @@ def _render_summary(
failed_chunks: Sequence[tuple[str, Sequence[str]]] = (),
include_verdict: bool = True,
inline_accounting: str | None = None,
+ spec_note: str = "",
+ ticket_note: str = "",
+ cost_label: str = "",
+ size_advisory_line: str = "",
) -> str:
+ """Render the PR summary comment body.
+
+ The template is filled in ONE pass (:func:`reviewer.fill_template`), so
+ a PR title, a note or a finding title containing ``{findings}``,
+ ``{attribution}`` or any other placeholder renders literally instead of
+ receiving that placeholder's value. ``spec_note`` and ``ticket_note``
+ ride ``{spec_note}{ticket_note}`` on the line after the counts; each
+ carries its own trailing newline when non-empty, so empty notes leave the
+ summary byte-identical. ``{findings}`` lists the in-ticket and unjudged
+ findings first; findings outside the ticket (scope ``"out"``) follow
+ under a bold ``Outside the ticket (N)`` heading led by
+ :data:`markers.OUT_OF_TICKET_MARKER`; when no other finding exists,
+ ``No in-ticket findings.`` stands in for the first list. Without an
+ active ticket every scope is ``"unknown"``, so the list stays flat.
+ ``cost_label`` is the attribution's last field
+ (:func:`_attribution`). ``size_advisory_line`` (``"> ⚠️ …\\n\\n"`` or
+ ``""``) is prepended to the finished body, after the partial-review
+ banner, so it is the first thing under the forge's summary marker.
+ """
try:
template = reviewer.load_prompt("summary")
except Exception as e: # noqa: BLE001
@@ -1212,35 +1745,42 @@ def _render_summary(
if not include_verdict:
template = _strip_verdict_stamp(template)
- counts = {"error": 0, "warning": 0, "outofscope": 0}
+ counts = {"error": 0, "warning": 0, "spec": 0, "outofscope": 0}
for f in findings_active:
counts[f.severity] = counts.get(f.severity, 0) + 1
- if findings_active:
- bullets = "\n".join(
- f"- {_SEVERITY_MARKERS.get(f.severity, '🟦')} "
- f"`{f.file}:{f.line if f.line > 0 else '—'}` — {f.title}"
- for f in findings_active
- )
+ inside = [f for f in findings_active if f.scope != SCOPE_OUT]
+ outside = [f for f in findings_active if f.scope == SCOPE_OUT]
+ if inside:
+ bullets = _summary_bullets(inside)
+ elif outside:
+ bullets = "No in-ticket findings."
else:
bullets = "No findings — nice work."
+ if outside:
+ bullets = (
+ f"{bullets}\n\n**{OUT_OF_TICKET_MARKER} Outside the ticket ({len(outside)})**"
+ f"\n\n{_summary_bullets(outside)}"
+ )
if inline_accounting:
bullets = f"{bullets}\n\n{inline_accounting}"
attribution = _attribution(
- model, input_tokens + output_tokens, elapsed_ms,
- )
- rendered = (
- template
- .replace("{verdict}", verdict)
- .replace("{title}", pr.title)
- .replace("{file_count}", str(len(files)))
- .replace("{error_count}", str(counts["error"]))
- .replace("{warning_count}", str(counts["warning"]))
- .replace("{outofscope_count}", str(counts["outofscope"]))
- .replace("{findings}", bullets)
- .replace("{attribution}", attribution)
+ model, input_tokens + output_tokens, elapsed_ms, cost_label=cost_label,
)
+ rendered = fill_template(template, {
+ "verdict": verdict,
+ "title": pr.title,
+ "file_count": str(len(files)),
+ "error_count": str(counts["error"]),
+ "warning_count": str(counts["warning"]),
+ "spec_count": str(counts["spec"]),
+ "outofscope_count": str(counts["outofscope"]),
+ "spec_note": spec_note,
+ "ticket_note": ticket_note,
+ "findings": bullets,
+ "attribution": attribution,
+ })
if attribution not in rendered:
rendered = f"{rendered}\n\n{attribution}"
if chunks_failed:
@@ -1257,7 +1797,127 @@ def _render_summary(
reason_lines = _failure_reason_lines(failed_chunks)
if reason_lines:
rendered += "\n>\n" + "\n".join(f"> {line}" for line in reason_lines)
- return rendered
+ return f"{size_advisory_line}{rendered}"
+
+
+def _size_advisory(
+ files,
+ *,
+ lines_limit: int | None,
+ files_limit: int | None,
+ ignore_globs: Sequence[str] = (),
+) -> dict | None:
+ """The PR-size advisory's stats, or ``None`` when both limits are unset.
+
+ The stats are ``{changed_lines, changed_files, lines_limit, files_limit,
+ triggered, message}``, computed whenever either limit is set, whether or
+ not it is exceeded. The counts come from
+ :func:`prxref.triage.count_size_relevant_changes`, which skips lockfiles
+ (:data:`prxref.heuristics.LOCKFILE_BASENAMES`), generated files and
+ ``ignore_globs``. A limit is exceeded strictly (``>``), so 0 is a real
+ threshold rather than "off". ``message`` is ``None`` unless a limit is
+ exceeded, and otherwise plain text naming only the exceeded limits, e.g.
+ ``This PR changes 812 lines in 24 files, above the team guideline of 500
+ lines and 20 files. Consider splitting it.`` The advisory never touches
+ the findings, so it cannot move the verdict or the exit code.
+ """
+ if lines_limit is None and files_limit is None:
+ return None
+ changed_lines, changed_files = count_size_relevant_changes(
+ files, lockfile_basenames=heuristics.LOCKFILE_BASENAMES, ignore_globs=ignore_globs,
+ )
+ exceeded = []
+ if lines_limit is not None and changed_lines > lines_limit:
+ exceeded.append(f"{lines_limit} {_plural(lines_limit, 'line')}")
+ if files_limit is not None and changed_files > files_limit:
+ exceeded.append(f"{files_limit} {_plural(files_limit, 'file')}")
+ message = None
+ if exceeded:
+ message = (
+ f"This PR changes {changed_lines} {_plural(changed_lines, 'line')} "
+ f"in {changed_files} {_plural(changed_files, 'file')}, above the team "
+ f"guideline of {' and '.join(exceeded)}. Consider splitting it."
+ )
+ return {
+ "changed_lines": changed_lines,
+ "changed_files": changed_files,
+ "lines_limit": lines_limit,
+ "files_limit": files_limit,
+ "triggered": message is not None,
+ "message": message,
+ }
+
+
+def _plural(n: int, unit: str) -> str:
+ """``unit`` for exactly one, else ``unit + "s"`` (0 lines, 1 line, 2 lines)."""
+ return unit if n == 1 else f"{unit}s"
+
+
+def _size_advisory_line(stats: Mapping[str, Any] | None) -> str:
+ """The blockquote a triggered size advisory prepends to the summary.
+
+ ``"> ⚠️ {message}\\n\\n"`` when ``stats`` carries a message, else ``""``,
+ which leaves the summary byte-identical to a run without the advisory.
+ """
+ message = stats.get("message") if stats else None
+ return f"> ⚠️ {message}\n\n" if message else ""
+
+
+def _spec_note(sources: Sequence[Any], digest: str) -> str:
+ """Render the summary's grounding note, ``""`` when nothing was requested.
+
+ One blockquote line counts what was injected
+ (:func:`prxref.specs.constraint_count`); one lists every failed source.
+ A failure is labelled by its 1-based position in the configured source
+ list and its kind, ``source 2 (url)``, or ``source 2`` when the kind was
+ never determined, never by its origin: a local path or a URL's query is
+ not the PR audience's business, and the operator can map the ordinal
+ back to the list. Each reason goes through :func:`redact_for_post`
+ first, because this text is posted. A run whose every source failed
+ renders ONLY the failure line — the review was un-grounded, and the note
+ must not dress it up as grounded. The note rides the ``{spec_note}``
+ placeholder on its own line between the counts and the findings, and a
+ non-empty note carries its own trailing newline, so an empty return
+ leaves the summary byte-identical to an ungrounded run's.
+ """
+ if not sources:
+ return ""
+ total = len(sources)
+ failed = [(i, s) for i, s in enumerate(sources, start=1) if s.error]
+ lines: list[str] = []
+ if len(failed) < total:
+ lines.append(
+ f"> {SEVERITY_MARKERS['spec']} Spec-grounded: {total} source(s) · "
+ f"{specs.constraint_count(digest)} constraint(s) injected"
+ )
+ if failed:
+ reasons = "; ".join(
+ f"source {i}{f' ({s.kind})' if s.kind else ''}: {redact_for_post(s.error)}"
+ for i, s in failed
+ )
+ lines.append(
+ f"> ⚠️ Spec fetch failed for {len(failed)} source(s): {reasons}"
+ )
+ return "\n".join(lines) + "\n"
+
+
+def _log_safe_origin(origin: str) -> str:
+ """Name a spec source for the operator's log, never its credentials.
+
+ A URL keeps ``scheme://host[:port]/path`` and loses its userinfo, query,
+ fragment and ``;params``, because CI logs are read more widely than the
+ operator's config. Anything without a scheme and a network location is
+ a local path and is returned verbatim, since it tells the operator which
+ file or directory to fix. A malformed URL is not echoed at all.
+ """
+ try:
+ parsed = urlparse(origin.strip())
+ except ValueError:
+ return "[unparseable origin]"
+ if not (parsed.scheme and parsed.netloc):
+ return origin
+ host = parsed.netloc.rpartition("@")[2]
+ return f"{parsed.scheme}://{host}{parsed.path}"
def _chunk_files_label(files: Sequence[str]) -> str:
@@ -1316,11 +1976,18 @@ def _failure_reason_lines(
+def _summary_bullets(findings: Sequence[Finding]) -> str:
+ """One ``- `file:line` — title`` summary bullet per finding, in order."""
+ return "\n".join(
+ f"- {marker_for(f.severity, f.scope)} "
+ f"`{f.file}:{f.line if f.line > 0 else '—'}` — {f.title}"
+ for f in findings
+ )
+
+
def _format_finding(f: Finding, model: str) -> str:
- marker = _SEVERITY_MARKERS.get(f.severity, "🟦")
- loc = f"{f.file}:{f.line}" if f.line > 0 else f.file
return (
- f"🤖 {marker} **[{f.severity.upper()}] {f.title}** (`{loc}`)\n\n"
+ f"{inline_header(f)}\n\n"
f"{f.body}\n\n"
f"---\n*Reviewed by prxref · model={model}*"
)
@@ -1356,6 +2023,7 @@ def _summary_only_run(
tracer: Tracer | None = None, sampling: dict | None = None,
release_shape_findings: list[Finding] | None = None,
confidence_floor: float | None = None, max_errors: int | None = None,
+ ticket_note: str = "", cost_label: str = "", size_advisory_line: str = "",
) -> dict:
"""The no-chunk exit: an empty diff, or every file binary.
@@ -1369,6 +2037,11 @@ def _summary_only_run(
``release_shape_findings=[]`` (fewer than 2 files can never be
release-shaped), so this degrades to exactly the prior empty-diff
behaviour: ``Approved``, no findings, no banner.
+
+ ``ticket_note``, ``cost_label`` and ``size_advisory_line`` are handed to
+ :func:`_render_summary` unchanged; all three default to ``""``, which
+ renders the summary exactly as before. The run-record keys are added by
+ the caller's :func:`_run_record`, not here.
"""
tracer = tracer if tracer is not None else get_tracer()
elapsed_ms = _elapsed_ms(t0)
@@ -1400,6 +2073,9 @@ def _summary_only_run(
pr, files, verdict, findings_active, "unknown", 0, 0, elapsed_ms,
chunks_reviewed=0, chunks_failed=0,
include_verdict=post_verdict,
+ ticket_note=ticket_note,
+ cost_label=cost_label,
+ size_advisory_line=size_advisory_line,
)
try:
forge.post_summary(ref, summary)
@@ -1435,7 +2111,23 @@ def _error_run(
post_mode: str = "summary+inline",
tracer: Tracer | None = None,
sampling: dict | None = None,
+ *,
+ cost_label: str = "",
+ chunks_reviewed: int = 0,
) -> dict:
+ """The error exit: post the failure notice when asked, return an Error run.
+
+ ``cost_label`` becomes the notice attribution's last field
+ (:func:`_attribution`); ``""`` leaves it as before. The notice never
+ carries a ticket note or a size advisory, and the run-record keys are
+ added by the caller's :func:`_run_record`, not here.
+
+ ``chunks_reviewed`` is how many of the ``chunk_count`` review units
+ succeeded; the rest are reported as failed. The default ``0`` fits every
+ exit taken before a review unit ran. The total-failure exit passes the
+ units that did succeed, so a sweep that answered over a dead worker pool
+ is counted as reviewed while the verdict stays ``Error``.
+ """
tracer = tracer if tracer is not None else get_tracer()
elapsed_ms = _elapsed_ms(t0)
posted = False
@@ -1447,6 +2139,7 @@ def _error_run(
if wanted:
attribution = _attribution(
model, input_tokens + output_tokens, elapsed_ms,
+ cost_label=cost_label,
)
# The same redaction the partial banner uses: this notice interpolates
# the reason into a public comment, and the caller has already logged
@@ -1468,8 +2161,8 @@ def _error_run(
"findings_active": [],
"findings_dropped": [],
"chunk_count": chunk_count,
- "chunks_reviewed": 0,
- "chunks_failed": chunk_count,
+ "chunks_reviewed": chunks_reviewed,
+ "chunks_failed": chunk_count - chunks_reviewed,
"elapsed_ms": elapsed_ms,
"input_tokens": input_tokens,
"output_tokens": output_tokens,
diff --git a/src/prxref/prompts/summary.md b/src/prxref/prompts/summary.md
index 6032eaa..bc11bd1 100644
--- a/src/prxref/prompts/summary.md
+++ b/src/prxref/prompts/summary.md
@@ -2,8 +2,8 @@
PR: {title} · files reviewed: {file_count}
-🟥 {error_count} error · 🟧 {warning_count} warning · 🟦 {outofscope_count} outofscope
-
+🟥 {error_count} error · 🟧 {warning_count} warning · 🔍 {spec_count} spec · ⬜ {outofscope_count} outofscope
+{spec_note}{ticket_note}
{findings}
---
diff --git a/src/prxref/prompts/systemic.md b/src/prxref/prompts/systemic.md
index 829f4ce..8a920fb 100644
--- a/src/prxref/prompts/systemic.md
+++ b/src/prxref/prompts/systemic.md
@@ -13,7 +13,7 @@ Per-chunk reviewers each see one slice of the diff and reliably miss classes tha
- A removed guard: a deleted numeric limit constant (`MAX_*_LENGTH`, `*_SIZE`, `*_BYTES`, `*_TIMEOUT`) or a deleted validator/sanitiser definition (`isValid*`, `validate*`, `sanitize*`, `check*`, `assert*`, `escape*`) on a path that consumes remote or third-party input. The digest shows these as `-` lines; the code that remains says nothing about the bound that is gone, so the deletion itself is the finding.
- Repo-config drift: two lockfiles for one package manager root — a lockfile newly added while another lockfile or a `packageManager` pin also appears in the PR. The digest states this collision on a `! repo-config:` line.
-Nothing else. Per-file bugs inside one chunk are the chunk workers' job; repeating them here only duplicates their findings, which are deduplicated away.
+Nothing else. Per-file bugs inside one chunk are the chunk workers' job; repeating them here only duplicates their findings, which are deduplicated away. One cross-file addition: with the whole-diff digest plus any spec constraints in view, this sweep is the natural seat for cross-file spec classes — naming rules, version pins, and `no component may` rules — while per-chunk seats catch line-local violations.
Do not raise a subject the reviewers already argued out under `### Existing discussion` — that decision was made with more context than the digest carries. If you raise it anyway, say in the body why the discussion's conclusion is wrong.
@@ -21,8 +21,13 @@ Do not raise a subject the reviewers already argued out under `### Existing disc
- `error` — the change will break at runtime or is a real bug: crash, wrong result, data loss, security hole, broken contract.
- `warning` — risk or smell the diff introduces or worsens: race-prone pattern, resource leak, missing error handling, load-bearing duplication.
+- `spec` — the diff violates a constraint quoted in the Spec constraints block below: a MUST/SHALL/required behaviour not implemented, a forbidden behaviour implemented, a version pin or naming rule broken. Only when specs were provided. Quote the violated constraint verbatim in the body, prefixed `Spec: "`.
- `outofscope` — minor: misleading naming, a TODO without context, dead code the diff adds.
+## Spec-grounded rules
+
+Emit `spec` only for a conflict between the diff and a constraint quoted in the Spec constraints block — never for a generic best practice not present in the block. This prompt's built-in classes (RLS, secrets, …) are never spec constraints. When the only basis for a finding is a constraint quoted in the Spec constraints block, its severity is `spec`. When the block reads `(no specs provided for this review)`, `spec` is not a legal severity. Cite the digest line that violates it — the same `file`/`line` contract as every finding — and quote the violated constraint verbatim in the body, prefixed `Spec: "`.
+
## Confidence
Each finding carries a confidence from 0.0 to 1.0 — how certain you are from this digest alone. Findings below the quality floor (default 0.6) are dropped downstream. 0.5 means "plausible but unverified". Reserve 0.9+ for defects provable from the digest text alone; an entry point with no visible auth check on a line that also names a paid API qualifies, a handler you merely suspect reaches a paid API does not.
@@ -44,6 +49,10 @@ PR description:
Repo: {repo_hint}
+{ticket_context}### Spec constraints
+
+{spec_digest}
+
The digest below lists every changed file (`## path`) with its hunk headers (`@@`), then its lines: a short file (or any file the migration DDL pattern touches) shows its FULL added content — every `+` line — so a statement you would expect and do not see inside such a file is evidence of absence; larger files show only the added (`+|`) and removed (`-|`) lines that matched a high-signal pattern, with secret, auth, and entry-point lines listed ahead of noisier matches in a capped file. A `! repo-config:` line is a synthetic note, not a diff line — cite it as a file-level finding (`line: 0`). `[full content omitted: ...]` means that file degraded to pattern lines only. `[digest truncated: token budget reached]` means the cap cut the text short; there is no more.
### Digest
@@ -65,7 +74,7 @@ Return exactly one JSON object, no prose, no fences:
"severity": "error",
"confidence": 0.9,
"title": "Paid API handler has no auth check",
- "body": "The digest shows the handler on line 42 reaching the billing API; no nonce or auth line for it appears anywhere in the digest."
+ "body": "The digest shows the handler on line 42 reaching the billing API; no nonce or auth line for it appears anywhere in the digest."{scope_example}
}
],
"escalations": []
diff --git a/src/prxref/prompts/worker.md b/src/prxref/prompts/worker.md
index b6ef715..79a8e21 100644
--- a/src/prxref/prompts/worker.md
+++ b/src/prxref/prompts/worker.md
@@ -8,8 +8,13 @@ Verify every claim against the diff itself. Every finding must cite a file and l
- `error` — the change will break at runtime or is a real bug: crash, wrong result, data loss, security hole, broken contract.
- `warning` — risk or smell the diff introduces or worsens: race-prone pattern, resource leak, missing error handling, load-bearing duplication.
+- `spec` — the diff violates a constraint quoted in the Spec constraints block below: a MUST/SHALL/required behaviour not implemented, a forbidden behaviour implemented, a version pin or naming rule broken. Only when specs were provided. Quote the violated constraint verbatim in the body, prefixed `Spec: "`.
- `outofscope` — minor: misleading naming, a TODO without context, dead code the diff adds.
+## Spec-grounded rules
+
+Emit `spec` only for a conflict between the diff and a constraint quoted in the Spec constraints block — never for a generic best practice not present in the block. When the only basis for a finding is a constraint quoted in the Spec constraints block, its severity is `spec`. When the block reads `(no specs provided for this review)`, `spec` is not a legal severity. Cite the diff line that violates it — the same `file`/`line` contract as every finding — and quote the violated constraint verbatim in the body, prefixed `Spec: "`.
+
## Confidence
Each finding carries a confidence from 0.0 to 1.0 — how certain you are from this diff alone. Findings below the quality floor (default 0.6) are dropped downstream. 0.5 means "plausible but unverified". Reserve 0.9+ for defects provable from the diff text alone.
@@ -37,6 +42,10 @@ PR description:
Repo: {repo_hint}
+{ticket_context}### Spec constraints
+
+{spec_digest}
+
The input stays under roughly 30k tokens; the diff below is the complete chunk.
### Diff
@@ -60,7 +69,7 @@ Return exactly one JSON object, no prose, no fences:
"severity": "error",
"confidence": 0.9,
"title": "Divide by zero when size is unset",
- "body": "size defaults to None and is used as a divisor on line 42; the diff adds no guard."
+ "body": "size defaults to None and is used as a divisor on line 42; the diff adds no guard."{scope_example}
}
],
"escalations": []
diff --git a/src/prxref/quality.py b/src/prxref/quality.py
index 5c468cb..506e742 100644
--- a/src/prxref/quality.py
+++ b/src/prxref/quality.py
@@ -1,17 +1,33 @@
"""Deterministic quality passes over worker findings.
-Eleven passes run before posting, in the order ``orchestrate_review``
-applies them. A twelfth deterministic check, the release-shaped-PR
+Thirteen passes run before posting, in the order ``orchestrate_review``
+applies them; pass 1 runs only when the team review rules declare a
+severity map. A fourteenth deterministic check, the release-shaped-PR
heuristic, is not a pass at all: ``heuristics.release_shape_findings``
ADDS a finding before pass 1 and it then flows through every pass below
exactly like a model finding. Every ``drop_reason`` prefix these passes
emit is tabulated for operators in ``docs/quality.md``.
-1. ``apply_location_validation``: drop findings whose ``file`` names no
+1. ``apply_severity_map``: when the team review rules declare a severity
+ map, rewrite a team severity word (``blocker``) to the prxref tier the
+ map gives it (``error``), compared after whitespace collapsing and
+ ``casefold()``. It runs first because every later pass reads the
+ severity. A dropped finding, an unmapped word and one of prxref's own
+ severities pass through unchanged; it drops nothing, and without a map
+ it is not called.
+2. ``apply_spec_grounding``: on a run that injected no spec constraint
+ (``specs.constraint_count`` of the digest is 0 — no sources, every
+ source failed, or nothing kept), relabel every ``spec`` finding as
+ ``warning``: the prompts showed the no-specs text, so the label has
+ nothing to be grounded in. It runs right after the team severity
+ map, so ``apply_severity_consistency`` never raises a same-title
+ sibling to ``spec`` on the strength of an ungrounded label. It drops
+ nothing.
+3. ``apply_location_validation``: drop findings whose ``file`` names no
path of the parsed diff — an empty, non-path, or invented location is
retained with ``drop_reason`` for the audit instead of rendering a
bullet anchored to nothing.
-2. ``apply_manifest_claim_check``: for findings on a manifest or
+4. ``apply_manifest_claim_check``: for findings on a manifest or
npm-family lockfile (``package.json``, ``bun.lock``, ...), drop a
claim whose named dependency is not the key on the anchored line
(``anchor mismatch:``) or sits under a different dependency section
@@ -19,7 +35,7 @@
own hunk holds no section header, the served full-file lines decide
the enclosing section. It runs BEFORE ``apply_line_align`` so it
reads the model's raw anchor.
-3. ``apply_line_align``: a line explicitly cited in the finding's own
+5. ``apply_line_align``: a line explicitly cited in the finding's own
title or body (``line 553``, ``at line 553``, an own-file
``path:line``) outranks a drifted ``line`` field whenever the cited
line lands on an added line — or a context line within tolerance of
@@ -35,43 +51,45 @@
an anchor survives only when it ties the file's best evidence match
or sits within tolerance of it, and a blank or pure-punctuation
anchor never survives while any token-bearing added line exists.
-4. ``apply_thread_dedup``: drop findings that duplicate an already-open
+6. ``apply_thread_dedup``: drop findings that duplicate an already-open
or existing thread on the PR (path + line-window + shared distinctive
tokens), with ``drop_reason`` ``duplicate of existing thread``.
-5. ``apply_settled_thread_suppression``: drop findings that re-litigate a
+7. ``apply_settled_thread_suppression``: drop findings that re-litigate a
subject an existing thread already argued out — same path plus shared
distinctive tokens, with NO line test, because line alignment has already
demoted a file-level finding to line 0 by this point
(``settled in thread: ``).
-6. ``apply_severity_consistency``: findings sharing one normalized title —
+8. ``apply_severity_consistency``: findings sharing one normalized title —
within a file or across sibling files — are all raised to the group's
maximum severity, so per-chunk workers cannot disagree about how
serious the same pattern is. Findings phrased differently but bound
by a shared rare code token, with a shared problem class or file,
join the same group (issue #30).
-7. ``apply_removal_claim_check``: drop findings whose removal verb governs
+9. ``apply_removal_claim_check``: drop findings whose removal verb governs
a path — ``removed src/app.py``, ``src/app.py was removed`` — when every
path the claim names is still present in the diff's post-image — the false positive a ``copy from``/``copy to``
header produces when a worker reads a copy as a move (issue #03).
Only a claim that NAMES a diff path is judged, so a finding about a
removed guard or constant is untouched.
-8. ``apply_hedge_gate``: drop findings whose title or body conditions the
- defect on a precondition the worker never established from the diff
- ("If X still leases a client", "unless the backfill already ran"),
- with ``drop_reason`` ``hedged: ""``.
-9. ``apply_quality_gate``: drop findings below the confidence floor
- (``confidence 0.40 below floor 0.60``), cap errors per review
- (``error cap exceeded (max N)``), and enforce the
- {error, warning, outofscope} severity vocabulary
- (``invalid severity: ''``). It RETURNS its findings sorted by
- ``finding_sort_key``, so the caller re-derives the chunk/sweep
- boundary from finding identity rather than carrying an index across it.
-10. ``apply_sweep_dedup``: drop a sweep finding that restates a chunk
+10. ``apply_hedge_gate``: drop findings whose title or body conditions the
+ defect on a precondition the worker never established from the diff
+ ("If X still leases a client", "unless the backfill already ran"),
+ with ``drop_reason`` ``hedged: ""``. A body's
+ ``Spec: "..."`` quote is not read for the text it copies verbatim from
+ the spec digest the workers were shown.
+11. ``apply_quality_gate``: drop findings below the confidence floor
+ (``confidence 0.40 below floor 0.60``), cap errors per review
+ (``error cap exceeded (max N)``), and enforce the
+ {error, warning, spec, outofscope} severity vocabulary
+ (``invalid severity: ''``). It RETURNS its findings sorted by
+ ``finding_sort_key``, so the caller re-derives the chunk/sweep
+ boundary from finding identity rather than carrying an index across it.
+12. ``apply_sweep_dedup``: drop a sweep finding that restates a chunk
finding which SURVIVED the gate, on file + normalized title
(``duplicate of chunk finding``). It runs after the gate so a
sub-floor chunk finding cannot suppress its higher-confidence sweep
duplicate and then die at the gate itself.
-11. ``apply_containment_note``: a finding that asserts a throw, panic,
+13. ``apply_containment_note``: a finding that asserts a throw, panic,
crash, or unhandled rejection and never names where it is caught or
where it propagates to has its body suffixed with
``" [containment boundary not stated]"`` — a purely textual
@@ -88,14 +106,14 @@
import os
import re
from collections import Counter
-from collections.abc import Callable, Sequence
+from collections.abc import Callable, Mapping, Sequence
from dataclasses import replace
from pathlib import PurePosixPath
from .forges.base import Thread
from .triage import DiffLine, FileDiff, Finding, Hunk
-SEVERITIES: frozenset[str] = frozenset({"error", "warning", "outofscope"})
+SEVERITIES: frozenset[str] = frozenset({"error", "warning", "spec", "outofscope"})
DEFAULT_CONFIDENCE_FLOOR: float = 0.6
DEFAULT_MAX_ERRORS: int = 10
@@ -170,6 +188,13 @@
_HEDGE_SPAN_MAX: int = 80
+# Where a finding quotes its constraint (``Spec: "..."``). Normative text is
+# conditional by nature ("If a session already exists, the server MUST reuse
+# it"), so the quoted text the digest really holds is the spec's precondition,
+# not the model's hedge, and is removed before the hedge rules read the body.
+_SPEC_QUOTE_OPEN_RE = re.compile(r"Spec:\s*[\"“‘']?")
+_SPEC_QUOTE_CLOSERS: frozenset[str] = frozenset("\"”’'")
+
def active(findings: Sequence[Finding]) -> list[Finding]:
"""Return only the findings that survived every quality pass."""
@@ -967,7 +992,9 @@ def apply_settled_thread_suppression(
return result
-_SEVERITY_RANK: dict[str, int] = {"error": 0, "warning": 1, "outofscope": 2}
+_SEVERITY_RANK: dict[str, int] = {
+ "error": 0, "warning": 1, "spec": 2, "outofscope": 3,
+}
_TITLE_PUNCT_RE = re.compile(r"[`*\"'\u2018\u2019\u201c\u201d]")
@@ -1225,8 +1252,8 @@ def apply_severity_consistency(findings: Sequence[Finding]) -> list[Finding]:
describing different problems stay apart.
Components bind transitively (A shares a token with B, B with C, so
- all three group). Each component is rewritten to its highest
- severity (error > warning > outofscope). A rewritten finding keeps
+ all three group). Each component is rewritten to its highest severity
+ (error > warning > spec > outofscope). A rewritten finding keeps
its own file, line, body, and confidence; only severity changes.
Findings carrying a ``drop_reason`` or a severity outside the
vocabulary pass through untouched. One summary line is logged when
@@ -1527,7 +1554,40 @@ def _hedge_span(text: str) -> str | None:
return None
-def apply_hedge_gate(findings: Sequence[Finding]) -> list[Finding]:
+def _spec_quote_len(rest: str, digest_lower: str) -> int:
+ lo, hi = 0, min(len(rest), len(digest_lower))
+ while lo < hi:
+ mid = (lo + hi + 1) // 2
+ if rest[:mid].lower() in digest_lower:
+ lo = mid
+ else:
+ hi = mid - 1
+ for k in range(min(lo, len(rest) - 1), 0, -1):
+ if rest[k] in _SPEC_QUOTE_CLOSERS:
+ return k
+ return 0
+
+
+def _blank_spec_quotes(body: str, spec_digest: str) -> str:
+ if not spec_digest:
+ return body
+ digest_lower = spec_digest.lower()
+ parts: list[str] = []
+ pos = 0
+ for m in _SPEC_QUOTE_OPEN_RE.finditer(body):
+ if m.start() < pos:
+ continue
+ k = _spec_quote_len(body[m.end():], digest_lower)
+ if k:
+ parts.append(body[pos:m.end()])
+ pos = m.end() + k
+ parts.append(body[pos:])
+ return "".join(parts)
+
+
+def apply_hedge_gate(
+ findings: Sequence[Finding], *, spec_digest: str = ""
+) -> list[Finding]:
"""Drop findings whose own text conditions the defect on an unverified fact.
A hedged finding ("If toolProxy.prepare still leases a client", "If they
@@ -1539,13 +1599,26 @@ def apply_hedge_gate(findings: Sequence[Finding]) -> list[Finding]:
Pure and order-preserving: already-dropped findings pass through
untouched, and a match sets ``drop_reason`` to ``hedged: ""``
naming the matched text so the drop is auditable in the run record.
+
+ ``spec_digest`` is the spec constraints block the workers were shown.
+ After each ``Spec:`` marker in the body (with or without an opening
+ quote), the longest following text that appears verbatim in the digest,
+ compared case-insensitively, is cut back to end just before a closing
+ quote and removed before the rules read the body; when no closing quote
+ follows any part of it, nothing is removed. A condition inside a real
+ constraint belongs to the spec, not to the model's reasoning, and this
+ holds for every severity. Text the digest does not hold, and every quote
+ when the digest is empty, is read like the rest of the body, so a model
+ cannot hide its own hedge inside a fabricated ``Spec: "..."``. The title
+ is always read as written.
"""
out: list[Finding] = []
for f in findings:
if f.drop_reason is not None:
out.append(f)
continue
- span = _hedge_span(f.title or "") or _hedge_span(f.body or "")
+ body = _blank_spec_quotes(f.body or "", spec_digest)
+ span = _hedge_span(f.title or "") or _hedge_span(body)
if span is None:
out.append(f)
continue
@@ -1562,12 +1635,15 @@ def apply_quality_gate(
"""Filter findings through vocabulary, confidence, and per-review error caps.
Order:
- 1. Severity vocabulary: non-empty lowercase must be in {error, warning, note};
- case-mismatches are normalized; invalid severities are dropped.
+ 1. Severity vocabulary: non-empty lowercase must be in
+ {error, warning, spec, outofscope}; case-mismatches are normalized;
+ invalid severities are dropped.
2. Confidence floor: drop findings below the threshold (default 0.6).
3. Error cap: among surviving errors, keep the top N ranked by
:func:`finding_rank_key` and drop the rest, so ties are broken by
- content rather than by arrival order.
+ content rather than by arrival order. ``spec`` findings never count
+ toward the cap: a spec-heavy review is neither crowded out by it nor
+ crowding it out.
The returned list is sorted by :func:`finding_sort_key`.
"""
@@ -1618,3 +1694,72 @@ def apply_quality_gate(
)
return sorted(staged, key=finding_sort_key)
+
+
+def apply_severity_map(
+ findings: Sequence[Finding], severity_map: Mapping[str, str],
+) -> list[Finding]:
+ """Rewrite a team severity word to the prxref severity it maps to.
+
+ ``severity_map`` is the review rules' front-matter map, team word to
+ prxref tier (``{"blocker": "error"}``). ``orchestrate_review`` runs this
+ before every other pass, so a mapped word reaches the gate as its tier
+ and an unmapped one still dies there as ``invalid severity``. Returns a
+ new list of the same length and order, rewritten findings being
+ :func:`dataclasses.replace` copies (every other field, ``scope``
+ included, is kept); it drops nothing.
+
+ A finding's severity matches a map word after ``strip()``, whitespace
+ collapsing and ``casefold()`` on both sides, so ``" Must FIX "`` meets
+ ``must fix``. A finding that already carries a ``drop_reason``, one whose
+ word is not in the map, and one that already names one of prxref's own
+ :data:`SEVERITIES` pass through as the same object: the map translates
+ team words only.
+ """
+ if not severity_map:
+ return list(findings)
+ table = {_severity_word(word): tier for word, tier in severity_map.items()}
+ out: list[Finding] = []
+ for f in findings:
+ word = _severity_word(f.severity)
+ tier = table.get(word)
+ if f.drop_reason is not None or tier is None or word in SEVERITIES:
+ out.append(f)
+ else:
+ out.append(replace(f, severity=tier))
+ return out
+
+
+def _severity_word(severity: object) -> str:
+ return " ".join(severity.split()).casefold() if isinstance(severity, str) else ""
+
+
+def apply_spec_grounding(
+ findings: Sequence[Finding], *, grounded: bool,
+) -> list[Finding]:
+ """Relabel ``spec`` findings as ``warning`` on a run with no spec grounding.
+
+ ``grounded`` says whether the run injected at least one spec constraint
+ into the prompts (:func:`prxref.specs.constraint_count` of the digest is
+ above 0). A run that injected none showed every review unit the no-specs
+ text, so a ``spec`` finding there has no quoted constraint behind it: it
+ is kept as a ``warning`` rather than posted under a label it has not
+ earned or dropped along with whatever it found. The severity is compared
+ after ``.strip().lower()``, so ``"SPEC"`` is relabelled too; every other
+ severity is left exactly as written, so this pass never raises a finding
+ to ``spec``. ``orchestrate_review`` runs it right after
+ :func:`apply_severity_map` and before every other pass.
+
+ Pure and order-preserving: returns a new list of the same length,
+ rewritten findings being :func:`dataclasses.replace` copies, and
+ already-dropped findings pass through untouched. Identity when
+ ``grounded`` is true.
+ """
+ if grounded:
+ return list(findings)
+ return [
+ replace(f, severity="warning")
+ if f.drop_reason is None and (f.severity or "").strip().lower() == "spec"
+ else f
+ for f in findings
+ ]
diff --git a/src/prxref/reviewer.py b/src/prxref/reviewer.py
index ed224ea..181ccc2 100644
--- a/src/prxref/reviewer.py
+++ b/src/prxref/reviewer.py
@@ -22,6 +22,20 @@
:func:`_render_discussion_block` under the ``DISCUSSION_MAX_*`` caps) so
the sweep stops re-raising subjects the team already argued out.
+Inputs an operator or a ticket supplies for the whole run ride one frozen
+:class:`PromptContext` through every hop, in one fixed order: team rules and
+the ticket-scope instructions are appended to the SYSTEM half (policy), while
+the ticket context and the spec digest are filled into the USER half ahead of
+the diff (per-PR data). While the ticket-scope instructions are in force, the
+``## Output Format`` JSON example that ends the USER half also shows a
+``"scope": "in"`` key on its finding, because a model copies the example it
+read last; the key is filled into the template's ``{scope_example}`` slot. Each
+template is filled in one pass by :func:`fill_template`, so a value that
+contains another placeholder (a PR description quoting ``{diff}``) renders
+literally. With :data:`NO_PROMPT_CONTEXT` the system prompt is the template
+head unchanged and the user prompt gains nothing: the ``{scope_example}`` slot
+renders empty, so the example is the pre-ticket one byte for byte.
+
Both ``prompts/worker.md`` and
``prompts/systemic.md`` require a throw/panic/crash/unhandled-rejection
finding to name its containment boundary; :func:`prxref.quality.apply_containment_note`
@@ -38,16 +52,19 @@
import json
import logging
import os
+import re
import time
-from collections.abc import Sequence
+from collections.abc import Mapping, Sequence
+from dataclasses import dataclass
from importlib import resources
from typing import Any
from .chunk_context import sibling_summary_block
+from .costs import valid_usd
from .forges.base import Thread
from .llm import LLMClient
from .parser import loads_lenient
-from .triage import FileDiff, Finding, trim_hunk_context
+from .triage import SCOPE_IN, SCOPE_UNKNOWN, FileDiff, Finding, normalize_scope, trim_hunk_context
logger = logging.getLogger("prxref")
@@ -56,6 +73,13 @@
_CONTEXT_MARKER = "## Review Context"
+_NO_SPECS_TEXT = "(no specs provided for this review)"
+
+# Fills the ``{scope_example}`` slot glued to the example finding's last value
+# in both templates' ``## Output Format``: the comma travels with the key, so
+# the empty value a no-ticket run gets leaves the example valid and unchanged.
+_SCOPE_EXAMPLE = f',\n "scope": "{SCOPE_IN}"'
+
_MAX_TOKENS_ENV = "PRXREF_LLM_MAX_TOKENS"
# Caps on the ``### Existing discussion`` block appended to the sweep prompt.
@@ -110,6 +134,71 @@ def load_prompt(name: str) -> str:
return resources.files("prxref").joinpath("prompts").joinpath(fname).read_text(encoding="utf-8")
+def fill_template(template: str, values: Mapping[str, str]) -> str:
+ """Replace each ``{name}`` in ``template`` whose name is a key of ``values``.
+
+ One :func:`re.sub` pass over the template: substituted text is never
+ scanned again, so a value that itself contains ``{diff}`` or any other
+ placeholder renders literally instead of receiving that placeholder's
+ value. Braces whose name is not a key (the JSON example in
+ ``## Output Format``, a literal ``{foo}``) stay as written. Values must
+ be strings and are inserted verbatim; backslashes are not interpreted.
+ An empty ``values`` returns the template unchanged.
+ """
+ if not values:
+ return template
+ pattern = re.compile(r"\{(" + "|".join(re.escape(k) for k in values) + r")\}")
+ return pattern.sub(lambda m: values[m.group(1)], template)
+
+
+@dataclass(frozen=True)
+class PromptContext:
+ """Run-wide inputs injected into every review unit's prompt, in one fixed order.
+
+ SYSTEM half, appended to the template head in this order:
+ ``rules_worker`` for chunk units or ``rules_sweep`` for the sweep (the
+ team review rules), then ``ticket_scope`` (the instructions that ask the
+ model for a per-finding ``scope``). USER half, after the Review Context
+ lines: ``ticket_context`` (the fenced ticket text), then ``spec_digest``
+ (the Spec constraints block), then the diff or digest.
+
+ Every field defaults to ``""``, which injects nothing; ``spec_digest``
+ empty renders ``(no specs provided for this review)`` as before.
+ :attr:`scope_active` is true only when the scope instructions are in the
+ prompt, and it alone decides whether a model-supplied ``scope`` is read
+ and whether the ``## Output Format`` example finding shows a ``"scope"``
+ key.
+ """
+
+ rules_worker: str = ""
+ rules_sweep: str = ""
+ ticket_scope: str = ""
+ ticket_context: str = ""
+ spec_digest: str = ""
+
+ @property
+ def scope_active(self) -> bool:
+ """True when the prompt asks for ``scope``, so the answer may be kept."""
+ return bool(self.ticket_scope)
+
+
+NO_PROMPT_CONTEXT = PromptContext()
+
+
+def _append_block(system: str, block: str) -> str:
+ block = block.strip()
+ return f"{system}\n\n{block}" if block else system
+
+
+def _ticket_context_value(prompt_context: PromptContext) -> str:
+ block = prompt_context.ticket_context.strip()
+ return f"{block}\n\n" if block else ""
+
+
+def _scope_example_value(prompt_context: PromptContext) -> str:
+ return _SCOPE_EXAMPLE if prompt_context.scope_active else ""
+
+
def _render_file(f: FileDiff, context_lines: int | None = None) -> str:
old = f.old_path or f.new_path or f.path
new = f.new_path or f.old_path or f.path
@@ -155,6 +244,8 @@ def _render_prompt(
context_lines: int | None = None,
context_blocks: str = "",
sibling_files: Sequence[FileDiff] = (),
+ *,
+ prompt_context: PromptContext = NO_PROMPT_CONTEXT,
) -> tuple[str, str]:
template = load_prompt("worker.md")
head, marker, tail = template.partition(_CONTEXT_MARKER)
@@ -162,20 +253,19 @@ def _render_prompt(
raise ValueError(f"worker.md is missing the {_CONTEXT_MARKER!r} split marker")
sibling_block = sibling_summary_block(chunk, sibling_files)
blocks = "\n\n".join(b for b in (sibling_block, context_blocks.strip()) if b)
- user = (
- marker + tail
- ).replace(
- "{pr_title}", pr_title.strip() or "(untitled)"
- ).replace(
- "{pr_description}", pr_description.strip() or "(none)"
- ).replace(
- "{repo_hint}", repo_hint.strip() or "(unspecified)"
- ).replace(
- "{context_blocks}", blocks
- ).replace(
- "{diff}", render_chunk(chunk, context_lines) or "(empty chunk)"
- )
- return head.strip(), user.strip()
+ user = fill_template(marker + tail, {
+ "pr_title": pr_title.strip() or "(untitled)",
+ "pr_description": pr_description.strip() or "(none)",
+ "repo_hint": repo_hint.strip() or "(unspecified)",
+ "ticket_context": _ticket_context_value(prompt_context),
+ "spec_digest": prompt_context.spec_digest.strip() or _NO_SPECS_TEXT,
+ "context_blocks": blocks,
+ "diff": render_chunk(chunk, context_lines) or "(empty chunk)",
+ "scope_example": _scope_example_value(prompt_context),
+ })
+ system = _append_block(head.strip(), prompt_context.rules_worker)
+ system = _append_block(system, prompt_context.ticket_scope)
+ return system, user.strip()
def _render_discussion_block(threads: Sequence[Thread]) -> str:
@@ -217,27 +307,29 @@ def _render_systemic_prompt(
pr_description: str,
repo_hint: str,
threads: Sequence[Thread] = (),
+ *,
+ prompt_context: PromptContext = NO_PROMPT_CONTEXT,
) -> tuple[str, str]:
template = load_prompt("systemic.md")
head, marker, tail = template.partition(_CONTEXT_MARKER)
if not marker:
raise ValueError(f"systemic.md is missing the {_CONTEXT_MARKER!r} split marker")
- user = (
- marker + tail
- ).replace(
- "{pr_title}", pr_title.strip() or "(untitled)"
- ).replace(
- "{pr_description}", pr_description.strip() or "(none)"
- ).replace(
- "{repo_hint}", repo_hint.strip() or "(unspecified)"
- ).replace(
- "{digest}", digest.strip() or "(empty digest)"
- )
+ user = fill_template(marker + tail, {
+ "pr_title": pr_title.strip() or "(untitled)",
+ "pr_description": pr_description.strip() or "(none)",
+ "repo_hint": repo_hint.strip() or "(unspecified)",
+ "ticket_context": _ticket_context_value(prompt_context),
+ "spec_digest": prompt_context.spec_digest.strip() or _NO_SPECS_TEXT,
+ "digest": digest.strip() or "(empty digest)",
+ "scope_example": _scope_example_value(prompt_context),
+ })
discussion = _render_discussion_block(threads)
user = user.strip()
if discussion:
user = f"{user}\n\n{discussion}"
- return head.strip(), user
+ system = _append_block(head.strip(), prompt_context.rules_sweep)
+ system = _append_block(system, prompt_context.ticket_scope)
+ return system, user
def _as_int(value: Any, default: int = 0) -> int:
@@ -247,7 +339,7 @@ def _as_int(value: Any, default: int = 0) -> int:
return default
-def _finding_from(raw: Any) -> Finding | None:
+def _finding_from(raw: Any, *, accept_scope: bool = False) -> Finding | None:
if not isinstance(raw, dict):
return None
file = str(raw.get("file") or raw.get("path") or "").strip()
@@ -264,6 +356,7 @@ def _finding_from(raw: Any) -> Finding | None:
confidence=confidence,
title=str(raw.get("title") or "").strip(),
body=str(raw.get("body") or "").strip(),
+ scope=normalize_scope(raw.get("scope")) if accept_scope else SCOPE_UNKNOWN,
)
@@ -282,7 +375,9 @@ def _write_trace_files(
prompt halves, ``