From f75f08971c765810015a0c88576dda204b68d6a5 Mon Sep 17 00:00:00 2001 From: Jehan Azad Date: Fri, 18 Sep 2026 23:50:04 +0000 Subject: [PATCH 1/6] Gate every node reputation penalty behind REPUTATION_PENALTY_SCALE, default 0 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A real mirrored node on the test droplet (node_ref ndebvzgeoij5t2l, node_id retce36dbb4) was permanently blocked off ONE trust sample. The chain: NodeAnalyticsManager.evaluate_reputations() acted on any node with at least one sample, a single out-of-threshold sample scores 0.0, NodeReputation .evaluate_trust charges 0.15 for that, and the evaluator runs every REPUTATION_INTERVAL_S = 60 s — so 1.0 crosses the 0.2 block threshold in six minutes. Once blocked, record_detection_frame returns False and every mirrored frame is dropped, apply_reward is a no-op so the node cannot climb back, and services/state_snapshot.py persists the block, so a restart brings it straight back. There is no admin unblock route, and NodeReputation .unblock() only resets to 0.3 — one penalty above re-blocking. Operator decision: for now, trust must never lower a node's reputation. One switch, default off, and a temporary stance rather than a change of intent — the only trust input today is a single claim residual from the identity-first lane, which is not enough evidence to act on. Trust is still computed and still reported; rewards are untouched. REPUTATION_PENALTY_SCALE (config/constants.py, default 0) multiplies every reputation penalty in the estate: trust 0.15/0.05, stale heartbeat 0.1, high detection rate 0.05, neighbour inconsistency 0.08, and the ADS-B cross-validation 0.1 charged from services/tasks/periodic.py. 0 records nothing at all, so no node can be blocked by any of those paths; 1 restores the historical behaviour. A negative or non-finite value logs a warning and reads as 0 — an unparseable gate must not read as "penalties on". core/state .py pushes it into the library at import, before restore_snapshot() rebuilds the reputations and before the evaluator's first pass, and logs it at INFO so a deploy log says which way it went. The scaling itself and the evaluator's new min-sample bar live in retina-analytics (separate submodule commit; the pin bump follows). services/node_bias.py now imports TRUST_MIN_SAMPLES from there instead of keeping its own literal 3, so the solver's reading of a young node's trust and the evaluator's willingness to act on it cannot drift apart. The switch deliberately does not unblock anything a snapshot already carries — that would hide from the operator which nodes had been blocked and why. backend/scripts/unblock_nodes.py does that instead: stdlib-only, operating on the schema-2 snapshot envelope with the server stopped. A script rather than a route because there is no unblock endpoint today and adding one is a separate decision (it would need an authz story and an audit trail, and this is a one-off cleanup after a bug, not an operation the product needs). Tests: penalties_on fixture in tests/conftest.py, applied to exactly the tests that assert a penalty lands, so the suite's ambient configuration is the deployed one. New coverage for the default recording nothing on a 100 km mismatch, the env parsing, that a restored block survives an evaluator pass untouched, and the unblock script's round trip through a verifying snapshot. Co-Authored-By: Claude Fable 5.1 --- backend/.env.example | 21 ++ backend/config/constants.py | 51 ++++ backend/core/state.py | 18 ++ backend/scripts/unblock_nodes.py | 246 ++++++++++++++++++ backend/services/node_bias.py | 10 +- backend/services/tasks/periodic.py | 7 + backend/tests/conftest.py | 26 ++ .../tests/test_adsb_cross_validation_gates.py | 6 + .../tests/test_adsb_freshness_regressions.py | 1 + backend/tests/test_analytics_refresh.py | 8 +- backend/tests/test_node_bias.py | 7 +- backend/tests/test_publication.py | 1 + .../tests/test_reputation_penalty_scale.py | 239 +++++++++++++++++ backend/tests/test_state_snapshot.py | 2 + backend/tests/test_unblock_nodes.py | 189 ++++++++++++++ docs/architecture.md | 1 + docs/runbook.md | 73 ++++++ 17 files changed, 902 insertions(+), 4 deletions(-) create mode 100644 backend/scripts/unblock_nodes.py create mode 100644 backend/tests/test_reputation_penalty_scale.py create mode 100644 backend/tests/test_unblock_nodes.py diff --git a/backend/.env.example b/backend/.env.example index eec740b58..de883f7e9 100644 --- a/backend/.env.example +++ b/backend/.env.example @@ -374,6 +374,27 @@ MENDER_PAT= # SOLVER_POOL=1 # SOLVER_POOL_CALL_TIMEOUT_S=30 +# Node reputation penalties (retina-analytics NodeReputation, evaluated every +# 60 s by services/tasks/periodic.reputation_evaluator). Multiplies EVERY +# penalty: trust 0.15/0.05, stale heartbeat 0.1, high detection rate 0.05, +# neighbour inconsistency 0.08, ADS-B cross-validation 0.1. Rewards are not +# scaled. 0 means no penalty is recorded at all, so no node can be blocked by +# any of these paths; 1 is the historical behaviour. +# +# The default is 0, and it is TEMPORARY (set 2026-09-18). The only trust input +# today is a single claim residual from the identity-first lane: one +# out-of-threshold residual scores a node 0.0, and 0.15 a pass takes it from +# 1.0 through the 0.2 block threshold in six minutes. That blocked a real +# mirrored node on the test droplet off ONE sample — and a blocked node has +# every frame dropped, cannot earn its way back (apply_reward is a no-op while +# blocked), and stays blocked across restarts because the block is persisted in +# the state snapshot. Put it back to 1 once trust is computed from enough +# evidence to act on. +# +# This only stops NEW penalties. Blocks a snapshot already carries are cleared +# with backend/scripts/unblock_nodes.py — see docs/runbook.md. +# REPUTATION_PENALTY_SCALE=0 + # Detection-archive buffering (services/frame_processor.py). Frames are kept in # the buffer across a failed Parquet write so a transient disk-full does not # drop data, and the per-node key is released only by a successful write that diff --git a/backend/config/constants.py b/backend/config/constants.py index 9bb6c274f..6dbe6f38b 100644 --- a/backend/config/constants.py +++ b/backend/config/constants.py @@ -8,6 +8,7 @@ retina_tracker YAML config stays separate (loaded at runtime via config.yaml). """ +import logging import math import os @@ -478,6 +479,56 @@ def _assoc_alt_layers_km() -> tuple[float, ...]: ADSB_TRUTH_INTERVAL_S = 120 # ADS-B truth fetcher sleep (s) ADSB_BACKOFF_S = 300 # Rate-limit backoff (s) +# ── Node reputation penalties ──────────────────────────────────────────────── +# Multiplier applied to EVERY reputation penalty in retina-analytics +# (NodeReputation.penalty_scale): trust 0.15/0.05 per evaluator pass, stale +# heartbeat 0.1, high detection rate 0.05, neighbour inconsistency 0.08, and +# the ADS-B cross-validation 0.1 charged from services/tasks/periodic.py. +# Rewards are untouched. +# +# 0 — no penalty is ever recorded, so no node can be blocked by any of +# these paths (a penalty of 0 is dropped entirely: no ledger entry, no +# reputation change). +# 1 — the historical behaviour. +# +# The default is 0, and that is a TEMPORARY stance taken 2026-09-18, not a +# decision that reputation should never bite. The only trust input today is a +# single claim residual from the identity-first lane, and one out-of-threshold +# residual scores a node 0.0: the evaluator then charges 0.15 every +# REPUTATION_INTERVAL_S (60 s) pass, so 1.0 crosses the 0.2 block threshold in +# six minutes. That is exactly what happened to a real mirrored node +# (node_ref ndebvzgeoij5t2l) on the test droplet from ONE sample — once +# blocked, record_detection_frame drops every frame it sends, apply_reward is +# a no-op, and the block is persisted by services/state_snapshot.py, so it +# survives restarts. Restore penalties (scale 1) once trust is computed from +# enough evidence to be worth acting on; the evaluator's own min-sample bar +# lives in retina_analytics.trust.TRUST_MIN_SAMPLES. +# +# Blocks already recorded in a snapshot are NOT cleared by this switch — it +# only stops new penalties. Use backend/scripts/unblock_nodes.py for those. + + +def _parse_penalty_scale(raw: str | None) -> float: + """REPUTATION_PENALTY_SCALE as a non-negative finite float, else 0. + + A malformed or negative value must not silently become "penalties on": + the safe reading of an unparseable gate here is the deployed default. + """ + if raw is None or not raw.strip(): + return 0.0 + try: + value = float(raw) + except (TypeError, ValueError): + logging.warning("REPUTATION_PENALTY_SCALE=%r is not a number — using 0 (no penalties)", raw) + return 0.0 + if not math.isfinite(value) or value < 0: + logging.warning("REPUTATION_PENALTY_SCALE=%r is not finite and >= 0 — using 0 (no penalties)", raw) + return 0.0 + return value + + +REPUTATION_PENALTY_SCALE = _parse_penalty_scale(os.getenv("REPUTATION_PENALTY_SCALE")) + # ── External ADS-B query regions ───────────────────────────────────────────── ADSB_CELL_SPACING_KM = 400.0 # Lattice cell size for grouping nodes into queries # Padding every query region carries around its members, sized against the diff --git a/backend/core/state.py b/backend/core/state.py index 33b8ce842..6ae3e842f 100644 --- a/backend/core/state.py +++ b/backend/core/state.py @@ -13,6 +13,7 @@ from retina_analytics.association import InterNodeAssociator from retina_analytics.manager import NodeAnalyticsManager +from retina_analytics.reputation import set_penalty_scale from retina_custody.crypto_backend import SignatureVerifier from retina_custody.models import NodeIdentity @@ -27,6 +28,7 @@ GROUND_TRUTH_MAX, # noqa: F401 — re-exported, used via state.GROUND_TRUTH_MAX N2_CONFIRM_MIN_EPOCHS, N2_CONFIRM_MIN_SPAN_S, + REPUTATION_PENALTY_SCALE, TRACK_HISTORY_MAX, # noqa: F401 — re-exported, used via state.TRACK_HISTORY_MAX as_num, ) @@ -148,6 +150,22 @@ node_analytics = NodeAnalyticsManager(storage_dir=COVERAGE_STORAGE_DIR, fov_mode=FOV_MODE) +# Every reputation penalty in retina-analytics is multiplied by this, and the +# default is 0 — no node can be blocked by a penalty (see +# config/constants.REPUTATION_PENALTY_SCALE for why, and +# backend/scripts/unblock_nodes.py for clearing blocks a snapshot already +# carries). It is a process-wide ClassVar, so it has to be set before +# anything reads it: at import here it lands before restore_snapshot() +# rebuilds the NodeReputation objects and before the reputation evaluator's +# first pass, which are the only two things that could act on it. Logged at +# INFO so a deploy log says plainly whether penalties are on. +set_penalty_scale(REPUTATION_PENALTY_SCALE) +logging.info( + "Node reputation penalty scale: %.3g (%s)", + REPUTATION_PENALTY_SCALE, + "penalties DISABLED — no node can be blocked" if REPUTATION_PENALTY_SCALE == 0 else "penalties active", +) + def _coverage_limit_for(node_id: str): """Shrink-only empirical prior for one node, from accumulated ADS-B fixes. diff --git a/backend/scripts/unblock_nodes.py b/backend/scripts/unblock_nodes.py new file mode 100644 index 000000000..9eb093c63 --- /dev/null +++ b/backend/scripts/unblock_nodes.py @@ -0,0 +1,246 @@ +#!/usr/bin/env python3 +"""Clear `blocked` reputations out of a persisted state snapshot. + +A node whose reputation fell below the block threshold stops being heard: +``NodeAnalyticsManager.record_detection_frame`` returns False for it, so every +frame it sends is dropped, and ``apply_reward`` is a no-op while blocked, so +the node cannot earn its way back. The block is persisted — ``reputations`` +in ``backend/data/state_snapshot.json`` — so a restart restores it. There is +no admin unblock route, and ``NodeReputation.unblock()`` only resets the +reputation to 0.3, one penalty above re-blocking. This script is the way out. + +REPUTATION_PENALTY_SCALE (default 0) stops *new* penalties from being +recorded; it deliberately does not unblock anything already written down. +That is this script's job. + +**The server must be stopped while this runs.** Two reasons, both fatal on +their own: the save loop rewrites the snapshot every 60 s and would overwrite +the edit, and the block that actually gates frames lives in memory — the file +only matters because the restart restores from it. Stop, edit, start. + +Node *ids*, not node_refs +------------------------- +``reputations`` is keyed by node_id (``retce36dbb4``, ``synth-GVL-0004``); the +analytics API renames entries to node_ref only at publication, so the ref you +read off ``/api/radar/analytics`` is not a key here. Resolve it first, inside +the running container, before you stop anything:: + + docker compose exec -w /app/backend server \\ + python3 -c "from services import node_refs as n; print(n.id_for_ref('ndebvzgeoij5t2l'))" + +(For a mirrored real node the ref comes from +``state.connected_nodes[node_id]["node_ref"]`` — see ``services/node_refs.py``, +``_mirrored_ref``.) Or skip the lookup entirely with ``--all-blocked``. + +Usage (droplet, snapshot on the backend-data volume):: + + docker compose stop server + docker run --rm -v retina-server_backend-data:/data -v $PWD/backend/scripts:/s \\ + python:3.12-slim python /s/unblock_nodes.py --path /data/state_snapshot.json --all-blocked + docker compose up -d server + +or straight on the host against the volume's mountpoint:: + + python3 backend/scripts/unblock_nodes.py \\ + --path /var/lib/docker/volumes/_backend-data/_data/state_snapshot.json \\ + --node retce36dbb4 + +Add ``--dry-run`` first to see what would change. Stdlib only, deliberately: +it has to run in a bare ``python:3.12-slim`` with the repo's dependencies +nowhere in sight. +""" + +import argparse +import hashlib +import json +import os +import sys + +# backend/scripts/unblock_nodes.py → backend/data/state_snapshot.json, the +# same path services/state_snapshot.py computes. Resolved here rather than +# imported because this script must not import the backend (no dependencies +# beyond the stdlib — see the module docstring). +_DEFAULT_PATH = os.path.join( + os.path.dirname(os.path.dirname(os.path.abspath(__file__))), + "data", + "state_snapshot.json", +) + + +def _read_envelope(path: str, force: bool) -> dict: + """Return the decoded payload dict, verifying the schema-2 checksum. + + Schema 1 (a bare payload, with a non-atomic ``.sha256`` side file) is read + too — it is written back as schema 2, which is what the server writes now. + """ + with open(path) as f: + raw = f.read() + parsed = json.loads(raw) + + if isinstance(parsed, dict) and "payload" in parsed and "sha256" in parsed: + actual = hashlib.sha256(parsed["payload"].encode()).hexdigest() + if actual != parsed["sha256"]: + msg = f"checksum mismatch (envelope says {str(parsed['sha256'])[:12]}, payload hashes to {actual[:12]})" + if not force: + raise SystemExit( + f"refusing to edit {path}: {msg}.\n" + "The file is corrupt or was written by something else; the server would " + "reject it on boot too. Re-run with --force to edit it anyway." + ) + print(f"WARNING: {msg} — continuing because --force was given", file=sys.stderr) + return json.loads(parsed["payload"]) + + # Legacy schema 1: the payload IS the file. + print(f"note: {path} is a legacy schema-1 snapshot; it will be written back as schema 2", file=sys.stderr) + return parsed + + +def _write_envelope(path: str, snap: dict) -> str: + """Write the schema-2 envelope and its side file, exactly as + services/state_snapshot.save_snapshot does: payload and checksum travel + together inside one atomic os.replace, and the ``.sha256`` side file is + refreshed after it. Returns the checksum. + """ + try: + st = os.stat(path) + except FileNotFoundError: + st = None + + payload = json.dumps(snap) + checksum = hashlib.sha256(payload.encode()).hexdigest() + envelope = json.dumps({"schema": 2, "sha256": checksum, "payload": payload}) + + tmp = path + ".tmp" + with open(tmp, "w") as f: + f.write(envelope) + _match_owner(tmp, st) + os.replace(tmp, path) + + sha_path = path + ".sha256" + sha_tmp = sha_path + ".tmp" + with open(sha_tmp, "w") as f: + f.write(checksum) + _match_owner(sha_tmp, st) + os.replace(sha_tmp, sha_path) + return checksum + + +def _match_owner(path: str, st) -> None: + """Give a freshly written temp file the mode and owner of the file it is + about to replace, so editing a container-owned snapshot as root does not + leave a file the server cannot rewrite.""" + if st is None: + return + try: + os.chmod(path, st.st_mode & 0o7777) + except OSError: + pass + try: + os.chown(path, st.st_uid, st.st_gid) + except (OSError, AttributeError): + pass + + +def _describe(entry: dict) -> str: + return ( + f"reputation={entry.get('reputation')!r} blocked={entry.get('blocked')!r} " + f"block_reason={entry.get('block_reason')!r} penalties={len(entry.get('penalties') or [])}" + ) + + +def _reset(entry: dict) -> None: + entry["reputation"] = 1.0 + entry["blocked"] = False + entry["block_reason"] = "" + entry["penalties"] = [] + # A condition that is still true (a heartbeat still stale) would not + # re-fire while its flag says "already active", so a leftover True hides + # the next onset rather than suppressing it harmlessly. Clear it: the + # next evaluator pass observes the condition fresh. + if "_condition_active" in entry: + entry["_condition_active"] = {} + + +def main(argv: list[str] | None = None) -> int: + ap = argparse.ArgumentParser( + description="Reset blocked node reputations in a RETINA state snapshot. " + "Stop the server first — the save loop rewrites the snapshot every 60 s.", + epilog="--node takes node_ids (retce36dbb4, synth-GVL-0004), NOT node_refs; " + "resolve a ref with services.node_refs.id_for_ref inside the container.", + ) + ap.add_argument( + "--path", + default=_DEFAULT_PATH, + help=f"state snapshot to edit (default: {_DEFAULT_PATH})", + ) + ap.add_argument( + "--node", + action="append", + default=[], + metavar="NODE_ID", + help="node_id to unblock; repeatable", + ) + ap.add_argument( + "--all-blocked", + action="store_true", + help="unblock every entry whose `blocked` is true", + ) + ap.add_argument("--dry-run", action="store_true", help="report what would change and write nothing") + ap.add_argument( + "--force", + action="store_true", + help="edit even if the snapshot fails its own checksum", + ) + args = ap.parse_args(argv) + + if not args.node and not args.all_blocked: + ap.error("nothing selected: pass --node NODE_ID (repeatable) and/or --all-blocked") + + if not os.path.exists(args.path): + print(f"no such snapshot: {args.path}", file=sys.stderr) + return 2 + + snap = _read_envelope(args.path, args.force) + reps = snap.get("reputations") + if not isinstance(reps, dict): + print(f"{args.path} has no `reputations` map — nothing to do", file=sys.stderr) + return 2 + + selected: list[str] = [] + missing: list[str] = [] + for node_id in args.node: + if node_id in reps: + selected.append(node_id) + else: + missing.append(node_id) + if args.all_blocked: + for node_id, entry in reps.items(): + if isinstance(entry, dict) and entry.get("blocked") and node_id not in selected: + selected.append(node_id) + + for node_id in missing: + print(f"NOT IN SNAPSHOT: {node_id}", file=sys.stderr) + + if not selected: + print(f"{args.path}: {len(reps)} reputations, none selected — nothing to change") + return 1 if missing else 0 + + for node_id in selected: + entry = reps[node_id] + print(f"{node_id}:") + print(f" before: {_describe(entry)}") + _reset(entry) + print(f" after: {_describe(entry)}") + + if args.dry_run: + print(f"\n--dry-run: {len(selected)} entries would be reset, {args.path} not written") + return 1 if missing else 0 + + checksum = _write_envelope(args.path, snap) + print(f"\n{len(selected)} entries reset; wrote {args.path} (sha256={checksum[:12]})") + print("Start the server now — it restores from this file.") + return 1 if missing else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/backend/services/node_bias.py b/backend/services/node_bias.py index 685a093b8..52d5772cd 100644 --- a/backend/services/node_bias.py +++ b/backend/services/node_bias.py @@ -51,7 +51,7 @@ import threading from collections import deque -from retina_analytics.trust import AdsReportEntry, TrustScoreState +from retina_analytics.trust import TRUST_MIN_SAMPLES, AdsReportEntry, TrustScoreState from core import state @@ -94,8 +94,14 @@ # Neutral prior for unknown nodes, and the M-of-N bar again before the real # score replaces it: below 3 samples the score quantizes to {0, 1/2, 1} and a # single unlucky residual would zero a brand-new node's solver weight. +# +# The bar itself now lives in retina-analytics, because the reputation +# evaluator there applies the same one: it used to act on any node with a +# single sample, which blocked a real node off one out-of-threshold residual. +# Sharing the constant keeps the solver's reading of a young node's trust and +# the evaluator's willingness to act on it from drifting apart. _TRUST_PRIOR = 0.5 -_TRUST_MIN_SAMPLES = 3 +_TRUST_MIN_SAMPLES = TRUST_MIN_SAMPLES _PROVENANCE = "claim_residual" diff --git a/backend/services/tasks/periodic.py b/backend/services/tasks/periodic.py index b2cf07a2e..ef879d8e3 100644 --- a/backend/services/tasks/periodic.py +++ b/backend/services/tasks/periodic.py @@ -631,6 +631,13 @@ def _cross_validate_adsb_reports(): - (0, 0) is routes/analytics.py's default for an omitted position, not a claim to have seen an aircraft off West Africa. + The 0.1 below, like every other reputation penalty, is multiplied by + NodeReputation.penalty_scale (config.constants.REPUTATION_PENALTY_SCALE), + which currently defaults to 0 — so on a default deployment this records + nothing and blocks nobody, and the gates above are what keeps that from + being the only thing standing between a truthful node and a block when the + scale is turned back up. + A sample is judged at most once because the age window (XVAL_MAX_AGE_S) is far shorter than the interval between cycles, and _adsb_truth_cycle calls this exactly once per cycle. Both halves of that must hold: shortening the diff --git a/backend/tests/conftest.py b/backend/tests/conftest.py index a6224112a..aa9eadcf8 100644 --- a/backend/tests/conftest.py +++ b/backend/tests/conftest.py @@ -537,3 +537,29 @@ async def registered_node(node_session): await node_pipeline.register_with_pipeline(node_session, node) await node_session.commit() return token, _NODE_ID + + +@pytest.fixture +def penalties_on(): + """Turn reputation penalties on for one test (or class), then restore. + + The deployed default is REPUTATION_PENALTY_SCALE=0 — core/state.py sets it + process-wide at import, so under the suite no penalty lands and no node + blocks. That is the behaviour worth having as the ambient one: a test + that never asks for penalties is then running against what production + runs. Every test that asserts a penalty *does* land asks for this fixture + explicitly, which also keeps the gate tests honest — a gate test asserting + "no penalty was recorded" proves nothing if penalties could not be + recorded at all. + + Deliberately not autouse: an autouse version would make the whole suite + exercise a configuration no deployment runs. + """ + from retina_analytics.reputation import NodeReputation, set_penalty_scale + + previous = NodeReputation.penalty_scale + set_penalty_scale(1.0) + try: + yield 1.0 + finally: + set_penalty_scale(previous) diff --git a/backend/tests/test_adsb_cross_validation_gates.py b/backend/tests/test_adsb_cross_validation_gates.py index 750d9d82a..5bba19850 100644 --- a/backend/tests/test_adsb_cross_validation_gates.py +++ b/backend/tests/test_adsb_cross_validation_gates.py @@ -21,6 +21,12 @@ from core import state from services.tasks.periodic import _cross_validate_adsb_reports +# Every test in this file is about which samples do and do not earn a penalty, +# which is only a question while penalties can be recorded at all — the +# deployed default (REPUTATION_PENALTY_SCALE=0) would pass the "penalises +# nobody" tests vacuously. +pytestmark = pytest.mark.usefixtures("penalties_on") + _NODE = "xval-node" _HEX = "cafe01" # Truth and claim ~30 km apart: unambiguously a mismatch, well past the 10 km bar. diff --git a/backend/tests/test_adsb_freshness_regressions.py b/backend/tests/test_adsb_freshness_regressions.py index ca69e3f3b..bad463811 100644 --- a/backend/tests/test_adsb_freshness_regressions.py +++ b/backend/tests/test_adsb_freshness_regressions.py @@ -211,6 +211,7 @@ def _truth(age_s=1.0): } +@pytest.mark.usefixtures("penalties_on") class TestCrossValidationCannotBeBypassed: @pytest.fixture(autouse=True) def _clean_nodes(self): diff --git a/backend/tests/test_analytics_refresh.py b/backend/tests/test_analytics_refresh.py index 9f0fb2a87..abbc8d67f 100644 --- a/backend/tests/test_analytics_refresh.py +++ b/backend/tests/test_analytics_refresh.py @@ -166,8 +166,14 @@ def test_snapshot_with_lock(self): # ── Reputation evaluations ──────────────────────────────────────────────────── +@pytest.mark.usefixtures("penalties_on") class TestReputationEvaluations: - """Test NodeReputation evaluation methods for trust, heartbeat, detection rate.""" + """Test NodeReputation evaluation methods for trust, heartbeat, detection rate. + + Penalties are off by default (REPUTATION_PENALTY_SCALE=0); this class is + about what the evaluation paths do when downrating is switched on, so it + asks for the scale explicitly. + """ def test_low_trust_blocks_node(self): from retina_analytics.reputation import NodeReputation diff --git a/backend/tests/test_node_bias.py b/backend/tests/test_node_bias.py index 034b59458..2b4248b33 100644 --- a/backend/tests/test_node_bias.py +++ b/backend/tests/test_node_bias.py @@ -137,6 +137,7 @@ def test_route_and_backend_samples_share_one_score(self, client, now_ms): assert len(ts.samples) == 3 assert ts.summary()["samples_by_provenance"] == {"self_report": 1, "claim_residual": 2} + @pytest.mark.usefixtures("penalties_on") def test_cross_validation_skips_backend_fed_samples(self, now_ms): """Claim residuals carry no position claim (lat/lon 0.0), so the external-truth cross-check must not read them as a >10 km mismatch.""" @@ -171,9 +172,13 @@ def _self_report(client, node_id, hex_, **position): ) +@pytest.mark.usefixtures("penalties_on") class TestCrossValidationRejectsUnusableFixes: """A sample the route accepted but haversine_km cannot measure must not be - charged for the distance it appears to be from truth.""" + charged for the distance it appears to be from truth. + + With penalties on, so "no penalty was recorded" is evidence about the + guard rather than about the deployed REPUTATION_PENALTY_SCALE=0.""" def _fresh_truth(self, hex_, lat, lon): # Stamped now, or the entry's age skips the sample and the assertion diff --git a/backend/tests/test_publication.py b/backend/tests/test_publication.py index 2f2a2c582..a6ba95e10 100644 --- a/backend/tests/test_publication.py +++ b/backend/tests/test_publication.py @@ -751,6 +751,7 @@ def test_the_real_only_analytics_variant_is_keyed_on_refs(self, seed_nodes): assert list(real) == [_seed_ref("ret1a2b3c4d")] +@pytest.mark.usefixtures("penalties_on") class TestAnalyticsPayloadIdentities: """/api/radar/analytics is the one surface that could give the map away. diff --git a/backend/tests/test_reputation_penalty_scale.py b/backend/tests/test_reputation_penalty_scale.py new file mode 100644 index 000000000..679f20772 --- /dev/null +++ b/backend/tests/test_reputation_penalty_scale.py @@ -0,0 +1,239 @@ +"""REPUTATION_PENALTY_SCALE: no node can be blocked while the scale is 0. + +The switch exists because one out-of-threshold trust sample was enough to +block a real mirrored node on the test droplet permanently: the evaluator acts +on any node with a sample, a single bad one scores 0.0, and 0.15 a pass at +60 s crosses the 0.2 block threshold in six minutes — after which every frame +that node sends is dropped and the block is persisted across restarts. + +Three things are pinned here, and the third is the one that is easy to get +wrong: the switch stops NEW penalties, it does not unblock what a snapshot +already carries. That is backend/scripts/unblock_nodes.py's job, and a switch +that quietly did it instead would hide from the operator which nodes had been +blocked and why. +""" + +import time + +import pytest +from retina_analytics.reputation import NodeReputation, set_penalty_scale +from retina_analytics.trust import AdsReportEntry + +from core import state +from services.tasks.periodic import _cross_validate_adsb_reports + +_NODE = "penalty-scale-node" +_HEX = "beef01" +# ~100 km apart: far past the 10 km mismatch bar, so nothing but the scale +# can be what keeps the penalty from landing. +_TRUTH = (51.5, -0.1) +_CLAIM = (51.5, 1.34) + + +@pytest.fixture(autouse=True) +def _clean(): + state.external_adsb_cache.clear() + state.node_analytics.trust_scores.pop(_NODE, None) + state.node_analytics.reputations.pop(_NODE, None) + yield + state.external_adsb_cache.clear() + state.node_analytics.trust_scores.pop(_NODE, None) + state.node_analytics.reputations.pop(_NODE, None) + + +def _seed_mismatch(): + """One fresh self-reported fix 100 km from an equally fresh external truth.""" + now = time.time() + state.node_analytics.record_adsb_correlation( + _NODE, + AdsReportEntry( + timestamp_ms=int(now * 1000), + predicted_delay=100.0, + predicted_doppler=0.0, + measured_delay=100.5, + measured_doppler=0.0, + adsb_hex=_HEX, + adsb_lat=_CLAIM[0], + adsb_lon=_CLAIM[1], + ), + ) + state.external_adsb_cache[_HEX] = { + "lat": _TRUTH[0], + "lon": _TRUTH[1], + "alt_m": 10000.0, + "last_seen_ms": int(now * 1000), + } + rep = NodeReputation(node_id=_NODE) + state.node_analytics.reputations[_NODE] = rep + return rep + + +# ── (a) The deployed default records nothing ───────────────────────────────── + + +class TestDefaultScaleRecordsNoPenalty: + def test_a_100km_mismatch_costs_nothing(self): + """core/state.py sets the scale from the environment at import, and + the suite runs with it unset — so this is the deployed configuration, + not a fixture's idea of one.""" + rep = _seed_mismatch() + _cross_validate_adsb_reports() + + assert rep.reputation == 1.0 + assert rep.penalties == [] + assert rep.blocked is False + + def test_the_same_mismatch_does_land_with_penalties_on(self, penalties_on): + """The control: without this the test above would pass just as well if + the gates had silently stopped finding the mismatch at all.""" + rep = _seed_mismatch() + _cross_validate_adsb_reports() + + assert rep.reputation < 1.0 + assert len(rep.penalties) == 1 + + def test_rewards_still_work_at_zero_scale(self): + """Only downrating is switched off — a node must still be able to + climb back, or the switch would freeze every reputation where it sits.""" + rep = NodeReputation(node_id="reward-node", reputation=0.5) + rep.evaluate_trust(0.9) + assert rep.reputation > 0.5 + + +# ── (b) Parsing the env var ────────────────────────────────────────────────── + + +class TestPenaltyScaleParsing: + """Straight at the helper: importing config.constants fresh per case would + mean reloading a module half the backend holds references into.""" + + def test_unset_is_zero(self): + from config.constants import _parse_penalty_scale + + assert _parse_penalty_scale(None) == 0.0 + assert _parse_penalty_scale("") == 0.0 + assert _parse_penalty_scale(" ") == 0.0 + + def test_one_restores_the_historical_behaviour(self): + from config.constants import _parse_penalty_scale + + assert _parse_penalty_scale("1") == 1.0 + assert _parse_penalty_scale("0.5") == 0.5 + assert _parse_penalty_scale("0") == 0.0 + + @pytest.mark.parametrize("raw", ["-1", "nan", "-inf", "banana", "1,0"]) + def test_a_value_that_is_not_a_scale_falls_back_to_zero(self, raw, caplog): + """Never the other way round: an unparseable gate must not read as + "penalties on", which is the setting that blocks nodes.""" + from config.constants import _parse_penalty_scale + + with caplog.at_level("WARNING"): + assert _parse_penalty_scale(raw) == 0.0 + assert any("REPUTATION_PENALTY_SCALE" in r.getMessage() for r in caplog.records) + + def test_infinity_is_refused(self): + from config.constants import _parse_penalty_scale + + assert _parse_penalty_scale("inf") == 0.0 + + def test_the_deployed_default_is_zero(self): + from config.constants import REPUTATION_PENALTY_SCALE + + assert REPUTATION_PENALTY_SCALE == 0.0 + assert NodeReputation.penalty_scale == 0.0, "core/state.py must push the constant into the library at import" + + +# ── (c) The switch does not unblock anything ───────────────────────────────── + + +class TestRestoredBlocksSurvive: + """A block a snapshot carries is state, not a live verdict — only + backend/scripts/unblock_nodes.py clears it.""" + + _NID = "restored-blocked-node" + + @pytest.fixture(autouse=True) + def _clean_restored(self): + yield + state.node_analytics.reputations.pop(self._NID, None) + state.node_analytics.trust_scores.pop(self._NID, None) + state.node_analytics.metrics.pop(self._NID, None) + + def test_a_restored_block_is_unchanged_by_an_evaluator_pass(self): + # Exactly the shape services/state_snapshot.restore_snapshot builds. + entry = { + "node_id": self._NID, + "reputation": 0.1, + "blocked": True, + "block_reason": "Reputation 0.10 below threshold", + "penalties": [{"time": 1.0, "amount": 0.15, "reason": "Trust score critically low: 0.000"}], + "_condition_active": {"heartbeat_stale": False}, + } + rep = NodeReputation(**entry) + state.node_analytics.reputations[self._NID] = rep + + # A trust score that would charge 0.15 a pass at scale 1. + from retina_analytics.trust import TrustScoreState + + bad = TrustScoreState(node_id=self._NID) + for _ in range(5): + bad.add_sample( + AdsReportEntry( + timestamp_ms=int(time.time() * 1000), + predicted_delay=100.0, + predicted_doppler=0.0, + measured_delay=900.0, # way past the 5 us threshold + measured_doppler=0.0, + adsb_hex=_HEX, + adsb_lat=0.0, + adsb_lon=0.0, + ) + ) + assert bad.score == 0.0 + state.node_analytics.trust_scores[self._NID] = bad + + state.node_analytics.evaluate_reputations() + + assert rep.blocked is True, "the switch must not silently unblock — the script does" + assert rep.block_reason == entry["block_reason"] + assert rep.reputation == 0.1, "no new penalty, and no reward while blocked" + assert len(rep.penalties) == 1, "nothing new recorded" + + def test_the_same_pass_would_have_charged_it_with_penalties_on(self, penalties_on): + from retina_analytics.trust import TrustScoreState + + rep = NodeReputation(node_id=self._NID, reputation=0.5) + state.node_analytics.reputations[self._NID] = rep + bad = TrustScoreState(node_id=self._NID) + for _ in range(5): + bad.add_sample( + AdsReportEntry( + timestamp_ms=int(time.time() * 1000), + predicted_delay=100.0, + predicted_doppler=0.0, + measured_delay=900.0, + measured_doppler=0.0, + adsb_hex=_HEX, + adsb_lat=0.0, + adsb_lon=0.0, + ) + ) + state.node_analytics.trust_scores[self._NID] = bad + + state.node_analytics.evaluate_reputations() + + assert rep.reputation < 0.5 + assert rep.penalties + + +def test_set_penalty_scale_refuses_a_nonsense_value(): + """The library validates too, so a caller that is not config.constants + cannot push the estate into an undefined state.""" + previous = NodeReputation.penalty_scale + try: + for bad in (-1.0, float("nan"), float("inf"), "x"): + with pytest.raises(ValueError): + set_penalty_scale(bad) + assert NodeReputation.penalty_scale == previous + finally: + set_penalty_scale(previous) diff --git a/backend/tests/test_state_snapshot.py b/backend/tests/test_state_snapshot.py index 859000750..7377d2f85 100644 --- a/backend/tests/test_state_snapshot.py +++ b/backend/tests/test_state_snapshot.py @@ -6,6 +6,7 @@ from collections import deque from unittest.mock import patch +import pytest from retina_analytics.reputation import NodeReputation from retina_analytics.trust import AdsReportEntry, TrustScoreState @@ -71,6 +72,7 @@ def test_trust_scores_survive_round_trip(self, tmp_path): # Cleanup state.node_analytics.trust_scores.pop("node-42", None) + @pytest.mark.usefixtures("penalties_on") def test_reputations_survive_round_trip(self, tmp_path): from core import state diff --git a/backend/tests/test_unblock_nodes.py b/backend/tests/test_unblock_nodes.py new file mode 100644 index 000000000..0de898efc --- /dev/null +++ b/backend/tests/test_unblock_nodes.py @@ -0,0 +1,189 @@ +"""backend/scripts/unblock_nodes.py — the only way out of a persisted block. + +REPUTATION_PENALTY_SCALE=0 stops new penalties; it deliberately leaves blocks +a snapshot already carries alone. So this script has to produce a file the +server will actually accept on boot: schema-2 envelope, self-consistent +checksum, and entries that still construct as NodeReputation(**entry) the way +services/state_snapshot.restore_snapshot does. +""" + +import hashlib +import importlib.util +import json +from dataclasses import asdict +from pathlib import Path + +import pytest +from retina_analytics.reputation import NodeReputation + +_SCRIPT = Path(__file__).resolve().parent.parent / "scripts" / "unblock_nodes.py" + + +@pytest.fixture(scope="module") +def unblock(): + """Load the script as a module. + + By path rather than by import: it is a stdlib-only operator tool that has + to run in a bare python:3.12-slim against a docker volume, so it is not on + the backend's import path and must not become dependent on being there. + """ + spec = importlib.util.spec_from_file_location("unblock_nodes", _SCRIPT) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def _blocked_entry(node_id: str) -> dict: + """A reputation as a real snapshot carries it — built through the dataclass + and asdict() so the shape cannot drift from what the server writes.""" + rep = NodeReputation( + node_id=node_id, + reputation=0.05, + blocked=True, + block_reason="Reputation 0.05 below threshold", + ) + rep.penalties = [ + {"time": 1.0, "amount": 0.15, "reason": "Trust score critically low: 0.000", "reputation_after": 0.85}, + {"time": 2.0, "amount": 0.15, "reason": "Trust score critically low: 0.000", "reputation_after": 0.70}, + ] + # What live entries look like: the evaluator records the state of each + # named condition every pass, keyed per neighbour. + rep._condition_active = { + "heartbeat_stale": True, + "neighbour_inconsistent:synth-GVL-0002": False, + "neighbour_inconsistent:synth-GVL-0003": False, + } + return asdict(rep) + + +def _healthy_entry(node_id: str) -> dict: + rep = NodeReputation(node_id=node_id, reputation=0.9) + rep._condition_active = {"heartbeat_stale": False} + return asdict(rep) + + +def _write_snapshot(path: Path, reputations: dict) -> None: + """A schema-2 envelope, written the way services/state_snapshot does.""" + payload = json.dumps({"saved_at": 1.0, "reputations": reputations, "trust_scores": {}}) + checksum = hashlib.sha256(payload.encode()).hexdigest() + path.write_text(json.dumps({"schema": 2, "sha256": checksum, "payload": payload})) + Path(str(path) + ".sha256").write_text(checksum) + + +def _read_payload(path: Path) -> dict: + """Read the file back the way the server does, verifying as it goes.""" + envelope = json.loads(path.read_text()) + assert envelope["schema"] == 2 + actual = hashlib.sha256(envelope["payload"].encode()).hexdigest() + assert actual == envelope["sha256"], "the file no longer verifies against its own checksum" + assert Path(str(path) + ".sha256").read_text().strip() == actual, "the side file is stale" + return json.loads(envelope["payload"]) + + +@pytest.fixture +def snapshot(tmp_path): + path = tmp_path / "state_snapshot.json" + _write_snapshot( + path, + { + # node_id keys, not node_refs: the analytics API renames entries to + # node_ref only at publication. + "retce36dbb4": _blocked_entry("retce36dbb4"), + "synth-GVL-0004": _blocked_entry("synth-GVL-0004"), + "synth-GVL-0005": _healthy_entry("synth-GVL-0005"), + }, + ) + return path + + +def test_a_named_node_is_reset_and_the_rest_are_untouched(unblock, snapshot): + before = _read_payload(snapshot) + + assert unblock.main(["--path", str(snapshot), "--node", "retce36dbb4"]) == 0 + + after = _read_payload(snapshot) + reset = after["reputations"]["retce36dbb4"] + assert reset["reputation"] == 1.0 + assert reset["blocked"] is False + assert reset["block_reason"] == "" + assert reset["penalties"] == [] + assert reset["_condition_active"] == {}, "a stale active flag would hide the next onset" + + for other in ("synth-GVL-0004", "synth-GVL-0005"): + assert after["reputations"][other] == before["reputations"][other] + + +def test_all_blocked_resets_every_block_and_nothing_else(unblock, snapshot): + assert unblock.main(["--path", str(snapshot), "--all-blocked"]) == 0 + + after = _read_payload(snapshot)["reputations"] + assert after["retce36dbb4"]["blocked"] is False + assert after["synth-GVL-0004"]["blocked"] is False + # The healthy node was never blocked, so --all-blocked must not have + # touched its reputation either. + assert after["synth-GVL-0005"]["reputation"] == 0.9 + + +def test_dry_run_writes_nothing(unblock, snapshot): + original = snapshot.read_text() + original_sha = Path(str(snapshot) + ".sha256").read_text() + + assert unblock.main(["--path", str(snapshot), "--all-blocked", "--dry-run"]) == 0 + + assert snapshot.read_text() == original + assert Path(str(snapshot) + ".sha256").read_text() == original_sha + + +def test_the_result_still_restores_as_a_NodeReputation(unblock, snapshot): + """restore_snapshot does NodeReputation(**entry) — an edit that added or + dropped a key would fail there, at boot, with the block still in place.""" + assert unblock.main(["--path", str(snapshot), "--all-blocked"]) == 0 + + for node_id, entry in _read_payload(snapshot)["reputations"].items(): + rep = NodeReputation(**entry) + assert rep.node_id == node_id + assert rep.blocked is False + + +def test_an_unknown_node_exits_non_zero(unblock, snapshot, capsys): + rc = unblock.main(["--path", str(snapshot), "--node", "retce36dbb4", "--node", "no-such-node"]) + assert rc != 0 + assert "no-such-node" in capsys.readouterr().err + # The one that did exist is still reset — a typo in a second --node must + # not silently roll back the unblock the operator came for. + assert _read_payload(snapshot)["reputations"]["retce36dbb4"]["blocked"] is False + + +def test_a_corrupt_snapshot_is_refused_unless_forced(unblock, snapshot): + envelope = json.loads(snapshot.read_text()) + envelope["sha256"] = "0" * 64 + snapshot.write_text(json.dumps(envelope)) + + with pytest.raises(SystemExit): + unblock.main(["--path", str(snapshot), "--all-blocked"]) + assert json.loads(snapshot.read_text())["sha256"] == "0" * 64, "nothing should have been written" + + assert unblock.main(["--path", str(snapshot), "--all-blocked", "--force"]) == 0 + # Forcing rewrites it with a checksum that matches, so the server will + # accept the repaired file on boot. + assert _read_payload(snapshot)["reputations"]["retce36dbb4"]["blocked"] is False + + +def test_a_legacy_schema_1_snapshot_is_upgraded(unblock, tmp_path): + path = tmp_path / "legacy.json" + payload = json.dumps({"saved_at": 1.0, "reputations": {"retce36dbb4": _blocked_entry("retce36dbb4")}}) + path.write_text(payload) + Path(str(path) + ".sha256").write_text(hashlib.sha256(payload.encode()).hexdigest()) + + assert unblock.main(["--path", str(path), "--all-blocked"]) == 0 + + assert _read_payload(path)["reputations"]["retce36dbb4"]["blocked"] is False + + +def test_selecting_nothing_is_an_error(unblock, snapshot): + with pytest.raises(SystemExit): + unblock.main(["--path", str(snapshot)]) + + +def test_a_missing_snapshot_exits_non_zero(unblock, tmp_path): + assert unblock.main(["--path", str(tmp_path / "nope.json"), "--all-blocked"]) == 2 diff --git a/docs/architecture.md b/docs/architecture.md index 447dc6865..2bf0822c1 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -130,6 +130,7 @@ gitignored `backend/.env`; unset = the safe default): | `ADSB_SEED_MODE` | `off/shadow/active` | `off` | `active` | ADS-B-seeded detection assignment: verified lit tracklets leave dark pairing, re-emitted as `mn-adsb-*` seeded solves | | `KNOWN_LANE_MODE` | `off/shadow/binding` | `binding` | `binding` | identity-first known-target claiming: per-frame detections bound to live ADS-B hexes (`state.known_claims`) leave the dark pool before the tracker/association ever see them | | `TRACK_SMOOTHER` | `kf/ewma/off` | `kf` | `kf` | display smoothing for multinode tracks (`ewma` is the rollback) | +| `REPUTATION_PENALTY_SCALE` | float ≥ 0 | `0` | `0` | multiplier on every node-reputation penalty; `0` means no node can be blocked, `1` is the historical behaviour (temporary — see [`runbook.md`](runbook.md)) | `shadow` computes and counts a stage's verdicts (exposed in `/api/test/solver-stats`) without letting them bind — the standard soak step diff --git a/docs/runbook.md b/docs/runbook.md index a46d10994..dcec1d488 100644 --- a/docs/runbook.md +++ b/docs/runbook.md @@ -111,6 +111,11 @@ arriving on production means it is working; confirm it positively by checking that the real nodes appear in the test droplet's `/api/radar/analytics`, which names them by `node_ref` rather than by node id. +A mirrored node that appears there with `total_frames` stuck at 0 while the +others rise is almost certainly **blocked** in its reputation, not missing from +the mirror: check `cross_node.blocked_nodes`, and see "Node reputation penalties" +below for the switch and the unblock procedure. + --- ## Server basics @@ -864,6 +869,74 @@ stage binding; flip to `active` only after the shadow soak looks sane. Instant rollbacks: any mode flag back to `shadow`/`off`, and `TRACK_SMOOTHER=ewma` for display smoothing. +### Node reputation penalties (`REPUTATION_PENALTY_SCALE`) + +Every node carries a reputation, and a node whose reputation falls below 0.2 is +**blocked**: `record_detection_frame` drops every frame it sends, `apply_reward` +is a no-op so it cannot climb back, and the block is written to the state +snapshot, so a restart restores it. There is no admin unblock route. + +`REPUTATION_PENALTY_SCALE` multiplies every penalty that can get a node there — +trust (0.15/0.05 per 60 s evaluator pass), stale heartbeat (0.1), high detection +rate (0.05), neighbour inconsistency (0.08), ADS-B cross-validation (0.1). +Rewards are not scaled. + +| Value | Effect | +|---|---| +| `0` (default) | no penalty is recorded at all — no node can be blocked by any of these paths | +| `1` | the historical behaviour | + +**The 0 is temporary** (set 2026-09-18). The only trust input today is a single +claim residual from the identity-first lane, and one out-of-threshold residual +scores a node 0.0 — 0.15 a pass then crosses the block threshold in six minutes. +That is how a real mirrored node on the test droplet was blocked permanently off +one sample. Put it back to `1` once trust is computed from enough evidence to +act on; the evaluator's min-sample bar is `TRUST_MIN_SAMPLES` in retina-analytics. + +To flip it: edit `backend/.env` (or the environment's compose overlay), then +`docker compose up -d` — env-only, no `--build`. The startup log says which way +it went (`Node reputation penalty scale: …`). + +Read it back from `/api/radar/analytics`: each node's `reputation` block carries +`penalty_scale` (0 here explains a reputation that never moves), `reputation`, +`blocked` and `n_penalties`; the fleet-wide list of blocked nodes is +`cross_node.blocked_nodes`. + +#### Unblocking a node + +The switch stops *new* penalties. It does not clear a block the snapshot already +carries — that is deliberate, so the operator sees which nodes were blocked and +why. Use `backend/scripts/unblock_nodes.py`. + +The snapshot keys reputations by **node_id**, while the analytics API names +nodes by `node_ref`. Resolve the ref first, while the server is still up: + +```bash +docker compose exec -w /app/backend server \ + python3 -c "from services import node_refs as n; print(n.id_for_ref(''))" +``` + +Then stop the server before editing — the save loop rewrites the snapshot every +60 s, and the block that actually gates frames lives in memory, so the file only +takes effect on the restart: + +```bash +docker compose stop server +docker run --rm -v retina-server_backend-data:/data -v $PWD/backend/scripts:/s \ + python:3.12-slim python /s/unblock_nodes.py --path /data/state_snapshot.json --all-blocked +docker compose up -d server +``` + +`--node ` (repeatable) instead of `--all-blocked` to pick individual +nodes, and `--dry-run` first to see what would change. The script is stdlib-only, +so it also runs straight on the host against the volume's mountpoint +(`/var/lib/docker/volumes/_backend-data/_data/state_snapshot.json`). It +verifies the snapshot's checksum before touching it, rewrites it atomically with +a fresh one, and resets each selected entry to reputation 1.0, unblocked, with an +empty penalty ledger. Confirm afterwards that `cross_node.blocked_nodes` in +`/api/radar/analytics` no longer names the node and that its `total_frames` +starts rising. + ### Empirical coverage / learned FOV health Per-node learned FOV state persists on the `coverage_data` volume and is From 7edd0e82eb370c80cc54d87fb56431ef9cdd1621 Mon Sep 17 00:00:00 2001 From: Jehan Azad Date: Fri, 18 Sep 2026 23:50:52 +0000 Subject: [PATCH 2/6] Bump retina-analytics to the penalty scale and 3-sample bar Picks up retina-analytics PR #34: NodeReputation.penalty_scale + set_penalty_scale(), which core/state.py now sets from REPUTATION_PENALTY_SCALE, and TRUST_MIN_SAMPLES, which node_bias imports. Co-Authored-By: Claude Fable 5.1 --- libs/retina-analytics | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/libs/retina-analytics b/libs/retina-analytics index 6df7c3df2..8b79a1a04 160000 --- a/libs/retina-analytics +++ b/libs/retina-analytics @@ -1 +1 @@ -Subproject commit 6df7c3df2e42271b5bccfda053c01835b0f41006 +Subproject commit 8b79a1a043641e2ce491b3e8bfaadcf3800b8410 From ff8997b57ff36f381d2a35ccd555d39a7d031c33 Mon Sep 17 00:00:00 2001 From: Jehan Azad Date: Fri, 18 Sep 2026 23:51:32 +0000 Subject: [PATCH 3/6] Say how to find a mirrored node's id for the unblock script id_for_ref answers from the registry, and a mirrored node has no row there: on the test droplet it printed None for ndebvzgeoij5t2l. The key was found by hand (the one ret* id among the synth-* blocked entries, whose single trust sample reproduces the published rms_delay_error_us). Co-Authored-By: Claude Fable 5.1 --- backend/scripts/unblock_nodes.py | 18 +++++++++++------- docs/runbook.md | 9 ++++++++- 2 files changed, 19 insertions(+), 8 deletions(-) diff --git a/backend/scripts/unblock_nodes.py b/backend/scripts/unblock_nodes.py index 9eb093c63..1e526f56d 100644 --- a/backend/scripts/unblock_nodes.py +++ b/backend/scripts/unblock_nodes.py @@ -22,15 +22,19 @@ ------------------------- ``reputations`` is keyed by node_id (``retce36dbb4``, ``synth-GVL-0004``); the analytics API renames entries to node_ref only at publication, so the ref you -read off ``/api/radar/analytics`` is not a key here. Resolve it first, inside -the running container, before you stop anything:: +read off ``/api/radar/analytics`` is not a key here. For a node registered on +this environment, resolve it from the registry before you stop anything:: docker compose exec -w /app/backend server \\ - python3 -c "from services import node_refs as n; print(n.id_for_ref('ndebvzgeoij5t2l'))" - -(For a mirrored real node the ref comes from -``state.connected_nodes[node_id]["node_ref"]`` — see ``services/node_refs.py``, -``_mirrored_ref``.) Or skip the lookup entirely with ``--all-blocked``. + python3 -c "from services import node_refs as n; print(n.id_for_ref(''))" + +A *mirrored* real node has no registry row here: its ref lives only in the +running process (``state.connected_nodes[node_id]["node_ref"]``, see +``services/node_refs.py`` ``_mirrored_ref``), which a ``docker compose exec`` +cannot see, so the lookup above prints None for it. Pick its key by hand +instead — it is a ``ret*`` id among the ``synth-*`` ones, and its +``trust_scores`` entry reproduces the ref's ``rms_delay_error_us`` — or skip +the question entirely with ``--all-blocked``. Usage (droplet, snapshot on the backend-data volume):: diff --git a/docs/runbook.md b/docs/runbook.md index dcec1d488..f5457e939 100644 --- a/docs/runbook.md +++ b/docs/runbook.md @@ -909,13 +909,20 @@ carries — that is deliberate, so the operator sees which nodes were blocked an why. Use `backend/scripts/unblock_nodes.py`. The snapshot keys reputations by **node_id**, while the analytics API names -nodes by `node_ref`. Resolve the ref first, while the server is still up: +nodes by `node_ref`. For a node registered on this environment, resolve the ref +from the registry while the server is still up: ```bash docker compose exec -w /app/backend server \ python3 -c "from services import node_refs as n; print(n.id_for_ref(''))" ``` +A **mirrored** real node has no registry row here; its ref lives only in the +running process (`state.connected_nodes`), which an `exec` cannot see, so that +prints `None`. Pick its key by hand instead: it is a `ret*` id among the +`synth-*` ones, and its `trust_scores` entry in the snapshot reproduces the +ref's `rms_delay_error_us` from `/api/radar/analytics`. Or use `--all-blocked`. + Then stop the server before editing — the save loop rewrites the snapshot every 60 s, and the block that actually gates frames lives in memory, so the file only takes effect on the restart: From 46772e0db87a82191cc2900ad46c195576bfacc8 Mon Sep 17 00:00:00 2001 From: Jehan Azad Date: Sat, 19 Sep 2026 00:02:17 +0000 Subject: [PATCH 4/6] Log the penalty scale from startup, where logging exists core/state.py runs at import, before main.py's logging.basicConfig, so its INFO line never reached the container log on the test droplet. The scale is still set at import (it must precede the snapshot restore); the line now prints from the lifespan, just before restore_snapshot(). Co-Authored-By: Claude Fable 5.1 --- backend/core/state.py | 10 +++------- backend/main.py | 10 ++++++++++ 2 files changed, 13 insertions(+), 7 deletions(-) diff --git a/backend/core/state.py b/backend/core/state.py index 6ae3e842f..d429be24e 100644 --- a/backend/core/state.py +++ b/backend/core/state.py @@ -157,14 +157,10 @@ # carries). It is a process-wide ClassVar, so it has to be set before # anything reads it: at import here it lands before restore_snapshot() # rebuilds the NodeReputation objects and before the reputation evaluator's -# first pass, which are the only two things that could act on it. Logged at -# INFO so a deploy log says plainly whether penalties are on. +# first pass, which are the only two things that could act on it. Logged +# from main.py's startup, not here: this runs before logging.basicConfig, +# where an INFO line is dropped. set_penalty_scale(REPUTATION_PENALTY_SCALE) -logging.info( - "Node reputation penalty scale: %.3g (%s)", - REPUTATION_PENALTY_SCALE, - "penalties DISABLED — no node can be blocked" if REPUTATION_PENALTY_SCALE == 0 else "penalties active", -) def _coverage_limit_for(node_id: str): diff --git a/backend/main.py b/backend/main.py index 0eb822f4b..f9039d7ff 100644 --- a/backend/main.py +++ b/backend/main.py @@ -36,6 +36,7 @@ from fastapi.responses import JSONResponse from starlette.middleware.base import BaseHTTPMiddleware +from config.constants import REPUTATION_PENALTY_SCALE from core import state from core.env_parsing import parse_comma_list from pipeline.passive_radar import DEFAULT_NODE_CONFIG, PassiveRadarPipeline @@ -185,6 +186,15 @@ async def lifespan(app: FastAPI): await prime_pipeline_at_startup() + # Set at import in core/state.py (it must precede the restore below); said + # here, after logging is configured, so a deploy log says plainly whether + # node reputation penalties are on. + logging.info( + "Node reputation penalty scale: %.3g (%s)", + REPUTATION_PENALTY_SCALE, + "penalties DISABLED — no node can be blocked" if REPUTATION_PENALTY_SCALE == 0 else "penalties active", + ) + # Restore persisted state before accepting connections restored = restore_snapshot() From b5bf9d92313e7a464898e1d8c51ea7ccadf02ff4 Mon Sep 17 00:00:00 2001 From: Jehan Azad Date: Sat, 19 Sep 2026 00:19:31 +0000 Subject: [PATCH 5/6] Keep the real node id out of the unblock script and its test test_no_real_identities caught retce36dbb4 in the docstring, the argparse epilog and the test fixture. This repo is public; a placeholder says the same thing. Co-Authored-By: Claude Fable 5.1 --- backend/scripts/unblock_nodes.py | 6 +++--- backend/tests/test_unblock_nodes.py | 18 +++++++++--------- 2 files changed, 12 insertions(+), 12 deletions(-) diff --git a/backend/scripts/unblock_nodes.py b/backend/scripts/unblock_nodes.py index 1e526f56d..86e531b04 100644 --- a/backend/scripts/unblock_nodes.py +++ b/backend/scripts/unblock_nodes.py @@ -20,7 +20,7 @@ Node *ids*, not node_refs ------------------------- -``reputations`` is keyed by node_id (``retce36dbb4``, ``synth-GVL-0004``); the +``reputations`` is keyed by node_id (``ret``, ``synth-GVL-0004``); the analytics API renames entries to node_ref only at publication, so the ref you read off ``/api/radar/analytics`` is not a key here. For a node registered on this environment, resolve it from the registry before you stop anything:: @@ -47,7 +47,7 @@ python3 backend/scripts/unblock_nodes.py \\ --path /var/lib/docker/volumes/_backend-data/_data/state_snapshot.json \\ - --node retce36dbb4 + --node Add ``--dry-run`` first to see what would change. Stdlib only, deliberately: it has to run in a bare ``python:3.12-slim`` with the repo's dependencies @@ -169,7 +169,7 @@ def main(argv: list[str] | None = None) -> int: ap = argparse.ArgumentParser( description="Reset blocked node reputations in a RETINA state snapshot. " "Stop the server first — the save loop rewrites the snapshot every 60 s.", - epilog="--node takes node_ids (retce36dbb4, synth-GVL-0004), NOT node_refs; " + epilog="--node takes node_ids (a ret id, synth-GVL-0004), NOT node_refs; " "resolve a ref with services.node_refs.id_for_ref inside the container.", ) ap.add_argument( diff --git a/backend/tests/test_unblock_nodes.py b/backend/tests/test_unblock_nodes.py index 0de898efc..9d31cc9a2 100644 --- a/backend/tests/test_unblock_nodes.py +++ b/backend/tests/test_unblock_nodes.py @@ -88,7 +88,7 @@ def snapshot(tmp_path): { # node_id keys, not node_refs: the analytics API renames entries to # node_ref only at publication. - "retce36dbb4": _blocked_entry("retce36dbb4"), + "blocked-real-1": _blocked_entry("blocked-real-1"), "synth-GVL-0004": _blocked_entry("synth-GVL-0004"), "synth-GVL-0005": _healthy_entry("synth-GVL-0005"), }, @@ -99,10 +99,10 @@ def snapshot(tmp_path): def test_a_named_node_is_reset_and_the_rest_are_untouched(unblock, snapshot): before = _read_payload(snapshot) - assert unblock.main(["--path", str(snapshot), "--node", "retce36dbb4"]) == 0 + assert unblock.main(["--path", str(snapshot), "--node", "blocked-real-1"]) == 0 after = _read_payload(snapshot) - reset = after["reputations"]["retce36dbb4"] + reset = after["reputations"]["blocked-real-1"] assert reset["reputation"] == 1.0 assert reset["blocked"] is False assert reset["block_reason"] == "" @@ -117,7 +117,7 @@ def test_all_blocked_resets_every_block_and_nothing_else(unblock, snapshot): assert unblock.main(["--path", str(snapshot), "--all-blocked"]) == 0 after = _read_payload(snapshot)["reputations"] - assert after["retce36dbb4"]["blocked"] is False + assert after["blocked-real-1"]["blocked"] is False assert after["synth-GVL-0004"]["blocked"] is False # The healthy node was never blocked, so --all-blocked must not have # touched its reputation either. @@ -146,12 +146,12 @@ def test_the_result_still_restores_as_a_NodeReputation(unblock, snapshot): def test_an_unknown_node_exits_non_zero(unblock, snapshot, capsys): - rc = unblock.main(["--path", str(snapshot), "--node", "retce36dbb4", "--node", "no-such-node"]) + rc = unblock.main(["--path", str(snapshot), "--node", "blocked-real-1", "--node", "no-such-node"]) assert rc != 0 assert "no-such-node" in capsys.readouterr().err # The one that did exist is still reset — a typo in a second --node must # not silently roll back the unblock the operator came for. - assert _read_payload(snapshot)["reputations"]["retce36dbb4"]["blocked"] is False + assert _read_payload(snapshot)["reputations"]["blocked-real-1"]["blocked"] is False def test_a_corrupt_snapshot_is_refused_unless_forced(unblock, snapshot): @@ -166,18 +166,18 @@ def test_a_corrupt_snapshot_is_refused_unless_forced(unblock, snapshot): assert unblock.main(["--path", str(snapshot), "--all-blocked", "--force"]) == 0 # Forcing rewrites it with a checksum that matches, so the server will # accept the repaired file on boot. - assert _read_payload(snapshot)["reputations"]["retce36dbb4"]["blocked"] is False + assert _read_payload(snapshot)["reputations"]["blocked-real-1"]["blocked"] is False def test_a_legacy_schema_1_snapshot_is_upgraded(unblock, tmp_path): path = tmp_path / "legacy.json" - payload = json.dumps({"saved_at": 1.0, "reputations": {"retce36dbb4": _blocked_entry("retce36dbb4")}}) + payload = json.dumps({"saved_at": 1.0, "reputations": {"blocked-real-1": _blocked_entry("blocked-real-1")}}) path.write_text(payload) Path(str(path) + ".sha256").write_text(hashlib.sha256(payload.encode()).hexdigest()) assert unblock.main(["--path", str(path), "--all-blocked"]) == 0 - assert _read_payload(path)["reputations"]["retce36dbb4"]["blocked"] is False + assert _read_payload(path)["reputations"]["blocked-real-1"]["blocked"] is False def test_selecting_nothing_is_an_error(unblock, snapshot): From a205806c0f33b03c02a67bc9a39c85b9d8fcabbf Mon Sep 17 00:00:00 2001 From: Jehan Azad Date: Sat, 19 Sep 2026 00:20:13 +0000 Subject: [PATCH 6/6] Fix the two backend tests #506 left red on main Unrelated to the penalty scale; here so this PR's CI can go green. - test_health_returns_ok: #506 added synthetic_fleet to /api/health (routes/health.py) and the test still expected {"status": "ok"} alone. Assert the contract instead: status ok, synthetic_fleet a bool, no other keys. - test_no_real_identities: #506's mapShell.test.tsx rendered /nodes/; any path segment does for that assertion. Co-Authored-By: Claude Fable 5.1 --- backend/tests/test_routes.py | 8 +++++++- dashboard/src/test/mapShell.test.tsx | 2 +- 2 files changed, 8 insertions(+), 2 deletions(-) diff --git a/backend/tests/test_routes.py b/backend/tests/test_routes.py index 3086690c0..fdb238ba0 100644 --- a/backend/tests/test_routes.py +++ b/backend/tests/test_routes.py @@ -12,7 +12,13 @@ class TestHealth: def test_health_returns_ok(self, client): r = client.get("/api/health") assert r.status_code == 200 - assert r.json() == {"status": "ok"} + body = r.json() + assert body["status"] == "ok" + # Not health: the console reads this to decide whether to show the + # simulator surface (routes/health.py). Present and boolean is the + # contract; its value depends on SYNTHETIC_FLEET_ENABLED. + assert isinstance(body["synthetic_fleet"], bool) + assert set(body) == {"status", "synthetic_fleet"} # ── Detections ──────────────────────────────────────────────────────────────── diff --git a/dashboard/src/test/mapShell.test.tsx b/dashboard/src/test/mapShell.test.tsx index a0583b0d9..81b5353f1 100644 --- a/dashboard/src/test/mapShell.test.tsx +++ b/dashboard/src/test/mapShell.test.tsx @@ -104,7 +104,7 @@ describe("the header's name for a page", () => { }); it("still names a page that owns its subtree by its first segment", () => { - const { container } = renderAt("/nodes/ret72b1909e"); + const { container } = renderAt("/nodes/nde0example0001"); expect(container.textContent).toContain("Node Detail"); }); });