diff --git a/contrib/derate_amp_control.py b/contrib/derate_amp_control.py index d03f333..071ac02 100755 --- a/contrib/derate_amp_control.py +++ b/contrib/derate_amp_control.py @@ -123,6 +123,21 @@ class Config: forecast_confidence_k: float = 2.0 min_amps: float = 6.0 + # Calibration probe. The degradation watch can only compare windows the + # charge current held steady through — and on an install where this + # daemon caps often, those are scarce and land at whatever current the + # capping happened to choose. A probe manufactures one deliberately: + # hold a current low enough that nothing wants to trim it, long enough + # for the handle to reach its plateau, on a fixed cadence. The result is + # a repeatable operating point that month-over-month comparison can use + # without extrapolating across currents. + # + # Disabled at 0 — this is opt-in, and an install whose charges already + # run unregulated at full rate does not need it. + probe_amps: float = 0.0 + probe_interval_days: float = 30.0 + probe_hold_min: float = 40.0 + @dataclass class State: @@ -137,6 +152,11 @@ class State: last_session_state: str | None = None restore_attempts: int = 0 last_step_up_ts: float | None = None + # Calibration probe: when the current hold began, and when one last ran + # to completion. Only a *completed* hold updates last_probe_ts, so a + # session that unplugs mid-probe does not consume the month's slot. + probe_started_ts: float | None = None + last_probe_ts: float | None = None @dataclass @@ -148,8 +168,9 @@ class Action: value: float | None = None -def decide(thermal: dict, state: State, cfg: Config) -> tuple[Action, State, str]: - """Pure decision logic: what to do, the state to persist, and why.""" +def _decide_thermal(thermal: dict, state: State, cfg: Config) -> tuple[Action, State, str]: + """The derate-avoidance decision: what to do, the state to persist, and + why. Knows nothing about the calibration probe — see _apply_probe.""" session_state = thermal.get("state") forecast = thermal.get("forecast") or {} basis = forecast.get("basis") @@ -403,6 +424,109 @@ def decide(thermal: dict, state: State, cfg: Config) -> tuple[Action, State, str ) +def _apply_probe( + action: Action, new_state: State, reason: str, thermal: dict, prev: State, cfg: Config +) -> tuple[Action, State, str]: + """Overlay the calibration probe on the derate decision. + + A layer rather than a branch inside _decide_thermal, because that logic + is tuned by a string of live incidents and the probe has no business + reaching into it. The precedence is the only thing that matters here: + + - **Safety always wins.** A cap below the probe current is a real + thermal decision; it is applied, and the probe is abandoned rather + than held over a window whose current just moved. An abandoned probe + does not update last_probe_ts, so the next session retries it. + - **The probe outranks restoring.** Stepping back up toward full rate is + exactly what would ruin the measurement, so while a probe holds, this + returns "none" and the step-up never happens. + - **A probe only starts from above.** If the current is already at or + under the probe current there is nothing to hold it down to. + """ + if cfg.probe_amps <= 0: + return action, new_state, reason + session_state = thermal.get("state") + now_ts = thermal.get("ts") + current_a = thermal.get("current_a") + + # Not charging: no probe can be running, and any half-finished one is + # abandoned (its window never completed, so it taught nothing). + if session_state != "charging": + return action, replace(new_state, probe_started_ts=None), reason + + probing = prev.probe_started_ts is not None + + if action.kind == "cap" and action.value is not None and action.value < cfg.probe_amps: + if probing: + return ( + action, + replace(new_state, probe_started_ts=None), + f"{reason}; probe abandoned (thermal cap below the {cfg.probe_amps:g}A probe current)", + ) + return action, new_state, reason + + if probing: + if not isinstance(now_ts, (int, float)): + return Action("none"), new_state, "probe holding (no timestamp to age it against)" + held_min = (now_ts - prev.probe_started_ts) / 60.0 + if held_min >= cfg.probe_hold_min: + return ( + Action("restore", cfg.normal_amps), + replace( + new_state, + probe_started_ts=None, + last_probe_ts=now_ts, + capped=False, + cap_value=None, + clear_streak=0, + trip_streak=0, + ), + ( + f"probe complete: held {cfg.probe_amps:g}A for {held_min:.0f}min, " + f"restoring to {cfg.normal_amps:g}A" + ), + ) + # Hold. clear_streak is pinned at zero so the restore path cannot + # bank confirming polls while the probe runs and then step up the + # instant it ends. + return ( + Action("none"), + replace( + new_state, + probe_started_ts=prev.probe_started_ts, + capped=True, + cap_value=cfg.probe_amps, + clear_streak=0, + ), + f"probe holding {cfg.probe_amps:g}A ({held_min:.0f}/{cfg.probe_hold_min:g}min)", + ) + + # Start one? Only when due, only from a current above the probe value, + # and never on top of a thermal cap that is already lower — that session + # has bigger problems than calibration. + due = prev.last_probe_ts is None or ( + isinstance(now_ts, (int, float)) and (now_ts - prev.last_probe_ts) >= cfg.probe_interval_days * 86400.0 + ) + already_lower = new_state.capped and new_state.cap_value is not None and new_state.cap_value <= cfg.probe_amps + if due and not already_lower and isinstance(current_a, (int, float)) and current_a > cfg.probe_amps + 1.0: + return ( + Action("cap", cfg.probe_amps), + replace(new_state, probe_started_ts=now_ts, capped=True, cap_value=cfg.probe_amps), + ( + f"calibration probe due: capping to {cfg.probe_amps:g}A for {cfg.probe_hold_min:g}min " + "to measure an unregulated plateau" + ), + ) + return action, new_state, reason + + +def decide(thermal: dict, state: State, cfg: Config) -> tuple[Action, State, str]: + """What to do, the state to persist, and why: the derate decision with + the calibration probe layered over it.""" + action, new_state, reason = _decide_thermal(thermal, state, cfg) + return _apply_probe(action, new_state, reason, thermal, state, cfg) + + # --------------------------------------------------------------------------- # wallmonitor / tesla-ble access @@ -569,9 +693,40 @@ def main(argv: list[str] | None = None) -> int: default="/tmp/derate_amp_control.state.json", help="remembers cap state and debounce streaks between runs", ) + parser.add_argument( + "--probe-amps", + type=float, + default=Config.probe_amps, + help=( + "calibration probe: hold this charge current on a fixed cadence so the degradation watch " + "gets a plateau nothing trimmed (0 disables, the default). Pick a current low enough that " + "neither this daemon nor the vehicle wants to reduce it" + ), + ) + parser.add_argument( + "--probe-interval-days", + type=float, + default=Config.probe_interval_days, + help="days between calibration probes (default: %(default)s)", + ) + parser.add_argument( + "--probe-hold-min", + type=float, + default=Config.probe_hold_min, + help=( + "minutes to hold the probe current; needs to exceed ~3x the install's thermal time " + "constant for the handle to reach its plateau (default: %(default)s)" + ), + ) parser.add_argument("--dry-run", action="store_true", help="print the decision without changing the charger") args = parser.parse_args(argv) + if args.probe_amps and not (args.min_amps <= args.probe_amps <= args.normal_amps): + parser.error( + f"--probe-amps must sit between --min-amps ({args.min_amps:g}) and " + f"--normal-amps ({args.normal_amps:g}); got {args.probe_amps:g}" + ) + cfg = Config( normal_amps=args.normal_amps, lead_time_min=args.lead_time_min, @@ -584,6 +739,9 @@ def main(argv: list[str] | None = None) -> int: reattempt_window_min=args.reattempt_window_min, forecast_confidence_k=args.forecast_confidence_k, min_amps=args.min_amps, + probe_amps=args.probe_amps, + probe_interval_days=args.probe_interval_days, + probe_hold_min=args.probe_hold_min, ) state = load_state(args.state_file) diff --git a/deploy/install-derate-amp-control.sh b/deploy/install-derate-amp-control.sh index 682a46a..05c3c6c 100755 --- a/deploy/install-derate-amp-control.sh +++ b/deploy/install-derate-amp-control.sh @@ -7,6 +7,9 @@ # Usage (run with sudo, from anywhere): # sudo ./install-derate-amp-control.sh --tesla-ble http:// # sudo ./install-derate-amp-control.sh --tesla-ble http:// --dry-run +# # add a monthly calibration probe (a held, unregulated charge the +# # degradation watch can compare month to month): +# sudo ./install-derate-amp-control.sh --tesla-ble http:// --probe-amps 32 # sudo ./install-derate-amp-control.sh --uninstall # # The ESP32 host/IP lands only in the local systemd unit — never commit it. @@ -28,6 +31,9 @@ NORMAL_AMPS="" LEAD_TIME_MIN="" CONFIRM_TICKS="" MIN_CAP_DELTA_A="" +PROBE_AMPS="" +PROBE_INTERVAL_DAYS="" +PROBE_HOLD_MIN="" STATE_FILE="" INTERVAL="30" DRY_RUN="0" @@ -45,6 +51,9 @@ while [[ $# -gt 0 ]]; do --lead-time-min) LEAD_TIME_MIN="$2"; shift 2 ;; --confirm-ticks) CONFIRM_TICKS="$2"; shift 2 ;; --min-cap-delta-a) MIN_CAP_DELTA_A="$2"; shift 2 ;; + --probe-amps) PROBE_AMPS="$2"; shift 2 ;; + --probe-interval-days) PROBE_INTERVAL_DAYS="$2"; shift 2 ;; + --probe-hold-min) PROBE_HOLD_MIN="$2"; shift 2 ;; --state-file) STATE_FILE="$2"; shift 2 ;; --interval) INTERVAL="$2"; shift 2 ;; --dry-run) DRY_RUN="1"; shift ;; @@ -87,6 +96,9 @@ DAEMON_ARGS="--tesla-ble ${TESLA_BLE} --wallmonitor ${WALLMONITOR}" [[ -n "$LEAD_TIME_MIN" ]] && DAEMON_ARGS+=" --lead-time-min ${LEAD_TIME_MIN}" [[ -n "$CONFIRM_TICKS" ]] && DAEMON_ARGS+=" --confirm-ticks ${CONFIRM_TICKS}" [[ -n "$MIN_CAP_DELTA_A" ]] && DAEMON_ARGS+=" --min-cap-delta-a ${MIN_CAP_DELTA_A}" +[[ -n "$PROBE_AMPS" ]] && DAEMON_ARGS+=" --probe-amps ${PROBE_AMPS}" +[[ -n "$PROBE_INTERVAL_DAYS" ]] && DAEMON_ARGS+=" --probe-interval-days ${PROBE_INTERVAL_DAYS}" +[[ -n "$PROBE_HOLD_MIN" ]] && DAEMON_ARGS+=" --probe-hold-min ${PROBE_HOLD_MIN}" [[ -n "$STATE_FILE" ]] && DAEMON_ARGS+=" --state-file ${STATE_FILE}" [[ "$DRY_RUN" == "1" ]] && DAEMON_ARGS+=" --dry-run" [[ -n "$EXTRA_ARGS" ]] && DAEMON_ARGS+=" ${EXTRA_ARGS}" diff --git a/docs/amp-control.md b/docs/amp-control.md index 5aa01ea..e6b60ec 100644 --- a/docs/amp-control.md +++ b/docs/amp-control.md @@ -101,6 +101,56 @@ records `amp_adjust_failed`, so bridge flakiness shows up in the same place as the decisions it blocked. Recording is best-effort by design: the event log is observability, never control flow. +## The calibration probe + +The daemon exists to move charge current, which makes it the one component +that can *stop* moving it on purpose — and that turns out to be worth as much +as the capping. + +The [degradation watch](thermal-model.md#degradation-watch) can only compare +charges whose current held steady through the ramp; anything else fits a +plateau that a current change produced rather than the connector. On an +install where this daemon caps often, those steady windows are scarce and +land at whatever current the capping happened to stop at — which is a moving +target, and one correlated with the weather, since a hot garage triggers +capping sooner. Measured on one install: 285 caps in a month, 7 of 18 fitted +windows contaminated, and the surviving ones split across three different +currents. + +`--probe-amps` fixes that by manufacturing a clean window on a cadence: + +```bash +sudo ./deploy/install-derate-amp-control.sh --tesla-ble http:// --probe-amps 32 +``` + +Once every `--probe-interval-days` (default 30), the first charging session +to come along is held at `--probe-amps` for `--probe-hold-min` (default 40) +minutes, then released. Pick a current low enough that neither this daemon +nor the vehicle wants to reduce it — on a 48 A install where foldback starts +around 61 °C, 32 A plateaus near 53 °C with room to spare. Hold it for more +than ~3x the install's time constant (`model.tau_min` in `/api/thermal`) so +the handle actually reaches that plateau instead of being extrapolated to it. + +The result is a repeatable operating point: same current, unregulated, once a +month. Comparing those to each other is a degradation test with no +extrapolation across currents in it at all. They also break the collinearity +between charge current and the calendar, which is what otherwise stops the +watch's regression from telling a cap apart from a trend. + +Precedence is the whole contract, and it is deliberately simple: + +- **A thermal cap below the probe current always wins.** It is applied, and + the probe is abandoned — a window whose current just moved teaches nothing. +- **The probe outranks restoring.** Stepping back toward full rate is exactly + what would ruin the measurement, so no step-up happens while it holds. +- **An abandoned probe does not count.** Only a hold that ran its full length + updates the cadence, so a session that unplugs early — or one that needed a + real cap — simply retries next time. + +The probe is off unless `--probe-amps` is set. An install whose charges +already run unregulated at full rate does not need one; `regulated_n` on the +Alerts page says whether yours does. + **Checking whether it would actually help, before or after deploying it:** `contrib/backtest_derate_amp_control.py` replays `decide()` against real historical sessions read straight from `wallmonitor.db` (point it at a copy, diff --git a/docs/thermal-model.md b/docs/thermal-model.md index 97e8bc2..cc1b24d 100644 --- a/docs/thermal-model.md +++ b/docs/thermal-model.md @@ -60,19 +60,26 @@ current actually flows. ## Free-running windows, and the ones the charger wrote -The other way a plateau can be fictitious is that the charger chose it. A -Gen 3 defends its own thermal limit long before alert 40: as the handle -warms it trims charge current back, and it can trim ~10 % without ever -leaving the fitter's steady-current band. The ramp then flattens *because -the current fell*, and the exponential reads that flattening as the -plateau — a lower rise paired with a faster τ, clearing every other gate -with an excellent RMSE. - -That bias is not random. Foldback starts sooner in a hot garage, so the -under-read arrives and leaves with the weather; and a session that runs at -a low enough current never triggers it at all. Measured on one install: -windows whose current sagged fitted a median +33.3 °C rise where the same -charger's steady windows fitted +37.2 °C. +The other way a plateau can be fictitious is that something chose it. The +fitter's steady-current band is 10 % of the reference current, which is wide +enough to hide a substantial reduction: 48.6 A trimmed to 44.7 A never +leaves it. The ramp then flattens *because the current fell*, and the +exponential reads that flattening as the plateau — a lower rise paired with +a faster τ, clearing every other gate with an excellent RMSE. + +Who moved the current matters less than that it moved. On one install it was +mostly the monitor's own doing: the optional [amp +controller](amp-control.md) caps on this model's forecast, 285 times in a +month, and the vehicle tapers on its own besides. The charger's *internal* +foldback — the one alert 40 raises, counted by lifetime `thermal_foldbacks` +— had not fired once in the same period. So this is first of all a feedback +loop: the forecast caps the current, the cap contaminates the fit, the fit +feeds the forecast. + +The bias is not random either. The controller caps sooner in a hot garage, +so the under-read arrives and leaves with the weather. Measured on that +install: windows whose current sagged fitted a median +33.3 °C rise where +the same charger's steady windows fitted +37.2 °C. So every fit records `current_sag_a` — how far current fell from the head of the window to its tail, compared by quarter-medians — and @@ -207,15 +214,26 @@ row. More sessions either confirm it or dissolve it. `regulated_n` says how many sat out, and the Alerts page says so too, because a verdict resting on four fits should not look like one resting on twelve. -- **Only sessions near the install's recent operating current.** Cap the - vehicle at a new amperage and the watch follows, rather than judging - forever against a current the install no longer uses. -- **Pooled across a wider current band when the fits are clean.** +- **Only sessions near the install's usual charge current** — the median + across its whole comparable history, not its newest few fits. The + regression holds current, so a cap is something to adjust for rather than + chase, and a stable band cannot be inverted by the occasional + off-current charge (a monthly calibration probe is exactly such a charge: + at "newest three", two of them landing together made the *probe* current + typical and pooled the operating current out of its own comparison). +- **Pooled across a wide current band when the fits are clean.** Ambient-bracketed fits join from a wider band, and the regression's own current term then *adjusts* them: residual error in the I² normalization lands on that coefficient instead of masquerading as a trend. On the install above that coefficient read −0.99 °C per amp, which is the whole - of the phantom +7.2 °C. + of the phantom +7.2 °C. The band is wide enough on purpose to admit a + [calibration probe](amp-control.md#the-calibration-probe). +- **Never only stale sessions.** If none of the newest few free-running + charges make it into the comparison, the install has moved to a current + the band excludes and the watch reports nothing rather than a verdict + about a way it no longer charges — which also lets a stale alert clear. + As charges at the new current accumulate they become the median and the + band follows them. ### How sure it is @@ -264,14 +282,16 @@ where full-rate charging always ends in foldback, the only free-running windows are the low-current ones, and the watch is judging a handful of fits at a current the install rarely uses. -The cheapest fix is a **fixed-condition probe**: once a month, charge at a -current low enough to run unregulated end to end (well under whatever -first triggers foldback), for at least 3 τ. That yields a plateau nobody -imposed, at a repeatable operating point, and comparing those month over -month is a degradation test with no extrapolation in it at all. A single -such charge is worth more to this watch than several at full rate. - -Deliberately varying the current — 32 / 40 / 48 A inside one week, at -similar ambient — is worth doing once for a different reason: it breaks the -collinearity between current and the calendar, which is the one thing that -stops the regression from separating a cap from a trend. +The fix is a **fixed-condition probe**: once a month, charge at a current +low enough to run unregulated end to end (well under whatever first trims +it), for at least 3 τ. That yields a plateau nobody imposed, at a +repeatable operating point, and comparing those month over month is a +degradation test with no extrapolation in it at all. A single such charge +is worth more to this watch than several at full rate. It also breaks the +collinearity between charge current and the calendar, which is the one +thing that stops the regression from separating a cap from a trend. + +The [amp controller](amp-control.md#the-calibration-probe) automates it — +`--probe-amps 32` — because the component that moves charge current is the +one that can hold it still on purpose. Without the controller, setting the +vehicle's charge limit by hand once a month does the same job. diff --git a/tests/test_derate_amp_control.py b/tests/test_derate_amp_control.py index 54c7235..65e592a 100644 --- a/tests/test_derate_amp_control.py +++ b/tests/test_derate_amp_control.py @@ -471,3 +471,123 @@ def test_confidence_guard_falls_back_to_fit_rmse_without_se(): action, state, reason = dac.decide(legacy, state, cfg) assert action.kind == "cap" and action.value == 43.0 assert "fit rmse" in reason + + +# --- calibration probe ------------------------------------------------- +# The degradation watch can only compare windows the current held steady +# through, and on an install where this daemon caps often those are scarce +# and land wherever the capping happened to stop. The probe manufactures one +# on a cadence. Its whole contract is precedence: safety outranks it, it +# outranks restoring, and it never reports a window it did not actually hold. + + +def _probe_cfg(**kw): + kw.setdefault("probe_amps", 32.0) + kw.setdefault("probe_interval_days", 30.0) + kw.setdefault("probe_hold_min", 40.0) + return _cfg(**kw) + + +def test_probe_disabled_by_default_changes_nothing(): + # Zero probe_amps must leave the daemon byte-for-byte as it was for + # every install that never asked for this. + thermal = _thermal(will_trip=False, handle_c=50.0) + plain = dac.decide(thermal, dac.State(last_session_state="charging"), _cfg()) + off = dac.decide(thermal, dac.State(last_session_state="charging"), _probe_cfg(probe_amps=0.0)) + assert plain == off + + +def test_probe_starts_when_due_and_caps_to_the_probe_current(): + state = dac.State(last_session_state="charging") # never probed + action, new, reason = dac.decide(_thermal(will_trip=False), state, _probe_cfg()) + assert action.kind == "cap" and action.value == 32.0 + assert new.probe_started_ts == 1_000_000.0 and new.last_probe_ts is None + assert "probe due" in reason + + +def test_probe_not_due_until_the_interval_elapses(): + recent = dac.State(last_session_state="charging", last_probe_ts=1_000_000.0 - 10 * 86400) + action, new, _ = dac.decide(_thermal(will_trip=False), recent, _probe_cfg()) + assert action.kind == "none" and new.probe_started_ts is None + due = dac.State(last_session_state="charging", last_probe_ts=1_000_000.0 - 31 * 86400) + action, _, _ = dac.decide(_thermal(will_trip=False), due, _probe_cfg()) + assert action.kind == "cap" and action.value == 32.0 + + +def test_probe_holds_against_the_restore_path(): + # Stepping back toward full rate is exactly what would ruin the window, + # so a trajectory-clear signal that would normally step up must not. + held = dac.State( + last_session_state="charging", capped=True, cap_value=32.0, + probe_started_ts=1_000_000.0 - 10 * 60, clear_streak=9, + ) + action, new, reason = dac.decide( + _thermal(will_trip=False, handle_c=45.0, ts=1_000_000.0), held, _probe_cfg()) + assert action.kind == "none" and "probe holding" in reason + assert new.cap_value == 32.0 and new.probe_started_ts is not None + # and it cannot bank confirming polls to spend the moment it ends + assert new.clear_streak == 0 + + +def test_probe_completes_after_the_hold_and_restores(): + held = dac.State( + last_session_state="charging", capped=True, cap_value=32.0, + probe_started_ts=1_000_000.0 - 41 * 60, + ) + action, new, reason = dac.decide(_thermal(will_trip=False), held, _probe_cfg()) + assert action.kind == "restore" and action.value == 48.0 + assert "probe complete" in reason + assert new.probe_started_ts is None and new.last_probe_ts == 1_000_000.0 + assert not new.capped + + +def test_safety_cap_below_probe_current_wins_and_abandons_the_probe(): + # A real thermal decision outranks calibration, and a window whose + # current just moved teaches nothing — so it is abandoned, not banked. + held = dac.State( + last_session_state="charging", capped=True, cap_value=32.0, + probe_started_ts=1_000_000.0 - 5 * 60, trip_streak=2, + ) + action, new, reason = dac.decide( + _thermal(will_trip=True, mtt=5.0, suggested=28.0), held, _probe_cfg()) + assert action.kind == "cap" and action.value == 28.0 + assert new.probe_started_ts is None + assert "probe abandoned" in reason + # Abandoned, not completed: the next session must retry it. + assert new.last_probe_ts is None + + +def test_probe_abandoned_by_an_unplug_does_not_consume_the_slot(): + held = dac.State( + last_session_state="charging", capped=True, cap_value=32.0, + probe_started_ts=1_000_000.0 - 5 * 60, + ) + _, new, _ = dac.decide(_thermal(state="idle"), held, _probe_cfg()) + assert new.probe_started_ts is None and new.last_probe_ts is None + + +def test_probe_does_not_start_from_a_lower_thermal_cap(): + # A session already capped under the probe current has bigger problems + # than calibration; forcing it up to 32 A would be actively unsafe. + capped_low = dac.State( + last_session_state="charging", capped=True, cap_value=26.0, clear_streak=0) + action, new, _ = dac.decide( + _thermal(will_trip=False, handle_c=63.0), capped_low, _probe_cfg()) + assert action.kind != "cap" or (action.value or 0) >= 32.0 + assert new.probe_started_ts is None + + +def test_probe_state_survives_an_old_state_file(tmp_path): + # A daemon upgrade reads state written before the probe existed. It must + # default the new fields rather than crash — and a probe must then be + # due, not silently skipped by a missing last_probe_ts. + path = tmp_path / "state.json" + path.write_text('{"capped": true, "cap_value": 40.0, "last_session_state": "charging"}') + state = dac.load_state(str(path)) + assert state.capped and state.cap_value == 40.0 + assert state.probe_started_ts is None and state.last_probe_ts is None + action, _, reason = dac.decide(_thermal(will_trip=False), state, _probe_cfg()) + assert action.kind == "cap" and action.value == 32.0 and "probe due" in reason + # and the round-trip back to disk keeps the new fields + dac.save_state(str(path), dac.State(probe_started_ts=1.0, last_probe_ts=2.0)) + assert dac.load_state(str(path)).last_probe_ts == 2.0 diff --git a/tests/test_wallmonitor.py b/tests/test_wallmonitor.py index 948b82b..59d5512 100644 --- a/tests/test_wallmonitor.py +++ b/tests/test_wallmonitor.py @@ -803,10 +803,12 @@ async def test_thermal_fit_debiases_in_window_ambient_drift(db): async def test_thermal_drift_follows_current_change(db): # The user caps the vehicle at a new charge current (e.g. 48 A -> 40 A to - # stay under the derate on hot days). The old all-history median kept - # "typical" at 48 A forever: every new session was off-current, the drift - # verdict froze on stale data, and an active alert could never clear or - # re-confirm. Typical must follow the install's recent operating point. + # stay under the derate on hot days). The failure to avoid is a verdict + # that freezes on stale data — judging from the old 48 A fits while every + # new charge sits outside the band, so an active alert can never clear or + # re-confirm. The rule is that the comparison must contain the newest + # free-running charge or produce nothing at all; as charges at the new + # current accumulate, the band follows them. now = time.time() for i, rise in enumerate([36.0, 36.5, 35.8, 36.2, 36.1, 36.4]): _seed_thermal_session(db, now - (14 - i) * 7200, ambient_c=25.0, rise_ref_c=rise) @@ -814,8 +816,9 @@ async def test_thermal_drift_follows_current_change(db): _seed_thermal_session(db, now - (7 - i) * 7200, ambient_c=25.0, rise_ref_c=rise, amps=40.6) fits = thermal.fit_sessions(db, now) drift = thermal.detect_drift(fits) - # Three 40 A fits and no 40 A baseline yet: the honest "can't judge yet" - # (which clears a stale alert) rather than a verdict frozen at 48 A. + # Three 40 A fits, unbracketed so they cannot pool across currents, and + # no 40 A baseline yet: the honest "can't judge yet" (which clears a + # stale alert) rather than a verdict frozen on the older 48 A fits. assert drift is None # More 40 A history accumulates — the watch re-arms at the new current # and a genuine same-current increase is still flagged. @@ -1067,8 +1070,12 @@ async def test_thermal_drift_pools_bracketed_cross_current_fits(db): drift = thermal.detect_drift(fits) assert drift is not None, "bracketed 48 A baseline must keep judging 40 A charges" assert drift["drifting"] is False - assert abs(drift["typical_current_a"] - 40.6) < 0.1 - assert drift["cross_current_n"] == 6 # the 48 A baseline, pooled in + # "Typical" is the install's usual current across its history, not its + # newest few fits — the regression holds current, so a cap is adjusted + # for rather than chased, and a stable band cannot be inverted by an + # occasional off-current charge. + assert abs(drift["typical_current_a"] - 48.6) < 0.1 + assert drift["cross_current_n"] == 3 # the 40 A charges, pooled in # Un-bracketed off-current fits must still be excluded (the old rule). # (Seeded clear of session 9's cool-down tail so neither ambient read is # contaminated by interleaved samples.) @@ -1158,6 +1165,33 @@ async def test_thermal_drift_holds_charge_current(db): assert abs(mixed["delta_c"]) < thermal.DRIFT_WARN_C +async def test_thermal_drift_admits_calibration_probe_fits(db): + # A calibration probe deliberately charges well under the operating + # current so the handle reaches a plateau nothing trimmed. Those are the + # best fits an install can produce and the pooling band used to discard + # them silently — leaving the automation running and its data ignored. + now = time.time() + # Ten charges at the 48.6 A operating current with a 32 A probe dropped + # in at positions 3 and 8, the way a monthly cadence actually lands. + rises = [36.0, 36.5, 35.8, 36.2, 36.1, 36.4, 36.0, 36.3, 35.9, 36.2] + for i, rise in enumerate(rises): + amps = 32.0 if i in (3, 8) else 48.6 + _seed_thermal_session(db, now - (24 - 2 * i) * 7200, ambient_c=25.0, rise_ref_c=rise, + amps=amps, cooldown_s=900.0, ambient_end_c=25.0) + fits = thermal.fit_sessions(db, now) + probes = [fit for fit in fits if fit["current_a"] < 36.0] + assert len(probes) == 2 and all(fit["free_plateau"] for fit in probes) + drift = thermal.detect_drift(fits) + assert drift is not None + assert drift["n"] == len(fits), "probe fits must reach the regression, not be pooled out" + assert drift["off_current_n"] == 0 + # Interleaving them is also what keeps current separable from the + # calendar, which is the other thing a probe buys. + assert "current" in drift["covariates"] + assert not drift["collinear_with_time"] + assert drift["drifting"] is False + + async def test_thermal_drift_reports_ambient_confound(db): # The premise of rise_ref is that subtracting ambient leaves a number # about the connector. On an install where it does not — the fits still diff --git a/wallmonitor/thermal.py b/wallmonitor/thermal.py index 5ad0498..1746526 100644 --- a/wallmonitor/thermal.py +++ b/wallmonitor/thermal.py @@ -209,22 +209,29 @@ def ambient_from_idle_handle(handle_c: float, model: IdleOffset = BUILTIN_IDLE_O PREFIX_SPAN_MIN_S = 1800.0 # The steady-prefix band (10% of the reference current) is wide enough to -# hide the charger's own thermal regulation: a Gen 3 that trims 48.6 A to -# 44.7 A as the handle nears its limit never leaves the band, so the ramp -# keeps collecting samples whose flattening is *caused by the current -# dropping*. The exponential then reads that as the plateau — a lower rise -# paired with a faster tau, passing every gate with a fine RMSE. Measured on -# one install: fits whose current sagged read a median 33.3 C rise against -# 37.2 C for the same charger's steady windows, and because foldback starts -# sooner in a hot garage the bias tracked ambient, manufacturing a 7 C -# "drift" verdict out of the seasons. +# hide a substantial current reduction: 48.6 A trimmed to 44.7 A never +# leaves it, so the ramp keeps collecting samples whose flattening is +# *caused by the current dropping*. The exponential then reads that as the +# plateau — a lower rise paired with a faster tau, passing every gate with a +# fine RMSE. Measured on one install: windows whose current sagged read a +# median 33.3 C rise against 37.2 C for the same charger's steady windows. +# +# Who moves the current matters less than that it moved, and on that install +# it was mostly *us*: the optional amp controller (contrib/) caps on this +# model's own forecast, 285 times in a month, and the vehicle tapers on its +# own besides. The charger's internal foldback — the one alert 40 raises, +# counted by lifetime `thermal_foldbacks` — had not fired once in the same +# period. So this is first of all a feedback loop: forecast caps the +# current, the cap contaminates the fit, the fit feeds the forecast. Because +# the controller caps sooner in a hot garage, the contamination also tracks +# ambient, which is what turned it into a 7 C "drift" verdict. # # So each fit records whether its window was *free-running*: current held # flat end to end, making the fitted plateau the connector's own equilibrium -# rather than one the charger imposed. Only free-running fits are compared -# by the degradation watch. The other half of "the plateau was real" — -# whether the window ran long enough to observe it — is MIN_SPAN_TAU above, -# already enforced before any fit is emitted. +# rather than one something else imposed. Only free-running fits are +# compared by the degradation watch. The other half of "the plateau was +# real" — whether the window ran long enough to observe it — is MIN_SPAN_TAU +# above, already enforced before any fit is emitted. # # The threshold separates the two populations with room to spare: on that # install steady windows sagged <= 0.6% while regulated ones sagged >= 3.9%. @@ -800,7 +807,9 @@ def fit_history(db: Database, now: float, lookback_days: float = 120.0, # time the collinearity inflates the slope's standard error and the verdict # declines to confirm — which is the honest outcome, reached automatically. DRIFT_MIN_N = 6 -DRIFT_TYPICAL_N = 3 # newest fits defining the install's current operating point +# How many of the newest fits the comparison has to reach into to count as +# describing the install as it charges now, rather than as it used to. +DRIFT_RECENCY_N = 3 DRIFT_WARN_C = 2.5 # materiality floor, not the trigger — see detect_drift DRIFT_ALERT = "Handle heat rise increasing (check connector/wiring)" @@ -810,7 +819,16 @@ def fit_history(db: Database, now: float, lookback_days: float = 120.0, # The regression carries a current term of its own, so pooled fits are # adjusted rather than merely admitted — a residual error in the I^2 # normalization lands on that coefficient instead of on the time slope. -DRIFT_POOL_BAND_FRAC = 0.25 +# +# The band is wide because the reason it was narrow is gone. It guarded a +# median against a single off-current fit swinging it; there is no median +# any more, every fit that gets here already held its current steady, and +# the current term adjusts what is left. It has to be this wide to admit a +# *calibration probe* — a charge deliberately held well under the operating +# current so the handle reaches a plateau nothing trimmed (see +# contrib/derate_amp_control.py --probe-amps). Those are the most valuable +# fits an install can produce, and a narrow band silently discarded them. +DRIFT_POOL_BAND_FRAC = 0.45 # A covariate earns a column only when the history actually moved in it. # Regressing on a covariate that barely varies buys nothing and spends a @@ -977,7 +995,15 @@ def detect_drift(fits: list[dict], anchor_ts: float | None = None) -> dict | Non if len(usable) < DRIFT_MIN_N: return None usable.sort(key=lambda fit: fit["start_ts"]) - typical_a = median(fit["current_a"] for fit in usable[-DRIFT_TYPICAL_N:]) + # The install's usual charge current, over its whole comparable history + # rather than its newest few fits. The old rule took the newest three so + # the watch would *follow* a cap — necessary when a median split could + # only compare like with like, and actively harmful now: the regression + # holds current, so a cap is something to adjust for rather than chase, + # and an occasional off-current charge taking over "typical" inverts the + # admission band and pools the operating current out of its own + # comparison. A monthly calibration probe is exactly such a charge. + typical_a = median(fit["current_a"] for fit in usable) band = max(2.0, 0.1 * typical_a) pool_band = DRIFT_POOL_BAND_FRAC * typical_a comparable = [ @@ -991,6 +1017,17 @@ def detect_drift(fits: list[dict], anchor_ts: float | None = None) -> dict | Non ] if len(comparable) < DRIFT_MIN_N: return None + recent_ts = {fit["start_ts"] for fit in usable[-DRIFT_RECENCY_N:]} + if not any(fit["start_ts"] in recent_ts for fit in comparable): + # None of the newest free-running charges made it into the + # comparison: the install has moved to a current the band excludes, + # and every fit that did make it describes a way it no longer + # charges. Report nothing rather than a verdict about the past — that + # also lets a stale alert clear. As charges at the new current + # accumulate they become the median and the band follows them. + # Deliberately "any of the newest few", not "the newest": a single + # odd charge must not blank a watch that is otherwise current. + return None origin = comparable[0]["start_ts"] days = [(fit["start_ts"] - origin) / 86400.0 for fit in comparable]