From 9b4fc25c25d63a560a4c2edaf11d93c35c68848c Mon Sep 17 00:00:00 2001 From: Dhruv Khara <66153057+BeLazy167@users.noreply.github.com> Date: Thu, 3 Sep 2026 19:53:05 -0500 Subject: [PATCH] fix: add /readyz Fly check so a dead DB fails health /healthz returns a constant 200 and never touches the store. It was the only Fly check, so argus-db filling its volume and crash-looping for three days still reported the app healthy while every data request failed. Add a second check on /readyz, which pings the store pool. Longer grace period covers pool open lagging the listener on boot. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01GJehDM4aXz1AkJrTaX2mSL --- backend/fly.toml | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/backend/fly.toml b/backend/fly.toml index c83ca8cd..15ff0877 100644 --- a/backend/fly.toml +++ b/backend/fly.toml @@ -46,6 +46,18 @@ swap_size_mb = 512 size = 'shared-cpu-1x' memory = '1024mb' +# Two checks, two questions. `health` asks "is the process alive?" and hits +# /healthz, which returns a constant 200 and touches nothing. `ready` asks "can +# this machine actually serve?" and hits /readyz, which pings the store pool. +# +# Only /healthz was wired up until 2026-09-03. That day argus-db filled its +# volume and Postgres crash-looped for three days; /healthz kept returning +# {"status":"ok"} the whole time, so Fly reported the app healthy while every +# data request failed. A dead database must turn a check red. +# +# `ready` carries the longer grace_period because a booting machine opens the +# pool after the listener, and the longer interval because each probe costs a +# round trip to Postgres. [checks] [checks.health] port = 8080 @@ -53,3 +65,11 @@ swap_size_mb = 512 interval = '15s' timeout = '2s' path = '/healthz' + + [checks.ready] + port = 8080 + type = 'http' + interval = '30s' + timeout = '5s' + grace_period = '20s' + path = '/readyz'