Skip to content
Closed
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 20 additions & 0 deletions backend/fly.toml
Original file line number Diff line number Diff line change
Expand Up @@ -46,10 +46,30 @@ swap_size_mb = 512
size = 'shared-cpu-1x'
memory = '1024mb'

# Two checks, two questions. `health` asks "is the process alive?" and hits
# /healthz, which returns a constant 200 and touches nothing. `ready` asks "can
# this machine actually serve?" and hits /readyz, which pings the store pool.
#
# Only /healthz was wired up until 2026-09-03. That day argus-db filled its
# volume and Postgres crash-looped for three days; /healthz kept returning
# {"status":"ok"} the whole time, so Fly reported the app healthy while every
# data request failed. A dead database must turn a check red.
#
# `ready` carries the longer grace_period because a booting machine opens the
# pool after the listener, and the longer interval because each probe costs a
# round trip to Postgres.
[checks]
[checks.health]
port = 8080
type = 'http'
interval = '15s'
timeout = '2s'
path = '/healthz'

[checks.ready]
port = 8080
type = 'http'
interval = '30s'
timeout = '5s'
grace_period = '20s'
path = '/readyz'
Loading