healthz.sh
1# forge-healthz: the forge's health for GET https://infra.rtw.run/healthz,
2# polled by the monitor (~/projects/monitor). Runs every 5 minutes and writes
3# the response for Caddy to serve as a static file:
4#
5# /run/forge-healthz/ok.json every check passes -> 200
6# /run/forge-healthz/degraded.json any check fails -> 503
7#
8# At most one of them exists; with neither (just booted) Caddy answers 404.
9# The body has the shape of sovrn's metrics-healthz:
10# {"status": "ok"|"degraded", "checked_at", "checks": {<name>: {"status":
11# "ok"|"fail", "detail", "error"}}}.
12#
13# Checks (thresholds from the Pushover-era monitor.sh):
14# disk root filesystem % used
15# load 15-minute load average; the 24h maximum of the 1-minute
16# samples is information only
17# backup age of the last successful backup (backup.sh stamps it)
18# soft-serve the unit is active
19
20OUT_DIR=/run/forge-healthz
21STATE_DIR=/var/lib/forge-healthz
22BACKUP_STAMP=/var/lib/forge-backup/last-success
23
24DISK_THRESHOLD=80 # % used on /
25LOAD_THRESHOLD=1.9 # 2 cores at ~95%
26BACKUP_MAX_AGE=$((26 * 3600)) # daily at 20:30 UTC, plus time to run
27
28# publish NAME BODY: make NAME.json the one Caddy serves.
29publish() {
30 local other=ok
31 [[ $1 == ok ]] && other=degraded
32 printf '%s\n' "$2" >"$OUT_DIR/.$1.json.tmp"
33 mv "$OUT_DIR/.$1.json.tmp" "$OUT_DIR/$1.json"
34 rm -f "$OUT_DIR/$other.json"
35}
36
37# If this script dies, the old answer must not stand: say so, as a 503.
38published=
39on_exit() {
40 [[ -n $published ]] && return
41 publish degraded "{\"status\":\"degraded\",\"checked_at\":\"$(date -u +%FT%TZ)\",\"checks\":{\"healthz\":{\"status\":\"fail\",\"error\":\"forge-healthz failed; see journalctl -u forge-healthz\"}}}"
42}
43trap on_exit EXIT
44
45now=$(date +%s)
46checks='{}'
47
48# check NAME ok|fail DETAIL
49check() {
50 checks=$(jq -c --arg name "$1" --arg status "$2" --arg detail "$3" \
51 '.[$name] = {status: $status, detail: $detail}
52 + (if $status == "ok" then {} else {error: $detail} end)' \
53 <<<"$checks")
54}
55
56# exceeds A B: true when A > B (decimals).
57exceeds() {
58 awk -v a="$1" -v b="$2" 'BEGIN { exit !(a > b) }'
59}
60
61# disk
62used=$(df --output=pcent / | tail -n 1 | tr -dc 0-9)
63detail="used=${used}% threshold=${DISK_THRESHOLD}%"
64if ((used > DISK_THRESHOLD)); then check disk fail "$detail"; else check disk ok "$detail"; fi
65
66# load: keep 24 hours of 1-minute samples, one per run.
67read -r load1 load5 load15 _ </proc/loadavg
68echo "$now $load1" >>"$STATE_DIR/load"
69awk -v cutoff=$((now - 86400)) '$1 >= cutoff' "$STATE_DIR/load" >"$STATE_DIR/load.tmp"
70mv "$STATE_DIR/load.tmp" "$STATE_DIR/load"
71max24h=$(awk 'NR == 1 || $2 > max { max = $2 } END { print max }' "$STATE_DIR/load")
72detail="load=${load1} ${load5} ${load15} max_24h=${max24h} threshold=${LOAD_THRESHOLD} (15m)"
73if exceeds "$load15" "$LOAD_THRESHOLD"; then check load fail "$detail"; else check load ok "$detail"; fi
74
75# backup
76if [[ -f $BACKUP_STAMP ]]; then
77 last=$(stat -c %Y "$BACKUP_STAMP")
78 age=$((now - last))
79 detail="last_success=$(date -u -d "@$last" +%FT%TZ) age=$((age / 3600))h max_age=$((BACKUP_MAX_AGE / 3600))h"
80 if ((age > BACKUP_MAX_AGE)); then check backup fail "$detail"; else check backup ok "$detail"; fi
81else
82 check backup fail "no successful backup recorded ($BACKUP_STAMP)"
83fi
84
85# soft-serve
86state=$(systemctl is-active soft-serve || true)
87if [[ $state == active ]]; then check soft-serve ok "$state"; else check soft-serve fail "$state"; fi
88
89status=$(jq -r 'if all(.[]; .status == "ok") then "ok" else "degraded" end' <<<"$checks")
90body=$(jq -n --arg status "$status" --arg at "$(date -u -d "@$now" +%FT%TZ)" --argjson checks "$checks" \
91 '{status: $status, checked_at: $at, checks: $checks}')
92publish "$status" "$body"
93published=1
94echo "$status: $(jq -c . <<<"$checks")"