healthz.sh

 1# forge-healthz: the forge's health for GET https://infra.rtw.run/healthz,
 2# polled by the monitor (~/projects/monitor). Runs every 5 minutes and writes
 3# the response for Caddy to serve as a static file:
 4#
 5#   /run/forge-healthz/ok.json        every check passes   -> 200
 6#   /run/forge-healthz/degraded.json  any check fails      -> 503
 7#
 8# At most one of them exists; with neither (just booted) Caddy answers 404.
 9# The body has the shape of sovrn's metrics-healthz:
10# {"status": "ok"|"degraded", "checked_at", "checks": {<name>: {"status":
11# "ok"|"fail", "detail", "error"}}}.
12#
13# Checks (thresholds from the Pushover-era monitor.sh):
14#   disk        root filesystem % used
15#   load        15-minute load average; the 24h maximum of the 1-minute
16#               samples is information only
17#   backup      age of the last successful backup (backup.sh stamps it)
18#   soft-serve  the unit is active
19
20OUT_DIR=/run/forge-healthz
21STATE_DIR=/var/lib/forge-healthz
22BACKUP_STAMP=/var/lib/forge-backup/last-success
23
24DISK_THRESHOLD=80              # % used on /
25LOAD_THRESHOLD=1.9             # 2 cores at ~95%
26BACKUP_MAX_AGE=$((26 * 3600))  # daily at 20:30 UTC, plus time to run
27
28# publish NAME BODY: make NAME.json the one Caddy serves.
29publish() {
30  local other=ok
31  [[ $1 == ok ]] && other=degraded
32  printf '%s\n' "$2" >"$OUT_DIR/.$1.json.tmp"
33  mv "$OUT_DIR/.$1.json.tmp" "$OUT_DIR/$1.json"
34  rm -f "$OUT_DIR/$other.json"
35}
36
37# If this script dies, the old answer must not stand: say so, as a 503.
38published=
39on_exit() {
40  [[ -n $published ]] && return
41  publish degraded "{\"status\":\"degraded\",\"checked_at\":\"$(date -u +%FT%TZ)\",\"checks\":{\"healthz\":{\"status\":\"fail\",\"error\":\"forge-healthz failed; see journalctl -u forge-healthz\"}}}"
42}
43trap on_exit EXIT
44
45now=$(date +%s)
46checks='{}'
47
48# check NAME ok|fail DETAIL
49check() {
50  checks=$(jq -c --arg name "$1" --arg status "$2" --arg detail "$3" \
51    '.[$name] = {status: $status, detail: $detail}
52     + (if $status == "ok" then {} else {error: $detail} end)' \
53    <<<"$checks")
54}
55
56# exceeds A B: true when A > B (decimals).
57exceeds() {
58  awk -v a="$1" -v b="$2" 'BEGIN { exit !(a > b) }'
59}
60
61# disk
62used=$(df --output=pcent / | tail -n 1 | tr -dc 0-9)
63detail="used=${used}% threshold=${DISK_THRESHOLD}%"
64if ((used > DISK_THRESHOLD)); then check disk fail "$detail"; else check disk ok "$detail"; fi
65
66# load: keep 24 hours of 1-minute samples, one per run.
67read -r load1 load5 load15 _ </proc/loadavg
68echo "$now $load1" >>"$STATE_DIR/load"
69awk -v cutoff=$((now - 86400)) '$1 >= cutoff' "$STATE_DIR/load" >"$STATE_DIR/load.tmp"
70mv "$STATE_DIR/load.tmp" "$STATE_DIR/load"
71max24h=$(awk 'NR == 1 || $2 > max { max = $2 } END { print max }' "$STATE_DIR/load")
72detail="load=${load1} ${load5} ${load15} max_24h=${max24h} threshold=${LOAD_THRESHOLD} (15m)"
73if exceeds "$load15" "$LOAD_THRESHOLD"; then check load fail "$detail"; else check load ok "$detail"; fi
74
75# backup
76if [[ -f $BACKUP_STAMP ]]; then
77  last=$(stat -c %Y "$BACKUP_STAMP")
78  age=$((now - last))
79  detail="last_success=$(date -u -d "@$last" +%FT%TZ) age=$((age / 3600))h max_age=$((BACKUP_MAX_AGE / 3600))h"
80  if ((age > BACKUP_MAX_AGE)); then check backup fail "$detail"; else check backup ok "$detail"; fi
81else
82  check backup fail "no successful backup recorded ($BACKUP_STAMP)"
83fi
84
85# soft-serve
86state=$(systemctl is-active soft-serve || true)
87if [[ $state == active ]]; then check soft-serve ok "$state"; else check soft-serve fail "$state"; fi
88
89status=$(jq -r 'if all(.[]; .status == "ok") then "ok" else "degraded" end' <<<"$checks")
90body=$(jq -n --arg status "$status" --arg at "$(date -u -d "@$now" +%FT%TZ)" --argjson checks "$checks" \
91  '{status: $status, checked_at: $at, checks: $checks}')
92publish "$status" "$body"
93published=1
94echo "$status: $(jq -c . <<<"$checks")"