legacy, forge: drop Pushover; the monitor alerts on /healthz (b660c34)
The monitor (~/projects/monitor) now sends every alert by email and Pushover, so the host's own Pushover pieces go: system-monitor and hosts/common/scripts (monitor.sh, exec-prestart.sh), soft-serve's start notification, backup.sh's notification and the pushover.env key. forge-healthz's error field is the detail alone (the monitor prefixes the check name).
12 files changed,  +20, -441
M README.md
+1, -1
 1@@ -52,8 +52,8 @@ Rollback: the previous generation in GRUB, `nixos-rebuild switch --rollback` on
 2 Secrets live in the `secrets` store, one age file per value, and reach hosts as Colmena keys: Colmena runs `secrets decrypt <path>` on the deploying machine and uploads the result, so nothing secret enters the Nix store or this repo.
 3 
 4 ```nix
 5-servers.secrets."pushover.env".scope = "shared";   # servers/shared/pushover.env
 6 servers.secrets."rclone.conf".scope = "host";      # servers/hosts/<fqdn>/rclone.conf
 7+servers.secrets."<name>".scope = "shared";         # servers/shared/<name>
 8 ```
 9 
10 A service reads `config.servers.secrets.<name>.path` (under `/var/lib/servers-keys`, root-owned 0400 by default) and orders itself after `config.servers.secrets.<name>.unit`. To add one: `printf '%s' "$VALUE" | secrets encrypt servers/shared/<name>`, declare it, deploy. sovrn and moods have the same scheme under their own prefixes and key directories.
M hosts/common/README.md
+3, -10
 1@@ -5,15 +5,8 @@ Configuration shared by legacy-layout hosts (`hosts.json` `"layout": "legacy"`;
 2 | File | What |
 3 | --- | --- |
 4 | `default.nix` | Imports the rest. |
 5-| `base.nix` | Locale, packages (git, htop, jq, lazyjournal, neovim), `system-monitor` (every 5 minutes), weekly Nix GC (generations older than 14 days). |
 6-| `secrets.nix` | Brings in `modules/secrets.nix` and declares the shared `pushover.env`. |
 7+| `base.nix` | Locale, packages (git, htop, jq, lazyjournal, neovim), weekly Nix GC (generations older than 14 days). |
 8+| `secrets.nix` | Brings in `modules/secrets.nix`. |
 9 | `users.nix` | Root keys (`keys/admins.pub` plus the scooter key) and the `btburke` user (wheel, password hash from the secrets store). |
10 
11-## Scripts
12-
13-Embedded into services with `pkgs.writeShellScriptBin`. Both read `PUSHOVER_TOKEN` and `PUSHOVER_USER` from `pushover.env` (the service's `EnvironmentFile`) and call curl through `$CURL`, since services have a minimal `PATH`.
14-
15-| Script | What |
16-| --- | --- |
17-| `scripts/monitor.sh` | Load (threshold 1.9, sized for 2 cores) and root disk (80%) checks; Pushover alerts on state changes only, plus a daily summary. State in `/var/lib/system-monitor`. |
18-| `scripts/exec-prestart.sh` | `ExecStartPre` helper: a low-priority Pushover note when a service (re)starts. Never fails the service. |
19+Monitoring is the forge's `/healthz` ([roles/forge](../../roles/forge/README.md#healthz)), watched by the monitor (~/projects/monitor), which alerts by email and Pushover. The Pushover scripts that were here (`system-monitor`, the service-start notes) are gone (b660c34).
M hosts/common/base.nix
+2, -34
 1@@ -2,9 +2,6 @@
 2 # Base system configuration - boot, filesystems, networking
 3 { config, lib, pkgs, ... }:
 4 
 5-let
 6-  monitorScript = pkgs.writeShellScriptBin "system-monitor" (lib.fileContents ./scripts/monitor.sh);
 7-in
 8 {
 9   # Boot and filesystems are defined in hardware-configuration.nix
10   # which is imported per-host
11@@ -34,12 +31,12 @@ in
12   i18n.defaultLocale = lib.mkDefault "en_US.UTF-8";
13 
14   # Essential packages
15-  environment.systemPackages = (with pkgs; [
16+  environment.systemPackages = with pkgs; [
17     git
18     htop
19     jq
20     lazyjournal
21-  ]) ++ [ monitorScript ];
22+  ];
23 
24   # Enable neovim (system-wide) - plain, no plugins
25   programs.neovim = {
26@@ -55,35 +52,6 @@ in
27   # Basic system state version
28   system.stateVersion = lib.mkDefault "24.11";
29 
30-  # Systemd service for monitoring
31-  systemd.services.system-monitor = {
32-    description = "System monitor - check load and disk space";
33-    wants = [ config.servers.secrets."pushover.env".unit ];
34-    after = [ config.servers.secrets."pushover.env".unit ];
35-    path = with pkgs; [ gawk ];  # Required for awk command in monitor.sh
36-    serviceConfig = {
37-      Type = "oneshot";
38-      ExecStart = "${monitorScript}/bin/system-monitor";
39-      User = "root";
40-      EnvironmentFile = config.servers.secrets."pushover.env".path;
41-      StateDirectory = "system-monitor";
42-    };
43-    environment = {
44-      CURL = "${pkgs.curl}/bin/curl";
45-      PUSHOVER_DEVICE = config.networking.hostName;
46-    };
47-  };
48-
49-  # Systemd timer - run every 5 minutes
50-  systemd.timers.system-monitor = {
51-    description = "System monitor timer (every 5 minutes)";
52-    wantedBy = [ "timers.target" ];
53-    timerConfig = {
54-      OnBootSec = "5min";
55-      OnUnitActiveSec = "5min";
56-    };
57-  };
58-
59   # Systemd service for Nix garbage collection
60   systemd.services.custom-nix-gc = {
61     description = "Nix garbage collection - remove generations older than 14 days";
D hosts/common/scripts/exec-prestart.sh
+0, -82
 1@@ -1,82 +0,0 @@
 2-#!/usr/bin/env bash
 3-set -euo pipefail
 4-
 5-# ExecStartPre notification script for systemd services
 6-# Sends a low-priority (-1) notification when a service starts/restarts
 7-# Use in systemd service files: ExecStartPre=/path/to/exec-prestart.sh <service-name>
 8-#
 9-# Required argument: service name
10-# Environment variables from EnvironmentFile (e.g., /etc/pushover.env):
11-#   PUSHOVER_TOKEN, PUSHOVER_USER
12-#   CURL (full path to curl binary)
13-
14-# Pushover configuration
15-PUSHOVER_DEVICE="${PUSHOVER_DEVICE:-$(hostname)}"
16-PUSHOVER_TITLE="${PUSHOVER_TITLE:-service+event}"
17-
18-# Get service name from first argument (required)
19-get_service_name() {
20-    if [[ $# -lt 1 ]]; then
21-        echo "Error: Service name required as first argument" >&2
22-        echo "Usage: $0 <service-name>" >&2
23-        exit 1
24-    fi
25-    echo "$1"
26-}
27-
28-# Send Pushover notification
29-send_notification() {
30-    local message="$1"
31-    local priority="${2:--1}"
32-
33-    # Only proceed if required variables are set
34-    if [[ -z "${CURL:-}" ]]; then
35-        echo "Warning: CURL environment variable not set" >&2
36-        return 0  # Don't fail the service startup
37-    fi
38-
39-    if [[ -z "${PUSHOVER_TOKEN:-}" ]] || [[ -z "${PUSHOVER_USER:-}" ]]; then
40-        echo "Warning: PUSHOVER_TOKEN or PUSHOVER_USER not set" >&2
41-        return 0  # Don't fail the service startup
42-    fi
43-
44-    local response
45-    response=$($CURL -s -X POST \
46-        "https://api.pushover.net/1/messages.json" \
47-        -d "token=${PUSHOVER_TOKEN}" \
48-        -d "user=${PUSHOVER_USER}" \
49-        -d "device=${PUSHOVER_DEVICE}" \
50-        -d "title=${PUSHOVER_TITLE}" \
51-        -d "message=${message}" \
52-        -d "priority=${priority}" 2>&1) || {
53-        echo "Warning: Pushover notification failed" >&2
54-        return 0  # Don't fail the service startup
55-    }
56-
57-    # Check if pushover API returned success
58-    if ! echo "$response" | grep -q '"status":1'; then
59-        echo "Warning: Pushover API error: $response" >&2
60-    fi
61-
62-    return 0  # Never fail the service startup
63-}
64-
65-main() {
66-    local service_name
67-    service_name=$(get_service_name "$@")
68-
69-    # Get current timestamp
70-    local timestamp
71-    timestamp=$(date -u +'%Y-%m-%dT%H:%M:%SZ')
72-
73-    # Build notification message (spaces encoded as + for URL encoding)
74-    local message="Service+starting:+${service_name}+at+${timestamp}"
75-
76-    # Send low-priority notification (-1 = no alert on phone, just in app history)
77-    send_notification "$message" -1
78-
79-    # Log to stdout (will appear in systemd journal)
80-    echo "${timestamp} Pre-start notification sent for ${service_name}"
81-}
82-
83-main "$@"
D hosts/common/scripts/monitor.sh
+0, -262
  1@@ -1,262 +0,0 @@
  2-#!/usr/bin/env bash
  3-set -euo pipefail
  4-
  5-# System Monitor - Alert on high load or low disk space
  6-# Designed for 2-core systems with 5-minute check intervals
  7-
  8-# Thresholds
  9-LOAD_THRESHOLD=1.9        # Load avg for 2-core @ ~95% utilization
 10-DISK_THRESHOLD=80           # Disk usage percentage for root filesystem
 11-
 12-# Statistics - 5 minute intervals = 12 per hour, 24 hours = 288 runs
 13-RUNS_PER_DAY=288
 14-
 15-# State file for tracking alert status
 16-STATE_DIR="/var/lib/system-monitor"
 17-STATE_FILE="${STATE_DIR}/state"
 18-
 19-# Pushover configuration - uses hostname as device name
 20-PUSHOVER_DEVICE="${PUSHOVER_DEVICE:-$(hostname)}"
 21-PUSHOVER_TITLE="${PUSHOVER_TITLE:-server+alert}"
 22-
 23-# Ensure state directory exists
 24-mkdir -p "${STATE_DIR}"
 25-
 26-# Read current state (0 = ok, 1 = alerting)
 27-read_state() {
 28-    local key="$1"
 29-    if [[ -f "${STATE_FILE}" ]]; then
 30-        grep "^${key}=" "${STATE_FILE}" 2>/dev/null | cut -d'=' -f2 || echo "0"
 31-    else
 32-        echo "0"
 33-    fi
 34-}
 35-
 36-# Write state only if curl succeeded
 37-write_state() {
 38-    local key="$1"
 39-    local value="$2"
 40-    local temp_file="${STATE_FILE}.tmp"
 41-
 42-    # Read existing state, update key, write back
 43-    if [[ -f "${STATE_FILE}" ]]; then
 44-        grep -v "^${key}=" "${STATE_FILE}" > "${temp_file}" 2>/dev/null || true
 45-    fi
 46-    echo "${key}=${value}" >> "${temp_file}"
 47-    mv "${temp_file}" "${STATE_FILE}"
 48-}
 49-
 50-# Compare two floating point numbers, return 1 if first > second
 51-# Uses integer comparison by removing decimal points
 52-greater_than() {
 53-    local a="$1"
 54-    local b="$2"
 55-
 56-    # Remove decimal points and pad to equal length
 57-    local int_a int_b
 58-    int_a=$(echo "$a" | tr -d '.')
 59-    int_b=$(echo "$b" | tr -d '.')
 60-
 61-    while (( ${#int_a} < ${#int_b} )); do
 62-        int_a="${int_a}0"
 63-    done
 64-    while (( ${#int_b} < ${#int_a} )); do
 65-        int_b="${int_b}0"
 66-    done
 67-
 68-    if (( int_a > int_b )); then
 69-        echo "1"
 70-    else
 71-        echo "0"
 72-    fi
 73-}
 74-
 75-# Send Pushover notification
 76-send_notification() {
 77-    local message="$1"
 78-    local priority="${2:-0}"
 79-
 80-    # Only proceed if curl variable is set
 81-    if [[ -z "${CURL:-}" ]]; then
 82-        echo "Error: CURL environment variable not set"
 83-        return 1
 84-    fi
 85-
 86-    # Validate required environment variables
 87-    if [[ -z "${PUSHOVER_TOKEN:-}" ]]; then
 88-        echo "Error: PUSHOVER_TOKEN environment variable not set"
 89-        return 1
 90-    fi
 91-    if [[ -z "${PUSHOVER_USER:-}" ]]; then
 92-        echo "Error: PUSHOVER_USER environment variable not set"
 93-        return 1
 94-    fi
 95-
 96-    local response
 97-    # Use POST with form data (required by Pushover API)
 98-    response=$($CURL -s -X POST \
 99-        --data-urlencode "token=${PUSHOVER_TOKEN}" \
100-        --data-urlencode "user=${PUSHOVER_USER}" \
101-        --data-urlencode "device=${PUSHOVER_DEVICE}" \
102-        --data-urlencode "title=${PUSHOVER_TITLE}" \
103-        --data-urlencode "message=${message}" \
104-        --data-urlencode "priority=${priority}" \
105-        "https://api.pushover.net/1/messages.json" 2>&1)
106-    local curl_exit=$?
107-
108-    # Check if curl command itself failed
109-    if [[ $curl_exit -ne 0 ]]; then
110-        echo "Pushover curl failed (exit $curl_exit): $response"
111-        return 1
112-    fi
113-
114-    # Check if pushover API returned success
115-    if echo "$response" | grep -q '"status":1'; then
116-        return 0
117-    else
118-        echo "Pushover API error: $response"
119-        return 1
120-    fi
121-}
122-
123-# Get 1-minute load average
124-get_load() {
125-    cut -d' ' -f1 /proc/loadavg
126-}
127-
128-# Get disk usage percentage for root filesystem
129-get_disk_usage() {
130-    df / | awk 'NR==2 {print $5}' | tr -d '%'
131-}
132-
133-# Check if load is above threshold (avoid bc by removing decimal point)
134-check_load() {
135-    local current_load
136-    current_load=$(get_load)
137-
138-    # Remove decimal point to compare as integers (e.g., 1.92 -> 192)
139-    local scaled_load scaled_threshold
140-    scaled_load=$(echo "$current_load" | tr -d '.')
141-    scaled_threshold=$(echo "$LOAD_THRESHOLD" | tr -d '.')
142-
143-    # Pad shorter number with zeros to ensure equal length
144-    while (( ${#scaled_load} < ${#scaled_threshold} )); do
145-        scaled_load="${scaled_load}0"
146-    done
147-    while (( ${#scaled_threshold} < ${#scaled_load} )); do
148-        scaled_threshold="${scaled_threshold}0"
149-    done
150-
151-    if (( scaled_load > scaled_threshold )); then
152-        echo "1"
153-    else
154-        echo "0"
155-    fi
156-}
157-
158-# Check if disk is above threshold
159-check_disk() {
160-    local current_usage
161-    current_usage=$(get_disk_usage)
162-
163-    if (( current_usage > DISK_THRESHOLD )); then
164-        echo "1"
165-    else
166-        echo "0"
167-    fi
168-}
169-
170-# Main monitoring logic
171-main() {
172-    local load_alert disk_alert current_load current_usage
173-    local prev_load_alert prev_disk_alert
174-    local highest_load highest_disk run_count
175-
176-    # Get current status
177-    load_alert=$(check_load)
178-    disk_alert=$(check_disk)
179-    current_load=$(get_load)
180-    current_usage=$(get_disk_usage)
181-
182-    # Get previous state
183-    prev_load_alert=$(read_state "load_alert")
184-    prev_disk_alert=$(read_state "disk_alert")
185-
186-    # Get 24-hour statistics
187-    highest_load=$(read_state "highest_load")
188-    highest_disk=$(read_state "highest_disk")
189-    run_count=$(read_state "run_count")
190-
191-    # Initialize stats if empty
192-    if [[ "$highest_load" == "0" ]]; then
193-        highest_load="0.0"
194-    fi
195-    if [[ "$highest_disk" == "0" ]]; then
196-        highest_disk="0"
197-    fi
198-
199-    # Update highest values if current values are higher
200-    if [[ $(greater_than "$current_load" "$highest_load") == "1" ]]; then
201-        highest_load="$current_load"
202-    fi
203-    if (( current_usage > highest_disk )); then
204-        highest_disk="$current_usage"
205-    fi
206-
207-    # Increment run counter
208-    ((run_count++)) || true
209-
210-    # Check if it's time to send 24-hour summary
211-    if (( run_count >= RUNS_PER_DAY )); then
212-        local summary_msg="24hr summary: max load=${highest_load}, disk=${highest_disk}%"
213-        if send_notification "$summary_msg" -1; then
214-            echo "$(date -u +'%Y-%m-%dT%H:%M:%SZ') 24-hour summary sent: load=${highest_load}, disk=${highest_disk}%"
215-            # Reset statistics
216-            highest_load="0.0"
217-            highest_disk="0"
218-            run_count=0
219-        else
220-            echo "$(date -u +'%Y-%m-%dT%H:%M:%SZ') Failed to send 24-hour summary"
221-        fi
222-    fi
223-
224-    # Save statistics
225-    write_state "highest_load" "$highest_load"
226-    write_state "highest_disk" "$highest_disk"
227-    write_state "run_count" "$run_count"
228-
229-    # Check load - only alert on transition from ok (0) to alerting (1)
230-    if [[ "$load_alert" == "1" && "$prev_load_alert" == "0" ]]; then
231-        if send_notification "High+load+alert:+${current_load}+(threshold:+${LOAD_THRESHOLD})" 0; then
232-            write_state "load_alert" "1"
233-            echo "$(date -u +'%Y-%m-%dT%H:%M:%SZ') Load alert sent: $current_load"
234-        else
235-            echo "$(date -u +'%Y-%m-%dT%H:%M:%SZ') Failed to send load alert"
236-        fi
237-    elif [[ "$load_alert" == "0" && "$prev_load_alert" == "1" ]]; then
238-        # Load recovered - update state (no notification on recovery to avoid spam)
239-        write_state "load_alert" "0"
240-        echo "$(date -u +'%Y-%m-%dT%H:%M:%SZ') Load recovered: $current_load"
241-    fi
242-
243-    # Check disk - only alert on transition from ok (0) to alerting (1)
244-    if [[ "$disk_alert" == "1" && "$prev_disk_alert" == "0" ]]; then
245-        if send_notification "Low+disk+space:+${current_usage}%+used+(threshold:+${DISK_THRESHOLD}%)" 0; then
246-            write_state "disk_alert" "1"
247-            echo "$(date -u +'%Y-%m-%dT%H:%M:%SZ') Disk alert sent: ${current_usage}%"
248-        else
249-            echo "$(date -u +'%Y-%m-%dT%H:%M:%SZ') Failed to send disk alert"
250-        fi
251-    elif [[ "$disk_alert" == "0" && "$prev_disk_alert" == "1" ]]; then
252-        # Disk recovered - update state
253-        write_state "disk_alert" "0"
254-        echo "$(date -u +'%Y-%m-%dT%H:%M:%SZ') Disk recovered: ${current_usage}%"
255-    fi
256-
257-    # Always log current status (useful for debugging)
258-    if [[ "$load_alert" == "0" && "$disk_alert" == "0" ]]; then
259-        echo "$(date -u +'%Y-%m-%dT%H:%M:%SZ') All OK - Load: $current_load, Disk: ${current_usage}% (24h highs: load=${highest_load}, disk=${highest_disk}%)"
260-    fi
261-}
262-
263-main "$@"
M hosts/common/secrets.nix
+0, -4
1@@ -4,8 +4,4 @@
2 { ... }:
3 {
4   imports = [ ../../modules/secrets.nix ];
5-
6-  # Pushover credentials (PUSHOVER_TOKEN, PUSHOVER_USER) for the monitor,
7-  # soft-serve's start notification and the backup job.
8-  servers.secrets."pushover.env".scope = "shared";
9 }
M hosts/infra-rtw-run/README.md
+1, -1
1@@ -10,7 +10,7 @@ Host-specific configuration for `infra.rtw.run` (Hetzner Cloud, 2 vCPU, x86_64),
2 | `secrets.nix` | btburke's password hash. |
3 | `zone.db` | An exported BIND zone for rtw.run, for reference only. |
4 
5-Shared with other legacy hosts: `hosts/common/` (monitor, Nix GC, users, secrets wiring).
6+Shared with other legacy hosts: `hosts/common/` (Nix GC, users, secrets wiring). Monitoring is the forge's `/healthz`, watched by the monitor (~/projects/monitor), which alerts by email and Pushover.
7 
8 ## Networking
9 
M modules/secrets.nix
+1, -1
1@@ -2,7 +2,7 @@
2 # uploaded by Colmena. The same scheme as sovrn's and moods' modules, under
3 # the `servers/` prefix. A role or host declares what it needs:
4 #
5-#   servers.secrets."pushover.env" = { scope = "shared"; };
6+#   servers.secrets."rclone.conf" = { scope = "host"; };
7 #
8 # and a service reads config.servers.secrets.<name>.path, ordering itself
9 # after config.servers.secrets.<name>.unit. Store naming:
M roles/forge/README.md
+3, -3
 1@@ -24,7 +24,7 @@ Personal git hosting on infra.rtw.run: [soft-serve](https://github.com/charmbrac
 2 | `soft-serve/config.yaml` | `/etc/soft-serve/config.yaml`. soft-serve restarts when it changes. |
 3 | `soft-serve/hooks/post-receive` | Global hook, linked from `/var/soft/data/hooks`. Backgrounds `forge-rebuild-site` and returns at once. |
 4 | `rebuild-site.sh` | `forge-rebuild-site`: rebuilds every public, non-hidden repo's pages and the index. |
 5-| `backup.sh` | Daily (20:30 UTC) `rclone sync` of `/var/soft/data` to `r2:soft-serve`, with a Pushover notification. |
 6+| `backup.sh` | Daily (20:30 UTC) `rclone sync` of `/var/soft/data` to `r2:soft-serve`; stamps `last-success` for `/healthz`. |
 7 | `healthz.sh` | `forge-healthz`, every 5 minutes: disk, load, backup age, soft-serve. Writes the `/healthz` response (below). |
 8 | `recover.sh` | Restores `/var/soft/data` from R2 and rebuilds the site. |
 9 
10@@ -39,11 +39,11 @@ Personal git hosting on infra.rtw.run: [soft-serve](https://github.com/charmbrac
11 | `/var/www/code` | The generated site. Not backed up: `forge-rebuild-site` recreates it. |
12 | `/var/log/pgit.log` | Output of every site build. |
13 | `/var/lib/caddy` | Caddy's certificates and ACME account (Caddy runs as `caddy`). |
14-| `/var/lib/servers-keys` | `pushover.env`, `rclone.conf` (Colmena keys). |
15+| `/var/lib/servers-keys` | `rclone.conf` (a Colmena key). |
16 
17 ## /healthz
18 
19-`https://infra.rtw.run/healthz` answers 200 when every check passes, 503 otherwise, with the body sovrn's `metrics-healthz` uses:
20+The monitor (~/projects/monitor) polls `https://infra.rtw.run/healthz` and alerts by email and Pushover. It answers 200 when every check passes, 503 otherwise, with the body sovrn's `metrics-healthz` uses:
21 
22 ```json
23 {
M roles/forge/backup.sh
+3, -23
 1@@ -1,41 +1,21 @@
 2 #!/usr/bin/env bash
 3 set -euo pipefail
 4 
 5-# Pushover credentials come via EnvironmentFile, the rclone config via
 6-# RCLONE_CONFIG (both Colmena keys in /var/lib/servers-keys)
 7-PUSHOVER_DEVICE="git"
 8-PUSHOVER_MESSAGE="git+-+backup+status"
 9+# The rclone config comes via RCLONE_CONFIG (a Colmena key in
10+# /var/lib/servers-keys). A failed backup shows on /healthz once the last
11+# success is over 26h old.
12 
13 log() {
14     echo "[$(date -u +'%Y-%m-%dT%H:%M:%SZ')] $1"
15 }
16 
17-send_notification() {
18-    local status="$1"
19-    local title
20-    local priority
21-    if [[ "$status" == "success" ]]; then
22-        title="backup+successful"
23-        priority=-1
24-    else
25-        title="backup+failed"
26-        priority=0
27-    fi
28-
29-    $CURL -X POST \
30-        "https://api.pushover.net/1/messages.json?priority=${priority}&token=${PUSHOVER_TOKEN}&user=${PUSHOVER_USER}&device=${PUSHOVER_DEVICE}&title=${title}&message=${PUSHOVER_MESSAGE}" \
31-        2>/tmp/backup.log || log "pushover message failed: $(cat /tmp/backup.log)" && true
32-}
33-
34 log "starting soft-serve backup"
35 
36 if $RCLONE sync -L /var/soft/data/ r2:soft-serve/; then
37     log "completed soft-serve backup"
38     # forge-healthz reports the backup stale when this is over 26h old.
39     touch /var/lib/forge-backup/last-success
40-    send_notification "success"
41 else
42     log "soft-serve backup failed"
43-    send_notification "failure"
44     exit 1
45 fi
M roles/forge/default.nix
+5, -19
 1@@ -18,8 +18,7 @@
 2 #   forge-healthz       every 5 minutes: disk, load, backup age, soft-serve;
 3 #                       writes the /healthz response Caddy serves
 4 #
 5-# Secrets (modules/secrets.nix): pushover.env (shared) for notifications,
 6-# rclone.conf (per host) for R2.
 7+# Secrets (modules/secrets.nix): rclone.conf (per host) for R2.
 8 {
 9   config,
10   lib,
11@@ -30,8 +29,6 @@
12 
13 let
14   backupScript = pkgs.writeShellScriptBin "backup" (lib.fileContents ./backup.sh);
15-  prestartNotify = pkgs.writeShellScriptBin "prestart-notify" (lib.fileContents ../../hosts/common/scripts/exec-prestart.sh);
16-  pushoverKey = config.servers.secrets."pushover.env";
17   rcloneKey = config.servers.secrets."rclone.conf";
18 
19   pgitPackage = pgit.packages.${pkgs.stdenv.hostPlatform.system}.default;
20@@ -71,10 +68,7 @@ let
21   '';
22 in
23 {
24-  servers.secrets = {
25-    "pushover.env".scope = "shared";
26-    "rclone.conf".scope = "host";
27-  };
28+  servers.secrets."rclone.conf".scope = "host";
29 
30   environment.systemPackages = with pkgs; [
31     soft-serve
32@@ -118,8 +112,7 @@ in
33 
34   systemd.services.soft-serve = {
35     description = "Soft Serve - A tasty, self-hosted Git server";
36-    after = [ "network.target" pushoverKey.unit ];
37-    wants = [ pushoverKey.unit ];
38+    after = [ "network.target" ];
39     wantedBy = [ "multi-user.target" ];
40     # soft-serve reads its config only at startup.
41     restartTriggers = [ softServeConfig ];
42@@ -127,16 +120,12 @@ in
43     environment = {
44       SOFT_SERVE_CONFIG_LOCATION = "/etc/soft-serve/config.yaml";
45       SOFT_SERVE_DATA_PATH = "/var/soft/data";
46-      CURL = "${pkgs.curl}/bin/curl";
47-      PUSHOVER_DEVICE = config.networking.hostName;
48     };
49 
50     serviceConfig = {
51       Type = "simple";
52       WorkingDirectory = "/var/soft";
53-      ExecStartPre = "${prestartNotify}/bin/prestart-notify soft-serve";
54       ExecStart = "${pkgs.soft-serve}/bin/soft serve";
55-      EnvironmentFile = pushoverKey.path;
56       Restart = "on-failure";
57       RestartSec = 5;
58       User = "root";
59@@ -146,21 +135,18 @@ in
60 
61   systemd.services.backup = {
62     description = "Daily backup service";
63-    wants = [ pushoverKey.unit rcloneKey.unit ];
64-    after = [ pushoverKey.unit rcloneKey.unit ];
65+    wants = [ rcloneKey.unit ];
66+    after = [ rcloneKey.unit ];
67     serviceConfig = {
68       Type = "oneshot";
69       ExecStart = "${backupScript}/bin/backup";
70       User = "root";
71-      EnvironmentFile = pushoverKey.path;
72       # last-success, for forge-healthz
73       StateDirectory = "forge-backup";
74     };
75     environment = {
76       RCLONE = "${pkgs.rclone}/bin/rclone";
77       RCLONE_CONFIG = rcloneKey.path;
78-      CURL = "${pkgs.curl}/bin/curl";
79-      PUSHOVER_DEVICE = config.networking.hostName;
80     };
81   };
82 
M roles/forge/healthz.sh
+1, -1
1@@ -49,7 +49,7 @@ checks='{}'
2 check() {
3   checks=$(jq -c --arg name "$1" --arg status "$2" --arg detail "$3" \
4     '.[$name] = {status: $status, detail: $detail}
5-     + (if $status == "ok" then {} else {error: "\($name): \($detail)"} end)' \
6+     + (if $status == "ok" then {} else {error: $detail} end)' \
7     <<<"$checks")
8 }
9