From d36e959a3eae90a6e071f99c7c1d28ee854b7195 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Fri, 25 Sep 2026 02:09:17 +0000 Subject: [PATCH] ops: make the disk alarm say what is eating the disk, and not invent rates MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two fixes from the alarm's first real firing. ATTRIBUTION. It fired CRITICAL at 57 GB/h and said nothing about the cause. The cause turned out to be 38 GB of docker build cache from an image build, not the databases at all, and establishing that cost a round trip. The alarm now reports docker usage and the largest databases inline, to syslog as well as mail, so the first question anyone asks is already answered. Parsing it needed --format rather than positional awk: "Build Cache" is two whitespace-separated fields in `docker system df`'s table output, which shifts every column and silently prints counts where you expect sizes. The first version reported "build-cache 40" — an object count — as though it were a size. RATE SANITY. Two samples seconds apart produce numbers like 166 GB/h from ordinary write jitter. That pollutes the growth record and can fire a spurious CRITICAL, since the 48h projection divides by it. Rates now need a 300s interval, and a too-short sample leaves the baseline alone so the next real sample still measures from a sensible point. Verified on dev2: attribution prints real sizes (Images 17.41GB, Build Cache 7.156GB, nichedb 163 GB), and two back-to-back runs both report rate=? instead of nonsense. Co-Authored-By: Claude Opus 5 (1M context) --- ops/selfhost/server/setup-supabase.sh | 51 ++++++++++++++++++++++++--- 1 file changed, 46 insertions(+), 5 deletions(-) diff --git a/ops/selfhost/server/setup-supabase.sh b/ops/selfhost/server/setup-supabase.sh index 0bfd672..e14c8d4 100755 --- a/ops/selfhost/server/setup-supabase.sh +++ b/ops/selfhost/server/setup-supabase.sh @@ -155,22 +155,63 @@ used_kb=$(df --output=used / | tail -1 | tr -dc 0-9) now=$(date -u +%s) stamp=$(date -u '+%F %T') +# A rate is only meaningful over a reasonable interval. Two samples seconds +# apart — a manual run right after a cron run, say — produce numbers like +# "166 GB/h" from ordinary write jitter, which pollutes the growth record and +# can fire a spurious CRITICAL. Ignore anything under MIN_INTERVAL, and do not +# overwrite the baseline either, so the next cron sample still measures from a +# sensible point. +MIN_INTERVAL=${MIN_INTERVAL:-300} rate="" +fresh_baseline=1 if [ -f "$STATE" ]; then read -r prev_ts prev_kb < "$STATE" 2>/dev/null || true - if [ -n "${prev_ts:-}" ] && [ "$now" -gt "$prev_ts" ]; then - rate=$(( (used_kb - prev_kb) * 3600 / (now - prev_ts) / 1048576 )) + if [ -n "${prev_ts:-}" ]; then + elapsed=$((now - prev_ts)) + if [ "$elapsed" -ge "$MIN_INTERVAL" ]; then + rate=$(( (used_kb - prev_kb) * 3600 / elapsed / 1048576 )) + else + fresh_baseline=0 + fi fi fi -echo "$now $used_kb" > "$STATE" +[ "$fresh_baseline" = 1 ] && echo "$now $used_kb" > "$STATE" echo "$stamp used=${pct}% ($(df -h --output=used / | tail -1 | tr -d ' ')) rate=${rate:-?}GB/h" >> "$LOG" tail -n 2000 "$LOG" > "$LOG.tmp" && mv "$LOG.tmp" "$LOG" +# What is eating the disk? The first alarm this fired was a 57 GB/h spike that +# turned out to be 38 GB of docker build cache from an image build, not the +# databases at all — and working that out cost a round trip. So the alarm +# attributes its own finding: reclaimable docker space and the biggest +# databases, inline. +whodunnit() { + local out="" + if command -v docker >/dev/null 2>&1; then + # --format, not positional awk: "Build Cache" is two whitespace-separated + # fields in the default table, which shifts every column and silently + # reports counts where you expected sizes. + while IFS='|' read -r type size recl; do + [ -n "$type" ] && out+="docker: ${type} ${size} (${recl} reclaimable) +" + done < <(docker system df --format '{{.Type}}|{{.Size}}|{{.Reclaimable}}' 2>/dev/null) + fi + # Sizes straight from the running cluster, largest first. + if docker exec -i supabase-db psql -U postgres -h localhost -d postgres -X -At \ + -c "select string_agg(datname || ' ' || pg_size_pretty(pg_database_size(datname)), ', ' order by pg_database_size(datname) desc) from pg_database where not datistemplate" >/tmp/.dbsz 2>/dev/null; then + out+="databases: $(cat /tmp/.dbsz) +" + rm -f /tmp/.dbsz + fi + printf '%s' "$out" +} + notify() { logger -t dev2-disk -p daemon.warning "$1: $2" - command -v mail >/dev/null 2>&1 && printf '%s\n\nRecent history:\n%s\n' \ - "$2" "$(tail -12 "$LOG")" | mail -s "dev2 disk $1: ${pct}% used" root 2>/dev/null || true + command -v mail >/dev/null 2>&1 && printf '%s\n\nWhat is using it:\n%s\nRecent history:\n%s\n' \ + "$2" "$(whodunnit)" "$(tail -12 "$LOG")" | mail -s "dev2 disk $1: ${pct}% used" root 2>/dev/null || true + # Attribution also goes to syslog, so it is there even with no mailer. + whodunnit | while IFS= read -r l; do [ -n "$l" ] && logger -t dev2-disk -p daemon.warning " $l"; done } if [ "$pct" -ge "$CRIT" ]; then