diff --git a/CHANGELOG.md b/CHANGELOG.md index d16471e..2d02ef4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,17 @@ Notable changes to this stack. Format follows ## [Unreleased] +### Upgrade + +- Re-vendor `templates/alloy/config.alloy` and `templates/run_scheduled.sh` on + each spoke. Host & Containers needs the first for the new panels and the + corrected host network panel. +- On the hub, `git pull` and restart Grafana once. `retired.yaml` removes the + old per-project rules; delete it after that restart. +- **Breaking:** `just ps`, `logs`, `pull`, `down` and `tail` are gone. Use + `docker compose ps|logs -f|pull|down --remove-orphans`, which read the same + `COMPOSE_FILE` from `.env`. + ### Added - **`HighLatencyP99`**: p99 request latency above 2s for 5 minutes, per @@ -15,6 +26,47 @@ Notable changes to this stack. Format follows minutes. - Service Health shows a **DB connection pool** panel (used and idle) from the OpenTelemetry SQLAlchemy instrumentation. +- **Host & Containers covers the USE method.** Host rows add CPU by mode, + load per core, disk utilisation and throughput, transmit traffic, + temperature and scheduler activity, from collectors the agent already ran. + Container rows add CPU, memory and I/O pressure (PSI wait time, which works + without limits), memory against the limit, and network and disk I/O. + +### Changed + +- **The spoke agent's cAdvisor skips its per-container filesystem walk**, a + third of its CPU on a 12-container host, by enabling only the metric kinds + the stack reads. +- **The spoke agent ships 8 of its own series instead of ~415**: `up`, the + export failures and the export queue gauges. +- The spoke agent batches for 5s instead of 200ms, so a trickle of logs is + one request per 5s rather than up to five a second. +- The hub drops its own Prometheus histogram buckets, like the other + stack jobs. +- GPU panels no longer request exemplars, which the GPU exporter never has. +- **One `ProjectTelemetrySilent` rule covers every project.** `bootstrap.sh` + renders it and the coverage backstop into `projects.yaml`, instead of a file + per project/env. Each silent pair is still its own alert instance, and + removing a project is now one edit to the `# COVERS:` line. +- `run_scheduled.sh` pings `/`; success pings still send + no job output. +- The GPU dashboard drops the MIG and NVLink fabric panels, which only + datacentre cards report. +- Removed config that restated defaults (Loki `server`, datasource `access`, + dashboard provider, Prometheus `evaluation_interval` and its unread + `origin` label, Grafana sign-up), the empty Cloudflare provider block, and + the pre-rename branches of `infra/generate-imports.sh`. +- The demo installs only the OTLP HTTP exporter, without gRPC, in a + single-stage image. + +### Fixed + +- **The host network panel showed the agent container's own interface.** + `/host/proc/net` resolves to the reading process's namespace; the agent now + reads PID 1's, and leaves out bridges and veths. Transmit is shown too. +- `just check` and `bootstrap.sh` no longer stop early on a checkout without + `.env`. +- ADR 0002 no longer claims three GPU alert rules; none exist. ## [0.3.1] - 2026-09-07 diff --git a/bootstrap.sh b/bootstrap.sh index e8d41d7..1d22e99 100755 --- a/bootstrap.sh +++ b/bootstrap.sh @@ -3,8 +3,9 @@ # # ./bootstrap.sh # -# 1. renders the ProjectTelemetrySilent rule and reloads Grafana; -# 2. regenerates the coverage rule that catches projects never bootstrapped; +# 1. renders the ProjectTelemetrySilent rule for every covered project/env; +# 2. renders the coverage rule that catches projects never bootstrapped, and +# reloads Grafana; # 3. creates the project's healthchecks.io checks (if an API key is present); # 4. prints the `.env` block for the project host and the curls that vendor the # templates at a pinned tag. @@ -31,7 +32,7 @@ cd "$root" # A bare ./bootstrap.sh does not load .env, so read single keys out of it. # Sourcing the whole file would drag the stack's secrets into scope. -env_get() { [[ -f .env ]] && sed -n "s/^$1=//p" .env | tail -1; } +env_get() { [[ ! -f .env ]] || sed -n "s/^$1=//p" .env | tail -1; } if [[ -z "${HEALTHCHECKS_API_KEY:-}" ]]; then HEALTHCHECKS_API_KEY="$(env_get HEALTHCHECKS_API_KEY)" @@ -58,37 +59,28 @@ if [[ -z "${BOOTSTRAP_OUT_DIR:-}" ]]; then repo_raw="https://raw.githubusercontent.com/CMLPlatform/monitoring/${tag}/templates" fi -# --------------------------------------------------------------- 1. the keystone rules -# Every covered project/env pair, read back from the COVERS marker in each -# rendered file, plus the pair being bootstrapped now. -pairs="$({ sed -n 's/^# COVERS: //p' "$out_dir"/project-*.yaml 2>/dev/null || true - echo "$project $env_name"; } | sort -u)" -# The markers come off disk and land in a sed replacement and a PromQL label -# value. Refuse the run rather than render a rule that silently never matches; -# the fix is to delete the edited file. +# ------------------------------------------------------- 1-2. keystone rule and backstop +# Every covered project/env pair, read back from the COVERS marker in the rendered +# file, plus the pair being bootstrapped now. +rendered="${out_dir}/projects.yaml" +pairs="$({ sed -n 's/^# COVERS: //p' "$rendered" 2>/dev/null | tr ' /' '\n ' || true + echo "$project $env_name"; } | sed '/^$/d' | sort -u)" +# The marker comes off disk and lands in a sed replacement and a PromQL label +# value. Refuse the run rather than render a rule that silently never matches. while read -r p e; do valid_pair "$p" "$e" \ - || { echo "error: bad '# COVERS:' marker in $out_dir: '$p $e'" >&2; exit 2; } + || { echo "error: bad '# COVERS:' marker in $rendered: '$p $e'" >&2; exit 2; } done <<<"$pairs" -# All pairs are re-rendered, so a template fix reaches every project. -while read -r p e; do - rendered="${out_dir}/project-${p}-${e}.yaml" - sed -e "s/__PROJECT__/${p}/g" \ - -e "s/__ENV__/${e}/g" \ - templates/alerting/project.yaml.tmpl > "$rendered" - echo "rendered $rendered" -done <<<"$pairs" - -# ------------------------------------------------------------ 2. the coverage backstop -covered="$(echo "$pairs" | sed 's| |/|' | paste -sd',' - | sed 's/,/, /g')" -covered_expr="$(echo "$pairs" \ - | sed 's@^\([^ ]*\) \([^ ]*\)$@{__name__=~"telemetry_.+_total", project="\1",env="\2"}@' \ - | paste -sd'@' - | sed 's/@/ or /g')" -sed -e "s@__COVERED__@${covered}@" \ +covers="$(echo "$pairs" | sed 's| |/|' | paste -sd' ' -)" +selector='s@^\([^ ]*\) \([^ ]*\)$@{__name__=~"telemetry_.+_total", project="\1",env="\2"}@' +silent_expr="$(echo "$pairs" | sed -e "$selector" -e 's/.*/absent(&)/' | paste -sd'@' - | sed 's/@/ or /g')" +covered_expr="$(echo "$pairs" | sed "$selector" | paste -sd'@' - | sed 's/@/ or /g')" +sed -e "s@__COVERS__@${covers}@" \ + -e "s@__SILENT_EXPR__@${silent_expr}@" \ -e "s@__COVERED_EXPR__@${covered_expr}@" \ - templates/alerting/coverage.yaml.tmpl > "${out_dir}/coverage.yaml" -echo "rendered ${out_dir}/coverage.yaml (covering: ${covered})" + templates/alerting/projects.yaml.tmpl > "$rendered" +echo "rendered $rendered (covering: ${covers})" [[ -z "${BOOTSTRAP_OUT_DIR:-}" ]] || exit 0 # ------------------------------------------------------------------ 3. reload Grafana @@ -109,26 +101,30 @@ if docker compose ps --status running --services 2>/dev/null | grep -qx grafana; # curl's config parser unescapes \ and " inside a quoted value, so escape both. gpw="${GRAFANA_ADMIN_PASSWORD//\\/\\\\}" gpw="${gpw//\"/\\\"}" - uid="proj-silent-${project}-${env_name}" + # One rule covers every pair, so read it back and look for this pair's selector. + want="project=\\\"${project}\\\",env=\\\"${env_name}\\\"" + verified="" for _ in $(seq 30); do sleep 2 # Password on stdin, never argv. A wrong password would otherwise poll # for a minute and then report the rule as missing. - code="$(printf 'user = "admin:%s"\n' "$gpw" \ - | curl -s -o /dev/null -w '%{http_code}' -K - "http://localhost:3000/api/v1/provisioning/alert-rules/${uid}")" || continue - case "$code" in + resp="$(printf 'user = "admin:%s"\n' "$gpw" \ + | curl -s -w '\n%{http_code}' -K - "http://localhost:3000/api/v1/provisioning/alert-rules/projects-silent")" || continue + case "${resp##*$'\n'}" in 200) - echo "verified rule ${uid} is provisioned" - uid="" - break + if grep -qF "$want" <<<"$resp"; then + echo "verified rule projects-silent covers ${project}/${env_name}" + verified=1 + break + fi ;; 401 | 403) - echo "error: grafana rejected the admin credentials (HTTP ${code}); check GRAFANA_ADMIN_PASSWORD" >&2 + echo "error: grafana rejected the admin credentials (HTTP ${resp##*$'\n'}); check GRAFANA_ADMIN_PASSWORD" >&2 exit 1 ;; esac done - [[ -z "$uid" ]] || { echo "error: grafana restarted but rule ${uid} is not provisioned; check 'just logs grafana' for the rejected file" >&2; exit 1; } + [[ -n "$verified" ]] || { echo "error: grafana restarted but rule projects-silent does not cover ${project}/${env_name}; check 'docker compose logs grafana' for the rejected file" >&2; exit 1; } else echo "note grafana is not running; the rules apply next time it starts" fi diff --git a/compose.yml b/compose.yml index d70abc6..a85dc9e 100644 --- a/compose.yml +++ b/compose.yml @@ -185,7 +185,6 @@ services: # Expanded by Grafana into the provisioned contact points. ALERT_WEBHOOK_URL: ${ALERT_WEBHOOK_URL:-} HEARTBEAT_URL: ${HEARTBEAT_URL:-} - GF_USERS_ALLOW_SIGN_UP: "false" GF_SERVER_ROOT_URL: ${GRAFANA_ROOT_URL:-http://localhost:3000} GF_DASHBOARDS_DEFAULT_HOME_DASHBOARD_PATH: /var/lib/grafana/dashboards/stack-health.json # Secure cookies break plain-http localhost logins, so opt-in. With the diff --git a/config/grafana/alerting/coverage.yaml b/config/grafana/alerting/coverage.yaml deleted file mode 100644 index 82109f5..0000000 --- a/config/grafana/alerting/coverage.yaml +++ /dev/null @@ -1,58 +0,0 @@ -# GENERATED by bootstrap.sh from templates/alerting/coverage.yaml.tmpl. Do not edit -# the rendered file; it is regenerated on every run. -# -# Fires on any series whose project/env pair has no rendered rule file: a project that -# ships telemetry but was never bootstrapped has no ProjectTelemetrySilent rule. -# -# Covered right now: relab/prod, relab/staging (plus demo/demo, exempt: see below) - -apiVersion: 1 - -groups: - - orgId: 1 - name: _coverage - folder: Stack alerts - interval: 5m - rules: - - uid: projects-uncovered - title: ProjectsUncovered - condition: FIRING - for: 15m - noDataState: OK - execErrState: Error - isPaused: false - data: - - refId: QUERY - datasourceUid: prometheus - relativeTimeRange: - from: 3600 - to: 0 - model: - refId: QUERY - instant: true - editorMode: code - # The collector's count connector (config/otel-collector.yaml) mints these - # counters for every sender, whatever signal it sends. A bare - # {project!=""} would scan every project-labelled series in the TSDB on - # each evaluation, and would still miss a logs-only or traces-only project. - # - # demo/demo is this repo's own `just demo` overlay. Bootstrapping it - # would leave ProjectTelemetrySilent firing forever after `just - # demo-down`, so it is exempt as a pair; a real project named demo in - # a real env is still caught. - expr: count by (project, env) ({__name__=~"telemetry_.+_total", project!="", env!=""} unless on (project, env) ({__name__=~"telemetry_.+_total", project="demo", env="demo"} or {__name__=~"telemetry_.+_total", project="relab",env="prod"} or {__name__=~"telemetry_.+_total", project="relab",env="staging"})) - - refId: FIRING - datasourceUid: __expr__ - model: - refId: FIRING - type: threshold - expression: QUERY - conditions: - - evaluator: - type: gt - params: [0] - labels: - severity: warning - annotations: - summary: "{{ $labels.project }}/{{ $labels.env }} is sending telemetry but has no alert rules" - description: "Telemetry is arriving for a project/environment bootstrap.sh was never run for, so nothing would notice if it went silent. Run ./bootstrap.sh {{ $labels.project }} {{ $labels.env }} on the monitoring host. A literal `unknown` means the sender set no project or env resource attribute at all: fix the sender (docs/ONBOARDING.md), do not bootstrap a project by that name." diff --git a/config/grafana/alerting/project-relab-prod.yaml b/config/grafana/alerting/project-relab-prod.yaml deleted file mode 100644 index 3480b3b..0000000 --- a/config/grafana/alerting/project-relab-prod.yaml +++ /dev/null @@ -1,57 +0,0 @@ -# Rendered by bootstrap.sh into config/grafana/alerting/. Do not edit the rendered -# files by hand; edit this template so every project gets the fix. -# -# Nothing on a project host can detect its own absence, so this rule lives here. -# -# bootstrap.sh reads the marker below to regenerate coverage.yaml. It carries the pair, -# not the filename, because both halves may contain a dash. -# COVERS: relab prod - -apiVersion: 1 - -groups: - - orgId: 1 - name: project-relab-prod - folder: Stack alerts - interval: 1m - rules: - # absent() yields nothing while telemetry flows, so NoData is the HEALTHY state - # and must map to OK. - - uid: proj-silent-relab-prod - title: ProjectTelemetrySilent - condition: FIRING - for: 15m - noDataState: OK - execErrState: Error - isPaused: false - data: - - refId: QUERY - datasourceUid: prometheus - relativeTimeRange: - from: 3600 - to: 0 - model: - refId: QUERY - instant: true - editorMode: code - # The collector's ingest counters, as in coverage.yaml.tmpl. absent() loads - # every series its selector matches, once a minute per project, so it is - # bounded to a few per service rather than everything the project sends. - expr: absent({__name__=~"telemetry_.+_total", project="relab",env="prod"}) - - refId: FIRING - datasourceUid: __expr__ - model: - refId: FIRING - type: threshold - expression: QUERY - conditions: - - evaluator: - type: gt - params: [0] - labels: - severity: critical - project: relab - env: prod - annotations: - summary: "No telemetry from relab/prod in 15m" - description: "Host down, Docker down, the agent down, tunnel down, token rotated wrong, or the collector is rejecting this project's data. Nothing on the relab side can detect this on its own. Expect this ~20m after the last sample: absent() needs the series to go stale first." diff --git a/config/grafana/alerting/project-relab-staging.yaml b/config/grafana/alerting/project-relab-staging.yaml deleted file mode 100644 index fda2de9..0000000 --- a/config/grafana/alerting/project-relab-staging.yaml +++ /dev/null @@ -1,57 +0,0 @@ -# Rendered by bootstrap.sh into config/grafana/alerting/. Do not edit the rendered -# files by hand; edit this template so every project gets the fix. -# -# Nothing on a project host can detect its own absence, so this rule lives here. -# -# bootstrap.sh reads the marker below to regenerate coverage.yaml. It carries the pair, -# not the filename, because both halves may contain a dash. -# COVERS: relab staging - -apiVersion: 1 - -groups: - - orgId: 1 - name: project-relab-staging - folder: Stack alerts - interval: 1m - rules: - # absent() yields nothing while telemetry flows, so NoData is the HEALTHY state - # and must map to OK. - - uid: proj-silent-relab-staging - title: ProjectTelemetrySilent - condition: FIRING - for: 15m - noDataState: OK - execErrState: Error - isPaused: false - data: - - refId: QUERY - datasourceUid: prometheus - relativeTimeRange: - from: 3600 - to: 0 - model: - refId: QUERY - instant: true - editorMode: code - # The collector's ingest counters, as in coverage.yaml.tmpl. absent() loads - # every series its selector matches, once a minute per project, so it is - # bounded to a few per service rather than everything the project sends. - expr: absent({__name__=~"telemetry_.+_total", project="relab",env="staging"}) - - refId: FIRING - datasourceUid: __expr__ - model: - refId: FIRING - type: threshold - expression: QUERY - conditions: - - evaluator: - type: gt - params: [0] - labels: - severity: critical - project: relab - env: staging - annotations: - summary: "No telemetry from relab/staging in 15m" - description: "Host down, Docker down, the agent down, tunnel down, token rotated wrong, or the collector is rejecting this project's data. Nothing on the relab side can detect this on its own. Expect this ~20m after the last sample: absent() needs the series to go stale first." diff --git a/config/grafana/alerting/projects.yaml b/config/grafana/alerting/projects.yaml new file mode 100644 index 0000000..e7174cd --- /dev/null +++ b/config/grafana/alerting/projects.yaml @@ -0,0 +1,98 @@ +# Rendered by bootstrap.sh into config/grafana/alerting/projects.yaml. Do not edit +# the rendered file except for the COVERS marker; edit this template instead. +# +# bootstrap.sh reads the covered project/env pairs back from this marker and adds the +# one it is run for. To remove a project, delete its pair here and re-run bootstrap.sh +# for any project that remains. +# COVERS: relab/prod relab/staging + +apiVersion: 1 + +groups: + # Nothing on a project host can detect its own absence, so this rule lives here. + - orgId: 1 + name: projects + folder: Stack alerts + interval: 1m + rules: + # absent() yields nothing while telemetry flows, so NoData is the HEALTHY state + # and must map to OK. Each absent() keeps its project/env equality matchers as + # labels, so every silent pair is its own alert instance. + - uid: projects-silent + title: ProjectTelemetrySilent + condition: FIRING + for: 15m + noDataState: OK + execErrState: Error + isPaused: false + data: + - refId: QUERY + datasourceUid: prometheus + relativeTimeRange: + from: 3600 + to: 0 + model: + refId: QUERY + instant: true + editorMode: code + # The collector's count connector (config/otel-collector.yaml) mints these + # ingest counters for every sender, whatever signal it sends. absent() + # loads every series its selector matches, so this is bounded to a few + # per service rather than everything a project sends. + expr: absent({__name__=~"telemetry_.+_total", project="relab",env="prod"}) or absent({__name__=~"telemetry_.+_total", project="relab",env="staging"}) + - &firing + refId: FIRING + datasourceUid: __expr__ + model: + refId: FIRING + type: threshold + expression: QUERY + conditions: + - evaluator: + type: gt + params: [0] + labels: + severity: critical + annotations: + summary: "No telemetry from {{ $labels.project }}/{{ $labels.env }} in 15m" + description: "Host down, Docker down, the agent down, tunnel down, token rotated wrong, or the collector is rejecting this project's data. Nothing on the project side can detect this on its own. Expect this ~20m after the last sample: absent() needs the series to go stale first." + + # The backstop: a project that ships telemetry but was never bootstrapped has no + # ProjectTelemetrySilent instance. + - orgId: 1 + name: _coverage + folder: Stack alerts + interval: 5m + rules: + - uid: projects-uncovered + title: ProjectsUncovered + condition: FIRING + for: 15m + noDataState: OK + execErrState: Error + isPaused: false + data: + - refId: QUERY + datasourceUid: prometheus + relativeTimeRange: + from: 3600 + to: 0 + model: + refId: QUERY + instant: true + editorMode: code + # A bare {project!=""} would scan every project-labelled series in the + # TSDB on each evaluation, and would still miss a logs-only or + # traces-only project. + # + # demo/demo is this repo's own `just demo` overlay. Bootstrapping it + # would leave ProjectTelemetrySilent firing forever after `just + # demo-down`, so it is exempt as a pair; a real project named demo in + # a real env is still caught. + expr: count by (project, env) ({__name__=~"telemetry_.+_total", project!="", env!=""} unless on (project, env) ({__name__=~"telemetry_.+_total", project="demo", env="demo"} or {__name__=~"telemetry_.+_total", project="relab",env="prod"} or {__name__=~"telemetry_.+_total", project="relab",env="staging"})) + - *firing + labels: + severity: warning + annotations: + summary: "{{ $labels.project }}/{{ $labels.env }} is sending telemetry but has no alert rules" + description: "Telemetry is arriving for a project/environment bootstrap.sh was never run for, so nothing would notice if it went silent. Run ./bootstrap.sh {{ $labels.project }} {{ $labels.env }} on the monitoring host. A literal `unknown` means the sender set no project or env resource attribute at all: fix the sender (docs/ONBOARDING.md), do not bootstrap a project by that name." diff --git a/config/grafana/alerting/retired.yaml b/config/grafana/alerting/retired.yaml new file mode 100644 index 0000000..bad1124 --- /dev/null +++ b/config/grafana/alerting/retired.yaml @@ -0,0 +1,11 @@ +# The per-project ProjectTelemetrySilent rules that projects.yaml replaced. Grafana +# keeps a provisioned rule after its file is gone and only drops it on a deleteRules +# entry. Delete this file once the hub has restarted Grafana on this release. + +apiVersion: 1 + +deleteRules: + - orgId: 1 + uid: proj-silent-relab-prod + - orgId: 1 + uid: proj-silent-relab-staging diff --git a/config/grafana/alerting/rules.yaml b/config/grafana/alerting/rules.yaml index 016d8d1..99a9e8b 100644 --- a/config/grafana/alerting/rules.yaml +++ b/config/grafana/alerting/rules.yaml @@ -78,7 +78,7 @@ groups: summary: "Grafana cannot deliver {{ $labels.integration }} notifications" description: >- Alerts are firing and going nowhere. Most likely an unset ALERT_WEBHOOK_URL, - otherwise the receiver is rejecting. Check `just logs grafana`. + otherwise the receiver is rejecting. Check `docker compose logs grafana`. - orgId: 1 name: stack-health folder: Stack alerts @@ -118,7 +118,7 @@ groups: summary: "OTel Collector failing to export via {{ $labels.exporter }}" description: >- The collector has been failing to deliver telemetry to a backend for - 5 minutes. Check `just logs otel-collector`. + 5 minutes. Check `docker compose logs otel-collector`. - orgId: 1 name: capacity folder: Stack alerts diff --git a/config/grafana/dashboards.yaml b/config/grafana/dashboards.yaml index 2b5d842..05aa281 100644 --- a/config/grafana/dashboards.yaml +++ b/config/grafana/dashboards.yaml @@ -2,9 +2,7 @@ apiVersion: 1 providers: - name: default - folder: "" type: file - allowUiUpdates: false updateIntervalSeconds: 30 options: path: /var/lib/grafana/dashboards diff --git a/config/grafana/datasources.yaml b/config/grafana/datasources.yaml index daefee2..939f98f 100644 --- a/config/grafana/datasources.yaml +++ b/config/grafana/datasources.yaml @@ -4,7 +4,6 @@ datasources: - name: Prometheus type: prometheus uid: prometheus - access: proxy url: http://prometheus:9090 isDefault: true jsonData: @@ -16,7 +15,6 @@ datasources: - name: Loki type: loki uid: loki - access: proxy url: http://loki:3100 jsonData: # trace_id in a log line links to the trace in Tempo. @@ -35,7 +33,6 @@ datasources: - name: Tempo type: tempo uid: tempo - access: proxy url: http://tempo:3200 jsonData: # No tracesToMetrics/serviceMap/nodeGraph: they need span-metrics this diff --git a/config/loki.yaml b/config/loki.yaml index faff492..9647e41 100644 --- a/config/loki.yaml +++ b/config/loki.yaml @@ -2,11 +2,6 @@ auth_enabled: false -server: - http_listen_port: 3100 - grpc_listen_port: 9095 - log_level: info - common: instance_addr: 127.0.0.1 path_prefix: /loki diff --git a/config/otel-collector.yaml b/config/otel-collector.yaml index a909abe..c201d02 100644 --- a/config/otel-collector.yaml +++ b/config/otel-collector.yaml @@ -5,7 +5,8 @@ extensions: bearertokenauth: token: ${env:OTLP_AUTH_TOKEN} # Backs the exporter queues so buffered telemetry survives a collector - # restart. The directory must be writable by uid 10001; `just up` chowns it. + # restart. otel-queue-init (compose.yml) chowns the volume to uid 10001. + # create_directory only matters to `just validate`, which has no volume. file_storage: directory: /var/lib/otelcol/queue create_directory: true diff --git a/config/prometheus.yaml b/config/prometheus.yaml index 92e2fdc..99d0015 100644 --- a/config/prometheus.yaml +++ b/config/prometheus.yaml @@ -1,8 +1,5 @@ global: scrape_interval: 30s - evaluation_interval: 30s - external_labels: - origin: monitoring-host storage: tsdb: @@ -18,9 +15,16 @@ otlp: # Apps push metrics via OTLP. Scrape only what this stack hosts. scrape_configs: + # Only `up`, prometheus_tsdb_head_series and grafana_alerting_notifications_failed_total + # are read from the stack's own jobs; their histogram buckets were 5.2k of 14.1k head + # series. The _sum and _count series stay, so latency is still visible in Explore. - job_name: prometheus static_configs: - targets: ["localhost:9090"] + metric_relabel_configs: &drop_buckets + - source_labels: [__name__] + regex: .*_bucket + action: drop - job_name: otel-collector static_configs: @@ -30,16 +34,10 @@ scrape_configs: static_configs: - targets: ["node-exporter:9100"] - # Only `up` and grafana_alerting_notifications_failed_total are read from these - # three; their histogram buckets were 5.2k of 14.1k head series. The _sum and - # _count series stay, so latency is still visible in Explore. - job_name: grafana static_configs: - targets: ["grafana:3000"] - metric_relabel_configs: &drop_buckets - - source_labels: [__name__] - regex: .*_bucket - action: drop + metric_relabel_configs: *drop_buckets - job_name: loki static_configs: diff --git a/dashboards/gpu.json b/dashboards/gpu.json index 90f570d..ca0414a 100644 --- a/dashboards/gpu.json +++ b/dashboards/gpu.json @@ -481,7 +481,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_pstate{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "interval": "", "legendFormat": "", @@ -729,7 +729,7 @@ "type": "prometheus", "uid": "prometheus" }, - "description": "Percent of time over the past sample period during which one or more kernels was executing on the GPU.\nThe sample period may be between 1 second and 1/6 second depending on the product. MIG-enabled cards do not report a whole-GPU number, which shows as N/A; per-instance activity is in the MIG panels.", + "description": "Percent of time over the past sample period during which one or more kernels was executing on the GPU.\nThe sample period may be between 1 second and 1/6 second depending on the product. MIG-enabled cards do not report a whole-GPU number, which shows as N/A.", "fieldConfig": { "defaults": { "color": { @@ -792,7 +792,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "sum(nvidia_smi_utilization_gpu_ratio{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}) or sum(nvidia_smi_gpu_info{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}) * 0 - 1", "interval": "", "legendFormat": "{{uuid}}", @@ -859,7 +859,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_power_draw_watts{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"} / (nvidia_smi_enforced_power_limit_watts{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"} or nvidia_smi_power_limit_watts{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"} or nvidia_smi_power_default_limit_watts{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"})", "interval": "", "legendFormat": "", @@ -937,7 +937,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "sum(nvidia_smi_fan_speed_ratio{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}) or sum(nvidia_smi_gpu_info{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}) * 0 - 1", "interval": "", "legendFormat": "", @@ -1012,7 +1012,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_temperature_gpu{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "interval": "", "legendFormat": "{{uuid}}", @@ -1079,7 +1079,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_clocks_current_graphics_clock_hz{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"} / nvidia_smi_clocks_max_graphics_clock_hz{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "interval": "", "legendFormat": "", @@ -1146,7 +1146,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_clocks_current_memory_clock_hz{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"} / nvidia_smi_clocks_max_memory_clock_hz{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "interval": "", "legendFormat": "", @@ -1221,7 +1221,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_memory_used_bytes{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"} / nvidia_smi_memory_total_bytes{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "interval": "", "legendFormat": "", @@ -1236,7 +1236,7 @@ "type": "prometheus", "uid": "prometheus" }, - "description": "Percent of time over the past sample period during which global (device) memory was being read or written.\nThe sample period may be between 1 second and 1/6 second depending on the product. MIG-enabled cards do not report a whole-GPU number, which shows as N/A; per-instance activity is in the MIG panels.", + "description": "Percent of time over the past sample period during which global (device) memory was being read or written.\nThe sample period may be between 1 second and 1/6 second depending on the product. MIG-enabled cards do not report a whole-GPU number, which shows as N/A.", "fieldConfig": { "defaults": { "color": { @@ -1299,7 +1299,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "sum(nvidia_smi_utilization_memory_ratio{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}) or sum(nvidia_smi_gpu_info{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}) * 0 - 1", "interval": "", "legendFormat": "", @@ -1397,7 +1397,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_utilization_memory_ratio{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "interval": "", "legendFormat": "{{uuid}}", @@ -1495,7 +1495,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_utilization_gpu_ratio{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "interval": "", "legendFormat": "", @@ -2015,7 +2015,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_memory_used_bytes{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "interval": "", "legendFormat": "{{uuid}}", @@ -2116,7 +2116,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_temperature_gpu{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "interval": "", "legendFormat": "{{uuid}}", @@ -2212,7 +2212,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_power_draw_watts{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "interval": "", "legendFormat": "{{uuid}}", @@ -2310,7 +2310,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_fan_speed_ratio{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "interval": "", "legendFormat": "{{uuid}}", @@ -2416,7 +2416,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_clocks_current_graphics_clock_hz{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "format": "time_series", "interval": "", @@ -2513,7 +2513,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_clocks_current_video_clock_hz{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "format": "time_series", "interval": "", @@ -2610,7 +2610,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_clocks_current_sm_clock_hz{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "format": "time_series", "interval": "", @@ -2707,7 +2707,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_clocks_current_memory_clock_hz{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "format": "time_series", "interval": "", @@ -3087,103 +3087,6 @@ }, "id": 48, "panels": [ - { - "datasource": { - "type": "prometheus", - "uid": "prometheus" - }, - "description": "NVLink fabric registration state. Completed is the healthy steady state. In Progress is normal briefly at boot, but persistent In Progress or Not Started means the GPU is not registered with the fabric manager and NVLink workloads will fail.", - "fieldConfig": { - "defaults": { - "color": { - "mode": "fixed", - "fixedColor": "text" - }, - "decimals": 0, - "mappings": [ - { - "type": "value", - "options": { - "0": { - "text": "Not Supported", - "color": "#6E7B8B", - "index": 0 - }, - "1": { - "text": "Not Started", - "color": "#FF9830", - "index": 1 - }, - "2": { - "text": "In Progress", - "color": "#F2CC0C", - "index": 2 - }, - "3": { - "text": "Completed", - "color": "#56A64B", - "index": 3 - } - } - } - ], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - } - ] - }, - "unit": "", - "noValue": "No NVLink fabric" - }, - "overrides": [] - }, - "gridPos": { - "h": 4, - "w": 6, - "x": 0, - "y": 30 - }, - "id": 37, - "options": { - "colorMode": "background", - "graphMode": "none", - "justifyMode": "auto", - "orientation": "auto", - "percentChangeColorMode": "standard", - "reduceOptions": { - "calcs": [ - "last" - ], - "fields": "", - "values": false - }, - "showPercentChange": false, - "text": {}, - "textMode": "value", - "wideLayout": true - }, - "pluginVersion": "11.2.0", - "targets": [ - { - "datasource": { - "type": "prometheus", - "uid": "prometheus" - }, - "editorMode": "code", - "expr": "nvidia_smi_fabric_state{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", - "instant": false, - "legendFormat": "__auto", - "range": true, - "refId": "A" - } - ], - "title": "Fabric State", - "type": "stat" - }, { "datasource": { "type": "prometheus", @@ -3254,7 +3157,7 @@ "type": "timeseries" } ], - "title": "Datacenter Health", + "title": "ECC", "type": "row" }, { @@ -3523,7 +3426,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_xid_errors_total{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "interval": "", "legendFormat": "XID {{xid}}", @@ -3618,7 +3521,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_pcie_throughput_tx_bytes_per_second{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "interval": "", "legendFormat": "TX", @@ -3629,7 +3532,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_pcie_throughput_rx_bytes_per_second{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "interval": "", "legendFormat": "RX", @@ -3723,7 +3626,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "rate(nvidia_smi_energy_joules_total{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}[$__rate_interval])", "interval": "", "legendFormat": "from energy counter", @@ -3734,7 +3637,7 @@ "type": "prometheus", "uid": "prometheus" }, - "exemplar": true, + "exemplar": false, "expr": "nvidia_smi_power_draw_watts{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}", "interval": "", "legendFormat": "sampled power draw", @@ -3811,294 +3714,9 @@ ], "title": "Energy (trailing 24h)", "type": "stat" - }, - { - "datasource": { - "type": "prometheus", - "uid": "prometheus" - }, - "description": "Framebuffer memory used per MIG GPU instance (memory is a GPU-instance-level resource, shared by its compute instances). Served only by the nvml and demo backends; absent under the default exec backend.", - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "axisBorderShow": false, - "axisCenteredZero": false, - "axisColorMode": "text", - "axisLabel": "", - "axisPlacement": "auto", - "barAlignment": 0, - "barWidthFactor": 0.6, - "drawStyle": "line", - "fillOpacity": 10, - "gradientMode": "none", - "hideFrom": { - "legend": false, - "tooltip": false, - "viz": false - }, - "insertNulls": false, - "lineInterpolation": "linear", - "lineWidth": 1, - "pointSize": 5, - "scaleDistribution": { - "type": "linear" - }, - "showPoints": "never", - "spanNulls": false, - "stacking": { - "group": "A", - "mode": "none" - }, - "thresholdsStyle": { - "mode": "off" - } - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "yellow", - "value": null - } - ] - }, - "unit": "bytes", - "noValue": "No MIG memory data" - }, - "overrides": [] - }, - "gridPos": { - "h": 6, - "w": 12, - "x": 12, - "y": 36 - }, - "id": 60, - "options": { - "legend": { - "calcs": [], - "displayMode": "list", - "placement": "bottom", - "showLegend": true - }, - "tooltip": { - "mode": "multi", - "sort": "none" - } - }, - "pluginVersion": "11.1.0", - "targets": [ - { - "datasource": { - "type": "prometheus", - "uid": "prometheus" - }, - "exemplar": true, - "expr": "(nvidia_smi_mig_memory_used_bytes{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}) * on(uuid, gpu_instance_id) group_left(profile) (max by (uuid, gpu_instance_id, profile) (label_replace(nvidia_smi_mig_info{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}, \"profile\", \"$1\", \"profile\", \"^[0-9]+c\\\\.(.*)$\")))", - "interval": "", - "legendFormat": "GI {{gpu_instance_id}} ({{profile}})", - "refId": "A" - } - ], - "title": "MIG Memory Used", - "type": "timeseries" - }, - { - "datasource": { - "type": "prometheus", - "uid": "prometheus" - }, - "description": "SM activity per MIG GPU instance. Empty on GPUs without MIG partitions; the first collection that sees an instance serves nothing (the sampling needs a pair of collections). Served only by the nvml and demo backends; absent under the default exec backend.", - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "axisBorderShow": false, - "axisCenteredZero": false, - "axisColorMode": "text", - "axisLabel": "", - "axisPlacement": "auto", - "barAlignment": 0, - "barWidthFactor": 0.6, - "drawStyle": "line", - "fillOpacity": 10, - "gradientMode": "none", - "hideFrom": { - "legend": false, - "tooltip": false, - "viz": false - }, - "insertNulls": false, - "lineInterpolation": "linear", - "lineWidth": 1, - "pointSize": 5, - "scaleDistribution": { - "type": "linear" - }, - "showPoints": "never", - "spanNulls": false, - "stacking": { - "group": "A", - "mode": "none" - }, - "thresholdsStyle": { - "mode": "off" - } - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "yellow", - "value": null - } - ] - }, - "unit": "percentunit", - "noValue": "No MIG activity data" - }, - "overrides": [] - }, - "gridPos": { - "h": 6, - "w": 12, - "x": 0, - "y": 42 - }, - "id": 58, - "options": { - "legend": { - "calcs": [], - "displayMode": "list", - "placement": "bottom", - "showLegend": true - }, - "tooltip": { - "mode": "multi", - "sort": "none" - } - }, - "pluginVersion": "11.1.0", - "targets": [ - { - "datasource": { - "type": "prometheus", - "uid": "prometheus" - }, - "exemplar": true, - "expr": "(nvidia_smi_mig_sm_activity_ratio{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}) * on(uuid, gpu_instance_id) group_left(profile) (max by (uuid, gpu_instance_id, profile) (label_replace(nvidia_smi_mig_info{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}, \"profile\", \"$1\", \"profile\", \"^[0-9]+c\\\\.(.*)$\")))", - "interval": "", - "legendFormat": "GI {{gpu_instance_id}} ({{profile}})", - "refId": "A" - } - ], - "title": "MIG SM Activity", - "type": "timeseries" - }, - { - "datasource": { - "type": "prometheus", - "uid": "prometheus" - }, - "description": "Tensor pipe activity per MIG GPU instance, the signal that shows whether an inference workload actually exercises the tensor cores. Served only by the nvml and demo backends; absent under the default exec backend.", - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "axisBorderShow": false, - "axisCenteredZero": false, - "axisColorMode": "text", - "axisLabel": "", - "axisPlacement": "auto", - "barAlignment": 0, - "barWidthFactor": 0.6, - "drawStyle": "line", - "fillOpacity": 10, - "gradientMode": "none", - "hideFrom": { - "legend": false, - "tooltip": false, - "viz": false - }, - "insertNulls": false, - "lineInterpolation": "linear", - "lineWidth": 1, - "pointSize": 5, - "scaleDistribution": { - "type": "linear" - }, - "showPoints": "never", - "spanNulls": false, - "stacking": { - "group": "A", - "mode": "none" - }, - "thresholdsStyle": { - "mode": "off" - } - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "yellow", - "value": null - } - ] - }, - "unit": "percentunit", - "noValue": "No MIG activity data" - }, - "overrides": [] - }, - "gridPos": { - "h": 6, - "w": 12, - "x": 12, - "y": 42 - }, - "id": 59, - "options": { - "legend": { - "calcs": [], - "displayMode": "list", - "placement": "bottom", - "showLegend": true - }, - "tooltip": { - "mode": "multi", - "sort": "none" - } - }, - "pluginVersion": "11.1.0", - "targets": [ - { - "datasource": { - "type": "prometheus", - "uid": "prometheus" - }, - "exemplar": true, - "expr": "(nvidia_smi_mig_tensor_activity_ratio{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}) * on(uuid, gpu_instance_id) group_left(profile) (max by (uuid, gpu_instance_id, profile) (label_replace(nvidia_smi_mig_info{uuid=\"$gpu\", host_name=\"$node\", job=\"$job\"}, \"profile\", \"$1\", \"profile\", \"^[0-9]+c\\\\.(.*)$\")))", - "interval": "", - "legendFormat": "GI {{gpu_instance_id}} ({{profile}})", - "refId": "A" - } - ], - "title": "MIG Tensor Activity", - "type": "timeseries" } ], - "title": "XID / MIG / Power / PCIe (NVML mode only)", + "title": "XID / Power / PCIe (NVML mode only)", "type": "row" } ], diff --git a/dashboards/host-containers.json b/dashboards/host-containers.json index e66e7aa..8a39ba1 100644 --- a/dashboards/host-containers.json +++ b/dashboards/host-containers.json @@ -79,25 +79,37 @@ ] }, "panels": [ + { + "type": "row", + "title": "Host", + "collapsed": false, + "gridPos": { + "h": 1, + "w": 24, + "x": 0, + "y": 0 + }, + "panels": [], + "id": 1 + }, { "type": "timeseries", "title": "Host CPU", - "description": "All cores, aggregated.", + "description": "Busy time by mode across all cores. Sustained iowait means disk-bound; steal means a noisy hypervisor neighbour.", "gridPos": { "h": 8, "w": 8, "x": 0, - "y": 0 + "y": 1 }, - "id": 1, "datasource": { "type": "prometheus", "uid": "prometheus" }, "targets": [ { - "expr": "100 * (1 - avg(rate(node_cpu_seconds_total{project=\"$project\", env=\"$env\", host_name=\"$host\", mode=\"idle\"}[$__rate_interval])))", - "legendFormat": "cpu", + "expr": "100 * avg by (mode) (rate(node_cpu_seconds_total{project=\"$project\", env=\"$env\", host_name=\"$host\", mode!=\"idle\"}[$__rate_interval]))", + "legendFormat": "{{mode}}", "refId": "A" } ], @@ -106,11 +118,17 @@ "unit": "percent", "custom": { "lineWidth": 2, - "fillOpacity": 10 - } + "fillOpacity": 30, + "stacking": { + "mode": "normal" + } + }, + "max": 100, + "min": 0 }, "overrides": [] - } + }, + "id": 2 }, { "type": "timeseries", @@ -120,9 +138,9 @@ "h": 8, "w": 8, "x": 8, - "y": 0 + "y": 1 }, - "id": 2, + "id": 3, "datasource": { "type": "prometheus", "uid": "prometheus" @@ -153,9 +171,9 @@ "h": 8, "w": 8, "x": 16, - "y": 0 + "y": 1 }, - "id": 3, + "id": 4, "datasource": { "type": "prometheus", "uid": "prometheus" @@ -178,6 +196,254 @@ "overrides": [] } }, + { + "type": "timeseries", + "title": "Load per Core", + "description": "Run-queue length divided by core count. Above 1 means work waits for a CPU.", + "gridPos": { + "h": 8, + "w": 8, + "x": 0, + "y": 9 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "node_load1{project=\"$project\", env=\"$env\", host_name=\"$host\"} / scalar(count(node_cpu_seconds_total{project=\"$project\", env=\"$env\", host_name=\"$host\", mode=\"idle\"}))", + "legendFormat": "1m", + "refId": "A" + }, + { + "expr": "node_load5{project=\"$project\", env=\"$env\", host_name=\"$host\"} / scalar(count(node_cpu_seconds_total{project=\"$project\", env=\"$env\", host_name=\"$host\", mode=\"idle\"}))", + "legendFormat": "5m", + "refId": "B" + }, + { + "expr": "node_load15{project=\"$project\", env=\"$env\", host_name=\"$host\"} / scalar(count(node_cpu_seconds_total{project=\"$project\", env=\"$env\", host_name=\"$host\", mode=\"idle\"}))", + "legendFormat": "15m", + "refId": "C" + } + ], + "fieldConfig": { + "defaults": { + "unit": "none", + "custom": { + "lineWidth": 2, + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "id": 5 + }, + { + "type": "timeseries", + "title": "Disk Utilisation", + "description": "Share of time each disk had I/O in flight. Near 100% the disk is saturated.", + "gridPos": { + "h": 8, + "w": 8, + "x": 8, + "y": 9 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "100 * rate(node_disk_io_time_seconds_total{project=\"$project\", env=\"$env\", host_name=\"$host\"}[$__rate_interval])", + "legendFormat": "{{device}}", + "refId": "A" + } + ], + "fieldConfig": { + "defaults": { + "unit": "percent", + "custom": { + "lineWidth": 2, + "fillOpacity": 10 + }, + "max": 100, + "min": 0 + }, + "overrides": [] + }, + "id": 6 + }, + { + "type": "timeseries", + "title": "Disk Throughput", + "description": "Bytes read (up) and written (down) per disk.", + "gridPos": { + "h": 8, + "w": 8, + "x": 16, + "y": 9 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "rate(node_disk_read_bytes_total{project=\"$project\", env=\"$env\", host_name=\"$host\"}[$__rate_interval])", + "legendFormat": "read {{device}}", + "refId": "A" + }, + { + "expr": "-rate(node_disk_written_bytes_total{project=\"$project\", env=\"$env\", host_name=\"$host\"}[$__rate_interval])", + "legendFormat": "write {{device}}", + "refId": "B" + } + ], + "fieldConfig": { + "defaults": { + "unit": "Bps", + "custom": { + "lineWidth": 2, + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "id": 7 + }, + { + "type": "timeseries", + "title": "Host Network", + "description": "Received (up) and transmitted (down) per host interface; container bridges and veths are excluded at the agent.", + "gridPos": { + "h": 8, + "w": 8, + "x": 0, + "y": 17 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "rate(node_network_receive_bytes_total{project=\"$project\", env=\"$env\", host_name=\"$host\"}[$__rate_interval])", + "legendFormat": "rx {{device}}", + "refId": "A" + }, + { + "expr": "-rate(node_network_transmit_bytes_total{project=\"$project\", env=\"$env\", host_name=\"$host\"}[$__rate_interval])", + "legendFormat": "tx {{device}}", + "refId": "B" + } + ], + "fieldConfig": { + "defaults": { + "unit": "Bps", + "custom": { + "lineWidth": 2, + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "id": 8 + }, + { + "type": "timeseries", + "title": "Temperature", + "description": "Hardware sensors by chip. Nothing alerts on this.", + "gridPos": { + "h": 8, + "w": 8, + "x": 8, + "y": 17 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "node_hwmon_temp_celsius{project=\"$project\", env=\"$env\", host_name=\"$host\"}", + "legendFormat": "{{chip}} {{sensor}}", + "refId": "A" + } + ], + "fieldConfig": { + "defaults": { + "unit": "celsius", + "custom": { + "lineWidth": 2, + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "id": 9 + }, + { + "type": "timeseries", + "title": "Scheduler", + "description": "Context switches and interrupts per second, plus runnable and I/O-blocked processes.", + "gridPos": { + "h": 8, + "w": 8, + "x": 16, + "y": 17 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "rate(node_context_switches_total{project=\"$project\", env=\"$env\", host_name=\"$host\"}[$__rate_interval])", + "legendFormat": "context switches/s", + "refId": "A" + }, + { + "expr": "rate(node_intr_total{project=\"$project\", env=\"$env\", host_name=\"$host\"}[$__rate_interval])", + "legendFormat": "interrupts/s", + "refId": "B" + }, + { + "expr": "node_procs_running{project=\"$project\", env=\"$env\", host_name=\"$host\"}", + "legendFormat": "running", + "refId": "C" + }, + { + "expr": "node_procs_blocked{project=\"$project\", env=\"$env\", host_name=\"$host\"}", + "legendFormat": "blocked", + "refId": "D" + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "lineWidth": 2, + "fillOpacity": 10 + } + }, + "overrides": [] + }, + "id": 10 + }, + { + "type": "row", + "title": "Containers", + "collapsed": false, + "gridPos": { + "h": 1, + "w": 24, + "x": 0, + "y": 25 + }, + "panels": [], + "id": 11 + }, { "type": "timeseries", "title": "Container CPU", @@ -186,9 +452,9 @@ "h": 8, "w": 12, "x": 0, - "y": 8 + "y": 26 }, - "id": 4, + "id": 12, "datasource": { "type": "prometheus", "uid": "prometheus" @@ -219,9 +485,9 @@ "h": 8, "w": 12, "x": 12, - "y": 8 + "y": 26 }, - "id": 5, + "id": 13, "datasource": { "type": "prometheus", "uid": "prometheus" @@ -246,57 +512,202 @@ }, { "type": "timeseries", - "title": "Container Restarts (1h)", - "description": "ContainerRestarting fires above 3.", + "title": "Container CPU Pressure", + "description": "Share of time tasks in the container waited on cpu (Linux PSI, `some`). Sustained above a few percent means the container is starved, limit or not.", "gridPos": { "h": 8, - "w": 12, + "w": 8, "x": 0, - "y": 16 + "y": 34 }, - "id": 6, "datasource": { "type": "prometheus", "uid": "prometheus" }, "targets": [ { - "expr": "changes(container_start_time_seconds{project=\"$project\", env=\"$env\", host_name=\"$host\", name!=\"\"}[1h])", + "expr": "100 * sum by (name) (rate(container_pressure_cpu_waiting_seconds_total{project=\"$project\", env=\"$env\", host_name=\"$host\", name!=\"\"}[$__rate_interval]))", "legendFormat": "{{name}}", "refId": "A" } ], "fieldConfig": { "defaults": { - "unit": "short", + "unit": "percent", + "custom": { + "lineWidth": 2, + "fillOpacity": 10 + }, + "min": 0 + }, + "overrides": [] + }, + "id": 14 + }, + { + "type": "timeseries", + "title": "Container Memory Pressure", + "description": "Share of time tasks in the container waited on memory (Linux PSI, `some`). Sustained above a few percent means the container is starved, limit or not.", + "gridPos": { + "h": 8, + "w": 8, + "x": 8, + "y": 34 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "100 * sum by (name) (rate(container_pressure_memory_waiting_seconds_total{project=\"$project\", env=\"$env\", host_name=\"$host\", name!=\"\"}[$__rate_interval]))", + "legendFormat": "{{name}}", + "refId": "A" + } + ], + "fieldConfig": { + "defaults": { + "unit": "percent", + "custom": { + "lineWidth": 2, + "fillOpacity": 10 + }, + "min": 0 + }, + "overrides": [] + }, + "id": 15 + }, + { + "type": "timeseries", + "title": "Container IO Pressure", + "description": "Share of time tasks in the container waited on io (Linux PSI, `some`). Sustained above a few percent means the container is starved, limit or not.", + "gridPos": { + "h": 8, + "w": 8, + "x": 16, + "y": 34 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "100 * sum by (name) (rate(container_pressure_io_waiting_seconds_total{project=\"$project\", env=\"$env\", host_name=\"$host\", name!=\"\"}[$__rate_interval]))", + "legendFormat": "{{name}}", + "refId": "A" + } + ], + "fieldConfig": { + "defaults": { + "unit": "percent", + "custom": { + "lineWidth": 2, + "fillOpacity": 10 + }, + "min": 0 + }, + "overrides": [] + }, + "id": 16 + }, + { + "type": "timeseries", + "title": "Container Memory vs Limit", + "description": "Working set as a share of the memory limit. At 100% the kernel OOM-kills. Only containers with a limit appear.", + "gridPos": { + "h": 8, + "w": 8, + "x": 0, + "y": 42 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "100 * sum by (name) (container_memory_working_set_bytes{project=\"$project\", env=\"$env\", host_name=\"$host\", name!=\"\"}) / sum by (name) (container_spec_memory_limit_bytes{project=\"$project\", env=\"$env\", host_name=\"$host\", name!=\"\"} > 0)", + "legendFormat": "{{name}}", + "refId": "A" + } + ], + "fieldConfig": { + "defaults": { + "unit": "percent", + "custom": { + "lineWidth": 2, + "fillOpacity": 10 + }, + "min": 0 + }, + "overrides": [] + }, + "id": 17 + }, + { + "type": "timeseries", + "title": "Container Network", + "description": "Received (up) and transmitted (down) per container.", + "gridPos": { + "h": 8, + "w": 8, + "x": 8, + "y": 42 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "sum by (name) (rate(container_network_receive_bytes_total{project=\"$project\", env=\"$env\", host_name=\"$host\", name!=\"\"}[$__rate_interval]))", + "legendFormat": "rx {{name}}", + "refId": "A" + }, + { + "expr": "-sum by (name) (rate(container_network_transmit_bytes_total{project=\"$project\", env=\"$env\", host_name=\"$host\", name!=\"\"}[$__rate_interval]))", + "legendFormat": "tx {{name}}", + "refId": "B" + } + ], + "fieldConfig": { + "defaults": { + "unit": "Bps", "custom": { "lineWidth": 2, "fillOpacity": 10 } }, "overrides": [] - } + }, + "id": 18 }, { "type": "timeseries", - "title": "Host Network", - "description": "Receive and transmit, all non-loopback interfaces.", + "title": "Container Disk I/O", + "description": "Bytes read (up) and written (down) per container, all devices.", "gridPos": { "h": 8, - "w": 12, - "x": 12, - "y": 16 + "w": 8, + "x": 16, + "y": 42 }, - "id": 7, "datasource": { "type": "prometheus", "uid": "prometheus" }, "targets": [ { - "expr": "sum by (device) (rate(node_network_receive_bytes_total{project=\"$project\", env=\"$env\", host_name=\"$host\", device!=\"lo\"}[$__rate_interval]))", - "legendFormat": "rx {{device}}", + "expr": "sum by (name) (rate(container_fs_reads_bytes_total{project=\"$project\", env=\"$env\", host_name=\"$host\", name!=\"\"}[$__rate_interval]))", + "legendFormat": "read {{name}}", "refId": "A" + }, + { + "expr": "-sum by (name) (rate(container_fs_writes_bytes_total{project=\"$project\", env=\"$env\", host_name=\"$host\", name!=\"\"}[$__rate_interval]))", + "legendFormat": "write {{name}}", + "refId": "B" } ], "fieldConfig": { @@ -308,6 +719,40 @@ } }, "overrides": [] + }, + "id": 19 + }, + { + "type": "timeseries", + "title": "Container Restarts (1h)", + "description": "ContainerRestarting fires above 3.", + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 50 + }, + "id": 20, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "changes(container_start_time_seconds{project=\"$project\", env=\"$env\", host_name=\"$host\", name!=\"\"}[1h])", + "legendFormat": "{{name}}", + "refId": "A" + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "lineWidth": 2, + "fillOpacity": 10 + } + }, + "overrides": [] } } ] diff --git a/dashboards/service-health.json b/dashboards/service-health.json index 63bcd17..da7be9e 100644 --- a/dashboards/service-health.json +++ b/dashboards/service-health.json @@ -161,7 +161,7 @@ { "type": "timeseries", "title": "Latency percentiles", - "description": "Quantiles over Tempo span-metrics latency. Dots are exemplars \u2014 click one to open the exact trace in Tempo.", + "description": "Quantiles over the applications' own request-duration histograms. Dots are exemplars \u2014 click one to open the exact trace in Tempo.", "gridPos": { "h": 8, "w": 12, diff --git a/demo/Dockerfile b/demo/Dockerfile index 275063e..7cbbbc3 100644 --- a/demo/Dockerfile +++ b/demo/Dockerfile @@ -1,13 +1,9 @@ -FROM python:3.14.7-slim@sha256:51dafde81dbdb6ebde285137a295cf18a47ca95234fe388a343719cb97305b3d AS builder -COPY --from=ghcr.io/astral-sh/uv:0.12.1 /uv /bin/ -WORKDIR /app -COPY pyproject.toml . -RUN uv pip install --system --no-cache -r pyproject.toml - FROM python:3.14.7-slim@sha256:51dafde81dbdb6ebde285137a295cf18a47ca95234fe388a343719cb97305b3d -COPY --from=builder /usr/local/lib/python3.14/site-packages /usr/local/lib/python3.14/site-packages -COPY --from=builder /usr/local/bin /usr/local/bin WORKDIR /app +COPY pyproject.toml . +# uv is mounted for this one step only, so it never lands in the image. +RUN --mount=from=ghcr.io/astral-sh/uv:0.12.1,source=/uv,target=/bin/uv \ + uv pip install --system --no-cache -r pyproject.toml COPY --chown=nobody:nogroup app.py . USER nobody diff --git a/demo/app.py b/demo/app.py index 43ad079..1063767 100644 --- a/demo/app.py +++ b/demo/app.py @@ -16,12 +16,6 @@ app = FastAPI() -@app.get("/") -def root() -> dict[str, bool]: - """Fast, always-successful endpoint.""" - return {"ok": True} - - @app.get("/work") def work() -> dict[str, bool]: """Simulate variable-latency work that sometimes fails.""" diff --git a/demo/pyproject.toml b/demo/pyproject.toml index b3edb9a..afb7af7 100644 --- a/demo/pyproject.toml +++ b/demo/pyproject.toml @@ -3,7 +3,7 @@ "fastapi==0.141.1", "uvicorn==0.54.0", "opentelemetry-distro==0.66b0", - "opentelemetry-exporter-otlp==1.45.0", + "opentelemetry-exporter-otlp-proto-http==1.45.0", "opentelemetry-instrumentation-fastapi==0.66b0", ] name = "demo" diff --git a/docs/RUNBOOK.md b/docs/RUNBOOK.md index 4cc5aed..23c25d0 100644 --- a/docs/RUNBOOK.md +++ b/docs/RUNBOOK.md @@ -1,7 +1,7 @@ # Runbook All commands run from the repo root on the monitoring host. Start with -`just ps` and the **Stack Health** dashboard. Between them they answer most +`docker compose ps` and the **Stack Health** dashboard. Between them they answer most "what is wrong" questions. ![Stack Health dashboard](img/stack-health.png) @@ -12,8 +12,8 @@ the ingest panel is a backup/restore drill. ## A service is down or misbehaving ```sh -just ps # what's running, what's restarting -just logs # follow logs (otel-collector, loki, tempo, prometheus, grafana) +docker compose ps # what's running, what's restarting +docker compose logs -f # follow logs (otel-collector, loki, tempo, prometheus, grafana) just restart ``` @@ -233,14 +233,14 @@ account. Only a plan proves the provider still maps the config to the same resources. Nothing watches the tool images pinned in the `justfile` (yamllint, -actionlint, shellcheck, ruff, gitleaks, OpenTofu, jq, Alloy, alpine). Bump -those by hand. The alpine pin appears twice: in the `justfile` and on -`otel-queue-init` in `compose.yml`. Bump both together. +actionlint, shellcheck, ruff, gitleaks, OpenTofu, jq). Bump those by hand. +The backup recipes reuse the alpine image of `otel-queue-init` in +`compose.yml`. After merging, on the host: ```sh -git pull && just pull && just up +git pull && docker compose pull && just up ``` `just up` adds the exposure guards, but a plain `docker compose up -d` is now diff --git a/docs/adr/0002-hub-and-spoke-observability.md b/docs/adr/0002-hub-and-spoke-observability.md index df9e716..1454c38 100644 --- a/docs/adr/0002-hub-and-spoke-observability.md +++ b/docs/adr/0002-hub-and-spoke-observability.md @@ -63,7 +63,9 @@ Contracts that make it scale: tag (three, plus one for a GPU host), add six `.env` variables, include the overlay, run `bootstrap.sh`. A GPU host is an ordinary host plus one opt-in overlay (`nvidia_gpu_exporter` scraped by Alloy, not dcgm-exporter, whose - profiling fields are datacentre-only) and three GPU alert rules. + profiling fields are datacentre-only) and the GPU dashboard. (Corrected + 2026-10-02: this said "three GPU alert rules", which were never added. GPU + state is diagnosed from the dashboard; nothing pages on it.) ## Alternatives considered diff --git a/infra/generate-imports.sh b/infra/generate-imports.sh index e674f5e..45c61b8 100755 --- a/infra/generate-imports.sh +++ b/infra/generate-imports.sh @@ -6,10 +6,6 @@ # blocks make the adoption reviewable: read the generated file, then the plan, then # apply. # -# The ingestion record was renamed from `otlp.` to `otel.` (2026-09). -# On a state that predates the rename, the live record is imported as -# `cloudflare_dns_record.otel`, so the apply renames it in place. -# # The generated imports.tf is a throwaway: delete it after the apply, or every plan # re-runs the imports. # @@ -69,47 +65,24 @@ lookup_dns_record() { printf '%s' "$id" } -tunnel_name="${MONITORING_TUNNEL_NAME:-cml-monitoring}" -tunnels="$(api "accounts/$account/cfd_tunnel?is_deleted=false")" -tunnel_id="$(jq -r --arg name "$tunnel_name" '.result[] | select(.name == $name) | .id' <<<"$tunnels" | head -1)" -if [[ -z "$tunnel_id" ]]; then - # The tunnel is usually present under another name; list what is there. - echo "error: no tunnel named $tunnel_name in account $account" >&2 - echo "tunnels that do exist in this account:" >&2 - jq -r '.result[]? | " \(.name)\t\(.id)\tconnections=\(.connections | length)"' <<<"$tunnels" >&2 - echo "Re-run with MONITORING_TUNNEL_NAME='' once you know which one serves monitoring." >&2 - exit 1 -fi +# main.tf names the tunnel literally, so an apply would rename any other one. +tunnel_id="$(api "accounts/$account/cfd_tunnel?is_deleted=false" | jq -r '.result[] | select(.name == "cml-monitoring") | .id' | head -1)" +[[ -n "$tunnel_id" ]] || die "no tunnel named cml-monitoring in account $account" grafana_record="$(lookup_dns_record "grafana.$domain")" || { [[ $? -ne 2 ]] || die "DNS lookup for grafana.$domain failed (token lacks DNS:Read?)" die "no CNAME found for grafana.$domain" } -# otel. after the rename, otlp. before it. New name first, so a re-run is a no-op. -for host in "otel.$domain" "otlp.$domain"; do - if otel_record="$(lookup_dns_record "$host")"; then - ingestion_host="$host" - break - elif [[ $? -eq 2 ]]; then - die "DNS lookup for $host failed (token lacks DNS:Read?)" - fi -done -[[ -n "${ingestion_host:-}" ]] || die "no CNAME found for otel.$domain or otlp.$domain" +otel_record="$(lookup_dns_record "otel.$domain")" || { + [[ $? -ne 2 ]] || die "DNS lookup for otel.$domain failed (token lacks DNS:Read?)" + die "no CNAME found for otel.$domain" +} -# The Access app may legitimately not exist; then the apply should create it. A -# zone-scoped app (predating account-scoped Access) cannot be adopted by the -# account-scoped resource, so that case gets its own message. +# The Access app may legitimately not exist; then the apply should create it. access_apps="$(api "accounts/$account/access/apps?per_page=100")" access_app_id="$(jq -r --arg d "grafana.$domain" '.result[] | select(.domain == $d) | .id' <<<"$access_apps" | head -1)" if [[ -z "$access_app_id" ]]; then - zone_app_id="$(api "zones/$zone/access/apps?per_page=100" \ - | jq -r --arg d "grafana.$domain" '.result[]? | select(.domain == $d) | .id' | head -1)" || zone_app_id="" - if [[ -n "$zone_app_id" ]]; then - die "the Access app on grafana.$domain is zone-scoped ($zone_app_id); this root declares an - account-scoped one. Recreate it at the account level, or give the resource - zone_id instead of account_id, before importing." - fi echo "note: no Access application on grafana.$domain; omitting its import block so the" >&2 echo " apply CREATES it. Until that apply lands, Grafana's hostname is protected only" >&2 echo " by its own login. Access apps that do exist in this account:" >&2 @@ -122,7 +95,6 @@ access_policies="$(api "accounts/$account/access/policies?per_page=100")" access_policy_id="$(jq -r '.result[] | select(.name == "monitoring: allowed emails") | .id' <<<"$access_policies" | head -1)" echo "# Generated by generate-imports.sh on $(date -u +%Y-%m-%dT%H:%M:%SZ). Delete after a successful apply." -echo "# Ingestion record adopted from $ingestion_host as cloudflare_dns_record.otel." cat </dev/null so an empty -# dashboards/ doesn't abort every recipe; `lint` refuses the empty list instead. -dash_paths := `ls dashboards/*.json 2>/dev/null | sed 's|^dashboards|/dashboards|' | tr '\n' ' '` +# dashboards/*.json as mounted at /dashboards. +dash_paths := `ls dashboards/*.json | sed 's|^dashboards|/dashboards|' | tr '\n' ' '` # List the recipes. default: @@ -70,10 +68,6 @@ _expose-guards: @for f in .env infra/terraform.tfvars infra/terraform.tfstate infra/terraform.tfstate.backup; do [ ! -e "$f" ] || case "$(stat -c %a "$f")" in *00) ;; *) echo "error: $f is readable by other users (mode $(stat -c %a "$f")); it holds live secrets, run: chmod 600 $f" >&2; exit 1;; esac; done @[ -n "${ALERT_WEBHOOK_URL:-}" ] || { echo "error: ALERT_WEBHOOK_URL is empty; every alert would fire into an empty webhook URL and be dropped. The heartbeat keeps pinging either way, so this failure looks healthy from the outside. Set it, or comment out this guard" >&2; exit 1; } -# Stop the stack; volumes stay. -down: - docker compose down --remove-orphans - # Core stack plus a demo telemetry source, isolated from any running stack (:3002). demo: {{compose_demo}} up -d --build @@ -86,26 +80,10 @@ demo-down: demo-destroy: {{compose_demo}} down --remove-orphans --volumes -# Follow logs, optionally of one service. -logs service="": - docker compose logs -f {{service}} - -# Container status of the stack. -ps: - docker compose ps - # Restart one service. restart service: _guard-if-exposed docker compose restart {{service}} -# Pull the pinned images. -pull: - docker compose pull - -# Tail a service's logs as JSON, decoded. Useful before Grafana is set up. -tail service: - docker compose logs -f --no-log-prefix {{service}} | jq -R 'fromjson? // .' - # Every check runs in a container: no host installs, no network. # Validate every config in the repo. check: lint validate @@ -129,10 +107,9 @@ lint: for bad in OTLP_AUTH_TOKEN=local-dev-token GRAFANA_ADMIN_PASSWORD=change-me GRAFANA_ROOT_URL=http://g.example GRAFANA_COOKIE_SECURE=false CF_ACCESS_AUD= ALERT_WEBHOOK_URL=; do \ ! env $good $bad just _expose-guards 2>/dev/null || { echo "error: exposure guards accepted $bad" >&2; exit 1; }; \ done - # The rendered project-*/coverage rules are skipped (their expr lines grow - # with every project); their templates are checked by rendering them into - # a scratch dir instead. - docker run --rm --network none -v .:/code:ro {{yamllint}} yamllint -d '{extends: relaxed, rules: {line-length: {max: 120, allow-non-breakable-inline-mappings: true}}, ignore: [.git/, backups/, infra/.terraform/, config/grafana/alerting/project-*.yaml, config/grafana/alerting/coverage.yaml]}' . + # The rendered projects.yaml is skipped (its expr lines grow with every + # project); its template is checked by rendering it into a scratch dir instead. + docker run --rm --network none -v .:/code:ro {{yamllint}} yamllint -d '{extends: relaxed, rules: {line-length: {max: 120, allow-non-breakable-inline-mappings: true}}, ignore: [.git/, backups/, infra/.terraform/, config/grafana/alerting/projects.yaml]}' . @d=$(mktemp -d) && BOOTSTRAP_OUT_DIR="$d" ./bootstrap.sh dummy dummy >/dev/null && docker run --rm --network none -v "$d":/code:ro {{yamllint}} yamllint -d '{extends: relaxed, rules: {line-length: disable}}' .; rc=$?; rm -rf "$d"; exit $rc docker run --rm --network none -v .:/repo:ro -w /repo {{actionlint}} -color docker run --rm --network none -v .:/mnt:ro {{shellcheck}} bootstrap.sh scripts/smoke.sh templates/run_scheduled.sh infra/generate-imports.sh @@ -143,7 +120,6 @@ lint: docker run --rm --network none -v ./infra:/infra:ro -w /infra {{tofu}} fmt -check # Dashboards: valid JSON, and every datasource uid they name is provisioned. # A typo provisions fine and renders empty panels. - @[ -n "{{dash_paths}}" ] || { echo "error: no dashboards/*.json to check" >&2; exit 1; } docker run --rm --network none -v ./dashboards:/dashboards:ro {{jq}} empty {{dash_paths}} @bad=$(docker run --rm --network none -v ./dashboards:/dashboards:ro {{jq}} -r '.. | objects | select(has("datasource")) | .datasource | (if type == "object" then .uid else . end) | strings' {{dash_paths}} | sort -u | grep -vxF "$(sed -n 's/^ *uid: *//p' config/grafana/datasources.yaml; echo grafana)"); \ [ -z "$bad" ] || { echo "error: dashboards reference datasource uids that are not provisioned:" $bad >&2; exit 1; } @@ -170,7 +146,7 @@ validate: # container, which has network access to fetch the provider. # Full OpenTofu validation (downloads the provider, so not part of `check`). infra-validate: - @d=$(mktemp -d) && cp infra/main.tf infra/.terraform.lock.hcl "$d"/ && docker run --rm --entrypoint sh -v "$d":/src:ro {{tofu}} -c 'mkdir /work && cp /src/main.tf /src/.terraform.lock.hcl /work && cd /work && tofu init -backend=false -input=false >/dev/null && tofu validate'; rc=$?; rm -rf "$d"; exit $rc + docker run --rm --entrypoint sh -v ./infra/main.tf:/src/main.tf:ro -v ./infra/.terraform.lock.hcl:/src/.terraform.lock.hcl:ro {{tofu}} -c 'cp -r /src /work && cd /work && tofu init -backend=false -input=false >/dev/null && tofu validate' # gitleaks over the staged diff (the pre-commit hook; see .pre-commit-config.yaml). _gitleaks-staged: @@ -196,7 +172,7 @@ _mounts project: _backup project dir: mkdir -p {{dir}} @-docker compose -p {{project}} unpause {{stateful}} >/dev/null 2>&1 - @rc=0; m=$(just _mounts {{project}}); docker compose -p {{project}} pause {{stateful}} && docker run --rm --network none $m -v {{absolute_path(dir)}}:/backups {{alpine}} sh -c 'set -o pipefail; umask 077 && tar cf - -C /data . | gzip -1 > /backups/monitoring-$(date +%Y%m%d-%H%M%S).tar.gz' || rc=$?; docker compose -p {{project}} unpause {{stateful}} || { echo "error: unpause failed; the stack is still paused" >&2; rc=1; }; exit $rc + @rc=0; img=$(just _image compose.yml otel-queue-init); m=$(just _mounts {{project}}); docker compose -p {{project}} pause {{stateful}} && docker run --rm --network none $m -v {{absolute_path(dir)}}:/backups "$img" sh -c 'set -o pipefail; umask 077 && tar cf - -C /data . | gzip -1 > /backups/monitoring-$(date +%Y%m%d-%H%M%S).tar.gz' || rc=$?; docker compose -p {{project}} unpause {{stateful}} || { echo "error: unpause failed; the stack is still paused" >&2; rc=1; }; exit $rc @ls -lh {{dir}}/ | tail -1 # Restore a backup tarball into the volumes (stops the stack; wipes current state). @@ -208,17 +184,17 @@ restore file: (_restore core_project file "backups") _restore project file dir: @[ -f "{{file}}" ] || { echo "error: {{file}} not found" >&2; exit 1; } @for s in {{stateful}}; do docker volume inspect {{project}}_${s}_data > /dev/null 2>&1 || { echo "error: volume {{project}}_${s}_data does not exist. 'docker run -v' would create it empty, so the pre-restore snapshot below would be a tarball of nothing. On a fresh host run 'just up' once first; otherwise check COMPOSE_PROJECT_NAME." >&2; exit 1; }; done - docker run --rm --network none -v {{absolute_path(file)}}:/backup.tar.gz:ro {{alpine}} tar tzf /backup.tar.gz > /dev/null + docker run --rm --network none -v {{absolute_path(file)}}:/backup.tar.gz:ro $(just _image compose.yml otel-queue-init) tar tzf /backup.tar.gz > /dev/null docker compose -p {{project}} down --remove-orphans mkdir -p {{dir}} - m=$(just _mounts {{project}}); docker run --rm --network none $m -v {{absolute_path(dir)}}:/backups {{alpine}} sh -c 'set -o pipefail; umask 077 && tar cf - -C /data . | gzip -1 > /backups/pre-restore-$(date +%Y%m%d-%H%M%S).tar.gz' - m=$(just _mounts {{project}}); docker run --rm --network none $m -v {{absolute_path(file)}}:/backup.tar.gz:ro {{alpine}} sh -c 'for d in /data/*; do find "$d" -mindepth 1 -delete; done && tar xzf /backup.tar.gz -C /data' + m=$(just _mounts {{project}}); docker run --rm --network none $m -v {{absolute_path(dir)}}:/backups $(just _image compose.yml otel-queue-init) sh -c 'set -o pipefail; umask 077 && tar cf - -C /data . | gzip -1 > /backups/pre-restore-$(date +%Y%m%d-%H%M%S).tar.gz' + m=$(just _mounts {{project}}); docker run --rm --network none $m -v {{absolute_path(file)}}:/backup.tar.gz:ro $(just _image compose.yml otel-queue-init) sh -c 'for d in /data/*; do find "$d" -mindepth 1 -delete; done && tar xzf /backup.tar.gz -C /data' # Needs a booted smoke stack (`just smoke`). Run it after touching _backup/_restore. # COMPOSE_FILE is pinned so the host's overlay list stays out, as for `smoke`. # Round-trip backup and restore on the smoke stack. restore-check: - @export COMPOSE_FILE=compose.yml:compose.sandbox.yml; d=$(mktemp -d) && just _backup {{smoke_project}} "$d" && f=$(ls "$d"/monitoring-*.tar.gz) && just _restore {{smoke_project}} "$f" "$d" && docker run --rm --network none -v {{smoke_project}}_grafana_data:/g:ro {{alpine}} test -s /g/grafana.db && echo "Backup round-trip ok"; rc=$?; rm -rf "$d"; exit $rc + @export COMPOSE_FILE=compose.yml:compose.sandbox.yml; d=$(mktemp -d) && just _backup {{smoke_project}} "$d" && f=$(ls "$d"/monitoring-*.tar.gz) && just _restore {{smoke_project}} "$f" "$d" && docker run --rm --network none -v {{smoke_project}}_grafana_data:/g:ro $(just _image compose.yml otel-queue-init) test -s /g/grafana.db && echo "Backup round-trip ok"; rc=$?; rm -rf "$d"; exit $rc # `--wait` blocks on the healthchecks and fails if any container exits, so a # crash-looping service is caught (after the full timeout). scripts/smoke.sh @@ -228,7 +204,7 @@ smoke: {{compose_smoke}} up -d --wait --wait-timeout 120 {{smoke_env}} SMOKE_URL=http://localhost:{{smoke_port}} SMOKE_PROJECT={{smoke_project}} scripts/smoke.sh -# Logs from the smoke stack (its own project, so `just logs` will not show it). +# Logs from the smoke stack (its own project, so `docker compose logs` will not show it). smoke-logs: {{compose_smoke}} logs --no-color --tail=200 diff --git a/templates/README.md b/templates/README.md index 12ff9c0..83d665b 100644 --- a/templates/README.md +++ b/templates/README.md @@ -32,8 +32,8 @@ stack's own alert rules. Skipping `bootstrap.sh` leaves a project unmonitored with no error anywhere. The rule that notices a host's *silence* lives on this stack, not on the host. The backstop is `ProjectsUncovered`, regenerated on every bootstrap -run: it fires on any project the gateway counts telemetry for that has no -rendered rule file, whichever signal that project sends. +run: it fires on any project the gateway counts telemetry for that is not +covered in `projects.yaml`, whichever signal that project sends. ## Two settings that silently produce nothing @@ -83,27 +83,10 @@ from [14574](https://grafana.com/grafana/dashboards/14574)) and ## Removing a project -Deleting `config/grafana/alerting/project--.yaml` is **not** -enough. Grafana provisioning never deletes a rule because its file vanished, -and the API refuses to delete a provisioned rule (409). The orphan keeps -evaluating and firing. - -1. Delete the project file. -2. Add a temporary provisioning file that drops the rule: - - ```yaml - # config/grafana/alerting/zz-delete.yaml (temporary) - apiVersion: 1 - deleteRules: - - orgId: 1 - uid: proj-silent-- - ``` - -3. Restart Grafana and confirm the rule group is gone. -4. Remove `zz-delete.yaml` and restart Grafana again. Left in place, it - deletes the rule again the next time `bootstrap.sh` renders it. -5. Re-run `bootstrap.sh` for a project that remains, so `coverage.yaml` stops - listing the removed one as covered. +Delete its pair from the `# COVERS:` line in +`config/grafana/alerting/projects.yaml`, re-run `bootstrap.sh` for any project +that remains, and commit the result. The one ProjectTelemetrySilent rule then +no longer checks it, and the coverage rule no longer counts it as covered. ## Limits diff --git a/templates/alerting/coverage.yaml.tmpl b/templates/alerting/coverage.yaml.tmpl deleted file mode 100644 index d337bf5..0000000 --- a/templates/alerting/coverage.yaml.tmpl +++ /dev/null @@ -1,58 +0,0 @@ -# GENERATED by bootstrap.sh from templates/alerting/coverage.yaml.tmpl. Do not edit -# the rendered file; it is regenerated on every run. -# -# Fires on any series whose project/env pair has no rendered rule file: a project that -# ships telemetry but was never bootstrapped has no ProjectTelemetrySilent rule. -# -# Covered right now: __COVERED__ (plus demo/demo, exempt: see below) - -apiVersion: 1 - -groups: - - orgId: 1 - name: _coverage - folder: Stack alerts - interval: 5m - rules: - - uid: projects-uncovered - title: ProjectsUncovered - condition: FIRING - for: 15m - noDataState: OK - execErrState: Error - isPaused: false - data: - - refId: QUERY - datasourceUid: prometheus - relativeTimeRange: - from: 3600 - to: 0 - model: - refId: QUERY - instant: true - editorMode: code - # The collector's count connector (config/otel-collector.yaml) mints these - # counters for every sender, whatever signal it sends. A bare - # {project!=""} would scan every project-labelled series in the TSDB on - # each evaluation, and would still miss a logs-only or traces-only project. - # - # demo/demo is this repo's own `just demo` overlay. Bootstrapping it - # would leave ProjectTelemetrySilent firing forever after `just - # demo-down`, so it is exempt as a pair; a real project named demo in - # a real env is still caught. - expr: count by (project, env) ({__name__=~"telemetry_.+_total", project!="", env!=""} unless on (project, env) ({__name__=~"telemetry_.+_total", project="demo", env="demo"} or __COVERED_EXPR__)) - - refId: FIRING - datasourceUid: __expr__ - model: - refId: FIRING - type: threshold - expression: QUERY - conditions: - - evaluator: - type: gt - params: [0] - labels: - severity: warning - annotations: - summary: "{{ $labels.project }}/{{ $labels.env }} is sending telemetry but has no alert rules" - description: "Telemetry is arriving for a project/environment bootstrap.sh was never run for, so nothing would notice if it went silent. Run ./bootstrap.sh {{ $labels.project }} {{ $labels.env }} on the monitoring host. A literal `unknown` means the sender set no project or env resource attribute at all: fix the sender (docs/ONBOARDING.md), do not bootstrap a project by that name." diff --git a/templates/alerting/project.yaml.tmpl b/templates/alerting/project.yaml.tmpl deleted file mode 100644 index 3afcedc..0000000 --- a/templates/alerting/project.yaml.tmpl +++ /dev/null @@ -1,57 +0,0 @@ -# Rendered by bootstrap.sh into config/grafana/alerting/. Do not edit the rendered -# files by hand; edit this template so every project gets the fix. -# -# Nothing on a project host can detect its own absence, so this rule lives here. -# -# bootstrap.sh reads the marker below to regenerate coverage.yaml. It carries the pair, -# not the filename, because both halves may contain a dash. -# COVERS: __PROJECT__ __ENV__ - -apiVersion: 1 - -groups: - - orgId: 1 - name: project-__PROJECT__-__ENV__ - folder: Stack alerts - interval: 1m - rules: - # absent() yields nothing while telemetry flows, so NoData is the HEALTHY state - # and must map to OK. - - uid: proj-silent-__PROJECT__-__ENV__ - title: ProjectTelemetrySilent - condition: FIRING - for: 15m - noDataState: OK - execErrState: Error - isPaused: false - data: - - refId: QUERY - datasourceUid: prometheus - relativeTimeRange: - from: 3600 - to: 0 - model: - refId: QUERY - instant: true - editorMode: code - # The collector's ingest counters, as in coverage.yaml.tmpl. absent() loads - # every series its selector matches, once a minute per project, so it is - # bounded to a few per service rather than everything the project sends. - expr: absent({__name__=~"telemetry_.+_total", project="__PROJECT__",env="__ENV__"}) - - refId: FIRING - datasourceUid: __expr__ - model: - refId: FIRING - type: threshold - expression: QUERY - conditions: - - evaluator: - type: gt - params: [0] - labels: - severity: critical - project: __PROJECT__ - env: __ENV__ - annotations: - summary: "No telemetry from __PROJECT__/__ENV__ in 15m" - description: "Host down, Docker down, the agent down, tunnel down, token rotated wrong, or the collector is rejecting this project's data. Nothing on the __PROJECT__ side can detect this on its own. Expect this ~20m after the last sample: absent() needs the series to go stale first." diff --git a/templates/alerting/projects.yaml.tmpl b/templates/alerting/projects.yaml.tmpl new file mode 100644 index 0000000..f3a4636 --- /dev/null +++ b/templates/alerting/projects.yaml.tmpl @@ -0,0 +1,98 @@ +# Rendered by bootstrap.sh into config/grafana/alerting/projects.yaml. Do not edit +# the rendered file except for the COVERS marker; edit this template instead. +# +# bootstrap.sh reads the covered project/env pairs back from this marker and adds the +# one it is run for. To remove a project, delete its pair here and re-run bootstrap.sh +# for any project that remains. +# COVERS: __COVERS__ + +apiVersion: 1 + +groups: + # Nothing on a project host can detect its own absence, so this rule lives here. + - orgId: 1 + name: projects + folder: Stack alerts + interval: 1m + rules: + # absent() yields nothing while telemetry flows, so NoData is the HEALTHY state + # and must map to OK. Each absent() keeps its project/env equality matchers as + # labels, so every silent pair is its own alert instance. + - uid: projects-silent + title: ProjectTelemetrySilent + condition: FIRING + for: 15m + noDataState: OK + execErrState: Error + isPaused: false + data: + - refId: QUERY + datasourceUid: prometheus + relativeTimeRange: + from: 3600 + to: 0 + model: + refId: QUERY + instant: true + editorMode: code + # The collector's count connector (config/otel-collector.yaml) mints these + # ingest counters for every sender, whatever signal it sends. absent() + # loads every series its selector matches, so this is bounded to a few + # per service rather than everything a project sends. + expr: __SILENT_EXPR__ + - &firing + refId: FIRING + datasourceUid: __expr__ + model: + refId: FIRING + type: threshold + expression: QUERY + conditions: + - evaluator: + type: gt + params: [0] + labels: + severity: critical + annotations: + summary: "No telemetry from {{ $labels.project }}/{{ $labels.env }} in 15m" + description: "Host down, Docker down, the agent down, tunnel down, token rotated wrong, or the collector is rejecting this project's data. Nothing on the project side can detect this on its own. Expect this ~20m after the last sample: absent() needs the series to go stale first." + + # The backstop: a project that ships telemetry but was never bootstrapped has no + # ProjectTelemetrySilent instance. + - orgId: 1 + name: _coverage + folder: Stack alerts + interval: 5m + rules: + - uid: projects-uncovered + title: ProjectsUncovered + condition: FIRING + for: 15m + noDataState: OK + execErrState: Error + isPaused: false + data: + - refId: QUERY + datasourceUid: prometheus + relativeTimeRange: + from: 3600 + to: 0 + model: + refId: QUERY + instant: true + editorMode: code + # A bare {project!=""} would scan every project-labelled series in the + # TSDB on each evaluation, and would still miss a logs-only or + # traces-only project. + # + # demo/demo is this repo's own `just demo` overlay. Bootstrapping it + # would leave ProjectTelemetrySilent firing forever after `just + # demo-down`, so it is exempt as a pair; a real project named demo in + # a real env is still caught. + expr: count by (project, env) ({__name__=~"telemetry_.+_total", project!="", env!=""} unless on (project, env) ({__name__=~"telemetry_.+_total", project="demo", env="demo"} or __COVERED_EXPR__)) + - *firing + labels: + severity: warning + annotations: + summary: "{{ $labels.project }}/{{ $labels.env }} is sending telemetry but has no alert rules" + description: "Telemetry is arriving for a project/environment bootstrap.sh was never run for, so nothing would notice if it went silent. Run ./bootstrap.sh {{ $labels.project }} {{ $labels.env }} on the monitoring host. A literal `unknown` means the sender set no project or env resource attribute at all: fix the sender (docs/ONBOARDING.md), do not bootstrap a project by that name." diff --git a/templates/alloy/config.alloy b/templates/alloy/config.alloy index 199c347..044c7a2 100644 --- a/templates/alloy/config.alloy +++ b/templates/alloy/config.alloy @@ -99,8 +99,8 @@ otelcol.processor.transform "resource_attributes" { // --------------------------------------------------------------------------- // Host metrics: node_exporter's collectors, scraped in-process and converted to OTLP. -// set_collectors is an allowlist, which keeps the series count down. `hwmon` feeds a -// dashboard panel only; nothing alerts on temperature. +// set_collectors is an allowlist, which keeps the series count down. Every collector +// feeds a Host & Containers panel; nothing alerts on temperature, load or disk I/O. prometheus.exporter.unix "host" { procfs_path = "/host/proc" sysfs_path = "/host/sys" @@ -113,7 +113,6 @@ prometheus.exporter.unix "host" { "hwmon", "loadavg", "meminfo", - "netdev", "stat", "uname", ] @@ -141,6 +140,13 @@ prometheus.exporter.cadvisor "containers" { "com.docker.compose.project", "com.docker.compose.service", ] + + // cAdvisor collects every enabled kind about once a second per container, long + // before the keep-list in cadvisor_scope runs. These cover the kept families + // (start_time and spec_memory_limit are always on). The defaults add `disk`, a + // filesystem walk per container per tick that was a third of cAdvisor's CPU, plus + // kinds nothing reads. Widen this together with that keep-list. + enabled_metrics = ["cpu", "memory", "oom_event", "network", "diskIO", "pressure"] } prometheus.scrape "cadvisor" { @@ -162,17 +168,49 @@ prometheus.relabel "cadvisor_scope" { action = "keep" } - // cAdvisor emits ~59 metric families per container (per-cpu, per-disk, per-NIC, - // TCP state); the stack reads four. The rest is the fastest-growing block in the - // series budget, since it scales per container per host. Add a name here when a - // dashboard or rule starts using one. + // The enabled kinds still emit ~40 families per container; Host & Containers and + // the alerts read these, one per USE question (usage, saturation as PSI wait time, + // errors as OOMs), with network and disk I/O traffic alongside. The rest is the + // fastest-growing block in the series budget, since it scales per container per + // host. Add a name here, and its kind to enabled_metrics above, when a dashboard or + // rule starts using one. rule { source_labels = ["__name__"] - regex = "container_(start_time_seconds|oom_events_total|cpu_usage_seconds_total|memory_working_set_bytes)" + regex = "container_(start_time_seconds|oom_events_total|cpu_usage_seconds_total|memory_working_set_bytes|spec_memory_limit_bytes|pressure_(cpu|memory|io)_waiting_seconds_total|network_(receive|transmit)_bytes_total|fs_(reads|writes)_bytes_total)" action = "keep" } } +// netdev reads /net/dev, and /host/proc/net links to self/net: this +// container's namespace, so the host collector above would report Alloy's own veth. +// PID 1's view is the host's. Bridges and veths are per-container noise. +prometheus.exporter.unix "host_net" { + procfs_path = "/host/proc/1" + set_collectors = ["netdev"] + + netdev { + device_exclude = "^(lo|veth.+|br-.+|docker[0-9]+)$" + } +} + +prometheus.scrape "host_net" { + targets = prometheus.exporter.unix.host_net.targets + forward_to = [prometheus.relabel.host_net.receiver] + scrape_interval = "30s" + job_name = "host_net" +} + +// Both unix exporters label their targets job="integrations/unix" and the same +// instance, whatever job_name says, so their `up` series would collide. +prometheus.relabel "host_net" { + forward_to = [prometheus.relabel.host.receiver] + + rule { + target_label = "job" + replacement = "integrations/unix_net" + } +} + prometheus.scrape "host" { targets = prometheus.exporter.unix.host.targets forward_to = [prometheus.relabel.host.receiver] @@ -187,20 +225,17 @@ prometheus.relabel "host" { rule { target_label = "env" replacement = sys.env("ENVIRONMENT") - action = "replace" } rule { target_label = "project" replacement = sys.env("PROJECT") - action = "replace" } // See the note on the log labels. rule { target_label = "host_name" replacement = coalesce(string.trim_space(local.file.hostname.content), constants.hostname) - action = "replace" } } @@ -216,15 +251,16 @@ prometheus.scrape "alloy" { job_name = "alloy" } -// Alloy is the only exporter in this file with histograms, and nothing central -// reads them: 345 of this job's 760 series. The _sum and _count series survive. +// Of this job's ~760 series the hub reads only the export failures; the queue gauges +// show how close an outage came to dropping data. `up` passes through relabelling +// here, unlike in Prometheus, so it is kept explicitly. prometheus.relabel "alloy_self" { forward_to = [prometheus.relabel.host.receiver] rule { source_labels = ["__name__"] - regex = ".*_bucket" - action = "drop" + regex = "up|otelcol_exporter_(send_failed|queue)_.+" + action = "keep" } } @@ -232,16 +268,11 @@ prometheus.relabel "alloy_self" { // GPU metrics, present only on hosts that run compose.telemetry.gpu.yml. Without the // exporter container this finds no targets and costs nothing. discovery.relabel "gpu_exporter" { - targets = discovery.docker.containers.targets - - rule { - source_labels = ["__meta_docker_container_label_com_docker_compose_project"] - regex = sys.env("COMPOSE_PROJECT_NAME") - action = "keep" - } + // Already scoped to this Compose project, with service_name set. + targets = discovery.relabel.containers.output rule { - source_labels = ["__meta_docker_container_label_com_docker_compose_service"] + source_labels = ["service_name"] regex = "nvidia-gpu-exporter" action = "keep" } @@ -249,7 +280,6 @@ discovery.relabel "gpu_exporter" { rule { target_label = "service_name" replacement = "gpu" - action = "replace" } } @@ -281,7 +311,12 @@ otelcol.processor.memory_limiter "default" { } } +// The 200ms default sends a request per trickle of log lines, each through the +// tunnel and the hub's auth. 5s matches the hub's own batching and no alert is +// that tight. otelcol.processor.batch "default" { + timeout = "5s" + output { logs = [otelcol.exporter.otlphttp.central.input] metrics = [otelcol.exporter.otlphttp.central.input] diff --git a/templates/run_scheduled.sh b/templates/run_scheduled.sh index 3237fa0..eff5579 100755 --- a/templates/run_scheduled.sh +++ b/templates/run_scheduled.sh @@ -46,11 +46,14 @@ ping_url="${!url_var:-}" output_file="$(mktemp)" trap 'rm -f "$output_file"' EXIT -# The job's output is the failure body, so the alert carries the reason. This puts job -# output in a third party's hands; keep credentials out of it. -ping_fail() { - curl -fsS -m 10 --retry 3 --data-binary "@${output_file}" "${ping_url}/fail" -o /dev/null \ - || echo "WARNING: failure ping to ${url_var} failed" >&2 +# healthchecks.io reads /: 0 is success, anything else a failure. +# A failure carries the job's output, so the alert has the reason. That puts job output +# in a third party's hands; keep credentials out of it. +ping_status() { + local body=() + [[ "$1" == 0 ]] || body=(--data-binary "@${output_file}") + curl -fsS -m 10 --retry 3 "${body[@]}" "${ping_url}/$1" -o /dev/null \ + || echo "WARNING: ping to ${url_var} failed" >&2 } # A killed job must still report: systemd's TimeoutStartSec TERMs the whole cgroup, and @@ -62,7 +65,7 @@ on_terminate() { echo "run_scheduled: received SIG${sig}; job killed (likely a systemd timeout)" >>"$output_file" cat "$output_file" if [[ -n "$ping_url" ]]; then - ping_fail + ping_status 143 fi rm -f "$output_file" exit 143 @@ -75,15 +78,5 @@ status=0 wait $! || status=$? cat "$output_file" -if [[ -z "$ping_url" ]]; then - exit "$status" -fi - -if [[ "$status" -eq 0 ]]; then - curl -fsS -m 10 --retry 3 "$ping_url" -o /dev/null \ - || echo "WARNING: success ping to ${url_var} failed" >&2 -else - ping_fail -fi - +[[ -z "$ping_url" ]] || ping_status "$status" exit "$status"