Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 3 additions & 2 deletions deploy/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -45,9 +45,10 @@ if onboarding needs one of these edited, report that upstream as a bug.
| `deploy/alloy/config.alloy` | `templates/alloy/config.alloy` |
| `scripts/run_scheduled.sh` | `templates/run_scheduled.sh` |

Vendored at **`v0.3.0`**. Update that tag in the same commit that re-vendors:
Vendored at **`v0.4.0`**, except that `compose.telemetry.gpu.yml` carries a newer
`nvidia_gpu_exporter` pin than that tag. Update the tag in the same commit that re-vendors:

```sh
git -C ../monitoring checkout v0.3.0 # or the newer tag being adopted
git -C ../monitoring checkout v0.4.0 # or the newer tag being adopted
diff -u ../monitoring/templates/compose.telemetry.yml compose.telemetry.yml
```
83 changes: 59 additions & 24 deletions deploy/alloy/config.alloy
Original file line number Diff line number Diff line change
Expand Up @@ -99,8 +99,8 @@ otelcol.processor.transform "resource_attributes" {

// ---------------------------------------------------------------------------
// Host metrics: node_exporter's collectors, scraped in-process and converted to OTLP.
// set_collectors is an allowlist, which keeps the series count down. `hwmon` feeds a
// dashboard panel only; nothing alerts on temperature.
// set_collectors is an allowlist, which keeps the series count down. Every collector
// feeds a Host & Containers panel; nothing alerts on temperature, load or disk I/O.
prometheus.exporter.unix "host" {
procfs_path = "/host/proc"
sysfs_path = "/host/sys"
Expand All @@ -113,7 +113,6 @@ prometheus.exporter.unix "host" {
"hwmon",
"loadavg",
"meminfo",
"netdev",
"stat",
"uname",
]
Expand Down Expand Up @@ -141,6 +140,13 @@ prometheus.exporter.cadvisor "containers" {
"com.docker.compose.project",
"com.docker.compose.service",
]

// cAdvisor collects every enabled kind about once a second per container, long
// before the keep-list in cadvisor_scope runs. These cover the kept families
// (start_time and spec_memory_limit are always on). The defaults add `disk`, a
// filesystem walk per container per tick that was a third of cAdvisor's CPU, plus
// kinds nothing reads. Widen this together with that keep-list.
enabled_metrics = ["cpu", "memory", "oom_event", "network", "diskIO", "pressure"]
}

prometheus.scrape "cadvisor" {
Expand All @@ -162,17 +168,49 @@ prometheus.relabel "cadvisor_scope" {
action = "keep"
}

// cAdvisor emits ~59 metric families per container (per-cpu, per-disk, per-NIC,
// TCP state); the stack reads four. The rest is the fastest-growing block in the
// series budget, since it scales per container per host. Add a name here when a
// dashboard or rule starts using one.
// The enabled kinds still emit ~40 families per container; Host & Containers and
// the alerts read these, one per USE question (usage, saturation as PSI wait time,
// errors as OOMs), with network and disk I/O traffic alongside. The rest is the
// fastest-growing block in the series budget, since it scales per container per
// host. Add a name here, and its kind to enabled_metrics above, when a dashboard or
// rule starts using one.
rule {
source_labels = ["__name__"]
regex = "container_(start_time_seconds|oom_events_total|cpu_usage_seconds_total|memory_working_set_bytes)"
regex = "container_(start_time_seconds|oom_events_total|cpu_usage_seconds_total|memory_working_set_bytes|spec_memory_limit_bytes|pressure_(cpu|memory|io)_waiting_seconds_total|network_(receive|transmit)_bytes_total|fs_(reads|writes)_bytes_total)"
action = "keep"
}
}

// netdev reads <procfs>/net/dev, and /host/proc/net links to self/net: this
// container's namespace, so the host collector above would report Alloy's own veth.
// PID 1's view is the host's. Bridges and veths are per-container noise.
prometheus.exporter.unix "host_net" {
procfs_path = "/host/proc/1"
set_collectors = ["netdev"]

netdev {
device_exclude = "^(lo|veth.+|br-.+|docker[0-9]+)$"
}
}

prometheus.scrape "host_net" {
targets = prometheus.exporter.unix.host_net.targets
forward_to = [prometheus.relabel.host_net.receiver]
scrape_interval = "30s"
job_name = "host_net"
}

// Both unix exporters label their targets job="integrations/unix" and the same
// instance, whatever job_name says, so their `up` series would collide.
prometheus.relabel "host_net" {
forward_to = [prometheus.relabel.host.receiver]

rule {
target_label = "job"
replacement = "integrations/unix_net"
}
}

prometheus.scrape "host" {
targets = prometheus.exporter.unix.host.targets
forward_to = [prometheus.relabel.host.receiver]
Expand All @@ -187,20 +225,17 @@ prometheus.relabel "host" {
rule {
target_label = "env"
replacement = sys.env("ENVIRONMENT")
action = "replace"
}

rule {
target_label = "project"
replacement = sys.env("PROJECT")
action = "replace"
}

// See the note on the log labels.
rule {
target_label = "host_name"
replacement = coalesce(string.trim_space(local.file.hostname.content), constants.hostname)
action = "replace"
}
}

Expand All @@ -216,40 +251,35 @@ prometheus.scrape "alloy" {
job_name = "alloy"
}

// Alloy is the only exporter in this file with histograms, and nothing central
// reads them: 345 of this job's 760 series. The _sum and _count series survive.
// Of this job's ~760 series the hub reads only the export failures; the queue gauges
// show how close an outage came to dropping data. `up` passes through relabelling
// here, unlike in Prometheus, so it is kept explicitly.
prometheus.relabel "alloy_self" {
forward_to = [prometheus.relabel.host.receiver]

rule {
source_labels = ["__name__"]
regex = ".*_bucket"
action = "drop"
regex = "up|otelcol_exporter_(send_failed|queue)_.+"
action = "keep"
}
}

// ---------------------------------------------------------------------------
// GPU metrics, present only on hosts that run compose.telemetry.gpu.yml. Without the
// exporter container this finds no targets and costs nothing.
discovery.relabel "gpu_exporter" {
targets = discovery.docker.containers.targets

rule {
source_labels = ["__meta_docker_container_label_com_docker_compose_project"]
regex = sys.env("COMPOSE_PROJECT_NAME")
action = "keep"
}
// Already scoped to this Compose project, with service_name set.
targets = discovery.relabel.containers.output

rule {
source_labels = ["__meta_docker_container_label_com_docker_compose_service"]
source_labels = ["service_name"]
regex = "nvidia-gpu-exporter"
action = "keep"
}

rule {
target_label = "service_name"
replacement = "gpu"
action = "replace"
}
}

Expand Down Expand Up @@ -281,7 +311,12 @@ otelcol.processor.memory_limiter "default" {
}
}

// The 200ms default sends a request per trickle of log lines, each through the
// tunnel and the hub's auth. 5s matches the hub's own batching and no alert is
// that tight.
otelcol.processor.batch "default" {
timeout = "5s"

output {
logs = [otelcol.exporter.otlphttp.central.input]
metrics = [otelcol.exporter.otlphttp.central.input]
Expand Down
27 changes: 10 additions & 17 deletions scripts/run_scheduled.sh
Original file line number Diff line number Diff line change
Expand Up @@ -46,11 +46,14 @@ ping_url="${!url_var:-}"
output_file="$(mktemp)"
trap 'rm -f "$output_file"' EXIT

# The job's output is the failure body, so the alert carries the reason. This puts job
# output in a third party's hands; keep credentials out of it.
ping_fail() {
curl -fsS -m 10 --retry 3 --data-binary "@${output_file}" "${ping_url}/fail" -o /dev/null \
|| echo "WARNING: failure ping to ${url_var} failed" >&2
# healthchecks.io reads <ping_url>/<exit status>: 0 is success, anything else a failure.
# A failure carries the job's output, so the alert has the reason. That puts job output
# in a third party's hands; keep credentials out of it.
ping_status() {
local body=()
[[ "$1" == 0 ]] || body=(--data-binary "@${output_file}")
curl -fsS -m 10 --retry 3 "${body[@]}" "${ping_url}/$1" -o /dev/null \
|| echo "WARNING: ping to ${url_var} failed" >&2
}

# A killed job must still report: systemd's TimeoutStartSec TERMs the whole cgroup, and
Expand All @@ -62,7 +65,7 @@ on_terminate() {
echo "run_scheduled: received SIG${sig}; job killed (likely a systemd timeout)" >>"$output_file"
cat "$output_file"
if [[ -n "$ping_url" ]]; then
ping_fail
ping_status 143
fi
rm -f "$output_file"
exit 143
Expand All @@ -75,15 +78,5 @@ status=0
wait $! || status=$?
cat "$output_file"

if [[ -z "$ping_url" ]]; then
exit "$status"
fi

if [[ "$status" -eq 0 ]]; then
curl -fsS -m 10 --retry 3 "$ping_url" -o /dev/null \
|| echo "WARNING: success ping to ${url_var} failed" >&2
else
ping_fail
fi

[[ -z "$ping_url" ]] || ping_status "$status"
exit "$status"
Loading