Compare commits

..
Author SHA1 Message Date
simon 3ea485bfc7 make disk usage per container metric optional 2026-09-08 10:30:59 +02:00
4 changed files with 12 additions and 27 deletions
+4 -1
View File
@@ -20,6 +20,9 @@ SECRET_BASIC_AUTH_VERSION=v1
# server is remote
# PROMETHEUS_REMOTE_WRITE_URL=https://prometheus.$DOMAIN/api/v1/write
# Enable container filesystem usage metrics (expensive du-scan, ~30% CPU on large hosts)
# CADVISOR_DISK_USAGE=1
# Enable authenticated scraping of containers that opt in via
# prometheus.io/auth=basic or prometheus.io/auth=bearer labels (used as
# password/bearer token respectively). Insert it with:
@@ -84,7 +87,7 @@ SECRET_BASIC_AUTH_VERSION=v1
# SECRET_GF_ADMINPASSWD_VERSION=v1
## Grafana's own domain. Defaults to $DOMAIN
## Change the value if you want Grafana on another domain.
GRAFANA_DOMAIN=$DOMAIN
# GRAFANA_DOMAIN=$DOMAIN
#
## Single-Sign-On with OIDC
# COMPOSE_FILE="$COMPOSE_FILE:compose.grafana-oidc.yml"
+1 -1
View File
@@ -10,7 +10,7 @@ export PROMETHEUS_YML_VERSION=v2
export MATRIX_ALERTMANAGER_CONFIG_VERSION=v1
export MATRIX_ALERTMANAGER_ENTRYPOINT_VERSION=v1
export GF_ALERTS_NODE_VERSION=v3
export CONFIG_ALLOY_VERSION=v1
export CONFIG_ALLOY_VERSION=v2
# migrates secrets from old names to new names by reading values from the
# running containers on the server and re-inserting them under the new names.
+4 -25
View File
@@ -14,7 +14,11 @@ discovery.docker "linux" {
{{ if ne (env "PROMETHEUS_REMOTE_WRITE_URL") "" }}
prometheus.exporter.cadvisor "docker" {
docker_only = true
{{ if eq (env "CADVISOR_DISK_USAGE") "1" }}
enabled_metrics = ["cpu", "cpuLoad", "disk", "diskIO", "memory", "network"]
{{ else }}
enabled_metrics = ["cpu", "cpuLoad", "diskIO", "memory", "network"]
{{ end }}
// host-wide totals already come from prometheus.exporter.unix
disable_root_cgroup_stats = true
}
@@ -48,32 +52,7 @@ prometheus.scrape "default" {
prometheus.exporter.cadvisor.docker.targets,
)
forward_to = [prometheus.relabel.container_meta.receiver]
}
prometheus.relabel "container_meta" {
forward_to = [prometheus.remote_write.prometheus.receiver]
// remove sha tail: nginx:1.31.1@sha256:608a... -> nginx:1.31.1
rule {
source_labels = ["image"]
regex = "([^@]+)@sha256:.*"
target_label = "image"
replacement = "$1"
}
// split image in name and tag
rule {
source_labels = ["image"]
regex = "(.+):[^:/]+"
target_label = "image_name"
replacement = "$1"
}
rule {
source_labels = ["image"]
regex = ".+:([^:/]+)"
target_label = "image_tag"
replacement = "$1"
}
}
prometheus.remote_write "prometheus" {
+3
View File
@@ -59,3 +59,6 @@ See the README's "Auto-discovering metrics from other apps" section.
The Swarm, Stacks and Traefik dashboards were reworked to show old (pull-model)
and new (Alloy push-model) data as one continuous line, so you don't lose history
across the migration.
The disk usage per container was disabled per default because of the CPU-expensive
filesystem scan (30% cpu increase). Enable by uncommenting env CADVISOR_DISK_USAGE=1