Compare commits
6
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
625e479c3d | ||
|
|
779dee5a0d | ||
|
|
0eab632ec4 | ||
|
|
e5f57c97f1 | ||
|
|
a7014f72ae | ||
|
|
f48aa64193 |
+8
-1
@@ -20,6 +20,9 @@ SECRET_BASIC_AUTH_VERSION=v1
|
||||
# server is remote
|
||||
# PROMETHEUS_REMOTE_WRITE_URL=https://prometheus.$DOMAIN/api/v1/write
|
||||
|
||||
# Enable container filesystem usage metrics (expensive du-scan, ~30% CPU on large hosts)
|
||||
# CADVISOR_DISK_USAGE=1
|
||||
|
||||
# Enable authenticated scraping of containers that opt in via
|
||||
# prometheus.io/auth=basic or prometheus.io/auth=bearer labels (used as
|
||||
# password/bearer token respectively). Insert it with:
|
||||
@@ -50,6 +53,10 @@ SECRET_BASIC_AUTH_VERSION=v1
|
||||
# Not needed if you just want to read local log files — use SYSLOG_FILES instead.
|
||||
# SYSLOG=1
|
||||
# COMPOSE_FILE="$COMPOSE_FILE:compose.syslog.yml"
|
||||
#
|
||||
# Scrape Docker-Events via official docker-cli image using docker socket
|
||||
# COMPOSE_FILE="$COMPOSE_FILE:compose.docker-events.yml"
|
||||
|
||||
|
||||
# Monitoring Server
|
||||
#
|
||||
@@ -84,7 +91,7 @@ SECRET_BASIC_AUTH_VERSION=v1
|
||||
# SECRET_GF_ADMINPASSWD_VERSION=v1
|
||||
## Grafana's own domain. Defaults to $DOMAIN
|
||||
## Change the value if you want Grafana on another domain.
|
||||
GRAFANA_DOMAIN=$DOMAIN
|
||||
# GRAFANA_DOMAIN=$DOMAIN
|
||||
#
|
||||
## Single-Sign-On with OIDC
|
||||
# COMPOSE_FILE="$COMPOSE_FILE:compose.grafana-oidc.yml"
|
||||
|
||||
@@ -54,6 +54,14 @@ This is what a gathering host pushes into. It also runs its own Alloy, so it mon
|
||||
|
||||
## Additional features
|
||||
|
||||
### Scrape docker events
|
||||
|
||||
Just uncomment the following line in your .env to collect docker events using
|
||||
the official docker-cli image and the docker socket:
|
||||
```
|
||||
COMPOSE_FILE="$COMPOSE_FILE:compose.docker-events.yml"
|
||||
```
|
||||
|
||||
### Discovering metrics from other apps
|
||||
|
||||
Alloy auto-discovers and scrapes other Docker Swarm services running on the same host, on the `proxy` network, that opt in via labels. No manual scrape config needed. On the app's `compose.yml`:
|
||||
|
||||
@@ -10,7 +10,7 @@ export PROMETHEUS_YML_VERSION=v2
|
||||
export MATRIX_ALERTMANAGER_CONFIG_VERSION=v1
|
||||
export MATRIX_ALERTMANAGER_ENTRYPOINT_VERSION=v1
|
||||
export GF_ALERTS_NODE_VERSION=v3
|
||||
export CONFIG_ALLOY_VERSION=v1
|
||||
export CONFIG_ALLOY_VERSION=v3
|
||||
|
||||
# migrates secrets from old names to new names by reading values from the
|
||||
# running containers on the server and re-inserting them under the new names.
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
version: "3.8"
|
||||
services:
|
||||
docker-events:
|
||||
image: docker:29.8.1-cli
|
||||
command: >
|
||||
docker events
|
||||
--format '{{json .}}'
|
||||
--filter type=container
|
||||
--filter event=start
|
||||
--filter event=die
|
||||
--filter event=kill
|
||||
--filter event=oom
|
||||
--filter event=restart
|
||||
--filter event=stop
|
||||
--filter event=health_status
|
||||
volumes:
|
||||
- /var/run/docker.sock:/var/run/docker.sock:ro
|
||||
deploy:
|
||||
restart_policy:
|
||||
condition: on-failure
|
||||
+1
-1
@@ -53,7 +53,7 @@ services:
|
||||
condition: on-failure
|
||||
labels:
|
||||
- "backupbot.backup=${ENABLE_BACKUPS:-true}"
|
||||
- "coop-cloud.${STACK_NAME}.version=1.6.0+v1.8.1"
|
||||
- "coop-cloud.${STACK_NAME}.version=2.2.0+v1.18.1"
|
||||
configs:
|
||||
config_alloy:
|
||||
template_driver: golang
|
||||
|
||||
@@ -14,7 +14,11 @@ discovery.docker "linux" {
|
||||
{{ if ne (env "PROMETHEUS_REMOTE_WRITE_URL") "" }}
|
||||
prometheus.exporter.cadvisor "docker" {
|
||||
docker_only = true
|
||||
{{ if eq (env "CADVISOR_DISK_USAGE") "1" }}
|
||||
enabled_metrics = ["cpu", "cpuLoad", "disk", "diskIO", "memory", "network"]
|
||||
{{ else }}
|
||||
enabled_metrics = ["cpu", "cpuLoad", "diskIO", "memory", "network"]
|
||||
{{ end }}
|
||||
// host-wide totals already come from prometheus.exporter.unix
|
||||
disable_root_cgroup_stats = true
|
||||
}
|
||||
@@ -48,7 +52,32 @@ prometheus.scrape "default" {
|
||||
prometheus.exporter.cadvisor.docker.targets,
|
||||
)
|
||||
|
||||
forward_to = [prometheus.relabel.container_meta.receiver]
|
||||
}
|
||||
|
||||
prometheus.relabel "container_meta" {
|
||||
forward_to = [prometheus.remote_write.prometheus.receiver]
|
||||
|
||||
// remove sha tail: nginx:1.31.1@sha256:608a... -> nginx:1.31.1
|
||||
rule {
|
||||
source_labels = ["image"]
|
||||
regex = "([^@]+)@sha256:.*"
|
||||
target_label = "image"
|
||||
replacement = "$1"
|
||||
}
|
||||
// split image in name and tag
|
||||
rule {
|
||||
source_labels = ["image"]
|
||||
regex = "(.+):[^:/]+"
|
||||
target_label = "image_name"
|
||||
replacement = "$1"
|
||||
}
|
||||
rule {
|
||||
source_labels = ["image"]
|
||||
regex = ".+:([^:/]+)"
|
||||
target_label = "image_tag"
|
||||
replacement = "$1"
|
||||
}
|
||||
}
|
||||
|
||||
prometheus.remote_write "prometheus" {
|
||||
@@ -286,7 +315,35 @@ loki.source.docker "docker" {
|
||||
loki.source.journal "journal" {
|
||||
path = "/rootfs/var/log/journal"
|
||||
labels = { job = "{{ env "DOMAIN" }}" }
|
||||
relabel_rules = loki.relabel.journal.rules
|
||||
forward_to = [loki.process.journal.receiver]
|
||||
}
|
||||
|
||||
loki.relabel "journal" {
|
||||
forward_to = []
|
||||
|
||||
rule {
|
||||
source_labels = ["__journal__systemd_unit"]
|
||||
target_label = "unit"
|
||||
}
|
||||
rule {
|
||||
source_labels = ["__journal_syslog_identifier"]
|
||||
target_label = "ident"
|
||||
}
|
||||
rule {
|
||||
source_labels = ["__journal_priority_keyword"]
|
||||
target_label = "priority"
|
||||
}
|
||||
}
|
||||
|
||||
// drop network db stats (15-20% of kernel logs)
|
||||
loki.process "journal" {
|
||||
forward_to = [loki.write.loki.receiver]
|
||||
|
||||
stage.drop {
|
||||
expression = ".*NetworkDB stats.*"
|
||||
drop_counter_reason = "networkdb_stats"
|
||||
}
|
||||
}
|
||||
{{ end }}
|
||||
|
||||
|
||||
@@ -59,3 +59,6 @@ See the README's "Auto-discovering metrics from other apps" section.
|
||||
The Swarm, Stacks and Traefik dashboards were reworked to show old (pull-model)
|
||||
and new (Alloy push-model) data as one continuous line, so you don't lose history
|
||||
across the migration.
|
||||
|
||||
The disk usage per container was disabled per default because of the CPU-expensive
|
||||
filesystem scan (30% cpu increase). Enable by uncommenting env CADVISOR_DISK_USAGE=1
|
||||
@@ -0,0 +1 @@
|
||||
alloy: split image label in "image_name" and "image_tag", relabel journald logs and drop NetworkDB stats
|
||||
@@ -0,0 +1 @@
|
||||
allow scraping of docker events via new compose.docker-events.yml
|
||||
Reference in New Issue
Block a user