Compare commits

..
Author SHA1 Message Date
simon 3ea485bfc7 make disk usage per container metric optional 2026-09-08 10:30:59 +02:00
9 changed files with 66 additions and 153 deletions
-4
View File
@@ -53,10 +53,6 @@ SECRET_BASIC_AUTH_VERSION=v1
# Not needed if you just want to read local log files — use SYSLOG_FILES instead.
# SYSLOG=1
# COMPOSE_FILE="$COMPOSE_FILE:compose.syslog.yml"
#
# Scrape Docker-Events via official docker-cli image using docker socket
# COMPOSE_FILE="$COMPOSE_FILE:compose.docker-events.yml"
# Monitoring Server
#
-8
View File
@@ -54,14 +54,6 @@ This is what a gathering host pushes into. It also runs its own Alloy, so it mon
## Additional features
### Scrape docker events
Just uncomment the following line in your .env to collect docker events using
the official docker-cli image and the docker socket:
```
COMPOSE_FILE="$COMPOSE_FILE:compose.docker-events.yml"
```
### Discovering metrics from other apps
Alloy auto-discovers and scrapes other Docker Swarm services running on the same host, on the `proxy` network, that opt in via labels. No manual scrape config needed. On the app's `compose.yml`:
+1 -1
View File
@@ -10,7 +10,7 @@ export PROMETHEUS_YML_VERSION=v2
export MATRIX_ALERTMANAGER_CONFIG_VERSION=v1
export MATRIX_ALERTMANAGER_ENTRYPOINT_VERSION=v1
export GF_ALERTS_NODE_VERSION=v3
export CONFIG_ALLOY_VERSION=v3
export CONFIG_ALLOY_VERSION=v2
# migrates secrets from old names to new names by reading values from the
# running containers on the server and re-inserting them under the new names.
-20
View File
@@ -1,20 +0,0 @@
version: "3.8"
services:
docker-events:
image: docker:29.8.1-cli
command: >
docker events
--format '{{json .}}'
--filter type=container
--filter event=start
--filter event=die
--filter event=kill
--filter event=oom
--filter event=restart
--filter event=stop
--filter event=health_status
volumes:
- /var/run/docker.sock:/var/run/docker.sock:ro
deploy:
restart_policy:
condition: on-failure
+1 -1
View File
@@ -53,7 +53,7 @@ services:
condition: on-failure
labels:
- "backupbot.backup=${ENABLE_BACKUPS:-true}"
- "coop-cloud.${STACK_NAME}.version=2.1.0+v1.18.1"
- "coop-cloud.${STACK_NAME}.version=1.6.0+v1.8.1"
configs:
config_alloy:
template_driver: golang
-53
View File
@@ -52,32 +52,7 @@ prometheus.scrape "default" {
prometheus.exporter.cadvisor.docker.targets,
)
forward_to = [prometheus.relabel.container_meta.receiver]
}
prometheus.relabel "container_meta" {
forward_to = [prometheus.remote_write.prometheus.receiver]
// remove sha tail: nginx:1.31.1@sha256:608a... -> nginx:1.31.1
rule {
source_labels = ["image"]
regex = "([^@]+)@sha256:.*"
target_label = "image"
replacement = "$1"
}
// split image in name and tag
rule {
source_labels = ["image"]
regex = "(.+):[^:/]+"
target_label = "image_name"
replacement = "$1"
}
rule {
source_labels = ["image"]
regex = ".+:([^:/]+)"
target_label = "image_tag"
replacement = "$1"
}
}
prometheus.remote_write "prometheus" {
@@ -315,35 +290,7 @@ loki.source.docker "docker" {
loki.source.journal "journal" {
path = "/rootfs/var/log/journal"
labels = { job = "{{ env "DOMAIN" }}" }
relabel_rules = loki.relabel.journal.rules
forward_to = [loki.process.journal.receiver]
}
loki.relabel "journal" {
forward_to = []
rule {
source_labels = ["__journal__systemd_unit"]
target_label = "unit"
}
rule {
source_labels = ["__journal_syslog_identifier"]
target_label = "ident"
}
rule {
source_labels = ["__journal_priority_keyword"]
target_label = "priority"
}
}
// drop network db stats (15-20% of kernel logs)
loki.process "journal" {
forward_to = [loki.write.loki.receiver]
stage.drop {
expression = ".*NetworkDB stats.*"
drop_counter_reason = "networkdb_stats"
}
}
{{ end }}
-64
View File
@@ -1,64 +0,0 @@
BREAKING CHANGE
Migration plan for upgrading from 1.6.0+v1.8.1.
## 1. Reinsert secrets with shortened names
Secret and config names were shortened to max 14 characters to prevent going over Docker's 64 character
limit when STACK_NAME and VERSION are added to it.
- `abra app secret list <domain>` to see which secrets are missing under their new name
- `abra app cmd --local <domain> migrate_secret_names` to reinsert all of them automatically
(or manually: `abra app secret insert <domain> <secret_name> v1 <value>` per secret)
## 2. If you use OIDC (moved to seperate compose file)
- Add to your .env: `COMPOSE_FILE="$COMPOSE_FILE:compose.grafana-oidc.yml"`
## 3. If you use SMTP (moved to a seperate compose file)
- Add to your .env: `COMPOSE_FILE="$COMPOSE_FILE:compose.grafana-smtp.yml"`
## 4. node_exporter/cadvisor/promtail replaced by Grafana Alloy
Metrics collection changed from Prometheus scraping endpoints
to Alloy pushing via `remote_write`/`loki push`.
- Remove `compose.promtail.yml`, `compose.expose-ports.yml` and
`compose.basic-auth.yml` from your .env if present. They no
longer exist. `compose.yml` now declares the `basic_auth` secret directly, so
`SECRET_BASIC_AUTH_VERSION` is always required.
- Add `PROMETHEUS_REMOTE_WRITE_URL=https://prometheus.$DOMAIN/api/v1/write`
(`$DOMAIN` if this host also runs `compose.prometheus.yml`, otherwise a remote
Prometheus' URL). Without this, Alloy collects no metrics at all.
- Add `LOKI_PUSH_URL` (existing var, still used) and pick a log source:
`JOURNALD=1` (systemd hosts), `SYSLOG_FILES=1` (non-systemd, tails
`/var/log/*log`), or `SYSLOG=1` + `compose.syslog.yml` (network syslog listener).
- If this host had its own `node.$DOMAIN`/`cadvisor.$DOMAIN` scrape target
configured on a central Prometheus, remove it. Those endpoints are gone.
- `scrape-config.example.yml` and the `add_node`/`add_domain` abra.sh commands
are gone. Replaced by label-based auto-discovery (see README).
- `docker stack deploy` doesn't prune removed services, so old `cadvisor`/
`promtail` containers keep running after a normal `abra app deploy`. Run
`abra app undeploy <domain>` then `abra app deploy <domain>` to clear them out.
- Diff your `.env` against the current `.env.sample`, to verify any other changes.
### New: label-based metrics auto-discovery
Alloy now auto-discovers and scrapes other Docker Swarm services on the same
host/`proxy` network that opt in via `prometheus.io/scrape=true` deploy labels.
See the README's "Auto-discovering metrics from other apps" section.
- If you scrape Traefik metrics: the old `metrics.traefik.$domain` pull-based
endpoint still works if you keep the scrape config in Prometheus and
existing dashboards keep showing its data, but it's recommended to get
Traefik onto the new label-based discovery.
### Dashboards
The Swarm, Stacks and Traefik dashboards were reworked to show old (pull-model)
and new (Alloy push-model) data as one continuous line, so you don't lose history
across the migration.
The disk usage per container was disabled per default because of the CPU-expensive
filesystem scan (30% cpu increase). Enable by uncommenting env CADVISOR_DISK_USAGE=1
-1
View File
@@ -1 +0,0 @@
alloy: split image label in "image_name" and "image_tag", relabel journald logs and drop NetworkDB stats
+64 -1
View File
@@ -1 +1,64 @@
allow scraping of docker events via new compose.docker-events.yml
BREAKING CHANGE
Migration plan for upgrading from 1.6.0+v1.8.1.
## 1. Reinsert secrets with shortened names
Secret and config names were shortened to max 14 characters to prevent going over Docker's 64 character
limit when STACK_NAME and VERSION are added to it.
- `abra app secret list <domain>` to see which secrets are missing under their new name
- `abra app cmd --local <domain> migrate_secret_names` to reinsert all of them automatically
(or manually: `abra app secret insert <domain> <secret_name> v1 <value>` per secret)
## 2. If you use OIDC (moved to seperate compose file)
- Add to your .env: `COMPOSE_FILE="$COMPOSE_FILE:compose.grafana-oidc.yml"`
## 3. If you use SMTP (moved to a seperate compose file)
- Add to your .env: `COMPOSE_FILE="$COMPOSE_FILE:compose.grafana-smtp.yml"`
## 4. node_exporter/cadvisor/promtail replaced by Grafana Alloy
Metrics collection changed from Prometheus scraping endpoints
to Alloy pushing via `remote_write`/`loki push`.
- Remove `compose.promtail.yml`, `compose.expose-ports.yml` and
`compose.basic-auth.yml` from your .env if present. They no
longer exist. `compose.yml` now declares the `basic_auth` secret directly, so
`SECRET_BASIC_AUTH_VERSION` is always required.
- Add `PROMETHEUS_REMOTE_WRITE_URL=https://prometheus.$DOMAIN/api/v1/write`
(`$DOMAIN` if this host also runs `compose.prometheus.yml`, otherwise a remote
Prometheus' URL). Without this, Alloy collects no metrics at all.
- Add `LOKI_PUSH_URL` (existing var, still used) and pick a log source:
`JOURNALD=1` (systemd hosts), `SYSLOG_FILES=1` (non-systemd, tails
`/var/log/*log`), or `SYSLOG=1` + `compose.syslog.yml` (network syslog listener).
- If this host had its own `node.$DOMAIN`/`cadvisor.$DOMAIN` scrape target
configured on a central Prometheus, remove it. Those endpoints are gone.
- `scrape-config.example.yml` and the `add_node`/`add_domain` abra.sh commands
are gone. Replaced by label-based auto-discovery (see README).
- `docker stack deploy` doesn't prune removed services, so old `cadvisor`/
`promtail` containers keep running after a normal `abra app deploy`. Run
`abra app undeploy <domain>` then `abra app deploy <domain>` to clear them out.
- Diff your `.env` against the current `.env.sample`, to verify any other changes.
### New: label-based metrics auto-discovery
Alloy now auto-discovers and scrapes other Docker Swarm services on the same
host/`proxy` network that opt in via `prometheus.io/scrape=true` deploy labels.
See the README's "Auto-discovering metrics from other apps" section.
- If you scrape Traefik metrics: the old `metrics.traefik.$domain` pull-based
endpoint still works if you keep the scrape config in Prometheus and
existing dashboards keep showing its data, but it's recommended to get
Traefik onto the new label-based discovery.
### Dashboards
The Swarm, Stacks and Traefik dashboards were reworked to show old (pull-model)
and new (Alloy push-model) data as one continuous line, so you don't lose history
across the migration.
The disk usage per container was disabled per default because of the CPU-expensive
filesystem scan (30% cpu increase). Enable by uncommenting env CADVISOR_DISK_USAGE=1