Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3ea485bfc7
|
@@ -53,10 +53,6 @@ SECRET_BASIC_AUTH_VERSION=v1
|
||||
# Not needed if you just want to read local log files — use SYSLOG_FILES instead.
|
||||
# SYSLOG=1
|
||||
# COMPOSE_FILE="$COMPOSE_FILE:compose.syslog.yml"
|
||||
#
|
||||
# Scrape Docker-Events via official docker-cli image using docker socket
|
||||
# COMPOSE_FILE="$COMPOSE_FILE:compose.docker-events.yml"
|
||||
|
||||
|
||||
# Monitoring Server
|
||||
#
|
||||
|
||||
@@ -54,14 +54,6 @@ This is what a gathering host pushes into. It also runs its own Alloy, so it mon
|
||||
|
||||
## Additional features
|
||||
|
||||
### Scrape docker events
|
||||
|
||||
Just uncomment the following line in your .env to collect docker events using
|
||||
the official docker-cli image and the docker socket:
|
||||
```
|
||||
COMPOSE_FILE="$COMPOSE_FILE:compose.docker-events.yml"
|
||||
```
|
||||
|
||||
### Discovering metrics from other apps
|
||||
|
||||
Alloy auto-discovers and scrapes other Docker Swarm services running on the same host, on the `proxy` network, that opt in via labels. No manual scrape config needed. On the app's `compose.yml`:
|
||||
|
||||
@@ -10,7 +10,7 @@ export PROMETHEUS_YML_VERSION=v2
|
||||
export MATRIX_ALERTMANAGER_CONFIG_VERSION=v1
|
||||
export MATRIX_ALERTMANAGER_ENTRYPOINT_VERSION=v1
|
||||
export GF_ALERTS_NODE_VERSION=v3
|
||||
export CONFIG_ALLOY_VERSION=v3
|
||||
export CONFIG_ALLOY_VERSION=v2
|
||||
|
||||
# migrates secrets from old names to new names by reading values from the
|
||||
# running containers on the server and re-inserting them under the new names.
|
||||
|
||||
@@ -1,20 +0,0 @@
|
||||
version: "3.8"
|
||||
services:
|
||||
docker-events:
|
||||
image: docker:29.8.1-cli
|
||||
command: >
|
||||
docker events
|
||||
--format '{{json .}}'
|
||||
--filter type=container
|
||||
--filter event=start
|
||||
--filter event=die
|
||||
--filter event=kill
|
||||
--filter event=oom
|
||||
--filter event=restart
|
||||
--filter event=stop
|
||||
--filter event=health_status
|
||||
volumes:
|
||||
- /var/run/docker.sock:/var/run/docker.sock:ro
|
||||
deploy:
|
||||
restart_policy:
|
||||
condition: on-failure
|
||||
+1
-1
@@ -53,7 +53,7 @@ services:
|
||||
condition: on-failure
|
||||
labels:
|
||||
- "backupbot.backup=${ENABLE_BACKUPS:-true}"
|
||||
- "coop-cloud.${STACK_NAME}.version=2.1.0+v1.18.1"
|
||||
- "coop-cloud.${STACK_NAME}.version=1.6.0+v1.8.1"
|
||||
configs:
|
||||
config_alloy:
|
||||
template_driver: golang
|
||||
|
||||
@@ -52,32 +52,7 @@ prometheus.scrape "default" {
|
||||
prometheus.exporter.cadvisor.docker.targets,
|
||||
)
|
||||
|
||||
forward_to = [prometheus.relabel.container_meta.receiver]
|
||||
}
|
||||
|
||||
prometheus.relabel "container_meta" {
|
||||
forward_to = [prometheus.remote_write.prometheus.receiver]
|
||||
|
||||
// remove sha tail: nginx:1.31.1@sha256:608a... -> nginx:1.31.1
|
||||
rule {
|
||||
source_labels = ["image"]
|
||||
regex = "([^@]+)@sha256:.*"
|
||||
target_label = "image"
|
||||
replacement = "$1"
|
||||
}
|
||||
// split image in name and tag
|
||||
rule {
|
||||
source_labels = ["image"]
|
||||
regex = "(.+):[^:/]+"
|
||||
target_label = "image_name"
|
||||
replacement = "$1"
|
||||
}
|
||||
rule {
|
||||
source_labels = ["image"]
|
||||
regex = ".+:([^:/]+)"
|
||||
target_label = "image_tag"
|
||||
replacement = "$1"
|
||||
}
|
||||
}
|
||||
|
||||
prometheus.remote_write "prometheus" {
|
||||
@@ -315,35 +290,7 @@ loki.source.docker "docker" {
|
||||
loki.source.journal "journal" {
|
||||
path = "/rootfs/var/log/journal"
|
||||
labels = { job = "{{ env "DOMAIN" }}" }
|
||||
relabel_rules = loki.relabel.journal.rules
|
||||
forward_to = [loki.process.journal.receiver]
|
||||
}
|
||||
|
||||
loki.relabel "journal" {
|
||||
forward_to = []
|
||||
|
||||
rule {
|
||||
source_labels = ["__journal__systemd_unit"]
|
||||
target_label = "unit"
|
||||
}
|
||||
rule {
|
||||
source_labels = ["__journal_syslog_identifier"]
|
||||
target_label = "ident"
|
||||
}
|
||||
rule {
|
||||
source_labels = ["__journal_priority_keyword"]
|
||||
target_label = "priority"
|
||||
}
|
||||
}
|
||||
|
||||
// drop network db stats (15-20% of kernel logs)
|
||||
loki.process "journal" {
|
||||
forward_to = [loki.write.loki.receiver]
|
||||
|
||||
stage.drop {
|
||||
expression = ".*NetworkDB stats.*"
|
||||
drop_counter_reason = "networkdb_stats"
|
||||
}
|
||||
}
|
||||
{{ end }}
|
||||
|
||||
|
||||
@@ -1,64 +0,0 @@
|
||||
BREAKING CHANGE
|
||||
Migration plan for upgrading from 1.6.0+v1.8.1.
|
||||
|
||||
## 1. Reinsert secrets with shortened names
|
||||
|
||||
Secret and config names were shortened to max 14 characters to prevent going over Docker's 64 character
|
||||
limit when STACK_NAME and VERSION are added to it.
|
||||
|
||||
- `abra app secret list <domain>` to see which secrets are missing under their new name
|
||||
- `abra app cmd --local <domain> migrate_secret_names` to reinsert all of them automatically
|
||||
(or manually: `abra app secret insert <domain> <secret_name> v1 <value>` per secret)
|
||||
|
||||
## 2. If you use OIDC (moved to seperate compose file)
|
||||
|
||||
- Add to your .env: `COMPOSE_FILE="$COMPOSE_FILE:compose.grafana-oidc.yml"`
|
||||
|
||||
## 3. If you use SMTP (moved to a seperate compose file)
|
||||
|
||||
- Add to your .env: `COMPOSE_FILE="$COMPOSE_FILE:compose.grafana-smtp.yml"`
|
||||
|
||||
## 4. node_exporter/cadvisor/promtail replaced by Grafana Alloy
|
||||
|
||||
Metrics collection changed from Prometheus scraping endpoints
|
||||
to Alloy pushing via `remote_write`/`loki push`.
|
||||
|
||||
- Remove `compose.promtail.yml`, `compose.expose-ports.yml` and
|
||||
`compose.basic-auth.yml` from your .env if present. They no
|
||||
longer exist. `compose.yml` now declares the `basic_auth` secret directly, so
|
||||
`SECRET_BASIC_AUTH_VERSION` is always required.
|
||||
- Add `PROMETHEUS_REMOTE_WRITE_URL=https://prometheus.$DOMAIN/api/v1/write`
|
||||
(`$DOMAIN` if this host also runs `compose.prometheus.yml`, otherwise a remote
|
||||
Prometheus' URL). Without this, Alloy collects no metrics at all.
|
||||
- Add `LOKI_PUSH_URL` (existing var, still used) and pick a log source:
|
||||
`JOURNALD=1` (systemd hosts), `SYSLOG_FILES=1` (non-systemd, tails
|
||||
`/var/log/*log`), or `SYSLOG=1` + `compose.syslog.yml` (network syslog listener).
|
||||
|
||||
- If this host had its own `node.$DOMAIN`/`cadvisor.$DOMAIN` scrape target
|
||||
configured on a central Prometheus, remove it. Those endpoints are gone.
|
||||
- `scrape-config.example.yml` and the `add_node`/`add_domain` abra.sh commands
|
||||
are gone. Replaced by label-based auto-discovery (see README).
|
||||
- `docker stack deploy` doesn't prune removed services, so old `cadvisor`/
|
||||
`promtail` containers keep running after a normal `abra app deploy`. Run
|
||||
`abra app undeploy <domain>` then `abra app deploy <domain>` to clear them out.
|
||||
- Diff your `.env` against the current `.env.sample`, to verify any other changes.
|
||||
|
||||
### New: label-based metrics auto-discovery
|
||||
|
||||
Alloy now auto-discovers and scrapes other Docker Swarm services on the same
|
||||
host/`proxy` network that opt in via `prometheus.io/scrape=true` deploy labels.
|
||||
See the README's "Auto-discovering metrics from other apps" section.
|
||||
|
||||
- If you scrape Traefik metrics: the old `metrics.traefik.$domain` pull-based
|
||||
endpoint still works if you keep the scrape config in Prometheus and
|
||||
existing dashboards keep showing its data, but it's recommended to get
|
||||
Traefik onto the new label-based discovery.
|
||||
|
||||
### Dashboards
|
||||
|
||||
The Swarm, Stacks and Traefik dashboards were reworked to show old (pull-model)
|
||||
and new (Alloy push-model) data as one continuous line, so you don't lose history
|
||||
across the migration.
|
||||
|
||||
The disk usage per container was disabled per default because of the CPU-expensive
|
||||
filesystem scan (30% cpu increase). Enable by uncommenting env CADVISOR_DISK_USAGE=1
|
||||
@@ -1 +0,0 @@
|
||||
alloy: split image label in "image_name" and "image_tag", relabel journald logs and drop NetworkDB stats
|
||||
+64
-1
@@ -1 +1,64 @@
|
||||
allow scraping of docker events via new compose.docker-events.yml
|
||||
BREAKING CHANGE
|
||||
Migration plan for upgrading from 1.6.0+v1.8.1.
|
||||
|
||||
## 1. Reinsert secrets with shortened names
|
||||
|
||||
Secret and config names were shortened to max 14 characters to prevent going over Docker's 64 character
|
||||
limit when STACK_NAME and VERSION are added to it.
|
||||
|
||||
- `abra app secret list <domain>` to see which secrets are missing under their new name
|
||||
- `abra app cmd --local <domain> migrate_secret_names` to reinsert all of them automatically
|
||||
(or manually: `abra app secret insert <domain> <secret_name> v1 <value>` per secret)
|
||||
|
||||
## 2. If you use OIDC (moved to seperate compose file)
|
||||
|
||||
- Add to your .env: `COMPOSE_FILE="$COMPOSE_FILE:compose.grafana-oidc.yml"`
|
||||
|
||||
## 3. If you use SMTP (moved to a seperate compose file)
|
||||
|
||||
- Add to your .env: `COMPOSE_FILE="$COMPOSE_FILE:compose.grafana-smtp.yml"`
|
||||
|
||||
## 4. node_exporter/cadvisor/promtail replaced by Grafana Alloy
|
||||
|
||||
Metrics collection changed from Prometheus scraping endpoints
|
||||
to Alloy pushing via `remote_write`/`loki push`.
|
||||
|
||||
- Remove `compose.promtail.yml`, `compose.expose-ports.yml` and
|
||||
`compose.basic-auth.yml` from your .env if present. They no
|
||||
longer exist. `compose.yml` now declares the `basic_auth` secret directly, so
|
||||
`SECRET_BASIC_AUTH_VERSION` is always required.
|
||||
- Add `PROMETHEUS_REMOTE_WRITE_URL=https://prometheus.$DOMAIN/api/v1/write`
|
||||
(`$DOMAIN` if this host also runs `compose.prometheus.yml`, otherwise a remote
|
||||
Prometheus' URL). Without this, Alloy collects no metrics at all.
|
||||
- Add `LOKI_PUSH_URL` (existing var, still used) and pick a log source:
|
||||
`JOURNALD=1` (systemd hosts), `SYSLOG_FILES=1` (non-systemd, tails
|
||||
`/var/log/*log`), or `SYSLOG=1` + `compose.syslog.yml` (network syslog listener).
|
||||
|
||||
- If this host had its own `node.$DOMAIN`/`cadvisor.$DOMAIN` scrape target
|
||||
configured on a central Prometheus, remove it. Those endpoints are gone.
|
||||
- `scrape-config.example.yml` and the `add_node`/`add_domain` abra.sh commands
|
||||
are gone. Replaced by label-based auto-discovery (see README).
|
||||
- `docker stack deploy` doesn't prune removed services, so old `cadvisor`/
|
||||
`promtail` containers keep running after a normal `abra app deploy`. Run
|
||||
`abra app undeploy <domain>` then `abra app deploy <domain>` to clear them out.
|
||||
- Diff your `.env` against the current `.env.sample`, to verify any other changes.
|
||||
|
||||
### New: label-based metrics auto-discovery
|
||||
|
||||
Alloy now auto-discovers and scrapes other Docker Swarm services on the same
|
||||
host/`proxy` network that opt in via `prometheus.io/scrape=true` deploy labels.
|
||||
See the README's "Auto-discovering metrics from other apps" section.
|
||||
|
||||
- If you scrape Traefik metrics: the old `metrics.traefik.$domain` pull-based
|
||||
endpoint still works if you keep the scrape config in Prometheus and
|
||||
existing dashboards keep showing its data, but it's recommended to get
|
||||
Traefik onto the new label-based discovery.
|
||||
|
||||
### Dashboards
|
||||
|
||||
The Swarm, Stacks and Traefik dashboards were reworked to show old (pull-model)
|
||||
and new (Alloy push-model) data as one continuous line, so you don't lose history
|
||||
across the migration.
|
||||
|
||||
The disk usage per container was disabled per default because of the CPU-expensive
|
||||
filesystem scan (30% cpu increase). Enable by uncommenting env CADVISOR_DISK_USAGE=1
|
||||
|
||||
Reference in New Issue
Block a user