Compare commits

..
Author SHA1 Message Date
simon 625e479c3d chore: publish 2.2.0+v1.18.1 release 2026-09-22 12:43:11 +02:00
simon 779dee5a0d allow scraping of docker events (#32)
another one! closes #15

again, the compose-snippet was generated using Claude, but verified and tested manually.
Thought of using a socket-proxy just as traefik does, but since alloy has socket access anyways and the official docker image should be as safe as it gets, I think this should be alright.

## Loki log preview
![grafik.png](/attachments/013605e9-9abc-47ef-8cb9-c118fbaec3b5)

Reviewed-on: #32
Reviewed-by: moritz <98+moritz@noreply.git.coopcloud.tech>
Co-authored-by: Simon <s.thiessen@local-it.org>
2026-09-22 10:33:47 +00:00
simon 0eab632ec4 chore: publish 2.1.0+v1.18.1 release 2026-09-17 16:05:27 +02:00
simon e5f57c97f1 Split image tags, relabel journald logs (#30)
Reviewed-on: #30
Reviewed-by: ammaratef45 <ammaratef45@proton.me>
Co-authored-by: Simon <s.thiessen@local-it.org>
2026-09-17 13:38:45 +00:00
simon a7014f72ae chore: publish 2.0.0+v1.18.1 release 2026-09-08 17:31:55 +02:00
simon f48aa64193 make disk usage per container metric optional (#29)
Following the discussion from the matrix channel, I made the disk usage metrics optional, as it currently causes ~35% of cpu usage on our system.

[Old cadvisor-config](02b01e5c23/compose.yml) from this repo had a `- "--housekeeping_interval=120s"` option enabled which seems currently not configurable for alloy - maybe this allows us reenabling disk metrics at a later point again.

dropped from 35% to 5% when removing `disk` from
`  enabled_metrics = ["cpu", "cpuLoad", "diskIO", "memory", "network"]` for cadvisor.

go profiling:

```
File: alloy
Build ID: de7ba8978f6753c25c13f6886ab3b896d3b05d3f
Type: cpu
Time: Sep 7, 2026 at 12:53pm (CEST)
Duration: 120s, Total samples = 37.23s (31.02%)
Showing nodes accounting for 32.20s, 86.49% of 37.23s total
Dropped 790 nodes (cum <= 0.19s)
      flat  flat%   sum%        cum   cum%
    28.72s 77.14% 77.14%     28.72s 77.14%  internal/runtime/syscall/linux.Syscall6
     0.47s  1.26% 78.40%      0.48s  1.29%  internal/filepathlite.(*lazybuf).append (inline)
     0.30s  0.81% 79.21%      0.94s  2.52%  internal/filepathlite.Clean
     0.26s   0.7% 79.91%      0.26s   0.7%  runtime.nextFreeFast (inline)
     0.23s  0.62% 80.53%      0.23s  0.62%  runtime.futex
     0.21s  0.56% 81.09%      0.21s  0.56%  runtime.memclrNoHeapPointers
     0.21s  0.56% 81.65%      0.21s  0.56%  runtime.memmove
     0.19s  0.51% 82.16%      6.19s 16.63%  os.(*File).readdir
     0.17s  0.46% 82.62%     31.24s 83.91%  path/filepath.walk
     0.12s  0.32% 82.94%      0.20s  0.54%  runtime.exitsyscall
     0.08s  0.21% 83.16%      0.24s  0.64%  runtime.scanObject
     0.08s  0.21% 83.37%      0.24s  0.64%  runtime.sweepone
     0.08s  0.21% 83.59%      0.29s  0.78%  slices.pdqsortOrdered[go.shape.string]
     0.07s  0.19% 83.78%      0.27s  0.73%  runtime.makeslicecopy
     0.06s  0.16% 83.94%     28.48s 76.50%  syscall.RawSyscall6
     0.05s  0.13% 84.07%      0.27s  0.73%  github.com/google/cadvisor/fs.GetDirUsage.func1
     0.05s  0.13% 84.21%     19.99s 53.69%  os.lstatNolog
     0.05s  0.13% 84.34%      1.26s  3.38%  runtime.mallocgc
     0.05s  0.13% 84.47%      0.43s  1.15%  runtime.mallocgcSmallNoscan
     0.04s  0.11% 84.58%      0.46s  1.24%  runtime.newobject
     0.04s  0.11% 84.69%      0.19s  0.51%  runtime.selectgo
     0.04s  0.11% 84.80%      1.18s  3.17%  runtime.systemstack
```

Reviewed-on: #29
Reviewed-by: Danny Groenewegen <247+dannygroenewegen@noreply.git.coopcloud.tech>
Co-authored-by: Simon <s.thiessen@local-it.org>
2026-09-08 14:25:47 +00:00
9 changed files with 89 additions and 2 deletions
+4
View File
@@ -53,6 +53,10 @@ SECRET_BASIC_AUTH_VERSION=v1
# Not needed if you just want to read local log files — use SYSLOG_FILES instead.
# SYSLOG=1
# COMPOSE_FILE="$COMPOSE_FILE:compose.syslog.yml"
#
# Scrape Docker-Events via official docker-cli image using docker socket
# COMPOSE_FILE="$COMPOSE_FILE:compose.docker-events.yml"
# Monitoring Server
#
+8
View File
@@ -54,6 +54,14 @@ This is what a gathering host pushes into. It also runs its own Alloy, so it mon
## Additional features
### Scrape docker events
Just uncomment the following line in your .env to collect docker events using
the official docker-cli image and the docker socket:
```
COMPOSE_FILE="$COMPOSE_FILE:compose.docker-events.yml"
```
### Discovering metrics from other apps
Alloy auto-discovers and scrapes other Docker Swarm services running on the same host, on the `proxy` network, that opt in via labels. No manual scrape config needed. On the app's `compose.yml`:
+1 -1
View File
@@ -10,7 +10,7 @@ export PROMETHEUS_YML_VERSION=v2
export MATRIX_ALERTMANAGER_CONFIG_VERSION=v1
export MATRIX_ALERTMANAGER_ENTRYPOINT_VERSION=v1
export GF_ALERTS_NODE_VERSION=v3
export CONFIG_ALLOY_VERSION=v2
export CONFIG_ALLOY_VERSION=v3
# migrates secrets from old names to new names by reading values from the
# running containers on the server and re-inserting them under the new names.
+20
View File
@@ -0,0 +1,20 @@
version: "3.8"
services:
docker-events:
image: docker:29.8.1-cli
command: >
docker events
--format '{{json .}}'
--filter type=container
--filter event=start
--filter event=die
--filter event=kill
--filter event=oom
--filter event=restart
--filter event=stop
--filter event=health_status
volumes:
- /var/run/docker.sock:/var/run/docker.sock:ro
deploy:
restart_policy:
condition: on-failure
+1 -1
View File
@@ -53,7 +53,7 @@ services:
condition: on-failure
labels:
- "backupbot.backup=${ENABLE_BACKUPS:-true}"
- "coop-cloud.${STACK_NAME}.version=1.6.0+v1.8.1"
- "coop-cloud.${STACK_NAME}.version=2.2.0+v1.18.1"
configs:
config_alloy:
template_driver: golang
+53
View File
@@ -52,7 +52,32 @@ prometheus.scrape "default" {
prometheus.exporter.cadvisor.docker.targets,
)
forward_to = [prometheus.relabel.container_meta.receiver]
}
prometheus.relabel "container_meta" {
forward_to = [prometheus.remote_write.prometheus.receiver]
// remove sha tail: nginx:1.31.1@sha256:608a... -> nginx:1.31.1
rule {
source_labels = ["image"]
regex = "([^@]+)@sha256:.*"
target_label = "image"
replacement = "$1"
}
// split image in name and tag
rule {
source_labels = ["image"]
regex = "(.+):[^:/]+"
target_label = "image_name"
replacement = "$1"
}
rule {
source_labels = ["image"]
regex = ".+:([^:/]+)"
target_label = "image_tag"
replacement = "$1"
}
}
prometheus.remote_write "prometheus" {
@@ -290,7 +315,35 @@ loki.source.docker "docker" {
loki.source.journal "journal" {
path = "/rootfs/var/log/journal"
labels = { job = "{{ env "DOMAIN" }}" }
relabel_rules = loki.relabel.journal.rules
forward_to = [loki.process.journal.receiver]
}
loki.relabel "journal" {
forward_to = []
rule {
source_labels = ["__journal__systemd_unit"]
target_label = "unit"
}
rule {
source_labels = ["__journal_syslog_identifier"]
target_label = "ident"
}
rule {
source_labels = ["__journal_priority_keyword"]
target_label = "priority"
}
}
// drop network db stats (15-20% of kernel logs)
loki.process "journal" {
forward_to = [loki.write.loki.receiver]
stage.drop {
expression = ".*NetworkDB stats.*"
drop_counter_reason = "networkdb_stats"
}
}
{{ end }}
+1
View File
@@ -0,0 +1 @@
alloy: split image label in "image_name" and "image_tag", relabel journald logs and drop NetworkDB stats
+1
View File
@@ -0,0 +1 @@
allow scraping of docker events via new compose.docker-events.yml