Compare commits

..
4 Commits
Author SHA1 Message Date
fauno f9820da529 feat: access log coop-cloud/traefik#126 2026-09-11 06:54:50 -03:00
simon a7014f72ae chore: publish 2.0.0+v1.18.1 release 2026-09-08 17:31:55 +02:00
simon f48aa64193 make disk usage per container metric optional (#29)
Following the discussion from the matrix channel, I made the disk usage metrics optional, as it currently causes ~35% of cpu usage on our system.

[Old cadvisor-config](02b01e5c23/compose.yml) from this repo had a `- "--housekeeping_interval=120s"` option enabled which seems currently not configurable for alloy - maybe this allows us reenabling disk metrics at a later point again.

dropped from 35% to 5% when removing `disk` from
`  enabled_metrics = ["cpu", "cpuLoad", "diskIO", "memory", "network"]` for cadvisor.

go profiling:

```
File: alloy
Build ID: de7ba8978f6753c25c13f6886ab3b896d3b05d3f
Type: cpu
Time: Sep 7, 2026 at 12:53pm (CEST)
Duration: 120s, Total samples = 37.23s (31.02%)
Showing nodes accounting for 32.20s, 86.49% of 37.23s total
Dropped 790 nodes (cum <= 0.19s)
      flat  flat%   sum%        cum   cum%
    28.72s 77.14% 77.14%     28.72s 77.14%  internal/runtime/syscall/linux.Syscall6
     0.47s  1.26% 78.40%      0.48s  1.29%  internal/filepathlite.(*lazybuf).append (inline)
     0.30s  0.81% 79.21%      0.94s  2.52%  internal/filepathlite.Clean
     0.26s   0.7% 79.91%      0.26s   0.7%  runtime.nextFreeFast (inline)
     0.23s  0.62% 80.53%      0.23s  0.62%  runtime.futex
     0.21s  0.56% 81.09%      0.21s  0.56%  runtime.memclrNoHeapPointers
     0.21s  0.56% 81.65%      0.21s  0.56%  runtime.memmove
     0.19s  0.51% 82.16%      6.19s 16.63%  os.(*File).readdir
     0.17s  0.46% 82.62%     31.24s 83.91%  path/filepath.walk
     0.12s  0.32% 82.94%      0.20s  0.54%  runtime.exitsyscall
     0.08s  0.21% 83.16%      0.24s  0.64%  runtime.scanObject
     0.08s  0.21% 83.37%      0.24s  0.64%  runtime.sweepone
     0.08s  0.21% 83.59%      0.29s  0.78%  slices.pdqsortOrdered[go.shape.string]
     0.07s  0.19% 83.78%      0.27s  0.73%  runtime.makeslicecopy
     0.06s  0.16% 83.94%     28.48s 76.50%  syscall.RawSyscall6
     0.05s  0.13% 84.07%      0.27s  0.73%  github.com/google/cadvisor/fs.GetDirUsage.func1
     0.05s  0.13% 84.21%     19.99s 53.69%  os.lstatNolog
     0.05s  0.13% 84.34%      1.26s  3.38%  runtime.mallocgc
     0.05s  0.13% 84.47%      0.43s  1.15%  runtime.mallocgcSmallNoscan
     0.04s  0.11% 84.58%      0.46s  1.24%  runtime.newobject
     0.04s  0.11% 84.69%      0.19s  0.51%  runtime.selectgo
     0.04s  0.11% 84.80%      1.18s  3.17%  runtime.systemstack
```

Reviewed-on: #29
Reviewed-by: Danny Groenewegen <247+dannygroenewegen@noreply.git.coopcloud.tech>
Co-authored-by: Simon <s.thiessen@local-it.org>
2026-09-08 14:25:47 +00:00
faunoanddannygroenewegen ff56438d7c BREAKING CHANGES: replace promtail and cadvisor for alloy (#21)
closes #20

---------

Co-authored-by: Danny Groenewegen <danny@ecommons.space>
Reviewed-on: #21
Reviewed-by: ammaratef45 <ammaratef45@proton.me>
Reviewed-by: Danny Groenewegen <247+dannygroenewegen@noreply.git.coopcloud.tech>
Co-authored-by: f <f@sutty.nl>
2026-09-02 22:57:39 +00:00
9 changed files with 23 additions and 3 deletions
+10 -1
View File
@@ -20,6 +20,9 @@ SECRET_BASIC_AUTH_VERSION=v1
# server is remote
# PROMETHEUS_REMOTE_WRITE_URL=https://prometheus.$DOMAIN/api/v1/write
# Enable container filesystem usage metrics (expensive du-scan, ~30% CPU on large hosts)
# CADVISOR_DISK_USAGE=1
# Enable authenticated scraping of containers that opt in via
# prometheus.io/auth=basic or prometheus.io/auth=bearer labels (used as
# password/bearer token respectively). Insert it with:
@@ -84,7 +87,7 @@ SECRET_BASIC_AUTH_VERSION=v1
# SECRET_GF_ADMINPASSWD_VERSION=v1
## Grafana's own domain. Defaults to $DOMAIN
## Change the value if you want Grafana on another domain.
GRAFANA_DOMAIN=$DOMAIN
# GRAFANA_DOMAIN=$DOMAIN
#
## Single-Sign-On with OIDC
# COMPOSE_FILE="$COMPOSE_FILE:compose.grafana-oidc.yml"
@@ -123,3 +126,9 @@ GRAFANA_DOMAIN=$DOMAIN
# Node memory usage alert will trigger when memory usage is above the given number in percent
#ALERT_NODE_MEMORY_USAGE=85
# Tell Traefik to keep access logs (could be very verbose)
ALLOY_ACCESS_LOGS=false
PROMETHEUS_ACCESS_LOGS=false
LOKI_ACCESS_LOGS=false
GRAFANA_ACCESS_LOGS=false
+1 -1
View File
@@ -10,7 +10,7 @@ export PROMETHEUS_YML_VERSION=v2
export MATRIX_ALERTMANAGER_CONFIG_VERSION=v1
export MATRIX_ALERTMANAGER_ENTRYPOINT_VERSION=v1
export GF_ALERTS_NODE_VERSION=v3
export CONFIG_ALLOY_VERSION=v1
export CONFIG_ALLOY_VERSION=v2
# migrates secrets from old names to new names by reading values from the
# running containers on the server and re-inserting them under the new names.
+1
View File
@@ -14,3 +14,4 @@ services:
- "traefik.http.routers.${STACK_NAME}-alloy.tls=true"
- "traefik.http.routers.${STACK_NAME}-alloy.tls.certresolver=${LETS_ENCRYPT_ENV}"
- "traefik.http.routers.${STACK_NAME}-alloy.middlewares=basicauth@file"
- "traefik.http.routers.${STACK_NAME}-alloy.observability.accesslogs=${ALLOY_ACCESS_LOGS:-false}"
+1
View File
@@ -43,6 +43,7 @@ services:
- "traefik.http.routers.${STACK_NAME}-grafana.entrypoints=web-secure"
- "traefik.http.routers.${STACK_NAME}-grafana.tls=true"
- "traefik.http.routers.${STACK_NAME}-grafana.tls.certresolver=${LETS_ENCRYPT_ENV}"
- "traefik.http.routers.${STACK_NAME}-grafana.observability.accesslogs=${GRAFANA_ACCESS_LOGS:-false}"
healthcheck:
test: "wget -q http://localhost:3000/healthz -O/dev/null"
interval: 5s
+1
View File
@@ -34,6 +34,7 @@ services:
- "traefik.http.routers.${STACK_NAME}-loki.tls=true"
- "traefik.http.routers.${STACK_NAME}-loki.tls.certresolver=${LETS_ENCRYPT_ENV}"
- "traefik.http.routers.${STACK_NAME}-loki.middlewares=basicauth@file"
- "traefik.http.routers.${STACK_NAME}-loki.observability.accesslogs=${LOKI_ACCESS_LOGS:-false}"
configs:
+1
View File
@@ -38,6 +38,7 @@ services:
- "traefik.http.routers.${STACK_NAME}-prometheus.tls=true"
- "traefik.http.routers.${STACK_NAME}-prometheus.tls.certresolver=${LETS_ENCRYPT_ENV}"
- "traefik.http.routers.${STACK_NAME}-prometheus.middlewares=basicauth@file"
- "traefik.http.routers.${STACK_NAME}-prometheus.observability.accesslogs=${PROMETHEUS_ACCESS_LOGS:-false}"
configs:
prometheus_yml:
+1 -1
View File
@@ -53,7 +53,7 @@ services:
condition: on-failure
labels:
- "backupbot.backup=${ENABLE_BACKUPS:-true}"
- "coop-cloud.${STACK_NAME}.version=1.6.0+v1.8.1"
- "coop-cloud.${STACK_NAME}.version=2.0.0+v1.18.1"
configs:
config_alloy:
template_driver: golang
+4
View File
@@ -14,7 +14,11 @@ discovery.docker "linux" {
{{ if ne (env "PROMETHEUS_REMOTE_WRITE_URL") "" }}
prometheus.exporter.cadvisor "docker" {
docker_only = true
{{ if eq (env "CADVISOR_DISK_USAGE") "1" }}
enabled_metrics = ["cpu", "cpuLoad", "disk", "diskIO", "memory", "network"]
{{ else }}
enabled_metrics = ["cpu", "cpuLoad", "diskIO", "memory", "network"]
{{ end }}
// host-wide totals already come from prometheus.exporter.unix
disable_root_cgroup_stats = true
}
+3
View File
@@ -59,3 +59,6 @@ See the README's "Auto-discovering metrics from other apps" section.
The Swarm, Stacks and Traefik dashboards were reworked to show old (pull-model)
and new (Alloy push-model) data as one continuous line, so you don't lose history
across the migration.
The disk usage per container was disabled per default because of the CPU-expensive
filesystem scan (30% cpu increase). Enable by uncommenting env CADVISOR_DISK_USAGE=1