Compare commits

..
Author SHA1 Message Date
fauno 5a10a45ced fix: remove unused vars 2026-09-24 12:17:34 -03:00
fauno 9211b98cc1 fix: /dev is not needed 2026-09-24 12:11:59 -03:00
fauno 48888a1fe9 fix: scrape once per hour 2026-09-11 06:49:19 -03:00
fauno b240c5273b feat: systemd timer 2026-09-11 06:49:19 -03:00
fauno 53e03e12a4 feat: smart monitoring 2026-09-11 06:49:19 -03:00
12 changed files with 105 additions and 88 deletions
+2 -3
View File
@@ -53,10 +53,9 @@ SECRET_BASIC_AUTH_VERSION=v1
# Not needed if you just want to read local log files — use SYSLOG_FILES instead.
# SYSLOG=1
# COMPOSE_FILE="$COMPOSE_FILE:compose.syslog.yml"
#
# Scrape Docker-Events via official docker-cli image using docker socket
# COMPOSE_FILE="$COMPOSE_FILE:compose.docker-events.yml"
# Monitor physical disks health
# COMPOSE_FILE="$COMPOSE_FILE:compose.smartctl.yml"
# Monitoring Server
#
+8 -8
View File
@@ -54,14 +54,6 @@ This is what a gathering host pushes into. It also runs its own Alloy, so it mon
## Additional features
### Scrape docker events
Just uncomment the following line in your .env to collect docker events using
the official docker-cli image and the docker socket:
```
COMPOSE_FILE="$COMPOSE_FILE:compose.docker-events.yml"
```
### Discovering metrics from other apps
Alloy auto-discovers and scrapes other Docker Swarm services running on the same host, on the `proxy` network, that opt in via labels. No manual scrape config needed. On the app's `compose.yml`:
@@ -174,3 +166,11 @@ It is possible to enable the following alerts, by uncommenting the corresponding
- node disk space: `ALERT_NODE_DISK_SPACE_LEFT`
- node memory usage: `ALERT_NODE_MEMORY_USAGE`
## smart monitoring
To be able monitor hard drive health data, you need to configure
`smartd` to run on the host system, and also the
`collect-smartctl-json.sh` script provided here (via cronjob or as
a `smartd` hook). This is a limitation on Docker Swarm, which prevents
the `smartctl_exporter` from running on privileged mode.
+1 -1
View File
@@ -10,7 +10,7 @@ export PROMETHEUS_YML_VERSION=v2
export MATRIX_ALERTMANAGER_CONFIG_VERSION=v1
export MATRIX_ALERTMANAGER_ENTRYPOINT_VERSION=v1
export GF_ALERTS_NODE_VERSION=v3
export CONFIG_ALLOY_VERSION=v3
export CONFIG_ALLOY_VERSION=v2
# migrates secrets from old names to new names by reading values from the
# running containers on the server and re-inserting them under the new names.
+6
View File
@@ -0,0 +1,6 @@
[Unit]
Description=Collect SMART data
[Service]
Type=oneshot
ExecStart=/usr/local/bin/collect-smartctl-json.sh
+61
View File
@@ -0,0 +1,61 @@
#! /bin/bash
# Adapted from https://github.com/prometheus-community/smartctl_exporter/blob/master/collect-smartctl-json.sh
# Data directory to dump smartctl output
# This directory will be created if it doesn't exist
data_dir="/var/lib/smartmontools/json"
# The original script used --xall but that doesn't work
# This matches the command in readSMARTctl()
smartctl_args="--json --info --health --attributes --tolerance=verypermissive \
--nocheck=standby --format=brief --log=error"
# Ignore this devices
smartctl_ignore_dev_regex="^(/dev/bus)"
# Determine the json query tool to use
if command -v jq >/dev/null; then
json_tool="jq"
json_args="--raw-output"
elif command -v yq >/dev/null; then
json_tool="yq"
json_args="--unwrapScalar"
else
echo -e "One of 'yq' or 'jq' is required. Please try again after \
installing one of them"
exit 1
fi
if [[ ! "${UID}" -eq 0 ]] && ! command -v sudo >/dev/null; then
# Not root and sudo doesn't exist
echo "sudo does not exist. Please run this as root"
exit 1
fi
SUDO="sudo"
if [[ "${UID}" -eq 0 ]]; then
# Don't use sudo if root
SUDO=""
fi
[[ ! -d "${data_dir}" ]] && mkdir --parents "${data_dir}"
if [[ $# -ne 0 ]]; then
devices="${1}"
else
devices="$(smartctl --scan --json | "${json_tool}" "${json_args}" \
".devices[].name | select(test(\"${smartctl_ignore_dev_regex}\") | not)")"
mapfile -t devices <<< "${devices[@]}"
fi
for device in "${devices[@]}"
do
echo -n "Collecting data for '${device}'..."
# shellcheck disable=SC2086
data="$($SUDO smartctl ${smartctl_args} ${device})"
# Accommodate a smartmontools pre-7.3 bug
data=${data#" Pending defect count:"}
device_name="$(basename "${device}")"
echo -e "\tSaving to ${device_name}.json"
echo "${data}" > "${data_dir}/${device_name}.json"
done
+9
View File
@@ -0,0 +1,9 @@
[Unit]
Description=Collect SMART data
[Timer]
OnCalendar=hourly
Persistent=true
[Install]
WantedBy=timers.target
-20
View File
@@ -1,20 +0,0 @@
version: "3.8"
services:
docker-events:
image: docker:29.8.1-cli
command: >
docker events
--format '{{json .}}'
--filter type=container
--filter event=start
--filter event=die
--filter event=kill
--filter event=oom
--filter event=restart
--filter event=stop
--filter event=health_status
volumes:
- /var/run/docker.sock:/var/run/docker.sock:ro
deploy:
restart_policy:
condition: on-failure
+17
View File
@@ -0,0 +1,17 @@
---
version: "3.8"
services:
smartctl:
image: "prometheuscommunity/smartctl-exporter:v0.14.0"
volumes:
- "/var/lib/smartmontools/json:/debug"
command:
- "--smartctl.fake-data"
- "--smartctl.interval=1h"
networks:
- "proxy"
deploy:
labels:
- "prometheus.io/scrape=true"
- "prometheus.io/port=9633"
- "prometheus.io/path=/metrics"
+1 -1
View File
@@ -53,7 +53,7 @@ services:
condition: on-failure
labels:
- "backupbot.backup=${ENABLE_BACKUPS:-true}"
- "coop-cloud.${STACK_NAME}.version=2.1.0+v1.18.1"
- "coop-cloud.${STACK_NAME}.version=2.0.0+v1.18.1"
configs:
config_alloy:
template_driver: golang
-53
View File
@@ -52,32 +52,7 @@ prometheus.scrape "default" {
prometheus.exporter.cadvisor.docker.targets,
)
forward_to = [prometheus.relabel.container_meta.receiver]
}
prometheus.relabel "container_meta" {
forward_to = [prometheus.remote_write.prometheus.receiver]
// remove sha tail: nginx:1.31.1@sha256:608a... -> nginx:1.31.1
rule {
source_labels = ["image"]
regex = "([^@]+)@sha256:.*"
target_label = "image"
replacement = "$1"
}
// split image in name and tag
rule {
source_labels = ["image"]
regex = "(.+):[^:/]+"
target_label = "image_name"
replacement = "$1"
}
rule {
source_labels = ["image"]
regex = ".+:([^:/]+)"
target_label = "image_tag"
replacement = "$1"
}
}
prometheus.remote_write "prometheus" {
@@ -315,35 +290,7 @@ loki.source.docker "docker" {
loki.source.journal "journal" {
path = "/rootfs/var/log/journal"
labels = { job = "{{ env "DOMAIN" }}" }
relabel_rules = loki.relabel.journal.rules
forward_to = [loki.process.journal.receiver]
}
loki.relabel "journal" {
forward_to = []
rule {
source_labels = ["__journal__systemd_unit"]
target_label = "unit"
}
rule {
source_labels = ["__journal_syslog_identifier"]
target_label = "ident"
}
rule {
source_labels = ["__journal_priority_keyword"]
target_label = "priority"
}
}
// drop network db stats (15-20% of kernel logs)
loki.process "journal" {
forward_to = [loki.write.loki.receiver]
stage.drop {
expression = ".*NetworkDB stats.*"
drop_counter_reason = "networkdb_stats"
}
}
{{ end }}
-1
View File
@@ -1 +0,0 @@
alloy: split image label in "image_name" and "image_tag", relabel journald logs and drop NetworkDB stats
-1
View File
@@ -1 +0,0 @@
allow scraping of docker events via new compose.docker-events.yml