Compare commits

..
Author SHA1 Message Date
simon cffb387580 split image tag 2026-09-08 16:30:54 +02:00
10 changed files with 28 additions and 126 deletions
+1 -7
View File
@@ -20,9 +20,6 @@ SECRET_BASIC_AUTH_VERSION=v1
# server is remote
# PROMETHEUS_REMOTE_WRITE_URL=https://prometheus.$DOMAIN/api/v1/write
# Enable container filesystem usage metrics (expensive du-scan, ~30% CPU on large hosts)
# CADVISOR_DISK_USAGE=1
# Enable authenticated scraping of containers that opt in via
# prometheus.io/auth=basic or prometheus.io/auth=bearer labels (used as
# password/bearer token respectively). Insert it with:
@@ -54,9 +51,6 @@ SECRET_BASIC_AUTH_VERSION=v1
# SYSLOG=1
# COMPOSE_FILE="$COMPOSE_FILE:compose.syslog.yml"
# Monitor physical disks health
# COMPOSE_FILE="$COMPOSE_FILE:compose.smartctl.yml"
# Monitoring Server
#
## Prometheus
@@ -90,7 +84,7 @@ SECRET_BASIC_AUTH_VERSION=v1
# SECRET_GF_ADMINPASSWD_VERSION=v1
## Grafana's own domain. Defaults to $DOMAIN
## Change the value if you want Grafana on another domain.
# GRAFANA_DOMAIN=$DOMAIN
GRAFANA_DOMAIN=$DOMAIN
#
## Single-Sign-On with OIDC
# COMPOSE_FILE="$COMPOSE_FILE:compose.grafana-oidc.yml"
-8
View File
@@ -166,11 +166,3 @@ It is possible to enable the following alerts, by uncommenting the corresponding
- node disk space: `ALERT_NODE_DISK_SPACE_LEFT`
- node memory usage: `ALERT_NODE_MEMORY_USAGE`
## smart monitoring
To be able monitor hard drive health data, you need to configure
`smartd` to run on the host system, and also the
`collect-smartctl-json.sh` script provided here (via cronjob or as
a `smartd` hook). This is a limitation on Docker Swarm, which prevents
the `smartctl_exporter` from running on privileged mode.
+1 -1
View File
@@ -10,7 +10,7 @@ export PROMETHEUS_YML_VERSION=v2
export MATRIX_ALERTMANAGER_CONFIG_VERSION=v1
export MATRIX_ALERTMANAGER_ENTRYPOINT_VERSION=v1
export GF_ALERTS_NODE_VERSION=v3
export CONFIG_ALLOY_VERSION=v2
export CONFIG_ALLOY_VERSION=v1
# migrates secrets from old names to new names by reading values from the
# running containers on the server and re-inserting them under the new names.
-6
View File
@@ -1,6 +0,0 @@
[Unit]
Description=Collect SMART data
[Service]
Type=oneshot
ExecStart=/usr/local/bin/collect-smartctl-json.sh
-69
View File
@@ -1,69 +0,0 @@
#! /bin/bash
# Adapted from https://github.com/prometheus-community/smartctl_exporter/blob/master/collect-smartctl-json.sh
script_dir=$( cd -- "$( dirname -- "${BASH_SOURCE[0]}" )" &> /dev/null && pwd )
# Data directory to dump smartctl output
# This directory will be created if it doesn't exist
data_dir="/var/lib/smartmontools/json"
# The original script used --xall but that doesn't work
# This matches the command in readSMARTctl()
smartctl_args="--json --info --health --attributes --tolerance=verypermissive \
--nocheck=standby --format=brief --log=error"
# Ignore this devices
smartctl_ignore_dev_regex="^(/dev/bus)"
# Determine the json query tool to use
if command -v jq >/dev/null; then
json_tool="jq"
json_args="--raw-output"
elif command -v yq >/dev/null; then
json_tool="yq"
json_args="--unwrapScalar"
else
echo -e "One of 'yq' or 'jq' is required. Please try again after \
installing one of them"
exit 1
fi
if [[ ! "${UID}" -eq 0 ]] && ! command -v sudo >/dev/null; then
# Not root and sudo doesn't exist
echo "sudo does not exist. Please run this as root"
exit 1
fi
SUDO="sudo"
if [[ "${UID}" -eq 0 ]]; then
# Don't use sudo if root
SUDO=""
fi
[[ ! -d "${data_dir}" ]] && mkdir --parents "${data_dir}"
if [[ $# -ne 0 ]]; then
devices="${1}"
else
devices="$(smartctl --scan --json | "${json_tool}" "${json_args}" \
".devices[].name | select(test(\"${smartctl_ignore_dev_regex}\") | not)")"
mapfile -t devices <<< "${devices[@]}"
fi
for device in "${devices[@]}"
do
echo -n "Collecting data for '${device}'..."
# shellcheck disable=SC2086
data="$($SUDO smartctl ${smartctl_args} ${device})"
# Accommodate a smartmontools pre-7.3 bug
data=${data#" Pending defect count:"}
type="$(echo "${data}" | "${json_tool}" "${json_args}" '.device.type')"
family="$(echo "${data}" | "${json_tool}" "${json_args}" \
'select(.model_family != null) | .model_family | sub(" |/" ; "_" ; "g")
| sub("\"|\\(|\\)" ; "" ; "g")')"
model="$(echo "${data}" | "${json_tool}" "${json_args}" \
'.model_name | sub(" |/" ; "_" ; "g") | sub("\"|\\(|\\)" ; "" ; "g")')"
device_name="$(basename "${device}")"
echo -e "\tSaving to ${device_name}.json"
echo "${data}" > "${data_dir}/${device_name}.json"
done
-9
View File
@@ -1,9 +0,0 @@
[Unit]
Description=Collect SMART data
[Timer]
OnCalendar=hourly
Persistent=true
[Install]
WantedBy=timers.target
-18
View File
@@ -1,18 +0,0 @@
---
version: "3.8"
services:
smartctl:
image: "prometheuscommunity/smartctl-exporter:v0.14.0"
volumes:
- "/dev:/dev"
- "/var/lib/smartmontools/json:/debug"
command:
- "--smartctl.fake-data"
- "--smartctl.interval=1h"
networks:
- "proxy"
deploy:
labels:
- "prometheus.io/scrape=true"
- "prometheus.io/port=9633"
- "prometheus.io/path=/metrics"
+1 -1
View File
@@ -53,7 +53,7 @@ services:
condition: on-failure
labels:
- "backupbot.backup=${ENABLE_BACKUPS:-true}"
- "coop-cloud.${STACK_NAME}.version=2.0.0+v1.18.1"
- "coop-cloud.${STACK_NAME}.version=1.6.0+v1.8.1"
configs:
config_alloy:
template_driver: golang
+25 -4
View File
@@ -14,11 +14,7 @@ discovery.docker "linux" {
{{ if ne (env "PROMETHEUS_REMOTE_WRITE_URL") "" }}
prometheus.exporter.cadvisor "docker" {
docker_only = true
{{ if eq (env "CADVISOR_DISK_USAGE") "1" }}
enabled_metrics = ["cpu", "cpuLoad", "disk", "diskIO", "memory", "network"]
{{ else }}
enabled_metrics = ["cpu", "cpuLoad", "diskIO", "memory", "network"]
{{ end }}
// host-wide totals already come from prometheus.exporter.unix
disable_root_cgroup_stats = true
}
@@ -52,7 +48,32 @@ prometheus.scrape "default" {
prometheus.exporter.cadvisor.docker.targets,
)
forward_to = [prometheus.relabel.container_meta.receiver]
}
prometheus.relabel "container_meta" {
forward_to = [prometheus.remote_write.prometheus.receiver]
// remove sha tail: nginx:1.31.1@sha256:608a... -> nginx:1.31.1
rule {
source_labels = ["image"]
regex = "([^@]+)@sha256:.*"
target_label = "image"
replacement = "$1"
}
// split image in name and tag
rule {
source_labels = ["image"]
regex = "(.+):[^:/]+"
target_label = "image_name"
replacement = "$1"
}
rule {
source_labels = ["image"]
regex = ".+:([^:/]+)"
target_label = "image_tag"
replacement = "$1"
}
}
prometheus.remote_write "prometheus" {
-3
View File
@@ -59,6 +59,3 @@ See the README's "Auto-discovering metrics from other apps" section.
The Swarm, Stacks and Traefik dashboards were reworked to show old (pull-model)
and new (Alloy push-model) data as one continuous line, so you don't lose history
across the migration.
The disk usage per container was disabled per default because of the CPU-expensive
filesystem scan (30% cpu increase). Enable by uncommenting env CADVISOR_DISK_USAGE=1