Compare commits

..
Author SHA1 Message Date
fauno 5a10a45ced fix: remove unused vars 2026-09-24 12:17:34 -03:00
fauno 9211b98cc1 fix: /dev is not needed 2026-09-24 12:11:59 -03:00
fauno 48888a1fe9 fix: scrape once per hour 2026-09-11 06:49:19 -03:00
fauno b240c5273b feat: systemd timer 2026-09-11 06:49:19 -03:00
fauno 53e03e12a4 feat: smart monitoring 2026-09-11 06:49:19 -03:00
7 changed files with 104 additions and 62 deletions
+3
View File
@@ -54,6 +54,9 @@ SECRET_BASIC_AUTH_VERSION=v1
# SYSLOG=1
# COMPOSE_FILE="$COMPOSE_FILE:compose.syslog.yml"
# Monitor physical disks health
# COMPOSE_FILE="$COMPOSE_FILE:compose.smartctl.yml"
# Monitoring Server
#
## Prometheus
+8
View File
@@ -166,3 +166,11 @@ It is possible to enable the following alerts, by uncommenting the corresponding
- node disk space: `ALERT_NODE_DISK_SPACE_LEFT`
- node memory usage: `ALERT_NODE_MEMORY_USAGE`
## smart monitoring
To be able monitor hard drive health data, you need to configure
`smartd` to run on the host system, and also the
`collect-smartctl-json.sh` script provided here (via cronjob or as
a `smartd` hook). This is a limitation on Docker Swarm, which prevents
the `smartctl_exporter` from running on privileged mode.
+6
View File
@@ -0,0 +1,6 @@
[Unit]
Description=Collect SMART data
[Service]
Type=oneshot
ExecStart=/usr/local/bin/collect-smartctl-json.sh
+61
View File
@@ -0,0 +1,61 @@
#! /bin/bash
# Adapted from https://github.com/prometheus-community/smartctl_exporter/blob/master/collect-smartctl-json.sh
# Data directory to dump smartctl output
# This directory will be created if it doesn't exist
data_dir="/var/lib/smartmontools/json"
# The original script used --xall but that doesn't work
# This matches the command in readSMARTctl()
smartctl_args="--json --info --health --attributes --tolerance=verypermissive \
--nocheck=standby --format=brief --log=error"
# Ignore this devices
smartctl_ignore_dev_regex="^(/dev/bus)"
# Determine the json query tool to use
if command -v jq >/dev/null; then
json_tool="jq"
json_args="--raw-output"
elif command -v yq >/dev/null; then
json_tool="yq"
json_args="--unwrapScalar"
else
echo -e "One of 'yq' or 'jq' is required. Please try again after \
installing one of them"
exit 1
fi
if [[ ! "${UID}" -eq 0 ]] && ! command -v sudo >/dev/null; then
# Not root and sudo doesn't exist
echo "sudo does not exist. Please run this as root"
exit 1
fi
SUDO="sudo"
if [[ "${UID}" -eq 0 ]]; then
# Don't use sudo if root
SUDO=""
fi
[[ ! -d "${data_dir}" ]] && mkdir --parents "${data_dir}"
if [[ $# -ne 0 ]]; then
devices="${1}"
else
devices="$(smartctl --scan --json | "${json_tool}" "${json_args}" \
".devices[].name | select(test(\"${smartctl_ignore_dev_regex}\") | not)")"
mapfile -t devices <<< "${devices[@]}"
fi
for device in "${devices[@]}"
do
echo -n "Collecting data for '${device}'..."
# shellcheck disable=SC2086
data="$($SUDO smartctl ${smartctl_args} ${device})"
# Accommodate a smartmontools pre-7.3 bug
data=${data#" Pending defect count:"}
device_name="$(basename "${device}")"
echo -e "\tSaving to ${device_name}.json"
echo "${data}" > "${data_dir}/${device_name}.json"
done
+9
View File
@@ -0,0 +1,9 @@
[Unit]
Description=Collect SMART data
[Timer]
OnCalendar=hourly
Persistent=true
[Install]
WantedBy=timers.target
+17
View File
@@ -0,0 +1,17 @@
---
version: "3.8"
services:
smartctl:
image: "prometheuscommunity/smartctl-exporter:v0.14.0"
volumes:
- "/var/lib/smartmontools/json:/debug"
command:
- "--smartctl.fake-data"
- "--smartctl.interval=1h"
networks:
- "proxy"
deploy:
labels:
- "prometheus.io/scrape=true"
- "prometheus.io/port=9633"
- "prometheus.io/path=/metrics"
-62
View File
@@ -52,32 +52,7 @@ prometheus.scrape "default" {
prometheus.exporter.cadvisor.docker.targets,
)
forward_to = [prometheus.relabel.container_meta.receiver]
}
prometheus.relabel "container_meta" {
forward_to = [prometheus.remote_write.prometheus.receiver]
// remove sha tail: nginx:1.31.1@sha256:608a... -> nginx:1.31.1
rule {
source_labels = ["image"]
regex = "([^@]+)@sha256:.*"
target_label = "image"
replacement = "$1"
}
// split image in name and tag
rule {
source_labels = ["image"]
regex = "(.+):[^:/]+"
target_label = "image_name"
replacement = "$1"
}
rule {
source_labels = ["image"]
regex = ".+:([^:/]+)"
target_label = "image_tag"
replacement = "$1"
}
}
prometheus.remote_write "prometheus" {
@@ -134,11 +109,6 @@ discovery.relabel "metrics" {
action = "keep"
}
rule {
action = "labelmap"
regex = "__meta_dockerswarm_container_label_coop_cloud_(.+)"
}
// default to port 80 when prometheus.io/port isn't set
rule {
source_labels = ["__meta_dockerswarm_service_label_prometheus_io_port"]
@@ -305,10 +275,6 @@ discovery.relabel "docker" {
source_labels = ["__meta_docker_container_log_stream"]
target_label = "stream"
}
rule {
action = "labelmap"
regex = "__meta_docker_container_label_coop_cloud_(.+)"
}
}
loki.source.docker "docker" {
@@ -324,35 +290,7 @@ loki.source.docker "docker" {
loki.source.journal "journal" {
path = "/rootfs/var/log/journal"
labels = { job = "{{ env "DOMAIN" }}" }
relabel_rules = loki.relabel.journal.rules
forward_to = [loki.process.journal.receiver]
}
loki.relabel "journal" {
forward_to = []
rule {
source_labels = ["__journal__systemd_unit"]
target_label = "unit"
}
rule {
source_labels = ["__journal_syslog_identifier"]
target_label = "ident"
}
rule {
source_labels = ["__journal_priority_keyword"]
target_label = "priority"
}
}
// drop network db stats (15-20% of kernel logs)
loki.process "journal" {
forward_to = [loki.write.loki.receiver]
stage.drop {
expression = ".*NetworkDB stats.*"
drop_counter_reason = "networkdb_stats"
}
}
{{ end }}