Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5a10a45ced
|
||
|
|
9211b98cc1
|
||
|
|
48888a1fe9
|
||
|
|
b240c5273b
|
||
|
|
53e03e12a4
|
+2
-3
@@ -53,10 +53,9 @@ SECRET_BASIC_AUTH_VERSION=v1
|
||||
# Not needed if you just want to read local log files — use SYSLOG_FILES instead.
|
||||
# SYSLOG=1
|
||||
# COMPOSE_FILE="$COMPOSE_FILE:compose.syslog.yml"
|
||||
#
|
||||
# Scrape Docker-Events via official docker-cli image using docker socket
|
||||
# COMPOSE_FILE="$COMPOSE_FILE:compose.docker-events.yml"
|
||||
|
||||
# Monitor physical disks health
|
||||
# COMPOSE_FILE="$COMPOSE_FILE:compose.smartctl.yml"
|
||||
|
||||
# Monitoring Server
|
||||
#
|
||||
|
||||
@@ -54,14 +54,6 @@ This is what a gathering host pushes into. It also runs its own Alloy, so it mon
|
||||
|
||||
## Additional features
|
||||
|
||||
### Scrape docker events
|
||||
|
||||
Just uncomment the following line in your .env to collect docker events using
|
||||
the official docker-cli image and the docker socket:
|
||||
```
|
||||
COMPOSE_FILE="$COMPOSE_FILE:compose.docker-events.yml"
|
||||
```
|
||||
|
||||
### Discovering metrics from other apps
|
||||
|
||||
Alloy auto-discovers and scrapes other Docker Swarm services running on the same host, on the `proxy` network, that opt in via labels. No manual scrape config needed. On the app's `compose.yml`:
|
||||
@@ -174,3 +166,11 @@ It is possible to enable the following alerts, by uncommenting the corresponding
|
||||
|
||||
- node disk space: `ALERT_NODE_DISK_SPACE_LEFT`
|
||||
- node memory usage: `ALERT_NODE_MEMORY_USAGE`
|
||||
|
||||
## smart monitoring
|
||||
|
||||
To be able monitor hard drive health data, you need to configure
|
||||
`smartd` to run on the host system, and also the
|
||||
`collect-smartctl-json.sh` script provided here (via cronjob or as
|
||||
a `smartd` hook). This is a limitation on Docker Swarm, which prevents
|
||||
the `smartctl_exporter` from running on privileged mode.
|
||||
|
||||
@@ -10,7 +10,7 @@ export PROMETHEUS_YML_VERSION=v2
|
||||
export MATRIX_ALERTMANAGER_CONFIG_VERSION=v1
|
||||
export MATRIX_ALERTMANAGER_ENTRYPOINT_VERSION=v1
|
||||
export GF_ALERTS_NODE_VERSION=v3
|
||||
export CONFIG_ALLOY_VERSION=v3
|
||||
export CONFIG_ALLOY_VERSION=v2
|
||||
|
||||
# migrates secrets from old names to new names by reading values from the
|
||||
# running containers on the server and re-inserting them under the new names.
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
[Unit]
|
||||
Description=Collect SMART data
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/bin/collect-smartctl-json.sh
|
||||
Executable
+61
@@ -0,0 +1,61 @@
|
||||
#! /bin/bash
|
||||
# Adapted from https://github.com/prometheus-community/smartctl_exporter/blob/master/collect-smartctl-json.sh
|
||||
|
||||
# Data directory to dump smartctl output
|
||||
# This directory will be created if it doesn't exist
|
||||
data_dir="/var/lib/smartmontools/json"
|
||||
|
||||
# The original script used --xall but that doesn't work
|
||||
# This matches the command in readSMARTctl()
|
||||
smartctl_args="--json --info --health --attributes --tolerance=verypermissive \
|
||||
--nocheck=standby --format=brief --log=error"
|
||||
|
||||
# Ignore this devices
|
||||
smartctl_ignore_dev_regex="^(/dev/bus)"
|
||||
|
||||
# Determine the json query tool to use
|
||||
if command -v jq >/dev/null; then
|
||||
json_tool="jq"
|
||||
json_args="--raw-output"
|
||||
elif command -v yq >/dev/null; then
|
||||
json_tool="yq"
|
||||
json_args="--unwrapScalar"
|
||||
else
|
||||
echo -e "One of 'yq' or 'jq' is required. Please try again after \
|
||||
installing one of them"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ ! "${UID}" -eq 0 ]] && ! command -v sudo >/dev/null; then
|
||||
# Not root and sudo doesn't exist
|
||||
echo "sudo does not exist. Please run this as root"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
SUDO="sudo"
|
||||
if [[ "${UID}" -eq 0 ]]; then
|
||||
# Don't use sudo if root
|
||||
SUDO=""
|
||||
fi
|
||||
|
||||
[[ ! -d "${data_dir}" ]] && mkdir --parents "${data_dir}"
|
||||
|
||||
if [[ $# -ne 0 ]]; then
|
||||
devices="${1}"
|
||||
else
|
||||
devices="$(smartctl --scan --json | "${json_tool}" "${json_args}" \
|
||||
".devices[].name | select(test(\"${smartctl_ignore_dev_regex}\") | not)")"
|
||||
mapfile -t devices <<< "${devices[@]}"
|
||||
fi
|
||||
|
||||
for device in "${devices[@]}"
|
||||
do
|
||||
echo -n "Collecting data for '${device}'..."
|
||||
# shellcheck disable=SC2086
|
||||
data="$($SUDO smartctl ${smartctl_args} ${device})"
|
||||
# Accommodate a smartmontools pre-7.3 bug
|
||||
data=${data#" Pending defect count:"}
|
||||
device_name="$(basename "${device}")"
|
||||
echo -e "\tSaving to ${device_name}.json"
|
||||
echo "${data}" > "${data_dir}/${device_name}.json"
|
||||
done
|
||||
@@ -0,0 +1,9 @@
|
||||
[Unit]
|
||||
Description=Collect SMART data
|
||||
|
||||
[Timer]
|
||||
OnCalendar=hourly
|
||||
Persistent=true
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
@@ -1,20 +0,0 @@
|
||||
version: "3.8"
|
||||
services:
|
||||
docker-events:
|
||||
image: docker:29.8.1-cli
|
||||
command: >
|
||||
docker events
|
||||
--format '{{json .}}'
|
||||
--filter type=container
|
||||
--filter event=start
|
||||
--filter event=die
|
||||
--filter event=kill
|
||||
--filter event=oom
|
||||
--filter event=restart
|
||||
--filter event=stop
|
||||
--filter event=health_status
|
||||
volumes:
|
||||
- /var/run/docker.sock:/var/run/docker.sock:ro
|
||||
deploy:
|
||||
restart_policy:
|
||||
condition: on-failure
|
||||
@@ -0,0 +1,17 @@
|
||||
---
|
||||
version: "3.8"
|
||||
services:
|
||||
smartctl:
|
||||
image: "prometheuscommunity/smartctl-exporter:v0.14.0"
|
||||
volumes:
|
||||
- "/var/lib/smartmontools/json:/debug"
|
||||
command:
|
||||
- "--smartctl.fake-data"
|
||||
- "--smartctl.interval=1h"
|
||||
networks:
|
||||
- "proxy"
|
||||
deploy:
|
||||
labels:
|
||||
- "prometheus.io/scrape=true"
|
||||
- "prometheus.io/port=9633"
|
||||
- "prometheus.io/path=/metrics"
|
||||
+1
-1
@@ -53,7 +53,7 @@ services:
|
||||
condition: on-failure
|
||||
labels:
|
||||
- "backupbot.backup=${ENABLE_BACKUPS:-true}"
|
||||
- "coop-cloud.${STACK_NAME}.version=2.1.0+v1.18.1"
|
||||
- "coop-cloud.${STACK_NAME}.version=2.0.0+v1.18.1"
|
||||
configs:
|
||||
config_alloy:
|
||||
template_driver: golang
|
||||
|
||||
@@ -52,32 +52,7 @@ prometheus.scrape "default" {
|
||||
prometheus.exporter.cadvisor.docker.targets,
|
||||
)
|
||||
|
||||
forward_to = [prometheus.relabel.container_meta.receiver]
|
||||
}
|
||||
|
||||
prometheus.relabel "container_meta" {
|
||||
forward_to = [prometheus.remote_write.prometheus.receiver]
|
||||
|
||||
// remove sha tail: nginx:1.31.1@sha256:608a... -> nginx:1.31.1
|
||||
rule {
|
||||
source_labels = ["image"]
|
||||
regex = "([^@]+)@sha256:.*"
|
||||
target_label = "image"
|
||||
replacement = "$1"
|
||||
}
|
||||
// split image in name and tag
|
||||
rule {
|
||||
source_labels = ["image"]
|
||||
regex = "(.+):[^:/]+"
|
||||
target_label = "image_name"
|
||||
replacement = "$1"
|
||||
}
|
||||
rule {
|
||||
source_labels = ["image"]
|
||||
regex = ".+:([^:/]+)"
|
||||
target_label = "image_tag"
|
||||
replacement = "$1"
|
||||
}
|
||||
}
|
||||
|
||||
prometheus.remote_write "prometheus" {
|
||||
@@ -315,35 +290,7 @@ loki.source.docker "docker" {
|
||||
loki.source.journal "journal" {
|
||||
path = "/rootfs/var/log/journal"
|
||||
labels = { job = "{{ env "DOMAIN" }}" }
|
||||
relabel_rules = loki.relabel.journal.rules
|
||||
forward_to = [loki.process.journal.receiver]
|
||||
}
|
||||
|
||||
loki.relabel "journal" {
|
||||
forward_to = []
|
||||
|
||||
rule {
|
||||
source_labels = ["__journal__systemd_unit"]
|
||||
target_label = "unit"
|
||||
}
|
||||
rule {
|
||||
source_labels = ["__journal_syslog_identifier"]
|
||||
target_label = "ident"
|
||||
}
|
||||
rule {
|
||||
source_labels = ["__journal_priority_keyword"]
|
||||
target_label = "priority"
|
||||
}
|
||||
}
|
||||
|
||||
// drop network db stats (15-20% of kernel logs)
|
||||
loki.process "journal" {
|
||||
forward_to = [loki.write.loki.receiver]
|
||||
|
||||
stage.drop {
|
||||
expression = ".*NetworkDB stats.*"
|
||||
drop_counter_reason = "networkdb_stats"
|
||||
}
|
||||
}
|
||||
{{ end }}
|
||||
|
||||
|
||||
@@ -1 +0,0 @@
|
||||
alloy: split image label in "image_name" and "image_tag", relabel journald logs and drop NetworkDB stats
|
||||
@@ -1 +0,0 @@
|
||||
allow scraping of docker events via new compose.docker-events.yml
|
||||
Reference in New Issue
Block a user