Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
cffb387580
|
+1
-7
@@ -20,9 +20,6 @@ SECRET_BASIC_AUTH_VERSION=v1
|
||||
# server is remote
|
||||
# PROMETHEUS_REMOTE_WRITE_URL=https://prometheus.$DOMAIN/api/v1/write
|
||||
|
||||
# Enable container filesystem usage metrics (expensive du-scan, ~30% CPU on large hosts)
|
||||
# CADVISOR_DISK_USAGE=1
|
||||
|
||||
# Enable authenticated scraping of containers that opt in via
|
||||
# prometheus.io/auth=basic or prometheus.io/auth=bearer labels (used as
|
||||
# password/bearer token respectively). Insert it with:
|
||||
@@ -54,9 +51,6 @@ SECRET_BASIC_AUTH_VERSION=v1
|
||||
# SYSLOG=1
|
||||
# COMPOSE_FILE="$COMPOSE_FILE:compose.syslog.yml"
|
||||
|
||||
# Monitor physical disks health
|
||||
# COMPOSE_FILE="$COMPOSE_FILE:compose.smartctl.yml"
|
||||
|
||||
# Monitoring Server
|
||||
#
|
||||
## Prometheus
|
||||
@@ -90,7 +84,7 @@ SECRET_BASIC_AUTH_VERSION=v1
|
||||
# SECRET_GF_ADMINPASSWD_VERSION=v1
|
||||
## Grafana's own domain. Defaults to $DOMAIN
|
||||
## Change the value if you want Grafana on another domain.
|
||||
# GRAFANA_DOMAIN=$DOMAIN
|
||||
GRAFANA_DOMAIN=$DOMAIN
|
||||
#
|
||||
## Single-Sign-On with OIDC
|
||||
# COMPOSE_FILE="$COMPOSE_FILE:compose.grafana-oidc.yml"
|
||||
|
||||
@@ -166,11 +166,3 @@ It is possible to enable the following alerts, by uncommenting the corresponding
|
||||
|
||||
- node disk space: `ALERT_NODE_DISK_SPACE_LEFT`
|
||||
- node memory usage: `ALERT_NODE_MEMORY_USAGE`
|
||||
|
||||
## smart monitoring
|
||||
|
||||
To be able monitor hard drive health data, you need to configure
|
||||
`smartd` to run on the host system, and also the
|
||||
`collect-smartctl-json.sh` script provided here (via cronjob or as
|
||||
a `smartd` hook). This is a limitation on Docker Swarm, which prevents
|
||||
the `smartctl_exporter` from running on privileged mode.
|
||||
|
||||
@@ -10,7 +10,7 @@ export PROMETHEUS_YML_VERSION=v2
|
||||
export MATRIX_ALERTMANAGER_CONFIG_VERSION=v1
|
||||
export MATRIX_ALERTMANAGER_ENTRYPOINT_VERSION=v1
|
||||
export GF_ALERTS_NODE_VERSION=v3
|
||||
export CONFIG_ALLOY_VERSION=v2
|
||||
export CONFIG_ALLOY_VERSION=v1
|
||||
|
||||
# migrates secrets from old names to new names by reading values from the
|
||||
# running containers on the server and re-inserting them under the new names.
|
||||
|
||||
@@ -1,6 +0,0 @@
|
||||
[Unit]
|
||||
Description=Collect SMART data
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/bin/collect-smartctl-json.sh
|
||||
@@ -1,69 +0,0 @@
|
||||
#! /bin/bash
|
||||
# Adapted from https://github.com/prometheus-community/smartctl_exporter/blob/master/collect-smartctl-json.sh
|
||||
|
||||
script_dir=$( cd -- "$( dirname -- "${BASH_SOURCE[0]}" )" &> /dev/null && pwd )
|
||||
|
||||
# Data directory to dump smartctl output
|
||||
# This directory will be created if it doesn't exist
|
||||
data_dir="/var/lib/smartmontools/json"
|
||||
|
||||
# The original script used --xall but that doesn't work
|
||||
# This matches the command in readSMARTctl()
|
||||
smartctl_args="--json --info --health --attributes --tolerance=verypermissive \
|
||||
--nocheck=standby --format=brief --log=error"
|
||||
|
||||
# Ignore this devices
|
||||
smartctl_ignore_dev_regex="^(/dev/bus)"
|
||||
|
||||
# Determine the json query tool to use
|
||||
if command -v jq >/dev/null; then
|
||||
json_tool="jq"
|
||||
json_args="--raw-output"
|
||||
elif command -v yq >/dev/null; then
|
||||
json_tool="yq"
|
||||
json_args="--unwrapScalar"
|
||||
else
|
||||
echo -e "One of 'yq' or 'jq' is required. Please try again after \
|
||||
installing one of them"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ ! "${UID}" -eq 0 ]] && ! command -v sudo >/dev/null; then
|
||||
# Not root and sudo doesn't exist
|
||||
echo "sudo does not exist. Please run this as root"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
SUDO="sudo"
|
||||
if [[ "${UID}" -eq 0 ]]; then
|
||||
# Don't use sudo if root
|
||||
SUDO=""
|
||||
fi
|
||||
|
||||
[[ ! -d "${data_dir}" ]] && mkdir --parents "${data_dir}"
|
||||
|
||||
if [[ $# -ne 0 ]]; then
|
||||
devices="${1}"
|
||||
else
|
||||
devices="$(smartctl --scan --json | "${json_tool}" "${json_args}" \
|
||||
".devices[].name | select(test(\"${smartctl_ignore_dev_regex}\") | not)")"
|
||||
mapfile -t devices <<< "${devices[@]}"
|
||||
fi
|
||||
|
||||
for device in "${devices[@]}"
|
||||
do
|
||||
echo -n "Collecting data for '${device}'..."
|
||||
# shellcheck disable=SC2086
|
||||
data="$($SUDO smartctl ${smartctl_args} ${device})"
|
||||
# Accommodate a smartmontools pre-7.3 bug
|
||||
data=${data#" Pending defect count:"}
|
||||
type="$(echo "${data}" | "${json_tool}" "${json_args}" '.device.type')"
|
||||
family="$(echo "${data}" | "${json_tool}" "${json_args}" \
|
||||
'select(.model_family != null) | .model_family | sub(" |/" ; "_" ; "g")
|
||||
| sub("\"|\\(|\\)" ; "" ; "g")')"
|
||||
model="$(echo "${data}" | "${json_tool}" "${json_args}" \
|
||||
'.model_name | sub(" |/" ; "_" ; "g") | sub("\"|\\(|\\)" ; "" ; "g")')"
|
||||
device_name="$(basename "${device}")"
|
||||
echo -e "\tSaving to ${device_name}.json"
|
||||
echo "${data}" > "${data_dir}/${device_name}.json"
|
||||
done
|
||||
@@ -1,9 +0,0 @@
|
||||
[Unit]
|
||||
Description=Collect SMART data
|
||||
|
||||
[Timer]
|
||||
OnCalendar=hourly
|
||||
Persistent=true
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
@@ -1,18 +0,0 @@
|
||||
---
|
||||
version: "3.8"
|
||||
services:
|
||||
smartctl:
|
||||
image: "prometheuscommunity/smartctl-exporter:v0.14.0"
|
||||
volumes:
|
||||
- "/dev:/dev"
|
||||
- "/var/lib/smartmontools/json:/debug"
|
||||
command:
|
||||
- "--smartctl.fake-data"
|
||||
- "--smartctl.interval=1h"
|
||||
networks:
|
||||
- "proxy"
|
||||
deploy:
|
||||
labels:
|
||||
- "prometheus.io/scrape=true"
|
||||
- "prometheus.io/port=9633"
|
||||
- "prometheus.io/path=/metrics"
|
||||
+1
-1
@@ -53,7 +53,7 @@ services:
|
||||
condition: on-failure
|
||||
labels:
|
||||
- "backupbot.backup=${ENABLE_BACKUPS:-true}"
|
||||
- "coop-cloud.${STACK_NAME}.version=2.0.0+v1.18.1"
|
||||
- "coop-cloud.${STACK_NAME}.version=1.6.0+v1.8.1"
|
||||
configs:
|
||||
config_alloy:
|
||||
template_driver: golang
|
||||
|
||||
+25
-4
@@ -14,11 +14,7 @@ discovery.docker "linux" {
|
||||
{{ if ne (env "PROMETHEUS_REMOTE_WRITE_URL") "" }}
|
||||
prometheus.exporter.cadvisor "docker" {
|
||||
docker_only = true
|
||||
{{ if eq (env "CADVISOR_DISK_USAGE") "1" }}
|
||||
enabled_metrics = ["cpu", "cpuLoad", "disk", "diskIO", "memory", "network"]
|
||||
{{ else }}
|
||||
enabled_metrics = ["cpu", "cpuLoad", "diskIO", "memory", "network"]
|
||||
{{ end }}
|
||||
// host-wide totals already come from prometheus.exporter.unix
|
||||
disable_root_cgroup_stats = true
|
||||
}
|
||||
@@ -52,7 +48,32 @@ prometheus.scrape "default" {
|
||||
prometheus.exporter.cadvisor.docker.targets,
|
||||
)
|
||||
|
||||
forward_to = [prometheus.relabel.container_meta.receiver]
|
||||
}
|
||||
|
||||
prometheus.relabel "container_meta" {
|
||||
forward_to = [prometheus.remote_write.prometheus.receiver]
|
||||
|
||||
// remove sha tail: nginx:1.31.1@sha256:608a... -> nginx:1.31.1
|
||||
rule {
|
||||
source_labels = ["image"]
|
||||
regex = "([^@]+)@sha256:.*"
|
||||
target_label = "image"
|
||||
replacement = "$1"
|
||||
}
|
||||
// split image in name and tag
|
||||
rule {
|
||||
source_labels = ["image"]
|
||||
regex = "(.+):[^:/]+"
|
||||
target_label = "image_name"
|
||||
replacement = "$1"
|
||||
}
|
||||
rule {
|
||||
source_labels = ["image"]
|
||||
regex = ".+:([^:/]+)"
|
||||
target_label = "image_tag"
|
||||
replacement = "$1"
|
||||
}
|
||||
}
|
||||
|
||||
prometheus.remote_write "prometheus" {
|
||||
|
||||
@@ -59,6 +59,3 @@ See the README's "Auto-discovering metrics from other apps" section.
|
||||
The Swarm, Stacks and Traefik dashboards were reworked to show old (pull-model)
|
||||
and new (Alloy push-model) data as one continuous line, so you don't lose history
|
||||
across the migration.
|
||||
|
||||
The disk usage per container was disabled per default because of the CPU-expensive
|
||||
filesystem scan (30% cpu increase). Enable by uncommenting env CADVISOR_DISK_USAGE=1
|
||||
Reference in New Issue
Block a user