diff --git a/zpool-health.sh b/zpool-health.sh new file mode 100755 index 0000000..b2d2030 --- /dev/null +++ b/zpool-health.sh @@ -0,0 +1,66 @@ +#!/bin/bash +# +# Pool health watchdog. ZED does not cover everything: a pool that gets +# SUSPENDED after an I/O failure keeps its vdev marked ONLINE, so +# statechange-notify.sh never fires, and data-class events are silent +# unless ZED_NOTIFY_DATA is set. Three separate incidents on nuc went +# unnoticed for days because of that. +# +# This checks `zpool status -x` and mails only on TRANSITIONS +# (healthy -> problem, problem -> healthy), so it is quiet by default +# and cannot spam. Run it often (every 15 min) from cron. +# +# Usage: zpool-health.sh [-m ] [-s ] [-t] +# -t send a test mail and exit (proves the path works) + +set -u + +MAILTO=root +STATE=/var/lib/zpool-health.state +TEST=0 +HOST=$(hostname -s) + +while getopts m:s:t flag; do + case "${flag}" in + m) MAILTO=${OPTARG};; + s) STATE=${OPTARG};; + t) TEST=1;; + *) echo "Usage: $0 [-m ] [-s ] [-t]" >&2; exit 2;; + esac +done + +notify() { # subject, body + if command -v mail >/dev/null 2>&1; then + printf '%s\n' "$2" | mail -s "$1" "$MAILTO" + else + logger -t zpool-health "$1" + printf '%s\n' "$2" | logger -t zpool-health + fi +} + +if [ "$TEST" -eq 1 ]; then + notify "[$HOST] zpool-health test" "$(zpool status -x 2>&1)" + echo "test notification sent to $MAILTO"; exit 0 +fi + +status=$(zpool status -x 2>&1) +if [ "$status" = "all pools are healthy" ]; then + now=ok +else + now=problem +fi + +was=$(cat "$STATE" 2>/dev/null || echo ok) +printf '%s' "$now" > "$STATE" + +[ "$now" = "$was" ] && exit 0 # no transition: stay quiet + +if [ "$now" = problem ]; then + notify "[$HOST] ZFS POOL PROBLEM" "$(printf '%s\n\n%s\n' "$status" "$(zpool status -v 2>&1)")" + echo "$(date '+%F %T') problem: $status" >&2 + exit 1 +else + notify "[$HOST] ZFS pools healthy again" "$(zpool status 2>&1)" + echo "$(date '+%F %T') recovered" +fi +exit 0