Files
Varaverk/unRAID_Essentials/system_watchdog.sh
T

889 lines
45 KiB
Bash
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/bin/bash
# ==============================================================================================
# ================================= System Watchdog ============================================
# ==============================================================================================
# Last line of defense — reboots the system cleanly if it is about to become unstable.
# Runs continuously as a background process — started by array_start.sh at array start.
# Works alongside docker_watchdog.sh which handles container-level healing first.
#
# ── THREE-TIER RESPONSE SYSTEM ────────────────────────────────────────────────────────────────
#
# TIER 1 — CRITICAL (bypass ALL strikes, reboot immediately)
# Docker daemon unresponsive — nothing can be healed, letting it run makes it worse
# rootfs at 99%+ — writes failing, SSH may stop, no recovery options
# Kernel oops/BUG in dmesg — kernel running with corrupted state
# File descriptor exhaustion — new connections and processes failing silently
# /boot read-only unexpectedly — state files and config writes silently failing
#
# TIER 2 — URGENT (bypass strikes when OOM confirms active crisis)
# RAM < MEM_GB AND OOM kills >= OOM_LIMIT in this cycle
# Rationale: OOM kills at this rate means system is dying faster than watchdogs heal
# Without OOM confirmation → standard strike system applies
#
# TIER 3 — STANDARD (N consecutive failures → reboot)
# RAM tiers, load, CPU temp, zombies, /var/log, /tmp, containers, NIC, mdstat
#
# ── RAM TIERS ─────────────────────────────────────────────────────────────────────────────────
# MEM_WARN_GB (10GB) — warn + notify only
# MEM_SHUTDOWN_GB (6GB) — stop non-essential containers, wait for recovery
# MEM_GB (4GB) — strike system → reboot (bypass if OOM confirms)
# MEM_RECOVER_GB (30GB) — RAM must reach this before containers restart
#
# ── CONTAINER SHUTDOWN LOGIC ──────────────────────────────────────────────────────────────────
# At MEM_SHUTDOWN_GB: stop all containers NOT in SYS_WATCHDOG_MEM_SHUTDOWN_EXCLUDED
# Excluded: NginxProxyManager, Authelia, Mariadb, Redis, Emby, Dispatcharr
# Stopped containers tracked in shutdown list — won't restart until RAM recovers
# Strike system prevents flip-flopping — shutdown only happens once per degradation event
#
# ── OOM TRACKING ──────────────────────────────────────────────────────────────────────────────
# /proc/vmstat oom_kill counter — read each cycle, delta = kills this cycle
# Included in reboot message with process names from dmesg (diagnostic context)
# Bypass trigger: RAM critical AND kills this cycle >= SYS_WATCHDOG_OOM_LIMIT
#
# ── NEW CHECKS THIS VERSION ───────────────────────────────────────────────────────────────────
# OOM rate tracking — delta from /proc/vmstat each cycle
# /boot read-only — write test on /boot each cycle
# Kernel oops detection — dmesg BUG/Oops count delta each cycle
# File descriptor exhaustion — /proc/sys/fs/file-nr utilisation
# /tmp usage — tmpfs fill detection with auto-clear attempt
# Array disk errors — mdstat error delta each cycle
# Runaway process — single process >N% CPU sustained (disabled by default)
# NIC state check — primary interface operstate
# sshd check — restart attempt before escalating
#
# ── EXISTING CHECKS ───────────────────────────────────────────────────────────────────────────
# rootfs usage, /var/log, free RAM, ZFS ARC, CPU temp, load avg,
# zombie processes, Docker daemon, required containers from skip list
#
# ── ABORT CONDITIONS ──────────────────────────────────────────────────────────────────────────
# ZFS pool unhealthy, parity running, mover running — toggleable
# CRITICAL tier bypasses abort conditions — imminent crash overrides data safety
#
# ── CONFIGURATION (master.conf) ───────────────────────────────────────────────────────────────
# Full config under System Watchdog section — see master.conf for all vars
#
# ── USAGE ─────────────────────────────────────────────────────────────────────────────────────
# system_watchdog.sh — normal start (continuous loop)
# system_watchdog.sh --dry-run — trigger detection without rebooting
# system_watchdog.sh --status — show config and thresholds
# system_watchdog.sh --log — verbose per-cycle output
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup — runs once at start ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
validate_unraid_cmd \
"/usr/local/emhttp/plugins/dynamix/scripts/notify" \
"" "" \
"unRAID notify script" || warn "unRAID notify script not found — native notifications disabled"
acquire_lock "continuous"
detect_hosts
TOTAL_CORES=$(nproc)
DOCKER_TIMEOUT=10
# Ensure state files exist
for state_file in "$SYS_WATCHDOG_STATE_FILE" "$SYS_WATCHDOG_REBOOT_LOG" \
"$SYS_WATCHDOG_FAILED_FILE" "$SYS_WATCHDOG_OOM_FILE"; do
touch "$state_file" 2>/dev/null || {
error "Cannot create state file: $state_file"
exit 1
}
done
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no reboots or container shutdowns will occur"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY SYSTEM WATCHDOG STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo ""
echo "── Tier 1 — CRITICAL (bypass strikes immediately) ──"
echo "$ICON_DISK rootfs critical: ${SYS_WATCHDOG_ROOTFS_CRITICAL_PCT}%"
echo "$ICON_GEAR FD critical: ${SYS_WATCHDOG_FD_CRITICAL_PCT}%"
echo "$ICON_GEAR /boot read-only: check=${SYS_WATCHDOG_CHECK_BOOT}"
echo "$ICON_GEAR Kernel oops: check=${SYS_WATCHDOG_CHECK_KERNEL_OOPS}"
echo "$ICON_CONTAINERS Docker daemon: check=${SYS_WATCHDOG_CHECK_DOCKER_DAEMON}"
echo ""
echo "── Tier 2 — URGENT (bypass strikes with OOM confirmation) ──"
echo "$ICON_MEM RAM critical: < ${SYS_WATCHDOG_MEM_GB}GB"
echo "$ICON_GEAR OOM limit: ${SYS_WATCHDOG_OOM_LIMIT} kills/cycle"
echo ""
echo "── Tier 3 — STANDARD (strike system) ──"
echo "$ICON_DISK rootfs warn: ${SYS_WATCHDOG_ROOTFS_PCT}%"
echo "$ICON_GEAR /var/log warn: ${SYS_WATCHDOG_LOG_PCT}%"
echo "$ICON_GEAR /tmp warn: ${SYS_WATCHDOG_TMP_PCT}%"
echo "$ICON_MEM RAM warn: < ${SYS_WATCHDOG_MEM_WARN_GB}GB"
echo "$ICON_MEM RAM shutdown: < ${SYS_WATCHDOG_MEM_SHUTDOWN_GB}GB"
echo "$ICON_MEM RAM recover: > ${SYS_WATCHDOG_MEM_RECOVER_GB}GB"
echo "$ICON_MEM RAM reboot: < ${SYS_WATCHDOG_MEM_GB}GB (+ strikes)"
echo "$ICON_ZFS ARC pinned: ${SYS_WATCHDOG_ARC_PINNED_PCT}%"
echo "$ICON_GEAR Load multiplier: ${SYS_WATCHDOG_LOAD_MULTIPLIER}x (= $(( TOTAL_CORES * SYS_WATCHDOG_LOAD_MULTIPLIER )) on $TOTAL_CORES cores)"
echo "$ICON_GEAR Zombie limit: ${SYS_WATCHDOG_ZOMBIE_LIMIT}"
echo "$ICON_GEAR CPU temp max: ${SYS_WATCHDOG_CPU_TEMP_MAX}°C"
echo "$ICON_GEAR Strike limit: ${SYS_WATCHDOG_STRIKE_LIMIT} cycles"
echo "$ICON_TIME Interval: ${SYSTEM_WATCHDOG_INTERVAL}s"
echo "$ICON_REBOOT_SMART Reboot limit: ${SYS_WATCHDOG_REBOOT_LIMIT} in ${SYS_WATCHDOG_REBOOT_WINDOW_HRS}hr"
echo ""
echo "── Container Shutdown Excluded ──"
for c in "${SYS_WATCHDOG_MEM_SHUTDOWN_EXCLUDED[@]:-}"; do
echo " $ICON_RUNNING $c"
done
echo ""
echo "── Check Toggles ──"
echo " rootfs=$SYS_WATCHDOG_CHECK_ROOTFS log=$SYS_WATCHDOG_CHECK_LOG ram=$SYS_WATCHDOG_CHECK_RAM"
echo " arc=$SYS_WATCHDOG_CHECK_ARC cpu_temp=$SYS_WATCHDOG_CHECK_CPU_TEMP load=$SYS_WATCHDOG_CHECK_LOAD"
echo " zombies=$SYS_WATCHDOG_CHECK_ZOMBIES docker=$SYS_WATCHDOG_CHECK_DOCKER_DAEMON"
echo " containers=$SYS_WATCHDOG_CHECK_CONTAINERS oom=$SYS_WATCHDOG_CHECK_OOM"
echo " tmp=$SYS_WATCHDOG_CHECK_TMP fd=$SYS_WATCHDOG_CHECK_FD boot=$SYS_WATCHDOG_CHECK_BOOT"
echo " kernel_oops=$SYS_WATCHDOG_CHECK_KERNEL_OOPS sshd=$SYS_WATCHDOG_CHECK_SSHD"
echo " network=$SYS_WATCHDOG_CHECK_NETWORK mdstat=$SYS_WATCHDOG_CHECK_MDSTAT"
echo " runaway=$SYS_WATCHDOG_CHECK_RUNAWAY"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ── STATE HELPERS ─────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
get_strikes() {
grep -E "^${1}:" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null | cut -d':' -f2
}
set_strikes() {
local key="$1" count="$2"
grep -vE "^${key}:" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null > "${SYS_WATCHDOG_STATE_FILE}.tmp"
echo "${key}:${count}" >> "${SYS_WATCHDOG_STATE_FILE}.tmp"
mv "${SYS_WATCHDOG_STATE_FILE}.tmp" "$SYS_WATCHDOG_STATE_FILE"
}
increment_strikes() {
local key="$1"
local current
current=$(get_strikes "$key")
[[ -z "$current" ]] && current=0
(( current++ ))
set_strikes "$key" "$current"
echo "$current"
}
reset_strikes() {
set_strikes "$1" 0
}
get_state_val() {
grep -E "^${1}=" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null | cut -d'=' -f2
}
set_state_val() {
local key="$1" val="$2"
grep -vE "^${key}=" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null > "${SYS_WATCHDOG_STATE_FILE}.tmp"
echo "${key}=${val}" >> "${SYS_WATCHDOG_STATE_FILE}.tmp"
mv "${SYS_WATCHDOG_STATE_FILE}.tmp" "$SYS_WATCHDOG_STATE_FILE"
}
purge_old_reboots() {
local now cutoff
now=$(date +%s)
cutoff=$(( now - SYS_WATCHDOG_REBOOT_WINDOW ))
grep -v "^$" "$SYS_WATCHDOG_REBOOT_LOG" 2>/dev/null | while IFS= read -r ts; do
[[ "$ts" -gt "$cutoff" ]] && echo "$ts"
done > "${SYS_WATCHDOG_REBOOT_LOG}.tmp"
mv "${SYS_WATCHDOG_REBOOT_LOG}.tmp" "$SYS_WATCHDOG_REBOOT_LOG"
}
count_recent_reboots() {
purge_old_reboots
grep -c "." "$SYS_WATCHDOG_REBOOT_LOG" 2>/dev/null || echo 0
}
log_reboot() {
date +%s >> "$SYS_WATCHDOG_REBOOT_LOG"
}
# ==============================================================================================
# ── OOM TRACKING ──────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Reads /proc/vmstat oom_kill counter — delta per cycle = rate of OOM kills
# Used for Tier 2 bypass and diagnostic context in reboot messages
get_oom_delta() {
local current_oom
current_oom=$(grep "^oom_kill " /proc/vmstat 2>/dev/null | awk '{print $2}')
[[ -z "$current_oom" ]] && echo 0 && return
local prev_oom
prev_oom=$(cat "$SYS_WATCHDOG_OOM_FILE" 2>/dev/null || echo 0)
echo "$current_oom" > "$SYS_WATCHDOG_OOM_FILE"
local delta=$(( current_oom - prev_oom ))
[[ "$delta" -lt 0 ]] && delta=0 # counter reset on reboot
echo "$delta"
}
get_oom_victims() {
# Get process names from dmesg that were OOM killed this boot
dmesg -T 2>/dev/null | grep -i "Killed process" | \
awk '{print $NF}' | sort | uniq -c | sort -rn | head -5 | \
awk '{printf "%s×%d ", $2, $1}' | sed 's/ $//'
}
# ==============================================================================================
# ── ABORT CONDITIONS ──────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Returns 1 if reboot should be aborted, 0 if reboot should proceed
# CRITICAL tier bypasses this function entirely
check_abort_conditions() {
local should_abort=false
if command -v zpool >/dev/null 2>&1; then
local unhealthy
unhealthy=$(zpool list -H -o health 2>/dev/null | grep -v ONLINE || true)
if [[ -n "$unhealthy" ]]; then
if [[ "$SYS_WATCHDOG_ABORT_ON_ZFS_UNHEALTHY" == true ]]; then
error "ZFS pool unhealthy — aborting reboot to prevent data loss"
notify "System watchdog aborted reboot on $(hostname) ($MY_ID) — ZFS pool unhealthy" \
"System Watchdog" "warning"
should_abort=true
else
warn "ZFS pool unhealthy — continuing reboot (ABORT_ON_ZFS_UNHEALTHY=false)"
fi
fi
fi
if grep -q "progress" /var/local/emhttp/parity-date.txt 2>/dev/null; then
if [[ "$SYS_WATCHDOG_ABORT_ON_PARITY" == true ]]; then
error "Parity check running — aborting reboot"
notify "System watchdog aborted reboot on $(hostname) ($MY_ID) — parity running" \
"System Watchdog" "warning"
should_abort=true
else
warn "Parity check running — continuing reboot (ABORT_ON_PARITY=false)"
fi
fi
if pgrep -f "mover" >/dev/null 2>&1; then
if [[ "$SYS_WATCHDOG_ABORT_ON_MOVER" == true ]]; then
error "Mover running — aborting reboot"
notify "System watchdog aborted reboot on $(hostname) ($MY_ID) — mover running" \
"System Watchdog" "warning"
should_abort=true
else
warn "Mover running — continuing reboot (ABORT_ON_MOVER=false)"
fi
fi
[[ "$should_abort" == true ]] && return 1
return 0
}
# ==============================================================================================
# ── STANDARD STRIKE CHECK ─────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Returns 0 = reboot now | 1 = not yet
run_strike_check() {
local key="$1" triggered="$2" description="$3"
if [[ "$triggered" == true ]]; then
local strikes
strikes=$(increment_strikes "$key")
warn "$description — strike $strikes/$SYS_WATCHDOG_STRIKE_LIMIT"
if (( strikes >= SYS_WATCHDOG_STRIKE_LIMIT )); then
error "$description — strike limit hit, reboot triggered"
reset_strikes "$key"
return 0
fi
else
local current
current=$(get_strikes "$key")
[[ -n "$current" && "$current" -gt 0 ]] && reset_strikes "$key"
fi
return 1
}
# ==============================================================================================
# ── CONTAINER SHUTDOWN (RAM EMERGENCY) ────────────────────────────────────────────────────────
# ==============================================================================================
shutdown_non_essential_containers() {
warn "RAM emergency — stopping non-essential containers"
local stopped=()
# Build exclusion map
declare -A EXCLUDED_MAP
for exc in "${SYS_WATCHDOG_MEM_SHUTDOWN_EXCLUDED[@]:-}"; do
[[ -n "$exc" ]] && EXCLUDED_MAP["$exc"]=1
done
# Stop all running containers not in exclusion list
while IFS= read -r container; do
[[ -z "$container" ]] && continue
if [[ -n "${EXCLUDED_MAP[$container]:-}" ]]; then
log "$container — excluded from RAM shutdown, leaving running"
continue
fi
if [[ "$DRY_RUN" == false ]]; then
timeout "$DOCKER_TIMEOUT" docker stop "$container" >/dev/null 2>&1 && \
warn "Stopped $container (RAM emergency)" && \
stopped+=("$container") || \
error "Failed to stop $container"
else
warn "DRY RUN — would stop $container (RAM emergency)"
stopped+=("$container")
fi
done < <(timeout "$DOCKER_TIMEOUT" docker ps --format "{{.Names}}" 2>/dev/null)
if [[ ${#stopped[@]} -gt 0 ]]; then
set_state_val "mem_shutdown_active" "true"
notify "RAM emergency on $(hostname) ($MY_ID) — stopped ${#stopped[@]} containers. Excluded: ${SYS_WATCHDOG_MEM_SHUTDOWN_EXCLUDED[*]}" \
"System Watchdog" "warning"
warn "Stopped ${#stopped[@]} containers — waiting for RAM to recover above ${SYS_WATCHDOG_MEM_RECOVER_GB}GB"
fi
}
restart_non_essential_containers() {
warn "RAM recovered — restarting containers that were stopped in emergency"
if [[ "$DRY_RUN" == false ]]; then
set_state_val "mem_shutdown_active" "false"
fi
# docker_watchdog.sh will detect stopped required containers and restart them
# We just clear the state flag here
warn "Cleared RAM emergency state — docker_watchdog.sh will restart required containers"
}
# ==============================================================================================
# ── DO REBOOT ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# tier: "critical" (bypass abort) | "urgent" | "standard"
do_reboot() {
local tier="${1:-standard}"
shift
local triggers=("$@")
# Get OOM context for reboot message
local oom_victims=""
if [[ "$SYS_WATCHDOG_CHECK_OOM" == true ]]; then
oom_victims=$(get_oom_victims)
[[ -n "$oom_victims" ]] && triggers+=("oom_victims: $oom_victims")
fi
# Abort check — CRITICAL bypasses this
if [[ "$tier" != "critical" ]]; then
if ! check_abort_conditions; then
return
fi
else
warn "CRITICAL tier — bypassing abort conditions"
fi
RECENT_REBOOTS=$(count_recent_reboots)
log "Recent reboots in ${SYS_WATCHDOG_REBOOT_WINDOW_HRS}hr window: $RECENT_REBOOTS / $SYS_WATCHDOG_REBOOT_LIMIT"
if [[ "$RECENT_REBOOTS" -ge "$SYS_WATCHDOG_REBOOT_LIMIT" ]]; then
error "Reboot loop detected — shutting down instead of rebooting"
notify "Reboot loop on $(hostname) ($MY_ID) — shutting down after $RECENT_REBOOTS reboots — ${triggers[*]}" \
"System Watchdog" "warning"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would shutdown now"
return
fi
sync
/sbin/poweroff
return
fi
echo ""
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo " $ICON_REBOOT_SMART SYSTEM WATCHDOG — REBOOT TRIGGERED"
echo " Tier: ${tier^^}"
echo " Host: $MY_ID ($LOCAL_SERVER_NAME)"
for t in "${triggers[@]}"; do
echo " → $t"
done
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
notify "System watchdog ${tier^^} reboot on $(hostname) ($MY_ID) — ${triggers[*]}" \
"System Watchdog" "warning"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — reboot sequence would begin now"
return
fi
log_reboot
# Graceful shutdown sequence
warn "Shutting down VMs..."
if command -v virsh >/dev/null 2>&1; then
for VM in $(virsh list --name 2>/dev/null); do
[[ -z "$VM" ]] && continue
virsh shutdown "$VM" >/dev/null 2>&1
done
sleep 30
fi
warn "Stopping Docker containers..."
if command -v docker >/dev/null 2>&1; then
timeout 60 docker ps -q 2>/dev/null | xargs -r docker stop >/dev/null 2>&1
fi
warn "Stopping User Scripts..."
pkill -f "/tmp/user.scripts" 2>/dev/null || true
warn "Syncing disks..."
sync
sleep 5
/sbin/reboot
}
# ==============================================================================================
# ━━━ Clean Shutdown ━━━
# ==============================================================================================
WATCHDOG_RUNNING=true
cleanup() {
echo ""
warn "System watchdog received shutdown signal — stopping cleanly"
WATCHDOG_RUNNING=false
exit 0
}
trap cleanup SIGTERM SIGINT
# ==============================================================================================
# ━━━ Continuous Monitoring Loop ━━━
# ==============================================================================================
warn "System watchdog started — $MY_ID — checking every ${SYSTEM_WATCHDOG_INTERVAL}s"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
CYCLE=0
while [[ "$WATCHDOG_RUNNING" == true ]]; do
(( CYCLE++ ))
# Re-source config each cycle — picks up config changes without restart
source "$SCRIPT_DIR/../load_config.sh"
detect_hosts
SYS_WATCHDOG_REBOOT_WINDOW=$(( SYS_WATCHDOG_REBOOT_WINDOW_HRS * 3600 ))
TOTAL_CORES=$(nproc)
TRIGGERS=()
CRITICAL_TRIGGERS=()
URGENT_OOM_CONFIRMED=false
# ── OOM Delta — read every cycle for bypass decisions ─────────────────────────────────────
OOM_DELTA=0
if [[ "$SYS_WATCHDOG_CHECK_OOM" == true ]]; then
OOM_DELTA=$(get_oom_delta)
[[ "$OOM_DELTA" -gt 0 ]] && \
log "OOM kills this cycle: $OOM_DELTA (limit: ${SYS_WATCHDOG_OOM_LIMIT})"
fi
# ==========================================================================================
# ━━━ TIER 1 — CRITICAL CHECKS (bypass all strikes, reboot immediately) ━━━
# ==========================================================================================
# ── Docker daemon — critical: nothing can heal without it ─────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_DOCKER_DAEMON" == true ]]; then
if ! timeout "$DOCKER_TIMEOUT" docker info >/dev/null 2>&1; then
error "Docker daemon unresponsive — CRITICAL"
# Attempt daemon restart before rebooting
warn "Attempting Docker daemon restart..."
if [[ "$DRY_RUN" == false ]]; then
/etc/rc.d/rc.docker restart >/dev/null 2>&1
sleep 15
if timeout "$DOCKER_TIMEOUT" docker info >/dev/null 2>&1; then
warn "Docker daemon restarted successfully — continuing monitoring"
else
error "Docker daemon restart failed — adding to CRITICAL triggers"
CRITICAL_TRIGGERS+=("docker_daemon_unresponsive")
fi
else
warn "DRY RUN — would attempt Docker daemon restart"
CRITICAL_TRIGGERS+=("docker_daemon_unresponsive")
fi
else
log "Docker daemon healthy ✅"
fi
fi
# ── rootfs critical — at 99%+ writes are failing ─────────────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_ROOTFS" == true ]]; then
ROOTFS_USED=$(df / --output=pcent 2>/dev/null | tail -1 | tr -d ' %')
if [[ "$ROOTFS_USED" -ge "${SYS_WATCHDOG_ROOTFS_CRITICAL_PCT:-99}" ]]; then
error "rootfs ${ROOTFS_USED}% — CRITICAL (writes failing)"
CRITICAL_TRIGGERS+=("rootfs_full=${ROOTFS_USED}%")
fi
fi
# ── Kernel oops/BUG — kernel running with corrupted state ────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_KERNEL_OOPS" == true ]]; then
PREV_OOPS=$(get_state_val "kernel_oops_count")
CURRENT_OOPS=$(dmesg 2>/dev/null | grep -cE "BUG:|kernel BUG|Oops:" || echo 0)
CURRENT_OOPS="${CURRENT_OOPS//[^0-9]/}"; CURRENT_OOPS="${CURRENT_OOPS:-0}"
set_state_val "kernel_oops_count" "$CURRENT_OOPS"
if [[ -n "$PREV_OOPS" && "$PREV_OOPS" =~ ^[0-9]+$ ]]; then
OOPS_DELTA=$(( CURRENT_OOPS - PREV_OOPS ))
if [[ "$OOPS_DELTA" -gt 0 ]]; then
error "Kernel oops/BUG detected — $OOPS_DELTA new since last cycle — CRITICAL"
CRITICAL_TRIGGERS+=("kernel_oops=${OOPS_DELTA}_new")
fi
fi
fi
# ── File descriptor exhaustion — new connections failing silently ─────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_FD" == true ]]; then
FD_LINE=$(cat /proc/sys/fs/file-nr 2>/dev/null)
FD_OPEN=$(echo "$FD_LINE" | awk '{print $1}')
FD_MAX=$(echo "$FD_LINE" | awk '{print $3}')
if [[ -n "$FD_OPEN" && -n "$FD_MAX" && "$FD_MAX" -gt 0 ]]; then
FD_PCT=$(( FD_OPEN * 100 / FD_MAX ))
if [[ "$FD_PCT" -ge "${SYS_WATCHDOG_FD_CRITICAL_PCT:-95}" ]]; then
error "File descriptors ${FD_PCT}% exhausted (${FD_OPEN}/${FD_MAX}) — CRITICAL"
CRITICAL_TRIGGERS+=("fd_exhaustion=${FD_PCT}%")
else
log "File descriptors: ${FD_PCT}% (${FD_OPEN}/${FD_MAX})"
fi
fi
fi
# ── /boot read-only — state and config writes failing silently ────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_BOOT" == true ]]; then
BOOT_TEST="/boot/.watchdog_write_test"
if ! touch "$BOOT_TEST" 2>/dev/null; then
error "/boot is read-only — config writes failing silently — CRITICAL"
CRITICAL_TRIGGERS+=("boot_read_only")
else
rm -f "$BOOT_TEST" 2>/dev/null
log "/boot is writable ✅"
fi
fi
# ── Act on CRITICAL triggers immediately ─────────────────────────────────────────────────
if [[ ${#CRITICAL_TRIGGERS[@]} -gt 0 ]]; then
echo ""
echo "━━━ $ICON_ERROR CRITICAL — IMMEDIATE REBOOT — Cycle $CYCLE ━━━"
for t in "${CRITICAL_TRIGGERS[@]}"; do
error " CRITICAL: $t"
done
do_reboot "critical" "${CRITICAL_TRIGGERS[@]}"
continue
fi
# ==========================================================================================
# ━━━ TIER 3 — STANDARD CHECKS (strike system) ━━━
# ==========================================================================================
# ── rootfs standard ──────────────────────────────────────────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_ROOTFS" == true ]]; then
ROOTFS_USED=$(df / --output=pcent 2>/dev/null | tail -1 | tr -d ' %')
TRIGGERED=false
[[ "$ROOTFS_USED" -ge "$SYS_WATCHDOG_ROOTFS_PCT" ]] && TRIGGERED=true
run_strike_check "rootfs" "$TRIGGERED" "rootfs ${ROOTFS_USED}%" && \
TRIGGERS+=("rootfs=${ROOTFS_USED}%")
fi
# ── /var/log ─────────────────────────────────────────────────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_LOG" == true ]]; then
LOG_USED=$(df -P /var/log 2>/dev/null | awk 'NR==2 {print $5}' | tr -d '%')
TRIGGERED=false
[[ "${LOG_USED:-0}" -ge "$SYS_WATCHDOG_LOG_PCT" ]] && TRIGGERED=true
run_strike_check "log" "$TRIGGERED" "/var/log ${LOG_USED}%" && \
TRIGGERS+=("log=${LOG_USED}%")
fi
# ── /tmp ─────────────────────────────────────────────────────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_TMP" == true ]]; then
TMP_USED=$(df -P /tmp 2>/dev/null | awk 'NR==2 {print $5}' | tr -d '%')
if [[ "${TMP_USED:-0}" -ge "${SYS_WATCHDOG_TMP_CRITICAL_PCT:-98}" ]]; then
# Try to clear before escalating
warn "/tmp ${TMP_USED}% — attempting cleanup..."
find /tmp -type f -mmin +60 -not -name "*.lock" -delete 2>/dev/null
TMP_USED_AFTER=$(df -P /tmp 2>/dev/null | awk 'NR==2 {print $5}' | tr -d '%')
if [[ "${TMP_USED_AFTER:-0}" -ge "${SYS_WATCHDOG_TMP_CRITICAL_PCT:-98}" ]]; then
error "/tmp still ${TMP_USED_AFTER}% after cleanup — adding to triggers"
TRIGGERED=true
else
warn "/tmp cleared to ${TMP_USED_AFTER}% ✅"
TRIGGERED=false
fi
elif [[ "${TMP_USED:-0}" -ge "${SYS_WATCHDOG_TMP_PCT:-90}" ]]; then
TRIGGERED=true
else
TRIGGERED=false
fi
run_strike_check "tmp" "$TRIGGERED" "/tmp ${TMP_USED}%" && \
TRIGGERS+=("tmp=${TMP_USED}%")
fi
# ── RAM tiers ─────────────────────────────────────────────────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_RAM" == true ]]; then
MEM_KB=$(awk '/MemAvailable/ {print $2}' /proc/meminfo)
MEM_GB=$(( MEM_KB / 1024 / 1024 ))
MEM_SHUTDOWN_ACTIVE=$(get_state_val "mem_shutdown_active")
if [[ "$MEM_GB" -lt "$SYS_WATCHDOG_MEM_GB" ]]; then
# Tier 2 check — bypass if OOM confirms crisis
if [[ "$SYS_WATCHDOG_CHECK_OOM" == true ]] && \
[[ "$OOM_DELTA" -ge "$SYS_WATCHDOG_OOM_LIMIT" ]]; then
error "RAM ${MEM_GB}GB + ${OOM_DELTA} OOM kills this cycle — URGENT bypass"
OOM_VICTIMS=$(get_oom_victims)
URGENT_TRIGGERS=("urgent_low_ram=${MEM_GB}GB" "oom_kills=${OOM_DELTA}")
[[ -n "$OOM_VICTIMS" ]] && URGENT_TRIGGERS+=("oom_victims: $OOM_VICTIMS")
do_reboot "urgent" "${URGENT_TRIGGERS[@]}"
continue
fi
# Standard strike path
run_strike_check "ram" true "RAM ${MEM_GB}GB free" && \
TRIGGERS+=("low_ram=${MEM_GB}GB")
elif [[ "$MEM_GB" -lt "$SYS_WATCHDOG_MEM_SHUTDOWN_GB" ]]; then
reset_strikes "ram"
# Container shutdown tier — but only once per event
if [[ "$MEM_SHUTDOWN_ACTIVE" != "true" ]]; then
warn "RAM ${MEM_GB}GB — below shutdown threshold ${SYS_WATCHDOG_MEM_SHUTDOWN_GB}GB"
run_strike_check "ram_shutdown" true "RAM shutdown tier ${MEM_GB}GB" && \
shutdown_non_essential_containers
else
# Already shutdown — check if recovered
if [[ "$MEM_GB" -ge "$SYS_WATCHDOG_MEM_RECOVER_GB" ]]; then
warn "RAM recovered to ${MEM_GB}GB — clearing emergency state"
restart_non_essential_containers
reset_strikes "ram_shutdown"
else
warn "RAM ${MEM_GB}GB — still in emergency shutdown (recover threshold: ${SYS_WATCHDOG_MEM_RECOVER_GB}GB)"
fi
fi
elif [[ "$MEM_GB" -lt "$SYS_WATCHDOG_MEM_WARN_GB" ]]; then
reset_strikes "ram"
reset_strikes "ram_shutdown"
warn "RAM ${MEM_GB}GB — below warning threshold ${SYS_WATCHDOG_MEM_WARN_GB}GB"
local prev_ram_warn
prev_ram_warn=$(get_strikes "ram_warn_notified")
if [[ "${prev_ram_warn:-0}" -eq 0 ]]; then
notify "RAM warning on $(hostname) ($MY_ID) — ${MEM_GB}GB free (threshold: ${SYS_WATCHDOG_MEM_WARN_GB}GB)" \
"System Watchdog" "warning"
set_strikes "ram_warn_notified" 1
fi
else
reset_strikes "ram"
reset_strikes "ram_shutdown"
set_strikes "ram_warn_notified" 0
log "RAM ${MEM_GB}GB free ✅"
fi
fi
# ── ZFS ARC ──────────────────────────────────────────────────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_ARC" == true ]] && [[ -f /proc/spl/kstat/zfs/arcstats ]]; then
ARC_SIZE=$(awk '/^size / {print $3}' /proc/spl/kstat/zfs/arcstats)
ARC_MAX=$(awk '/^c_max / {print $3}' /proc/spl/kstat/zfs/arcstats)
ARC_PCT=$(( ARC_SIZE * 100 / ARC_MAX ))
TRIGGERED=false
if [[ "$ARC_PCT" -ge "$SYS_WATCHDOG_ARC_PINNED_PCT" ]]; then
sync; echo 3 > /proc/sys/vm/drop_caches; sleep 5
ARC_AFTER=$(awk '/^size / {print $3}' /proc/spl/kstat/zfs/arcstats)
ARC_AFTER_PCT=$(( ARC_AFTER * 100 / ARC_MAX ))
[[ "$ARC_AFTER_PCT" -ge "$SYS_WATCHDOG_ARC_RELEASE_PCT" ]] && TRIGGERED=true
fi
run_strike_check "arc" "$TRIGGERED" "ZFS ARC pinned ${ARC_PCT}%" && \
TRIGGERS+=("arc_pinned=${ARC_PCT}%")
fi
# ── CPU temperature ───────────────────────────────────────────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_CPU_TEMP" == true ]]; then
CPU_TEMP=""
if command -v sensors >/dev/null 2>&1; then
CPU_TEMP=$(sensors 2>/dev/null | \
grep -i "Package id 0\|Tctl\|CPU Temp" | \
awk '{print $NF}' | tr -d '+°C' | head -1)
fi
if [[ -n "$CPU_TEMP" ]]; then
CPU_TEMP_INT=$(printf "%.0f" "$CPU_TEMP")
TRIGGERED=false
[[ "$CPU_TEMP_INT" -ge "$SYS_WATCHDOG_CPU_TEMP_MAX" ]] && TRIGGERED=true
run_strike_check "cpu_temp" "$TRIGGERED" "CPU temp ${CPU_TEMP_INT}°C" && \
TRIGGERS+=("cpu_temp=${CPU_TEMP_INT}C")
fi
fi
# ── Load average ─────────────────────────────────────────────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_LOAD" == true ]]; then
LOAD=$(awk '{print $1}' /proc/loadavg)
LOAD_INT=$(printf "%.0f" "$LOAD")
LOAD_THRESHOLD=$(( TOTAL_CORES * SYS_WATCHDOG_LOAD_MULTIPLIER ))
TRIGGERED=false
[[ "$LOAD_INT" -ge "$LOAD_THRESHOLD" ]] && TRIGGERED=true
run_strike_check "load" "$TRIGGERED" "load avg ${LOAD}" && \
TRIGGERS+=("load=${LOAD}")
fi
# ── Zombie processes ─────────────────────────────────────────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_ZOMBIES" == true ]]; then
ZOMBIE_COUNT=$(ps aux 2>/dev/null | awk '{print $8}' | grep -c "^Z$" || echo 0)
ZOMBIE_COUNT="${ZOMBIE_COUNT//[^0-9]/}"; ZOMBIE_COUNT="${ZOMBIE_COUNT:-0}"
TRIGGERED=false
[[ "$ZOMBIE_COUNT" -ge "$SYS_WATCHDOG_ZOMBIE_LIMIT" ]] && TRIGGERED=true
run_strike_check "zombies" "$TRIGGERED" "zombies ${ZOMBIE_COUNT}" && \
TRIGGERS+=("zombies=${ZOMBIE_COUNT}")
fi
# ── Array disk errors — accumulating mdstat errors ────────────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_MDSTAT" == true ]]; then
PREV_MD_ERRORS=$(get_state_val "mdstat_errors")
CURRENT_MD_ERRORS=$(grep -oP "(?<=\[)[^\]]*[U_][^\]]*(?=\])" \
/proc/mdstat 2>/dev/null | grep -o "_" | wc -l || echo 0)
CURRENT_MD_ERRORS="${CURRENT_MD_ERRORS//[^0-9]/}"; CURRENT_MD_ERRORS="${CURRENT_MD_ERRORS:-0}"
set_state_val "mdstat_errors" "$CURRENT_MD_ERRORS"
if [[ -n "$PREV_MD_ERRORS" && "$PREV_MD_ERRORS" =~ ^[0-9]+$ ]]; then
MD_DELTA=$(( CURRENT_MD_ERRORS - PREV_MD_ERRORS ))
if [[ "$MD_DELTA" -ge "${SYS_WATCHDOG_MDSTAT_ERROR_LIMIT:-5}" ]]; then
TRIGGERED=true
run_strike_check "mdstat" "$TRIGGERED" \
"mdstat errors +${MD_DELTA} (total: ${CURRENT_MD_ERRORS})" && \
TRIGGERS+=("mdstat_errors=+${MD_DELTA}")
else
run_strike_check "mdstat" false "mdstat" > /dev/null
fi
fi
fi
# ── Network interface state ───────────────────────────────────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_NETWORK" == true ]]; then
NIC="${SYS_WATCHDOG_NIC:-eth0}"
NIC_STATE=$(cat "/sys/class/net/${NIC}/operstate" 2>/dev/null || echo "unknown")
TRIGGERED=false
[[ "$NIC_STATE" != "up" ]] && TRIGGERED=true
run_strike_check "network" "$TRIGGERED" "${NIC} state: ${NIC_STATE}" && \
TRIGGERS+=("nic_down=${NIC}")
fi
# ── sshd — try restart before escalating ─────────────────────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_SSHD" == true ]]; then
if ! pgrep -x sshd >/dev/null 2>&1; then
warn "sshd not running — attempting restart..."
if [[ "$DRY_RUN" == false ]]; then
/etc/rc.d/rc.sshd start >/dev/null 2>&1
sleep 3
if pgrep -x sshd >/dev/null 2>&1; then
warn "sshd restarted successfully ✅"
reset_strikes "sshd"
notify "sshd was down on $(hostname) ($MY_ID) — restarted automatically" \
"System Watchdog" "warning"
else
error "sshd restart failed — remote access unavailable"
run_strike_check "sshd" true "sshd not running" && \
TRIGGERS+=("sshd_down")
fi
else
warn "DRY RUN — would restart sshd"
fi
else
reset_strikes "sshd"
fi
fi
# ── Runaway process ───────────────────────────────────────────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_RUNAWAY" == true ]]; then
RUNAWAY_PCT="${SYS_WATCHDOG_RUNAWAY_CPU_PCT:-90}"
TOP_CPU_PCT=$(ps aux 2>/dev/null | awk 'NR>1{print $3}' | sort -rn | head -1)
TOP_CPU_INT=$(printf "%.0f" "${TOP_CPU_PCT:-0}")
TOP_CPU_NAME=$(ps aux 2>/dev/null | sort -k3 -rn | awk 'NR==2{print $11}')
TRIGGERED=false
[[ "$TOP_CPU_INT" -ge "$RUNAWAY_PCT" ]] && TRIGGERED=true
# Runaway uses SYS_WATCHDOG_RUNAWAY_STRIKES not global strike limit
if [[ "$TRIGGERED" == true ]]; then
RAWAY_S=$(increment_strikes "runaway")
RLIMIT="${SYS_WATCHDOG_RUNAWAY_STRIKES:-3}"
warn "Runaway ${TOP_CPU_NAME} ${TOP_CPU_PCT}% CPU -- strike $RAWAY_S/$RLIMIT"
if (( RAWAY_S >= RLIMIT )); then
error "Runaway process ${TOP_CPU_NAME} -- strike limit hit"
reset_strikes "runaway"
TRIGGERS+=("runaway=${TOP_CPU_NAME}@${TOP_CPU_PCT}%")
fi
else
RAWAY_CUR=$(get_strikes "runaway")
[[ "${RAWAY_CUR:-0}" -gt 0 ]] && reset_strikes "runaway"
fi
fi
# ── Required containers from docker_watchdog skip list ────────────────────────────────────
if [[ "$SYS_WATCHDOG_CHECK_CONTAINERS" == true ]] && [[ -s "$SYS_WATCHDOG_FAILED_FILE" ]]; then
FAILED_CONTAINERS=()
while IFS= read -r container; do
[[ -z "$container" ]] && continue
STATUS=$(timeout "$DOCKER_TIMEOUT" docker inspect -f \
'{{.State.Running}}' "$container" 2>/dev/null || echo "unknown")
[[ "$STATUS" != "true" ]] && FAILED_CONTAINERS+=("$container")
done < "$SYS_WATCHDOG_FAILED_FILE"
TRIGGERED=false
[[ ${#FAILED_CONTAINERS[@]} -gt 0 ]] && TRIGGERED=true
run_strike_check "failed_containers" "$TRIGGERED" \
"required containers stopped: ${FAILED_CONTAINERS[*]:-}" && \
TRIGGERS+=("containers=${FAILED_CONTAINERS[*]:-}")
fi
# ==========================================================================================
# ━━━ Evaluate Standard Triggers ━━━
# ==========================================================================================
if [[ ${#TRIGGERS[@]} -gt 0 ]]; then
echo ""
echo "━━━ $ICON_REBOOT_SMART System Watchdog — Cycle $CYCLE$(date '+%Y-%m-%d %H:%M:%S') ━━━"
for t in "${TRIGGERS[@]}"; do
echo " $ICON_REBOOT_SMART $t"
done
[[ "$OOM_DELTA" -gt 0 ]] && echo " OOM kills this cycle: $OOM_DELTA"
echo ""
do_reboot "standard" "${TRIGGERS[@]}"
else
log "Cycle $CYCLE — system healthy ($(date '+%H:%M:%S'))"
# Heartbeat — periodic proof of life
if [[ "${SYSTEM_WATCHDOG_HEARTBEAT:-true}" == true ]]; then
HB_SECONDS=$(( ${SYSTEM_WATCHDOG_HEARTBEAT_HOURS:-1} * 3600 ))
UPTIME_APPROX=$(( CYCLE * SYSTEM_WATCHDOG_INTERVAL ))
if [[ "$HB_SECONDS" -gt 0 ]] && \
(( UPTIME_APPROX % HB_SECONDS < SYSTEM_WATCHDOG_INTERVAL )) && \
[[ "$UPTIME_APPROX" -gt 0 ]]; then
HB_HR=$(( UPTIME_APPROX / 3600 ))
warn "♥ system_watchdog alive — $MY_ID — ~${HB_HR}hr uptime ($(date '+%H:%M:%S'))"
fi
fi
fi
# ── State file heartbeat — keep mtime fresh every cycle ───────────────────────────────────
# docker_watchdog.sh uses state file mtime to detect stale RAM emergency flags.
# If all checks pass with no set_state_val calls (e.g. KERNEL_OOPS + MDSTAT both disabled),
# mtime would not update and stale guard would incorrectly resume docker_watchdog.sh.
# Writing watchdog_cycle each tick guarantees mtime stays current while watchdog runs.
set_state_val "watchdog_cycle" "$CYCLE"
# Sleep until next cycle — interruptible by SIGTERM
sleep "$SYSTEM_WATCHDOG_INTERVAL" &
wait $!
done