The stability_watchdog.sh script was being called "System Watchdog" everywhere, conflicting with system_watchdog.sh (which orchestrates storage/webgui/network sub-watchdogs). Fixes: - --status header: "SYSTEM WATCHDOG STATUS" → "STABILITY WATCHDOG STATUS" - reboot banner: "SYSTEM WATCHDOG — REBOOT TRIGGERED" → "STABILITY WATCHDOG" - Interval line: replace undefined SYSTEM_WATCHDOG_INTERVAL variable with hardcoded "60s (cron — every minute)" - --status check toggles: remove containers= (dead config — container check was removed from stability_watchdog in a prior refactor) Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
790 lines
39 KiB
Bash
Executable File
790 lines
39 KiB
Bash
Executable File
#!/bin/bash
|
||
# ==============================================================================================
|
||
# ================================= System Watchdog ============================================
|
||
# ==============================================================================================
|
||
#
|
||
# PURPOSE
|
||
# ─────────────────────────────────────────────────────────────────────────────
|
||
# Last line of defense — reboots the system cleanly if it is about to become
|
||
# unstable. Runs continuously as a background process started by
|
||
# array_started.sh at array start. Works alongside docker_watchdog.sh which
|
||
# handles container-level healing first. Only escalates to reboot when
|
||
# docker_watchdog.sh cannot resolve the condition.
|
||
#
|
||
# ==============================================================================================
|
||
# OPERATIONAL MODEL
|
||
# ==============================================================================================
|
||
#
|
||
# Three-Tier Response System
|
||
#
|
||
# Tier 1 — CRITICAL (bypass all strikes, reboot immediately)
|
||
# rootfs at 99%+ — writes failing; SSH may stop; no recovery options
|
||
# Kernel oops/BUG in dmesg — kernel running with corrupted state
|
||
# File descriptor exhaustion — new connections and processes failing silently
|
||
# /boot read-only unexpectedly — state files and config writes silently failing
|
||
# (Docker daemon: owned by docker_watchdog — writes daemon_confirmed_down flag → standard strikes)
|
||
#
|
||
# Tier 2 — URGENT (bypass strikes when OOM confirms active crisis)
|
||
# RAM < MEM_GB AND OOM kills >= OOM_LIMIT in this cycle.
|
||
# OOM kills at this rate means the system is dying faster than watchdogs can heal.
|
||
# Without OOM confirmation → standard strike system applies.
|
||
#
|
||
# Tier 3 — STANDARD (N consecutive failures → reboot)
|
||
# RAM tiers, load, CPU temp, zombies, /var/log, /tmp, containers, NIC, mdstat.
|
||
#
|
||
# RAM Tiers
|
||
# MEM_WARN_GB (10GB) — warn + notify only
|
||
# MEM_SHUTDOWN_GB (6GB) — stop non-essential containers, wait for recovery
|
||
# MEM_GB (4GB) — strike system → reboot (bypass with OOM confirmation)
|
||
# MEM_RECOVER_GB (30GB) — RAM must reach this before stopped containers restart
|
||
#
|
||
# Container Shutdown Logic (at MEM_SHUTDOWN_GB)
|
||
# Stops all containers not in SYS_WATCHDOG_MEM_SHUTDOWN_EXCLUDED.
|
||
# Stopped containers tracked in shutdown list — won't restart until RAM recovers.
|
||
# Strike system prevents flip-flopping — shutdown only once per degradation event.
|
||
#
|
||
# Abort Conditions (prevent reboot during sensitive operations)
|
||
# ZFS pool unhealthy, parity running, mover running — each toggleable.
|
||
# CRITICAL tier bypasses all abort conditions — imminent crash overrides data safety.
|
||
#
|
||
# Checks Run Every Cycle
|
||
# rootfs usage, /var/log, /tmp, free RAM, ZFS ARC, CPU temp, load avg,
|
||
# zombie processes, Docker daemon, OOM rate, /boot read-only, kernel oops,
|
||
# file descriptor exhaustion, array disk errors, NIC state.
|
||
# (Container health is owned by docker_watchdog — not checked here.)
|
||
#
|
||
# ==============================================================================================
|
||
# OPERATIONAL SAFEGUARDS
|
||
# ==============================================================================================
|
||
#
|
||
# Root Required
|
||
# Reboot and container stop require root.
|
||
#
|
||
# Single Instance Lock
|
||
# acquire_lock prevents a second watchdog instance from starting.
|
||
#
|
||
# State File Verification
|
||
# All state files verified writable at startup — errors if any cannot be created.
|
||
#
|
||
# ==============================================================================================
|
||
# CONFIGURATION
|
||
# ==============================================================================================
|
||
#
|
||
# master.conf — System Watchdog section
|
||
# Full variable listing in master.conf. Key variables:
|
||
#
|
||
# SYS_WATCHDOG_REBOOT_WINDOW_HRS — reboot rate limit window (default: 2)
|
||
# SYS_WATCHDOG_MAX_REBOOTS — max reboots in window before giving up (default: 3)
|
||
# SYS_WATCHDOG_STRIKES — consecutive failures before reboot (default: 3)
|
||
# SYS_WATCHDOG_OOM_LIMIT — OOM kills/cycle to trigger URGENT bypass (default: 3)
|
||
# SYS_WATCHDOG_MEM_SHUTDOWN_EXCLUDED — containers exempt from memory shutdown
|
||
#
|
||
# ==============================================================================================
|
||
# STATE FILES
|
||
# ==============================================================================================
|
||
#
|
||
# SYS_WATCHDOG_STATE_FILE — strike counters and cycle state
|
||
# SYS_WATCHDOG_REBOOT_LOG — reboot history for rate limiting
|
||
# SYS_WATCHDOG_OOM_FILE — OOM kill counter from previous cycle
|
||
#
|
||
# ==============================================================================================
|
||
# RUNTIME MODES
|
||
# ==============================================================================================
|
||
#
|
||
# stability_watchdog.sh
|
||
# Start continuous monitoring loop. Runs until stopped or system reboots.
|
||
#
|
||
# stability_watchdog.sh --dry-run
|
||
# Run detection logic without rebooting or stopping containers.
|
||
#
|
||
# stability_watchdog.sh --status
|
||
# Show config, thresholds, current system state, and strike counts.
|
||
#
|
||
# stability_watchdog.sh --log
|
||
# Verbose per-cycle output — show every check result and threshold comparison.
|
||
#
|
||
# ==============================================================================================
|
||
|
||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||
|
||
source "$SCRIPT_DIR/../load_config.sh"
|
||
|
||
parse_args "$@"
|
||
|
||
# ==============================================================================================
|
||
# ━━━ Setup — runs once at start ━━━
|
||
# ==============================================================================================
|
||
if [[ "$EUID" -ne 0 ]]; then
|
||
error "Must be run as root"
|
||
exit 1
|
||
fi
|
||
|
||
validate_unraid_cmd \
|
||
"/usr/local/emhttp/plugins/dynamix/scripts/notify" \
|
||
"" "" \
|
||
"unRAID notify script" || warn "unRAID notify script not found — native notifications disabled"
|
||
|
||
acquire_lock
|
||
|
||
detect_hosts
|
||
|
||
TOTAL_CORES=$(nproc)
|
||
SYS_WATCHDOG_REBOOT_WINDOW=$(( SYS_WATCHDOG_REBOOT_WINDOW_HRS * 3600 ))
|
||
|
||
# Ensure state files exist
|
||
for state_file in "$SYS_WATCHDOG_STATE_FILE" "$SYS_WATCHDOG_REBOOT_LOG" \
|
||
"$SYS_WATCHDOG_OOM_FILE"; do
|
||
touch "$state_file" 2>/dev/null || {
|
||
error "Cannot create state file: $state_file"
|
||
exit 1
|
||
}
|
||
done
|
||
|
||
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no reboots or container shutdowns will occur"
|
||
|
||
# ==============================================================================================
|
||
# ━━━ Status ━━━
|
||
# ==============================================================================================
|
||
if [[ "$SHOW_STATUS" == true ]]; then
|
||
echo ""
|
||
echo "━━━━━ $ICON_SUMMARY STABILITY WATCHDOG STATUS ━━━━━"
|
||
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
|
||
echo ""
|
||
echo "── Tier 1 — CRITICAL (bypass strikes immediately) ──"
|
||
echo "$ICON_DISK rootfs critical: ${SYS_WATCHDOG_ROOTFS_CRITICAL_PCT}%"
|
||
echo "$ICON_GEAR FD critical: ${SYS_WATCHDOG_FD_CRITICAL_PCT}%"
|
||
echo "$ICON_GEAR /boot read-only: check=${SYS_WATCHDOG_CHECK_BOOT}"
|
||
echo "$ICON_GEAR Kernel oops: check=${SYS_WATCHDOG_CHECK_KERNEL_OOPS}"
|
||
echo "$ICON_CONTAINERS Docker daemon: check=${SYS_WATCHDOG_CHECK_DOCKER_DAEMON}"
|
||
echo ""
|
||
echo "── Tier 2 — URGENT (bypass strikes with OOM confirmation) ──"
|
||
echo "$ICON_MEM RAM critical: < ${SYS_WATCHDOG_MEM_GB}GB"
|
||
echo "$ICON_GEAR OOM limit: ${SYS_WATCHDOG_OOM_LIMIT} kills/cycle"
|
||
echo ""
|
||
echo "── Tier 3 — STANDARD (strike system) ──"
|
||
echo "$ICON_DISK rootfs warn: ${SYS_WATCHDOG_ROOTFS_PCT}%"
|
||
echo "$ICON_GEAR /var/log warn: ${SYS_WATCHDOG_LOG_PCT}%"
|
||
echo "$ICON_GEAR /tmp warn: ${SYS_WATCHDOG_TMP_PCT}%"
|
||
echo "$ICON_MEM RAM reboot: < ${SYS_WATCHDOG_MEM_GB}GB (+ strikes)"
|
||
echo " (RAM warn/shutdown/recover managed by resource_watchdog.sh)"
|
||
echo "$ICON_ZFS ARC pinned: ${SYS_WATCHDOG_ARC_PINNED_PCT}%"
|
||
echo "$ICON_GEAR Load multiplier: ${SYS_WATCHDOG_LOAD_MULTIPLIER}x (= $(( TOTAL_CORES * SYS_WATCHDOG_LOAD_MULTIPLIER )) on $TOTAL_CORES cores)"
|
||
echo "$ICON_GEAR Zombie limit: ${SYS_WATCHDOG_ZOMBIE_LIMIT}"
|
||
echo "$ICON_GEAR CPU temp max: ${SYS_WATCHDOG_CPU_TEMP_MAX}°C"
|
||
echo "$ICON_GEAR Strike limit: ${SYS_WATCHDOG_STRIKE_LIMIT} cycles"
|
||
echo "$ICON_TIME Interval: 60s (cron — every minute)"
|
||
echo "$ICON_REBOOT_SMART Reboot limit: ${SYS_WATCHDOG_REBOOT_LIMIT} in ${SYS_WATCHDOG_REBOOT_WINDOW_HRS}hr"
|
||
echo ""
|
||
echo ""
|
||
echo "── Check Toggles ──"
|
||
echo " rootfs=$SYS_WATCHDOG_CHECK_ROOTFS log=$SYS_WATCHDOG_CHECK_LOG ram=$SYS_WATCHDOG_CHECK_RAM"
|
||
echo " arc=$SYS_WATCHDOG_CHECK_ARC cpu_temp=$SYS_WATCHDOG_CHECK_CPU_TEMP load=$SYS_WATCHDOG_CHECK_LOAD"
|
||
echo " zombies=$SYS_WATCHDOG_CHECK_ZOMBIES docker=$SYS_WATCHDOG_CHECK_DOCKER_DAEMON"
|
||
echo " oom=$SYS_WATCHDOG_CHECK_OOM"
|
||
echo " tmp=$SYS_WATCHDOG_CHECK_TMP fd=$SYS_WATCHDOG_CHECK_FD boot=$SYS_WATCHDOG_CHECK_BOOT"
|
||
echo " kernel_oops=$SYS_WATCHDOG_CHECK_KERNEL_OOPS sshd=$SYS_WATCHDOG_CHECK_SSHD"
|
||
echo " network=$SYS_WATCHDOG_CHECK_NETWORK mdstat=$SYS_WATCHDOG_CHECK_MDSTAT"
|
||
echo " runaway=$SYS_WATCHDOG_CHECK_RUNAWAY"
|
||
echo "━━━━━━━━━━━━━━━━━━━━━━━"
|
||
exit 0
|
||
fi
|
||
|
||
# ==============================================================================================
|
||
# ── STATE HELPERS ─────────────────────────────────────────────────────────────────────────────
|
||
# ==============================================================================================
|
||
|
||
get_strikes() {
|
||
grep -E "^${1}:" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null | cut -d':' -f2
|
||
}
|
||
|
||
set_strikes() {
|
||
local key="$1" count="$2"
|
||
grep -vE "^${key}:" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null > "${SYS_WATCHDOG_STATE_FILE}.tmp"
|
||
echo "${key}:${count}" >> "${SYS_WATCHDOG_STATE_FILE}.tmp"
|
||
mv "${SYS_WATCHDOG_STATE_FILE}.tmp" "$SYS_WATCHDOG_STATE_FILE"
|
||
}
|
||
|
||
increment_strikes() {
|
||
local key="$1"
|
||
local current
|
||
current=$(get_strikes "$key")
|
||
[[ -z "$current" ]] && current=0
|
||
(( current++ ))
|
||
set_strikes "$key" "$current"
|
||
echo "$current"
|
||
}
|
||
|
||
reset_strikes() {
|
||
set_strikes "$1" 0
|
||
}
|
||
|
||
get_state_val() {
|
||
grep -E "^${1}=" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null | cut -d'=' -f2
|
||
}
|
||
|
||
set_state_val() {
|
||
local key="$1" val="$2"
|
||
grep -vE "^${key}=" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null > "${SYS_WATCHDOG_STATE_FILE}.tmp"
|
||
echo "${key}=${val}" >> "${SYS_WATCHDOG_STATE_FILE}.tmp"
|
||
mv "${SYS_WATCHDOG_STATE_FILE}.tmp" "$SYS_WATCHDOG_STATE_FILE"
|
||
}
|
||
|
||
purge_old_reboots() {
|
||
local now cutoff
|
||
now=$(date +%s)
|
||
cutoff=$(( now - SYS_WATCHDOG_REBOOT_WINDOW ))
|
||
grep -v "^$" "$SYS_WATCHDOG_REBOOT_LOG" 2>/dev/null | while IFS= read -r ts; do
|
||
[[ "$ts" -gt "$cutoff" ]] && echo "$ts"
|
||
done > "${SYS_WATCHDOG_REBOOT_LOG}.tmp"
|
||
mv "${SYS_WATCHDOG_REBOOT_LOG}.tmp" "$SYS_WATCHDOG_REBOOT_LOG"
|
||
}
|
||
|
||
count_recent_reboots() {
|
||
purge_old_reboots
|
||
grep -c "." "$SYS_WATCHDOG_REBOOT_LOG" 2>/dev/null || echo 0
|
||
}
|
||
|
||
log_reboot() {
|
||
date +%s >> "$SYS_WATCHDOG_REBOOT_LOG"
|
||
}
|
||
|
||
# ==============================================================================================
|
||
# ── OOM TRACKING ──────────────────────────────────────────────────────────────────────────────
|
||
# ==============================================================================================
|
||
# Reads /proc/vmstat oom_kill counter — delta per cycle = rate of OOM kills
|
||
# Used for Tier 2 bypass and diagnostic context in reboot messages
|
||
|
||
get_oom_delta() {
|
||
local current_oom
|
||
current_oom=$(grep "^oom_kill " /proc/vmstat 2>/dev/null | awk '{print $2}')
|
||
[[ -z "$current_oom" ]] && echo 0 && return
|
||
|
||
local prev_oom
|
||
prev_oom=$(cat "$SYS_WATCHDOG_OOM_FILE" 2>/dev/null || echo 0)
|
||
echo "$current_oom" > "$SYS_WATCHDOG_OOM_FILE"
|
||
|
||
local delta=$(( current_oom - prev_oom ))
|
||
[[ "$delta" -lt 0 ]] && delta=0 # counter reset on reboot
|
||
echo "$delta"
|
||
}
|
||
|
||
get_oom_victims() {
|
||
# Get process names from dmesg that were OOM killed this boot
|
||
dmesg -T 2>/dev/null | grep -i "Killed process" | \
|
||
awk '{print $NF}' | sort | uniq -c | sort -rn | head -5 | \
|
||
awk '{printf "%s×%d ", $2, $1}' | sed 's/ $//'
|
||
}
|
||
|
||
# ==============================================================================================
|
||
# ── ABORT CONDITIONS ──────────────────────────────────────────────────────────────────────────
|
||
# ==============================================================================================
|
||
# Returns 1 if reboot should be aborted, 0 if reboot should proceed
|
||
# CRITICAL tier bypasses this function entirely
|
||
|
||
check_abort_conditions() {
|
||
local should_abort=false
|
||
|
||
if command -v zpool >/dev/null 2>&1; then
|
||
local unhealthy
|
||
unhealthy=$(zpool list -H -o health 2>/dev/null | grep -v ONLINE || true)
|
||
if [[ -n "$unhealthy" ]]; then
|
||
if [[ "$SYS_WATCHDOG_ABORT_ON_ZFS_UNHEALTHY" == true ]]; then
|
||
error "ZFS pool unhealthy — aborting reboot to prevent data loss"
|
||
notify "System watchdog aborted reboot on $(hostname) ($MY_ID) — ZFS pool unhealthy" \
|
||
"System Watchdog" "warning"
|
||
should_abort=true
|
||
else
|
||
warn "ZFS pool unhealthy — continuing reboot (ABORT_ON_ZFS_UNHEALTHY=false)"
|
||
fi
|
||
fi
|
||
fi
|
||
|
||
# Unraid 7.3+: parity state is in var.ini (mdResync != 0 means check/sync in progress)
|
||
# Older: parity-date.txt contained "progress" — check both for compatibility
|
||
_md_resync=$(awk -F'"' '/^mdResync=/{print $2}' /var/local/emhttp/var.ini 2>/dev/null)
|
||
_parity_running=false
|
||
[[ -n "$_md_resync" && "$_md_resync" != "0" ]] && _parity_running=true
|
||
grep -q "progress" /var/local/emhttp/parity-date.txt 2>/dev/null && _parity_running=true
|
||
if [[ "$_parity_running" == true ]]; then
|
||
if [[ "$SYS_WATCHDOG_ABORT_ON_PARITY" == true ]]; then
|
||
error "Parity check running — aborting reboot"
|
||
notify "System watchdog aborted reboot on $(hostname) ($MY_ID) — parity running" \
|
||
"System Watchdog" "warning"
|
||
should_abort=true
|
||
else
|
||
warn "Parity check running — continuing reboot (ABORT_ON_PARITY=false)"
|
||
fi
|
||
fi
|
||
|
||
if pgrep -f "mover" >/dev/null 2>&1; then
|
||
if [[ "$SYS_WATCHDOG_ABORT_ON_MOVER" == true ]]; then
|
||
error "Mover running — aborting reboot"
|
||
notify "System watchdog aborted reboot on $(hostname) ($MY_ID) — mover running" \
|
||
"System Watchdog" "warning"
|
||
should_abort=true
|
||
else
|
||
warn "Mover running — continuing reboot (ABORT_ON_MOVER=false)"
|
||
fi
|
||
fi
|
||
|
||
[[ "$should_abort" == true ]] && return 1
|
||
return 0
|
||
}
|
||
|
||
# ==============================================================================================
|
||
# ── STANDARD STRIKE CHECK ─────────────────────────────────────────────────────────────────────
|
||
# ==============================================================================================
|
||
# Returns 0 = reboot now | 1 = not yet
|
||
|
||
run_strike_check() {
|
||
local key="$1" triggered="$2" description="$3"
|
||
if [[ "$triggered" == true ]]; then
|
||
local strikes
|
||
strikes=$(increment_strikes "$key")
|
||
warn "$description — strike $strikes/$SYS_WATCHDOG_STRIKE_LIMIT"
|
||
if (( strikes >= SYS_WATCHDOG_STRIKE_LIMIT )); then
|
||
error "$description — strike limit hit, reboot triggered"
|
||
reset_strikes "$key"
|
||
return 0
|
||
fi
|
||
else
|
||
local current
|
||
current=$(get_strikes "$key")
|
||
[[ -n "$current" && "$current" -gt 0 ]] && reset_strikes "$key"
|
||
fi
|
||
return 1
|
||
}
|
||
|
||
# ── Exit Trap — restart containers stopped before an aborted reboot ───────────────────────────
|
||
_SYS_REBOOT_STOPPED=()
|
||
_trap_sys_reboot_restart() {
|
||
[[ ${#_SYS_REBOOT_STOPPED[@]} -eq 0 ]] && return
|
||
warn "Exit trap: restarting containers stopped before aborted reboot"
|
||
for c in "${_SYS_REBOOT_STOPPED[@]}"; do
|
||
[[ -z "$c" ]] && continue
|
||
docker inspect "$c" >/dev/null 2>&1 && docker start "$c" >/dev/null 2>&1 || true
|
||
done
|
||
}
|
||
|
||
# ==============================================================================================
|
||
# ── DO REBOOT ─────────────────────────────────────────────────────────────────────────────────
|
||
# ==============================================================================================
|
||
# tier: "critical" (bypass abort) | "urgent" | "standard"
|
||
|
||
do_reboot() {
|
||
local tier="${1:-standard}"
|
||
shift
|
||
local triggers=("$@")
|
||
|
||
# Get OOM context for reboot message
|
||
local oom_victims=""
|
||
if [[ "$SYS_WATCHDOG_CHECK_OOM" == true ]]; then
|
||
oom_victims=$(get_oom_victims)
|
||
[[ -n "$oom_victims" ]] && triggers+=("oom_victims: $oom_victims")
|
||
fi
|
||
|
||
# Abort check — CRITICAL bypasses this
|
||
if [[ "$tier" != "critical" ]]; then
|
||
if ! check_abort_conditions; then
|
||
return
|
||
fi
|
||
else
|
||
warn "CRITICAL tier — bypassing abort conditions"
|
||
fi
|
||
|
||
RECENT_REBOOTS=$(count_recent_reboots)
|
||
log "Recent reboots in ${SYS_WATCHDOG_REBOOT_WINDOW_HRS}hr window: $RECENT_REBOOTS / $SYS_WATCHDOG_REBOOT_LIMIT"
|
||
|
||
if [[ "$RECENT_REBOOTS" -ge "$SYS_WATCHDOG_REBOOT_LIMIT" ]]; then
|
||
error "Reboot loop detected — shutting down instead of rebooting"
|
||
notify "Reboot loop on $(hostname) ($MY_ID) — shutting down after $RECENT_REBOOTS reboots — ${triggers[*]}" \
|
||
"System Watchdog" "warning"
|
||
if [[ "$DRY_RUN" == true ]]; then
|
||
warn "DRY RUN — would shutdown now"
|
||
return
|
||
fi
|
||
sync
|
||
/sbin/poweroff
|
||
return
|
||
fi
|
||
|
||
echo ""
|
||
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
|
||
echo " $ICON_REBOOT_SMART STABILITY WATCHDOG — REBOOT TRIGGERED"
|
||
echo " Tier: ${tier^^}"
|
||
echo " Host: $MY_ID ($LOCAL_SERVER_NAME)"
|
||
for t in "${triggers[@]}"; do
|
||
echo " → $t"
|
||
done
|
||
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
|
||
|
||
notify "System watchdog ${tier^^} reboot on $(hostname) ($MY_ID) — ${triggers[*]}" \
|
||
"System Watchdog" "warning"
|
||
|
||
if [[ "$DRY_RUN" == true ]]; then
|
||
warn "DRY RUN — reboot sequence would begin now"
|
||
return
|
||
fi
|
||
|
||
log_reboot
|
||
|
||
# Graceful shutdown sequence
|
||
if is_vm_manager_enabled && command -v virsh >/dev/null 2>&1; then
|
||
warn "Shutting down VMs..."
|
||
for VM in $(virsh list --name 2>/dev/null); do
|
||
[[ -z "$VM" ]] && continue
|
||
virsh shutdown "$VM" >/dev/null 2>&1
|
||
done
|
||
sleep 30
|
||
else
|
||
log "VM Manager not enabled — skipping VM shutdown"
|
||
fi
|
||
|
||
warn "Stopping Docker containers..."
|
||
if is_docker_enabled && command -v docker >/dev/null 2>&1; then
|
||
mapfile -t _SYS_REBOOT_STOPPED < <(docker ps --format '{{.Names}}' 2>/dev/null)
|
||
trap _trap_sys_reboot_restart EXIT
|
||
timeout 60 docker ps -q 2>/dev/null | xargs -r docker stop >/dev/null 2>&1
|
||
fi
|
||
|
||
warn "Stopping User Scripts..."
|
||
pkill -f "/tmp/user.scripts" 2>/dev/null || true
|
||
|
||
warn "Syncing disks..."
|
||
sync
|
||
|
||
trap - EXIT # committed to reboot — containers should stay down
|
||
sleep 5
|
||
/sbin/reboot
|
||
}
|
||
|
||
# ==============================================================================================
|
||
# ━━━ Single-Pass Health Check ━━━
|
||
# ==============================================================================================
|
||
echo "━━━ $ICON_REBOOT Stability Watchdog — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
|
||
|
||
TRIGGERS=()
|
||
CRITICAL_TRIGGERS=()
|
||
URGENT_OOM_CONFIRMED=false
|
||
|
||
# ── OOM Delta — read every cycle for bypass decisions ─────────────────────────────────────
|
||
OOM_DELTA=0
|
||
if [[ "$SYS_WATCHDOG_CHECK_OOM" == true ]]; then
|
||
OOM_DELTA=$(get_oom_delta)
|
||
[[ "$OOM_DELTA" -gt 0 ]] && \
|
||
log "OOM kills this cycle: $OOM_DELTA (limit: ${SYS_WATCHDOG_OOM_LIMIT})"
|
||
fi
|
||
|
||
# ==========================================================================================
|
||
# ━━━ TIER 1 — CRITICAL CHECKS (bypass all strikes, reboot immediately) ━━━
|
||
# ==========================================================================================
|
||
|
||
# ── Docker daemon — delegated to docker_watchdog ─────────────────────────────────────────
|
||
# docker_watchdog owns daemon restart attempts (strike system + rc.docker restart).
|
||
# When restart fails and daemon is confirmed down, it writes daemon_confirmed_down=true
|
||
# to WATCHDOG_STATE_FILE. We read that flag and run through the standard strike system.
|
||
if [[ "$SYS_WATCHDOG_CHECK_DOCKER_DAEMON" == true ]] && is_docker_enabled; then
|
||
_daemon_down=$(grep -oP "(?<=^daemon_confirmed_down:)[^:]*" "$WATCHDOG_STATE_FILE" 2>/dev/null || echo "false")
|
||
TRIGGERED=false
|
||
[[ "$_daemon_down" == "true" ]] && TRIGGERED=true
|
||
run_strike_check "docker_daemon" "$TRIGGERED" "Docker daemon confirmed down by docker_watchdog" && \
|
||
TRIGGERS+=("docker_daemon_unresponsive")
|
||
[[ "$TRIGGERED" == true ]] && log "Docker daemon confirmed down — strike toward reboot" || \
|
||
log "Docker daemon flag clear ✅"
|
||
elif [[ "$SYS_WATCHDOG_CHECK_DOCKER_DAEMON" == true ]]; then
|
||
log "Docker not enabled — skipping daemon check"
|
||
fi
|
||
|
||
# ── rootfs critical — at 99%+ writes are failing ─────────────────────────────────────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_ROOTFS" == true ]]; then
|
||
ROOTFS_USED=$(df / --output=pcent 2>/dev/null | tail -1 | tr -d ' %')
|
||
if [[ "$ROOTFS_USED" -ge "${SYS_WATCHDOG_ROOTFS_CRITICAL_PCT:-99}" ]]; then
|
||
error "rootfs ${ROOTFS_USED}% — CRITICAL (writes failing)"
|
||
CRITICAL_TRIGGERS+=("rootfs_full=${ROOTFS_USED}%")
|
||
fi
|
||
fi
|
||
|
||
# ── Kernel oops/BUG — kernel running with corrupted state ────────────────────────────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_KERNEL_OOPS" == true ]]; then
|
||
PREV_OOPS=$(get_state_val "kernel_oops_count")
|
||
CURRENT_OOPS=$(dmesg 2>/dev/null | grep -cE "BUG:|kernel BUG|Oops:" || echo 0)
|
||
CURRENT_OOPS="${CURRENT_OOPS//[^0-9]/}"; CURRENT_OOPS="${CURRENT_OOPS:-0}"
|
||
set_state_val "kernel_oops_count" "$CURRENT_OOPS"
|
||
|
||
if [[ -n "$PREV_OOPS" && "$PREV_OOPS" =~ ^[0-9]+$ ]]; then
|
||
OOPS_DELTA=$(( CURRENT_OOPS - PREV_OOPS ))
|
||
if [[ "$OOPS_DELTA" -gt 0 ]]; then
|
||
error "Kernel oops/BUG detected — $OOPS_DELTA new since last cycle — CRITICAL"
|
||
CRITICAL_TRIGGERS+=("kernel_oops=${OOPS_DELTA}_new")
|
||
fi
|
||
fi
|
||
fi
|
||
|
||
# ── File descriptor exhaustion — new connections failing silently ─────────────────────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_FD" == true ]]; then
|
||
FD_LINE=$(cat /proc/sys/fs/file-nr 2>/dev/null)
|
||
FD_OPEN=$(echo "$FD_LINE" | awk '{print $1}')
|
||
FD_MAX=$(echo "$FD_LINE" | awk '{print $3}')
|
||
if [[ -n "$FD_OPEN" && -n "$FD_MAX" && "$FD_MAX" -gt 0 ]]; then
|
||
FD_PCT=$(( FD_OPEN * 100 / FD_MAX ))
|
||
if [[ "$FD_PCT" -ge "${SYS_WATCHDOG_FD_CRITICAL_PCT:-95}" ]]; then
|
||
error "File descriptors ${FD_PCT}% exhausted (${FD_OPEN}/${FD_MAX}) — CRITICAL"
|
||
CRITICAL_TRIGGERS+=("fd_exhaustion=${FD_PCT}%")
|
||
else
|
||
log "File descriptors: ${FD_PCT}% (${FD_OPEN}/${FD_MAX})"
|
||
fi
|
||
fi
|
||
fi
|
||
|
||
# ── /boot read-only — state and config writes failing silently ────────────────────────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_BOOT" == true ]]; then
|
||
BOOT_TEST="/boot/.watchdog_write_test"
|
||
if ! touch "$BOOT_TEST" 2>/dev/null; then
|
||
error "/boot is read-only — config writes failing silently — CRITICAL"
|
||
CRITICAL_TRIGGERS+=("boot_read_only")
|
||
else
|
||
rm -f "$BOOT_TEST" 2>/dev/null
|
||
log "/boot is writable ✅"
|
||
fi
|
||
fi
|
||
|
||
# ── Act on CRITICAL triggers immediately ─────────────────────────────────────────────────
|
||
if [[ ${#CRITICAL_TRIGGERS[@]} -gt 0 ]]; then
|
||
echo ""
|
||
echo "━━━ $ICON_ERROR CRITICAL — IMMEDIATE REBOOT ━━━"
|
||
for t in "${CRITICAL_TRIGGERS[@]}"; do
|
||
error " CRITICAL: $t"
|
||
done
|
||
do_reboot "critical" "${CRITICAL_TRIGGERS[@]}"
|
||
exit 0
|
||
fi
|
||
|
||
# ==========================================================================================
|
||
# ━━━ TIER 3 — STANDARD CHECKS (strike system) ━━━
|
||
# ==========================================================================================
|
||
|
||
# ── rootfs standard ──────────────────────────────────────────────────────────────────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_ROOTFS" == true ]]; then
|
||
ROOTFS_USED=$(df / --output=pcent 2>/dev/null | tail -1 | tr -d ' %')
|
||
TRIGGERED=false
|
||
[[ "$ROOTFS_USED" -ge "$SYS_WATCHDOG_ROOTFS_PCT" ]] && TRIGGERED=true
|
||
run_strike_check "rootfs" "$TRIGGERED" "rootfs ${ROOTFS_USED}%" && \
|
||
TRIGGERS+=("rootfs=${ROOTFS_USED}%")
|
||
fi
|
||
|
||
# ── /var/log ─────────────────────────────────────────────────────────────────────────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_LOG" == true ]]; then
|
||
LOG_USED=$(df -P /var/log 2>/dev/null | awk 'NR==2 {print $5}' | tr -d '%')
|
||
TRIGGERED=false
|
||
[[ "${LOG_USED:-0}" -ge "$SYS_WATCHDOG_LOG_PCT" ]] && TRIGGERED=true
|
||
run_strike_check "log" "$TRIGGERED" "/var/log ${LOG_USED}%" && \
|
||
TRIGGERS+=("log=${LOG_USED}%")
|
||
fi
|
||
|
||
# ── /tmp ─────────────────────────────────────────────────────────────────────────────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_TMP" == true ]]; then
|
||
TMP_USED=$(df -P /tmp 2>/dev/null | awk 'NR==2 {print $5}' | tr -d '%')
|
||
if [[ "${TMP_USED:-0}" -ge "${SYS_WATCHDOG_TMP_CRITICAL_PCT:-98}" ]]; then
|
||
# Try to clear before escalating
|
||
warn "/tmp ${TMP_USED}% — attempting cleanup..."
|
||
find /tmp -type f -mmin +60 -not -name "*.lock" -delete 2>/dev/null
|
||
TMP_USED_AFTER=$(df -P /tmp 2>/dev/null | awk 'NR==2 {print $5}' | tr -d '%')
|
||
if [[ "${TMP_USED_AFTER:-0}" -ge "${SYS_WATCHDOG_TMP_CRITICAL_PCT:-98}" ]]; then
|
||
error "/tmp still ${TMP_USED_AFTER}% after cleanup — adding to triggers"
|
||
TRIGGERED=true
|
||
else
|
||
warn "/tmp cleared to ${TMP_USED_AFTER}% ✅"
|
||
TRIGGERED=false
|
||
fi
|
||
elif [[ "${TMP_USED:-0}" -ge "${SYS_WATCHDOG_TMP_PCT:-90}" ]]; then
|
||
TRIGGERED=true
|
||
else
|
||
TRIGGERED=false
|
||
fi
|
||
run_strike_check "tmp" "$TRIGGERED" "/tmp ${TMP_USED}%" && \
|
||
TRIGGERS+=("tmp=${TMP_USED}%")
|
||
fi
|
||
|
||
# ── RAM — reboot tier only (warn/shutdown/recover handled by resource_watchdog.sh) ──────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_RAM" == true ]]; then
|
||
MEM_KB=$(awk '/MemAvailable/ {print $2}' /proc/meminfo)
|
||
MEM_GB=$(( MEM_KB / 1024 / 1024 ))
|
||
|
||
if [[ "$MEM_GB" -lt "$SYS_WATCHDOG_MEM_GB" ]]; then
|
||
# Tier 2 — bypass strike system if OOM confirms active crisis
|
||
if [[ "$SYS_WATCHDOG_CHECK_OOM" == true ]] && \
|
||
[[ "$OOM_DELTA" -ge "$SYS_WATCHDOG_OOM_LIMIT" ]]; then
|
||
error "RAM ${MEM_GB}GB + ${OOM_DELTA} OOM kills this run — URGENT bypass"
|
||
OOM_VICTIMS=$(get_oom_victims)
|
||
URGENT_TRIGGERS=("urgent_low_ram=${MEM_GB}GB" "oom_kills=${OOM_DELTA}")
|
||
[[ -n "$OOM_VICTIMS" ]] && URGENT_TRIGGERS+=("oom_victims: $OOM_VICTIMS")
|
||
do_reboot "urgent" "${URGENT_TRIGGERS[@]}"
|
||
exit 0
|
||
fi
|
||
# Standard strike path
|
||
run_strike_check "ram" true "RAM ${MEM_GB}GB free" && \
|
||
TRIGGERS+=("low_ram=${MEM_GB}GB")
|
||
else
|
||
reset_strikes "ram"
|
||
log "RAM ${MEM_GB}GB free ✅"
|
||
fi
|
||
fi
|
||
|
||
# ── ZFS ARC ──────────────────────────────────────────────────────────────────────────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_ARC" == true ]] && [[ -f /proc/spl/kstat/zfs/arcstats ]]; then
|
||
ARC_SIZE=$(awk '/^size / {print $3}' /proc/spl/kstat/zfs/arcstats)
|
||
ARC_MAX=$(awk '/^c_max / {print $3}' /proc/spl/kstat/zfs/arcstats)
|
||
ARC_PCT=$(( ARC_SIZE * 100 / ARC_MAX ))
|
||
TRIGGERED=false
|
||
if [[ "$ARC_PCT" -ge "$SYS_WATCHDOG_ARC_PINNED_PCT" ]]; then
|
||
sync; echo 3 > /proc/sys/vm/drop_caches; sleep 5
|
||
ARC_AFTER=$(awk '/^size / {print $3}' /proc/spl/kstat/zfs/arcstats)
|
||
ARC_AFTER_PCT=$(( ARC_AFTER * 100 / ARC_MAX ))
|
||
[[ "$ARC_AFTER_PCT" -ge "$SYS_WATCHDOG_ARC_RELEASE_PCT" ]] && TRIGGERED=true
|
||
fi
|
||
run_strike_check "arc" "$TRIGGERED" "ZFS ARC pinned ${ARC_PCT}%" && \
|
||
TRIGGERS+=("arc_pinned=${ARC_PCT}%")
|
||
fi
|
||
|
||
# ── CPU temperature ───────────────────────────────────────────────────────────────────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_CPU_TEMP" == true ]]; then
|
||
CPU_TEMP=""
|
||
if command -v sensors >/dev/null 2>&1; then
|
||
CPU_TEMP=$(sensors 2>/dev/null | \
|
||
grep -i "Package id 0\|Tctl\|CPU Temp" | \
|
||
awk '{print $NF}' | tr -d '+°C' | head -1)
|
||
fi
|
||
if [[ -n "$CPU_TEMP" ]]; then
|
||
CPU_TEMP_INT=$(printf "%.0f" "$CPU_TEMP")
|
||
TRIGGERED=false
|
||
[[ "$CPU_TEMP_INT" -ge "$SYS_WATCHDOG_CPU_TEMP_MAX" ]] && TRIGGERED=true
|
||
run_strike_check "cpu_temp" "$TRIGGERED" "CPU temp ${CPU_TEMP_INT}°C" && \
|
||
TRIGGERS+=("cpu_temp=${CPU_TEMP_INT}C")
|
||
fi
|
||
fi
|
||
|
||
# ── Load average ─────────────────────────────────────────────────────────────────────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_LOAD" == true ]]; then
|
||
LOAD=$(awk '{print $1}' /proc/loadavg)
|
||
LOAD_INT=$(printf "%.0f" "$LOAD")
|
||
LOAD_THRESHOLD=$(( TOTAL_CORES * SYS_WATCHDOG_LOAD_MULTIPLIER ))
|
||
TRIGGERED=false
|
||
[[ "$LOAD_INT" -ge "$LOAD_THRESHOLD" ]] && TRIGGERED=true
|
||
run_strike_check "load" "$TRIGGERED" "load avg ${LOAD}" && \
|
||
TRIGGERS+=("load=${LOAD}")
|
||
fi
|
||
|
||
# ── Zombie processes ─────────────────────────────────────────────────────────────────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_ZOMBIES" == true ]]; then
|
||
ZOMBIE_COUNT=$(ps aux 2>/dev/null | awk '{print $8}' | grep -c "^Z$" || echo 0)
|
||
ZOMBIE_COUNT="${ZOMBIE_COUNT//[^0-9]/}"; ZOMBIE_COUNT="${ZOMBIE_COUNT:-0}"
|
||
TRIGGERED=false
|
||
[[ "$ZOMBIE_COUNT" -ge "$SYS_WATCHDOG_ZOMBIE_LIMIT" ]] && TRIGGERED=true
|
||
run_strike_check "zombies" "$TRIGGERED" "zombies ${ZOMBIE_COUNT}" && \
|
||
TRIGGERS+=("zombies=${ZOMBIE_COUNT}")
|
||
fi
|
||
|
||
# ── Array disk errors — accumulating mdstat errors ────────────────────────────────────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_MDSTAT" == true ]]; then
|
||
PREV_MD_ERRORS=$(get_state_val "mdstat_errors")
|
||
CURRENT_MD_ERRORS=$(grep -oP "(?<=\[)[^\]]*[U_][^\]]*(?=\])" \
|
||
/proc/mdstat 2>/dev/null | grep -o "_" | wc -l || echo 0)
|
||
CURRENT_MD_ERRORS="${CURRENT_MD_ERRORS//[^0-9]/}"; CURRENT_MD_ERRORS="${CURRENT_MD_ERRORS:-0}"
|
||
set_state_val "mdstat_errors" "$CURRENT_MD_ERRORS"
|
||
|
||
if [[ -n "$PREV_MD_ERRORS" && "$PREV_MD_ERRORS" =~ ^[0-9]+$ ]]; then
|
||
MD_DELTA=$(( CURRENT_MD_ERRORS - PREV_MD_ERRORS ))
|
||
if [[ "$MD_DELTA" -ge "${SYS_WATCHDOG_MDSTAT_ERROR_LIMIT:-5}" ]]; then
|
||
TRIGGERED=true
|
||
run_strike_check "mdstat" "$TRIGGERED" \
|
||
"mdstat errors +${MD_DELTA} (total: ${CURRENT_MD_ERRORS})" && \
|
||
TRIGGERS+=("mdstat_errors=+${MD_DELTA}")
|
||
else
|
||
run_strike_check "mdstat" false "mdstat" > /dev/null
|
||
fi
|
||
fi
|
||
fi
|
||
|
||
# ── Network interface state ───────────────────────────────────────────────────────────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_NETWORK" == true ]]; then
|
||
NIC="${SYS_WATCHDOG_NIC:-eth0}"
|
||
NIC_STATE=$(cat "/sys/class/net/${NIC}/operstate" 2>/dev/null || echo "unknown")
|
||
TRIGGERED=false
|
||
[[ "$NIC_STATE" != "up" ]] && TRIGGERED=true
|
||
run_strike_check "network" "$TRIGGERED" "${NIC} state: ${NIC_STATE}" && \
|
||
TRIGGERS+=("nic_down=${NIC}")
|
||
fi
|
||
|
||
# ── sshd — try restart before escalating ─────────────────────────────────────────────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_SSHD" == true ]]; then
|
||
if ! pgrep -x sshd >/dev/null 2>&1; then
|
||
warn "sshd not running — attempting restart..."
|
||
if [[ "$DRY_RUN" == false ]]; then
|
||
/etc/rc.d/rc.sshd start >/dev/null 2>&1
|
||
sleep 3
|
||
if pgrep -x sshd >/dev/null 2>&1; then
|
||
warn "sshd restarted successfully ✅"
|
||
reset_strikes "sshd"
|
||
notify "sshd was down on $(hostname) ($MY_ID) — restarted automatically" \
|
||
"System Watchdog" "warning"
|
||
else
|
||
error "sshd restart failed — remote access unavailable"
|
||
run_strike_check "sshd" true "sshd not running" && \
|
||
TRIGGERS+=("sshd_down")
|
||
fi
|
||
else
|
||
warn "DRY RUN — would restart sshd"
|
||
fi
|
||
else
|
||
reset_strikes "sshd"
|
||
fi
|
||
fi
|
||
|
||
# ── Runaway process ───────────────────────────────────────────────────────────────────────
|
||
if [[ "$SYS_WATCHDOG_CHECK_RUNAWAY" == true ]]; then
|
||
RUNAWAY_PCT="${SYS_WATCHDOG_RUNAWAY_CPU_PCT:-90}"
|
||
TOP_CPU_PCT=$(ps aux 2>/dev/null | awk 'NR>1{print $3}' | sort -rn | head -1)
|
||
TOP_CPU_INT=$(printf "%.0f" "${TOP_CPU_PCT:-0}")
|
||
TOP_CPU_NAME=$(ps aux 2>/dev/null | sort -k3 -rn | awk 'NR==2{print $11}')
|
||
TRIGGERED=false
|
||
[[ "$TOP_CPU_INT" -ge "$RUNAWAY_PCT" ]] && TRIGGERED=true
|
||
# Runaway uses SYS_WATCHDOG_RUNAWAY_STRIKES not global strike limit
|
||
if [[ "$TRIGGERED" == true ]]; then
|
||
RAWAY_S=$(increment_strikes "runaway")
|
||
RLIMIT="${SYS_WATCHDOG_RUNAWAY_STRIKES:-3}"
|
||
warn "Runaway ${TOP_CPU_NAME} ${TOP_CPU_PCT}% CPU -- strike $RAWAY_S/$RLIMIT"
|
||
if (( RAWAY_S >= RLIMIT )); then
|
||
error "Runaway process ${TOP_CPU_NAME} -- strike limit hit"
|
||
reset_strikes "runaway"
|
||
TRIGGERS+=("runaway=${TOP_CPU_NAME}@${TOP_CPU_PCT}%")
|
||
fi
|
||
else
|
||
RAWAY_CUR=$(get_strikes "runaway")
|
||
[[ "${RAWAY_CUR:-0}" -gt 0 ]] && reset_strikes "runaway"
|
||
fi
|
||
fi
|
||
|
||
# ── Required containers — removed from stability_watchdog ────────────────────────────────
|
||
# Container health is owned entirely by docker_watchdog: strike system, restart attempts,
|
||
# skip-listing, and critical notifications. Rebooting here when docker_watchdog already
|
||
# gave up creates a reboot loop — the container is still broken after reboot.
|
||
|
||
# ==========================================================================================
|
||
# ━━━ Evaluate Standard Triggers ━━━
|
||
# ==========================================================================================
|
||
if [[ ${#TRIGGERS[@]} -gt 0 ]]; then
|
||
echo ""
|
||
echo "━━━ $ICON_REBOOT_SMART Stability Watchdog — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
|
||
for t in "${TRIGGERS[@]}"; do
|
||
echo " $ICON_REBOOT_SMART $t"
|
||
done
|
||
[[ "$OOM_DELTA" -gt 0 ]] && echo " OOM kills this run: $OOM_DELTA"
|
||
echo ""
|
||
do_reboot "standard" "${TRIGGERS[@]}"
|
||
exit 0
|
||
else
|
||
echo "System healthy ✅ ($(date '+%H:%M:%S'))"
|
||
fi
|
||
|
||
# Keep state file mtime fresh — docker_watchdog stale guard checks this
|
||
set_state_val "watchdog_cycle" "$(date +%s)" |