#!/bin/bash # ============================================================================================== # ================================= System Watchdog ============================================ # ============================================================================================== # Last line of defense — reboots the system cleanly if it is about to become unstable. # Runs continuously as a background process — started by array_start.sh at array start. # Works alongside docker_watchdog.sh which handles container-level healing first. # # ── THREE-TIER RESPONSE SYSTEM ──────────────────────────────────────────────────────────────── # # TIER 1 — CRITICAL (bypass ALL strikes, reboot immediately) # Docker daemon unresponsive — nothing can be healed, letting it run makes it worse # rootfs at 99%+ — writes failing, SSH may stop, no recovery options # Kernel oops/BUG in dmesg — kernel running with corrupted state # File descriptor exhaustion — new connections and processes failing silently # /boot read-only unexpectedly — state files and config writes silently failing # # TIER 2 — URGENT (bypass strikes when OOM confirms active crisis) # RAM < MEM_GB AND OOM kills >= OOM_LIMIT in this cycle # Rationale: OOM kills at this rate means system is dying faster than watchdogs heal # Without OOM confirmation → standard strike system applies # # TIER 3 — STANDARD (N consecutive failures → reboot) # RAM tiers, load, CPU temp, zombies, /var/log, /tmp, containers, NIC, mdstat # # ── RAM TIERS ───────────────────────────────────────────────────────────────────────────────── # MEM_WARN_GB (10GB) — warn + notify only # MEM_SHUTDOWN_GB (6GB) — stop non-essential containers, wait for recovery # MEM_GB (4GB) — strike system → reboot (bypass if OOM confirms) # MEM_RECOVER_GB (30GB) — RAM must reach this before containers restart # # ── CONTAINER SHUTDOWN LOGIC ────────────────────────────────────────────────────────────────── # At MEM_SHUTDOWN_GB: stop all containers NOT in SYS_WATCHDOG_MEM_SHUTDOWN_EXCLUDED # Excluded: NginxProxyManager, Authelia, Mariadb, Redis, Emby, Dispatcharr # Stopped containers tracked in shutdown list — won't restart until RAM recovers # Strike system prevents flip-flopping — shutdown only happens once per degradation event # # ── OOM TRACKING ────────────────────────────────────────────────────────────────────────────── # /proc/vmstat oom_kill counter — read each cycle, delta = kills this cycle # Included in reboot message with process names from dmesg (diagnostic context) # Bypass trigger: RAM critical AND kills this cycle >= SYS_WATCHDOG_OOM_LIMIT # # ── NEW CHECKS THIS VERSION ─────────────────────────────────────────────────────────────────── # OOM rate tracking — delta from /proc/vmstat each cycle # /boot read-only — write test on /boot each cycle # Kernel oops detection — dmesg BUG/Oops count delta each cycle # File descriptor exhaustion — /proc/sys/fs/file-nr utilisation # /tmp usage — tmpfs fill detection with auto-clear attempt # Array disk errors — mdstat error delta each cycle # Runaway process — single process >N% CPU sustained (disabled by default) # NIC state check — primary interface operstate # sshd check — restart attempt before escalating # # ── EXISTING CHECKS ─────────────────────────────────────────────────────────────────────────── # rootfs usage, /var/log, free RAM, ZFS ARC, CPU temp, load avg, # zombie processes, Docker daemon, required containers from skip list # # ── ABORT CONDITIONS ────────────────────────────────────────────────────────────────────────── # ZFS pool unhealthy, parity running, mover running — toggleable # CRITICAL tier bypasses abort conditions — imminent crash overrides data safety # # ── CONFIGURATION (master.conf) ─────────────────────────────────────────────────────────────── # Full config under System Watchdog section — see master.conf for all vars # # ── USAGE ───────────────────────────────────────────────────────────────────────────────────── # system_watchdog.sh — normal start (continuous loop) # system_watchdog.sh --dry-run — trigger detection without rebooting # system_watchdog.sh --status — show config and thresholds # system_watchdog.sh --log — verbose per-cycle output # ============================================================================================== SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" source "$SCRIPT_DIR/../load_config.sh" parse_args "$@" # ============================================================================================== # ━━━ Setup — runs once at start ━━━ # ============================================================================================== if [[ "$EUID" -ne 0 ]]; then error "Must be run as root" exit 1 fi validate_unraid_cmd \ "/usr/local/emhttp/plugins/dynamix/scripts/notify" \ "" "" \ "unRAID notify script" || warn "unRAID notify script not found — native notifications disabled" acquire_lock "continuous" detect_hosts TOTAL_CORES=$(nproc) DOCKER_TIMEOUT=10 # Ensure state files exist for state_file in "$SYS_WATCHDOG_STATE_FILE" "$SYS_WATCHDOG_REBOOT_LOG" \ "$SYS_WATCHDOG_FAILED_FILE" "$SYS_WATCHDOG_OOM_FILE"; do touch "$state_file" 2>/dev/null || { error "Cannot create state file: $state_file" exit 1 } done [[ "$DRY_RUN" == true ]] && warn "DRY RUN — no reboots or container shutdowns will occur" # ============================================================================================== # ━━━ Status ━━━ # ============================================================================================== if [[ "$SHOW_STATUS" == true ]]; then echo "" echo "━━━━━ $ICON_SUMMARY SYSTEM WATCHDOG STATUS ━━━━━" echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)" echo "" echo "── Tier 1 — CRITICAL (bypass strikes immediately) ──" echo "$ICON_DISK rootfs critical: ${SYS_WATCHDOG_ROOTFS_CRITICAL_PCT}%" echo "$ICON_GEAR FD critical: ${SYS_WATCHDOG_FD_CRITICAL_PCT}%" echo "$ICON_GEAR /boot read-only: check=${SYS_WATCHDOG_CHECK_BOOT}" echo "$ICON_GEAR Kernel oops: check=${SYS_WATCHDOG_CHECK_KERNEL_OOPS}" echo "$ICON_CONTAINERS Docker daemon: check=${SYS_WATCHDOG_CHECK_DOCKER_DAEMON}" echo "" echo "── Tier 2 — URGENT (bypass strikes with OOM confirmation) ──" echo "$ICON_MEM RAM critical: < ${SYS_WATCHDOG_MEM_GB}GB" echo "$ICON_GEAR OOM limit: ${SYS_WATCHDOG_OOM_LIMIT} kills/cycle" echo "" echo "── Tier 3 — STANDARD (strike system) ──" echo "$ICON_DISK rootfs warn: ${SYS_WATCHDOG_ROOTFS_PCT}%" echo "$ICON_GEAR /var/log warn: ${SYS_WATCHDOG_LOG_PCT}%" echo "$ICON_GEAR /tmp warn: ${SYS_WATCHDOG_TMP_PCT}%" echo "$ICON_MEM RAM warn: < ${SYS_WATCHDOG_MEM_WARN_GB}GB" echo "$ICON_MEM RAM shutdown: < ${SYS_WATCHDOG_MEM_SHUTDOWN_GB}GB" echo "$ICON_MEM RAM recover: > ${SYS_WATCHDOG_MEM_RECOVER_GB}GB" echo "$ICON_MEM RAM reboot: < ${SYS_WATCHDOG_MEM_GB}GB (+ strikes)" echo "$ICON_ZFS ARC pinned: ${SYS_WATCHDOG_ARC_PINNED_PCT}%" echo "$ICON_GEAR Load multiplier: ${SYS_WATCHDOG_LOAD_MULTIPLIER}x (= $(( TOTAL_CORES * SYS_WATCHDOG_LOAD_MULTIPLIER )) on $TOTAL_CORES cores)" echo "$ICON_GEAR Zombie limit: ${SYS_WATCHDOG_ZOMBIE_LIMIT}" echo "$ICON_GEAR CPU temp max: ${SYS_WATCHDOG_CPU_TEMP_MAX}°C" echo "$ICON_GEAR Strike limit: ${SYS_WATCHDOG_STRIKE_LIMIT} cycles" echo "$ICON_TIME Interval: ${SYSTEM_WATCHDOG_INTERVAL}s" echo "$ICON_REBOOT_SMART Reboot limit: ${SYS_WATCHDOG_REBOOT_LIMIT} in ${SYS_WATCHDOG_REBOOT_WINDOW_HRS}hr" echo "" echo "── Container Shutdown Excluded ──" for c in "${SYS_WATCHDOG_MEM_SHUTDOWN_EXCLUDED[@]:-}"; do echo " $ICON_RUNNING $c" done echo "" echo "── Check Toggles ──" echo " rootfs=$SYS_WATCHDOG_CHECK_ROOTFS log=$SYS_WATCHDOG_CHECK_LOG ram=$SYS_WATCHDOG_CHECK_RAM" echo " arc=$SYS_WATCHDOG_CHECK_ARC cpu_temp=$SYS_WATCHDOG_CHECK_CPU_TEMP load=$SYS_WATCHDOG_CHECK_LOAD" echo " zombies=$SYS_WATCHDOG_CHECK_ZOMBIES docker=$SYS_WATCHDOG_CHECK_DOCKER_DAEMON" echo " containers=$SYS_WATCHDOG_CHECK_CONTAINERS oom=$SYS_WATCHDOG_CHECK_OOM" echo " tmp=$SYS_WATCHDOG_CHECK_TMP fd=$SYS_WATCHDOG_CHECK_FD boot=$SYS_WATCHDOG_CHECK_BOOT" echo " kernel_oops=$SYS_WATCHDOG_CHECK_KERNEL_OOPS sshd=$SYS_WATCHDOG_CHECK_SSHD" echo " network=$SYS_WATCHDOG_CHECK_NETWORK mdstat=$SYS_WATCHDOG_CHECK_MDSTAT" echo " runaway=$SYS_WATCHDOG_CHECK_RUNAWAY" echo "━━━━━━━━━━━━━━━━━━━━━━━" exit 0 fi # ============================================================================================== # ── STATE HELPERS ───────────────────────────────────────────────────────────────────────────── # ============================================================================================== get_strikes() { grep -E "^${1}:" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null | cut -d':' -f2 } set_strikes() { local key="$1" count="$2" grep -vE "^${key}:" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null > "${SYS_WATCHDOG_STATE_FILE}.tmp" echo "${key}:${count}" >> "${SYS_WATCHDOG_STATE_FILE}.tmp" mv "${SYS_WATCHDOG_STATE_FILE}.tmp" "$SYS_WATCHDOG_STATE_FILE" } increment_strikes() { local key="$1" local current current=$(get_strikes "$key") [[ -z "$current" ]] && current=0 (( current++ )) set_strikes "$key" "$current" echo "$current" } reset_strikes() { set_strikes "$1" 0 } get_state_val() { grep -E "^${1}=" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null | cut -d'=' -f2 } set_state_val() { local key="$1" val="$2" grep -vE "^${key}=" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null > "${SYS_WATCHDOG_STATE_FILE}.tmp" echo "${key}=${val}" >> "${SYS_WATCHDOG_STATE_FILE}.tmp" mv "${SYS_WATCHDOG_STATE_FILE}.tmp" "$SYS_WATCHDOG_STATE_FILE" } purge_old_reboots() { local now cutoff now=$(date +%s) cutoff=$(( now - SYS_WATCHDOG_REBOOT_WINDOW )) grep -v "^$" "$SYS_WATCHDOG_REBOOT_LOG" 2>/dev/null | while IFS= read -r ts; do [[ "$ts" -gt "$cutoff" ]] && echo "$ts" done > "${SYS_WATCHDOG_REBOOT_LOG}.tmp" mv "${SYS_WATCHDOG_REBOOT_LOG}.tmp" "$SYS_WATCHDOG_REBOOT_LOG" } count_recent_reboots() { purge_old_reboots grep -c "." "$SYS_WATCHDOG_REBOOT_LOG" 2>/dev/null || echo 0 } log_reboot() { date +%s >> "$SYS_WATCHDOG_REBOOT_LOG" } # ============================================================================================== # ── OOM TRACKING ────────────────────────────────────────────────────────────────────────────── # ============================================================================================== # Reads /proc/vmstat oom_kill counter — delta per cycle = rate of OOM kills # Used for Tier 2 bypass and diagnostic context in reboot messages get_oom_delta() { local current_oom current_oom=$(grep "^oom_kill " /proc/vmstat 2>/dev/null | awk '{print $2}') [[ -z "$current_oom" ]] && echo 0 && return local prev_oom prev_oom=$(cat "$SYS_WATCHDOG_OOM_FILE" 2>/dev/null || echo 0) echo "$current_oom" > "$SYS_WATCHDOG_OOM_FILE" local delta=$(( current_oom - prev_oom )) [[ "$delta" -lt 0 ]] && delta=0 # counter reset on reboot echo "$delta" } get_oom_victims() { # Get process names from dmesg that were OOM killed this boot dmesg -T 2>/dev/null | grep -i "Killed process" | \ awk '{print $NF}' | sort | uniq -c | sort -rn | head -5 | \ awk '{printf "%s×%d ", $2, $1}' | sed 's/ $//' } # ============================================================================================== # ── ABORT CONDITIONS ────────────────────────────────────────────────────────────────────────── # ============================================================================================== # Returns 1 if reboot should be aborted, 0 if reboot should proceed # CRITICAL tier bypasses this function entirely check_abort_conditions() { local should_abort=false if command -v zpool >/dev/null 2>&1; then local unhealthy unhealthy=$(zpool list -H -o health 2>/dev/null | grep -v ONLINE || true) if [[ -n "$unhealthy" ]]; then if [[ "$SYS_WATCHDOG_ABORT_ON_ZFS_UNHEALTHY" == true ]]; then error "ZFS pool unhealthy — aborting reboot to prevent data loss" notify "System watchdog aborted reboot on $(hostname) ($MY_ID) — ZFS pool unhealthy" \ "System Watchdog" "warning" should_abort=true else warn "ZFS pool unhealthy — continuing reboot (ABORT_ON_ZFS_UNHEALTHY=false)" fi fi fi if grep -q "progress" /var/local/emhttp/parity-date.txt 2>/dev/null; then if [[ "$SYS_WATCHDOG_ABORT_ON_PARITY" == true ]]; then error "Parity check running — aborting reboot" notify "System watchdog aborted reboot on $(hostname) ($MY_ID) — parity running" \ "System Watchdog" "warning" should_abort=true else warn "Parity check running — continuing reboot (ABORT_ON_PARITY=false)" fi fi if pgrep -f "mover" >/dev/null 2>&1; then if [[ "$SYS_WATCHDOG_ABORT_ON_MOVER" == true ]]; then error "Mover running — aborting reboot" notify "System watchdog aborted reboot on $(hostname) ($MY_ID) — mover running" \ "System Watchdog" "warning" should_abort=true else warn "Mover running — continuing reboot (ABORT_ON_MOVER=false)" fi fi [[ "$should_abort" == true ]] && return 1 return 0 } # ============================================================================================== # ── STANDARD STRIKE CHECK ───────────────────────────────────────────────────────────────────── # ============================================================================================== # Returns 0 = reboot now | 1 = not yet run_strike_check() { local key="$1" triggered="$2" description="$3" if [[ "$triggered" == true ]]; then local strikes strikes=$(increment_strikes "$key") warn "$description — strike $strikes/$SYS_WATCHDOG_STRIKE_LIMIT" if (( strikes >= SYS_WATCHDOG_STRIKE_LIMIT )); then error "$description — strike limit hit, reboot triggered" reset_strikes "$key" return 0 fi else local current current=$(get_strikes "$key") [[ -n "$current" && "$current" -gt 0 ]] && reset_strikes "$key" fi return 1 } # ============================================================================================== # ── CONTAINER SHUTDOWN (RAM EMERGENCY) ──────────────────────────────────────────────────────── # ============================================================================================== shutdown_non_essential_containers() { warn "RAM emergency — stopping non-essential containers" local stopped=() # Build exclusion map declare -A EXCLUDED_MAP for exc in "${SYS_WATCHDOG_MEM_SHUTDOWN_EXCLUDED[@]:-}"; do [[ -n "$exc" ]] && EXCLUDED_MAP["$exc"]=1 done # Stop all running containers not in exclusion list while IFS= read -r container; do [[ -z "$container" ]] && continue if [[ -n "${EXCLUDED_MAP[$container]:-}" ]]; then log "$container — excluded from RAM shutdown, leaving running" continue fi if [[ "$DRY_RUN" == false ]]; then timeout "$DOCKER_TIMEOUT" docker stop "$container" >/dev/null 2>&1 && \ warn "Stopped $container (RAM emergency)" && \ stopped+=("$container") || \ error "Failed to stop $container" else warn "DRY RUN — would stop $container (RAM emergency)" stopped+=("$container") fi done < <(timeout "$DOCKER_TIMEOUT" docker ps --format "{{.Names}}" 2>/dev/null) if [[ ${#stopped[@]} -gt 0 ]]; then set_state_val "mem_shutdown_active" "true" notify "RAM emergency on $(hostname) ($MY_ID) — stopped ${#stopped[@]} containers. Excluded: ${SYS_WATCHDOG_MEM_SHUTDOWN_EXCLUDED[*]}" \ "System Watchdog" "warning" warn "Stopped ${#stopped[@]} containers — waiting for RAM to recover above ${SYS_WATCHDOG_MEM_RECOVER_GB}GB" fi } restart_non_essential_containers() { warn "RAM recovered — restarting containers that were stopped in emergency" if [[ "$DRY_RUN" == false ]]; then set_state_val "mem_shutdown_active" "false" fi # docker_watchdog.sh will detect stopped required containers and restart them # We just clear the state flag here warn "Cleared RAM emergency state — docker_watchdog.sh will restart required containers" } # ============================================================================================== # ── DO REBOOT ───────────────────────────────────────────────────────────────────────────────── # ============================================================================================== # tier: "critical" (bypass abort) | "urgent" | "standard" do_reboot() { local tier="${1:-standard}" shift local triggers=("$@") # Get OOM context for reboot message local oom_victims="" if [[ "$SYS_WATCHDOG_CHECK_OOM" == true ]]; then oom_victims=$(get_oom_victims) [[ -n "$oom_victims" ]] && triggers+=("oom_victims: $oom_victims") fi # Abort check — CRITICAL bypasses this if [[ "$tier" != "critical" ]]; then if ! check_abort_conditions; then return fi else warn "CRITICAL tier — bypassing abort conditions" fi RECENT_REBOOTS=$(count_recent_reboots) log "Recent reboots in ${SYS_WATCHDOG_REBOOT_WINDOW_HRS}hr window: $RECENT_REBOOTS / $SYS_WATCHDOG_REBOOT_LIMIT" if [[ "$RECENT_REBOOTS" -ge "$SYS_WATCHDOG_REBOOT_LIMIT" ]]; then error "Reboot loop detected — shutting down instead of rebooting" notify "Reboot loop on $(hostname) ($MY_ID) — shutting down after $RECENT_REBOOTS reboots — ${triggers[*]}" \ "System Watchdog" "warning" if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would shutdown now" return fi sync /sbin/poweroff return fi echo "" echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━" echo " $ICON_REBOOT_SMART SYSTEM WATCHDOG — REBOOT TRIGGERED" echo " Tier: ${tier^^}" echo " Host: $MY_ID ($LOCAL_SERVER_NAME)" for t in "${triggers[@]}"; do echo " → $t" done echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━" notify "System watchdog ${tier^^} reboot on $(hostname) ($MY_ID) — ${triggers[*]}" \ "System Watchdog" "warning" if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — reboot sequence would begin now" return fi log_reboot # Graceful shutdown sequence warn "Shutting down VMs..." if command -v virsh >/dev/null 2>&1; then for VM in $(virsh list --name 2>/dev/null); do [[ -z "$VM" ]] && continue virsh shutdown "$VM" >/dev/null 2>&1 done sleep 30 fi warn "Stopping Docker containers..." if command -v docker >/dev/null 2>&1; then timeout 60 docker ps -q 2>/dev/null | xargs -r docker stop >/dev/null 2>&1 fi warn "Stopping User Scripts..." pkill -f "/tmp/user.scripts" 2>/dev/null || true warn "Syncing disks..." sync sleep 5 /sbin/reboot } # ============================================================================================== # ━━━ Clean Shutdown ━━━ # ============================================================================================== WATCHDOG_RUNNING=true cleanup() { echo "" warn "System watchdog received shutdown signal — stopping cleanly" WATCHDOG_RUNNING=false exit 0 } trap cleanup SIGTERM SIGINT # ============================================================================================== # ━━━ Continuous Monitoring Loop ━━━ # ============================================================================================== warn "System watchdog started — $MY_ID — checking every ${SYSTEM_WATCHDOG_INTERVAL}s" echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━" CYCLE=0 while [[ "$WATCHDOG_RUNNING" == true ]]; do (( CYCLE++ )) # Re-source config each cycle — picks up config changes without restart source "$SCRIPT_DIR/../load_config.sh" detect_hosts SYS_WATCHDOG_REBOOT_WINDOW=$(( SYS_WATCHDOG_REBOOT_WINDOW_HRS * 3600 )) TOTAL_CORES=$(nproc) TRIGGERS=() CRITICAL_TRIGGERS=() URGENT_OOM_CONFIRMED=false # ── OOM Delta — read every cycle for bypass decisions ───────────────────────────────────── OOM_DELTA=0 if [[ "$SYS_WATCHDOG_CHECK_OOM" == true ]]; then OOM_DELTA=$(get_oom_delta) [[ "$OOM_DELTA" -gt 0 ]] && \ log "OOM kills this cycle: $OOM_DELTA (limit: ${SYS_WATCHDOG_OOM_LIMIT})" fi # ========================================================================================== # ━━━ TIER 1 — CRITICAL CHECKS (bypass all strikes, reboot immediately) ━━━ # ========================================================================================== # ── Docker daemon — critical: nothing can heal without it ───────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_DOCKER_DAEMON" == true ]]; then if ! timeout "$DOCKER_TIMEOUT" docker info >/dev/null 2>&1; then error "Docker daemon unresponsive — CRITICAL" # Attempt daemon restart before rebooting warn "Attempting Docker daemon restart..." if [[ "$DRY_RUN" == false ]]; then /etc/rc.d/rc.docker restart >/dev/null 2>&1 sleep 15 if timeout "$DOCKER_TIMEOUT" docker info >/dev/null 2>&1; then warn "Docker daemon restarted successfully — continuing monitoring" else error "Docker daemon restart failed — adding to CRITICAL triggers" CRITICAL_TRIGGERS+=("docker_daemon_unresponsive") fi else warn "DRY RUN — would attempt Docker daemon restart" CRITICAL_TRIGGERS+=("docker_daemon_unresponsive") fi else log "Docker daemon healthy ✅" fi fi # ── rootfs critical — at 99%+ writes are failing ───────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_ROOTFS" == true ]]; then ROOTFS_USED=$(df / --output=pcent 2>/dev/null | tail -1 | tr -d ' %') if [[ "$ROOTFS_USED" -ge "${SYS_WATCHDOG_ROOTFS_CRITICAL_PCT:-99}" ]]; then error "rootfs ${ROOTFS_USED}% — CRITICAL (writes failing)" CRITICAL_TRIGGERS+=("rootfs_full=${ROOTFS_USED}%") fi fi # ── Kernel oops/BUG — kernel running with corrupted state ──────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_KERNEL_OOPS" == true ]]; then PREV_OOPS=$(get_state_val "kernel_oops_count") CURRENT_OOPS=$(dmesg 2>/dev/null | grep -cE "BUG:|kernel BUG|Oops:" || echo 0) CURRENT_OOPS="${CURRENT_OOPS//[^0-9]/}"; CURRENT_OOPS="${CURRENT_OOPS:-0}" set_state_val "kernel_oops_count" "$CURRENT_OOPS" if [[ -n "$PREV_OOPS" && "$PREV_OOPS" =~ ^[0-9]+$ ]]; then OOPS_DELTA=$(( CURRENT_OOPS - PREV_OOPS )) if [[ "$OOPS_DELTA" -gt 0 ]]; then error "Kernel oops/BUG detected — $OOPS_DELTA new since last cycle — CRITICAL" CRITICAL_TRIGGERS+=("kernel_oops=${OOPS_DELTA}_new") fi fi fi # ── File descriptor exhaustion — new connections failing silently ───────────────────────── if [[ "$SYS_WATCHDOG_CHECK_FD" == true ]]; then FD_LINE=$(cat /proc/sys/fs/file-nr 2>/dev/null) FD_OPEN=$(echo "$FD_LINE" | awk '{print $1}') FD_MAX=$(echo "$FD_LINE" | awk '{print $3}') if [[ -n "$FD_OPEN" && -n "$FD_MAX" && "$FD_MAX" -gt 0 ]]; then FD_PCT=$(( FD_OPEN * 100 / FD_MAX )) if [[ "$FD_PCT" -ge "${SYS_WATCHDOG_FD_CRITICAL_PCT:-95}" ]]; then error "File descriptors ${FD_PCT}% exhausted (${FD_OPEN}/${FD_MAX}) — CRITICAL" CRITICAL_TRIGGERS+=("fd_exhaustion=${FD_PCT}%") else log "File descriptors: ${FD_PCT}% (${FD_OPEN}/${FD_MAX})" fi fi fi # ── /boot read-only — state and config writes failing silently ──────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_BOOT" == true ]]; then BOOT_TEST="/boot/.watchdog_write_test" if ! touch "$BOOT_TEST" 2>/dev/null; then error "/boot is read-only — config writes failing silently — CRITICAL" CRITICAL_TRIGGERS+=("boot_read_only") else rm -f "$BOOT_TEST" 2>/dev/null log "/boot is writable ✅" fi fi # ── Act on CRITICAL triggers immediately ───────────────────────────────────────────────── if [[ ${#CRITICAL_TRIGGERS[@]} -gt 0 ]]; then echo "" echo "━━━ $ICON_ERROR CRITICAL — IMMEDIATE REBOOT — Cycle $CYCLE ━━━" for t in "${CRITICAL_TRIGGERS[@]}"; do error " CRITICAL: $t" done do_reboot "critical" "${CRITICAL_TRIGGERS[@]}" continue fi # ========================================================================================== # ━━━ TIER 3 — STANDARD CHECKS (strike system) ━━━ # ========================================================================================== # ── rootfs standard ────────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_ROOTFS" == true ]]; then ROOTFS_USED=$(df / --output=pcent 2>/dev/null | tail -1 | tr -d ' %') TRIGGERED=false [[ "$ROOTFS_USED" -ge "$SYS_WATCHDOG_ROOTFS_PCT" ]] && TRIGGERED=true run_strike_check "rootfs" "$TRIGGERED" "rootfs ${ROOTFS_USED}%" && \ TRIGGERS+=("rootfs=${ROOTFS_USED}%") fi # ── /var/log ───────────────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_LOG" == true ]]; then LOG_USED=$(df -P /var/log 2>/dev/null | awk 'NR==2 {print $5}' | tr -d '%') TRIGGERED=false [[ "${LOG_USED:-0}" -ge "$SYS_WATCHDOG_LOG_PCT" ]] && TRIGGERED=true run_strike_check "log" "$TRIGGERED" "/var/log ${LOG_USED}%" && \ TRIGGERS+=("log=${LOG_USED}%") fi # ── /tmp ───────────────────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_TMP" == true ]]; then TMP_USED=$(df -P /tmp 2>/dev/null | awk 'NR==2 {print $5}' | tr -d '%') if [[ "${TMP_USED:-0}" -ge "${SYS_WATCHDOG_TMP_CRITICAL_PCT:-98}" ]]; then # Try to clear before escalating warn "/tmp ${TMP_USED}% — attempting cleanup..." find /tmp -type f -mmin +60 -not -name "*.lock" -delete 2>/dev/null TMP_USED_AFTER=$(df -P /tmp 2>/dev/null | awk 'NR==2 {print $5}' | tr -d '%') if [[ "${TMP_USED_AFTER:-0}" -ge "${SYS_WATCHDOG_TMP_CRITICAL_PCT:-98}" ]]; then error "/tmp still ${TMP_USED_AFTER}% after cleanup — adding to triggers" TRIGGERED=true else warn "/tmp cleared to ${TMP_USED_AFTER}% ✅" TRIGGERED=false fi elif [[ "${TMP_USED:-0}" -ge "${SYS_WATCHDOG_TMP_PCT:-90}" ]]; then TRIGGERED=true else TRIGGERED=false fi run_strike_check "tmp" "$TRIGGERED" "/tmp ${TMP_USED}%" && \ TRIGGERS+=("tmp=${TMP_USED}%") fi # ── RAM tiers ───────────────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_RAM" == true ]]; then MEM_KB=$(awk '/MemAvailable/ {print $2}' /proc/meminfo) MEM_GB=$(( MEM_KB / 1024 / 1024 )) MEM_SHUTDOWN_ACTIVE=$(get_state_val "mem_shutdown_active") if [[ "$MEM_GB" -lt "$SYS_WATCHDOG_MEM_GB" ]]; then # Tier 2 check — bypass if OOM confirms crisis if [[ "$SYS_WATCHDOG_CHECK_OOM" == true ]] && \ [[ "$OOM_DELTA" -ge "$SYS_WATCHDOG_OOM_LIMIT" ]]; then error "RAM ${MEM_GB}GB + ${OOM_DELTA} OOM kills this cycle — URGENT bypass" OOM_VICTIMS=$(get_oom_victims) URGENT_TRIGGERS=("urgent_low_ram=${MEM_GB}GB" "oom_kills=${OOM_DELTA}") [[ -n "$OOM_VICTIMS" ]] && URGENT_TRIGGERS+=("oom_victims: $OOM_VICTIMS") do_reboot "urgent" "${URGENT_TRIGGERS[@]}" continue fi # Standard strike path run_strike_check "ram" true "RAM ${MEM_GB}GB free" && \ TRIGGERS+=("low_ram=${MEM_GB}GB") elif [[ "$MEM_GB" -lt "$SYS_WATCHDOG_MEM_SHUTDOWN_GB" ]]; then reset_strikes "ram" # Container shutdown tier — but only once per event if [[ "$MEM_SHUTDOWN_ACTIVE" != "true" ]]; then warn "RAM ${MEM_GB}GB — below shutdown threshold ${SYS_WATCHDOG_MEM_SHUTDOWN_GB}GB" run_strike_check "ram_shutdown" true "RAM shutdown tier ${MEM_GB}GB" && \ shutdown_non_essential_containers else # Already shutdown — check if recovered if [[ "$MEM_GB" -ge "$SYS_WATCHDOG_MEM_RECOVER_GB" ]]; then warn "RAM recovered to ${MEM_GB}GB — clearing emergency state" restart_non_essential_containers reset_strikes "ram_shutdown" else warn "RAM ${MEM_GB}GB — still in emergency shutdown (recover threshold: ${SYS_WATCHDOG_MEM_RECOVER_GB}GB)" fi fi elif [[ "$MEM_GB" -lt "$SYS_WATCHDOG_MEM_WARN_GB" ]]; then reset_strikes "ram" reset_strikes "ram_shutdown" warn "RAM ${MEM_GB}GB — below warning threshold ${SYS_WATCHDOG_MEM_WARN_GB}GB" local prev_ram_warn prev_ram_warn=$(get_strikes "ram_warn_notified") if [[ "${prev_ram_warn:-0}" -eq 0 ]]; then notify "RAM warning on $(hostname) ($MY_ID) — ${MEM_GB}GB free (threshold: ${SYS_WATCHDOG_MEM_WARN_GB}GB)" \ "System Watchdog" "warning" set_strikes "ram_warn_notified" 1 fi else reset_strikes "ram" reset_strikes "ram_shutdown" set_strikes "ram_warn_notified" 0 log "RAM ${MEM_GB}GB free ✅" fi fi # ── ZFS ARC ────────────────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_ARC" == true ]] && [[ -f /proc/spl/kstat/zfs/arcstats ]]; then ARC_SIZE=$(awk '/^size / {print $3}' /proc/spl/kstat/zfs/arcstats) ARC_MAX=$(awk '/^c_max / {print $3}' /proc/spl/kstat/zfs/arcstats) ARC_PCT=$(( ARC_SIZE * 100 / ARC_MAX )) TRIGGERED=false if [[ "$ARC_PCT" -ge "$SYS_WATCHDOG_ARC_PINNED_PCT" ]]; then sync; echo 3 > /proc/sys/vm/drop_caches; sleep 5 ARC_AFTER=$(awk '/^size / {print $3}' /proc/spl/kstat/zfs/arcstats) ARC_AFTER_PCT=$(( ARC_AFTER * 100 / ARC_MAX )) [[ "$ARC_AFTER_PCT" -ge "$SYS_WATCHDOG_ARC_RELEASE_PCT" ]] && TRIGGERED=true fi run_strike_check "arc" "$TRIGGERED" "ZFS ARC pinned ${ARC_PCT}%" && \ TRIGGERS+=("arc_pinned=${ARC_PCT}%") fi # ── CPU temperature ─────────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_CPU_TEMP" == true ]]; then CPU_TEMP="" if command -v sensors >/dev/null 2>&1; then CPU_TEMP=$(sensors 2>/dev/null | \ grep -i "Package id 0\|Tctl\|CPU Temp" | \ awk '{print $NF}' | tr -d '+°C' | head -1) fi if [[ -n "$CPU_TEMP" ]]; then CPU_TEMP_INT=$(printf "%.0f" "$CPU_TEMP") TRIGGERED=false [[ "$CPU_TEMP_INT" -ge "$SYS_WATCHDOG_CPU_TEMP_MAX" ]] && TRIGGERED=true run_strike_check "cpu_temp" "$TRIGGERED" "CPU temp ${CPU_TEMP_INT}°C" && \ TRIGGERS+=("cpu_temp=${CPU_TEMP_INT}C") fi fi # ── Load average ───────────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_LOAD" == true ]]; then LOAD=$(awk '{print $1}' /proc/loadavg) LOAD_INT=$(printf "%.0f" "$LOAD") LOAD_THRESHOLD=$(( TOTAL_CORES * SYS_WATCHDOG_LOAD_MULTIPLIER )) TRIGGERED=false [[ "$LOAD_INT" -ge "$LOAD_THRESHOLD" ]] && TRIGGERED=true run_strike_check "load" "$TRIGGERED" "load avg ${LOAD}" && \ TRIGGERS+=("load=${LOAD}") fi # ── Zombie processes ───────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_ZOMBIES" == true ]]; then ZOMBIE_COUNT=$(ps aux 2>/dev/null | awk '{print $8}' | grep -c "^Z$" || echo 0) ZOMBIE_COUNT="${ZOMBIE_COUNT//[^0-9]/}"; ZOMBIE_COUNT="${ZOMBIE_COUNT:-0}" TRIGGERED=false [[ "$ZOMBIE_COUNT" -ge "$SYS_WATCHDOG_ZOMBIE_LIMIT" ]] && TRIGGERED=true run_strike_check "zombies" "$TRIGGERED" "zombies ${ZOMBIE_COUNT}" && \ TRIGGERS+=("zombies=${ZOMBIE_COUNT}") fi # ── Array disk errors — accumulating mdstat errors ──────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_MDSTAT" == true ]]; then PREV_MD_ERRORS=$(get_state_val "mdstat_errors") CURRENT_MD_ERRORS=$(grep -oP "(?<=\[)[^\]]*[U_][^\]]*(?=\])" \ /proc/mdstat 2>/dev/null | grep -o "_" | wc -l || echo 0) CURRENT_MD_ERRORS="${CURRENT_MD_ERRORS//[^0-9]/}"; CURRENT_MD_ERRORS="${CURRENT_MD_ERRORS:-0}" set_state_val "mdstat_errors" "$CURRENT_MD_ERRORS" if [[ -n "$PREV_MD_ERRORS" && "$PREV_MD_ERRORS" =~ ^[0-9]+$ ]]; then MD_DELTA=$(( CURRENT_MD_ERRORS - PREV_MD_ERRORS )) if [[ "$MD_DELTA" -ge "${SYS_WATCHDOG_MDSTAT_ERROR_LIMIT:-5}" ]]; then TRIGGERED=true run_strike_check "mdstat" "$TRIGGERED" \ "mdstat errors +${MD_DELTA} (total: ${CURRENT_MD_ERRORS})" && \ TRIGGERS+=("mdstat_errors=+${MD_DELTA}") else run_strike_check "mdstat" false "mdstat" > /dev/null fi fi fi # ── Network interface state ─────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_NETWORK" == true ]]; then NIC="${SYS_WATCHDOG_NIC:-eth0}" NIC_STATE=$(cat "/sys/class/net/${NIC}/operstate" 2>/dev/null || echo "unknown") TRIGGERED=false [[ "$NIC_STATE" != "up" ]] && TRIGGERED=true run_strike_check "network" "$TRIGGERED" "${NIC} state: ${NIC_STATE}" && \ TRIGGERS+=("nic_down=${NIC}") fi # ── sshd — try restart before escalating ───────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_SSHD" == true ]]; then if ! pgrep -x sshd >/dev/null 2>&1; then warn "sshd not running — attempting restart..." if [[ "$DRY_RUN" == false ]]; then /etc/rc.d/rc.sshd start >/dev/null 2>&1 sleep 3 if pgrep -x sshd >/dev/null 2>&1; then warn "sshd restarted successfully ✅" reset_strikes "sshd" notify "sshd was down on $(hostname) ($MY_ID) — restarted automatically" \ "System Watchdog" "warning" else error "sshd restart failed — remote access unavailable" run_strike_check "sshd" true "sshd not running" && \ TRIGGERS+=("sshd_down") fi else warn "DRY RUN — would restart sshd" fi else reset_strikes "sshd" fi fi # ── Runaway process ─────────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_RUNAWAY" == true ]]; then RUNAWAY_PCT="${SYS_WATCHDOG_RUNAWAY_CPU_PCT:-90}" TOP_CPU_PCT=$(ps aux 2>/dev/null | awk 'NR>1{print $3}' | sort -rn | head -1) TOP_CPU_INT=$(printf "%.0f" "${TOP_CPU_PCT:-0}") TOP_CPU_NAME=$(ps aux 2>/dev/null | sort -k3 -rn | awk 'NR==2{print $11}') TRIGGERED=false [[ "$TOP_CPU_INT" -ge "$RUNAWAY_PCT" ]] && TRIGGERED=true # Runaway uses SYS_WATCHDOG_RUNAWAY_STRIKES not global strike limit if [[ "$TRIGGERED" == true ]]; then RAWAY_S=$(increment_strikes "runaway") RLIMIT="${SYS_WATCHDOG_RUNAWAY_STRIKES:-3}" warn "Runaway ${TOP_CPU_NAME} ${TOP_CPU_PCT}% CPU -- strike $RAWAY_S/$RLIMIT" if (( RAWAY_S >= RLIMIT )); then error "Runaway process ${TOP_CPU_NAME} -- strike limit hit" reset_strikes "runaway" TRIGGERS+=("runaway=${TOP_CPU_NAME}@${TOP_CPU_PCT}%") fi else RAWAY_CUR=$(get_strikes "runaway") [[ "${RAWAY_CUR:-0}" -gt 0 ]] && reset_strikes "runaway" fi fi # ── Required containers from docker_watchdog skip list ──────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_CONTAINERS" == true ]] && [[ -s "$SYS_WATCHDOG_FAILED_FILE" ]]; then FAILED_CONTAINERS=() while IFS= read -r container; do [[ -z "$container" ]] && continue STATUS=$(timeout "$DOCKER_TIMEOUT" docker inspect -f \ '{{.State.Running}}' "$container" 2>/dev/null || echo "unknown") [[ "$STATUS" != "true" ]] && FAILED_CONTAINERS+=("$container") done < "$SYS_WATCHDOG_FAILED_FILE" TRIGGERED=false [[ ${#FAILED_CONTAINERS[@]} -gt 0 ]] && TRIGGERED=true run_strike_check "failed_containers" "$TRIGGERED" \ "required containers stopped: ${FAILED_CONTAINERS[*]:-}" && \ TRIGGERS+=("containers=${FAILED_CONTAINERS[*]:-}") fi # ========================================================================================== # ━━━ Evaluate Standard Triggers ━━━ # ========================================================================================== if [[ ${#TRIGGERS[@]} -gt 0 ]]; then echo "" echo "━━━ $ICON_REBOOT_SMART System Watchdog — Cycle $CYCLE — $(date '+%Y-%m-%d %H:%M:%S') ━━━" for t in "${TRIGGERS[@]}"; do echo " $ICON_REBOOT_SMART $t" done [[ "$OOM_DELTA" -gt 0 ]] && echo " OOM kills this cycle: $OOM_DELTA" echo "" do_reboot "standard" "${TRIGGERS[@]}" else log "Cycle $CYCLE — system healthy ($(date '+%H:%M:%S'))" # Heartbeat — periodic proof of life if [[ "${SYSTEM_WATCHDOG_HEARTBEAT:-true}" == true ]]; then HB_SECONDS=$(( ${SYSTEM_WATCHDOG_HEARTBEAT_HOURS:-1} * 3600 )) UPTIME_APPROX=$(( CYCLE * SYSTEM_WATCHDOG_INTERVAL )) if [[ "$HB_SECONDS" -gt 0 ]] && \ (( UPTIME_APPROX % HB_SECONDS < SYSTEM_WATCHDOG_INTERVAL )) && \ [[ "$UPTIME_APPROX" -gt 0 ]]; then HB_HR=$(( UPTIME_APPROX / 3600 )) warn "♥ system_watchdog alive — $MY_ID — ~${HB_HR}hr uptime ($(date '+%H:%M:%S'))" fi fi fi # ── State file heartbeat — keep mtime fresh every cycle ─────────────────────────────────── # docker_watchdog.sh uses state file mtime to detect stale RAM emergency flags. # If all checks pass with no set_state_val calls (e.g. KERNEL_OOPS + MDSTAT both disabled), # mtime would not update and stale guard would incorrectly resume docker_watchdog.sh. # Writing watchdog_cycle each tick guarantees mtime stays current while watchdog runs. set_state_val "watchdog_cycle" "$CYCLE" # Sleep until next cycle — interruptible by SIGTERM sleep "$SYSTEM_WATCHDOG_INTERVAL" & wait $! done