#!/bin/bash # ============================================================================================== # ================================= Stability Watchdog ========================================= # ============================================================================================== # # PURPOSE # ───────────────────────────────────────────────────────────────────────────── # Last line of defense — reboots the system cleanly if it is about to become # unstable. Called by watchdog_orchestrator.sh via cron every 15 minutes as a # single-pass run. Works alongside docker_watchdog.sh which handles # container-level healing first. Only escalates to reboot when # docker_watchdog.sh cannot resolve the condition. # # ============================================================================================== # OPERATIONAL MODEL # ============================================================================================== # # Three-Tier Response System # # Tier 1 — CRITICAL (bypass all strikes, reboot immediately) # rootfs at 99%+ — writes failing; SSH may stop; no recovery options # Kernel oops/BUG in dmesg — kernel running with corrupted state # File descriptor exhaustion — new connections and processes failing silently # /boot read-only unexpectedly — state files and config writes silently failing # (Docker daemon: owned by docker_watchdog — writes daemon_confirmed_down flag → standard strikes) # # Tier 2 — URGENT (bypass strikes when OOM confirms active crisis) # RAM < MEM_GB AND OOM kills >= OOM_LIMIT in this cycle. # OOM kills at this rate means the system is dying faster than watchdogs can heal. # Without OOM confirmation → standard strike system applies. # # Tier 3 — STANDARD (N consecutive failures → reboot) # RAM tiers, load, CPU temp, zombies, /var/log, /tmp, containers, NIC, mdstat. # # RAM Tiers # MEM_WARN_GB (10GB) — warn + notify only # MEM_SHUTDOWN_GB (6GB) — stop non-essential containers, wait for recovery # MEM_GB (4GB) — strike system → reboot (bypass with OOM confirmation) # MEM_RECOVER_GB (30GB) — RAM must reach this before stopped containers restart # # Container Shutdown Logic (at MEM_SHUTDOWN_GB) # Stops all containers not in SYS_WATCHDOG_MEM_SHUTDOWN_EXCLUDED. # Stopped containers tracked in shutdown list — won't restart until RAM recovers. # Strike system prevents flip-flopping — shutdown only once per degradation event. # # Abort Conditions (prevent reboot during sensitive operations) # ZFS pool unhealthy, parity running, mover running — each toggleable. # CRITICAL tier bypasses all abort conditions — imminent crash overrides data safety. # # Checks Run Every Cycle # rootfs usage, /var/log, /tmp, free RAM, ZFS ARC, CPU temp, load avg, # zombie processes, Docker daemon, OOM rate, /boot read-only, kernel oops, # file descriptor exhaustion, array disk errors, NIC state. # (Container health is owned by docker_watchdog — not checked here.) # # ============================================================================================== # DESIGN PRINCIPLES # ============================================================================================== # # Last Line of Defense # Every other watchdog tries to heal a specific subsystem. This one assumes # those attempts have already failed and holds the only irreversible remedy in # the ecosystem — a reboot. That authority is why nearly every check here is # gated behind strikes, tiers and abort conditions. # # Evidence Before Reboot # Strikes are the default; bypassing them requires corroboration, not just a # worse number. Tier 2 needs low RAM AND active OOM kills before it acts — # low RAM alone is a reading, low RAM plus processes being killed is a crisis. # # Unrecoverable Conditions Skip the Queue # Tier 1 conditions share one property: the system cannot heal from them and # waiting makes recovery less likely. A full rootfs or a kernel oops degrades # further every cycle, and strike-counting through it only guarantees the # reboot happens from a worse state. # # Data Safety Outranks Uptime # Reboots abort while a ZFS pool is unhealthy, parity is running, or the mover # is active. Interrupting those risks the data itself, which no amount of # uptime justifies. Tier 1 is the sole exception — an imminent crash will # interrupt them anyway, less gracefully. # # Clear Ownership Boundaries # Container health belongs to docker_watchdog.sh and is deliberately not # checked here. The Docker daemon check writes a flag for docker_watchdog # rather than acting on it. Two watchdogs remediating the same subsystem # would race. # # ============================================================================================== # OPERATIONAL SAFEGUARDS # ============================================================================================== # # Root Required # Reboot and container stop require root. # # Single Instance Lock # acquire_lock prevents a second watchdog instance from starting. Two instances # could each count strikes against the same condition and reach the reboot # threshold in half the intended time. # # State File Verification # All state files verified writable at startup — errors if any cannot be created. # Strike counts and the reboot log live in these files; if they silently failed # to persist, every cycle would look like strike 1 and the reboot rate limit # would never accumulate. # # Reboot Rate Limiting # No more than SYS_WATCHDOG_REBOOT_LIMIT reboots within # SYS_WATCHDOG_REBOOT_WINDOW_HRS. On hitting the limit the host powers off # instead of rebooting — a fault that survives repeated reboots will not be # fixed by more of them, and a box cycling endlessly is worse than one that # is cleanly down and obviously needs attention. # # Abort Conditions # Reboots are aborted while a ZFS pool is unhealthy, parity is running, or the # mover is active — each individually toggleable. Interrupting any of these # risks the data itself. # # Critical Tier Override # Tier 1 conditions bypass both strikes and abort conditions. These are states # the system cannot recover from and which degrade every cycle; waiting only # guarantees the eventual reboot happens from a worse position. # # Strike Threshold # Tier 3 requires SYS_WATCHDOG_STRIKE_LIMIT consecutive failing cycles. A # single bad sample — a momentary load spike, a transient RAM dip — never # reboots the system. # # OOM Corroboration # Tier 2 escalation requires low RAM AND active OOM kills in the same cycle. # Low RAM alone stays in the strike system. # # Ownership Boundary # Container health is not checked here — docker_watchdog.sh owns it. The Docker # daemon check writes daemon_confirmed_down for docker_watchdog rather than # remediating, so the two never act on the same subsystem. # # Aborted-Reboot Recovery # An EXIT trap is armed the moment containers start being stopped for a reboot # and disarmed only once the reboot is committed. If the script dies anywhere # in between, the trap restarts everything it stopped — the failure mode is a # running system, never a host left with all containers down and no reboot. # # Sync Before Reboot # sync is issued before both /sbin/poweroff and /sbin/reboot so pending writes # are flushed. Container stop is additionally bounded by a 60 second timeout so # one unresponsive container cannot hold the shutdown sequence open forever. # # Dry Run Support # --dry-run runs the full detection path and reports the reboot or shutdown # that would occur without issuing either. # # ============================================================================================== # CONFIGURATION # ============================================================================================== # # master.conf — System Watchdog section # Full variable listing in master.conf. Key variables: # # SYS_WATCHDOG_REBOOT_WINDOW_HRS — reboot rate limit window (default: 12) # SYS_WATCHDOG_REBOOT_LIMIT — max reboots in window before giving up (default: 3) # SYS_WATCHDOG_STRIKE_LIMIT — consecutive failures before reboot (default: 2) # SYS_WATCHDOG_OOM_LIMIT — OOM kills/cycle to trigger URGENT bypass (default: 3) # SYS_WATCHDOG_MEM_SHUTDOWN_EXCLUDED — containers exempt from memory shutdown # # ============================================================================================== # STATE FILES # ============================================================================================== # # SYS_WATCHDOG_STATE_FILE — strike counters and cycle state # SYS_WATCHDOG_REBOOT_LOG — reboot history for rate limiting # SYS_WATCHDOG_OOM_FILE — OOM kill counter from previous cycle # # ============================================================================================== # RUNTIME MODES # ============================================================================================== # # stability_watchdog.sh # Single-pass stability check — called by watchdog_orchestrator.sh every 15 min # # stability_watchdog.sh --dry-run # Run detection logic without rebooting or stopping containers. # # stability_watchdog.sh --status # Show config, thresholds, current system state, and strike counts. # # stability_watchdog.sh --log # Verbose per-cycle output — show every check result and threshold comparison. # # ============================================================================================== SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" source "$SCRIPT_DIR/../load_config.sh" parse_args "$@" # ============================================================================================== # ━━━ Setup — runs once at start ━━━ # ============================================================================================== if [[ "$EUID" -ne 0 ]]; then error "Must be run as root" exit 1 fi acquire_lock detect_hosts TOTAL_CORES=$(nproc) SYS_WATCHDOG_REBOOT_WINDOW=$(( SYS_WATCHDOG_REBOOT_WINDOW_HRS * 3600 )) # Ensure state files exist for state_file in "$SYS_WATCHDOG_STATE_FILE" "$SYS_WATCHDOG_REBOOT_LOG" \ "$SYS_WATCHDOG_OOM_FILE"; do touch "$state_file" 2>/dev/null || { error "Cannot create state file: $state_file" exit 1 } done [[ "$DRY_RUN" == true ]] && warn "DRY RUN — no reboots or container shutdowns will occur" log "$ICON_GEAR Config: strikes=${SYS_WATCHDOG_STRIKE_LIMIT} reboot-limit=${SYS_WATCHDOG_REBOOT_LIMIT}/${SYS_WATCHDOG_REBOOT_WINDOW_HRS}hr oom-limit=${SYS_WATCHDOG_OOM_LIMIT}" log "$ICON_GEAR Tiers: rootfs-crit=${SYS_WATCHDOG_ROOTFS_CRITICAL_PCT}% rootfs-warn=${SYS_WATCHDOG_ROOTFS_PCT}% ram-reboot=${SYS_WATCHDOG_MEM_GB}GB load=${SYS_WATCHDOG_LOAD_MULTIPLIER}x(${TOTAL_CORES}cores) cpu-temp=${SYS_WATCHDOG_CPU_TEMP_MAX}°C zombies=${SYS_WATCHDOG_ZOMBIE_LIMIT}" # ============================================================================================== # ━━━ Status ━━━ # ============================================================================================== if [[ "$SHOW_STATUS" == true ]]; then echo "" echo "━━━━━ $ICON_SUMMARY STABILITY WATCHDOG STATUS ━━━━━" echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)" echo "" echo "── Tier 1 — CRITICAL (bypass strikes immediately) ──" echo "$ICON_DISK rootfs critical: ${SYS_WATCHDOG_ROOTFS_CRITICAL_PCT}%" echo "$ICON_GEAR FD critical: ${SYS_WATCHDOG_FD_CRITICAL_PCT}%" echo "$ICON_GEAR /boot read-only: check=${SYS_WATCHDOG_CHECK_BOOT}" echo "$ICON_GEAR Kernel oops: check=${SYS_WATCHDOG_CHECK_KERNEL_OOPS}" echo "$ICON_CONTAINERS Docker daemon: check=${SYS_WATCHDOG_CHECK_DOCKER_DAEMON}" echo "" echo "── Tier 2 — URGENT (bypass strikes with OOM confirmation) ──" echo "$ICON_MEM RAM critical: < ${SYS_WATCHDOG_MEM_GB}GB" echo "$ICON_GEAR OOM limit: ${SYS_WATCHDOG_OOM_LIMIT} kills/cycle" echo "" echo "── Tier 3 — STANDARD (strike system) ──" echo "$ICON_DISK rootfs warn: ${SYS_WATCHDOG_ROOTFS_PCT}%" echo "$ICON_GEAR /var/log warn: ${SYS_WATCHDOG_LOG_PCT}%" echo "$ICON_GEAR /tmp warn: ${SYS_WATCHDOG_TMP_PCT}%" echo "$ICON_MEM RAM reboot: < ${SYS_WATCHDOG_MEM_GB}GB (+ strikes)" echo " (RAM warn/shutdown/recover managed by resource_watchdog.sh)" echo "$ICON_ZFS ARC pinned: ${SYS_WATCHDOG_ARC_PINNED_PCT}%" echo "$ICON_GEAR Load multiplier: ${SYS_WATCHDOG_LOAD_MULTIPLIER}x (= $(( TOTAL_CORES * SYS_WATCHDOG_LOAD_MULTIPLIER )) on $TOTAL_CORES cores)" echo "$ICON_GEAR Zombie limit: ${SYS_WATCHDOG_ZOMBIE_LIMIT}" echo "$ICON_GEAR CPU temp max: ${SYS_WATCHDOG_CPU_TEMP_MAX}°C" echo "$ICON_GEAR Strike limit: ${SYS_WATCHDOG_STRIKE_LIMIT} cycles" echo "$ICON_TIME Schedule: every 15 min (cron via watchdog_orchestrator)" echo "$ICON_REBOOT_SMART Reboot limit: ${SYS_WATCHDOG_REBOOT_LIMIT} in ${SYS_WATCHDOG_REBOOT_WINDOW_HRS}hr" echo "" echo "" echo "── Check Toggles ──" echo " rootfs=$SYS_WATCHDOG_CHECK_ROOTFS log=$SYS_WATCHDOG_CHECK_LOG ram=$SYS_WATCHDOG_CHECK_RAM" echo " arc=$SYS_WATCHDOG_CHECK_ARC cpu_temp=$SYS_WATCHDOG_CHECK_CPU_TEMP load=$SYS_WATCHDOG_CHECK_LOAD" echo " zombies=$SYS_WATCHDOG_CHECK_ZOMBIES docker=$SYS_WATCHDOG_CHECK_DOCKER_DAEMON" echo " oom=$SYS_WATCHDOG_CHECK_OOM" echo " tmp=$SYS_WATCHDOG_CHECK_TMP fd=$SYS_WATCHDOG_CHECK_FD boot=$SYS_WATCHDOG_CHECK_BOOT" echo " kernel_oops=$SYS_WATCHDOG_CHECK_KERNEL_OOPS sshd=$SYS_WATCHDOG_CHECK_SSHD" echo " network=$SYS_WATCHDOG_CHECK_NETWORK mdstat=$SYS_WATCHDOG_CHECK_MDSTAT" echo " runaway=$SYS_WATCHDOG_CHECK_RUNAWAY" echo "━━━━━━━━━━━━━━━━━━━━━━━" exit 0 fi # ============================================================================================== # ── STATE HELPERS ───────────────────────────────────────────────────────────────────────────── # ============================================================================================== # Wrap common.sh's wd_state_get()/wd_state_set() — file is implicit here (this script's # own state file), unlike the generic helper which always takes it as an argument. get_strikes() { wd_state_get "$1" "$SYS_WATCHDOG_STATE_FILE" } set_strikes() { local key="$1" count="$2" wd_state_set "$key" "$count" "$SYS_WATCHDOG_STATE_FILE" } increment_strikes() { local key="$1" local current current=$(get_strikes "$key") [[ -z "$current" ]] && current=0 (( current++ )) set_strikes "$key" "$current" echo "$current" } reset_strikes() { set_strikes "$1" 0 } get_state_val() { wd_state_get "$1" "$SYS_WATCHDOG_STATE_FILE" "=" } set_state_val() { local key="$1" val="$2" wd_state_set "$key" "$val" "$SYS_WATCHDOG_STATE_FILE" "=" } purge_old_reboots() { local now cutoff now=$(date +%s) cutoff=$(( now - SYS_WATCHDOG_REBOOT_WINDOW )) grep -v "^$" "$SYS_WATCHDOG_REBOOT_LOG" 2>/dev/null | while IFS= read -r ts; do [[ "$ts" -gt "$cutoff" ]] && echo "$ts" done > "${SYS_WATCHDOG_REBOOT_LOG}.tmp" mv "${SYS_WATCHDOG_REBOOT_LOG}.tmp" "$SYS_WATCHDOG_REBOOT_LOG" } # Always emits exactly one number, because the caller compares it against the reboot limit and a # malformed value there decides whether a looping server reboots again or shuts down. # # Two failure paths have to collapse to 0, and the old one-liner got both wrong. An EMPTY log — # the normal state — made grep -c print 0 and exit 1, so "|| echo 0" fired as well and the # function returned two lines. A MISSING log makes grep print nothing at all, so "|| true" alone # would return the empty string. Either way the caller's [[ -ge ]] dies with a syntax error. count_recent_reboots() { purge_old_reboots local n n=$(grep -c "." "$SYS_WATCHDOG_REBOOT_LOG" 2>/dev/null || true) n="${n//[^0-9]/}" echo "${n:-0}" } log_reboot() { date +%s >> "$SYS_WATCHDOG_REBOOT_LOG" } # ============================================================================================== # ── OOM TRACKING ────────────────────────────────────────────────────────────────────────────── # ============================================================================================== # Reads /proc/vmstat oom_kill counter — delta per cycle = rate of OOM kills # Used for Tier 2 bypass and diagnostic context in reboot messages get_oom_delta() { local current_oom current_oom=$(grep "^oom_kill " /proc/vmstat 2>/dev/null | awk '{print $2}') [[ -z "$current_oom" ]] && echo 0 && return local prev_oom prev_oom=$(cat "$SYS_WATCHDOG_OOM_FILE" 2>/dev/null || echo 0) echo "$current_oom" > "$SYS_WATCHDOG_OOM_FILE" local delta=$(( current_oom - prev_oom )) [[ "$delta" -lt 0 ]] && delta=0 # counter reset on reboot echo "$delta" } get_oom_victims() { # Get process names from dmesg that were OOM killed this boot dmesg -T 2>/dev/null | grep -i "Killed process" | \ awk '{print $NF}' | sort | uniq -c | sort -rn | head -5 | \ awk '{printf "%s×%d ", $2, $1}' | sed 's/ $//' } # ============================================================================================== # ── ABORT CONDITIONS ────────────────────────────────────────────────────────────────────────── # ============================================================================================== # Returns 1 if reboot should be aborted, 0 if reboot should proceed # CRITICAL tier bypasses this function entirely check_abort_conditions() { local should_abort=false if command -v zpool >/dev/null 2>&1; then local unhealthy unhealthy=$(zpool list -H -o health 2>/dev/null | grep -v ONLINE || true) if [[ -n "$unhealthy" ]]; then if [[ "$SYS_WATCHDOG_ABORT_ON_ZFS_UNHEALTHY" == true ]]; then error "ZFS pool unhealthy — aborting reboot to prevent data loss" notify "System watchdog aborted reboot on $(hostname) ($MY_ID) — ZFS pool unhealthy" \ "System Watchdog" "warning" should_abort=true else warn "ZFS pool unhealthy — continuing reboot (ABORT_ON_ZFS_UNHEALTHY=false)" fi fi fi _parity_running=false platform_is_maintenance_running && _parity_running=true if [[ "$_parity_running" == true ]]; then if [[ "$SYS_WATCHDOG_ABORT_ON_PARITY" == true ]]; then error "Parity check running — aborting reboot" notify "System watchdog aborted reboot on $(hostname) ($MY_ID) — parity running" \ "System Watchdog" "warning" should_abort=true else warn "Parity check running — continuing reboot (ABORT_ON_PARITY=false)" fi fi if platform_is_mover_running; then if [[ "$SYS_WATCHDOG_ABORT_ON_MOVER" == true ]]; then error "Mover running — aborting reboot" notify "System watchdog aborted reboot on $(hostname) ($MY_ID) — mover running" \ "System Watchdog" "warning" should_abort=true else warn "Mover running — continuing reboot (ABORT_ON_MOVER=false)" fi fi [[ "$should_abort" == true ]] && return 1 return 0 } # ============================================================================================== # ── STANDARD STRIKE CHECK ───────────────────────────────────────────────────────────────────── # ============================================================================================== # Returns 0 = reboot now | 1 = not yet run_strike_check() { local key="$1" triggered="$2" description="$3" if [[ "$triggered" == true ]]; then local strikes strikes=$(increment_strikes "$key") warn "$description — strike $strikes/$SYS_WATCHDOG_STRIKE_LIMIT" if (( strikes >= SYS_WATCHDOG_STRIKE_LIMIT )); then error "$description — strike limit hit, reboot triggered" reset_strikes "$key" return 0 fi else local current current=$(get_strikes "$key") [[ -n "$current" && "$current" -gt 0 ]] && reset_strikes "$key" fi return 1 } # ── Exit Trap — restart containers stopped before an aborted reboot ─────────────────────────── _SYS_REBOOT_STOPPED=() _trap_sys_reboot_restart() { [[ ${#_SYS_REBOOT_STOPPED[@]} -eq 0 ]] && return warn "Exit trap: restarting containers stopped before aborted reboot" for c in "${_SYS_REBOOT_STOPPED[@]}"; do [[ -z "$c" ]] && continue docker inspect "$c" >/dev/null 2>&1 && docker start "$c" >/dev/null 2>&1 || true done } # ============================================================================================== # ── DO REBOOT ───────────────────────────────────────────────────────────────────────────────── # ============================================================================================== # tier: "critical" (bypass abort) | "urgent" | "standard" do_reboot() { local tier="${1:-standard}" shift local triggers=("$@") # Get OOM context for reboot message local oom_victims="" if [[ "$SYS_WATCHDOG_CHECK_OOM" == true ]]; then oom_victims=$(get_oom_victims) [[ -n "$oom_victims" ]] && triggers+=("oom_victims: $oom_victims") fi # Abort check — CRITICAL bypasses this if [[ "$tier" != "critical" ]]; then if ! check_abort_conditions; then return fi else warn "CRITICAL tier — bypassing abort conditions" fi RECENT_REBOOTS=$(count_recent_reboots) log "Recent reboots in ${SYS_WATCHDOG_REBOOT_WINDOW_HRS}hr window: $RECENT_REBOOTS / $SYS_WATCHDOG_REBOOT_LIMIT" if [[ "$RECENT_REBOOTS" -ge "$SYS_WATCHDOG_REBOOT_LIMIT" ]]; then error "Reboot loop detected — shutting down instead of rebooting" notify "Reboot loop on $(hostname) ($MY_ID) — shutting down after $RECENT_REBOOTS reboots — ${triggers[*]}" \ "System Watchdog" "warning" if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would shutdown now" return fi sync /sbin/poweroff return fi echo "" echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━" echo " $ICON_REBOOT_SMART STABILITY WATCHDOG — REBOOT TRIGGERED" echo " Tier: ${tier^^}" echo " Host: $MY_ID ($LOCAL_SERVER_NAME)" for t in "${triggers[@]}"; do echo " → $t" done echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━" notify "System watchdog ${tier^^} reboot on $(hostname) ($MY_ID) — ${triggers[*]}" \ "System Watchdog" "warning" if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — reboot sequence would begin now" return fi log_reboot # Graceful shutdown sequence if is_vm_manager_enabled && command -v virsh >/dev/null 2>&1; then warn "Shutting down VMs..." for VM in $(virsh list --name 2>/dev/null); do [[ -z "$VM" ]] && continue virsh shutdown "$VM" >/dev/null 2>&1 done sleep 30 else log "VM Manager not enabled — skipping VM shutdown" fi warn "Stopping Docker containers..." if is_docker_enabled && command -v docker >/dev/null 2>&1; then mapfile -t _SYS_REBOOT_STOPPED < <(docker ps --format '{{.Names}}' 2>/dev/null) trap _trap_sys_reboot_restart EXIT timeout 60 docker ps -q 2>/dev/null | xargs -r docker stop >/dev/null 2>&1 fi warn "Stopping User Scripts..." platform_stop_user_scripts warn "Syncing disks..." sync trap - EXIT # committed to reboot — containers should stay down sleep 5 /sbin/reboot } # ============================================================================================== # ━━━ Single-Pass Health Check ━━━ # ============================================================================================== echo "━━━ $ICON_REBOOT Stability Watchdog — $(date '+%Y-%m-%d %H:%M:%S') ━━━" TRIGGERS=() CRITICAL_TRIGGERS=() URGENT_OOM_CONFIRMED=false # ── OOM Delta — read every cycle for bypass decisions ───────────────────────────────────── OOM_DELTA=0 if [[ "$SYS_WATCHDOG_CHECK_OOM" == true ]]; then OOM_DELTA=$(get_oom_delta) [[ "$OOM_DELTA" -gt 0 ]] && \ log "OOM kills this cycle: $OOM_DELTA (limit: ${SYS_WATCHDOG_OOM_LIMIT})" fi # ========================================================================================== # ━━━ TIER 1 — CRITICAL CHECKS (bypass all strikes, reboot immediately) ━━━ # ========================================================================================== # ── Docker daemon — delegated to docker_watchdog ───────────────────────────────────────── # docker_watchdog owns daemon restart attempts (strike system + rc.docker restart). # When restart fails and daemon is confirmed down, it writes daemon_confirmed_down=true # to WATCHDOG_STATE_FILE. We read that flag and run through the standard strike system. if [[ "$SYS_WATCHDOG_CHECK_DOCKER_DAEMON" == true ]] && is_docker_enabled; then _daemon_down=$(grep -oP "(?<=^daemon_confirmed_down:)[^:]*" "$WATCHDOG_STATE_FILE" 2>/dev/null || echo "false") TRIGGERED=false [[ "$_daemon_down" == "true" ]] && TRIGGERED=true run_strike_check "docker_daemon" "$TRIGGERED" "Docker daemon confirmed down by docker_watchdog" && \ TRIGGERS+=("docker_daemon_unresponsive") [[ "$TRIGGERED" == true ]] && log "Docker daemon confirmed down — strike toward reboot" || \ log "Docker daemon flag clear ✅" elif [[ "$SYS_WATCHDOG_CHECK_DOCKER_DAEMON" == true ]]; then log "Docker not enabled — skipping daemon check" fi # ── rootfs critical — at 99%+ writes are failing ───────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_ROOTFS" == true ]]; then ROOTFS_USED=$(df / --output=pcent 2>/dev/null | tail -1 | tr -d ' %') if [[ "$ROOTFS_USED" -ge "${SYS_WATCHDOG_ROOTFS_CRITICAL_PCT:-99}" ]]; then error "rootfs ${ROOTFS_USED}% — CRITICAL (writes failing)" CRITICAL_TRIGGERS+=("rootfs_full=${ROOTFS_USED}%") fi fi # ── Kernel oops/BUG — kernel running with corrupted state ──────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_KERNEL_OOPS" == true ]]; then PREV_OOPS=$(get_state_val "kernel_oops_count") CURRENT_OOPS=$(dmesg 2>/dev/null | grep -cE "BUG:|kernel BUG|Oops:" || true) CURRENT_OOPS="${CURRENT_OOPS//[^0-9]/}"; CURRENT_OOPS="${CURRENT_OOPS:-0}" set_state_val "kernel_oops_count" "$CURRENT_OOPS" if [[ -n "$PREV_OOPS" && "$PREV_OOPS" =~ ^[0-9]+$ ]]; then OOPS_DELTA=$(( CURRENT_OOPS - PREV_OOPS )) if [[ "$OOPS_DELTA" -gt 0 ]]; then error "Kernel oops/BUG detected — $OOPS_DELTA new since last cycle — CRITICAL" CRITICAL_TRIGGERS+=("kernel_oops=${OOPS_DELTA}_new") fi fi fi # ── File descriptor exhaustion — new connections failing silently ───────────────────────── if [[ "$SYS_WATCHDOG_CHECK_FD" == true ]]; then FD_LINE=$(cat /proc/sys/fs/file-nr 2>/dev/null) FD_OPEN=$(echo "$FD_LINE" | awk '{print $1}') FD_MAX=$(echo "$FD_LINE" | awk '{print $3}') if [[ -n "$FD_OPEN" && -n "$FD_MAX" && "$FD_MAX" -gt 0 ]]; then FD_PCT=$(( FD_OPEN * 100 / FD_MAX )) if [[ "$FD_PCT" -ge "${SYS_WATCHDOG_FD_CRITICAL_PCT:-95}" ]]; then error "File descriptors ${FD_PCT}% exhausted (${FD_OPEN}/${FD_MAX}) — CRITICAL" CRITICAL_TRIGGERS+=("fd_exhaustion=${FD_PCT}%") else log "File descriptors: ${FD_PCT}% (${FD_OPEN}/${FD_MAX})" fi fi fi # ── /boot read-only — state and config writes failing silently ──────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_BOOT" == true ]]; then BOOT_TEST="/boot/.watchdog_write_test" if ! touch "$BOOT_TEST" 2>/dev/null; then error "/boot is read-only — config writes failing silently — CRITICAL" CRITICAL_TRIGGERS+=("boot_read_only") else rm -f "$BOOT_TEST" 2>/dev/null log "/boot is writable ✅" fi fi # ── Act on CRITICAL triggers immediately ───────────────────────────────────────────────── if [[ ${#CRITICAL_TRIGGERS[@]} -gt 0 ]]; then echo "" echo "━━━ $ICON_ERROR CRITICAL — IMMEDIATE REBOOT ━━━" for t in "${CRITICAL_TRIGGERS[@]}"; do error " CRITICAL: $t" done do_reboot "critical" "${CRITICAL_TRIGGERS[@]}" exit 0 fi # ========================================================================================== # ━━━ TIER 3 — STANDARD CHECKS (strike system) ━━━ # ========================================================================================== # ── rootfs standard ────────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_ROOTFS" == true ]]; then ROOTFS_USED=$(df / --output=pcent 2>/dev/null | tail -1 | tr -d ' %') TRIGGERED=false [[ "$ROOTFS_USED" -ge "$SYS_WATCHDOG_ROOTFS_PCT" ]] && TRIGGERED=true run_strike_check "rootfs" "$TRIGGERED" "rootfs ${ROOTFS_USED}%" && \ TRIGGERS+=("rootfs=${ROOTFS_USED}%") [[ "$TRIGGERED" == false ]] && log "rootfs ${ROOTFS_USED}% ✅ (warn at ${SYS_WATCHDOG_ROOTFS_PCT}%)" fi # ── /var/log ───────────────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_LOG" == true ]]; then LOG_USED=$(df -P /var/log 2>/dev/null | awk 'NR==2 {print $5}' | tr -d '%') TRIGGERED=false [[ "${LOG_USED:-0}" -ge "$SYS_WATCHDOG_LOG_PCT" ]] && TRIGGERED=true run_strike_check "log" "$TRIGGERED" "/var/log ${LOG_USED}%" && \ TRIGGERS+=("log=${LOG_USED}%") [[ "$TRIGGERED" == false ]] && log "/var/log ${LOG_USED:-?}% ✅ (warn at ${SYS_WATCHDOG_LOG_PCT}%)" fi # ── /tmp ───────────────────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_TMP" == true ]]; then TMP_USED=$(df -P /tmp 2>/dev/null | awk 'NR==2 {print $5}' | tr -d '%') if [[ "${TMP_USED:-0}" -ge "${SYS_WATCHDOG_TMP_CRITICAL_PCT:-98}" ]]; then # Try to clear before escalating warn "/tmp ${TMP_USED}% — attempting cleanup..." find /tmp -type f -mmin +60 -not -name "*.lock" -delete 2>/dev/null TMP_USED_AFTER=$(df -P /tmp 2>/dev/null | awk 'NR==2 {print $5}' | tr -d '%') if [[ "${TMP_USED_AFTER:-0}" -ge "${SYS_WATCHDOG_TMP_CRITICAL_PCT:-98}" ]]; then error "/tmp still ${TMP_USED_AFTER}% after cleanup — adding to triggers" TRIGGERED=true else warn "/tmp cleared to ${TMP_USED_AFTER}% ✅" TRIGGERED=false fi elif [[ "${TMP_USED:-0}" -ge "${SYS_WATCHDOG_TMP_PCT:-90}" ]]; then TRIGGERED=true else TRIGGERED=false fi run_strike_check "tmp" "$TRIGGERED" "/tmp ${TMP_USED}%" && \ TRIGGERS+=("tmp=${TMP_USED}%") fi # ── RAM — reboot tier only (warn/shutdown/recover handled by resource_watchdog.sh) ────────── if [[ "$SYS_WATCHDOG_CHECK_RAM" == true ]]; then MEM_KB=$(awk '/MemAvailable/ {print $2}' /proc/meminfo) MEM_GB=$(( MEM_KB / 1024 / 1024 )) if [[ "$MEM_GB" -lt "$SYS_WATCHDOG_MEM_GB" ]]; then # Tier 2 — bypass strike system if OOM confirms active crisis if [[ "$SYS_WATCHDOG_CHECK_OOM" == true ]] && \ [[ "$OOM_DELTA" -ge "$SYS_WATCHDOG_OOM_LIMIT" ]]; then error "RAM ${MEM_GB}GB + ${OOM_DELTA} OOM kills this run — URGENT bypass" OOM_VICTIMS=$(get_oom_victims) URGENT_TRIGGERS=("urgent_low_ram=${MEM_GB}GB" "oom_kills=${OOM_DELTA}") [[ -n "$OOM_VICTIMS" ]] && URGENT_TRIGGERS+=("oom_victims: $OOM_VICTIMS") do_reboot "urgent" "${URGENT_TRIGGERS[@]}" exit 0 fi # Standard strike path run_strike_check "ram" true "RAM ${MEM_GB}GB free" && \ TRIGGERS+=("low_ram=${MEM_GB}GB") else reset_strikes "ram" log "RAM ${MEM_GB}GB free ✅" fi fi # ── ZFS ARC ────────────────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_ARC" == true ]] && [[ -f /proc/spl/kstat/zfs/arcstats ]]; then ARC_SIZE=$(awk '/^size / {print $3}' /proc/spl/kstat/zfs/arcstats) ARC_MAX=$(awk '/^c_max / {print $3}' /proc/spl/kstat/zfs/arcstats) ARC_PCT=$(( ARC_SIZE * 100 / ARC_MAX )) TRIGGERED=false if [[ "$ARC_PCT" -ge "$SYS_WATCHDOG_ARC_PINNED_PCT" ]]; then sync; echo 3 > /proc/sys/vm/drop_caches; sleep 5 ARC_AFTER=$(awk '/^size / {print $3}' /proc/spl/kstat/zfs/arcstats) ARC_AFTER_PCT=$(( ARC_AFTER * 100 / ARC_MAX )) [[ "$ARC_AFTER_PCT" -ge "$SYS_WATCHDOG_ARC_RELEASE_PCT" ]] && TRIGGERED=true fi run_strike_check "arc" "$TRIGGERED" "ZFS ARC pinned ${ARC_PCT}%" && \ TRIGGERS+=("arc_pinned=${ARC_PCT}%") fi # ── CPU temperature ─────────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_CPU_TEMP" == true ]]; then CPU_TEMP="" if command -v sensors >/dev/null 2>&1; then CPU_TEMP=$(sensors 2>/dev/null | \ grep -i "Package id 0\|Tctl\|CPU Temp" | \ awk '{print $NF}' | tr -d '+°C' | head -1) fi if [[ -n "$CPU_TEMP" ]]; then CPU_TEMP_INT=$(printf "%.0f" "$CPU_TEMP") TRIGGERED=false [[ "$CPU_TEMP_INT" -ge "$SYS_WATCHDOG_CPU_TEMP_MAX" ]] && TRIGGERED=true run_strike_check "cpu_temp" "$TRIGGERED" "CPU temp ${CPU_TEMP_INT}°C" && \ TRIGGERS+=("cpu_temp=${CPU_TEMP_INT}C") fi fi # ── Load average ───────────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_LOAD" == true ]]; then LOAD=$(awk '{print $1}' /proc/loadavg) LOAD_INT=$(printf "%.0f" "$LOAD") LOAD_THRESHOLD=$(( TOTAL_CORES * SYS_WATCHDOG_LOAD_MULTIPLIER )) TRIGGERED=false [[ "$LOAD_INT" -ge "$LOAD_THRESHOLD" ]] && TRIGGERED=true run_strike_check "load" "$TRIGGERED" "load avg ${LOAD}" && \ TRIGGERS+=("load=${LOAD}") [[ "$TRIGGERED" == false ]] && log "load avg ${LOAD} ✅ (warn at ${LOAD_THRESHOLD} = ${SYS_WATCHDOG_LOAD_MULTIPLIER}×${TOTAL_CORES} cores)" fi # ── Zombie processes ───────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_ZOMBIES" == true ]]; then ZOMBIE_COUNT=$(ps aux 2>/dev/null | awk '{print $8}' | grep -c "^Z$" || true) ZOMBIE_COUNT="${ZOMBIE_COUNT//[^0-9]/}"; ZOMBIE_COUNT="${ZOMBIE_COUNT:-0}" TRIGGERED=false [[ "$ZOMBIE_COUNT" -ge "$SYS_WATCHDOG_ZOMBIE_LIMIT" ]] && TRIGGERED=true run_strike_check "zombies" "$TRIGGERED" "zombies ${ZOMBIE_COUNT}" && \ TRIGGERS+=("zombies=${ZOMBIE_COUNT}") [[ "$TRIGGERED" == false ]] && log "zombies ${ZOMBIE_COUNT} ✅ (warn at ${SYS_WATCHDOG_ZOMBIE_LIMIT})" fi # ── Array disk errors — accumulating mdstat errors ──────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_MDSTAT" == true ]]; then PREV_MD_ERRORS=$(get_state_val "mdstat_errors") CURRENT_MD_ERRORS=$(grep -oP "(?<=\[)[^\]]*[U_][^\]]*(?=\])" \ /proc/mdstat 2>/dev/null | grep -o "_" | wc -l || echo 0) CURRENT_MD_ERRORS="${CURRENT_MD_ERRORS//[^0-9]/}"; CURRENT_MD_ERRORS="${CURRENT_MD_ERRORS:-0}" set_state_val "mdstat_errors" "$CURRENT_MD_ERRORS" if [[ -n "$PREV_MD_ERRORS" && "$PREV_MD_ERRORS" =~ ^[0-9]+$ ]]; then MD_DELTA=$(( CURRENT_MD_ERRORS - PREV_MD_ERRORS )) if [[ "$MD_DELTA" -ge "${SYS_WATCHDOG_MDSTAT_ERROR_LIMIT:-5}" ]]; then TRIGGERED=true run_strike_check "mdstat" "$TRIGGERED" \ "mdstat errors +${MD_DELTA} (total: ${CURRENT_MD_ERRORS})" && \ TRIGGERS+=("mdstat_errors=+${MD_DELTA}") else run_strike_check "mdstat" false "mdstat" > /dev/null fi fi fi # ── Network interface state ─────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_NETWORK" == true ]]; then NIC="${SYS_WATCHDOG_NIC:-eth0}" NIC_STATE=$(cat "/sys/class/net/${NIC}/operstate" 2>/dev/null || echo "unknown") NIC_SPEED=$(cat "/sys/class/net/${NIC}/speed" 2>/dev/null || echo "?") TRIGGERED=false [[ "$NIC_STATE" != "up" ]] && TRIGGERED=true run_strike_check "network" "$TRIGGERED" "${NIC} state: ${NIC_STATE}" && \ TRIGGERS+=("nic_down=${NIC}") [[ "$TRIGGERED" == false ]] && log "$NIC ${NIC_STATE} @ ${NIC_SPEED}Mbps ✅" fi # ── sshd — try restart before escalating ───────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_SSHD" == true ]]; then if ! platform_is_service_running sshd; then warn "sshd not running — attempting restart..." if [[ "$DRY_RUN" == false ]]; then platform_restart_service sshd sleep 3 if platform_is_service_running sshd; then warn "sshd restarted successfully ✅" reset_strikes "sshd" notify "sshd was down on $(hostname) ($MY_ID) — restarted automatically" \ "System Watchdog" "warning" else error "sshd restart failed — remote access unavailable" run_strike_check "sshd" true "sshd not running" && \ TRIGGERS+=("sshd_down") fi else warn "DRY RUN — would restart sshd" fi else reset_strikes "sshd" fi fi # ── Runaway process ─────────────────────────────────────────────────────────────────────── if [[ "$SYS_WATCHDOG_CHECK_RUNAWAY" == true ]]; then RUNAWAY_PCT="${SYS_WATCHDOG_RUNAWAY_CPU_PCT:-90}" TOP_CPU_PCT=$(ps aux 2>/dev/null | awk 'NR>1{print $3}' | sort -rn | head -1) TOP_CPU_INT=$(printf "%.0f" "${TOP_CPU_PCT:-0}") TOP_CPU_NAME=$(ps aux 2>/dev/null | sort -k3 -rn | awk 'NR==2{print $11}') TRIGGERED=false [[ "$TOP_CPU_INT" -ge "$RUNAWAY_PCT" ]] && TRIGGERED=true # Runaway uses SYS_WATCHDOG_RUNAWAY_STRIKES not global strike limit if [[ "$TRIGGERED" == true ]]; then RAWAY_S=$(increment_strikes "runaway") RLIMIT="${SYS_WATCHDOG_RUNAWAY_STRIKES:-3}" warn "Runaway ${TOP_CPU_NAME} ${TOP_CPU_PCT}% CPU -- strike $RAWAY_S/$RLIMIT" if (( RAWAY_S >= RLIMIT )); then error "Runaway process ${TOP_CPU_NAME} -- strike limit hit" reset_strikes "runaway" TRIGGERS+=("runaway=${TOP_CPU_NAME}@${TOP_CPU_PCT}%") fi else RAWAY_CUR=$(get_strikes "runaway") [[ "${RAWAY_CUR:-0}" -gt 0 ]] && reset_strikes "runaway" fi fi # ── Required containers — removed from stability_watchdog ──────────────────────────────── # Container health is owned entirely by docker_watchdog: strike system, restart attempts, # skip-listing, and critical notifications. Rebooting here when docker_watchdog already # gave up creates a reboot loop — the container is still broken after reboot. # ========================================================================================== # ━━━ Evaluate Standard Triggers ━━━ # ========================================================================================== if [[ ${#TRIGGERS[@]} -gt 0 ]]; then echo "" echo "━━━ $ICON_REBOOT_SMART Stability Watchdog — $(date '+%Y-%m-%d %H:%M:%S') ━━━" for t in "${TRIGGERS[@]}"; do echo " $ICON_REBOOT_SMART $t" done [[ "$OOM_DELTA" -gt 0 ]] && echo " OOM kills this run: $OOM_DELTA" echo "" do_reboot "standard" "${TRIGGERS[@]}" exit 0 else echo "System healthy ✅ ($(date '+%H:%M:%S'))" fi # Keep state file mtime fresh — docker_watchdog stale guard checks this set_state_val "watchdog_cycle" "$(date +%s)"