From f826672b80c5440094387cadd87bf578be65605a Mon Sep 17 00:00:00 2001 From: FailedProxy Date: Fri, 24 Apr 2026 22:52:32 -0400 Subject: [PATCH] added more descriptions to master.conf --- Docker_Essentials/docker_watchdog.sh | 788 ++++++++++++++++----------- Master.conf | 478 ++++++++++++---- unRAID_Essentials/system_watchdog.sh | 9 + 3 files changed, 844 insertions(+), 431 deletions(-) diff --git a/Docker_Essentials/docker_watchdog.sh b/Docker_Essentials/docker_watchdog.sh index a423d33..99ec46d 100644 --- a/Docker_Essentials/docker_watchdog.sh +++ b/Docker_Essentials/docker_watchdog.sh @@ -1,40 +1,44 @@ #!/bin/bash # ----------------------------------------------------------------------------------------------- -# --------------------------------- System Watchdog -------------------------------------------- +# --------------------------------- Docker Watchdog -------------------------------------------- # ----------------------------------------------------------------------------------------------- -# Last line of defense — reboots the system cleanly if it is about to become unstable. -# Runs continuously as a background process — started by array_start.sh at array start. -# Works alongside docker_watchdog.sh which handles container-level healing first. +# Two-tier self-healing container monitoring system — runs continuously as a background process. +# Started by array_start.sh at array start — runs until array stops or SIGTERM received. # -# Checks (all toggleable in Master.conf): -# rootfs usage — high rootfs fills rapidly when array is down, crash imminent -# /var/log usage — log spam can fill rootfs, indicates something is broken -# free RAM — critically low RAM means OOM or swap imminent -# ZFS ARC pinned — ARC not releasing after reclaim means memory is stuck -# CPU temperature — sustained tjmax causes throttling or kernel panic -# load average — sustained high load means something is stuck or runaway -# zombie processes — large zombie count indicates serious process management failure -# Docker daemon — unresponsive daemon means containers cannot be managed -# Required containers — stopped containers that should be running (after watchdog skip list) +# Tier 1 — Strict monitoring (configured containers only) +# Memory hard limits — immediate restart if exceeded +# CPU thresholds — strike system, restart after CPU_FAIL_LIMIT strikes +# HTTP responsiveness — strike system, restart after RESP_FAIL_LIMIT strikes +# Required containers — must always be running, strike system with skip list # -# Abort conditions (toggleable): -# ZFS pool unhealthy — reboot with bad pool risks data loss -# Parity running — aborting parity is better than crashing mid-check -# Mover running — aborting move is better than crashing mid-move +# Tier 2 — Global health scan (all running containers) +# Unhealthy status — Docker HEALTHCHECK unhealthy → restart +# OOM killed — kernel killed container → restart + notify +# Crash loop detection — RestartCount climbing → notify, critical above limit +# Dead containers — remove and restart +# Unexpected exits — non-zero exit code → restart # -# Reboot loop protection: -# Tracks reboot timestamps in persistent log on /boot/ -# Rolling window — old entries purge automatically -# If reboot count hits limit in window → shutdown instead of reboot +# Cross-cutting intelligence: +# Startup grace period — skip restarts while system is still booting +# Dependency ordering — restart database before app +# Restart loop protect — stop restarting after X restarts in X hours → skip list +# Skip list auto-clear — clears when container recovers +# Notification batching — one clean summary per cycle, not one ping per event +# Quiet when healthy — only logs when something needs attention +# Parity awareness — skips restarts during parity check # # Continuous loop: -# Checks run every SYSTEM_WATCHDOG_INTERVAL seconds (default 300 = 5min) -# Master.conf re-sourced each cycle — config changes picked up without restart -# Silent when healthy — only verbose when trigger or reboot +# Checks run every DOCKER_WATCHDOG_INTERVAL seconds (default 900 = 15min) # Clean shutdown on SIGTERM/SIGINT — sent by array stop +# Variables scoped per-cycle — no state accumulation between cycles # -# All configuration in Master.conf under System Watchdog section. -# Supports --dry-run to show triggered conditions without rebooting. +# State files: +# WATCHDOG_STATE_FILE — strike counts (/tmp — resets on reboot) +# SYS_WATCHDOG_FAILED_FILE — persistent skip list (/boot — survives reboots) +# WATCHDOG_CONTAINER_RESTART_LOG — restart history for loop detection +# +# All configuration in Master.conf under Docker Watchdog section. +# Supports --dry-run and --status. # ----------------------------------------------------------------------------------------------- SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -59,23 +63,37 @@ success "Running as root" acquire_lock "continuous" -TOTAL_CORES=$(nproc) +# Select correct per-host watchdog lists — done once at startup +detect_hosts -# Ensure state and persistent files exist -touch "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null || { - error "Cannot create state file: $SYS_WATCHDOG_STATE_FILE" - exit 1 -} +if [[ "$LOCAL_SERVER_NAME" == "$HOST1" ]]; then + WATCHDOG_REQUIRED_CONTAINERS=("${HOST1_WATCHDOG_REQUIRED_CONTAINERS[@]}") + declare -A WATCHDOG_CONTAINER_URLS + for key in "${!HOST1_WATCHDOG_CONTAINER_URLS[@]}"; do + WATCHDOG_CONTAINER_URLS["$key"]="${HOST1_WATCHDOG_CONTAINER_URLS[$key]}" + done +else + WATCHDOG_REQUIRED_CONTAINERS=("${HOST2_WATCHDOG_REQUIRED_CONTAINERS[@]}") + declare -A WATCHDOG_CONTAINER_URLS + for key in "${!HOST2_WATCHDOG_CONTAINER_URLS[@]}"; do + WATCHDOG_CONTAINER_URLS["$key"]="${HOST2_WATCHDOG_CONTAINER_URLS[$key]}" + done +fi -touch "$SYS_WATCHDOG_REBOOT_LOG" 2>/dev/null || { - error "Cannot create reboot log: $SYS_WATCHDOG_REBOOT_LOG" - exit 1 -} +info "Watchdog running as: $LOCAL_SERVER_NAME" +info "Check interval: ${DOCKER_WATCHDOG_INTERVAL}s" -touch "$SYS_WATCHDOG_FAILED_FILE" 2>/dev/null || { - error "Cannot create failed container list: $SYS_WATCHDOG_FAILED_FILE" +if ! command -v docker >/dev/null 2>&1; then + error "Docker not found" exit 1 -} +fi + +success "Docker found" +[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no containers will be restarted" + +# Ensure state files exist +touch "$WATCHDOG_STATE_FILE" "$WATCHDOG_CONTAINER_RESTART_LOG" \ + "$SYS_WATCHDOG_FAILED_FILE" 2>/dev/null # ----------------------------------------------------------------------------------------------- # ━━━ $ICON_SUMMARY Status ━━━ @@ -83,198 +101,158 @@ touch "$SYS_WATCHDOG_FAILED_FILE" 2>/dev/null || { if [[ "$SHOW_STATUS" == true ]]; then echo "" echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━" - echo "$ICON_HEALTH rootfs threshold: ${SYS_WATCHDOG_ROOTFS_PCT}% toggle: $SYS_WATCHDOG_CHECK_ROOTFS" - echo "$ICON_HEALTH log threshold: ${SYS_WATCHDOG_LOG_PCT}% toggle: $SYS_WATCHDOG_CHECK_LOG" - echo "$ICON_MEM RAM threshold: ${SYS_WATCHDOG_MEM_GB}GB free toggle: $SYS_WATCHDOG_CHECK_RAM" - echo "$ICON_ZFS ARC pinned: ${SYS_WATCHDOG_ARC_PINNED_PCT}% toggle: $SYS_WATCHDOG_CHECK_ARC" - echo "$ICON_ZFS ARC release: ${SYS_WATCHDOG_ARC_RELEASE_PCT}%" - echo "$ICON_GEAR CPU temp max: ${SYS_WATCHDOG_CPU_TEMP_MAX}°C toggle: $SYS_WATCHDOG_CHECK_CPU_TEMP" - echo "$ICON_GEAR Load multiplier: ${SYS_WATCHDOG_LOAD_MULTIPLIER}x cores toggle: $SYS_WATCHDOG_CHECK_LOAD" - echo "$ICON_GEAR Zombie limit: ${SYS_WATCHDOG_ZOMBIE_LIMIT} toggle: $SYS_WATCHDOG_CHECK_ZOMBIES" - echo "$ICON_CONTAINERS Docker daemon: toggle: $SYS_WATCHDOG_CHECK_DOCKER_DAEMON" - echo "$ICON_CONTAINERS Containers: toggle: $SYS_WATCHDOG_CHECK_CONTAINERS" - echo "$ICON_SHIELD Strike limit: $SYS_WATCHDOG_STRIKE_LIMIT" - echo "$ICON_TIME Interval: ${SYSTEM_WATCHDOG_INTERVAL}s" - echo "$ICON_REBOOT_SMART Reboot limit: $SYS_WATCHDOG_REBOOT_LIMIT in ${SYS_WATCHDOG_REBOOT_WINDOW_HRS}hrs" - echo "$ICON_ZFS Abort ZFS unhealthy: $SYS_WATCHDOG_ABORT_ON_ZFS_UNHEALTHY" - echo "$ICON_GEAR Abort on parity: $SYS_WATCHDOG_ABORT_ON_PARITY" - echo "$ICON_MOVER Abort on mover: $SYS_WATCHDOG_ABORT_ON_MOVER" - echo "$ICON_GEAR Dry Run: $DRY_RUN" + echo "$ICON_CONTAINERS Watched: ${!WATCHDOG_CONTAINERS[@]}" + echo "$ICON_CONTAINERS Required: ${WATCHDOG_REQUIRED_CONTAINERS[*]}" + echo "$ICON_WATCHDOG Scan all: $WATCHDOG_SCAN_ALL" + echo "$ICON_WATCHDOG Interval: ${DOCKER_WATCHDOG_INTERVAL}s" + echo "$ICON_WATCHDOG Startup grace: ${WATCHDOG_STARTUP_GRACE}s" + echo "$ICON_WATCHDOG Restart limit: $WATCHDOG_CONTAINER_RESTART_LIMIT in ${WATCHDOG_CONTAINER_RESTART_WINDOW}h" + echo "$ICON_WATCHDOG Batch notify: $WATCHDOG_BATCH_NOTIFY" + echo "$ICON_WATCHDOG Ignore list: ${WATCHDOG_SCAN_IGNORE[*]:-none}" + echo "$ICON_GEAR Dry Run: $DRY_RUN" + UPTIME_S=$(awk '{print int($1)}' /proc/uptime) + echo "$ICON_TIME System uptime: $(format_duration $UPTIME_S)" + [[ "$UPTIME_S" -lt "$WATCHDOG_STARTUP_GRACE" ]] && \ + warn "Within startup grace period — restarts suppressed" echo "━━━━━━━━━━━━━━━━━━━━━━━" exit 0 fi -[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no reboot will be executed" - # ----------------------------------------------------------------------------------------------- -# STATE HELPERS — defined once, used every cycle +# HELPERS — defined once, used every cycle # ----------------------------------------------------------------------------------------------- get_strikes() { - local key="$1" - grep -E "^${key}:" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null | cut -d':' -f2 + grep "^${1}:" "${2}" 2>/dev/null | cut -d: -f2 || echo "0" } set_strikes() { - local key="$1" count="$2" - grep -vE "^${key}:" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null > "${SYS_WATCHDOG_STATE_FILE}.tmp" - echo "${key}:${count}" >> "${SYS_WATCHDOG_STATE_FILE}.tmp" - mv "${SYS_WATCHDOG_STATE_FILE}.tmp" "$SYS_WATCHDOG_STATE_FILE" + local container="$1" count="$2" file="$3" + if grep -q "^${container}:" "$file" 2>/dev/null; then + sed -i "s/^${container}:.*/${container}:${count}/" "$file" + else + echo "${container}:${count}" >> "$file" + fi } -increment_strikes() { - local key="$1" - local current - current=$(get_strikes "$key") - [[ -z "$current" ]] && current=0 - ((current++)) - set_strikes "$key" "$current" - echo "$current" +is_skipped() { + grep -q "^${1}$" "$SYS_WATCHDOG_FAILED_FILE" 2>/dev/null } -reset_strikes() { - local key="$1" - set_strikes "$key" 0 +add_to_skip_list() { + local container="$1" reason="$2" + if ! is_skipped "$container"; then + echo "$container" >> "$SYS_WATCHDOG_FAILED_FILE" + error "$container added to skip list — $reason" + queue_notify "$container added to skip list on $(hostname) — $reason — manual intervention needed" "critical" + fi } -purge_old_reboots() { +remove_from_skip_list() { + sed -i "/^${1}$/d" "$SYS_WATCHDOG_FAILED_FILE" 2>/dev/null + success "$1 removed from skip list — recovered" +} + +log_restart() { + local container="$1" local now - now=$(date +%s) - local cutoff=$(( now - SYS_WATCHDOG_REBOOT_WINDOW )) - grep -v "^$" "$SYS_WATCHDOG_REBOOT_LOG" 2>/dev/null | while IFS= read -r ts; do - [[ "$ts" -gt "$cutoff" ]] && echo "$ts" - done > "${SYS_WATCHDOG_REBOOT_LOG}.tmp" - mv "${SYS_WATCHDOG_REBOOT_LOG}.tmp" "$SYS_WATCHDOG_REBOOT_LOG" + now=$(date '+%Y-%m-%d %H:%M:%S') + local cutoff + cutoff=$(date -d "${WATCHDOG_CONTAINER_RESTART_WINDOW} hours ago" '+%Y-%m-%d %H:%M:%S') + echo "${container}|${now}" >> "$WATCHDOG_CONTAINER_RESTART_LOG" + local tmp="${WATCHDOG_CONTAINER_RESTART_LOG}.tmp" + awk -F'|' -v cutoff="$cutoff" '$2 >= cutoff' \ + "$WATCHDOG_CONTAINER_RESTART_LOG" > "$tmp" && \ + mv "$tmp" "$WATCHDOG_CONTAINER_RESTART_LOG" } -count_recent_reboots() { - purge_old_reboots - grep -c "." "$SYS_WATCHDOG_REBOOT_LOG" 2>/dev/null || echo 0 +get_restart_count() { + local container="$1" + local cutoff + cutoff=$(date -d "${WATCHDOG_CONTAINER_RESTART_WINDOW} hours ago" '+%Y-%m-%d %H:%M:%S') + awk -F'|' -v c="$container" -v cutoff="$cutoff" \ + '$1==c && $2>=cutoff' "$WATCHDOG_CONTAINER_RESTART_LOG" 2>/dev/null | wc -l } -log_reboot() { - date +%s >> "$SYS_WATCHDOG_REBOOT_LOG" -} - -check_abort_conditions() { - local should_abort=false - - if command -v zpool >/dev/null 2>&1; then - local unhealthy - unhealthy=$(zpool list -H -o health 2>/dev/null | grep -v ONLINE || true) - if [[ -n "$unhealthy" ]]; then - if [[ "$SYS_WATCHDOG_ABORT_ON_ZFS_UNHEALTHY" == true ]]; then - error "$ICON_ZFS ZFS pool unhealthy — aborting reboot" - notify "System watchdog aborted reboot on $(hostname) — ZFS pool unhealthy" "System Watchdog" "warning" - should_abort=true - else - warn "$ICON_ZFS ZFS pool unhealthy — continuing reboot" - fi +dependencies_satisfied() { + local container="$1" + local deps="${WATCHDOG_DEPENDENCIES[$container]:-}" + [[ -z "$deps" ]] && return 0 + for dep in $deps; do + local status + status=$(docker inspect -f '{{.State.Running}}' "$dep" 2>/dev/null) + if [[ "$status" != "true" ]]; then + warn "$container — dependency $dep is not running — skipping restart this cycle" + return 1 fi - fi - - if [[ -f /var/local/emhttp/parity-date.txt ]]; then - if grep -q "progress" /var/local/emhttp/parity-date.txt 2>/dev/null; then - if [[ "$SYS_WATCHDOG_ABORT_ON_PARITY" == true ]]; then - error "$ICON_DISK Parity check running — aborting reboot" - notify "System watchdog aborted reboot on $(hostname) — parity running" "System Watchdog" "warning" - should_abort=true - else - warn "$ICON_DISK Parity check running — continuing reboot" - fi - fi - fi - - if pgrep -f "mover" >/dev/null 2>&1; then - if [[ "$SYS_WATCHDOG_ABORT_ON_MOVER" == true ]]; then - error "$ICON_MOVER Mover running — aborting reboot" - notify "System watchdog aborted reboot on $(hostname) — mover running" "System Watchdog" "warning" - should_abort=true - else - warn "$ICON_MOVER Mover running — continuing reboot" - fi - fi - - [[ "$should_abort" == true ]] && return 1 + done return 0 } -run_strike_check() { - local key="$1" triggered="$2" description="$3" - if [[ "$triggered" == true ]]; then - local strikes - strikes=$(increment_strikes "$key") - warn "$description — strike $strikes/$SYS_WATCHDOG_STRIKE_LIMIT" - if (( strikes >= SYS_WATCHDOG_STRIKE_LIMIT )); then - error "$description hit strike limit — reboot triggered" - reset_strikes "$key" - return 0 - fi +safe_restart() { + local container="$1" reason="$2" + local restart_count + restart_count=$(get_restart_count "$container") + if [[ "$restart_count" -ge "$WATCHDOG_CONTAINER_RESTART_LIMIT" ]]; then + add_to_skip_list "$container" \ + "restarted $restart_count times in ${WATCHDOG_CONTAINER_RESTART_WINDOW}h" + return 2 + fi + dependencies_satisfied "$container" || return 1 + local uptime_s + uptime_s=$(awk '{print int($1)}' /proc/uptime) + if [[ "$uptime_s" -lt "$WATCHDOG_STARTUP_GRACE" ]]; then + warn "$container — within startup grace period — skipping restart" + return 1 + fi + if [[ "$DRY_RUN" == true ]]; then + warn "DRY RUN — would restart $container ($reason)" + return 0 + fi + info "Restarting $container ($reason) [restart $((restart_count + 1))/$WATCHDOG_CONTAINER_RESTART_LIMIT in window]..." + if docker restart "$container" >/dev/null 2>&1; then + success "$ICON_STARTED $container restarted" + log_restart "$container" + return 0 else - local current - current=$(get_strikes "$key") - if [[ -n "$current" && "$current" -gt 0 ]]; then - reset_strikes "$key" - fi + error "Failed to restart $container" + return 1 fi - return 1 } -do_reboot() { - local triggers=("$@") +queue_notify() { + local message="$1" severity="${2:-warning}" + NOTIFY_EVENTS+=("${severity}|${message}") + log "Queued: $message" +} - if ! check_abort_conditions; then - return - fi - - RECENT_REBOOTS=$(count_recent_reboots) - info "Recent reboots in ${SYS_WATCHDOG_REBOOT_WINDOW_HRS}hr window: $RECENT_REBOOTS / $SYS_WATCHDOG_REBOOT_LIMIT" - - if [[ "$RECENT_REBOOTS" -ge "$SYS_WATCHDOG_REBOOT_LIMIT" ]]; then - error "Reboot limit hit — shutting down instead" - notify "Reboot loop detected on $(hostname) — shutting down after $RECENT_REBOOTS reboots — conditions: ${triggers[*]}" "System Watchdog" "warning" - if [[ "$DRY_RUN" == true ]]; then - warn "DRY RUN — would shutdown now" - return - fi - sync - /sbin/poweroff - return - fi - - notify "System watchdog reboot triggered on $(hostname) — conditions: ${triggers[*]}" "System Watchdog" "warning" - - if [[ "$DRY_RUN" == true ]]; then - warn "DRY RUN — reboot sequence would begin now" - return - fi - - log_reboot - - info "Shutting down VMs..." - if command -v virsh >/dev/null 2>&1; then - for VM in $(virsh list --name 2>/dev/null); do - [[ -z "$VM" ]] && continue - virsh shutdown "$VM" >/dev/null 2>&1 +flush_notify() { + [[ ${#NOTIFY_EVENTS[@]} -eq 0 ]] && return + if [[ "$WATCHDOG_BATCH_NOTIFY" == "true" ]]; then + local highest_severity="normal" + local messages=() + for event in "${NOTIFY_EVENTS[@]}"; do + local sev="${event%%|*}" msg="${event#*|}" + messages+=("$msg") + [[ "$sev" == "critical" ]] && highest_severity="warning" + [[ "$sev" == "warning" && "$highest_severity" == "normal" ]] && highest_severity="warning" + done + local summary + summary=$(printf '%s. ' "${messages[@]}") + notify "Docker Watchdog on $(hostname) — ${#NOTIFY_EVENTS[@]} event(s): $summary" \ + "Docker Watchdog" "$highest_severity" + else + for event in "${NOTIFY_EVENTS[@]}"; do + local sev="${event%%|*}" msg="${event#*|}" + [[ "$sev" == "critical" ]] && sev="warning" + notify "$msg" "Docker Watchdog" "$sev" done - sleep 30 fi + NOTIFY_EVENTS=() +} - info "Stopping Docker containers..." - if command -v docker >/dev/null 2>&1; then - docker ps -q | xargs -r docker stop >/dev/null 2>&1 - fi - - info "Stopping User Scripts..." - pkill -f "/tmp/user.scripts" 2>/dev/null || true - - info "Syncing disks..." - sync - - echo "" - echo "$ICON_REBOOT_SMART Rebooting system NOW..." - sleep 5 - /sbin/reboot +is_parity_running() { + grep -q "progress" /var/local/emhttp/parity-date.txt 2>/dev/null } # ----------------------------------------------------------------------------------------------- @@ -284,7 +262,7 @@ WATCHDOG_RUNNING=true cleanup() { echo "" - info "System watchdog received shutdown signal — stopping cleanly" + info "Docker watchdog received shutdown signal — stopping cleanly" WATCHDOG_RUNNING=false exit 0 } @@ -294,140 +272,312 @@ trap cleanup SIGTERM SIGINT # ----------------------------------------------------------------------------------------------- # ━━━ CONTINUOUS MONITORING LOOP ━━━ # ----------------------------------------------------------------------------------------------- -info "System watchdog started — checking every ${SYSTEM_WATCHDOG_INTERVAL}s" +info "Docker watchdog started — checking every ${DOCKER_WATCHDOG_INTERVAL}s" echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━" CYCLE=0 while [[ "$WATCHDOG_RUNNING" == true ]]; do ((CYCLE++)) + CYCLE_START=$(date +%s) - # Re-source Master.conf each cycle — picks up config changes without restart + # Re-source Master.conf each cycle — picks up any config changes without restart source "$SCRIPT_DIR/../Master.conf" - SYS_WATCHDOG_REBOOT_WINDOW=$(( SYS_WATCHDOG_REBOOT_WINDOW_HRS * 3600 )) - TOTAL_CORES=$(nproc) - # Per-cycle triggers — cleared each iteration - TRIGGERS=() - - # ── rootfs usage ───────────────────────────────────────────────────────────────────────── - if [[ "$SYS_WATCHDOG_CHECK_ROOTFS" == true ]]; then - ROOTFS_USED=$(df / --output=pcent | tail -1 | tr -d ' %') - TRIGGERED=false - [[ "$ROOTFS_USED" -ge "$SYS_WATCHDOG_ROOTFS_PCT" ]] && TRIGGERED=true - run_strike_check "rootfs" "$TRIGGERED" "rootfs ${ROOTFS_USED}%" && \ - TRIGGERS+=("rootfs=${ROOTFS_USED}%") - fi - - # ── /var/log usage ──────────────────────────────────────────────────────────────────────── - if [[ "$SYS_WATCHDOG_CHECK_LOG" == true ]]; then - LOG_USED=$(df -P /var/log | awk 'NR==2 {print $5}' | tr -d '%') - TRIGGERED=false - [[ "$LOG_USED" -ge "$SYS_WATCHDOG_LOG_PCT" ]] && TRIGGERED=true - run_strike_check "log" "$TRIGGERED" "/var/log ${LOG_USED}%" && \ - TRIGGERS+=("log=${LOG_USED}%") - fi - - # ── Free RAM ───────────────────────────────────────────────────────────────────────────── - if [[ "$SYS_WATCHDOG_CHECK_RAM" == true ]]; then - MEM_KB=$(awk '/MemAvailable/ {print $2}' /proc/meminfo) - MEM_GB=$((MEM_KB / 1024 / 1024)) - TRIGGERED=false - [[ "$MEM_GB" -lt "$SYS_WATCHDOG_MEM_GB" ]] && TRIGGERED=true - run_strike_check "ram" "$TRIGGERED" "RAM ${MEM_GB}GB free" && \ - TRIGGERS+=("low_ram=${MEM_GB}GB") - fi - - # ── ZFS ARC ────────────────────────────────────────────────────────────────────────────── - if [[ "$SYS_WATCHDOG_CHECK_ARC" == true ]] && [[ -f /proc/spl/kstat/zfs/arcstats ]]; then - ARC_SIZE=$(awk '/^size / {print $3}' /proc/spl/kstat/zfs/arcstats) - ARC_MAX=$(awk '/^c_max / {print $3}' /proc/spl/kstat/zfs/arcstats) - ARC_PCT=$(( ARC_SIZE * 100 / ARC_MAX )) - TRIGGERED=false - if [[ "$ARC_PCT" -ge "$SYS_WATCHDOG_ARC_PINNED_PCT" ]]; then - sync; echo 3 > /proc/sys/vm/drop_caches; sleep 5 - ARC_AFTER=$(awk '/^size / {print $3}' /proc/spl/kstat/zfs/arcstats) - ARC_AFTER_PCT=$(( ARC_AFTER * 100 / ARC_MAX )) - [[ "$ARC_AFTER_PCT" -ge "$SYS_WATCHDOG_ARC_RELEASE_PCT" ]] && TRIGGERED=true - fi - run_strike_check "arc" "$TRIGGERED" "ZFS ARC pinned" && \ - TRIGGERS+=("arc_pinned=${ARC_PCT}%") - fi - - # ── CPU temperature ────────────────────────────────────────────────────────────────────── - if [[ "$SYS_WATCHDOG_CHECK_CPU_TEMP" == true ]]; then - CPU_TEMP="" - if command -v sensors >/dev/null 2>&1; then - CPU_TEMP=$(sensors 2>/dev/null | grep -i "Package id 0\|Tctl\|CPU Temp" | \ - awk '{print $NF}' | tr -d '+°C' | head -1) - fi - if [[ -n "$CPU_TEMP" ]]; then - CPU_TEMP_INT=$(printf "%.0f" "$CPU_TEMP") - TRIGGERED=false - [[ "$CPU_TEMP_INT" -ge "$SYS_WATCHDOG_CPU_TEMP_MAX" ]] && TRIGGERED=true - run_strike_check "cpu_temp" "$TRIGGERED" "CPU temp ${CPU_TEMP_INT}°C" && \ - TRIGGERS+=("cpu_temp=${CPU_TEMP_INT}C") - fi - fi - - # ── Load average ───────────────────────────────────────────────────────────────────────── - if [[ "$SYS_WATCHDOG_CHECK_LOAD" == true ]]; then - LOAD=$(awk '{print $1}' /proc/loadavg) - LOAD_INT=$(printf "%.0f" "$LOAD") - LOAD_THRESHOLD=$(( TOTAL_CORES * SYS_WATCHDOG_LOAD_MULTIPLIER )) - TRIGGERED=false - [[ "$LOAD_INT" -ge "$LOAD_THRESHOLD" ]] && TRIGGERED=true - run_strike_check "load" "$TRIGGERED" "load avg ${LOAD}" && \ - TRIGGERS+=("load=${LOAD}") - fi - - # ── Zombie processes ───────────────────────────────────────────────────────────────────── - if [[ "$SYS_WATCHDOG_CHECK_ZOMBIES" == true ]]; then - ZOMBIE_COUNT=$(ps aux | awk '{print $8}' | grep -c "^Z$" || echo 0) - TRIGGERED=false - [[ "$ZOMBIE_COUNT" -ge "$SYS_WATCHDOG_ZOMBIE_LIMIT" ]] && TRIGGERED=true - run_strike_check "zombies" "$TRIGGERED" "zombies ${ZOMBIE_COUNT}" && \ - TRIGGERS+=("zombies=${ZOMBIE_COUNT}") - fi - - # ── Docker daemon ──────────────────────────────────────────────────────────────────────── - if [[ "$SYS_WATCHDOG_CHECK_DOCKER_DAEMON" == true ]]; then - TRIGGERED=false - ! timeout 10 docker ps >/dev/null 2>&1 && TRIGGERED=true - run_strike_check "docker_daemon" "$TRIGGERED" "Docker daemon unresponsive" && \ - TRIGGERS+=("docker_daemon") - fi - - # ── Required containers from skip list ─────────────────────────────────────────────────── - if [[ "$SYS_WATCHDOG_CHECK_CONTAINERS" == true ]] && [[ -s "$SYS_WATCHDOG_FAILED_FILE" ]]; then - FAILED_CONTAINERS=() - while IFS= read -r container; do - [[ -z "$container" ]] && continue - STATUS=$(docker inspect -f '{{.State.Running}}' "$container" 2>/dev/null || echo "unknown") - [[ "$STATUS" != "true" ]] && FAILED_CONTAINERS+=("$container") - done < "$SYS_WATCHDOG_FAILED_FILE" - - if [[ ${#FAILED_CONTAINERS[@]} -gt 0 ]]; then - run_strike_check "failed_containers" "true" "required containers stopped" && \ - TRIGGERS+=("containers=${FAILED_CONTAINERS[*]}") - fi - fi - - # ── Evaluate triggers ──────────────────────────────────────────────────────────────────── - if [[ ${#TRIGGERS[@]} -gt 0 ]]; then - echo "" - echo "━━━ $ICON_REBOOT_SMART System Watchdog — Cycle $CYCLE — $(date '+%Y-%m-%d %H:%M:%S') ━━━" - for t in "${TRIGGERS[@]}"; do - echo " $ICON_REBOOT_SMART $t" + # Rebuild per-host lists after re-source in case they changed + if [[ "$LOCAL_SERVER_NAME" == "$HOST1" ]]; then + WATCHDOG_REQUIRED_CONTAINERS=("${HOST1_WATCHDOG_REQUIRED_CONTAINERS[@]}") + for key in "${!HOST1_WATCHDOG_CONTAINER_URLS[@]}"; do + WATCHDOG_CONTAINER_URLS["$key"]="${HOST1_WATCHDOG_CONTAINER_URLS[$key]}" done - echo "" - do_reboot "${TRIGGERS[@]}" else - log "Cycle $CYCLE — system healthy ($(date '+%H:%M:%S'))" + WATCHDOG_REQUIRED_CONTAINERS=("${HOST2_WATCHDOG_REQUIRED_CONTAINERS[@]}") + for key in "${!HOST2_WATCHDOG_CONTAINER_URLS[@]}"; do + WATCHDOG_CONTAINER_URLS["$key"]="${HOST2_WATCHDOG_CONTAINER_URLS[$key]}" + done + fi + + # Per-cycle variables — cleared each iteration, no accumulation + NOTIFY_EVENTS=() + T1_RESTARTS=0 + T1_WARNINGS=0 + T2_RESTARTS=0 + T2_WARNINGS=0 + + # Rebuild ignore map each cycle in case config was updated + declare -A IGNORE_MAP + for c in "${WATCHDOG_SCAN_IGNORE[@]}"; do + [[ -n "$c" ]] && IGNORE_MAP["$c"]=1 + done + + # Startup grace check + UPTIME_SECONDS=$(awk '{print int($1)}' /proc/uptime) + IN_GRACE_PERIOD=false + if [[ "$UPTIME_SECONDS" -lt "$WATCHDOG_STARTUP_GRACE" ]]; then + IN_GRACE_PERIOD=true + fi + + # Parity check — skip restarts during parity to avoid I/O interference + if is_parity_running; then + log "Parity check in progress — skipping restart actions this cycle" + sleep "$DOCKER_WATCHDOG_INTERVAL" + continue + fi + + # ── TIER 1 — Strict Monitoring ────────────────────────────────────────────────────────── + + # Required containers + if [[ ${#WATCHDOG_REQUIRED_CONTAINERS[@]} -gt 0 ]]; then + for container in "${WATCHDOG_REQUIRED_CONTAINERS[@]}"; do + [[ -z "$container" ]] && continue + STATUS=$(docker inspect -f '{{.State.Running}}' "$container" 2>/dev/null) + + if is_skipped "$container"; then + if [[ "$STATUS" == "true" ]]; then + remove_from_skip_list "$container" + set_strikes "$container" 0 "$WATCHDOG_STATE_FILE" + sed -i "/^${container}|/d" "$WATCHDOG_CONTAINER_RESTART_LOG" 2>/dev/null + queue_notify "$container recovered on $(hostname) — removed from skip list" "normal" + else + warn "$container — on skip list, manual intervention needed" + fi + continue + fi + + if [[ "$STATUS" == "true" ]]; then + set_strikes "$container" 0 "$WATCHDOG_STATE_FILE" + else + STRIKES=$(get_strikes "$container" "$WATCHDOG_STATE_FILE") + STRIKES=$(( STRIKES + 1 )) + set_strikes "$container" "$STRIKES" "$WATCHDOG_STATE_FILE" + warn "$container — not running (strike $STRIKES/$SYS_WATCHDOG_STRIKE_LIMIT)" + ((T1_WARNINGS++)) + + if [[ "$STRIKES" -ge "$SYS_WATCHDOG_STRIKE_LIMIT" ]]; then + result=0 + safe_restart "$container" "required container down" || result=$? + case $result in + 0) set_strikes "$container" 0 "$WATCHDOG_STATE_FILE" + ((T1_RESTARTS++)) + queue_notify "$container was down and restarted on $(hostname)" "warning" ;; + 2) : ;; + *) queue_notify "$container failed to restart on $(hostname)" "warning" ;; + esac + fi + fi + done + fi + + # Memory and CPU monitoring + if [[ ${#WATCHDOG_CONTAINERS[@]} -gt 0 ]]; then + STATS=$(docker stats --no-stream \ + --format "{{.Name}}|{{.MemUsage}}|{{.CPUPerc}}" 2>/dev/null) + TOTAL_CORES=$(nproc 2>/dev/null || echo 1) + + for container in "${!WATCHDOG_CONTAINERS[@]}"; do + MEM_LIMIT_MB="${WATCHDOG_CONTAINERS[$container]}" + CONTAINER_STATS=$(echo "$STATS" | grep "^${container}|" | head -1) + [[ -z "$CONTAINER_STATS" ]] && continue + + MEM_USAGE=$(echo "$CONTAINER_STATS" | cut -d'|' -f2 | awk '{print $1}') + MEM_UNIT=$(echo "$MEM_USAGE" | grep -oE '[A-Za-z]+') + MEM_VALUE=$(echo "$MEM_USAGE" | grep -oE '[0-9.]+') + + case "$MEM_UNIT" in + GiB|GB) MEM_MB=$(awk "BEGIN {printf \"%.0f\", $MEM_VALUE * 1024}") ;; + MiB|MB) MEM_MB=$(awk "BEGIN {printf \"%.0f\", $MEM_VALUE}") ;; + KiB|KB) MEM_MB=$(awk "BEGIN {printf \"%.0f\", $MEM_VALUE / 1024}") ;; + *) MEM_MB=0 ;; + esac + + CPU_RAW=$(echo "$CONTAINER_STATS" | cut -d'|' -f3 | tr -d '%') + CPU_NORM=$(awk "BEGIN {printf \"%.1f\", $CPU_RAW / $TOTAL_CORES}") + CPU_INT=$(printf "%.0f" "$CPU_NORM") + + if [[ "$MEM_MB" -ge "$MEM_LIMIT_MB" ]]; then + error "$container — memory exceeded hard limit ${MEM_LIMIT_MB}MB" + safe_restart "$container" "memory hard limit exceeded" + ((T1_RESTARTS++)) + queue_notify "$container exceeded memory limit on $(hostname) — restarted" "warning" + fi + + if [[ "$CPU_INT" -ge "$HARD_CPU_THRESHOLD" ]]; then + CPU_STRIKES=$(get_strikes "${container}_cpu" "$WATCHDOG_STATE_FILE") + CPU_STRIKES=$(( CPU_STRIKES + 1 )) + set_strikes "${container}_cpu" "$CPU_STRIKES" "$WATCHDOG_STATE_FILE" + if [[ "$CPU_STRIKES" -ge "$CPU_FAIL_LIMIT" ]]; then + safe_restart "$container" "CPU threshold exceeded" + set_strikes "${container}_cpu" 0 "$WATCHDOG_STATE_FILE" + ((T1_RESTARTS++)) + queue_notify "$container CPU ${CPU_NORM}% on $(hostname) — restarted" "warning" + fi + else + set_strikes "${container}_cpu" 0 "$WATCHDOG_STATE_FILE" + fi + done + fi + + # HTTP responsiveness + if [[ ${#WATCHDOG_CONTAINER_URLS[@]} -gt 0 ]]; then + for container in "${!WATCHDOG_CONTAINER_URLS[@]}"; do + URL="${WATCHDOG_CONTAINER_URLS[$container]}" + if curl -sf --max-time "$CURL_TIMEOUT" "$URL" >/dev/null 2>&1; then + set_strikes "${container}_http" 0 "$WATCHDOG_STATE_FILE" + else + HTTP_STRIKES=$(get_strikes "${container}_http" "$WATCHDOG_STATE_FILE") + HTTP_STRIKES=$(( HTTP_STRIKES + 1 )) + set_strikes "${container}_http" "$HTTP_STRIKES" "$WATCHDOG_STATE_FILE" + warn "$container — not responding at $URL (strike $HTTP_STRIKES/$RESP_FAIL_LIMIT)" + ((T1_WARNINGS++)) + if [[ "$HTTP_STRIKES" -ge "$RESP_FAIL_LIMIT" ]]; then + safe_restart "$container" "HTTP unresponsive" + set_strikes "${container}_http" 0 "$WATCHDOG_STATE_FILE" + ((T1_RESTARTS++)) + queue_notify "$container unresponsive at $URL on $(hostname) — restarted" "warning" + fi + fi + done + fi + + # ── TIER 2 — Global Health Scan ───────────────────────────────────────────────────────── + if [[ "$WATCHDOG_SCAN_ALL" == "true" ]]; then + ALL_CONTAINERS=$(docker ps --format "{{.Names}}" 2>/dev/null) + + # Unhealthy containers + if [[ "$WATCHDOG_RESTART_UNHEALTHY" == "true" ]]; then + UNHEALTHY=$(docker ps --filter health=unhealthy --format "{{.Names}}" 2>/dev/null) + while IFS= read -r container; do + [[ -z "$container" ]] && continue + [[ -n "${IGNORE_MAP[$container]:-}" ]] && continue + is_skipped "$container" && continue + error "$container — unhealthy" + ((T2_WARNINGS++)) + result=0 + safe_restart "$container" "unhealthy health status" || result=$? + [[ $result -eq 0 ]] && ((T2_RESTARTS++)) && \ + queue_notify "$container unhealthy on $(hostname) — restarted" "warning" + done <<< "$UNHEALTHY" + fi + + # OOM killed + if [[ "$WATCHDOG_NOTIFY_OOM" == "true" ]]; then + while IFS= read -r container; do + [[ -z "$container" ]] && continue + [[ -n "${IGNORE_MAP[$container]:-}" ]] && continue + is_skipped "$container" && continue + OOM=$(docker inspect -f '{{.State.OOMKilled}}' "$container" 2>/dev/null) + if [[ "$OOM" == "true" ]]; then + error "$container — OOM killed" + ((T2_WARNINGS++)) + result=0 + safe_restart "$container" "OOM killed" || result=$? + [[ $result -eq 0 ]] && ((T2_RESTARTS++)) && \ + queue_notify "$container OOM killed on $(hostname) — restarted" "warning" + fi + done <<< "$ALL_CONTAINERS" + fi + + # Crash loop detection + if [[ "$WATCHDOG_NOTIFY_CRASHLOOP" == "true" ]]; then + while IFS= read -r container; do + [[ -z "$container" ]] && continue + [[ -n "${IGNORE_MAP[$container]:-}" ]] && continue + is_skipped "$container" && continue + RESTART_COUNT=$(docker inspect -f '{{.RestartCount}}' "$container" 2>/dev/null || echo 0) + PREV_COUNT=$(grep "^${container}_docker:" "$WATCHDOG_STATE_FILE" 2>/dev/null | cut -d: -f2 || echo 0) + set_strikes "${container}_docker" "$RESTART_COUNT" "$WATCHDOG_STATE_FILE" + if [[ "$RESTART_COUNT" -gt "$PREV_COUNT" && "$RESTART_COUNT" -gt 0 ]]; then + ((T2_WARNINGS++)) + if [[ "$RESTART_COUNT" -ge "$WATCHDOG_CRASH_LIMIT" ]]; then + error "$container — crash loop CRITICAL: $RESTART_COUNT restarts" + queue_notify "$container crash loop CRITICAL on $(hostname) — manual intervention needed" "critical" + else + warn "$container — restarted since last check (total: $RESTART_COUNT)" + queue_notify "$container restarted on $(hostname) — count: $RESTART_COUNT" "warning" + fi + fi + done <<< "$ALL_CONTAINERS" + fi + + # Dead containers + if [[ "$WATCHDOG_RESTART_DEAD" == "true" ]]; then + DEAD=$(docker ps -a --filter status=dead --format "{{.Names}}" 2>/dev/null) + while IFS= read -r container; do + [[ -z "$container" ]] && continue + [[ -n "${IGNORE_MAP[$container]:-}" ]] && continue + is_skipped "$container" && continue + error "$container — dead" + ((T2_WARNINGS++)) + RESTART_COUNT=$(get_restart_count "$container") + if [[ "$RESTART_COUNT" -ge "$WATCHDOG_CONTAINER_RESTART_LIMIT" ]]; then + add_to_skip_list "$container" "dead — restarted $RESTART_COUNT times" + elif [[ "$DRY_RUN" == false ]]; then + docker rm "$container" >/dev/null 2>&1 + if docker start "$container" >/dev/null 2>&1; then + success "$container removed from dead state and restarted" + log_restart "$container" + ((T2_RESTARTS++)) + queue_notify "$container was dead on $(hostname) — restarted" "warning" + fi + fi + done <<< "$DEAD" + fi + + # Unexpected exits + if [[ "$WATCHDOG_RESTART_CRASHED" == "true" ]]; then + CRASHED=$(docker ps -a \ + --filter status=exited \ + --format "{{.Names}}|{{.Status}}" 2>/dev/null | \ + grep -v "Exited (0)") + while IFS='|' read -r container status; do + [[ -z "$container" ]] && continue + [[ -n "${IGNORE_MAP[$container]:-}" ]] && continue + is_skipped "$container" && continue + SKIP=false + for req in "${WATCHDOG_REQUIRED_CONTAINERS[@]}"; do + [[ "$container" == "$req" ]] && SKIP=true && break + done + [[ "$SKIP" == true ]] && continue + error "$container — $status (unexpected exit)" + ((T2_WARNINGS++)) + result=0 + safe_restart "$container" "unexpected exit" || result=$? + [[ $result -eq 0 ]] && ((T2_RESTARTS++)) && \ + queue_notify "$container crashed on $(hostname) ($status) — restarted" "warning" + done <<< "$CRASHED" + fi + fi + + # Send notifications if any events this cycle + flush_notify + + # Only log summary if something happened — quiet when all healthy + TOTAL_RESTARTS=$(( T1_RESTARTS + T2_RESTARTS )) + TOTAL_WARNINGS=$(( T1_WARNINGS + T2_WARNINGS )) + + if [[ "$TOTAL_RESTARTS" -gt 0 || "$TOTAL_WARNINGS" -gt 0 ]]; then + CYCLE_END=$(date +%s) + echo "" + echo "━━━ $ICON_WATCHDOG Cycle $CYCLE — $(date '+%Y-%m-%d %H:%M:%S') ━━━" + echo "$ICON_WATCHDOG T1: $T1_RESTARTS restarts / $T1_WARNINGS warnings" + echo "$ICON_WATCHDOG T2: $T2_RESTARTS restarts / $T2_WARNINGS warnings" + echo "$ICON_TIME Duration: $(format_duration $(( CYCLE_END - CYCLE_START )))" + echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━" + else + log "Cycle $CYCLE — all healthy ($(date '+%H:%M:%S'))" + # Heartbeat — periodic proof of life + if [[ "${DOCKER_WATCHDOG_HEARTBEAT:-true}" == true ]]; then + local hb_seconds=$(( ${DOCKER_WATCHDOG_HEARTBEAT_HOURS:-1} * 3600 )) + local uptime_seconds=$(( CYCLE * DOCKER_WATCHDOG_INTERVAL )) + if (( uptime_seconds % hb_seconds < DOCKER_WATCHDOG_INTERVAL )); then + local uptime_hr=$(( uptime_seconds / 3600 )) + info "♥ docker_watchdog alive — ~${uptime_hr}hr uptime ($(date '+%H:%M:%S'))" + fi + fi fi # Sleep until next cycle — interruptible by SIGTERM - sleep "$SYSTEM_WATCHDOG_INTERVAL" & + sleep "$DOCKER_WATCHDOG_INTERVAL" & wait $! done \ No newline at end of file diff --git a/Master.conf b/Master.conf index 4d3c778..676e44c 100644 --- a/Master.conf +++ b/Master.conf @@ -134,12 +134,12 @@ # Gitea self-hosted repository — used by git_pull_execute.sh. # Detects Gitea container location at runtime — works through failover automatically. # Falls back to GITEA_DOMAIN if local and Tailscale both fail. - GITEA_CONTAINER="Gitea" - GITEA_REPO_PATH="FailedProxy/Unraid_Scripts.git" - GITEA_DOMAIN="" # e.g. git.gmer4lfe.com — requires NPM + DNS - TARGET_DIR="/mnt/user/appdata/unraid_scripts" - GITEA_SSH_KEY="/root/.ssh/unraid_gitea" - SSH_PORT=221 + GITEA_CONTAINER="Gitea" # exact Docker container name + GITEA_REPO_PATH="FailedProxy/Unraid_Scripts.git" # repo path on Gitea server + GITEA_DOMAIN="" # e.g. git.gmer4lfe.com — requires NPM + DNS setup + TARGET_DIR="/mnt/user/appdata/unraid_scripts" # where scripts are cloned to + GITEA_SSH_KEY="/root/.ssh/unraid_gitea" # SSH key for authenticating to Gitea + SSH_PORT=221 # Gitea SSH port (default 22, Gitea often uses 222/221) # ============================================================================================== # ── ORCHESTRATORS ────────────────────────────────────────────────────────────────────────────── @@ -261,13 +261,19 @@ MEDIA_MANAGEMENT_JOBS=( # Global fallback values used when no profile match is found. # Media shares in HOST*_DAILY_SYNC_SHARES always use these globals — no profile needed. # Appdata shares match profiles by directory basename (lowercased). - BW_LIMIT=12500 - RETRY_COUNT=3 - SLEEP=300 - CRITICAL_CONTAINER_NAMES=() - DELAYED_CONTAINERS=() - CONTAINER_DELAY=5 - EXCLUDE_DIRS=() +# If a profile key exists it overrides the global. If missing the global is used. + + BW_LIMIT=12500 # KB/s — 12500 ≈ 100Mbit — network transfer speed cap + RETRY_COUNT=3 # retry attempts if rsync fails before giving up + SLEEP=300 # seconds between retry attempts + CRITICAL_CONTAINER_NAMES=() # containers to stop on REMOTE before rsync — profiles override + DELAYED_CONTAINERS=() # containers needing delay before starting — profiles override + CONTAINER_DELAY=5 # seconds to wait before starting delayed containers + EXCLUDE_DIRS=() # directories to exclude from transfer — profiles override + +# --delete removes files on remote that no longer exist on source (mirror behaviour) +# --inplace writes directly to destination — better for large files, avoids temp copies +# --no-whole-file forces delta transfer even on fast connections — sends only changed blocks DEFAULT_RSYNC_OPTS=(-av --info=progress2 --human-readable --bwlimit="$BW_LIMIT" --delete --inplace --no-whole-file) # ━━━ Remote Health Checks ━━━ @@ -302,15 +308,18 @@ declare -A PROFILE_RSYNC_OPTS=( [emby-failover]="-av --human-readable --bwlimit=$BW_LIMIT --delete --inplace --no-whole-file" ) +# Per-profile bandwidth limits in KB/s — overrides global BW_LIMIT for that profile only +# Lower for shares running alongside other jobs, higher for time-sensitive critical data declare -A PROFILE_BW_LIMIT=( - [arrs_stack]=5000 - [critical-data]=9500 + [arrs_stack]=5000 # lower — runs alongside other syncs, avoids saturating link + [critical-data]=9500 # high — small dataset, get it synced fast and clean [gmer4lfe]=8000 - [important-data]=9500 - [emby]=8000 - [emby-failover]=9500 + [important-data]=9500 # high — database sync needs to be fast + [emby]=8000 # medium — large full mirror, steady transfer + [emby-failover]=9500 # high — small critical dataset, sync as fast as possible ) +# Retry attempts per profile — how many times to retry before giving up on a failed sync declare -A PROFILE_RETRY_COUNT=( [arrs_stack]=3 [critical-data]=3 @@ -320,48 +329,60 @@ declare -A PROFILE_RETRY_COUNT=( [emby-failover]=3 ) +# Seconds to wait between retry attempts +# emby-failover shorter — frequent sync, faster retry on transient failures declare -A PROFILE_SLEEP=( [arrs_stack]=300 [critical-data]=300 [gmer4lfe]=300 [important-data]=300 [emby]=300 - [emby-failover]=120 + [emby-failover]=120 # shorter — frequent dirty sync, retry faster ) # Containers stopped on BOTH LOCAL and REMOTE servers before rsync. -# Local stops first — flushes databases cleanly. Remote stops next — prevents writes while receiving. +# Local stops first — flushes databases cleanly before pushing data out. +# Remote stops next — prevents writes to destination while receiving. +# Only containers that were running get restarted — stopped containers stay stopped. # Same container names on both servers — consistent naming is required by this ecosystem. -# If a container is not found it is skipped gracefully, not errored. +# If a container is not found on a server it is skipped gracefully, not errored. # SPACE-SEPARATED STRINGS — converted to array at runtime declare -A PROFILE_CRITICAL_CONTAINER_NAMES=( [arrs_stack]="Sonarr Lidarr Readarr Radarr Prowlarr Bazarr Pinchflat" [critical-data]="Mariadb-Authelia Mariadb-Authelia-Secondary Redis-Authelia Redis-Authelia-Secondary Lldap-Gmer4Lfe NginxProxyManager Authelia Authelia-Secondary" [gmer4lfe]="Organizrv2-Gmer4Lfe UptimeKuma-Gmer4Lfe VaultWarden-Gmer4Lfe" [important-data]="Postgres-NextCloud NextCloud" - [emby]="Emby" - [emby-failover]="" + [emby]="Emby" # weekly clean sync — both Emby instances stopped, WAL checkpointed + [emby-failover]="" # dirty sync — Emby stays running both sides, WAL excluded from sync ) +# Containers that need a delay before starting after rsync completes. +# Database containers must be accepting connections before dependent apps start. +# Authelia waits for Mariadb + Redis. NextCloud waits for Postgres. # SPACE-SEPARATED STRINGS — converted to array at runtime declare -A PROFILE_DELAYED_CONTAINERS=( [arrs_stack]="" - [critical-data]="Authelia Authelia-Secondary" + [critical-data]="Authelia Authelia-Secondary" # wait for Mariadb + Redis to be ready [gmer4lfe]="" - [important-data]="NextCloud" + [important-data]="NextCloud" # wait for Postgres to accept connections [emby]="" [emby-failover]="" ) +# Seconds to wait before starting delayed containers +# 15s gives Mariadb, Redis, and LLDAP time to accept connections before Authelia starts declare -A PROFILE_CONTAINER_DELAY=( [arrs_stack]=5 - [critical-data]=15 + [critical-data]=15 # Mariadb + Redis need time to accept connections [gmer4lfe]=5 - [important-data]=10 + [important-data]=10 # Postgres needs time before NextCloud [emby]=5 [emby-failover]=5 ) +# Directories excluded from rsync transfer per profile +# emby-failover excludes WAL files — safe to sync while Emby is running +# emby clean sync only excludes logs, transcodes, cache — full metadata mirror # SPACE-SEPARATED STRINGS — converted to array at runtime declare -A PROFILE_EXCLUDE_DIRS=( [arrs_stack]="logs *.tmp" @@ -369,9 +390,13 @@ declare -A PROFILE_EXCLUDE_DIRS=( [important-data]="logs *.tmp" [critical-data]="logs *.tmp *.log nginx/temp nginx/cache __pycache__ notification.txt" [emby]="logs transcodes cache crash*" + # emby-failover: Emby running, WAL excluded — only safe critical data synced + # users.db, library.db, authentication.db, config/ — everything else excluded [emby-failover]="logs transcodes cache metadata *.db-wal *.db-shm crash* plugins root" ) +# Skip per-disk space check for these profiles — appdata syncs go to cache/appdata +# not to array disks, so disk space check is irrelevant and just slows things down declare -A PROFILE_SKIP_DISK_CHECK=( [arrs_stack]=true [critical-data]=true @@ -489,29 +514,45 @@ FAILOVER_HOST2_RUNS_FOR_HOST1_18HR=( ) # ━━━ Tier Delay Settings ━━━ -# Minutes before each tier activates. Tier 1 is always immediate. -HOST1_TIER2_DELAY=240 -HOST1_TIER3_DELAY=720 -HOST1_TIER4_DELAY=1440 +# How long the primary server must be down before each tier activates — in minutes. +# Tier 1 is always immediate — Live TV and media can't wait. +# Set independently per host — adjust based on hardware and what's worth starting. +# Longer delays = less resource usage on covering server but slower recovery. +# +# HOST1's containers running on HOST2 (HOST1 is down): +HOST1_TIER2_DELAY=240 # 4 hours — NextCloud, Immich — can wait +HOST1_TIER3_DELAY=720 # 12 hours — secondary services — Gitea etc. +HOST1_TIER4_DELAY=1440 # 24 hours — full workflow — arrs and downloaders +# HOST2's containers running on HOST1 (HOST2 is down): HOST2_TIER2_DELAY=240 HOST2_TIER3_DELAY=720 HOST2_TIER4_DELAY=1440 # ━━━ Rsync Writeback Jobs ━━━ -# Syncs critical appdata back to primary on handback — containers stopped before this runs. -# Short outages skip writeback — primary state is more reliable than dirty sync data. -# Tier 4 automatically syncs HOST*_DAILY_SYNC_SHARES — add edge cases here only. +# Syncs critical appdata BACK to primary server during handback after failover. +# Containers are stopped before writeback runs — clean source, no competing writes. +# Purpose: primary comes back online with the state that built up during its outage +# (watch states, auth changes, library updates that happened on HOST2) +# +# HOST*_TIER1_WRITEBACK_DELAY: +# Short outages skip Tier 1 writeback — primary state is more reliable than dirty sync data +# Only writeback if outage lasted longer than this many minutes +# 60 minutes = if HOST1 was down less than 1hr, don't bother writing back Emby +# +# Tier 4 writeback automatically syncs HOST*_DAILY_SYNC_SHARES back — no need to list those here +# Only add paths that are NOT in DAILY_SYNC_SHARES and need writeback after extended outage -HOST1_TIER1_WRITEBACK_DELAY=60 +HOST1_TIER1_WRITEBACK_DELAY=60 # minutes — skip Emby writeback if outage under 1hr HOST2_TIER1_WRITEBACK_DELAY=60 +# HOST1 writeback — run by HOST2 during HOST1 handback FAILOVER_HOST1_WRITEBACK_TIER1=( - "/mnt/user/Media_Server/Emby" + "/mnt/user/Media_Server/Emby" # watch states, playstates built up during outage ) FAILOVER_HOST1_WRITEBACK_TIER2=( - "/mnt/user/appdata-Failover/Important-Data" + "/mnt/user/appdata-Failover/Important-Data" # NextCloud + Postgres — files added during outage ) FAILOVER_HOST1_WRITEBACK_TIER3=( @@ -519,9 +560,11 @@ FAILOVER_HOST1_WRITEBACK_TIER3=( ) FAILOVER_HOST1_WRITEBACK_TIER4=( - "/mnt/user/appdata-Failover/Arrs_Stack" + # Edge cases outside HOST1_DAILY_SYNC_SHARES + "/mnt/user/appdata-Failover/Arrs_Stack" # arr databases — downloads queued during outage ) +# HOST2 writeback — run by HOST1 during HOST2 handback FAILOVER_HOST2_WRITEBACK_TIER1=( # "/mnt/user/appdata-Failover/Jayred365-Emby" ) @@ -535,6 +578,7 @@ FAILOVER_HOST2_WRITEBACK_TIER3=( ) FAILOVER_HOST2_WRITEBACK_TIER4=( + # Edge cases outside HOST2_DAILY_SYNC_SHARES "/mnt/user/appdata-Failover/Arrs_Stack" ) @@ -543,18 +587,25 @@ FAILOVER_HOST2_WRITEBACK_TIER4=( # ============================================================================================== # ━━━ Docker Daily Restart ━━━ -# Restarted by docker_daily_restart.sh via daily_sync_maintenance.sh — 1am daily. +# Containers restarted every day by docker_daily_restart.sh via daily_sync_maintenance.sh. +# These containers run better with a daily restart — not just "keeping things fresh". +# Dispatcharr specifically degrades over time without restart — daily is intentional. +# Schedule is set in daily_sync_maintenance.sh — runs at 1am as part of daily window. +# Case-sensitive — must match exact Docker container names. DAILY_RESTART_CONTAINERS=( "NginxProxyManager" "Authelia" "Dispatcharr-Iptv-Users" - "Dispatcharr" + "Dispatcharr" # Live TV scheduler — degrades without daily restart "Dispatcharr-Basic" "ErsatzTV-Emby" ) # ━━━ Docker Weekly Restart ━━━ -# Restarted by docker_weekly_restart.sh via weekly_sync_maintenance.sh — Sunday 2:30am. +# Less critical services restarted weekly by docker_weekly_restart.sh. +# Called by weekly_sync_maintenance.sh Sunday 2:30am — containers already stopped +# for the weekly sync window so restart adds zero extra downtime. +# Weekly restarts also catch any pending image updates not applied during weekly sync. WEEKLY_RESTART_CONTAINERS=( "NextCloud" "Organizrv2-Gmer4Lfe" @@ -567,14 +618,25 @@ WEEKLY_RESTART_CONTAINERS=( # Started by array_start.sh — runs until array stops. # Re-sources Master.conf each cycle — add/remove containers without restarting watchdog. # Silent when all healthy — only logs when something needs attention. +# Heartbeat fires periodically as proof of life even when everything is healthy. # -# Tier 1 — strict monitoring of explicitly configured containers -# Memory hard limits, CPU thresholds, HTTP responsiveness, required container checks -# Tier 2 — global health scan of ALL running containers -# Unhealthy status, OOM kills, crash loops, dead containers, unexpected exits +# Tier 1 — strict monitoring of explicitly configured containers: +# Memory hard limits — immediate restart if container exceeds limit +# CPU thresholds — strike system, restart after CPU_FAIL_LIMIT sustained strikes +# HTTP responsiveness — strike system, restart after RESP_FAIL_LIMIT failed checks +# Required containers — must always be running, strike + skip list with auto-clear +# +# Tier 2 — global health scan of ALL running containers: +# Unhealthy status — Docker HEALTHCHECK unhealthy → restart +# OOM killed — kernel killed container → restart + notify +# Crash loop detection — RestartCount climbing → notify, critical above limit +# Dead containers — remove and restart +# Unexpected exits — non-zero exit code → restart # Memory hard limits in MB — immediate restart if exceeded -# 20GB=20480 16GB=16384 12GB=12288 8GB=8192 4GB=4096 2GB=2048 1GB=1024 +# Container restarted the moment it crosses this line — no strike system +# 20GB=20480 16GB=16384 12GB=12288 10GB=10240 +# 8GB=8192 6GB=6144 4GB=4096 2GB=2048 1GB=1024 declare -A WATCHDOG_CONTAINERS=( ["Emby"]=16384 ["LidaTube"]=6144 @@ -582,6 +644,9 @@ declare -A WATCHDOG_CONTAINERS=( ["Code-Server"]=1024 ) +# HTTP health check URLs — checked every cycle, strike system before restart +# Container must respond with HTTP 200 within CURL_TIMEOUT seconds +# Per-host — HOST1 and HOST2 may run different containers on different ports declare -A HOST1_WATCHDOG_CONTAINER_URLS=( ["Emby"]="http://localhost:8096" ) @@ -590,6 +655,11 @@ declare -A HOST2_WATCHDOG_CONTAINER_URLS=( ["Emby"]="http://localhost:8096" ) +# Required containers — must always be running +# Strike system: SYS_WATCHDOG_STRIKE_LIMIT strikes before restart attempt +# Persistent skip list: added after WATCHDOG_CONTAINER_RESTART_LIMIT restarts in window +# Skip list auto-clears when container recovers — no manual intervention needed +# Per-host — each server has different critical containers HOST1_WATCHDOG_REQUIRED_CONTAINERS=( "NginxProxyManager" "Lldap-Gmer4Lfe" @@ -605,33 +675,76 @@ HOST2_WATCHDOG_REQUIRED_CONTAINERS=( # add HOST2 required containers here ) +# Strike state file — /tmp resets on reboot which is correct +# Fresh start after reboot means no stale strikes carrying over WATCHDOG_STATE_FILE="/tmp/container_watchdog_state.db" - SOFT_CPU_THRESHOLD=80 - HARD_CPU_THRESHOLD=85 - CPU_FAIL_LIMIT=2 - SOFT_MEM_THRESHOLD=80 - RESP_FAIL_LIMIT=2 - CURL_TIMEOUT=5 - DOCKER_WATCHDOG_INTERVAL=900 +# CPU thresholds — normalised against total core count automatically at runtime +# SOFT = warn only, HARD = strike toward restart +# CPU_FAIL_LIMIT = consecutive HARD strikes before restart + SOFT_CPU_THRESHOLD=80 # warn at this % of total system CPU + HARD_CPU_THRESHOLD=85 # strike at this % of total system CPU + CPU_FAIL_LIMIT=2 # consecutive hard CPU strikes before container restart + +# Memory soft threshold — warn when container reaches this % of its WATCHDOG_CONTAINERS hard limit +# Does not trigger restart — informational only + SOFT_MEM_THRESHOLD=80 + +# HTTP responsiveness — consecutive failed checks before restart +# CURL_TIMEOUT = seconds before curl gives up on a single check + RESP_FAIL_LIMIT=2 # consecutive failed checks before restart + CURL_TIMEOUT=5 # seconds per check before timeout + +# How often the watchdog runs its checks +# 900 = 15 minutes — long enough to not be noisy, short enough to catch issues quickly +# Containers have this long to recover before next check + DOCKER_WATCHDOG_INTERVAL=900 # seconds between watchdog cycles + +# Heartbeat — proof of life logged periodically even when everything is healthy +# Useful to confirm the watchdog is still running without flooding logs + DOCKER_WATCHDOG_HEARTBEAT=true # true = log heartbeat / false = completely silent + DOCKER_WATCHDOG_HEARTBEAT_HOURS=1 # how often to log heartbeat (hours) + +# Tier 2 master toggle — set false to disable global container scanning entirely +# When false only WATCHDOG_CONTAINERS and required containers are monitored WATCHDOG_SCAN_ALL=true +# Containers to skip in Tier 2 scan entirely +# Useful for containers that legitimately exit/restart frequently WATCHDOG_SCAN_IGNORE=( # "container-name" ) - WATCHDOG_RESTART_UNHEALTHY=true - WATCHDOG_RESTART_DEAD=true - WATCHDOG_RESTART_CRASHED=true - WATCHDOG_NOTIFY_OOM=true - WATCHDOG_NOTIFY_CRASHLOOP=true +# Individual Tier 2 check toggles — disable specific checks without disabling Tier 2 + WATCHDOG_RESTART_UNHEALTHY=true # restart containers with Docker HEALTHCHECK = unhealthy + WATCHDOG_RESTART_DEAD=true # restart containers in dead state + WATCHDOG_RESTART_CRASHED=true # restart containers that exited with non-zero code + WATCHDOG_NOTIFY_OOM=true # notify + restart OOM killed containers + WATCHDOG_NOTIFY_CRASHLOOP=true # notify when Docker RestartCount keeps climbing + +# Crash loop threshold — notify critical if Docker has restarted this many times total +# Above this number the notification escalates to critical — manual intervention needed WATCHDOG_CRASH_LIMIT=5 - WATCHDOG_STARTUP_GRACE=600 - WATCHDOG_CONTAINER_RESTART_LIMIT=3 - WATCHDOG_CONTAINER_RESTART_WINDOW=1 + +# Startup grace period — skip restarts while system is still booting after array start +# Prevents watchdog from restarting containers that are legitimately still initializing + WATCHDOG_STARTUP_GRACE=600 # seconds after boot before watchdog acts on failures + +# Restart loop protection — stops hammering a broken container +# If watchdog restarts a container more than LIMIT times in WINDOW hours → skip list +# Skip list auto-clears when container recovers healthy + WATCHDOG_CONTAINER_RESTART_LIMIT=3 # max watchdog restarts allowed in window + WATCHDOG_CONTAINER_RESTART_WINDOW=1 # rolling window in hours WATCHDOG_CONTAINER_RESTART_LOG="/boot/config/container_restart_history.db" + +# Notification batching — one clean summary per cycle instead of one ping per event +# true = batch all events into one notification at end of cycle +# false = send one notification per event (noisy on busy systems) WATCHDOG_BATCH_NOTIFY=true +# Dependency ordering — skip restarting a container if its dependency is also down +# Prevents restarting Authelia before its database is ready +# Space-separated list of dependencies per container declare -A WATCHDOG_DEPENDENCIES=( ["Authelia"]="Mariadb-Authelia Redis-Authelia" ["Authelia-Secondary"]="Mariadb-Authelia Redis-Authelia-Secondary" @@ -639,14 +752,18 @@ declare -A WATCHDOG_DEPENDENCIES=( ) # ━━━ Docker Network Connect ━━━ -# Connects containers to extra networks on array start via array_start.sh. +# Connects containers to extra Docker networks on array start via array_start.sh. +# Useful for containers that need their own custom network but also need to be +# reachable from your main custom bridge network. +# Every container in NETWORK_CONNECT_CONTAINERS is connected to every network in +# NETWORK_CONNECT_NETWORKS — containers not found are skipped gracefully. NETWORK_CONNECT_CONTAINERS=( "memcached" "Npm-CrowdSec" ) NETWORK_CONNECT_NETWORKS=( - "high-availability" + "high-availability" # must exist before array start — create in Docker settings ) # ============================================================================================== @@ -654,33 +771,52 @@ NETWORK_CONNECT_NETWORKS=( # ============================================================================================== # ━━━ Reboot ━━━ +# Seconds of warning broadcast to logged-in users before server_reboot.sh reboots. +# Gives users time to save work — 300s = 5 minutes REBOOT_SLEEP=300 # ━━━ Mover ━━━ +# Seconds to wait before mover_stop.sh sends SIGTERM to the mover process. +# Gives mover time to finish current file transfer before being interrupted. MOVER_STOP_TIMEOUT=300 # ━━━ Syslog Filter ━━━ +# Path for the rsyslog filter file that suppresses Docker veth interface noise. +# Docker creates a new veth interface for each container — generates hundreds of +# log lines per hour that have no diagnostic value. Filter removes them at source. FILTER_FILE="/etc/rsyslog.d/ignore-docker-veth.conf" # ━━━ PHP-FPM ━━━ +# Higher max_children allows more concurrent PHP requests to the unRAID WebGUI. +# Default is very low — increasing it prevents WebGUI slowdowns under load. +# 250 is safe for servers with 32GB+ RAM. PHP_CONF="/etc/php-fpm.d/www.conf" PHP_MAX_CHILDREN=250 # ━━━ Clear Logs ━━━ +# System log files cleared weekly to prevent rootfs fill over time. +# These grow continuously — without clearing they eventually consume all rootfs space. LOG_FILES=(/var/log/syslog /var/log/messages /var/log/dmesg) # ━━━ WebGUI Watchdog ━━━ +# Monitors unRAID WebGUI responsiveness — escalates through nginx restart → emhttp restart. +# Separate from docker_watchdog — this monitors the unRAID UI itself, not containers. +# WEBGUI_NGINX_WAIT = seconds after nginx restart before rechecking +# WEBGUI_EMHTTP_WAIT = seconds after emhttp restart before rechecking WEBGUI_URL="http://localhost" - WEBGUI_TIMEOUT=5 - WEBGUI_NGINX_WAIT=15 - WEBGUI_EMHTTP_WAIT=30 + WEBGUI_TIMEOUT=5 # seconds before curl gives up on WebGUI check + WEBGUI_NGINX_WAIT=15 # seconds after nginx restart before rechecking + WEBGUI_EMHTTP_WAIT=30 # seconds after emhttp restart before rechecking # ============================================================================================== # ── MEDIA ───────────────────────────────────────────────────────────────────────────────────── # ============================================================================================== # ━━━ Media Permissions ━━━ -# Applied recursively by media_shares_permissions.sh via MEDIA_MANAGEMENT_JOBS. +# Applied recursively to all shares in MEDIA_PERMISSION_SHARES by media_shares_permissions.sh. +# Runs first in MEDIA_MANAGEMENT_JOBS — arr cleanup scripts depend on correct ownership. +# 777 mode = read/write/execute for all users — standard for unRAID media shares +# nobody:users = standard unRAID media share ownership PERMISSIONS_MODE="777" PERMISSIONS_OWNER="nobody:users" @@ -759,6 +895,7 @@ MEDIA_FILE_PATTERNS=( HOST1_LIDARR_URL="http://192.168.50.2:8686" HOST1_LIDARR_API_KEY="b2977e71ef074bc0a0529d9fcce3b2dc" HOST1_LIDARR_MUSIC_ROOT="/mnt/user/Music-New" +LIDARR_LOCK_WARN_AGE=3600 # 1hr — large libraries take time, not stuck # Container path → host path translation # Lidarr stores file paths using container paths — script scans host paths @@ -771,12 +908,18 @@ declare -A HOST2_LIDARR_PATH_MAP=( # ["/ext-music"]="/mnt/user/Music-New" ) -LIDARR_ORPHAN_AGE=7 +LIDARR_ORPHAN_AGE=7 # days — files must be older than this before eligible for deletion + # protects files that may still be mid-import or recently downloaded LIDARR_EXTENSIONS=("flac" "mp3" "m4a" "wav" "aac" "ogg" "opus" "wma") LIDARR_PROTECTED_PATTERNS=("*.jpg" "*.jpeg" "*.png" "*.nfo" "*.lrc") + # NEVER deleted — cover art, metadata, lyrics + # Lidarr generates these but doesn't include them in trackFile API + # Without this protection cleanup would delete all your artwork LIDARR_MAX_DELETE_GB=1 # require --i-know-what-im-doing if deletion exceeds this LIDARR_MIN_TRACKED_PCT=80 # abort if tracked count drops below this % of last run + # protects against API returning partial data on a bad day LIDARR_TRACKED_COUNT_FILE="/boot/config/lidarr_tracked.count" + # persists last known tracked count for percentage comparison # ── Sonarr ──────────────────────────────────────────────────────────────────────────────────── HOST1_SONARR_URL="http://192.168.50.2:8989" @@ -802,9 +945,11 @@ declare -A HOST2_SONARR_PATH_MAP=( # ["/tv"]="/mnt/user/Anime_Shows" ) -SONARR_ORPHAN_AGE=7 +SONARR_ORPHAN_AGE=7 # days — files must be older than this before eligible for deletion SONARR_EXTENSIONS=("mkv" "mp4" "avi" "m4v" "ts" "wmv" "mov") SONARR_PROTECTED_PATTERNS=("*.jpg" "*.jpeg" "*.png" "*.nfo" "*.srt" "*.sub" "*.ass" "*.ssa") + # NEVER deleted — cover art, metadata, subtitles + # Sonarr generates these but doesn't include them in episodefile API # ── Radarr ──────────────────────────────────────────────────────────────────────────────────── HOST1_RADARR_URL="http://192.168.50.2:7878" @@ -830,9 +975,11 @@ declare -A HOST2_RADARR_PATH_MAP=( # ["/anime-movies"]="/mnt/user/Anime_Movies" ) -RADARR_ORPHAN_AGE=7 +RADARR_ORPHAN_AGE=7 # days — files must be older than this before eligible for deletion RADARR_EXTENSIONS=("mkv" "mp4" "avi" "m4v" "wmv" "mov") RADARR_PROTECTED_PATTERNS=("*.jpg" "*.jpeg" "*.png" "*.nfo" "*.srt" "*.sub" "*.ass" "*.ssa") + # NEVER deleted — cover art, metadata, subtitles + # Radarr generates these but doesn't include them in moviefile API # ━━━ Arr Failed/Stalled Recovery ━━━ # Auto blocklist + re-search failed imports and stalled downloads. @@ -849,12 +996,16 @@ RADARR_PROTECTED_PATTERNS=("*.jpg" "*.jpeg" "*.png" "*.nfo" "*.srt" "*.sub" "*.a # Lidarr runs on HOST1 only — exits cleanly on HOST2. ARR_IMPORT_RECOVERY_AGE=6 # hours — skip items newer than this + # gives the arr time to retry on its own before we intervene + # matches cron interval — items are eligible after one missed cycle -HOST1_SONARR_RECOVERY=true -HOST1_RADARR_RECOVERY=true -HOST1_LIDARR_RECOVERY=true # HOST1 only -HOST2_SONARR_RECOVERY=true -HOST2_RADARR_RECOVERY=true +# Per-arr enable/disable toggles — set false to temporarily disable without removing from cron +# Useful if an arr is having issues and you want to skip it for a few runs +HOST1_SONARR_RECOVERY=true # Tv_Shows import recovery +HOST1_RADARR_RECOVERY=true # Movies import recovery +HOST1_LIDARR_RECOVERY=true # Music import recovery — HOST1 only, exits cleanly on HOST2 +HOST2_SONARR_RECOVERY=true # Anime_Shows import recovery +HOST2_RADARR_RECOVERY=true # Anime_Movies import recovery # ============================================================================================== # ── TRANSCODES ──────────────────────────────────────────────────────────────────────────────── @@ -873,16 +1024,47 @@ HOST2_RADARR_RECOVERY=true # Standard rprivate bind mounts lock the inode — sessions drift to SSD permanently. # ━━━ Transcode Manager ━━━ +# tmpfs mount point — created at array start by ramdisk_setup.sh +# Must exist before Emby starts so the symlink resolves correctly RAMDISK_PATH="/mnt/ramdisk_transcodes" + +# Ramdisk size ceiling — tmpfs only uses RAM actually needed, not the full size upfront +# Set this to a comfortable limit based on your typical concurrent stream count +# Real-world: 9 streams peaked at ~5.5GB — 8G gives comfortable headroom RAMDISK_SIZE="8G" + +# Symlink that Emby points at — this path NEVER changes regardless of ramdisk/SSD state +# Emby resolves the symlink once per session at start — symlink flips are transparent +# Must match the container path configured in Emby's Extra Parameters TRANSCODE_LINK="/mnt/ram-transcode" + +# SSD fallback location — where transcodes land when ramdisk is too full +# Must have enough free space to handle peak session load TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/" + +# Usage thresholds in GB — hysteresis gap between WARN and LOW prevents flip-flop +# RAMDISK_WARN_GB: flip symlink to SSD when ramdisk usage reaches this +# RAMDISK_LOW_GB: flip symlink back to ramdisk when usage drops to this +# Gap (6.8 - 5.5 = 1.3GB) means ramdisk must drop 1.3GB before flipping back +# Without hysteresis a session right at the threshold causes rapid flipping RAMDISK_WARN_GB=6.8 RAMDISK_LOW_GB=5.5 + +# Minimum free GB on SSD before allowing a flip to SSD +# Prevents flipping to SSD when it's almost full — that would be worse than a full ramdisk RAMDISK_SSD_MIN_GB=20 + +# File age thresholds in minutes before cleanup eligibility +# TRANSCODE_MAX_AGE: HLS segment files older than this with no active session = clean up +# TRANSCODE_ORPHAN_AGE: files with no matching session at all = clean up TRANSCODE_MAX_AGE=20 TRANSCODE_ORPHAN_AGE=30 + +# Notify if symlink flips this many times in one hour +# Frequent flips indicate the ramdisk is too small or thresholds need adjustment TRANSCODE_FLIP_WARN=3 + +# Permissions applied to ramdisk and SSD transcode directories TRANSCODE_OWNER="nobody:users" TRANSCODE_CHMOD="755" @@ -895,8 +1077,12 @@ HOST2_RADARR_RECOVERY=true # ssd — always uses SSD, never flips to ramdisk # use during ramdisk maintenance or after a ramdisk issue TRANSCODE_MANAGER_MODE="smart" # smart | ramdisk | ssd + +# Daily statistics log — read by weekly_health_digest.sh for transcode summary +# Tracks peak usage, flip count, session ratio, files cleaned per day +# Bounded to TRANSCODE_LOG_RETENTION days — auto-purges old entries on each write TRANSCODE_DAILY_LOG="/boot/config/transcode_daily.db" - TRANSCODE_LOG_RETENTION=90 + TRANSCODE_LOG_RETENTION=90 # days before old entries are purged # ━━━ Transcode Server Array ━━━ # All media servers sharing the ramdisk transcode space. @@ -918,30 +1104,48 @@ TRANSCODE_SERVERS=( # ============================================================================================== # ━━━ Certificate Monitor ━━━ +# Checks SSL certificate expiry via direct openssl connection — no NPM dependency. +# Checks the actual certificate served by each domain, not what NPM thinks it has. +# CERT_WARN_DAYS = notify this many days before expiry +# CERT_CRIT_DAYS = escalate to critical this many days before expiry +# CERT_TIMEOUT = seconds before giving up on the openssl connection CERT_MONITOR_DOMAINS=( "Gmer4Lfe.com" "Gmer4Lfe.us" ) - CERT_WARN_DAYS=30 - CERT_CRIT_DAYS=7 - CERT_TIMEOUT=10 + CERT_WARN_DAYS=30 # warn when cert expires within this many days + CERT_CRIT_DAYS=7 # critical alert within this many days + CERT_TIMEOUT=10 # seconds per domain check # ━━━ Backup Verify ━━━ -# Leave empty to use HOST*_DAILY_SYNC_SHARES automatically. +# Verifies rsync mirror health by comparing random file checksums between servers. +# Catches silent corruption or incomplete syncs that rsync itself wouldn't detect. +# Leave BACKUP_VERIFY_SHARES empty to use HOST*_DAILY_SYNC_SHARES automatically. +# BACKUP_VERIFY_SAMPLE = number of random files to checksum per share +# BACKUP_VERIFY_MIN_SIZE = skip files smaller than this (small files are rarely corrupted) BACKUP_VERIFY_SHARES=( - # leave empty to use daily sync shares automatically + # leave empty to use HOST*_DAILY_SYNC_SHARES automatically ) - BACKUP_VERIFY_SAMPLE=10 - BACKUP_VERIFY_MIN_SIZE=1M + BACKUP_VERIFY_SAMPLE=10 # random files to check per share + BACKUP_VERIFY_MIN_SIZE=1M # minimum file size to include in sample # ━━━ SMART Health ━━━ - SMART_TEMP_WARN=45 - SMART_TEMP_CRIT=55 +# Monitors drive SMART attributes — discovers all drives automatically via /dev/sd* and /dev/nvme*. +# Reads live SMART data — no persistent writes. +# SMART_IGNORE_DRIVES = drives to skip (boot USB, drives without meaningful SMART data) + SMART_TEMP_WARN=45 # Celsius — warn above this temperature + SMART_TEMP_CRIT=55 # Celsius — critical above this temperature SMART_IGNORE_DRIVES=( - "sda" + "sda" # boot USB — SMART not meaningful on flash drives ) # ━━━ ZFS Memory Snapshot ━━━ +# Weekly ZFS pool health and memory diagnostic report — informational only, no action taken. +# ZFS_REPORT_ARC_WARN_PCT = warn if ARC is using more than this % of its max +# ZFS_REPORT_FREE_WARN_GB = warn if less than this GB free RAM +# ZFS_REPORT_AVAIL_WARN_GB = warn if less than this GB available on ZFS pool +# ZFS_REPORT_DOCKER_TOP = how many top Docker containers to show by memory usage +# ZFS_REPORT_IGNORE_POOLS = individual disk pools to skip (unRAID array disks as ZFS) ZFS_REPORT_LOG="/var/log/zfs-weekly-health.log" ZFS_REPORT_ARC_WARN_PCT=90 ZFS_REPORT_FREE_WARN_GB=10 @@ -956,26 +1160,35 @@ ZFS_REPORT_IGNORE_POOLS=( ) # ━━━ Bandwidth Monitor ━━━ -# Called by rsync.sh after each sync — bounded write, minimal flash wear. +# Called automatically by rsync.sh after each sync — one bounded write per run. +# Tracks transfer size, duration and profile per sync for weekly summary reporting. +# BANDWIDTH_LOG_RETENTION = days to keep entries before auto-purging old records +# BANDWIDTH_WARN_GB = flag in weekly summary if a single sync exceeded this size BANDWIDTH_LOG="/boot/config/bandwidth_history.db" - BANDWIDTH_LOG_RETENTION=90 - BANDWIDTH_WARN_GB=50 + BANDWIDTH_LOG_RETENTION=90 # days before old entries are purged + BANDWIDTH_WARN_GB=50 # flag syncs larger than this in weekly report # ━━━ Health Digest ━━━ -# Reads existing state files — no new flash writes. -# Profiles: always | smart | weekly - DIGEST_PROFILE="weekly" - DIGEST_DAY="Sunday" - DIGEST_SMART_ON_WATCHDOG=true - DIGEST_SMART_ON_FAILOVER=true - DIGEST_SMART_ON_CERT_WARN=true - DIGEST_SMART_ON_BANDWIDTH=true +# Aggregated system health summary — reads existing state files, no new writes. +# Three profiles control when the digest email is sent: +# always — sends every run regardless of findings +# smart — sends only when DIGEST_SMART_ON_* conditions are found +# weekly — sends once per week on DIGEST_DAY only +# Smart profile triggers — set true to send digest when finding is detected: + DIGEST_PROFILE="weekly" # always | smart | weekly + DIGEST_DAY="Sunday" # day of week for weekly profile + DIGEST_SMART_ON_WATCHDOG=true # send if any watchdog strikes are active + DIGEST_SMART_ON_FAILOVER=true # send if failover state is not NORMAL + DIGEST_SMART_ON_CERT_WARN=true # send if any cert is under CERT_WARN_DAYS + DIGEST_SMART_ON_BANDWIDTH=true # send if any transfer exceeded BANDWIDTH_WARN_GB # ━━━ Emby Session Report ━━━ -# Weekly Emby usage statistics via API — no persistent writes. -# URL and API key from HOST1/HOST2_EMBY_URL and HOST1/HOST2_EMBY_API_KEY in Host Configuration. - EMBY_REPORT_DAYS=7 - EMBY_REPORT_TOP_N=10 +# Weekly Emby usage statistics via API — no persistent writes, queries fresh each run. +# Shows top content, most active users, session counts over the report period. +# URL and API key pulled from HOST1/HOST2_EMBY_URL and HOST1/HOST2_EMBY_API_KEY +# defined in Host Configuration at the top of this file — no duplication needed. + EMBY_REPORT_DAYS=7 # days to include in the report period + EMBY_REPORT_TOP_N=10 # number of top content items to show # ============================================================================================== # ── SYSTEM WATCHDOG ─────────────────────────────────────────────────────────────────────────── @@ -993,37 +1206,78 @@ ZFS_REPORT_IGNORE_POOLS=( SYS_WATCHDOG_REBOOT_LOG="/boot/config/system_watchdog_reboots.db" # reboot loop detection # ━━━ Strike and Reboot Loop Settings ━━━ - SYS_WATCHDOG_STRIKE_LIMIT=2 - SYSTEM_WATCHDOG_INTERVAL=300 # seconds between cycles (5min default) - SYS_WATCHDOG_REBOOT_LIMIT=3 - SYS_WATCHDOG_REBOOT_WINDOW_HRS=12 +# Strike system: a check must fail this many consecutive cycles before action is taken +# Single spikes (one bad reading) are ignored — sustained problems trigger reboot + SYS_WATCHDOG_STRIKE_LIMIT=2 # consecutive failures before reboot trigger + +# How often checks run — 300s = 5 minutes +# At STRIKE_LIMIT=2 and INTERVAL=300: problem must persist 10min before reboot + SYSTEM_WATCHDOG_INTERVAL=300 # seconds between watchdog cycles + +# Reboot loop protection — if system keeps rebooting something is seriously wrong +# After REBOOT_LIMIT reboots in REBOOT_WINDOW_HRS hours → shutdown instead of reboot +# Prevents infinite reboot loops when the underlying problem can't be fixed by rebooting + SYS_WATCHDOG_REBOOT_LIMIT=3 # max reboots before shutdown instead + SYS_WATCHDOG_REBOOT_WINDOW_HRS=12 # rolling window in hours + +# Heartbeat — proof of life logged periodically even when everything is healthy + SYSTEM_WATCHDOG_HEARTBEAT=true # true = log heartbeat / false = completely silent + SYSTEM_WATCHDOG_HEARTBEAT_HOURS=1 # how often to log heartbeat (hours) # ━━━ Thresholds ━━━ +# Set at "about to become unstable" levels — not "things are a bit high" +# These should be high enough that normal operation never triggers them + +# rootfs (/) usage percentage — when array is down rsync writes land on rootfs +# fills rapidly and can crash the server — 95% is almost too late, act fast SYS_WATCHDOG_ROOTFS_PCT=95 + +# /var/log usage percentage — log spam can fill rootfs, indicates something broken SYS_WATCHDOG_LOG_PCT=95 + +# Free RAM in GB — below this is critically low, OOM or swap imminent +# Your server has 128GB — 4GB free means something is consuming everything SYS_WATCHDOG_MEM_GB=4 + +# ZFS ARC pinned percentage — ARC not releasing after reclaim = memory stuck +# SYS_WATCHDOG_ARC_RELEASE_PCT = after reclaim attempt, if still above this → trigger SYS_WATCHDOG_ARC_PINNED_PCT=98 SYS_WATCHDOG_ARC_RELEASE_PCT=95 + +# Load average multiplier — threshold = MULTIPLIER × CPU core count +# MULTIPLIER=3 on 16-core = load average of 48 before triggering +# Set high — transcoding causes legitimate high load spikes SYS_WATCHDOG_LOAD_MULTIPLIER=3 + +# Zombie process count — large numbers indicate serious process management failure +# A few zombies are normal — 50 means something is very wrong SYS_WATCHDOG_ZOMBIE_LIMIT=50 + +# CPU temperature in Celsius — sustained high temp causes throttling or kernel panic +# 95°C is close to tjmax on most CPUs — triggers before thermal shutdown SYS_WATCHDOG_CPU_TEMP_MAX=95 # ━━━ Check Toggles ━━━ +# Disable individual checks without disabling the whole watchdog +# All enabled by default except load — transcoding causes legitimate load spikes SYS_WATCHDOG_CHECK_ROOTFS=true SYS_WATCHDOG_CHECK_LOG=true SYS_WATCHDOG_CHECK_RAM=true SYS_WATCHDOG_CHECK_ARC=true SYS_WATCHDOG_CHECK_CPU_TEMP=true - SYS_WATCHDOG_CHECK_LOAD=false # disabled — load spikes during transcoding are normal + SYS_WATCHDOG_CHECK_LOAD=false # disabled — load spikes during transcoding are normal SYS_WATCHDOG_CHECK_ZOMBIES=true - SYS_WATCHDOG_CHECK_CONTAINERS=true - SYS_WATCHDOG_CHECK_DOCKER_DAEMON=true + SYS_WATCHDOG_CHECK_CONTAINERS=true # checks docker_watchdog persistent skip list + SYS_WATCHDOG_CHECK_DOCKER_DAEMON=true # checks if Docker daemon is responding # ━━━ Abort Toggles ━━━ -# true = abort reboot if condition active / false = reboot anyway - SYS_WATCHDOG_ABORT_ON_ZFS_UNHEALTHY=true - SYS_WATCHDOG_ABORT_ON_PARITY=false - SYS_WATCHDOG_ABORT_ON_MOVER=false +# Conditions that prevent reboot even when a threshold is hit +# true = abort reboot if this condition is active (conservative — avoid data loss) +# false = reboot anyway (aggressive — a clean reboot beats a hard crash) +# Philosophy: aborting is safer for data, rebooting is safer for stability + SYS_WATCHDOG_ABORT_ON_ZFS_UNHEALTHY=true # rebooting with bad pool risks data loss + SYS_WATCHDOG_ABORT_ON_PARITY=false # aborting parity is better than crashing mid-check + SYS_WATCHDOG_ABORT_ON_MOVER=false # aborting move is better than crashing mid-move # ============================================================================================== # ──────────────────────── End Of User Variables ─────────────────────────────────────────────── diff --git a/unRAID_Essentials/system_watchdog.sh b/unRAID_Essentials/system_watchdog.sh index 0e87385..153b8b5 100644 --- a/unRAID_Essentials/system_watchdog.sh +++ b/unRAID_Essentials/system_watchdog.sh @@ -427,6 +427,15 @@ while [[ "$WATCHDOG_RUNNING" == true ]]; do do_reboot "${TRIGGERS[@]}" else log "Cycle $CYCLE — system healthy ($(date '+%H:%M:%S'))" + # Heartbeat — periodic proof of life + if [[ "${SYSTEM_WATCHDOG_HEARTBEAT:-true}" == true ]]; then + local hb_seconds=$(( ${SYSTEM_WATCHDOG_HEARTBEAT_HOURS:-1} * 3600 )) + local uptime_seconds=$(( CYCLE * SYSTEM_WATCHDOG_INTERVAL )) + if (( uptime_seconds % hb_seconds < SYSTEM_WATCHDOG_INTERVAL )); then + local uptime_hr=$(( uptime_seconds / 3600 )) + info "♥ system_watchdog alive — ~${uptime_hr}hr uptime ($(date '+%H:%M:%S'))" + fi + fi fi # Sleep until next cycle — interruptible by SIGTERM