#!/bin/bash # ----------------------------------------------------------------------------------------------- # --------------------------------- Docker Watchdog -------------------------------------------- # ----------------------------------------------------------------------------------------------- # First line of defense — monitors Docker containers for memory, CPU, HTTP responsiveness, # and unexpected stops. Restarts containers that exceed thresholds or go offline. # # Works alongside system_watchdog.sh: # docker_watchdog.sh — container level, minimal disruption, tries to self-heal # system_watchdog.sh — system level, last resort, reboots when healing fails # # Behaviour: # Memory — immediate restart if hard limit exceeded # CPU — strike system, restarts after CPU_FAIL_LIMIT consecutive hits # HTTP — strike system, restarts after RESP_FAIL_LIMIT consecutive failures # Required — strike system, restarts stopped containers, persistent skip list # prevents reboot loops, auto-clears when container recovers # Daemon — immediate notify if Docker daemon is unresponsive # # Strike cadence depends on cron schedule: # Every 15min + 2 strikes = 30min sustained before restart # Every 10min + 2 strikes = 20min sustained before restart # Every 5min + 2 strikes = 10min sustained before restart # # All configuration in Master.conf under Docker Watchdog section. # Supports --dry-run to show what would happen without acting. # ----------------------------------------------------------------------------------------------- SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" source "$SCRIPT_DIR/../Master.conf" source "$SCRIPT_DIR/../common.sh" parse_args "$@" TOTAL_CORES=$(nproc) # Persistent skip list — shared with system_watchdog.sh # Containers in this list are skipped until they recover SKIP_LIST_FILE="$SYS_WATCHDOG_FAILED_FILE" # ----------------------------------------------------------------------------------------------- # ━━━ $ICON_GEAR Setup ━━━ # ----------------------------------------------------------------------------------------------- echo "" echo "━━━ $ICON_GEAR Setup ━━━" if [[ "$EUID" -ne 0 ]]; then error "Must be run as root" exit 1 fi success "Running as root" info "$ICON_WATCHDOG Watchdog initialising — $TOTAL_CORES cores detected" touch "$WATCHDOG_STATE_FILE" 2>/dev/null || { error "Cannot create state file: $WATCHDOG_STATE_FILE" exit 1 } touch "$SKIP_LIST_FILE" 2>/dev/null || { error "Cannot create skip list: $SKIP_LIST_FILE" exit 1 } # ----------------------------------------------------------------------------------------------- # ━━━ $ICON_SUMMARY Status ━━━ # ----------------------------------------------------------------------------------------------- if [[ "$SHOW_STATUS" == true ]]; then echo "" echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━" echo "$ICON_WATCHDOG Monitored: ${!WATCHDOG_CONTAINERS[*]}" echo "$ICON_CONTAINERS Required: ${WATCHDOG_REQUIRED_CONTAINERS[*]}" echo "$ICON_MEM Soft mem: ${SOFT_MEM_THRESHOLD}% of limit" echo "$ICON_ZFS CPU soft: ${SOFT_CPU_THRESHOLD}%" echo "$ICON_ZFS CPU hard: ${HARD_CPU_THRESHOLD}%" echo "$ICON_RETRY CPU strikes: ${CPU_FAIL_LIMIT}" echo "$ICON_RETRY Resp strikes: ${RESP_FAIL_LIMIT}" echo "$ICON_TIME Curl timeout: ${CURL_TIMEOUT}s" echo "$ICON_NOTIFY Notify: unRAID=${NOTIFY_UNRAID:-false} Discord=$([[ -n "${DISCORD_WEBHOOK:-}" ]] && echo enabled || echo disabled)" echo "$ICON_GEAR Dry Run: $DRY_RUN" echo "━━━━━━━━━━━━━━━━━━━━━━━" exit 0 fi [[ "$DRY_RUN" == true ]] && warn "DRY RUN — no containers will be restarted" # ----------------------------------------------------------------------------------------------- # STATE HELPERS # ----------------------------------------------------------------------------------------------- get_strikes() { local container="$1" metric="$2" grep -E "^${container}:${metric}:" "$WATCHDOG_STATE_FILE" 2>/dev/null | cut -d':' -f3 } set_strikes() { local container="$1" metric="$2" count="$3" grep -vE "^${container}:${metric}:" "$WATCHDOG_STATE_FILE" 2>/dev/null > "${WATCHDOG_STATE_FILE}.tmp" echo "${container}:${metric}:${count}" >> "${WATCHDOG_STATE_FILE}.tmp" mv "${WATCHDOG_STATE_FILE}.tmp" "$WATCHDOG_STATE_FILE" } # ----------------------------------------------------------------------------------------------- # SKIP LIST HELPERS # Container skip list — persistent across reboots via /boot/ # Auto-clears entries when container is found running again. # ----------------------------------------------------------------------------------------------- is_in_skip_list() { local container="$1" grep -qE "^${container}$" "$SKIP_LIST_FILE" 2>/dev/null } add_to_skip_list() { local container="$1" if ! is_in_skip_list "$container"; then echo "$container" >> "$SKIP_LIST_FILE" warn "$ICON_WATCHDOG $container added to persistent skip list" notify "$container added to watchdog skip list on $(hostname) — manual check recommended" "Docker Watchdog" "warning" fi } remove_from_skip_list() { local container="$1" grep -vE "^${container}$" "$SKIP_LIST_FILE" 2>/dev/null > "${SKIP_LIST_FILE}.tmp" mv "${SKIP_LIST_FILE}.tmp" "$SKIP_LIST_FILE" success "$ICON_WATCHDOG $container recovered — removed from skip list" notify "$container recovered and removed from watchdog skip list on $(hostname)" "Docker Watchdog" "normal" } # ----------------------------------------------------------------------------------------------- # SKIP LIST AUTO-HEAL CHECK # On every run check if any skipped containers are now running. # If running remove from skip list — could have recovered after reboot or manual fix. # ----------------------------------------------------------------------------------------------- check_skip_list_recovery() { [[ ! -s "$SKIP_LIST_FILE" ]] && return info "$ICON_WATCHDOG Checking skip list for recovered containers..." while IFS= read -r container; do [[ -z "$container" ]] && continue STATUS=$(docker inspect -f '{{.State.Running}}' "$container" 2>/dev/null || echo "unknown") if [[ "$STATUS" == "true" ]]; then remove_from_skip_list "$container" else log "$container still not running — remains on skip list" fi done < "$SKIP_LIST_FILE" } # ----------------------------------------------------------------------------------------------- # DOCKER DAEMON HEALTH CHECK # Verifies Docker daemon is responding before attempting any container operations. # A hung daemon means all checks will fail — notify immediately and exit. # ----------------------------------------------------------------------------------------------- check_docker_daemon() { info "$ICON_CONTAINERS Checking Docker daemon..." if ! timeout 10 docker ps >/dev/null 2>&1; then error "Docker daemon is not responding" notify "Docker daemon unresponsive on $(hostname) — immediate attention required" "Docker Watchdog" "warning" exit 1 fi success "Docker daemon is healthy" } # ----------------------------------------------------------------------------------------------- # HELPERS # ----------------------------------------------------------------------------------------------- parse_stats() { local container="$1" docker stats --no-stream --format "{{.MemUsage}}|{{.CPUPerc}}" "$container" } convert_to_mb() { local value="$1" unit="$2" case "$unit" in KiB) awk "BEGIN {print $value / 1024}" ;; MiB) echo "$value" ;; GiB) awk "BEGIN {print $value * 1024}" ;; *) echo "UNKNOWN" ;; esac } # Restarts a container and sends notification. # Failed restarts also notify — system_watchdog.sh is the next line of defense. restart_container() { local container="$1" reason="$2" if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would restart $container ($reason)" return 0 fi info "Restarting $container ($reason)..." if docker restart "$container" >/dev/null 2>&1; then echo "$ICON_STARTED $container restarted" notify "$container restarted on $(hostname) — $reason" "Docker Watchdog" "warning" return 0 else error "Failed to restart $container" notify "Failed to restart $container on $(hostname) — $reason" "Docker Watchdog" "warning" return 1 fi } # ----------------------------------------------------------------------------------------------- # MEMORY CHECK # Restarts immediately if container exceeds hard memory limit. # Warns if approaching soft threshold. # Usage: check_memory "Emby" 16384 # ----------------------------------------------------------------------------------------------- check_memory() { local container="$1" limit_mb="$2" local stats mem_raw mem_val mem_unit mem_mb usage_pct stats=$(parse_stats "$container") mem_raw=$(echo "$stats" | awk -F'|' '{print $1}' | awk '{print $1}') mem_val=$(echo "$mem_raw" | sed -E 's/([0-9.]+).*/\1/') mem_unit=$(echo "$mem_raw" | sed -E 's/[0-9.]+([a-zA-Z]+).*/\1/') mem_mb=$(convert_to_mb "$mem_val" "$mem_unit") local mem_int mem_int=$(printf "%.0f" "$mem_mb") usage_pct=$(( (mem_int * 100) / limit_mb )) if (( usage_pct >= 100 )); then error "$ICON_MEM $container memory ${mem_int}MB (${usage_pct}%) — exceeded ${limit_mb}MB hard limit" restart_container "$container" "memory hard limit" set_strikes "$container" "CPU" 0 set_strikes "$container" "RESP" 0 elif (( usage_pct >= SOFT_MEM_THRESHOLD )); then warn "$ICON_MEM $container memory ${mem_int}MB (${usage_pct}% of ${limit_mb}MB limit)" else success "$ICON_MEM $container memory ${mem_int}MB (${usage_pct}% of ${limit_mb}MB limit)" fi } # ----------------------------------------------------------------------------------------------- # CPU CHECK # Strike system — restarts after CPU_FAIL_LIMIT consecutive over-threshold checks. # Resets strikes on recovery or restart. # Usage: check_cpu "Emby" # ----------------------------------------------------------------------------------------------- check_cpu() { local container="$1" local stats cpu_raw cpu_norm cpu_int violations stats=$(parse_stats "$container") cpu_raw=$(echo "$stats" | awk -F'|' '{print $2}' | tr -d '%') cpu_norm=$(awk "BEGIN {print $cpu_raw / $TOTAL_CORES}") cpu_int=$(printf "%.0f" "$cpu_norm") violations=$(get_strikes "$container" "CPU") [[ -z "$violations" ]] && violations=0 if (( cpu_int >= HARD_CPU_THRESHOLD )); then ((violations++)) error "$ICON_ZFS $container CPU ${cpu_int}% — hard threshold ($violations/$CPU_FAIL_LIMIT strikes)" set_strikes "$container" "CPU" "$violations" elif (( cpu_int >= SOFT_CPU_THRESHOLD )); then ((violations++)) warn "$ICON_ZFS $container CPU ${cpu_int}% — soft threshold ($violations/$CPU_FAIL_LIMIT strikes)" set_strikes "$container" "CPU" "$violations" else [[ $violations -gt 0 ]] && info "$ICON_ZFS $container CPU ${cpu_int}% — recovered, resetting strikes" [[ $violations -eq 0 ]] && success "$ICON_ZFS $container CPU ${cpu_int}%" set_strikes "$container" "CPU" 0 violations=0 fi if (( violations >= CPU_FAIL_LIMIT )); then error "$ICON_ZFS $container CPU limit hit for $CPU_FAIL_LIMIT consecutive checks" restart_container "$container" "sustained CPU abuse" set_strikes "$container" "CPU" 0 fi } # ----------------------------------------------------------------------------------------------- # RESPONSIVENESS CHECK # Strike system — restarts after RESP_FAIL_LIMIT consecutive failed HTTP checks. # Skips containers with no URL defined in WATCHDOG_CONTAINER_URLS. # Usage: check_responsiveness "Emby" # ----------------------------------------------------------------------------------------------- check_responsiveness() { local container="$1" local url="${WATCHDOG_CONTAINER_URLS[$container]:-}" [[ -z "$url" ]] && return local fails fails=$(get_strikes "$container" "RESP") [[ -z "$fails" ]] && fails=0 if ! curl -s --max-time "$CURL_TIMEOUT" "$url" >/dev/null 2>&1; then ((fails++)) warn "$ICON_PING $container unresponsive at $url ($fails/$RESP_FAIL_LIMIT strikes)" set_strikes "$container" "RESP" "$fails" else [[ $fails -gt 0 ]] && info "$ICON_PING $container responsive again — resetting strikes" [[ $fails -eq 0 ]] && success "$ICON_PING $container responsive at $url" set_strikes "$container" "RESP" 0 fails=0 fi if (( fails >= RESP_FAIL_LIMIT )); then error "$ICON_PING $container unresponsive for $RESP_FAIL_LIMIT consecutive checks" restart_container "$container" "HTTP unresponsive" set_strikes "$container" "RESP" 0 fi } # ----------------------------------------------------------------------------------------------- # REQUIRED CONTAINER CHECK # Monitors WATCHDOG_REQUIRED_CONTAINERS for unexpected stops. # Strike system — attempts restart on each strike. # After strike limit hit — adds to persistent skip list and notifies system_watchdog handoff. # Skip list auto-clears at start of each run if container has recovered. # Usage: check_required_containers # ----------------------------------------------------------------------------------------------- check_required_containers() { [[ ${#WATCHDOG_REQUIRED_CONTAINERS[@]} -eq 0 ]] && return info "$ICON_CONTAINERS Checking required containers..." for container in "${WATCHDOG_REQUIRED_CONTAINERS[@]}"; do [[ -z "$container" ]] && continue # Skip if on persistent skip list if is_in_skip_list "$container"; then warn "$ICON_NOT_RUNNING $container is on skip list — skipping until recovered" continue fi STATUS=$(docker inspect -f '{{.State.Running}}' "$container" 2>/dev/null || echo "unknown") if [[ "$STATUS" == "true" ]]; then success "$ICON_RUNNING $container is running" set_strikes "$container" "STOP" 0 continue fi if [[ "$STATUS" == "unknown" ]]; then warn "$container not found on this host — skipping" continue fi # Container is stopped — apply strike local strikes strikes=$(get_strikes "$container" "STOP") [[ -z "$strikes" ]] && strikes=0 ((strikes++)) warn "$ICON_NOT_RUNNING $container is stopped ($strikes/$SYS_WATCHDOG_STRIKE_LIMIT strikes)" set_strikes "$container" "STOP" "$strikes" # Attempt restart on each strike if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would attempt restart of $container" else if restart_container "$container" "unexpected stop"; then set_strikes "$container" "STOP" 0 else # Restart failed if (( strikes >= SYS_WATCHDOG_STRIKE_LIMIT )); then error "$container failed to restart after $SYS_WATCHDOG_STRIKE_LIMIT attempts" add_to_skip_list "$container" set_strikes "$container" "STOP" 0 notify "$container handed off to system_watchdog on $(hostname) — added to skip list" "Docker Watchdog" "warning" fi fi fi done } # ----------------------------------------------------------------------------------------------- # ━━━ $ICON_WATCHDOG Watchdog Run ━━━ # ----------------------------------------------------------------------------------------------- echo "" echo "━━━ $ICON_WATCHDOG Watchdog Run — $(date '+%Y-%m-%d %H:%M:%S') ━━━" echo "" START=$(date +%s) SKIPPED=() # Daemon check first — if daemon is down nothing else works check_docker_daemon # Auto-heal skip list before processing check_skip_list_recovery # ----------------------------------------------------------------------------------------------- # Resource monitoring — WATCHDOG_CONTAINERS # ----------------------------------------------------------------------------------------------- if [[ ${#WATCHDOG_CONTAINERS[@]} -gt 0 ]]; then echo "" echo "━━━ $ICON_MEM Resource Monitoring ━━━" for container in "${!WATCHDOG_CONTAINERS[@]}"; do echo "" info "$ICON_CONTAINERS $container" if ! docker inspect "$container" &>/dev/null; then warn "$container not found — skipping" SKIPPED+=("$container") continue fi if ! docker ps --filter "name=^/${container}$" --format "{{.Names}}" | grep -qw "$container"; then warn "$ICON_NOT_RUNNING $container is not running — skipping resource checks" SKIPPED+=("$container") continue fi check_memory "$container" "${WATCHDOG_CONTAINERS[$container]}" check_cpu "$container" check_responsiveness "$container" done fi # ----------------------------------------------------------------------------------------------- # Required container monitoring — WATCHDOG_REQUIRED_CONTAINERS # ----------------------------------------------------------------------------------------------- echo "" echo "━━━ $ICON_CONTAINERS Required Container Check ━━━" check_required_containers END=$(date +%s) # ----------------------------------------------------------------------------------------------- # ━━━ $ICON_SUMMARY Summary ━━━ # ----------------------------------------------------------------------------------------------- echo "" echo "━━━━━ $ICON_SUMMARY WATCHDOG SUMMARY ━━━━━" echo "$ICON_TIME $(date '+%Y-%m-%d %H:%M:%S')" echo "$ICON_TIME Duration: $(format_duration $((END - START)))" echo "$ICON_WATCHDOG Monitored: ${#WATCHDOG_CONTAINERS[@]} containers" echo "$ICON_CONTAINERS Required: ${#WATCHDOG_REQUIRED_CONTAINERS[@]} containers" [[ ${#SKIPPED[@]} -gt 0 ]] && echo "$ICON_NOT_RUNNING Skipped: ${SKIPPED[*]}" [[ "$DRY_RUN" == true ]] && echo "$ICON_WARN Dry Run: no actions taken" echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"