Files
Varaverk/Watchdogs/docker_watchdog.sh
T
Gmer4Lfe 35f870c650 Auto-prune skip-list/intentional-stop entries for uninstalled containers
Neither remove_from_skip_list() nor clear_intentional_stop() can ever
fire for a container that's been fully uninstalled — both only trigger
on seeing it running again. Confirmed live 2026-07-19: Healarr sat on
the skip list nagging "manual intervention needed" every single cycle
for weeks after being removed. Now checks docker inspect for each
entry every cycle and drops it automatically if the container no
longer exists at all.
2026-07-19 17:22:35 -04:00

1107 lines
58 KiB
Bash
Executable File

#!/bin/bash
# ==============================================================================================
# ================================= Docker Watchdog ============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Two-tier self-healing container monitoring system. Called by
# watchdog_orchestrator.sh via cron every 15 minutes as a single-pass run.
# Tier 1 applies specific thresholds to explicitly configured containers.
# Tier 2 scans everything else for generic health problems. Silent on clean
# cycles, loud when something needs attention.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# Tier 1 — Strict Per-Container Monitoring
# Applies only to containers explicitly configured in host*.conf.
#
# Memory hard limits — immediate restart if container exceeds MB ceiling
# Memory soft threshold — warn at SOFT_MEM_THRESHOLD % of limit (no restart)
# CPU thresholds — strike system: warn at SOFT_CPU_THRESHOLD, restart after
# CPU_FAIL_LIMIT consecutive strikes at HARD_CPU_THRESHOLD
# HTTP responsiveness — strike system: restart after RESP_FAIL_LIMIT consecutive
# failures against the configured endpoint
# Required containers — must always be running; strike system before restart;
# skip list after WATCHDOG_CONTAINER_RESTART_LIMIT in window
#
# Tier 2 — Global Health Scan
# Scans ALL running containers when WATCHDOG_SCAN_ALL=true.
# Containers in WATCHDOG_SCAN_IGNORE are excluded.
#
# Unhealthy status — Docker HEALTHCHECK unhealthy → restart
# OOM killed — kernel OOM kill detected → restart + notify
# Crash loop — RestartCount climbing → notify; above WATCHDOG_CRASH_LIMIT
# → restart → skip list if restart limit hit
# Dead containers — remove + start (dead state cannot be restarted directly)
# Unexpected exits — non-zero exit code → restart
#
# Cross-Cutting Intelligence
# Applies to both tiers on every cycle.
#
# Startup grace period — no restarts for WATCHDOG_STARTUP_GRACE seconds after boot
# Dependency ordering — dependency restarted first, dependent skipped this cycle
# Restart loop protect — skip list after WATCHDOG_CONTAINER_RESTART_LIMIT in window
# Skip list auto-clear — removed when container seen running again
# Notification batching — one summary per cycle, not one ping per event
# Parity awareness — skips restart actions during parity check
# Timeout protection — all docker commands wrapped in timeout
# Docker daemon check — each cycle begins with daemon health check; hung daemon →
# restart via rc.docker → stability_watchdog.sh escalates if needed
# RAM emergency defer — reads RW_STATE_FILE; stands down while
# resource_watchdog.sh is managing a RAM emergency
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Tiered Monitoring
# Not all containers need the same monitoring strategy. Tier 1 gives explicit
# control over the containers that matter most. Tier 2 is the catch-all that
# requires no configuration and protects everything else.
#
# Strike vs Immediate
# CPU spikes and HTTP failures are transient — brief spikes are normal during
# transcoding or library scans. Memory leaks are not transient. CPU and HTTP
# use a strike system to distinguish sustained problems from momentary ones.
# Memory triggers immediate restart because a container at its ceiling is
# actively leaking, not spiking.
#
# Loop Protection Over Persistence
# A watchdog that keeps restarting a broken container is not helpful — it risks
# making a database corruption worse. After WATCHDOG_CONTAINER_RESTART_LIMIT
# attempts the container is skip-listed and the operator is notified. Automated
# recovery stops. Human investigation begins.
#
# Dependency-Safe Ordering
# When a container and its dependency are both down, restart the dependency
# first and skip the dependent this cycle. Prevents false-alarm skip-listing
# of containers whose only failure was starting before their dependency was ready.
#
# Silent When Healthy
# Runs 96 times per day. Producing output on every clean cycle would make
# logs useless. Output only when something needs attention — alive heartbeat
# is handled by watchdog_orchestrator.sh, the caller, not this script.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Lock Acquisition
# Prevents concurrent execution via acquire_lock(). Safe at array start —
# only one watchdog instance runs at a time.
#
# Host Detection
# detect_hosts() identifies which server is running the script and aliases
# all HOST*_WATCHDOG_* arrays to the correct host's values.
#
# Startup Grace Period
# Restart actions suppressed for WATCHDOG_STARTUP_GRACE seconds after the
# watchdog starts. Checks still run and log — only restarts are suppressed.
# Prevents false-positive restarts while containers are still initialising.
#
# Restart Loop Protection
# WATCHDOG_CONTAINER_RESTART_LIMIT restarts within WATCHDOG_CONTAINER_RESTART_WINDOW
# hours triggers skip-listing and a critical notification. Skip list persists on
# /boot/config/ — survives reboots intentionally. Auto-clears when container
# is seen running again.
#
# Docker Daemon Health Check
# First operation every cycle. Daemon not responding within DOCKER_TIMEOUT →
# restart via /etc/rc.d/rc.docker → verify recovery. If still hung: log
# critical, skip cycle. stability_watchdog.sh handles further escalation.
#
# RAM Emergency Deferral
# Reads RW_STATE_FILE each cycle. If resource_watchdog.sh has set
# mem_shutdown_active=true, all restart logic defers until the flag clears.
# Stale state guard: if file is >2 hours old with flag still set,
# resource_watchdog.sh has likely stopped — watchdog resumes normal operation.
#
# Timeout Protection
# All docker commands wrapped in timeout. Daemon hangs cannot stall the
# watchdog and leave containers unmonitored between cycles.
#
# Notification Batching
# Events collected across a full cycle and sent as a single summary.
# Prevents notification floods when a shared dependency failure cascades.
#
# ==============================================================================================
# STATE FILES
# ==============================================================================================
#
# WATCHDOG_STATE_FILE — strike counts, daemon flags (STATE_DIR — survives reboots)
# DOCKER_WATCHDOG_FAILED_FILE — container skip list (STATE_DIR — survives reboots)
# DOCKER_WATCHDOG_INTENTIONAL_FILE — intentional stops list (STATE_DIR — survives reboots)
# WATCHDOG_CONTAINER_RESTART_LOG — restart history for loop detection (DATA_DIR)
# RW_STATE_FILE — read-only: resource_watchdog RAM emergency flag
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_WATCHDOG_CONTAINERS
# Memory hard limits per container. Format: "ContainerName:LimitInMB"
# Aliased by detect_hosts() → WATCHDOG_CONTAINERS
#
# HOST*_WATCHDOG_CONTAINER_URLS
# HTTP health check endpoints. Format: "ContainerName:http://host:port"
# Aliased by detect_hosts() → WATCHDOG_CONTAINER_URLS
#
# HOST*_WATCHDOG_CONTAINER_API_CHECKS
# API liveness checks — deeper than HTTP. Format: "ContainerName:URL|APIKey"
# Use endpoints that require a live DB round-trip (e.g. Emby /System/Info).
# Aliased by detect_hosts() → WATCHDOG_CONTAINER_API_CHECKS
#
# HOST*_WATCHDOG_REQUIRED_CONTAINERS
# Containers that must always be running. Aliased by detect_hosts() →
# WATCHDOG_REQUIRED_CONTAINERS
#
# HOST*_WATCHDOG_SCAN_IGNORE
# Containers excluded from Tier 2 global scan. Aliased by detect_hosts() →
# WATCHDOG_SCAN_IGNORE
#
# HOST*_WATCHDOG_DEPENDENCIES
# Dependency ordering. Format: "Dependent:dep1 dep2". Aliased by
# detect_hosts() → WATCHDOG_DEPENDENCIES
#
# master.conf
#
# WATCHDOG_STARTUP_GRACE
# Seconds before restart actions begin after watchdog starts (default: 600)
#
# SOFT_MEM_THRESHOLD
# Warn at this % of hard memory limit — no restart (default: 80)
#
# SOFT_CPU_THRESHOLD / HARD_CPU_THRESHOLD / CPU_FAIL_LIMIT
# CPU monitoring thresholds and strike limit
#
# CURL_TIMEOUT / RESP_FAIL_LIMIT
# HTTP health check timeout and consecutive failure limit
#
# WATCHDOG_SCAN_ALL
# Enable Tier 2 global health scan (default: true)
#
# WATCHDOG_RESTART_UNHEALTHY / WATCHDOG_RESTART_DEAD / WATCHDOG_RESTART_CRASHED
# Tier 2 action toggles
#
# WATCHDOG_NOTIFY_OOM / WATCHDOG_NOTIFY_CRASHLOOP / WATCHDOG_CRASH_LIMIT
# OOM and crash loop detection toggles and threshold
#
# WATCHDOG_CONTAINER_RESTART_LIMIT / WATCHDOG_CONTAINER_RESTART_WINDOW
# Restart loop protection: attempt limit and rolling window in hours
#
# WATCHDOG_REQUIRED_STRIKE_LIMIT
# Consecutive down-checks on a required container before a restart is attempted (default: 2)
#
# WATCHDOG_BATCH_NOTIFY
# Collect cycle events and send as one notification (default: true)
#
# WATCHDOG_DAEMON_TIMEOUT / WATCHDOG_DAEMON_STRIKE_LIMIT / WATCHDOG_DAEMON_RESTART_WAIT
# Docker daemon health check: command timeout, strikes before restart attempt,
# seconds to wait after restart before verifying
#
# DOCKER_WATCHDOG_INTENTIONAL_FILE
# Path to intentional stops state file (STATE_DIR). Containers in this file
# are never restarted by the watchdog, regardless of exit code.
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# docker_watchdog.sh
# Single-pass monitoring cycle — called by watchdog_orchestrator.sh every 15 min
#
# docker_watchdog.sh --dry-run
# Run a full watchdog cycle without restarting anything. Shows what would
# happen based on current container states. Use to verify configuration.
#
# docker_watchdog.sh --status
# Show strike counts, skip list contents, intentional stops, grace period status,
# RAM emergency deferral state, and last cycle timing. Then exit.
#
# docker_watchdog.sh --log
# Verbose output — full detail for every container checked and every decision.
# Use to debug why a container is or is not being restarted.
#
# docker_watchdog.sh --pause ContainerName
# Add ContainerName to the intentional stops list. Watchdog will not strike or
# restart it until it is seen running again or --resume is called. Persists across
# reboots. Use when stopping a required container for planned maintenance.
#
# docker_watchdog.sh --resume ContainerName
# Remove ContainerName from the intentional stops list. Normal watchdog monitoring
# resumes on the next cycle. The container is not started — it remains stopped
# until started manually.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
# Extract watchdog-specific flags not handled by common parse_args
WATCHDOG_PAUSE_CONTAINER=""
WATCHDOG_RESUME_CONTAINER=""
for (( _wdi=0; _wdi<${#PARSED_ARGS[@]}; _wdi++ )); do
case "${PARSED_ARGS[$_wdi]}" in
--pause) ((_wdi++)); WATCHDOG_PAUSE_CONTAINER="${PARSED_ARGS[$_wdi]:-}" ;;
--resume) ((_wdi++)); WATCHDOG_RESUME_CONTAINER="${PARSED_ARGS[$_wdi]:-}" ;;
esac
done
unset _wdi
# ==============================================================================================
# ━━━ Setup — runs once at start ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
# Skip if another instance is running — no pile-up during long operations
acquire_lock
# detect_hosts() sets MY_ID and aliases all HOST*_WATCHDOG_* arrays
detect_hosts
if ! is_docker_enabled; then
echo "Docker not enabled in Unraid settings — skipping cycle"
exit 0
fi
if ! command -v docker >/dev/null 2>&1; then
error "Docker not found — cannot start watchdog"
exit 1
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no containers will be restarted"
log "$ICON_GEAR Config: grace=${WATCHDOG_STARTUP_GRACE}s mem-soft=${SOFT_MEM_THRESHOLD}% cpu-soft=${SOFT_CPU_THRESHOLD}% cpu-hard=${HARD_CPU_THRESHOLD}% cpu-limit=${CPU_FAIL_LIMIT} http-limit=${RESP_FAIL_LIMIT} daemon-timeout=${DOCKER_TIMEOUT}s"
log "$ICON_CONTAINERS Tier1: watched=${#WATCHDOG_CONTAINERS[@]} required=${#WATCHDOG_REQUIRED_CONTAINERS[@]} urls=${#WATCHDOG_CONTAINER_URLS[@]} restart-limit=${WATCHDOG_CONTAINER_RESTART_LIMIT}/${WATCHDOG_CONTAINER_RESTART_WINDOW}h"
# Ensure state files exist
touch "$WATCHDOG_STATE_FILE" "$WATCHDOG_CONTAINER_RESTART_LOG" \
"$DOCKER_WATCHDOG_FAILED_FILE" "$DOCKER_WATCHDOG_INTENTIONAL_FILE" 2>/dev/null
# Timeout for all docker commands — configurable via WATCHDOG_DAEMON_TIMEOUT in master.conf
DOCKER_TIMEOUT="${WATCHDOG_DAEMON_TIMEOUT:-20}"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
_watched="${!WATCHDOG_CONTAINERS[*]}"; echo "$ICON_CONTAINERS Watched: ${_watched:-none}"
echo "$ICON_CONTAINERS Required: ${WATCHDOG_REQUIRED_CONTAINERS[*]:-none}"
_intentional=$(cat "$DOCKER_WATCHDOG_INTENTIONAL_FILE" 2>/dev/null | tr '\n' ' ' | xargs)
echo "$ICON_SKIP Intentional: ${_intentional:-none}"
echo "$ICON_WATCHDOG Scan all: $WATCHDOG_SCAN_ALL"
echo "$ICON_WATCHDOG Ignore: ${WATCHDOG_SCAN_IGNORE[*]:-none}"
echo "$ICON_WATCHDOG Schedule: every 15 min (cron via watchdog_orchestrator)"
echo "$ICON_WATCHDOG Startup grace: ${WATCHDOG_STARTUP_GRACE}s"
echo "$ICON_WATCHDOG Restart limit: $WATCHDOG_CONTAINER_RESTART_LIMIT in ${WATCHDOG_CONTAINER_RESTART_WINDOW}h"
echo "$ICON_WATCHDOG Batch notify: $WATCHDOG_BATCH_NOTIFY"
echo "$ICON_WATCHDOG Docker timeout: ${DOCKER_TIMEOUT}s"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
UPTIME_S=$(awk '{print int($1)}' /proc/uptime)
echo "$ICON_TIME System uptime: $(format_duration $UPTIME_S)"
[[ "$UPTIME_S" -lt "$WATCHDOG_STARTUP_GRACE" ]] && \
warn "Within startup grace period — restarts suppressed"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ── Intentional stop management — --pause / --resume ─────────────────────────────────────────
if [[ -n "$WATCHDOG_PAUSE_CONTAINER" || -n "$WATCHDOG_RESUME_CONTAINER" ]]; then
if [[ -n "$WATCHDOG_PAUSE_CONTAINER" ]]; then
if grep -q "^${WATCHDOG_PAUSE_CONTAINER}$" "$DOCKER_WATCHDOG_INTENTIONAL_FILE" 2>/dev/null; then
warn "$WATCHDOG_PAUSE_CONTAINER already in intentional stops"
else
echo "$WATCHDOG_PAUSE_CONTAINER" >> "$DOCKER_WATCHDOG_INTENTIONAL_FILE"
success "$WATCHDOG_PAUSE_CONTAINER added to intentional stops — watchdog will not restart it"
fi
fi
if [[ -n "$WATCHDOG_RESUME_CONTAINER" ]]; then
sed -i "/^${WATCHDOG_RESUME_CONTAINER}$/d" "$DOCKER_WATCHDOG_INTENTIONAL_FILE" 2>/dev/null
success "$WATCHDOG_RESUME_CONTAINER removed from intentional stops — normal monitoring resumes next cycle"
fi
exit 0
fi
# ==============================================================================================
# ── HELPER FUNCTIONS ──────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Get strike count for a container from state file — wraps common.sh's wd_state_get()
get_strikes() {
wd_state_get "$1" "$2"
}
# Set strike count for a container in state file — wraps common.sh's wd_state_set()
set_strikes() {
local container="$1" count="$2" file="$3"
wd_state_set "$container" "$count" "$file"
}
# Check if container is on the persistent skip list
is_skipped() {
grep -q "^${1}$" "$DOCKER_WATCHDOG_FAILED_FILE" 2>/dev/null
}
# Add container to persistent skip list — manual intervention required to recover
add_to_skip_list() {
local container="$1" reason="$2"
if ! is_skipped "$container"; then
echo "$container" >> "$DOCKER_WATCHDOG_FAILED_FILE"
error "$container added to skip list — $reason"
queue_notify "$container added to skip list on $(hostname)$reason — manual intervention needed" "critical"
fi
}
# Remove container from skip list — called when container is seen running again
remove_from_skip_list() {
sed -i "/^${1}$/d" "$DOCKER_WATCHDOG_FAILED_FILE" 2>/dev/null
warn "$1 recovered — removed from skip list ✅"
}
# Check if container is in the intentional stops list
is_intentional_stop() {
grep -q "^${1}$" "$DOCKER_WATCHDOG_INTENTIONAL_FILE" 2>/dev/null
}
# Remove from intentional stops — called when container is seen running again
clear_intentional_stop() {
if grep -q "^${1}$" "$DOCKER_WATCHDOG_INTENTIONAL_FILE" 2>/dev/null; then
sed -i "/^${1}$/d" "$DOCKER_WATCHDOG_INTENTIONAL_FILE" 2>/dev/null
warn "$1 running — removed from intentional stops ✅"
fi
}
# Log a restart event to the rolling restart history file
log_restart() {
local container="$1"
local now
now=$(date '+%Y-%m-%d %H:%M:%S')
local cutoff
cutoff=$(date -d "${WATCHDOG_CONTAINER_RESTART_WINDOW} hours ago" '+%Y-%m-%d %H:%M:%S')
echo "${container}|${now}" >> "$WATCHDOG_CONTAINER_RESTART_LOG"
# Trim entries older than the rolling window
local tmp="${WATCHDOG_CONTAINER_RESTART_LOG}.tmp"
awk -F'|' -v cutoff="$cutoff" '$2 >= cutoff' \
"$WATCHDOG_CONTAINER_RESTART_LOG" > "$tmp" && \
mv "$tmp" "$WATCHDOG_CONTAINER_RESTART_LOG"
}
# Get number of times a container was restarted within the rolling window
get_restart_count() {
local container="$1"
local cutoff
cutoff=$(date -d "${WATCHDOG_CONTAINER_RESTART_WINDOW} hours ago" '+%Y-%m-%d %H:%M:%S')
awk -F'|' -v c="$container" -v cutoff="$cutoff" \
'$1==c && $2>=cutoff' "$WATCHDOG_CONTAINER_RESTART_LOG" 2>/dev/null | wc -l
}
# Check if all dependencies of a container are currently running.
# Returns 0 if all deps running (or no deps), 1 if any dep is down.
dependencies_satisfied() {
local container="$1"
local deps="${WATCHDOG_DEPENDENCIES[$container]:-}"
[[ -z "$deps" ]] && return 0
for dep in $deps; do
local status
status=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Running}}' "$dep" 2>/dev/null)
if [[ "$status" != "true" ]]; then
warn "$container — dependency $dep is not running — skipping restart this cycle"
return 1
fi
done
return 0
}
# Safe restart with all guards:
# - Restart loop protection (skip list after limit)
# - Dependency check (don't restart if deps down)
# - Startup grace period (no restarts while booting)
# - Dry run support
# - Timeout protection on docker restart
#
# Returns: 0=restarted, 1=skipped, 2=added to skip list
safe_restart() {
local container="$1" reason="$2"
# Restart loop protection — skip list if over limit
local restart_count
restart_count=$(get_restart_count "$container")
if [[ "$restart_count" -ge "$WATCHDOG_CONTAINER_RESTART_LIMIT" ]]; then
add_to_skip_list "$container" \
"restarted $restart_count times in ${WATCHDOG_CONTAINER_RESTART_WINDOW}h"
return 2
fi
# Dependency check
dependencies_satisfied "$container" || return 1
# Startup grace period
local uptime_s
uptime_s=$(awk '{print int($1)}' /proc/uptime)
if [[ "$uptime_s" -lt "$WATCHDOG_STARTUP_GRACE" ]]; then
warn "$container — within startup grace period (${uptime_s}s < ${WATCHDOG_STARTUP_GRACE}s) — skipping"
return 1
fi
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would restart $container ($reason)"
return 0
fi
log "Restarting $container ($reason) [restart $((restart_count + 1))/$WATCHDOG_CONTAINER_RESTART_LIMIT in window]"
if timeout "$DOCKER_TIMEOUT" docker restart "$container" >/dev/null 2>&1; then
success "$ICON_STARTED $container restarted"
log_restart "$container"
return 0
else
error "Failed to restart $container (timeout or error)"
return 1
fi
}
# Queue a notification event for batch sending at end of cycle
queue_notify() {
local message="$1" severity="${2:-warning}"
NOTIFY_EVENTS+=("${severity}|${message}")
log "Queued: $message"
}
# Send all queued notifications — one batched summary or individual per event
flush_notify() {
[[ ${#NOTIFY_EVENTS[@]} -eq 0 ]] && return
if [[ "$WATCHDOG_BATCH_NOTIFY" == "true" ]]; then
local highest_severity="normal"
local messages=()
for event in "${NOTIFY_EVENTS[@]}"; do
local sev="${event%%|*}" msg="${event#*|}"
messages+=("$msg")
[[ "$sev" == "critical" ]] && highest_severity="warning"
[[ "$sev" == "warning" && "$highest_severity" == "normal" ]] && highest_severity="warning"
done
local summary
summary=$(printf '%s. ' "${messages[@]}")
notify "Docker Watchdog on $(hostname)${#NOTIFY_EVENTS[@]} event(s): $summary" \
"Docker Watchdog" "$highest_severity"
else
for event in "${NOTIFY_EVENTS[@]}"; do
local sev="${event%%|*}" msg="${event#*|}"
[[ "$sev" == "critical" ]] && sev="warning"
notify "$msg" "Docker Watchdog" "$sev"
done
fi
NOTIFY_EVENTS=()
}
# is_parity_running — delegates to adapter
is_parity_running() { platform_is_maintenance_running; }
# ==============================================================================================
# ── DOCKER DAEMON HEALTH CHECK ────────────────────────────────────────────────────────────────
# ==============================================================================================
# Checks Docker daemon responsiveness at the start of every cycle.
# A hung daemon makes all container operations useless — check first, short-circuit if down.
#
# Strike system:
# Each consecutive failed check adds a strike
# At WATCHDOG_DAEMON_STRIKE_LIMIT → attempt daemon restart via rc.docker
# After restart → wait WATCHDOG_DAEMON_RESTART_WAIT seconds → verify
# If verified → clear strikes, continue cycle ✅
# If still hung → notify critical, skip cycle → stability_watchdog.sh escalates from here
#
# Returns: 0 = daemon healthy | 1 = daemon down, skip this cycle
WATCHDOG_DAEMON_STRIKE_LIMIT="${WATCHDOG_DAEMON_STRIKE_LIMIT:-3}" # consecutive failed checks before restart attempt (master.conf)
WATCHDOG_DAEMON_RESTART_WAIT="${WATCHDOG_DAEMON_RESTART_WAIT:-900}" # seconds to wait after restart before verifying (master.conf)
# Loaded from state file — persists across single-pass runs
WATCHDOG_DAEMON_STRIKES=$(get_strikes "daemon_strikes" "$WATCHDOG_STATE_FILE")
WATCHDOG_DAEMON_STRIKES="${WATCHDOG_DAEMON_STRIKES//[^0-9]/}"; WATCHDOG_DAEMON_STRIKES="${WATCHDOG_DAEMON_STRIKES:-0}"
_dr_raw=$(get_strikes "daemon_restarted_flag" "$WATCHDOG_STATE_FILE")
[[ "$_dr_raw" == "true" ]] && WATCHDOG_DAEMON_RESTARTED=true || WATCHDOG_DAEMON_RESTARTED=false
# ==============================================================================================
# ── SYSTEM WATCHDOG COORDINATION ──────────────────────────────────────────────────────────────
# ==============================================================================================
# Reads system_watchdog.sh state file to check if a RAM emergency shutdown is active.
# During RAM emergency: system_watchdog.sh has stopped non-essential containers to free RAM.
# docker_watchdog.sh must not restart them — that would undo the emergency shutdown and
# prevent RAM from recovering, creating an infinite restart/shutdown loop.
#
# Returns:
# 0 = normal — run all checks
# 1 = RAM emergency active — defer container management this cycle
check_resource_watchdog_state() {
# Returns 0 = normal operation | 1 = defer, resource_watchdog RAM emergency active
local state_file="$RW_STATE_FILE"
# No state file = resource_watchdog not yet run — assume normal
[[ ! -f "$state_file" ]] && return 0
local mem_shutdown
mem_shutdown=$(grep "^mem_shutdown_active=" "$state_file" 2>/dev/null | cut -d= -f2)
[[ "$mem_shutdown" != "true" ]] && return 0
# ── Stale state guard ─────────────────────────────────────────────────────────────────────
# If mem_shutdown_active=true but state file hasn't been updated in > 2 hours,
# resource_watchdog.sh may not be running — don't defer indefinitely on stale state.
local state_mtime now age_seconds stale_limit=7200 # 2 hours
state_mtime=$(stat -c %Y "$state_file" 2>/dev/null || echo 0)
now=$(date +%s)
age_seconds=$(( now - state_mtime ))
if [[ "$age_seconds" -gt "$stale_limit" ]]; then
warn "mem_shutdown_active=true but state file is ${age_seconds}s old — may be stale"
warn "resource_watchdog.sh may not be running — resuming normal container management"
return 0 # Resume normal — don't defer indefinitely on stale state
fi
return 1 # Defer — RAM emergency confirmed and state is fresh
}
check_docker_daemon() {
# docker info is more definitive than docker ps for daemon health
if timeout "$DOCKER_TIMEOUT" docker info >/dev/null 2>&1; then
if [[ "$WATCHDOG_DAEMON_STRIKES" -gt 0 ]]; then
log "$ICON_STARTED Docker daemon recovered — clearing strikes"
queue_notify "Docker daemon recovered on $(hostname)" "normal"
WATCHDOG_DAEMON_STRIKES=0
WATCHDOG_DAEMON_RESTARTED=false
set_strikes "daemon_strikes" 0 "$WATCHDOG_STATE_FILE"
set_strikes "daemon_restarted_flag" "false" "$WATCHDOG_STATE_FILE"
fi
# Always clear confirmed-down flag when daemon is healthy — unconditional so it
# clears correctly after a reboot (strikes reset to 0 but flag persists on flash).
set_strikes "daemon_confirmed_down" "false" "$WATCHDOG_STATE_FILE"
return 0
fi
WATCHDOG_DAEMON_STRIKES=$(( WATCHDOG_DAEMON_STRIKES + 1 ))
set_strikes "daemon_strikes" "$WATCHDOG_DAEMON_STRIKES" "$WATCHDOG_STATE_FILE"
warn "$ICON_WATCHDOG Docker daemon not responding (strike $WATCHDOG_DAEMON_STRIKES/$WATCHDOG_DAEMON_STRIKE_LIMIT)"
if [[ "$WATCHDOG_DAEMON_STRIKES" -lt "$WATCHDOG_DAEMON_STRIKE_LIMIT" ]]; then
warn "Skipping monitoring cycle — waiting for daemon to recover"
return 1
fi
if [[ "$WATCHDOG_DAEMON_RESTARTED" == true ]]; then
# Restart was already attempted last cycle and daemon is still down.
# Write confirmed-down flag — stability_watchdog reads this and strikes toward reboot.
error "Docker daemon still unresponsive after restart attempt"
set_strikes "daemon_confirmed_down" "true" "$WATCHDOG_STATE_FILE"
queue_notify "Docker daemon hung on $(hostname) — restart failed — stability_watchdog escalating" "critical"
flush_notify
return 1
fi
error "Docker daemon unresponsive — attempting restart"
notify "Docker daemon hung on $(hostname) — attempting restart" "Docker Watchdog" "warning"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would restart Docker daemon via platform_restart_service docker"
return 1
fi
# timeout 180: Docker daemon stop can block for 30+ min on a busy host.
WATCHDOG_DAEMON_RESTARTED=true
set_strikes "daemon_restarted_flag" "true" "$WATCHDOG_STATE_FILE"
if timeout 180 platform_restart_service docker; then
log "Docker daemon restart issued — waiting ${WATCHDOG_DAEMON_RESTART_WAIT}s..."
sleep "$WATCHDOG_DAEMON_RESTART_WAIT"
if timeout "$DOCKER_TIMEOUT" docker info >/dev/null 2>&1; then
success "Docker daemon restarted successfully ✅"
notify "Docker daemon restarted successfully on $(hostname)" "Docker Watchdog" "normal"
WATCHDOG_DAEMON_STRIKES=0
WATCHDOG_DAEMON_RESTARTED=false
set_strikes "daemon_strikes" 0 "$WATCHDOG_STATE_FILE"
set_strikes "daemon_restarted_flag" "false" "$WATCHDOG_STATE_FILE"
set_strikes "daemon_confirmed_down" "false" "$WATCHDOG_STATE_FILE"
return 0
else
# Restart issued but daemon still down — flag for stability_watchdog on next cycle
error "Docker daemon did not recover after restart"
set_strikes "daemon_confirmed_down" "true" "$WATCHDOG_STATE_FILE"
queue_notify "Docker daemon restart failed on $(hostname) — stability_watchdog escalating" "critical"
flush_notify
return 1
fi
else
# platform_restart_service returned non-zero — flag immediately, stability_watchdog escalates
error "Failed to issue Docker daemon restart — platform_restart_service docker failed"
set_strikes "daemon_confirmed_down" "true" "$WATCHDOG_STATE_FILE"
queue_notify "Docker daemon restart command failed on $(hostname) — stability_watchdog escalating" "critical"
flush_notify
return 1
fi
}
# ==============================================================================================
# ━━━ Single-Pass Monitoring Run ━━━
# ==============================================================================================
echo "━━━ $ICON_WATCHDOG Docker Watchdog — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
CYCLE_START=$(date +%s)
# ── Per-run state ─────────────────────────────────────────────────────────────────────────
NOTIFY_EVENTS=()
T1_RESTARTS=0
T1_WARNINGS=0
T2_RESTARTS=0
T2_WARNINGS=0
# OOM handled set — tracks containers already handled for OOM this cycle
# Prevents restart loop from OOMKilled flag persisting after restart
declare -A OOM_HANDLED
# Rebuild ignore map each cycle (config may have changed)
declare -A IGNORE_MAP
for c in "${WATCHDOG_SCAN_IGNORE[@]:-}"; do
[[ -n "$c" ]] && IGNORE_MAP["$c"]=1
done
# ── Skip list and intentional stops visibility ───────────────────────────────────────────
# Prune entries for containers that no longer exist at all (uninstalled/removed) from both
# state files. Neither remove_from_skip_list() nor clear_intentional_stop() can ever fire
# for one of these — both only trigger when a container is "seen running again," which never
# happens for something that's been uninstalled — so without this an entry nags every cycle
# forever (confirmed live 2026-07-19: Healarr, uninstalled weeks earlier, still flagged every
# run). Snapshot into an array first — sed -i rewriting the same file a `while read < file`
# loop is still iterating is the classic gotcha this avoids.
for _prune_file in "$DOCKER_WATCHDOG_FAILED_FILE" "$DOCKER_WATCHDOG_INTENTIONAL_FILE"; do
[[ -s "$_prune_file" ]] || continue
mapfile -t _prune_snapshot < "$_prune_file"
for _prune_container in "${_prune_snapshot[@]}"; do
[[ -z "$_prune_container" ]] && continue
if ! timeout "$DOCKER_TIMEOUT" docker inspect "$_prune_container" &>/dev/null; then
sed -i "/^${_prune_container}$/d" "$_prune_file" 2>/dev/null
warn "$_prune_container no longer exists — removed from $(basename "$_prune_file")"
fi
done
done
unset _prune_file _prune_snapshot _prune_container
_skip_contents=$(cat "$DOCKER_WATCHDOG_FAILED_FILE" 2>/dev/null | tr '\n' ' ' | xargs)
[[ -n "$_skip_contents" ]] && warn "$ICON_SKIP Skip list active: $_skip_contents — manual intervention needed"
_intentional_contents=$(cat "$DOCKER_WATCHDOG_INTENTIONAL_FILE" 2>/dev/null | tr '\n' ' ' | xargs)
[[ -n "$_intentional_contents" ]] && warn "$ICON_SKIP Intentional stops: $_intentional_contents — watchdog will not restart these"
# ── Docker daemon health check — first check every run ──────────────────────────────────
# If daemon is hung all container operations will fail — check first, skip run if down
if ! check_docker_daemon; then
exit 0
fi
# ── Parity check — skip restarts during parity ───────────────────────────────────────────
if is_parity_running; then
echo "Parity in progress — restart actions skipped ($(date '+%H:%M:%S'))"
exit 0
fi
# ── RAM emergency check — resource_watchdog.sh managing containers ────────────────────────
# If resource_watchdog.sh has triggered a hard RAM shutdown, defer all container
# management this run to prevent undoing the emergency stop and re-pressuring RAM.
if ! check_resource_watchdog_state; then
MEM_KB=$(awk '/MemAvailable/ {print $2}' /proc/meminfo)
MEM_GB=$(( MEM_KB / 1024 / 1024 ))
warn "RAM emergency active (${MEM_GB}GB free) — resource_watchdog.sh managing containers"
warn "Deferring all container restart logic this run"
log "Waiting for RAM to recover above ${RW_RAM_RECOVER_GB:-20}GB before resuming"
exit 0
fi
# ── Startup grace period ──────────────────────────────────────────────────────────────────
UPTIME_SECONDS=$(awk '{print int($1)}' /proc/uptime)
IN_GRACE_PERIOD=false
[[ "$UPTIME_SECONDS" -lt "$WATCHDOG_STARTUP_GRACE" ]] && IN_GRACE_PERIOD=true
# ==========================================================================================
# ── TIER 1 — Strict Monitoring ────────────────────────────────────────────────────────────
# ==========================================================================================
# ── Required containers ───────────────────────────────────────────────────────────────────
# Must always be running — strike system before restart, skip list after limit
if [[ ${#WATCHDOG_REQUIRED_CONTAINERS[@]} -gt 0 ]]; then
for container in "${WATCHDOG_REQUIRED_CONTAINERS[@]}"; do
[[ -z "$container" ]] && continue
STATUS=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Running}}' \
"$container" 2>/dev/null)
if is_skipped "$container"; then
if [[ "$STATUS" == "true" ]]; then
# Container recovered — remove from skip list and clear its history
remove_from_skip_list "$container"
set_strikes "$container" 0 "$WATCHDOG_STATE_FILE"
sed -i "/^${container}|/d" "$WATCHDOG_CONTAINER_RESTART_LOG" 2>/dev/null
queue_notify "$container recovered on $(hostname) — removed from skip list" "normal"
else
warn "$container — on skip list, manual intervention needed"
fi
continue
fi
if [[ "$STATUS" == "true" ]]; then
# Running — clear strikes and any intentional-stop flag set from a prior cycle
set_strikes "$container" 0 "$WATCHDOG_STATE_FILE"
clear_intentional_stop "$container"
log "$ICON_RUNNING $container — running ✅"
else
# Check intentional stops first — explicit operator instruction beats everything
if is_intentional_stop "$container"; then
log "$container — intentionally stopped (on pause list) — skipping"
continue
fi
# Exit code 0 = cleanly stopped (docker stop, Unraid UI stop, clean shutdown).
# Don't strike or restart — operator almost certainly stopped it on purpose.
# Use --pause to make this permanent across reboots.
LAST_EXIT=$(timeout "$DOCKER_TIMEOUT" docker inspect -f \
'{{.State.ExitCode}}' "$container" 2>/dev/null)
if [[ "$LAST_EXIT" == "0" ]]; then
log "$container — stopped cleanly (exit 0) — treating as intentional; use --pause to suppress permanently"
set_strikes "$container" 0 "$WATCHDOG_STATE_FILE"
continue
fi
STRIKES=$(get_strikes "$container" "$WATCHDOG_STATE_FILE")
STRIKES=$(( STRIKES + 1 ))
set_strikes "$container" "$STRIKES" "$WATCHDOG_STATE_FILE"
warn "$container — not running, exit ${LAST_EXIT} (strike $STRIKES/${WATCHDOG_REQUIRED_STRIKE_LIMIT:-2})"
((T1_WARNINGS++))
if [[ "$STRIKES" -ge "${WATCHDOG_REQUIRED_STRIKE_LIMIT:-2}" ]]; then
result=0
safe_restart "$container" "required container down (exit ${LAST_EXIT})" || result=$?
case $result in
0) set_strikes "$container" 0 "$WATCHDOG_STATE_FILE"
((T1_RESTARTS++))
queue_notify "$container was down (exit ${LAST_EXIT}) and restarted on $(hostname)" "warning" ;;
2) : ;; # Added to skip list — already notified
*) queue_notify "$container failed to restart on $(hostname)" "warning" ;;
esac
fi
fi
done
fi
# ── Memory and CPU monitoring ─────────────────────────────────────────────────────────────
if [[ ${#WATCHDOG_CONTAINERS[@]} -gt 0 ]]; then
STATS=$(timeout "$DOCKER_TIMEOUT" docker stats --no-stream \
--format "{{.Name}}|{{.MemUsage}}|{{.CPUPerc}}" 2>/dev/null)
TOTAL_CORES=$(nproc 2>/dev/null || echo 1)
for container in "${!WATCHDOG_CONTAINERS[@]}"; do
MEM_LIMIT_MB="${WATCHDOG_CONTAINERS[$container]}"
CONTAINER_STATS=$(echo "$STATS" | grep "^${container}|" | head -1)
[[ -z "$CONTAINER_STATS" ]] && continue
# ── Memory ────────────────────────────────────────────────────────────────────────
MEM_USAGE=$(echo "$CONTAINER_STATS" | cut -d'|' -f2 | awk '{print $1}')
MEM_UNIT=$(echo "$MEM_USAGE" | grep -oE '[A-Za-z]+')
MEM_VALUE=$(echo "$MEM_USAGE" | grep -oE '[0-9.]+')
case "$MEM_UNIT" in
GiB|GB) MEM_MB=$(awk "BEGIN {printf \"%.0f\", $MEM_VALUE * 1024}") ;;
MiB|MB) MEM_MB=$(awk "BEGIN {printf \"%.0f\", $MEM_VALUE}") ;;
KiB|KB) MEM_MB=$(awk "BEGIN {printf \"%.0f\", $MEM_VALUE / 1024}") ;;
*) MEM_MB=0 ;;
esac
# Soft memory threshold — warn when approaching hard limit
SOFT_MEM_MB=$(awk "BEGIN {printf \"%.0f\", $MEM_LIMIT_MB * $SOFT_MEM_THRESHOLD / 100}")
if [[ "$MEM_MB" -ge "$MEM_LIMIT_MB" ]]; then
# Hard limit exceeded — immediate restart
error "$container — memory ${MEM_MB}MB exceeded hard limit ${MEM_LIMIT_MB}MB"
safe_restart "$container" "memory hard limit exceeded"
((T1_RESTARTS++))
queue_notify "$container exceeded memory hard limit on $(hostname) — restarted" "warning"
elif [[ "$MEM_MB" -ge "$SOFT_MEM_MB" ]]; then
# Soft threshold — warn only, no restart
warn "$container — memory ${MEM_MB}MB approaching limit (${SOFT_MEM_THRESHOLD}% of ${MEM_LIMIT_MB}MB)"
((T1_WARNINGS++))
fi
# ── CPU ───────────────────────────────────────────────────────────────────────────
CPU_RAW=$(echo "$CONTAINER_STATS" | cut -d'|' -f3 | tr -d '%')
CPU_NORM=$(awk "BEGIN {printf \"%.1f\", $CPU_RAW / $TOTAL_CORES}")
CPU_INT=$(printf "%.0f" "$CPU_NORM")
if [[ "$CPU_INT" -ge "$HARD_CPU_THRESHOLD" ]]; then
# Hard CPU threshold — strike system → restart
CPU_STRIKES=$(get_strikes "${container}_cpu" "$WATCHDOG_STATE_FILE")
CPU_STRIKES=$(( CPU_STRIKES + 1 ))
set_strikes "${container}_cpu" "$CPU_STRIKES" "$WATCHDOG_STATE_FILE"
warn "$container — CPU ${CPU_NORM}% (strike $CPU_STRIKES/$CPU_FAIL_LIMIT)"
if [[ "$CPU_STRIKES" -ge "$CPU_FAIL_LIMIT" ]]; then
safe_restart "$container" "CPU hard threshold exceeded ${CPU_NORM}%"
set_strikes "${container}_cpu" 0 "$WATCHDOG_STATE_FILE"
((T1_RESTARTS++))
queue_notify "$container CPU ${CPU_NORM}% on $(hostname) — restarted" "warning"
fi
elif [[ "$CPU_INT" -ge "$SOFT_CPU_THRESHOLD" ]]; then
# Soft CPU threshold — warn only, no restart, clear strikes
warn "$container — CPU ${CPU_NORM}% (above soft threshold ${SOFT_CPU_THRESHOLD}%)"
((T1_WARNINGS++))
set_strikes "${container}_cpu" 0 "$WATCHDOG_STATE_FILE"
else
# Normal — clear CPU strikes
set_strikes "${container}_cpu" 0 "$WATCHDOG_STATE_FILE"
log "$container — CPU ${CPU_NORM}% | MEM ${MEM_MB}MB / ${MEM_LIMIT_MB}MB ✅"
fi
done
fi
# ── HTTP responsiveness ───────────────────────────────────────────────────────────────────
if [[ ${#WATCHDOG_CONTAINER_URLS[@]} -gt 0 ]]; then
for container in "${!WATCHDOG_CONTAINER_URLS[@]}"; do
URL="${WATCHDOG_CONTAINER_URLS[$container]}"
if curl -sf --max-time "$CURL_TIMEOUT" "$URL" >/dev/null 2>&1; then
set_strikes "${container}_http" 0 "$WATCHDOG_STATE_FILE"
else
HTTP_STRIKES=$(get_strikes "${container}_http" "$WATCHDOG_STATE_FILE")
HTTP_STRIKES=$(( HTTP_STRIKES + 1 ))
set_strikes "${container}_http" "$HTTP_STRIKES" "$WATCHDOG_STATE_FILE"
warn "$container — not responding at $URL (strike $HTTP_STRIKES/$RESP_FAIL_LIMIT)"
((T1_WARNINGS++))
if [[ "$HTTP_STRIKES" -ge "$RESP_FAIL_LIMIT" ]]; then
result=0
safe_restart "$container" "HTTP unresponsive at $URL" || result=$?
if [[ $result -eq 0 ]]; then
set_strikes "${container}_http" 0 "$WATCHDOG_STATE_FILE"
((T1_RESTARTS++))
queue_notify "$container unresponsive at $URL on $(hostname) — restarted" "warning"
fi
fi
fi
done
fi
# ── API liveness checks ───────────────────────────────────────────────────────────────────
# Catches containers that serve HTTP 200 but are internally frozen (DB lock, deadlocked
# thread, etc.). Endpoint must require a live DB round-trip to respond successfully.
# Format per entry: "URL|APIKey"
if [[ ${#WATCHDOG_CONTAINER_API_CHECKS[@]} -gt 0 ]]; then
for container in "${!WATCHDOG_CONTAINER_API_CHECKS[@]}"; do
IFS='|' read -r _api_url _api_key <<< "${WATCHDOG_CONTAINER_API_CHECKS[$container]}"
# Skip entirely if key is absent or a placeholder — check is optional protection
if [[ -z "$_api_key" || "$_api_key" == "YOUR_API_KEY"* || "$_api_key" == "placeholder"* ]]; then
log "$container — API check skipped (no key configured)"
continue
fi
_api_http=$(curl -s --max-time "$CURL_TIMEOUT" \
-H "X-Emby-Token: ${_api_key}" \
-o /tmp/_varaverk_api_check \
-w "%{http_code}" \
"$_api_url" 2>/dev/null)
_resp=$(cat /tmp/_varaverk_api_check 2>/dev/null)
# 401/403 = wrong key — skip silently, don't penalise the container
if [[ "$_api_http" == "401" || "$_api_http" == "403" ]]; then
log "$container — API check skipped (HTTP $_api_http — key may be wrong or revoked)"
continue
fi
# Object-shaped responses (e.g. /System/Info) pass via ServerName/Id/Version.
# Array-shaped responses (e.g. /Users) pass on any valid array — indexing an
# array with a string key would itself error in jq, so branch on type first.
if echo "$_resp" | jq -e 'if type == "array" then true else (.ServerName // .Id // .Version) != null end' >/dev/null 2>&1; then
set_strikes "${container}_api" 0 "$WATCHDOG_STATE_FILE"
else
API_STRIKES=$(get_strikes "${container}_api" "$WATCHDOG_STATE_FILE")
API_STRIKES=$(( API_STRIKES + 1 ))
set_strikes "${container}_api" "$API_STRIKES" "$WATCHDOG_STATE_FILE"
warn "$container — API unresponsive at $_api_url (strike $API_STRIKES/$RESP_FAIL_LIMIT)"
((T1_WARNINGS++))
if [[ "$API_STRIKES" -ge "$RESP_FAIL_LIMIT" ]]; then
result=0
safe_restart "$container" "API unresponsive at $_api_url" || result=$?
if [[ $result -eq 0 ]]; then
set_strikes "${container}_api" 0 "$WATCHDOG_STATE_FILE"
((T1_RESTARTS++))
queue_notify "$container API unresponsive at $_api_url on $(hostname) — restarted" "warning"
fi
fi
fi
done
fi
# ==========================================================================================
# ── TIER 2 — Global Health Scan ───────────────────────────────────────────────────────────
# ==========================================================================================
if [[ "$WATCHDOG_SCAN_ALL" == "true" ]]; then
ALL_CONTAINERS=$(timeout "$DOCKER_TIMEOUT" docker ps --format "{{.Names}}" 2>/dev/null)
# ── Unhealthy containers ──────────────────────────────────────────────────────────────
if [[ "$WATCHDOG_RESTART_UNHEALTHY" == "true" ]]; then
UNHEALTHY=$(timeout "$DOCKER_TIMEOUT" docker ps \
--filter health=unhealthy --format "{{.Names}}" 2>/dev/null)
while IFS= read -r container; do
[[ -z "$container" ]] && continue
[[ -n "${IGNORE_MAP[$container]:-}" ]] && continue
is_skipped "$container" && continue
error "$container — Docker HEALTHCHECK unhealthy"
((T2_WARNINGS++))
result=0
safe_restart "$container" "unhealthy health status" || result=$?
if [[ $result -eq 0 ]]; then
((T2_RESTARTS++))
queue_notify "$container unhealthy on $(hostname) — restarted" "warning"
fi
done <<< "$UNHEALTHY"
fi
# ── OOM killed ────────────────────────────────────────────────────────────────────────
# OOMKilled flag persists after restart — track handled containers per-cycle
# to prevent the same container triggering a restart loop every cycle
if [[ "$WATCHDOG_NOTIFY_OOM" == "true" ]]; then
while IFS= read -r container; do
[[ -z "$container" ]] && continue
[[ -n "${IGNORE_MAP[$container]:-}" ]] && continue
is_skipped "$container" && continue
[[ -n "${OOM_HANDLED[$container]:-}" ]] && continue # already handled this cycle
OOM=$(timeout "$DOCKER_TIMEOUT" docker inspect -f \
'{{.State.OOMKilled}}' "$container" 2>/dev/null)
if [[ "$OOM" == "true" ]]; then
error "$container — OOM killed by kernel"
((T2_WARNINGS++))
OOM_HANDLED["$container"]=1
result=0
safe_restart "$container" "OOM killed" || result=$?
if [[ $result -eq 0 ]]; then
((T2_RESTARTS++))
queue_notify "$container OOM killed on $(hostname) — restarted" "warning"
fi
fi
done <<< "$ALL_CONTAINERS"
fi
# ── Crash loop detection ──────────────────────────────────────────────────────────────
# Tracks Docker's own RestartCount climbing between cycles.
# Below WATCHDOG_CRASH_LIMIT: notify only — Docker's restart policy is handling it.
# At or above WATCHDOG_CRASH_LIMIT: safe_restart() which will add to skip list
# if WATCHDOG_CONTAINER_RESTART_LIMIT is also hit — ensures eventual quarantine.
if [[ "$WATCHDOG_NOTIFY_CRASHLOOP" == "true" ]]; then
while IFS= read -r container; do
[[ -z "$container" ]] && continue
[[ -n "${IGNORE_MAP[$container]:-}" ]] && continue
is_skipped "$container" && continue
RESTART_COUNT=$(timeout "$DOCKER_TIMEOUT" docker inspect -f \
'{{.RestartCount}}' "$container" 2>/dev/null || echo 0)
EXIT_CODE=$(timeout "$DOCKER_TIMEOUT" docker inspect -f \
'{{.State.ExitCode}}' "$container" 2>/dev/null || echo 0)
PREV_COUNT=$(grep "^${container}_docker:" "$WATCHDOG_STATE_FILE" \
2>/dev/null | cut -d: -f2 || echo 0)
set_strikes "${container}_docker" "$RESTART_COUNT" "$WATCHDOG_STATE_FILE"
# Skip clean-exit containers (exit 0 = completed, not crashed)
[[ "$EXIT_CODE" == "0" ]] && continue
if [[ "$RESTART_COUNT" -gt "$PREV_COUNT" && "$RESTART_COUNT" -gt 0 ]]; then
((T2_WARNINGS++))
if [[ "$RESTART_COUNT" -ge "$WATCHDOG_CRASH_LIMIT" ]]; then
error "$container — crash loop CRITICAL: $RESTART_COUNT restarts"
# Attempt restart via safe_restart — will add to skip list if over limit
result=0
safe_restart "$container" "crash loop — $RESTART_COUNT restarts" || result=$?
case $result in
0) ((T2_RESTARTS++))
queue_notify "$container crash loop on $(hostname) — restarted" "critical" ;;
2) : ;; # Added to skip list
*) queue_notify "$container crash loop CRITICAL on $(hostname) — manual intervention needed" "critical" ;;
esac
else
warn "$container — restarted since last check (Docker count: $RESTART_COUNT)"
queue_notify "$container restarted on $(hostname) — Docker count: $RESTART_COUNT" "warning"
fi
fi
done <<< "$ALL_CONTAINERS"
fi
# ── Dead containers ───────────────────────────────────────────────────────────────────
# Routes through safe_restart() — ensures restart loop protection applies
if [[ "$WATCHDOG_RESTART_DEAD" == "true" ]]; then
DEAD=$(timeout "$DOCKER_TIMEOUT" docker ps -a \
--filter status=dead --format "{{.Names}}" 2>/dev/null)
while IFS= read -r container; do
[[ -z "$container" ]] && continue
[[ -n "${IGNORE_MAP[$container]:-}" ]] && continue
is_skipped "$container" && continue
error "$container — dead"
((T2_WARNINGS++))
if [[ "$DRY_RUN" == false ]]; then
timeout "$DOCKER_TIMEOUT" docker rm "$container" >/dev/null 2>&1
fi
result=0
safe_restart "$container" "dead container" || result=$?
if [[ $result -eq 0 ]]; then
((T2_RESTARTS++))
queue_notify "$container was dead on $(hostname) — removed and restarted" "warning"
fi
done <<< "$DEAD"
fi
# ── Unexpected exits ──────────────────────────────────────────────────────────────────
# Only non-zero exit codes — exit 0 is a clean stop, not a crash
# Skips containers already covered by WATCHDOG_REQUIRED_CONTAINERS (handled in Tier 1)
if [[ "$WATCHDOG_RESTART_CRASHED" == "true" ]]; then
CRASHED=$(timeout "$DOCKER_TIMEOUT" docker ps -a \
--filter status=exited \
--format "{{.Names}}|{{.Status}}" 2>/dev/null | \
grep -v "Exited (0)")
while IFS='|' read -r container status; do
[[ -z "$container" ]] && continue
[[ -n "${IGNORE_MAP[$container]:-}" ]] && continue
is_skipped "$container" && continue
# Skip containers already monitored by required containers (Tier 1)
already_required=false
for req in "${WATCHDOG_REQUIRED_CONTAINERS[@]:-}"; do
[[ "$container" == "$req" ]] && already_required=true && break
done
[[ "$already_required" == true ]] && continue
error "$container$status (unexpected exit)"
((T2_WARNINGS++))
result=0
safe_restart "$container" "unexpected exit: $status" || result=$?
if [[ $result -eq 0 ]]; then
((T2_RESTARTS++))
queue_notify "$container crashed on $(hostname) ($status) — restarted" "warning"
fi
done <<< "$CRASHED"
fi
fi # WATCHDOG_SCAN_ALL
# ── Send notifications ────────────────────────────────────────────────────────────────────
flush_notify
# ── Cycle summary — quiet when healthy ───────────────────────────────────────────────────
TOTAL_RESTARTS=$(( T1_RESTARTS + T2_RESTARTS ))
TOTAL_WARNINGS=$(( T1_WARNINGS + T2_WARNINGS ))
CYCLE_END=$(date +%s)
CONTAINER_COUNT=$(docker ps -q 2>/dev/null | wc -l | tr -d ' ')
if [[ "$TOTAL_RESTARTS" -gt 0 || "$TOTAL_WARNINGS" -gt 0 ]]; then
echo "$ICON_WATCHDOG T1: $T1_RESTARTS restarts / $T1_WARNINGS warnings T2: $T2_RESTARTS restarts / $T2_WARNINGS warnings"
echo "$ICON_TIME Duration: $(format_duration $(( CYCLE_END - CYCLE_START ))) Containers: $CONTAINER_COUNT"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
else
echo "All healthy ✅ — ${CONTAINER_COUNT} containers ($(date '+%H:%M:%S'))"
fi