Files
Varaverk/Docker_Essentials/docker_weekly_restart.sh
T
Gmer4Lfe e8b114094a Bring script headers onto the template and close safeguard gaps
Headers claimed protections the code never had, and several destructive paths had no
guard against a collapsed config value.
2026-08-01 20:37:59 -04:00

376 lines
17 KiB
Bash
Executable File

#!/bin/bash
# ==============================================================================================
# ================================= Docker Weekly Restart ======================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Restarts configured containers every Sunday at 2:30am as proactive maintenance.
#
# Called by weekly_sync_maintenance.sh via WEEKLY_MAINTENANCE_SCRIPTS. Runs after
# the sync window has completed and already restarted its own critical containers
# (Emby, auth stack). Targets a separate set of less-critical services that benefit
# from a weekly clean start but do not need to be stopped for the sync itself.
#
# Same behavioural rules as docker_daily_restart.sh: running → restart,
# stopped → leave, missing → skip. Container state is always respected.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# Identical to docker_daily_restart.sh, against WEEKLY_RESTART_CONTAINERS:
#
# 1. Build restart order
# → build_restart_order() sorts WEEKLY_RESTART_CONTAINERS by WATCHDOG_DEPENDENCIES
#
# 2. Skip anything docker_update.sh --weekly already rebuilt this run
# → a rebuild onto a new image already restarted it moments ago
#
# 3. Inspect container state
# missing → skip, not an error
# stopped → skip, stopped state is respected
# running → restart
#
# 4. Restart with retry
# → retry_docker wraps each attempt in a timeout, up to RETRY_COUNT
#
# 5. Verify it stayed running
# → verify_running() settles for RESTART_VERIFY_WAIT then checks State.Running
# → a container that crashes immediately is marked failed and notified
#
# 6. Prune dangling images
# → restarts swap onto new images, leaving the old ones dangling
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Weekly Cadence
# Weekly restarts target services that accumulate state on a slower schedule
# than daily targets — less-critical containers that benefit from a periodic
# clean start but do not need nightly intervention. Daily restarts handle
# high-churn containers; weekly handles the longer-cycle ones.
#
# State Respect
# Running containers are restarted. Stopped containers are left stopped — they
# were intentionally halted and this script has no authority to override that
# decision. This rule is consistent across the entire ecosystem.
#
# Dependency-Safe Ordering
# Restarts follow the same dependency ordering used by docker_watchdog.sh.
# Services that other containers depend on restart first. A dependent is never
# restarted while its dependency is still coming up.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Root Enforcement
# Docker operations require root privileges.
#
# Docker Presence Check
# Verifies the docker binary exists before execution. Notifies on absence.
#
# Docker Daemon Check
# Verifies the daemon is responsive before any restart work. Every container
# would otherwise fail its inspect and be logged as an unknown-status failure,
# burying one daemon fault under a list of bogus per-container errors.
#
# Host Detection
# detect_hosts() identifies which server is running the script and aliases
# HOST*_WEEKLY_RESTART_CONTAINERS and HOST*_WATCHDOG_DEPENDENCIES to the
# correct host's values.
#
# Empty List Guard
# Exits cleanly with a pointer to the relevant conf key if
# WEEKLY_RESTART_CONTAINERS is unconfigured for this host.
#
# Dependency Ordering
# Containers restart in dependency-safe order using HOST*_WATCHDOG_DEPENDENCIES.
# CONTAINER_DELAY seconds between dependency restart and dependent restart.
#
# Restart Verification
# Container state checked after a settle period. A container that crashes
# immediately after restart is marked failed with a notification sent.
#
# Timeout Protection
# All docker commands wrapped in a 30 second timeout. A hung Docker daemon
# cannot cause this script to hang indefinitely.
#
# Stale Rebuild-List Guard
# The rebuilt-container list written by docker_update.sh --weekly is discarded
# if older than DOCKER_UPDATE_REBUILT_STALE_HOURS. A stale file would otherwise
# suppress real restarts based on an update run that never happened this week.
#
# Lock Acquisition
# acquire_lock() prevents concurrent execution.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_WEEKLY_RESTART_CONTAINERS
# Containers restarted weekly. Aliased by detect_hosts() →
# WEEKLY_RESTART_CONTAINERS
#
# HOST*_WATCHDOG_DEPENDENCIES
# Dependency ordering shared with docker_watchdog.sh. Aliased by
# detect_hosts() → WATCHDOG_DEPENDENCIES
#
# master.conf
#
# RETRY_COUNT
# Retry attempts before giving up on a container
#
# SLEEP
# Seconds between retry attempts
#
# CONTAINER_DELAY
# Seconds to wait after restarting a dependency before starting its dependents
#
# RESTART_VERIFY_WAIT
# Seconds verify_running() waits after docker restart before checking the
# container is running. Gives the process time to initialise before the
# state is sampled. (default: 3)
#
# DOCKER_UPDATE_REBUILT_WEEKLY_FILE / DOCKER_UPDATE_REBUILT_STALE_HOURS
# List of containers docker_update.sh --weekly already rebuilt onto a new image
# this run — read here so they're not restarted a second time. Discarded as
# stale (and every container restarts normally) if older than
# DOCKER_UPDATE_REBUILT_STALE_HOURS.
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# docker_weekly_restart.sh
# Restart all containers in WEEKLY_RESTART_CONTAINERS
#
# docker_weekly_restart.sh --dry-run
# Preview which containers would be restarted and which would be skipped
#
# docker_weekly_restart.sh --status
# Show configured restart list, container states, and dependency ordering
#
# docker_weekly_restart.sh --log
# Verbose per-container execution output
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
acquire_lock
if ! command -v docker &>/dev/null; then
error "Docker command not found — check PATH or Docker installation"
notify "Docker weekly restart failed — Docker not found on $(hostname)" "Docker Weekly Restart" "warning"
exit 1
fi
# detect_hosts() sets MY_ID and aliases HOST*_WEEKLY_RESTART_CONTAINERS → WEEKLY_RESTART_CONTAINERS
detect_hosts
# Without this, a hung daemon fails every container's inspect individually and the summary
# reports a list of unknown-status failures instead of the one fault that caused them.
if ! timeout "$DOCKER_TIMEOUT" docker info >/dev/null 2>&1; then
error "Docker daemon not responding — skipping weekly restart"
notify "Weekly restart skipped on $(hostname) — Docker daemon not responding" "Docker Weekly Restart" "warning"
exit 1
fi
if [[ ${#WEEKLY_RESTART_CONTAINERS[@]} -eq 0 ]]; then
warn "WEEKLY_RESTART_CONTAINERS is empty for $MY_ID — nothing to restart"
warn "Check HOST*_WEEKLY_RESTART_CONTAINERS in host*.conf"
exit 0
fi
log "$MY_ID ($LOCAL_SERVER_NAME) — ${#WEEKLY_RESTART_CONTAINERS[@]} containers configured"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_CONTAINERS Containers: ${WEEKLY_RESTART_CONTAINERS[*]}"
echo "$ICON_RETRY Retries: $RETRY_COUNT"
echo "$ICON_TIME Sleep: ${SLEEP}s between retries"
echo "$ICON_NOTIFY Notify: unRAID=${NOTIFY_UNRAID:-false} Discord=$([[ -n "${MY_DISCORD_WEBHOOK:-}" ]] && echo enabled || echo disabled)"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no containers will be restarted"
# ==============================================================================================
# ── FUNCTIONS ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# docker_cmd, retry_docker, verify_running — defined in common.sh
# build_restart_order() / check_dependency_delay() — provided by common.sh
# ==============================================================================================
# ━━━ Weekly Restart ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CONTAINERS Weekly Restart — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
log "$ICON_CONTAINERS Containers: ${WEEKLY_RESTART_CONTAINERS[*]}"
log "$ICON_RETRY Retries: $RETRY_COUNT"
log "$ICON_GEAR Config: sleep=${SLEEP}s delay=${CONTAINER_DELAY}s verify-wait=${RESTART_VERIFY_WAIT}s cmd-timeout=${DOCKER_TIMEOUT}s"
START=$(date +%s)
FAILED=()
RESTARTED=()
SKIPPED=()
ALREADY_UPDATED=()
# Build dependency-safe restart order
build_restart_order WEEKLY_RESTART_CONTAINERS
# ── Load containers docker_update.sh --weekly already rebuilt this run ──────────────────────────
# Same reasoning as docker_daily_restart.sh: a container docker_update.sh already rebuilt onto a
# new image doesn't need a plain restart right after. A file older than
# DOCKER_UPDATE_REBUILT_STALE_HOURS is discarded as untrustworthy rather than trusted, and every
# container restarts as normal.
declare -A ALREADY_REBUILT_MAP
if [[ -n "${DOCKER_UPDATE_REBUILT_WEEKLY_FILE:-}" && -f "$DOCKER_UPDATE_REBUILT_WEEKLY_FILE" ]]; then
_rebuilt_age=$(( $(date +%s) - $(stat -c %Y "$DOCKER_UPDATE_REBUILT_WEEKLY_FILE" 2>/dev/null || echo 0) ))
_rebuilt_stale_seconds=$(( ${DOCKER_UPDATE_REBUILT_STALE_HOURS:-12} * 3600 ))
if [[ "$_rebuilt_age" -gt "$_rebuilt_stale_seconds" ]]; then
warn "Rebuilt-container list is stale ($(( _rebuilt_age / 3600 ))h old) — discarding, restarting all"
rm -f "$DOCKER_UPDATE_REBUILT_WEEKLY_FILE"
else
while IFS= read -r _c; do
[[ -n "$_c" ]] && ALREADY_REBUILT_MAP["$_c"]=1
done < "$DOCKER_UPDATE_REBUILT_WEEKLY_FILE"
[[ "${#ALREADY_REBUILT_MAP[@]}" -gt 0 ]] && \
log "Already rebuilt today by docker_update.sh, skipping restart: ${!ALREADY_REBUILT_MAP[*]}"
fi
unset _rebuilt_age _rebuilt_stale_seconds
fi
LAST_RESTARTED=""
for container in "${ORDERED_RESTART[@]}"; do
[[ -z "$container" ]] && continue
c_start=$(date +%s)
c_image=$(timeout "$DOCKER_TIMEOUT" docker inspect --format '{{.Config.Image}}' "$container" 2>/dev/null || echo "unknown")
log "━━━ $ICON_CONTAINERS $container ($c_image) ━━━"
if ! timeout "$DOCKER_TIMEOUT" docker inspect "$container" &>/dev/null; then
warn "$container does not exist — skipping"
continue
fi
STATUS=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Running}}' "$container" 2>/dev/null)
case "$STATUS" in
true)
if [[ -n "${ALREADY_REBUILT_MAP[$container]:-}" ]]; then
log "$ICON_RUNNING $container already rebuilt onto new image by docker_update.sh — skipping redundant restart"
ALREADY_UPDATED+=("$container")
LAST_RESTARTED="$container" # it did restart, just moments ago via the rebuild
continue
fi
log "$ICON_RUNNING $container is running — restarting..."
# Wait if this container depends on the last one restarted
check_dependency_delay "$container" "$LAST_RESTARTED"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would restart $container"
RESTARTED+=("$container")
else
if retry_docker docker restart "$container"; then
if verify_running "$container"; then
echo "$ICON_STARTED $container restarted and running in $(format_duration $(( $(date +%s) - c_start ))) ✅"
RESTARTED+=("$container")
LAST_RESTARTED="$container"
else
error "$container restarted but crashed immediately"
notify "$container crashed after restart on $(hostname)" "Docker Weekly Restart" "warning"
FAILED+=("$container")
fi
else
error "Failed to restart $container after $RETRY_COUNT attempts"
notify "$container failed to restart on $(hostname)" "Docker Weekly Restart" "warning"
FAILED+=("$container")
fi
fi
;;
false)
# Container was stopped — leave it stopped
log "$ICON_NOT_RUNNING $container is stopped — skipping (respecting stopped state)"
SKIPPED+=("$container")
;;
*)
error "Unknown status for $container: $STATUS"
FAILED+=("$container")
;;
esac
done
END=$(date +%s)
# ==============================================================================================
# ━━━ Prune Old Images ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SYNC Pruning Dangling Images — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would prune dangling images"
PRUNED_SUMMARY="(dry run)"
else
PRUNED_OUTPUT=$(timeout "$DOCKER_TIMEOUT" docker image prune -f 2>&1)
[[ "$ENABLE_LOGGING" == "true" ]] && echo "$PRUNED_OUTPUT" | sed 's/^/ /'
PRUNED_SUMMARY=$(echo "$PRUNED_OUTPUT" | grep -E "^Total reclaimed" || echo "nothing reclaimed")
fi
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo "━━━━━ $ICON_SUMMARY WEEKLY RESTART SUMMARY ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_TIME Duration: $(format_duration $((END - START)))"
if [[ ${#RESTARTED[@]} -gt 0 ]]; then
echo "$ICON_STARTED Restarted: ${#RESTARTED[@]}"
log " Names: ${RESTARTED[*]}"
fi
[[ ${#ALREADY_UPDATED[@]} -gt 0 ]] && log "$ICON_DONE Already updated (skipped): ${ALREADY_UPDATED[*]}"
[[ ${#SKIPPED[@]} -gt 0 ]] && log "$ICON_NOT_RUNNING Skipped: ${SKIPPED[*]} (were stopped)"
[[ ${#FAILED[@]} -gt 0 ]] && echo "$ICON_ERROR Failed: ${FAILED[*]}"
echo "$ICON_SYNC Pruned: ${PRUNED_SUMMARY:-none}"
if [[ "$DRY_RUN" == true ]]; then
echo "$ICON_WARN Status: DRY RUN — no changes made"
elif [[ ${#FAILED[@]} -eq 0 ]]; then
echo "$ICON_DONE Status: $ICON_SUCCESS ALL DONE"
notify "Weekly restart complete — ${#RESTARTED[@]} restarted, ${#ALREADY_UPDATED[@]} already updated, ${#SKIPPED[@]} skipped (stopped) on $(hostname)" "Docker Weekly Restart" "normal"
else
echo "$ICON_ERROR Status: $ICON_ERROR ${#FAILED[@]} container(s) failed"
notify "Weekly restart completed with errors on $(hostname) — failed: ${FAILED[*]}" "Docker Weekly Restart" "warning"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
[[ ${#FAILED[@]} -gt 0 ]] && exit 1
exit 0