#!/bin/bash # ============================================================================================== # ================================= Resource Manager =========================================== # ============================================================================================== # # PURPOSE # ───────────────────────────────────────────────────────────────────────────── # Pressure reduction layer — detects rising system load and reduces it before # things break. Called by watchdog_orchestrator.sh every 15 minutes as a # single-pass run. The middle layer between docker_watchdog.sh (fixes broken # containers) and stability_watchdog.sh (reboots). Does neither of those things. # # ============================================================================================== # OPERATIONAL MODEL # ============================================================================================== # # Three-Level Pressure Response # # Level 1 — SOFT (RAM < RW_RAM_SOFT_GB OR load > RW_LOAD_SOFT_MULTIPLIER × cores): # Throttle SABnzbd download speed to RW_SABNZBD_SPEED_SOFT. # Throttle qBittorrent download to RW_QBIT_DL_SOFT KB/s. # # Level 2 — MEDIUM (RAM < RW_RAM_MEDIUM_GB OR load > RW_LOAD_MEDIUM_MULTIPLIER × cores): # Further throttle SABnzbd + qBittorrent to medium limits. # docker pause RW_PAUSE_CONTAINERS — suspend without losing state, instantly reversible. # # Level 3 — HARD (RAM < RW_RAM_HARD_GB): # docker stop RW_STOP_CONTAINERS — optional/heavy services (games, LocalAI, etc.). # Write mem_shutdown_active=true → signals docker_watchdog to defer container restarts. # # Recovery # Pressure must stay below current threshold for RW_RECOVER_CYCLES consecutive # runs before restoring. De-escalates one level at a time — prevents re-triggering # immediately after recovery. Level 3 additionally requires RAM >= RW_RAM_RECOVER_GB # before containers are un-stopped. # # Coordination with docker_watchdog.sh # At level 3, writes mem_shutdown_active=true to RW_STATE_FILE. # docker_watchdog.sh reads this and defers all container restart logic. # Without this, docker_watchdog would immediately restart containers that were # just stopped to free RAM — defeating the purpose of level 3. # Cleared when pressure resolves and containers are restarted. # # ============================================================================================== # DESIGN PRINCIPLES # ============================================================================================== # # Graduated Response # Pressure is answered with the smallest effective action first — throttle, # then pause, then stop. Each level is only reached because the level below it # failed to relieve pressure. Nothing jumps straight to stopping containers. # # Reversibility First # docker pause suspends a container without losing its state and is instantly # reversible, so it is preferred at level 2. docker stop, which discards # in-memory state, is held back to level 3 and applied only to services # explicitly listed as expendable in RW_STOP_CONTAINERS. # # Hysteresis on Recovery # Restoring requires RW_RECOVER_CYCLES consecutive clear cycles and # de-escalates one level at a time. Recovering instantly on a single good # reading would flap — restore, re-trigger, restore — under sustained load. # # Cross-Watchdog Coordination # Level 3 publishes mem_shutdown_active=true so docker_watchdog.sh defers its # restart logic. Two watchdogs acting on the same containers with opposite # intent would otherwise fight: one stopping to free RAM, the other restarting # to restore health. # # ============================================================================================== # OPERATIONAL SAFEGUARDS # ============================================================================================== # # Root Required # docker pause/stop require root. # # Single Instance Lock # acquire_lock prevents concurrent runs from racing on state file writes. # # Docker Presence Check # Verifies the docker binary exists before any pressure response — every # level-2 and level-3 action depends on it. # # Host Detection # detect_hosts() aliases HOST*_RW_PAUSE_CONTAINERS, HOST*_RW_STOP_CONTAINERS # and the downloader credentials to the correct host's values. # # State File Verification # Exits if RW_STATE_FILE cannot be created. Without durable state the script # cannot track recovery cycles or know which containers it paused, and would # never restore them. # # RW_CRITICAL_CONTAINERS # Containers listed here are never paused or stopped regardless of pressure # level. Enforced by is_critical(), which gates both the pause and the stop # path — not just the configuration lists. # # RW_ENABLED Flag # Set RW_ENABLED=false to disable the entire script without removing it from # the orchestrator schedule. # # Timeout Protection # All docker commands wrapped in a 15 second timeout. Pressure response runs # during a degraded system, which is exactly when the daemon is most likely # to be slow — a hang here would stall the whole watchdog chain for that cycle. # # Downloader Availability Guards # SABnzbd and qBittorrent throttling no-ops when the service is disabled or # its URL/credentials are unset. A missing downloader never blocks the # container-level pressure response. # # Recovery Hysteresis # Restoration requires RW_RECOVER_CYCLES consecutive clear cycles and # de-escalates one level per cycle, preventing flapping under sustained load. # # Dry Run Support # --dry-run reports every throttle, pause and stop without performing any. # # ============================================================================================== # CONFIGURATION # ============================================================================================== # # master.conf # RW_ENABLED, RW_STATE_FILE # RW_RAM_SOFT_GB, RW_RAM_MEDIUM_GB, RW_RAM_HARD_GB, RW_RAM_RECOVER_GB # RW_LOAD_SOFT_MULTIPLIER, RW_LOAD_MEDIUM_MULTIPLIER # RW_RECOVER_CYCLES # RW_SABNZBD_ENABLED, RW_SABNZBD_SPEED_SOFT, RW_SABNZBD_SPEED_MEDIUM # RW_QBIT_ENABLED, RW_QBIT_DL_SOFT, RW_QBIT_DL_MEDIUM # RW_CRITICAL_CONTAINERS — never paused or stopped regardless of pressure # # host*.conf (aliased by detect_hosts()) # HOST*_RW_PAUSE_CONTAINERS — docker pause at medium pressure # HOST*_RW_STOP_CONTAINERS — docker stop at hard pressure # HOST*_SABNZBD_URL, HOST*_SABNZBD_API_KEY # HOST*_QBIT_URL, HOST*_QBIT_USERNAME, HOST*_QBIT_PASSWORD # # ============================================================================================== # STATE FILES # ============================================================================================== # # RW_STATE_FILE # Pressure level, recovery cycle count, stopped container list, and the # mem_shutdown_active coordination flag read by docker_watchdog.sh. # # ============================================================================================== # RUNTIME MODES # ============================================================================================== # # resource_watchdog.sh # Single-pass pressure check. Apply actions if threshold crossed. Silent if below. # # resource_watchdog.sh --dry-run # Show current pressure level and what would be throttled/paused/stopped. No changes. # # resource_watchdog.sh --status # Show current pressure level, active actions, recovery cycle count, stopped containers. # # resource_watchdog.sh --log # Verbose per-check output — show RAM, load, each threshold comparison, each action. # # ============================================================================================== SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" source "$SCRIPT_DIR/../load_config.sh" parse_args "$@" # ============================================================================================== # ━━━ Setup ━━━ # ============================================================================================== if [[ "$EUID" -ne 0 ]]; then error "Must be run as root" exit 1 fi if [[ "${RW_ENABLED:-true}" != "true" ]]; then echo "Resource Manager disabled (RW_ENABLED=false)" exit 0 fi acquire_lock if ! command -v docker &>/dev/null; then error "Docker command not found" exit 1 fi detect_hosts DOCKER_TIMEOUT=15 TOTAL_CORES=$(nproc) RW_LOAD_SOFT_THRESH=$(awk "BEGIN{printf \"%.0f\", $TOTAL_CORES * ${RW_LOAD_SOFT_MULTIPLIER:-2.0}}") RW_LOAD_MEDIUM_THRESH=$(awk "BEGIN{printf \"%.0f\", $TOTAL_CORES * ${RW_LOAD_MEDIUM_MULTIPLIER:-3.0}}") log "$ICON_GEAR Config: soft=RAM<${RW_RAM_SOFT_GB}GB/load≥${RW_LOAD_SOFT_THRESH} medium=RAM<${RW_RAM_MEDIUM_GB}GB/load≥${RW_LOAD_MEDIUM_THRESH} hard=RAM<${RW_RAM_HARD_GB}GB recover=RAM≥${RW_RAM_RECOVER_GB}GB cycles=${RW_RECOVER_CYCLES}" log "$ICON_CONTAINERS Pause at medium: ${RW_PAUSE_CONTAINERS[*]:-none} Stop at hard: ${RW_STOP_CONTAINERS[*]:-none}" touch "$RW_STATE_FILE" 2>/dev/null || { error "Cannot create state file: $RW_STATE_FILE" exit 1 } # ── Exit Trap — restart containers stopped this run if script crashes ────────────────────────── declare -a _RW_TRAP_STOPPED=() _rw_trap_restart_stopped() { [[ ${#_RW_TRAP_STOPPED[@]} -eq 0 ]] && return for c in "${_RW_TRAP_STOPPED[@]}"; do [[ -z "$c" ]] && continue # Bounded, because this runs from the EXIT trap. An unbounded docker call here means a # hung daemon stops the script exiting at all — it keeps its lock, and the containers # this trap exists to bring back stay down. if timeout "$DOCKER_TIMEOUT" docker inspect "$c" >/dev/null 2>&1; then warn "Exit trap: restarting $c (stopped but state not persisted)" timeout "$DOCKER_TIMEOUT" docker start "$c" >/dev/null 2>&1 || warn " Failed to restart $c" fi done } trap "_release_all_locks; _rw_trap_restart_stopped" EXIT # ============================================================================================== # ━━━ State Helpers ━━━ # ============================================================================================== # rm_state_get/set use : separator for RM-internal state # rm_state_get_eq/set_eq use = separator for docker_watchdog coordination flags # Both wrap common.sh's wd_state_get()/wd_state_set() — file is implicit here (this # script's own state file), unlike the generic helper which always takes it as an argument. rm_state_get() { wd_state_get "$1" "$RW_STATE_FILE" } rm_state_set() { local key="$1" val="$2" wd_state_set "$key" "$val" "$RW_STATE_FILE" } rm_state_get_eq() { wd_state_get "$1" "$RW_STATE_FILE" "=" } rm_state_set_eq() { local key="$1" val="$2" wd_state_set "$key" "$val" "$RW_STATE_FILE" "=" } # ============================================================================================== # ━━━ Load State ━━━ # ============================================================================================== CURRENT_LEVEL=$(rm_state_get "rm_action_level"); CURRENT_LEVEL=${CURRENT_LEVEL:-0} RECOVER_CYCLES=$(rm_state_get "rm_recover_cycles"); RECOVER_CYCLES=${RECOVER_CYCLES:-0} PAUSED_LIST=$(rm_state_get "rm_paused_containers"); PAUSED_LIST=${PAUSED_LIST:-""} STOPPED_LIST=$(rm_state_get "rm_stopped_containers"); STOPPED_LIST=${STOPPED_LIST:-""} # ============================================================================================== # ━━━ Pressure Calculation ━━━ # ============================================================================================== MEM_KB=$(awk '/MemAvailable/ {print $2}' /proc/meminfo) MEM_GB=$(( MEM_KB / 1024 / 1024 )) LOAD=$(awk '{print $1}' /proc/loadavg) LOAD_INT=$(printf "%.0f" "$LOAD") TARGET_LEVEL=0 TARGET_REASON="" if [[ "$MEM_GB" -lt "${RW_RAM_HARD_GB:-6}" ]]; then TARGET_LEVEL=3 TARGET_REASON="RAM ${MEM_GB}GB < hard threshold ${RW_RAM_HARD_GB}GB" elif [[ "$MEM_GB" -lt "${RW_RAM_MEDIUM_GB:-8}" ]] || [[ "$LOAD_INT" -ge "$RW_LOAD_MEDIUM_THRESH" ]]; then TARGET_LEVEL=2 [[ "$MEM_GB" -lt "${RW_RAM_MEDIUM_GB:-8}" ]] && TARGET_REASON="RAM ${MEM_GB}GB < medium threshold ${RW_RAM_MEDIUM_GB}GB" [[ "$LOAD_INT" -ge "$RW_LOAD_MEDIUM_THRESH" ]] && TARGET_REASON="${TARGET_REASON:+$TARGET_REASON, }load ${LOAD} >= medium threshold ${RW_LOAD_MEDIUM_THRESH}" elif [[ "$MEM_GB" -lt "${RW_RAM_SOFT_GB:-12}" ]] || [[ "$LOAD_INT" -ge "$RW_LOAD_SOFT_THRESH" ]]; then TARGET_LEVEL=1 [[ "$MEM_GB" -lt "${RW_RAM_SOFT_GB:-12}" ]] && TARGET_REASON="RAM ${MEM_GB}GB < soft threshold ${RW_RAM_SOFT_GB}GB" [[ "$LOAD_INT" -ge "$RW_LOAD_SOFT_THRESH" ]] && TARGET_REASON="${TARGET_REASON:+$TARGET_REASON, }load ${LOAD} >= soft threshold ${RW_LOAD_SOFT_THRESH}" fi LEVEL_NAMES=("normal" "soft" "medium" "hard") # ============================================================================================== # ━━━ Status ━━━ # ============================================================================================== if [[ "$SHOW_STATUS" == true ]]; then echo "" echo "━━━━━ $ICON_SUMMARY RESOURCE MANAGER STATUS ━━━━━" echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)" echo "" echo "── Current State ──" echo " Action level: $CURRENT_LEVEL (${LEVEL_NAMES[$CURRENT_LEVEL]:-unknown})" echo " Target level: $TARGET_LEVEL (${LEVEL_NAMES[$TARGET_LEVEL]:-unknown})" echo " Recover cycles: $RECOVER_CYCLES / ${RW_RECOVER_CYCLES:-3}" [[ -n "$PAUSED_LIST" ]] && echo " Paused: $PAUSED_LIST" [[ -n "$STOPPED_LIST" ]] && echo " Stopped: $STOPPED_LIST" MEM_SHUTDOWN_ACTIVE=$(rm_state_get_eq "mem_shutdown_active") [[ "$MEM_SHUTDOWN_ACTIVE" == "true" ]] && warn " docker_watchdog DEFERRED (mem_shutdown_active=true)" echo "" echo "── System Pressure ──" echo " RAM free: ${MEM_GB}GB (soft:<${RW_RAM_SOFT_GB} medium:<${RW_RAM_MEDIUM_GB} hard:<${RW_RAM_HARD_GB} recover:>=${RW_RAM_RECOVER_GB})" echo " Load avg: ${LOAD} (soft:>=${RW_LOAD_SOFT_THRESH} medium:>=${RW_LOAD_MEDIUM_THRESH} cores:${TOTAL_CORES})" echo "" echo "── Configuration ──" echo " SABnzbd throttle: ${RW_SABNZBD_ENABLED:-true} soft=${RW_SABNZBD_SPEED_SOFT} medium=${RW_SABNZBD_SPEED_MEDIUM}" echo " qBit throttle: ${RW_QBIT_ENABLED:-true} soft=${RW_QBIT_DL_SOFT}KB/s medium=${RW_QBIT_DL_MEDIUM}KB/s" echo "" echo "── Container Lists (this host) ──" echo " Pause at medium: ${RW_PAUSE_CONTAINERS[*]:-none configured}" echo " Stop at hard: ${RW_STOP_CONTAINERS[*]:-none configured}" echo " Critical (never touched): ${RW_CRITICAL_CONTAINERS[*]:-none}" echo "━━━━━━━━━━━━━━━━━━━━━━━" exit 0 fi # ============================================================================================== # ━━━ Critical Container Guard ━━━ # ============================================================================================== # Returns 0 if container is safe to pause/stop, 1 if it is critical is_critical() { local container="$1" for c in "${RW_CRITICAL_CONTAINERS[@]:-}"; do [[ "$c" == "$container" ]] && return 1 done return 0 } # ============================================================================================== # ━━━ SABnzbd API ━━━ # ============================================================================================== sabnzbd_set_speed() { local speed="$1" [[ "${RW_SABNZBD_ENABLED:-true}" != "true" ]] && return 0 [[ -z "$SABNZBD_URL" || -z "$SABNZBD_API_KEY" ]] && return 0 if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would set SABnzbd speed to $speed" return 0 fi curl -sf --max-time 10 \ "${SABNZBD_URL}/api?mode=config&name=speedlimit&value=${speed}&apikey=${SABNZBD_API_KEY}" \ >/dev/null 2>&1 && log "SABnzbd speed → $speed" || warn "SABnzbd API call failed" } # ============================================================================================== # ━━━ qBittorrent API ━━━ # ============================================================================================== QBIT_COOKIE="/tmp/rm_qbit_cookie.txt" qbit_login() { [[ "${RW_QBIT_ENABLED:-true}" != "true" ]] && return 0 [[ -z "$QBIT_URL" || -z "$QBIT_USERNAME" || -z "$QBIT_PASSWORD" ]] && return 0 curl -sf --max-time 10 -c "$QBIT_COOKIE" \ -X POST "${QBIT_URL}/api/v2/auth/login" \ -d "username=${QBIT_USERNAME}&password=${QBIT_PASSWORD}" >/dev/null 2>&1 } qbit_set_dl_limit() { local kbps="$1" # KB/s — 0 = unlimited [[ "${RW_QBIT_ENABLED:-true}" != "true" ]] && return 0 [[ -z "$QBIT_URL" ]] && return 0 if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would set qBit download limit to ${kbps}KB/s" return 0 fi local bps=$(( kbps * 1024 )) qbit_login curl -sf --max-time 10 -b "$QBIT_COOKIE" \ -X POST "${QBIT_URL}/api/v2/transfer/setDownloadLimit" \ -d "limit=${bps}" >/dev/null 2>&1 && log "qBit download limit → ${kbps}KB/s" || warn "qBit API call failed" } # ============================================================================================== # ━━━ Container Actions ━━━ # ============================================================================================== # Pause a list of containers — returns newline-separated list of actually-paused containers pause_containers() { local actually_paused=() for container in "$@"; do [[ -z "$container" ]] && continue is_critical "$container" || { log "$container — critical, skipping pause"; continue; } local status status=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Status}}' "$container" 2>/dev/null) if [[ "$status" != "running" ]]; then log "$container — not running (status: ${status:-unknown}), skipping pause" continue fi if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would docker pause $container" actually_paused+=("$container") continue fi if timeout "$DOCKER_TIMEOUT" docker pause "$container" >/dev/null 2>&1; then warn "Paused $container (medium pressure)" actually_paused+=("$container") else error "Failed to pause $container" fi done printf '%s,' "${actually_paused[@]}" | sed 's/,$//' } # Unpause a comma-separated list of containers unpause_containers() { local IFS=',' for container in $1; do [[ -z "$container" ]] && continue local status status=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Status}}' "$container" 2>/dev/null) if [[ "$status" != "paused" ]]; then log "$container — not paused (status: ${status:-unknown}), skipping unpause" continue fi if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would docker unpause $container" continue fi if timeout "$DOCKER_TIMEOUT" docker unpause "$container" >/dev/null 2>&1; then warn "Unpaused $container (pressure reduced)" else error "Failed to unpause $container" fi done } # Stop a list of containers — returns comma-separated list of actually-stopped containers # Named rw_ to avoid colliding with common.sh's stop_containers() (different signature/semantics) rw_stop_containers() { local actually_stopped=() for container in "$@"; do [[ -z "$container" ]] && continue is_critical "$container" || { log "$container — critical, skipping stop"; continue; } local status status=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Status}}' "$container" 2>/dev/null) if [[ "$status" != "running" && "$status" != "paused" ]]; then log "$container — not running (status: ${status:-unknown}), skipping stop" continue fi if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would docker stop $container" actually_stopped+=("$container") continue fi if timeout "$DOCKER_TIMEOUT" docker stop "$container" >/dev/null 2>&1; then warn "Stopped $container (hard pressure)" actually_stopped+=("$container") _RW_TRAP_STOPPED+=("$container") else error "Failed to stop $container" fi done printf '%s,' "${actually_stopped[@]}" | sed 's/,$//' } # Start a comma-separated list of containers (only those RM stopped) # Named rw_ to avoid colliding with common.sh's start_containers() (different signature/semantics) rw_start_containers() { local IFS=',' for container in $1; do [[ -z "$container" ]] && continue local status status=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Status}}' "$container" 2>/dev/null) if [[ "$status" == "running" ]]; then log "$container — already running" continue fi if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would docker start $container" continue fi if timeout "$DOCKER_TIMEOUT" docker start "$container" >/dev/null 2>&1; then warn "Started $container (pressure cleared)" else error "Failed to start $container" fi done } # ============================================================================================== # ━━━ Apply Level Actions ━━━ # ============================================================================================== apply_level_1() { log "Applying level 1 (soft) — throttling downloaders" sabnzbd_set_speed "${RW_SABNZBD_SPEED_SOFT:-50M}" qbit_set_dl_limit "${RW_QBIT_DL_SOFT:-51200}" } apply_level_2() { log "Applying level 2 (medium) — throttling + pausing background containers" sabnzbd_set_speed "${RW_SABNZBD_SPEED_MEDIUM:-10M}" qbit_set_dl_limit "${RW_QBIT_DL_MEDIUM:-10240}" if [[ ${#RW_PAUSE_CONTAINERS[@]} -gt 0 ]]; then local newly_paused newly_paused=$(pause_containers "${RW_PAUSE_CONTAINERS[@]}") # Merge with existing paused list (avoid duplicates on re-escalation) if [[ -n "$newly_paused" ]]; then if [[ -n "$PAUSED_LIST" ]]; then PAUSED_LIST="${PAUSED_LIST},${newly_paused}" else PAUSED_LIST="$newly_paused" fi fi fi } apply_level_3() { log "Applying level 3 (hard) — stopping optional containers" sabnzbd_set_speed "${RW_SABNZBD_SPEED_MEDIUM:-10M}" # already at medium from level 2 qbit_set_dl_limit "${RW_QBIT_DL_MEDIUM:-10240}" if [[ ${#RW_STOP_CONTAINERS[@]} -gt 0 ]]; then local newly_stopped newly_stopped=$(rw_stop_containers "${RW_STOP_CONTAINERS[@]}") if [[ -n "$newly_stopped" ]]; then if [[ -n "$STOPPED_LIST" ]]; then STOPPED_LIST="${STOPPED_LIST},${newly_stopped}" else STOPPED_LIST="$newly_stopped" fi fi fi # Signal docker_watchdog to defer container restarts rm_state_set_eq "mem_shutdown_active" "true" warn "mem_shutdown_active=true — docker_watchdog will defer restarts" } # ============================================================================================== # ━━━ Restore Level Actions ━━━ # ============================================================================================== restore_level_3() { echo "Restoring from level 3 — starting stopped containers" if [[ -n "$STOPPED_LIST" ]]; then rw_start_containers "$STOPPED_LIST" STOPPED_LIST="" fi rm_state_set_eq "mem_shutdown_active" "false" warn "mem_shutdown_active=false — docker_watchdog restoring normal operation" } restore_level_2() { echo "Restoring from level 2 — unpausing containers" if [[ -n "$PAUSED_LIST" ]]; then unpause_containers "$PAUSED_LIST" PAUSED_LIST="" fi } restore_level_1() { echo "Restoring from level 1 — removing downloader throttle" sabnzbd_set_speed "0" qbit_set_dl_limit 0 } # ============================================================================================== # ━━━ Pressure Decision ━━━ # ============================================================================================== echo "" echo "━━━ $ICON_GEAR Resource Manager — $MY_ID — $(date '+%H:%M:%S') ━━━" echo " RAM: ${MEM_GB}GB free Load: ${LOAD} Action: ${CURRENT_LEVEL} → ${TARGET_LEVEL} (${LEVEL_NAMES[$TARGET_LEVEL]:-unknown})" if [[ "$TARGET_LEVEL" -gt "$CURRENT_LEVEL" ]]; then # ── Escalate ──────────────────────────────────────────────────────────────────────────── warn "Pressure escalating to level $TARGET_LEVEL — $TARGET_REASON" notify "Resource Manager: pressure level $TARGET_LEVEL on $(hostname) ($MY_ID) — $TARGET_REASON" \ "Resource Manager" "warning" for (( lvl = CURRENT_LEVEL + 1; lvl <= TARGET_LEVEL; lvl++ )); do case "$lvl" in 1) apply_level_1 ;; 2) apply_level_2 ;; 3) apply_level_3 ;; esac done rm_state_set "rm_action_level" "$TARGET_LEVEL" rm_state_set "rm_recover_cycles" 0 elif [[ "$TARGET_LEVEL" -lt "$CURRENT_LEVEL" ]]; then # ── Tracking recovery ─────────────────────────────────────────────────────────────────── RECOVER_CYCLES=$(( RECOVER_CYCLES + 1 )) rm_state_set "rm_recover_cycles" "$RECOVER_CYCLES" log "Pressure at level $TARGET_LEVEL — recovery cycle $RECOVER_CYCLES/${RW_RECOVER_CYCLES:-3} before restoring level $CURRENT_LEVEL actions" if [[ "$RECOVER_CYCLES" -ge "${RW_RECOVER_CYCLES:-3}" ]]; then # Level 3 de-escalation requires RAM above recover threshold if [[ "$CURRENT_LEVEL" -ge 3 && "$MEM_GB" -lt "${RW_RAM_RECOVER_GB:-20}" ]]; then warn "Level 3 restore blocked — RAM ${MEM_GB}GB still below recover threshold ${RW_RAM_RECOVER_GB}GB" else warn "Pressure sustained below level $CURRENT_LEVEL — restoring" case "$CURRENT_LEVEL" in 3) restore_level_3 ;; 2) restore_level_2 ;; 1) restore_level_1 ;; esac NEW_LEVEL=$(( CURRENT_LEVEL - 1 )) rm_state_set "rm_action_level" "$NEW_LEVEL" rm_state_set "rm_recover_cycles" 0 rm_state_set "rm_paused_containers" "$PAUSED_LIST" rm_state_set "rm_stopped_containers" "$STOPPED_LIST" if [[ "$NEW_LEVEL" -gt 0 ]]; then warn "De-escalated to level $NEW_LEVEL (${LEVEL_NAMES[$NEW_LEVEL]}) — ${RW_RECOVER_CYCLES:-3} more cycles to fully clear" else echo "All pressure cleared — system at normal operation ✅" notify "Resource Manager: pressure resolved on $(hostname) ($MY_ID) — system back to normal" \ "Resource Manager" "normal" fi fi fi else # ── Steady state ──────────────────────────────────────────────────────────────────────── if [[ "$CURRENT_LEVEL" -gt 0 ]]; then echo "Pressure holding at level $CURRENT_LEVEL — waiting for sustained recovery" else log "System normal — RAM ${MEM_GB}GB free | load ${LOAD} | ${TOTAL_CORES} cores | SABnzbd=${RW_SABNZBD_ENABLED:-true} qBit=${RW_QBIT_ENABLED:-true} ✅" fi rm_state_set "rm_recover_cycles" 0 fi # ============================================================================================== # ━━━ Persist State ━━━ # ============================================================================================== rm_state_set "rm_paused_containers" "$PAUSED_LIST" rm_state_set "rm_stopped_containers" "$STOPPED_LIST" trap "_release_all_locks" EXIT # state persisted — disable restart trap, keep lock cleanup # Touch state file each run so docker_watchdog stale guard sees fresh mtime touch "$RW_STATE_FILE" 2>/dev/null