#!/bin/bash # ============================================================================================== # ================================= Resource Manager =========================================== # ============================================================================================== # Pressure reduction layer — detects rising system load and reduces it intelligently. # Called by watchdog_orchestrator.sh every minute — single-pass, not a continuous loop. # # ── RESPONSIBILITY ──────────────────────────────────────────────────────────────────────────── # Reduce system pressure before things break. NOT fixing broken containers (docker_watchdog) # and NOT rebooting (system_watchdog). The middle layer that keeps the system comfortable. # # "Pressure is rising — reduce load intelligently." # # ── THREE-LEVEL PRESSURE RESPONSE ───────────────────────────────────────────────────────────── # # Level 1 — SOFT (RAM < RM_RAM_SOFT_GB OR load > RM_LOAD_SOFT_MULTIPLIER × cores): # Throttle SABnzbd download speed to RM_SABNZBD_SPEED_SOFT # Throttle qBittorrent download to RM_QBIT_DL_SOFT KB/s # # Level 2 — MEDIUM (RAM < RM_RAM_MEDIUM_GB OR load > RM_LOAD_MEDIUM_MULTIPLIER × cores): # Further throttle SABnzbd + qBittorrent to medium limits # docker pause RM_PAUSE_CONTAINERS — suspend without losing state, instant reversible # # Level 3 — HARD (RAM < RM_RAM_HARD_GB): # docker stop RM_STOP_CONTAINERS — optional/heavy services (games, LocalAI, etc.) # Write mem_shutdown_active=true — signals docker_watchdog to defer container restarts # # ── RECOVERY ────────────────────────────────────────────────────────────────────────────────── # Pressure must stay below current action threshold for RM_RECOVER_CYCLES consecutive runs # before restoring. De-escalates one level at a time to avoid re-triggering immediately. # Level 3 de-escalation additionally requires RAM >= RM_RAM_RECOVER_GB before un-stopping. # # ── COORDINATION WITH DOCKER WATCHDOG ───────────────────────────────────────────────────────── # At level 3: writes mem_shutdown_active=true to RM_STATE_FILE. # docker_watchdog.sh reads this and defers all container restart logic. # Cleared when pressure fully resolves and containers are restarted. # This prevents docker_watchdog from restarting containers that RM just stopped to free RAM. # # ── NOT RESPONSIBLE FOR ─────────────────────────────────────────────────────────────────────── # Restarting broken containers — docker_watchdog.sh # Rebooting the system — system_watchdog.sh # Reacting to single data points — RM_RECOVER_CYCLES prevents flip-flopping # # ── CONFIGURATION (master.conf) ─────────────────────────────────────────────────────────────── # RM_ENABLED, RM_STATE_FILE # RM_RAM_SOFT_GB, RM_RAM_MEDIUM_GB, RM_RAM_HARD_GB, RM_RAM_RECOVER_GB # RM_LOAD_SOFT_MULTIPLIER, RM_LOAD_MEDIUM_MULTIPLIER # RM_RECOVER_CYCLES # RM_SABNZBD_ENABLED, RM_SABNZBD_SPEED_SOFT, RM_SABNZBD_SPEED_MEDIUM # RM_QBIT_ENABLED, RM_QBIT_DL_SOFT, RM_QBIT_DL_MEDIUM # RM_CRITICAL_CONTAINERS — never paused or stopped regardless of pressure # # ── CONFIGURATION (master_host*.conf) ───────────────────────────────────────────────────────── # HOST*_RM_PAUSE_CONTAINERS — docker pause at medium pressure (aliased by detect_hosts) # HOST*_RM_STOP_CONTAINERS — docker stop at hard pressure (aliased by detect_hosts) # HOST*_SABNZBD_URL, HOST*_SABNZBD_API_KEY # HOST*_QBIT_URL, HOST*_QBIT_USERNAME, HOST*_QBIT_PASSWORD # # ── USAGE ───────────────────────────────────────────────────────────────────────────────────── # resource_manager.sh — normal run (via watchdog_orchestrator.sh) # resource_manager.sh --dry-run — show what would happen without acting # resource_manager.sh --status — current pressure level and active actions # resource_manager.sh --log — verbose per-check output # ============================================================================================== SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" source "$SCRIPT_DIR/../load_config.sh" parse_args "$@" # ============================================================================================== # ━━━ Setup ━━━ # ============================================================================================== if [[ "$EUID" -ne 0 ]]; then error "Must be run as root" exit 1 fi if [[ "${RM_ENABLED:-true}" != "true" ]]; then log "Resource Manager disabled (RM_ENABLED=false)" exit 0 fi acquire_lock detect_hosts DOCKER_TIMEOUT=15 touch "$RM_STATE_FILE" 2>/dev/null || { error "Cannot create state file: $RM_STATE_FILE" exit 1 } # ============================================================================================== # ━━━ State Helpers ━━━ # ============================================================================================== # rm_state_get/set use : separator for RM-internal state # rm_state_get_eq/set_eq use = separator for docker_watchdog coordination flags rm_state_get() { grep -E "^${1}:" "$RM_STATE_FILE" 2>/dev/null | cut -d: -f2- } rm_state_set() { local key="$1" val="$2" grep -vE "^${key}:" "$RM_STATE_FILE" 2>/dev/null > "${RM_STATE_FILE}.tmp" echo "${key}:${val}" >> "${RM_STATE_FILE}.tmp" mv "${RM_STATE_FILE}.tmp" "$RM_STATE_FILE" } rm_state_get_eq() { grep -E "^${1}=" "$RM_STATE_FILE" 2>/dev/null | cut -d= -f2- } rm_state_set_eq() { local key="$1" val="$2" grep -vE "^${key}=" "$RM_STATE_FILE" 2>/dev/null > "${RM_STATE_FILE}.tmp" echo "${key}=${val}" >> "${RM_STATE_FILE}.tmp" mv "${RM_STATE_FILE}.tmp" "$RM_STATE_FILE" } # ============================================================================================== # ━━━ Load State ━━━ # ============================================================================================== CURRENT_LEVEL=$(rm_state_get "rm_action_level"); CURRENT_LEVEL=${CURRENT_LEVEL:-0} RECOVER_CYCLES=$(rm_state_get "rm_recover_cycles"); RECOVER_CYCLES=${RECOVER_CYCLES:-0} PAUSED_LIST=$(rm_state_get "rm_paused_containers"); PAUSED_LIST=${PAUSED_LIST:-""} STOPPED_LIST=$(rm_state_get "rm_stopped_containers"); STOPPED_LIST=${STOPPED_LIST:-""} # ============================================================================================== # ━━━ Pressure Calculation ━━━ # ============================================================================================== TOTAL_CORES=$(nproc) MEM_KB=$(awk '/MemAvailable/ {print $2}' /proc/meminfo) MEM_GB=$(( MEM_KB / 1024 / 1024 )) LOAD=$(awk '{print $1}' /proc/loadavg) LOAD_INT=$(printf "%.0f" "$LOAD") RM_LOAD_SOFT_THRESH=$(awk "BEGIN{printf \"%.0f\", $TOTAL_CORES * ${RM_LOAD_SOFT_MULTIPLIER:-2.0}}") RM_LOAD_MEDIUM_THRESH=$(awk "BEGIN{printf \"%.0f\", $TOTAL_CORES * ${RM_LOAD_MEDIUM_MULTIPLIER:-3.0}}") TARGET_LEVEL=0 TARGET_REASON="" if [[ "$MEM_GB" -lt "${RM_RAM_HARD_GB:-6}" ]]; then TARGET_LEVEL=3 TARGET_REASON="RAM ${MEM_GB}GB < hard threshold ${RM_RAM_HARD_GB}GB" elif [[ "$MEM_GB" -lt "${RM_RAM_MEDIUM_GB:-8}" ]] || [[ "$LOAD_INT" -ge "$RM_LOAD_MEDIUM_THRESH" ]]; then TARGET_LEVEL=2 [[ "$MEM_GB" -lt "${RM_RAM_MEDIUM_GB:-8}" ]] && TARGET_REASON="RAM ${MEM_GB}GB < medium threshold ${RM_RAM_MEDIUM_GB}GB" [[ "$LOAD_INT" -ge "$RM_LOAD_MEDIUM_THRESH" ]] && TARGET_REASON="${TARGET_REASON:+$TARGET_REASON, }load ${LOAD} >= medium threshold ${RM_LOAD_MEDIUM_THRESH}" elif [[ "$MEM_GB" -lt "${RM_RAM_SOFT_GB:-12}" ]] || [[ "$LOAD_INT" -ge "$RM_LOAD_SOFT_THRESH" ]]; then TARGET_LEVEL=1 [[ "$MEM_GB" -lt "${RM_RAM_SOFT_GB:-12}" ]] && TARGET_REASON="RAM ${MEM_GB}GB < soft threshold ${RM_RAM_SOFT_GB}GB" [[ "$LOAD_INT" -ge "$RM_LOAD_SOFT_THRESH" ]] && TARGET_REASON="${TARGET_REASON:+$TARGET_REASON, }load ${LOAD} >= soft threshold ${RM_LOAD_SOFT_THRESH}" fi LEVEL_NAMES=("normal" "soft" "medium" "hard") # ============================================================================================== # ━━━ Status ━━━ # ============================================================================================== if [[ "$SHOW_STATUS" == true ]]; then echo "" echo "━━━━━ $ICON_SUMMARY RESOURCE MANAGER STATUS ━━━━━" echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)" echo "" echo "── Current State ──" echo " Action level: $CURRENT_LEVEL (${LEVEL_NAMES[$CURRENT_LEVEL]:-unknown})" echo " Target level: $TARGET_LEVEL (${LEVEL_NAMES[$TARGET_LEVEL]:-unknown})" echo " Recover cycles: $RECOVER_CYCLES / ${RM_RECOVER_CYCLES:-3}" [[ -n "$PAUSED_LIST" ]] && echo " Paused: $PAUSED_LIST" [[ -n "$STOPPED_LIST" ]] && echo " Stopped: $STOPPED_LIST" MEM_SHUTDOWN_ACTIVE=$(rm_state_get_eq "mem_shutdown_active") [[ "$MEM_SHUTDOWN_ACTIVE" == "true" ]] && warn " docker_watchdog DEFERRED (mem_shutdown_active=true)" echo "" echo "── System Pressure ──" echo " RAM free: ${MEM_GB}GB (soft:<${RM_RAM_SOFT_GB} medium:<${RM_RAM_MEDIUM_GB} hard:<${RM_RAM_HARD_GB} recover:>=${RM_RAM_RECOVER_GB})" echo " Load avg: ${LOAD} (soft:>=${RM_LOAD_SOFT_THRESH} medium:>=${RM_LOAD_MEDIUM_THRESH} cores:${TOTAL_CORES})" echo "" echo "── Configuration ──" echo " SABnzbd throttle: ${RM_SABNZBD_ENABLED:-true} soft=${RM_SABNZBD_SPEED_SOFT} medium=${RM_SABNZBD_SPEED_MEDIUM}" echo " qBit throttle: ${RM_QBIT_ENABLED:-true} soft=${RM_QBIT_DL_SOFT}KB/s medium=${RM_QBIT_DL_MEDIUM}KB/s" echo "" echo "── Container Lists (this host) ──" echo " Pause at medium: ${RM_PAUSE_CONTAINERS[*]:-none configured}" echo " Stop at hard: ${RM_STOP_CONTAINERS[*]:-none configured}" echo " Critical (never touched): ${RM_CRITICAL_CONTAINERS[*]:-none}" echo "━━━━━━━━━━━━━━━━━━━━━━━" exit 0 fi # ============================================================================================== # ━━━ Critical Container Guard ━━━ # ============================================================================================== # Returns 0 if container is safe to pause/stop, 1 if it is critical is_critical() { local container="$1" for c in "${RM_CRITICAL_CONTAINERS[@]:-}"; do [[ "$c" == "$container" ]] && return 1 done return 0 } # ============================================================================================== # ━━━ SABnzbd API ━━━ # ============================================================================================== sabnzbd_set_speed() { local speed="$1" [[ "${RM_SABNZBD_ENABLED:-true}" != "true" ]] && return 0 [[ -z "$SABNZBD_URL" || -z "$SABNZBD_API_KEY" ]] && return 0 if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would set SABnzbd speed to $speed" return 0 fi curl -sf --max-time 10 \ "${SABNZBD_URL}/api?mode=config&name=speedlimit&value=${speed}&apikey=${SABNZBD_API_KEY}" \ >/dev/null 2>&1 && log "SABnzbd speed → $speed" || warn "SABnzbd API call failed" } # ============================================================================================== # ━━━ qBittorrent API ━━━ # ============================================================================================== QBIT_COOKIE="/tmp/rm_qbit_cookie.txt" qbit_login() { [[ "${RM_QBIT_ENABLED:-true}" != "true" ]] && return 0 [[ -z "$QBIT_URL" || -z "$QBIT_USERNAME" || -z "$QBIT_PASSWORD" ]] && return 0 curl -sf --max-time 10 -c "$QBIT_COOKIE" \ -X POST "${QBIT_URL}/api/v2/auth/login" \ -d "username=${QBIT_USERNAME}&password=${QBIT_PASSWORD}" >/dev/null 2>&1 } qbit_set_dl_limit() { local kbps="$1" # KB/s — 0 = unlimited [[ "${RM_QBIT_ENABLED:-true}" != "true" ]] && return 0 [[ -z "$QBIT_URL" ]] && return 0 if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would set qBit download limit to ${kbps}KB/s" return 0 fi local bps=$(( kbps * 1024 )) qbit_login curl -sf --max-time 10 -b "$QBIT_COOKIE" \ -X POST "${QBIT_URL}/api/v2/transfer/setDownloadLimit" \ -d "limit=${bps}" >/dev/null 2>&1 && log "qBit download limit → ${kbps}KB/s" || warn "qBit API call failed" } # ============================================================================================== # ━━━ Container Actions ━━━ # ============================================================================================== # Pause a list of containers — returns newline-separated list of actually-paused containers pause_containers() { local actually_paused=() for container in "$@"; do [[ -z "$container" ]] && continue is_critical "$container" || { log "$container — critical, skipping pause"; continue; } local status status=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Status}}' "$container" 2>/dev/null) if [[ "$status" != "running" ]]; then log "$container — not running (status: ${status:-unknown}), skipping pause" continue fi if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would docker pause $container" actually_paused+=("$container") continue fi if timeout "$DOCKER_TIMEOUT" docker pause "$container" >/dev/null 2>&1; then warn "Paused $container (medium pressure)" actually_paused+=("$container") else error "Failed to pause $container" fi done printf '%s,' "${actually_paused[@]}" | sed 's/,$//' } # Unpause a comma-separated list of containers unpause_containers() { local IFS=',' for container in $1; do [[ -z "$container" ]] && continue local status status=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Status}}' "$container" 2>/dev/null) if [[ "$status" != "paused" ]]; then log "$container — not paused (status: ${status:-unknown}), skipping unpause" continue fi if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would docker unpause $container" continue fi if timeout "$DOCKER_TIMEOUT" docker unpause "$container" >/dev/null 2>&1; then warn "Unpaused $container (pressure reduced)" else error "Failed to unpause $container" fi done } # Stop a list of containers — returns comma-separated list of actually-stopped containers stop_containers() { local actually_stopped=() for container in "$@"; do [[ -z "$container" ]] && continue is_critical "$container" || { log "$container — critical, skipping stop"; continue; } local status status=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Status}}' "$container" 2>/dev/null) if [[ "$status" != "running" && "$status" != "paused" ]]; then log "$container — not running (status: ${status:-unknown}), skipping stop" continue fi if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would docker stop $container" actually_stopped+=("$container") continue fi if timeout "$DOCKER_TIMEOUT" docker stop "$container" >/dev/null 2>&1; then warn "Stopped $container (hard pressure)" actually_stopped+=("$container") else error "Failed to stop $container" fi done printf '%s,' "${actually_stopped[@]}" | sed 's/,$//' } # Start a comma-separated list of containers (only those RM stopped) start_containers() { local IFS=',' for container in $1; do [[ -z "$container" ]] && continue local status status=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Status}}' "$container" 2>/dev/null) if [[ "$status" == "running" ]]; then log "$container — already running" continue fi if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would docker start $container" continue fi if timeout "$DOCKER_TIMEOUT" docker start "$container" >/dev/null 2>&1; then warn "Started $container (pressure cleared)" else error "Failed to start $container" fi done } # ============================================================================================== # ━━━ Apply Level Actions ━━━ # ============================================================================================== apply_level_1() { log "Applying level 1 (soft) — throttling downloaders" sabnzbd_set_speed "${RM_SABNZBD_SPEED_SOFT:-50M}" qbit_set_dl_limit "${RM_QBIT_DL_SOFT:-51200}" } apply_level_2() { log "Applying level 2 (medium) — throttling + pausing background containers" sabnzbd_set_speed "${RM_SABNZBD_SPEED_MEDIUM:-10M}" qbit_set_dl_limit "${RM_QBIT_DL_MEDIUM:-10240}" if [[ ${#RM_PAUSE_CONTAINERS[@]} -gt 0 ]]; then local newly_paused newly_paused=$(pause_containers "${RM_PAUSE_CONTAINERS[@]}") # Merge with existing paused list (avoid duplicates on re-escalation) if [[ -n "$newly_paused" ]]; then if [[ -n "$PAUSED_LIST" ]]; then PAUSED_LIST="${PAUSED_LIST},${newly_paused}" else PAUSED_LIST="$newly_paused" fi fi fi } apply_level_3() { log "Applying level 3 (hard) — stopping optional containers" sabnzbd_set_speed "${RM_SABNZBD_SPEED_MEDIUM:-10M}" # already at medium from level 2 qbit_set_dl_limit "${RM_QBIT_DL_MEDIUM:-10240}" if [[ ${#RM_STOP_CONTAINERS[@]} -gt 0 ]]; then local newly_stopped newly_stopped=$(stop_containers "${RM_STOP_CONTAINERS[@]}") if [[ -n "$newly_stopped" ]]; then if [[ -n "$STOPPED_LIST" ]]; then STOPPED_LIST="${STOPPED_LIST},${newly_stopped}" else STOPPED_LIST="$newly_stopped" fi fi fi # Signal docker_watchdog to defer container restarts rm_state_set_eq "mem_shutdown_active" "true" warn "mem_shutdown_active=true — docker_watchdog will defer restarts" } # ============================================================================================== # ━━━ Restore Level Actions ━━━ # ============================================================================================== restore_level_3() { log "Restoring from level 3 — starting stopped containers" if [[ -n "$STOPPED_LIST" ]]; then start_containers "$STOPPED_LIST" STOPPED_LIST="" fi rm_state_set_eq "mem_shutdown_active" "false" warn "mem_shutdown_active=false — docker_watchdog restoring normal operation" } restore_level_2() { log "Restoring from level 2 — unpausing containers" if [[ -n "$PAUSED_LIST" ]]; then unpause_containers "$PAUSED_LIST" PAUSED_LIST="" fi } restore_level_1() { log "Restoring from level 1 — removing downloader throttle" sabnzbd_set_speed "0" qbit_set_dl_limit 0 } # ============================================================================================== # ━━━ Pressure Decision ━━━ # ============================================================================================== echo "" echo "━━━ $ICON_GEAR Resource Manager — $MY_ID — $(date '+%H:%M:%S') ━━━" echo " RAM: ${MEM_GB}GB free Load: ${LOAD} Action: ${CURRENT_LEVEL} → ${TARGET_LEVEL} (${LEVEL_NAMES[$TARGET_LEVEL]:-unknown})" if [[ "$TARGET_LEVEL" -gt "$CURRENT_LEVEL" ]]; then # ── Escalate ──────────────────────────────────────────────────────────────────────────── warn "Pressure escalating to level $TARGET_LEVEL — $TARGET_REASON" notify "Resource Manager: pressure level $TARGET_LEVEL on $(hostname) ($MY_ID) — $TARGET_REASON" \ "Resource Manager" "warning" for (( lvl = CURRENT_LEVEL + 1; lvl <= TARGET_LEVEL; lvl++ )); do case "$lvl" in 1) apply_level_1 ;; 2) apply_level_2 ;; 3) apply_level_3 ;; esac done rm_state_set "rm_action_level" "$TARGET_LEVEL" rm_state_set "rm_recover_cycles" 0 elif [[ "$TARGET_LEVEL" -lt "$CURRENT_LEVEL" ]]; then # ── Tracking recovery ─────────────────────────────────────────────────────────────────── RECOVER_CYCLES=$(( RECOVER_CYCLES + 1 )) rm_state_set "rm_recover_cycles" "$RECOVER_CYCLES" log "Pressure at level $TARGET_LEVEL — recovery cycle $RECOVER_CYCLES/${RM_RECOVER_CYCLES:-3} before restoring level $CURRENT_LEVEL actions" if [[ "$RECOVER_CYCLES" -ge "${RM_RECOVER_CYCLES:-3}" ]]; then # Level 3 de-escalation requires RAM above recover threshold if [[ "$CURRENT_LEVEL" -ge 3 && "$MEM_GB" -lt "${RM_RAM_RECOVER_GB:-20}" ]]; then warn "Level 3 restore blocked — RAM ${MEM_GB}GB still below recover threshold ${RM_RAM_RECOVER_GB}GB" else warn "Pressure sustained below level $CURRENT_LEVEL — restoring" case "$CURRENT_LEVEL" in 3) restore_level_3 ;; 2) restore_level_2 ;; 1) restore_level_1 ;; esac NEW_LEVEL=$(( CURRENT_LEVEL - 1 )) rm_state_set "rm_action_level" "$NEW_LEVEL" rm_state_set "rm_recover_cycles" 0 rm_state_set "rm_paused_containers" "$PAUSED_LIST" rm_state_set "rm_stopped_containers" "$STOPPED_LIST" if [[ "$NEW_LEVEL" -gt 0 ]]; then warn "De-escalated to level $NEW_LEVEL (${LEVEL_NAMES[$NEW_LEVEL]}) — ${RM_RECOVER_CYCLES:-3} more cycles to fully clear" else log "All pressure cleared — system at normal operation ✅" notify "Resource Manager: pressure resolved on $(hostname) ($MY_ID) — system back to normal" \ "Resource Manager" "normal" fi fi fi else # ── Steady state ──────────────────────────────────────────────────────────────────────── if [[ "$CURRENT_LEVEL" -gt 0 ]]; then log "Pressure holding at level $CURRENT_LEVEL — waiting for sustained recovery" else log "System at normal pressure ✅" fi rm_state_set "rm_recover_cycles" 0 fi # ============================================================================================== # ━━━ Persist State ━━━ # ============================================================================================== rm_state_set "rm_paused_containers" "$PAUSED_LIST" rm_state_set "rm_stopped_containers" "$STOPPED_LIST" # Touch state file each run so docker_watchdog stale guard sees fresh mtime touch "$RM_STATE_FILE" 2>/dev/null