1241 lines
54 KiB
Bash
1241 lines
54 KiB
Bash
#!/bin/bash
|
||
# -----------------------------------------------------------------------------------------------
|
||
# ----------------- UNRAID OPS COMMON LIBRARY (STABLE FRAMEWORK v1) ----------------------------
|
||
# -----------------------------------------------------------------------------------------------
|
||
# Version: 2.9
|
||
# -----------------------------------------------------------------------------------------------
|
||
# Changelog:
|
||
# v1.0 — Initial stable framework
|
||
# v1.1 — format_duration moved here from daily_sync_maintenance.sh for shared use
|
||
# SSH_KEY collision resolved — gitea key renamed GITEA_SSH_KEY in Master.conf
|
||
# Version and changelog tracking added
|
||
# v1.2 — Consistent function header comment blocks across all functions
|
||
# check_connectivity added as standalone function
|
||
# check_connectivity friendlier error output with tailscale hint
|
||
# v1.3 — check_remote_rootfs added — aborts if remote rootfs exceeds ROOTFS_WARN threshold
|
||
# check_remote_share added — aborts if target directory is missing or empty on remote
|
||
# Both protect against rsync running when remote array is down or drives are missing
|
||
# v1.4 — check_remote_disks added — verifies all physical disks backing a share are mounted
|
||
# Discovers disk layout automatically at runtime, no configuration required
|
||
# Aborts if any single disk backing the share is offline or unmounted
|
||
# v1.5 — Full icon set expanded — each operation and state has its own distinct icon
|
||
# All function output updated to use correct icon per context
|
||
# Icons grouped and commented by category for clarity
|
||
# v1.6 — ICON_CONTAINERS added — 📦 anchors all container sections for visual consistency
|
||
# ICON_NOT_RUNNING changed to ⭕ — distinct from ICON_STOPPED 🔴
|
||
# Section dividers updated from --- to ━━━ for cleaner log readability
|
||
# Summary passed/failed lines use ICON_SUCCESS and ICON_ERROR consistently
|
||
# v1.7 — ICON_MOVER added for mover operations
|
||
# ICON_CONTAINERS replaces ICON_DOCKER for docker/container operations
|
||
# validate_int added — reusable integer validation for any script
|
||
# v1.8 — ICON_PHP added for PHP-FPM operations
|
||
# v1.9 — ICON_REBOOT added for server reboot operations
|
||
# v2.0 — ICON_PLUGIN added for User Scripts plugin operations
|
||
# v2.1 — ICON_ZFS and ICON_MEM added for ZFS and memory diagnostics
|
||
# Diagnostics icon group added to icon block
|
||
# v2.2 — ICON_WATCHDOG added for Docker watchdog monitoring operations
|
||
# v2.3 — ICON_NOTIFY added for notification operations
|
||
# notify() added — shared notification function supporting unRAID native and Discord
|
||
# NOTIFY_UNRAID and DISCORD_WEBHOOK configured in Master.conf
|
||
# v2.4 — ICON_CLEAN, ICON_TRASH added for media cleaner operations
|
||
# ICON_PERMS, ICON_UNLOCKED added for media permissions operations
|
||
# ICON_REBOOT_SMART added for smart conditional reboot
|
||
# v2.5 — ICON_RAM added for ramdisk operations
|
||
# ICON_LINK added for symlink state and management
|
||
# Transcode scripts group added to ecosystem
|
||
# v2.6 — ICON_FAILOVER added for failover operations
|
||
# check_local_array added — verifies local /mnt/user is mounted and healthy
|
||
# check_remote_array added — verifies remote /mnt/user is mounted and healthy
|
||
# check_remote_docker added — verifies remote Docker daemon is responding
|
||
# ping_remote added — non-fatal ping returning status for failover use
|
||
# ping_internet added — non-fatal external ping for failover use
|
||
# v2.7 — ICON_WEBGUI added for WebGUI watchdog operations
|
||
# ICON_DOCKER_NET added for Docker network connect operations
|
||
# v2.8 — ICON_CERT added for SSL certificate monitoring operations
|
||
# v2.9 — ICON_MONITOR added for monitoring section headers
|
||
# v3.0 — Script locking system added — prevents concurrent execution conflicts
|
||
# acquire_lock() — create PID lock file, register EXIT trap
|
||
# release_lock() — remove lock file on exit
|
||
# acquire_rsync_lock() — per-profile lock + global concurrent limit
|
||
# release_rsync_lock() — decrement global counter, remove profile lock
|
||
# check_api() — pre-flight API reachability check
|
||
# ICON_LOCK added — 🔏 script instance lock acquired/released
|
||
# ICON_SMART added for drive SMART health operations
|
||
# ICON_BANDWIDTH added for bandwidth tracking operations
|
||
# ICON_DIGEST added for health digest operations
|
||
# ICON_EMBY added for Emby session reporting
|
||
# ICON_VERIFY added for backup verification operations
|
||
# Monitor/ folder added to ecosystem
|
||
# -----------------------------------------------------------------------------------------------
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# ICONS
|
||
# Each icon has one job — do not reuse across different contexts.
|
||
# Adding a new icon: add it to the appropriate group below with a comment describing its job.
|
||
# -----------------------------------------------------------------------------------------------
|
||
|
||
# System / Host
|
||
ICON_HOST="🖥️" # host detection
|
||
ICON_NET="🌐" # network / IP resolution
|
||
ICON_PING="📡" # connectivity check
|
||
ICON_GEAR="⚙️" # setup section header / profile load
|
||
|
||
# Health Checks
|
||
ICON_DISK="💾" # disk checks
|
||
ICON_HEALTH="🩺" # rootfs / share health checks
|
||
ICON_SHIELD="🛡️" # pre-flight section header
|
||
|
||
# Containers
|
||
ICON_CONTAINERS="📦" # container section anchor
|
||
ICON_STOP="⛔" # stop command being issued
|
||
ICON_STOPPED="🔴" # container confirmed stopped
|
||
ICON_START="▶️" # start command being issued
|
||
ICON_STARTED="💚" # container confirmed started
|
||
ICON_RUNNING="🟢" # container already running when checked
|
||
ICON_SKIP="⏭️" # skipping — already running healthy instance
|
||
ICON_PROTECTED="🔰" # file/item is protected — never delete
|
||
ICON_NOT_RUNNING="⭕" # container already stopped when checked
|
||
|
||
# Transfer
|
||
ICON_SYNC="🔄" # transfer section header
|
||
ICON_RUN="🚀" # sync starting / rsync attempt
|
||
ICON_RETRY="🔁" # retry attempt
|
||
ICON_DONE="🏁" # transfer complete
|
||
|
||
# Summary
|
||
ICON_SUMMARY="📋" # summary section header
|
||
ICON_TIME="⏱️" # duration line
|
||
|
||
# System Operations
|
||
ICON_MOVER="🔃" # mover operations
|
||
ICON_REBOOT="⚡" # scheduled server reboot
|
||
ICON_REBOOT_SMART="🚨" # smart conditional reboot triggered
|
||
ICON_PLUGIN="🧩" # user scripts plugin operations
|
||
ICON_PHP="👥" # PHP-FPM operations
|
||
ICON_WEBGUI="💻" # WebGUI / nginx / emhttp operations
|
||
|
||
# Diagnostics
|
||
ICON_ZFS="📊" # ZFS ARC statistics
|
||
ICON_MEM="🧠" # memory status
|
||
ICON_WATCHDOG="🐾" # docker watchdog monitoring operations
|
||
|
||
# Media Operations
|
||
ICON_CLEAN="🧹" # media cleaner operations
|
||
ICON_TRASH="🗑️" # files being deleted
|
||
ICON_PERMS="🔐" # permissions operation / section header
|
||
ICON_UNLOCKED="🔓" # permissions successfully applied to a share
|
||
ICON_LOCK="🔏" # script instance lock — acquired/released
|
||
|
||
# Transcode Operations
|
||
ICON_RAM="💨" # ramdisk operations — fast ephemeral storage
|
||
ICON_LINK="🔗" # symlink state and management
|
||
|
||
# Failover Operations
|
||
ICON_FAILOVER="🔀" # failover state changes and operations
|
||
|
||
# Docker Network Operations
|
||
ICON_DOCKER_NET="🔌" # Docker network connect operations
|
||
|
||
# Security / Certificate Operations
|
||
ICON_CERT="🔒" # SSL certificate monitoring
|
||
|
||
# Monitor Operations
|
||
ICON_MONITOR="📈" # monitoring section headers and general monitoring
|
||
ICON_SMART="🔧" # drive SMART health attribute monitoring
|
||
ICON_BANDWIDTH="📶" # bandwidth usage tracking and reporting
|
||
ICON_DIGEST="📰" # health digest — aggregated system summary
|
||
ICON_EMBY="🎬" # Emby media server session reporting
|
||
ICON_VERIFY="✔️" # backup verification — checksum comparison
|
||
|
||
# Notifications
|
||
ICON_NOTIFY="🔔" # notification operations
|
||
|
||
# Output
|
||
ICON_INFO="ℹ️"
|
||
ICON_WARN="⚠️"
|
||
ICON_ERROR="❌"
|
||
ICON_SUCCESS="✅"
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# OUTPUT HELPERS
|
||
# Standardised output functions used across all scripts.
|
||
# log() is gated by ENABLE_LOGGING — set in Master.conf or via --log flag.
|
||
# -----------------------------------------------------------------------------------------------
|
||
info() { echo "$ICON_INFO [INFO] $*"; }
|
||
warn() { echo "$ICON_WARN [WARN] $*"; }
|
||
error() { echo "$ICON_ERROR [ERROR] $*"; }
|
||
success() { echo "$ICON_SUCCESS [OK] $*"; }
|
||
|
||
log() {
|
||
[[ "${ENABLE_LOGGING:-false}" == true ]] && echo "[LOG] $*"
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# NOTIFICATION
|
||
# Sends a notification via unRAID native system and/or Discord webhook.
|
||
# Both channels are optional and independently controlled via Master.conf.
|
||
# Severity levels: normal, warning, alert
|
||
# Usage: notify "message" "subject" "severity"
|
||
# -----------------------------------------------------------------------------------------------
|
||
notify() {
|
||
local message="$1"
|
||
local subject="${2:-unRAID Notification}"
|
||
local severity="${3:-normal}"
|
||
|
||
log "$ICON_NOTIFY Sending notification: $subject — $message"
|
||
|
||
if [[ "${NOTIFY_UNRAID:-false}" == true ]]; then
|
||
local notify_script="/usr/local/emhttp/plugins/dynamix/scripts/notify"
|
||
if [[ -x "$notify_script" ]]; then
|
||
"$notify_script" -s "$subject" -d "$message" -i "$severity" 2>/dev/null
|
||
log "$ICON_NOTIFY unRAID notification sent"
|
||
else
|
||
log "$ICON_NOTIFY unRAID notify script not found — skipping"
|
||
fi
|
||
fi
|
||
|
||
if [[ -n "${DISCORD_WEBHOOK:-}" ]]; then
|
||
local payload
|
||
payload=$(printf '{"content": "%s — **%s**\\n%s"}' \
|
||
"$ICON_NOTIFY" "$subject" "$message")
|
||
if curl -s -H "Content-Type: application/json" \
|
||
-d "$payload" "$DISCORD_WEBHOOK" >/dev/null 2>&1; then
|
||
log "$ICON_NOTIFY Discord notification sent"
|
||
else
|
||
warn "Discord notification failed — check DISCORD_WEBHOOK in Master.conf"
|
||
fi
|
||
fi
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# DURATION FORMATTER
|
||
# Converts raw seconds into a human readable string — e.g. 10m53s or 47s
|
||
# -----------------------------------------------------------------------------------------------
|
||
format_duration() {
|
||
local secs=$1
|
||
local mins=$((secs / 60))
|
||
local rem=$((secs % 60))
|
||
[[ $mins -gt 0 ]] && echo "${mins}m${rem}s" || echo "${rem}s"
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# ARG PARSER
|
||
# Processes all flags and key=value pairs passed to any script.
|
||
# Supported flags: --dry-run, --log, --no-log, --status, --help
|
||
# Supported key=value: LOG=true/false, or any declared variable e.g. BW_LIMIT=5000
|
||
# Unparsed positional args returned in PARSED_ARGS array.
|
||
# -----------------------------------------------------------------------------------------------
|
||
parse_args() {
|
||
ENABLE_LOGGING=${ENABLE_LOGGING:-false}
|
||
DRY_RUN=${DRY_RUN:-false}
|
||
SHOW_STATUS=${SHOW_STATUS:-false}
|
||
CLEAN_ARGS=()
|
||
|
||
for ARG in "$@"; do
|
||
if [[ "$ARG" == *=* ]]; then
|
||
VAR="${ARG%%=*}"
|
||
VAL="${ARG#*=}"
|
||
case "$VAR" in
|
||
LOG)
|
||
[[ "$VAL" == "true" ]] && ENABLE_LOGGING=true
|
||
[[ "$VAL" == "false" ]] && ENABLE_LOGGING=false
|
||
;;
|
||
*)
|
||
if declare -p "$VAR" &>/dev/null; then
|
||
printf -v "$VAR" '%s' "$VAL"
|
||
log "Set $VAR=$VAL"
|
||
else
|
||
warn "Unknown variable: $VAR"
|
||
fi
|
||
;;
|
||
esac
|
||
else
|
||
case "$ARG" in
|
||
--dry-run|-n) DRY_RUN=true ;;
|
||
--log) ENABLE_LOGGING=true ;;
|
||
--no-log) ENABLE_LOGGING=false ;;
|
||
--status|--summary) SHOW_STATUS=true ;;
|
||
--help|-h)
|
||
echo "Usage: script [--dry-run] [--log] [--status]"
|
||
exit 0
|
||
;;
|
||
*) CLEAN_ARGS+=("$ARG") ;;
|
||
esac
|
||
fi
|
||
done
|
||
|
||
PARSED_ARGS=("${CLEAN_ARGS[@]}")
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# VALIDATION HELPERS
|
||
# -----------------------------------------------------------------------------------------------
|
||
|
||
# Exits with error if a required variable is empty or unset.
|
||
# Usage: require_var VAR_NAME
|
||
require_var() {
|
||
[[ -z "${!1:-}" ]] && error "Missing required: $1" && exit 1
|
||
}
|
||
|
||
# Exits with error if a variable is not a valid positive integer.
|
||
# Usage: validate_int VAR_NAME "$VAR_VALUE"
|
||
validate_int() {
|
||
local name="$1" value="$2"
|
||
if [[ -z "$value" ]]; then
|
||
error "$name is not set — check Master.conf"
|
||
exit 1
|
||
fi
|
||
if ! [[ "$value" =~ ^[0-9]+$ ]]; then
|
||
error "$name must be a positive integer — got: '$value'"
|
||
exit 1
|
||
fi
|
||
log "$name validated: $value"
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# HOST DETECTION
|
||
# Determines which server is local and which is remote by comparing hostname against
|
||
# HOST1 and HOST2 in Master.conf. Sets LOCAL_SERVER_NAME, REMOTE_SERVER_NAME and SSH_KEY.
|
||
# Both servers run identical scripts — this is what makes them bidirectional.
|
||
# -----------------------------------------------------------------------------------------------
|
||
detect_hosts() {
|
||
LOCAL_HOSTNAME="$(hostname)"
|
||
|
||
if [[ "$LOCAL_HOSTNAME" == "$HOST1" ]]; then
|
||
LOCAL_SERVER_NAME="$HOST1"
|
||
REMOTE_SERVER_NAME="$HOST2"
|
||
elif [[ "$LOCAL_HOSTNAME" == "$HOST2" ]]; then
|
||
LOCAL_SERVER_NAME="$HOST2"
|
||
REMOTE_SERVER_NAME="$HOST1"
|
||
else
|
||
error "Unknown host: $LOCAL_HOSTNAME"
|
||
exit 1
|
||
fi
|
||
|
||
declare -A SSH_KEYS
|
||
SSH_KEYS["$HOST1|$HOST2"]="$HOST1_SSH_KEY"
|
||
SSH_KEYS["$HOST2|$HOST1"]="$HOST2_SSH_KEY"
|
||
SSH_KEY="${SSH_KEYS[$LOCAL_SERVER_NAME|$REMOTE_SERVER_NAME]}"
|
||
|
||
[[ -z "$SSH_KEY" ]] && error "Missing SSH key mapping" && exit 1
|
||
info "$ICON_HOST Host: $LOCAL_SERVER_NAME → $REMOTE_SERVER_NAME"
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# REMOTE IP RESOLUTION
|
||
# Resolves the Tailscale IPv4 address of the remote server.
|
||
# Sets REMOTE_SERVER used by all subsequent SSH and rsync calls.
|
||
# Exits if resolution fails — Tailscale may be down or peer offline.
|
||
# -----------------------------------------------------------------------------------------------
|
||
resolve_remote_ip() {
|
||
log "Resolving remote IP for $REMOTE_SERVER_NAME..."
|
||
REMOTE_SERVER=$(tailscale ip -4 "$REMOTE_SERVER_NAME" 2>/dev/null)
|
||
[[ -z "$REMOTE_SERVER" ]] && error "Failed to resolve Tailscale IP for $REMOTE_SERVER_NAME" && exit 1
|
||
info "$ICON_NET Remote IP: $REMOTE_SERVER"
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# CONNECTIVITY CHECK — fatal, used by rsync scripts
|
||
# Pings remote and exits if unreachable.
|
||
# For failover use ping_remote() which returns status without exiting.
|
||
# -----------------------------------------------------------------------------------------------
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# check_rsync_enabled — two-tier rsync gate check
|
||
# Tier 1: RSYNC_ENABLED — global, overrides everything
|
||
# Tier 2: orchestrator-specific flag passed as argument
|
||
#
|
||
# Usage:
|
||
# check_rsync_enabled "DAILY" ← checks RSYNC_ENABLED + DAILY_RSYNC_ENABLED
|
||
# check_rsync_enabled "WEEKLY" ← checks RSYNC_ENABLED + WEEKLY_RSYNC_ENABLED
|
||
# check_rsync_enabled "FAILOVER" ← checks RSYNC_ENABLED + FAILOVER_RSYNC_ENABLED
|
||
# check_rsync_enabled ← checks RSYNC_ENABLED only (rsync.sh direct call)
|
||
#
|
||
# Returns:
|
||
# 0 = rsync enabled, proceed
|
||
# 1 = rsync disabled, skip cleanly
|
||
# -----------------------------------------------------------------------------------------------
|
||
check_rsync_enabled() {
|
||
local orchestrator="${1:-}"
|
||
|
||
# Tier 1 — global gate
|
||
if [[ "${RSYNC_ENABLED:-true}" == false ]]; then
|
||
warn "RSYNC_ENABLED=false — rsync globally disabled, skipping all syncs"
|
||
return 1
|
||
fi
|
||
|
||
# Tier 2 — per-orchestrator
|
||
if [[ -n "$orchestrator" ]]; then
|
||
local var_name="${orchestrator}_RSYNC_ENABLED"
|
||
local var_value="${!var_name:-true}"
|
||
if [[ "$var_value" == false ]]; then
|
||
info "${var_name}=false — rsync disabled for $orchestrator orchestrator"
|
||
info "All other $orchestrator jobs will still run normally"
|
||
return 1
|
||
fi
|
||
fi
|
||
|
||
return 0
|
||
}
|
||
|
||
check_connectivity() {
|
||
log "Checking connectivity to $REMOTE_SERVER..."
|
||
if ! ping -c1 -W3 "$REMOTE_SERVER" &>/dev/null; then
|
||
error "$ICON_PING Remote $REMOTE_SERVER ($REMOTE_SERVER_NAME) is unreachable"
|
||
info "Hint: tailscale status | grep $REMOTE_SERVER_NAME"
|
||
exit 1
|
||
fi
|
||
info "$ICON_PING $REMOTE_SERVER_NAME is reachable"
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# PING REMOTE — non-fatal, used by failover
|
||
# Returns 0 if reachable, 1 if not — does NOT exit.
|
||
# -----------------------------------------------------------------------------------------------
|
||
ping_remote() {
|
||
ping -c2 -W3 "$REMOTE_SERVER" &>/dev/null
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# PING INTERNET — non-fatal external connectivity check
|
||
# Returns 0 if internet reachable, 1 if not — does NOT exit.
|
||
# -----------------------------------------------------------------------------------------------
|
||
ping_internet() {
|
||
ping -c2 -W3 "${EXTERNAL_IP:-8.8.8.8}" &>/dev/null
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# LOCAL ARRAY CHECK — non-fatal, returns status
|
||
# Verifies local /mnt/user is mounted and has shares.
|
||
# Used by failover before starting remote containers locally.
|
||
# Returns 0 if healthy, 1 if not.
|
||
# -----------------------------------------------------------------------------------------------
|
||
check_local_array() {
|
||
log "Checking local array..."
|
||
if ! mountpoint -q /mnt/user 2>/dev/null; then
|
||
error "$ICON_DISK Local array is not started — /mnt/user is not mounted"
|
||
return 1
|
||
fi
|
||
local file_count
|
||
file_count=$(ls /mnt/user 2>/dev/null | wc -l)
|
||
if [[ "$file_count" -eq 0 ]]; then
|
||
error "$ICON_DISK Local array appears empty — shares may not be available"
|
||
return 1
|
||
fi
|
||
log "Local array is healthy"
|
||
return 0
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# REMOTE ARRAY CHECK — non-fatal, returns status
|
||
# Verifies remote /mnt/user is mounted via SSH.
|
||
# Used before handback rsync — syncing to remote with no array fills rootfs.
|
||
# Returns 0 if healthy, 1 if not.
|
||
# -----------------------------------------------------------------------------------------------
|
||
check_remote_array() {
|
||
log "Checking remote array on $REMOTE_SERVER_NAME..."
|
||
local result
|
||
result=$(ssh -i "$SSH_KEY" -o ConnectTimeout=10 root@"$REMOTE_SERVER" \
|
||
"mountpoint -q /mnt/user && echo yes || echo no" 2>/dev/null)
|
||
if [[ "$result" != "yes" ]]; then
|
||
error "$ICON_DISK Remote array not started on $REMOTE_SERVER_NAME"
|
||
return 1
|
||
fi
|
||
log "Remote array is healthy"
|
||
return 0
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# REMOTE DOCKER CHECK — non-fatal, returns status
|
||
# Verifies remote Docker daemon is responding before container operations.
|
||
# A hung daemon means start/stop commands will silently fail.
|
||
# Returns 0 if healthy, 1 if not.
|
||
# -----------------------------------------------------------------------------------------------
|
||
check_remote_docker() {
|
||
log "Checking remote Docker daemon on $REMOTE_SERVER_NAME..."
|
||
if ! ssh -i "$SSH_KEY" -o ConnectTimeout=10 root@"$REMOTE_SERVER" \
|
||
"timeout 10 docker ps" >/dev/null 2>&1; then
|
||
error "$ICON_CONTAINERS Remote Docker daemon not responding on $REMOTE_SERVER_NAME"
|
||
return 1
|
||
fi
|
||
log "Remote Docker daemon is healthy"
|
||
return 0
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# REMOTE ROOTFS SPACE CHECK — fatal
|
||
# Aborts if remote rootfs exceeds ROOTFS_WARN threshold.
|
||
# -----------------------------------------------------------------------------------------------
|
||
check_remote_rootfs() {
|
||
log "Checking remote rootfs usage..."
|
||
REMOTE_USAGE=$(ssh -i "$SSH_KEY" root@"$REMOTE_SERVER" \
|
||
"df / --output=pcent | tail -1 | tr -d ' %'" 2>/dev/null)
|
||
if [[ -z "$REMOTE_USAGE" ]]; then
|
||
error "Could not retrieve rootfs usage from $REMOTE_SERVER_NAME"
|
||
exit 1
|
||
fi
|
||
if [[ "$REMOTE_USAGE" -ge "${ROOTFS_WARN:-75}" ]]; then
|
||
error "$ICON_HEALTH Remote rootfs ${REMOTE_USAGE}% — threshold ${ROOTFS_WARN:-75}%"
|
||
exit 1
|
||
fi
|
||
info "$ICON_HEALTH Remote rootfs: ${REMOTE_USAGE}% (threshold: ${ROOTFS_WARN:-75}%)"
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# REMOTE SHARE VALIDATION — fatal
|
||
# Verifies target directory exists and is not empty on remote.
|
||
# Usage: check_remote_share "/mnt/user/Movies"
|
||
# -----------------------------------------------------------------------------------------------
|
||
check_remote_share() {
|
||
local dir="$1"
|
||
log "Checking remote share: $dir..."
|
||
|
||
SHARE_EXISTS=$(ssh -i "$SSH_KEY" root@"$REMOTE_SERVER" \
|
||
"[[ -d '$dir' ]] && echo yes || echo no" 2>/dev/null)
|
||
if [[ "$SHARE_EXISTS" != "yes" ]]; then
|
||
error "$ICON_HEALTH Remote share does not exist: $dir"
|
||
exit 1
|
||
fi
|
||
|
||
SHARE_EMPTY=$(ssh -i "$SSH_KEY" root@"$REMOTE_SERVER" \
|
||
"[[ -z \"\$(ls -A '$dir' 2>/dev/null)\" ]] && echo yes || echo no" 2>/dev/null)
|
||
if [[ "$SHARE_EMPTY" == "yes" ]]; then
|
||
warn "$ICON_HEALTH Remote share exists but is empty: $dir — aborting to protect data"
|
||
exit 1
|
||
fi
|
||
|
||
info "$ICON_HEALTH Remote share verified: $dir"
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# get_unraid_temp_thresholds — read disk temp thresholds from unRAID's dynamix.cfg
|
||
# Sets globals: UNRAID_DISK_HOT UNRAID_DISK_MAX UNRAID_SSD_HOT UNRAID_SSD_MAX
|
||
# Falls back to safe defaults if file not found
|
||
# -----------------------------------------------------------------------------------------------
|
||
get_unraid_temp_thresholds() {
|
||
local cfg="/boot/config/plugins/dynamix/dynamix.cfg"
|
||
if [[ -f "$cfg" ]]; then
|
||
UNRAID_DISK_HOT=$(grep '^hot=' "$cfg" 2>/dev/null | cut -d= -f2 | tr -d '"')
|
||
UNRAID_DISK_MAX=$(grep '^max=' "$cfg" 2>/dev/null | cut -d= -f2 | tr -d '"')
|
||
UNRAID_SSD_HOT=$(grep '^hotssd=' "$cfg" 2>/dev/null | cut -d= -f2 | tr -d '"')
|
||
UNRAID_SSD_MAX=$(grep '^maxssd=' "$cfg" 2>/dev/null | cut -d= -f2 | tr -d '"')
|
||
fi
|
||
# Safe defaults if not found
|
||
UNRAID_DISK_HOT="${UNRAID_DISK_HOT:-45}"
|
||
UNRAID_DISK_MAX="${UNRAID_DISK_MAX:-55}"
|
||
UNRAID_SSD_HOT="${UNRAID_SSD_HOT:-60}"
|
||
UNRAID_SSD_MAX="${UNRAID_SSD_MAX:-70}"
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# check_local_disk_temps — check local disk temps before rsync
|
||
# Reads temps and rotational flag from /var/local/emhttp/disks.ini
|
||
# Uses unRAID's own thresholds from dynamix.cfg
|
||
#
|
||
# Returns:
|
||
# 0 = all temps OK
|
||
# 1 = warn threshold exceeded (skip this profile)
|
||
# 2 = critical threshold exceeded (abort all remaining profiles)
|
||
#
|
||
# Sets global TEMP_CHECK_RESULT with human readable summary
|
||
# -----------------------------------------------------------------------------------------------
|
||
check_local_disk_temps() {
|
||
get_unraid_temp_thresholds
|
||
|
||
local disks_ini="/var/local/emhttp/disks.ini"
|
||
if [[ ! -f "$disks_ini" ]]; then
|
||
warn "disks.ini not found — skipping temp check"
|
||
TEMP_CHECK_RESULT="temp check skipped (disks.ini not found)"
|
||
return 0
|
||
fi
|
||
|
||
local worst_result=0
|
||
local hot_drives=()
|
||
local crit_drives=()
|
||
local current_name=""
|
||
local current_device=""
|
||
local current_rotational=""
|
||
local current_temp=""
|
||
|
||
check_drive() {
|
||
[[ -z "$current_name" ]] || [[ -z "$current_temp" ]] && return
|
||
[[ "$current_temp" -eq 0 ]] && return # spun down
|
||
|
||
local warn_thresh crit_thresh
|
||
if [[ "$current_rotational" == "0" ]]; then
|
||
warn_thresh="$UNRAID_SSD_HOT"
|
||
crit_thresh="$UNRAID_SSD_MAX"
|
||
else
|
||
warn_thresh="$UNRAID_DISK_HOT"
|
||
crit_thresh="$UNRAID_DISK_MAX"
|
||
fi
|
||
|
||
if [[ "$current_temp" -ge "$crit_thresh" ]]; then
|
||
crit_drives+=("${current_name}(${current_device}):${current_temp}°C≥${crit_thresh}°C")
|
||
[[ $worst_result -lt 2 ]] && worst_result=2
|
||
elif [[ "$current_temp" -ge "$warn_thresh" ]]; then
|
||
hot_drives+=("${current_name}(${current_device}):${current_temp}°C≥${warn_thresh}°C")
|
||
[[ $worst_result -lt 1 ]] && worst_result=1
|
||
fi
|
||
}
|
||
|
||
while IFS= read -r ini_line; do
|
||
if echo "$ini_line" | grep -qE '^\["(disk[0-9]+|parity[0-9]?|cache[0-9]?)"\]'; then
|
||
# Save previous drive before starting new one
|
||
check_drive
|
||
current_name=$(echo "$ini_line" | grep -o '"[^"]*"' | head -1 | tr -d '"')
|
||
current_device=""
|
||
current_rotational="1" # default HDD
|
||
current_temp=""
|
||
elif echo "$ini_line" | grep -q '^device='; then
|
||
current_device=$(echo "$ini_line" | cut -d= -f2 | tr -d '"')
|
||
elif echo "$ini_line" | grep -q '^rotational='; then
|
||
current_rotational=$(echo "$ini_line" | cut -d= -f2 | tr -d '"')
|
||
elif echo "$ini_line" | grep -q '^temp='; then
|
||
current_temp=$(echo "$ini_line" | cut -d= -f2 | tr -d '"')
|
||
current_temp="${current_temp//[^0-9]/}"
|
||
current_temp="${current_temp:-0}"
|
||
fi
|
||
done < "$disks_ini"
|
||
check_drive # process last drive
|
||
|
||
if [[ ${#crit_drives[@]} -gt 0 ]]; then
|
||
TEMP_CHECK_RESULT="CRITICAL temps: ${crit_drives[*]}"
|
||
error "$ICON_WARN Drive temp CRITICAL — aborting all remaining syncs: ${crit_drives[*]}"
|
||
notify "Rsync aborted on $(hostname) — drive temp CRITICAL: ${crit_drives[*]}" "Rsync Temp Check" "warning"
|
||
return 2
|
||
elif [[ ${#hot_drives[@]} -gt 0 ]]; then
|
||
TEMP_CHECK_RESULT="HOT drives: ${hot_drives[*]}"
|
||
warn "$ICON_WARN Drive temp WARNING — skipping this profile: ${hot_drives[*]}"
|
||
notify "Rsync profile skipped on $(hostname) — drive temp WARNING: ${hot_drives[*]}" "Rsync Temp Check" "normal"
|
||
return 1
|
||
else
|
||
TEMP_CHECK_RESULT="all normal (HDD warn:${UNRAID_DISK_HOT}°C crit:${UNRAID_DISK_MAX}°C SSD warn:${UNRAID_SSD_HOT}°C crit:${UNRAID_SSD_MAX}°C)"
|
||
return 0
|
||
fi
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# check_remote_disks — verify all disks backing a share are healthy on the remote server
|
||
# Auto-detects filesystem type from disks.ini — handles XFS and ZFS correctly
|
||
# No SKIP_DISK_CHECK needed — detection is automatic
|
||
#
|
||
# XFS array disks: checks mountpoint is active via mountpoint -q
|
||
# ZFS disks/pools: checks zpool status is ONLINE
|
||
# Shares spanning multiple disks: all must pass
|
||
#
|
||
# Usage: check_remote_disks "/mnt/user/Movies"
|
||
# -----------------------------------------------------------------------------------------------
|
||
check_remote_disks() {
|
||
local dir="$1"
|
||
local share_name
|
||
share_name=$(basename "$dir")
|
||
|
||
info "$ICON_DISK Checking disks backing $share_name on $REMOTE_SERVER_NAME..."
|
||
|
||
# Get disk→fsType mapping from remote disks.ini
|
||
local disks_ini_content
|
||
disks_ini_content=$(ssh -i "$SSH_KEY" root@"$REMOTE_SERVER" \
|
||
"cat /var/local/emhttp/disks.ini 2>/dev/null" 2>/dev/null)
|
||
|
||
if [[ -z "$disks_ini_content" ]]; then
|
||
error "$ICON_DISK Cannot read disks.ini from $REMOTE_SERVER_NAME"
|
||
exit 1
|
||
fi
|
||
|
||
# Find which disk(s) back this share on remote — check full path first
|
||
# Share may be nested (e.g. /mnt/user/appdata-Failover/Critical-Data lives at
|
||
# /mnt/cache/appdata-Failover/Critical-Data on the remote cache pool)
|
||
local backing_disks
|
||
backing_disks=$(ssh -i "$SSH_KEY" root@"$REMOTE_SERVER" \
|
||
"ls -d /mnt/disk*/$share_name 2>/dev/null | awk -F/ '{print \$3}'" 2>/dev/null)
|
||
|
||
# Check ZFS standalone pools for the share — by basename AND nested path
|
||
local zfs_pool_paths
|
||
zfs_pool_paths=$(ssh -i "$SSH_KEY" root@"$REMOTE_SERVER" \
|
||
"zpool list -H -o name 2>/dev/null | while read pool; do
|
||
# Check direct: /mnt/pool/sharename
|
||
[[ -d \"/mnt/\${pool}/$share_name\" ]] && echo \"\$pool\" && continue
|
||
# Check nested: /mnt/pool/**/sharename (one level deep)
|
||
find \"/mnt/\${pool}\" -maxdepth 2 -name '$share_name' -type d 2>/dev/null | \
|
||
grep -q . && echo \"\$pool\"
|
||
done | sort -u" 2>/dev/null)
|
||
|
||
# Also check unRAID shfs — share exists under /mnt/user which is the union filesystem
|
||
# If the path exists under /mnt/user on remote it's accessible regardless of backing disk
|
||
if [[ -z "$backing_disks" ]] && [[ -z "$zfs_pool_paths" ]]; then
|
||
local on_user
|
||
on_user=$(ssh -i "$SSH_KEY" root@"$REMOTE_SERVER" \
|
||
"find /mnt/user -maxdepth 3 -name '$share_name' -type d 2>/dev/null | head -1" \
|
||
2>/dev/null)
|
||
if [[ -n "$on_user" ]]; then
|
||
# Share exists on remote shfs — find which pool backs it
|
||
local actual_path
|
||
actual_path=$(ssh -i "$SSH_KEY" root@"$REMOTE_SERVER" \
|
||
"readlink -f '$on_user' 2>/dev/null || echo '$on_user'" 2>/dev/null)
|
||
local pool_name
|
||
pool_name=$(echo "$actual_path" | awk -F/ '{print $3}')
|
||
if [[ -n "$pool_name" ]] && [[ "$pool_name" != "user" ]]; then
|
||
zfs_pool_paths="$pool_name"
|
||
info "$ICON_DISK $share_name found at $actual_path on $REMOTE_SERVER_NAME"
|
||
else
|
||
# Accessible via shfs — verify shfs is mounted
|
||
local shfs_ok
|
||
shfs_ok=$(ssh -i "$SSH_KEY" root@"$REMOTE_SERVER" \
|
||
"mountpoint -q /mnt/user && echo yes || echo no" 2>/dev/null)
|
||
if [[ "$shfs_ok" == "yes" ]]; then
|
||
success "All disks backing $share_name are online ✅ (via shfs)"
|
||
return 0
|
||
fi
|
||
fi
|
||
fi
|
||
fi
|
||
|
||
if [[ -z "$backing_disks" ]] && [[ -z "$zfs_pool_paths" ]]; then
|
||
error "$ICON_DISK No disks found backing $share_name on $REMOTE_SERVER_NAME"
|
||
exit 1
|
||
fi
|
||
|
||
local all_ok=true
|
||
|
||
# Check array disks (XFS or ZFS single-disk-in-array)
|
||
if [[ -n "$backing_disks" ]]; then
|
||
while IFS= read -r disk_name; do
|
||
[[ -z "$disk_name" ]] && continue
|
||
|
||
local fs_type
|
||
fs_type=$(echo "$disks_ini_content" | awk -F= -v disk="$disk_name" '
|
||
/^\["'"'"'?/ { current=substr($0,3,length($0)-4) }
|
||
current==disk && /^fsType=/ { print $2; exit }
|
||
' | tr -d '"')
|
||
fs_type="${fs_type:-xfs}"
|
||
|
||
if [[ "$fs_type" == "zfs" ]]; then
|
||
local pool_health
|
||
pool_health=$(ssh -i "$SSH_KEY" root@"$REMOTE_SERVER" \
|
||
"zpool list -H -o health '$disk_name' 2>/dev/null" 2>/dev/null)
|
||
if [[ "$pool_health" == "ONLINE" ]]; then
|
||
info "$ICON_DISK $disk_name (ZFS) $ICON_RUNNING — $share_name ONLINE"
|
||
else
|
||
error "$ICON_DISK $disk_name (ZFS) $ICON_STOPPED — pool ${pool_health:-offline}"
|
||
all_ok=false
|
||
fi
|
||
else
|
||
local mounted
|
||
mounted=$(ssh -i "$SSH_KEY" root@"$REMOTE_SERVER" \
|
||
"mountpoint -q '/mnt/$disk_name' && echo yes || echo no" 2>/dev/null)
|
||
if [[ "$mounted" == "yes" ]]; then
|
||
info "$ICON_DISK $disk_name (XFS) $ICON_RUNNING — $share_name present"
|
||
else
|
||
error "$ICON_DISK $disk_name (XFS) $ICON_STOPPED — not mounted"
|
||
all_ok=false
|
||
fi
|
||
fi
|
||
done <<< "$backing_disks"
|
||
fi
|
||
|
||
# Check ZFS standalone pools (cache, gaming, media-servers etc.)
|
||
if [[ -n "$zfs_pool_paths" ]]; then
|
||
while IFS= read -r pool_name; do
|
||
[[ -z "$pool_name" ]] && continue
|
||
local pool_health
|
||
pool_health=$(ssh -i "$SSH_KEY" root@"$REMOTE_SERVER" \
|
||
"zpool list -H -o health '$pool_name' 2>/dev/null" 2>/dev/null)
|
||
if [[ "$pool_health" == "ONLINE" ]]; then
|
||
info "$ICON_DISK $pool_name (ZFS pool) $ICON_RUNNING — $share_name ONLINE"
|
||
else
|
||
error "$ICON_DISK $pool_name (ZFS pool) $ICON_STOPPED — pool ${pool_health:-offline}"
|
||
all_ok=false
|
||
fi
|
||
done <<< "$zfs_pool_paths"
|
||
fi
|
||
|
||
if [[ "$all_ok" == false ]]; then
|
||
error "One or more disks backing $share_name are offline on $REMOTE_SERVER_NAME"
|
||
exit 1
|
||
fi
|
||
success "All disks backing $share_name are online ✅"
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# CONTAINER MANAGEMENT — STOP (remote via SSH)
|
||
# Stops containers in CRITICAL_CONTAINER_NAMES on remote server.
|
||
# Tracks running containers in RUNNING_CONTAINERS for restart after rsync.
|
||
# -----------------------------------------------------------------------------------------------
|
||
RUNNING_CONTAINERS=()
|
||
|
||
stop_containers() {
|
||
if [[ ${#CRITICAL_CONTAINER_NAMES[@]} -eq 0 ]] || \
|
||
[[ "${CRITICAL_CONTAINER_NAMES[*]}" == "" ]]; then
|
||
log "No remote containers configured for this profile, skipping stop."
|
||
return
|
||
fi
|
||
info "Stopping remote containers..."
|
||
RUNNING_CONTAINERS=()
|
||
for c in "${CRITICAL_CONTAINER_NAMES[@]}"; do
|
||
[[ -z "$c" ]] && continue
|
||
STATUS=$(ssh -i "$SSH_KEY" root@"$REMOTE_SERVER" \
|
||
"docker inspect -f '{{.State.Running}}' $c 2>/dev/null" 2>/dev/null || echo "unknown")
|
||
if [[ "$STATUS" == "true" ]]; then
|
||
echo "$ICON_STOP Stopping $c..."
|
||
RUNNING_CONTAINERS+=("$c")
|
||
if ssh -i "$SSH_KEY" root@"$REMOTE_SERVER" "docker stop $c" >/dev/null; then
|
||
echo "$ICON_STOPPED $c stopped"
|
||
else
|
||
error "Failed to stop $c"
|
||
fi
|
||
elif [[ "$STATUS" == "false" ]]; then
|
||
echo "$ICON_NOT_RUNNING $c is not running, skipping"
|
||
else
|
||
log "$c not found on remote — skipping"
|
||
fi
|
||
done
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# CONTAINER MANAGEMENT — STOP LOCAL
|
||
# Stops containers on the LOCAL server before rsync pushes data out.
|
||
# Uses PROFILE_LOCAL_CRITICAL_CONTAINER_NAMES — same naming scheme as remote.
|
||
# If container not found on this server → skipped gracefully, not errored.
|
||
# Only containers that were running get tracked for restart.
|
||
# -----------------------------------------------------------------------------------------------
|
||
stop_local_containers() {
|
||
if [[ ${#LOCAL_CRITICAL_CONTAINER_NAMES[@]} -eq 0 ]] || \
|
||
[[ "${LOCAL_CRITICAL_CONTAINER_NAMES[*]}" == "" ]]; then
|
||
log "No local containers configured for this profile, skipping local stop."
|
||
return
|
||
fi
|
||
info "Stopping local containers..."
|
||
LOCAL_RUNNING_CONTAINERS=()
|
||
for c in "${LOCAL_CRITICAL_CONTAINER_NAMES[@]}"; do
|
||
[[ -z "$c" ]] && continue
|
||
STATUS=$(docker inspect -f '{{.State.Running}}' "$c" 2>/dev/null || echo "unknown")
|
||
if [[ "$STATUS" == "true" ]]; then
|
||
echo "$ICON_STOP Stopping local $c..."
|
||
LOCAL_RUNNING_CONTAINERS+=("$c")
|
||
if docker stop "$c" >/dev/null; then
|
||
echo "$ICON_STOPPED $c stopped"
|
||
else
|
||
error "Failed to stop local $c"
|
||
fi
|
||
elif [[ "$STATUS" == "false" ]]; then
|
||
echo "$ICON_NOT_RUNNING $c is not running, skipping"
|
||
else
|
||
log "$c not found locally — skipping"
|
||
fi
|
||
done
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# CONTAINER MANAGEMENT — START (remote via SSH)
|
||
# Restarts only containers tracked in RUNNING_CONTAINERS.
|
||
# Delayed containers receive CONTAINER_DELAY seconds before starting.
|
||
# -----------------------------------------------------------------------------------------------
|
||
start_containers() {
|
||
if [[ ${#RUNNING_CONTAINERS[@]} -eq 0 ]]; then
|
||
log "No remote containers to restart."
|
||
return
|
||
fi
|
||
info "Starting remote containers..."
|
||
for c in "${RUNNING_CONTAINERS[@]}"; do
|
||
[[ -z "$c" ]] && continue
|
||
local needs_delay=false
|
||
for d in "${DELAYED_CONTAINERS[@]}"; do
|
||
[[ "$c" == "$d" ]] && needs_delay=true && break
|
||
done
|
||
if [[ "$needs_delay" == true ]]; then
|
||
info "Waiting ${CONTAINER_DELAY}s before starting $c..."
|
||
sleep "$CONTAINER_DELAY"
|
||
fi
|
||
echo "$ICON_START Starting $c..."
|
||
if ssh -i "$SSH_KEY" root@"$REMOTE_SERVER" "docker start $c" >/dev/null 2>&1; then
|
||
echo "$ICON_STARTED $c started"
|
||
else
|
||
error "Failed to start $c — start manually if needed"
|
||
fi
|
||
done
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# CONTAINER MANAGEMENT — START LOCAL
|
||
# Restarts only containers tracked in LOCAL_RUNNING_CONTAINERS.
|
||
# Respects DELAYED_CONTAINERS and CONTAINER_DELAY same as remote start.
|
||
# If container not found → skipped gracefully.
|
||
# -----------------------------------------------------------------------------------------------
|
||
start_local_containers() {
|
||
if [[ ${#LOCAL_RUNNING_CONTAINERS[@]} -eq 0 ]]; then
|
||
log "No local containers to restart."
|
||
return
|
||
fi
|
||
info "Starting local containers..."
|
||
for c in "${LOCAL_RUNNING_CONTAINERS[@]}"; do
|
||
[[ -z "$c" ]] && continue
|
||
local needs_delay=false
|
||
for d in "${DELAYED_CONTAINERS[@]}"; do
|
||
[[ "$c" == "$d" ]] && needs_delay=true && break
|
||
done
|
||
if [[ "$needs_delay" == true ]]; then
|
||
info "Waiting ${CONTAINER_DELAY}s before starting local $c..."
|
||
sleep "$CONTAINER_DELAY"
|
||
fi
|
||
echo "$ICON_START Starting local $c..."
|
||
if docker start "$c" >/dev/null 2>&1; then
|
||
echo "$ICON_STARTED $c started"
|
||
else
|
||
error "Failed to start local $c — start manually if needed"
|
||
fi
|
||
done
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# RSYNC OPTIONS
|
||
# Loads rsync options for current profile. Falls back to DEFAULT_RSYNC_OPTS if no match.
|
||
# Profile opts do NOT inherit from defaults — list all desired flags explicitly.
|
||
# -----------------------------------------------------------------------------------------------
|
||
get_rsync_opts() {
|
||
if [[ -n "${PROFILE_RSYNC_OPTS[$PROFILE_NAME]:-}" ]]; then
|
||
read -r -a RSYNC_OPTS <<< "${PROFILE_RSYNC_OPTS[$PROFILE_NAME]}"
|
||
log "Using profile rsync opts for $PROFILE_NAME: ${RSYNC_OPTS[*]}"
|
||
else
|
||
RSYNC_OPTS=("${DEFAULT_RSYNC_OPTS[@]}")
|
||
log "Using default rsync opts: ${RSYNC_OPTS[*]}"
|
||
fi
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# SCRIPT LOCKING — v3.0
|
||
# Prevents multiple instances of the same script running simultaneously.
|
||
# All lock files live in /tmp/unraid_locks/ — auto-cleared on reboot.
|
||
#
|
||
# Usage in scripts:
|
||
# acquire_lock — strict: exit immediately if already running
|
||
# acquire_lock "wait" — wait mode: wait briefly then exit if still locked
|
||
# acquire_rsync_lock "$profile" — per-profile + global concurrent limit
|
||
#
|
||
# Stale lock detection — if lock file exists but PID is dead, clears and proceeds.
|
||
# Lock age warning — if lock is older than expected, warns but does not override.
|
||
# EXIT trap registered automatically — lock always released on exit, crash, or kill.
|
||
# -----------------------------------------------------------------------------------------------
|
||
|
||
LOCK_DIR="/tmp/unraid_locks"
|
||
|
||
# Ensure DATA_DIR exists — created here so every script that sources common.sh
|
||
# can safely write to it without checking first
|
||
if [[ -n "${DATA_DIR:-}" ]] && [[ ! -d "$DATA_DIR" ]]; then
|
||
mkdir -p "$DATA_DIR" 2>/dev/null || true
|
||
fi
|
||
RSYNC_COUNT_FILE="$LOCK_DIR/rsync_active_count"
|
||
RSYNC_MAX_CONCURRENT=3
|
||
LOCK_WARN_AGE=300 # seconds — warn if lock older than this (5min default)
|
||
LOCK_WAIT_TIMEOUT=30 # seconds — how long "wait" mode waits before giving up
|
||
|
||
# Internal — script name used as lock identifier
|
||
_lock_name() {
|
||
basename "${BASH_SOURCE[1]:-$0}" .sh
|
||
}
|
||
|
||
# Internal — lock file path for this script
|
||
_lock_file() {
|
||
echo "$LOCK_DIR/${1:-$(_lock_name)}.lock"
|
||
}
|
||
|
||
# Internal — release lock on exit
|
||
_release_on_exit() {
|
||
local lockfile="$1"
|
||
[[ -f "$lockfile" ]] && rm -f "$lockfile"
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# acquire_lock — acquire exclusive lock for this script
|
||
# Mode: strict (default) — exit immediately if locked
|
||
# wait — wait LOCK_WAIT_TIMEOUT seconds then exit
|
||
# continuous — for long-running scripts: skip gracefully if healthy
|
||
# Lock file stores PID:scriptname — prevents PID reuse false positives
|
||
# Stale lock: if PID dead OR PID belongs to different process → clear and acquire
|
||
# Age warning: if lock older than LOCK_WARN_AGE → warn (skipped for continuous)
|
||
# -----------------------------------------------------------------------------------------------
|
||
acquire_lock() {
|
||
local mode="${1:-strict}"
|
||
local script_name
|
||
script_name=$(basename "${BASH_SOURCE[1]:-$0}" .sh)
|
||
local lockfile
|
||
lockfile="$(_lock_file "$script_name")"
|
||
|
||
mkdir -p "$LOCK_DIR"
|
||
|
||
# Check for existing lock
|
||
if [[ -f "$lockfile" ]]; then
|
||
local lock_content existing_pid locked_name
|
||
lock_content=$(cat "$lockfile" 2>/dev/null)
|
||
existing_pid="${lock_content%%:*}"
|
||
locked_name="${lock_content##*:}"
|
||
|
||
# Stale lock — PID dead
|
||
if [[ -z "$existing_pid" ]] || ! kill -0 "$existing_pid" 2>/dev/null; then
|
||
warn "Stale lock detected for $script_name (PID $existing_pid gone) — clearing"
|
||
rm -f "$lockfile"
|
||
|
||
# PID reuse — PID alive but belongs to a different process
|
||
elif [[ "$locked_name" != "$script_name" ]]; then
|
||
warn "Lock PID $existing_pid reused by different process ($locked_name ≠ $script_name) — clearing stale lock"
|
||
rm -f "$lockfile"
|
||
|
||
else
|
||
# Lock is genuinely active — check age (skip warning for continuous scripts)
|
||
local lock_age
|
||
lock_age=$(( $(date +%s) - $(stat -c %Y "$lockfile" 2>/dev/null || echo 0) ))
|
||
if [[ "$lock_age" -gt "$LOCK_WARN_AGE" ]] && [[ "$mode" != "continuous" ]]; then
|
||
warn "$script_name has been running for ${lock_age}s — may be stuck (PID $existing_pid)"
|
||
fi
|
||
|
||
if [[ "$mode" == "continuous" ]]; then
|
||
# Continuous scripts — healthy instance = always skip gracefully
|
||
log "$ICON_SKIP $script_name already running healthy (PID $existing_pid) — skipping"
|
||
exit 0
|
||
elif [[ "$mode" == "wait" ]]; then
|
||
info "Another instance of $script_name is running — waiting up to ${LOCK_WAIT_TIMEOUT}s"
|
||
local waited=0
|
||
while [[ -f "$lockfile" ]] && [[ "$waited" -lt "$LOCK_WAIT_TIMEOUT" ]]; do
|
||
sleep 1
|
||
((waited++))
|
||
lock_content=$(cat "$lockfile" 2>/dev/null)
|
||
existing_pid="${lock_content%%:*}"
|
||
locked_name="${lock_content##*:}"
|
||
if [[ -z "$existing_pid" ]] || ! kill -0 "$existing_pid" 2>/dev/null; then
|
||
warn "Lock became stale while waiting — clearing"
|
||
rm -f "$lockfile"
|
||
break
|
||
elif [[ "$locked_name" != "$script_name" ]]; then
|
||
warn "Lock PID reused while waiting — clearing"
|
||
rm -f "$lockfile"
|
||
break
|
||
fi
|
||
done
|
||
if [[ -f "$lockfile" ]]; then
|
||
error "$script_name still locked after ${LOCK_WAIT_TIMEOUT}s — exiting"
|
||
exit 1
|
||
fi
|
||
else
|
||
error "Another instance of $script_name is already running (PID $existing_pid) — exiting"
|
||
case "$script_name" in
|
||
failover|transcode_management|media_management|daily_sync_maintenance|system_watchdog)
|
||
notify "$script_name lock collision on $(hostname) — concurrent instance detected" "$script_name" "warning"
|
||
;;
|
||
esac
|
||
exit 1
|
||
fi
|
||
fi
|
||
fi
|
||
|
||
# Acquire lock — store PID:scriptname to prevent PID reuse false positives
|
||
echo "$$:$script_name" > "$lockfile"
|
||
|
||
# Register EXIT trap to always release lock
|
||
trap "_release_on_exit '$lockfile'" EXIT
|
||
|
||
log "$ICON_LOCK Lock acquired: $script_name (PID $$)"
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# acquire_rsync_lock — per-profile lock + global concurrent limit
|
||
# Prevents same profile running twice and limits total concurrent rsync instances
|
||
# -----------------------------------------------------------------------------------------------
|
||
acquire_rsync_lock() {
|
||
local profile="$1"
|
||
local profile_lock
|
||
profile_lock="$(_lock_file "rsync_${profile}")"
|
||
|
||
mkdir -p "$LOCK_DIR"
|
||
|
||
# Per-profile lock — same profile cannot run twice
|
||
if [[ -f "$profile_lock" ]]; then
|
||
local lock_content existing_pid locked_name
|
||
lock_content=$(cat "$profile_lock" 2>/dev/null)
|
||
existing_pid="${lock_content%%:*}"
|
||
locked_name="${lock_content##*:}"
|
||
if [[ -n "$existing_pid" ]] && kill -0 "$existing_pid" 2>/dev/null && [[ "$locked_name" == "rsync_${profile}" ]]; then
|
||
error "rsync profile '$profile' is already running (PID $existing_pid) — exiting"
|
||
exit 1
|
||
else
|
||
warn "Stale rsync lock for profile '$profile' — clearing"
|
||
rm -f "$profile_lock"
|
||
fi
|
||
fi
|
||
|
||
# Global concurrent limit
|
||
local current_count=0
|
||
if [[ -f "$RSYNC_COUNT_FILE" ]]; then
|
||
current_count=$(cat "$RSYNC_COUNT_FILE" 2>/dev/null || echo 0)
|
||
# Validate count — clean up if stale
|
||
local actual_count=0
|
||
for lf in "$LOCK_DIR"/rsync_*.lock; do
|
||
[[ -f "$lf" ]] || continue
|
||
local lpid
|
||
lpid=$(cat "$lf" 2>/dev/null)
|
||
kill -0 "$lpid" 2>/dev/null && ((actual_count++))
|
||
done
|
||
if [[ "$actual_count" -ne "$current_count" ]]; then
|
||
log "rsync count corrected: $current_count → $actual_count"
|
||
current_count=$actual_count
|
||
echo "$current_count" > "$RSYNC_COUNT_FILE"
|
||
fi
|
||
fi
|
||
|
||
if [[ "$current_count" -ge "$RSYNC_MAX_CONCURRENT" ]]; then
|
||
error "Maximum concurrent rsync limit ($RSYNC_MAX_CONCURRENT) reached — exiting"
|
||
info "Active rsync locks: $(ls "$LOCK_DIR"/rsync_*.lock 2>/dev/null | xargs -I{} basename {} .lock | tr '\n' ' ')"
|
||
exit 1
|
||
fi
|
||
|
||
# Acquire profile lock and increment counter
|
||
echo "$$:rsync_${profile}" > "$profile_lock"
|
||
echo $(( current_count + 1 )) > "$RSYNC_COUNT_FILE"
|
||
|
||
# Register EXIT trap
|
||
trap "_release_rsync_on_exit '$profile_lock'" EXIT
|
||
|
||
log "$ICON_LOCK rsync lock acquired: profile '$profile' (PID $$, active: $(( current_count + 1 ))/$RSYNC_MAX_CONCURRENT)"
|
||
}
|
||
|
||
# Internal — release rsync lock on exit
|
||
_release_rsync_on_exit() {
|
||
local profile_lock="$1"
|
||
[[ -f "$profile_lock" ]] && rm -f "$profile_lock"
|
||
# Decrement global counter
|
||
if [[ -f "$RSYNC_COUNT_FILE" ]]; then
|
||
local count
|
||
count=$(cat "$RSYNC_COUNT_FILE" 2>/dev/null || echo 1)
|
||
count=$(( count - 1 ))
|
||
[[ "$count" -lt 0 ]] && count=0
|
||
echo "$count" > "$RSYNC_COUNT_FILE"
|
||
fi
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# check_api — pre-flight API reachability check
|
||
# Verifies API endpoint is reachable before attempting operations
|
||
# Usage: check_api "http://localhost:8989" "Sonarr" || exit 1
|
||
# -----------------------------------------------------------------------------------------------
|
||
# -----------------------------------------------------------------------------------------------
|
||
# translate_path — translate container path to host path
|
||
# Uses ARR_PATH_MAP associative array — must be declared in calling script before use
|
||
#
|
||
# Usage:
|
||
# declare -A ARR_PATH_MAP=(["/ext-music"]="/mnt/user/Music-New")
|
||
# translate_path "/ext-music/Artist/Album/track.flac"
|
||
# Returns: /mnt/user/Music-New/Artist/Album/track.flac
|
||
#
|
||
# If no match found — returns path unchanged
|
||
# Handles longest-match first to avoid partial path collisions
|
||
# ARR_PATH_MAP must be exported or declared global before calling
|
||
# -----------------------------------------------------------------------------------------------
|
||
translate_path() {
|
||
local api_path="$1"
|
||
local best_match=""
|
||
local best_len=0
|
||
|
||
# Find longest matching container path prefix
|
||
for container_path in "${!ARR_PATH_MAP[@]}"; do
|
||
if [[ "$api_path" == "$container_path"* ]]; then
|
||
if [[ "${#container_path}" -gt "$best_len" ]]; then
|
||
best_match="$container_path"
|
||
best_len="${#container_path}"
|
||
fi
|
||
fi
|
||
done
|
||
|
||
if [[ -n "$best_match" ]]; then
|
||
echo "${ARR_PATH_MAP[$best_match]}${api_path#$best_match}"
|
||
else
|
||
echo "$api_path" # no match — return unchanged
|
||
fi
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# check_arr_version — verify arr major version matches tested version in Master.conf
|
||
# Exits the calling script if version doesn't match — prevents running against untested API
|
||
#
|
||
# Usage:
|
||
# check_arr_version "$RADARR_URL" "$RADARR_API_KEY" "v3" "$RADARR_VERSION_MAJOR" "Radarr"
|
||
#
|
||
# Arguments:
|
||
# $1 = arr base URL
|
||
# $2 = API key
|
||
# $3 = API path prefix (v3 or v1)
|
||
# $4 = expected major version number
|
||
# $5 = arr name for error messages
|
||
# -----------------------------------------------------------------------------------------------
|
||
check_arr_version() {
|
||
local url="$1"
|
||
local api_key="$2"
|
||
local api_prefix="$3"
|
||
local expected_major="$4"
|
||
local arr_name="${5:-Arr}"
|
||
|
||
local status_response version major_version
|
||
|
||
status_response=$(curl -sf --max-time 10 \
|
||
-H "X-Api-Key: $api_key" \
|
||
"${url}/api/${api_prefix}/system/status" 2>/dev/null)
|
||
|
||
if [[ -z "$status_response" ]]; then
|
||
warn "$arr_name version check failed — could not reach system/status endpoint"
|
||
warn "Proceeding without version verification — monitor for API errors"
|
||
return 0
|
||
fi
|
||
|
||
version=$(echo "$status_response" | \
|
||
grep -o '"version":"[^"]*"' | \
|
||
grep -o '[0-9][^"]*' | head -1)
|
||
|
||
if [[ -z "$version" ]]; then
|
||
warn "$arr_name version check failed — could not parse version from response"
|
||
warn "Proceeding without version verification — monitor for API errors"
|
||
return 0
|
||
fi
|
||
|
||
major_version="${version%%.*}"
|
||
|
||
if [[ "$major_version" == "$expected_major" ]]; then
|
||
success "$arr_name version: $version (major $major_version — tested ✅)"
|
||
return 0
|
||
else
|
||
error "$arr_name version mismatch — running v${major_version}, tested against v${expected_major}"
|
||
error "The API endpoint structure may have changed — exiting to protect your library"
|
||
error "Update ${arr_name^^}_VERSION_MAJOR in Master.conf after verifying the script works with v${major_version}"
|
||
notify "$arr_name version mismatch on $(hostname) — running v${major_version}, script tested against v${expected_major}" "$arr_name Cleanup" "warning"
|
||
exit 1
|
||
fi
|
||
}
|
||
|
||
check_api() {
|
||
local url="$1"
|
||
local service="${2:-API}"
|
||
local timeout="${3:-10}"
|
||
|
||
if curl -sf --max-time "$timeout" "$url" >/dev/null 2>&1; then
|
||
log "$service API reachable: $url"
|
||
return 0
|
||
else
|
||
error "$service API not reachable: $url"
|
||
return 1
|
||
fi
|
||
}
|
||
|
||
# -----------------------------------------------------------------------------------------------
|
||
# STATUS DISPLAY
|
||
# Prints current runtime configuration — triggered by --status flag.
|
||
# -----------------------------------------------------------------------------------------------
|
||
show_status() {
|
||
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
|
||
echo "Local: $LOCAL_SERVER_NAME"
|
||
echo "Remote: $REMOTE_SERVER_NAME"
|
||
echo "IP: $REMOTE_SERVER"
|
||
echo "Profile: ${PROFILE_NAME:-n/a}"
|
||
echo "DryRun: $DRY_RUN"
|
||
echo "Logging: $ENABLE_LOGGING"
|
||
echo "Containers: ${CRITICAL_CONTAINER_NAMES[*]:-n/a}"
|
||
echo "Delayed: ${DELAYED_CONTAINERS[*]:-n/a}"
|
||
echo "Excludes: ${EXCLUDE_DIRS[*]:-n/a}"
|
||
echo "━━━━━━━━━━━━━━━━━━━━━━━"
|
||
} |