Files
Varaverk/Tools/failover_state_reset.sh
T

229 lines
9.9 KiB
Bash

#!/bin/bash
# ==============================================================================================
# ============================= Failover State Reset ===========================================
# ==============================================================================================
# Resets the failover state file to NORMAL and clears all tier flags.
# Use when the failover state file is stuck in a non-NORMAL state after:
# - Failover testing that left state as FAILOVER
# - A failed handback that did not complete cleanly
# - Manual intervention that left state inconsistent
# - failover.sh was killed mid-cycle and state is unknown
#
# ── WHAT THIS DOES ────────────────────────────────────────────────────────────────────────────
# Writes a fresh state file with:
# state=NORMAL
# failover_start=0
# handback_strikes=0
# tier2_started=false / tier3_started=false / tier4_started=false
#
# Does NOT start or stop any containers — state file only.
# After reset, failover.sh will resume from NORMAL on its next cycle.
#
# ── ⚠️ ONLY RUN WHEN SAFE ────────────────────────────────────────────────────────────────────
# Verify BEFORE resetting:
# ✓ Right containers running on the right server
# ✓ DDNS pointing at the correct server
# ✓ No active failover actually in progress
# ✓ Both servers can see each other
#
# Resetting state while a real failover is happening causes failover.sh to stop
# covering the remote server — services go offline until next detection cycle.
#
# ── SAFEGUARDS ────────────────────────────────────────────────────────────────────────────────
# failover.sh running check — warns if failover.sh is active when reset is attempted
# acquire_lock — prevents concurrent resets
# flock on state write — prevents race with failover.sh mid-cycle read
# Confirmation required — interactive: type YES | non-interactive: --force flag
# validate_unraid_cmd — notify validated before use
#
# ── USAGE ─────────────────────────────────────────────────────────────────────────────────────
# failover_state_reset.sh — interactive reset (prompts for YES)
# failover_state_reset.sh --dry-run — show current state, show what would be written
# failover_state_reset.sh --status — show current state file contents and exit
# failover_state_reset.sh --force — non-interactive reset (no prompt, use in scripts)
# failover_state_reset.sh --force --dry-run — dry run without prompt
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
# ── Handle --force flag before parse_args ─────────────────────────────────────────────────────
FORCE=false
FILTERED_ARGS=()
for arg in "$@"; do
case "$arg" in
--force) FORCE=true ;;
*) FILTERED_ARGS+=("$arg") ;;
esac
done
parse_args "${FILTERED_ARGS[@]}"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
validate_unraid_cmd \
"/usr/local/emhttp/plugins/dynamix/scripts/notify" \
"" "" \
"unRAID notify script" || warn "unRAID notify script not found — native notifications disabled"
acquire_lock
# detect_hosts() sets MY_ID — used in summary and notification
detect_hosts
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes will be made"
[[ "$FORCE" == true ]] && warn "FORCE mode — confirmation prompt skipped"
# ==============================================================================================
# ━━━ Current State ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_FAILOVER Current Failover State ━━━"
echo "$ICON_HOST My ID: $MY_ID ($LOCAL_SERVER_NAME)"
echo ""
if [[ ! -f "$FAILOVER_STATE_FILE" ]]; then
warn "State file not found: $FAILOVER_STATE_FILE"
warn "Will be created fresh on reset"
CURRENT_STATE="NOT FOUND"
else
log "State file: $FAILOVER_STATE_FILE"
echo ""
while IFS='=' read -r key value; do
[[ -z "$key" ]] && continue
echo " $ICON_INFO $key = $value"
done < "$FAILOVER_STATE_FILE"
CURRENT_STATE=$(grep "^state=" "$FAILOVER_STATE_FILE" 2>/dev/null | cut -d= -f2)
fi
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
# Check if failover.sh is running — informational in status mode
if pgrep -f "failover.sh" >/dev/null 2>&1; then
warn "failover.sh is currently RUNNING — any reset would race with active cycle"
else
log "failover.sh is not running"
fi
exit 0
fi
# ==============================================================================================
# ━━━ Safety Checks ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SHIELD Safety Checks ━━━"
# Check if failover.sh is actively running
FAILOVER_RUNNING=false
if pgrep -f "failover.sh" >/dev/null 2>&1; then
FAILOVER_RUNNING=true
warn "⚠️ failover.sh is currently RUNNING"
warn "Resetting state mid-cycle may cause incorrect decisions on the next iteration"
warn "Consider stopping failover.sh first (click Abort in User Scripts)"
warn "Then reset state, then restart failover.sh"
echo ""
warn "If you are sure you want to proceed anyway, confirm below"
else
log "failover.sh is not running — safe to reset ✅"
fi
# Check current state — if already NORMAL warn user
if [[ "$CURRENT_STATE" == "NORMAL" ]]; then
warn "State is already NORMAL — reset may not be necessary"
warn "Proceeding anyway (will refresh the state file)"
fi
# ==============================================================================================
# ━━━ Confirmation ━━━
# ==============================================================================================
echo ""
warn "This will reset failover state to NORMAL on $MY_ID ($LOCAL_SERVER_NAME)"
warn "Verify before proceeding:"
warn " ✓ Right containers running on the right server"
warn " ✓ DDNS pointing at correct server"
warn " ✓ No real failover actually in progress"
warn " ✓ Both servers can reach each other"
echo ""
if [[ "$DRY_RUN" == false ]]; then
if [[ "$FORCE" == true ]]; then
log "FORCE flag set — skipping confirmation prompt"
elif [[ -t 0 ]]; then
# Interactive terminal — prompt for confirmation
read -r -p "Type YES to confirm reset: " CONFIRM
if [[ "$CONFIRM" != "YES" ]]; then
warn "Reset cancelled"
exit 0
fi
else
# Non-interactive — no terminal, cannot prompt
error "Non-interactive mode — use --force flag to skip confirmation"
error "Usage: failover_state_reset.sh --force"
exit 1
fi
fi
# ==============================================================================================
# ━━━ Reset State File ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_FAILOVER Resetting State File ━━━"
NEW_STATE_CONTENT="state=NORMAL
failover_start=0
handback_strikes=0
tier2_started=false
tier3_started=false
tier4_started=false
last_reset=$(date '+%Y-%m-%d %H:%M:%S')
reset_by=$MY_ID"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would write to $FAILOVER_STATE_FILE:"
echo ""
echo "$NEW_STATE_CONTENT" | while IFS= read -r line; do
echo " $line"
done
else
mkdir -p "$(dirname "$FAILOVER_STATE_FILE")"
# flock prevents race with failover.sh mid-cycle read/write
(
flock -x 200
echo "$NEW_STATE_CONTENT" > "$FAILOVER_STATE_FILE"
) 200>"${FAILOVER_STATE_FILE}.lock"
warn "State file reset to NORMAL ✅"
log "Written to: $FAILOVER_STATE_FILE"
fi
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY FAILOVER STATE RESET SUMMARY ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_FAILOVER File: $FAILOVER_STATE_FILE"
echo "$ICON_TIME Reset at: $(date '+%Y-%m-%d %H:%M:%S')"
echo ""
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no changes made"
else
warn "$ICON_DONE State reset to NORMAL"
log "failover.sh will resume from NORMAL on next cycle"
log "No containers were started or stopped"
echo ""
[[ "$FAILOVER_RUNNING" == true ]] && \
warn "⚠️ failover.sh was running during reset — monitor next cycle carefully"
notify "Failover state manually reset to NORMAL on $(hostname) ($MY_ID)" \
"Failover State Reset" "warning"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"