#!/bin/bash # ══════════════════════════════════════════════════════════════════════════════════════════════ # PURPOSE # Put the containers this host has marked for fallback coverage onto the partner, so that the # partner can actually start them during an outage — and take them off again on request. # # OPERATIONAL MODEL # fallback.sh covers a host by running `docker start ` on the partner. It never creates # anything. So a name in FALLBACK__TIER* is a promise that only holds if the partner already # has that container built. Measured 2026-08-23: all 12 of HOST1's covered containers were absent # from HOST2, meaning every tier would have failed on the first real outage while the UI showed # coverage as configured. This script is what closes that gap. # # Push and remove are separate, deliberate actions, never a side effect of saving the tier list. # Editing coverage is a cheap config write; deploying a dozen containers onto another machine is # not, and the two should not share a button. # # DESIGN PRINCIPLES # Deployed, then verified STOPPED. # A container built here and left running on the partner would be a second live instance of # NextCloud, Gitea or PostgreSQL_Immich against the same data while this host is healthy. # That is the danger_rsync_live_database_appdata failure with worse odds. Every deploy is # followed by a stop and a re-inspect, and a container that will not stay stopped is an # error, not a warning. # # Remove takes the container AND its appdata. # Operator decision 2026-08-23: the button is explicit, so a removal should leave nothing # behind to reason about later. The risk it accepts is narrow and worth naming — if the # partner ever covered for us, ITS appdata is the newer copy and is what a handback rsyncs # home. The NORMAL-state gate below closes the live-failover window; what it cannot see is # a handback that partially failed and then returned to NORMAL, so the UI says so before # asking. # # Two guards on the deletion itself: only paths under /mnt/*/appdata* are ever touched, and # a bind of the appdata ROOT is refused outright — a container mounting /mnt/user/appdata # would otherwise turn one removal into wiping every application on the partner. # # Refuses to run unless fallback state is NORMAL. # Pushing or removing containers mid-outage edits the thing currently keeping services up. # # Coverage names are resolved to templates by , not by filename. # my-Foo.xml routinely holds a container called something else. Matching on the filename # silently pushes the wrong template, or nothing at all. # # USAGE # coverage_deploy.sh --push deploy every covered container onto the partner (stopped) # coverage_deploy.sh --remove stop, remove, and delete the pushed template on the partner # coverage_deploy.sh --status report, per covered container, whether it exists there # any mode supports --dry-run # # DEPENDS ON # Plugin//Partnership/containers.sh deploy_container_from_xml(), GPU transform # FALLBACK__TIER1-4 the coverage list this acts on # ══════════════════════════════════════════════════════════════════════════════════════════════ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" source "$SCRIPT_DIR/../load_config.sh" source "$SCRIPT_DIR/../Plugin/$PLATFORM/Partnership/containers.sh" SSH_TIMEOUT="${SSH_TIMEOUT:-15}" MODE="" DRY_RUN="${DRY_RUN:-false}" for arg in "$@"; do case "$arg" in --push) MODE="push" ;; --remove) MODE="remove" ;; --status) MODE="status" ;; --dry-run) DRY_RUN=true ;; esac done if [[ -z "$MODE" ]]; then error "No mode given — use --push, --remove or --status" exit 1 fi detect_hosts if [[ -z "$REMOTE_ID" || "$REMOTE_SERVER_NAME" == "unknown" ]]; then error "No partner configured — nothing to push to" exit 1 fi # ── Gate: only with fallback idle ───────────────────────────────────────────────────────────── # Read rather than assumed. A missing state file means fallback has never run, which is idle # enough; a file that says anything other than NORMAL means services are in motion right now. FALLBACK_STATE_FILE="${FALLBACK_STATE_FILE:-${STATE_DIR}/fallback_state.db}" _fb_state="NORMAL" if [[ -f "$FALLBACK_STATE_FILE" ]]; then _fb_state=$(grep -m1 '^state=' "$FALLBACK_STATE_FILE" 2>/dev/null | cut -d= -f2) _fb_state="${_fb_state:-NORMAL}" fi if [[ "$_fb_state" != "NORMAL" && "$MODE" != "status" ]]; then error "Fallback state is $_fb_state, not NORMAL — refusing to $MODE" error "Changing what the partner holds while a failover is live edits the thing keeping services up." exit 1 fi # ── The coverage list ───────────────────────────────────────────────────────────────────────── COVERED=() for _t in 1 2 3 4; do _var="FALLBACK_${MY_ID}_TIER${_t}[@]" for _c in "${!_var}"; do [[ -n "$_c" ]] && COVERED+=("$_c") done done if [[ ${#COVERED[@]} -eq 0 ]]; then warn "No containers are covered in FALLBACK_${MY_ID}_TIER1-4 — nothing to do" exit 0 fi log "$ICON_FALLBACK Coverage: ${#COVERED[@]} container(s) for $REMOTE_SERVER_NAME to start during an outage" resolve_remote_ip MIRROR="$REMOTE_SERVER_NAME" MIRROR_IP="$REMOTE_SERVER" _key_var="${MY_ID}_SSH_KEY" MIRROR_SSH_KEY="${!_key_var}" if [[ ! -f "$MIRROR_SSH_KEY" ]]; then error "SSH key $MIRROR_SSH_KEY not found — cannot reach $MIRROR" exit 1 fi # ── name -> template ────────────────────────────────────────────────────────────────────────── # Matched on the element. Filenames lie often enough that trusting them would push the # wrong container without saying so. xml_for_container() { local want="$1" f n for f in "$TEMPLATES_DIR"/*.xml; do [[ -f "$f" ]] || continue n=$(awk 'match($0,/([^<]+)<\/Name>/,a){print a[1];exit}' "$f") [[ "$n" == "$want" ]] && { echo "$f"; return 0; } done return 1 } remote_has_container() { timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" -o ConnectTimeout="$SSH_TIMEOUT" \ -o BatchMode=yes -o StrictHostKeyChecking=no root@"$MIRROR_IP" \ "docker inspect $(printf '%q' "$1") >/dev/null 2>&1" 2>/dev/null } remote_state_of() { timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" -o ConnectTimeout="$SSH_TIMEOUT" \ -o BatchMode=yes -o StrictHostKeyChecking=no root@"$MIRROR_IP" \ "docker inspect -f '{{.State.Status}}' $(printf '%q' "$1") 2>/dev/null" 2>/dev/null } OK=0; FAIL=0; SKIP=0 case "$MODE" in status) # Written as a cache as well as printed. The assistant's fallback_state block cannot afford an # SSH round trip per container mid-question, so it reads this file and reports its AGE — a stale # answer stated as stale is useful, stated as current it is the exact failure this feature # exists to prevent. _present="" _missing="" for c in "${COVERED[@]}"; do if remote_has_container "$c"; then _st=$(remote_state_of "$c") printf ' %-28s on %s (%s)\n' "$c" "$MIRROR" "$_st" _present+="\"$c\":\"${_st:-unknown}\"," OK=$((OK+1)) else printf ' %-28s MISSING on %s — docker start would fail\n' "$c" "$MIRROR" _missing+="\"$c\"," FAIL=$((FAIL+1)) fi done log "$ICON_FALLBACK Coverage present: $OK · missing: $FAIL" mkdir -p "$VV_CACHE_ROOT/api" 2>/dev/null || mkdir -p /tmp/varaverk/api 2>/dev/null _cache="${VV_CACHE_ROOT:-/tmp/varaverk}/api/fallback_presence.json" # Written atomically — a half-written cache read mid-question would report containers as # missing that are merely unparsed. printf '{"present":{%s},"missing":[%s],"partner":"%s","checked":%s}\n' \ "${_present%,}" "${_missing%,}" "$MIRROR" "$(date +%s)" > "$_cache.tmp" \ && mv -f "$_cache.tmp" "$_cache" [[ "$FAIL" -gt 0 ]] && exit 2 || exit 0 ;; push) # Networks first — a container whose network is absent is created and then cannot start, # which is the failure that read as "auth 0/8, arr 0/5" during onboarding. _nets=() for c in "${COVERED[@]}"; do x=$(xml_for_container "$c") || continue net=$(sed -n 's/.*\([^<]*\)<\/Network>.*/\1/p' "$x" 2>/dev/null | head -1) net="${net//[[:space:]]/}" # br* is host hardware. wg* is a WireGuard-backed bridge whose meaning does NOT travel: # recreating it on the partner as a plain bridge yields a network that exists, starts its # containers, and routes their traffic OUTSIDE the tunnel. ChannelTube rides wg0 here. case "$net" in ''|bridge|host|none|br[0-9]*) continue ;; wg[0-9]*) warn "$c uses $net — a WireGuard-backed network. NOT created on $MIRROR: a plain" warn " bridge of the same name would route its traffic outside the tunnel. Build the" warn " matching tunnel there first, or drop $c from coverage." continue ;; esac _seen=false for n in "${_nets[@]}"; do [[ "$n" == "$net" ]] && { _seen=true; break; }; done [[ "$_seen" == false ]] && _nets+=("$net") done for net in "${_nets[@]}"; do driver=$(timeout "${DOCKER_TIMEOUT:-30}" docker network inspect "$net" --format '{{.Driver}}' 2>/dev/null) if [[ "$driver" != "bridge" ]]; then warn "Network $net is '${driver:-absent}' here, not bridge — create it on $MIRROR by hand" continue fi if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would ensure network $net on $MIRROR" continue fi timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" -o ConnectTimeout="$SSH_TIMEOUT" \ -o BatchMode=yes -o StrictHostKeyChecking=no root@"$MIRROR_IP" \ "docker network inspect $(printf '%q' "$net") >/dev/null 2>&1 \ || docker network create --driver bridge $(printf '%q' "$net") >/dev/null" 2>/dev/null \ && log " network $net ready on $MIRROR" \ || warn " could not ensure network $net on $MIRROR" done for c in "${COVERED[@]}"; do x=$(xml_for_container "$c") || { warn "$c — no template in $TEMPLATES_DIR names it; skipped" SKIP=$((SKIP+1)); continue } if ! deploy_container_from_xml "$x" "$MIRROR_IP" "$MIRROR_SSH_KEY"; then error "$c — deploy failed" FAIL=$((FAIL+1)); continue fi if [[ "$DRY_RUN" == true ]]; then OK=$((OK+1)); continue; fi # Deployed containers must not run here. Stop, then re-inspect — a stop that did not take # is the one outcome that silently duplicates a live service against shared data. timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" -o ConnectTimeout="$SSH_TIMEOUT" \ -o BatchMode=yes -o StrictHostKeyChecking=no root@"$MIRROR_IP" \ "docker stop $(printf '%q' "$c") >/dev/null 2>&1" 2>/dev/null st=$(remote_state_of "$c") if [[ "$st" == "running" ]]; then error "$c is RUNNING on $MIRROR after deploy and would not stop — stop it there before continuing" FAIL=$((FAIL+1)) else log " $c deployed and ${st:-stopped} on $MIRROR ✅" OK=$((OK+1)) fi done log "$ICON_FALLBACK Push complete — deployed $OK · failed $FAIL · skipped $SKIP" [[ "$FAIL" -gt 0 ]] && exit 1 || exit 0 ;; remove) for c in "${COVERED[@]}"; do if ! remote_has_container "$c"; then log " $c not on $MIRROR — nothing to remove" SKIP=$((SKIP+1)); continue fi # Binds are read BEFORE the container goes — once it is removed there is nothing left to # enumerate, and a path list gathered afterwards would silently be empty. _binds=$(timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" -o ConnectTimeout="$SSH_TIMEOUT" \ -o BatchMode=yes -o StrictHostKeyChecking=no root@"$MIRROR_IP" \ "docker inspect --format '{{range .HostConfig.Binds}}{{println .}}{{end}}' $(printf '%q' "$c") 2>/dev/null \ | awk -F: '{print \$1}'" 2>/dev/null) _wipe=() while IFS= read -r _p; do [[ -z "$_p" ]] && continue # Only appdata, and never an appdata root. /mnt/user/appdata as a bind would make one # container removal delete every application on the partner. [[ "$_p" =~ ^/mnt/[^/]+/appdata[^/]*/.+ ]] || { [[ "$_p" =~ ^/mnt/[^/]+/appdata[^/]*/?$ ]] && \ warn " $c binds the appdata ROOT ($_p) — refusing to delete it" continue } _wipe+=("$_p") done <<< "$_binds" if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — would stop and remove $c on $MIRROR" for _p in "${_wipe[@]}"; do warn " DRY RUN — would delete appdata $_p on $MIRROR"; done OK=$((OK+1)); continue fi x=$(xml_for_container "$c") && xml_name=$(basename "$x") || xml_name="" if timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" -o ConnectTimeout="$SSH_TIMEOUT" \ -o BatchMode=yes -o StrictHostKeyChecking=no root@"$MIRROR_IP" \ "docker stop $(printf '%q' "$c") >/dev/null 2>&1; \ docker rm $(printf '%q' "$c") >/dev/null 2>&1; \ ${xml_name:+rm -f ${TEMPLATES_DIR}/$(printf '%q' "$xml_name");} \ ! docker inspect $(printf '%q' "$c") >/dev/null 2>&1" 2>/dev/null; then log " $c removed from $MIRROR ✅" for _p in "${_wipe[@]}"; do if timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" -o ConnectTimeout="$SSH_TIMEOUT" \ -o BatchMode=yes -o StrictHostKeyChecking=no root@"$MIRROR_IP" \ "rm -rf -- $(printf '%q' "$_p") && ! [ -e $(printf '%q' "$_p") ]" 2>/dev/null; then log " appdata deleted on $MIRROR: $_p" else warn " could not delete appdata on $MIRROR: $_p" FAIL=$((FAIL+1)) fi done OK=$((OK+1)) else error "$c — removal failed or it still exists on $MIRROR" FAIL=$((FAIL+1)) fi done log "$ICON_FALLBACK Remove complete — removed $OK · failed $FAIL · skipped $SKIP" [[ "$FAIL" -gt 0 ]] && exit 1 || exit 0 ;; esac