Files
Varaverk/Plugin/unraid/Partnership/containers.sh
T

767 lines
36 KiB
Bash
Executable File

#!/bin/bash
# ==============================================================================================
# ====================== Partnership — Unraid Container Adapter ================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Platform adapter providing container deploy/cleanup functions for the Unraid
# partnership system. Sourced (not executed) by partnership_onboard.sh and
# partnership_offboard.sh via:
# source "$SCRIPTS_ROOT/Plugin/$PLATFORM/Partnership/containers.sh"
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# Functions operate on variables from the calling script's scope:
# MIRROR, MIRROR_IP, MIRROR_SSH_KEY, SSH_TIMEOUT, DRY_RUN, SCRIPTS_ROOT
#
# Capabilities:
# - Docker container deployment from Unraid CA XML templates
# - Remote GPU type detection (cached per session — one SSH call per onboard)
# - Deployed stack tracking via _STACK_DEPLOYED / _STACK_FAILED counters
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Platform Adapter, Same Contract as adapter.sh
# Container deployment is Unraid-specific — CA XML templates, dockerMan paths, the
# templates-user directory. Confining it here means the partnership scripts contain no
# Unraid knowledge and a second platform is a new Partnership/containers.sh, not edits
# scattered through onboard and offboard.
#
# Deploy From the Template, Not a Copy
# Containers are created from the CA XML the operator already maintains, so a partnership
# deployment produces the same container the Unraid UI would. Hand-built docker run lines
# would drift from the template the moment anyone edited it in the UI.
#
# GPU Detection Cached Per Session
# The remote GPU type is probed once and reused. Onboarding deploys several containers and
# each would otherwise repeat the same SSH round-trip to learn an answer that cannot change
# mid-run.
#
# Count Outcomes, Do Not Abort
# Failures increment _STACK_FAILED rather than exiting. A stack deployment that fails on
# one container should report which one and continue — the caller owns whether a partial
# stack is acceptable, and it is the only side with the context to decide.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# No Root, No Lock, No detect_hosts — Deliberate
# Sourced by partnership_onboard.sh and partnership_offboard.sh, both of which already
# enforce root and hold their own strict locks. Re-checking here would be redundant, and
# taking a lock would deadlock against the caller's. Do not add them.
#
# Caller Scope Is the Contract
# Functions read MIRROR, MIRROR_IP, MIRROR_SSH_KEY, SSH_TIMEOUT, DRY_RUN and SCRIPTS_ROOT
# from the caller. That coupling is deliberate — it keeps one definition of who the mirror
# is — but it means these functions are only valid inside the partnership scripts and
# cannot be sourced standalone.
#
# DRY_RUN Honoured Throughout
# Every deploy and cleanup path checks the caller's DRY_RUN, so a dry-run onboard makes no
# remote container changes.
#
# SSH Timeouts on Every Remote Call
# All remote operations use the caller's SSH_TIMEOUT — an unreachable mirror cannot hang
# an onboard partway through a stack deployment.
#
# Template Existence Checked
# A missing CA XML is counted as a failure for that container rather than producing a
# container built from nothing.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# No config vars of its own. Inputs come from the calling script's scope (see above).
#
# Platform paths it owns:
#
# /boot/config/plugins/dockerMan/templates-user
# Unraid CA template directory. Source of every container definition deployed here.
#
# The container lists themselves live in host*.conf as HOST*_PARTNERSHIP_AUTH_STACK and
# HOST*_PARTNERSHIP_ARR_STACK — read by the partnership scripts, passed in as arguments.
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# None — sourced, never executed:
#
# source "$SCRIPTS_ROOT/Plugin/$PLATFORM/Partnership/containers.sh"
#
# No argument parsing and no flags. Dry-run behaviour comes from the caller's DRY_RUN.
#
# ==============================================================================================
TEMPLATES_DIR="/boot/config/plugins/dockerMan/templates-user"
_STACK_DEPLOYED=0
_STACK_FAILED=0
_REMOTE_GPU_TYPE="" # cached after first detection
# ==============================================================================================
# ── Detect GPU type on a remote host ─────────────────────────────────────────────────────────
#
# Returns one of: nvidia | intel | amd | dri | none
# nvidia — /dev/nvidia0 present (NVIDIA driver loaded)
# intel — /dev/dri present, vendor 0x8086
# amd — /dev/dri present, vendor 0x1002 (also exposes /dev/kfd)
# dri — /dev/dri present but vendor unreadable
# none — no GPU device found
#
# Result is cached in _REMOTE_GPU_TYPE for the session — SSH'd once per onboard run.
# ==============================================================================================
detect_remote_gpu() {
local remote_ip="$1" ssh_key="$2"
if [[ -n "$_REMOTE_GPU_TYPE" ]]; then
echo "$_REMOTE_GPU_TYPE"
return 0
fi
local result
result=$(timeout 10 ssh -i "$ssh_key" \
-o ConnectTimeout=10 -o BatchMode=yes root@"$remote_ip" '
if [ -c /dev/nvidia0 ]; then
echo nvidia
elif [ -d /dev/dri ]; then
vendor=""
for f in /sys/class/drm/card*/device/vendor; do
[ -f "$f" ] && { vendor=$(cat "$f" 2>/dev/null); break; }
done
case "$vendor" in
0x8086) echo intel ;;
0x1002) echo amd ;;
*) echo dri ;;
esac
else
echo none
fi
' 2>/dev/null)
_REMOTE_GPU_TYPE="${result:-none}"
echo "$_REMOTE_GPU_TYPE"
}
# ==============================================================================================
# ── Rewrite GPU config in an XML for a target GPU type ───────────────────────────────────────
#
# Called when deploying to a remote whose GPU differs from the owner's. Takes the owner's
# XML (NVIDIA-configured) and rewrites it for the remote's hardware without modifying the
# original on disk.
#
# Returns the path to a temp file — caller must clean it up.
# Returns the original path unchanged if the XML has no NVIDIA markers (not GPU-aware).
#
# Handles both GPU config styles:
# New: --gpus "device=GPU-UUID" in ExtraParams (current approach)
# Old: --runtime=nvidia in ExtraParams + NVIDIA_VISIBLE_DEVICES Variable Config
#
# Transforms applied:
# nvidia → nvidia: normalise device UUID to "all" (both styles)
# nvidia → intel/dri: strip NVIDIA params, inject /dev/dri Device Config
# nvidia → amd: strip NVIDIA params, inject /dev/dri + /dev/kfd Device Configs
# nvidia → none: strip NVIDIA params, no device added
# ==============================================================================================
transform_xml_for_gpu() {
local src_xml="$1" gpu_type="$2"
# Detect GPU-aware XMLs — new style (--gpus "device=) or old style (--runtime=nvidia)
if ! grep -qE '--gpus[[:space:]]+"device=|--runtime=nvidia|NVIDIA_VISIBLE_DEVICES' "$src_xml" 2>/dev/null; then
echo "$src_xml"
return 0
fi
local tmp_xml
tmp_xml=$(mktemp /tmp/vv_xml_gpu_XXXXXX.xml)
case "$gpu_type" in
nvidia)
# Normalise to --gpus all (new style) or NVIDIA_VISIBLE_DEVICES=all (old style)
sed \
-e 's/--gpus[[:space:]]*"device=[^"]*"/--gpus all/g' \
-e 's/\(Target="NVIDIA_VISIBLE_DEVICES"[^>]*>\)[^<]*/\1all/' \
"$src_xml" > "$tmp_xml"
;;
intel|dri)
# Strip NVIDIA params, add /dev/dri device
sed \
-e 's/--gpus[[:space:]]*"device=[^"]*"[[:space:]]*//' \
-e 's/--runtime=nvidia[[:space:]]*//' \
-e '/Target="NVIDIA_VISIBLE_DEVICES"/d' \
"$src_xml" > "$tmp_xml"
sed -i 's|</Container>| <Config Name="GPU" Target="/dev/dri" Default="/dev/dri" Mode="rwm" Description="" Type="Device" Display="always" Required="false" Mask="false">/dev/dri</Config>\n</Container>|' "$tmp_xml"
;;
amd)
# AMD needs /dev/dri for VA-API and /dev/kfd for ROCm/OpenCL
sed \
-e 's/--gpus[[:space:]]*"device=[^"]*"[[:space:]]*//' \
-e 's/--runtime=nvidia[[:space:]]*//' \
-e '/Target="NVIDIA_VISIBLE_DEVICES"/d' \
"$src_xml" > "$tmp_xml"
sed -i 's|</Container>| <Config Name="GPU" Target="/dev/dri" Default="/dev/dri" Mode="rwm" Description="" Type="Device" Display="always" Required="false" Mask="false">/dev/dri</Config>\n <Config Name="GPU KFD" Target="/dev/kfd" Default="/dev/kfd" Mode="rwm" Description="" Type="Device" Display="always" Required="false" Mask="false">/dev/kfd</Config>\n</Container>|' "$tmp_xml"
;;
none)
# No GPU — strip all GPU params, no device added
sed \
-e 's/--gpus[[:space:]]*"device=[^"]*"[[:space:]]*//' \
-e 's/--runtime=nvidia[[:space:]]*//' \
-e '/Target="NVIDIA_VISIBLE_DEVICES"/d' \
"$src_xml" > "$tmp_xml"
;;
esac
echo "$tmp_xml"
}
# ==============================================================================================
# ── Wait for a container on the remote to be healthy/running ─────────────────────────────────
#
# Polls docker inspect on the remote. Prefers the health status if a healthcheck is defined;
# falls back to the running state. Non-fatal after timeout — some containers take time to
# fully initialize but the deploy itself succeeded.
# ==============================================================================================
wait_for_container_healthy() {
local name="$1" remote_ip="$2" ssh_key="$3"
local max_wait=60 interval=5 elapsed=0
[[ "$DRY_RUN" == true ]] && return 0
log " Waiting for $name to be ready..."
while (( elapsed < max_wait )); do
local status
status=$(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$remote_ip" \
"h=\$(docker inspect --format '{{.State.Health.Status}}' '$name' 2>/dev/null)
r=\$(docker inspect --format '{{.State.Running}}' '$name' 2>/dev/null)
echo \${h:-\$r}" 2>/dev/null)
case "$status" in
healthy|true)
echo " $name ready ✅"
return 0
;;
*)
sleep "$interval"
(( elapsed += interval ))
;;
esac
done
warn " $name not confirmed healthy after ${max_wait}s — continuing (may affect dependents)"
return 0
}
# ==============================================================================================
# ── Deploy a container from a local Unraid CA XML template to a remote host ──────────────────
#
# Parses Port / Path / Variable Config entries from the Unraid XML, SCPs the template and a
# self-contained deploy script to the remote, executes it, then cleans up both sides.
# Credentials are never passed as SSH command-line args — they stay in the SCPed script.
# ==============================================================================================
deploy_container_from_xml() {
local xml_file="$1" remote_ip="$2" ssh_key="$3"
local xml_name
xml_name=$(basename "$xml_file")
# GPU transform — rewrite GPU params for the remote's hardware before parsing or SCP.
# detect_remote_gpu is cached after the first SSH call.
local _gpu_type _transformed_xml _gpu_tmp=""
_gpu_type=$(detect_remote_gpu "$remote_ip" "$ssh_key")
_transformed_xml=$(transform_xml_for_gpu "$xml_file" "$_gpu_type")
[[ "$_transformed_xml" != "$xml_file" ]] && _gpu_tmp="$_transformed_xml"
xml_file="$_transformed_xml"
local name repo network extra privileged webui icon
# WebUI and Icon become Unraid labels below — see the docker create line for why.
webui=$( awk 'match($0,/<WebUI>([^<]*)<\/WebUI>/, a){print a[1];exit}' "$xml_file")
icon=$( awk 'match($0,/<Icon>([^<]*)<\/Icon>/, a){print a[1];exit}' "$xml_file")
name=$( awk 'match($0,/<Name>([^<]+)<\/Name>/, a){print a[1];exit}' "$xml_file")
repo=$( awk 'match($0,/<Repository>([^<]+)<\/Repository>/,a){print a[1];exit}' "$xml_file")
network=$( awk 'match($0,/<Network>([^<]+)<\/Network>/, a){print a[1];exit}' "$xml_file")
extra=$( awk 'match($0,/<ExtraParams>([^<]*)<\/ExtraParams>/,a){print a[1];exit}' "$xml_file")
privileged=$( awk 'match($0,/<Privileged>([^<]+)<\/Privileged>/,a){print a[1];exit}' "$xml_file")
if [[ -z "$name" || -z "$repo" ]]; then
rm -f "$_gpu_tmp"
warn " Cannot parse Name/Repository from $xml_name — skipping"
return 1
fi
[[ -n "$_gpu_tmp" ]] && log " GPU: ${_gpu_type} (rewritten from owner NVIDIA config)"
log "Deploying $name..."
if [[ "$DRY_RUN" == false ]]; then
timeout "$SSH_TIMEOUT" scp -i "$ssh_key" -o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes \
"$xml_file" "root@${remote_ip}:${TEMPLATES_DIR}/${xml_name}" 2>/dev/null || {
rm -f "$_gpu_tmp"
warn " SCP failed for $xml_name — skipping $name"
return 1
}
else
warn " DRY RUN — would SCP $xml_name$MIRROR:${TEMPLATES_DIR}/"
fi
local tmp_script
tmp_script=$(mktemp /tmp/deploy_XXXXXX.sh)
chmod 600 "$tmp_script"
{
echo "#!/bin/bash"
echo "set -e"
echo ""
printf "docker pull %q 2>/dev/null || true\n" "$repo"
printf "docker stop %q 2>/dev/null || true\n" "$name"
printf "docker rm %q 2>/dev/null || true\n" "$name"
echo ""
# Unraid's Docker Manager decides what it owns by label, not by template presence. The
# XML is SCPed to the mirror's templates-user above, but without these three the WebGUI
# lists the container as third-party: no Edit button, no WebUI link, no icon — the
# operator can see it running and cannot do anything with it.
#
# The values go in verbatim, placeholders and all: Unraid stores the literal
# "http://[IP]:[PORT:8989]/..." form in the label and substitutes at render time, so
# resolving them here would produce a link that stops being right the moment the
# container's port mapping changes.
printf "docker create --name %q --restart=unless-stopped" "$name"
printf " --label %q" "net.unraid.docker.managed=dockerman"
[[ -n "$webui" ]] && printf " --label %q" "net.unraid.docker.webui=${webui}"
[[ -n "$icon" ]] && printf " --label %q" "net.unraid.docker.icon=${icon}"
[[ -n "$network" ]] && printf " --network=%q" "$network"
[[ "$privileged" == "true" ]] && printf " --privileged"
[[ -n "$extra" ]] && printf " %s" "$extra"
# Port mappings → -p host:container/proto
awk '/Type="Port"/ {
match($0, /Target="([^"]+)"/, t)
match($0, /Mode="([^"]+)"/, m)
match($0, />([^<]+)<\/Config>/, v)
if (t[1] != "" && v[1] != "") {
proto = (m[1] == "udp") ? "udp" : "tcp"
printf " -p %s:%s/%s", v[1], t[1], proto
}
}' "$xml_file"
# Volume mappings → -v 'host:container:mode'
awk 'BEGIN{q=sprintf("%c",39)} /Type="Path"/ {
match($0, /Target="([^"]+)"/, t)
match($0, /Mode="([^"]+)"/, m)
match($0, />([^<]+)<\/Config>/, v)
if (t[1] != "" && v[1] != "") {
mode = (m[1] == "ro") ? "ro" : "rw"
printf " -v %s%s:%s:%s%s", q, v[1], t[1], mode, q
}
}' "$xml_file"
# Environment variables → -e 'KEY=VALUE' (single-quoted to protect $ and special chars)
awk 'BEGIN{q=sprintf("%c",39)} /Type="Variable"/ {
match($0, /Target="([^"]+)"/, t)
match($0, />([^<]+)<\/Config>/, v)
if (t[1] != "" && v[1] != "") {
printf " -e %s%s=%s%s", q, t[1], v[1], q
}
}' "$xml_file"
printf " %q\n" "$repo"
echo ""
printf "docker start %q && echo 'deployed:%s'\n" "$name" "$name"
} > "$tmp_script"
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would deploy $name on $MIRROR"
rm -f "$tmp_script" "$_gpu_tmp"
return 0
fi
local remote_script="/tmp/deploy_${name//[^a-zA-Z0-9_]/_}.sh"
if timeout "$SSH_TIMEOUT" scp -i "$ssh_key" -o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes \
"$tmp_script" "root@${remote_ip}:${remote_script}" 2>/dev/null && \
timeout 120 ssh -i "$ssh_key" -o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$remote_ip" \
"bash '$remote_script' 2>&1; rc=\$?; rm -f '$remote_script'; exit \$rc" 2>/dev/null | \
grep -q "deployed:${name}"; then
echo " $name deployed ✅"
rm -f "$tmp_script" "$_gpu_tmp"
return 0
else
warn " $name deployment failed — check $MIRROR manually"
rm -f "$tmp_script" "$_gpu_tmp"
return 1
fi
}
# ==============================================================================================
# ── Deploy a stack of Unraid CA XMLs to the mirror ───────────────────────────────────────────
#
# Sets globals _STACK_DEPLOYED and _STACK_FAILED rather than printing to stdout.
# Health-checks database deps (Mariadb/Redis/Postgres) between batches so dependents
# (e.g. Authelia) start cleanly.
# ==============================================================================================
deploy_xml_stack() {
local -n xml_array_ref="$1"
_STACK_DEPLOYED=0
_STACK_FAILED=0
for xml_name in "${xml_array_ref[@]}"; do
local xml_file="${TEMPLATES_DIR}/${xml_name}"
if [[ ! -f "$xml_file" ]]; then
warn "$xml_name not found in $TEMPLATES_DIR — skipping"
(( _STACK_FAILED++ ))
continue
fi
local cname
cname=$(awk 'match($0,/<Name>([^<]+)<\/Name>/,a){print a[1];exit}' "$xml_file")
if deploy_container_from_xml "$xml_file" "$MIRROR_IP" "$MIRROR_SSH_KEY"; then
(( _STACK_DEPLOYED++ ))
if [[ -n "$cname" ]] && echo "$cname" | grep -qiE 'mariadb|redis|postgres|mysql'; then
wait_for_container_healthy "$cname" "$MIRROR_IP" "$MIRROR_SSH_KEY"
fi
else
(( _STACK_FAILED++ ))
fi
done
}
# ==============================================================================================
# ── Ensure the networks our pushed templates reference exist on a remote host ─────────────────
#
# The owner deploys the mirror's containers from the owner's own XMLs, and those XMLs name a
# network. `docker create` fails outright if that network is missing, so the network has to
# exist on the mirror before any stack is deployed.
#
# This used to be left entirely to docker_network_connect.sh running on the mirror, which
# iterates the *mirror's* NETWORK_CONNECT_NETWORKS. host.conf.template ships that array with its
# only entry commented out, so on a fresh node it is empty — nothing was created, and every
# container in both stacks was created against a network that did not exist and could never
# start. Twelve containers stuck in `Created`, reported as "0 deployed, 8 failed" and
# "0 deployed, 5 failed" as though each container had its own problem.
#
# The owner knows what it is about to push, so it derives the requirement from the templates
# rather than trusting the mirror's conf to have been filled in.
#
# Only bridge networks are created. br0 and friends are ipvlan/macvlan bound to real host
# hardware — the parent interface cannot be inferred from here, and guessing one would attach
# the mirror's containers to the wrong segment.
#
# Usage: ensure_stack_networks_on_remote "$MIRROR_IP" "$MIRROR_SSH_KEY"
# ==============================================================================================
# ==============================================================================================
# ── Container names this onboard deploys, read from the templates it deploys them from ────────
#
# Echoes one name per line. The <Name> element is the same value deploy_xml_stack() passes to
# `docker create --name`, so this is the deployed set by construction rather than by asking the
# mirror what it ended up with — which would also pick up whatever the mirror already ran.
#
# Usage: mapfile -t names < <(deployed_stack_container_names)
# ==============================================================================================
deployed_stack_container_names() {
local -a xml_names=()
[[ ${#PARTNERSHIP_AUTH_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_AUTH_STACK[@]}")
[[ ${#PARTNERSHIP_ARR_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_ARR_STACK[@]}")
[[ ${#PARTNERSHIP_SERVICES_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_SERVICES_STACK[@]}")
local xml_name xml_file cname
for xml_name in "${xml_names[@]}"; do
[[ -z "$xml_name" ]] && continue
xml_file="${TEMPLATES_DIR}/${xml_name}"
[[ -f "$xml_file" ]] || continue
cname=$(awk 'match($0,/<Name>([^<]+)<\/Name>/,a){print a[1];exit}' "$xml_file")
[[ -n "$cname" ]] && echo "$cname"
done
}
ensure_stack_networks_on_remote() {
local remote_ip="$1" ssh_key="$2"
local -a xml_names=() nets=()
[[ ${#PARTNERSHIP_AUTH_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_AUTH_STACK[@]}")
[[ ${#PARTNERSHIP_ARR_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_ARR_STACK[@]}")
[[ ${#PARTNERSHIP_SERVICES_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_SERVICES_STACK[@]}")
local xml_name xml_file net n seen
for xml_name in "${xml_names[@]}"; do
[[ -z "$xml_name" ]] && continue
xml_file="${TEMPLATES_DIR}/${xml_name}"
[[ -f "$xml_file" ]] || continue
net=$(sed -n 's/.*<Network>\([^<]*\)<\/Network>.*/\1/p' "$xml_file" 2>/dev/null | head -1)
net="${net//[[:space:]]/}"
# Built-ins exist on every host; br* is host hardware, handled above.
case "$net" in ''|bridge|host|none|br[0-9]*) continue ;; esac
seen=false
for n in "${nets[@]}"; do [[ "$n" == "$net" ]] && { seen=true; break; }; done
[[ "$seen" == false ]] && nets+=("$net")
done
if [[ ${#nets[@]} -eq 0 ]]; then
log "No custom networks referenced by the pushed templates"
return 0
fi
local rc=0 driver
for net in "${nets[@]}"; do
driver=$(timeout "${DOCKER_TIMEOUT:-30}" docker network inspect "$net" \
--format '{{.Driver}}' 2>/dev/null)
if [[ -z "$driver" ]]; then
warn "$net is referenced by a pushed template but does not exist here either — skipping"
rc=1
continue
fi
if [[ "$driver" != "bridge" ]]; then
warn "$net is $driver here, not bridge — create it on $MIRROR by hand, its parent interface is host-specific"
rc=1
continue
fi
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would ensure network '$net' (bridge) exists on $MIRROR"
continue
fi
if timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes -o StrictHostKeyChecking=no \
root@"$remote_ip" \
"docker network inspect $(printf '%q' "$net") >/dev/null 2>&1 \
|| docker network create --driver bridge $(printf '%q' "$net") >/dev/null" 2>/dev/null; then
echo " network $net (bridge) ready on $MIRROR ✅"
else
warn " could not ensure network $net on $MIRROR — its containers will not start"
rc=1
fi
done
return "$rc"
}
# ==============================================================================================
# ── Remove owner-deployed containers from a remote host ──────────────────────────────────────
#
# Uses PARTNERSHIP_AUTH_STACK + PARTNERSHIP_ARR_STACK (owner's conf) to derive container
# names from local XML templates. SSHes to remote to stop, remove, and delete appdata.
# Appdata paths collected via docker inspect before removal. Safety gate: only
# /mnt/*/appdata* paths are deleted.
# ==============================================================================================
cleanup_deployed_stack_on_remote() {
local remote_ip="$1" ssh_key="$2"
local _rc=0
local -a xml_names=()
[[ ${#PARTNERSHIP_AUTH_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_AUTH_STACK[@]}")
[[ ${#PARTNERSHIP_ARR_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_ARR_STACK[@]}")
[[ ${#PARTNERSHIP_SERVICES_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_SERVICES_STACK[@]}")
if [[ ${#xml_names[@]} -eq 0 ]]; then
log "No auth/arr/services stack arrays configured — skipping deployed stack cleanup"
return 0
fi
log "Removing owner-deployed containers (auth/arr/services stacks) from $MIRROR..."
for xml_name in "${xml_names[@]}"; do
[[ -z "$xml_name" ]] && continue
local xml_file="${TEMPLATES_DIR}/${xml_name}"
if [[ ! -f "$xml_file" ]]; then
warn " $xml_name not found in local $TEMPLATES_DIR — skipping"
continue
fi
local cname
cname=$(awk 'match($0,/<Name>([^<]+)<\/Name>/,a){print a[1];exit}' "$xml_file")
[[ -z "$cname" ]] && continue
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would stop + rm $cname on $MIRROR"
warn " DRY RUN — would delete appdata for $cname on $MIRROR"
continue
fi
local appdata_paths
appdata_paths=$(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$remote_ip" \
"docker inspect --format '{{range .HostConfig.Binds}}{{println .}}{{end}}' '$cname' 2>/dev/null \
| awk -F: '{print \$1}' | grep '^/mnt/.*/appdata'" 2>/dev/null)
timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$remote_ip" \
"docker stop '$cname' >/dev/null 2>&1
docker rm '$cname' >/dev/null 2>&1 && echo removed" 2>/dev/null | \
grep -q removed && \
echo " $cname removed from $MIRROR ✅" || \
log " $cname not found on $MIRROR — skipping"
while IFS= read -r path; do
[[ -z "$path" ]] && continue
if timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$remote_ip" \
"rm -rf '$path' && echo removed" 2>/dev/null | grep -q removed; then
echo " Appdata removed on $MIRROR: $path ✅"
else
warn " Failed to remove appdata on $MIRROR: $path"
_rc=1
fi
done <<< "$appdata_paths"
done
# Only appdata failures are counted. The container branch above cannot tell "removal failed"
# from "already gone" — both produce no `removed` echo — and an offboard re-run on a
# half-finished teardown is a normal case, so treating that as failure would cry wolf.
return "$_rc"
}
# ==============================================================================================
# ── Remove owner-deployed containers locally (mirror-initiated offboard) ─────────────────────
#
# SSHes to owner to read PARTNERSHIP_AUTH_STACK + PARTNERSHIP_ARR_STACK, then uses the
# local templates-user/ copies (SCPed there during onboard) to get container names and
# appdata paths. Appdata collected before removal. Skips gracefully if owner unreachable.
# ==============================================================================================
cleanup_deployed_stack_locally() {
local owner_ip="$1" ssh_key="$2"
local -a xml_names=()
if [[ -n "$owner_ip" ]]; then
local -a auth_arr arr_arr
mapfile -t auth_arr < <(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$owner_ip" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${PARTNERSHIP_AUTH_STACK[@]:-}\"" 2>/dev/null | grep -v '^$')
mapfile -t arr_arr < <(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$owner_ip" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${PARTNERSHIP_ARR_STACK[@]:-}\"" 2>/dev/null | grep -v '^$')
local -a svc_arr
mapfile -t svc_arr < <(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$owner_ip" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${PARTNERSHIP_SERVICES_STACK[@]:-}\"" 2>/dev/null | grep -v '^$')
xml_names=("${auth_arr[@]}" "${arr_arr[@]}" "${svc_arr[@]}")
fi
if [[ ${#xml_names[@]} -eq 0 ]]; then
log "Could not read deployed stack from owner — skipping auth/arr/services cleanup"
return 0
fi
local _local_short
_local_short=$(derive_short_name "${LOCAL_SERVER_NAME:-}")
log "Removing owner-deployed containers (auth/arr/services stacks) locally..."
for xml_name in "${xml_names[@]}"; do
[[ -z "$xml_name" ]] && continue
local xml_file="${TEMPLATES_DIR}/${xml_name}"
if [[ ! -f "$xml_file" ]]; then
warn " $xml_name not found locally — skipping"
continue
fi
local cname
cname=$(awk 'match($0,/<Name>([^<]+)<\/Name>/,a){print a[1];exit}' "$xml_file")
[[ -z "$cname" ]] && continue
if [[ -n "$_local_short" && "$cname" == *"$_local_short"* ]]; then
warn " $cname contains local server identity ($_local_short) — skipping to protect local containers"
continue
fi
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would stop + rm $cname"
warn " DRY RUN — would delete appdata for $cname"
continue
fi
local appdata_paths=""
if timeout "${DOCKER_TIMEOUT:-30}" docker inspect "$cname" >/dev/null 2>&1; then
appdata_paths=$(docker inspect \
--format '{{range .HostConfig.Binds}}{{println .}}{{end}}' \
"$cname" 2>/dev/null | awk -F: '{print $1}' | grep '^/mnt/.*/appdata')
timeout "${DOCKER_TIMEOUT:-30}" docker stop "$cname" >/dev/null 2>&1 || true
_PM_TRAP_STOPPED+=("$cname")
timeout "${DOCKER_TIMEOUT:-30}" docker rm "$cname" >/dev/null 2>&1 && \
echo " $cname removed ✅" || warn " $cname rm failed"
else
log " $cname not found locally — skipping"
fi
while IFS= read -r path; do
[[ -z "$path" ]] && continue
rm -rf "$path" && echo " Appdata removed: $path ✅" || warn " Failed to remove: $path"
done <<< "$appdata_paths"
done
}
# ==============================================================================================
# ── Reconfigure a container's WebUI on the remote server ─────────────────────────────────────
# ==============================================================================================
reconfigure_webui() {
local container="$1" port="$2" target_ip="$3"
local ssh_key="$4" remote_ip="$5" label="${6:-remote}"
log "Reconfiguring $container WebUI → ${target_ip}:${port} on $label..."
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would reconfigure $container WebUI to http://${target_ip}:${port}/"
return 0
fi
local template
template=$(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$remote_ip" \
"grep -rl '<WebUI>' '$TEMPLATES_DIR/' 2>/dev/null | \
xargs grep -l '\"$container\"' 2>/dev/null | head -1" 2>/dev/null)
if [[ -z "$template" ]]; then
warn "$container template not found on $label — WebUI needs manual reconfiguration"
return 1
fi
timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$remote_ip" \
"sed -i 's|<WebUI>.*</WebUI>|<WebUI>http://${target_ip}:${port}/</WebUI>|g' '$template'" \
2>/dev/null && \
echo "$container → http://${target_ip}:${port}/ ✅" || {
error "Failed to reconfigure $container WebUI on $label"
return 1
}
}
# ==============================================================================================
# ── Reconfigure local auth WebUIs to target IP ───────────────────────────────────────────────
# ==============================================================================================
reconfigure_local_webuis() {
local target_ip="$1"
log "Reconfiguring local auth WebUIs → ${target_ip}..."
local failures=0
for entry in "${PARTNERSHIP_AUTH_WEBUIS[@]}"; do
[[ -z "$entry" ]] && continue
local container="${entry%%|*}"
local port="${entry##*|}"
local template
template=$(grep -rl '<WebUI>' "$TEMPLATES_DIR/" 2>/dev/null | \
xargs grep -l "\"$container\"" 2>/dev/null | head -1)
if [[ -z "$template" ]]; then
warn "$container template not found locally"
(( failures++ ))
continue
fi
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would reconfigure $container → http://${target_ip}:${port}/"
continue
fi
sed -i "s|<WebUI>.*</WebUI>|<WebUI>http://${target_ip}:${port}/</WebUI>|g" \
"$template" 2>/dev/null && \
echo "$container → http://${target_ip}:${port}/ ✅" || \
{ error "Failed to reconfigure $container"; (( failures++ )); }
done
return $failures
}