Codebase-wide audit pass: fixed real bugs (SSH hangs missing BatchMode, local-outside-function no-ops, variable name collisions, a truncated ratio calc, wrong state-dir path, DARK vs NO_INTERNET drift, and more), then pulled logic that was duplicated across multiple scripts — arr cleanup safety gates, docker restart ordering, container maintenance stop/restart, watchdog state-file helpers, partnership role resolution, cert expiry checks, remote node discovery, and TMDB discovery scoring — into common.sh so each now has a single implementation.
311 lines
14 KiB
Bash
Executable File
311 lines
14 KiB
Bash
Executable File
#!/bin/bash
|
|
# ==============================================================================================
|
|
# ================================= Network Watchdog ===========================================
|
|
# ==============================================================================================
|
|
#
|
|
# PURPOSE
|
|
# ─────────────────────────────────────────────────────────────────────────────
|
|
# Services-layer connectivity — checks that the outside world can actually reach
|
|
# what it needs to reach. Silent when everything is reachable. Only fires when
|
|
# something in the connectivity chain has broken.
|
|
#
|
|
# ==============================================================================================
|
|
# OPERATIONAL MODEL
|
|
# ==============================================================================================
|
|
#
|
|
# Check 1 — Internet Connectivity
|
|
# curl external endpoint → fail = alert + skip all remaining checks.
|
|
# Internet down means DDNS and NPM checks would false-positive — gating prevents noise.
|
|
#
|
|
# Check 2 — DDNS (Cloudflare)
|
|
# Public IP via ifconfig.me vs DNS record via dig @1.1.1.1.
|
|
# Match → pass (silent). Mismatch → restart DDNS container + notify.
|
|
# Container restart triggers an immediate Cloudflare record update.
|
|
#
|
|
# Check 3 — Tailscale
|
|
# tailscale status → Running → pass (silent). Not running → notify.
|
|
# Notify only — no restart attempt. Tailscale state issues warrant human review.
|
|
#
|
|
# Check 4 — NPM Proxy (external check)
|
|
# curl external URL → 2-strike system before restarting NginxProxyManager.
|
|
# Strike 1: warn + notify. Strike 2: restart NPM + notify + clear strikes.
|
|
# Strikes auto-clear when the external URL becomes reachable again.
|
|
# External check — verifies the full stack (DNS → NPM → backend), not just NPM running.
|
|
#
|
|
# ==============================================================================================
|
|
# OPERATIONAL SAFEGUARDS
|
|
# ==============================================================================================
|
|
#
|
|
# Root Required
|
|
# docker restart requires root.
|
|
#
|
|
# Single Instance Lock
|
|
# acquire_lock prevents concurrent runs.
|
|
#
|
|
# Internet Gates All Checks
|
|
# If internet is down, DDNS and NPM checks are skipped — no cascade of false positives.
|
|
#
|
|
# Strike Before Acting on NPM
|
|
# Single curl failure could be transient DNS hiccup or CDN blip.
|
|
# Two consecutive failures confirms NPM is the problem.
|
|
#
|
|
# ==============================================================================================
|
|
# CONFIGURATION
|
|
# ==============================================================================================
|
|
#
|
|
# master.conf
|
|
#
|
|
# NETWORK_WATCHDOG_ENABLED toggle entire watchdog (default: true)
|
|
# NETWORK_WATCHDOG_INTERNET_URL endpoint for internet reachability check
|
|
# NETWORK_WATCHDOG_INTERNET_TIMEOUT curl timeout in seconds for internet check
|
|
# NETWORK_WATCHDOG_CHECK_TAILSCALE toggle tailscale check (default: true)
|
|
# NETWORK_WATCHDOG_NPM_TIMEOUT curl timeout for NPM external check
|
|
# NETWORK_WATCHDOG_NPM_STRIKE_LIMIT consecutive failures before NPM restart
|
|
# NETWORK_WATCHDOG_NPM_STATE_FILE strike count persistence (/tmp — resets on reboot)
|
|
#
|
|
# host*.conf (host-specific)
|
|
#
|
|
# HOST*_NETWORK_WATCHDOG_DDNS_DOMAIN domain to resolve and compare to public IP
|
|
# HOST*_NETWORK_WATCHDOG_DDNS_CONTAINER container to restart on DDNS mismatch
|
|
# HOST*_NETWORK_WATCHDOG_NPM_URL external URL to test full proxy stack
|
|
#
|
|
# ==============================================================================================
|
|
# RUNTIME MODES
|
|
# ==============================================================================================
|
|
#
|
|
# network_watchdog.sh
|
|
# Run all connectivity checks. Silent when healthy.
|
|
#
|
|
# network_watchdog.sh --dry-run
|
|
# Run all checks without restarting any containers.
|
|
#
|
|
# network_watchdog.sh --status
|
|
# Show configuration, current public IP, DNS record, NPM strike state.
|
|
#
|
|
# network_watchdog.sh --log
|
|
# Verbose output — show each check result even when passing.
|
|
#
|
|
# ==============================================================================================
|
|
|
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
|
|
source "$SCRIPT_DIR/../../load_config.sh"
|
|
|
|
parse_args "$@"
|
|
|
|
# ==============================================================================================
|
|
# ━━━ Resolve host-specific config ━━━
|
|
# ==============================================================================================
|
|
_ddns_domain_var="${MY_ID}_NETWORK_WATCHDOG_DDNS_DOMAIN"
|
|
_ddns_container_var="${MY_ID}_NETWORK_WATCHDOG_DDNS_CONTAINER"
|
|
_npm_url_var="${MY_ID}_NETWORK_WATCHDOG_NPM_URL"
|
|
|
|
DDNS_DOMAIN="${!_ddns_domain_var:-}"
|
|
DDNS_CONTAINER="${!_ddns_container_var:-}"
|
|
NPM_URL="${!_npm_url_var:-}"
|
|
|
|
# ==============================================================================================
|
|
# ━━━ Setup ━━━
|
|
# ==============================================================================================
|
|
if [[ "$EUID" -ne 0 ]]; then
|
|
error "Must be run as root"
|
|
exit 1
|
|
fi
|
|
|
|
|
|
acquire_lock
|
|
|
|
detect_hosts
|
|
|
|
[[ "${NETWORK_WATCHDOG_ENABLED:-true}" != "true" ]] && echo "Network watchdog disabled" && exit 0
|
|
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no containers will be restarted"
|
|
|
|
log "$ICON_GEAR Config: internet=${NETWORK_WATCHDOG_INTERNET_URL:-https://1.1.1.1} timeout=${NETWORK_WATCHDOG_INTERNET_TIMEOUT:-5}s tailscale=${NETWORK_WATCHDOG_CHECK_TAILSCALE:-true} npm-strikes=${NETWORK_WATCHDOG_NPM_STRIKE_LIMIT:-2}"
|
|
log "$ICON_NET DDNS: ${DDNS_DOMAIN:-not configured} → ${DDNS_CONTAINER:-no container} NPM: ${NPM_URL:-not configured}"
|
|
|
|
touch "${NETWORK_WATCHDOG_NPM_STATE_FILE}" 2>/dev/null
|
|
|
|
# ━━━ Strike helpers — wrap common.sh's wd_state_get()/wd_state_set() ━━━
|
|
get_strikes() { wd_state_get "$1" "$2"; }
|
|
set_strikes() { wd_state_set "$1" "$2" "$3"; }
|
|
|
|
# ==============================================================================================
|
|
# ━━━ Status ━━━
|
|
# ==============================================================================================
|
|
if [[ "$SHOW_STATUS" == true ]]; then
|
|
echo ""
|
|
echo "━━━━━ $ICON_SUMMARY NETWORK WATCHDOG STATUS ━━━━━"
|
|
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
|
|
echo "$ICON_GEAR Internet URL: ${NETWORK_WATCHDOG_INTERNET_URL:-https://1.1.1.1}"
|
|
echo "$ICON_GEAR DDNS domain: ${DDNS_DOMAIN:-not configured}"
|
|
echo "$ICON_GEAR DDNS container: ${DDNS_CONTAINER:-not configured}"
|
|
echo "$ICON_GEAR Tailscale: ${NETWORK_WATCHDOG_CHECK_TAILSCALE:-true}"
|
|
echo "$ICON_GEAR NPM URL: ${NPM_URL:-not configured}"
|
|
echo "$ICON_GEAR NPM strikes: $(get_strikes "npm" "${NETWORK_WATCHDOG_NPM_STATE_FILE}") / ${NETWORK_WATCHDOG_NPM_STRIKE_LIMIT:-2}"
|
|
echo ""
|
|
echo "── Current State ──"
|
|
|
|
if curl -sf --max-time "${NETWORK_WATCHDOG_INTERNET_TIMEOUT:-5}" \
|
|
"${NETWORK_WATCHDOG_INTERNET_URL:-https://1.1.1.1}" >/dev/null 2>&1; then
|
|
echo " $ICON_SUCCESS Internet: reachable"
|
|
else
|
|
echo " $ICON_ERROR Internet: NOT reachable"
|
|
fi
|
|
|
|
if [[ -n "$DDNS_DOMAIN" ]]; then
|
|
_pub=$(curl -sf --max-time 5 https://ifconfig.me 2>/dev/null | tr -d '[:space:]')
|
|
_dns=$(dig +short "$DDNS_DOMAIN" @1.1.1.1 2>/dev/null | grep -Eo '[0-9]+\.[0-9]+\.[0-9]+\.[0-9]+' | head -1)
|
|
echo " $ICON_GEAR Public IP: ${_pub:-unknown}"
|
|
echo " $ICON_GEAR DNS record: ${_dns:-unknown}"
|
|
[[ "$_pub" == "$_dns" ]] && \
|
|
echo " $ICON_SUCCESS DDNS: in sync" || \
|
|
echo " $ICON_ERROR DDNS: MISMATCH — public=$_pub dns=$_dns"
|
|
else
|
|
echo " $ICON_GEAR DDNS: not configured for $MY_ID"
|
|
fi
|
|
|
|
if [[ "${NETWORK_WATCHDOG_CHECK_TAILSCALE:-true}" == "true" ]] && command -v tailscale >/dev/null 2>&1; then
|
|
if tailscale status --json 2>/dev/null | grep -qE '"BackendState":\s*"Running"'; then
|
|
echo " $ICON_SUCCESS Tailscale: running"
|
|
else
|
|
echo " $ICON_ERROR Tailscale: NOT running"
|
|
fi
|
|
else
|
|
echo " $ICON_GEAR Tailscale: check disabled or not installed"
|
|
fi
|
|
|
|
if [[ -n "$NPM_URL" ]]; then
|
|
if curl -sf --max-time "${NETWORK_WATCHDOG_NPM_TIMEOUT:-10}" "$NPM_URL" >/dev/null 2>&1; then
|
|
echo " $ICON_SUCCESS NPM proxy: reachable ($NPM_URL)"
|
|
else
|
|
echo " $ICON_ERROR NPM proxy: NOT reachable ($NPM_URL)"
|
|
fi
|
|
else
|
|
echo " $ICON_GEAR NPM proxy: not configured for $MY_ID"
|
|
fi
|
|
|
|
echo "━━━━━━━━━━━━━━━━━━━━━━━"
|
|
exit 0
|
|
fi
|
|
|
|
# ==============================================================================================
|
|
# ━━━ Check 1 — Internet Connectivity ━━━
|
|
# ==============================================================================================
|
|
echo "━━━ $ICON_NET Network Watchdog — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
|
|
|
|
ISSUES=0
|
|
|
|
if ! curl -sf --max-time "${NETWORK_WATCHDOG_INTERNET_TIMEOUT:-5}" \
|
|
"${NETWORK_WATCHDOG_INTERNET_URL:-https://1.1.1.1}" >/dev/null 2>&1; then
|
|
warn "$ICON_ERROR Internet not reachable — skipping DDNS, Tailscale, and NPM checks"
|
|
notify "Network watchdog: internet not reachable on $(hostname) ($MY_ID)" \
|
|
"Network Watchdog" "warning"
|
|
exit 1
|
|
fi
|
|
|
|
log "$ICON_SUCCESS Internet reachable"
|
|
|
|
# ==============================================================================================
|
|
# ━━━ Check 2 — DDNS ━━━
|
|
# ==============================================================================================
|
|
if [[ -n "$DDNS_DOMAIN" ]] && [[ -n "$DDNS_CONTAINER" ]]; then
|
|
PUBLIC_IP=$(curl -sf --max-time 5 https://ifconfig.me 2>/dev/null | tr -d '[:space:]')
|
|
DNS_IP=$(dig +short "$DDNS_DOMAIN" @1.1.1.1 2>/dev/null | grep -Eo '[0-9]+\.[0-9]+\.[0-9]+\.[0-9]+' | head -1)
|
|
|
|
if [[ -z "$PUBLIC_IP" ]]; then
|
|
warn "Could not determine public IP — skipping DDNS check"
|
|
elif [[ -z "$DNS_IP" ]]; then
|
|
warn "Could not resolve $DDNS_DOMAIN — skipping DDNS check"
|
|
elif [[ "$PUBLIC_IP" == "$DNS_IP" ]]; then
|
|
log "$ICON_SUCCESS DDNS in sync — $DDNS_DOMAIN → $DNS_IP"
|
|
else
|
|
warn "$ICON_ERROR DDNS mismatch — public=$PUBLIC_IP dns=$DNS_IP"
|
|
(( ISSUES++ ))
|
|
if [[ "$DRY_RUN" == true ]]; then
|
|
warn "DRY RUN — would restart $DDNS_CONTAINER"
|
|
else
|
|
warn "Restarting $DDNS_CONTAINER to trigger Cloudflare update..."
|
|
if docker restart "$DDNS_CONTAINER" >/dev/null 2>&1; then
|
|
warn "$DDNS_CONTAINER restarted ✅"
|
|
notify "DDNS mismatch on $(hostname) ($MY_ID) — $DDNS_DOMAIN was $DNS_IP, public is $PUBLIC_IP — $DDNS_CONTAINER restarted" \
|
|
"Network Watchdog" "warning"
|
|
else
|
|
warn "$DDNS_CONTAINER restart failed"
|
|
notify "DDNS mismatch on $(hostname) ($MY_ID) — $DDNS_CONTAINER restart failed — manual intervention needed" \
|
|
"Network Watchdog" "warning"
|
|
fi
|
|
fi
|
|
fi
|
|
else
|
|
log "DDNS check not configured for $MY_ID — skipping"
|
|
fi
|
|
|
|
# ==============================================================================================
|
|
# ━━━ Check 3 — Tailscale ━━━
|
|
# ==============================================================================================
|
|
if [[ "${NETWORK_WATCHDOG_CHECK_TAILSCALE:-true}" == "true" ]]; then
|
|
if ! command -v tailscale >/dev/null 2>&1; then
|
|
log "Tailscale not installed — skipping"
|
|
elif tailscale status --json 2>/dev/null | grep -qE '"BackendState":\s*"Running"'; then
|
|
log "$ICON_SUCCESS Tailscale running"
|
|
else
|
|
warn "$ICON_ERROR Tailscale not in Running state"
|
|
notify "Tailscale not running on $(hostname) ($MY_ID) — manual check needed" \
|
|
"Network Watchdog" "warning"
|
|
fi
|
|
fi
|
|
|
|
# ==============================================================================================
|
|
# ━━━ Check 4 — NPM Proxy (external) ━━━
|
|
# ==============================================================================================
|
|
if [[ -n "$NPM_URL" ]]; then
|
|
NPM_STRIKES=$(get_strikes "npm" "${NETWORK_WATCHDOG_NPM_STATE_FILE}")
|
|
NPM_STRIKE_LIMIT="${NETWORK_WATCHDOG_NPM_STRIKE_LIMIT:-2}"
|
|
|
|
if curl -sf --max-time "${NETWORK_WATCHDOG_NPM_TIMEOUT:-10}" "$NPM_URL" >/dev/null 2>&1; then
|
|
log "$ICON_SUCCESS NPM proxy reachable — $NPM_URL"
|
|
if [[ "$NPM_STRIKES" -gt 0 ]]; then
|
|
echo "NPM strikes cleared (was $NPM_STRIKES)"
|
|
set_strikes "npm" 0 "${NETWORK_WATCHDOG_NPM_STATE_FILE}"
|
|
fi
|
|
else
|
|
NPM_STRIKES=$(( NPM_STRIKES + 1 ))
|
|
set_strikes "npm" "$NPM_STRIKES" "${NETWORK_WATCHDOG_NPM_STATE_FILE}"
|
|
(( ISSUES++ ))
|
|
|
|
if [[ "$NPM_STRIKES" -lt "$NPM_STRIKE_LIMIT" ]]; then
|
|
warn "$ICON_ERROR NPM proxy not reachable — $NPM_URL (strike $NPM_STRIKES/$NPM_STRIKE_LIMIT)"
|
|
notify "NPM proxy not reachable on $(hostname) ($MY_ID) — $NPM_URL (strike $NPM_STRIKES/$NPM_STRIKE_LIMIT)" \
|
|
"Network Watchdog" "warning"
|
|
else
|
|
warn "$ICON_ERROR NPM proxy strike limit reached ($NPM_STRIKES) — restarting NginxProxyManager"
|
|
if [[ "$DRY_RUN" == true ]]; then
|
|
warn "DRY RUN — would restart NginxProxyManager"
|
|
else
|
|
if docker restart NginxProxyManager >/dev/null 2>&1; then
|
|
warn "NginxProxyManager restarted ✅"
|
|
set_strikes "npm" 0 "${NETWORK_WATCHDOG_NPM_STATE_FILE}"
|
|
notify "NPM proxy restarted on $(hostname) ($MY_ID) — $NPM_URL was unreachable for $NPM_STRIKES cycles" \
|
|
"Network Watchdog" "warning"
|
|
else
|
|
warn "NginxProxyManager restart failed — manual intervention needed"
|
|
notify "NPM proxy restart FAILED on $(hostname) ($MY_ID) — manual intervention needed" \
|
|
"Network Watchdog" "warning"
|
|
fi
|
|
fi
|
|
fi
|
|
fi
|
|
else
|
|
log "NPM check not configured for $MY_ID — skipping"
|
|
fi
|
|
|
|
# ==============================================================================================
|
|
# ━━━ Exit ━━━
|
|
# ==============================================================================================
|
|
if [[ "$ISSUES" -gt 0 ]]; then
|
|
exit 1
|
|
else
|
|
echo "Network healthy ✅ ($(date '+%H:%M:%S'))"
|
|
exit 0
|
|
fi
|