Files
Varaverk/Master.conf
T

1329 lines
72 KiB
Bash

#!/bin/bash
# ==============================================================================================
# ================================= MASTER CONFIGURATION =======================================
# ==============================================================================================
# All user-facing variables for the unRAID script ecosystem.
# Scripts source this file — edit here, changes apply everywhere on next git pull.
#
# ── HOW THIS FILE WORKS ───────────────────────────────────────────────────────────────────────
# Every script sources Master.conf and common.sh at startup.
# Change a value here and it affects all scripts that use it — no hunting through files.
# To disable something: comment it out with # rather than deleting it.
# To add a new rsync profile: add a key to each PROFILE_* array.
# To add a new media maintenance job: add a line to MEDIA_MAINTENANCE_JOBS.
#
# ── INDEX ─────────────────────────────────────────────────────────────────────────────────────
#
# Section Description
# ───────────────────────────────────────────────────────────────────────────────────────────
# HOST CONFIGURATION Server hostnames and SSH key paths
# LOGGING Enable or disable verbose logging
# NOTIFICATIONS unRAID native and Discord webhook settings
# GIT / REPO Gitea repository and SSH settings
#
# ── RSYNC ──────────────────────────────────────────────────────────────────────────────────
# RSYNC DEFAULTS Global fallback rsync settings
# REMOTE HEALTH CHECKS Rootfs threshold for pre-flight abort
# DAILY SYNC SHARES Media shares synced by media_shares_sync.sh (per-host)
# PERSONAL ENCRYPTED SHARES Per-user private shares — ZFS encrypted, synced separately
# RSYNC PROFILE SYSTEM Per-profile overrides (appdata profiles)
#
# ── FAILOVER ───────────────────────────────────────────────────────────────────────────────
# FAILOVER Mutual container failover between two servers
# FAILOVER TEST Simulated outage settings for failover_test.sh
# DDNS Script-controlled DDNS — absolute rules
# INTERNET LOSS Containers to stop when internet is lost
# TIERED CONTAINER LISTS What each server runs for the other per tier
# TIER DELAY SETTINGS How long before each tier activates (minutes)
# RSYNC WRITEBACK JOBS Appdata synced back to primary on handback
#
# ── DOCKER ESSENTIALS ──────────────────────────────────────────────────────────────────────
# DOCKER DAILY RESTART Containers restarted daily
# DOCKER WEEKLY RESTART Containers restarted weekly
# DOCKER WATCHDOG Two-tier self-healing container monitoring
# DOCKER NETWORK CONNECT Connect containers to extra networks on boot
#
# ── UNRAID ESSENTIALS ──────────────────────────────────────────────────────────────────────
# REBOOT User warning delay before scheduled reboot
# MOVER Mover stop timeout
# SYSLOG FILTER Docker veth noise filter file path
# PHP-FPM PHP-FPM max children config
# CLEAR LOGS System log file paths
# WEBGUI WATCHDOG WebGUI nginx + emhttp monitoring and restart
#
# ── MEDIA ──────────────────────────────────────────────────────────────────────────────────
# MEDIA PERMISSIONS Share list, mode and owner for permissions script
# MEDIA CLEANER Anime and media folder lists and file patterns
# MEDIA MANAGEMENT Orchestrator job list for media_management.sh
# ARR CLEANUP Lidarr, Sonarr, Radarr orphan file cleanup
# ARR FAILED/STALLED RECOVERY Auto blocklist + re-search failed imports and stalled downloads
#
# ── TRANSCODES ─────────────────────────────────────────────────────────────────────────────
# TRANSCODE MANAGER Ramdisk and SSD fallback transcode management
# TRANSCODE SERVER ARRAY Multi-server session monitoring (Emby, Jellyfin, Plex)
#
# ── MONITORS ───────────────────────────────────────────────────────────────────────────────
# CERTIFICATE MONITOR SSL certificate expiry monitoring
# BACKUP VERIFY Random sample checksum verification against remote
# SMART HEALTH Drive SMART attribute monitoring
# ZFS MEMORY SNAPSHOT Weekly ZFS health and memory diagnostic report
# BANDWIDTH MONITOR Daily rsync transfer logging and weekly summary
# HEALTH DIGEST Aggregated system health digest — always/smart/weekly
# CRITICAL SHARES MAINTENANCE Weekly clean sync + container updates (Emby + auth stack)
# EMBY Emby URL and API key — used by multiple scripts
# EMBY SESSION REPORT Weekly Emby usage statistics via API
#
# ── SYSTEM WATCHDOG ────────────────────────────────────────────────────────────────────────
# SYSTEM WATCHDOG System health monitoring — last line of defense
#
# ==============================================================================================
# ━━━ Host Configuration ━━━
# Hostnames must match Tailscale machine names exactly — case sensitive.
# Used by detect_hosts() in common.sh to determine which server is local and which is remote.
# Both servers run identical scripts — host detection makes them bidirectional.
HOST1="unRAID-Gmer4Lfe"
HOST2="unRAID-Jayred365"
# SSH keys for server-to-server rsync and failover container operations.
# HOST1_SSH_KEY is used when HOST1 SSHes to HOST2 and vice versa.
# Both keys must be in /root/.ssh/ and authorised in the remote server's authorized_keys.
HOST1_SSH_KEY="/root/.ssh/Gmer4Lfe-rsync-key"
HOST2_SSH_KEY="/root/.ssh/Jayred365-rsync-key"
# Emby — per-host container name, URL and API key.
# Defined once here — referenced by transcode_manager.sh, emby_session_report.sh,
# emby_database_repair.sh, critical_shares_maintenance.sh and TRANSCODE_SERVERS array.
# API key: Emby Dashboard → API Keys → + New Key
HOST1_EMBY_CONTAINER="Emby"
HOST1_EMBY_URL="http://localhost:8096"
HOST1_EMBY_API_KEY="0c27448d93a7431f9ac63569f7655829"
HOST2_EMBY_CONTAINER="Emby-Jayred365"
HOST2_EMBY_URL="http://localhost:8096" # same port — different server, different key
HOST2_EMBY_API_KEY="your-host2-emby-api-key"
# ━━━ Logging ━━━
# Controls verbose [LOG] output across all scripts.
# true = show detailed [LOG] lines — useful for debugging or first-time setup
# false = show only user-facing output — cleaner for scheduled runs
ENABLE_LOGGING=true
# ━━━ Notifications ━━━
# Two independent notification channels — either or both can be active simultaneously.
# unRAID native notification system — integrates with the bell icon in the WebGUI.
# Recommended: set Settings → Notification Settings to errors/warnings only so
# normal completions don't create noise. The ecosystem sends:
# normal — job completed successfully (informational)
# warning — something failed or needs attention
NOTIFY_UNRAID=true
# Discord webhook URL — paste the full webhook URL from your Discord server settings.
# Leave blank to disable Discord notifications entirely.
DISCORD_WEBHOOK=""
# ━━━ Git / Repo ━━━
# Gitea self-hosted repository settings used by git_pull_execute.sh.
# Running git_pull_execute.sh on either server pulls the latest scripts and sets
# executable permissions automatically — keeps both servers in sync with one command.
#
# git_pull_execute.sh detects where the Gitea container is running at runtime:
# Gitea running locally → connects via local IP
# Gitea running remotely → connects via remote server's Tailscale IP
# No hardcoded assumptions about which server hosts Gitea — works through failover.
# If Gitea fails over to HOST2, HOST1 automatically finds it there and vice versa.
GITEA_CONTAINER="Gitea" # exact Docker container name
GITEA_REPO_PATH="FailedProxy/Unraid_Scripts.git" # repo path on Gitea
GITEA_DOMAIN="" # public domain fallback — e.g. git.gmer4lfe.com
# used if local and Tailscale both fail
# requires NPM + DNS setup before enabling
TARGET_DIR="/mnt/user/appdata/unraid_scripts"
GITEA_SSH_KEY="/root/.ssh/unraid_gitea" # SSH key for authenticating to Gitea
SSH_PORT=221 # Gitea SSH port
# ==============================================================================================
# ── RSYNC ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Rsync Defaults ━━━
# Global fallback values used when no profile match is found for a directory.
# Shares in HOST1_DAILY_SYNC_SHARES and HOST2_DAILY_SYNC_SHARES always use these globals — no profile is defined for them.
# Appdata shares (Arrs_Stack, Critical-Data etc.) match profiles by directory basename.
# If a profile key exists in a PROFILE_* array that value overrides the global.
# If a profile key is missing the global below is used as the fallback.
BW_LIMIT=12500 # network transfer speed cap in KB/s — 12500 ≈ 100Mbit
RETRY_COUNT=3 # number of retry attempts if rsync fails before giving up
SLEEP=300 # seconds to wait between retry attempts
CRITICAL_CONTAINER_NAMES=() # containers to stop on REMOTE before rsync — profiles override
DELAYED_CONTAINERS=() # containers needing delay before starting — profiles override
CONTAINER_DELAY=5 # seconds to wait before starting delayed containers
EXCLUDE_DIRS=() # directories to exclude from transfer — profiles override
# Default rsync options used when no profile match is found.
# --delete removes files on remote that no longer exist on source (mirror behaviour)
# --inplace writes directly to destination file — better for large files, avoids temp copies
# --no-whole-file forces delta transfer even on fast local-like connections
DEFAULT_RSYNC_OPTS=(-av --info=progress2 --human-readable --bwlimit="$BW_LIMIT" --delete --inplace --no-whole-file)
# ━━━ Remote Health Checks ━━━
# Pre-flight check run before every rsync — aborts if remote rootfs (/) usage is at or
# above this percentage. When the remote array is down or drives are missing, rsync
# writes land on rootfs instead of /mnt/user — this fills the filesystem rapidly and
# can crash the remote server. 75% gives headroom to detect the problem early.
ROOTFS_WARN=75
# ━━━ Daily Sync Shares ━━━
# Media shares synced once daily by Orchestrators/media_shares_sync.sh.
# Each server only syncs the shares it is source of truth for — direction is automatic.
# detect_hosts() determines which server is running and picks the correct list.
#
# HOST1 pushes its truth shares TO HOST2.
# HOST2 pushes its truth shares TO HOST1.
# Never both pushing the same share — one server is always the truth holder.
#
# These shares use DEFAULT_RSYNC_OPTS — no profile entry needed.
# For shares needing custom bandwidth or container stops — create a profile below instead.
# HOST1 truth shares — pushed from HOST1 to HOST2 nightly
HOST1_DAILY_SYNC_SHARES=(
/mnt/user/Books
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Nextcloud
/mnt/user/stand-up_comedy
/mnt/user/Sports
/mnt/user/Tv_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/Anime_Movies-Old
)
# HOST2 truth shares — pushed from HOST2 to HOST1 nightly
# HOST2 is source of truth for anime — his arrs manage these shares
HOST2_DAILY_SYNC_SHARES=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Shows
)
# ━━━ Personal Encrypted Shares ━━━
# Personal shares synced to the remote server for offsite backup.
# These are independent of the failover container stack — data backup only.
# Each user syncs their own personal share to the other server.
#
# ── ZFS ENCRYPTION SETUP (unRAID 7) ─────────────────────────────────────────────────────────
# Encrypting your personal share means the remote admin can see the share exists
# and its file sizes but cannot read any content without your passphrase or keyfile.
# ZFS encrypts at the dataset level — rsync copies encrypted blocks as-is.
# The remote server never needs your key.
#
# Setup steps on HOST1:
# 1. In unRAID UI → go to your ZFS pool (Main tab → pool name)
# 2. Click the pool to expand it
# 3. Click "+ Dataset" to create a new dataset
# 4. Name it: e.g. Gmer4Lfe-Personal
# 5. Enable Encryption → set your passphrase (or keyfile path)
# ⚠️ Write your passphrase down — if lost, data is unrecoverable
# 6. Go to Settings → Shares → Add Share
# 7. Set the share path to your new encrypted dataset
# 8. Set Use cache: Only (keeps data on ZFS pool, not array)
#
# Auto-unlock on boot (optional — keyfile approach):
# 1. Create a keyfile: dd if=/dev/urandom bs=32 count=1 | base64 > /root/.zfs-keys/personal.key
# 2. Store keyfile on HOST1 only — never sync it to HOST2
# 3. Set the dataset to use keyfile instead of passphrase
# 4. Add to /etc/rc.local or a startup script:
# zfs load-key -L file:///root/.zfs-keys/personal.key poolname/Gmer4Lfe-Personal
# zfs mount poolname/Gmer4Lfe-Personal
# Manual unlock alternative (most secure):
# zfs load-key poolname/Gmer4Lfe-Personal (prompts for passphrase)
# zfs mount poolname/Gmer4Lfe-Personal
#
# Verify encryption is active before syncing:
# zfs get encryption poolname/Gmer4Lfe-Personal
# Should show: encryption aes-256-gcm (or similar)
#
# Once set up — add the share to PERSONAL_SYNC_SHARES below.
# rsync copies encrypted blocks to remote — remote admin cannot decrypt without your key.
# ─────────────────────────────────────────────────────────────────────────────────────────────
# HOST1 personal shares synced to HOST2 for offsite backup
# These sync via media_shares_sync.sh or on their own schedule
# Encrypted datasets sync as encrypted — remote cannot read content
HOST1_PERSONAL_SHARES=(
# /mnt/user/Gmer4Lfe-Personal # uncomment after creating encrypted dataset
)
# HOST2 personal shares synced to HOST1 for offsite backup
HOST2_PERSONAL_SHARES=(
# /mnt/user/Jayred365-Personal # uncomment after creating encrypted dataset
)
# ━━━ Rsync Profile System ━━━
# Profiles allow per-share rsync behaviour without touching script logic.
# The profile key is matched automatically by the basename of the directory
# passed to rsync.sh (lowercased).
#
# Example:
# rsync.sh /mnt/user/appdata-Failover/Arrs_Stack
# basename = Arrs_Stack → lowercased = arrs_stack → matches [arrs_stack] profile
#
# To add a new profile:
# 1. Add a key to each PROFILE_* array below using your chosen name
# 2. Call rsync.sh with a directory whose basename matches that key
# 3. Any array you omit falls back to its global default automatically
#
# IMPORTANT: PROFILE_RSYNC_OPTS does NOT inherit from DEFAULT_RSYNC_OPTS.
# If you define a profile entry you must list ALL desired options explicitly.
#
# Current profiles:
# arrs_stack — Sonarr, Radarr, Lidarr, Readarr, Prowlarr, Bazarr, Pinchflat
# Lower bandwidth — runs alongside media syncs
# Containers stopped during sync for data consistency
# critical-data — Auth stack: NPM, Authelia, Mariadb-Authelia, Redis-Authelia, LLDAP
# High bandwidth — small data, synced frequently
# Authelia needs delayed start — database containers must be ready first
# gmer4lfe — Server-specific appdata: Organizr, UptimeKuma, VaultWarden
# Medium bandwidth — personal services, no container stop needed
# important-data — NextCloud + Postgres database
# High bandwidth — NextCloud needs graceful stop before sync
# NextCloud needs delayed start — Postgres must be accepting connections
# emby — Emby media server appdata and metadata only
# Medium bandwidth — large appdata directory, no containers stopped
# Rsync options per profile — replaces DEFAULT_RSYNC_OPTS entirely for that profile run
# SPACE-SEPARATED STRINGS — converted to array at runtime by rsync.sh
declare -A PROFILE_RSYNC_OPTS=(
[arrs_stack]="-av --info=progress2 --human-readable --bwlimit=$BW_LIMIT --delete --inplace"
[critical-data]="-av --human-readable --bwlimit=$BW_LIMIT --delete"
[gmer4lfe]="-av --info=progress2 --bwlimit=$BW_LIMIT"
[important-data]="-av --human-readable --bwlimit=$BW_LIMIT"
# emby — nightly clean sync, both Emby instances stopped
# WAL checkpointed on shutdown — full consistent mirror, minimal excludes
[emby]="-av --human-readable --bwlimit=$BW_LIMIT --delete --inplace --no-whole-file"
# emby-failover — frequent dirty sync every 30-60min, Emby stays running
# Only critical failover data — what users need immediately on failover
# WAL excluded — safe dirty write while Emby is running
# This is also the list written back during failover handback
[emby-failover]="-av --human-readable --bwlimit=$BW_LIMIT --delete --inplace --no-whole-file"
)
# Bandwidth limit in KB/s per profile — overrides global BW_LIMIT for this profile only
declare -A PROFILE_BW_LIMIT=(
[arrs_stack]=5000 # lower — runs alongside other jobs, avoids saturating link
[critical-data]=9500 # high — small data, get it synced fast
[gmer4lfe]=8000 # medium
[important-data]=9500 # high — database sync needs to be fast and clean
[emby]=8000 # medium — full mirror, steady transfer
[emby-failover]=9500 # high — small critical dataset, get it synced fast
)
# Retry attempts per profile before giving up — overrides global RETRY_COUNT
declare -A PROFILE_RETRY_COUNT=(
[arrs_stack]=3
[critical-data]=3
[gmer4lfe]=3
[important-data]=3
[emby]=3
[emby-failover]=3
)
# Seconds between retry attempts — overrides global SLEEP
declare -A PROFILE_SLEEP=(
[arrs_stack]=300
[critical-data]=300
[gmer4lfe]=300
[important-data]=300
[emby]=300
[emby-failover]=120 # shorter — frequent sync, retry faster
)
# Containers stopped on BOTH LOCAL and REMOTE servers before rsync.
# Local stops first — flushes databases cleanly before push.
# Remote stops next — prevents writes while receiving.
# Only containers that were running get restarted — stopped containers stay stopped.
# Same container names used on both HOST1 and HOST2 — naming scheme is consistent.
# If a container is not found on a server it is skipped gracefully, not errored.
# SPACE-SEPARATED STRINGS — converted to array at runtime
declare -A PROFILE_CRITICAL_CONTAINER_NAMES=(
[arrs_stack]="Sonarr Lidarr Readarr Radarr Prowlarr Bazarr Pinchflat"
[critical-data]="Mariadb-Authelia Mariadb-Authelia-Secondary Redis-Authelia Redis-Authelia-Secondary Lldap-Gmer4Lfe NginxProxyManager Authelia Authelia-Secondary"
[gmer4lfe]="Organizrv2-Gmer4Lfe UptimeKuma-Gmer4Lfe VaultWarden-Gmer4Lfe"
[important-data]="Postgres-NextCloud NextCloud"
[emby]="Emby" # nightly clean sync — Emby stopped both sides, WAL checkpointed
[emby-failover]="" # dirty sync — Emby stays running both sides
)
# Containers that need a delay before starting after rsync completes.
# Used when a container depends on another that was also stopped.
# Startup order for critical-data:
# Immediate: Mariadb x2, Redis x2, Lldap, NginxProxyManager
# Delayed: Authelia, Authelia-Secondary (need databases ready)
# SPACE-SEPARATED STRINGS — converted to array at runtime
declare -A PROFILE_DELAYED_CONTAINERS=(
[arrs_stack]=""
[critical-data]="Authelia Authelia-Secondary"
[gmer4lfe]=""
[important-data]="NextCloud" # NextCloud needs Postgres accepting connections first
[emby]=""
[emby-failover]=""
)
# Seconds to wait before starting delayed containers
declare -A PROFILE_CONTAINER_DELAY=(
[arrs_stack]=5
[critical-data]=15 # 15s gives Mariadb, Redis, and LLDAP time to accept connections
[gmer4lfe]=5
[important-data]=10
[emby]=5
[emby-failover]=5
)
# Directories excluded from rsync transfer per profile.
# SPACE-SEPARATED STRINGS — converted to array at runtime
declare -A PROFILE_EXCLUDE_DIRS=(
[arrs_stack]="logs *.tmp"
[gmer4lfe]="logs *.tmp"
[important-data]="logs *.tmp"
# Critical-Data — auth stack
# Containers stopped during sync — databases flush cleanly
# Generated configs, logs, and temp files excluded
[critical-data]="logs *.tmp *.log nginx/temp nginx/cache __pycache__ notification.txt"
# Emby nightly clean sync — both Emby instances stopped
# WAL checkpointed on shutdown — safe to push everything except true junk
# Full faithful mirror: metadata, plugins, config all included
[emby]="logs transcodes cache crash*"
# Emby failover dirty sync — Emby stays running
# Only what users need immediately on failover:
# users.db ← watch states, continue watching, next up, played state
# library.db ← library structure
# authentication.db ← API keys, sessions
# config/ ← server settings
# WAL and SHM excluded — unsafe while Emby is running
# Also the list written back during failover handback
[emby-failover]="logs transcodes cache metadata *.db-wal *.db-shm crash* plugins root"
)
# Per-disk check toggle
declare -A PROFILE_SKIP_DISK_CHECK=(
[arrs_stack]=true
[critical-data]=true
[gmer4lfe]=true
[important-data]=true
[emby]=true
[emby-failover]=true
)
# ==============================================================================================
# ── FAILOVER ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Mutual container failover between two unRAID servers.
# Each server runs Failover/failover.sh independently — no coordination between servers.
# All decisions based solely on two pings: remote reachable + internet reachable.
#
# ── STATES ────────────────────────────────────────────────────────────────────────────────────
# NORMAL — remote up, internet up — own containers only, DDNS ON, silent
# FAILOVER — remote down, internet up — start remote containers (tiered by time)
# NO_INTERNET — internet down — stop own DDNS immediately, wait for recovery
# DARK — remote down AND internet down — same as NO_INTERNET
#
# ── DDNS RULES — ABSOLUTE ─────────────────────────────────────────────────────────────────────
# Each server owns its own DDNS — ON when that server has internet
# Script controls DDNS exclusively — network state NEVER auto-starts DDNS
# Internet loss → stop own DDNS immediately
# Failover → start remote DDNS as first action (Tier 1)
# Handback → stop remote DDNS FIRST → rsync → start local containers
# → start local DDNS LAST — only after containers confirmed up
# One DDNS per domain active at all times — never two, never zero for long
# 1 minute TTL + 1 minute check interval = minimal user impact
#
# ── HANDBACK SEQUENCE ─────────────────────────────────────────────────────────────────────────
# Strike confirmation → pre-flight → stop remote DDNS → stop remote containers
# → rsync writeback → start local containers → start local DDNS → NORMAL
# Containers only down during rsync window — minimise this time
#
# ── TIERED FAILOVER ───────────────────────────────────────────────────────────────────────────
# Tier 1 — Immediate — vital services + Live TV — people are watching, can't wait
# Tier 2 — configurable delay — shared productivity services
# Tier 3 — configurable delay — secondary services
# Tier 4 — configurable delay — arrs + downloaders — workflow continuity
# Delays set independently per host below
EXTERNAL_IP="8.8.8.8" # external IP to ping for internet connectivity check
FAILOVER_CHECK_INTERVAL=120 # seconds between state checks
# 1 minute TTL + 2 minute interval = minimal gap
FAILOVER_HANDBACK_STRIKES=2 # consecutive remote-up checks before handback
# 2 strikes x 120s = 4 min confirmation window
FAILOVER_STATE_FILE="/boot/config/failover_state.db"
# persists on /boot/ — survives reboots
# tracks: state, failover_start, strikes, tier flags
# ━━━ Failover Test ━━━
# Used by Failover/failover_test.sh — controlled simulation via iptables block.
# All failover logic stays in failover.sh — test script is the harness only.
# ⚠️ Run during maintenance window — real containers start and stop during the test.
# Use --dry-run first to walk through phases without touching anything.
FAILOVER_TEST_BLOCK_WAIT=150 # seconds to hold iptables block
# must be > FAILOVER_CHECK_INTERVAL + buffer
FAILOVER_TEST_HANDBACK_WAIT=360 # seconds to wait for handback completion
# covers FAILOVER_HANDBACK_STRIKES x INTERVAL + rsync
# ━━━ DDNS — Script Controlled Exclusively ━━━
# Each server owns its own DDNS containers — one domain per server.
# DDNS is started and stopped ONLY by this script — never by network state returning.
# HOST1 DDNS starts last in handback (after containers confirmed up).
# HOST1 DDNS stops first on internet loss.
# HOST2 DDNS starts when HOST2 detects HOST1 is down (Tier 1).
# HOST2 DDNS stops before handback rsync begins.
HOST1_DDNS_CONTAINERS=(
"Gmer4Lfe.com" # HOST1's own DDNS — ON when HOST1 has internet
# covers gmer4lfe.com pointing to HOST1 IP
)
HOST2_DDNS_CONTAINERS=(
"Gmer4Lfe.us" # HOST2's own DDNS — ON when HOST2 has internet
# covers gmer4lfe.us pointing to HOST2 IP
)
# ━━━ Containers to stop on internet loss ━━━
# Own DDNS handled separately above — list additional containers here if needed
# These stop when this server loses internet — regardless of remote state
FAILOVER_HOST1_STOP_ON_NO_NET=(
"Gmer4Lfe.com" # stop when HOST1 loses internet
)
FAILOVER_HOST2_STOP_ON_NO_NET=(
"Gmer4Lfe.us" # stop when HOST2 loses internet
)
# ━━━ HOST1 runs these for HOST2 when HOST2 goes down ━━━
# HOST2's DDNS listed in Tier 1 — starts immediately as first action
# List HOST2's specific services here — HOST2's own containers only
# Do NOT list shared services that HOST1 already runs
# Tier 1 — Immediate — starts as soon as HOST2 is detected down
FAILOVER_HOST1_RUNS_FOR_HOST2_IMMEDIATE=(
"Gmer4Lfe.us" # HOST2's DDNS — start first, covers HOST2's domain
"VaultWarden-Jayred365" # HOST2's password manager — immediate access needed
# "container-placeholder" # add HOST2 specific services here
)
# Tier 2 — starts after HOST2_TIER2_DELAY minutes
FAILOVER_HOST1_RUNS_FOR_HOST2_2HR=(
# "container-placeholder"
)
# Tier 3 — starts after HOST2_TIER3_DELAY minutes
FAILOVER_HOST1_RUNS_FOR_HOST2_6HR=(
# "container-placeholder"
)
# Tier 4 — starts after HOST2_TIER4_DELAY minutes
FAILOVER_HOST1_RUNS_FOR_HOST2_18HR=(
# "container-placeholder"
)
# ━━━ HOST2 runs these for HOST1 when HOST1 goes down ━━━
# HOST1's DDNS listed in Tier 1 — starts immediately to cover HOST1's domain
# Live TV in Tier 1 — people are watching, cannot wait for tiered startup
# Auth stack in Tier 1 — everything proxied through NPM needs auth
# Tier 1 — Immediate — vital services and Live TV cannot wait
FAILOVER_HOST2_RUNS_FOR_HOST1_IMMEDIATE=(
"Gmer4Lfe.com" # HOST1's DDNS — start first, covers HOST1's domain
"Emby" # media server — users are actively watching
"VaultWarden-Gmer4Lfe" # password manager — critical, immediate access needed
"Dispatcharr" # Live TV — people are watching, cannot wait
"Dispatcharr-Basic" # Live TV basic profile
"Dispatcharr-Iptv-Users" # Live TV IPTV users
"ErsatzTV-Emby" # Live TV scheduling and channel management
)
# Tier 2 — starts after HOST1_TIER2_DELAY minutes
# Productivity services — important but can wait a couple of hours
FAILOVER_HOST2_RUNS_FOR_HOST1_2HR=(
"Postgres-NextCloud" # NextCloud database — must start before NextCloud
"NextCloud" # file access and collaboration
"PostgreSQL_Immich" # Immich database
"Immich-Gmer4Lfe" # photo management
# "container-placeholder"
)
# Tier 3 — starts after HOST1_TIER3_DELAY minutes
# Secondary services — useful but not immediately critical
FAILOVER_HOST2_RUNS_FOR_HOST1_6HR=(
"Gitea" # git server
# "container-placeholder"
)
# Tier 4 — starts after HOST1_TIER4_DELAY minutes
# Full workflow mode — arrs and downloaders
# Minimal writeback on handback — start fresh is cleaner than syncing download state
FAILOVER_HOST2_RUNS_FOR_HOST1_18HR=(
"Sonarr" # TV show management
"Radarr" # movie management
"Lidarr" # music management
"Readarr" # book management
"Prowlarr" # indexer management
"Bazarr" # subtitle management
"SABnzbd-Gmer4Lfe" # usenet downloader
"Qbittorrent-Gmer4Lfe" # torrent downloader
"LidaTube" # YouTube music downloader
"Pinchflat" # YouTube channel downloader
"ChannelTube" # YouTube channel management
# "container-placeholder"
)
# ━━━ Tier Delay Settings ━━━
# How long primary must be down before each tier activates — set in minutes
# Tier 1 is always immediate — no delay
# Set independently per host — a large server may want longer delays than a small one
# Adjust based on your tolerance for resource usage on the covering server
# Delays for HOST1's containers running on HOST2 (HOST1 is down)
HOST1_TIER2_DELAY=240 # 4 hours — NextCloud, Immich can wait
HOST1_TIER3_DELAY=720 # 12 hours — secondary services
HOST1_TIER4_DELAY=1440 # 24 hours — full workflow, arrs and downloaders
# Delays for HOST2's containers running on HOST1 (HOST2 is down)
HOST2_TIER2_DELAY=240
HOST2_TIER3_DELAY=720
HOST2_TIER4_DELAY=1440
# ━━━ Rsync Writeback Jobs ━━━
# Run during handback — syncs critical appdata back to primary before containers restart.
# Containers are stopped before this runs — clean source, no competing writes.
# Full bandwidth available — DDNS stopped, containers stopped, nothing competing.
#
# ── WRITEBACK SKIP WINDOW ─────────────────────────────────────────────────────────────────────
# Short outages do not benefit from writeback — the covering server accumulated dirty
# or minimal data not worth writing over the primary's cleaner state.
#
# Emby syncs every 30min from a live running container (dirty sync).
# A clean full sync runs nightly at 2:30am with Emby stopped.
# After a short outage HOST1's nightly clean state is more reliable
# than HOST2's dirty 30min sync data — skip writeback entirely.
#
# Real world outage profile:
# 2-10 minutes — power blip, most common → skip writeback
# 10-60 minutes — ISP issue, router restart → skip writeback
# 1hr+ — actual problem → writeback worthwhile
# 18hr+ — Tier 4 activated → always writeback
#
# Tier 1 writeback delay — separate from Tier 1 start delay (Tier 1 always starts immediately)
# Tier 2+ reuse their existing TIER_DELAY vars — if containers started, time passed = writeback warranted
# Tier 4 writeback — skips if under HOST*_TIER4_DELAY (same threshold as container start)
# Only runs if outage was long enough to activate Tier 4 containers
# Automatically uses the opposing host's daily sync share list (HOST*_DAILY_SYNC_SHARES)
# FAILOVER_HOST*_WRITEBACK_TIER4 is for edge cases only — empty by default
HOST1_TIER1_WRITEBACK_DELAY=60 # minutes — skip Tier 1 writeback if outage under this
HOST2_TIER1_WRITEBACK_DELAY=60 # minutes — skip Tier 1 writeback if outage under this
# Tier 2 writeback threshold = HOST1_TIER2_DELAY (reused)
# Tier 3 writeback threshold = HOST1_TIER3_DELAY (reused)
# Tier 4 always writebacks — no threshold
# ── WRITEBACK JOB LISTS ───────────────────────────────────────────────────────────────────────
# Organised by tier — writeback runs per tier based on outage duration
# Priority:
# Tier 1 — Emby userdata, auth stack — small, fast, most important
# Tier 2 — NextCloud, Immich — user files that may have changed
# Tier 3 — secondary services
# Tier 4 — always runs if Tier 4 activated — arrs accumulated meaningful state
# Skip — media files (already on primary, never moved)
# Skip — downloads (start fresh — cleaner than syncing partial state)
# HOST1 writeback tiers — run by HOST2 during HOST1 handback
FAILOVER_HOST1_WRITEBACK_TIER1=(
"/mnt/user/Media_Server/Emby" # Emby userdata, playstates, metadata
# "location-placeholder"
)
FAILOVER_HOST1_WRITEBACK_TIER2=(
"/mnt/user/appdata-Failover/Important-Data" # NextCloud + Postgres
)
FAILOVER_HOST1_WRITEBACK_TIER3=(
# "location-placeholder"
)
FAILOVER_HOST1_WRITEBACK_TIER4=(
# Edge case paths outside of normal HOST1_DAILY_SYNC_SHARES
# Tier 4 automatically syncs HOST1_DAILY_SYNC_SHARES back — add extras here only
"/mnt/user/appdata-Failover/Arrs_Stack"
)
# HOST2 writeback tiers — run by HOST1 during HOST2 handback
FAILOVER_HOST2_WRITEBACK_TIER1=(
# "/mnt/user/appdata-Failover/Jayred365-Emby"
# "/mnt/user/appdata-Failover/Jayred365-Critical"
)
FAILOVER_HOST2_WRITEBACK_TIER2=(
# "/mnt/user/appdata-Failover/Jayred365-Important"
)
FAILOVER_HOST2_WRITEBACK_TIER3=(
# "location-placeholder"
)
FAILOVER_HOST2_WRITEBACK_TIER4=(
# Edge case paths outside of normal HOST2_DAILY_SYNC_SHARES
# Tier 4 automatically syncs HOST2_DAILY_SYNC_SHARES back — add extras here only
"/mnt/user/appdata-Failover/Arrs_Stack"
)
# ==============================================================================================
# ── DOCKER ESSENTIALS ─────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Docker Daily Restart ━━━
# Containers restarted every day — keeps services fresh, clears memory leaks.
# Case-sensitive — must match exact Docker container names in the unRAID Docker tab.
DAILY_RESTART_CONTAINERS=(
"NginxProxyManager"
"Authelia"
"Dispatcharr-Iptv-Users"
"Dispatcharr"
"Dispatcharr-Basic"
"ErsatzTV-Emby"
)
# ━━━ Docker Weekly Restart ━━━
# Less critical services that benefit from periodic restart but don't need daily cycling.
# Case-sensitive — must match exact Docker container names in the unRAID Docker tab.
WEEKLY_RESTART_CONTAINERS=(
"NextCloud"
"Organizrv2-Gmer4Lfe"
"AdGuard-Home"
"Immich-Gmer4Lfe"
)
# ━━━ Docker Watchdog ━━━
# Two-tier self-healing container monitoring.
# Tier 1 — strict monitoring of explicitly configured containers
# Tier 2 — global health scan of ALL running containers
#
# Cross-cutting intelligence:
# Startup grace — skip restarts while system is still booting
# Dependency order — restart database before app
# Restart loop — stop restarting after limit hit → skip list → notify critical
# Skip list — persistent across reboots, auto-clears when container recovers
# Batch notify — one clean summary per run
# Memory hard limits in MB — immediate restart if exceeded
# 20GB=20480 16GB=16384 14GB=14336 12GB=12288 10GB=10240
# 8GB=8192 6GB=6144 4GB=4096 2GB=2048 1GB=1024
declare -A WATCHDOG_CONTAINERS=(
["Emby"]=16384
["LidaTube"]=6144
["Tdarr"]=6144
["Code-Server"]=1024
)
# Per-host HTTP health check URLs — docker_watchdog.sh picks correct list via detect_hosts()
declare -A HOST1_WATCHDOG_CONTAINER_URLS=(
["Emby"]="http://localhost:8096"
)
declare -A HOST2_WATCHDOG_CONTAINER_URLS=(
["Emby"]="http://localhost:8096" # his Emby — same port, different server
)
# Per-host required containers — core stack that must always be running
# docker_watchdog.sh picks correct list via detect_hosts()
# Strike system — persistent skip list on /boot/, auto-clears on recovery
HOST1_WATCHDOG_REQUIRED_CONTAINERS=(
"NginxProxyManager"
"Lldap-Gmer4Lfe"
"Authelia"
"Mariadb-Authelia"
"Redis-Authelia"
"Authelia-Secondary"
"Redis-Authelia-Secondary"
)
HOST2_WATCHDOG_REQUIRED_CONTAINERS=(
"NginxProxyManager"
# add HOST2's required containers here
)
# Strike state file — /tmp resets on reboot which is correct for strike tracking
WATCHDOG_STATE_FILE="/tmp/container_watchdog_state.db"
# CPU thresholds — normalised against total core count automatically at runtime
SOFT_CPU_THRESHOLD=80 # warn at this % of total system CPU
HARD_CPU_THRESHOLD=85 # strike at this % of total system CPU
CPU_FAIL_LIMIT=2 # consecutive CPU strikes before container restart
# Memory soft threshold — warn when container reaches this % of its hard limit
# Hard limit exceeded triggers immediate restart regardless of strikes
SOFT_MEM_THRESHOLD=80
# HTTP responsiveness check settings
RESP_FAIL_LIMIT=2 # consecutive failed curl checks before restart
CURL_TIMEOUT=5 # seconds before curl gives up per check
# Tier 2 master toggle — false disables global scan entirely
WATCHDOG_SCAN_ALL=true
# Containers to skip in Tier 2 scan entirely
# Add intentionally stopped containers or containers managed by other systems
WATCHDOG_SCAN_IGNORE=(
# "container-name"
)
# Individual Tier 2 check toggles — disable checks that cause false positives
WATCHDOG_RESTART_UNHEALTHY=true # restart containers with unhealthy Docker health status
WATCHDOG_RESTART_DEAD=true # remove and restart containers in dead state
WATCHDOG_RESTART_CRASHED=true # restart containers that exited with non-zero exit code
WATCHDOG_NOTIFY_OOM=true # restart and notify when OOM killed by kernel
WATCHDOG_NOTIFY_CRASHLOOP=true # notify when Docker restart count is climbing
# Crash loop threshold — notify critical if Docker has restarted container this many times
WATCHDOG_CRASH_LIMIT=5
# Startup grace — skip restarts while system is still booting
# Prevents false positives while containers are coming up after array start
WATCHDOG_STARTUP_GRACE=600 # seconds after boot before watchdog acts on failures
# Restart loop protection — stops hammering broken containers
# After limit hit → skip list → notify critical → manual intervention needed
# Skip list auto-clears when container is found running again
WATCHDOG_CONTAINER_RESTART_LIMIT=3 # max watchdog restarts allowed in window
WATCHDOG_CONTAINER_RESTART_WINDOW=1 # hours — rolling window for restart count
WATCHDOG_CONTAINER_RESTART_LOG="/boot/config/container_restart_history.db"
# /boot/ survives reboots — bounded, auto-purges
# Dependency ordering — skip restarting a container if its dependency is also down
# Dependency gets restarted first, dependent picked up on the next watchdog cycle
# Prevents Authelia restarting before its database is ready — it would just fail again
# Format: ["dependent"]="dependency1 dependency2"
declare -A WATCHDOG_DEPENDENCIES=(
["Authelia"]="Mariadb-Authelia Redis-Authelia"
["Authelia-Secondary"]="Mariadb-Authelia Redis-Authelia-Secondary"
["NextCloud"]="Postgres-NextCloud"
)
# Notification batching — one clean summary per run instead of one ping per event
# true = batch all events into a single notification at end of run
# false = send individual notification per event as it happens
WATCHDOG_BATCH_NOTIFY=true
# ━━━ Docker Network Connect ━━━
# Connects containers to extra Docker networks on array start — many-to-many.
# Every container in the list connects to every network in the list.
# Useful when containers need to communicate across networks they were not originally
# configured with — e.g. memcached needing access to the nextcloud-aio network.
NETWORK_CONNECT_CONTAINERS=(
"memcached"
"Npm-CrowdSec"
)
NETWORK_CONNECT_NETWORKS=(
"high-availability" # Docker network name — must exist before array start
# containers needing their own network but accessible from main custom network
)
# ==============================================================================================
# ── UNRAID ESSENTIALS ─────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Reboot ━━━
# Seconds of warning broadcast to logged-in users before server_reboot.sh reboots.
# Gives users time to save work before the system goes down.
REBOOT_SLEEP=300
# ━━━ Mover ━━━
# Seconds to wait before mover_stop.sh sends SIGTERM to the mover process.
# Gives the mover time to finish its current file operation cleanly before being killed.
MOVER_STOP_TIMEOUT=300
# ━━━ Syslog Filter ━━━
# Path for the rsyslog filter file that suppresses Docker veth noise from syslog.
# Without this filter every Docker network interface change floods the syslog on boot.
FILTER_FILE="/etc/rsyslog.d/ignore-docker-veth.conf"
# ━━━ PHP-FPM ━━━
# Higher max_children allows more concurrent PHP requests to the unRAID WebGUI.
# Set based on available RAM — too high can cause memory pressure on low-RAM systems.
PHP_CONF="/etc/php-fpm.d/www.conf"
PHP_MAX_CHILDREN=250
# ━━━ Clear Logs ━━━
# System log files cleared weekly to prevent rootfs fill over time.
LOG_FILES=(/var/log/syslog /var/log/messages /var/log/dmesg)
# ━━━ WebGUI Watchdog ━━━
# Escalation: nginx restart → recheck → emhttp restart → recheck → notify warning.
# emhttp is the core unRAID daemon — restarting is more disruptive but recovers cleanly.
WEBGUI_URL="http://localhost" # adjust if running non-standard port
WEBGUI_TIMEOUT=5 # seconds before curl gives up on the WebGUI check
WEBGUI_NGINX_WAIT=15 # seconds to wait after nginx restart before rechecking
WEBGUI_EMHTTP_WAIT=30 # seconds to wait after emhttp restart — takes longer
# ==============================================================================================
# ── MEDIA ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Media Permissions ━━━
# Mode and owner applied recursively to all shares in MEDIA_PERMISSION_SHARES.
# Run by Media/media_shares_permissions.sh via the media_management.sh orchestrator.
# 777 and nobody:users is standard for unRAID media shares accessible by Docker containers.
# Applied recursively so large shares take time — run overnight via orchestrator.
PERMISSIONS_MODE="777"
PERMISSIONS_OWNER="nobody:users"
# Shares to apply permissions to — add or remove paths as your library grows.
MEDIA_PERMISSION_SHARES=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/appcache
/mnt/user/Books
/mnt/user/Downloads
/mnt/user/Games
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movie_Recordings
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Photo
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Temp_Storage
/mnt/user/Tv_Recordings
/mnt/user/Tv_Shows
/mnt/user/YouTube
)
# ━━━ Media Cleaner ━━━
# Removes junk files from media shares using configurable file pattern lists.
# Two profiles: anime and media — each with their own folder list and patterns.
# Run via Media/media_cleaner.sh anime or Media/media_cleaner.sh media
# Called automatically by media_management.sh via MEDIA_MAINTENANCE_JOBS below.
# Folders scanned by the anime profile — anime downloads commonly include these junk files
ANIME_CLEAN_FOLDERS=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
)
# Folders scanned by the media profile
MEDIA_CLEAN_FOLDERS=(
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Tv_Shows
)
# File patterns deleted by the anime profile — common junk from anime download groups
ANIME_FILE_PATTERNS=(
'*.sfv' '*.md5' '*.sha1' '*.txt' '*.url' '*.lnk'
'*.rar' '*.zip' '*.info' '*.torrent' '*.sample*' '*.proof*'
'*sync-conflict*' '*.scr' '*.srr' '*.exe' '*.webp'
'*.log' '*.json'
)
# File patterns deleted by the media profile
# Includes *.iso and *.lrc not needed in anime profile
MEDIA_FILE_PATTERNS=(
'*.sfv' '*.md5' '*.sha1' '*.txt' '*.url' '*.lnk'
'*.rar' '*.zip' '*.info' '*.torrent' '*.sample*' '*.proof*'
'*sync-conflict*' '*.scr' '*.srr' '*.exe' '*.webp'
'*.log' '*.json' '*.iso' '*.lrc'
)
# ━━━ Media Management Orchestrator ━━━
# Job list for Orchestrators/media_management.sh — runs scripts sequentially in order.
# Format: "folder/script.sh optional_argument"
# Order matters — permissions runs first so cleaners and arr scripts see correct ownership.
# Arr cleanup scripts run last — they depend on clean folders from the cleaner steps.
# Comment out any job to disable without removing it — easy to re-enable later.
MEDIA_MAINTENANCE_JOBS=(
"Media/media_shares_permissions.sh" # apply permissions first
"Media/media_cleaner.sh anime" # remove junk from anime shares
"Media/media_cleaner.sh media" # remove junk from media shares
"Media/lidarr_cleanup.sh" # remove orphaned music files
"Media/sonarr_cleanup.sh" # remove orphaned TV files
"Media/radarr_cleanup.sh" # remove orphaned movie files
)
# ━━━ Arr Cleanup ━━━
# Lidarr, Sonarr and Radarr orphan file cleanup via their respective APIs.
# Each arr script queries its API to get all tracked file paths, then compares against
# what exists on disk. Files not tracked by the arr and older than ORPHAN_AGE days are deleted.
#
# Each server runs different arrs managing different shares:
# HOST1: Sonarr (Tv_Shows), Radarr (Movies), Lidarr (Music)
# HOST2: Sonarr (Anime_Shows), Radarr (Anime_Movies)
#
# detect_hosts() selects the correct URL, API key, and root path at runtime.
# Scripts run identically on both servers — configuration drives behavior.
#
# Why the age threshold matters:
# The arr downloads a file then processes it — there is a window where the file exists
# on disk but the arr hasn't imported it yet. ORPHAN_AGE prevents deleting files that
# are mid-import. 7 days is conservative and safe for any normal workflow.
#
# Protected patterns are NEVER deleted regardless of tracking status or age.
# These protect arr-generated metadata (cover art, .nfo files, subtitles) that the arr
# depends on but does not include in its tracked file API response.
# ── Lidarr ────────────────────────────────────────────────────────────────────────────────────
# Lidarr runs on HOST1 only — music library management
# HOST2 does not run Lidarr — no HOST2 Lidarr config needed
HOST1_LIDARR_URL="http://192.168.50.2:8686"
HOST1_LIDARR_API_KEY="b2977e71ef074bc0a0529d9fcce3b2dc"
HOST1_LIDARR_MUSIC_ROOT="/mnt/user/Music-New" # must match root path set in Lidarr exactly
LIDARR_ORPHAN_AGE=7 # days before untracked file eligible for deletion
LIDARR_EXTENSIONS=("flac" "mp3" "m4a" "wav" "aac" "ogg" "opus" "wma")
LIDARR_PROTECTED_PATTERNS=("*.jpg" "*.jpeg" "*.png" "*.nfo" "*.lrc")
# never deleted — cover art, metadata, lyrics
# ── Sonarr ────────────────────────────────────────────────────────────────────────────────────
# HOST1 Sonarr manages Tv_Shows — HOST2 Sonarr manages Anime_Shows
# Scripts run on each server and hit their own local Sonarr instance
HOST1_SONARR_URL="http://192.168.50.2:8989"
HOST1_SONARR_API_KEY="130decd3db5b4c25afad64864cd03f9f"
HOST1_SONARR_TV_ROOT="/mnt/user/Tv_Shows" # must match root path set in HOST1 Sonarr exactly
HOST2_SONARR_URL="http://localhost:8989" # HOST2 local Sonarr — update with actual port
HOST2_SONARR_API_KEY="your-host2-sonarr-api-key"
HOST2_SONARR_TV_ROOT="/mnt/user/Anime_Shows" # must match root path set in HOST2 Sonarr exactly
SONARR_ORPHAN_AGE=7
SONARR_EXTENSIONS=("mkv" "mp4" "avi" "m4v" "ts" "wmv" "mov")
SONARR_PROTECTED_PATTERNS=("*.jpg" "*.jpeg" "*.png" "*.nfo" "*.srt" "*.sub" "*.ass" "*.ssa")
# never deleted — artwork, metadata, subtitles
# ── Radarr ────────────────────────────────────────────────────────────────────────────────────
# HOST1 Radarr manages Movies — HOST2 Radarr manages Anime_Movies
# Scripts run on each server and hit their own local Radarr instance
HOST1_RADARR_URL="http://192.168.50.2:7878"
HOST1_RADARR_API_KEY="d43a3ec6cf1549edb4af0cc63f98b2a9"
HOST1_RADARR_MOVIES_ROOT="/mnt/user/Movies" # must match root path set in HOST1 Radarr exactly
HOST2_RADARR_URL="http://localhost:7878" # HOST2 local Radarr — update with actual port
HOST2_RADARR_API_KEY="your-host2-radarr-api-key"
HOST2_RADARR_MOVIES_ROOT="/mnt/user/Anime_Movies" # must match root path set in HOST2 Radarr exactly
RADARR_ORPHAN_AGE=7
RADARR_EXTENSIONS=("mkv" "mp4" "avi" "m4v" "wmv" "mov")
RADARR_PROTECTED_PATTERNS=("*.jpg" "*.jpeg" "*.png" "*.nfo" "*.srt" "*.sub" "*.ass" "*.ssa")
# ━━━ Arr Import Recovery ━━━
# Automatically blocklists and re-searches failed imports and stalled downloads.
# Runs daily at 5am — by this time overnight downloads are complete and any
# failures have had time to surface. Items under ARR_IMPORT_RECOVERY_AGE are
# skipped — gives the arr time to retry on its own before we intervene.
#
# Targets two problem types from the queue API:
# importFailed — downloaded successfully but arr couldn't import it
# stalled — download stuck with no connections or progress
#
# Action: blocklist the release + remove from queue + trigger new search
# Blocklist prevents the same bad release being grabbed again
# New search finds a different release automatically
# If all releases are bad → arr will exhaust options, manual check needed
# Notification tells you what was actioned so you can monitor
#
# Per-arr toggles — disable temporarily if an arr is having issues
# detect_hosts() selects correct URL and API key per server at runtime
# Lidarr runs on HOST1 only — exits cleanly on HOST2
ARR_IMPORT_RECOVERY_AGE=12 # hours — skip items newer than this, give arr time to retry
# Per-arr enable/disable
HOST1_SONARR_RECOVERY=true # Tv_Shows import recovery
HOST1_RADARR_RECOVERY=true # Movies import recovery
HOST1_LIDARR_RECOVERY=true # Music import recovery — HOST1 only
HOST2_SONARR_RECOVERY=true # Anime_Shows import recovery
HOST2_RADARR_RECOVERY=true # Anime_Movies import recovery
# ==============================================================================================
# ── TRANSCODES ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Session-based storage allocator using filesystem symlink indirection.
# ffmpeg resolves the symlink ONCE at session start — existing sessions are never affected.
# Only new sessions care about where the symlink currently points.
#
# How it works:
# ramdisk_setup.sh — run once at array start, creates tmpfs and sets symlink
# transcode_management.sh — every 3 min, runs cleanup then manager in correct order
# transcode_cleanup.sh — called by transcode_management.sh — removes old inactive files
# transcode_manager.sh — called by transcode_management.sh — manages symlink direction
#
# ⚠️ Docker mount warning:
# Mount must use shared propagation so symlink flips are visible inside the container.
# In unRAID Extra Parameters — do NOT use standard path mapping for this mount:
# --mount type=bind,source=/mnt/ram-transcode,target=/ext-ram-transcode,bind-propagation=shared
# Standard bind mounts use rprivate — Docker locks the inode on first symlink flip
# and new sessions land on SSD permanently for that container run.
RAMDISK_PATH="/mnt/ramdisk_transcodes" # tmpfs mount point created at array start
RAMDISK_SIZE="8G" # ceiling — tmpfs only uses RAM actually needed
TRANSCODE_LINK="/mnt/ram-transcode" # symlink Emby points at — location never changes
TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/" # SSD fallback location
# Usage thresholds in GB — hysteresis gap between WARN and LOW prevents flip-flop
RAMDISK_WARN_GB=6.8 # flip symlink to SSD at or above this usage
RAMDISK_LOW_GB=5.5 # flip symlink back to ramdisk when usage drops here
RAMDISK_SSD_MIN_GB=20 # minimum free GB on SSD required before allowing flip to SSD
# Cleanup age thresholds — files must be older than these AND not open by any process
TRANSCODE_MAX_AGE=20 # minutes before a transcode file is eligible for cleanup
TRANSCODE_ORPHAN_AGE=30 # minutes before an orphaned file is eligible — extra caution buffer
# Flip frequency alert — too many flips per hour indicates ramdisk needs to be larger
TRANSCODE_FLIP_WARN=3 # notify if symlink flips this many times in one hour
# Permissions — must match your Emby container user
TRANSCODE_OWNER="nobody:users"
TRANSCODE_CHMOD="755" # renamed from TRANSCODE_MODE to avoid ambiguity with manager mode
# Operating mode — controls symlink routing behavior
# smart — auto-flips between ramdisk and SSD based on usage thresholds (default)
# ramdisk — always uses ramdisk, never flips to SSD regardless of usage
# useful when load is light and you want guaranteed ramdisk performance
# warns if usage exceeds threshold but does not flip
# ssd — always uses SSD, never uses ramdisk
# useful during ramdisk maintenance, testing, or after a flip issue
# switch to this mode to drain ramdisk sessions gracefully
TRANSCODE_MANAGER_MODE="smart" # smart | ramdisk | ssd
# ━━━ Transcode Server Array ━━━
# All media servers that share the ramdisk transcode space.
# transcode_manager.sh reads this array and queries each server's API for active sessions.
# Session display, storage detection, and threshold decisions cover ALL servers combined.
#
# One entry → works exactly as before — single server behavior
# Many entries → aggregates sessions from all servers, one threshold on total ramdisk usage
#
# Each server writes to its own subfolder inside transcoding-temp:
# /mnt/ram-transcode/transcoding-temp/ABC123/ ← Emby session
# /mnt/ram-transcode/transcoding-temp/XYZ789/ ← Jellyfin session
# They never touch each other's files — the ramdisk is shared scratch space.
#
# Docker Extra Parameters — each media server container that uses the ramdisk needs:
# --mount type=bind,source=/mnt/ram-transcode,target=/ext-ram-transcode,bind-propagation=shared
# The target path (/ext-ram-transcode) must match what's configured in each server's
# transcoding settings. Use a different target path per container if needed.
#
# Format: "ContainerName|URL|APIKey|Type"
# ContainerName — exact Docker container name (used for running check)
# URL — API base URL including port
# APIKey — server API key (Emby/Jellyfin token, Plex token etc.)
# Type — emby | jellyfin | plex (controls API endpoint format)
#
# The first entry is the primary server — used for container running check
# (TRANSCODE_CHECK_EMBY still applies to first entry).
# Additional entries are queried if their container is running.
# Entries with placeholder APIKey values are skipped automatically.
#
# ⚠️ Tdarr does NOT belong here — Tdarr encodes full files, not HLS segments.
# Large working files would fill the ramdisk rapidly and cause constant flips.
# Keep Tdarr on SSD. Use tdarr_cleanup.sh for Tdarr orphan management.
TRANSCODE_SERVERS=(
"${HOST1_EMBY_CONTAINER}|${HOST1_EMBY_URL}|${HOST1_EMBY_API_KEY}|emby"
# "${HOST2_EMBY_CONTAINER}|${HOST2_EMBY_URL}|${HOST2_EMBY_API_KEY}|emby" # his Emby temporarily
# "Jellyfin|http://localhost:8097|jellyfin-api-key|jellyfin" # test Jellyfin instance
# "Plex|http://localhost:32400|plex-token|plex" # Plex if needed
)
# Container running check — still applies — skips threshold checks when no servers active
# Checked against the first entry in TRANSCODE_SERVERS automatically
TRANSCODE_CHECK_EMBY=true
# Daily transcode statistics log — read by weekly_health_digest.sh
# Tracks peak usage, flip count, session ratio, files cleaned per day
# Bounded to TRANSCODE_LOG_RETENTION days — auto-purges old entries
TRANSCODE_DAILY_LOG="/boot/config/transcode_daily.db"
TRANSCODE_LOG_RETENTION=90
# ==============================================================================================
# ── MONITORS ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
# Checks SSL cert expiry via direct openssl connection — no NPM dependency.
# Reads the actual cert the server is presenting — catches real-world issues API checks miss.
# Each domain and subdomain is a separate entry — they have independent certs.
CERT_MONITOR_DOMAINS=(
"Gmer4Lfe.com"
"Gmer4Lfe.us"
)
CERT_WARN_DAYS=30 # notify warning when cert expires within this many days
CERT_CRIT_DAYS=7 # notify critical when cert expires within this many days
CERT_TIMEOUT=10 # seconds before openssl connection attempt gives up per domain
# ━━━ Backup Verify ━━━
# Verifies the rsync mirror is healthy by comparing random file checksums between servers.
# Uses existing SSH keys — no additional configuration needed beyond the share list.
# Leave BACKUP_VERIFY_SHARES empty to automatically use the local host's daily sync shares
# (HOST1_DAILY_SYNC_SHARES or HOST2_DAILY_SYNC_SHARES based on detect_hosts()).
BACKUP_VERIFY_SHARES=(
# leave empty to use host-specific daily sync shares automatically
)
BACKUP_VERIFY_SAMPLE=10 # number of files to randomly sample per share per run
BACKUP_VERIFY_MIN_SIZE=1M # skip files smaller than this — avoids tiny junk files
# ━━━ SMART Health ━━━
# Monitors drive SMART attributes — reads live from each drive, no persistent writes.
# Discovers all drives automatically via /dev/sd* and /dev/nvme* — no drive list needed.
# Add drives to SMART_IGNORE_DRIVES to skip specific drives (e.g. your unRAID boot USB).
SMART_TEMP_WARN=45 # degrees C — warn if drive temperature exceeds this
SMART_TEMP_CRIT=55 # degrees C — critical if drive temperature exceeds this
SMART_IGNORE_DRIVES=(
"sda" # boot USB — SMART not meaningful on flash drives
)
# ━━━ ZFS Memory Snapshot ━━━
# Weekly ZFS pool health and memory diagnostic report — informational only, no action taken.
# system_watchdog.sh handles threshold-based intervention.
# Output written to ZFS_REPORT_LOG for historical review in addition to console output.
# Pools in ZFS_REPORT_IGNORE_POOLS are excluded from reporting — still monitored by unRAID.
ZFS_REPORT_LOG="/var/log/zfs-weekly-health.log"
ZFS_REPORT_ARC_WARN_PCT=90 # warn in report if ARC utilization above this %
ZFS_REPORT_FREE_WARN_GB=10 # warn in report if free RAM drops below this GB
ZFS_REPORT_AVAIL_WARN_GB=20 # warn in report if available RAM drops below this GB
ZFS_REPORT_DOCKER_TOP=10 # number of top Docker memory users to show in report
ZFS_REPORT_IGNORE_POOLS=(
# Pools excluded from health reporting — expected to run at high usage
# All pools still monitored by unRAID regardless of this list
"disk10"
"disk9"
"disk8"
"disk6"
"disk5"
)
# ━━━ Bandwidth Monitor ━━━
# Called by rsync.sh after each sync — one bounded write per run, minimal flash wear.
# Log format: YYYY-MM-DD|HH:MM|profile|duration_seconds|status — version-proof
# File stays bounded to BANDWIDTH_LOG_RETENTION days — old entries auto-purged on write.
BANDWIDTH_LOG="/boot/config/bandwidth_history.db"
BANDWIDTH_LOG_RETENTION=90 # days to keep — file never grows beyond ~90 lines
BANDWIDTH_WARN_GB=50 # flag in reports if a single sync transfer exceeds this GB
# ━━━ Health Digest ━━━
# Aggregated system health summary — reads existing state files, no new writes to flash.
# Three profiles — switch by changing DIGEST_PROFILE, no cron changes needed:
# always — sends every run
# smart — sends only if findings worth reporting
# weekly — sends once per week on DIGEST_DAY only
DIGEST_PROFILE="weekly" # always | smart | weekly
DIGEST_DAY="Sunday" # must match date +%A output
# Smart profile triggers — set true to send digest when this condition is found
DIGEST_SMART_ON_WATCHDOG=true # send if any watchdog strikes are active
DIGEST_SMART_ON_FAILOVER=true # send if failover state is not NORMAL
DIGEST_SMART_ON_CERT_WARN=true # send if any cert is under CERT_WARN_DAYS
DIGEST_SMART_ON_BANDWIDTH=true # send if any transfer exceeded BANDWIDTH_WARN_GB
# ━━━ Critical Shares Maintenance ━━━
# Controls container update behaviour in critical_shares_maintenance.sh
# Both local and remote containers are already stopped for the sync window
# Updates pull new images while containers are down — starts fresh on new version
#
# CRITICAL_SYNC_UPDATES — pull updates on LOCAL server during maintenance window
# CRITICAL_SYNC_UPDATES_REMOTE — pull updates on REMOTE server during maintenance window
#
# Both false → sync only, no updates (sync only, no updates)
# Both true → full maintenance both servers — recommended for Sunday window
# Toggle false temporarily to skip updates without changing the schedule
CRITICAL_SYNC_UPDATES=true # pull container updates locally during maintenance
CRITICAL_SYNC_UPDATES_REMOTE=true # pull container updates on remote via SSH
# ━━━ Emby ━━━
# ━━━ Emby Session Report ━━━
# No persistent writes — queries fresh each run.
# Emby URL and API key pulled from HOST1/HOST2_EMBY_URL and HOST1/HOST2_EMBY_API_KEY
# defined in Host Configuration at the top of this file.
EMBY_REPORT_DAYS=7 # number of days to include in the report period
EMBY_REPORT_TOP_N=10 # number of top content items to show in report
# ==============================================================================================
# ── SYSTEM WATCHDOG ───────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Last line of defense — reboots cleanly when system is about to become unstable.
# Strike system: sustained threshold hits trigger reboot — single spikes ignored.
# Reboot loop protection: shuts down instead if reboot limit hit in window.
# ━━━ State Files ━━━
# Strike counts reset on reboot — /tmp is correct (fresh start after each reboot)
SYS_WATCHDOG_STATE_FILE="/tmp/system_watchdog_state.db"
# Persistent container skip list — on /boot/ so it survives reboots
# Containers added here when docker_watchdog.sh exhausts all restart attempts
# Auto-clears when container is found running again after reboot or manual fix
SYS_WATCHDOG_FAILED_FILE="/boot/config/system_watchdog_failed.db"
# Reboot timestamp log — on /boot/ for reboot loop detection across reboots
SYS_WATCHDOG_REBOOT_LOG="/boot/config/system_watchdog_reboots.db"
# ━━━ Strike and Reboot Loop Settings ━━━
# Consecutive threshold hits required before triggering reboot
SYS_WATCHDOG_STRIKE_LIMIT=2
# Maximum reboots allowed within the window before shutting down instead
# A reboot loop means something fundamental is broken that rebooting is not fixing
SYS_WATCHDOG_REBOOT_LIMIT=3
# Window in hours — controls BOTH the reboot count window AND the rolling log purge
# Entries older than this many hours are automatically removed from the reboot log
SYS_WATCHDOG_REBOOT_WINDOW_HRS=12
# ━━━ Thresholds ━━━
# Set these at "I am about to become unstable" levels — not just "things are a bit high"
SYS_WATCHDOG_ROOTFS_PCT=95 # rootfs % — at 95% something is seriously wrong
SYS_WATCHDOG_LOG_PCT=95 # /var/log % — log spam filling the filesystem
SYS_WATCHDOG_MEM_GB=4 # free RAM GB — 4GB free on 128GB system is critical
SYS_WATCHDOG_ARC_PINNED_PCT=98 # ZFS ARC % of max before attempting reclaim
SYS_WATCHDOG_ARC_RELEASE_PCT=95 # ZFS ARC % after reclaim that still triggers reboot
SYS_WATCHDOG_LOAD_MULTIPLIER=3 # strike if load avg > cores x this multiplier
SYS_WATCHDOG_ZOMBIE_LIMIT=50 # zombie process count before strike
SYS_WATCHDOG_CPU_TEMP_MAX=95 # degrees C — adjust for your specific CPU tjmax
# ━━━ Check Toggles ━━━
# true = run this check on every watchdog cycle / false = skip entirely
# Disable checks not relevant to your hardware or that cause false positives
SYS_WATCHDOG_CHECK_ROOTFS=true
SYS_WATCHDOG_CHECK_LOG=true
SYS_WATCHDOG_CHECK_RAM=true
SYS_WATCHDOG_CHECK_ARC=true
SYS_WATCHDOG_CHECK_CPU_TEMP=true
SYS_WATCHDOG_CHECK_LOAD=false # disabled — load spikes during transcoding are normal
SYS_WATCHDOG_CHECK_ZOMBIES=true
SYS_WATCHDOG_CHECK_CONTAINERS=true # checks persistent skip list from docker_watchdog.sh
SYS_WATCHDOG_CHECK_DOCKER_DAEMON=true
# ━━━ Abort Toggles ━━━
# Controls whether certain conditions prevent a reboot from happening.
# true = abort reboot if this condition is active (conservative — default)
# false = reboot anyway regardless of this condition (aggressive)
# Philosophy: a graceful reboot before crash is always better than a hard crash mid-operation
SYS_WATCHDOG_ABORT_ON_ZFS_UNHEALTHY=true # unhealthy pool + reboot risks data loss
SYS_WATCHDOG_ABORT_ON_PARITY=false # aborting parity check beats crashing mid-check
SYS_WATCHDOG_ABORT_ON_MOVER=false # aborting mover beats crashing mid-move
# ==============================================================================================
# ──────────────────────── End Of User Variables ───────────────────────────────────────────────
# ==============================================================================================