856 lines
47 KiB
Bash
856 lines
47 KiB
Bash
#!/bin/bash
|
|
# ==============================================================================================
|
|
# ================================= MASTER CONFIGURATION =======================================
|
|
# ==============================================================================================
|
|
# All user-facing variables for the unRAID script ecosystem.
|
|
# Scripts source this file — edit here, changes apply everywhere on next git pull.
|
|
#
|
|
# ── HOW THIS FILE WORKS ───────────────────────────────────────────────────────────────────────
|
|
# Every script sources Master.conf and common.sh at startup.
|
|
# Change a value here and it affects all scripts that use it — no hunting through files.
|
|
# To disable something: comment it out with # rather than deleting it.
|
|
# To add a new rsync profile: add a key to each PROFILE_* array.
|
|
# To add a new media maintenance job: add a line to MEDIA_MAINTENANCE_JOBS.
|
|
#
|
|
# ── INDEX ─────────────────────────────────────────────────────────────────────────────────────
|
|
#
|
|
# Section Description
|
|
# ───────────────────────────────────────────────────────────────────────────────────────────
|
|
# HOST CONFIGURATION Server hostnames and SSH key paths
|
|
# LOGGING Enable or disable verbose logging
|
|
# NOTIFICATIONS unRAID native and Discord webhook settings
|
|
# GIT / REPO Gitea repository and SSH settings
|
|
#
|
|
# ── RSYNC ──────────────────────────────────────────────────────────────────────────────────
|
|
# RSYNC DEFAULTS Global fallback rsync settings
|
|
# REMOTE HEALTH CHECKS Rootfs threshold for pre-flight abort
|
|
# DAILY SYNC SHARES Media shares synced by daily_sync.sh
|
|
# RSYNC PROFILE SYSTEM Per-profile overrides (appdata profiles)
|
|
#
|
|
# ── FAILOVER ───────────────────────────────────────────────────────────────────────────────
|
|
# FAILOVER Mutual container failover between two servers
|
|
#
|
|
# ── DOCKER ESSENTIALS ──────────────────────────────────────────────────────────────────────
|
|
# DOCKER DAILY RESTART Containers restarted daily
|
|
# DOCKER WEEKLY RESTART Containers restarted weekly
|
|
# DOCKER WATCHDOG Container health monitoring — memory, CPU, HTTP
|
|
# DOCKER NETWORK CONNECT Connect containers to extra networks on boot
|
|
#
|
|
# ── UNRAID ESSENTIALS ──────────────────────────────────────────────────────────────────────
|
|
# REBOOT User warning delay before scheduled reboot
|
|
# MOVER Mover stop timeout
|
|
# SYSLOG FILTER Docker veth noise filter file path
|
|
# PHP-FPM PHP-FPM max children config
|
|
# CLEAR LOGS System log file paths
|
|
# WEBGUI WATCHDOG WebGUI nginx + emhttp monitoring and restart
|
|
#
|
|
# ── MEDIA ──────────────────────────────────────────────────────────────────────────────────
|
|
# MEDIA PERMISSIONS Share list, mode and owner for permissions script
|
|
# MEDIA CLEANER Anime and media folder lists and file patterns
|
|
# MEDIA MANAGEMENT Orchestrator job list for media_management.sh
|
|
# ARR CLEANUP Lidarr, Sonarr, Radarr orphan file cleanup
|
|
#
|
|
# ── TRANSCODES ─────────────────────────────────────────────────────────────────────────────
|
|
# TRANSCODE MANAGER Ramdisk and SSD fallback transcode management
|
|
#
|
|
# ── MONITORS ────────────────────────────────────────────────────────────────────────────────
|
|
# CERTIFICATE MONITOR SSL certificate expiry monitoring
|
|
# BACKUP VERIFY Random sample checksum verification against remote
|
|
# SMART HEALTH Drive SMART attribute monitoring
|
|
# BANDWIDTH MONITOR Daily rsync transfer logging and weekly summary
|
|
# HEALTH DIGEST Aggregated system health digest — always/smart/weekly
|
|
# EMBY SESSION REPORT Weekly Emby usage statistics via API
|
|
#
|
|
# ── SYSTEM WATCHDOG ────────────────────────────────────────────────────────────────────────
|
|
# SYSTEM WATCHDOG System health monitoring — last line of defense
|
|
#
|
|
# ==============================================================================================
|
|
|
|
# ━━━ Host Configuration ━━━
|
|
# Hostnames must match Tailscale machine names exactly — case sensitive.
|
|
# Used by detect_hosts() in common.sh to determine which server is local and which is remote.
|
|
# Both servers run identical scripts — host detection makes them bidirectional.
|
|
HOST1="unRAID-Gmer4Lfe"
|
|
HOST2="unRAID-Jayred365"
|
|
|
|
# SSH keys for server-to-server rsync and failover container operations.
|
|
# HOST1_SSH_KEY is used when HOST1 SSHes to HOST2 and vice versa.
|
|
# Both keys must be in /root/.ssh/ and authorised in the remote server's authorized_keys.
|
|
HOST1_SSH_KEY="/root/.ssh/Gmer4Lfe-rsync-key"
|
|
HOST2_SSH_KEY="/root/.ssh/Jayred365-rsync-key"
|
|
|
|
# ━━━ Logging ━━━
|
|
# Controls verbose [LOG] output across all scripts.
|
|
# true = show detailed [LOG] lines — useful for debugging or first-time setup
|
|
# false = show only user-facing output — cleaner for scheduled runs
|
|
ENABLE_LOGGING=true
|
|
|
|
# ━━━ Notifications ━━━
|
|
# Two independent notification channels — either or both can be active simultaneously.
|
|
|
|
# unRAID native notification system — integrates with the bell icon in the WebGUI.
|
|
# Recommended: set Settings → Notification Settings to errors/warnings only so
|
|
# normal completions don't create noise. The ecosystem sends:
|
|
# normal — job completed successfully (informational)
|
|
# warning — something failed or needs attention
|
|
NOTIFY_UNRAID=true
|
|
|
|
# Discord webhook URL — paste the full webhook URL from your Discord server settings.
|
|
# Leave blank to disable Discord notifications entirely.
|
|
DISCORD_WEBHOOK=""
|
|
|
|
# ━━━ Git / Repo ━━━
|
|
# Gitea self-hosted repository settings used by git_pull_execute.sh.
|
|
# Running git_pull_execute.sh on either server pulls the latest scripts and sets
|
|
# executable permissions automatically — keeps both servers in sync with one command.
|
|
REPO_SSH="git@192.168.50.2:FailedProxy/Unraid_Scripts.git"
|
|
TARGET_DIR="/mnt/user/appdata/unraid_scripts"
|
|
GITEA_SSH_KEY="/root/.ssh/unraid_gitea" # SSH key for authenticating to Gitea
|
|
SSH_PORT=221 # Gitea SSH port — default Gitea uses 22
|
|
|
|
# ==============================================================================================
|
|
# ── RSYNC ─────────────────────────────────────────────────────────────────────────────────────
|
|
# ==============================================================================================
|
|
|
|
# ━━━ Rsync Defaults ━━━
|
|
# Global fallback values used when no profile match is found for a directory.
|
|
# Shares in DAILY_SYNC_SHARES always use these globals — no profile is defined for them.
|
|
# Appdata shares (Arrs_Stack, Critical-Data etc.) match profiles by directory basename.
|
|
# If a profile key exists in a PROFILE_* array that value overrides the global.
|
|
# If a profile key is missing the global below is used as the fallback.
|
|
|
|
BW_LIMIT=12500 # network transfer speed cap in KB/s — 12500 ≈ 100Mbit
|
|
RETRY_COUNT=3 # number of retry attempts if rsync fails before giving up
|
|
SLEEP=300 # seconds to wait between retry attempts
|
|
CRITICAL_CONTAINER_NAMES=() # containers to stop on REMOTE before rsync — profiles override
|
|
DELAYED_CONTAINERS=() # containers needing delay before starting — profiles override
|
|
CONTAINER_DELAY=5 # seconds to wait before starting delayed containers
|
|
EXCLUDE_DIRS=() # directories to exclude from transfer — profiles override
|
|
|
|
# Default rsync options used when no profile match is found.
|
|
# --delete removes files on remote that no longer exist on source (mirror behaviour)
|
|
# --inplace writes directly to destination file — better for large files, avoids temp copies
|
|
# --no-whole-file forces delta transfer even on fast local-like connections
|
|
DEFAULT_RSYNC_OPTS=(-av --info=progress2 --human-readable --bwlimit="$BW_LIMIT" --delete --inplace --no-whole-file)
|
|
|
|
# ━━━ Remote Health Checks ━━━
|
|
# Pre-flight check run before every rsync — aborts if remote rootfs (/) usage is at or
|
|
# above this percentage. When the remote array is down or drives are missing, rsync
|
|
# writes land on rootfs instead of /mnt/user — this fills the filesystem rapidly and
|
|
# can crash the remote server. 75% gives headroom to detect the problem early.
|
|
ROOTFS_WARN=75
|
|
|
|
# ━━━ Daily Sync Shares ━━━
|
|
# Media shares synced once daily by Orchestrators/daily_sync.sh.
|
|
# These shares have no profile entry — all use DEFAULT_RSYNC_OPTS above.
|
|
# Add or remove paths here to control what syncs each night.
|
|
# For shares needing custom bandwidth or container stops — create a profile below instead.
|
|
DAILY_SYNC_SHARES=(
|
|
/mnt/user/Anime_Movies-Old
|
|
/mnt/user/Anime_Shows-Old
|
|
/mnt/user/Anime_Shows
|
|
/mnt/user/Books
|
|
/mnt/user/Intros
|
|
/mnt/user/Kids_Movies
|
|
/mnt/user/Kids_Tv_Shows
|
|
/mnt/user/Movies
|
|
/mnt/user/Music_Videos
|
|
/mnt/user/Nextcloud
|
|
/mnt/user/stand-up_comedy
|
|
/mnt/user/Tv_Shows
|
|
)
|
|
|
|
# ━━━ Rsync Profile System ━━━
|
|
# Profiles allow per-share rsync behaviour without touching script logic.
|
|
# The profile key is matched automatically by the basename of the directory
|
|
# passed to rsync.sh (lowercased).
|
|
#
|
|
# Example:
|
|
# rsync.sh /mnt/user/appdata-Failover/Arrs_Stack
|
|
# basename = Arrs_Stack → lowercased = arrs_stack → matches [arrs_stack] profile
|
|
#
|
|
# To add a new profile:
|
|
# 1. Add a key to each PROFILE_* array below using your chosen name
|
|
# 2. Call rsync.sh with a directory whose basename matches that key
|
|
# 3. Any array you omit falls back to its global default automatically
|
|
#
|
|
# IMPORTANT: PROFILE_RSYNC_OPTS does NOT inherit from DEFAULT_RSYNC_OPTS.
|
|
# If you define a profile entry you must list ALL desired options explicitly.
|
|
#
|
|
# Current profiles:
|
|
# arrs_stack — Sonarr, Radarr, Lidarr, Readarr, Prowlarr, Bazarr, Pinchflat
|
|
# Lower bandwidth — runs alongside media syncs
|
|
# Containers stopped during sync for data consistency
|
|
# critical-data — Auth stack: NPM, Authelia, Mariadb-Authelia, Redis-Authelia, LLDAP
|
|
# High bandwidth — small data, synced frequently
|
|
# Authelia needs delayed start — database containers must be ready first
|
|
# gmer4lfe — Server-specific appdata: Organizr, UptimeKuma, VaultWarden
|
|
# Medium bandwidth — personal services, no container stop needed
|
|
# important-data — NextCloud + Postgres database
|
|
# High bandwidth — NextCloud needs graceful stop before sync
|
|
# NextCloud needs delayed start — Postgres must be accepting connections
|
|
# emby — Emby media server appdata and metadata only
|
|
# Medium bandwidth — large appdata directory, no containers stopped
|
|
|
|
# Rsync options per profile — replaces DEFAULT_RSYNC_OPTS entirely for that profile run
|
|
# SPACE-SEPARATED STRINGS — converted to array at runtime by rsync.sh
|
|
declare -A PROFILE_RSYNC_OPTS=(
|
|
[arrs_stack]="-av --info=progress2 --human-readable --bwlimit=$BW_LIMIT --delete --inplace"
|
|
[critical-data]="-av --human-readable --bwlimit=$BW_LIMIT --delete"
|
|
[gmer4lfe]="-av --info=progress2 --bwlimit=$BW_LIMIT"
|
|
[important-data]="-av --human-readable --bwlimit=$BW_LIMIT"
|
|
[emby]="-av --human-readable --bwlimit=$BW_LIMIT"
|
|
)
|
|
|
|
# Bandwidth limit in KB/s per profile — overrides global BW_LIMIT for this profile only
|
|
declare -A PROFILE_BW_LIMIT=(
|
|
[arrs_stack]=5000 # lower — runs alongside other jobs, avoids saturating link
|
|
[critical-data]=9500 # high — small data, get it synced fast
|
|
[gmer4lfe]=8000 # medium
|
|
[important-data]=9500 # high — database sync needs to be fast and clean
|
|
[emby]=8000 # medium — large files, steady sustained transfer
|
|
)
|
|
|
|
# Retry attempts per profile before giving up — overrides global RETRY_COUNT
|
|
declare -A PROFILE_RETRY_COUNT=(
|
|
[arrs_stack]=3
|
|
[critical-data]=3
|
|
[gmer4lfe]=3
|
|
[important-data]=3
|
|
[emby]=3
|
|
)
|
|
|
|
# Seconds between retry attempts — overrides global SLEEP
|
|
declare -A PROFILE_SLEEP=(
|
|
[arrs_stack]=300
|
|
[critical-data]=300
|
|
[gmer4lfe]=300
|
|
[important-data]=300
|
|
[emby]=300
|
|
)
|
|
|
|
# Containers stopped on REMOTE before rsync and restarted after completion.
|
|
# Only include containers that need to be stopped for data consistency.
|
|
# Databases and auth stacks need clean state — media servers generally do not.
|
|
# SPACE-SEPARATED STRINGS — converted to array at runtime
|
|
declare -A PROFILE_CRITICAL_CONTAINER_NAMES=(
|
|
[arrs_stack]="Sonarr Lidarr Readarr Radarr Prowlarr Bazarr Pinchflat"
|
|
[critical-data]="Mariadb-Authelia Redis-Authelia Lldap-Gmer4Lfe NginxProxyManager Authelia"
|
|
[gmer4lfe]="Organizrv2-Gmer4Lfe UptimeKuma-Gmer4Lfe VaultWarden-Gmer4Lfe"
|
|
[important-data]="Postgres-NextCloud NextCloud"
|
|
[emby]=""
|
|
)
|
|
|
|
# Containers that need a delay before starting after rsync completes.
|
|
# Used when a container depends on another that was also stopped — it needs its
|
|
# dependency to be ready before it can start successfully.
|
|
# SPACE-SEPARATED STRINGS — converted to array at runtime
|
|
declare -A PROFILE_DELAYED_CONTAINERS=(
|
|
[arrs_stack]=""
|
|
[critical-data]="Authelia" # Authelia needs Mariadb + Redis ready before starting
|
|
[gmer4lfe]=""
|
|
[important-data]="NextCloud" # NextCloud needs Postgres accepting connections first
|
|
[emby]=""
|
|
)
|
|
|
|
# Seconds to wait before starting delayed containers — gives dependencies time to initialise
|
|
declare -A PROFILE_CONTAINER_DELAY=(
|
|
[arrs_stack]=5
|
|
[critical-data]=10 # 10s gives Mariadb and Redis time to accept connections
|
|
[gmer4lfe]=5
|
|
[important-data]=10 # 10s gives Postgres time to accept connections
|
|
[emby]=5
|
|
)
|
|
|
|
# Directories excluded from rsync transfer per profile.
|
|
# logs and *.tmp are safe to exclude — they are ephemeral and regenerated on container start.
|
|
# SPACE-SEPARATED STRINGS — converted to array at runtime
|
|
declare -A PROFILE_EXCLUDE_DIRS=(
|
|
[arrs_stack]="logs *.tmp"
|
|
[critical-data]="logs *.tmp"
|
|
[gmer4lfe]="logs *.tmp"
|
|
[important-data]="logs *.tmp"
|
|
[emby]="logs *.tmp"
|
|
)
|
|
|
|
# Per-disk check toggle — controls whether rsync.sh runs check_remote_disks() for this profile.
|
|
# true = skip per-disk check — use when remote share lives on a ZFS pool
|
|
# ZFS pools don't have /mnt/disk* structure so the check always fails incorrectly
|
|
# false = run per-disk check — use for traditional unRAID array with individual disk mounts
|
|
# Verifies all disks backing the share are online before syncing
|
|
# Note: rootfs and share existence checks always run regardless of this setting
|
|
declare -A PROFILE_SKIP_DISK_CHECK=(
|
|
[arrs_stack]=true # remote uses ZFS pool — no individual disk mounts
|
|
[critical-data]=true
|
|
[gmer4lfe]=true
|
|
[important-data]=true
|
|
[emby]=true
|
|
)
|
|
|
|
# ==============================================================================================
|
|
# ── FAILOVER ──────────────────────────────────────────────────────────────────────────────────
|
|
# ==============================================================================================
|
|
# Mutual container failover between two unRAID servers 50 miles apart.
|
|
# Each server runs Failover/failover.sh independently — no coordination between servers.
|
|
# All decisions are based solely on two ping checks: remote reachable + internet reachable.
|
|
#
|
|
# States:
|
|
# NORMAL — remote up, internet up — own containers only, silent operation
|
|
# FAILOVER — remote down, internet up — start remote's containers locally (additive)
|
|
# NO_INTERNET — internet down — stop public-facing containers, wait for recovery
|
|
# DARK — remote down + internet down — same actions as NO_INTERNET
|
|
#
|
|
# Handback: strike confirmation → pre-flight → rsync → start remote → stop local
|
|
|
|
EXTERNAL_IP="8.8.8.8" # external IP to ping for internet connectivity check
|
|
FAILOVER_CHECK_INTERVAL=120 # seconds between state checks
|
|
FAILOVER_HANDBACK_STRIKES=2 # consecutive remote-up confirmations before handback
|
|
FAILOVER_STATE_FILE="/boot/config/failover_state.db"
|
|
|
|
# ━━━ Failover Test ━━━
|
|
# Used by Failover/failover_test.sh — controlled simulation of the failover lifecycle.
|
|
# failover_test.sh blocks remote connectivity via iptables then observes failover.sh behavior.
|
|
# All failover logic stays in failover.sh — test script is the harness only.
|
|
#
|
|
# ⚠️ Run during a maintenance window — real containers start and stop during the test.
|
|
# Use --dry-run first to walk through phases without touching anything.
|
|
|
|
# Seconds to hold the iptables block — must be longer than FAILOVER_CHECK_INTERVAL
|
|
# so failover.sh has time to detect the outage and change state
|
|
FAILOVER_TEST_BLOCK_WAIT=150 # 150s = FAILOVER_CHECK_INTERVAL + 30s buffer
|
|
|
|
# Seconds to wait for handback after restoring connectivity
|
|
# Must cover FAILOVER_HANDBACK_STRIKES x FAILOVER_CHECK_INTERVAL plus rsync time
|
|
# 2 strikes x 120s = 240s minimum — add buffer for rsync handback jobs
|
|
FAILOVER_TEST_HANDBACK_WAIT=360 # 360s = 6 minutes — adjust if rsync takes longer
|
|
|
|
|
|
FAILOVER_HOST1_STARTS_FOR_HOST2=(
|
|
"Gmer4Lfe.com"
|
|
"Gmer4Lfe.us"
|
|
)
|
|
FAILOVER_HOST1_STOP_ON_NO_NET=(
|
|
"Gmer4Lfe.com"
|
|
"Gmer4Lfe.us"
|
|
)
|
|
FAILOVER_HOST1_RSYNC_JOBS=(
|
|
# "/mnt/user/appdata-Failover/Jayred365"
|
|
# "/mnt/user/Media_Server/Emby-Jayred"
|
|
)
|
|
|
|
# HOST2 (unRAID-Jayred365 — Secondary)
|
|
FAILOVER_HOST2_STARTS_FOR_HOST1=(
|
|
"Emby"
|
|
"Gmer4Lfe.com"
|
|
"Gmer4Lfe.us"
|
|
)
|
|
FAILOVER_HOST2_STOP_ON_NO_NET=(
|
|
"Gmer4Lfe.com"
|
|
"Gmer4Lfe.us"
|
|
)
|
|
FAILOVER_HOST2_RSYNC_JOBS=(
|
|
# "/mnt/user/appdata-Failover/Gmer4Lfe"
|
|
)
|
|
|
|
# ==============================================================================================
|
|
# ── DOCKER ESSENTIALS ─────────────────────────────────────────────────────────────────────────
|
|
# ==============================================================================================
|
|
|
|
# ━━━ Docker Daily Restart ━━━
|
|
# Containers restarted every day by Docker_Essentials/docker_daily_restart.sh.
|
|
# Keeps services fresh and clears memory leaks that accumulate over time.
|
|
# Case-sensitive — must match exact Docker container names shown in the unRAID Docker tab.
|
|
DAILY_RESTART_CONTAINERS=(
|
|
"NginxProxyManager"
|
|
"Authelia"
|
|
"Dispatcharr-Iptv-Users"
|
|
"Dispatcharr"
|
|
"Dispatcharr-Basic"
|
|
"ErsatzTV-Emby"
|
|
)
|
|
|
|
# ━━━ Docker Weekly Restart ━━━
|
|
# Containers restarted once per week by Docker_Essentials/docker_weekly_restart.sh.
|
|
# For less critical services that benefit from periodic restart but don't need daily cycling.
|
|
WEEKLY_RESTART_CONTAINERS=(
|
|
"NextCloud"
|
|
"Organizrv2-Gmer4Lfe"
|
|
"AdGuard-Home"
|
|
"Immich-Gmer4Lfe"
|
|
)
|
|
|
|
# ━━━ Docker Watchdog ━━━
|
|
# First line of defense for container health — runs every 15 minutes via cron.
|
|
# Monitors memory usage, CPU usage and HTTP responsiveness per container.
|
|
# Uses a strike system to avoid restarting on brief spikes — sustained issues trigger restart.
|
|
# Works alongside system_watchdog.sh — containers first, system reboot is the last resort.
|
|
|
|
# Containers to monitor with their memory hard limits in MB.
|
|
# Memory hard limit exceeded → immediate restart (no strike system for memory).
|
|
# CPU and HTTP use strike system — see CPU_FAIL_LIMIT and RESP_FAIL_LIMIT below.
|
|
# 20GB=20480 16GB=16384 14GB=14336 12GB=12288 10GB=10240
|
|
# 8GB=8192 6GB=6144 4GB=4096 2GB=2048 1GB=1024
|
|
declare -A WATCHDOG_CONTAINERS=(
|
|
["Emby"]=16384 # 16GB — media server, transcoding can spike high
|
|
["LidaTube"]=6144 # 6GB — YouTube downloader
|
|
["Tdarr"]=6144 # 6GB — transcoding node
|
|
["Code-Server"]=1024 # 1GB — VS Code server
|
|
)
|
|
|
|
# Containers to check HTTP responsiveness via curl — omit a container to skip its HTTP check.
|
|
# curl checks the URL and considers the container unresponsive if it times out or errors.
|
|
declare -A WATCHDOG_CONTAINER_URLS=(
|
|
["Emby"]="http://localhost:8096"
|
|
)
|
|
|
|
# Containers that should always be running — monitored for unexpected stops.
|
|
# Strike system used — tries restart on each strike up to SYS_WATCHDOG_STRIKE_LIMIT.
|
|
# If restart fails after strike limit → added to persistent skip list on /boot/
|
|
# Skip list auto-clears when container recovers after reboot or manual fix.
|
|
WATCHDOG_REQUIRED_CONTAINERS=(
|
|
"NginxProxyManager"
|
|
"Lldap-Gmer4Lfe"
|
|
"Authelia"
|
|
"Mariadb-Authelia"
|
|
"Redis-Authelia"
|
|
"Authelia-Secondary"
|
|
"Redis-Authelia-Secondary"
|
|
)
|
|
|
|
# Strike state file — /tmp resets on reboot which is correct behaviour for strike tracking
|
|
WATCHDOG_STATE_FILE="/tmp/container_watchdog_state.db"
|
|
|
|
# CPU thresholds — normalised against total core count automatically at runtime.
|
|
# A container using 85% of one core on a 16-core system = ~5.3% normalised.
|
|
SOFT_CPU_THRESHOLD=80 # warn at this % of total system CPU
|
|
HARD_CPU_THRESHOLD=85 # strike at this % of total system CPU
|
|
CPU_FAIL_LIMIT=2 # consecutive strikes before container restart
|
|
|
|
# Memory soft threshold — warn when container reaches this % of its hard limit.
|
|
# Hard limit exceeded triggers immediate restart regardless of strikes.
|
|
SOFT_MEM_THRESHOLD=80
|
|
|
|
# HTTP responsiveness check settings
|
|
RESP_FAIL_LIMIT=2 # consecutive failed curl checks before restart
|
|
CURL_TIMEOUT=5 # seconds before curl gives up per check
|
|
|
|
# ━━━ Docker Network Connect ━━━
|
|
# Connects containers to extra Docker networks on array start.
|
|
# Useful when containers need to communicate across networks they were not originally
|
|
# configured with — e.g. memcached needing access to the nextcloud-aio network.
|
|
# Every container in the list connects to every network in the list (many-to-many).
|
|
# Comment out entries to disable without removing them.
|
|
NETWORK_CONNECT_CONTAINERS=(
|
|
"memcached"
|
|
"Npm-CrowdSec"
|
|
)
|
|
|
|
NETWORK_CONNECT_NETWORKS=(
|
|
"nextcloud-aio" # Docker network name — must exist before array start
|
|
)
|
|
|
|
# ==============================================================================================
|
|
# ── UNRAID ESSENTIALS ─────────────────────────────────────────────────────────────────────────
|
|
# ==============================================================================================
|
|
|
|
# ━━━ Reboot ━━━
|
|
# Seconds of warning broadcast to all logged-in users before server_reboot.sh reboots.
|
|
# Gives users time to save work or finish what they are doing before the system goes down.
|
|
REBOOT_SLEEP=300
|
|
|
|
# ━━━ Mover ━━━
|
|
# Seconds to wait after warning users before mover_stop.sh sends SIGTERM to the mover.
|
|
# Gives the mover time to finish its current file operation cleanly before being killed.
|
|
MOVER_STOP_TIMEOUT=300
|
|
|
|
# ━━━ Syslog Filter ━━━
|
|
# Path for the rsyslog filter file created by docker_syslog_filter.sh.
|
|
# The filter suppresses noisy Docker veth and docker0 messages from syslog on boot.
|
|
# Without this filter, every Docker network interface change floods the syslog.
|
|
FILTER_FILE="/etc/rsyslog.d/ignore-docker-veth.conf"
|
|
|
|
# ━━━ PHP-FPM ━━━
|
|
# Config file path and max children value for php_fpm_max_children.sh.
|
|
# Higher max_children allows more concurrent PHP requests to the unRAID WebGUI.
|
|
# Set based on available RAM — too high can cause memory pressure on low-RAM systems.
|
|
PHP_CONF="/etc/php-fpm.d/www.conf"
|
|
PHP_MAX_CHILDREN=250
|
|
|
|
# ━━━ Clear Logs ━━━
|
|
# System log files cleared by clear_logs.sh — Docker container logs are cleared too.
|
|
# Run weekly to prevent logs from filling the rootfs over time.
|
|
LOG_FILES=(/var/log/syslog /var/log/messages /var/log/dmesg)
|
|
|
|
# ━━━ WebGUI Watchdog ━━━
|
|
# Monitors the unRAID WebGUI and restarts services if it becomes unresponsive.
|
|
# Escalation path: nginx restart → recheck → emhttp restart → recheck → notify warning.
|
|
# emhttp is the core unRAID management daemon — restarting it is more disruptive than nginx
|
|
# but both recover cleanly. Notification sent on any restart so you know what happened.
|
|
WEBGUI_URL="http://localhost" # adjust port if non-standard e.g. http://localhost:8080
|
|
WEBGUI_TIMEOUT=5 # seconds before curl gives up on the WebGUI check
|
|
WEBGUI_NGINX_WAIT=15 # seconds to wait after nginx restart before rechecking
|
|
WEBGUI_EMHTTP_WAIT=30 # seconds to wait after emhttp restart — emhttp takes longer
|
|
|
|
# ==============================================================================================
|
|
# ── MEDIA ─────────────────────────────────────────────────────────────────────────────────────
|
|
# ==============================================================================================
|
|
|
|
# ━━━ Media Permissions ━━━
|
|
# Mode and owner applied recursively to all shares in MEDIA_PERMISSION_SHARES.
|
|
# Run by Media/media_shares_permissions.sh via the media_management.sh orchestrator.
|
|
# 777 and nobody:users is standard for unRAID media shares accessible by Docker containers.
|
|
PERMISSIONS_MODE="777"
|
|
PERMISSIONS_OWNER="nobody:users"
|
|
|
|
# Shares to apply permissions to — add or remove paths as your library grows.
|
|
# These are applied recursively so large shares take time — run overnight via orchestrator.
|
|
MEDIA_PERMISSION_SHARES=(
|
|
/mnt/user/Anime_Movies
|
|
/mnt/user/Anime_Movies-Old
|
|
/mnt/user/Anime_Shows
|
|
/mnt/user/Anime_Shows-Old
|
|
/mnt/user/appcache
|
|
/mnt/user/Books
|
|
/mnt/user/Downloads
|
|
/mnt/user/Games
|
|
/mnt/user/Intros
|
|
/mnt/user/Kids_Movies
|
|
/mnt/user/Kids_Tv_Shows
|
|
/mnt/user/Movie_Recordings
|
|
/mnt/user/Movies
|
|
/mnt/user/Music
|
|
/mnt/user/Music_Videos
|
|
/mnt/user/Photo
|
|
/mnt/user/Sports
|
|
/mnt/user/stand-up_comedy
|
|
/mnt/user/Temp_Storage
|
|
/mnt/user/Tv_Recordings
|
|
/mnt/user/Tv_Shows
|
|
/mnt/user/YouTube
|
|
)
|
|
|
|
# ━━━ Media Cleaner ━━━
|
|
# Removes junk files from media shares using configurable file pattern lists.
|
|
# Two profiles: anime and media — each with their own folder list and patterns.
|
|
# Run via Media/media_cleaner.sh anime or Media/media_cleaner.sh media
|
|
# Called automatically by media_management.sh via MEDIA_MAINTENANCE_JOBS below.
|
|
|
|
# Folders scanned by the anime profile — anime downloads commonly include these junk files
|
|
ANIME_CLEAN_FOLDERS=(
|
|
/mnt/user/Anime_Movies
|
|
/mnt/user/Anime_Movies-Old
|
|
/mnt/user/Anime_Shows
|
|
/mnt/user/Anime_Shows-Old
|
|
)
|
|
|
|
# Folders scanned by the media profile
|
|
MEDIA_CLEAN_FOLDERS=(
|
|
/mnt/user/Kids_Movies
|
|
/mnt/user/Kids_Tv_Shows
|
|
/mnt/user/Movies
|
|
/mnt/user/Music
|
|
/mnt/user/Sports
|
|
/mnt/user/stand-up_comedy
|
|
/mnt/user/Tv_Shows
|
|
)
|
|
|
|
# File patterns deleted by the anime profile — common junk from anime download groups
|
|
ANIME_FILE_PATTERNS=(
|
|
'*.sfv' '*.md5' '*.sha1' '*.txt' '*.url' '*.lnk'
|
|
'*.rar' '*.zip' '*.info' '*.torrent' '*.sample*' '*.proof*'
|
|
'*sync-conflict*' '*.scr' '*.srr' '*.exe' '*.webp'
|
|
'*.log' '*.json'
|
|
)
|
|
|
|
# File patterns deleted by the media profile — includes *.iso and *.lrc not needed in anime
|
|
MEDIA_FILE_PATTERNS=(
|
|
'*.sfv' '*.md5' '*.sha1' '*.txt' '*.url' '*.lnk'
|
|
'*.rar' '*.zip' '*.info' '*.torrent' '*.sample*' '*.proof*'
|
|
'*sync-conflict*' '*.scr' '*.srr' '*.exe' '*.webp'
|
|
'*.log' '*.json' '*.iso' '*.lrc'
|
|
)
|
|
|
|
# ━━━ Media Management Orchestrator ━━━
|
|
# Job list for Orchestrators/media_management.sh — runs scripts sequentially in order.
|
|
# Format: "folder/script.sh optional_argument"
|
|
# Order matters — permissions runs first so cleaners and arr scripts see correct ownership.
|
|
# Arr cleanup scripts run last — they depend on clean folders from the cleaner steps.
|
|
# Comment out any job to disable without removing it — easy to re-enable later.
|
|
MEDIA_MAINTENANCE_JOBS=(
|
|
"Media/media_shares_permissions.sh" # apply permissions first
|
|
"Media/media_cleaner.sh anime" # remove junk from anime shares
|
|
"Media/media_cleaner.sh media" # remove junk from media shares
|
|
"Media/lidarr_cleanup.sh" # remove orphaned music files
|
|
"Media/sonarr_cleanup.sh" # remove orphaned TV files
|
|
"Media/radarr_cleanup.sh" # remove orphaned movie files
|
|
)
|
|
|
|
# ━━━ Arr Cleanup ━━━
|
|
# Lidarr, Sonarr and Radarr orphan file cleanup via their respective APIs.
|
|
# Each arr script queries its API to get all tracked file paths, then compares against
|
|
# what exists on disk. Files not tracked by the arr and older than ORPHAN_AGE days are deleted.
|
|
#
|
|
# Why the age threshold matters:
|
|
# The arr downloads a file then processes it — there is a window where the file exists
|
|
# on disk but the arr hasn't imported it yet. ORPHAN_AGE prevents deleting files that
|
|
# are mid-import. 7 days is conservative and safe for any normal workflow.
|
|
#
|
|
# Protected patterns are NEVER deleted regardless of tracking status or age.
|
|
# These protect arr-generated metadata (cover art, .nfo files, subtitles) that the arr
|
|
# depends on but does not include in its tracked file API response.
|
|
# Add new patterns here if arr metadata formats change in future versions.
|
|
|
|
# Lidarr — music library
|
|
LIDARR_URL="http://192.168.50.2:8686"
|
|
LIDARR_API_KEY="b2977e71ef074bc0a0529d9fcce3b2dc"
|
|
LIDARR_MUSIC_ROOT="/mnt/user/Music-New" # must match the root path set in Lidarr
|
|
LIDARR_ORPHAN_AGE=7 # days before untracked file is eligible for deletion
|
|
LIDARR_EXTENSIONS=("flac" "mp3" "m4a" "wav" "aac" "ogg" "opus" "wma")
|
|
# file extensions considered valid music files
|
|
LIDARR_PROTECTED_PATTERNS=("*.jpg" "*.jpeg" "*.png" "*.nfo" "*.lrc")
|
|
# never deleted — cover art, metadata, lyrics
|
|
|
|
# Sonarr — TV library
|
|
SONARR_URL="http://192.168.50.2:8989"
|
|
SONARR_API_KEY="130decd3db5b4c25afad64864cd03f9f"
|
|
SONARR_TV_ROOT="/mnt/user/Tv_Shows" # must match the root path set in Sonarr
|
|
SONARR_ORPHAN_AGE=7
|
|
SONARR_EXTENSIONS=("mkv" "mp4" "avi" "m4v" "ts" "wmv" "mov")
|
|
SONARR_PROTECTED_PATTERNS=("*.jpg" "*.jpeg" "*.png" "*.nfo" "*.srt" "*.sub" "*.ass" "*.ssa")
|
|
# never deleted — artwork, metadata, subtitles
|
|
|
|
# Radarr — movie library
|
|
RADARR_URL="http://192.168.50.2:7878"
|
|
RADARR_API_KEY="d43a3ec6cf1549edb4af0cc63f98b2a9"
|
|
RADARR_MOVIES_ROOT="/mnt/user/Movies" # must match the root path set in Radarr
|
|
RADARR_ORPHAN_AGE=7
|
|
RADARR_EXTENSIONS=("mkv" "mp4" "avi" "m4v" "wmv" "mov")
|
|
RADARR_PROTECTED_PATTERNS=("*.jpg" "*.jpeg" "*.png" "*.nfo" "*.srt" "*.sub" "*.ass" "*.ssa")
|
|
|
|
# ==============================================================================================
|
|
# ── TRANSCODES ────────────────────────────────────────────────────────────────────────────────
|
|
# ==============================================================================================
|
|
# Session-based storage allocator using filesystem symlink indirection.
|
|
# ffmpeg resolves the symlink ONCE at session start — existing sessions are never affected.
|
|
# Only new sessions care about where the symlink currently points.
|
|
#
|
|
# How it works:
|
|
# ramdisk_setup.sh — run once at array start, creates tmpfs and sets symlink
|
|
# transcode_manager.sh — every 3 min, monitors usage and manages symlink direction
|
|
# transcode_cleanup.sh — every 5 min, removes old inactive files from both locations
|
|
#
|
|
# ⚠️ Docker mount warning:
|
|
# Do NOT add a static SSD transcode path as a second volume mount in your Emby container.
|
|
# If the SSD path is mounted inside the container Emby can see it and will use it
|
|
# independently of the symlink — breaking symlink-based routing entirely.
|
|
# The symlink IS your emergency lever — one mount only:
|
|
# /mnt/ram-transcode → /ext-ram-transcode
|
|
|
|
RAMDISK_PATH="/mnt/ramdisk_transcodes" # tmpfs mount point created at array start
|
|
RAMDISK_SIZE="8G" # ceiling — tmpfs only uses RAM actually needed
|
|
TRANSCODE_LINK="/mnt/ram-transcode" # symlink Emby points at — location never changes
|
|
TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/" # SSD fallback location
|
|
|
|
# Usage thresholds in GB — hysteresis gap prevents flip-flop near threshold
|
|
RAMDISK_WARN_GB=6.8 # flip symlink to SSD at or above this usage
|
|
RAMDISK_LOW_GB=5.5 # flip symlink back to ramdisk when usage drops here
|
|
RAMDISK_SSD_MIN_GB=20 # minimum free GB on SSD required before allowing flip to SSD
|
|
|
|
# Cleanup age thresholds — files must be older than these AND not open by any process
|
|
TRANSCODE_MAX_AGE=20 # minutes before a transcode file is eligible for cleanup
|
|
TRANSCODE_ORPHAN_AGE=30 # minutes before an orphaned file is eligible
|
|
|
|
# Flip frequency alert — too many flips per hour may indicate ramdisk needs to be larger
|
|
TRANSCODE_FLIP_WARN=3 # notify if symlink flips this many times in one hour
|
|
|
|
# Permissions — must match your Emby container user
|
|
TRANSCODE_OWNER="nobody:users"
|
|
TRANSCODE_CHMOD="755" # renamed from TRANSCODE_MODE to avoid ambiguity with manager mode
|
|
|
|
# Operating mode — controls symlink routing behavior
|
|
# smart — auto-flips between ramdisk and SSD based on usage thresholds (default)
|
|
# ramdisk — always uses ramdisk, never flips to SSD regardless of usage
|
|
# useful when load is light and you want guaranteed ramdisk performance
|
|
# warns if usage exceeds threshold but does not flip
|
|
# ssd — always uses SSD, never uses ramdisk
|
|
# useful during ramdisk maintenance, testing, or after a flip issue
|
|
# switch to this mode to drain ramdisk sessions gracefully
|
|
TRANSCODE_MANAGER_MODE="smart"
|
|
|
|
# Emby container check — skips threshold checks when Emby is not running
|
|
# Prevents unnecessary symlink flips when no transcoding is happening
|
|
TRANSCODE_CHECK_EMBY=true
|
|
TRANSCODE_EMBY_CONTAINER="Emby" # exact Docker container name — case sensitive
|
|
|
|
# ==============================================================================================
|
|
# ── MONITORS ───────────────────────────────────────────────────────────────────────────────────
|
|
# ==============================================================================================
|
|
# Monitoring scripts — watch and report only, never take action.
|
|
# Lives in Monitor/ folder — distinct from unRAID_Essentials (which acts) and
|
|
# Docker_Essentials (which manages containers).
|
|
# These scripts are the data sources for the future plugin dashboard.
|
|
|
|
# ━━━ Certificate Monitor ━━━
|
|
# Checks SSL cert expiry via direct openssl connection — no NPM dependency.
|
|
# Reads the actual cert the server is presenting — catches real-world issues API checks miss.
|
|
# Each domain and subdomain is a separate entry — they have independent certs.
|
|
# Add your public-facing domains — uncomment and replace with your actual domains.
|
|
CERT_MONITOR_DOMAINS=(
|
|
"Gmer4Lfe.com"
|
|
"Gmer4Lfe.us"
|
|
)
|
|
CERT_WARN_DAYS=30 # notify warning when cert expires within this many days
|
|
CERT_CRIT_DAYS=7 # notify critical when cert expires within this many days
|
|
CERT_TIMEOUT=10 # seconds before openssl connection attempt gives up per domain
|
|
|
|
# ━━━ Backup Verify ━━━
|
|
# Verifies the rsync mirror is healthy by comparing random file checksums between servers.
|
|
# Uses existing SSH keys — no additional configuration needed beyond the share list.
|
|
# Leave BACKUP_VERIFY_SHARES empty to automatically use DAILY_SYNC_SHARES as the target list.
|
|
BACKUP_VERIFY_SHARES=(
|
|
# leave empty to use DAILY_SYNC_SHARES automatically
|
|
# or specify individual shares to verify:
|
|
# /mnt/user/Movies
|
|
# /mnt/user/Tv_Shows
|
|
)
|
|
BACKUP_VERIFY_SAMPLE=10 # number of files to randomly sample per share per run
|
|
BACKUP_VERIFY_MIN_SIZE=1M # skip files smaller than this — avoids tiny junk files
|
|
|
|
# ━━━ SMART Health ━━━
|
|
# Monitors drive SMART attributes — reads live from each drive, no persistent writes.
|
|
# Discovers all drives automatically via /dev/sd* and /dev/nvme* — no drive list needed.
|
|
# Add drives to SMART_IGNORE_DRIVES to skip specific drives (e.g. your unRAID boot USB).
|
|
SMART_TEMP_WARN=45 # degrees C — warn if drive temperature exceeds this
|
|
SMART_TEMP_CRIT=55 # degrees C — critical if drive temperature exceeds this
|
|
SMART_IGNORE_DRIVES=(
|
|
"sda" # uncomment to ignore sda — common choice if sda is your unRAID boot USB
|
|
)
|
|
|
|
# ━━━ Bandwidth Monitor ━━━
|
|
# Logs daily rsync transfer totals to a bounded file on /boot/ — minimal flash wear.
|
|
# bandwidth_monitor.sh --log-transfer is called by rsync.sh after each successful sync.
|
|
# bandwidth_monitor.sh --report generates the weekly summary standalone.
|
|
# File stays bounded to BANDWIDTH_LOG_RETENTION lines — old entries auto-purged on each write.
|
|
BANDWIDTH_LOG="/boot/config/bandwidth_history.db"
|
|
BANDWIDTH_LOG_RETENTION=90 # days to keep — file never grows beyond ~90 lines
|
|
BANDWIDTH_WARN_GB=50 # flag in reports if a single sync transfer exceeds this GB
|
|
|
|
# ━━━ ZFS Memory Snapshot ━━━
|
|
ZFS_REPORT_LOG="/var/log/zfs-weekly-health.log"
|
|
ZFS_REPORT_ARC_WARN_PCT=90
|
|
ZFS_REPORT_FREE_WARN_GB=10
|
|
ZFS_REPORT_AVAIL_WARN_GB=20
|
|
ZFS_REPORT_DOCKER_TOP=10
|
|
|
|
# Pools to exclude from health reporting — still monitored by unRAID but skipped in report
|
|
# Useful for pools that are expected to be heavily used or are managed separately
|
|
ZFS_REPORT_IGNORE_POOLS=(
|
|
"disk10" # Docker overlay storage — high usage is normal
|
|
"disk9" # Cache pool — usage varies widely, not meaningful to report
|
|
"disk8"
|
|
"disk6"
|
|
"disk5"
|
|
)
|
|
|
|
# ━━━ Health Digest ━━━
|
|
# Aggregated system health summary from across the ecosystem.
|
|
# Reads existing state files — no new writes to flash drive.
|
|
#
|
|
# Three profiles — switch by changing DIGEST_PROFILE, no cron changes needed:
|
|
# always — sends every run (schedule daily = daily digest, weekly = weekly digest)
|
|
# smart — sends only if findings worth reporting (intelligent quiet operation)
|
|
# weekly — sends once per week on DIGEST_DAY only, silent all other days
|
|
#
|
|
# Data sources (reads only — no writes):
|
|
# Transcode ramdisk state, container watchdog strikes, system watchdog strikes,
|
|
# failover state, container skip list, bandwidth history, SSL cert days remaining
|
|
DIGEST_PROFILE="weekly" # always | smart | weekly
|
|
DIGEST_DAY="Sunday" # day name for weekly profile — must match date +%A output
|
|
|
|
# Smart profile triggers — set true to send digest when this condition is found
|
|
DIGEST_SMART_ON_WATCHDOG=true # send if any watchdog strikes are active
|
|
DIGEST_SMART_ON_FAILOVER=true # send if failover state is not NORMAL
|
|
DIGEST_SMART_ON_CERT_WARN=true # send if any cert is under CERT_WARN_DAYS
|
|
DIGEST_SMART_ON_BANDWIDTH=true # send if any transfer exceeded BANDWIDTH_WARN_GB
|
|
|
|
# ━━━ Emby Session Report ━━━
|
|
# Weekly Emby usage report via API — no persistent writes, queries fresh each run.
|
|
# Shows active streams, library counts, transcode vs direct play ratio.
|
|
# Requires an API key from Emby Settings → API Keys in the Emby WebUI.
|
|
EMBY_URL="http://localhost:8096"
|
|
EMBY_API_KEY="0c27448d93a7431f9ac63569f7655829" # paste your Emby API key here
|
|
EMBY_REPORT_DAYS=7 # number of days to include in the report period
|
|
EMBY_REPORT_TOP_N=10 # number of top content items to show in report
|
|
|
|
# ==============================================================================================
|
|
# ── SYSTEM WATCHDOG ───────────────────────────────────────────────────────────────────────────
|
|
# ==============================================================================================
|
|
# Last line of defense — reboots the system cleanly when it is about to become unstable.
|
|
# Runs every 15 minutes via cron. Works alongside docker_watchdog.sh:
|
|
# docker_watchdog.sh — container level, minimal disruption, tries to self-heal first
|
|
# system_watchdog.sh — system level, last resort, reboots when healing has failed
|
|
#
|
|
# Strike system: sustained threshold hits trigger reboot — single spikes are ignored.
|
|
# Each check that exceeds its threshold adds a strike. Strikes reset when recovered.
|
|
# When strike limit is hit the reboot sequence begins.
|
|
#
|
|
# Reboot loop protection: tracks reboot timestamps on /boot/ (survives reboots).
|
|
# If the server reboots too many times in the window it shuts down instead — a reboot
|
|
# loop means something is fundamentally wrong that a reboot is not fixing.
|
|
|
|
# ━━━ State Files ━━━
|
|
# Strike counts reset on reboot — /tmp is correct (fresh start after each reboot)
|
|
SYS_WATCHDOG_STATE_FILE="/tmp/system_watchdog_state.db"
|
|
# Persistent container skip list — on /boot/ so it survives reboots
|
|
# Containers added here when docker_watchdog.sh exhausts all restart attempts
|
|
# Auto-clears when container is found running again after reboot or manual fix
|
|
SYS_WATCHDOG_FAILED_FILE="/boot/config/system_watchdog_failed.db"
|
|
# Reboot timestamp log — on /boot/ for reboot loop detection across reboots
|
|
SYS_WATCHDOG_REBOOT_LOG="/boot/config/system_watchdog_reboots.db"
|
|
|
|
# ━━━ Strike and Reboot Loop Settings ━━━
|
|
# Consecutive threshold hits required before triggering reboot
|
|
SYS_WATCHDOG_STRIKE_LIMIT=2
|
|
# Maximum reboots allowed within the window before shutting down instead
|
|
# A reboot loop means something fundamental is broken that rebooting is not fixing
|
|
SYS_WATCHDOG_REBOOT_LIMIT=3
|
|
# Window in hours — controls BOTH the reboot count window AND the rolling log purge
|
|
# Entries older than this many hours are automatically removed from the reboot log
|
|
SYS_WATCHDOG_REBOOT_WINDOW_HRS=12
|
|
|
|
# ━━━ Thresholds ━━━
|
|
# Set these at "I am about to become unstable" levels — not just "things are a bit high"
|
|
SYS_WATCHDOG_ROOTFS_PCT=95 # rootfs % — at 95% something is seriously wrong
|
|
SYS_WATCHDOG_LOG_PCT=95 # /var/log % — log spam filling the filesystem
|
|
SYS_WATCHDOG_MEM_GB=4 # free RAM GB — 4GB free on 128GB system is critical
|
|
SYS_WATCHDOG_ARC_PINNED_PCT=98 # ZFS ARC % of max before attempting reclaim
|
|
SYS_WATCHDOG_ARC_RELEASE_PCT=95 # ZFS ARC % after reclaim that still triggers reboot
|
|
SYS_WATCHDOG_LOAD_MULTIPLIER=3 # strike if load avg > cores x this multiplier
|
|
SYS_WATCHDOG_ZOMBIE_LIMIT=50 # zombie process count before strike
|
|
SYS_WATCHDOG_CPU_TEMP_MAX=95 # degrees C — adjust for your specific CPU tjmax
|
|
|
|
# ━━━ Check Toggles ━━━
|
|
# true = run this check on every watchdog cycle / false = skip entirely
|
|
# Disable checks not relevant to your hardware or that cause false positives
|
|
SYS_WATCHDOG_CHECK_ROOTFS=true
|
|
SYS_WATCHDOG_CHECK_LOG=true
|
|
SYS_WATCHDOG_CHECK_RAM=true
|
|
SYS_WATCHDOG_CHECK_ARC=true
|
|
SYS_WATCHDOG_CHECK_CPU_TEMP=true
|
|
SYS_WATCHDOG_CHECK_LOAD=false # disabled — load spikes during transcoding are normal
|
|
SYS_WATCHDOG_CHECK_ZOMBIES=true
|
|
SYS_WATCHDOG_CHECK_CONTAINERS=true # checks persistent skip list from docker_watchdog.sh
|
|
SYS_WATCHDOG_CHECK_DOCKER_DAEMON=true
|
|
|
|
# ━━━ Abort Toggles ━━━
|
|
# Controls whether certain conditions prevent a reboot from happening.
|
|
# true = abort reboot if this condition is active (conservative — default)
|
|
# false = reboot anyway regardless of this condition (aggressive)
|
|
# Philosophy: a graceful reboot before crash is always better than a hard crash mid-operation
|
|
SYS_WATCHDOG_ABORT_ON_ZFS_UNHEALTHY=true # unhealthy pool + reboot risks data loss
|
|
SYS_WATCHDOG_ABORT_ON_PARITY=false # aborting parity check beats crashing mid-check
|
|
SYS_WATCHDOG_ABORT_ON_MOVER=false # aborting mover beats crashing mid-move
|
|
|
|
# ==============================================================================================
|
|
# ──────────────────────── End Of User Variables ───────────────────────────────────────────────
|
|
# ============================================================================================== |