#!/bin/bash # ============================================================================================== # ============================= Health Digest ================================================== # ============================================================================================== # # PURPOSE # ───────────────────────────────────────────────────────────────────────────── # Full ecosystem health aggregation from existing state files. Scheduled daily # (8am). DIGEST_PROFILE controls when notifications actually send — the cron # schedule never changes, only the profile in master.conf. # # Reads state files from across the system (watchdog strikes, fallback state, # skip list, bandwidth history, transcode stats, cert status) and compiles them # into a single digest. Reads only — writes nothing, changes nothing. # # ============================================================================================== # OPERATIONAL MODEL # ============================================================================================== # # Three profiles — switch by changing DIGEST_PROFILE in master.conf: # # always — sends every run regardless of findings # Use: daily digest of everything, even when healthy # # smart — sends only when something worth reporting was found # Stays silent on clean days. DIGEST_SMART_ON_* toggles control # what triggers a send — all independently configurable. # # weekly — sends once per week on DIGEST_DAY (default Sunday), silent all other days # Use: one weekly summary without daily noise # # Data sources (reads only): # FALLBACK_STATE_FILE — current fallback state # DOCKER_WATCHDOG_FAILED_FILE — container skip list (manual intervention needed) # WATCHDOG_STATE_FILE — active container watchdog strikes # SYS_WATCHDOG_STATE_FILE — active system watchdog strikes # BANDWIDTH_LOG — yesterday's transfer totals # TRANSCODE_DAILY_LOG — weekly transcode statistics # CERT_MONITOR_DOMAINS — live SSL cert check via openssl # RAMDISK_PATH / TRANSCODE_LINK — current transcode location and usage # # ============================================================================================== # DESIGN PRINCIPLES # ============================================================================================== # # Aggregator, Not Generator # This script reads state files that other scripts maintain. It never produces # health data itself — it only presents what is already there. Each source # script remains responsible for its own state; this script is the envelope. # # Profile-Driven Notification # The cron schedule never changes. The DIGEST_PROFILE in master.conf controls # when notifications actually send — switching from daily noise to weekly # summaries is a one-line conf change, not a cron edit. # # Read-Only, No Side Effects # Writes nothing, changes nothing, triggers nothing. Safe to run at any time # for a health snapshot without affecting any running service or state file. # # ============================================================================================== # OPERATIONAL SAFEGUARDS # ============================================================================================== # # Single Instance Lock # acquire_lock prevents duplicate reports — report generation takes time. # # Per-Host Variables # detect_hosts() aliases CERT_MONITOR_DOMAINS, RAMDISK_WARN_GB, RAMDISK_SIZE, # RAMDISK_LOW_GB, and all other host-specific vars used in the report. # # Per-Section Guards # Each data source section checks whether its state file exists before reading. # A missing state file is skipped cleanly — it does not abort the report. # # Silent Smart Profile # smart profile produces no output and no notification when nothing worth # reporting is found. # # openssl Validated — Non-Fatal # platform_require_cmd checks openssl and, unlike the other monitors, only warns if it # is missing: the SSL section is skipped and the rest of the digest still runs. The # notify script is validated separately by the platform adapter. # # ============================================================================================== # CONFIGURATION # ============================================================================================== # # master.conf # # DIGEST_PROFILE # Notification frequency: always | smart | weekly. (default: weekly) # # DIGEST_DAY # Day name for weekly profile — must match `date +%A` output. (default: Sunday) # # DIGEST_SMART_ON_WATCHDOG # Send smart profile notification if any active watchdog strikes. (default: true) # # DIGEST_SMART_ON_FALLBACK # Send smart profile notification if fallback state is not NORMAL. (default: true) # # DIGEST_SMART_ON_CERT_WARN # Send smart profile notification if any cert is within CERT_WARN_DAYS. (default: true) # # DIGEST_SMART_ON_BANDWIDTH # Send smart profile notification if any transfer exceeded BANDWIDTH_WARN_GB. (default: true) # # CERT_WARN_DAYS / CERT_CRIT_DAYS / CERT_TIMEOUT # Cert check thresholds — shared with cert_monitor.sh. # # BANDWIDTH_WARN_GB # High-transfer threshold — shared with bandwidth_monitor.sh. # # TRANSCODE_DAILY_LOG # Path to the transcode statistics log. # # ============================================================================================== # RUNTIME MODES # ============================================================================================== # # weekly_health_digest.sh # Run digest. DIGEST_PROFILE determines whether a notification is sent. # # weekly_health_digest.sh --dry-run # Generate and display digest output. No notification sent regardless of profile. # # weekly_health_digest.sh --status # Show profile, day, and smart trigger configuration. Then exit. # # weekly_health_digest.sh --log # Verbose per-section output during digest generation. # # ============================================================================================== SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" source "$SCRIPT_DIR/../load_config.sh" parse_args "$@" # ============================================================================================== # ━━━ Setup ━━━ # ============================================================================================== if [[ "$EUID" -ne 0 ]]; then error "Must be run as root" exit 1 fi platform_require_cmd \ "$(command -v openssl 2>/dev/null || echo /usr/bin/openssl)" \ "version" "OpenSSL" \ "openssl" || warn "openssl not found — SSL cert checks will be skipped" acquire_lock # detect_hosts() sets MY_ID and aliases all host-specific vars used in this report detect_hosts log "$ICON_GEAR Config: profile=${DIGEST_PROFILE} day=${DIGEST_DAY}" log "$ICON_GEAR Smart triggers: watchdog=${DIGEST_SMART_ON_WATCHDOG} fallback=${DIGEST_SMART_ON_FALLBACK} cert=${DIGEST_SMART_ON_CERT_WARN} bandwidth=${DIGEST_SMART_ON_BANDWIDTH}" [[ "$DRY_RUN" == true ]] && warn "DRY RUN — report generated but no notification sent" # ============================================================================================== # ━━━ Status ━━━ # ============================================================================================== if [[ "$SHOW_STATUS" == true ]]; then echo "" echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━" echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)" echo "$ICON_DIGEST Profile: $DIGEST_PROFILE" echo "$ICON_DIGEST Digest day: $DIGEST_DAY" echo "$ICON_DIGEST Smart triggers: watchdog=$DIGEST_SMART_ON_WATCHDOG fallback=$DIGEST_SMART_ON_FALLBACK cert=$DIGEST_SMART_ON_CERT_WARN bandwidth=$DIGEST_SMART_ON_BANDWIDTH" echo "$ICON_CERT Cert domains: ${CERT_MONITOR_DOMAINS[*]:-none}" echo "$ICON_BANDWIDTH Bandwidth warn: ${BANDWIDTH_WARN_GB}GB" echo "$ICON_GEAR Dry Run: $DRY_RUN" echo "━━━━━━━━━━━━━━━━━━━━━━━" exit 0 fi # ============================================================================================== # ── Profile gate — should we send today? ────────────────────────────────────────────────────── # ============================================================================================== SHOULD_SEND=false case "$DIGEST_PROFILE" in always) SHOULD_SEND=true log "Profile: always — will send" ;; weekly) TODAY_NAME=$(date '+%A') if [[ "$TODAY_NAME" == "$DIGEST_DAY" ]]; then SHOULD_SEND=true log "Profile: weekly — today is $DIGEST_DAY — will send" else echo "Profile: weekly — today is $TODAY_NAME, digest day is $DIGEST_DAY — no-op" exit 0 fi ;; smart) log "Profile: smart — evaluating findings before deciding" SHOULD_SEND=false ;; *) warn "Unknown DIGEST_PROFILE: $DIGEST_PROFILE — defaulting to weekly" TODAY_NAME=$(date '+%A') [[ "$TODAY_NAME" == "$DIGEST_DAY" ]] && SHOULD_SEND=true ;; esac # ============================================================================================== # ── Data Gathering ──────────────────────────────────────────────────────────────────────────── # ============================================================================================== FINDINGS=() # notable but not critical ISSUES=() # need attention DIGEST_LINES=() # full report lines # ── fallback State ──────────────────────────────────────────────────────────────────────────── log "Reading: fallback=$FALLBACK_STATE_FILE skip=$DOCKER_WATCHDOG_FAILED_FILE watchdog=$WATCHDOG_STATE_FILE sys=$SYS_WATCHDOG_STATE_FILE" if [[ -f "$FALLBACK_STATE_FILE" ]]; then FALLBACK_STATE=$(grep "^state=" "$FALLBACK_STATE_FILE" 2>/dev/null | cut -d= -f2) if [[ -n "$FALLBACK_STATE" ]]; then DIGEST_LINES+=("$ICON_FALLBACK Fallback: $FALLBACK_STATE") if [[ "$FALLBACK_STATE" != "NORMAL" ]]; then ISSUES+=("Fallback state: $FALLBACK_STATE") [[ "$DIGEST_SMART_ON_FALLBACK" == true ]] && SHOULD_SEND=true fi fi else DIGEST_LINES+=("$ICON_FALLBACK Fallback: state file not found") fi # ── Container Skip List ─────────────────────────────────────────────────────────────────────── if [[ -f "$DOCKER_WATCHDOG_FAILED_FILE" ]] && [[ -s "$DOCKER_WATCHDOG_FAILED_FILE" ]]; then SKIP_COUNT=$(wc -l < "$DOCKER_WATCHDOG_FAILED_FILE") SKIP_LIST=$(cat "$DOCKER_WATCHDOG_FAILED_FILE" | tr '\n' ' ') DIGEST_LINES+=("$ICON_NOT_RUNNING Skip list: $SKIP_COUNT containers — $SKIP_LIST") ISSUES+=("Containers on skip list (manual intervention needed): $SKIP_LIST") SHOULD_SEND=true else DIGEST_LINES+=("$ICON_RUNNING Skip list: empty ✅") fi # ── Container Watchdog Strikes ──────────────────────────────────────────────────────────────── if [[ -f "$WATCHDOG_STATE_FILE" ]]; then ACTIVE_STRIKES=$(grep -v ":0$" "$WATCHDOG_STATE_FILE" 2>/dev/null | grep -c ".") if [[ "$ACTIVE_STRIKES" -gt 0 ]]; then STRIKE_LIST=$(grep -v ":0$" "$WATCHDOG_STATE_FILE" 2>/dev/null | tr '\n' ' ') DIGEST_LINES+=("$ICON_WATCHDOG Container strikes: $ACTIVE_STRIKES active — $STRIKE_LIST") FINDINGS+=("Container watchdog: $ACTIVE_STRIKES active strikes") [[ "$DIGEST_SMART_ON_WATCHDOG" == true ]] && SHOULD_SEND=true else DIGEST_LINES+=("$ICON_WATCHDOG Container watchdog: no active strikes ✅") fi fi # ── System Watchdog Strikes ─────────────────────────────────────────────────────────────────── if [[ -f "$SYS_WATCHDOG_STATE_FILE" ]]; then SYS_STRIKES=$(grep "^[^=]*:" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null | grep -cv -E "(:0$|:false$)") if [[ "$SYS_STRIKES" -gt 0 ]]; then SYS_STRIKE_LIST=$(grep "^[^=]*:" "$SYS_WATCHDOG_STATE_FILE" 2>/dev/null | grep -v -E "(:0$|:false$)" | tr '\n' ' ') DIGEST_LINES+=("$ICON_REBOOT_SMART System strikes: $SYS_STRIKES active — $SYS_STRIKE_LIST") FINDINGS+=("System watchdog: $SYS_STRIKES active strikes") [[ "$DIGEST_SMART_ON_WATCHDOG" == true ]] && SHOULD_SEND=true else DIGEST_LINES+=("$ICON_REBOOT_SMART System watchdog: no active strikes ✅") fi fi # ── Transcode Ramdisk ───────────────────────────────────────────────────────────────────────── if mountpoint -q "$RAMDISK_PATH" 2>/dev/null; then RAMDISK_USED_KB=$(df "$RAMDISK_PATH" --output=used 2>/dev/null | tail -1 | tr -d ' ') RAMDISK_USED_GB=$(kb_to_gb "$RAMDISK_USED_KB") SYMLINK_TARGET=$(readlink "$TRANSCODE_LINK" 2>/dev/null || echo "unknown") DIGEST_LINES+=("$ICON_RAM Transcodes: ${RAMDISK_USED_GB}GB used → $SYMLINK_TARGET") # Weekly transcode stats from TRANSCODE_DAILY_LOG if [[ -f "${TRANSCODE_DAILY_LOG:-}" ]]; then WEEK_CUTOFF=$(date -d "7 days ago" '+%Y-%m-%d') WEEK_PEAK=$(awk -F'|' -v c="$WEEK_CUTOFF" \ '$1>=c{if($2>max)max=$2} END{printf "%.2f",max+0}' "$TRANSCODE_DAILY_LOG") WEEK_FLIPS=$(awk -F'|' -v c="$WEEK_CUTOFF" \ '$1>=c{sum+=$3} END{print sum+0}' "$TRANSCODE_DAILY_LOG") WEEK_RAM=$(awk -F'|' -v c="$WEEK_CUTOFF" \ '$1>=c{sum+=$4} END{print sum+0}' "$TRANSCODE_DAILY_LOG") WEEK_SSD=$(awk -F'|' -v c="$WEEK_CUTOFF" \ '$1>=c{sum+=$5} END{print sum+0}' "$TRANSCODE_DAILY_LOG") WEEK_FILES=$(awk -F'|' -v c="$WEEK_CUTOFF" \ '$1>=c{sum+=$6} END{print sum+0}' "$TRANSCODE_DAILY_LOG") DIGEST_LINES+=("$ICON_RAM Transcode week: peak ${WEEK_PEAK}GB | flips: $WEEK_FLIPS | sessions: ${WEEK_RAM} ramdisk / ${WEEK_SSD} SSD | cleaned: ${WEEK_FILES} files") PEAK_INT=$(printf "%.0f" "$WEEK_PEAK") WARN_INT=$(printf "%.0f" "${RAMDISK_WARN_GB:-6.8}") if [[ "$PEAK_INT" -ge "$WARN_INT" ]]; then DIGEST_LINES+=("$ICON_WARN Peak ${WEEK_PEAK}GB near threshold ${RAMDISK_WARN_GB}GB — consider increasing HOST*_RAMDISK_SIZE") FINDINGS+=("Transcode ramdisk near threshold: ${WEEK_PEAK}GB / ${RAMDISK_WARN_GB}GB") fi fi else DIGEST_LINES+=("$ICON_RAM Transcodes: ramdisk not mounted") ISSUES+=("Ramdisk not mounted at $RAMDISK_PATH") SHOULD_SEND=true fi # ── Bandwidth ───────────────────────────────────────────────────────────────────────────────── # Updated for new log format: date|time|profile|duration|status|bytes|warn_flag if [[ -f "${BANDWIDTH_LOG:-}" ]] && [[ -s "$BANDWIDTH_LOG" ]]; then YESTERDAY=$(date -d "yesterday" '+%Y-%m-%d') YESTERDAY_BYTES=$(awk -F'|' -v d="$YESTERDAY" '$1==d{sum+=$6} END{print sum+0}' \ "$BANDWIDTH_LOG") YESTERDAY_GB=$(bytes_to_gb "$YESTERDAY_BYTES") YESTERDAY_LARGE=$(awk -F'|' -v d="$YESTERDAY" '$1==d && $7=="LARGE"' \ "$BANDWIDTH_LOG" | wc -l) if [[ "${YESTERDAY_LARGE:-0}" -gt 0 ]]; then DIGEST_LINES+=("$ICON_BANDWIDTH Yesterday: ${YESTERDAY_GB}GB — $YESTERDAY_LARGE large transfer(s) exceeded ${BANDWIDTH_WARN_GB}GB") FINDINGS+=("High bandwidth day: ${YESTERDAY_GB}GB transferred") [[ "$DIGEST_SMART_ON_BANDWIDTH" == true ]] && SHOULD_SEND=true else DIGEST_LINES+=("$ICON_BANDWIDTH Yesterday: ${YESTERDAY_GB}GB") fi else DIGEST_LINES+=("$ICON_BANDWIDTH Bandwidth: no data yet") fi # ── SSL Certificates ────────────────────────────────────────────────────────────────────────── if [[ ${#CERT_MONITOR_DOMAINS[@]} -gt 0 ]] && command -v openssl >/dev/null 2>&1; then CERT_ISSUES=() for domain in "${CERT_MONITOR_DOMAINS[@]}"; do [[ -z "$domain" ]] && continue if check_cert_expiry "$domain" 443 "${CERT_TIMEOUT:-10}"; then days_remaining="$_CERT_DAYS" if [[ "$days_remaining" -le "${CERT_CRIT_DAYS:-7}" ]]; then CERT_ISSUES+=("$domain: ${days_remaining}d CRITICAL") SHOULD_SEND=true elif [[ "$days_remaining" -le "${CERT_WARN_DAYS:-30}" ]]; then CERT_ISSUES+=("$domain: ${days_remaining}d warning") [[ "$DIGEST_SMART_ON_CERT_WARN" == true ]] && SHOULD_SEND=true else log "$ICON_CERT $domain: ${days_remaining}d remaining ✅" fi fi done if [[ ${#CERT_ISSUES[@]} -gt 0 ]]; then DIGEST_LINES+=("$ICON_CERT Certificates: ${CERT_ISSUES[*]}") FINDINGS+=("Cert issues: ${CERT_ISSUES[*]}") else DIGEST_LINES+=("$ICON_CERT Certificates: all healthy ✅") fi fi # ============================================================================================== # ── Smart profile — exit silently if nothing to report ──────────────────────────────────────── # ============================================================================================== if [[ "$DIGEST_PROFILE" == "smart" && "$SHOULD_SEND" == false ]]; then echo "Profile: smart — no findings worth reporting — silent exit" exit 0 fi # ============================================================================================== # ━━━ Build and Send Digest ━━━ # ============================================================================================== echo "" echo "━━━ $ICON_DIGEST Health Digest — $(date '+%Y-%m-%d %H:%M:%S') ━━━" echo "$ICON_HOST $MY_ID ($LOCAL_SERVER_NAME)" echo "" for line in "${DIGEST_LINES[@]}"; do echo " $line" done echo "" echo "━━━━━ $ICON_SUMMARY DIGEST SUMMARY ━━━━━" echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)" echo "$ICON_DIGEST Profile: $DIGEST_PROFILE" echo "$ICON_ERROR Issues: ${#ISSUES[@]}" echo "$ICON_WARN Findings: ${#FINDINGS[@]}" echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━" # Build notification message NOTIFY_MSG="Health Digest — $MY_ID ($LOCAL_SERVER_NAME)" [[ ${#ISSUES[@]} -gt 0 ]] && NOTIFY_MSG+=" | Issues: ${ISSUES[*]}" [[ ${#FINDINGS[@]} -gt 0 ]] && NOTIFY_MSG+=" | Findings: ${FINDINGS[*]}" [[ ${#ISSUES[@]} -eq 0 && ${#FINDINGS[@]} -eq 0 ]] && NOTIFY_MSG+=" | All systems healthy" NOTIFY_SEV="normal" [[ ${#ISSUES[@]} -gt 0 ]] && NOTIFY_SEV="warning" if [[ "$DRY_RUN" == true ]]; then warn "DRY RUN — digest generated but not sent" elif [[ "$SHOULD_SEND" == true ]]; then notify "$NOTIFY_MSG" "Health Digest" "$NOTIFY_SEV" echo "Digest sent" fi