untrack claude-bin and claude-data directories; add to gitignore

This commit is contained in:
Gmer4Lfe
2026-06-14 12:41:53 -04:00
parent 3964f6fb46
commit af76900aa6
963 changed files with 4 additions and 377635 deletions
+4
View File
@@ -27,6 +27,10 @@ varaverk-*.txz
# .txz packages are attached to GitHub releases, not committed to the repo. # .txz packages are attached to GitHub releases, not committed to the repo.
Plugin/dist/ Plugin/dist/
# ── Claude Code installation (lives alongside repo on flash, not source) ──────
claude-bin/
claude-data/
# ── OS / editor ─────────────────────────────────────────────────────────────── # ── OS / editor ───────────────────────────────────────────────────────────────
.DS_Store .DS_Store
*.swp *.swp
Binary file not shown.
-1
View File
@@ -1 +0,0 @@
{"claudeAiOauth":{"accessToken":"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","refreshToken":"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","expiresAt":1781475960207,"scopes":["user:file_upload","user:inference","user:mcp_servers","user:profile","user:sessions:claude_code"],"subscriptionType":"pro","rateLimitTier":"default_claude_ai"}}
-1
View File
@@ -1 +0,0 @@
2026-06-14T15:34:48.524Z
-1
View File
@@ -1 +0,0 @@
{"timestamp":"2026-06-13T13:53:50.240Z","path":"native","outcome":"success","status":"success","version_from":"2.1.176","version_to":"2.1.177","error_code":null}
@@ -1,785 +0,0 @@
{
"numStartups": 22,
"installMethod": "native",
"autoUpdates": false,
"hasSeenTasksHint": true,
"tipsHistory": {
"fotw-campaign-upsell": 13,
"new-user-warmup": 6,
"plan-mode-for-complex-tasks": 22,
"memory-command": 16,
"theme-command": 21,
"colorterm-truecolor": 1,
"status-line": 1,
"prompt-queue": 11,
"enter-to-steer-in-relatime": 21,
"todo-list": 21,
"ide-upsell-external-terminal": 19,
"install-github-app": 22,
"install-slack-app": 22,
"drag-and-drop-images": 14,
"double-esc-code-restore": 14,
"continue": 14,
"shift-tab": 15,
"image-paste": 4,
"web-app": 19,
"color-when-multi-clauding": 6,
"custom-agents": 21,
"remote-control": 21,
"voice-mode": 16,
"goal-command-nudge": 16,
"guest-passes": 22,
"feedback-command": 22,
"frontend-design-plugin": 6,
"permissions": 22,
"rename-conversation": 11,
"custom-commands": 11,
"c4e-remote-sessions": 18,
"subagent-fanout-nudge": 18,
"no-flicker": 19
},
"promptQueueUseCount": 43,
"cachedGrowthBookFeatures": {
"tengu_slate_kestrel": true,
"tengu_bridge_repl_v2": true,
"tengu_basalt_meadow": true,
"tengu_sage_compass2": {
"enabled": true
},
"tengu_kairos_loop_dynamic": true,
"tengu_sepia_cormorant": [],
"tengu_amber_heron": false,
"tengu_log_datadog_events": true,
"tengu-fable-off-switch": {
"activated": false
},
"tengu_quiet_slate_wren": false,
"tengu_birch_compass": true,
"tengu_bramble_lintel": 7,
"tengu_malort_pedway": {
"enabled": true,
"pixelValidation": false,
"clipboardPasteMultiline": true,
"screenshotFilter": true,
"mouseAnimation": true,
"hideBeforeAction": true,
"autoTargetDisplay": false,
"coordinateMode": "pixels"
},
"tengu_lilac_loom": {},
"tengu_sub_nomdrep_q7k": true,
"tengu_lantern_spool": false,
"tengu_hawthorn_steeple": false,
"tengu_version_config": {
"minVersion": "1.0.24"
},
"tengu_auto_notice_once": true,
"tengu_sparrow_ledger": false,
"tengu_loggia_carousel": false,
"tengu_ccr_bridge": true,
"tengu_basalt_sundial": false,
"tengu_mcp_stateless_skip_init": true,
"tengu_lapis_anchor": "off",
"tengu_sage_compass": {},
"tengu_kairos_cron": true,
"tengu_kairos_loop_prompt": true,
"tengu_jade_anvil_4": false,
"tengu_skills_dashboard_enabled": false,
"tengu_sedge_lantern_holdback": false,
"tengu_dunwich_bell": false,
"tengu_desktop_upsell": {
"enable_shortcut_tip": true,
"enable_startup_dialog": false
},
"tengu_code_diff_cli": true,
"tengu_anchor_tide": true,
"tengu_garnet_finch": false,
"tengu_cobalt_heron": true,
"tengu_ccr_v2_send_events_cli": true,
"tengu_onyx_plover": {
"enabled": false,
"minHours": 24,
"minSessions": 3,
"remoteEnabled": false
},
"tengu_react_vulnerability_warning": false,
"tengu_prompt_cache_1h_config": {
"allowlist": [
"repl_main_thread*",
"sdk",
"auto_mode",
"rolling_compact",
"memdir_relevance",
"agent_classifier",
"prompt_suggestion",
"away_summary",
"extract_memories",
"compact"
]
},
"tengu_timber_lark": "copy_a",
"tengu_ladder_mq7": false,
"tengu_birthday_hat": false,
"tengu_prompt_cache_diagnostics": true,
"tengu_worktree_mode": true,
"tengu_willow_refresh_ttl_hours": 0,
"tengu_pewter_kestrel": {
"global": 50000,
"Bash": 30000,
"PowerShell": 30000,
"Grep": 20000,
"Snip": 1000,
"StrReplaceBasedEditTool": 30000,
"BashSearchTool": 20000
},
"tengu_slate_finch": true,
"tengu_workflows_enabled": true,
"tengu_permission_friction": true,
"tengu_marble_lark": false,
"tengu_copper_fox": false,
"tengu_bridge_repl_v2_config": {
"init_retry_max_attempts": 3,
"init_retry_base_delay_ms": 500,
"init_retry_jitter_fraction": 0.25,
"init_retry_max_delay_ms": 4000,
"http_timeout_ms": 10000,
"uuid_dedup_buffer_size": 2000,
"heartbeat_interval_ms": 20000,
"heartbeat_jitter_fraction": 0.1,
"token_refresh_buffer_ms": 600000,
"teardown_archive_timeout_ms": 1500,
"connect_timeout_ms": 15000,
"min_version": "2.1.70",
"should_show_app_upgrade_message": false
},
"tengu_marble_whisper": true,
"tengu_maple_sundial": false,
"tengu_velvet_cascade": {},
"tengu_passport_quail": false,
"tengu_ember_latch": true,
"tengu_vscode_onboarding": false,
"tengu_fennel_kite_model": "",
"tengu_nimble_amber_prose": false,
"tengu_bridge_poll_interval_ms": 0,
"tengu_cobalt_wren": false,
"tengu_harbor_permissions": true,
"tengu_orchid_trellis": false,
"tengu_ccr_bridge_multi_session": true,
"tengu_bad_survey_transcript_ask_config": {
"probability": 1
},
"tengu_good_survey_transcript_ask_config": {
"probability": 0.5
},
"tengu_amber_sentinel": true,
"tengu_crimson_vector": false,
"tengu_drift_lantern": false,
"tengu_kestrel_arch": "OFF",
"tengu_read_dedup_killswitch": false,
"tengu_saffron_lattice": {
"enabled": false,
"planLimitsEndDate": "2026-06-22T10:00:00Z",
"hideRateLimitsDescription": true
},
"tengu_cloth_snorkel": false,
"tengu_system_prompt_global_cache": true,
"tengu_slate_moth": true,
"tengu_bridge_poll_interval_config": {
"poll_interval_ms_not_at_capacity": 2000,
"poll_interval_ms_at_capacity": 600000,
"heartbeat_interval_ms": 0,
"multisession_poll_interval_ms_not_at_capacity": 5000,
"multisession_poll_interval_ms_at_capacity": 60000,
"multisession_poll_interval_ms_partial_capacity": 5000,
"non_exclusive_heartbeat_interval_ms": 180000,
"session_keepalive_interval_ms": 0,
"session_keepalive_interval_v2_ms": 0
},
"tengu_gouda_loop": true,
"tengu_otk_slot_v1": false,
"tengu_pewter_lark": "off",
"tengu_walnut_prism": false,
"tengu_immediate_model_command": false,
"tengu_pewter_summit": true,
"tengu_fg_left_arrow_agents": true,
"tengu_willow_sentinel_ttl_hours": 1,
"tengu_pewter_lantern": false,
"tengu_desktop_upsell_v2": {
"enabled": false
},
"tengu_vellum_siding": false,
"tengu_vscode_feedback_survey": true,
"tengu_mcp_singleton_unwrap": true,
"tengu_coral_fern": false,
"tengu_trace_lantern": false,
"tengu_review_bughunter_config": {
"fleet_size": 5,
"max_duration_minutes": 10,
"agent_timeout_seconds": 600,
"total_wallclock_minutes": 22,
"model": "claude-opus-4-7",
"cost_note": "$5-$25",
"duration_note": "~5-10 min",
"enabled": true
},
"tengu_basalt_spur": false,
"tengu_crystal_beam": {
"budgetTokens": 0
},
"tengu_hawthorn_window": 200000,
"tengu_flint_harbor_share": false,
"tengu_bridge_attestation_enforce": false,
"tengu_compass_dial": true,
"tengu_moss_anchor": false,
"tengu_willow_census_ttl_hours": 24,
"tengu_compact_cache_prefix": true,
"tengu_cedar_hollow_7m": {},
"tengu_prompt_suggestion": true,
"tengu_crimson_echo": {},
"tengu_cork_m4q": true,
"tengu_classifier_summary_llm_emit": true,
"tengu_tide_elm": "off",
"tengu_ccr_bundle_seed_enabled": true,
"tengu_copper_wren": false,
"tengu_ember_trail": "0",
"tengu_gha_plugin_code_review": false,
"tengu_keybinding_customization_release": true,
"tengu_kairos_cron_durable": false,
"tengu_canary": {},
"tengu_mocha_barista": true,
"tengu_negative_interaction_transcript_ask_config": {
"probability": 0
},
"tengu_steady_lantern": false,
"tengu_malformed_tool_use_clean_retry": false,
"tengu_agent_list_attach": false,
"tengu_ultraplan_timeout_seconds": 5400,
"tengu_hazel_osprey_floor": 75000,
"tengu_brick_follow": false,
"tengu_slate_ribbon": true,
"tengu_slate_siskin": {
"enabled": false,
"timeoutMs": 8000,
"throttleMs": 30000,
"summaryLineThreshold": 5
},
"tengu_amber_rokovoko": 0.2,
"tengu_penguin_mode_promo": {
"discountPercent": 0,
"endDate": "Feb 16"
},
"tengu_slate_harrier": "off",
"tengu_lapis_thicket": false,
"tengu_harbor_willow": false,
"tengu_amber_anchor": false,
"tengu_tussock_oriole": false,
"tengu_tern_alloy": "copy_a",
"tengu_fgts": true,
"tengu_vellum_lantern": false,
"tengu_saffron_anchor": true,
"tengu_miraculo_the_bard": false,
"tengu_red_coaster": false,
"tengu_cobalt_compass": true,
"tengu_plum_vx3": true,
"tengu_mcp_subagent_prompt": true,
"tengu_mcp_local_oauth_blocked_hosts": {
"hosts": [
"microsoft365.mcp.claude.com",
"gmail.mcp.claude.com",
"gcal.mcp.claude.com"
]
},
"tengu_byte_stream_idle_timeout_ms": 180000,
"tengu_umber_petrel": false,
"tengu_prism_ledger": false,
"tengu_ccr_bundle_max_bytes": 104857600,
"tengu_amber_sextant": true,
"tengu_pewter_ledger": "OFF",
"tengu_amber_flint": true,
"tengu_disable_bypass_permissions_mode": false,
"tengu_walrus_canteen": false,
"tengu_ashen_kelp": true,
"tengu_plugin_official_mkt_git_fallback": true,
"tengu_max_version_config": {},
"tengu_cobalt_lantern": true,
"tengu_ultraplan_prompt_identifier": "visual_plan",
"tengu_swann_brevity": "focused",
"tengu_hazel_osprey": false,
"tengu_slate_meadow": true,
"tengu_amber_redwood2": "",
"tengu_frond_boric": {},
"tengu_slate_thimble": false,
"tengu_slate_nexus": true,
"tengu_chert_bezel": true,
"tengu_streaming_tool_execution2": true,
"tengu_event_watchdog_default_on": false,
"tengu_auto_mode_config": {
"enabled": "enabled",
"twoStageClassifier": true
},
"tengu_grey_step2": {
"enabled": true,
"dialogTitle": "We recommend medium effort for Opus",
"dialogDescription": "Effort determines how long Claude thinks for when completing your task. We recommend medium effort for most tasks to balance speed and intelligence and maximize rate limits. Use ultrathink to trigger high effort when needed."
},
"tengu_dune_wren": false,
"tengu_cedar_lantern": true,
"tengu_velvet_moth": 0.2,
"tengu_harbor_ledger": [
{
"marketplace": "claude-plugins-official",
"plugin": "discord"
},
{
"marketplace": "claude-plugins-official",
"plugin": "telegram"
},
{
"marketplace": "claude-plugins-official",
"plugin": "fakechat"
},
{
"marketplace": "claude-plugins-official",
"plugin": "imessage"
}
],
"tengu_harbor": true,
"tengu_amber_lynx": false,
"tengu_doorbell_agave": false,
"tengu_maple_tide": false,
"tengu_fennel_kite": false,
"tengu_collage_kaleidoscope": true,
"tengu_file_write_optimization": true,
"tengu_startup_notice": "",
"tengu_mcp_retry_failed_remote": false,
"tengu_session_memory": false,
"tengu_flint_harbor_prompt": {
"prompt": "You are helping a power user generate an onboarding guide for teammates who are new to Claude Code. The guide will live in the team's onboarding docs and can be pasted into Claude for an interactive walkthrough.\n\nYou're co-authoring this with them — collaborative and helpful, like a teammate who's done this before and is happy to share.\n\n## Usage data (last {{WINDOW_DAYS}} days)\n\nThis was scanned from the guide creator's local Claude Code transcripts:\n\n```json\n{{USAGE_DATA}}\n```\n\n## Your task\n\nBefore anything else — including before thinking through the classification — output exactly this line as your first visible text:\n\n> Looking at how you've used Claude over the last {{WINDOW_DAYS}} days to put together an onboarding guide for teammates new to Claude Code.\n\nThis must come before any extended thinking about session descriptors. The guide creator is staring at a blank screen until you do. Classification is step 2, not step 1.\n\nGenerate the guide immediately, then ask for revisions. Don't wait for answers first — it's easier for the guide creator to edit a concrete draft than answer abstract questions.\n\n1. **Output the acknowledgment line above.** No thinking, no classification, no tool calls before this. One line, then move on.\n\n2. **Derive the work-type breakdown.** Read the `sessionDescriptors` array — each entry describes one session via its title, any linked code reviews (`prNumbers`), and first user message. Classify each session into one of these task types:\n\n - **build_feature** — new functionality, scripts, tools, config/CI/env setup\n - **debug_fix** — investigating and fixing bugs\n - **improve_quality** — refactoring, tests, cleanup, code review\n - **analyze_data** — queries, metrics, number crunching\n - **plan_design** — architecture, approach, strategy, understanding unfamiliar code, design review\n - **prototype** — spikes, POCs, throwaway exploration\n - **write_docs** — PRDs, RFCs, READMEs, design docs, copy/doc review\n\n Categories describe the *type of task*, not the project or domain — a teammate on any project should recognize them. Review sessions belong with whatever's being reviewed: code review is improve_quality, doc review is write_docs, design review is plan_design. Most sessions fit the list; only invent a new category if it's genuinely a different type of task. Pick the top 3-5 with rough percentages. First messages alone are usually enough; titles and code-review links are enrichment. If first messages are uninformative, use tool and MCP counts as a weak hint. If there are ~0 sessions, leave the breakdown as a TODO.\n\n In the rendered guide, display categories with spaces and title case (e.g. \"Build Feature\" not \"build_feature\").\n\n3. **Gather the remaining pieces.** For repos, start with `currentRepo` and check the workspace for sibling repo directories. For MCP server setup, use each entry's `name` (and `urlOrigin` where present) to infer what the server does and how a teammate would get access. Leave the Team Tips and Get Started sections as TODO placeholders — you'll ask for these in Review and fill them in after.\n\n4. **Write the guide to `ONBOARDING.md`** following this template:\n\n```\n{{GUIDE_TEMPLATE}}\n```\n\n Fill in real numbers from the usage data (not placeholders). Use `generatedBy` for the name; if it's missing, omit the name. Ascii bar charts: `█` for filled, `░` for empty, 20 chars wide. Keep the HTML comment instruction at the bottom exactly as shown.\n\n5. **Render the guide in a code block, then close out the first turn.** You're co-authoring this guide with the guide creator — frame the follow-up as collaboration, not corrections.\n\n After the code block, add a `---` horizontal rule and a `**Review**` heading so the guide is visually separated from your questions. Under the heading, number these three questions:\n\n 1. \"I went with '[X]' for the team name — let me know if that sounds right.\" (or if you couldn't tell: \"What's the team name? I'll add it in.\")\n 2. Is there a starter task for someone new to Claude Code? (ticket or doc link — optional)\n 3. Any team tips you'd tell a new teammate that aren't already in CLAUDE.md?\n\n After they answer, update `ONBOARDING.md` with their team name, tips, and starter task. Then close with this exact line (not numbered, not paraphrased):\n\n Saved to `ONBOARDING.md`. Drop it in your team docs and channels — when a new teammate pastes it into Claude Code, they get a guided onboarding tour from there.\n\n Apply any edits they come back with to the file.",
"guideTemplate": "# Welcome to [Team Name]\n\n## How We Use Claude\n\nBased on [name]'s usage over the last [N] days:\n\nWork Type Breakdown:\n [Category 1] [ascii bar] [N]%\n [Category 2] [ascii bar] [N]%\n [Category 3] [ascii bar] [N]%\n ...\n\nTop Skills & Commands:\n [/command] [ascii bar] [N]x/month\n ...\n\nTop MCP Servers:\n [Server] [ascii bar] [N] calls\n ...\n\n## Your Setup Checklist\n\n### Codebases\n- [ ] [repo-name] — [repo url]\n...\n\n### MCP Servers to Activate\n- [ ] [Server] — [what it's for]. [How to get access]\n...\n\n### Skills to Know About\n- [/command] — [what it does, when the team uses it]\n...\n\n## Team Tips\n\n_TODO_\n\n## Get Started\n\n_TODO_\n\n<!-- INSTRUCTION FOR CLAUDE: A new teammate just pasted this guide for how the\nteam uses Claude Code. You're their onboarding buddy — warm, conversational,\nnot lecture-y.\n\nOpen with a warm welcome — include the team name from the title. Then: \"Your\nteammate uses Claude Code for [list all the work types]. Let's get you started.\"\n\nCheck what's already in place against everything under Setup Checklist\n(including skills), using markdown checkboxes — [x] done, [ ] not yet. Lead\nwith what they already have. One sentence per item, all in one message.\n\nTell them you'll help with setup, cover the actionable team tips, then the\nstarter task (if there is one). Offer to start with the first unchecked item,\nget their go-ahead, then work through the rest one by one.\n\nAfter setup, walk them through the remaining sections — offer to help where you\ncan (e.g. link to channels), and just surface the purely informational bits.\n\nDon't invent sections or summaries that aren't in the guide. The stats are the\nguide creator's personal usage data — don't extrapolate them into a \"team\nworkflow\" narrative. -->",
"windowDays": 30
},
"tengu_slim_subagent_claudemd": true,
"tengu_tangerine_ladder_boost": true,
"tengu_chair_sermon": false,
"tengu_gypsum_kite": true,
"tengu_quartz_heron": false,
"tengu_xterm_atlas_reset": true,
"tengu-model-error-overrides": {
"claude-fable-5": {
"block": "Claude Fable 5 is currently unavailable. Learn more: https://www.anthropic.com/news/fable-mythos-access"
}
},
"tengu_orchid_mantis_v2": true,
"tengu-off-switch": {
"activated": false
},
"tengu_feedback_survey_config": {
"minTimeBeforeFeedbackMs": 600000,
"minTimeBetweenFeedbackMs": 43200000,
"minTimeBetweenGlobalFeedbackMs": 43200000,
"minUserTurnsBeforeFeedback": 5,
"minUserTurnsBetweenFeedback": 25,
"hideThanksAfterMs": 3000,
"onForModels": [
"*"
],
"probability": 0.05
},
"tengu_cork_lantern": false,
"tengu_mint_lanes": false,
"tengu_bridge_attestation_enforce_config": {
"accept_level": "VERIFIED_BY_GATE",
"accept_statuses": []
},
"tengu_marble_sandcastle": false,
"tengu_bg_attach_stall_ms": 5000,
"tengu_workout2": true,
"tengu_orford_ness": false,
"tengu_porch_bell_9f": "",
"tengu_auto_mode_default_on": false,
"tengu_birch_kettle": false,
"tengu_classifier_summary_heuristic_emit": true,
"tengu_cobalt_thicket": false,
"tengu_destructive_command_warning": false,
"tengu_cinder_plover": "",
"tengu_cedar_halo": false,
"tengu_sotto_voce": true,
"tengu_sepia_moth": false,
"tengu_cedar_sundial": false,
"tengu_penguins_enabled": true,
"tengu_quiet_basalt_echo": false,
"tengu_ochre_hollow": true,
"tengu_coral_beacon": true,
"tengu_copper_thistle": false,
"tengu_1p_event_batch_config": {
"scheduledDelayMillis": 10000,
"maxExportBatchSize": 400,
"maxQueueSize": 8192,
"path": "/api/event_logging/v2/batch"
},
"tengu_amber_wren": {
"targetedRangeNudge": true,
"maxTokens": 25000
},
"tengu_amber_prism": true,
"tengu_cobalt_plinth": false,
"tengu_silent_harbor": false,
"tengu_chomp_inflection": true,
"tengu_mcp_elicitation": true,
"tengu_sm_config": {
"minimumMessageTokensToInit": 150000,
"minimumTokensBetweenUpdate": 40000,
"toolCallsBetweenUpdates": 10
},
"tengu_bridge_min_version": {
"minVersion": "2.1.70"
},
"tengu_kairos_input_needed_push": true,
"tengu_quiet_harbor": false,
"tengu_slate_wren": false,
"tengu_tool_search_unsupported_models": [
"claude-3-5-haiku",
"claude-3-haiku"
],
"tengu_native_cursor": true,
"tengu_orchid_mantis": false,
"tengu_amber_lark": true,
"tengu_shale_finch": true,
"tengu_cedar_plume": false,
"tengu_kairos_push_notifications": true,
"tengu_marble_whisper2": true,
"tengu_lichen_compass": false,
"tengu_c4w_usage_limit_notifications_enabled": true,
"tengu_scarf_coffee": false,
"tengu_copper_bridge": true,
"tengu_tool_pear": false,
"tengu_claudeai_mcp_connectors": true,
"tengu_ccr_post_turn_summary": false,
"tengu_sedge_lantern": true,
"tengu_feature_template": false,
"tengu_harbor_prism": true,
"tengu_cedar_inlet": "step",
"tengu_flax_grouse": false,
"tengu_event_sampling_config": {},
"tengu_herring_clock": false,
"tengu_quartz_vireo": "",
"tengu_team_discovery": false,
"tengu_gleaming_fair": true,
"tengu_marble_anvil": true,
"tengu_classifier_disabled_surfaces": "",
"tengu_pewter_brook": false,
"tengu_vscode_review_upsell": false,
"claude_code_skills_dashboard_enabled_cli": false,
"tengu_post_compact_survey": false,
"tengu_reactive_compact_remote": false,
"tengu_idle_amber_finch": false,
"tengu_noreread_q7m_velvet": false,
"tengu_ultraplan_config": {
"enabled": true
},
"tengu_scratch": false,
"tengu_alder_compass": false,
"tengu_olive_hinge": "",
"tengu_shining_fractals": false,
"tengu_maple_pier": false,
"tengu_sessions_elevated_auth_enforcement": true,
"tengu_turtle_carbon": true,
"tengu_billiard_aviary": false,
"tengu_cinder_almanac": true,
"tengu_osprey_lantern": false,
"tengu-top-of-feed-tip": {
"tip": "Claude Fable 5 is currently unavailable. Please use Opus 4.8 or another available model. Learn more: https://www.anthropic.com/news/fable-mythos-access",
"color": "warning"
},
"tengu_cobalt_raccoon": true,
"tengu_loud_sugary_rock": false,
"tengu_willow_mode": "hint_v2",
"tengu_blue_coaster": false,
"tengu_snippet_save": false,
"tengu_amber_lattice": {
"plugins": [
"security-guidance",
"code-review",
"commit-commands",
"code-simplifier",
"hookify",
"feature-dev",
"frontend-design",
"pr-review-toolkit",
"skill-creator",
"plugin-dev",
"agent-sdk-dev",
"mcp-server-dev",
"claude-code-setup",
"claude-md-management",
"playground",
"ralph-loop",
"explanatory-output-style",
"learning-output-style",
"clangd-lsp",
"csharp-lsp",
"gopls-lsp",
"jdtls-lsp",
"kotlin-lsp",
"lua-lsp",
"php-lsp",
"pyright-lsp",
"ruby-lsp",
"rust-analyzer-lsp",
"swift-lsp",
"typescript-lsp"
]
},
"tengu_slate_harbor_experiment": false,
"tengu_velvet_ibis": {},
"tengu_bridge_requires_action_details": true,
"tengu_lapis_finch": true,
"tengu_satin_quoll": {},
"tengu_moth_copse": false,
"tengu_silk_hinge": false,
"tengu_surreal_dali": true,
"tengu_cobalt_ridge": true,
"tengu_flint_harbor": false,
"tengu_plank_river_frost": "user_intent",
"tengu_velvet_mallet_haiku": false,
"tengu_velvet_mallet": false,
"tengu_velvet_mallet_haiku_4_5": false,
"tengu_velvet_hammer_falcon": false,
"tengu_loud_sugary_rock2": false,
"tengu_velvet_hammer_sonnet_4_5": false,
"tengu_velvet_hammer_sonnet": false,
"tengu_tab_read_sep": false,
"tengu_quill_harbor": "acceptEdits",
"tengu_velvet_hammer": false,
"tengu_velvet_hammer_opus": false,
"tengu_c4e_slash_upsell": true,
"tengu_velvet_hammer_haiku_4_5": false,
"tengu_feature_claudified_template": false,
"tengu_slate_quill": true,
"tengu_ax_screen_reader": false,
"tengu_windows_credman": false,
"tengu_basalt_tern": false,
"tengu_velvet_mallet_opus": false,
"tengu_velvet_hammer_haiku": false,
"tengu_velvet_static": true,
"tengu_velvet_mallet_sonnet": false,
"tengu_soft_slate_nudge": "baseline",
"tengu_lantern_hearth": "off",
"tengu_velvet_mallet_falcon": false,
"tengu_velvet_mallet_sonnet_4_5": false
},
"firstStartTime": "2026-06-05T19:39:28.542Z",
"opusProMigrationComplete": true,
"sonnet1m45MigrationComplete": true,
"seenNotifications": {},
"migrationVersion": 13,
"userID": "9d89994d486a4884b8cf33372d8a4cd61ebf7d34009e9d3cbce9db24e2e971a4",
"changelogLastFetched": 1781361371930,
"autoUpdatesProtectedForNative": true,
"claudeCodeFirstTokenDate": "2026-04-11T19:03:48.223040Z",
"hasCompletedOnboarding": true,
"lastOnboardingVersion": "2.1.165",
"groveConfigCache": {
"09792e21-2287-4348-b4d4-34cddbbfabc5": {
"grove_enabled": true,
"timestamp": 1781406640065
}
},
"cachedExperimentFeatures": [
"tengu_amber_prism",
"tengu_basalt_spur",
"tengu_cedar_inlet",
"tengu_coral_beacon",
"tengu_flint_harbor",
"tengu_mcp_subagent_prompt",
"tengu_ochre_hollow",
"tengu_orchid_mantis_v2",
"tengu_plank_river_frost",
"tengu_read_dedup_killswitch"
],
"cachedGrowthBookFeaturesAt": 1781406639973,
"lastReleaseNotesSeen": "2.1.177",
"projects": {
"/root": {
"allowedTools": [],
"mcpContextUris": [],
"mcpServers": {},
"enabledMcpjsonServers": [],
"disabledMcpjsonServers": [],
"hasTrustDialogAccepted": false,
"projectOnboardingSeenCount": 3,
"hasClaudeMdExternalIncludesApproved": false,
"hasClaudeMdExternalIncludesWarningShown": false,
"exampleFiles": [],
"lastGracefulShutdown": false,
"lastVersionBase": "2.1.177",
"lastCost": 1.0676417999999999,
"lastAPIDuration": 276732,
"lastAPIDurationWithoutRetries": 276675,
"lastToolDuration": 9607,
"lastDuration": 2130140,
"lastLinesAdded": 29,
"lastLinesRemoved": 15,
"lastTotalInputTokens": 4397,
"lastTotalOutputTokens": 16093,
"lastTotalCacheCreationInputTokens": 53595,
"lastTotalCacheReadInputTokens": 1642666,
"lastTotalWebSearchRequests": 0,
"lastFpsAverage": 1.82,
"lastFpsLow1Pct": 313.42,
"lastModelUsage": {
"claude-haiku-4-5-20251001": {
"inputTokens": 572,
"outputTokens": 17,
"cacheReadInputTokens": 0,
"cacheCreationInputTokens": 0,
"webSearchRequests": 0,
"costUSD": 0.000657
},
"claude-sonnet-4-6": {
"inputTokens": 3825,
"outputTokens": 16076,
"cacheReadInputTokens": 1642666,
"cacheCreationInputTokens": 53595,
"webSearchRequests": 0,
"costUSD": 1.0669847999999997
}
},
"lastSessionId": "96cf6b2d-d6a0-405b-81e5-95c657e1922a",
"lastSessionMetrics": {
"frame_duration_ms_count": 16776,
"frame_duration_ms_min": 0.11423300000024028,
"frame_duration_ms_max": 21.985366000095382,
"frame_duration_ms_avg": 0.7730811968292047,
"frame_duration_ms_p50": 0.5600509999203496,
"frame_duration_ms_p95": 1.786581499991007,
"frame_duration_ms_p99": 4.282493569953367,
"pre_tool_hook_duration_ms_count": 108,
"pre_tool_hook_duration_ms_min": 0,
"pre_tool_hook_duration_ms_max": 15,
"pre_tool_hook_duration_ms_avg": 0.24074074074074073,
"pre_tool_hook_duration_ms_p50": 0,
"pre_tool_hook_duration_ms_p95": 1,
"pre_tool_hook_duration_ms_p99": 4.789999999999978,
"hook_duration_ms_count": 40,
"hook_duration_ms_min": 0,
"hook_duration_ms_max": 8,
"hook_duration_ms_avg": 0.35,
"hook_duration_ms_p50": 0,
"hook_duration_ms_p95": 1,
"hook_duration_ms_p99": 5.269999999999996
},
"hasCompletedProjectOnboarding": true
}
},
"routineFiredWatermark": "2026-06-05T19:47:09.178Z",
"penguinModeOrgEnabled": true,
"closedIssuesLastChecked": 1781406639965,
"passesEligibilityCache": {
"4bb43199-0efc-4d5c-b552-79865cb0361b": {
"eligible": true,
"referral_code_details": {
"code": "BeGGjphr1g",
"campaign": "claude_code_guest_pass_a47c",
"referral_link": "https://claude.ai/referral/BeGGjphr1g"
},
"referrer_reward": {
"amount_minor_units": 1000,
"currency": "USD"
},
"remaining_passes": 3,
"limit": 3,
"share_link": "https://claude.ai/referral/BeGGjphr1g",
"terms_url": "https://support.claude.com/en/articles/12875061-claude-code-guest-passes",
"timestamp": 1781406640514
}
},
"cachedExtraUsageDisabledReason": "out_of_credits",
"passesUpsellSeenCount": 3,
"hasVisitedPasses": false,
"passesLastSeenRemaining": 3,
"officialMarketplaceAutoInstallAttempted": true,
"officialMarketplaceAutoInstalled": true,
"tipLifetimeShownCounts": {
"fotw-campaign-upsell": 6,
"new-user-warmup": 2,
"plan-mode-for-complex-tasks": 5,
"memory-command": 2,
"theme-command": 2,
"colorterm-truecolor": 1,
"status-line": 1,
"prompt-queue": 3,
"enter-to-steer-in-relatime": 2,
"todo-list": 2,
"ide-upsell-external-terminal": 5,
"install-github-app": 3,
"install-slack-app": 3,
"drag-and-drop-images": 2,
"double-esc-code-restore": 2,
"continue": 2,
"shift-tab": 2,
"image-paste": 1,
"web-app": 2,
"color-when-multi-clauding": 1,
"custom-agents": 2,
"remote-control": 2,
"voice-mode": 2,
"goal-command-nudge": 4,
"guest-passes": 6,
"feedback-command": 2,
"frontend-design-plugin": 1,
"permissions": 2,
"rename-conversation": 1,
"custom-commands": 1,
"c4e-remote-sessions": 1,
"subagent-fanout-nudge": 1,
"no-flicker": 1
},
"feedbackSurveyState": {
"lastShownTime": 1781411703066
},
"hasUsedBackslashReturn": true,
"agentLastUsed": {
"bg": 1780696781055
},
"remoteControlUpsellSeenCount": 3,
"fullscreenUpsellSeenCount": 3,
"lastShownEmergencyTip": "Claude Fable 5 is currently unavailable. Please use Opus 4.8 or another available model. Learn more: https://www.anthropic.com/news/fable-mythos-access",
"oauthAccount": {
"accountUuid": "09792e21-2287-4348-b4d4-34cddbbfabc5",
"emailAddress": "gmer4lfe@gmail.com",
"organizationUuid": "4bb43199-0efc-4d5c-b552-79865cb0361b",
"hasExtraUsageEnabled": true,
"billingType": "stripe_subscription",
"accountCreatedAt": "2026-04-03T21:52:35.642439Z",
"subscriptionCreatedAt": "2026-04-11T13:14:49.905923Z",
"ccOnboardingFlags": {},
"claudeCodeTrialEndsAt": null,
"claudeCodeTrialDurationDays": null,
"seatTier": null,
"displayName": "Gmer4Lfe",
"organizationRole": "admin",
"workspaceRole": null,
"organizationName": "gmer4lfe@gmail.com's Organization",
"organizationType": "claude_pro",
"organizationRateLimitTier": "default_claude_ai",
"userRateLimitTier": null
},
"clientDataCache": {
"cedar_lagoon": {
"claude-fable": true,
"claude-mythos": true
},
"pewter_owl_tool": true,
"pewter_owl_model": "claude-fable"
},
"additionalModelOptionsCache": [
{
"value": "claude-fable-5[1m]",
"label": "Fable (disabled)",
"description": "Claude Fable 5 is currently unavailable. Learn more: https://www.anthropic.com/news/fable-mythos-access",
"disabled": true
}
],
"additionalModelCostsCache": {}
}
@@ -1,785 +0,0 @@
{
"numStartups": 22,
"installMethod": "native",
"autoUpdates": false,
"hasSeenTasksHint": true,
"tipsHistory": {
"fotw-campaign-upsell": 13,
"new-user-warmup": 6,
"plan-mode-for-complex-tasks": 22,
"memory-command": 16,
"theme-command": 21,
"colorterm-truecolor": 1,
"status-line": 1,
"prompt-queue": 11,
"enter-to-steer-in-relatime": 21,
"todo-list": 21,
"ide-upsell-external-terminal": 19,
"install-github-app": 22,
"install-slack-app": 22,
"drag-and-drop-images": 14,
"double-esc-code-restore": 14,
"continue": 14,
"shift-tab": 15,
"image-paste": 4,
"web-app": 19,
"color-when-multi-clauding": 6,
"custom-agents": 21,
"remote-control": 21,
"voice-mode": 16,
"goal-command-nudge": 16,
"guest-passes": 22,
"feedback-command": 22,
"frontend-design-plugin": 6,
"permissions": 22,
"rename-conversation": 11,
"custom-commands": 11,
"c4e-remote-sessions": 18,
"subagent-fanout-nudge": 18,
"no-flicker": 19
},
"promptQueueUseCount": 44,
"cachedGrowthBookFeatures": {
"tengu_slate_kestrel": true,
"tengu_bridge_repl_v2": true,
"tengu_basalt_meadow": true,
"tengu_sage_compass2": {
"enabled": true
},
"tengu_kairos_loop_dynamic": true,
"tengu_sepia_cormorant": [],
"tengu_amber_heron": false,
"tengu_log_datadog_events": true,
"tengu-fable-off-switch": {
"activated": false
},
"tengu_quiet_slate_wren": false,
"tengu_birch_compass": true,
"tengu_bramble_lintel": 7,
"tengu_malort_pedway": {
"enabled": true,
"pixelValidation": false,
"clipboardPasteMultiline": true,
"screenshotFilter": true,
"mouseAnimation": true,
"hideBeforeAction": true,
"autoTargetDisplay": false,
"coordinateMode": "pixels"
},
"tengu_lilac_loom": {},
"tengu_sub_nomdrep_q7k": true,
"tengu_lantern_spool": false,
"tengu_hawthorn_steeple": false,
"tengu_version_config": {
"minVersion": "1.0.24"
},
"tengu_auto_notice_once": true,
"tengu_sparrow_ledger": false,
"tengu_loggia_carousel": false,
"tengu_ccr_bridge": true,
"tengu_basalt_sundial": false,
"tengu_mcp_stateless_skip_init": true,
"tengu_lapis_anchor": "off",
"tengu_sage_compass": {},
"tengu_kairos_cron": true,
"tengu_kairos_loop_prompt": true,
"tengu_jade_anvil_4": false,
"tengu_skills_dashboard_enabled": false,
"tengu_sedge_lantern_holdback": false,
"tengu_dunwich_bell": false,
"tengu_desktop_upsell": {
"enable_shortcut_tip": true,
"enable_startup_dialog": false
},
"tengu_code_diff_cli": true,
"tengu_anchor_tide": true,
"tengu_garnet_finch": false,
"tengu_cobalt_heron": true,
"tengu_ccr_v2_send_events_cli": true,
"tengu_onyx_plover": {
"enabled": false,
"minHours": 24,
"minSessions": 3,
"remoteEnabled": false
},
"tengu_react_vulnerability_warning": false,
"tengu_prompt_cache_1h_config": {
"allowlist": [
"repl_main_thread*",
"sdk",
"auto_mode",
"rolling_compact",
"memdir_relevance",
"agent_classifier",
"prompt_suggestion",
"away_summary",
"extract_memories",
"compact"
]
},
"tengu_timber_lark": "copy_a",
"tengu_ladder_mq7": false,
"tengu_birthday_hat": false,
"tengu_prompt_cache_diagnostics": true,
"tengu_worktree_mode": true,
"tengu_willow_refresh_ttl_hours": 0,
"tengu_pewter_kestrel": {
"global": 50000,
"Bash": 30000,
"PowerShell": 30000,
"Grep": 20000,
"Snip": 1000,
"StrReplaceBasedEditTool": 30000,
"BashSearchTool": 20000
},
"tengu_slate_finch": true,
"tengu_workflows_enabled": true,
"tengu_permission_friction": true,
"tengu_marble_lark": false,
"tengu_copper_fox": false,
"tengu_bridge_repl_v2_config": {
"init_retry_max_attempts": 3,
"init_retry_base_delay_ms": 500,
"init_retry_jitter_fraction": 0.25,
"init_retry_max_delay_ms": 4000,
"http_timeout_ms": 10000,
"uuid_dedup_buffer_size": 2000,
"heartbeat_interval_ms": 20000,
"heartbeat_jitter_fraction": 0.1,
"token_refresh_buffer_ms": 600000,
"teardown_archive_timeout_ms": 1500,
"connect_timeout_ms": 15000,
"min_version": "2.1.70",
"should_show_app_upgrade_message": false
},
"tengu_marble_whisper": true,
"tengu_maple_sundial": false,
"tengu_velvet_cascade": {},
"tengu_passport_quail": false,
"tengu_ember_latch": true,
"tengu_vscode_onboarding": false,
"tengu_fennel_kite_model": "",
"tengu_nimble_amber_prose": false,
"tengu_bridge_poll_interval_ms": 0,
"tengu_cobalt_wren": false,
"tengu_harbor_permissions": true,
"tengu_orchid_trellis": false,
"tengu_ccr_bridge_multi_session": true,
"tengu_bad_survey_transcript_ask_config": {
"probability": 1
},
"tengu_good_survey_transcript_ask_config": {
"probability": 0.5
},
"tengu_amber_sentinel": true,
"tengu_crimson_vector": false,
"tengu_drift_lantern": false,
"tengu_kestrel_arch": "OFF",
"tengu_read_dedup_killswitch": false,
"tengu_saffron_lattice": {
"enabled": false,
"planLimitsEndDate": "2026-06-22T10:00:00Z",
"hideRateLimitsDescription": true
},
"tengu_cloth_snorkel": false,
"tengu_system_prompt_global_cache": true,
"tengu_slate_moth": true,
"tengu_bridge_poll_interval_config": {
"poll_interval_ms_not_at_capacity": 2000,
"poll_interval_ms_at_capacity": 600000,
"heartbeat_interval_ms": 0,
"multisession_poll_interval_ms_not_at_capacity": 5000,
"multisession_poll_interval_ms_at_capacity": 60000,
"multisession_poll_interval_ms_partial_capacity": 5000,
"non_exclusive_heartbeat_interval_ms": 180000,
"session_keepalive_interval_ms": 0,
"session_keepalive_interval_v2_ms": 0
},
"tengu_gouda_loop": true,
"tengu_otk_slot_v1": false,
"tengu_pewter_lark": "off",
"tengu_walnut_prism": false,
"tengu_immediate_model_command": false,
"tengu_pewter_summit": true,
"tengu_fg_left_arrow_agents": true,
"tengu_willow_sentinel_ttl_hours": 1,
"tengu_pewter_lantern": false,
"tengu_desktop_upsell_v2": {
"enabled": false
},
"tengu_vellum_siding": false,
"tengu_vscode_feedback_survey": true,
"tengu_mcp_singleton_unwrap": true,
"tengu_coral_fern": false,
"tengu_trace_lantern": false,
"tengu_review_bughunter_config": {
"fleet_size": 5,
"max_duration_minutes": 10,
"agent_timeout_seconds": 600,
"total_wallclock_minutes": 22,
"model": "claude-opus-4-7",
"cost_note": "$5-$25",
"duration_note": "~5-10 min",
"enabled": true
},
"tengu_basalt_spur": false,
"tengu_crystal_beam": {
"budgetTokens": 0
},
"tengu_hawthorn_window": 200000,
"tengu_flint_harbor_share": false,
"tengu_bridge_attestation_enforce": false,
"tengu_compass_dial": true,
"tengu_moss_anchor": false,
"tengu_willow_census_ttl_hours": 24,
"tengu_compact_cache_prefix": true,
"tengu_cedar_hollow_7m": {},
"tengu_prompt_suggestion": true,
"tengu_crimson_echo": {},
"tengu_cork_m4q": true,
"tengu_classifier_summary_llm_emit": true,
"tengu_tide_elm": "off",
"tengu_ccr_bundle_seed_enabled": true,
"tengu_copper_wren": false,
"tengu_ember_trail": "0",
"tengu_gha_plugin_code_review": false,
"tengu_keybinding_customization_release": true,
"tengu_kairos_cron_durable": false,
"tengu_canary": {},
"tengu_mocha_barista": true,
"tengu_negative_interaction_transcript_ask_config": {
"probability": 0
},
"tengu_steady_lantern": false,
"tengu_malformed_tool_use_clean_retry": false,
"tengu_agent_list_attach": false,
"tengu_ultraplan_timeout_seconds": 5400,
"tengu_hazel_osprey_floor": 75000,
"tengu_brick_follow": false,
"tengu_slate_ribbon": true,
"tengu_slate_siskin": {
"enabled": false,
"timeoutMs": 8000,
"throttleMs": 30000,
"summaryLineThreshold": 5
},
"tengu_amber_rokovoko": 0.2,
"tengu_penguin_mode_promo": {
"discountPercent": 0,
"endDate": "Feb 16"
},
"tengu_slate_harrier": "off",
"tengu_lapis_thicket": false,
"tengu_harbor_willow": false,
"tengu_amber_anchor": false,
"tengu_tussock_oriole": false,
"tengu_tern_alloy": "copy_a",
"tengu_fgts": true,
"tengu_vellum_lantern": false,
"tengu_saffron_anchor": true,
"tengu_miraculo_the_bard": false,
"tengu_red_coaster": false,
"tengu_cobalt_compass": true,
"tengu_plum_vx3": true,
"tengu_mcp_subagent_prompt": true,
"tengu_mcp_local_oauth_blocked_hosts": {
"hosts": [
"microsoft365.mcp.claude.com",
"gmail.mcp.claude.com",
"gcal.mcp.claude.com"
]
},
"tengu_byte_stream_idle_timeout_ms": 180000,
"tengu_umber_petrel": false,
"tengu_prism_ledger": false,
"tengu_ccr_bundle_max_bytes": 104857600,
"tengu_amber_sextant": true,
"tengu_pewter_ledger": "OFF",
"tengu_amber_flint": true,
"tengu_disable_bypass_permissions_mode": false,
"tengu_walrus_canteen": false,
"tengu_ashen_kelp": true,
"tengu_plugin_official_mkt_git_fallback": true,
"tengu_max_version_config": {},
"tengu_cobalt_lantern": true,
"tengu_ultraplan_prompt_identifier": "visual_plan",
"tengu_swann_brevity": "focused",
"tengu_hazel_osprey": false,
"tengu_slate_meadow": true,
"tengu_amber_redwood2": "",
"tengu_frond_boric": {},
"tengu_slate_thimble": false,
"tengu_slate_nexus": true,
"tengu_chert_bezel": true,
"tengu_streaming_tool_execution2": true,
"tengu_event_watchdog_default_on": false,
"tengu_auto_mode_config": {
"enabled": "enabled",
"twoStageClassifier": true
},
"tengu_grey_step2": {
"enabled": true,
"dialogTitle": "We recommend medium effort for Opus",
"dialogDescription": "Effort determines how long Claude thinks for when completing your task. We recommend medium effort for most tasks to balance speed and intelligence and maximize rate limits. Use ultrathink to trigger high effort when needed."
},
"tengu_dune_wren": false,
"tengu_cedar_lantern": true,
"tengu_velvet_moth": 0.2,
"tengu_harbor_ledger": [
{
"marketplace": "claude-plugins-official",
"plugin": "discord"
},
{
"marketplace": "claude-plugins-official",
"plugin": "telegram"
},
{
"marketplace": "claude-plugins-official",
"plugin": "fakechat"
},
{
"marketplace": "claude-plugins-official",
"plugin": "imessage"
}
],
"tengu_harbor": true,
"tengu_amber_lynx": false,
"tengu_doorbell_agave": false,
"tengu_maple_tide": false,
"tengu_fennel_kite": false,
"tengu_collage_kaleidoscope": true,
"tengu_file_write_optimization": true,
"tengu_startup_notice": "",
"tengu_mcp_retry_failed_remote": false,
"tengu_session_memory": false,
"tengu_flint_harbor_prompt": {
"prompt": "You are helping a power user generate an onboarding guide for teammates who are new to Claude Code. The guide will live in the team's onboarding docs and can be pasted into Claude for an interactive walkthrough.\n\nYou're co-authoring this with them — collaborative and helpful, like a teammate who's done this before and is happy to share.\n\n## Usage data (last {{WINDOW_DAYS}} days)\n\nThis was scanned from the guide creator's local Claude Code transcripts:\n\n```json\n{{USAGE_DATA}}\n```\n\n## Your task\n\nBefore anything else — including before thinking through the classification — output exactly this line as your first visible text:\n\n> Looking at how you've used Claude over the last {{WINDOW_DAYS}} days to put together an onboarding guide for teammates new to Claude Code.\n\nThis must come before any extended thinking about session descriptors. The guide creator is staring at a blank screen until you do. Classification is step 2, not step 1.\n\nGenerate the guide immediately, then ask for revisions. Don't wait for answers first — it's easier for the guide creator to edit a concrete draft than answer abstract questions.\n\n1. **Output the acknowledgment line above.** No thinking, no classification, no tool calls before this. One line, then move on.\n\n2. **Derive the work-type breakdown.** Read the `sessionDescriptors` array — each entry describes one session via its title, any linked code reviews (`prNumbers`), and first user message. Classify each session into one of these task types:\n\n - **build_feature** — new functionality, scripts, tools, config/CI/env setup\n - **debug_fix** — investigating and fixing bugs\n - **improve_quality** — refactoring, tests, cleanup, code review\n - **analyze_data** — queries, metrics, number crunching\n - **plan_design** — architecture, approach, strategy, understanding unfamiliar code, design review\n - **prototype** — spikes, POCs, throwaway exploration\n - **write_docs** — PRDs, RFCs, READMEs, design docs, copy/doc review\n\n Categories describe the *type of task*, not the project or domain — a teammate on any project should recognize them. Review sessions belong with whatever's being reviewed: code review is improve_quality, doc review is write_docs, design review is plan_design. Most sessions fit the list; only invent a new category if it's genuinely a different type of task. Pick the top 3-5 with rough percentages. First messages alone are usually enough; titles and code-review links are enrichment. If first messages are uninformative, use tool and MCP counts as a weak hint. If there are ~0 sessions, leave the breakdown as a TODO.\n\n In the rendered guide, display categories with spaces and title case (e.g. \"Build Feature\" not \"build_feature\").\n\n3. **Gather the remaining pieces.** For repos, start with `currentRepo` and check the workspace for sibling repo directories. For MCP server setup, use each entry's `name` (and `urlOrigin` where present) to infer what the server does and how a teammate would get access. Leave the Team Tips and Get Started sections as TODO placeholders — you'll ask for these in Review and fill them in after.\n\n4. **Write the guide to `ONBOARDING.md`** following this template:\n\n```\n{{GUIDE_TEMPLATE}}\n```\n\n Fill in real numbers from the usage data (not placeholders). Use `generatedBy` for the name; if it's missing, omit the name. Ascii bar charts: `█` for filled, `░` for empty, 20 chars wide. Keep the HTML comment instruction at the bottom exactly as shown.\n\n5. **Render the guide in a code block, then close out the first turn.** You're co-authoring this guide with the guide creator — frame the follow-up as collaboration, not corrections.\n\n After the code block, add a `---` horizontal rule and a `**Review**` heading so the guide is visually separated from your questions. Under the heading, number these three questions:\n\n 1. \"I went with '[X]' for the team name — let me know if that sounds right.\" (or if you couldn't tell: \"What's the team name? I'll add it in.\")\n 2. Is there a starter task for someone new to Claude Code? (ticket or doc link — optional)\n 3. Any team tips you'd tell a new teammate that aren't already in CLAUDE.md?\n\n After they answer, update `ONBOARDING.md` with their team name, tips, and starter task. Then close with this exact line (not numbered, not paraphrased):\n\n Saved to `ONBOARDING.md`. Drop it in your team docs and channels — when a new teammate pastes it into Claude Code, they get a guided onboarding tour from there.\n\n Apply any edits they come back with to the file.",
"guideTemplate": "# Welcome to [Team Name]\n\n## How We Use Claude\n\nBased on [name]'s usage over the last [N] days:\n\nWork Type Breakdown:\n [Category 1] [ascii bar] [N]%\n [Category 2] [ascii bar] [N]%\n [Category 3] [ascii bar] [N]%\n ...\n\nTop Skills & Commands:\n [/command] [ascii bar] [N]x/month\n ...\n\nTop MCP Servers:\n [Server] [ascii bar] [N] calls\n ...\n\n## Your Setup Checklist\n\n### Codebases\n- [ ] [repo-name] — [repo url]\n...\n\n### MCP Servers to Activate\n- [ ] [Server] — [what it's for]. [How to get access]\n...\n\n### Skills to Know About\n- [/command] — [what it does, when the team uses it]\n...\n\n## Team Tips\n\n_TODO_\n\n## Get Started\n\n_TODO_\n\n<!-- INSTRUCTION FOR CLAUDE: A new teammate just pasted this guide for how the\nteam uses Claude Code. You're their onboarding buddy — warm, conversational,\nnot lecture-y.\n\nOpen with a warm welcome — include the team name from the title. Then: \"Your\nteammate uses Claude Code for [list all the work types]. Let's get you started.\"\n\nCheck what's already in place against everything under Setup Checklist\n(including skills), using markdown checkboxes — [x] done, [ ] not yet. Lead\nwith what they already have. One sentence per item, all in one message.\n\nTell them you'll help with setup, cover the actionable team tips, then the\nstarter task (if there is one). Offer to start with the first unchecked item,\nget their go-ahead, then work through the rest one by one.\n\nAfter setup, walk them through the remaining sections — offer to help where you\ncan (e.g. link to channels), and just surface the purely informational bits.\n\nDon't invent sections or summaries that aren't in the guide. The stats are the\nguide creator's personal usage data — don't extrapolate them into a \"team\nworkflow\" narrative. -->",
"windowDays": 30
},
"tengu_slim_subagent_claudemd": true,
"tengu_tangerine_ladder_boost": true,
"tengu_chair_sermon": false,
"tengu_gypsum_kite": true,
"tengu_quartz_heron": false,
"tengu_xterm_atlas_reset": true,
"tengu-model-error-overrides": {
"claude-fable-5": {
"block": "Claude Fable 5 is currently unavailable. Learn more: https://www.anthropic.com/news/fable-mythos-access"
}
},
"tengu_orchid_mantis_v2": true,
"tengu-off-switch": {
"activated": false
},
"tengu_feedback_survey_config": {
"minTimeBeforeFeedbackMs": 600000,
"minTimeBetweenFeedbackMs": 43200000,
"minTimeBetweenGlobalFeedbackMs": 43200000,
"minUserTurnsBeforeFeedback": 5,
"minUserTurnsBetweenFeedback": 25,
"hideThanksAfterMs": 3000,
"onForModels": [
"*"
],
"probability": 0.05
},
"tengu_cork_lantern": false,
"tengu_mint_lanes": false,
"tengu_bridge_attestation_enforce_config": {
"accept_level": "VERIFIED_BY_GATE",
"accept_statuses": []
},
"tengu_marble_sandcastle": false,
"tengu_bg_attach_stall_ms": 5000,
"tengu_workout2": true,
"tengu_orford_ness": false,
"tengu_porch_bell_9f": "",
"tengu_auto_mode_default_on": false,
"tengu_birch_kettle": false,
"tengu_classifier_summary_heuristic_emit": true,
"tengu_cobalt_thicket": false,
"tengu_destructive_command_warning": false,
"tengu_cinder_plover": "",
"tengu_cedar_halo": false,
"tengu_sotto_voce": true,
"tengu_sepia_moth": false,
"tengu_cedar_sundial": false,
"tengu_penguins_enabled": true,
"tengu_quiet_basalt_echo": false,
"tengu_ochre_hollow": true,
"tengu_coral_beacon": true,
"tengu_copper_thistle": false,
"tengu_1p_event_batch_config": {
"scheduledDelayMillis": 10000,
"maxExportBatchSize": 400,
"maxQueueSize": 8192,
"path": "/api/event_logging/v2/batch"
},
"tengu_amber_wren": {
"targetedRangeNudge": true,
"maxTokens": 25000
},
"tengu_amber_prism": true,
"tengu_cobalt_plinth": false,
"tengu_silent_harbor": false,
"tengu_chomp_inflection": true,
"tengu_mcp_elicitation": true,
"tengu_sm_config": {
"minimumMessageTokensToInit": 150000,
"minimumTokensBetweenUpdate": 40000,
"toolCallsBetweenUpdates": 10
},
"tengu_bridge_min_version": {
"minVersion": "2.1.70"
},
"tengu_kairos_input_needed_push": true,
"tengu_quiet_harbor": false,
"tengu_slate_wren": false,
"tengu_tool_search_unsupported_models": [
"claude-3-5-haiku",
"claude-3-haiku"
],
"tengu_native_cursor": true,
"tengu_orchid_mantis": false,
"tengu_amber_lark": true,
"tengu_shale_finch": true,
"tengu_cedar_plume": false,
"tengu_kairos_push_notifications": true,
"tengu_marble_whisper2": true,
"tengu_lichen_compass": false,
"tengu_c4w_usage_limit_notifications_enabled": true,
"tengu_scarf_coffee": false,
"tengu_copper_bridge": true,
"tengu_tool_pear": false,
"tengu_claudeai_mcp_connectors": true,
"tengu_ccr_post_turn_summary": false,
"tengu_sedge_lantern": true,
"tengu_feature_template": false,
"tengu_harbor_prism": true,
"tengu_cedar_inlet": "step",
"tengu_flax_grouse": false,
"tengu_event_sampling_config": {},
"tengu_herring_clock": false,
"tengu_quartz_vireo": "",
"tengu_team_discovery": false,
"tengu_gleaming_fair": true,
"tengu_marble_anvil": true,
"tengu_classifier_disabled_surfaces": "",
"tengu_pewter_brook": false,
"tengu_vscode_review_upsell": false,
"claude_code_skills_dashboard_enabled_cli": false,
"tengu_post_compact_survey": false,
"tengu_reactive_compact_remote": false,
"tengu_idle_amber_finch": false,
"tengu_noreread_q7m_velvet": false,
"tengu_ultraplan_config": {
"enabled": true
},
"tengu_scratch": false,
"tengu_alder_compass": false,
"tengu_olive_hinge": "",
"tengu_shining_fractals": false,
"tengu_maple_pier": false,
"tengu_sessions_elevated_auth_enforcement": true,
"tengu_turtle_carbon": true,
"tengu_billiard_aviary": false,
"tengu_cinder_almanac": true,
"tengu_osprey_lantern": false,
"tengu-top-of-feed-tip": {
"tip": "Claude Fable 5 is currently unavailable. Please use Opus 4.8 or another available model. Learn more: https://www.anthropic.com/news/fable-mythos-access",
"color": "warning"
},
"tengu_cobalt_raccoon": true,
"tengu_loud_sugary_rock": false,
"tengu_willow_mode": "hint_v2",
"tengu_blue_coaster": false,
"tengu_snippet_save": false,
"tengu_amber_lattice": {
"plugins": [
"security-guidance",
"code-review",
"commit-commands",
"code-simplifier",
"hookify",
"feature-dev",
"frontend-design",
"pr-review-toolkit",
"skill-creator",
"plugin-dev",
"agent-sdk-dev",
"mcp-server-dev",
"claude-code-setup",
"claude-md-management",
"playground",
"ralph-loop",
"explanatory-output-style",
"learning-output-style",
"clangd-lsp",
"csharp-lsp",
"gopls-lsp",
"jdtls-lsp",
"kotlin-lsp",
"lua-lsp",
"php-lsp",
"pyright-lsp",
"ruby-lsp",
"rust-analyzer-lsp",
"swift-lsp",
"typescript-lsp"
]
},
"tengu_slate_harbor_experiment": false,
"tengu_velvet_ibis": {},
"tengu_bridge_requires_action_details": true,
"tengu_lapis_finch": true,
"tengu_satin_quoll": {},
"tengu_moth_copse": false,
"tengu_silk_hinge": false,
"tengu_surreal_dali": true,
"tengu_cobalt_ridge": true,
"tengu_flint_harbor": false,
"tengu_plank_river_frost": "user_intent",
"tengu_velvet_mallet_haiku": false,
"tengu_velvet_mallet": false,
"tengu_velvet_mallet_haiku_4_5": false,
"tengu_velvet_hammer_falcon": false,
"tengu_loud_sugary_rock2": false,
"tengu_velvet_hammer_sonnet_4_5": false,
"tengu_velvet_hammer_sonnet": false,
"tengu_tab_read_sep": false,
"tengu_quill_harbor": "acceptEdits",
"tengu_velvet_hammer": false,
"tengu_velvet_hammer_opus": false,
"tengu_c4e_slash_upsell": true,
"tengu_velvet_hammer_haiku_4_5": false,
"tengu_feature_claudified_template": false,
"tengu_slate_quill": true,
"tengu_ax_screen_reader": false,
"tengu_windows_credman": false,
"tengu_basalt_tern": false,
"tengu_velvet_mallet_opus": false,
"tengu_velvet_hammer_haiku": false,
"tengu_velvet_static": true,
"tengu_velvet_mallet_sonnet": false,
"tengu_soft_slate_nudge": "baseline",
"tengu_lantern_hearth": "off",
"tengu_velvet_mallet_falcon": false,
"tengu_velvet_mallet_sonnet_4_5": false
},
"firstStartTime": "2026-06-05T19:39:28.542Z",
"opusProMigrationComplete": true,
"sonnet1m45MigrationComplete": true,
"seenNotifications": {},
"migrationVersion": 13,
"userID": "9d89994d486a4884b8cf33372d8a4cd61ebf7d34009e9d3cbce9db24e2e971a4",
"changelogLastFetched": 1781361371930,
"autoUpdatesProtectedForNative": true,
"claudeCodeFirstTokenDate": "2026-04-11T19:03:48.223040Z",
"hasCompletedOnboarding": true,
"lastOnboardingVersion": "2.1.165",
"groveConfigCache": {
"09792e21-2287-4348-b4d4-34cddbbfabc5": {
"grove_enabled": true,
"timestamp": 1781406640065
}
},
"cachedExperimentFeatures": [
"tengu_amber_prism",
"tengu_basalt_spur",
"tengu_cedar_inlet",
"tengu_coral_beacon",
"tengu_flint_harbor",
"tengu_mcp_subagent_prompt",
"tengu_ochre_hollow",
"tengu_orchid_mantis_v2",
"tengu_plank_river_frost",
"tengu_read_dedup_killswitch"
],
"cachedGrowthBookFeaturesAt": 1781406639973,
"lastReleaseNotesSeen": "2.1.177",
"projects": {
"/root": {
"allowedTools": [],
"mcpContextUris": [],
"mcpServers": {},
"enabledMcpjsonServers": [],
"disabledMcpjsonServers": [],
"hasTrustDialogAccepted": false,
"projectOnboardingSeenCount": 3,
"hasClaudeMdExternalIncludesApproved": false,
"hasClaudeMdExternalIncludesWarningShown": false,
"exampleFiles": [],
"lastGracefulShutdown": false,
"lastVersionBase": "2.1.177",
"lastCost": 1.0676417999999999,
"lastAPIDuration": 276732,
"lastAPIDurationWithoutRetries": 276675,
"lastToolDuration": 9607,
"lastDuration": 2130140,
"lastLinesAdded": 29,
"lastLinesRemoved": 15,
"lastTotalInputTokens": 4397,
"lastTotalOutputTokens": 16093,
"lastTotalCacheCreationInputTokens": 53595,
"lastTotalCacheReadInputTokens": 1642666,
"lastTotalWebSearchRequests": 0,
"lastFpsAverage": 1.82,
"lastFpsLow1Pct": 313.42,
"lastModelUsage": {
"claude-haiku-4-5-20251001": {
"inputTokens": 572,
"outputTokens": 17,
"cacheReadInputTokens": 0,
"cacheCreationInputTokens": 0,
"webSearchRequests": 0,
"costUSD": 0.000657
},
"claude-sonnet-4-6": {
"inputTokens": 3825,
"outputTokens": 16076,
"cacheReadInputTokens": 1642666,
"cacheCreationInputTokens": 53595,
"webSearchRequests": 0,
"costUSD": 1.0669847999999997
}
},
"lastSessionId": "96cf6b2d-d6a0-405b-81e5-95c657e1922a",
"lastSessionMetrics": {
"frame_duration_ms_count": 16776,
"frame_duration_ms_min": 0.11423300000024028,
"frame_duration_ms_max": 21.985366000095382,
"frame_duration_ms_avg": 0.7730811968292047,
"frame_duration_ms_p50": 0.5600509999203496,
"frame_duration_ms_p95": 1.786581499991007,
"frame_duration_ms_p99": 4.282493569953367,
"pre_tool_hook_duration_ms_count": 108,
"pre_tool_hook_duration_ms_min": 0,
"pre_tool_hook_duration_ms_max": 15,
"pre_tool_hook_duration_ms_avg": 0.24074074074074073,
"pre_tool_hook_duration_ms_p50": 0,
"pre_tool_hook_duration_ms_p95": 1,
"pre_tool_hook_duration_ms_p99": 4.789999999999978,
"hook_duration_ms_count": 40,
"hook_duration_ms_min": 0,
"hook_duration_ms_max": 8,
"hook_duration_ms_avg": 0.35,
"hook_duration_ms_p50": 0,
"hook_duration_ms_p95": 1,
"hook_duration_ms_p99": 5.269999999999996
},
"hasCompletedProjectOnboarding": true
}
},
"routineFiredWatermark": "2026-06-05T19:47:09.178Z",
"penguinModeOrgEnabled": true,
"closedIssuesLastChecked": 1781406639965,
"passesEligibilityCache": {
"4bb43199-0efc-4d5c-b552-79865cb0361b": {
"eligible": true,
"referral_code_details": {
"code": "BeGGjphr1g",
"campaign": "claude_code_guest_pass_a47c",
"referral_link": "https://claude.ai/referral/BeGGjphr1g"
},
"referrer_reward": {
"amount_minor_units": 1000,
"currency": "USD"
},
"remaining_passes": 3,
"limit": 3,
"share_link": "https://claude.ai/referral/BeGGjphr1g",
"terms_url": "https://support.claude.com/en/articles/12875061-claude-code-guest-passes",
"timestamp": 1781406640514
}
},
"cachedExtraUsageDisabledReason": "out_of_credits",
"passesUpsellSeenCount": 3,
"hasVisitedPasses": false,
"passesLastSeenRemaining": 3,
"officialMarketplaceAutoInstallAttempted": true,
"officialMarketplaceAutoInstalled": true,
"tipLifetimeShownCounts": {
"fotw-campaign-upsell": 6,
"new-user-warmup": 2,
"plan-mode-for-complex-tasks": 5,
"memory-command": 2,
"theme-command": 2,
"colorterm-truecolor": 1,
"status-line": 1,
"prompt-queue": 3,
"enter-to-steer-in-relatime": 2,
"todo-list": 2,
"ide-upsell-external-terminal": 5,
"install-github-app": 3,
"install-slack-app": 3,
"drag-and-drop-images": 2,
"double-esc-code-restore": 2,
"continue": 2,
"shift-tab": 2,
"image-paste": 1,
"web-app": 2,
"color-when-multi-clauding": 1,
"custom-agents": 2,
"remote-control": 2,
"voice-mode": 2,
"goal-command-nudge": 4,
"guest-passes": 6,
"feedback-command": 2,
"frontend-design-plugin": 1,
"permissions": 2,
"rename-conversation": 1,
"custom-commands": 1,
"c4e-remote-sessions": 1,
"subagent-fanout-nudge": 1,
"no-flicker": 1
},
"feedbackSurveyState": {
"lastShownTime": 1781411703066
},
"hasUsedBackslashReturn": true,
"agentLastUsed": {
"bg": 1780696781055
},
"remoteControlUpsellSeenCount": 3,
"fullscreenUpsellSeenCount": 3,
"lastShownEmergencyTip": "Claude Fable 5 is currently unavailable. Please use Opus 4.8 or another available model. Learn more: https://www.anthropic.com/news/fable-mythos-access",
"oauthAccount": {
"accountUuid": "09792e21-2287-4348-b4d4-34cddbbfabc5",
"emailAddress": "gmer4lfe@gmail.com",
"organizationUuid": "4bb43199-0efc-4d5c-b552-79865cb0361b",
"hasExtraUsageEnabled": true,
"billingType": "stripe_subscription",
"accountCreatedAt": "2026-04-03T21:52:35.642439Z",
"subscriptionCreatedAt": "2026-04-11T13:14:49.905923Z",
"ccOnboardingFlags": {},
"claudeCodeTrialEndsAt": null,
"claudeCodeTrialDurationDays": null,
"seatTier": null,
"displayName": "Gmer4Lfe",
"organizationRole": "admin",
"workspaceRole": null,
"organizationName": "gmer4lfe@gmail.com's Organization",
"organizationType": "claude_pro",
"organizationRateLimitTier": "default_claude_ai",
"userRateLimitTier": null
},
"clientDataCache": {
"cedar_lagoon": {
"claude-fable": true,
"claude-mythos": true
},
"pewter_owl_tool": true,
"pewter_owl_model": "claude-fable"
},
"additionalModelOptionsCache": [
{
"value": "claude-fable-5[1m]",
"label": "Fable (disabled)",
"description": "Claude Fable 5 is currently unavailable. Learn more: https://www.anthropic.com/news/fable-mythos-access",
"disabled": true
}
],
"additionalModelCostsCache": {}
}
@@ -1,785 +0,0 @@
{
"numStartups": 22,
"installMethod": "native",
"autoUpdates": false,
"hasSeenTasksHint": true,
"tipsHistory": {
"fotw-campaign-upsell": 13,
"new-user-warmup": 6,
"plan-mode-for-complex-tasks": 22,
"memory-command": 16,
"theme-command": 21,
"colorterm-truecolor": 1,
"status-line": 1,
"prompt-queue": 11,
"enter-to-steer-in-relatime": 21,
"todo-list": 21,
"ide-upsell-external-terminal": 19,
"install-github-app": 22,
"install-slack-app": 22,
"drag-and-drop-images": 14,
"double-esc-code-restore": 14,
"continue": 14,
"shift-tab": 15,
"image-paste": 4,
"web-app": 19,
"color-when-multi-clauding": 6,
"custom-agents": 21,
"remote-control": 21,
"voice-mode": 16,
"goal-command-nudge": 16,
"guest-passes": 22,
"feedback-command": 22,
"frontend-design-plugin": 6,
"permissions": 22,
"rename-conversation": 11,
"custom-commands": 11,
"c4e-remote-sessions": 18,
"subagent-fanout-nudge": 18,
"no-flicker": 19
},
"promptQueueUseCount": 44,
"cachedGrowthBookFeatures": {
"tengu_slate_kestrel": true,
"tengu_bridge_repl_v2": true,
"tengu_basalt_meadow": true,
"tengu_sage_compass2": {
"enabled": true
},
"tengu_kairos_loop_dynamic": true,
"tengu_sepia_cormorant": [],
"tengu_amber_heron": false,
"tengu_log_datadog_events": true,
"tengu-fable-off-switch": {
"activated": false
},
"tengu_quiet_slate_wren": false,
"tengu_birch_compass": true,
"tengu_bramble_lintel": 7,
"tengu_malort_pedway": {
"enabled": true,
"pixelValidation": false,
"clipboardPasteMultiline": true,
"screenshotFilter": true,
"mouseAnimation": true,
"hideBeforeAction": true,
"autoTargetDisplay": false,
"coordinateMode": "pixels"
},
"tengu_lilac_loom": {},
"tengu_sub_nomdrep_q7k": true,
"tengu_lantern_spool": false,
"tengu_hawthorn_steeple": false,
"tengu_version_config": {
"minVersion": "1.0.24"
},
"tengu_auto_notice_once": true,
"tengu_sparrow_ledger": false,
"tengu_loggia_carousel": false,
"tengu_ccr_bridge": true,
"tengu_basalt_sundial": false,
"tengu_mcp_stateless_skip_init": true,
"tengu_lapis_anchor": "off",
"tengu_sage_compass": {},
"tengu_kairos_cron": true,
"tengu_kairos_loop_prompt": true,
"tengu_jade_anvil_4": false,
"tengu_skills_dashboard_enabled": false,
"tengu_sedge_lantern_holdback": false,
"tengu_dunwich_bell": false,
"tengu_desktop_upsell": {
"enable_shortcut_tip": true,
"enable_startup_dialog": false
},
"tengu_code_diff_cli": true,
"tengu_anchor_tide": true,
"tengu_garnet_finch": false,
"tengu_cobalt_heron": true,
"tengu_ccr_v2_send_events_cli": true,
"tengu_onyx_plover": {
"enabled": false,
"minHours": 24,
"minSessions": 3,
"remoteEnabled": false
},
"tengu_react_vulnerability_warning": false,
"tengu_prompt_cache_1h_config": {
"allowlist": [
"repl_main_thread*",
"sdk",
"auto_mode",
"rolling_compact",
"memdir_relevance",
"agent_classifier",
"prompt_suggestion",
"away_summary",
"extract_memories",
"compact"
]
},
"tengu_timber_lark": "copy_a",
"tengu_ladder_mq7": false,
"tengu_birthday_hat": false,
"tengu_prompt_cache_diagnostics": true,
"tengu_worktree_mode": true,
"tengu_willow_refresh_ttl_hours": 0,
"tengu_pewter_kestrel": {
"global": 50000,
"Bash": 30000,
"PowerShell": 30000,
"Grep": 20000,
"Snip": 1000,
"StrReplaceBasedEditTool": 30000,
"BashSearchTool": 20000
},
"tengu_slate_finch": true,
"tengu_workflows_enabled": true,
"tengu_permission_friction": true,
"tengu_marble_lark": false,
"tengu_copper_fox": false,
"tengu_bridge_repl_v2_config": {
"init_retry_max_attempts": 3,
"init_retry_base_delay_ms": 500,
"init_retry_jitter_fraction": 0.25,
"init_retry_max_delay_ms": 4000,
"http_timeout_ms": 10000,
"uuid_dedup_buffer_size": 2000,
"heartbeat_interval_ms": 20000,
"heartbeat_jitter_fraction": 0.1,
"token_refresh_buffer_ms": 600000,
"teardown_archive_timeout_ms": 1500,
"connect_timeout_ms": 15000,
"min_version": "2.1.70",
"should_show_app_upgrade_message": false
},
"tengu_marble_whisper": true,
"tengu_maple_sundial": false,
"tengu_velvet_cascade": {},
"tengu_passport_quail": false,
"tengu_ember_latch": true,
"tengu_vscode_onboarding": false,
"tengu_fennel_kite_model": "",
"tengu_nimble_amber_prose": false,
"tengu_bridge_poll_interval_ms": 0,
"tengu_cobalt_wren": false,
"tengu_harbor_permissions": true,
"tengu_orchid_trellis": false,
"tengu_ccr_bridge_multi_session": true,
"tengu_bad_survey_transcript_ask_config": {
"probability": 1
},
"tengu_good_survey_transcript_ask_config": {
"probability": 0.5
},
"tengu_amber_sentinel": true,
"tengu_crimson_vector": false,
"tengu_drift_lantern": false,
"tengu_kestrel_arch": "OFF",
"tengu_read_dedup_killswitch": false,
"tengu_saffron_lattice": {
"enabled": false,
"planLimitsEndDate": "2026-06-22T10:00:00Z",
"hideRateLimitsDescription": true
},
"tengu_cloth_snorkel": false,
"tengu_system_prompt_global_cache": true,
"tengu_slate_moth": true,
"tengu_bridge_poll_interval_config": {
"poll_interval_ms_not_at_capacity": 2000,
"poll_interval_ms_at_capacity": 600000,
"heartbeat_interval_ms": 0,
"multisession_poll_interval_ms_not_at_capacity": 5000,
"multisession_poll_interval_ms_at_capacity": 60000,
"multisession_poll_interval_ms_partial_capacity": 5000,
"non_exclusive_heartbeat_interval_ms": 180000,
"session_keepalive_interval_ms": 0,
"session_keepalive_interval_v2_ms": 0
},
"tengu_gouda_loop": true,
"tengu_otk_slot_v1": false,
"tengu_pewter_lark": "off",
"tengu_walnut_prism": false,
"tengu_immediate_model_command": false,
"tengu_pewter_summit": true,
"tengu_fg_left_arrow_agents": true,
"tengu_willow_sentinel_ttl_hours": 1,
"tengu_pewter_lantern": false,
"tengu_desktop_upsell_v2": {
"enabled": false
},
"tengu_vellum_siding": false,
"tengu_vscode_feedback_survey": true,
"tengu_mcp_singleton_unwrap": true,
"tengu_coral_fern": false,
"tengu_trace_lantern": false,
"tengu_review_bughunter_config": {
"fleet_size": 5,
"max_duration_minutes": 10,
"agent_timeout_seconds": 600,
"total_wallclock_minutes": 22,
"model": "claude-opus-4-7",
"cost_note": "$5-$25",
"duration_note": "~5-10 min",
"enabled": true
},
"tengu_basalt_spur": false,
"tengu_crystal_beam": {
"budgetTokens": 0
},
"tengu_hawthorn_window": 200000,
"tengu_flint_harbor_share": false,
"tengu_bridge_attestation_enforce": false,
"tengu_compass_dial": true,
"tengu_moss_anchor": false,
"tengu_willow_census_ttl_hours": 24,
"tengu_compact_cache_prefix": true,
"tengu_cedar_hollow_7m": {},
"tengu_prompt_suggestion": true,
"tengu_crimson_echo": {},
"tengu_cork_m4q": true,
"tengu_classifier_summary_llm_emit": true,
"tengu_tide_elm": "off",
"tengu_ccr_bundle_seed_enabled": true,
"tengu_copper_wren": false,
"tengu_ember_trail": "0",
"tengu_gha_plugin_code_review": false,
"tengu_keybinding_customization_release": true,
"tengu_kairos_cron_durable": false,
"tengu_canary": {},
"tengu_mocha_barista": true,
"tengu_negative_interaction_transcript_ask_config": {
"probability": 0
},
"tengu_steady_lantern": false,
"tengu_malformed_tool_use_clean_retry": false,
"tengu_agent_list_attach": false,
"tengu_ultraplan_timeout_seconds": 5400,
"tengu_hazel_osprey_floor": 75000,
"tengu_brick_follow": false,
"tengu_slate_ribbon": true,
"tengu_slate_siskin": {
"enabled": false,
"timeoutMs": 8000,
"throttleMs": 30000,
"summaryLineThreshold": 5
},
"tengu_amber_rokovoko": 0.2,
"tengu_penguin_mode_promo": {
"discountPercent": 0,
"endDate": "Feb 16"
},
"tengu_slate_harrier": "off",
"tengu_lapis_thicket": false,
"tengu_harbor_willow": false,
"tengu_amber_anchor": false,
"tengu_tussock_oriole": false,
"tengu_tern_alloy": "copy_a",
"tengu_fgts": true,
"tengu_vellum_lantern": false,
"tengu_saffron_anchor": true,
"tengu_miraculo_the_bard": false,
"tengu_red_coaster": false,
"tengu_cobalt_compass": true,
"tengu_plum_vx3": true,
"tengu_mcp_subagent_prompt": true,
"tengu_mcp_local_oauth_blocked_hosts": {
"hosts": [
"microsoft365.mcp.claude.com",
"gmail.mcp.claude.com",
"gcal.mcp.claude.com"
]
},
"tengu_byte_stream_idle_timeout_ms": 180000,
"tengu_umber_petrel": false,
"tengu_prism_ledger": false,
"tengu_ccr_bundle_max_bytes": 104857600,
"tengu_amber_sextant": true,
"tengu_pewter_ledger": "OFF",
"tengu_amber_flint": true,
"tengu_disable_bypass_permissions_mode": false,
"tengu_walrus_canteen": false,
"tengu_ashen_kelp": true,
"tengu_plugin_official_mkt_git_fallback": true,
"tengu_max_version_config": {},
"tengu_cobalt_lantern": true,
"tengu_ultraplan_prompt_identifier": "visual_plan",
"tengu_swann_brevity": "focused",
"tengu_hazel_osprey": false,
"tengu_slate_meadow": true,
"tengu_amber_redwood2": "",
"tengu_frond_boric": {},
"tengu_slate_thimble": false,
"tengu_slate_nexus": true,
"tengu_chert_bezel": true,
"tengu_streaming_tool_execution2": true,
"tengu_event_watchdog_default_on": false,
"tengu_auto_mode_config": {
"enabled": "enabled",
"twoStageClassifier": true
},
"tengu_grey_step2": {
"enabled": true,
"dialogTitle": "We recommend medium effort for Opus",
"dialogDescription": "Effort determines how long Claude thinks for when completing your task. We recommend medium effort for most tasks to balance speed and intelligence and maximize rate limits. Use ultrathink to trigger high effort when needed."
},
"tengu_dune_wren": false,
"tengu_cedar_lantern": true,
"tengu_velvet_moth": 0.2,
"tengu_harbor_ledger": [
{
"marketplace": "claude-plugins-official",
"plugin": "discord"
},
{
"marketplace": "claude-plugins-official",
"plugin": "telegram"
},
{
"marketplace": "claude-plugins-official",
"plugin": "fakechat"
},
{
"marketplace": "claude-plugins-official",
"plugin": "imessage"
}
],
"tengu_harbor": true,
"tengu_amber_lynx": false,
"tengu_doorbell_agave": false,
"tengu_maple_tide": false,
"tengu_fennel_kite": false,
"tengu_collage_kaleidoscope": true,
"tengu_file_write_optimization": true,
"tengu_startup_notice": "",
"tengu_mcp_retry_failed_remote": false,
"tengu_session_memory": false,
"tengu_flint_harbor_prompt": {
"prompt": "You are helping a power user generate an onboarding guide for teammates who are new to Claude Code. The guide will live in the team's onboarding docs and can be pasted into Claude for an interactive walkthrough.\n\nYou're co-authoring this with them — collaborative and helpful, like a teammate who's done this before and is happy to share.\n\n## Usage data (last {{WINDOW_DAYS}} days)\n\nThis was scanned from the guide creator's local Claude Code transcripts:\n\n```json\n{{USAGE_DATA}}\n```\n\n## Your task\n\nBefore anything else — including before thinking through the classification — output exactly this line as your first visible text:\n\n> Looking at how you've used Claude over the last {{WINDOW_DAYS}} days to put together an onboarding guide for teammates new to Claude Code.\n\nThis must come before any extended thinking about session descriptors. The guide creator is staring at a blank screen until you do. Classification is step 2, not step 1.\n\nGenerate the guide immediately, then ask for revisions. Don't wait for answers first — it's easier for the guide creator to edit a concrete draft than answer abstract questions.\n\n1. **Output the acknowledgment line above.** No thinking, no classification, no tool calls before this. One line, then move on.\n\n2. **Derive the work-type breakdown.** Read the `sessionDescriptors` array — each entry describes one session via its title, any linked code reviews (`prNumbers`), and first user message. Classify each session into one of these task types:\n\n - **build_feature** — new functionality, scripts, tools, config/CI/env setup\n - **debug_fix** — investigating and fixing bugs\n - **improve_quality** — refactoring, tests, cleanup, code review\n - **analyze_data** — queries, metrics, number crunching\n - **plan_design** — architecture, approach, strategy, understanding unfamiliar code, design review\n - **prototype** — spikes, POCs, throwaway exploration\n - **write_docs** — PRDs, RFCs, READMEs, design docs, copy/doc review\n\n Categories describe the *type of task*, not the project or domain — a teammate on any project should recognize them. Review sessions belong with whatever's being reviewed: code review is improve_quality, doc review is write_docs, design review is plan_design. Most sessions fit the list; only invent a new category if it's genuinely a different type of task. Pick the top 3-5 with rough percentages. First messages alone are usually enough; titles and code-review links are enrichment. If first messages are uninformative, use tool and MCP counts as a weak hint. If there are ~0 sessions, leave the breakdown as a TODO.\n\n In the rendered guide, display categories with spaces and title case (e.g. \"Build Feature\" not \"build_feature\").\n\n3. **Gather the remaining pieces.** For repos, start with `currentRepo` and check the workspace for sibling repo directories. For MCP server setup, use each entry's `name` (and `urlOrigin` where present) to infer what the server does and how a teammate would get access. Leave the Team Tips and Get Started sections as TODO placeholders — you'll ask for these in Review and fill them in after.\n\n4. **Write the guide to `ONBOARDING.md`** following this template:\n\n```\n{{GUIDE_TEMPLATE}}\n```\n\n Fill in real numbers from the usage data (not placeholders). Use `generatedBy` for the name; if it's missing, omit the name. Ascii bar charts: `█` for filled, `░` for empty, 20 chars wide. Keep the HTML comment instruction at the bottom exactly as shown.\n\n5. **Render the guide in a code block, then close out the first turn.** You're co-authoring this guide with the guide creator — frame the follow-up as collaboration, not corrections.\n\n After the code block, add a `---` horizontal rule and a `**Review**` heading so the guide is visually separated from your questions. Under the heading, number these three questions:\n\n 1. \"I went with '[X]' for the team name — let me know if that sounds right.\" (or if you couldn't tell: \"What's the team name? I'll add it in.\")\n 2. Is there a starter task for someone new to Claude Code? (ticket or doc link — optional)\n 3. Any team tips you'd tell a new teammate that aren't already in CLAUDE.md?\n\n After they answer, update `ONBOARDING.md` with their team name, tips, and starter task. Then close with this exact line (not numbered, not paraphrased):\n\n Saved to `ONBOARDING.md`. Drop it in your team docs and channels — when a new teammate pastes it into Claude Code, they get a guided onboarding tour from there.\n\n Apply any edits they come back with to the file.",
"guideTemplate": "# Welcome to [Team Name]\n\n## How We Use Claude\n\nBased on [name]'s usage over the last [N] days:\n\nWork Type Breakdown:\n [Category 1] [ascii bar] [N]%\n [Category 2] [ascii bar] [N]%\n [Category 3] [ascii bar] [N]%\n ...\n\nTop Skills & Commands:\n [/command] [ascii bar] [N]x/month\n ...\n\nTop MCP Servers:\n [Server] [ascii bar] [N] calls\n ...\n\n## Your Setup Checklist\n\n### Codebases\n- [ ] [repo-name] — [repo url]\n...\n\n### MCP Servers to Activate\n- [ ] [Server] — [what it's for]. [How to get access]\n...\n\n### Skills to Know About\n- [/command] — [what it does, when the team uses it]\n...\n\n## Team Tips\n\n_TODO_\n\n## Get Started\n\n_TODO_\n\n<!-- INSTRUCTION FOR CLAUDE: A new teammate just pasted this guide for how the\nteam uses Claude Code. You're their onboarding buddy — warm, conversational,\nnot lecture-y.\n\nOpen with a warm welcome — include the team name from the title. Then: \"Your\nteammate uses Claude Code for [list all the work types]. Let's get you started.\"\n\nCheck what's already in place against everything under Setup Checklist\n(including skills), using markdown checkboxes — [x] done, [ ] not yet. Lead\nwith what they already have. One sentence per item, all in one message.\n\nTell them you'll help with setup, cover the actionable team tips, then the\nstarter task (if there is one). Offer to start with the first unchecked item,\nget their go-ahead, then work through the rest one by one.\n\nAfter setup, walk them through the remaining sections — offer to help where you\ncan (e.g. link to channels), and just surface the purely informational bits.\n\nDon't invent sections or summaries that aren't in the guide. The stats are the\nguide creator's personal usage data — don't extrapolate them into a \"team\nworkflow\" narrative. -->",
"windowDays": 30
},
"tengu_slim_subagent_claudemd": true,
"tengu_tangerine_ladder_boost": true,
"tengu_chair_sermon": false,
"tengu_gypsum_kite": true,
"tengu_quartz_heron": false,
"tengu_xterm_atlas_reset": true,
"tengu-model-error-overrides": {
"claude-fable-5": {
"block": "Claude Fable 5 is currently unavailable. Learn more: https://www.anthropic.com/news/fable-mythos-access"
}
},
"tengu_orchid_mantis_v2": true,
"tengu-off-switch": {
"activated": false
},
"tengu_feedback_survey_config": {
"minTimeBeforeFeedbackMs": 600000,
"minTimeBetweenFeedbackMs": 43200000,
"minTimeBetweenGlobalFeedbackMs": 43200000,
"minUserTurnsBeforeFeedback": 5,
"minUserTurnsBetweenFeedback": 25,
"hideThanksAfterMs": 3000,
"onForModels": [
"*"
],
"probability": 0.05
},
"tengu_cork_lantern": false,
"tengu_mint_lanes": false,
"tengu_bridge_attestation_enforce_config": {
"accept_level": "VERIFIED_BY_GATE",
"accept_statuses": []
},
"tengu_marble_sandcastle": false,
"tengu_bg_attach_stall_ms": 5000,
"tengu_workout2": true,
"tengu_orford_ness": false,
"tengu_porch_bell_9f": "",
"tengu_auto_mode_default_on": false,
"tengu_birch_kettle": false,
"tengu_classifier_summary_heuristic_emit": true,
"tengu_cobalt_thicket": false,
"tengu_destructive_command_warning": false,
"tengu_cinder_plover": "",
"tengu_cedar_halo": false,
"tengu_sotto_voce": true,
"tengu_sepia_moth": false,
"tengu_cedar_sundial": false,
"tengu_penguins_enabled": true,
"tengu_quiet_basalt_echo": false,
"tengu_ochre_hollow": true,
"tengu_coral_beacon": true,
"tengu_copper_thistle": false,
"tengu_1p_event_batch_config": {
"scheduledDelayMillis": 10000,
"maxExportBatchSize": 400,
"maxQueueSize": 8192,
"path": "/api/event_logging/v2/batch"
},
"tengu_amber_wren": {
"targetedRangeNudge": true,
"maxTokens": 25000
},
"tengu_amber_prism": true,
"tengu_cobalt_plinth": false,
"tengu_silent_harbor": false,
"tengu_chomp_inflection": true,
"tengu_mcp_elicitation": true,
"tengu_sm_config": {
"minimumMessageTokensToInit": 150000,
"minimumTokensBetweenUpdate": 40000,
"toolCallsBetweenUpdates": 10
},
"tengu_bridge_min_version": {
"minVersion": "2.1.70"
},
"tengu_kairos_input_needed_push": true,
"tengu_quiet_harbor": false,
"tengu_slate_wren": false,
"tengu_tool_search_unsupported_models": [
"claude-3-5-haiku",
"claude-3-haiku"
],
"tengu_native_cursor": true,
"tengu_orchid_mantis": false,
"tengu_amber_lark": true,
"tengu_shale_finch": true,
"tengu_cedar_plume": false,
"tengu_kairos_push_notifications": true,
"tengu_marble_whisper2": true,
"tengu_lichen_compass": false,
"tengu_c4w_usage_limit_notifications_enabled": true,
"tengu_scarf_coffee": false,
"tengu_copper_bridge": true,
"tengu_tool_pear": false,
"tengu_claudeai_mcp_connectors": true,
"tengu_ccr_post_turn_summary": false,
"tengu_sedge_lantern": true,
"tengu_feature_template": false,
"tengu_harbor_prism": true,
"tengu_cedar_inlet": "step",
"tengu_flax_grouse": false,
"tengu_event_sampling_config": {},
"tengu_herring_clock": false,
"tengu_quartz_vireo": "",
"tengu_team_discovery": false,
"tengu_gleaming_fair": true,
"tengu_marble_anvil": true,
"tengu_classifier_disabled_surfaces": "",
"tengu_pewter_brook": false,
"tengu_vscode_review_upsell": false,
"claude_code_skills_dashboard_enabled_cli": false,
"tengu_post_compact_survey": false,
"tengu_reactive_compact_remote": false,
"tengu_idle_amber_finch": false,
"tengu_noreread_q7m_velvet": false,
"tengu_ultraplan_config": {
"enabled": true
},
"tengu_scratch": false,
"tengu_alder_compass": false,
"tengu_olive_hinge": "",
"tengu_shining_fractals": false,
"tengu_maple_pier": false,
"tengu_sessions_elevated_auth_enforcement": true,
"tengu_turtle_carbon": true,
"tengu_billiard_aviary": false,
"tengu_cinder_almanac": true,
"tengu_osprey_lantern": false,
"tengu-top-of-feed-tip": {
"tip": "Claude Fable 5 is currently unavailable. Please use Opus 4.8 or another available model. Learn more: https://www.anthropic.com/news/fable-mythos-access",
"color": "warning"
},
"tengu_cobalt_raccoon": true,
"tengu_loud_sugary_rock": false,
"tengu_willow_mode": "hint_v2",
"tengu_blue_coaster": false,
"tengu_snippet_save": false,
"tengu_amber_lattice": {
"plugins": [
"security-guidance",
"code-review",
"commit-commands",
"code-simplifier",
"hookify",
"feature-dev",
"frontend-design",
"pr-review-toolkit",
"skill-creator",
"plugin-dev",
"agent-sdk-dev",
"mcp-server-dev",
"claude-code-setup",
"claude-md-management",
"playground",
"ralph-loop",
"explanatory-output-style",
"learning-output-style",
"clangd-lsp",
"csharp-lsp",
"gopls-lsp",
"jdtls-lsp",
"kotlin-lsp",
"lua-lsp",
"php-lsp",
"pyright-lsp",
"ruby-lsp",
"rust-analyzer-lsp",
"swift-lsp",
"typescript-lsp"
]
},
"tengu_slate_harbor_experiment": false,
"tengu_velvet_ibis": {},
"tengu_bridge_requires_action_details": true,
"tengu_lapis_finch": true,
"tengu_satin_quoll": {},
"tengu_moth_copse": false,
"tengu_silk_hinge": false,
"tengu_surreal_dali": true,
"tengu_cobalt_ridge": true,
"tengu_flint_harbor": false,
"tengu_plank_river_frost": "user_intent",
"tengu_velvet_mallet_haiku": false,
"tengu_velvet_mallet": false,
"tengu_velvet_mallet_haiku_4_5": false,
"tengu_velvet_hammer_falcon": false,
"tengu_loud_sugary_rock2": false,
"tengu_velvet_hammer_sonnet_4_5": false,
"tengu_velvet_hammer_sonnet": false,
"tengu_tab_read_sep": false,
"tengu_quill_harbor": "acceptEdits",
"tengu_velvet_hammer": false,
"tengu_velvet_hammer_opus": false,
"tengu_c4e_slash_upsell": true,
"tengu_velvet_hammer_haiku_4_5": false,
"tengu_feature_claudified_template": false,
"tengu_slate_quill": true,
"tengu_ax_screen_reader": false,
"tengu_windows_credman": false,
"tengu_basalt_tern": false,
"tengu_velvet_mallet_opus": false,
"tengu_velvet_hammer_haiku": false,
"tengu_velvet_static": true,
"tengu_velvet_mallet_sonnet": false,
"tengu_soft_slate_nudge": "baseline",
"tengu_lantern_hearth": "off",
"tengu_velvet_mallet_falcon": false,
"tengu_velvet_mallet_sonnet_4_5": false
},
"firstStartTime": "2026-06-05T19:39:28.542Z",
"opusProMigrationComplete": true,
"sonnet1m45MigrationComplete": true,
"seenNotifications": {},
"migrationVersion": 13,
"userID": "9d89994d486a4884b8cf33372d8a4cd61ebf7d34009e9d3cbce9db24e2e971a4",
"changelogLastFetched": 1781361371930,
"autoUpdatesProtectedForNative": true,
"claudeCodeFirstTokenDate": "2026-04-11T19:03:48.223040Z",
"hasCompletedOnboarding": true,
"lastOnboardingVersion": "2.1.165",
"groveConfigCache": {
"09792e21-2287-4348-b4d4-34cddbbfabc5": {
"grove_enabled": true,
"timestamp": 1781406640065
}
},
"cachedExperimentFeatures": [
"tengu_amber_prism",
"tengu_basalt_spur",
"tengu_cedar_inlet",
"tengu_coral_beacon",
"tengu_flint_harbor",
"tengu_mcp_subagent_prompt",
"tengu_ochre_hollow",
"tengu_orchid_mantis_v2",
"tengu_plank_river_frost",
"tengu_read_dedup_killswitch"
],
"cachedGrowthBookFeaturesAt": 1781406639973,
"lastReleaseNotesSeen": "2.1.177",
"projects": {
"/root": {
"allowedTools": [],
"mcpContextUris": [],
"mcpServers": {},
"enabledMcpjsonServers": [],
"disabledMcpjsonServers": [],
"hasTrustDialogAccepted": false,
"projectOnboardingSeenCount": 3,
"hasClaudeMdExternalIncludesApproved": false,
"hasClaudeMdExternalIncludesWarningShown": false,
"exampleFiles": [],
"lastGracefulShutdown": false,
"lastVersionBase": "2.1.177",
"lastCost": 1.0676417999999999,
"lastAPIDuration": 276732,
"lastAPIDurationWithoutRetries": 276675,
"lastToolDuration": 9607,
"lastDuration": 2130140,
"lastLinesAdded": 29,
"lastLinesRemoved": 15,
"lastTotalInputTokens": 4397,
"lastTotalOutputTokens": 16093,
"lastTotalCacheCreationInputTokens": 53595,
"lastTotalCacheReadInputTokens": 1642666,
"lastTotalWebSearchRequests": 0,
"lastFpsAverage": 1.82,
"lastFpsLow1Pct": 313.42,
"lastModelUsage": {
"claude-haiku-4-5-20251001": {
"inputTokens": 572,
"outputTokens": 17,
"cacheReadInputTokens": 0,
"cacheCreationInputTokens": 0,
"webSearchRequests": 0,
"costUSD": 0.000657
},
"claude-sonnet-4-6": {
"inputTokens": 3825,
"outputTokens": 16076,
"cacheReadInputTokens": 1642666,
"cacheCreationInputTokens": 53595,
"webSearchRequests": 0,
"costUSD": 1.0669847999999997
}
},
"lastSessionId": "96cf6b2d-d6a0-405b-81e5-95c657e1922a",
"lastSessionMetrics": {
"frame_duration_ms_count": 16776,
"frame_duration_ms_min": 0.11423300000024028,
"frame_duration_ms_max": 21.985366000095382,
"frame_duration_ms_avg": 0.7730811968292047,
"frame_duration_ms_p50": 0.5600509999203496,
"frame_duration_ms_p95": 1.786581499991007,
"frame_duration_ms_p99": 4.282493569953367,
"pre_tool_hook_duration_ms_count": 108,
"pre_tool_hook_duration_ms_min": 0,
"pre_tool_hook_duration_ms_max": 15,
"pre_tool_hook_duration_ms_avg": 0.24074074074074073,
"pre_tool_hook_duration_ms_p50": 0,
"pre_tool_hook_duration_ms_p95": 1,
"pre_tool_hook_duration_ms_p99": 4.789999999999978,
"hook_duration_ms_count": 40,
"hook_duration_ms_min": 0,
"hook_duration_ms_max": 8,
"hook_duration_ms_avg": 0.35,
"hook_duration_ms_p50": 0,
"hook_duration_ms_p95": 1,
"hook_duration_ms_p99": 5.269999999999996
},
"hasCompletedProjectOnboarding": true
}
},
"routineFiredWatermark": "2026-06-05T19:47:09.178Z",
"penguinModeOrgEnabled": true,
"closedIssuesLastChecked": 1781406639965,
"passesEligibilityCache": {
"4bb43199-0efc-4d5c-b552-79865cb0361b": {
"eligible": true,
"referral_code_details": {
"code": "BeGGjphr1g",
"campaign": "claude_code_guest_pass_a47c",
"referral_link": "https://claude.ai/referral/BeGGjphr1g"
},
"referrer_reward": {
"amount_minor_units": 1000,
"currency": "USD"
},
"remaining_passes": 3,
"limit": 3,
"share_link": "https://claude.ai/referral/BeGGjphr1g",
"terms_url": "https://support.claude.com/en/articles/12875061-claude-code-guest-passes",
"timestamp": 1781406640514
}
},
"cachedExtraUsageDisabledReason": null,
"passesUpsellSeenCount": 3,
"hasVisitedPasses": false,
"passesLastSeenRemaining": 3,
"officialMarketplaceAutoInstallAttempted": true,
"officialMarketplaceAutoInstalled": true,
"tipLifetimeShownCounts": {
"fotw-campaign-upsell": 6,
"new-user-warmup": 2,
"plan-mode-for-complex-tasks": 5,
"memory-command": 2,
"theme-command": 2,
"colorterm-truecolor": 1,
"status-line": 1,
"prompt-queue": 3,
"enter-to-steer-in-relatime": 2,
"todo-list": 2,
"ide-upsell-external-terminal": 5,
"install-github-app": 3,
"install-slack-app": 3,
"drag-and-drop-images": 2,
"double-esc-code-restore": 2,
"continue": 2,
"shift-tab": 2,
"image-paste": 1,
"web-app": 2,
"color-when-multi-clauding": 1,
"custom-agents": 2,
"remote-control": 2,
"voice-mode": 2,
"goal-command-nudge": 4,
"guest-passes": 6,
"feedback-command": 2,
"frontend-design-plugin": 1,
"permissions": 2,
"rename-conversation": 1,
"custom-commands": 1,
"c4e-remote-sessions": 1,
"subagent-fanout-nudge": 1,
"no-flicker": 1
},
"feedbackSurveyState": {
"lastShownTime": 1781411703066
},
"hasUsedBackslashReturn": true,
"agentLastUsed": {
"bg": 1780696781055
},
"remoteControlUpsellSeenCount": 3,
"fullscreenUpsellSeenCount": 3,
"lastShownEmergencyTip": "Claude Fable 5 is currently unavailable. Please use Opus 4.8 or another available model. Learn more: https://www.anthropic.com/news/fable-mythos-access",
"oauthAccount": {
"accountUuid": "09792e21-2287-4348-b4d4-34cddbbfabc5",
"emailAddress": "gmer4lfe@gmail.com",
"organizationUuid": "4bb43199-0efc-4d5c-b552-79865cb0361b",
"hasExtraUsageEnabled": true,
"billingType": "stripe_subscription",
"accountCreatedAt": "2026-04-03T21:52:35.642439Z",
"subscriptionCreatedAt": "2026-04-11T13:14:49.905923Z",
"ccOnboardingFlags": {},
"claudeCodeTrialEndsAt": null,
"claudeCodeTrialDurationDays": null,
"seatTier": null,
"displayName": "Gmer4Lfe",
"organizationRole": "admin",
"workspaceRole": null,
"organizationName": "gmer4lfe@gmail.com's Organization",
"organizationType": "claude_pro",
"organizationRateLimitTier": "default_claude_ai",
"userRateLimitTier": null
},
"clientDataCache": {
"cedar_lagoon": {
"claude-fable": true,
"claude-mythos": true
},
"pewter_owl_tool": true,
"pewter_owl_model": "claude-fable"
},
"additionalModelOptionsCache": [
{
"value": "claude-fable-5[1m]",
"label": "Fable (disabled)",
"description": "Claude Fable 5 is currently unavailable. Learn more: https://www.anthropic.com/news/fable-mythos-access",
"disabled": true
}
],
"additionalModelCostsCache": {}
}
@@ -1,785 +0,0 @@
{
"numStartups": 22,
"installMethod": "native",
"autoUpdates": false,
"hasSeenTasksHint": true,
"tipsHistory": {
"fotw-campaign-upsell": 13,
"new-user-warmup": 6,
"plan-mode-for-complex-tasks": 22,
"memory-command": 16,
"theme-command": 21,
"colorterm-truecolor": 1,
"status-line": 1,
"prompt-queue": 11,
"enter-to-steer-in-relatime": 21,
"todo-list": 21,
"ide-upsell-external-terminal": 19,
"install-github-app": 22,
"install-slack-app": 22,
"drag-and-drop-images": 14,
"double-esc-code-restore": 14,
"continue": 14,
"shift-tab": 15,
"image-paste": 4,
"web-app": 19,
"color-when-multi-clauding": 6,
"custom-agents": 21,
"remote-control": 21,
"voice-mode": 16,
"goal-command-nudge": 16,
"guest-passes": 22,
"feedback-command": 22,
"frontend-design-plugin": 6,
"permissions": 22,
"rename-conversation": 11,
"custom-commands": 11,
"c4e-remote-sessions": 18,
"subagent-fanout-nudge": 18,
"no-flicker": 19
},
"promptQueueUseCount": 44,
"cachedGrowthBookFeatures": {
"tengu_slate_kestrel": true,
"tengu_bridge_repl_v2": true,
"tengu_basalt_meadow": true,
"tengu_sage_compass2": {
"enabled": true
},
"tengu_kairos_loop_dynamic": true,
"tengu_sepia_cormorant": [],
"tengu_amber_heron": false,
"tengu_log_datadog_events": true,
"tengu-fable-off-switch": {
"activated": false
},
"tengu_quiet_slate_wren": false,
"tengu_birch_compass": true,
"tengu_bramble_lintel": 7,
"tengu_malort_pedway": {
"enabled": true,
"pixelValidation": false,
"clipboardPasteMultiline": true,
"screenshotFilter": true,
"mouseAnimation": true,
"hideBeforeAction": true,
"autoTargetDisplay": false,
"coordinateMode": "pixels"
},
"tengu_lilac_loom": {},
"tengu_sub_nomdrep_q7k": true,
"tengu_lantern_spool": false,
"tengu_hawthorn_steeple": false,
"tengu_version_config": {
"minVersion": "1.0.24"
},
"tengu_auto_notice_once": true,
"tengu_sparrow_ledger": false,
"tengu_loggia_carousel": false,
"tengu_ccr_bridge": true,
"tengu_basalt_sundial": false,
"tengu_mcp_stateless_skip_init": true,
"tengu_lapis_anchor": "off",
"tengu_sage_compass": {},
"tengu_kairos_cron": true,
"tengu_kairos_loop_prompt": true,
"tengu_jade_anvil_4": false,
"tengu_skills_dashboard_enabled": false,
"tengu_sedge_lantern_holdback": false,
"tengu_dunwich_bell": false,
"tengu_desktop_upsell": {
"enable_shortcut_tip": true,
"enable_startup_dialog": false
},
"tengu_code_diff_cli": true,
"tengu_anchor_tide": true,
"tengu_garnet_finch": false,
"tengu_cobalt_heron": true,
"tengu_ccr_v2_send_events_cli": true,
"tengu_onyx_plover": {
"enabled": false,
"minHours": 24,
"minSessions": 3,
"remoteEnabled": false
},
"tengu_react_vulnerability_warning": false,
"tengu_prompt_cache_1h_config": {
"allowlist": [
"repl_main_thread*",
"sdk",
"auto_mode",
"rolling_compact",
"memdir_relevance",
"agent_classifier",
"prompt_suggestion",
"away_summary",
"extract_memories",
"compact"
]
},
"tengu_timber_lark": "copy_a",
"tengu_ladder_mq7": false,
"tengu_birthday_hat": false,
"tengu_prompt_cache_diagnostics": true,
"tengu_worktree_mode": true,
"tengu_willow_refresh_ttl_hours": 0,
"tengu_pewter_kestrel": {
"global": 50000,
"Bash": 30000,
"PowerShell": 30000,
"Grep": 20000,
"Snip": 1000,
"StrReplaceBasedEditTool": 30000,
"BashSearchTool": 20000
},
"tengu_slate_finch": true,
"tengu_workflows_enabled": true,
"tengu_permission_friction": true,
"tengu_marble_lark": false,
"tengu_copper_fox": false,
"tengu_bridge_repl_v2_config": {
"init_retry_max_attempts": 3,
"init_retry_base_delay_ms": 500,
"init_retry_jitter_fraction": 0.25,
"init_retry_max_delay_ms": 4000,
"http_timeout_ms": 10000,
"uuid_dedup_buffer_size": 2000,
"heartbeat_interval_ms": 20000,
"heartbeat_jitter_fraction": 0.1,
"token_refresh_buffer_ms": 600000,
"teardown_archive_timeout_ms": 1500,
"connect_timeout_ms": 15000,
"min_version": "2.1.70",
"should_show_app_upgrade_message": false
},
"tengu_marble_whisper": true,
"tengu_maple_sundial": false,
"tengu_velvet_cascade": {},
"tengu_passport_quail": false,
"tengu_ember_latch": true,
"tengu_vscode_onboarding": false,
"tengu_fennel_kite_model": "",
"tengu_nimble_amber_prose": false,
"tengu_bridge_poll_interval_ms": 0,
"tengu_cobalt_wren": false,
"tengu_harbor_permissions": true,
"tengu_orchid_trellis": false,
"tengu_ccr_bridge_multi_session": true,
"tengu_bad_survey_transcript_ask_config": {
"probability": 1
},
"tengu_good_survey_transcript_ask_config": {
"probability": 0.5
},
"tengu_amber_sentinel": true,
"tengu_crimson_vector": false,
"tengu_drift_lantern": false,
"tengu_kestrel_arch": "OFF",
"tengu_read_dedup_killswitch": false,
"tengu_saffron_lattice": {
"enabled": false,
"planLimitsEndDate": "2026-06-22T10:00:00Z",
"hideRateLimitsDescription": true
},
"tengu_cloth_snorkel": false,
"tengu_system_prompt_global_cache": true,
"tengu_slate_moth": true,
"tengu_bridge_poll_interval_config": {
"poll_interval_ms_not_at_capacity": 2000,
"poll_interval_ms_at_capacity": 600000,
"heartbeat_interval_ms": 0,
"multisession_poll_interval_ms_not_at_capacity": 5000,
"multisession_poll_interval_ms_at_capacity": 60000,
"multisession_poll_interval_ms_partial_capacity": 5000,
"non_exclusive_heartbeat_interval_ms": 180000,
"session_keepalive_interval_ms": 0,
"session_keepalive_interval_v2_ms": 0
},
"tengu_gouda_loop": true,
"tengu_otk_slot_v1": false,
"tengu_pewter_lark": "off",
"tengu_walnut_prism": false,
"tengu_immediate_model_command": false,
"tengu_pewter_summit": true,
"tengu_fg_left_arrow_agents": true,
"tengu_willow_sentinel_ttl_hours": 1,
"tengu_pewter_lantern": false,
"tengu_desktop_upsell_v2": {
"enabled": false
},
"tengu_vellum_siding": false,
"tengu_vscode_feedback_survey": true,
"tengu_mcp_singleton_unwrap": true,
"tengu_coral_fern": false,
"tengu_trace_lantern": false,
"tengu_review_bughunter_config": {
"fleet_size": 5,
"max_duration_minutes": 10,
"agent_timeout_seconds": 600,
"total_wallclock_minutes": 22,
"model": "claude-opus-4-7",
"cost_note": "$5-$25",
"duration_note": "~5-10 min",
"enabled": true
},
"tengu_basalt_spur": false,
"tengu_crystal_beam": {
"budgetTokens": 0
},
"tengu_hawthorn_window": 200000,
"tengu_flint_harbor_share": false,
"tengu_bridge_attestation_enforce": false,
"tengu_compass_dial": true,
"tengu_moss_anchor": false,
"tengu_willow_census_ttl_hours": 24,
"tengu_compact_cache_prefix": true,
"tengu_cedar_hollow_7m": {},
"tengu_prompt_suggestion": true,
"tengu_crimson_echo": {},
"tengu_cork_m4q": true,
"tengu_classifier_summary_llm_emit": true,
"tengu_tide_elm": "off",
"tengu_ccr_bundle_seed_enabled": true,
"tengu_copper_wren": false,
"tengu_ember_trail": "0",
"tengu_gha_plugin_code_review": false,
"tengu_keybinding_customization_release": true,
"tengu_kairos_cron_durable": false,
"tengu_canary": {},
"tengu_mocha_barista": true,
"tengu_negative_interaction_transcript_ask_config": {
"probability": 0
},
"tengu_steady_lantern": false,
"tengu_malformed_tool_use_clean_retry": false,
"tengu_agent_list_attach": false,
"tengu_ultraplan_timeout_seconds": 5400,
"tengu_hazel_osprey_floor": 75000,
"tengu_brick_follow": false,
"tengu_slate_ribbon": true,
"tengu_slate_siskin": {
"enabled": false,
"timeoutMs": 8000,
"throttleMs": 30000,
"summaryLineThreshold": 5
},
"tengu_amber_rokovoko": 0.2,
"tengu_penguin_mode_promo": {
"discountPercent": 0,
"endDate": "Feb 16"
},
"tengu_slate_harrier": "off",
"tengu_lapis_thicket": false,
"tengu_harbor_willow": false,
"tengu_amber_anchor": false,
"tengu_tussock_oriole": false,
"tengu_tern_alloy": "copy_a",
"tengu_fgts": true,
"tengu_vellum_lantern": false,
"tengu_saffron_anchor": true,
"tengu_miraculo_the_bard": false,
"tengu_red_coaster": false,
"tengu_cobalt_compass": true,
"tengu_plum_vx3": true,
"tengu_mcp_subagent_prompt": true,
"tengu_mcp_local_oauth_blocked_hosts": {
"hosts": [
"microsoft365.mcp.claude.com",
"gmail.mcp.claude.com",
"gcal.mcp.claude.com"
]
},
"tengu_byte_stream_idle_timeout_ms": 180000,
"tengu_umber_petrel": false,
"tengu_prism_ledger": false,
"tengu_ccr_bundle_max_bytes": 104857600,
"tengu_amber_sextant": true,
"tengu_pewter_ledger": "OFF",
"tengu_amber_flint": true,
"tengu_disable_bypass_permissions_mode": false,
"tengu_walrus_canteen": false,
"tengu_ashen_kelp": true,
"tengu_plugin_official_mkt_git_fallback": true,
"tengu_max_version_config": {},
"tengu_cobalt_lantern": true,
"tengu_ultraplan_prompt_identifier": "visual_plan",
"tengu_swann_brevity": "focused",
"tengu_hazel_osprey": false,
"tengu_slate_meadow": true,
"tengu_amber_redwood2": "",
"tengu_frond_boric": {},
"tengu_slate_thimble": false,
"tengu_slate_nexus": true,
"tengu_chert_bezel": true,
"tengu_streaming_tool_execution2": true,
"tengu_event_watchdog_default_on": false,
"tengu_auto_mode_config": {
"enabled": "enabled",
"twoStageClassifier": true
},
"tengu_grey_step2": {
"enabled": true,
"dialogTitle": "We recommend medium effort for Opus",
"dialogDescription": "Effort determines how long Claude thinks for when completing your task. We recommend medium effort for most tasks to balance speed and intelligence and maximize rate limits. Use ultrathink to trigger high effort when needed."
},
"tengu_dune_wren": false,
"tengu_cedar_lantern": true,
"tengu_velvet_moth": 0.2,
"tengu_harbor_ledger": [
{
"marketplace": "claude-plugins-official",
"plugin": "discord"
},
{
"marketplace": "claude-plugins-official",
"plugin": "telegram"
},
{
"marketplace": "claude-plugins-official",
"plugin": "fakechat"
},
{
"marketplace": "claude-plugins-official",
"plugin": "imessage"
}
],
"tengu_harbor": true,
"tengu_amber_lynx": false,
"tengu_doorbell_agave": false,
"tengu_maple_tide": false,
"tengu_fennel_kite": false,
"tengu_collage_kaleidoscope": true,
"tengu_file_write_optimization": true,
"tengu_startup_notice": "",
"tengu_mcp_retry_failed_remote": false,
"tengu_session_memory": false,
"tengu_flint_harbor_prompt": {
"prompt": "You are helping a power user generate an onboarding guide for teammates who are new to Claude Code. The guide will live in the team's onboarding docs and can be pasted into Claude for an interactive walkthrough.\n\nYou're co-authoring this with them — collaborative and helpful, like a teammate who's done this before and is happy to share.\n\n## Usage data (last {{WINDOW_DAYS}} days)\n\nThis was scanned from the guide creator's local Claude Code transcripts:\n\n```json\n{{USAGE_DATA}}\n```\n\n## Your task\n\nBefore anything else — including before thinking through the classification — output exactly this line as your first visible text:\n\n> Looking at how you've used Claude over the last {{WINDOW_DAYS}} days to put together an onboarding guide for teammates new to Claude Code.\n\nThis must come before any extended thinking about session descriptors. The guide creator is staring at a blank screen until you do. Classification is step 2, not step 1.\n\nGenerate the guide immediately, then ask for revisions. Don't wait for answers first — it's easier for the guide creator to edit a concrete draft than answer abstract questions.\n\n1. **Output the acknowledgment line above.** No thinking, no classification, no tool calls before this. One line, then move on.\n\n2. **Derive the work-type breakdown.** Read the `sessionDescriptors` array — each entry describes one session via its title, any linked code reviews (`prNumbers`), and first user message. Classify each session into one of these task types:\n\n - **build_feature** — new functionality, scripts, tools, config/CI/env setup\n - **debug_fix** — investigating and fixing bugs\n - **improve_quality** — refactoring, tests, cleanup, code review\n - **analyze_data** — queries, metrics, number crunching\n - **plan_design** — architecture, approach, strategy, understanding unfamiliar code, design review\n - **prototype** — spikes, POCs, throwaway exploration\n - **write_docs** — PRDs, RFCs, READMEs, design docs, copy/doc review\n\n Categories describe the *type of task*, not the project or domain — a teammate on any project should recognize them. Review sessions belong with whatever's being reviewed: code review is improve_quality, doc review is write_docs, design review is plan_design. Most sessions fit the list; only invent a new category if it's genuinely a different type of task. Pick the top 3-5 with rough percentages. First messages alone are usually enough; titles and code-review links are enrichment. If first messages are uninformative, use tool and MCP counts as a weak hint. If there are ~0 sessions, leave the breakdown as a TODO.\n\n In the rendered guide, display categories with spaces and title case (e.g. \"Build Feature\" not \"build_feature\").\n\n3. **Gather the remaining pieces.** For repos, start with `currentRepo` and check the workspace for sibling repo directories. For MCP server setup, use each entry's `name` (and `urlOrigin` where present) to infer what the server does and how a teammate would get access. Leave the Team Tips and Get Started sections as TODO placeholders — you'll ask for these in Review and fill them in after.\n\n4. **Write the guide to `ONBOARDING.md`** following this template:\n\n```\n{{GUIDE_TEMPLATE}}\n```\n\n Fill in real numbers from the usage data (not placeholders). Use `generatedBy` for the name; if it's missing, omit the name. Ascii bar charts: `█` for filled, `░` for empty, 20 chars wide. Keep the HTML comment instruction at the bottom exactly as shown.\n\n5. **Render the guide in a code block, then close out the first turn.** You're co-authoring this guide with the guide creator — frame the follow-up as collaboration, not corrections.\n\n After the code block, add a `---` horizontal rule and a `**Review**` heading so the guide is visually separated from your questions. Under the heading, number these three questions:\n\n 1. \"I went with '[X]' for the team name — let me know if that sounds right.\" (or if you couldn't tell: \"What's the team name? I'll add it in.\")\n 2. Is there a starter task for someone new to Claude Code? (ticket or doc link — optional)\n 3. Any team tips you'd tell a new teammate that aren't already in CLAUDE.md?\n\n After they answer, update `ONBOARDING.md` with their team name, tips, and starter task. Then close with this exact line (not numbered, not paraphrased):\n\n Saved to `ONBOARDING.md`. Drop it in your team docs and channels — when a new teammate pastes it into Claude Code, they get a guided onboarding tour from there.\n\n Apply any edits they come back with to the file.",
"guideTemplate": "# Welcome to [Team Name]\n\n## How We Use Claude\n\nBased on [name]'s usage over the last [N] days:\n\nWork Type Breakdown:\n [Category 1] [ascii bar] [N]%\n [Category 2] [ascii bar] [N]%\n [Category 3] [ascii bar] [N]%\n ...\n\nTop Skills & Commands:\n [/command] [ascii bar] [N]x/month\n ...\n\nTop MCP Servers:\n [Server] [ascii bar] [N] calls\n ...\n\n## Your Setup Checklist\n\n### Codebases\n- [ ] [repo-name] — [repo url]\n...\n\n### MCP Servers to Activate\n- [ ] [Server] — [what it's for]. [How to get access]\n...\n\n### Skills to Know About\n- [/command] — [what it does, when the team uses it]\n...\n\n## Team Tips\n\n_TODO_\n\n## Get Started\n\n_TODO_\n\n<!-- INSTRUCTION FOR CLAUDE: A new teammate just pasted this guide for how the\nteam uses Claude Code. You're their onboarding buddy — warm, conversational,\nnot lecture-y.\n\nOpen with a warm welcome — include the team name from the title. Then: \"Your\nteammate uses Claude Code for [list all the work types]. Let's get you started.\"\n\nCheck what's already in place against everything under Setup Checklist\n(including skills), using markdown checkboxes — [x] done, [ ] not yet. Lead\nwith what they already have. One sentence per item, all in one message.\n\nTell them you'll help with setup, cover the actionable team tips, then the\nstarter task (if there is one). Offer to start with the first unchecked item,\nget their go-ahead, then work through the rest one by one.\n\nAfter setup, walk them through the remaining sections — offer to help where you\ncan (e.g. link to channels), and just surface the purely informational bits.\n\nDon't invent sections or summaries that aren't in the guide. The stats are the\nguide creator's personal usage data — don't extrapolate them into a \"team\nworkflow\" narrative. -->",
"windowDays": 30
},
"tengu_slim_subagent_claudemd": true,
"tengu_tangerine_ladder_boost": true,
"tengu_chair_sermon": false,
"tengu_gypsum_kite": true,
"tengu_quartz_heron": false,
"tengu_xterm_atlas_reset": true,
"tengu-model-error-overrides": {
"claude-fable-5": {
"block": "Claude Fable 5 is currently unavailable. Learn more: https://www.anthropic.com/news/fable-mythos-access"
}
},
"tengu_orchid_mantis_v2": true,
"tengu-off-switch": {
"activated": false
},
"tengu_feedback_survey_config": {
"minTimeBeforeFeedbackMs": 600000,
"minTimeBetweenFeedbackMs": 43200000,
"minTimeBetweenGlobalFeedbackMs": 43200000,
"minUserTurnsBeforeFeedback": 5,
"minUserTurnsBetweenFeedback": 25,
"hideThanksAfterMs": 3000,
"onForModels": [
"*"
],
"probability": 0.05
},
"tengu_cork_lantern": false,
"tengu_mint_lanes": false,
"tengu_bridge_attestation_enforce_config": {
"accept_level": "VERIFIED_BY_GATE",
"accept_statuses": []
},
"tengu_marble_sandcastle": false,
"tengu_bg_attach_stall_ms": 5000,
"tengu_workout2": true,
"tengu_orford_ness": false,
"tengu_porch_bell_9f": "",
"tengu_auto_mode_default_on": false,
"tengu_birch_kettle": false,
"tengu_classifier_summary_heuristic_emit": true,
"tengu_cobalt_thicket": false,
"tengu_destructive_command_warning": false,
"tengu_cinder_plover": "",
"tengu_cedar_halo": false,
"tengu_sotto_voce": true,
"tengu_sepia_moth": false,
"tengu_cedar_sundial": false,
"tengu_penguins_enabled": true,
"tengu_quiet_basalt_echo": false,
"tengu_ochre_hollow": true,
"tengu_coral_beacon": true,
"tengu_copper_thistle": false,
"tengu_1p_event_batch_config": {
"scheduledDelayMillis": 10000,
"maxExportBatchSize": 400,
"maxQueueSize": 8192,
"path": "/api/event_logging/v2/batch"
},
"tengu_amber_wren": {
"targetedRangeNudge": true,
"maxTokens": 25000
},
"tengu_amber_prism": true,
"tengu_cobalt_plinth": false,
"tengu_silent_harbor": false,
"tengu_chomp_inflection": true,
"tengu_mcp_elicitation": true,
"tengu_sm_config": {
"minimumMessageTokensToInit": 150000,
"minimumTokensBetweenUpdate": 40000,
"toolCallsBetweenUpdates": 10
},
"tengu_bridge_min_version": {
"minVersion": "2.1.70"
},
"tengu_kairos_input_needed_push": true,
"tengu_quiet_harbor": false,
"tengu_slate_wren": false,
"tengu_tool_search_unsupported_models": [
"claude-3-5-haiku",
"claude-3-haiku"
],
"tengu_native_cursor": true,
"tengu_orchid_mantis": false,
"tengu_amber_lark": true,
"tengu_shale_finch": true,
"tengu_cedar_plume": false,
"tengu_kairos_push_notifications": true,
"tengu_marble_whisper2": true,
"tengu_lichen_compass": false,
"tengu_c4w_usage_limit_notifications_enabled": true,
"tengu_scarf_coffee": false,
"tengu_copper_bridge": true,
"tengu_tool_pear": false,
"tengu_claudeai_mcp_connectors": true,
"tengu_ccr_post_turn_summary": false,
"tengu_sedge_lantern": true,
"tengu_feature_template": false,
"tengu_harbor_prism": true,
"tengu_cedar_inlet": "step",
"tengu_flax_grouse": false,
"tengu_event_sampling_config": {},
"tengu_herring_clock": false,
"tengu_quartz_vireo": "",
"tengu_team_discovery": false,
"tengu_gleaming_fair": true,
"tengu_marble_anvil": true,
"tengu_classifier_disabled_surfaces": "",
"tengu_pewter_brook": false,
"tengu_vscode_review_upsell": false,
"claude_code_skills_dashboard_enabled_cli": false,
"tengu_post_compact_survey": false,
"tengu_reactive_compact_remote": false,
"tengu_idle_amber_finch": false,
"tengu_noreread_q7m_velvet": false,
"tengu_ultraplan_config": {
"enabled": true
},
"tengu_scratch": false,
"tengu_alder_compass": false,
"tengu_olive_hinge": "",
"tengu_shining_fractals": false,
"tengu_maple_pier": false,
"tengu_sessions_elevated_auth_enforcement": true,
"tengu_turtle_carbon": true,
"tengu_billiard_aviary": false,
"tengu_cinder_almanac": true,
"tengu_osprey_lantern": false,
"tengu-top-of-feed-tip": {
"tip": "Claude Fable 5 is currently unavailable. Please use Opus 4.8 or another available model. Learn more: https://www.anthropic.com/news/fable-mythos-access",
"color": "warning"
},
"tengu_cobalt_raccoon": true,
"tengu_loud_sugary_rock": false,
"tengu_willow_mode": "hint_v2",
"tengu_blue_coaster": false,
"tengu_snippet_save": false,
"tengu_amber_lattice": {
"plugins": [
"security-guidance",
"code-review",
"commit-commands",
"code-simplifier",
"hookify",
"feature-dev",
"frontend-design",
"pr-review-toolkit",
"skill-creator",
"plugin-dev",
"agent-sdk-dev",
"mcp-server-dev",
"claude-code-setup",
"claude-md-management",
"playground",
"ralph-loop",
"explanatory-output-style",
"learning-output-style",
"clangd-lsp",
"csharp-lsp",
"gopls-lsp",
"jdtls-lsp",
"kotlin-lsp",
"lua-lsp",
"php-lsp",
"pyright-lsp",
"ruby-lsp",
"rust-analyzer-lsp",
"swift-lsp",
"typescript-lsp"
]
},
"tengu_slate_harbor_experiment": false,
"tengu_velvet_ibis": {},
"tengu_bridge_requires_action_details": true,
"tengu_lapis_finch": true,
"tengu_satin_quoll": {},
"tengu_moth_copse": false,
"tengu_silk_hinge": false,
"tengu_surreal_dali": true,
"tengu_cobalt_ridge": true,
"tengu_flint_harbor": false,
"tengu_plank_river_frost": "user_intent",
"tengu_velvet_mallet_haiku": false,
"tengu_velvet_mallet": false,
"tengu_velvet_mallet_haiku_4_5": false,
"tengu_velvet_hammer_falcon": false,
"tengu_loud_sugary_rock2": false,
"tengu_velvet_hammer_sonnet_4_5": false,
"tengu_velvet_hammer_sonnet": false,
"tengu_tab_read_sep": false,
"tengu_quill_harbor": "acceptEdits",
"tengu_velvet_hammer": false,
"tengu_velvet_hammer_opus": false,
"tengu_c4e_slash_upsell": true,
"tengu_velvet_hammer_haiku_4_5": false,
"tengu_feature_claudified_template": false,
"tengu_slate_quill": true,
"tengu_ax_screen_reader": false,
"tengu_windows_credman": false,
"tengu_basalt_tern": false,
"tengu_velvet_mallet_opus": false,
"tengu_velvet_hammer_haiku": false,
"tengu_velvet_static": true,
"tengu_velvet_mallet_sonnet": false,
"tengu_soft_slate_nudge": "baseline",
"tengu_lantern_hearth": "off",
"tengu_velvet_mallet_falcon": false,
"tengu_velvet_mallet_sonnet_4_5": false
},
"firstStartTime": "2026-06-05T19:39:28.542Z",
"opusProMigrationComplete": true,
"sonnet1m45MigrationComplete": true,
"seenNotifications": {},
"migrationVersion": 13,
"userID": "9d89994d486a4884b8cf33372d8a4cd61ebf7d34009e9d3cbce9db24e2e971a4",
"changelogLastFetched": 1781361371930,
"autoUpdatesProtectedForNative": true,
"claudeCodeFirstTokenDate": "2026-04-11T19:03:48.223040Z",
"hasCompletedOnboarding": true,
"lastOnboardingVersion": "2.1.165",
"groveConfigCache": {
"09792e21-2287-4348-b4d4-34cddbbfabc5": {
"grove_enabled": true,
"timestamp": 1781406640065
}
},
"cachedExperimentFeatures": [
"tengu_amber_prism",
"tengu_basalt_spur",
"tengu_cedar_inlet",
"tengu_coral_beacon",
"tengu_flint_harbor",
"tengu_mcp_subagent_prompt",
"tengu_ochre_hollow",
"tengu_orchid_mantis_v2",
"tengu_plank_river_frost",
"tengu_read_dedup_killswitch"
],
"cachedGrowthBookFeaturesAt": 1781406639973,
"lastReleaseNotesSeen": "2.1.177",
"projects": {
"/root": {
"allowedTools": [],
"mcpContextUris": [],
"mcpServers": {},
"enabledMcpjsonServers": [],
"disabledMcpjsonServers": [],
"hasTrustDialogAccepted": false,
"projectOnboardingSeenCount": 3,
"hasClaudeMdExternalIncludesApproved": false,
"hasClaudeMdExternalIncludesWarningShown": false,
"exampleFiles": [],
"lastGracefulShutdown": true,
"lastVersionBase": "2.1.177",
"lastCost": 22.33646404999996,
"lastAPIDuration": 5145196,
"lastAPIDurationWithoutRetries": 5144336,
"lastToolDuration": 506403,
"lastDuration": 11598574,
"lastLinesAdded": 652,
"lastLinesRemoved": 392,
"lastTotalInputTokens": 32237,
"lastTotalOutputTokens": 290269,
"lastTotalCacheCreationInputTokens": 1509902,
"lastTotalCacheReadInputTokens": 44894244,
"lastTotalWebSearchRequests": 0,
"lastFpsAverage": 6.03,
"lastFpsLow1Pct": 451.66,
"lastModelUsage": {
"claude-haiku-4-5-20251001": {
"inputTokens": 21206,
"outputTokens": 30015,
"cacheReadInputTokens": 2624502,
"cacheCreationInputTokens": 791709,
"webSearchRequests": 0,
"costUSD": 1.4233674499999998
},
"claude-sonnet-4-6": {
"inputTokens": 11031,
"outputTokens": 260254,
"cacheReadInputTokens": 42269742,
"cacheCreationInputTokens": 718193,
"webSearchRequests": 0,
"costUSD": 20.913096599999978
}
},
"lastSessionId": "685e6c5b-62c1-40bd-9cfd-2c9f7e15c50f",
"lastSessionMetrics": {
"frame_duration_ms_count": 69975,
"frame_duration_ms_min": 0.06756199989467859,
"frame_duration_ms_max": 100.53212600015104,
"frame_duration_ms_avg": 0.663704399986351,
"frame_duration_ms_p50": 0.4998550007585436,
"frame_duration_ms_p95": 1.5134988494683035,
"frame_duration_ms_p99": 2.4773533696774384,
"pre_tool_hook_duration_ms_count": 655,
"pre_tool_hook_duration_ms_min": 0,
"pre_tool_hook_duration_ms_max": 12,
"pre_tool_hook_duration_ms_avg": 0.1267175572519084,
"pre_tool_hook_duration_ms_p50": 0,
"pre_tool_hook_duration_ms_p95": 1,
"pre_tool_hook_duration_ms_p99": 1,
"hook_duration_ms_count": 465,
"hook_duration_ms_min": 0,
"hook_duration_ms_max": 22,
"hook_duration_ms_avg": 0.3204301075268817,
"hook_duration_ms_p50": 0,
"hook_duration_ms_p95": 1,
"hook_duration_ms_p99": 8.360000000000014
},
"hasCompletedProjectOnboarding": true
}
},
"routineFiredWatermark": "2026-06-05T19:47:09.178Z",
"penguinModeOrgEnabled": true,
"closedIssuesLastChecked": 1781406639965,
"passesEligibilityCache": {
"4bb43199-0efc-4d5c-b552-79865cb0361b": {
"eligible": true,
"referral_code_details": {
"code": "BeGGjphr1g",
"campaign": "claude_code_guest_pass_a47c",
"referral_link": "https://claude.ai/referral/BeGGjphr1g"
},
"referrer_reward": {
"amount_minor_units": 1000,
"currency": "USD"
},
"remaining_passes": 3,
"limit": 3,
"share_link": "https://claude.ai/referral/BeGGjphr1g",
"terms_url": "https://support.claude.com/en/articles/12875061-claude-code-guest-passes",
"timestamp": 1781406640514
}
},
"cachedExtraUsageDisabledReason": null,
"passesUpsellSeenCount": 3,
"hasVisitedPasses": false,
"passesLastSeenRemaining": 3,
"officialMarketplaceAutoInstallAttempted": true,
"officialMarketplaceAutoInstalled": true,
"tipLifetimeShownCounts": {
"fotw-campaign-upsell": 6,
"new-user-warmup": 2,
"plan-mode-for-complex-tasks": 5,
"memory-command": 2,
"theme-command": 2,
"colorterm-truecolor": 1,
"status-line": 1,
"prompt-queue": 3,
"enter-to-steer-in-relatime": 2,
"todo-list": 2,
"ide-upsell-external-terminal": 5,
"install-github-app": 3,
"install-slack-app": 3,
"drag-and-drop-images": 2,
"double-esc-code-restore": 2,
"continue": 2,
"shift-tab": 2,
"image-paste": 1,
"web-app": 2,
"color-when-multi-clauding": 1,
"custom-agents": 2,
"remote-control": 2,
"voice-mode": 2,
"goal-command-nudge": 4,
"guest-passes": 6,
"feedback-command": 2,
"frontend-design-plugin": 1,
"permissions": 2,
"rename-conversation": 1,
"custom-commands": 1,
"c4e-remote-sessions": 1,
"subagent-fanout-nudge": 1,
"no-flicker": 1
},
"feedbackSurveyState": {
"lastShownTime": 1781411703066
},
"hasUsedBackslashReturn": true,
"agentLastUsed": {
"bg": 1780696781055
},
"remoteControlUpsellSeenCount": 3,
"fullscreenUpsellSeenCount": 3,
"lastShownEmergencyTip": "Claude Fable 5 is currently unavailable. Please use Opus 4.8 or another available model. Learn more: https://www.anthropic.com/news/fable-mythos-access",
"oauthAccount": {
"accountUuid": "09792e21-2287-4348-b4d4-34cddbbfabc5",
"emailAddress": "gmer4lfe@gmail.com",
"organizationUuid": "4bb43199-0efc-4d5c-b552-79865cb0361b",
"hasExtraUsageEnabled": true,
"billingType": "stripe_subscription",
"accountCreatedAt": "2026-04-03T21:52:35.642439Z",
"subscriptionCreatedAt": "2026-04-11T13:14:49.905923Z",
"ccOnboardingFlags": {},
"claudeCodeTrialEndsAt": null,
"claudeCodeTrialDurationDays": null,
"seatTier": null,
"displayName": "Gmer4Lfe",
"organizationRole": "admin",
"workspaceRole": null,
"organizationName": "gmer4lfe@gmail.com's Organization",
"organizationType": "claude_pro",
"organizationRateLimitTier": "default_claude_ai",
"userRateLimitTier": null
},
"clientDataCache": {
"cedar_lagoon": {
"claude-fable": true,
"claude-mythos": true
},
"pewter_owl_tool": true,
"pewter_owl_model": "claude-fable"
},
"additionalModelOptionsCache": [
{
"value": "claude-fable-5[1m]",
"label": "Fable (disabled)",
"description": "Claude Fable 5 is currently unavailable. Learn more: https://www.anthropic.com/news/fable-mythos-access",
"disabled": true
}
],
"additionalModelCostsCache": {}
}
@@ -1,785 +0,0 @@
{
"numStartups": 23,
"installMethod": "native",
"autoUpdates": false,
"hasSeenTasksHint": true,
"tipsHistory": {
"fotw-campaign-upsell": 13,
"new-user-warmup": 6,
"plan-mode-for-complex-tasks": 22,
"memory-command": 16,
"theme-command": 21,
"colorterm-truecolor": 1,
"status-line": 1,
"prompt-queue": 11,
"enter-to-steer-in-relatime": 21,
"todo-list": 21,
"ide-upsell-external-terminal": 19,
"install-github-app": 22,
"install-slack-app": 22,
"drag-and-drop-images": 14,
"double-esc-code-restore": 14,
"continue": 14,
"shift-tab": 15,
"image-paste": 4,
"web-app": 19,
"color-when-multi-clauding": 6,
"custom-agents": 21,
"remote-control": 21,
"voice-mode": 16,
"goal-command-nudge": 16,
"guest-passes": 22,
"feedback-command": 22,
"frontend-design-plugin": 6,
"permissions": 22,
"rename-conversation": 11,
"custom-commands": 11,
"c4e-remote-sessions": 18,
"subagent-fanout-nudge": 18,
"no-flicker": 19
},
"promptQueueUseCount": 44,
"cachedGrowthBookFeatures": {
"tengu_flint_harbor_share": false,
"tengu_cinder_plover": "",
"tengu_velvet_cascade": {},
"tengu_post_compact_survey": false,
"tengu_flint_harbor": false,
"tengu_sedge_lantern": true,
"tengu_harbor": true,
"tengu_slim_subagent_claudemd": true,
"tengu_cedar_plume": false,
"tengu_bridge_poll_interval_config": {
"poll_interval_ms_not_at_capacity": 2000,
"poll_interval_ms_at_capacity": 600000,
"heartbeat_interval_ms": 0,
"multisession_poll_interval_ms_not_at_capacity": 5000,
"multisession_poll_interval_ms_at_capacity": 60000,
"multisession_poll_interval_ms_partial_capacity": 5000,
"non_exclusive_heartbeat_interval_ms": 180000,
"session_keepalive_interval_ms": 0,
"session_keepalive_interval_v2_ms": 0
},
"tengu_bg_attach_stall_ms": 5000,
"tengu_quiet_harbor": false,
"tengu_mcp_singleton_unwrap": true,
"tengu_sage_compass": {},
"tengu_slate_moth": true,
"tengu_event_watchdog_default_on": false,
"tengu_claudeai_mcp_connectors": true,
"tengu_feedback_survey_config": {
"minTimeBeforeFeedbackMs": 600000,
"minTimeBetweenFeedbackMs": 43200000,
"minTimeBetweenGlobalFeedbackMs": 43200000,
"minUserTurnsBeforeFeedback": 5,
"minUserTurnsBetweenFeedback": 25,
"hideThanksAfterMs": 3000,
"onForModels": [
"*"
],
"probability": 0.05
},
"tengu_loggia_carousel": false,
"tengu_file_write_optimization": true,
"tengu_sepia_moth": false,
"tengu_harbor_willow": false,
"tengu_amber_sextant": true,
"tengu_event_sampling_config": {},
"tengu_cedar_halo": false,
"tengu_ember_trail": "0",
"tengu_slate_meadow": true,
"tengu_c4w_usage_limit_notifications_enabled": true,
"tengu_lichen_compass": false,
"tengu_osprey_lantern": false,
"tengu_desktop_upsell_v2": {
"enabled": false
},
"tengu_ccr_bridge": true,
"tengu_drift_lantern": false,
"tengu_herring_clock": false,
"tengu_sm_config": {
"minimumMessageTokensToInit": 150000,
"minimumTokensBetweenUpdate": 40000,
"toolCallsBetweenUpdates": 10
},
"tengu_feature_template": false,
"tengu_bridge_attestation_enforce": false,
"tengu_nimble_amber_prose": false,
"tengu_destructive_command_warning": false,
"tengu_ladder_mq7": false,
"tengu_crimson_echo": {},
"tengu-off-switch": {
"activated": false
},
"tengu_scratch": false,
"tengu_session_memory": false,
"tengu_orchid_mantis_v2": true,
"tengu_prompt_cache_1h_config": {
"allowlist": [
"repl_main_thread*",
"sdk",
"auto_mode",
"rolling_compact",
"memdir_relevance",
"agent_classifier",
"prompt_suggestion",
"away_summary",
"extract_memories",
"compact"
]
},
"tengu_amber_redwood2": "",
"tengu_kairos_cron": true,
"tengu_marble_anvil": true,
"tengu_billiard_aviary": false,
"tengu_basalt_spur": false,
"tengu_ochre_hollow": true,
"tengu_maple_tide": false,
"tengu_crimson_vector": false,
"tengu_cedar_sundial": false,
"tengu_skills_dashboard_enabled": false,
"tengu_red_coaster": false,
"tengu_good_survey_transcript_ask_config": {
"probability": 0.5
},
"tengu_system_prompt_global_cache": true,
"tengu_slate_kestrel": true,
"tengu_harbor_prism": true,
"tengu_disable_bypass_permissions_mode": false,
"tengu_slate_ribbon": true,
"tengu_1p_event_batch_config": {
"scheduledDelayMillis": 10000,
"maxExportBatchSize": 400,
"maxQueueSize": 8192,
"path": "/api/event_logging/v2/batch"
},
"tengu_cobalt_compass": true,
"tengu_shining_fractals": false,
"tengu_marble_sandcastle": false,
"tengu_pewter_summit": true,
"tengu_slate_finch": true,
"tengu_kairos_loop_prompt": true,
"tengu_version_config": {
"minVersion": "1.0.24"
},
"tengu_miraculo_the_bard": false,
"tengu_copper_fox": false,
"tengu_marble_whisper2": true,
"tengu_orchid_mantis": false,
"tengu_willow_sentinel_ttl_hours": 1,
"tengu_startup_notice": "",
"tengu_amber_flint": true,
"tengu_kairos_loop_dynamic": true,
"tengu_walrus_canteen": false,
"tengu_kairos_push_notifications": true,
"tengu_maple_sundial": false,
"tengu_malformed_tool_use_clean_retry": false,
"tengu_kestrel_arch": "OFF",
"tengu_gha_plugin_code_review": false,
"tengu_ember_latch": true,
"tengu_plum_vx3": true,
"tengu_bridge_repl_v2": true,
"tengu_cobalt_thicket": false,
"tengu_orchid_trellis": false,
"tengu_cobalt_lantern": true,
"tengu_cloth_snorkel": false,
"tengu_passport_quail": false,
"tengu_amber_sentinel": true,
"tengu_cork_lantern": false,
"tengu_penguins_enabled": true,
"tengu_velvet_ibis": {},
"tengu_snippet_save": false,
"tengu_maple_pier": false,
"tengu_cobalt_raccoon": true,
"tengu_ultraplan_prompt_identifier": "visual_plan",
"tengu_copper_bridge": true,
"tengu_willow_refresh_ttl_hours": 0,
"tengu_ultraplan_timeout_seconds": 5400,
"tengu_cedar_hollow_7m": {},
"tengu_quartz_vireo": "",
"claude_code_skills_dashboard_enabled_cli": false,
"tengu_cork_m4q": true,
"tengu_mocha_barista": true,
"tengu_harbor_ledger": [
{
"marketplace": "claude-plugins-official",
"plugin": "discord"
},
{
"marketplace": "claude-plugins-official",
"plugin": "telegram"
},
{
"marketplace": "claude-plugins-official",
"plugin": "fakechat"
},
{
"marketplace": "claude-plugins-official",
"plugin": "imessage"
}
],
"tengu-model-error-overrides": {
"claude-fable-5": {
"block": "Claude Fable 5 is currently unavailable. Learn more: https://www.anthropic.com/news/fable-mythos-access"
}
},
"tengu_pewter_lantern": false,
"tengu_mcp_stateless_skip_init": true,
"tengu_jade_anvil_4": false,
"tengu_mcp_retry_failed_remote": false,
"tengu_pewter_ledger": "OFF",
"tengu_fg_left_arrow_agents": true,
"tengu_lantern_spool": false,
"tengu_auto_notice_once": true,
"tengu_shale_finch": true,
"tengu_immediate_model_command": false,
"tengu_ccr_v2_send_events_cli": true,
"tengu_compact_cache_prefix": true,
"tengu_mcp_elicitation": true,
"tengu_harbor_permissions": true,
"tengu_sepia_cormorant": [],
"tengu_grey_step2": {
"enabled": true,
"dialogTitle": "We recommend medium effort for Opus",
"dialogDescription": "Effort determines how long Claude thinks for when completing your task. We recommend medium effort for most tasks to balance speed and intelligence and maximize rate limits. Use ultrathink to trigger high effort when needed."
},
"tengu_quartz_heron": false,
"tengu_ccr_bridge_multi_session": true,
"tengu_team_discovery": false,
"tengu_otk_slot_v1": false,
"tengu_blue_coaster": false,
"tengu_vscode_onboarding": false,
"tengu_amber_lynx": false,
"tengu_cinder_almanac": true,
"tengu_mint_lanes": false,
"tengu_max_version_config": {},
"tengu_basalt_sundial": false,
"tengu_copper_thistle": false,
"tengu_saffron_lattice": {
"enabled": false,
"planLimitsEndDate": "2026-06-22T10:00:00Z",
"hideRateLimitsDescription": true
},
"tengu_penguin_mode_promo": {
"discountPercent": 0,
"endDate": "Feb 16"
},
"tengu_bridge_requires_action_details": true,
"tengu_walnut_prism": false,
"tengu_amber_rokovoko": 0.2,
"tengu_loud_sugary_rock": false,
"tengu_steady_lantern": false,
"tengu_lapis_anchor": "off",
"tengu_amber_wren": {
"targetedRangeNudge": true,
"maxTokens": 25000
},
"tengu_ccr_post_turn_summary": false,
"tengu_sessions_elevated_auth_enforcement": true,
"tengu_hazel_osprey_floor": 75000,
"tengu_flax_grouse": false,
"tengu_dune_wren": false,
"tengu_hawthorn_window": 200000,
"tengu_slate_wren": false,
"tengu_permission_friction": true,
"tengu_amber_anchor": false,
"tengu_fgts": true,
"tengu_chomp_inflection": true,
"tengu_birthday_hat": false,
"tengu_olive_hinge": "",
"tengu_brick_follow": false,
"tengu_doorbell_agave": false,
"tengu_sage_compass2": {
"enabled": true
},
"tengu_lilac_loom": {},
"tengu_compass_dial": true,
"tengu_sparrow_ledger": false,
"tengu_pewter_brook": false,
"tengu_prompt_cache_diagnostics": true,
"tengu_chert_bezel": true,
"tengu_birch_compass": true,
"tengu_timber_lark": "copy_a",
"tengu_coral_beacon": true,
"tengu_worktree_mode": true,
"tengu_turtle_carbon": true,
"tengu_workout2": true,
"tengu_vellum_siding": false,
"tengu_vscode_review_upsell": false,
"tengu_cedar_lantern": true,
"tengu_kairos_cron_durable": false,
"tengu_anchor_tide": true,
"tengu_cobalt_ridge": true,
"tengu_bridge_repl_v2_config": {
"init_retry_max_attempts": 3,
"init_retry_base_delay_ms": 500,
"init_retry_jitter_fraction": 0.25,
"init_retry_max_delay_ms": 4000,
"http_timeout_ms": 10000,
"uuid_dedup_buffer_size": 2000,
"heartbeat_interval_ms": 20000,
"heartbeat_jitter_fraction": 0.1,
"token_refresh_buffer_ms": 600000,
"teardown_archive_timeout_ms": 1500,
"connect_timeout_ms": 15000,
"min_version": "2.1.70",
"should_show_app_upgrade_message": false
},
"tengu_malort_pedway": {
"enabled": true,
"pixelValidation": false,
"clipboardPasteMultiline": true,
"screenshotFilter": true,
"mouseAnimation": true,
"hideBeforeAction": true,
"autoTargetDisplay": false,
"coordinateMode": "pixels"
},
"tengu_tool_search_unsupported_models": [
"claude-3-5-haiku",
"claude-3-haiku"
],
"tengu_tussock_oriole": false,
"tengu_reactive_compact_remote": false,
"tengu_ccr_bundle_seed_enabled": true,
"tengu_silent_harbor": false,
"tengu_plank_river_frost": "user_intent",
"tengu_idle_amber_finch": false,
"tengu_xterm_atlas_reset": true,
"tengu_flint_harbor_prompt": {
"prompt": "You are helping a power user generate an onboarding guide for teammates who are new to Claude Code. The guide will live in the team's onboarding docs and can be pasted into Claude for an interactive walkthrough.\n\nYou're co-authoring this with them — collaborative and helpful, like a teammate who's done this before and is happy to share.\n\n## Usage data (last {{WINDOW_DAYS}} days)\n\nThis was scanned from the guide creator's local Claude Code transcripts:\n\n```json\n{{USAGE_DATA}}\n```\n\n## Your task\n\nBefore anything else — including before thinking through the classification — output exactly this line as your first visible text:\n\n> Looking at how you've used Claude over the last {{WINDOW_DAYS}} days to put together an onboarding guide for teammates new to Claude Code.\n\nThis must come before any extended thinking about session descriptors. The guide creator is staring at a blank screen until you do. Classification is step 2, not step 1.\n\nGenerate the guide immediately, then ask for revisions. Don't wait for answers first — it's easier for the guide creator to edit a concrete draft than answer abstract questions.\n\n1. **Output the acknowledgment line above.** No thinking, no classification, no tool calls before this. One line, then move on.\n\n2. **Derive the work-type breakdown.** Read the `sessionDescriptors` array — each entry describes one session via its title, any linked code reviews (`prNumbers`), and first user message. Classify each session into one of these task types:\n\n - **build_feature** — new functionality, scripts, tools, config/CI/env setup\n - **debug_fix** — investigating and fixing bugs\n - **improve_quality** — refactoring, tests, cleanup, code review\n - **analyze_data** — queries, metrics, number crunching\n - **plan_design** — architecture, approach, strategy, understanding unfamiliar code, design review\n - **prototype** — spikes, POCs, throwaway exploration\n - **write_docs** — PRDs, RFCs, READMEs, design docs, copy/doc review\n\n Categories describe the *type of task*, not the project or domain — a teammate on any project should recognize them. Review sessions belong with whatever's being reviewed: code review is improve_quality, doc review is write_docs, design review is plan_design. Most sessions fit the list; only invent a new category if it's genuinely a different type of task. Pick the top 3-5 with rough percentages. First messages alone are usually enough; titles and code-review links are enrichment. If first messages are uninformative, use tool and MCP counts as a weak hint. If there are ~0 sessions, leave the breakdown as a TODO.\n\n In the rendered guide, display categories with spaces and title case (e.g. \"Build Feature\" not \"build_feature\").\n\n3. **Gather the remaining pieces.** For repos, start with `currentRepo` and check the workspace for sibling repo directories. For MCP server setup, use each entry's `name` (and `urlOrigin` where present) to infer what the server does and how a teammate would get access. Leave the Team Tips and Get Started sections as TODO placeholders — you'll ask for these in Review and fill them in after.\n\n4. **Write the guide to `ONBOARDING.md`** following this template:\n\n```\n{{GUIDE_TEMPLATE}}\n```\n\n Fill in real numbers from the usage data (not placeholders). Use `generatedBy` for the name; if it's missing, omit the name. Ascii bar charts: `█` for filled, `░` for empty, 20 chars wide. Keep the HTML comment instruction at the bottom exactly as shown.\n\n5. **Render the guide in a code block, then close out the first turn.** You're co-authoring this guide with the guide creator — frame the follow-up as collaboration, not corrections.\n\n After the code block, add a `---` horizontal rule and a `**Review**` heading so the guide is visually separated from your questions. Under the heading, number these three questions:\n\n 1. \"I went with '[X]' for the team name — let me know if that sounds right.\" (or if you couldn't tell: \"What's the team name? I'll add it in.\")\n 2. Is there a starter task for someone new to Claude Code? (ticket or doc link — optional)\n 3. Any team tips you'd tell a new teammate that aren't already in CLAUDE.md?\n\n After they answer, update `ONBOARDING.md` with their team name, tips, and starter task. Then close with this exact line (not numbered, not paraphrased):\n\n Saved to `ONBOARDING.md`. Drop it in your team docs and channels — when a new teammate pastes it into Claude Code, they get a guided onboarding tour from there.\n\n Apply any edits they come back with to the file.",
"guideTemplate": "# Welcome to [Team Name]\n\n## How We Use Claude\n\nBased on [name]'s usage over the last [N] days:\n\nWork Type Breakdown:\n [Category 1] [ascii bar] [N]%\n [Category 2] [ascii bar] [N]%\n [Category 3] [ascii bar] [N]%\n ...\n\nTop Skills & Commands:\n [/command] [ascii bar] [N]x/month\n ...\n\nTop MCP Servers:\n [Server] [ascii bar] [N] calls\n ...\n\n## Your Setup Checklist\n\n### Codebases\n- [ ] [repo-name] — [repo url]\n...\n\n### MCP Servers to Activate\n- [ ] [Server] — [what it's for]. [How to get access]\n...\n\n### Skills to Know About\n- [/command] — [what it does, when the team uses it]\n...\n\n## Team Tips\n\n_TODO_\n\n## Get Started\n\n_TODO_\n\n<!-- INSTRUCTION FOR CLAUDE: A new teammate just pasted this guide for how the\nteam uses Claude Code. You're their onboarding buddy — warm, conversational,\nnot lecture-y.\n\nOpen with a warm welcome — include the team name from the title. Then: \"Your\nteammate uses Claude Code for [list all the work types]. Let's get you started.\"\n\nCheck what's already in place against everything under Setup Checklist\n(including skills), using markdown checkboxes — [x] done, [ ] not yet. Lead\nwith what they already have. One sentence per item, all in one message.\n\nTell them you'll help with setup, cover the actionable team tips, then the\nstarter task (if there is one). Offer to start with the first unchecked item,\nget their go-ahead, then work through the rest one by one.\n\nAfter setup, walk them through the remaining sections — offer to help where you\ncan (e.g. link to channels), and just surface the purely informational bits.\n\nDon't invent sections or summaries that aren't in the guide. The stats are the\nguide creator's personal usage data — don't extrapolate them into a \"team\nworkflow\" narrative. -->",
"windowDays": 30
},
"tengu_native_cursor": true,
"tengu_pewter_lark": "off",
"tengu_streaming_tool_execution2": true,
"tengu_cobalt_plinth": false,
"tengu_porch_bell_9f": "",
"tengu_marble_whisper": true,
"tengu_classifier_summary_heuristic_emit": true,
"tengu_willow_mode": "hint_v2",
"tengu_birch_kettle": false,
"tengu_bridge_min_version": {
"minVersion": "2.1.70"
},
"tengu_classifier_disabled_surfaces": "",
"tengu_cedar_inlet": "step",
"tengu_velvet_moth": 0.2,
"tengu_auto_mode_default_on": false,
"tengu_workflows_enabled": true,
"tengu_tangerine_ladder_boost": true,
"tengu_gypsum_kite": true,
"tengu_gleaming_fair": true,
"tengu_noreread_q7m_velvet": false,
"tengu_crystal_beam": {
"budgetTokens": 0
},
"tengu_quiet_slate_wren": false,
"tengu_vscode_feedback_survey": true,
"tengu_sotto_voce": true,
"tengu_slate_harrier": "off",
"tengu_tool_pear": false,
"tengu_surreal_dali": true,
"tengu_collage_kaleidoscope": true,
"tengu_pewter_kestrel": {
"global": 50000,
"Bash": 30000,
"PowerShell": 30000,
"Grep": 20000,
"Snip": 1000,
"StrReplaceBasedEditTool": 30000,
"BashSearchTool": 20000
},
"tengu_slate_thimble": false,
"tengu_negative_interaction_transcript_ask_config": {
"probability": 0
},
"tengu_moss_anchor": false,
"tengu_mcp_local_oauth_blocked_hosts": {
"hosts": [
"microsoft365.mcp.claude.com",
"gmail.mcp.claude.com",
"gcal.mcp.claude.com"
]
},
"tengu_bridge_attestation_enforce_config": {
"accept_level": "VERIFIED_BY_GATE",
"accept_statuses": []
},
"tengu_lapis_thicket": false,
"tengu_auto_mode_config": {
"enabled": "enabled",
"twoStageClassifier": true
},
"tengu_basalt_meadow": true,
"tengu_byte_stream_idle_timeout_ms": 180000,
"tengu_lapis_finch": true,
"tengu_prism_ledger": false,
"tengu_prompt_suggestion": true,
"tengu_react_vulnerability_warning": false,
"tengu_amber_prism": true,
"tengu_plugin_official_mkt_git_fallback": true,
"tengu_cobalt_wren": false,
"tengu_coral_fern": false,
"tengu_log_datadog_events": true,
"tengu_amber_heron": false,
"tengu_saffron_anchor": true,
"tengu_tern_alloy": "copy_a",
"tengu_gouda_loop": true,
"tengu_dunwich_bell": false,
"tengu_mcp_subagent_prompt": true,
"tengu_quiet_basalt_echo": false,
"tengu_sedge_lantern_holdback": false,
"tengu_garnet_finch": false,
"tengu_chair_sermon": false,
"tengu_umber_petrel": false,
"tengu_bramble_lintel": 7,
"tengu_sub_nomdrep_q7k": true,
"tengu_swann_brevity": "focused",
"tengu_marble_lark": false,
"tengu_scarf_coffee": false,
"tengu_bridge_poll_interval_ms": 0,
"tengu_moth_copse": false,
"tengu_bad_survey_transcript_ask_config": {
"probability": 1
},
"tengu_desktop_upsell": {
"enable_shortcut_tip": true,
"enable_startup_dialog": false
},
"tengu_agent_list_attach": true,
"tengu_amber_lark": true,
"tengu_slate_siskin": {
"enabled": false,
"timeoutMs": 8000,
"throttleMs": 30000,
"summaryLineThreshold": 5
},
"tengu_tide_elm": "off",
"tengu_alder_compass": false,
"tengu-fable-off-switch": {
"activated": false
},
"tengu_hazel_osprey": false,
"tengu_cobalt_heron": true,
"tengu_code_diff_cli": true,
"tengu_trace_lantern": false,
"tengu_silk_hinge": false,
"tengu_amber_lattice": {
"plugins": [
"security-guidance",
"code-review",
"commit-commands",
"code-simplifier",
"hookify",
"feature-dev",
"frontend-design",
"pr-review-toolkit",
"skill-creator",
"plugin-dev",
"agent-sdk-dev",
"mcp-server-dev",
"claude-code-setup",
"claude-md-management",
"playground",
"ralph-loop",
"explanatory-output-style",
"learning-output-style",
"clangd-lsp",
"csharp-lsp",
"gopls-lsp",
"jdtls-lsp",
"kotlin-lsp",
"lua-lsp",
"php-lsp",
"pyright-lsp",
"ruby-lsp",
"rust-analyzer-lsp",
"swift-lsp",
"typescript-lsp"
]
},
"tengu_willow_census_ttl_hours": 24,
"tengu_fennel_kite": false,
"tengu_orford_ness": false,
"tengu_read_dedup_killswitch": false,
"tengu_onyx_plover": {
"enabled": false,
"minHours": 24,
"minSessions": 3,
"remoteEnabled": false
},
"tengu_kairos_input_needed_push": true,
"tengu_fennel_kite_model": "",
"tengu_ccr_bundle_max_bytes": 104857600,
"tengu-top-of-feed-tip": {
"tip": "Claude Fable 5 is currently unavailable. Please use Opus 4.8 or another available model. Learn more: https://www.anthropic.com/news/fable-mythos-access",
"color": "warning"
},
"tengu_copper_wren": false,
"tengu_frond_boric": {},
"tengu_satin_quoll": {},
"tengu_hawthorn_steeple": false,
"tengu_review_bughunter_config": {
"fleet_size": 5,
"max_duration_minutes": 10,
"agent_timeout_seconds": 600,
"total_wallclock_minutes": 22,
"model": "claude-opus-4-7",
"cost_note": "$5-$25",
"duration_note": "~5-10 min",
"enabled": true
},
"tengu_slate_nexus": true,
"tengu_keybinding_customization_release": true,
"tengu_canary": {},
"tengu_classifier_summary_llm_emit": true,
"tengu_ultraplan_config": {
"enabled": true
},
"tengu_vellum_lantern": false,
"tengu_slate_harbor_experiment": false,
"tengu_ashen_kelp": true,
"tengu_soft_slate_nudge": "baseline",
"tengu_velvet_hammer_haiku_4_5": false,
"tengu_velvet_hammer_haiku": false,
"tengu_velvet_static": true,
"tengu_velvet_mallet_opus": false,
"tengu_velvet_hammer_sonnet_4_5": false,
"tengu_c4e_slash_upsell": true,
"tengu_velvet_mallet_sonnet": false,
"tengu_loud_sugary_rock2": false,
"tengu_windows_credman": false,
"tengu_velvet_hammer": false,
"tengu_lantern_hearth": "off",
"tengu_velvet_mallet_falcon": false,
"tengu_ax_screen_reader": false,
"tengu_velvet_mallet": false,
"tengu_velvet_hammer_sonnet": false,
"tengu_velvet_mallet_haiku": false,
"tengu_velvet_mallet_sonnet_4_5": false,
"tengu_velvet_hammer_opus": false,
"tengu_velvet_mallet_haiku_4_5": false,
"tengu_tab_read_sep": false,
"tengu_velvet_hammer_falcon": false,
"tengu_feature_claudified_template": false,
"tengu_quill_harbor": "acceptEdits",
"tengu_slate_quill": true,
"tengu_basalt_tern": false
},
"firstStartTime": "2026-06-05T19:39:28.542Z",
"opusProMigrationComplete": true,
"sonnet1m45MigrationComplete": true,
"seenNotifications": {},
"migrationVersion": 13,
"userID": "9d89994d486a4884b8cf33372d8a4cd61ebf7d34009e9d3cbce9db24e2e971a4",
"changelogLastFetched": 1781361371930,
"autoUpdatesProtectedForNative": true,
"claudeCodeFirstTokenDate": "2026-04-11T19:03:48.223040Z",
"hasCompletedOnboarding": true,
"lastOnboardingVersion": "2.1.165",
"groveConfigCache": {
"09792e21-2287-4348-b4d4-34cddbbfabc5": {
"grove_enabled": true,
"timestamp": 1781406640065
}
},
"cachedExperimentFeatures": [
"tengu_amber_prism",
"tengu_basalt_spur",
"tengu_cedar_inlet",
"tengu_coral_beacon",
"tengu_flint_harbor",
"tengu_mcp_subagent_prompt",
"tengu_ochre_hollow",
"tengu_orchid_mantis_v2",
"tengu_plank_river_frost",
"tengu_read_dedup_killswitch"
],
"cachedGrowthBookFeaturesAt": 1781447286954,
"lastReleaseNotesSeen": "2.1.177",
"projects": {
"/root": {
"allowedTools": [],
"mcpContextUris": [],
"mcpServers": {},
"enabledMcpjsonServers": [],
"disabledMcpjsonServers": [],
"hasTrustDialogAccepted": false,
"projectOnboardingSeenCount": 3,
"hasClaudeMdExternalIncludesApproved": false,
"hasClaudeMdExternalIncludesWarningShown": false,
"exampleFiles": [],
"lastGracefulShutdown": false,
"lastVersionBase": "2.1.177",
"lastCost": 22.33646404999996,
"lastAPIDuration": 5145196,
"lastAPIDurationWithoutRetries": 5144336,
"lastToolDuration": 506403,
"lastDuration": 11598574,
"lastLinesAdded": 652,
"lastLinesRemoved": 392,
"lastTotalInputTokens": 32237,
"lastTotalOutputTokens": 290269,
"lastTotalCacheCreationInputTokens": 1509902,
"lastTotalCacheReadInputTokens": 44894244,
"lastTotalWebSearchRequests": 0,
"lastFpsAverage": 6.03,
"lastFpsLow1Pct": 451.66,
"lastModelUsage": {
"claude-haiku-4-5-20251001": {
"inputTokens": 21206,
"outputTokens": 30015,
"cacheReadInputTokens": 2624502,
"cacheCreationInputTokens": 791709,
"webSearchRequests": 0,
"costUSD": 1.4233674499999998
},
"claude-sonnet-4-6": {
"inputTokens": 11031,
"outputTokens": 260254,
"cacheReadInputTokens": 42269742,
"cacheCreationInputTokens": 718193,
"webSearchRequests": 0,
"costUSD": 20.913096599999978
}
},
"lastSessionId": "685e6c5b-62c1-40bd-9cfd-2c9f7e15c50f",
"lastSessionMetrics": {
"frame_duration_ms_count": 69975,
"frame_duration_ms_min": 0.06756199989467859,
"frame_duration_ms_max": 100.53212600015104,
"frame_duration_ms_avg": 0.663704399986351,
"frame_duration_ms_p50": 0.4998550007585436,
"frame_duration_ms_p95": 1.5134988494683035,
"frame_duration_ms_p99": 2.4773533696774384,
"pre_tool_hook_duration_ms_count": 655,
"pre_tool_hook_duration_ms_min": 0,
"pre_tool_hook_duration_ms_max": 12,
"pre_tool_hook_duration_ms_avg": 0.1267175572519084,
"pre_tool_hook_duration_ms_p50": 0,
"pre_tool_hook_duration_ms_p95": 1,
"pre_tool_hook_duration_ms_p99": 1,
"hook_duration_ms_count": 465,
"hook_duration_ms_min": 0,
"hook_duration_ms_max": 22,
"hook_duration_ms_avg": 0.3204301075268817,
"hook_duration_ms_p50": 0,
"hook_duration_ms_p95": 1,
"hook_duration_ms_p99": 8.360000000000014
},
"hasCompletedProjectOnboarding": true
}
},
"routineFiredWatermark": "2026-06-05T19:47:09.178Z",
"penguinModeOrgEnabled": true,
"closedIssuesLastChecked": 1781406639965,
"passesEligibilityCache": {
"4bb43199-0efc-4d5c-b552-79865cb0361b": {
"eligible": true,
"referral_code_details": {
"code": "BeGGjphr1g",
"campaign": "claude_code_guest_pass_a47c",
"referral_link": "https://claude.ai/referral/BeGGjphr1g"
},
"referrer_reward": {
"amount_minor_units": 1000,
"currency": "USD"
},
"remaining_passes": 3,
"limit": 3,
"share_link": "https://claude.ai/referral/BeGGjphr1g",
"terms_url": "https://support.claude.com/en/articles/12875061-claude-code-guest-passes",
"timestamp": 1781406640514
}
},
"cachedExtraUsageDisabledReason": null,
"passesUpsellSeenCount": 3,
"hasVisitedPasses": false,
"passesLastSeenRemaining": 3,
"officialMarketplaceAutoInstallAttempted": true,
"officialMarketplaceAutoInstalled": true,
"tipLifetimeShownCounts": {
"fotw-campaign-upsell": 6,
"new-user-warmup": 2,
"plan-mode-for-complex-tasks": 5,
"memory-command": 2,
"theme-command": 2,
"colorterm-truecolor": 1,
"status-line": 1,
"prompt-queue": 3,
"enter-to-steer-in-relatime": 2,
"todo-list": 2,
"ide-upsell-external-terminal": 5,
"install-github-app": 3,
"install-slack-app": 3,
"drag-and-drop-images": 2,
"double-esc-code-restore": 2,
"continue": 2,
"shift-tab": 2,
"image-paste": 1,
"web-app": 2,
"color-when-multi-clauding": 1,
"custom-agents": 2,
"remote-control": 2,
"voice-mode": 2,
"goal-command-nudge": 4,
"guest-passes": 6,
"feedback-command": 2,
"frontend-design-plugin": 1,
"permissions": 2,
"rename-conversation": 1,
"custom-commands": 1,
"c4e-remote-sessions": 1,
"subagent-fanout-nudge": 1,
"no-flicker": 1
},
"feedbackSurveyState": {
"lastShownTime": 1781411703066
},
"hasUsedBackslashReturn": true,
"agentLastUsed": {
"bg": 1780696781055
},
"remoteControlUpsellSeenCount": 3,
"fullscreenUpsellSeenCount": 3,
"lastShownEmergencyTip": "Claude Fable 5 is currently unavailable. Please use Opus 4.8 or another available model. Learn more: https://www.anthropic.com/news/fable-mythos-access",
"oauthAccount": {
"accountUuid": "09792e21-2287-4348-b4d4-34cddbbfabc5",
"emailAddress": "gmer4lfe@gmail.com",
"organizationUuid": "4bb43199-0efc-4d5c-b552-79865cb0361b",
"hasExtraUsageEnabled": true,
"billingType": "stripe_subscription",
"accountCreatedAt": "2026-04-03T21:52:35.642439Z",
"subscriptionCreatedAt": "2026-04-11T13:14:49.905923Z",
"ccOnboardingFlags": {},
"claudeCodeTrialEndsAt": null,
"claudeCodeTrialDurationDays": null,
"seatTier": null,
"displayName": "Gmer4Lfe",
"organizationRole": "admin",
"workspaceRole": null,
"organizationName": "gmer4lfe@gmail.com's Organization",
"organizationType": "claude_pro",
"organizationRateLimitTier": "default_claude_ai",
"userRateLimitTier": null
},
"clientDataCache": {
"cedar_lagoon": {
"claude-fable": true,
"claude-mythos": true
},
"pewter_owl_tool": true,
"pewter_owl_model": "claude-fable"
},
"additionalModelOptionsCache": [
{
"value": "claude-fable-5[1m]",
"label": "Fable (disabled)",
"description": "Claude Fable 5 is currently unavailable. Learn more: https://www.anthropic.com/news/fable-mythos-access",
"disabled": true
}
],
"additionalModelCostsCache": {}
}
-4385
View File
File diff suppressed because it is too large Load Diff
-6
View File
@@ -1,6 +0,0 @@
{
"proto": 1,
"supervisorPid": 64919,
"updatedAt": 1780720722982,
"workers": {}
}
@@ -1,505 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ========================== HOSTN CONFIGURATION — (hostname) ==================================
# ==============================================================================================
# HOSTN-specific variables — credentials, container names, share paths, failover lists.
# Sourced after master.conf — values here extend shared profile arrays and add HOSTN-specific
# identity, credentials, and container configuration.
#
# Sparse checkout (git) ensures other hosts never receive this file.
#
# DO NOT put shared config here — thresholds, toggles, profiles belong in master.conf.
# DO NOT put other hosts' variables here — they belong in their own host*.conf files.
#
# ── HOW TO USE THIS TEMPLATE ──────────────────────────────────────────────────────────────────
# This file was generated by the Varaverk first-run wizard.
# Fill in the sections that apply to your setup — leave unused sections empty.
# All scripts self-guard against empty values — safe to leave sections blank until needed.
#
# ── INDEX ─────────────────────────────────────────────────────────────────────────────────────
#
# ── IDENTITY & CONNECTIVITY ────────────────────────────────────────────────────────────────
# IDENTITY hostname, SSH key, Unraid API key
# EMBY container name, URL, API key
# JELLYFIN container name, URL, API key
# GITEA API token for SSH key registration
# NOTIFICATIONS Discord webhook
#
# ── PARTNERSHIP ────────────────────────────────────────────────────────────────────────────
# PARTNERSHIP auth containers, backup paths, emby provisioning
#
# ── RSYNC ──────────────────────────────────────────────────────────────────────────────────
# DAILY SYNC SHARES media shares this host owns and pushes
# PERSONAL SHARES private encrypted shares for offsite backup
# WEEKLY SYNC SHARES appdata shares synced weekly
# INTERMEDIATE SYNC mid-day appdata propagation
# CRITICAL SYNC SHARES appdata shares synced every 30 minutes
# BACKUP VERIFY shares for checksum verification against remote
# HOSTN RSYNC PROFILE host-specific appdata sync profile
#
# ── FALLBACK ───────────────────────────────────────────────────────────────────────────────
# DDNS DDNS containers managed by this host
# INTERNET LOSS containers stopped when internet is lost
# FALLBACK TIERS what this host runs for the remote per tier
# TIER DELAYS delays before each tier activates
# RSYNC WRITEBACK appdata synced back on handback
#
# ── DOCKER ─────────────────────────────────────────────────────────────────────────────────
# DOCKER DAILY RESTART containers restarted daily
# DOCKER WEEKLY RESTART containers restarted weekly
# DOCKER WATCHDOG memory limits, health URLs, required containers
# NETWORK WATCHDOG DDNS domain, NPM URL for connectivity checks
# DOCKER NETWORK CONNECT networks and containers for array start
#
# ── MEDIA ──────────────────────────────────────────────────────────────────────────────────
# MEDIA PERMISSIONS share list for permissions script
# MEDIA CLEANER folder lists for media_cleaner.sh
#
# ── ARR STACK ──────────────────────────────────────────────────────────────────────────────
# DOWNLOADERS slskd, SABnzbd, qBittorrent credentials and URLs
# LIDARR / SONARR / RADARR URL, API key, path map
# ARR RECOVERY per-arr recovery toggles
#
# ── TRANSCODES ─────────────────────────────────────────────────────────────────────────────
# TRANSCODES ramdisk size, thresholds, SSD path, server array
#
# ── MONITORS ───────────────────────────────────────────────────────────────────────────────
# CERTIFICATE MONITOR domains checked for SSL expiry
# SMART HEALTH drives to skip in SMART monitoring
# ZFS REPORT pools to exclude from ZFS health report
#
# ── RESOURCE MANAGER ───────────────────────────────────────────────────────────────────────
# RESOURCE MANAGER containers paused/stopped under memory pressure
#
# ── SYSTEM WATCHDOG ────────────────────────────────────────────────────────────────────────
# SYSTEM WATCHDOG per-host check toggles and NIC configuration
#
# ==============================================================================================
# ==============================================================================================
# ── STORAGE MODE ──────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Storage mode ━━━
# Controls where Varaverk stores scripts, conf, and state files.
# true = internal NVMe/SSD — /boot/config/plugins/varaverk (write-safe, git-direct)
# false = USB flash boot — /mnt/user/appdata/Varaverk (preserves flash lifetime)
# Auto-detected from boot device transport on first setup.
# To change: Settings → Storage → Migrate.
HOSTN_STORAGE_MODE_INTERNAL=true
# ==============================================================================================
# ── IDENTITY & CONNECTIVITY ───────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Identity ━━━
# HOSTN hostname lives in master.conf (not a credential — safe for all servers).
# SSH key used for all server-to-server operations — rsync, failover, conf sync.
# Convention: /root/.ssh/<hostname-lowercase-no-unraid-prefix>_rsync_automation
# Must be in /root/.ssh/ and authorised in the partner's /root/.ssh/authorized_keys.
# Run Partnership/ssh_setup.sh to generate the key and copy it to the partner.
HOSTN_SSH_KEY="" # e.g. /root/.ssh/myserver_rsync_automation
HOSTN_OWNER="" # short identifier for this server (e.g. myserver)
HOSTN_OWNER_EMAIL=""
# ━━━ Unraid API ━━━
# Used by the Varaverk plugin to query this server's Unraid GraphQL API.
# Generate in Unraid: Settings → Management Access → API Keys → + New Key
HOSTN_UNRAID_API_KEY=""
# ━━━ Emby ━━━
HOSTN_EMBY_CONTAINER="Emby"
HOSTN_EMBY_URL="http://localhost:8096"
HOSTN_EMBY_API_KEY="" # Emby Dashboard → API Keys → + New Key
# ━━━ Jellyfin ━━━
HOSTN_JELLYFIN_CONTAINER="Jellyfin"
HOSTN_JELLYFIN_URL="http://localhost:8095"
HOSTN_JELLYFIN_API_KEY="" # Jellyfin Dashboard → Administration → API Keys
# ━━━ Gitea ━━━
# Personal access token for gitea_ssh_setup.sh.
# Create in Gitea: Settings → Applications → Generate Token → scope: write:user
HOSTN_GITEA_API_TOKEN=""
# ━━━ Notifications ━━━
# Discord webhook — leave blank to disable.
HOSTN_DISCORD_WEBHOOK=""
# ==============================================================================================
# ── PARTNERSHIP ───────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Auth containers reconfigured on onboard/offboard.
# Format: "ContainerName|WebUIPort"
HOSTN_PARTNERSHIP_AUTH_WEBUIS=(
# "NginxProxyManager|81"
# "Authelia|9091"
)
# XML templates pushed to mirror during onboard — auth stack.
# Dependencies (databases) must come before apps that depend on them.
HOSTN_PARTNERSHIP_AUTH_STACK=(
# "my-Authelia.xml"
# "my-NginxProxyManager.xml"
)
# XML templates pushed to mirror during onboard — arr stack.
HOSTN_PARTNERSHIP_ARR_STACK=(
# "my-Sonarr.xml"
# "my-Radarr.xml"
)
# Paths the partner should collect during the grace window after offboard.
HOSTN_PARTNERSHIP_MIRROR_BACKUPS=(
# "/mnt/user/appdata-Fallback/Partner-Emby"
)
# Containers parked on this server when partnership is active.
HOSTN_PARTNERSHIP_OWN_CONTAINERS=(
# "Emby"
)
# Containers stopped on THIS server before deploying the mirror's stack on onboard.
HOSTN_PARTNERSHIP_REPLACE_CONTAINERS=(
)
# Arr containers stopped on this server when mirror's arr stack is deployed.
HOSTN_PARTNERSHIP_ARR_REPLACE_CONTAINERS=(
)
# Emby admin provisioning — owner controls whether Emby is shared.
HOSTN_PARTNERSHIP_PROVISION_EMBY_ADMIN=false
HOSTN_PARTNERSHIP_EMBY_PORT=8096
HOSTN_PARTNERSHIP_EMBY_ADMIN_USER=""
HOSTN_PARTNERSHIP_EMBY_ADMIN_PASS=""
# ==============================================================================================
# ── RSYNC ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Daily Sync Shares ━━━
# Media shares this host pushes to all other nodes every night.
# Uses DEFAULT_RSYNC_OPTS from master.conf — no profile needed.
HOSTN_DAILY_SYNC_SHARES=(
# /mnt/user/Movies
# /mnt/user/Tv_Shows
# /mnt/user/Music
)
# ━━━ Personal Shares ━━━
# Private encrypted shares synced for offsite backup, independent of media shares.
HOSTN_PERSONAL_SHARES=(
# /mnt/user/Personal # e.g. ZFS-encrypted dataset
)
# ━━━ Weekly Sync Shares ━━━
# Appdata shares synced during the weekly maintenance window.
# Profiles (emby, critical-data) drive container stops — define in master.conf.
HOSTN_WEEKLY_SYNC_SHARES=(
# "/mnt/user/Media_Server/Emby" # emby profile
# "/mnt/user/appdata-Fallback/Critical-Data" # critical-data profile
)
# ━━━ Intermediate Sync Shares ━━━
# Shares synced every 4 hours. Leave empty to skip mid-day rsync.
HOSTN_INTERMEDIATE_SYNC_SHARES=(
# Add shares here to enable mid-day rsync
)
# ━━━ Critical Sync Shares ━━━
# Appdata shares synced every 30 minutes.
# Format: "/path/to/share" or "/path/to/share|profile-name"
HOSTN_CRITICAL_SYNC_SHARES=(
# "/mnt/user/appdata-Fallback/Critical-Data|critical-fallback"
# "/mnt/user/Media_Server/Emby|emby-fallback"
)
# ━━━ Backup Verify ━━━
# Leave empty to use HOSTN_DAILY_SYNC_SHARES automatically.
HOSTN_BACKUP_VERIFY_SHARES=(
# leave empty to use HOSTN_DAILY_SYNC_SHARES automatically
)
# ━━━ HOSTN Rsync Profile — hostn-appdata ━━━
# Host-specific appdata sync profile.
PROFILE_RSYNC_OPTS[hostn-appdata]="-av --info=progress2 --bwlimit=${PROFILE_BW_LIMIT[hostn-appdata]:-8000}"
PROFILE_BW_LIMIT[hostn-appdata]=8000
PROFILE_RETRY_COUNT[hostn-appdata]=3
PROFILE_SLEEP[hostn-appdata]=300
PROFILE_CRITICAL_CONTAINER_NAMES[hostn-appdata]=""
PROFILE_DELAYED_CONTAINERS[hostn-appdata]=""
PROFILE_CONTAINER_DELAY[hostn-appdata]=5
PROFILE_EXCLUDE_DIRS[hostn-appdata]="logs *.tmp"
# ==============================================================================================
# ── FALLBACK ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ DDNS ━━━
# DDNS containers this host manages.
HOSTN_DDNS_CONTAINERS=(
# "MyServer.com"
)
# ━━━ Internet Loss ━━━
# Containers stopped immediately when internet is lost.
FALLBACK_HOSTN_STOP_ON_NO_NET=(
# "MyServer.com"
)
# ━━━ Fallback Tiers — HOSTN Runs for Partner ━━━
# Containers this host starts when the partner goes down.
# Replace REMOTE_ID below with the actual remote host ID (HOST1, HOST2, etc.)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER1=(
# "Partner-DDNS-Container"
)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER2=(
# "container-placeholder"
)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER3=(
# "container-placeholder"
)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER4=(
# "container-placeholder"
)
# ━━━ Tier Delays — This Host's Outage Timers ━━━
# How long THIS host must be down before each tier activates on the partner.
HOSTN_TIER2_DELAY=240 # 4 hours
HOSTN_TIER3_DELAY=720 # 12 hours
HOSTN_TIER4_DELAY=1440 # 24 hours
# ━━━ Rsync Writeback ━━━
HOSTN_TIER1_WRITEBACK_DELAY=60 # skip Emby writeback if outage under 1hr
FALLBACK_HOSTN_WRITEBACK_TIER1=(
# "/mnt/user/Media_Server/Emby"
)
FALLBACK_HOSTN_WRITEBACK_TIER2=(
# "/mnt/user/appdata-Fallback/Important-Data"
)
FALLBACK_HOSTN_WRITEBACK_TIER3=(
# "location-placeholder"
)
FALLBACK_HOSTN_WRITEBACK_TIER4=(
# "/mnt/user/appdata-Fallback/Arrs_Stack"
)
# ==============================================================================================
# ── DOCKER ────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Docker Daily Restart ━━━
HOSTN_DAILY_RESTART_CONTAINERS=(
# "NginxProxyManager"
# "Authelia"
)
# ━━━ Docker Weekly Restart ━━━
HOSTN_WEEKLY_RESTART_CONTAINERS=(
# "NextCloud"
# "AdGuard-Home"
)
# ━━━ Docker Watchdog ━━━
# Memory hard limits in MB — immediate restart if exceeded.
# 20GB=20480 16GB=16384 8GB=8192 4GB=4096 2GB=2048 1GB=1024
declare -A HOSTN_WATCHDOG_CONTAINERS=(
# ["Emby"]=18432
)
# HTTP health check URLs — checked every cycle.
declare -A HOSTN_WATCHDOG_CONTAINER_URLS=(
# ["Emby"]="http://localhost:8096"
)
# API-level health checks. Format: ["ContainerName"]="url|expected_json_key|expected_value"
declare -A HOSTN_WATCHDOG_CONTAINER_API_CHECKS=(
)
# Required containers — must always be running.
HOSTN_WATCHDOG_REQUIRED_CONTAINERS=(
# "NginxProxyManager"
# "Authelia"
)
# Containers to skip in Tier 2 global scan.
HOSTN_WATCHDOG_SCAN_IGNORE=(
# "my-occasional-container"
)
# Dependency ordering — skip restarting a container if its dependency is also down.
declare -A HOSTN_WATCHDOG_DEPENDENCIES=(
# ["Authelia"]="Mariadb Redis-Authelia"
)
# Per-container appdata growth suppress ceilings in MB.
declare -A HOSTN_WATCHDOG_APPDATA_SIZES=(
# ["Tdarr"]="25600"
)
# ━━━ Network Watchdog ━━━
HOSTN_NETWORK_WATCHDOG_DDNS_DOMAIN="" # e.g. myserver.com
HOSTN_NETWORK_WATCHDOG_DDNS_CONTAINER="" # e.g. MyServer.com
HOSTN_NETWORK_WATCHDOG_NPM_URL="" # e.g. https://myserver.com
# ━━━ Docker Network Connect ━━━
HOSTN_NETWORK_CONNECT_CONTAINERS=(
# "memcached"
)
HOSTN_NETWORK_CONNECT_NETWORKS=(
# "high-availability"
)
# ==============================================================================================
# ── MEDIA ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Media Permissions ━━━
HOSTN_MEDIA_PERMISSION_SHARES=(
# /mnt/user/Movies
# /mnt/user/Tv_Shows
# /mnt/user/Music
# /mnt/user/Downloads
)
# ━━━ Media Cleaner ━━━
HOSTN_ANIME_CLEAN_FOLDERS=(
# /mnt/user/Anime_Movies
# /mnt/user/Anime_Shows
)
HOSTN_MEDIA_CLEAN_FOLDERS=(
# /mnt/user/Movies
# /mnt/user/Tv_Shows
)
# ==============================================================================================
# ── ARR STACK ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Downloaders ━━━
HOSTN_SLSKD_URL="http://localhost:8980"
HOSTN_SLSKD_API_KEY=""
HOSTN_SLSKD_FAILED_IMPORTS_DIR=""
HOSTN_SABNZBD_URL="http://localhost:8180"
HOSTN_SABNZBD_API_KEY=""
HOSTN_QBIT_URL="http://localhost:8080"
HOSTN_QBIT_USERNAME="admin"
HOSTN_QBIT_PASSWORD=""
# ━━━ Lidarr ━━━
HOSTN_LIDARR_URL="http://localhost:8686"
HOSTN_LIDARR_API_KEY=""
HOSTN_LIDARR_MUSIC_ROOT="/mnt/user/Music"
HOSTN_FANART_API_KEY=""
HOSTN_LASTFM_API_KEY=""
declare -A HOSTN_LIDARR_PATH_MAP=(
# ["/music"]="/mnt/user/Music"
)
# ━━━ Sonarr ━━━
HOSTN_SONARR_URL="http://localhost:8989"
HOSTN_SONARR_API_KEY=""
HOSTN_SONARR_TV_ROOT="/mnt/user/Tv_Shows"
declare -A HOSTN_SONARR_PATH_MAP=(
# ["/tv"]="/mnt/user/Tv_Shows"
)
# ━━━ Radarr ━━━
HOSTN_RADARR_URL="http://localhost:7878"
HOSTN_RADARR_API_KEY=""
HOSTN_TMDB_API_KEY=""
HOSTN_RADARR_MOVIES_ROOT="/mnt/user/Movies"
declare -A HOSTN_RADARR_PATH_MAP=(
# ["/movies"]="/mnt/user/Movies"
)
# ━━━ Arr Recovery Toggles ━━━
HOSTN_LIDARR_RECOVERY=false
HOSTN_SONARR_RECOVERY=true
HOSTN_RADARR_RECOVERY=true
# ==============================================================================================
# ── TRANSCODES ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOSTN_RAMDISK_SIZE="10G"
HOSTN_RAMDISK_WARN_GB=8.5
HOSTN_RAMDISK_LOW_GB=7
HOSTN_TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/"
HOSTN_TRANSCODE_SERVERS=(
"${HOSTN_EMBY_CONTAINER}|${HOSTN_EMBY_URL}|${HOSTN_EMBY_API_KEY}|emby"
)
# ==============================================================================================
# ── MONITORS ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
HOSTN_CERT_MONITOR_DOMAINS=(
# "myserver.com"
)
# ━━━ SMART Health ━━━
HOSTN_SMART_IGNORE_DRIVES=(
"sda" # boot USB — SMART not meaningful on flash drives
)
# ━━━ ZFS Report ━━━
HOSTN_ZFS_REPORT_IGNORE_POOLS=(
# "disk5"
)
# ==============================================================================================
# ── RESOURCE MANAGER ──────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOSTN_RW_PAUSE_CONTAINERS=(
# "Tdarr"
# "LidaTube"
)
HOSTN_RW_STOP_CONTAINERS=(
# "Tdarr"
)
# ==============================================================================================
# ── SYSTEM WATCHDOG ───────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOSTN_SYS_WATCHDOG_NIC="" # e.g. eth0 — for network monitoring
HOSTN_SYS_WATCHDOG_CHECK_DOCKER_DAEMON=true
HOSTN_SYS_WATCHDOG_CHECK_ROOTFS=true
HOSTN_SYS_WATCHDOG_CHECK_KERNEL_OOPS=true
HOSTN_SYS_WATCHDOG_CHECK_FD=true
HOSTN_SYS_WATCHDOG_CHECK_BOOT=true
HOSTN_SYS_WATCHDOG_CHECK_OOM=true
HOSTN_SYS_WATCHDOG_CHECK_RAM=true
HOSTN_SYS_WATCHDOG_CHECK_LOG=true
HOSTN_SYS_WATCHDOG_CHECK_ARC=true
HOSTN_SYS_WATCHDOG_CHECK_CPU_TEMP=true
HOSTN_SYS_WATCHDOG_CHECK_LOAD=true
HOSTN_SYS_WATCHDOG_CHECK_ZOMBIES=true
HOSTN_SYS_WATCHDOG_CHECK_CONTAINERS=true
HOSTN_SYS_WATCHDOG_CHECK_TMP=true
HOSTN_SYS_WATCHDOG_CHECK_MDSTAT=true
HOSTN_SYS_WATCHDOG_CHECK_NETWORK=true
HOSTN_SYS_WATCHDOG_CHECK_SSHD=true
HOSTN_SYS_WATCHDOG_CHECK_RUNAWAY=false
@@ -1,529 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ========================== HOSTN CONFIGURATION — (hostname) ==================================
# ==============================================================================================
# HOSTN-specific variables — credentials, container names, share paths, failover lists.
# Sourced after master.conf — values here extend shared profile arrays and add HOSTN-specific
# identity, credentials, and container configuration.
#
# Sparse checkout (git) ensures other hosts never receive this file.
#
# DO NOT put shared config here — thresholds, toggles, profiles belong in master.conf.
# DO NOT put other hosts' variables here — they belong in their own host*.conf files.
#
# ── HOW TO USE THIS TEMPLATE ──────────────────────────────────────────────────────────────────
# This file was generated by the Varaverk first-run wizard.
# Fill in the sections that apply to your setup — leave unused sections empty.
# All scripts self-guard against empty values — safe to leave sections blank until needed.
#
# ── INDEX ─────────────────────────────────────────────────────────────────────────────────────
#
# ── IDENTITY & CONNECTIVITY ────────────────────────────────────────────────────────────────
# IDENTITY hostname, SSH key, Unraid API key
# EMBY container name, URL, API key
# JELLYFIN container name, URL, API key
# GITEA API token for SSH key registration
# NOTIFICATIONS Discord webhook
#
# ── PARTNERSHIP ────────────────────────────────────────────────────────────────────────────
# PARTNERSHIP auth containers, backup paths, emby provisioning
#
# ── RSYNC ──────────────────────────────────────────────────────────────────────────────────
# DAILY SYNC SHARES media shares this host owns and pushes
# PERSONAL SHARES private encrypted shares for offsite backup
# WEEKLY SYNC SHARES appdata shares synced weekly
# INTERMEDIATE SYNC mid-day appdata propagation
# CRITICAL SYNC SHARES appdata shares synced every 30 minutes
# BACKUP VERIFY shares for checksum verification against remote
# HOSTN RSYNC PROFILE host-specific appdata sync profile
#
# ── FALLBACK ───────────────────────────────────────────────────────────────────────────────
# DDNS DDNS containers managed by this host
# INTERNET LOSS containers stopped when internet is lost
# FALLBACK TIERS what this host runs for the remote per tier
# TIER DELAYS delays before each tier activates
# RSYNC WRITEBACK appdata synced back on handback
#
# ── DOCKER ─────────────────────────────────────────────────────────────────────────────────
# DOCKER DAILY RESTART containers restarted daily
# DOCKER WEEKLY RESTART containers restarted weekly
# DOCKER WATCHDOG memory limits, health URLs, required containers
# NETWORK WATCHDOG DDNS domain, NPM URL for connectivity checks
# DOCKER NETWORK CONNECT networks and containers for array start
#
# ── MEDIA ──────────────────────────────────────────────────────────────────────────────────
# MEDIA PERMISSIONS share list for permissions script
# MEDIA CLEANER folder lists for media_cleaner.sh
#
# ── ARR STACK ──────────────────────────────────────────────────────────────────────────────
# DOWNLOADERS slskd, SABnzbd, qBittorrent credentials and URLs
# LIDARR / SONARR / RADARR URL, API key, path map
# ARR RECOVERY per-arr recovery toggles
#
# ── TRANSCODES ─────────────────────────────────────────────────────────────────────────────
# TRANSCODES ramdisk size, thresholds, SSD path, server array
#
# ── MONITORS ───────────────────────────────────────────────────────────────────────────────
# CERTIFICATE MONITOR domains checked for SSL expiry
# SMART HEALTH drives to skip in SMART monitoring
# ZFS REPORT pools to exclude from ZFS health report
#
# ── RESOURCE MANAGER ───────────────────────────────────────────────────────────────────────
# RESOURCE MANAGER containers paused/stopped under memory pressure
#
# ── SYSTEM WATCHDOG ────────────────────────────────────────────────────────────────────────
# SYSTEM WATCHDOG per-host check toggles and NIC configuration
#
# ==============================================================================================
# ==============================================================================================
# ── STORAGE MODE ──────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Storage mode ━━━
# Controls where Varaverk stores scripts, conf, and state files.
# true = internal NVMe/SSD — /boot/config/plugins/varaverk (write-safe, git-direct)
# false = USB flash boot — /mnt/user/appdata/Varaverk (preserves flash lifetime)
# Auto-detected from boot device transport on first setup.
# To change: Settings → Storage → Migrate.
HOSTN_STORAGE_MODE_INTERNAL=true
# ==============================================================================================
# ── IDENTITY & CONNECTIVITY ───────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Identity ━━━
# HOSTN hostname lives in master.conf (not a credential — safe for all servers).
# SSH key used for all server-to-server operations — rsync, failover, conf sync.
# Convention: /root/.ssh/<hostname-lowercase-no-unraid-prefix>_rsync_automation
# Must be in /root/.ssh/ and authorised in the partner's /root/.ssh/authorized_keys.
# Run Partnership/ssh_setup.sh to generate the key and copy it to the partner.
HOSTN_SSH_KEY="" # e.g. /root/.ssh/myserver_rsync_automation
HOSTN_OWNER="" # short identifier for this server (e.g. myserver)
HOSTN_OWNER_EMAIL=""
# ━━━ Unraid API ━━━
# Used by the Varaverk plugin to query this server's Unraid GraphQL API.
# Generate in Unraid: Settings → Management Access → API Keys → + New Key
HOSTN_UNRAID_API_KEY=""
# ━━━ Emby ━━━
HOSTN_EMBY_CONTAINER="Emby"
HOSTN_EMBY_URL="http://localhost:8096"
HOSTN_EMBY_API_KEY="" # Emby Dashboard → API Keys → + New Key
# ━━━ Jellyfin ━━━
HOSTN_JELLYFIN_CONTAINER="Jellyfin"
HOSTN_JELLYFIN_URL="http://localhost:8095"
HOSTN_JELLYFIN_API_KEY="" # Jellyfin Dashboard → Administration → API Keys
# ━━━ Gitea ━━━
# Personal access token for gitea_ssh_setup.sh.
# Create in Gitea: Settings → Applications → Generate Token → scope: write:user
HOSTN_GITEA_API_TOKEN=""
# ━━━ Notifications ━━━
# Discord webhook — leave blank to disable.
HOSTN_DISCORD_WEBHOOK=""
# ==============================================================================================
# ── PARTNERSHIP ───────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Auth containers reconfigured on onboard/offboard.
# Format: "ContainerName|WebUIPort"
HOSTN_PARTNERSHIP_AUTH_WEBUIS=(
# "NginxProxyManager|81"
# "Authelia|9091"
)
# XML templates pushed to mirror during onboard — auth stack.
# Dependencies (databases) must come before apps that depend on them.
HOSTN_PARTNERSHIP_AUTH_STACK=(
# "my-Authelia.xml"
# "my-NginxProxyManager.xml"
)
# XML templates pushed to mirror during onboard — arr stack.
HOSTN_PARTNERSHIP_ARR_STACK=(
# "my-Sonarr.xml"
# "my-Radarr.xml"
)
# Paths the partner should collect during the grace window after offboard.
HOSTN_PARTNERSHIP_MIRROR_BACKUPS=(
# "/mnt/user/appdata-Fallback/Partner-Emby"
)
# Containers parked on this server when partnership is active.
HOSTN_PARTNERSHIP_OWN_CONTAINERS=(
# "Emby"
)
# Containers stopped on THIS server before deploying the mirror's stack on onboard.
HOSTN_PARTNERSHIP_REPLACE_CONTAINERS=(
)
# Arr containers stopped on this server when mirror's arr stack is deployed.
HOSTN_PARTNERSHIP_ARR_REPLACE_CONTAINERS=(
)
# Emby admin provisioning — owner controls whether Emby is shared.
HOSTN_PARTNERSHIP_PROVISION_EMBY_ADMIN=false
HOSTN_PARTNERSHIP_EMBY_PORT=8096
HOSTN_PARTNERSHIP_EMBY_ADMIN_USER=""
HOSTN_PARTNERSHIP_EMBY_ADMIN_PASS=""
# ==============================================================================================
# ── RSYNC ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Daily Sync Shares ━━━
# Media shares this host pushes to all other nodes every night.
# Uses DEFAULT_RSYNC_OPTS from master.conf — no profile needed.
HOSTN_DAILY_SYNC_SHARES=(
# /mnt/user/Movies
# /mnt/user/Tv_Shows
# /mnt/user/Music
)
# ━━━ Personal Shares ━━━
# Private encrypted shares synced for offsite backup, independent of media shares.
HOSTN_PERSONAL_SHARES=(
# /mnt/user/Personal # e.g. ZFS-encrypted dataset
)
# ━━━ Weekly Sync Shares ━━━
# Appdata shares synced during the weekly maintenance window.
# Profiles (emby, critical-data) drive container stops — define in master.conf.
HOSTN_WEEKLY_SYNC_SHARES=(
# "/mnt/user/Media_Server/Emby" # emby profile
# "/mnt/user/appdata-Fallback/Critical-Data" # critical-data profile
)
# ━━━ Intermediate Sync Shares ━━━
# Shares synced every 4 hours. Leave empty to skip mid-day rsync.
HOSTN_INTERMEDIATE_SYNC_SHARES=(
# Add shares here to enable mid-day rsync
)
# ━━━ Critical Sync Shares ━━━
# Appdata shares synced every 30 minutes.
# Format: "/path/to/share" or "/path/to/share|profile-name"
HOSTN_CRITICAL_SYNC_SHARES=(
# "/mnt/user/appdata-Fallback/Critical-Data|critical-fallback"
# "/mnt/user/Media_Server/Emby|emby-fallback"
)
# ━━━ Backup Verify ━━━
# Leave empty to use HOSTN_DAILY_SYNC_SHARES automatically.
HOSTN_BACKUP_VERIFY_SHARES=(
# leave empty to use HOSTN_DAILY_SYNC_SHARES automatically
)
# ━━━ HOSTN Rsync Profile — hostn-appdata ━━━
# Host-specific appdata sync profile.
PROFILE_RSYNC_OPTS[hostn-appdata]="-av --info=progress2 --bwlimit=${PROFILE_BW_LIMIT[hostn-appdata]:-8000}"
PROFILE_BW_LIMIT[hostn-appdata]=8000
PROFILE_RETRY_COUNT[hostn-appdata]=3
PROFILE_SLEEP[hostn-appdata]=300
PROFILE_CRITICAL_CONTAINER_NAMES[hostn-appdata]=""
PROFILE_DELAYED_CONTAINERS[hostn-appdata]=""
PROFILE_CONTAINER_DELAY[hostn-appdata]=5
PROFILE_EXCLUDE_DIRS[hostn-appdata]="logs *.tmp"
# ==============================================================================================
# ── FALLBACK ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ DDNS ━━━
# DDNS containers this host manages.
HOSTN_DDNS_CONTAINERS=(
# "MyServer.com"
)
# ━━━ Internet Loss ━━━
# Containers stopped immediately when internet is lost.
FALLBACK_HOSTN_STOP_ON_NO_NET=(
# "MyServer.com"
)
# ━━━ Fallback Tiers — HOSTN Runs for Partner ━━━
# Containers this host starts when the partner goes down.
# Replace REMOTE_ID below with the actual remote host ID (HOST1, HOST2, etc.)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER1=(
# "Partner-DDNS-Container"
)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER2=(
# "container-placeholder"
)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER3=(
# "container-placeholder"
)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER4=(
# "container-placeholder"
)
# ━━━ Tier Delays — This Host's Outage Timers ━━━
# How long THIS host must be down before each tier activates on the partner.
HOSTN_TIER2_DELAY=240 # 4 hours
HOSTN_TIER3_DELAY=720 # 12 hours
HOSTN_TIER4_DELAY=1440 # 24 hours
# ━━━ Rsync Writeback ━━━
HOSTN_TIER1_WRITEBACK_DELAY=60 # skip Emby writeback if outage under 1hr
FALLBACK_HOSTN_WRITEBACK_TIER1=(
# "/mnt/user/Media_Server/Emby"
)
FALLBACK_HOSTN_WRITEBACK_TIER2=(
# "/mnt/user/appdata-Fallback/Important-Data"
)
FALLBACK_HOSTN_WRITEBACK_TIER3=(
# "location-placeholder"
)
FALLBACK_HOSTN_WRITEBACK_TIER4=(
# "/mnt/user/appdata-Fallback/Arrs_Stack"
)
# ==============================================================================================
# ── DOCKER ────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Docker Daily Restart ━━━
HOSTN_DAILY_RESTART_CONTAINERS=(
# "NginxProxyManager"
# "Authelia"
)
# ━━━ Docker Weekly Restart ━━━
HOSTN_WEEKLY_RESTART_CONTAINERS=(
# "NextCloud"
# "AdGuard-Home"
)
# ━━━ Docker Watchdog ━━━
# Memory hard limits in MB — immediate restart if exceeded.
# 20GB=20480 16GB=16384 8GB=8192 4GB=4096 2GB=2048 1GB=1024
declare -A HOSTN_WATCHDOG_CONTAINERS=(
# ["Emby"]=18432
)
# HTTP health check URLs — checked every cycle.
declare -A HOSTN_WATCHDOG_CONTAINER_URLS=(
# ["Emby"]="http://localhost:8096"
)
# API-level health checks. Format: ["ContainerName"]="url|expected_json_key|expected_value"
declare -A HOSTN_WATCHDOG_CONTAINER_API_CHECKS=(
)
# Required containers — must always be running.
HOSTN_WATCHDOG_REQUIRED_CONTAINERS=(
# "NginxProxyManager"
# "Authelia"
)
# Containers to skip in Tier 2 global scan.
HOSTN_WATCHDOG_SCAN_IGNORE=(
# "my-occasional-container"
)
# Dependency ordering — skip restarting a container if its dependency is also down.
declare -A HOSTN_WATCHDOG_DEPENDENCIES=(
# ["Authelia"]="Mariadb Redis-Authelia"
)
# Per-container appdata growth suppress ceilings in MB.
declare -A HOSTN_WATCHDOG_APPDATA_SIZES=(
# ["Tdarr"]="25600"
)
# ━━━ Network Watchdog ━━━
HOSTN_NETWORK_WATCHDOG_DDNS_DOMAIN="" # e.g. myserver.com
HOSTN_NETWORK_WATCHDOG_DDNS_CONTAINER="" # e.g. MyServer.com
HOSTN_NETWORK_WATCHDOG_NPM_URL="" # e.g. https://myserver.com
# ━━━ Docker Network Connect ━━━
HOSTN_NETWORK_CONNECT_CONTAINERS=(
# "memcached"
)
HOSTN_NETWORK_CONNECT_NETWORKS=(
# "high-availability"
)
# ==============================================================================================
# ── MEDIA ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Media Permissions ━━━
HOSTN_MEDIA_PERMISSION_SHARES=(
# /mnt/user/Movies
# /mnt/user/Tv_Shows
# /mnt/user/Music
# /mnt/user/Downloads
)
# ━━━ Media Cleaner ━━━
HOSTN_ANIME_CLEAN_FOLDERS=(
# /mnt/user/Anime_Movies
# /mnt/user/Anime_Shows
)
HOSTN_MEDIA_CLEAN_FOLDERS=(
# /mnt/user/Movies
# /mnt/user/Tv_Shows
)
# ==============================================================================================
# ── ARR STACK ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Downloaders ━━━
HOSTN_SLSKD_URL="http://localhost:8980"
HOSTN_SLSKD_API_KEY=""
HOSTN_SLSKD_FAILED_IMPORTS_DIR=""
HOSTN_SABNZBD_URL="http://localhost:8180"
HOSTN_SABNZBD_API_KEY=""
HOSTN_QBIT_URL="http://localhost:8080"
HOSTN_QBIT_USERNAME="admin"
HOSTN_QBIT_PASSWORD=""
# ━━━ Lidarr ━━━
HOSTN_LIDARR_URL="http://localhost:8686"
HOSTN_LIDARR_API_KEY=""
HOSTN_LIDARR_MUSIC_ROOT="/mnt/user/Music"
HOSTN_FANART_API_KEY=""
HOSTN_LASTFM_API_KEY=""
declare -A HOSTN_LIDARR_PATH_MAP=(
# ["/music"]="/mnt/user/Music"
)
# ━━━ Sonarr ━━━
HOSTN_SONARR_URL="http://localhost:8989"
HOSTN_SONARR_API_KEY=""
HOSTN_SONARR_TV_ROOT="/mnt/user/Tv_Shows"
declare -A HOSTN_SONARR_PATH_MAP=(
# ["/tv"]="/mnt/user/Tv_Shows"
)
# ━━━ Radarr ━━━
HOSTN_RADARR_URL="http://localhost:7878"
HOSTN_RADARR_API_KEY=""
HOSTN_TMDB_API_KEY=""
HOSTN_RADARR_MOVIES_ROOT="/mnt/user/Movies"
declare -A HOSTN_RADARR_PATH_MAP=(
# ["/movies"]="/mnt/user/Movies"
)
# ━━━ Arr Recovery Toggles ━━━
HOSTN_LIDARR_RECOVERY=false
HOSTN_SONARR_RECOVERY=true
HOSTN_RADARR_RECOVERY=true
# ==============================================================================================
# ── TRANSCODES ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOSTN_RAMDISK_SIZE="10G"
HOSTN_RAMDISK_WARN_GB=8.5
HOSTN_RAMDISK_LOW_GB=7
HOSTN_TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/"
HOSTN_TRANSCODE_SERVERS=(
"${HOSTN_EMBY_CONTAINER}|${HOSTN_EMBY_URL}|${HOSTN_EMBY_API_KEY}|emby"
)
# ==============================================================================================
# ── MONITORS ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
HOSTN_CERT_MONITOR_DOMAINS=(
# "myserver.com"
)
# ━━━ SMART Health ━━━
HOSTN_SMART_IGNORE_DRIVES=(
"sda" # boot USB — SMART not meaningful on flash drives
)
# ━━━ ZFS Report ━━━
HOSTN_ZFS_REPORT_IGNORE_POOLS=(
# "disk5"
)
# ==============================================================================================
# ── RESOURCE MANAGER ──────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOSTN_RW_PAUSE_CONTAINERS=(
# "Tdarr"
# "LidaTube"
)
HOSTN_RW_STOP_CONTAINERS=(
# "Tdarr"
)
# ==============================================================================================
# ── SYSTEM WATCHDOG ───────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOSTN_SYS_WATCHDOG_NIC="" # e.g. eth0 — for network monitoring
HOSTN_SYS_WATCHDOG_CHECK_DOCKER_DAEMON=true
HOSTN_SYS_WATCHDOG_CHECK_ROOTFS=true
HOSTN_SYS_WATCHDOG_CHECK_KERNEL_OOPS=true
HOSTN_SYS_WATCHDOG_CHECK_FD=true
HOSTN_SYS_WATCHDOG_CHECK_BOOT=true
HOSTN_SYS_WATCHDOG_CHECK_OOM=true
HOSTN_SYS_WATCHDOG_CHECK_RAM=true
HOSTN_SYS_WATCHDOG_CHECK_LOG=true
HOSTN_SYS_WATCHDOG_CHECK_ARC=true
HOSTN_SYS_WATCHDOG_CHECK_CPU_TEMP=true
HOSTN_SYS_WATCHDOG_CHECK_LOAD=true
HOSTN_SYS_WATCHDOG_CHECK_ZOMBIES=true
HOSTN_SYS_WATCHDOG_CHECK_CONTAINERS=true
HOSTN_SYS_WATCHDOG_CHECK_TMP=true
HOSTN_SYS_WATCHDOG_CHECK_MDSTAT=true
HOSTN_SYS_WATCHDOG_CHECK_NETWORK=true
HOSTN_SYS_WATCHDOG_CHECK_SSHD=true
HOSTN_SYS_WATCHDOG_CHECK_RUNAWAY=false
# ==============================================================================================
# ── AUTH STACK ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Credentials for the Varaverk Auth Stack page (NPM, lldap, Authelia).
# ━━━ NginxProxyManager ━━━
# Admin API runs on 7818 (not 81 — 81 is the partnership WebUI port).
HOSTN_NPM_URL="http://localhost:7818"
HOSTN_NPM_USER="" # NPM admin email
HOSTN_NPM_PASS="" # NPM admin password
# ━━━ lldap ━━━
HOSTN_LLDAP_URL="http://localhost:17170"
HOSTN_LLDAP_USER="admin" # lldap admin username
HOSTN_LLDAP_PASS="" # lldap admin password
# ━━━ Authelia ━━━
HOSTN_AUTHELIA_CONFIG="/mnt/user/appdata/Authelia/configuration.yml"
HOSTN_AUTHELIA_CONTAINER="Authelia"
# ==============================================================================================
# ──────────────────────── End Of HOSTn Variables ──────────────────────────────────────────────
# ==============================================================================================
@@ -1,807 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ========================== HOST1 CONFIGURATION — unRAID-Gmer4Lfe ============================
# ==============================================================================================
# HOST1-specific variables — credentials, container names, share paths, failover lists.
# Sourced after master.conf — values here extend shared profile arrays and add HOST1-specific
# identity, credentials, and container configuration.
#
# Sparse checkout (git) ensures HOST2 never receives this file.
# HOST2 never sees HOST1 credentials — clean separation at the file level.
#
# DO NOT put shared config here — thresholds, toggles, profiles belong in master.conf.
# DO NOT put HOST2 variables here — they belong in host2.conf.
#
# ── INDEX ─────────────────────────────────────────────────────────────────────────────────────
#
# ── IDENTITY & CONNECTIVITY ────────────────────────────────────────────────────────────────
# IDENTITY hostname, SSH key, Unraid API key
# EMBY container name, URL, API key
# JELLYFIN container name, URL, API key
# GITEA API token for SSH key registration
# NOTIFICATIONS Discord webhook
#
# ── PARTNERSHIP ────────────────────────────────────────────────────────────────────────────
# PARTNERSHIP auth containers, backup paths, emby provisioning
#
# ── RSYNC ──────────────────────────────────────────────────────────────────────────────────
# DAILY SYNC SHARES media shares HOST1 owns and pushes to HOST2
# WEEKLY SYNC SHARES appdata shares synced weekly (Sunday 2:30am)
# CRITICAL SYNC SHARES appdata shares synced every 30 minutes
# BACKUP VERIFY shares for checksum verification against remote
# HOST1 RSYNC PROFILE host1-appdata profile for HOST1-specific appdata syncs
#
# ── FALLBACK ───────────────────────────────────────────────────────────────────────────────
# DDNS DDNS containers managed by HOST1
# INTERNET LOSS containers stopped when internet is lost
# FALLBACK TIERS what HOST1 runs for HOST2 per tier
# TIER DELAYS how long HOST1 must be down before each tier activates on HOST2
# RSYNC WRITEBACK HOST1 appdata synced back on handback
#
# ── DOCKER ─────────────────────────────────────────────────────────────────────────────────
# DOCKER DAILY RESTART containers restarted daily
# DOCKER WEEKLY RESTART containers restarted weekly
# DOCKER WATCHDOG memory limits, health URLs, required containers, ignore list
# NETWORK WATCHDOG DDNS domain, NPM URL for connectivity checks
# DOCKER NETWORK CONNECT networks and containers for docker_network_connect.sh
#
# ── MEDIA ──────────────────────────────────────────────────────────────────────────────────
# MEDIA PERMISSIONS share list for media_shares_permissions.sh
# MEDIA CLEANER folder lists for media_cleaner.sh
#
# ── ARR STACK ──────────────────────────────────────────────────────────────────────────────
# DOWNLOADERS slskd, SABnzbd, qBittorrent credentials and URLs
# LIDARR URL, API key, path map
# SONARR URL, API key, path map
# RADARR URL, API key, path map
# ARR RECOVERY per-arr recovery toggles
#
# ── TRANSCODES ─────────────────────────────────────────────────────────────────────────────
# TRANSCODES ramdisk size, thresholds, SSD path, server array
#
# ── MONITORS ───────────────────────────────────────────────────────────────────────────────
# CERTIFICATE MONITOR domains checked for SSL expiry
# SMART HEALTH drives to skip in SMART monitoring
# ZFS REPORT pools to exclude from ZFS health report
#
# ── RESOURCE MANAGER ───────────────────────────────────────────────────────────────────────
# RESOURCE MANAGER containers paused/stopped under memory pressure
#
# ── SYSTEM WATCHDOG ────────────────────────────────────────────────────────────────────────
# SYSTEM WATCHDOG per-host check toggles and NIC configuration
#
# ==============================================================================================
# ==============================================================================================
# ── STORAGE MODE ──────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Storage mode ━━━
# Controls where Varaverk stores scripts, conf, and state files.
# true = internal NVMe/SSD — /boot/config/plugins/varaverk (write-safe, git-direct)
# false = USB flash boot — /mnt/user/appdata/Varaverk (preserves flash lifetime)
HOST1_STORAGE_MODE_INTERNAL=true
# ==============================================================================================
# ── IDENTITY & CONNECTIVITY ───────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Identity ━━━
# HOST1 hostname lives in master.conf (not a credential — safe for all servers).
# SSH key used for all server-to-server operations — rsync, failover container commands.
# Must be in /root/.ssh/ and authorised in HOST2's /root/.ssh/authorized_keys.
HOST1_SSH_KEY="/root/.ssh/gmer4lfe_rsync_automation"
HOST1_OWNER="gmer4lfe"
HOST1_OWNER_EMAIL="gmer4lfe@gmail.com"
# ━━━ Unraid API ━━━
# Used by the Varaverk plugin to query this server's Unraid GraphQL API.
# Generate in Unraid: Settings → Management Access → API Keys → + New Key
HOST1_UNRAID_API_KEY="1825c3a2e03ea5089974f4da2e171aa2d5907a1dea23cc479c33e492c8ff4dbb"
# ━━━ Emby ━━━
# Referenced by transcode_manager.sh, emby_session_report.sh, emby_database_repair.sh,
# weekly_sync_maintenance.sh, and HOST1_TRANSCODE_SERVERS below.
# API key: Emby Dashboard → API Keys → + New Key
HOST1_EMBY_CONTAINER="Emby"
HOST1_EMBY_URL="http://localhost:8096"
HOST1_EMBY_API_KEY="0c27448d93a7431f9ac63569f7655829"
# ━━━ Jellyfin ━━━
# API key: Jellyfin Dashboard → Administration → API Keys → + New Key
HOST1_JELLYFIN_CONTAINER="Jellyfin"
HOST1_JELLYFIN_URL="http://localhost:8095"
HOST1_JELLYFIN_API_KEY="4e820e7df74c4933acec212b1996314e"
# ━━━ Gitea ━━━
# Personal access token for gitea_ssh_setup.sh — registers this server's SSH public key
# with Gitea so git operations use key auth instead of passwords.
# Create in Gitea: Settings → Applications → Generate Token → scope: write:user
HOST1_GITEA_API_TOKEN=""
# ━━━ Notifications ━━━
# Discord webhook — leave blank to disable.
# Per-host so HOST1 and HOST2 can post to different channels or only one server notifies.
HOST1_DISCORD_WEBHOOK=""
# ==============================================================================================
# ── PARTNERSHIP ───────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# HOST1 is always the owner (source of truth) unless --transfer has been run.
# See README-Partnership.md and master.conf PARTNERSHIP section for full lifecycle docs.
# Auth containers reconfigured on onboard/offboard.
# Format: "ContainerName|WebUIPort"
# On onboard → WebUI pointed at owner's Tailscale IP (mirror clicks NPM, gets owner's auth)
# On offboard → WebUI pointed back at localhost
HOST1_PARTNERSHIP_AUTH_WEBUIS=(
"NginxProxyManager|81"
"Lldap-Gmer4Lfe|17170"
"Authelia|9091"
"Authelia-Secondary|9092"
)
# XML templates (from this server's templates-user/) pushed to mirror during onboard.
# These become the mirror's active auth stack, backed by the rsync-synced appdata.
# Update filename if Lldap is renamed to drop the host suffix.
HOST1_PARTNERSHIP_AUTH_STACK=(
# Dependencies first — Mariadb/Redis must be healthy before Authelia starts
"my-Mariadb-Authelia.xml"
"my-Mariadb-Authelia-Secondary.xml"
"my-Redis-Authelia.xml"
"my-Redis-Authelia-Secondary.xml"
# Auth apps — deployed after their deps are confirmed healthy
"my-Authelia.xml"
"my-Authelia-Secondary.xml"
"my-NginxProxyManager.xml"
"my-Lldap-Gmer4Lfe.xml"
# Source of truth — must be available on HOST2 independently of the auth stack
"my-Gitea.xml"
)
# XML templates pushed to mirror for the arr stack during onboard.
# Deps (e.g. databases) first if any — same ordering rule as auth stack.
HOST1_PARTNERSHIP_ARR_STACK=(
# "my-Sonarr.xml"
# "my-Radarr.xml"
# "my-Lidarr.xml"
# "my-Prowlarr.xml"
# "my-Bazarr.xml"
)
# Containers stopped on THIS server before deploying the mirror's stack on onboard.
# Only needed when this server parks its own stack to make room for the mirror's.
HOST1_PARTNERSHIP_REPLACE_CONTAINERS=(
)
# Arr containers stopped on this server when mirror's arr stack is deployed.
HOST1_PARTNERSHIP_ARR_REPLACE_CONTAINERS=(
)
# Paths HOST2 should collect during the grace window after offboard.
# Notified on offboard — no auto-deletion, HOST2 must collect manually within PARTNERSHIP_GRACE_HOURS.
HOST1_PARTNERSHIP_MIRROR_BACKUPS=(
# "/mnt/user/appdata-Fallback/Jayred365-Emby"
)
# Containers parked on this server when partnership is active.
# Stopped on onboard (owner deploys its stack instead), restarted on offboard.
HOST1_PARTNERSHIP_OWN_CONTAINERS=(
# "Emby"
# "NginxProxyManager"
)
# Emby admin provisioning — toggle is owner-only, credentials are per-host.
# Owner enables/disables the feature. Each host sets the account they want on the shared Emby.
# On onboard: owner reads mirror's HOST*_PARTNERSHIP_EMBY_ADMIN_* and creates that account.
# On offboard: account is deleted. Username collision → onboard exits with error.
HOST1_PARTNERSHIP_PROVISION_EMBY_ADMIN=false # owner controls whether Emby is shared
HOST1_PARTNERSHIP_EMBY_PORT=8096
HOST1_PARTNERSHIP_EMBY_ADMIN_USER="" # this server's desired Emby username
HOST1_PARTNERSHIP_EMBY_ADMIN_PASS="" # this server's desired Emby password
# ==============================================================================================
# ── RSYNC ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Daily Sync Shares ━━━
# Shares HOST1 pushes to all other nodes every night (1am via daily_sync_maintenance.sh).
# Mesh model: every node pushes every media share — no ownership, no mirrors.
# arr_sync ensures all arr libraries converge (union). rsync spreads files (additive, no --delete).
# arr_cleanup removes true orphans based on local arr state.
# Any node can download content to any share — it propagates to all nodes on the next cycle.
# Nextcloud is intentionally one-directional (HOST1→HOST2 offsite backup — not arr-managed).
# Uses DEFAULT_RSYNC_OPTS from master.conf — no profile needed.
# For shares needing container stops or custom options — add a profile in master.conf.
HOST1_DAILY_SYNC_SHARES=(
/mnt/user/Books
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Nextcloud
/mnt/user/stand-up_comedy
/mnt/user/Sports
# /mnt/user/Tv_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Movies
/mnt/user/Anime_Shows
)
# Personal encrypted shares — synced for offsite backup, independent of media shares.
# ZFS encrypted at dataset level — remote receives encrypted blocks, cannot read content.
# See README-Rsync_Setup.md for ZFS encryption setup before uncommenting.
HOST1_PERSONAL_SHARES=(
# /mnt/user/HOST1-Personal # uncomment after creating encrypted dataset
)
# ━━━ Weekly Sync Shares ━━━
# Appdata shares synced during the weekly maintenance window (Sunday 2:30am).
# Containers stopped both sides before sync — full clean state guaranteed.
# Profiles drive container stops, excludes, and options — configured in master.conf RSYNC section.
# Order matters — Emby first (larger transfer), then Critical-Data (auth stack).
HOST1_WEEKLY_SYNC_SHARES=(
"/mnt/user/Media_Server/Emby" # emby profile — full clean mirror
"/mnt/user/appdata-Fallback/Critical-Data" # critical-data profile — auth stack
)
# ━━━ Intermediate Sync Shares ━━━
# Shares synced every 4 hours by intermediate_sync_maintenance.sh.
# Uses DEFAULT_RSYNC_OPTS (no --delete) — for sub-daily propagation of metadata or watch state.
# Full media share sync stays in the daily window. Leave empty to skip mid-day rsync entirely.
HOST1_INTERMEDIATE_SYNC_SHARES=(
# Add shares here to enable mid-day rsync
# Example: "/mnt/user/Emby_Metadata"
)
# ━━━ Critical Sync Shares ━━━
# Appdata shares synced every 30 minutes by critical_sync_maintenance.sh.
# Format: "/path/to/share" or "/path/to/share|profile-name"
# Order matters — Critical-Data first (auth stack), then Emby dirty sync.
HOST1_CRITICAL_SYNC_SHARES=(
"/mnt/user/appdata-Fallback/Critical-Data|critical-fallback" # auth dirty sync — stays running
"/mnt/user/Media_Server/Emby|emby-fallback" # Emby dirty sync — stays running
)
# ━━━ Backup Verify ━━━
# Shares verified by backup_verify.sh — random file checksum comparison against remote.
# Leave empty to use HOST1_DAILY_SYNC_SHARES automatically.
# Sample size and minimum file size defined in master.conf.
HOST1_BACKUP_VERIFY_SHARES=(
# leave empty to use HOST1_DAILY_SYNC_SHARES automatically
)
# ━━━ HOST1 Rsync Profile — host1-appdata ━━━
# HOST1-specific appdata sync profile — extends the shared PROFILE_* arrays in master.conf.
# Use for appdata unique to HOST1 (Organizrv2, VaultWarden, UptimeKuma etc.)
# Shared appdata (auth stack, Emby) use dedicated profiles defined in master.conf.
# Run manually: bash Rsync/rsync.sh /mnt/user/appdata-Fallback/HOST1-Appdata --profile=host1-appdata
PROFILE_RSYNC_OPTS[host1-appdata]="-av --info=progress2 --bwlimit=${PROFILE_BW_LIMIT[host1-appdata]:-8000}"
PROFILE_BW_LIMIT[host1-appdata]=8000
PROFILE_RETRY_COUNT[host1-appdata]=3
PROFILE_SLEEP[host1-appdata]=300
PROFILE_CRITICAL_CONTAINER_NAMES[host1-appdata]="Organizrv2-Gmer4Lfe UptimeKuma-Gmer4Lfe VaultWarden-Gmer4Lfe"
PROFILE_DELAYED_CONTAINERS[host1-appdata]=""
PROFILE_CONTAINER_DELAY[host1-appdata]=5
PROFILE_EXCLUDE_DIRS[host1-appdata]="logs *.tmp"
# ==============================================================================================
# ── FALLBACK ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ DDNS ━━━
# DDNS containers HOST1 manages — started/stopped by fallback.sh per DDNS absolute rules:
# Internet loss → stop immediately
# Failover → HOST2 starts HOST1's DDNS as Tier 1 (before any other containers)
# Handback → stop HOST1's DDNS on HOST2 → rsync → start containers → start local DDNS last
HOST1_DDNS_CONTAINERS=(
"Gmer4Lfe.com"
)
# ━━━ Internet Loss ━━━
# Containers stopped immediately on HOST1 when internet connection is lost.
# Prevents external-facing services from operating without connectivity.
FALLBACK_HOST1_STOP_ON_NO_NET=(
"Gmer4Lfe.com"
)
# ━━━ Fallback Tiers — HOST1 Runs for HOST2 ━━━
# Containers HOST1 starts when HOST2 goes down.
# Tier 1 is always immediate — vital services cannot wait.
# Higher tiers activate after HOST2_TIER*_DELAY minutes (set in host2.conf).
FALLBACK_HOST1_COVERS_HOST2_TIER1=(
"Gmer4Lfe.us"
"VaultWarden-Jayred365"
)
FALLBACK_HOST1_COVERS_HOST2_TIER2=(
# "container-placeholder"
)
FALLBACK_HOST1_COVERS_HOST2_TIER3=(
# "container-placeholder"
)
FALLBACK_HOST1_COVERS_HOST2_TIER4=(
# "container-placeholder"
)
# ━━━ Tier Delays — HOST1's Containers on HOST2 ━━━
# How long HOST1 must be down before each tier activates on HOST2 — in minutes.
# Tier 1 is always immediate — no delay var needed.
HOST1_TIER2_DELAY=240 # 4 hours — NextCloud, Immich
HOST1_TIER3_DELAY=720 # 12 hours — secondary services
HOST1_TIER4_DELAY=1440 # 24 hours — arrs + downloaders
# ━━━ Rsync Writeback — HOST1 Appdata Back on Handback ━━━
# Syncs HOST1 appdata BACK to HOST1 when it comes back online after a failover.
# Containers stopped before writeback — clean source, no competing writes.
#
# HOST1_TIER1_WRITEBACK_DELAY: short outages skip Tier 1 writeback — primary state
# is more reliable than dirty sync data for brief outages.
HOST1_TIER1_WRITEBACK_DELAY=60 # skip Emby writeback if outage under 1hr
# Tier 4 automatically syncs HOST1_DAILY_SYNC_SHARES — only list paths NOT in that array.
FALLBACK_HOST1_WRITEBACK_TIER1=(
"/mnt/user/Media_Server/Emby" # watch states built up during outage
)
FALLBACK_HOST1_WRITEBACK_TIER2=(
"/mnt/user/appdata-Fallback/Important-Data" # NextCloud + Postgres — files added during outage
)
FALLBACK_HOST1_WRITEBACK_TIER3=(
# "location-placeholder"
)
FALLBACK_HOST1_WRITEBACK_TIER4=(
"/mnt/user/appdata-Fallback/Arrs_Stack" # arr databases — downloads queued during outage
)
# ==============================================================================================
# ── DOCKER ────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Docker Daily Restart ━━━
# Containers restarted every day via DAILY_MAINTENANCE_SCRIPTS.
# Dispatcharr degrades over time without restart — daily is intentional, not just housekeeping.
# Order matters — auth stack first, then media services.
HOST1_DAILY_RESTART_CONTAINERS=(
"NginxProxyManager"
"Lldap-Gmer4Lfe"
"Authelia"
"Authelia-Secondary"
"Dispatcharr-Iptv-Users"
"Dispatcharr" # Live TV scheduler — degrades without daily restart
"Dispatcharr-Basic"
"ErsatzTV-Emby"
"slskd" # Soulseek connection drops after extended uptime; restart refreshes share index
)
# ━━━ Docker Weekly Restart ━━━
# Less critical services restarted weekly via WEEKLY_MAINTENANCE_SCRIPTS (Sunday 2:30am).
# Containers already stopped for weekly sync — restart adds zero extra downtime.
HOST1_WEEKLY_RESTART_CONTAINERS=(
"NextCloud"
"Organizrv2-Gmer4Lfe"
"AdGuard-Home"
"Immich-Gmer4Lfe"
)
# ━━━ Docker Watchdog ━━━
# Per-HOST1 container configuration for docker_watchdog.sh.
# Shared thresholds and toggles live in master.conf.
# Memory hard limits in MB — immediate restart if exceeded.
# Set at "container is clearly broken" not "container is busy".
# 20GB=20480 18GB=18432 16GB=16384 12GB=12288 8GB=8192 4GB=4096 2GB=2048 1GB=1024
declare -A HOST1_WATCHDOG_CONTAINERS=(
["Emby"]=20480 # 20GB — large library + active transcodes
["LidaTube"]=6144 # 6GB — memory leak over time
["Tdarr"]=6144 # 6GB — encoding is memory intensive
["Code-Server"]=1024 # 1GB — should never need more
)
# HTTP health check URLs — checked every cycle, strike system before restart.
# Only add containers with a meaningful web interface to check.
declare -A HOST1_WATCHDOG_CONTAINER_URLS=(
["Emby"]="http://localhost:8096"
["NginxProxyManager"]="http://localhost:7818"
["Authelia"]="http://localhost:9091/api/health"
["Authelia-Secondary"]="http://localhost:9092/api/health"
["Lldap-Gmer4Lfe"]="http://localhost:17170"
)
# Required containers — must always be running on HOST1.
# Strike system before restart — repeated failures go on skip list, auto-clears on recovery.
# Listed in dependency order — dependencies before dependents.
HOST1_WATCHDOG_REQUIRED_CONTAINERS=(
"NginxProxyManager"
"Lldap-Gmer4Lfe"
"Mariadb-Authelia"
"Mariadb-Authelia-Secondary"
"Redis-Authelia"
"Redis-Authelia-Secondary"
"Authelia"
"Authelia-Secondary"
)
# Containers to skip in Tier 2 global scan — legitimately stopped or frequently restarting.
# Watchdog leaves these alone entirely — no restart attempts, no crash loop tracking.
HOST1_WATCHDOG_SCAN_IGNORE=(
"DashGate"
"PIA-WG-Config-Generator"
"Aperture"
"Aperture-Kids"
"pgvector-18-Apeture-Kids"
"Pgvector18-Aperture"
"emby-test" # broken test container (exit 127 — bad image)
)
# Dependency ordering — skip restarting a container if its dependency is also down.
# Prevents watchdog from restarting Authelia before Mariadb is back up.
# SPACE-SEPARATED STRINGS — converted to array at runtime.
declare -A HOST1_WATCHDOG_DEPENDENCIES=(
["Authelia"]="Mariadb-Authelia Redis-Authelia"
["Authelia-Secondary"]="Mariadb-Authelia Redis-Authelia-Secondary"
["NextCloud"]="Postgres-NextCloud"
)
# Per-container appdata growth suppress ceilings in MB.
# ONLY needed in specific cases — growth rate detection covers all containers automatically.
# Use this when a container legitimately has large stable data and you want to guarantee
# it never triggers a false-positive growth alert. Growth warnings are suppressed while the
# container's dir stays below this ceiling; above it, warnings resume as normal.
# 50GB=51200 25GB=25600 20GB=20480 15GB=15360 10GB=10240 5GB=5120
declare -A HOST1_WATCHDOG_APPDATA_SIZES=(
["Tdarr"]="25600" # 25GB — transcode cache grows legitimately during active jobs
["7dtd"]="20480" # 20GB — game server world data, expected to be large
)
# API-level health checks — checked every cycle alongside HTTP URL checks.
# Format: ["ContainerName"]="url|expected_json_key|expected_value"
# Empty = no API checks for this host.
declare -A HOST1_WATCHDOG_CONTAINER_API_CHECKS=(
)
# ━━━ Network Watchdog ━━━
# Host-specific connectivity config for Watchdogs/System/network_watchdog.sh.
HOST1_NETWORK_WATCHDOG_DDNS_DOMAIN="gmer4lfe.com"
HOST1_NETWORK_WATCHDOG_DDNS_CONTAINER="Gmer4Lfe.com"
HOST1_NETWORK_WATCHDOG_NPM_URL="https://gmer4lfe.com"
# ━━━ Docker Network Connect ━━━
# Containers connected to custom networks at array start by docker_network_connect.sh.
# Networks created if they don't exist — idempotent, safe to re-run.
HOST1_NETWORK_CONNECT_CONTAINERS=(
"memcached"
"Npm-CrowdSec"
)
HOST1_NETWORK_CONNECT_NETWORKS=(
"high-availability"
)
# ==============================================================================================
# ── MEDIA ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Media Permissions ━━━
# Shares that media_shares_permissions.sh applies PERMISSIONS_MODE and PERMISSIONS_OWNER to.
# Runs first in DAILY_MAINTENANCE_SCRIPTS — arr cleanup depends on correct ownership.
HOST1_MEDIA_PERMISSION_SHARES=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/appcache
/mnt/user/Books
/mnt/user/Downloads
/mnt/user/Games
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movie_Recordings
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Photo
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Tv_Recordings
/mnt/user/Tv_Shows
/mnt/user/YouTube
)
# ━━━ Media Cleaner ━━━
# Folder lists for media_cleaner.sh — two profiles: anime and media.
# File patterns shared across all servers — defined in master.conf.
# Called via DAILY_MAINTENANCE_SCRIPTS. Run manually: Media/media_cleaner.sh anime|media
HOST1_ANIME_CLEAN_FOLDERS=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
)
HOST1_MEDIA_CLEAN_FOLDERS=(
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Tv_Shows
)
# ==============================================================================================
# ── ARR STACK ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Used by arr cleanup scripts and arrs_failed_stalled_recovery.sh.
# detect_hosts() selects HOST1 vars when running on HOST1.
#
# PATH MAPS — container path → host path translation.
# Arr stores file paths using container-internal paths — scripts need host paths to scan.
# Add one entry per root folder in arr Settings → Media Management → Root Folders.
# ━━━ Downloaders ━━━
# Used by downloaders_reset.sh — runs every 30min via CRITICAL_MAINTENANCE_SCRIPTS.
# Clears stuck states, purges old history, prepares each client for a clean cycle.
# slskd — clears stuck searches, dead transfers, purges expired failed imports.
# SLSKD_FAILED_IMPORTS_DIR: where Soularr moves albums Lidarr rejected.
HOST1_SLSKD_URL="http://localhost:8980"
HOST1_SLSKD_API_KEY="4bF9kL2mNpQrT7vWxYz1A3dEgHjKoRsU"
HOST1_SLSKD_FAILED_IMPORTS_DIR="/mnt/user/Temp_Storage/Slskd/completed/failed_imports"
# SABnzbd
HOST1_SABNZBD_URL="http://localhost:8180"
HOST1_SABNZBD_API_KEY="8bfefe41d83b4d50883e32859b55ca9a"
# qBittorrent — deleteFiles=false removes torrent from qBit but leaves files on disk.
# Radarr/Sonarr manage actual files independently.
HOST1_QBIT_URL="http://localhost:8080"
HOST1_QBIT_USERNAME="root"
HOST1_QBIT_PASSWORD="Stay0utD!ck"
# ━━━ Lidarr — HOST1 only ━━━
# HOST2 does not run Lidarr — HOST1_LIDARR_RECOVERY flag handles the exit cleanly.
HOST1_LIDARR_URL="http://localhost:8686"
HOST1_LIDARR_API_KEY="b2977e71ef074bc0a0529d9fcce3b2dc"
HOST1_LIDARR_MUSIC_ROOT="/mnt/user/Music-New"
HOST1_FANART_API_KEY="Yd7147a43b692df0b364b94dc47efb81"
HOST1_LASTFM_API_KEY="be6dc169c33ae263e690c30d18b7491d"
declare -A HOST1_LIDARR_PATH_MAP=(
["/ext-music"]="/mnt/user/Music-New"
)
# ━━━ Sonarr ━━━
HOST1_SONARR_URL="http://localhost:8989"
HOST1_SONARR_API_KEY="130decd3db5b4c25afad64864cd03f9f"
HOST1_SONARR_TV_ROOT="/mnt/user/Tv_Shows"
# Note: stand-up_comedy in both Sonarr + Radarr — TV specials and movie specials, one folder
declare -A HOST1_SONARR_PATH_MAP=(
["/tv"]="/mnt/user/Tv_Shows"
["/ext-standup-comedy"]="/mnt/user/stand-up_comedy/series"
["/kids tv"]="/mnt/user/Kids_Tv_Shows"
["/ext-anime-shows"]="/mnt/user/Anime_Shows-Old"
)
# ━━━ Radarr ━━━
HOST1_RADARR_URL="http://localhost:7878"
HOST1_RADARR_API_KEY="d43a3ec6cf1549edb4af0cc63f98b2a9"
HOST1_TMDB_API_KEY="3dac5e2e49b5540472d2eafec4f01260"
HOST1_RADARR_MOVIES_ROOT="/mnt/user/Movies"
# Note: stand-up_comedy in both Radarr + Sonarr — movie specials and TV specials, one folder
declare -A HOST1_RADARR_PATH_MAP=(
["/movies"]="/mnt/user/Movies"
["/kids movies"]="/mnt/user/Kids_Movies"
["/ext-stand-up-comedy"]="/mnt/user/stand-up_comedy/specials"
["/anime-movies"]="/mnt/user/Anime_Movies-Old"
["/ext-anime-movies"]="/mnt/user/Anime_Movies-Old"
)
# ━━━ Arr Recovery Toggles ━━━
# false = skip that arr on this host — exits cleanly without error
HOST1_SONARR_RECOVERY=true
HOST1_RADARR_RECOVERY=true
HOST1_LIDARR_RECOVERY=true # HOST1 only — exits cleanly on HOST2
# ==============================================================================================
# ── TRANSCODES ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Ramdisk size ceiling — tmpfs only uses RAM actually needed, not the full size upfront.
# Real-world: 9 streams peaked at ~5.5GB — 10G gives generous headroom on 128GB RAM.
HOST1_RAMDISK_SIZE="10G"
# Usage thresholds — coupled to HOST1_RAMDISK_SIZE, adjust all three together if size changes.
# Hysteresis gap (8.5 - 7 = 1.5GB) prevents flip-flop between ramdisk and SSD.
HOST1_RAMDISK_WARN_GB=8.5 # flip to SSD when ramdisk usage reaches this
HOST1_RAMDISK_LOW_GB=7 # flip back to ramdisk when usage drops to this
# SSD fallback path — where transcodes land when ramdisk exceeds HOST1_RAMDISK_WARN_GB.
# Must be on cache pool — array disks too slow for active transcode writes.
HOST1_TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/"
# Media servers sharing the ramdisk transcode space on HOST1.
# Format: "ContainerName|URL|APIKey|Type" — Type: emby | jellyfin | plex
# Entries with placeholder API keys are skipped automatically.
# ⚠️ Tdarr does NOT belong here — keep Tdarr on SSD, not ramdisk.
HOST1_TRANSCODE_SERVERS=(
"${HOST1_EMBY_CONTAINER}|${HOST1_EMBY_URL}|${HOST1_EMBY_API_KEY}|emby"
"${HOST1_JELLYFIN_CONTAINER}|${HOST1_JELLYFIN_URL}|${HOST1_JELLYFIN_API_KEY}|jellyfin"
)
# ==============================================================================================
# ── MONITORS ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
# Domains checked via direct openssl connection — not relying on NPM's certificate state.
# Checks the actual certificate served, not what NPM thinks it has.
# Thresholds (CERT_WARN_DAYS, CERT_CRIT_DAYS) defined in master.conf.
HOST1_CERT_MONITOR_DOMAINS=(
"Gmer4Lfe.com"
"Gmer4Lfe.us"
)
# ━━━ SMART Health ━━━
# Drives skipped in SMART attribute monitoring — hardware is server-specific.
# Thresholds read from dynamix.cfg at runtime — fallbacks in master.conf.
HOST1_SMART_IGNORE_DRIVES=(
"sda" # boot USB — SMART not meaningful on flash drives
)
# ━━━ ZFS Report ━━━
# Pools excluded from the weekly ZFS health report — reduces noise from single-disk array pools.
# These are individual array disks formatted as ZFS — converting to XFS over time via unBalance.
# Pool health thresholds defined in master.conf.
HOST1_ZFS_REPORT_IGNORE_POOLS=(
"disk5"
"disk6"
"disk8"
"disk9"
"disk10"
)
# ==============================================================================================
# ── RESOURCE MANAGER ──────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Containers to manage under pressure — see master.conf RW_CRITICAL_CONTAINERS for exclusions.
# docker pause at medium pressure (RAM < RW_RAM_MEDIUM_GB or load > medium threshold)
# Suspended in-place — instant to pause/unpause, no state lost, no restart delay.
HOST1_RW_PAUSE_CONTAINERS=(
"Huntarr" # arr search automation — safe to suspend
"Cleanuparr" # download cleanup — safe to suspend
"Healarr" # arr health checks — safe to suspend
"Soularr" # Slskd automation — background only
"ChannelTube" # YouTube archiver — background only
"Pinchflat" # YouTube archiver — background only
)
# docker stop at hard pressure (RAM < RW_RAM_HARD_GB)
# Full stop — these are optional/heavy services that free significant RAM when stopped.
# resource_watchdog.sh restarts them when pressure fully clears (RAM >= RW_RAM_RECOVER_GB).
HOST1_RW_STOP_CONTAINERS=(
"LocalAI" # GPU/CPU heavy — largest RAM consumer when idle
"7DaysToDie" # game server — optional
"V-Rising" # game server — optional
"Code-Server" # IDE — not needed during pressure events
)
# ==============================================================================================
# ── SYSTEM WATCHDOG ───────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Per-host check toggles and NIC config for system_watchdog.sh.
# Aliased by detect_hosts() — script uses unprefixed SYS_WATCHDOG_* names.
# HOST1: TR1950X 128GB — full media server, active transcoding, ZFS cache pools.
#
# Three-tier response — all critical checks enabled by default on HOST1:
# Tier 1 (bypass strikes, reboot now): docker daemon, rootfs full, kernel oops, FD, /boot
# Tier 2 (bypass strikes with OOM): RAM critical + OOM kills in cycle
# Tier 3 (standard strike system): everything else
#
# RAM tiers, OOM limits, and reboot loop settings in master.conf System Watchdog section.
# ━━━ Primary NIC ━━━
# Network interface for NIC state check — verify with: ip link show | grep "^[0-9]"
# Common values: eth0, bond0, br0, eno1
HOST1_SYS_WATCHDOG_NIC="eth0"
# ━━━ Tier 1 — Critical Checks ━━━
# These bypass the strike system — a single hit triggers immediate reboot.
# Disabling any of these is not recommended — they protect against acute system failure.
# Docker daemon unresponsive → try restart, reboot if restart fails.
# Without a working daemon docker_watchdog.sh is blind and containers cannot be managed.
HOST1_SYS_WATCHDOG_CHECK_DOCKER_DAEMON=true
# rootfs at critical threshold (ROOTFS_CRITICAL_PCT=99) → reboot immediately.
# At 99% rootfs writes fail silently — logs stop, Docker errors out, SSH may stop working.
# Standard 95% threshold still uses strike system — only 99%+ is critical tier.
HOST1_SYS_WATCHDOG_CHECK_ROOTFS=true
# Kernel BUG/Oops in dmesg delta since last cycle → reboot immediately.
# A kernel oops means the kernel ran with a corrupted state — stability is not guaranteed.
HOST1_SYS_WATCHDOG_CHECK_KERNEL_OOPS=true
# File descriptor exhaustion at FD_CRITICAL_PCT (95%) → reboot immediately.
# At 95% FD: new connections fail, Docker can't spawn processes, SSH drops.
HOST1_SYS_WATCHDOG_CHECK_FD=true
# /boot read-only detected → reboot immediately.
# Unexpected read-only /boot means state files and config writes are silently failing.
# Fallback state, watchdog reboot log, and lock files all go stale silently.
HOST1_SYS_WATCHDOG_CHECK_BOOT=true
# ━━━ Tier 2 — Urgent OOM Check ━━━
# Bypass strikes when RAM is critically low AND OOM kill rate confirms active crisis.
# Both must be enabled for Tier 2 bypass to function — disable either to always use strikes.
# Track kernel OOM kills each cycle via /proc/vmstat oom_kill delta.
# Also provides diagnostic context in reboot messages (which processes were killed).
HOST1_SYS_WATCHDOG_CHECK_OOM=true
# Free RAM check — required for both Tier 2 bypass and RAM tier logic.
# Tiers: MEM_WARN_GB(10) → notify | MEM_SHUTDOWN_GB(6) → stop containers | MEM_GB(4) → strikes
HOST1_SYS_WATCHDOG_CHECK_RAM=true
# ━━━ Tier 3 — Standard Checks (strike system) ━━━
# Each check must fail SYS_WATCHDOG_STRIKE_LIMIT consecutive cycles before action is taken.
# Single spikes are ignored — sustained problems trigger reboot.
# /var/log filesystem usage above SYS_WATCHDOG_LOG_PCT.
# Log spam (Docker log storms, syslog loops) fills rootfs — indicates something broken.
HOST1_SYS_WATCHDOG_CHECK_LOG=true
# ZFS ARC memory pinned above SYS_WATCHDOG_ARC_PINNED_PCT after cache drop.
# Enabled on HOST1 — ZFS cache pools actively used. Disable on hosts without ZFS.
HOST1_SYS_WATCHDOG_CHECK_ARC=true
# CPU temperature above SYS_WATCHDOG_CPU_TEMP_MAX (95°C).
# Sustained high temp causes kernel throttling or panic. Requires lm-sensors.
HOST1_SYS_WATCHDOG_CHECK_CPU_TEMP=true
# Load average above SYS_WATCHDOG_LOAD_MULTIPLIER × core count.
# DISABLED on HOST1 — Tdarr and Emby cause legitimate sustained load spikes during encoding.
# Enable on idle servers or adjust SYS_WATCHDOG_LOAD_MULTIPLIER if load is always high.
HOST1_SYS_WATCHDOG_CHECK_LOAD=false
# Zombie process count above SYS_WATCHDOG_ZOMBIE_LIMIT (50).
# Large zombie counts indicate serious process management failure — something is stuck.
HOST1_SYS_WATCHDOG_CHECK_ZOMBIES=true
# Check docker_watchdog.sh persistent skip list — required containers on skip list.
# Cross-watchdog coordination: if docker_watchdog gave up, system_watchdog escalates.
# ENABLED — HOST1 fully built and operational, skip list is meaningful.
HOST1_SYS_WATCHDOG_CHECK_CONTAINERS=true
# /tmp filesystem usage above SYS_WATCHDOG_TMP_PCT with auto-clear attempt.
# Script tries to clear aged /tmp files first — only strikes if clear fails.
# Lock files, rsync temp files, and Docker ops use /tmp — 100% means lock failures.
HOST1_SYS_WATCHDOG_CHECK_TMP=true
# Array disk error count delta in /proc/mdstat — accumulating errors = disk failing now.
# Triggers on SYS_WATCHDOG_MDSTAT_ERROR_LIMIT new errors in one cycle.
HOST1_SYS_WATCHDOG_CHECK_MDSTAT=true
# Primary NIC operstate — detects NIC going down (physical or driver failure).
# Uses HOST1_SYS_WATCHDOG_NIC above. Strike system — brief flaps don't trigger reboot.
HOST1_SYS_WATCHDOG_CHECK_NETWORK=true
# sshd running check — attempts restart before escalating.
# sshd down = no remote access. Script tries rc.sshd start, notifies, strikes on failure.
HOST1_SYS_WATCHDOG_CHECK_SSHD=true
# Runaway process detection — single process above SYS_WATCHDOG_RUNAWAY_CPU_PCT sustained.
# DISABLED — Tdarr encoding and Emby transcoding legitimately peg CPU for extended periods.
# Enable only if HOST1 has no CPU-intensive workloads.
HOST1_SYS_WATCHDOG_CHECK_RUNAWAY=false
@@ -1,832 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ========================== HOST1 CONFIGURATION — unRAID-Gmer4Lfe ============================
# ==============================================================================================
# HOST1-specific variables — credentials, container names, share paths, failover lists.
# Sourced after master.conf — values here extend shared profile arrays and add HOST1-specific
# identity, credentials, and container configuration.
#
# Sparse checkout (git) ensures HOST2 never receives this file.
# HOST2 never sees HOST1 credentials — clean separation at the file level.
#
# DO NOT put shared config here — thresholds, toggles, profiles belong in master.conf.
# DO NOT put HOST2 variables here — they belong in host2.conf.
#
# ── INDEX ─────────────────────────────────────────────────────────────────────────────────────
#
# ── IDENTITY & CONNECTIVITY ────────────────────────────────────────────────────────────────
# IDENTITY hostname, SSH key, Unraid API key
# EMBY container name, URL, API key
# JELLYFIN container name, URL, API key
# GITEA API token for SSH key registration
# NOTIFICATIONS Discord webhook
#
# ── PARTNERSHIP ────────────────────────────────────────────────────────────────────────────
# PARTNERSHIP auth containers, backup paths, emby provisioning
#
# ── RSYNC ──────────────────────────────────────────────────────────────────────────────────
# DAILY SYNC SHARES media shares HOST1 owns and pushes to HOST2
# WEEKLY SYNC SHARES appdata shares synced weekly (Sunday 2:30am)
# CRITICAL SYNC SHARES appdata shares synced every 30 minutes
# BACKUP VERIFY shares for checksum verification against remote
# HOST1 RSYNC PROFILE host1-appdata profile for HOST1-specific appdata syncs
#
# ── FALLBACK ───────────────────────────────────────────────────────────────────────────────
# DDNS DDNS containers managed by HOST1
# INTERNET LOSS containers stopped when internet is lost
# FALLBACK TIERS what HOST1 runs for HOST2 per tier
# TIER DELAYS how long HOST1 must be down before each tier activates on HOST2
# RSYNC WRITEBACK HOST1 appdata synced back on handback
#
# ── DOCKER ─────────────────────────────────────────────────────────────────────────────────
# DOCKER DAILY RESTART containers restarted daily
# DOCKER WEEKLY RESTART containers restarted weekly
# DOCKER WATCHDOG memory limits, health URLs, required containers, ignore list
# NETWORK WATCHDOG DDNS domain, NPM URL for connectivity checks
# DOCKER NETWORK CONNECT networks and containers for docker_network_connect.sh
#
# ── MEDIA ──────────────────────────────────────────────────────────────────────────────────
# MEDIA PERMISSIONS share list for media_shares_permissions.sh
# MEDIA CLEANER folder lists for media_cleaner.sh
#
# ── ARR STACK ──────────────────────────────────────────────────────────────────────────────
# DOWNLOADERS slskd, SABnzbd, qBittorrent credentials and URLs
# LIDARR URL, API key, path map
# SONARR URL, API key, path map
# RADARR URL, API key, path map
# ARR RECOVERY per-arr recovery toggles
#
# ── TRANSCODES ─────────────────────────────────────────────────────────────────────────────
# TRANSCODES ramdisk size, thresholds, SSD path, server array
#
# ── MONITORS ───────────────────────────────────────────────────────────────────────────────
# CERTIFICATE MONITOR domains checked for SSL expiry
# SMART HEALTH drives to skip in SMART monitoring
# ZFS REPORT pools to exclude from ZFS health report
#
# ── RESOURCE MANAGER ───────────────────────────────────────────────────────────────────────
# RESOURCE MANAGER containers paused/stopped under memory pressure
#
# ── SYSTEM WATCHDOG ────────────────────────────────────────────────────────────────────────
# SYSTEM WATCHDOG per-host check toggles and NIC configuration
#
# ==============================================================================================
# ==============================================================================================
# ── STORAGE MODE ──────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Storage mode ━━━
# Controls where Varaverk stores scripts, conf, and state files.
# true = internal NVMe/SSD — /boot/config/plugins/varaverk (write-safe, git-direct)
# false = USB flash boot — /mnt/user/appdata/Varaverk (preserves flash lifetime)
HOST1_STORAGE_MODE_INTERNAL=true
# ==============================================================================================
# ── IDENTITY & CONNECTIVITY ───────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Identity ━━━
# HOST1 hostname lives in master.conf (not a credential — safe for all servers).
# SSH key used for all server-to-server operations — rsync, failover container commands.
# Must be in /root/.ssh/ and authorised in HOST2's /root/.ssh/authorized_keys.
HOST1_SSH_KEY="/root/.ssh/gmer4lfe_rsync_automation"
HOST1_OWNER="gmer4lfe"
HOST1_OWNER_EMAIL="gmer4lfe@gmail.com"
# ━━━ Unraid API ━━━
# Used by the Varaverk plugin to query this server's Unraid GraphQL API.
# Generate in Unraid: Settings → Management Access → API Keys → + New Key
HOST1_UNRAID_API_KEY="1825c3a2e03ea5089974f4da2e171aa2d5907a1dea23cc479c33e492c8ff4dbb"
# ━━━ Emby ━━━
# Referenced by transcode_manager.sh, emby_session_report.sh, emby_database_repair.sh,
# weekly_sync_maintenance.sh, and HOST1_TRANSCODE_SERVERS below.
# API key: Emby Dashboard → API Keys → + New Key
HOST1_EMBY_CONTAINER="Emby"
HOST1_EMBY_URL="http://localhost:8096"
HOST1_EMBY_API_KEY="0c27448d93a7431f9ac63569f7655829"
# ━━━ Jellyfin ━━━
# API key: Jellyfin Dashboard → Administration → API Keys → + New Key
HOST1_JELLYFIN_CONTAINER="Jellyfin"
HOST1_JELLYFIN_URL="http://localhost:8095"
HOST1_JELLYFIN_API_KEY="4e820e7df74c4933acec212b1996314e"
# ━━━ Gitea ━━━
# Personal access token for gitea_ssh_setup.sh — registers this server's SSH public key
# with Gitea so git operations use key auth instead of passwords.
# Create in Gitea: Settings → Applications → Generate Token → scope: write:user
HOST1_GITEA_API_TOKEN=""
# ━━━ Notifications ━━━
# Discord webhook — leave blank to disable.
# Per-host so HOST1 and HOST2 can post to different channels or only one server notifies.
HOST1_DISCORD_WEBHOOK=""
# ==============================================================================================
# ── PARTNERSHIP ───────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# HOST1 is always the owner (source of truth) unless --transfer has been run.
# See README-Partnership.md and master.conf PARTNERSHIP section for full lifecycle docs.
# Auth containers reconfigured on onboard/offboard.
# Format: "ContainerName|WebUIPort"
# On onboard → WebUI pointed at owner's Tailscale IP (mirror clicks NPM, gets owner's auth)
# On offboard → WebUI pointed back at localhost
HOST1_PARTNERSHIP_AUTH_WEBUIS=(
"NginxProxyManager|81"
"Lldap-Gmer4Lfe|17170"
"Authelia|9091"
"Authelia-Secondary|9092"
)
# XML templates (from this server's templates-user/) pushed to mirror during onboard.
# These become the mirror's active auth stack, backed by the rsync-synced appdata.
# Update filename if Lldap is renamed to drop the host suffix.
HOST1_PARTNERSHIP_AUTH_STACK=(
# Dependencies first — Mariadb/Redis must be healthy before Authelia starts
"my-Mariadb-Authelia.xml"
"my-Mariadb-Authelia-Secondary.xml"
"my-Redis-Authelia.xml"
"my-Redis-Authelia-Secondary.xml"
# Auth apps — deployed after their deps are confirmed healthy
"my-Authelia.xml"
"my-Authelia-Secondary.xml"
"my-NginxProxyManager.xml"
"my-Lldap-Gmer4Lfe.xml"
# Source of truth — must be available on HOST2 independently of the auth stack
"my-Gitea.xml"
)
# XML templates pushed to mirror for the arr stack during onboard.
# Deps (e.g. databases) first if any — same ordering rule as auth stack.
HOST1_PARTNERSHIP_ARR_STACK=(
# "my-Sonarr.xml"
# "my-Radarr.xml"
# "my-Lidarr.xml"
# "my-Prowlarr.xml"
# "my-Bazarr.xml"
)
# Containers stopped on THIS server before deploying the mirror's stack on onboard.
# Only needed when this server parks its own stack to make room for the mirror's.
HOST1_PARTNERSHIP_REPLACE_CONTAINERS=(
)
# Arr containers stopped on this server when mirror's arr stack is deployed.
HOST1_PARTNERSHIP_ARR_REPLACE_CONTAINERS=(
)
# Paths HOST2 should collect during the grace window after offboard.
# Notified on offboard — no auto-deletion, HOST2 must collect manually within PARTNERSHIP_GRACE_HOURS.
HOST1_PARTNERSHIP_MIRROR_BACKUPS=(
# "/mnt/user/appdata-Fallback/Jayred365-Emby"
)
# Containers parked on this server when partnership is active.
# Stopped on onboard (owner deploys its stack instead), restarted on offboard.
HOST1_PARTNERSHIP_OWN_CONTAINERS=(
# "Emby"
# "NginxProxyManager"
)
# Emby admin provisioning — toggle is owner-only, credentials are per-host.
# Owner enables/disables the feature. Each host sets the account they want on the shared Emby.
# On onboard: owner reads mirror's HOST*_PARTNERSHIP_EMBY_ADMIN_* and creates that account.
# On offboard: account is deleted. Username collision → onboard exits with error.
HOST1_PARTNERSHIP_PROVISION_EMBY_ADMIN=false # owner controls whether Emby is shared
HOST1_PARTNERSHIP_EMBY_PORT=8096
HOST1_PARTNERSHIP_EMBY_ADMIN_USER="" # this server's desired Emby username
HOST1_PARTNERSHIP_EMBY_ADMIN_PASS="" # this server's desired Emby password
# ==============================================================================================
# ── RSYNC ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Daily Sync Shares ━━━
# Shares HOST1 pushes to all other nodes every night (1am via daily_sync_maintenance.sh).
# Mesh model: every node pushes every media share — no ownership, no mirrors.
# arr_sync ensures all arr libraries converge (union). rsync spreads files (additive, no --delete).
# arr_cleanup removes true orphans based on local arr state.
# Any node can download content to any share — it propagates to all nodes on the next cycle.
# Nextcloud is intentionally one-directional (HOST1→HOST2 offsite backup — not arr-managed).
# Uses DEFAULT_RSYNC_OPTS from master.conf — no profile needed.
# For shares needing container stops or custom options — add a profile in master.conf.
HOST1_DAILY_SYNC_SHARES=(
/mnt/user/Books
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Nextcloud
/mnt/user/stand-up_comedy
/mnt/user/Sports
# /mnt/user/Tv_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Movies
/mnt/user/Anime_Shows
)
# Personal encrypted shares — synced for offsite backup, independent of media shares.
# ZFS encrypted at dataset level — remote receives encrypted blocks, cannot read content.
# See README-Rsync_Setup.md for ZFS encryption setup before uncommenting.
HOST1_PERSONAL_SHARES=(
# /mnt/user/HOST1-Personal # uncomment after creating encrypted dataset
)
# ━━━ Weekly Sync Shares ━━━
# Appdata shares synced during the weekly maintenance window (Sunday 2:30am).
# Containers stopped both sides before sync — full clean state guaranteed.
# Profiles drive container stops, excludes, and options — configured in master.conf RSYNC section.
# Order matters — Emby first (larger transfer), then Critical-Data (auth stack).
HOST1_WEEKLY_SYNC_SHARES=(
"/mnt/user/Media_Server/Emby" # emby profile — full clean mirror
"/mnt/user/appdata-Fallback/Critical-Data" # critical-data profile — auth stack
)
# ━━━ Intermediate Sync Shares ━━━
# Shares synced every 4 hours by intermediate_sync_maintenance.sh.
# Uses DEFAULT_RSYNC_OPTS (no --delete) — for sub-daily propagation of metadata or watch state.
# Full media share sync stays in the daily window. Leave empty to skip mid-day rsync entirely.
HOST1_INTERMEDIATE_SYNC_SHARES=(
# Add shares here to enable mid-day rsync
# Example: "/mnt/user/Emby_Metadata"
)
# ━━━ Critical Sync Shares ━━━
# Appdata shares synced every 30 minutes by critical_sync_maintenance.sh.
# Format: "/path/to/share" or "/path/to/share|profile-name"
# Order matters — Critical-Data first (auth stack), then Emby dirty sync.
HOST1_CRITICAL_SYNC_SHARES=(
"/mnt/user/appdata-Fallback/Critical-Data|critical-fallback" # auth dirty sync — stays running
"/mnt/user/Media_Server/Emby|emby-fallback" # Emby dirty sync — stays running
)
# ━━━ Backup Verify ━━━
# Shares verified by backup_verify.sh — random file checksum comparison against remote.
# Leave empty to use HOST1_DAILY_SYNC_SHARES automatically.
# Sample size and minimum file size defined in master.conf.
HOST1_BACKUP_VERIFY_SHARES=(
# leave empty to use HOST1_DAILY_SYNC_SHARES automatically
)
# ━━━ HOST1 Rsync Profile — host1-appdata ━━━
# HOST1-specific appdata sync profile — extends the shared PROFILE_* arrays in master.conf.
# Use for appdata unique to HOST1 (Organizrv2, VaultWarden, UptimeKuma etc.)
# Shared appdata (auth stack, Emby) use dedicated profiles defined in master.conf.
# Run manually: bash Rsync/rsync.sh /mnt/user/appdata-Fallback/HOST1-Appdata --profile=host1-appdata
PROFILE_RSYNC_OPTS[host1-appdata]="-av --info=progress2 --bwlimit=${PROFILE_BW_LIMIT[host1-appdata]:-8000}"
PROFILE_BW_LIMIT[host1-appdata]=8000
PROFILE_RETRY_COUNT[host1-appdata]=3
PROFILE_SLEEP[host1-appdata]=300
PROFILE_CRITICAL_CONTAINER_NAMES[host1-appdata]="Organizrv2-Gmer4Lfe UptimeKuma-Gmer4Lfe VaultWarden-Gmer4Lfe"
PROFILE_DELAYED_CONTAINERS[host1-appdata]=""
PROFILE_CONTAINER_DELAY[host1-appdata]=5
PROFILE_EXCLUDE_DIRS[host1-appdata]="logs *.tmp"
# ==============================================================================================
# ── FALLBACK ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ DDNS ━━━
# DDNS containers HOST1 manages — started/stopped by fallback.sh per DDNS absolute rules:
# Internet loss → stop immediately
# Failover → HOST2 starts HOST1's DDNS as Tier 1 (before any other containers)
# Handback → stop HOST1's DDNS on HOST2 → rsync → start containers → start local DDNS last
HOST1_DDNS_CONTAINERS=(
"Gmer4Lfe.com"
)
# ━━━ Internet Loss ━━━
# Containers stopped immediately on HOST1 when internet connection is lost.
# Prevents external-facing services from operating without connectivity.
FALLBACK_HOST1_STOP_ON_NO_NET=(
"Gmer4Lfe.com"
)
# ━━━ Fallback Tiers — HOST1 Runs for HOST2 ━━━
# Containers HOST1 starts when HOST2 goes down.
# Tier 1 is always immediate — vital services cannot wait.
# Higher tiers activate after HOST2_TIER*_DELAY minutes (set in host2.conf).
FALLBACK_HOST1_COVERS_HOST2_TIER1=(
"Gmer4Lfe.us"
"VaultWarden-Jayred365"
)
FALLBACK_HOST1_COVERS_HOST2_TIER2=(
# "container-placeholder"
)
FALLBACK_HOST1_COVERS_HOST2_TIER3=(
# "container-placeholder"
)
FALLBACK_HOST1_COVERS_HOST2_TIER4=(
# "container-placeholder"
)
# ━━━ Tier Delays — HOST1's Containers on HOST2 ━━━
# How long HOST1 must be down before each tier activates on HOST2 — in minutes.
# Tier 1 is always immediate — no delay var needed.
HOST1_TIER2_DELAY=240 # 4 hours — NextCloud, Immich
HOST1_TIER3_DELAY=720 # 12 hours — secondary services
HOST1_TIER4_DELAY=1440 # 24 hours — arrs + downloaders
# ━━━ Rsync Writeback — HOST1 Appdata Back on Handback ━━━
# Syncs HOST1 appdata BACK to HOST1 when it comes back online after a failover.
# Containers stopped before writeback — clean source, no competing writes.
#
# HOST1_TIER1_WRITEBACK_DELAY: short outages skip Tier 1 writeback — primary state
# is more reliable than dirty sync data for brief outages.
HOST1_TIER1_WRITEBACK_DELAY=60 # skip Emby writeback if outage under 1hr
# Tier 4 automatically syncs HOST1_DAILY_SYNC_SHARES — only list paths NOT in that array.
FALLBACK_HOST1_WRITEBACK_TIER1=(
"/mnt/user/Media_Server/Emby" # watch states built up during outage
)
FALLBACK_HOST1_WRITEBACK_TIER2=(
"/mnt/user/appdata-Fallback/Important-Data" # NextCloud + Postgres — files added during outage
)
FALLBACK_HOST1_WRITEBACK_TIER3=(
# "location-placeholder"
)
FALLBACK_HOST1_WRITEBACK_TIER4=(
"/mnt/user/appdata-Fallback/Arrs_Stack" # arr databases — downloads queued during outage
)
# ==============================================================================================
# ── DOCKER ────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Docker Daily Restart ━━━
# Containers restarted every day via DAILY_MAINTENANCE_SCRIPTS.
# Dispatcharr degrades over time without restart — daily is intentional, not just housekeeping.
# Order matters — auth stack first, then media services.
HOST1_DAILY_RESTART_CONTAINERS=(
"NginxProxyManager"
"Lldap-Gmer4Lfe"
"Authelia"
"Authelia-Secondary"
"Dispatcharr-Iptv-Users"
"Dispatcharr" # Live TV scheduler — degrades without daily restart
"Dispatcharr-Basic"
"ErsatzTV-Emby"
"slskd" # Soulseek connection drops after extended uptime; restart refreshes share index
)
# ━━━ Docker Weekly Restart ━━━
# Less critical services restarted weekly via WEEKLY_MAINTENANCE_SCRIPTS (Sunday 2:30am).
# Containers already stopped for weekly sync — restart adds zero extra downtime.
HOST1_WEEKLY_RESTART_CONTAINERS=(
"NextCloud"
"Organizrv2-Gmer4Lfe"
"AdGuard-Home"
"Immich-Gmer4Lfe"
)
# ━━━ Docker Watchdog ━━━
# Per-HOST1 container configuration for docker_watchdog.sh.
# Shared thresholds and toggles live in master.conf.
# Memory hard limits in MB — immediate restart if exceeded.
# Set at "container is clearly broken" not "container is busy".
# 20GB=20480 18GB=18432 16GB=16384 12GB=12288 8GB=8192 4GB=4096 2GB=2048 1GB=1024
declare -A HOST1_WATCHDOG_CONTAINERS=(
["Emby"]=20480 # 20GB — large library + active transcodes
["LidaTube"]=6144 # 6GB — memory leak over time
["Tdarr"]=6144 # 6GB — encoding is memory intensive
["Code-Server"]=1024 # 1GB — should never need more
)
# HTTP health check URLs — checked every cycle, strike system before restart.
# Only add containers with a meaningful web interface to check.
declare -A HOST1_WATCHDOG_CONTAINER_URLS=(
["Emby"]="http://localhost:8096"
["NginxProxyManager"]="http://localhost:7818"
["Authelia"]="http://localhost:9091/api/health"
["Authelia-Secondary"]="http://localhost:9092/api/health"
["Lldap-Gmer4Lfe"]="http://localhost:17170"
)
# Required containers — must always be running on HOST1.
# Strike system before restart — repeated failures go on skip list, auto-clears on recovery.
# Listed in dependency order — dependencies before dependents.
HOST1_WATCHDOG_REQUIRED_CONTAINERS=(
"NginxProxyManager"
"Lldap-Gmer4Lfe"
"Mariadb-Authelia"
"Mariadb-Authelia-Secondary"
"Redis-Authelia"
"Redis-Authelia-Secondary"
"Authelia"
"Authelia-Secondary"
)
# Containers to skip in Tier 2 global scan — legitimately stopped or frequently restarting.
# Watchdog leaves these alone entirely — no restart attempts, no crash loop tracking.
HOST1_WATCHDOG_SCAN_IGNORE=(
"DashGate"
"PIA-WG-Config-Generator"
"Aperture"
"Aperture-Kids"
"pgvector-18-Apeture-Kids"
"Pgvector18-Aperture"
"emby-test" # broken test container (exit 127 — bad image)
)
# Dependency ordering — skip restarting a container if its dependency is also down.
# Prevents watchdog from restarting Authelia before Mariadb is back up.
# SPACE-SEPARATED STRINGS — converted to array at runtime.
declare -A HOST1_WATCHDOG_DEPENDENCIES=(
["Authelia"]="Mariadb-Authelia Redis-Authelia"
["Authelia-Secondary"]="Mariadb-Authelia Redis-Authelia-Secondary"
["NextCloud"]="Postgres-NextCloud"
)
# Per-container appdata growth suppress ceilings in MB.
# ONLY needed in specific cases — growth rate detection covers all containers automatically.
# Use this when a container legitimately has large stable data and you want to guarantee
# it never triggers a false-positive growth alert. Growth warnings are suppressed while the
# container's dir stays below this ceiling; above it, warnings resume as normal.
# 50GB=51200 25GB=25600 20GB=20480 15GB=15360 10GB=10240 5GB=5120
declare -A HOST1_WATCHDOG_APPDATA_SIZES=(
["Tdarr"]="25600" # 25GB — transcode cache grows legitimately during active jobs
["7dtd"]="20480" # 20GB — game server world data, expected to be large
)
# API-level health checks — checked every cycle alongside HTTP URL checks.
# Format: ["ContainerName"]="url|expected_json_key|expected_value"
# Empty = no API checks for this host.
declare -A HOST1_WATCHDOG_CONTAINER_API_CHECKS=(
)
# ━━━ Network Watchdog ━━━
# Host-specific connectivity config for Watchdogs/System/network_watchdog.sh.
HOST1_NETWORK_WATCHDOG_DDNS_DOMAIN="gmer4lfe.com"
HOST1_NETWORK_WATCHDOG_DDNS_CONTAINER="Gmer4Lfe.com"
HOST1_NETWORK_WATCHDOG_NPM_URL="https://gmer4lfe.com"
# ━━━ Docker Network Connect ━━━
# Containers connected to custom networks at array start by docker_network_connect.sh.
# Networks created if they don't exist — idempotent, safe to re-run.
HOST1_NETWORK_CONNECT_CONTAINERS=(
"memcached"
"Npm-CrowdSec"
)
HOST1_NETWORK_CONNECT_NETWORKS=(
"high-availability"
)
# ==============================================================================================
# ── MEDIA ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Media Permissions ━━━
# Shares that media_shares_permissions.sh applies PERMISSIONS_MODE and PERMISSIONS_OWNER to.
# Runs first in DAILY_MAINTENANCE_SCRIPTS — arr cleanup depends on correct ownership.
HOST1_MEDIA_PERMISSION_SHARES=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/appcache
/mnt/user/Books
/mnt/user/Downloads
/mnt/user/Games
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movie_Recordings
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Photo
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Tv_Recordings
/mnt/user/Tv_Shows
/mnt/user/YouTube
)
# ━━━ Media Cleaner ━━━
# Folder lists for media_cleaner.sh — two profiles: anime and media.
# File patterns shared across all servers — defined in master.conf.
# Called via DAILY_MAINTENANCE_SCRIPTS. Run manually: Media/media_cleaner.sh anime|media
HOST1_ANIME_CLEAN_FOLDERS=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
)
HOST1_MEDIA_CLEAN_FOLDERS=(
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Tv_Shows
)
# ==============================================================================================
# ── ARR STACK ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Used by arr cleanup scripts and arrs_failed_stalled_recovery.sh.
# detect_hosts() selects HOST1 vars when running on HOST1.
#
# PATH MAPS — container path → host path translation.
# Arr stores file paths using container-internal paths — scripts need host paths to scan.
# Add one entry per root folder in arr Settings → Media Management → Root Folders.
# ━━━ Downloaders ━━━
# Used by downloaders_reset.sh — runs every 30min via CRITICAL_MAINTENANCE_SCRIPTS.
# Clears stuck states, purges old history, prepares each client for a clean cycle.
# slskd — clears stuck searches, dead transfers, purges expired failed imports.
# SLSKD_FAILED_IMPORTS_DIR: where Soularr moves albums Lidarr rejected.
HOST1_SLSKD_URL="http://localhost:8980"
HOST1_SLSKD_API_KEY="4bF9kL2mNpQrT7vWxYz1A3dEgHjKoRsU"
HOST1_SLSKD_FAILED_IMPORTS_DIR="/mnt/user/Temp_Storage/Slskd/completed/failed_imports"
# SABnzbd
HOST1_SABNZBD_URL="http://localhost:8180"
HOST1_SABNZBD_API_KEY="8bfefe41d83b4d50883e32859b55ca9a"
# qBittorrent — deleteFiles=false removes torrent from qBit but leaves files on disk.
# Radarr/Sonarr manage actual files independently.
HOST1_QBIT_URL="http://localhost:8080"
HOST1_QBIT_USERNAME="root"
HOST1_QBIT_PASSWORD="Stay0utD!ck"
# ━━━ Lidarr — HOST1 only ━━━
# HOST2 does not run Lidarr — HOST1_LIDARR_RECOVERY flag handles the exit cleanly.
HOST1_LIDARR_URL="http://localhost:8686"
HOST1_LIDARR_API_KEY="b2977e71ef074bc0a0529d9fcce3b2dc"
HOST1_LIDARR_MUSIC_ROOT="/mnt/user/Music-New"
HOST1_FANART_API_KEY="Yd7147a43b692df0b364b94dc47efb81"
HOST1_LASTFM_API_KEY="be6dc169c33ae263e690c30d18b7491d"
declare -A HOST1_LIDARR_PATH_MAP=(
["/ext-music"]="/mnt/user/Music-New"
)
# ━━━ Sonarr ━━━
HOST1_SONARR_URL="http://localhost:8989"
HOST1_SONARR_API_KEY="130decd3db5b4c25afad64864cd03f9f"
HOST1_SONARR_TV_ROOT="/mnt/user/Tv_Shows"
# Note: stand-up_comedy in both Sonarr + Radarr — TV specials and movie specials, one folder
declare -A HOST1_SONARR_PATH_MAP=(
["/tv"]="/mnt/user/Tv_Shows"
["/ext-standup-comedy"]="/mnt/user/stand-up_comedy/series"
["/kids tv"]="/mnt/user/Kids_Tv_Shows"
["/ext-anime-shows"]="/mnt/user/Anime_Shows-Old"
)
# ━━━ Radarr ━━━
HOST1_RADARR_URL="http://localhost:7878"
HOST1_RADARR_API_KEY="d43a3ec6cf1549edb4af0cc63f98b2a9"
HOST1_TMDB_API_KEY="3dac5e2e49b5540472d2eafec4f01260"
HOST1_RADARR_MOVIES_ROOT="/mnt/user/Movies"
# Note: stand-up_comedy in both Radarr + Sonarr — movie specials and TV specials, one folder
declare -A HOST1_RADARR_PATH_MAP=(
["/movies"]="/mnt/user/Movies"
["/kids movies"]="/mnt/user/Kids_Movies"
["/ext-stand-up-comedy"]="/mnt/user/stand-up_comedy/specials"
["/anime-movies"]="/mnt/user/Anime_Movies-Old"
["/ext-anime-movies"]="/mnt/user/Anime_Movies-Old"
)
# ━━━ Arr Recovery Toggles ━━━
# false = skip that arr on this host — exits cleanly without error
HOST1_SONARR_RECOVERY=true
HOST1_RADARR_RECOVERY=true
HOST1_LIDARR_RECOVERY=true # HOST1 only — exits cleanly on HOST2
# ==============================================================================================
# ── TRANSCODES ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Ramdisk size ceiling — tmpfs only uses RAM actually needed, not the full size upfront.
# Real-world: 9 streams peaked at ~5.5GB — 10G gives generous headroom on 128GB RAM.
HOST1_RAMDISK_SIZE="10G"
# Usage thresholds — coupled to HOST1_RAMDISK_SIZE, adjust all three together if size changes.
# Hysteresis gap (8.5 - 7 = 1.5GB) prevents flip-flop between ramdisk and SSD.
HOST1_RAMDISK_WARN_GB=8.5 # flip to SSD when ramdisk usage reaches this
HOST1_RAMDISK_LOW_GB=7 # flip back to ramdisk when usage drops to this
# SSD fallback path — where transcodes land when ramdisk exceeds HOST1_RAMDISK_WARN_GB.
# Must be on cache pool — array disks too slow for active transcode writes.
HOST1_TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/"
# Media servers sharing the ramdisk transcode space on HOST1.
# Format: "ContainerName|URL|APIKey|Type" — Type: emby | jellyfin | plex
# Entries with placeholder API keys are skipped automatically.
# ⚠️ Tdarr does NOT belong here — keep Tdarr on SSD, not ramdisk.
HOST1_TRANSCODE_SERVERS=(
"${HOST1_EMBY_CONTAINER}|${HOST1_EMBY_URL}|${HOST1_EMBY_API_KEY}|emby"
"${HOST1_JELLYFIN_CONTAINER}|${HOST1_JELLYFIN_URL}|${HOST1_JELLYFIN_API_KEY}|jellyfin"
)
# ==============================================================================================
# ── MONITORS ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
# Domains checked via direct openssl connection — not relying on NPM's certificate state.
# Checks the actual certificate served, not what NPM thinks it has.
# Thresholds (CERT_WARN_DAYS, CERT_CRIT_DAYS) defined in master.conf.
HOST1_CERT_MONITOR_DOMAINS=(
"Gmer4Lfe.com"
"Gmer4Lfe.us"
)
# ━━━ SMART Health ━━━
# Drives skipped in SMART attribute monitoring — hardware is server-specific.
# Thresholds read from dynamix.cfg at runtime — fallbacks in master.conf.
HOST1_SMART_IGNORE_DRIVES=(
"sda" # boot USB — SMART not meaningful on flash drives
)
# ━━━ ZFS Report ━━━
# Pools excluded from the weekly ZFS health report — reduces noise from single-disk array pools.
# These are individual array disks formatted as ZFS — converting to XFS over time via unBalance.
# Pool health thresholds defined in master.conf.
HOST1_ZFS_REPORT_IGNORE_POOLS=(
"disk5"
"disk6"
"disk8"
"disk9"
"disk10"
)
# ==============================================================================================
# ── RESOURCE MANAGER ──────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Containers to manage under pressure — see master.conf RW_CRITICAL_CONTAINERS for exclusions.
# docker pause at medium pressure (RAM < RW_RAM_MEDIUM_GB or load > medium threshold)
# Suspended in-place — instant to pause/unpause, no state lost, no restart delay.
HOST1_RW_PAUSE_CONTAINERS=(
"Huntarr" # arr search automation — safe to suspend
"Cleanuparr" # download cleanup — safe to suspend
"Healarr" # arr health checks — safe to suspend
"Soularr" # Slskd automation — background only
"ChannelTube" # YouTube archiver — background only
"Pinchflat" # YouTube archiver — background only
)
# docker stop at hard pressure (RAM < RW_RAM_HARD_GB)
# Full stop — these are optional/heavy services that free significant RAM when stopped.
# resource_watchdog.sh restarts them when pressure fully clears (RAM >= RW_RAM_RECOVER_GB).
HOST1_RW_STOP_CONTAINERS=(
"LocalAI" # GPU/CPU heavy — largest RAM consumer when idle
"7DaysToDie" # game server — optional
"V-Rising" # game server — optional
"Code-Server" # IDE — not needed during pressure events
)
# ==============================================================================================
# ── SYSTEM WATCHDOG ───────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Per-host check toggles and NIC config for system_watchdog.sh.
# Aliased by detect_hosts() — script uses unprefixed SYS_WATCHDOG_* names.
# HOST1: TR1950X 128GB — full media server, active transcoding, ZFS cache pools.
#
# Three-tier response — all critical checks enabled by default on HOST1:
# Tier 1 (bypass strikes, reboot now): docker daemon, rootfs full, kernel oops, FD, /boot
# Tier 2 (bypass strikes with OOM): RAM critical + OOM kills in cycle
# Tier 3 (standard strike system): everything else
#
# RAM tiers, OOM limits, and reboot loop settings in master.conf System Watchdog section.
# ━━━ Primary NIC ━━━
# Network interface for NIC state check — verify with: ip link show | grep "^[0-9]"
# Common values: eth0, bond0, br0, eno1
HOST1_SYS_WATCHDOG_NIC="eth0"
# ━━━ Tier 1 — Critical Checks ━━━
# These bypass the strike system — a single hit triggers immediate reboot.
# Disabling any of these is not recommended — they protect against acute system failure.
# Docker daemon unresponsive → try restart, reboot if restart fails.
# Without a working daemon docker_watchdog.sh is blind and containers cannot be managed.
HOST1_SYS_WATCHDOG_CHECK_DOCKER_DAEMON=true
# rootfs at critical threshold (ROOTFS_CRITICAL_PCT=99) → reboot immediately.
# At 99% rootfs writes fail silently — logs stop, Docker errors out, SSH may stop working.
# Standard 95% threshold still uses strike system — only 99%+ is critical tier.
HOST1_SYS_WATCHDOG_CHECK_ROOTFS=true
# Kernel BUG/Oops in dmesg delta since last cycle → reboot immediately.
# A kernel oops means the kernel ran with a corrupted state — stability is not guaranteed.
HOST1_SYS_WATCHDOG_CHECK_KERNEL_OOPS=true
# File descriptor exhaustion at FD_CRITICAL_PCT (95%) → reboot immediately.
# At 95% FD: new connections fail, Docker can't spawn processes, SSH drops.
HOST1_SYS_WATCHDOG_CHECK_FD=true
# /boot read-only detected → reboot immediately.
# Unexpected read-only /boot means state files and config writes are silently failing.
# Fallback state, watchdog reboot log, and lock files all go stale silently.
HOST1_SYS_WATCHDOG_CHECK_BOOT=true
# ━━━ Tier 2 — Urgent OOM Check ━━━
# Bypass strikes when RAM is critically low AND OOM kill rate confirms active crisis.
# Both must be enabled for Tier 2 bypass to function — disable either to always use strikes.
# Track kernel OOM kills each cycle via /proc/vmstat oom_kill delta.
# Also provides diagnostic context in reboot messages (which processes were killed).
HOST1_SYS_WATCHDOG_CHECK_OOM=true
# Free RAM check — required for both Tier 2 bypass and RAM tier logic.
# Tiers: MEM_WARN_GB(10) → notify | MEM_SHUTDOWN_GB(6) → stop containers | MEM_GB(4) → strikes
HOST1_SYS_WATCHDOG_CHECK_RAM=true
# ━━━ Tier 3 — Standard Checks (strike system) ━━━
# Each check must fail SYS_WATCHDOG_STRIKE_LIMIT consecutive cycles before action is taken.
# Single spikes are ignored — sustained problems trigger reboot.
# /var/log filesystem usage above SYS_WATCHDOG_LOG_PCT.
# Log spam (Docker log storms, syslog loops) fills rootfs — indicates something broken.
HOST1_SYS_WATCHDOG_CHECK_LOG=true
# ZFS ARC memory pinned above SYS_WATCHDOG_ARC_PINNED_PCT after cache drop.
# Enabled on HOST1 — ZFS cache pools actively used. Disable on hosts without ZFS.
HOST1_SYS_WATCHDOG_CHECK_ARC=true
# CPU temperature above SYS_WATCHDOG_CPU_TEMP_MAX (95°C).
# Sustained high temp causes kernel throttling or panic. Requires lm-sensors.
HOST1_SYS_WATCHDOG_CHECK_CPU_TEMP=true
# Load average above SYS_WATCHDOG_LOAD_MULTIPLIER × core count.
# DISABLED on HOST1 — Tdarr and Emby cause legitimate sustained load spikes during encoding.
# Enable on idle servers or adjust SYS_WATCHDOG_LOAD_MULTIPLIER if load is always high.
HOST1_SYS_WATCHDOG_CHECK_LOAD=false
# Zombie process count above SYS_WATCHDOG_ZOMBIE_LIMIT (50).
# Large zombie counts indicate serious process management failure — something is stuck.
HOST1_SYS_WATCHDOG_CHECK_ZOMBIES=true
# Check docker_watchdog.sh persistent skip list — required containers on skip list.
# Cross-watchdog coordination: if docker_watchdog gave up, system_watchdog escalates.
# ENABLED — HOST1 fully built and operational, skip list is meaningful.
HOST1_SYS_WATCHDOG_CHECK_CONTAINERS=true
# /tmp filesystem usage above SYS_WATCHDOG_TMP_PCT with auto-clear attempt.
# Script tries to clear aged /tmp files first — only strikes if clear fails.
# Lock files, rsync temp files, and Docker ops use /tmp — 100% means lock failures.
HOST1_SYS_WATCHDOG_CHECK_TMP=true
# Array disk error count delta in /proc/mdstat — accumulating errors = disk failing now.
# Triggers on SYS_WATCHDOG_MDSTAT_ERROR_LIMIT new errors in one cycle.
HOST1_SYS_WATCHDOG_CHECK_MDSTAT=true
# Primary NIC operstate — detects NIC going down (physical or driver failure).
# Uses HOST1_SYS_WATCHDOG_NIC above. Strike system — brief flaps don't trigger reboot.
HOST1_SYS_WATCHDOG_CHECK_NETWORK=true
# sshd running check — attempts restart before escalating.
# sshd down = no remote access. Script tries rc.sshd start, notifies, strikes on failure.
HOST1_SYS_WATCHDOG_CHECK_SSHD=true
# Runaway process detection — single process above SYS_WATCHDOG_RUNAWAY_CPU_PCT sustained.
# DISABLED — Tdarr encoding and Emby transcoding legitimately peg CPU for extended periods.
# Enable only if HOST1 has no CPU-intensive workloads.
HOST1_SYS_WATCHDOG_CHECK_RUNAWAY=false
# ==============================================================================================
# ── AUTH STACK ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Credentials for the Varaverk Auth Stack page (NPM, lldap, Authelia).
# HOST1 is the auth source of truth — these are the live production credentials.
# ━━━ NginxProxyManager ━━━
# Admin API runs on 7818 (not 81 — 81 is the partnership WebUI port).
HOST1_NPM_URL="http://localhost:7818"
HOST1_NPM_USER="" # NPM admin email
HOST1_NPM_PASS="" # NPM admin password
# ━━━ lldap ━━━
HOST1_LLDAP_URL="http://localhost:17170"
HOST1_LLDAP_USER="admin" # lldap admin username
HOST1_LLDAP_PASS="" # lldap admin password
# ━━━ Authelia ━━━
HOST1_AUTHELIA_CONFIG="/mnt/user/appdata/Authelia/configuration.yml"
HOST1_AUTHELIA_CONTAINER="Authelia"
# ==============================================================================================
# ──────────────────────── End Of HOST1 Variables ──────────────────────────────────────────────
# ==============================================================================================
@@ -1,58 +0,0 @@
---
name: project_varaverk
description: Varaverk — self-healing mutually-redundant two-server Unraid home media ecosystem
metadata:
node_type: memory
type: project
originSessionId: ffec43cd-13e3-4911-878f-40459f7d16a9
---
**Varaverk** is a complete self-healing, self-maintaining, mutually-redundant two-server home server ecosystem. One codebase runs on both servers. No primary/standby — both servers run independently and cover each other when one goes down.
## The Two Servers
**HOST1 — unRAID-Gmer4Lfe**
- Hardware: Threadripper 1950X, 128GB RAM, ZFS cache pools
- Location: Primary site
- Domain: Gmer4Lfe.com
- Runs: Arrs (Movies, TV, Music), Auth stack (source of truth), Emby (primary)
**HOST2 — unRAID-Jayred365**
- Hardware: Intel i5 10th gen, 64GB RAM
- Location: Remote — different building, different power utility
- Domain: Gmer4Lfe.us
## Architecture
- Platform adapter layer (Plugin/unraid/adapter.sh) isolates OS-specific calls — scripts never branch on OS
- Self-healing, not enterprise HA — goal is minimal media stack disruption
- Tailscale for mesh networking between hosts
## Session State — 2026-06-13
**What was done this session:**
- New Claude Code install after a reinstall. Old data was at /boot/config/claude and /boot/config/claude-bin.
- Memory files restored from old install into current install.
- claude_startup.sh run manually — created claude-data and claude-bin dirs under /boot/config/plugins/varaverk/, migrated all data, symlinks confirmed working.
- Verified Varaverk is fully running from /boot — nothing in appdata. varaverk.cfg SCRIPTS_DIR, DATA_DIR, STATE_DIR, all point to /boot/config/plugins/varaverk.
- No code changes made — session was setup/verification only.
**Stale note in .plg:** The ###2026.05.31 CHANGES entry says "Scripts are git-cloned to appdata on first install" — this is wrong, the actual code clones to /boot/config/plugins/varaverk. Worth fixing on next package build.
**Flash wear note:** /boot is on USB flash (flash/boot). HOST1_STORAGE_MODE_INTERNAL=true was designed for NVMe/SSD boot. Git writes, logs, and claude data all land on flash — acceptable for now but worth migrating boot to NVMe eventually.
## Active To-Dos (from Notes_To-Do.md)
- Fix fallback strike list timing: ~30s first, ~90s for 3-strike trigger — needs testing
- Verify silent toggle switches back on good notifications
- Rename folder Unraid_Scripts → Varaverk everywhere, update git script, all traces/scripts
- Delete old /boot/config/claude and /boot/config/claude-bin dirs (migrated, no longer needed)
## Future Design Ideas
- Shared auth stack for partner hosts to start shared services
- When owner offboards with 2+ servers: auto-promote strongest server (by compute + bandwidth)
- Overall setup script that pulls vars automatically (docker names, etc.)
- App layer as king: no more direct git — app opens/edits settings, partnership deploys to servers, pushes correct host.conf
- Web UI: on initial launch with no state file, open master.conf; lock orchs until setup complete
- First-launch guide: owner sets up master.conf → host1.conf → Tailscale shares → onboard → host2/3 install and see state file, default to mirror mode
**Why:** User is building this as a personal project on Unraid. Design philosophy favors simplicity and independence over enterprise tooling.
**How to apply:** Understand the two-server mesh model when suggesting architecture. The app layer / web UI direction is the current strategic focus — moving away from raw git/scripts toward a proper application.
@@ -1,60 +0,0 @@
---
name: project_varaverk
description: Varaverk — self-healing mutually-redundant two-server Unraid home media ecosystem
metadata:
node_type: memory
type: project
originSessionId: ffec43cd-13e3-4911-878f-40459f7d16a9
---
**Varaverk** is a complete self-healing, self-maintaining, mutually-redundant two-server home server ecosystem. One codebase runs on both servers. No primary/standby — both servers run independently and cover each other when one goes down.
## The Two Servers
**HOST1 — unRAID-Gmer4Lfe**
- Hardware: Threadripper 1950X, 128GB RAM, ZFS cache pools
- Location: Primary site
- Domain: Gmer4Lfe.com
- Runs: Arrs (Movies, TV, Music), Auth stack (source of truth), Emby (primary)
**HOST2 — unRAID-Jayred365**
- Hardware: Intel i5 10th gen, 64GB RAM
- Location: Remote — different building, different power utility
- Domain: Gmer4Lfe.us
## Architecture
- Platform adapter layer (Plugin/unraid/adapter.sh) isolates OS-specific calls — scripts never branch on OS
- Self-healing, not enterprise HA — goal is minimal media stack disruption
- Tailscale for mesh networking between hosts
## Session State — 2026-06-13
**What was done this session:**
- New Claude Code install after a reinstall. Old data was at /boot/config/claude and /boot/config/claude-bin.
- Memory files restored from old install into current install.
- claude_startup.sh run manually — created claude-data and claude-bin dirs under /boot/config/plugins/varaverk/, migrated all data, symlinks confirmed working.
- Verified Varaverk is fully running from /boot — nothing in appdata. varaverk.cfg SCRIPTS_DIR, DATA_DIR, STATE_DIR, all point to /boot/config/plugins/varaverk.
- No code changes made — session was setup/verification only.
**Stale note in .plg:** The ###2026.05.31 CHANGES entry says "Scripts are git-cloned to appdata on first install" — this is wrong, the actual code clones to /boot/config/plugins/varaverk. Worth fixing on next package build.
**Flash wear note:** /boot is on USB flash (flash/boot). HOST1_STORAGE_MODE_INTERNAL=true was designed for NVMe/SSD boot. Git writes, logs, and claude data all land on flash — acceptable for now but worth migrating boot to NVMe eventually.
**Plugin install flow:** Plugin installs to appdata first, then during the setup wizard the user can select "normal" or set `internal_boot=true` to pin it to /boot. This is why the .plg note about appdata isn't wrong per se — it's the staging location before the wizard runs.
## Active To-Dos (from Notes_To-Do.md)
- Fix fallback strike list timing: ~30s first, ~90s for 3-strike trigger — needs testing
- Verify silent toggle switches back on good notifications
- Rename folder Unraid_Scripts → Varaverk everywhere, update git script, all traces/scripts
- Delete old /boot/config/claude and /boot/config/claude-bin dirs (migrated, no longer needed)
## Future Design Ideas
- Shared auth stack for partner hosts to start shared services
- When owner offboards with 2+ servers: auto-promote strongest server (by compute + bandwidth)
- Overall setup script that pulls vars automatically (docker names, etc.)
- App layer as king: no more direct git — app opens/edits settings, partnership deploys to servers, pushes correct host.conf
- Web UI: on initial launch with no state file, open master.conf; lock orchs until setup complete
- First-launch guide: owner sets up master.conf → host1.conf → Tailscale shares → onboard → host2/3 install and see state file, default to mirror mode
**Why:** User is building this as a personal project on Unraid. Design philosophy favors simplicity and independence over enterprise tooling.
**How to apply:** Understand the two-server mesh model when suggesting architecture. The app layer / web UI direction is the current strategic focus — moving away from raw git/scripts toward a proper application.
@@ -1,664 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ========================== HOST2 CONFIGURATION — unRAID-Jayred365 ===========================
# ==============================================================================================
# HOST2-specific variables — credentials, container names, share paths, failover lists.
# Sourced after master.conf — values here extend shared profile arrays and add HOST2-specific
# identity, credentials, and container configuration.
#
# Sparse checkout (git) ensures HOST1 never receives this file.
# HOST1 never sees HOST2 credentials — clean separation at the file level.
#
# DO NOT put shared config here — thresholds, toggles, profiles belong in master.conf.
# DO NOT put HOST1 variables here — they belong in host1.conf.
#
# ── STATUS ────────────────────────────────────────────────────────────────────────────────────
# HOST2 is currently being rebuilt — most sections scaffolded, fill in when back online.
# When ready: set FALLBACK_ENABLED=true and DAILY_RSYNC_ENABLED=true in master.conf.
#
# ── INDEX ─────────────────────────────────────────────────────────────────────────────────────
#
# ── IDENTITY & CONNECTIVITY ────────────────────────────────────────────────────────────────
# IDENTITY hostname, SSH key
# EMBY container name, URL, API key
# NOTIFICATIONS Discord webhook
# PARTNERSHIP auth containers, backup paths
#
# ── RSYNC ──────────────────────────────────────────────────────────────────────────────────
# DAILY SYNC SHARES media shares HOST2 owns and pushes to HOST1
# WEEKLY SYNC SHARES appdata shares synced weekly (Sunday 2:30am)
# CRITICAL SYNC SHARES appdata shares synced every 30 minutes
# BACKUP VERIFY shares for checksum verification against remote
# HOST2 RSYNC PROFILE host2-appdata profile for HOST2-specific appdata syncs
#
# ── DOCKER ─────────────────────────────────────────────────────────────────────────────────
# DOCKER DAILY RESTART containers restarted daily
# DOCKER WEEKLY RESTART containers restarted weekly
# DOCKER WATCHDOG memory limits, health URLs, required containers, ignore list
# DOCKER NETWORK CONNECT networks and containers for docker_network_connect.sh
#
# ── FALLBACK ───────────────────────────────────────────────────────────────────────────────
# DDNS DDNS containers managed by HOST2
# INTERNET LOSS containers stopped when internet is lost
# FALLBACK TIERS what HOST2 runs for HOST1 per tier
# TIER DELAYS how long HOST2 must be down before each tier activates on HOST1
# RSYNC WRITEBACK HOST2 appdata synced back on handback
#
# ── MEDIA ──────────────────────────────────────────────────────────────────────────────────
# MEDIA PERMISSIONS share list for media_shares_permissions.sh
# MEDIA CLEANER folder lists for media_cleaner.sh
#
# ── MONITORS ───────────────────────────────────────────────────────────────────────────────
# CERTIFICATE MONITOR domains checked for SSL expiry
# SMART HEALTH drives to skip in SMART monitoring
# ZFS REPORT pools to exclude from ZFS health report
#
# ── TRANSCODES ─────────────────────────────────────────────────────────────────────────────
# TRANSCODES ramdisk size, thresholds, SSD path, server array
#
# ── ARR STACK ──────────────────────────────────────────────────────────────────────────────
# SONARR URL, API key, path map
# RADARR URL, API key, path map
# ARR RECOVERY per-arr recovery toggles (no Lidarr on HOST2)
#
# ── SYSTEM WATCHDOG ────────────────────────────────────────────────────────────────────────
# SYSTEM WATCHDOG per-host check toggles and NIC configuration
#
# ── RESOURCE MANAGER ───────────────────────────────────────────────────────────────────────
# RESOURCE MANAGER containers paused/stopped under memory pressure
#
# ==============================================================================================
# ==============================================================================================
# ── IDENTITY & CONNECTIVITY ───────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Identity ━━━
# HOST2 hostname lives in master.conf (not a credential — safe for all servers).
# SSH key used for all server-to-server operations — rsync, failover container commands.
# Must be in /root/.ssh/ and authorised in HOST1's /root/.ssh/authorized_keys.
HOST2_SSH_KEY="/root/.ssh/Jayred365-rsync-key"
HOST2_OWNER="jayred365"
HOST2_OWNER_EMAIL="" # fill in when HOST2 is back online
# ━━━ Unraid API ━━━
# Generate in Unraid: Settings → Management Access → API Keys → + New Key
HOST2_UNRAID_API_KEY="2bdf5119d61eefa3023434748bd1c171bd23dc0b2ebc8586e24abe07df986acc"
# ━━━ Emby ━━━
# Referenced by transcode_manager.sh, emby_session_report.sh, emby_database_repair.sh,
# weekly_sync_maintenance.sh, and HOST2_TRANSCODE_SERVERS below.
# API key: Emby Dashboard → API Keys → + New Key
HOST2_EMBY_CONTAINER="Emby-Jayred365"
HOST2_EMBY_URL="http://localhost:8096"
HOST2_EMBY_API_KEY="your-host2-emby-api-key"
# ━━━ Jellyfin ━━━
# API key: Jellyfin Dashboard → Administration → API Keys → + New Key
HOST2_JELLYFIN_CONTAINER="Jellyfin"
HOST2_JELLYFIN_URL="http://localhost:8095"
HOST2_JELLYFIN_API_KEY="956d0168987f4e4680626653abb080f0"
# ━━━ Notifications ━━━
# Discord webhook — leave blank to disable.
# Per-host so HOST1 and HOST2 can post to different channels or only one server notifies.
HOST2_DISCORD_WEBHOOK=""
# ━━━ Partnership ━━━
# HOST2 is the mirror — HOST1 is always the owner unless --transfer has been run.
# See README-Partnership.md and master.conf PARTNERSHIP section for full lifecycle docs.
# Auth containers reconfigured on onboard/offboard.
# Format: "ContainerName|WebUIPort"
# On onboard → WebUI pointed at owner's Tailscale IP (mirror clicks NPM, gets owner's auth)
# On offboard → WebUI pointed back at localhost
HOST2_PARTNERSHIP_AUTH_WEBUIS=(
# fill in when HOST2 is back online
# "NginxProxyManager|81"
)
# Containers to stop on this server before the owner deploys the auth stack during onboard.
# List whatever auth/proxy containers are currently running here.
HOST2_PARTNERSHIP_REPLACE_CONTAINERS=(
"NginxProxyManager"
"Authelia"
"Authelia-Secondary"
"Mariadb-Authelia"
"Mariadb-Authelia-Secondary"
"Redis-Authelia"
"Redis-Authelia-Secondary"
"Lldap-Gmer4Lfe"
)
# Arr containers to stop on this server before the owner deploys the arr stack during onboard.
HOST2_PARTNERSHIP_ARR_REPLACE_CONTAINERS=(
# "Sonarr"
# "Radarr"
# "Lidarr"
# "Prowlarr"
# "Bazarr"
)
# Paths HOST1 should collect during the grace window after offboard.
# Notified on offboard — no auto-deletion, HOST1 must collect manually within PARTNERSHIP_GRACE_HOURS.
HOST2_PARTNERSHIP_MIRROR_BACKUPS=(
# fill in when HOST2 is back online
)
# Containers parked on this server when partnership is active.
# Stopped on onboard (owner deploys its stack instead), restarted on offboard.
HOST2_PARTNERSHIP_OWN_CONTAINERS=(
# "Emby"
# "NginxProxyManager"
)
# This server's desired Emby admin account on the shared Emby instance.
# Set these — owner reads them during --onboard to create the account.
HOST2_PARTNERSHIP_EMBY_ADMIN_USER="" # desired Emby username
HOST2_PARTNERSHIP_EMBY_ADMIN_PASS="" # desired Emby password
# ==============================================================================================
# ── RSYNC ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Daily Sync Shares ━━━
# Shares HOST2 pushes to all other nodes every night (1am via daily_sync_maintenance.sh).
# Mesh model: every node pushes every media share — no ownership, no mirrors.
# arr_sync ensures all arr libraries converge (union). rsync spreads files (additive, no --delete).
# arr_cleanup removes true orphans based on local arr state.
# Any node can download content to any share — it propagates to all nodes on the next cycle.
# Nextcloud excluded — personal data, not arr-managed, synced HOST1→HOST2 only as offsite backup.
# Uses DEFAULT_RSYNC_OPTS from master.conf — no profile needed.
# For shares needing container stops or custom options — add a profile in master.conf.
HOST2_DAILY_SYNC_SHARES=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Shows
/mnt/user/Books
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/stand-up_comedy
/mnt/user/Sports
/mnt/user/Tv_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/Anime_Movies-Old
)
# Personal encrypted shares — synced for offsite backup, independent of media shares.
# ZFS encrypted at dataset level — remote receives encrypted blocks, cannot read content.
# See README-Rsync_Setup.md for ZFS encryption setup before uncommenting.
HOST2_PERSONAL_SHARES=(
# /mnt/user/HOST2-Personal # uncomment after creating encrypted dataset
)
# ━━━ Weekly Sync Shares ━━━
# Appdata shares synced during the weekly maintenance window (Sunday 2:30am).
# Containers stopped both sides before sync — full clean state guaranteed.
# Profiles drive container stops, excludes, and options — configured in master.conf RSYNC section.
HOST2_WEEKLY_SYNC_SHARES=(
# fill in when HOST2 is back online
# "/mnt/user/Media_Server/Emby"
# "/mnt/user/appdata-Fallback/Critical-Data"
)
# ━━━ Intermediate Sync Shares ━━━
# Shares synced every 4 hours by intermediate_sync_maintenance.sh.
# Uses DEFAULT_RSYNC_OPTS (no --delete) — for sub-daily propagation of metadata or watch state.
# Full media share sync stays in the daily window. Leave empty to skip mid-day rsync entirely.
HOST2_INTERMEDIATE_SYNC_SHARES=(
# fill in when HOST2 is back online
# Example: "/mnt/user/Emby_Metadata"
)
# ━━━ Critical Sync Shares ━━━
# Appdata shares synced every 30 minutes by critical_sync_maintenance.sh.
# Format: "/path/to/share" or "/path/to/share|profile-name"
HOST2_CRITICAL_SYNC_SHARES=(
# fill in when HOST2 is back online
# "/mnt/user/appdata-Fallback/Critical-Data|critical-fallback"
# "/mnt/user/Media_Server/Emby|emby-fallback"
)
# ━━━ Backup Verify ━━━
# Shares verified by backup_verify.sh — random file checksum comparison against remote.
# Leave empty to use HOST2_DAILY_SYNC_SHARES automatically.
# Sample size and minimum file size defined in master.conf.
HOST2_BACKUP_VERIFY_SHARES=(
# leave empty to use HOST2_DAILY_SYNC_SHARES automatically
)
# ━━━ HOST2 Rsync Profile — host2-appdata ━━━
# HOST2-specific appdata sync profile — extends the shared PROFILE_* arrays in master.conf.
# Use for appdata unique to HOST2.
# Shared appdata (auth stack, Emby) use dedicated profiles defined in master.conf.
# Run manually: bash Rsync/rsync.sh /mnt/user/appdata-Fallback/HOST2-Appdata --profile=host2-appdata
PROFILE_RSYNC_OPTS[host2-appdata]="-av --info=progress2 --bwlimit=${PROFILE_BW_LIMIT[host2-appdata]:-8000}"
PROFILE_BW_LIMIT[host2-appdata]=8000
PROFILE_RETRY_COUNT[host2-appdata]=3
PROFILE_SLEEP[host2-appdata]=300
PROFILE_CRITICAL_CONTAINER_NAMES[host2-appdata]="" # fill in when HOST2 is back online
PROFILE_DELAYED_CONTAINERS[host2-appdata]=""
PROFILE_CONTAINER_DELAY[host2-appdata]=5
PROFILE_EXCLUDE_DIRS[host2-appdata]="logs *.tmp"
# ==============================================================================================
# ── DOCKER ────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Docker Daily Restart ━━━
# Containers restarted every day via DAILY_MAINTENANCE_SCRIPTS.
# Fill in when HOST2 is back online — add containers that degrade without daily restart.
HOST2_DAILY_RESTART_CONTAINERS=(
"NginxProxyManager"
# add HOST2 daily restart containers here
)
# ━━━ Docker Weekly Restart ━━━
# Less critical services restarted weekly via WEEKLY_MAINTENANCE_SCRIPTS (Sunday 2:30am).
# Containers already stopped for weekly sync — restart adds zero extra downtime.
HOST2_WEEKLY_RESTART_CONTAINERS=(
# add HOST2 weekly restart containers here
)
# ━━━ Docker Watchdog ━━━
# Per-HOST2 container configuration for docker_watchdog.sh.
# Shared thresholds and toggles live in master.conf.
# Memory hard limits in MB — immediate restart if exceeded.
# Set at "container is clearly broken" not "container is busy".
# 20GB=20480 16GB=16384 12GB=12288 10GB=10240 8GB=8192 4GB=4096 2GB=2048 1GB=1024
declare -A HOST2_WATCHDOG_CONTAINERS=(
["Emby"]=16384 # fill in correct limit when HOST2 is back online
)
# HTTP health check URLs — checked every cycle, strike system before restart.
# Only add containers with a meaningful web interface to check.
declare -A HOST2_WATCHDOG_CONTAINER_URLS=(
["Emby"]="http://localhost:8096"
)
# Required containers — must always be running on HOST2.
# Strike system before restart — repeated failures go on skip list, auto-clears on recovery.
# Listed in dependency order — dependencies before dependents.
HOST2_WATCHDOG_REQUIRED_CONTAINERS=(
"NginxProxyManager"
# add HOST2 required containers here when back online
)
# Containers to skip in Tier 2 global scan — legitimately stopped or frequently restarting.
# Watchdog leaves these alone entirely — no restart attempts, no crash loop tracking.
HOST2_WATCHDOG_SCAN_IGNORE=(
# add HOST2 scan ignore containers here when back online
)
# Dependency ordering — skip restarting a container if its dependency is also down.
# Prevents watchdog from restarting dependent services before their dependencies are up.
# SPACE-SEPARATED STRINGS — converted to array at runtime.
declare -A HOST2_WATCHDOG_DEPENDENCIES=(
# add HOST2 dependencies here when containers are defined
# ["Authelia"]="Mariadb-Authelia Redis-Authelia"
)
# Per-container appdata growth suppress ceilings in MB.
# ONLY needed in specific cases — growth rate detection covers all containers automatically.
# Use when a container legitimately has large stable data and you want to suppress false-positive
# growth alerts. Add entries here only when a container triggers warnings it shouldn't.
declare -A HOST2_WATCHDOG_APPDATA_SIZES=(
# add HOST2 suppress entries here only as needed
)
# ━━━ Docker Network Connect ━━━
# Containers connected to custom networks at array start by docker_network_connect.sh.
# Networks created if they don't exist — idempotent, safe to re-run.
HOST2_NETWORK_CONNECT_CONTAINERS=(
# fill in when HOST2 is back online
)
HOST2_NETWORK_CONNECT_NETWORKS=(
# fill in when HOST2 is back online
)
# ==============================================================================================
# ── FALLBACK ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ DDNS ━━━
# DDNS containers HOST2 manages — started/stopped by fallback.sh per DDNS absolute rules:
# Internet loss → stop immediately
# Failover → HOST1 starts HOST2's DDNS as Tier 1 (before any other containers)
# Handback → stop HOST2's DDNS on HOST1 → rsync → start containers → start local DDNS last
HOST2_DDNS_CONTAINERS=(
"Gmer4Lfe.us"
)
# ━━━ Internet Loss ━━━
# Containers stopped immediately on HOST2 when internet connection is lost.
# Prevents external-facing services from operating without connectivity.
FALLBACK_HOST2_STOP_ON_NO_NET=(
"Gmer4Lfe.us"
)
# ━━━ Fallback Tiers — HOST2 Runs for HOST1 ━━━
# Containers HOST2 starts when HOST1 goes down.
# Tier 1 is always immediate — vital services cannot wait.
# Higher tiers activate after HOST1_TIER*_DELAY minutes (set in host1.conf).
FALLBACK_HOST2_COVERS_HOST1_TIER1=(
"Gmer4Lfe.com"
"Gitea" # source of truth — must be reachable even when HOST1 auth stack is down
"Emby"
"VaultWarden-Gmer4Lfe"
"Dispatcharr"
"Dispatcharr-Basic"
"Dispatcharr-Iptv-Users"
"ErsatzTV-Emby"
)
FALLBACK_HOST2_COVERS_HOST1_TIER2=(
"Postgres-NextCloud"
"NextCloud"
"PostgreSQL_Immich"
"Immich-Gmer4Lfe"
)
FALLBACK_HOST2_COVERS_HOST1_TIER3=(
"Gitea"
)
FALLBACK_HOST2_COVERS_HOST1_TIER4=(
"Sonarr"
"Radarr"
"Lidarr"
"Readarr"
"Prowlarr"
"Bazarr"
"SABnzbd-Gmer4Lfe"
"Qbittorrent-Gmer4Lfe"
"LidaTube"
"Pinchflat"
"ChannelTube"
)
# ━━━ Tier Delays — HOST2's Containers on HOST1 ━━━
# How long HOST2 must be down before each tier activates on HOST1 — in minutes.
# Tier 1 is always immediate — no delay var needed.
HOST2_TIER2_DELAY=240 # 4 hours — productivity services
HOST2_TIER3_DELAY=720 # 12 hours — secondary services
HOST2_TIER4_DELAY=1440 # 24 hours — arrs + downloaders
# ━━━ Rsync Writeback — HOST2 Appdata Back on Handback ━━━
# Syncs HOST2 appdata BACK to HOST2 when it comes back online after a failover.
# Containers stopped before writeback — clean source, no competing writes.
#
# HOST2_TIER1_WRITEBACK_DELAY: short outages skip Tier 1 writeback — primary state
# is more reliable than dirty sync data for brief outages.
HOST2_TIER1_WRITEBACK_DELAY=60 # skip writeback if outage under 1hr
# Tier 4 automatically syncs HOST2_DAILY_SYNC_SHARES — only list paths NOT in that array.
FALLBACK_HOST2_WRITEBACK_TIER1=(
# "/mnt/user/appdata-Fallback/Jayred365-Emby"
)
FALLBACK_HOST2_WRITEBACK_TIER2=(
# "/mnt/user/appdata-Fallback/Jayred365-Important"
)
FALLBACK_HOST2_WRITEBACK_TIER3=(
# "location-placeholder"
)
FALLBACK_HOST2_WRITEBACK_TIER4=(
"/mnt/user/appdata-Fallback/Arrs_Stack" # arr databases — downloads queued during outage
)
# ==============================================================================================
# ── MEDIA ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Media Permissions ━━━
# Shares that media_shares_permissions.sh applies PERMISSIONS_MODE and PERMISSIONS_OWNER to.
# Runs first in DAILY_MAINTENANCE_SCRIPTS — arr cleanup depends on correct ownership.
HOST2_MEDIA_PERMISSION_SHARES=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Shows
)
# ━━━ Media Cleaner ━━━
# Folder lists for media_cleaner.sh — two profiles: anime and media.
# File patterns shared across all servers — defined in master.conf.
# Called via DAILY_MAINTENANCE_SCRIPTS. Run manually: Media/media_cleaner.sh anime|media
HOST2_ANIME_CLEAN_FOLDERS=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Shows
)
HOST2_MEDIA_CLEAN_FOLDERS=(
# fill in when HOST2 is back online
)
# ==============================================================================================
# ── MONITORS ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
# Domains checked via direct openssl connection — not relying on NPM's certificate state.
# Checks the actual certificate served, not what NPM thinks it has.
# Thresholds (CERT_WARN_DAYS, CERT_CRIT_DAYS) defined in master.conf.
HOST2_CERT_MONITOR_DOMAINS=(
# fill in when HOST2 is back online
)
# ━━━ SMART Health ━━━
# Drives skipped in SMART attribute monitoring — hardware is server-specific.
# Thresholds read from dynamix.cfg at runtime — fallbacks in master.conf.
HOST2_SMART_IGNORE_DRIVES=(
"sda" # boot USB — SMART not meaningful on flash drives
)
# ━━━ ZFS Report ━━━
# Pools excluded from the weekly ZFS health report — reduces noise from single-disk array pools.
# Pool health thresholds defined in master.conf.
HOST2_ZFS_REPORT_IGNORE_POOLS=(
# fill in when HOST2 is back online
)
# ==============================================================================================
# ── TRANSCODES ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Ramdisk size ceiling — tmpfs only uses RAM actually needed, not the full size upfront.
# Adjust HOST2_RAMDISK_WARN_GB and HOST2_RAMDISK_LOW_GB together if this changes.
HOST2_RAMDISK_SIZE="8G"
# Usage thresholds — coupled to HOST2_RAMDISK_SIZE, adjust all three together if size changes.
# Hysteresis gap (6.8 - 5.5 = 1.3GB) prevents flip-flop between ramdisk and SSD.
HOST2_RAMDISK_WARN_GB=6.8 # flip to SSD when ramdisk usage reaches this
HOST2_RAMDISK_LOW_GB=5.5 # flip back to ramdisk when usage drops to this
# SSD fallback path — where transcodes land when ramdisk exceeds HOST2_RAMDISK_WARN_GB.
# Must be on cache pool — array disks too slow for active transcode writes.
HOST2_TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/"
# Media servers sharing the ramdisk transcode space on HOST2.
# Format: "ContainerName|URL|APIKey|Type" — Type: emby | jellyfin | plex
# Entries with placeholder API keys are skipped automatically.
# ⚠️ Tdarr does NOT belong here — keep Tdarr on SSD, not ramdisk.
HOST2_TRANSCODE_SERVERS=(
"${HOST2_EMBY_CONTAINER}|${HOST2_EMBY_URL}|${HOST2_EMBY_API_KEY}|emby"
"${HOST2_JELLYFIN_CONTAINER}|${HOST2_JELLYFIN_URL}|${HOST2_JELLYFIN_API_KEY}|jellyfin"
)
# ==============================================================================================
# ── ARR STACK ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Used by arr cleanup scripts and arrs_failed_stalled_recovery.sh.
# detect_hosts() selects HOST2 vars when running on HOST2.
# Lidarr does not run on HOST2 — HOST1_LIDARR_RECOVERY flag handles the exit cleanly.
#
# PATH MAPS — container path → host path translation.
# Arr stores file paths using container-internal paths — scripts need host paths to scan.
# Add one entry per root folder in arr Settings → Media Management → Root Folders.
HOST2_FANART_API_KEY="Yd7147a43b692df0b364b94dc47efb81"
HOST2_LASTFM_API_KEY="be6dc169c33ae263e690c30d18b7491d"
# ━━━ Sonarr ━━━
HOST2_SONARR_URL="http://localhost:8989"
HOST2_SONARR_API_KEY="130decd3db5b4c25afad64864cd03f9f"
HOST2_SONARR_TV_ROOT="/mnt/user/Anime_Shows"
declare -A HOST2_SONARR_PATH_MAP=(
# fill in when HOST2 is back online
# ["/tv"]="/mnt/user/Anime_Shows"
)
# ━━━ Radarr ━━━
HOST2_RADARR_URL="http://localhost:7878"
HOST2_RADARR_API_KEY="d43a3ec6cf1549edb4af0cc63f98b2a9"
HOST2_RADARR_MOVIES_ROOT="/mnt/user/Anime_Movies"
declare -A HOST2_RADARR_PATH_MAP=(
# fill in when HOST2 is back online
# ["/anime-movies"]="/mnt/user/Anime_Movies"
)
# ━━━ Arr Recovery Toggles ━━━
# false = skip that arr on this host — exits cleanly without error
HOST2_SONARR_RECOVERY=true
HOST2_RADARR_RECOVERY=true
# HOST2_LIDARR_RECOVERY not set — Lidarr does not run on HOST2
# ==============================================================================================
# ── SYSTEM WATCHDOG ───────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Per-host check toggles and NIC config for system_watchdog.sh.
# Aliased by detect_hosts() — script uses unprefixed SYS_WATCHDOG_* names.
# HOST2: i5 10th gen 64GB — being rebuilt, lighter workload, no ZFS cache pools.
#
# Conservative defaults during rebuild — re-enable checks as HOST2 stabilises.
# Three-tier response — all critical checks enabled regardless of rebuild state:
# Tier 1 (bypass strikes, reboot now): docker daemon, rootfs full, kernel oops, FD, /boot
# Tier 2 (bypass strikes with OOM): RAM critical + OOM kills in cycle
# Tier 3 (standard strike system): selectively disabled during rebuild
#
# RAM tiers, OOM limits, and reboot loop settings in master.conf System Watchdog section.
# ━━━ Primary NIC ━━━
# Network interface for NIC state check — verify with: ip link show | grep "^[0-9]"
# Common values: eth0, bond0, br0, eno1
HOST2_SYS_WATCHDOG_NIC="eth0"
# ━━━ Tier 1 — Critical Checks ━━━
# All critical checks always enabled — these protect against acute failure regardless of
# rebuild state. Disabling any is not recommended.
# Docker daemon unresponsive → try restart, reboot if restart fails.
HOST2_SYS_WATCHDOG_CHECK_DOCKER_DAEMON=true
# rootfs at critical threshold (ROOTFS_CRITICAL_PCT=99) → reboot immediately.
HOST2_SYS_WATCHDOG_CHECK_ROOTFS=true
# Kernel BUG/Oops in dmesg delta since last cycle → reboot immediately.
HOST2_SYS_WATCHDOG_CHECK_KERNEL_OOPS=true
# File descriptor exhaustion at FD_CRITICAL_PCT (95%) → reboot immediately.
HOST2_SYS_WATCHDOG_CHECK_FD=true
# /boot read-only detected → reboot immediately.
HOST2_SYS_WATCHDOG_CHECK_BOOT=true
# ━━━ Tier 2 — Urgent OOM Check ━━━
# Both must be enabled for Tier 2 bypass to function.
# Track kernel OOM kills each cycle via /proc/vmstat oom_kill delta.
HOST2_SYS_WATCHDOG_CHECK_OOM=true
# Free RAM check — 64GB RAM on HOST2, tiers adjusted relative to HOST1.
# Update master.conf SYS_WATCHDOG_MEM_* thresholds if HOST2 needs different values.
# Currently inheriting shared master.conf values — may want lower thresholds on 64GB.
HOST2_SYS_WATCHDOG_CHECK_RAM=true
# ━━━ Tier 3 — Standard Checks (strike system) ━━━
# Several checks disabled during rebuild — enable progressively as HOST2 stabilises.
# Each check must fail SYS_WATCHDOG_STRIKE_LIMIT consecutive cycles before action.
# /var/log filesystem usage above SYS_WATCHDOG_LOG_PCT.
HOST2_SYS_WATCHDOG_CHECK_LOG=true
# ZFS ARC memory check.
# DISABLED — HOST2 has no ZFS cache pools. Enable if ZFS pools are added later.
HOST2_SYS_WATCHDOG_CHECK_ARC=false
# CPU temperature above SYS_WATCHDOG_CPU_TEMP_MAX (95°C).
HOST2_SYS_WATCHDOG_CHECK_CPU_TEMP=true
# Load average above SYS_WATCHDOG_LOAD_MULTIPLIER × core count.
# DISABLED — rebuild operations cause legitimate load spikes. Enable after rebuild.
HOST2_SYS_WATCHDOG_CHECK_LOAD=false
# Zombie process count above SYS_WATCHDOG_ZOMBIE_LIMIT (50).
HOST2_SYS_WATCHDOG_CHECK_ZOMBIES=true
# docker_watchdog.sh persistent skip list check.
# DISABLED during rebuild — skip list may be unreliable mid-rebuild, avoid false reboots.
# Enable once HOST2 is fully operational and docker_watchdog.sh is running stably.
HOST2_SYS_WATCHDOG_CHECK_CONTAINERS=false
# /tmp filesystem usage with auto-clear attempt.
HOST2_SYS_WATCHDOG_CHECK_TMP=true
# Array disk error count delta in /proc/mdstat.
HOST2_SYS_WATCHDOG_CHECK_MDSTAT=true
# Primary NIC operstate — uses HOST2_SYS_WATCHDOG_NIC above.
HOST2_SYS_WATCHDOG_CHECK_NETWORK=true
# sshd running check — restart attempt before escalating.
HOST2_SYS_WATCHDOG_CHECK_SSHD=true
# Runaway process detection.
# DISABLED — rebuild workloads may legitimately peg CPU. Enable after rebuild.
HOST2_SYS_WATCHDOG_CHECK_RUNAWAY=false
# ==============================================================================================
# ── RESOURCE MANAGER ──────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Containers to manage under pressure — see master.conf RW_CRITICAL_CONTAINERS for exclusions.
# docker pause at medium pressure (RAM < RW_RAM_MEDIUM_GB or load > medium threshold)
# Suspended in-place — instant to pause/unpause, no state lost, no restart delay.
HOST2_RW_PAUSE_CONTAINERS=(
# fill in when HOST2 is back online
)
# docker stop at hard pressure (RAM < RW_RAM_HARD_GB)
# Full stop — these are optional/heavy services that free significant RAM when stopped.
# resource_watchdog.sh restarts them when pressure fully clears (RAM >= RW_RAM_RECOVER_GB).
HOST2_RW_STOP_CONTAINERS=(
# fill in when HOST2 is back online
)
# ==============================================================================================
# ── AUTH STACK ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Credentials for the Varaverk Auth Stack page (NPM, lldap, Authelia).
# Fill in when HOST2 is back online.
# ━━━ NginxProxyManager ━━━
HOST2_NPM_URL="http://localhost:81"
HOST2_NPM_USER="" # NPM admin email
HOST2_NPM_PASS="" # NPM admin password
# ━━━ lldap ━━━
HOST2_LLDAP_URL="http://localhost:17170"
HOST2_LLDAP_USER="admin" # lldap admin username
HOST2_LLDAP_PASS="" # lldap admin password
# ━━━ Authelia ━━━
HOST2_AUTHELIA_CONFIG="/mnt/user/appdata-Fallback/Critical-Data/Authelia/configuration.yml"
HOST2_AUTHELIA_CONTAINER="Authelia"
# ==============================================================================================
# ──────────────────────── End Of HOST2 Variables ──────────────────────────────────────────────
# ==============================================================================================
@@ -1,665 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ========================== HOST2 CONFIGURATION — unRAID-Jayred365 ===========================
# ==============================================================================================
# HOST2-specific variables — credentials, container names, share paths, failover lists.
# Sourced after master.conf — values here extend shared profile arrays and add HOST2-specific
# identity, credentials, and container configuration.
#
# Sparse checkout (git) ensures HOST1 never receives this file.
# HOST1 never sees HOST2 credentials — clean separation at the file level.
#
# DO NOT put shared config here — thresholds, toggles, profiles belong in master.conf.
# DO NOT put HOST1 variables here — they belong in host1.conf.
#
# ── STATUS ────────────────────────────────────────────────────────────────────────────────────
# HOST2 is currently being rebuilt — most sections scaffolded, fill in when back online.
# When ready: set FALLBACK_ENABLED=true and DAILY_RSYNC_ENABLED=true in master.conf.
#
# ── INDEX ─────────────────────────────────────────────────────────────────────────────────────
#
# ── IDENTITY & CONNECTIVITY ────────────────────────────────────────────────────────────────
# IDENTITY hostname, SSH key
# EMBY container name, URL, API key
# NOTIFICATIONS Discord webhook
# PARTNERSHIP auth containers, backup paths
#
# ── RSYNC ──────────────────────────────────────────────────────────────────────────────────
# DAILY SYNC SHARES media shares HOST2 owns and pushes to HOST1
# WEEKLY SYNC SHARES appdata shares synced weekly (Sunday 2:30am)
# CRITICAL SYNC SHARES appdata shares synced every 30 minutes
# BACKUP VERIFY shares for checksum verification against remote
# HOST2 RSYNC PROFILE host2-appdata profile for HOST2-specific appdata syncs
#
# ── DOCKER ─────────────────────────────────────────────────────────────────────────────────
# DOCKER DAILY RESTART containers restarted daily
# DOCKER WEEKLY RESTART containers restarted weekly
# DOCKER WATCHDOG memory limits, health URLs, required containers, ignore list
# DOCKER NETWORK CONNECT networks and containers for docker_network_connect.sh
#
# ── FALLBACK ───────────────────────────────────────────────────────────────────────────────
# DDNS DDNS containers managed by HOST2
# INTERNET LOSS containers stopped when internet is lost
# FALLBACK TIERS what HOST2 runs for HOST1 per tier
# TIER DELAYS how long HOST2 must be down before each tier activates on HOST1
# RSYNC WRITEBACK HOST2 appdata synced back on handback
#
# ── MEDIA ──────────────────────────────────────────────────────────────────────────────────
# MEDIA PERMISSIONS share list for media_shares_permissions.sh
# MEDIA CLEANER folder lists for media_cleaner.sh
#
# ── MONITORS ───────────────────────────────────────────────────────────────────────────────
# CERTIFICATE MONITOR domains checked for SSL expiry
# SMART HEALTH drives to skip in SMART monitoring
# ZFS REPORT pools to exclude from ZFS health report
#
# ── TRANSCODES ─────────────────────────────────────────────────────────────────────────────
# TRANSCODES ramdisk size, thresholds, SSD path, server array
#
# ── ARR STACK ──────────────────────────────────────────────────────────────────────────────
# SONARR URL, API key, path map
# RADARR URL, API key, path map
# ARR RECOVERY per-arr recovery toggles (no Lidarr on HOST2)
#
# ── SYSTEM WATCHDOG ────────────────────────────────────────────────────────────────────────
# SYSTEM WATCHDOG per-host check toggles and NIC configuration
#
# ── RESOURCE MANAGER ───────────────────────────────────────────────────────────────────────
# RESOURCE MANAGER containers paused/stopped under memory pressure
#
# ==============================================================================================
# ==============================================================================================
# ── IDENTITY & CONNECTIVITY ───────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Identity ━━━
# HOST2 hostname lives in master.conf (not a credential — safe for all servers).
# SSH key used for all server-to-server operations — rsync, failover container commands.
# Must be in /root/.ssh/ and authorised in HOST1's /root/.ssh/authorized_keys.
HOST2_SSH_KEY="/root/.ssh/Jayred365-rsync-key"
HOST2_OWNER="jayred365"
HOST2_OWNER_EMAIL="" # fill in when HOST2 is back online
# ━━━ Unraid API ━━━
# Generate in Unraid: Settings → Management Access → API Keys → + New Key
HOST2_UNRAID_API_KEY="2bdf5119d61eefa3023434748bd1c171bd23dc0b2ebc8586e24abe07df986acc"
# ━━━ Emby ━━━
# Referenced by transcode_manager.sh, emby_session_report.sh, emby_database_repair.sh,
# weekly_sync_maintenance.sh, and HOST2_TRANSCODE_SERVERS below.
# API key: Emby Dashboard → API Keys → + New Key
HOST2_EMBY_CONTAINER="Emby-Jayred365"
HOST2_EMBY_URL="http://localhost:8096"
HOST2_EMBY_API_KEY="your-host2-emby-api-key"
# ━━━ Jellyfin ━━━
# API key: Jellyfin Dashboard → Administration → API Keys → + New Key
HOST2_JELLYFIN_CONTAINER="Jellyfin"
HOST2_JELLYFIN_URL="http://localhost:8095"
HOST2_JELLYFIN_API_KEY="956d0168987f4e4680626653abb080f0"
# ━━━ Notifications ━━━
# Discord webhook — leave blank to disable.
# Per-host so HOST1 and HOST2 can post to different channels or only one server notifies.
HOST2_DISCORD_WEBHOOK=""
# ━━━ Partnership ━━━
# HOST2 is the mirror — HOST1 is always the owner unless --transfer has been run.
# See README-Partnership.md and master.conf PARTNERSHIP section for full lifecycle docs.
# Auth containers reconfigured on onboard/offboard.
# Format: "ContainerName|WebUIPort"
# On onboard → WebUI pointed at owner's Tailscale IP (mirror clicks NPM, gets owner's auth)
# On offboard → WebUI pointed back at localhost
HOST2_PARTNERSHIP_AUTH_WEBUIS=(
# fill in when HOST2 is back online
# "NginxProxyManager|81"
)
# Containers to stop on this server before the owner deploys the auth stack during onboard.
# List whatever auth/proxy containers are currently running here.
HOST2_PARTNERSHIP_REPLACE_CONTAINERS=(
"NginxProxyManager"
"Authelia"
"Authelia-Secondary"
"Mariadb-Authelia"
"Mariadb-Authelia-Secondary"
"Redis-Authelia"
"Redis-Authelia-Secondary"
"Lldap-Gmer4Lfe"
)
# Arr containers to stop on this server before the owner deploys the arr stack during onboard.
HOST2_PARTNERSHIP_ARR_REPLACE_CONTAINERS=(
# "Sonarr"
# "Radarr"
# "Lidarr"
# "Prowlarr"
# "Bazarr"
)
# Paths HOST1 should collect during the grace window after offboard.
# Notified on offboard — no auto-deletion, HOST1 must collect manually within PARTNERSHIP_GRACE_HOURS.
HOST2_PARTNERSHIP_MIRROR_BACKUPS=(
# fill in when HOST2 is back online
)
# Containers parked on this server when partnership is active.
# Stopped on onboard (owner deploys its stack instead), restarted on offboard.
HOST2_PARTNERSHIP_OWN_CONTAINERS=(
# "Emby"
# "NginxProxyManager"
)
# This server's desired Emby admin account on the shared Emby instance.
# Set these — owner reads them during --onboard to create the account.
HOST2_PARTNERSHIP_EMBY_ADMIN_USER="" # desired Emby username
HOST2_PARTNERSHIP_EMBY_ADMIN_PASS="" # desired Emby password
# ==============================================================================================
# ── RSYNC ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Daily Sync Shares ━━━
# Shares HOST2 pushes to all other nodes every night (1am via daily_sync_maintenance.sh).
# Mesh model: every node pushes every media share — no ownership, no mirrors.
# arr_sync ensures all arr libraries converge (union). rsync spreads files (additive, no --delete).
# arr_cleanup removes true orphans based on local arr state.
# Any node can download content to any share — it propagates to all nodes on the next cycle.
# Nextcloud excluded — personal data, not arr-managed, synced HOST1→HOST2 only as offsite backup.
# Uses DEFAULT_RSYNC_OPTS from master.conf — no profile needed.
# For shares needing container stops or custom options — add a profile in master.conf.
HOST2_DAILY_SYNC_SHARES=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Shows
/mnt/user/Books
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/stand-up_comedy
/mnt/user/Sports
/mnt/user/Tv_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/Anime_Movies-Old
)
# Personal encrypted shares — synced for offsite backup, independent of media shares.
# ZFS encrypted at dataset level — remote receives encrypted blocks, cannot read content.
# See README-Rsync_Setup.md for ZFS encryption setup before uncommenting.
HOST2_PERSONAL_SHARES=(
# /mnt/user/HOST2-Personal # uncomment after creating encrypted dataset
)
# ━━━ Weekly Sync Shares ━━━
# Appdata shares synced during the weekly maintenance window (Sunday 2:30am).
# Containers stopped both sides before sync — full clean state guaranteed.
# Profiles drive container stops, excludes, and options — configured in master.conf RSYNC section.
HOST2_WEEKLY_SYNC_SHARES=(
# fill in when HOST2 is back online
# "/mnt/user/Media_Server/Emby"
# "/mnt/user/appdata-Fallback/Critical-Data"
)
# ━━━ Intermediate Sync Shares ━━━
# Shares synced every 4 hours by intermediate_sync_maintenance.sh.
# Uses DEFAULT_RSYNC_OPTS (no --delete) — for sub-daily propagation of metadata or watch state.
# Full media share sync stays in the daily window. Leave empty to skip mid-day rsync entirely.
HOST2_INTERMEDIATE_SYNC_SHARES=(
# fill in when HOST2 is back online
# Example: "/mnt/user/Emby_Metadata"
)
# ━━━ Critical Sync Shares ━━━
# Appdata shares synced every 30 minutes by critical_sync_maintenance.sh.
# Format: "/path/to/share" or "/path/to/share|profile-name"
HOST2_CRITICAL_SYNC_SHARES=(
# fill in when HOST2 is back online
# "/mnt/user/appdata-Fallback/Critical-Data|critical-fallback"
# "/mnt/user/Media_Server/Emby|emby-fallback"
)
# ━━━ Backup Verify ━━━
# Shares verified by backup_verify.sh — random file checksum comparison against remote.
# Leave empty to use HOST2_DAILY_SYNC_SHARES automatically.
# Sample size and minimum file size defined in master.conf.
HOST2_BACKUP_VERIFY_SHARES=(
# leave empty to use HOST2_DAILY_SYNC_SHARES automatically
)
# ━━━ HOST2 Rsync Profile — host2-appdata ━━━
# HOST2-specific appdata sync profile — extends the shared PROFILE_* arrays in master.conf.
# Use for appdata unique to HOST2.
# Shared appdata (auth stack, Emby) use dedicated profiles defined in master.conf.
# Run manually: bash Rsync/rsync.sh /mnt/user/appdata-Fallback/HOST2-Appdata --profile=host2-appdata
PROFILE_RSYNC_OPTS[host2-appdata]="-av --info=progress2 --bwlimit=${PROFILE_BW_LIMIT[host2-appdata]:-8000}"
PROFILE_BW_LIMIT[host2-appdata]=8000
PROFILE_RETRY_COUNT[host2-appdata]=3
PROFILE_SLEEP[host2-appdata]=300
PROFILE_CRITICAL_CONTAINER_NAMES[host2-appdata]="" # fill in when HOST2 is back online
PROFILE_DELAYED_CONTAINERS[host2-appdata]=""
PROFILE_CONTAINER_DELAY[host2-appdata]=5
PROFILE_EXCLUDE_DIRS[host2-appdata]="logs *.tmp"
# ==============================================================================================
# ── DOCKER ────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Docker Daily Restart ━━━
# Containers restarted every day via DAILY_MAINTENANCE_SCRIPTS.
# Fill in when HOST2 is back online — add containers that degrade without daily restart.
HOST2_DAILY_RESTART_CONTAINERS=(
"NginxProxyManager"
# add HOST2 daily restart containers here
)
# ━━━ Docker Weekly Restart ━━━
# Less critical services restarted weekly via WEEKLY_MAINTENANCE_SCRIPTS (Sunday 2:30am).
# Containers already stopped for weekly sync — restart adds zero extra downtime.
HOST2_WEEKLY_RESTART_CONTAINERS=(
# add HOST2 weekly restart containers here
)
# ━━━ Docker Watchdog ━━━
# Per-HOST2 container configuration for docker_watchdog.sh.
# Shared thresholds and toggles live in master.conf.
# Memory hard limits in MB — immediate restart if exceeded.
# Set at "container is clearly broken" not "container is busy".
# 20GB=20480 16GB=16384 12GB=12288 10GB=10240 8GB=8192 4GB=4096 2GB=2048 1GB=1024
declare -A HOST2_WATCHDOG_CONTAINERS=(
["Emby"]=16384 # fill in correct limit when HOST2 is back online
)
# HTTP health check URLs — checked every cycle, strike system before restart.
# Only add containers with a meaningful web interface to check.
declare -A HOST2_WATCHDOG_CONTAINER_URLS=(
["Emby"]="http://localhost:8096"
)
# Required containers — must always be running on HOST2.
# Strike system before restart — repeated failures go on skip list, auto-clears on recovery.
# Listed in dependency order — dependencies before dependents.
HOST2_WATCHDOG_REQUIRED_CONTAINERS=(
"NginxProxyManager"
# add HOST2 required containers here when back online
)
# Containers to skip in Tier 2 global scan — legitimately stopped or frequently restarting.
# Watchdog leaves these alone entirely — no restart attempts, no crash loop tracking.
HOST2_WATCHDOG_SCAN_IGNORE=(
# add HOST2 scan ignore containers here when back online
)
# Dependency ordering — skip restarting a container if its dependency is also down.
# Prevents watchdog from restarting dependent services before their dependencies are up.
# SPACE-SEPARATED STRINGS — converted to array at runtime.
declare -A HOST2_WATCHDOG_DEPENDENCIES=(
# add HOST2 dependencies here when containers are defined
# ["Authelia"]="Mariadb-Authelia Redis-Authelia"
)
# Per-container appdata growth suppress ceilings in MB.
# ONLY needed in specific cases — growth rate detection covers all containers automatically.
# Use when a container legitimately has large stable data and you want to suppress false-positive
# growth alerts. Add entries here only when a container triggers warnings it shouldn't.
declare -A HOST2_WATCHDOG_APPDATA_SIZES=(
# add HOST2 suppress entries here only as needed
)
# ━━━ Docker Network Connect ━━━
# Containers connected to custom networks at array start by docker_network_connect.sh.
# Networks created if they don't exist — idempotent, safe to re-run.
HOST2_NETWORK_CONNECT_CONTAINERS=(
# fill in when HOST2 is back online
)
HOST2_NETWORK_CONNECT_NETWORKS=(
# fill in when HOST2 is back online
)
# ==============================================================================================
# ── FALLBACK ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ DDNS ━━━
# DDNS containers HOST2 manages — started/stopped by fallback.sh per DDNS absolute rules:
# Internet loss → stop immediately
# Failover → HOST1 starts HOST2's DDNS as Tier 1 (before any other containers)
# Handback → stop HOST2's DDNS on HOST1 → rsync → start containers → start local DDNS last
HOST2_DDNS_CONTAINERS=(
"Gmer4Lfe.us"
)
# ━━━ Internet Loss ━━━
# Containers stopped immediately on HOST2 when internet connection is lost.
# Prevents external-facing services from operating without connectivity.
FALLBACK_HOST2_STOP_ON_NO_NET=(
"Gmer4Lfe.us"
)
# ━━━ Fallback Tiers — HOST2 Runs for HOST1 ━━━
# Containers HOST2 starts when HOST1 goes down.
# Tier 1 is always immediate — vital services cannot wait.
# Higher tiers activate after HOST1_TIER*_DELAY minutes (set in host1.conf).
FALLBACK_HOST2_COVERS_HOST1_TIER1=(
"Gmer4Lfe.com"
"Gitea" # source of truth — must be reachable even when HOST1 auth stack is down
"Emby"
"VaultWarden-Gmer4Lfe"
"Dispatcharr"
"Dispatcharr-Basic"
"Dispatcharr-Iptv-Users"
"ErsatzTV-Emby"
)
FALLBACK_HOST2_COVERS_HOST1_TIER2=(
"Postgres-NextCloud"
"NextCloud"
"PostgreSQL_Immich"
"Immich-Gmer4Lfe"
)
FALLBACK_HOST2_COVERS_HOST1_TIER3=(
"Gitea"
)
FALLBACK_HOST2_COVERS_HOST1_TIER4=(
"Sonarr"
"Radarr"
"Lidarr"
"Readarr"
"Prowlarr"
"Bazarr"
"SABnzbd-Gmer4Lfe"
"Qbittorrent-Gmer4Lfe"
"LidaTube"
"Pinchflat"
"ChannelTube"
)
# ━━━ Tier Delays — HOST2's Containers on HOST1 ━━━
# How long HOST2 must be down before each tier activates on HOST1 — in minutes.
# Tier 1 is always immediate — no delay var needed.
HOST2_TIER2_DELAY=240 # 4 hours — productivity services
HOST2_TIER3_DELAY=720 # 12 hours — secondary services
HOST2_TIER4_DELAY=1440 # 24 hours — arrs + downloaders
# ━━━ Rsync Writeback — HOST2 Appdata Back on Handback ━━━
# Syncs HOST2 appdata BACK to HOST2 when it comes back online after a failover.
# Containers stopped before writeback — clean source, no competing writes.
#
# HOST2_TIER1_WRITEBACK_DELAY: short outages skip Tier 1 writeback — primary state
# is more reliable than dirty sync data for brief outages.
HOST2_TIER1_WRITEBACK_DELAY=60 # skip writeback if outage under 1hr
# Tier 4 automatically syncs HOST2_DAILY_SYNC_SHARES — only list paths NOT in that array.
FALLBACK_HOST2_WRITEBACK_TIER1=(
# "/mnt/user/appdata-Fallback/Jayred365-Emby"
)
FALLBACK_HOST2_WRITEBACK_TIER2=(
# "/mnt/user/appdata-Fallback/Jayred365-Important"
)
FALLBACK_HOST2_WRITEBACK_TIER3=(
# "location-placeholder"
)
FALLBACK_HOST2_WRITEBACK_TIER4=(
"/mnt/user/appdata-Fallback/Arrs_Stack" # arr databases — downloads queued during outage
)
# ==============================================================================================
# ── MEDIA ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Media Permissions ━━━
# Shares that media_shares_permissions.sh applies PERMISSIONS_MODE and PERMISSIONS_OWNER to.
# Runs first in DAILY_MAINTENANCE_SCRIPTS — arr cleanup depends on correct ownership.
HOST2_MEDIA_PERMISSION_SHARES=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Shows
)
# ━━━ Media Cleaner ━━━
# Folder lists for media_cleaner.sh — two profiles: anime and media.
# File patterns shared across all servers — defined in master.conf.
# Called via DAILY_MAINTENANCE_SCRIPTS. Run manually: Media/media_cleaner.sh anime|media
HOST2_ANIME_CLEAN_FOLDERS=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Shows
)
HOST2_MEDIA_CLEAN_FOLDERS=(
# fill in when HOST2 is back online
)
# ==============================================================================================
# ── MONITORS ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
# Domains checked via direct openssl connection — not relying on NPM's certificate state.
# Checks the actual certificate served, not what NPM thinks it has.
# Thresholds (CERT_WARN_DAYS, CERT_CRIT_DAYS) defined in master.conf.
HOST2_CERT_MONITOR_DOMAINS=(
# fill in when HOST2 is back online
)
# ━━━ SMART Health ━━━
# Drives skipped in SMART attribute monitoring — hardware is server-specific.
# Thresholds read from dynamix.cfg at runtime — fallbacks in master.conf.
HOST2_SMART_IGNORE_DRIVES=(
"sda" # boot USB — SMART not meaningful on flash drives
)
# ━━━ ZFS Report ━━━
# Pools excluded from the weekly ZFS health report — reduces noise from single-disk array pools.
# Pool health thresholds defined in master.conf.
HOST2_ZFS_REPORT_IGNORE_POOLS=(
# fill in when HOST2 is back online
)
# ==============================================================================================
# ── TRANSCODES ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Ramdisk size ceiling — tmpfs only uses RAM actually needed, not the full size upfront.
# Adjust HOST2_RAMDISK_WARN_GB and HOST2_RAMDISK_LOW_GB together if this changes.
HOST2_RAMDISK_SIZE="8G"
# Usage thresholds — coupled to HOST2_RAMDISK_SIZE, adjust all three together if size changes.
# Hysteresis gap (6.8 - 5.5 = 1.3GB) prevents flip-flop between ramdisk and SSD.
HOST2_RAMDISK_WARN_GB=6.8 # flip to SSD when ramdisk usage reaches this
HOST2_RAMDISK_LOW_GB=5.5 # flip back to ramdisk when usage drops to this
# SSD fallback path — where transcodes land when ramdisk exceeds HOST2_RAMDISK_WARN_GB.
# Must be on cache pool — array disks too slow for active transcode writes.
HOST2_TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/"
# Media servers sharing the ramdisk transcode space on HOST2.
# Format: "ContainerName|URL|APIKey|Type" — Type: emby | jellyfin | plex
# Entries with placeholder API keys are skipped automatically.
# ⚠️ Tdarr does NOT belong here — keep Tdarr on SSD, not ramdisk.
HOST2_TRANSCODE_SERVERS=(
"${HOST2_EMBY_CONTAINER}|${HOST2_EMBY_URL}|${HOST2_EMBY_API_KEY}|emby"
"${HOST2_JELLYFIN_CONTAINER}|${HOST2_JELLYFIN_URL}|${HOST2_JELLYFIN_API_KEY}|jellyfin"
)
# ==============================================================================================
# ── ARR STACK ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Used by arr cleanup scripts and arrs_failed_stalled_recovery.sh.
# detect_hosts() selects HOST2 vars when running on HOST2.
# Lidarr does not run on HOST2 — HOST1_LIDARR_RECOVERY flag handles the exit cleanly.
#
# PATH MAPS — container path → host path translation.
# Arr stores file paths using container-internal paths — scripts need host paths to scan.
# Add one entry per root folder in arr Settings → Media Management → Root Folders.
HOST2_FANART_API_KEY="Yd7147a43b692df0b364b94dc47efb81"
HOST2_LASTFM_API_KEY="be6dc169c33ae263e690c30d18b7491d"
# ━━━ Sonarr ━━━
HOST2_SONARR_URL="http://localhost:8989"
HOST2_SONARR_API_KEY="130decd3db5b4c25afad64864cd03f9f"
HOST2_SONARR_TV_ROOT="/mnt/user/Anime_Shows"
declare -A HOST2_SONARR_PATH_MAP=(
# fill in when HOST2 is back online
# ["/tv"]="/mnt/user/Anime_Shows"
)
# ━━━ Radarr ━━━
HOST2_RADARR_URL="http://localhost:7878"
HOST2_RADARR_API_KEY="d43a3ec6cf1549edb4af0cc63f98b2a9"
HOST2_RADARR_MOVIES_ROOT="/mnt/user/Anime_Movies"
declare -A HOST2_RADARR_PATH_MAP=(
# fill in when HOST2 is back online
# ["/anime-movies"]="/mnt/user/Anime_Movies"
)
# ━━━ Arr Recovery Toggles ━━━
# false = skip that arr on this host — exits cleanly without error
HOST2_SONARR_RECOVERY=true
HOST2_RADARR_RECOVERY=true
# HOST2_LIDARR_RECOVERY not set — Lidarr does not run on HOST2
# ==============================================================================================
# ── SYSTEM WATCHDOG ───────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Per-host check toggles and NIC config for system_watchdog.sh.
# Aliased by detect_hosts() — script uses unprefixed SYS_WATCHDOG_* names.
# HOST2: i5 10th gen 64GB — being rebuilt, lighter workload, no ZFS cache pools.
#
# Conservative defaults during rebuild — re-enable checks as HOST2 stabilises.
# Three-tier response — all critical checks enabled regardless of rebuild state:
# Tier 1 (bypass strikes, reboot now): docker daemon, rootfs full, kernel oops, FD, /boot
# Tier 2 (bypass strikes with OOM): RAM critical + OOM kills in cycle
# Tier 3 (standard strike system): selectively disabled during rebuild
#
# RAM tiers, OOM limits, and reboot loop settings in master.conf System Watchdog section.
# ━━━ Primary NIC ━━━
# Network interface for NIC state check — verify with: ip link show | grep "^[0-9]"
# Common values: eth0, bond0, br0, eno1
HOST2_SYS_WATCHDOG_NIC="eth0"
# ━━━ Tier 1 — Critical Checks ━━━
# All critical checks always enabled — these protect against acute failure regardless of
# rebuild state. Disabling any is not recommended.
# Docker daemon unresponsive → try restart, reboot if restart fails.
HOST2_SYS_WATCHDOG_CHECK_DOCKER_DAEMON=true
# rootfs at critical threshold (ROOTFS_CRITICAL_PCT=99) → reboot immediately.
HOST2_SYS_WATCHDOG_CHECK_ROOTFS=true
# Kernel BUG/Oops in dmesg delta since last cycle → reboot immediately.
HOST2_SYS_WATCHDOG_CHECK_KERNEL_OOPS=true
# File descriptor exhaustion at FD_CRITICAL_PCT (95%) → reboot immediately.
HOST2_SYS_WATCHDOG_CHECK_FD=true
# /boot read-only detected → reboot immediately.
HOST2_SYS_WATCHDOG_CHECK_BOOT=true
# ━━━ Tier 2 — Urgent OOM Check ━━━
# Both must be enabled for Tier 2 bypass to function.
# Track kernel OOM kills each cycle via /proc/vmstat oom_kill delta.
HOST2_SYS_WATCHDOG_CHECK_OOM=true
# Free RAM check — 64GB RAM on HOST2, tiers adjusted relative to HOST1.
# Update master.conf SYS_WATCHDOG_MEM_* thresholds if HOST2 needs different values.
# Currently inheriting shared master.conf values — may want lower thresholds on 64GB.
HOST2_SYS_WATCHDOG_CHECK_RAM=true
# ━━━ Tier 3 — Standard Checks (strike system) ━━━
# Several checks disabled during rebuild — enable progressively as HOST2 stabilises.
# Each check must fail SYS_WATCHDOG_STRIKE_LIMIT consecutive cycles before action.
# /var/log filesystem usage above SYS_WATCHDOG_LOG_PCT.
HOST2_SYS_WATCHDOG_CHECK_LOG=true
# ZFS ARC memory check.
# DISABLED — HOST2 has no ZFS cache pools. Enable if ZFS pools are added later.
HOST2_SYS_WATCHDOG_CHECK_ARC=false
# CPU temperature above SYS_WATCHDOG_CPU_TEMP_MAX (95°C).
HOST2_SYS_WATCHDOG_CHECK_CPU_TEMP=true
# Load average above SYS_WATCHDOG_LOAD_MULTIPLIER × core count.
# DISABLED — rebuild operations cause legitimate load spikes. Enable after rebuild.
HOST2_SYS_WATCHDOG_CHECK_LOAD=false
# Zombie process count above SYS_WATCHDOG_ZOMBIE_LIMIT (50).
HOST2_SYS_WATCHDOG_CHECK_ZOMBIES=true
# docker_watchdog.sh persistent skip list check.
# DISABLED during rebuild — skip list may be unreliable mid-rebuild, avoid false reboots.
# Enable once HOST2 is fully operational and docker_watchdog.sh is running stably.
HOST2_SYS_WATCHDOG_CHECK_CONTAINERS=false
# /tmp filesystem usage with auto-clear attempt.
HOST2_SYS_WATCHDOG_CHECK_TMP=true
# Array disk error count delta in /proc/mdstat.
HOST2_SYS_WATCHDOG_CHECK_MDSTAT=true
# Primary NIC operstate — uses HOST2_SYS_WATCHDOG_NIC above.
HOST2_SYS_WATCHDOG_CHECK_NETWORK=true
# sshd running check — restart attempt before escalating.
HOST2_SYS_WATCHDOG_CHECK_SSHD=true
# Runaway process detection.
# DISABLED — rebuild workloads may legitimately peg CPU. Enable after rebuild.
HOST2_SYS_WATCHDOG_CHECK_RUNAWAY=false
# ==============================================================================================
# ── RESOURCE MANAGER ──────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Containers to manage under pressure — see master.conf RW_CRITICAL_CONTAINERS for exclusions.
# docker pause at medium pressure (RAM < RW_RAM_MEDIUM_GB or load > medium threshold)
# Suspended in-place — instant to pause/unpause, no state lost, no restart delay.
HOST2_RW_PAUSE_CONTAINERS=(
# fill in when HOST2 is back online
)
# docker stop at hard pressure (RAM < RW_RAM_HARD_GB)
# Full stop — these are optional/heavy services that free significant RAM when stopped.
# resource_watchdog.sh restarts them when pressure fully clears (RAM >= RW_RAM_RECOVER_GB).
HOST2_RW_STOP_CONTAINERS=(
# fill in when HOST2 is back online
)
# ==============================================================================================
# ── AUTH STACK ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Credentials for the Varaverk Auth Stack page (NPM, lldap, Authelia).
# Credentials empty — fill in when HOST2 is back online.
# ━━━ NginxProxyManager ━━━
# Admin API runs on 7818 (not 81 — 81 is the partnership WebUI port).
HOST2_NPM_URL="http://localhost:7818"
HOST2_NPM_USER="" # NPM admin email
HOST2_NPM_PASS="" # NPM admin password
# ━━━ lldap ━━━
HOST2_LLDAP_URL="http://localhost:17170"
HOST2_LLDAP_USER="admin" # lldap admin username
HOST2_LLDAP_PASS="" # lldap admin password
# ━━━ Authelia ━━━
HOST2_AUTHELIA_CONFIG="/mnt/user/appdata-Fallback/Critical-Data/Authelia/configuration.yml"
HOST2_AUTHELIA_CONTAINER="Authelia"
# ==============================================================================================
# ──────────────────────── End Of HOST2 Variables ──────────────────────────────────────────────
# ==============================================================================================
@@ -1,386 +0,0 @@
<?php
// First-run setup wizard — uniform flow for all hosts.
// Step 1: auto-detect environment + server identity form.
// Step 2: auto-populate + guide + checklist.
// master.conf pull (for partner servers) lives in the checklist, not here.
$detectedHostname = vv_get_hostname();
?>
<link rel="stylesheet" href="/plugins/varaverk/css/varaverk.css">
<style>
#vv-setup {
max-width: 580px; margin: 40px auto 0;
background: #141414; border: 1px solid #2a2a2a;
border-radius: 6px; padding: 36px 40px 40px;
font-family: monospace; color: #ccc;
}
#vv-setup h1 { margin: 0 0 4px; font-size: 17px; color: #e0e0e0; font-weight: normal; letter-spacing: .04em; }
.vv-sub { font-size: 12px; color: #555; margin-bottom: 28px; }
.vv-field { margin-bottom: 18px; }
.vv-field label { display: block; font-size: 11px; color: #888; margin-bottom: 5px; text-transform: uppercase; letter-spacing: .06em; }
.vv-field input[type=text],
.vv-field select {
width: 100%; box-sizing: border-box; background: #0d0d0d;
border: 1px solid #333; color: #ddd; padding: 7px 10px;
border-radius: 3px; font-family: monospace; font-size: 13px;
}
.vv-field input:focus, .vv-field select:focus { outline: none; border-color: #555; }
.vv-hint { font-size: 11px; color: #555; margin-top: 4px; }
.vv-role-row { display: flex; gap: 10px; margin-bottom: 22px; }
.vv-role-btn { flex: 1; padding: 9px 0; background: #1a1a1a; border: 1px solid #333;
border-radius: 3px; color: #777; font-family: monospace; font-size: 12px;
cursor: pointer; text-align: center; transition: border-color .15s, color .15s; }
.vv-role-btn.active { border-color: #555; color: #ccc; background: #1e1e1e; }
.vv-cond { display: none; }
.vv-cond.show { display: block; }
hr.vv-hr { border: none; border-top: 1px solid #1e1e1e; margin: 22px 0; }
.vv-btn { width: 100%; padding: 10px; background: #1e1e1e; border: 1px solid #444;
color: #ccc; font-family: monospace; font-size: 13px; border-radius: 3px;
cursor: pointer; letter-spacing: .03em; }
.vv-btn:hover { border-color: #666; color: #eee; }
.vv-btn:disabled { opacity: .4; cursor: default; }
#vv-status { margin-top: 10px; font-size: 12px; color: #666; text-align: center; min-height: 16px; }
#vv-status.ok { color: #4a8; }
#vv-status.err { color: #a44; }
/* Detection banner */
#vv-detect-banner {
background: #0d0d0d; border: 1px solid #2a2a2a; border-radius: 3px;
padding: 11px 14px; margin-bottom: 22px; font-size: 12px; line-height: 1.8; color: #666;
}
#vv-detect-banner .vv-det-row { display: flex; gap: 8px; }
#vv-detect-banner .vv-det-lbl { color: #555; min-width: 100px; }
#vv-detect-banner .vv-det-val { color: #999; }
#vv-detect-banner .loading { color: #444; font-style: italic; }
/* Step 2 */
#vv-step2 { display: none; }
.vv-guide {
background: #0d0d0d; border: 1px solid #2a2a2a; border-radius: 3px;
padding: 13px 16px; margin-bottom: 20px; font-size: 12px; color: #666; line-height: 1.9;
}
.vv-guide ol { margin: 8px 0 0 16px; padding: 0; }
.vv-guide li { margin-bottom: 3px; }
.vv-cl-title { font-size: 11px; color: #555; text-transform: uppercase; letter-spacing: .06em; margin-bottom: 10px; }
.vv-cl-item { display: flex; align-items: flex-start; gap: 10px; padding: 7px 0;
border-bottom: 1px solid #1a1a1a; font-size: 12px; }
.vv-cl-item:last-child { border-bottom: none; }
.vv-cl-icon { font-size: 13px; min-width: 16px; margin-top: 1px; }
.vv-cl-body { flex: 1; }
.vv-cl-label { color: #bbb; }
.vv-cl-detail{ color: #555; font-size: 11px; margin-top: 2px; }
.vv-cl-act { margin-top: 5px; }
.vv-cl-act button { padding: 4px 10px; background: #1a1a1a; border: 1px solid #333; color: #888;
font-family: monospace; font-size: 11px; border-radius: 2px; cursor: pointer; }
.vv-cl-act button:hover { border-color: #555; color: #bbb; }
.vv-cl-err { font-size: 11px; color: #a44; margin-top: 4px; }
</style>
<div id="vv-setup">
<h1>⬡ Varaverk — First Run</h1>
<div class="vv-sub">Set up this server before the plugin can start.</div>
<!-- ── Step 1: Detection + identity ──────────────────────────────────────── -->
<div id="vv-step1">
<div id="vv-detect-banner"><div class="loading">Detecting environment…</div></div>
<div class="vv-field">
<label>This server's hostname</label>
<input type="text" id="vv-hostname" value="<?= htmlspecialchars($detectedHostname) ?>" autocomplete="off" spellcheck="false">
<div class="vv-hint">Must match Unraid Settings → Identification exactly (case-sensitive)</div>
</div>
<hr class="vv-hr">
<label style="display:block;font-size:11px;color:#888;text-transform:uppercase;letter-spacing:.06em;margin-bottom:10px;">Server role</label>
<div class="vv-role-row">
<div class="vv-role-btn active" id="vv-role-primary" onclick="vvSetRole('primary')">
Primary<br><span style="color:#555;font-size:10px;">HOST1 · first server</span>
</div>
<div class="vv-role-btn" id="vv-role-partner" onclick="vvSetRole('partner')">
Partner<br><span style="color:#555;font-size:10px;">HOST2+ · joining primary</span>
</div>
</div>
<div class="vv-cond" id="vv-cond-primary">
<div class="vv-field">
<label>Partner's hostname <span style="color:#444;font-size:10px;">(optional — can fill in later)</span></label>
<input type="text" id="vv-partner-hostname" value="" placeholder="unRAID-PartnerServer" autocomplete="off" spellcheck="false">
</div>
</div>
<div class="vv-cond" id="vv-cond-partner">
<div class="vv-field">
<label>Primary server's hostname <span style="color:#a44;font-size:10px;">required</span></label>
<input type="text" id="vv-primary-hostname" value="" placeholder="unRAID-PrimaryServer" autocomplete="off" spellcheck="false">
</div>
<div class="vv-field">
<label>Your slot</label>
<select id="vv-partner-slot">
<option value="host2">HOST2</option>
<option value="host3">HOST3</option>
<option value="host4">HOST4</option>
</select>
</div>
<div style="font-size:11px;color:#555;margin-bottom:4px;">
SSH key and master.conf pull are handled automatically after save.
</div>
</div>
<button class="vv-btn" id="vv-main-btn" onclick="vvDoSave()">Save and continue →</button>
<div id="vv-status"></div>
</div>
<!-- ── Step 2: Populate + guide + checklist ───────────────────────────────── -->
<div id="vv-step2">
<hr class="vv-hr">
<div style="font-size:10px;color:#555;text-transform:uppercase;letter-spacing:.06em;margin-bottom:14px;">Step 2 of 2</div>
<div id="vv-populate-status" style="font-size:12px;color:#555;margin-bottom:14px;">⟳ Running auto-populate…</div>
<div class="vv-guide">
<strong style="color:#888;">Quick start</strong>
<ol>
<li>Create your Unraid API key below — needed for live monitor stats</li>
<li>Open <strong>Scheduler → Edit host.conf</strong> — only three things need manual entry:<br>
<span style="color:#444;">
<code>EMBY_API_KEY</code> — Emby Dashboard → API Keys → + New Key<br>
<code>DISCORD_WEBHOOK</code> — for notifications (optional)<br>
<code>DAILY_SYNC_SHARES</code> — media paths to rsync nightly<br>
Everything else was auto-populated or has working defaults
</span></li>
<li>If partnering: the checklist below will guide you through pulling HOST1's config and running onboard</li>
</ol>
</div>
<div style="display:flex;gap:10px;align-items:center;margin-bottom:14px;">
<button id="vv-key-btn" onclick="vvCreateKey(this)" class="vv-btn" style="flex:1;background:#1a3a1a;border-color:#2e6b2e;color:#6fcf97;">
Create API Key
</button>
<a href="#" onclick="vvGoScheduler(event)" style="font-size:11px;color:#444;text-decoration:none;white-space:nowrap;">Skip →</a>
</div>
<div id="vv-key-status" style="font-size:12px;min-height:14px;margin-bottom:18px;"></div>
<hr class="vv-hr">
<div class="vv-cl-title">Setup checklist</div>
<div id="vv-checklist"><div style="font-size:12px;color:#444;">Loading…</div></div>
<div style="margin-top:18px;text-align:right;">
<a href="#" onclick="vvGoScheduler(event)" style="font-size:12px;color:#444;text-decoration:none;">Go to Scheduler →</a>
</div>
</div>
</div>
<script>
let _vvRedirect = '?tab=scheduler';
// ── Detection banner ──────────────────────────────────────────────────────────
(function() {
const _ac = new AbortController();
setTimeout(() => _ac.abort(), 6000);
fetch('/plugins/varaverk/api/setup.php?action=detect&_=' + Date.now(), {signal: _ac.signal})
.then(r => r.json()).then(d => {
const b = document.getElementById('vv-detect-banner');
if (!d.ok) { b.innerHTML = '<span style="color:#555">Detection unavailable</span>'; return; }
const modeLabel = d.mode === 'internal'
? '<span style="color:#4a8">internal (NVMe/SSD)</span>'
: '<span style="color:#a84">flash mode (USB boot)</span>';
b.innerHTML =
'<div class="vv-det-row"><span class="vv-det-lbl">OS</span><span class="vv-det-val">Unraid ' + (d.unraid_ver||'') + '</span></div>' +
'<div class="vv-det-row"><span class="vv-det-lbl">Boot device</span><span class="vv-det-val">' + d.boot_device + ' (' + d.transport + ')</span></div>' +
'<div class="vv-det-row"><span class="vv-det-lbl">Storage mode</span><span class="vv-det-val">' + modeLabel + '</span></div>' +
'<div class="vv-det-row"><span class="vv-det-lbl">Scripts dir</span><span class="vv-det-val" style="color:#666">' + d.scripts_dir + '</span></div>';
const hf = document.getElementById('vv-hostname');
if (hf && !hf.value.trim()) hf.value = d.hostname;
}).catch(() => {
document.getElementById('vv-detect-banner').innerHTML = '<span style="color:#444">Detection unavailable</span>';
});
})();
// ── Role toggle ───────────────────────────────────────────────────────────────
let vvRole = 'primary';
function vvSetRole(role) {
vvRole = role;
document.getElementById('vv-role-primary')?.classList.toggle('active', role === 'primary');
document.getElementById('vv-role-partner')?.classList.toggle('active', role === 'partner');
document.getElementById('vv-cond-primary')?.classList.toggle('show', role === 'primary');
document.getElementById('vv-cond-partner')?.classList.toggle('show', role === 'partner');
}
// ── Helpers ───────────────────────────────────────────────────────────────────
function vvSetStatus(msg, cls) {
const s = document.getElementById('vv-status');
s.textContent = msg; s.className = cls || '';
}
function vvSetBtn(text, disabled) {
const b = document.getElementById('vv-main-btn');
if (b) { b.textContent = text; b.disabled = disabled; }
}
function vvGoScheduler(e) {
if (e) e.preventDefault();
window.location.href = _vvRedirect || '?tab=scheduler';
}
// ── Step 2 ────────────────────────────────────────────────────────────────────
function vvShowStep2(redirect, apiKey) {
_vvRedirect = redirect || '?tab=scheduler';
document.getElementById('vv-step1').style.display = 'none';
document.getElementById('vv-step2').style.display = 'block';
if (apiKey && apiKey.ok) {
const btn = document.getElementById('vv-key-btn');
const status = document.getElementById('vv-key-status');
if (btn) { btn.textContent = 'Created ✓'; btn.disabled = true; btn.style.opacity = '.6'; }
if (status) { status.textContent = '✓ API key created automatically'; status.style.color = '#4a8'; }
}
vvRunPopulate();
vvLoadChecklist();
}
// ── Populate ──────────────────────────────────────────────────────────────────
function vvRunPopulate() {
const el = document.getElementById('vv-populate-status');
fetch('/plugins/varaverk/api/setup.php', {
method: 'POST', headers: {'Content-Type': 'application/x-www-form-urlencoded'},
body: new URLSearchParams({action: 'populate'})
}).then(r => r.json()).then(d => {
if (d.ok) {
const found = (d.lines || []).filter(l => /✅|found|detected/i.test(l));
el.textContent = found.length
? '✓ Auto-populate: ' + found.length + ' field' + (found.length > 1 ? 's' : '') + ' detected'
: '✓ Auto-populate ran — arr keys will fill once services are running';
el.style.color = '#4a8';
} else {
el.textContent = 'Auto-populate skipped — run Tools/conf_populate.sh once your arr containers are up';
el.style.color = '#555';
}
vvLoadChecklist();
}).catch(() => {
el.textContent = 'Auto-populate unavailable — run manually from Scheduler';
el.style.color = '#555';
});
}
// ── Checklist ─────────────────────────────────────────────────────────────────
const vvActionLabels = {
create_key: 'Create API key',
ssh_setup: 'SSH guide →',
run_populate: 'Run now',
pull_master: 'Pull from HOST1',
onboard: 'Partnership tab →',
};
const vvActionHref = {
ssh_setup: '?tab=partnership',
onboard: '?tab=partnership',
};
function vvLoadChecklist() {
fetch('/plugins/varaverk/api/checklist.php?_=' + Date.now())
.then(r => r.json()).then(d => {
const el = document.getElementById('vv-checklist');
if (!d.ok || !d.items) { el.innerHTML = '<span style="color:#555">Unable to load checklist</span>'; return; }
el.innerHTML = d.items.map(item => {
const icon = item.ok === null ? '○' : (item.ok ? '✓' : '✗');
const iclr = item.ok === null ? '#444' : (item.ok ? '#4a8' : '#a66');
let act = '';
if (item.action) {
const lbl = vvActionLabels[item.action] || item.action;
const href = vvActionHref[item.action];
if (href) {
act = `<div class="vv-cl-act"><a href="${href}" style="font-size:11px;color:#556;">${lbl}</a></div>`;
} else if (item.action === 'create_key') {
act = `<div class="vv-cl-act"><button onclick="vvCreateKey(this)">${lbl}</button></div>`;
} else if (item.action === 'run_populate') {
act = `<div class="vv-cl-act"><button onclick="vvRunPopulateBtn(this)">${lbl}</button></div>`;
} else if (item.action === 'pull_master') {
act = `<div class="vv-cl-act"><button onclick="vvPullMaster(this)">${lbl}</button><div id="vv-pull-err" class="vv-cl-err"></div></div>`;
}
}
return `<div class="vv-cl-item">
<div class="vv-cl-icon" style="color:${iclr}">${icon}</div>
<div class="vv-cl-body">
<div class="vv-cl-label">${item.label}</div>
<div class="vv-cl-detail">${item.detail || ''}</div>
${act}
</div>
</div>`;
}).join('');
}).catch(() => {});
}
function vvRunPopulateBtn(btn) {
btn.disabled = true; btn.textContent = '…';
fetch('/plugins/varaverk/api/setup.php', {
method: 'POST', headers: {'Content-Type': 'application/x-www-form-urlencoded'},
body: new URLSearchParams({action: 'populate'})
}).then(() => { btn.textContent = 'Done'; vvLoadChecklist(); })
.catch(() => { btn.disabled = false; btn.textContent = 'Retry'; });
}
function vvPullMaster(btn) {
btn.disabled = true; btn.textContent = '⟳ Pulling…';
const errEl = document.getElementById('vv-pull-err');
if (errEl) errEl.textContent = '';
fetch('/plugins/varaverk/api/setup.php', {
method: 'POST', headers: {'Content-Type': 'application/x-www-form-urlencoded'},
body: new URLSearchParams({action: 'pull'})
}).then(r => r.json()).then(d => {
if (d.ok) {
btn.textContent = '✓ Done';
setTimeout(vvLoadChecklist, 600);
} else {
if (errEl) errEl.textContent = d.error || 'Pull failed';
btn.disabled = false; btn.textContent = 'Retry';
}
}).catch(() => { btn.disabled = false; btn.textContent = 'Retry'; });
}
// ── API key ───────────────────────────────────────────────────────────────────
function vvCreateKey(btn) {
const status = document.getElementById('vv-key-status');
btn.disabled = true; btn.textContent = '⟳ Creating…';
fetch('/plugins/varaverk/api/create_api_key.php?_=' + Date.now())
.then(r => r.json()).then(d => {
if (d.ok) {
status.textContent = '✓ Key created — ' + d.key_preview;
status.style.color = '#4a8';
btn.textContent = 'Created ✓'; btn.style.opacity = '.6';
vvLoadChecklist();
} else {
status.textContent = '✗ ' + (d.error || 'Failed');
status.style.color = '#a44';
btn.disabled = false; btn.textContent = 'Retry';
}
}).catch(e => {
status.textContent = '✗ ' + e; status.style.color = '#a44';
btn.disabled = false; btn.textContent = 'Retry';
});
}
// ── Save ──────────────────────────────────────────────────────────────────────
function vvDoSave() {
const hostname = document.getElementById('vv-hostname')?.value.trim();
if (!hostname) { vvSetStatus('✗ Hostname is required', 'err'); return; }
let host1 = '', host2 = '', mySlot = 'host1';
if (vvRole === 'primary') {
host1 = hostname;
host2 = document.getElementById('vv-partner-hostname')?.value.trim() || '';
mySlot = 'host1';
} else {
const primary = document.getElementById('vv-primary-hostname')?.value.trim();
if (!primary) { vvSetStatus('✗ Primary hostname required', 'err'); return; }
mySlot = document.getElementById('vv-partner-slot')?.value || 'host2';
host1 = primary;
if (mySlot === 'host2') host2 = hostname;
}
vvSetBtn('Saving…', true);
fetch('/plugins/varaverk/api/setup.php', {
method: 'POST', headers: {'Content-Type': 'application/x-www-form-urlencoded'},
body: new URLSearchParams({action:'save', host1, host2, my_slot:mySlot, my_hostname:hostname})
}).then(r => r.json()).then(d => {
if (d.ok) { vvShowStep2(d.redirect || '?tab=scheduler', d.api_key); }
else { vvSetBtn('Save and continue →', false); vvSetStatus('✗ ' + (d.error||'Error'), 'err'); }
}).catch(() => { vvSetBtn('Save and continue →', false); vvSetStatus('✗ Request failed', 'err'); });
}
</script>
@@ -1,433 +0,0 @@
<?php
// First-run setup wizard — uniform flow for all hosts.
// Step 1: auto-detect environment + server identity form.
// Step 2: auto-populate + guide + checklist.
// master.conf pull (for partner servers) lives in the checklist, not here.
$detectedHostname = vv_get_hostname();
?>
<link rel="stylesheet" href="/plugins/varaverk/css/varaverk.css">
<style>
#vv-setup {
max-width: 580px; margin: 40px auto 0;
background: #141414; border: 1px solid #2a2a2a;
border-radius: 6px; padding: 36px 40px 40px;
font-family: monospace; color: #ccc;
}
#vv-setup h1 { margin: 0 0 4px; font-size: 17px; color: #e0e0e0; font-weight: normal; letter-spacing: .04em; }
.vv-sub { font-size: 12px; color: #555; margin-bottom: 28px; }
.vv-field { margin-bottom: 18px; }
.vv-field label { display: block; font-size: 11px; color: #888; margin-bottom: 5px; text-transform: uppercase; letter-spacing: .06em; }
.vv-field input[type=text],
.vv-field select {
width: 100%; box-sizing: border-box; background: #0d0d0d;
border: 1px solid #333; color: #ddd; padding: 7px 10px;
border-radius: 3px; font-family: monospace; font-size: 13px;
}
.vv-field input:focus, .vv-field select:focus { outline: none; border-color: #555; }
.vv-hint { font-size: 11px; color: #555; margin-top: 4px; }
.vv-role-row { display: flex; gap: 10px; margin-bottom: 22px; }
.vv-role-btn { flex: 1; padding: 9px 0; background: #1a1a1a; border: 1px solid #333;
border-radius: 3px; color: #777; font-family: monospace; font-size: 12px;
cursor: pointer; text-align: center; transition: border-color .15s, color .15s; }
.vv-role-btn.active { border-color: #555; color: #ccc; background: #1e1e1e; }
.vv-cond { display: none; }
.vv-cond.show { display: block; }
hr.vv-hr { border: none; border-top: 1px solid #1e1e1e; margin: 22px 0; }
.vv-btn { width: 100%; padding: 10px; background: #1e1e1e; border: 1px solid #444;
color: #ccc; font-family: monospace; font-size: 13px; border-radius: 3px;
cursor: pointer; letter-spacing: .03em; }
.vv-btn:hover { border-color: #666; color: #eee; }
.vv-btn:disabled { opacity: .4; cursor: default; }
#vv-status { margin-top: 10px; font-size: 12px; color: #666; text-align: center; min-height: 16px; }
#vv-status.ok { color: #4a8; }
#vv-status.err { color: #a44; }
/* Detection banner */
#vv-detect-banner {
background: #0d0d0d; border: 1px solid #2a2a2a; border-radius: 3px;
padding: 11px 14px; margin-bottom: 22px; font-size: 12px; line-height: 1.8; color: #666;
}
#vv-detect-banner .vv-det-row { display: flex; gap: 8px; }
#vv-detect-banner .vv-det-lbl { color: #555; min-width: 100px; }
#vv-detect-banner .vv-det-val { color: #999; }
#vv-detect-banner .loading { color: #444; font-style: italic; }
/* Step 2 */
#vv-step2 { display: none; }
.vv-guide {
background: #0d0d0d; border: 1px solid #2a2a2a; border-radius: 3px;
padding: 13px 16px; margin-bottom: 20px; font-size: 12px; color: #666; line-height: 1.9;
}
.vv-guide ol { margin: 8px 0 0 16px; padding: 0; }
.vv-guide li { margin-bottom: 3px; }
.vv-cl-title { font-size: 11px; color: #555; text-transform: uppercase; letter-spacing: .06em; margin-bottom: 10px; }
.vv-cl-item { display: flex; align-items: flex-start; gap: 10px; padding: 7px 0;
border-bottom: 1px solid #1a1a1a; font-size: 12px; }
.vv-cl-item:last-child { border-bottom: none; }
.vv-cl-icon { font-size: 13px; min-width: 16px; margin-top: 1px; }
.vv-cl-body { flex: 1; }
.vv-cl-label { color: #bbb; }
.vv-cl-detail{ color: #555; font-size: 11px; margin-top: 2px; }
.vv-cl-act { margin-top: 5px; }
.vv-cl-act button { padding: 4px 10px; background: #1a1a1a; border: 1px solid #333; color: #888;
font-family: monospace; font-size: 11px; border-radius: 2px; cursor: pointer; }
.vv-cl-act button:hover { border-color: #555; color: #bbb; }
.vv-cl-err { font-size: 11px; color: #a44; margin-top: 4px; }
</style>
<div id="vv-setup">
<h1>⬡ Varaverk — First Run</h1>
<div class="vv-sub">Set up this server before the plugin can start.</div>
<!-- ── Step 1: Detection + identity ──────────────────────────────────────── -->
<div id="vv-step1">
<div id="vv-detect-banner"><div class="loading">Detecting environment…</div></div>
<div class="vv-field">
<label>Storage mode</label>
<div class="vv-role-row" style="margin-bottom:4px">
<div class="vv-role-btn" id="vv-store-flash" onclick="vvSetStorage('flash')">
Appdata<br><span style="color:#555;font-size:10px;">USB boot · requires array</span>
</div>
<div class="vv-role-btn" id="vv-store-internal" onclick="vvSetStorage('internal')">
Internal Boot<br><span style="color:#555;font-size:10px;">NVMe/SSD · no array dep</span>
</div>
</div>
<div id="vv-store-hint" class="vv-hint"></div>
</div>
<div class="vv-field">
<label>This server's hostname</label>
<input type="text" id="vv-hostname" value="<?= htmlspecialchars($detectedHostname) ?>" autocomplete="off" spellcheck="false">
<div class="vv-hint">Must match Unraid Settings → Identification exactly (case-sensitive)</div>
</div>
<hr class="vv-hr">
<label style="display:block;font-size:11px;color:#888;text-transform:uppercase;letter-spacing:.06em;margin-bottom:10px;">Server role</label>
<div class="vv-role-row">
<div class="vv-role-btn active" id="vv-role-primary" onclick="vvSetRole('primary')">
Primary<br><span style="color:#555;font-size:10px;">HOST1 · first server</span>
</div>
<div class="vv-role-btn" id="vv-role-partner" onclick="vvSetRole('partner')">
Partner<br><span style="color:#555;font-size:10px;">HOST2+ · joining primary</span>
</div>
</div>
<div class="vv-cond" id="vv-cond-primary">
<div class="vv-field">
<label>Partner's hostname <span style="color:#444;font-size:10px;">(optional — can fill in later)</span></label>
<input type="text" id="vv-partner-hostname" value="" placeholder="unRAID-PartnerServer" autocomplete="off" spellcheck="false">
</div>
</div>
<div class="vv-cond" id="vv-cond-partner">
<div class="vv-field">
<label>Primary server's hostname <span style="color:#a44;font-size:10px;">required</span></label>
<input type="text" id="vv-primary-hostname" value="" placeholder="unRAID-PrimaryServer" autocomplete="off" spellcheck="false">
</div>
<div class="vv-field">
<label>Your slot</label>
<select id="vv-partner-slot">
<option value="host2">HOST2</option>
<option value="host3">HOST3</option>
<option value="host4">HOST4</option>
</select>
</div>
<div style="font-size:11px;color:#555;margin-bottom:4px;">
SSH key and master.conf pull are handled automatically after save.
</div>
</div>
<button class="vv-btn" id="vv-main-btn" onclick="vvDoSave()">Save and continue →</button>
<div id="vv-status"></div>
</div>
<!-- ── Step 2: Populate + guide + checklist ───────────────────────────────── -->
<div id="vv-step2">
<hr class="vv-hr">
<div style="font-size:10px;color:#555;text-transform:uppercase;letter-spacing:.06em;margin-bottom:14px;">Step 2 of 2</div>
<div id="vv-populate-status" style="font-size:12px;color:#555;margin-bottom:14px;">⟳ Running auto-populate…</div>
<div class="vv-guide">
<strong style="color:#888;">Quick start</strong>
<ol>
<li>Create your Unraid API key below — needed for live monitor stats</li>
<li>Open <strong>Scheduler → Edit host.conf</strong> — only three things need manual entry:<br>
<span style="color:#444;">
<code>EMBY_API_KEY</code> — Emby Dashboard → API Keys → + New Key<br>
<code>DISCORD_WEBHOOK</code> — for notifications (optional)<br>
<code>DAILY_SYNC_SHARES</code> — media paths to rsync nightly<br>
Everything else was auto-populated or has working defaults
</span></li>
<li>If partnering: the checklist below will guide you through pulling HOST1's config and running onboard</li>
</ol>
</div>
<div style="display:flex;gap:10px;align-items:center;margin-bottom:14px;">
<button id="vv-key-btn" onclick="vvCreateKey(this)" class="vv-btn" style="flex:1;background:#1a3a1a;border-color:#2e6b2e;color:#6fcf97;">
Create API Key
</button>
<a href="#" onclick="vvGoScheduler(event)" style="font-size:11px;color:#444;text-decoration:none;white-space:nowrap;">Skip →</a>
</div>
<div id="vv-key-status" style="font-size:12px;min-height:14px;margin-bottom:18px;"></div>
<hr class="vv-hr">
<div class="vv-cl-title">Setup checklist</div>
<div id="vv-checklist"><div style="font-size:12px;color:#444;">Loading…</div></div>
<div style="margin-top:18px;text-align:right;">
<a href="#" onclick="vvGoScheduler(event)" style="font-size:12px;color:#444;text-decoration:none;">Go to Scheduler →</a>
</div>
</div>
</div>
<script>
let _vvRedirect = '?tab=scheduler';
let _vvStorageMode = 'flash';
let _vvCurrentDir = '';
function vvSetStorage(mode) {
_vvStorageMode = mode;
document.getElementById('vv-store-flash')?.classList.toggle('active', mode === 'flash');
document.getElementById('vv-store-internal')?.classList.toggle('active', mode === 'internal');
const hint = document.getElementById('vv-store-hint');
if (hint) hint.textContent = mode === 'flash'
? 'Scripts live in appdata — requires array to be started. Recommended for USB flash boot.'
: 'Scripts live on /boot — available before array mounts. Requires NVMe/SSD boot.';
}
// ── Detection banner ──────────────────────────────────────────────────────────
(function() {
const _ac = new AbortController();
setTimeout(() => _ac.abort(), 6000);
fetch('/plugins/varaverk/api/setup.php?action=detect&_=' + Date.now(), {signal: _ac.signal})
.then(r => r.json()).then(d => {
const b = document.getElementById('vv-detect-banner');
if (!d.ok) { b.innerHTML = '<span style="color:#555">Detection unavailable</span>'; return; }
_vvCurrentDir = d.scripts_dir || '';
b.innerHTML =
'<div class="vv-det-row"><span class="vv-det-lbl">OS</span><span class="vv-det-val">Unraid ' + (d.unraid_ver||'') + '</span></div>' +
'<div class="vv-det-row"><span class="vv-det-lbl">Boot device</span><span class="vv-det-val">' + d.boot_device + ' (' + d.transport + ')</span></div>' +
'<div class="vv-det-row"><span class="vv-det-lbl">Scripts dir</span><span class="vv-det-val" style="color:#666">' + d.scripts_dir + '</span></div>';
vvSetStorage(d.mode);
const hf = document.getElementById('vv-hostname');
if (hf && !hf.value.trim()) hf.value = d.hostname;
}).catch(() => {
document.getElementById('vv-detect-banner').innerHTML = '<span style="color:#444">Detection unavailable</span>';
});
})();
// ── Role toggle ───────────────────────────────────────────────────────────────
let vvRole = 'primary';
function vvSetRole(role) {
vvRole = role;
document.getElementById('vv-role-primary')?.classList.toggle('active', role === 'primary');
document.getElementById('vv-role-partner')?.classList.toggle('active', role === 'partner');
document.getElementById('vv-cond-primary')?.classList.toggle('show', role === 'primary');
document.getElementById('vv-cond-partner')?.classList.toggle('show', role === 'partner');
}
// ── Helpers ───────────────────────────────────────────────────────────────────
function vvSetStatus(msg, cls) {
const s = document.getElementById('vv-status');
s.textContent = msg; s.className = cls || '';
}
function vvSetBtn(text, disabled) {
const b = document.getElementById('vv-main-btn');
if (b) { b.textContent = text; b.disabled = disabled; }
}
function vvGoScheduler(e) {
if (e) e.preventDefault();
window.location.href = _vvRedirect || '?tab=scheduler';
}
// ── Step 2 ────────────────────────────────────────────────────────────────────
function vvShowStep2(redirect, apiKey) {
_vvRedirect = redirect || '?tab=scheduler';
document.getElementById('vv-step1').style.display = 'none';
document.getElementById('vv-step2').style.display = 'block';
if (apiKey && apiKey.ok) {
const btn = document.getElementById('vv-key-btn');
const status = document.getElementById('vv-key-status');
if (btn) { btn.textContent = 'Created ✓'; btn.disabled = true; btn.style.opacity = '.6'; }
if (status) { status.textContent = '✓ API key created automatically'; status.style.color = '#4a8'; }
}
vvRunPopulate();
vvLoadChecklist();
}
// ── Populate ──────────────────────────────────────────────────────────────────
function vvRunPopulate() {
const el = document.getElementById('vv-populate-status');
fetch('/plugins/varaverk/api/setup.php', {
method: 'POST', headers: {'Content-Type': 'application/x-www-form-urlencoded'},
body: new URLSearchParams({action: 'populate'})
}).then(r => r.json()).then(d => {
if (d.ok) {
const found = (d.lines || []).filter(l => /✅|found|detected/i.test(l));
el.textContent = found.length
? '✓ Auto-populate: ' + found.length + ' field' + (found.length > 1 ? 's' : '') + ' detected'
: '✓ Auto-populate ran — arr keys will fill once services are running';
el.style.color = '#4a8';
} else {
el.textContent = 'Auto-populate skipped — run Tools/conf_populate.sh once your arr containers are up';
el.style.color = '#555';
}
vvLoadChecklist();
}).catch(() => {
el.textContent = 'Auto-populate unavailable — run manually from Scheduler';
el.style.color = '#555';
});
}
// ── Checklist ─────────────────────────────────────────────────────────────────
const vvActionLabels = {
create_key: 'Create API key',
ssh_setup: 'SSH guide →',
run_populate: 'Run now',
pull_master: 'Pull from HOST1',
onboard: 'Partnership tab →',
};
const vvActionHref = {
ssh_setup: '?tab=partnership',
onboard: '?tab=partnership',
};
function vvLoadChecklist() {
fetch('/plugins/varaverk/api/checklist.php?_=' + Date.now())
.then(r => r.json()).then(d => {
const el = document.getElementById('vv-checklist');
if (!d.ok || !d.items) { el.innerHTML = '<span style="color:#555">Unable to load checklist</span>'; return; }
el.innerHTML = d.items.map(item => {
const icon = item.ok === null ? '○' : (item.ok ? '✓' : '✗');
const iclr = item.ok === null ? '#444' : (item.ok ? '#4a8' : '#a66');
let act = '';
if (item.action) {
const lbl = vvActionLabels[item.action] || item.action;
const href = vvActionHref[item.action];
if (href) {
act = `<div class="vv-cl-act"><a href="${href}" style="font-size:11px;color:#556;">${lbl}</a></div>`;
} else if (item.action === 'create_key') {
act = `<div class="vv-cl-act"><button onclick="vvCreateKey(this)">${lbl}</button></div>`;
} else if (item.action === 'run_populate') {
act = `<div class="vv-cl-act"><button onclick="vvRunPopulateBtn(this)">${lbl}</button></div>`;
} else if (item.action === 'pull_master') {
act = `<div class="vv-cl-act"><button onclick="vvPullMaster(this)">${lbl}</button><div id="vv-pull-err" class="vv-cl-err"></div></div>`;
}
}
return `<div class="vv-cl-item">
<div class="vv-cl-icon" style="color:${iclr}">${icon}</div>
<div class="vv-cl-body">
<div class="vv-cl-label">${item.label}</div>
<div class="vv-cl-detail">${item.detail || ''}</div>
${act}
</div>
</div>`;
}).join('');
}).catch(() => {});
}
function vvRunPopulateBtn(btn) {
btn.disabled = true; btn.textContent = '…';
fetch('/plugins/varaverk/api/setup.php', {
method: 'POST', headers: {'Content-Type': 'application/x-www-form-urlencoded'},
body: new URLSearchParams({action: 'populate'})
}).then(() => { btn.textContent = 'Done'; vvLoadChecklist(); })
.catch(() => { btn.disabled = false; btn.textContent = 'Retry'; });
}
function vvPullMaster(btn) {
btn.disabled = true; btn.textContent = '⟳ Pulling…';
const errEl = document.getElementById('vv-pull-err');
if (errEl) errEl.textContent = '';
fetch('/plugins/varaverk/api/setup.php', {
method: 'POST', headers: {'Content-Type': 'application/x-www-form-urlencoded'},
body: new URLSearchParams({action: 'pull'})
}).then(r => r.json()).then(d => {
if (d.ok) {
btn.textContent = '✓ Done';
setTimeout(vvLoadChecklist, 600);
} else {
if (errEl) errEl.textContent = d.error || 'Pull failed';
btn.disabled = false; btn.textContent = 'Retry';
}
}).catch(() => { btn.disabled = false; btn.textContent = 'Retry'; });
}
// ── API key ───────────────────────────────────────────────────────────────────
function vvCreateKey(btn) {
const status = document.getElementById('vv-key-status');
btn.disabled = true; btn.textContent = '⟳ Creating…';
fetch('/plugins/varaverk/api/create_api_key.php?_=' + Date.now())
.then(r => r.json()).then(d => {
if (d.ok) {
status.textContent = '✓ Key created — ' + d.key_preview;
status.style.color = '#4a8';
btn.textContent = 'Created ✓'; btn.style.opacity = '.6';
vvLoadChecklist();
} else {
status.textContent = '✗ ' + (d.error || 'Failed');
status.style.color = '#a44';
btn.disabled = false; btn.textContent = 'Retry';
}
}).catch(e => {
status.textContent = '✗ ' + e; status.style.color = '#a44';
btn.disabled = false; btn.textContent = 'Retry';
});
}
// ── Save ──────────────────────────────────────────────────────────────────────
function vvDoSave() {
const hostname = document.getElementById('vv-hostname')?.value.trim();
if (!hostname) { vvSetStatus('✗ Hostname is required', 'err'); return; }
let host1 = '', host2 = '', mySlot = 'host1';
if (vvRole === 'primary') {
host1 = hostname;
host2 = document.getElementById('vv-partner-hostname')?.value.trim() || '';
mySlot = 'host1';
} else {
const primary = document.getElementById('vv-primary-hostname')?.value.trim();
if (!primary) { vvSetStatus('✗ Primary hostname required', 'err'); return; }
mySlot = document.getElementById('vv-partner-slot')?.value || 'host2';
host1 = primary;
if (mySlot === 'host2') host2 = hostname;
}
vvSetBtn('Saving…', true);
fetch('/plugins/varaverk/api/setup.php', {
method: 'POST', headers: {'Content-Type': 'application/x-www-form-urlencoded'},
body: new URLSearchParams({action:'save', host1, host2, my_slot:mySlot, my_hostname:hostname, storage_mode:_vvStorageMode})
}).then(r => r.json()).then(d => {
if (d.ok) {
if (d.needs_migration) {
const dest = d.migrate_to === 'flash' ? 'appdata' : '/boot';
vvSetStatus('⟳ Migrating scripts to ' + dest + '…', '');
vvDoMigration(d.migrate_to, d.redirect || '?tab=scheduler', d.api_key);
} else {
vvShowStep2(d.redirect || '?tab=scheduler', d.api_key);
}
} else { vvSetBtn('Save and continue →', false); vvSetStatus('✗ ' + (d.error||'Error'), 'err'); }
}).catch(() => { vvSetBtn('Save and continue →', false); vvSetStatus('✗ Request failed', 'err'); });
}
function vvDoMigration(to, redirect, apiKey) {
fetch('/plugins/varaverk/api/storage.php', {
method: 'POST', headers: {'Content-Type': 'application/x-www-form-urlencoded'},
body: new URLSearchParams({action: 'migrate', to})
}).then(r => r.json()).then(d => {
if (d.ok) {
vvShowStep2(redirect, apiKey);
} else {
vvSetBtn('Save and continue →', false);
vvSetStatus('✗ Migration failed — ' + (d.error || 'check install.log'), 'err');
}
}).catch(() => {
vvSetBtn('Save and continue →', false);
vvSetStatus('✗ Migration request failed', 'err');
});
}
</script>
@@ -1,259 +0,0 @@
<?php
header('Content-Type: application/json');
require_once dirname(__DIR__) . '/include/config.php';
$action = ($_SERVER['REQUEST_METHOD'] === 'GET')
? trim($_GET['action'] ?? '')
: trim($_POST['action'] ?? 'save');
// ── GET: detect environment ────────────────────────────────────────────────────────────────────
if ($action === 'detect') {
$bootPart = trim(shell_exec('findmnt -n -o SOURCE /boot 2>/dev/null') ?: '');
$bootDisk = $bootPart
? trim(shell_exec('lsblk -no pkname ' . escapeshellarg($bootPart) . ' 2>/dev/null') ?: '')
: '';
$transport = $bootDisk
? strtolower(trim(shell_exec('lsblk -dno TRAN /dev/' . escapeshellarg($bootDisk) . ' 2>/dev/null') ?: ''))
: 'unknown';
$isUsb = ($transport === 'usb');
preg_match('/version="([^"]+)"/', @file_get_contents('/etc/unraid-version') ?: '', $vm);
echo json_encode([
'ok' => true,
'hostname' => vv_get_hostname(),
'unraid_ver' => $vm[1] ?? 'unknown',
'transport' => $transport,
'boot_device' => $bootDisk ? '/dev/' . $bootDisk : 'unknown',
'mode' => $isUsb ? 'flash' : 'internal',
'scripts_dir' => SCRIPTS_DIR,
]);
exit;
}
// ── GET/POST: generate local SSH keypair ──────────────────────────────────────────────────────
if ($action === 'ssh_generate') {
$script = SCRIPTS_DIR . '/Partnership/ssh_setup.sh';
if (!file_exists($script)) {
echo json_encode(['ok' => false, 'error' => 'ssh_setup.sh not found']);
exit;
}
exec('bash ' . escapeshellarg($script) . ' --local-only 2>&1', $out, $rc);
// Derive pubkey path from hostname
$hostname = vv_get_hostname();
$shortName = strtolower(preg_replace('/^unraid-/i', '', $hostname));
$pubPath = '/root/.ssh/' . $shortName . '_rsync_automation.pub';
$pubKey = trim(@file_get_contents($pubPath) ?: '');
echo json_encode([
'ok' => $rc === 0 && !empty($pubKey),
'pubkey' => $pubKey,
'error' => ($rc !== 0) ? implode(' ', array_slice(array_filter(array_map('trim', $out)), -3)) : null,
]);
exit;
}
// ── POST: run conf_populate.sh ─────────────────────────────────────────────────────────────────
if ($action === 'populate') {
$script = SCRIPTS_DIR . '/Plugin/unraid/Tools/conf_populate.sh';
if (!file_exists($script)) {
echo json_encode(['ok' => false, 'error' => 'conf_populate.sh not found']);
exit;
}
exec('bash ' . escapeshellarg($script) . ' --no-push 2>&1', $out, $rc);
$lines = array_values(array_filter(array_map('trim', $out)));
echo json_encode(['ok' => $rc === 0, 'lines' => array_slice($lines, 0, 20)]);
exit;
}
if ($_SERVER['REQUEST_METHOD'] !== 'POST') {
echo json_encode(['ok' => false, 'error' => 'Method not allowed']);
exit;
}
$sshScript = SCRIPTS_DIR . '/Partnership/ssh_setup.sh';
// ── Pull master.conf from HOST1 via SSH (wizard or checklist) ────────────────────────────────
if ($action === 'pull') {
$mySlot = trim($_POST['my_slot'] ?? '') ?: strtolower(vv_detect_host());
$myHostname = trim($_POST['my_hostname'] ?? '') ?: vv_get_hostname();
$host1Hostname = trim($_POST['host1_hostname'] ?? '');
if (!$host1Hostname) {
$masterRaw = vv_read_conf_raw('master.conf');
preg_match('/^\s*HOST1\s*=\s*"([^"]*)"/m', $masterRaw, $_mh);
$host1Hostname = trim($_mh[1] ?? '');
}
if (!$host1Hostname) {
echo json_encode(['ok' => false, 'error' => 'HOST1 hostname not set — fill in master.conf first']);
exit;
}
if (!preg_match('/^host\d+$/', $mySlot)) {
echo json_encode(['ok' => false, 'error' => 'Invalid slot']);
exit;
}
$hostId = strtoupper($mySlot);
$hostIdLow = strtolower($mySlot);
// Derive SSH key path from this server's hostname
$sshOwner = strtolower(preg_replace('/^unraid-/i', '', $myHostname ?: vv_get_hostname()));
$sshKey = '/root/.ssh/' . $sshOwner . '_rsync_automation';
if (!file_exists($sshKey)) {
echo json_encode(['ok' => false, 'error' =>
"SSH key not found at $sshKey — run Partnership/ssh_setup.sh first"]);
exit;
}
// Resolve HOST1 Tailscale IP
$ip = trim(shell_exec('tailscale ip -4 ' . escapeshellarg($host1Hostname) . ' 2>/dev/null') ?: '');
if (!$ip) {
echo json_encode(['ok' => false, 'error' =>
"Cannot resolve Tailscale IP for $host1Hostname — is Tailscale running on both servers?"]);
exit;
}
// Get HOST1's SCRIPTS_DIR from their varaverk.cfg
$sshBase = 'ssh -i ' . escapeshellarg($sshKey)
. ' -o ConnectTimeout=10 -o StrictHostKeyChecking=no root@' . $ip;
$remoteCfg = trim(shell_exec($sshBase . ' "grep SCRIPTS_DIR /boot/config/plugins/varaverk/varaverk.cfg 2>/dev/null"') ?: '');
preg_match('/SCRIPTS_DIR\s*=\s*["\']?([^"\']+)["\']?/', $remoteCfg, $sm);
$remoteConf = rtrim($sm[1] ?? '/boot/config/plugins/varaverk', '/') . '/Configurations';
// SCP master.conf from HOST1
$localMaster = CONF_DIR . '/master.conf';
$src = escapeshellarg('root@' . $ip . ':' . $remoteConf . '/master.conf');
$cmd = 'scp -i ' . escapeshellarg($sshKey)
. ' -o ConnectTimeout=10 -o StrictHostKeyChecking=no'
. ' ' . $src . ' ' . escapeshellarg($localMaster) . ' 2>&1';
exec($cmd, $out, $rc);
if ($rc !== 0) {
echo json_encode(['ok' => false, 'error' =>
'SCP failed: ' . implode('; ', $out) .
' — ensure your SSH key is authorised on HOST1 (run Partnership/ssh_setup.sh)']);
exit;
}
// Create host conf from template if it doesn't exist
$confFile = $hostIdLow . '.conf';
if (!file_exists(CONF_DIR . '/' . $confFile)) {
$template = @file_get_contents(CONF_DIR . '/host.conf.template') ?: '';
if ($template) {
$bootPart2 = trim(shell_exec('findmnt -n -o SOURCE /boot 2>/dev/null') ?: '');
$bootDisk2 = $bootPart2 ? trim(shell_exec('lsblk -no pkname ' . escapeshellarg($bootPart2) . ' 2>/dev/null') ?: '') : '';
$transport2 = $bootDisk2 ? strtolower(trim(shell_exec('lsblk -dno TRAN /dev/' . escapeshellarg($bootDisk2) . ' 2>/dev/null') ?: '')) : '';
$storageInternal2 = ($transport2 !== 'usb') ? 'true' : 'false';
$conf = str_replace('HOSTN', $hostId, $template);
$conf = str_replace('hostn', $hostIdLow, $conf);
$conf = preg_replace('/^(\s*' . $hostId . '_SSH_KEY\s*=\s*)""/m',
'${1}"' . $sshKey . '"', $conf);
$conf = preg_replace('/^(\s*' . $hostId . '_STORAGE_MODE_INTERNAL\s*=\s*)\S+/m',
'${1}' . $storageInternal2, $conf);
vv_write_conf_raw($confFile, $conf);
}
}
if (file_exists($sshScript)) {
exec('bash ' . escapeshellarg($sshScript) . ' --local-only 2>/dev/null');
}
$apiKeyResult = vv_auto_create_api_key($hostId, $confFile);
$state = vv_setup_state_read();
$state['master_conf_pulled'] = 'true';
vv_setup_state_write($state);
echo json_encode(['ok' => true, 'host_id' => $hostId, 'conf_file' => $confFile,
'api_key' => $apiKeyResult,
'redirect' => '?tab=scheduler&vv_setup=' . $confFile]);
exit;
}
// ── Default action: save (HOST1 first-run wizard) ────────────────────────────────────────────
$host1 = trim($_POST['host1'] ?? '');
$host2 = trim($_POST['host2'] ?? '');
$mySlot = trim($_POST['my_slot'] ?? 'host1');
$myHostname = trim($_POST['my_hostname'] ?? '');
if (empty($host1)) {
echo json_encode(['ok' => false, 'error' => 'HOST1 hostname is required']);
exit;
}
if (!preg_match('/^host\d+$/', $mySlot)) {
echo json_encode(['ok' => false, 'error' => 'Invalid slot']);
exit;
}
// Write HOST1 / HOST2 into master.conf
$master = vv_read_conf_raw('master.conf');
if ($master === '') {
echo json_encode(['ok' => false, 'error' => 'master.conf not found — check SCRIPTS_DIR in varaverk.cfg']);
exit;
}
$master = preg_replace('/^(\s*HOST1\s*=\s*).*$/m', '${1}"' . addslashes($host1) . '"', $master);
$master = preg_replace('/^(\s*HOST2\s*=\s*).*$/m', '${1}"' . addslashes($host2) . '"', $master);
$slotNum = (int) preg_replace('/\D/', '', $mySlot);
if ($slotNum > 2 && !empty($myHostname)) {
$hostKey = 'HOST' . $slotNum;
if (!preg_match('/^\s*' . $hostKey . '\s*=/m', $master)) {
$master = preg_replace('/^(\s*HOST2\s*=.*$)/m',
'$1' . "\n {$hostKey}=\"" . addslashes($myHostname) . '"', $master);
} else {
$master = preg_replace('/^(\s*' . $hostKey . '\s*=\s*).*$/m',
'${1}"' . addslashes($myHostname) . '"', $master);
}
}
if (!vv_write_conf_raw('master.conf', $master)) {
echo json_encode(['ok' => false, 'error' => 'Failed to write master.conf']);
exit;
}
// Create host*.conf from template
$hostId = strtoupper($mySlot);
$hostIdLow = strtolower($mySlot);
$confFile = $hostIdLow . '.conf';
if (!file_exists(CONF_DIR . '/' . $confFile)) {
$template = @file_get_contents(CONF_DIR . '/host.conf.template') ?: '';
if ($template) {
$sshOwner = strtolower(preg_replace('/^unraid-/i', '', $myHostname));
$sshKeyPath = '/root/.ssh/' . $sshOwner . '_rsync_automation';
// Auto-detect storage mode from boot device transport
$bootPart = trim(shell_exec('findmnt -n -o SOURCE /boot 2>/dev/null') ?: '');
$bootDisk = $bootPart ? trim(shell_exec('lsblk -no pkname ' . escapeshellarg($bootPart) . ' 2>/dev/null') ?: '') : '';
$transport = $bootDisk ? strtolower(trim(shell_exec('lsblk -dno TRAN /dev/' . escapeshellarg($bootDisk) . ' 2>/dev/null') ?: '')) : '';
$storageInternal = ($transport !== 'usb') ? 'true' : 'false';
$conf = str_replace('HOSTN', $hostId, $template);
$conf = str_replace('hostn', $hostIdLow, $conf);
$conf = preg_replace('/^(\s*' . $hostId . '_SSH_KEY\s*=\s*)""/m',
'${1}"' . $sshKeyPath . '"', $conf);
$conf = preg_replace('/^(\s*' . $hostId . '_STORAGE_MODE_INTERNAL\s*=\s*)\S+/m',
'${1}' . $storageInternal, $conf);
if (!vv_write_conf_raw($confFile, $conf)) {
echo json_encode(['ok' => false, 'error' => "Failed to write $confFile"]);
exit;
}
}
}
// Write setup state file — lets partner servers know HOST1 is configured
vv_setup_state_write(['host1_hostname' => $host1]);
// Auto-generate SSH keypair (local only — remote copy happens during onboarding)
if (file_exists($sshScript)) {
exec('bash ' . escapeshellarg($sshScript) . ' --local-only 2>/dev/null');
}
// Auto-create Unraid API key and write into the fresh conf
$apiKeyResult = vv_auto_create_api_key($hostId, $confFile);
echo json_encode([
'ok' => true,
'host_id' => $hostId,
'api_key' => $apiKeyResult,
'redirect' => '?tab=scheduler&vv_setup=master.conf',
]);
@@ -1,271 +0,0 @@
<?php
header('Content-Type: application/json');
require_once dirname(__DIR__) . '/include/config.php';
$action = ($_SERVER['REQUEST_METHOD'] === 'GET')
? trim($_GET['action'] ?? '')
: trim($_POST['action'] ?? 'save');
// ── GET: detect environment ────────────────────────────────────────────────────────────────────
if ($action === 'detect') {
$bootPart = trim(shell_exec('findmnt -n -o SOURCE /boot 2>/dev/null') ?: '');
$bootDisk = $bootPart
? trim(shell_exec('lsblk -no pkname ' . escapeshellarg($bootPart) . ' 2>/dev/null') ?: '')
: '';
$transport = $bootDisk
? strtolower(trim(shell_exec('lsblk -dno TRAN /dev/' . escapeshellarg($bootDisk) . ' 2>/dev/null') ?: ''))
: 'unknown';
$isUsb = ($transport === 'usb');
preg_match('/version="([^"]+)"/', @file_get_contents('/etc/unraid-version') ?: '', $vm);
echo json_encode([
'ok' => true,
'hostname' => vv_get_hostname(),
'unraid_ver' => $vm[1] ?? 'unknown',
'transport' => $transport,
'boot_device' => $bootDisk ? '/dev/' . $bootDisk : 'unknown',
'mode' => $isUsb ? 'flash' : 'internal',
'scripts_dir' => SCRIPTS_DIR,
]);
exit;
}
// ── GET/POST: generate local SSH keypair ──────────────────────────────────────────────────────
if ($action === 'ssh_generate') {
$script = SCRIPTS_DIR . '/Partnership/ssh_setup.sh';
if (!file_exists($script)) {
echo json_encode(['ok' => false, 'error' => 'ssh_setup.sh not found']);
exit;
}
exec('bash ' . escapeshellarg($script) . ' --local-only 2>&1', $out, $rc);
// Derive pubkey path from hostname
$hostname = vv_get_hostname();
$shortName = strtolower(preg_replace('/^unraid-/i', '', $hostname));
$pubPath = '/root/.ssh/' . $shortName . '_rsync_automation.pub';
$pubKey = trim(@file_get_contents($pubPath) ?: '');
echo json_encode([
'ok' => $rc === 0 && !empty($pubKey),
'pubkey' => $pubKey,
'error' => ($rc !== 0) ? implode(' ', array_slice(array_filter(array_map('trim', $out)), -3)) : null,
]);
exit;
}
// ── POST: run conf_populate.sh ─────────────────────────────────────────────────────────────────
if ($action === 'populate') {
$script = SCRIPTS_DIR . '/Plugin/unraid/Tools/conf_populate.sh';
if (!file_exists($script)) {
echo json_encode(['ok' => false, 'error' => 'conf_populate.sh not found']);
exit;
}
exec('bash ' . escapeshellarg($script) . ' --no-push 2>&1', $out, $rc);
$lines = array_values(array_filter(array_map('trim', $out)));
echo json_encode(['ok' => $rc === 0, 'lines' => array_slice($lines, 0, 20)]);
exit;
}
if ($_SERVER['REQUEST_METHOD'] !== 'POST') {
echo json_encode(['ok' => false, 'error' => 'Method not allowed']);
exit;
}
$sshScript = SCRIPTS_DIR . '/Partnership/ssh_setup.sh';
// ── Pull master.conf from HOST1 via SSH (wizard or checklist) ────────────────────────────────
if ($action === 'pull') {
$mySlot = trim($_POST['my_slot'] ?? '') ?: strtolower(vv_detect_host());
$myHostname = trim($_POST['my_hostname'] ?? '') ?: vv_get_hostname();
$host1Hostname = trim($_POST['host1_hostname'] ?? '');
if (!$host1Hostname) {
$masterRaw = vv_read_conf_raw('master.conf');
preg_match('/^\s*HOST1\s*=\s*"([^"]*)"/m', $masterRaw, $_mh);
$host1Hostname = trim($_mh[1] ?? '');
}
if (!$host1Hostname) {
echo json_encode(['ok' => false, 'error' => 'HOST1 hostname not set — fill in master.conf first']);
exit;
}
if (!preg_match('/^host\d+$/', $mySlot)) {
echo json_encode(['ok' => false, 'error' => 'Invalid slot']);
exit;
}
$hostId = strtoupper($mySlot);
$hostIdLow = strtolower($mySlot);
// Derive SSH key path from this server's hostname
$sshOwner = strtolower(preg_replace('/^unraid-/i', '', $myHostname ?: vv_get_hostname()));
$sshKey = '/root/.ssh/' . $sshOwner . '_rsync_automation';
if (!file_exists($sshKey)) {
echo json_encode(['ok' => false, 'error' =>
"SSH key not found at $sshKey — run Partnership/ssh_setup.sh first"]);
exit;
}
// Resolve HOST1 Tailscale IP
$ip = trim(shell_exec('tailscale ip -4 ' . escapeshellarg($host1Hostname) . ' 2>/dev/null') ?: '');
if (!$ip) {
echo json_encode(['ok' => false, 'error' =>
"Cannot resolve Tailscale IP for $host1Hostname — is Tailscale running on both servers?"]);
exit;
}
// Get HOST1's SCRIPTS_DIR from their varaverk.cfg
$sshBase = 'ssh -i ' . escapeshellarg($sshKey)
. ' -o ConnectTimeout=10 -o StrictHostKeyChecking=no root@' . $ip;
$remoteCfg = trim(shell_exec($sshBase . ' "grep SCRIPTS_DIR /boot/config/plugins/varaverk/varaverk.cfg 2>/dev/null"') ?: '');
preg_match('/SCRIPTS_DIR\s*=\s*["\']?([^"\']+)["\']?/', $remoteCfg, $sm);
$remoteConf = rtrim($sm[1] ?? '/boot/config/plugins/varaverk', '/') . '/Configurations';
// SCP master.conf from HOST1
$localMaster = CONF_DIR . '/master.conf';
$src = escapeshellarg('root@' . $ip . ':' . $remoteConf . '/master.conf');
$cmd = 'scp -i ' . escapeshellarg($sshKey)
. ' -o ConnectTimeout=10 -o StrictHostKeyChecking=no'
. ' ' . $src . ' ' . escapeshellarg($localMaster) . ' 2>&1';
exec($cmd, $out, $rc);
if ($rc !== 0) {
echo json_encode(['ok' => false, 'error' =>
'SCP failed: ' . implode('; ', $out) .
' — ensure your SSH key is authorised on HOST1 (run Partnership/ssh_setup.sh)']);
exit;
}
// Create host conf from template if it doesn't exist
$confFile = $hostIdLow . '.conf';
if (!file_exists(CONF_DIR . '/' . $confFile)) {
$template = @file_get_contents(CONF_DIR . '/host.conf.template') ?: '';
if ($template) {
$bootPart2 = trim(shell_exec('findmnt -n -o SOURCE /boot 2>/dev/null') ?: '');
$bootDisk2 = $bootPart2 ? trim(shell_exec('lsblk -no pkname ' . escapeshellarg($bootPart2) . ' 2>/dev/null') ?: '') : '';
$transport2 = $bootDisk2 ? strtolower(trim(shell_exec('lsblk -dno TRAN /dev/' . escapeshellarg($bootDisk2) . ' 2>/dev/null') ?: '')) : '';
$storageInternal2 = ($transport2 !== 'usb') ? 'true' : 'false';
$conf = str_replace('HOSTN', $hostId, $template);
$conf = str_replace('hostn', $hostIdLow, $conf);
$conf = preg_replace('/^(\s*' . $hostId . '_SSH_KEY\s*=\s*)""/m',
'${1}"' . $sshKey . '"', $conf);
$conf = preg_replace('/^(\s*' . $hostId . '_STORAGE_MODE_INTERNAL\s*=\s*)\S+/m',
'${1}' . $storageInternal2, $conf);
vv_write_conf_raw($confFile, $conf);
}
}
if (file_exists($sshScript)) {
exec('bash ' . escapeshellarg($sshScript) . ' --local-only 2>/dev/null');
}
$apiKeyResult = vv_auto_create_api_key($hostId, $confFile);
$state = vv_setup_state_read();
$state['master_conf_pulled'] = 'true';
vv_setup_state_write($state);
echo json_encode(['ok' => true, 'host_id' => $hostId, 'conf_file' => $confFile,
'api_key' => $apiKeyResult,
'redirect' => '?tab=scheduler&vv_setup=' . $confFile]);
exit;
}
// ── Default action: save (HOST1 first-run wizard) ────────────────────────────────────────────
$host1 = trim($_POST['host1'] ?? '');
$host2 = trim($_POST['host2'] ?? '');
$mySlot = trim($_POST['my_slot'] ?? 'host1');
$myHostname = trim($_POST['my_hostname'] ?? '');
if (empty($host1)) {
echo json_encode(['ok' => false, 'error' => 'HOST1 hostname is required']);
exit;
}
if (!preg_match('/^host\d+$/', $mySlot)) {
echo json_encode(['ok' => false, 'error' => 'Invalid slot']);
exit;
}
// Write HOST1 / HOST2 into master.conf
$master = vv_read_conf_raw('master.conf');
if ($master === '') {
echo json_encode(['ok' => false, 'error' => 'master.conf not found — check SCRIPTS_DIR in varaverk.cfg']);
exit;
}
$master = preg_replace('/^(\s*HOST1\s*=\s*).*$/m', '${1}"' . addslashes($host1) . '"', $master);
$master = preg_replace('/^(\s*HOST2\s*=\s*).*$/m', '${1}"' . addslashes($host2) . '"', $master);
$slotNum = (int) preg_replace('/\D/', '', $mySlot);
if ($slotNum > 2 && !empty($myHostname)) {
$hostKey = 'HOST' . $slotNum;
if (!preg_match('/^\s*' . $hostKey . '\s*=/m', $master)) {
$master = preg_replace('/^(\s*HOST2\s*=.*$)/m',
'$1' . "\n {$hostKey}=\"" . addslashes($myHostname) . '"', $master);
} else {
$master = preg_replace('/^(\s*' . $hostKey . '\s*=\s*).*$/m',
'${1}"' . addslashes($myHostname) . '"', $master);
}
}
if (!vv_write_conf_raw('master.conf', $master)) {
echo json_encode(['ok' => false, 'error' => 'Failed to write master.conf']);
exit;
}
// Create host*.conf from template
$hostId = strtoupper($mySlot);
$hostIdLow = strtolower($mySlot);
$confFile = $hostIdLow . '.conf';
// Storage mode: use wizard selection, fall back to auto-detect from boot transport
$smParam = trim($_POST['storage_mode'] ?? '');
if ($smParam === 'flash') {
$storageInternal = 'false';
} elseif ($smParam === 'internal') {
$storageInternal = 'true';
} else {
$bootPart = trim(shell_exec('findmnt -n -o SOURCE /boot 2>/dev/null') ?: '');
$bootDisk = $bootPart ? trim(shell_exec('lsblk -no pkname ' . escapeshellarg($bootPart) . ' 2>/dev/null') ?: '') : '';
$transport = $bootDisk ? strtolower(trim(shell_exec('lsblk -dno TRAN /dev/' . escapeshellarg($bootDisk) . ' 2>/dev/null') ?: '')) : '';
$storageInternal = ($transport !== 'usb') ? 'true' : 'false';
}
if (!file_exists(CONF_DIR . '/' . $confFile)) {
$template = @file_get_contents(CONF_DIR . '/host.conf.template') ?: '';
if ($template) {
$sshOwner = strtolower(preg_replace('/^unraid-/i', '', $myHostname));
$sshKeyPath = '/root/.ssh/' . $sshOwner . '_rsync_automation';
$conf = str_replace('HOSTN', $hostId, $template);
$conf = str_replace('hostn', $hostIdLow, $conf);
$conf = preg_replace('/^(\s*' . $hostId . '_SSH_KEY\s*=\s*)""/m',
'${1}"' . $sshKeyPath . '"', $conf);
$conf = preg_replace('/^(\s*' . $hostId . '_STORAGE_MODE_INTERNAL\s*=\s*)\S+/m',
'${1}' . $storageInternal, $conf);
if (!vv_write_conf_raw($confFile, $conf)) {
echo json_encode(['ok' => false, 'error' => "Failed to write $confFile"]);
exit;
}
}
}
// Write setup state file — lets partner servers know HOST1 is configured
vv_setup_state_write(['host1_hostname' => $host1]);
// Auto-generate SSH keypair (local only — remote copy happens during onboarding)
if (file_exists($sshScript)) {
exec('bash ' . escapeshellarg($sshScript) . ' --local-only 2>/dev/null');
}
// Auto-create Unraid API key and write into the fresh conf
$apiKeyResult = vv_auto_create_api_key($hostId, $confFile);
$targetDir = ($storageInternal === 'true') ? '/boot/config/plugins/varaverk' : '/mnt/user/appdata/Varaverk';
$needsMigration = (defined('SCRIPTS_DIR') && SCRIPTS_DIR !== $targetDir);
echo json_encode([
'ok' => true,
'host_id' => $hostId,
'api_key' => $apiKeyResult,
'needs_migration'=> $needsMigration,
'migrate_to' => $needsMigration ? ($storageInternal === 'true' ? 'internal' : 'flash') : null,
'redirect' => '?tab=scheduler&vv_setup=master.conf',
]);
@@ -1,256 +0,0 @@
<?xml version='1.0' standalone='yes'?>
<!DOCTYPE PLUGIN [
<!ENTITY name "varaverk">
<!ENTITY author "gmer4lfe">
<!ENTITY version "2026.05.31">
<!ENTITY sha256 "d588470d6cc7f284601cb56039d5dfea6fb5bb1ee2c800657438a8edacafbd01">
<!ENTITY launch "varaverk/monitor">
<!ENTITY github "https://github.com/FailedProxy/Varaverk">
<!ENTITY branch "main">
<!ENTITY cfgdir "/boot/config/plugins/varaverk">
<!ENTITY plugdir "/usr/local/emhttp/plugins/varaverk">
<!ENTITY pkg "varaverk-&version;-noarch-1.txz">
]>
<PLUGIN name="&name;" author="&author;" version="&version;" launch="&launch;"
support="https://github.com/FailedProxy/Varaverk/issues"
icon="/plugins/varaverk/icons/varaverk.png">
<CHANGES>
###2026.05.31
- Packaged release: web files now ship as a .txz that Unraid reinstalls to RAM on every boot
- Survives reboots with zero manual steps (no symlink, no go script) — fixes plugin vanishing after OS upgrades
- Scripts are git-cloned to appdata on first install; web files stay on flash (~200KB)
- Updates handled in-UI (git pull); the plugin no longer pulls on every boot
###2026.05.30
- First-run setup wizard: auto-detects hostname, creates master.conf + host conf from templates
- Scheduler setup mode: after wizard, master.conf and host conf open sequentially with forced save flow
- Partnership tab: Onboard button highlighted on arrival from wizard; disabled until partner is configured
- HOST2 install paths: state-file pull, master.conf push detection, conf-only flow
- Onboard Step 9: master.conf automatically pushed to all listed hosts on onboard completion
- Graceful pre-onboard state: neutral banners instead of error warnings before SSH is configured
- GitHub link in tab bar and Settings page; Community Apps support URL
###2026.05.28
- Initial release: Monitor, Scheduler, Docker, Watchdog, Partnership, Fallback, Arrs tabs
- Mutual container fallback with tiered escalation and strike-confirmed handback
- Partnership lifecycle: onboard, offboard, transfer
- Rsync profile system with per-share container stops and writeback
- Watchdog: Tier 1 (explicit) + Tier 2 (global scan) container monitoring
- master.conf push-on-save to all configured partners via SSH
</CHANGES>
<!--
── 1. Web files symlink (runs on every boot) ──────────────────────────────────
Instead of extracting a .txz, we symlink the installed plugin web dir directly
to the workspace on flash. Changes to Plugin/unraid/ are live instantly — no
package build, no sync step. /boot is always mounted before this runs.
────────────────────────────────────────────────────────────────────────────────
-->
<FILE Run="/bin/bash">
<INLINE>
<![CDATA[
#!/bin/bash
WEB_DIR="/usr/local/emhttp/plugins/varaverk"
SRC="/boot/config/plugins/varaverk/Plugin/unraid"
[[ -L "$WEB_DIR" ]] && rm -f "$WEB_DIR"
[[ -d "$WEB_DIR" ]] && rm -rf "$WEB_DIR"
ln -sf "$SRC" "$WEB_DIR"
echo "[Varaverk] web dir symlinked → $SRC"
]]>
</INLINE>
</FILE>
<!--
── 2. Scripts bootstrap (first install only) ──────────────────────────────────
Clones the repo directly into the plugin config dir on flash (/boot/config/plugins/varaverk).
No array dependency — scripts live on flash (64GB NVMe) and are available at boot.
Uses git init+fetch+reset so the clone works into the non-empty cfgdir (varaverk.cfg,
varaverk-*.txz etc. are already there). Never auto-pulls — updates via the UI git pull.
Clone source priority:
1. Gitea (internal) — reads settings from varaverk.cfg; detects container IP at runtime
2. GitHub (public) — HTTPS fallback if Gitea is unreachable
────────────────────────────────────────────────────────────────────────────────
-->
<FILE Run="/bin/bash" Method="install">
<INLINE>
<![CDATA[
#!/bin/bash
PLUGIN="varaverk"
CFG_DIR="/boot/config/plugins/$PLUGIN"
CFG_FILE="$CFG_DIR/varaverk.cfg"
GITHUB="https://github.com/FailedProxy/Varaverk"
BRANCH="main"
LOG="$CFG_DIR/install.log"
mkdir -p "$CFG_DIR"
log() { echo "[$(date '+%H:%M:%S')] $*" | tee -a "$LOG"; }
# Scripts live in the plugin dir on flash — no array needed.
SCRIPTS_DIR="$CFG_DIR"
CONF_DIR="$SCRIPTS_DIR/Configurations"
# ── Boot device check ─────────────────────────────────────────────────────────
# Warn if /boot is on a USB/removable device. Varaverk is designed for internal
# NVMe/SSD boot — git repo + state files + data writes on USB will wear it out
# fast and may run out of space. Install proceeds but user is warned.
_boot_dev=$(df /boot --output=source 2>/dev/null | tail -1)
_boot_base=$(lsblk -no pkname "$_boot_dev" 2>/dev/null || basename "${_boot_dev%[0-9p]*}")
_removable=$(cat "/sys/block/${_boot_base}/removable" 2>/dev/null || echo "0")
if [[ "$_removable" == "1" ]]; then
log "WARNING: /boot is on a removable/USB device ($_boot_dev)"
log "WARNING: Varaverk is designed for internal NVMe/SSD boot."
log "WARNING: Running from USB risks drive wear and space exhaustion."
log "WARNING: Strongly recommend migrating boot to an internal NVMe/SSD drive."
fi
unset _boot_dev _boot_base _removable
# Seed varaverk.cfg with defaults (SCRIPTS_DIR + Gitea settings) if not present.
# Requires internal NVMe/SSD boot — scripts live on flash, available before array mounts.
if [[ ! -f "$CFG_FILE" ]]; then
cat > "$CFG_FILE" <<'CFGEOF'
SCRIPTS_DIR="/boot/config/plugins/varaverk"
GITEA_CONTAINER="Gitea"
GITEA_REPO_PATH="FailedProxy/Varaverk.git"
GITEA_SSH_KEY="/root/.ssh/unraid_gitea"
SSH_PORT="221"
CFGEOF
log "seeded varaverk.cfg"
fi
# Read Gitea settings from varaverk.cfg (allows override without editing .plg).
_read_cfg() { grep -oP "(?<=^${1}=\")[^\"]*" "$CFG_FILE" 2>/dev/null || echo "${2}"; }
GITEA_CONTAINER=$(_read_cfg GITEA_CONTAINER "Gitea")
GITEA_REPO_PATH=$(_read_cfg GITEA_REPO_PATH "FailedProxy/Varaverk.git")
GITEA_SSH_KEY=$(_read_cfg GITEA_SSH_KEY "/root/.ssh/unraid_gitea")
SSH_PORT=$(_read_cfg SSH_PORT "221")
# Clone on first install only; never auto-pull (updates via the UI git pull).
if [[ ! -d "$SCRIPTS_DIR/.git" ]]; then
log "initialising repo in $SCRIPTS_DIR ($BRANCH)..."
# Locate Gitea: local container → local IP; else Tailscale; else fall back to GitHub.
GITEA_IP=""
if command -v docker >/dev/null 2>&1 && \
docker ps --format "{{.Names}}" 2>/dev/null | grep -q "^${GITEA_CONTAINER}$"; then
GITEA_IP=$(hostname -I | awk '{print $1}')
log "Gitea running locally — using $GITEA_IP"
elif command -v tailscale >/dev/null 2>&1; then
# Try each known peer until we find one hosting Gitea
while IFS= read -r peer_ip; do
if ssh -i "$GITEA_SSH_KEY" -p "$SSH_PORT" \
-o ConnectTimeout=3 -o StrictHostKeyChecking=no \
-o BatchMode=yes "git@${peer_ip}" info 2>/dev/null | grep -q "varaverk\|Gitea\|gitea"; then
GITEA_IP="$peer_ip"
log "Gitea found on Tailscale peer $GITEA_IP"
break
fi
done < <(tailscale status --json 2>/dev/null | \
python3 -c "import json,sys; d=json.load(sys.stdin); \
[print(v['TailscaleIPs'][0]) for v in d.get('Peer',{}).values() \
if v.get('TailscaleIPs')]" 2>/dev/null)
fi
# init-in-place — git clone would fail because the dir already has files.
git -C "$SCRIPTS_DIR" init >> "$LOG" 2>&1
CLONED=false
if [[ -n "$GITEA_IP" && -f "$GITEA_SSH_KEY" ]]; then
GITEA_URL="ssh://git@${GITEA_IP}:${SSH_PORT}/${GITEA_REPO_PATH}"
log "trying Gitea: $GITEA_URL"
git -C "$SCRIPTS_DIR" remote add origin "$GITEA_URL" >> "$LOG" 2>&1
if GIT_SSH_COMMAND="ssh -i $GITEA_SSH_KEY -p $SSH_PORT -o StrictHostKeyChecking=no" \
GIT_TERMINAL_PROMPT=0 \
git -C "$SCRIPTS_DIR" fetch --depth=1 origin "$BRANCH" >> "$LOG" 2>&1; then
git -C "$SCRIPTS_DIR" reset --hard FETCH_HEAD >> "$LOG" 2>&1
git -C "$SCRIPTS_DIR" branch -M "$BRANCH" >> "$LOG" 2>&1
git -C "$SCRIPTS_DIR" branch --set-upstream-to=origin/"$BRANCH" "$BRANCH" >> "$LOG" 2>&1
log "scripts installed from Gitea ($GITEA_IP)"
CLONED=true
else
log "Gitea fetch failed — falling back to GitHub"
git -C "$SCRIPTS_DIR" remote remove origin >> "$LOG" 2>&1 || true
fi
fi
if [[ "$CLONED" == false ]]; then
log "trying GitHub: $GITHUB"
git -C "$SCRIPTS_DIR" remote add origin "$GITHUB" >> "$LOG" 2>&1
if GIT_TERMINAL_PROMPT=0 \
git -C "$SCRIPTS_DIR" fetch --depth=1 origin "$BRANCH" >> "$LOG" 2>&1; then
git -C "$SCRIPTS_DIR" reset --hard FETCH_HEAD >> "$LOG" 2>&1
git -C "$SCRIPTS_DIR" branch -M "$BRANCH" >> "$LOG" 2>&1
git -C "$SCRIPTS_DIR" branch --set-upstream-to=origin/"$BRANCH" "$BRANCH" >> "$LOG" 2>&1
log "scripts installed from GitHub"
CLONED=true
else
log "WARNING: both Gitea and GitHub failed — scripts not installed, retry when network is up"
exit 0
fi
fi
else
log "repo present — leaving scripts untouched (update from the UI)"
fi
# Seed master.conf from template if absent.
mkdir -p "$CONF_DIR"
if [[ ! -f "$CONF_DIR/master.conf" && -f "$SCRIPTS_DIR/Deployment/conf_templates/master.conf" ]]; then
cp "$SCRIPTS_DIR/Deployment/conf_templates/master.conf" "$CONF_DIR/master.conf"
log "seeded master.conf from template"
fi
log "install step complete"
]]>
</INLINE>
</FILE>
<!--
── 3. Remove ───────────────────────────────────────────────────────────────────
Stops background scripts, removes cron, the installed package, and flash config.
Scripts/conf in appdata are left intact (delete manually for a full wipe).
────────────────────────────────────────────────────────────────────────────────
-->
<FILE Run="/bin/bash" Method="remove">
<INLINE>
<![CDATA[
#!/bin/bash
PLUGIN="varaverk"
CFG_DIR="/boot/config/plugins/$PLUGIN"
CFG_FILE="$CFG_DIR/varaverk.cfg"
CRON_FILE="$CFG_DIR/varaverk.cron"
WEB_DIR="/usr/local/emhttp/plugins/$PLUGIN"
log() { echo "[Varaverk remove] $*"; }
[[ -f "$CFG_FILE" ]] && _sd=$(grep -oP '(?<=SCRIPTS_DIR=")[^"]+' "$CFG_FILE" 2>/dev/null)
SCRIPTS_DIR="${_sd:-$CFG_DIR}"
# Stop continuous background scripts.
if [[ -f "$SCRIPTS_DIR/Fallback/fallback.sh" ]]; then
bash "$SCRIPTS_DIR/Fallback/fallback.sh" --stop 2>/dev/null && log "fallback.sh stopped" || true
fi
pkill -f "run_job.sh" 2>/dev/null || true
pkill -f "watchdog_orchestrator.sh" 2>/dev/null || true
# Remove cron entries.
if [[ -f "$CRON_FILE" ]]; then
rm -f "$CRON_FILE"
/usr/local/sbin/update_cron 2>/dev/null || true
log "cron removed"
fi
rm -f /etc/cron.d/varaverk
# Remove the installed package (and its RAM files).
removepkg "$PLUGIN" 2>/dev/null || true
[[ -L "$WEB_DIR" ]] && rm -f "$WEB_DIR"
[[ -d "$WEB_DIR" ]] && rm -rf "$WEB_DIR"
log "web files removed"
# Remove flash config (incl. cached .txz).
rm -rf "$CFG_DIR"
log "flash config removed"
log "done — scripts/conf in $SCRIPTS_DIR preserved (delete manually for full wipe)"
]]>
</INLINE>
</FILE>
</PLUGIN>
@@ -1,256 +0,0 @@
<?xml version='1.0' standalone='yes'?>
<!DOCTYPE PLUGIN [
<!ENTITY name "varaverk">
<!ENTITY author "gmer4lfe">
<!ENTITY version "2026.05.31">
<!ENTITY sha256 "d588470d6cc7f284601cb56039d5dfea6fb5bb1ee2c800657438a8edacafbd01">
<!ENTITY launch "varaverk/monitor">
<!ENTITY github "https://github.com/FailedProxy/Varaverk">
<!ENTITY branch "main">
<!ENTITY cfgdir "/boot/config/plugins/varaverk">
<!ENTITY plugdir "/usr/local/emhttp/plugins/varaverk">
<!ENTITY pkg "varaverk-&version;-noarch-1.txz">
]>
<PLUGIN name="&name;" author="&author;" version="&version;" launch="&launch;"
support="https://github.com/FailedProxy/Varaverk/issues"
icon="/plugins/varaverk/icons/varaverk.png">
<CHANGES>
###2026.05.31
- Packaged release: web files now ship as a .txz that Unraid reinstalls to RAM on every boot
- Survives reboots with zero manual steps (no symlink, no go script) — fixes plugin vanishing after OS upgrades
- Scripts are git-cloned to appdata on first install; web files stay on flash (~200KB)
- Updates handled in-UI (git pull); the plugin no longer pulls on every boot
###2026.05.30
- First-run setup wizard: auto-detects hostname, creates master.conf + host conf from templates
- Scheduler setup mode: after wizard, master.conf and host conf open sequentially with forced save flow
- Partnership tab: Onboard button highlighted on arrival from wizard; disabled until partner is configured
- HOST2 install paths: state-file pull, master.conf push detection, conf-only flow
- Onboard Step 9: master.conf automatically pushed to all listed hosts on onboard completion
- Graceful pre-onboard state: neutral banners instead of error warnings before SSH is configured
- GitHub link in tab bar and Settings page; Community Apps support URL
###2026.05.28
- Initial release: Monitor, Scheduler, Docker, Watchdog, Partnership, Fallback, Arrs tabs
- Mutual container fallback with tiered escalation and strike-confirmed handback
- Partnership lifecycle: onboard, offboard, transfer
- Rsync profile system with per-share container stops and writeback
- Watchdog: Tier 1 (explicit) + Tier 2 (global scan) container monitoring
- master.conf push-on-save to all configured partners via SSH
</CHANGES>
<!--
── 1. Web files symlink (runs on every boot) ──────────────────────────────────
Instead of extracting a .txz, we symlink the installed plugin web dir directly
to the workspace on flash. Changes to Plugin/unraid/ are live instantly — no
package build, no sync step. /boot is always mounted before this runs.
────────────────────────────────────────────────────────────────────────────────
-->
<FILE Run="/bin/bash">
<INLINE>
<![CDATA[
#!/bin/bash
WEB_DIR="/usr/local/emhttp/plugins/varaverk"
SRC="/boot/config/plugins/varaverk/Plugin/unraid"
[[ -L "$WEB_DIR" ]] && rm -f "$WEB_DIR"
[[ -d "$WEB_DIR" ]] && rm -rf "$WEB_DIR"
ln -sf "$SRC" "$WEB_DIR"
echo "[Varaverk] web dir symlinked → $SRC"
]]>
</INLINE>
</FILE>
<!--
── 2. Scripts bootstrap (first install only) ──────────────────────────────────
Clones the repo directly into the plugin config dir on flash (/boot/config/plugins/varaverk).
No array dependency — scripts live on flash (64GB NVMe) and are available at boot.
Uses git init+fetch+reset so the clone works into the non-empty cfgdir (varaverk.cfg,
varaverk-*.txz etc. are already there). Never auto-pulls — updates via the UI git pull.
Clone source priority:
1. Gitea (internal) — reads settings from varaverk.cfg; detects container IP at runtime
2. GitHub (public) — HTTPS fallback if Gitea is unreachable
────────────────────────────────────────────────────────────────────────────────
-->
<FILE Run="/bin/bash" Method="install">
<INLINE>
<![CDATA[
#!/bin/bash
PLUGIN="varaverk"
CFG_DIR="/boot/config/plugins/$PLUGIN"
CFG_FILE="$CFG_DIR/varaverk.cfg"
GITHUB="https://github.com/FailedProxy/Varaverk"
BRANCH="main"
LOG="$CFG_DIR/install.log"
mkdir -p "$CFG_DIR"
log() { echo "[$(date '+%H:%M:%S')] $*" | tee -a "$LOG"; }
# Scripts live in the plugin dir on flash — no array needed.
SCRIPTS_DIR="$CFG_DIR"
CONF_DIR="$SCRIPTS_DIR/Configurations"
# ── Boot device check ─────────────────────────────────────────────────────────
# Warn if /boot is on a USB/removable device. Varaverk is designed for internal
# NVMe/SSD boot — git repo + state files + data writes on USB will wear it out
# fast and may run out of space. Install proceeds but user is warned.
_boot_dev=$(df /boot --output=source 2>/dev/null | tail -1)
_boot_base=$(lsblk -no pkname "$_boot_dev" 2>/dev/null || basename "${_boot_dev%[0-9p]*}")
_removable=$(cat "/sys/block/${_boot_base}/removable" 2>/dev/null || echo "0")
if [[ "$_removable" == "1" ]]; then
log "WARNING: /boot is on a removable/USB device ($_boot_dev)"
log "WARNING: Varaverk is designed for internal NVMe/SSD boot."
log "WARNING: Running from USB risks drive wear and space exhaustion."
log "WARNING: Strongly recommend migrating boot to an internal NVMe/SSD drive."
fi
unset _boot_dev _boot_base _removable
# Seed varaverk.cfg with defaults (SCRIPTS_DIR + Gitea settings) if not present.
# Requires internal NVMe/SSD boot — scripts live on flash, available before array mounts.
if [[ ! -f "$CFG_FILE" ]]; then
cat > "$CFG_FILE" <<'CFGEOF'
SCRIPTS_DIR="/boot/config/plugins/varaverk"
GITEA_CONTAINER="Gitea"
GITEA_REPO_PATH="FailedProxy/Varaverk.git"
GITEA_SSH_KEY="/root/.ssh/unraid_gitea"
SSH_PORT="221"
CFGEOF
log "seeded varaverk.cfg"
fi
# Read Gitea settings from varaverk.cfg (allows override without editing .plg).
_read_cfg() { grep -oP "(?<=^${1}=\")[^\"]*" "$CFG_FILE" 2>/dev/null || echo "${2}"; }
GITEA_CONTAINER=$(_read_cfg GITEA_CONTAINER "Gitea")
GITEA_REPO_PATH=$(_read_cfg GITEA_REPO_PATH "FailedProxy/Varaverk.git")
GITEA_SSH_KEY=$(_read_cfg GITEA_SSH_KEY "/root/.ssh/unraid_gitea")
SSH_PORT=$(_read_cfg SSH_PORT "221")
# Clone on first install only; never auto-pull (updates via the UI git pull).
if [[ ! -d "$SCRIPTS_DIR/.git" ]]; then
log "initialising repo in $SCRIPTS_DIR ($BRANCH)..."
# Locate Gitea: local container → local IP; else Tailscale; else fall back to GitHub.
GITEA_IP=""
if command -v docker >/dev/null 2>&1 && \
docker ps --format "{{.Names}}" 2>/dev/null | grep -q "^${GITEA_CONTAINER}$"; then
GITEA_IP=$(hostname -I | awk '{print $1}')
log "Gitea running locally — using $GITEA_IP"
elif command -v tailscale >/dev/null 2>&1; then
# Try each known peer until we find one hosting Gitea
while IFS= read -r peer_ip; do
if ssh -i "$GITEA_SSH_KEY" -p "$SSH_PORT" \
-o ConnectTimeout=3 -o StrictHostKeyChecking=no \
-o BatchMode=yes "git@${peer_ip}" info 2>/dev/null | grep -q "varaverk\|Gitea\|gitea"; then
GITEA_IP="$peer_ip"
log "Gitea found on Tailscale peer $GITEA_IP"
break
fi
done < <(tailscale status --json 2>/dev/null | \
python3 -c "import json,sys; d=json.load(sys.stdin); \
[print(v['TailscaleIPs'][0]) for v in d.get('Peer',{}).values() \
if v.get('TailscaleIPs')]" 2>/dev/null)
fi
# init-in-place — git clone would fail because the dir already has files.
git -C "$SCRIPTS_DIR" init >> "$LOG" 2>&1
CLONED=false
if [[ -n "$GITEA_IP" && -f "$GITEA_SSH_KEY" ]]; then
GITEA_URL="ssh://git@${GITEA_IP}:${SSH_PORT}/${GITEA_REPO_PATH}"
log "trying Gitea: $GITEA_URL"
git -C "$SCRIPTS_DIR" remote add origin "$GITEA_URL" >> "$LOG" 2>&1
if GIT_SSH_COMMAND="ssh -i $GITEA_SSH_KEY -p $SSH_PORT -o StrictHostKeyChecking=no" \
GIT_TERMINAL_PROMPT=0 \
git -C "$SCRIPTS_DIR" fetch --depth=1 origin "$BRANCH" >> "$LOG" 2>&1; then
git -C "$SCRIPTS_DIR" reset --hard FETCH_HEAD >> "$LOG" 2>&1
git -C "$SCRIPTS_DIR" branch -M "$BRANCH" >> "$LOG" 2>&1
git -C "$SCRIPTS_DIR" branch --set-upstream-to=origin/"$BRANCH" "$BRANCH" >> "$LOG" 2>&1
log "scripts installed from Gitea ($GITEA_IP)"
CLONED=true
else
log "Gitea fetch failed — falling back to GitHub"
git -C "$SCRIPTS_DIR" remote remove origin >> "$LOG" 2>&1 || true
fi
fi
if [[ "$CLONED" == false ]]; then
log "trying GitHub: $GITHUB"
git -C "$SCRIPTS_DIR" remote add origin "$GITHUB" >> "$LOG" 2>&1
if GIT_TERMINAL_PROMPT=0 \
git -C "$SCRIPTS_DIR" fetch --depth=1 origin "$BRANCH" >> "$LOG" 2>&1; then
git -C "$SCRIPTS_DIR" reset --hard FETCH_HEAD >> "$LOG" 2>&1
git -C "$SCRIPTS_DIR" branch -M "$BRANCH" >> "$LOG" 2>&1
git -C "$SCRIPTS_DIR" branch --set-upstream-to=origin/"$BRANCH" "$BRANCH" >> "$LOG" 2>&1
log "scripts installed from GitHub"
CLONED=true
else
log "WARNING: both Gitea and GitHub failed — scripts not installed, retry when network is up"
exit 0
fi
fi
else
log "repo present — leaving scripts untouched (update from the UI)"
fi
# Seed master.conf from template if absent.
mkdir -p "$CONF_DIR"
if [[ ! -f "$CONF_DIR/master.conf" && -f "$CONF_DIR/master.conf.template" ]]; then
cp "$CONF_DIR/master.conf.template" "$CONF_DIR/master.conf"
log "seeded master.conf from template"
fi
log "install step complete"
]]>
</INLINE>
</FILE>
<!--
── 3. Remove ───────────────────────────────────────────────────────────────────
Stops background scripts, removes cron, the installed package, and flash config.
Scripts/conf in appdata are left intact (delete manually for a full wipe).
────────────────────────────────────────────────────────────────────────────────
-->
<FILE Run="/bin/bash" Method="remove">
<INLINE>
<![CDATA[
#!/bin/bash
PLUGIN="varaverk"
CFG_DIR="/boot/config/plugins/$PLUGIN"
CFG_FILE="$CFG_DIR/varaverk.cfg"
CRON_FILE="$CFG_DIR/varaverk.cron"
WEB_DIR="/usr/local/emhttp/plugins/$PLUGIN"
log() { echo "[Varaverk remove] $*"; }
[[ -f "$CFG_FILE" ]] && _sd=$(grep -oP '(?<=SCRIPTS_DIR=")[^"]+' "$CFG_FILE" 2>/dev/null)
SCRIPTS_DIR="${_sd:-$CFG_DIR}"
# Stop continuous background scripts.
if [[ -f "$SCRIPTS_DIR/Fallback/fallback.sh" ]]; then
bash "$SCRIPTS_DIR/Fallback/fallback.sh" --stop 2>/dev/null && log "fallback.sh stopped" || true
fi
pkill -f "run_job.sh" 2>/dev/null || true
pkill -f "watchdog_orchestrator.sh" 2>/dev/null || true
# Remove cron entries.
if [[ -f "$CRON_FILE" ]]; then
rm -f "$CRON_FILE"
/usr/local/sbin/update_cron 2>/dev/null || true
log "cron removed"
fi
rm -f /etc/cron.d/varaverk
# Remove the installed package (and its RAM files).
removepkg "$PLUGIN" 2>/dev/null || true
[[ -L "$WEB_DIR" ]] && rm -f "$WEB_DIR"
[[ -d "$WEB_DIR" ]] && rm -rf "$WEB_DIR"
log "web files removed"
# Remove flash config (incl. cached .txz).
rm -rf "$CFG_DIR"
log "flash config removed"
log "done — scripts/conf in $SCRIPTS_DIR preserved (delete manually for full wipe)"
]]>
</INLINE>
</FILE>
</PLUGIN>
@@ -1,469 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ========================== HOSTN CONFIGURATION — (hostname) ==================================
# ==============================================================================================
# HOSTN-specific variables — credentials, container names, share paths, failover lists.
# Sourced after master.conf — values here extend shared profile arrays and add HOSTN-specific
# identity, credentials, and container configuration.
#
# Sparse checkout (git) ensures other hosts never receive this file.
#
# DO NOT put shared config here — thresholds, toggles, profiles belong in master.conf.
# DO NOT put other hosts' variables here — they belong in their own host*.conf files.
#
# ── HOW TO USE THIS TEMPLATE ──────────────────────────────────────────────────────────────────
# This file was generated by the Varaverk first-run wizard.
# Fill in the sections that apply to your setup — leave unused sections empty.
# All scripts self-guard against empty values — safe to leave sections blank until needed.
#
# ── INDEX ─────────────────────────────────────────────────────────────────────────────────────
#
# ── IDENTITY & CONNECTIVITY ────────────────────────────────────────────────────────────────
# IDENTITY hostname, SSH key
# EMBY container name, URL, API key
# NOTIFICATIONS Discord webhook
# PARTNERSHIP auth containers, backup paths
#
# ── RSYNC ──────────────────────────────────────────────────────────────────────────────────
# DAILY SYNC SHARES media shares this host owns and pushes
# WEEKLY SYNC SHARES appdata shares synced weekly
# CRITICAL SYNC SHARES appdata shares synced every 30 minutes
# HOSTN RSYNC PROFILE host-specific appdata sync profile
#
# ── DOCKER ─────────────────────────────────────────────────────────────────────────────────
# DOCKER DAILY RESTART containers restarted daily
# DOCKER WEEKLY RESTART containers restarted weekly
# DOCKER WATCHDOG memory limits, health URLs, required containers
# DOCKER NETWORK CONNECT networks and containers for array start
#
# ── FALLBACK ───────────────────────────────────────────────────────────────────────────────
# DDNS DDNS containers managed by this host
# INTERNET LOSS containers stopped when internet is lost
# FALLBACK TIERS what this host runs for the remote per tier
# TIER DELAYS delays before each tier activates
# RSYNC WRITEBACK appdata synced back on handback
#
# ── MEDIA ──────────────────────────────────────────────────────────────────────────────────
# MEDIA PERMISSIONS share list for permissions script
# MEDIA CLEANER folder lists for media_cleaner.sh
#
# ── MONITORS ───────────────────────────────────────────────────────────────────────────────
# CERTIFICATE MONITOR domains checked for SSL expiry
# SMART HEALTH drives to skip in SMART monitoring
# ZFS REPORT pools to exclude from ZFS health report
#
# ── TRANSCODES ─────────────────────────────────────────────────────────────────────────────
# TRANSCODES ramdisk size, thresholds, SSD path, server array
#
# ── ARR STACK ──────────────────────────────────────────────────────────────────────────────
# DOWNLOADERS slskd, SABnzbd, qBittorrent credentials and URLs
# LIDARR / SONARR / RADARR URL, API key, path map
# ARR RECOVERY per-arr recovery toggles
#
# ── SYSTEM WATCHDOG ────────────────────────────────────────────────────────────────────────
# SYSTEM WATCHDOG per-host check toggles and NIC configuration
#
# ── RESOURCE MANAGER ───────────────────────────────────────────────────────────────────────
# RESOURCE MANAGER containers paused/stopped under memory pressure
#
# ==============================================================================================
# ==============================================================================================
# ── STORAGE MODE ──────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Storage mode ━━━
# Controls where Varaverk stores scripts, conf, and state files.
# true = internal NVMe/SSD — /boot/config/plugins/varaverk (write-safe, git-direct)
# false = USB flash boot — /mnt/user/appdata/Varaverk (preserves flash lifetime)
# Auto-detected from boot device transport on first setup.
# To change: Settings → Storage → Migrate.
HOSTN_STORAGE_MODE_INTERNAL=true
# ==============================================================================================
# ── IDENTITY & CONNECTIVITY ───────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Identity ━━━
# HOSTN hostname lives in master.conf (not a credential — safe for all servers).
# SSH key used for all server-to-server operations — rsync, failover, conf sync.
# Convention: /root/.ssh/<hostname-lowercase-no-unraid-prefix>_rsync_automation
# Must be in /root/.ssh/ and authorised in the partner's /root/.ssh/authorized_keys.
# Run Partnership/ssh_setup.sh to generate the key and copy it to the partner.
HOSTN_SSH_KEY="" # e.g. /root/.ssh/myserver_rsync_automation
HOSTN_OWNER="" # short identifier for this server (e.g. myserver)
HOSTN_OWNER_EMAIL=""
# ━━━ Unraid API ━━━
# Used by the Varaverk plugin to query this server's Unraid GraphQL API.
# Generate in Unraid: Settings → Management Access → API Keys → + New Key
HOSTN_UNRAID_API_KEY=""
# ━━━ Emby ━━━
HOSTN_EMBY_CONTAINER="Emby"
HOSTN_EMBY_URL="http://localhost:8096"
HOSTN_EMBY_API_KEY="" # Emby Dashboard → API Keys → + New Key
# ━━━ Jellyfin ━━━
HOSTN_JELLYFIN_CONTAINER="Jellyfin"
HOSTN_JELLYFIN_URL="http://localhost:8095"
HOSTN_JELLYFIN_API_KEY="" # Jellyfin Dashboard → Administration → API Keys
# ━━━ Gitea ━━━
# Personal access token for gitea_ssh_setup.sh.
# Create in Gitea: Settings → Applications → Generate Token → scope: write:user
HOSTN_GITEA_API_TOKEN=""
# ━━━ Notifications ━━━
# Discord webhook — leave blank to disable.
HOSTN_DISCORD_WEBHOOK=""
# ━━━ Partnership ━━━
# Auth containers reconfigured on onboard/offboard.
# Format: "ContainerName|WebUIPort"
HOSTN_PARTNERSHIP_AUTH_WEBUIS=(
# "NginxProxyManager|81"
# "Authelia|9091"
)
# XML templates pushed to mirror during onboard — auth stack.
# Dependencies (databases) must come before apps that depend on them.
HOSTN_PARTNERSHIP_AUTH_STACK=(
# "my-Authelia.xml"
# "my-NginxProxyManager.xml"
)
# XML templates pushed to mirror during onboard — arr stack.
HOSTN_PARTNERSHIP_ARR_STACK=(
# "my-Sonarr.xml"
# "my-Radarr.xml"
)
# Paths the partner should collect during the grace window after offboard.
HOSTN_PARTNERSHIP_MIRROR_BACKUPS=(
# "/mnt/user/appdata-Fallback/Partner-Emby"
)
# Containers parked on this server when partnership is active.
HOSTN_PARTNERSHIP_OWN_CONTAINERS=(
# "Emby"
)
# Emby admin provisioning — owner controls whether Emby is shared.
HOSTN_PARTNERSHIP_PROVISION_EMBY_ADMIN=false
HOSTN_PARTNERSHIP_EMBY_PORT=8096
HOSTN_PARTNERSHIP_EMBY_ADMIN_USER=""
HOSTN_PARTNERSHIP_EMBY_ADMIN_PASS=""
# ==============================================================================================
# ── RSYNC ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Daily Sync Shares ━━━
# Media shares this host pushes to all other nodes every night.
# Uses DEFAULT_RSYNC_OPTS from master.conf — no profile needed.
HOSTN_DAILY_SYNC_SHARES=(
# /mnt/user/Movies
# /mnt/user/Tv_Shows
# /mnt/user/Music
)
# ━━━ Weekly Sync Shares ━━━
# Appdata shares synced during the weekly maintenance window.
# Profiles (emby, critical-data) drive container stops — define in master.conf.
HOSTN_WEEKLY_SYNC_SHARES=(
# "/mnt/user/Media_Server/Emby" # emby profile
# "/mnt/user/appdata-Fallback/Critical-Data" # critical-data profile
)
# ━━━ Intermediate Sync Shares ━━━
# Shares synced every 4 hours. Leave empty to skip mid-day rsync.
HOSTN_INTERMEDIATE_SYNC_SHARES=(
# Add shares here to enable mid-day rsync
)
# ━━━ Critical Sync Shares ━━━
# Appdata shares synced every 30 minutes.
# Format: "/path/to/share" or "/path/to/share|profile-name"
HOSTN_CRITICAL_SYNC_SHARES=(
# "/mnt/user/appdata-Fallback/Critical-Data|critical-fallback"
# "/mnt/user/Media_Server/Emby|emby-fallback"
)
# ━━━ Backup Verify ━━━
# Leave empty to use HOSTN_DAILY_SYNC_SHARES automatically.
HOSTN_BACKUP_VERIFY_SHARES=(
# leave empty to use HOSTN_DAILY_SYNC_SHARES automatically
)
# ━━━ HOSTN Rsync Profile — hostn-appdata ━━━
# Host-specific appdata sync profile.
PROFILE_RSYNC_OPTS[hostn-appdata]="-av --info=progress2 --bwlimit=${PROFILE_BW_LIMIT[hostn-appdata]:-8000}"
PROFILE_BW_LIMIT[hostn-appdata]=8000
PROFILE_RETRY_COUNT[hostn-appdata]=3
PROFILE_SLEEP[hostn-appdata]=300
PROFILE_CRITICAL_CONTAINER_NAMES[hostn-appdata]=""
PROFILE_DELAYED_CONTAINERS[hostn-appdata]=""
PROFILE_CONTAINER_DELAY[hostn-appdata]=5
PROFILE_EXCLUDE_DIRS[hostn-appdata]="logs *.tmp"
# ==============================================================================================
# ── DOCKER ────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Docker Daily Restart ━━━
HOSTN_DAILY_RESTART_CONTAINERS=(
# "NginxProxyManager"
# "Authelia"
)
# ━━━ Docker Weekly Restart ━━━
HOSTN_WEEKLY_RESTART_CONTAINERS=(
# "NextCloud"
# "AdGuard-Home"
)
# ━━━ Docker Watchdog ━━━
# Memory hard limits in MB — immediate restart if exceeded.
# 20GB=20480 16GB=16384 8GB=8192 4GB=4096 2GB=2048 1GB=1024
declare -A HOSTN_WATCHDOG_CONTAINERS=(
# ["Emby"]=18432
)
# HTTP health check URLs — checked every cycle.
declare -A HOSTN_WATCHDOG_CONTAINER_URLS=(
# ["Emby"]="http://localhost:8096"
)
# Required containers — must always be running.
HOSTN_WATCHDOG_REQUIRED_CONTAINERS=(
# "NginxProxyManager"
# "Authelia"
)
# Containers to skip in Tier 2 global scan.
HOSTN_WATCHDOG_SCAN_IGNORE=(
# "my-occasional-container"
)
# Dependency ordering — skip restarting a container if its dependency is also down.
declare -A HOSTN_WATCHDOG_DEPENDENCIES=(
# ["Authelia"]="Mariadb Redis-Authelia"
)
# Per-container appdata growth suppress ceilings in MB.
declare -A HOSTN_WATCHDOG_APPDATA_SIZES=(
# ["Tdarr"]="25600"
)
# ━━━ Network Watchdog ━━━
HOSTN_NETWORK_WATCHDOG_DDNS_DOMAIN="" # e.g. myserver.com
HOSTN_NETWORK_WATCHDOG_DDNS_CONTAINER="" # e.g. MyServer.com
HOSTN_NETWORK_WATCHDOG_NPM_URL="" # e.g. https://myserver.com
# ━━━ Docker Network Connect ━━━
HOSTN_NETWORK_CONNECT_CONTAINERS=(
# "memcached"
)
HOSTN_NETWORK_CONNECT_NETWORKS=(
# "high-availability"
)
# ==============================================================================================
# ── FALLBACK ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ DDNS ━━━
# DDNS containers this host manages.
HOSTN_DDNS_CONTAINERS=(
# "MyServer.com"
)
# ━━━ Internet Loss ━━━
# Containers stopped immediately when internet is lost.
FALLBACK_HOSTN_STOP_ON_NO_NET=(
# "MyServer.com"
)
# ━━━ Fallback Tiers — HOSTN Runs for Partner ━━━
# Containers this host starts when the partner goes down.
# Replace REMOTE_ID below with the actual remote host ID (HOST1, HOST2, etc.)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER1=(
# "Partner-DDNS-Container"
)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER2=(
# "container-placeholder"
)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER3=(
# "container-placeholder"
)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER4=(
# "container-placeholder"
)
# ━━━ Tier Delays — Partner's Containers on this Host ━━━
# How long the partner must be down before each tier activates here — in minutes.
# Replace REMOTE_ID with the actual remote host ID (HOST1, HOST2, etc.)
REMOTE_ID_TIER2_DELAY=240 # 4 hours
REMOTE_ID_TIER3_DELAY=720 # 12 hours
REMOTE_ID_TIER4_DELAY=1440 # 24 hours
# ━━━ Rsync Writeback ━━━
HOSTN_TIER1_WRITEBACK_DELAY=60 # skip Emby writeback if outage under 1hr
FALLBACK_HOSTN_WRITEBACK_TIER1=(
# "/mnt/user/Media_Server/Emby"
)
FALLBACK_HOSTN_WRITEBACK_TIER2=(
# "/mnt/user/appdata-Fallback/Important-Data"
)
FALLBACK_HOSTN_WRITEBACK_TIER3=(
# "location-placeholder"
)
FALLBACK_HOSTN_WRITEBACK_TIER4=(
# "/mnt/user/appdata-Fallback/Arrs_Stack"
)
# ==============================================================================================
# ── MEDIA ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Media Permissions ━━━
HOSTN_MEDIA_PERMISSION_SHARES=(
# /mnt/user/Movies
# /mnt/user/Tv_Shows
# /mnt/user/Music
# /mnt/user/Downloads
)
# ━━━ Media Cleaner ━━━
HOSTN_ANIME_CLEAN_FOLDERS=(
# /mnt/user/Anime_Movies
# /mnt/user/Anime_Shows
)
HOSTN_MEDIA_CLEAN_FOLDERS=(
# /mnt/user/Movies
# /mnt/user/Tv_Shows
)
# ==============================================================================================
# ── MONITORS ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
HOSTN_CERT_MONITOR_DOMAINS=(
# "myserver.com"
)
# ━━━ SMART Health ━━━
HOSTN_SMART_IGNORE_DRIVES=(
"sda" # boot USB — SMART not meaningful on flash drives
)
# ━━━ ZFS Report ━━━
HOSTN_ZFS_REPORT_IGNORE_POOLS=(
# "disk5"
)
# ==============================================================================================
# ── TRANSCODES ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOSTN_RAMDISK_SIZE="10G"
HOSTN_RAMDISK_WARN_GB=8.5
HOSTN_RAMDISK_LOW_GB=7
HOSTN_TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/"
HOSTN_TRANSCODE_SERVERS=(
"${HOSTN_EMBY_CONTAINER}|${HOSTN_EMBY_URL}|${HOSTN_EMBY_API_KEY}|emby"
)
# ==============================================================================================
# ── ARR STACK ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Downloaders ━━━
HOSTN_SLSKD_URL="http://localhost:8980"
HOSTN_SLSKD_API_KEY=""
HOSTN_SLSKD_FAILED_IMPORTS_DIR=""
HOSTN_SABNZBD_URL="http://localhost:8180"
HOSTN_SABNZBD_API_KEY=""
HOSTN_QBIT_URL="http://localhost:8080"
HOSTN_QBIT_USERNAME="admin"
HOSTN_QBIT_PASSWORD=""
# ━━━ Lidarr ━━━
HOSTN_LIDARR_URL="http://localhost:8686"
HOSTN_LIDARR_API_KEY=""
HOSTN_LIDARR_MUSIC_ROOT="/mnt/user/Music"
HOSTN_FANART_API_KEY=""
HOSTN_LASTFM_API_KEY=""
declare -A HOSTN_LIDARR_PATH_MAP=(
# ["/music"]="/mnt/user/Music"
)
# ━━━ Sonarr ━━━
HOSTN_SONARR_URL="http://localhost:8989"
HOSTN_SONARR_API_KEY=""
HOSTN_SONARR_TV_ROOT="/mnt/user/Tv_Shows"
declare -A HOSTN_SONARR_PATH_MAP=(
# ["/tv"]="/mnt/user/Tv_Shows"
)
# ━━━ Radarr ━━━
HOSTN_RADARR_URL="http://localhost:7878"
HOSTN_RADARR_API_KEY=""
HOSTN_RADARR_MOVIE_ROOT="/mnt/user/Movies"
declare -A HOSTN_RADARR_PATH_MAP=(
# ["/movies"]="/mnt/user/Movies"
)
# ━━━ Arr Recovery Toggles ━━━
HOSTN_LIDARR_RECOVERY=false
HOSTN_SONARR_RECOVERY=true
HOSTN_RADARR_RECOVERY=true
# ==============================================================================================
# ── SYSTEM WATCHDOG ───────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOSTN_SYS_WATCHDOG_CHECK_DOCKER=true
HOSTN_SYS_WATCHDOG_CHECK_ROOTFS=true
HOSTN_SYS_WATCHDOG_CHECK_RAM=true
HOSTN_SYS_WATCHDOG_CHECK_LOAD=true
HOSTN_SYS_WATCHDOG_CHECK_TEMP=true
HOSTN_SYS_WATCHDOG_CHECK_ZOMBIES=true
HOSTN_SYS_WATCHDOG_CHECK_LOG=true
HOSTN_SYS_WATCHDOG_CHECK_TMP=true
HOSTN_SYS_WATCHDOG_CHECK_FD=true
HOSTN_SYS_WATCHDOG_CHECK_RUNAWAY=true
HOSTN_SYS_WATCHDOG_NIC="" # e.g. eth0 — for network monitoring
# ==============================================================================================
# ── RESOURCE MANAGER ──────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOSTN_RW_PAUSE_CONTAINERS=(
# "Tdarr"
# "LidaTube"
)
HOSTN_RW_STOP_CONTAINERS=(
# "Tdarr"
)
@@ -1,484 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ========================== HOSTN CONFIGURATION — (hostname) ==================================
# ==============================================================================================
# HOSTN-specific variables — credentials, container names, share paths, failover lists.
# Sourced after master.conf — values here extend shared profile arrays and add HOSTN-specific
# identity, credentials, and container configuration.
#
# Sparse checkout (git) ensures other hosts never receive this file.
#
# DO NOT put shared config here — thresholds, toggles, profiles belong in master.conf.
# DO NOT put other hosts' variables here — they belong in their own host*.conf files.
#
# ── HOW TO USE THIS TEMPLATE ──────────────────────────────────────────────────────────────────
# This file was generated by the Varaverk first-run wizard.
# Fill in the sections that apply to your setup — leave unused sections empty.
# All scripts self-guard against empty values — safe to leave sections blank until needed.
#
# ── INDEX ─────────────────────────────────────────────────────────────────────────────────────
#
# ── IDENTITY & CONNECTIVITY ────────────────────────────────────────────────────────────────
# IDENTITY hostname, SSH key
# EMBY container name, URL, API key
# NOTIFICATIONS Discord webhook
# PARTNERSHIP auth containers, backup paths
#
# ── RSYNC ──────────────────────────────────────────────────────────────────────────────────
# DAILY SYNC SHARES media shares this host owns and pushes
# WEEKLY SYNC SHARES appdata shares synced weekly
# CRITICAL SYNC SHARES appdata shares synced every 30 minutes
# HOSTN RSYNC PROFILE host-specific appdata sync profile
#
# ── DOCKER ─────────────────────────────────────────────────────────────────────────────────
# DOCKER DAILY RESTART containers restarted daily
# DOCKER WEEKLY RESTART containers restarted weekly
# DOCKER WATCHDOG memory limits, health URLs, required containers
# DOCKER NETWORK CONNECT networks and containers for array start
#
# ── FALLBACK ───────────────────────────────────────────────────────────────────────────────
# DDNS DDNS containers managed by this host
# INTERNET LOSS containers stopped when internet is lost
# FALLBACK TIERS what this host runs for the remote per tier
# TIER DELAYS delays before each tier activates
# RSYNC WRITEBACK appdata synced back on handback
#
# ── MEDIA ──────────────────────────────────────────────────────────────────────────────────
# MEDIA PERMISSIONS share list for permissions script
# MEDIA CLEANER folder lists for media_cleaner.sh
#
# ── MONITORS ───────────────────────────────────────────────────────────────────────────────
# CERTIFICATE MONITOR domains checked for SSL expiry
# SMART HEALTH drives to skip in SMART monitoring
# ZFS REPORT pools to exclude from ZFS health report
#
# ── TRANSCODES ─────────────────────────────────────────────────────────────────────────────
# TRANSCODES ramdisk size, thresholds, SSD path, server array
#
# ── ARR STACK ──────────────────────────────────────────────────────────────────────────────
# DOWNLOADERS slskd, SABnzbd, qBittorrent credentials and URLs
# LIDARR / SONARR / RADARR URL, API key, path map
# ARR RECOVERY per-arr recovery toggles
#
# ── SYSTEM WATCHDOG ────────────────────────────────────────────────────────────────────────
# SYSTEM WATCHDOG per-host check toggles and NIC configuration
#
# ── RESOURCE MANAGER ───────────────────────────────────────────────────────────────────────
# RESOURCE MANAGER containers paused/stopped under memory pressure
#
# ==============================================================================================
# ==============================================================================================
# ── STORAGE MODE ──────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Storage mode ━━━
# Controls where Varaverk stores scripts, conf, and state files.
# true = internal NVMe/SSD — /boot/config/plugins/varaverk (write-safe, git-direct)
# false = USB flash boot — /mnt/user/appdata/Varaverk (preserves flash lifetime)
# Auto-detected from boot device transport on first setup.
# To change: Settings → Storage → Migrate.
HOSTN_STORAGE_MODE_INTERNAL=true
# ==============================================================================================
# ── IDENTITY & CONNECTIVITY ───────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Identity ━━━
# HOSTN hostname lives in master.conf (not a credential — safe for all servers).
# SSH key used for all server-to-server operations — rsync, failover, conf sync.
# Convention: /root/.ssh/<hostname-lowercase-no-unraid-prefix>_rsync_automation
# Must be in /root/.ssh/ and authorised in the partner's /root/.ssh/authorized_keys.
# Run Partnership/ssh_setup.sh to generate the key and copy it to the partner.
HOSTN_SSH_KEY="" # e.g. /root/.ssh/myserver_rsync_automation
HOSTN_OWNER="" # short identifier for this server (e.g. myserver)
HOSTN_OWNER_EMAIL=""
# ━━━ Unraid API ━━━
# Used by the Varaverk plugin to query this server's Unraid GraphQL API.
# Generate in Unraid: Settings → Management Access → API Keys → + New Key
HOSTN_UNRAID_API_KEY=""
# ━━━ Emby ━━━
HOSTN_EMBY_CONTAINER="Emby"
HOSTN_EMBY_URL="http://localhost:8096"
HOSTN_EMBY_API_KEY="" # Emby Dashboard → API Keys → + New Key
# ━━━ Jellyfin ━━━
HOSTN_JELLYFIN_CONTAINER="Jellyfin"
HOSTN_JELLYFIN_URL="http://localhost:8095"
HOSTN_JELLYFIN_API_KEY="" # Jellyfin Dashboard → Administration → API Keys
# ━━━ Gitea ━━━
# Personal access token for gitea_ssh_setup.sh.
# Create in Gitea: Settings → Applications → Generate Token → scope: write:user
HOSTN_GITEA_API_TOKEN=""
# ━━━ Notifications ━━━
# Discord webhook — leave blank to disable.
HOSTN_DISCORD_WEBHOOK=""
# ━━━ Partnership ━━━
# Auth containers reconfigured on onboard/offboard.
# Format: "ContainerName|WebUIPort"
HOSTN_PARTNERSHIP_AUTH_WEBUIS=(
# "NginxProxyManager|81"
# "Authelia|9091"
)
# XML templates pushed to mirror during onboard — auth stack.
# Dependencies (databases) must come before apps that depend on them.
HOSTN_PARTNERSHIP_AUTH_STACK=(
# "my-Authelia.xml"
# "my-NginxProxyManager.xml"
)
# XML templates pushed to mirror during onboard — arr stack.
HOSTN_PARTNERSHIP_ARR_STACK=(
# "my-Sonarr.xml"
# "my-Radarr.xml"
)
# Paths the partner should collect during the grace window after offboard.
HOSTN_PARTNERSHIP_MIRROR_BACKUPS=(
# "/mnt/user/appdata-Fallback/Partner-Emby"
)
# Containers parked on this server when partnership is active.
HOSTN_PARTNERSHIP_OWN_CONTAINERS=(
# "Emby"
)
# Emby admin provisioning — owner controls whether Emby is shared.
HOSTN_PARTNERSHIP_PROVISION_EMBY_ADMIN=false
HOSTN_PARTNERSHIP_EMBY_PORT=8096
HOSTN_PARTNERSHIP_EMBY_ADMIN_USER=""
HOSTN_PARTNERSHIP_EMBY_ADMIN_PASS=""
# ==============================================================================================
# ── RSYNC ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Daily Sync Shares ━━━
# Media shares this host pushes to all other nodes every night.
# Uses DEFAULT_RSYNC_OPTS from master.conf — no profile needed.
HOSTN_DAILY_SYNC_SHARES=(
# /mnt/user/Movies
# /mnt/user/Tv_Shows
# /mnt/user/Music
)
# ━━━ Weekly Sync Shares ━━━
# Appdata shares synced during the weekly maintenance window.
# Profiles (emby, critical-data) drive container stops — define in master.conf.
HOSTN_WEEKLY_SYNC_SHARES=(
# "/mnt/user/Media_Server/Emby" # emby profile
# "/mnt/user/appdata-Fallback/Critical-Data" # critical-data profile
)
# ━━━ Intermediate Sync Shares ━━━
# Shares synced every 4 hours. Leave empty to skip mid-day rsync.
HOSTN_INTERMEDIATE_SYNC_SHARES=(
# Add shares here to enable mid-day rsync
)
# ━━━ Critical Sync Shares ━━━
# Appdata shares synced every 30 minutes.
# Format: "/path/to/share" or "/path/to/share|profile-name"
HOSTN_CRITICAL_SYNC_SHARES=(
# "/mnt/user/appdata-Fallback/Critical-Data|critical-fallback"
# "/mnt/user/Media_Server/Emby|emby-fallback"
)
# ━━━ Personal Shares ━━━
# Private encrypted shares synced for offsite backup, independent of media shares.
HOSTN_PERSONAL_SHARES=(
# /mnt/user/Personal # e.g. ZFS-encrypted dataset
)
# ━━━ Backup Verify ━━━
# Leave empty to use HOSTN_DAILY_SYNC_SHARES automatically.
HOSTN_BACKUP_VERIFY_SHARES=(
# leave empty to use HOSTN_DAILY_SYNC_SHARES automatically
)
# ━━━ HOSTN Rsync Profile — hostn-appdata ━━━
# Host-specific appdata sync profile.
PROFILE_RSYNC_OPTS[hostn-appdata]="-av --info=progress2 --bwlimit=${PROFILE_BW_LIMIT[hostn-appdata]:-8000}"
PROFILE_BW_LIMIT[hostn-appdata]=8000
PROFILE_RETRY_COUNT[hostn-appdata]=3
PROFILE_SLEEP[hostn-appdata]=300
PROFILE_CRITICAL_CONTAINER_NAMES[hostn-appdata]=""
PROFILE_DELAYED_CONTAINERS[hostn-appdata]=""
PROFILE_CONTAINER_DELAY[hostn-appdata]=5
PROFILE_EXCLUDE_DIRS[hostn-appdata]="logs *.tmp"
# ==============================================================================================
# ── DOCKER ────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Docker Daily Restart ━━━
HOSTN_DAILY_RESTART_CONTAINERS=(
# "NginxProxyManager"
# "Authelia"
)
# ━━━ Docker Weekly Restart ━━━
HOSTN_WEEKLY_RESTART_CONTAINERS=(
# "NextCloud"
# "AdGuard-Home"
)
# ━━━ Docker Watchdog ━━━
# Memory hard limits in MB — immediate restart if exceeded.
# 20GB=20480 16GB=16384 8GB=8192 4GB=4096 2GB=2048 1GB=1024
declare -A HOSTN_WATCHDOG_CONTAINERS=(
# ["Emby"]=18432
)
# HTTP health check URLs — checked every cycle.
declare -A HOSTN_WATCHDOG_CONTAINER_URLS=(
# ["Emby"]="http://localhost:8096"
)
# Required containers — must always be running.
HOSTN_WATCHDOG_REQUIRED_CONTAINERS=(
# "NginxProxyManager"
# "Authelia"
)
# Containers to skip in Tier 2 global scan.
HOSTN_WATCHDOG_SCAN_IGNORE=(
# "my-occasional-container"
)
# Dependency ordering — skip restarting a container if its dependency is also down.
declare -A HOSTN_WATCHDOG_DEPENDENCIES=(
# ["Authelia"]="Mariadb Redis-Authelia"
)
# Per-container appdata growth suppress ceilings in MB.
declare -A HOSTN_WATCHDOG_APPDATA_SIZES=(
# ["Tdarr"]="25600"
)
# ━━━ Network Watchdog ━━━
HOSTN_NETWORK_WATCHDOG_DDNS_DOMAIN="" # e.g. myserver.com
HOSTN_NETWORK_WATCHDOG_DDNS_CONTAINER="" # e.g. MyServer.com
HOSTN_NETWORK_WATCHDOG_NPM_URL="" # e.g. https://myserver.com
# ━━━ Docker Network Connect ━━━
HOSTN_NETWORK_CONNECT_CONTAINERS=(
# "memcached"
)
HOSTN_NETWORK_CONNECT_NETWORKS=(
# "high-availability"
)
# ==============================================================================================
# ── FALLBACK ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ DDNS ━━━
# DDNS containers this host manages.
HOSTN_DDNS_CONTAINERS=(
# "MyServer.com"
)
# ━━━ Internet Loss ━━━
# Containers stopped immediately when internet is lost.
FALLBACK_HOSTN_STOP_ON_NO_NET=(
# "MyServer.com"
)
# ━━━ Fallback Tiers — HOSTN Runs for Partner ━━━
# Containers this host starts when the partner goes down.
# Replace REMOTE_ID below with the actual remote host ID (HOST1, HOST2, etc.)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER1=(
# "Partner-DDNS-Container"
)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER2=(
# "container-placeholder"
)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER3=(
# "container-placeholder"
)
FALLBACK_HOSTN_COVERS_REMOTE_ID_TIER4=(
# "container-placeholder"
)
# ━━━ Tier Delays — Partner's Containers on this Host ━━━
# How long the partner must be down before each tier activates here — in minutes.
# Replace REMOTE_ID with the actual remote host ID (HOST1, HOST2, etc.)
REMOTE_ID_TIER2_DELAY=240 # 4 hours
REMOTE_ID_TIER3_DELAY=720 # 12 hours
REMOTE_ID_TIER4_DELAY=1440 # 24 hours
# ━━━ Rsync Writeback ━━━
HOSTN_TIER1_WRITEBACK_DELAY=60 # skip Emby writeback if outage under 1hr
FALLBACK_HOSTN_WRITEBACK_TIER1=(
# "/mnt/user/Media_Server/Emby"
)
FALLBACK_HOSTN_WRITEBACK_TIER2=(
# "/mnt/user/appdata-Fallback/Important-Data"
)
FALLBACK_HOSTN_WRITEBACK_TIER3=(
# "location-placeholder"
)
FALLBACK_HOSTN_WRITEBACK_TIER4=(
# "/mnt/user/appdata-Fallback/Arrs_Stack"
)
# ==============================================================================================
# ── MEDIA ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Media Permissions ━━━
HOSTN_MEDIA_PERMISSION_SHARES=(
# /mnt/user/Movies
# /mnt/user/Tv_Shows
# /mnt/user/Music
# /mnt/user/Downloads
)
# ━━━ Media Cleaner ━━━
HOSTN_ANIME_CLEAN_FOLDERS=(
# /mnt/user/Anime_Movies
# /mnt/user/Anime_Shows
)
HOSTN_MEDIA_CLEAN_FOLDERS=(
# /mnt/user/Movies
# /mnt/user/Tv_Shows
)
# ==============================================================================================
# ── MONITORS ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
HOSTN_CERT_MONITOR_DOMAINS=(
# "myserver.com"
)
# ━━━ SMART Health ━━━
HOSTN_SMART_IGNORE_DRIVES=(
"sda" # boot USB — SMART not meaningful on flash drives
)
# ━━━ ZFS Report ━━━
HOSTN_ZFS_REPORT_IGNORE_POOLS=(
# "disk5"
)
# ==============================================================================================
# ── TRANSCODES ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOSTN_RAMDISK_SIZE="10G"
HOSTN_RAMDISK_WARN_GB=8.5
HOSTN_RAMDISK_LOW_GB=7
HOSTN_TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/"
HOSTN_TRANSCODE_SERVERS=(
"${HOSTN_EMBY_CONTAINER}|${HOSTN_EMBY_URL}|${HOSTN_EMBY_API_KEY}|emby"
)
# ==============================================================================================
# ── ARR STACK ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Downloaders ━━━
HOSTN_SLSKD_URL="http://localhost:8980"
HOSTN_SLSKD_API_KEY=""
HOSTN_SLSKD_FAILED_IMPORTS_DIR=""
HOSTN_SABNZBD_URL="http://localhost:8180"
HOSTN_SABNZBD_API_KEY=""
HOSTN_QBIT_URL="http://localhost:8080"
HOSTN_QBIT_USERNAME="admin"
HOSTN_QBIT_PASSWORD=""
# ━━━ Lidarr ━━━
HOSTN_LIDARR_URL="http://localhost:8686"
HOSTN_LIDARR_API_KEY=""
HOSTN_LIDARR_MUSIC_ROOT="/mnt/user/Music"
HOSTN_FANART_API_KEY=""
HOSTN_LASTFM_API_KEY=""
declare -A HOSTN_LIDARR_PATH_MAP=(
# ["/music"]="/mnt/user/Music"
)
# ━━━ Sonarr ━━━
HOSTN_SONARR_URL="http://localhost:8989"
HOSTN_SONARR_API_KEY=""
HOSTN_SONARR_TV_ROOT="/mnt/user/Tv_Shows"
declare -A HOSTN_SONARR_PATH_MAP=(
# ["/tv"]="/mnt/user/Tv_Shows"
)
# ━━━ Radarr ━━━
HOSTN_RADARR_URL="http://localhost:7878"
HOSTN_RADARR_API_KEY=""
HOSTN_TMDB_API_KEY=""
HOSTN_RADARR_MOVIE_ROOT="/mnt/user/Movies"
declare -A HOSTN_RADARR_PATH_MAP=(
# ["/movies"]="/mnt/user/Movies"
)
# ━━━ Arr Recovery Toggles ━━━
HOSTN_LIDARR_RECOVERY=false
HOSTN_SONARR_RECOVERY=true
HOSTN_RADARR_RECOVERY=true
# ==============================================================================================
# ── SYSTEM WATCHDOG ───────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOSTN_SYS_WATCHDOG_CHECK_DOCKER=true
HOSTN_SYS_WATCHDOG_CHECK_ROOTFS=true
HOSTN_SYS_WATCHDOG_CHECK_KERNEL_OOPS=true
HOSTN_SYS_WATCHDOG_CHECK_FD=true
HOSTN_SYS_WATCHDOG_CHECK_BOOT=true
HOSTN_SYS_WATCHDOG_CHECK_OOM=true
HOSTN_SYS_WATCHDOG_CHECK_RAM=true
HOSTN_SYS_WATCHDOG_CHECK_LOG=true
HOSTN_SYS_WATCHDOG_CHECK_ARC=true
HOSTN_SYS_WATCHDOG_CHECK_TEMP=true
HOSTN_SYS_WATCHDOG_CHECK_LOAD=true
HOSTN_SYS_WATCHDOG_CHECK_ZOMBIES=true
HOSTN_SYS_WATCHDOG_CHECK_CONTAINERS=true
HOSTN_SYS_WATCHDOG_CHECK_TMP=true
HOSTN_SYS_WATCHDOG_CHECK_MDSTAT=true
HOSTN_SYS_WATCHDOG_CHECK_NETWORK=true
HOSTN_SYS_WATCHDOG_CHECK_SSHD=true
HOSTN_SYS_WATCHDOG_CHECK_RUNAWAY=false
HOSTN_SYS_WATCHDOG_NIC="" # e.g. eth0 — for network monitoring
# ==============================================================================================
# ── RESOURCE MANAGER ──────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOSTN_RW_PAUSE_CONTAINERS=(
# "Tdarr"
# "LidaTube"
)
HOSTN_RW_STOP_CONTAINERS=(
# "Tdarr"
)
@@ -1,239 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= conf_upgrade.sh ================================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Merges a new conf template into an existing user conf while preserving every
# value the user has already set. Run manually when the conf schema changes
# between versions — adds new keys, removes deprecated ones, and keeps the
# structure of the new template exactly.
#
# Keys in template only → ADDED (placeholder/default — user fills in once)
# Keys in target only → REMOVED (deprecated in new version)
# Keys in both → KEPT (target's value always wins, template ignored)
# Comments / blank lines → always from template (structure follows new version)
#
# Supports all conf variable patterns: simple scalars, indexed arrays, and
# associative arrays (declare -A).
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Dry-run Mode
# --dry-run prints the full change report (ADDED / REMOVED / KEPT) then exits
# without writing anything. Always preview before applying to production confs.
#
# Backup Option
# --backup writes a .bak copy of the target before overwriting. Use when
# applying to a conf that has never been upgraded before.
#
# File Existence Guards
# Both --template and --target are validated before any parsing begins.
# Missing files abort immediately with a clear error.
#
# Atomic Write
# Merged output is written to a tempfile first, then copied to the target.
# A partial write cannot corrupt the original.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# No conf vars. All inputs are CLI flags.
#
# --template <file> New version conf file (source of structure and defaults)
# --target <file> Existing user conf (source of real values — always preserved)
# --dry-run Show what would change without writing
# --backup Write a .bak copy of target before modifying
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# conf_upgrade.sh --template master.conf.template --target Configurations/master.conf --dry-run
# Preview what would be added, removed, and kept — no changes written.
#
# conf_upgrade.sh --template master.conf.template --target Configurations/master.conf --backup
# Apply the upgrade, writing a .bak first.
#
# conf_upgrade.sh --template master.conf.template --target Configurations/master.conf
# Apply the upgrade in-place with no backup.
#
# ==============================================================================================
set -uo pipefail
# ── Arguments ────────────────────────────────────────────────────────────────────────────────
TEMPLATE=""
TARGET=""
DRY_RUN=false
BACKUP=false
while [[ $# -gt 0 ]]; do
case "$1" in
--template) TEMPLATE="$2"; shift 2 ;;
--target) TARGET="$2"; shift 2 ;;
--dry-run) DRY_RUN=true; shift ;;
--backup) BACKUP=true; shift ;;
*) echo "Unknown option: $1" >&2; exit 1 ;;
esac
done
[[ -z "$TEMPLATE" ]] && { echo "Error: --template required" >&2; exit 1; }
[[ -z "$TARGET" ]] && { echo "Error: --target required" >&2; exit 1; }
[[ -f "$TEMPLATE" ]] || { echo "Error: template not found: $TEMPLATE" >&2; exit 1; }
[[ -f "$TARGET" ]] || { echo "Error: target not found: $TARGET" >&2; exit 1; }
# ── Parse target → KEY → full definition block ───────────────────────────────────────────────
declare -A HOST_MAP # KEY → complete definition line(s) from user's conf
_parse_target() {
local in_block=false cur_key="" cur_block="" line
while IFS= read -r line || [[ -n "$line" ]]; do
if [[ "$in_block" == true ]]; then
cur_block+="$line"$'\n'
# Closing ) — optional trailing whitespace and comment
if [[ "$line" =~ ^[[:space:]]*\)[[:space:]]*(#.*)?$ ]]; then
HOST_MAP["$cur_key"]="$cur_block"
in_block=false; cur_key=""; cur_block=""
fi
else
# declare -A KEY=(
if [[ "$line" =~ ^[[:space:]]*declare[[:space:]]+-[a-zA-Z]+[[:space:]]+([A-Z0-9_]+)[[:space:]]*=\( ]]; then
cur_key="${BASH_REMATCH[1]}"; in_block=true; cur_block="$line"$'\n'
# KEY=(
elif [[ "$line" =~ ^[[:space:]]*([A-Z0-9_]+)[[:space:]]*=\( ]]; then
cur_key="${BASH_REMATCH[1]}"; in_block=true; cur_block="$line"$'\n'
# KEY=value (simple scalar)
elif [[ "$line" =~ ^[[:space:]]*([A-Z0-9_]+)[[:space:]]*= ]]; then
HOST_MAP["${BASH_REMATCH[1]}"]="$line"$'\n'
fi
fi
done < "$TARGET"
}
# ── Walk template — collect stats (must run in current shell so arrays persist) ──────────────
declare -a ADDED=() KEPT=() REMOVED=()
declare -A TMPL_SEEN=()
_collect_stats() {
local in_block=false cur_key="" line
while IFS= read -r line || [[ -n "$line" ]]; do
if [[ "$in_block" == true ]]; then
if [[ "$line" =~ ^[[:space:]]*\)[[:space:]]*(#.*)?$ ]]; then
in_block=false
TMPL_SEEN["$cur_key"]=1
if [[ -n "${HOST_MAP[$cur_key]+_}" ]]; then KEPT+=("$cur_key")
else ADDED+=("$cur_key"); fi
cur_key=""
fi
else
if [[ "$line" =~ ^[[:space:]]*declare[[:space:]]+-[a-zA-Z]+[[:space:]]+([A-Z0-9_]+)[[:space:]]*=\( ]]; then
cur_key="${BASH_REMATCH[1]}"; in_block=true
elif [[ "$line" =~ ^[[:space:]]*([A-Z0-9_]+)[[:space:]]*=\( ]]; then
cur_key="${BASH_REMATCH[1]}"; in_block=true
elif [[ "$line" =~ ^[[:space:]]*([A-Z0-9_]+)[[:space:]]*= ]]; then
local k="${BASH_REMATCH[1]}"
TMPL_SEEN["$k"]=1
if [[ -n "${HOST_MAP[$k]+_}" ]]; then KEPT+=("$k")
else ADDED+=("$k"); fi
fi
fi
done < "$TEMPLATE"
for key in "${!HOST_MAP[@]}"; do
[[ -z "${TMPL_SEEN[$key]+_}" ]] && REMOVED+=("$key")
done
}
# ── Walk template — write merged output ──────────────────────────────────────────────────────
# Runs in a subshell (stdout redirected) — array mutations are intentionally discarded here.
_write_merged() {
local in_block=false cur_key="" cur_block="" line
while IFS= read -r line || [[ -n "$line" ]]; do
if [[ "$in_block" == true ]]; then
cur_block+="$line"$'\n'
if [[ "$line" =~ ^[[:space:]]*\)[[:space:]]*(#.*)?$ ]]; then
in_block=false
if [[ -n "${HOST_MAP[$cur_key]+_}" ]]; then printf '%s' "${HOST_MAP[$cur_key]}"
else printf '%s' "$cur_block"; fi
cur_key=""; cur_block=""
fi
else
if [[ "$line" =~ ^[[:space:]]*declare[[:space:]]+-[a-zA-Z]+[[:space:]]+([A-Z0-9_]+)[[:space:]]*=\( ]]; then
cur_key="${BASH_REMATCH[1]}"; in_block=true; cur_block="$line"$'\n'
elif [[ "$line" =~ ^[[:space:]]*([A-Z0-9_]+)[[:space:]]*=\( ]]; then
cur_key="${BASH_REMATCH[1]}"; in_block=true; cur_block="$line"$'\n'
elif [[ "$line" =~ ^[[:space:]]*([A-Z0-9_]+)[[:space:]]*= ]]; then
local k="${BASH_REMATCH[1]}"
if [[ -n "${HOST_MAP[$k]+_}" ]]; then printf '%s' "${HOST_MAP[$k]}"
else printf '%s\n' "$line"; fi
else
printf '%s\n' "$line"
fi
fi
done < "$TEMPLATE"
}
# ── Run ───────────────────────────────────────────────────────────────────────────────────────
_parse_target
_collect_stats
# ── Report ────────────────────────────────────────────────────────────────────────────────────
TARGET_NAME="$(basename "$TARGET")"
echo ""
echo "── conf_upgrade: $TARGET_NAME ──────────────────────────────────────────"
if [[ ${#ADDED[@]} -gt 0 ]]; then
echo " ADDED (new — fill in your values where needed):"
for k in "${ADDED[@]}"; do echo " + $k"; done
fi
if [[ ${#REMOVED[@]} -gt 0 ]]; then
echo " REMOVED (deprecated — no longer in this version):"
for k in "${REMOVED[@]}"; do echo " - $k"; done
fi
echo " KEPT ${#KEPT[@]} existing vars — your values preserved"
if [[ ${#ADDED[@]} -eq 0 && ${#REMOVED[@]} -eq 0 ]]; then
echo " Already up to date — no changes needed."
echo "────────────────────────────────────────────────────────────────────────"
echo ""
exit 0
fi
echo "────────────────────────────────────────────────────────────────────────"
echo ""
# ── Apply ─────────────────────────────────────────────────────────────────────────────────────
if [[ "$DRY_RUN" == true ]]; then
echo "(dry-run — no changes written)"
exit 0
fi
TMPOUT="$(mktemp)"
trap 'rm -f "$TMPOUT"' EXIT
_write_merged > "$TMPOUT"
if [[ "$BACKUP" == true ]]; then
cp "$TARGET" "${TARGET}.bak"
echo "Backup: ${TARGET}.bak"
fi
cp "$TMPOUT" "$TARGET"
echo "Updated: $TARGET"
@@ -1,239 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= conf_upgrade.sh ================================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Merges a new conf template into an existing user conf while preserving every
# value the user has already set. Run manually when the conf schema changes
# between versions — adds new keys, removes deprecated ones, and keeps the
# structure of the new template exactly.
#
# Keys in template only → ADDED (placeholder/default — user fills in once)
# Keys in target only → REMOVED (deprecated in new version)
# Keys in both → KEPT (target's value always wins, template ignored)
# Comments / blank lines → always from template (structure follows new version)
#
# Supports all conf variable patterns: simple scalars, indexed arrays, and
# associative arrays (declare -A).
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Dry-run Mode
# --dry-run prints the full change report (ADDED / REMOVED / KEPT) then exits
# without writing anything. Always preview before applying to production confs.
#
# Backup Option
# --backup writes a .bak copy of the target before overwriting. Use when
# applying to a conf that has never been upgraded before.
#
# File Existence Guards
# Both --template and --target are validated before any parsing begins.
# Missing files abort immediately with a clear error.
#
# Atomic Write
# Merged output is written to a tempfile first, then copied to the target.
# A partial write cannot corrupt the original.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# No conf vars. All inputs are CLI flags.
#
# --template <file> New version conf file (source of structure and defaults)
# --target <file> Existing user conf (source of real values — always preserved)
# --dry-run Show what would change without writing
# --backup Write a .bak copy of target before modifying
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# conf_upgrade.sh --template Configurations/master.conf.template --target Configurations/master.conf --dry-run
# Preview what would be added, removed, and kept — no changes written.
#
# conf_upgrade.sh --template Configurations/master.conf.template --target Configurations/master.conf --backup
# Apply the upgrade, writing a .bak first.
#
# conf_upgrade.sh --template Configurations/master.conf.template --target Configurations/master.conf
# Apply the upgrade in-place with no backup.
#
# ==============================================================================================
set -uo pipefail
# ── Arguments ────────────────────────────────────────────────────────────────────────────────
TEMPLATE=""
TARGET=""
DRY_RUN=false
BACKUP=false
while [[ $# -gt 0 ]]; do
case "$1" in
--template) TEMPLATE="$2"; shift 2 ;;
--target) TARGET="$2"; shift 2 ;;
--dry-run) DRY_RUN=true; shift ;;
--backup) BACKUP=true; shift ;;
*) echo "Unknown option: $1" >&2; exit 1 ;;
esac
done
[[ -z "$TEMPLATE" ]] && { echo "Error: --template required" >&2; exit 1; }
[[ -z "$TARGET" ]] && { echo "Error: --target required" >&2; exit 1; }
[[ -f "$TEMPLATE" ]] || { echo "Error: template not found: $TEMPLATE" >&2; exit 1; }
[[ -f "$TARGET" ]] || { echo "Error: target not found: $TARGET" >&2; exit 1; }
# ── Parse target → KEY → full definition block ───────────────────────────────────────────────
declare -A HOST_MAP # KEY → complete definition line(s) from user's conf
_parse_target() {
local in_block=false cur_key="" cur_block="" line
while IFS= read -r line || [[ -n "$line" ]]; do
if [[ "$in_block" == true ]]; then
cur_block+="$line"$'\n'
# Closing ) — optional trailing whitespace and comment
if [[ "$line" =~ ^[[:space:]]*\)[[:space:]]*(#.*)?$ ]]; then
HOST_MAP["$cur_key"]="$cur_block"
in_block=false; cur_key=""; cur_block=""
fi
else
# declare -A KEY=(
if [[ "$line" =~ ^[[:space:]]*declare[[:space:]]+-[a-zA-Z]+[[:space:]]+([A-Z0-9_]+)[[:space:]]*=\( ]]; then
cur_key="${BASH_REMATCH[1]}"; in_block=true; cur_block="$line"$'\n'
# KEY=(
elif [[ "$line" =~ ^[[:space:]]*([A-Z0-9_]+)[[:space:]]*=\( ]]; then
cur_key="${BASH_REMATCH[1]}"; in_block=true; cur_block="$line"$'\n'
# KEY=value (simple scalar)
elif [[ "$line" =~ ^[[:space:]]*([A-Z0-9_]+)[[:space:]]*= ]]; then
HOST_MAP["${BASH_REMATCH[1]}"]="$line"$'\n'
fi
fi
done < "$TARGET"
}
# ── Walk template — collect stats (must run in current shell so arrays persist) ──────────────
declare -a ADDED=() KEPT=() REMOVED=()
declare -A TMPL_SEEN=()
_collect_stats() {
local in_block=false cur_key="" line
while IFS= read -r line || [[ -n "$line" ]]; do
if [[ "$in_block" == true ]]; then
if [[ "$line" =~ ^[[:space:]]*\)[[:space:]]*(#.*)?$ ]]; then
in_block=false
TMPL_SEEN["$cur_key"]=1
if [[ -n "${HOST_MAP[$cur_key]+_}" ]]; then KEPT+=("$cur_key")
else ADDED+=("$cur_key"); fi
cur_key=""
fi
else
if [[ "$line" =~ ^[[:space:]]*declare[[:space:]]+-[a-zA-Z]+[[:space:]]+([A-Z0-9_]+)[[:space:]]*=\( ]]; then
cur_key="${BASH_REMATCH[1]}"; in_block=true
elif [[ "$line" =~ ^[[:space:]]*([A-Z0-9_]+)[[:space:]]*=\( ]]; then
cur_key="${BASH_REMATCH[1]}"; in_block=true
elif [[ "$line" =~ ^[[:space:]]*([A-Z0-9_]+)[[:space:]]*= ]]; then
local k="${BASH_REMATCH[1]}"
TMPL_SEEN["$k"]=1
if [[ -n "${HOST_MAP[$k]+_}" ]]; then KEPT+=("$k")
else ADDED+=("$k"); fi
fi
fi
done < "$TEMPLATE"
for key in "${!HOST_MAP[@]}"; do
[[ -z "${TMPL_SEEN[$key]+_}" ]] && REMOVED+=("$key")
done
}
# ── Walk template — write merged output ──────────────────────────────────────────────────────
# Runs in a subshell (stdout redirected) — array mutations are intentionally discarded here.
_write_merged() {
local in_block=false cur_key="" cur_block="" line
while IFS= read -r line || [[ -n "$line" ]]; do
if [[ "$in_block" == true ]]; then
cur_block+="$line"$'\n'
if [[ "$line" =~ ^[[:space:]]*\)[[:space:]]*(#.*)?$ ]]; then
in_block=false
if [[ -n "${HOST_MAP[$cur_key]+_}" ]]; then printf '%s' "${HOST_MAP[$cur_key]}"
else printf '%s' "$cur_block"; fi
cur_key=""; cur_block=""
fi
else
if [[ "$line" =~ ^[[:space:]]*declare[[:space:]]+-[a-zA-Z]+[[:space:]]+([A-Z0-9_]+)[[:space:]]*=\( ]]; then
cur_key="${BASH_REMATCH[1]}"; in_block=true; cur_block="$line"$'\n'
elif [[ "$line" =~ ^[[:space:]]*([A-Z0-9_]+)[[:space:]]*=\( ]]; then
cur_key="${BASH_REMATCH[1]}"; in_block=true; cur_block="$line"$'\n'
elif [[ "$line" =~ ^[[:space:]]*([A-Z0-9_]+)[[:space:]]*= ]]; then
local k="${BASH_REMATCH[1]}"
if [[ -n "${HOST_MAP[$k]+_}" ]]; then printf '%s' "${HOST_MAP[$k]}"
else printf '%s\n' "$line"; fi
else
printf '%s\n' "$line"
fi
fi
done < "$TEMPLATE"
}
# ── Run ───────────────────────────────────────────────────────────────────────────────────────
_parse_target
_collect_stats
# ── Report ────────────────────────────────────────────────────────────────────────────────────
TARGET_NAME="$(basename "$TARGET")"
echo ""
echo "── conf_upgrade: $TARGET_NAME ──────────────────────────────────────────"
if [[ ${#ADDED[@]} -gt 0 ]]; then
echo " ADDED (new — fill in your values where needed):"
for k in "${ADDED[@]}"; do echo " + $k"; done
fi
if [[ ${#REMOVED[@]} -gt 0 ]]; then
echo " REMOVED (deprecated — no longer in this version):"
for k in "${REMOVED[@]}"; do echo " - $k"; done
fi
echo " KEPT ${#KEPT[@]} existing vars — your values preserved"
if [[ ${#ADDED[@]} -eq 0 && ${#REMOVED[@]} -eq 0 ]]; then
echo " Already up to date — no changes needed."
echo "────────────────────────────────────────────────────────────────────────"
echo ""
exit 0
fi
echo "────────────────────────────────────────────────────────────────────────"
echo ""
# ── Apply ─────────────────────────────────────────────────────────────────────────────────────
if [[ "$DRY_RUN" == true ]]; then
echo "(dry-run — no changes written)"
exit 0
fi
TMPOUT="$(mktemp)"
trap 'rm -f "$TMPOUT"' EXIT
_write_merged > "$TMPOUT"
if [[ "$BACKUP" == true ]]; then
cp "$TARGET" "${TARGET}.bak"
echo "Backup: ${TARGET}.bak"
fi
cp "$TMPOUT" "$TARGET"
echo "Updated: $TARGET"
@@ -1,775 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ========================== HOST1 CONFIGURATION — unRAID-Gmer4Lfe ============================
# ==============================================================================================
# HOST1-specific variables — credentials, container names, share paths, failover lists.
# Sourced after master.conf — values here extend shared profile arrays and add HOST1-specific
# identity, credentials, and container configuration.
#
# Sparse checkout (git) ensures HOST2 never receives this file.
# HOST2 never sees HOST1 credentials — clean separation at the file level.
#
# DO NOT put shared config here — thresholds, toggles, profiles belong in master.conf.
# DO NOT put HOST2 variables here — they belong in host2.conf.
#
# ── INDEX ─────────────────────────────────────────────────────────────────────────────────────
#
# ── IDENTITY & CONNECTIVITY ────────────────────────────────────────────────────────────────
# IDENTITY hostname, SSH key
# EMBY container name, URL, API key
# NOTIFICATIONS Discord webhook
# PARTNERSHIP auth containers, backup paths
#
# ── RSYNC ──────────────────────────────────────────────────────────────────────────────────
# DAILY SYNC SHARES media shares HOST1 owns and pushes to HOST2
# WEEKLY SYNC SHARES appdata shares synced weekly (Sunday 2:30am)
# CRITICAL SYNC SHARES appdata shares synced every 30 minutes
# BACKUP VERIFY shares for checksum verification against remote
# HOST1 RSYNC PROFILE host1-appdata profile for HOST1-specific appdata syncs
#
# ── DOCKER ─────────────────────────────────────────────────────────────────────────────────
# DOCKER DAILY RESTART containers restarted daily
# DOCKER WEEKLY RESTART containers restarted weekly
# DOCKER WATCHDOG memory limits, health URLs, required containers, ignore list
# DOCKER NETWORK CONNECT networks and containers for docker_network_connect.sh
#
# ── FALLBACK ───────────────────────────────────────────────────────────────────────────────
# DDNS DDNS containers managed by HOST1
# INTERNET LOSS containers stopped when internet is lost
# FALLBACK TIERS what HOST1 runs for HOST2 per tier
# TIER DELAYS how long HOST1 must be down before each tier activates on HOST2
# RSYNC WRITEBACK HOST1 appdata synced back on handback
#
# ── MEDIA ──────────────────────────────────────────────────────────────────────────────────
# MEDIA PERMISSIONS share list for media_shares_permissions.sh
# MEDIA CLEANER folder lists for media_cleaner.sh
#
# ── MONITORS ───────────────────────────────────────────────────────────────────────────────
# CERTIFICATE MONITOR domains checked for SSL expiry
# SMART HEALTH drives to skip in SMART monitoring
# ZFS REPORT pools to exclude from ZFS health report
#
# ── TRANSCODES ─────────────────────────────────────────────────────────────────────────────
# TRANSCODES ramdisk size, thresholds, SSD path, server array
#
# ── ARR STACK ──────────────────────────────────────────────────────────────────────────────
# DOWNLOADERS slskd, SABnzbd, qBittorrent credentials and URLs
# LIDARR URL, API key, path map
# SONARR URL, API key, path map
# RADARR URL, API key, path map
# ARR RECOVERY per-arr recovery toggles
#
# ── SYSTEM WATCHDOG ────────────────────────────────────────────────────────────────────────
# SYSTEM WATCHDOG per-host check toggles and NIC configuration
#
# ── RESOURCE MANAGER ───────────────────────────────────────────────────────────────────────
# RESOURCE MANAGER containers paused/stopped under memory pressure
#
# ==============================================================================================
# ==============================================================================================
# ── IDENTITY & CONNECTIVITY ───────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Identity ━━━
# HOST1 hostname lives in master.conf (not a credential — safe for all servers).
# SSH key used for all server-to-server operations — rsync, failover container commands.
# Must be in /root/.ssh/ and authorised in HOST2's /root/.ssh/authorized_keys.
HOST1_SSH_KEY="/root/.ssh/gmer4lfe_rsync_automation"
HOST1_OWNER="gmer4lfe"
HOST1_OWNER_EMAIL="gmer4lfe@gmail.com"
# ━━━ Emby ━━━
# Referenced by transcode_manager.sh, emby_session_report.sh, emby_database_repair.sh,
# weekly_sync_maintenance.sh, and HOST1_TRANSCODE_SERVERS below.
# API key: Emby Dashboard → API Keys → + New Key
HOST1_EMBY_CONTAINER="Emby"
HOST1_EMBY_URL="http://localhost:8096"
HOST1_EMBY_API_KEY="0c27448d93a7431f9ac63569f7655829"
# ━━━ Jellyfin ━━━
# API key: Jellyfin Dashboard → Administration → API Keys → + New Key
HOST1_JELLYFIN_CONTAINER="Jellyfin"
HOST1_JELLYFIN_URL="http://localhost:8095"
HOST1_JELLYFIN_API_KEY="4e820e7df74c4933acec212b1996314e"
# ━━━ Gitea ━━━
# Personal access token for gitea_ssh_setup.sh — registers this server's SSH public key
# with Gitea so git operations use key auth instead of passwords.
# Create in Gitea: Settings → Applications → Generate Token → scope: write:user
HOST1_GITEA_API_TOKEN=""
# ━━━ Notifications ━━━
# Discord webhook — leave blank to disable.
# Per-host so HOST1 and HOST2 can post to different channels or only one server notifies.
HOST1_DISCORD_WEBHOOK=""
# ━━━ Partnership ━━━
# HOST1 is always the owner (source of truth) unless --transfer has been run.
# See README-Partnership.md and master.conf PARTNERSHIP section for full lifecycle docs.
# Auth containers reconfigured on onboard/offboard.
# Format: "ContainerName|WebUIPort"
# On onboard → WebUI pointed at owner's Tailscale IP (mirror clicks NPM, gets owner's auth)
# On offboard → WebUI pointed back at localhost
HOST1_PARTNERSHIP_AUTH_WEBUIS=(
"NginxProxyManager|81"
"Lldap-Gmer4Lfe|17170"
"Authelia|9091"
"Authelia-Secondary|9092"
)
# XML templates (from this server's templates-user/) pushed to mirror during onboard.
# These become the mirror's active auth stack, backed by the rsync-synced appdata.
# Update filename if Lldap is renamed to drop the host suffix.
HOST1_PARTNERSHIP_AUTH_STACK=(
# Dependencies first — Mariadb/Redis must be healthy before Authelia starts
"my-Mariadb-Authelia.xml"
"my-Mariadb-Authelia-Secondary.xml"
"my-Redis-Authelia.xml"
"my-Redis-Authelia-Secondary.xml"
# Auth apps — deployed after their deps are confirmed healthy
"my-Authelia.xml"
"my-Authelia-Secondary.xml"
"my-NginxProxyManager.xml"
"my-Lldap-Gmer4Lfe.xml"
# Source of truth — must be available on HOST2 independently of the auth stack
"my-Gitea.xml"
)
# XML templates pushed to mirror for the arr stack during onboard.
# Deps (e.g. databases) first if any — same ordering rule as auth stack.
HOST1_PARTNERSHIP_ARR_STACK=(
# "my-Sonarr.xml"
# "my-Radarr.xml"
# "my-Lidarr.xml"
# "my-Prowlarr.xml"
# "my-Bazarr.xml"
)
# Paths HOST2 should collect during the grace window after offboard.
# Notified on offboard — no auto-deletion, HOST2 must collect manually within PARTNERSHIP_GRACE_HOURS.
HOST1_PARTNERSHIP_MIRROR_BACKUPS=(
# "/mnt/user/appdata-Fallback/Jayred365-Emby"
)
# Containers parked on this server when partnership is active.
# Stopped on onboard (owner deploys its stack instead), restarted on offboard.
HOST1_PARTNERSHIP_OWN_CONTAINERS=(
# "Emby"
# "NginxProxyManager"
)
# Emby admin provisioning — toggle is owner-only, credentials are per-host.
# Owner enables/disables the feature. Each host sets the account they want on the shared Emby.
# On onboard: owner reads mirror's HOST*_PARTNERSHIP_EMBY_ADMIN_* and creates that account.
# On offboard: account is deleted. Username collision → onboard exits with error.
HOST1_PARTNERSHIP_PROVISION_EMBY_ADMIN=false # owner controls whether Emby is shared
HOST1_PARTNERSHIP_EMBY_PORT=8096
HOST1_PARTNERSHIP_EMBY_ADMIN_USER="" # this server's desired Emby username
HOST1_PARTNERSHIP_EMBY_ADMIN_PASS="" # this server's desired Emby password
# ==============================================================================================
# ── RSYNC ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Daily Sync Shares ━━━
# Shares HOST1 pushes to all other nodes every night (1am via daily_sync_maintenance.sh).
# Mesh model: every node pushes every media share — no ownership, no mirrors.
# arr_sync ensures all arr libraries converge (union). rsync spreads files (additive, no --delete).
# arr_cleanup removes true orphans based on local arr state.
# Any node can download content to any share — it propagates to all nodes on the next cycle.
# Nextcloud is intentionally one-directional (HOST1→HOST2 offsite backup — not arr-managed).
# Uses DEFAULT_RSYNC_OPTS from master.conf — no profile needed.
# For shares needing container stops or custom options — add a profile in master.conf.
HOST1_DAILY_SYNC_SHARES=(
/mnt/user/Books
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Nextcloud
/mnt/user/stand-up_comedy
/mnt/user/Sports
# /mnt/user/Tv_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Movies
/mnt/user/Anime_Shows
)
# Personal encrypted shares — synced for offsite backup, independent of media shares.
# ZFS encrypted at dataset level — remote receives encrypted blocks, cannot read content.
# See README-Rsync_Setup.md for ZFS encryption setup before uncommenting.
HOST1_PERSONAL_SHARES=(
# /mnt/user/HOST1-Personal # uncomment after creating encrypted dataset
)
# ━━━ Weekly Sync Shares ━━━
# Appdata shares synced during the weekly maintenance window (Sunday 2:30am).
# Containers stopped both sides before sync — full clean state guaranteed.
# Profiles drive container stops, excludes, and options — configured in master.conf RSYNC section.
# Order matters — Emby first (larger transfer), then Critical-Data (auth stack).
HOST1_WEEKLY_SYNC_SHARES=(
"/mnt/user/Media_Server/Emby" # emby profile — full clean mirror
"/mnt/user/appdata-Fallback/Critical-Data" # critical-data profile — auth stack
)
# ━━━ Intermediate Sync Shares ━━━
# Shares synced every 4 hours by intermediate_sync_maintenance.sh.
# Uses DEFAULT_RSYNC_OPTS (no --delete) — for sub-daily propagation of metadata or watch state.
# Full media share sync stays in the daily window. Leave empty to skip mid-day rsync entirely.
HOST1_INTERMEDIATE_SYNC_SHARES=(
# Add shares here to enable mid-day rsync
# Example: "/mnt/user/Emby_Metadata"
)
# ━━━ Critical Sync Shares ━━━
# Appdata shares synced every 30 minutes by critical_sync_maintenance.sh.
# Format: "/path/to/share" or "/path/to/share|profile-name"
# Order matters — Critical-Data first (auth stack), then Emby dirty sync.
HOST1_CRITICAL_SYNC_SHARES=(
"/mnt/user/appdata-Fallback/Critical-Data|critical-fallback" # auth dirty sync — stays running
"/mnt/user/Media_Server/Emby|emby-fallback" # Emby dirty sync — stays running
)
# ━━━ Backup Verify ━━━
# Shares verified by backup_verify.sh — random file checksum comparison against remote.
# Leave empty to use HOST1_DAILY_SYNC_SHARES automatically.
# Sample size and minimum file size defined in master.conf.
HOST1_BACKUP_VERIFY_SHARES=(
# leave empty to use HOST1_DAILY_SYNC_SHARES automatically
)
# ━━━ HOST1 Rsync Profile — host1-appdata ━━━
# HOST1-specific appdata sync profile — extends the shared PROFILE_* arrays in master.conf.
# Use for appdata unique to HOST1 (Organizrv2, VaultWarden, UptimeKuma etc.)
# Shared appdata (auth stack, Emby) use dedicated profiles defined in master.conf.
# Run manually: bash Rsync/rsync.sh /mnt/user/appdata-Fallback/HOST1-Appdata --profile=host1-appdata
PROFILE_RSYNC_OPTS[host1-appdata]="-av --info=progress2 --bwlimit=${PROFILE_BW_LIMIT[host1-appdata]:-8000}"
PROFILE_BW_LIMIT[host1-appdata]=8000
PROFILE_RETRY_COUNT[host1-appdata]=3
PROFILE_SLEEP[host1-appdata]=300
PROFILE_CRITICAL_CONTAINER_NAMES[host1-appdata]="Organizrv2-Gmer4Lfe UptimeKuma-Gmer4Lfe VaultWarden-Gmer4Lfe"
PROFILE_DELAYED_CONTAINERS[host1-appdata]=""
PROFILE_CONTAINER_DELAY[host1-appdata]=5
PROFILE_EXCLUDE_DIRS[host1-appdata]="logs *.tmp"
# ==============================================================================================
# ── DOCKER ────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Docker Daily Restart ━━━
# Containers restarted every day via DAILY_MAINTENANCE_SCRIPTS.
# Dispatcharr degrades over time without restart — daily is intentional, not just housekeeping.
# Order matters — auth stack first, then media services.
HOST1_DAILY_RESTART_CONTAINERS=(
"NginxProxyManager"
"Lldap-Gmer4Lfe"
"Authelia"
"Authelia-Secondary"
"Dispatcharr-Iptv-Users"
"Dispatcharr" # Live TV scheduler — degrades without daily restart
"Dispatcharr-Basic"
"ErsatzTV-Emby"
"slskd" # Soulseek connection drops after extended uptime; restart refreshes share index
)
# ━━━ Docker Weekly Restart ━━━
# Less critical services restarted weekly via WEEKLY_MAINTENANCE_SCRIPTS (Sunday 2:30am).
# Containers already stopped for weekly sync — restart adds zero extra downtime.
HOST1_WEEKLY_RESTART_CONTAINERS=(
"NextCloud"
"Organizrv2-Gmer4Lfe"
"AdGuard-Home"
"Immich-Gmer4Lfe"
)
# ━━━ Docker Watchdog ━━━
# Per-HOST1 container configuration for docker_watchdog.sh.
# Shared thresholds and toggles live in master.conf.
# Memory hard limits in MB — immediate restart if exceeded.
# Set at "container is clearly broken" not "container is busy".
# 20GB=20480 18GB=18432 16GB=16384 12GB=12288 8GB=8192 4GB=4096 2GB=2048 1GB=1024
declare -A HOST1_WATCHDOG_CONTAINERS=(
["Emby"]=20480 # 20GB — large library + active transcodes
["LidaTube"]=6144 # 6GB — memory leak over time
["Tdarr"]=6144 # 6GB — encoding is memory intensive
["Code-Server"]=1024 # 1GB — should never need more
)
# HTTP health check URLs — checked every cycle, strike system before restart.
# Only add containers with a meaningful web interface to check.
declare -A HOST1_WATCHDOG_CONTAINER_URLS=(
["Emby"]="http://localhost:8096"
["NginxProxyManager"]="http://localhost:7818"
["Authelia"]="http://localhost:9091/api/health"
["Authelia-Secondary"]="http://localhost:9092/api/health"
["Lldap-Gmer4Lfe"]="http://localhost:17170"
)
# Required containers — must always be running on HOST1.
# Strike system before restart — repeated failures go on skip list, auto-clears on recovery.
# Listed in dependency order — dependencies before dependents.
HOST1_WATCHDOG_REQUIRED_CONTAINERS=(
"NginxProxyManager"
"Lldap-Gmer4Lfe"
"Mariadb-Authelia"
"Mariadb-Authelia-Secondary"
"Redis-Authelia"
"Redis-Authelia-Secondary"
"Authelia"
"Authelia-Secondary"
)
# Containers to skip in Tier 2 global scan — legitimately stopped or frequently restarting.
# Watchdog leaves these alone entirely — no restart attempts, no crash loop tracking.
HOST1_WATCHDOG_SCAN_IGNORE=(
"DashGate"
"PIA-WG-Config-Generator"
"Aperture"
"Aperture-Kids"
"pgvector-18-Apeture-Kids"
"Pgvector18-Aperture"
"emby-test" # broken test container (exit 127 — bad image)
)
# Dependency ordering — skip restarting a container if its dependency is also down.
# Prevents watchdog from restarting Authelia before Mariadb is back up.
# SPACE-SEPARATED STRINGS — converted to array at runtime.
declare -A HOST1_WATCHDOG_DEPENDENCIES=(
["Authelia"]="Mariadb-Authelia Redis-Authelia"
["Authelia-Secondary"]="Mariadb-Authelia Redis-Authelia-Secondary"
["NextCloud"]="Postgres-NextCloud"
)
# Per-container appdata growth suppress ceilings in MB.
# ONLY needed in specific cases — growth rate detection covers all containers automatically.
# Use this when a container legitimately has large stable data and you want to guarantee
# it never triggers a false-positive growth alert. Growth warnings are suppressed while the
# container's dir stays below this ceiling; above it, warnings resume as normal.
# 50GB=51200 25GB=25600 20GB=20480 15GB=15360 10GB=10240 5GB=5120
declare -A HOST1_WATCHDOG_APPDATA_SIZES=(
["Tdarr"]="25600" # 25GB — transcode cache grows legitimately during active jobs
["7dtd"]="20480" # 20GB — game server world data, expected to be large
)
# ━━━ Network Watchdog ━━━
# Host-specific connectivity config for Watchdogs/System/network_watchdog.sh.
HOST1_NETWORK_WATCHDOG_DDNS_DOMAIN="gmer4lfe.com"
HOST1_NETWORK_WATCHDOG_DDNS_CONTAINER="Gmer4Lfe.com"
HOST1_NETWORK_WATCHDOG_NPM_URL="https://gmer4lfe.com"
# ━━━ Docker Network Connect ━━━
# Containers connected to custom networks at array start by docker_network_connect.sh.
# Networks created if they don't exist — idempotent, safe to re-run.
HOST1_NETWORK_CONNECT_CONTAINERS=(
"memcached"
"Npm-CrowdSec"
)
HOST1_NETWORK_CONNECT_NETWORKS=(
"high-availability"
)
# ==============================================================================================
# ── FALLBACK ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ DDNS ━━━
# DDNS containers HOST1 manages — started/stopped by fallback.sh per DDNS absolute rules:
# Internet loss → stop immediately
# Failover → HOST2 starts HOST1's DDNS as Tier 1 (before any other containers)
# Handback → stop HOST1's DDNS on HOST2 → rsync → start containers → start local DDNS last
HOST1_DDNS_CONTAINERS=(
"Gmer4Lfe.com"
)
# ━━━ Internet Loss ━━━
# Containers stopped immediately on HOST1 when internet connection is lost.
# Prevents external-facing services from operating without connectivity.
FALLBACK_HOST1_STOP_ON_NO_NET=(
"Gmer4Lfe.com"
)
# ━━━ Fallback Tiers — HOST1 Runs for HOST2 ━━━
# Containers HOST1 starts when HOST2 goes down.
# Tier 1 is always immediate — vital services cannot wait.
# Higher tiers activate after HOST2_TIER*_DELAY minutes (set in host2.conf).
FALLBACK_HOST1_COVERS_HOST2_TIER1=(
"Gmer4Lfe.us"
"VaultWarden-Jayred365"
)
FALLBACK_HOST1_COVERS_HOST2_TIER2=(
# "container-placeholder"
)
FALLBACK_HOST1_COVERS_HOST2_TIER3=(
# "container-placeholder"
)
FALLBACK_HOST1_COVERS_HOST2_TIER4=(
# "container-placeholder"
)
# ━━━ Tier Delays — HOST1's Containers on HOST2 ━━━
# How long HOST1 must be down before each tier activates on HOST2 — in minutes.
# Tier 1 is always immediate — no delay var needed.
HOST1_TIER2_DELAY=240 # 4 hours — NextCloud, Immich
HOST1_TIER3_DELAY=720 # 12 hours — secondary services
HOST1_TIER4_DELAY=1440 # 24 hours — arrs + downloaders
# ━━━ Rsync Writeback — HOST1 Appdata Back on Handback ━━━
# Syncs HOST1 appdata BACK to HOST1 when it comes back online after a failover.
# Containers stopped before writeback — clean source, no competing writes.
#
# HOST1_TIER1_WRITEBACK_DELAY: short outages skip Tier 1 writeback — primary state
# is more reliable than dirty sync data for brief outages.
HOST1_TIER1_WRITEBACK_DELAY=60 # skip Emby writeback if outage under 1hr
# Tier 4 automatically syncs HOST1_DAILY_SYNC_SHARES — only list paths NOT in that array.
FALLBACK_HOST1_WRITEBACK_TIER1=(
"/mnt/user/Media_Server/Emby" # watch states built up during outage
)
FALLBACK_HOST1_WRITEBACK_TIER2=(
"/mnt/user/appdata-Fallback/Important-Data" # NextCloud + Postgres — files added during outage
)
FALLBACK_HOST1_WRITEBACK_TIER3=(
# "location-placeholder"
)
FALLBACK_HOST1_WRITEBACK_TIER4=(
"/mnt/user/appdata-Fallback/Arrs_Stack" # arr databases — downloads queued during outage
)
# ==============================================================================================
# ── MEDIA ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Media Permissions ━━━
# Shares that media_shares_permissions.sh applies PERMISSIONS_MODE and PERMISSIONS_OWNER to.
# Runs first in DAILY_MAINTENANCE_SCRIPTS — arr cleanup depends on correct ownership.
HOST1_MEDIA_PERMISSION_SHARES=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/appcache
/mnt/user/Books
/mnt/user/Downloads
/mnt/user/Games
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movie_Recordings
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Photo
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Tv_Recordings
/mnt/user/Tv_Shows
/mnt/user/YouTube
)
# ━━━ Media Cleaner ━━━
# Folder lists for media_cleaner.sh — two profiles: anime and media.
# File patterns shared across all servers — defined in master.conf.
# Called via DAILY_MAINTENANCE_SCRIPTS. Run manually: Media/media_cleaner.sh anime|media
HOST1_ANIME_CLEAN_FOLDERS=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
)
HOST1_MEDIA_CLEAN_FOLDERS=(
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Tv_Shows
)
# ==============================================================================================
# ── MONITORS ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
# Domains checked via direct openssl connection — not relying on NPM's certificate state.
# Checks the actual certificate served, not what NPM thinks it has.
# Thresholds (CERT_WARN_DAYS, CERT_CRIT_DAYS) defined in master.conf.
HOST1_CERT_MONITOR_DOMAINS=(
"Gmer4Lfe.com"
"Gmer4Lfe.us"
)
# ━━━ SMART Health ━━━
# Drives skipped in SMART attribute monitoring — hardware is server-specific.
# Thresholds read from dynamix.cfg at runtime — fallbacks in master.conf.
HOST1_SMART_IGNORE_DRIVES=(
"sda" # boot USB — SMART not meaningful on flash drives
)
# ━━━ ZFS Report ━━━
# Pools excluded from the weekly ZFS health report — reduces noise from single-disk array pools.
# These are individual array disks formatted as ZFS — converting to XFS over time via unBalance.
# Pool health thresholds defined in master.conf.
HOST1_ZFS_REPORT_IGNORE_POOLS=(
"disk5"
"disk6"
"disk8"
"disk9"
"disk10"
)
# ==============================================================================================
# ── TRANSCODES ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Ramdisk size ceiling — tmpfs only uses RAM actually needed, not the full size upfront.
# Real-world: 9 streams peaked at ~5.5GB — 10G gives generous headroom on 128GB RAM.
HOST1_RAMDISK_SIZE="10G"
# Usage thresholds — coupled to HOST1_RAMDISK_SIZE, adjust all three together if size changes.
# Hysteresis gap (8.5 - 7 = 1.5GB) prevents flip-flop between ramdisk and SSD.
HOST1_RAMDISK_WARN_GB=8.5 # flip to SSD when ramdisk usage reaches this
HOST1_RAMDISK_LOW_GB=7 # flip back to ramdisk when usage drops to this
# SSD fallback path — where transcodes land when ramdisk exceeds HOST1_RAMDISK_WARN_GB.
# Must be on cache pool — array disks too slow for active transcode writes.
HOST1_TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/"
# Media servers sharing the ramdisk transcode space on HOST1.
# Format: "ContainerName|URL|APIKey|Type" — Type: emby | jellyfin | plex
# Entries with placeholder API keys are skipped automatically.
# ⚠️ Tdarr does NOT belong here — keep Tdarr on SSD, not ramdisk.
HOST1_TRANSCODE_SERVERS=(
"${HOST1_EMBY_CONTAINER}|${HOST1_EMBY_URL}|${HOST1_EMBY_API_KEY}|emby"
"${HOST1_JELLYFIN_CONTAINER}|${HOST1_JELLYFIN_URL}|${HOST1_JELLYFIN_API_KEY}|jellyfin"
)
# ==============================================================================================
# ── ARR STACK ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Used by arr cleanup scripts and arrs_failed_stalled_recovery.sh.
# detect_hosts() selects HOST1 vars when running on HOST1.
#
# PATH MAPS — container path → host path translation.
# Arr stores file paths using container-internal paths — scripts need host paths to scan.
# Add one entry per root folder in arr Settings → Media Management → Root Folders.
# ━━━ Downloaders ━━━
# Used by downloaders_reset.sh — runs every 30min via CRITICAL_MAINTENANCE_SCRIPTS.
# Clears stuck states, purges old history, prepares each client for a clean cycle.
# slskd — clears stuck searches, dead transfers, purges expired failed imports.
# SLSKD_FAILED_IMPORTS_DIR: where Soularr moves albums Lidarr rejected.
HOST1_SLSKD_URL="http://localhost:8980"
HOST1_SLSKD_API_KEY="4bF9kL2mNpQrT7vWxYz1A3dEgHjKoRsU"
HOST1_SLSKD_FAILED_IMPORTS_DIR="/mnt/user/Temp_Storage/Slskd/completed/failed_imports"
# SABnzbd
HOST1_SABNZBD_URL="http://localhost:8180"
HOST1_SABNZBD_API_KEY="8bfefe41d83b4d50883e32859b55ca9a"
# qBittorrent — deleteFiles=false removes torrent from qBit but leaves files on disk.
# Radarr/Sonarr manage actual files independently.
HOST1_QBIT_URL="http://localhost:8080"
HOST1_QBIT_USERNAME="root"
HOST1_QBIT_PASSWORD="Stay0utD!ck"
# ━━━ Lidarr — HOST1 only ━━━
# HOST2 does not run Lidarr — HOST1_LIDARR_RECOVERY flag handles the exit cleanly.
HOST1_LIDARR_URL="http://localhost:8686"
HOST1_LIDARR_API_KEY="b2977e71ef074bc0a0529d9fcce3b2dc"
HOST1_LIDARR_MUSIC_ROOT="/mnt/user/Music-New"
HOST1_FANART_API_KEY="Yd7147a43b692df0b364b94dc47efb81"
HOST1_LASTFM_API_KEY="be6dc169c33ae263e690c30d18b7491d"
declare -A HOST1_LIDARR_PATH_MAP=(
["/ext-music"]="/mnt/user/Music-New"
)
# ━━━ Sonarr ━━━
HOST1_SONARR_URL="http://localhost:8989"
HOST1_SONARR_API_KEY="130decd3db5b4c25afad64864cd03f9f"
HOST1_SONARR_TV_ROOT="/mnt/user/Tv_Shows"
# Note: stand-up_comedy in both Sonarr + Radarr — TV specials and movie specials, one folder
declare -A HOST1_SONARR_PATH_MAP=(
["/tv"]="/mnt/user/Tv_Shows"
["/ext-standup-comedy"]="/mnt/user/stand-up_comedy/series"
["/kids tv"]="/mnt/user/Kids_Tv_Shows"
["/ext-anime-shows"]="/mnt/user/Anime_Shows-Old"
)
# ━━━ Radarr ━━━
HOST1_RADARR_URL="http://localhost:7878"
HOST1_RADARR_API_KEY="d43a3ec6cf1549edb4af0cc63f98b2a9"
HOST1_TMDB_API_KEY="3dac5e2e49b5540472d2eafec4f01260"
HOST1_RADARR_MOVIES_ROOT="/mnt/user/Movies"
# Note: stand-up_comedy in both Radarr + Sonarr — movie specials and TV specials, one folder
declare -A HOST1_RADARR_PATH_MAP=(
["/movies"]="/mnt/user/Movies"
["/kids movies"]="/mnt/user/Kids_Movies"
["/ext-stand-up-comedy"]="/mnt/user/stand-up_comedy/specials"
["/anime-movies"]="/mnt/user/Anime_Movies-Old"
["/ext-anime-movies"]="/mnt/user/Anime_Movies-Old"
)
# ━━━ Arr Recovery Toggles ━━━
# false = skip that arr on this host — exits cleanly without error
HOST1_SONARR_RECOVERY=true
HOST1_RADARR_RECOVERY=true
HOST1_LIDARR_RECOVERY=true # HOST1 only — exits cleanly on HOST2
# ==============================================================================================
# ── SYSTEM WATCHDOG ───────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Per-host check toggles and NIC config for system_watchdog.sh.
# Aliased by detect_hosts() — script uses unprefixed SYS_WATCHDOG_* names.
# HOST1: TR1950X 128GB — full media server, active transcoding, ZFS cache pools.
#
# Three-tier response — all critical checks enabled by default on HOST1:
# Tier 1 (bypass strikes, reboot now): docker daemon, rootfs full, kernel oops, FD, /boot
# Tier 2 (bypass strikes with OOM): RAM critical + OOM kills in cycle
# Tier 3 (standard strike system): everything else
#
# RAM tiers, OOM limits, and reboot loop settings in master.conf System Watchdog section.
# ━━━ Primary NIC ━━━
# Network interface for NIC state check — verify with: ip link show | grep "^[0-9]"
# Common values: eth0, bond0, br0, eno1
HOST1_SYS_WATCHDOG_NIC="eth0"
# ━━━ Tier 1 — Critical Checks ━━━
# These bypass the strike system — a single hit triggers immediate reboot.
# Disabling any of these is not recommended — they protect against acute system failure.
# Docker daemon unresponsive → try restart, reboot if restart fails.
# Without a working daemon docker_watchdog.sh is blind and containers cannot be managed.
HOST1_SYS_WATCHDOG_CHECK_DOCKER_DAEMON=true
# rootfs at critical threshold (ROOTFS_CRITICAL_PCT=99) → reboot immediately.
# At 99% rootfs writes fail silently — logs stop, Docker errors out, SSH may stop working.
# Standard 95% threshold still uses strike system — only 99%+ is critical tier.
HOST1_SYS_WATCHDOG_CHECK_ROOTFS=true
# Kernel BUG/Oops in dmesg delta since last cycle → reboot immediately.
# A kernel oops means the kernel ran with a corrupted state — stability is not guaranteed.
HOST1_SYS_WATCHDOG_CHECK_KERNEL_OOPS=true
# File descriptor exhaustion at FD_CRITICAL_PCT (95%) → reboot immediately.
# At 95% FD: new connections fail, Docker can't spawn processes, SSH drops.
HOST1_SYS_WATCHDOG_CHECK_FD=true
# /boot read-only detected → reboot immediately.
# Unexpected read-only /boot means state files and config writes are silently failing.
# Fallback state, watchdog reboot log, and lock files all go stale silently.
HOST1_SYS_WATCHDOG_CHECK_BOOT=true
# ━━━ Tier 2 — Urgent OOM Check ━━━
# Bypass strikes when RAM is critically low AND OOM kill rate confirms active crisis.
# Both must be enabled for Tier 2 bypass to function — disable either to always use strikes.
# Track kernel OOM kills each cycle via /proc/vmstat oom_kill delta.
# Also provides diagnostic context in reboot messages (which processes were killed).
HOST1_SYS_WATCHDOG_CHECK_OOM=true
# Free RAM check — required for both Tier 2 bypass and RAM tier logic.
# Tiers: MEM_WARN_GB(10) → notify | MEM_SHUTDOWN_GB(6) → stop containers | MEM_GB(4) → strikes
HOST1_SYS_WATCHDOG_CHECK_RAM=true
# ━━━ Tier 3 — Standard Checks (strike system) ━━━
# Each check must fail SYS_WATCHDOG_STRIKE_LIMIT consecutive cycles before action is taken.
# Single spikes are ignored — sustained problems trigger reboot.
# /var/log filesystem usage above SYS_WATCHDOG_LOG_PCT.
# Log spam (Docker log storms, syslog loops) fills rootfs — indicates something broken.
HOST1_SYS_WATCHDOG_CHECK_LOG=true
# ZFS ARC memory pinned above SYS_WATCHDOG_ARC_PINNED_PCT after cache drop.
# Enabled on HOST1 — ZFS cache pools actively used. Disable on hosts without ZFS.
HOST1_SYS_WATCHDOG_CHECK_ARC=true
# CPU temperature above SYS_WATCHDOG_CPU_TEMP_MAX (95°C).
# Sustained high temp causes kernel throttling or panic. Requires lm-sensors.
HOST1_SYS_WATCHDOG_CHECK_CPU_TEMP=true
# Load average above SYS_WATCHDOG_LOAD_MULTIPLIER × core count.
# DISABLED on HOST1 — Tdarr and Emby cause legitimate sustained load spikes during encoding.
# Enable on idle servers or adjust SYS_WATCHDOG_LOAD_MULTIPLIER if load is always high.
HOST1_SYS_WATCHDOG_CHECK_LOAD=false
# Zombie process count above SYS_WATCHDOG_ZOMBIE_LIMIT (50).
# Large zombie counts indicate serious process management failure — something is stuck.
HOST1_SYS_WATCHDOG_CHECK_ZOMBIES=true
# Check docker_watchdog.sh persistent skip list — required containers on skip list.
# Cross-watchdog coordination: if docker_watchdog gave up, system_watchdog escalates.
# ENABLED — HOST1 fully built and operational, skip list is meaningful.
HOST1_SYS_WATCHDOG_CHECK_CONTAINERS=true
# /tmp filesystem usage above SYS_WATCHDOG_TMP_PCT with auto-clear attempt.
# Script tries to clear aged /tmp files first — only strikes if clear fails.
# Lock files, rsync temp files, and Docker ops use /tmp — 100% means lock failures.
HOST1_SYS_WATCHDOG_CHECK_TMP=true
# Array disk error count delta in /proc/mdstat — accumulating errors = disk failing now.
# Triggers on SYS_WATCHDOG_MDSTAT_ERROR_LIMIT new errors in one cycle.
HOST1_SYS_WATCHDOG_CHECK_MDSTAT=true
# Primary NIC operstate — detects NIC going down (physical or driver failure).
# Uses HOST1_SYS_WATCHDOG_NIC above. Strike system — brief flaps don't trigger reboot.
HOST1_SYS_WATCHDOG_CHECK_NETWORK=true
# sshd running check — attempts restart before escalating.
# sshd down = no remote access. Script tries rc.sshd start, notifies, strikes on failure.
HOST1_SYS_WATCHDOG_CHECK_SSHD=true
# Runaway process detection — single process above SYS_WATCHDOG_RUNAWAY_CPU_PCT sustained.
# DISABLED — Tdarr encoding and Emby transcoding legitimately peg CPU for extended periods.
# Enable only if HOST1 has no CPU-intensive workloads.
HOST1_SYS_WATCHDOG_CHECK_RUNAWAY=false
# ==============================================================================================
# ── RESOURCE MANAGER ──────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Containers to manage under pressure — see master.conf RW_CRITICAL_CONTAINERS for exclusions.
# docker pause at medium pressure (RAM < RW_RAM_MEDIUM_GB or load > medium threshold)
# Suspended in-place — instant to pause/unpause, no state lost, no restart delay.
HOST1_RW_PAUSE_CONTAINERS=(
"Huntarr" # arr search automation — safe to suspend
"Cleanuparr" # download cleanup — safe to suspend
"Healarr" # arr health checks — safe to suspend
"Soularr" # Slskd automation — background only
"ChannelTube" # YouTube archiver — background only
"Pinchflat" # YouTube archiver — background only
)
# docker stop at hard pressure (RAM < RW_RAM_HARD_GB)
# Full stop — these are optional/heavy services that free significant RAM when stopped.
# resource_watchdog.sh restarts them when pressure fully clears (RAM >= RW_RAM_RECOVER_GB).
HOST1_RW_STOP_CONTAINERS=(
"LocalAI" # GPU/CPU heavy — largest RAM consumer when idle
"7DaysToDie" # game server — optional
"V-Rising" # game server — optional
"Code-Server" # IDE — not needed during pressure events
)
# ==============================================================================================
# ──────────────────────── End Of HOST1 Variables ──────────────────────────────────────────────
# ==============================================================================================
HOST1_UNRAID_API_KEY="1825c3a2e03ea5089974f4da2e171aa2d5907a1dea23cc479c33e492c8ff4dbb"
@@ -1,334 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ================================= Git Pull & Execute =========================================
# ==============================================================================================
# Pulls the latest scripts from the Gitea repository via SSH.
# Lives at the repo root — sources load_config.sh from the same directory.
#
# ── WHAT THIS SCRIPT DOES ─────────────────────────────────────────────────────────────────────
# 1. Detects which server it's running on via detect_hosts() (MY_ID)
# 2. Configures sparse checkout to exclude other servers' credential files
# Each server only pulls its own host*.conf — never sees peer credentials
# 3. Pulls or clones latest scripts from Gitea
# 4. Sets executable permissions on all .sh files
#
# ── SPARSE CHECKOUT ───────────────────────────────────────────────────────────────────────────
# Sparse checkout ensures each server only receives its own host conf:
# HOST1 pulls: master.conf + host1.conf + all scripts
# HOST1 skips: host2.conf, host3.conf etc.
# HOST2 pulls: master.conf + host2.conf + all scripts
# HOST2 skips: host1.conf, host3.conf etc.
#
# Adding a new server:
# Create host3.conf in the repo
# All existing servers automatically exclude it on next pull
# New server gets only its own conf ✅
#
# ── GITEA LOCATION DETECTION ──────────────────────────────────────────────────────────────────
# Detects where Gitea is running at runtime — works through fallback:
# Gitea local → connects via local IP
# Gitea remote → connects via Tailscale IP
# Both fail → falls back to GITEA_DOMAIN if configured
#
# ── CONFIGURATION (master.conf) ───────────────────────────────────────────────────────────────
# GITEA_CONTAINER — Docker container name for Gitea
# GITEA_REPO_PATH — repo path on Gitea (e.g. Varaverk/varaverk.git)
# GITEA_DOMAIN — public domain fallback (optional)
# TARGET_DIR — local path to clone/pull into
# GITEA_SSH_KEY — SSH key for Gitea authentication
# SSH_PORT — Gitea SSH port (often 221 or 222)
#
# ── USAGE ─────────────────────────────────────────────────────────────────────────────────────
# git_pull_execute.sh — normal pull
# git_pull_execute.sh --dry-run — preview without making changes
# git_pull_execute.sh --log — verbose output
# git_pull_execute.sh --status — show config and exit
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
# Root level script — load_config.sh is in the same directory
source "$SCRIPT_DIR/load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
acquire_lock
if ! command -v docker &>/dev/null; then
error "Docker command not found"
exit 1
fi
# detect_hosts() sets MY_ID — needed for sparse checkout configuration
detect_hosts
# ==============================================================================================
# ━━━ Locate Gitea ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CONTAINERS Locate Gitea ━━━"
if docker ps --format "{{.Names}}" 2>/dev/null | grep -q "^${GITEA_CONTAINER}$"; then
# Gitea is running on this server — use local IP
GITEA_IP=$(hostname -I | awk '{print $1}')
log "Gitea running locally — connecting via $GITEA_IP"
else
# Gitea not running locally — find it on the remote server via Tailscale
log "Gitea not running locally — checking remote server"
GITEA_IP=$(resolve_tailscale_ip "${REMOTE_SERVER_NAME}")
if [[ -n "$GITEA_IP" ]]; then
echo " Gitea on $REMOTE_SERVER_NAME — connecting via Tailscale $GITEA_IP"
elif [[ -n "${GITEA_DOMAIN:-}" ]]; then
warn "Tailscale resolution failed — falling back to $GITEA_DOMAIN"
GITEA_IP="$GITEA_DOMAIN"
else
error "Cannot find Gitea — local: not running, Tailscale: failed, domain: not configured"
notify "Git pull failed on $(hostname) — cannot locate Gitea container" "Git Sync" "alert"
exit 1
fi
fi
REPO_SSH="git@${GITEA_IP}:${GITEA_REPO_PATH}"
require_var REPO_SSH
require_var TARGET_DIR
require_var GITEA_SSH_KEY
require_var SSH_PORT
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_NET Repo: $REPO_SSH"
echo "$ICON_GEAR Target: $TARGET_DIR"
echo "$ICON_GEAR SSH Key: $GITEA_SSH_KEY"
echo "$ICON_GEAR SSH Port: $SSH_PORT"
echo "$ICON_HOST My ID: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_HOST Remote ID: $REMOTE_ID ($REMOTE_SERVER_NAME)"
echo "$ICON_NOTIFY Notify: unRAID=${NOTIFY_UNRAID:-false} Discord=$([[ -n "${MY_DISCORD_WEBHOOK:-}" ]] && echo enabled || echo disabled)"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes will be made"
# ==============================================================================================
# ━━━ Sparse Checkout Configuration ━━━
# ==============================================================================================
# Build the list of host*.conf files that belong to OTHER servers.
# This server pulls everything EXCEPT those files.
# MY_ID is set by detect_hosts() — e.g. "HOST1"
configure_sparse_checkout() {
local repo_dir="$1"
log "Configuring sparse checkout for $MY_ID..."
# Enable sparse checkout
git -C "$repo_dir" config core.sparseCheckout true 2>/dev/null
# Build exclusion list — all host*.conf files except MY_ID's
local sparse_file="$repo_dir/.git/info/sparse-checkout"
mkdir -p "$(dirname "$sparse_file")"
# Start with: pull everything
echo "/*" > "$sparse_file"
# Exclude each other server's conf file
# Find all host*.conf files present in the repo
local excluded=0
for conf_file in "$repo_dir"/host*.conf; do
[[ -f "$conf_file" ]] || continue
local conf_name
conf_name=$(basename "$conf_file")
# Determine which HOST ID owns this conf by grepping its hostname var
# Pattern: HOST1="unRAID-..." or HOST2="unRAID-..."
local conf_host_id
conf_host_id=$(grep -m1 -oP '^\s+HOST[0-9]+(?==)' "$conf_file" 2>/dev/null | tr -d ' ')
if [[ -z "$conf_host_id" ]]; then
log "Cannot determine HOST ID for $conf_name — including in pull (safe default)"
continue
fi
if [[ "$conf_host_id" != "$MY_ID" ]]; then
echo "!$conf_name" >> "$sparse_file"
log "Sparse checkout: excluding $conf_name (belongs to $conf_host_id)"
((excluded++))
else
log "Sparse checkout: including $conf_name (belongs to $MY_ID — this server)"
fi
done
if [[ "$excluded" -gt 0 ]]; then
echo " Sparse checkout: excluding $excluded peer conf file(s) — credentials protected"
else
log "Sparse checkout: no peer conf files to exclude (single server or first run)"
fi
}
# ==============================================================================================
# ━━━ Git Sync ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SYNC Git Sync ━━━"
echo "$ICON_NET Repo: $REPO_SSH"
echo "$ICON_GEAR Target: $TARGET_DIR"
echo ""
START=$(date +%s)
SYNC_SUCCESS=false
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would sync $REPO_SSH → $TARGET_DIR"
warn "DRY RUN — would configure sparse checkout for $MY_ID"
warn "DRY RUN — would exclude peer host*.conf files"
SYNC_SUCCESS=true
else
mkdir -p "$TARGET_DIR"
git config --global --add safe.directory "$TARGET_DIR"
cd "$TARGET_DIR" || { error "Cannot cd into $TARGET_DIR"; exit 1; }
if [[ -d ".git" ]]; then
# ── Existing repository ──────────────────────────────────────────────
echo " Existing repository — updating"
# Configure sparse checkout BEFORE pull
# Uses conf files already present from last pull to determine exclusions
configure_sparse_checkout "$TARGET_DIR"
echo " Pulling latest changes..."
if GIT_SSH_COMMAND="ssh -i $GITEA_SSH_KEY -p $SSH_PORT" git pull --ff-only; then
echo " Git pull successful"
SYNC_SUCCESS=true
else
# ff-only fails when local commits or tracked changes exist that can't
# fast-forward. Fail loudly — never silently destroy local work.
error "Git pull failed — local changes conflict with remote (will not force-reset)"
notify "Git pull failed on $(hostname) — local changes conflict, manual resolve needed" "Git Sync" "alert"
exit 1
fi
else
# ── Fresh clone ──────────────────────────────────────────────────────
echo " No repository found — cloning"
# Clone first — need the repo to exist before configuring sparse checkout
if GIT_SSH_COMMAND="ssh -i $GITEA_SSH_KEY -p $SSH_PORT" git clone "$REPO_SSH" .; then
echo " Clone successful"
# Configure sparse checkout after clone
# Now all host*.conf files are present — can detect exclusions
configure_sparse_checkout "$TARGET_DIR"
# Apply sparse checkout — removes excluded files from working tree
echo " Applying sparse checkout..."
git read-tree -mu HEAD
echo " Sparse checkout applied — peer credentials removed from working tree"
SYNC_SUCCESS=true
else
error "Clone failed"
notify "Git clone failed on $(hostname) — check Gitea connectivity" "Git Sync" "alert"
exit 1
fi
fi
# ── Permissions ──────────────────────────────────────────────────────────
echo ""
echo "━━━ $ICON_GEAR Permissions ━━━"
log "Setting executable permissions on all .sh files..."
find "$TARGET_DIR" -type f -name "*.sh" -exec chmod +x {} \;
echo " Permissions set on .sh files"
# ── Flash mode: sync Plugin/ to /boot/ so the webUI picks up updates ─────
# In flash mode SCRIPTS_DIR is in appdata — Plugin/ lives in the repo there
# but Unraid serves PHP from /boot/. Sync after every pull to keep them in step.
_BOOT_DIR="/boot/config/plugins/varaverk"
if [[ "$TARGET_DIR" != "$_BOOT_DIR" ]]; then
echo ""
echo "━━━ $ICON_SYNC Flash mode: sync Plugin/ → /boot/ ━━━"
if rsync -a --delete "$TARGET_DIR/Plugin/" "$_BOOT_DIR/Plugin/" 2>/dev/null; then
echo " Plugin/ synced to /boot/ ✅"
else
warn "Plugin/ sync to /boot/ failed — webUI may be stale until next pull"
fi
fi
fi
END=$(date +%s)
# ==============================================================================================
# ━━━ Conf Upgrade ━━━
# ==============================================================================================
# Merges new conf structure into the live conf files after every pull.
# New keys → added with template defaults (user fills in once).
# Removed keys → dropped. Existing values → always preserved.
# Silent when already up to date — no overhead on unchanged pulls.
echo ""
echo "━━━ $ICON_GEAR Conf Upgrade ━━━"
UPGRADE_SCRIPT="$TARGET_DIR/Deployment/conf_upgrade.sh"
TMPL_DIR="$TARGET_DIR/Deployment/conf_templates"
CONF_DIR="$TARGET_DIR/Configurations"
if [[ ! -f "$UPGRADE_SCRIPT" ]]; then
log "conf_upgrade.sh not found — skipping (pre-deployment-folder repo)"
elif [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would upgrade master.conf and ${MY_ID,,}.conf"
elif [[ "$SYNC_SUCCESS" == true ]]; then
_DRY=""
# master.conf
if [[ -f "$TMPL_DIR/master.conf" && -f "$CONF_DIR/master.conf" ]]; then
bash "$UPGRADE_SCRIPT" \
--template "$TMPL_DIR/master.conf" \
--target "$CONF_DIR/master.conf" \
--backup $_DRY
else
warn "master.conf template or target not found — skipping"
fi
# This server's host conf only — sparse checkout ensures we have it
HOST_CONF="$CONF_DIR/${MY_ID,,}.conf"
if [[ -f "$TMPL_DIR/host.conf.template" && -f "$HOST_CONF" ]]; then
bash "$UPGRADE_SCRIPT" \
--template "$TMPL_DIR/host.conf.template" \
--target "$HOST_CONF" \
--backup $_DRY
else
warn "${MY_ID,,}.conf or host.conf.template not found — skipping"
fi
fi
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY GIT SYNC SUMMARY ━━━━━"
echo "$ICON_NET Repo: $REPO_SSH"
echo "$ICON_GEAR Target: $TARGET_DIR"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_LOCK Excluded: peer host*.conf files"
echo "$ICON_TIME Duration: $(format_duration $((END - START)))"
if [[ "$DRY_RUN" == true ]]; then
echo "$ICON_WARN Status: DRY RUN — no changes made"
elif [[ "$SYNC_SUCCESS" == true ]]; then
echo "$ICON_DONE Status: $ICON_SUCCESS DONE"
notify "Repository synced successfully on $(hostname)" "Git Sync" "normal"
else
echo "$ICON_ERROR Status: $ICON_ERROR FAILED"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
@@ -1,333 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ================================= Git Pull & Execute =========================================
# ==============================================================================================
# Pulls the latest scripts from the Gitea repository via SSH.
# Lives at the repo root — sources load_config.sh from the same directory.
#
# ── WHAT THIS SCRIPT DOES ─────────────────────────────────────────────────────────────────────
# 1. Detects which server it's running on via detect_hosts() (MY_ID)
# 2. Configures sparse checkout to exclude other servers' credential files
# Each server only pulls its own host*.conf — never sees peer credentials
# 3. Pulls or clones latest scripts from Gitea
# 4. Sets executable permissions on all .sh files
#
# ── SPARSE CHECKOUT ───────────────────────────────────────────────────────────────────────────
# Sparse checkout ensures each server only receives its own host conf:
# HOST1 pulls: master.conf + host1.conf + all scripts
# HOST1 skips: host2.conf, host3.conf etc.
# HOST2 pulls: master.conf + host2.conf + all scripts
# HOST2 skips: host1.conf, host3.conf etc.
#
# Adding a new server:
# Create host3.conf in the repo
# All existing servers automatically exclude it on next pull
# New server gets only its own conf ✅
#
# ── GITEA LOCATION DETECTION ──────────────────────────────────────────────────────────────────
# Detects where Gitea is running at runtime — works through fallback:
# Gitea local → connects via local IP
# Gitea remote → connects via Tailscale IP
# Both fail → falls back to GITEA_DOMAIN if configured
#
# ── CONFIGURATION (master.conf) ───────────────────────────────────────────────────────────────
# GITEA_CONTAINER — Docker container name for Gitea
# GITEA_REPO_PATH — repo path on Gitea (e.g. Varaverk/varaverk.git)
# GITEA_DOMAIN — public domain fallback (optional)
# TARGET_DIR — local path to clone/pull into
# GITEA_SSH_KEY — SSH key for Gitea authentication
# SSH_PORT — Gitea SSH port (often 221 or 222)
#
# ── USAGE ─────────────────────────────────────────────────────────────────────────────────────
# git_pull_execute.sh — normal pull
# git_pull_execute.sh --dry-run — preview without making changes
# git_pull_execute.sh --log — verbose output
# git_pull_execute.sh --status — show config and exit
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
# Root level script — load_config.sh is in the same directory
source "$SCRIPT_DIR/load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
acquire_lock
if ! command -v docker &>/dev/null; then
error "Docker command not found"
exit 1
fi
# detect_hosts() sets MY_ID — needed for sparse checkout configuration
detect_hosts
# ==============================================================================================
# ━━━ Locate Gitea ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CONTAINERS Locate Gitea ━━━"
if docker ps --format "{{.Names}}" 2>/dev/null | grep -q "^${GITEA_CONTAINER}$"; then
# Gitea is running on this server — use local IP
GITEA_IP=$(hostname -I | awk '{print $1}')
log "Gitea running locally — connecting via $GITEA_IP"
else
# Gitea not running locally — find it on the remote server via Tailscale
log "Gitea not running locally — checking remote server"
GITEA_IP=$(resolve_tailscale_ip "${REMOTE_SERVER_NAME}")
if [[ -n "$GITEA_IP" ]]; then
echo " Gitea on $REMOTE_SERVER_NAME — connecting via Tailscale $GITEA_IP"
elif [[ -n "${GITEA_DOMAIN:-}" ]]; then
warn "Tailscale resolution failed — falling back to $GITEA_DOMAIN"
GITEA_IP="$GITEA_DOMAIN"
else
error "Cannot find Gitea — local: not running, Tailscale: failed, domain: not configured"
notify "Git pull failed on $(hostname) — cannot locate Gitea container" "Git Sync" "alert"
exit 1
fi
fi
REPO_SSH="git@${GITEA_IP}:${GITEA_REPO_PATH}"
require_var REPO_SSH
require_var TARGET_DIR
require_var GITEA_SSH_KEY
require_var SSH_PORT
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_NET Repo: $REPO_SSH"
echo "$ICON_GEAR Target: $TARGET_DIR"
echo "$ICON_GEAR SSH Key: $GITEA_SSH_KEY"
echo "$ICON_GEAR SSH Port: $SSH_PORT"
echo "$ICON_HOST My ID: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_HOST Remote ID: $REMOTE_ID ($REMOTE_SERVER_NAME)"
echo "$ICON_NOTIFY Notify: unRAID=${NOTIFY_UNRAID:-false} Discord=$([[ -n "${MY_DISCORD_WEBHOOK:-}" ]] && echo enabled || echo disabled)"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes will be made"
# ==============================================================================================
# ━━━ Sparse Checkout Configuration ━━━
# ==============================================================================================
# Build the list of host*.conf files that belong to OTHER servers.
# This server pulls everything EXCEPT those files.
# MY_ID is set by detect_hosts() — e.g. "HOST1"
configure_sparse_checkout() {
local repo_dir="$1"
log "Configuring sparse checkout for $MY_ID..."
# Enable sparse checkout
git -C "$repo_dir" config core.sparseCheckout true 2>/dev/null
# Build exclusion list — all host*.conf files except MY_ID's
local sparse_file="$repo_dir/.git/info/sparse-checkout"
mkdir -p "$(dirname "$sparse_file")"
# Start with: pull everything
echo "/*" > "$sparse_file"
# Exclude each other server's conf file
# Find all host*.conf files present in the repo
local excluded=0
for conf_file in "$repo_dir"/host*.conf; do
[[ -f "$conf_file" ]] || continue
local conf_name
conf_name=$(basename "$conf_file")
# Determine which HOST ID owns this conf by grepping its hostname var
# Pattern: HOST1="unRAID-..." or HOST2="unRAID-..."
local conf_host_id
conf_host_id=$(grep -m1 -oP '^\s+HOST[0-9]+(?==)' "$conf_file" 2>/dev/null | tr -d ' ')
if [[ -z "$conf_host_id" ]]; then
log "Cannot determine HOST ID for $conf_name — including in pull (safe default)"
continue
fi
if [[ "$conf_host_id" != "$MY_ID" ]]; then
echo "!$conf_name" >> "$sparse_file"
log "Sparse checkout: excluding $conf_name (belongs to $conf_host_id)"
((excluded++))
else
log "Sparse checkout: including $conf_name (belongs to $MY_ID — this server)"
fi
done
if [[ "$excluded" -gt 0 ]]; then
echo " Sparse checkout: excluding $excluded peer conf file(s) — credentials protected"
else
log "Sparse checkout: no peer conf files to exclude (single server or first run)"
fi
}
# ==============================================================================================
# ━━━ Git Sync ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SYNC Git Sync ━━━"
echo "$ICON_NET Repo: $REPO_SSH"
echo "$ICON_GEAR Target: $TARGET_DIR"
echo ""
START=$(date +%s)
SYNC_SUCCESS=false
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would sync $REPO_SSH → $TARGET_DIR"
warn "DRY RUN — would configure sparse checkout for $MY_ID"
warn "DRY RUN — would exclude peer host*.conf files"
SYNC_SUCCESS=true
else
mkdir -p "$TARGET_DIR"
git config --global --add safe.directory "$TARGET_DIR"
cd "$TARGET_DIR" || { error "Cannot cd into $TARGET_DIR"; exit 1; }
if [[ -d ".git" ]]; then
# ── Existing repository ──────────────────────────────────────────────
echo " Existing repository — updating"
# Configure sparse checkout BEFORE pull
# Uses conf files already present from last pull to determine exclusions
configure_sparse_checkout "$TARGET_DIR"
echo " Pulling latest changes..."
if GIT_SSH_COMMAND="ssh -i $GITEA_SSH_KEY -p $SSH_PORT" git pull --ff-only; then
echo " Git pull successful"
SYNC_SUCCESS=true
else
# ff-only fails when local commits or tracked changes exist that can't
# fast-forward. Fail loudly — never silently destroy local work.
error "Git pull failed — local changes conflict with remote (will not force-reset)"
notify "Git pull failed on $(hostname) — local changes conflict, manual resolve needed" "Git Sync" "alert"
exit 1
fi
else
# ── Fresh clone ──────────────────────────────────────────────────────
echo " No repository found — cloning"
# Clone first — need the repo to exist before configuring sparse checkout
if GIT_SSH_COMMAND="ssh -i $GITEA_SSH_KEY -p $SSH_PORT" git clone "$REPO_SSH" .; then
echo " Clone successful"
# Configure sparse checkout after clone
# Now all host*.conf files are present — can detect exclusions
configure_sparse_checkout "$TARGET_DIR"
# Apply sparse checkout — removes excluded files from working tree
echo " Applying sparse checkout..."
git read-tree -mu HEAD
echo " Sparse checkout applied — peer credentials removed from working tree"
SYNC_SUCCESS=true
else
error "Clone failed"
notify "Git clone failed on $(hostname) — check Gitea connectivity" "Git Sync" "alert"
exit 1
fi
fi
# ── Permissions ──────────────────────────────────────────────────────────
echo ""
echo "━━━ $ICON_GEAR Permissions ━━━"
log "Setting executable permissions on all .sh files..."
find "$TARGET_DIR" -type f -name "*.sh" -exec chmod +x {} \;
echo " Permissions set on .sh files"
# ── Flash mode: sync Plugin/ to /boot/ so the webUI picks up updates ─────
# In flash mode SCRIPTS_DIR is in appdata — Plugin/ lives in the repo there
# but Unraid serves PHP from /boot/. Sync after every pull to keep them in step.
_BOOT_DIR="/boot/config/plugins/varaverk"
if [[ "$TARGET_DIR" != "$_BOOT_DIR" ]]; then
echo ""
echo "━━━ $ICON_SYNC Flash mode: sync Plugin/ → /boot/ ━━━"
if rsync -a --delete "$TARGET_DIR/Plugin/" "$_BOOT_DIR/Plugin/" 2>/dev/null; then
echo " Plugin/ synced to /boot/ ✅"
else
warn "Plugin/ sync to /boot/ failed — webUI may be stale until next pull"
fi
fi
fi
END=$(date +%s)
# ==============================================================================================
# ━━━ Conf Upgrade ━━━
# ==============================================================================================
# Merges new conf structure into the live conf files after every pull.
# New keys → added with template defaults (user fills in once).
# Removed keys → dropped. Existing values → always preserved.
# Silent when already up to date — no overhead on unchanged pulls.
echo ""
echo "━━━ $ICON_GEAR Conf Upgrade ━━━"
UPGRADE_SCRIPT="$TARGET_DIR/Deployment/conf_upgrade.sh"
CONF_DIR="$TARGET_DIR/Configurations"
if [[ ! -f "$UPGRADE_SCRIPT" ]]; then
log "conf_upgrade.sh not found — skipping (pre-deployment-folder repo)"
elif [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would upgrade master.conf and ${MY_ID,,}.conf"
elif [[ "$SYNC_SUCCESS" == true ]]; then
_DRY=""
# master.conf
if [[ -f "$CONF_DIR/master.conf.template" && -f "$CONF_DIR/master.conf" ]]; then
bash "$UPGRADE_SCRIPT" \
--template "$CONF_DIR/master.conf.template" \
--target "$CONF_DIR/master.conf" \
--backup $_DRY
else
warn "master.conf.template or master.conf not found — skipping"
fi
# This server's host conf only — sparse checkout ensures we have it
HOST_CONF="$CONF_DIR/${MY_ID,,}.conf"
if [[ -f "$CONF_DIR/host.conf.template" && -f "$HOST_CONF" ]]; then
bash "$UPGRADE_SCRIPT" \
--template "$CONF_DIR/host.conf.template" \
--target "$HOST_CONF" \
--backup $_DRY
else
warn "${MY_ID,,}.conf or host.conf.template not found — skipping"
fi
fi
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY GIT SYNC SUMMARY ━━━━━"
echo "$ICON_NET Repo: $REPO_SSH"
echo "$ICON_GEAR Target: $TARGET_DIR"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_LOCK Excluded: peer host*.conf files"
echo "$ICON_TIME Duration: $(format_duration $((END - START)))"
if [[ "$DRY_RUN" == true ]]; then
echo "$ICON_WARN Status: DRY RUN — no changes made"
elif [[ "$SYNC_SUCCESS" == true ]]; then
echo "$ICON_DONE Status: $ICON_SUCCESS DONE"
notify "Repository synced successfully on $(hostname)" "Git Sync" "normal"
else
echo "$ICON_ERROR Status: $ICON_ERROR FAILED"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
@@ -1,122 +0,0 @@
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
# 🔌 PLUGIN
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
**The Varaverk Unraid plugin — a web UI that wraps the entire script ecosystem.**
Scheduler, Monitor, Docker management, Partnership sync, Fallback state, and Arrs —
all surfaced inside the Unraid web interface as a first-class plugin.
> **Why this folder exists:** The scripts need a control surface. Managing a 50+ container
> homelab ecosystem from terminal windows is friction. The plugin turns configuration files
> into editable forms, cron schedules into a visual scheduler, and runtime log output into
> a live dashboard — without duplicating any of the logic that already lives in common.sh
> and the conf files.
---
## ━━━ THE PROBLEM THAT BUILT THIS ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
The script ecosystem works well from the command line, but day-to-day operation is not
the command line. Checking whether the nightly sync ran, adjusting a container's watchdog
limit, confirming the partnership fallback is active — all of that requires SSH sessions,
knowing which log files to look at, and remembering which conf variable controls what.
The plugin solves the visibility problem: one URL on any browser, on any device on the
Tailscale network, shows everything running and lets you act on it. No extra tooling,
no separate monitoring stack, no third-party dashboards.
---
## ━━━ WHAT THIS FOLDER CONTAINS ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
```
Plugin/
├── dev_install.sh # One-time developer setup: symlinks plugin into web server
├── Icons/ # Source icon assets (1024px master files)
└── unraid/ # The Unraid platform adapter + plugin application
├── adapter.sh # Platform adapter — provides platform_*() API to all scripts
├── Varaverk.page # Main plugin entry point (Tasks menu)
├── VaraverkSettings.page # Unraid Settings → Other Settings entry
├── api/ # PHP API endpoints (called by JS via fetch)
├── css/ # Plugin stylesheet
├── event/ # Unraid event hooks (boot-time cron setup, array lifecycle)
├── icons/ # Plugin icons served by emhttp
├── images/ # Plugin images
├── include/ # PHP business logic shared across pages
├── js/ # Frontend JavaScript
├── pages/ # Per-tab page includes (monitor, scheduler, docker, ...)
└── run_job.sh # Script runner invoked by the Scheduler
# Future platform adapters follow the same structure:
# Plugin/truenas/adapter.sh — TrueNAS adapter (future)
# Plugin/ubuntu/adapter.sh — Ubuntu/Debian adapter (future)
```
---
## ━━━ RELATIONSHIP TO THE REST OF THE REPO ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
**Plugin is a wrapper, never a reimplementation.** Every setting the plugin reads or writes
lives in `Configurations/master.conf` or `Configurations/host*.conf` — the same files the
shell scripts read. The plugin has no separate data store. If a conf file changes outside
the plugin (by hand, by SSH), the plugin reflects it on next load.
The one exception is `varaverk.cfg` on flash (`/boot/config/plugins/varaverk/varaverk.cfg`),
which holds a single bootstrap value: `SCRIPTS_DIR`. This is the path the plugin uses to
find the Configurations directory and all scripts. Everything else flows from there.
The plugin also taps `common.sh` indirectly — `include/config.php` mirrors
`resolve_tailscale_ip()` and `detect_host()` exactly, using the same logic as common.sh
so behaviour stays consistent without a shell dependency.
---
## ━━━ SCRIPTS IN THIS FOLDER ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
| Script | Role | When It Runs |
|--------|------|--------------|
| `dev_install.sh` | Symlinks `Plugin/unraid/` into Unraid's web server | Once, manually, after cloning or moving the repo |
---
## ━━━ THE PLATFORM ADAPTER ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
`Plugin/unraid/adapter.sh` is the Unraid platform adapter. It is sourced automatically
by `load_config.sh` whenever `PLATFORM=unraid` is detected (via `/etc/unraid-version`).
Every bash script in the ecosystem calls `platform_*()` functions instead of OS-specific
commands directly. The adapter translates those calls into Unraid-specific implementations.
```
platform_storage_healthy # is the array up and shfs mounted?
platform_is_maintenance_running # parity check or sync in progress?
platform_is_service_running # is a named service process alive?
platform_restart_service # restart via rc.d (Unraid) or systemctl (future)
platform_stop_service # stop a named service
platform_is_mover_running # Unraid mover active?
platform_get_mover_pid # PID of the mover process
platform_stop_user_scripts # kill Unraid user.scripts background jobs
platform_send_os_notification # dynamix notify (Unraid) or equivalent
platform_get_disk_states # reads disks.ini (Unraid) or equivalent
platform_get_temp_thresholds # reads dynamix.cfg (Unraid) or equivalent
platform_is_service_enabled # docker.cfg / domain.cfg enabled check
platform_require_cmd # verify a platform command exists
```
**Adding a new platform:** Create `Plugin/<platform>/adapter.sh` implementing the same
function names. `load_config.sh` detects the OS at runtime and sources the correct adapter.
No other files need changing.
---
## ━━━ UNRAID INTEGRATION POINTS ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
| File | Where it appears in Unraid |
|------|---------------------------|
| `Varaverk.page` | Tasks menu item |
| `VaraverkSettings.page` | Settings → Other Settings tile |
| `event/disks_mounted/rebuild_cron` | Fires on every boot — copies `.plg`, rebuilds cron |
| `event/disks_mounted/array_start_jobs` | Fires when array starts |
| `event/disks_unmounting/array_stop_jobs` | Fires when array stops |
| `/boot/config/plugins/varaverk.plg` | Registers the plugin with Unraid's plugin system (lives on flash, not in repo) |
@@ -1,650 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ================================= Downloaders Reset ==========================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Maintenance reset for all download clients on this server. Clears accumulated
# state that download clients generate but never clean up themselves — stuck
# searches, dead transfers, failed imports, stale queue entries, completed history.
#
# Called every 30 minutes by critical_sync_maintenance.sh via
# CRITICAL_MAINTENANCE_SCRIPTS. Can also be run manually for ad hoc cleanup.
# If a downloader is not configured for this host, that section skips cleanly.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# slskd
# Stuck searches — clears Completed/Errored searches left by Soularr crashes
# prevents 409 Conflict on next Soularr startup
# Dead transfers — removes completed/errored/aborted transfer records per user
# prevents Soularr 404 loop when polling a user whose transfer is gone
# NEVER removes InProgress or Queued transfers
# Failed imports — purges albums Soularr downloaded but Lidarr rejected
# Soularr moves these to failed_imports/ and never cleans them up
#
# SABnzbd
# Completed history — removes completed download records older than DOWNLOADER_RETENTION_DAYS
# Failed history — removes failed download records older than DOWNLOADER_RETENTION_DAYS
# Stalled queue — removes Paused or Stuck queue items no longer progressing
# active downloading items are never touched
#
# qBittorrent
# Age failsafe — removes torrents older than QBIT_FAILSAFE_MIN_DAYS
# deleteFiles=false — removes from qBit, leaves files for arrs to manage
# optional ratio requirement via QBIT_FAILSAFE_MIN_RATIO
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Never Interrupt Active Downloads
# Each downloader section checks for active state before any removal. slskd
# skips users with InProgress or Queued transfers. SABnzbd only removes items
# past the retention threshold. qBittorrent applies minimum age and optional
# ratio requirements. In-progress work is never touched.
#
# Graceful Skip on Unavailability
# If a downloader's URL is empty or the service is unreachable, that section
# skips cleanly with a log message. The script never exits fatally on a single
# unreachable downloader — the others still run.
#
# Host-Aware Configuration
# detect_hosts() aliases all HOST*_SLSKD_*, HOST*_SABNZBD_*, HOST*_QBIT_* vars
# to their unprefixed names. Downloaders not configured for this host are absent
# from the aliased vars and skip automatically.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Active Transfer Protection
# slskd: skips users with InProgress or Queued transfers before any removal.
# SABnzbd: age threshold enforced before deletion.
# qBittorrent: minimum age plus optional ratio gate before failsafe removal.
#
# Reachability Check
# Each section validates its downloader URL before API calls. Missing or
# unreachable downloaders skip without affecting other sections.
#
# Host Detection
# detect_hosts() identifies which server is running the script and aliases
# all HOST*_SLSKD_*, HOST*_SABNZBD_*, and HOST*_QBIT_* vars to the correct
# host's values. Downloaders not configured on this host skip automatically.
#
# Lock Acquisition
# acquire_lock "wait" — waits for previous run to finish since this runs every
# 15 minutes and prior execution may still be completing.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_SLSKD_URL / HOST*_SLSKD_API_KEY / HOST*_SLSKD_FAILED_IMPORTS_DIR
# slskd connection and failed imports path. Aliased by detect_hosts()
#
# HOST*_SABNZBD_URL / HOST*_SABNZBD_API_KEY
# SABnzbd connection details. Aliased by detect_hosts()
#
# HOST*_QBIT_URL / HOST*_QBIT_USERNAME / HOST*_QBIT_PASSWORD
# qBittorrent connection details. Aliased by detect_hosts()
#
# master.conf
#
# DOWNLOADER_RETENTION_DAYS
# Days before SABnzbd history entries (completed or failed) are removed
#
# QBIT_FAILSAFE_MIN_DAYS
# Minimum torrent age in days before failsafe removal is considered
#
# QBIT_FAILSAFE_MIN_RATIO
# Minimum seeding ratio required alongside age gate (0 = age only)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# downloaders_reset.sh
# Run maintenance reset for all configured download clients
#
# downloaders_reset.sh --dry-run
# Preview what would be removed without making any changes
#
# downloaders_reset.sh --status
# Show configured downloaders, current queue depths, and retention settings
#
# downloaders_reset.sh --log
# Verbose per-client per-item output
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
# Lock first — wait mode since this runs every 30min and previous may still be finishing
acquire_lock "wait"
# detect_hosts() sets MY_ID and aliases all HOST*_SLSKD_*, HOST*_SABNZBD_*, HOST*_QBIT_* vars
detect_hosts
START_TIME=$(date +%s)
CUTOFF=$(( $(date +%s) - (DOWNLOADER_RETENTION_DAYS * 86400) ))
TOTAL_PASS=0
TOTAL_FAIL=0
log "$ICON_GEAR Config: retention=${DOWNLOADER_RETENTION_DAYS}d qbit-age=${QBIT_FAILSAFE_MIN_DAYS}d qbit-ratio=${QBIT_FAILSAFE_MIN_RATIO}"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_GEAR slskd: ${SLSKD_URL:-not configured}"
echo "$ICON_GEAR SABnzbd: ${SABNZBD_URL:-not configured}"
echo "$ICON_GEAR qBittorrent: ${QBIT_URL:-not configured}"
echo "$ICON_TIME Retention: ${DOWNLOADER_RETENTION_DAYS} days"
echo "$ICON_GEAR qBit age: ${QBIT_FAILSAFE_MIN_DAYS} days"
echo "$ICON_GEAR qBit ratio: ${QBIT_FAILSAFE_MIN_RATIO} (0=age only)"
echo "$ICON_NOTIFY Notify: unRAID=${NOTIFY_UNRAID:-false} Discord=$([[ -n "${MY_DISCORD_WEBHOOK:-}" ]] && echo enabled || echo disabled)"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes will be made"
# Log which downloaders are active on this host
if [[ -z "$SLSKD_URL" ]] && [[ -z "$SABNZBD_URL" ]] && [[ -z "$QBIT_URL" ]]; then
warn "No downloaders configured for $MY_ID — nothing to reset"
exit 0
fi
[[ -n "$SLSKD_URL" ]] && log "slskd active on $MY_ID"
[[ -n "$SABNZBD_URL" ]] && log "SABnzbd active on $MY_ID"
[[ -n "$QBIT_URL" ]] && log "qBittorrent active on $MY_ID"
# ==============================================================================================
# ━━━ slskd — Connection Check ━━━
# ==============================================================================================
# slskd's internal watchdog doesn't always recover from disconnection. Check before
# running API-dependent sections; attempt reconnect if down.
SLSKD_CONNECTED=false
if [[ -n "$SLSKD_URL" ]] && [[ -n "$SLSKD_API_KEY" ]]; then
echo ""
echo "━━━ $ICON_SYNC slskd — Connection Check ━━━"
_slskd_is_connected() {
local state
state=$(curl -sf --max-time 10 \
-H "X-Api-Key: $SLSKD_API_KEY" \
"$SLSKD_URL/api/v0/application" 2>/dev/null | \
jq -r '.server.isConnected // false' 2>/dev/null)
[[ "$state" == "true" ]]
}
if _slskd_is_connected; then
log "slskd connected to Soulseek ✅"
SLSKD_CONNECTED=true
else
warn "slskd disconnected — triggering reconnect"
curl -sf --max-time 10 -X PUT \
-H "X-Api-Key: $SLSKD_API_KEY" \
-H "Content-Type: application/json" \
"$SLSKD_URL/api/v0/server" \
-d '{"address":"server.slsknet.org","port":2242}' \
>/dev/null 2>&1
_ELAPSED=0
while [[ "$_ELAPSED" -lt 60 ]]; do
sleep 10
_ELAPSED=$(( _ELAPSED + 10 ))
if _slskd_is_connected; then
log "slskd reconnected after ${_ELAPSED}s ✅"
SLSKD_CONNECTED=true
break
fi
log " waiting... (${_ELAPSED}s / 60s)"
done
[[ "$SLSKD_CONNECTED" != true ]] && \
warn "slskd still disconnected after 60s — skipping API-dependent sections"
fi
fi
# ==============================================================================================
# ━━━ slskd — Stuck Searches ━━━
# ==============================================================================================
# Clears searches in Completed/Errored state left by Soularr crashes.
# Prevents 409 Conflict error on next Soularr startup when it tries to
# create a search with the same ID that already exists in a terminal state.
if [[ -n "$SLSKD_URL" ]] && [[ -n "$SLSKD_API_KEY" ]] && [[ "$SLSKD_CONNECTED" == true ]]; then
echo ""
echo "━━━ 🔍 slskd — Stuck Searches ━━━"
SEARCHES=$(curl -sf --max-time 10 -X GET "$SLSKD_URL/api/v0/searches" \
-H "X-Api-Key: $SLSKD_API_KEY" 2>/dev/null)
if [[ -z "$SEARCHES" ]]; then
warn "slskd not reachable — skipping searches"
else
IDS=$(echo "$SEARCHES" | tr '{' '\n' | \
grep '"isComplete":true' | grep '"searchText":' | \
grep -o '"id":"[^"]*"' | sed 's/"id":"//;s/"//')
COUNT=$(echo "$IDS" | grep -c . 2>/dev/null || echo 0)
COUNT="${COUNT//[^0-9]/}"; COUNT="${COUNT:-0}"
if [[ "$COUNT" -eq 0 ]]; then
success "No stuck searches found ✅"
else
log "Found $COUNT stuck search(es)"
SUCCESS=0; FAIL=0
while IFS= read -r ID; do
[[ -z "$ID" ]] && continue
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would delete search: $ID"
((SUCCESS++))
continue
fi
RESULT=$(curl -sf --max-time 10 -o /dev/null -w "%{http_code}" -X DELETE \
"$SLSKD_URL/api/v0/searches/$ID" \
-H "X-Api-Key: $SLSKD_API_KEY")
if [[ "$RESULT" == "200" || "$RESULT" == "204" ]]; then
log "$ICON_TRASH Cleared search: $ID"
((SUCCESS++))
else
error "Failed: $ID (HTTP $RESULT)"
((FAIL++))
fi
done <<< "$IDS"
success "Searches: $SUCCESS cleared, $FAIL failed"
(( TOTAL_FAIL += FAIL ))
(( TOTAL_PASS += SUCCESS ))
fi
fi
fi
# ==============================================================================================
# ━━━ slskd — Dead Transfer Records ━━━
# ==============================================================================================
# Removes completed/errored/aborted transfer records per user.
# Prevents Soularr 404 loop when polling a user whose transfer no longer exists.
# Safety: NEVER removes transfers that are InProgress or Queued — active downloads protected.
if [[ -n "$SLSKD_URL" ]] && [[ -n "$SLSKD_API_KEY" ]] && [[ "$SLSKD_CONNECTED" == true ]]; then
echo ""
echo "━━━ 🔍 slskd — Dead Transfer Records ━━━"
TRANSFERS=$(curl -sf --max-time 10 -X GET "$SLSKD_URL/api/v0/transfers/downloads" \
-H "X-Api-Key: $SLSKD_API_KEY" 2>/dev/null)
if [[ -z "$TRANSFERS" ]]; then
warn "slskd not reachable — skipping transfers"
else
USERNAMES=$(echo "$TRANSFERS" | grep -o '"username":"[^"]*"' | \
sed 's/"username":"//;s/"//' | sort -u)
if [[ -z "$USERNAMES" ]]; then
success "No transfer records found ✅"
else
USER_COUNT=$(echo "$USERNAMES" | grep -c . 2>/dev/null || echo 0)
log "Found $USER_COUNT user(s) with transfer records"
SUCCESS=0; SKIPPED=0; FAIL=0
while IFS= read -r USER; do
[[ -z "$USER" ]] && continue
USER_DATA=$(curl -sf --max-time 10 \
"$SLSKD_URL/api/v0/transfers/downloads/$USER" \
-H "X-Api-Key: $SLSKD_API_KEY" 2>/dev/null)
# Skip users with any active or queued transfers — never interrupt downloads
ACTIVE=$(echo "$USER_DATA" | grep -c '"state":"InProgress"\|"state":"Queued"')
if [[ "${ACTIVE:-0}" -gt 0 ]]; then
log "$ICON_SKIP Skipping $USER — has active/queued transfer(s)"
((SKIPPED++))
continue
fi
# Extract IDs of terminal-state file transfers
# Split at { so each file object lands on its own line, then grep for state
FILE_IDS=$(echo "$USER_DATA" | tr '{' '\n' | \
grep '"state":"Completed"\|"state":"Errored"\|"state":"Aborted"\|"state":"Cancelled"' | \
grep -o '"id":"[^"]*"' | sed 's/"id":"//;s/"//')
if [[ -z "$FILE_IDS" ]]; then
log "$ICON_SKIP Skipping $USER — no terminal-state transfers"
((SKIPPED++))
continue
fi
if [[ "$DRY_RUN" == true ]]; then
F_COUNT=$(echo "$FILE_IDS" | grep -c .)
warn "DRY RUN — would clear $F_COUNT transfer(s) for: $USER"
((SUCCESS++))
continue
fi
F_SUCCESS=0; F_FAIL=0
while IFS= read -r FILE_ID; do
[[ -z "$FILE_ID" ]] && continue
RESULT=$(curl -sf --max-time 10 -o /dev/null -w "%{http_code}" -X DELETE \
"$SLSKD_URL/api/v0/transfers/downloads/$USER/$FILE_ID" \
-H "X-Api-Key: $SLSKD_API_KEY")
if [[ "$RESULT" == "200" || "$RESULT" == "204" ]]; then
((F_SUCCESS++))
else
((F_FAIL++))
fi
done <<< "$FILE_IDS"
log "$ICON_TRASH Cleared $F_SUCCESS transfer(s) for: $USER ($F_FAIL failed)"
((SUCCESS += F_SUCCESS))
((FAIL += F_FAIL))
done <<< "$USERNAMES"
success "Transfers: $SUCCESS cleared, $SKIPPED skipped (active/empty), $FAIL failed"
(( TOTAL_FAIL += FAIL ))
(( TOTAL_PASS += SUCCESS ))
fi
fi
fi
# ==============================================================================================
# ━━━ slskd — Purge Expired Failed Imports ━━━
# ==============================================================================================
# Removes albums Soularr downloaded but Lidarr rejected.
# Soularr moves rejected albums to failed_imports/ and never cleans them up.
# Purges directories older than DOWNLOADER_RETENTION_DAYS to prevent unbounded growth.
if [[ -n "$SLSKD_FAILED_IMPORTS_DIR" ]]; then
echo ""
echo "━━━ 🔍 slskd — Failed Imports (older than ${DOWNLOADER_RETENTION_DAYS} days) ━━━"
if [[ ! -d "$SLSKD_FAILED_IMPORTS_DIR" ]]; then
warn "Directory not found: $SLSKD_FAILED_IMPORTS_DIR — skipping"
else
OLD_IMPORTS=$(find "$SLSKD_FAILED_IMPORTS_DIR" \
-mindepth 1 -maxdepth 1 -mtime +"${DOWNLOADER_RETENTION_DAYS}")
IMPORT_COUNT=$(echo "$OLD_IMPORTS" | grep -c . 2>/dev/null || echo 0)
IMPORT_COUNT="${IMPORT_COUNT//[^0-9]/}"; IMPORT_COUNT="${IMPORT_COUNT:-0}"
if [[ "$IMPORT_COUNT" -eq 0 ]]; then
success "No expired failed imports found ✅"
else
log "Found $IMPORT_COUNT expired failed import(s)"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would delete:"
echo "$OLD_IMPORTS"
else
find "$SLSKD_FAILED_IMPORTS_DIR" \
-mindepth 1 -maxdepth 1 -mtime +"${DOWNLOADER_RETENTION_DAYS}" \
-exec rm -rf {} \;
success "$ICON_TRASH Purged $IMPORT_COUNT expired failed import(s)"
(( TOTAL_PASS += IMPORT_COUNT ))
fi
fi
fi
fi
# ==============================================================================================
# ━━━ SABnzbd — Clear Completed History ━━━
# ==============================================================================================
# Removes completed download history older than DOWNLOADER_RETENTION_DAYS.
# Keeps recent history for reference — only purges what's past the retention window.
if [[ -n "$SABNZBD_URL" ]] && [[ -n "$SABNZBD_API_KEY" ]]; then
echo ""
echo "━━━ 🔍 SABnzbd — Completed History (older than ${DOWNLOADER_RETENTION_DAYS} days) ━━━"
HISTORY=$(curl -sf --max-time 15 \
"$SABNZBD_URL/api?mode=history&output=json&limit=1000&apikey=$SABNZBD_API_KEY" 2>/dev/null)
if [[ -z "$HISTORY" ]]; then
warn "SABnzbd not reachable — skipping completed history"
else
COMPLETED_IDS=$(echo "$HISTORY" | grep -o '"nzo_id":"[^"]*"' | \
sed 's/"nzo_id":"//;s/"//')
if [[ -z "$COMPLETED_IDS" ]]; then
success "No completed history found ✅"
else
HIST_TOTAL=$(echo "$COMPLETED_IDS" | grep -c . 2>/dev/null || echo 0)
log "Found $HIST_TOTAL completed history entries"
DELETED=0; SKIPPED=0
while IFS= read -r NZO_ID; do
[[ -z "$NZO_ID" ]] && continue
JOB_TIME=$(echo "$HISTORY" | grep -A5 "$NZO_ID" | \
grep -o '"completed":[0-9]*' | grep -o '[0-9]*' | head -1)
[[ -z "$JOB_TIME" ]] && continue
[[ "$JOB_TIME" -gt "$CUTOFF" ]] && ((SKIPPED++)) && continue
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would delete completed job: $NZO_ID"
((DELETED++))
else
curl -sf --max-time 10 \
"$SABNZBD_URL/api?mode=history&name=delete&value=$NZO_ID&apikey=$SABNZBD_API_KEY" \
>/dev/null
log "$ICON_TRASH Deleted: $NZO_ID"
((DELETED++))
fi
done <<< "$COMPLETED_IDS"
success "Completed: $DELETED deleted, $SKIPPED within retention"
(( TOTAL_PASS += DELETED ))
fi
fi
fi
# ==============================================================================================
# ━━━ SABnzbd — Clear Failed History ━━━
# ==============================================================================================
# Removes failed download history older than DOWNLOADER_RETENTION_DAYS.
# Failed history is kept briefly for diagnosis but purged after the retention window.
if [[ -n "$SABNZBD_URL" ]] && [[ -n "$SABNZBD_API_KEY" ]]; then
echo ""
echo "━━━ 🔍 SABnzbd — Failed History (older than ${DOWNLOADER_RETENTION_DAYS} days) ━━━"
FAILED_HIST=$(curl -sf --max-time 15 \
"$SABNZBD_URL/api?mode=history&output=json&limit=1000&failed_only=1&apikey=$SABNZBD_API_KEY" 2>/dev/null)
if [[ -z "$FAILED_HIST" ]]; then
warn "SABnzbd not reachable — skipping failed history"
else
FAILED_IDS=$(echo "$FAILED_HIST" | grep -o '"nzo_id":"[^"]*"' | \
sed 's/"nzo_id":"//;s/"//')
if [[ -z "$FAILED_IDS" ]]; then
success "No failed history found ✅"
else
FAILED_TOTAL=$(echo "$FAILED_IDS" | grep -c . 2>/dev/null || echo 0)
log "Found $FAILED_TOTAL failed history entries"
DELETED=0; SKIPPED=0
while IFS= read -r NZO_ID; do
[[ -z "$NZO_ID" ]] && continue
JOB_TIME=$(echo "$FAILED_HIST" | grep -A5 "$NZO_ID" | \
grep -o '"completed":[0-9]*' | grep -o '[0-9]*' | head -1)
[[ -z "$JOB_TIME" ]] && continue
[[ "$JOB_TIME" -gt "$CUTOFF" ]] && ((SKIPPED++)) && continue
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would delete failed job: $NZO_ID"
((DELETED++))
else
curl -sf --max-time 10 \
"$SABNZBD_URL/api?mode=history&name=delete&value=$NZO_ID&apikey=$SABNZBD_API_KEY" \
>/dev/null
log "$ICON_TRASH Deleted: $NZO_ID"
((DELETED++))
fi
done <<< "$FAILED_IDS"
success "Failed: $DELETED deleted, $SKIPPED within retention"
(( TOTAL_PASS += DELETED ))
fi
fi
fi
# ==============================================================================================
# ━━━ SABnzbd — Remove Stalled Queue Items ━━━
# ==============================================================================================
# Removes queue items in Paused or Stuck state that are no longer progressing.
# Active downloading items (Downloading, Grabbing) are never touched.
# Paused items may be intentional pauses — but in an automated environment
# a Paused item sitting in the queue indefinitely is effectively stalled.
if [[ -n "$SABNZBD_URL" ]] && [[ -n "$SABNZBD_API_KEY" ]]; then
echo ""
echo "━━━ 🔍 SABnzbd — Stalled Queue Items ━━━"
QUEUE=$(curl -sf --max-time 10 \
"$SABNZBD_URL/api?mode=queue&output=json&apikey=$SABNZBD_API_KEY" 2>/dev/null)
if [[ -z "$QUEUE" ]]; then
warn "SABnzbd not reachable — skipping queue"
else
STALLED_IDS=$(echo "$QUEUE" | grep -o '"nzo_id":"[^"]*"' | \
sed 's/"nzo_id":"//;s/"//')
if [[ -z "$STALLED_IDS" ]]; then
success "No stalled queue items found ✅"
else
QUEUE_TOTAL=$(echo "$STALLED_IDS" | grep -c . 2>/dev/null || echo 0)
log "Found $QUEUE_TOTAL queue item(s) — checking status"
DELETED=0; SKIPPED=0
while IFS= read -r NZO_ID; do
[[ -z "$NZO_ID" ]] && continue
STATUS=$(echo "$QUEUE" | grep -A10 "$NZO_ID" | \
grep -o '"status":"[^"]*"' | sed 's/"status":"//;s/"//')
# Only remove Paused or Stuck items — Downloading/Grabbing are active
if [[ "$STATUS" != "Paused" ]] && [[ "$STATUS" != "Stuck" ]]; then
((SKIPPED++))
continue
fi
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would remove stalled item: $NZO_ID ($STATUS)"
((DELETED++))
else
curl -sf --max-time 10 \
"$SABNZBD_URL/api?mode=queue&name=delete&value=$NZO_ID&apikey=$SABNZBD_API_KEY" \
>/dev/null
log "$ICON_TRASH Removed stalled ($STATUS): $NZO_ID"
((DELETED++))
fi
done <<< "$STALLED_IDS"
success "Queue: $DELETED removed, $SKIPPED active (skipped)"
(( TOTAL_PASS += DELETED ))
fi
fi
fi
# ==============================================================================================
# ━━━ qBittorrent — Age Failsafe Cleanup ━━━
# ==============================================================================================
# Last-chance cleanup for torrents that have been sitting in qBit past their useful life.
# deleteFiles=false — removes the torrent record from qBit but leaves files on disk.
# Radarr/Sonarr manage actual files independently — this only cleans up the qBit entry.
#
# Safety checks before deletion:
# Age must exceed QBIT_FAILSAFE_MIN_DAYS
# Ratio must meet QBIT_FAILSAFE_MIN_RATIO (0 = age only, no ratio requirement)
if [[ -n "$QBIT_URL" ]] && [[ -n "$QBIT_USERNAME" ]]; then
echo ""
echo "━━━ 🔍 qBittorrent — Failsafe (older than ${QBIT_FAILSAFE_MIN_DAYS} days) ━━━"
[[ "$QBIT_FAILSAFE_MIN_RATIO" != "0" ]] && \
log "Ratio requirement: >= ${QBIT_FAILSAFE_MIN_RATIO}"
QBIT_COOKIE=$(curl -sf --max-time 10 -c - \
"$QBIT_URL/api/v2/auth/login" \
--data "username=$QBIT_USERNAME&password=$QBIT_PASSWORD" 2>/dev/null | \
grep SID | awk '{print "SID="$NF}')
if [[ -z "$QBIT_COOKIE" ]]; then
error "Failed to authenticate with qBittorrent — check QBIT_USERNAME/PASSWORD"
notify "qBittorrent auth failed on $(hostname) — check credentials in host*.conf" "Downloaders Reset" "warning"
((TOTAL_FAIL++))
else
TORRENTS=$(curl -sf --max-time 15 \
"$QBIT_URL/api/v2/torrents/info" \
-H "Cookie: $QBIT_COOKIE" 2>/dev/null)
NOW=$(date +%s)
TORRENT_TOTAL=$(echo "$TORRENTS" | tr '}' '\n' | grep -c '"hash"' 2>/dev/null || echo 0)
log "Found $TORRENT_TOTAL torrent(s) — applying age/ratio filter"
DELETED=0; SKIPPED=0
while read -r TORRENT; do
[[ -z "$TORRENT" ]] && continue
HASH=$(echo "$TORRENT" | grep -o '"hash":"[^"]*"' | sed 's/"hash":"//;s/"//')
NAME=$(echo "$TORRENT" | grep -o '"name":"[^"]*"' | sed 's/"name":"//;s/"//')
ADDED=$(echo "$TORRENT" | grep -o '"added_on":[0-9]*' | grep -o '[0-9]*')
RATIO=$(echo "$TORRENT" | grep -o '"ratio":[0-9.]*' | grep -o '[0-9.]*')
[[ -z "$HASH" || -z "$ADDED" ]] && continue
AGE_DAYS=$(( (NOW - ADDED) / 86400 ))
# Age check — must be old enough
[[ "$AGE_DAYS" -lt "$QBIT_FAILSAFE_MIN_DAYS" ]] && ((SKIPPED++)) && continue
# Ratio check — if configured
if [[ "$QBIT_FAILSAFE_MIN_RATIO" != "0" ]]; then
RATIO_INT="${RATIO%.*}"
MIN_RATIO_INT="${QBIT_FAILSAFE_MIN_RATIO%.*}"
[[ "$RATIO_INT" -lt "$MIN_RATIO_INT" ]] && ((SKIPPED++)) && continue
fi
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would delete: $NAME (${AGE_DAYS}d old, ratio: $RATIO)"
((DELETED++))
else
curl -sf --max-time 10 -X POST \
"$QBIT_URL/api/v2/torrents/delete" \
-H "Cookie: $QBIT_COOKIE" \
--data "hashes=$HASH&deleteFiles=false" >/dev/null
log "$ICON_TRASH Deleted: $NAME (${AGE_DAYS}d old, ratio: $RATIO)"
((DELETED++))
fi
done < <(echo "$TORRENTS" | tr '}' '\n')
success "qBittorrent: $DELETED deleted, $SKIPPED skipped (under threshold)"
(( TOTAL_PASS += DELETED ))
fi
fi
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY DOWNLOADERS RESET SUMMARY ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_TIME Duration: $(format_duration $(( $(date +%s) - START_TIME )))"
echo "$ICON_SUCCESS Actions: $TOTAL_PASS"
echo "$ICON_ERROR Failures: $TOTAL_FAIL"
echo ""
if [[ "$DRY_RUN" == true ]]; then
echo "$ICON_WARN Status: DRY RUN — no changes made"
elif [[ "$TOTAL_FAIL" -gt 0 ]]; then
echo "$ICON_ERROR Status: $TOTAL_FAIL failure(s) — check logs"
notify "Downloaders reset completed with failures on $(hostname)" "Downloaders Reset" "warning"
exit 1
else
echo "$ICON_DONE Status: $ICON_SUCCESS DONE"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
@@ -1,650 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ================================= Downloaders Reset ==========================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Maintenance reset for all download clients on this server. Clears accumulated
# state that download clients generate but never clean up themselves — stuck
# searches, dead transfers, failed imports, stale queue entries, completed history.
#
# Called every 30 minutes by critical_sync_maintenance.sh via
# CRITICAL_MAINTENANCE_SCRIPTS. Can also be run manually for ad hoc cleanup.
# If a downloader is not configured for this host, that section skips cleanly.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# slskd
# Stuck searches — clears Completed/Errored searches left by Soularr crashes
# prevents 409 Conflict on next Soularr startup
# Dead transfers — removes completed/errored/aborted transfer records per user
# prevents Soularr 404 loop when polling a user whose transfer is gone
# NEVER removes InProgress or Queued transfers
# Failed imports — purges albums Soularr downloaded but Lidarr rejected
# Soularr moves these to failed_imports/ and never cleans them up
#
# SABnzbd
# Completed history — removes completed download records older than DOWNLOADER_RETENTION_DAYS
# Failed history — removes failed download records older than DOWNLOADER_RETENTION_DAYS
# Stalled queue — removes Paused or Stuck queue items no longer progressing
# active downloading items are never touched
#
# qBittorrent
# Age failsafe — removes torrents older than QBIT_FAILSAFE_MIN_DAYS
# deleteFiles=false — removes from qBit, leaves files for arrs to manage
# optional ratio requirement via QBIT_FAILSAFE_MIN_RATIO
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Never Interrupt Active Downloads
# Each downloader section checks for active state before any removal. slskd
# skips users with InProgress or Queued transfers. SABnzbd only removes items
# past the retention threshold. qBittorrent applies minimum age and optional
# ratio requirements. In-progress work is never touched.
#
# Graceful Skip on Unavailability
# If a downloader's URL is empty or the service is unreachable, that section
# skips cleanly with a log message. The script never exits fatally on a single
# unreachable downloader — the others still run.
#
# Host-Aware Configuration
# detect_hosts() aliases all HOST*_SLSKD_*, HOST*_SABNZBD_*, HOST*_QBIT_* vars
# to their unprefixed names. Downloaders not configured for this host are absent
# from the aliased vars and skip automatically.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Active Transfer Protection
# slskd: skips users with InProgress or Queued transfers before any removal.
# SABnzbd: age threshold enforced before deletion.
# qBittorrent: minimum age plus optional ratio gate before failsafe removal.
#
# Reachability Check
# Each section validates its downloader URL before API calls. Missing or
# unreachable downloaders skip without affecting other sections.
#
# Host Detection
# detect_hosts() identifies which server is running the script and aliases
# all HOST*_SLSKD_*, HOST*_SABNZBD_*, and HOST*_QBIT_* vars to the correct
# host's values. Downloaders not configured on this host skip automatically.
#
# Lock Acquisition
# acquire_lock "wait" — waits for previous run to finish since this runs every
# 30 minutes and prior execution may still be completing.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_SLSKD_URL / HOST*_SLSKD_API_KEY / HOST*_SLSKD_FAILED_IMPORTS_DIR
# slskd connection and failed imports path. Aliased by detect_hosts()
#
# HOST*_SABNZBD_URL / HOST*_SABNZBD_API_KEY
# SABnzbd connection details. Aliased by detect_hosts()
#
# HOST*_QBIT_URL / HOST*_QBIT_USERNAME / HOST*_QBIT_PASSWORD
# qBittorrent connection details. Aliased by detect_hosts()
#
# master.conf
#
# DOWNLOADER_RETENTION_DAYS
# Days before SABnzbd history entries (completed or failed) are removed
#
# QBIT_FAILSAFE_MIN_DAYS
# Minimum torrent age in days before failsafe removal is considered
#
# QBIT_FAILSAFE_MIN_RATIO
# Minimum seeding ratio required alongside age gate (0 = age only)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# downloaders_reset.sh
# Run maintenance reset for all configured download clients
#
# downloaders_reset.sh --dry-run
# Preview what would be removed without making any changes
#
# downloaders_reset.sh --status
# Show configured downloaders, current queue depths, and retention settings
#
# downloaders_reset.sh --log
# Verbose per-client per-item output
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
# Lock first — wait mode since this runs every 30min and previous may still be finishing
acquire_lock "wait"
# detect_hosts() sets MY_ID and aliases all HOST*_SLSKD_*, HOST*_SABNZBD_*, HOST*_QBIT_* vars
detect_hosts
START_TIME=$(date +%s)
CUTOFF=$(( $(date +%s) - (DOWNLOADER_RETENTION_DAYS * 86400) ))
TOTAL_PASS=0
TOTAL_FAIL=0
log "$ICON_GEAR Config: retention=${DOWNLOADER_RETENTION_DAYS}d qbit-age=${QBIT_FAILSAFE_MIN_DAYS}d qbit-ratio=${QBIT_FAILSAFE_MIN_RATIO}"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_GEAR slskd: ${SLSKD_URL:-not configured}"
echo "$ICON_GEAR SABnzbd: ${SABNZBD_URL:-not configured}"
echo "$ICON_GEAR qBittorrent: ${QBIT_URL:-not configured}"
echo "$ICON_TIME Retention: ${DOWNLOADER_RETENTION_DAYS} days"
echo "$ICON_GEAR qBit age: ${QBIT_FAILSAFE_MIN_DAYS} days"
echo "$ICON_GEAR qBit ratio: ${QBIT_FAILSAFE_MIN_RATIO} (0=age only)"
echo "$ICON_NOTIFY Notify: unRAID=${NOTIFY_UNRAID:-false} Discord=$([[ -n "${MY_DISCORD_WEBHOOK:-}" ]] && echo enabled || echo disabled)"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes will be made"
# Log which downloaders are active on this host
if [[ -z "$SLSKD_URL" ]] && [[ -z "$SABNZBD_URL" ]] && [[ -z "$QBIT_URL" ]]; then
warn "No downloaders configured for $MY_ID — nothing to reset"
exit 0
fi
[[ -n "$SLSKD_URL" ]] && log "slskd active on $MY_ID"
[[ -n "$SABNZBD_URL" ]] && log "SABnzbd active on $MY_ID"
[[ -n "$QBIT_URL" ]] && log "qBittorrent active on $MY_ID"
# ==============================================================================================
# ━━━ slskd — Connection Check ━━━
# ==============================================================================================
# slskd's internal watchdog doesn't always recover from disconnection. Check before
# running API-dependent sections; attempt reconnect if down.
SLSKD_CONNECTED=false
if [[ -n "$SLSKD_URL" ]] && [[ -n "$SLSKD_API_KEY" ]]; then
echo ""
echo "━━━ $ICON_SYNC slskd — Connection Check ━━━"
_slskd_is_connected() {
local state
state=$(curl -sf --max-time 10 \
-H "X-Api-Key: $SLSKD_API_KEY" \
"$SLSKD_URL/api/v0/application" 2>/dev/null | \
jq -r '.server.isConnected // false' 2>/dev/null)
[[ "$state" == "true" ]]
}
if _slskd_is_connected; then
log "slskd connected to Soulseek ✅"
SLSKD_CONNECTED=true
else
warn "slskd disconnected — triggering reconnect"
curl -sf --max-time 10 -X PUT \
-H "X-Api-Key: $SLSKD_API_KEY" \
-H "Content-Type: application/json" \
"$SLSKD_URL/api/v0/server" \
-d '{"address":"server.slsknet.org","port":2242}' \
>/dev/null 2>&1
_ELAPSED=0
while [[ "$_ELAPSED" -lt 60 ]]; do
sleep 10
_ELAPSED=$(( _ELAPSED + 10 ))
if _slskd_is_connected; then
log "slskd reconnected after ${_ELAPSED}s ✅"
SLSKD_CONNECTED=true
break
fi
log " waiting... (${_ELAPSED}s / 60s)"
done
[[ "$SLSKD_CONNECTED" != true ]] && \
warn "slskd still disconnected after 60s — skipping API-dependent sections"
fi
fi
# ==============================================================================================
# ━━━ slskd — Stuck Searches ━━━
# ==============================================================================================
# Clears searches in Completed/Errored state left by Soularr crashes.
# Prevents 409 Conflict error on next Soularr startup when it tries to
# create a search with the same ID that already exists in a terminal state.
if [[ -n "$SLSKD_URL" ]] && [[ -n "$SLSKD_API_KEY" ]] && [[ "$SLSKD_CONNECTED" == true ]]; then
echo ""
echo "━━━ 🔍 slskd — Stuck Searches ━━━"
SEARCHES=$(curl -sf --max-time 10 -X GET "$SLSKD_URL/api/v0/searches" \
-H "X-Api-Key: $SLSKD_API_KEY" 2>/dev/null)
if [[ -z "$SEARCHES" ]]; then
warn "slskd not reachable — skipping searches"
else
IDS=$(echo "$SEARCHES" | tr '{' '\n' | \
grep '"isComplete":true' | grep '"searchText":' | \
grep -o '"id":"[^"]*"' | sed 's/"id":"//;s/"//')
COUNT=$(echo "$IDS" | grep -c . 2>/dev/null || echo 0)
COUNT="${COUNT//[^0-9]/}"; COUNT="${COUNT:-0}"
if [[ "$COUNT" -eq 0 ]]; then
success "No stuck searches found ✅"
else
log "Found $COUNT stuck search(es)"
SUCCESS=0; FAIL=0
while IFS= read -r ID; do
[[ -z "$ID" ]] && continue
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would delete search: $ID"
((SUCCESS++))
continue
fi
RESULT=$(curl -sf --max-time 10 -o /dev/null -w "%{http_code}" -X DELETE \
"$SLSKD_URL/api/v0/searches/$ID" \
-H "X-Api-Key: $SLSKD_API_KEY")
if [[ "$RESULT" == "200" || "$RESULT" == "204" ]]; then
log "$ICON_TRASH Cleared search: $ID"
((SUCCESS++))
else
error "Failed: $ID (HTTP $RESULT)"
((FAIL++))
fi
done <<< "$IDS"
success "Searches: $SUCCESS cleared, $FAIL failed"
(( TOTAL_FAIL += FAIL ))
(( TOTAL_PASS += SUCCESS ))
fi
fi
fi
# ==============================================================================================
# ━━━ slskd — Dead Transfer Records ━━━
# ==============================================================================================
# Removes completed/errored/aborted transfer records per user.
# Prevents Soularr 404 loop when polling a user whose transfer no longer exists.
# Safety: NEVER removes transfers that are InProgress or Queued — active downloads protected.
if [[ -n "$SLSKD_URL" ]] && [[ -n "$SLSKD_API_KEY" ]] && [[ "$SLSKD_CONNECTED" == true ]]; then
echo ""
echo "━━━ 🔍 slskd — Dead Transfer Records ━━━"
TRANSFERS=$(curl -sf --max-time 10 -X GET "$SLSKD_URL/api/v0/transfers/downloads" \
-H "X-Api-Key: $SLSKD_API_KEY" 2>/dev/null)
if [[ -z "$TRANSFERS" ]]; then
warn "slskd not reachable — skipping transfers"
else
USERNAMES=$(echo "$TRANSFERS" | grep -o '"username":"[^"]*"' | \
sed 's/"username":"//;s/"//' | sort -u)
if [[ -z "$USERNAMES" ]]; then
success "No transfer records found ✅"
else
USER_COUNT=$(echo "$USERNAMES" | grep -c . 2>/dev/null || echo 0)
log "Found $USER_COUNT user(s) with transfer records"
SUCCESS=0; SKIPPED=0; FAIL=0
while IFS= read -r USER; do
[[ -z "$USER" ]] && continue
USER_DATA=$(curl -sf --max-time 10 \
"$SLSKD_URL/api/v0/transfers/downloads/$USER" \
-H "X-Api-Key: $SLSKD_API_KEY" 2>/dev/null)
# Skip users with any active or queued transfers — never interrupt downloads
ACTIVE=$(echo "$USER_DATA" | grep -c '"state":"InProgress"\|"state":"Queued"')
if [[ "${ACTIVE:-0}" -gt 0 ]]; then
log "$ICON_SKIP Skipping $USER — has active/queued transfer(s)"
((SKIPPED++))
continue
fi
# Extract IDs of terminal-state file transfers
# Split at { so each file object lands on its own line, then grep for state
FILE_IDS=$(echo "$USER_DATA" | tr '{' '\n' | \
grep '"state":"Completed"\|"state":"Errored"\|"state":"Aborted"\|"state":"Cancelled"' | \
grep -o '"id":"[^"]*"' | sed 's/"id":"//;s/"//')
if [[ -z "$FILE_IDS" ]]; then
log "$ICON_SKIP Skipping $USER — no terminal-state transfers"
((SKIPPED++))
continue
fi
if [[ "$DRY_RUN" == true ]]; then
F_COUNT=$(echo "$FILE_IDS" | grep -c .)
warn "DRY RUN — would clear $F_COUNT transfer(s) for: $USER"
((SUCCESS++))
continue
fi
F_SUCCESS=0; F_FAIL=0
while IFS= read -r FILE_ID; do
[[ -z "$FILE_ID" ]] && continue
RESULT=$(curl -sf --max-time 10 -o /dev/null -w "%{http_code}" -X DELETE \
"$SLSKD_URL/api/v0/transfers/downloads/$USER/$FILE_ID" \
-H "X-Api-Key: $SLSKD_API_KEY")
if [[ "$RESULT" == "200" || "$RESULT" == "204" ]]; then
((F_SUCCESS++))
else
((F_FAIL++))
fi
done <<< "$FILE_IDS"
log "$ICON_TRASH Cleared $F_SUCCESS transfer(s) for: $USER ($F_FAIL failed)"
((SUCCESS += F_SUCCESS))
((FAIL += F_FAIL))
done <<< "$USERNAMES"
success "Transfers: $SUCCESS cleared, $SKIPPED skipped (active/empty), $FAIL failed"
(( TOTAL_FAIL += FAIL ))
(( TOTAL_PASS += SUCCESS ))
fi
fi
fi
# ==============================================================================================
# ━━━ slskd — Purge Expired Failed Imports ━━━
# ==============================================================================================
# Removes albums Soularr downloaded but Lidarr rejected.
# Soularr moves rejected albums to failed_imports/ and never cleans them up.
# Purges directories older than DOWNLOADER_RETENTION_DAYS to prevent unbounded growth.
if [[ -n "$SLSKD_FAILED_IMPORTS_DIR" ]]; then
echo ""
echo "━━━ 🔍 slskd — Failed Imports (older than ${DOWNLOADER_RETENTION_DAYS} days) ━━━"
if [[ ! -d "$SLSKD_FAILED_IMPORTS_DIR" ]]; then
warn "Directory not found: $SLSKD_FAILED_IMPORTS_DIR — skipping"
else
OLD_IMPORTS=$(find "$SLSKD_FAILED_IMPORTS_DIR" \
-mindepth 1 -maxdepth 1 -mtime +"${DOWNLOADER_RETENTION_DAYS}")
IMPORT_COUNT=$(echo "$OLD_IMPORTS" | grep -c . 2>/dev/null || echo 0)
IMPORT_COUNT="${IMPORT_COUNT//[^0-9]/}"; IMPORT_COUNT="${IMPORT_COUNT:-0}"
if [[ "$IMPORT_COUNT" -eq 0 ]]; then
success "No expired failed imports found ✅"
else
log "Found $IMPORT_COUNT expired failed import(s)"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would delete:"
echo "$OLD_IMPORTS"
else
find "$SLSKD_FAILED_IMPORTS_DIR" \
-mindepth 1 -maxdepth 1 -mtime +"${DOWNLOADER_RETENTION_DAYS}" \
-exec rm -rf {} \;
success "$ICON_TRASH Purged $IMPORT_COUNT expired failed import(s)"
(( TOTAL_PASS += IMPORT_COUNT ))
fi
fi
fi
fi
# ==============================================================================================
# ━━━ SABnzbd — Clear Completed History ━━━
# ==============================================================================================
# Removes completed download history older than DOWNLOADER_RETENTION_DAYS.
# Keeps recent history for reference — only purges what's past the retention window.
if [[ -n "$SABNZBD_URL" ]] && [[ -n "$SABNZBD_API_KEY" ]]; then
echo ""
echo "━━━ 🔍 SABnzbd — Completed History (older than ${DOWNLOADER_RETENTION_DAYS} days) ━━━"
HISTORY=$(curl -sf --max-time 15 \
"$SABNZBD_URL/api?mode=history&output=json&limit=1000&apikey=$SABNZBD_API_KEY" 2>/dev/null)
if [[ -z "$HISTORY" ]]; then
warn "SABnzbd not reachable — skipping completed history"
else
COMPLETED_IDS=$(echo "$HISTORY" | grep -o '"nzo_id":"[^"]*"' | \
sed 's/"nzo_id":"//;s/"//')
if [[ -z "$COMPLETED_IDS" ]]; then
success "No completed history found ✅"
else
HIST_TOTAL=$(echo "$COMPLETED_IDS" | grep -c . 2>/dev/null || echo 0)
log "Found $HIST_TOTAL completed history entries"
DELETED=0; SKIPPED=0
while IFS= read -r NZO_ID; do
[[ -z "$NZO_ID" ]] && continue
JOB_TIME=$(echo "$HISTORY" | grep -A5 "$NZO_ID" | \
grep -o '"completed":[0-9]*' | grep -o '[0-9]*' | head -1)
[[ -z "$JOB_TIME" ]] && continue
[[ "$JOB_TIME" -gt "$CUTOFF" ]] && ((SKIPPED++)) && continue
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would delete completed job: $NZO_ID"
((DELETED++))
else
curl -sf --max-time 10 \
"$SABNZBD_URL/api?mode=history&name=delete&value=$NZO_ID&apikey=$SABNZBD_API_KEY" \
>/dev/null
log "$ICON_TRASH Deleted: $NZO_ID"
((DELETED++))
fi
done <<< "$COMPLETED_IDS"
success "Completed: $DELETED deleted, $SKIPPED within retention"
(( TOTAL_PASS += DELETED ))
fi
fi
fi
# ==============================================================================================
# ━━━ SABnzbd — Clear Failed History ━━━
# ==============================================================================================
# Removes failed download history older than DOWNLOADER_RETENTION_DAYS.
# Failed history is kept briefly for diagnosis but purged after the retention window.
if [[ -n "$SABNZBD_URL" ]] && [[ -n "$SABNZBD_API_KEY" ]]; then
echo ""
echo "━━━ 🔍 SABnzbd — Failed History (older than ${DOWNLOADER_RETENTION_DAYS} days) ━━━"
FAILED_HIST=$(curl -sf --max-time 15 \
"$SABNZBD_URL/api?mode=history&output=json&limit=1000&failed_only=1&apikey=$SABNZBD_API_KEY" 2>/dev/null)
if [[ -z "$FAILED_HIST" ]]; then
warn "SABnzbd not reachable — skipping failed history"
else
FAILED_IDS=$(echo "$FAILED_HIST" | grep -o '"nzo_id":"[^"]*"' | \
sed 's/"nzo_id":"//;s/"//')
if [[ -z "$FAILED_IDS" ]]; then
success "No failed history found ✅"
else
FAILED_TOTAL=$(echo "$FAILED_IDS" | grep -c . 2>/dev/null || echo 0)
log "Found $FAILED_TOTAL failed history entries"
DELETED=0; SKIPPED=0
while IFS= read -r NZO_ID; do
[[ -z "$NZO_ID" ]] && continue
JOB_TIME=$(echo "$FAILED_HIST" | grep -A5 "$NZO_ID" | \
grep -o '"completed":[0-9]*' | grep -o '[0-9]*' | head -1)
[[ -z "$JOB_TIME" ]] && continue
[[ "$JOB_TIME" -gt "$CUTOFF" ]] && ((SKIPPED++)) && continue
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would delete failed job: $NZO_ID"
((DELETED++))
else
curl -sf --max-time 10 \
"$SABNZBD_URL/api?mode=history&name=delete&value=$NZO_ID&apikey=$SABNZBD_API_KEY" \
>/dev/null
log "$ICON_TRASH Deleted: $NZO_ID"
((DELETED++))
fi
done <<< "$FAILED_IDS"
success "Failed: $DELETED deleted, $SKIPPED within retention"
(( TOTAL_PASS += DELETED ))
fi
fi
fi
# ==============================================================================================
# ━━━ SABnzbd — Remove Stalled Queue Items ━━━
# ==============================================================================================
# Removes queue items in Paused or Stuck state that are no longer progressing.
# Active downloading items (Downloading, Grabbing) are never touched.
# Paused items may be intentional pauses — but in an automated environment
# a Paused item sitting in the queue indefinitely is effectively stalled.
if [[ -n "$SABNZBD_URL" ]] && [[ -n "$SABNZBD_API_KEY" ]]; then
echo ""
echo "━━━ 🔍 SABnzbd — Stalled Queue Items ━━━"
QUEUE=$(curl -sf --max-time 10 \
"$SABNZBD_URL/api?mode=queue&output=json&apikey=$SABNZBD_API_KEY" 2>/dev/null)
if [[ -z "$QUEUE" ]]; then
warn "SABnzbd not reachable — skipping queue"
else
STALLED_IDS=$(echo "$QUEUE" | grep -o '"nzo_id":"[^"]*"' | \
sed 's/"nzo_id":"//;s/"//')
if [[ -z "$STALLED_IDS" ]]; then
success "No stalled queue items found ✅"
else
QUEUE_TOTAL=$(echo "$STALLED_IDS" | grep -c . 2>/dev/null || echo 0)
log "Found $QUEUE_TOTAL queue item(s) — checking status"
DELETED=0; SKIPPED=0
while IFS= read -r NZO_ID; do
[[ -z "$NZO_ID" ]] && continue
STATUS=$(echo "$QUEUE" | grep -A10 "$NZO_ID" | \
grep -o '"status":"[^"]*"' | sed 's/"status":"//;s/"//')
# Only remove Paused or Stuck items — Downloading/Grabbing are active
if [[ "$STATUS" != "Paused" ]] && [[ "$STATUS" != "Stuck" ]]; then
((SKIPPED++))
continue
fi
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would remove stalled item: $NZO_ID ($STATUS)"
((DELETED++))
else
curl -sf --max-time 10 \
"$SABNZBD_URL/api?mode=queue&name=delete&value=$NZO_ID&apikey=$SABNZBD_API_KEY" \
>/dev/null
log "$ICON_TRASH Removed stalled ($STATUS): $NZO_ID"
((DELETED++))
fi
done <<< "$STALLED_IDS"
success "Queue: $DELETED removed, $SKIPPED active (skipped)"
(( TOTAL_PASS += DELETED ))
fi
fi
fi
# ==============================================================================================
# ━━━ qBittorrent — Age Failsafe Cleanup ━━━
# ==============================================================================================
# Last-chance cleanup for torrents that have been sitting in qBit past their useful life.
# deleteFiles=false — removes the torrent record from qBit but leaves files on disk.
# Radarr/Sonarr manage actual files independently — this only cleans up the qBit entry.
#
# Safety checks before deletion:
# Age must exceed QBIT_FAILSAFE_MIN_DAYS
# Ratio must meet QBIT_FAILSAFE_MIN_RATIO (0 = age only, no ratio requirement)
if [[ -n "$QBIT_URL" ]] && [[ -n "$QBIT_USERNAME" ]]; then
echo ""
echo "━━━ 🔍 qBittorrent — Failsafe (older than ${QBIT_FAILSAFE_MIN_DAYS} days) ━━━"
[[ "$QBIT_FAILSAFE_MIN_RATIO" != "0" ]] && \
log "Ratio requirement: >= ${QBIT_FAILSAFE_MIN_RATIO}"
QBIT_COOKIE=$(curl -sf --max-time 10 -c - \
"$QBIT_URL/api/v2/auth/login" \
--data "username=$QBIT_USERNAME&password=$QBIT_PASSWORD" 2>/dev/null | \
grep SID | awk '{print "SID="$NF}')
if [[ -z "$QBIT_COOKIE" ]]; then
error "Failed to authenticate with qBittorrent — check QBIT_USERNAME/PASSWORD"
notify "qBittorrent auth failed on $(hostname) — check credentials in host*.conf" "Downloaders Reset" "warning"
((TOTAL_FAIL++))
else
TORRENTS=$(curl -sf --max-time 15 \
"$QBIT_URL/api/v2/torrents/info" \
-H "Cookie: $QBIT_COOKIE" 2>/dev/null)
NOW=$(date +%s)
TORRENT_TOTAL=$(echo "$TORRENTS" | tr '}' '\n' | grep -c '"hash"' 2>/dev/null || echo 0)
log "Found $TORRENT_TOTAL torrent(s) — applying age/ratio filter"
DELETED=0; SKIPPED=0
while read -r TORRENT; do
[[ -z "$TORRENT" ]] && continue
HASH=$(echo "$TORRENT" | grep -o '"hash":"[^"]*"' | sed 's/"hash":"//;s/"//')
NAME=$(echo "$TORRENT" | grep -o '"name":"[^"]*"' | sed 's/"name":"//;s/"//')
ADDED=$(echo "$TORRENT" | grep -o '"added_on":[0-9]*' | grep -o '[0-9]*')
RATIO=$(echo "$TORRENT" | grep -o '"ratio":[0-9.]*' | grep -o '[0-9.]*')
[[ -z "$HASH" || -z "$ADDED" ]] && continue
AGE_DAYS=$(( (NOW - ADDED) / 86400 ))
# Age check — must be old enough
[[ "$AGE_DAYS" -lt "$QBIT_FAILSAFE_MIN_DAYS" ]] && ((SKIPPED++)) && continue
# Ratio check — if configured
if [[ "$QBIT_FAILSAFE_MIN_RATIO" != "0" ]]; then
RATIO_INT="${RATIO%.*}"
MIN_RATIO_INT="${QBIT_FAILSAFE_MIN_RATIO%.*}"
[[ "$RATIO_INT" -lt "$MIN_RATIO_INT" ]] && ((SKIPPED++)) && continue
fi
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would delete: $NAME (${AGE_DAYS}d old, ratio: $RATIO)"
((DELETED++))
else
curl -sf --max-time 10 -X POST \
"$QBIT_URL/api/v2/torrents/delete" \
-H "Cookie: $QBIT_COOKIE" \
--data "hashes=$HASH&deleteFiles=false" >/dev/null
log "$ICON_TRASH Deleted: $NAME (${AGE_DAYS}d old, ratio: $RATIO)"
((DELETED++))
fi
done < <(echo "$TORRENTS" | tr '}' '\n')
success "qBittorrent: $DELETED deleted, $SKIPPED skipped (under threshold)"
(( TOTAL_PASS += DELETED ))
fi
fi
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY DOWNLOADERS RESET SUMMARY ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_TIME Duration: $(format_duration $(( $(date +%s) - START_TIME )))"
echo "$ICON_SUCCESS Actions: $TOTAL_PASS"
echo "$ICON_ERROR Failures: $TOTAL_FAIL"
echo ""
if [[ "$DRY_RUN" == true ]]; then
echo "$ICON_WARN Status: DRY RUN — no changes made"
elif [[ "$TOTAL_FAIL" -gt 0 ]]; then
echo "$ICON_ERROR Status: $TOTAL_FAIL failure(s) — check logs"
notify "Downloaders reset completed with failures on $(hostname)" "Downloaders Reset" "warning"
exit 1
else
echo "$ICON_DONE Status: $ICON_SUCCESS DONE"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
@@ -1,330 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= Backup Verify ==================================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# rsync mirror integrity verification via independent MD5 checksums. Scheduled
# weekly (Sunday 10am). Randomly samples BACKUP_VERIFY_SAMPLE files per share
# above BACKUP_VERIFY_MIN_SIZE, computes checksums locally, then computes the
# same checksums on the remote via SSH and compares.
#
# Per file: MATCH (checksums identical) | MISMATCH (file exists on both but
# checksums differ — sync failure or corruption) | MISSING (file exists locally
# but not on remote). All MISMATCHes and significant MISSINGs trigger notification.
# rsync exit code 0 is not trusted — this script verifies actual content.
#
# Share list from HOST*_BACKUP_VERIFY_SHARES if defined, otherwise falls back
# to HOST*_DAILY_SYNC_SHARES. Both aliased by detect_hosts().
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Independent Verification
# rsync reports success when the transfer completed without network errors and
# file sizes and modification times match. It does not detect silent corruption
# during transfer (bitflip in transit), corruption written to storage at rest
# (faulty drive sector), or files that matched size/mtime but had wrong content.
# All of these produce exit code 0. This script checks whether "done" means "correct."
#
# Intentionally Small Sample
# 10 files per share (default) — a spot check, not an exhaustive verify.
# Catches systematic problems and hardware issues while running in minutes, not
# hours. Full verification would take longer than the rsync itself.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Single Instance Lock
# acquire_lock prevents concurrent runs producing conflicting results.
#
# Remote Connectivity Check
# check_connectivity() verifies the remote Tailscale IP is reachable before
# any SSH calls. Without this, all files show as MISSING on a network hiccup.
#
# Remote Array Check
# check_remote_array() verifies /mnt/user is mounted on the remote before
# computing checksums. Array not started = all files "missing" = false alarm.
#
# Version Parity
# Refuses to run if remote unRAID version doesn't match local. A mismatch
# may mean the remote is in an unexpected state.
#
# SSH Timeout
# SSH_TIMEOUT caps all SSH calls. One hung connection does not block the run.
#
# Notification Validated
# platform_require_cmd confirms the notify script is present before use.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_BACKUP_VERIFY_SHARES
# Shares to verify. Leave empty to use HOST*_DAILY_SYNC_SHARES automatically.
# Aliased by detect_hosts() → BACKUP_VERIFY_SHARES.
#
# HOST*_DAILY_SYNC_SHARES
# Fallback share list if BACKUP_VERIFY_SHARES is empty. Aliased by detect_hosts().
#
# master.conf
#
# BACKUP_VERIFY_SAMPLE
# Random files checked per share per run. (default: 10)
#
# BACKUP_VERIFY_MIN_SIZE
# Minimum file size to include in sample — tiny files have low corruption
# risk and slow checksums. (default: 1M)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# backup_verify.sh
# Sample files from all shares and compare checksums. Notify on MISMATCH
# or significant MISSING count. Silent when all samples match.
#
# backup_verify.sh --dry-run
# Show which files would be sampled. No checksums computed, no notifications.
#
# backup_verify.sh --status
# Show share list, sample size, and min file size configuration. Then exit.
#
# backup_verify.sh --log
# Verbose per-file checksum comparison output during the run.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
SSH_TIMEOUT=15
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_GEAR Setup ━━━"
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
acquire_lock
# detect_hosts() sets MY_ID and aliases BACKUP_VERIFY_SHARES + DAILY_SYNC_SHARES
detect_hosts
# Share selection — configured list or fallback to daily sync shares
if [[ ${#BACKUP_VERIFY_SHARES[@]} -gt 0 ]]; then
VERIFY_SHARES=("${BACKUP_VERIFY_SHARES[@]}")
log "Using BACKUP_VERIFY_SHARES (${#VERIFY_SHARES[@]} shares)"
else
VERIFY_SHARES=("${DAILY_SYNC_SHARES[@]}")
log "BACKUP_VERIFY_SHARES not set — using DAILY_SYNC_SHARES (${#VERIFY_SHARES[@]} shares)"
fi
if [[ ${#VERIFY_SHARES[@]} -eq 0 ]]; then
warn "No shares configured for $MY_ID — nothing to verify"
warn "Check HOST*_BACKUP_VERIFY_SHARES or HOST*_DAILY_SYNC_SHARES in host*.conf"
exit 0
fi
log "$ICON_GEAR Config: sample=${BACKUP_VERIFY_SAMPLE} min-size=${BACKUP_VERIFY_MIN_SIZE} ssh-timeout=${SSH_TIMEOUT}s"
log "$ICON_GEAR Remote: $REMOTE_ID ($REMOTE_SERVER_NAME — $REMOTE_SERVER)"
log "$ICON_GEAR Shares: ${VERIFY_SHARES[*]}"
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — showing sample selection only, no checksums computed"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST My ID: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_HOST Remote: $REMOTE_ID ($REMOTE_SERVER_NAME — $REMOTE_SERVER)"
echo "$ICON_VERIFY Shares: ${#VERIFY_SHARES[@]}"
echo "$ICON_VERIFY Sample: $BACKUP_VERIFY_SAMPLE files per share"
echo "$ICON_VERIFY Min size: $BACKUP_VERIFY_MIN_SIZE"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
echo ""
echo " Shares to verify:"
for share in "${VERIFY_SHARES[@]}"; do
echo " $share"
done
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ━━━ Pre-flight ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SHIELD Pre-flight ━━━"
resolve_remote_ip
# Connectivity — no point making 100+ SSH calls if remote is unreachable
check_connectivity
log "Connectivity to $REMOTE_SERVER_NAME ✅"
# Version parity — mismatched unRAID could cause md5sum path differences
check_unraid_version_parity || {
warn "Version parity check failed — proceeding with caution"
warn "Checksum results may be unreliable if md5sum path changed between versions"
}
log "Version parity with $REMOTE_SERVER_NAME ✅"
# Remote array — if array is down all files appear "missing" = false alarm
if ! check_remote_array; then
error "Remote array not mounted on $REMOTE_SERVER_NAME"
error "All files would appear as MISSING — aborting to prevent false alarm"
notify "Backup verify aborted on $(hostname) — remote array not mounted on $REMOTE_SERVER_NAME" \
"Backup Verify" "warning"
exit 1
fi
log "Remote array mounted on $REMOTE_SERVER_NAME ✅"
echo "Pre-flight passed ✅"
# ==============================================================================================
# ━━━ Backup Verification ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_VERIFY Backup Verification — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
echo "$ICON_HOST $MY_ID ($LOCAL_SERVER_NAME) → $REMOTE_ID ($REMOTE_SERVER_NAME)"
echo "$ICON_VERIFY Sample: $BACKUP_VERIFY_SAMPLE files per share (min: $BACKUP_VERIFY_MIN_SIZE)"
echo ""
START=$(date +%s)
TOTAL_CHECKED=0
TOTAL_MATCH=0
TOTAL_MISMATCH=0
TOTAL_MISSING=0
SHARES_WITH_ISSUES=()
for share in "${VERIFY_SHARES[@]}"; do
SHARE_NAME=$(basename "$share")
echo "━━━ $ICON_VERIFY $SHARE_NAME ━━━"
if [[ ! -d "$share" ]]; then
warn "$SHARE_NAME not found locally — skipping"
echo ""
continue
fi
# Sample random files above minimum size
mapfile -t SAMPLE_FILES < <(
find "$share" -type f -size +"$BACKUP_VERIFY_MIN_SIZE" 2>/dev/null | \
shuf | head -n "$BACKUP_VERIFY_SAMPLE"
)
if [[ ${#SAMPLE_FILES[@]} -eq 0 ]]; then
log "$SHARE_NAME — no files found above $BACKUP_VERIFY_MIN_SIZE"
echo ""
continue
fi
log "$SHARE_NAME — sampled ${#SAMPLE_FILES[@]} files"
if [[ "$DRY_RUN" == true ]]; then
for f in "${SAMPLE_FILES[@]}"; do
warn "DRY RUN — would check: $(basename "$f")"
done
echo ""
continue
fi
SHARE_MATCH=0
SHARE_MISMATCH=0
SHARE_MISSING=0
for local_file in "${SAMPLE_FILES[@]}"; do
[[ -z "$local_file" ]] && continue
# Local checksum
local_md5=$(md5sum "$local_file" 2>/dev/null | awk '{print $1}')
if [[ -z "$local_md5" ]]; then
warn "Could not checksum locally: $(basename "$local_file") — skipping"
continue
fi
# Remote checksum via SSH — timeout protected
remote_md5=$(timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" \
-o StrictHostKeyChecking=no \
root@"$REMOTE_SERVER" \
"md5sum '$local_file' 2>/dev/null | awk '{print \$1}'" 2>/dev/null)
(( TOTAL_CHECKED++ ))
if [[ -z "$remote_md5" ]]; then
warn "$ICON_ERROR MISSING: $(basename "$local_file")"
(( SHARE_MISSING++ ))
(( TOTAL_MISSING++ ))
elif [[ "$local_md5" == "$remote_md5" ]]; then
log "MATCH: $(basename "$local_file")"
(( SHARE_MATCH++ ))
(( TOTAL_MATCH++ ))
else
error "MISMATCH: $(basename "$local_file")"
error " local: $local_md5"
error " remote: $remote_md5"
(( SHARE_MISMATCH++ ))
(( TOTAL_MISMATCH++ ))
fi
done
# Per-share result — only visible if issues found
if [[ "$SHARE_MISMATCH" -gt 0 || "$SHARE_MISSING" -gt 0 ]]; then
warn "$SHARE_NAME — match: $SHARE_MATCH missing: $SHARE_MISSING mismatch: $SHARE_MISMATCH"
SHARES_WITH_ISSUES+=("$SHARE_NAME")
else
log "$SHARE_NAME — all $SHARE_MATCH files match ✅"
fi
echo ""
done
END=$(date +%s)
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo "━━━━━ $ICON_SUMMARY BACKUP VERIFY SUMMARY ━━━━━"
echo "$ICON_HOST My ID: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_HOST Remote: $REMOTE_ID ($REMOTE_SERVER_NAME)"
echo "$ICON_VERIFY Checked: $TOTAL_CHECKED files"
echo "$ICON_TIME Duration: $(format_duration $(( END - START )))"
echo ""
if [[ "$TOTAL_MISMATCH" -gt 0 || "$TOTAL_MISSING" -gt 0 ]]; then
echo "$ICON_SUCCESS Match: $TOTAL_MATCH"
warn "Missing: $TOTAL_MISSING"
[[ "$TOTAL_MISMATCH" -gt 0 ]] && echo "$ICON_ERROR Mismatch: $TOTAL_MISMATCH"
fi
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no checksums computed"
elif [[ "$TOTAL_MISMATCH" -gt 0 || "$TOTAL_MISSING" -gt 0 ]]; then
echo "$ICON_ERROR Status: ISSUES FOUND — ${#SHARES_WITH_ISSUES[@]} share(s) need attention: ${SHARES_WITH_ISSUES[*]}"
notify "Backup verify FAILED on $(hostname) → $REMOTE_SERVER_NAME — mismatches: $TOTAL_MISMATCH missing: $TOTAL_MISSING — shares: ${SHARES_WITH_ISSUES[*]}" \
"Backup Verify" "warning"
else
echo "$ICON_DONE Status: all $TOTAL_CHECKED files match across ${#VERIFY_SHARES[@]} shares ✅"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
[[ "$TOTAL_MISMATCH" -gt 0 ]] && exit 1
exit 0
@@ -1,330 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= Backup Verify ==================================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# rsync mirror integrity verification via independent MD5 checksums. Scheduled
# weekly (Sunday 10am). Randomly samples BACKUP_VERIFY_SAMPLE files per share
# above BACKUP_VERIFY_MIN_SIZE, computes checksums locally, then computes the
# same checksums on the remote via SSH and compares.
#
# Per file: MATCH (checksums identical) | MISMATCH (file exists on both but
# checksums differ — sync failure or corruption) | MISSING (file exists locally
# but not on remote). All MISMATCHes and significant MISSINGs trigger notification.
# rsync exit code 0 is not trusted — this script verifies actual content.
#
# Share list from HOST*_BACKUP_VERIFY_SHARES if defined, otherwise falls back
# to HOST*_DAILY_SYNC_SHARES. Both aliased by detect_hosts().
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Independent Verification
# rsync reports success when the transfer completed without network errors and
# file sizes and modification times match. It does not detect silent corruption
# during transfer (bitflip in transit), corruption written to storage at rest
# (faulty drive sector), or files that matched size/mtime but had wrong content.
# All of these produce exit code 0. This script checks whether "done" means "correct."
#
# Intentionally Small Sample
# 10 files per share (default) — a spot check, not an exhaustive verify.
# Catches systematic problems and hardware issues while running in minutes, not
# hours. Full verification would take longer than the rsync itself.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Single Instance Lock
# acquire_lock prevents concurrent runs producing conflicting results.
#
# Remote Connectivity Check
# check_connectivity() verifies the remote Tailscale IP is reachable before
# any SSH calls. Without this, all files show as MISSING on a network hiccup.
#
# Remote Array Check
# check_remote_array() verifies /mnt/user is mounted on the remote before
# computing checksums. Array not started = all files "missing" = false alarm.
#
# Version Parity
# Refuses to run if remote unRAID version doesn't match local. A mismatch
# may mean the remote is in an unexpected state.
#
# SSH Timeout
# SSH_TIMEOUT caps all SSH calls. One hung connection does not block the run.
#
# Notification Validated
# platform_require_cmd confirms the notify script is present before use.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_BACKUP_VERIFY_SHARES
# Shares to verify. Leave empty to use HOST*_DAILY_SYNC_SHARES automatically.
# Aliased by detect_hosts() → BACKUP_VERIFY_SHARES.
#
# HOST*_DAILY_SYNC_SHARES
# Fallback share list if BACKUP_VERIFY_SHARES is empty. Aliased by detect_hosts().
#
# master.conf
#
# BACKUP_VERIFY_SAMPLE
# Random files checked per share per run. (default: 10)
#
# BACKUP_VERIFY_MIN_SIZE
# Minimum file size to include in sample — tiny files have low corruption
# risk and slow checksums. (default: 1M)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# backup_verify.sh
# Sample files from all shares and compare checksums. Notify on MISMATCH
# or significant MISSING count. Silent when all samples match.
#
# backup_verify.sh --dry-run
# Show which files would be sampled. No checksums computed, no notifications.
#
# backup_verify.sh --status
# Show share list, sample size, and min file size configuration. Then exit.
#
# backup_verify.sh --log
# Verbose per-file checksum comparison output during the run.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
SSH_TIMEOUT=15
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_GEAR Setup ━━━"
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
acquire_lock
# detect_hosts() sets MY_ID and aliases BACKUP_VERIFY_SHARES + DAILY_SYNC_SHARES
detect_hosts
# Share selection — configured list or fallback to daily sync shares
if [[ ${#BACKUP_VERIFY_SHARES[@]} -gt 0 ]]; then
VERIFY_SHARES=("${BACKUP_VERIFY_SHARES[@]}")
log "Using BACKUP_VERIFY_SHARES (${#VERIFY_SHARES[@]} shares)"
else
VERIFY_SHARES=("${DAILY_SYNC_SHARES[@]}")
log "BACKUP_VERIFY_SHARES not set — using DAILY_SYNC_SHARES (${#VERIFY_SHARES[@]} shares)"
fi
if [[ ${#VERIFY_SHARES[@]} -eq 0 ]]; then
warn "No shares configured for $MY_ID — nothing to verify"
warn "Check HOST*_BACKUP_VERIFY_SHARES or HOST*_DAILY_SYNC_SHARES in host*.conf"
exit 0
fi
log "$ICON_GEAR Config: sample=${BACKUP_VERIFY_SAMPLE} min-size=${BACKUP_VERIFY_MIN_SIZE} ssh-timeout=${SSH_TIMEOUT}s"
log "$ICON_GEAR Remote: $REMOTE_ID ($REMOTE_SERVER_NAME — $REMOTE_SERVER)"
log "$ICON_GEAR Shares: ${VERIFY_SHARES[*]}"
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — showing sample selection only, no checksums computed"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST My ID: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_HOST Remote: $REMOTE_ID ($REMOTE_SERVER_NAME — $REMOTE_SERVER)"
echo "$ICON_VERIFY Shares: ${#VERIFY_SHARES[@]}"
echo "$ICON_VERIFY Sample: $BACKUP_VERIFY_SAMPLE files per share"
echo "$ICON_VERIFY Min size: $BACKUP_VERIFY_MIN_SIZE"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
echo ""
echo " Shares to verify:"
for share in "${VERIFY_SHARES[@]}"; do
echo " $share"
done
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ━━━ Pre-flight ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SHIELD Pre-flight ━━━"
resolve_remote_ip
# Connectivity — no point making 100+ SSH calls if remote is unreachable
check_connectivity
log "Connectivity to $REMOTE_SERVER_NAME ✅"
# Version parity — mismatched unRAID could cause md5sum path differences
check_os_version_parity || {
warn "Version parity check failed — proceeding with caution"
warn "Checksum results may be unreliable if md5sum path changed between versions"
}
log "Version parity with $REMOTE_SERVER_NAME ✅"
# Remote array — if array is down all files appear "missing" = false alarm
if ! check_remote_array; then
error "Remote array not mounted on $REMOTE_SERVER_NAME"
error "All files would appear as MISSING — aborting to prevent false alarm"
notify "Backup verify aborted on $(hostname) — remote array not mounted on $REMOTE_SERVER_NAME" \
"Backup Verify" "warning"
exit 1
fi
log "Remote array mounted on $REMOTE_SERVER_NAME ✅"
echo "Pre-flight passed ✅"
# ==============================================================================================
# ━━━ Backup Verification ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_VERIFY Backup Verification — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
echo "$ICON_HOST $MY_ID ($LOCAL_SERVER_NAME) → $REMOTE_ID ($REMOTE_SERVER_NAME)"
echo "$ICON_VERIFY Sample: $BACKUP_VERIFY_SAMPLE files per share (min: $BACKUP_VERIFY_MIN_SIZE)"
echo ""
START=$(date +%s)
TOTAL_CHECKED=0
TOTAL_MATCH=0
TOTAL_MISMATCH=0
TOTAL_MISSING=0
SHARES_WITH_ISSUES=()
for share in "${VERIFY_SHARES[@]}"; do
SHARE_NAME=$(basename "$share")
echo "━━━ $ICON_VERIFY $SHARE_NAME ━━━"
if [[ ! -d "$share" ]]; then
warn "$SHARE_NAME not found locally — skipping"
echo ""
continue
fi
# Sample random files above minimum size
mapfile -t SAMPLE_FILES < <(
find "$share" -type f -size +"$BACKUP_VERIFY_MIN_SIZE" 2>/dev/null | \
shuf | head -n "$BACKUP_VERIFY_SAMPLE"
)
if [[ ${#SAMPLE_FILES[@]} -eq 0 ]]; then
log "$SHARE_NAME — no files found above $BACKUP_VERIFY_MIN_SIZE"
echo ""
continue
fi
log "$SHARE_NAME — sampled ${#SAMPLE_FILES[@]} files"
if [[ "$DRY_RUN" == true ]]; then
for f in "${SAMPLE_FILES[@]}"; do
warn "DRY RUN — would check: $(basename "$f")"
done
echo ""
continue
fi
SHARE_MATCH=0
SHARE_MISMATCH=0
SHARE_MISSING=0
for local_file in "${SAMPLE_FILES[@]}"; do
[[ -z "$local_file" ]] && continue
# Local checksum
local_md5=$(md5sum "$local_file" 2>/dev/null | awk '{print $1}')
if [[ -z "$local_md5" ]]; then
warn "Could not checksum locally: $(basename "$local_file") — skipping"
continue
fi
# Remote checksum via SSH — timeout protected
remote_md5=$(timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" \
-o StrictHostKeyChecking=no \
root@"$REMOTE_SERVER" \
"md5sum '$local_file' 2>/dev/null | awk '{print \$1}'" 2>/dev/null)
(( TOTAL_CHECKED++ ))
if [[ -z "$remote_md5" ]]; then
warn "$ICON_ERROR MISSING: $(basename "$local_file")"
(( SHARE_MISSING++ ))
(( TOTAL_MISSING++ ))
elif [[ "$local_md5" == "$remote_md5" ]]; then
log "MATCH: $(basename "$local_file")"
(( SHARE_MATCH++ ))
(( TOTAL_MATCH++ ))
else
error "MISMATCH: $(basename "$local_file")"
error " local: $local_md5"
error " remote: $remote_md5"
(( SHARE_MISMATCH++ ))
(( TOTAL_MISMATCH++ ))
fi
done
# Per-share result — only visible if issues found
if [[ "$SHARE_MISMATCH" -gt 0 || "$SHARE_MISSING" -gt 0 ]]; then
warn "$SHARE_NAME — match: $SHARE_MATCH missing: $SHARE_MISSING mismatch: $SHARE_MISMATCH"
SHARES_WITH_ISSUES+=("$SHARE_NAME")
else
log "$SHARE_NAME — all $SHARE_MATCH files match ✅"
fi
echo ""
done
END=$(date +%s)
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo "━━━━━ $ICON_SUMMARY BACKUP VERIFY SUMMARY ━━━━━"
echo "$ICON_HOST My ID: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_HOST Remote: $REMOTE_ID ($REMOTE_SERVER_NAME)"
echo "$ICON_VERIFY Checked: $TOTAL_CHECKED files"
echo "$ICON_TIME Duration: $(format_duration $(( END - START )))"
echo ""
if [[ "$TOTAL_MISMATCH" -gt 0 || "$TOTAL_MISSING" -gt 0 ]]; then
echo "$ICON_SUCCESS Match: $TOTAL_MATCH"
warn "Missing: $TOTAL_MISSING"
[[ "$TOTAL_MISMATCH" -gt 0 ]] && echo "$ICON_ERROR Mismatch: $TOTAL_MISMATCH"
fi
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no checksums computed"
elif [[ "$TOTAL_MISMATCH" -gt 0 || "$TOTAL_MISSING" -gt 0 ]]; then
echo "$ICON_ERROR Status: ISSUES FOUND — ${#SHARES_WITH_ISSUES[@]} share(s) need attention: ${SHARES_WITH_ISSUES[*]}"
notify "Backup verify FAILED on $(hostname) → $REMOTE_SERVER_NAME — mismatches: $TOTAL_MISMATCH missing: $TOTAL_MISSING — shares: ${SHARES_WITH_ISSUES[*]}" \
"Backup Verify" "warning"
else
echo "$ICON_DONE Status: all $TOTAL_CHECKED files match across ${#VERIFY_SHARES[@]} shares ✅"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
[[ "$TOTAL_MISMATCH" -gt 0 ]] && exit 1
exit 0
@@ -1,339 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ================================= Docker Daily Restart =======================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Restarts configured containers every night at 1am as proactive maintenance.
#
# Called by daily_sync_maintenance.sh via DAILY_MAINTENANCE_SCRIPTS. Runs inside
# the daily maintenance window — any service downtime is absorbed by a window
# that is already happening. Also drives docker_update.sh in normal mode: the
# same DAILY_RESTART_CONTAINERS list is used for both restarts and image pulls,
# so there is no second list to maintain.
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Proactive Maintenance
# Daily restarts target containers known to degrade over time without
# crossing a clear failure threshold — connection table growth, scheduler
# state accumulation, session cache bloat. The watchdog cannot detect this
# class of degradation. Scheduled restarts clear it before it becomes visible.
#
# State Respect
# Running containers are restarted. Stopped containers are left stopped — they
# were intentionally halted and this script has no authority to override that
# decision. This rule is consistent across the entire ecosystem.
#
# Dependency-Safe Ordering
# Restarts follow the same dependency ordering used by docker_watchdog.sh.
# Services that other containers depend on restart first. A dependent is never
# restarted while its dependency is still coming up.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Dependency Ordering
# Containers restart in dependency-safe order using HOST*_WATCHDOG_DEPENDENCIES.
# CONTAINER_DELAY seconds between dependency restart and dependent restart gives
# the dependency time to fully initialise before dependents try to connect.
#
# Restart Verification
# After each restart, container state is checked after a settle period. A
# container that starts and immediately crashes is marked failed and a
# notification is sent — the script does not silently pass a restart that
# did not stick.
#
# Timeout Protection
# All docker commands wrapped in a 30 second timeout. A hung Docker daemon
# cannot cause this script to hang indefinitely. Timed-out commands retry
# per RETRY_COUNT before marking as failed.
#
# Lock Acquisition
# acquire_lock() prevents concurrent execution if a previous run is still active.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_DAILY_RESTART_CONTAINERS
# Containers restarted nightly. Also used by docker_update.sh normal mode
# for image pulls — add a container once, it gets both. Aliased by
# detect_hosts() → DAILY_RESTART_CONTAINERS
#
# HOST*_WATCHDOG_DEPENDENCIES
# Dependency ordering shared with docker_watchdog.sh. Aliased by
# detect_hosts() → WATCHDOG_DEPENDENCIES
#
# master.conf
#
# RETRY_COUNT
# Retry attempts before giving up on a container
#
# SLEEP
# Seconds between retry attempts
#
# CONTAINER_DELAY
# Seconds to wait after restarting a dependency before starting its dependents
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# docker_daily_restart.sh
# Restart all containers in DAILY_RESTART_CONTAINERS
#
# docker_daily_restart.sh --dry-run
# Preview which containers would be restarted and which would be skipped
#
# docker_daily_restart.sh --status
# Show configured restart list, container states, and dependency ordering
#
# docker_daily_restart.sh --log
# Verbose per-container execution output
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
acquire_lock
if ! command -v docker &>/dev/null; then
error "Docker command not found — check PATH or Docker installation"
notify "Docker daily restart failed — Docker not found on $(hostname)" "Docker Daily Restart" "warning"
exit 1
fi
# detect_hosts() sets MY_ID and aliases HOST*_DAILY_RESTART_CONTAINERS → DAILY_RESTART_CONTAINERS
detect_hosts
if [[ ${#DAILY_RESTART_CONTAINERS[@]} -eq 0 ]]; then
warn "DAILY_RESTART_CONTAINERS is empty for $MY_ID — nothing to restart"
warn "Check HOST${MY_ID#HOST}_DAILY_RESTART_CONTAINERS in host*.conf"
exit 0
fi
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_CONTAINERS Containers: ${DAILY_RESTART_CONTAINERS[*]}"
echo "$ICON_RETRY Retries: $RETRY_COUNT"
echo "$ICON_TIME Sleep: ${SLEEP}s between retries"
echo "$ICON_NOTIFY Notify: unRAID=${NOTIFY_UNRAID:-false} Discord=$([[ -n "${MY_DISCORD_WEBHOOK:-}" ]] && echo enabled || echo disabled)"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no containers will be restarted"
# ==============================================================================================
# ── FUNCTIONS ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# docker_cmd, verify_running, retry_docker — defined in common.sh
# Builds a dependency-safe restart order from DAILY_RESTART_CONTAINERS.
# Containers that are dependencies of others restart first.
# Returns ordered list in ORDERED_RESTART array.
build_restart_order() {
ORDERED_RESTART=()
local remaining=("${DAILY_RESTART_CONTAINERS[@]}")
local placed=()
# First pass — add dependency containers that appear in our list
for container in "${remaining[@]}"; do
[[ -z "$container" ]] && continue
local is_dependency=false
# Check if this container is a dependency of any other in our list
for dep_string in "${WATCHDOG_DEPENDENCIES[@]}"; do
if [[ "$dep_string" == *"$container"* ]]; then
is_dependency=true
break
fi
done
# Also check associative array format
for dependent in "${!WATCHDOG_DEPENDENCIES[@]}"; do
if [[ "${WATCHDOG_DEPENDENCIES[$dependent]}" == *"$container"* ]]; then
is_dependency=true
break
fi
done
if [[ "$is_dependency" == true ]]; then
# Check not already placed
local already=false
for p in "${placed[@]}"; do [[ "$p" == "$container" ]] && already=true && break; done
if [[ "$already" == false ]]; then
ORDERED_RESTART+=("$container")
placed+=("$container")
fi
fi
done
# Second pass — add remaining containers (dependents and independents)
for container in "${remaining[@]}"; do
[[ -z "$container" ]] && continue
local already=false
for p in "${placed[@]}"; do [[ "$p" == "$container" ]] && already=true && break; done
if [[ "$already" == false ]]; then
ORDERED_RESTART+=("$container")
placed+=("$container")
fi
done
}
# Checks if a container is a dependent of the previously restarted container.
# If so, waits CONTAINER_DELAY before restarting to allow dependency to settle.
# Usage: check_dependency_delay "$container" "$last_restarted"
check_dependency_delay() {
local container="$1"
local last="$2"
[[ -z "$last" ]] && return
local deps="${WATCHDOG_DEPENDENCIES[$container]:-}"
if [[ -n "$deps" ]] && [[ "$deps" == *"$last"* ]]; then
echo " Waiting ${CONTAINER_DELAY}s — $container depends on $last..."
sleep "$CONTAINER_DELAY"
fi
}
# ==============================================================================================
# ━━━ Daily Restart ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CONTAINERS Daily Restart — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
echo "$ICON_HOST $MY_ID ($LOCAL_SERVER_NAME) — ${#DAILY_RESTART_CONTAINERS[@]} container(s)"
log "$ICON_CONTAINERS Containers: ${DAILY_RESTART_CONTAINERS[*]}"
log "$ICON_RETRY Retries: $RETRY_COUNT"
log "$ICON_GEAR Config: sleep=${SLEEP}s delay=${CONTAINER_DELAY}s verify-wait=${RESTART_VERIFY_WAIT}s cmd-timeout=${DOCKER_TIMEOUT}s"
START=$(date +%s)
FAILED=()
RESTARTED=()
SKIPPED=()
# Build dependency-safe restart order
build_restart_order
log "$ICON_GEAR Restart order: ${ORDERED_RESTART[*]}"
LAST_RESTARTED=""
for container in "${ORDERED_RESTART[@]}"; do
[[ -z "$container" ]] && continue
c_start=$(date +%s)
c_image=$(docker inspect --format '{{.Config.Image}}' "$container" 2>/dev/null || echo "unknown")
log "━━━ $ICON_CONTAINERS $container ($c_image) ━━━"
if ! timeout "$DOCKER_TIMEOUT" docker inspect "$container" &>/dev/null; then
warn "$container does not exist — skipping"
continue
fi
STATUS=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Running}}' "$container" 2>/dev/null)
case "$STATUS" in
true)
log "$ICON_RUNNING $container is running — restarting..."
# Wait if this container depends on the last one restarted
check_dependency_delay "$container" "$LAST_RESTARTED"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would restart $container"
RESTARTED+=("$container")
else
if retry_docker docker restart "$container"; then
# Verify container stayed running after restart
if verify_running "$container"; then
log "$ICON_STARTED $container restarted and running in $(format_duration $(( $(date +%s) - c_start ))) ✅"
RESTARTED+=("$container")
LAST_RESTARTED="$container"
else
error "$container restarted but crashed immediately"
notify "$container crashed after restart on $(hostname)" "Docker Daily Restart" "warning"
FAILED+=("$container")
fi
else
error "Failed to restart $container after $RETRY_COUNT attempts"
notify "$container failed to restart on $(hostname)" "Docker Daily Restart" "warning"
FAILED+=("$container")
fi
fi
;;
false)
log "$ICON_NOT_RUNNING $container is stopped — skipping"
SKIPPED+=("$container")
;;
*)
error "Unknown status for $container: $STATUS"
FAILED+=("$container")
;;
esac
done
END=$(date +%s)
# ==============================================================================================
# ━━━ Prune Old Images ━━━
# ==============================================================================================
# Restarts above swap containers onto new images — old images are now dangling. Prune immediately.
echo ""
echo "━━━ $ICON_SYNC Pruning Dangling Images — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would prune dangling images"
PRUNED_SUMMARY="(dry run)"
else
PRUNED_OUTPUT=$(docker image prune -f 2>&1)
[[ "$ENABLE_LOGGING" == "true" ]] && echo "$PRUNED_OUTPUT" | sed 's/^/ /'
PRUNED_SUMMARY=$(echo "$PRUNED_OUTPUT" | grep -E "^Total reclaimed" || echo "nothing reclaimed")
fi
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo "━━━━━ $ICON_SUMMARY DAILY RESTART SUMMARY ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_TIME Duration: $(format_duration $((END - START)))"
echo "$ICON_CONTAINERS Scope: ${#RESTARTED[@]} restarted, ${#SKIPPED[@]} skipped, ${#FAILED[@]} failed"
[[ ${#RESTARTED[@]} -gt 0 ]] && log "$ICON_STARTED Restarted: ${RESTARTED[*]}"
[[ ${#SKIPPED[@]} -gt 0 ]] && log "$ICON_NOT_RUNNING Skipped: ${SKIPPED[*]}"
[[ ${#FAILED[@]} -gt 0 ]] && echo "$ICON_ERROR Failed: ${FAILED[*]}"
echo "$ICON_SYNC Pruned: ${PRUNED_SUMMARY:-none}"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no changes made"
elif [[ ${#FAILED[@]} -eq 0 ]]; then
echo "$ICON_DONE Status: ALL DONE ✅"
notify "Daily restart complete — ${#RESTARTED[@]} restarted, ${#SKIPPED[@]} skipped on $(hostname)" "Docker Daily Restart" "normal"
else
echo "$ICON_ERROR Status: ${#FAILED[@]} container(s) failed"
notify "Daily restart completed with errors on $(hostname) — failed: ${FAILED[*]}" "Docker Daily Restart" "warning"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
[[ ${#FAILED[@]} -gt 0 ]] && exit 1
exit 0
@@ -1,344 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ================================= Docker Daily Restart =======================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Restarts configured containers every night at 1am as proactive maintenance.
#
# Called by daily_sync_maintenance.sh via DAILY_MAINTENANCE_SCRIPTS. Runs inside
# the daily maintenance window — any service downtime is absorbed by a window
# that is already happening. Also drives docker_update.sh in normal mode: the
# same DAILY_RESTART_CONTAINERS list is used for both restarts and image pulls,
# so there is no second list to maintain.
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Proactive Maintenance
# Daily restarts target containers known to degrade over time without
# crossing a clear failure threshold — connection table growth, scheduler
# state accumulation, session cache bloat. The watchdog cannot detect this
# class of degradation. Scheduled restarts clear it before it becomes visible.
#
# State Respect
# Running containers are restarted. Stopped containers are left stopped — they
# were intentionally halted and this script has no authority to override that
# decision. This rule is consistent across the entire ecosystem.
#
# Dependency-Safe Ordering
# Restarts follow the same dependency ordering used by docker_watchdog.sh.
# Services that other containers depend on restart first. A dependent is never
# restarted while its dependency is still coming up.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Dependency Ordering
# Containers restart in dependency-safe order using HOST*_WATCHDOG_DEPENDENCIES.
# CONTAINER_DELAY seconds between dependency restart and dependent restart gives
# the dependency time to fully initialise before dependents try to connect.
#
# Restart Verification
# After each restart, container state is checked after a settle period. A
# container that starts and immediately crashes is marked failed and a
# notification is sent — the script does not silently pass a restart that
# did not stick.
#
# Timeout Protection
# All docker commands wrapped in a 30 second timeout. A hung Docker daemon
# cannot cause this script to hang indefinitely. Timed-out commands retry
# per RETRY_COUNT before marking as failed.
#
# Lock Acquisition
# acquire_lock() prevents concurrent execution if a previous run is still active.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_DAILY_RESTART_CONTAINERS
# Containers restarted nightly. Also used by docker_update.sh normal mode
# for image pulls — add a container once, it gets both. Aliased by
# detect_hosts() → DAILY_RESTART_CONTAINERS
#
# HOST*_WATCHDOG_DEPENDENCIES
# Dependency ordering shared with docker_watchdog.sh. Aliased by
# detect_hosts() → WATCHDOG_DEPENDENCIES
#
# master.conf
#
# RETRY_COUNT
# Retry attempts before giving up on a container
#
# SLEEP
# Seconds between retry attempts
#
# CONTAINER_DELAY
# Seconds to wait after restarting a dependency before starting its dependents
#
# RESTART_VERIFY_WAIT
# Seconds to wait after docker restart before checking the container is running.
# Gives the process time to initialise before verify_running samples the state.
# (default: 3)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# docker_daily_restart.sh
# Restart all containers in DAILY_RESTART_CONTAINERS
#
# docker_daily_restart.sh --dry-run
# Preview which containers would be restarted and which would be skipped
#
# docker_daily_restart.sh --status
# Show configured restart list, container states, and dependency ordering
#
# docker_daily_restart.sh --log
# Verbose per-container execution output
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
acquire_lock
if ! command -v docker &>/dev/null; then
error "Docker command not found — check PATH or Docker installation"
notify "Docker daily restart failed — Docker not found on $(hostname)" "Docker Daily Restart" "warning"
exit 1
fi
# detect_hosts() sets MY_ID and aliases HOST*_DAILY_RESTART_CONTAINERS → DAILY_RESTART_CONTAINERS
detect_hosts
if [[ ${#DAILY_RESTART_CONTAINERS[@]} -eq 0 ]]; then
warn "DAILY_RESTART_CONTAINERS is empty for $MY_ID — nothing to restart"
warn "Check HOST${MY_ID#HOST}_DAILY_RESTART_CONTAINERS in host*.conf"
exit 0
fi
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_CONTAINERS Containers: ${DAILY_RESTART_CONTAINERS[*]}"
echo "$ICON_RETRY Retries: $RETRY_COUNT"
echo "$ICON_TIME Sleep: ${SLEEP}s between retries"
echo "$ICON_NOTIFY Notify: unRAID=${NOTIFY_UNRAID:-false} Discord=$([[ -n "${MY_DISCORD_WEBHOOK:-}" ]] && echo enabled || echo disabled)"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no containers will be restarted"
# ==============================================================================================
# ── FUNCTIONS ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# docker_cmd, verify_running, retry_docker — defined in common.sh
# Builds a dependency-safe restart order from DAILY_RESTART_CONTAINERS.
# Containers that are dependencies of others restart first.
# Returns ordered list in ORDERED_RESTART array.
build_restart_order() {
ORDERED_RESTART=()
local remaining=("${DAILY_RESTART_CONTAINERS[@]}")
local placed=()
# First pass — add dependency containers that appear in our list
for container in "${remaining[@]}"; do
[[ -z "$container" ]] && continue
local is_dependency=false
# Check if this container is a dependency of any other in our list
for dep_string in "${WATCHDOG_DEPENDENCIES[@]}"; do
if [[ "$dep_string" == *"$container"* ]]; then
is_dependency=true
break
fi
done
# Also check associative array format
for dependent in "${!WATCHDOG_DEPENDENCIES[@]}"; do
if [[ "${WATCHDOG_DEPENDENCIES[$dependent]}" == *"$container"* ]]; then
is_dependency=true
break
fi
done
if [[ "$is_dependency" == true ]]; then
# Check not already placed
local already=false
for p in "${placed[@]}"; do [[ "$p" == "$container" ]] && already=true && break; done
if [[ "$already" == false ]]; then
ORDERED_RESTART+=("$container")
placed+=("$container")
fi
fi
done
# Second pass — add remaining containers (dependents and independents)
for container in "${remaining[@]}"; do
[[ -z "$container" ]] && continue
local already=false
for p in "${placed[@]}"; do [[ "$p" == "$container" ]] && already=true && break; done
if [[ "$already" == false ]]; then
ORDERED_RESTART+=("$container")
placed+=("$container")
fi
done
}
# Checks if a container is a dependent of the previously restarted container.
# If so, waits CONTAINER_DELAY before restarting to allow dependency to settle.
# Usage: check_dependency_delay "$container" "$last_restarted"
check_dependency_delay() {
local container="$1"
local last="$2"
[[ -z "$last" ]] && return
local deps="${WATCHDOG_DEPENDENCIES[$container]:-}"
if [[ -n "$deps" ]] && [[ "$deps" == *"$last"* ]]; then
echo " Waiting ${CONTAINER_DELAY}s — $container depends on $last..."
sleep "$CONTAINER_DELAY"
fi
}
# ==============================================================================================
# ━━━ Daily Restart ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CONTAINERS Daily Restart — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
echo "$ICON_HOST $MY_ID ($LOCAL_SERVER_NAME) — ${#DAILY_RESTART_CONTAINERS[@]} container(s)"
log "$ICON_CONTAINERS Containers: ${DAILY_RESTART_CONTAINERS[*]}"
log "$ICON_RETRY Retries: $RETRY_COUNT"
log "$ICON_GEAR Config: sleep=${SLEEP}s delay=${CONTAINER_DELAY}s verify-wait=${RESTART_VERIFY_WAIT}s cmd-timeout=${DOCKER_TIMEOUT}s"
START=$(date +%s)
FAILED=()
RESTARTED=()
SKIPPED=()
# Build dependency-safe restart order
build_restart_order
log "$ICON_GEAR Restart order: ${ORDERED_RESTART[*]}"
LAST_RESTARTED=""
for container in "${ORDERED_RESTART[@]}"; do
[[ -z "$container" ]] && continue
c_start=$(date +%s)
c_image=$(docker inspect --format '{{.Config.Image}}' "$container" 2>/dev/null || echo "unknown")
log "━━━ $ICON_CONTAINERS $container ($c_image) ━━━"
if ! timeout "$DOCKER_TIMEOUT" docker inspect "$container" &>/dev/null; then
warn "$container does not exist — skipping"
continue
fi
STATUS=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Running}}' "$container" 2>/dev/null)
case "$STATUS" in
true)
log "$ICON_RUNNING $container is running — restarting..."
# Wait if this container depends on the last one restarted
check_dependency_delay "$container" "$LAST_RESTARTED"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would restart $container"
RESTARTED+=("$container")
else
if retry_docker docker restart "$container"; then
[[ "${RESTART_VERIFY_WAIT:-3}" -gt 0 ]] && sleep "${RESTART_VERIFY_WAIT:-3}"
if verify_running "$container"; then
log "$ICON_STARTED $container restarted and running in $(format_duration $(( $(date +%s) - c_start ))) ✅"
RESTARTED+=("$container")
LAST_RESTARTED="$container"
else
error "$container restarted but crashed immediately"
notify "$container crashed after restart on $(hostname)" "Docker Daily Restart" "warning"
FAILED+=("$container")
fi
else
error "Failed to restart $container after $RETRY_COUNT attempts"
notify "$container failed to restart on $(hostname)" "Docker Daily Restart" "warning"
FAILED+=("$container")
fi
fi
;;
false)
log "$ICON_NOT_RUNNING $container is stopped — skipping"
SKIPPED+=("$container")
;;
*)
error "Unknown status for $container: $STATUS"
FAILED+=("$container")
;;
esac
done
END=$(date +%s)
# ==============================================================================================
# ━━━ Prune Old Images ━━━
# ==============================================================================================
# Restarts above swap containers onto new images — old images are now dangling. Prune immediately.
echo ""
echo "━━━ $ICON_SYNC Pruning Dangling Images — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would prune dangling images"
PRUNED_SUMMARY="(dry run)"
else
PRUNED_OUTPUT=$(docker image prune -f 2>&1)
[[ "$ENABLE_LOGGING" == "true" ]] && echo "$PRUNED_OUTPUT" | sed 's/^/ /'
PRUNED_SUMMARY=$(echo "$PRUNED_OUTPUT" | grep -E "^Total reclaimed" || echo "nothing reclaimed")
fi
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo "━━━━━ $ICON_SUMMARY DAILY RESTART SUMMARY ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_TIME Duration: $(format_duration $((END - START)))"
echo "$ICON_CONTAINERS Scope: ${#RESTARTED[@]} restarted, ${#SKIPPED[@]} skipped, ${#FAILED[@]} failed"
[[ ${#RESTARTED[@]} -gt 0 ]] && log "$ICON_STARTED Restarted: ${RESTARTED[*]}"
[[ ${#SKIPPED[@]} -gt 0 ]] && log "$ICON_NOT_RUNNING Skipped: ${SKIPPED[*]}"
[[ ${#FAILED[@]} -gt 0 ]] && echo "$ICON_ERROR Failed: ${FAILED[*]}"
echo "$ICON_SYNC Pruned: ${PRUNED_SUMMARY:-none}"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no changes made"
elif [[ ${#FAILED[@]} -eq 0 ]]; then
echo "$ICON_DONE Status: ALL DONE ✅"
notify "Daily restart complete — ${#RESTARTED[@]} restarted, ${#SKIPPED[@]} skipped on $(hostname)" "Docker Daily Restart" "normal"
else
echo "$ICON_ERROR Status: ${#FAILED[@]} container(s) failed"
notify "Daily restart completed with errors on $(hostname) — failed: ${FAILED[*]}" "Docker Daily Restart" "warning"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
[[ ${#FAILED[@]} -gt 0 ]] && exit 1
exit 0
@@ -1,978 +0,0 @@
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
# 🎯 ORCHESTRATORS
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
**Sequential job runners that coordinate multiple scripts into single scheduled units.**
Orchestrators contain no business logic — they call other scripts in order, track
pass/fail per job, and produce one clean summary. Configuration lives in `master.conf`.
Adding or removing a job never requires touching the orchestrator script itself.
> **The Varaverk scheduler runs only orchestrators.** Every cron entry, every array
> start/stop event, every scheduled operation runs through an orchestrator. The individual
> scripts it calls are never scheduled directly — they run in a defined order inside a
> coordinated window, with a unified summary at the end.
---
## ━━━ THE PROBLEM THAT BUILT THIS ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
---
### 🔴 Race Conditions From Independent Scheduling
Media permissions ran at 01:05. Arr cleanup ran at 01:00. Arr cleanup started
five minutes before permissions were applied — running against files that were still
owned by root, silently failing to delete the ones it should have cleaned up. Both
scripts reported success. Neither knew about the other. The result was a library
that looked cleaned but wasn't.
The fix: orchestrators enforce order. `daily_sync_maintenance.sh` runs permissions
first, then arr cleanup. The arr scripts see correct ownership every time because the
orchestrator guarantees it. No race, no silent failure, no coordinating cron entries.
---
### 🔴 Six Separate Notifications Instead of One
Before orchestrators, each script sent its own notification on completion. A single
daily run produced six separate notification pings — one for permissions, one for
each arr cleanup, one for the cleaner, one for docker restart. Six bells for one
maintenance window. Worse, if something failed in the middle, you'd get some
notifications and not others, and figuring out which step failed meant correlating
timestamps across multiple notification messages.
The fix: orchestrators collect all results and send one notification at the end.
One summary. One bell. Clear pass/fail count. If something failed, the summary
tells you which job and what happened — no correlation needed.
---
### 🔴 Transcode Manager Triggering Unnecessary SSD Flips
`transcode_manager.sh` ran every 7 minutes on its own. It checked ramdisk usage —
saw 6.8GB used, threshold is 6.5GB, flipped sessions to SSD. One minute later
`transcode_cleanup.sh` ran and removed 4GB of stale segment files from ended sessions.
Actual usage was 2.8GB. Sessions were now on SSD for no reason. Users experiencing
slightly worse performance. The flip counter incremented for nothing.
The fix: `transcode_management.sh` runs cleanup first, manager second, every cycle.
The manager always sees post-cleanup usage. Stale files can't trigger a flip because
they're gone before the manager looks. The correct order requires exactly one
orchestrator to enforce it.
---
### 🔴 Failed Imports Sitting Stalled for Days
A release downloads successfully but Lidarr can't import it — wrong format, incorrect
tags, file already exists. Lidarr marks it `importFailed` and stops. Nobody notices.
The download client has the file, Lidarr has given up, and nothing is going to happen
until someone manually opens Lidarr, identifies the problem, blocklists the release,
and triggers a new search. This takes minutes to do — but nobody does it at 3am
when it usually happens.
The fix: `arrs_failed_stalled_recovery.sh` runs every 6 hours. It finds all
`importFailed`, `importPending`, `error`, and `stalled` items, blocklists them, removes
them from the queue, and triggers a new search — automatically. By morning the failed
import has already been replaced by a working one. No manual intervention required.
---
### 🔴 Array Start Scripts Running in Wrong Order or Not at All
Startup scripts configured individually ran in an unpredictable order. The ramdisk
setup might run after Emby starts. The syslog filter might run after containers have
already created veth interfaces. PHP-FPM tuning might run after the WebGUI has already
served its first requests. Each script competed for the same startup slot with no
guaranteed order.
The fix: `array_started.sh` is the only array-start entry in the Varaverk scheduler.
It launches every startup script in a defined order, with one-second settle between
each, and reports which succeeded and which failed. Order is guaranteed. Nothing starts
before its dependency. Everything is visible in a single summary.
---
## ━━━ THE ORCHESTRATOR MODEL ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
The Varaverk scheduler contains exactly these entries:
```bash
# Array start event (Varaverk disks_mounted hook → cron: "array_start"):
array_started.sh
# Cron — managed via Varaverk Scheduler:
*/7 * * * * transcode_management.sh
*/15 * * * * watchdog_orchestrator.sh ← resource → docker → system → stability
*/30 * * * * critical_sync_maintenance.sh ← auth + Emby dirty sync + partnership
0 */4 * * * intermediate_sync_maintenance.sh ← arr sync + failed recovery + optional rsync
0 1 * * * daily_sync_maintenance.sh
0 7 * * 0 sunday_morning_coffee_report.sh
30 2 * * 0 weekly_sync_maintenance.sh
0 0 15 * * monthly_maintenance.sh ← uptime-gated: ZFS scrub, SMART tests
# Manual only (not scheduled):
fallback_test.sh, emby_database_repair.sh, repair tools
```
Every job list is configured in the `ORCHESTRATORS` section of `master.conf`.
No orchestrator script ever changes when jobs are added or removed — only `master.conf` changes.
---
## ━━━ THE ORCHESTRATOR PATTERN ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
All orchestrators follow the same structure:
```
1. Setup → validate config, detect_hosts(), acquire_lock
2. Pre-flight → fail fast checks before any work begins
3. Job loop → run each job, track pass/fail, continue on failure
4. Summary → one clean report of all job results
5. Notification → one notify per run on failure (never per job)
```
Properties that apply to every orchestrator:
```
Consistent output → every orchestrator looks the same in logs
No silent failures → pass/fail tracked per job, all in summary
Resilient → one job failing does not stop the rest
Single notification → one bell per run, not one per job
--dry-run cascade → passes --dry-run through to every child script
--status support → show configured jobs and exit
```
---
## ━━━ OUTPUT TIERS ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
All scripts use a two-tier output model: `echo` lines are always visible; `log`
lines only appear when `--log` is passed.
**One-shot orchestrators** (`array_started.sh`, `array_stopping.sh`, `sunday_morning_coffee_report.sh`):
without `--log`, section headers, per-phase results, and the final summary are
visible. Per-item detail suppressed.
**Periodic orchestrators** (`critical_sync_maintenance.sh`, `daily_sync_maintenance.sh`,
`intermediate_sync_maintenance.sh`, `weekly_sync_maintenance.sh`): without `--log`,
phase headers, per-phase completion status, and the final summary are visible. Per-share
and per-job detail suppressed.
**High-frequency orchestrators** (`watchdog_orchestrator.sh`, `transcode_management.sh`): silent
during clean cycles. Only state transitions, errors, and startup-grace expiry shown
without `--log`.
---
## ━━━ SCRIPTS AT A GLANCE ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
| Script | What It Orchestrates | Schedule |
|--------|---------------------|----------|
| `array_started.sh` | All array startup scripts in order | `array_start` event (Varaverk plugin hook) |
| `watchdog_orchestrator.sh` | resource → docker → system → api_renew → stability watchdogs | Every 15 minutes |
| `transcode_management.sh` | Cleanup then manager — order critical | Every 7 minutes |
| `arrs_failed_stalled_recovery.sh` | Failed import + stalled download recovery | Every 6 hours |
| `daily_sync_maintenance.sh` | git pull → sync → media maintenance → restarts | 1am daily |
| `weekly_sync_maintenance.sh` | Stop → update → clean sync → start → weekly restarts | 2:30am Sunday |
| `monthly_maintenance.sh` | Uptime-triggered heavy tasks — ZFS scrub, SMART tests | Daily check, fires when uptime ≥ 30d |
| `intermediate_sync_maintenance.sh` | arr library sync, artwork, failed recovery | Every 4 hours |
---
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
## 🚀 array_started.sh
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
Single array-start entry for the entire ecosystem. Fired by the Varaverk plugin's
`disks_mounted` event hook. Launches every startup script in order — each as a
background process — and reports which succeeded and which failed.
```bash
# Triggered by: Plugin/unraid/event/disks_mounted/array_start_jobs
# schedule.json entry: "Orchestrators/array_started.sh" → cron: "array_start"
```
---
### ── Execution Order ──────────────────────────────────────────────────────────
```bash
# master.conf
# ─────────────────────────────────────────────────────────────────────────────
# Order matters — each entry depends on the previous ones having run.
# See comments for why each position is correct.
#
ARRAY_START_SCRIPTS=(
# ── One-shot scripts — run and exit naturally ─────────────────────────────
"System_Essentials/unraid_api_key_renew.sh" # re-register API key FIRST — unraid-api
# registry is ephemeral, lost on service restart
"System_Essentials/inotify_tuning.sh" # raise inotify BEFORE containers start
# containers inherit limits at startup —
# if Code-Server starts with low limits
# it keeps them until restart
"System_Essentials/docker_syslog_filter.sh" # suppress veth noise BEFORE containers create
# veth interfaces — otherwise the first boot
# always has unfiltered veth spam
"System_Essentials/php_fpm_max_children.sh" # WebGUI tuning — before any WebGUI requests
"Transcodes/ramdisk_setup.sh" # create tmpfs + symlink BEFORE Emby starts —
# Emby needs the transcode path to exist
"Docker_Essentials/docker_network_connect.sh" # ensure networks + connections BEFORE
# watchdogs check container states
# ── Continuous scripts — run until array stops ─────────────────────────────
"Fallback/fallback.sh" # fallback LAST — needs everything else stable
)
# NOTE: watchdogs (docker_watchdog, system_watchdog, stability_watchdog) are NOT here.
# They run via watchdog_orchestrator.sh on cron every 15 minutes — not as daemons.
```
---
### ── One-Shot vs Continuous Detection ───────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# Each script is launched with `bash script.sh &` — background process.
# After 1 second:
# PID still alive → continuous script (running in background)
# logged as: "fallback.sh — running (PID 12345)"
# PID dead + exit 0 → one-shot completed successfully
# logged as: "inotify_tuning.sh — completed (one-shot)"
# PID dead + exit N → failure
# logged as: "ramdisk_setup.sh — exited with code 1"
# full path printed — debugging is immediate
#
# This means the orchestrator correctly identifies and reports all startup
# scripts without needing to know in advance which ones are continuous.
```
---
### ── Auto-Fix Permissions ────────────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# Scripts that are not executable are chmod +x'd automatically before launch.
# A permissions problem on a startup script does not cause a silent skip —
# the orchestrator fixes it and proceeds, then logs that it did so.
# This prevents "why didn't X run on startup" questions.
```
---
### ── Usage ───────────────────────────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# Normal — fired by Varaverk disks_mounted event hook. Never run manually in production.
# array_started.sh runs once and exits — the continuous scripts it launched
# keep running as background processes.
# ─────────────────────────────────────────────────────────────────────────────
array_started.sh
# ─────────────────────────────────────────────────────────────────────────────
# Dry run — show what would be launched, in order, without launching anything.
# Use to verify the ARRAY_START_SCRIPTS list before an array restart.
# ─────────────────────────────────────────────────────────────────────────────
array_started.sh --dry-run
# ─────────────────────────────────────────────────────────────────────────────
# Status — show each configured script with its current running state.
# RUNNING (PID XXXXX) — continuous script currently active
# not running — one-shot that has completed, or continuous not yet started
# FILE NOT FOUND — script path wrong or missing
# ─────────────────────────────────────────────────────────────────────────────
array_started.sh --status
```
---
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
## 🎬 transcode_management.sh
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
Runs `transcode_cleanup.sh` then `transcode_manager.sh` in the correct order every
7 minutes. Exists because the order is not optional — the manager must always see
post-cleanup usage to make accurate flip decisions.
```bash
# Scheduled: */7 * * * * (every 7 minutes)
```
---
### ── Why Order Is Non-Negotiable ─────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# Emby writes segment files to the transcode directory as it buffers streams.
# When a stream ends, Emby deletes its own active files — but may leave behind
# stale segment files from sessions that ended uncleanly. These consume real
# ramdisk space. The manager has no way to know if they're active or stale.
#
# Without correct order:
# Manager runs → sees 6.8GB used (stale files inflating) → exceeds threshold
# → flips sessions to SSD → flip counter incremented
# Cleanup runs → removes 4GB of stale files → actual usage was 2.8GB
# → flip was unnecessary — sessions now on SSD for no reason
#
# With correct order (this orchestrator):
# Cleanup runs → removes stale files → actual usage 2.8GB
# Manager runs → sees 2.8GB → below threshold → stays on ramdisk ✅
# → no flip, no wasted counter, correct decision every time
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── What Each Child Script Does ────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# transcode_cleanup.sh:
# Identifies segment files not currently open by any process (via lsof)
# Removes them from the transcode directory
# If usage drops enough after cleanup → triggers flip-back to ramdisk
# (handles the recovery direction so manager doesn't have to)
#
# transcode_manager.sh:
# Reads current ramdisk usage after cleanup has run
# Compares against RAMDISK_WARN_GB threshold
# Flips the symlink if needed (ramdisk → SSD or SSD → ramdisk)
# Writes one entry to TRANSCODE_DAILY_LOG for the weekly coffee report
# Shows active Emby sessions with their play method
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Daily Log ────────────────────────────────────────────────────────────────
```bash
# master.conf
# ─────────────────────────────────────────────────────────────────────────────
# transcode_manager.sh writes to TRANSCODE_DAILY_LOG after each cycle.
# Format: DATE|RAMDISK_USED_GB|FLIP_COUNT|RAM_SESSIONS|SSD_SESSIONS
#
# The sunday_morning_coffee_report.sh reads this log for weekly stats:
# Peak ramdisk usage across the week
# Total flip count (unnecessary flips visible here)
# Session split: how often ramdisk vs SSD was used
#
# Log trimmed to TRANSCODE_LOG_RETENTION days on every write — bounded, never grows.
#
TRANSCODE_DAILY_LOG="$DATA_DIR/transcode_daily.db"
TRANSCODE_LOG_RETENTION=90 # days
TRANSCODE_STATE_FILE="/tmp/transcode_state.db" # /tmp — resets on reboot
```
---
### ── Usage ───────────────────────────────────────────────────────────────────
```bash
transcode_management.sh # normal run (every 7 min via cron)
transcode_management.sh --dry-run # passes --dry-run to both child scripts
transcode_management.sh --status # show config, current state, daily log stats
transcode_management.sh --log # verbose output from both child scripts
```
---
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
## 🔄 arrs_failed_stalled_recovery.sh
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
Automatic recovery from failed imports and stalled downloads across Sonarr, Radarr,
and Lidarr. Blocklists the bad release, removes it from the queue, and triggers a
new search — hands-free recovery while you sleep.
```bash
# Scheduled: 0 */6 * * * (every 6 hours)
```
---
### ── Four Problem Types ───────────────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# The arr queue API reports these status values for problem items:
#
# importFailed — downloaded successfully but arr couldn't import the file
# Common causes: wrong format for quality profile, corrupted file,
# duplicate already in library, permission issue on import path
# Self-resolution: never — arr stops trying after first failure
#
# importPending — downloaded, stuck waiting for import to begin
# Common causes: import queue backed up, arr paused, API error
# Self-resolution: sometimes — but stuck for hours is always wrong
#
# error — serious failure state not covered by the above
# Common causes: indexer issues, download client unreachable, disk full
#
# stalled — download stuck with no connections or no progress
# Common causes: no seeders, VPN routing issue, tracker ban
# Self-resolution: never without a source change
#
# NOT touched — items with status "downloading" or "imported" are never touched.
# Safe to run at any time — only processes items that are already broken.
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── What It Does Per Problem Item ──────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# For each problem item found, in order:
#
# 1. Blocklist the release
# Prevents the arr from grabbing the exact same release again immediately.
# The bad indexer result goes into the blocklist — future searches skip it.
#
# 2. Remove from queue
# Tells the download client to stop and remove the failed download.
# Frees up the slot for the replacement.
#
# 3. Trigger new search
# Arr searches for a different release meeting the quality profile.
# If a suitable alternative exists, it starts downloading automatically.
# If not, the item is marked as "awaiting upgrade" — arr will retry when
# a new indexer result appears.
#
# The entire cycle from "failed import" to "replacement downloading" happens
# without any human involvement.
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Age Threshold ────────────────────────────────────────────────────────────
```bash
# master.conf
# ─────────────────────────────────────────────────────────────────────────────
# Items newer than ARR_IMPORT_RECOVERY_AGE hours are skipped.
# Arrs retry on their own schedule after initial failures — a 30-minute-old
# importFailed may still resolve itself. Waiting 6 hours before intervening
# gives the arr a full retry cycle before this script steps in.
#
ARR_IMPORT_RECOVERY_AGE=6 # hours — skip items newer than this
```
---
### ── Host Awareness ───────────────────────────────────────────────────────────
```bash
# master.conf + host*.conf
# ─────────────────────────────────────────────────────────────────────────────
# Each arr is independently toggled per host.
# Lidarr only runs on HOST1 — exits cleanly on HOST2 with no action.
# HOST2 has its own Sonarr and Radarr for its anime shares.
#
HOST1_SONARR_RECOVERY=true # HOST1 Sonarr — Tv_Shows
HOST1_RADARR_RECOVERY=true # HOST1 Radarr — Movies
HOST1_LIDARR_RECOVERY=true # HOST1 Lidarr — Music (HOST1 only)
HOST2_SONARR_RECOVERY=true # HOST2 Sonarr — Anime_Shows
HOST2_RADARR_RECOVERY=true # HOST2 Radarr — Anime_Movies
#
# API versions (current — update if arr major version changes):
# Sonarr v4 → /api/v3/
# Radarr v6 → /api/v3/
# Lidarr v3 → /api/v1/
```
---
### ── Usage ───────────────────────────────────────────────────────────────────
```bash
arrs_failed_stalled_recovery.sh # normal run
arrs_failed_stalled_recovery.sh --dry-run # show what would be actioned, no changes
arrs_failed_stalled_recovery.sh --log # verbose — show each queue item evaluated
arrs_failed_stalled_recovery.sh --status # show arr config and API connectivity
```
---
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
## 📅 daily_sync_maintenance.sh
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
Full daily maintenance window orchestrator. The entire 1am window — git pull, media
share sync, media maintenance, and docker daily restarts — in one scheduled entry.
Runs on both servers; `detect_hosts()` determines which direction each sync goes.
```bash
# Scheduled: 0 1 * * * (1am daily — on BOTH servers)
```
---
### ── Execution Order ──────────────────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# The split into pre-sync and post-sync is based on a simple rule:
# git pull runs before anything — maintenance uses the latest scripts
# rsync runs in the middle — media management uses the post-sync state
# docker restart runs last — after everything else has completed
#
# 1. Pre-sync: git_pull_execute.sh
# Pull latest scripts first. Everything that follows runs on current code.
#
# 2. Rsync window: HOST*_DAILY_SYNC_SHARES + HOST*_PERSONAL_SHARES
# Each server pushes its own truth shares to the other.
# HOST1 → pushes Movies, Tv_Shows, Music → HOST2
# HOST2 → pushes Anime_Shows, Anime_Movies → HOST1
# Personal encrypted shares appended after standard shares.
# Drive temperature exit codes respected — skip share or abort all on CRIT.
#
# 3. Post-sync: DAILY_MAINTENANCE_SCRIPTS (everything except git pull)
# media_management.sh → permissions + cleaners + arr cleanup
# docker_daily_restart.sh → nightly container restarts
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Bidirectional — Same Script, Correct Direction Automatic ────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# detect_hosts() sets MY_ID at runtime and aliases HOST*_DAILY_SYNC_SHARES
# to DAILY_SYNC_SHARES. The script uses DAILY_SYNC_SHARES — always the right
# list for whichever server is running.
#
# No HOST1/HOST2 comparisons in the script. Configuration drives direction.
#
# HOST1 runs this script at 1am:
# → pushes HOST1_DAILY_SYNC_SHARES (Movies, Tv_Shows, Music) → HOST2
# → media_management.sh on HOST1's shares
# → docker_daily_restart.sh on HOST1's containers
#
# HOST2 runs this script at 1am:
# → pushes HOST2_DAILY_SYNC_SHARES (Anime_Shows, Anime_Movies) → HOST1
# → media_management.sh on HOST2's shares
# → docker_daily_restart.sh on HOST2's containers
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Configuration ────────────────────────────────────────────────────────────
```bash
# master.conf
# ─────────────────────────────────────────────────────────────────────────────
# Script lists — git_pull is split out as pre-sync; everything else is post-sync.
# The orchestrator recognises git_pull_execute.sh by name and routes it correctly.
#
DAILY_MAINTENANCE_SCRIPTS=(
"git_pull_execute.sh" # PRE-SYNC — always first
"Media/media_shares_permissions.sh" # POST-SYNC — permissions before arr
"Media/media_cleaner.sh anime" # POST-SYNC — junk before orphan scan
"Media/media_cleaner.sh media" # POST-SYNC
"Media/lidarr_cleanup.sh" # POST-SYNC — orphan cleanup last
"Media/sonarr_cleanup.sh" # POST-SYNC
"Media/radarr_cleanup.sh" # POST-SYNC
"Docker_Essentials/docker_daily_restart.sh" # POST-SYNC — restarts after everything
)
# host1.conf
HOST1_DAILY_SYNC_SHARES=(
"/mnt/user/Movies" # both servers — arr_sync union, rsync spreads files
"/mnt/user/Tv_Shows" # both servers
"/mnt/user/Music" # both servers
"/mnt/user/Kids_Movies"
"/mnt/user/Kids_Tv_Shows"
"/mnt/user/Sports"
"/mnt/user/stand-up_comedy"
)
HOST1_PERSONAL_SHARES=(
"/mnt/user/Personal" # encrypted personal share — appended after standard
)
# host2.conf
HOST2_DAILY_SYNC_SHARES=(
"/mnt/user/Anime_Shows" # both servers — arr_sync union, rsync spreads files
"/mnt/user/Anime_Movies" # both servers
)
```
---
### ── Adding or Removing a Job ────────────────────────────────────────────────
```bash
# master.conf
# ─────────────────────────────────────────────────────────────────────────────
# Add a new media script — just insert it in the right position:
DAILY_MAINTENANCE_SCRIPTS=(
"git_pull_execute.sh"
"Media/media_shares_permissions.sh"
"Media/media_cleaner.sh anime"
"Media/media_cleaner.sh media"
"Media/my_new_script.sh" # ← add here in the correct order
"Media/lidarr_cleanup.sh"
"Media/sonarr_cleanup.sh"
"Media/radarr_cleanup.sh"
"Docker_Essentials/docker_daily_restart.sh"
)
# Disable a job temporarily — comment it out, do not delete:
DAILY_MAINTENANCE_SCRIPTS=(
"git_pull_execute.sh"
"Media/media_shares_permissions.sh"
# "Media/media_cleaner.sh anime" # ← temporarily disabled
"Media/media_cleaner.sh media"
"Media/lidarr_cleanup.sh"
...
)
# ─────────────────────────────────────────────────────────────────────────────
# No changes to daily_sync_maintenance.sh needed in either case.
```
---
### ── Drive Temperature Exit Codes ───────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# rsync.sh returns specific exit codes for temperature issues.
# The orchestrator handles these correctly — one hot drive does not abort all others.
#
# exit 0 → success — continue to next share
# exit 1 → temp WARN — skip this share, continue to next share
# exit 2 → temp CRITICAL — abort ALL remaining shares in this window
# notify immediately with which share triggered the abort
# exit N → other failure — skip this share, continue to next share
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Relationship to Fallback Writeback ──────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# The same HOST*_DAILY_SYNC_SHARES lists are used by fallback.sh for Tier 4
# writeback — but in the opposite direction.
#
# Normal (daily_sync_maintenance.sh):
# HOST1 → pushes Movies, Tv_Shows → HOST2
#
# Tier 4 fallback writeback (HOST1 returns after 24hr+ outage):
# HOST2 → pushes Movies, Tv_Shows → HOST1
# (HOST2 was running HOST1's arrs and accumulated content)
#
# Same list, correct direction for the situation, zero duplication.
# No separate writeback list to maintain.
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Usage ───────────────────────────────────────────────────────────────────
```bash
daily_sync_maintenance.sh # normal run
daily_sync_maintenance.sh --dry-run # preview all jobs without syncing or changing
daily_sync_maintenance.sh --log # verbose per-share and per-job output
daily_sync_maintenance.sh --status # show configured shares and jobs for this host
```
---
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
## 📅 weekly_sync_maintenance.sh
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
Weekly maintenance window orchestrator — clean sync of Emby and auth stack, container
image updates, and weekly docker restarts. Runs Sunday 2:30am; fits inside the Sunday
maintenance block before the 7am coffee report.
```bash
# Scheduled: 30 2 * * 0 (Sunday 2:30am)
```
---
### ── Execution Order ──────────────────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# Containers stop BEFORE sync — clean static source, full bandwidth.
# Containers start AFTER sync — on fresh data, in dependency order.
# DDNS and fallback continue running throughout — only managed containers stop.
#
# 1. Pre-flight checks — connectivity, remote Docker daemon, remote rootfs
# 2. Stop local containers — Emby + auth stack stopped on this server
# 3. Stop remote containers — Emby + auth stack stopped on remote via SSH
# 4. Pull updates locally — if WEEKLY_SYNC_UPDATES=true (containers already stopped)
# 5. Pull updates remotely — if WEEKLY_SYNC_UPDATES_REMOTE=true
# 6. rsync WEEKLY_SYNC_SHARES — full clean mirror at full bandwidth
# 7. Start remote containers — new image, correct dependency order
# 8. Start local containers — new image, correct dependency order
# Post-sync jobs: docker_weekly_restart.sh
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Why Weekly Not Nightly for Emby ─────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# Two Emby syncs run in parallel — dirty and clean:
#
# emby-fallback dirty sync (every 30 minutes via critical_sync_maintenance.sh, Emby running):
# watch states, library deltas, user activity — continuous coverage
# WAL files excluded — safe to copy while Emby writes
# HOST2 always within 30 minutes of HOST1 on playback state
#
# weekly clean sync (Sunday 2:30am, Emby stopped):
# Full clean mirror — all databases checkpointed and flushed
# Metadata, plugins, config all included
# ~30 seconds of Emby downtime — both instances stopped during rsync only
#
# Why not nightly:
# Emby builds a warm image thumbnail cache on HOST2 throughout the week.
# Nightly sync resets this cache — cold loads every morning for users.
# Weekly sync: cache stays warm for 6 days. Resets Sunday night while users sleep.
# One weekly reset at an acceptable time is better than six unnecessary resets.
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Container Update Window ─────────────────────────────────────────────────
```bash
# master.conf
# ─────────────────────────────────────────────────────────────────────────────
# Containers are already stopped for the sync — pulling updated images costs
# nothing extra in downtime. Both servers start on the same image version
# after the window completes.
#
WEEKLY_SYNC_UPDATES=true # pull updated images locally
WEEKLY_SYNC_UPDATES_REMOTE=true # pull updated images on remote via SSH
#
# Toggle false to skip updates without changing the schedule:
# WEEKLY_SYNC_UPDATES=false # skips pulls, containers still restart on current image
```
---
### ── Why Auth Stack Weekly Sync Matters ──────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# NPM, Authelia, LLDAP, Mariadb run warm on both servers continuously.
# HOST1 is source of truth — changes propagate to HOST2 via weekly clean sync.
#
# What propagates automatically every Sunday:
# New users added in LLDAP on HOST1 → appear on HOST2 by Monday
# Proxy rules changed in NPM on HOST1 → live on HOST2 by Monday
# Authelia policies updated on HOST1 → enforced on HOST2 by Monday
# TLS certificates renewed on HOST1 → valid on HOST2 by Monday
#
# No manual sync needed for routine auth administration.
# Anything done on HOST1 is on HOST2 within a week.
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Sunday Maintenance Window ───────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# This script is part of a coordinated Sunday maintenance block:
#
# 2:30am weekly_sync_maintenance.sh ← clean sync + image updates (~5-10min)
# 2:50am CA Auto Update plugin ← unRAID plugin updates
# 2:55am CA container updates ← additional container updates
# 3:00am Network reboot ← router/switch restart
#
# Everything comes back clean:
# Network fresh, Emby + auth updated, containers on latest images.
# All in one window while users sleep.
# Sunday morning coffee report at 7am shows the post-maintenance state.
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Configuration ────────────────────────────────────────────────────────────
```bash
# master.conf
WEEKLY_SYNC_SHARES=(
"/mnt/user/Media_Server/Emby" # emby profile — full clean mirror
"/mnt/user/appdata-Fallback/Critical-Data" # critical-data — auth stack clean state
)
WEEKLY_MAINTENANCE_SCRIPTS=(
"Docker_Essentials/docker_weekly_restart.sh" # weekly restart of less-critical services
)
WEEKLY_SYNC_UPDATES=true
WEEKLY_SYNC_UPDATES_REMOTE=true
```
---
### ── Usage ───────────────────────────────────────────────────────────────────
```bash
weekly_sync_maintenance.sh # normal run
weekly_sync_maintenance.sh --dry-run # preview — no stops, no syncs, no starts
weekly_sync_maintenance.sh --log # verbose per-share per-job output
weekly_sync_maintenance.sh --status # show configured shares, jobs, update toggles
```
---
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
## 📆 monthly_maintenance.sh
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
Uptime-triggered orchestrator for long-running system tasks — ZFS scrub, SMART
long tests — that should only run on stable systems that have been up for at least
30 days. Called daily by cron; most invocations are silent no-ops.
Two gates must both pass before any job runs:
1. Server uptime ≥ `MONTHLY_UPTIME_THRESHOLD_DAYS`
2. Last run ≥ `MONTHLY_RUN_INTERVAL_DAYS` ago (state file on `/boot/config/` — survives reboots)
If either gate fails, the script exits 0 with no output. This is expected — it runs
daily and most days are no-ops.
`--force` bypasses both gates and runs the job list immediately. Use for testing or
when a scrub was missed and the server hasn't reached the uptime threshold yet.
### Configuration (master.conf)
```bash
MONTHLY_MAINTENANCE_SCRIPTS=(
#"Tools/zfs_pool_scrub.sh"
#"Tools/smart_long_test.sh"
)
MONTHLY_UPTIME_THRESHOLD_DAYS=30
MONTHLY_RUN_INTERVAL_DAYS=30
MONTHLY_LAST_RUN_FILE="/boot/config/monthly_maintenance_last_run.db"
```
Scripts are commented out by default — uncomment what applies to your hardware.
### Usage
```bash
monthly_maintenance.sh # normal run (daily cron — silent no-op when gates not met)
monthly_maintenance.sh --force # bypass both gates — run immediately
monthly_maintenance.sh --dry-run # show what would run without running it
monthly_maintenance.sh --status # show gate state: uptime, last run, next eligible
monthly_maintenance.sh --log # verbose output from each child script
```
---
## ━━━ COMPLETE SCHEDULE ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
```bash
# ─────────────────────────────────────────────────────────────────────────────
# Array start event (Varaverk disks_mounted hook):
# ─────────────────────────────────────────────────────────────────────────────
array_started.sh
# ─────────────────────────────────────────────────────────────────────────────
# Every 7 minutes:
# ─────────────────────────────────────────────────────────────────────────────
*/7 * * * * transcode_management.sh
# ─────────────────────────────────────────────────────────────────────────────
# Every 15 minutes — watchdog cycle:
# ─────────────────────────────────────────────────────────────────────────────
*/15 * * * * watchdog_orchestrator.sh
# ─────────────────────────────────────────────────────────────────────────────
# Every 30 minutes — auth stack + Emby dirty sync + partnership check:
# ─────────────────────────────────────────────────────────────────────────────
*/30 * * * * critical_sync_maintenance.sh
# ─────────────────────────────────────────────────────────────────────────────
# Every 4 hours — arr library sync + failed import recovery + optional rsync:
# ─────────────────────────────────────────────────────────────────────────────
0 */4 * * * intermediate_sync_maintenance.sh
# ─────────────────────────────────────────────────────────────────────────────
# Daily — 1am:
# git pull → rsync all truth shares → permissions + cleaners + arr cleanup → docker restart
# ─────────────────────────────────────────────────────────────────────────────
0 1 * * * daily_sync_maintenance.sh
# ─────────────────────────────────────────────────────────────────────────────
# Weekly — Sunday maintenance block:
# ─────────────────────────────────────────────────────────────────────────────
30 2 * * 0 weekly_sync_maintenance.sh # clean sync + updates (~5-10min)
50 2 * * 0 CA Auto Update plugin # plugin updates
55 2 * * 0 CA container updates # container image updates
0 3 * * 0 Network reboot # router/switch restart
0 7 * * 0 sunday_morning_coffee_report.sh
# ─────────────────────────────────────────────────────────────────────────────
# 15th of each month (uptime-gated — silent no-op if uptime < 30 days):
# ─────────────────────────────────────────────────────────────────────────────
0 0 15 * * monthly_maintenance.sh
```
---
## ━━━ ADDING A NEW ORCHESTRATOR ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
If you find yourself running 3+ related scripts on the same schedule, wrap them
in a new orchestrator. Model directly on `media_management.sh` which has the
complete pattern — dry-run passthrough, status display, pass/fail tracking, summary.
```bash
# Minimal skeleton — the full pattern in its simplest form:
#!/bin/bash
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
SCRIPTS_ROOT="$SCRIPT_DIR/.."
PASS=()
FAIL=()
# Read job list from master.conf — never hardcode jobs in the orchestrator
for script_entry in "${MY_MAINTENANCE_JOBS[@]:-}"; do
[[ -z "$script_entry" ]] && continue
read -r -a parts <<< "$script_entry"
script_path="$SCRIPTS_ROOT/${parts[0]}"
script_name=$(basename "${parts[0]}")
extra_args=("${parts[@]:1}")
[[ "$DRY_RUN" == true ]] && extra_args+=("--dry-run")
if bash "$script_path" "${extra_args[@]}"; then
PASS+=("$script_name")
else
FAIL+=("$script_name")
fi
done
# One summary — one notification
echo "Passed: ${#PASS[@]} Failed: ${#FAIL[@]}"
[[ ${#FAIL[@]} -gt 0 ]] && \
notify "My maintenance failed on $(hostname) ($MY_ID) — ${FAIL[*]}" \
"My Orchestrator" "warning"
```
@@ -1,977 +0,0 @@
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
# 🎯 ORCHESTRATORS
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
**Sequential job runners that coordinate multiple scripts into single scheduled units.**
Orchestrators contain no business logic — they call other scripts in order, track
pass/fail per job, and produce one clean summary. Configuration lives in `master.conf`.
Adding or removing a job never requires touching the orchestrator script itself.
> **The Varaverk scheduler runs only orchestrators.** Every cron entry, every array
> start/stop event, every scheduled operation runs through an orchestrator. The individual
> scripts it calls are never scheduled directly — they run in a defined order inside a
> coordinated window, with a unified summary at the end.
---
## ━━━ THE PROBLEM THAT BUILT THIS ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
---
### 🔴 Race Conditions From Independent Scheduling
Media permissions ran at 01:05. Arr cleanup ran at 01:00. Arr cleanup started
five minutes before permissions were applied — running against files that were still
owned by root, silently failing to delete the ones it should have cleaned up. Both
scripts reported success. Neither knew about the other. The result was a library
that looked cleaned but wasn't.
The fix: orchestrators enforce order. `daily_sync_maintenance.sh` runs permissions
first, then arr cleanup. The arr scripts see correct ownership every time because the
orchestrator guarantees it. No race, no silent failure, no coordinating cron entries.
---
### 🔴 Six Separate Notifications Instead of One
Before orchestrators, each script sent its own notification on completion. A single
daily run produced six separate notification pings — one for permissions, one for
each arr cleanup, one for the cleaner, one for docker restart. Six bells for one
maintenance window. Worse, if something failed in the middle, you'd get some
notifications and not others, and figuring out which step failed meant correlating
timestamps across multiple notification messages.
The fix: orchestrators collect all results and send one notification at the end.
One summary. One bell. Clear pass/fail count. If something failed, the summary
tells you which job and what happened — no correlation needed.
---
### 🔴 Transcode Manager Triggering Unnecessary SSD Flips
`transcode_manager.sh` ran every 7 minutes on its own. It checked ramdisk usage —
saw 6.8GB used, threshold is 6.5GB, flipped sessions to SSD. One minute later
`transcode_cleanup.sh` ran and removed 4GB of stale segment files from ended sessions.
Actual usage was 2.8GB. Sessions were now on SSD for no reason. Users experiencing
slightly worse performance. The flip counter incremented for nothing.
The fix: `transcode_management.sh` runs cleanup first, manager second, every cycle.
The manager always sees post-cleanup usage. Stale files can't trigger a flip because
they're gone before the manager looks. The correct order requires exactly one
orchestrator to enforce it.
---
### 🔴 Failed Imports Sitting Stalled for Days
A release downloads successfully but Lidarr can't import it — wrong format, incorrect
tags, file already exists. Lidarr marks it `importFailed` and stops. Nobody notices.
The download client has the file, Lidarr has given up, and nothing is going to happen
until someone manually opens Lidarr, identifies the problem, blocklists the release,
and triggers a new search. This takes minutes to do — but nobody does it at 3am
when it usually happens.
The fix: `arrs_failed_stalled_recovery.sh` runs every 6 hours. It finds all
`importFailed`, `importPending`, `error`, and `stalled` items, blocklists them, removes
them from the queue, and triggers a new search — automatically. By morning the failed
import has already been replaced by a working one. No manual intervention required.
---
### 🔴 Array Start Scripts Running in Wrong Order or Not at All
Startup scripts configured individually ran in an unpredictable order. The ramdisk
setup might run after Emby starts. The syslog filter might run after containers have
already created veth interfaces. PHP-FPM tuning might run after the WebGUI has already
served its first requests. Each script competed for the same startup slot with no
guaranteed order.
The fix: `array_started.sh` is the only array-start entry in the Varaverk scheduler.
It launches every startup script in a defined order, with one-second settle between
each, and reports which succeeded and which failed. Order is guaranteed. Nothing starts
before its dependency. Everything is visible in a single summary.
---
## ━━━ THE ORCHESTRATOR MODEL ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
The Varaverk scheduler contains exactly these entries:
```bash
# Array start event (Varaverk disks_mounted hook → cron: "array_start"):
array_started.sh
# Cron — managed via Varaverk Scheduler:
*/7 * * * * transcode_management.sh
*/15 * * * * watchdog_orchestrator.sh ← resource → docker → system → stability
*/30 * * * * critical_sync_maintenance.sh ← auth + Emby dirty sync + partnership
0 */4 * * * intermediate_sync_maintenance.sh ← arr sync + failed recovery + optional rsync
0 1 * * * daily_sync_maintenance.sh
0 7 * * 0 sunday_morning_coffee_report.sh
30 2 * * 0 weekly_sync_maintenance.sh
0 0 15 * * monthly_maintenance.sh ← uptime-gated: ZFS scrub, SMART tests
# Manual only (not scheduled):
fallback_test.sh, emby_database_repair.sh, repair tools
```
Every job list is configured in the `ORCHESTRATORS` section of `master.conf`.
No orchestrator script ever changes when jobs are added or removed — only `master.conf` changes.
---
## ━━━ THE ORCHESTRATOR PATTERN ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
All orchestrators follow the same structure:
```
1. Setup → validate config, detect_hosts(), acquire_lock
2. Pre-flight → fail fast checks before any work begins
3. Job loop → run each job, track pass/fail, continue on failure
4. Summary → one clean report of all job results
5. Notification → one notify per run on failure (never per job)
```
Properties that apply to every orchestrator:
```
Consistent output → every orchestrator looks the same in logs
No silent failures → pass/fail tracked per job, all in summary
Resilient → one job failing does not stop the rest
Single notification → one bell per run, not one per job
--dry-run cascade → passes --dry-run through to every child script
--status support → show configured jobs and exit
```
---
## ━━━ OUTPUT TIERS ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
All scripts use a two-tier output model: `echo` lines are always visible; `log`
lines only appear when `--log` is passed.
**One-shot orchestrators** (`array_started.sh`, `array_stopping.sh`, `sunday_morning_coffee_report.sh`):
without `--log`, section headers, per-phase results, and the final summary are
visible. Per-item detail suppressed.
**Periodic orchestrators** (`critical_sync_maintenance.sh`, `daily_sync_maintenance.sh`,
`intermediate_sync_maintenance.sh`, `weekly_sync_maintenance.sh`): without `--log`,
phase headers, per-phase completion status, and the final summary are visible. Per-share
and per-job detail suppressed.
**High-frequency orchestrators** (`watchdog_orchestrator.sh`, `transcode_management.sh`): silent
during clean cycles. Only state transitions, errors, and startup-grace expiry shown
without `--log`.
---
## ━━━ SCRIPTS AT A GLANCE ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
| Script | What It Orchestrates | Schedule |
|--------|---------------------|----------|
| `array_started.sh` | All array startup scripts in order | `array_start` event (Varaverk plugin hook) |
| `watchdog_orchestrator.sh` | resource → docker → system → api_renew → stability watchdogs | Every 15 minutes |
| `transcode_management.sh` | Cleanup then manager — order critical | Every 7 minutes |
| `daily_sync_maintenance.sh` | git pull → sync → media maintenance → restarts | 1am daily |
| `weekly_sync_maintenance.sh` | Stop → update → clean sync → start → weekly restarts | 2:30am Sunday |
| `monthly_maintenance.sh` | Uptime-triggered heavy tasks — ZFS scrub, SMART tests | Daily check, fires when uptime ≥ 30d |
| `intermediate_sync_maintenance.sh` | arr sync + arrs_failed_stalled_recovery + optional rsync | Every 4 hours |
---
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
## 🚀 array_started.sh
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
Single array-start entry for the entire ecosystem. Fired by the Varaverk plugin's
`disks_mounted` event hook. Launches every startup script in order — each as a
background process — and reports which succeeded and which failed.
```bash
# Triggered by: Plugin/unraid/event/disks_mounted/array_start_jobs
# schedule.json entry: "Orchestrators/array_started.sh" → cron: "array_start"
```
---
### ── Execution Order ──────────────────────────────────────────────────────────
```bash
# master.conf
# ─────────────────────────────────────────────────────────────────────────────
# Order matters — each entry depends on the previous ones having run.
# See comments for why each position is correct.
#
ARRAY_START_SCRIPTS=(
# ── One-shot scripts — run and exit naturally ─────────────────────────────
"System_Essentials/unraid_api_key_renew.sh" # re-register API key FIRST — unraid-api
# registry is ephemeral, lost on service restart
"System_Essentials/inotify_tuning.sh" # raise inotify BEFORE containers start
# containers inherit limits at startup —
# if Code-Server starts with low limits
# it keeps them until restart
"System_Essentials/docker_syslog_filter.sh" # suppress veth noise BEFORE containers create
# veth interfaces — otherwise the first boot
# always has unfiltered veth spam
"System_Essentials/php_fpm_max_children.sh" # WebGUI tuning — before any WebGUI requests
"Transcodes/ramdisk_setup.sh" # create tmpfs + symlink BEFORE Emby starts —
# Emby needs the transcode path to exist
"Docker_Essentials/docker_network_connect.sh" # ensure networks + connections BEFORE
# watchdogs check container states
# ── Continuous scripts — run until array stops ─────────────────────────────
"Fallback/fallback.sh" # fallback LAST — needs everything else stable
)
# NOTE: watchdogs (docker_watchdog, system_watchdog, stability_watchdog) are NOT here.
# They run via watchdog_orchestrator.sh on cron every 15 minutes — not as daemons.
```
---
### ── One-Shot vs Continuous Detection ───────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# Each script is launched with `bash script.sh &` — background process.
# After 1 second:
# PID still alive → continuous script (running in background)
# logged as: "fallback.sh — running (PID 12345)"
# PID dead + exit 0 → one-shot completed successfully
# logged as: "inotify_tuning.sh — completed (one-shot)"
# PID dead + exit N → failure
# logged as: "ramdisk_setup.sh — exited with code 1"
# full path printed — debugging is immediate
#
# This means the orchestrator correctly identifies and reports all startup
# scripts without needing to know in advance which ones are continuous.
```
---
### ── Auto-Fix Permissions ────────────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# Scripts that are not executable are chmod +x'd automatically before launch.
# A permissions problem on a startup script does not cause a silent skip —
# the orchestrator fixes it and proceeds, then logs that it did so.
# This prevents "why didn't X run on startup" questions.
```
---
### ── Usage ───────────────────────────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# Normal — fired by Varaverk disks_mounted event hook. Never run manually in production.
# array_started.sh runs once and exits — the continuous scripts it launched
# keep running as background processes.
# ─────────────────────────────────────────────────────────────────────────────
array_started.sh
# ─────────────────────────────────────────────────────────────────────────────
# Dry run — show what would be launched, in order, without launching anything.
# Use to verify the ARRAY_START_SCRIPTS list before an array restart.
# ─────────────────────────────────────────────────────────────────────────────
array_started.sh --dry-run
# ─────────────────────────────────────────────────────────────────────────────
# Status — show each configured script with its current running state.
# RUNNING (PID XXXXX) — continuous script currently active
# not running — one-shot that has completed, or continuous not yet started
# FILE NOT FOUND — script path wrong or missing
# ─────────────────────────────────────────────────────────────────────────────
array_started.sh --status
```
---
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
## 🎬 transcode_management.sh
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
Runs `transcode_cleanup.sh` then `transcode_manager.sh` in the correct order every
7 minutes. Exists because the order is not optional — the manager must always see
post-cleanup usage to make accurate flip decisions.
```bash
# Scheduled: */7 * * * * (every 7 minutes)
```
---
### ── Why Order Is Non-Negotiable ─────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# Emby writes segment files to the transcode directory as it buffers streams.
# When a stream ends, Emby deletes its own active files — but may leave behind
# stale segment files from sessions that ended uncleanly. These consume real
# ramdisk space. The manager has no way to know if they're active or stale.
#
# Without correct order:
# Manager runs → sees 6.8GB used (stale files inflating) → exceeds threshold
# → flips sessions to SSD → flip counter incremented
# Cleanup runs → removes 4GB of stale files → actual usage was 2.8GB
# → flip was unnecessary — sessions now on SSD for no reason
#
# With correct order (this orchestrator):
# Cleanup runs → removes stale files → actual usage 2.8GB
# Manager runs → sees 2.8GB → below threshold → stays on ramdisk ✅
# → no flip, no wasted counter, correct decision every time
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── What Each Child Script Does ────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# transcode_cleanup.sh:
# Identifies segment files not currently open by any process (via lsof)
# Removes them from the transcode directory
# If usage drops enough after cleanup → triggers flip-back to ramdisk
# (handles the recovery direction so manager doesn't have to)
#
# transcode_manager.sh:
# Reads current ramdisk usage after cleanup has run
# Compares against RAMDISK_WARN_GB threshold
# Flips the symlink if needed (ramdisk → SSD or SSD → ramdisk)
# Writes one entry to TRANSCODE_DAILY_LOG for the weekly coffee report
# Shows active Emby sessions with their play method
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Daily Log ────────────────────────────────────────────────────────────────
```bash
# master.conf
# ─────────────────────────────────────────────────────────────────────────────
# transcode_manager.sh writes to TRANSCODE_DAILY_LOG after each cycle.
# Format: DATE|RAMDISK_USED_GB|FLIP_COUNT|RAM_SESSIONS|SSD_SESSIONS
#
# The sunday_morning_coffee_report.sh reads this log for weekly stats:
# Peak ramdisk usage across the week
# Total flip count (unnecessary flips visible here)
# Session split: how often ramdisk vs SSD was used
#
# Log trimmed to TRANSCODE_LOG_RETENTION days on every write — bounded, never grows.
#
TRANSCODE_DAILY_LOG="$DATA_DIR/transcode_daily.db"
TRANSCODE_LOG_RETENTION=90 # days
TRANSCODE_STATE_FILE="/tmp/transcode_state.db" # /tmp — resets on reboot
```
---
### ── Usage ───────────────────────────────────────────────────────────────────
```bash
transcode_management.sh # normal run (every 7 min via cron)
transcode_management.sh --dry-run # passes --dry-run to both child scripts
transcode_management.sh --status # show config, current state, daily log stats
transcode_management.sh --log # verbose output from both child scripts
```
---
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
## 🔄 arrs_failed_stalled_recovery.sh
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
Automatic recovery from failed imports and stalled downloads across Sonarr, Radarr,
and Lidarr. Blocklists the bad release, removes it from the queue, and triggers a
new search — hands-free recovery while you sleep.
```bash
# Scheduled: 0 */6 * * * (every 6 hours)
```
---
### ── Four Problem Types ───────────────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# The arr queue API reports these status values for problem items:
#
# importFailed — downloaded successfully but arr couldn't import the file
# Common causes: wrong format for quality profile, corrupted file,
# duplicate already in library, permission issue on import path
# Self-resolution: never — arr stops trying after first failure
#
# importPending — downloaded, stuck waiting for import to begin
# Common causes: import queue backed up, arr paused, API error
# Self-resolution: sometimes — but stuck for hours is always wrong
#
# error — serious failure state not covered by the above
# Common causes: indexer issues, download client unreachable, disk full
#
# stalled — download stuck with no connections or no progress
# Common causes: no seeders, VPN routing issue, tracker ban
# Self-resolution: never without a source change
#
# NOT touched — items with status "downloading" or "imported" are never touched.
# Safe to run at any time — only processes items that are already broken.
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── What It Does Per Problem Item ──────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# For each problem item found, in order:
#
# 1. Blocklist the release
# Prevents the arr from grabbing the exact same release again immediately.
# The bad indexer result goes into the blocklist — future searches skip it.
#
# 2. Remove from queue
# Tells the download client to stop and remove the failed download.
# Frees up the slot for the replacement.
#
# 3. Trigger new search
# Arr searches for a different release meeting the quality profile.
# If a suitable alternative exists, it starts downloading automatically.
# If not, the item is marked as "awaiting upgrade" — arr will retry when
# a new indexer result appears.
#
# The entire cycle from "failed import" to "replacement downloading" happens
# without any human involvement.
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Age Threshold ────────────────────────────────────────────────────────────
```bash
# master.conf
# ─────────────────────────────────────────────────────────────────────────────
# Items newer than ARR_IMPORT_RECOVERY_AGE hours are skipped.
# Arrs retry on their own schedule after initial failures — a 30-minute-old
# importFailed may still resolve itself. Waiting 6 hours before intervening
# gives the arr a full retry cycle before this script steps in.
#
ARR_IMPORT_RECOVERY_AGE=6 # hours — skip items newer than this
```
---
### ── Host Awareness ───────────────────────────────────────────────────────────
```bash
# master.conf + host*.conf
# ─────────────────────────────────────────────────────────────────────────────
# Each arr is independently toggled per host.
# Lidarr only runs on HOST1 — exits cleanly on HOST2 with no action.
# HOST2 has its own Sonarr and Radarr for its anime shares.
#
HOST1_SONARR_RECOVERY=true # HOST1 Sonarr — Tv_Shows
HOST1_RADARR_RECOVERY=true # HOST1 Radarr — Movies
HOST1_LIDARR_RECOVERY=true # HOST1 Lidarr — Music (HOST1 only)
HOST2_SONARR_RECOVERY=true # HOST2 Sonarr — Anime_Shows
HOST2_RADARR_RECOVERY=true # HOST2 Radarr — Anime_Movies
#
# API versions (current — update if arr major version changes):
# Sonarr v4 → /api/v3/
# Radarr v6 → /api/v3/
# Lidarr v3 → /api/v1/
```
---
### ── Usage ───────────────────────────────────────────────────────────────────
```bash
arrs_failed_stalled_recovery.sh # normal run
arrs_failed_stalled_recovery.sh --dry-run # show what would be actioned, no changes
arrs_failed_stalled_recovery.sh --log # verbose — show each queue item evaluated
arrs_failed_stalled_recovery.sh --status # show arr config and API connectivity
```
---
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
## 📅 daily_sync_maintenance.sh
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
Full daily maintenance window orchestrator. The entire 1am window — git pull, media
share sync, media maintenance, and docker daily restarts — in one scheduled entry.
Runs on both servers; `detect_hosts()` determines which direction each sync goes.
```bash
# Scheduled: 0 1 * * * (1am daily — on BOTH servers)
```
---
### ── Execution Order ──────────────────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# The split into pre-sync and post-sync is based on a simple rule:
# git pull runs before anything — maintenance uses the latest scripts
# rsync runs in the middle — media management uses the post-sync state
# docker restart runs last — after everything else has completed
#
# 1. Pre-sync: git_pull_execute.sh
# Pull latest scripts first. Everything that follows runs on current code.
#
# 2. Rsync window: HOST*_DAILY_SYNC_SHARES + HOST*_PERSONAL_SHARES
# Each server pushes its own truth shares to the other.
# HOST1 → pushes Movies, Tv_Shows, Music → HOST2
# HOST2 → pushes Anime_Shows, Anime_Movies → HOST1
# Personal encrypted shares appended after standard shares.
# Drive temperature exit codes respected — skip share or abort all on CRIT.
#
# 3. Post-sync: DAILY_MAINTENANCE_SCRIPTS (everything except git pull)
# media_management.sh → permissions + cleaners + arr cleanup
# docker_daily_restart.sh → nightly container restarts
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Bidirectional — Same Script, Correct Direction Automatic ────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# detect_hosts() sets MY_ID at runtime and aliases HOST*_DAILY_SYNC_SHARES
# to DAILY_SYNC_SHARES. The script uses DAILY_SYNC_SHARES — always the right
# list for whichever server is running.
#
# No HOST1/HOST2 comparisons in the script. Configuration drives direction.
#
# HOST1 runs this script at 1am:
# → pushes HOST1_DAILY_SYNC_SHARES (Movies, Tv_Shows, Music) → HOST2
# → media_management.sh on HOST1's shares
# → docker_daily_restart.sh on HOST1's containers
#
# HOST2 runs this script at 1am:
# → pushes HOST2_DAILY_SYNC_SHARES (Anime_Shows, Anime_Movies) → HOST1
# → media_management.sh on HOST2's shares
# → docker_daily_restart.sh on HOST2's containers
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Configuration ────────────────────────────────────────────────────────────
```bash
# master.conf
# ─────────────────────────────────────────────────────────────────────────────
# Script lists — git_pull is split out as pre-sync; everything else is post-sync.
# The orchestrator recognises git_pull_execute.sh by name and routes it correctly.
#
DAILY_MAINTENANCE_SCRIPTS=(
"git_pull_execute.sh" # PRE-SYNC — always first
"Media/media_shares_permissions.sh" # POST-SYNC — permissions before arr
"Media/media_cleaner.sh anime" # POST-SYNC — junk before orphan scan
"Media/media_cleaner.sh media" # POST-SYNC
"Media/lidarr_cleanup.sh" # POST-SYNC — orphan cleanup last
"Media/sonarr_cleanup.sh" # POST-SYNC
"Media/radarr_cleanup.sh" # POST-SYNC
"Docker_Essentials/docker_daily_restart.sh" # POST-SYNC — restarts after everything
)
# host1.conf
HOST1_DAILY_SYNC_SHARES=(
"/mnt/user/Movies" # both servers — arr_sync union, rsync spreads files
"/mnt/user/Tv_Shows" # both servers
"/mnt/user/Music" # both servers
"/mnt/user/Kids_Movies"
"/mnt/user/Kids_Tv_Shows"
"/mnt/user/Sports"
"/mnt/user/stand-up_comedy"
)
HOST1_PERSONAL_SHARES=(
"/mnt/user/Personal" # encrypted personal share — appended after standard
)
# host2.conf
HOST2_DAILY_SYNC_SHARES=(
"/mnt/user/Anime_Shows" # both servers — arr_sync union, rsync spreads files
"/mnt/user/Anime_Movies" # both servers
)
```
---
### ── Adding or Removing a Job ────────────────────────────────────────────────
```bash
# master.conf
# ─────────────────────────────────────────────────────────────────────────────
# Add a new media script — just insert it in the right position:
DAILY_MAINTENANCE_SCRIPTS=(
"git_pull_execute.sh"
"Media/media_shares_permissions.sh"
"Media/media_cleaner.sh anime"
"Media/media_cleaner.sh media"
"Media/my_new_script.sh" # ← add here in the correct order
"Media/lidarr_cleanup.sh"
"Media/sonarr_cleanup.sh"
"Media/radarr_cleanup.sh"
"Docker_Essentials/docker_daily_restart.sh"
)
# Disable a job temporarily — comment it out, do not delete:
DAILY_MAINTENANCE_SCRIPTS=(
"git_pull_execute.sh"
"Media/media_shares_permissions.sh"
# "Media/media_cleaner.sh anime" # ← temporarily disabled
"Media/media_cleaner.sh media"
"Media/lidarr_cleanup.sh"
...
)
# ─────────────────────────────────────────────────────────────────────────────
# No changes to daily_sync_maintenance.sh needed in either case.
```
---
### ── Drive Temperature Exit Codes ───────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# rsync.sh returns specific exit codes for temperature issues.
# The orchestrator handles these correctly — one hot drive does not abort all others.
#
# exit 0 → success — continue to next share
# exit 1 → temp WARN — skip this share, continue to next share
# exit 2 → temp CRITICAL — abort ALL remaining shares in this window
# notify immediately with which share triggered the abort
# exit N → other failure — skip this share, continue to next share
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Relationship to Fallback Writeback ──────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# The same HOST*_DAILY_SYNC_SHARES lists are used by fallback.sh for Tier 4
# writeback — but in the opposite direction.
#
# Normal (daily_sync_maintenance.sh):
# HOST1 → pushes Movies, Tv_Shows → HOST2
#
# Tier 4 fallback writeback (HOST1 returns after 24hr+ outage):
# HOST2 → pushes Movies, Tv_Shows → HOST1
# (HOST2 was running HOST1's arrs and accumulated content)
#
# Same list, correct direction for the situation, zero duplication.
# No separate writeback list to maintain.
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Usage ───────────────────────────────────────────────────────────────────
```bash
daily_sync_maintenance.sh # normal run
daily_sync_maintenance.sh --dry-run # preview all jobs without syncing or changing
daily_sync_maintenance.sh --log # verbose per-share and per-job output
daily_sync_maintenance.sh --status # show configured shares and jobs for this host
```
---
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
## 📅 weekly_sync_maintenance.sh
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
Weekly maintenance window orchestrator — clean sync of Emby and auth stack, container
image updates, and weekly docker restarts. Runs Sunday 2:30am; fits inside the Sunday
maintenance block before the 7am coffee report.
```bash
# Scheduled: 30 2 * * 0 (Sunday 2:30am)
```
---
### ── Execution Order ──────────────────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# Containers stop BEFORE sync — clean static source, full bandwidth.
# Containers start AFTER sync — on fresh data, in dependency order.
# DDNS and fallback continue running throughout — only managed containers stop.
#
# 1. Pre-flight checks — connectivity, remote Docker daemon, remote rootfs
# 2. Stop local containers — Emby + auth stack stopped on this server
# 3. Stop remote containers — Emby + auth stack stopped on remote via SSH
# 4. Pull updates locally — if WEEKLY_SYNC_UPDATES=true (containers already stopped)
# 5. Pull updates remotely — if WEEKLY_SYNC_UPDATES_REMOTE=true
# 6. rsync WEEKLY_SYNC_SHARES — full clean mirror at full bandwidth
# 7. Start remote containers — new image, correct dependency order
# 8. Start local containers — new image, correct dependency order
# Post-sync jobs: docker_weekly_restart.sh
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Why Weekly Not Nightly for Emby ─────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# Two Emby syncs run in parallel — dirty and clean:
#
# emby-fallback dirty sync (every 30 minutes via critical_sync_maintenance.sh, Emby running):
# watch states, library deltas, user activity — continuous coverage
# WAL files excluded — safe to copy while Emby writes
# HOST2 always within 30 minutes of HOST1 on playback state
#
# weekly clean sync (Sunday 2:30am, Emby stopped):
# Full clean mirror — all databases checkpointed and flushed
# Metadata, plugins, config all included
# ~30 seconds of Emby downtime — both instances stopped during rsync only
#
# Why not nightly:
# Emby builds a warm image thumbnail cache on HOST2 throughout the week.
# Nightly sync resets this cache — cold loads every morning for users.
# Weekly sync: cache stays warm for 6 days. Resets Sunday night while users sleep.
# One weekly reset at an acceptable time is better than six unnecessary resets.
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Container Update Window ─────────────────────────────────────────────────
```bash
# master.conf
# ─────────────────────────────────────────────────────────────────────────────
# Containers are already stopped for the sync — pulling updated images costs
# nothing extra in downtime. Both servers start on the same image version
# after the window completes.
#
WEEKLY_SYNC_UPDATES=true # pull updated images locally
WEEKLY_SYNC_UPDATES_REMOTE=true # pull updated images on remote via SSH
#
# Toggle false to skip updates without changing the schedule:
# WEEKLY_SYNC_UPDATES=false # skips pulls, containers still restart on current image
```
---
### ── Why Auth Stack Weekly Sync Matters ──────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# NPM, Authelia, LLDAP, Mariadb run warm on both servers continuously.
# HOST1 is source of truth — changes propagate to HOST2 via weekly clean sync.
#
# What propagates automatically every Sunday:
# New users added in LLDAP on HOST1 → appear on HOST2 by Monday
# Proxy rules changed in NPM on HOST1 → live on HOST2 by Monday
# Authelia policies updated on HOST1 → enforced on HOST2 by Monday
# TLS certificates renewed on HOST1 → valid on HOST2 by Monday
#
# No manual sync needed for routine auth administration.
# Anything done on HOST1 is on HOST2 within a week.
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Sunday Maintenance Window ───────────────────────────────────────────────
```bash
# ─────────────────────────────────────────────────────────────────────────────
# This script is part of a coordinated Sunday maintenance block:
#
# 2:30am weekly_sync_maintenance.sh ← clean sync + image updates (~5-10min)
# 2:50am CA Auto Update plugin ← unRAID plugin updates
# 2:55am CA container updates ← additional container updates
# 3:00am Network reboot ← router/switch restart
#
# Everything comes back clean:
# Network fresh, Emby + auth updated, containers on latest images.
# All in one window while users sleep.
# Sunday morning coffee report at 7am shows the post-maintenance state.
# ─────────────────────────────────────────────────────────────────────────────
```
---
### ── Configuration ────────────────────────────────────────────────────────────
```bash
# master.conf
WEEKLY_SYNC_SHARES=(
"/mnt/user/Media_Server/Emby" # emby profile — full clean mirror
"/mnt/user/appdata-Fallback/Critical-Data" # critical-data — auth stack clean state
)
WEEKLY_MAINTENANCE_SCRIPTS=(
"Docker_Essentials/docker_weekly_restart.sh" # weekly restart of less-critical services
)
WEEKLY_SYNC_UPDATES=true
WEEKLY_SYNC_UPDATES_REMOTE=true
```
---
### ── Usage ───────────────────────────────────────────────────────────────────
```bash
weekly_sync_maintenance.sh # normal run
weekly_sync_maintenance.sh --dry-run # preview — no stops, no syncs, no starts
weekly_sync_maintenance.sh --log # verbose per-share per-job output
weekly_sync_maintenance.sh --status # show configured shares, jobs, update toggles
```
---
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
## 📆 monthly_maintenance.sh
## ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
Uptime-triggered orchestrator for long-running system tasks — ZFS scrub, SMART
long tests — that should only run on stable systems that have been up for at least
30 days. Called daily by cron; most invocations are silent no-ops.
Two gates must both pass before any job runs:
1. Server uptime ≥ `MONTHLY_UPTIME_THRESHOLD_DAYS`
2. Last run ≥ `MONTHLY_RUN_INTERVAL_DAYS` ago (state file on `/boot/config/` — survives reboots)
If either gate fails, the script exits 0 with no output. This is expected — it runs
daily and most days are no-ops.
`--force` bypasses both gates and runs the job list immediately. Use for testing or
when a scrub was missed and the server hasn't reached the uptime threshold yet.
### Configuration (master.conf)
```bash
MONTHLY_MAINTENANCE_SCRIPTS=(
#"Tools/zfs_pool_scrub.sh"
#"Tools/smart_long_test.sh"
)
MONTHLY_UPTIME_THRESHOLD_DAYS=30
MONTHLY_RUN_INTERVAL_DAYS=30
MONTHLY_LAST_RUN_FILE="/boot/config/monthly_maintenance_last_run.db"
```
Scripts are commented out by default — uncomment what applies to your hardware.
### Usage
```bash
monthly_maintenance.sh # normal run (daily cron — silent no-op when gates not met)
monthly_maintenance.sh --force # bypass both gates — run immediately
monthly_maintenance.sh --dry-run # show what would run without running it
monthly_maintenance.sh --status # show gate state: uptime, last run, next eligible
monthly_maintenance.sh --log # verbose output from each child script
```
---
## ━━━ COMPLETE SCHEDULE ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
```bash
# ─────────────────────────────────────────────────────────────────────────────
# Array start event (Varaverk disks_mounted hook):
# ─────────────────────────────────────────────────────────────────────────────
array_started.sh
# ─────────────────────────────────────────────────────────────────────────────
# Every 7 minutes:
# ─────────────────────────────────────────────────────────────────────────────
*/7 * * * * transcode_management.sh
# ─────────────────────────────────────────────────────────────────────────────
# Every 15 minutes — watchdog cycle:
# ─────────────────────────────────────────────────────────────────────────────
*/15 * * * * watchdog_orchestrator.sh
# ─────────────────────────────────────────────────────────────────────────────
# Every 30 minutes — auth stack + Emby dirty sync + partnership check:
# ─────────────────────────────────────────────────────────────────────────────
*/30 * * * * critical_sync_maintenance.sh
# ─────────────────────────────────────────────────────────────────────────────
# Every 4 hours — arr library sync + failed import recovery + optional rsync:
# ─────────────────────────────────────────────────────────────────────────────
0 */4 * * * intermediate_sync_maintenance.sh
# ─────────────────────────────────────────────────────────────────────────────
# Daily — 1am:
# git pull → rsync all truth shares → permissions + cleaners + arr cleanup → docker restart
# ─────────────────────────────────────────────────────────────────────────────
0 1 * * * daily_sync_maintenance.sh
# ─────────────────────────────────────────────────────────────────────────────
# Weekly — Sunday maintenance block:
# ─────────────────────────────────────────────────────────────────────────────
30 2 * * 0 weekly_sync_maintenance.sh # clean sync + updates (~5-10min)
50 2 * * 0 CA Auto Update plugin # plugin updates
55 2 * * 0 CA container updates # container image updates
0 3 * * 0 Network reboot # router/switch restart
0 7 * * 0 sunday_morning_coffee_report.sh
# ─────────────────────────────────────────────────────────────────────────────
# 15th of each month (uptime-gated — silent no-op if uptime < 30 days):
# ─────────────────────────────────────────────────────────────────────────────
0 0 15 * * monthly_maintenance.sh
```
---
## ━━━ ADDING A NEW ORCHESTRATOR ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
If you find yourself running 3+ related scripts on the same schedule, wrap them
in a new orchestrator. Model directly on `media_management.sh` which has the
complete pattern — dry-run passthrough, status display, pass/fail tracking, summary.
```bash
# Minimal skeleton — the full pattern in its simplest form:
#!/bin/bash
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
SCRIPTS_ROOT="$SCRIPT_DIR/.."
PASS=()
FAIL=()
# Read job list from master.conf — never hardcode jobs in the orchestrator
for script_entry in "${MY_MAINTENANCE_JOBS[@]:-}"; do
[[ -z "$script_entry" ]] && continue
read -r -a parts <<< "$script_entry"
script_path="$SCRIPTS_ROOT/${parts[0]}"
script_name=$(basename "${parts[0]}")
extra_args=("${parts[@]:1}")
[[ "$DRY_RUN" == true ]] && extra_args+=("--dry-run")
if bash "$script_path" "${extra_args[@]}"; then
PASS+=("$script_name")
else
FAIL+=("$script_name")
fi
done
# One summary — one notification
echo "Passed: ${#PASS[@]} Failed: ${#FAIL[@]}"
[[ ${#FAIL[@]} -gt 0 ]] && \
notify "My maintenance failed on $(hostname) ($MY_ID) — ${FAIL[*]}" \
"My Orchestrator" "warning"
```
@@ -1,335 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ================================= Ramdisk Stop ===============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Safely stops the transcode ramdisk: redirects the transcode symlink to the
# SSD fallback before unmounting so Emby continues writing without interruption,
# then unmounts the tmpfs and updates the state file.
#
# Primary use case: stopping the current ramdisk before re-running
# ramdisk_setup.sh with new size or threshold values (setup is idempotent —
# if the ramdisk is mounted, it skips the mount and reports status, so you
# must stop it first to change the size).
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# Executes in safe order:
# 1. Validate — ramdisk mounted, SSD fallback exists
# 2. Redirect symlink → SSD (Emby immediately writes to SSD instead)
# 3. Warn if active transcode files still on ramdisk (informational — not a blocker)
# 4. Unmount ramdisk tmpfs
# 5. Update /tmp/transcode_state.db → current_target=TRANSCODE_SSD
#
# The symlink redirect happens before unmount so there is no window where Emby
# has nowhere to write. Existing in-progress transcode files on the ramdisk are
# lost on unmount — warn the user but proceed (this is expected for maintenance).
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Root Required
# umount requires root.
#
# Single Instance Lock
# acquire_lock prevents concurrent stop attempts.
#
# Mounted Check
# Exits cleanly if ramdisk is not mounted — nothing to do.
#
# Symlink-First Order
# Symlink is redirected before unmount — Emby never sees a broken path.
#
# transcode_manager Warning
# Warns if transcode_manager.sh is running — it may flip the symlink back
# to ramdisk on its next cycle. Stop transcode_manager before running this
# if you need the SSD redirect to hold.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
# HOST*_TRANSCODE_SSD SSD fallback directory — redirect target during stop.
# Aliased by detect_hosts() → TRANSCODE_SSD.
#
# master.conf
# TRANSCODE_LINK Symlink Emby uses. Must match Emby's transcode path setting.
# RAMDISK_PATH tmpfs mount point.
#
# ==============================================================================================
# STATE FILES
# ==============================================================================================
#
# /tmp/transcode_state.db — updated to current_target=TRANSCODE_SSD after stop.
# transcode_manager.sh reads this on its next cycle.
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# ramdisk_stop.sh
# Stop the ramdisk: redirect symlink → SSD, unmount, update state.
#
# ramdisk_stop.sh --dry-run
# Show what would happen without making any changes.
#
# ramdisk_stop.sh --status
# Show current mount state, symlink target, active files on ramdisk. Exit.
#
# ramdisk_stop.sh --log
# Verbose output for each step.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root — umount requires root"
exit 1
fi
acquire_lock
detect_hosts
STATE_FILE="${TRANSCODE_STATE_FILE:-${STATE_DIR:-/tmp}/transcode_state.db}"
log "Identity: $MY_ID ($LOCAL_SERVER_NAME)"
log "Ramdisk: $RAMDISK_PATH"
log "Fallback: $TRANSCODE_SSD"
log "Symlink: $TRANSCODE_LINK"
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes will be made"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_RAM Ramdisk path: $RAMDISK_PATH"
echo "$ICON_DISK SSD fallback: $TRANSCODE_SSD"
echo "$ICON_LINK Symlink: $TRANSCODE_LINK"
echo ""
if mountpoint -q "$RAMDISK_PATH" 2>/dev/null; then
USAGE=$(df -BG "$RAMDISK_PATH" | awk 'NR==2 {print $3}')
AVAIL=$(df -BG "$RAMDISK_PATH" | awk 'NR==2 {print $4}')
echo " $ICON_RAM Ramdisk: mounted — $USAGE used / $AVAIL available ✅"
FILE_COUNT=$(find "$RAMDISK_PATH" -type f 2>/dev/null | wc -l)
echo " $ICON_RAM Active files on ramdisk: $FILE_COUNT"
else
echo " $ICON_RAM Ramdisk: NOT mounted"
fi
if [[ -L "$TRANSCODE_LINK" ]]; then
TARGET=$(readlink "$TRANSCODE_LINK")
echo " $ICON_LINK Symlink: $TRANSCODE_LINK → $TARGET"
else
echo " $ICON_LINK Symlink: not set"
fi
if [[ -f "$STATE_FILE" ]]; then
echo ""
echo " State file ($STATE_FILE):"
while IFS='=' read -r key value; do
[[ -z "$key" ]] && continue
echo " $key = $value"
done < "$STATE_FILE"
else
echo " $ICON_INFO State file: not found (ramdisk never started this boot)"
fi
if pgrep -f "transcode_manager.sh" >/dev/null 2>&1; then
echo ""
warn "transcode_manager.sh is currently RUNNING"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ━━━ Preflight ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_GEAR Preflight ━━━"
# Bail if not mounted — nothing to do
if ! mountpoint -q "$RAMDISK_PATH" 2>/dev/null; then
warn "Ramdisk is not mounted at $RAMDISK_PATH — nothing to stop"
exit 0
fi
log "Ramdisk is mounted ✅"
# Warn if transcode_manager is running — it may flip symlink back on next cycle
if pgrep -f "transcode_manager.sh" >/dev/null 2>&1; then
warn "transcode_manager.sh is currently RUNNING"
warn "It may flip the symlink back to ramdisk on its next cycle"
warn "Stop transcode_manager.sh first if you need the SSD redirect to hold"
echo ""
fi
# Confirm SSD fallback exists
if [[ ! -d "$TRANSCODE_SSD" ]]; then
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — SSD fallback does not exist: $TRANSCODE_SSD"
warn "DRY RUN — would create it before redirecting symlink"
else
warn "SSD fallback does not exist — creating: $TRANSCODE_SSD"
mkdir -p "$TRANSCODE_SSD" || {
error "Failed to create SSD fallback: $TRANSCODE_SSD"
error "Cannot safely redirect symlink — aborting"
exit 1
}
log "SSD fallback created ✅"
fi
else
log "SSD fallback exists: $TRANSCODE_SSD ✅"
fi
START=$(date +%s)
STOP_SUCCESS=true
# ==============================================================================================
# ━━━ Redirect Symlink → SSD ━━━
# ==============================================================================================
# Redirect BEFORE unmount — Emby continues writing to SSD with no broken path window.
echo ""
echo "━━━ $ICON_LINK Redirect Symlink → SSD ━━━"
if [[ -L "$TRANSCODE_LINK" ]]; then
CURRENT_TARGET=$(readlink "$TRANSCODE_LINK")
if [[ "$CURRENT_TARGET" == "$TRANSCODE_SSD" ]]; then
echo "Symlink already points to SSD — no change needed ✅"
else
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would redirect: $TRANSCODE_LINK → $TRANSCODE_SSD"
else
ln -sfn "$TRANSCODE_SSD" "$TRANSCODE_LINK" && \
warn "Symlink redirected: $TRANSCODE_LINK → $TRANSCODE_SSD ✅" || {
error "Failed to redirect symlink"
STOP_SUCCESS=false
}
fi
fi
elif [[ -e "$TRANSCODE_LINK" ]]; then
warn "$TRANSCODE_LINK exists but is not a symlink — leaving as-is"
else
warn "Symlink $TRANSCODE_LINK does not exist — nothing to redirect"
fi
# ==============================================================================================
# ━━━ Active Files Warning ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_RAM Active Files Check ━━━"
FILE_COUNT=$(find "$RAMDISK_PATH" -type f 2>/dev/null | wc -l)
if [[ "$FILE_COUNT" -gt 0 ]]; then
warn "⚠️ $FILE_COUNT file(s) still on ramdisk — will be lost on unmount"
warn "Active transcode sessions should be stopped before unmounting"
warn "Proceeding regardless (this is expected for maintenance)"
if [[ "$LOG" == true ]]; then
find "$RAMDISK_PATH" -type f 2>/dev/null | while read -r f; do
log " $f"
done
fi
else
log "No active files on ramdisk ✅"
fi
# ==============================================================================================
# ━━━ Unmount ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_RAM Unmount Ramdisk ━━━"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would unmount: $RAMDISK_PATH"
else
if umount "$RAMDISK_PATH" 2>/dev/null; then
warn "Ramdisk unmounted: $RAMDISK_PATH ✅"
else
# Regular unmount failed — check if only directory handles are open (no active writes)
OPEN_FILES=$(lsof +D "$RAMDISK_PATH" 2>/dev/null | awk 'NR>1 && $5 != "DIR"' | wc -l)
if [[ "$OPEN_FILES" -eq 0 ]]; then
warn "Busy — only directory handles open, no active writes — trying lazy unmount"
if umount -l "$RAMDISK_PATH"; then
warn "Ramdisk lazy-unmounted: $RAMDISK_PATH ✅"
warn "Handles will release when owning processes next check the directory"
else
error "Lazy unmount also failed — $RAMDISK_PATH"
STOP_SUCCESS=false
fi
else
error "Failed to unmount $RAMDISK_PATH — $OPEN_FILES file(s) still open for writing"
error "Stop active transcode sessions and retry"
STOP_SUCCESS=false
fi
fi
fi
# ==============================================================================================
# ━━━ Update State File ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_GEAR State File ━━━"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would update $STATE_FILE: current_target=$TRANSCODE_SSD"
elif [[ "$STOP_SUCCESS" == true ]]; then
NOW=$(date +%s)
cat > "$STATE_FILE" <<EOF
current_target=$TRANSCODE_SSD
last_flip_time=$NOW
flip_count_hour=0
flip_hour_start=$NOW
EOF
log "State file updated: current_target=$TRANSCODE_SSD"
else
warn "Skipping state file update — stop had errors"
fi
END=$(date +%s)
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY RAMDISK STOP SUMMARY ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_RAM Ramdisk: $RAMDISK_PATH"
echo "$ICON_DISK Fallback: $TRANSCODE_SSD"
echo "$ICON_LINK Symlink: $TRANSCODE_LINK → $(readlink "$TRANSCODE_LINK" 2>/dev/null || echo "not set")"
echo "$ICON_TIME Duration: $(format_duration $(( END - START )))"
echo ""
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no changes made"
elif [[ "$STOP_SUCCESS" == true ]]; then
echo "$ICON_DONE Status: done ✅"
echo "Run ramdisk_setup.sh to remount with new configuration"
else
echo "$ICON_ERROR Status: STOP HAD ERRORS"
notify "Ramdisk stop errors on $(hostname) ($MY_ID) — check output" \
"Ramdisk Stop" "warning"
exit 1
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
@@ -1,336 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ================================= Ramdisk Stop ===============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Safely stops the transcode ramdisk: redirects the transcode symlink to the
# SSD fallback before unmounting so Emby continues writing without interruption,
# then unmounts the tmpfs and updates the state file.
#
# Primary use case: stopping the current ramdisk before re-running
# ramdisk_setup.sh with new size or threshold values (setup is idempotent —
# if the ramdisk is mounted, it skips the mount and reports status, so you
# must stop it first to change the size).
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# Executes in safe order:
# 1. Validate — ramdisk mounted, SSD fallback exists
# 2. Redirect symlink → SSD (Emby immediately writes to SSD instead)
# 3. Warn if active transcode files still on ramdisk (informational — not a blocker)
# 4. Unmount ramdisk tmpfs
# 5. Update /tmp/transcode_state.db → current_target=TRANSCODE_SSD
#
# The symlink redirect happens before unmount so there is no window where Emby
# has nowhere to write. Existing in-progress transcode files on the ramdisk are
# lost on unmount — warn the user but proceed (this is expected for maintenance).
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Root Required
# umount requires root.
#
# Single Instance Lock
# acquire_lock prevents concurrent stop attempts.
#
# Mounted Check
# Exits cleanly if ramdisk is not mounted — nothing to do.
#
# Symlink-First Order
# Symlink is redirected before unmount — Emby never sees a broken path.
#
# transcode_manager Warning
# Warns if transcode_manager.sh is running — it may flip the symlink back
# to ramdisk on its next cycle. Stop transcode_manager before running this
# if you need the SSD redirect to hold.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
# HOST*_TRANSCODE_SSD SSD fallback directory — redirect target during stop.
# Aliased by detect_hosts() → TRANSCODE_SSD.
#
# master.conf
# TRANSCODE_LINK Symlink Emby uses. Must match Emby's transcode path setting.
# RAMDISK_PATH tmpfs mount point.
# TRANSCODE_STATE_FILE Override state file path (default: ${STATE_DIR}/transcode_state.db).
#
# ==============================================================================================
# STATE FILES
# ==============================================================================================
#
# /tmp/transcode_state.db — updated to current_target=TRANSCODE_SSD after stop.
# transcode_manager.sh reads this on its next cycle.
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# ramdisk_stop.sh
# Stop the ramdisk: redirect symlink → SSD, unmount, update state.
#
# ramdisk_stop.sh --dry-run
# Show what would happen without making any changes.
#
# ramdisk_stop.sh --status
# Show current mount state, symlink target, active files on ramdisk. Exit.
#
# ramdisk_stop.sh --log
# Verbose output for each step.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root — umount requires root"
exit 1
fi
acquire_lock
detect_hosts
STATE_FILE="${TRANSCODE_STATE_FILE:-${STATE_DIR:-/tmp}/transcode_state.db}"
log "Identity: $MY_ID ($LOCAL_SERVER_NAME)"
log "Ramdisk: $RAMDISK_PATH"
log "Fallback: $TRANSCODE_SSD"
log "Symlink: $TRANSCODE_LINK"
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes will be made"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_RAM Ramdisk path: $RAMDISK_PATH"
echo "$ICON_DISK SSD fallback: $TRANSCODE_SSD"
echo "$ICON_LINK Symlink: $TRANSCODE_LINK"
echo ""
if mountpoint -q "$RAMDISK_PATH" 2>/dev/null; then
USAGE=$(df -BG "$RAMDISK_PATH" | awk 'NR==2 {print $3}')
AVAIL=$(df -BG "$RAMDISK_PATH" | awk 'NR==2 {print $4}')
echo " $ICON_RAM Ramdisk: mounted — $USAGE used / $AVAIL available ✅"
FILE_COUNT=$(find "$RAMDISK_PATH" -type f 2>/dev/null | wc -l)
echo " $ICON_RAM Active files on ramdisk: $FILE_COUNT"
else
echo " $ICON_RAM Ramdisk: NOT mounted"
fi
if [[ -L "$TRANSCODE_LINK" ]]; then
TARGET=$(readlink "$TRANSCODE_LINK")
echo " $ICON_LINK Symlink: $TRANSCODE_LINK → $TARGET"
else
echo " $ICON_LINK Symlink: not set"
fi
if [[ -f "$STATE_FILE" ]]; then
echo ""
echo " State file ($STATE_FILE):"
while IFS='=' read -r key value; do
[[ -z "$key" ]] && continue
echo " $key = $value"
done < "$STATE_FILE"
else
echo " $ICON_INFO State file: not found (ramdisk never started this boot)"
fi
if pgrep -f "transcode_manager.sh" >/dev/null 2>&1; then
echo ""
warn "transcode_manager.sh is currently RUNNING"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ━━━ Preflight ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_GEAR Preflight ━━━"
# Bail if not mounted — nothing to do
if ! mountpoint -q "$RAMDISK_PATH" 2>/dev/null; then
warn "Ramdisk is not mounted at $RAMDISK_PATH — nothing to stop"
exit 0
fi
log "Ramdisk is mounted ✅"
# Warn if transcode_manager is running — it may flip symlink back on next cycle
if pgrep -f "transcode_manager.sh" >/dev/null 2>&1; then
warn "transcode_manager.sh is currently RUNNING"
warn "It may flip the symlink back to ramdisk on its next cycle"
warn "Stop transcode_manager.sh first if you need the SSD redirect to hold"
echo ""
fi
# Confirm SSD fallback exists
if [[ ! -d "$TRANSCODE_SSD" ]]; then
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — SSD fallback does not exist: $TRANSCODE_SSD"
warn "DRY RUN — would create it before redirecting symlink"
else
warn "SSD fallback does not exist — creating: $TRANSCODE_SSD"
mkdir -p "$TRANSCODE_SSD" || {
error "Failed to create SSD fallback: $TRANSCODE_SSD"
error "Cannot safely redirect symlink — aborting"
exit 1
}
log "SSD fallback created ✅"
fi
else
log "SSD fallback exists: $TRANSCODE_SSD ✅"
fi
START=$(date +%s)
STOP_SUCCESS=true
# ==============================================================================================
# ━━━ Redirect Symlink → SSD ━━━
# ==============================================================================================
# Redirect BEFORE unmount — Emby continues writing to SSD with no broken path window.
echo ""
echo "━━━ $ICON_LINK Redirect Symlink → SSD ━━━"
if [[ -L "$TRANSCODE_LINK" ]]; then
CURRENT_TARGET=$(readlink "$TRANSCODE_LINK")
if [[ "$CURRENT_TARGET" == "$TRANSCODE_SSD" ]]; then
echo "Symlink already points to SSD — no change needed ✅"
else
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would redirect: $TRANSCODE_LINK → $TRANSCODE_SSD"
else
ln -sfn "$TRANSCODE_SSD" "$TRANSCODE_LINK" && \
warn "Symlink redirected: $TRANSCODE_LINK → $TRANSCODE_SSD ✅" || {
error "Failed to redirect symlink"
STOP_SUCCESS=false
}
fi
fi
elif [[ -e "$TRANSCODE_LINK" ]]; then
warn "$TRANSCODE_LINK exists but is not a symlink — leaving as-is"
else
warn "Symlink $TRANSCODE_LINK does not exist — nothing to redirect"
fi
# ==============================================================================================
# ━━━ Active Files Warning ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_RAM Active Files Check ━━━"
FILE_COUNT=$(find "$RAMDISK_PATH" -type f 2>/dev/null | wc -l)
if [[ "$FILE_COUNT" -gt 0 ]]; then
warn "⚠️ $FILE_COUNT file(s) still on ramdisk — will be lost on unmount"
warn "Active transcode sessions should be stopped before unmounting"
warn "Proceeding regardless (this is expected for maintenance)"
if [[ "$LOG" == true ]]; then
find "$RAMDISK_PATH" -type f 2>/dev/null | while read -r f; do
log " $f"
done
fi
else
log "No active files on ramdisk ✅"
fi
# ==============================================================================================
# ━━━ Unmount ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_RAM Unmount Ramdisk ━━━"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would unmount: $RAMDISK_PATH"
else
if umount "$RAMDISK_PATH" 2>/dev/null; then
warn "Ramdisk unmounted: $RAMDISK_PATH ✅"
else
# Regular unmount failed — check if only directory handles are open (no active writes)
OPEN_FILES=$(lsof +D "$RAMDISK_PATH" 2>/dev/null | awk 'NR>1 && $5 != "DIR"' | wc -l)
if [[ "$OPEN_FILES" -eq 0 ]]; then
warn "Busy — only directory handles open, no active writes — trying lazy unmount"
if umount -l "$RAMDISK_PATH"; then
warn "Ramdisk lazy-unmounted: $RAMDISK_PATH ✅"
warn "Handles will release when owning processes next check the directory"
else
error "Lazy unmount also failed — $RAMDISK_PATH"
STOP_SUCCESS=false
fi
else
error "Failed to unmount $RAMDISK_PATH — $OPEN_FILES file(s) still open for writing"
error "Stop active transcode sessions and retry"
STOP_SUCCESS=false
fi
fi
fi
# ==============================================================================================
# ━━━ Update State File ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_GEAR State File ━━━"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would update $STATE_FILE: current_target=$TRANSCODE_SSD"
elif [[ "$STOP_SUCCESS" == true ]]; then
NOW=$(date +%s)
cat > "$STATE_FILE" <<EOF
current_target=$TRANSCODE_SSD
last_flip_time=$NOW
flip_count_hour=0
flip_hour_start=$NOW
EOF
log "State file updated: current_target=$TRANSCODE_SSD"
else
warn "Skipping state file update — stop had errors"
fi
END=$(date +%s)
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY RAMDISK STOP SUMMARY ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_RAM Ramdisk: $RAMDISK_PATH"
echo "$ICON_DISK Fallback: $TRANSCODE_SSD"
echo "$ICON_LINK Symlink: $TRANSCODE_LINK → $(readlink "$TRANSCODE_LINK" 2>/dev/null || echo "not set")"
echo "$ICON_TIME Duration: $(format_duration $(( END - START )))"
echo ""
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no changes made"
elif [[ "$STOP_SUCCESS" == true ]]; then
echo "$ICON_DONE Status: done ✅"
echo "Run ramdisk_setup.sh to remount with new configuration"
else
echo "$ICON_ERROR Status: STOP HAD ERRORS"
notify "Ramdisk stop errors on $(hostname) ($MY_ID) — check output" \
"Ramdisk Stop" "warning"
exit 1
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
@@ -1,508 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ================================= Fallback Test ==============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Controlled simulation of the fallback lifecycle. Validates the entire sequence
# without waiting for a real outage. Contains no fallback logic — exercises the
# real fallback.sh via an iptables DROP rule on the remote Tailscale IP.
#
# Run during a maintenance window. Users will experience a brief service
# interruption. Use --dry-run to walk through all phases without real changes.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# Phase 1 — Pre-flight Both servers reachable, Docker daemons healthy,
# version parity, fallback.sh exists, state NORMAL
# Phase 2 — Block Remote iptables DROP rule added — remote appears unreachable
# Phase 3 — Fallback Detection Wait FALLBACK_TEST_BLOCK_WAIT for fallback.sh to
# detect the outage and enter FALLBACK state
# Phase 4 — Container Start Verify Tier 1 containers started locally
# Phase 5 — Restore iptables rule removed — remote reachable again
# Phase 6 — Handback Wait FALLBACK_TEST_HANDBACK_WAIT for fallback.sh to
# complete full handback and return to NORMAL
# Phase 7 — Container Handback Verify Tier 1 containers stopped locally
# Report — Full pass/fail per phase with timing
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Test Harness Only
# Contains zero fallback logic. All fallback is exercised through fallback.sh.
# Any change to fallback.sh is automatically reflected in the test result.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# iptables Safety Trap
# The DROP rule is removed via trap on ANY exit — normal completion, crash, error,
# ctrl-c. Remote connectivity is always restored regardless of test outcome.
# You cannot accidentally leave the remote permanently blocked.
#
# FALLBACK_ENABLED Gate
# Aborts if FALLBACK_ENABLED=false. Testing a disabled fallback system is
# misleading and potentially destructive.
#
# State Must Be NORMAL
# Pre-flight fails if state is not NORMAL. Running a test during an actual
# fallback event would interfere with the real event.
#
# Version Parity Check
# Pre-flight verifies unRAID version parity before any iptables rules are
# added. A mismatch makes the test result unreliable.
#
# Remote Docker Daemon Check
# Pre-flight confirms remote Docker daemon is responsive before Phase 2.
#
# Lock Acquisition
# acquire_lock() prevents concurrent test runs. Running two tests simultaneously
# would produce conflicting iptables rules and unreliable results.
#
# Host Detection
# detect_hosts() resolves MY_ID / REMOTE_ID from master.conf at startup.
# Exits if the host cannot be identified — prevents testing on an unknown machine.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# master.conf
#
# FALLBACK_TEST_BLOCK_WAIT
# Seconds to wait in Phase 3 for fallback.sh to detect the outage.
# Must be > FALLBACK_CHECK_INTERVAL + buffer. At 30s interval: use ≥60s.
# (default: 60)
#
# FALLBACK_TEST_HANDBACK_WAIT
# Seconds to wait in Phase 6 for fallback.sh to complete handback.
# Must cover: FALLBACK_HANDBACK_STRIKES × FALLBACK_CHECK_INTERVAL + rsync
# duration + container start time. At 3 strikes × 30s + ~2min rsync +
# ~1min container start: use ≥240s. (default: 300)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# fallback_test.sh --dry-run
# Walk through all 7 phases with output but no iptables changes and no
# container starts/stops. ALWAYS run this before a live test.
#
# fallback_test.sh
# Full live test — real iptables DROP rule, real container lifecycle.
# Users will experience a brief service interruption. Run during a
# maintenance window.
#
# fallback_test.sh --status
# Show current fallback state and test timing configuration. No test run.
#
# fallback_test.sh --log
# Verbose output on every check in every phase.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
FALLBACK_SCRIPT="$SCRIPT_DIR/fallback.sh"
DOCKER_TIMEOUT=15
# ==============================================================================================
# ── SAFETY TRAP — always remove iptables rule on exit ─────────────────────────────────────────
# ==============================================================================================
# Fires on normal exit, error exit, ctrl-c, and script crashes.
# Remote connectivity is ALWAYS restored regardless of test outcome.
IPTABLES_RULE_ACTIVE=false
cleanup() {
if [[ "$IPTABLES_RULE_ACTIVE" == true ]]; then
echo ""
warn "$ICON_SHIELD Cleanup — removing iptables block on $REMOTE_SERVER..."
if [[ "$DRY_RUN" == false ]]; then
iptables -D OUTPUT -d "$REMOTE_SERVER" -j DROP 2>/dev/null
IPTABLES_RULE_ACTIVE=false
warn "iptables rule removed — remote connectivity restored"
else
warn "DRY RUN — would remove iptables rule"
fi
fi
}
trap cleanup EXIT
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
# FALLBACK_ENABLED gate — no point testing if fallback is disabled
if [[ "${FALLBACK_ENABLED:-false}" == false ]]; then
warn "FALLBACK_ENABLED=false — fallback test aborted"
warn "Enable fallback in master.conf before running this test"
exit 0
fi
acquire_lock # strict single instance — modifies iptables and containers
if ! command -v docker &>/dev/null; then
error "Docker command not found"
exit 1
fi
detect_hosts
resolve_remote_ip
# Validate commands used by this script
platform_require_cmd \
"$(which iptables 2>/dev/null || echo /sbin/iptables)" \
"--version" "iptables" \
"iptables" || { error "iptables not found — required for connectivity simulation"; exit 1; }
if [[ ! -f "$FALLBACK_SCRIPT" ]]; then
error "fallback.sh not found at $FALLBACK_SCRIPT"
exit 1
fi
log "fallback.sh found at $FALLBACK_SCRIPT"
log "$ICON_GEAR Config: remote=${REMOTE_SERVER_NAME} (${REMOTE_SERVER}) fallback-script=${FALLBACK_SCRIPT}"
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no iptables rules or container changes will be made"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
local_ver=$(grep -oP '(?<=version=")[^"]+' /etc/unraid-version 2>/dev/null || echo "unknown")
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST My ID: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_HOST Remote ID: $REMOTE_ID ($REMOTE_SERVER_NAME — $REMOTE_SERVER)"
echo "$ICON_GEAR unRAID ver: $local_ver"
echo "$ICON_FALLBACK Block wait: ${FALLBACK_TEST_BLOCK_WAIT}s"
echo "$ICON_FALLBACK Handback wait: ${FALLBACK_TEST_HANDBACK_WAIT}s"
echo "$ICON_FALLBACK Check interval: ${FALLBACK_CHECK_INTERVAL}s"
echo "$ICON_FALLBACK Handback strikes: ${FALLBACK_HANDBACK_STRIKES}"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
if [[ -f "$FALLBACK_STATE_FILE" ]]; then
CURRENT_STATE=$(grep "^state=" "$FALLBACK_STATE_FILE" 2>/dev/null | cut -d= -f2)
echo "$ICON_FALLBACK Current state: ${CURRENT_STATE:-unknown}"
else
echo "$ICON_FALLBACK Current state: no state file"
fi
# Show Tier 1 containers for this host
TIER1_VAR="FALLBACK_${MY_ID}_COVERS_${REMOTE_ID}_TIER1"
eval "TIER1_CONTAINERS=(\"\${${TIER1_VAR}[@]:-}\")"
echo "$ICON_CONTAINERS Tier 1 to test: ${TIER1_CONTAINERS[*]:-none configured}"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ── PHASE TRACKING ────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
PHASES_PASS=()
PHASES_FAIL=()
TOTAL_START=$(date +%s)
phase_pass() { PHASES_PASS+=("$1"); warn "$ICON_DONE Phase: $1 — PASSED ✅"; }
phase_fail() { PHASES_FAIL+=("$1"); error "Phase: $1 — FAILED ❌"; }
# Get Tier 1 containers for this server's fallback responsibility
TIER1_VAR="FALLBACK_${MY_ID}_COVERS_${REMOTE_ID}_TIER1"
eval "TIER1_CONTAINERS=(\"\${${TIER1_VAR}[@]:-}\")"
# ==============================================================================================
# ━━━ Phase 1 — Pre-flight ━━━
# ==============================================================================================
echo ""
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo " $ICON_SHIELD FALLBACK TEST — $(date '+%Y-%m-%d %H:%M:%S')"
echo " $ICON_HOST $MY_ID ($LOCAL_SERVER_NAME) → $REMOTE_ID ($REMOTE_SERVER_NAME)"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo ""
echo "━━━ $ICON_SHIELD Phase 1 — Pre-flight ━━━"
# Remote reachable
if ping_remote; then
log "$REMOTE_SERVER_NAME is reachable"
else
error "$REMOTE_SERVER_NAME is not reachable — cannot run test"
phase_fail "Pre-flight"
exit 1
fi
# Internet reachable
if ping_internet; then
log "Internet is reachable"
else
error "No internet connectivity — cannot run test"
phase_fail "Pre-flight"
exit 1
fi
# Version parity — test may produce misleading results on mismatch
if ! check_unraid_version_parity; then
error "unRAID version mismatch — test aborted to prevent misleading results"
phase_fail "Pre-flight"
exit 1
fi
# Remote Docker daemon — must be responsive before test manipulates containers
if ! check_remote_docker_daemon; then
error "Remote Docker daemon not responsive — cannot run test"
phase_fail "Pre-flight"
exit 1
fi
# Fallback state must be NORMAL before test
if [[ -f "$FALLBACK_STATE_FILE" ]]; then
CURRENT_STATE=$(grep "^state=" "$FALLBACK_STATE_FILE" 2>/dev/null | cut -d= -f2)
if [[ "$CURRENT_STATE" != "NORMAL" ]]; then
error "Fallback state is $CURRENT_STATE — must be NORMAL before running test"
phase_fail "Pre-flight"
exit 1
fi
log "Fallback state is NORMAL"
else
warn "No state file found — assuming NORMAL (first run)"
fi
# Tier 1 containers configured
if [[ ${#TIER1_CONTAINERS[@]} -eq 0 ]]; then
error "No Tier 1 containers configured for $MY_ID → $REMOTE_ID"
error "Check FALLBACK_${MY_ID}_COVERS_${REMOTE_ID}_TIER1 in host*.conf"
phase_fail "Pre-flight"
exit 1
fi
log "Tier 1 containers: ${TIER1_CONTAINERS[*]}"
phase_pass "Pre-flight"
# ==============================================================================================
# ━━━ Phase 2 — Block Remote Connectivity ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_PING Phase 2 — Block Remote Connectivity ━━━"
warn "Adding iptables rule — dropping all traffic to $REMOTE_SERVER ($REMOTE_SERVER_NAME)"
if [[ "$DRY_RUN" == false ]]; then
iptables -I OUTPUT -d "$REMOTE_SERVER" -j DROP
IPTABLES_RULE_ACTIVE=true
# Verify block is working
sleep 2
if ! ping -c1 -W2 "$REMOTE_SERVER" &>/dev/null; then
log "Connectivity block confirmed — ping to remote fails as expected"
phase_pass "Block Remote"
else
error "iptables rule did not block connectivity — ping still succeeds"
phase_fail "Block Remote"
exit 1
fi
else
warn "DRY RUN — would block $REMOTE_SERVER with iptables DROP rule"
phase_pass "Block Remote"
fi
# ==============================================================================================
# ━━━ Phase 3 — Fallback Detection ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_FALLBACK Phase 3 — Fallback Detection ━━━"
warn "Waiting ${FALLBACK_TEST_BLOCK_WAIT}s for fallback.sh to detect outage..."
log "fallback.sh check interval: ${FALLBACK_CHECK_INTERVAL}s"
if [[ "$DRY_RUN" == false ]]; then
sleep "$FALLBACK_TEST_BLOCK_WAIT"
if [[ -f "$FALLBACK_STATE_FILE" ]]; then
NEW_STATE=$(grep "^state=" "$FALLBACK_STATE_FILE" 2>/dev/null | cut -d= -f2)
if [[ "$NEW_STATE" == "FALLBACK" ]]; then
log "State changed to FALLBACK — outage detected correctly ✅"
phase_pass "Fallback Detection"
else
error "State is $NEW_STATE — expected FALLBACK after ${FALLBACK_TEST_BLOCK_WAIT}s"
warn "Is fallback.sh running? Check User Scripts plugin"
phase_fail "Fallback Detection"
fi
else
error "No state file found after wait — fallback.sh may not be running"
phase_fail "Fallback Detection"
fi
else
warn "DRY RUN — would wait ${FALLBACK_TEST_BLOCK_WAIT}s then check for FALLBACK state"
phase_pass "Fallback Detection"
fi
# ==============================================================================================
# ━━━ Phase 4 — Container Start Verification ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CONTAINERS Phase 4 — Tier 1 Containers Started Locally ━━━"
log "Checking Tier 1 containers: ${TIER1_CONTAINERS[*]}"
if [[ "$DRY_RUN" == false ]]; then
CONTAINERS_OK=true
for container in "${TIER1_CONTAINERS[@]}"; do
[[ -z "$container" ]] && continue
STATUS=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Running}}' \
"$container" 2>/dev/null)
if [[ "$STATUS" == "true" ]]; then
log "$ICON_RUNNING $container is running locally ✅"
else
error "$ICON_NOT_RUNNING $container is NOT running locally"
CONTAINERS_OK=false
fi
done
if [[ "$CONTAINERS_OK" == true ]]; then
phase_pass "Container Start"
else
phase_fail "Container Start"
fi
else
warn "DRY RUN — would verify these Tier 1 containers started: ${TIER1_CONTAINERS[*]}"
phase_pass "Container Start"
fi
# ==============================================================================================
# ━━━ Phase 5 — Restore Remote Connectivity ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_PING Phase 5 — Restore Remote Connectivity ━━━"
warn "Removing iptables block — $REMOTE_SERVER_NAME becomes reachable again"
if [[ "$DRY_RUN" == false ]]; then
iptables -D OUTPUT -d "$REMOTE_SERVER" -j DROP 2>/dev/null
IPTABLES_RULE_ACTIVE=false
sleep 3
if ping_remote; then
log "$REMOTE_SERVER_NAME is reachable again ✅"
phase_pass "Restore Connectivity"
else
error "$REMOTE_SERVER_NAME still unreachable after removing iptables rule"
phase_fail "Restore Connectivity"
fi
else
warn "DRY RUN — would remove iptables rule"
phase_pass "Restore Connectivity"
fi
# ==============================================================================================
# ━━━ Phase 6 — Handback ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_FALLBACK Phase 6 — Handback ━━━"
warn "Waiting ${FALLBACK_TEST_HANDBACK_WAIT}s for fallback.sh to complete handback..."
log "Requires $FALLBACK_HANDBACK_STRIKES consecutive checks at ${FALLBACK_CHECK_INTERVAL}s"
log "Minimum handback time: $(( FALLBACK_HANDBACK_STRIKES * FALLBACK_CHECK_INTERVAL ))s"
if [[ "$DRY_RUN" == false ]]; then
sleep "$FALLBACK_TEST_HANDBACK_WAIT"
if [[ -f "$FALLBACK_STATE_FILE" ]]; then
FINAL_STATE=$(grep "^state=" "$FALLBACK_STATE_FILE" 2>/dev/null | cut -d= -f2)
if [[ "$FINAL_STATE" == "NORMAL" ]]; then
log "State returned to NORMAL — handback completed ✅"
phase_pass "Handback"
else
error "State is $FINAL_STATE — expected NORMAL after ${FALLBACK_TEST_HANDBACK_WAIT}s"
warn "Handback may still be in progress — check fallback.sh output"
phase_fail "Handback"
fi
else
error "No state file found"
phase_fail "Handback"
fi
else
warn "DRY RUN — would wait ${FALLBACK_TEST_HANDBACK_WAIT}s then verify NORMAL state"
phase_pass "Handback"
fi
# ==============================================================================================
# ━━━ Phase 7 — Container Handback Verification ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CONTAINERS Phase 7 — Tier 1 Containers Stopped Locally ━━━"
log "Verifying Tier 1 containers returned to $REMOTE_SERVER_NAME"
if [[ "$DRY_RUN" == false ]]; then
HANDBACK_OK=true
for container in "${TIER1_CONTAINERS[@]}"; do
[[ -z "$container" ]] && continue
STATUS=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Running}}' \
"$container" 2>/dev/null)
if [[ "$STATUS" != "true" ]]; then
log "$ICON_NOT_RUNNING $container stopped locally — handed back ✅"
else
error "$ICON_RUNNING $container still running locally — handback may have failed"
HANDBACK_OK=false
fi
done
if [[ "$HANDBACK_OK" == true ]]; then
phase_pass "Container Handback"
else
phase_fail "Container Handback"
fi
else
warn "DRY RUN — would verify Tier 1 containers stopped locally after handback"
phase_pass "Container Handback"
fi
TOTAL_END=$(date +%s)
# ==============================================================================================
# ━━━ Test Report ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY FALLBACK TEST REPORT ━━━━━"
echo "$ICON_HOST My ID: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_HOST Remote: $REMOTE_ID ($REMOTE_SERVER_NAME)"
echo "$ICON_TIME Duration: $(format_duration $((TOTAL_END - TOTAL_START)))"
echo ""
echo " Phase Results:"
for phase in "${PHASES_PASS[@]}"; do
echo " $ICON_SUCCESS $phase"
done
for phase in "${PHASES_FAIL[@]}"; do
echo " $ICON_ERROR $phase"
done
echo ""
PASS_COUNT=${#PHASES_PASS[@]}
FAIL_COUNT=${#PHASES_FAIL[@]}
TOTAL_PHASES=$(( PASS_COUNT + FAIL_COUNT ))
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no changes made"
elif [[ "$FAIL_COUNT" -eq 0 ]]; then
warn "$ICON_DONE ALL $TOTAL_PHASES PHASES PASSED"
notify "Fallback test PASSED on $(hostname) — all $TOTAL_PHASES phases completed" \
"Fallback Test" "normal"
else
error "$FAIL_COUNT/$TOTAL_PHASES PHASES FAILED"
notify "Fallback test FAILED on $(hostname) — $FAIL_COUNT/$TOTAL_PHASES phases failed: ${PHASES_FAIL[*]}" \
"Fallback Test" "warning"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
[[ "$FAIL_COUNT" -gt 0 ]] && exit 1
exit 0
@@ -1,508 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ================================= Fallback Test ==============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Controlled simulation of the fallback lifecycle. Validates the entire sequence
# without waiting for a real outage. Contains no fallback logic — exercises the
# real fallback.sh via an iptables DROP rule on the remote Tailscale IP.
#
# Run during a maintenance window. Users will experience a brief service
# interruption. Use --dry-run to walk through all phases without real changes.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# Phase 1 — Pre-flight Both servers reachable, Docker daemons healthy,
# version parity, fallback.sh exists, state NORMAL
# Phase 2 — Block Remote iptables DROP rule added — remote appears unreachable
# Phase 3 — Fallback Detection Wait FALLBACK_TEST_BLOCK_WAIT for fallback.sh to
# detect the outage and enter FALLBACK state
# Phase 4 — Container Start Verify Tier 1 containers started locally
# Phase 5 — Restore iptables rule removed — remote reachable again
# Phase 6 — Handback Wait FALLBACK_TEST_HANDBACK_WAIT for fallback.sh to
# complete full handback and return to NORMAL
# Phase 7 — Container Handback Verify Tier 1 containers stopped locally
# Report — Full pass/fail per phase with timing
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Test Harness Only
# Contains zero fallback logic. All fallback is exercised through fallback.sh.
# Any change to fallback.sh is automatically reflected in the test result.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# iptables Safety Trap
# The DROP rule is removed via trap on ANY exit — normal completion, crash, error,
# ctrl-c. Remote connectivity is always restored regardless of test outcome.
# You cannot accidentally leave the remote permanently blocked.
#
# FALLBACK_ENABLED Gate
# Aborts if FALLBACK_ENABLED=false. Testing a disabled fallback system is
# misleading and potentially destructive.
#
# State Must Be NORMAL
# Pre-flight fails if state is not NORMAL. Running a test during an actual
# fallback event would interfere with the real event.
#
# Version Parity Check
# Pre-flight verifies unRAID version parity before any iptables rules are
# added. A mismatch makes the test result unreliable.
#
# Remote Docker Daemon Check
# Pre-flight confirms remote Docker daemon is responsive before Phase 2.
#
# Lock Acquisition
# acquire_lock() prevents concurrent test runs. Running two tests simultaneously
# would produce conflicting iptables rules and unreliable results.
#
# Host Detection
# detect_hosts() resolves MY_ID / REMOTE_ID from master.conf at startup.
# Exits if the host cannot be identified — prevents testing on an unknown machine.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# master.conf
#
# FALLBACK_TEST_BLOCK_WAIT
# Seconds to wait in Phase 3 for fallback.sh to detect the outage.
# Must be > FALLBACK_CHECK_INTERVAL + buffer. At 30s interval: use ≥60s.
# (default: 60)
#
# FALLBACK_TEST_HANDBACK_WAIT
# Seconds to wait in Phase 6 for fallback.sh to complete handback.
# Must cover: FALLBACK_HANDBACK_STRIKES × FALLBACK_CHECK_INTERVAL + rsync
# duration + container start time. At 3 strikes × 30s + ~2min rsync +
# ~1min container start: use ≥240s. (default: 300)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# fallback_test.sh --dry-run
# Walk through all 7 phases with output but no iptables changes and no
# container starts/stops. ALWAYS run this before a live test.
#
# fallback_test.sh
# Full live test — real iptables DROP rule, real container lifecycle.
# Users will experience a brief service interruption. Run during a
# maintenance window.
#
# fallback_test.sh --status
# Show current fallback state and test timing configuration. No test run.
#
# fallback_test.sh --log
# Verbose output on every check in every phase.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
FALLBACK_SCRIPT="$SCRIPT_DIR/fallback.sh"
DOCKER_TIMEOUT=15
# ==============================================================================================
# ── SAFETY TRAP — always remove iptables rule on exit ─────────────────────────────────────────
# ==============================================================================================
# Fires on normal exit, error exit, ctrl-c, and script crashes.
# Remote connectivity is ALWAYS restored regardless of test outcome.
IPTABLES_RULE_ACTIVE=false
cleanup() {
if [[ "$IPTABLES_RULE_ACTIVE" == true ]]; then
echo ""
warn "$ICON_SHIELD Cleanup — removing iptables block on $REMOTE_SERVER..."
if [[ "$DRY_RUN" == false ]]; then
iptables -D OUTPUT -d "$REMOTE_SERVER" -j DROP 2>/dev/null
IPTABLES_RULE_ACTIVE=false
warn "iptables rule removed — remote connectivity restored"
else
warn "DRY RUN — would remove iptables rule"
fi
fi
}
trap cleanup EXIT
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
# FALLBACK_ENABLED gate — no point testing if fallback is disabled
if [[ "${FALLBACK_ENABLED:-false}" == false ]]; then
warn "FALLBACK_ENABLED=false — fallback test aborted"
warn "Enable fallback in master.conf before running this test"
exit 0
fi
acquire_lock # strict single instance — modifies iptables and containers
if ! command -v docker &>/dev/null; then
error "Docker command not found"
exit 1
fi
detect_hosts
resolve_remote_ip
# Validate commands used by this script
platform_require_cmd \
"$(which iptables 2>/dev/null || echo /sbin/iptables)" \
"--version" "iptables" \
"iptables" || { error "iptables not found — required for connectivity simulation"; exit 1; }
if [[ ! -f "$FALLBACK_SCRIPT" ]]; then
error "fallback.sh not found at $FALLBACK_SCRIPT"
exit 1
fi
log "fallback.sh found at $FALLBACK_SCRIPT"
log "$ICON_GEAR Config: remote=${REMOTE_SERVER_NAME} (${REMOTE_SERVER}) fallback-script=${FALLBACK_SCRIPT}"
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no iptables rules or container changes will be made"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
local_ver=$(platform_get_os_version 2>/dev/null || echo "unknown")
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST My ID: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_HOST Remote ID: $REMOTE_ID ($REMOTE_SERVER_NAME — $REMOTE_SERVER)"
echo "$ICON_GEAR OS ver: $local_ver"
echo "$ICON_FALLBACK Block wait: ${FALLBACK_TEST_BLOCK_WAIT}s"
echo "$ICON_FALLBACK Handback wait: ${FALLBACK_TEST_HANDBACK_WAIT}s"
echo "$ICON_FALLBACK Check interval: ${FALLBACK_CHECK_INTERVAL}s"
echo "$ICON_FALLBACK Handback strikes: ${FALLBACK_HANDBACK_STRIKES}"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
if [[ -f "$FALLBACK_STATE_FILE" ]]; then
CURRENT_STATE=$(grep "^state=" "$FALLBACK_STATE_FILE" 2>/dev/null | cut -d= -f2)
echo "$ICON_FALLBACK Current state: ${CURRENT_STATE:-unknown}"
else
echo "$ICON_FALLBACK Current state: no state file"
fi
# Show Tier 1 containers for this host
TIER1_VAR="FALLBACK_${MY_ID}_COVERS_${REMOTE_ID}_TIER1"
eval "TIER1_CONTAINERS=(\"\${${TIER1_VAR}[@]:-}\")"
echo "$ICON_CONTAINERS Tier 1 to test: ${TIER1_CONTAINERS[*]:-none configured}"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ── PHASE TRACKING ────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
PHASES_PASS=()
PHASES_FAIL=()
TOTAL_START=$(date +%s)
phase_pass() { PHASES_PASS+=("$1"); warn "$ICON_DONE Phase: $1 — PASSED ✅"; }
phase_fail() { PHASES_FAIL+=("$1"); error "Phase: $1 — FAILED ❌"; }
# Get Tier 1 containers for this server's fallback responsibility
TIER1_VAR="FALLBACK_${MY_ID}_COVERS_${REMOTE_ID}_TIER1"
eval "TIER1_CONTAINERS=(\"\${${TIER1_VAR}[@]:-}\")"
# ==============================================================================================
# ━━━ Phase 1 — Pre-flight ━━━
# ==============================================================================================
echo ""
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo " $ICON_SHIELD FALLBACK TEST — $(date '+%Y-%m-%d %H:%M:%S')"
echo " $ICON_HOST $MY_ID ($LOCAL_SERVER_NAME) → $REMOTE_ID ($REMOTE_SERVER_NAME)"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo ""
echo "━━━ $ICON_SHIELD Phase 1 — Pre-flight ━━━"
# Remote reachable
if ping_remote; then
log "$REMOTE_SERVER_NAME is reachable"
else
error "$REMOTE_SERVER_NAME is not reachable — cannot run test"
phase_fail "Pre-flight"
exit 1
fi
# Internet reachable
if ping_internet; then
log "Internet is reachable"
else
error "No internet connectivity — cannot run test"
phase_fail "Pre-flight"
exit 1
fi
# Version parity — test may produce misleading results on mismatch
if ! check_unraid_version_parity; then
error "unRAID version mismatch — test aborted to prevent misleading results"
phase_fail "Pre-flight"
exit 1
fi
# Remote Docker daemon — must be responsive before test manipulates containers
if ! check_remote_docker_daemon; then
error "Remote Docker daemon not responsive — cannot run test"
phase_fail "Pre-flight"
exit 1
fi
# Fallback state must be NORMAL before test
if [[ -f "$FALLBACK_STATE_FILE" ]]; then
CURRENT_STATE=$(grep "^state=" "$FALLBACK_STATE_FILE" 2>/dev/null | cut -d= -f2)
if [[ "$CURRENT_STATE" != "NORMAL" ]]; then
error "Fallback state is $CURRENT_STATE — must be NORMAL before running test"
phase_fail "Pre-flight"
exit 1
fi
log "Fallback state is NORMAL"
else
warn "No state file found — assuming NORMAL (first run)"
fi
# Tier 1 containers configured
if [[ ${#TIER1_CONTAINERS[@]} -eq 0 ]]; then
error "No Tier 1 containers configured for $MY_ID → $REMOTE_ID"
error "Check FALLBACK_${MY_ID}_COVERS_${REMOTE_ID}_TIER1 in host*.conf"
phase_fail "Pre-flight"
exit 1
fi
log "Tier 1 containers: ${TIER1_CONTAINERS[*]}"
phase_pass "Pre-flight"
# ==============================================================================================
# ━━━ Phase 2 — Block Remote Connectivity ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_PING Phase 2 — Block Remote Connectivity ━━━"
warn "Adding iptables rule — dropping all traffic to $REMOTE_SERVER ($REMOTE_SERVER_NAME)"
if [[ "$DRY_RUN" == false ]]; then
iptables -I OUTPUT -d "$REMOTE_SERVER" -j DROP
IPTABLES_RULE_ACTIVE=true
# Verify block is working
sleep 2
if ! ping -c1 -W2 "$REMOTE_SERVER" &>/dev/null; then
log "Connectivity block confirmed — ping to remote fails as expected"
phase_pass "Block Remote"
else
error "iptables rule did not block connectivity — ping still succeeds"
phase_fail "Block Remote"
exit 1
fi
else
warn "DRY RUN — would block $REMOTE_SERVER with iptables DROP rule"
phase_pass "Block Remote"
fi
# ==============================================================================================
# ━━━ Phase 3 — Fallback Detection ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_FALLBACK Phase 3 — Fallback Detection ━━━"
warn "Waiting ${FALLBACK_TEST_BLOCK_WAIT}s for fallback.sh to detect outage..."
log "fallback.sh check interval: ${FALLBACK_CHECK_INTERVAL}s"
if [[ "$DRY_RUN" == false ]]; then
sleep "$FALLBACK_TEST_BLOCK_WAIT"
if [[ -f "$FALLBACK_STATE_FILE" ]]; then
NEW_STATE=$(grep "^state=" "$FALLBACK_STATE_FILE" 2>/dev/null | cut -d= -f2)
if [[ "$NEW_STATE" == "FALLBACK" ]]; then
log "State changed to FALLBACK — outage detected correctly ✅"
phase_pass "Fallback Detection"
else
error "State is $NEW_STATE — expected FALLBACK after ${FALLBACK_TEST_BLOCK_WAIT}s"
warn "Is fallback.sh running? Check User Scripts plugin"
phase_fail "Fallback Detection"
fi
else
error "No state file found after wait — fallback.sh may not be running"
phase_fail "Fallback Detection"
fi
else
warn "DRY RUN — would wait ${FALLBACK_TEST_BLOCK_WAIT}s then check for FALLBACK state"
phase_pass "Fallback Detection"
fi
# ==============================================================================================
# ━━━ Phase 4 — Container Start Verification ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CONTAINERS Phase 4 — Tier 1 Containers Started Locally ━━━"
log "Checking Tier 1 containers: ${TIER1_CONTAINERS[*]}"
if [[ "$DRY_RUN" == false ]]; then
CONTAINERS_OK=true
for container in "${TIER1_CONTAINERS[@]}"; do
[[ -z "$container" ]] && continue
STATUS=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Running}}' \
"$container" 2>/dev/null)
if [[ "$STATUS" == "true" ]]; then
log "$ICON_RUNNING $container is running locally ✅"
else
error "$ICON_NOT_RUNNING $container is NOT running locally"
CONTAINERS_OK=false
fi
done
if [[ "$CONTAINERS_OK" == true ]]; then
phase_pass "Container Start"
else
phase_fail "Container Start"
fi
else
warn "DRY RUN — would verify these Tier 1 containers started: ${TIER1_CONTAINERS[*]}"
phase_pass "Container Start"
fi
# ==============================================================================================
# ━━━ Phase 5 — Restore Remote Connectivity ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_PING Phase 5 — Restore Remote Connectivity ━━━"
warn "Removing iptables block — $REMOTE_SERVER_NAME becomes reachable again"
if [[ "$DRY_RUN" == false ]]; then
iptables -D OUTPUT -d "$REMOTE_SERVER" -j DROP 2>/dev/null
IPTABLES_RULE_ACTIVE=false
sleep 3
if ping_remote; then
log "$REMOTE_SERVER_NAME is reachable again ✅"
phase_pass "Restore Connectivity"
else
error "$REMOTE_SERVER_NAME still unreachable after removing iptables rule"
phase_fail "Restore Connectivity"
fi
else
warn "DRY RUN — would remove iptables rule"
phase_pass "Restore Connectivity"
fi
# ==============================================================================================
# ━━━ Phase 6 — Handback ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_FALLBACK Phase 6 — Handback ━━━"
warn "Waiting ${FALLBACK_TEST_HANDBACK_WAIT}s for fallback.sh to complete handback..."
log "Requires $FALLBACK_HANDBACK_STRIKES consecutive checks at ${FALLBACK_CHECK_INTERVAL}s"
log "Minimum handback time: $(( FALLBACK_HANDBACK_STRIKES * FALLBACK_CHECK_INTERVAL ))s"
if [[ "$DRY_RUN" == false ]]; then
sleep "$FALLBACK_TEST_HANDBACK_WAIT"
if [[ -f "$FALLBACK_STATE_FILE" ]]; then
FINAL_STATE=$(grep "^state=" "$FALLBACK_STATE_FILE" 2>/dev/null | cut -d= -f2)
if [[ "$FINAL_STATE" == "NORMAL" ]]; then
log "State returned to NORMAL — handback completed ✅"
phase_pass "Handback"
else
error "State is $FINAL_STATE — expected NORMAL after ${FALLBACK_TEST_HANDBACK_WAIT}s"
warn "Handback may still be in progress — check fallback.sh output"
phase_fail "Handback"
fi
else
error "No state file found"
phase_fail "Handback"
fi
else
warn "DRY RUN — would wait ${FALLBACK_TEST_HANDBACK_WAIT}s then verify NORMAL state"
phase_pass "Handback"
fi
# ==============================================================================================
# ━━━ Phase 7 — Container Handback Verification ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CONTAINERS Phase 7 — Tier 1 Containers Stopped Locally ━━━"
log "Verifying Tier 1 containers returned to $REMOTE_SERVER_NAME"
if [[ "$DRY_RUN" == false ]]; then
HANDBACK_OK=true
for container in "${TIER1_CONTAINERS[@]}"; do
[[ -z "$container" ]] && continue
STATUS=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Running}}' \
"$container" 2>/dev/null)
if [[ "$STATUS" != "true" ]]; then
log "$ICON_NOT_RUNNING $container stopped locally — handed back ✅"
else
error "$ICON_RUNNING $container still running locally — handback may have failed"
HANDBACK_OK=false
fi
done
if [[ "$HANDBACK_OK" == true ]]; then
phase_pass "Container Handback"
else
phase_fail "Container Handback"
fi
else
warn "DRY RUN — would verify Tier 1 containers stopped locally after handback"
phase_pass "Container Handback"
fi
TOTAL_END=$(date +%s)
# ==============================================================================================
# ━━━ Test Report ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY FALLBACK TEST REPORT ━━━━━"
echo "$ICON_HOST My ID: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_HOST Remote: $REMOTE_ID ($REMOTE_SERVER_NAME)"
echo "$ICON_TIME Duration: $(format_duration $((TOTAL_END - TOTAL_START)))"
echo ""
echo " Phase Results:"
for phase in "${PHASES_PASS[@]}"; do
echo " $ICON_SUCCESS $phase"
done
for phase in "${PHASES_FAIL[@]}"; do
echo " $ICON_ERROR $phase"
done
echo ""
PASS_COUNT=${#PHASES_PASS[@]}
FAIL_COUNT=${#PHASES_FAIL[@]}
TOTAL_PHASES=$(( PASS_COUNT + FAIL_COUNT ))
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no changes made"
elif [[ "$FAIL_COUNT" -eq 0 ]]; then
warn "$ICON_DONE ALL $TOTAL_PHASES PHASES PASSED"
notify "Fallback test PASSED on $(hostname) — all $TOTAL_PHASES phases completed" \
"Fallback Test" "normal"
else
error "$FAIL_COUNT/$TOTAL_PHASES PHASES FAILED"
notify "Fallback test FAILED on $(hostname) — $FAIL_COUNT/$TOTAL_PHASES phases failed: ${PHASES_FAIL[*]}" \
"Fallback Test" "warning"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
[[ "$FAIL_COUNT" -gt 0 ]] && exit 1
exit 0
@@ -1,508 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ================================= Fallback Test ==============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Controlled simulation of the fallback lifecycle. Validates the entire sequence
# without waiting for a real outage. Contains no fallback logic — exercises the
# real fallback.sh via an iptables DROP rule on the remote Tailscale IP.
#
# Run during a maintenance window. Users will experience a brief service
# interruption. Use --dry-run to walk through all phases without real changes.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# Phase 1 — Pre-flight Both servers reachable, Docker daemons healthy,
# version parity, fallback.sh exists, state NORMAL
# Phase 2 — Block Remote iptables DROP rule added — remote appears unreachable
# Phase 3 — Fallback Detection Wait FALLBACK_TEST_BLOCK_WAIT for fallback.sh to
# detect the outage and enter FALLBACK state
# Phase 4 — Container Start Verify Tier 1 containers started locally
# Phase 5 — Restore iptables rule removed — remote reachable again
# Phase 6 — Handback Wait FALLBACK_TEST_HANDBACK_WAIT for fallback.sh to
# complete full handback and return to NORMAL
# Phase 7 — Container Handback Verify Tier 1 containers stopped locally
# Report — Full pass/fail per phase with timing
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Test Harness Only
# Contains zero fallback logic. All fallback is exercised through fallback.sh.
# Any change to fallback.sh is automatically reflected in the test result.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# iptables Safety Trap
# The DROP rule is removed via trap on ANY exit — normal completion, crash, error,
# ctrl-c. Remote connectivity is always restored regardless of test outcome.
# You cannot accidentally leave the remote permanently blocked.
#
# FALLBACK_ENABLED Gate
# Aborts if FALLBACK_ENABLED=false. Testing a disabled fallback system is
# misleading and potentially destructive.
#
# State Must Be NORMAL
# Pre-flight fails if state is not NORMAL. Running a test during an actual
# fallback event would interfere with the real event.
#
# Version Parity Check
# Pre-flight verifies unRAID version parity before any iptables rules are
# added. A mismatch makes the test result unreliable.
#
# Remote Docker Daemon Check
# Pre-flight confirms remote Docker daemon is responsive before Phase 2.
#
# Lock Acquisition
# acquire_lock() prevents concurrent test runs. Running two tests simultaneously
# would produce conflicting iptables rules and unreliable results.
#
# Host Detection
# detect_hosts() resolves MY_ID / REMOTE_ID from master.conf at startup.
# Exits if the host cannot be identified — prevents testing on an unknown machine.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# master.conf
#
# FALLBACK_TEST_BLOCK_WAIT
# Seconds to wait in Phase 3 for fallback.sh to detect the outage.
# Must be > FALLBACK_CHECK_INTERVAL + buffer. At 30s interval: use ≥60s.
# (default: 60)
#
# FALLBACK_TEST_HANDBACK_WAIT
# Seconds to wait in Phase 6 for fallback.sh to complete handback.
# Must cover: FALLBACK_HANDBACK_STRIKES × FALLBACK_CHECK_INTERVAL + rsync
# duration + container start time. At 3 strikes × 30s + ~2min rsync +
# ~1min container start: use ≥240s. (default: 300)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# fallback_test.sh --dry-run
# Walk through all 7 phases with output but no iptables changes and no
# container starts/stops. ALWAYS run this before a live test.
#
# fallback_test.sh
# Full live test — real iptables DROP rule, real container lifecycle.
# Users will experience a brief service interruption. Run during a
# maintenance window.
#
# fallback_test.sh --status
# Show current fallback state and test timing configuration. No test run.
#
# fallback_test.sh --log
# Verbose output on every check in every phase.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
FALLBACK_SCRIPT="$SCRIPT_DIR/fallback.sh"
DOCKER_TIMEOUT=15
# ==============================================================================================
# ── SAFETY TRAP — always remove iptables rule on exit ─────────────────────────────────────────
# ==============================================================================================
# Fires on normal exit, error exit, ctrl-c, and script crashes.
# Remote connectivity is ALWAYS restored regardless of test outcome.
IPTABLES_RULE_ACTIVE=false
cleanup() {
if [[ "$IPTABLES_RULE_ACTIVE" == true ]]; then
echo ""
warn "$ICON_SHIELD Cleanup — removing iptables block on $REMOTE_SERVER..."
if [[ "$DRY_RUN" == false ]]; then
iptables -D OUTPUT -d "$REMOTE_SERVER" -j DROP 2>/dev/null
IPTABLES_RULE_ACTIVE=false
warn "iptables rule removed — remote connectivity restored"
else
warn "DRY RUN — would remove iptables rule"
fi
fi
}
trap cleanup EXIT
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
# FALLBACK_ENABLED gate — no point testing if fallback is disabled
if [[ "${FALLBACK_ENABLED:-false}" == false ]]; then
warn "FALLBACK_ENABLED=false — fallback test aborted"
warn "Enable fallback in master.conf before running this test"
exit 0
fi
acquire_lock # strict single instance — modifies iptables and containers
if ! command -v docker &>/dev/null; then
error "Docker command not found"
exit 1
fi
detect_hosts
resolve_remote_ip
# Validate commands used by this script
platform_require_cmd \
"$(which iptables 2>/dev/null || echo /sbin/iptables)" \
"--version" "iptables" \
"iptables" || { error "iptables not found — required for connectivity simulation"; exit 1; }
if [[ ! -f "$FALLBACK_SCRIPT" ]]; then
error "fallback.sh not found at $FALLBACK_SCRIPT"
exit 1
fi
log "fallback.sh found at $FALLBACK_SCRIPT"
log "$ICON_GEAR Config: remote=${REMOTE_SERVER_NAME} (${REMOTE_SERVER}) fallback-script=${FALLBACK_SCRIPT}"
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no iptables rules or container changes will be made"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
local_ver=$(platform_get_os_version 2>/dev/null || echo "unknown")
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST My ID: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_HOST Remote ID: $REMOTE_ID ($REMOTE_SERVER_NAME — $REMOTE_SERVER)"
echo "$ICON_GEAR OS ver: $local_ver"
echo "$ICON_FALLBACK Block wait: ${FALLBACK_TEST_BLOCK_WAIT}s"
echo "$ICON_FALLBACK Handback wait: ${FALLBACK_TEST_HANDBACK_WAIT}s"
echo "$ICON_FALLBACK Check interval: ${FALLBACK_CHECK_INTERVAL}s"
echo "$ICON_FALLBACK Handback strikes: ${FALLBACK_HANDBACK_STRIKES}"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
if [[ -f "$FALLBACK_STATE_FILE" ]]; then
CURRENT_STATE=$(grep "^state=" "$FALLBACK_STATE_FILE" 2>/dev/null | cut -d= -f2)
echo "$ICON_FALLBACK Current state: ${CURRENT_STATE:-unknown}"
else
echo "$ICON_FALLBACK Current state: no state file"
fi
# Show Tier 1 containers for this host
TIER1_VAR="FALLBACK_${MY_ID}_COVERS_${REMOTE_ID}_TIER1"
eval "TIER1_CONTAINERS=(\"\${${TIER1_VAR}[@]:-}\")"
echo "$ICON_CONTAINERS Tier 1 to test: ${TIER1_CONTAINERS[*]:-none configured}"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ── PHASE TRACKING ────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
PHASES_PASS=()
PHASES_FAIL=()
TOTAL_START=$(date +%s)
phase_pass() { PHASES_PASS+=("$1"); warn "$ICON_DONE Phase: $1 — PASSED ✅"; }
phase_fail() { PHASES_FAIL+=("$1"); error "Phase: $1 — FAILED ❌"; }
# Get Tier 1 containers for this server's fallback responsibility
TIER1_VAR="FALLBACK_${MY_ID}_COVERS_${REMOTE_ID}_TIER1"
eval "TIER1_CONTAINERS=(\"\${${TIER1_VAR}[@]:-}\")"
# ==============================================================================================
# ━━━ Phase 1 — Pre-flight ━━━
# ==============================================================================================
echo ""
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo " $ICON_SHIELD FALLBACK TEST — $(date '+%Y-%m-%d %H:%M:%S')"
echo " $ICON_HOST $MY_ID ($LOCAL_SERVER_NAME) → $REMOTE_ID ($REMOTE_SERVER_NAME)"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo ""
echo "━━━ $ICON_SHIELD Phase 1 — Pre-flight ━━━"
# Remote reachable
if ping_remote; then
log "$REMOTE_SERVER_NAME is reachable"
else
error "$REMOTE_SERVER_NAME is not reachable — cannot run test"
phase_fail "Pre-flight"
exit 1
fi
# Internet reachable
if ping_internet; then
log "Internet is reachable"
else
error "No internet connectivity — cannot run test"
phase_fail "Pre-flight"
exit 1
fi
# Version parity — test may produce misleading results on mismatch
if ! check_os_version_parity; then
error "unRAID version mismatch — test aborted to prevent misleading results"
phase_fail "Pre-flight"
exit 1
fi
# Remote Docker daemon — must be responsive before test manipulates containers
if ! check_remote_docker_daemon; then
error "Remote Docker daemon not responsive — cannot run test"
phase_fail "Pre-flight"
exit 1
fi
# Fallback state must be NORMAL before test
if [[ -f "$FALLBACK_STATE_FILE" ]]; then
CURRENT_STATE=$(grep "^state=" "$FALLBACK_STATE_FILE" 2>/dev/null | cut -d= -f2)
if [[ "$CURRENT_STATE" != "NORMAL" ]]; then
error "Fallback state is $CURRENT_STATE — must be NORMAL before running test"
phase_fail "Pre-flight"
exit 1
fi
log "Fallback state is NORMAL"
else
warn "No state file found — assuming NORMAL (first run)"
fi
# Tier 1 containers configured
if [[ ${#TIER1_CONTAINERS[@]} -eq 0 ]]; then
error "No Tier 1 containers configured for $MY_ID → $REMOTE_ID"
error "Check FALLBACK_${MY_ID}_COVERS_${REMOTE_ID}_TIER1 in host*.conf"
phase_fail "Pre-flight"
exit 1
fi
log "Tier 1 containers: ${TIER1_CONTAINERS[*]}"
phase_pass "Pre-flight"
# ==============================================================================================
# ━━━ Phase 2 — Block Remote Connectivity ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_PING Phase 2 — Block Remote Connectivity ━━━"
warn "Adding iptables rule — dropping all traffic to $REMOTE_SERVER ($REMOTE_SERVER_NAME)"
if [[ "$DRY_RUN" == false ]]; then
iptables -I OUTPUT -d "$REMOTE_SERVER" -j DROP
IPTABLES_RULE_ACTIVE=true
# Verify block is working
sleep 2
if ! ping -c1 -W2 "$REMOTE_SERVER" &>/dev/null; then
log "Connectivity block confirmed — ping to remote fails as expected"
phase_pass "Block Remote"
else
error "iptables rule did not block connectivity — ping still succeeds"
phase_fail "Block Remote"
exit 1
fi
else
warn "DRY RUN — would block $REMOTE_SERVER with iptables DROP rule"
phase_pass "Block Remote"
fi
# ==============================================================================================
# ━━━ Phase 3 — Fallback Detection ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_FALLBACK Phase 3 — Fallback Detection ━━━"
warn "Waiting ${FALLBACK_TEST_BLOCK_WAIT}s for fallback.sh to detect outage..."
log "fallback.sh check interval: ${FALLBACK_CHECK_INTERVAL}s"
if [[ "$DRY_RUN" == false ]]; then
sleep "$FALLBACK_TEST_BLOCK_WAIT"
if [[ -f "$FALLBACK_STATE_FILE" ]]; then
NEW_STATE=$(grep "^state=" "$FALLBACK_STATE_FILE" 2>/dev/null | cut -d= -f2)
if [[ "$NEW_STATE" == "FALLBACK" ]]; then
log "State changed to FALLBACK — outage detected correctly ✅"
phase_pass "Fallback Detection"
else
error "State is $NEW_STATE — expected FALLBACK after ${FALLBACK_TEST_BLOCK_WAIT}s"
warn "Is fallback.sh running? Check User Scripts plugin"
phase_fail "Fallback Detection"
fi
else
error "No state file found after wait — fallback.sh may not be running"
phase_fail "Fallback Detection"
fi
else
warn "DRY RUN — would wait ${FALLBACK_TEST_BLOCK_WAIT}s then check for FALLBACK state"
phase_pass "Fallback Detection"
fi
# ==============================================================================================
# ━━━ Phase 4 — Container Start Verification ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CONTAINERS Phase 4 — Tier 1 Containers Started Locally ━━━"
log "Checking Tier 1 containers: ${TIER1_CONTAINERS[*]}"
if [[ "$DRY_RUN" == false ]]; then
CONTAINERS_OK=true
for container in "${TIER1_CONTAINERS[@]}"; do
[[ -z "$container" ]] && continue
STATUS=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Running}}' \
"$container" 2>/dev/null)
if [[ "$STATUS" == "true" ]]; then
log "$ICON_RUNNING $container is running locally ✅"
else
error "$ICON_NOT_RUNNING $container is NOT running locally"
CONTAINERS_OK=false
fi
done
if [[ "$CONTAINERS_OK" == true ]]; then
phase_pass "Container Start"
else
phase_fail "Container Start"
fi
else
warn "DRY RUN — would verify these Tier 1 containers started: ${TIER1_CONTAINERS[*]}"
phase_pass "Container Start"
fi
# ==============================================================================================
# ━━━ Phase 5 — Restore Remote Connectivity ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_PING Phase 5 — Restore Remote Connectivity ━━━"
warn "Removing iptables block — $REMOTE_SERVER_NAME becomes reachable again"
if [[ "$DRY_RUN" == false ]]; then
iptables -D OUTPUT -d "$REMOTE_SERVER" -j DROP 2>/dev/null
IPTABLES_RULE_ACTIVE=false
sleep 3
if ping_remote; then
log "$REMOTE_SERVER_NAME is reachable again ✅"
phase_pass "Restore Connectivity"
else
error "$REMOTE_SERVER_NAME still unreachable after removing iptables rule"
phase_fail "Restore Connectivity"
fi
else
warn "DRY RUN — would remove iptables rule"
phase_pass "Restore Connectivity"
fi
# ==============================================================================================
# ━━━ Phase 6 — Handback ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_FALLBACK Phase 6 — Handback ━━━"
warn "Waiting ${FALLBACK_TEST_HANDBACK_WAIT}s for fallback.sh to complete handback..."
log "Requires $FALLBACK_HANDBACK_STRIKES consecutive checks at ${FALLBACK_CHECK_INTERVAL}s"
log "Minimum handback time: $(( FALLBACK_HANDBACK_STRIKES * FALLBACK_CHECK_INTERVAL ))s"
if [[ "$DRY_RUN" == false ]]; then
sleep "$FALLBACK_TEST_HANDBACK_WAIT"
if [[ -f "$FALLBACK_STATE_FILE" ]]; then
FINAL_STATE=$(grep "^state=" "$FALLBACK_STATE_FILE" 2>/dev/null | cut -d= -f2)
if [[ "$FINAL_STATE" == "NORMAL" ]]; then
log "State returned to NORMAL — handback completed ✅"
phase_pass "Handback"
else
error "State is $FINAL_STATE — expected NORMAL after ${FALLBACK_TEST_HANDBACK_WAIT}s"
warn "Handback may still be in progress — check fallback.sh output"
phase_fail "Handback"
fi
else
error "No state file found"
phase_fail "Handback"
fi
else
warn "DRY RUN — would wait ${FALLBACK_TEST_HANDBACK_WAIT}s then verify NORMAL state"
phase_pass "Handback"
fi
# ==============================================================================================
# ━━━ Phase 7 — Container Handback Verification ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CONTAINERS Phase 7 — Tier 1 Containers Stopped Locally ━━━"
log "Verifying Tier 1 containers returned to $REMOTE_SERVER_NAME"
if [[ "$DRY_RUN" == false ]]; then
HANDBACK_OK=true
for container in "${TIER1_CONTAINERS[@]}"; do
[[ -z "$container" ]] && continue
STATUS=$(timeout "$DOCKER_TIMEOUT" docker inspect -f '{{.State.Running}}' \
"$container" 2>/dev/null)
if [[ "$STATUS" != "true" ]]; then
log "$ICON_NOT_RUNNING $container stopped locally — handed back ✅"
else
error "$ICON_RUNNING $container still running locally — handback may have failed"
HANDBACK_OK=false
fi
done
if [[ "$HANDBACK_OK" == true ]]; then
phase_pass "Container Handback"
else
phase_fail "Container Handback"
fi
else
warn "DRY RUN — would verify Tier 1 containers stopped locally after handback"
phase_pass "Container Handback"
fi
TOTAL_END=$(date +%s)
# ==============================================================================================
# ━━━ Test Report ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY FALLBACK TEST REPORT ━━━━━"
echo "$ICON_HOST My ID: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_HOST Remote: $REMOTE_ID ($REMOTE_SERVER_NAME)"
echo "$ICON_TIME Duration: $(format_duration $((TOTAL_END - TOTAL_START)))"
echo ""
echo " Phase Results:"
for phase in "${PHASES_PASS[@]}"; do
echo " $ICON_SUCCESS $phase"
done
for phase in "${PHASES_FAIL[@]}"; do
echo " $ICON_ERROR $phase"
done
echo ""
PASS_COUNT=${#PHASES_PASS[@]}
FAIL_COUNT=${#PHASES_FAIL[@]}
TOTAL_PHASES=$(( PASS_COUNT + FAIL_COUNT ))
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no changes made"
elif [[ "$FAIL_COUNT" -eq 0 ]]; then
warn "$ICON_DONE ALL $TOTAL_PHASES PHASES PASSED"
notify "Fallback test PASSED on $(hostname) — all $TOTAL_PHASES phases completed" \
"Fallback Test" "normal"
else
error "$FAIL_COUNT/$TOTAL_PHASES PHASES FAILED"
notify "Fallback test FAILED on $(hostname) — $FAIL_COUNT/$TOTAL_PHASES phases failed: ${PHASES_FAIL[*]}" \
"Fallback Test" "warning"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
[[ "$FAIL_COUNT" -gt 0 ]] && exit 1
exit 0
@@ -1,171 +0,0 @@
# ━━━━━ TRANSCODES ━━━━━
**Ramdisk-based transcode storage with automatic SSD fallback.** Emby transcodes to RAM
at full speed. When the ramdisk fills, new sessions shift to SSD automatically —
without interrupting anything already playing. When pressure drops, new sessions shift
back to RAM.
> **Three configuration requirements that are not obvious and were all discovered the
> hard way in production.** The `transcoding-temp` directory must be pre-created on
> the ramdisk or Emby finds the SSD version and routes all sessions there until
> restarted. GPU containers require `--gpus "device=UUID"` in Extra Parameters —
> not `--runtime=nvidia`. Do **not** use `bind-propagation=shared` on Unraid 7.3+
> (Docker 29.x / runc v1.3.5+) — it crashes container start; the ramdisk is already
> MS\_SHARED at the kernel level. Additionally: any GPU-accelerated sidecar (OCR plugins,
> credit detection) that holds VRAM and never releases it will starve Emby and Jellyfin
> of VRAM for transcoding — Jellyfin hard-fails, Emby silently falls back to CPU.
> All three are documented in Manual-Transcoding.md.
---
## ━━━ REQUIRED EXTRA PARAMETERS ━━━
> **Stop. Set this before starting Emby or Jellyfin. If you Google how to add GPU
> access to a Docker container on unRAID you will find the wrong answer.** Every
> forum post and guide shows `--runtime=nvidia` + `NVIDIA_VISIBLE_DEVICES`. That
> method breaks on container rebuilds. Use `--gpus` instead.
In the unRAID Docker template, open **Advanced View** and paste the following into
the **Extra Parameters** field. Do not use the path mapping UI for the transcode
directory — it does not support the `--mount` syntax.
**GPU-accelerated (Emby, Jellyfin with NVENC/NVDEC) — use this:**
```
--gpus "device=GPU-62e1659d-1ed4-935f-3df3-4bb4339438f1" --pids-limit=0 --mount type=bind,source=/mnt/ram-transcode,target=/ext-ram-transcode
```
**Non-GPU — use this:**
```
--mount type=bind,source=/mnt/ram-transcode,target=/ext-ram-transcode
```
In Emby and Jellyfin's transcoding settings, set the transcode temp path to `/ext-ram-transcode`.
Find your GPU UUID: `nvidia-smi -L`
HOST1 GPU UUID (Quadro P2000): `GPU-62e1659d-1ed4-935f-3df3-4bb4339438f1`
**→ Full explanation: [Manual-Transcoding.md](Manual-Transcoding.md)**
---
## ━━━ THE PROBLEM THAT BUILT THIS ━━━
**Three Storage Options, None Perfect on Their Own**
Hard drives: seek times cause buffering on multi-stream transcoding. SSD: fast enough,
but constant small file writes at Emby volume accelerate wear over months. RAM: fastest,
no wear, files vanish instantly on session end — but limited by available memory.
Fix: RAM by default, SSD as a safety net. The system manages the transition automatically.
**Changing Transcode Location Requires Restarting Emby**
Configuring Emby to switch between ramdisk and SSD requires a restart. Restarting
during active streams drops everyone. A 7-person household with 5 Live TV streams at
9pm is not a good moment to restart Emby.
Fix: symlink indirection. Emby points at a fixed path. The symlink target changes.
ffmpeg resolves the symlink once at session start — existing sessions are completely
unaffected by flips. Only new sessions follow the new target.
**Docker Bind Mount Silently Ignored After First Flip** *(historical — resolved differently)*
Symlink flip from ramdisk → SSD worked. Flip back: nothing. All new sessions still land
on SSD. The symlink on the host is correct. Emby doesn't see it.
Original fix was `bind-propagation=shared` — but this crashes runc v1.3.5+ (Docker 29.x,
Unraid 7.3+) on any container start. The ramdisk tmpfs is `MS_SHARED` at the kernel level,
so propagation is inherited automatically without specifying it in Docker. Do not add
`bind-propagation=shared` to Extra Parameters.
**Sessions Drifting to SSD After a Day of Operation**
System working correctly for hours, then sessions gradually drift to SSD despite the
ramdisk having plenty of space.
Cause: cleanup was removing the empty `transcoding-temp` directory from the ramdisk.
Emby then found the SSD fallback version and routed all sessions there.
Fix: `transcoding-temp` is excluded from cleanup by name. `ramdisk_setup.sh` pre-creates
it at mount time. Both protections together prevent this permanently.
**lsof Per File on a Live TV System**
Early cleanup called `lsof filename` per file to check if anything had it open. On a busy
Live TV night with 5 simultaneous streams, the ramdisk contains thousands of HLS segment
files — thousands of subprocess calls every 7 minutes.
Fix: lsof called once per location to build a complete open-file map. All subsequent
checks are O(1) lookups against that map.
---
## ━━━ WHAT THIS FOLDER DOES ━━━
Three scripts, one goal: keep transcodes on RAM, fall back to SSD when needed.
`ramdisk_setup.sh` runs at array start — creates the tmpfs, SSD fallback directory,
symlink, and pre-creates `transcoding-temp`. Everything that must exist before Emby starts.
`transcode_cleanup.sh` runs first in every 7-minute cycle — removes stale files from both
ramdisk and SSD. Cleans up before usage is measured, so the manager sees real load.
`transcode_manager.sh` runs second — measures ramdisk usage, flips the symlink if
thresholds are crossed, runs safety checks, displays active sessions, writes the daily log.
The symlink is the mechanism that makes this seamless. Emby writes to a fixed path. That
path is a symlink whose target is managed at runtime. Sessions in progress never notice.
---
## ━━━ RELATIONSHIP TO OTHER FOLDERS ━━━
```
System_Essentials/
array_started.sh ──────────────────────────────► ramdisk_setup.sh (at array start)
Orchestrators/
transcode_management.sh ──── cleanup first ──► transcode_cleanup.sh
──── then manager ──► transcode_manager.sh
(every 7 minutes — order non-negotiable)
Monitors/
weekly_health_digest.sh ◄─── reads ──────────── TRANSCODE_DAILY_LOG
```
Do not schedule `transcode_cleanup.sh` or `transcode_manager.sh` directly.
Both are called by `transcode_management.sh` in the correct order.
---
## ━━━ SCRIPTS IN THIS FOLDER ━━━
| Script | Role | When It Runs |
|--------|------|-------------|
| `ramdisk_setup.sh` | Create tmpfs, SSD fallback dir, symlink, transcoding-temp | At array start (via array_started.sh) |
| `transcode_cleanup.sh` | Remove stale files, check for flip-back opportunity | Every 3 min via transcode_management.sh — runs first |
| `transcode_manager.sh` | Check usage, flip symlink, safety checks, session display, daily log | Every 3 min via transcode_management.sh — runs second |
---
## ━━━ HOW THE SCRIPTS RELATE ━━━
```
Array starts
ramdisk_setup.sh
Creates: /mnt/ramdisk_transcodes (tmpfs)
/mnt/ramdisk_transcodes/transcoding-temp/
/mnt/cache/Temp_Storage/Emby/Transcodes/ (SSD fallback)
/mnt/ram-transcode → /mnt/ramdisk_transcodes (symlink)
Emby starts, reads transcode path from config
Sees: /ext-ram-transcode (bind-mounted from /mnt/ram-transcode)
All new sessions write to: /mnt/ram-transcode → /mnt/ramdisk_transcodes/
Every 7 minutes (transcode_management.sh):
├─ transcode_cleanup.sh
│ Remove files older than TRANSCODE_MAX_AGE, not open by any process
│ transcoding-temp: never deleted
│ If ramdisk recovered below RAMDISK_LOW_GB → trigger flip-back
└─ transcode_manager.sh
Safety checks (symlink, ramdisk mount, transcoding-temp, permissions)
smart mode: ramdisk > RAMDISK_WARN_GB → flip symlink to SSD
ramdisk < RAMDISK_LOW_GB → flip symlink back to ramdisk
Session display (all TRANSCODE_SERVERS)
Append to TRANSCODE_DAILY_LOG
```
@@ -1,171 +0,0 @@
# ━━━━━ TRANSCODES ━━━━━
**Ramdisk-based transcode storage with automatic SSD fallback.** Emby transcodes to RAM
at full speed. When the ramdisk fills, new sessions shift to SSD automatically —
without interrupting anything already playing. When pressure drops, new sessions shift
back to RAM.
> **Three configuration requirements that are not obvious and were all discovered the
> hard way in production.** The `transcoding-temp` directory must be pre-created on
> the ramdisk or Emby finds the SSD version and routes all sessions there until
> restarted. GPU containers require `--gpus "device=UUID"` in Extra Parameters —
> not `--runtime=nvidia`. Do **not** use `bind-propagation=shared` on Unraid 7.3+
> (Docker 29.x / runc v1.3.5+) — it crashes container start; the ramdisk is already
> MS\_SHARED at the kernel level. Additionally: any GPU-accelerated sidecar (OCR plugins,
> credit detection) that holds VRAM and never releases it will starve Emby and Jellyfin
> of VRAM for transcoding — Jellyfin hard-fails, Emby silently falls back to CPU.
> All three are documented in Manual-Transcoding.md.
---
## ━━━ REQUIRED EXTRA PARAMETERS ━━━
> **Stop. Set this before starting Emby or Jellyfin. If you Google how to add GPU
> access to a Docker container on unRAID you will find the wrong answer.** Every
> forum post and guide shows `--runtime=nvidia` + `NVIDIA_VISIBLE_DEVICES`. That
> method breaks on container rebuilds. Use `--gpus` instead.
In the unRAID Docker template, open **Advanced View** and paste the following into
the **Extra Parameters** field. Do not use the path mapping UI for the transcode
directory — it does not support the `--mount` syntax.
**GPU-accelerated (Emby, Jellyfin with NVENC/NVDEC) — use this:**
```
--gpus "device=GPU-62e1659d-1ed4-935f-3df3-4bb4339438f1" --pids-limit=0 --mount type=bind,source=/mnt/ram-transcode,target=/ext-ram-transcode
```
**Non-GPU — use this:**
```
--mount type=bind,source=/mnt/ram-transcode,target=/ext-ram-transcode
```
In Emby and Jellyfin's transcoding settings, set the transcode temp path to `/ext-ram-transcode`.
Find your GPU UUID: `nvidia-smi -L`
HOST1 GPU UUID (Quadro P2000): `GPU-62e1659d-1ed4-935f-3df3-4bb4339438f1`
**→ Full explanation: [Manual-Transcoding.md](Manual-Transcoding.md)**
---
## ━━━ THE PROBLEM THAT BUILT THIS ━━━
**Three Storage Options, None Perfect on Their Own**
Hard drives: seek times cause buffering on multi-stream transcoding. SSD: fast enough,
but constant small file writes at Emby volume accelerate wear over months. RAM: fastest,
no wear, files vanish instantly on session end — but limited by available memory.
Fix: RAM by default, SSD as a safety net. The system manages the transition automatically.
**Changing Transcode Location Requires Restarting Emby**
Configuring Emby to switch between ramdisk and SSD requires a restart. Restarting
during active streams drops everyone. A 7-person household with 5 Live TV streams at
9pm is not a good moment to restart Emby.
Fix: symlink indirection. Emby points at a fixed path. The symlink target changes.
ffmpeg resolves the symlink once at session start — existing sessions are completely
unaffected by flips. Only new sessions follow the new target.
**Docker Bind Mount Silently Ignored After First Flip** *(historical — resolved differently)*
Symlink flip from ramdisk → SSD worked. Flip back: nothing. All new sessions still land
on SSD. The symlink on the host is correct. Emby doesn't see it.
Original fix was `bind-propagation=shared` — but this crashes runc v1.3.5+ (Docker 29.x,
Unraid 7.3+) on any container start. The ramdisk tmpfs is `MS_SHARED` at the kernel level,
so propagation is inherited automatically without specifying it in Docker. Do not add
`bind-propagation=shared` to Extra Parameters.
**Sessions Drifting to SSD After a Day of Operation**
System working correctly for hours, then sessions gradually drift to SSD despite the
ramdisk having plenty of space.
Cause: cleanup was removing the empty `transcoding-temp` directory from the ramdisk.
Emby then found the SSD fallback version and routed all sessions there.
Fix: `transcoding-temp` is excluded from cleanup by name. `ramdisk_setup.sh` pre-creates
it at mount time. Both protections together prevent this permanently.
**lsof Per File on a Live TV System**
Early cleanup called `lsof filename` per file to check if anything had it open. On a busy
Live TV night with 5 simultaneous streams, the ramdisk contains thousands of HLS segment
files — thousands of subprocess calls every 7 minutes.
Fix: lsof called once per location to build a complete open-file map. All subsequent
checks are O(1) lookups against that map.
---
## ━━━ WHAT THIS FOLDER DOES ━━━
Three scripts, one goal: keep transcodes on RAM, fall back to SSD when needed.
`ramdisk_setup.sh` runs at array start — creates the tmpfs, SSD fallback directory,
symlink, and pre-creates `transcoding-temp`. Everything that must exist before Emby starts.
`transcode_cleanup.sh` runs first in every 7-minute cycle — removes stale files from both
ramdisk and SSD. Cleans up before usage is measured, so the manager sees real load.
`transcode_manager.sh` runs second — measures ramdisk usage, flips the symlink if
thresholds are crossed, runs safety checks, displays active sessions, writes the daily log.
The symlink is the mechanism that makes this seamless. Emby writes to a fixed path. That
path is a symlink whose target is managed at runtime. Sessions in progress never notice.
---
## ━━━ RELATIONSHIP TO OTHER FOLDERS ━━━
```
System_Essentials/
array_started.sh ──────────────────────────────► ramdisk_setup.sh (at array start)
Orchestrators/
transcode_management.sh ──── cleanup first ──► transcode_cleanup.sh
──── then manager ──► transcode_manager.sh
(every 7 minutes — order non-negotiable)
Monitors/
weekly_health_digest.sh ◄─── reads ──────────── TRANSCODE_DAILY_LOG
```
Do not schedule `transcode_cleanup.sh` or `transcode_manager.sh` directly.
Both are called by `transcode_management.sh` in the correct order.
---
## ━━━ SCRIPTS IN THIS FOLDER ━━━
| Script | Role | When It Runs |
|--------|------|-------------|
| `ramdisk_setup.sh` | Create tmpfs, SSD fallback dir, symlink, transcoding-temp | At array start (via array_started.sh) |
| `transcode_cleanup.sh` | Remove stale files, check for flip-back opportunity | Every 7 minutes via transcode_management.sh — runs first |
| `transcode_manager.sh` | Check usage, flip symlink, safety checks, session display, daily log | Every 7 minutes via transcode_management.sh — runs second |
---
## ━━━ HOW THE SCRIPTS RELATE ━━━
```
Array starts
ramdisk_setup.sh
Creates: /mnt/ramdisk_transcodes (tmpfs)
/mnt/ramdisk_transcodes/transcoding-temp/
/mnt/cache/Temp_Storage/Emby/Transcodes/ (SSD fallback)
/mnt/ram-transcode → /mnt/ramdisk_transcodes (symlink)
Emby starts, reads transcode path from config
Sees: /ext-ram-transcode (bind-mounted from /mnt/ram-transcode)
All new sessions write to: /mnt/ram-transcode → /mnt/ramdisk_transcodes/
Every 7 minutes (transcode_management.sh):
├─ transcode_cleanup.sh
│ Remove files older than TRANSCODE_MAX_AGE, not open by any process
│ transcoding-temp: never deleted
│ If ramdisk recovered below RAMDISK_LOW_GB → trigger flip-back
└─ transcode_manager.sh
Safety checks (symlink, ramdisk mount, transcoding-temp, permissions)
smart mode: ramdisk > RAMDISK_WARN_GB → flip symlink to SSD
ramdisk < RAMDISK_LOW_GB → flip symlink back to ramdisk
Session display (all TRANSCODE_SERVERS)
Append to TRANSCODE_DAILY_LOG
```
@@ -1,650 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= Partnership Onboard ============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Runs once on both servers to establish a new partnership. Role is detected
# automatically via detect_hosts() — no flags needed to declare which side you are.
# Run on the mirror first (generates its SSH key), then on the owner to complete
# setup remotely.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# MIRROR PATH (1 step)
# Step 1: SSH key setup — generate keypair, copy to owner, update conf
# Owner completes the rest remotely. Mirror is done.
#
# OWNER PATH (8 steps)
# Step 1: SSH key setup — generate keypair, install on mirror, update conf
# Step 2: Stop mirror auth — stop mirror's existing auth containers before replacing
# Step 3: Deploy auth stack — push XMLs, pull images, create + start on mirror
# Mariadb/Redis health-checked before Authelia deploys
# Step 4: Stop mirror arr — stop mirror's existing arr containers before replacing
# Step 5: Deploy arr stack — push arr XMLs, pull images, create + start on mirror
# Step 6: Partnership onboard — configure WebUIs → owner IP, write state, Emby
# Step 7: Arr bootstrap — bidirectional library sync (arr_sync.sh)
# Step 8: Conf push — push master.conf + setup state to all listed hosts
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Credentials never in SSH command strings
# Auth stack containers hold API keys, DB passwords, etc. The deploy script is written
# locally, SCPed to the remote, and executed there. Command-line args are never used
# to pass credentials — they'd appear in `ps` output and shell history on both servers.
#
# XML templates are the single source of truth for deployed containers
# The owner's templates-user/ XMLs define every container deployed on the mirror.
# The same XMLs that Unraid's Docker Manager uses are what get SCPed — the mirror's
# Docker Manager can manage the containers after onboard without additional config.
#
# Dependency ordering in the auth stack is owner-enforced
# PARTNERSHIP_AUTH_STACK order matters: Mariadb and Redis must come before Authelia.
# The array is ordered correctly in host1.conf. After each Mariadb/Redis deploy,
# the script waits for the container to be healthy before continuing. This is a remote
# health check — the container must be running (or report healthy) before the next
# dependent is deployed.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Root check
# All operations run as root — SSH key management, docker operations, conf updates.
#
# SSH timeout on all remote calls
# Every ssh/scp call uses SSH_TIMEOUT. No operation hangs indefinitely on a
# slow or unreachable mirror.
#
# --dry-run shows exact actions without executing
# Every step prints what it would do. SCP, deploy, plugin install, arr sync —
# all dry-run safe.
#
# Step skip flags for partial re-runs
# --skip-ssh, --skip-auth-stack, --skip-arr-stack, --skip-arr-sync allow
# resuming after a partial failure without re-running completed steps.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_PARTNERSHIP_AUTH_STACK
# XML filenames (from this server's templates-user/) to push and deploy on the
# mirror as its auth stack. Order matters: database deps before Authelia.
# Aliased by detect_hosts() → PARTNERSHIP_AUTH_STACK
#
# HOST*_PARTNERSHIP_REPLACE_CONTAINERS
# Containers to stop on the mirror before deploying the auth stack.
# Defined in the MIRROR's own conf (host*.conf on HOST2) — never in HOST1's conf.
# Read live from the mirror via SSH during Step 3 (sources mirror's load_config.sh at
# the same $SCRIPTS_ROOT path — convention: both servers use the same repo location).
# Leave empty on HOST2 if no conflicting containers exist (fresh mirror: nothing to stop).
# Aliased by detect_hosts() → PARTNERSHIP_REPLACE_CONTAINERS (on the mirror)
#
# HOST*_PARTNERSHIP_ARR_STACK
# XML filenames to push and deploy on the mirror as its arr stack.
# Leave empty to skip arr stack deploy.
# Aliased by detect_hosts() → PARTNERSHIP_ARR_STACK
#
# HOST*_PARTNERSHIP_ARR_REPLACE_CONTAINERS
# Arr containers to stop on the mirror before deploying the arr stack.
# Same rule as PARTNERSHIP_REPLACE_CONTAINERS: defined in mirror's own conf, never HOST1's.
# Aliased by detect_hosts() → PARTNERSHIP_ARR_REPLACE_CONTAINERS (on the mirror)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# Partnership/partnership_onboard.sh
# Full onboard — role detected automatically
#
# Partnership/partnership_onboard.sh --dry-run
# Preview all steps without making changes
#
# Partnership/partnership_onboard.sh --log
# Verbose per-step output
#
# Partnership/partnership_onboard.sh --skip-ssh
# Skip SSH key setup (key already in place)
#
# Partnership/partnership_onboard.sh --skip-auth-stack
# Skip auth stack stop + deploy (Steps 3-4)
#
# Partnership/partnership_onboard.sh --skip-arr-stack
# Skip arr stack stop + deploy (Steps 5-6)
#
# Partnership/partnership_onboard.sh --skip-arr-sync
# Skip arr library bootstrap (Step 8)
#
# Partnership/partnership_onboard.sh --phase1-only
# OWNER only: SSH key exchange + conf push. Safe to run before HOST2 has Varaverk.
# Writes HOST2_PHASE1_DONE=true to varaverk_setup.db.
#
# Partnership/partnership_onboard.sh --phase2-only
# OWNER only: container deploy + arr + onboard (skips SSH). Triggered automatically
# by HOST2 after it completes its Mirror-path onboard. Can also be run manually.
# Writes HOST2_PHASE2_DONE=true to varaverk_setup.db.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
SCRIPTS_ROOT="$SCRIPT_DIR/.."
SSH_TIMEOUT=15
source "$SCRIPTS_ROOT/load_config.sh"
source "$SCRIPTS_ROOT/Plugin/$PLATFORM/Partnership/containers.sh"
# ── Parse flags ───────────────────────────────────────────────────────────────────────────────
SKIP_SSH=false
SKIP_AUTH_STACK=false
SKIP_ARR_STACK=false
SKIP_ARR_SYNC=false
PHASE1_ONLY=false # OWNER: SSH + conf push only (HOST2 not yet installed)
PHASE2_ONLY=false # OWNER: containers/arr/onboard only (triggered by HOST2 after it onboards)
FILTERED_ARGS=()
for arg in "$@"; do
case "$arg" in
--skip-ssh) SKIP_SSH=true ;;
--skip-auth-stack) SKIP_AUTH_STACK=true ;;
--skip-arr-stack) SKIP_ARR_STACK=true ;;
--skip-arr-sync) SKIP_ARR_SYNC=true ;;
--phase1-only) PHASE1_ONLY=true ;;
--phase2-only) PHASE2_ONLY=true; SKIP_SSH=true ;;
*) FILTERED_ARGS+=("$arg") ;;
esac
done
parse_args "${FILTERED_ARGS[@]}"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
[[ "$EUID" -ne 0 ]] && { error "Must be run as root"; exit 1; }
acquire_lock
if ! command -v docker &>/dev/null; then
error "Docker command not found"
exit 1
fi
detect_hosts
OWNER_ID="${PARTNERSHIP_OWNER_HOST:-HOST1}"
MIRROR_ID=$( [[ "$OWNER_ID" == "HOST1" ]] && echo "HOST2" || echo "HOST1" )
OWNER="${!OWNER_ID}"
MIRROR="${!MIRROR_ID}"
# SSH_KEY (set by detect_hosts) is this server's own private key.
# The remote accepts it because this server's PUBLIC key was installed there via ssh_setup.sh.
# HOST{N}_SSH_KEY lives in host{N}.conf — with sparse checkout, the other server's
# conf is never present here. Always use SSH_KEY (local private key) for outbound SSH.
MIRROR_SSH_KEY="$SSH_KEY"
AM_OWNER=false
AM_MIRROR=false
[[ "$MY_ID" == "$OWNER_ID" ]] && AM_OWNER=true
[[ "$MY_ID" == "$MIRROR_ID" ]] && AM_MIRROR=true
EXTRA_FLAGS=()
[[ "$DRY_RUN" == true ]] && EXTRA_FLAGS+=("--dry-run")
[[ "$LOG_MODE" == true ]] && EXTRA_FLAGS+=("--log")
START=$(date +%s)
# ── Helper: write phase completion flag to setup.db + push to remotes ─────────────────────────
write_onboard_phase() {
local target_id="$1" phase="$2"
local key="${target_id}_PHASE${phase}_DONE"
local state_file="${VARAVERK_SETUP_FILE:-${STATE_DIR:-/boot/config}/varaverk_setup.db}"
[[ "$DRY_RUN" == true ]] && { warn "DRY RUN — would write ${key}=true"; return 0; }
if grep -q "^${key}=" "$state_file" 2>/dev/null; then
sed -i "s|^${key}=.*|${key}=true|" "$state_file"
else
echo "${key}=true" >> "$state_file"
fi
command -v php &>/dev/null && \
php -r "require_once '/usr/local/emhttp/plugins/varaverk/include/config.php'; vv_push_setup_state();" 2>/dev/null || true
}
echo ""
echo "━━━ $ICON_FALLBACK Partnership Onboard — $MY_ID ($LOCAL_SERVER_NAME) — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
echo ""
echo " Role: $( [[ "$AM_OWNER" == true ]] && echo "OWNER" || echo "MIRROR" )"
echo " This: $MY_ID ($LOCAL_SERVER_NAME)"
echo " Partner: $( [[ "$AM_OWNER" == true ]] && echo "$MIRROR_ID ($MIRROR)" || echo "$OWNER_ID ($OWNER)" )"
echo ""
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no permanent changes will be made"
# ==============================================================================================
# ── HELPER: stop containers on the mirror by reading its own conf via SSH ────────────────────
#
# SSHes to the mirror, sources its load_config.sh at the same $SCRIPTS_ROOT path (both servers
# use the same convention), and reads the named config array from the mirror's own conf.
# HOST2's container list stays in HOST2's host2.conf — not duplicated in HOST1's conf.
# Fails gracefully if scripts aren't present yet or the array is empty (nothing to stop).
#
# deploy_container_from_xml() already stops/removes containers with the same name as what's
# being deployed. This step handles containers with DIFFERENT names that conflict.
# ==============================================================================================
stop_mirror_stack() {
local config_var="$1" label="$2"
local -a to_stop=()
mapfile -t to_stop < <(
timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$MIRROR_IP" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${${config_var}[@]:-}\"" 2>/dev/null | grep -v '^$'
)
if [[ ${#to_stop[@]} -eq 0 ]]; then
log "No $label containers to stop on $MIRROR — skipping"
return 0
fi
log "Stopping $label on $MIRROR: ${to_stop[*]}"
for container in "${to_stop[@]}"; do
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would stop + rm $container on $MIRROR"
continue
fi
timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" root@"$MIRROR_IP" \
"docker stop '$container' 2>/dev/null
docker rm '$container' 2>/dev/null && echo removed" 2>/dev/null | \
grep -q removed && \
log " $container removed ✅" || \
log " $container not found on $MIRROR — skipping"
done
}
# ==============================================================================================
# ── MIRROR PATH ───────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
if [[ "$AM_MIRROR" == true ]]; then
echo "━━━ Step 1/2 — SSH Key Setup (Mirror) ━━━"
echo ""
echo " Mirror sets up SSH keys, then notifies Owner to run Phase 2."
echo ""
if [[ "$SKIP_SSH" == true ]]; then
warn "Skipping SSH setup (--skip-ssh)"
elif bash "$SCRIPT_DIR/ssh_setup.sh" "${EXTRA_FLAGS[@]}"; then
log "SSH key ready ✅"
else
error "SSH key setup failed"
exit 1
fi
echo ""
echo "━━━ Step 2/2 — Notify Owner to Run Phase 2 ━━━"
echo ""
OWNER_IP=$(resolve_tailscale_ip "$OWNER" 2>/dev/null || true)
PHASE2_TRIGGERED=false
if [[ -n "$OWNER_IP" ]]; then
# Read OWNER's SCRIPTS_DIR from their varaverk.cfg — don't assume same path as mirror
OWNER_SCRIPTS_DIR=$(timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$OWNER_IP" \
'grep SCRIPTS_DIR /boot/config/plugins/varaverk/varaverk.cfg 2>/dev/null | cut -d= -f2 | tr -d "\"'"'"'" 2>/dev/null' 2>/dev/null | tr -d '[:space:]')
OWNER_SCRIPTS_DIR="${OWNER_SCRIPTS_DIR:-/boot/config/plugins/varaverk}"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would SSH to $OWNER ($OWNER_IP) and trigger Phase 2"
PHASE2_TRIGGERED=true
elif timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$OWNER_IP" \
"nohup bash '${OWNER_SCRIPTS_DIR}/Partnership/partnership_onboard.sh' --phase2-only > /tmp/vv_phase2_onboard.log 2>&1 & echo triggered" \
2>/dev/null | grep -q triggered; then
log "Phase 2 triggered on $OWNER ✅"
log "Watch progress on $OWNER: tail -f /tmp/vv_phase2_onboard.log"
PHASE2_TRIGGERED=true
else
warn "Could not auto-trigger Phase 2 on $OWNER"
fi
else
warn "Cannot resolve $OWNER Tailscale IP"
fi
echo ""
echo "━━━━━ $ICON_SUMMARY MIRROR SETUP COMPLETE ━━━━━"
echo " SSH key: ready"
echo " Phase 2 on $OWNER: $( [[ "$PHASE2_TRIGGERED" == true ]] && echo "triggered ✅" || echo "needs manual trigger ⚠" )"
if [[ "$PHASE2_TRIGGERED" == false ]]; then
echo ""
echo " Run manually on $OWNER:"
echo " bash Partnership/partnership_onboard.sh --phase2-only"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ── OWNER PATH ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
MIRROR_IP=$(resolve_tailscale_ip "$MIRROR")
[[ -z "$MIRROR_IP" ]] && { error "Cannot resolve $MIRROR Tailscale IP — is Tailscale running?"; exit 1; }
log "Mirror: $MIRROR ($MIRROR_IP)"
[[ "$PHASE1_ONLY" == true ]] && log "Mode: Phase 1 only (SSH + conf push)"
[[ "$PHASE2_ONLY" == true ]] && log "Mode: Phase 2 only (containers + arr + onboard)"
echo ""
STEP_SSH_OK=false
STEP_STOP_AUTH_OK=true
STEP_AUTH_OK=true
AUTH_DEPLOYED=0
AUTH_FAILED=0
STEP_STOP_ARR_OK=true
STEP_ARR_OK=true
ARR_DEPLOYED=0
ARR_FAILED=0
ONBOARD_OK=false
ARR_SYNC_OK=false
MASTER_PUSH_OK=false
# ── Step 1: SSH ───────────────────────────────────────────────────────────────────────────────
# Skipped when --phase2-only (SSH was already done in Phase 1).
echo "━━━ Step 1 — SSH Key Setup ━━━"
if [[ "$SKIP_SSH" == true ]]; then
warn "Skipping (--skip-ssh)"
STEP_SSH_OK=true
elif [[ "$PHASE1_ONLY" == true ]]; then
# Phase 1 in background: test if SSH already works first — avoids ssh-copy-id
# hanging for a password prompt with no TTY.
if timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$MIRROR_IP" exit 0 2>/dev/null; then
log "SSH to $MIRROR already works ✅ — skipping key install"
STEP_SSH_OK=true
else
# Key not yet on HOST2 — try ssh_setup.sh (works interactively, may fail in background)
if bash "$SCRIPT_DIR/ssh_setup.sh" "${EXTRA_FLAGS[@]}"; then
log "SSH keys ready ✅"
STEP_SSH_OK=true
else
# Soft-fail: generate key locally if not present, then tell user to install manually
warn "Could not install key on $MIRROR automatically (no terminal for password prompt)"
if [[ -f "$SSH_KEY" ]]; then
log "Local key exists at: $SSH_KEY"
else
bash "$SCRIPT_DIR/ssh_setup.sh" --key-only "${EXTRA_FLAGS[@]}" 2>/dev/null || true
fi
if [[ -f "${SSH_KEY}.pub" ]]; then
echo ""
echo " Install this key on $MIRROR to complete SSH setup:"
echo " ┌─────────────────────────────────────────────────────"
cat "${SSH_KEY}.pub" | sed 's/^/ │ /'
echo " └─────────────────────────────────────────────────────"
echo " Run on a terminal: ssh-copy-id -i ${SSH_KEY}.pub root@${MIRROR_IP}"
echo " Then click 'Push Conf' in the Partnership tab."
# Write key-ready flag so UI can show the manual-install state
[[ "$DRY_RUN" == false ]] && {
local kflag="${MIRROR_ID}_KEY_READY"
local _setup_f="${VARAVERK_SETUP_FILE:-${STATE_DIR:-/boot/config}/varaverk_setup.db}"
grep -q "^${kflag}=" "$_setup_f" 2>/dev/null \
&& sed -i "s|^${kflag}=.*|${kflag}=true|" "$_setup_f" \
|| echo "${kflag}=true" >> "$_setup_f"
}
fi
STEP_SSH_OK=false
fi
fi
elif bash "$SCRIPT_DIR/ssh_setup.sh" "${EXTRA_FLAGS[@]}"; then
log "SSH keys ready ✅"
STEP_SSH_OK=true
else
error "SSH key setup failed — aborting"
error "Re-run or use --skip-ssh if key is already set up"
exit 1
fi
# ── Phase 1 exit point ────────────────────────────────────────────────────────────────────────
# --phase1-only: SSH + conf push is all HOST1 needs to do before HOST2 installs Varaverk.
# HOST2's wizard will detect the pushed master.conf + state file and take the correct path.
if [[ "$PHASE1_ONLY" == true ]]; then
if [[ "$STEP_SSH_OK" == false ]]; then
# SSH key not yet installed on HOST2 — can't push conf, but local setup still runs.
# UI will show "key ready, install manually" state via HOST2_KEY_READY flag.
echo ""
echo "━━━ Phase 1 — HOST1 Local Setup (SSH pending) ━━━"
bash "$SCRIPT_DIR/partnership_manager.sh" --onboard --local-only "${EXTRA_FLAGS[@]}" || \
warn "Local setup had issues — check partnership_manager.sh output above"
END=$(date +%s)
echo ""
echo "━━━━━ $ICON_SUMMARY PHASE 1 — SSH PENDING ━━━━━"
echo " SSH keys: key generated ✅ — NOT yet installed on $MIRROR ⚠"
echo " Conf push: skipped (needs SSH access to $MIRROR)"
echo " HOST1 setup: done ✅"
echo " Duration: $(format_duration $(( END - START )))"
echo ""
echo " ACTION NEEDED: install the key on $MIRROR:"
echo " ssh-copy-id -i ${SSH_KEY}.pub root@${MIRROR_IP}"
echo " Then click 'Push Conf' in Partnership tab, or run:"
echo " bash Partnership/partnership_onboard.sh --phase1-only --skip-ssh"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
echo ""
echo "━━━ Phase 1 — Conf Push ━━━"
CONF_PUSH_OK=false
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would push master.conf + state file to $MIRROR"
CONF_PUSH_OK=true
elif ! command -v php &>/dev/null; then
warn "php not available — push master.conf manually via Scheduler → master.conf → Save Conf"
else
push_output=$(php -r "
require_once '/usr/local/emhttp/plugins/varaverk/include/config.php';
\$results = vv_push_master_conf();
vv_push_setup_state();
if (empty(\$results)) { echo 'no remote hosts'; exit(0); }
\$failed = 0;
foreach (\$results as \$r) {
echo \$r['host'] . ': ' . (\$r['ok'] ? 'pushed' : 'FAILED — ' . \$r['error']) . PHP_EOL;
if (!\$r['ok']) \$failed++;
}
exit(\$failed > 0 ? 1 : 0);
" 2>/dev/null)
push_rc=$?
echo "$push_output"
if [[ $push_rc -eq 0 ]]; then
log "Conf push complete ✅"
CONF_PUSH_OK=true
else
warn "Conf push had failures — retry via Scheduler → master.conf → Save Conf"
fi
fi
# HOST1 local setup — runs immediately without needing HOST2
echo ""
echo "━━━ Phase 1 — HOST1 Local Setup ━━━"
bash "$SCRIPT_DIR/partnership_manager.sh" --onboard --local-only "${EXTRA_FLAGS[@]}" || \
warn "Local setup had issues — check partnership_manager.sh output above"
[[ "$DRY_RUN" == false ]] && write_onboard_phase "$MIRROR_ID" 1
END=$(date +%s)
echo ""
echo "━━━━━ $ICON_SUMMARY PHASE 1 COMPLETE ━━━━━"
echo " SSH keys: $( [[ "$STEP_SSH_OK" == true ]] && echo "ready ✅" || echo "skipped" )"
echo " Conf push: $( [[ "$CONF_PUSH_OK" == true ]] && echo "done ✅" || echo "⚠ manual needed" )"
echo " HOST1 setup: done ✅"
echo " Duration: $(format_duration $(( END - START )))"
echo ""
echo " HOST1 is fully set up. HOST2 ($MIRROR) can now install the Varaverk plugin."
echo " The wizard will detect the pushed conf and take the correct path."
echo " When HOST2 completes its onboard, it will automatically trigger Phase 2 here."
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ── Step 2: Stop mirror's existing auth stack ─────────────────────────────────────────────────
echo ""
echo "━━━ Step 2 — Stop Mirror Auth Stack ━━━"
if [[ "$SKIP_AUTH_STACK" == true ]]; then
warn "Skipping (--skip-auth-stack)"
else
stop_mirror_stack "PARTNERSHIP_REPLACE_CONTAINERS" "auth stack"
fi
# ── Step 4: Deploy auth stack on mirror ───────────────────────────────────────────────────────
echo ""
echo "━━━ Step 3 — Deploy Auth Stack on Mirror ━━━"
if [[ "$SKIP_AUTH_STACK" == true ]]; then
warn "Skipping (--skip-auth-stack)"
elif [[ ${#PARTNERSHIP_AUTH_STACK[@]} -eq 0 ]]; then
warn "PARTNERSHIP_AUTH_STACK not set in ${MY_ID} conf — skipping auth stack deploy"
warn "Add HOST${MY_ID: -1}_PARTNERSHIP_AUTH_STACK to host${MY_ID: -1}.conf"
STEP_AUTH_OK=false
else
deploy_xml_stack PARTNERSHIP_AUTH_STACK
AUTH_DEPLOYED=$_STACK_DEPLOYED
AUTH_FAILED=$_STACK_FAILED
echo "Auth stack: $AUTH_DEPLOYED deployed, $AUTH_FAILED failed"
[[ "$AUTH_FAILED" -gt 0 ]] && STEP_AUTH_OK=false
fi
# ── Step 5: Stop mirror's existing arr stack ──────────────────────────────────────────────────
echo ""
echo "━━━ Step 4 — Stop Mirror Arr Stack ━━━"
if [[ "$SKIP_ARR_STACK" == true ]]; then
warn "Skipping (--skip-arr-stack)"
elif [[ ${#PARTNERSHIP_ARR_STACK[@]} -eq 0 ]]; then
log "PARTNERSHIP_ARR_STACK not configured — skipping arr stack deploy"
SKIP_ARR_STACK=true
else
stop_mirror_stack "PARTNERSHIP_ARR_REPLACE_CONTAINERS" "arr stack"
fi
# ── Step 6: Deploy arr stack on mirror ───────────────────────────────────────────────────────
echo ""
echo "━━━ Step 5 — Deploy Arr Stack on Mirror ━━━"
if [[ "$SKIP_ARR_STACK" == true ]]; then
warn "Skipping (--skip-arr-stack)"
else
deploy_xml_stack PARTNERSHIP_ARR_STACK
ARR_DEPLOYED=$_STACK_DEPLOYED
ARR_FAILED=$_STACK_FAILED
echo "Arr stack: $ARR_DEPLOYED deployed, $ARR_FAILED failed"
[[ "$ARR_FAILED" -gt 0 ]] && STEP_ARR_OK=false
fi
# ── Step 7: Partnership onboard ───────────────────────────────────────────────────────────────
echo ""
echo "━━━ Step 6 — Partnership Onboard ━━━"
if bash "$SCRIPTS_ROOT/Partnership/partnership_manager.sh" --onboard "${EXTRA_FLAGS[@]}"; then
echo "Partnership onboard complete ✅"
ONBOARD_OK=true
else
error "Partnership onboard failed"
ONBOARD_OK=false
fi
# ── Step 8: Arr library bootstrap ─────────────────────────────────────────────────────────────
echo ""
echo "━━━ Step 7 — Arr Library Bootstrap ━━━"
if [[ "$ONBOARD_OK" == false ]]; then
warn "Skipping — onboard did not complete"
elif [[ "$SKIP_ARR_SYNC" == true ]]; then
warn "Skipping (--skip-arr-sync)"
elif [[ ! -f "$SCRIPTS_ROOT/Media/arr_sync.sh" ]]; then
warn "arr_sync.sh not found — run Media/arr_sync.sh manually once arrs are live"
elif bash "$SCRIPTS_ROOT/Media/arr_sync.sh" "${EXTRA_FLAGS[@]}"; then
echo "Arr bootstrap complete ✅"
ARR_SYNC_OK=true
else
warn "Arr sync had errors — partnership still valid"
warn "Re-run Media/arr_sync.sh once all arr containers are live"
fi
# ── Step 9: Push master.conf to all listed hosts ──────────────────────────────────────────────
# SSH is now established and all partners have the plugin installed.
# Push the authoritative master.conf so every listed host is in sync immediately.
echo ""
echo "━━━ $ICON_GEAR Step 8 — master.conf Push ━━━"
if [[ "$ONBOARD_OK" == false ]]; then
warn "Skipping — onboard did not complete"
elif [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would push master.conf to all listed hosts"
MASTER_PUSH_OK=true
elif ! command -v php &>/dev/null; then
warn "php not available — push master.conf manually via Scheduler → master.conf → Save Conf"
else
push_output=$(php -r "
require_once '/usr/local/emhttp/plugins/varaverk/include/config.php';
\$results = vv_push_master_conf();
vv_push_setup_state();
if (empty(\$results)) { echo 'no remote hosts'; exit(0); }
\$failed = 0;
foreach (\$results as \$r) {
echo \$r['host'] . ': ' . (\$r['ok'] ? 'pushed' : 'FAILED — ' . \$r['error']) . PHP_EOL;
if (!\$r['ok']) \$failed++;
}
exit(\$failed > 0 ? 1 : 0);
" 2>/dev/null)
push_rc=$?
echo "$push_output"
if [[ $push_rc -eq 0 ]]; then
echo "master.conf sync complete ✅"
MASTER_PUSH_OK=true
else
warn "master.conf push had failures — retry via Scheduler → master.conf → Save Conf"
fi
fi
# ── Write Phase 2 completion state ────────────────────────────────────────────────────────────
[[ "$ONBOARD_OK" == true && "$DRY_RUN" == false ]] && write_onboard_phase "$MIRROR_ID" 2
# ── Summary ───────────────────────────────────────────────────────────────────────────────────
END=$(date +%s)
echo ""
echo "━━━━━ $ICON_SUMMARY ONBOARD SUMMARY ━━━━━"
echo " Owner: $MY_ID ($LOCAL_SERVER_NAME)"
echo " Mirror: $MIRROR ($MIRROR_IP)"
[[ "$PHASE2_ONLY" == true ]] && echo " Mode: Phase 2 (triggered by HOST2 notification)"
echo " Duration: $(format_duration $(( END - START )))"
echo ""
_ok() { [[ "$1" == true ]] && echo "✅" || echo "❌"; }
_skip() { [[ "$1" == true ]] && echo "skipped" || echo "$(_ok "$2")"; }
echo " Step 1 — SSH keys: $(_skip "$SKIP_SSH" "$STEP_SSH_OK")"
echo " Step 2 — Stop auth: $(_skip "$SKIP_AUTH_STACK" "$STEP_STOP_AUTH_OK")"
echo " Step 3 — Auth stack: $( [[ "$SKIP_AUTH_STACK" == true ]] && echo "skipped" || echo "${AUTH_DEPLOYED} deployed, ${AUTH_FAILED} failed" )"
echo " Step 4 — Stop arr: $(_skip "$SKIP_ARR_STACK" "$STEP_STOP_ARR_OK")"
echo " Step 5 — Arr stack: $( [[ "$SKIP_ARR_STACK" == true ]] && echo "skipped" || echo "${ARR_DEPLOYED} deployed, ${ARR_FAILED} failed" )"
echo " Step 6 — Onboard: $(_ok "$ONBOARD_OK")"
echo " Step 7 — Arr bootstrap: $( [[ "$SKIP_ARR_SYNC" == true || "$ONBOARD_OK" == false ]] && echo "skipped" || echo "$(_ok "$ARR_SYNC_OK")" )"
echo " Step 8 — Conf push: $( [[ "$ONBOARD_OK" == false ]] && echo "skipped" || echo "$(_ok "$MASTER_PUSH_OK")" )"
echo ""
if [[ "$ONBOARD_OK" == true ]]; then
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes made" || \
echo "$ICON_DONE DONE — partnership established ✅"
echo "Verify with: Partnership/partnership_manager.sh --status"
else
error "Setup incomplete — resolve errors above and re-run"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━"
[[ "$ONBOARD_OK" == false ]] && exit 1
exit 0
@@ -1,700 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= Partnership Onboard ============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Runs once on both servers to establish a new partnership. Role is detected
# automatically via detect_hosts() — no flags needed to declare which side you are.
# Run on the mirror first (generates its SSH key), then on the owner to complete
# setup remotely.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# MIRROR PATH (1 step)
# Step 1: SSH key setup — generate keypair, copy to owner, update conf
# Owner completes the rest remotely. Mirror is done.
#
# OWNER PATH (10 steps)
# Step 1: SSH key setup — generate keypair, install on mirror, update conf
# Step 2: Stop mirror auth — stop mirror's existing auth containers before replacing
# Step 3: Deploy auth stack — push XMLs, pull images, create + start on mirror
# Mariadb/Redis health-checked before Authelia deploys
# Step 4: Stop mirror arr — stop mirror's existing arr containers before replacing
# Step 5: Deploy arr stack — push arr XMLs, pull images, create + start on mirror
# Step 6: Stop mirror services — stop mirror's existing services containers before replacing
# Step 7: Deploy services stack — push Emby/Jellyfin/Seerr XMLs, pull images, create + start
# Step 8: Partnership onboard — configure WebUIs → owner IP, write state, Emby
# Step 9: Arr bootstrap — bidirectional library sync (arr_sync.sh)
# Step 10: Conf push — push master.conf + setup state to all listed hosts
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Credentials never in SSH command strings
# Auth stack containers hold API keys, DB passwords, etc. The deploy script is written
# locally, SCPed to the remote, and executed there. Command-line args are never used
# to pass credentials — they'd appear in `ps` output and shell history on both servers.
#
# XML templates are the single source of truth for deployed containers
# The owner's templates-user/ XMLs define every container deployed on the mirror.
# The same XMLs that Unraid's Docker Manager uses are what get SCPed — the mirror's
# Docker Manager can manage the containers after onboard without additional config.
#
# Dependency ordering in the auth stack is owner-enforced
# PARTNERSHIP_AUTH_STACK order matters: Mariadb and Redis must come before Authelia.
# The array is ordered correctly in host1.conf. After each Mariadb/Redis deploy,
# the script waits for the container to be healthy before continuing. This is a remote
# health check — the container must be running (or report healthy) before the next
# dependent is deployed.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Root check
# All operations run as root — SSH key management, docker operations, conf updates.
#
# SSH timeout on all remote calls
# Every ssh/scp call uses SSH_TIMEOUT. No operation hangs indefinitely on a
# slow or unreachable mirror.
#
# --dry-run shows exact actions without executing
# Every step prints what it would do. SCP, deploy, plugin install, arr sync —
# all dry-run safe.
#
# Step skip flags for partial re-runs
# --skip-ssh, --skip-auth-stack, --skip-arr-stack, --skip-arr-sync allow
# resuming after a partial failure without re-running completed steps.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_PARTNERSHIP_AUTH_STACK
# XML filenames (from this server's templates-user/) to push and deploy on the
# mirror as its auth stack. Order matters: database deps before Authelia.
# Aliased by detect_hosts() → PARTNERSHIP_AUTH_STACK
#
# HOST*_PARTNERSHIP_REPLACE_CONTAINERS
# Containers to stop on the mirror before deploying the auth stack.
# Defined in the MIRROR's own conf (host*.conf on HOST2) — never in HOST1's conf.
# Read live from the mirror via SSH during Step 3 (sources mirror's load_config.sh at
# the same $SCRIPTS_ROOT path — convention: both servers use the same repo location).
# Leave empty on HOST2 if no conflicting containers exist (fresh mirror: nothing to stop).
# Aliased by detect_hosts() → PARTNERSHIP_REPLACE_CONTAINERS (on the mirror)
#
# HOST*_PARTNERSHIP_ARR_STACK
# XML filenames to push and deploy on the mirror as its arr stack.
# Leave empty to skip arr stack deploy.
# Aliased by detect_hosts() → PARTNERSHIP_ARR_STACK
#
# HOST*_PARTNERSHIP_ARR_REPLACE_CONTAINERS
# Arr containers to stop on the mirror before deploying the arr stack.
# Same rule as PARTNERSHIP_REPLACE_CONTAINERS: defined in mirror's own conf, never HOST1's.
# Aliased by detect_hosts() → PARTNERSHIP_ARR_REPLACE_CONTAINERS (on the mirror)
#
# HOST*_PARTNERSHIP_SERVICES_STACK
# XML filenames to push and deploy on the mirror as its shared services stack.
# Includes Emby, Jellyfin, Seerr, SeerrFin. Leave empty to skip services stack deploy.
# Aliased by detect_hosts() → PARTNERSHIP_SERVICES_STACK
#
# HOST*_PARTNERSHIP_SERVICES_REPLACE_CONTAINERS
# Services containers to stop on the mirror before deploying the services stack.
# Same rule as PARTNERSHIP_REPLACE_CONTAINERS: defined in mirror's own conf, never HOST1's.
# Aliased by detect_hosts() → PARTNERSHIP_SERVICES_REPLACE_CONTAINERS (on the mirror)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# Partnership/partnership_onboard.sh
# Full onboard — role detected automatically
#
# Partnership/partnership_onboard.sh --dry-run
# Preview all steps without making changes
#
# Partnership/partnership_onboard.sh --log
# Verbose per-step output
#
# Partnership/partnership_onboard.sh --skip-ssh
# Skip SSH key setup (key already in place)
#
# Partnership/partnership_onboard.sh --skip-auth-stack
# Skip auth stack stop + deploy (Steps 3-4)
#
# Partnership/partnership_onboard.sh --skip-arr-stack
# Skip arr stack stop + deploy (Steps 4-5)
#
# Partnership/partnership_onboard.sh --skip-services-stack
# Skip services stack stop + deploy (Steps 6-7)
#
# Partnership/partnership_onboard.sh --skip-arr-sync
# Skip arr library bootstrap (Step 9)
#
# Partnership/partnership_onboard.sh --phase1-only
# OWNER only: SSH key exchange + conf push. Safe to run before HOST2 has Varaverk.
# Writes HOST2_PHASE1_DONE=true to varaverk_setup.db.
#
# Partnership/partnership_onboard.sh --phase2-only
# OWNER only: container deploy + arr + onboard (skips SSH). Triggered automatically
# by HOST2 after it completes its Mirror-path onboard. Can also be run manually.
# Writes HOST2_PHASE2_DONE=true to varaverk_setup.db.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
SCRIPTS_ROOT="$SCRIPT_DIR/.."
SSH_TIMEOUT=15
source "$SCRIPTS_ROOT/load_config.sh"
source "$SCRIPTS_ROOT/Plugin/$PLATFORM/Partnership/containers.sh"
# ── Parse flags ───────────────────────────────────────────────────────────────────────────────
SKIP_SSH=false
SKIP_AUTH_STACK=false
SKIP_ARR_STACK=false
SKIP_SERVICES_STACK=false
SKIP_ARR_SYNC=false
PHASE1_ONLY=false # OWNER: SSH + conf push only (HOST2 not yet installed)
PHASE2_ONLY=false # OWNER: containers/arr/onboard only (triggered by HOST2 after it onboards)
FILTERED_ARGS=()
for arg in "$@"; do
case "$arg" in
--skip-ssh) SKIP_SSH=true ;;
--skip-auth-stack) SKIP_AUTH_STACK=true ;;
--skip-arr-stack) SKIP_ARR_STACK=true ;;
--skip-services-stack) SKIP_SERVICES_STACK=true ;;
--skip-arr-sync) SKIP_ARR_SYNC=true ;;
--phase1-only) PHASE1_ONLY=true ;;
--phase2-only) PHASE2_ONLY=true; SKIP_SSH=true ;;
*) FILTERED_ARGS+=("$arg") ;;
esac
done
parse_args "${FILTERED_ARGS[@]}"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
[[ "$EUID" -ne 0 ]] && { error "Must be run as root"; exit 1; }
acquire_lock
if ! command -v docker &>/dev/null; then
error "Docker command not found"
exit 1
fi
detect_hosts
OWNER_ID="${PARTNERSHIP_OWNER_HOST:-HOST1}"
MIRROR_ID=$( [[ "$OWNER_ID" == "HOST1" ]] && echo "HOST2" || echo "HOST1" )
OWNER="${!OWNER_ID}"
MIRROR="${!MIRROR_ID}"
# SSH_KEY (set by detect_hosts) is this server's own private key.
# The remote accepts it because this server's PUBLIC key was installed there via ssh_setup.sh.
# HOST{N}_SSH_KEY lives in host{N}.conf — with sparse checkout, the other server's
# conf is never present here. Always use SSH_KEY (local private key) for outbound SSH.
MIRROR_SSH_KEY="$SSH_KEY"
AM_OWNER=false
AM_MIRROR=false
[[ "$MY_ID" == "$OWNER_ID" ]] && AM_OWNER=true
[[ "$MY_ID" == "$MIRROR_ID" ]] && AM_MIRROR=true
EXTRA_FLAGS=()
[[ "$DRY_RUN" == true ]] && EXTRA_FLAGS+=("--dry-run")
[[ "$LOG_MODE" == true ]] && EXTRA_FLAGS+=("--log")
START=$(date +%s)
# ── Helper: write phase completion flag to setup.db + push to remotes ─────────────────────────
write_onboard_phase() {
local target_id="$1" phase="$2"
local key="${target_id}_PHASE${phase}_DONE"
local state_file="${VARAVERK_SETUP_FILE:-${STATE_DIR:-/boot/config}/varaverk_setup.db}"
[[ "$DRY_RUN" == true ]] && { warn "DRY RUN — would write ${key}=true"; return 0; }
if grep -q "^${key}=" "$state_file" 2>/dev/null; then
sed -i "s|^${key}=.*|${key}=true|" "$state_file"
else
echo "${key}=true" >> "$state_file"
fi
command -v php &>/dev/null && \
php -r "require_once '/usr/local/emhttp/plugins/varaverk/include/config.php'; vv_push_setup_state();" 2>/dev/null || true
}
echo ""
echo "━━━ $ICON_FALLBACK Partnership Onboard — $MY_ID ($LOCAL_SERVER_NAME) — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
echo ""
echo " Role: $( [[ "$AM_OWNER" == true ]] && echo "OWNER" || echo "MIRROR" )"
echo " This: $MY_ID ($LOCAL_SERVER_NAME)"
echo " Partner: $( [[ "$AM_OWNER" == true ]] && echo "$MIRROR_ID ($MIRROR)" || echo "$OWNER_ID ($OWNER)" )"
echo ""
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no permanent changes will be made"
# ==============================================================================================
# ── HELPER: stop containers on the mirror by reading its own conf via SSH ────────────────────
#
# SSHes to the mirror, sources its load_config.sh at the same $SCRIPTS_ROOT path (both servers
# use the same convention), and reads the named config array from the mirror's own conf.
# HOST2's container list stays in HOST2's host2.conf — not duplicated in HOST1's conf.
# Fails gracefully if scripts aren't present yet or the array is empty (nothing to stop).
#
# deploy_container_from_xml() already stops/removes containers with the same name as what's
# being deployed. This step handles containers with DIFFERENT names that conflict.
# ==============================================================================================
stop_mirror_stack() {
local config_var="$1" label="$2"
local -a to_stop=()
mapfile -t to_stop < <(
timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$MIRROR_IP" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${${config_var}[@]:-}\"" 2>/dev/null | grep -v '^$'
)
if [[ ${#to_stop[@]} -eq 0 ]]; then
log "No $label containers to stop on $MIRROR — skipping"
return 0
fi
log "Stopping $label on $MIRROR: ${to_stop[*]}"
for container in "${to_stop[@]}"; do
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would stop + rm $container on $MIRROR"
continue
fi
timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" root@"$MIRROR_IP" \
"docker stop '$container' 2>/dev/null
docker rm '$container' 2>/dev/null && echo removed" 2>/dev/null | \
grep -q removed && \
log " $container removed ✅" || \
log " $container not found on $MIRROR — skipping"
done
}
# ==============================================================================================
# ── MIRROR PATH ───────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
if [[ "$AM_MIRROR" == true ]]; then
echo "━━━ Step 1/2 — SSH Key Setup (Mirror) ━━━"
echo ""
echo " Mirror sets up SSH keys, then notifies Owner to run Phase 2."
echo ""
if [[ "$SKIP_SSH" == true ]]; then
warn "Skipping SSH setup (--skip-ssh)"
elif bash "$SCRIPT_DIR/ssh_setup.sh" "${EXTRA_FLAGS[@]}"; then
log "SSH key ready ✅"
else
error "SSH key setup failed"
exit 1
fi
echo ""
echo "━━━ Step 2/2 — Notify Owner to Run Phase 2 ━━━"
echo ""
OWNER_IP=$(resolve_tailscale_ip "$OWNER" 2>/dev/null || true)
PHASE2_TRIGGERED=false
if [[ -n "$OWNER_IP" ]]; then
# Read OWNER's SCRIPTS_DIR from their varaverk.cfg — don't assume same path as mirror
OWNER_SCRIPTS_DIR=$(timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$OWNER_IP" \
'grep SCRIPTS_DIR /boot/config/plugins/varaverk/varaverk.cfg 2>/dev/null | cut -d= -f2 | tr -d "\"'"'"'" 2>/dev/null' 2>/dev/null | tr -d '[:space:]')
OWNER_SCRIPTS_DIR="${OWNER_SCRIPTS_DIR:-/boot/config/plugins/varaverk}"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would SSH to $OWNER ($OWNER_IP) and trigger Phase 2"
PHASE2_TRIGGERED=true
elif timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$OWNER_IP" \
"nohup bash '${OWNER_SCRIPTS_DIR}/Partnership/partnership_onboard.sh' --phase2-only > /tmp/vv_phase2_onboard.log 2>&1 & echo triggered" \
2>/dev/null | grep -q triggered; then
log "Phase 2 triggered on $OWNER ✅"
log "Watch progress on $OWNER: tail -f /tmp/vv_phase2_onboard.log"
PHASE2_TRIGGERED=true
else
warn "Could not auto-trigger Phase 2 on $OWNER"
fi
else
warn "Cannot resolve $OWNER Tailscale IP"
fi
echo ""
echo "━━━━━ $ICON_SUMMARY MIRROR SETUP COMPLETE ━━━━━"
echo " SSH key: ready"
echo " Phase 2 on $OWNER: $( [[ "$PHASE2_TRIGGERED" == true ]] && echo "triggered ✅" || echo "needs manual trigger ⚠" )"
if [[ "$PHASE2_TRIGGERED" == false ]]; then
echo ""
echo " Run manually on $OWNER:"
echo " bash Partnership/partnership_onboard.sh --phase2-only"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ── OWNER PATH ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
MIRROR_IP=$(resolve_tailscale_ip "$MIRROR")
[[ -z "$MIRROR_IP" ]] && { error "Cannot resolve $MIRROR Tailscale IP — is Tailscale running?"; exit 1; }
log "Mirror: $MIRROR ($MIRROR_IP)"
[[ "$PHASE1_ONLY" == true ]] && log "Mode: Phase 1 only (SSH + conf push)"
[[ "$PHASE2_ONLY" == true ]] && log "Mode: Phase 2 only (containers + arr + onboard)"
echo ""
STEP_SSH_OK=false
STEP_STOP_AUTH_OK=true
STEP_AUTH_OK=true
AUTH_DEPLOYED=0
AUTH_FAILED=0
STEP_STOP_ARR_OK=true
STEP_ARR_OK=true
ARR_DEPLOYED=0
ARR_FAILED=0
STEP_STOP_SERVICES_OK=true
STEP_SERVICES_OK=true
SERVICES_DEPLOYED=0
SERVICES_FAILED=0
ONBOARD_OK=false
ARR_SYNC_OK=false
MASTER_PUSH_OK=false
# ── Step 1: SSH ───────────────────────────────────────────────────────────────────────────────
# Skipped when --phase2-only (SSH was already done in Phase 1).
echo "━━━ Step 1 — SSH Key Setup ━━━"
if [[ "$SKIP_SSH" == true ]]; then
warn "Skipping (--skip-ssh)"
STEP_SSH_OK=true
elif [[ "$PHASE1_ONLY" == true ]]; then
# Phase 1 in background: test if SSH already works first — avoids ssh-copy-id
# hanging for a password prompt with no TTY.
if timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$MIRROR_IP" exit 0 2>/dev/null; then
log "SSH to $MIRROR already works ✅ — skipping key install"
STEP_SSH_OK=true
else
# Key not yet on HOST2 — try ssh_setup.sh (works interactively, may fail in background)
if bash "$SCRIPT_DIR/ssh_setup.sh" "${EXTRA_FLAGS[@]}"; then
log "SSH keys ready ✅"
STEP_SSH_OK=true
else
# Soft-fail: generate key locally if not present, then tell user to install manually
warn "Could not install key on $MIRROR automatically (no terminal for password prompt)"
if [[ -f "$SSH_KEY" ]]; then
log "Local key exists at: $SSH_KEY"
else
bash "$SCRIPT_DIR/ssh_setup.sh" --key-only "${EXTRA_FLAGS[@]}" 2>/dev/null || true
fi
if [[ -f "${SSH_KEY}.pub" ]]; then
echo ""
echo " Install this key on $MIRROR to complete SSH setup:"
echo " ┌─────────────────────────────────────────────────────"
cat "${SSH_KEY}.pub" | sed 's/^/ │ /'
echo " └─────────────────────────────────────────────────────"
echo " Run on a terminal: ssh-copy-id -i ${SSH_KEY}.pub root@${MIRROR_IP}"
echo " Then click 'Push Conf' in the Partnership tab."
# Write key-ready flag so UI can show the manual-install state
[[ "$DRY_RUN" == false ]] && {
local kflag="${MIRROR_ID}_KEY_READY"
local _setup_f="${VARAVERK_SETUP_FILE:-${STATE_DIR:-/boot/config}/varaverk_setup.db}"
grep -q "^${kflag}=" "$_setup_f" 2>/dev/null \
&& sed -i "s|^${kflag}=.*|${kflag}=true|" "$_setup_f" \
|| echo "${kflag}=true" >> "$_setup_f"
}
fi
STEP_SSH_OK=false
fi
fi
elif bash "$SCRIPT_DIR/ssh_setup.sh" "${EXTRA_FLAGS[@]}"; then
log "SSH keys ready ✅"
STEP_SSH_OK=true
else
error "SSH key setup failed — aborting"
error "Re-run or use --skip-ssh if key is already set up"
exit 1
fi
# ── Phase 1 exit point ────────────────────────────────────────────────────────────────────────
# --phase1-only: SSH + conf push is all HOST1 needs to do before HOST2 installs Varaverk.
# HOST2's wizard will detect the pushed master.conf + state file and take the correct path.
if [[ "$PHASE1_ONLY" == true ]]; then
if [[ "$STEP_SSH_OK" == false ]]; then
# SSH key not yet installed on HOST2 — can't push conf, but local setup still runs.
# UI will show "key ready, install manually" state via HOST2_KEY_READY flag.
echo ""
echo "━━━ Phase 1 — HOST1 Local Setup (SSH pending) ━━━"
bash "$SCRIPT_DIR/partnership_manager.sh" --onboard --local-only "${EXTRA_FLAGS[@]}" || \
warn "Local setup had issues — check partnership_manager.sh output above"
END=$(date +%s)
echo ""
echo "━━━━━ $ICON_SUMMARY PHASE 1 — SSH PENDING ━━━━━"
echo " SSH keys: key generated ✅ — NOT yet installed on $MIRROR ⚠"
echo " Conf push: skipped (needs SSH access to $MIRROR)"
echo " HOST1 setup: done ✅"
echo " Duration: $(format_duration $(( END - START )))"
echo ""
echo " ACTION NEEDED: install the key on $MIRROR:"
echo " ssh-copy-id -i ${SSH_KEY}.pub root@${MIRROR_IP}"
echo " Then click 'Push Conf' in Partnership tab, or run:"
echo " bash Partnership/partnership_onboard.sh --phase1-only --skip-ssh"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
echo ""
echo "━━━ Phase 1 — Conf Push ━━━"
CONF_PUSH_OK=false
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would push master.conf + state file to $MIRROR"
CONF_PUSH_OK=true
elif ! command -v php &>/dev/null; then
warn "php not available — push master.conf manually via Scheduler → master.conf → Save Conf"
else
push_output=$(php -r "
require_once '/usr/local/emhttp/plugins/varaverk/include/config.php';
\$results = vv_push_master_conf();
vv_push_setup_state();
if (empty(\$results)) { echo 'no remote hosts'; exit(0); }
\$failed = 0;
foreach (\$results as \$r) {
echo \$r['host'] . ': ' . (\$r['ok'] ? 'pushed' : 'FAILED — ' . \$r['error']) . PHP_EOL;
if (!\$r['ok']) \$failed++;
}
exit(\$failed > 0 ? 1 : 0);
" 2>/dev/null)
push_rc=$?
echo "$push_output"
if [[ $push_rc -eq 0 ]]; then
log "Conf push complete ✅"
CONF_PUSH_OK=true
else
warn "Conf push had failures — retry via Scheduler → master.conf → Save Conf"
fi
fi
# HOST1 local setup — runs immediately without needing HOST2
echo ""
echo "━━━ Phase 1 — HOST1 Local Setup ━━━"
bash "$SCRIPT_DIR/partnership_manager.sh" --onboard --local-only "${EXTRA_FLAGS[@]}" || \
warn "Local setup had issues — check partnership_manager.sh output above"
[[ "$DRY_RUN" == false ]] && write_onboard_phase "$MIRROR_ID" 1
END=$(date +%s)
echo ""
echo "━━━━━ $ICON_SUMMARY PHASE 1 COMPLETE ━━━━━"
echo " SSH keys: $( [[ "$STEP_SSH_OK" == true ]] && echo "ready ✅" || echo "skipped" )"
echo " Conf push: $( [[ "$CONF_PUSH_OK" == true ]] && echo "done ✅" || echo "⚠ manual needed" )"
echo " HOST1 setup: done ✅"
echo " Duration: $(format_duration $(( END - START )))"
echo ""
echo " HOST1 is fully set up. HOST2 ($MIRROR) can now install the Varaverk plugin."
echo " The wizard will detect the pushed conf and take the correct path."
echo " When HOST2 completes its onboard, it will automatically trigger Phase 2 here."
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ── Step 2: Stop mirror's existing auth stack ─────────────────────────────────────────────────
echo ""
echo "━━━ Step 2 — Stop Mirror Auth Stack ━━━"
if [[ "$SKIP_AUTH_STACK" == true ]]; then
warn "Skipping (--skip-auth-stack)"
else
stop_mirror_stack "PARTNERSHIP_REPLACE_CONTAINERS" "auth stack"
fi
# ── Step 4: Deploy auth stack on mirror ───────────────────────────────────────────────────────
echo ""
echo "━━━ Step 3 — Deploy Auth Stack on Mirror ━━━"
if [[ "$SKIP_AUTH_STACK" == true ]]; then
warn "Skipping (--skip-auth-stack)"
elif [[ ${#PARTNERSHIP_AUTH_STACK[@]} -eq 0 ]]; then
warn "PARTNERSHIP_AUTH_STACK not set in ${MY_ID} conf — skipping auth stack deploy"
warn "Add HOST${MY_ID: -1}_PARTNERSHIP_AUTH_STACK to host${MY_ID: -1}.conf"
STEP_AUTH_OK=false
else
deploy_xml_stack PARTNERSHIP_AUTH_STACK
AUTH_DEPLOYED=$_STACK_DEPLOYED
AUTH_FAILED=$_STACK_FAILED
echo "Auth stack: $AUTH_DEPLOYED deployed, $AUTH_FAILED failed"
[[ "$AUTH_FAILED" -gt 0 ]] && STEP_AUTH_OK=false
fi
# ── Step 5: Stop mirror's existing arr stack ──────────────────────────────────────────────────
echo ""
echo "━━━ Step 4 — Stop Mirror Arr Stack ━━━"
if [[ "$SKIP_ARR_STACK" == true ]]; then
warn "Skipping (--skip-arr-stack)"
elif [[ ${#PARTNERSHIP_ARR_STACK[@]} -eq 0 ]]; then
log "PARTNERSHIP_ARR_STACK not configured — skipping arr stack deploy"
SKIP_ARR_STACK=true
else
stop_mirror_stack "PARTNERSHIP_ARR_REPLACE_CONTAINERS" "arr stack"
fi
# ── Step 5: Deploy arr stack on mirror ───────────────────────────────────────────────────────
echo ""
echo "━━━ Step 5 — Deploy Arr Stack on Mirror ━━━"
if [[ "$SKIP_ARR_STACK" == true ]]; then
warn "Skipping (--skip-arr-stack)"
else
deploy_xml_stack PARTNERSHIP_ARR_STACK
ARR_DEPLOYED=$_STACK_DEPLOYED
ARR_FAILED=$_STACK_FAILED
echo "Arr stack: $ARR_DEPLOYED deployed, $ARR_FAILED failed"
[[ "$ARR_FAILED" -gt 0 ]] && STEP_ARR_OK=false
fi
# ── Step 6: Stop mirror's existing services stack ─────────────────────────────────────────────
echo ""
echo "━━━ Step 6 — Stop Mirror Services Stack ━━━"
if [[ "$SKIP_SERVICES_STACK" == true ]]; then
warn "Skipping (--skip-services-stack)"
elif [[ ${#PARTNERSHIP_SERVICES_STACK[@]} -eq 0 ]]; then
log "PARTNERSHIP_SERVICES_STACK not configured — skipping services stack deploy"
SKIP_SERVICES_STACK=true
else
stop_mirror_stack "PARTNERSHIP_SERVICES_REPLACE_CONTAINERS" "services stack"
fi
# ── Step 7: Deploy services stack on mirror ───────────────────────────────────────────────────
echo ""
echo "━━━ Step 7 — Deploy Services Stack on Mirror ━━━"
if [[ "$SKIP_SERVICES_STACK" == true ]]; then
warn "Skipping (--skip-services-stack)"
else
deploy_xml_stack PARTNERSHIP_SERVICES_STACK
SERVICES_DEPLOYED=$_STACK_DEPLOYED
SERVICES_FAILED=$_STACK_FAILED
echo "Services stack: $SERVICES_DEPLOYED deployed, $SERVICES_FAILED failed"
[[ "$SERVICES_FAILED" -gt 0 ]] && STEP_SERVICES_OK=false
fi
# ── Step 8: Partnership onboard ───────────────────────────────────────────────────────────────
echo ""
echo "━━━ Step 8 — Partnership Onboard ━━━"
if bash "$SCRIPTS_ROOT/Partnership/partnership_manager.sh" --onboard "${EXTRA_FLAGS[@]}"; then
echo "Partnership onboard complete ✅"
ONBOARD_OK=true
else
error "Partnership onboard failed"
ONBOARD_OK=false
fi
# ── Step 9: Arr library bootstrap ─────────────────────────────────────────────────────────────
echo ""
echo "━━━ Step 9 — Arr Library Bootstrap ━━━"
if [[ "$ONBOARD_OK" == false ]]; then
warn "Skipping — onboard did not complete"
elif [[ "$SKIP_ARR_SYNC" == true ]]; then
warn "Skipping (--skip-arr-sync)"
elif [[ ! -f "$SCRIPTS_ROOT/Media/arr_sync.sh" ]]; then
warn "arr_sync.sh not found — run Media/arr_sync.sh manually once arrs are live"
elif bash "$SCRIPTS_ROOT/Media/arr_sync.sh" "${EXTRA_FLAGS[@]}"; then
echo "Arr bootstrap complete ✅"
ARR_SYNC_OK=true
else
warn "Arr sync had errors — partnership still valid"
warn "Re-run Media/arr_sync.sh once all arr containers are live"
fi
# ── Step 10: Push master.conf to all listed hosts ─────────────────────────────────────────────
# SSH is now established and all partners have the plugin installed.
# Push the authoritative master.conf so every listed host is in sync immediately.
echo ""
echo "━━━ $ICON_GEAR Step 10 — master.conf Push ━━━"
if [[ "$ONBOARD_OK" == false ]]; then
warn "Skipping — onboard did not complete"
elif [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would push master.conf to all listed hosts"
MASTER_PUSH_OK=true
elif ! command -v php &>/dev/null; then
warn "php not available — push master.conf manually via Scheduler → master.conf → Save Conf"
else
push_output=$(php -r "
require_once '/usr/local/emhttp/plugins/varaverk/include/config.php';
\$results = vv_push_master_conf();
vv_push_setup_state();
if (empty(\$results)) { echo 'no remote hosts'; exit(0); }
\$failed = 0;
foreach (\$results as \$r) {
echo \$r['host'] . ': ' . (\$r['ok'] ? 'pushed' : 'FAILED — ' . \$r['error']) . PHP_EOL;
if (!\$r['ok']) \$failed++;
}
exit(\$failed > 0 ? 1 : 0);
" 2>/dev/null)
push_rc=$?
echo "$push_output"
if [[ $push_rc -eq 0 ]]; then
echo "master.conf sync complete ✅"
MASTER_PUSH_OK=true
else
warn "master.conf push had failures — retry via Scheduler → master.conf → Save Conf"
fi
fi
# ── Write Phase 2 completion state ────────────────────────────────────────────────────────────
[[ "$ONBOARD_OK" == true && "$DRY_RUN" == false ]] && write_onboard_phase "$MIRROR_ID" 2
# ── Summary ───────────────────────────────────────────────────────────────────────────────────
END=$(date +%s)
echo ""
echo "━━━━━ $ICON_SUMMARY ONBOARD SUMMARY ━━━━━"
echo " Owner: $MY_ID ($LOCAL_SERVER_NAME)"
echo " Mirror: $MIRROR ($MIRROR_IP)"
[[ "$PHASE2_ONLY" == true ]] && echo " Mode: Phase 2 (triggered by HOST2 notification)"
echo " Duration: $(format_duration $(( END - START )))"
echo ""
_ok() { [[ "$1" == true ]] && echo "✅" || echo "❌"; }
_skip() { [[ "$1" == true ]] && echo "skipped" || echo "$(_ok "$2")"; }
echo " Step 1 — SSH keys: $(_skip "$SKIP_SSH" "$STEP_SSH_OK")"
echo " Step 2 — Stop auth: $(_skip "$SKIP_AUTH_STACK" "$STEP_STOP_AUTH_OK")"
echo " Step 3 — Auth stack: $( [[ "$SKIP_AUTH_STACK" == true ]] && echo "skipped" || echo "${AUTH_DEPLOYED} deployed, ${AUTH_FAILED} failed" )"
echo " Step 4 — Stop arr: $(_skip "$SKIP_ARR_STACK" "$STEP_STOP_ARR_OK")"
echo " Step 5 — Arr stack: $( [[ "$SKIP_ARR_STACK" == true ]] && echo "skipped" || echo "${ARR_DEPLOYED} deployed, ${ARR_FAILED} failed" )"
echo " Step 6 — Stop services: $(_skip "$SKIP_SERVICES_STACK" "$STEP_STOP_SERVICES_OK")"
echo " Step 7 — Services stack: $( [[ "$SKIP_SERVICES_STACK" == true ]] && echo "skipped" || echo "${SERVICES_DEPLOYED} deployed, ${SERVICES_FAILED} failed" )"
echo " Step 8 — Onboard: $(_ok "$ONBOARD_OK")"
echo " Step 9 — Arr bootstrap: $( [[ "$SKIP_ARR_SYNC" == true || "$ONBOARD_OK" == false ]] && echo "skipped" || echo "$(_ok "$ARR_SYNC_OK")" )"
echo " Step 10 — Conf push: $( [[ "$ONBOARD_OK" == false ]] && echo "skipped" || echo "$(_ok "$MASTER_PUSH_OK")" )"
echo ""
if [[ "$ONBOARD_OK" == true ]]; then
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes made" || \
echo "$ICON_DONE DONE — partnership established ✅"
echo "Verify with: Partnership/partnership_manager.sh --status"
else
error "Setup incomplete — resolve errors above and re-run"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━"
[[ "$ONBOARD_OK" == false ]] && exit 1
exit 0
@@ -1,675 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= Partnership Onboard ============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Runs once on both servers to establish a new partnership. Role is detected
# automatically via detect_hosts() — no flags needed to declare which side you are.
# Run on the mirror first (generates its SSH key), then on the owner to complete
# setup remotely.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# MIRROR PATH (1 step)
# Step 1: SSH key setup — generate keypair, copy to owner, update conf
# Owner completes the rest remotely. Mirror is done.
#
# OWNER PATH (10 steps)
# Step 1: SSH key setup — generate keypair, install on mirror, update conf
# Step 2: Stop mirror auth — stop mirror's existing auth containers before replacing
# Step 3: Deploy auth stack — push XMLs, pull images, create + start on mirror
# Mariadb/Redis health-checked before Authelia deploys
# Step 4: Stop mirror arr — stop mirror's existing arr containers before replacing
# Step 5: Deploy arr stack — push arr XMLs, pull images, create + start on mirror
# Step 6: Stop mirror services — stop mirror's existing services containers before replacing
# Step 7: Deploy services stack — push Emby/Jellyfin/Seerr XMLs, pull images, create + start
# Step 8: Partnership onboard — configure WebUIs → owner IP, write state, Emby
# Step 9: Arr bootstrap — bidirectional library sync (arr_sync.sh)
# Step 10: Conf push — push master.conf + setup state to all listed hosts
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Credentials never in SSH command strings
# Auth stack containers hold API keys, DB passwords, etc. The deploy script is written
# locally, SCPed to the remote, and executed there. Command-line args are never used
# to pass credentials — they'd appear in `ps` output and shell history on both servers.
#
# XML templates are the single source of truth for deployed containers
# The owner's templates-user/ XMLs define every container deployed on the mirror.
# The same XMLs that Unraid's Docker Manager uses are what get SCPed — the mirror's
# Docker Manager can manage the containers after onboard without additional config.
#
# Dependency ordering in the auth stack is owner-enforced
# PARTNERSHIP_AUTH_STACK order matters: Mariadb and Redis must come before Authelia.
# The array is ordered correctly in host1.conf. After each Mariadb/Redis deploy,
# the script waits for the container to be healthy before continuing. This is a remote
# health check — the container must be running (or report healthy) before the next
# dependent is deployed.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Root check
# All operations run as root — SSH key management, docker operations, conf updates.
#
# SSH timeout on all remote calls
# Every ssh/scp call uses SSH_TIMEOUT. No operation hangs indefinitely on a
# slow or unreachable mirror.
#
# --dry-run shows exact actions without executing
# Every step prints what it would do. SCP, deploy, plugin install, arr sync —
# all dry-run safe.
#
# Step skip flags for partial re-runs
# --skip-ssh, --skip-auth-stack, --skip-arr-stack, --skip-arr-sync allow
# resuming after a partial failure without re-running completed steps.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_PARTNERSHIP_AUTH_STACK
# XML filenames (from this server's templates-user/) to push and deploy on the
# mirror as its auth stack. Order matters: database deps before Authelia.
# Aliased by detect_hosts() → PARTNERSHIP_AUTH_STACK
#
# HOST*_PARTNERSHIP_REPLACE_CONTAINERS
# Containers to stop on the mirror before deploying the auth stack.
# Defined in the MIRROR's own conf (host*.conf on HOST2) — never in HOST1's conf.
# Read live from the mirror via SSH during Step 3 (sources mirror's load_config.sh at
# the same $SCRIPTS_ROOT path — convention: both servers use the same repo location).
# Leave empty on HOST2 if no conflicting containers exist (fresh mirror: nothing to stop).
# Aliased by detect_hosts() → PARTNERSHIP_REPLACE_CONTAINERS (on the mirror)
#
# HOST*_PARTNERSHIP_ARR_STACK
# XML filenames to push and deploy on the mirror as its arr stack.
# Leave empty to skip arr stack deploy.
# Aliased by detect_hosts() → PARTNERSHIP_ARR_STACK
#
# HOST*_PARTNERSHIP_ARR_REPLACE_CONTAINERS
# Arr containers to stop on the mirror before deploying the arr stack.
# Same rule as PARTNERSHIP_REPLACE_CONTAINERS: defined in mirror's own conf, never HOST1's.
# Aliased by detect_hosts() → PARTNERSHIP_ARR_REPLACE_CONTAINERS (on the mirror)
#
# HOST*_PARTNERSHIP_SERVICES_STACK
# XML filenames to push and deploy on the mirror as its shared services stack.
# Includes Emby, Jellyfin, Seerr, SeerrFin. Leave empty to skip services stack deploy.
# Aliased by detect_hosts() → PARTNERSHIP_SERVICES_STACK
#
# HOST*_PARTNERSHIP_SERVICES_REPLACE_CONTAINERS
# Services containers to stop on the mirror before deploying the services stack.
# Same rule as PARTNERSHIP_REPLACE_CONTAINERS: defined in mirror's own conf, never HOST1's.
# Aliased by detect_hosts() → PARTNERSHIP_SERVICES_REPLACE_CONTAINERS (on the mirror)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# Partnership/partnership_onboard.sh
# Full onboard — role detected automatically
#
# Partnership/partnership_onboard.sh --dry-run
# Preview all steps without making changes
#
# Partnership/partnership_onboard.sh --log
# Verbose per-step output
#
# Partnership/partnership_onboard.sh --skip-ssh
# Skip SSH key setup (key already in place)
#
# Partnership/partnership_onboard.sh --skip-auth-stack
# Skip auth stack stop + deploy (Steps 3-4)
#
# Partnership/partnership_onboard.sh --skip-arr-stack
# Skip arr stack stop + deploy (Steps 4-5)
#
# Partnership/partnership_onboard.sh --skip-services-stack
# Skip services stack stop + deploy (Steps 6-7)
#
# Partnership/partnership_onboard.sh --skip-arr-sync
# Skip arr library bootstrap (Step 9)
#
# Partnership/partnership_onboard.sh --phase1-only
# OWNER only: SSH key exchange + conf push. Safe to run before HOST2 has Varaverk.
# Writes HOST2_PHASE1_DONE=true to varaverk_setup.db.
#
# Partnership/partnership_onboard.sh --phase2-only
# OWNER only: container deploy + arr + onboard (skips SSH). Triggered automatically
# by HOST2 after it completes its Mirror-path onboard. Can also be run manually.
# Writes HOST2_PHASE2_DONE=true to varaverk_setup.db.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
SCRIPTS_ROOT="$SCRIPT_DIR/.."
SSH_TIMEOUT=15
source "$SCRIPTS_ROOT/load_config.sh"
source "$SCRIPTS_ROOT/Plugin/$PLATFORM/Partnership/containers.sh"
# ── Parse flags ───────────────────────────────────────────────────────────────────────────────
SKIP_SSH=false
SKIP_AUTH_STACK=false
SKIP_ARR_STACK=false
SKIP_SERVICES_STACK=false
SKIP_ARR_SYNC=false
PHASE1_ONLY=false # OWNER: SSH + conf push only (HOST2 not yet installed)
PHASE2_ONLY=false # OWNER: containers/arr/onboard only (triggered by HOST2 after it onboards)
FILTERED_ARGS=()
for arg in "$@"; do
case "$arg" in
--skip-ssh) SKIP_SSH=true ;;
--skip-auth-stack) SKIP_AUTH_STACK=true ;;
--skip-arr-stack) SKIP_ARR_STACK=true ;;
--skip-services-stack) SKIP_SERVICES_STACK=true ;;
--skip-arr-sync) SKIP_ARR_SYNC=true ;;
--phase1-only) PHASE1_ONLY=true ;;
--phase2-only) PHASE2_ONLY=true; SKIP_SSH=true ;;
*) FILTERED_ARGS+=("$arg") ;;
esac
done
parse_args "${FILTERED_ARGS[@]}"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
[[ "$EUID" -ne 0 ]] && { error "Must be run as root"; exit 1; }
acquire_lock
if ! command -v docker &>/dev/null; then
error "Docker command not found"
exit 1
fi
detect_hosts
OWNER_ID="${PARTNERSHIP_OWNER_HOST:-HOST1}"
MIRROR_ID=$( [[ "$OWNER_ID" == "HOST1" ]] && echo "HOST2" || echo "HOST1" )
OWNER="${!OWNER_ID}"
MIRROR="${!MIRROR_ID}"
# SSH_KEY (set by detect_hosts) is this server's own private key.
# The remote accepts it because this server's PUBLIC key was installed there via ssh_setup.sh.
# HOST{N}_SSH_KEY lives in host{N}.conf — with sparse checkout, the other server's
# conf is never present here. Always use SSH_KEY (local private key) for outbound SSH.
MIRROR_SSH_KEY="$SSH_KEY"
AM_OWNER=false
AM_MIRROR=false
[[ "$MY_ID" == "$OWNER_ID" ]] && AM_OWNER=true
[[ "$MY_ID" == "$MIRROR_ID" ]] && AM_MIRROR=true
EXTRA_FLAGS=()
[[ "$DRY_RUN" == true ]] && EXTRA_FLAGS+=("--dry-run")
[[ "$LOG_MODE" == true ]] && EXTRA_FLAGS+=("--log")
START=$(date +%s)
# ── Helper: write phase completion flag to setup.db + push to remotes ─────────────────────────
write_onboard_phase() {
local target_id="$1" phase="$2"
local key="${target_id}_PHASE${phase}_DONE"
local state_file="${VARAVERK_SETUP_FILE:-${STATE_DIR:-/boot/config}/varaverk_setup.db}"
[[ "$DRY_RUN" == true ]] && { warn "DRY RUN — would write ${key}=true"; return 0; }
if grep -q "^${key}=" "$state_file" 2>/dev/null; then
sed -i "s|^${key}=.*|${key}=true|" "$state_file"
else
echo "${key}=true" >> "$state_file"
fi
platform_push_setup_state
}
echo ""
echo "━━━ $ICON_FALLBACK Partnership Onboard — $MY_ID ($LOCAL_SERVER_NAME) — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
echo ""
echo " Role: $( [[ "$AM_OWNER" == true ]] && echo "OWNER" || echo "MIRROR" )"
echo " This: $MY_ID ($LOCAL_SERVER_NAME)"
echo " Partner: $( [[ "$AM_OWNER" == true ]] && echo "$MIRROR_ID ($MIRROR)" || echo "$OWNER_ID ($OWNER)" )"
echo ""
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no permanent changes will be made"
# ==============================================================================================
# ── HELPER: stop containers on the mirror by reading its own conf via SSH ────────────────────
#
# SSHes to the mirror, sources its load_config.sh at the same $SCRIPTS_ROOT path (both servers
# use the same convention), and reads the named config array from the mirror's own conf.
# HOST2's container list stays in HOST2's host2.conf — not duplicated in HOST1's conf.
# Fails gracefully if scripts aren't present yet or the array is empty (nothing to stop).
#
# deploy_container_from_xml() already stops/removes containers with the same name as what's
# being deployed. This step handles containers with DIFFERENT names that conflict.
# ==============================================================================================
stop_mirror_stack() {
local config_var="$1" label="$2"
local -a to_stop=()
mapfile -t to_stop < <(
timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$MIRROR_IP" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${${config_var}[@]:-}\"" 2>/dev/null | grep -v '^$'
)
if [[ ${#to_stop[@]} -eq 0 ]]; then
log "No $label containers to stop on $MIRROR — skipping"
return 0
fi
log "Stopping $label on $MIRROR: ${to_stop[*]}"
for container in "${to_stop[@]}"; do
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would stop + rm $container on $MIRROR"
continue
fi
timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" root@"$MIRROR_IP" \
"docker stop '$container' 2>/dev/null
docker rm '$container' 2>/dev/null && echo removed" 2>/dev/null | \
grep -q removed && \
log " $container removed ✅" || \
log " $container not found on $MIRROR — skipping"
done
}
# ==============================================================================================
# ── MIRROR PATH ───────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
if [[ "$AM_MIRROR" == true ]]; then
echo "━━━ Step 1/2 — SSH Key Setup (Mirror) ━━━"
echo ""
echo " Mirror sets up SSH keys, then notifies Owner to run Phase 2."
echo ""
if [[ "$SKIP_SSH" == true ]]; then
warn "Skipping SSH setup (--skip-ssh)"
elif bash "$SCRIPT_DIR/ssh_setup.sh" "${EXTRA_FLAGS[@]}"; then
log "SSH key ready ✅"
else
error "SSH key setup failed"
exit 1
fi
echo ""
echo "━━━ Step 2/2 — Notify Owner to Run Phase 2 ━━━"
echo ""
OWNER_IP=$(resolve_tailscale_ip "$OWNER" 2>/dev/null || true)
PHASE2_TRIGGERED=false
if [[ -n "$OWNER_IP" ]]; then
# Read OWNER's SCRIPTS_DIR from their varaverk.cfg — don't assume same path as mirror
OWNER_SCRIPTS_DIR=$(timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$OWNER_IP" \
'grep SCRIPTS_DIR /boot/config/plugins/varaverk/varaverk.cfg 2>/dev/null | cut -d= -f2 | tr -d "\"'"'"'" 2>/dev/null' 2>/dev/null | tr -d '[:space:]')
OWNER_SCRIPTS_DIR="${OWNER_SCRIPTS_DIR:-/boot/config/plugins/varaverk}"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would SSH to $OWNER ($OWNER_IP) and trigger Phase 2"
PHASE2_TRIGGERED=true
elif timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$OWNER_IP" \
"nohup bash '${OWNER_SCRIPTS_DIR}/Partnership/partnership_onboard.sh' --phase2-only > /tmp/vv_phase2_onboard.log 2>&1 & echo triggered" \
2>/dev/null | grep -q triggered; then
log "Phase 2 triggered on $OWNER ✅"
log "Watch progress on $OWNER: tail -f /tmp/vv_phase2_onboard.log"
PHASE2_TRIGGERED=true
else
warn "Could not auto-trigger Phase 2 on $OWNER"
fi
else
warn "Cannot resolve $OWNER Tailscale IP"
fi
echo ""
echo "━━━━━ $ICON_SUMMARY MIRROR SETUP COMPLETE ━━━━━"
echo " SSH key: ready"
echo " Phase 2 on $OWNER: $( [[ "$PHASE2_TRIGGERED" == true ]] && echo "triggered ✅" || echo "needs manual trigger ⚠" )"
if [[ "$PHASE2_TRIGGERED" == false ]]; then
echo ""
echo " Run manually on $OWNER:"
echo " bash Partnership/partnership_onboard.sh --phase2-only"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ── OWNER PATH ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
MIRROR_IP=$(resolve_tailscale_ip "$MIRROR")
[[ -z "$MIRROR_IP" ]] && { error "Cannot resolve $MIRROR Tailscale IP — is Tailscale running?"; exit 1; }
log "Mirror: $MIRROR ($MIRROR_IP)"
[[ "$PHASE1_ONLY" == true ]] && log "Mode: Phase 1 only (SSH + conf push)"
[[ "$PHASE2_ONLY" == true ]] && log "Mode: Phase 2 only (containers + arr + onboard)"
echo ""
STEP_SSH_OK=false
STEP_STOP_AUTH_OK=true
STEP_AUTH_OK=true
AUTH_DEPLOYED=0
AUTH_FAILED=0
STEP_STOP_ARR_OK=true
STEP_ARR_OK=true
ARR_DEPLOYED=0
ARR_FAILED=0
STEP_STOP_SERVICES_OK=true
STEP_SERVICES_OK=true
SERVICES_DEPLOYED=0
SERVICES_FAILED=0
ONBOARD_OK=false
ARR_SYNC_OK=false
MASTER_PUSH_OK=false
# ── Step 1: SSH ───────────────────────────────────────────────────────────────────────────────
# Skipped when --phase2-only (SSH was already done in Phase 1).
echo "━━━ Step 1 — SSH Key Setup ━━━"
if [[ "$SKIP_SSH" == true ]]; then
warn "Skipping (--skip-ssh)"
STEP_SSH_OK=true
elif [[ "$PHASE1_ONLY" == true ]]; then
# Phase 1 in background: test if SSH already works first — avoids ssh-copy-id
# hanging for a password prompt with no TTY.
if timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$MIRROR_IP" exit 0 2>/dev/null; then
log "SSH to $MIRROR already works ✅ — skipping key install"
STEP_SSH_OK=true
else
# Key not yet on HOST2 — try ssh_setup.sh (works interactively, may fail in background)
if bash "$SCRIPT_DIR/ssh_setup.sh" "${EXTRA_FLAGS[@]}"; then
log "SSH keys ready ✅"
STEP_SSH_OK=true
else
# Soft-fail: generate key locally if not present, then tell user to install manually
warn "Could not install key on $MIRROR automatically (no terminal for password prompt)"
if [[ -f "$SSH_KEY" ]]; then
log "Local key exists at: $SSH_KEY"
else
bash "$SCRIPT_DIR/ssh_setup.sh" --key-only "${EXTRA_FLAGS[@]}" 2>/dev/null || true
fi
if [[ -f "${SSH_KEY}.pub" ]]; then
echo ""
echo " Install this key on $MIRROR to complete SSH setup:"
echo " ┌─────────────────────────────────────────────────────"
cat "${SSH_KEY}.pub" | sed 's/^/ │ /'
echo " └─────────────────────────────────────────────────────"
echo " Run on a terminal: ssh-copy-id -i ${SSH_KEY}.pub root@${MIRROR_IP}"
echo " Then click 'Push Conf' in the Partnership tab."
# Write key-ready flag so UI can show the manual-install state
[[ "$DRY_RUN" == false ]] && {
local kflag="${MIRROR_ID}_KEY_READY"
local _setup_f="${VARAVERK_SETUP_FILE:-${STATE_DIR:-/boot/config}/varaverk_setup.db}"
grep -q "^${kflag}=" "$_setup_f" 2>/dev/null \
&& sed -i "s|^${kflag}=.*|${kflag}=true|" "$_setup_f" \
|| echo "${kflag}=true" >> "$_setup_f"
}
fi
STEP_SSH_OK=false
fi
fi
elif bash "$SCRIPT_DIR/ssh_setup.sh" "${EXTRA_FLAGS[@]}"; then
log "SSH keys ready ✅"
STEP_SSH_OK=true
else
error "SSH key setup failed — aborting"
error "Re-run or use --skip-ssh if key is already set up"
exit 1
fi
# ── Phase 1 exit point ────────────────────────────────────────────────────────────────────────
# --phase1-only: SSH + conf push is all HOST1 needs to do before HOST2 installs Varaverk.
# HOST2's wizard will detect the pushed master.conf + state file and take the correct path.
if [[ "$PHASE1_ONLY" == true ]]; then
if [[ "$STEP_SSH_OK" == false ]]; then
# SSH key not yet installed on HOST2 — can't push conf, but local setup still runs.
# UI will show "key ready, install manually" state via HOST2_KEY_READY flag.
echo ""
echo "━━━ Phase 1 — HOST1 Local Setup (SSH pending) ━━━"
bash "$SCRIPT_DIR/partnership_manager.sh" --onboard --local-only "${EXTRA_FLAGS[@]}" || \
warn "Local setup had issues — check partnership_manager.sh output above"
END=$(date +%s)
echo ""
echo "━━━━━ $ICON_SUMMARY PHASE 1 — SSH PENDING ━━━━━"
echo " SSH keys: key generated ✅ — NOT yet installed on $MIRROR ⚠"
echo " Conf push: skipped (needs SSH access to $MIRROR)"
echo " HOST1 setup: done ✅"
echo " Duration: $(format_duration $(( END - START )))"
echo ""
echo " ACTION NEEDED: install the key on $MIRROR:"
echo " ssh-copy-id -i ${SSH_KEY}.pub root@${MIRROR_IP}"
echo " Then click 'Push Conf' in Partnership tab, or run:"
echo " bash Partnership/partnership_onboard.sh --phase1-only --skip-ssh"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
echo ""
echo "━━━ Phase 1 — Conf Push ━━━"
CONF_PUSH_OK=false
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would push master.conf + state file to $MIRROR"
CONF_PUSH_OK=true
else
push_output=$(platform_push_conf)
push_rc=$?
[[ -n "$push_output" ]] && echo "$push_output"
platform_push_setup_state
if [[ $push_rc -eq 0 ]]; then
log "Conf push complete ✅"
CONF_PUSH_OK=true
else
warn "Conf push had failures — retry via Scheduler → master.conf → Save Conf"
fi
fi
# HOST1 local setup — runs immediately without needing HOST2
echo ""
echo "━━━ Phase 1 — HOST1 Local Setup ━━━"
bash "$SCRIPT_DIR/partnership_manager.sh" --onboard --local-only "${EXTRA_FLAGS[@]}" || \
warn "Local setup had issues — check partnership_manager.sh output above"
[[ "$DRY_RUN" == false ]] && write_onboard_phase "$MIRROR_ID" 1
END=$(date +%s)
echo ""
echo "━━━━━ $ICON_SUMMARY PHASE 1 COMPLETE ━━━━━"
echo " SSH keys: $( [[ "$STEP_SSH_OK" == true ]] && echo "ready ✅" || echo "skipped" )"
echo " Conf push: $( [[ "$CONF_PUSH_OK" == true ]] && echo "done ✅" || echo "⚠ manual needed" )"
echo " HOST1 setup: done ✅"
echo " Duration: $(format_duration $(( END - START )))"
echo ""
echo " HOST1 is fully set up. HOST2 ($MIRROR) can now install the Varaverk plugin."
echo " The wizard will detect the pushed conf and take the correct path."
echo " When HOST2 completes its onboard, it will automatically trigger Phase 2 here."
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ── Step 2: Stop mirror's existing auth stack ─────────────────────────────────────────────────
echo ""
echo "━━━ Step 2 — Stop Mirror Auth Stack ━━━"
if [[ "$SKIP_AUTH_STACK" == true ]]; then
warn "Skipping (--skip-auth-stack)"
else
stop_mirror_stack "PARTNERSHIP_REPLACE_CONTAINERS" "auth stack"
fi
# ── Step 4: Deploy auth stack on mirror ───────────────────────────────────────────────────────
echo ""
echo "━━━ Step 3 — Deploy Auth Stack on Mirror ━━━"
if [[ "$SKIP_AUTH_STACK" == true ]]; then
warn "Skipping (--skip-auth-stack)"
elif [[ ${#PARTNERSHIP_AUTH_STACK[@]} -eq 0 ]]; then
warn "PARTNERSHIP_AUTH_STACK not set in ${MY_ID} conf — skipping auth stack deploy"
warn "Add HOST${MY_ID: -1}_PARTNERSHIP_AUTH_STACK to host${MY_ID: -1}.conf"
STEP_AUTH_OK=false
else
deploy_xml_stack PARTNERSHIP_AUTH_STACK
AUTH_DEPLOYED=$_STACK_DEPLOYED
AUTH_FAILED=$_STACK_FAILED
echo "Auth stack: $AUTH_DEPLOYED deployed, $AUTH_FAILED failed"
[[ "$AUTH_FAILED" -gt 0 ]] && STEP_AUTH_OK=false
fi
# ── Step 5: Stop mirror's existing arr stack ──────────────────────────────────────────────────
echo ""
echo "━━━ Step 4 — Stop Mirror Arr Stack ━━━"
if [[ "$SKIP_ARR_STACK" == true ]]; then
warn "Skipping (--skip-arr-stack)"
elif [[ ${#PARTNERSHIP_ARR_STACK[@]} -eq 0 ]]; then
log "PARTNERSHIP_ARR_STACK not configured — skipping arr stack deploy"
SKIP_ARR_STACK=true
else
stop_mirror_stack "PARTNERSHIP_ARR_REPLACE_CONTAINERS" "arr stack"
fi
# ── Step 5: Deploy arr stack on mirror ───────────────────────────────────────────────────────
echo ""
echo "━━━ Step 5 — Deploy Arr Stack on Mirror ━━━"
if [[ "$SKIP_ARR_STACK" == true ]]; then
warn "Skipping (--skip-arr-stack)"
else
deploy_xml_stack PARTNERSHIP_ARR_STACK
ARR_DEPLOYED=$_STACK_DEPLOYED
ARR_FAILED=$_STACK_FAILED
echo "Arr stack: $ARR_DEPLOYED deployed, $ARR_FAILED failed"
[[ "$ARR_FAILED" -gt 0 ]] && STEP_ARR_OK=false
fi
# ── Step 6: Stop mirror's existing services stack ─────────────────────────────────────────────
echo ""
echo "━━━ Step 6 — Stop Mirror Services Stack ━━━"
if [[ "$SKIP_SERVICES_STACK" == true ]]; then
warn "Skipping (--skip-services-stack)"
elif [[ ${#PARTNERSHIP_SERVICES_STACK[@]} -eq 0 ]]; then
log "PARTNERSHIP_SERVICES_STACK not configured — skipping services stack deploy"
SKIP_SERVICES_STACK=true
else
stop_mirror_stack "PARTNERSHIP_SERVICES_REPLACE_CONTAINERS" "services stack"
fi
# ── Step 7: Deploy services stack on mirror ───────────────────────────────────────────────────
echo ""
echo "━━━ Step 7 — Deploy Services Stack on Mirror ━━━"
if [[ "$SKIP_SERVICES_STACK" == true ]]; then
warn "Skipping (--skip-services-stack)"
else
deploy_xml_stack PARTNERSHIP_SERVICES_STACK
SERVICES_DEPLOYED=$_STACK_DEPLOYED
SERVICES_FAILED=$_STACK_FAILED
echo "Services stack: $SERVICES_DEPLOYED deployed, $SERVICES_FAILED failed"
[[ "$SERVICES_FAILED" -gt 0 ]] && STEP_SERVICES_OK=false
fi
# ── Step 8: Partnership onboard ───────────────────────────────────────────────────────────────
echo ""
echo "━━━ Step 8 — Partnership Onboard ━━━"
if bash "$SCRIPTS_ROOT/Partnership/partnership_manager.sh" --onboard "${EXTRA_FLAGS[@]}"; then
echo "Partnership onboard complete ✅"
ONBOARD_OK=true
else
error "Partnership onboard failed"
ONBOARD_OK=false
fi
# ── Step 9: Arr library bootstrap ─────────────────────────────────────────────────────────────
echo ""
echo "━━━ Step 9 — Arr Library Bootstrap ━━━"
if [[ "$ONBOARD_OK" == false ]]; then
warn "Skipping — onboard did not complete"
elif [[ "$SKIP_ARR_SYNC" == true ]]; then
warn "Skipping (--skip-arr-sync)"
elif [[ ! -f "$SCRIPTS_ROOT/Media/arr_sync.sh" ]]; then
warn "arr_sync.sh not found — run Media/arr_sync.sh manually once arrs are live"
elif bash "$SCRIPTS_ROOT/Media/arr_sync.sh" "${EXTRA_FLAGS[@]}"; then
echo "Arr bootstrap complete ✅"
ARR_SYNC_OK=true
else
warn "Arr sync had errors — partnership still valid"
warn "Re-run Media/arr_sync.sh once all arr containers are live"
fi
# ── Step 10: Push master.conf to all listed hosts ─────────────────────────────────────────────
# SSH is now established and all partners have the plugin installed.
# Push the authoritative master.conf so every listed host is in sync immediately.
echo ""
echo "━━━ $ICON_GEAR Step 10 — master.conf Push ━━━"
if [[ "$ONBOARD_OK" == false ]]; then
warn "Skipping — onboard did not complete"
elif [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would push master.conf to all listed hosts"
MASTER_PUSH_OK=true
else
push_output=$(platform_push_conf)
push_rc=$?
[[ -n "$push_output" ]] && echo "$push_output"
platform_push_setup_state
if [[ $push_rc -eq 0 ]]; then
echo "master.conf sync complete ✅"
MASTER_PUSH_OK=true
else
warn "master.conf push had failures — retry via Scheduler → master.conf → Save Conf"
fi
fi
# ── Write Phase 2 completion state ────────────────────────────────────────────────────────────
[[ "$ONBOARD_OK" == true && "$DRY_RUN" == false ]] && write_onboard_phase "$MIRROR_ID" 2
# ── Summary ───────────────────────────────────────────────────────────────────────────────────
END=$(date +%s)
echo ""
echo "━━━━━ $ICON_SUMMARY ONBOARD SUMMARY ━━━━━"
echo " Owner: $MY_ID ($LOCAL_SERVER_NAME)"
echo " Mirror: $MIRROR ($MIRROR_IP)"
[[ "$PHASE2_ONLY" == true ]] && echo " Mode: Phase 2 (triggered by HOST2 notification)"
echo " Duration: $(format_duration $(( END - START )))"
echo ""
_ok() { [[ "$1" == true ]] && echo "✅" || echo "❌"; }
_skip() { [[ "$1" == true ]] && echo "skipped" || echo "$(_ok "$2")"; }
echo " Step 1 — SSH keys: $(_skip "$SKIP_SSH" "$STEP_SSH_OK")"
echo " Step 2 — Stop auth: $(_skip "$SKIP_AUTH_STACK" "$STEP_STOP_AUTH_OK")"
echo " Step 3 — Auth stack: $( [[ "$SKIP_AUTH_STACK" == true ]] && echo "skipped" || echo "${AUTH_DEPLOYED} deployed, ${AUTH_FAILED} failed" )"
echo " Step 4 — Stop arr: $(_skip "$SKIP_ARR_STACK" "$STEP_STOP_ARR_OK")"
echo " Step 5 — Arr stack: $( [[ "$SKIP_ARR_STACK" == true ]] && echo "skipped" || echo "${ARR_DEPLOYED} deployed, ${ARR_FAILED} failed" )"
echo " Step 6 — Stop services: $(_skip "$SKIP_SERVICES_STACK" "$STEP_STOP_SERVICES_OK")"
echo " Step 7 — Services stack: $( [[ "$SKIP_SERVICES_STACK" == true ]] && echo "skipped" || echo "${SERVICES_DEPLOYED} deployed, ${SERVICES_FAILED} failed" )"
echo " Step 8 — Onboard: $(_ok "$ONBOARD_OK")"
echo " Step 9 — Arr bootstrap: $( [[ "$SKIP_ARR_SYNC" == true || "$ONBOARD_OK" == false ]] && echo "skipped" || echo "$(_ok "$ARR_SYNC_OK")" )"
echo " Step 10 — Conf push: $( [[ "$ONBOARD_OK" == false ]] && echo "skipped" || echo "$(_ok "$MASTER_PUSH_OK")" )"
echo ""
if [[ "$ONBOARD_OK" == true ]]; then
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes made" || \
echo "$ICON_DONE DONE — partnership established ✅"
echo "Verify with: Partnership/partnership_manager.sh --status"
else
error "Setup incomplete — resolve errors above and re-run"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━"
[[ "$ONBOARD_OK" == false ]] && exit 1
exit 0
@@ -1,676 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= Partnership Onboard ============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Runs once on both servers to establish a new partnership. Role is detected
# automatically via detect_hosts() — no flags needed to declare which side you are.
# Run on the mirror first (generates its SSH key), then on the owner to complete
# setup remotely.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# MIRROR PATH (1 step)
# Step 1: SSH key setup — generate keypair, copy to owner, update conf
# Owner completes the rest remotely. Mirror is done.
#
# OWNER PATH (10 steps)
# Step 1: SSH key setup — generate keypair, install on mirror, update conf
# Step 2: Stop mirror auth — stop mirror's existing auth containers before replacing
# Step 3: Deploy auth stack — push XMLs, pull images, create + start on mirror
# Mariadb/Redis health-checked before Authelia deploys
# Step 4: Stop mirror arr — stop mirror's existing arr containers before replacing
# Step 5: Deploy arr stack — push arr XMLs, pull images, create + start on mirror
# Step 6: Stop mirror services — stop mirror's existing services containers before replacing
# Step 7: Deploy services stack — push Emby/Jellyfin/Seerr XMLs, pull images, create + start
# Step 8: Partnership onboard — configure WebUIs → owner IP, write state, Emby
# Step 9: Arr bootstrap — bidirectional library sync (arr_sync.sh)
# Step 10: Conf push — push master.conf + setup state to all listed hosts
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Credentials never in SSH command strings
# Auth stack containers hold API keys, DB passwords, etc. The deploy script is written
# locally, SCPed to the remote, and executed there. Command-line args are never used
# to pass credentials — they'd appear in `ps` output and shell history on both servers.
#
# XML templates are the single source of truth for deployed containers
# The owner's templates-user/ XMLs define every container deployed on the mirror.
# The same XMLs that Unraid's Docker Manager uses are what get SCPed — the mirror's
# Docker Manager can manage the containers after onboard without additional config.
#
# Dependency ordering in the auth stack is owner-enforced
# PARTNERSHIP_AUTH_STACK order matters: Mariadb and Redis must come before Authelia.
# The array is ordered correctly in host1.conf. After each Mariadb/Redis deploy,
# the script waits for the container to be healthy before continuing. This is a remote
# health check — the container must be running (or report healthy) before the next
# dependent is deployed.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Root check
# All operations run as root — SSH key management, docker operations, conf updates.
#
# SSH timeout on all remote calls
# Every ssh/scp call uses SSH_TIMEOUT. No operation hangs indefinitely on a
# slow or unreachable mirror.
#
# --dry-run shows exact actions without executing
# Every step prints what it would do. SCP, deploy, plugin install, arr sync —
# all dry-run safe.
#
# Step skip flags for partial re-runs
# --skip-ssh, --skip-auth-stack, --skip-arr-stack, --skip-arr-sync allow
# resuming after a partial failure without re-running completed steps.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_PARTNERSHIP_AUTH_STACK
# XML filenames (from this server's templates-user/) to push and deploy on the
# mirror as its auth stack. Order matters: database deps before Authelia.
# Aliased by detect_hosts() → PARTNERSHIP_AUTH_STACK
#
# HOST*_PARTNERSHIP_REPLACE_CONTAINERS
# Containers to stop on the mirror before deploying the auth stack.
# Defined in the MIRROR's own conf (host*.conf on HOST2) — never in HOST1's conf.
# Read live from the mirror via SSH during Step 3 (sources mirror's load_config.sh at
# the same $SCRIPTS_ROOT path — convention: both servers use the same repo location).
# Leave empty on HOST2 if no conflicting containers exist (fresh mirror: nothing to stop).
# Aliased by detect_hosts() → PARTNERSHIP_REPLACE_CONTAINERS (on the mirror)
#
# HOST*_PARTNERSHIP_ARR_STACK
# XML filenames to push and deploy on the mirror as its arr stack.
# Leave empty to skip arr stack deploy.
# Aliased by detect_hosts() → PARTNERSHIP_ARR_STACK
#
# HOST*_PARTNERSHIP_ARR_REPLACE_CONTAINERS
# Arr containers to stop on the mirror before deploying the arr stack.
# Same rule as PARTNERSHIP_REPLACE_CONTAINERS: defined in mirror's own conf, never HOST1's.
# Aliased by detect_hosts() → PARTNERSHIP_ARR_REPLACE_CONTAINERS (on the mirror)
#
# HOST*_PARTNERSHIP_SERVICES_STACK
# XML filenames to push and deploy on the mirror as its shared services stack.
# Includes Emby, Jellyfin, Seerr, SeerrFin. Leave empty to skip services stack deploy.
# Aliased by detect_hosts() → PARTNERSHIP_SERVICES_STACK
#
# HOST*_PARTNERSHIP_SERVICES_REPLACE_CONTAINERS
# Services containers to stop on the mirror before deploying the services stack.
# Same rule as PARTNERSHIP_REPLACE_CONTAINERS: defined in mirror's own conf, never HOST1's.
# Aliased by detect_hosts() → PARTNERSHIP_SERVICES_REPLACE_CONTAINERS (on the mirror)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# Partnership/partnership_onboard.sh
# Full onboard — role detected automatically
#
# Partnership/partnership_onboard.sh --dry-run
# Preview all steps without making changes
#
# Partnership/partnership_onboard.sh --log
# Verbose per-step output
#
# Partnership/partnership_onboard.sh --skip-ssh
# Skip SSH key setup (key already in place)
#
# Partnership/partnership_onboard.sh --skip-auth-stack
# Skip auth stack stop + deploy (Steps 3-4)
#
# Partnership/partnership_onboard.sh --skip-arr-stack
# Skip arr stack stop + deploy (Steps 4-5)
#
# Partnership/partnership_onboard.sh --skip-services-stack
# Skip services stack stop + deploy (Steps 6-7)
#
# Partnership/partnership_onboard.sh --skip-arr-sync
# Skip arr library bootstrap (Step 9)
#
# Partnership/partnership_onboard.sh --phase1-only
# OWNER only: SSH key exchange + conf push. Safe to run before HOST2 has Varaverk.
# Writes HOST2_PHASE1_DONE=true to varaverk_setup.db.
#
# Partnership/partnership_onboard.sh --phase2-only
# OWNER only: container deploy + arr + onboard (skips SSH). Triggered automatically
# by HOST2 after it completes its Mirror-path onboard. Can also be run manually.
# Writes HOST2_PHASE2_DONE=true to varaverk_setup.db.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
SCRIPTS_ROOT="$SCRIPT_DIR/.."
SSH_TIMEOUT=15
source "$SCRIPTS_ROOT/load_config.sh"
source "$SCRIPTS_ROOT/Plugin/$PLATFORM/Partnership/containers.sh"
# ── Parse flags ───────────────────────────────────────────────────────────────────────────────
SKIP_SSH=false
SKIP_AUTH_STACK=false
SKIP_ARR_STACK=false
SKIP_SERVICES_STACK=false
SKIP_ARR_SYNC=false
PHASE1_ONLY=false # OWNER: SSH + conf push only (HOST2 not yet installed)
PHASE2_ONLY=false # OWNER: containers/arr/onboard only (triggered by HOST2 after it onboards)
FILTERED_ARGS=()
for arg in "$@"; do
case "$arg" in
--skip-ssh) SKIP_SSH=true ;;
--skip-auth-stack) SKIP_AUTH_STACK=true ;;
--skip-arr-stack) SKIP_ARR_STACK=true ;;
--skip-services-stack) SKIP_SERVICES_STACK=true ;;
--skip-arr-sync) SKIP_ARR_SYNC=true ;;
--phase1-only) PHASE1_ONLY=true ;;
--phase2-only) PHASE2_ONLY=true; SKIP_SSH=true ;;
*) FILTERED_ARGS+=("$arg") ;;
esac
done
parse_args "${FILTERED_ARGS[@]}"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
[[ "$EUID" -ne 0 ]] && { error "Must be run as root"; exit 1; }
acquire_lock
if ! command -v docker &>/dev/null; then
error "Docker command not found"
exit 1
fi
detect_hosts
OWNER_ID="${PARTNERSHIP_OWNER_HOST:-HOST1}"
MIRROR_ID=$( [[ "$OWNER_ID" == "HOST1" ]] && echo "HOST2" || echo "HOST1" )
OWNER="${!OWNER_ID}"
MIRROR="${!MIRROR_ID}"
# SSH_KEY (set by detect_hosts) is this server's own private key.
# The remote accepts it because this server's PUBLIC key was installed there via ssh_setup.sh.
# HOST{N}_SSH_KEY lives in host{N}.conf — with sparse checkout, the other server's
# conf is never present here. Always use SSH_KEY (local private key) for outbound SSH.
MIRROR_SSH_KEY="$SSH_KEY"
AM_OWNER=false
AM_MIRROR=false
[[ "$MY_ID" == "$OWNER_ID" ]] && AM_OWNER=true
[[ "$MY_ID" == "$MIRROR_ID" ]] && AM_MIRROR=true
EXTRA_FLAGS=()
[[ "$DRY_RUN" == true ]] && EXTRA_FLAGS+=("--dry-run")
[[ "$LOG_MODE" == true ]] && EXTRA_FLAGS+=("--log")
START=$(date +%s)
# ── Helper: write phase completion flag to setup.db + push to remotes ─────────────────────────
write_onboard_phase() {
local target_id="$1" phase="$2"
local key="${target_id}_PHASE${phase}_DONE"
local state_file="${VARAVERK_SETUP_FILE:-${STATE_DIR:-/boot/config}/varaverk_setup.db}"
[[ "$DRY_RUN" == true ]] && { warn "DRY RUN — would write ${key}=true"; return 0; }
if grep -q "^${key}=" "$state_file" 2>/dev/null; then
sed -i "s|^${key}=.*|${key}=true|" "$state_file"
else
echo "${key}=true" >> "$state_file"
fi
platform_push_setup_state
}
echo ""
echo "━━━ $ICON_FALLBACK Partnership Onboard — $MY_ID ($LOCAL_SERVER_NAME) — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
echo ""
echo " Role: $( [[ "$AM_OWNER" == true ]] && echo "OWNER" || echo "MIRROR" )"
echo " This: $MY_ID ($LOCAL_SERVER_NAME)"
echo " Partner: $( [[ "$AM_OWNER" == true ]] && echo "$MIRROR_ID ($MIRROR)" || echo "$OWNER_ID ($OWNER)" )"
echo ""
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no permanent changes will be made"
# ==============================================================================================
# ── HELPER: stop containers on the mirror by reading its own conf via SSH ────────────────────
#
# SSHes to the mirror, sources its load_config.sh at the same $SCRIPTS_ROOT path (both servers
# use the same convention), and reads the named config array from the mirror's own conf.
# HOST2's container list stays in HOST2's host2.conf — not duplicated in HOST1's conf.
# Fails gracefully if scripts aren't present yet or the array is empty (nothing to stop).
#
# deploy_container_from_xml() already stops/removes containers with the same name as what's
# being deployed. This step handles containers with DIFFERENT names that conflict.
# ==============================================================================================
stop_mirror_stack() {
local config_var="$1" label="$2"
local -a to_stop=()
mapfile -t to_stop < <(
timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$MIRROR_IP" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${${config_var}[@]:-}\"" 2>/dev/null | grep -v '^$'
)
if [[ ${#to_stop[@]} -eq 0 ]]; then
log "No $label containers to stop on $MIRROR — skipping"
return 0
fi
log "Stopping $label on $MIRROR: ${to_stop[*]}"
for container in "${to_stop[@]}"; do
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would stop + rm $container on $MIRROR"
continue
fi
timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" root@"$MIRROR_IP" \
"docker stop '$container' 2>/dev/null
docker rm '$container' 2>/dev/null && echo removed" 2>/dev/null | \
grep -q removed && \
log " $container removed ✅" || \
log " $container not found on $MIRROR — skipping"
done
}
# ==============================================================================================
# ── MIRROR PATH ───────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
if [[ "$AM_MIRROR" == true ]]; then
echo "━━━ Step 1/2 — SSH Key Setup (Mirror) ━━━"
echo ""
echo " Mirror sets up SSH keys, then notifies Owner to run Phase 2."
echo ""
if [[ "$SKIP_SSH" == true ]]; then
warn "Skipping SSH setup (--skip-ssh)"
elif bash "$SCRIPT_DIR/ssh_setup.sh" "${EXTRA_FLAGS[@]}"; then
log "SSH key ready ✅"
else
error "SSH key setup failed"
exit 1
fi
echo ""
echo "━━━ Step 2/2 — Notify Owner to Run Phase 2 ━━━"
echo ""
OWNER_IP=$(resolve_tailscale_ip "$OWNER" 2>/dev/null || true)
PHASE2_TRIGGERED=false
if [[ -n "$OWNER_IP" ]]; then
# Read OWNER's SCRIPTS_DIR via platform probe command — don't assume same path as mirror
_probe_cmd=$(platform_scripts_dir_probe_cmd)
OWNER_SCRIPTS_DIR=$(timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$OWNER_IP" \
"$_probe_cmd" 2>/dev/null | tr -d '[:space:]')
OWNER_SCRIPTS_DIR="${OWNER_SCRIPTS_DIR:-$SCRIPTS_DIR}"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would SSH to $OWNER ($OWNER_IP) and trigger Phase 2"
PHASE2_TRIGGERED=true
elif timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$OWNER_IP" \
"nohup bash '${OWNER_SCRIPTS_DIR}/Partnership/partnership_onboard.sh' --phase2-only > /tmp/vv_phase2_onboard.log 2>&1 & echo triggered" \
2>/dev/null | grep -q triggered; then
log "Phase 2 triggered on $OWNER ✅"
log "Watch progress on $OWNER: tail -f /tmp/vv_phase2_onboard.log"
PHASE2_TRIGGERED=true
else
warn "Could not auto-trigger Phase 2 on $OWNER"
fi
else
warn "Cannot resolve $OWNER Tailscale IP"
fi
echo ""
echo "━━━━━ $ICON_SUMMARY MIRROR SETUP COMPLETE ━━━━━"
echo " SSH key: ready"
echo " Phase 2 on $OWNER: $( [[ "$PHASE2_TRIGGERED" == true ]] && echo "triggered ✅" || echo "needs manual trigger ⚠" )"
if [[ "$PHASE2_TRIGGERED" == false ]]; then
echo ""
echo " Run manually on $OWNER:"
echo " bash Partnership/partnership_onboard.sh --phase2-only"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ── OWNER PATH ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
MIRROR_IP=$(resolve_tailscale_ip "$MIRROR")
[[ -z "$MIRROR_IP" ]] && { error "Cannot resolve $MIRROR Tailscale IP — is Tailscale running?"; exit 1; }
log "Mirror: $MIRROR ($MIRROR_IP)"
[[ "$PHASE1_ONLY" == true ]] && log "Mode: Phase 1 only (SSH + conf push)"
[[ "$PHASE2_ONLY" == true ]] && log "Mode: Phase 2 only (containers + arr + onboard)"
echo ""
STEP_SSH_OK=false
STEP_STOP_AUTH_OK=true
STEP_AUTH_OK=true
AUTH_DEPLOYED=0
AUTH_FAILED=0
STEP_STOP_ARR_OK=true
STEP_ARR_OK=true
ARR_DEPLOYED=0
ARR_FAILED=0
STEP_STOP_SERVICES_OK=true
STEP_SERVICES_OK=true
SERVICES_DEPLOYED=0
SERVICES_FAILED=0
ONBOARD_OK=false
ARR_SYNC_OK=false
MASTER_PUSH_OK=false
# ── Step 1: SSH ───────────────────────────────────────────────────────────────────────────────
# Skipped when --phase2-only (SSH was already done in Phase 1).
echo "━━━ Step 1 — SSH Key Setup ━━━"
if [[ "$SKIP_SSH" == true ]]; then
warn "Skipping (--skip-ssh)"
STEP_SSH_OK=true
elif [[ "$PHASE1_ONLY" == true ]]; then
# Phase 1 in background: test if SSH already works first — avoids ssh-copy-id
# hanging for a password prompt with no TTY.
if timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$MIRROR_IP" exit 0 2>/dev/null; then
log "SSH to $MIRROR already works ✅ — skipping key install"
STEP_SSH_OK=true
else
# Key not yet on HOST2 — try ssh_setup.sh (works interactively, may fail in background)
if bash "$SCRIPT_DIR/ssh_setup.sh" "${EXTRA_FLAGS[@]}"; then
log "SSH keys ready ✅"
STEP_SSH_OK=true
else
# Soft-fail: generate key locally if not present, then tell user to install manually
warn "Could not install key on $MIRROR automatically (no terminal for password prompt)"
if [[ -f "$SSH_KEY" ]]; then
log "Local key exists at: $SSH_KEY"
else
bash "$SCRIPT_DIR/ssh_setup.sh" --key-only "${EXTRA_FLAGS[@]}" 2>/dev/null || true
fi
if [[ -f "${SSH_KEY}.pub" ]]; then
echo ""
echo " Install this key on $MIRROR to complete SSH setup:"
echo " ┌─────────────────────────────────────────────────────"
cat "${SSH_KEY}.pub" | sed 's/^/ │ /'
echo " └─────────────────────────────────────────────────────"
echo " Run on a terminal: ssh-copy-id -i ${SSH_KEY}.pub root@${MIRROR_IP}"
echo " Then click 'Push Conf' in the Partnership tab."
# Write key-ready flag so UI can show the manual-install state
[[ "$DRY_RUN" == false ]] && {
local kflag="${MIRROR_ID}_KEY_READY"
local _setup_f="${VARAVERK_SETUP_FILE:-${STATE_DIR:-/boot/config}/varaverk_setup.db}"
grep -q "^${kflag}=" "$_setup_f" 2>/dev/null \
&& sed -i "s|^${kflag}=.*|${kflag}=true|" "$_setup_f" \
|| echo "${kflag}=true" >> "$_setup_f"
}
fi
STEP_SSH_OK=false
fi
fi
elif bash "$SCRIPT_DIR/ssh_setup.sh" "${EXTRA_FLAGS[@]}"; then
log "SSH keys ready ✅"
STEP_SSH_OK=true
else
error "SSH key setup failed — aborting"
error "Re-run or use --skip-ssh if key is already set up"
exit 1
fi
# ── Phase 1 exit point ────────────────────────────────────────────────────────────────────────
# --phase1-only: SSH + conf push is all HOST1 needs to do before HOST2 installs Varaverk.
# HOST2's wizard will detect the pushed master.conf + state file and take the correct path.
if [[ "$PHASE1_ONLY" == true ]]; then
if [[ "$STEP_SSH_OK" == false ]]; then
# SSH key not yet installed on HOST2 — can't push conf, but local setup still runs.
# UI will show "key ready, install manually" state via HOST2_KEY_READY flag.
echo ""
echo "━━━ Phase 1 — HOST1 Local Setup (SSH pending) ━━━"
bash "$SCRIPT_DIR/partnership_manager.sh" --onboard --local-only "${EXTRA_FLAGS[@]}" || \
warn "Local setup had issues — check partnership_manager.sh output above"
END=$(date +%s)
echo ""
echo "━━━━━ $ICON_SUMMARY PHASE 1 — SSH PENDING ━━━━━"
echo " SSH keys: key generated ✅ — NOT yet installed on $MIRROR ⚠"
echo " Conf push: skipped (needs SSH access to $MIRROR)"
echo " HOST1 setup: done ✅"
echo " Duration: $(format_duration $(( END - START )))"
echo ""
echo " ACTION NEEDED: install the key on $MIRROR:"
echo " ssh-copy-id -i ${SSH_KEY}.pub root@${MIRROR_IP}"
echo " Then click 'Push Conf' in Partnership tab, or run:"
echo " bash Partnership/partnership_onboard.sh --phase1-only --skip-ssh"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
echo ""
echo "━━━ Phase 1 — Conf Push ━━━"
CONF_PUSH_OK=false
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would push master.conf + state file to $MIRROR"
CONF_PUSH_OK=true
else
push_output=$(platform_push_conf)
push_rc=$?
[[ -n "$push_output" ]] && echo "$push_output"
platform_push_setup_state
if [[ $push_rc -eq 0 ]]; then
log "Conf push complete ✅"
CONF_PUSH_OK=true
else
warn "Conf push had failures — retry via Scheduler → master.conf → Save Conf"
fi
fi
# HOST1 local setup — runs immediately without needing HOST2
echo ""
echo "━━━ Phase 1 — HOST1 Local Setup ━━━"
bash "$SCRIPT_DIR/partnership_manager.sh" --onboard --local-only "${EXTRA_FLAGS[@]}" || \
warn "Local setup had issues — check partnership_manager.sh output above"
[[ "$DRY_RUN" == false ]] && write_onboard_phase "$MIRROR_ID" 1
END=$(date +%s)
echo ""
echo "━━━━━ $ICON_SUMMARY PHASE 1 COMPLETE ━━━━━"
echo " SSH keys: $( [[ "$STEP_SSH_OK" == true ]] && echo "ready ✅" || echo "skipped" )"
echo " Conf push: $( [[ "$CONF_PUSH_OK" == true ]] && echo "done ✅" || echo "⚠ manual needed" )"
echo " HOST1 setup: done ✅"
echo " Duration: $(format_duration $(( END - START )))"
echo ""
echo " HOST1 is fully set up. HOST2 ($MIRROR) can now install the Varaverk plugin."
echo " The wizard will detect the pushed conf and take the correct path."
echo " When HOST2 completes its onboard, it will automatically trigger Phase 2 here."
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ── Step 2: Stop mirror's existing auth stack ─────────────────────────────────────────────────
echo ""
echo "━━━ Step 2 — Stop Mirror Auth Stack ━━━"
if [[ "$SKIP_AUTH_STACK" == true ]]; then
warn "Skipping (--skip-auth-stack)"
else
stop_mirror_stack "PARTNERSHIP_REPLACE_CONTAINERS" "auth stack"
fi
# ── Step 4: Deploy auth stack on mirror ───────────────────────────────────────────────────────
echo ""
echo "━━━ Step 3 — Deploy Auth Stack on Mirror ━━━"
if [[ "$SKIP_AUTH_STACK" == true ]]; then
warn "Skipping (--skip-auth-stack)"
elif [[ ${#PARTNERSHIP_AUTH_STACK[@]} -eq 0 ]]; then
warn "PARTNERSHIP_AUTH_STACK not set in ${MY_ID} conf — skipping auth stack deploy"
warn "Add HOST${MY_ID: -1}_PARTNERSHIP_AUTH_STACK to host${MY_ID: -1}.conf"
STEP_AUTH_OK=false
else
deploy_xml_stack PARTNERSHIP_AUTH_STACK
AUTH_DEPLOYED=$_STACK_DEPLOYED
AUTH_FAILED=$_STACK_FAILED
echo "Auth stack: $AUTH_DEPLOYED deployed, $AUTH_FAILED failed"
[[ "$AUTH_FAILED" -gt 0 ]] && STEP_AUTH_OK=false
fi
# ── Step 5: Stop mirror's existing arr stack ──────────────────────────────────────────────────
echo ""
echo "━━━ Step 4 — Stop Mirror Arr Stack ━━━"
if [[ "$SKIP_ARR_STACK" == true ]]; then
warn "Skipping (--skip-arr-stack)"
elif [[ ${#PARTNERSHIP_ARR_STACK[@]} -eq 0 ]]; then
log "PARTNERSHIP_ARR_STACK not configured — skipping arr stack deploy"
SKIP_ARR_STACK=true
else
stop_mirror_stack "PARTNERSHIP_ARR_REPLACE_CONTAINERS" "arr stack"
fi
# ── Step 5: Deploy arr stack on mirror ───────────────────────────────────────────────────────
echo ""
echo "━━━ Step 5 — Deploy Arr Stack on Mirror ━━━"
if [[ "$SKIP_ARR_STACK" == true ]]; then
warn "Skipping (--skip-arr-stack)"
else
deploy_xml_stack PARTNERSHIP_ARR_STACK
ARR_DEPLOYED=$_STACK_DEPLOYED
ARR_FAILED=$_STACK_FAILED
echo "Arr stack: $ARR_DEPLOYED deployed, $ARR_FAILED failed"
[[ "$ARR_FAILED" -gt 0 ]] && STEP_ARR_OK=false
fi
# ── Step 6: Stop mirror's existing services stack ─────────────────────────────────────────────
echo ""
echo "━━━ Step 6 — Stop Mirror Services Stack ━━━"
if [[ "$SKIP_SERVICES_STACK" == true ]]; then
warn "Skipping (--skip-services-stack)"
elif [[ ${#PARTNERSHIP_SERVICES_STACK[@]} -eq 0 ]]; then
log "PARTNERSHIP_SERVICES_STACK not configured — skipping services stack deploy"
SKIP_SERVICES_STACK=true
else
stop_mirror_stack "PARTNERSHIP_SERVICES_REPLACE_CONTAINERS" "services stack"
fi
# ── Step 7: Deploy services stack on mirror ───────────────────────────────────────────────────
echo ""
echo "━━━ Step 7 — Deploy Services Stack on Mirror ━━━"
if [[ "$SKIP_SERVICES_STACK" == true ]]; then
warn "Skipping (--skip-services-stack)"
else
deploy_xml_stack PARTNERSHIP_SERVICES_STACK
SERVICES_DEPLOYED=$_STACK_DEPLOYED
SERVICES_FAILED=$_STACK_FAILED
echo "Services stack: $SERVICES_DEPLOYED deployed, $SERVICES_FAILED failed"
[[ "$SERVICES_FAILED" -gt 0 ]] && STEP_SERVICES_OK=false
fi
# ── Step 8: Partnership onboard ───────────────────────────────────────────────────────────────
echo ""
echo "━━━ Step 8 — Partnership Onboard ━━━"
if bash "$SCRIPTS_ROOT/Partnership/partnership_manager.sh" --onboard "${EXTRA_FLAGS[@]}"; then
echo "Partnership onboard complete ✅"
ONBOARD_OK=true
else
error "Partnership onboard failed"
ONBOARD_OK=false
fi
# ── Step 9: Arr library bootstrap ─────────────────────────────────────────────────────────────
echo ""
echo "━━━ Step 9 — Arr Library Bootstrap ━━━"
if [[ "$ONBOARD_OK" == false ]]; then
warn "Skipping — onboard did not complete"
elif [[ "$SKIP_ARR_SYNC" == true ]]; then
warn "Skipping (--skip-arr-sync)"
elif [[ ! -f "$SCRIPTS_ROOT/Media/arr_sync.sh" ]]; then
warn "arr_sync.sh not found — run Media/arr_sync.sh manually once arrs are live"
elif bash "$SCRIPTS_ROOT/Media/arr_sync.sh" "${EXTRA_FLAGS[@]}"; then
echo "Arr bootstrap complete ✅"
ARR_SYNC_OK=true
else
warn "Arr sync had errors — partnership still valid"
warn "Re-run Media/arr_sync.sh once all arr containers are live"
fi
# ── Step 10: Push master.conf to all listed hosts ─────────────────────────────────────────────
# SSH is now established and all partners have the plugin installed.
# Push the authoritative master.conf so every listed host is in sync immediately.
echo ""
echo "━━━ $ICON_GEAR Step 10 — master.conf Push ━━━"
if [[ "$ONBOARD_OK" == false ]]; then
warn "Skipping — onboard did not complete"
elif [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would push master.conf to all listed hosts"
MASTER_PUSH_OK=true
else
push_output=$(platform_push_conf)
push_rc=$?
[[ -n "$push_output" ]] && echo "$push_output"
platform_push_setup_state
if [[ $push_rc -eq 0 ]]; then
echo "master.conf sync complete ✅"
MASTER_PUSH_OK=true
else
warn "master.conf push had failures — retry via Scheduler → master.conf → Save Conf"
fi
fi
# ── Write Phase 2 completion state ────────────────────────────────────────────────────────────
[[ "$ONBOARD_OK" == true && "$DRY_RUN" == false ]] && write_onboard_phase "$MIRROR_ID" 2
# ── Summary ───────────────────────────────────────────────────────────────────────────────────
END=$(date +%s)
echo ""
echo "━━━━━ $ICON_SUMMARY ONBOARD SUMMARY ━━━━━"
echo " Owner: $MY_ID ($LOCAL_SERVER_NAME)"
echo " Mirror: $MIRROR ($MIRROR_IP)"
[[ "$PHASE2_ONLY" == true ]] && echo " Mode: Phase 2 (triggered by HOST2 notification)"
echo " Duration: $(format_duration $(( END - START )))"
echo ""
_ok() { [[ "$1" == true ]] && echo "✅" || echo "❌"; }
_skip() { [[ "$1" == true ]] && echo "skipped" || echo "$(_ok "$2")"; }
echo " Step 1 — SSH keys: $(_skip "$SKIP_SSH" "$STEP_SSH_OK")"
echo " Step 2 — Stop auth: $(_skip "$SKIP_AUTH_STACK" "$STEP_STOP_AUTH_OK")"
echo " Step 3 — Auth stack: $( [[ "$SKIP_AUTH_STACK" == true ]] && echo "skipped" || echo "${AUTH_DEPLOYED} deployed, ${AUTH_FAILED} failed" )"
echo " Step 4 — Stop arr: $(_skip "$SKIP_ARR_STACK" "$STEP_STOP_ARR_OK")"
echo " Step 5 — Arr stack: $( [[ "$SKIP_ARR_STACK" == true ]] && echo "skipped" || echo "${ARR_DEPLOYED} deployed, ${ARR_FAILED} failed" )"
echo " Step 6 — Stop services: $(_skip "$SKIP_SERVICES_STACK" "$STEP_STOP_SERVICES_OK")"
echo " Step 7 — Services stack: $( [[ "$SKIP_SERVICES_STACK" == true ]] && echo "skipped" || echo "${SERVICES_DEPLOYED} deployed, ${SERVICES_FAILED} failed" )"
echo " Step 8 — Onboard: $(_ok "$ONBOARD_OK")"
echo " Step 9 — Arr bootstrap: $( [[ "$SKIP_ARR_SYNC" == true || "$ONBOARD_OK" == false ]] && echo "skipped" || echo "$(_ok "$ARR_SYNC_OK")" )"
echo " Step 10 — Conf push: $( [[ "$ONBOARD_OK" == false ]] && echo "skipped" || echo "$(_ok "$MASTER_PUSH_OK")" )"
echo ""
if [[ "$ONBOARD_OK" == true ]]; then
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes made" || \
echo "$ICON_DONE DONE — partnership established ✅"
echo "Verify with: Partnership/partnership_manager.sh --status"
else
error "Setup incomplete — resolve errors above and re-run"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━"
[[ "$ONBOARD_OK" == false ]] && exit 1
exit 0
@@ -1,676 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= Partnership Onboard ============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Runs once on both servers to establish a new partnership. Role is detected
# automatically via detect_hosts() — no flags needed to declare which side you are.
# Run on the mirror first (generates its SSH key), then on the owner to complete
# setup remotely.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# MIRROR PATH (1 step)
# Step 1: SSH key setup — generate keypair, copy to owner, update conf
# Owner completes the rest remotely. Mirror is done.
#
# OWNER PATH (10 steps)
# Step 1: SSH key setup — generate keypair, install on mirror, update conf
# Step 2: Stop mirror auth — stop mirror's existing auth containers before replacing
# Step 3: Deploy auth stack — push XMLs, pull images, create + start on mirror
# Mariadb/Redis health-checked before Authelia deploys
# Step 4: Stop mirror arr — stop mirror's existing arr containers before replacing
# Step 5: Deploy arr stack — push arr XMLs, pull images, create + start on mirror
# Step 6: Stop mirror services — stop mirror's existing services containers before replacing
# Step 7: Deploy services stack — push Emby/Jellyfin/Seerr XMLs, pull images, create + start
# Step 8: Partnership onboard — configure WebUIs → owner IP, write state, Emby
# Step 9: Arr bootstrap — bidirectional library sync (arr_sync.sh)
# Step 10: Conf push — push master.conf + setup state to all listed hosts
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Credentials never in SSH command strings
# Auth stack containers hold API keys, DB passwords, etc. The deploy script is written
# locally, SCPed to the remote, and executed there. Command-line args are never used
# to pass credentials — they'd appear in `ps` output and shell history on both servers.
#
# XML templates are the single source of truth for deployed containers
# The owner's templates-user/ XMLs define every container deployed on the mirror.
# The same XMLs that Unraid's Docker Manager uses are what get SCPed — the mirror's
# Docker Manager can manage the containers after onboard without additional config.
#
# Dependency ordering in the auth stack is owner-enforced
# PARTNERSHIP_AUTH_STACK order matters: Mariadb and Redis must come before Authelia.
# The array is ordered correctly in host1.conf. After each Mariadb/Redis deploy,
# the script waits for the container to be healthy before continuing. This is a remote
# health check — the container must be running (or report healthy) before the next
# dependent is deployed.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Root check
# All operations run as root — SSH key management, docker operations, conf updates.
#
# SSH timeout on all remote calls
# Every ssh/scp call uses SSH_TIMEOUT. No operation hangs indefinitely on a
# slow or unreachable mirror.
#
# --dry-run shows exact actions without executing
# Every step prints what it would do. SCP, deploy, plugin install, arr sync —
# all dry-run safe.
#
# Step skip flags for partial re-runs
# --skip-ssh, --skip-auth-stack, --skip-arr-stack, --skip-arr-sync allow
# resuming after a partial failure without re-running completed steps.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_PARTNERSHIP_AUTH_STACK
# XML filenames (from this server's templates-user/) to push and deploy on the
# mirror as its auth stack. Order matters: database deps before Authelia.
# Aliased by detect_hosts() → PARTNERSHIP_AUTH_STACK
#
# HOST*_PARTNERSHIP_REPLACE_CONTAINERS
# Containers to stop on the mirror before deploying the auth stack.
# Defined in the MIRROR's own conf (host*.conf on HOST2) — never in HOST1's conf.
# Read live from the mirror via SSH during Step 3 (sources mirror's load_config.sh at
# the same $SCRIPTS_ROOT path — convention: both servers use the same repo location).
# Leave empty on HOST2 if no conflicting containers exist (fresh mirror: nothing to stop).
# Aliased by detect_hosts() → PARTNERSHIP_REPLACE_CONTAINERS (on the mirror)
#
# HOST*_PARTNERSHIP_ARR_STACK
# XML filenames to push and deploy on the mirror as its arr stack.
# Leave empty to skip arr stack deploy.
# Aliased by detect_hosts() → PARTNERSHIP_ARR_STACK
#
# HOST*_PARTNERSHIP_ARR_REPLACE_CONTAINERS
# Arr containers to stop on the mirror before deploying the arr stack.
# Same rule as PARTNERSHIP_REPLACE_CONTAINERS: defined in mirror's own conf, never HOST1's.
# Aliased by detect_hosts() → PARTNERSHIP_ARR_REPLACE_CONTAINERS (on the mirror)
#
# HOST*_PARTNERSHIP_SERVICES_STACK
# XML filenames to push and deploy on the mirror as its shared services stack.
# Includes Emby, Jellyfin, Seerr, SeerrFin. Leave empty to skip services stack deploy.
# Aliased by detect_hosts() → PARTNERSHIP_SERVICES_STACK
#
# HOST*_PARTNERSHIP_SERVICES_REPLACE_CONTAINERS
# Services containers to stop on the mirror before deploying the services stack.
# Same rule as PARTNERSHIP_REPLACE_CONTAINERS: defined in mirror's own conf, never HOST1's.
# Aliased by detect_hosts() → PARTNERSHIP_SERVICES_REPLACE_CONTAINERS (on the mirror)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# Partnership/partnership_onboard.sh
# Full onboard — role detected automatically
#
# Partnership/partnership_onboard.sh --dry-run
# Preview all steps without making changes
#
# Partnership/partnership_onboard.sh --log
# Verbose per-step output
#
# Partnership/partnership_onboard.sh --skip-ssh
# Skip SSH key setup (key already in place)
#
# Partnership/partnership_onboard.sh --skip-auth-stack
# Skip auth stack stop + deploy (Steps 3-4)
#
# Partnership/partnership_onboard.sh --skip-arr-stack
# Skip arr stack stop + deploy (Steps 4-5)
#
# Partnership/partnership_onboard.sh --skip-services-stack
# Skip services stack stop + deploy (Steps 6-7)
#
# Partnership/partnership_onboard.sh --skip-arr-sync
# Skip arr library bootstrap (Step 9)
#
# Partnership/partnership_onboard.sh --phase1-only
# OWNER only: SSH key exchange + conf push. Safe to run before HOST2 has Varaverk.
# Writes HOST2_PHASE1_DONE=true to varaverk_setup.db.
#
# Partnership/partnership_onboard.sh --phase2-only
# OWNER only: container deploy + arr + onboard (skips SSH). Triggered automatically
# by HOST2 after it completes its Mirror-path onboard. Can also be run manually.
# Writes HOST2_PHASE2_DONE=true to varaverk_setup.db.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
SCRIPTS_ROOT="$SCRIPT_DIR/.."
SSH_TIMEOUT=15
source "$SCRIPTS_ROOT/load_config.sh"
source "$SCRIPTS_ROOT/Plugin/$PLATFORM/Partnership/containers.sh"
# ── Parse flags ───────────────────────────────────────────────────────────────────────────────
SKIP_SSH=false
SKIP_AUTH_STACK=false
SKIP_ARR_STACK=false
SKIP_SERVICES_STACK=false
SKIP_ARR_SYNC=false
PHASE1_ONLY=false # OWNER: SSH + conf push only (HOST2 not yet installed)
PHASE2_ONLY=false # OWNER: containers/arr/onboard only (triggered by HOST2 after it onboards)
FILTERED_ARGS=()
for arg in "$@"; do
case "$arg" in
--skip-ssh) SKIP_SSH=true ;;
--skip-auth-stack) SKIP_AUTH_STACK=true ;;
--skip-arr-stack) SKIP_ARR_STACK=true ;;
--skip-services-stack) SKIP_SERVICES_STACK=true ;;
--skip-arr-sync) SKIP_ARR_SYNC=true ;;
--phase1-only) PHASE1_ONLY=true ;;
--phase2-only) PHASE2_ONLY=true; SKIP_SSH=true ;;
*) FILTERED_ARGS+=("$arg") ;;
esac
done
parse_args "${FILTERED_ARGS[@]}"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
[[ "$EUID" -ne 0 ]] && { error "Must be run as root"; exit 1; }
acquire_lock
if ! command -v docker &>/dev/null; then
error "Docker command not found"
exit 1
fi
detect_hosts
OWNER_ID="${PARTNERSHIP_OWNER_HOST:-HOST1}"
MIRROR_ID=$( [[ "$OWNER_ID" == "HOST1" ]] && echo "HOST2" || echo "HOST1" )
OWNER="${!OWNER_ID}"
MIRROR="${!MIRROR_ID}"
# SSH_KEY (set by detect_hosts) is this server's own private key.
# The remote accepts it because this server's PUBLIC key was installed there via ssh_setup.sh.
# HOST{N}_SSH_KEY lives in host{N}.conf — with sparse checkout, the other server's
# conf is never present here. Always use SSH_KEY (local private key) for outbound SSH.
MIRROR_SSH_KEY="$SSH_KEY"
AM_OWNER=false
AM_MIRROR=false
[[ "$MY_ID" == "$OWNER_ID" ]] && AM_OWNER=true
[[ "$MY_ID" == "$MIRROR_ID" ]] && AM_MIRROR=true
EXTRA_FLAGS=()
[[ "$DRY_RUN" == true ]] && EXTRA_FLAGS+=("--dry-run")
[[ "$LOG_MODE" == true ]] && EXTRA_FLAGS+=("--log")
START=$(date +%s)
# ── Helper: write phase completion flag to setup.db + push to remotes ─────────────────────────
write_onboard_phase() {
local target_id="$1" phase="$2"
local key="${target_id}_PHASE${phase}_DONE"
local state_file="$(platform_setup_db_path)"
[[ "$DRY_RUN" == true ]] && { warn "DRY RUN — would write ${key}=true"; return 0; }
if grep -q "^${key}=" "$state_file" 2>/dev/null; then
sed -i "s|^${key}=.*|${key}=true|" "$state_file"
else
echo "${key}=true" >> "$state_file"
fi
platform_push_setup_state
}
echo ""
echo "━━━ $ICON_FALLBACK Partnership Onboard — $MY_ID ($LOCAL_SERVER_NAME) — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
echo ""
echo " Role: $( [[ "$AM_OWNER" == true ]] && echo "OWNER" || echo "MIRROR" )"
echo " This: $MY_ID ($LOCAL_SERVER_NAME)"
echo " Partner: $( [[ "$AM_OWNER" == true ]] && echo "$MIRROR_ID ($MIRROR)" || echo "$OWNER_ID ($OWNER)" )"
echo ""
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no permanent changes will be made"
# ==============================================================================================
# ── HELPER: stop containers on the mirror by reading its own conf via SSH ────────────────────
#
# SSHes to the mirror, sources its load_config.sh at the same $SCRIPTS_ROOT path (both servers
# use the same convention), and reads the named config array from the mirror's own conf.
# HOST2's container list stays in HOST2's host2.conf — not duplicated in HOST1's conf.
# Fails gracefully if scripts aren't present yet or the array is empty (nothing to stop).
#
# deploy_container_from_xml() already stops/removes containers with the same name as what's
# being deployed. This step handles containers with DIFFERENT names that conflict.
# ==============================================================================================
stop_mirror_stack() {
local config_var="$1" label="$2"
local -a to_stop=()
mapfile -t to_stop < <(
timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$MIRROR_IP" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${${config_var}[@]:-}\"" 2>/dev/null | grep -v '^$'
)
if [[ ${#to_stop[@]} -eq 0 ]]; then
log "No $label containers to stop on $MIRROR — skipping"
return 0
fi
log "Stopping $label on $MIRROR: ${to_stop[*]}"
for container in "${to_stop[@]}"; do
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would stop + rm $container on $MIRROR"
continue
fi
timeout "$SSH_TIMEOUT" ssh -i "$MIRROR_SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" root@"$MIRROR_IP" \
"docker stop '$container' 2>/dev/null
docker rm '$container' 2>/dev/null && echo removed" 2>/dev/null | \
grep -q removed && \
log " $container removed ✅" || \
log " $container not found on $MIRROR — skipping"
done
}
# ==============================================================================================
# ── MIRROR PATH ───────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
if [[ "$AM_MIRROR" == true ]]; then
echo "━━━ Step 1/2 — SSH Key Setup (Mirror) ━━━"
echo ""
echo " Mirror sets up SSH keys, then notifies Owner to run Phase 2."
echo ""
if [[ "$SKIP_SSH" == true ]]; then
warn "Skipping SSH setup (--skip-ssh)"
elif bash "$SCRIPT_DIR/ssh_setup.sh" "${EXTRA_FLAGS[@]}"; then
log "SSH key ready ✅"
else
error "SSH key setup failed"
exit 1
fi
echo ""
echo "━━━ Step 2/2 — Notify Owner to Run Phase 2 ━━━"
echo ""
OWNER_IP=$(resolve_tailscale_ip "$OWNER" 2>/dev/null || true)
PHASE2_TRIGGERED=false
if [[ -n "$OWNER_IP" ]]; then
# Read OWNER's SCRIPTS_DIR via platform probe command — don't assume same path as mirror
_probe_cmd=$(platform_scripts_dir_probe_cmd)
OWNER_SCRIPTS_DIR=$(timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$OWNER_IP" \
"$_probe_cmd" 2>/dev/null | tr -d '[:space:]')
OWNER_SCRIPTS_DIR="${OWNER_SCRIPTS_DIR:-$SCRIPTS_DIR}"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would SSH to $OWNER ($OWNER_IP) and trigger Phase 2"
PHASE2_TRIGGERED=true
elif timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$OWNER_IP" \
"nohup bash '${OWNER_SCRIPTS_DIR}/Partnership/partnership_onboard.sh' --phase2-only > /tmp/vv_phase2_onboard.log 2>&1 & echo triggered" \
2>/dev/null | grep -q triggered; then
log "Phase 2 triggered on $OWNER ✅"
log "Watch progress on $OWNER: tail -f /tmp/vv_phase2_onboard.log"
PHASE2_TRIGGERED=true
else
warn "Could not auto-trigger Phase 2 on $OWNER"
fi
else
warn "Cannot resolve $OWNER Tailscale IP"
fi
echo ""
echo "━━━━━ $ICON_SUMMARY MIRROR SETUP COMPLETE ━━━━━"
echo " SSH key: ready"
echo " Phase 2 on $OWNER: $( [[ "$PHASE2_TRIGGERED" == true ]] && echo "triggered ✅" || echo "needs manual trigger ⚠" )"
if [[ "$PHASE2_TRIGGERED" == false ]]; then
echo ""
echo " Run manually on $OWNER:"
echo " bash Partnership/partnership_onboard.sh --phase2-only"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ── OWNER PATH ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
MIRROR_IP=$(resolve_tailscale_ip "$MIRROR")
[[ -z "$MIRROR_IP" ]] && { error "Cannot resolve $MIRROR Tailscale IP — is Tailscale running?"; exit 1; }
log "Mirror: $MIRROR ($MIRROR_IP)"
[[ "$PHASE1_ONLY" == true ]] && log "Mode: Phase 1 only (SSH + conf push)"
[[ "$PHASE2_ONLY" == true ]] && log "Mode: Phase 2 only (containers + arr + onboard)"
echo ""
STEP_SSH_OK=false
STEP_STOP_AUTH_OK=true
STEP_AUTH_OK=true
AUTH_DEPLOYED=0
AUTH_FAILED=0
STEP_STOP_ARR_OK=true
STEP_ARR_OK=true
ARR_DEPLOYED=0
ARR_FAILED=0
STEP_STOP_SERVICES_OK=true
STEP_SERVICES_OK=true
SERVICES_DEPLOYED=0
SERVICES_FAILED=0
ONBOARD_OK=false
ARR_SYNC_OK=false
MASTER_PUSH_OK=false
# ── Step 1: SSH ───────────────────────────────────────────────────────────────────────────────
# Skipped when --phase2-only (SSH was already done in Phase 1).
echo "━━━ Step 1 — SSH Key Setup ━━━"
if [[ "$SKIP_SSH" == true ]]; then
warn "Skipping (--skip-ssh)"
STEP_SSH_OK=true
elif [[ "$PHASE1_ONLY" == true ]]; then
# Phase 1 in background: test if SSH already works first — avoids ssh-copy-id
# hanging for a password prompt with no TTY.
if timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$MIRROR_IP" exit 0 2>/dev/null; then
log "SSH to $MIRROR already works ✅ — skipping key install"
STEP_SSH_OK=true
else
# Key not yet on HOST2 — try ssh_setup.sh (works interactively, may fail in background)
if bash "$SCRIPT_DIR/ssh_setup.sh" "${EXTRA_FLAGS[@]}"; then
log "SSH keys ready ✅"
STEP_SSH_OK=true
else
# Soft-fail: generate key locally if not present, then tell user to install manually
warn "Could not install key on $MIRROR automatically (no terminal for password prompt)"
if [[ -f "$SSH_KEY" ]]; then
log "Local key exists at: $SSH_KEY"
else
bash "$SCRIPT_DIR/ssh_setup.sh" --key-only "${EXTRA_FLAGS[@]}" 2>/dev/null || true
fi
if [[ -f "${SSH_KEY}.pub" ]]; then
echo ""
echo " Install this key on $MIRROR to complete SSH setup:"
echo " ┌─────────────────────────────────────────────────────"
cat "${SSH_KEY}.pub" | sed 's/^/ │ /'
echo " └─────────────────────────────────────────────────────"
echo " Run on a terminal: ssh-copy-id -i ${SSH_KEY}.pub root@${MIRROR_IP}"
echo " Then click 'Push Conf' in the Partnership tab."
# Write key-ready flag so UI can show the manual-install state
[[ "$DRY_RUN" == false ]] && {
local kflag="${MIRROR_ID}_KEY_READY"
local _setup_f="$(platform_setup_db_path)"
grep -q "^${kflag}=" "$_setup_f" 2>/dev/null \
&& sed -i "s|^${kflag}=.*|${kflag}=true|" "$_setup_f" \
|| echo "${kflag}=true" >> "$_setup_f"
}
fi
STEP_SSH_OK=false
fi
fi
elif bash "$SCRIPT_DIR/ssh_setup.sh" "${EXTRA_FLAGS[@]}"; then
log "SSH keys ready ✅"
STEP_SSH_OK=true
else
error "SSH key setup failed — aborting"
error "Re-run or use --skip-ssh if key is already set up"
exit 1
fi
# ── Phase 1 exit point ────────────────────────────────────────────────────────────────────────
# --phase1-only: SSH + conf push is all HOST1 needs to do before HOST2 installs Varaverk.
# HOST2's wizard will detect the pushed master.conf + state file and take the correct path.
if [[ "$PHASE1_ONLY" == true ]]; then
if [[ "$STEP_SSH_OK" == false ]]; then
# SSH key not yet installed on HOST2 — can't push conf, but local setup still runs.
# UI will show "key ready, install manually" state via HOST2_KEY_READY flag.
echo ""
echo "━━━ Phase 1 — HOST1 Local Setup (SSH pending) ━━━"
bash "$SCRIPT_DIR/partnership_manager.sh" --onboard --local-only "${EXTRA_FLAGS[@]}" || \
warn "Local setup had issues — check partnership_manager.sh output above"
END=$(date +%s)
echo ""
echo "━━━━━ $ICON_SUMMARY PHASE 1 — SSH PENDING ━━━━━"
echo " SSH keys: key generated ✅ — NOT yet installed on $MIRROR ⚠"
echo " Conf push: skipped (needs SSH access to $MIRROR)"
echo " HOST1 setup: done ✅"
echo " Duration: $(format_duration $(( END - START )))"
echo ""
echo " ACTION NEEDED: install the key on $MIRROR:"
echo " ssh-copy-id -i ${SSH_KEY}.pub root@${MIRROR_IP}"
echo " Then click 'Push Conf' in Partnership tab, or run:"
echo " bash Partnership/partnership_onboard.sh --phase1-only --skip-ssh"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
echo ""
echo "━━━ Phase 1 — Conf Push ━━━"
CONF_PUSH_OK=false
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would push master.conf + state file to $MIRROR"
CONF_PUSH_OK=true
else
push_output=$(platform_push_conf)
push_rc=$?
[[ -n "$push_output" ]] && echo "$push_output"
platform_push_setup_state
if [[ $push_rc -eq 0 ]]; then
log "Conf push complete ✅"
CONF_PUSH_OK=true
else
warn "Conf push had failures — retry via Scheduler → master.conf → Save Conf"
fi
fi
# HOST1 local setup — runs immediately without needing HOST2
echo ""
echo "━━━ Phase 1 — HOST1 Local Setup ━━━"
bash "$SCRIPT_DIR/partnership_manager.sh" --onboard --local-only "${EXTRA_FLAGS[@]}" || \
warn "Local setup had issues — check partnership_manager.sh output above"
[[ "$DRY_RUN" == false ]] && write_onboard_phase "$MIRROR_ID" 1
END=$(date +%s)
echo ""
echo "━━━━━ $ICON_SUMMARY PHASE 1 COMPLETE ━━━━━"
echo " SSH keys: $( [[ "$STEP_SSH_OK" == true ]] && echo "ready ✅" || echo "skipped" )"
echo " Conf push: $( [[ "$CONF_PUSH_OK" == true ]] && echo "done ✅" || echo "⚠ manual needed" )"
echo " HOST1 setup: done ✅"
echo " Duration: $(format_duration $(( END - START )))"
echo ""
echo " HOST1 is fully set up. HOST2 ($MIRROR) can now install the Varaverk plugin."
echo " The wizard will detect the pushed conf and take the correct path."
echo " When HOST2 completes its onboard, it will automatically trigger Phase 2 here."
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ── Step 2: Stop mirror's existing auth stack ─────────────────────────────────────────────────
echo ""
echo "━━━ Step 2 — Stop Mirror Auth Stack ━━━"
if [[ "$SKIP_AUTH_STACK" == true ]]; then
warn "Skipping (--skip-auth-stack)"
else
stop_mirror_stack "PARTNERSHIP_REPLACE_CONTAINERS" "auth stack"
fi
# ── Step 4: Deploy auth stack on mirror ───────────────────────────────────────────────────────
echo ""
echo "━━━ Step 3 — Deploy Auth Stack on Mirror ━━━"
if [[ "$SKIP_AUTH_STACK" == true ]]; then
warn "Skipping (--skip-auth-stack)"
elif [[ ${#PARTNERSHIP_AUTH_STACK[@]} -eq 0 ]]; then
warn "PARTNERSHIP_AUTH_STACK not set in ${MY_ID} conf — skipping auth stack deploy"
warn "Add HOST${MY_ID: -1}_PARTNERSHIP_AUTH_STACK to host${MY_ID: -1}.conf"
STEP_AUTH_OK=false
else
deploy_xml_stack PARTNERSHIP_AUTH_STACK
AUTH_DEPLOYED=$_STACK_DEPLOYED
AUTH_FAILED=$_STACK_FAILED
echo "Auth stack: $AUTH_DEPLOYED deployed, $AUTH_FAILED failed"
[[ "$AUTH_FAILED" -gt 0 ]] && STEP_AUTH_OK=false
fi
# ── Step 5: Stop mirror's existing arr stack ──────────────────────────────────────────────────
echo ""
echo "━━━ Step 4 — Stop Mirror Arr Stack ━━━"
if [[ "$SKIP_ARR_STACK" == true ]]; then
warn "Skipping (--skip-arr-stack)"
elif [[ ${#PARTNERSHIP_ARR_STACK[@]} -eq 0 ]]; then
log "PARTNERSHIP_ARR_STACK not configured — skipping arr stack deploy"
SKIP_ARR_STACK=true
else
stop_mirror_stack "PARTNERSHIP_ARR_REPLACE_CONTAINERS" "arr stack"
fi
# ── Step 5: Deploy arr stack on mirror ───────────────────────────────────────────────────────
echo ""
echo "━━━ Step 5 — Deploy Arr Stack on Mirror ━━━"
if [[ "$SKIP_ARR_STACK" == true ]]; then
warn "Skipping (--skip-arr-stack)"
else
deploy_xml_stack PARTNERSHIP_ARR_STACK
ARR_DEPLOYED=$_STACK_DEPLOYED
ARR_FAILED=$_STACK_FAILED
echo "Arr stack: $ARR_DEPLOYED deployed, $ARR_FAILED failed"
[[ "$ARR_FAILED" -gt 0 ]] && STEP_ARR_OK=false
fi
# ── Step 6: Stop mirror's existing services stack ─────────────────────────────────────────────
echo ""
echo "━━━ Step 6 — Stop Mirror Services Stack ━━━"
if [[ "$SKIP_SERVICES_STACK" == true ]]; then
warn "Skipping (--skip-services-stack)"
elif [[ ${#PARTNERSHIP_SERVICES_STACK[@]} -eq 0 ]]; then
log "PARTNERSHIP_SERVICES_STACK not configured — skipping services stack deploy"
SKIP_SERVICES_STACK=true
else
stop_mirror_stack "PARTNERSHIP_SERVICES_REPLACE_CONTAINERS" "services stack"
fi
# ── Step 7: Deploy services stack on mirror ───────────────────────────────────────────────────
echo ""
echo "━━━ Step 7 — Deploy Services Stack on Mirror ━━━"
if [[ "$SKIP_SERVICES_STACK" == true ]]; then
warn "Skipping (--skip-services-stack)"
else
deploy_xml_stack PARTNERSHIP_SERVICES_STACK
SERVICES_DEPLOYED=$_STACK_DEPLOYED
SERVICES_FAILED=$_STACK_FAILED
echo "Services stack: $SERVICES_DEPLOYED deployed, $SERVICES_FAILED failed"
[[ "$SERVICES_FAILED" -gt 0 ]] && STEP_SERVICES_OK=false
fi
# ── Step 8: Partnership onboard ───────────────────────────────────────────────────────────────
echo ""
echo "━━━ Step 8 — Partnership Onboard ━━━"
if bash "$SCRIPTS_ROOT/Partnership/partnership_manager.sh" --onboard "${EXTRA_FLAGS[@]}"; then
echo "Partnership onboard complete ✅"
ONBOARD_OK=true
else
error "Partnership onboard failed"
ONBOARD_OK=false
fi
# ── Step 9: Arr library bootstrap ─────────────────────────────────────────────────────────────
echo ""
echo "━━━ Step 9 — Arr Library Bootstrap ━━━"
if [[ "$ONBOARD_OK" == false ]]; then
warn "Skipping — onboard did not complete"
elif [[ "$SKIP_ARR_SYNC" == true ]]; then
warn "Skipping (--skip-arr-sync)"
elif [[ ! -f "$SCRIPTS_ROOT/Media/arr_sync.sh" ]]; then
warn "arr_sync.sh not found — run Media/arr_sync.sh manually once arrs are live"
elif bash "$SCRIPTS_ROOT/Media/arr_sync.sh" "${EXTRA_FLAGS[@]}"; then
echo "Arr bootstrap complete ✅"
ARR_SYNC_OK=true
else
warn "Arr sync had errors — partnership still valid"
warn "Re-run Media/arr_sync.sh once all arr containers are live"
fi
# ── Step 10: Push master.conf to all listed hosts ─────────────────────────────────────────────
# SSH is now established and all partners have the plugin installed.
# Push the authoritative master.conf so every listed host is in sync immediately.
echo ""
echo "━━━ $ICON_GEAR Step 10 — master.conf Push ━━━"
if [[ "$ONBOARD_OK" == false ]]; then
warn "Skipping — onboard did not complete"
elif [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would push master.conf to all listed hosts"
MASTER_PUSH_OK=true
else
push_output=$(platform_push_conf)
push_rc=$?
[[ -n "$push_output" ]] && echo "$push_output"
platform_push_setup_state
if [[ $push_rc -eq 0 ]]; then
echo "master.conf sync complete ✅"
MASTER_PUSH_OK=true
else
warn "master.conf push had failures — retry via Scheduler → master.conf → Save Conf"
fi
fi
# ── Write Phase 2 completion state ────────────────────────────────────────────────────────────
[[ "$ONBOARD_OK" == true && "$DRY_RUN" == false ]] && write_onboard_phase "$MIRROR_ID" 2
# ── Summary ───────────────────────────────────────────────────────────────────────────────────
END=$(date +%s)
echo ""
echo "━━━━━ $ICON_SUMMARY ONBOARD SUMMARY ━━━━━"
echo " Owner: $MY_ID ($LOCAL_SERVER_NAME)"
echo " Mirror: $MIRROR ($MIRROR_IP)"
[[ "$PHASE2_ONLY" == true ]] && echo " Mode: Phase 2 (triggered by HOST2 notification)"
echo " Duration: $(format_duration $(( END - START )))"
echo ""
_ok() { [[ "$1" == true ]] && echo "✅" || echo "❌"; }
_skip() { [[ "$1" == true ]] && echo "skipped" || echo "$(_ok "$2")"; }
echo " Step 1 — SSH keys: $(_skip "$SKIP_SSH" "$STEP_SSH_OK")"
echo " Step 2 — Stop auth: $(_skip "$SKIP_AUTH_STACK" "$STEP_STOP_AUTH_OK")"
echo " Step 3 — Auth stack: $( [[ "$SKIP_AUTH_STACK" == true ]] && echo "skipped" || echo "${AUTH_DEPLOYED} deployed, ${AUTH_FAILED} failed" )"
echo " Step 4 — Stop arr: $(_skip "$SKIP_ARR_STACK" "$STEP_STOP_ARR_OK")"
echo " Step 5 — Arr stack: $( [[ "$SKIP_ARR_STACK" == true ]] && echo "skipped" || echo "${ARR_DEPLOYED} deployed, ${ARR_FAILED} failed" )"
echo " Step 6 — Stop services: $(_skip "$SKIP_SERVICES_STACK" "$STEP_STOP_SERVICES_OK")"
echo " Step 7 — Services stack: $( [[ "$SKIP_SERVICES_STACK" == true ]] && echo "skipped" || echo "${SERVICES_DEPLOYED} deployed, ${SERVICES_FAILED} failed" )"
echo " Step 8 — Onboard: $(_ok "$ONBOARD_OK")"
echo " Step 9 — Arr bootstrap: $( [[ "$SKIP_ARR_SYNC" == true || "$ONBOARD_OK" == false ]] && echo "skipped" || echo "$(_ok "$ARR_SYNC_OK")" )"
echo " Step 10 — Conf push: $( [[ "$ONBOARD_OK" == false ]] && echo "skipped" || echo "$(_ok "$MASTER_PUSH_OK")" )"
echo ""
if [[ "$ONBOARD_OK" == true ]]; then
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes made" || \
echo "$ICON_DONE DONE — partnership established ✅"
echo "Verify with: Partnership/partnership_manager.sh --status"
else
error "Setup incomplete — resolve errors above and re-run"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━"
[[ "$ONBOARD_OK" == false ]] && exit 1
exit 0
@@ -1,191 +0,0 @@
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
# 🤝 PARTNERSHIP
# ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
**Managed lifecycle for a two-server partnership — setup, ongoing operation,
and clean separation.** One server owns the shared services. The other mirrors
them and benefits from them. Every phase of the relationship has the same
engineering discipline as the rest of the ecosystem.
> **This folder exists because a clean exit should be as easy as a clean setup.**
> The partnership is not a permanent commitment. `--offboard` works from either
> server at any time. Everything the mirror needs to run independently is already
> there. The only thing that stops on separation is the sync — and that's intentional.
---
## ━━━ THE PROBLEM THAT BUILT THIS ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
---
### 🔴 Two Servers, One Auth Stack, No Clean Way to Share It
The auth stack — NginxProxyManager, LLDAP, Authelia, MariaDB, Redis — runs on
HOST1. HOST2 serves its own domain to its own household. It needs its own auth.
But maintaining two independent auth stacks means double the work: two places to
add users, two places to update proxy rules, two places to renew certs, two
configurations that inevitably drift apart. One change on HOST1 has to be manually
replicated to HOST2 — or it isn't, and the configurations diverge silently.
The real cost isn't the initial setup. It's the maintenance burden that accumulates
over months — every new user, every proxy rule change, every config update applied
in one place and forgotten in the other.
**The fix:** one auth stack with a managed mirror. HOST1 owns the configuration.
HOST2 runs a warm copy that stays current via 30-minute sync. HOST2's operator makes
zero auth management decisions — clicking an auth container opens HOST1's WebUI via
Tailscale. Changes happen there, propagate to HOST2 in 30 minutes. One place to
manage everything for both households.
---
### 🔴 No Structure Around the Relationship Itself
Setting up the mirror was a manual process. SSH in, reconfigure container WebUI
URLs one by one, copy auth config, verify connectivity, update state tracking. No
defined sequence. No dry-run capability. No verification that each step worked. If
something went wrong midway, the mirror was in an inconsistent state with no clear
way to understand what had and hadn't been done. Offboard was worse — it involves
stopping a sync that's been running for months, making a final copy of data,
reconfiguring WebUIs back to local addresses, removing Tailscale access, and
notifying both servers. A manual process with that many steps, taken under pressure,
leaves one or both parties in a bad state.
**The fix:** `partnership_manager.sh` with explicit modes for each lifecycle phase.
Each mode is a defined sequence. Every step is verified. Dry-run shows exactly what
will happen before anything changes. State files make the current relationship status
unambiguous from either server.
---
### 🔴 No Safe Way to Check If the Other Server Has Gone Away
After months of operation, HOST2 goes quiet. The sync starts failing. The offline
counter increments. But nothing actually happens — the ecosystem just keeps failing
the same sync, incrementing the same counter, sending the same notifications.
Without a defined threshold and an automated response, "partner gone for 30 days"
looks exactly like "partner gone for 3 years."
**The fix:** `PARTNERSHIP_OFFLINE_THRESHOLD`. After this many days of missed sync
cycles, both servers independently auto-offboard. HOST1 removes HOST2 from Tailscale,
disables critical sync, writes INACTIVE state. HOST2 — if it eventually comes back —
reads HOST1's INACTIVE state and cleans up its own side. The relationship is formally
ended from both sides without anyone needing to be present.
---
### 🔴 Ownership Transfer Had No Safe Path
The arrangement was always intended to be flexible — HOST1 owns the auth stack now,
but circumstances change. Swapping ownership manually meant reconfiguring WebUIs on
both servers, swapping sync direction, updating master.conf on both, and hoping the
sequence was correct. A misstep — like flipping sync direction before the final sync
completed — leaves both servers with different auth configurations and no clear source
of truth.
**The fix:** `--transfer` with a required confirmation string, a consecutive health
check system, and a strict sequence. The confirmation string cannot be typed
accidentally. Health strikes require both servers to be healthy on multiple
consecutive checks before the transfer begins. A final sync in the current direction
runs before anything is flipped.
---
## ━━━ WHAT THIS FOLDER DOES ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
**Onboarding** (`partnership_onboard.sh`) — one-time setup run from both servers.
Generates SSH keys, installs the auth stack and arr stack on the mirror from XML
templates, configures WebUI redirects to the owner, and bootstraps the arr library.
Role is detected automatically — no flags needed to declare which side you are.
**Offboarding** (`partnership_offboard.sh`) — handles clean separation from either
role. Owner path: final sync, WebUI reconfigure, remote container + appdata cleanup
(auth/arr stack by XML array, fallback coverage by naming), SSH key revocation,
Tailscale removal. Mirror path: local WebUI reconfigure, remove owner-deployed
containers locally, disable sync, revoke Emby admin, SSH key revocation, signal owner.
Called by `partnership_manager.sh --offboard` but runnable directly.
**Lifecycle management** (`partnership_manager.sh`) — dispatcher and monitor.
- Manually: `--onboard`, `--offboard` (delegates to offboard script), `--transfer`, `--status`, `--unblock`
- Automatically: `--check` called every 30 minutes by `critical_sync_maintenance.sh`
**SSH management** (`ssh_setup.sh`) — generates the keypair for rsync automation,
installs it on the remote, and tracks auth failures with a configurable strike system.
Called by `partnership_onboard.sh` but runnable independently for validation and
re-keying.
---
## ━━━ RELATIONSHIP TO OTHER FOLDERS ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
**Orchestrators/** — `critical_sync_maintenance.sh` calls `partnership_manager.sh --check`
every 30 minutes, passing `--remote-seen` or `--remote-unseen` based on whether the
rsync to the partner succeeded. The rsync outcome is the connectivity signal — no
separate ping needed.
**Rsync/** — Critical-Data rsync keeps the auth stack appdata current on the mirror
(NPM rules, Authelia config, LLDAP database, certs). Partnership manages the
relationship; Rsync delivers the actual data. On offboard, `rsync_stop.sh --rsync-only`
stops any running rsync before the final sync runs.
**Media/** — `arr_sync.sh` bootstraps the mirror's arr library during onboard, ensuring
both servers have each other's full library from day one. Ongoing arr sync runs
independently at 4-hour cadence.
---
## ━━━ SCRIPTS IN THIS FOLDER ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
| Script | Role | When It Runs |
|--------|------|--------------|
| `partnership_onboard.sh` | One-time setup — SSH keys, stack deploy, arr bootstrap | Manually, once per server per partnership |
| `partnership_offboard.sh` | Clean separation — both paths, both roles | Via `partnership_manager.sh --offboard`; or directly |
| `partnership_manager.sh` | Dispatcher + monitor — onboard WebUIs, health check, transfer, status | `--check` every 30min; all other modes manually |
| `ssh_setup.sh` | SSH key generation, remote install, auth validation | Called by onboard; manually for re-keying or validation |
---
## ━━━ HOW THE SCRIPTS RELATE ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
```
INITIAL SETUP (run once)
─────────────────────────────────────────────────────────────────────────────
HOST2 (mirror) runs:
partnership_onboard.sh
└─ ssh_setup.sh generates keypair, copies to owner
HOST1 (owner) runs:
partnership_onboard.sh
├─ ssh_setup.sh generates keypair, copies to mirror
├─ [stop mirror auth stack] PARTNERSHIP_REPLACE_CONTAINERS via SSH
├─ deploy_container_from_xml() pushes auth XMLs to mirror + starts containers
│ └─ wait_for_container_healthy() Mariadb/Redis health-checked before Authelia
├─ [stop mirror arr stack] PARTNERSHIP_ARR_REPLACE_CONTAINERS via SSH
├─ deploy_container_from_xml() pushes arr XMLs to mirror + starts containers
├─ partnership_manager.sh --onboard reconfigures WebUIs, writes ACTIVE state
└─ arr_sync.sh bootstraps full library on both servers
ONGOING OPERATION (every 30min)
─────────────────────────────────────────────────────────────────────────────
critical_sync_maintenance.sh
├─ Critical-Data rsync keeps auth appdata current on mirror
└─ partnership_manager.sh --check reads state files, tracks offline counter
├─ --remote-seen path rsync succeeded → reset counter
└─ --remote-unseen path rsync failed → increment counter → auto-offboard at threshold
OFFBOARD (manual or auto)
─────────────────────────────────────────────────────────────────────────────
partnership_manager.sh --offboard
└─ partnership_offboard.sh (exec'd — holds own lock)
├─ [owner-initiated] stop rsync → final sync → reconfigure mirror WebUIs
│ → disable critical sync → write INACTIVE state
│ → local fallback cleanup → restart own stack
│ → remote: remove auth/arr stack + fallback containers + appdata
│ → restart mirror stack → Emby revoke → SSH key revocation
│ → Tailscale removal after grace window
└─ [mirror-initiated] stop rsync → reconfigure own WebUIs
→ remove owner-deployed containers locally (reads owner's stack arrays)
→ remove local fallback containers → disable critical sync
→ revoke own Emby admin → restart own stack
→ SSH key revocation → write INACTIVE → signal owner
```
@@ -1,215 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================ Watchdog Orchestrator ===========================================
# ==============================================================================================
# Runs WATCHDOG_ORCHESTRATOR_SCRIPTS in order each cron cycle.
# Schedule: */15 * * * * (every 15 minutes)
#
# ── EXECUTION ORDER ───────────────────────────────────────────────────────────────────────────
# Driven by WATCHDOG_ORCHESTRATOR_SCRIPTS in master.conf — add, remove, or reorder there.
# Default: resource_watchdog → docker_watchdog → system_watchdog → unraid_api_key_renew → stability_watchdog
#
# ── WHY ORDER MATTERS ─────────────────────────────────────────────────────────────────────────
# Resource Watchdog first — frees RAM and CPU before healing attempts container restarts.
# Containers restarted into a resource-pressured system just fail again.
# Docker Watchdog second — restarts with pressure already reduced, more likely to stabilise.
# System Watchdog third — system component health after containers are healed.
# API key renew fourth — self-heals unraid-api registry loss; check-first, silent when valid.
# Stability Watchdog last — only reboots when all prior layers could not resolve the issue.
#
# ── ARRAY CHECK ───────────────────────────────────────────────────────────────────────────────
# Exits immediately if /mnt/user is not mounted as shfs (array not started).
# Watchdogs check Docker containers and storage — meaningless without the array.
# Prevents false positives and unnecessary reboots when array is stopped or stopping.
#
# ── STARTUP GRACE ─────────────────────────────────────────────────────────────────────────────
# No action until system uptime >= WATCHDOG_STARTUP_GRACE seconds.
# Prevents false positives from containers still starting at array launch.
# Each sub-script enforces this independently — orchestrator exits early to avoid log noise.
#
# ── OVERLAP PROTECTION ────────────────────────────────────────────────────────────────────────
# acquire_lock() — exits immediately if a prior cycle is still in progress.
# Prevents pile-up when a cycle runs long (daemon restart attempt = 30s, etc.).
#
# ── REPLACES ──────────────────────────────────────────────────────────────────────────────────
# Continuous loops previously in system_watchdog.sh and docker_watchdog.sh.
# Those scripts are now single-pass — this orchestrator provides the cadence.
#
# ── CONFIGURATION (master.conf) ───────────────────────────────────────────────────────────────
# WATCHDOG_ORCHESTRATOR_SCRIPTS — watchdogs to run, in order
# WATCHDOG_STARTUP_GRACE — seconds after boot before checks activate
# WATCHDOG_ORCHESTRATOR_HEARTBEAT — periodic heartbeat log toggle
# WATCHDOG_ORCHESTRATOR_HEARTBEAT_HOURS — heartbeat interval in hours
#
# ── USAGE ─────────────────────────────────────────────────────────────────────────────────────
# watchdog_orchestrator.sh — normal run (called by cron every 15 minutes)
# watchdog_orchestrator.sh --dry-run — pass --dry-run to all sub-scripts
# watchdog_orchestrator.sh --status — show script paths and current grace state
# watchdog_orchestrator.sh --log — verbose output from all sub-scripts
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
ECOSYSTEM_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
source "$ECOSYSTEM_ROOT/load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
# Skip immediately if another cycle is still running — no pile-up
acquire_lock
detect_hosts
log "$ICON_GEAR Config: grace=${WATCHDOG_STARTUP_GRACE}s heartbeat=${WATCHDOG_ORCHESTRATOR_HEARTBEAT:-true}/${WATCHDOG_ORCHESTRATOR_HEARTBEAT_HOURS:-1}hr scripts=${#WATCHDOG_ORCHESTRATOR_SCRIPTS[@]}"
log "$ICON_WATCHDOG Order: $(for s in "${WATCHDOG_ORCHESTRATOR_SCRIPTS[@]}"; do printf '%s ' "${s##*/}"; done)"
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — passing --dry-run to all sub-scripts"
# Derive a display name from a script path: "resource_watchdog.sh" → "Resource Watchdog"
_watchdog_display_name() {
local path="$1"
local base="${path##*/}"
base="${base%.sh}"
base="${base//_/ }"
echo "$base" | awk '{for(i=1;i<=NF;i++) $i=toupper(substr($i,1,1)) substr($i,2); print}'
}
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY WATCHDOG ORCHESTRATOR STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo ""
UPTIME_S=$(awk '{print int($1)}' /proc/uptime)
if [[ "$UPTIME_S" -lt "$WATCHDOG_STARTUP_GRACE" ]]; then
warn "Within startup grace — $(format_duration $UPTIME_S) / $(format_duration $WATCHDOG_STARTUP_GRACE)"
else
echo "Past startup grace — $(format_duration $UPTIME_S) uptime"
fi
echo ""
echo "── Sub-scripts ──"
for entry in "${WATCHDOG_ORCHESTRATOR_SCRIPTS[@]}"; do
local_path="$ECOSYSTEM_ROOT/$entry"
label="$(_watchdog_display_name "$entry")"
if [[ -f "$local_path" ]]; then
[[ -x "$local_path" ]] && icon="$ICON_DONE" || icon="$ICON_WARN"
echo " $icon $label — ${local_path##*/}"
else
echo " $ICON_ERROR $label — NOT FOUND: $local_path"
fi
done
echo ""
echo " Schedule: */15 * * * * (every 15 minutes)"
echo " Heartbeat: ${WATCHDOG_ORCHESTRATOR_HEARTBEAT:-true} / every ${WATCHDOG_ORCHESTRATOR_HEARTBEAT_HOURS:-1}hr"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ━━━ Array Check ━━━
# ==============================================================================================
if ! platform_storage_healthy; then
echo "Array not started — skipping watchdog cycle"
exit 0
fi
log "$ICON_DISK Array: /mnt/user mounted (shfs) ✅"
# ==============================================================================================
# ━━━ Startup Grace ━━━
# ==============================================================================================
UPTIME_SECONDS=$(awk '{print int($1)}' /proc/uptime)
if [[ "$UPTIME_SECONDS" -lt "$WATCHDOG_STARTUP_GRACE" ]]; then
echo "Startup grace — $(format_duration $UPTIME_SECONDS) / $(format_duration $WATCHDOG_STARTUP_GRACE) — skipping cycle"
exit 0
fi
log "Startup grace: past — uptime $(format_duration $UPTIME_SECONDS)"
# ==============================================================================================
# ━━━ Run Watchdog Cycle ━━━
# ==============================================================================================
CYCLE_START=$(date +%s)
PASS=()
FAIL=()
run_watchdog() {
local name="$1" script="$2"
if [[ ! -f "$script" ]]; then
error "$name — not found: $script"
FAIL+=("$name:missing")
return 1
fi
[[ ! -x "$script" ]] && chmod +x "$script"
local extra_args=()
[[ "$DRY_RUN" == true ]] && extra_args+=("--dry-run")
[[ "$VERBOSE" == true ]] && extra_args+=("--log")
local _ws
_ws=$(date +%s)
log "$ICON_START $name"
if bash "$script" "${extra_args[@]}"; then
log "$ICON_DONE $name — done in $(format_duration $(( $(date +%s) - _ws )))"
PASS+=("$name")
return 0
else
error "$name — non-zero exit ($(format_duration $(( $(date +%s) - _ws ))))"
FAIL+=("$name")
return 1
fi
}
for _entry in "${WATCHDOG_ORCHESTRATOR_SCRIPTS[@]}"; do
run_watchdog "$(_watchdog_display_name "$_entry")" "$ECOSYSTEM_ROOT/$_entry"
done
CYCLE_END=$(date +%s)
DURATION=$(( CYCLE_END - CYCLE_START ))
# ==============================================================================================
# ━━━ Heartbeat ━━━
# ==============================================================================================
if [[ "${WATCHDOG_ORCHESTRATOR_HEARTBEAT:-true}" == true ]]; then
HB_SECONDS=$(( ${WATCHDOG_ORCHESTRATOR_HEARTBEAT_HOURS:-1} * 3600 ))
HB_COUNT_FILE="${STATE_DIR:-/tmp}/watchdog_orch_hb.count"
HB_COUNT=$(cat "$HB_COUNT_FILE" 2>/dev/null || echo 0)
HB_COUNT=$(( HB_COUNT + 1 ))
echo "$HB_COUNT" > "$HB_COUNT_FILE"
# Each cron run = ~60s — use count × 60 as uptime approximation
HB_ELAPSED=$(( HB_COUNT * 60 ))
if [[ "$HB_SECONDS" -gt 0 ]] && (( HB_ELAPSED % HB_SECONDS < 60 )) && [[ "$HB_COUNT" -gt 1 ]]; then
HB_HR=$(( HB_ELAPSED / 3600 ))
warn "♥ watchdog_orchestrator alive — $MY_ID — ~${HB_HR}hr ($(date '+%H:%M:%S'))"
fi
fi
# ==============================================================================================
# ━━━ Summary — only shown on failures or --log ━━━
# ==============================================================================================
if [[ "${#FAIL[@]}" -gt 0 || "$VERBOSE" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY WATCHDOG CYCLE — $MY_ID — $(date '+%H:%M:%S') ━━━━━"
for p in "${PASS[@]}"; do log " $ICON_DONE $p"; done
for f in "${FAIL[@]}"; do error " $ICON_ERROR $f"; done
echo "$ICON_TIME Duration: $(format_duration $DURATION)"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
if [[ "${#FAIL[@]}" -gt 0 ]]; then
notify "Watchdog cycle failure on $(hostname) ($MY_ID) — ${FAIL[*]}" \
"Watchdog Orchestrator" "warning"
fi
fi
@@ -1,215 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================ Watchdog Orchestrator ===========================================
# ==============================================================================================
# Runs WATCHDOG_ORCHESTRATOR_SCRIPTS in order each cron cycle.
# Schedule: */15 * * * * (every 15 minutes)
#
# ── EXECUTION ORDER ───────────────────────────────────────────────────────────────────────────
# Driven by WATCHDOG_ORCHESTRATOR_SCRIPTS in master.conf — add, remove, or reorder there.
# Default: resource_watchdog → docker_watchdog → system_watchdog → unraid_api_key_renew → stability_watchdog
#
# ── WHY ORDER MATTERS ─────────────────────────────────────────────────────────────────────────
# Resource Watchdog first — frees RAM and CPU before healing attempts container restarts.
# Containers restarted into a resource-pressured system just fail again.
# Docker Watchdog second — restarts with pressure already reduced, more likely to stabilise.
# System Watchdog third — system component health after containers are healed.
# API key renew fourth — self-heals unraid-api registry loss; check-first, silent when valid.
# Stability Watchdog last — only reboots when all prior layers could not resolve the issue.
#
# ── ARRAY CHECK ───────────────────────────────────────────────────────────────────────────────
# Exits immediately if /mnt/user is not mounted as shfs (array not started).
# Watchdogs check Docker containers and storage — meaningless without the array.
# Prevents false positives and unnecessary reboots when array is stopped or stopping.
#
# ── STARTUP GRACE ─────────────────────────────────────────────────────────────────────────────
# No action until system uptime >= WATCHDOG_STARTUP_GRACE seconds.
# Prevents false positives from containers still starting at array launch.
# Each sub-script enforces this independently — orchestrator exits early to avoid log noise.
#
# ── OVERLAP PROTECTION ────────────────────────────────────────────────────────────────────────
# acquire_lock() — exits immediately if a prior cycle is still in progress.
# Prevents pile-up when a cycle runs long (daemon restart attempt = 30s, etc.).
#
# ── REPLACES ──────────────────────────────────────────────────────────────────────────────────
# Continuous loops previously in system_watchdog.sh and docker_watchdog.sh.
# Those scripts are now single-pass — this orchestrator provides the cadence.
#
# ── CONFIGURATION (master.conf) ───────────────────────────────────────────────────────────────
# WATCHDOG_ORCHESTRATOR_SCRIPTS — watchdogs to run, in order
# WATCHDOG_STARTUP_GRACE — seconds after boot before checks activate
# WATCHDOG_ORCHESTRATOR_HEARTBEAT — periodic heartbeat log toggle
# WATCHDOG_ORCHESTRATOR_HEARTBEAT_HOURS — heartbeat interval in hours
#
# ── USAGE ─────────────────────────────────────────────────────────────────────────────────────
# watchdog_orchestrator.sh — normal run (called by cron every 15 minutes)
# watchdog_orchestrator.sh --dry-run — pass --dry-run to all sub-scripts
# watchdog_orchestrator.sh --status — show script paths and current grace state
# watchdog_orchestrator.sh --log — verbose output from all sub-scripts
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
ECOSYSTEM_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
source "$ECOSYSTEM_ROOT/load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
# Skip immediately if another cycle is still running — no pile-up
acquire_lock
detect_hosts
log "$ICON_GEAR Config: grace=${WATCHDOG_STARTUP_GRACE}s heartbeat=${WATCHDOG_ORCHESTRATOR_HEARTBEAT:-true}/${WATCHDOG_ORCHESTRATOR_HEARTBEAT_HOURS:-1}hr scripts=${#WATCHDOG_ORCHESTRATOR_SCRIPTS[@]}"
log "$ICON_WATCHDOG Order: $(for s in "${WATCHDOG_ORCHESTRATOR_SCRIPTS[@]}"; do printf '%s ' "${s##*/}"; done)"
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — passing --dry-run to all sub-scripts"
# Derive a display name from a script path: "resource_watchdog.sh" → "Resource Watchdog"
_watchdog_display_name() {
local path="$1"
local base="${path##*/}"
base="${base%.sh}"
base="${base//_/ }"
echo "$base" | awk '{for(i=1;i<=NF;i++) $i=toupper(substr($i,1,1)) substr($i,2); print}'
}
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY WATCHDOG ORCHESTRATOR STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo ""
UPTIME_S=$(awk '{print int($1)}' /proc/uptime)
if [[ "$UPTIME_S" -lt "$WATCHDOG_STARTUP_GRACE" ]]; then
warn "Within startup grace — $(format_duration $UPTIME_S) / $(format_duration $WATCHDOG_STARTUP_GRACE)"
else
echo "Past startup grace — $(format_duration $UPTIME_S) uptime"
fi
echo ""
echo "── Sub-scripts ──"
for entry in "${WATCHDOG_ORCHESTRATOR_SCRIPTS[@]}"; do
local_path="$ECOSYSTEM_ROOT/$entry"
label="$(_watchdog_display_name "$entry")"
if [[ -f "$local_path" ]]; then
[[ -x "$local_path" ]] && icon="$ICON_DONE" || icon="$ICON_WARN"
echo " $icon $label — ${local_path##*/}"
else
echo " $ICON_ERROR $label — NOT FOUND: $local_path"
fi
done
echo ""
echo " Schedule: */15 * * * * (every 15 minutes)"
echo " Heartbeat: ${WATCHDOG_ORCHESTRATOR_HEARTBEAT:-true} / every ${WATCHDOG_ORCHESTRATOR_HEARTBEAT_HOURS:-1}hr"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ━━━ Array Check ━━━
# ==============================================================================================
if ! platform_storage_healthy; then
echo "Array not started — skipping watchdog cycle"
exit 0
fi
log "$ICON_DISK Array: $(platform_storage_path) mounted ✅"
# ==============================================================================================
# ━━━ Startup Grace ━━━
# ==============================================================================================
UPTIME_SECONDS=$(awk '{print int($1)}' /proc/uptime)
if [[ "$UPTIME_SECONDS" -lt "$WATCHDOG_STARTUP_GRACE" ]]; then
echo "Startup grace — $(format_duration $UPTIME_SECONDS) / $(format_duration $WATCHDOG_STARTUP_GRACE) — skipping cycle"
exit 0
fi
log "Startup grace: past — uptime $(format_duration $UPTIME_SECONDS)"
# ==============================================================================================
# ━━━ Run Watchdog Cycle ━━━
# ==============================================================================================
CYCLE_START=$(date +%s)
PASS=()
FAIL=()
run_watchdog() {
local name="$1" script="$2"
if [[ ! -f "$script" ]]; then
error "$name — not found: $script"
FAIL+=("$name:missing")
return 1
fi
[[ ! -x "$script" ]] && chmod +x "$script"
local extra_args=()
[[ "$DRY_RUN" == true ]] && extra_args+=("--dry-run")
[[ "$VERBOSE" == true ]] && extra_args+=("--log")
local _ws
_ws=$(date +%s)
log "$ICON_START $name"
if bash "$script" "${extra_args[@]}"; then
log "$ICON_DONE $name — done in $(format_duration $(( $(date +%s) - _ws )))"
PASS+=("$name")
return 0
else
error "$name — non-zero exit ($(format_duration $(( $(date +%s) - _ws ))))"
FAIL+=("$name")
return 1
fi
}
for _entry in "${WATCHDOG_ORCHESTRATOR_SCRIPTS[@]}"; do
run_watchdog "$(_watchdog_display_name "$_entry")" "$ECOSYSTEM_ROOT/$_entry"
done
CYCLE_END=$(date +%s)
DURATION=$(( CYCLE_END - CYCLE_START ))
# ==============================================================================================
# ━━━ Heartbeat ━━━
# ==============================================================================================
if [[ "${WATCHDOG_ORCHESTRATOR_HEARTBEAT:-true}" == true ]]; then
HB_SECONDS=$(( ${WATCHDOG_ORCHESTRATOR_HEARTBEAT_HOURS:-1} * 3600 ))
HB_COUNT_FILE="${STATE_DIR:-/tmp}/watchdog_orch_hb.count"
HB_COUNT=$(cat "$HB_COUNT_FILE" 2>/dev/null || echo 0)
HB_COUNT=$(( HB_COUNT + 1 ))
echo "$HB_COUNT" > "$HB_COUNT_FILE"
# Each cron run = ~60s — use count × 60 as uptime approximation
HB_ELAPSED=$(( HB_COUNT * 60 ))
if [[ "$HB_SECONDS" -gt 0 ]] && (( HB_ELAPSED % HB_SECONDS < 60 )) && [[ "$HB_COUNT" -gt 1 ]]; then
HB_HR=$(( HB_ELAPSED / 3600 ))
warn "♥ watchdog_orchestrator alive — $MY_ID — ~${HB_HR}hr ($(date '+%H:%M:%S'))"
fi
fi
# ==============================================================================================
# ━━━ Summary — only shown on failures or --log ━━━
# ==============================================================================================
if [[ "${#FAIL[@]}" -gt 0 || "$VERBOSE" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY WATCHDOG CYCLE — $MY_ID — $(date '+%H:%M:%S') ━━━━━"
for p in "${PASS[@]}"; do log " $ICON_DONE $p"; done
for f in "${FAIL[@]}"; do error " $ICON_ERROR $f"; done
echo "$ICON_TIME Duration: $(format_duration $DURATION)"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
if [[ "${#FAIL[@]}" -gt 0 ]]; then
notify "Watchdog cycle failure on $(hostname) ($MY_ID) — ${FAIL[*]}" \
"Watchdog Orchestrator" "warning"
fi
fi
@@ -1,305 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= Certificate Monitor ============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# SSL certificate expiry monitoring for all configured domains. Scheduled weekly
# (Sunday 9am). Connects via openssl directly to each domain — not to NPM's API,
# not to any internal check, but to the actual TLS handshake the outside world sees.
#
# Per domain: HEALTHY (> CERT_WARN_DAYS remaining, silent) | WARNING (≤ CERT_WARN_DAYS)
# | CRITICAL (≤ CERT_CRIT_DAYS) | FAILED (could not connect or parse cert).
# Notifications batched by severity — one message lists all WARNING domains, a
# separate message lists all CRITICAL domains. Not one notification per domain.
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Direct openssl, Not an API
# API-based cert checks ask the certificate manager whether the cert is valid.
# openssl checks ask the server what cert it is actually serving. These are not
# the same question and the answers can differ. Catches: cert renewed in NPM but
# server not reloaded (old cert still serving), wrong cert being served to external
# clients, chain issues visible externally but not internally, NPM reporting healthy
# while the outside world sees an expired cert.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Single Instance Lock
# acquire_lock prevents concurrent runs producing duplicate notifications.
#
# Per-Host Domain List
# detect_hosts() aliases HOST*_CERT_MONITOR_DOMAINS → CERT_MONITOR_DOMAINS.
# Each server monitors its own domains only.
#
# Empty Array Guard
# Warns and exits cleanly if CERT_MONITOR_DOMAINS is empty — no silent no-op.
#
# Connection Timeout
# CERT_TIMEOUT caps each openssl connection attempt. One unreachable domain
# does not block the remaining domains.
#
# Notification Validated
# platform_require_cmd confirms openssl and notify script are present before use.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_CERT_MONITOR_DOMAINS
# Domains this host monitors. Each domain and subdomain is a separate entry —
# they have independent certs. Aliased by detect_hosts() → CERT_MONITOR_DOMAINS.
#
# master.conf
#
# CERT_WARN_DAYS
# Days before expiry at which to send a warning notification. (default: 30)
#
# CERT_CRIT_DAYS
# Days before expiry at which to send a critical notification. (default: 7)
#
# CERT_TIMEOUT
# Seconds to wait per domain before declaring FAILED. (default: 10)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# cert_monitor.sh
# Check all configured domains and notify on WARNING, CRITICAL, or FAILED.
# Silent when all domains are healthy.
#
# cert_monitor.sh --dry-run
# Check all domains and show results. No notifications sent regardless of result.
#
# cert_monitor.sh --status
# Show domain list, warning thresholds, and timeout. Then exit.
#
# cert_monitor.sh --log
# Verbose per-domain output during the run.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_GEAR Setup ━━━"
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
# Validate openssl — required for all cert checks
platform_require_cmd \
"$(command -v openssl 2>/dev/null || echo /usr/bin/openssl)" \
"version" "OpenSSL" \
"openssl" || { error "openssl not found — required for certificate checks"; exit 1; }
acquire_lock
# detect_hosts() sets MY_ID and aliases HOST*_CERT_MONITOR_DOMAINS
detect_hosts
# Empty array guard
if [[ ${#CERT_MONITOR_DOMAINS[@]} -eq 0 ]]; then
warn "CERT_MONITOR_DOMAINS is empty for $MY_ID"
warn "Check HOST*_CERT_MONITOR_DOMAINS in host*.conf"
exit 0
fi
info "Domains to check: ${#CERT_MONITOR_DOMAINS[@]}"
log "$ICON_GEAR Config: warn=${CERT_WARN_DAYS}d crit=${CERT_CRIT_DAYS}d timeout=${CERT_TIMEOUT}s"
log "$ICON_GEAR Domains: ${CERT_MONITOR_DOMAINS[*]}"
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — results shown but no notifications sent"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_CERT Domains: ${CERT_MONITOR_DOMAINS[*]}"
echo "$ICON_WARN Warn at: ${CERT_WARN_DAYS} days remaining"
echo "$ICON_ERROR Crit at: ${CERT_CRIT_DAYS} days remaining"
echo "$ICON_TIME Timeout: ${CERT_TIMEOUT}s per domain"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ── CERT CHECK FUNCTION ───────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Connects to domain:443 via openssl, extracts expiry date, calculates days remaining.
# Returns:
# 0 = healthy (> CERT_WARN_DAYS)
# 1 = warning (<= CERT_WARN_DAYS)
# 2 = critical (<= CERT_CRIT_DAYS)
# 3 = failed (could not connect or parse)
check_cert() {
local domain="$1"
local port="${2:-443}"
_CERT_DAYS=""
_CERT_EXPIRY=""
local expiry_str
expiry_str=$(echo | timeout "$CERT_TIMEOUT" openssl s_client \
-connect "${domain}:${port}" \
-servername "$domain" \
2>/dev/null | openssl x509 -noout -enddate 2>/dev/null | cut -d= -f2)
if [[ -z "$expiry_str" ]]; then
error "$ICON_CERT $domain — could not retrieve certificate (unreachable or no TLS)"
return 3
fi
local expiry_epoch
expiry_epoch=$(date -d "$expiry_str" +%s 2>/dev/null)
if [[ -z "$expiry_epoch" ]]; then
error "$ICON_CERT $domain — could not parse expiry date: $expiry_str"
return 3
fi
local now days_remaining expiry_display
now=$(date +%s)
days_remaining=$(( (expiry_epoch - now) / 86400 ))
expiry_display=$(date -d "$expiry_str" '+%Y-%m-%d' 2>/dev/null)
_CERT_DAYS=$days_remaining
_CERT_EXPIRY=$expiry_display
if [[ "$days_remaining" -le "$CERT_CRIT_DAYS" ]]; then
error "$ICON_CERT $domain — CRITICAL: ${days_remaining} days remaining (expires $expiry_display)"
return 2
elif [[ "$days_remaining" -le "$CERT_WARN_DAYS" ]]; then
warn "$ICON_CERT $domain — WARNING: ${days_remaining} days remaining (expires $expiry_display)"
return 1
else
log "$ICON_CERT $domain — OK: ${days_remaining} days remaining (expires $expiry_display)"
return 0
fi
}
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CERT Certificate Monitor — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
echo "$ICON_HOST $MY_ID ($LOCAL_SERVER_NAME)"
info "Warn threshold: ${CERT_WARN_DAYS} days"
info "Crit threshold: ${CERT_CRIT_DAYS} days"
echo ""
START=$(date +%s)
HEALTHY=()
WARNING=()
CRITICAL=()
FAILED=()
declare -A DOMAIN_STATUS DOMAIN_DAYS DOMAIN_EXPIRY
for domain in "${CERT_MONITOR_DOMAINS[@]}"; do
[[ -z "$domain" ]] && continue
check_cert "$domain"
result=$?
DOMAIN_DAYS["$domain"]="${_CERT_DAYS:-}"
DOMAIN_EXPIRY["$domain"]="${_CERT_EXPIRY:-}"
case $result in
0) HEALTHY+=("$domain"); DOMAIN_STATUS["$domain"]="OK" ;;
1) WARNING+=("$domain"); DOMAIN_STATUS["$domain"]="WARN" ;;
2) CRITICAL+=("$domain"); DOMAIN_STATUS["$domain"]="CRIT" ;;
3) FAILED+=("$domain"); DOMAIN_STATUS["$domain"]="FAIL" ;;
esac
done
END=$(date +%s)
# ── Send notifications — batched per severity ─────────────────────────────────────────────────
if [[ "$DRY_RUN" == false ]]; then
[[ ${#CRITICAL[@]} -gt 0 ]] && \
notify "Certificate CRITICAL on $(hostname) — expiring within ${CERT_CRIT_DAYS} days: ${CRITICAL[*]}" \
"Certificate Monitor" "warning"
[[ ${#WARNING[@]} -gt 0 ]] && \
notify "Certificate WARNING on $(hostname) — expiring within ${CERT_WARN_DAYS} days: ${WARNING[*]}" \
"Certificate Monitor" "warning"
[[ ${#FAILED[@]} -gt 0 ]] && \
notify "Certificate check FAILED on $(hostname) — could not reach: ${FAILED[*]}" \
"Certificate Monitor" "warning"
fi
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY CERTIFICATE MONITOR SUMMARY ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_TIME Duration: $(format_duration $(( END - START )))"
echo ""
echo " $ICON_SUCCESS Healthy: ${#HEALTHY[@]}"
[[ ${#WARNING[@]} -gt 0 ]] && warn "Warning: ${#WARNING[@]} — renewal recommended"
[[ ${#CRITICAL[@]} -gt 0 ]] && echo "$ICON_ERROR Critical: ${#CRITICAL[@]} — ACTION REQUIRED"
[[ ${#FAILED[@]} -gt 0 ]] && echo "$ICON_ERROR Failed: ${#FAILED[@]} — unreachable"
echo ""
# Per-domain results — only show problems, healthy ones stay in log()
for domain in "${CERT_MONITOR_DOMAINS[@]}"; do
[[ -z "$domain" ]] && continue
case "${DOMAIN_STATUS[$domain]:-UNKN}" in
OK) log " $ICON_SUCCESS $domain — healthy" ;;
WARN) warn " $ICON_WARN $domain — warning" ;;
CRIT) echo " $ICON_ERROR $domain — CRITICAL" ;;
FAIL) echo " $ICON_ERROR $domain — unreachable" ;;
esac
done
echo ""
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no notifications sent"
elif [[ ${#CRITICAL[@]} -gt 0 || ${#FAILED[@]} -gt 0 ]]; then
echo "$ICON_ERROR Status: ACTION REQUIRED"
elif [[ ${#WARNING[@]} -gt 0 ]]; then
warn "Status: WARNINGS — renewal recommended"
else
echo "$ICON_DONE Status: all ${#HEALTHY[@]} certs healthy ✅"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
# ── Write JSON status cache ───────────────────────────────────────────────────
_CERT_CACHE_FILE="$SCRIPTS_DIR/State_Files/cert_status.json"
{
printf '{"checked_at":%d,"host":"%s","warn_days":%d,"crit_days":%d,"dry_run":%s,"domains":[\n' \
"$(date +%s)" "$MY_ID" "$CERT_WARN_DAYS" "$CERT_CRIT_DAYS" \
"$([[ $DRY_RUN == true ]] && echo true || echo false)"
_first=true
for _d in "${CERT_MONITOR_DOMAINS[@]}"; do
[[ -z "$_d" ]] && continue
[[ "$_first" != true ]] && printf ','
_first=false
_days="${DOMAIN_DAYS[$_d]:-null}"
_exp="${DOMAIN_EXPIRY[$_d]:-}"
printf '{"domain":"%s","status":"%s","days":%s,"expires":"%s"}\n' \
"$_d" "${DOMAIN_STATUS[$_d]:-UNKN}" "$_days" "$_exp"
done
printf ']}\n'
} > "$_CERT_CACHE_FILE" 2>/dev/null
[[ ${#CRITICAL[@]} -gt 0 || ${#FAILED[@]} -gt 0 ]] && exit 1
exit 0
@@ -1,307 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= Certificate Monitor ============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# SSL certificate expiry monitoring for all configured domains. Scheduled weekly
# (Sunday 9am). Connects via openssl directly to each domain — not to NPM's API,
# not to any internal check, but to the actual TLS handshake the outside world sees.
#
# Per domain: HEALTHY (> CERT_WARN_DAYS remaining, silent) | WARNING (≤ CERT_WARN_DAYS)
# | CRITICAL (≤ CERT_CRIT_DAYS) | FAILED (could not connect or parse cert).
# Notifications batched by severity — one message lists all WARNING domains, a
# separate message lists all CRITICAL domains. Not one notification per domain.
#
# ==============================================================================================
# DESIGN PRINCIPLES
# ==============================================================================================
#
# Direct openssl, Not an API
# API-based cert checks ask the certificate manager whether the cert is valid.
# openssl checks ask the server what cert it is actually serving. These are not
# the same question and the answers can differ. Catches: cert renewed in NPM but
# server not reloaded (old cert still serving), wrong cert being served to external
# clients, chain issues visible externally but not internally, NPM reporting healthy
# while the outside world sees an expired cert.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Single Instance Lock
# acquire_lock prevents concurrent runs producing duplicate notifications.
#
# Per-Host Domain List
# detect_hosts() aliases HOST*_CERT_MONITOR_DOMAINS → CERT_MONITOR_DOMAINS.
# Each server monitors its own domains only.
#
# Empty Array Guard
# Warns and exits cleanly if CERT_MONITOR_DOMAINS is empty — no silent no-op.
#
# Connection Timeout
# CERT_TIMEOUT caps each openssl connection attempt. One unreachable domain
# does not block the remaining domains.
#
# Notification Validated
# platform_require_cmd confirms openssl and notify script are present before use.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_CERT_MONITOR_DOMAINS
# Domains this host monitors. Each domain and subdomain is a separate entry —
# they have independent certs. Aliased by detect_hosts() → CERT_MONITOR_DOMAINS.
#
# master.conf
#
# CERT_WARN_DAYS
# Days before expiry at which to send a warning notification. (default: 30)
#
# CERT_CRIT_DAYS
# Days before expiry at which to send a critical notification. (default: 7)
#
# CERT_TIMEOUT
# Seconds to wait per domain before declaring FAILED. (default: 10)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# cert_monitor.sh
# Check all configured domains and notify on WARNING, CRITICAL, or FAILED.
# Silent when all domains are healthy.
#
# cert_monitor.sh --dry-run
# Check all domains and show results. No notifications sent regardless of result.
#
# cert_monitor.sh --status
# Show domain list, warning thresholds, and timeout. Then exit.
#
# cert_monitor.sh --log
# Include healthy domains in per-domain output with expiry date and days remaining.
# Problems (WARN/CRIT/FAIL) always show with their details regardless of this flag.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_GEAR Setup ━━━"
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
# Validate openssl — required for all cert checks
platform_require_cmd \
"$(command -v openssl 2>/dev/null || echo /usr/bin/openssl)" \
"version" "OpenSSL" \
"openssl" || { error "openssl not found — required for certificate checks"; exit 1; }
acquire_lock
# detect_hosts() sets MY_ID and aliases HOST*_CERT_MONITOR_DOMAINS
detect_hosts
# Empty array guard
if [[ ${#CERT_MONITOR_DOMAINS[@]} -eq 0 ]]; then
warn "CERT_MONITOR_DOMAINS is empty for $MY_ID"
warn "Check HOST*_CERT_MONITOR_DOMAINS in host*.conf"
exit 0
fi
info "Domains to check: ${#CERT_MONITOR_DOMAINS[@]}"
log "$ICON_GEAR Config: warn=${CERT_WARN_DAYS}d crit=${CERT_CRIT_DAYS}d timeout=${CERT_TIMEOUT}s"
log "$ICON_GEAR Domains: ${CERT_MONITOR_DOMAINS[*]}"
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — results shown but no notifications sent"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_CERT Domains: ${CERT_MONITOR_DOMAINS[*]}"
echo "$ICON_WARN Warn at: ${CERT_WARN_DAYS} days remaining"
echo "$ICON_ERROR Crit at: ${CERT_CRIT_DAYS} days remaining"
echo "$ICON_TIME Timeout: ${CERT_TIMEOUT}s per domain"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ── CERT CHECK FUNCTION ───────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Connects to domain:443 via openssl, extracts expiry date, calculates days remaining.
# Returns:
# 0 = healthy (> CERT_WARN_DAYS)
# 1 = warning (<= CERT_WARN_DAYS)
# 2 = critical (<= CERT_CRIT_DAYS)
# 3 = failed (could not connect or parse)
check_cert() {
local domain="$1"
local port="${2:-443}"
_CERT_DAYS=""
_CERT_EXPIRY=""
local expiry_str
expiry_str=$(echo | timeout "$CERT_TIMEOUT" openssl s_client \
-connect "${domain}:${port}" \
-servername "$domain" \
2>/dev/null | openssl x509 -noout -enddate 2>/dev/null | cut -d= -f2)
if [[ -z "$expiry_str" ]]; then
error "$ICON_CERT $domain — could not retrieve certificate (unreachable or no TLS)"
return 3
fi
local expiry_epoch
expiry_epoch=$(date -d "$expiry_str" +%s 2>/dev/null)
if [[ -z "$expiry_epoch" ]]; then
error "$ICON_CERT $domain — could not parse expiry date: $expiry_str"
return 3
fi
local now days_remaining expiry_display
now=$(date +%s)
days_remaining=$(( (expiry_epoch - now) / 86400 ))
expiry_display=$(date -d "$expiry_str" '+%Y-%m-%d' 2>/dev/null)
_CERT_DAYS=$days_remaining
_CERT_EXPIRY=$expiry_display
if [[ "$days_remaining" -le "$CERT_CRIT_DAYS" ]]; then
error "$ICON_CERT $domain — CRITICAL: ${days_remaining} days remaining (expires $expiry_display)"
return 2
elif [[ "$days_remaining" -le "$CERT_WARN_DAYS" ]]; then
warn "$ICON_CERT $domain — WARNING: ${days_remaining} days remaining (expires $expiry_display)"
return 1
else
log "$ICON_CERT $domain — OK: ${days_remaining} days remaining (expires $expiry_display)"
return 0
fi
}
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CERT Certificate Monitor — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
echo "$ICON_HOST $MY_ID ($LOCAL_SERVER_NAME)"
info "Warn threshold: ${CERT_WARN_DAYS} days"
info "Crit threshold: ${CERT_CRIT_DAYS} days"
echo ""
START=$(date +%s)
HEALTHY=()
WARNING=()
CRITICAL=()
FAILED=()
declare -A DOMAIN_STATUS DOMAIN_DAYS DOMAIN_EXPIRY
for domain in "${CERT_MONITOR_DOMAINS[@]}"; do
[[ -z "$domain" ]] && continue
check_cert "$domain"
result=$?
DOMAIN_DAYS["$domain"]="${_CERT_DAYS:-}"
DOMAIN_EXPIRY["$domain"]="${_CERT_EXPIRY:-}"
case $result in
0) HEALTHY+=("$domain"); DOMAIN_STATUS["$domain"]="OK" ;;
1) WARNING+=("$domain"); DOMAIN_STATUS["$domain"]="WARN" ;;
2) CRITICAL+=("$domain"); DOMAIN_STATUS["$domain"]="CRIT" ;;
3) FAILED+=("$domain"); DOMAIN_STATUS["$domain"]="FAIL" ;;
esac
done
END=$(date +%s)
# ── Send notifications — batched per severity ─────────────────────────────────────────────────
if [[ "$DRY_RUN" == false ]]; then
[[ ${#CRITICAL[@]} -gt 0 ]] && \
notify "Certificate CRITICAL on $(hostname) — expiring within ${CERT_CRIT_DAYS} days: ${CRITICAL[*]}" \
"Certificate Monitor" "warning"
[[ ${#WARNING[@]} -gt 0 ]] && \
notify "Certificate WARNING on $(hostname) — expiring within ${CERT_WARN_DAYS} days: ${WARNING[*]}" \
"Certificate Monitor" "warning"
[[ ${#FAILED[@]} -gt 0 ]] && \
notify "Certificate check FAILED on $(hostname) — could not reach: ${FAILED[*]}" \
"Certificate Monitor" "warning"
fi
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY CERTIFICATE MONITOR SUMMARY ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_TIME Duration: $(format_duration $(( END - START )))"
echo ""
echo " $ICON_SUCCESS Healthy: ${#HEALTHY[@]}"
[[ ${#WARNING[@]} -gt 0 ]] && warn "Warning: ${#WARNING[@]} — renewal recommended"
[[ ${#CRITICAL[@]} -gt 0 ]] && echo "$ICON_ERROR Critical: ${#CRITICAL[@]} — ACTION REQUIRED"
[[ ${#FAILED[@]} -gt 0 ]] && echo "$ICON_ERROR Failed: ${#FAILED[@]} — unreachable"
echo ""
# Per-domain results — problems always shown with days remaining; healthy only with --log
for domain in "${CERT_MONITOR_DOMAINS[@]}"; do
[[ -z "$domain" ]] && continue
local _days="${DOMAIN_DAYS[$domain]:-?}" _exp="${DOMAIN_EXPIRY[$domain]:-unknown}"
case "${DOMAIN_STATUS[$domain]:-UNKN}" in
OK) log " $ICON_SUCCESS $domain — healthy (${_days}d, expires ${_exp})" ;;
WARN) warn " $ICON_WARN $domain — warning (${_days}d, expires ${_exp})" ;;
CRIT) echo " $ICON_ERROR $domain — CRITICAL (${_days}d, expires ${_exp})" ;;
FAIL) echo " $ICON_ERROR $domain — unreachable" ;;
esac
done
echo ""
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no notifications sent"
elif [[ ${#CRITICAL[@]} -gt 0 || ${#FAILED[@]} -gt 0 ]]; then
echo "$ICON_ERROR Status: ACTION REQUIRED"
elif [[ ${#WARNING[@]} -gt 0 ]]; then
warn "Status: WARNINGS — renewal recommended"
else
echo "$ICON_DONE Status: all ${#HEALTHY[@]} certs healthy ✅"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
# ── Write JSON status cache ───────────────────────────────────────────────────
_CERT_CACHE_FILE="$SCRIPTS_DIR/State_Files/cert_status.json"
{
printf '{"checked_at":%d,"host":"%s","warn_days":%d,"crit_days":%d,"dry_run":%s,"domains":[\n' \
"$(date +%s)" "$MY_ID" "$CERT_WARN_DAYS" "$CERT_CRIT_DAYS" \
"$([[ $DRY_RUN == true ]] && echo true || echo false)"
_first=true
for _d in "${CERT_MONITOR_DOMAINS[@]}"; do
[[ -z "$_d" ]] && continue
[[ "$_first" != true ]] && printf ','
_first=false
_days="${DOMAIN_DAYS[$_d]:-null}"
_exp="${DOMAIN_EXPIRY[$_d]:-}"
printf '{"domain":"%s","status":"%s","days":%s,"expires":"%s"}\n' \
"$_d" "${DOMAIN_STATUS[$_d]:-UNKN}" "$_days" "$_exp"
done
printf ']}\n'
} > "$_CERT_CACHE_FILE" 2>/dev/null
[[ ${#CRITICAL[@]} -gt 0 || ${#FAILED[@]} -gt 0 ]] && exit 1
exit 0
@@ -1,343 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= Partnership Transfer ============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Transfers ownership from the current owner to the current mirror. After
# transfer the roles are swapped: what was the mirror becomes the new owner,
# and what was the owner becomes the new mirror.
#
# No containers are moved — only config and WebUI targets are updated. Both
# servers remain in the partnership; the sync direction reverses on the next
# fallback.sh / critical_sync_maintenance.sh cycle.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# OWNER ONLY — mirror cannot run this script.
#
# Step 1: Health Verification — both servers pass N consecutive health checks
# Step 2: Pre-transfer Sync — final sync in current direction (owner → mirror)
# Step 3: Reconfigure WebUIs — new owner WebUIs → localhost
# new mirror WebUIs → new owner IP
# Step 4: Flip Ownership — update PARTNERSHIP_OWNER_HOST in master.conf
# on both servers
# Step 5: Write State — ACTIVE written locally and pushed to new mirror
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Owner-Only Enforcement
# The script reads MY_ID from detect_hosts() and exits immediately if it is
# not the current PARTNERSHIP_OWNER_HOST. The mirror cannot run a transfer.
#
# Explicit Confirmation String
# Requires the exact passphrase from PARTNERSHIP_TRANSFER_CONFIRM via
# --confirm=<value>. Without a matching string the transfer is cancelled
# before any steps execute. Prevents accidental ownership changes.
#
# Active Partnership Guard
# Reads the local state file and exits if the current state is INACTIVE.
# A transfer without an active partnership has no defined outcome.
#
# Dual Health Verification
# Both servers must pass PARTNERSHIP_TRANSFER_STRIKES consecutive health
# checks before proceeding. A single failure resets the strike counter.
# After PARTNERSHIP_TRANSFER_MAX_ATTEMPTS total attempts the transfer aborts.
#
# Pre-transfer Final Sync
# A full sync in the current direction (owner → mirror) runs immediately
# before roles flip. Ensures the mirror is current before it becomes the owner.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# master.conf
#
# PARTNERSHIP_OWNER_HOST
# Current owner host ID (e.g. "HOST1"). Updated on both servers after transfer.
#
# PARTNERSHIP_TRANSFER_CONFIRM
# Exact string required to confirm transfer (default: "i-understand-this-transfers-ownership").
# Pass via --confirm=<value>.
#
# PARTNERSHIP_TRANSFER_STRIKES
# Consecutive health checks both servers must pass before transfer proceeds (default: 3).
#
# PARTNERSHIP_TRANSFER_MAX_ATTEMPTS
# Max health check attempts before giving up (default: 20).
#
# CRITICAL_SYNC_SHARES
# Array of "path|profile" or "path" entries for do_final_sync().
#
# PARTNERSHIP_AUTH_WEBUIS
# Array of "ContainerName|WebUIPort" entries reconfigured during transfer.
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# Partnership/partnership_transfer.sh --confirm=i-understand-this-transfers-ownership
# Full transfer — owner detected automatically.
#
# Partnership/partnership_transfer.sh --confirm=i-understand-this-transfers-ownership --dry-run
# Preview all steps without executing. Confirmation check is skipped in dry-run mode.
#
# Partnership/partnership_transfer.sh --confirm=i-understand-this-transfers-ownership --log
# Verbose per-step output.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
SCRIPTS_ROOT="$SCRIPT_DIR/.."
SSH_TIMEOUT=15
source "$SCRIPTS_ROOT/load_config.sh"
# ── Parse flags ───────────────────────────────────────────────────────────────────────────────
TRANSFER_CONFIRM_INPUT=""
FILTERED_ARGS=()
for arg in "$@"; do
case "$arg" in
--confirm=*) TRANSFER_CONFIRM_INPUT="${arg#--confirm=}" ;;
*) FILTERED_ARGS+=("$arg") ;;
esac
done
parse_args "${FILTERED_ARGS[@]}"
# ── Source partnership_manager.sh for shared helpers ──────────────────────────────────────────
# PARTNERSHIP_LIB_MODE=1 skips mode dispatch — functions are defined, nothing is executed.
PARTNERSHIP_LIB_MODE=1 source "$SCRIPT_DIR/partnership_manager.sh"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
[[ "$EUID" -ne 0 ]] && { error "Must be run as root"; exit 1; }
if ! command -v docker &>/dev/null; then
error "Docker command not found"
exit 1
fi
platform_require_cmd \
"/usr/local/emhttp/plugins/dynamix/scripts/notify" \
"" "" \
"unRAID notify script" || warn "unRAID notify script not found — native notifications disabled"
detect_hosts
OWNER_ID="${PARTNERSHIP_OWNER_HOST:-HOST1}"
MIRROR_ID=$( [[ "$OWNER_ID" == "HOST1" ]] && echo "HOST2" || echo "HOST1" )
OWNER="${!OWNER_ID}"
MIRROR="${!MIRROR_ID}"
MIRROR_SSH_KEY="$SSH_KEY"
OWNER_SSH_KEY="$SSH_KEY"
AM_OWNER=false
AM_MIRROR=false
[[ "$MY_ID" == "$OWNER_ID" ]] && AM_OWNER=true
[[ "$MY_ID" == "$MIRROR_ID" ]] && AM_MIRROR=true
LOCAL_STATE_FILE="${STATE_DIR:-/boot/config}/partnership_${LOCAL_SERVER_NAME}.db"
REMOTE_STATE_FILE="${STATE_DIR:-/boot/config}/partnership_${REMOTE_SERVER_NAME}.db"
acquire_lock "strict"
# ==============================================================================================
# ━━━ Preflight ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_FALLBACK Partnership Transfer — $MY_ID ($LOCAL_SERVER_NAME) — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
echo ""
if [[ "$AM_MIRROR" == true ]]; then
error "Only the owner ($OWNER / $OWNER_ID) can run --transfer"
error "Run from $OWNER, or use --offboard and re-onboard with roles swapped"
exit 1
fi
if [[ -f "$LOCAL_STATE_FILE" ]]; then
CURRENT_STATE=$(read_state_file "$LOCAL_STATE_FILE" "state")
if [[ "$CURRENT_STATE" == "INACTIVE" ]]; then
error "No active partnership — transfer requires an active partnership"
error "If roles are already correct, check PARTNERSHIP_OWNER_HOST in master.conf"
exit 1
fi
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no permanent changes will be made"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo "⚠️ WARNING — OWNERSHIP TRANSFER"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo " Current owner: $OWNER_ID ($OWNER)"
echo " Current mirror: $MIRROR_ID ($MIRROR)"
echo ""
echo " After transfer:"
echo " New owner: $MIRROR_ID ($MIRROR)"
echo " New mirror: $OWNER_ID ($OWNER)"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
# ==============================================================================================
# ━━━ Confirmation ━━━
# ==============================================================================================
if [[ "$DRY_RUN" == false ]]; then
if [[ -z "$TRANSFER_CONFIRM_INPUT" ]]; then
echo ""
echo "To proceed, pass exactly:"
echo " --confirm=${PARTNERSHIP_TRANSFER_CONFIRM}"
echo ""
error "Transfer cancelled — confirmation required"
exit 1
fi
if [[ "$TRANSFER_CONFIRM_INPUT" != "$PARTNERSHIP_TRANSFER_CONFIRM" ]]; then
error "Confirmation string does not match — transfer cancelled"
exit 1
fi
log "Confirmation accepted"
else
warn "DRY RUN — confirmation check skipped"
fi
# ==============================================================================================
# ━━━ Step 1: Health Verification ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SHIELD Step 1: Health Verification ━━━"
log "Both servers must pass ${PARTNERSHIP_TRANSFER_STRIKES} consecutive health checks"
STRIKES=0
ATTEMPTS=0
MAX_ATTEMPTS="${PARTNERSHIP_TRANSFER_MAX_ATTEMPTS:-20}"
while [[ "$STRIKES" -lt "$PARTNERSHIP_TRANSFER_STRIKES" ]]; do
(( ATTEMPTS++ ))
if [[ "$ATTEMPTS" -gt "$MAX_ATTEMPTS" ]]; then
error "Health checks failed after $MAX_ATTEMPTS attempts — servers not stable"
error "Transfer cancelled — try again when both servers are healthy"
exit 1
fi
if check_both_healthy; then
(( STRIKES++ ))
log "Health check passed ($STRIKES/${PARTNERSHIP_TRANSFER_STRIKES})"
[[ "$STRIKES" -lt "$PARTNERSHIP_TRANSFER_STRIKES" ]] && sleep 10
else
warn "Health check failed — resetting (attempt $ATTEMPTS/$MAX_ATTEMPTS)"
STRIKES=0
sleep 30
fi
done
warn "Both servers healthy — proceeding ✅"
# Resolve IPs after health checks confirm reachability
MIRROR_IP=$(resolve_tailscale_ip "$MIRROR")
OWNER_IP=$(resolve_tailscale_ip "$OWNER")
[[ -z "$MIRROR_IP" ]] && { error "Cannot resolve mirror Tailscale IP"; exit 1; }
# Compute post-transfer roles
NEW_OWNER_ID="$MIRROR_ID"
NEW_MIRROR_ID="$OWNER_ID"
NEW_OWNER="$MIRROR"
NEW_MIRROR="$OWNER"
NEW_OWNER_IP="$MIRROR_IP"
# ==============================================================================================
# ━━━ Step 2: Pre-transfer Sync ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SYNC Step 2: Pre-transfer Sync ━━━"
do_final_sync
# ==============================================================================================
# ━━━ Step 3: Reconfigure WebUIs ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CONTAINERS Step 3: Reconfigure WebUIs ━━━"
WEBUI_FAILURES=0
# New owner (current mirror, HOST2) WebUIs → localhost — it now manages itself directly
log "New owner ($NEW_OWNER) WebUIs → localhost"
for entry in "${PARTNERSHIP_AUTH_WEBUIS[@]}"; do
[[ -z "$entry" ]] && continue
container="${entry%%|*}"
port="${entry##*|}"
reconfigure_webui "$container" "$port" "localhost" \
"$SSH_KEY" "$MIRROR_IP" "$MIRROR" || (( WEBUI_FAILURES++ ))
done
# New mirror (us, HOST1) WebUIs → new owner IP — defers to new owner going forward
log "New mirror ($NEW_MIRROR) WebUIs → $NEW_OWNER_IP"
reconfigure_local_webuis "$NEW_OWNER_IP"
WEBUI_RC=$?
[[ "$WEBUI_RC" -gt 0 ]] && (( WEBUI_FAILURES += WEBUI_RC ))
# ==============================================================================================
# ━━━ Step 4: Flip Ownership in master.conf ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_GEAR Step 4: Flip Ownership ━━━"
if [[ "$DRY_RUN" == false ]]; then
update_master_conf "PARTNERSHIP_OWNER_HOST" "\"$NEW_OWNER_ID\""
# Push updated master.conf to new owner so both servers agree immediately.
# master.conf is shared — host-specific credentials live in host*.conf.
_REMOTE_SD=$(ssh -i "$SSH_KEY" -o ConnectTimeout=5 -o StrictHostKeyChecking=no \
"root@${MIRROR_IP}" \
"grep -m1 '^SCRIPTS_DIR' /boot/config/plugins/varaverk/varaverk.cfg 2>/dev/null | cut -d= -f2 | tr -d '\"'" \
2>/dev/null | tr -d '[:space:]')
_REMOTE_SD="${_REMOTE_SD:-/boot/config/plugins/varaverk}"
scp -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" \
-o StrictHostKeyChecking=no \
"$SCRIPTS_ROOT/Configurations/master.conf" \
"root@${MIRROR_IP}:${_REMOTE_SD}/Configurations/master.conf" 2>/dev/null && \
log "master.conf pushed to $NEW_OWNER ✅" || \
error "Failed to push master.conf to $NEW_OWNER — set PARTNERSHIP_OWNER_HOST=\"$NEW_OWNER_ID\" manually"
else
warn "DRY RUN — would set PARTNERSHIP_OWNER_HOST=$NEW_OWNER_ID on both servers"
fi
# ==============================================================================================
# ━━━ Step 5: Write State ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SYNC Step 5: Write State ━━━"
NOW=$(date '+%Y-%m-%d %H:%M:%S')
write_state_file "$LOCAL_STATE_FILE" "ACTIVE" "$NOW" "" "$LOCAL_SERVER_NAME" "transfer"
push_state_to_remote "$LOCAL_STATE_FILE" "$MIRROR_IP" "$SSH_KEY"
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY TRANSFER SUMMARY ━━━━━"
echo " New owner: $NEW_OWNER_ID ($NEW_OWNER — $NEW_OWNER_IP)"
echo " New mirror: $NEW_MIRROR_ID ($NEW_MIRROR)"
echo " WebUI failures: $WEBUI_FAILURES"
echo " Sync direction: $NEW_OWNER → $NEW_MIRROR (next cycle)"
echo " Ownership: PARTNERSHIP_OWNER_HOST=$NEW_OWNER_ID"
echo ""
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no changes made"
else
warn "$ICON_DONE DONE — ownership transferred to $NEW_OWNER_ID ($NEW_OWNER) ✅"
echo "fallback.sh and critical_sync_maintenance.sh will adapt on next cycle"
echo "No containers were moved — only config and WebUI targets updated"
[[ "$WEBUI_FAILURES" -gt 0 ]] && \
warn "$WEBUI_FAILURES WebUI(s) failed — check templates manually"
notify "Partnership ownership transferred — new owner: $NEW_OWNER ($NEW_OWNER_ID)" \
"Partnership" "normal"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
@@ -1,338 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= Partnership Transfer ============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Transfers ownership from the current owner to the current mirror. After
# transfer the roles are swapped: what was the mirror becomes the new owner,
# and what was the owner becomes the new mirror.
#
# No containers are moved — only config and WebUI targets are updated. Both
# servers remain in the partnership; the sync direction reverses on the next
# fallback.sh / critical_sync_maintenance.sh cycle.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# OWNER ONLY — mirror cannot run this script.
#
# Step 1: Health Verification — both servers pass N consecutive health checks
# Step 2: Pre-transfer Sync — final sync in current direction (owner → mirror)
# Step 3: Reconfigure WebUIs — new owner WebUIs → localhost
# new mirror WebUIs → new owner IP
# Step 4: Flip Ownership — update PARTNERSHIP_OWNER_HOST in master.conf
# on both servers
# Step 5: Write State — ACTIVE written locally and pushed to new mirror
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Owner-Only Enforcement
# The script reads MY_ID from detect_hosts() and exits immediately if it is
# not the current PARTNERSHIP_OWNER_HOST. The mirror cannot run a transfer.
#
# Explicit Confirmation String
# Requires the exact passphrase from PARTNERSHIP_TRANSFER_CONFIRM via
# --confirm=<value>. Without a matching string the transfer is cancelled
# before any steps execute. Prevents accidental ownership changes.
#
# Active Partnership Guard
# Reads the local state file and exits if the current state is INACTIVE.
# A transfer without an active partnership has no defined outcome.
#
# Dual Health Verification
# Both servers must pass PARTNERSHIP_TRANSFER_STRIKES consecutive health
# checks before proceeding. A single failure resets the strike counter.
# After PARTNERSHIP_TRANSFER_MAX_ATTEMPTS total attempts the transfer aborts.
#
# Pre-transfer Final Sync
# A full sync in the current direction (owner → mirror) runs immediately
# before roles flip. Ensures the mirror is current before it becomes the owner.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# master.conf
#
# PARTNERSHIP_OWNER_HOST
# Current owner host ID (e.g. "HOST1"). Updated on both servers after transfer.
#
# PARTNERSHIP_TRANSFER_CONFIRM
# Exact string required to confirm transfer (default: "i-understand-this-transfers-ownership").
# Pass via --confirm=<value>.
#
# PARTNERSHIP_TRANSFER_STRIKES
# Consecutive health checks both servers must pass before transfer proceeds (default: 3).
#
# PARTNERSHIP_TRANSFER_MAX_ATTEMPTS
# Max health check attempts before giving up (default: 20).
#
# CRITICAL_SYNC_SHARES
# Array of "path|profile" or "path" entries for do_final_sync().
#
# PARTNERSHIP_AUTH_WEBUIS
# Array of "ContainerName|WebUIPort" entries reconfigured during transfer.
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# Partnership/partnership_transfer.sh --confirm=i-understand-this-transfers-ownership
# Full transfer — owner detected automatically.
#
# Partnership/partnership_transfer.sh --confirm=i-understand-this-transfers-ownership --dry-run
# Preview all steps without executing. Confirmation check is skipped in dry-run mode.
#
# Partnership/partnership_transfer.sh --confirm=i-understand-this-transfers-ownership --log
# Verbose per-step output.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
SCRIPTS_ROOT="$SCRIPT_DIR/.."
SSH_TIMEOUT=15
source "$SCRIPTS_ROOT/load_config.sh"
# ── Parse flags ───────────────────────────────────────────────────────────────────────────────
TRANSFER_CONFIRM_INPUT=""
FILTERED_ARGS=()
for arg in "$@"; do
case "$arg" in
--confirm=*) TRANSFER_CONFIRM_INPUT="${arg#--confirm=}" ;;
*) FILTERED_ARGS+=("$arg") ;;
esac
done
parse_args "${FILTERED_ARGS[@]}"
# ── Source partnership_manager.sh for shared helpers ──────────────────────────────────────────
# PARTNERSHIP_LIB_MODE=1 skips mode dispatch — functions are defined, nothing is executed.
PARTNERSHIP_LIB_MODE=1 source "$SCRIPT_DIR/partnership_manager.sh"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
[[ "$EUID" -ne 0 ]] && { error "Must be run as root"; exit 1; }
if ! command -v docker &>/dev/null; then
error "Docker command not found"
exit 1
fi
detect_hosts
OWNER_ID="${PARTNERSHIP_OWNER_HOST:-HOST1}"
MIRROR_ID=$( [[ "$OWNER_ID" == "HOST1" ]] && echo "HOST2" || echo "HOST1" )
OWNER="${!OWNER_ID}"
MIRROR="${!MIRROR_ID}"
MIRROR_SSH_KEY="$SSH_KEY"
OWNER_SSH_KEY="$SSH_KEY"
AM_OWNER=false
AM_MIRROR=false
[[ "$MY_ID" == "$OWNER_ID" ]] && AM_OWNER=true
[[ "$MY_ID" == "$MIRROR_ID" ]] && AM_MIRROR=true
LOCAL_STATE_FILE="${STATE_DIR:-/boot/config}/partnership_${LOCAL_SERVER_NAME}.db"
REMOTE_STATE_FILE="${STATE_DIR:-/boot/config}/partnership_${REMOTE_SERVER_NAME}.db"
acquire_lock "strict"
# ==============================================================================================
# ━━━ Preflight ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_FALLBACK Partnership Transfer — $MY_ID ($LOCAL_SERVER_NAME) — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
echo ""
if [[ "$AM_MIRROR" == true ]]; then
error "Only the owner ($OWNER / $OWNER_ID) can run --transfer"
error "Run from $OWNER, or use --offboard and re-onboard with roles swapped"
exit 1
fi
if [[ -f "$LOCAL_STATE_FILE" ]]; then
CURRENT_STATE=$(read_state_file "$LOCAL_STATE_FILE" "state")
if [[ "$CURRENT_STATE" == "INACTIVE" ]]; then
error "No active partnership — transfer requires an active partnership"
error "If roles are already correct, check PARTNERSHIP_OWNER_HOST in master.conf"
exit 1
fi
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no permanent changes will be made"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo "⚠️ WARNING — OWNERSHIP TRANSFER"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo " Current owner: $OWNER_ID ($OWNER)"
echo " Current mirror: $MIRROR_ID ($MIRROR)"
echo ""
echo " After transfer:"
echo " New owner: $MIRROR_ID ($MIRROR)"
echo " New mirror: $OWNER_ID ($OWNER)"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
# ==============================================================================================
# ━━━ Confirmation ━━━
# ==============================================================================================
if [[ "$DRY_RUN" == false ]]; then
if [[ -z "$TRANSFER_CONFIRM_INPUT" ]]; then
echo ""
echo "To proceed, pass exactly:"
echo " --confirm=${PARTNERSHIP_TRANSFER_CONFIRM}"
echo ""
error "Transfer cancelled — confirmation required"
exit 1
fi
if [[ "$TRANSFER_CONFIRM_INPUT" != "$PARTNERSHIP_TRANSFER_CONFIRM" ]]; then
error "Confirmation string does not match — transfer cancelled"
exit 1
fi
log "Confirmation accepted"
else
warn "DRY RUN — confirmation check skipped"
fi
# ==============================================================================================
# ━━━ Step 1: Health Verification ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SHIELD Step 1: Health Verification ━━━"
log "Both servers must pass ${PARTNERSHIP_TRANSFER_STRIKES} consecutive health checks"
STRIKES=0
ATTEMPTS=0
MAX_ATTEMPTS="${PARTNERSHIP_TRANSFER_MAX_ATTEMPTS:-20}"
while [[ "$STRIKES" -lt "$PARTNERSHIP_TRANSFER_STRIKES" ]]; do
(( ATTEMPTS++ ))
if [[ "$ATTEMPTS" -gt "$MAX_ATTEMPTS" ]]; then
error "Health checks failed after $MAX_ATTEMPTS attempts — servers not stable"
error "Transfer cancelled — try again when both servers are healthy"
exit 1
fi
if check_both_healthy; then
(( STRIKES++ ))
log "Health check passed ($STRIKES/${PARTNERSHIP_TRANSFER_STRIKES})"
[[ "$STRIKES" -lt "$PARTNERSHIP_TRANSFER_STRIKES" ]] && sleep 10
else
warn "Health check failed — resetting (attempt $ATTEMPTS/$MAX_ATTEMPTS)"
STRIKES=0
sleep 30
fi
done
warn "Both servers healthy — proceeding ✅"
# Resolve IPs after health checks confirm reachability
MIRROR_IP=$(resolve_tailscale_ip "$MIRROR")
OWNER_IP=$(resolve_tailscale_ip "$OWNER")
[[ -z "$MIRROR_IP" ]] && { error "Cannot resolve mirror Tailscale IP"; exit 1; }
# Compute post-transfer roles
NEW_OWNER_ID="$MIRROR_ID"
NEW_MIRROR_ID="$OWNER_ID"
NEW_OWNER="$MIRROR"
NEW_MIRROR="$OWNER"
NEW_OWNER_IP="$MIRROR_IP"
# ==============================================================================================
# ━━━ Step 2: Pre-transfer Sync ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SYNC Step 2: Pre-transfer Sync ━━━"
do_final_sync
# ==============================================================================================
# ━━━ Step 3: Reconfigure WebUIs ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CONTAINERS Step 3: Reconfigure WebUIs ━━━"
WEBUI_FAILURES=0
# New owner (current mirror, HOST2) WebUIs → localhost — it now manages itself directly
log "New owner ($NEW_OWNER) WebUIs → localhost"
for entry in "${PARTNERSHIP_AUTH_WEBUIS[@]}"; do
[[ -z "$entry" ]] && continue
container="${entry%%|*}"
port="${entry##*|}"
reconfigure_webui "$container" "$port" "localhost" \
"$SSH_KEY" "$MIRROR_IP" "$MIRROR" || (( WEBUI_FAILURES++ ))
done
# New mirror (us, HOST1) WebUIs → new owner IP — defers to new owner going forward
log "New mirror ($NEW_MIRROR) WebUIs → $NEW_OWNER_IP"
reconfigure_local_webuis "$NEW_OWNER_IP"
WEBUI_RC=$?
[[ "$WEBUI_RC" -gt 0 ]] && (( WEBUI_FAILURES += WEBUI_RC ))
# ==============================================================================================
# ━━━ Step 4: Flip Ownership in master.conf ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_GEAR Step 4: Flip Ownership ━━━"
if [[ "$DRY_RUN" == false ]]; then
update_master_conf "PARTNERSHIP_OWNER_HOST" "\"$NEW_OWNER_ID\""
# Push updated master.conf to new owner so both servers agree immediately.
# master.conf is shared — host-specific credentials live in host*.conf.
_REMOTE_SD=$(ssh -i "$SSH_KEY" -o ConnectTimeout=5 -o StrictHostKeyChecking=no \
"root@${MIRROR_IP}" \
"grep -m1 '^SCRIPTS_DIR' /boot/config/plugins/varaverk/varaverk.cfg 2>/dev/null | cut -d= -f2 | tr -d '\"'" \
2>/dev/null | tr -d '[:space:]')
_REMOTE_SD="${_REMOTE_SD:-/boot/config/plugins/varaverk}"
scp -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" \
-o StrictHostKeyChecking=no \
"$SCRIPTS_ROOT/Configurations/master.conf" \
"root@${MIRROR_IP}:${_REMOTE_SD}/Configurations/master.conf" 2>/dev/null && \
log "master.conf pushed to $NEW_OWNER ✅" || \
error "Failed to push master.conf to $NEW_OWNER — set PARTNERSHIP_OWNER_HOST=\"$NEW_OWNER_ID\" manually"
else
warn "DRY RUN — would set PARTNERSHIP_OWNER_HOST=$NEW_OWNER_ID on both servers"
fi
# ==============================================================================================
# ━━━ Step 5: Write State ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SYNC Step 5: Write State ━━━"
NOW=$(date '+%Y-%m-%d %H:%M:%S')
write_state_file "$LOCAL_STATE_FILE" "ACTIVE" "$NOW" "" "$LOCAL_SERVER_NAME" "transfer"
push_state_to_remote "$LOCAL_STATE_FILE" "$MIRROR_IP" "$SSH_KEY"
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY TRANSFER SUMMARY ━━━━━"
echo " New owner: $NEW_OWNER_ID ($NEW_OWNER — $NEW_OWNER_IP)"
echo " New mirror: $NEW_MIRROR_ID ($NEW_MIRROR)"
echo " WebUI failures: $WEBUI_FAILURES"
echo " Sync direction: $NEW_OWNER → $NEW_MIRROR (next cycle)"
echo " Ownership: PARTNERSHIP_OWNER_HOST=$NEW_OWNER_ID"
echo ""
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no changes made"
else
warn "$ICON_DONE DONE — ownership transferred to $NEW_OWNER_ID ($NEW_OWNER) ✅"
echo "fallback.sh and critical_sync_maintenance.sh will adapt on next cycle"
echo "No containers were moved — only config and WebUI targets updated"
[[ "$WEBUI_FAILURES" -gt 0 ]] && \
warn "$WEBUI_FAILURES WebUI(s) failed — check templates manually"
notify "Partnership ownership transferred — new owner: $NEW_OWNER ($NEW_OWNER_ID)" \
"Partnership" "normal"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
@@ -1,338 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= Partnership Transfer ============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Transfers ownership from the current owner to the current mirror. After
# transfer the roles are swapped: what was the mirror becomes the new owner,
# and what was the owner becomes the new mirror.
#
# No containers are moved — only config and WebUI targets are updated. Both
# servers remain in the partnership; the sync direction reverses on the next
# fallback.sh / critical_sync_maintenance.sh cycle.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# OWNER ONLY — mirror cannot run this script.
#
# Step 1: Health Verification — both servers pass N consecutive health checks
# Step 2: Pre-transfer Sync — final sync in current direction (owner → mirror)
# Step 3: Reconfigure WebUIs — new owner WebUIs → localhost
# new mirror WebUIs → new owner IP
# Step 4: Flip Ownership — update PARTNERSHIP_OWNER_HOST in master.conf
# on both servers
# Step 5: Write State — ACTIVE written locally and pushed to new mirror
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Owner-Only Enforcement
# The script reads MY_ID from detect_hosts() and exits immediately if it is
# not the current PARTNERSHIP_OWNER_HOST. The mirror cannot run a transfer.
#
# Explicit Confirmation String
# Requires the exact passphrase from PARTNERSHIP_TRANSFER_CONFIRM via
# --confirm=<value>. Without a matching string the transfer is cancelled
# before any steps execute. Prevents accidental ownership changes.
#
# Active Partnership Guard
# Reads the local state file and exits if the current state is INACTIVE.
# A transfer without an active partnership has no defined outcome.
#
# Dual Health Verification
# Both servers must pass PARTNERSHIP_TRANSFER_STRIKES consecutive health
# checks before proceeding. A single failure resets the strike counter.
# After PARTNERSHIP_TRANSFER_MAX_ATTEMPTS total attempts the transfer aborts.
#
# Pre-transfer Final Sync
# A full sync in the current direction (owner → mirror) runs immediately
# before roles flip. Ensures the mirror is current before it becomes the owner.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# master.conf
#
# PARTNERSHIP_OWNER_HOST
# Current owner host ID (e.g. "HOST1"). Updated on both servers after transfer.
#
# PARTNERSHIP_TRANSFER_CONFIRM
# Exact string required to confirm transfer (default: "i-understand-this-transfers-ownership").
# Pass via --confirm=<value>.
#
# PARTNERSHIP_TRANSFER_STRIKES
# Consecutive health checks both servers must pass before transfer proceeds (default: 3).
#
# PARTNERSHIP_TRANSFER_MAX_ATTEMPTS
# Max health check attempts before giving up (default: 20).
#
# CRITICAL_SYNC_SHARES
# Array of "path|profile" or "path" entries for do_final_sync().
#
# PARTNERSHIP_AUTH_WEBUIS
# Array of "ContainerName|WebUIPort" entries reconfigured during transfer.
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# Partnership/partnership_transfer.sh --confirm=i-understand-this-transfers-ownership
# Full transfer — owner detected automatically.
#
# Partnership/partnership_transfer.sh --confirm=i-understand-this-transfers-ownership --dry-run
# Preview all steps without executing. Confirmation check is skipped in dry-run mode.
#
# Partnership/partnership_transfer.sh --confirm=i-understand-this-transfers-ownership --log
# Verbose per-step output.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
SCRIPTS_ROOT="$SCRIPT_DIR/.."
SSH_TIMEOUT=15
source "$SCRIPTS_ROOT/load_config.sh"
# ── Parse flags ───────────────────────────────────────────────────────────────────────────────
TRANSFER_CONFIRM_INPUT=""
FILTERED_ARGS=()
for arg in "$@"; do
case "$arg" in
--confirm=*) TRANSFER_CONFIRM_INPUT="${arg#--confirm=}" ;;
*) FILTERED_ARGS+=("$arg") ;;
esac
done
parse_args "${FILTERED_ARGS[@]}"
# ── Source partnership_manager.sh for shared helpers ──────────────────────────────────────────
# PARTNERSHIP_LIB_MODE=1 skips mode dispatch — functions are defined, nothing is executed.
PARTNERSHIP_LIB_MODE=1 source "$SCRIPT_DIR/partnership_manager.sh"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
[[ "$EUID" -ne 0 ]] && { error "Must be run as root"; exit 1; }
if ! command -v docker &>/dev/null; then
error "Docker command not found"
exit 1
fi
detect_hosts
OWNER_ID="${PARTNERSHIP_OWNER_HOST:-HOST1}"
MIRROR_ID=$( [[ "$OWNER_ID" == "HOST1" ]] && echo "HOST2" || echo "HOST1" )
OWNER="${!OWNER_ID}"
MIRROR="${!MIRROR_ID}"
MIRROR_SSH_KEY="$SSH_KEY"
OWNER_SSH_KEY="$SSH_KEY"
AM_OWNER=false
AM_MIRROR=false
[[ "$MY_ID" == "$OWNER_ID" ]] && AM_OWNER=true
[[ "$MY_ID" == "$MIRROR_ID" ]] && AM_MIRROR=true
LOCAL_STATE_FILE="${STATE_DIR:-/boot/config}/partnership_${LOCAL_SERVER_NAME}.db"
REMOTE_STATE_FILE="${STATE_DIR:-/boot/config}/partnership_${REMOTE_SERVER_NAME}.db"
acquire_lock "strict"
# ==============================================================================================
# ━━━ Preflight ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_FALLBACK Partnership Transfer — $MY_ID ($LOCAL_SERVER_NAME) — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
echo ""
if [[ "$AM_MIRROR" == true ]]; then
error "Only the owner ($OWNER / $OWNER_ID) can run --transfer"
error "Run from $OWNER, or use --offboard and re-onboard with roles swapped"
exit 1
fi
if [[ -f "$LOCAL_STATE_FILE" ]]; then
CURRENT_STATE=$(read_state_file "$LOCAL_STATE_FILE" "state")
if [[ "$CURRENT_STATE" == "INACTIVE" ]]; then
error "No active partnership — transfer requires an active partnership"
error "If roles are already correct, check PARTNERSHIP_OWNER_HOST in master.conf"
exit 1
fi
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no permanent changes will be made"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo "⚠️ WARNING — OWNERSHIP TRANSFER"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo " Current owner: $OWNER_ID ($OWNER)"
echo " Current mirror: $MIRROR_ID ($MIRROR)"
echo ""
echo " After transfer:"
echo " New owner: $MIRROR_ID ($MIRROR)"
echo " New mirror: $OWNER_ID ($OWNER)"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
# ==============================================================================================
# ━━━ Confirmation ━━━
# ==============================================================================================
if [[ "$DRY_RUN" == false ]]; then
if [[ -z "$TRANSFER_CONFIRM_INPUT" ]]; then
echo ""
echo "To proceed, pass exactly:"
echo " --confirm=${PARTNERSHIP_TRANSFER_CONFIRM}"
echo ""
error "Transfer cancelled — confirmation required"
exit 1
fi
if [[ "$TRANSFER_CONFIRM_INPUT" != "$PARTNERSHIP_TRANSFER_CONFIRM" ]]; then
error "Confirmation string does not match — transfer cancelled"
exit 1
fi
log "Confirmation accepted"
else
warn "DRY RUN — confirmation check skipped"
fi
# ==============================================================================================
# ━━━ Step 1: Health Verification ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SHIELD Step 1: Health Verification ━━━"
log "Both servers must pass ${PARTNERSHIP_TRANSFER_STRIKES} consecutive health checks"
STRIKES=0
ATTEMPTS=0
MAX_ATTEMPTS="${PARTNERSHIP_TRANSFER_MAX_ATTEMPTS:-20}"
while [[ "$STRIKES" -lt "$PARTNERSHIP_TRANSFER_STRIKES" ]]; do
(( ATTEMPTS++ ))
if [[ "$ATTEMPTS" -gt "$MAX_ATTEMPTS" ]]; then
error "Health checks failed after $MAX_ATTEMPTS attempts — servers not stable"
error "Transfer cancelled — try again when both servers are healthy"
exit 1
fi
if check_both_healthy; then
(( STRIKES++ ))
log "Health check passed ($STRIKES/${PARTNERSHIP_TRANSFER_STRIKES})"
[[ "$STRIKES" -lt "$PARTNERSHIP_TRANSFER_STRIKES" ]] && sleep 10
else
warn "Health check failed — resetting (attempt $ATTEMPTS/$MAX_ATTEMPTS)"
STRIKES=0
sleep 30
fi
done
warn "Both servers healthy — proceeding ✅"
# Resolve IPs after health checks confirm reachability
MIRROR_IP=$(resolve_tailscale_ip "$MIRROR")
OWNER_IP=$(resolve_tailscale_ip "$OWNER")
[[ -z "$MIRROR_IP" ]] && { error "Cannot resolve mirror Tailscale IP"; exit 1; }
# Compute post-transfer roles
NEW_OWNER_ID="$MIRROR_ID"
NEW_MIRROR_ID="$OWNER_ID"
NEW_OWNER="$MIRROR"
NEW_MIRROR="$OWNER"
NEW_OWNER_IP="$MIRROR_IP"
# ==============================================================================================
# ━━━ Step 2: Pre-transfer Sync ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SYNC Step 2: Pre-transfer Sync ━━━"
do_final_sync
# ==============================================================================================
# ━━━ Step 3: Reconfigure WebUIs ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CONTAINERS Step 3: Reconfigure WebUIs ━━━"
WEBUI_FAILURES=0
# New owner (current mirror, HOST2) WebUIs → localhost — it now manages itself directly
log "New owner ($NEW_OWNER) WebUIs → localhost"
for entry in "${PARTNERSHIP_AUTH_WEBUIS[@]}"; do
[[ -z "$entry" ]] && continue
container="${entry%%|*}"
port="${entry##*|}"
reconfigure_webui "$container" "$port" "localhost" \
"$SSH_KEY" "$MIRROR_IP" "$MIRROR" || (( WEBUI_FAILURES++ ))
done
# New mirror (us, HOST1) WebUIs → new owner IP — defers to new owner going forward
log "New mirror ($NEW_MIRROR) WebUIs → $NEW_OWNER_IP"
reconfigure_local_webuis "$NEW_OWNER_IP"
WEBUI_RC=$?
[[ "$WEBUI_RC" -gt 0 ]] && (( WEBUI_FAILURES += WEBUI_RC ))
# ==============================================================================================
# ━━━ Step 4: Flip Ownership in master.conf ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_GEAR Step 4: Flip Ownership ━━━"
if [[ "$DRY_RUN" == false ]]; then
update_master_conf "PARTNERSHIP_OWNER_HOST" "\"$NEW_OWNER_ID\""
# Push updated master.conf to new owner so both servers agree immediately.
# master.conf is shared — host-specific credentials live in host*.conf.
_probe_cmd=$(platform_scripts_dir_probe_cmd)
_REMOTE_SD=$(ssh -i "$SSH_KEY" -o ConnectTimeout=5 -o StrictHostKeyChecking=no \
"root@${MIRROR_IP}" "$_probe_cmd" \
2>/dev/null | tr -d '[:space:]')
_REMOTE_SD="${_REMOTE_SD:-$SCRIPTS_DIR}"
scp -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" \
-o StrictHostKeyChecking=no \
"$SCRIPTS_ROOT/Configurations/master.conf" \
"root@${MIRROR_IP}:${_REMOTE_SD}/Configurations/master.conf" 2>/dev/null && \
log "master.conf pushed to $NEW_OWNER ✅" || \
error "Failed to push master.conf to $NEW_OWNER — set PARTNERSHIP_OWNER_HOST=\"$NEW_OWNER_ID\" manually"
else
warn "DRY RUN — would set PARTNERSHIP_OWNER_HOST=$NEW_OWNER_ID on both servers"
fi
# ==============================================================================================
# ━━━ Step 5: Write State ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SYNC Step 5: Write State ━━━"
NOW=$(date '+%Y-%m-%d %H:%M:%S')
write_state_file "$LOCAL_STATE_FILE" "ACTIVE" "$NOW" "" "$LOCAL_SERVER_NAME" "transfer"
push_state_to_remote "$LOCAL_STATE_FILE" "$MIRROR_IP" "$SSH_KEY"
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY TRANSFER SUMMARY ━━━━━"
echo " New owner: $NEW_OWNER_ID ($NEW_OWNER — $NEW_OWNER_IP)"
echo " New mirror: $NEW_MIRROR_ID ($NEW_MIRROR)"
echo " WebUI failures: $WEBUI_FAILURES"
echo " Sync direction: $NEW_OWNER → $NEW_MIRROR (next cycle)"
echo " Ownership: PARTNERSHIP_OWNER_HOST=$NEW_OWNER_ID"
echo ""
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no changes made"
else
warn "$ICON_DONE DONE — ownership transferred to $NEW_OWNER_ID ($NEW_OWNER) ✅"
echo "fallback.sh and critical_sync_maintenance.sh will adapt on next cycle"
echo "No containers were moved — only config and WebUI targets updated"
[[ "$WEBUI_FAILURES" -gt 0 ]] && \
warn "$WEBUI_FAILURES WebUI(s) failed — check templates manually"
notify "Partnership ownership transferred — new owner: $NEW_OWNER ($NEW_OWNER_ID)" \
"Partnership" "normal"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
@@ -1,338 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= Partnership Transfer ============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Transfers ownership from the current owner to the current mirror. After
# transfer the roles are swapped: what was the mirror becomes the new owner,
# and what was the owner becomes the new mirror.
#
# No containers are moved — only config and WebUI targets are updated. Both
# servers remain in the partnership; the sync direction reverses on the next
# fallback.sh / critical_sync_maintenance.sh cycle.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# OWNER ONLY — mirror cannot run this script.
#
# Step 1: Health Verification — both servers pass N consecutive health checks
# Step 2: Pre-transfer Sync — final sync in current direction (owner → mirror)
# Step 3: Reconfigure WebUIs — new owner WebUIs → localhost
# new mirror WebUIs → new owner IP
# Step 4: Flip Ownership — update PARTNERSHIP_OWNER_HOST in master.conf
# on both servers
# Step 5: Write State — ACTIVE written locally and pushed to new mirror
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Owner-Only Enforcement
# The script reads MY_ID from detect_hosts() and exits immediately if it is
# not the current PARTNERSHIP_OWNER_HOST. The mirror cannot run a transfer.
#
# Explicit Confirmation String
# Requires the exact passphrase from PARTNERSHIP_TRANSFER_CONFIRM via
# --confirm=<value>. Without a matching string the transfer is cancelled
# before any steps execute. Prevents accidental ownership changes.
#
# Active Partnership Guard
# Reads the local state file and exits if the current state is INACTIVE.
# A transfer without an active partnership has no defined outcome.
#
# Dual Health Verification
# Both servers must pass PARTNERSHIP_TRANSFER_STRIKES consecutive health
# checks before proceeding. A single failure resets the strike counter.
# After PARTNERSHIP_TRANSFER_MAX_ATTEMPTS total attempts the transfer aborts.
#
# Pre-transfer Final Sync
# A full sync in the current direction (owner → mirror) runs immediately
# before roles flip. Ensures the mirror is current before it becomes the owner.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# master.conf
#
# PARTNERSHIP_OWNER_HOST
# Current owner host ID (e.g. "HOST1"). Updated on both servers after transfer.
#
# PARTNERSHIP_TRANSFER_CONFIRM
# Exact string required to confirm transfer (default: "i-understand-this-transfers-ownership").
# Pass via --confirm=<value>.
#
# PARTNERSHIP_TRANSFER_STRIKES
# Consecutive health checks both servers must pass before transfer proceeds (default: 3).
#
# PARTNERSHIP_TRANSFER_MAX_ATTEMPTS
# Max health check attempts before giving up (default: 20).
#
# CRITICAL_SYNC_SHARES
# Array of "path|profile" or "path" entries for do_final_sync().
#
# PARTNERSHIP_AUTH_WEBUIS
# Array of "ContainerName|WebUIPort" entries reconfigured during transfer.
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# Partnership/partnership_transfer.sh --confirm=i-understand-this-transfers-ownership
# Full transfer — owner detected automatically.
#
# Partnership/partnership_transfer.sh --confirm=i-understand-this-transfers-ownership --dry-run
# Preview all steps without executing. Confirmation check is skipped in dry-run mode.
#
# Partnership/partnership_transfer.sh --confirm=i-understand-this-transfers-ownership --log
# Verbose per-step output.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
SCRIPTS_ROOT="$SCRIPT_DIR/.."
SSH_TIMEOUT=15
source "$SCRIPTS_ROOT/load_config.sh"
# ── Parse flags ───────────────────────────────────────────────────────────────────────────────
TRANSFER_CONFIRM_INPUT=""
FILTERED_ARGS=()
for arg in "$@"; do
case "$arg" in
--confirm=*) TRANSFER_CONFIRM_INPUT="${arg#--confirm=}" ;;
*) FILTERED_ARGS+=("$arg") ;;
esac
done
parse_args "${FILTERED_ARGS[@]}"
# ── Source partnership_manager.sh for shared helpers ──────────────────────────────────────────
# PARTNERSHIP_LIB_MODE=1 skips mode dispatch — functions are defined, nothing is executed.
PARTNERSHIP_LIB_MODE=1 source "$SCRIPT_DIR/partnership_manager.sh"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
[[ "$EUID" -ne 0 ]] && { error "Must be run as root"; exit 1; }
if ! command -v docker &>/dev/null; then
error "Docker command not found"
exit 1
fi
detect_hosts
OWNER_ID="${PARTNERSHIP_OWNER_HOST:-HOST1}"
MIRROR_ID=$( [[ "$OWNER_ID" == "HOST1" ]] && echo "HOST2" || echo "HOST1" )
OWNER="${!OWNER_ID}"
MIRROR="${!MIRROR_ID}"
MIRROR_SSH_KEY="$SSH_KEY"
OWNER_SSH_KEY="$SSH_KEY"
AM_OWNER=false
AM_MIRROR=false
[[ "$MY_ID" == "$OWNER_ID" ]] && AM_OWNER=true
[[ "$MY_ID" == "$MIRROR_ID" ]] && AM_MIRROR=true
LOCAL_STATE_FILE="${STATE_DIR}/partnership_${LOCAL_SERVER_NAME}.db"
REMOTE_STATE_FILE="${STATE_DIR}/partnership_${REMOTE_SERVER_NAME}.db"
acquire_lock "strict"
# ==============================================================================================
# ━━━ Preflight ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_FALLBACK Partnership Transfer — $MY_ID ($LOCAL_SERVER_NAME) — $(date '+%Y-%m-%d %H:%M:%S') ━━━"
echo ""
if [[ "$AM_MIRROR" == true ]]; then
error "Only the owner ($OWNER / $OWNER_ID) can run --transfer"
error "Run from $OWNER, or use --offboard and re-onboard with roles swapped"
exit 1
fi
if [[ -f "$LOCAL_STATE_FILE" ]]; then
CURRENT_STATE=$(read_state_file "$LOCAL_STATE_FILE" "state")
if [[ "$CURRENT_STATE" == "INACTIVE" ]]; then
error "No active partnership — transfer requires an active partnership"
error "If roles are already correct, check PARTNERSHIP_OWNER_HOST in master.conf"
exit 1
fi
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no permanent changes will be made"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo "⚠️ WARNING — OWNERSHIP TRANSFER"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo " Current owner: $OWNER_ID ($OWNER)"
echo " Current mirror: $MIRROR_ID ($MIRROR)"
echo ""
echo " After transfer:"
echo " New owner: $MIRROR_ID ($MIRROR)"
echo " New mirror: $OWNER_ID ($OWNER)"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
# ==============================================================================================
# ━━━ Confirmation ━━━
# ==============================================================================================
if [[ "$DRY_RUN" == false ]]; then
if [[ -z "$TRANSFER_CONFIRM_INPUT" ]]; then
echo ""
echo "To proceed, pass exactly:"
echo " --confirm=${PARTNERSHIP_TRANSFER_CONFIRM}"
echo ""
error "Transfer cancelled — confirmation required"
exit 1
fi
if [[ "$TRANSFER_CONFIRM_INPUT" != "$PARTNERSHIP_TRANSFER_CONFIRM" ]]; then
error "Confirmation string does not match — transfer cancelled"
exit 1
fi
log "Confirmation accepted"
else
warn "DRY RUN — confirmation check skipped"
fi
# ==============================================================================================
# ━━━ Step 1: Health Verification ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SHIELD Step 1: Health Verification ━━━"
log "Both servers must pass ${PARTNERSHIP_TRANSFER_STRIKES} consecutive health checks"
STRIKES=0
ATTEMPTS=0
MAX_ATTEMPTS="${PARTNERSHIP_TRANSFER_MAX_ATTEMPTS:-20}"
while [[ "$STRIKES" -lt "$PARTNERSHIP_TRANSFER_STRIKES" ]]; do
(( ATTEMPTS++ ))
if [[ "$ATTEMPTS" -gt "$MAX_ATTEMPTS" ]]; then
error "Health checks failed after $MAX_ATTEMPTS attempts — servers not stable"
error "Transfer cancelled — try again when both servers are healthy"
exit 1
fi
if check_both_healthy; then
(( STRIKES++ ))
log "Health check passed ($STRIKES/${PARTNERSHIP_TRANSFER_STRIKES})"
[[ "$STRIKES" -lt "$PARTNERSHIP_TRANSFER_STRIKES" ]] && sleep 10
else
warn "Health check failed — resetting (attempt $ATTEMPTS/$MAX_ATTEMPTS)"
STRIKES=0
sleep 30
fi
done
warn "Both servers healthy — proceeding ✅"
# Resolve IPs after health checks confirm reachability
MIRROR_IP=$(resolve_tailscale_ip "$MIRROR")
OWNER_IP=$(resolve_tailscale_ip "$OWNER")
[[ -z "$MIRROR_IP" ]] && { error "Cannot resolve mirror Tailscale IP"; exit 1; }
# Compute post-transfer roles
NEW_OWNER_ID="$MIRROR_ID"
NEW_MIRROR_ID="$OWNER_ID"
NEW_OWNER="$MIRROR"
NEW_MIRROR="$OWNER"
NEW_OWNER_IP="$MIRROR_IP"
# ==============================================================================================
# ━━━ Step 2: Pre-transfer Sync ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SYNC Step 2: Pre-transfer Sync ━━━"
do_final_sync
# ==============================================================================================
# ━━━ Step 3: Reconfigure WebUIs ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_CONTAINERS Step 3: Reconfigure WebUIs ━━━"
WEBUI_FAILURES=0
# New owner (current mirror, HOST2) WebUIs → localhost — it now manages itself directly
log "New owner ($NEW_OWNER) WebUIs → localhost"
for entry in "${PARTNERSHIP_AUTH_WEBUIS[@]}"; do
[[ -z "$entry" ]] && continue
container="${entry%%|*}"
port="${entry##*|}"
reconfigure_webui "$container" "$port" "localhost" \
"$SSH_KEY" "$MIRROR_IP" "$MIRROR" || (( WEBUI_FAILURES++ ))
done
# New mirror (us, HOST1) WebUIs → new owner IP — defers to new owner going forward
log "New mirror ($NEW_MIRROR) WebUIs → $NEW_OWNER_IP"
reconfigure_local_webuis "$NEW_OWNER_IP"
WEBUI_RC=$?
[[ "$WEBUI_RC" -gt 0 ]] && (( WEBUI_FAILURES += WEBUI_RC ))
# ==============================================================================================
# ━━━ Step 4: Flip Ownership in master.conf ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_GEAR Step 4: Flip Ownership ━━━"
if [[ "$DRY_RUN" == false ]]; then
update_master_conf "PARTNERSHIP_OWNER_HOST" "\"$NEW_OWNER_ID\""
# Push updated master.conf to new owner so both servers agree immediately.
# master.conf is shared — host-specific credentials live in host*.conf.
_probe_cmd=$(platform_scripts_dir_probe_cmd)
_REMOTE_SD=$(ssh -i "$SSH_KEY" -o ConnectTimeout=5 -o StrictHostKeyChecking=no \
"root@${MIRROR_IP}" "$_probe_cmd" \
2>/dev/null | tr -d '[:space:]')
_REMOTE_SD="${_REMOTE_SD:-$SCRIPTS_DIR}"
scp -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" \
-o StrictHostKeyChecking=no \
"$SCRIPTS_ROOT/Configurations/master.conf" \
"root@${MIRROR_IP}:${_REMOTE_SD}/Configurations/master.conf" 2>/dev/null && \
log "master.conf pushed to $NEW_OWNER ✅" || \
error "Failed to push master.conf to $NEW_OWNER — set PARTNERSHIP_OWNER_HOST=\"$NEW_OWNER_ID\" manually"
else
warn "DRY RUN — would set PARTNERSHIP_OWNER_HOST=$NEW_OWNER_ID on both servers"
fi
# ==============================================================================================
# ━━━ Step 5: Write State ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_SYNC Step 5: Write State ━━━"
NOW=$(date '+%Y-%m-%d %H:%M:%S')
write_state_file "$LOCAL_STATE_FILE" "ACTIVE" "$NOW" "" "$LOCAL_SERVER_NAME" "transfer"
push_state_to_remote "$LOCAL_STATE_FILE" "$MIRROR_IP" "$SSH_KEY"
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY TRANSFER SUMMARY ━━━━━"
echo " New owner: $NEW_OWNER_ID ($NEW_OWNER — $NEW_OWNER_IP)"
echo " New mirror: $NEW_MIRROR_ID ($NEW_MIRROR)"
echo " WebUI failures: $WEBUI_FAILURES"
echo " Sync direction: $NEW_OWNER → $NEW_MIRROR (next cycle)"
echo " Ownership: PARTNERSHIP_OWNER_HOST=$NEW_OWNER_ID"
echo ""
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no changes made"
else
warn "$ICON_DONE DONE — ownership transferred to $NEW_OWNER_ID ($NEW_OWNER) ✅"
echo "fallback.sh and critical_sync_maintenance.sh will adapt on next cycle"
echo "No containers were moved — only config and WebUI targets updated"
[[ "$WEBUI_FAILURES" -gt 0 ]] && \
warn "$WEBUI_FAILURES WebUI(s) failed — check templates manually"
notify "Partnership ownership transferred — new owner: $NEW_OWNER ($NEW_OWNER_ID)" \
"Partnership" "normal"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
@@ -1,785 +0,0 @@
# ━━━━━ TOOLS — Manual ━━━━━
Configuration reference, usage procedures, and field guides for every script
in `Tools/`. Run any script with `--status` first — it shows current state before
making any changes.
---
## ━━━ CONTENTS ━━━
- [emby_to_lidarr_sync.sh](#emby_to_lidarr_syncsh)
- [emby_to_sonarr_sync.sh](#emby_to_sonarr_syncsh)
- [emby_to_radarr_sync.sh](#emby_to_radarr_syncsh)
- [fallback_state_reset.sh](#fallback_state_resetsh) *(not yet built — manual workaround)*
- [watchdog_skip_list_manager.sh](#watchdog_skip_list_managersh)
- [bulk_permissions_repair.sh](#bulk_permissions_repairsh)
- [container_data_export.sh](#container_data_exportsh)
- [emby_database_repair.sh](#emby_database_repairsh)
- [zfs_pool_scrub.sh](#zfs_pool_scrubsh)
- [recreate_shares.sh](#recreate_sharessh)
- [continuous_scripts_status.sh](#continuous_scripts_statussh)
- [claude_startup.sh](#claude_startupsh)
- [ramdisk_stop.sh](#ramdisk_stopsh)
- [Adding a New Tool](#adding-a-new-tool)
---
## Output Tiers
All tools have two output levels controlled by `--log`.
Without `--log`, each script processes and always concludes with a summary block
showing identity, duration, counts, and a status line. Warnings and errors are
always visible. State-display scripts (watchdog_skip_list_manager, zfs_pool_scrub
`--status`, fallback_state_reset current-state section) always show their state
output — `--log` adds configuration detail and per-item resolution within each
section.
With `--log`, per-item detail appears: individual items added/skipped, per-database
check results, per-pool scan lines, per-container stop/start state, per-directory
creation results. Use when debugging unexpected results or confirming a first run.
---
## emby_to_lidarr_sync.sh
One-shot bootstrap tool. Scans Emby play history, finds artists you've actually
listened to that are not yet tracked in Lidarr, and adds them. No scoring — if
it was played, Lidarr should monitor it. Not scheduled; run manually when you
want to close the gap between what's in your library and what Lidarr watches.
### When to Use
- After initial Lidarr setup — bring it in line with existing listening history
- After a Lidarr database wipe or migration
- Any time you suspect artists you listen to are slipping through unmonitored
### Usage
```bash
# See what would be added (no changes)
bash Tools/emby_to_lidarr_sync.sh --dry-run
# Limit to recent plays only
bash Tools/emby_to_lidarr_sync.sh --dry-run --days 30
# Run for real
bash Tools/emby_to_lidarr_sync.sh
```
### Notes
- Reads all MusicAlbum items from Emby and extracts AlbumArtist — primary album
artists only, not guest features or tag credits
- Filters out VA, Various Artists, and other metadata placeholders
- Triggers `ArtistSearch` immediately after each successful add — no manual search needed
- Activity log has a finite history; use `--days N` if the log has been pruned
---
## emby_to_sonarr_sync.sh
One-shot bootstrap tool. Finds TV series present in Emby that are not tracked in
Sonarr and adds them. Uses TVDB ID matching when available (more reliable than
title matching), falling back to case-insensitive title comparison.
### When to Use
- After initial Sonarr setup — bring it in line with your existing library
- After a Sonarr database wipe or migration
- Any time series you own are slipping through unmonitored
### Usage
```bash
# See what would be added (no changes)
bash Tools/emby_to_sonarr_sync.sh --dry-run
# Run for real
bash Tools/emby_to_sonarr_sync.sh
```
### Notes
- Triggers `SeriesSearch` immediately after each successful add — Sonarr begins
searching for missing episodes right away
- Adds to the first accessible root folder in Sonarr
- TVDB ID match preferred over title; title fallback handles edge cases
- Requires `SONARR_EMBY_LIBRARIES` configured in `master.conf`
---
## emby_to_radarr_sync.sh
One-shot bootstrap tool. Finds movies present in Emby that are not tracked in
Radarr and adds them. Uses TMDB ID matching when available, falling back to
case-insensitive title comparison.
### When to Use
- After initial Radarr setup — bring it in line with your existing library
- After a Radarr database wipe or migration
- Any time movies you own are slipping through unmonitored
### Usage
```bash
# See what would be added (no changes)
bash Tools/emby_to_radarr_sync.sh --dry-run
# Run for real
bash Tools/emby_to_radarr_sync.sh
```
### Notes
- Triggers `MoviesSearch` immediately after each successful add — Radarr begins
searching for the movie right away
- Adds to the first accessible root folder in Radarr
- TMDB ID match preferred over title; title fallback handles edge cases
---
## fallback_state_reset.sh
> **Not yet built.** Use the manual workaround below.
Planned: reset the fallback state file to NORMAL and clear all tier flags. State file
only — does NOT start or stop any containers.
### When to Use
```
After fallback_test.sh didn't complete cleanly
→ state left in FALLBACK but containers are actually back to normal
After a failed handback
→ state shows FALLBACK but remote is back up and containers are split
After killing fallback.sh directly (not gracefully via SIGTERM)
→ state is unknown, cycle was interrupted mid-operation
After a dev/debug session
→ state left in a non-NORMAL state from testing
```
### Manual Workaround
```bash
# Verify before resetting:
# Right containers on right server?
continuous_scripts_status.sh # shows fallback current state
# DDNS pointing correctly?
nslookup Gmer4Lfe.com # confirm it resolves to the right IP
# fallback.sh not running?
pgrep -f "fallback.sh" # empty output = not running
# Both servers Tailscale connected?
tailscale status # both hosts should show active
# Check current state file:
cat /boot/config/fallback_state.db
# Reset to NORMAL (only after confirming containers and DDNS are correct):
echo "state=NORMAL" > /boot/config/fallback_state.db
```
Resetting during an actual fallback causes fallback.sh to think everything is normal
and stop covering the remote — services go offline until the next detection cycle.
### What the State File Contains
```bash
state=NORMAL
fallback_start=0
handback_strikes=0
tier2_started=false
tier3_started=false
tier4_started=false
```
---
## watchdog_skip_list_manager.sh
View and manage the persistent container skip list used by `docker_watchdog.sh`.
### When to Use
```
docker_watchdog.sh restarts the same container N times within the rolling window
→ container added to skip list on /boot/config/
→ critical notification sent
→ watchdog stops touching it entirely
You fix the underlying problem (database, config, dependencies).
You need to clear the container from the skip list so monitoring resumes.
```
### Recovery Workflow
```bash
# 1. Understand the situation — always start here:
watchdog_skip_list_manager.sh --status
# Shows: skip list contents, which are running vs. stopped, restart history
# 2. Fix the underlying problem first
# Check logs: docker logs ContainerName --tail 100
# Check disk: df -h /mnt/user
# Check db: docker exec ContainerName sqlite3 /path/to.db ".tables"
# 3. Clear from skip list + restart history:
watchdog_skip_list_manager.sh --clear ContainerName
# 4. Start the container manually — confirm your fix worked:
docker start ContainerName
# 5. Watchdog resumes normal monitoring on next cycle — no further action needed
```
### State Files Managed
```bash
# Both live on /boot/config — survive reboots intentionally.
# A container that was skip-listed before a reboot is still broken after it.
$SYS_WATCHDOG_FAILED_FILE # persistent skip list
$WATCHDOG_CONTAINER_RESTART_LOG # restart loop tracking
```
### Configuration (master.conf)
```bash
WATCHDOG_CONTAINER_RESTART_LIMIT=3 # restarts before skip-listing
WATCHDOG_CONTAINER_RESTART_WINDOW=1 # rolling window in hours
```
### Usage
```bash
watchdog_skip_list_manager.sh # show status (default)
watchdog_skip_list_manager.sh --status # explicit status
watchdog_skip_list_manager.sh --clear ContainerName # clear specific + restart history
watchdog_skip_list_manager.sh --clear ContainerName --force # no confirmation prompt
watchdog_skip_list_manager.sh --clear-all # clear everything
watchdog_skip_list_manager.sh --clear-all --force # non-interactive
watchdog_skip_list_manager.sh --dry-run # preview any clear action
```
---
## bulk_permissions_repair.sh
Applies correct ownership and permissions to specific paths. Faster than running
`media_shares_permissions.sh` which processes every configured share — use this when
you know exactly what needs fixing and don't want to wait for a full library walk.
### When to Use
```
Admin copy left root:root files — scp, cp, direct file transfer
New share needs permissions now — can't wait for nightly run
Container wrote as root — before PUID/PGID was fixed
Specific directory has wrong perms — targeted fix, not a full library walk
```
Use the full `media_shares_permissions.sh` instead for:
- Regular nightly maintenance (already scheduled in daily_sync_maintenance.sh)
- After confirming a container's PUID/PGID is now correct
- Initial permissions setup on a new server
### Diagnosing High Wrong-Owner Counts
The script counts files with wrong ownership before applying the fix. A high count
on a share that was recently written means a container has wrong PUID/PGID.
```bash
# Fix: add to the container's Docker template:
PUID=99
PGID=100
# Common culprits writing as root:
# SABnzbd, qBittorrent, slskd — check each one's Docker env vars
```
### Configuration (master.conf)
```bash
PERMISSIONS_OWNER="nobody:users" # matches PUID=99 PGID=100
PERMISSIONS_DIR_MODE="755" # directories — enter, list, no world-write
PERMISSIONS_FILE_MODE="664" # files — owner+group rw, others read-only
```
### Usage
```bash
# Single path:
bulk_permissions_repair.sh /mnt/user/Movies
# Multiple paths — all corrected in one run:
bulk_permissions_repair.sh /mnt/user/Movies /mnt/user/Tv_Shows /mnt/user/Music
# Dry run first — shows count of files with wrong ownership per path:
bulk_permissions_repair.sh /mnt/user/Movies --dry-run
# Verbose — show each corrected file:
bulk_permissions_repair.sh /mnt/user/Movies --log
```
---
## container_data_export.sh
Exports a container's appdata directory to a compressed tar archive. Stops the
container first for a clean consistent backup, verifies the archive after creation,
then restarts the container.
### When to Use
```
Before major container updates — especially "database migration — no rollback" changelogs
Before pool migrations — clean backup before moving appdata to a new pool
Before removing a container from the stack — archive its data before deletion
Manual point-in-time backup before risky config changes
```
### Export Sequence
```
1. Space check
Estimates required space from appdata size × 1.1
Aborts if output directory doesn't have enough free space
Container is NOT stopped until the space check passes
2. Stop container cleanly
docker stop ContainerName — graceful shutdown
3. Create archive
tar -czf ContainerName_YYYY-MM-DD_HH-MM.tar.gz /path/to/appdata
4. Verify archive integrity
tar --test-file archive.tar.gz — confirms archive is valid and complete
If verification fails → restart container anyway, report error
5. Restart container
docker start ContainerName — always happens, even if archiving failed
```
### Usage
```bash
# Syntax: container_data_export.sh ContainerName AppDataPath OutputDir
# Emby backup:
container_data_export.sh \
Emby \
/mnt/media-servers/Media_Server/Emby \
/mnt/user/Backups/
# Dry run — verify space and paths without stopping anything:
container_data_export.sh \
Emby \
/mnt/media-servers/Media_Server/Emby \
/mnt/user/Backups/ \
--dry-run
# Output filename: Emby_2026-05-14_02-30.tar.gz
# Timestamped — safe to run multiple times, no overwrite
```
---
## emby_database_repair.sh
Stops Emby, runs SQLite `PRAGMA integrity_check` on every Emby database, and restarts.
Reports per-database — does NOT automatically repair. Recovery requires judgment.
### When to Use
```
Emby logs show database errors → run this first
Emby crashing repeatedly with no clear cause → likely database corruption
Playback history or user data behaving strangely → users.db or library.db issue
After a hard shutdown or power loss with Emby running → check for WAL corruption
```
### Recovery Guide by Database
```
library.db — media library metadata: titles, seasons, episodes, artwork
CORRUPT → safe to delete — Emby fully rebuilds from media files on next start
Rebuild takes time (hours on large libraries) but loses nothing permanent
users.db — user accounts, watch history, playback positions, settings
CORRUPT → deleting resets ALL user accounts and watch history
Check for a recent backup (weekly_sync_maintenance.sh mirrors Emby/)
before deleting — restore from remote if available
authentication.db — API keys, session tokens
CORRUPT → safe to delete — API keys regenerated on restart
Any connected clients will need to re-authenticate once
activity.db — activity/access log
CORRUPT → safe to delete — it's a log, losing it is acceptable
library.db-wal — write-ahead log (uncommitted transactions)
PRESENT + CORRUPT → check library.db first; WAL corruption usually means
the main library.db is also affected
```
### Configuration (host*.conf)
```bash
HOST1_EMBY_CONTAINER="Emby" # aliased by detect_hosts() → EMBY_CONTAINER
HOST2_EMBY_CONTAINER="Emby"
```
Emby's config path is detected automatically from Docker volume mounts — no manual
path configuration needed.
### Usage
```bash
emby_database_repair.sh # stop Emby, check all databases, restart
emby_database_repair.sh --dry-run # show what would be checked, no Emby stop
emby_database_repair.sh --log # verbose — show SQLite output per database
emby_database_repair.sh --status # show Emby config path and database locations
```
---
## zfs_pool_scrub.sh
Triggers ZFS scrub on all pools (or a specific named pool) and waits for completion.
Notifies when done with a summary of any errors found.
### Why Run ZFS Scrub
ZFS stores a checksum with every block of data. Scrub reads every block and verifies
the checksum matches the stored hash. Silent data corruption can sit on disk for months
without triggering any error — until you try to read that specific file. By then:
- It may already be mirrored to HOST2 in its corrupted state
- The original source may no longer exist
- ZFS can self-repair during scrub if redundancy exists (RAIDZ or mirrors)
Run monthly. Also run after any disk replacement or power event.
Safe to run while the system is in use — scrub runs at low I/O priority.
### Configuration (host*.conf)
```bash
HOST1_ZFS_REPORT_IGNORE_POOLS=(
"disk10" # JBOD member — no redundancy, skipped from default scrub
"disk9"
"disk8"
)
HOST2_ZFS_REPORT_IGNORE_POOLS=(
"cache" # example — single-disk pool excluded from default
)
```
To scrub a pool in the ignore list, specify it by name explicitly.
### Usage
```bash
# Scrub all pools except those in ZFS_REPORT_IGNORE_POOLS:
zfs_pool_scrub.sh
# Scrub a specific pool by name — bypasses the ignore list:
zfs_pool_scrub.sh gaming
# Check current scrub status without starting a new one:
zfs_pool_scrub.sh --status
# Dry run — show which pools would be scrubbed:
zfs_pool_scrub.sh --dry-run
# Verbose — show scrub progress every 60s poll:
zfs_pool_scrub.sh --log
```
---
## recreate_shares.sh
Creates share directories on the correct disks after a fresh unRAID install or disk
rebuild. Run once on HOST2 before the first rsync from HOST1.
### When to Use
```
After a fresh unRAID install where /boot/config/shares/*.cfg were restored:
The share definitions exist → UI shows shares → directories are missing on disk
rsync.sh tries to write to /mnt/user/Movies → path doesn't exist → aborts
After a disk replacement or rebuild where share folders were lost:
Replacement disk is blank → no share directories on the new disk
unRAID won't create them automatically
```
### What It Does
```
For each .cfg file in /boot/config/shares/:
1. Read the share name (e.g., Movies)
2. Read the shareInclude list (e.g., disk1,disk2,disk5)
3. Create /mnt/disk1/Movies, /mnt/disk2/Movies, /mnt/disk5/Movies
4. Place a .recovery marker in /mnt/user/Movies/
The .recovery marker tells rsync.sh this is a fresh share:
.recovery present → rsync WITHOUT --delete (safe — new files only, nothing removed)
.recovery absent → rsync WITH --delete (normal mirror mode)
Self-cleaning: after the first successful rsync, the source side has no .recovery file,
so the second nightly run deletes it from the mirror and normal --delete resumes.
No manual cleanup needed.
```
### Usage
```bash
recreate_shares.sh # create all missing share directories + .recovery markers
recreate_shares.sh --dry-run # show what would be created without creating
recreate_shares.sh --log # verbose — show each directory created per disk
recreate_shares.sh --status # show share configs and current directory state
```
---
## continuous_scripts_status.sh
Live status dashboard for all continuously running scripts. Read-only — makes no
changes to any running process, container, or state file.
### What It Shows
```
stability_watchdog
Running state, PID, uptime, approximate cycle count
Active strikes, recent restart history
Live snapshot: rootfs, RAM, ZFS ARC, load, zombie count, CPU temp
docker_watchdog
Running state, PID, uptime
Running / stopped / unhealthy container counts
Required containers status
Memory-monitored containers
Recent restart history + skip list
fallback (fallback.sh)
Current state (NORMAL / FALLBACK / HANDBACK)
Tier flags and timestamps
Remote Tailscale visibility
```
State files are read as-is — if a script is mid-cycle, the display reflects the last
completed cycle, not the current in-progress state.
### Usage
```bash
continuous_scripts_status.sh # show full dashboard
continuous_scripts_status.sh --log # verbose output with additional detail per section
```
---
## claude_startup.sh
Restores Claude Code's persistent data after an unRAID reboot and optionally launches
Claude. Standalone script — no common.sh dependency.
### Why This Exists
unRAID's root filesystem lives in RAM — `/root/.claude` and `/root/.local` are wiped on
every reboot. This script symlinks both directories back to persistent appdata storage
at `/mnt/user/appdata/claude-code/` before launching Claude.
### First Run Migration
On first run, if persistent storage is empty, the script migrates from current live locations:
```
/root/.claude → /mnt/user/appdata/claude-code/.claude
/root/.local/share/claude → /mnt/user/appdata/claude-code/local/share/claude
```
Subsequent runs skip the migration and only create the symlinks.
### Calling from array_started.sh
To auto-restore Claude data on every boot without launching an interactive session:
```bash
# In /boot/config/go or array_started.sh:
/path/to/Tools/claude_startup.sh --setup
```
### Usage
```bash
claude_startup.sh # set up persistent symlinks and launch Claude
claude_startup.sh --setup # set up symlinks only — no launch (for array_started.sh)
```
---
## ramdisk_stop.sh
Safely stops the transcode ramdisk: redirects the transcode symlink to the SSD
fallback first (so Emby continues writing without interruption), then unmounts the
tmpfs and updates the state file. Primary use case is stopping the current ramdisk
before re-running `ramdisk_setup.sh` with new size or threshold values.
### When to Use
```
Bumping RAMDISK_SIZE — setup script is idempotent, skips remount if already mounted
→ stop first, then re-run ramdisk_setup.sh with new HOST*_RAMDISK_SIZE value
Adjusting RAMDISK_WARN_GB / RAMDISK_LOW_GB thresholds
→ no need to stop for threshold changes (transcode_manager reads vars live)
→ only needed if you're also changing the size
Temporarily freeing ramdisk RAM — reclaim tmpfs back to general memory pool
→ stop, restart later with ramdisk_setup.sh
```
### Stop Sequence
```
1. Redirect symlink: TRANSCODE_LINK → TRANSCODE_SSD
Emby immediately writes to SSD — no broken-path window during unmount
2. Check for active transcode files on ramdisk (warn, don't block)
Files in progress on the ramdisk are lost on unmount — expected for maintenance
3. Unmount ramdisk
Regular umount first; if busy (directory handles only, no active writes)
falls back to lazy unmount automatically
4. Update /tmp/transcode_state.db → current_target=TRANSCODE_SSD
transcode_manager.sh reads this on its next cycle
```
### transcode_manager Warning
If `transcode_manager.sh` is running, it may flip the symlink back to the ramdisk
on its next cycle (once the ramdisk is unmounted, that flip will fail). Stop
`transcode_manager.sh` first if you need the SSD redirect to hold before remounting.
### After Stopping
```bash
# Update host*.conf with new size values:
# HOST1_RAMDISK_SIZE="10G"
# HOST1_RAMDISK_WARN_GB=8.5
# HOST1_RAMDISK_LOW_GB=7
# Remount at new size:
bash Transcodes/ramdisk_setup.sh
```
### Usage
```bash
ramdisk_stop.sh --status # show mount state, symlink, active files — always check first
ramdisk_stop.sh --dry-run # show what would happen without making changes
ramdisk_stop.sh # stop the ramdisk
ramdisk_stop.sh --log # verbose — show each step
```
---
## Adding a New Tool
Write the tool when you solve a problem manually with bash commands. You'll face it again.
The cost of writing the tool is 30 minutes. The cost of reconstructing the commands at 2am
is much higher.
### Checklist
```
✓ Header explains the specific situation that requires this tool
✓ Root check — most tools need root
✓ --dry-run support — always
✓ --status support — show current state before acting
✓ Confirmation for destructive operations (interactive YES or --force flag)
✓ Notify on completion — success and failure
✓ Leave system in clean state on any exit — trap for cleanup
✓ Add to README-Tools.md scripts table and HOW THE SCRIPTS RELATE diagram
```
### Minimal Skeleton
```bash
#!/bin/bash
# ==============================================================================================
# ============================= Your Tool Name ================================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# One sentence: what situation this solves and when to use it.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Root Required
# chown / docker / etc. require root.
#
# Confirmation Required
# Interactive mode prompts for YES. Use --force to bypass in scripts.
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# your_tool.sh
# Normal run.
#
# your_tool.sh --dry-run
# Preview without making changes.
#
# your_tool.sh --status
# Show current state and exit.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
if [[ "$EUID" -ne 0 ]]; then error "Must be run as root"; exit 1; fi
platform_require_cmd \
"/usr/local/emhttp/plugins/dynamix/scripts/notify" \
"" "" "unRAID notify script" || warn "notify not found — notifications disabled"
acquire_lock
detect_hosts
if [[ "$SHOW_STATUS" == true ]]; then
log "Current state: ..."
exit 0
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes will be made"
if [[ "$FORCE" != true ]]; then
read -r -p "Type YES to proceed: " CONFIRM
[[ "$CONFIRM" != "YES" ]] && { warn "Aborted."; exit 0; }
fi
# Do the work
# ...
notify "Tool completed on $(hostname) ($MY_ID)" "Tool Name" "normal"
```
@@ -1,805 +0,0 @@
# ━━━━━ TOOLS — Manual ━━━━━
Configuration reference, usage procedures, and field guides for every script
in `Tools/`. Run any script with `--status` first — it shows current state before
making any changes.
---
## ━━━ CONTENTS ━━━
- [emby_to_lidarr_sync.sh](#emby_to_lidarr_syncsh)
- [emby_to_sonarr_sync.sh](#emby_to_sonarr_syncsh)
- [emby_to_radarr_sync.sh](#emby_to_radarr_syncsh)
- [fallback_state_reset.sh](#fallback_state_resetsh)
- [docker_prune_images.sh](#docker_prune_imagessh)
- [watchdog_skip_list_manager.sh](#watchdog_skip_list_managersh)
- [bulk_permissions_repair.sh](#bulk_permissions_repairsh)
- [container_data_export.sh](#container_data_exportsh)
- [emby_database_repair.sh](#emby_database_repairsh)
- [zfs_pool_scrub.sh](#zfs_pool_scrubsh)
- [recreate_shares.sh](#recreate_sharessh)
- [continuous_scripts_status.sh](#continuous_scripts_statussh)
- [claude_startup.sh](#claude_startupsh)
- [ramdisk_stop.sh](#ramdisk_stopsh)
- [Adding a New Tool](#adding-a-new-tool)
---
## Output Tiers
All tools have two output levels controlled by `--log`.
Without `--log`, each script processes and always concludes with a summary block
showing identity, duration, counts, and a status line. Warnings and errors are
always visible. State-display scripts (watchdog_skip_list_manager, zfs_pool_scrub
`--status`, fallback_state_reset current-state section) always show their state
output — `--log` adds configuration detail and per-item resolution within each
section.
With `--log`, per-item detail appears: individual items added/skipped, per-database
check results, per-pool scan lines, per-container stop/start state, per-directory
creation results. Use when debugging unexpected results or confirming a first run.
---
## emby_to_lidarr_sync.sh
One-shot bootstrap tool. Scans Emby play history, finds artists you've actually
listened to that are not yet tracked in Lidarr, and adds them. No scoring — if
it was played, Lidarr should monitor it. Not scheduled; run manually when you
want to close the gap between what's in your library and what Lidarr watches.
### When to Use
- After initial Lidarr setup — bring it in line with existing listening history
- After a Lidarr database wipe or migration
- Any time you suspect artists you listen to are slipping through unmonitored
### Usage
```bash
# See what would be added (no changes)
bash Tools/emby_to_lidarr_sync.sh --dry-run
# Limit to recent plays only
bash Tools/emby_to_lidarr_sync.sh --dry-run --days 30
# Run for real
bash Tools/emby_to_lidarr_sync.sh
```
### Notes
- Reads all MusicAlbum items from Emby and extracts AlbumArtist — primary album
artists only, not guest features or tag credits
- Filters out VA, Various Artists, and other metadata placeholders
- Triggers `ArtistSearch` immediately after each successful add — no manual search needed
- Activity log has a finite history; use `--days N` if the log has been pruned
---
## emby_to_sonarr_sync.sh
One-shot bootstrap tool. Finds TV series present in Emby that are not tracked in
Sonarr and adds them. Uses TVDB ID matching when available (more reliable than
title matching), falling back to case-insensitive title comparison.
### When to Use
- After initial Sonarr setup — bring it in line with your existing library
- After a Sonarr database wipe or migration
- Any time series you own are slipping through unmonitored
### Usage
```bash
# See what would be added (no changes)
bash Tools/emby_to_sonarr_sync.sh --dry-run
# Run for real
bash Tools/emby_to_sonarr_sync.sh
```
### Notes
- Triggers `SeriesSearch` immediately after each successful add — Sonarr begins
searching for missing episodes right away
- Adds to the first accessible root folder in Sonarr
- TVDB ID match preferred over title; title fallback handles edge cases
- Requires `SONARR_EMBY_LIBRARIES` configured in `master.conf`
---
## emby_to_radarr_sync.sh
One-shot bootstrap tool. Finds movies present in Emby that are not tracked in
Radarr and adds them. Uses TMDB ID matching when available, falling back to
case-insensitive title comparison.
### When to Use
- After initial Radarr setup — bring it in line with your existing library
- After a Radarr database wipe or migration
- Any time movies you own are slipping through unmonitored
### Usage
```bash
# See what would be added (no changes)
bash Tools/emby_to_radarr_sync.sh --dry-run
# Run for real
bash Tools/emby_to_radarr_sync.sh
```
### Notes
- Triggers `MoviesSearch` immediately after each successful add — Radarr begins
searching for the movie right away
- Adds to the first accessible root folder in Radarr
- TMDB ID match preferred over title; title fallback handles edge cases
---
## fallback_state_reset.sh
Resets the fallback state file to NORMAL and clears all tier flags. State file only —
does NOT start or stop any containers. After reset, fallback.sh resumes from NORMAL on
its next cycle.
**Only run after verifying the stack is actually in a normal state** — right containers
on the right server, DDNS correct, no active fallback in progress. Resetting state during
a real fallback causes fallback.sh to stop covering the remote until the next detection cycle.
### When to Use
```
After fallback_test.sh didn't complete cleanly
→ state left in FALLBACK but containers are actually back to normal
After a failed handback
→ state shows FALLBACK but remote is back up and containers are split
After killing fallback.sh directly (not gracefully via SIGTERM)
→ state is unknown, cycle was interrupted mid-operation
After a dev/debug session
→ state left in a non-NORMAL state from testing
```
### Usage
```bash
fallback_state_reset.sh # show current state, prompt for YES before resetting
fallback_state_reset.sh --status # show current state file contents only
fallback_state_reset.sh --dry-run # show what the new state file would contain, no write
fallback_state_reset.sh --force # reset without confirmation prompt (for scripted use)
```
### Verify Before Resetting
```bash
# Right containers on right server?
continuous_scripts_status.sh # shows fallback current state
# DDNS pointing correctly?
nslookup Gmer4Lfe.com # confirm it resolves to the right IP
# fallback.sh not running?
pgrep -f "fallback.sh" # empty output = not running
# Both servers Tailscale connected?
tailscale status # both hosts should show active
```
### What the State File Contains
```bash
state=NORMAL
fallback_start=0
handback_strikes=0
tier2_started=false
tier3_started=false
tier4_started=false
```
---
## watchdog_skip_list_manager.sh
View and manage the persistent container skip list used by `docker_watchdog.sh`.
### When to Use
```
docker_watchdog.sh restarts the same container N times within the rolling window
→ container added to skip list on /boot/config/
→ critical notification sent
→ watchdog stops touching it entirely
You fix the underlying problem (database, config, dependencies).
You need to clear the container from the skip list so monitoring resumes.
```
### Recovery Workflow
```bash
# 1. Understand the situation — always start here:
watchdog_skip_list_manager.sh --status
# Shows: skip list contents, which are running vs. stopped, restart history
# 2. Fix the underlying problem first
# Check logs: docker logs ContainerName --tail 100
# Check disk: df -h /mnt/user
# Check db: docker exec ContainerName sqlite3 /path/to.db ".tables"
# 3. Clear from skip list + restart history:
watchdog_skip_list_manager.sh --clear ContainerName
# 4. Start the container manually — confirm your fix worked:
docker start ContainerName
# 5. Watchdog resumes normal monitoring on next cycle — no further action needed
```
### State Files Managed
```bash
# Both live on /boot/config — survive reboots intentionally.
# A container that was skip-listed before a reboot is still broken after it.
$SYS_WATCHDOG_FAILED_FILE # persistent skip list
$WATCHDOG_CONTAINER_RESTART_LOG # restart loop tracking
```
### Configuration (master.conf)
```bash
WATCHDOG_CONTAINER_RESTART_LIMIT=3 # restarts before skip-listing
WATCHDOG_CONTAINER_RESTART_WINDOW=1 # rolling window in hours
```
### Usage
```bash
watchdog_skip_list_manager.sh # show status (default)
watchdog_skip_list_manager.sh --status # explicit status
watchdog_skip_list_manager.sh --clear ContainerName # clear specific + restart history
watchdog_skip_list_manager.sh --clear ContainerName --force # no confirmation prompt
watchdog_skip_list_manager.sh --clear-all # clear everything
watchdog_skip_list_manager.sh --clear-all --force # non-interactive
watchdog_skip_list_manager.sh --dry-run # preview any clear action
```
---
## bulk_permissions_repair.sh
Applies correct ownership and permissions to specific paths. Faster than running
`media_shares_permissions.sh` which processes every configured share — use this when
you know exactly what needs fixing and don't want to wait for a full library walk.
### When to Use
```
Admin copy left root:root files — scp, cp, direct file transfer
New share needs permissions now — can't wait for nightly run
Container wrote as root — before PUID/PGID was fixed
Specific directory has wrong perms — targeted fix, not a full library walk
```
Use the full `media_shares_permissions.sh` instead for:
- Regular nightly maintenance (already scheduled in daily_sync_maintenance.sh)
- After confirming a container's PUID/PGID is now correct
- Initial permissions setup on a new server
### Diagnosing High Wrong-Owner Counts
The script counts files with wrong ownership before applying the fix. A high count
on a share that was recently written means a container has wrong PUID/PGID.
```bash
# Fix: add to the container's Docker template:
PUID=99
PGID=100
# Common culprits writing as root:
# SABnzbd, qBittorrent, slskd — check each one's Docker env vars
```
### Configuration (master.conf)
```bash
PERMISSIONS_OWNER="nobody:users" # matches PUID=99 PGID=100
PERMISSIONS_DIR_MODE="755" # directories — enter, list, no world-write
PERMISSIONS_FILE_MODE="664" # files — owner+group rw, others read-only
```
### Usage
```bash
# Single path:
bulk_permissions_repair.sh /mnt/user/Movies
# Multiple paths — all corrected in one run:
bulk_permissions_repair.sh /mnt/user/Movies /mnt/user/Tv_Shows /mnt/user/Music
# Dry run first — shows count of files with wrong ownership per path:
bulk_permissions_repair.sh /mnt/user/Movies --dry-run
# Verbose — show each corrected file:
bulk_permissions_repair.sh /mnt/user/Movies --log
```
---
## container_data_export.sh
Exports a container's appdata directory to a compressed tar archive. Stops the
container first for a clean consistent backup, verifies the archive after creation,
then restarts the container.
### When to Use
```
Before major container updates — especially "database migration — no rollback" changelogs
Before pool migrations — clean backup before moving appdata to a new pool
Before removing a container from the stack — archive its data before deletion
Manual point-in-time backup before risky config changes
```
### Export Sequence
```
1. Space check
Estimates required space from appdata size × 1.1
Aborts if output directory doesn't have enough free space
Container is NOT stopped until the space check passes
2. Stop container cleanly
docker stop ContainerName — graceful shutdown
3. Create archive
tar -czf ContainerName_YYYY-MM-DD_HH-MM.tar.gz /path/to/appdata
4. Verify archive integrity
tar --test-file archive.tar.gz — confirms archive is valid and complete
If verification fails → restart container anyway, report error
5. Restart container
docker start ContainerName — always happens, even if archiving failed
```
### Usage
```bash
# Syntax: container_data_export.sh ContainerName AppDataPath OutputDir
# Emby backup:
container_data_export.sh \
Emby \
/mnt/media-servers/Media_Server/Emby \
/mnt/user/Backups/
# Dry run — verify space and paths without stopping anything:
container_data_export.sh \
Emby \
/mnt/media-servers/Media_Server/Emby \
/mnt/user/Backups/ \
--dry-run
# Output filename: Emby_2026-05-14_02-30.tar.gz
# Timestamped — safe to run multiple times, no overwrite
```
---
## emby_database_repair.sh
Stops Emby, runs SQLite `PRAGMA integrity_check` on every Emby database, and restarts.
Reports per-database — does NOT automatically repair. Recovery requires judgment.
### When to Use
```
Emby logs show database errors → run this first
Emby crashing repeatedly with no clear cause → likely database corruption
Playback history or user data behaving strangely → users.db or library.db issue
After a hard shutdown or power loss with Emby running → check for WAL corruption
```
### Recovery Guide by Database
```
library.db — media library metadata: titles, seasons, episodes, artwork
CORRUPT → safe to delete — Emby fully rebuilds from media files on next start
Rebuild takes time (hours on large libraries) but loses nothing permanent
users.db — user accounts, watch history, playback positions, settings
CORRUPT → deleting resets ALL user accounts and watch history
Check for a recent backup (weekly_sync_maintenance.sh mirrors Emby/)
before deleting — restore from remote if available
authentication.db — API keys, session tokens
CORRUPT → safe to delete — API keys regenerated on restart
Any connected clients will need to re-authenticate once
activity.db — activity/access log
CORRUPT → safe to delete — it's a log, losing it is acceptable
library.db-wal — write-ahead log (uncommitted transactions)
PRESENT + CORRUPT → check library.db first; WAL corruption usually means
the main library.db is also affected
```
### Configuration (host*.conf)
```bash
HOST1_EMBY_CONTAINER="Emby" # aliased by detect_hosts() → EMBY_CONTAINER
HOST2_EMBY_CONTAINER="Emby"
```
Emby's config path is detected automatically from Docker volume mounts — no manual
path configuration needed.
### Usage
```bash
emby_database_repair.sh # stop Emby, check all databases, restart
emby_database_repair.sh --dry-run # show what would be checked, no Emby stop
emby_database_repair.sh --log # verbose — show SQLite output per database
emby_database_repair.sh --status # show Emby config path and database locations
```
---
## zfs_pool_scrub.sh
Triggers ZFS scrub on all pools (or a specific named pool) and waits for completion.
Notifies when done with a summary of any errors found.
### Why Run ZFS Scrub
ZFS stores a checksum with every block of data. Scrub reads every block and verifies
the checksum matches the stored hash. Silent data corruption can sit on disk for months
without triggering any error — until you try to read that specific file. By then:
- It may already be mirrored to HOST2 in its corrupted state
- The original source may no longer exist
- ZFS can self-repair during scrub if redundancy exists (RAIDZ or mirrors)
Run monthly. Also run after any disk replacement or power event.
Safe to run while the system is in use — scrub runs at low I/O priority.
### Configuration (host*.conf)
```bash
HOST1_ZFS_REPORT_IGNORE_POOLS=(
"disk10" # JBOD member — no redundancy, skipped from default scrub
"disk9"
"disk8"
)
HOST2_ZFS_REPORT_IGNORE_POOLS=(
"cache" # example — single-disk pool excluded from default
)
```
To scrub a pool in the ignore list, specify it by name explicitly.
### Usage
```bash
# Scrub all pools except those in ZFS_REPORT_IGNORE_POOLS:
zfs_pool_scrub.sh
# Scrub a specific pool by name — bypasses the ignore list:
zfs_pool_scrub.sh gaming
# Check current scrub status without starting a new one:
zfs_pool_scrub.sh --status
# Dry run — show which pools would be scrubbed:
zfs_pool_scrub.sh --dry-run
# Verbose — show scrub progress every 60s poll:
zfs_pool_scrub.sh --log
```
---
## recreate_shares.sh
Creates share directories on the correct disks after a fresh unRAID install or disk
rebuild. Run once on HOST2 before the first rsync from HOST1.
### When to Use
```
After a fresh unRAID install where /boot/config/shares/*.cfg were restored:
The share definitions exist → UI shows shares → directories are missing on disk
rsync.sh tries to write to /mnt/user/Movies → path doesn't exist → aborts
After a disk replacement or rebuild where share folders were lost:
Replacement disk is blank → no share directories on the new disk
unRAID won't create them automatically
```
### What It Does
```
For each .cfg file in /boot/config/shares/:
1. Read the share name (e.g., Movies)
2. Read the shareInclude list (e.g., disk1,disk2,disk5)
3. Create /mnt/disk1/Movies, /mnt/disk2/Movies, /mnt/disk5/Movies
4. Place a .recovery marker in /mnt/user/Movies/
The .recovery marker tells rsync.sh this is a fresh share:
.recovery present → rsync WITHOUT --delete (safe — new files only, nothing removed)
.recovery absent → rsync WITH --delete (normal mirror mode)
Self-cleaning: after the first successful rsync, the source side has no .recovery file,
so the second nightly run deletes it from the mirror and normal --delete resumes.
No manual cleanup needed.
```
### Usage
```bash
recreate_shares.sh # create all missing share directories + .recovery markers
recreate_shares.sh --dry-run # show what would be created without creating
recreate_shares.sh --log # verbose — show each directory created per disk
recreate_shares.sh --status # show share configs and current directory state
```
---
## continuous_scripts_status.sh
Live status dashboard for all continuously running scripts. Read-only — makes no
changes to any running process, container, or state file.
### What It Shows
```
stability_watchdog
Running state, PID, uptime, approximate cycle count
Active strikes, recent restart history
Live snapshot: rootfs, RAM, ZFS ARC, load, zombie count, CPU temp
docker_watchdog
Running state, PID, uptime
Running / stopped / unhealthy container counts
Required containers status
Memory-monitored containers
Recent restart history + skip list
fallback (fallback.sh)
Current state (NORMAL / FALLBACK / HANDBACK)
Tier flags and timestamps
Remote Tailscale visibility
```
State files are read as-is — if a script is mid-cycle, the display reflects the last
completed cycle, not the current in-progress state.
### Usage
```bash
continuous_scripts_status.sh # show full dashboard
continuous_scripts_status.sh --log # verbose output with additional detail per section
```
---
## claude_startup.sh
Restores Claude Code's persistent data after an unRAID reboot and optionally launches
Claude. Standalone script — no common.sh dependency.
### Why This Exists
unRAID's root filesystem lives in RAM — `/root/.claude` and `/root/.local` are wiped on
every reboot. This script symlinks both directories back to persistent appdata storage
at `/mnt/user/appdata/claude-code/` before launching Claude.
### First Run Migration
On first run, if persistent storage is empty, the script migrates from current live locations:
```
/root/.claude → /mnt/user/appdata/claude-code/.claude
/root/.local/share/claude → /mnt/user/appdata/claude-code/local/share/claude
```
Subsequent runs skip the migration and only create the symlinks.
### Calling from array_started.sh
`array_started.sh` calls `claude_startup.sh` directly (no flags). This sets up the
symlinks only — no interactive session is launched. That is the default behavior.
### Usage
```bash
claude_startup.sh # set up persistent symlinks only (default — used by array_started.sh)
claude_startup.sh --launch # set up symlinks and launch Claude interactively
```
---
## ramdisk_stop.sh
Safely stops the transcode ramdisk: redirects the transcode symlink to the SSD
fallback first (so Emby continues writing without interruption), then unmounts the
tmpfs and updates the state file. Primary use case is stopping the current ramdisk
before re-running `ramdisk_setup.sh` with new size or threshold values.
### When to Use
```
Bumping RAMDISK_SIZE — setup script is idempotent, skips remount if already mounted
→ stop first, then re-run ramdisk_setup.sh with new HOST*_RAMDISK_SIZE value
Adjusting RAMDISK_WARN_GB / RAMDISK_LOW_GB thresholds
→ no need to stop for threshold changes (transcode_manager reads vars live)
→ only needed if you're also changing the size
Temporarily freeing ramdisk RAM — reclaim tmpfs back to general memory pool
→ stop, restart later with ramdisk_setup.sh
```
### Stop Sequence
```
1. Redirect symlink: TRANSCODE_LINK → TRANSCODE_SSD
Emby immediately writes to SSD — no broken-path window during unmount
2. Check for active transcode files on ramdisk (warn, don't block)
Files in progress on the ramdisk are lost on unmount — expected for maintenance
3. Unmount ramdisk
Regular umount first; if busy (directory handles only, no active writes)
falls back to lazy unmount automatically
4. Update /tmp/transcode_state.db → current_target=TRANSCODE_SSD
transcode_manager.sh reads this on its next cycle
```
### transcode_manager Warning
If `transcode_manager.sh` is running, it may flip the symlink back to the ramdisk
on its next cycle (once the ramdisk is unmounted, that flip will fail). Stop
`transcode_manager.sh` first if you need the SSD redirect to hold before remounting.
### After Stopping
```bash
# Update host*.conf with new size values:
# HOST1_RAMDISK_SIZE="10G"
# HOST1_RAMDISK_WARN_GB=8.5
# HOST1_RAMDISK_LOW_GB=7
# Remount at new size:
bash Transcodes/ramdisk_setup.sh
```
### Usage
```bash
ramdisk_stop.sh --status # show mount state, symlink, active files — always check first
ramdisk_stop.sh --dry-run # show what would happen without making changes
ramdisk_stop.sh # stop the ramdisk
ramdisk_stop.sh --log # verbose — show each step
```
---
## docker_prune_images.sh
Removes orphaned Docker images that accumulate after container updates. Two modes:
**Default (dangling only)** — removes untagged images (no name, no container reference).
Safe — running containers are never affected. Use routinely after update cycles.
**`--all` (full orphan cleanup)** — first removes stopped/exited containers, then removes
all images not used by any running container. Use when you've removed apps and want to
recover the disk space. CAUTION: also removes intentionally stopped containers.
### Usage
```bash
docker_prune_images.sh # remove dangling (untagged) images only
docker_prune_images.sh --all # remove stopped containers, then all unused images
docker_prune_images.sh --dry-run # show what would be removed without making changes
docker_prune_images.sh --status # show dangling images and stopped containers
```
---
## Adding a New Tool
Write the tool when you solve a problem manually with bash commands. You'll face it again.
The cost of writing the tool is 30 minutes. The cost of reconstructing the commands at 2am
is much higher.
### Checklist
```
✓ Header explains the specific situation that requires this tool
✓ Root check — most tools need root
✓ --dry-run support — always
✓ --status support — show current state before acting
✓ Confirmation for destructive operations (interactive YES or --force flag)
✓ Notify on completion — success and failure
✓ Leave system in clean state on any exit — trap for cleanup
✓ Add to README-Tools.md scripts table and HOW THE SCRIPTS RELATE diagram
```
### Minimal Skeleton
```bash
#!/bin/bash
# ==============================================================================================
# ============================= Your Tool Name ================================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# One sentence: what situation this solves and when to use it.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Root Required
# chown / docker / etc. require root.
#
# Confirmation Required
# Interactive mode prompts for YES. Use --force to bypass in scripts.
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# your_tool.sh
# Normal run.
#
# your_tool.sh --dry-run
# Preview without making changes.
#
# your_tool.sh --status
# Show current state and exit.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
if [[ "$EUID" -ne 0 ]]; then error "Must be run as root"; exit 1; fi
platform_require_cmd \
"/usr/local/emhttp/plugins/dynamix/scripts/notify" \
"" "" "unRAID notify script" || warn "notify not found — notifications disabled"
acquire_lock
detect_hosts
if [[ "$SHOW_STATUS" == true ]]; then
log "Current state: ..."
exit 0
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes will be made"
if [[ "$FORCE" != true ]]; then
read -r -p "Type YES to proceed: " CONFIRM
[[ "$CONFIRM" != "YES" ]] && { warn "Aborted."; exit 0; }
fi
# Do the work
# ...
notify "Tool completed on $(hostname) ($MY_ID)" "Tool Name" "normal"
```
@@ -1,158 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= Conf Cache Sync ================================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Maintains a RAM-resident conf cache at /tmp/.vv/config/cached/.confs/.
# Credentials and partner keys live in RAM only — never on disk across hosts.
#
# On array start (default / --array-start):
# 1. Copy own conf to local cache
# 2. Pull each available partner's conf from their disk → local cache
# 3. Push own conf to each available partner's /tmp/.vv/ cache
#
# On conf save (--push-only):
# Fast path — push updated own conf to all partners' /tmp/.vv/ cache only.
# No pulls, no local cache rebuild.
#
# Cache is /tmp (tmpfs) — cleared every reboot, repopulated by this script
# on next array start. Scripts source from cache for partner vars; own vars
# always come from disk (load_config.sh skips cached copy of own conf).
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# conf_sync.sh Full sync: pull from all partners + push to all partners
# conf_sync.sh --push-only Push own conf to all partners (fast, for conf-save hook)
# conf_sync.sh --pull-only Pull partner confs into local cache only (for intermediate orch)
# conf_sync.sh --dry-run Show what would happen, no changes
# conf_sync.sh --log Verbose output
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
PUSH_ONLY=false
PULL_ONLY=false
FILTERED_ARGS=()
for arg in "$@"; do
case "$arg" in
--push-only) PUSH_ONLY=true ;;
--pull-only) PULL_ONLY=true ;;
*) FILTERED_ARGS+=("$arg") ;;
esac
done
parse_args "${FILTERED_ARGS[@]}"
detect_hosts
if [[ "${CONF_SYNC_ENABLED:-true}" == false ]]; then
log "CONF_SYNC_ENABLED=false — skipping"
exit 0
fi
CACHE_DIR="/tmp/.vv/config/cached/.confs"
MY_CONF="$SCRIPTS_ROOT/Configurations/${MY_ID,,}.conf"
SSH_TIMEOUT=10
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes will be made"
# ── Ensure cache dir exists ───────────────────────────────────────────────────
if [[ "$DRY_RUN" == false ]]; then
mkdir -p "$CACHE_DIR"
fi
# ── Copy own conf into local cache ───────────────────────────────────────────
if [[ "$PUSH_ONLY" == false ]] && [[ "$PULL_ONLY" == false ]]; then
if [[ -f "$MY_CONF" ]]; then
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would copy $(basename "$MY_CONF") → $CACHE_DIR/"
else
cp "$MY_CONF" "$CACHE_DIR/${MY_ID,,}.conf" && \
log "Own conf cached ✅" || warn "Failed to cache own conf"
fi
else
warn "Own conf not found: $MY_CONF"
fi
fi
# ── Per-partner sync ──────────────────────────────────────────────────────────
PUSHED=0
PULLED=0
FAILED=0
for host_var in $(compgen -v | grep -E '^HOST[0-9]+$' | sort); do
partner_host="${!host_var}"
[[ -z "$partner_host" ]] && continue
[[ "${host_var,,}" == "${MY_ID,,}" ]] && continue
partner_slot="${host_var,,}" # e.g. host2
partner_ip=$(resolve_tailscale_ip "$partner_host" 2>/dev/null || true)
if [[ -z "$partner_ip" ]]; then
warn "$partner_host — cannot resolve Tailscale IP, skipping"
(( FAILED++ ))
continue
fi
# ── Pull: grab partner's conf from their disk → our local cache ──────────
if [[ "$PUSH_ONLY" == false ]]; then
remote_conf="/boot/config/plugins/varaverk/Configurations/${partner_slot}.conf"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would pull $partner_host:$remote_conf → $CACHE_DIR/${partner_slot}.conf"
elif timeout "$SSH_TIMEOUT" scp -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes -o StrictHostKeyChecking=no \
"root@${partner_ip}:${remote_conf}" \
"$CACHE_DIR/${partner_slot}.conf" 2>/dev/null; then
log "Pulled ${partner_slot}.conf from $partner_host ✅"
(( PULLED++ ))
else
warn "Could not pull ${partner_slot}.conf from $partner_host"
(( FAILED++ ))
fi
fi
# ── Push: send own conf to partner's /tmp/.vv/ cache ────────────────────
if [[ "$PULL_ONLY" == true ]]; then
continue
fi
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would push ${MY_ID,,}.conf → $partner_host:/tmp/.vv/config/cached/.confs/"
continue
fi
# Ensure partner's cache dir exists, then SCP own conf into it
timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes -o StrictHostKeyChecking=no \
"root@${partner_ip}" "mkdir -p '$CACHE_DIR'" 2>/dev/null
if timeout "$SSH_TIMEOUT" scp -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes -o StrictHostKeyChecking=no \
"$MY_CONF" \
"root@${partner_ip}:${CACHE_DIR}/${MY_ID,,}.conf" 2>/dev/null; then
log "Pushed ${MY_ID,,}.conf to $partner_host ✅"
(( PUSHED++ ))
else
warn "Could not push to $partner_host"
(( FAILED++ ))
fi
done
# ── Summary ───────────────────────────────────────────────────────────────────
if [[ "$PUSH_ONLY" == true ]]; then
info "Conf push complete — pushed to $PUSHED host(s)${FAILED:+, $FAILED failed}"
elif [[ "$PULL_ONLY" == true ]]; then
info "Conf pull complete — pulled $PULLED partner conf(s)${FAILED:+, $FAILED failed}"
else
info "Conf sync complete — pulled $PULLED, pushed $PUSHED${FAILED:+, $FAILED failed}"
fi
if [[ "$FAILED" -gt 0 ]]; then
notify "Conf sync on $LOCAL_SERVER_NAME ($MY_ID) — $FAILED partner(s) failed. Partner config cache may be stale." \
"Conf Sync" "warning"
exit 1
fi
@@ -1,158 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= Conf Cache Sync ================================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Maintains a RAM-resident conf cache at /tmp/.vv/config/cached/.confs/.
# Credentials and partner keys live in RAM only — never on disk across hosts.
#
# On array start (default / --array-start):
# 1. Copy own conf to local cache
# 2. Pull each available partner's conf from their disk → local cache
# 3. Push own conf to each available partner's /tmp/.vv/ cache
#
# On conf save (--push-only):
# Fast path — push updated own conf to all partners' /tmp/.vv/ cache only.
# No pulls, no local cache rebuild.
#
# Cache is /tmp (tmpfs) — cleared every reboot, repopulated by this script
# on next array start. Scripts source from cache for partner vars; own vars
# always come from disk (load_config.sh skips cached copy of own conf).
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# conf_sync.sh Full sync: pull from all partners + push to all partners
# conf_sync.sh --push-only Push own conf to all partners (fast, for conf-save hook)
# conf_sync.sh --pull-only Pull partner confs into local cache only (for intermediate orch)
# conf_sync.sh --dry-run Show what would happen, no changes
# conf_sync.sh --log Verbose output
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
PUSH_ONLY=false
PULL_ONLY=false
FILTERED_ARGS=()
for arg in "$@"; do
case "$arg" in
--push-only) PUSH_ONLY=true ;;
--pull-only) PULL_ONLY=true ;;
*) FILTERED_ARGS+=("$arg") ;;
esac
done
parse_args "${FILTERED_ARGS[@]}"
detect_hosts
if [[ "${CONF_SYNC_ENABLED:-true}" == false ]]; then
log "CONF_SYNC_ENABLED=false — skipping"
exit 0
fi
CACHE_DIR="/tmp/.vv/config/cached/.confs"
MY_CONF="$SCRIPTS_ROOT/Configurations/${MY_ID,,}.conf"
SSH_TIMEOUT=10
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes will be made"
# ── Ensure cache dir exists ───────────────────────────────────────────────────
if [[ "$DRY_RUN" == false ]]; then
mkdir -p "$CACHE_DIR"
fi
# ── Copy own conf into local cache ───────────────────────────────────────────
if [[ "$PUSH_ONLY" == false ]] && [[ "$PULL_ONLY" == false ]]; then
if [[ -f "$MY_CONF" ]]; then
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would copy $(basename "$MY_CONF") → $CACHE_DIR/"
else
cp "$MY_CONF" "$CACHE_DIR/${MY_ID,,}.conf" && \
log "Own conf cached ✅" || warn "Failed to cache own conf"
fi
else
warn "Own conf not found: $MY_CONF"
fi
fi
# ── Per-partner sync ──────────────────────────────────────────────────────────
PUSHED=0
PULLED=0
FAILED=0
for host_var in $(compgen -v | grep -E '^HOST[0-9]+$' | sort); do
partner_host="${!host_var}"
[[ -z "$partner_host" ]] && continue
[[ "${host_var,,}" == "${MY_ID,,}" ]] && continue
partner_slot="${host_var,,}" # e.g. host2
partner_ip=$(resolve_tailscale_ip "$partner_host" 2>/dev/null || true)
if [[ -z "$partner_ip" ]]; then
warn "$partner_host — cannot resolve Tailscale IP, skipping"
(( FAILED++ ))
continue
fi
# ── Pull: grab partner's conf from their disk → our local cache ──────────
if [[ "$PUSH_ONLY" == false ]]; then
remote_conf="${SCRIPTS_DIR}/Configurations/${partner_slot}.conf"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would pull $partner_host:$remote_conf → $CACHE_DIR/${partner_slot}.conf"
elif timeout "$SSH_TIMEOUT" scp -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes -o StrictHostKeyChecking=no \
"root@${partner_ip}:${remote_conf}" \
"$CACHE_DIR/${partner_slot}.conf" 2>/dev/null; then
log "Pulled ${partner_slot}.conf from $partner_host ✅"
(( PULLED++ ))
else
warn "Could not pull ${partner_slot}.conf from $partner_host"
(( FAILED++ ))
fi
fi
# ── Push: send own conf to partner's /tmp/.vv/ cache ────────────────────
if [[ "$PULL_ONLY" == true ]]; then
continue
fi
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would push ${MY_ID,,}.conf → $partner_host:/tmp/.vv/config/cached/.confs/"
continue
fi
# Ensure partner's cache dir exists, then SCP own conf into it
timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes -o StrictHostKeyChecking=no \
"root@${partner_ip}" "mkdir -p '$CACHE_DIR'" 2>/dev/null
if timeout "$SSH_TIMEOUT" scp -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes -o StrictHostKeyChecking=no \
"$MY_CONF" \
"root@${partner_ip}:${CACHE_DIR}/${MY_ID,,}.conf" 2>/dev/null; then
log "Pushed ${MY_ID,,}.conf to $partner_host ✅"
(( PUSHED++ ))
else
warn "Could not push to $partner_host"
(( FAILED++ ))
fi
done
# ── Summary ───────────────────────────────────────────────────────────────────
if [[ "$PUSH_ONLY" == true ]]; then
info "Conf push complete — pushed to $PUSHED host(s)${FAILED:+, $FAILED failed}"
elif [[ "$PULL_ONLY" == true ]]; then
info "Conf pull complete — pulled $PULLED partner conf(s)${FAILED:+, $FAILED failed}"
else
info "Conf sync complete — pulled $PULLED, pushed $PUSHED${FAILED:+, $FAILED failed}"
fi
if [[ "$FAILED" -gt 0 ]]; then
notify "Conf sync on $LOCAL_SERVER_NAME ($MY_ID) — $FAILED partner(s) failed. Partner config cache may be stale." \
"Conf Sync" "warning"
exit 1
fi
@@ -1,358 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= ZFS Memory Snapshot ============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Weekly ZFS pool health and memory diagnostic report. Scheduled Sunday 6am —
# first in the Sunday monitoring block, before other scripts run. Informational
# only — system_watchdog.sh handles threshold-based intervention.
#
# Combines ZFS pool status, ARC statistics, Docker memory usage, and kernel
# memory pressure into a single snapshot. Output goes to both console (for User
# Scripts output log) and ZFS_REPORT_LOG for week-over-week comparison.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# Five report sections (each skips gracefully if its data source is unavailable):
#
# ZFS pool health — status, state, errors per pool. Pools in
# ZFS_REPORT_IGNORE_POOLS excluded from the report
# (still fully monitored by unRAID — report-only exclusion).
# ARC statistics — current ARC vs max, metadata pressure, hit rate.
# Warns if ARC utilisation exceeds ZFS_REPORT_ARC_WARN_PCT.
# Memory status — total, free, available RAM.
# Warns if free < ZFS_REPORT_FREE_WARN_GB or
# available < ZFS_REPORT_AVAIL_WARN_GB.
# Docker memory — top ZFS_REPORT_DOCKER_TOP containers by memory usage.
# Useful for spotting containers approaching watchdog limits.
# Kernel pressure — vmstat snapshot (3 samples).
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Single Instance Lock
# acquire_lock prevents duplicate runs — zpool and docker stats are slow.
#
# Per-Host Pool Ignore List
# detect_hosts() aliases HOST*_ZFS_REPORT_IGNORE_POOLS → ZFS_REPORT_IGNORE_POOLS.
# Single-disk JBOD members excluded from report noise per server.
#
# ZFS Availability Guard
# Skips pool and ARC sections gracefully if ZFS is not available on this server.
#
# Docker Availability Guard
# Skips container memory section gracefully if Docker is not responding.
#
# Docker Stats Timeout
# DOCKER_TIMEOUT caps docker stats calls. A hung daemon does not block the report.
#
# Notification Validated
# platform_require_cmd confirms the notify script is present before use.
#
# ==============================================================================================
# STATE FILES
# ==============================================================================================
#
# ZFS_REPORT_LOG — /var/log/zfs-weekly-health.log (tmpfs, resets on reboot)
# Weekly report written here for comparison across runs. Open the log to
# see pool health trend week over week without remembering last week's values.
# In dry-run mode, console only — nothing written.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_ZFS_REPORT_IGNORE_POOLS
# Pools excluded from health reporting. Single-disk JBOD members generate
# expected high-usage warnings — exclude them to reduce report noise.
# Aliased by detect_hosts() → ZFS_REPORT_IGNORE_POOLS.
#
# master.conf
#
# ZFS_REPORT_LOG
# Log file path for weekly reports. (default: /var/log/zfs-weekly-health.log)
#
# ZFS_REPORT_ARC_WARN_PCT
# Warn if ARC is using more than this percentage of its configured max. (default: 90)
#
# ZFS_REPORT_FREE_WARN_GB
# Warn if free RAM is below this threshold in GB. (default: 10)
#
# ZFS_REPORT_AVAIL_WARN_GB
# Warn if available RAM is below this threshold in GB. (default: 20)
#
# ZFS_REPORT_DOCKER_TOP
# Number of top Docker containers by memory usage to include. (default: 10)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# zfs_memory_snapshot.sh
# Generate report, write to ZFS_REPORT_LOG and console. Notify on warnings.
#
# zfs_memory_snapshot.sh --dry-run
# Generate report to console only. No log write, no notifications.
#
# zfs_memory_snapshot.sh --status
# Show pool ignore list and threshold configuration. Then exit.
#
# zfs_memory_snapshot.sh --log
# Verbose output during report generation.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
DOCKER_TIMEOUT=15
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
acquire_lock
# detect_hosts() sets MY_ID and aliases HOST*_ZFS_REPORT_IGNORE_POOLS
detect_hosts
# Build ignore pool lookup map — O(1) check per pool
declare -A IGNORE_POOL_MAP
for pool in "${ZFS_REPORT_IGNORE_POOLS[@]}"; do
[[ -n "$pool" ]] && IGNORE_POOL_MAP["$pool"]=1
done
log "Identity: $MY_ID ($LOCAL_SERVER_NAME)"
log "Ignoring pools: ${ZFS_REPORT_IGNORE_POOLS[*]:-none}"
log "$ICON_GEAR Config: arc-warn=${ZFS_REPORT_ARC_WARN_PCT}% free-warn=${ZFS_REPORT_FREE_WARN_GB}GB avail-warn=${ZFS_REPORT_AVAIL_WARN_GB}GB docker-top=${ZFS_REPORT_DOCKER_TOP}"
# Tee output to log file unless dry run
if [[ "$DRY_RUN" == false ]]; then
mkdir -p "$(dirname "$ZFS_REPORT_LOG")"
exec > >(tee -a "$ZFS_REPORT_LOG") 2>&1
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — output will not be written to log"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_ZFS Log file: $ZFS_REPORT_LOG"
echo "$ICON_ZFS ARC warn: ${ZFS_REPORT_ARC_WARN_PCT}%"
echo "$ICON_MEM Free RAM warn: ${ZFS_REPORT_FREE_WARN_GB}GB"
echo "$ICON_MEM Avail warn: ${ZFS_REPORT_AVAIL_WARN_GB}GB"
echo "$ICON_CONTAINERS Docker top: $ZFS_REPORT_DOCKER_TOP"
echo "$ICON_ZFS Ignore pools: ${ZFS_REPORT_IGNORE_POOLS[*]:-none}"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ━━━ Report ━━━
# ==============================================================================================
WARNINGS=()
START=$(date +%s)
DATE=$(date '+%Y-%m-%d %H:%M:%S')
echo ""
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo " $ICON_ZFS ZFS WEEKLY HEALTH REPORT — $DATE"
echo " $ICON_HOST $MY_ID — $LOCAL_SERVER_NAME"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
# ── ZFS Pool Health ───────────────────────────────────────────────────────────────────────────
echo ""
echo "━━━ $ICON_ZFS ZFS Pool Health ━━━"
if ! command -v zpool >/dev/null 2>&1; then
warn "ZFS not available on this system — skipping pool checks"
else
# Pool status — filtered to key lines, ignoring specified pools
CURRENT_POOL=""
while IFS= read -r line; do
if [[ "$line" =~ ^[[:space:]]*pool:[[:space:]]*(.+) ]]; then
CURRENT_POOL="${BASH_REMATCH[1]// /}"
fi
[[ -n "${IGNORE_POOL_MAP[$CURRENT_POOL]:-}" ]] && continue
echo " $line"
done < <(zpool status 2>/dev/null | grep -E "pool:|state:|status:|errors:|scan:")
echo ""
# Pool list — filter out ignored pools
zpool list 2>/dev/null | while IFS= read -r line; do
if [[ "$line" == NAME* ]]; then
echo " $line"
continue
fi
pool_name=$(echo "$line" | awk '{print $1}')
[[ -n "${IGNORE_POOL_MAP[$pool_name]:-}" ]] && continue
echo " $line"
done
# Check for unhealthy non-ignored pools
UNHEALTHY=$(zpool list -H -o name,health 2>/dev/null | \
while IFS=$'\t' read -r name health; do
[[ -n "${IGNORE_POOL_MAP[$name]:-}" ]] && continue
[[ "$health" != "ONLINE" ]] && echo "$name: $health"
done)
if [[ -n "$UNHEALTHY" ]]; then
error "One or more ZFS pools are NOT ONLINE: $UNHEALTHY"
WARNINGS+=("ZFS pool unhealthy: $UNHEALTHY")
else
echo "All monitored ZFS pools are ONLINE ✅"
fi
if [[ ${#ZFS_REPORT_IGNORE_POOLS[@]} -gt 0 ]]; then
log "Ignored pools: ${ZFS_REPORT_IGNORE_POOLS[*]}"
fi
fi
# ── ARC Statistics ────────────────────────────────────────────────────────────────────────────
echo ""
echo "━━━ $ICON_ZFS ARC Statistics ━━━"
if [[ ! -f /proc/spl/kstat/zfs/arcstats ]]; then
warn "ZFS arcstats not available — skipping ARC section"
else
ARC_MAX=$(cat /sys/module/zfs/parameters/zfs_arc_max 2>/dev/null || \
awk '/^c_max / {print $3}' /proc/spl/kstat/zfs/arcstats)
ARC_SIZE=$(awk '/^size / {print $3}' /proc/spl/kstat/zfs/arcstats)
ARC_META_USED=$(awk '/^arc_meta_used / {print $3}' /proc/spl/kstat/zfs/arcstats)
ARC_MAX_GB=$(awk "BEGIN {printf \"%.1f\", $ARC_MAX / 1073741824}")
ARC_CUR_GB=$(awk "BEGIN {printf \"%.1f\", $ARC_SIZE / 1073741824}")
ARC_META_GB=$(awk "BEGIN {printf \"%.1f\", $ARC_META_USED / 1073741824}")
ARC_PCT=$(awk "BEGIN {printf \"%.1f\", $ARC_SIZE * 100 / $ARC_MAX}")
ARC_PCT_INT=$(printf "%.0f" "$ARC_PCT")
echo " $ICON_ZFS ARC Max: ${ARC_MAX_GB}GB"
echo " $ICON_ZFS ARC Current: ${ARC_CUR_GB}GB"
echo " $ICON_ZFS ARC Meta Used: ${ARC_META_GB}GB"
echo " $ICON_ZFS ARC Utilization: ${ARC_PCT}%"
if [[ "$ARC_PCT_INT" -ge "$ZFS_REPORT_ARC_WARN_PCT" ]]; then
warn "ARC utilization ${ARC_PCT}% — above ${ZFS_REPORT_ARC_WARN_PCT}% threshold"
WARNINGS+=("ARC high: ${ARC_PCT}%")
else
log "ARC utilization ${ARC_PCT}% — within threshold ✅"
fi
echo ""
META_MRU_GHOST=$(awk '/^mru_ghost_metadata / {print $3}' \
/proc/spl/kstat/zfs/arcstats 2>/dev/null || echo 0)
META_MFU_GHOST=$(awk '/^mfu_ghost_metadata / {print $3}' \
/proc/spl/kstat/zfs/arcstats 2>/dev/null || echo 0)
META_MISSES=$(awk '/^demand_metadata_misses / {print $3}' \
/proc/spl/kstat/zfs/arcstats 2>/dev/null || echo 0)
MRU_GB=$(awk "BEGIN {printf \"%.2f\", $META_MRU_GHOST / 1073741824}")
MFU_GB=$(awk "BEGIN {printf \"%.2f\", $META_MFU_GHOST / 1073741824}")
echo " $ICON_ZFS MRU Ghost: ${MRU_GB}GB"
echo " $ICON_ZFS MFU Ghost: ${MFU_GB}GB"
echo " $ICON_ZFS Metadata Misses: ${META_MISSES}"
fi
# ── Memory Status ─────────────────────────────────────────────────────────────────────────────
echo ""
echo "━━━ $ICON_MEM Memory Status ━━━"
FREE_HUMAN=$(free -h | awk '/Mem:/ {print $4}')
AVAIL_HUMAN=$(free -h | awk '/Mem:/ {print $7}')
TOTAL_HUMAN=$(free -h | awk '/Mem:/ {print $2}')
FREE_GB=$(free -g | awk '/Mem:/ {print $4}')
AVAIL_GB=$(free -g | awk '/Mem:/ {print $7}')
echo " $ICON_MEM Total RAM: $TOTAL_HUMAN"
echo " $ICON_MEM Free RAM: $FREE_HUMAN"
echo " $ICON_MEM Available RAM: $AVAIL_HUMAN"
if [[ "$FREE_GB" -lt "$ZFS_REPORT_FREE_WARN_GB" ]]; then
warn "Free RAM ${FREE_HUMAN} — below ${ZFS_REPORT_FREE_WARN_GB}GB threshold"
WARNINGS+=("Low free RAM: ${FREE_HUMAN}")
else
log "Free RAM ${FREE_HUMAN} — within threshold ✅"
fi
if [[ "$AVAIL_GB" -lt "$ZFS_REPORT_AVAIL_WARN_GB" ]]; then
warn "Available RAM ${AVAIL_HUMAN} — below ${ZFS_REPORT_AVAIL_WARN_GB}GB threshold"
WARNINGS+=("Low available RAM: ${AVAIL_HUMAN}")
else
log "Available RAM ${AVAIL_HUMAN} — within threshold ✅"
fi
# ── Docker Memory ─────────────────────────────────────────────────────────────────────────────
echo ""
echo "━━━ $ICON_CONTAINERS Top $ZFS_REPORT_DOCKER_TOP Docker Memory Users ━━━"
if ! command -v docker >/dev/null 2>&1; then
warn "Docker not available — skipping container memory section"
else
timeout "$DOCKER_TIMEOUT" docker stats --no-stream \
--format "table {{.Name}}\t{{.MemUsage}}\t{{.MemPerc}}" \
2>/dev/null | head -n $(( ZFS_REPORT_DOCKER_TOP + 1 )) | \
while IFS= read -r line; do
echo " $line"
done
fi
# ── Kernel Pressure ───────────────────────────────────────────────────────────────────────────
echo ""
echo "━━━ $ICON_GEAR Kernel Pressure ━━━"
if ! command -v vmstat >/dev/null 2>&1; then
warn "vmstat not available — skipping kernel pressure section"
else
vmstat 1 3 2>/dev/null | while IFS= read -r line; do
echo " $line"
done
fi
END=$(date +%s)
# ── Summary ───────────────────────────────────────────────────────────────────────────────────
echo ""
echo "━━━━━ $ICON_SUMMARY ZFS REPORT SUMMARY ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_TIME Duration: $(format_duration $(( END - START )))"
echo "$ICON_ZFS Log: $ZFS_REPORT_LOG"
[[ ${#ZFS_REPORT_IGNORE_POOLS[@]} -gt 0 ]] && \
log "Ignored: ${ZFS_REPORT_IGNORE_POOLS[*]}"
echo ""
if [[ ${#WARNINGS[@]} -eq 0 ]]; then
echo "$ICON_DONE All checks within thresholds ✅"
else
echo "$ICON_WARN Warnings: ${#WARNINGS[@]}"
for w in "${WARNINGS[@]}"; do
echo " $ICON_WARN $w"
done
notify "ZFS weekly report on $(hostname) — ${#WARNINGS[@]} warning(s): ${WARNINGS[*]}" \
"ZFS Report" "warning"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
[[ ${#WARNINGS[@]} -gt 0 ]] && exit 1
exit 0
@@ -1,359 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ============================= ZFS Memory Snapshot ============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Weekly ZFS pool health and memory diagnostic report. Scheduled Sunday 6am —
# first in the Sunday monitoring block, before other scripts run. Informational
# only — system_watchdog.sh handles threshold-based intervention.
#
# Combines ZFS pool status, ARC statistics, Docker memory usage, and kernel
# memory pressure into a single snapshot. In normal mode output goes to both
# console (for User Scripts output log) and ZFS_REPORT_LOG for week-over-week
# comparison. In --dry-run mode, console only — nothing written to the log.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# Five report sections (each skips gracefully if its data source is unavailable):
#
# ZFS pool health — status, state, errors per pool. Pools in
# ZFS_REPORT_IGNORE_POOLS excluded from the report
# (still fully monitored by unRAID — report-only exclusion).
# ARC statistics — current ARC vs max, metadata pressure, hit rate.
# Warns if ARC utilisation exceeds ZFS_REPORT_ARC_WARN_PCT.
# Memory status — total, free, available RAM.
# Warns if free < ZFS_REPORT_FREE_WARN_GB or
# available < ZFS_REPORT_AVAIL_WARN_GB.
# Docker memory — top ZFS_REPORT_DOCKER_TOP containers by memory usage.
# Useful for spotting containers approaching watchdog limits.
# Kernel pressure — vmstat snapshot (3 samples).
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Single Instance Lock
# acquire_lock prevents duplicate runs — zpool and docker stats are slow.
#
# Per-Host Pool Ignore List
# detect_hosts() aliases HOST*_ZFS_REPORT_IGNORE_POOLS → ZFS_REPORT_IGNORE_POOLS.
# Single-disk JBOD members excluded from report noise per server.
#
# ZFS Availability Guard
# Skips pool and ARC sections gracefully if ZFS is not available on this server.
#
# Docker Availability Guard
# Skips container memory section gracefully if Docker is not responding.
#
# Docker Stats Timeout
# DOCKER_TIMEOUT caps docker stats calls. A hung daemon does not block the report.
#
# Notification Validated
# platform_require_cmd confirms the notify script is present before use.
#
# ==============================================================================================
# STATE FILES
# ==============================================================================================
#
# ZFS_REPORT_LOG — /var/log/zfs-weekly-health.log (tmpfs, resets on reboot)
# Weekly report written here for comparison across runs. Open the log to
# see pool health trend week over week without remembering last week's values.
# In dry-run mode, console only — nothing written.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_ZFS_REPORT_IGNORE_POOLS
# Pools excluded from health reporting. Single-disk JBOD members generate
# expected high-usage warnings — exclude them to reduce report noise.
# Aliased by detect_hosts() → ZFS_REPORT_IGNORE_POOLS.
#
# master.conf
#
# ZFS_REPORT_LOG
# Log file path for weekly reports. (default: /var/log/zfs-weekly-health.log)
#
# ZFS_REPORT_ARC_WARN_PCT
# Warn if ARC is using more than this percentage of its configured max. (default: 90)
#
# ZFS_REPORT_FREE_WARN_GB
# Warn if free RAM is below this threshold in GB. (default: 10)
#
# ZFS_REPORT_AVAIL_WARN_GB
# Warn if available RAM is below this threshold in GB. (default: 20)
#
# ZFS_REPORT_DOCKER_TOP
# Number of top Docker containers by memory usage to include. (default: 10)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# zfs_memory_snapshot.sh
# Generate report, write to ZFS_REPORT_LOG and console. Notify on warnings.
#
# zfs_memory_snapshot.sh --dry-run
# Generate report to console only. No log write, no notifications.
#
# zfs_memory_snapshot.sh --status
# Show pool ignore list and threshold configuration. Then exit.
#
# zfs_memory_snapshot.sh --log
# Verbose output during report generation.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
DOCKER_TIMEOUT=15
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
acquire_lock
# detect_hosts() sets MY_ID and aliases HOST*_ZFS_REPORT_IGNORE_POOLS
detect_hosts
# Build ignore pool lookup map — O(1) check per pool
declare -A IGNORE_POOL_MAP
for pool in "${ZFS_REPORT_IGNORE_POOLS[@]}"; do
[[ -n "$pool" ]] && IGNORE_POOL_MAP["$pool"]=1
done
log "Identity: $MY_ID ($LOCAL_SERVER_NAME)"
log "Ignoring pools: ${ZFS_REPORT_IGNORE_POOLS[*]:-none}"
log "$ICON_GEAR Config: arc-warn=${ZFS_REPORT_ARC_WARN_PCT}% free-warn=${ZFS_REPORT_FREE_WARN_GB}GB avail-warn=${ZFS_REPORT_AVAIL_WARN_GB}GB docker-top=${ZFS_REPORT_DOCKER_TOP}"
# Tee output to log file unless dry run
if [[ "$DRY_RUN" == false ]]; then
mkdir -p "$(dirname "$ZFS_REPORT_LOG")"
exec > >(tee -a "$ZFS_REPORT_LOG") 2>&1
fi
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — output will not be written to log"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_ZFS Log file: $ZFS_REPORT_LOG"
echo "$ICON_ZFS ARC warn: ${ZFS_REPORT_ARC_WARN_PCT}%"
echo "$ICON_MEM Free RAM warn: ${ZFS_REPORT_FREE_WARN_GB}GB"
echo "$ICON_MEM Avail warn: ${ZFS_REPORT_AVAIL_WARN_GB}GB"
echo "$ICON_CONTAINERS Docker top: $ZFS_REPORT_DOCKER_TOP"
echo "$ICON_ZFS Ignore pools: ${ZFS_REPORT_IGNORE_POOLS[*]:-none}"
echo "$ICON_GEAR Dry Run: $DRY_RUN"
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ━━━ Report ━━━
# ==============================================================================================
WARNINGS=()
START=$(date +%s)
DATE=$(date '+%Y-%m-%d %H:%M:%S')
echo ""
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo " $ICON_ZFS ZFS WEEKLY HEALTH REPORT — $DATE"
echo " $ICON_HOST $MY_ID — $LOCAL_SERVER_NAME"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
# ── ZFS Pool Health ───────────────────────────────────────────────────────────────────────────
echo ""
echo "━━━ $ICON_ZFS ZFS Pool Health ━━━"
if ! command -v zpool >/dev/null 2>&1; then
warn "ZFS not available on this system — skipping pool checks"
else
# Pool status — filtered to key lines, ignoring specified pools
CURRENT_POOL=""
while IFS= read -r line; do
if [[ "$line" =~ ^[[:space:]]*pool:[[:space:]]*(.+) ]]; then
CURRENT_POOL="${BASH_REMATCH[1]// /}"
fi
[[ -n "${IGNORE_POOL_MAP[$CURRENT_POOL]:-}" ]] && continue
echo " $line"
done < <(zpool status 2>/dev/null | grep -E "pool:|state:|status:|errors:|scan:")
echo ""
# Pool list — filter out ignored pools
zpool list 2>/dev/null | while IFS= read -r line; do
if [[ "$line" == NAME* ]]; then
echo " $line"
continue
fi
pool_name=$(echo "$line" | awk '{print $1}')
[[ -n "${IGNORE_POOL_MAP[$pool_name]:-}" ]] && continue
echo " $line"
done
# Check for unhealthy non-ignored pools
UNHEALTHY=$(zpool list -H -o name,health 2>/dev/null | \
while IFS=$'\t' read -r name health; do
[[ -n "${IGNORE_POOL_MAP[$name]:-}" ]] && continue
[[ "$health" != "ONLINE" ]] && echo "$name: $health"
done)
if [[ -n "$UNHEALTHY" ]]; then
error "One or more ZFS pools are NOT ONLINE: $UNHEALTHY"
WARNINGS+=("ZFS pool unhealthy: $UNHEALTHY")
else
echo "All monitored ZFS pools are ONLINE ✅"
fi
if [[ ${#ZFS_REPORT_IGNORE_POOLS[@]} -gt 0 ]]; then
log "Ignored pools: ${ZFS_REPORT_IGNORE_POOLS[*]}"
fi
fi
# ── ARC Statistics ────────────────────────────────────────────────────────────────────────────
echo ""
echo "━━━ $ICON_ZFS ARC Statistics ━━━"
if [[ ! -f /proc/spl/kstat/zfs/arcstats ]]; then
warn "ZFS arcstats not available — skipping ARC section"
else
ARC_MAX=$(cat /sys/module/zfs/parameters/zfs_arc_max 2>/dev/null || \
awk '/^c_max / {print $3}' /proc/spl/kstat/zfs/arcstats)
ARC_SIZE=$(awk '/^size / {print $3}' /proc/spl/kstat/zfs/arcstats)
ARC_META_USED=$(awk '/^arc_meta_used / {print $3}' /proc/spl/kstat/zfs/arcstats)
ARC_MAX_GB=$(awk "BEGIN {printf \"%.1f\", $ARC_MAX / 1073741824}")
ARC_CUR_GB=$(awk "BEGIN {printf \"%.1f\", $ARC_SIZE / 1073741824}")
ARC_META_GB=$(awk "BEGIN {printf \"%.1f\", $ARC_META_USED / 1073741824}")
ARC_PCT=$(awk "BEGIN {printf \"%.1f\", $ARC_SIZE * 100 / $ARC_MAX}")
ARC_PCT_INT=$(printf "%.0f" "$ARC_PCT")
echo " $ICON_ZFS ARC Max: ${ARC_MAX_GB}GB"
echo " $ICON_ZFS ARC Current: ${ARC_CUR_GB}GB"
echo " $ICON_ZFS ARC Meta Used: ${ARC_META_GB}GB"
echo " $ICON_ZFS ARC Utilization: ${ARC_PCT}%"
if [[ "$ARC_PCT_INT" -ge "$ZFS_REPORT_ARC_WARN_PCT" ]]; then
warn "ARC utilization ${ARC_PCT}% — above ${ZFS_REPORT_ARC_WARN_PCT}% threshold"
WARNINGS+=("ARC high: ${ARC_PCT}%")
else
log "ARC utilization ${ARC_PCT}% — within threshold ✅"
fi
echo ""
META_MRU_GHOST=$(awk '/^mru_ghost_metadata / {print $3}' \
/proc/spl/kstat/zfs/arcstats 2>/dev/null || echo 0)
META_MFU_GHOST=$(awk '/^mfu_ghost_metadata / {print $3}' \
/proc/spl/kstat/zfs/arcstats 2>/dev/null || echo 0)
META_MISSES=$(awk '/^demand_metadata_misses / {print $3}' \
/proc/spl/kstat/zfs/arcstats 2>/dev/null || echo 0)
MRU_GB=$(awk "BEGIN {printf \"%.2f\", $META_MRU_GHOST / 1073741824}")
MFU_GB=$(awk "BEGIN {printf \"%.2f\", $META_MFU_GHOST / 1073741824}")
echo " $ICON_ZFS MRU Ghost: ${MRU_GB}GB"
echo " $ICON_ZFS MFU Ghost: ${MFU_GB}GB"
echo " $ICON_ZFS Metadata Misses: ${META_MISSES}"
fi
# ── Memory Status ─────────────────────────────────────────────────────────────────────────────
echo ""
echo "━━━ $ICON_MEM Memory Status ━━━"
FREE_HUMAN=$(free -h | awk '/Mem:/ {print $4}')
AVAIL_HUMAN=$(free -h | awk '/Mem:/ {print $7}')
TOTAL_HUMAN=$(free -h | awk '/Mem:/ {print $2}')
FREE_GB=$(free -g | awk '/Mem:/ {print $4}')
AVAIL_GB=$(free -g | awk '/Mem:/ {print $7}')
echo " $ICON_MEM Total RAM: $TOTAL_HUMAN"
echo " $ICON_MEM Free RAM: $FREE_HUMAN"
echo " $ICON_MEM Available RAM: $AVAIL_HUMAN"
if [[ "$FREE_GB" -lt "$ZFS_REPORT_FREE_WARN_GB" ]]; then
warn "Free RAM ${FREE_HUMAN} — below ${ZFS_REPORT_FREE_WARN_GB}GB threshold"
WARNINGS+=("Low free RAM: ${FREE_HUMAN}")
else
log "Free RAM ${FREE_HUMAN} — within threshold ✅"
fi
if [[ "$AVAIL_GB" -lt "$ZFS_REPORT_AVAIL_WARN_GB" ]]; then
warn "Available RAM ${AVAIL_HUMAN} — below ${ZFS_REPORT_AVAIL_WARN_GB}GB threshold"
WARNINGS+=("Low available RAM: ${AVAIL_HUMAN}")
else
log "Available RAM ${AVAIL_HUMAN} — within threshold ✅"
fi
# ── Docker Memory ─────────────────────────────────────────────────────────────────────────────
echo ""
echo "━━━ $ICON_CONTAINERS Top $ZFS_REPORT_DOCKER_TOP Docker Memory Users ━━━"
if ! command -v docker >/dev/null 2>&1; then
warn "Docker not available — skipping container memory section"
else
timeout "$DOCKER_TIMEOUT" docker stats --no-stream \
--format "table {{.Name}}\t{{.MemUsage}}\t{{.MemPerc}}" \
2>/dev/null | head -n $(( ZFS_REPORT_DOCKER_TOP + 1 )) | \
while IFS= read -r line; do
echo " $line"
done
fi
# ── Kernel Pressure ───────────────────────────────────────────────────────────────────────────
echo ""
echo "━━━ $ICON_GEAR Kernel Pressure ━━━"
if ! command -v vmstat >/dev/null 2>&1; then
warn "vmstat not available — skipping kernel pressure section"
else
vmstat 1 3 2>/dev/null | while IFS= read -r line; do
echo " $line"
done
fi
END=$(date +%s)
# ── Summary ───────────────────────────────────────────────────────────────────────────────────
echo ""
echo "━━━━━ $ICON_SUMMARY ZFS REPORT SUMMARY ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_TIME Duration: $(format_duration $(( END - START )))"
echo "$ICON_ZFS Log: $ZFS_REPORT_LOG"
[[ ${#ZFS_REPORT_IGNORE_POOLS[@]} -gt 0 ]] && \
log "Ignored: ${ZFS_REPORT_IGNORE_POOLS[*]}"
echo ""
if [[ ${#WARNINGS[@]} -eq 0 ]]; then
echo "$ICON_DONE All checks within thresholds ✅"
else
echo "$ICON_WARN Warnings: ${#WARNINGS[@]}"
for w in "${WARNINGS[@]}"; do
echo " $ICON_WARN $w"
done
notify "ZFS weekly report on $(hostname) — ${#WARNINGS[@]} warning(s): ${WARNINGS[*]}" \
"ZFS Report" "warning"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
[[ ${#WARNINGS[@]} -gt 0 ]] && exit 1
exit 0
@@ -1,338 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ====================== Partnership — Unraid Container Adapter ================================
# ==============================================================================================
#
# Sourced by partnership_onboard.sh and partnership_offboard.sh via:
# source "$SCRIPTS_ROOT/Plugin/$PLATFORM/Partnership/containers.sh"
#
# Provides container deploy/cleanup functions specific to the Unraid platform:
# - Docker container deployment from Unraid CA XML templates
#
# Functions use variables from the calling script's scope (sourced, not exec'd):
# MIRROR, MIRROR_IP, MIRROR_SSH_KEY, SSH_TIMEOUT, DRY_RUN, SCRIPTS_ROOT
#
# ==============================================================================================
TEMPLATES_DIR="/boot/config/plugins/dockerMan/templates-user"
_STACK_DEPLOYED=0
_STACK_FAILED=0
# ==============================================================================================
# ── Wait for a container on the remote to be healthy/running ─────────────────────────────────
#
# Polls docker inspect on the remote. Prefers the health status if a healthcheck is defined;
# falls back to the running state. Non-fatal after timeout — some containers take time to
# fully initialize but the deploy itself succeeded.
# ==============================================================================================
wait_for_container_healthy() {
local name="$1" remote_ip="$2" ssh_key="$3"
local max_wait=60 interval=5 elapsed=0
[[ "$DRY_RUN" == true ]] && return 0
log " Waiting for $name to be ready..."
while (( elapsed < max_wait )); do
local status
status=$(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$remote_ip" \
"h=\$(docker inspect --format '{{.State.Health.Status}}' '$name' 2>/dev/null)
r=\$(docker inspect --format '{{.State.Running}}' '$name' 2>/dev/null)
echo \${h:-\$r}" 2>/dev/null)
case "$status" in
healthy|true)
log " $name ready ✅"
return 0
;;
*)
sleep "$interval"
(( elapsed += interval ))
;;
esac
done
warn " $name not confirmed healthy after ${max_wait}s — continuing (may affect dependents)"
return 0
}
# ==============================================================================================
# ── Deploy a container from a local Unraid CA XML template to a remote host ──────────────────
#
# Parses Port / Path / Variable Config entries from the Unraid XML, SCPs the template and a
# self-contained deploy script to the remote, executes it, then cleans up both sides.
# Credentials are never passed as SSH command-line args — they stay in the SCPed script.
# ==============================================================================================
deploy_container_from_xml() {
local xml_file="$1" remote_ip="$2" ssh_key="$3"
local xml_name
xml_name=$(basename "$xml_file")
local name repo network extra privileged
name=$( awk 'match($0,/<Name>([^<]+)<\/Name>/, a){print a[1];exit}' "$xml_file")
repo=$( awk 'match($0,/<Repository>([^<]+)<\/Repository>/,a){print a[1];exit}' "$xml_file")
network=$( awk 'match($0,/<Network>([^<]+)<\/Network>/, a){print a[1];exit}' "$xml_file")
extra=$( awk 'match($0,/<ExtraParams>([^<]*)<\/ExtraParams>/,a){print a[1];exit}' "$xml_file")
privileged=$( awk 'match($0,/<Privileged>([^<]+)<\/Privileged>/,a){print a[1];exit}' "$xml_file")
if [[ -z "$name" || -z "$repo" ]]; then
warn " Cannot parse Name/Repository from $xml_name — skipping"
return 1
fi
log "Deploying $name..."
if [[ "$DRY_RUN" == false ]]; then
timeout "$SSH_TIMEOUT" scp -i "$ssh_key" -o ConnectTimeout="$SSH_TIMEOUT" \
"$xml_file" "root@${remote_ip}:${TEMPLATES_DIR}/${xml_name}" 2>/dev/null || {
warn " SCP failed for $xml_name — skipping $name"
return 1
}
else
warn " DRY RUN — would SCP $xml_name → $MIRROR:${TEMPLATES_DIR}/"
fi
local tmp_script
tmp_script=$(mktemp /tmp/deploy_XXXXXX.sh)
chmod 600 "$tmp_script"
{
echo "#!/bin/bash"
echo "set -e"
echo ""
printf "docker pull %q 2>/dev/null || true\n" "$repo"
printf "docker stop %q 2>/dev/null || true\n" "$name"
printf "docker rm %q 2>/dev/null || true\n" "$name"
echo ""
printf "docker create --name %q --restart=unless-stopped" "$name"
[[ -n "$network" ]] && printf " --network=%q" "$network"
[[ "$privileged" == "true" ]] && printf " --privileged"
[[ -n "$extra" ]] && printf " %s" "$extra"
# Port mappings → -p host:container/proto
awk '/Type="Port"/ {
match($0, /Target="([^"]+)"/, t)
match($0, /Mode="([^"]+)"/, m)
match($0, />([^<]+)<\/Config>/, v)
if (t[1] != "" && v[1] != "") {
proto = (m[1] == "udp") ? "udp" : "tcp"
printf " -p %s:%s/%s", v[1], t[1], proto
}
}' "$xml_file"
# Volume mappings → -v 'host:container:mode'
awk 'BEGIN{q=sprintf("%c",39)} /Type="Path"/ {
match($0, /Target="([^"]+)"/, t)
match($0, /Mode="([^"]+)"/, m)
match($0, />([^<]+)<\/Config>/, v)
if (t[1] != "" && v[1] != "") {
mode = (m[1] == "ro") ? "ro" : "rw"
printf " -v %s%s:%s:%s%s", q, v[1], t[1], mode, q
}
}' "$xml_file"
# Environment variables → -e 'KEY=VALUE' (single-quoted to protect $ and special chars)
awk 'BEGIN{q=sprintf("%c",39)} /Type="Variable"/ {
match($0, /Target="([^"]+)"/, t)
match($0, />([^<]+)<\/Config>/, v)
if (t[1] != "" && v[1] != "") {
printf " -e %s%s=%s%s", q, t[1], v[1], q
}
}' "$xml_file"
printf " %q\n" "$repo"
echo ""
printf "docker start %q && echo 'deployed:%s'\n" "$name" "$name"
} > "$tmp_script"
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would deploy $name on $MIRROR"
rm -f "$tmp_script"
return 0
fi
local remote_script="/tmp/deploy_${name//[^a-zA-Z0-9_]/_}.sh"
if timeout "$SSH_TIMEOUT" scp -i "$ssh_key" -o ConnectTimeout="$SSH_TIMEOUT" \
"$tmp_script" "root@${remote_ip}:${remote_script}" 2>/dev/null && \
timeout 120 ssh -i "$ssh_key" -o ConnectTimeout="$SSH_TIMEOUT" root@"$remote_ip" \
"bash '$remote_script' 2>&1; rc=\$?; rm -f '$remote_script'; exit \$rc" 2>/dev/null | \
grep -q "deployed:${name}"; then
log " $name deployed ✅"
rm -f "$tmp_script"
return 0
else
warn " $name deployment failed — check $MIRROR manually"
rm -f "$tmp_script"
return 1
fi
}
# ==============================================================================================
# ── Deploy a stack of Unraid CA XMLs to the mirror ───────────────────────────────────────────
#
# Sets globals _STACK_DEPLOYED and _STACK_FAILED rather than printing to stdout.
# Health-checks database deps (Mariadb/Redis/Postgres) between batches so dependents
# (e.g. Authelia) start cleanly.
# ==============================================================================================
deploy_xml_stack() {
local -n xml_array_ref="$1"
_STACK_DEPLOYED=0
_STACK_FAILED=0
for xml_name in "${xml_array_ref[@]}"; do
local xml_file="${TEMPLATES_DIR}/${xml_name}"
if [[ ! -f "$xml_file" ]]; then
warn "$xml_name not found in $TEMPLATES_DIR — skipping"
(( _STACK_FAILED++ ))
continue
fi
local cname
cname=$(awk 'match($0,/<Name>([^<]+)<\/Name>/,a){print a[1];exit}' "$xml_file")
if deploy_container_from_xml "$xml_file" "$MIRROR_IP" "$MIRROR_SSH_KEY"; then
(( _STACK_DEPLOYED++ ))
if [[ -n "$cname" ]] && echo "$cname" | grep -qiE 'mariadb|redis|postgres|mysql'; then
wait_for_container_healthy "$cname" "$MIRROR_IP" "$MIRROR_SSH_KEY"
fi
else
(( _STACK_FAILED++ ))
fi
done
}
# ==============================================================================================
# ── Remove owner-deployed containers from a remote host ──────────────────────────────────────
#
# Uses PARTNERSHIP_AUTH_STACK + PARTNERSHIP_ARR_STACK (owner's conf) to derive container
# names from local XML templates. SSHes to remote to stop, remove, and delete appdata.
# Appdata paths collected via docker inspect before removal. Safety gate: only
# /mnt/*/appdata* paths are deleted.
# ==============================================================================================
cleanup_deployed_stack_on_remote() {
local remote_ip="$1" ssh_key="$2"
local -a xml_names=()
[[ ${#PARTNERSHIP_AUTH_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_AUTH_STACK[@]}")
[[ ${#PARTNERSHIP_ARR_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_ARR_STACK[@]}")
if [[ ${#xml_names[@]} -eq 0 ]]; then
log "No auth/arr stack arrays configured — skipping deployed stack cleanup"
return 0
fi
log "Removing owner-deployed containers (auth/arr stacks) from $MIRROR..."
for xml_name in "${xml_names[@]}"; do
[[ -z "$xml_name" ]] && continue
local xml_file="${TEMPLATES_DIR}/${xml_name}"
if [[ ! -f "$xml_file" ]]; then
warn " $xml_name not found in local $TEMPLATES_DIR — skipping"
continue
fi
local cname
cname=$(awk 'match($0,/<Name>([^<]+)<\/Name>/,a){print a[1];exit}' "$xml_file")
[[ -z "$cname" ]] && continue
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would stop + rm $cname on $MIRROR"
warn " DRY RUN — would delete appdata for $cname on $MIRROR"
continue
fi
local appdata_paths
appdata_paths=$(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$remote_ip" \
"docker inspect --format '{{range .HostConfig.Binds}}{{println .}}{{end}}' '$cname' 2>/dev/null \
| awk -F: '{print \$1}' | grep '^/mnt/.*/appdata'" 2>/dev/null)
timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" root@"$remote_ip" \
"docker stop '$cname' >/dev/null 2>&1
docker rm '$cname' >/dev/null 2>&1 && echo removed" 2>/dev/null | \
grep -q removed && \
log " $cname removed from $MIRROR ✅" || \
log " $cname not found on $MIRROR — skipping"
while IFS= read -r path; do
[[ -z "$path" ]] && continue
timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" root@"$remote_ip" \
"rm -rf '$path' && echo removed" 2>/dev/null | grep -q removed && \
log " Appdata removed on $MIRROR: $path ✅" || \
warn " Failed to remove appdata on $MIRROR: $path"
done <<< "$appdata_paths"
done
}
# ==============================================================================================
# ── Remove owner-deployed containers locally (mirror-initiated offboard) ─────────────────────
#
# SSHes to owner to read PARTNERSHIP_AUTH_STACK + PARTNERSHIP_ARR_STACK, then uses the
# local templates-user/ copies (SCPed there during onboard) to get container names and
# appdata paths. Appdata collected before removal. Skips gracefully if owner unreachable.
# ==============================================================================================
cleanup_deployed_stack_locally() {
local owner_ip="$1" ssh_key="$2"
local -a xml_names=()
if [[ -n "$owner_ip" ]]; then
local -a auth_arr arr_arr
mapfile -t auth_arr < <(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$owner_ip" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${PARTNERSHIP_AUTH_STACK[@]:-}\"" 2>/dev/null | grep -v '^$')
mapfile -t arr_arr < <(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$owner_ip" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${PARTNERSHIP_ARR_STACK[@]:-}\"" 2>/dev/null | grep -v '^$')
xml_names=("${auth_arr[@]}" "${arr_arr[@]}")
fi
if [[ ${#xml_names[@]} -eq 0 ]]; then
log "Could not read deployed stack from owner — skipping auth/arr cleanup"
return 0
fi
log "Removing owner-deployed containers (auth/arr stacks) locally..."
for xml_name in "${xml_names[@]}"; do
[[ -z "$xml_name" ]] && continue
local xml_file="${TEMPLATES_DIR}/${xml_name}"
if [[ ! -f "$xml_file" ]]; then
warn " $xml_name not found locally — skipping"
continue
fi
local cname
cname=$(awk 'match($0,/<Name>([^<]+)<\/Name>/,a){print a[1];exit}' "$xml_file")
[[ -z "$cname" ]] && continue
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would stop + rm $cname"
warn " DRY RUN — would delete appdata for $cname"
continue
fi
local appdata_paths=""
if timeout "${DOCKER_TIMEOUT:-30}" docker inspect "$cname" >/dev/null 2>&1; then
appdata_paths=$(docker inspect \
--format '{{range .HostConfig.Binds}}{{println .}}{{end}}' \
"$cname" 2>/dev/null | awk -F: '{print $1}' | grep '^/mnt/.*/appdata')
timeout "${DOCKER_TIMEOUT:-30}" docker stop "$cname" >/dev/null 2>&1 || true
_PM_TRAP_STOPPED+=("$cname")
timeout "${DOCKER_TIMEOUT:-30}" docker rm "$cname" >/dev/null 2>&1 && \
log " $cname removed ✅" || warn " $cname rm failed"
else
log " $cname not found locally — skipping"
fi
while IFS= read -r path; do
[[ -z "$path" ]] && continue
rm -rf "$path" && log " Appdata removed: $path ✅" || warn " Failed to remove: $path"
done <<< "$appdata_paths"
done
}
@@ -1,345 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ====================== Partnership — Unraid Container Adapter ================================
# ==============================================================================================
#
# Sourced by partnership_onboard.sh and partnership_offboard.sh via:
# source "$SCRIPTS_ROOT/Plugin/$PLATFORM/Partnership/containers.sh"
#
# Provides container deploy/cleanup functions specific to the Unraid platform:
# - Docker container deployment from Unraid CA XML templates
#
# Functions use variables from the calling script's scope (sourced, not exec'd):
# MIRROR, MIRROR_IP, MIRROR_SSH_KEY, SSH_TIMEOUT, DRY_RUN, SCRIPTS_ROOT
#
# ==============================================================================================
TEMPLATES_DIR="/boot/config/plugins/dockerMan/templates-user"
_STACK_DEPLOYED=0
_STACK_FAILED=0
# ==============================================================================================
# ── Wait for a container on the remote to be healthy/running ─────────────────────────────────
#
# Polls docker inspect on the remote. Prefers the health status if a healthcheck is defined;
# falls back to the running state. Non-fatal after timeout — some containers take time to
# fully initialize but the deploy itself succeeded.
# ==============================================================================================
wait_for_container_healthy() {
local name="$1" remote_ip="$2" ssh_key="$3"
local max_wait=60 interval=5 elapsed=0
[[ "$DRY_RUN" == true ]] && return 0
log " Waiting for $name to be ready..."
while (( elapsed < max_wait )); do
local status
status=$(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$remote_ip" \
"h=\$(docker inspect --format '{{.State.Health.Status}}' '$name' 2>/dev/null)
r=\$(docker inspect --format '{{.State.Running}}' '$name' 2>/dev/null)
echo \${h:-\$r}" 2>/dev/null)
case "$status" in
healthy|true)
log " $name ready ✅"
return 0
;;
*)
sleep "$interval"
(( elapsed += interval ))
;;
esac
done
warn " $name not confirmed healthy after ${max_wait}s — continuing (may affect dependents)"
return 0
}
# ==============================================================================================
# ── Deploy a container from a local Unraid CA XML template to a remote host ──────────────────
#
# Parses Port / Path / Variable Config entries from the Unraid XML, SCPs the template and a
# self-contained deploy script to the remote, executes it, then cleans up both sides.
# Credentials are never passed as SSH command-line args — they stay in the SCPed script.
# ==============================================================================================
deploy_container_from_xml() {
local xml_file="$1" remote_ip="$2" ssh_key="$3"
local xml_name
xml_name=$(basename "$xml_file")
local name repo network extra privileged
name=$( awk 'match($0,/<Name>([^<]+)<\/Name>/, a){print a[1];exit}' "$xml_file")
repo=$( awk 'match($0,/<Repository>([^<]+)<\/Repository>/,a){print a[1];exit}' "$xml_file")
network=$( awk 'match($0,/<Network>([^<]+)<\/Network>/, a){print a[1];exit}' "$xml_file")
extra=$( awk 'match($0,/<ExtraParams>([^<]*)<\/ExtraParams>/,a){print a[1];exit}' "$xml_file")
privileged=$( awk 'match($0,/<Privileged>([^<]+)<\/Privileged>/,a){print a[1];exit}' "$xml_file")
if [[ -z "$name" || -z "$repo" ]]; then
warn " Cannot parse Name/Repository from $xml_name — skipping"
return 1
fi
log "Deploying $name..."
if [[ "$DRY_RUN" == false ]]; then
timeout "$SSH_TIMEOUT" scp -i "$ssh_key" -o ConnectTimeout="$SSH_TIMEOUT" \
"$xml_file" "root@${remote_ip}:${TEMPLATES_DIR}/${xml_name}" 2>/dev/null || {
warn " SCP failed for $xml_name — skipping $name"
return 1
}
else
warn " DRY RUN — would SCP $xml_name → $MIRROR:${TEMPLATES_DIR}/"
fi
local tmp_script
tmp_script=$(mktemp /tmp/deploy_XXXXXX.sh)
chmod 600 "$tmp_script"
{
echo "#!/bin/bash"
echo "set -e"
echo ""
printf "docker pull %q 2>/dev/null || true\n" "$repo"
printf "docker stop %q 2>/dev/null || true\n" "$name"
printf "docker rm %q 2>/dev/null || true\n" "$name"
echo ""
printf "docker create --name %q --restart=unless-stopped" "$name"
[[ -n "$network" ]] && printf " --network=%q" "$network"
[[ "$privileged" == "true" ]] && printf " --privileged"
[[ -n "$extra" ]] && printf " %s" "$extra"
# Port mappings → -p host:container/proto
awk '/Type="Port"/ {
match($0, /Target="([^"]+)"/, t)
match($0, /Mode="([^"]+)"/, m)
match($0, />([^<]+)<\/Config>/, v)
if (t[1] != "" && v[1] != "") {
proto = (m[1] == "udp") ? "udp" : "tcp"
printf " -p %s:%s/%s", v[1], t[1], proto
}
}' "$xml_file"
# Volume mappings → -v 'host:container:mode'
awk 'BEGIN{q=sprintf("%c",39)} /Type="Path"/ {
match($0, /Target="([^"]+)"/, t)
match($0, /Mode="([^"]+)"/, m)
match($0, />([^<]+)<\/Config>/, v)
if (t[1] != "" && v[1] != "") {
mode = (m[1] == "ro") ? "ro" : "rw"
printf " -v %s%s:%s:%s%s", q, v[1], t[1], mode, q
}
}' "$xml_file"
# Environment variables → -e 'KEY=VALUE' (single-quoted to protect $ and special chars)
awk 'BEGIN{q=sprintf("%c",39)} /Type="Variable"/ {
match($0, /Target="([^"]+)"/, t)
match($0, />([^<]+)<\/Config>/, v)
if (t[1] != "" && v[1] != "") {
printf " -e %s%s=%s%s", q, t[1], v[1], q
}
}' "$xml_file"
printf " %q\n" "$repo"
echo ""
printf "docker start %q && echo 'deployed:%s'\n" "$name" "$name"
} > "$tmp_script"
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would deploy $name on $MIRROR"
rm -f "$tmp_script"
return 0
fi
local remote_script="/tmp/deploy_${name//[^a-zA-Z0-9_]/_}.sh"
if timeout "$SSH_TIMEOUT" scp -i "$ssh_key" -o ConnectTimeout="$SSH_TIMEOUT" \
"$tmp_script" "root@${remote_ip}:${remote_script}" 2>/dev/null && \
timeout 120 ssh -i "$ssh_key" -o ConnectTimeout="$SSH_TIMEOUT" root@"$remote_ip" \
"bash '$remote_script' 2>&1; rc=\$?; rm -f '$remote_script'; exit \$rc" 2>/dev/null | \
grep -q "deployed:${name}"; then
log " $name deployed ✅"
rm -f "$tmp_script"
return 0
else
warn " $name deployment failed — check $MIRROR manually"
rm -f "$tmp_script"
return 1
fi
}
# ==============================================================================================
# ── Deploy a stack of Unraid CA XMLs to the mirror ───────────────────────────────────────────
#
# Sets globals _STACK_DEPLOYED and _STACK_FAILED rather than printing to stdout.
# Health-checks database deps (Mariadb/Redis/Postgres) between batches so dependents
# (e.g. Authelia) start cleanly.
# ==============================================================================================
deploy_xml_stack() {
local -n xml_array_ref="$1"
_STACK_DEPLOYED=0
_STACK_FAILED=0
for xml_name in "${xml_array_ref[@]}"; do
local xml_file="${TEMPLATES_DIR}/${xml_name}"
if [[ ! -f "$xml_file" ]]; then
warn "$xml_name not found in $TEMPLATES_DIR — skipping"
(( _STACK_FAILED++ ))
continue
fi
local cname
cname=$(awk 'match($0,/<Name>([^<]+)<\/Name>/,a){print a[1];exit}' "$xml_file")
if deploy_container_from_xml "$xml_file" "$MIRROR_IP" "$MIRROR_SSH_KEY"; then
(( _STACK_DEPLOYED++ ))
if [[ -n "$cname" ]] && echo "$cname" | grep -qiE 'mariadb|redis|postgres|mysql'; then
wait_for_container_healthy "$cname" "$MIRROR_IP" "$MIRROR_SSH_KEY"
fi
else
(( _STACK_FAILED++ ))
fi
done
}
# ==============================================================================================
# ── Remove owner-deployed containers from a remote host ──────────────────────────────────────
#
# Uses PARTNERSHIP_AUTH_STACK + PARTNERSHIP_ARR_STACK (owner's conf) to derive container
# names from local XML templates. SSHes to remote to stop, remove, and delete appdata.
# Appdata paths collected via docker inspect before removal. Safety gate: only
# /mnt/*/appdata* paths are deleted.
# ==============================================================================================
cleanup_deployed_stack_on_remote() {
local remote_ip="$1" ssh_key="$2"
local -a xml_names=()
[[ ${#PARTNERSHIP_AUTH_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_AUTH_STACK[@]}")
[[ ${#PARTNERSHIP_ARR_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_ARR_STACK[@]}")
[[ ${#PARTNERSHIP_SERVICES_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_SERVICES_STACK[@]}")
if [[ ${#xml_names[@]} -eq 0 ]]; then
log "No auth/arr/services stack arrays configured — skipping deployed stack cleanup"
return 0
fi
log "Removing owner-deployed containers (auth/arr/services stacks) from $MIRROR..."
for xml_name in "${xml_names[@]}"; do
[[ -z "$xml_name" ]] && continue
local xml_file="${TEMPLATES_DIR}/${xml_name}"
if [[ ! -f "$xml_file" ]]; then
warn " $xml_name not found in local $TEMPLATES_DIR — skipping"
continue
fi
local cname
cname=$(awk 'match($0,/<Name>([^<]+)<\/Name>/,a){print a[1];exit}' "$xml_file")
[[ -z "$cname" ]] && continue
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would stop + rm $cname on $MIRROR"
warn " DRY RUN — would delete appdata for $cname on $MIRROR"
continue
fi
local appdata_paths
appdata_paths=$(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$remote_ip" \
"docker inspect --format '{{range .HostConfig.Binds}}{{println .}}{{end}}' '$cname' 2>/dev/null \
| awk -F: '{print \$1}' | grep '^/mnt/.*/appdata'" 2>/dev/null)
timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" root@"$remote_ip" \
"docker stop '$cname' >/dev/null 2>&1
docker rm '$cname' >/dev/null 2>&1 && echo removed" 2>/dev/null | \
grep -q removed && \
log " $cname removed from $MIRROR ✅" || \
log " $cname not found on $MIRROR — skipping"
while IFS= read -r path; do
[[ -z "$path" ]] && continue
timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" root@"$remote_ip" \
"rm -rf '$path' && echo removed" 2>/dev/null | grep -q removed && \
log " Appdata removed on $MIRROR: $path ✅" || \
warn " Failed to remove appdata on $MIRROR: $path"
done <<< "$appdata_paths"
done
}
# ==============================================================================================
# ── Remove owner-deployed containers locally (mirror-initiated offboard) ─────────────────────
#
# SSHes to owner to read PARTNERSHIP_AUTH_STACK + PARTNERSHIP_ARR_STACK, then uses the
# local templates-user/ copies (SCPed there during onboard) to get container names and
# appdata paths. Appdata collected before removal. Skips gracefully if owner unreachable.
# ==============================================================================================
cleanup_deployed_stack_locally() {
local owner_ip="$1" ssh_key="$2"
local -a xml_names=()
if [[ -n "$owner_ip" ]]; then
local -a auth_arr arr_arr
mapfile -t auth_arr < <(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$owner_ip" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${PARTNERSHIP_AUTH_STACK[@]:-}\"" 2>/dev/null | grep -v '^$')
mapfile -t arr_arr < <(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$owner_ip" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${PARTNERSHIP_ARR_STACK[@]:-}\"" 2>/dev/null | grep -v '^$')
local -a svc_arr
mapfile -t svc_arr < <(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$owner_ip" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${PARTNERSHIP_SERVICES_STACK[@]:-}\"" 2>/dev/null | grep -v '^$')
xml_names=("${auth_arr[@]}" "${arr_arr[@]}" "${svc_arr[@]}")
fi
if [[ ${#xml_names[@]} -eq 0 ]]; then
log "Could not read deployed stack from owner — skipping auth/arr/services cleanup"
return 0
fi
log "Removing owner-deployed containers (auth/arr/services stacks) locally..."
for xml_name in "${xml_names[@]}"; do
[[ -z "$xml_name" ]] && continue
local xml_file="${TEMPLATES_DIR}/${xml_name}"
if [[ ! -f "$xml_file" ]]; then
warn " $xml_name not found locally — skipping"
continue
fi
local cname
cname=$(awk 'match($0,/<Name>([^<]+)<\/Name>/,a){print a[1];exit}' "$xml_file")
[[ -z "$cname" ]] && continue
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would stop + rm $cname"
warn " DRY RUN — would delete appdata for $cname"
continue
fi
local appdata_paths=""
if timeout "${DOCKER_TIMEOUT:-30}" docker inspect "$cname" >/dev/null 2>&1; then
appdata_paths=$(docker inspect \
--format '{{range .HostConfig.Binds}}{{println .}}{{end}}' \
"$cname" 2>/dev/null | awk -F: '{print $1}' | grep '^/mnt/.*/appdata')
timeout "${DOCKER_TIMEOUT:-30}" docker stop "$cname" >/dev/null 2>&1 || true
_PM_TRAP_STOPPED+=("$cname")
timeout "${DOCKER_TIMEOUT:-30}" docker rm "$cname" >/dev/null 2>&1 && \
log " $cname removed ✅" || warn " $cname rm failed"
else
log " $cname not found locally — skipping"
fi
while IFS= read -r path; do
[[ -z "$path" ]] && continue
rm -rf "$path" && log " Appdata removed: $path ✅" || warn " Failed to remove: $path"
done <<< "$appdata_paths"
done
}
@@ -1,416 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ====================== Partnership — Unraid Container Adapter ================================
# ==============================================================================================
#
# Sourced by partnership_onboard.sh and partnership_offboard.sh via:
# source "$SCRIPTS_ROOT/Plugin/$PLATFORM/Partnership/containers.sh"
#
# Provides container deploy/cleanup functions specific to the Unraid platform:
# - Docker container deployment from Unraid CA XML templates
#
# Functions use variables from the calling script's scope (sourced, not exec'd):
# MIRROR, MIRROR_IP, MIRROR_SSH_KEY, SSH_TIMEOUT, DRY_RUN, SCRIPTS_ROOT
#
# ==============================================================================================
TEMPLATES_DIR="/boot/config/plugins/dockerMan/templates-user"
_STACK_DEPLOYED=0
_STACK_FAILED=0
# ==============================================================================================
# ── Wait for a container on the remote to be healthy/running ─────────────────────────────────
#
# Polls docker inspect on the remote. Prefers the health status if a healthcheck is defined;
# falls back to the running state. Non-fatal after timeout — some containers take time to
# fully initialize but the deploy itself succeeded.
# ==============================================================================================
wait_for_container_healthy() {
local name="$1" remote_ip="$2" ssh_key="$3"
local max_wait=60 interval=5 elapsed=0
[[ "$DRY_RUN" == true ]] && return 0
log " Waiting for $name to be ready..."
while (( elapsed < max_wait )); do
local status
status=$(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$remote_ip" \
"h=\$(docker inspect --format '{{.State.Health.Status}}' '$name' 2>/dev/null)
r=\$(docker inspect --format '{{.State.Running}}' '$name' 2>/dev/null)
echo \${h:-\$r}" 2>/dev/null)
case "$status" in
healthy|true)
log " $name ready ✅"
return 0
;;
*)
sleep "$interval"
(( elapsed += interval ))
;;
esac
done
warn " $name not confirmed healthy after ${max_wait}s — continuing (may affect dependents)"
return 0
}
# ==============================================================================================
# ── Deploy a container from a local Unraid CA XML template to a remote host ──────────────────
#
# Parses Port / Path / Variable Config entries from the Unraid XML, SCPs the template and a
# self-contained deploy script to the remote, executes it, then cleans up both sides.
# Credentials are never passed as SSH command-line args — they stay in the SCPed script.
# ==============================================================================================
deploy_container_from_xml() {
local xml_file="$1" remote_ip="$2" ssh_key="$3"
local xml_name
xml_name=$(basename "$xml_file")
local name repo network extra privileged
name=$( awk 'match($0,/<Name>([^<]+)<\/Name>/, a){print a[1];exit}' "$xml_file")
repo=$( awk 'match($0,/<Repository>([^<]+)<\/Repository>/,a){print a[1];exit}' "$xml_file")
network=$( awk 'match($0,/<Network>([^<]+)<\/Network>/, a){print a[1];exit}' "$xml_file")
extra=$( awk 'match($0,/<ExtraParams>([^<]*)<\/ExtraParams>/,a){print a[1];exit}' "$xml_file")
privileged=$( awk 'match($0,/<Privileged>([^<]+)<\/Privileged>/,a){print a[1];exit}' "$xml_file")
if [[ -z "$name" || -z "$repo" ]]; then
warn " Cannot parse Name/Repository from $xml_name — skipping"
return 1
fi
log "Deploying $name..."
if [[ "$DRY_RUN" == false ]]; then
timeout "$SSH_TIMEOUT" scp -i "$ssh_key" -o ConnectTimeout="$SSH_TIMEOUT" \
"$xml_file" "root@${remote_ip}:${TEMPLATES_DIR}/${xml_name}" 2>/dev/null || {
warn " SCP failed for $xml_name — skipping $name"
return 1
}
else
warn " DRY RUN — would SCP $xml_name → $MIRROR:${TEMPLATES_DIR}/"
fi
local tmp_script
tmp_script=$(mktemp /tmp/deploy_XXXXXX.sh)
chmod 600 "$tmp_script"
{
echo "#!/bin/bash"
echo "set -e"
echo ""
printf "docker pull %q 2>/dev/null || true\n" "$repo"
printf "docker stop %q 2>/dev/null || true\n" "$name"
printf "docker rm %q 2>/dev/null || true\n" "$name"
echo ""
printf "docker create --name %q --restart=unless-stopped" "$name"
[[ -n "$network" ]] && printf " --network=%q" "$network"
[[ "$privileged" == "true" ]] && printf " --privileged"
[[ -n "$extra" ]] && printf " %s" "$extra"
# Port mappings → -p host:container/proto
awk '/Type="Port"/ {
match($0, /Target="([^"]+)"/, t)
match($0, /Mode="([^"]+)"/, m)
match($0, />([^<]+)<\/Config>/, v)
if (t[1] != "" && v[1] != "") {
proto = (m[1] == "udp") ? "udp" : "tcp"
printf " -p %s:%s/%s", v[1], t[1], proto
}
}' "$xml_file"
# Volume mappings → -v 'host:container:mode'
awk 'BEGIN{q=sprintf("%c",39)} /Type="Path"/ {
match($0, /Target="([^"]+)"/, t)
match($0, /Mode="([^"]+)"/, m)
match($0, />([^<]+)<\/Config>/, v)
if (t[1] != "" && v[1] != "") {
mode = (m[1] == "ro") ? "ro" : "rw"
printf " -v %s%s:%s:%s%s", q, v[1], t[1], mode, q
}
}' "$xml_file"
# Environment variables → -e 'KEY=VALUE' (single-quoted to protect $ and special chars)
awk 'BEGIN{q=sprintf("%c",39)} /Type="Variable"/ {
match($0, /Target="([^"]+)"/, t)
match($0, />([^<]+)<\/Config>/, v)
if (t[1] != "" && v[1] != "") {
printf " -e %s%s=%s%s", q, t[1], v[1], q
}
}' "$xml_file"
printf " %q\n" "$repo"
echo ""
printf "docker start %q && echo 'deployed:%s'\n" "$name" "$name"
} > "$tmp_script"
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would deploy $name on $MIRROR"
rm -f "$tmp_script"
return 0
fi
local remote_script="/tmp/deploy_${name//[^a-zA-Z0-9_]/_}.sh"
if timeout "$SSH_TIMEOUT" scp -i "$ssh_key" -o ConnectTimeout="$SSH_TIMEOUT" \
"$tmp_script" "root@${remote_ip}:${remote_script}" 2>/dev/null && \
timeout 120 ssh -i "$ssh_key" -o ConnectTimeout="$SSH_TIMEOUT" root@"$remote_ip" \
"bash '$remote_script' 2>&1; rc=\$?; rm -f '$remote_script'; exit \$rc" 2>/dev/null | \
grep -q "deployed:${name}"; then
log " $name deployed ✅"
rm -f "$tmp_script"
return 0
else
warn " $name deployment failed — check $MIRROR manually"
rm -f "$tmp_script"
return 1
fi
}
# ==============================================================================================
# ── Deploy a stack of Unraid CA XMLs to the mirror ───────────────────────────────────────────
#
# Sets globals _STACK_DEPLOYED and _STACK_FAILED rather than printing to stdout.
# Health-checks database deps (Mariadb/Redis/Postgres) between batches so dependents
# (e.g. Authelia) start cleanly.
# ==============================================================================================
deploy_xml_stack() {
local -n xml_array_ref="$1"
_STACK_DEPLOYED=0
_STACK_FAILED=0
for xml_name in "${xml_array_ref[@]}"; do
local xml_file="${TEMPLATES_DIR}/${xml_name}"
if [[ ! -f "$xml_file" ]]; then
warn "$xml_name not found in $TEMPLATES_DIR — skipping"
(( _STACK_FAILED++ ))
continue
fi
local cname
cname=$(awk 'match($0,/<Name>([^<]+)<\/Name>/,a){print a[1];exit}' "$xml_file")
if deploy_container_from_xml "$xml_file" "$MIRROR_IP" "$MIRROR_SSH_KEY"; then
(( _STACK_DEPLOYED++ ))
if [[ -n "$cname" ]] && echo "$cname" | grep -qiE 'mariadb|redis|postgres|mysql'; then
wait_for_container_healthy "$cname" "$MIRROR_IP" "$MIRROR_SSH_KEY"
fi
else
(( _STACK_FAILED++ ))
fi
done
}
# ==============================================================================================
# ── Remove owner-deployed containers from a remote host ──────────────────────────────────────
#
# Uses PARTNERSHIP_AUTH_STACK + PARTNERSHIP_ARR_STACK (owner's conf) to derive container
# names from local XML templates. SSHes to remote to stop, remove, and delete appdata.
# Appdata paths collected via docker inspect before removal. Safety gate: only
# /mnt/*/appdata* paths are deleted.
# ==============================================================================================
cleanup_deployed_stack_on_remote() {
local remote_ip="$1" ssh_key="$2"
local -a xml_names=()
[[ ${#PARTNERSHIP_AUTH_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_AUTH_STACK[@]}")
[[ ${#PARTNERSHIP_ARR_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_ARR_STACK[@]}")
[[ ${#PARTNERSHIP_SERVICES_STACK[@]} -gt 0 ]] && xml_names+=("${PARTNERSHIP_SERVICES_STACK[@]}")
if [[ ${#xml_names[@]} -eq 0 ]]; then
log "No auth/arr/services stack arrays configured — skipping deployed stack cleanup"
return 0
fi
log "Removing owner-deployed containers (auth/arr/services stacks) from $MIRROR..."
for xml_name in "${xml_names[@]}"; do
[[ -z "$xml_name" ]] && continue
local xml_file="${TEMPLATES_DIR}/${xml_name}"
if [[ ! -f "$xml_file" ]]; then
warn " $xml_name not found in local $TEMPLATES_DIR — skipping"
continue
fi
local cname
cname=$(awk 'match($0,/<Name>([^<]+)<\/Name>/,a){print a[1];exit}' "$xml_file")
[[ -z "$cname" ]] && continue
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would stop + rm $cname on $MIRROR"
warn " DRY RUN — would delete appdata for $cname on $MIRROR"
continue
fi
local appdata_paths
appdata_paths=$(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$remote_ip" \
"docker inspect --format '{{range .HostConfig.Binds}}{{println .}}{{end}}' '$cname' 2>/dev/null \
| awk -F: '{print \$1}' | grep '^/mnt/.*/appdata'" 2>/dev/null)
timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" root@"$remote_ip" \
"docker stop '$cname' >/dev/null 2>&1
docker rm '$cname' >/dev/null 2>&1 && echo removed" 2>/dev/null | \
grep -q removed && \
log " $cname removed from $MIRROR ✅" || \
log " $cname not found on $MIRROR — skipping"
while IFS= read -r path; do
[[ -z "$path" ]] && continue
timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" root@"$remote_ip" \
"rm -rf '$path' && echo removed" 2>/dev/null | grep -q removed && \
log " Appdata removed on $MIRROR: $path ✅" || \
warn " Failed to remove appdata on $MIRROR: $path"
done <<< "$appdata_paths"
done
}
# ==============================================================================================
# ── Remove owner-deployed containers locally (mirror-initiated offboard) ─────────────────────
#
# SSHes to owner to read PARTNERSHIP_AUTH_STACK + PARTNERSHIP_ARR_STACK, then uses the
# local templates-user/ copies (SCPed there during onboard) to get container names and
# appdata paths. Appdata collected before removal. Skips gracefully if owner unreachable.
# ==============================================================================================
cleanup_deployed_stack_locally() {
local owner_ip="$1" ssh_key="$2"
local -a xml_names=()
if [[ -n "$owner_ip" ]]; then
local -a auth_arr arr_arr
mapfile -t auth_arr < <(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$owner_ip" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${PARTNERSHIP_AUTH_STACK[@]:-}\"" 2>/dev/null | grep -v '^$')
mapfile -t arr_arr < <(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$owner_ip" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${PARTNERSHIP_ARR_STACK[@]:-}\"" 2>/dev/null | grep -v '^$')
local -a svc_arr
mapfile -t svc_arr < <(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes root@"$owner_ip" \
"source '$SCRIPTS_ROOT/load_config.sh' 2>/dev/null
detect_hosts 2>/dev/null
printf '%s\n' \"\${PARTNERSHIP_SERVICES_STACK[@]:-}\"" 2>/dev/null | grep -v '^$')
xml_names=("${auth_arr[@]}" "${arr_arr[@]}" "${svc_arr[@]}")
fi
if [[ ${#xml_names[@]} -eq 0 ]]; then
log "Could not read deployed stack from owner — skipping auth/arr/services cleanup"
return 0
fi
log "Removing owner-deployed containers (auth/arr/services stacks) locally..."
for xml_name in "${xml_names[@]}"; do
[[ -z "$xml_name" ]] && continue
local xml_file="${TEMPLATES_DIR}/${xml_name}"
if [[ ! -f "$xml_file" ]]; then
warn " $xml_name not found locally — skipping"
continue
fi
local cname
cname=$(awk 'match($0,/<Name>([^<]+)<\/Name>/,a){print a[1];exit}' "$xml_file")
[[ -z "$cname" ]] && continue
if [[ "$DRY_RUN" == true ]]; then
warn " DRY RUN — would stop + rm $cname"
warn " DRY RUN — would delete appdata for $cname"
continue
fi
local appdata_paths=""
if timeout "${DOCKER_TIMEOUT:-30}" docker inspect "$cname" >/dev/null 2>&1; then
appdata_paths=$(docker inspect \
--format '{{range .HostConfig.Binds}}{{println .}}{{end}}' \
"$cname" 2>/dev/null | awk -F: '{print $1}' | grep '^/mnt/.*/appdata')
timeout "${DOCKER_TIMEOUT:-30}" docker stop "$cname" >/dev/null 2>&1 || true
_PM_TRAP_STOPPED+=("$cname")
timeout "${DOCKER_TIMEOUT:-30}" docker rm "$cname" >/dev/null 2>&1 && \
log " $cname removed ✅" || warn " $cname rm failed"
else
log " $cname not found locally — skipping"
fi
while IFS= read -r path; do
[[ -z "$path" ]] && continue
rm -rf "$path" && log " Appdata removed: $path ✅" || warn " Failed to remove: $path"
done <<< "$appdata_paths"
done
}
# ==============================================================================================
# ── Reconfigure a container's WebUI on the remote server ─────────────────────────────────────
# ==============================================================================================
reconfigure_webui() {
local container="$1" port="$2" target_ip="$3"
local ssh_key="$4" remote_ip="$5" label="${6:-remote}"
log "Reconfiguring $container WebUI → ${target_ip}:${port} on $label..."
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would reconfigure $container WebUI to http://${target_ip}:${port}/"
return 0
fi
local template
template=$(timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" root@"$remote_ip" \
"grep -rl '<WebUI>' '$TEMPLATES_DIR/' 2>/dev/null | \
xargs grep -l '\"$container\"' 2>/dev/null | head -1" 2>/dev/null)
if [[ -z "$template" ]]; then
warn "$container template not found on $label — WebUI needs manual reconfiguration"
return 1
fi
timeout "$SSH_TIMEOUT" ssh -i "$ssh_key" \
-o ConnectTimeout="$SSH_TIMEOUT" root@"$remote_ip" \
"sed -i 's|<WebUI>.*</WebUI>|<WebUI>http://${target_ip}:${port}/</WebUI>|g' '$template'" \
2>/dev/null && \
log "$container → http://${target_ip}:${port}/ ✅" || {
error "Failed to reconfigure $container WebUI on $label"
return 1
}
}
# ==============================================================================================
# ── Reconfigure local auth WebUIs to target IP ───────────────────────────────────────────────
# ==============================================================================================
reconfigure_local_webuis() {
local target_ip="$1"
log "Reconfiguring local auth WebUIs → ${target_ip}..."
local failures=0
for entry in "${PARTNERSHIP_AUTH_WEBUIS[@]}"; do
[[ -z "$entry" ]] && continue
local container="${entry%%|*}"
local port="${entry##*|}"
local template
template=$(grep -rl '<WebUI>' "$TEMPLATES_DIR/" 2>/dev/null | \
xargs grep -l "\"$container\"" 2>/dev/null | head -1)
if [[ -z "$template" ]]; then
warn "$container template not found locally"
(( failures++ ))
continue
fi
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would reconfigure $container → http://${target_ip}:${port}/"
continue
fi
sed -i "s|<WebUI>.*</WebUI>|<WebUI>http://${target_ip}:${port}/</WebUI>|g" \
"$template" 2>/dev/null && \
log "$container → http://${target_ip}:${port}/ ✅" || \
{ error "Failed to reconfigure $container"; (( failures++ )); }
done
return $failures
}
@@ -1,361 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ================================= Ramdisk Setup ==============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Creates the tmpfs ramdisk, SSD fallback directory, transcode symlink, and
# pre-creates transcoding-temp on the ramdisk. Run once at array start via
# array_started.sh (System_Essentials/). Idempotent — already-mounted ramdisk
# reports status and exits cleanly. Always resets the symlink to the ramdisk
# on boot, ensuring a clean state regardless of what state it was in before
# shutdown.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# Creates four things in order:
# 1. RAMDISK_PATH — tmpfs mount (size: HOST*_RAMDISK_SIZE ceiling, not a reservation)
# 2. TRANSCODE_SSD — SSD fallback directory and transcoding-temp inside it
# 3. TRANSCODE_LINK — symlink reset to RAMDISK_PATH (clean state at every boot)
# 4. transcoding-temp/ inside RAMDISK_PATH — pre-created before Emby starts
#
# The transcoding-temp pre-creation is critical: if it doesn't exist on the ramdisk
# when Emby starts, Emby searches all accessible paths for an existing one and finds
# the SSD fallback version — routing all sessions there until Emby restarts.
#
# Initialises /tmp/transcode_state.db with current target and flip counters.
# /tmp resets on reboot — correct, transcode state should not persist across boots.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Root Required
# mount and symlink creation require root.
#
# Single Instance Lock
# acquire_lock prevents duplicate runs at array start.
#
# Idempotent Mount Check
# If RAMDISK_PATH is already a mountpoint, reports status and exits cleanly
# without attempting to remount or changing anything.
#
# Notification Validated
# platform_require_cmd confirms the notify script is present before use.
#
# Silent on Success
# Startup script runs on every boot — no output when healthy.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_RAMDISK_SIZE
# tmpfs ceiling (e.g. 10G). Must change together with WARN_GB and LOW_GB.
# Aliased by detect_hosts() → RAMDISK_SIZE.
#
# HOST*_RAMDISK_WARN_GB
# Usage level at which transcode_manager.sh flips symlink to SSD.
#
# HOST*_RAMDISK_LOW_GB
# Usage level at which transcode_manager.sh flips back to ramdisk.
#
# HOST*_TRANSCODE_SSD
# SSD fallback directory path.
# Aliased by detect_hosts() → TRANSCODE_SSD.
#
# master.conf
#
# TRANSCODE_LINK
# Symlink path Emby uses as its transcode directory. Must match the path
# configured in Emby's transcoding settings.
#
# TRANSCODE_CHMOD / TRANSCODE_OWNER
# Permissions applied to both ramdisk and SSD directories. (default: 755 / nobody:users)
#
# ==============================================================================================
# STATE FILES
# ==============================================================================================
#
# /tmp/transcode_state.db — current symlink target + flip count tracking
# Lives in /tmp (ephemeral — resets on reboot correctly)
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# ramdisk_setup.sh
# Normal setup run. Called by array_started.sh at boot.
#
# ramdisk_setup.sh --dry-run
# Show what would be created without creating anything.
#
# ramdisk_setup.sh --status
# Show current ramdisk mount state, symlink target, and SSD directory state.
#
# ramdisk_setup.sh --log
# Verbose output showing each creation step.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root — mount and symlink require root"
exit 1
fi
acquire_lock
# detect_hosts() sets MY_ID and aliases RAMDISK_SIZE, TRANSCODE_SSD etc.
detect_hosts
log "Identity: $MY_ID ($LOCAL_SERVER_NAME)"
log "Ramdisk: $RAMDISK_PATH ($RAMDISK_SIZE)"
log "Fallback: $TRANSCODE_SSD"
log "Symlink: $TRANSCODE_LINK"
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes will be made"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_RAM Ramdisk path: $RAMDISK_PATH"
echo "$ICON_RAM Ramdisk size: $RAMDISK_SIZE"
echo "$ICON_RAM Warn at: ${RAMDISK_WARN_GB}GB"
echo "$ICON_RAM Flip at: ${RAMDISK_LOW_GB}GB"
echo "$ICON_DISK SSD fallback: $TRANSCODE_SSD"
echo "$ICON_LINK Symlink: $TRANSCODE_LINK"
echo "$ICON_GEAR Owner: $TRANSCODE_OWNER"
echo "$ICON_GEAR Mode: $TRANSCODE_CHMOD"
echo ""
if mountpoint -q "$RAMDISK_PATH" 2>/dev/null; then
USAGE=$(df -BG "$RAMDISK_PATH" | awk 'NR==2 {print $3}')
AVAIL=$(df -BG "$RAMDISK_PATH" | awk 'NR==2 {print $4}')
echo " $ICON_RAM Ramdisk: mounted — $USAGE used / $AVAIL available ✅"
else
echo " $ICON_RAM Ramdisk: NOT mounted"
fi
if [[ -L "$TRANSCODE_LINK" ]]; then
TARGET=$(readlink "$TRANSCODE_LINK")
echo " $ICON_LINK Symlink: $TRANSCODE_LINK → $TARGET"
else
echo " $ICON_LINK Symlink: not set"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ━━━ Ramdisk ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_RAM Ramdisk — $MY_ID ━━━"
log "$ICON_RAM Path: $RAMDISK_PATH"
log "$ICON_RAM Size: $RAMDISK_SIZE"
echo ""
START=$(date +%s)
SETUP_SUCCESS=true
if mountpoint -q "$RAMDISK_PATH" 2>/dev/null; then
USAGE=$(df -BG "$RAMDISK_PATH" | awk 'NR==2 {print $3}')
AVAIL=$(df -BG "$RAMDISK_PATH" | awk 'NR==2 {print $4}')
log "Ramdisk already mounted — $USAGE used / $AVAIL available"
log "Skipping mount — verifying symlink and permissions"
else
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would create $RAMDISK_PATH"
warn "DRY RUN — would mount tmpfs ${RAMDISK_SIZE} at $RAMDISK_PATH"
else
log "Creating ramdisk mount point: $RAMDISK_PATH"
mkdir -p "$RAMDISK_PATH" || {
error "Failed to create $RAMDISK_PATH"
notify "Ramdisk setup failed on $(hostname) ($MY_ID) — could not create mount point" \
"Ramdisk Setup" "warning"
exit 1
}
log "Mounting tmpfs ${RAMDISK_SIZE} at $RAMDISK_PATH..."
if mount -t tmpfs -o size="$RAMDISK_SIZE" tmpfs "$RAMDISK_PATH"; then
warn "Ramdisk mounted — ${RAMDISK_SIZE} at $RAMDISK_PATH ✅"
else
error "Failed to mount ramdisk at $RAMDISK_PATH"
notify "Ramdisk setup failed on $(hostname) ($MY_ID) — mount failed" \
"Ramdisk Setup" "warning"
exit 1
fi
fi
fi
# ==============================================================================================
# ━━━ SSD Fallback ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_DISK SSD Fallback ━━━"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would create SSD fallback: $TRANSCODE_SSD"
else
if [[ -d "$TRANSCODE_SSD" ]]; then
log "SSD fallback already exists: $TRANSCODE_SSD"
else
log "Creating SSD fallback directory: $TRANSCODE_SSD"
if mkdir -p "$TRANSCODE_SSD"; then
log "SSD fallback created: $TRANSCODE_SSD ✅"
else
error "Failed to create SSD fallback: $TRANSCODE_SSD"
SETUP_SUCCESS=false
fi
fi
fi
# ==============================================================================================
# ━━━ Symlink ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_LINK Symlink ━━━"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would set $TRANSCODE_LINK → $RAMDISK_PATH"
else
if [[ -L "$TRANSCODE_LINK" ]]; then
CURRENT_TARGET=$(readlink "$TRANSCODE_LINK")
if [[ "$CURRENT_TARGET" == "$RAMDISK_PATH" ]]; then
log "Symlink already points to ramdisk — no change needed ✅"
else
log "Updating symlink: $CURRENT_TARGET → $RAMDISK_PATH"
ln -sfn "$RAMDISK_PATH" "$TRANSCODE_LINK" || {
error "Failed to update symlink"
SETUP_SUCCESS=false
}
fi
elif [[ -e "$TRANSCODE_LINK" ]]; then
warn "$TRANSCODE_LINK exists but is not a symlink — removing and replacing"
rm -rf "$TRANSCODE_LINK"
ln -sfn "$RAMDISK_PATH" "$TRANSCODE_LINK" || {
error "Failed to create symlink"
SETUP_SUCCESS=false
}
else
log "Creating symlink: $TRANSCODE_LINK → $RAMDISK_PATH"
mkdir -p "$(dirname "$TRANSCODE_LINK")"
ln -sfn "$RAMDISK_PATH" "$TRANSCODE_LINK" || {
error "Failed to create symlink"
SETUP_SUCCESS=false
}
fi
[[ "$SETUP_SUCCESS" == true ]] && log "Symlink: $TRANSCODE_LINK → $RAMDISK_PATH ✅"
fi
# ==============================================================================================
# ━━━ Transcoding-temp Directory ━━━
# ==============================================================================================
# Pre-created inside ramdisk so Emby always finds it there at session start.
# Without this Emby creates it at its own first-writable path — which may be
# SSD even when the symlink points at the ramdisk — locking all sessions onto SSD.
echo ""
echo "━━━ $ICON_GEAR Transcoding Temp Directory ━━━"
TRANSCODE_TEMP_DIR="${RAMDISK_PATH}/transcoding-temp"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would create $TRANSCODE_TEMP_DIR"
else
if [[ -d "$TRANSCODE_TEMP_DIR" ]]; then
log "transcoding-temp already exists on ramdisk"
else
if mkdir -p "$TRANSCODE_TEMP_DIR"; then
log "Created transcoding-temp on ramdisk ✅"
else
error "Failed to create transcoding-temp on ramdisk"
SETUP_SUCCESS=false
fi
fi
if [[ -d "$TRANSCODE_TEMP_DIR" ]]; then
chmod "$TRANSCODE_CHMOD" "$TRANSCODE_TEMP_DIR"
chown "$TRANSCODE_OWNER" "$TRANSCODE_TEMP_DIR"
log "Permissions set on transcoding-temp ($TRANSCODE_CHMOD $TRANSCODE_OWNER)"
fi
fi
# ==============================================================================================
# ━━━ Permissions ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_GEAR Permissions ━━━"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would apply $TRANSCODE_CHMOD $TRANSCODE_OWNER to $RAMDISK_PATH and $TRANSCODE_SSD"
else
for path in "$RAMDISK_PATH" "$TRANSCODE_SSD"; do
if [[ -d "$path" ]]; then
chmod "$TRANSCODE_CHMOD" "$path"
chown "$TRANSCODE_OWNER" "$path"
log "Permissions set: $path ($TRANSCODE_CHMOD $TRANSCODE_OWNER)"
fi
done
fi
# ==============================================================================================
# ━━━ Initialise State File ━━━
# ==============================================================================================
if [[ "$DRY_RUN" == false ]]; then
STATE_FILE="${TRANSCODE_STATE_FILE:-${STATE_DIR:-/tmp}/transcode_state.db}"
NOW=$(date +%s)
cat > "$STATE_FILE" <<EOF
current_target=$RAMDISK_PATH
last_flip_time=$NOW
flip_count_hour=0
flip_hour_start=$NOW
EOF
log "State file initialised: $STATE_FILE"
fi
END=$(date +%s)
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY RAMDISK SETUP SUMMARY ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_RAM Ramdisk: $RAMDISK_PATH ($RAMDISK_SIZE)"
echo "$ICON_DISK Fallback: $TRANSCODE_SSD"
echo "$ICON_LINK Symlink: $TRANSCODE_LINK"
echo "$ICON_TIME Duration: $(format_duration $(( END - START )))"
echo ""
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no changes made"
elif [[ "$SETUP_SUCCESS" == true ]]; then
echo "$ICON_DONE Status: done ✅"
else
echo "$ICON_ERROR Status: SETUP HAD ERRORS"
notify "Ramdisk setup errors on $(hostname) ($MY_ID) — check output" \
"Ramdisk Setup" "warning"
exit 1
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
@@ -1,365 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ================================= Ramdisk Setup ==============================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Creates the tmpfs ramdisk, SSD fallback directory, transcode symlink, and
# pre-creates transcoding-temp on the ramdisk. Run once at array start via
# array_started.sh (System_Essentials/). Idempotent — already-mounted ramdisk
# reports status and exits cleanly. Always resets the symlink to the ramdisk
# on boot, ensuring a clean state regardless of what state it was in before
# shutdown.
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# Creates four things in order:
# 1. RAMDISK_PATH — tmpfs mount (size: HOST*_RAMDISK_SIZE ceiling, not a reservation)
# 2. TRANSCODE_SSD — SSD fallback directory and transcoding-temp inside it
# 3. TRANSCODE_LINK — symlink reset to RAMDISK_PATH (clean state at every boot)
# 4. transcoding-temp/ inside RAMDISK_PATH — pre-created before Emby starts
#
# The transcoding-temp pre-creation is critical: if it doesn't exist on the ramdisk
# when Emby starts, Emby searches all accessible paths for an existing one and finds
# the SSD fallback version — routing all sessions there until Emby restarts.
#
# Initialises /tmp/transcode_state.db with current target and flip counters.
# /tmp resets on reboot — correct, transcode state should not persist across boots.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# Root Required
# mount and symlink creation require root.
#
# Single Instance Lock
# acquire_lock prevents duplicate runs at array start.
#
# Idempotent Mount Check
# If RAMDISK_PATH is already a mountpoint, reports status and exits cleanly
# without attempting to remount or changing anything.
#
# Notification Validated
# platform_require_cmd confirms the notify script is present before use.
#
# Silent on Success
# Startup script runs on every boot — no output when healthy.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# host*.conf
#
# HOST*_RAMDISK_SIZE
# tmpfs ceiling (e.g. 10G). Must change together with WARN_GB and LOW_GB.
# Aliased by detect_hosts() → RAMDISK_SIZE.
#
# HOST*_RAMDISK_WARN_GB
# Usage level at which transcode_manager.sh flips symlink to SSD.
#
# HOST*_RAMDISK_LOW_GB
# Usage level at which transcode_manager.sh flips back to ramdisk.
#
# HOST*_TRANSCODE_SSD
# SSD fallback directory path.
# Aliased by detect_hosts() → TRANSCODE_SSD.
#
# master.conf
#
# TRANSCODE_LINK
# Symlink path Emby uses as its transcode directory. Must match the path
# configured in Emby's transcoding settings.
#
# TRANSCODE_CHMOD / TRANSCODE_OWNER
# Permissions applied to both ramdisk and SSD directories. (default: 755 / nobody:users)
#
# TRANSCODE_STATE_FILE
# Override state file path. (default: ${STATE_DIR}/transcode_state.db)
#
# ==============================================================================================
# STATE FILES
# ==============================================================================================
#
# TRANSCODE_STATE_FILE (default: ${STATE_DIR}/transcode_state.db)
# Current symlink target + flip count tracking. Lives in STATE_DIR
# (ephemeral on Unraid — resets on reboot correctly).
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# ramdisk_setup.sh
# Normal setup run. Called by array_started.sh at boot.
#
# ramdisk_setup.sh --dry-run
# Show what would be created without creating anything.
#
# ramdisk_setup.sh --status
# Show current ramdisk mount state, symlink target, and SSD directory state.
#
# ramdisk_setup.sh --log
# Verbose output showing each creation step.
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
# ==============================================================================================
# ━━━ Setup ━━━
# ==============================================================================================
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root — mount and symlink require root"
exit 1
fi
acquire_lock
# detect_hosts() sets MY_ID and aliases RAMDISK_SIZE, TRANSCODE_SSD etc.
detect_hosts
log "Identity: $MY_ID ($LOCAL_SERVER_NAME)"
log "Ramdisk: $RAMDISK_PATH ($RAMDISK_SIZE)"
log "Fallback: $TRANSCODE_SSD"
log "Symlink: $TRANSCODE_LINK"
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes will be made"
# ==============================================================================================
# ━━━ Status ━━━
# ==============================================================================================
if [[ "$SHOW_STATUS" == true ]]; then
echo ""
echo "━━━━━ $ICON_SUMMARY STATUS ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_RAM Ramdisk path: $RAMDISK_PATH"
echo "$ICON_RAM Ramdisk size: $RAMDISK_SIZE"
echo "$ICON_RAM Warn at: ${RAMDISK_WARN_GB}GB"
echo "$ICON_RAM Flip at: ${RAMDISK_LOW_GB}GB"
echo "$ICON_DISK SSD fallback: $TRANSCODE_SSD"
echo "$ICON_LINK Symlink: $TRANSCODE_LINK"
echo "$ICON_GEAR Owner: $TRANSCODE_OWNER"
echo "$ICON_GEAR Mode: $TRANSCODE_CHMOD"
echo ""
if mountpoint -q "$RAMDISK_PATH" 2>/dev/null; then
USAGE=$(df -BG "$RAMDISK_PATH" | awk 'NR==2 {print $3}')
AVAIL=$(df -BG "$RAMDISK_PATH" | awk 'NR==2 {print $4}')
echo " $ICON_RAM Ramdisk: mounted — $USAGE used / $AVAIL available ✅"
else
echo " $ICON_RAM Ramdisk: NOT mounted"
fi
if [[ -L "$TRANSCODE_LINK" ]]; then
TARGET=$(readlink "$TRANSCODE_LINK")
echo " $ICON_LINK Symlink: $TRANSCODE_LINK → $TARGET"
else
echo " $ICON_LINK Symlink: not set"
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━"
exit 0
fi
# ==============================================================================================
# ━━━ Ramdisk ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_RAM Ramdisk — $MY_ID ━━━"
log "$ICON_RAM Path: $RAMDISK_PATH"
log "$ICON_RAM Size: $RAMDISK_SIZE"
echo ""
START=$(date +%s)
SETUP_SUCCESS=true
if mountpoint -q "$RAMDISK_PATH" 2>/dev/null; then
USAGE=$(df -BG "$RAMDISK_PATH" | awk 'NR==2 {print $3}')
AVAIL=$(df -BG "$RAMDISK_PATH" | awk 'NR==2 {print $4}')
log "Ramdisk already mounted — $USAGE used / $AVAIL available"
log "Skipping mount — verifying symlink and permissions"
else
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would create $RAMDISK_PATH"
warn "DRY RUN — would mount tmpfs ${RAMDISK_SIZE} at $RAMDISK_PATH"
else
log "Creating ramdisk mount point: $RAMDISK_PATH"
mkdir -p "$RAMDISK_PATH" || {
error "Failed to create $RAMDISK_PATH"
notify "Ramdisk setup failed on $(hostname) ($MY_ID) — could not create mount point" \
"Ramdisk Setup" "warning"
exit 1
}
log "Mounting tmpfs ${RAMDISK_SIZE} at $RAMDISK_PATH..."
if mount -t tmpfs -o size="$RAMDISK_SIZE" tmpfs "$RAMDISK_PATH"; then
warn "Ramdisk mounted — ${RAMDISK_SIZE} at $RAMDISK_PATH ✅"
else
error "Failed to mount ramdisk at $RAMDISK_PATH"
notify "Ramdisk setup failed on $(hostname) ($MY_ID) — mount failed" \
"Ramdisk Setup" "warning"
exit 1
fi
fi
fi
# ==============================================================================================
# ━━━ SSD Fallback ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_DISK SSD Fallback ━━━"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would create SSD fallback: $TRANSCODE_SSD"
else
if [[ -d "$TRANSCODE_SSD" ]]; then
log "SSD fallback already exists: $TRANSCODE_SSD"
else
log "Creating SSD fallback directory: $TRANSCODE_SSD"
if mkdir -p "$TRANSCODE_SSD"; then
log "SSD fallback created: $TRANSCODE_SSD ✅"
else
error "Failed to create SSD fallback: $TRANSCODE_SSD"
SETUP_SUCCESS=false
fi
fi
fi
# ==============================================================================================
# ━━━ Symlink ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_LINK Symlink ━━━"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would set $TRANSCODE_LINK → $RAMDISK_PATH"
else
if [[ -L "$TRANSCODE_LINK" ]]; then
CURRENT_TARGET=$(readlink "$TRANSCODE_LINK")
if [[ "$CURRENT_TARGET" == "$RAMDISK_PATH" ]]; then
log "Symlink already points to ramdisk — no change needed ✅"
else
log "Updating symlink: $CURRENT_TARGET → $RAMDISK_PATH"
ln -sfn "$RAMDISK_PATH" "$TRANSCODE_LINK" || {
error "Failed to update symlink"
SETUP_SUCCESS=false
}
fi
elif [[ -e "$TRANSCODE_LINK" ]]; then
warn "$TRANSCODE_LINK exists but is not a symlink — removing and replacing"
rm -rf "$TRANSCODE_LINK"
ln -sfn "$RAMDISK_PATH" "$TRANSCODE_LINK" || {
error "Failed to create symlink"
SETUP_SUCCESS=false
}
else
log "Creating symlink: $TRANSCODE_LINK → $RAMDISK_PATH"
mkdir -p "$(dirname "$TRANSCODE_LINK")"
ln -sfn "$RAMDISK_PATH" "$TRANSCODE_LINK" || {
error "Failed to create symlink"
SETUP_SUCCESS=false
}
fi
[[ "$SETUP_SUCCESS" == true ]] && log "Symlink: $TRANSCODE_LINK → $RAMDISK_PATH ✅"
fi
# ==============================================================================================
# ━━━ Transcoding-temp Directory ━━━
# ==============================================================================================
# Pre-created inside ramdisk so Emby always finds it there at session start.
# Without this Emby creates it at its own first-writable path — which may be
# SSD even when the symlink points at the ramdisk — locking all sessions onto SSD.
echo ""
echo "━━━ $ICON_GEAR Transcoding Temp Directory ━━━"
TRANSCODE_TEMP_DIR="${RAMDISK_PATH}/transcoding-temp"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would create $TRANSCODE_TEMP_DIR"
else
if [[ -d "$TRANSCODE_TEMP_DIR" ]]; then
log "transcoding-temp already exists on ramdisk"
else
if mkdir -p "$TRANSCODE_TEMP_DIR"; then
log "Created transcoding-temp on ramdisk ✅"
else
error "Failed to create transcoding-temp on ramdisk"
SETUP_SUCCESS=false
fi
fi
if [[ -d "$TRANSCODE_TEMP_DIR" ]]; then
chmod "$TRANSCODE_CHMOD" "$TRANSCODE_TEMP_DIR"
chown "$TRANSCODE_OWNER" "$TRANSCODE_TEMP_DIR"
log "Permissions set on transcoding-temp ($TRANSCODE_CHMOD $TRANSCODE_OWNER)"
fi
fi
# ==============================================================================================
# ━━━ Permissions ━━━
# ==============================================================================================
echo ""
echo "━━━ $ICON_GEAR Permissions ━━━"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would apply $TRANSCODE_CHMOD $TRANSCODE_OWNER to $RAMDISK_PATH and $TRANSCODE_SSD"
else
for path in "$RAMDISK_PATH" "$TRANSCODE_SSD"; do
if [[ -d "$path" ]]; then
chmod "$TRANSCODE_CHMOD" "$path"
chown "$TRANSCODE_OWNER" "$path"
log "Permissions set: $path ($TRANSCODE_CHMOD $TRANSCODE_OWNER)"
fi
done
fi
# ==============================================================================================
# ━━━ Initialise State File ━━━
# ==============================================================================================
if [[ "$DRY_RUN" == false ]]; then
STATE_FILE="${TRANSCODE_STATE_FILE:-${STATE_DIR:-/tmp}/transcode_state.db}"
NOW=$(date +%s)
cat > "$STATE_FILE" <<EOF
current_target=$RAMDISK_PATH
last_flip_time=$NOW
flip_count_hour=0
flip_hour_start=$NOW
EOF
log "State file initialised: $STATE_FILE"
fi
END=$(date +%s)
# ==============================================================================================
# ━━━ Summary ━━━
# ==============================================================================================
echo ""
echo "━━━━━ $ICON_SUMMARY RAMDISK SETUP SUMMARY ━━━━━"
echo "$ICON_HOST Identity: $MY_ID ($LOCAL_SERVER_NAME)"
echo "$ICON_RAM Ramdisk: $RAMDISK_PATH ($RAMDISK_SIZE)"
echo "$ICON_DISK Fallback: $TRANSCODE_SSD"
echo "$ICON_LINK Symlink: $TRANSCODE_LINK"
echo "$ICON_TIME Duration: $(format_duration $(( END - START )))"
echo ""
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — no changes made"
elif [[ "$SETUP_SUCCESS" == true ]]; then
echo "$ICON_DONE Status: done ✅"
else
echo "$ICON_ERROR Status: SETUP HAD ERRORS"
notify "Ramdisk setup errors on $(hostname) ($MY_ID) — check output" \
"Ramdisk Setup" "warning"
exit 1
fi
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
@@ -1,830 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ========================== HOST1 CONFIGURATION — unRAID-Gmer4Lfe ============================
# ==============================================================================================
# HOST1-specific variables — credentials, container names, share paths, failover lists.
# Sourced after master.conf — values here extend shared profile arrays and add HOST1-specific
# identity, credentials, and container configuration.
#
# Sparse checkout (git) ensures HOST2 never receives this file.
# HOST2 never sees HOST1 credentials — clean separation at the file level.
#
# DO NOT put shared config here — thresholds, toggles, profiles belong in master.conf.
# DO NOT put HOST2 variables here — they belong in host2.conf.
#
# ── INDEX ─────────────────────────────────────────────────────────────────────────────────────
#
# ── IDENTITY & CONNECTIVITY ────────────────────────────────────────────────────────────────
# IDENTITY hostname, SSH key, Unraid API key
# EMBY container name, URL, API key
# JELLYFIN container name, URL, API key
# GITEA API token for SSH key registration
# NOTIFICATIONS Discord webhook
#
# ── PARTNERSHIP ────────────────────────────────────────────────────────────────────────────
# PARTNERSHIP auth containers, backup paths, emby provisioning
#
# ── RSYNC ──────────────────────────────────────────────────────────────────────────────────
# DAILY SYNC SHARES media shares HOST1 owns and pushes to HOST2
# WEEKLY SYNC SHARES appdata shares synced weekly (Sunday 2:30am)
# CRITICAL SYNC SHARES appdata shares synced every 30 minutes
# BACKUP VERIFY shares for checksum verification against remote
# HOST1 RSYNC PROFILE host1-appdata profile for HOST1-specific appdata syncs
#
# ── FALLBACK ───────────────────────────────────────────────────────────────────────────────
# DDNS DDNS containers managed by HOST1
# INTERNET LOSS containers stopped when internet is lost
# FALLBACK TIERS what HOST1 runs for HOST2 per tier
# TIER DELAYS how long HOST1 must be down before each tier activates on HOST2
# RSYNC WRITEBACK HOST1 appdata synced back on handback
#
# ── DOCKER ─────────────────────────────────────────────────────────────────────────────────
# DOCKER DAILY RESTART containers restarted daily
# DOCKER WEEKLY RESTART containers restarted weekly
# DOCKER WATCHDOG memory limits, health URLs, required containers, ignore list
# NETWORK WATCHDOG DDNS domain, NPM URL for connectivity checks
# DOCKER NETWORK CONNECT networks and containers for docker_network_connect.sh
#
# ── MEDIA ──────────────────────────────────────────────────────────────────────────────────
# MEDIA PERMISSIONS share list for media_shares_permissions.sh
# MEDIA CLEANER folder lists for media_cleaner.sh
#
# ── ARR STACK ──────────────────────────────────────────────────────────────────────────────
# DOWNLOADERS slskd, SABnzbd, qBittorrent credentials and URLs
# LIDARR URL, API key, path map
# SONARR URL, API key, path map
# RADARR URL, API key, path map
# ARR RECOVERY per-arr recovery toggles
#
# ── TRANSCODES ─────────────────────────────────────────────────────────────────────────────
# TRANSCODES ramdisk size, thresholds, SSD path, server array
#
# ── MONITORS ───────────────────────────────────────────────────────────────────────────────
# CERTIFICATE MONITOR domains checked for SSL expiry
# SMART HEALTH drives to skip in SMART monitoring
# ZFS REPORT pools to exclude from ZFS health report
#
# ── RESOURCE MANAGER ───────────────────────────────────────────────────────────────────────
# RESOURCE MANAGER containers paused/stopped under memory pressure
#
# ── SYSTEM WATCHDOG ────────────────────────────────────────────────────────────────────────
# SYSTEM WATCHDOG per-host check toggles and NIC configuration
#
# ==============================================================================================
# ==============================================================================================
# ── STORAGE MODE ──────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Storage mode ━━━
# Controls where Varaverk stores scripts, conf, and state files.
# true = internal NVMe/SSD — /boot/config/plugins/varaverk (write-safe, git-direct)
# false = USB flash boot — /mnt/user/appdata/Varaverk (preserves flash lifetime)
HOST1_STORAGE_MODE_INTERNAL=true
# ==============================================================================================
# ── IDENTITY & CONNECTIVITY ───────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Identity ━━━
# HOST1 hostname lives in master.conf (not a credential — safe for all servers).
# SSH key used for all server-to-server operations — rsync, failover container commands.
# Must be in /root/.ssh/ and authorised in HOST2's /root/.ssh/authorized_keys.
HOST1_SSH_KEY="/root/.ssh/gmer4lfe_rsync_automation"
HOST1_OWNER="gmer4lfe"
HOST1_OWNER_EMAIL="gmer4lfe@gmail.com"
# ━━━ Unraid API ━━━
# Used by the Varaverk plugin to query this server's Unraid GraphQL API.
# Generate in Unraid: Settings → Management Access → API Keys → + New Key
HOST1_UNRAID_API_KEY="1825c3a2e03ea5089974f4da2e171aa2d5907a1dea23cc479c33e492c8ff4dbb"
# ━━━ Emby ━━━
# Referenced by transcode_manager.sh, emby_session_report.sh, emby_database_repair.sh,
# weekly_sync_maintenance.sh, and HOST1_TRANSCODE_SERVERS below.
# API key: Emby Dashboard → API Keys → + New Key
HOST1_EMBY_CONTAINER="Emby"
HOST1_EMBY_URL="http://localhost:8096"
HOST1_EMBY_API_KEY="0c27448d93a7431f9ac63569f7655829"
# ━━━ Jellyfin ━━━
# API key: Jellyfin Dashboard → Administration → API Keys → + New Key
HOST1_JELLYFIN_CONTAINER="Jellyfin"
HOST1_JELLYFIN_URL="http://localhost:8095"
HOST1_JELLYFIN_API_KEY="4e820e7df74c4933acec212b1996314e"
# ━━━ Gitea ━━━
# Personal access token for gitea_ssh_setup.sh — registers this server's SSH public key
# with Gitea so git operations use key auth instead of passwords.
# Create in Gitea: Settings → Applications → Generate Token → scope: write:user
HOST1_GITEA_API_TOKEN=""
# ━━━ Notifications ━━━
# Discord webhook — leave blank to disable.
# Per-host so HOST1 and HOST2 can post to different channels or only one server notifies.
HOST1_DISCORD_WEBHOOK=""
# ==============================================================================================
# ── PARTNERSHIP ───────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# HOST1 is always the owner (source of truth) unless --transfer has been run.
# See README-Partnership.md and master.conf PARTNERSHIP section for full lifecycle docs.
# Auth containers reconfigured on onboard/offboard.
# Format: "ContainerName|WebUIPort"
# On onboard → WebUI pointed at owner's Tailscale IP (mirror clicks NPM, gets owner's auth)
# On offboard → WebUI pointed back at localhost
HOST1_PARTNERSHIP_AUTH_WEBUIS=(
"NginxProxyManager|81"
"Lldap-Gmer4Lfe|17170"
"Authelia|9091"
"Authelia-Secondary|9092"
)
# XML templates (from this server's templates-user/) pushed to mirror during onboard.
# These become the mirror's active auth stack, backed by the rsync-synced appdata.
# Update filename if Lldap is renamed to drop the host suffix.
HOST1_PARTNERSHIP_AUTH_STACK=(
# Dependencies first — Mariadb/Redis must be healthy before Authelia starts
"my-Mariadb-Authelia.xml"
"my-Mariadb-Authelia-Secondary.xml"
"my-Redis-Authelia.xml"
"my-Redis-Authelia-Secondary.xml"
# Auth apps — deployed after their deps are confirmed healthy
"my-Authelia.xml"
"my-Authelia-Secondary.xml"
"my-NginxProxyManager.xml"
"my-Lldap-Gmer4Lfe.xml"
)
# XML templates pushed to mirror for the arr stack during onboard.
# Deps (e.g. databases) first if any — same ordering rule as auth stack.
HOST1_PARTNERSHIP_ARR_STACK=(
"my-Sonarr.xml"
"my-Radarr.xml"
"my-Lidarr.xml"
"my-Prowlarr.xml"
"my-Bazarr.xml"
)
# Containers stopped on THIS server before deploying the mirror's stack on onboard.
# Only needed when this server parks its own stack to make room for the mirror's.
HOST1_PARTNERSHIP_REPLACE_CONTAINERS=(
)
# Arr containers stopped on this server when mirror's arr stack is deployed.
HOST1_PARTNERSHIP_ARR_REPLACE_CONTAINERS=(
)
# Paths HOST2 should collect during the grace window after offboard.
# Notified on offboard — no auto-deletion, HOST2 must collect manually within PARTNERSHIP_GRACE_HOURS.
HOST1_PARTNERSHIP_MIRROR_BACKUPS=(
# "/mnt/user/appdata-Fallback/Jayred365-Emby"
)
# Containers parked on this server when partnership is active.
# Stopped on onboard (owner deploys its stack instead), restarted on offboard.
HOST1_PARTNERSHIP_OWN_CONTAINERS=(
# "Emby"
# "NginxProxyManager"
)
# Emby admin provisioning — toggle is owner-only, credentials are per-host.
# Owner enables/disables the feature. Each host sets the account they want on the shared Emby.
# On onboard: owner reads mirror's HOST*_PARTNERSHIP_EMBY_ADMIN_* and creates that account.
# On offboard: account is deleted. Username collision → onboard exits with error.
HOST1_PARTNERSHIP_PROVISION_EMBY_ADMIN=false # owner controls whether Emby is shared
HOST1_PARTNERSHIP_EMBY_PORT=8096
HOST1_PARTNERSHIP_EMBY_ADMIN_USER="" # this server's desired Emby username
HOST1_PARTNERSHIP_EMBY_ADMIN_PASS="" # this server's desired Emby password
# ==============================================================================================
# ── RSYNC ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Daily Sync Shares ━━━
# Shares HOST1 pushes to all other nodes every night (1am via daily_sync_maintenance.sh).
# Mesh model: every node pushes every media share — no ownership, no mirrors.
# arr_sync ensures all arr libraries converge (union). rsync spreads files (additive, no --delete).
# arr_cleanup removes true orphans based on local arr state.
# Any node can download content to any share — it propagates to all nodes on the next cycle.
# Nextcloud is intentionally one-directional (HOST1→HOST2 offsite backup — not arr-managed).
# Uses DEFAULT_RSYNC_OPTS from master.conf — no profile needed.
# For shares needing container stops or custom options — add a profile in master.conf.
HOST1_DAILY_SYNC_SHARES=(
/mnt/user/Books
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Nextcloud
/mnt/user/stand-up_comedy
/mnt/user/Sports
# /mnt/user/Tv_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Movies
/mnt/user/Anime_Shows
)
# Personal encrypted shares — synced for offsite backup, independent of media shares.
# ZFS encrypted at dataset level — remote receives encrypted blocks, cannot read content.
# See README-Rsync_Setup.md for ZFS encryption setup before uncommenting.
HOST1_PERSONAL_SHARES=(
# /mnt/user/HOST1-Personal # uncomment after creating encrypted dataset
)
# ━━━ Weekly Sync Shares ━━━
# Appdata shares synced during the weekly maintenance window (Sunday 2:30am).
# Containers stopped both sides before sync — full clean state guaranteed.
# Profiles drive container stops, excludes, and options — configured in master.conf RSYNC section.
# Order matters — Emby first (larger transfer), then Critical-Data (auth stack).
HOST1_WEEKLY_SYNC_SHARES=(
"/mnt/user/Media_Server/Emby" # emby profile — full clean mirror
"/mnt/user/appdata-Fallback/Critical-Data" # critical-data profile — auth stack
)
# ━━━ Intermediate Sync Shares ━━━
# Shares synced every 4 hours by intermediate_sync_maintenance.sh.
# Uses DEFAULT_RSYNC_OPTS (no --delete) — for sub-daily propagation of metadata or watch state.
# Full media share sync stays in the daily window. Leave empty to skip mid-day rsync entirely.
HOST1_INTERMEDIATE_SYNC_SHARES=(
# Add shares here to enable mid-day rsync
# Example: "/mnt/user/Emby_Metadata"
)
# ━━━ Critical Sync Shares ━━━
# Appdata shares synced every 30 minutes by critical_sync_maintenance.sh.
# Format: "/path/to/share" or "/path/to/share|profile-name"
# Order matters — Critical-Data first (auth stack), then Emby dirty sync.
HOST1_CRITICAL_SYNC_SHARES=(
"/mnt/user/appdata-Fallback/Critical-Data|critical-fallback" # auth dirty sync — stays running
"/mnt/user/Media_Server/Emby|emby-fallback" # Emby dirty sync — stays running
)
# ━━━ Backup Verify ━━━
# Shares verified by backup_verify.sh — random file checksum comparison against remote.
# Leave empty to use HOST1_DAILY_SYNC_SHARES automatically.
# Sample size and minimum file size defined in master.conf.
HOST1_BACKUP_VERIFY_SHARES=(
# leave empty to use HOST1_DAILY_SYNC_SHARES automatically
)
# ━━━ HOST1 Rsync Profile — host1-appdata ━━━
# HOST1-specific appdata sync profile — extends the shared PROFILE_* arrays in master.conf.
# Use for appdata unique to HOST1 (Organizrv2, VaultWarden, UptimeKuma etc.)
# Shared appdata (auth stack, Emby) use dedicated profiles defined in master.conf.
# Run manually: bash Rsync/rsync.sh /mnt/user/appdata-Fallback/HOST1-Appdata --profile=host1-appdata
PROFILE_RSYNC_OPTS[host1-appdata]="-av --info=progress2 --bwlimit=${PROFILE_BW_LIMIT[host1-appdata]:-8000}"
PROFILE_BW_LIMIT[host1-appdata]=8000
PROFILE_RETRY_COUNT[host1-appdata]=3
PROFILE_SLEEP[host1-appdata]=300
PROFILE_CRITICAL_CONTAINER_NAMES[host1-appdata]="Organizrv2-Gmer4Lfe UptimeKuma-Gmer4Lfe VaultWarden-Gmer4Lfe"
PROFILE_DELAYED_CONTAINERS[host1-appdata]=""
PROFILE_CONTAINER_DELAY[host1-appdata]=5
PROFILE_EXCLUDE_DIRS[host1-appdata]="logs *.tmp"
# ==============================================================================================
# ── FALLBACK ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ DDNS ━━━
# DDNS containers HOST1 manages — started/stopped by fallback.sh per DDNS absolute rules:
# Internet loss → stop immediately
# Failover → HOST2 starts HOST1's DDNS as Tier 1 (before any other containers)
# Handback → stop HOST1's DDNS on HOST2 → rsync → start containers → start local DDNS last
HOST1_DDNS_CONTAINERS=(
"Gmer4Lfe.com"
)
# ━━━ Internet Loss ━━━
# Containers stopped immediately on HOST1 when internet connection is lost.
# Prevents external-facing services from operating without connectivity.
FALLBACK_HOST1_STOP_ON_NO_NET=(
"Gmer4Lfe.com"
)
# ━━━ Fallback Tiers — HOST1 Runs for HOST2 ━━━
# Containers HOST1 starts when HOST2 goes down.
# Tier 1 is always immediate — vital services cannot wait.
# Higher tiers activate after HOST2_TIER*_DELAY minutes (set in host2.conf).
FALLBACK_HOST1_COVERS_HOST2_TIER1=(
"Gmer4Lfe.us"
"VaultWarden-Jayred365"
)
FALLBACK_HOST1_COVERS_HOST2_TIER2=(
# "container-placeholder"
)
FALLBACK_HOST1_COVERS_HOST2_TIER3=(
# "container-placeholder"
)
FALLBACK_HOST1_COVERS_HOST2_TIER4=(
# "container-placeholder"
)
# ━━━ Tier Delays — HOST1's Containers on HOST2 ━━━
# How long HOST1 must be down before each tier activates on HOST2 — in minutes.
# Tier 1 is always immediate — no delay var needed.
HOST1_TIER2_DELAY=240 # 4 hours — NextCloud, Immich
HOST1_TIER3_DELAY=720 # 12 hours — secondary services
HOST1_TIER4_DELAY=1440 # 24 hours — arrs + downloaders
# ━━━ Rsync Writeback — HOST1 Appdata Back on Handback ━━━
# Syncs HOST1 appdata BACK to HOST1 when it comes back online after a failover.
# Containers stopped before writeback — clean source, no competing writes.
#
# HOST1_TIER1_WRITEBACK_DELAY: short outages skip Tier 1 writeback — primary state
# is more reliable than dirty sync data for brief outages.
HOST1_TIER1_WRITEBACK_DELAY=60 # skip Emby writeback if outage under 1hr
# Tier 4 automatically syncs HOST1_DAILY_SYNC_SHARES — only list paths NOT in that array.
FALLBACK_HOST1_WRITEBACK_TIER1=(
"/mnt/user/Media_Server/Emby" # watch states built up during outage
)
FALLBACK_HOST1_WRITEBACK_TIER2=(
"/mnt/user/appdata-Fallback/Important-Data" # NextCloud + Postgres — files added during outage
)
FALLBACK_HOST1_WRITEBACK_TIER3=(
# "location-placeholder"
)
FALLBACK_HOST1_WRITEBACK_TIER4=(
"/mnt/user/appdata-Fallback/Arrs_Stack" # arr databases — downloads queued during outage
)
# ==============================================================================================
# ── DOCKER ────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Docker Daily Restart ━━━
# Containers restarted every day via DAILY_MAINTENANCE_SCRIPTS.
# Dispatcharr degrades over time without restart — daily is intentional, not just housekeeping.
# Order matters — auth stack first, then media services.
HOST1_DAILY_RESTART_CONTAINERS=(
"NginxProxyManager"
"Lldap-Gmer4Lfe"
"Authelia"
"Authelia-Secondary"
"Dispatcharr-Iptv-Users"
"Dispatcharr" # Live TV scheduler — degrades without daily restart
"Dispatcharr-Basic"
"ErsatzTV-Emby"
"slskd" # Soulseek connection drops after extended uptime; restart refreshes share index
)
# ━━━ Docker Weekly Restart ━━━
# Less critical services restarted weekly via WEEKLY_MAINTENANCE_SCRIPTS (Sunday 2:30am).
# Containers already stopped for weekly sync — restart adds zero extra downtime.
HOST1_WEEKLY_RESTART_CONTAINERS=(
"NextCloud"
"Organizrv2-Gmer4Lfe"
"AdGuard-Home"
"Immich-Gmer4Lfe"
)
# ━━━ Docker Watchdog ━━━
# Per-HOST1 container configuration for docker_watchdog.sh.
# Shared thresholds and toggles live in master.conf.
# Memory hard limits in MB — immediate restart if exceeded.
# Set at "container is clearly broken" not "container is busy".
# 20GB=20480 18GB=18432 16GB=16384 12GB=12288 8GB=8192 4GB=4096 2GB=2048 1GB=1024
declare -A HOST1_WATCHDOG_CONTAINERS=(
["Emby"]=20480 # 20GB — large library + active transcodes
["LidaTube"]=6144 # 6GB — memory leak over time
["Tdarr"]=6144 # 6GB — encoding is memory intensive
["Code-Server"]=1024 # 1GB — should never need more
)
# HTTP health check URLs — checked every cycle, strike system before restart.
# Only add containers with a meaningful web interface to check.
declare -A HOST1_WATCHDOG_CONTAINER_URLS=(
["Emby"]="http://localhost:8096"
["NginxProxyManager"]="http://localhost:7818"
["Authelia"]="http://localhost:9091/api/health"
["Authelia-Secondary"]="http://localhost:9092/api/health"
["Lldap-Gmer4Lfe"]="http://localhost:17170"
)
# Required containers — must always be running on HOST1.
# Strike system before restart — repeated failures go on skip list, auto-clears on recovery.
# Listed in dependency order — dependencies before dependents.
HOST1_WATCHDOG_REQUIRED_CONTAINERS=(
"NginxProxyManager"
"Lldap-Gmer4Lfe"
"Mariadb-Authelia"
"Mariadb-Authelia-Secondary"
"Redis-Authelia"
"Redis-Authelia-Secondary"
"Authelia"
"Authelia-Secondary"
)
# Containers to skip in Tier 2 global scan — legitimately stopped or frequently restarting.
# Watchdog leaves these alone entirely — no restart attempts, no crash loop tracking.
HOST1_WATCHDOG_SCAN_IGNORE=(
"DashGate"
"PIA-WG-Config-Generator"
"Aperture"
"Aperture-Kids"
"pgvector-18-Apeture-Kids"
"Pgvector18-Aperture"
"emby-test" # broken test container (exit 127 — bad image)
)
# Dependency ordering — skip restarting a container if its dependency is also down.
# Prevents watchdog from restarting Authelia before Mariadb is back up.
# SPACE-SEPARATED STRINGS — converted to array at runtime.
declare -A HOST1_WATCHDOG_DEPENDENCIES=(
["Authelia"]="Mariadb-Authelia Redis-Authelia"
["Authelia-Secondary"]="Mariadb-Authelia Redis-Authelia-Secondary"
["NextCloud"]="Postgres-NextCloud"
)
# Per-container appdata growth suppress ceilings in MB.
# ONLY needed in specific cases — growth rate detection covers all containers automatically.
# Use this when a container legitimately has large stable data and you want to guarantee
# it never triggers a false-positive growth alert. Growth warnings are suppressed while the
# container's dir stays below this ceiling; above it, warnings resume as normal.
# 50GB=51200 25GB=25600 20GB=20480 15GB=15360 10GB=10240 5GB=5120
declare -A HOST1_WATCHDOG_APPDATA_SIZES=(
["Tdarr"]="25600" # 25GB — transcode cache grows legitimately during active jobs
["7dtd"]="20480" # 20GB — game server world data, expected to be large
)
# API-level health checks — checked every cycle alongside HTTP URL checks.
# Format: ["ContainerName"]="url|expected_json_key|expected_value"
# Empty = no API checks for this host.
declare -A HOST1_WATCHDOG_CONTAINER_API_CHECKS=(
)
# ━━━ Network Watchdog ━━━
# Host-specific connectivity config for Watchdogs/System/network_watchdog.sh.
HOST1_NETWORK_WATCHDOG_DDNS_DOMAIN="gmer4lfe.com"
HOST1_NETWORK_WATCHDOG_DDNS_CONTAINER="Gmer4Lfe.com"
HOST1_NETWORK_WATCHDOG_NPM_URL="https://gmer4lfe.com"
# ━━━ Docker Network Connect ━━━
# Containers connected to custom networks at array start by docker_network_connect.sh.
# Networks created if they don't exist — idempotent, safe to re-run.
HOST1_NETWORK_CONNECT_CONTAINERS=(
"memcached"
"Npm-CrowdSec"
)
HOST1_NETWORK_CONNECT_NETWORKS=(
"high-availability"
)
# ==============================================================================================
# ── MEDIA ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Media Permissions ━━━
# Shares that media_shares_permissions.sh applies PERMISSIONS_MODE and PERMISSIONS_OWNER to.
# Runs first in DAILY_MAINTENANCE_SCRIPTS — arr cleanup depends on correct ownership.
HOST1_MEDIA_PERMISSION_SHARES=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/appcache
/mnt/user/Books
/mnt/user/Downloads
/mnt/user/Games
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movie_Recordings
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Photo
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Tv_Recordings
/mnt/user/Tv_Shows
/mnt/user/YouTube
)
# ━━━ Media Cleaner ━━━
# Folder lists for media_cleaner.sh — two profiles: anime and media.
# File patterns shared across all servers — defined in master.conf.
# Called via DAILY_MAINTENANCE_SCRIPTS. Run manually: Media/media_cleaner.sh anime|media
HOST1_ANIME_CLEAN_FOLDERS=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
)
HOST1_MEDIA_CLEAN_FOLDERS=(
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Tv_Shows
)
# ==============================================================================================
# ── ARR STACK ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Used by arr cleanup scripts and arrs_failed_stalled_recovery.sh.
# detect_hosts() selects HOST1 vars when running on HOST1.
#
# PATH MAPS — container path → host path translation.
# Arr stores file paths using container-internal paths — scripts need host paths to scan.
# Add one entry per root folder in arr Settings → Media Management → Root Folders.
# ━━━ Downloaders ━━━
# Used by downloaders_reset.sh — runs every 30min via CRITICAL_MAINTENANCE_SCRIPTS.
# Clears stuck states, purges old history, prepares each client for a clean cycle.
# slskd — clears stuck searches, dead transfers, purges expired failed imports.
# SLSKD_FAILED_IMPORTS_DIR: where Soularr moves albums Lidarr rejected.
HOST1_SLSKD_URL="http://localhost:8980"
HOST1_SLSKD_API_KEY="4bF9kL2mNpQrT7vWxYz1A3dEgHjKoRsU"
HOST1_SLSKD_FAILED_IMPORTS_DIR="/mnt/user/Temp_Storage/Slskd/completed/failed_imports"
# SABnzbd
HOST1_SABNZBD_URL="http://localhost:8180"
HOST1_SABNZBD_API_KEY="8bfefe41d83b4d50883e32859b55ca9a"
# qBittorrent — deleteFiles=false removes torrent from qBit but leaves files on disk.
# Radarr/Sonarr manage actual files independently.
HOST1_QBIT_URL="http://localhost:8080"
HOST1_QBIT_USERNAME="root"
HOST1_QBIT_PASSWORD="Stay0utD!ck"
# ━━━ Lidarr — HOST1 only ━━━
# HOST2 does not run Lidarr — HOST1_LIDARR_RECOVERY flag handles the exit cleanly.
HOST1_LIDARR_URL="http://localhost:8686"
HOST1_LIDARR_API_KEY="b2977e71ef074bc0a0529d9fcce3b2dc"
HOST1_LIDARR_MUSIC_ROOT="/mnt/user/Music-New"
HOST1_FANART_API_KEY="Yd7147a43b692df0b364b94dc47efb81"
HOST1_LASTFM_API_KEY="be6dc169c33ae263e690c30d18b7491d"
declare -A HOST1_LIDARR_PATH_MAP=(
["/ext-music"]="/mnt/user/Music-New"
)
# ━━━ Sonarr ━━━
HOST1_SONARR_URL="http://localhost:8989"
HOST1_SONARR_API_KEY="130decd3db5b4c25afad64864cd03f9f"
HOST1_SONARR_TV_ROOT="/mnt/user/Tv_Shows"
# Note: stand-up_comedy in both Sonarr + Radarr — TV specials and movie specials, one folder
declare -A HOST1_SONARR_PATH_MAP=(
["/tv"]="/mnt/user/Tv_Shows"
["/ext-standup-comedy"]="/mnt/user/stand-up_comedy/series"
["/kids tv"]="/mnt/user/Kids_Tv_Shows"
["/ext-anime-shows"]="/mnt/user/Anime_Shows-Old"
)
# ━━━ Radarr ━━━
HOST1_RADARR_URL="http://localhost:7878"
HOST1_RADARR_API_KEY="d43a3ec6cf1549edb4af0cc63f98b2a9"
HOST1_TMDB_API_KEY="3dac5e2e49b5540472d2eafec4f01260"
HOST1_RADARR_MOVIES_ROOT="/mnt/user/Movies"
# Note: stand-up_comedy in both Radarr + Sonarr — movie specials and TV specials, one folder
declare -A HOST1_RADARR_PATH_MAP=(
["/movies"]="/mnt/user/Movies"
["/kids movies"]="/mnt/user/Kids_Movies"
["/ext-stand-up-comedy"]="/mnt/user/stand-up_comedy/specials"
["/anime-movies"]="/mnt/user/Anime_Movies-Old"
["/ext-anime-movies"]="/mnt/user/Anime_Movies-Old"
)
# ━━━ Arr Recovery Toggles ━━━
# false = skip that arr on this host — exits cleanly without error
HOST1_SONARR_RECOVERY=true
HOST1_RADARR_RECOVERY=true
HOST1_LIDARR_RECOVERY=true # HOST1 only — exits cleanly on HOST2
# ==============================================================================================
# ── TRANSCODES ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Ramdisk size ceiling — tmpfs only uses RAM actually needed, not the full size upfront.
# Real-world: 9 streams peaked at ~5.5GB — 10G gives generous headroom on 128GB RAM.
HOST1_RAMDISK_SIZE="10G"
# Usage thresholds — coupled to HOST1_RAMDISK_SIZE, adjust all three together if size changes.
# Hysteresis gap (8.5 - 7 = 1.5GB) prevents flip-flop between ramdisk and SSD.
HOST1_RAMDISK_WARN_GB=8.5 # flip to SSD when ramdisk usage reaches this
HOST1_RAMDISK_LOW_GB=7 # flip back to ramdisk when usage drops to this
# SSD fallback path — where transcodes land when ramdisk exceeds HOST1_RAMDISK_WARN_GB.
# Must be on cache pool — array disks too slow for active transcode writes.
HOST1_TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/"
# Media servers sharing the ramdisk transcode space on HOST1.
# Format: "ContainerName|URL|APIKey|Type" — Type: emby | jellyfin | plex
# Entries with placeholder API keys are skipped automatically.
# ⚠️ Tdarr does NOT belong here — keep Tdarr on SSD, not ramdisk.
HOST1_TRANSCODE_SERVERS=(
"${HOST1_EMBY_CONTAINER}|${HOST1_EMBY_URL}|${HOST1_EMBY_API_KEY}|emby"
"${HOST1_JELLYFIN_CONTAINER}|${HOST1_JELLYFIN_URL}|${HOST1_JELLYFIN_API_KEY}|jellyfin"
)
# ==============================================================================================
# ── MONITORS ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
# Domains checked via direct openssl connection — not relying on NPM's certificate state.
# Checks the actual certificate served, not what NPM thinks it has.
# Thresholds (CERT_WARN_DAYS, CERT_CRIT_DAYS) defined in master.conf.
HOST1_CERT_MONITOR_DOMAINS=(
"Gmer4Lfe.com"
"Gmer4Lfe.us"
)
# ━━━ SMART Health ━━━
# Drives skipped in SMART attribute monitoring — hardware is server-specific.
# Thresholds read from dynamix.cfg at runtime — fallbacks in master.conf.
HOST1_SMART_IGNORE_DRIVES=(
"sda" # boot USB — SMART not meaningful on flash drives
)
# ━━━ ZFS Report ━━━
# Pools excluded from the weekly ZFS health report — reduces noise from single-disk array pools.
# These are individual array disks formatted as ZFS — converting to XFS over time via unBalance.
# Pool health thresholds defined in master.conf.
HOST1_ZFS_REPORT_IGNORE_POOLS=(
"disk5"
"disk6"
"disk8"
"disk9"
"disk10"
)
# ==============================================================================================
# ── RESOURCE MANAGER ──────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Containers to manage under pressure — see master.conf RW_CRITICAL_CONTAINERS for exclusions.
# docker pause at medium pressure (RAM < RW_RAM_MEDIUM_GB or load > medium threshold)
# Suspended in-place — instant to pause/unpause, no state lost, no restart delay.
HOST1_RW_PAUSE_CONTAINERS=(
"Huntarr" # arr search automation — safe to suspend
"Cleanuparr" # download cleanup — safe to suspend
"Healarr" # arr health checks — safe to suspend
"Soularr" # Slskd automation — background only
"ChannelTube" # YouTube archiver — background only
"Pinchflat" # YouTube archiver — background only
)
# docker stop at hard pressure (RAM < RW_RAM_HARD_GB)
# Full stop — these are optional/heavy services that free significant RAM when stopped.
# resource_watchdog.sh restarts them when pressure fully clears (RAM >= RW_RAM_RECOVER_GB).
HOST1_RW_STOP_CONTAINERS=(
"LocalAI" # GPU/CPU heavy — largest RAM consumer when idle
"7DaysToDie" # game server — optional
"V-Rising" # game server — optional
"Code-Server" # IDE — not needed during pressure events
)
# ==============================================================================================
# ── SYSTEM WATCHDOG ───────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Per-host check toggles and NIC config for system_watchdog.sh.
# Aliased by detect_hosts() — script uses unprefixed SYS_WATCHDOG_* names.
# HOST1: TR1950X 128GB — full media server, active transcoding, ZFS cache pools.
#
# Three-tier response — all critical checks enabled by default on HOST1:
# Tier 1 (bypass strikes, reboot now): docker daemon, rootfs full, kernel oops, FD, /boot
# Tier 2 (bypass strikes with OOM): RAM critical + OOM kills in cycle
# Tier 3 (standard strike system): everything else
#
# RAM tiers, OOM limits, and reboot loop settings in master.conf System Watchdog section.
# ━━━ Primary NIC ━━━
# Network interface for NIC state check — verify with: ip link show | grep "^[0-9]"
# Common values: eth0, bond0, br0, eno1
HOST1_SYS_WATCHDOG_NIC="eth0"
# ━━━ Tier 1 — Critical Checks ━━━
# These bypass the strike system — a single hit triggers immediate reboot.
# Disabling any of these is not recommended — they protect against acute system failure.
# Docker daemon unresponsive → try restart, reboot if restart fails.
# Without a working daemon docker_watchdog.sh is blind and containers cannot be managed.
HOST1_SYS_WATCHDOG_CHECK_DOCKER_DAEMON=true
# rootfs at critical threshold (ROOTFS_CRITICAL_PCT=99) → reboot immediately.
# At 99% rootfs writes fail silently — logs stop, Docker errors out, SSH may stop working.
# Standard 95% threshold still uses strike system — only 99%+ is critical tier.
HOST1_SYS_WATCHDOG_CHECK_ROOTFS=true
# Kernel BUG/Oops in dmesg delta since last cycle → reboot immediately.
# A kernel oops means the kernel ran with a corrupted state — stability is not guaranteed.
HOST1_SYS_WATCHDOG_CHECK_KERNEL_OOPS=true
# File descriptor exhaustion at FD_CRITICAL_PCT (95%) → reboot immediately.
# At 95% FD: new connections fail, Docker can't spawn processes, SSH drops.
HOST1_SYS_WATCHDOG_CHECK_FD=true
# /boot read-only detected → reboot immediately.
# Unexpected read-only /boot means state files and config writes are silently failing.
# Fallback state, watchdog reboot log, and lock files all go stale silently.
HOST1_SYS_WATCHDOG_CHECK_BOOT=true
# ━━━ Tier 2 — Urgent OOM Check ━━━
# Bypass strikes when RAM is critically low AND OOM kill rate confirms active crisis.
# Both must be enabled for Tier 2 bypass to function — disable either to always use strikes.
# Track kernel OOM kills each cycle via /proc/vmstat oom_kill delta.
# Also provides diagnostic context in reboot messages (which processes were killed).
HOST1_SYS_WATCHDOG_CHECK_OOM=true
# Free RAM check — required for both Tier 2 bypass and RAM tier logic.
# Tiers: MEM_WARN_GB(10) → notify | MEM_SHUTDOWN_GB(6) → stop containers | MEM_GB(4) → strikes
HOST1_SYS_WATCHDOG_CHECK_RAM=true
# ━━━ Tier 3 — Standard Checks (strike system) ━━━
# Each check must fail SYS_WATCHDOG_STRIKE_LIMIT consecutive cycles before action is taken.
# Single spikes are ignored — sustained problems trigger reboot.
# /var/log filesystem usage above SYS_WATCHDOG_LOG_PCT.
# Log spam (Docker log storms, syslog loops) fills rootfs — indicates something broken.
HOST1_SYS_WATCHDOG_CHECK_LOG=true
# ZFS ARC memory pinned above SYS_WATCHDOG_ARC_PINNED_PCT after cache drop.
# Enabled on HOST1 — ZFS cache pools actively used. Disable on hosts without ZFS.
HOST1_SYS_WATCHDOG_CHECK_ARC=true
# CPU temperature above SYS_WATCHDOG_CPU_TEMP_MAX (95°C).
# Sustained high temp causes kernel throttling or panic. Requires lm-sensors.
HOST1_SYS_WATCHDOG_CHECK_CPU_TEMP=true
# Load average above SYS_WATCHDOG_LOAD_MULTIPLIER × core count.
# DISABLED on HOST1 — Tdarr and Emby cause legitimate sustained load spikes during encoding.
# Enable on idle servers or adjust SYS_WATCHDOG_LOAD_MULTIPLIER if load is always high.
HOST1_SYS_WATCHDOG_CHECK_LOAD=false
# Zombie process count above SYS_WATCHDOG_ZOMBIE_LIMIT (50).
# Large zombie counts indicate serious process management failure — something is stuck.
HOST1_SYS_WATCHDOG_CHECK_ZOMBIES=true
# Check docker_watchdog.sh persistent skip list — required containers on skip list.
# Cross-watchdog coordination: if docker_watchdog gave up, system_watchdog escalates.
# ENABLED — HOST1 fully built and operational, skip list is meaningful.
HOST1_SYS_WATCHDOG_CHECK_CONTAINERS=true
# /tmp filesystem usage above SYS_WATCHDOG_TMP_PCT with auto-clear attempt.
# Script tries to clear aged /tmp files first — only strikes if clear fails.
# Lock files, rsync temp files, and Docker ops use /tmp — 100% means lock failures.
HOST1_SYS_WATCHDOG_CHECK_TMP=true
# Array disk error count delta in /proc/mdstat — accumulating errors = disk failing now.
# Triggers on SYS_WATCHDOG_MDSTAT_ERROR_LIMIT new errors in one cycle.
HOST1_SYS_WATCHDOG_CHECK_MDSTAT=true
# Primary NIC operstate — detects NIC going down (physical or driver failure).
# Uses HOST1_SYS_WATCHDOG_NIC above. Strike system — brief flaps don't trigger reboot.
HOST1_SYS_WATCHDOG_CHECK_NETWORK=true
# sshd running check — attempts restart before escalating.
# sshd down = no remote access. Script tries rc.sshd start, notifies, strikes on failure.
HOST1_SYS_WATCHDOG_CHECK_SSHD=true
# Runaway process detection — single process above SYS_WATCHDOG_RUNAWAY_CPU_PCT sustained.
# DISABLED — Tdarr encoding and Emby transcoding legitimately peg CPU for extended periods.
# Enable only if HOST1 has no CPU-intensive workloads.
HOST1_SYS_WATCHDOG_CHECK_RUNAWAY=false
# ==============================================================================================
# ── AUTH STACK ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Credentials for the Varaverk Auth Stack page (NPM, lldap, Authelia).
# HOST1 is the auth source of truth — these are the live production credentials.
# ━━━ NginxProxyManager ━━━
# Admin API runs on 7818 (not 81 — 81 is the partnership WebUI port).
HOST1_NPM_URL="http://localhost:7818"
HOST1_NPM_USER="" # NPM admin email
HOST1_NPM_PASS="" # NPM admin password
# ━━━ lldap ━━━
HOST1_LLDAP_URL="http://localhost:17170"
HOST1_LLDAP_USER="admin" # lldap admin username
HOST1_LLDAP_PASS="" # lldap admin password
# ━━━ Authelia ━━━
HOST1_AUTHELIA_CONFIG="/mnt/user/appdata/Authelia/configuration.yml"
HOST1_AUTHELIA_CONTAINER="Authelia"
# ==============================================================================================
# ──────────────────────── End Of HOST1 Variables ──────────────────────────────────────────────
# ==============================================================================================
@@ -1,843 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ========================== HOST1 CONFIGURATION — unRAID-Gmer4Lfe ============================
# ==============================================================================================
# HOST1-specific variables — credentials, container names, share paths, failover lists.
# Sourced after master.conf — values here extend shared profile arrays and add HOST1-specific
# identity, credentials, and container configuration.
#
# Sparse checkout (git) ensures HOST2 never receives this file.
# HOST2 never sees HOST1 credentials — clean separation at the file level.
#
# DO NOT put shared config here — thresholds, toggles, profiles belong in master.conf.
# DO NOT put HOST2 variables here — they belong in host2.conf.
#
# ── INDEX ─────────────────────────────────────────────────────────────────────────────────────
#
# ── IDENTITY & CONNECTIVITY ────────────────────────────────────────────────────────────────
# IDENTITY hostname, SSH key, Unraid API key
# EMBY container name, URL, API key
# JELLYFIN container name, URL, API key
# GITEA API token for SSH key registration
# NOTIFICATIONS Discord webhook
#
# ── PARTNERSHIP ────────────────────────────────────────────────────────────────────────────
# PARTNERSHIP auth containers, backup paths, emby provisioning
#
# ── RSYNC ──────────────────────────────────────────────────────────────────────────────────
# DAILY SYNC SHARES media shares HOST1 owns and pushes to HOST2
# WEEKLY SYNC SHARES appdata shares synced weekly (Sunday 2:30am)
# CRITICAL SYNC SHARES appdata shares synced every 30 minutes
# BACKUP VERIFY shares for checksum verification against remote
# HOST1 RSYNC PROFILE host1-appdata profile for HOST1-specific appdata syncs
#
# ── FALLBACK ───────────────────────────────────────────────────────────────────────────────
# DDNS DDNS containers managed by HOST1
# INTERNET LOSS containers stopped when internet is lost
# FALLBACK TIERS what HOST1 runs for HOST2 per tier
# TIER DELAYS how long HOST1 must be down before each tier activates on HOST2
# RSYNC WRITEBACK HOST1 appdata synced back on handback
#
# ── DOCKER ─────────────────────────────────────────────────────────────────────────────────
# DOCKER DAILY RESTART containers restarted daily
# DOCKER WEEKLY RESTART containers restarted weekly
# DOCKER WATCHDOG memory limits, health URLs, required containers, ignore list
# NETWORK WATCHDOG DDNS domain, NPM URL for connectivity checks
# DOCKER NETWORK CONNECT networks and containers for docker_network_connect.sh
#
# ── MEDIA ──────────────────────────────────────────────────────────────────────────────────
# MEDIA PERMISSIONS share list for media_shares_permissions.sh
# MEDIA CLEANER folder lists for media_cleaner.sh
#
# ── ARR STACK ──────────────────────────────────────────────────────────────────────────────
# DOWNLOADERS slskd, SABnzbd, qBittorrent credentials and URLs
# LIDARR URL, API key, path map
# SONARR URL, API key, path map
# RADARR URL, API key, path map
# ARR RECOVERY per-arr recovery toggles
#
# ── TRANSCODES ─────────────────────────────────────────────────────────────────────────────
# TRANSCODES ramdisk size, thresholds, SSD path, server array
#
# ── MONITORS ───────────────────────────────────────────────────────────────────────────────
# CERTIFICATE MONITOR domains checked for SSL expiry
# SMART HEALTH drives to skip in SMART monitoring
# ZFS REPORT pools to exclude from ZFS health report
#
# ── RESOURCE MANAGER ───────────────────────────────────────────────────────────────────────
# RESOURCE MANAGER containers paused/stopped under memory pressure
#
# ── SYSTEM WATCHDOG ────────────────────────────────────────────────────────────────────────
# SYSTEM WATCHDOG per-host check toggles and NIC configuration
#
# ==============================================================================================
# ==============================================================================================
# ── STORAGE MODE ──────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Storage mode ━━━
# Controls where Varaverk stores scripts, conf, and state files.
# true = internal NVMe/SSD — /boot/config/plugins/varaverk (write-safe, git-direct)
# false = USB flash boot — /mnt/user/appdata/Varaverk (preserves flash lifetime)
HOST1_STORAGE_MODE_INTERNAL=true
# ==============================================================================================
# ── IDENTITY & CONNECTIVITY ───────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Identity ━━━
# HOST1 hostname lives in master.conf (not a credential — safe for all servers).
# SSH key used for all server-to-server operations — rsync, failover container commands.
# Must be in /root/.ssh/ and authorised in HOST2's /root/.ssh/authorized_keys.
HOST1_SSH_KEY="/root/.ssh/gmer4lfe_rsync_automation"
HOST1_OWNER="gmer4lfe"
HOST1_OWNER_EMAIL="gmer4lfe@gmail.com"
# ━━━ Unraid API ━━━
# Used by the Varaverk plugin to query this server's Unraid GraphQL API.
# Generate in Unraid: Settings → Management Access → API Keys → + New Key
HOST1_UNRAID_API_KEY="1825c3a2e03ea5089974f4da2e171aa2d5907a1dea23cc479c33e492c8ff4dbb"
# ━━━ Emby ━━━
# Referenced by transcode_manager.sh, emby_session_report.sh, emby_database_repair.sh,
# weekly_sync_maintenance.sh, and HOST1_TRANSCODE_SERVERS below.
# API key: Emby Dashboard → API Keys → + New Key
HOST1_EMBY_CONTAINER="Emby"
HOST1_EMBY_URL="http://localhost:8096"
HOST1_EMBY_API_KEY="0c27448d93a7431f9ac63569f7655829"
# ━━━ Jellyfin ━━━
# API key: Jellyfin Dashboard → Administration → API Keys → + New Key
HOST1_JELLYFIN_CONTAINER="Jellyfin"
HOST1_JELLYFIN_URL="http://localhost:8095"
HOST1_JELLYFIN_API_KEY="4e820e7df74c4933acec212b1996314e"
# ━━━ Gitea ━━━
# Personal access token for gitea_ssh_setup.sh — registers this server's SSH public key
# with Gitea so git operations use key auth instead of passwords.
# Create in Gitea: Settings → Applications → Generate Token → scope: write:user
HOST1_GITEA_API_TOKEN=""
# ━━━ Notifications ━━━
# Discord webhook — leave blank to disable.
# Per-host so HOST1 and HOST2 can post to different channels or only one server notifies.
HOST1_DISCORD_WEBHOOK=""
# ==============================================================================================
# ── PARTNERSHIP ───────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# HOST1 is always the owner (source of truth) unless --transfer has been run.
# See README-Partnership.md and master.conf PARTNERSHIP section for full lifecycle docs.
# Auth containers reconfigured on onboard/offboard.
# Format: "ContainerName|WebUIPort"
# On onboard → WebUI pointed at owner's Tailscale IP (mirror clicks NPM, gets owner's auth)
# On offboard → WebUI pointed back at localhost
HOST1_PARTNERSHIP_AUTH_WEBUIS=(
"NginxProxyManager|81"
"Lldap-Gmer4Lfe|17170"
"Authelia|9091"
"Authelia-Secondary|9092"
)
# XML templates (from this server's templates-user/) pushed to mirror during onboard.
# These become the mirror's active auth stack, backed by the rsync-synced appdata.
# Update filename if Lldap is renamed to drop the host suffix.
HOST1_PARTNERSHIP_AUTH_STACK=(
# Dependencies first — Mariadb/Redis must be healthy before Authelia starts
"my-Mariadb-Authelia.xml"
"my-Mariadb-Authelia-Secondary.xml"
"my-Redis-Authelia.xml"
"my-Redis-Authelia-Secondary.xml"
# Auth apps — deployed after their deps are confirmed healthy
"my-Authelia.xml"
"my-Authelia-Secondary.xml"
"my-NginxProxyManager.xml"
"my-Lldap-Gmer4Lfe.xml"
)
# XML templates pushed to mirror for the arr stack during onboard.
# Deps (e.g. databases) first if any — same ordering rule as auth stack.
HOST1_PARTNERSHIP_ARR_STACK=(
"my-Sonarr.xml"
"my-Radarr.xml"
"my-Lidarr.xml"
"my-Prowlarr.xml"
"my-Bazarr.xml"
)
# Shared media services deployed on the mirror during onboard.
# Emby, Jellyfin, and request managers — deployed after arrs so library paths exist.
HOST1_PARTNERSHIP_SERVICES_STACK=(
"my-Emby.xml"
"my-jellyfin.xml"
"my-Seerr.xml"
"my-SeerrFin.xml"
)
# Containers stopped on THIS server before deploying the mirror's stack on onboard.
# Only needed when this server parks its own stack to make room for the mirror's.
HOST1_PARTNERSHIP_REPLACE_CONTAINERS=(
)
# Arr containers stopped on this server when mirror's arr stack is deployed.
HOST1_PARTNERSHIP_ARR_REPLACE_CONTAINERS=(
)
# Services containers stopped on this server when mirror's services stack is deployed.
HOST1_PARTNERSHIP_SERVICES_REPLACE_CONTAINERS=(
)
# Paths HOST2 should collect during the grace window after offboard.
# Notified on offboard — no auto-deletion, HOST2 must collect manually within PARTNERSHIP_GRACE_HOURS.
HOST1_PARTNERSHIP_MIRROR_BACKUPS=(
# "/mnt/user/appdata-Fallback/Jayred365-Emby"
)
# Containers parked on this server when partnership is active.
# Stopped on onboard (owner deploys its stack instead), restarted on offboard.
HOST1_PARTNERSHIP_OWN_CONTAINERS=(
# "Emby"
# "NginxProxyManager"
)
# Emby admin provisioning — toggle is owner-only, credentials are per-host.
# Owner enables/disables the feature. Each host sets the account they want on the shared Emby.
# On onboard: owner reads mirror's HOST*_PARTNERSHIP_EMBY_ADMIN_* and creates that account.
# On offboard: account is deleted. Username collision → onboard exits with error.
HOST1_PARTNERSHIP_PROVISION_EMBY_ADMIN=false # owner controls whether Emby is shared
HOST1_PARTNERSHIP_EMBY_PORT=8096
HOST1_PARTNERSHIP_EMBY_ADMIN_USER="" # this server's desired Emby username
HOST1_PARTNERSHIP_EMBY_ADMIN_PASS="" # this server's desired Emby password
# ==============================================================================================
# ── RSYNC ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Daily Sync Shares ━━━
# Shares HOST1 pushes to all other nodes every night (1am via daily_sync_maintenance.sh).
# Mesh model: every node pushes every media share — no ownership, no mirrors.
# arr_sync ensures all arr libraries converge (union). rsync spreads files (additive, no --delete).
# arr_cleanup removes true orphans based on local arr state.
# Any node can download content to any share — it propagates to all nodes on the next cycle.
# Nextcloud is intentionally one-directional (HOST1→HOST2 offsite backup — not arr-managed).
# Uses DEFAULT_RSYNC_OPTS from master.conf — no profile needed.
# For shares needing container stops or custom options — add a profile in master.conf.
HOST1_DAILY_SYNC_SHARES=(
/mnt/user/Books
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Nextcloud
/mnt/user/stand-up_comedy
/mnt/user/Sports
# /mnt/user/Tv_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Movies
/mnt/user/Anime_Shows
)
# Personal encrypted shares — synced for offsite backup, independent of media shares.
# ZFS encrypted at dataset level — remote receives encrypted blocks, cannot read content.
# See README-Rsync_Setup.md for ZFS encryption setup before uncommenting.
HOST1_PERSONAL_SHARES=(
# /mnt/user/HOST1-Personal # uncomment after creating encrypted dataset
)
# ━━━ Weekly Sync Shares ━━━
# Appdata shares synced during the weekly maintenance window (Sunday 2:30am).
# Containers stopped both sides before sync — full clean state guaranteed.
# Profiles drive container stops, excludes, and options — configured in master.conf RSYNC section.
# Order matters — Emby first (larger transfer), then Critical-Data (auth stack).
HOST1_WEEKLY_SYNC_SHARES=(
"/mnt/user/Media_Server/Emby" # emby profile — full clean mirror
"/mnt/user/appdata-Fallback/Critical-Data" # critical-data profile — auth stack
)
# ━━━ Intermediate Sync Shares ━━━
# Shares synced every 4 hours by intermediate_sync_maintenance.sh.
# Uses DEFAULT_RSYNC_OPTS (no --delete) — for sub-daily propagation of metadata or watch state.
# Full media share sync stays in the daily window. Leave empty to skip mid-day rsync entirely.
HOST1_INTERMEDIATE_SYNC_SHARES=(
# Add shares here to enable mid-day rsync
# Example: "/mnt/user/Emby_Metadata"
)
# ━━━ Critical Sync Shares ━━━
# Appdata shares synced every 30 minutes by critical_sync_maintenance.sh.
# Format: "/path/to/share" or "/path/to/share|profile-name"
# Order matters — Critical-Data first (auth stack), then Emby dirty sync.
HOST1_CRITICAL_SYNC_SHARES=(
"/mnt/user/appdata-Fallback/Critical-Data|critical-fallback" # auth dirty sync — stays running
"/mnt/user/Media_Server/Emby|emby-fallback" # Emby dirty sync — stays running
)
# ━━━ Backup Verify ━━━
# Shares verified by backup_verify.sh — random file checksum comparison against remote.
# Leave empty to use HOST1_DAILY_SYNC_SHARES automatically.
# Sample size and minimum file size defined in master.conf.
HOST1_BACKUP_VERIFY_SHARES=(
# leave empty to use HOST1_DAILY_SYNC_SHARES automatically
)
# ━━━ HOST1 Rsync Profile — host1-appdata ━━━
# HOST1-specific appdata sync profile — extends the shared PROFILE_* arrays in master.conf.
# Use for appdata unique to HOST1 (Organizrv2, VaultWarden, UptimeKuma etc.)
# Shared appdata (auth stack, Emby) use dedicated profiles defined in master.conf.
# Run manually: bash Rsync/rsync.sh /mnt/user/appdata-Fallback/HOST1-Appdata --profile=host1-appdata
PROFILE_RSYNC_OPTS[host1-appdata]="-av --info=progress2 --bwlimit=${PROFILE_BW_LIMIT[host1-appdata]:-8000}"
PROFILE_BW_LIMIT[host1-appdata]=8000
PROFILE_RETRY_COUNT[host1-appdata]=3
PROFILE_SLEEP[host1-appdata]=300
PROFILE_CRITICAL_CONTAINER_NAMES[host1-appdata]="Organizrv2-Gmer4Lfe UptimeKuma-Gmer4Lfe VaultWarden-Gmer4Lfe"
PROFILE_DELAYED_CONTAINERS[host1-appdata]=""
PROFILE_CONTAINER_DELAY[host1-appdata]=5
PROFILE_EXCLUDE_DIRS[host1-appdata]="logs *.tmp"
# ==============================================================================================
# ── FALLBACK ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ DDNS ━━━
# DDNS containers HOST1 manages — started/stopped by fallback.sh per DDNS absolute rules:
# Internet loss → stop immediately
# Failover → HOST2 starts HOST1's DDNS as Tier 1 (before any other containers)
# Handback → stop HOST1's DDNS on HOST2 → rsync → start containers → start local DDNS last
HOST1_DDNS_CONTAINERS=(
"Gmer4Lfe.com"
)
# ━━━ Internet Loss ━━━
# Containers stopped immediately on HOST1 when internet connection is lost.
# Prevents external-facing services from operating without connectivity.
FALLBACK_HOST1_STOP_ON_NO_NET=(
"Gmer4Lfe.com"
)
# ━━━ Fallback Tiers — HOST1 Runs for HOST2 ━━━
# Containers HOST1 starts when HOST2 goes down.
# Tier 1 is always immediate — vital services cannot wait.
# Higher tiers activate after HOST2_TIER*_DELAY minutes (set in host2.conf).
FALLBACK_HOST1_COVERS_HOST2_TIER1=(
"Gmer4Lfe.us"
"VaultWarden-Jayred365"
)
FALLBACK_HOST1_COVERS_HOST2_TIER2=(
# "container-placeholder"
)
FALLBACK_HOST1_COVERS_HOST2_TIER3=(
# "container-placeholder"
)
FALLBACK_HOST1_COVERS_HOST2_TIER4=(
# "container-placeholder"
)
# ━━━ Tier Delays — HOST1's Containers on HOST2 ━━━
# How long HOST1 must be down before each tier activates on HOST2 — in minutes.
# Tier 1 is always immediate — no delay var needed.
HOST1_TIER2_DELAY=240 # 4 hours — NextCloud, Immich
HOST1_TIER3_DELAY=720 # 12 hours — secondary services
HOST1_TIER4_DELAY=1440 # 24 hours — arrs + downloaders
# ━━━ Rsync Writeback — HOST1 Appdata Back on Handback ━━━
# Syncs HOST1 appdata BACK to HOST1 when it comes back online after a failover.
# Containers stopped before writeback — clean source, no competing writes.
#
# HOST1_TIER1_WRITEBACK_DELAY: short outages skip Tier 1 writeback — primary state
# is more reliable than dirty sync data for brief outages.
HOST1_TIER1_WRITEBACK_DELAY=60 # skip Emby writeback if outage under 1hr
# Tier 4 automatically syncs HOST1_DAILY_SYNC_SHARES — only list paths NOT in that array.
FALLBACK_HOST1_WRITEBACK_TIER1=(
"/mnt/user/Media_Server/Emby" # watch states built up during outage
)
FALLBACK_HOST1_WRITEBACK_TIER2=(
"/mnt/user/appdata-Fallback/Important-Data" # NextCloud + Postgres — files added during outage
)
FALLBACK_HOST1_WRITEBACK_TIER3=(
# "location-placeholder"
)
FALLBACK_HOST1_WRITEBACK_TIER4=(
"/mnt/user/appdata-Fallback/Arrs_Stack" # arr databases — downloads queued during outage
)
# ==============================================================================================
# ── DOCKER ────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Docker Daily Restart ━━━
# Containers restarted every day via DAILY_MAINTENANCE_SCRIPTS.
# Dispatcharr degrades over time without restart — daily is intentional, not just housekeeping.
# Order matters — auth stack first, then media services.
HOST1_DAILY_RESTART_CONTAINERS=(
"NginxProxyManager"
"Lldap-Gmer4Lfe"
"Authelia"
"Authelia-Secondary"
"Dispatcharr-Iptv-Users"
"Dispatcharr" # Live TV scheduler — degrades without daily restart
"Dispatcharr-Basic"
"ErsatzTV-Emby"
"slskd" # Soulseek connection drops after extended uptime; restart refreshes share index
)
# ━━━ Docker Weekly Restart ━━━
# Less critical services restarted weekly via WEEKLY_MAINTENANCE_SCRIPTS (Sunday 2:30am).
# Containers already stopped for weekly sync — restart adds zero extra downtime.
HOST1_WEEKLY_RESTART_CONTAINERS=(
"NextCloud"
"Organizrv2-Gmer4Lfe"
"AdGuard-Home"
"Immich-Gmer4Lfe"
)
# ━━━ Docker Watchdog ━━━
# Per-HOST1 container configuration for docker_watchdog.sh.
# Shared thresholds and toggles live in master.conf.
# Memory hard limits in MB — immediate restart if exceeded.
# Set at "container is clearly broken" not "container is busy".
# 20GB=20480 18GB=18432 16GB=16384 12GB=12288 8GB=8192 4GB=4096 2GB=2048 1GB=1024
declare -A HOST1_WATCHDOG_CONTAINERS=(
["Emby"]=20480 # 20GB — large library + active transcodes
["LidaTube"]=6144 # 6GB — memory leak over time
["Tdarr"]=6144 # 6GB — encoding is memory intensive
["Code-Server"]=1024 # 1GB — should never need more
)
# HTTP health check URLs — checked every cycle, strike system before restart.
# Only add containers with a meaningful web interface to check.
declare -A HOST1_WATCHDOG_CONTAINER_URLS=(
["Emby"]="http://localhost:8096"
["NginxProxyManager"]="http://localhost:7818"
["Authelia"]="http://localhost:9091/api/health"
["Authelia-Secondary"]="http://localhost:9092/api/health"
["Lldap-Gmer4Lfe"]="http://localhost:17170"
)
# Required containers — must always be running on HOST1.
# Strike system before restart — repeated failures go on skip list, auto-clears on recovery.
# Listed in dependency order — dependencies before dependents.
HOST1_WATCHDOG_REQUIRED_CONTAINERS=(
"NginxProxyManager"
"Lldap-Gmer4Lfe"
"Mariadb-Authelia"
"Mariadb-Authelia-Secondary"
"Redis-Authelia"
"Redis-Authelia-Secondary"
"Authelia"
"Authelia-Secondary"
)
# Containers to skip in Tier 2 global scan — legitimately stopped or frequently restarting.
# Watchdog leaves these alone entirely — no restart attempts, no crash loop tracking.
HOST1_WATCHDOG_SCAN_IGNORE=(
"DashGate"
"PIA-WG-Config-Generator"
"Aperture"
"Aperture-Kids"
"pgvector-18-Apeture-Kids"
"Pgvector18-Aperture"
"emby-test" # broken test container (exit 127 — bad image)
)
# Dependency ordering — skip restarting a container if its dependency is also down.
# Prevents watchdog from restarting Authelia before Mariadb is back up.
# SPACE-SEPARATED STRINGS — converted to array at runtime.
declare -A HOST1_WATCHDOG_DEPENDENCIES=(
["Authelia"]="Mariadb-Authelia Redis-Authelia"
["Authelia-Secondary"]="Mariadb-Authelia Redis-Authelia-Secondary"
["NextCloud"]="Postgres-NextCloud"
)
# Per-container appdata growth suppress ceilings in MB.
# ONLY needed in specific cases — growth rate detection covers all containers automatically.
# Use this when a container legitimately has large stable data and you want to guarantee
# it never triggers a false-positive growth alert. Growth warnings are suppressed while the
# container's dir stays below this ceiling; above it, warnings resume as normal.
# 50GB=51200 25GB=25600 20GB=20480 15GB=15360 10GB=10240 5GB=5120
declare -A HOST1_WATCHDOG_APPDATA_SIZES=(
["Tdarr"]="25600" # 25GB — transcode cache grows legitimately during active jobs
["7dtd"]="20480" # 20GB — game server world data, expected to be large
)
# API-level health checks — checked every cycle alongside HTTP URL checks.
# Format: ["ContainerName"]="url|expected_json_key|expected_value"
# Empty = no API checks for this host.
declare -A HOST1_WATCHDOG_CONTAINER_API_CHECKS=(
)
# ━━━ Network Watchdog ━━━
# Host-specific connectivity config for Watchdogs/System/network_watchdog.sh.
HOST1_NETWORK_WATCHDOG_DDNS_DOMAIN="gmer4lfe.com"
HOST1_NETWORK_WATCHDOG_DDNS_CONTAINER="Gmer4Lfe.com"
HOST1_NETWORK_WATCHDOG_NPM_URL="https://gmer4lfe.com"
# ━━━ Docker Network Connect ━━━
# Containers connected to custom networks at array start by docker_network_connect.sh.
# Networks created if they don't exist — idempotent, safe to re-run.
HOST1_NETWORK_CONNECT_CONTAINERS=(
"memcached"
"Npm-CrowdSec"
)
HOST1_NETWORK_CONNECT_NETWORKS=(
"high-availability"
)
# ==============================================================================================
# ── MEDIA ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Media Permissions ━━━
# Shares that media_shares_permissions.sh applies PERMISSIONS_MODE and PERMISSIONS_OWNER to.
# Runs first in DAILY_MAINTENANCE_SCRIPTS — arr cleanup depends on correct ownership.
HOST1_MEDIA_PERMISSION_SHARES=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/appcache
/mnt/user/Books
/mnt/user/Downloads
/mnt/user/Games
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movie_Recordings
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Photo
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Tv_Recordings
/mnt/user/Tv_Shows
/mnt/user/YouTube
)
# ━━━ Media Cleaner ━━━
# Folder lists for media_cleaner.sh — two profiles: anime and media.
# File patterns shared across all servers — defined in master.conf.
# Called via DAILY_MAINTENANCE_SCRIPTS. Run manually: Media/media_cleaner.sh anime|media
HOST1_ANIME_CLEAN_FOLDERS=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
)
HOST1_MEDIA_CLEAN_FOLDERS=(
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Tv_Shows
)
# ==============================================================================================
# ── ARR STACK ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Used by arr cleanup scripts and arrs_failed_stalled_recovery.sh.
# detect_hosts() selects HOST1 vars when running on HOST1.
#
# PATH MAPS — container path → host path translation.
# Arr stores file paths using container-internal paths — scripts need host paths to scan.
# Add one entry per root folder in arr Settings → Media Management → Root Folders.
# ━━━ Downloaders ━━━
# Used by downloaders_reset.sh — runs every 30min via CRITICAL_MAINTENANCE_SCRIPTS.
# Clears stuck states, purges old history, prepares each client for a clean cycle.
# slskd — clears stuck searches, dead transfers, purges expired failed imports.
# SLSKD_FAILED_IMPORTS_DIR: where Soularr moves albums Lidarr rejected.
HOST1_SLSKD_URL="http://localhost:8980"
HOST1_SLSKD_API_KEY="4bF9kL2mNpQrT7vWxYz1A3dEgHjKoRsU"
HOST1_SLSKD_FAILED_IMPORTS_DIR="/mnt/user/Temp_Storage/Slskd/completed/failed_imports"
# SABnzbd
HOST1_SABNZBD_URL="http://localhost:8180"
HOST1_SABNZBD_API_KEY="8bfefe41d83b4d50883e32859b55ca9a"
# qBittorrent — deleteFiles=false removes torrent from qBit but leaves files on disk.
# Radarr/Sonarr manage actual files independently.
HOST1_QBIT_URL="http://localhost:8080"
HOST1_QBIT_USERNAME="root"
HOST1_QBIT_PASSWORD="Stay0utD!ck"
# ━━━ Lidarr — HOST1 only ━━━
# HOST2 does not run Lidarr — HOST1_LIDARR_RECOVERY flag handles the exit cleanly.
HOST1_LIDARR_URL="http://localhost:8686"
HOST1_LIDARR_API_KEY="b2977e71ef074bc0a0529d9fcce3b2dc"
HOST1_LIDARR_MUSIC_ROOT="/mnt/user/Music-New"
HOST1_FANART_API_KEY="Yd7147a43b692df0b364b94dc47efb81"
HOST1_LASTFM_API_KEY="be6dc169c33ae263e690c30d18b7491d"
declare -A HOST1_LIDARR_PATH_MAP=(
["/ext-music"]="/mnt/user/Music-New"
)
# ━━━ Sonarr ━━━
HOST1_SONARR_URL="http://localhost:8989"
HOST1_SONARR_API_KEY="130decd3db5b4c25afad64864cd03f9f"
HOST1_SONARR_TV_ROOT="/mnt/user/Tv_Shows"
# Note: stand-up_comedy in both Sonarr + Radarr — TV specials and movie specials, one folder
declare -A HOST1_SONARR_PATH_MAP=(
["/tv"]="/mnt/user/Tv_Shows"
["/ext-standup-comedy"]="/mnt/user/stand-up_comedy/series"
["/kids tv"]="/mnt/user/Kids_Tv_Shows"
["/ext-anime-shows"]="/mnt/user/Anime_Shows-Old"
)
# ━━━ Radarr ━━━
HOST1_RADARR_URL="http://localhost:7878"
HOST1_RADARR_API_KEY="d43a3ec6cf1549edb4af0cc63f98b2a9"
HOST1_TMDB_API_KEY="3dac5e2e49b5540472d2eafec4f01260"
HOST1_RADARR_MOVIES_ROOT="/mnt/user/Movies"
# Note: stand-up_comedy in both Radarr + Sonarr — movie specials and TV specials, one folder
declare -A HOST1_RADARR_PATH_MAP=(
["/movies"]="/mnt/user/Movies"
["/kids movies"]="/mnt/user/Kids_Movies"
["/ext-stand-up-comedy"]="/mnt/user/stand-up_comedy/specials"
["/anime-movies"]="/mnt/user/Anime_Movies-Old"
["/ext-anime-movies"]="/mnt/user/Anime_Movies-Old"
)
# ━━━ Arr Recovery Toggles ━━━
# false = skip that arr on this host — exits cleanly without error
HOST1_SONARR_RECOVERY=true
HOST1_RADARR_RECOVERY=true
HOST1_LIDARR_RECOVERY=true # HOST1 only — exits cleanly on HOST2
# ==============================================================================================
# ── TRANSCODES ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Ramdisk size ceiling — tmpfs only uses RAM actually needed, not the full size upfront.
# Real-world: 9 streams peaked at ~5.5GB — 10G gives generous headroom on 128GB RAM.
HOST1_RAMDISK_SIZE="10G"
# Usage thresholds — coupled to HOST1_RAMDISK_SIZE, adjust all three together if size changes.
# Hysteresis gap (8.5 - 7 = 1.5GB) prevents flip-flop between ramdisk and SSD.
HOST1_RAMDISK_WARN_GB=8.5 # flip to SSD when ramdisk usage reaches this
HOST1_RAMDISK_LOW_GB=7 # flip back to ramdisk when usage drops to this
# SSD fallback path — where transcodes land when ramdisk exceeds HOST1_RAMDISK_WARN_GB.
# Must be on cache pool — array disks too slow for active transcode writes.
HOST1_TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/"
# Media servers sharing the ramdisk transcode space on HOST1.
# Format: "ContainerName|URL|APIKey|Type" — Type: emby | jellyfin | plex
# Entries with placeholder API keys are skipped automatically.
# ⚠️ Tdarr does NOT belong here — keep Tdarr on SSD, not ramdisk.
HOST1_TRANSCODE_SERVERS=(
"${HOST1_EMBY_CONTAINER}|${HOST1_EMBY_URL}|${HOST1_EMBY_API_KEY}|emby"
"${HOST1_JELLYFIN_CONTAINER}|${HOST1_JELLYFIN_URL}|${HOST1_JELLYFIN_API_KEY}|jellyfin"
)
# ==============================================================================================
# ── MONITORS ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
# Domains checked via direct openssl connection — not relying on NPM's certificate state.
# Checks the actual certificate served, not what NPM thinks it has.
# Thresholds (CERT_WARN_DAYS, CERT_CRIT_DAYS) defined in master.conf.
HOST1_CERT_MONITOR_DOMAINS=(
"Gmer4Lfe.com"
"Gmer4Lfe.us"
)
# ━━━ SMART Health ━━━
# Drives skipped in SMART attribute monitoring — hardware is server-specific.
# Thresholds read from dynamix.cfg at runtime — fallbacks in master.conf.
HOST1_SMART_IGNORE_DRIVES=(
"sda" # boot USB — SMART not meaningful on flash drives
)
# ━━━ ZFS Report ━━━
# Pools excluded from the weekly ZFS health report — reduces noise from single-disk array pools.
# These are individual array disks formatted as ZFS — converting to XFS over time via unBalance.
# Pool health thresholds defined in master.conf.
HOST1_ZFS_REPORT_IGNORE_POOLS=(
"disk5"
"disk6"
"disk8"
"disk9"
"disk10"
)
# ==============================================================================================
# ── RESOURCE MANAGER ──────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Containers to manage under pressure — see master.conf RW_CRITICAL_CONTAINERS for exclusions.
# docker pause at medium pressure (RAM < RW_RAM_MEDIUM_GB or load > medium threshold)
# Suspended in-place — instant to pause/unpause, no state lost, no restart delay.
HOST1_RW_PAUSE_CONTAINERS=(
"Huntarr" # arr search automation — safe to suspend
"Cleanuparr" # download cleanup — safe to suspend
"Healarr" # arr health checks — safe to suspend
"Soularr" # Slskd automation — background only
"ChannelTube" # YouTube archiver — background only
"Pinchflat" # YouTube archiver — background only
)
# docker stop at hard pressure (RAM < RW_RAM_HARD_GB)
# Full stop — these are optional/heavy services that free significant RAM when stopped.
# resource_watchdog.sh restarts them when pressure fully clears (RAM >= RW_RAM_RECOVER_GB).
HOST1_RW_STOP_CONTAINERS=(
"LocalAI" # GPU/CPU heavy — largest RAM consumer when idle
"7DaysToDie" # game server — optional
"V-Rising" # game server — optional
"Code-Server" # IDE — not needed during pressure events
)
# ==============================================================================================
# ── SYSTEM WATCHDOG ───────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Per-host check toggles and NIC config for system_watchdog.sh.
# Aliased by detect_hosts() — script uses unprefixed SYS_WATCHDOG_* names.
# HOST1: TR1950X 128GB — full media server, active transcoding, ZFS cache pools.
#
# Three-tier response — all critical checks enabled by default on HOST1:
# Tier 1 (bypass strikes, reboot now): docker daemon, rootfs full, kernel oops, FD, /boot
# Tier 2 (bypass strikes with OOM): RAM critical + OOM kills in cycle
# Tier 3 (standard strike system): everything else
#
# RAM tiers, OOM limits, and reboot loop settings in master.conf System Watchdog section.
# ━━━ Primary NIC ━━━
# Network interface for NIC state check — verify with: ip link show | grep "^[0-9]"
# Common values: eth0, bond0, br0, eno1
HOST1_SYS_WATCHDOG_NIC="eth0"
# ━━━ Tier 1 — Critical Checks ━━━
# These bypass the strike system — a single hit triggers immediate reboot.
# Disabling any of these is not recommended — they protect against acute system failure.
# Docker daemon unresponsive → try restart, reboot if restart fails.
# Without a working daemon docker_watchdog.sh is blind and containers cannot be managed.
HOST1_SYS_WATCHDOG_CHECK_DOCKER_DAEMON=true
# rootfs at critical threshold (ROOTFS_CRITICAL_PCT=99) → reboot immediately.
# At 99% rootfs writes fail silently — logs stop, Docker errors out, SSH may stop working.
# Standard 95% threshold still uses strike system — only 99%+ is critical tier.
HOST1_SYS_WATCHDOG_CHECK_ROOTFS=true
# Kernel BUG/Oops in dmesg delta since last cycle → reboot immediately.
# A kernel oops means the kernel ran with a corrupted state — stability is not guaranteed.
HOST1_SYS_WATCHDOG_CHECK_KERNEL_OOPS=true
# File descriptor exhaustion at FD_CRITICAL_PCT (95%) → reboot immediately.
# At 95% FD: new connections fail, Docker can't spawn processes, SSH drops.
HOST1_SYS_WATCHDOG_CHECK_FD=true
# /boot read-only detected → reboot immediately.
# Unexpected read-only /boot means state files and config writes are silently failing.
# Fallback state, watchdog reboot log, and lock files all go stale silently.
HOST1_SYS_WATCHDOG_CHECK_BOOT=true
# ━━━ Tier 2 — Urgent OOM Check ━━━
# Bypass strikes when RAM is critically low AND OOM kill rate confirms active crisis.
# Both must be enabled for Tier 2 bypass to function — disable either to always use strikes.
# Track kernel OOM kills each cycle via /proc/vmstat oom_kill delta.
# Also provides diagnostic context in reboot messages (which processes were killed).
HOST1_SYS_WATCHDOG_CHECK_OOM=true
# Free RAM check — required for both Tier 2 bypass and RAM tier logic.
# Tiers: MEM_WARN_GB(10) → notify | MEM_SHUTDOWN_GB(6) → stop containers | MEM_GB(4) → strikes
HOST1_SYS_WATCHDOG_CHECK_RAM=true
# ━━━ Tier 3 — Standard Checks (strike system) ━━━
# Each check must fail SYS_WATCHDOG_STRIKE_LIMIT consecutive cycles before action is taken.
# Single spikes are ignored — sustained problems trigger reboot.
# /var/log filesystem usage above SYS_WATCHDOG_LOG_PCT.
# Log spam (Docker log storms, syslog loops) fills rootfs — indicates something broken.
HOST1_SYS_WATCHDOG_CHECK_LOG=true
# ZFS ARC memory pinned above SYS_WATCHDOG_ARC_PINNED_PCT after cache drop.
# Enabled on HOST1 — ZFS cache pools actively used. Disable on hosts without ZFS.
HOST1_SYS_WATCHDOG_CHECK_ARC=true
# CPU temperature above SYS_WATCHDOG_CPU_TEMP_MAX (95°C).
# Sustained high temp causes kernel throttling or panic. Requires lm-sensors.
HOST1_SYS_WATCHDOG_CHECK_CPU_TEMP=true
# Load average above SYS_WATCHDOG_LOAD_MULTIPLIER × core count.
# DISABLED on HOST1 — Tdarr and Emby cause legitimate sustained load spikes during encoding.
# Enable on idle servers or adjust SYS_WATCHDOG_LOAD_MULTIPLIER if load is always high.
HOST1_SYS_WATCHDOG_CHECK_LOAD=false
# Zombie process count above SYS_WATCHDOG_ZOMBIE_LIMIT (50).
# Large zombie counts indicate serious process management failure — something is stuck.
HOST1_SYS_WATCHDOG_CHECK_ZOMBIES=true
# Check docker_watchdog.sh persistent skip list — required containers on skip list.
# Cross-watchdog coordination: if docker_watchdog gave up, system_watchdog escalates.
# ENABLED — HOST1 fully built and operational, skip list is meaningful.
HOST1_SYS_WATCHDOG_CHECK_CONTAINERS=true
# /tmp filesystem usage above SYS_WATCHDOG_TMP_PCT with auto-clear attempt.
# Script tries to clear aged /tmp files first — only strikes if clear fails.
# Lock files, rsync temp files, and Docker ops use /tmp — 100% means lock failures.
HOST1_SYS_WATCHDOG_CHECK_TMP=true
# Array disk error count delta in /proc/mdstat — accumulating errors = disk failing now.
# Triggers on SYS_WATCHDOG_MDSTAT_ERROR_LIMIT new errors in one cycle.
HOST1_SYS_WATCHDOG_CHECK_MDSTAT=true
# Primary NIC operstate — detects NIC going down (physical or driver failure).
# Uses HOST1_SYS_WATCHDOG_NIC above. Strike system — brief flaps don't trigger reboot.
HOST1_SYS_WATCHDOG_CHECK_NETWORK=true
# sshd running check — attempts restart before escalating.
# sshd down = no remote access. Script tries rc.sshd start, notifies, strikes on failure.
HOST1_SYS_WATCHDOG_CHECK_SSHD=true
# Runaway process detection — single process above SYS_WATCHDOG_RUNAWAY_CPU_PCT sustained.
# DISABLED — Tdarr encoding and Emby transcoding legitimately peg CPU for extended periods.
# Enable only if HOST1 has no CPU-intensive workloads.
HOST1_SYS_WATCHDOG_CHECK_RUNAWAY=false
# ==============================================================================================
# ── AUTH STACK ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Credentials for the Varaverk Auth Stack page (NPM, lldap, Authelia).
# HOST1 is the auth source of truth — these are the live production credentials.
# ━━━ NginxProxyManager ━━━
# Admin API runs on 7818 (not 81 — 81 is the partnership WebUI port).
HOST1_NPM_URL="http://localhost:7818"
HOST1_NPM_USER="" # NPM admin email
HOST1_NPM_PASS="" # NPM admin password
# ━━━ lldap ━━━
HOST1_LLDAP_URL="http://localhost:17170"
HOST1_LLDAP_USER="admin" # lldap admin username
HOST1_LLDAP_PASS="" # lldap admin password
# ━━━ Authelia ━━━
HOST1_AUTHELIA_CONFIG="/mnt/user/appdata/Authelia/configuration.yml"
HOST1_AUTHELIA_CONTAINER="Authelia"
# ==============================================================================================
# ──────────────────────── End Of HOST1 Variables ──────────────────────────────────────────────
# ==============================================================================================
@@ -1,633 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ========================== HOSTN CONFIGURATION — (hostname) ==================================
# ==============================================================================================
# HOSTN-specific variables — credentials, container names, share paths, failover lists.
# Sourced after master.conf — values here extend shared profile arrays and add HOSTN-specific
# identity, credentials, and container configuration.
#
# Sparse checkout (git) ensures other hosts never receive this file.
#
# DO NOT put shared config here — thresholds, toggles, profiles belong in master.conf.
# DO NOT put other hosts' variables here — they belong in their own host*.conf files.
#
# ── HOW TO USE THIS TEMPLATE ──────────────────────────────────────────────────────────────────
# This file was generated by the Varaverk first-run wizard.
# Fill in the sections that apply to your setup — leave unused sections empty.
# All scripts self-guard against empty values — safe to leave sections blank until needed.
#
# ── INDEX ─────────────────────────────────────────────────────────────────────────────────────
#
# ── IDENTITY & CONNECTIVITY ────────────────────────────────────────────────────────────────
# IDENTITY hostname, SSH key, Unraid API key
# EMBY container name, URL, API key
# JELLYFIN container name, URL, API key
# GITEA API token for SSH key registration
# NOTIFICATIONS Discord webhook
#
# ── PARTNERSHIP ────────────────────────────────────────────────────────────────────────────
# PARTNERSHIP auth containers, backup paths, emby provisioning
#
# ── RSYNC ──────────────────────────────────────────────────────────────────────────────────
# DAILY SYNC SHARES media shares this host owns and pushes
# PERSONAL SHARES private encrypted shares for offsite backup
# WEEKLY SYNC SHARES appdata shares synced weekly
# INTERMEDIATE SYNC mid-day appdata propagation
# CRITICAL SYNC SHARES appdata shares synced every 30 minutes
# BACKUP VERIFY shares for checksum verification against remote
# HOSTN RSYNC PROFILE host-specific appdata sync profile
#
# ── FALLBACK ───────────────────────────────────────────────────────────────────────────────
# DDNS DDNS containers managed by this host
# INTERNET LOSS containers stopped when internet is lost
# FALLBACK TIERS what this host runs for the remote per tier
# TIER DELAYS delays before each tier activates
# RSYNC WRITEBACK appdata synced back on handback
#
# ── DOCKER ─────────────────────────────────────────────────────────────────────────────────
# DOCKER DAILY RESTART containers restarted daily
# DOCKER WEEKLY RESTART containers restarted weekly
# DOCKER WATCHDOG memory limits, health URLs, required containers
# NETWORK WATCHDOG DDNS domain, NPM URL for connectivity checks
# DOCKER NETWORK CONNECT networks and containers for array start
#
# ── MEDIA ──────────────────────────────────────────────────────────────────────────────────
# MEDIA PERMISSIONS share list for permissions script
# MEDIA CLEANER folder lists for media_cleaner.sh
#
# ── ARR STACK ──────────────────────────────────────────────────────────────────────────────
# DOWNLOADERS slskd, SABnzbd, qBittorrent credentials and URLs
# LIDARR / SONARR / RADARR URL, API key, path map
# ARR RECOVERY per-arr recovery toggles
#
# ── TRANSCODES ─────────────────────────────────────────────────────────────────────────────
# TRANSCODES ramdisk size, thresholds, SSD path, server array
#
# ── MONITORS ───────────────────────────────────────────────────────────────────────────────
# CERTIFICATE MONITOR domains checked for SSL expiry
# SMART HEALTH drives to skip in SMART monitoring
# ZFS REPORT pools to exclude from ZFS health report
#
# ── RESOURCE MANAGER ───────────────────────────────────────────────────────────────────────
# RESOURCE MANAGER containers paused/stopped under memory pressure
#
# ── SYSTEM WATCHDOG ────────────────────────────────────────────────────────────────────────
# SYSTEM WATCHDOG per-host check toggles and NIC configuration
#
# ==============================================================================================
# ==============================================================================================
# ── STORAGE MODE ──────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Storage mode ━━━
# Controls where Varaverk stores scripts, conf, and state files.
# true = internal NVMe/SSD — /boot/config/plugins/varaverk (write-safe, git-direct)
# false = USB flash boot — /mnt/user/appdata/Varaverk (preserves flash lifetime)
# Auto-detected from boot device transport on first setup.
# To change: Settings → Storage → Migrate.
HOST1_STORAGE_MODE_INTERNAL=true
# ==============================================================================================
# ── IDENTITY & CONNECTIVITY ───────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Identity ━━━
# HOSTN hostname lives in master.conf (not a credential — safe for all servers).
# SSH key used for all server-to-server operations — rsync, failover, conf sync.
# Convention: /root/.ssh/<hostname-lowercase-no-unraid-prefix>_rsync_automation
# Must be in /root/.ssh/ and authorised in the partner's /root/.ssh/authorized_keys.
# Run Partnership/ssh_setup.sh to generate the key and copy it to the partner.
HOST1_SSH_KEY="/root/.ssh/gmer4lfe_rsync_automation"
HOST1_OWNER="gmer4lfe"
HOST1_OWNER_EMAIL="gmer4lfe@gmail.com"
# ━━━ Unraid API ━━━
# Used by the Varaverk plugin to query this server's Unraid GraphQL API.
# Generate in Unraid: Settings → Management Access → API Keys → + New Key
HOST1_UNRAID_API_KEY="1825c3a2e03ea5089974f4da2e171aa2d5907a1dea23cc479c33e492c8ff4dbb"
# ━━━ Emby ━━━
HOST1_EMBY_CONTAINER="Emby"
HOST1_EMBY_URL="http://localhost:8096"
HOST1_EMBY_API_KEY="0c27448d93a7431f9ac63569f7655829"
# ━━━ Jellyfin ━━━
HOST1_JELLYFIN_CONTAINER="Jellyfin"
HOST1_JELLYFIN_URL="http://localhost:8095"
HOST1_JELLYFIN_API_KEY="4e820e7df74c4933acec212b1996314e"
# ━━━ Gitea ━━━
# Personal access token for gitea_ssh_setup.sh.
# Create in Gitea: Settings → Applications → Generate Token → scope: write:user
HOST1_GITEA_API_TOKEN=""
# ━━━ Notifications ━━━
# Discord webhook — leave blank to disable.
HOST1_DISCORD_WEBHOOK=""
# ==============================================================================================
# ── PARTNERSHIP ───────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Auth containers reconfigured on onboard/offboard.
# Format: "ContainerName|WebUIPort"
HOST1_PARTNERSHIP_AUTH_WEBUIS=(
"NginxProxyManager|81"
"Lldap-Gmer4Lfe|17170"
"Authelia|9091"
"Authelia-Secondary|9092"
)
# XML templates pushed to mirror during onboard — auth stack.
# Dependencies (databases) must come before apps that depend on them.
HOST1_PARTNERSHIP_AUTH_STACK=(
# Dependencies first — Mariadb/Redis must be healthy before Authelia starts
"my-Mariadb-Authelia.xml"
"my-Mariadb-Authelia-Secondary.xml"
"my-Redis-Authelia.xml"
"my-Redis-Authelia-Secondary.xml"
# Auth apps — deployed after their deps are confirmed healthy
"my-Authelia.xml"
"my-Authelia-Secondary.xml"
"my-NginxProxyManager.xml"
"my-Lldap-Gmer4Lfe.xml"
)
# XML templates pushed to mirror during onboard — arr stack.
HOST1_PARTNERSHIP_ARR_STACK=(
"my-Sonarr.xml"
"my-Radarr.xml"
"my-Lidarr.xml"
"my-Prowlarr.xml"
"my-Bazarr.xml"
)
# Paths the partner should collect during the grace window after offboard.
HOST1_PARTNERSHIP_MIRROR_BACKUPS=(
# "/mnt/user/appdata-Fallback/Jayred365-Emby"
)
# Containers parked on this server when partnership is active.
HOST1_PARTNERSHIP_OWN_CONTAINERS=(
# "Emby"
# "NginxProxyManager"
)
# Containers stopped on THIS server before deploying the mirror's stack on onboard.
HOST1_PARTNERSHIP_REPLACE_CONTAINERS=(
)
# Arr containers stopped on this server when mirror's arr stack is deployed.
HOST1_PARTNERSHIP_ARR_REPLACE_CONTAINERS=(
)
# Emby admin provisioning — owner controls whether Emby is shared.
HOST1_PARTNERSHIP_PROVISION_EMBY_ADMIN=false # owner controls whether Emby is shared
HOST1_PARTNERSHIP_EMBY_PORT=8096
HOST1_PARTNERSHIP_EMBY_ADMIN_USER="" # this server's desired Emby username
HOST1_PARTNERSHIP_EMBY_ADMIN_PASS="" # this server's desired Emby password
# ==============================================================================================
# ── RSYNC ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Daily Sync Shares ━━━
# Media shares this host pushes to all other nodes every night.
# Uses DEFAULT_RSYNC_OPTS from master.conf — no profile needed.
HOST1_DAILY_SYNC_SHARES=(
/mnt/user/Books
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Nextcloud
/mnt/user/stand-up_comedy
/mnt/user/Sports
# /mnt/user/Tv_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Movies
/mnt/user/Anime_Shows
)
# ━━━ Personal Shares ━━━
# Private encrypted shares synced for offsite backup, independent of media shares.
HOST1_PERSONAL_SHARES=(
# /mnt/user/HOST1-Personal # uncomment after creating encrypted dataset
)
# ━━━ Weekly Sync Shares ━━━
# Appdata shares synced during the weekly maintenance window.
# Profiles (emby, critical-data) drive container stops — define in master.conf.
HOST1_WEEKLY_SYNC_SHARES=(
"/mnt/user/Media_Server/Emby" # emby profile — full clean mirror
"/mnt/user/appdata-Fallback/Critical-Data" # critical-data profile — auth stack
)
# ━━━ Intermediate Sync Shares ━━━
# Shares synced every 4 hours. Leave empty to skip mid-day rsync.
HOST1_INTERMEDIATE_SYNC_SHARES=(
# Add shares here to enable mid-day rsync
# Example: "/mnt/user/Emby_Metadata"
)
# ━━━ Critical Sync Shares ━━━
# Appdata shares synced every 30 minutes.
# Format: "/path/to/share" or "/path/to/share|profile-name"
HOST1_CRITICAL_SYNC_SHARES=(
"/mnt/user/appdata-Fallback/Critical-Data|critical-fallback" # auth dirty sync — stays running
"/mnt/user/Media_Server/Emby|emby-fallback" # Emby dirty sync — stays running
)
# ━━━ Backup Verify ━━━
# Leave empty to use HOST1_DAILY_SYNC_SHARES automatically.
HOST1_BACKUP_VERIFY_SHARES=(
# leave empty to use HOST1_DAILY_SYNC_SHARES automatically
)
# ━━━ HOSTN Rsync Profile — hostn-appdata ━━━
# Host-specific appdata sync profile.
PROFILE_RSYNC_OPTS[hostn-appdata]="-av --info=progress2 --bwlimit=${PROFILE_BW_LIMIT[hostn-appdata]:-8000}"
PROFILE_BW_LIMIT[hostn-appdata]=8000
PROFILE_RETRY_COUNT[hostn-appdata]=3
PROFILE_SLEEP[hostn-appdata]=300
PROFILE_CRITICAL_CONTAINER_NAMES[hostn-appdata]=""
PROFILE_DELAYED_CONTAINERS[hostn-appdata]=""
PROFILE_CONTAINER_DELAY[hostn-appdata]=5
PROFILE_EXCLUDE_DIRS[hostn-appdata]="logs *.tmp"
# ==============================================================================================
# ── FALLBACK ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ DDNS ━━━
# DDNS containers this host manages.
HOST1_DDNS_CONTAINERS=(
"Gmer4Lfe.com"
)
# ━━━ Internet Loss ━━━
# Containers stopped immediately when internet is lost.
FALLBACK_HOST1_STOP_ON_NO_NET=(
"Gmer4Lfe.com"
)
# ━━━ Fallback Tiers — HOSTN Runs for Partner ━━━
# Containers this host starts when the partner goes down.
# Replace HOST2 below with the actual remote host ID (HOST1, HOST2, etc.)
FALLBACK_HOST1_COVERS_HOST2_TIER1=(
"Gmer4Lfe.us"
"VaultWarden-Jayred365"
)
FALLBACK_HOST1_COVERS_HOST2_TIER2=(
# "container-placeholder"
)
FALLBACK_HOST1_COVERS_HOST2_TIER3=(
# "container-placeholder"
)
FALLBACK_HOST1_COVERS_HOST2_TIER4=(
# "container-placeholder"
)
# ━━━ Tier Delays — This Host's Outage Timers ━━━
# How long THIS host must be down before each tier activates on the partner.
HOST1_TIER2_DELAY=240 # 4 hours — NextCloud, Immich
HOST1_TIER3_DELAY=720 # 12 hours — secondary services
HOST1_TIER4_DELAY=1440 # 24 hours — arrs + downloaders
# ━━━ Rsync Writeback ━━━
HOST1_TIER1_WRITEBACK_DELAY=60 # skip Emby writeback if outage under 1hr
FALLBACK_HOST1_WRITEBACK_TIER1=(
"/mnt/user/Media_Server/Emby" # watch states built up during outage
)
FALLBACK_HOST1_WRITEBACK_TIER2=(
"/mnt/user/appdata-Fallback/Important-Data" # NextCloud + Postgres — files added during outage
)
FALLBACK_HOST1_WRITEBACK_TIER3=(
# "location-placeholder"
)
FALLBACK_HOST1_WRITEBACK_TIER4=(
"/mnt/user/appdata-Fallback/Arrs_Stack" # arr databases — downloads queued during outage
)
# ==============================================================================================
# ── DOCKER ────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Docker Daily Restart ━━━
HOST1_DAILY_RESTART_CONTAINERS=(
"NginxProxyManager"
"Lldap-Gmer4Lfe"
"Authelia"
"Authelia-Secondary"
"Dispatcharr-Iptv-Users"
"Dispatcharr" # Live TV scheduler — degrades without daily restart
"Dispatcharr-Basic"
"ErsatzTV-Emby"
"slskd" # Soulseek connection drops after extended uptime; restart refreshes share index
)
# ━━━ Docker Weekly Restart ━━━
HOST1_WEEKLY_RESTART_CONTAINERS=(
"NextCloud"
"Organizrv2-Gmer4Lfe"
"AdGuard-Home"
"Immich-Gmer4Lfe"
)
# ━━━ Docker Watchdog ━━━
# Memory hard limits in MB — immediate restart if exceeded.
# 20GB=20480 16GB=16384 8GB=8192 4GB=4096 2GB=2048 1GB=1024
declare -A HOST1_WATCHDOG_CONTAINERS=(
["Emby"]=20480 # 20GB — large library + active transcodes
["LidaTube"]=6144 # 6GB — memory leak over time
["Tdarr"]=6144 # 6GB — encoding is memory intensive
["Code-Server"]=1024 # 1GB — should never need more
)
# HTTP health check URLs — checked every cycle.
declare -A HOST1_WATCHDOG_CONTAINER_URLS=(
["Emby"]="http://localhost:8096"
["NginxProxyManager"]="http://localhost:7818"
["Authelia"]="http://localhost:9091/api/health"
["Authelia-Secondary"]="http://localhost:9092/api/health"
["Lldap-Gmer4Lfe"]="http://localhost:17170"
)
# API-level health checks. Format: ["ContainerName"]="url|expected_json_key|expected_value"
declare -A HOST1_WATCHDOG_CONTAINER_API_CHECKS=(
)
# Required containers — must always be running.
HOST1_WATCHDOG_REQUIRED_CONTAINERS=(
"NginxProxyManager"
"Lldap-Gmer4Lfe"
"Mariadb-Authelia"
"Mariadb-Authelia-Secondary"
"Redis-Authelia"
"Redis-Authelia-Secondary"
"Authelia"
"Authelia-Secondary"
)
# Containers to skip in Tier 2 global scan.
HOST1_WATCHDOG_SCAN_IGNORE=(
"DashGate"
"PIA-WG-Config-Generator"
"Aperture"
"Aperture-Kids"
"pgvector-18-Apeture-Kids"
"Pgvector18-Aperture"
"emby-test" # broken test container (exit 127 — bad image)
)
# Dependency ordering — skip restarting a container if its dependency is also down.
declare -A HOST1_WATCHDOG_DEPENDENCIES=(
["Authelia"]="Mariadb-Authelia Redis-Authelia"
["Authelia-Secondary"]="Mariadb-Authelia Redis-Authelia-Secondary"
["NextCloud"]="Postgres-NextCloud"
)
# Per-container appdata growth suppress ceilings in MB.
declare -A HOST1_WATCHDOG_APPDATA_SIZES=(
["Tdarr"]="25600" # 25GB — transcode cache grows legitimately during active jobs
["7dtd"]="20480" # 20GB — game server world data, expected to be large
)
# ━━━ Network Watchdog ━━━
HOST1_NETWORK_WATCHDOG_DDNS_DOMAIN="gmer4lfe.com"
HOST1_NETWORK_WATCHDOG_DDNS_CONTAINER="Gmer4Lfe.com"
HOST1_NETWORK_WATCHDOG_NPM_URL="https://gmer4lfe.com"
# ━━━ Docker Network Connect ━━━
HOST1_NETWORK_CONNECT_CONTAINERS=(
"memcached"
"Npm-CrowdSec"
)
HOST1_NETWORK_CONNECT_NETWORKS=(
"high-availability"
)
# ==============================================================================================
# ── MEDIA ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Media Permissions ━━━
HOST1_MEDIA_PERMISSION_SHARES=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/appcache
/mnt/user/Books
/mnt/user/Downloads
/mnt/user/Games
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movie_Recordings
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Photo
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Tv_Recordings
/mnt/user/Tv_Shows
/mnt/user/YouTube
)
# ━━━ Media Cleaner ━━━
HOST1_ANIME_CLEAN_FOLDERS=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
)
HOST1_MEDIA_CLEAN_FOLDERS=(
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Tv_Shows
)
# ==============================================================================================
# ── ARR STACK ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Downloaders ━━━
HOST1_SLSKD_URL="http://localhost:8980"
HOST1_SLSKD_API_KEY="4bF9kL2mNpQrT7vWxYz1A3dEgHjKoRsU"
HOST1_SLSKD_FAILED_IMPORTS_DIR="/mnt/user/Temp_Storage/Slskd/completed/failed_imports"
HOST1_SABNZBD_URL="http://localhost:8180"
HOST1_SABNZBD_API_KEY="8bfefe41d83b4d50883e32859b55ca9a"
HOST1_QBIT_URL="http://localhost:8080"
HOST1_QBIT_USERNAME="root"
HOST1_QBIT_PASSWORD="Stay0utD!ck"
# ━━━ Lidarr ━━━
HOST1_LIDARR_URL="http://localhost:8686"
HOST1_LIDARR_API_KEY="b2977e71ef074bc0a0529d9fcce3b2dc"
HOST1_LIDARR_MUSIC_ROOT="/mnt/user/Music-New"
HOST1_FANART_API_KEY="Yd7147a43b692df0b364b94dc47efb81"
HOST1_LASTFM_API_KEY="be6dc169c33ae263e690c30d18b7491d"
declare -A HOST1_LIDARR_PATH_MAP=(
["/ext-music"]="/mnt/user/Music-New"
)
# ━━━ Sonarr ━━━
HOST1_SONARR_URL="http://localhost:8989"
HOST1_SONARR_API_KEY="130decd3db5b4c25afad64864cd03f9f"
HOST1_SONARR_TV_ROOT="/mnt/user/Tv_Shows"
declare -A HOST1_SONARR_PATH_MAP=(
["/tv"]="/mnt/user/Tv_Shows"
["/ext-standup-comedy"]="/mnt/user/stand-up_comedy/series"
["/kids tv"]="/mnt/user/Kids_Tv_Shows"
["/ext-anime-shows"]="/mnt/user/Anime_Shows-Old"
)
# ━━━ Radarr ━━━
HOST1_RADARR_URL="http://localhost:7878"
HOST1_RADARR_API_KEY="d43a3ec6cf1549edb4af0cc63f98b2a9"
HOST1_TMDB_API_KEY="3dac5e2e49b5540472d2eafec4f01260"
HOST1_RADARR_MOVIES_ROOT="/mnt/user/Movies"
declare -A HOST1_RADARR_PATH_MAP=(
["/movies"]="/mnt/user/Movies"
["/kids movies"]="/mnt/user/Kids_Movies"
["/ext-stand-up-comedy"]="/mnt/user/stand-up_comedy/specials"
["/anime-movies"]="/mnt/user/Anime_Movies-Old"
["/ext-anime-movies"]="/mnt/user/Anime_Movies-Old"
)
# ━━━ Arr Recovery Toggles ━━━
HOST1_LIDARR_RECOVERY=true # HOST1 only — exits cleanly on HOST2
HOST1_SONARR_RECOVERY=true
HOST1_RADARR_RECOVERY=true
# ==============================================================================================
# ── TRANSCODES ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOST1_RAMDISK_SIZE="10G"
HOST1_RAMDISK_WARN_GB=8.5 # flip to SSD when ramdisk usage reaches this
HOST1_RAMDISK_LOW_GB=7 # flip back to ramdisk when usage drops to this
HOST1_TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/"
HOST1_TRANSCODE_SERVERS=(
"${HOST1_EMBY_CONTAINER}|${HOST1_EMBY_URL}|${HOST1_EMBY_API_KEY}|emby"
"${HOST1_JELLYFIN_CONTAINER}|${HOST1_JELLYFIN_URL}|${HOST1_JELLYFIN_API_KEY}|jellyfin"
)
# ==============================================================================================
# ── MONITORS ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
HOST1_CERT_MONITOR_DOMAINS=(
"Gmer4Lfe.com"
"Gmer4Lfe.us"
)
# ━━━ SMART Health ━━━
HOST1_SMART_IGNORE_DRIVES=(
"sda" # boot USB — SMART not meaningful on flash drives
)
# ━━━ ZFS Report ━━━
HOST1_ZFS_REPORT_IGNORE_POOLS=(
"disk5"
"disk6"
"disk8"
"disk9"
"disk10"
)
# ==============================================================================================
# ── RESOURCE MANAGER ──────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOST1_RW_PAUSE_CONTAINERS=(
"Huntarr" # arr search automation — safe to suspend
"Cleanuparr" # download cleanup — safe to suspend
"Healarr" # arr health checks — safe to suspend
"Soularr" # Slskd automation — background only
"ChannelTube" # YouTube archiver — background only
"Pinchflat" # YouTube archiver — background only
)
HOST1_RW_STOP_CONTAINERS=(
"LocalAI" # GPU/CPU heavy — largest RAM consumer when idle
"7DaysToDie" # game server — optional
"V-Rising" # game server — optional
"Code-Server" # IDE — not needed during pressure events
)
# ==============================================================================================
# ── SYSTEM WATCHDOG ───────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOST1_SYS_WATCHDOG_NIC="eth0"
HOST1_SYS_WATCHDOG_CHECK_DOCKER_DAEMON=true
HOST1_SYS_WATCHDOG_CHECK_ROOTFS=true
HOST1_SYS_WATCHDOG_CHECK_KERNEL_OOPS=true
HOST1_SYS_WATCHDOG_CHECK_FD=true
HOST1_SYS_WATCHDOG_CHECK_BOOT=true
HOST1_SYS_WATCHDOG_CHECK_OOM=true
HOST1_SYS_WATCHDOG_CHECK_RAM=true
HOST1_SYS_WATCHDOG_CHECK_LOG=true
HOST1_SYS_WATCHDOG_CHECK_ARC=true
HOST1_SYS_WATCHDOG_CHECK_CPU_TEMP=true
HOST1_SYS_WATCHDOG_CHECK_LOAD=false
HOST1_SYS_WATCHDOG_CHECK_ZOMBIES=true
HOST1_SYS_WATCHDOG_CHECK_CONTAINERS=true
HOST1_SYS_WATCHDOG_CHECK_TMP=true
HOST1_SYS_WATCHDOG_CHECK_MDSTAT=true
HOST1_SYS_WATCHDOG_CHECK_NETWORK=true
HOST1_SYS_WATCHDOG_CHECK_SSHD=true
HOST1_SYS_WATCHDOG_CHECK_RUNAWAY=false
# ==============================================================================================
# ── AUTH STACK ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Credentials for the Varaverk Auth Stack page (NPM, lldap, Authelia).
# ━━━ NginxProxyManager ━━━
# Admin API runs on 7818 (not 81 — 81 is the partnership WebUI port).
HOST1_NPM_URL="http://localhost:7818"
HOST1_NPM_USER="" # NPM admin email
HOST1_NPM_PASS="" # NPM admin password
# ━━━ lldap ━━━
HOST1_LLDAP_URL="http://localhost:17170"
HOST1_LLDAP_USER="admin" # lldap admin username
HOST1_LLDAP_PASS="" # lldap admin password
# ━━━ Authelia ━━━
HOST1_AUTHELIA_CONFIG="/mnt/user/appdata/Authelia/configuration.yml"
HOST1_AUTHELIA_CONTAINER="Authelia"
# ==============================================================================================
# ──────────────────────── End Of HOSTn Variables ──────────────────────────────────────────────
# ==============================================================================================
@@ -1,634 +0,0 @@
#!/bin/bash
# ==============================================================================================
# ========================== HOSTN CONFIGURATION — (hostname) ==================================
# ==============================================================================================
# HOSTN-specific variables — credentials, container names, share paths, failover lists.
# Sourced after master.conf — values here extend shared profile arrays and add HOSTN-specific
# identity, credentials, and container configuration.
#
# Sparse checkout (git) ensures other hosts never receive this file.
#
# DO NOT put shared config here — thresholds, toggles, profiles belong in master.conf.
# DO NOT put other hosts' variables here — they belong in their own host*.conf files.
#
# ── HOW TO USE THIS TEMPLATE ──────────────────────────────────────────────────────────────────
# This file was generated by the Varaverk first-run wizard.
# Fill in the sections that apply to your setup — leave unused sections empty.
# All scripts self-guard against empty values — safe to leave sections blank until needed.
#
# ── INDEX ─────────────────────────────────────────────────────────────────────────────────────
#
# ── IDENTITY & CONNECTIVITY ────────────────────────────────────────────────────────────────
# IDENTITY hostname, SSH key, Unraid API key
# EMBY container name, URL, API key
# JELLYFIN container name, URL, API key
# GITEA API token for SSH key registration
# NOTIFICATIONS Discord webhook
#
# ── PARTNERSHIP ────────────────────────────────────────────────────────────────────────────
# PARTNERSHIP auth containers, backup paths, emby provisioning
#
# ── RSYNC ──────────────────────────────────────────────────────────────────────────────────
# DAILY SYNC SHARES media shares this host owns and pushes
# PERSONAL SHARES private encrypted shares for offsite backup
# WEEKLY SYNC SHARES appdata shares synced weekly
# INTERMEDIATE SYNC mid-day appdata propagation
# CRITICAL SYNC SHARES appdata shares synced every 30 minutes
# BACKUP VERIFY shares for checksum verification against remote
# HOSTN RSYNC PROFILE host-specific appdata sync profile
#
# ── FALLBACK ───────────────────────────────────────────────────────────────────────────────
# DDNS DDNS containers managed by this host
# INTERNET LOSS containers stopped when internet is lost
# FALLBACK TIERS what this host runs for the remote per tier
# TIER DELAYS delays before each tier activates
# RSYNC WRITEBACK appdata synced back on handback
#
# ── DOCKER ─────────────────────────────────────────────────────────────────────────────────
# DOCKER DAILY RESTART containers restarted daily
# DOCKER WEEKLY RESTART containers restarted weekly
# DOCKER WATCHDOG memory limits, health URLs, required containers
# NETWORK WATCHDOG DDNS domain, NPM URL for connectivity checks
# DOCKER NETWORK CONNECT networks and containers for array start
#
# ── MEDIA ──────────────────────────────────────────────────────────────────────────────────
# MEDIA PERMISSIONS share list for permissions script
# MEDIA CLEANER folder lists for media_cleaner.sh
#
# ── ARR STACK ──────────────────────────────────────────────────────────────────────────────
# DOWNLOADERS slskd, SABnzbd, qBittorrent credentials and URLs
# LIDARR / SONARR / RADARR URL, API key, path map
# ARR RECOVERY per-arr recovery toggles
#
# ── TRANSCODES ─────────────────────────────────────────────────────────────────────────────
# TRANSCODES ramdisk size, thresholds, SSD path, server array
#
# ── MONITORS ───────────────────────────────────────────────────────────────────────────────
# CERTIFICATE MONITOR domains checked for SSL expiry
# SMART HEALTH drives to skip in SMART monitoring
# ZFS REPORT pools to exclude from ZFS health report
#
# ── RESOURCE MANAGER ───────────────────────────────────────────────────────────────────────
# RESOURCE MANAGER containers paused/stopped under memory pressure
#
# ── SYSTEM WATCHDOG ────────────────────────────────────────────────────────────────────────
# SYSTEM WATCHDOG per-host check toggles and NIC configuration
#
# ==============================================================================================
# ==============================================================================================
# ── STORAGE MODE ──────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Storage mode ━━━
# Controls where Varaverk stores scripts, conf, and state files.
# true = internal NVMe/SSD — /boot/config/plugins/varaverk (write-safe, git-direct)
# false = USB flash boot — /mnt/user/appdata/Varaverk (preserves flash lifetime)
# Auto-detected from boot device transport on first setup.
# To change: Settings → Storage → Migrate.
HOST1_STORAGE_MODE_INTERNAL=true
# ==============================================================================================
# ── IDENTITY & CONNECTIVITY ───────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Identity ━━━
# HOSTN hostname lives in master.conf (not a credential — safe for all servers).
# SSH key used for all server-to-server operations — rsync, failover, conf sync.
# Convention: /root/.ssh/<hostname-lowercase-no-unraid-prefix>_rsync_automation
# Must be in /root/.ssh/ and authorised in the partner's /root/.ssh/authorized_keys.
# Run Partnership/ssh_setup.sh to generate the key and copy it to the partner.
HOST1_SSH_KEY="/root/.ssh/gmer4lfe_rsync_automation"
HOST1_STORAGE_PATH="/mnt/user"
HOST1_OWNER="gmer4lfe"
HOST1_OWNER_EMAIL="gmer4lfe@gmail.com"
# ━━━ Unraid API ━━━
# Used by the Varaverk plugin to query this server's Unraid GraphQL API.
# Generate in Unraid: Settings → Management Access → API Keys → + New Key
HOST1_UNRAID_API_KEY="1825c3a2e03ea5089974f4da2e171aa2d5907a1dea23cc479c33e492c8ff4dbb"
# ━━━ Emby ━━━
HOST1_EMBY_CONTAINER="Emby"
HOST1_EMBY_URL="http://localhost:8096"
HOST1_EMBY_API_KEY="0c27448d93a7431f9ac63569f7655829"
# ━━━ Jellyfin ━━━
HOST1_JELLYFIN_CONTAINER="Jellyfin"
HOST1_JELLYFIN_URL="http://localhost:8095"
HOST1_JELLYFIN_API_KEY="4e820e7df74c4933acec212b1996314e"
# ━━━ Gitea ━━━
# Personal access token for gitea_ssh_setup.sh.
# Create in Gitea: Settings → Applications → Generate Token → scope: write:user
HOST1_GITEA_API_TOKEN=""
# ━━━ Notifications ━━━
# Discord webhook — leave blank to disable.
HOST1_DISCORD_WEBHOOK=""
# ==============================================================================================
# ── PARTNERSHIP ───────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Auth containers reconfigured on onboard/offboard.
# Format: "ContainerName|WebUIPort"
HOST1_PARTNERSHIP_AUTH_WEBUIS=(
"NginxProxyManager|81"
"Lldap-Gmer4Lfe|17170"
"Authelia|9091"
"Authelia-Secondary|9092"
)
# XML templates pushed to mirror during onboard — auth stack.
# Dependencies (databases) must come before apps that depend on them.
HOST1_PARTNERSHIP_AUTH_STACK=(
# Dependencies first — Mariadb/Redis must be healthy before Authelia starts
"my-Mariadb-Authelia.xml"
"my-Mariadb-Authelia-Secondary.xml"
"my-Redis-Authelia.xml"
"my-Redis-Authelia-Secondary.xml"
# Auth apps — deployed after their deps are confirmed healthy
"my-Authelia.xml"
"my-Authelia-Secondary.xml"
"my-NginxProxyManager.xml"
"my-Lldap-Gmer4Lfe.xml"
)
# XML templates pushed to mirror during onboard — arr stack.
HOST1_PARTNERSHIP_ARR_STACK=(
"my-Sonarr.xml"
"my-Radarr.xml"
"my-Lidarr.xml"
"my-Prowlarr.xml"
"my-Bazarr.xml"
)
# Paths the partner should collect during the grace window after offboard.
HOST1_PARTNERSHIP_MIRROR_BACKUPS=(
# "/mnt/user/appdata-Fallback/Jayred365-Emby"
)
# Containers parked on this server when partnership is active.
HOST1_PARTNERSHIP_OWN_CONTAINERS=(
# "Emby"
# "NginxProxyManager"
)
# Containers stopped on THIS server before deploying the mirror's stack on onboard.
HOST1_PARTNERSHIP_REPLACE_CONTAINERS=(
)
# Arr containers stopped on this server when mirror's arr stack is deployed.
HOST1_PARTNERSHIP_ARR_REPLACE_CONTAINERS=(
)
# Emby admin provisioning — owner controls whether Emby is shared.
HOST1_PARTNERSHIP_PROVISION_EMBY_ADMIN=false # owner controls whether Emby is shared
HOST1_PARTNERSHIP_EMBY_PORT=8096
HOST1_PARTNERSHIP_EMBY_ADMIN_USER="" # this server's desired Emby username
HOST1_PARTNERSHIP_EMBY_ADMIN_PASS="" # this server's desired Emby password
# ==============================================================================================
# ── RSYNC ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Daily Sync Shares ━━━
# Media shares this host pushes to all other nodes every night.
# Uses DEFAULT_RSYNC_OPTS from master.conf — no profile needed.
HOST1_DAILY_SYNC_SHARES=(
/mnt/user/Books
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Nextcloud
/mnt/user/stand-up_comedy
/mnt/user/Sports
# /mnt/user/Tv_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Movies
/mnt/user/Anime_Shows
)
# ━━━ Personal Shares ━━━
# Private encrypted shares synced for offsite backup, independent of media shares.
HOST1_PERSONAL_SHARES=(
# /mnt/user/HOST1-Personal # uncomment after creating encrypted dataset
)
# ━━━ Weekly Sync Shares ━━━
# Appdata shares synced during the weekly maintenance window.
# Profiles (emby, critical-data) drive container stops — define in master.conf.
HOST1_WEEKLY_SYNC_SHARES=(
"/mnt/user/Media_Server/Emby" # emby profile — full clean mirror
"/mnt/user/appdata-Fallback/Critical-Data" # critical-data profile — auth stack
)
# ━━━ Intermediate Sync Shares ━━━
# Shares synced every 4 hours. Leave empty to skip mid-day rsync.
HOST1_INTERMEDIATE_SYNC_SHARES=(
# Add shares here to enable mid-day rsync
# Example: "/mnt/user/Emby_Metadata"
)
# ━━━ Critical Sync Shares ━━━
# Appdata shares synced every 30 minutes.
# Format: "/path/to/share" or "/path/to/share|profile-name"
HOST1_CRITICAL_SYNC_SHARES=(
"/mnt/user/appdata-Fallback/Critical-Data|critical-fallback" # auth dirty sync — stays running
"/mnt/user/Media_Server/Emby|emby-fallback" # Emby dirty sync — stays running
)
# ━━━ Backup Verify ━━━
# Leave empty to use HOST1_DAILY_SYNC_SHARES automatically.
HOST1_BACKUP_VERIFY_SHARES=(
# leave empty to use HOST1_DAILY_SYNC_SHARES automatically
)
# ━━━ HOSTN Rsync Profile — hostn-appdata ━━━
# Host-specific appdata sync profile.
PROFILE_RSYNC_OPTS[hostn-appdata]="-av --info=progress2 --bwlimit=${PROFILE_BW_LIMIT[hostn-appdata]:-8000}"
PROFILE_BW_LIMIT[hostn-appdata]=8000
PROFILE_RETRY_COUNT[hostn-appdata]=3
PROFILE_SLEEP[hostn-appdata]=300
PROFILE_CRITICAL_CONTAINER_NAMES[hostn-appdata]=""
PROFILE_DELAYED_CONTAINERS[hostn-appdata]=""
PROFILE_CONTAINER_DELAY[hostn-appdata]=5
PROFILE_EXCLUDE_DIRS[hostn-appdata]="logs *.tmp"
# ==============================================================================================
# ── FALLBACK ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ DDNS ━━━
# DDNS containers this host manages.
HOST1_DDNS_CONTAINERS=(
"Gmer4Lfe.com"
)
# ━━━ Internet Loss ━━━
# Containers stopped immediately when internet is lost.
FALLBACK_HOST1_STOP_ON_NO_NET=(
"Gmer4Lfe.com"
)
# ━━━ Fallback Tiers — HOSTN Runs for Partner ━━━
# Containers this host starts when the partner goes down.
# Replace HOST2 below with the actual remote host ID (HOST1, HOST2, etc.)
FALLBACK_HOST1_COVERS_HOST2_TIER1=(
"Gmer4Lfe.us"
"VaultWarden-Jayred365"
)
FALLBACK_HOST1_COVERS_HOST2_TIER2=(
# "container-placeholder"
)
FALLBACK_HOST1_COVERS_HOST2_TIER3=(
# "container-placeholder"
)
FALLBACK_HOST1_COVERS_HOST2_TIER4=(
# "container-placeholder"
)
# ━━━ Tier Delays — This Host's Outage Timers ━━━
# How long THIS host must be down before each tier activates on the partner.
HOST1_TIER2_DELAY=240 # 4 hours — NextCloud, Immich
HOST1_TIER3_DELAY=720 # 12 hours — secondary services
HOST1_TIER4_DELAY=1440 # 24 hours — arrs + downloaders
# ━━━ Rsync Writeback ━━━
HOST1_TIER1_WRITEBACK_DELAY=60 # skip Emby writeback if outage under 1hr
FALLBACK_HOST1_WRITEBACK_TIER1=(
"/mnt/user/Media_Server/Emby" # watch states built up during outage
)
FALLBACK_HOST1_WRITEBACK_TIER2=(
"/mnt/user/appdata-Fallback/Important-Data" # NextCloud + Postgres — files added during outage
)
FALLBACK_HOST1_WRITEBACK_TIER3=(
# "location-placeholder"
)
FALLBACK_HOST1_WRITEBACK_TIER4=(
"/mnt/user/appdata-Fallback/Arrs_Stack" # arr databases — downloads queued during outage
)
# ==============================================================================================
# ── DOCKER ────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Docker Daily Restart ━━━
HOST1_DAILY_RESTART_CONTAINERS=(
"NginxProxyManager"
"Lldap-Gmer4Lfe"
"Authelia"
"Authelia-Secondary"
"Dispatcharr-Iptv-Users"
"Dispatcharr" # Live TV scheduler — degrades without daily restart
"Dispatcharr-Basic"
"ErsatzTV-Emby"
"slskd" # Soulseek connection drops after extended uptime; restart refreshes share index
)
# ━━━ Docker Weekly Restart ━━━
HOST1_WEEKLY_RESTART_CONTAINERS=(
"NextCloud"
"Organizrv2-Gmer4Lfe"
"AdGuard-Home"
"Immich-Gmer4Lfe"
)
# ━━━ Docker Watchdog ━━━
# Memory hard limits in MB — immediate restart if exceeded.
# 20GB=20480 16GB=16384 8GB=8192 4GB=4096 2GB=2048 1GB=1024
declare -A HOST1_WATCHDOG_CONTAINERS=(
["Emby"]=20480 # 20GB — large library + active transcodes
["LidaTube"]=6144 # 6GB — memory leak over time
["Tdarr"]=6144 # 6GB — encoding is memory intensive
["Code-Server"]=1024 # 1GB — should never need more
)
# HTTP health check URLs — checked every cycle.
declare -A HOST1_WATCHDOG_CONTAINER_URLS=(
["Emby"]="http://localhost:8096"
["NginxProxyManager"]="http://localhost:7818"
["Authelia"]="http://localhost:9091/api/health"
["Authelia-Secondary"]="http://localhost:9092/api/health"
["Lldap-Gmer4Lfe"]="http://localhost:17170"
)
# API-level health checks. Format: ["ContainerName"]="url|expected_json_key|expected_value"
declare -A HOST1_WATCHDOG_CONTAINER_API_CHECKS=(
)
# Required containers — must always be running.
HOST1_WATCHDOG_REQUIRED_CONTAINERS=(
"NginxProxyManager"
"Lldap-Gmer4Lfe"
"Mariadb-Authelia"
"Mariadb-Authelia-Secondary"
"Redis-Authelia"
"Redis-Authelia-Secondary"
"Authelia"
"Authelia-Secondary"
)
# Containers to skip in Tier 2 global scan.
HOST1_WATCHDOG_SCAN_IGNORE=(
"DashGate"
"PIA-WG-Config-Generator"
"Aperture"
"Aperture-Kids"
"pgvector-18-Apeture-Kids"
"Pgvector18-Aperture"
"emby-test" # broken test container (exit 127 — bad image)
)
# Dependency ordering — skip restarting a container if its dependency is also down.
declare -A HOST1_WATCHDOG_DEPENDENCIES=(
["Authelia"]="Mariadb-Authelia Redis-Authelia"
["Authelia-Secondary"]="Mariadb-Authelia Redis-Authelia-Secondary"
["NextCloud"]="Postgres-NextCloud"
)
# Per-container appdata growth suppress ceilings in MB.
declare -A HOST1_WATCHDOG_APPDATA_SIZES=(
["Tdarr"]="25600" # 25GB — transcode cache grows legitimately during active jobs
["7dtd"]="20480" # 20GB — game server world data, expected to be large
)
# ━━━ Network Watchdog ━━━
HOST1_NETWORK_WATCHDOG_DDNS_DOMAIN="gmer4lfe.com"
HOST1_NETWORK_WATCHDOG_DDNS_CONTAINER="Gmer4Lfe.com"
HOST1_NETWORK_WATCHDOG_NPM_URL="https://gmer4lfe.com"
# ━━━ Docker Network Connect ━━━
HOST1_NETWORK_CONNECT_CONTAINERS=(
"memcached"
"Npm-CrowdSec"
)
HOST1_NETWORK_CONNECT_NETWORKS=(
"high-availability"
)
# ==============================================================================================
# ── MEDIA ─────────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Media Permissions ━━━
HOST1_MEDIA_PERMISSION_SHARES=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
/mnt/user/appcache
/mnt/user/Books
/mnt/user/Downloads
/mnt/user/Games
/mnt/user/Intros
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movie_Recordings
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Music_Videos
/mnt/user/Photo
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Tv_Recordings
/mnt/user/Tv_Shows
/mnt/user/YouTube
)
# ━━━ Media Cleaner ━━━
HOST1_ANIME_CLEAN_FOLDERS=(
/mnt/user/Anime_Movies
/mnt/user/Anime_Movies-Old
/mnt/user/Anime_Shows
/mnt/user/Anime_Shows-Old
)
HOST1_MEDIA_CLEAN_FOLDERS=(
/mnt/user/Kids_Movies
/mnt/user/Kids_Tv_Shows
/mnt/user/Movies
/mnt/user/Music
/mnt/user/Sports
/mnt/user/stand-up_comedy
/mnt/user/Tv_Shows
)
# ==============================================================================================
# ── ARR STACK ─────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Downloaders ━━━
HOST1_SLSKD_URL="http://localhost:8980"
HOST1_SLSKD_API_KEY="4bF9kL2mNpQrT7vWxYz1A3dEgHjKoRsU"
HOST1_SLSKD_FAILED_IMPORTS_DIR="/mnt/user/Temp_Storage/Slskd/completed/failed_imports"
HOST1_SABNZBD_URL="http://localhost:8180"
HOST1_SABNZBD_API_KEY="8bfefe41d83b4d50883e32859b55ca9a"
HOST1_QBIT_URL="http://localhost:8080"
HOST1_QBIT_USERNAME="root"
HOST1_QBIT_PASSWORD="Stay0utD!ck"
# ━━━ Lidarr ━━━
HOST1_LIDARR_URL="http://localhost:8686"
HOST1_LIDARR_API_KEY="b2977e71ef074bc0a0529d9fcce3b2dc"
HOST1_LIDARR_MUSIC_ROOT="/mnt/user/Music-New"
HOST1_FANART_API_KEY="Yd7147a43b692df0b364b94dc47efb81"
HOST1_LASTFM_API_KEY="be6dc169c33ae263e690c30d18b7491d"
declare -A HOST1_LIDARR_PATH_MAP=(
["/ext-music"]="/mnt/user/Music-New"
)
# ━━━ Sonarr ━━━
HOST1_SONARR_URL="http://localhost:8989"
HOST1_SONARR_API_KEY="130decd3db5b4c25afad64864cd03f9f"
HOST1_SONARR_TV_ROOT="/mnt/user/Tv_Shows"
declare -A HOST1_SONARR_PATH_MAP=(
["/tv"]="/mnt/user/Tv_Shows"
["/ext-standup-comedy"]="/mnt/user/stand-up_comedy/series"
["/kids tv"]="/mnt/user/Kids_Tv_Shows"
["/ext-anime-shows"]="/mnt/user/Anime_Shows-Old"
)
# ━━━ Radarr ━━━
HOST1_RADARR_URL="http://localhost:7878"
HOST1_RADARR_API_KEY="d43a3ec6cf1549edb4af0cc63f98b2a9"
HOST1_TMDB_API_KEY="3dac5e2e49b5540472d2eafec4f01260"
HOST1_RADARR_MOVIES_ROOT="/mnt/user/Movies"
declare -A HOST1_RADARR_PATH_MAP=(
["/movies"]="/mnt/user/Movies"
["/kids movies"]="/mnt/user/Kids_Movies"
["/ext-stand-up-comedy"]="/mnt/user/stand-up_comedy/specials"
["/anime-movies"]="/mnt/user/Anime_Movies-Old"
["/ext-anime-movies"]="/mnt/user/Anime_Movies-Old"
)
# ━━━ Arr Recovery Toggles ━━━
HOST1_LIDARR_RECOVERY=true # HOST1 only — exits cleanly on HOST2
HOST1_SONARR_RECOVERY=true
HOST1_RADARR_RECOVERY=true
# ==============================================================================================
# ── TRANSCODES ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOST1_RAMDISK_SIZE="10G"
HOST1_RAMDISK_WARN_GB=8.5 # flip to SSD when ramdisk usage reaches this
HOST1_RAMDISK_LOW_GB=7 # flip back to ramdisk when usage drops to this
HOST1_TRANSCODE_SSD="/mnt/cache/Temp_Storage/Emby/Transcodes/"
HOST1_TRANSCODE_SERVERS=(
"${HOST1_EMBY_CONTAINER}|${HOST1_EMBY_URL}|${HOST1_EMBY_API_KEY}|emby"
"${HOST1_JELLYFIN_CONTAINER}|${HOST1_JELLYFIN_URL}|${HOST1_JELLYFIN_API_KEY}|jellyfin"
)
# ==============================================================================================
# ── MONITORS ──────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# ━━━ Certificate Monitor ━━━
HOST1_CERT_MONITOR_DOMAINS=(
"Gmer4Lfe.com"
"Gmer4Lfe.us"
)
# ━━━ SMART Health ━━━
HOST1_SMART_IGNORE_DRIVES=(
"sda" # boot USB — SMART not meaningful on flash drives
)
# ━━━ ZFS Report ━━━
HOST1_ZFS_REPORT_IGNORE_POOLS=(
"disk5"
"disk6"
"disk8"
"disk9"
"disk10"
)
# ==============================================================================================
# ── RESOURCE MANAGER ──────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOST1_RW_PAUSE_CONTAINERS=(
"Huntarr" # arr search automation — safe to suspend
"Cleanuparr" # download cleanup — safe to suspend
"Healarr" # arr health checks — safe to suspend
"Soularr" # Slskd automation — background only
"ChannelTube" # YouTube archiver — background only
"Pinchflat" # YouTube archiver — background only
)
HOST1_RW_STOP_CONTAINERS=(
"LocalAI" # GPU/CPU heavy — largest RAM consumer when idle
"7DaysToDie" # game server — optional
"V-Rising" # game server — optional
"Code-Server" # IDE — not needed during pressure events
)
# ==============================================================================================
# ── SYSTEM WATCHDOG ───────────────────────────────────────────────────────────────────────────
# ==============================================================================================
HOST1_SYS_WATCHDOG_NIC="eth0"
HOST1_SYS_WATCHDOG_CHECK_DOCKER_DAEMON=true
HOST1_SYS_WATCHDOG_CHECK_ROOTFS=true
HOST1_SYS_WATCHDOG_CHECK_KERNEL_OOPS=true
HOST1_SYS_WATCHDOG_CHECK_FD=true
HOST1_SYS_WATCHDOG_CHECK_BOOT=true
HOST1_SYS_WATCHDOG_CHECK_OOM=true
HOST1_SYS_WATCHDOG_CHECK_RAM=true
HOST1_SYS_WATCHDOG_CHECK_LOG=true
HOST1_SYS_WATCHDOG_CHECK_ARC=true
HOST1_SYS_WATCHDOG_CHECK_CPU_TEMP=true
HOST1_SYS_WATCHDOG_CHECK_LOAD=false
HOST1_SYS_WATCHDOG_CHECK_ZOMBIES=true
HOST1_SYS_WATCHDOG_CHECK_CONTAINERS=true
HOST1_SYS_WATCHDOG_CHECK_TMP=true
HOST1_SYS_WATCHDOG_CHECK_MDSTAT=true
HOST1_SYS_WATCHDOG_CHECK_NETWORK=true
HOST1_SYS_WATCHDOG_CHECK_SSHD=true
HOST1_SYS_WATCHDOG_CHECK_RUNAWAY=false
# ==============================================================================================
# ── AUTH STACK ────────────────────────────────────────────────────────────────────────────────
# ==============================================================================================
# Credentials for the Varaverk Auth Stack page (NPM, lldap, Authelia).
# ━━━ NginxProxyManager ━━━
# Admin API runs on 7818 (not 81 — 81 is the partnership WebUI port).
HOST1_NPM_URL="http://localhost:7818"
HOST1_NPM_USER="" # NPM admin email
HOST1_NPM_PASS="" # NPM admin password
# ━━━ lldap ━━━
HOST1_LLDAP_URL="http://localhost:17170"
HOST1_LLDAP_USER="admin" # lldap admin username
HOST1_LLDAP_PASS="" # lldap admin password
# ━━━ Authelia ━━━
HOST1_AUTHELIA_CONFIG="/mnt/user/appdata/Authelia/configuration.yml"
HOST1_AUTHELIA_CONTAINER="Authelia"
# ==============================================================================================
# ──────────────────────── End Of HOSTn Variables ──────────────────────────────────────────────
# ==============================================================================================
@@ -1,165 +0,0 @@
# ━━━━━ TOOLS ━━━━━
**Situational utilities — run when something needs fixing, not on a schedule.**
Recovery, repair, migration, and inspection tools for situations that arise outside
the scheduled maintenance model. These scripts sit ready for the moment you need them.
> **None of these scripts run on a schedule.** A script belongs here when it solves
> a specific operational situation — something you run in response to a problem, before
> a risky operation, or during a one-time setup task. Having a dedicated folder keeps
> the other folders clean and makes it obvious what runs routinely vs. situationally.
---
## ━━━ THE PROBLEM THAT BUILT THIS ━━━
**Fallback State Stuck After Testing**
Run a fallback test, something exits uncleanly, state file shows `FALLBACK`.
`fallback.sh` resumes and reads FALLBACK — starts containers it shouldn't, makes
decisions based on a state that doesn't reflect reality. Manual recovery means
knowing the exact file format and every field to reset. At 2am after a failed test.
Fix: `fallback_state_reset.sh` — shows current state, prompts for confirmation,
resets cleanly to NORMAL.
**Container Stuck on Watchdog Skip List After Fixing the Problem**
Authelia hit the restart loop limit — went on the skip list. Problem fixed. But
the watchdog still isn't monitoring it because the skip list persists on `/boot/config`
across reboots. Where's the file? What format? How do you clear restart history?
Fix: `watchdog_skip_list_manager.sh` — shows the skip list and which containers are
running vs. stopped, clears specific containers with confirmation.
**Emby Crashing With No Clear Cause After a Power Cut**
Server lost power with Emby running. Emby comes back, runs for 20 minutes, crashes.
Logs show database errors. Which database? library.db? users.db? Each has different
recovery implications — deleting the wrong one resets all user watch history.
Fix: `emby_database_repair.sh` — stops Emby, runs `PRAGMA integrity_check` on every
database, reports per-database with specific guidance on what to do about each one.
**Files Owned by Root After an Admin Copy**
`scp` a file into a media share. File arrives as `root:root`. Radarr fails to import —
permission denied. The daily permissions script won't run for another 20 hours. Running
`media_shares_permissions.sh` on the whole library takes 30 minutes just to fix one dir.
Fix: `bulk_permissions_repair.sh` — takes specific paths, applies correct ownership and
permissions in seconds.
**No Way to Back Up a Container Before a Risky Update**
Major version update, changelog says "database migration — no rollback." You want a
point-in-time backup. But `cp -r` while the container is running produces an
inconsistent backup, and tar without stopping the container is equally unreliable.
Fix: `container_data_export.sh` — stops the container cleanly, archives appdata to a
timestamped `.tar.gz`, verifies archive integrity, restarts the container.
**Fresh HOST2 Has Shares Configured But Directories Missing**
Fresh install on HOST2. Restored `/boot/config/shares/` from backup. Array starts.
Shares show in the UI. But the actual `/mnt/diskN/sharename` directories don't exist —
unRAID created the share definitions but not the directories. rsync.sh aborts.
Fix: `recreate_shares.sh` — reads every `.cfg` file, creates directories on each
included disk, places `.recovery` markers so the first rsync won't delete anything.
---
## ━━━ WHAT THIS FOLDER DOES ━━━
One role: hold scripts for situations the scheduled maintenance model can't handle.
Every script here was written because a specific situation arose that required bash
commands to resolve — and that situation is guaranteed to arise again. When you encounter
something new, write the tool. Store it here. Find it at 2am next time.
**Recovery Tools** — Restore known-good state after a failure
`fallback_state_reset.sh`, `watchdog_skip_list_manager.sh`
**Diagnostic Tools** — Inspect and verify before acting
`emby_database_repair.sh`, `continuous_scripts_status.sh`
**Repair Tools** — Fix a specific known problem
`bulk_permissions_repair.sh`, `zfs_pool_scrub.sh`
**Lifecycle Tools** — Backup, setup, and migration support
`container_data_export.sh`, `recreate_shares.sh`, `claude_startup.sh`, `ramdisk_stop.sh`
**Library Sync Bootstrap** — Close the gap between Emby and arr libraries
`emby_to_lidarr_sync.sh`, `emby_to_sonarr_sync.sh`, `emby_to_radarr_sync.sh`
---
## ━━━ RELATIONSHIP TO OTHER FOLDERS ━━━
```
System_Essentials/ ← regular system maintenance — scheduled
Docker_Essentials/ ← regular container management — scheduled
Monitors/ ← regular health reporting — scheduled
Orchestrators/ ← regular maintenance windows — scheduled
Fallback/ ← automated fallback/handback — event-driven
Tools/ ← situational utilities — run when needed
```
Some tools interact with state written by other folders:
```
Fallback/
fallback.sh ──────── writes FALLBACK_STATE_FILE ──► fallback_state_reset.sh reads/writes it
Docker_Essentials/
docker_watchdog.sh ── writes skip list + history ──► watchdog_skip_list_manager.sh manages them
Docker_Essentials/ + System_Essentials/ + Fallback/
All continuous scripts ──────────────────────────► continuous_scripts_status.sh reads their state
```
Tools never call scripts in other folders. Other folders never call Tools scripts.
The relationship is one-way: Tools act on state that other scripts have written.
---
## ━━━ SCRIPTS IN THIS FOLDER ━━━
| Script | What It Fixes | When to Run |
|--------|--------------|-------------|
| `fallback_state_reset.sh` | State file stuck in FALLBACK after test or failed handback | After fallback testing or manual intervention |
| `watchdog_skip_list_manager.sh` | Container stuck on watchdog skip list after fixing root cause | After fixing a container that hit the restart loop limit |
| `bulk_permissions_repair.sh` | Files owned by wrong user after admin copy or bad container config | When arr operations fail due to permissions |
| `container_data_export.sh` | Need a clean backup before a risky container update or migration | Before major updates, appdata migrations, or container removals |
| `emby_database_repair.sh` | Emby crashing with database errors after power loss or crash | When Emby logs show corruption or repeated crashes |
| `zfs_pool_scrub.sh` | Verify ZFS pool integrity — catch silent corruption before it spreads | Monthly, or after any disk or power event |
| `recreate_shares.sh` | Share directories missing after fresh install or disk rebuild | After fresh unRAID install or disk replacement on HOST2 |
| `continuous_scripts_status.sh` | Need a live view of watchdog and fallback state | Any time — manual dashboard, no schedule |
| `claude_startup.sh` | Claude Code session setup after reboot — symlinks persistent storage | After each unRAID reboot, or called by array_started.sh |
| `ramdisk_stop.sh` | Safely stop the transcode ramdisk — redirect symlink to SSD, unmount, update state | Before re-running ramdisk_setup.sh with new size or thresholds |
| `emby_to_lidarr_sync.sh` | Add all Emby album artists not yet tracked in Lidarr | After Lidarr setup, database wipe, or when you suspect gaps |
| `emby_to_sonarr_sync.sh` | Add all Emby TV series not yet tracked in Sonarr | After Sonarr setup, database wipe, or when you suspect gaps |
| `emby_to_radarr_sync.sh` | Add all Emby movies not yet tracked in Radarr | After Radarr setup, database wipe, or when you suspect gaps |
---
## ━━━ HOW THE SCRIPTS RELATE ━━━
All Tools scripts are independent — none call each other, none are called by other Tools.
```
Situation arises
┌──────────────────────────────────────────────────────────────────────┐
│ Tools/ Run directly when needed │
│ │
│ fallback_state_reset.sh ◄── after fallback test / failed handback│
│ watchdog_skip_list_manager ◄── after fixing a crash-looping container│
│ bulk_permissions_repair ◄── wrong ownership after copy or rsync │
│ container_data_export ◄── before a risky update or migration │
│ emby_database_repair ◄── Emby logs show corruption │
│ zfs_pool_scrub ◄── monthly integrity check / post-event │
│ recreate_shares ◄── fresh HOST2 setup or disk rebuild │
│ continuous_scripts_status ◄── manual status check at any time │
│ claude_startup ◄── after each unRAID reboot │
│ ramdisk_stop ◄── before ramdisk resize / remount │
│ │
│ emby_to_lidarr_sync ◄── Lidarr setup / database wipe / gap │
│ emby_to_sonarr_sync ◄── Sonarr setup / database wipe / gap │
│ emby_to_radarr_sync ◄── Radarr setup / database wipe / gap │
└──────────────────────────────────────────────────────────────────────┘
State files in other folders (Fallback/, Docker_Essentials/) may be read or written.
No other scripts call into Tools/.
```

Some files were not shown because too many files have changed in this diff Show More