Comments asserting a partner is down, being rebuilt or not yet installed are true on the day they are written and wrong afterwards, and nothing prompts anyone to revisit them. Behaviour is described generically; current state belongs in operator memory, not in tracked files.
212 lines
9.2 KiB
Bash
Executable File
212 lines
9.2 KiB
Bash
Executable File
#!/bin/bash
|
|
# ==============================================================================================
|
|
# ============================= AI Token Ledger Sync ===========================================
|
|
# ==============================================================================================
|
|
#
|
|
# PURPOSE
|
|
# ─────────────────────────────────────────────────────────────────────────────
|
|
# Pulls each partner's AI token ledger into a RAM cache at /tmp/.cache/vv/ai/, so the AI tab
|
|
# can report usage for the whole fleet instead of only the host the browser happens to be on.
|
|
#
|
|
# Each host records its own turns to data/ai_token_history.db and nothing syncs that file, so
|
|
# without this a host can only ever see its own totals. The tab is careful to say "not collected
|
|
# here" rather than 0 for a partner it cannot see; this script is what turns that into a number.
|
|
#
|
|
# Same trick as conf_sync.sh, and deliberately so — resolve the partner over Tailscale, scp one
|
|
# small file into a tmpfs cache, let the reader treat a missing file as "unknown".
|
|
#
|
|
# ==============================================================================================
|
|
# OPERATIONAL MODEL
|
|
# ==============================================================================================
|
|
#
|
|
# 1. Gates — PARTNERSHIP_ENABLED, AI_ENABLED, AI_TOKEN_SYNC_ENABLED
|
|
# 2. Per partner:
|
|
# a. Resolve their Tailscale IP
|
|
# b. Resolve their SCRIPTS_DIR from their varaverk.cfg (they may be in appdata mode)
|
|
# c. scp their data/ai_token_history.db → $AI_TOKEN_CACHE_DIR/<slot>.tokens.db
|
|
#
|
|
# Pull only, no push. conf_sync.sh pushes as well because a partner may be unable to reach us
|
|
# and still needs our credentials; nothing here is needed by anyone else, and a reader that
|
|
# fetches its own data controls its own freshness rather than depending on the partner's cron.
|
|
#
|
|
# ==============================================================================================
|
|
# OPERATIONAL SAFEGUARDS
|
|
# ==============================================================================================
|
|
#
|
|
# An unreachable partner is not a failure.
|
|
# HOST2 is expected to be down for long stretches during onboarding. A warn every four
|
|
# hours would train the operator to ignore this script's output, and the AI diagnostic
|
|
# path treats every log WARN as actionable. Unresolvable partners are counted and
|
|
# reported once at info level; only a partner that resolves and then fails to transfer
|
|
# is treated as an error.
|
|
#
|
|
# The cache is never written directly.
|
|
# scp lands on a .part file that is renamed into place, so a transfer interrupted halfway
|
|
# cannot leave the reader parsing half a ledger. A truncated final row would be skipped by
|
|
# the field-count check on the PHP side, but a torn file should not reach it at all.
|
|
#
|
|
# Nothing is ever written back to the partner.
|
|
# This script only reads. A bug here cannot corrupt a partner's accounting.
|
|
#
|
|
# The cache is tmpfs and deliberately not preserved.
|
|
# Unlike the conf cache there is no save/restore pair. Stale counters are worse than
|
|
# absent ones: absent reads as "not collected here", stale reads as fact.
|
|
#
|
|
# ==============================================================================================
|
|
# CONFIGURATION
|
|
# ==============================================================================================
|
|
#
|
|
# master.conf
|
|
#
|
|
# AI_ENABLED Whole AI subsystem gate
|
|
# AI_TOKEN_SYNC_ENABLED This script's own toggle (default: true)
|
|
# PARTNERSHIP_ENABLED Checked via require_partnership()
|
|
# SSH_KEY Key used for all partner ssh/scp operations
|
|
#
|
|
# load_config.sh
|
|
#
|
|
# AI_TOKEN_CACHE_DIR tmpfs directory the tab reads partner ledgers from
|
|
#
|
|
# host*.conf
|
|
#
|
|
# HOST* — hostnames used to build the partner list via detect_hosts()
|
|
#
|
|
# ==============================================================================================
|
|
# RUNTIME MODES
|
|
# ==============================================================================================
|
|
#
|
|
# ai_token_sync.sh Pull every reachable partner's ledger
|
|
# ai_token_sync.sh --dry-run Report what would be pulled, transfer nothing
|
|
# ai_token_sync.sh --log Verbose output
|
|
#
|
|
# ==============================================================================================
|
|
|
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
source "$SCRIPT_DIR/../load_config.sh"
|
|
|
|
parse_args "$@"
|
|
|
|
if [[ "$EUID" -ne 0 ]]; then
|
|
error "Must be run as root"
|
|
exit 1
|
|
fi
|
|
|
|
acquire_lock
|
|
|
|
detect_hosts
|
|
require_partnership
|
|
|
|
if [[ "${AI_ENABLED:-false}" != true ]]; then
|
|
log "AI_ENABLED=false — skipping"
|
|
exit 0
|
|
fi
|
|
|
|
if [[ "${AI_TOKEN_SYNC_ENABLED:-true}" == false ]]; then
|
|
log "AI_TOKEN_SYNC_ENABLED=false — skipping"
|
|
exit 0
|
|
fi
|
|
|
|
CACHE_DIR="$AI_TOKEN_CACHE_DIR"
|
|
SSH_TIMEOUT=10
|
|
|
|
# Mirrors conf_sync.sh — the remote may be in appdata storage mode, so its ledger is not
|
|
# necessarily under /boot.
|
|
_remote_scripts_dir() {
|
|
local ip="$1"
|
|
local cfg line sd
|
|
cfg=$(timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
|
|
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes -o StrictHostKeyChecking=no \
|
|
"root@${ip}" "cat /boot/config/plugins/varaverk/varaverk.cfg 2>/dev/null" 2>/dev/null) || true
|
|
while IFS= read -r line; do
|
|
[[ "$line" == SCRIPTS_DIR=* ]] || continue
|
|
sd="${line#SCRIPTS_DIR=}"; sd="${sd//\"/}"; sd="${sd//\'/}"
|
|
echo "$sd"; return
|
|
done <<< "$cfg"
|
|
echo "/boot/config/plugins/varaverk"
|
|
}
|
|
|
|
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes will be made"
|
|
|
|
if [[ "$DRY_RUN" == false ]]; then
|
|
mkdir -p "$CACHE_DIR" && chmod 755 "$CACHE_DIR"
|
|
fi
|
|
|
|
PULLED=0
|
|
OFFLINE=0
|
|
FAILED=0
|
|
|
|
for host_var in $(compgen -v | grep -E '^HOST[0-9]+$' | sort); do
|
|
partner_host="${!host_var}"
|
|
[[ -z "$partner_host" ]] && continue
|
|
[[ "${host_var,,}" == "${MY_ID,,}" ]] && continue
|
|
|
|
partner_slot="${host_var,,}"
|
|
partner_ip=$(resolve_tailscale_ip "$partner_host" 2>/dev/null || true)
|
|
|
|
if [[ -z "$partner_ip" ]]; then
|
|
log "$partner_host — unresolvable, leaving any cached ledger as-is"
|
|
(( OFFLINE++ ))
|
|
continue
|
|
fi
|
|
|
|
# Liveness and ledger presence are separate questions, probed separately on purpose.
|
|
# Tailscale hands back an IP for a peer that is registered but powered off, so resolving
|
|
# proves nothing. And a single `ssh test -f` answers both questions at once: it fails
|
|
# identically whether the host is down or the file is simply absent. Treating that one
|
|
# failure as "no ledger" would delete a perfectly good cached copy every time the partner
|
|
# blinked — turning "synced 3h ago" into "not collected here" on a transient.
|
|
if ! timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
|
|
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes -o StrictHostKeyChecking=no \
|
|
"root@${partner_ip}" true 2>/dev/null; then
|
|
log "$partner_host — not answering, keeping any cached ledger as-is"
|
|
(( OFFLINE++ ))
|
|
continue
|
|
fi
|
|
|
|
remote_sd=$(_remote_scripts_dir "$partner_ip")
|
|
remote_db="${remote_sd}/data/ai_token_history.db"
|
|
|
|
if [[ "$DRY_RUN" == true ]]; then
|
|
warn "DRY RUN — would pull $partner_host:$remote_db → $CACHE_DIR/${partner_slot}.tokens.db"
|
|
continue
|
|
fi
|
|
|
|
# Reachable, but nothing recorded there — a partner with AI off, or one that has simply
|
|
# never been asked anything. Now that liveness is established this is a real answer, so the
|
|
# stale copy goes: the tab should say "not collected here", not quote a number from before.
|
|
if ! timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
|
|
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes -o StrictHostKeyChecking=no \
|
|
"root@${partner_ip}" "[[ -f '$remote_db' ]]" 2>/dev/null; then
|
|
log "$partner_host — reachable, but no ledger there yet"
|
|
rm -f "$CACHE_DIR/${partner_slot}.tokens.db"
|
|
(( OFFLINE++ ))
|
|
continue
|
|
fi
|
|
|
|
if timeout "$SSH_TIMEOUT" scp -i "$SSH_KEY" \
|
|
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes -o StrictHostKeyChecking=no \
|
|
"root@${partner_ip}:${remote_db}" \
|
|
"$CACHE_DIR/${partner_slot}.tokens.db.part" 2>/dev/null \
|
|
&& mv -f "$CACHE_DIR/${partner_slot}.tokens.db.part" "$CACHE_DIR/${partner_slot}.tokens.db"; then
|
|
chmod 644 "$CACHE_DIR/${partner_slot}.tokens.db" 2>/dev/null
|
|
echo "Pulled ${partner_slot} ledger from $partner_host ✅"
|
|
(( PULLED++ ))
|
|
else
|
|
rm -f "$CACHE_DIR/${partner_slot}.tokens.db.part"
|
|
warn "Could not pull ${partner_slot} ledger from $partner_host"
|
|
(( FAILED++ ))
|
|
fi
|
|
done
|
|
|
|
# Zero counts are omitted rather than printed. ${VAR:+...} keeps "0" because it is a non-empty
|
|
# string, and a summary that always ends "0 failed" is what teaches you to stop reading it.
|
|
_summary="AI token sync complete — pulled $PULLED"
|
|
(( OFFLINE > 0 )) && _summary+=", $OFFLINE unavailable"
|
|
(( FAILED > 0 )) && _summary+=", $FAILED failed"
|
|
info "$_summary"
|
|
|
|
# Only a partner that answered and then failed the transfer is worth an exit code. An absent
|
|
# partner is the normal state whenever a partner is not yet onboarded.
|
|
[[ "$FAILED" -gt 0 ]] && exit 1
|
|
exit 0
|