Files
Varaverk/AI/ai_token_sync.sh
T

212 lines
9.2 KiB
Bash
Executable File

#!/bin/bash
# ==============================================================================================
# ============================= AI Token Ledger Sync ===========================================
# ==============================================================================================
#
# PURPOSE
# ─────────────────────────────────────────────────────────────────────────────
# Pulls each partner's AI token ledger into a RAM cache at /tmp/.cache/vv/ai/, so the AI tab
# can report usage for the whole fleet instead of only the host the browser happens to be on.
#
# Each host records its own turns to data/ai_token_history.db and nothing syncs that file, so
# without this a host can only ever see its own totals. The tab is careful to say "not collected
# here" rather than 0 for a partner it cannot see; this script is what turns that into a number.
#
# Same trick as conf_sync.sh, and deliberately so — resolve the partner over Tailscale, scp one
# small file into a tmpfs cache, let the reader treat a missing file as "unknown".
#
# ==============================================================================================
# OPERATIONAL MODEL
# ==============================================================================================
#
# 1. Gates — PARTNERSHIP_ENABLED, AI_ENABLED, AI_TOKEN_SYNC_ENABLED
# 2. Per partner:
# a. Resolve their Tailscale IP
# b. Resolve their SCRIPTS_DIR from their varaverk.cfg (they may be in appdata mode)
# c. scp their data/ai_token_history.db → $AI_TOKEN_CACHE_DIR/<slot>.tokens.db
#
# Pull only, no push. conf_sync.sh pushes as well because a partner may be unable to reach us
# and still needs our credentials; nothing here is needed by anyone else, and a reader that
# fetches its own data controls its own freshness rather than depending on the partner's cron.
#
# ==============================================================================================
# OPERATIONAL SAFEGUARDS
# ==============================================================================================
#
# An unreachable partner is not a failure.
# HOST2 is expected to be down for long stretches during onboarding. A warn every four
# hours would train the operator to ignore this script's output, and the AI diagnostic
# path treats every log WARN as actionable. Unresolvable partners are counted and
# reported once at info level; only a partner that resolves and then fails to transfer
# is treated as an error.
#
# The cache is never written directly.
# scp lands on a .part file that is renamed into place, so a transfer interrupted halfway
# cannot leave the reader parsing half a ledger. A truncated final row would be skipped by
# the field-count check on the PHP side, but a torn file should not reach it at all.
#
# Nothing is ever written back to the partner.
# This script only reads. A bug here cannot corrupt a partner's accounting.
#
# The cache is tmpfs and deliberately not preserved.
# Unlike the conf cache there is no save/restore pair. Stale counters are worse than
# absent ones: absent reads as "not collected here", stale reads as fact.
#
# ==============================================================================================
# CONFIGURATION
# ==============================================================================================
#
# master.conf
#
# AI_ENABLED Whole AI subsystem gate
# AI_TOKEN_SYNC_ENABLED This script's own toggle (default: true)
# PARTNERSHIP_ENABLED Checked via require_partnership()
# SSH_KEY Key used for all partner ssh/scp operations
#
# load_config.sh
#
# AI_TOKEN_CACHE_DIR tmpfs directory the tab reads partner ledgers from
#
# host*.conf
#
# HOST* — hostnames used to build the partner list via detect_hosts()
#
# ==============================================================================================
# RUNTIME MODES
# ==============================================================================================
#
# ai_token_sync.sh Pull every reachable partner's ledger
# ai_token_sync.sh --dry-run Report what would be pulled, transfer nothing
# ai_token_sync.sh --log Verbose output
#
# ==============================================================================================
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/../load_config.sh"
parse_args "$@"
if [[ "$EUID" -ne 0 ]]; then
error "Must be run as root"
exit 1
fi
acquire_lock
detect_hosts
require_partnership
if [[ "${AI_ENABLED:-false}" != true ]]; then
log "AI_ENABLED=false — skipping"
exit 0
fi
if [[ "${AI_TOKEN_SYNC_ENABLED:-true}" == false ]]; then
log "AI_TOKEN_SYNC_ENABLED=false — skipping"
exit 0
fi
CACHE_DIR="$AI_TOKEN_CACHE_DIR"
SSH_TIMEOUT=10
# Mirrors conf_sync.sh — the remote may be in appdata storage mode, so its ledger is not
# necessarily under /boot.
_remote_scripts_dir() {
local ip="$1"
local cfg line sd
cfg=$(timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes -o StrictHostKeyChecking=no \
"root@${ip}" "cat /boot/config/plugins/varaverk/varaverk.cfg 2>/dev/null" 2>/dev/null) || true
while IFS= read -r line; do
[[ "$line" == SCRIPTS_DIR=* ]] || continue
sd="${line#SCRIPTS_DIR=}"; sd="${sd//\"/}"; sd="${sd//\'/}"
echo "$sd"; return
done <<< "$cfg"
echo "/boot/config/plugins/varaverk"
}
[[ "$DRY_RUN" == true ]] && warn "DRY RUN — no changes will be made"
if [[ "$DRY_RUN" == false ]]; then
mkdir -p "$CACHE_DIR" && chmod 755 "$CACHE_DIR"
fi
PULLED=0
OFFLINE=0
FAILED=0
for host_var in $(compgen -v | grep -E '^HOST[0-9]+$' | sort); do
partner_host="${!host_var}"
[[ -z "$partner_host" ]] && continue
[[ "${host_var,,}" == "${MY_ID,,}" ]] && continue
partner_slot="${host_var,,}"
partner_ip=$(resolve_tailscale_ip "$partner_host" 2>/dev/null || true)
if [[ -z "$partner_ip" ]]; then
log "$partner_host — unresolvable, leaving any cached ledger as-is"
(( OFFLINE++ ))
continue
fi
# Liveness and ledger presence are separate questions, probed separately on purpose.
# Tailscale hands back an IP for a peer that is registered but powered off, so resolving
# proves nothing. And a single `ssh test -f` answers both questions at once: it fails
# identically whether the host is down or the file is simply absent. Treating that one
# failure as "no ledger" would delete a perfectly good cached copy every time the partner
# blinked — turning "synced 3h ago" into "not collected here" on a transient.
if ! timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes -o StrictHostKeyChecking=no \
"root@${partner_ip}" true 2>/dev/null; then
log "$partner_host — not answering, keeping any cached ledger as-is"
(( OFFLINE++ ))
continue
fi
remote_sd=$(_remote_scripts_dir "$partner_ip")
remote_db="${remote_sd}/data/ai_token_history.db"
if [[ "$DRY_RUN" == true ]]; then
warn "DRY RUN — would pull $partner_host:$remote_db$CACHE_DIR/${partner_slot}.tokens.db"
continue
fi
# Reachable, but nothing recorded there — a partner with AI off, or one that has simply
# never been asked anything. Now that liveness is established this is a real answer, so the
# stale copy goes: the tab should say "not collected here", not quote a number from before.
if ! timeout "$SSH_TIMEOUT" ssh -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes -o StrictHostKeyChecking=no \
"root@${partner_ip}" "[[ -f '$remote_db' ]]" 2>/dev/null; then
log "$partner_host — reachable, but no ledger there yet"
rm -f "$CACHE_DIR/${partner_slot}.tokens.db"
(( OFFLINE++ ))
continue
fi
if timeout "$SSH_TIMEOUT" scp -i "$SSH_KEY" \
-o ConnectTimeout="$SSH_TIMEOUT" -o BatchMode=yes -o StrictHostKeyChecking=no \
"root@${partner_ip}:${remote_db}" \
"$CACHE_DIR/${partner_slot}.tokens.db.part" 2>/dev/null \
&& mv -f "$CACHE_DIR/${partner_slot}.tokens.db.part" "$CACHE_DIR/${partner_slot}.tokens.db"; then
chmod 644 "$CACHE_DIR/${partner_slot}.tokens.db" 2>/dev/null
echo "Pulled ${partner_slot} ledger from $partner_host ✅"
(( PULLED++ ))
else
rm -f "$CACHE_DIR/${partner_slot}.tokens.db.part"
warn "Could not pull ${partner_slot} ledger from $partner_host"
(( FAILED++ ))
fi
done
# Zero counts are omitted rather than printed. ${VAR:+...} keeps "0" because it is a non-empty
# string, and a summary that always ends "0 failed" is what teaches you to stop reading it.
_summary="AI token sync complete — pulled $PULLED"
(( OFFLINE > 0 )) && _summary+=", $OFFLINE unavailable"
(( FAILED > 0 )) && _summary+=", $FAILED failed"
info "$_summary"
# Only a partner that answered and then failed the transfer is worth an exit code. An absent
# partner is the normal state while HOST2 is being rebuilt.
[[ "$FAILED" -gt 0 ]] && exit 1
exit 0