210 lines
8.8 KiB
Bash
Executable File
210 lines
8.8 KiB
Bash
Executable File
#!/bin/bash
|
|
# ==============================================================================================
|
|
# ━━━ AI Retrieval Query ━━━
|
|
# ==============================================================================================
|
|
#
|
|
# PURPOSE
|
|
# ==============================================================================================
|
|
# Answers questions about Varaverk from Varaverk's own documentation. Embeds the question,
|
|
# retrieves the closest chunks from the index AI/ai_index.sh built, and either prints them
|
|
# directly or passes them to the generation model as grounding context.
|
|
#
|
|
# ==============================================================================================
|
|
# OPERATIONAL MODEL
|
|
# ==============================================================================================
|
|
# Two modes over the same retrieval:
|
|
#
|
|
# --search print the matching chunks and their sources. No generation model involved,
|
|
# so it is fast and its output is verbatim repo text.
|
|
# (default) retrieve, then ask the generation model to answer strictly from what was
|
|
# retrieved, citing each claim.
|
|
#
|
|
# Retrieval is steered by question shape. A question about what prevents something is pushed
|
|
# toward OPERATIONAL SAFEGUARDS chunks, one about a variable toward CONFIGURATION, one asking
|
|
# why toward DESIGN PRINCIPLES. This is a score boost, not a filter — a heuristic must not be
|
|
# able to exclude the chunk that actually holds the answer.
|
|
#
|
|
# ==============================================================================================
|
|
# DESIGN PRINCIPLES
|
|
# ==============================================================================================
|
|
#
|
|
# Grounded Or Silent
|
|
# The prompt instructs the model to answer only from retrieved context and to say what is
|
|
# missing rather than fill the gap. This repo's conventions are frequently not the
|
|
# conventional ones, and a confident generic answer about rsync or Docker is worse here
|
|
# than no answer.
|
|
#
|
|
# Sources Are Always Shown
|
|
# Every answer prints the chunks it drew on. An answer that cannot be traced back to a file
|
|
# is not usable for changing anything.
|
|
#
|
|
# Search Is The Trustworthy Mode
|
|
# --search returns repo text with nothing generated. When an answer matters, use it.
|
|
#
|
|
# Read-Only
|
|
# Retrieves and answers. Nothing here writes conf, touches state, or runs another script.
|
|
#
|
|
# ==============================================================================================
|
|
# OPERATIONAL SAFEGUARDS
|
|
# ==============================================================================================
|
|
#
|
|
# Fail-Closed Gate
|
|
# Exits cleanly unless AI_ENABLED is exactly "true".
|
|
#
|
|
# No Root Required
|
|
# Reads the index and calls Ollama. Nothing it does needs privilege, so it does not ask for
|
|
# any — this is the one AI script an ordinary user should be able to run.
|
|
#
|
|
# Missing Index Is Reported, Not Built
|
|
# An absent index exits with guidance to run ai_index.sh. Building a corpus-wide index as a
|
|
# side effect of a question would turn a two-second query into a several-minute one.
|
|
#
|
|
# Reachability Pre-flight
|
|
# Probes Ollama with AI_CONNECT_TIMEOUT before embedding, so an unreachable endpoint fails
|
|
# immediately with a clear message.
|
|
#
|
|
# Bounded Generation
|
|
# The request is capped at AI_REQUEST_TIMEOUT. A wedged model cannot hang the caller.
|
|
#
|
|
# ==============================================================================================
|
|
# CONFIGURATION
|
|
# ==============================================================================================
|
|
#
|
|
# master.conf
|
|
#
|
|
# AI_ENABLED
|
|
# Master switch. Fail-closed — must be exactly "true".
|
|
#
|
|
# AI_INDEX_DB
|
|
# SQLite index to search. (shipped default: $DATA_DIR/ai_index.db)
|
|
#
|
|
# AI_SEARCH_K
|
|
# Chunks retrieved per query. (shipped default: 8)
|
|
#
|
|
# AI_SEARCH_PER_FILE
|
|
# Cap per file, so one document cannot fill the context. (shipped default: 3)
|
|
#
|
|
# AI_REQUEST_TIMEOUT
|
|
# Seconds allowed for generation. (shipped default: 240)
|
|
#
|
|
# host*.conf
|
|
#
|
|
# HOST*_OLLAMA_URL / HOST*_OLLAMA_MODEL / HOST*_OLLAMA_EMBED_MODEL
|
|
#
|
|
# ==============================================================================================
|
|
# RUNTIME MODES
|
|
# ==============================================================================================
|
|
#
|
|
# ai_query.sh "your question"
|
|
# Retrieve and answer, with sources.
|
|
#
|
|
# ai_query.sh --search "your question"
|
|
# Print matching chunks only. No generation.
|
|
#
|
|
# ai_query.sh --section=CONFIGURATION "your question"
|
|
# Restrict retrieval to one header section.
|
|
#
|
|
# ai_query.sh --kind=readme "what is Varaverk"
|
|
# Restrict retrieval to one chunk origin: header, readme, manual, template, doc.
|
|
# Use this for definitional and narrative questions. Intent routing boosts header
|
|
# sections such as PURPOSE, which answers "what does this script do" well but buries
|
|
# the top-level prose that explains what the system *is*. --kind=readme reaches it.
|
|
#
|
|
# ai_query.sh --json "your question"
|
|
# Machine-readable output for other scripts.
|
|
#
|
|
# ai_query.sh --status
|
|
# Show index and endpoint state.
|
|
#
|
|
# ==============================================================================================
|
|
|
|
set -uo pipefail
|
|
|
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
source "${SCRIPT_DIR}/../load_config.sh"
|
|
|
|
detect_hosts
|
|
|
|
SEARCH_ONLY=false; JSON=false; STATUS=false; SECTION=""; KIND=""; QUERY=""
|
|
for arg in "$@"; do
|
|
case "$arg" in
|
|
--search) SEARCH_ONLY=true ;;
|
|
--json) JSON=true ;;
|
|
--status) STATUS=true ;;
|
|
--section=*) SECTION="${arg#*=}" ;;
|
|
--kind=*) KIND="${arg#*=}" ;;
|
|
--*) echo "Unknown option: $arg" >&2; exit 1 ;;
|
|
*) QUERY="$arg" ;;
|
|
esac
|
|
done
|
|
|
|
# --kind is a hard filter on chunk origin; --section filters the header section within a
|
|
# chunk. They answer different questions and compose: --kind=readme --section=PURPOSE is
|
|
# meaningful. Validated here rather than in the CLI so a typo costs nothing — an unknown kind
|
|
# silently matches no rows, which reads as "the index has no answer" and is the most
|
|
# misleading failure this tool can produce.
|
|
if [[ -n "$KIND" ]]; then
|
|
case "$KIND" in
|
|
header|readme|manual|template|doc) ;;
|
|
*) echo "Unknown --kind=$KIND (expected: header, readme, manual, template, doc)" >&2
|
|
exit 1 ;;
|
|
esac
|
|
fi
|
|
|
|
CLI="${SCRIPT_DIR}/lib/cli.js"
|
|
DB="${AI_INDEX_DB:-${DATA_DIR}/ai_index.db}"
|
|
|
|
_url_var="${MY_ID}_OLLAMA_URL"
|
|
_gen_var="${MY_ID}_OLLAMA_MODEL"
|
|
_emb_var="${MY_ID}_OLLAMA_EMBED_MODEL"
|
|
OLLAMA_URL="${!_url_var:-}"
|
|
GEN_MODEL="${!_gen_var:-}"
|
|
EMBED_MODEL="${!_emb_var:-nomic-embed-text}"
|
|
|
|
if [[ "$STATUS" == true ]]; then
|
|
echo "$ICON_GEAR AI Query Status"
|
|
echo " Enabled: ${AI_ENABLED:-false}"
|
|
echo " Index: $DB $([[ -f "$DB" ]] && echo "($(sqlite3 "$DB" 'SELECT COUNT(*) FROM vv_chunks;' 2>/dev/null) chunks)" || echo '(not built)')"
|
|
echo " Ollama: ${OLLAMA_URL:-<none>}"
|
|
echo " Generate: ${GEN_MODEL:-<unset>}"
|
|
echo " Embed: $EMBED_MODEL"
|
|
exit 0
|
|
fi
|
|
|
|
if [[ "${AI_ENABLED:-false}" != "true" ]]; then
|
|
echo "AI_ENABLED is not true — AI features are off" >&2
|
|
exit 0
|
|
fi
|
|
|
|
[[ -z "$QUERY" ]] && { echo "usage: ai_query.sh [--search] [--section=NAME] [--kind=KIND] \"your question\"" >&2; exit 1; }
|
|
[[ -f "$DB" ]] || { error "No index at $DB — run AI/ai_index.sh first"; exit 1; }
|
|
[[ -f "$CLI" ]] || { error "missing $CLI"; exit 1; }
|
|
command -v node >/dev/null 2>&1 || { error "node not found"; exit 1; }
|
|
[[ -z "$OLLAMA_URL" ]] && { error "${MY_ID}_OLLAMA_URL is empty"; exit 1; }
|
|
|
|
if ! curl -sf --max-time "${AI_CONNECT_TIMEOUT:-5}" "${OLLAMA_URL%/}/api/tags" >/dev/null 2>&1; then
|
|
error "Ollama unreachable at $OLLAMA_URL"
|
|
exit 1
|
|
fi
|
|
|
|
_args=("--db=${DB}" "--url=${OLLAMA_URL}" "--query=${QUERY}")
|
|
[[ -n "$SECTION" ]] && _args+=("--section=${SECTION}")
|
|
[[ -n "$KIND" ]] && _args+=("--kind=${KIND}")
|
|
[[ "$JSON" == true ]] && _args+=(--json)
|
|
|
|
if [[ "$SEARCH_ONLY" == true ]]; then
|
|
node --no-warnings "$CLI" search "${_args[@]}" \
|
|
"--model=${EMBED_MODEL}" \
|
|
"--k=${AI_SEARCH_K:-8}" "--per-file=${AI_SEARCH_PER_FILE:-3}"
|
|
else
|
|
[[ -z "$GEN_MODEL" ]] && { error "${MY_ID}_OLLAMA_MODEL is empty — needed for generation"; exit 1; }
|
|
# Token accounting. Passed in rather than re-read in node, so the conf stays the shell's job
|
|
# and cli.js keeps taking everything it needs as arguments. Omitting either flag simply
|
|
# skips the row — the CLI must still work when called by hand outside this wrapper.
|
|
node --no-warnings "$CLI" ask "${_args[@]}" \
|
|
"--model=${GEN_MODEL}" "--embed-model=${EMBED_MODEL}" \
|
|
"--k=${AI_SEARCH_K:-8}" "--per-file=${AI_SEARCH_PER_FILE:-3}" \
|
|
"--timeout=$(( ${AI_REQUEST_TIMEOUT:-240} * 1000 ))" \
|
|
"--token-db=${AI_TOKEN_DB:-}" "--token-host=$(echo "$MY_ID" | tr '[:upper:]' '[:lower:]')"
|
|
fi
|