Add config-vs-reality health checks, loaded models, and log evidence

The failures this subsystem actually has are configuration drift, so each
check names the setting to change rather than reporting that retrieval
failed. Notably it catches a conf model tag that is no longer installed, and
an index built by a different embedder than the one configured — vectors
from two models are not comparable, and that failure returns confident
nonsense rather than erroring. Diagnostic questions also get recent log
warnings, attached only then because they cost budget the passages need.
This commit is contained in:
Gmer4Lfe
2026-08-02 17:40:43 -04:00
parent 264ba57cbb
commit 237156c46f
3 changed files with 322 additions and 7 deletions
+199 -1
View File
@@ -29,6 +29,17 @@
// The index only ever contains git-tracked content, so comparing to an untracked scratch
// file would report permanent staleness. git ls-files is the same source the indexer uses.
//
// Health checks compare configuration against reality, and carry their remedy.
// "Retrieval failed" is not actionable; "HOST1_OLLAMA_MODEL names a tag that is not
// installed" is. Every check names the setting to change, because the failures this
// subsystem actually has are configuration drift — a model pulled, wired into conf, then
// removed — rather than software faults.
//
// Logs are read live, never indexed.
// They churn constantly, would dominate a 3341-chunk corpus, and embedding similarity
// retrieves log lines poorly compared with recency and a severity filter. Indexing them
// would also mean re-embedding every few minutes for content that is stale immediately.
//
// OPERATIONAL SAFEGUARDS
// Read-only. Nothing here writes to the index, the conf, or the job files — it reads state
// and runs a retrieval query. The worker owns every write.
@@ -53,7 +64,9 @@
//
// EXPORTS
// Config vv_ai_enabled(), vv_ai_config()
// Status vv_ai_stats(), vv_ai_index_stats(), vv_ai_runtime_stats()
// Status vv_ai_stats(), vv_ai_index_stats(), vv_ai_runtime_stats(), vv_ai_index_meta()
// Models vv_ai_models_available(), vv_ai_models_loaded()
// Diagnosis vv_ai_health(), vv_ai_recent_logs()
// Retrieval vv_ai_retrieve()
// Jobs vv_ai_job_dir(), vv_ai_job_path(), vv_ai_job_read()
//
@@ -178,6 +191,189 @@ function vv_ai_runtime_stats(): array {
return $out;
}
// Models Ollama actually has on disk. The configured model naming a tag that does not exist is
// the single most likely misconfiguration here — it is what happens whenever a model is pulled,
// wired into conf, and later removed.
function vv_ai_models_available(): ?array {
$cfg = vv_ai_config();
if ($cfg['url'] === '') return null;
$ctx = stream_context_create(['http' => ['timeout' => max(2, $cfg['connect']), 'ignore_errors' => true]]);
$raw = @file_get_contents($cfg['url'] . '/api/tags', false, $ctx);
if ($raw === false) return null;
$d = json_decode($raw, true);
if (!isset($d['models'])) return null;
return array_map(fn($m) => [
'name' => $m['name'] ?? '',
'size' => (int)($m['size'] ?? 0),
], $d['models']);
}
// Currently resident models with their offload split.
function vv_ai_models_loaded(): ?array {
$cfg = vv_ai_config();
if ($cfg['url'] === '') return null;
$ctx = stream_context_create(['http' => ['timeout' => max(2, $cfg['connect']), 'ignore_errors' => true]]);
$raw = @file_get_contents($cfg['url'] . '/api/ps', false, $ctx);
if ($raw === false) return null;
$d = json_decode($raw, true);
if (!isset($d['models'])) return null;
$out = [];
foreach ($d['models'] as $m) {
$size = (int)($m['size'] ?? 0);
$vram = (int)($m['size_vram'] ?? 0);
$out[] = [
'name' => $m['name'] ?? '',
'context' => $m['context_length'] ?? null,
'size' => $size,
'vram' => $vram,
'offload' => $size > 0 ? (int)round($vram / $size * 100) : null,
];
}
return $out;
}
// vv_meta records the embed model and dimensionality the index was built with.
function vv_ai_index_meta(): array {
$cfg = vv_ai_config();
if (!file_exists($cfg['db'])) return [];
$raw = trim((string)@shell_exec(
'sqlite3 ' . escapeshellarg($cfg['db']) . ' ' . escapeshellarg('SELECT key, value FROM vv_meta;') . ' 2>/dev/null'
));
$out = [];
foreach (explode("\n", $raw) as $line) {
if ($line === '') continue;
[$k, $v] = array_pad(explode('|', $line, 2), 2, '');
if ($k !== '') $out[$k] = $v;
}
return $out;
}
// Config checked against reality. Each check is ok | warn | bad, with the remedy attached —
// the point is to name the setting that is wrong, not merely report that something failed.
function vv_ai_health(): array {
$cfg = vv_ai_config();
$checks = [];
$add = function (string $id, string $label, string $state, string $detail, string $fix = '')
use (&$checks) {
$checks[] = ['id' => $id, 'label' => $label, 'state' => $state,
'detail' => $detail, 'fix' => $fix];
};
$add('enabled', 'AI enabled', $cfg['enabled'] ? 'ok' : 'bad',
$cfg['enabled'] ? 'AI_ENABLED=true' : 'AI_ENABLED is false',
$cfg['enabled'] ? '' : 'Set AI_ENABLED=true in master.conf');
if (!command_exists_node()) {
$add('node', 'node runtime', 'bad', 'node not found on PATH',
'Retrieval shells to AI/lib/cli.js and cannot run without it');
}
if ($cfg['url'] === '') {
$add('url', 'Ollama URL', 'bad', 'not configured',
'Set ' . strtoupper(vv_detect_host()) . '_OLLAMA_URL in the host conf');
return $checks;
}
$avail = vv_ai_models_available();
if ($avail === null) {
$add('reach', 'Ollama reachable', 'bad', 'no response from ' . $cfg['url'],
'Check the Ollama container is running and the URL is correct');
return $checks;
}
$add('reach', 'Ollama reachable', 'ok', $cfg['url']);
$names = array_column($avail, 'name');
// The configured tag not existing is the failure this whole section is for.
if ($cfg['model'] === '') {
$add('gen', 'Generation model', 'bad', 'not configured',
'Set ' . strtoupper(vv_detect_host()) . '_OLLAMA_MODEL in the host conf');
} elseif (!in_array($cfg['model'], $names, true)) {
$add('gen', 'Generation model', 'bad', $cfg['model'] . ' is not installed',
'Either `ollama pull` it, or point _OLLAMA_MODEL at one of: ' . implode(', ', array_slice($names, 0, 4)));
} else {
$add('gen', 'Generation model', 'ok', $cfg['model']);
}
if ($cfg['embed_model'] === '' || !in_array($cfg['embed_model'], $names, true)) {
$add('embed', 'Embedding model', 'bad',
($cfg['embed_model'] ?: 'not configured') . ' is not installed',
'Retrieval cannot embed a query without it — `ollama pull ' . ($cfg['embed_model'] ?: 'nomic-embed-text') . '`');
} else {
$add('embed', 'Embedding model', 'ok', $cfg['embed_model']);
}
$ix = vv_ai_index_stats();
$meta = vv_ai_index_meta();
if (!$ix['exists']) {
$add('index', 'Index', 'bad', 'not built', 'Run AI/ai_index.sh');
} else {
$add('index', 'Index', 'ok', number_format($ix['chunks']) . ' chunks from ' . $ix['files'] . ' files');
// A vector built by one embedding model is meaningless to another. Changing the embed
// model without reindexing does not error — it silently returns nonsense, scored
// confidently, which is the hardest failure here to notice from the answers alone.
$builtWith = $meta['embed_model'] ?? '';
if ($builtWith !== '' && $cfg['embed_model'] !== '' && $builtWith !== $cfg['embed_model']) {
$add('embedmatch', 'Index / embedder match', 'bad',
'index built with ' . $builtWith . ', conf says ' . $cfg['embed_model'],
'Vectors from different models are not comparable — rerun AI/ai_index.sh --force');
} else {
$add('embedmatch', 'Index / embedder match', 'ok', $builtWith ?: 'unrecorded');
}
if ($ix['stale'] === true) {
$add('fresh', 'Index freshness', 'warn', 'tracked files are newer than the index',
'Answers may cite code that has changed — rerun AI/ai_index.sh');
} else {
$add('fresh', 'Index freshness', 'ok', 'current');
}
}
$loaded = vv_ai_models_loaded();
if ($loaded !== null && $cfg['model'] !== '') {
$hit = null;
foreach ($loaded as $m) if ($m['name'] === $cfg['model']) { $hit = $m; break; }
if ($hit === null) {
$add('resident', 'Model resident', 'warn', 'not loaded — first question will load it');
} elseif ($hit['offload'] !== null && $hit['offload'] < 100) {
$add('resident', 'GPU offload', 'bad', $hit['offload'] . '% on GPU — layers on CPU',
'Roughly 4x slower on this card. Lower num_ctx or use a smaller quant');
} else {
$add('resident', 'GPU offload', 'ok', '100% on GPU at ' . ($hit['context'] ?? '?') . ' ctx');
}
}
return $checks;
}
function command_exists_node(): bool {
static $has = null;
if ($has !== null) return $has;
return $has = trim((string)@shell_exec('command -v node 2>/dev/null')) !== '';
}
// Recent warnings and errors across the orchestrator logs. Read live rather than indexed:
// logs churn constantly, would dominate a 3341-chunk index, and embedding similarity retrieves
// them poorly compared with recency plus a severity filter.
function vv_ai_recent_logs(int $max = 40): array {
$dir = '/var/log/varaverk';
if (!is_dir($dir)) return [];
$cmd = 'grep -rhE "\[(ERROR|WARN|CRITICAL|FAILED)\]|✗" ' . escapeshellarg($dir)
. ' --include="*.log" 2>/dev/null | tail -' . max(1, min($max, 200));
$raw = (string)@shell_exec($cmd);
$out = [];
foreach (explode("\n", $raw) as $line) {
$line = trim(preg_replace('/\033\[[0-9;]*[mK]/', '', $line));
if ($line === '') continue;
$out[] = mb_substr($line, 0, 300);
}
return $out;
}
function vv_ai_stats(): array {
$cfg = vv_ai_config();
return [
@@ -188,6 +384,8 @@ function vv_ai_stats(): array {
'k' => $cfg['k'],
'index' => vv_ai_index_stats(),
'runtime' => vv_ai_runtime_stats(),
'loaded' => vv_ai_models_loaded(),
'health' => vv_ai_health(),
'ts' => time(),
];
}