Files
Varaverk/AI/lib/search.js
T
Gmer4Lfe ec3084db12 Answer "what is Varaverk" with Varaverk's own README
The PURPOSE intent cannot tell the project from a script inside it, so asking
about the whole promoted every component's summary at once.
2026-08-12 19:23:44 -04:00

140 lines
6.8 KiB
JavaScript

'use strict';
// ═══════════════════════════════════════════════════════════════════════════════════════════════
// Search — embed a question, score it against the index, return the best chunks.
//
// nomic-embed-text returns L2-normalised vectors, so cosine similarity is a plain dot product.
// At this corpus size (~1.7k chunks x 768 dims) that is a couple of million multiply-adds —
// under a millisecond, with no vector database and no index structure to maintain.
//
// Section routing is the payoff from the header audit. Every chunk knows whether it is a
// PURPOSE, a DESIGN PRINCIPLES, an OPERATIONAL SAFEGUARDS and so on, so a question's shape can
// steer retrieval before similarity is even considered. It is applied as a score boost rather
// than a hard filter — intent detection is a heuristic, and a heuristic should not be able to
// exclude the one chunk that actually holds the answer.
// ═══════════════════════════════════════════════════════════════════════════════════════════════
const { DatabaseSync } = require('node:sqlite');
// Question shape → the section most likely to answer it.
const INTENTS = [
{ section: 'OPERATIONAL SAFEGUARDS',
re: /\b(safe|safety|guard|protect|prevent|fail|failure|abort|refuse|lock|root|timeout|dry.?run|what stops|what happens if|race|corrupt|data.?loss)\b/i },
{ section: 'CONFIGURATION',
re: /\b(variable|var|setting|conf|config|threshold|toggle|which key|what controls|where is .* set|default value|env)\b/i },
{ section: 'RUNTIME MODES',
re: /\b(flag|argument|option|--\w+|how do i run|invoke|cli|command line|status mode|usage)\b/i },
{ section: 'DESIGN PRINCIPLES',
re: /\b(why|rationale|reason|design|decision|deliberate|intentional|on purpose|trade.?off|chose|approach)\b/i },
{ section: 'OPERATIONAL MODEL',
re: /\b(how does .* work|flow|sequence|order|tier|lifecycle|state machine|when does)\b/i },
{ section: 'EXPORTS',
re: /\b(function|export|api surface|what does .* provide|helper|vv_\w+)\b/i },
{ section: 'PURPOSE',
re: /\b(what is|what does .* do|purpose|responsible for|job of)\b/i },
];
const SECTION_BOOST = 0.06; // enough to reorder near-ties, not enough to beat a real match
const KIND_BOOST = 0.02; // docs answer "how do I" better than a script header does
function detectIntent(q) {
const hits = [];
for (const i of INTENTS) if (i.re.test(q)) hits.push(i.section);
return hits;
}
function blobToVec(buf) {
const b = Buffer.from(buf);
return new Float32Array(b.buffer, b.byteOffset, b.length / 4);
}
function dot(a, b) {
let s = 0;
for (let i = 0; i < a.length; i++) s += a[i] * b[i];
return s;
}
async function embedQuery(url, model, text, timeoutMs = 60000) {
const res = await fetch(`${url.replace(/\/$/, '')}/api/embed`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ model, input: text }),
signal: AbortSignal.timeout(timeoutMs),
});
if (!res.ok) throw new Error(`embed HTTP ${res.status}`);
const j = await res.json();
if (!j.embeddings || !j.embeddings[0]) throw new Error('embed returned no vector');
return Float32Array.from(j.embeddings[0]);
}
// Keep at most `perFile` chunks from any one file, so a single large document cannot fill the
// entire context window and crowd out a better answer living somewhere else.
function diversify(rows, k, perFile) {
const seen = new Map();
const out = [];
for (const r of rows) {
const n = seen.get(r.path) || 0;
if (n >= perFile) continue;
seen.set(r.path, n + 1);
out.push(r);
if (out.length >= k) break;
}
return out;
}
async function search(opts) {
const {
dbPath, url, embedModel, query,
k = 8, perFile = 3, section = null, kind = null, minScore = 0.0,
} = opts;
const db = new DatabaseSync(dbPath, { readOnly: true });
let sql = 'SELECT id,path,kind,section,heading,part,content,vector FROM vv_chunks';
const where = [], args = [];
if (section) { where.push('section = ?'); args.push(section); }
if (kind) { where.push('kind = ?'); args.push(kind); }
if (where.length) sql += ' WHERE ' + where.join(' AND ');
const rows = db.prepare(sql).all(...args);
if (!rows.length) { db.close(); return { results: [], intents: [], scanned: 0 }; }
const qv = await embedQuery(url, embedModel, query);
let intents = section ? [] : detectIntent(query);
// "What is Varaverk" and "what is arr_sync.sh" are not the same question, and the PURPOSE
// intent cannot tell them apart — it fires on both and boosts every PURPOSE block in the
// repository at once. There are a couple of hundred, each genuinely describing the purpose of
// something, and each a short sentence containing the word Varaverk. The project's own README
// then loses to a script that migrates storage modes, because a paragraph is more diluted
// than a one-line summary.
//
// A question that names the project and no component inside it is asking about the whole, so
// PURPOSE is precisely the wrong section to promote. Dropping only that intent, rather than
// all of them, leaves "why was Varaverk built this way" still routed to DESIGN PRINCIPLES.
const namesProject = /\bvaraverk\b/i.test(query);
const namesComponent = /\b[\w.-]+\.(sh|php|js)\b|\b[A-Z][A-Z0-9]*(_[A-Z0-9]+)+\b/.test(query);
const projectLevel = namesProject && !namesComponent;
if (projectLevel) intents = intents.filter(s => s !== 'PURPOSE');
// The same question wants the top-level prose, which is what the doc kinds are.
const wantDoc = projectLevel
|| /\b(how do i|steps|procedure|setup|install|troubleshoot|guide)\b/i.test(query);
const scored = rows.map(r => {
let s = dot(qv, blobToVec(r.vector));
if (intents.includes(r.section)) s += SECTION_BOOST;
if (wantDoc && (r.kind === 'manual' || r.kind === 'readme')) s += KIND_BOOST;
return {
id: r.id, path: r.path, kind: r.kind, section: r.section,
heading: r.heading, part: r.part, content: r.content, score: s,
};
});
scored.sort((a, b) => b.score - a.score);
const kept = diversify(scored.filter(r => r.score >= minScore), k, perFile);
db.close();
return { results: kept, intents, scanned: rows.length };
}
module.exports = { search, detectIntent, blobToVec, dot, embedQuery, INTENTS };