Three # rules open most docs here, so the title became a section too small to survive and the paragraph defining the project got a chunk with no heading.
277 lines
13 KiB
JavaScript
277 lines
13 KiB
JavaScript
'use strict';
|
|
// ═══════════════════════════════════════════════════════════════════════════════════════════════
|
|
// Chunker — turns repo files into retrieval units.
|
|
//
|
|
// The whole point of the header audit is that chunk boundaries are deterministic here. Bash
|
|
// scripts split on their six section names, markdown on its headings, conf templates on their
|
|
// ━━━ section rules. Nothing is split on a fixed token window, so no chunk ever contains half
|
|
// of one idea and half of another.
|
|
//
|
|
// Every chunk carries its section name as its own field, because that is the metadata that
|
|
// lets retrieval filter by question shape before it ever computes similarity.
|
|
// ═══════════════════════════════════════════════════════════════════════════════════════════════
|
|
|
|
const fs = require('fs');
|
|
const path = require('path');
|
|
|
|
const BASH_SECTIONS = [
|
|
'PURPOSE', 'OPERATIONAL MODEL', 'DESIGN PRINCIPLES',
|
|
'OPERATIONAL SAFEGUARDS', 'CONFIGURATION', 'RUNTIME MODES',
|
|
];
|
|
|
|
// PHP headers reuse the first three names deliberately, then diverge per layer.
|
|
const PHP_SECTIONS = [
|
|
'PURPOSE', 'OPERATIONAL MODEL', 'DESIGN PRINCIPLES', 'OPERATIONAL SAFEGUARDS',
|
|
'STATUS', 'EXPORTS', 'REQUEST CONTRACT', 'SIDE EFFECTS', 'RENDERS', 'DEPENDS ON',
|
|
'CONFIGURATION',
|
|
];
|
|
|
|
const MIN_CHARS = 40; // below this a chunk carries no retrievable meaning
|
|
const MAX_CHARS = 6000; // above this, split on blank lines — protects the embed window
|
|
|
|
// Banner rules and box-drawing art are everywhere in this repo's headers. They carry no
|
|
// meaning to embed, and a chunk that is mostly rule characters is pure noise in the index.
|
|
// Measure a chunk by what is left after the decoration is removed, not by raw length.
|
|
function meaningful(s) {
|
|
return s.replace(/[═─━=_#\/*\s|+.-]/g, '').length;
|
|
}
|
|
const MIN_MEANINGFUL = 30;
|
|
|
|
function isRealHeading(h) {
|
|
return !!h && /[A-Za-z0-9]/.test(h.replace(/[═─━=_]/g, ''));
|
|
}
|
|
|
|
function stripPrefix(line, prefix) {
|
|
// '# text' -> 'text' '// text' -> 'text'
|
|
const re = new RegExp('^\\s*' + prefix + '\\s?');
|
|
return line.replace(re, '');
|
|
}
|
|
|
|
// ── Comment-header sectioning, shared by bash (#) and PHP (//) ────────────────────────────────
|
|
function sectionsFromCommentHeader(text, prefix, names) {
|
|
const lines = text.split('\n');
|
|
const nameSet = new Set(names);
|
|
const found = [];
|
|
|
|
// headerEnd matters as much as the section starts. The last section (RUNTIME MODES in bash,
|
|
// DEPENDS ON in a page) would otherwise run to EOF and sweep up every unrelated comment in
|
|
// the file — scheduler.php alone contributed an 11k-char chunk of unrelated inline comments.
|
|
let headerEnd = lines.length;
|
|
for (let i = 0; i < lines.length; i++) {
|
|
const raw = lines[i];
|
|
if (!new RegExp('^\\s*' + prefix).test(raw)) {
|
|
// Header block ends at the first non-comment, non-blank line past the shebang.
|
|
// '<?php' and '?>' bracket a PHP header block and are not the end of it.
|
|
const t = raw.trim();
|
|
if (t !== '' && !/^#!/.test(t) && t !== '<?php' && t !== '?>' && found.length) {
|
|
headerEnd = i;
|
|
break;
|
|
}
|
|
continue;
|
|
}
|
|
const inner = stripPrefix(raw, prefix).trim();
|
|
if (nameSet.has(inner)) found.push({ name: inner, start: i });
|
|
}
|
|
|
|
const out = [];
|
|
for (let k = 0; k < found.length; k++) {
|
|
const start = found[k].start + 1;
|
|
const end = k + 1 < found.length ? found[k + 1].start : headerEnd;
|
|
const body = lines.slice(start, end)
|
|
.filter(l => new RegExp('^\\s*' + prefix).test(l))
|
|
.map(l => stripPrefix(l, prefix))
|
|
// drop pure separator rules (════, ────, ━━━) — they carry no meaning
|
|
.filter(l => !/^[\s═─━=_-]*$/.test(l) || l.trim() === '')
|
|
.join('\n')
|
|
.replace(/\n{3,}/g, '\n\n')
|
|
.trim();
|
|
if (meaningful(body) >= MIN_MEANINGFUL) {
|
|
for (const p of splitNamedParagraphs(body))
|
|
out.push({ section: found[k].name, title: p.title, content: p.content });
|
|
}
|
|
}
|
|
return out;
|
|
}
|
|
|
|
// ── Named-paragraph sub-chunking ──────────────────────────────────────────────────────────────
|
|
// The header convention writes safeguards and principles as named paragraphs: an unindented
|
|
// title line followed by an indented body. Embedding a whole section as one unit dilutes them —
|
|
// rsync.sh's OPERATIONAL SAFEGUARDS holds eight distinct guarantees in 2.8k chars, and a query
|
|
// about one of them scored below unrelated chunks because the other seven dominated the vector.
|
|
// Splitting on the title lines is what makes a specific question find a specific answer.
|
|
//
|
|
// The section name is carried onto every sub-chunk, so section routing still works; the
|
|
// paragraph title becomes the chunk's heading.
|
|
function splitNamedParagraphs(body) {
|
|
const lines = body.split('\n');
|
|
const marks = [];
|
|
for (let i = 0; i < lines.length; i++) {
|
|
const l = lines[i];
|
|
if (!l.trim()) continue;
|
|
if (/^\s/.test(l)) continue; // indented => body, not a title
|
|
if (/^[-*•]/.test(l.trim())) continue; // list item, not a title
|
|
if (l.trim().length > 80) continue; // a long line is prose, not a heading
|
|
if (/[.:;,]$/.test(l.trim())) continue; // ends like a sentence
|
|
|
|
// The decisive signal: a real title is followed by an indented body. Wrapped prose is
|
|
// followed by more unindented prose. Without this check, any short line in a paragraph
|
|
// that happened to wrap became a spurious chunk boundary mid-sentence.
|
|
let j = i + 1;
|
|
while (j < lines.length && !lines[j].trim()) j++;
|
|
if (j >= lines.length || !/^\s+\S/.test(lines[j])) continue;
|
|
|
|
marks.push(i);
|
|
}
|
|
// Fewer than two titles means this section is not written as named paragraphs — keep it whole.
|
|
if (marks.length < 2) return [{ title: null, content: body }];
|
|
|
|
const out = [];
|
|
if (marks[0] > 0) {
|
|
const pre = lines.slice(0, marks[0]).join('\n').trim();
|
|
if (meaningful(pre) >= MIN_MEANINGFUL) out.push({ title: null, content: pre });
|
|
}
|
|
for (let k = 0; k < marks.length; k++) {
|
|
const start = marks[k];
|
|
const end = k + 1 < marks.length ? marks[k + 1] : lines.length;
|
|
const title = lines[start].trim();
|
|
const content = lines.slice(start, end).join('\n').trim();
|
|
if (meaningful(content) >= MIN_MEANINGFUL) out.push({ title, content });
|
|
}
|
|
return out;
|
|
}
|
|
|
|
// ── Markdown: split on ## headings, keep the heading with its body ─────────────────────────────
|
|
function sectionsFromMarkdown(text) {
|
|
const lines = text.split('\n');
|
|
const marks = [];
|
|
let fence = false;
|
|
|
|
for (let i = 0; i < lines.length; i++) {
|
|
if (/^\s*```/.test(lines[i])) { fence = !fence; continue; }
|
|
if (fence) continue;
|
|
// A heading whose text is nothing but rule characters is a banner, not a section. The
|
|
// house style opens a document with three of them —
|
|
// # ━━━━━━━━
|
|
// # 🏠 VARAVERK
|
|
// # ━━━━━━━━
|
|
// — and treating each as a boundary split the title into a section of its own, too small
|
|
// to survive, then gave the paragraph that actually defines the project a chunk headed by
|
|
// the rule beneath it: no heading, and content opening with 75 identical glyphs. That is
|
|
// why "what is Varaverk" returned five script PURPOSE headers and never README.md, which
|
|
// has been indexed the whole time. 35 of the 40 markdown files here open this way.
|
|
//
|
|
// Whole-string test, so an ordinary heading containing a dash is unaffected — "Set-up"
|
|
// does not reduce to empty, and a line of dashes does.
|
|
if (!/^#{1,3}\s+\S/.test(lines[i])) continue;
|
|
if (lines[i].replace(/^#+\s*/, '').replace(/[━─═=~_*\-\s]+/gu, '') === '') continue;
|
|
marks.push(i);
|
|
}
|
|
if (!marks.length) return [{ heading: null, content: text.trim() }];
|
|
|
|
const out = [];
|
|
// preamble before the first heading
|
|
if (marks[0] > 0) {
|
|
const pre = lines.slice(0, marks[0]).join('\n').trim();
|
|
if (meaningful(pre) >= MIN_MEANINGFUL) out.push({ heading: null, content: pre });
|
|
}
|
|
for (let k = 0; k < marks.length; k++) {
|
|
const start = marks[k];
|
|
const end = k + 1 < marks.length ? marks[k + 1] : lines.length;
|
|
const heading = lines[start].replace(/^#+\s*/, '').replace(/[━─═]+/g, '').trim();
|
|
const content = lines.slice(start, end).join('\n').trim();
|
|
if (meaningful(content) >= MIN_MEANINGFUL) out.push({ heading: isRealHeading(heading) ? heading : null, content });
|
|
}
|
|
return out;
|
|
}
|
|
|
|
// ── Conf templates: split on the ━━━ / ── section rules ───────────────────────────────────────
|
|
function sectionsFromConfTemplate(text) {
|
|
const lines = text.split('\n');
|
|
const marks = [];
|
|
for (let i = 0; i < lines.length; i++) {
|
|
const m = lines[i].match(/^#\s*[━─]{2,}\s*(.+?)\s*[━─]{2,}\s*$/);
|
|
if (m && isRealHeading(m[1])) marks.push({ i, name: m[1].trim() });
|
|
}
|
|
if (!marks.length) return [];
|
|
|
|
const out = [];
|
|
for (let k = 0; k < marks.length; k++) {
|
|
const start = marks[k].i;
|
|
const end = k + 1 < marks.length ? marks[k + 1].i : lines.length;
|
|
const content = lines.slice(start, end).join('\n').replace(/\n{3,}/g, '\n\n').trim();
|
|
if (meaningful(content) >= MIN_MEANINGFUL) out.push({ heading: marks[k].name, content });
|
|
}
|
|
return out;
|
|
}
|
|
|
|
// Oversized chunks split on blank lines rather than mid-sentence.
|
|
function capSize(chunks) {
|
|
const out = [];
|
|
for (const c of chunks) {
|
|
if (c.content.length <= MAX_CHARS) { out.push(c); continue; }
|
|
const paras = c.content.split(/\n\s*\n/);
|
|
let buf = [], len = 0, part = 1;
|
|
const flush = () => {
|
|
if (!buf.length) return;
|
|
out.push({ ...c, content: buf.join('\n\n'), part: part++ });
|
|
buf = []; len = 0;
|
|
};
|
|
for (const p of paras) {
|
|
if (len + p.length > MAX_CHARS && buf.length) flush();
|
|
buf.push(p); len += p.length + 2;
|
|
}
|
|
flush();
|
|
}
|
|
return out;
|
|
}
|
|
|
|
function classify(rel) {
|
|
const base = path.basename(rel);
|
|
if (rel.startsWith('Deployment/') && rel.endsWith('.template')) return 'template';
|
|
// WebGUI page docs, written for whoever is using the tab rather than maintaining it. Their
|
|
// own kind because every other kind here answers a maintainer's question: an operator asking
|
|
// "how do I stop this" needs the click path, and a corpus that is three-quarters script
|
|
// headers will otherwise always answer in conf edits. Matched on the folder, not the
|
|
// filename, so these can be named whatever reads best.
|
|
if (rel.startsWith('Plugin/unraid/pages/readme/') && base.endsWith('.md')) return 'ui';
|
|
if (base.endsWith('.md')) {
|
|
if (base.startsWith('Manual')) return 'manual';
|
|
if (base.startsWith('README') || base === 'README.md') return 'readme';
|
|
return 'doc';
|
|
}
|
|
if (base.endsWith('.sh')) return 'header';
|
|
if (base.endsWith('.php')) return 'header';
|
|
return 'other';
|
|
}
|
|
|
|
function chunkFile(absPath, rel) {
|
|
const text = fs.readFileSync(absPath, 'utf8');
|
|
const kind = classify(rel);
|
|
let raw = [];
|
|
|
|
if (kind === 'header' && rel.endsWith('.sh')) {
|
|
raw = sectionsFromCommentHeader(text, '#', BASH_SECTIONS)
|
|
.map(s => ({ section: s.section, heading: s.title || null, content: s.content }));
|
|
} else if (kind === 'header' && rel.endsWith('.php')) {
|
|
raw = sectionsFromCommentHeader(text, '//', PHP_SECTIONS)
|
|
.map(s => ({ section: s.section, heading: s.title || null, content: s.content }));
|
|
} else if (kind === 'template') {
|
|
raw = sectionsFromConfTemplate(text)
|
|
.map(s => ({ section: null, heading: s.heading, content: s.content }));
|
|
} else if (kind === 'readme' || kind === 'manual' || kind === 'doc' || kind === 'ui') {
|
|
raw = sectionsFromMarkdown(text)
|
|
.map(s => ({ section: null, heading: s.heading, content: s.content }));
|
|
}
|
|
|
|
return capSize(raw).map(c => ({
|
|
path: rel,
|
|
kind,
|
|
section: c.section || null,
|
|
heading: c.heading || null,
|
|
part: c.part || null,
|
|
content: c.content,
|
|
}));
|
|
}
|
|
|
|
module.exports = { chunkFile, classify, BASH_SECTIONS, PHP_SECTIONS };
|