Phase 02 (story: import documents):
- fence-aware markdown chunker (heading sections, 200-char overlap,
heading anchor on every chunk, 1200-char hard cap, fence blocks
kept atomic and split under the cap)
- LLMClient over aipi (LiteLLM) reusing the openai client's httpx
transport to send a clean {model, input} payload — the openai SDK
injects encoding_format, which aipi's openai_like group rejects;
token-budget batching + halving retry for the endpoint's
~1024-token per-request input cap
- two-phase per-file upsert importer: sha256 delta (unchanged skip),
atomic commit, A9 exclusion walk, per-source prune, per-file error
tolerance (rollback + log + continue, non-zero CLI exit), adaptive
re-chunk at half target for URL-dense files the endpoint rejects
- scripts/import_docs CLI (repeatable --source, --prune, --limit,
defaults ~/Homelab + ~/Deployments)
- GET /api/docs with per-doc chunk counts; Sources page wired to the
real endpoint (stat cards, full-width a11y table, designed empty
state, DOM-built rows — no innerHTML)
- tests: 63 passed (chunker/llm/importer units, docs API + importer
integration), story E2E 3/3 (real endpoints, in-thread import);
app/ coverage 98%
- real KB imported: 672 docs / 8969 chunks in ~3m, idempotent
re-run (672 unchanged, 0 batches)
- harness: .agent/validate.sh now gates through uv (pytest +
coverage >90% + ruff + pyright) instead of system python3
78 lines
2.1 KiB
JavaScript
78 lines
2.1 KiB
JavaScript
/* Brain of Reese — Sources page (knowledge base index view).
|
||
*
|
||
* Wires the real `GET /api/docs` endpoint (import phase): stat cards +
|
||
* full-width document table, or the designed empty state when nothing is
|
||
* indexed yet. Cells are built with DOM APIs (textContent) — never
|
||
* innerHTML with document-derived data (XSS-safe by construction).
|
||
*/
|
||
|
||
const tbody = document.querySelector("#docs-tbody");
|
||
const emptyEl = document.querySelector("#sources-empty");
|
||
const tableWrap = document.querySelector(".table-wrap");
|
||
const statDocs = document.querySelector("#stat-docs");
|
||
const statChunks = document.querySelector("#stat-chunks");
|
||
const statLast = document.querySelector("#stat-last");
|
||
|
||
function fmtDate(iso) {
|
||
try {
|
||
return new Date(iso).toLocaleString();
|
||
} catch {
|
||
return iso;
|
||
}
|
||
}
|
||
|
||
async function loadDocs() {
|
||
let r;
|
||
try {
|
||
r = await fetch("/api/docs");
|
||
} catch {
|
||
showEmpty();
|
||
return;
|
||
}
|
||
if (!r.ok) {
|
||
showEmpty();
|
||
return;
|
||
}
|
||
const { documents } = await r.json();
|
||
if (!documents.length) {
|
||
showEmpty();
|
||
return;
|
||
}
|
||
|
||
tbody.replaceChildren();
|
||
let totalChunks = 0;
|
||
let last = "";
|
||
for (const d of documents) {
|
||
totalChunks += d.chunks;
|
||
if (d.indexed_at > last) last = d.indexed_at;
|
||
tbody.appendChild(makeRow(d));
|
||
}
|
||
statDocs.textContent = String(documents.length);
|
||
statChunks.textContent = String(totalChunks);
|
||
statLast.textContent = last ? fmtDate(last) : "–";
|
||
emptyEl.hidden = true;
|
||
tableWrap.hidden = false;
|
||
}
|
||
|
||
function makeRow(d) {
|
||
const tr = document.createElement("tr");
|
||
const cells = [d.source, d.path, d.title, String(d.chunks), fmtDate(d.indexed_at)];
|
||
for (const value of cells) {
|
||
const td = document.createElement("td");
|
||
td.textContent = value; // document-derived text — never innerHTML
|
||
tr.appendChild(td);
|
||
}
|
||
tr.children[1].title = d.path; // full path on hover (column is ellipsized)
|
||
return tr;
|
||
}
|
||
|
||
function showEmpty() {
|
||
statDocs.textContent = "0";
|
||
statChunks.textContent = "0";
|
||
statLast.textContent = "–";
|
||
emptyEl.hidden = false;
|
||
if (tableWrap) tableWrap.hidden = true;
|
||
}
|
||
|
||
loadDocs();
|