Files
brain-of-reese/frontend/assets/sources.js
T
ducoterra 99c48cbe06 feat(rag): index markdown KB — chunker, embed client, delta importer, Sources page
Phase 02 (story: import documents):

- fence-aware markdown chunker (heading sections, 200-char overlap,
  heading anchor on every chunk, 1200-char hard cap, fence blocks
  kept atomic and split under the cap)
- LLMClient over aipi (LiteLLM) reusing the openai client's httpx
  transport to send a clean {model, input} payload — the openai SDK
  injects encoding_format, which aipi's openai_like group rejects;
  token-budget batching + halving retry for the endpoint's
  ~1024-token per-request input cap
- two-phase per-file upsert importer: sha256 delta (unchanged skip),
  atomic commit, A9 exclusion walk, per-source prune, per-file error
  tolerance (rollback + log + continue, non-zero CLI exit), adaptive
  re-chunk at half target for URL-dense files the endpoint rejects
- scripts/import_docs CLI (repeatable --source, --prune, --limit,
  defaults ~/Homelab + ~/Deployments)
- GET /api/docs with per-doc chunk counts; Sources page wired to the
  real endpoint (stat cards, full-width a11y table, designed empty
  state, DOM-built rows — no innerHTML)
- tests: 63 passed (chunker/llm/importer units, docs API + importer
  integration), story E2E 3/3 (real endpoints, in-thread import);
  app/ coverage 98%
- real KB imported: 672 docs / 8969 chunks in ~3m, idempotent
  re-run (672 unchanged, 0 batches)
- harness: .agent/validate.sh now gates through uv (pytest +
  coverage >90% + ruff + pyright) instead of system python3
2026-08-21 16:24:45 -04:00

78 lines
2.1 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/* Brain of Reese — Sources page (knowledge base index view).
*
* Wires the real `GET /api/docs` endpoint (import phase): stat cards +
* full-width document table, or the designed empty state when nothing is
* indexed yet. Cells are built with DOM APIs (textContent) — never
* innerHTML with document-derived data (XSS-safe by construction).
*/
const tbody = document.querySelector("#docs-tbody");
const emptyEl = document.querySelector("#sources-empty");
const tableWrap = document.querySelector(".table-wrap");
const statDocs = document.querySelector("#stat-docs");
const statChunks = document.querySelector("#stat-chunks");
const statLast = document.querySelector("#stat-last");
function fmtDate(iso) {
try {
return new Date(iso).toLocaleString();
} catch {
return iso;
}
}
async function loadDocs() {
let r;
try {
r = await fetch("/api/docs");
} catch {
showEmpty();
return;
}
if (!r.ok) {
showEmpty();
return;
}
const { documents } = await r.json();
if (!documents.length) {
showEmpty();
return;
}
tbody.replaceChildren();
let totalChunks = 0;
let last = "";
for (const d of documents) {
totalChunks += d.chunks;
if (d.indexed_at > last) last = d.indexed_at;
tbody.appendChild(makeRow(d));
}
statDocs.textContent = String(documents.length);
statChunks.textContent = String(totalChunks);
statLast.textContent = last ? fmtDate(last) : "–";
emptyEl.hidden = true;
tableWrap.hidden = false;
}
function makeRow(d) {
const tr = document.createElement("tr");
const cells = [d.source, d.path, d.title, String(d.chunks), fmtDate(d.indexed_at)];
for (const value of cells) {
const td = document.createElement("td");
td.textContent = value; // document-derived text — never innerHTML
tr.appendChild(td);
}
tr.children[1].title = d.path; // full path on hover (column is ellipsized)
return tr;
}
function showEmpty() {
statDocs.textContent = "0";
statChunks.textContent = "0";
statLast.textContent = "–";
emptyEl.hidden = false;
if (tableWrap) tableWrap.hidden = true;
}
loadDocs();