feat(rag): index markdown KB — chunker, embed client, delta importer, Sources page

Phase 02 (story: import documents):

- fence-aware markdown chunker (heading sections, 200-char overlap,
  heading anchor on every chunk, 1200-char hard cap, fence blocks
  kept atomic and split under the cap)
- LLMClient over aipi (LiteLLM) reusing the openai client's httpx
  transport to send a clean {model, input} payload — the openai SDK
  injects encoding_format, which aipi's openai_like group rejects;
  token-budget batching + halving retry for the endpoint's
  ~1024-token per-request input cap
- two-phase per-file upsert importer: sha256 delta (unchanged skip),
  atomic commit, A9 exclusion walk, per-source prune, per-file error
  tolerance (rollback + log + continue, non-zero CLI exit), adaptive
  re-chunk at half target for URL-dense files the endpoint rejects
- scripts/import_docs CLI (repeatable --source, --prune, --limit,
  defaults ~/Homelab + ~/Deployments)
- GET /api/docs with per-doc chunk counts; Sources page wired to the
  real endpoint (stat cards, full-width a11y table, designed empty
  state, DOM-built rows — no innerHTML)
- tests: 63 passed (chunker/llm/importer units, docs API + importer
  integration), story E2E 3/3 (real endpoints, in-thread import);
  app/ coverage 98%
- real KB imported: 672 docs / 8969 chunks in ~3m, idempotent
  re-run (672 unchanged, 0 batches)
- harness: .agent/validate.sh now gates through uv (pytest +
  coverage >90% + ruff + pyright) instead of system python3
This commit is contained in:
2026-08-21 16:24:45 -04:00
parent dce6d0d6f1
commit 99c48cbe06
19 changed files with 1822 additions and 11 deletions
+79
View File
@@ -0,0 +1,79 @@
"""Integration tests: GET /api/docs — empty shape + populated shape.
Uses the real compose Postgres (``db`` fixture) and FastAPI's TestClient.
"""
from __future__ import annotations
import uuid
from datetime import UTC, datetime
from sqlalchemy import text
from app.models import Chunk, Document
def test_docs_empty_shape(client, db) -> None:
db.execute(text("TRUNCATE chunks, documents"))
db.commit()
r = client.get("/api/docs")
assert r.status_code == 200
assert r.json() == {"documents": []}
def test_docs_populated_shape_sorted_with_chunk_counts(client, db) -> None:
db.execute(text("TRUNCATE chunks, documents"))
db.commit()
now = datetime.now(UTC)
k8s = Document(
source="Homelab",
path="kubernetes.md",
full_path="/tmp/kubernetes.md",
title="Kubernetes Homelab Cluster",
content="# Kubernetes Homelab Cluster\n\nTalos on 3 nodes.",
content_hash="a" * 64,
indexed_at=now,
)
empty = Document(
source="Deployments",
path="empty.md",
full_path="/tmp/empty.md",
title="No Chunks Yet",
content="two-phase: doc exists, embeddings pending",
content_hash="b" * 64,
indexed_at=now,
)
db.add_all([empty, k8s])
db.flush()
db.add_all(
Chunk(document_id=k8s.id, position=i, content=f"chunk {i}", embedding=[0.01] * 768)
for i in range(3)
)
db.commit()
r = client.get("/api/docs")
assert r.status_code == 200
body = r.json()
# Ordered by (source, path): Deployments < Homelab.
assert [d["path"] for d in body["documents"]] == ["empty.md", "kubernetes.md"]
by_path = {d["path"]: d for d in body["documents"]}
k = by_path["kubernetes.md"]
assert k["source"] == "Homelab"
assert k["title"] == "Kubernetes Homelab Cluster"
assert k["chunks"] == 3
datetime.fromisoformat(k["indexed_at"]) # raises if not valid ISO-8601
uuid.UUID(k["id"]) # raises if not a valid UUID
assert by_path["empty.md"]["chunks"] == 0 # outerjoin → zero, not missing
db.execute(text("TRUNCATE chunks, documents"))
db.commit()
def test_docs_response_matches_schema_shape(client, db) -> None:
r = client.get("/api/docs")
assert r.status_code == 200
body = r.json()
assert set(body) == {"documents"}
for d in body["documents"]:
assert set(d) == {"id", "source", "path", "title", "chunks", "indexed_at"}
assert isinstance(d["chunks"], int) and d["chunks"] >= 0