feat(rag): index markdown KB — chunker, embed client, delta importer, Sources page
Phase 02 (story: import documents):
- fence-aware markdown chunker (heading sections, 200-char overlap,
heading anchor on every chunk, 1200-char hard cap, fence blocks
kept atomic and split under the cap)
- LLMClient over aipi (LiteLLM) reusing the openai client's httpx
transport to send a clean {model, input} payload — the openai SDK
injects encoding_format, which aipi's openai_like group rejects;
token-budget batching + halving retry for the endpoint's
~1024-token per-request input cap
- two-phase per-file upsert importer: sha256 delta (unchanged skip),
atomic commit, A9 exclusion walk, per-source prune, per-file error
tolerance (rollback + log + continue, non-zero CLI exit), adaptive
re-chunk at half target for URL-dense files the endpoint rejects
- scripts/import_docs CLI (repeatable --source, --prune, --limit,
defaults ~/Homelab + ~/Deployments)
- GET /api/docs with per-doc chunk counts; Sources page wired to the
real endpoint (stat cards, full-width a11y table, designed empty
state, DOM-built rows — no innerHTML)
- tests: 63 passed (chunker/llm/importer units, docs API + importer
integration), story E2E 3/3 (real endpoints, in-thread import);
app/ coverage 98%
- real KB imported: 672 docs / 8969 chunks in ~3m, idempotent
re-run (672 unchanged, 0 batches)
- harness: .agent/validate.sh now gates through uv (pytest +
coverage >90% + ruff + pyright) instead of system python3
This commit is contained in:
@@ -0,0 +1,79 @@
|
||||
"""Integration tests: GET /api/docs — empty shape + populated shape.
|
||||
|
||||
Uses the real compose Postgres (``db`` fixture) and FastAPI's TestClient.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import uuid
|
||||
from datetime import UTC, datetime
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.models import Chunk, Document
|
||||
|
||||
|
||||
def test_docs_empty_shape(client, db) -> None:
|
||||
db.execute(text("TRUNCATE chunks, documents"))
|
||||
db.commit()
|
||||
r = client.get("/api/docs")
|
||||
assert r.status_code == 200
|
||||
assert r.json() == {"documents": []}
|
||||
|
||||
|
||||
def test_docs_populated_shape_sorted_with_chunk_counts(client, db) -> None:
|
||||
db.execute(text("TRUNCATE chunks, documents"))
|
||||
db.commit()
|
||||
now = datetime.now(UTC)
|
||||
k8s = Document(
|
||||
source="Homelab",
|
||||
path="kubernetes.md",
|
||||
full_path="/tmp/kubernetes.md",
|
||||
title="Kubernetes Homelab Cluster",
|
||||
content="# Kubernetes Homelab Cluster\n\nTalos on 3 nodes.",
|
||||
content_hash="a" * 64,
|
||||
indexed_at=now,
|
||||
)
|
||||
empty = Document(
|
||||
source="Deployments",
|
||||
path="empty.md",
|
||||
full_path="/tmp/empty.md",
|
||||
title="No Chunks Yet",
|
||||
content="two-phase: doc exists, embeddings pending",
|
||||
content_hash="b" * 64,
|
||||
indexed_at=now,
|
||||
)
|
||||
db.add_all([empty, k8s])
|
||||
db.flush()
|
||||
db.add_all(
|
||||
Chunk(document_id=k8s.id, position=i, content=f"chunk {i}", embedding=[0.01] * 768)
|
||||
for i in range(3)
|
||||
)
|
||||
db.commit()
|
||||
|
||||
r = client.get("/api/docs")
|
||||
assert r.status_code == 200
|
||||
body = r.json()
|
||||
# Ordered by (source, path): Deployments < Homelab.
|
||||
assert [d["path"] for d in body["documents"]] == ["empty.md", "kubernetes.md"]
|
||||
by_path = {d["path"]: d for d in body["documents"]}
|
||||
|
||||
k = by_path["kubernetes.md"]
|
||||
assert k["source"] == "Homelab"
|
||||
assert k["title"] == "Kubernetes Homelab Cluster"
|
||||
assert k["chunks"] == 3
|
||||
datetime.fromisoformat(k["indexed_at"]) # raises if not valid ISO-8601
|
||||
uuid.UUID(k["id"]) # raises if not a valid UUID
|
||||
assert by_path["empty.md"]["chunks"] == 0 # outerjoin → zero, not missing
|
||||
|
||||
db.execute(text("TRUNCATE chunks, documents"))
|
||||
db.commit()
|
||||
|
||||
|
||||
def test_docs_response_matches_schema_shape(client, db) -> None:
|
||||
r = client.get("/api/docs")
|
||||
assert r.status_code == 200
|
||||
body = r.json()
|
||||
assert set(body) == {"documents"}
|
||||
for d in body["documents"]:
|
||||
assert set(d) == {"id", "source", "path", "title", "chunks", "indexed_at"}
|
||||
assert isinstance(d["chunks"], int) and d["chunks"] >= 0
|
||||
@@ -0,0 +1,65 @@
|
||||
"""Integration test: importer end-to-end against ``tests/fixtures/docs/``.
|
||||
|
||||
Runs the real import pipeline (walk → chunk → embed → upsert) into the
|
||||
local compose Postgres, then checks the DB state *and* the API shape a
|
||||
browser would consume. Embeddings come from a deterministic in-process
|
||||
fake, so no network is needed.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
from pathlib import Path
|
||||
|
||||
from sqlalchemy import func, select, text
|
||||
|
||||
from app.models import Chunk, Document
|
||||
from app.rag.importer import import_sources
|
||||
from tests.fakes import FakeEmbedder
|
||||
|
||||
FIXTURES = Path(__file__).resolve().parents[1] / "fixtures" / "docs"
|
||||
|
||||
EXPECTED_DOCS = {
|
||||
("docs", "homelab/kubernetes.md"),
|
||||
("docs", "homelab/backups.md"),
|
||||
("docs", "deployments/new-service.md"),
|
||||
}
|
||||
|
||||
|
||||
def test_import_fixtures_end_to_end(client, db) -> None:
|
||||
db.execute(text("TRUNCATE chunks, documents, query_log"))
|
||||
db.commit()
|
||||
llm = FakeEmbedder()
|
||||
|
||||
summary = asyncio.run(import_sources([FIXTURES], llm, session=db))
|
||||
assert (summary.files, summary.added, summary.unchanged) == (3, 3, 0)
|
||||
assert summary.chunks >= 3
|
||||
|
||||
docs = db.scalars(select(Document)).all()
|
||||
assert {(d.source, d.path) for d in docs} == EXPECTED_DOCS
|
||||
titles = {d.path: d.title for d in docs}
|
||||
assert titles["homelab/kubernetes.md"] == "Kubernetes Homelab Cluster"
|
||||
assert titles["deployments/new-service.md"] == "Deploying a New Service"
|
||||
# Full content is stored — that is what the RAG context will be.
|
||||
k8s = next(d for d in docs if d.path == "homelab/kubernetes.md")
|
||||
assert "Talos Linux" in k8s.content and k8s.content_hash
|
||||
|
||||
n_chunks = db.scalar(select(func.count()).select_from(Chunk))
|
||||
assert n_chunks == summary.chunks
|
||||
for c in db.scalars(select(Chunk)).all():
|
||||
assert c.embedding is not None and len(c.embedding) == 768
|
||||
|
||||
# The Sources page consumes exactly this shape.
|
||||
r = client.get("/api/docs")
|
||||
assert r.status_code == 200
|
||||
body = r.json()
|
||||
assert len(body["documents"]) == 3
|
||||
assert all(d["chunks"] >= 1 for d in body["documents"])
|
||||
|
||||
# Idempotent re-run: nothing re-embedded.
|
||||
calls_before = len(llm.calls)
|
||||
s2 = asyncio.run(import_sources([FIXTURES], llm, session=db))
|
||||
assert s2.unchanged == 3 and s2.added == 0
|
||||
assert len(llm.calls) == calls_before # unchanged → no embedding requests
|
||||
|
||||
db.execute(text("TRUNCATE chunks, documents, query_log"))
|
||||
db.commit()
|
||||
Reference in New Issue
Block a user