Phase 02 (story: import documents):
- fence-aware markdown chunker (heading sections, 200-char overlap,
heading anchor on every chunk, 1200-char hard cap, fence blocks
kept atomic and split under the cap)
- LLMClient over aipi (LiteLLM) reusing the openai client's httpx
transport to send a clean {model, input} payload — the openai SDK
injects encoding_format, which aipi's openai_like group rejects;
token-budget batching + halving retry for the endpoint's
~1024-token per-request input cap
- two-phase per-file upsert importer: sha256 delta (unchanged skip),
atomic commit, A9 exclusion walk, per-source prune, per-file error
tolerance (rollback + log + continue, non-zero CLI exit), adaptive
re-chunk at half target for URL-dense files the endpoint rejects
- scripts/import_docs CLI (repeatable --source, --prune, --limit,
defaults ~/Homelab + ~/Deployments)
- GET /api/docs with per-doc chunk counts; Sources page wired to the
real endpoint (stat cards, full-width a11y table, designed empty
state, DOM-built rows — no innerHTML)
- tests: 63 passed (chunker/llm/importer units, docs API + importer
integration), story E2E 3/3 (real endpoints, in-thread import);
app/ coverage 98%
- real KB imported: 672 docs / 8969 chunks in ~3m, idempotent
re-run (672 unchanged, 0 batches)
- harness: .agent/validate.sh now gates through uv (pytest +
coverage >90% + ruff + pyright) instead of system python3
80 lines
2.5 KiB
Python
80 lines
2.5 KiB
Python
"""Integration tests: GET /api/docs — empty shape + populated shape.
|
|
|
|
Uses the real compose Postgres (``db`` fixture) and FastAPI's TestClient.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import uuid
|
|
from datetime import UTC, datetime
|
|
|
|
from sqlalchemy import text
|
|
|
|
from app.models import Chunk, Document
|
|
|
|
|
|
def test_docs_empty_shape(client, db) -> None:
|
|
db.execute(text("TRUNCATE chunks, documents"))
|
|
db.commit()
|
|
r = client.get("/api/docs")
|
|
assert r.status_code == 200
|
|
assert r.json() == {"documents": []}
|
|
|
|
|
|
def test_docs_populated_shape_sorted_with_chunk_counts(client, db) -> None:
|
|
db.execute(text("TRUNCATE chunks, documents"))
|
|
db.commit()
|
|
now = datetime.now(UTC)
|
|
k8s = Document(
|
|
source="Homelab",
|
|
path="kubernetes.md",
|
|
full_path="/tmp/kubernetes.md",
|
|
title="Kubernetes Homelab Cluster",
|
|
content="# Kubernetes Homelab Cluster\n\nTalos on 3 nodes.",
|
|
content_hash="a" * 64,
|
|
indexed_at=now,
|
|
)
|
|
empty = Document(
|
|
source="Deployments",
|
|
path="empty.md",
|
|
full_path="/tmp/empty.md",
|
|
title="No Chunks Yet",
|
|
content="two-phase: doc exists, embeddings pending",
|
|
content_hash="b" * 64,
|
|
indexed_at=now,
|
|
)
|
|
db.add_all([empty, k8s])
|
|
db.flush()
|
|
db.add_all(
|
|
Chunk(document_id=k8s.id, position=i, content=f"chunk {i}", embedding=[0.01] * 768)
|
|
for i in range(3)
|
|
)
|
|
db.commit()
|
|
|
|
r = client.get("/api/docs")
|
|
assert r.status_code == 200
|
|
body = r.json()
|
|
# Ordered by (source, path): Deployments < Homelab.
|
|
assert [d["path"] for d in body["documents"]] == ["empty.md", "kubernetes.md"]
|
|
by_path = {d["path"]: d for d in body["documents"]}
|
|
|
|
k = by_path["kubernetes.md"]
|
|
assert k["source"] == "Homelab"
|
|
assert k["title"] == "Kubernetes Homelab Cluster"
|
|
assert k["chunks"] == 3
|
|
datetime.fromisoformat(k["indexed_at"]) # raises if not valid ISO-8601
|
|
uuid.UUID(k["id"]) # raises if not a valid UUID
|
|
assert by_path["empty.md"]["chunks"] == 0 # outerjoin → zero, not missing
|
|
|
|
db.execute(text("TRUNCATE chunks, documents"))
|
|
db.commit()
|
|
|
|
|
|
def test_docs_response_matches_schema_shape(client, db) -> None:
|
|
r = client.get("/api/docs")
|
|
assert r.status_code == 200
|
|
body = r.json()
|
|
assert set(body) == {"documents"}
|
|
for d in body["documents"]:
|
|
assert set(d) == {"id", "source", "path", "title", "chunks", "indexed_at"}
|
|
assert isinstance(d["chunks"], int) and d["chunks"] >= 0
|