Phase 02 (story: import documents):
- fence-aware markdown chunker (heading sections, 200-char overlap,
heading anchor on every chunk, 1200-char hard cap, fence blocks
kept atomic and split under the cap)
- LLMClient over aipi (LiteLLM) reusing the openai client's httpx
transport to send a clean {model, input} payload — the openai SDK
injects encoding_format, which aipi's openai_like group rejects;
token-budget batching + halving retry for the endpoint's
~1024-token per-request input cap
- two-phase per-file upsert importer: sha256 delta (unchanged skip),
atomic commit, A9 exclusion walk, per-source prune, per-file error
tolerance (rollback + log + continue, non-zero CLI exit), adaptive
re-chunk at half target for URL-dense files the endpoint rejects
- scripts/import_docs CLI (repeatable --source, --prune, --limit,
defaults ~/Homelab + ~/Deployments)
- GET /api/docs with per-doc chunk counts; Sources page wired to the
real endpoint (stat cards, full-width a11y table, designed empty
state, DOM-built rows — no innerHTML)
- tests: 63 passed (chunker/llm/importer units, docs API + importer
integration), story E2E 3/3 (real endpoints, in-thread import);
app/ coverage 98%
- real KB imported: 672 docs / 8969 chunks in ~3m, idempotent
re-run (672 unchanged, 0 batches)
- harness: .agent/validate.sh now gates through uv (pytest +
coverage >90% + ruff + pyright) instead of system python3
48 lines
1.5 KiB
Python
48 lines
1.5 KiB
Python
"""GET /api/docs — the indexed document list (feeds the Sources page)."""
|
|
from __future__ import annotations
|
|
|
|
from fastapi import APIRouter, Depends
|
|
from sqlalchemy import func, select
|
|
from sqlalchemy.orm import Session
|
|
|
|
from app.db import get_db
|
|
from app.models import Chunk, Document
|
|
from app.schemas import DocList, DocSummary
|
|
|
|
router = APIRouter(tags=["kb"])
|
|
|
|
|
|
@router.get("/docs", response_model=DocList)
|
|
def list_documents(db: Session = Depends(get_db)) -> DocList: # noqa: B008
|
|
"""All indexed documents with per-document chunk counts.
|
|
|
|
An empty list means the knowledge base has not been imported yet —
|
|
the Sources page renders its designed empty state in that case.
|
|
"""
|
|
rows = db.execute(
|
|
select(
|
|
Document.id,
|
|
Document.source,
|
|
Document.path,
|
|
Document.title,
|
|
func.count(Chunk.id).label("chunks"),
|
|
Document.indexed_at,
|
|
)
|
|
.outerjoin(Chunk, Chunk.document_id == Document.id)
|
|
.group_by(Document.id, Document.source, Document.path, Document.title, Document.indexed_at)
|
|
.order_by(Document.source, Document.path)
|
|
).all()
|
|
return DocList(
|
|
documents=[
|
|
DocSummary(
|
|
id=str(row.id),
|
|
source=row.source,
|
|
path=row.path,
|
|
title=row.title,
|
|
chunks=row.chunks,
|
|
indexed_at=row.indexed_at.isoformat(),
|
|
)
|
|
for row in rows
|
|
]
|
|
)
|