93 lines
3.2 KiB
Python
93 lines
3.2 KiB
Python
"""GET /api/docs — the indexed document list (feeds the Sources page).
|
|
|
|
GET /api/documents/content — one indexed document's full content (feeds the
|
|
clickable document viewer, phase 10). DB-only by design: the (source, path)
|
|
pair is looked up as a row, so there is no filesystem access and no
|
|
path-traversal surface — ``../``-style values simply aren't rows (→ 404).
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
|
|
from fastapi import APIRouter, Depends, HTTPException
|
|
from sqlalchemy import func, select
|
|
from sqlalchemy.orm import Session
|
|
|
|
from app.db import get_db
|
|
from app.models import Chunk, Document
|
|
from app.schemas import DocContent, DocList, DocSummary
|
|
|
|
router = APIRouter(tags=["kb"])
|
|
|
|
|
|
def doc_format(path: str) -> str:
|
|
"""Lowercased path suffix without its dot (``kubernetes.md`` → ``md``,
|
|
``notes/deep.Markdown`` → ``markdown``); ``text`` when the path has no
|
|
suffix — the value shown in the viewer's format badge."""
|
|
return Path(path).suffix.lower().lstrip(".") or "text"
|
|
|
|
|
|
@router.get("/docs", response_model=DocList)
|
|
def list_documents(db: Session = Depends(get_db)) -> DocList: # noqa: B008
|
|
"""All indexed documents with per-document chunk counts.
|
|
|
|
An empty list means the knowledge base has not been imported yet —
|
|
the Sources page renders its designed empty state in that case.
|
|
"""
|
|
rows = db.execute(
|
|
select(
|
|
Document.id,
|
|
Document.source,
|
|
Document.path,
|
|
Document.title,
|
|
func.count(Chunk.id).label("chunks"),
|
|
Document.indexed_at,
|
|
)
|
|
.outerjoin(Chunk, Chunk.document_id == Document.id)
|
|
.group_by(Document.id, Document.source, Document.path, Document.title, Document.indexed_at)
|
|
.order_by(Document.source, Document.path)
|
|
).all()
|
|
return DocList(
|
|
documents=[
|
|
DocSummary(
|
|
id=str(row.id),
|
|
source=row.source,
|
|
path=row.path,
|
|
title=row.title,
|
|
chunks=row.chunks,
|
|
indexed_at=row.indexed_at.isoformat(),
|
|
)
|
|
for row in rows
|
|
]
|
|
)
|
|
|
|
|
|
@router.get("/documents/content", response_model=DocContent)
|
|
def get_document_content(
|
|
source: str, path: str, db: Session = Depends(get_db) # noqa: B008
|
|
) -> DocContent:
|
|
"""Full content of one indexed document, looked up by ``(source, path)``.
|
|
|
|
Stateless (A10) and database-only: unknown pairs — including traversal
|
|
strings such as ``../../etc/passwd`` — are just non-existent rows and
|
|
map to 404 ``{detail: "document not found"}``.
|
|
"""
|
|
row = db.execute(
|
|
select(Document, func.count(Chunk.id).label("chunks"))
|
|
.outerjoin(Chunk, Chunk.document_id == Document.id)
|
|
.where(Document.source == source, Document.path == path)
|
|
.group_by(Document.id)
|
|
).first()
|
|
if row is None:
|
|
raise HTTPException(status_code=404, detail="document not found")
|
|
doc, chunks = row
|
|
return DocContent(
|
|
source=doc.source,
|
|
path=doc.path,
|
|
title=doc.title,
|
|
format=doc_format(doc.path),
|
|
content=doc.content,
|
|
indexed_at=doc.indexed_at.isoformat(),
|
|
chunks=chunks,
|
|
)
|