feat(rag): stream grounded RAG answers over SSE with source citations
Phase 03 (Story: Chat RAG Answer — happy path):
- app/rag/retriever.py: top-k cosine search + parent-doc selection with
per-doc dedupe and BOR_MAX_CONTEXT_CHARS cap ([…truncated…] marker)
- app/rag/prompts.py: locked persona + HIGH/DEFLECT prompt builders
- app/rag/llm.py: LLMError + chat_stream (turbo, temp 0.4, max 700, stream)
- app/api/chat.py: POST /api/chat SSE — delta* then done{deflected,
sources, suggestions}; query_log row + PLAN §9 per-turn log line;
structured error event on mid-stream failure, JSON 503 when DB down
- frontend: SSE reader, live bubble streaming, source chips -> /sources.html,
red role=alert banner, Send button state that always recovers
- fix(scaffold): [hidden] { display: none !important } — .kb-banner's
display:flex was overriding the hidden attribute (banner always visible)
- tests: unit (retriever/prompts/sse/llm) + integration (real Postgres RAG
turn, query_log, error + 503 paths, mid-turn failures) + Playwright story
suite (grounded answer, log row, raw SSE shape); smoke placeholder test
replaced with the real never-stale-button contract
This commit is contained in:
+39
-3
@@ -1,7 +1,7 @@
|
||||
"""Async OpenAI-compatible client for the self-hosted aipi endpoint (PLAN A5).
|
||||
|
||||
Phase 02 adds the embeddings surface (the importer — and, from phase 03,
|
||||
retrieval — need it). Chat streaming lands in phase 03 on this same client.
|
||||
Provides the embeddings surface (importer, retrieval) and chat streaming
|
||||
(PLAN A15) for the RAG pipeline.
|
||||
|
||||
Fail-loud rule (PLAN A6): the ``chunks.embedding`` column is fixed at 768
|
||||
dimensions when the table is created, so a model that returns a different
|
||||
@@ -11,8 +11,11 @@ vectors that pgvector rejects.
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from collections.abc import AsyncIterator
|
||||
from typing import cast
|
||||
|
||||
from openai import AsyncOpenAI
|
||||
from openai.types.chat import ChatCompletionMessageParam
|
||||
|
||||
from app.config import Settings, get_settings
|
||||
|
||||
@@ -27,6 +30,10 @@ class EmbeddingDimensionError(EmbeddingError):
|
||||
"""Embedding dimension != BOR_EMBEDDING_DIM — import must fail loudly."""
|
||||
|
||||
|
||||
class LLMError(RuntimeError):
|
||||
"""The chat-completions endpoint failed (network, HTTP, or mid-stream)."""
|
||||
|
||||
|
||||
# aipi's local embedding model rejects requests over ~1024 input tokens
|
||||
# ("input is too large to process"). Batch by estimated tokens, with a
|
||||
# safety margin under that cap — code-dense text can tokenize at ~3
|
||||
@@ -157,6 +164,35 @@ class LLMClient:
|
||||
return out
|
||||
|
||||
async def embed_one(self, text: str) -> list[float]:
|
||||
"""Convenience: embed a single text (retrieval path, phase 03)."""
|
||||
"""Convenience: embed a single text (retrieval path)."""
|
||||
(vec,) = await self.embed([text])
|
||||
return vec
|
||||
|
||||
async def chat_stream(self, messages: list[dict[str, str]]) -> AsyncIterator[str]:
|
||||
"""Stream assistant text deltas from the chat model (PLAN A5/A15).
|
||||
|
||||
``stream=True`` against the OpenAI-compatible endpoint; yields only
|
||||
non-empty ``delta.content`` pieces. Any failure (network, HTTP,
|
||||
malformed stream) surfaces as :class:`LLMError` so the API layer can
|
||||
turn it into an SSE ``error`` event instead of a hung request.
|
||||
"""
|
||||
try:
|
||||
# ``{role, content}`` dicts are exactly what the message params
|
||||
# accept; the cast keeps pyright honest about the SDK's union.
|
||||
stream = await self._client.chat.completions.create(
|
||||
model=self.settings.llm_chat_model,
|
||||
messages=cast("list[ChatCompletionMessageParam]", messages),
|
||||
temperature=0.4,
|
||||
max_tokens=700,
|
||||
stream=True,
|
||||
)
|
||||
async for chunk in stream:
|
||||
if not chunk.choices:
|
||||
continue
|
||||
piece = chunk.choices[0].delta.content
|
||||
if piece:
|
||||
yield piece
|
||||
except LLMError:
|
||||
raise
|
||||
except Exception as e: # noqa: BLE001 — wrap transport-level failures
|
||||
raise LLMError(f"chat stream from {self.settings.llm_base_url} failed: {e}") from e
|
||||
|
||||
@@ -0,0 +1,67 @@
|
||||
"""Locked system-prompt builder (PLAN §6).
|
||||
|
||||
The persona + HONESTY GATE text is **locked verbatim** — change it through
|
||||
the plan, not here. Two modes:
|
||||
|
||||
* ``HIGH`` — grounded turn: full top-document texts under ``<documents>``.
|
||||
* ``LOW`` — deflection turn: weak-hit *titles only* plus the
|
||||
``DEFLECT_MODE`` marker (the E2E mock LLM keys on that marker).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Sequence
|
||||
|
||||
from app.models import Document
|
||||
|
||||
#: PLAN §6 verbatim (line wrapping included); ``{relevance}`` is filled by
|
||||
#: :func:`_base`.
|
||||
PERSONA: str = (
|
||||
'You are "Brain of Reese" — the digital brain of Reese, a self-hoster and\n'
|
||||
"homelab tinkerer. Personality: chippy, upbeat, warm, and genuinely\n"
|
||||
'optimistic about the user\'s ability to do things ("you\'ve got this").\n'
|
||||
"\n"
|
||||
"Rules:\n"
|
||||
"1. Answer ONLY from the provided document context. Cite which document(s)\n"
|
||||
" you used, by path.\n"
|
||||
"2. Be concrete: names, versions, ports, hosts, schedules — the specifics in\n"
|
||||
" the docs are the value.\n"
|
||||
'3. HONESTY GATE: if <relevance> is "LOW", you must NOT pretend to know.\n'
|
||||
' Start your answer with a variant of: "I haven\'t done anything like that."\n'
|
||||
" Then offer 2-3 alternative questions about things you DO have notes on.\n"
|
||||
"4. Never invent facts, hosts, or steps that are not in the context.\n"
|
||||
"5. Keep answers tight: short paragraphs, bullets where helpful.\n"
|
||||
"\n"
|
||||
"<relevance>{relevance}</relevance>"
|
||||
)
|
||||
|
||||
|
||||
def _base(relevance: str) -> str:
|
||||
if relevance not in ("HIGH", "LOW"):
|
||||
raise ValueError(f"relevance must be HIGH or LOW, got {relevance!r}")
|
||||
return PERSONA.replace("{relevance}", relevance)
|
||||
|
||||
|
||||
def build_high_prompt(documents: Sequence[Document]) -> str:
|
||||
"""Grounded turn: locked persona + full texts of the top documents."""
|
||||
blocks = [
|
||||
f'<document source="{doc.source}" path="{doc.path}" title="{doc.title}">\n'
|
||||
f"{doc.content}\n"
|
||||
"</document>"
|
||||
for doc in documents
|
||||
]
|
||||
body = "\n\n".join(blocks) if blocks else (
|
||||
"(no documents matched — do not invent specifics)"
|
||||
)
|
||||
return _base("HIGH") + "\n<documents>\n" + body + "\n</documents>"
|
||||
|
||||
|
||||
def build_deflect_prompt(titles: Sequence[str]) -> str:
|
||||
"""Deflection turn: weak-hit titles only (no document content)."""
|
||||
weak = "\n".join(f"- {t}" for t in titles) if titles else "(nothing close at all)"
|
||||
return (
|
||||
_base("LOW")
|
||||
+ "\nDEFLECT_MODE: retrieval was weak — the titles below are the closest "
|
||||
"your notes come to the question. They are titles only; do not pretend "
|
||||
"they answer it. Use them to propose 2-3 alternative questions.\n"
|
||||
+ weak
|
||||
)
|
||||
@@ -0,0 +1,101 @@
|
||||
"""pgvector cosine retrieval → parent-document mapping (PLAN §3/§6, A7).
|
||||
|
||||
Retrieval returns the *chunks* closest to the question embedding (top-K by
|
||||
cosine distance). The product requirement is that the LLM receives the
|
||||
**entire relevant document**, not just the chunk (LOCKED A7) — so
|
||||
:meth:`select_documents` maps chunk hits back to their parent documents
|
||||
(``chunks.document_id → documents``), dedupes, ranks by best chunk score,
|
||||
and caps the combined context at ``BOR_MAX_CONTEXT_CHARS``.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import uuid
|
||||
from collections.abc import Sequence
|
||||
from dataclasses import dataclass
|
||||
|
||||
from sqlalchemy import select
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from app.config import get_settings
|
||||
from app.models import Chunk, Document
|
||||
|
||||
#: Marker appended when the context budget is exceeded (PLAN §6).
|
||||
TRUNCATION_MARKER = "[…truncated…]"
|
||||
|
||||
|
||||
@dataclass
|
||||
class RetrievedChunk:
|
||||
"""One chunk hit: its cosine score plus the parent document row."""
|
||||
|
||||
chunk_id: uuid.UUID
|
||||
position: int
|
||||
content: str
|
||||
score: float # 1 − cosine_distance (higher is more similar)
|
||||
document: Document
|
||||
|
||||
|
||||
def retrieve(
|
||||
db: Session, question_embedding: list[float], top_k: int | None = None
|
||||
) -> list[RetrievedChunk]:
|
||||
"""Top-*top_k* chunks by pgvector cosine distance (``<=>``).
|
||||
|
||||
``score = 1 − distance``. Results are ordered by ascending distance, so
|
||||
index 0 is the best hit. Chunks whose embedding is still NULL (two-phase
|
||||
import in progress) are skipped.
|
||||
"""
|
||||
k = top_k if top_k is not None else get_settings().top_k_chunks
|
||||
distance = Chunk.embedding.cosine_distance(question_embedding)
|
||||
rows = db.execute(
|
||||
select(Chunk, distance.label("distance"), Document)
|
||||
.join(Document, Chunk.document_id == Document.id)
|
||||
.where(Chunk.embedding.is_not(None))
|
||||
.order_by(distance)
|
||||
.limit(k)
|
||||
).all()
|
||||
return [
|
||||
RetrievedChunk(
|
||||
chunk_id=chunk.id,
|
||||
position=chunk.position,
|
||||
content=chunk.content,
|
||||
score=round(1.0 - float(dist), 6),
|
||||
document=doc,
|
||||
)
|
||||
for chunk, dist, doc in rows
|
||||
]
|
||||
|
||||
|
||||
def select_documents(
|
||||
chunks: Sequence[RetrievedChunk],
|
||||
n: int | None = None,
|
||||
max_chars: int | None = None,
|
||||
) -> list[Document]:
|
||||
"""Map chunk hits to distinct parent documents, ranked by best chunk score.
|
||||
|
||||
At most *n* documents are returned (default ``BOR_TOP_N_DOCS``). The
|
||||
returned rows carry the full document content; if the combined content
|
||||
would exceed *max_chars* (default ``BOR_MAX_CONTEXT_CHARS``), the
|
||||
lowest-ranked overflowing document is truncated in place with the
|
||||
``[…truncated…]`` marker so the assembled context never exceeds the
|
||||
budget (PLAN §6).
|
||||
"""
|
||||
top_n = n if n is not None else get_settings().top_n_docs
|
||||
budget = max_chars if max_chars is not None else get_settings().max_context_chars
|
||||
|
||||
docs: list[Document] = []
|
||||
seen: set[uuid.UUID] = set()
|
||||
for rc in sorted(chunks, key=lambda c: c.score, reverse=True):
|
||||
if rc.document.id in seen:
|
||||
continue
|
||||
seen.add(rc.document.id)
|
||||
docs.append(rc.document)
|
||||
docs = docs[:top_n]
|
||||
|
||||
remaining = budget
|
||||
for doc in docs:
|
||||
if len(doc.content) <= remaining:
|
||||
remaining -= len(doc.content)
|
||||
else:
|
||||
keep = max(0, remaining - len(TRUNCATION_MARKER))
|
||||
doc.content = doc.content[:keep] + TRUNCATION_MARKER
|
||||
remaining = 0
|
||||
return docs
|
||||
Reference in New Issue
Block a user