Files
brain-of-reese/tests/integration/test_kb_overview_api.py
T
ducoterra 5d679f5184 feat(import): index quadlet unit files and jinja templates (A9 revision)
Phase 47 (owner permission 2026-08-27, TODO.md L10–11, roadmap R1): the
full Podman quadlet family (.container, .network, .volume, .image,
.pod, .kube, .swap, .os, .endpoint) and .j2 Jinja templates join the
allowed + default A9 import formats, chunked as plain text (owner
decision — no TOML/Jinja-aware splitter). No env configuration needed:
a default import now indexes them.

- app/config.py: _ALLOWED_IMPORT_EXTENSIONS + the default
  import_extensions CSV gain the ten names (the original seven first);
  the never-widen BOR_IMPORT_EXTENSIONS validator is untouched and
  still rejects truly unknown extensions.
- app/rag/chunker.py: ten _FORMAT_CHUNKERS entries -> chunk_text
  (HARD_MAX_CHARS 1200 honored, unknown-suffix fallback unchanged);
  docstring/comments cite the A9 revision 2026-08-27.
- tests/fixtures/docs/homelab/: quadlet/compose.container (realistic
  quadlet TOML, >1500 chars, [Unit]/[Service]/[Container] sections,
  RESE-QUADLET-SENTINEL-77aa), quadlet/lan.network,
  quadlet/cache.volume, templates/deploy.j2 (for/set/if Jinja
  constructs + RESE-JINJA-SENTINEL-33dd). Every suite that seeds the
  fixture tree updates its 9 -> 13 document-count constants.
- tests/unit/test_config.py: allowed set carries all seventeen formats,
  default CSV + dotted import_extension_set include the ten, the
  validator accepts the new names and still rejects unknowns.
- tests/unit/test_chunker.py: dispatch parity with chunk_text for every
  new suffix (parametrized), the .container fixture chunks >=2 under
  the cap with the sentinel surviving, the .j2 fixture keeps {{ }}
  verbatim, the unknown-suffix fallback is unchanged.
- tests/unit/test_importer.py: a default-extensions walk over a temp
  tree indexes exactly the ten new files (unknown/hidden/excluded
  filtered), the original seven still walk, stem-title fallback holds.
- tests/integration/test_import_quadlet_jinja.py (new): import_sources
  over a temp tree with .container/.volume/.j2 -> documents + chunks
  rows with stem titles; delta re-import updates only the changed .j2
  doc; prune drops the deleted .volume doc with cascade.
- tests/e2e/test_quadlet_jinja_import.py (new, story suite, mock-only,
  isolation): GET /api/docs (admin session) lists the four new-format
  docs with non-zero chunk counts and stem titles; the Sources table
  renders a row + .doc-link per file; the phase-26 modal shows the
  .container TOML ([Container] section + sentinel) with stem title and
  the container format badge; a RESE-JINJA-SENTINEL-33dd question
  FTS-matches the .j2 chunk -> honest-positive (A8: LOW requires zero
  FTS hits) — the bubble is not .is-deflected and a source chip names
  templates/deploy.j2.
- README.md + .env.example: the extended default format set (A9
  revised 2026-08-27, plain-text chunking, narrow-only rule intact).
- .agent/PLAN.md: the A9 revision (owner-locked R1) — A9 row status,
  the revision note under the anchors table, and the §5 chunking-policy
  + §11 workflow lines. The only PLAN edit this phase.

Gates: uv run pytest 795 passed; app/ coverage TOTAL 99% (>90%);
ruff check + pyright clean; story E2E 4/4 in isolation (DB up);
regression E2E suites test_import_documents (3) / test_sync_button
(3) / test_git_sources_admin (6) green in isolation.

Also records the 47_quadlet_jinja_import task-file moves (01–03)
todo/ -> complete/.
2026-08-28 07:02:24 -04:00

250 lines
9.5 KiB
Python

"""Integration: KB overview (phase 31) — the ``<knowledge_base>`` section
of the chat system prompt.
Real Postgres (``podman compose up -d db``) seeded from
``tests/fixtures/docs/`` through the real importer; the chat path reuses
the deterministic capturing fake LLM from ``test_chat_api``
(token-overlap embeddings), so the stored row's journey —
``kb_overview`` row → per-turn PK lookup → ``<knowledge_base>`` section
of the **exact** captured system prompt (HIGH and LOW) — is verified
end-to-end without a network.
The byte-identity contract (phase 15 convention): with no row, the
captured system prompt equals the pre-phase construction
(``build_high_prompt`` / ``build_deflect_prompt`` with
``kb_overview=None``) — asserted with ``==``, not ``in``.
Requires: podman compose up -d db
"""
from __future__ import annotations
import asyncio
import logging
from collections.abc import Iterator
from pathlib import Path
import pytest
from fastapi.testclient import TestClient
from sqlalchemy import select, text
from test_chat_api import FakeRagLLM, _stream_chat, _token_vec
from app.api import chat as chat_api
from app.main import app as fastapi_app
from app.models import Document, KbOverview
from app.rag.importer import import_sources
from app.rag.prompts import build_deflect_prompt, build_high_prompt
from app.rag.retriever import retrieve, weak_hit_titles
FIXTURES = Path(__file__).resolve().parents[1] / "fixtures" / "docs"
QUESTION = "How is my Kubernetes cluster set up?"
OFF_TOPIC = "How do I bake sourdough bread?"
#: A multi-line, multi-bullet outline: the section must carry it whole
#: (well within ``BOR_KB_OVERVIEW_MAX_CHARS``) and the per-turn log line
#: records its length.
OVERVIEW = (
"- Kubernetes cluster and node maintenance notes\n"
"- Backup schedules and restore runbooks\n"
"- Networking: static DNS and kafkabridge"
)
@pytest.fixture(autouse=True)
def clean_kb_overview(db) -> Iterator[None]:
"""The outline row + query log are global state: reset around every
test so no test inherits another test's row."""
db.execute(text("TRUNCATE kb_overview, query_log"))
db.commit()
yield
db.execute(text("TRUNCATE kb_overview, query_log"))
db.commit()
@pytest.fixture()
def seeded_kb(db) -> Iterator[FakeRagLLM]:
"""Fresh Postgres with the fixture docs imported (real pipeline)."""
db.execute(text("TRUNCATE chunks, documents, query_log, kb_overview"))
db.commit()
llm = FakeRagLLM()
summary = asyncio.run(import_sources([FIXTURES], llm, session=db))
assert summary.added == 13 # A9 formats (phase 47 added quadlet+j2); .hidden/ skipped
yield llm
db.execute(text("TRUNCATE chunks, documents, query_log, kb_overview"))
db.commit()
def _seed_overview(db) -> None:
db.add(KbOverview(id=1, content=OVERVIEW))
db.commit()
def _cited_docs(db, frames: list[dict]) -> list[Document]:
"""The documents the done event cited, in citation order — the same
list ``plan_turn`` passed to the prompt builder."""
docs = []
for s in frames[-1]["sources"]:
doc = db.scalar(select(Document).where(Document.path == s["path"]))
assert doc is not None, f"done source {s['path']!r} missing from the KB"
docs.append(doc)
return docs
def _turn_log_lines(caplog: pytest.LogCaptureFixture) -> list[str]:
"""The per-turn ``question=…`` log lines (PLAN §9) from this test."""
return [r.getMessage() for r in caplog.records if "question=" in r.getMessage()]
# ---------- no row → byte-identical to the pre-phase prompts ----------
def test_no_row_high_prompt_byte_identical_to_pre_phase(
client: TestClient, db, seeded_kb: FakeRagLLM, caplog: pytest.LogCaptureFixture
) -> None:
"""No ``kb_overview`` row: the captured HIGH system prompt EQUALS the
pre-phase construction exactly — the section is absent, not empty."""
caplog.set_level(logging.INFO, logger="app.chat")
fastapi_app.dependency_overrides[chat_api.get_llm] = lambda: seeded_kb
try:
_, _, frames = _stream_chat(client, QUESTION)
finally:
fastapi_app.dependency_overrides.clear()
assert frames[-1]["deflected"] is False
(system, user) = seeded_kb.seen_messages[0][0], seeded_kb.seen_messages[0][1]
assert user["content"] == QUESTION
expected = build_high_prompt(_cited_docs(db, frames), notes=[], kb_overview=None)
assert system["content"] == expected
assert "<knowledge_base>" not in system["content"]
lines = _turn_log_lines(caplog)
assert lines and "kb_chars=0" in lines[-1]
def test_no_row_low_prompt_byte_identical_to_pre_phase(
client: TestClient, db, seeded_kb: FakeRagLLM, caplog: pytest.LogCaptureFixture
) -> None:
"""No row, off-topic question: the captured LOW (deflection) prompt
EQUALS the pre-phase construction exactly."""
caplog.set_level(logging.INFO, logger="app.chat")
fastapi_app.dependency_overrides[chat_api.get_llm] = lambda: seeded_kb
try:
_, _, frames = _stream_chat(client, OFF_TOPIC)
finally:
fastapi_app.dependency_overrides.clear()
assert frames[-1]["deflected"] is True
(system, user) = seeded_kb.seen_messages[0][0], seeded_kb.seen_messages[0][1]
assert user["content"] == OFF_TOPIC
# Reconstruct the LOW prompt the way plan_turn does — the pre-phase
# construction (kb_overview=None), the same deterministic retrieval.
chunks = retrieve(db, OFF_TOPIC, _token_vec(OFF_TOPIC))
expected = build_deflect_prompt(
weak_hit_titles(chunks), notes=[], kb_overview=None
)
assert system["content"] == expected
assert "<knowledge_base>" not in system["content"]
assert "DEFLECT_MODE" in system["content"]
lines = _turn_log_lines(caplog)
assert lines and "kb_chars=0" in lines[-1]
# ---------- row present → section in BOTH prompts, exactly ----------
def test_row_high_prompt_carries_kb_section_exactly(
client: TestClient, db, seeded_kb: FakeRagLLM, caplog: pytest.LogCaptureFixture
) -> None:
"""Stored row: the captured HIGH prompt EQUALS the construction with
the outline — section present, ordered before ``<documents>``."""
_seed_overview(db)
caplog.set_level(logging.INFO, logger="app.chat")
fastapi_app.dependency_overrides[chat_api.get_llm] = lambda: seeded_kb
try:
_, _, frames = _stream_chat(client, QUESTION)
finally:
fastapi_app.dependency_overrides.clear()
assert frames[-1]["deflected"] is False
(system, _user) = seeded_kb.seen_messages[0][0], seeded_kb.seen_messages[0][1]
expected = build_high_prompt(
_cited_docs(db, frames), notes=[], kb_overview=OVERVIEW
)
assert system["content"] == expected
# Section shape + order: <relevance> → <knowledge_base> → <documents>.
prompt = system["content"]
assert (
prompt.index("<relevance>HIGH</relevance>")
< prompt.index("<knowledge_base>")
< prompt.index(OVERVIEW)
< prompt.index("</knowledge_base>")
< prompt.index("<documents>")
)
# The per-turn log line records the outline's length (PLAN §9).
lines = _turn_log_lines(caplog)
assert lines and f"kb_chars={len(OVERVIEW)}" in lines[-1]
def test_row_low_prompt_carries_kb_section_exactly(
client: TestClient, db, seeded_kb: FakeRagLLM, caplog: pytest.LogCaptureFixture
) -> None:
"""Stored row, off-topic question: the LOW prompt EQUALS the
construction with the outline — the section is in the deflection
prompt too (real alternatives, not hallucinated ones)."""
_seed_overview(db)
caplog.set_level(logging.INFO, logger="app.chat")
fastapi_app.dependency_overrides[chat_api.get_llm] = lambda: seeded_kb
try:
_, _, frames = _stream_chat(client, OFF_TOPIC)
finally:
fastapi_app.dependency_overrides.clear()
assert frames[-1]["deflected"] is True
(system, _user) = seeded_kb.seen_messages[0][0], seeded_kb.seen_messages[0][1]
chunks = retrieve(db, OFF_TOPIC, _token_vec(OFF_TOPIC))
expected = build_deflect_prompt(
weak_hit_titles(chunks), notes=[], kb_overview=OVERVIEW
)
assert system["content"] == expected
prompt = system["content"]
assert (
prompt.index("<relevance>LOW</relevance>")
< prompt.index("<knowledge_base>")
< prompt.index(OVERVIEW)
< prompt.index("</knowledge_base>")
< prompt.index("DEFLECT_MODE")
)
# Deflection still sees titles only — never document content.
assert "Talos Linux" not in prompt
lines = _turn_log_lines(caplog)
assert lines and f"kb_chars={len(OVERVIEW)}" in lines[-1]
def test_row_reread_every_turn_and_deleted_row_stops_it(
client: TestClient, db, seeded_kb: FakeRagLLM
) -> None:
"""The row is read per turn (not cached): it steers every turn until
it is deleted, and the following turn is section-free again."""
_seed_overview(db)
fastapi_app.dependency_overrides[chat_api.get_llm] = lambda: seeded_kb
try:
_stream_chat(client, QUESTION)
_stream_chat(client, QUESTION)
assert len(seeded_kb.seen_messages) == 2
for messages in seeded_kb.seen_messages:
assert "<knowledge_base>" in messages[0]["content"]
assert OVERVIEW in messages[0]["content"]
# Delete the row → the next turn's prompt drops the section.
db.execute(text("TRUNCATE kb_overview"))
db.commit()
_stream_chat(client, QUESTION)
assert len(seeded_kb.seen_messages) == 3
assert "<knowledge_base>" not in seeded_kb.seen_messages[-1][0]["content"]
finally:
fastapi_app.dependency_overrides.clear()