All verification complete. Final report: **Phase 119 final verification pass — all criteria verified, one stale pin fixed.** - Verified implementation of all 6 tasks: D1 component name-hit rule (`name_hit` flag, titles never matched, retired length tie-break), D2 `BOR_NAME_HIT_BONUS` (0.005 default, 0 = byte-identical kill switch, negative fails startup, selection-layer only, `eval_retrieval` `suggested:` line), D3 suggested-folder lines (after `SUGGEST_INTRO`, before first block), D4 cite-discipline `SUGGEST_INTRO` sentence (PERSONA/LOW/`TOOLS_SECTION` byte-pins intact), D5 `done.sources` = read docs only (frontend no-op on empty confirmed), D6 mock `repeat your folder map` echo + new suite + telemetry. - Battery (replica restored per skill, fingerprint docs=1000/chunks=8866 verified, `eval_retrieval --from-file tests/fixtures/retrieval_battery.txt` re-run): **GATE PASS** — gitea README #4 in suggested top-5, forgejo 5/5 (README #1), gateway README in top-5 (#4), qwen3.8-27b quadlets top-5, Mongolia HIGH/fts=5 unchanged. - New E2E in isolation: `4 passed` ×2 (deterministic). All 27 modified E2E suites in isolation: 26 green; **1 stale pin fixed** — `test_source_chip_quality.py` durable-record order pin pre-dated the D1 re-rank (`aliases` stem sub-component name-hits `ssh_aliases.txt`, deterministically lifting `backups.md` over `kubernetes.md`; probe-verified 0.016277 vs 0.016036, 4/4 stable) — re-pinned with the phase-119 rationale; suite green ×2. - Gates: `uv run pytest --cov=app --cov-report=term-missing` → **2547 passed, app coverage 99%** (>90%); `uv run ruff check .` → All checks passed; `uv run pyright` → 0 errors. - Completion criteria: 1 ✅ (battery, recorded), 2 ✅ (folder lines; block/LOW byte-identical pins green), 3 ✅ (read-only chips, zero-read chips nothing, related row + durable record untouched — unit+E2E agree), 4 ✅ (all green), 5 → commit/phase-move left to the harness per pass rules (nothing committed). - Deviations: battery output + real-model telemetry recorded in `.agents/reports/119_name_signal_read_chips/task06_battery_and_e2e.md` and `TOOL_CALLING_TESTING.md` §11 (task files in `complete/` are immutable to this pass); gateway canonical doc at #4 vs overview's #3 was already documented at task 06 (containment gate met). - Next pending phase: **none** — `todo/` holds only phase 119.
289 lines
11 KiB
Python
289 lines
11 KiB
Python
"""Integration: KB overview (phase 31) — the ``<knowledge_base>`` section
|
|
of the chat system prompt.
|
|
|
|
Real Postgres (``podman compose up -d db``) seeded from
|
|
``tests/fixtures/docs/`` through the real importer; the chat path reuses
|
|
the deterministic capturing fake LLM from ``test_chat_api``
|
|
(token-overlap embeddings), so the stored row's journey —
|
|
``kb_overview`` row → per-turn PK lookup → ``<knowledge_base>`` section
|
|
of the **exact** captured system prompt (HIGH and LOW) — is verified
|
|
end-to-end without a network.
|
|
|
|
The byte-identity contract (phase 15 convention): with no row, the
|
|
captured system prompt equals the pre-phase construction
|
|
(``build_high_prompt`` / ``build_deflect_prompt`` with
|
|
``kb_overview=None``) — asserted with ``==``, not ``in``.
|
|
|
|
Requires: podman compose up -d db
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import logging
|
|
from collections.abc import Iterator
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
from fastapi.testclient import TestClient
|
|
from sqlalchemy import text
|
|
from test_chat_api import FakeRagLLM, _stream_chat, _token_vec
|
|
|
|
from app.api import chat as chat_api
|
|
from app.config import get_settings
|
|
from app.main import app as fastapi_app
|
|
from app.models import Document, KbOverview
|
|
from app.rag.agent import suggested_folder_lines
|
|
from app.rag.importer import import_sources
|
|
from app.rag.prompts import build_deflect_prompt, build_high_prompt
|
|
from app.rag.retriever import retrieve, select_suggested, weak_hit_titles
|
|
from tests.conftest import ADMIN_PASSWORD
|
|
|
|
FIXTURES = Path(__file__).resolve().parents[1] / "fixtures" / "docs"
|
|
QUESTION = "How is my Kubernetes cluster set up?"
|
|
OFF_TOPIC = "How do I bake sourdough bread?"
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def _admin_signed_in(client: TestClient) -> None:
|
|
"""Phase 79 (task 03): ``POST /api/chat`` is user-gated — this module
|
|
reuses ``test_chat_api._stream_chat`` with the shared (module-local
|
|
to THIS file) ``client``, so it signs the admin in once per test
|
|
(the autouse in ``test_chat_api`` does not apply across the import).
|
|
"""
|
|
r = client.post("/api/login", json={"password": ADMIN_PASSWORD})
|
|
assert r.status_code == 204, f"admin login failed: {r.status_code} {r.text}"
|
|
|
|
#: A multi-line, multi-bullet outline: the section must carry it whole
|
|
#: (well within ``BOR_KB_OVERVIEW_MAX_CHARS``) and the per-turn log line
|
|
#: records its length.
|
|
OVERVIEW = (
|
|
"- Kubernetes cluster and node maintenance notes\n"
|
|
"- Backup schedules and restore runbooks\n"
|
|
"- Networking: static DNS and kafkabridge"
|
|
)
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def clean_kb_overview(db) -> Iterator[None]:
|
|
"""The outline row + query log are global state: reset around every
|
|
test so no test inherits another test's row."""
|
|
db.execute(text("TRUNCATE kb_overview, query_log"))
|
|
db.commit()
|
|
yield
|
|
db.execute(text("TRUNCATE kb_overview, query_log"))
|
|
db.commit()
|
|
|
|
|
|
@pytest.fixture()
|
|
def seeded_kb(db) -> Iterator[FakeRagLLM]:
|
|
"""Fresh Postgres with the fixture docs imported (real pipeline)."""
|
|
db.execute(text("TRUNCATE chunks, documents, query_log, kb_overview"))
|
|
db.commit()
|
|
llm = FakeRagLLM()
|
|
summary = asyncio.run(import_sources([FIXTURES], llm, session=db))
|
|
assert summary.added == 13 # A9 formats (phase 47 added quadlet+j2); .hidden/ skipped
|
|
yield llm
|
|
db.execute(text("TRUNCATE chunks, documents, query_log, kb_overview"))
|
|
db.commit()
|
|
|
|
|
|
def _seed_overview(db) -> None:
|
|
db.add(KbOverview(id=1, content=OVERVIEW))
|
|
db.commit()
|
|
|
|
|
|
def _suggested_docs(db) -> list[Document]:
|
|
"""The documents the grounded turn's HIGH prompt seeded, in the same
|
|
order ``plan_turn`` walked them — the SAME deterministic suggested
|
|
walk over the retrieval. Phase 119 (A1) moved the citation surface
|
|
to the agent's READ docs, so the seeded tier can no longer be read
|
|
off the done frame (the canned LLM here reads nothing ⇒
|
|
``done.sources`` is empty) — reconstructing from the retrieval is
|
|
the faithful source now."""
|
|
chunks = retrieve(db, QUESTION, _token_vec(QUESTION))
|
|
return list(select_suggested(chunks, n=get_settings().suggested_docs))
|
|
|
|
|
|
def _suggested_folder_lines(db) -> list[str]:
|
|
"""The suggested-folder context lines the endpoint computed (phase
|
|
119, D3, LOCKED A4) — the SAME deterministic suggested walk
|
|
``plan_turn`` performs internally (one extra walk, no shared
|
|
state), over the real seeded catalog."""
|
|
chunks = retrieve(db, QUESTION, _token_vec(QUESTION))
|
|
return suggested_folder_lines(
|
|
db, select_suggested(chunks, n=get_settings().suggested_docs)
|
|
)
|
|
|
|
|
|
def _turn_log_lines(caplog: pytest.LogCaptureFixture) -> list[str]:
|
|
"""The per-turn ``question=…`` log lines (PLAN §9) from this test."""
|
|
return [r.getMessage() for r in caplog.records if "question=" in r.getMessage()]
|
|
|
|
|
|
# ---------- no row → byte-identical to the pre-phase prompts ----------
|
|
|
|
|
|
def test_no_row_high_prompt_byte_identical_to_pre_phase(
|
|
client: TestClient, db, seeded_kb: FakeRagLLM, caplog: pytest.LogCaptureFixture
|
|
) -> None:
|
|
"""No ``kb_overview`` row: the captured HIGH system prompt EQUALS the
|
|
pre-phase construction exactly — the section is absent, not empty."""
|
|
caplog.set_level(logging.INFO, logger="app.chat")
|
|
fastapi_app.dependency_overrides[chat_api.get_llm] = lambda: seeded_kb
|
|
try:
|
|
_, _, frames = _stream_chat(client, QUESTION)
|
|
finally:
|
|
fastapi_app.dependency_overrides.clear()
|
|
|
|
assert frames[-1]["deflected"] is False
|
|
(system, user) = seeded_kb.seen_messages[0][0], seeded_kb.seen_messages[0][1]
|
|
assert user["content"] == QUESTION
|
|
# Phase 119 (D3): the endpoint's suggested-folder lines ride the
|
|
# HIGH prompt — reconstructed the same deterministic way.
|
|
expected = build_high_prompt(
|
|
_suggested_docs(db),
|
|
notes=[],
|
|
kb_overview=None,
|
|
folder_lines=_suggested_folder_lines(db),
|
|
)
|
|
assert system["content"] == expected
|
|
assert "<knowledge_base>" not in system["content"]
|
|
|
|
lines = _turn_log_lines(caplog)
|
|
assert lines and "kb_chars=0" in lines[-1]
|
|
|
|
|
|
def test_no_row_low_prompt_byte_identical_to_pre_phase(
|
|
client: TestClient, db, seeded_kb: FakeRagLLM, caplog: pytest.LogCaptureFixture
|
|
) -> None:
|
|
"""No row, off-topic question: the captured LOW (deflection) prompt
|
|
EQUALS the pre-phase construction exactly."""
|
|
caplog.set_level(logging.INFO, logger="app.chat")
|
|
fastapi_app.dependency_overrides[chat_api.get_llm] = lambda: seeded_kb
|
|
try:
|
|
_, _, frames = _stream_chat(client, OFF_TOPIC)
|
|
finally:
|
|
fastapi_app.dependency_overrides.clear()
|
|
|
|
assert frames[-1]["deflected"] is True
|
|
(system, user) = seeded_kb.seen_messages[0][0], seeded_kb.seen_messages[0][1]
|
|
assert user["content"] == OFF_TOPIC
|
|
# Reconstruct the LOW prompt the way plan_turn does — the pre-phase
|
|
# construction (kb_overview=None), the same deterministic retrieval.
|
|
chunks = retrieve(db, OFF_TOPIC, _token_vec(OFF_TOPIC))
|
|
expected = build_deflect_prompt(
|
|
weak_hit_titles(chunks), notes=[], kb_overview=None
|
|
)
|
|
assert system["content"] == expected
|
|
assert "<knowledge_base>" not in system["content"]
|
|
assert "DEFLECT_MODE" in system["content"]
|
|
|
|
lines = _turn_log_lines(caplog)
|
|
assert lines and "kb_chars=0" in lines[-1]
|
|
|
|
|
|
# ---------- row present → section in BOTH prompts, exactly ----------
|
|
|
|
|
|
def test_row_high_prompt_carries_kb_section_exactly(
|
|
client: TestClient, db, seeded_kb: FakeRagLLM, caplog: pytest.LogCaptureFixture
|
|
) -> None:
|
|
"""Stored row: the captured HIGH prompt EQUALS the construction with
|
|
the outline — section present, ordered before ``<documents>``."""
|
|
_seed_overview(db)
|
|
caplog.set_level(logging.INFO, logger="app.chat")
|
|
fastapi_app.dependency_overrides[chat_api.get_llm] = lambda: seeded_kb
|
|
try:
|
|
_, _, frames = _stream_chat(client, QUESTION)
|
|
finally:
|
|
fastapi_app.dependency_overrides.clear()
|
|
|
|
assert frames[-1]["deflected"] is False
|
|
(system, _user) = seeded_kb.seen_messages[0][0], seeded_kb.seen_messages[0][1]
|
|
# Phase 119 (D3): the suggested-folder lines ride the HIGH prompt
|
|
# alongside the <knowledge_base> section — reconstructed the same
|
|
# deterministic way.
|
|
expected = build_high_prompt(
|
|
_suggested_docs(db),
|
|
notes=[],
|
|
kb_overview=OVERVIEW,
|
|
folder_lines=_suggested_folder_lines(db),
|
|
)
|
|
assert system["content"] == expected
|
|
|
|
# Section shape + order: <relevance> → <knowledge_base> → <documents>.
|
|
prompt = system["content"]
|
|
assert (
|
|
prompt.index("<relevance>HIGH</relevance>")
|
|
< prompt.index("<knowledge_base>")
|
|
< prompt.index(OVERVIEW)
|
|
< prompt.index("</knowledge_base>")
|
|
< prompt.index("<documents>")
|
|
)
|
|
|
|
# The per-turn log line records the outline's length (PLAN §9).
|
|
lines = _turn_log_lines(caplog)
|
|
assert lines and f"kb_chars={len(OVERVIEW)}" in lines[-1]
|
|
|
|
|
|
def test_row_low_prompt_carries_kb_section_exactly(
|
|
client: TestClient, db, seeded_kb: FakeRagLLM, caplog: pytest.LogCaptureFixture
|
|
) -> None:
|
|
"""Stored row, off-topic question: the LOW prompt EQUALS the
|
|
construction with the outline — the section is in the deflection
|
|
prompt too (real alternatives, not hallucinated ones)."""
|
|
_seed_overview(db)
|
|
caplog.set_level(logging.INFO, logger="app.chat")
|
|
fastapi_app.dependency_overrides[chat_api.get_llm] = lambda: seeded_kb
|
|
try:
|
|
_, _, frames = _stream_chat(client, OFF_TOPIC)
|
|
finally:
|
|
fastapi_app.dependency_overrides.clear()
|
|
|
|
assert frames[-1]["deflected"] is True
|
|
(system, _user) = seeded_kb.seen_messages[0][0], seeded_kb.seen_messages[0][1]
|
|
chunks = retrieve(db, OFF_TOPIC, _token_vec(OFF_TOPIC))
|
|
expected = build_deflect_prompt(
|
|
weak_hit_titles(chunks), notes=[], kb_overview=OVERVIEW
|
|
)
|
|
assert system["content"] == expected
|
|
|
|
prompt = system["content"]
|
|
assert (
|
|
prompt.index("<relevance>LOW</relevance>")
|
|
< prompt.index("<knowledge_base>")
|
|
< prompt.index(OVERVIEW)
|
|
< prompt.index("</knowledge_base>")
|
|
< prompt.index("DEFLECT_MODE")
|
|
)
|
|
# Deflection still sees titles only — never document content.
|
|
assert "Talos Linux" not in prompt
|
|
|
|
lines = _turn_log_lines(caplog)
|
|
assert lines and f"kb_chars={len(OVERVIEW)}" in lines[-1]
|
|
|
|
|
|
def test_row_reread_every_turn_and_deleted_row_stops_it(
|
|
client: TestClient, db, seeded_kb: FakeRagLLM
|
|
) -> None:
|
|
"""The row is read per turn (not cached): it steers every turn until
|
|
it is deleted, and the following turn is section-free again."""
|
|
_seed_overview(db)
|
|
fastapi_app.dependency_overrides[chat_api.get_llm] = lambda: seeded_kb
|
|
try:
|
|
_stream_chat(client, QUESTION)
|
|
_stream_chat(client, QUESTION)
|
|
assert len(seeded_kb.seen_messages) == 2
|
|
for messages in seeded_kb.seen_messages:
|
|
assert "<knowledge_base>" in messages[0]["content"]
|
|
assert OVERVIEW in messages[0]["content"]
|
|
|
|
# Delete the row → the next turn's prompt drops the section.
|
|
db.execute(text("TRUNCATE kb_overview"))
|
|
db.commit()
|
|
_stream_chat(client, QUESTION)
|
|
assert len(seeded_kb.seen_messages) == 3
|
|
assert "<knowledge_base>" not in seeded_kb.seen_messages[-1][0]["content"]
|
|
finally:
|
|
fastapi_app.dependency_overrides.clear()
|