Files
brain-of-reese/tests/e2e/test_retrieval_quality.py
T
ducoterra 7fce6572d0
Build and Push Containers / build-and-push-app (push) Successful in 1m45s
Build and Push Containers / build-and-push-db (push) Successful in 13s
feat: phases 77–80 — navbar view refresh, static background, API tokens, history suggestion chips
Single consolidated commit for four completed, validated phases (77, 78,
79, 80). The pipeline run left all work uncommitted because the harness
commits only with PHASE_COMMIT=1 while child executors are forbidden from
committing; the phases themselves all passed validation and moved to
.agents/phases/complete/.

Phase 77 — navbar view refresh
- router.js dispatches bor:view-refresh on re-show / active re-click /
  popstate (gated on wasMounted; first show and boot exempt)
- History / RAG / Sources / Tuning re-fetch on refresh (admin branch);
  Chat deliberately excluded (stream survival)
- History "Refresh" button (admin-only, in-flight disable + status line)
- New story suite tests/e2e/test_navbar_refresh.py (7 tests)

Phase 78 — static background
- Removed the animated glow layers; static 44px grid over the flat --bg
  canvas; default and reduced-motion renders byte-identical
- Updated background/theme E2E suites; removed bg-glow test pins

Phase 79 — API tokens
- api_tokens model + migration 0012; hash-only token service
- Admin tokens API + Tokens admin view; POST /api/token-auth;
  live-revoking require_user on chat / suggestions / document content
- Frontend token gate with localStorage cache; anonymous E2E suites
  migrated to token login
- New story suite tests/e2e/test_api_tokens.py (9 tests)

Phase 80 — history suggestion chips
- last_questions() endpoint with SEED fallback; startNewChat() refetch
- Seed-semantics docs (config.py, .env.example, README)
- Integration state matrix + E2E suite rewritten to the 4 chip states

Also included: phase-76 report artifacts and the repo restore-test-db
skill (previously untracked), scripts/* ruff fixes from phase 77.

Final gate state (phase 80 final pass, covers everything above):
- uv run pytest --cov=app → 1637 passed, 0 failed, app/ coverage 99%
- uv run ruff check . && uv run pyright → clean, 0 errors
- Per-phase story E2E suites green in isolation
2026-09-07 12:39:01 -04:00

211 lines
8.2 KiB
Python

"""Phase 09 E2E (Playwright): retrieval quality — hybrid search end to end.
Story: ``.agents/user_stories/retrieval-quality.md``
Run in isolation (DB must be up: ``podman compose up -d db``):
uv run pytest tests/e2e/test_retrieval_quality.py -v --no-cov
Seeding reuses the real importer against ``tests/fixtures/docs/`` with the
deterministic mock embeddings (same pattern as the earlier story suites).
The four tests map the story's acceptance criteria:
1. multi-format fixture import — hidden doc excluded, ``/api/docs`` counts
2. "How did I install gitlab?" — grounded (not deflected), gitlab chip,
``query_log`` row with the gitlab doc in ``sources``
3. keyword-only question ("kafkabridge") beats the vector ranking — the
FTS-OR gate grounds it end to end despite weak cosine
4. "sourdough" — deflected bubble + ≥2 "Maybe try" chips
"""
from __future__ import annotations
import asyncio
from pathlib import Path
from threading import Thread
from typing import Any
import httpx
from playwright.sync_api import Page, expect
from sqlalchemy import select, text
from app.config import Settings, get_settings
from app.db import SessionLocal
from app.models import QueryLog
from app.rag.importer import ImportSummary, import_sources
from app.rag.llm import LLMClient
from e2e.auth_helpers import login
REPO = Path(__file__).resolve().parents[2]
FIXTURES = REPO / "tests" / "fixtures" / "docs"
GITLAB_QUESTION = "How did I install gitlab?"
KEYWORD_QUESTION = "How does kafkabridge work?"
OFF_TOPIC = "sourdough starter"
MOCK_ANSWER_MARKER = "Deterministic mock answer for E2E"
async def _import_fixtures(mock_port: int) -> ImportSummary:
kwargs: dict[str, Any] = {"_env_file": None, "llm_base_url": f"http://127.0.0.1:{mock_port}/v1"}
settings = Settings(**kwargs) # pyright: ignore[reportCallIssue]
return await import_sources([FIXTURES], LLMClient(settings))
def _run_in_thread(coro: Any) -> Any:
"""Run a coroutine on a worker thread (Playwright owns the test loop)."""
box: dict[str, Any] = {}
def runner() -> None:
try:
box["value"] = asyncio.run(coro)
except BaseException as e: # noqa: BLE001 — re-raised on the test thread
box["error"] = e
t = Thread(target=runner)
t.start()
t.join()
if "error" in box:
raise box["error"]
return box["value"]
def _reset_db(mock_port: int, seed: bool) -> ImportSummary | None:
"""Truncate the KB (and query log), then optionally re-import fixtures."""
with SessionLocal() as db:
db.execute(text("TRUNCATE chunks, documents, query_log"))
db.commit()
if not seed:
return None
return _run_in_thread(_import_fixtures(mock_port))
def _ask(page: Page, message: str) -> None:
page.fill("#message-input", message)
page.click("#send-btn")
def test_multi_format_import_hidden_doc_excluded(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
"""A9 (revised): all seventeen formats import; hidden (dot) paths never do."""
summary = _reset_db(mock_llm, seed=True)
assert summary is not None
# Thirteen A9-format fixture files (phase 44 added homelab/tables.md,
# phase 47 added the quadlet + j2 fixtures); .hidden/junk.md must never
# be walked.
assert summary.added == 13
assert summary.formats == {
"md": 5, "yaml": 1, "json": 1, "py": 1, "txt": 1,
"container": 1, "network": 1, "volume": 1, "j2": 1, # phase 47
}
# Phase 16: the catalog is admin-only — perform the real form login,
# then call the API with the signed cookie the browser now holds.
login(page, app_url, next="/sources.html")
cookies = {
c["name"]: c["value"]
for c in page.context.cookies()
if "name" in c and "value" in c
}
r = httpx.get(f"{app_url}/api/docs", timeout=10, cookies=cookies)
assert r.status_code == 200
docs = r.json()["documents"]
assert len(docs) == 13
assert all(".hidden" not in d["path"] for d in docs)
assert {d["path"] for d in docs} >= {
"homelab/container_gitlab/gitlab.md",
"homelab/container_gitlab/gitlab-compose.yaml",
"homelab/networking/static-dns.json",
"homelab/scripts/uptime_probe.py",
"homelab/ssh/ssh_aliases.txt",
}
# The Sources page (we're already on it, signed in) reflects the set.
expect(page.locator("#stat-docs")).to_have_text("13")
expect(page.locator("#docs-tbody tr", has_text=".hidden")).to_have_count(0)
def test_gitlab_question_is_grounded_with_gitlab_chip(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
"""The ranking problem that motivated this phase: a tool-name question
must land on the tool's own document — not a generic template."""
_reset_db(mock_llm, seed=True)
page.set_default_timeout(30_000)
login(page, app_url, next="/") # phase 79: chat is require_user-gated
_ask(page, GITLAB_QUESTION)
expect(page.locator(".msg.user .bubble")).to_contain_text(GITLAB_QUESTION)
bubble = page.locator(".msg.brain .bubble").first
bubble.wait_for(state="visible", timeout=30_000)
expect(bubble).to_contain_text(MOCK_ANSWER_MARKER, timeout=30_000)
# Grounded: no deflected bubble at all.
expect(page.locator(".msg.brain.is-deflected")).to_have_count(0)
# The gitlab document is cited (a chip carrying its path).
chip = page.locator(".msg.brain .source-chip", has_text="container_gitlab/gitlab.md")
expect(chip).to_have_count(1, timeout=30_000)
# Durable record: not deflected, and the gitlab doc is in sources.
with SessionLocal() as db:
row = db.scalars(select(QueryLog)).one()
assert row.question == GITLAB_QUESTION
assert row.deflected is False
assert "container_gitlab/gitlab.md" in row.sources
def test_keyword_only_question_beats_vector_ranking(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
"""The FTS-OR gate end to end: "kafkabridge" appears in exactly one
fixture doc (static-dns.json) and the question's cosine overlap is
weak — the lexical branch is what grounds the answer."""
_reset_db(mock_llm, seed=True)
page.set_default_timeout(30_000)
login(page, app_url, next="/") # phase 79: chat is require_user-gated
_ask(page, KEYWORD_QUESTION)
bubble = page.locator(".msg.brain .bubble").first
bubble.wait_for(state="visible", timeout=30_000)
expect(bubble).to_contain_text(MOCK_ANSWER_MARKER, timeout=30_000)
expect(page.locator(".msg.brain.is-deflected")).to_have_count(0)
# The FTS-matched doc is the TOP source chip (it beats the vector rank).
first_chip = page.locator(".msg.brain .source-chip").first
first_chip.wait_for(state="visible", timeout=30_000)
expect(first_chip).to_contain_text("static-dns.json")
with SessionLocal() as db:
row = db.scalars(select(QueryLog)).one()
# Weak vector score…
assert row.top_score < get_settings().relevance_threshold
# …but a lexical hit grounded it (the FTS-OR branch).
assert (row.fts_hits or 0) >= 1
assert row.deflected is False
assert "homelab/networking/static-dns.json" in row.sources
def test_off_topic_still_deflects_with_chips(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
"""A8 (revised): deflection requires weak cosine AND zero FTS hits.
"sourdough" matches nothing in the KB lexically → honest deflection."""
_reset_db(mock_llm, seed=True)
page.set_default_timeout(30_000)
login(page, app_url, next="/") # phase 79: chat is require_user-gated
_ask(page, OFF_TOPIC)
bubble = page.locator(".msg.brain.is-deflected .bubble").first
bubble.wait_for(state="visible", timeout=30_000)
expect(page.locator(".msg.brain.is-deflected")).to_have_count(1)
chips = page.locator(".msg.brain.is-deflected .maybe-try .suggestion-chip")
expect(chips.first).to_be_visible(timeout=30_000)
assert chips.count() >= 2, "deflection must offer 2-3 alternative chips"
assert all(c.strip() for c in chips.all_inner_texts())
with SessionLocal() as db:
row = db.scalars(select(QueryLog)).one()
assert row.question == OFF_TOPIC
assert row.deflected is True
assert 0.0 < row.top_score < get_settings().relevance_threshold
assert row.fts_hits == 0 # deflection is only reached with zero hits