**Phase 112 — final verification pass (all 4 tasks already complete in `complete/`):** - Verified gate fix: `app/api/chat.py::plan_turn` — HIGH iff `best_cosine >= relevance_threshold` OR (`fts_hits > 0` AND `best_cosine >= lexical_support_floor`); `lexical_support_floor` (default 0.35, `BOR_LEXICAL_SUPPORT_FLOOR`, bounds-validated) in `app/config.py` + `.env.example`; A8 revision note (2026-09-14) in `.agents/PLAN.md`. - Verified prompt contract: `app/rag/prompts.py` diff is docstring-only (dated owner-decision-iii entry); `tests/unit/test_prompt_lock.py` byte-pins PERSONA/TOOLS_SECTION/DEFLECT body (sha256+length). - Verified README: L11 + L575 deflection copy refreshed; `grep "haven't done anything" README.md` → no hits; disclosed-answer behavior documented. - Tests: `uv run pytest --cov=app --cov-report=term-missing` → **2378 passed, 99% coverage (>90%)**; includes Mongolia-quadrant unit pins (fts>0 + cosine<floor → LOW). - E2E in isolation: `uv run pytest tests/e2e/test_honest_deflection.py -v --no-cov` → **4 passed** (out-of-KB question: `deflected=true`, `sources==[]`, 2–3 suggestions); regression `test_chat_rag.py` + `test_retrieval_quality.py` → **7 passed**. - Lint/types: `uv run ruff check .` → clean; `uv run pyright` → 0 errors. **Completion criteria:** weak-FTS→LOW unit-pinned ✅ · no false citations + 2–3 alternatives E2E ✅ · prompts byte-identical (test-pinned) + README matches ✅ · suite/coverage/e2e/lint all green ✅ · commit + phase move → left to harness (no `git commit` run, per rules; changes in working tree). **Deviations:** none. Next pending phase: `113_source_chip_quality`.
231 lines
9.7 KiB
Python
231 lines
9.7 KiB
Python
"""Phase 09 E2E (Playwright): retrieval quality — hybrid search end to end.
|
|
|
|
Story: ``.agents/user_stories/retrieval-quality.md``
|
|
Run in isolation (DB must be up: ``podman compose up -d db``):
|
|
|
|
uv run pytest tests/e2e/test_retrieval_quality.py -v --no-cov
|
|
|
|
Seeding reuses the real importer against ``tests/fixtures/docs/`` with the
|
|
deterministic mock embeddings (same pattern as the earlier story suites).
|
|
The four tests map the story's acceptance criteria:
|
|
|
|
1. multi-format fixture import — hidden doc excluded, ``/api/docs`` counts
|
|
2. "How did I install gitlab?" — grounded (not deflected), gitlab chip,
|
|
``query_log`` row with the gitlab doc in ``sources``
|
|
3. keyword-only question ("kafkabridge") beats the vector ranking — the
|
|
corroborated-lexical gate (A8 revised 2026-09-14) grounds it end to
|
|
end: weak cosine, but an FTS hit AND cosine >= lexical_support_floor
|
|
4. "sourdough" — deflected bubble + ≥2 "Maybe try" chips
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
from pathlib import Path
|
|
from threading import Thread
|
|
from typing import Any
|
|
|
|
import httpx
|
|
from playwright.sync_api import Page, expect
|
|
from sqlalchemy import select, text
|
|
|
|
from app.config import Settings, get_settings
|
|
from app.db import SessionLocal
|
|
from app.models import QueryLog
|
|
from app.rag.importer import ImportSummary, import_sources
|
|
from app.rag.llm import LLMClient
|
|
from e2e.auth_helpers import login
|
|
|
|
REPO = Path(__file__).resolve().parents[2]
|
|
FIXTURES = REPO / "tests" / "fixtures" / "docs"
|
|
GITLAB_QUESTION = "How did I install gitlab?"
|
|
# Phase 112 (A8 revised): the pre-phase question ("How does kafkabridge
|
|
# work?" — mock cosine 0.134) now sits BELOW lexical_support_floor
|
|
# (0.15, mock-calibrated) with fts>0 — the new gate's deflection
|
|
# quadrant, so it can no longer demonstrate the grounded lexical path.
|
|
# "handle DNS" adds static-dns.json's own tokens: cosine ≈0.24 — still
|
|
# weak (below the 0.30 threshold) yet corroborated (>= floor) with the
|
|
# same single-doc FTS hit, and the FTS-matched doc still tops the fused
|
|
# ranking (the test's actual assertion).
|
|
KEYWORD_QUESTION = "How does kafkabridge handle DNS?"
|
|
OFF_TOPIC = "sourdough starter"
|
|
MOCK_ANSWER_MARKER = "Deterministic mock answer for E2E"
|
|
|
|
|
|
async def _import_fixtures(mock_port: int) -> ImportSummary:
|
|
kwargs: dict[str, Any] = {"_env_file": None, "llm_base_url": f"http://127.0.0.1:{mock_port}/v1"}
|
|
settings = Settings(**kwargs) # pyright: ignore[reportCallIssue]
|
|
return await import_sources([FIXTURES], LLMClient(settings))
|
|
|
|
|
|
def _run_in_thread(coro: Any) -> Any:
|
|
"""Run a coroutine on a worker thread (Playwright owns the test loop)."""
|
|
box: dict[str, Any] = {}
|
|
|
|
def runner() -> None:
|
|
try:
|
|
box["value"] = asyncio.run(coro)
|
|
except BaseException as e: # noqa: BLE001 — re-raised on the test thread
|
|
box["error"] = e
|
|
|
|
t = Thread(target=runner)
|
|
t.start()
|
|
t.join()
|
|
if "error" in box:
|
|
raise box["error"]
|
|
return box["value"]
|
|
|
|
|
|
def _reset_db(mock_port: int, seed: bool) -> ImportSummary | None:
|
|
"""Truncate the KB (and query log), then optionally re-import fixtures."""
|
|
with SessionLocal() as db:
|
|
db.execute(text("TRUNCATE chunks, documents, query_log"))
|
|
db.commit()
|
|
if not seed:
|
|
return None
|
|
return _run_in_thread(_import_fixtures(mock_port))
|
|
|
|
|
|
def _ask(page: Page, message: str) -> None:
|
|
page.fill("#message-input", message)
|
|
page.click("#send-btn")
|
|
|
|
|
|
def test_multi_format_import_hidden_doc_excluded(
|
|
page: Page, app_url: str, mock_llm: int, db_ready: None
|
|
) -> None:
|
|
"""A9 (revised): all seventeen formats import; hidden (dot) paths never do."""
|
|
summary = _reset_db(mock_llm, seed=True)
|
|
assert summary is not None
|
|
# Thirteen A9-format fixture files (phase 44 added homelab/tables.md,
|
|
# phase 47 added the quadlet + j2 fixtures); .hidden/junk.md must never
|
|
# be walked.
|
|
assert summary.added == 13
|
|
assert summary.formats == {
|
|
"md": 5, "yaml": 1, "json": 1, "py": 1, "txt": 1,
|
|
"container": 1, "network": 1, "volume": 1, "j2": 1, # phase 47
|
|
}
|
|
|
|
# Phase 16: the catalog is admin-only — perform the real form login,
|
|
# then call the API with the signed cookie the browser now holds.
|
|
login(page, app_url, next="/sources.html")
|
|
cookies = {
|
|
c["name"]: c["value"]
|
|
for c in page.context.cookies()
|
|
if "name" in c and "value" in c
|
|
}
|
|
r = httpx.get(f"{app_url}/api/docs", timeout=10, cookies=cookies)
|
|
assert r.status_code == 200
|
|
docs = r.json()["documents"]
|
|
assert len(docs) == 13
|
|
assert all(".hidden" not in d["path"] for d in docs)
|
|
assert {d["path"] for d in docs} >= {
|
|
"homelab/container_gitlab/gitlab.md",
|
|
"homelab/container_gitlab/gitlab-compose.yaml",
|
|
"homelab/networking/static-dns.json",
|
|
"homelab/scripts/uptime_probe.py",
|
|
"homelab/ssh/ssh_aliases.txt",
|
|
}
|
|
|
|
# The Sources page (we're already on it, signed in) reflects the set.
|
|
# Phase 97: the file table is hidden at the top level (the source
|
|
# row is the tree's top) — drill into the source, where a
|
|
# `.hidden` junk row WOULD appear, and scope the absence there
|
|
# (the stat card pins the 13 total: the junk is counted nowhere).
|
|
expect(page.locator("#stat-docs")).to_have_text("13")
|
|
page.locator("#folders-tbody .folder-link").first.wait_for(state="visible")
|
|
page.click('#folders-tbody a.folder-link:text-is("docs")')
|
|
expect(page.locator("#docs-tbody tr", has_text=".hidden")).to_have_count(0)
|
|
expect(page.locator("#folders-tbody .folder-link", has_text=".hidden")).to_have_count(0)
|
|
|
|
|
|
def test_gitlab_question_is_grounded_with_gitlab_chip(
|
|
page: Page, app_url: str, mock_llm: int, db_ready: None
|
|
) -> None:
|
|
"""The ranking problem that motivated this phase: a tool-name question
|
|
must land on the tool's own document — not a generic template."""
|
|
_reset_db(mock_llm, seed=True)
|
|
page.set_default_timeout(30_000)
|
|
login(page, app_url, next="/") # phase 79: chat is require_user-gated
|
|
|
|
_ask(page, GITLAB_QUESTION)
|
|
expect(page.locator(".msg.user .bubble")).to_contain_text(GITLAB_QUESTION)
|
|
|
|
bubble = page.locator(".msg.brain .bubble").first
|
|
bubble.wait_for(state="visible", timeout=30_000)
|
|
expect(bubble).to_contain_text(MOCK_ANSWER_MARKER, timeout=30_000)
|
|
# Grounded: no deflected bubble at all.
|
|
expect(page.locator(".msg.brain.is-deflected")).to_have_count(0)
|
|
|
|
# The gitlab document is cited (a chip carrying its path).
|
|
chip = page.locator(".msg.brain .source-chip", has_text="container_gitlab/gitlab.md")
|
|
expect(chip).to_have_count(1, timeout=30_000)
|
|
|
|
# Durable record: not deflected, and the gitlab doc is in sources.
|
|
with SessionLocal() as db:
|
|
row = db.scalars(select(QueryLog)).one()
|
|
assert row.question == GITLAB_QUESTION
|
|
assert row.deflected is False
|
|
assert "container_gitlab/gitlab.md" in row.sources
|
|
|
|
|
|
def test_keyword_only_question_beats_vector_ranking(
|
|
page: Page, app_url: str, mock_llm: int, db_ready: None
|
|
) -> None:
|
|
"""The corroborated-lexical gate end to end (A8 revised 2026-09-14):
|
|
"kafkabridge" appears in exactly one fixture doc (static-dns.json)
|
|
and the question's cosine overlap is weak (below the threshold) —
|
|
the lexical hit plus cosine >= lexical_support_floor is what grounds
|
|
the answer (a lexical-only hit below the floor would deflect)."""
|
|
_reset_db(mock_llm, seed=True)
|
|
page.set_default_timeout(30_000)
|
|
login(page, app_url, next="/") # phase 79: chat is require_user-gated
|
|
|
|
_ask(page, KEYWORD_QUESTION)
|
|
bubble = page.locator(".msg.brain .bubble").first
|
|
bubble.wait_for(state="visible", timeout=30_000)
|
|
expect(bubble).to_contain_text(MOCK_ANSWER_MARKER, timeout=30_000)
|
|
expect(page.locator(".msg.brain.is-deflected")).to_have_count(0)
|
|
|
|
# The FTS-matched doc is the TOP source chip (it beats the vector rank).
|
|
first_chip = page.locator(".msg.brain .source-chip").first
|
|
first_chip.wait_for(state="visible", timeout=30_000)
|
|
expect(first_chip).to_contain_text("static-dns.json")
|
|
|
|
with SessionLocal() as db:
|
|
row = db.scalars(select(QueryLog)).one()
|
|
# Weak vector score — below the answer threshold…
|
|
assert row.top_score < get_settings().relevance_threshold
|
|
# …but cleared the lexical support floor, and a lexical hit fired —
|
|
# the corroborated-lexical path (A8 revised 2026-09-14) grounded it.
|
|
assert row.top_score >= get_settings().lexical_support_floor
|
|
assert (row.fts_hits or 0) >= 1
|
|
assert row.deflected is False
|
|
assert "homelab/networking/static-dns.json" in row.sources
|
|
|
|
|
|
def test_off_topic_still_deflects_with_chips(
|
|
page: Page, app_url: str, mock_llm: int, db_ready: None
|
|
) -> None:
|
|
"""A8 (revised): deflection requires weak cosine AND zero FTS hits.
|
|
"sourdough" matches nothing in the KB lexically → honest deflection."""
|
|
_reset_db(mock_llm, seed=True)
|
|
page.set_default_timeout(30_000)
|
|
login(page, app_url, next="/") # phase 79: chat is require_user-gated
|
|
|
|
_ask(page, OFF_TOPIC)
|
|
bubble = page.locator(".msg.brain.is-deflected .bubble").first
|
|
bubble.wait_for(state="visible", timeout=30_000)
|
|
expect(page.locator(".msg.brain.is-deflected")).to_have_count(1)
|
|
|
|
chips = page.locator(".msg.brain.is-deflected .maybe-try .suggestion-chip")
|
|
expect(chips.first).to_be_visible(timeout=30_000)
|
|
assert chips.count() >= 2, "deflection must offer 2-3 alternative chips"
|
|
assert all(c.strip() for c in chips.all_inner_texts())
|
|
|
|
with SessionLocal() as db:
|
|
row = db.scalars(select(QueryLog)).one()
|
|
assert row.question == OFF_TOPIC
|
|
assert row.deflected is True
|
|
assert 0.0 < row.top_score < get_settings().relevance_threshold
|
|
assert row.fts_hits == 0 # deflection is only reached with zero hits
|