feat(rag): hybrid FTS+vector retrieval and multi-format ingestion — name-your-tool questions find the right document

This commit is contained in:
2026-08-22 01:27:02 -04:00
parent 2f738a7f19
commit 7e8d14702e
36 changed files with 2018 additions and 290 deletions
+197
View File
@@ -0,0 +1,197 @@
"""Phase 09 E2E (Playwright): retrieval quality — hybrid search end to end.
Story: ``.agent/user_stories/retrieval-quality.md``
Run in isolation (DB must be up: ``podman compose up -d db``):
uv run pytest tests/e2e/test_retrieval_quality.py -v --no-cov
Seeding reuses the real importer against ``tests/fixtures/docs/`` with the
deterministic mock embeddings (same pattern as the earlier story suites).
The four tests map the story's acceptance criteria:
1. multi-format fixture import — hidden doc excluded, ``/api/docs`` counts
2. "How did I install gitlab?" — grounded (not deflected), gitlab chip,
``query_log`` row with the gitlab doc in ``sources``
3. keyword-only question ("kafkabridge") beats the vector ranking — the
FTS-OR gate grounds it end to end despite weak cosine
4. "sourdough" — deflected bubble + ≥2 "Maybe try" chips
"""
from __future__ import annotations
import asyncio
from pathlib import Path
from threading import Thread
from typing import Any
import httpx
from playwright.sync_api import Page, expect
from sqlalchemy import select, text
from app.config import Settings, get_settings
from app.db import SessionLocal
from app.models import QueryLog
from app.rag.importer import ImportSummary, import_sources
from app.rag.llm import LLMClient
REPO = Path(__file__).resolve().parents[2]
FIXTURES = REPO / "tests" / "fixtures" / "docs"
GITLAB_QUESTION = "How did I install gitlab?"
KEYWORD_QUESTION = "How does kafkabridge work?"
OFF_TOPIC = "sourdough starter"
MOCK_ANSWER_MARKER = "Deterministic mock answer for E2E"
async def _import_fixtures(mock_port: int) -> ImportSummary:
kwargs: dict[str, Any] = {"_env_file": None, "llm_base_url": f"http://127.0.0.1:{mock_port}/v1"}
settings = Settings(**kwargs) # pyright: ignore[reportCallIssue]
return await import_sources([FIXTURES], LLMClient(settings))
def _run_in_thread(coro: Any) -> Any:
"""Run a coroutine on a worker thread (Playwright owns the test loop)."""
box: dict[str, Any] = {}
def runner() -> None:
try:
box["value"] = asyncio.run(coro)
except BaseException as e: # noqa: BLE001 — re-raised on the test thread
box["error"] = e
t = Thread(target=runner)
t.start()
t.join()
if "error" in box:
raise box["error"]
return box["value"]
def _reset_db(mock_port: int, seed: bool) -> ImportSummary | None:
"""Truncate the KB (and query log), then optionally re-import fixtures."""
with SessionLocal() as db:
db.execute(text("TRUNCATE chunks, documents, query_log"))
db.commit()
if not seed:
return None
return _run_in_thread(_import_fixtures(mock_port))
def _ask(page: Page, message: str) -> None:
page.fill("#message-input", message)
page.click("#send-btn")
def test_multi_format_import_hidden_doc_excluded(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
"""A9 (revised): all seven formats import; hidden (dot) paths never do."""
summary = _reset_db(mock_llm, seed=True)
assert summary is not None
# Eight A9-format fixture files; .hidden/junk.md must never be walked.
assert summary.added == 8
assert summary.formats == {"md": 4, "yaml": 1, "json": 1, "py": 1, "txt": 1}
r = httpx.get(f"{app_url}/api/docs", timeout=10)
assert r.status_code == 200
docs = r.json()["documents"]
assert len(docs) == 8
assert all(".hidden" not in d["path"] for d in docs)
assert {d["path"] for d in docs} >= {
"homelab/container_gitlab/gitlab.md",
"homelab/container_gitlab/gitlab-compose.yaml",
"homelab/networking/static-dns.json",
"homelab/scripts/uptime_probe.py",
"homelab/ssh/ssh_aliases.txt",
}
# The Sources page reflects the same set.
page.goto(f"{app_url}/sources.html")
expect(page.locator("#stat-docs")).to_have_text("8")
expect(page.locator("#docs-tbody tr", has_text=".hidden")).to_have_count(0)
def test_gitlab_question_is_grounded_with_gitlab_chip(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
"""The ranking problem that motivated this phase: a tool-name question
must land on the tool's own document — not a generic template."""
_reset_db(mock_llm, seed=True)
page.set_default_timeout(30_000)
page.goto(app_url)
_ask(page, GITLAB_QUESTION)
expect(page.locator(".msg.user .bubble")).to_contain_text(GITLAB_QUESTION)
bubble = page.locator(".msg.brain .bubble").first
bubble.wait_for(state="visible", timeout=30_000)
expect(bubble).to_contain_text(MOCK_ANSWER_MARKER, timeout=30_000)
# Grounded: no deflected bubble at all.
expect(page.locator(".msg.brain.is-deflected")).to_have_count(0)
# The gitlab document is cited (a chip carrying its path).
chip = page.locator(".msg.brain .source-chip", has_text="container_gitlab/gitlab.md")
expect(chip).to_have_count(1, timeout=30_000)
# Durable record: not deflected, and the gitlab doc is in sources.
with SessionLocal() as db:
row = db.scalars(select(QueryLog)).one()
assert row.question == GITLAB_QUESTION
assert row.deflected is False
assert "container_gitlab/gitlab.md" in row.sources
def test_keyword_only_question_beats_vector_ranking(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
"""The FTS-OR gate end to end: "kafkabridge" appears in exactly one
fixture doc (static-dns.json) and the question's cosine overlap is
weak — the lexical branch is what grounds the answer."""
_reset_db(mock_llm, seed=True)
page.set_default_timeout(30_000)
page.goto(app_url)
_ask(page, KEYWORD_QUESTION)
bubble = page.locator(".msg.brain .bubble").first
bubble.wait_for(state="visible", timeout=30_000)
expect(bubble).to_contain_text(MOCK_ANSWER_MARKER, timeout=30_000)
expect(page.locator(".msg.brain.is-deflected")).to_have_count(0)
# The FTS-matched doc is the TOP source chip (it beats the vector rank).
first_chip = page.locator(".msg.brain .source-chip").first
first_chip.wait_for(state="visible", timeout=30_000)
expect(first_chip).to_contain_text("static-dns.json")
with SessionLocal() as db:
row = db.scalars(select(QueryLog)).one()
# Weak vector score…
assert row.top_score < get_settings().relevance_threshold
# …but a lexical hit grounded it (the FTS-OR branch).
assert (row.fts_hits or 0) >= 1
assert row.deflected is False
assert "homelab/networking/static-dns.json" in row.sources
def test_off_topic_still_deflects_with_chips(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
"""A8 (revised): deflection requires weak cosine AND zero FTS hits.
"sourdough" matches nothing in the KB lexically → honest deflection."""
_reset_db(mock_llm, seed=True)
page.set_default_timeout(30_000)
page.goto(app_url)
_ask(page, OFF_TOPIC)
bubble = page.locator(".msg.brain.is-deflected .bubble").first
bubble.wait_for(state="visible", timeout=30_000)
expect(page.locator(".msg.brain.is-deflected")).to_have_count(1)
chips = page.locator(".msg.brain.is-deflected .maybe-try .suggestion-chip")
expect(chips.first).to_be_visible(timeout=30_000)
assert chips.count() >= 2, "deflection must offer 2-3 alternative chips"
assert all(c.strip() for c in chips.all_inner_texts())
with SessionLocal() as db:
row = db.scalars(select(QueryLog)).one()
assert row.question == OFF_TOPIC
assert row.deflected is True
assert 0.0 < row.top_score < get_settings().relevance_threshold
assert row.fts_hits == 0 # deflection is only reached with zero hits