feat(rag): hybrid FTS+vector retrieval and multi-format ingestion — name-your-tool questions find the right document
This commit is contained in:
@@ -82,6 +82,12 @@ def app_server(mock_llm: int) -> Iterator[str]:
|
||||
if USE_REAL_LLM
|
||||
else f"http://127.0.0.1:{mock_llm}/v1"
|
||||
)
|
||||
# The E2E mock's token-overlap embeddings have their own score
|
||||
# distribution (phase 09) — the app under test gets the mock-calibrated
|
||||
# threshold so every story suite keeps its deterministic gate behavior.
|
||||
# The production default stays 0.62 (re-tuned against the real
|
||||
# `embed` model's 0.41–0.84 cosine range, PLAN A8).
|
||||
env["BOR_RELEVANCE_THRESHOLD"] = "0.30"
|
||||
env.setdefault("BOR_DATABASE_URL", "postgresql+psycopg://reese:reese@localhost:5432/brain_of_reese")
|
||||
proc = subprocess.Popen(
|
||||
[sys.executable, "-m", "uvicorn", "app.main:app",
|
||||
|
||||
@@ -76,7 +76,7 @@ def test_on_topic_question_streams_grounded_answer(
|
||||
page: Page, app_url: str, mock_llm: int, db_ready: None
|
||||
) -> None:
|
||||
summary = _reset_db(mock_llm, seed=True)
|
||||
assert summary is not None and summary.added == 3
|
||||
assert summary is not None and summary.added == 8 # A9 formats
|
||||
page.set_default_timeout(30_000)
|
||||
page.goto(app_url)
|
||||
|
||||
|
||||
@@ -111,7 +111,7 @@ def _seed_kb(mock_port: int) -> ImportSummary:
|
||||
db.execute(text("TRUNCATE chunks, documents, query_log"))
|
||||
db.commit()
|
||||
summary = _run_in_thread(_import_fixtures(mock_port))
|
||||
assert summary is not None and summary.added == 3
|
||||
assert summary is not None and summary.added == 8 # A9 formats
|
||||
return summary
|
||||
|
||||
|
||||
|
||||
@@ -84,7 +84,7 @@ def test_off_topic_question_deflects_honestly(
|
||||
page: Page, app_url: str, mock_llm: int, db_ready: None
|
||||
) -> None:
|
||||
summary = _reset_db(mock_llm, seed=True)
|
||||
assert summary is not None and summary.added == 3
|
||||
assert summary is not None and summary.added == 8 # A9 formats
|
||||
page.set_default_timeout(30_000)
|
||||
page.goto(app_url)
|
||||
expect(page.locator("#kb-banner")).to_be_hidden()
|
||||
|
||||
@@ -31,6 +31,11 @@ EXPECTED_ROWS = (
|
||||
"homelab/kubernetes.md",
|
||||
"homelab/backups.md",
|
||||
"deployments/new-service.md",
|
||||
"homelab/container_gitlab/gitlab.md",
|
||||
"homelab/container_gitlab/gitlab-compose.yaml",
|
||||
"homelab/networking/static-dns.json",
|
||||
"homelab/scripts/uptime_probe.py",
|
||||
"homelab/ssh/ssh_aliases.txt",
|
||||
)
|
||||
|
||||
|
||||
@@ -76,16 +81,21 @@ def test_sources_page_lists_indexed_docs(
|
||||
page: Page, app_url: str, mock_llm: int, db_ready: None
|
||||
) -> None:
|
||||
summary = _reset_db(mock_llm, seed=True)
|
||||
assert summary is not None and summary.added == 3
|
||||
# Eight A9-format files are imported; .hidden/junk.md is out of scope
|
||||
# (A9 revised — hidden path components are never walked).
|
||||
assert summary is not None and summary.added == 8
|
||||
assert summary.formats == {"md": 4, "yaml": 1, "json": 1, "py": 1, "txt": 1}
|
||||
|
||||
page.goto(f"{app_url}/sources.html")
|
||||
expect(page.locator("#stat-docs")).to_have_text("3")
|
||||
expect(page.locator("#stat-docs")).to_have_text("8")
|
||||
expect(page.locator("#stat-chunks")).to_have_text(str(summary.chunks))
|
||||
expect(page.locator("#stat-last")).not_to_have_text("–")
|
||||
expect(page.locator("#sources-empty")).to_be_hidden()
|
||||
|
||||
for row_path in EXPECTED_ROWS:
|
||||
expect(page.locator("#docs-tbody tr", has_text=row_path)).to_have_count(1)
|
||||
# The hidden junk was never indexed (A9 scope).
|
||||
expect(page.locator("#docs-tbody tr", has_text=".hidden")).to_have_count(0)
|
||||
# The path column carries the full path for hover (ellipsis is visual only).
|
||||
expect(page.locator("#docs-tbody tr", has_text="homelab/kubernetes.md")
|
||||
.get_by_role("cell").nth(1)).to_have_attribute("title", "homelab/kubernetes.md")
|
||||
|
||||
@@ -176,7 +176,7 @@ def test_typing_indicator_during_slow_think(
|
||||
"""AC1/AC5: the 3s mock warm-up must show the typing indicator for
|
||||
>=2s before any text appears, then it is gone once the answer lands."""
|
||||
summary = _reset_db(mock_llm, seed=True)
|
||||
assert summary is not None and summary.added == 3
|
||||
assert summary is not None and summary.added == 8 # A9 formats
|
||||
page.set_default_timeout(30_000)
|
||||
page.goto(app_url)
|
||||
|
||||
|
||||
@@ -0,0 +1,197 @@
|
||||
"""Phase 09 E2E (Playwright): retrieval quality — hybrid search end to end.
|
||||
|
||||
Story: ``.agent/user_stories/retrieval-quality.md``
|
||||
Run in isolation (DB must be up: ``podman compose up -d db``):
|
||||
|
||||
uv run pytest tests/e2e/test_retrieval_quality.py -v --no-cov
|
||||
|
||||
Seeding reuses the real importer against ``tests/fixtures/docs/`` with the
|
||||
deterministic mock embeddings (same pattern as the earlier story suites).
|
||||
The four tests map the story's acceptance criteria:
|
||||
|
||||
1. multi-format fixture import — hidden doc excluded, ``/api/docs`` counts
|
||||
2. "How did I install gitlab?" — grounded (not deflected), gitlab chip,
|
||||
``query_log`` row with the gitlab doc in ``sources``
|
||||
3. keyword-only question ("kafkabridge") beats the vector ranking — the
|
||||
FTS-OR gate grounds it end to end despite weak cosine
|
||||
4. "sourdough" — deflected bubble + ≥2 "Maybe try" chips
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
from pathlib import Path
|
||||
from threading import Thread
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
from playwright.sync_api import Page, expect
|
||||
from sqlalchemy import select, text
|
||||
|
||||
from app.config import Settings, get_settings
|
||||
from app.db import SessionLocal
|
||||
from app.models import QueryLog
|
||||
from app.rag.importer import ImportSummary, import_sources
|
||||
from app.rag.llm import LLMClient
|
||||
|
||||
REPO = Path(__file__).resolve().parents[2]
|
||||
FIXTURES = REPO / "tests" / "fixtures" / "docs"
|
||||
GITLAB_QUESTION = "How did I install gitlab?"
|
||||
KEYWORD_QUESTION = "How does kafkabridge work?"
|
||||
OFF_TOPIC = "sourdough starter"
|
||||
MOCK_ANSWER_MARKER = "Deterministic mock answer for E2E"
|
||||
|
||||
|
||||
async def _import_fixtures(mock_port: int) -> ImportSummary:
|
||||
kwargs: dict[str, Any] = {"_env_file": None, "llm_base_url": f"http://127.0.0.1:{mock_port}/v1"}
|
||||
settings = Settings(**kwargs) # pyright: ignore[reportCallIssue]
|
||||
return await import_sources([FIXTURES], LLMClient(settings))
|
||||
|
||||
|
||||
def _run_in_thread(coro: Any) -> Any:
|
||||
"""Run a coroutine on a worker thread (Playwright owns the test loop)."""
|
||||
box: dict[str, Any] = {}
|
||||
|
||||
def runner() -> None:
|
||||
try:
|
||||
box["value"] = asyncio.run(coro)
|
||||
except BaseException as e: # noqa: BLE001 — re-raised on the test thread
|
||||
box["error"] = e
|
||||
|
||||
t = Thread(target=runner)
|
||||
t.start()
|
||||
t.join()
|
||||
if "error" in box:
|
||||
raise box["error"]
|
||||
return box["value"]
|
||||
|
||||
|
||||
def _reset_db(mock_port: int, seed: bool) -> ImportSummary | None:
|
||||
"""Truncate the KB (and query log), then optionally re-import fixtures."""
|
||||
with SessionLocal() as db:
|
||||
db.execute(text("TRUNCATE chunks, documents, query_log"))
|
||||
db.commit()
|
||||
if not seed:
|
||||
return None
|
||||
return _run_in_thread(_import_fixtures(mock_port))
|
||||
|
||||
|
||||
def _ask(page: Page, message: str) -> None:
|
||||
page.fill("#message-input", message)
|
||||
page.click("#send-btn")
|
||||
|
||||
|
||||
def test_multi_format_import_hidden_doc_excluded(
|
||||
page: Page, app_url: str, mock_llm: int, db_ready: None
|
||||
) -> None:
|
||||
"""A9 (revised): all seven formats import; hidden (dot) paths never do."""
|
||||
summary = _reset_db(mock_llm, seed=True)
|
||||
assert summary is not None
|
||||
# Eight A9-format fixture files; .hidden/junk.md must never be walked.
|
||||
assert summary.added == 8
|
||||
assert summary.formats == {"md": 4, "yaml": 1, "json": 1, "py": 1, "txt": 1}
|
||||
|
||||
r = httpx.get(f"{app_url}/api/docs", timeout=10)
|
||||
assert r.status_code == 200
|
||||
docs = r.json()["documents"]
|
||||
assert len(docs) == 8
|
||||
assert all(".hidden" not in d["path"] for d in docs)
|
||||
assert {d["path"] for d in docs} >= {
|
||||
"homelab/container_gitlab/gitlab.md",
|
||||
"homelab/container_gitlab/gitlab-compose.yaml",
|
||||
"homelab/networking/static-dns.json",
|
||||
"homelab/scripts/uptime_probe.py",
|
||||
"homelab/ssh/ssh_aliases.txt",
|
||||
}
|
||||
|
||||
# The Sources page reflects the same set.
|
||||
page.goto(f"{app_url}/sources.html")
|
||||
expect(page.locator("#stat-docs")).to_have_text("8")
|
||||
expect(page.locator("#docs-tbody tr", has_text=".hidden")).to_have_count(0)
|
||||
|
||||
|
||||
def test_gitlab_question_is_grounded_with_gitlab_chip(
|
||||
page: Page, app_url: str, mock_llm: int, db_ready: None
|
||||
) -> None:
|
||||
"""The ranking problem that motivated this phase: a tool-name question
|
||||
must land on the tool's own document — not a generic template."""
|
||||
_reset_db(mock_llm, seed=True)
|
||||
page.set_default_timeout(30_000)
|
||||
page.goto(app_url)
|
||||
|
||||
_ask(page, GITLAB_QUESTION)
|
||||
expect(page.locator(".msg.user .bubble")).to_contain_text(GITLAB_QUESTION)
|
||||
|
||||
bubble = page.locator(".msg.brain .bubble").first
|
||||
bubble.wait_for(state="visible", timeout=30_000)
|
||||
expect(bubble).to_contain_text(MOCK_ANSWER_MARKER, timeout=30_000)
|
||||
# Grounded: no deflected bubble at all.
|
||||
expect(page.locator(".msg.brain.is-deflected")).to_have_count(0)
|
||||
|
||||
# The gitlab document is cited (a chip carrying its path).
|
||||
chip = page.locator(".msg.brain .source-chip", has_text="container_gitlab/gitlab.md")
|
||||
expect(chip).to_have_count(1, timeout=30_000)
|
||||
|
||||
# Durable record: not deflected, and the gitlab doc is in sources.
|
||||
with SessionLocal() as db:
|
||||
row = db.scalars(select(QueryLog)).one()
|
||||
assert row.question == GITLAB_QUESTION
|
||||
assert row.deflected is False
|
||||
assert "container_gitlab/gitlab.md" in row.sources
|
||||
|
||||
|
||||
def test_keyword_only_question_beats_vector_ranking(
|
||||
page: Page, app_url: str, mock_llm: int, db_ready: None
|
||||
) -> None:
|
||||
"""The FTS-OR gate end to end: "kafkabridge" appears in exactly one
|
||||
fixture doc (static-dns.json) and the question's cosine overlap is
|
||||
weak — the lexical branch is what grounds the answer."""
|
||||
_reset_db(mock_llm, seed=True)
|
||||
page.set_default_timeout(30_000)
|
||||
page.goto(app_url)
|
||||
|
||||
_ask(page, KEYWORD_QUESTION)
|
||||
bubble = page.locator(".msg.brain .bubble").first
|
||||
bubble.wait_for(state="visible", timeout=30_000)
|
||||
expect(bubble).to_contain_text(MOCK_ANSWER_MARKER, timeout=30_000)
|
||||
expect(page.locator(".msg.brain.is-deflected")).to_have_count(0)
|
||||
|
||||
# The FTS-matched doc is the TOP source chip (it beats the vector rank).
|
||||
first_chip = page.locator(".msg.brain .source-chip").first
|
||||
first_chip.wait_for(state="visible", timeout=30_000)
|
||||
expect(first_chip).to_contain_text("static-dns.json")
|
||||
|
||||
with SessionLocal() as db:
|
||||
row = db.scalars(select(QueryLog)).one()
|
||||
# Weak vector score…
|
||||
assert row.top_score < get_settings().relevance_threshold
|
||||
# …but a lexical hit grounded it (the FTS-OR branch).
|
||||
assert (row.fts_hits or 0) >= 1
|
||||
assert row.deflected is False
|
||||
assert "homelab/networking/static-dns.json" in row.sources
|
||||
|
||||
|
||||
def test_off_topic_still_deflects_with_chips(
|
||||
page: Page, app_url: str, mock_llm: int, db_ready: None
|
||||
) -> None:
|
||||
"""A8 (revised): deflection requires weak cosine AND zero FTS hits.
|
||||
"sourdough" matches nothing in the KB lexically → honest deflection."""
|
||||
_reset_db(mock_llm, seed=True)
|
||||
page.set_default_timeout(30_000)
|
||||
page.goto(app_url)
|
||||
|
||||
_ask(page, OFF_TOPIC)
|
||||
bubble = page.locator(".msg.brain.is-deflected .bubble").first
|
||||
bubble.wait_for(state="visible", timeout=30_000)
|
||||
expect(page.locator(".msg.brain.is-deflected")).to_have_count(1)
|
||||
|
||||
chips = page.locator(".msg.brain.is-deflected .maybe-try .suggestion-chip")
|
||||
expect(chips.first).to_be_visible(timeout=30_000)
|
||||
assert chips.count() >= 2, "deflection must offer 2-3 alternative chips"
|
||||
assert all(c.strip() for c in chips.all_inner_texts())
|
||||
|
||||
with SessionLocal() as db:
|
||||
row = db.scalars(select(QueryLog)).one()
|
||||
assert row.question == OFF_TOPIC
|
||||
assert row.deflected is True
|
||||
assert 0.0 < row.top_score < get_settings().relevance_threshold
|
||||
assert row.fts_hits == 0 # deflection is only reached with zero hits
|
||||
@@ -63,7 +63,7 @@ def _seed_kb(mock_port: int) -> ImportSummary:
|
||||
db.execute(text("TRUNCATE chunks, documents, query_log"))
|
||||
db.commit()
|
||||
summary = _run_in_thread(_import_fixtures(mock_port))
|
||||
assert summary is not None and summary.added == 3
|
||||
assert summary is not None and summary.added == 8 # A9 formats
|
||||
return summary
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user