All verification complete. Final report: **Phase 119 final verification pass — all criteria verified, one stale pin fixed.** - Verified implementation of all 6 tasks: D1 component name-hit rule (`name_hit` flag, titles never matched, retired length tie-break), D2 `BOR_NAME_HIT_BONUS` (0.005 default, 0 = byte-identical kill switch, negative fails startup, selection-layer only, `eval_retrieval` `suggested:` line), D3 suggested-folder lines (after `SUGGEST_INTRO`, before first block), D4 cite-discipline `SUGGEST_INTRO` sentence (PERSONA/LOW/`TOOLS_SECTION` byte-pins intact), D5 `done.sources` = read docs only (frontend no-op on empty confirmed), D6 mock `repeat your folder map` echo + new suite + telemetry. - Battery (replica restored per skill, fingerprint docs=1000/chunks=8866 verified, `eval_retrieval --from-file tests/fixtures/retrieval_battery.txt` re-run): **GATE PASS** — gitea README #4 in suggested top-5, forgejo 5/5 (README #1), gateway README in top-5 (#4), qwen3.8-27b quadlets top-5, Mongolia HIGH/fts=5 unchanged. - New E2E in isolation: `4 passed` ×2 (deterministic). All 27 modified E2E suites in isolation: 26 green; **1 stale pin fixed** — `test_source_chip_quality.py` durable-record order pin pre-dated the D1 re-rank (`aliases` stem sub-component name-hits `ssh_aliases.txt`, deterministically lifting `backups.md` over `kubernetes.md`; probe-verified 0.016277 vs 0.016036, 4/4 stable) — re-pinned with the phase-119 rationale; suite green ×2. - Gates: `uv run pytest --cov=app --cov-report=term-missing` → **2547 passed, app coverage 99%** (>90%); `uv run ruff check .` → All checks passed; `uv run pyright` → 0 errors. - Completion criteria: 1 ✅ (battery, recorded), 2 ✅ (folder lines; block/LOW byte-identical pins green), 3 ✅ (read-only chips, zero-read chips nothing, related row + durable record untouched — unit+E2E agree), 4 ✅ (all green), 5 → commit/phase-move left to the harness per pass rules (nothing committed). - Deviations: battery output + real-model telemetry recorded in `.agents/reports/119_name_signal_read_chips/task06_battery_and_e2e.md` and `TOOL_CALLING_TESTING.md` §11 (task files in `complete/` are immutable to this pass); gateway canonical doc at #4 vs overview's #3 was already documented at task 06 (containment gate met). - Next pending phase: **none** — `todo/` holds only phase 119.
160 lines
6.0 KiB
Python
160 lines
6.0 KiB
Python
"""Evaluate hybrid retrieval against the live knowledge base (phase 09).
|
|
|
|
Embeds each question via aipi, runs the same hybrid search the chat API
|
|
uses (cosine top-N + FTS top-N, RRF-fused — plus the phase-106 recency
|
|
boost, which ``retrieve()`` applies after the fusion), and prints the
|
|
top-5 documents with their cosine / fts / fused scores (labelled
|
|
``effective`` when the recency boost is on — the post-boost score)
|
|
and the document's creation date, the seeded suggestion tier
|
|
(``suggested:`` — the ``select_suggested`` walk, the phase-119 name-hit
|
|
bonus included), plus the honesty-gate verdict:
|
|
|
|
uv run python -m scripts.eval_retrieval "How did I install gitlab?"
|
|
uv run python -m scripts.eval_retrieval --from-file questions.txt
|
|
|
|
Requires ``AIPI_KEY`` in the environment (same convention as
|
|
``scripts/llm_probe.py``) and an imported knowledge base
|
|
(``python -m scripts.import_docs``). Exit code 0 when all questions were
|
|
scored (deflections are a normal result — the verdict column shows them).
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import asyncio
|
|
import os
|
|
import sys
|
|
|
|
from dotenv import load_dotenv
|
|
|
|
load_dotenv()
|
|
|
|
|
|
def build_parser() -> argparse.ArgumentParser:
|
|
p = argparse.ArgumentParser(
|
|
prog="python -m scripts.eval_retrieval",
|
|
description="Rank hybrid retrieval results for one or more questions.",
|
|
)
|
|
p.add_argument(
|
|
"questions",
|
|
nargs="*",
|
|
metavar="QUESTION",
|
|
help="one or more questions to evaluate",
|
|
)
|
|
p.add_argument(
|
|
"--from-file",
|
|
type=str,
|
|
default=None,
|
|
metavar="PATH",
|
|
help="read questions from a file (one per line, blanks/# skipped)",
|
|
)
|
|
p.add_argument(
|
|
"--top",
|
|
type=int,
|
|
default=5,
|
|
help="documents to print per question (default: 5)",
|
|
)
|
|
return p
|
|
|
|
|
|
def read_questions(args: argparse.Namespace) -> list[str]:
|
|
questions = list(args.questions)
|
|
if args.from_file:
|
|
with open(args.from_file, encoding="utf-8") as f:
|
|
questions.extend(
|
|
line.strip() for line in f if line.strip() and not line.lstrip().startswith("#")
|
|
)
|
|
return questions
|
|
|
|
|
|
async def _embed_all(llm, questions: list[str]) -> list[list[float]]:
|
|
return await llm.embed(questions)
|
|
|
|
|
|
def main(argv: list[str] | None = None) -> int:
|
|
args = build_parser().parse_args(argv)
|
|
api_key = os.environ.get("BOR_LLM_API_KEY") or os.environ.get("AIPI_KEY", "")
|
|
if not api_key or api_key == "not-needed":
|
|
print(
|
|
"eval_retrieval: AIPI_KEY is required in the environment "
|
|
"(same convention as scripts/llm_probe.py).",
|
|
file=sys.stderr,
|
|
)
|
|
return 1
|
|
questions = read_questions(args)
|
|
if not questions:
|
|
print("eval_retrieval: no questions given (positional or --from-file).", file=sys.stderr)
|
|
return 1
|
|
|
|
from app.config import get_settings
|
|
from app.db import SessionLocal, db_available
|
|
from app.rag.llm import LLMClient
|
|
from app.rag.retriever import RetrievedChunk, retrieve, select_suggested
|
|
|
|
settings = get_settings()
|
|
if not db_available():
|
|
print("eval_retrieval: Postgres is down — run `podman compose up -d db`.", file=sys.stderr)
|
|
return 1
|
|
|
|
llm = LLMClient(settings)
|
|
vectors = asyncio.run(_embed_all(llm, questions))
|
|
|
|
print(
|
|
f"eval: threshold={settings.relevance_threshold} "
|
|
f"vector_candidates={settings.hybrid_vector_candidates} "
|
|
f"lexical_candidates={settings.hybrid_lexical_candidates} rrf_k={settings.rrf_k} "
|
|
f"recency_boost={settings.recency_boost} "
|
|
f"recency_half_life_days={settings.recency_half_life_days} "
|
|
f"name_hit_bonus={settings.name_hit_bonus}"
|
|
)
|
|
with SessionLocal() as db:
|
|
for question, vec in zip(questions, vectors, strict=True):
|
|
chunks = retrieve(db, question, vec)
|
|
best_cosine = max((c.cosine for c in chunks), default=0.0)
|
|
fts_hits = sum(1 for c in chunks if c.fts_hit)
|
|
verdict = (
|
|
"LOW (deflect)"
|
|
if best_cosine < settings.relevance_threshold and fts_hits == 0
|
|
else "HIGH (answer)"
|
|
)
|
|
print(f"\nquestion: {question!r}")
|
|
print(f" gate: best_cosine={best_cosine:.4f} fts_hits={fts_hits} -> {verdict}")
|
|
# Best chunk per document, in fused rank order.
|
|
best_by_doc: dict[str, RetrievedChunk] = {}
|
|
for c in chunks:
|
|
key = f"{c.document.source}/{c.document.path}"
|
|
if key not in best_by_doc:
|
|
best_by_doc[key] = c
|
|
# Phase 106, D6: the post-recency-boost score is the
|
|
# effective score ``retrieve()`` ranked by — labelled
|
|
# ``effective`` while the boost is on (``fused`` when the
|
|
# kill switch is off), with the document's creation date
|
|
# alongside (the boost's input).
|
|
score_label = "effective" if settings.recency_boost > 0 else "fused"
|
|
for i, c in enumerate(list(best_by_doc.values())[: args.top], start=1):
|
|
print(
|
|
f" {i}. {c.document.source}/{c.document.path} "
|
|
f"cosine={c.cosine:.4f} fts={int(c.fts_hit)} "
|
|
f"{score_label}={c.score:.5f} created={c.document.created_at:%Y-%m-%d} "
|
|
f"({c.document.title})"
|
|
)
|
|
# Phase 119, D2: the seeded suggestion tier — the
|
|
# ``select_suggested`` walk over the same list (the name-hit
|
|
# bonus included, default settings) — the tuning tool must
|
|
# report the tier the prompt actually seeds.
|
|
suggested = select_suggested(chunks)
|
|
print(
|
|
" suggested: "
|
|
+ (
|
|
" ".join(
|
|
f"{i}. {d.source}/{d.path}"
|
|
for i, d in enumerate(suggested, start=1)
|
|
)
|
|
or "(none)"
|
|
)
|
|
)
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|