phase: 112_honesty_gate_weak_hits
**Phase 112 — final verification pass (all 4 tasks already complete in `complete/`):** - Verified gate fix: `app/api/chat.py::plan_turn` — HIGH iff `best_cosine >= relevance_threshold` OR (`fts_hits > 0` AND `best_cosine >= lexical_support_floor`); `lexical_support_floor` (default 0.35, `BOR_LEXICAL_SUPPORT_FLOOR`, bounds-validated) in `app/config.py` + `.env.example`; A8 revision note (2026-09-14) in `.agents/PLAN.md`. - Verified prompt contract: `app/rag/prompts.py` diff is docstring-only (dated owner-decision-iii entry); `tests/unit/test_prompt_lock.py` byte-pins PERSONA/TOOLS_SECTION/DEFLECT body (sha256+length). - Verified README: L11 + L575 deflection copy refreshed; `grep "haven't done anything" README.md` → no hits; disclosed-answer behavior documented. - Tests: `uv run pytest --cov=app --cov-report=term-missing` → **2378 passed, 99% coverage (>90%)**; includes Mongolia-quadrant unit pins (fts>0 + cosine<floor → LOW). - E2E in isolation: `uv run pytest tests/e2e/test_honest_deflection.py -v --no-cov` → **4 passed** (out-of-KB question: `deflected=true`, `sources==[]`, 2–3 suggestions); regression `test_chat_rag.py` + `test_retrieval_quality.py` → **7 passed**. - Lint/types: `uv run ruff check .` → clean; `uv run pyright` → 0 errors. **Completion criteria:** weak-FTS→LOW unit-pinned ✅ · no false citations + 2–3 alternatives E2E ✅ · prompts byte-identical (test-pinned) + README matches ✅ · suite/coverage/e2e/lint all green ✅ · commit + phase move → left to harness (no `git commit` run, per rules; changes in working tree). **Deviations:** none. Next pending phase: `113_source_chip_quality`.
This commit is contained in:
@@ -15,6 +15,8 @@ from sqlalchemy.orm import Session
|
||||
# conftest.py. Must be set before ``app.main`` (below) caches settings.
|
||||
# The production default stays 0.62 (app/config.py, A8 revised).
|
||||
os.environ.setdefault("BOR_RELEVANCE_THRESHOLD", "0.30")
|
||||
# A8 revised 2026-09-14: lexical_support_floor must be <= relevance_threshold.
|
||||
os.environ.setdefault("BOR_LEXICAL_SUPPORT_FLOOR", "0.15")
|
||||
|
||||
# Phase 16: single-admin auth is fail-loud — create_app() refuses to boot
|
||||
# without both vars, and app.main (imported below) builds the app at
|
||||
|
||||
@@ -34,6 +34,17 @@ from e2e.auth_helpers import ADMIN_PASSWORD, login
|
||||
REPO = Path(__file__).resolve().parents[2]
|
||||
FIXTURES = REPO / "tests" / "fixtures" / "docs"
|
||||
OFF_TOPIC = "How do I bake sourdough bread?"
|
||||
# Phase 112 (A8 revised 2026-09-14, TODO L2a — the "Mongolia case"): a
|
||||
# question the LLM knows (Gershwin) but the fixture KB does not cover.
|
||||
# Unlike the plain no-hit deflection above, it carries WEAK FTS hits
|
||||
# (the "compos" stem matches the compose fixture docs — fts_hits >= 1)
|
||||
# while its best mock token-overlap cosine (~0.12) sits BELOW
|
||||
# lexical_support_floor (0.15, the mock-calibrated conftest value). The
|
||||
# pre-phase gate (fts>0 → HIGH) grounded it and injected irrelevant docs
|
||||
# into the prompt; the revised gate (cosine corroboration) must keep it
|
||||
# LOW. The mock keys on DEFLECT_MODE, so the test pins the gate, not
|
||||
# model compliance.
|
||||
OUT_OF_KB = "Who composed Rhapsody in Blue?"
|
||||
# The mock's deflection answer (tests/e2e/mock_llm.py) must match this.
|
||||
DEFLECT_PHRASE = r"haven't done anything like that"
|
||||
MOCK_ANSWER_MARKER = "Deterministic mock answer for E2E"
|
||||
@@ -201,3 +212,67 @@ def test_deflected_done_event_and_query_log(app_url: str, mock_llm: int, db_read
|
||||
assert row.deflected is True
|
||||
assert 0.0 < row.top_score < get_settings().relevance_threshold
|
||||
assert row.chunk_hits >= 1
|
||||
|
||||
|
||||
def test_out_of_kb_question_deflects_without_citations(
|
||||
app_url: str, mock_llm: int, db_ready: None
|
||||
) -> None:
|
||||
"""Phase 112 acceptance (TODO L2): a known-out-of-KB question whose
|
||||
weak lexical hits NO LONGER promote (the fts>0 / cosine<floor
|
||||
quadrant, pinned end-to-end) deflects with ZERO source citations —
|
||||
done.sources is empty (the UI chips nothing under a deflected
|
||||
answer) and 2-3 concrete alternative questions are offered.
|
||||
|
||||
Raw SSE (like the done-event test above): the done frame is the
|
||||
contract surface; the mock's DEFLECT_MODE phrasing proves the
|
||||
server sent the LOW prompt (the gate, not the model, decides).
|
||||
"""
|
||||
_reset_db(mock_llm, seed=True)
|
||||
|
||||
client = httpx.Client(timeout=60.0)
|
||||
r = client.post(f"{app_url}/api/login", json={"password": ADMIN_PASSWORD})
|
||||
assert r.status_code == 204
|
||||
|
||||
frames: list[dict[str, Any]] = []
|
||||
with client.stream(
|
||||
"POST", f"{app_url}/api/chat", json={"message": OUT_OF_KB}, timeout=60.0
|
||||
) as r:
|
||||
assert r.status_code == 200
|
||||
assert r.headers["content-type"].startswith("text/event-stream")
|
||||
buf = ""
|
||||
for part in r.iter_text():
|
||||
buf += part
|
||||
while "\n\n" in buf:
|
||||
frame, buf = buf.split("\n\n", 1)
|
||||
if frame.strip().startswith("data:"):
|
||||
frames.append(json.loads(frame.strip().removeprefix("data:").strip()))
|
||||
assert buf.strip() == "" # stream ends cleanly on a frame boundary
|
||||
|
||||
deltas = [f for f in frames if f.get("type") == "delta"]
|
||||
answer = "".join(d["text"] for d in deltas)
|
||||
# The DEFLECT_MODE phrasing streamed ⇒ the LOW prompt reached the
|
||||
# model (the mock answers it only for the deflection system prompt).
|
||||
assert re.search(DEFLECT_PHRASE, answer, re.IGNORECASE)
|
||||
|
||||
done = frames[-1]
|
||||
assert done["type"] == "done"
|
||||
assert done["deflected"] is True
|
||||
# No false citations (TODO L2): the weak hits never ride the wire as
|
||||
# sources — a deflected answer cites nothing.
|
||||
assert done["sources"] == []
|
||||
# 2-3 concrete alternative questions, all non-empty.
|
||||
assert 2 <= len(done["suggestions"]) <= 3
|
||||
assert all(s.strip() for s in done["suggestions"])
|
||||
|
||||
# Durable record: the quadrant pinned end-to-end — the lexical leg
|
||||
# FIRED (fts_hits > 0, the pre-phase gate's promotion trigger) while
|
||||
# the vector signal never cleared lexical_support_floor, so the
|
||||
# revised gate deflected. The retrieval itself stays recorded
|
||||
# (query_log = observability, not citations).
|
||||
with SessionLocal() as db:
|
||||
row = db.scalars(select(QueryLog)).one()
|
||||
assert row.question == OUT_OF_KB
|
||||
assert row.deflected is True
|
||||
assert (row.fts_hits or 0) >= 1
|
||||
assert row.top_score < get_settings().lexical_support_floor
|
||||
assert row.sources
|
||||
|
||||
@@ -13,7 +13,8 @@ The four tests map the story's acceptance criteria:
|
||||
2. "How did I install gitlab?" — grounded (not deflected), gitlab chip,
|
||||
``query_log`` row with the gitlab doc in ``sources``
|
||||
3. keyword-only question ("kafkabridge") beats the vector ranking — the
|
||||
FTS-OR gate grounds it end to end despite weak cosine
|
||||
corroborated-lexical gate (A8 revised 2026-09-14) grounds it end to
|
||||
end: weak cosine, but an FTS hit AND cosine >= lexical_support_floor
|
||||
4. "sourdough" — deflected bubble + ≥2 "Maybe try" chips
|
||||
"""
|
||||
from __future__ import annotations
|
||||
@@ -37,7 +38,15 @@ from e2e.auth_helpers import login
|
||||
REPO = Path(__file__).resolve().parents[2]
|
||||
FIXTURES = REPO / "tests" / "fixtures" / "docs"
|
||||
GITLAB_QUESTION = "How did I install gitlab?"
|
||||
KEYWORD_QUESTION = "How does kafkabridge work?"
|
||||
# Phase 112 (A8 revised): the pre-phase question ("How does kafkabridge
|
||||
# work?" — mock cosine 0.134) now sits BELOW lexical_support_floor
|
||||
# (0.15, mock-calibrated) with fts>0 — the new gate's deflection
|
||||
# quadrant, so it can no longer demonstrate the grounded lexical path.
|
||||
# "handle DNS" adds static-dns.json's own tokens: cosine ≈0.24 — still
|
||||
# weak (below the 0.30 threshold) yet corroborated (>= floor) with the
|
||||
# same single-doc FTS hit, and the FTS-matched doc still tops the fused
|
||||
# ranking (the test's actual assertion).
|
||||
KEYWORD_QUESTION = "How does kafkabridge handle DNS?"
|
||||
OFF_TOPIC = "sourdough starter"
|
||||
MOCK_ANSWER_MARKER = "Deterministic mock answer for E2E"
|
||||
|
||||
@@ -162,9 +171,11 @@ def test_gitlab_question_is_grounded_with_gitlab_chip(
|
||||
def test_keyword_only_question_beats_vector_ranking(
|
||||
page: Page, app_url: str, mock_llm: int, db_ready: None
|
||||
) -> None:
|
||||
"""The FTS-OR gate end to end: "kafkabridge" appears in exactly one
|
||||
fixture doc (static-dns.json) and the question's cosine overlap is
|
||||
weak — the lexical branch is what grounds the answer."""
|
||||
"""The corroborated-lexical gate end to end (A8 revised 2026-09-14):
|
||||
"kafkabridge" appears in exactly one fixture doc (static-dns.json)
|
||||
and the question's cosine overlap is weak (below the threshold) —
|
||||
the lexical hit plus cosine >= lexical_support_floor is what grounds
|
||||
the answer (a lexical-only hit below the floor would deflect)."""
|
||||
_reset_db(mock_llm, seed=True)
|
||||
page.set_default_timeout(30_000)
|
||||
login(page, app_url, next="/") # phase 79: chat is require_user-gated
|
||||
@@ -182,9 +193,11 @@ def test_keyword_only_question_beats_vector_ranking(
|
||||
|
||||
with SessionLocal() as db:
|
||||
row = db.scalars(select(QueryLog)).one()
|
||||
# Weak vector score…
|
||||
# Weak vector score — below the answer threshold…
|
||||
assert row.top_score < get_settings().relevance_threshold
|
||||
# …but a lexical hit grounded it (the FTS-OR branch).
|
||||
# …but cleared the lexical support floor, and a lexical hit fired —
|
||||
# the corroborated-lexical path (A8 revised 2026-09-14) grounded it.
|
||||
assert row.top_score >= get_settings().lexical_support_floor
|
||||
assert (row.fts_hits or 0) >= 1
|
||||
assert row.deflected is False
|
||||
assert "homelab/networking/static-dns.json" in row.sources
|
||||
|
||||
@@ -413,7 +413,11 @@ def test_off_topic_question_deflects_honestly(client, db, seeded_kb: FakeRagLLM)
|
||||
assert any(
|
||||
"Deploying a New Service" in s for s in done["suggestions"]
|
||||
), "the best weak-hit title must be offered as a chip"
|
||||
assert done["sources"], "weak hits are still reported as the closest sources"
|
||||
# Phase 112 (A8 revised, TODO L2): a deflected turn cites nothing —
|
||||
# done.sources is the citation surface (the UI chips every entry as
|
||||
# "the answer used this"), and the weak hits are scored docs, not
|
||||
# citations. (Pre-phase: they rode the wire as sources.)
|
||||
assert done["sources"] == []
|
||||
|
||||
# The LLM saw the LOW prompt: DEFLECT_MODE + titles, never doc content.
|
||||
(system, user) = seeded_kb.seen_messages[0][0], seeded_kb.seen_messages[0][1]
|
||||
@@ -433,19 +437,34 @@ def test_off_topic_question_deflects_honestly(client, db, seeded_kb: FakeRagLLM)
|
||||
assert 0.0 < row.top_score < get_settings().relevance_threshold
|
||||
assert row.fts_hits == 0
|
||||
assert row.chunk_hits >= 1
|
||||
# The retrieval stays durably recorded for threshold tuning
|
||||
# (observability unchanged — query_log records retrieval, not
|
||||
# citations; the done frame's [] above is the citation surface).
|
||||
assert row.sources
|
||||
|
||||
|
||||
def test_keyword_question_grounded_by_lexical_hit_despite_weak_cosine(
|
||||
client, db, seeded_kb: FakeRagLLM
|
||||
client, db, seeded_kb: FakeRagLLM,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""Phase 09: a name-your-tool question the vector model barely ranks
|
||||
("kafkabridge" only appears in static-dns.json) must still be grounded
|
||||
via the FTS branch — LOW only fires at weak cosine AND zero hits."""
|
||||
via the FTS branch — HIGH when cosine >= lexical_support_floor AND
|
||||
fts_hits > 0 (A8 revised 2026-09-14).
|
||||
|
||||
The conftest floor (0.15) is above the mock's cosine (~0.134), so we
|
||||
lower the floor here so the corroborated-lexical path fires."""
|
||||
from app.config import get_settings # noqa: E402
|
||||
|
||||
monkeypatch.setenv("BOR_LEXICAL_SUPPORT_FLOOR", "0.10")
|
||||
# get_settings is lru_cached — clear the cache so the new env var takes effect.
|
||||
get_settings.cache_clear()
|
||||
fastapi_app.dependency_overrides[chat_api.get_llm] = lambda: seeded_kb
|
||||
try:
|
||||
_, _, frames = _stream_chat(client, "How does kafkabridge work?")
|
||||
finally:
|
||||
fastapi_app.dependency_overrides.clear()
|
||||
get_settings.cache_clear()
|
||||
|
||||
done = frames[-1]
|
||||
assert done["type"] == "done"
|
||||
|
||||
@@ -36,10 +36,13 @@ ANSWER = "I haven't done anything like that — try one of these instead!"
|
||||
KB_OVERVIEW = "- Homelab\n - Kubernetes (k3s)\n- Deployments\n - Borg backups"
|
||||
|
||||
|
||||
def _settings(threshold: float = 0.30) -> Settings:
|
||||
def _settings(threshold: float = 0.30, floor: float | None = None) -> Settings:
|
||||
if floor is None:
|
||||
floor = threshold * 0.5 # half the threshold — keeps existing tests green
|
||||
return Settings(
|
||||
_env_file=None, # pyright: ignore[reportCallIssue]
|
||||
relevance_threshold=threshold,
|
||||
lexical_support_floor=floor,
|
||||
)
|
||||
|
||||
|
||||
@@ -118,12 +121,16 @@ def test_gate_is_env_tunable_via_settings() -> None:
|
||||
|
||||
|
||||
def test_gate_weak_cosine_with_fts_hit_still_answers() -> None:
|
||||
"""cosine < threshold but a lexical hit ⇒ HIGH — the FTS-OR branch.
|
||||
This is the name-your-tool case: "kafkabridge" grounds despite weak
|
||||
vector overlap."""
|
||||
"""cosine < threshold but a lexical hit corroborated by cosine >= floor
|
||||
⇒ HIGH — the FTS-OR branch. This is the name-your-tool case:
|
||||
"kafkabridge" grounds despite weak vector overlap.
|
||||
|
||||
A8 revised 2026-09-14: FTS alone no longer promotes; cosine must also
|
||||
clear lexical_support_floor (here 0.15 = half of threshold 0.30)."""
|
||||
doc = _doc("Static DNS", "DNS_DOC_CONTENT")
|
||||
plan = chat_api.plan_turn(
|
||||
[_chunk(doc, 0.02, cosine=0.10, fts_hit=True)], _settings(threshold=0.30)
|
||||
[_chunk(doc, 0.02, cosine=0.10, fts_hit=True)],
|
||||
_settings(threshold=0.30, floor=0.05), # floor=0.05 so 0.10 >= floor
|
||||
)
|
||||
assert plan.deflected is False
|
||||
assert plan.top_score == pytest.approx(0.10) # gate input is the cosine
|
||||
@@ -155,7 +162,7 @@ def test_gate_fts_hits_counts_all_lexical_candidates() -> None:
|
||||
_chunk(a, 0.02, cosine=0.04, fts_hit=True), # same doc, second chunk
|
||||
_chunk(b, 0.01, cosine=0.03),
|
||||
]
|
||||
plan = chat_api.plan_turn(chunks, _settings(threshold=0.30))
|
||||
plan = chat_api.plan_turn(chunks, _settings(threshold=0.30, floor=0.03))
|
||||
assert plan.deflected is False
|
||||
assert plan.fts_hits == 2 # per chunk, not per doc
|
||||
|
||||
@@ -176,6 +183,184 @@ def test_gate_lexical_only_chunk_does_not_inflate_cosine() -> None:
|
||||
assert plan.docs[0].title == "Beta"
|
||||
|
||||
|
||||
# ---------- lexical support floor (A8 revised 2026-09-14) ----------
|
||||
|
||||
|
||||
def test_gate_fts_hit_below_floor_deflects() -> None:
|
||||
"""The Mongolia case: FTS hit with cosine below lexical_support_floor
|
||||
→ LOW (deflected). The lexical-only hit no longer promotes to HIGH.
|
||||
This is the regression pin for phase 112."""
|
||||
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
|
||||
plan = chat_api.plan_turn(
|
||||
[_chunk(doc, 0.05, cosine=0.10, fts_hit=True)],
|
||||
_settings(threshold=0.62),
|
||||
)
|
||||
assert plan.deflected is True
|
||||
assert plan.top_score == pytest.approx(0.10)
|
||||
assert plan.fts_hits == 1
|
||||
assert "DEFLECT_MODE" in plan.system_prompt
|
||||
assert "QUEST_DOC_CONTENT" not in plan.system_prompt
|
||||
assert "Capital Quest" in plan.system_prompt # title only
|
||||
assert plan.suggestions # derived from weak-hit titles
|
||||
|
||||
|
||||
def test_gate_fts_hit_at_floor_answers() -> None:
|
||||
"""FTS hit with cosine exactly at lexical_support_floor → HIGH.
|
||||
The floor is inclusive (>=), not strict (<)."""
|
||||
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
|
||||
plan = chat_api.plan_turn(
|
||||
[_chunk(doc, 0.40, cosine=0.35, fts_hit=True)],
|
||||
_settings(threshold=0.62),
|
||||
)
|
||||
assert plan.deflected is False
|
||||
assert plan.top_score == pytest.approx(0.35)
|
||||
assert plan.fts_hits == 1
|
||||
assert "QUEST_DOC_CONTENT" in plan.system_prompt
|
||||
assert plan.suggestions == []
|
||||
|
||||
|
||||
def test_gate_fts_hit_above_floor_below_threshold_answers() -> None:
|
||||
"""FTS hit with cosine between floor and threshold → HIGH.
|
||||
The corroborated-lexical path fires."""
|
||||
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
|
||||
plan = chat_api.plan_turn(
|
||||
[_chunk(doc, 0.50, cosine=0.50, fts_hit=True)],
|
||||
_settings(threshold=0.62),
|
||||
)
|
||||
assert plan.deflected is False
|
||||
assert plan.top_score == pytest.approx(0.50)
|
||||
assert plan.fts_hits == 1
|
||||
assert "QUEST_DOC_CONTENT" in plan.system_prompt
|
||||
assert plan.suggestions == []
|
||||
|
||||
|
||||
def test_gate_fts_hit_above_code_default_floor_answers() -> None:
|
||||
"""The quadrant table's "0.50 with default settings" row: the CODE
|
||||
defaults (``test_lexical_support_floor_validation_default`` pins them:
|
||||
threshold 0.62 / floor 0.35) — fts>0 + cosine 0.50 >= 0.35 → HIGH.
|
||||
Named literally (not via the helper's half-threshold floor) so the
|
||||
production-default path is pinned on its own."""
|
||||
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
|
||||
plan = chat_api.plan_turn(
|
||||
[_chunk(doc, 0.50, cosine=0.50, fts_hit=True)],
|
||||
_settings(threshold=0.62, floor=0.35),
|
||||
)
|
||||
assert plan.deflected is False
|
||||
assert plan.top_score == pytest.approx(0.50)
|
||||
assert plan.fts_hits == 1
|
||||
assert "QUEST_DOC_CONTENT" in plan.system_prompt
|
||||
assert plan.suggestions == []
|
||||
|
||||
|
||||
def test_gate_high_cosine_overrides_fts_deflection() -> None:
|
||||
"""Strong cosine (>= threshold) → HIGH regardless of FTS status.
|
||||
The cosine-primary path is unchanged."""
|
||||
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
|
||||
plan = chat_api.plan_turn(
|
||||
[_chunk(doc, 0.90, cosine=0.80, fts_hit=True)],
|
||||
_settings(threshold=0.62),
|
||||
)
|
||||
assert plan.deflected is False
|
||||
assert plan.top_score == pytest.approx(0.80)
|
||||
assert plan.fts_hits == 1
|
||||
assert "QUEST_DOC_CONTENT" in plan.system_prompt
|
||||
assert plan.suggestions == []
|
||||
|
||||
|
||||
def test_gate_fts_no_cosine_deflects() -> None:
|
||||
"""FTS hit with cosine = 0.0 → LOW (the extreme Mongolia case)."""
|
||||
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
|
||||
plan = chat_api.plan_turn(
|
||||
[_chunk(doc, 0.90, cosine=0.0, fts_hit=True)],
|
||||
_settings(threshold=0.62),
|
||||
)
|
||||
assert plan.deflected is True
|
||||
assert plan.top_score == 0.0
|
||||
assert plan.fts_hits == 1
|
||||
assert "DEFLECT_MODE" in plan.system_prompt
|
||||
|
||||
|
||||
def test_gate_multiple_fts_below_floor_deflects() -> None:
|
||||
"""Multiple FTS hits, all below lexical_support_floor → LOW.
|
||||
The gate requires the BEST cosine to clear the floor, not just any hit."""
|
||||
a = _doc("Alpha Quest", "ALPHA_CONTENT")
|
||||
b = _doc("Beta Quest", "BETA_CONTENT")
|
||||
chunks = [
|
||||
_chunk(a, 0.30, cosine=0.20, fts_hit=True),
|
||||
_chunk(b, 0.25, cosine=0.15, fts_hit=True),
|
||||
]
|
||||
plan = chat_api.plan_turn(chunks, _settings(threshold=0.62))
|
||||
assert plan.deflected is True
|
||||
assert plan.fts_hits == 2
|
||||
assert "DEFLECT_MODE" in plan.system_prompt
|
||||
|
||||
|
||||
def test_gate_one_fts_above_floor_answers() -> None:
|
||||
"""Multiple chunks, one FTS hit above floor → HIGH.
|
||||
The best cosine (from the corroborated hit) clears the floor."""
|
||||
a = _doc("Alpha Quest", "ALPHA_CONTENT")
|
||||
b = _doc("Beta Quest", "BETA_CONTENT")
|
||||
chunks = [
|
||||
_chunk(a, 0.30, cosine=0.20, fts_hit=True), # below floor
|
||||
_chunk(b, 0.25, cosine=0.40, fts_hit=True), # above floor
|
||||
]
|
||||
plan = chat_api.plan_turn(chunks, _settings(threshold=0.62))
|
||||
assert plan.deflected is False
|
||||
assert plan.fts_hits == 2
|
||||
assert "ALPHA_CONTENT" in plan.system_prompt
|
||||
assert "BETA_CONTENT" in plan.system_prompt
|
||||
|
||||
|
||||
# ---------- config validation (lexical_support_floor) ----------
|
||||
|
||||
|
||||
def test_lexical_support_floor_validation_floor_above_threshold_fails() -> None:
|
||||
"""lexical_support_floor > relevance_threshold is rejected at startup."""
|
||||
with pytest.raises(ValueError, match="lexical_support_floor"):
|
||||
Settings(
|
||||
_env_file=None, # pyright: ignore[reportCallIssue]
|
||||
relevance_threshold=0.62,
|
||||
lexical_support_floor=0.70,
|
||||
)
|
||||
|
||||
|
||||
def test_lexical_support_floor_validation_negative_fails() -> None:
|
||||
"""Negative lexical_support_floor is rejected."""
|
||||
with pytest.raises(ValueError, match="lexical_support_floor"):
|
||||
Settings(
|
||||
_env_file=None, # pyright: ignore[reportCallIssue]
|
||||
lexical_support_floor=-0.1,
|
||||
)
|
||||
|
||||
|
||||
def test_lexical_support_floor_validation_at_threshold_succeeds() -> None:
|
||||
"""lexical_support_floor == relevance_threshold is legal."""
|
||||
s = Settings(
|
||||
_env_file=None, # pyright: ignore[reportCallIssue]
|
||||
relevance_threshold=0.62,
|
||||
lexical_support_floor=0.62,
|
||||
)
|
||||
assert s.lexical_support_floor == 0.62
|
||||
|
||||
|
||||
def test_lexical_support_floor_validation_default() -> None:
|
||||
"""Default lexical_support_floor is 0.35."""
|
||||
import os
|
||||
# Conftest sets BOR_RELEVANCE_THRESHOLD=0.30 and BOR_LEXICAL_SUPPORT_FLOOR=0.15.
|
||||
# We need the CODE defaults, so clear both and let the class defaults apply.
|
||||
saved_relevance = os.environ.pop("BOR_RELEVANCE_THRESHOLD", None)
|
||||
saved_floor = os.environ.pop("BOR_LEXICAL_SUPPORT_FLOOR", None)
|
||||
try:
|
||||
s = Settings(_env_file=None) # pyright: ignore[reportCallIssue]
|
||||
assert s.lexical_support_floor == 0.35
|
||||
assert s.relevance_threshold == 0.62
|
||||
finally:
|
||||
if saved_relevance is not None:
|
||||
os.environ["BOR_RELEVANCE_THRESHOLD"] = saved_relevance
|
||||
if saved_floor is not None:
|
||||
os.environ["BOR_LEXICAL_SUPPORT_FLOOR"] = saved_floor
|
||||
|
||||
|
||||
def test_gate_zero_chunks_deflects_with_fallback_chips() -> None:
|
||||
plan = chat_api.plan_turn([], _settings())
|
||||
assert plan.deflected is True
|
||||
@@ -581,6 +766,11 @@ def test_endpoint_just_below_threshold_deflects(
|
||||
assert 2 <= len(done["suggestions"]) <= MAX_SUGGESTIONS # title chip + fallback
|
||||
assert all(s.strip() for s in done["suggestions"])
|
||||
assert any("Deploying a New Service" in s for s in done["suggestions"])
|
||||
# Phase 112 (A8 revised, TODO L2): a deflected turn cites nothing —
|
||||
# done.sources is the citation surface (the UI chips every entry as
|
||||
# "the answer used this"), and the weak hits are scored docs, not
|
||||
# citations.
|
||||
assert done["sources"] == []
|
||||
|
||||
# The LLM saw the LOW prompt: DEFLECT_MODE + titles, never doc content.
|
||||
(system, user) = llm.seen[0][0], llm.seen[0][1]
|
||||
@@ -588,11 +778,14 @@ def test_endpoint_just_below_threshold_deflects(
|
||||
assert "DEFLECT_MODE" in system["content"]
|
||||
assert "DOC_CONTENT_NEVER_SENT" not in system["content"]
|
||||
|
||||
# Durable record: deflected + the weak score.
|
||||
# Durable record: deflected + the weak score. The retrieval itself
|
||||
# stays recorded (observability unchanged — query_log records
|
||||
# retrieval, not citations; the phase-113 A3 precedent).
|
||||
(row,) = session.added
|
||||
assert isinstance(row, QueryLog)
|
||||
assert row.deflected is True
|
||||
assert row.top_score == pytest.approx(0.2999)
|
||||
assert row.sources # the weak-hit doc's path, for threshold tuning
|
||||
assert session.commits == 1
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,122 @@
|
||||
"""Prompt-lock pin (phase 112, task 03 — owner decision iii, 2026-09-14).
|
||||
|
||||
The persona + HONESTY GATE text is **locked verbatim** (PLAN §6): it
|
||||
changes through the plan, never in code. This module byte-pins the
|
||||
locked prompt constants against their pre-phase-112 anchor values —
|
||||
sha256 + exact prefix/suffix + total length, so *any* byte change
|
||||
(option (i)'s copy tightening, option (ii)'s plan amendment, or an
|
||||
accidental edit) fails loudly until the anchors are re-cut as part of
|
||||
the same plan revision. ``tests.unit.test_prompts`` pins the
|
||||
assembled-prompt structure and the behavioral contracts on top of
|
||||
these constants; this file pins the constants themselves.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
|
||||
from app.rag.prompts import (
|
||||
PERSONA,
|
||||
TOOLS_SECTION,
|
||||
_base,
|
||||
build_deflect_prompt,
|
||||
)
|
||||
|
||||
|
||||
def _sha256(text: str) -> str:
|
||||
return hashlib.sha256(text.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
# ---------- PERSONA (the locked base of BOTH the HIGH and LOW prompts) ----------
|
||||
|
||||
#: Pre-phase-112 anchors for ``PERSONA`` (the ``{relevance}`` placeholder
|
||||
#: and the line wrapping included).
|
||||
PERSONA_SHA256 = "e31792a73e64c53853097e0f7b6df8b96c5f2d05c286944c3edba16dde7777fe"
|
||||
PERSONA_LEN = 706
|
||||
PERSONA_PREFIX = (
|
||||
'You are "Brain of Reese" — the digital brain of Reese, a self-hoster and\n'
|
||||
"homelab tinkerer. Personality: chippy, upbeat, warm, and genuinely\n"
|
||||
"optimistic about the user's ability to do things.\n"
|
||||
"\n"
|
||||
"Rules:\n"
|
||||
)
|
||||
PERSONA_SUFFIX = (
|
||||
"4. Never invent facts, hosts, or steps that are not in the context.\n"
|
||||
"5. Keep answers tight: short paragraphs, bullets where helpful.\n"
|
||||
"\n"
|
||||
"<relevance>{relevance}</relevance>"
|
||||
)
|
||||
|
||||
|
||||
def test_persona_byte_locked() -> None:
|
||||
"""Any byte change to the locked persona (opening, any rule line, the
|
||||
``<relevance>`` placeholder) fails on the sha256; the prefix/suffix
|
||||
anchors name the damaged region for the diff."""
|
||||
assert len(PERSONA) == PERSONA_LEN
|
||||
assert _sha256(PERSONA) == PERSONA_SHA256
|
||||
assert PERSONA.startswith(PERSONA_PREFIX)
|
||||
assert PERSONA.endswith(PERSONA_SUFFIX)
|
||||
|
||||
|
||||
def test_high_and_low_bases_byte_locked() -> None:
|
||||
"""``_base`` only substitutes ``{relevance}`` — the per-mode base
|
||||
lengths pin the substitution against a moved or re-spelled
|
||||
placeholder in the locked text."""
|
||||
assert len(_base("HIGH")) == 699 # PERSONA_LEN - 11 + 4
|
||||
assert len(_base("LOW")) == 698 # PERSONA_LEN - 11 + 3
|
||||
|
||||
|
||||
# ---------- TOOLS_SECTION (the HIGH prompt's locked ``<tools>`` copy) ----------
|
||||
|
||||
#: Pre-phase-112 anchors for ``TOOLS_SECTION``.
|
||||
TOOLS_SECTION_SHA256 = "b834cbe368055e65da82ae3e37a91e6c658c713954703fc79b849a6ebdf4aa53"
|
||||
TOOLS_SECTION_LEN = 2273
|
||||
TOOLS_SECTION_PREFIX = (
|
||||
"<tools>\n"
|
||||
"You may extend your context with three tools. `ls` lists the "
|
||||
"knowledge base as a tree, one level at a time: "
|
||||
)
|
||||
TOOLS_SECTION_SUFFIX = (
|
||||
"Never repeat a call that was refused or already succeeded — the refusal "
|
||||
"already told you the correct form. Answer as soon as you have what "
|
||||
"you need.\n</tools>"
|
||||
)
|
||||
|
||||
|
||||
def test_tools_section_byte_locked() -> None:
|
||||
"""The ``<tools>`` teaching is LOCKED verbatim too (the E2E mock keys
|
||||
on the ``<tools>`` marker's presence; the wording is the owner's):
|
||||
sha256 + exact prefix/suffix + total length."""
|
||||
assert len(TOOLS_SECTION) == TOOLS_SECTION_LEN
|
||||
assert _sha256(TOOLS_SECTION) == TOOLS_SECTION_SHA256
|
||||
assert TOOLS_SECTION.startswith(TOOLS_SECTION_PREFIX)
|
||||
assert TOOLS_SECTION.endswith(TOOLS_SECTION_SUFFIX)
|
||||
|
||||
|
||||
# ---------- the LOW prompt's locked DEFLECT_MODE body ----------
|
||||
|
||||
#: The ``DEFLECT_MODE`` body exactly as it was pre-phase-112 (the
|
||||
#: phase-71 plain-text line included) — inline in
|
||||
#: :func:`app.rag.prompts.build_deflect_prompt`, so it is pinned through
|
||||
#: the built prompt rather than a module constant.
|
||||
LOW_BODY_SHA256 = "c9868cfccdd0ff79d7c1de5df5f6dea0182d3726ca912c9ef609ab54b563a0a4"
|
||||
LOW_BODY_LEN = 259
|
||||
LOW_BODY = (
|
||||
"DEFLECT_MODE: retrieval was weak — the titles below are the closest "
|
||||
"your notes come to the question. They are titles only; do not pretend "
|
||||
"they answer it. Use them to propose 2-3 alternative questions.\n"
|
||||
"Reply in plain text only — you have no tools in this mode."
|
||||
)
|
||||
|
||||
|
||||
def test_deflect_body_byte_locked() -> None:
|
||||
"""The LOW build = locked base + exactly the locked DEFLECT_MODE body
|
||||
+ the weak-hit title list — byte for byte (the mock keys on the
|
||||
``DEFLECT_MODE`` marker's presence; the body wording is locked)."""
|
||||
assert len(LOW_BODY) == LOW_BODY_LEN
|
||||
assert _sha256(LOW_BODY) == LOW_BODY_SHA256
|
||||
assert build_deflect_prompt(["T1", "T2"]) == _base("LOW") + "\n" + LOW_BODY + "\n- T1\n- T2"
|
||||
prompt = build_deflect_prompt(["T1", "T2"])
|
||||
assert prompt.count(LOW_BODY) == 1
|
||||
assert prompt.index("DEFLECT_MODE") < prompt.index(
|
||||
"Reply in plain text only"
|
||||
) # the marker precedes the plain-text line
|
||||
Reference in New Issue
Block a user