phase: 112_honesty_gate_weak_hits
**Phase 112 — final verification pass (all 4 tasks already complete in `complete/`):** - Verified gate fix: `app/api/chat.py::plan_turn` — HIGH iff `best_cosine >= relevance_threshold` OR (`fts_hits > 0` AND `best_cosine >= lexical_support_floor`); `lexical_support_floor` (default 0.35, `BOR_LEXICAL_SUPPORT_FLOOR`, bounds-validated) in `app/config.py` + `.env.example`; A8 revision note (2026-09-14) in `.agents/PLAN.md`. - Verified prompt contract: `app/rag/prompts.py` diff is docstring-only (dated owner-decision-iii entry); `tests/unit/test_prompt_lock.py` byte-pins PERSONA/TOOLS_SECTION/DEFLECT body (sha256+length). - Verified README: L11 + L575 deflection copy refreshed; `grep "haven't done anything" README.md` → no hits; disclosed-answer behavior documented. - Tests: `uv run pytest --cov=app --cov-report=term-missing` → **2378 passed, 99% coverage (>90%)**; includes Mongolia-quadrant unit pins (fts>0 + cosine<floor → LOW). - E2E in isolation: `uv run pytest tests/e2e/test_honest_deflection.py -v --no-cov` → **4 passed** (out-of-KB question: `deflected=true`, `sources==[]`, 2–3 suggestions); regression `test_chat_rag.py` + `test_retrieval_quality.py` → **7 passed**. - Lint/types: `uv run ruff check .` → clean; `uv run pyright` → 0 errors. **Completion criteria:** weak-FTS→LOW unit-pinned ✅ · no false citations + 2–3 alternatives E2E ✅ · prompts byte-identical (test-pinned) + README matches ✅ · suite/coverage/e2e/lint all green ✅ · commit + phase move → left to harness (no `git commit` run, per rules; changes in working tree). **Deviations:** none. Next pending phase: `113_source_chip_quality`.
This commit is contained in:
@@ -36,10 +36,13 @@ ANSWER = "I haven't done anything like that — try one of these instead!"
|
||||
KB_OVERVIEW = "- Homelab\n - Kubernetes (k3s)\n- Deployments\n - Borg backups"
|
||||
|
||||
|
||||
def _settings(threshold: float = 0.30) -> Settings:
|
||||
def _settings(threshold: float = 0.30, floor: float | None = None) -> Settings:
|
||||
if floor is None:
|
||||
floor = threshold * 0.5 # half the threshold — keeps existing tests green
|
||||
return Settings(
|
||||
_env_file=None, # pyright: ignore[reportCallIssue]
|
||||
relevance_threshold=threshold,
|
||||
lexical_support_floor=floor,
|
||||
)
|
||||
|
||||
|
||||
@@ -118,12 +121,16 @@ def test_gate_is_env_tunable_via_settings() -> None:
|
||||
|
||||
|
||||
def test_gate_weak_cosine_with_fts_hit_still_answers() -> None:
|
||||
"""cosine < threshold but a lexical hit ⇒ HIGH — the FTS-OR branch.
|
||||
This is the name-your-tool case: "kafkabridge" grounds despite weak
|
||||
vector overlap."""
|
||||
"""cosine < threshold but a lexical hit corroborated by cosine >= floor
|
||||
⇒ HIGH — the FTS-OR branch. This is the name-your-tool case:
|
||||
"kafkabridge" grounds despite weak vector overlap.
|
||||
|
||||
A8 revised 2026-09-14: FTS alone no longer promotes; cosine must also
|
||||
clear lexical_support_floor (here 0.15 = half of threshold 0.30)."""
|
||||
doc = _doc("Static DNS", "DNS_DOC_CONTENT")
|
||||
plan = chat_api.plan_turn(
|
||||
[_chunk(doc, 0.02, cosine=0.10, fts_hit=True)], _settings(threshold=0.30)
|
||||
[_chunk(doc, 0.02, cosine=0.10, fts_hit=True)],
|
||||
_settings(threshold=0.30, floor=0.05), # floor=0.05 so 0.10 >= floor
|
||||
)
|
||||
assert plan.deflected is False
|
||||
assert plan.top_score == pytest.approx(0.10) # gate input is the cosine
|
||||
@@ -155,7 +162,7 @@ def test_gate_fts_hits_counts_all_lexical_candidates() -> None:
|
||||
_chunk(a, 0.02, cosine=0.04, fts_hit=True), # same doc, second chunk
|
||||
_chunk(b, 0.01, cosine=0.03),
|
||||
]
|
||||
plan = chat_api.plan_turn(chunks, _settings(threshold=0.30))
|
||||
plan = chat_api.plan_turn(chunks, _settings(threshold=0.30, floor=0.03))
|
||||
assert plan.deflected is False
|
||||
assert plan.fts_hits == 2 # per chunk, not per doc
|
||||
|
||||
@@ -176,6 +183,184 @@ def test_gate_lexical_only_chunk_does_not_inflate_cosine() -> None:
|
||||
assert plan.docs[0].title == "Beta"
|
||||
|
||||
|
||||
# ---------- lexical support floor (A8 revised 2026-09-14) ----------
|
||||
|
||||
|
||||
def test_gate_fts_hit_below_floor_deflects() -> None:
|
||||
"""The Mongolia case: FTS hit with cosine below lexical_support_floor
|
||||
→ LOW (deflected). The lexical-only hit no longer promotes to HIGH.
|
||||
This is the regression pin for phase 112."""
|
||||
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
|
||||
plan = chat_api.plan_turn(
|
||||
[_chunk(doc, 0.05, cosine=0.10, fts_hit=True)],
|
||||
_settings(threshold=0.62),
|
||||
)
|
||||
assert plan.deflected is True
|
||||
assert plan.top_score == pytest.approx(0.10)
|
||||
assert plan.fts_hits == 1
|
||||
assert "DEFLECT_MODE" in plan.system_prompt
|
||||
assert "QUEST_DOC_CONTENT" not in plan.system_prompt
|
||||
assert "Capital Quest" in plan.system_prompt # title only
|
||||
assert plan.suggestions # derived from weak-hit titles
|
||||
|
||||
|
||||
def test_gate_fts_hit_at_floor_answers() -> None:
|
||||
"""FTS hit with cosine exactly at lexical_support_floor → HIGH.
|
||||
The floor is inclusive (>=), not strict (<)."""
|
||||
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
|
||||
plan = chat_api.plan_turn(
|
||||
[_chunk(doc, 0.40, cosine=0.35, fts_hit=True)],
|
||||
_settings(threshold=0.62),
|
||||
)
|
||||
assert plan.deflected is False
|
||||
assert plan.top_score == pytest.approx(0.35)
|
||||
assert plan.fts_hits == 1
|
||||
assert "QUEST_DOC_CONTENT" in plan.system_prompt
|
||||
assert plan.suggestions == []
|
||||
|
||||
|
||||
def test_gate_fts_hit_above_floor_below_threshold_answers() -> None:
|
||||
"""FTS hit with cosine between floor and threshold → HIGH.
|
||||
The corroborated-lexical path fires."""
|
||||
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
|
||||
plan = chat_api.plan_turn(
|
||||
[_chunk(doc, 0.50, cosine=0.50, fts_hit=True)],
|
||||
_settings(threshold=0.62),
|
||||
)
|
||||
assert plan.deflected is False
|
||||
assert plan.top_score == pytest.approx(0.50)
|
||||
assert plan.fts_hits == 1
|
||||
assert "QUEST_DOC_CONTENT" in plan.system_prompt
|
||||
assert plan.suggestions == []
|
||||
|
||||
|
||||
def test_gate_fts_hit_above_code_default_floor_answers() -> None:
|
||||
"""The quadrant table's "0.50 with default settings" row: the CODE
|
||||
defaults (``test_lexical_support_floor_validation_default`` pins them:
|
||||
threshold 0.62 / floor 0.35) — fts>0 + cosine 0.50 >= 0.35 → HIGH.
|
||||
Named literally (not via the helper's half-threshold floor) so the
|
||||
production-default path is pinned on its own."""
|
||||
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
|
||||
plan = chat_api.plan_turn(
|
||||
[_chunk(doc, 0.50, cosine=0.50, fts_hit=True)],
|
||||
_settings(threshold=0.62, floor=0.35),
|
||||
)
|
||||
assert plan.deflected is False
|
||||
assert plan.top_score == pytest.approx(0.50)
|
||||
assert plan.fts_hits == 1
|
||||
assert "QUEST_DOC_CONTENT" in plan.system_prompt
|
||||
assert plan.suggestions == []
|
||||
|
||||
|
||||
def test_gate_high_cosine_overrides_fts_deflection() -> None:
|
||||
"""Strong cosine (>= threshold) → HIGH regardless of FTS status.
|
||||
The cosine-primary path is unchanged."""
|
||||
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
|
||||
plan = chat_api.plan_turn(
|
||||
[_chunk(doc, 0.90, cosine=0.80, fts_hit=True)],
|
||||
_settings(threshold=0.62),
|
||||
)
|
||||
assert plan.deflected is False
|
||||
assert plan.top_score == pytest.approx(0.80)
|
||||
assert plan.fts_hits == 1
|
||||
assert "QUEST_DOC_CONTENT" in plan.system_prompt
|
||||
assert plan.suggestions == []
|
||||
|
||||
|
||||
def test_gate_fts_no_cosine_deflects() -> None:
|
||||
"""FTS hit with cosine = 0.0 → LOW (the extreme Mongolia case)."""
|
||||
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
|
||||
plan = chat_api.plan_turn(
|
||||
[_chunk(doc, 0.90, cosine=0.0, fts_hit=True)],
|
||||
_settings(threshold=0.62),
|
||||
)
|
||||
assert plan.deflected is True
|
||||
assert plan.top_score == 0.0
|
||||
assert plan.fts_hits == 1
|
||||
assert "DEFLECT_MODE" in plan.system_prompt
|
||||
|
||||
|
||||
def test_gate_multiple_fts_below_floor_deflects() -> None:
|
||||
"""Multiple FTS hits, all below lexical_support_floor → LOW.
|
||||
The gate requires the BEST cosine to clear the floor, not just any hit."""
|
||||
a = _doc("Alpha Quest", "ALPHA_CONTENT")
|
||||
b = _doc("Beta Quest", "BETA_CONTENT")
|
||||
chunks = [
|
||||
_chunk(a, 0.30, cosine=0.20, fts_hit=True),
|
||||
_chunk(b, 0.25, cosine=0.15, fts_hit=True),
|
||||
]
|
||||
plan = chat_api.plan_turn(chunks, _settings(threshold=0.62))
|
||||
assert plan.deflected is True
|
||||
assert plan.fts_hits == 2
|
||||
assert "DEFLECT_MODE" in plan.system_prompt
|
||||
|
||||
|
||||
def test_gate_one_fts_above_floor_answers() -> None:
|
||||
"""Multiple chunks, one FTS hit above floor → HIGH.
|
||||
The best cosine (from the corroborated hit) clears the floor."""
|
||||
a = _doc("Alpha Quest", "ALPHA_CONTENT")
|
||||
b = _doc("Beta Quest", "BETA_CONTENT")
|
||||
chunks = [
|
||||
_chunk(a, 0.30, cosine=0.20, fts_hit=True), # below floor
|
||||
_chunk(b, 0.25, cosine=0.40, fts_hit=True), # above floor
|
||||
]
|
||||
plan = chat_api.plan_turn(chunks, _settings(threshold=0.62))
|
||||
assert plan.deflected is False
|
||||
assert plan.fts_hits == 2
|
||||
assert "ALPHA_CONTENT" in plan.system_prompt
|
||||
assert "BETA_CONTENT" in plan.system_prompt
|
||||
|
||||
|
||||
# ---------- config validation (lexical_support_floor) ----------
|
||||
|
||||
|
||||
def test_lexical_support_floor_validation_floor_above_threshold_fails() -> None:
|
||||
"""lexical_support_floor > relevance_threshold is rejected at startup."""
|
||||
with pytest.raises(ValueError, match="lexical_support_floor"):
|
||||
Settings(
|
||||
_env_file=None, # pyright: ignore[reportCallIssue]
|
||||
relevance_threshold=0.62,
|
||||
lexical_support_floor=0.70,
|
||||
)
|
||||
|
||||
|
||||
def test_lexical_support_floor_validation_negative_fails() -> None:
|
||||
"""Negative lexical_support_floor is rejected."""
|
||||
with pytest.raises(ValueError, match="lexical_support_floor"):
|
||||
Settings(
|
||||
_env_file=None, # pyright: ignore[reportCallIssue]
|
||||
lexical_support_floor=-0.1,
|
||||
)
|
||||
|
||||
|
||||
def test_lexical_support_floor_validation_at_threshold_succeeds() -> None:
|
||||
"""lexical_support_floor == relevance_threshold is legal."""
|
||||
s = Settings(
|
||||
_env_file=None, # pyright: ignore[reportCallIssue]
|
||||
relevance_threshold=0.62,
|
||||
lexical_support_floor=0.62,
|
||||
)
|
||||
assert s.lexical_support_floor == 0.62
|
||||
|
||||
|
||||
def test_lexical_support_floor_validation_default() -> None:
|
||||
"""Default lexical_support_floor is 0.35."""
|
||||
import os
|
||||
# Conftest sets BOR_RELEVANCE_THRESHOLD=0.30 and BOR_LEXICAL_SUPPORT_FLOOR=0.15.
|
||||
# We need the CODE defaults, so clear both and let the class defaults apply.
|
||||
saved_relevance = os.environ.pop("BOR_RELEVANCE_THRESHOLD", None)
|
||||
saved_floor = os.environ.pop("BOR_LEXICAL_SUPPORT_FLOOR", None)
|
||||
try:
|
||||
s = Settings(_env_file=None) # pyright: ignore[reportCallIssue]
|
||||
assert s.lexical_support_floor == 0.35
|
||||
assert s.relevance_threshold == 0.62
|
||||
finally:
|
||||
if saved_relevance is not None:
|
||||
os.environ["BOR_RELEVANCE_THRESHOLD"] = saved_relevance
|
||||
if saved_floor is not None:
|
||||
os.environ["BOR_LEXICAL_SUPPORT_FLOOR"] = saved_floor
|
||||
|
||||
|
||||
def test_gate_zero_chunks_deflects_with_fallback_chips() -> None:
|
||||
plan = chat_api.plan_turn([], _settings())
|
||||
assert plan.deflected is True
|
||||
@@ -581,6 +766,11 @@ def test_endpoint_just_below_threshold_deflects(
|
||||
assert 2 <= len(done["suggestions"]) <= MAX_SUGGESTIONS # title chip + fallback
|
||||
assert all(s.strip() for s in done["suggestions"])
|
||||
assert any("Deploying a New Service" in s for s in done["suggestions"])
|
||||
# Phase 112 (A8 revised, TODO L2): a deflected turn cites nothing —
|
||||
# done.sources is the citation surface (the UI chips every entry as
|
||||
# "the answer used this"), and the weak hits are scored docs, not
|
||||
# citations.
|
||||
assert done["sources"] == []
|
||||
|
||||
# The LLM saw the LOW prompt: DEFLECT_MODE + titles, never doc content.
|
||||
(system, user) = llm.seen[0][0], llm.seen[0][1]
|
||||
@@ -588,11 +778,14 @@ def test_endpoint_just_below_threshold_deflects(
|
||||
assert "DEFLECT_MODE" in system["content"]
|
||||
assert "DOC_CONTENT_NEVER_SENT" not in system["content"]
|
||||
|
||||
# Durable record: deflected + the weak score.
|
||||
# Durable record: deflected + the weak score. The retrieval itself
|
||||
# stays recorded (observability unchanged — query_log records
|
||||
# retrieval, not citations; the phase-113 A3 precedent).
|
||||
(row,) = session.added
|
||||
assert isinstance(row, QueryLog)
|
||||
assert row.deflected is True
|
||||
assert row.top_score == pytest.approx(0.2999)
|
||||
assert row.sources # the weak-hit doc's path, for threshold tuning
|
||||
assert session.commits == 1
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user