phase: 112_honesty_gate_weak_hits
Build and Push Containers / build-and-push-app (push) Successful in 2m15s
Build and Push Containers / build-and-push-db (push) Successful in 14s

**Phase 112 — final verification pass (all 4 tasks already complete in `complete/`):**

- Verified gate fix: `app/api/chat.py::plan_turn` — HIGH iff `best_cosine >= relevance_threshold` OR (`fts_hits > 0` AND `best_cosine >= lexical_support_floor`); `lexical_support_floor` (default 0.35, `BOR_LEXICAL_SUPPORT_FLOOR`, bounds-validated) in `app/config.py` + `.env.example`; A8 revision note (2026-09-14) in `.agents/PLAN.md`.
- Verified prompt contract: `app/rag/prompts.py` diff is docstring-only (dated owner-decision-iii entry); `tests/unit/test_prompt_lock.py` byte-pins PERSONA/TOOLS_SECTION/DEFLECT body (sha256+length).
- Verified README: L11 + L575 deflection copy refreshed; `grep "haven't done anything" README.md` → no hits; disclosed-answer behavior documented.
- Tests: `uv run pytest --cov=app --cov-report=term-missing` → **2378 passed, 99% coverage (>90%)**; includes Mongolia-quadrant unit pins (fts>0 + cosine<floor → LOW).
- E2E in isolation: `uv run pytest tests/e2e/test_honest_deflection.py -v --no-cov` → **4 passed** (out-of-KB question: `deflected=true`, `sources==[]`, 2–3 suggestions); regression `test_chat_rag.py` + `test_retrieval_quality.py` → **7 passed**.
- Lint/types: `uv run ruff check .` → clean; `uv run pyright` → 0 errors.

**Completion criteria:** weak-FTS→LOW unit-pinned ✅ · no false citations + 2–3 alternatives E2E ✅ · prompts byte-identical (test-pinned) + README matches ✅ · suite/coverage/e2e/lint all green ✅ · commit + phase move → left to harness (no `git commit` run, per rules; changes in working tree).

**Deviations:** none. Next pending phase: `113_source_chip_quality`.
This commit is contained in:
2026-09-15 00:37:38 -04:00
parent 2683128876
commit 1374faf136
36 changed files with 1240 additions and 67 deletions
+200 -7
View File
@@ -36,10 +36,13 @@ ANSWER = "I haven't done anything like that — try one of these instead!"
KB_OVERVIEW = "- Homelab\n - Kubernetes (k3s)\n- Deployments\n - Borg backups"
def _settings(threshold: float = 0.30) -> Settings:
def _settings(threshold: float = 0.30, floor: float | None = None) -> Settings:
if floor is None:
floor = threshold * 0.5 # half the threshold — keeps existing tests green
return Settings(
_env_file=None, # pyright: ignore[reportCallIssue]
relevance_threshold=threshold,
lexical_support_floor=floor,
)
@@ -118,12 +121,16 @@ def test_gate_is_env_tunable_via_settings() -> None:
def test_gate_weak_cosine_with_fts_hit_still_answers() -> None:
"""cosine < threshold but a lexical hit ⇒ HIGH — the FTS-OR branch.
This is the name-your-tool case: "kafkabridge" grounds despite weak
vector overlap."""
"""cosine < threshold but a lexical hit corroborated by cosine >= floor
⇒ HIGH — the FTS-OR branch. This is the name-your-tool case:
"kafkabridge" grounds despite weak vector overlap.
A8 revised 2026-09-14: FTS alone no longer promotes; cosine must also
clear lexical_support_floor (here 0.15 = half of threshold 0.30)."""
doc = _doc("Static DNS", "DNS_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.02, cosine=0.10, fts_hit=True)], _settings(threshold=0.30)
[_chunk(doc, 0.02, cosine=0.10, fts_hit=True)],
_settings(threshold=0.30, floor=0.05), # floor=0.05 so 0.10 >= floor
)
assert plan.deflected is False
assert plan.top_score == pytest.approx(0.10) # gate input is the cosine
@@ -155,7 +162,7 @@ def test_gate_fts_hits_counts_all_lexical_candidates() -> None:
_chunk(a, 0.02, cosine=0.04, fts_hit=True), # same doc, second chunk
_chunk(b, 0.01, cosine=0.03),
]
plan = chat_api.plan_turn(chunks, _settings(threshold=0.30))
plan = chat_api.plan_turn(chunks, _settings(threshold=0.30, floor=0.03))
assert plan.deflected is False
assert plan.fts_hits == 2 # per chunk, not per doc
@@ -176,6 +183,184 @@ def test_gate_lexical_only_chunk_does_not_inflate_cosine() -> None:
assert plan.docs[0].title == "Beta"
# ---------- lexical support floor (A8 revised 2026-09-14) ----------
def test_gate_fts_hit_below_floor_deflects() -> None:
"""The Mongolia case: FTS hit with cosine below lexical_support_floor
→ LOW (deflected). The lexical-only hit no longer promotes to HIGH.
This is the regression pin for phase 112."""
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.05, cosine=0.10, fts_hit=True)],
_settings(threshold=0.62),
)
assert plan.deflected is True
assert plan.top_score == pytest.approx(0.10)
assert plan.fts_hits == 1
assert "DEFLECT_MODE" in plan.system_prompt
assert "QUEST_DOC_CONTENT" not in plan.system_prompt
assert "Capital Quest" in plan.system_prompt # title only
assert plan.suggestions # derived from weak-hit titles
def test_gate_fts_hit_at_floor_answers() -> None:
"""FTS hit with cosine exactly at lexical_support_floor → HIGH.
The floor is inclusive (>=), not strict (<)."""
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.40, cosine=0.35, fts_hit=True)],
_settings(threshold=0.62),
)
assert plan.deflected is False
assert plan.top_score == pytest.approx(0.35)
assert plan.fts_hits == 1
assert "QUEST_DOC_CONTENT" in plan.system_prompt
assert plan.suggestions == []
def test_gate_fts_hit_above_floor_below_threshold_answers() -> None:
"""FTS hit with cosine between floor and threshold → HIGH.
The corroborated-lexical path fires."""
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.50, cosine=0.50, fts_hit=True)],
_settings(threshold=0.62),
)
assert plan.deflected is False
assert plan.top_score == pytest.approx(0.50)
assert plan.fts_hits == 1
assert "QUEST_DOC_CONTENT" in plan.system_prompt
assert plan.suggestions == []
def test_gate_fts_hit_above_code_default_floor_answers() -> None:
"""The quadrant table's "0.50 with default settings" row: the CODE
defaults (``test_lexical_support_floor_validation_default`` pins them:
threshold 0.62 / floor 0.35) — fts>0 + cosine 0.50 >= 0.35 → HIGH.
Named literally (not via the helper's half-threshold floor) so the
production-default path is pinned on its own."""
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.50, cosine=0.50, fts_hit=True)],
_settings(threshold=0.62, floor=0.35),
)
assert plan.deflected is False
assert plan.top_score == pytest.approx(0.50)
assert plan.fts_hits == 1
assert "QUEST_DOC_CONTENT" in plan.system_prompt
assert plan.suggestions == []
def test_gate_high_cosine_overrides_fts_deflection() -> None:
"""Strong cosine (>= threshold) → HIGH regardless of FTS status.
The cosine-primary path is unchanged."""
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.90, cosine=0.80, fts_hit=True)],
_settings(threshold=0.62),
)
assert plan.deflected is False
assert plan.top_score == pytest.approx(0.80)
assert plan.fts_hits == 1
assert "QUEST_DOC_CONTENT" in plan.system_prompt
assert plan.suggestions == []
def test_gate_fts_no_cosine_deflects() -> None:
"""FTS hit with cosine = 0.0 → LOW (the extreme Mongolia case)."""
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.90, cosine=0.0, fts_hit=True)],
_settings(threshold=0.62),
)
assert plan.deflected is True
assert plan.top_score == 0.0
assert plan.fts_hits == 1
assert "DEFLECT_MODE" in plan.system_prompt
def test_gate_multiple_fts_below_floor_deflects() -> None:
"""Multiple FTS hits, all below lexical_support_floor → LOW.
The gate requires the BEST cosine to clear the floor, not just any hit."""
a = _doc("Alpha Quest", "ALPHA_CONTENT")
b = _doc("Beta Quest", "BETA_CONTENT")
chunks = [
_chunk(a, 0.30, cosine=0.20, fts_hit=True),
_chunk(b, 0.25, cosine=0.15, fts_hit=True),
]
plan = chat_api.plan_turn(chunks, _settings(threshold=0.62))
assert plan.deflected is True
assert plan.fts_hits == 2
assert "DEFLECT_MODE" in plan.system_prompt
def test_gate_one_fts_above_floor_answers() -> None:
"""Multiple chunks, one FTS hit above floor → HIGH.
The best cosine (from the corroborated hit) clears the floor."""
a = _doc("Alpha Quest", "ALPHA_CONTENT")
b = _doc("Beta Quest", "BETA_CONTENT")
chunks = [
_chunk(a, 0.30, cosine=0.20, fts_hit=True), # below floor
_chunk(b, 0.25, cosine=0.40, fts_hit=True), # above floor
]
plan = chat_api.plan_turn(chunks, _settings(threshold=0.62))
assert plan.deflected is False
assert plan.fts_hits == 2
assert "ALPHA_CONTENT" in plan.system_prompt
assert "BETA_CONTENT" in plan.system_prompt
# ---------- config validation (lexical_support_floor) ----------
def test_lexical_support_floor_validation_floor_above_threshold_fails() -> None:
"""lexical_support_floor > relevance_threshold is rejected at startup."""
with pytest.raises(ValueError, match="lexical_support_floor"):
Settings(
_env_file=None, # pyright: ignore[reportCallIssue]
relevance_threshold=0.62,
lexical_support_floor=0.70,
)
def test_lexical_support_floor_validation_negative_fails() -> None:
"""Negative lexical_support_floor is rejected."""
with pytest.raises(ValueError, match="lexical_support_floor"):
Settings(
_env_file=None, # pyright: ignore[reportCallIssue]
lexical_support_floor=-0.1,
)
def test_lexical_support_floor_validation_at_threshold_succeeds() -> None:
"""lexical_support_floor == relevance_threshold is legal."""
s = Settings(
_env_file=None, # pyright: ignore[reportCallIssue]
relevance_threshold=0.62,
lexical_support_floor=0.62,
)
assert s.lexical_support_floor == 0.62
def test_lexical_support_floor_validation_default() -> None:
"""Default lexical_support_floor is 0.35."""
import os
# Conftest sets BOR_RELEVANCE_THRESHOLD=0.30 and BOR_LEXICAL_SUPPORT_FLOOR=0.15.
# We need the CODE defaults, so clear both and let the class defaults apply.
saved_relevance = os.environ.pop("BOR_RELEVANCE_THRESHOLD", None)
saved_floor = os.environ.pop("BOR_LEXICAL_SUPPORT_FLOOR", None)
try:
s = Settings(_env_file=None) # pyright: ignore[reportCallIssue]
assert s.lexical_support_floor == 0.35
assert s.relevance_threshold == 0.62
finally:
if saved_relevance is not None:
os.environ["BOR_RELEVANCE_THRESHOLD"] = saved_relevance
if saved_floor is not None:
os.environ["BOR_LEXICAL_SUPPORT_FLOOR"] = saved_floor
def test_gate_zero_chunks_deflects_with_fallback_chips() -> None:
plan = chat_api.plan_turn([], _settings())
assert plan.deflected is True
@@ -581,6 +766,11 @@ def test_endpoint_just_below_threshold_deflects(
assert 2 <= len(done["suggestions"]) <= MAX_SUGGESTIONS # title chip + fallback
assert all(s.strip() for s in done["suggestions"])
assert any("Deploying a New Service" in s for s in done["suggestions"])
# Phase 112 (A8 revised, TODO L2): a deflected turn cites nothing —
# done.sources is the citation surface (the UI chips every entry as
# "the answer used this"), and the weak hits are scored docs, not
# citations.
assert done["sources"] == []
# The LLM saw the LOW prompt: DEFLECT_MODE + titles, never doc content.
(system, user) = llm.seen[0][0], llm.seen[0][1]
@@ -588,11 +778,14 @@ def test_endpoint_just_below_threshold_deflects(
assert "DEFLECT_MODE" in system["content"]
assert "DOC_CONTENT_NEVER_SENT" not in system["content"]
# Durable record: deflected + the weak score.
# Durable record: deflected + the weak score. The retrieval itself
# stays recorded (observability unchanged — query_log records
# retrieval, not citations; the phase-113 A3 precedent).
(row,) = session.added
assert isinstance(row, QueryLog)
assert row.deflected is True
assert row.top_score == pytest.approx(0.2999)
assert row.sources # the weak-hit doc's path, for threshold tuning
assert session.commits == 1