Files
brain-of-reese/tests/unit/test_chat_gate.py
T
ducoterra a5b63f83ad
Build and Push Containers / build-and-push-app (push) Successful in 2m1s
Build and Push Containers / build-and-push-db (push) Successful in 18s
phase: 119_name_signal_read_chips
All verification complete. Final report:

**Phase 119 final verification pass — all criteria verified, one stale pin fixed.**
- Verified implementation of all 6 tasks: D1 component name-hit rule (`name_hit` flag, titles never matched, retired length tie-break), D2 `BOR_NAME_HIT_BONUS` (0.005 default, 0 = byte-identical kill switch, negative fails startup, selection-layer only, `eval_retrieval` `suggested:` line), D3 suggested-folder lines (after `SUGGEST_INTRO`, before first block), D4 cite-discipline `SUGGEST_INTRO` sentence (PERSONA/LOW/`TOOLS_SECTION` byte-pins intact), D5 `done.sources` = read docs only (frontend no-op on empty confirmed), D6 mock `repeat your folder map` echo + new suite + telemetry.
- Battery (replica restored per skill, fingerprint docs=1000/chunks=8866 verified, `eval_retrieval --from-file tests/fixtures/retrieval_battery.txt` re-run): **GATE PASS** — gitea README #4 in suggested top-5, forgejo 5/5 (README #1), gateway README in top-5 (#4), qwen3.8-27b quadlets top-5, Mongolia HIGH/fts=5 unchanged.
- New E2E in isolation: `4 passed` ×2 (deterministic). All 27 modified E2E suites in isolation: 26 green; **1 stale pin fixed** — `test_source_chip_quality.py` durable-record order pin pre-dated the D1 re-rank (`aliases` stem sub-component name-hits `ssh_aliases.txt`, deterministically lifting `backups.md` over `kubernetes.md`; probe-verified 0.016277 vs 0.016036, 4/4 stable) — re-pinned with the phase-119 rationale; suite green ×2.
- Gates: `uv run pytest --cov=app --cov-report=term-missing` → **2547 passed, app coverage 99%** (>90%); `uv run ruff check .` → All checks passed; `uv run pyright` → 0 errors.
- Completion criteria: 1 ✅ (battery, recorded), 2 ✅ (folder lines; block/LOW byte-identical pins green), 3 ✅ (read-only chips, zero-read chips nothing, related row + durable record untouched — unit+E2E agree), 4 ✅ (all green), 5 → commit/phase-move left to the harness per pass rules (nothing committed).
- Deviations: battery output + real-model telemetry recorded in `.agents/reports/119_name_signal_read_chips/task06_battery_and_e2e.md` and `TOOL_CALLING_TESTING.md` §11 (task files in `complete/` are immutable to this pass); gateway canonical doc at #4 vs overview's #3 was already documented at task 06 (containment gate met).
- Next pending phase: **none** — `todo/` holds only phase 119.
2026-09-16 15:50:48 -04:00

1429 lines
58 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Unit: the honesty gate (A8) — boundary, prompts, and suggestion chips.
Pure gate logic runs against fake retriever output (``RetrievedChunk``
rows from a fake retriever) with no Postgres and no network. The
endpoint-level tests drive ``POST /api/chat`` with the retriever, the DB
session, and the LLM all faked, so the whole deflection contract
(prompt → deltas → done event → query_log) is verified without a stack.
"""
from __future__ import annotations
import hashlib
import json
import uuid
from collections.abc import Iterator, Sequence
from datetime import UTC, datetime
from typing import TYPE_CHECKING, Any
import pytest
from fastapi.testclient import TestClient
from app.api import chat as chat_api
from app.config import Settings
from app.main import app as fastapi_app
from app.models import Document, KbOverview, QueryLog
from app.rag import agent as rag_agent
from app.rag.agent import AGENT_TOOLS
from app.rag.llm import StreamPiece, ToolCallPiece
from app.rag.prompts import SUGGEST_INTRO, build_deflect_prompt
from app.rag.retriever import RetrievedChunk, weak_hit_titles
from app.rag.suggestions import MAX_SUGGESTIONS, derive_suggestions
from tests.conftest import ADMIN_PASSWORD
if TYPE_CHECKING:
from app.rag.scaffolding import ScaffoldingFilter
ANSWER = "I haven't done anything like that — try one of these instead!"
#: A small KB outline standing in for the lite-generated one (phase 31).
KB_OVERVIEW = "- Homelab\n - Kubernetes (k3s)\n- Deployments\n - Borg backups"
def _settings(threshold: float = 0.30, floor: float | None = None) -> Settings:
if floor is None:
floor = threshold * 0.5 # half the threshold — keeps existing tests green
return Settings(
_env_file=None, # pyright: ignore[reportCallIssue]
relevance_threshold=threshold,
lexical_support_floor=floor,
)
def _doc(
title: str, content: str, summary: str | None = None
) -> Document:
return Document(
id=uuid.uuid4(),
source="Homelab",
path=f"{title.lower().replace(' ', '-')}.md",
full_path="/tmp/doc.md",
title=title,
content=content,
# Phase 30/118: the stored lite-model summary — the HIGH block's
# BODY (task 03). ``None`` exercises the A5 preview fallback
# (the first ``suggestion_preview_chars`` content chars).
summary=summary,
content_hash="0" * 64,
# Phase 106, D5: the HIGH block formats the row's created_at
# UTC date part — the detached fixture carries it (the NOT NULL
# DB column guarantees it for real rows).
created_at=datetime(2024, 6, 15, 12, 0, 0, tzinfo=UTC),
)
def _chunk(
doc: Document,
score: float,
cosine: float | None = None,
fts_hit: bool = False,
is_summary: bool = False,
) -> RetrievedChunk:
"""Fake candidate: *score* is the fused rank score; *cosine* (defaults to
*score*) is the vector-similarity gate input."""
return RetrievedChunk(
chunk_id=uuid.uuid4(),
position=-1 if is_summary else 0,
content=doc.content[:32],
score=score,
document=doc,
cosine=score if cosine is None else cosine,
fts_hit=fts_hit,
is_summary=is_summary,
)
# ---------- gate boundary (fake retriever rows, no LLM) ----------
def test_gate_boundary_score_at_threshold_answers() -> None:
"""Score exactly at the threshold ⇒ HIGH (the gate is strict <)."""
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_CONTENT")
plan = chat_api.plan_turn([_chunk(doc, 0.30)], _settings(threshold=0.30))
assert plan.deflected is False
assert plan.top_score == pytest.approx(0.30)
assert "<relevance>HIGH</relevance>" in plan.system_prompt
assert "DEFLECT_MODE" not in plan.system_prompt
assert "TALOS_DOC_CONTENT" in plan.system_prompt
assert plan.suggestions == []
def test_gate_boundary_just_below_threshold_deflects() -> None:
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_CONTENT")
plan = chat_api.plan_turn([_chunk(doc, 0.2999)], _settings(threshold=0.30))
assert plan.deflected is True
assert plan.top_score == pytest.approx(0.2999)
assert "<relevance>LOW</relevance>" in plan.system_prompt
assert "DEFLECT_MODE" in plan.system_prompt
# Titles only: the full document content must never reach the LLM.
assert "TALOS_DOC_CONTENT" not in plan.system_prompt
assert "Kubernetes Homelab Cluster" in plan.system_prompt
def test_gate_is_env_tunable_via_settings() -> None:
doc = _doc("Backup Strategy", "BACKUP_DOC_CONTENT")
hits = [_chunk(doc, 0.30)]
assert chat_api.plan_turn(hits, _settings(threshold=0.35)).deflected is True
assert chat_api.plan_turn(hits, _settings(threshold=0.25)).deflected is False
# ---------- hybrid gate matrix (A8, revised: cosine AND fts) ----------
def test_gate_weak_cosine_with_fts_hit_still_answers() -> None:
"""cosine < threshold but a lexical hit corroborated by cosine >= floor
⇒ HIGH — the FTS-OR branch. This is the name-your-tool case:
"kafkabridge" grounds despite weak vector overlap.
A8 revised 2026-09-14: FTS alone no longer promotes; cosine must also
clear lexical_support_floor (here 0.15 = half of threshold 0.30)."""
doc = _doc("Static DNS", "DNS_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.02, cosine=0.10, fts_hit=True)],
_settings(threshold=0.30, floor=0.05), # floor=0.05 so 0.10 >= floor
)
assert plan.deflected is False
assert plan.top_score == pytest.approx(0.10) # gate input is the cosine
assert plan.fts_hits == 1
assert "DNS_DOC_CONTENT" in plan.system_prompt
assert plan.suggestions == []
def test_gate_weak_cosine_zero_fts_deflects() -> None:
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_CONTENT")
plan = chat_api.plan_turn([_chunk(doc, 0.02, cosine=0.10)], _settings(threshold=0.30))
assert plan.deflected is True
assert plan.top_score == pytest.approx(0.10)
assert plan.fts_hits == 0
def test_gate_strong_cosine_without_fts_answers() -> None:
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_CONTENT")
plan = chat_api.plan_turn([_chunk(doc, 0.90, cosine=0.90)], _settings(threshold=0.30))
assert plan.deflected is False
assert plan.fts_hits == 0
def test_gate_fts_hits_counts_all_lexical_candidates() -> None:
a = _doc("Alpha", "ALPHA_CONTENT")
b = _doc("Beta", "BETA_CONTENT")
chunks = [
_chunk(a, 0.03, cosine=0.05, fts_hit=True),
_chunk(a, 0.02, cosine=0.04, fts_hit=True), # same doc, second chunk
_chunk(b, 0.01, cosine=0.03),
]
plan = chat_api.plan_turn(chunks, _settings(threshold=0.30, floor=0.03))
assert plan.deflected is False
assert plan.fts_hits == 2 # per chunk, not per doc
def test_gate_lexical_only_chunk_does_not_inflate_cosine() -> None:
"""top_score stays the best *vector* cosine even when a lexical-only
chunk (cosine 0.0 by construction) carries the highest fused score."""
a = _doc("Alpha", "ALPHA_CONTENT")
b = _doc("Beta", "BETA_CONTENT")
chunks = [
_chunk(a, 0.50, cosine=0.55), # vector rank 1
_chunk(b, 0.90, cosine=0.0, fts_hit=True), # lexical rank 1 wins the ranking
]
plan = chat_api.plan_turn(chunks, _settings(threshold=0.30))
assert plan.top_score == pytest.approx(0.55)
assert plan.deflected is False # 0.55 >= 0.30 anyway
# Phase 118 (A3): the suggestion tier has NO floor — the lexical-
# only doc (cosine 0.0 by construction) is SUGGESTED when it ranks.
# Rank order is the fused score, so Beta (0.90) leads Alpha (0.50);
# with two docs nothing is left for the related tier (rank 6+).
assert [d.title for d in plan.suggested_docs] == ["Beta", "Alpha"]
assert plan.related_docs == []
# ---------- lexical support floor (A8 revised 2026-09-14) ----------
def test_gate_fts_hit_below_floor_deflects() -> None:
"""The Mongolia case: FTS hit with cosine below lexical_support_floor
→ LOW (deflected). The lexical-only hit no longer promotes to HIGH.
This is the regression pin for phase 112."""
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.05, cosine=0.10, fts_hit=True)],
_settings(threshold=0.62),
)
assert plan.deflected is True
assert plan.top_score == pytest.approx(0.10)
assert plan.fts_hits == 1
assert "DEFLECT_MODE" in plan.system_prompt
assert "QUEST_DOC_CONTENT" not in plan.system_prompt
assert "Capital Quest" in plan.system_prompt # title only
assert plan.suggestions # derived from weak-hit titles
def test_gate_fts_hit_at_floor_answers() -> None:
"""FTS hit with cosine exactly at lexical_support_floor → HIGH.
The floor is inclusive (>=), not strict (<)."""
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.40, cosine=0.35, fts_hit=True)],
_settings(threshold=0.62),
)
assert plan.deflected is False
assert plan.top_score == pytest.approx(0.35)
assert plan.fts_hits == 1
assert "QUEST_DOC_CONTENT" in plan.system_prompt
assert plan.suggestions == []
def test_gate_fts_hit_above_floor_below_threshold_answers() -> None:
"""FTS hit with cosine between floor and threshold → HIGH.
The corroborated-lexical path fires."""
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.50, cosine=0.50, fts_hit=True)],
_settings(threshold=0.62),
)
assert plan.deflected is False
assert plan.top_score == pytest.approx(0.50)
assert plan.fts_hits == 1
assert "QUEST_DOC_CONTENT" in plan.system_prompt
assert plan.suggestions == []
def test_gate_fts_hit_above_code_default_floor_answers() -> None:
"""The quadrant table's "0.50 with default settings" row: the CODE
defaults (``test_lexical_support_floor_validation_default`` pins them:
threshold 0.62 / floor 0.35) — fts>0 + cosine 0.50 >= 0.35 → HIGH.
Named literally (not via the helper's half-threshold floor) so the
production-default path is pinned on its own."""
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.50, cosine=0.50, fts_hit=True)],
_settings(threshold=0.62, floor=0.35),
)
assert plan.deflected is False
assert plan.top_score == pytest.approx(0.50)
assert plan.fts_hits == 1
assert "QUEST_DOC_CONTENT" in plan.system_prompt
assert plan.suggestions == []
def test_gate_high_cosine_overrides_fts_deflection() -> None:
"""Strong cosine (>= threshold) → HIGH regardless of FTS status.
The cosine-primary path is unchanged."""
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.90, cosine=0.80, fts_hit=True)],
_settings(threshold=0.62),
)
assert plan.deflected is False
assert plan.top_score == pytest.approx(0.80)
assert plan.fts_hits == 1
assert "QUEST_DOC_CONTENT" in plan.system_prompt
assert plan.suggestions == []
def test_gate_fts_no_cosine_deflects() -> None:
"""FTS hit with cosine = 0.0 → LOW (the extreme Mongolia case)."""
doc = _doc("Capital Quest", "QUEST_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.90, cosine=0.0, fts_hit=True)],
_settings(threshold=0.62),
)
assert plan.deflected is True
assert plan.top_score == 0.0
assert plan.fts_hits == 1
assert "DEFLECT_MODE" in plan.system_prompt
def test_gate_multiple_fts_below_floor_deflects() -> None:
"""Multiple FTS hits, all below lexical_support_floor → LOW.
The gate requires the BEST cosine to clear the floor, not just any hit."""
a = _doc("Alpha Quest", "ALPHA_CONTENT")
b = _doc("Beta Quest", "BETA_CONTENT")
chunks = [
_chunk(a, 0.30, cosine=0.20, fts_hit=True),
_chunk(b, 0.25, cosine=0.15, fts_hit=True),
]
plan = chat_api.plan_turn(chunks, _settings(threshold=0.62))
assert plan.deflected is True
assert plan.fts_hits == 2
assert "DEFLECT_MODE" in plan.system_prompt
def test_gate_one_fts_above_floor_answers() -> None:
"""Multiple chunks, one FTS hit above floor → HIGH.
The best cosine (from the corroborated hit) clears the floor."""
a = _doc("Alpha Quest", "ALPHA_CONTENT")
b = _doc("Beta Quest", "BETA_CONTENT")
chunks = [
_chunk(a, 0.30, cosine=0.20, fts_hit=True), # below floor
_chunk(b, 0.25, cosine=0.40, fts_hit=True), # above floor
]
plan = chat_api.plan_turn(chunks, _settings(threshold=0.62))
assert plan.deflected is False
assert plan.fts_hits == 2
assert "ALPHA_CONTENT" in plan.system_prompt
assert "BETA_CONTENT" in plan.system_prompt
# ---------- config validation (lexical_support_floor) ----------
def test_lexical_support_floor_validation_floor_above_threshold_fails() -> None:
"""lexical_support_floor > relevance_threshold is rejected at startup."""
with pytest.raises(ValueError, match="lexical_support_floor"):
Settings(
_env_file=None, # pyright: ignore[reportCallIssue]
relevance_threshold=0.62,
lexical_support_floor=0.70,
)
def test_lexical_support_floor_validation_negative_fails() -> None:
"""Negative lexical_support_floor is rejected."""
with pytest.raises(ValueError, match="lexical_support_floor"):
Settings(
_env_file=None, # pyright: ignore[reportCallIssue]
lexical_support_floor=-0.1,
)
def test_lexical_support_floor_validation_at_threshold_succeeds() -> None:
"""lexical_support_floor == relevance_threshold is legal."""
s = Settings(
_env_file=None, # pyright: ignore[reportCallIssue]
relevance_threshold=0.62,
lexical_support_floor=0.62,
)
assert s.lexical_support_floor == 0.62
def test_lexical_support_floor_validation_default() -> None:
"""Default lexical_support_floor is 0.35."""
import os
# Conftest sets BOR_RELEVANCE_THRESHOLD=0.30 and BOR_LEXICAL_SUPPORT_FLOOR=0.15.
# We need the CODE defaults, so clear both and let the class defaults apply.
saved_relevance = os.environ.pop("BOR_RELEVANCE_THRESHOLD", None)
saved_floor = os.environ.pop("BOR_LEXICAL_SUPPORT_FLOOR", None)
try:
s = Settings(_env_file=None) # pyright: ignore[reportCallIssue]
assert s.lexical_support_floor == 0.35
assert s.relevance_threshold == 0.62
finally:
if saved_relevance is not None:
os.environ["BOR_RELEVANCE_THRESHOLD"] = saved_relevance
if saved_floor is not None:
os.environ["BOR_LEXICAL_SUPPORT_FLOOR"] = saved_floor
def test_gate_zero_chunks_deflects_with_fallback_chips() -> None:
plan = chat_api.plan_turn([], _settings())
assert plan.deflected is True
assert plan.top_score == 0.0
assert "nothing close at all" in plan.system_prompt
# No weak hits ⇒ onboarding fallback fills the chips.
assert 2 <= len(plan.suggestions) <= MAX_SUGGESTIONS
# ---------- summary-seed tiering (phase 118, LOCKED A3/A4/A6) ----------
def _tier_settings(
threshold: float = 0.62,
lex_floor: float = 0.35,
suggested_cap: int = 5,
related_cap: int = 2,
) -> Settings:
"""Explicit code defaults (production calibration) — the env's mock-
calibrated values (tests/conftest.py) are overridden per test.
``source_usefulness_floor`` / ``top_n_docs`` are deliberately left
at their code defaults: phase 118 retired their seeding role (A6) —
``plan_turn`` never consults them (pinned in
``test_plan_turn_does_not_consult_retired_seeding_settings``)."""
return Settings(
_env_file=None, # pyright: ignore[reportCallIssue]
relevance_threshold=threshold,
lexical_support_floor=lex_floor,
suggested_docs=suggested_cap,
related_max_docs=related_cap,
)
def _seven_docs_with_summaries() -> list[Document]:
"""The 7-doc fixture (rank 1–7 by fused score): every document has a
stored SUMMARY (distinct sentinel) and a distinct FULL-CONTENT
sentinel that must never reach the prompt (A6: the ``read`` tool is
the only full-text path)."""
return [
_doc(
f"Doc {i}",
f"FULL_CONTENT_SENTINEL_{i}_SHOULD_NEVER_REACH_THE_PROMPT",
summary=f"SUMMARY_TEXT_{i}",
)
for i in range(7)
]
def test_plan_turn_high_seeds_top5_suggested_related_is_rank6plus() -> None:
"""The phase-118 core pin (LOCKED A3/A6): the HIGH prompt seeds
exactly the top-5 suggested documents' SUMMARY text and NONE of
their full content; the related tier is rank 6+ (docs 6–7, capped
by ``related_max_docs``)."""
docs_in = _seven_docs_with_summaries()
chunks = [
_chunk(d, 0.9 - 0.1 * i, cosine=0.8 - 0.05 * i) for i, d in enumerate(docs_in)
]
plan = chat_api.plan_turn(chunks, _tier_settings())
assert plan.deflected is False
assert [d.title for d in plan.suggested_docs] == [f"Doc {i}" for i in range(5)]
assert [d.title for d in plan.related_docs] == ["Doc 5", "Doc 6"]
# The prompt seeds the five summaries …
for i in range(5):
assert f"SUMMARY_TEXT_{i}" in plan.system_prompt
# … and NONE of the seven documents' full content (suggested OR
# related) reaches the LLM (A6).
for i in range(7):
assert f"FULL_CONTENT_SENTINEL_{i}" not in plan.system_prompt
def test_plan_turn_folder_lines_ride_the_high_prompt() -> None:
"""Phase 119 (D3, LOCKED A4): plan_turn passes *folder_lines* through
to the HIGH prompt — after ``SUGGEST_INTRO``, before the first
``<document>`` block (each on its own line)."""
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_CONTENT")
lines = ["Homelab/: kubernetes-homelab-cluster.md (2 docs)"]
plan = chat_api.plan_turn(
[_chunk(doc, 0.90)], _settings(threshold=0.30), folder_lines=lines
)
assert plan.deflected is False
prompt = plan.system_prompt
i_open = prompt.index("<documents>")
i_block = prompt.index("<document ")
assert prompt[i_open:i_block] == f"<documents>\n{SUGGEST_INTRO}\n{lines[0]}\n\n"
def test_plan_turn_folder_lines_default_keeps_phase_118_high_prompt() -> None:
"""Omitted *folder_lines* (the default ``()``) ⇒ the HIGH prompt is
the phase-118 shape, byte-identical."""
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_CONTENT")
plan = chat_api.plan_turn([_chunk(doc, 0.90)], _settings(threshold=0.30))
assert plan.deflected is False
prompt = plan.system_prompt
i_open = prompt.index("<documents>")
i_block = prompt.index("<document ")
assert prompt[i_open:i_block] == f"<documents>\n{SUGGEST_INTRO}\n\n"
def test_plan_turn_folder_lines_ignored_on_the_low_branch() -> None:
"""The LOW (deflected) branch IGNORES *folder_lines* — the
deflected prompt stays byte-identical (LOCKED A4)."""
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_CONTENT")
low_with = chat_api.plan_turn(
[_chunk(doc, 0.05, cosine=0.05)],
_settings(threshold=0.30),
folder_lines=["Homelab/: x.md"],
)
low_without = chat_api.plan_turn(
[_chunk(doc, 0.05, cosine=0.05)], _settings(threshold=0.30)
)
assert low_with.deflected is True
assert low_with.system_prompt == low_without.system_prompt
assert "Homelab/: x.md" not in low_with.system_prompt
def test_plan_turn_high_single_strong_doc_yields_one_suggested() -> None:
"""The suggested cap is a CEILING, not a quota: one doc ⇒ one
suggested doc, an empty related tier (nothing beyond rank 1)."""
strong = _doc("Kubernetes Homelab Cluster", "STRONG_DOC_CONTENT")
plan = chat_api.plan_turn([_chunk(strong, 0.90, cosine=0.80)], _tier_settings())
assert plan.deflected is False
assert [d.title for d in plan.suggested_docs] == ["Kubernetes Homelab Cluster"]
assert plan.related_docs == []
# No stored summary (the fixture default) → the A5 preview fallback
# carries the short content whole (under the 400-char cap).
assert "STRONG_DOC_CONTENT" in plan.system_prompt
def test_plan_turn_high_suggests_strong_and_weak_no_floor() -> None:
"""The recurring incident under phase 118 (A3): the weak 2nd doc no
longer loses a seeding slot to a bar — the floor never filters, so
BOTH docs are suggested (rank order) and ride the durable record
(118-A3 — the phase-119 A1 citation surface is the read docs only);
the HIGH prompt seeds both summaries (the A5 fallback carries the
short fixture content whole)."""
strong = _doc("Kubernetes Homelab Cluster", "STRONG_DOC_CONTENT")
weak = _doc("Backup Strategy", "WEAK_DOC_CONTENT")
chunks = [_chunk(strong, 0.90, cosine=0.80), _chunk(weak, 0.80, cosine=0.20)]
plan = chat_api.plan_turn(chunks, _tier_settings())
assert plan.deflected is False
assert [d.title for d in plan.suggested_docs] == [
"Kubernetes Homelab Cluster",
"Backup Strategy",
]
assert plan.related_docs == []
assert "STRONG_DOC_CONTENT" in plan.system_prompt
assert "WEAK_DOC_CONTENT" in plan.system_prompt
def test_plan_turn_low_weak_hits_are_suggested_record() -> None:
"""Deflected turn: the weak hits are SUGGESTED too (no floor, A3) —
the TurnPlan carries suggested + related for the durable record —
while the LOW prompt itself stays byte-identical (weak-hit titles
only, never content)."""
a = _doc("Alpha", "ALPHA_DOC_NEVER_SENT")
b = _doc("Beta", "BETA_DOC_NEVER_SENT")
chunks = [_chunk(a, 0.30, cosine=0.20), _chunk(b, 0.20, cosine=0.15)]
plan = chat_api.plan_turn(chunks, _tier_settings())
assert plan.deflected is True
assert [d.title for d in plan.suggested_docs] == ["Alpha", "Beta"] # rank order
assert plan.related_docs == [] # nothing beyond rank 2 for 2 docs
assert "ALPHA_DOC_NEVER_SENT" not in plan.system_prompt
assert "Beta" in plan.system_prompt # weak-hit titles still carried
assert plan.suggestions # chips unchanged
def test_plan_turn_related_cap_zero_kills_the_related_tier() -> None:
"""related_max_docs=0 is the kill switch: rank-6+ docs are scored
and suggested-adjacent but neither suggested nor related — the
done frame's row stays empty."""
docs_in = _seven_docs_with_summaries()
chunks = [
_chunk(d, 0.9 - 0.1 * i, cosine=0.8 - 0.05 * i) for i, d in enumerate(docs_in)
]
plan = chat_api.plan_turn(chunks, _tier_settings(related_cap=0))
assert plan.deflected is False
assert len(plan.suggested_docs) == 5 # the suggestion tier is untouched
assert plan.related_docs == []
# The cap restores the rank-6+ row (rank order, capped).
wide = chat_api.plan_turn(chunks, _tier_settings(related_cap=3))
assert [d.title for d in wide.related_docs] == ["Doc 5", "Doc 6"]
def test_plan_turn_suggested_docs_setting_caps_the_suggested_tier() -> None:
"""``suggested_docs`` (``BOR_SUGGESTED_DOCS``) is honored as the
suggestion cap: 3 here ⇒ the top-3 rank-ordered docs are suggested
and rank 4+ falls to the related tier (capped at ``related_max_docs``)."""
docs_in = _seven_docs_with_summaries()
chunks = [
_chunk(d, 0.9 - 0.1 * i, cosine=0.8 - 0.05 * i) for i, d in enumerate(docs_in)
]
plan = chat_api.plan_turn(chunks, _tier_settings(suggested_cap=3))
assert [d.title for d in plan.suggested_docs] == ["Doc 0", "Doc 1", "Doc 2"]
assert [d.title for d in plan.related_docs] == ["Doc 3", "Doc 4"] # cap 2
def test_plan_turn_does_not_consult_retired_seeding_settings() -> None:
"""Phase 118 (A6): ``top_n_docs`` and ``source_usefulness_floor``
lost their seeding role — ``plan_turn`` never consults them. A
degenerate config (the maximum legal bar — 0.62, above every chunk's
cosine of 0.50 — and a top-N of 1) changes nothing: the suggested
tier is still the no-floor top-5 in rank order and the related tier
is still rank 6+.
(Behavioral pin — the settings themselves stay, env back-compat.)"""
docs_in = _seven_docs_with_summaries()
chunks = [
_chunk(docs_in[0], 0.9, cosine=0.50, fts_hit=True),
] + [
_chunk(d, 0.9 - 0.1 * i, cosine=0.50)
for i, d in enumerate(docs_in[1:], start=1)
]
settings = Settings(
_env_file=None, # pyright: ignore[reportCallIssue]
relevance_threshold=0.62,
lexical_support_floor=0.35,
top_n_docs=1, # retired: the old full-text seeding ceiling
# retired: the maximum legal bar (== threshold) — above every
# cosine here (0.50), so the OLD tiering would cite nothing.
source_usefulness_floor=0.62,
related_max_docs=2,
)
plan = chat_api.plan_turn(chunks, settings)
assert plan.deflected is False # 0.50 >= the 0.35 lex floor, fts fired
assert [d.title for d in plan.suggested_docs] == [f"Doc {i}" for i in range(5)]
assert [d.title for d in plan.related_docs] == ["Doc 5", "Doc 6"]
# ---------- summary hits (phase 30: summary → full source document) ----------
def test_summary_hit_on_suggested_doc_counts() -> None:
"""HIGH branch: a suggested (rank-1) document hit via its summary
chunk ⇒ 1. Phase 118 (A6): the *summary* is what the LLM sees in
the prompt (no stored summary here → the A5 preview fallback carries
the short fixture content whole)."""
a = _doc("Alpha", "ALPHA_FULL_SOURCE_CONTENT")
b = _doc("Beta", "BETA_FULL_SOURCE_CONTENT")
chunks = [
_chunk(a, 0.90, is_summary=True), # suggested doc reached through its summary
_chunk(b, 0.50),
]
plan = chat_api.plan_turn(chunks, _settings(threshold=0.30))
assert plan.deflected is False
assert plan.summary_hits == 1
# The A5 preview fallback (short content, under the 400-char cap)
# carries the content whole — the block body, never more.
assert "ALPHA_FULL_SOURCE_CONTENT" in plan.system_prompt
def test_summary_hits_counts_suggested_parent_only() -> None:
"""Phase 118 redefinition (redefined from the phase-113 cited set):
a summary chunk counts ONLY when its parent document is in the
SUGGESTED set — a rank-1 (suggested) parent counts, a rank-6
(related-tier-only) parent does not."""
docs_in = [_doc(f"Doc {i}", f"CONTENT_{i}") for i in range(7)]
chunks = [
_chunk(docs_in[0], 0.90, is_summary=True), # suggested parent — counts
_chunk(docs_in[1], 0.80),
_chunk(docs_in[2], 0.70),
_chunk(docs_in[3], 0.60),
_chunk(docs_in[4], 0.50),
_chunk(docs_in[5], 0.40, is_summary=True), # related-only parent — does not
_chunk(docs_in[6], 0.30),
]
plan = chat_api.plan_turn(chunks, _settings(threshold=0.30))
assert plan.deflected is False
assert [d.title for d in plan.suggested_docs] == [f"Doc {i}" for i in range(5)]
assert [d.title for d in plan.related_docs] == ["Doc 5", "Doc 6"]
assert plan.summary_hits == 1
def test_low_branch_counts_summary_hit_on_suggested_doc() -> None:
"""LOW (deflected) branch records ``summary_hits`` too: the weak hit's
parent is still the suggested (weak-hit) document (no floor, A3)."""
a = _doc("Gamma", "GAMMA_DOC_CONTENT")
b = _doc("Delta", "DELTA_DOC_CONTENT")
chunks = [
_chunk(a, 0.05, cosine=0.05, is_summary=True), # weak cosine, no FTS
_chunk(b, 0.03, cosine=0.03),
]
plan = chat_api.plan_turn(chunks, _settings(threshold=0.30))
assert plan.deflected is True
assert plan.summary_hits == 1
def test_no_summary_chunks_yields_zero_summary_hits() -> None:
"""Legacy chunks (``is_summary=false``) keep ``summary_hits == 0``."""
a = _doc("Alpha", "ALPHA_CONTENT")
b = _doc("Beta", "BETA_CONTENT")
plan = chat_api.plan_turn([_chunk(a, 0.90), _chunk(b, 0.40)], _settings(threshold=0.30))
assert plan.summary_hits == 0
plan_low = chat_api.plan_turn([_chunk(a, 0.05, cosine=0.05)], _settings(threshold=0.30))
assert plan_low.deflected is True
assert plan_low.summary_hits == 0
# ---------- KB overview (phase 31: <knowledge_base> section + kb_chars) ----------
def test_plan_turn_high_injects_kb_overview() -> None:
"""HIGH branch: the stored outline lands in the prompt between
``<relevance>`` and ``<tuning>`` (or the ``<documents>`` body with no
notes), and ``kb_chars`` records the outline's length."""
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.90)], _settings(threshold=0.30), kb_overview=KB_OVERVIEW
)
assert plan.deflected is False
assert plan.kb_chars == len(KB_OVERVIEW)
prompt = plan.system_prompt
assert "<knowledge_base>" in prompt
assert KB_OVERVIEW in prompt
i_rel = prompt.index("<relevance>HIGH</relevance>")
i_kb = prompt.index("<knowledge_base>")
i_docs = prompt.index("<documents>")
assert i_rel < i_kb < i_docs
def test_plan_turn_high_kb_section_ordered_before_tuning() -> None:
"""Both sections present: ``<relevance>`` → ``<knowledge_base>`` →
``<tuning>`` → ``<documents>`` (the locked phase-31 order)."""
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_CONTENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.90)],
_settings(threshold=0.30),
notes=["be concise"],
kb_overview=KB_OVERVIEW,
)
prompt = plan.system_prompt
i_rel = prompt.index("<relevance>HIGH</relevance>")
i_kb = prompt.index("<knowledge_base>")
i_kb_close = prompt.index("</knowledge_base>")
i_tuning = prompt.index("<tuning>")
i_docs = prompt.index("<documents>")
assert i_rel < i_kb < i_kb_close < i_tuning < i_docs
assert plan.kb_chars == len(KB_OVERVIEW)
def test_plan_turn_low_injects_kb_overview() -> None:
"""LOW (deflected) branch: the outline is injected there too, ahead
of the DEFLECT_MODE body, and document content stays excluded."""
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_NEVER_SENT")
plan = chat_api.plan_turn(
[_chunk(doc, 0.10)], _settings(threshold=0.30), kb_overview=KB_OVERVIEW
)
assert plan.deflected is True
assert plan.kb_chars == len(KB_OVERVIEW)
prompt = plan.system_prompt
assert "<knowledge_base>" in prompt
assert KB_OVERVIEW in prompt
i_rel = prompt.index("<relevance>LOW</relevance>")
i_kb = prompt.index("<knowledge_base>")
i_mode = prompt.index("DEFLECT_MODE")
assert i_rel < i_kb < i_mode
assert "TALOS_DOC_NEVER_SENT" not in prompt # titles only, still
def test_plan_turn_empty_overview_keeps_prompt_and_zero_kb_chars() -> None:
"""No outline (None/empty/blank) → ``kb_chars == 0`` and a prompt
byte-identical to the no-overview build in both branches."""
high_doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_CONTENT")
low_doc = _doc("Backup Strategy", "BACKUP_DOC_CONTENT")
for chunks, kb in (
([_chunk(high_doc, 0.90)], None),
([_chunk(high_doc, 0.90)], ""),
([_chunk(high_doc, 0.90)], " \n\t "),
([_chunk(low_doc, 0.10)], None),
([_chunk(low_doc, 0.10)], ""),
):
plan = chat_api.plan_turn(chunks, _settings(threshold=0.30), kb_overview=kb)
baseline = chat_api.plan_turn(chunks, _settings(threshold=0.30))
assert plan.kb_chars == 0
assert baseline.kb_chars == 0
assert plan.system_prompt == baseline.system_prompt # byte-identical
assert "<knowledge_base>" not in plan.system_prompt
# ---------- prompt content (LOW vs HIGH) ----------
def test_low_prompt_has_titles_only_no_content() -> None:
a = _doc("Kubernetes Homelab Cluster", "ALPHA_DOC_CONTENT")
b = _doc("Backup Strategy", "BETA_DOC_CONTENT")
plan = chat_api.plan_turn([_chunk(b, 0.10), _chunk(a, 0.20)], _settings())
prompt = plan.system_prompt
assert "<relevance>LOW</relevance>" in prompt
assert "DEFLECT_MODE" in prompt
assert "HONESTY GATE" in prompt # the LOW rule is what the model follows
# Phase 71 (owner-permitted 2026-09-03): the deflection plain-text
# line — the LOW turn offers no tools, so any tool markup there is
# always wrong (prevention at the prompt; the filter + recovery is
# the backstop).
assert "Reply in plain text only — you have no tools in this mode." in prompt
assert "- Kubernetes Homelab Cluster" in prompt
assert "- Backup Strategy" in prompt
assert "ALPHA_DOC_CONTENT" not in prompt
assert "BETA_DOC_CONTENT" not in prompt
assert "<documents>" not in prompt
def test_high_path_unaffected() -> None:
a = _doc("Kubernetes Homelab Cluster", "ALPHA_DOC_CONTENT")
b = _doc("Backup Strategy", "BETA_DOC_CONTENT")
plan = chat_api.plan_turn([_chunk(a, 0.90), _chunk(b, 0.40)], _settings())
assert plan.deflected is False
assert plan.suggestions == []
assert "<relevance>HIGH</relevance>" in plan.system_prompt
assert "DEFLECT_MODE" not in plan.system_prompt
# Phase 71: the deflection plain-text line never leaks into HIGH.
assert "Reply in plain text only" not in plan.system_prompt
assert "ALPHA_DOC_CONTENT" in plan.system_prompt
assert "BETA_DOC_CONTENT" in plan.system_prompt
# Phase 118 (A3): both docs are suggested (no floor) — the durable-
# record input — with nothing left for the related tier (rank 6+).
assert [d.title for d in plan.suggested_docs] == [
"Kubernetes Homelab Cluster",
"Backup Strategy",
]
assert plan.related_docs == []
# ---------- deflected branch: byte-identical (A8 gate untouched) ----------
def test_low_prompt_byte_identical_to_pre_task_sha_pin() -> None:
"""Phase 118 (A8 gate untouched): the deflected prompt is
BYTE-IDENTICAL to pre-task on the same chunks — it is exactly
``build_deflect_prompt(weak_hit_titles(…))`` (the tiering feeds the
TurnPlan's durable-record fields, never the LOW prompt). The sha256
pin makes any future LOW-body drift loud; the suggestions/chips and
the deflected flag are unchanged."""
a = _doc("Kubernetes Homelab Cluster", "ALPHA_DOC_CONTENT")
b = _doc("Backup Strategy", "BETA_DOC_CONTENT")
chunks = [_chunk(b, 0.10), _chunk(a, 0.20)]
plan = chat_api.plan_turn(chunks, _settings())
assert plan.deflected is True
expected = build_deflect_prompt(weak_hit_titles(chunks))
assert plan.system_prompt == expected # byte-identical to the pre-task build
assert (
hashlib.sha256(plan.system_prompt.encode("utf-8")).hexdigest()
== "603395e013c97be8a13837eda533c0b7bd5da4f7a0806b0ae8ad7d6c52420913"
)
assert plan.suggestions # the "Maybe try" chips are unchanged
# Both tiers still ride the plan (the durable-record inputs, A3).
assert [d.title for d in plan.suggested_docs] == [
"Kubernetes Homelab Cluster",
"Backup Strategy",
]
assert plan.related_docs == []
# ---------- weak_hit_titles (fake retriever mapping) ----------
def test_weak_hit_titles_dedupe_and_rank_by_best_score() -> None:
a = _doc("Kubernetes Homelab Cluster", "AAA")
b = _doc("Backup Strategy", "BBB")
chunks = [_chunk(b, 0.5), _chunk(a, 0.2), _chunk(a, 0.9)]
assert weak_hit_titles(chunks) == ["Kubernetes Homelab Cluster", "Backup Strategy"]
# ---------- suggestions derivation ----------
def test_suggestions_derived_from_titles_in_order() -> None:
got = derive_suggestions(
["Kubernetes Homelab Cluster", "Backup Strategy", "Deploying a New Service"],
fallback=["should not appear"],
)
assert len(got) == 3
assert all(s.strip() for s in got)
assert "Kubernetes Homelab Cluster" in got[0]
assert "Backup Strategy" in got[1]
assert "Deploying a New Service" in got[2]
def test_suggestions_capped_at_three() -> None:
got = derive_suggestions([f"Title {i}" for i in range(6)], fallback=["F"])
assert len(got) == MAX_SUGGESTIONS == 3
def test_suggestions_top_up_from_fallback_when_titles_thin() -> None:
got = derive_suggestions(
["Backup Strategy"],
fallback=["How is my Kubernetes cluster set up?", "What's my backup strategy?"],
)
assert len(got) == 3
assert got[0] == "What's in your notes about Backup Strategy?"
assert got[1] == "How is my Kubernetes cluster set up?"
def test_suggestions_dedupes_and_ignores_blank() -> None:
got = derive_suggestions(
["Backup Strategy", "backup strategy", " "],
fallback=["What's my backup strategy?", " "],
)
# "backup strategy" is a case-insensitive dup; blank title/fallback are
# skipped — including ones that only look blank after formatting. Only
# two valid items remain, and the list never pads with junk.
assert got == [
"What's in your notes about Backup Strategy?",
"What's my backup strategy?",
]
assert all("about ?" not in s and s == s.strip() for s in got)
def test_suggestions_empty_input_yields_fallback_only() -> None:
assert derive_suggestions([], fallback=[]) == []
got = derive_suggestions([], fallback=["One?", "Two?"])
assert got == ["One?", "Two?"]
# ---------- endpoint-level gate (fake retriever + fake LLM + fake session) ----------
class _CannedLLM:
"""Records the messages it is given; streams a canned answer.
Without *read_paths* it never emits tool calls, so a grounded turn
through the phase-37 agent loop ends after the single (tools-offered)
request. With *read_paths*, the request whose conversation carries
*i* tool results (i < len(read_paths)) emits ``read(read_paths[i])``
(phase 119: the read-only done-sources pins drive multi-read turns),
and once the list is exhausted the request streams the answer.
*seen_tools* records each request's ``tools`` value for the phase-37
wiring pins.
"""
def __init__(self, answer: str = ANSWER, read_paths: Sequence[str] = ()) -> None:
self.settings = Settings(_env_file=None) # pyright: ignore[reportCallIssue]
self.embed_batches = 0
self.answer = answer
self.read_paths = list(read_paths)
self.seen: list[list[dict[str, str]]] = []
self.seen_tools: list[list[dict[str, Any]] | None] = []
async def embed_one(self, _text: str) -> list[float]:
return [0.0] * 768
async def chat_stream(
self,
messages: list[dict[str, str]],
tools: list[dict[str, Any]] | None = None,
scaffolding: ScaffoldingFilter | None = None, # phase 71 pass-through
):
self.seen.append(messages)
self.seen_tools.append(tools)
if tools is not None and self.read_paths:
tool_results = sum(1 for m in messages if m.get("role") == "tool")
if tool_results < len(self.read_paths):
yield ToolCallPiece(
id=f"call_{tool_results + 1}",
name="read",
arguments={"path": self.read_paths[tool_results]},
)
return
for i in range(0, len(self.answer), 12):
yield StreamPiece("content", self.answer[i : i + 12])
class _FakeSteeringResult:
"""Empty steering-note result (no stored notes in these unit tests)."""
def all(self) -> list[Any]:
return []
class _FakeSession:
"""Stands in for the DB session: records the QueryLog row it is given.
``scalars`` always yields no steering notes (phase 15) so the chat
turn's ``load_steering_notes`` call stays a no-op here, and ``get``
returns the single ``kb_overview`` row when one is configured
(phase 31) — ``None`` by default, i.e. no stored outline.
"""
def __init__(self, kb_overview: str = "") -> None:
self.added: list[Any] = []
self.commits = 0
self.kb_overview = kb_overview
def __enter__(self) -> _FakeSession:
return self
def __exit__(self, *args: Any) -> None:
pass
def add(self, obj: Any) -> None:
self.added.append(obj)
def commit(self) -> None:
self.commits += 1
def scalars(self, _stmt: Any) -> _FakeSteeringResult:
return _FakeSteeringResult()
def get(self, model: Any, pk: Any) -> Any:
if model is KbOverview and self.kb_overview:
return KbOverview(id=1, content=self.kb_overview)
return None
def execute(self, *args: Any, **kwargs: Any) -> list[Any]:
"""Phase 119 (D3): the endpoint's suggested-folder lines run the
``ls`` catalog fetchers (``_source_document_rows`` /
``_source_folder_summaries``) on this session — the fake catalog
is empty, so each suggested doc's line is its header alone
(``Homelab/:``), keeping the prompt builds deterministic in
these gate tests."""
return []
@pytest.fixture(autouse=True)
def _admin_signed_in(client: TestClient) -> None:
"""Phase 79 (task 03): ``POST /api/chat`` is user-gated — the
endpoint-level tests run as the signed-in ADMIN, so the shared
``client`` logs in once per test. The admin session short-circuits
``require_user`` before any DB touch, so the fake-session wiring in
``gate_env`` is untouched."""
r = client.post("/api/login", json={"password": ADMIN_PASSWORD})
assert r.status_code == 204, f"admin login failed: {r.status_code} {r.text}"
@pytest.fixture()
def gate_env(monkeypatch: pytest.MonkeyPatch) -> Iterator[tuple[_FakeSession, _CannedLLM]]:
"""``POST /api/chat`` with retriever, session, and LLM all faked.
SEC-14-04: the chat endpoint uses short-lived sessions via
``SessionLocal()`` — monkeypatch ``chat_api.SessionLocal`` instead
of overriding ``get_db``.
"""
monkeypatch.setattr(chat_api, "db_available", lambda: True)
session = _FakeSession()
llm = _CannedLLM()
monkeypatch.setattr(chat_api, "SessionLocal", lambda: session)
monkeypatch.setitem(fastapi_app.dependency_overrides, chat_api.get_llm, lambda: llm)
# These tests assert against a specific gate threshold; keep it stable
# regardless of the production default (0.62) or any .env.
monkeypatch.setattr(
chat_api,
"get_settings",
lambda: Settings(_env_file=None, relevance_threshold=0.30), # pyright: ignore[reportCallIssue]
)
yield session, llm
fastapi_app.dependency_overrides.clear()
def _ask(client: TestClient, message: str) -> list[dict[str, Any]]:
with client.stream("POST", "/api/chat", json={"message": message}) as r:
assert r.status_code == 200
frames: list[dict[str, Any]] = []
buf = ""
for part in r.iter_text():
buf += part
while "\n\n" in buf:
frame, buf = buf.split("\n\n", 1)
frame = frame.strip()
if frame.startswith("data:"):
frames.append(json.loads(frame.removeprefix("data:").strip()))
assert buf.strip() == ""
return frames
def _fake_retriever(chunks: list[RetrievedChunk]) -> Any:
def retrieve(_db: Any, _question: str, _vec: list[float]) -> list[RetrievedChunk]:
return chunks
return retrieve
def _find_docs(docs: dict[tuple[str, str], Document]) -> Any:
"""A fake ``find_document`` resolving *docs* by ``(source, path)``
(the agent module's hook — the fake sessions here have no catalog)."""
def find_document(_db: Any, source: str, path: str) -> Document | None:
return docs.get((source, path))
return find_document
def test_endpoint_just_below_threshold_deflects(
client: TestClient,
gate_env: tuple[_FakeSession, _CannedLLM],
monkeypatch: pytest.MonkeyPatch,
) -> None:
session, llm = gate_env
doc = _doc("Deploying a New Service", "DOC_CONTENT_NEVER_SENT")
monkeypatch.setattr(chat_api, "retrieve", _fake_retriever([_chunk(doc, 0.2999)]))
frames = _ask(client, "How do I bake sourdough bread?")
deltas = [f for f in frames if f["type"] == "delta"]
assert "".join(d["text"] for d in deltas) == ANSWER # the LLM was still called
done = frames[-1]
assert done["type"] == "done"
assert done["deflected"] is True
assert 2 <= len(done["suggestions"]) <= MAX_SUGGESTIONS # title chip + fallback
assert all(s.strip() for s in done["suggestions"])
assert any("Deploying a New Service" in s for s in done["suggestions"])
# Phase 112 (A8 revised, TODO L2): a deflected turn cites nothing —
# done.sources is the citation surface (the UI chips every entry as
# "the answer used this"), and the weak hits are scored docs, not
# citations.
assert done["sources"] == []
# The LLM saw the LOW prompt: DEFLECT_MODE + titles, never doc content.
(system, user) = llm.seen[0][0], llm.seen[0][1]
assert user["content"] == "How do I bake sourdough bread?"
assert "DEFLECT_MODE" in system["content"]
assert "DOC_CONTENT_NEVER_SENT" not in system["content"]
# Durable record: deflected + the weak score. The retrieval itself
# stays recorded (observability unchanged — query_log records
# retrieval, not citations; the phase-113 A3 precedent).
(row,) = session.added
assert isinstance(row, QueryLog)
assert row.deflected is True
assert row.top_score == pytest.approx(0.2999)
assert row.sources # the weak-hit doc's path, for threshold tuning
assert session.commits == 1
def test_endpoint_grounded_turn_runs_agent_loop_with_tools(
client: TestClient,
gate_env: tuple[_FakeSession, _CannedLLM],
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""Phase 37: a grounded endpoint turn runs the agent loop — the
single no-tool-call request carries ``AGENT_TOOLS`` (the default
round cap keeps the tools offered), no ``tool`` frames stream, and
the ``done`` event is the plain retrieval shape (the tool-free
answer is byte-identical)."""
_session, llm = gate_env
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_SENT")
monkeypatch.setattr(chat_api, "retrieve", _fake_retriever([_chunk(doc, 0.90)]))
frames = _ask(client, "How is my Kubernetes cluster set up?")
assert frames[-1]["type"] == "done"
assert frames[-1]["deflected"] is False
assert not any(f["type"] == "tool" for f in frames)
assert len(llm.seen) == 1
assert llm.seen_tools == [AGENT_TOOLS] # one request, tools offered
# The system prompt is the HIGH prompt with the <tools> instructions
# (phase 70: the harness-aligned ls/read/grep copy — new names in,
# old phase-37/68 names out).
(system, _user) = llm.seen[0][0], llm.seen[0][1]
assert "<relevance>HIGH</relevance>" in system["content"]
assert "<tools>" in system["content"]
for tool in ("`ls`", "`grep`", "`read`"):
assert tool in system["content"]
for old in ("list_documents", "read_document", "search_documents"):
assert old not in system["content"]
def test_endpoint_deflected_turn_never_offers_tools(
client: TestClient,
gate_env: tuple[_FakeSession, _CannedLLM],
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""Phase 37: a deflected endpoint turn keeps the direct
``chat_stream`` — the single request carries no ``tools`` key
(``seen_tools == [None]``), A8 byte-identical."""
_session, llm = gate_env
doc = _doc("Deploying a New Service", "DOC_CONTENT_NEVER_SENT")
monkeypatch.setattr(chat_api, "retrieve", _fake_retriever([_chunk(doc, 0.2999)]))
frames = _ask(client, "How do I bake sourdough bread?")
assert frames[-1]["type"] == "done"
assert frames[-1]["deflected"] is True
assert not any(f["type"] == "tool" for f in frames)
assert len(llm.seen) == 1
assert llm.seen_tools == [None]
(system, _user) = llm.seen[0][0], llm.seen[0][1]
assert "<tools>" not in system["content"] # the LOW prompt never carries it
# Phase 70: the rewritten <tools> copy stays out of the deflected path
# (the LOW prompt is byte-identical to the pre-phase text).
assert "You may extend your context with three tools" not in system["content"]
def test_endpoint_score_at_threshold_answers_zero_read_grounded_turn_chips_nothing(
client: TestClient,
gate_env: tuple[_FakeSession, _CannedLLM],
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""Phase 119 (LOCKED A1 — the phase-118 "always chips" pin
re-targeted): a grounded turn on which the agent reads NOTHING (the
summary-only fast path — the canned LLM never emits a tool call)
chips nothing: ``done.sources`` is the READ DOCS only, so the
never-read suggested doc appears NOWHERE in it (the explicit contrast
against the retired A4 union). The durable record keeps the retrieval
(118-A3 stands)."""
session, llm = gate_env
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_SENT")
monkeypatch.setattr(chat_api, "retrieve", _fake_retriever([_chunk(doc, 0.30)]))
frames = _ask(client, "How is my Kubernetes cluster set up?")
done = frames[-1]
assert done["type"] == "done"
assert done["deflected"] is False
assert done["suggestions"] == []
# A1: read docs only — nothing was read ⇒ no chips. The suggested
# doc (rank 1) appears NOWHERE in done.sources (the retired A4 union
# would have carried it).
assert done["sources"] == []
(system, _user) = llm.seen[0][0], llm.seen[0][1]
assert "<relevance>HIGH</relevance>" in system["content"]
assert "DEFLECT_MODE" not in system["content"]
assert "TALOS_DOC_SENT" in system["content"]
# The durable record still records the retrieval (118-A3 stands).
(row,) = session.added
assert isinstance(row, QueryLog)
assert row.deflected is False
assert "kubernetes-homelab-cluster.md" in row.sources
def test_endpoint_done_sources_are_the_read_docs_in_read_order(
client: TestClient,
gate_env: tuple[_FakeSession, _CannedLLM],
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""Phase 119 (LOCKED A1): the done frame's ``sources`` is exactly
the agent's READ documents — deduped by (source, path), READ order
(not the suggested tier's rank order). Suggested docs that were
never read appear NOWHERE in the frame (the contrast pin against
the retired A4 union); the durable record keeps the full retrieval
(118-A3 stands)."""
session, _llm = gate_env
a = _doc("Alpha", "ALPHA_DOC_CONTENT")
b = _doc("Beta", "BETA_DOC_CONTENT")
c = _doc("Gamma", "GAMMA_DOC_CONTENT")
chunks = [_chunk(a, 0.90), _chunk(b, 0.80), _chunk(c, 0.70)]
monkeypatch.setattr(chat_api, "retrieve", _fake_retriever(chunks))
llm = _CannedLLM(read_paths=[f"Homelab/{b.path}", f"Homelab/{a.path}"])
fastapi_app.dependency_overrides[chat_api.get_llm] = lambda: llm
monkeypatch.setattr(
rag_agent,
"find_document",
_find_docs({(a.source, a.path): a, (b.source, b.path): b}),
)
frames = _ask(client, "Which of these documents do I have?")
done = frames[-1]
assert done["deflected"] is False
# Both reads streamed as tool frames (the agent loop ran).
reads = [f for f in frames if f["type"] == "tool"]
assert [f["argument"] for f in reads] == [
f"Homelab/{b.path}",
f"Homelab/{a.path}",
]
# A1: read docs only, read order — B first (it was read first), A
# second; the never-read suggested doc C is NOWHERE in the frame, and
# the suggested tier's rank order (A, B, C) is NOT the chip order.
assert [(s["source"], s["path"]) for s in done["sources"]] == [
(b.source, b.path),
(a.source, a.path),
]
assert not any(s["path"] == c.path for s in done["sources"])
assert len(done["sources"]) == len({(s["source"], s["path"]) for s in done["sources"]})
assert done["related"] == [] # nothing beyond rank 3 for 3 docs
# The durable record keeps the full retrieval (118-A3 stands).
(row,) = session.added
assert isinstance(row, QueryLog)
for doc in (a, b, c):
assert f"Homelab/{doc.path}" in row.sources
def test_endpoint_read_related_doc_is_cited_not_related(
client: TestClient,
gate_env: tuple[_FakeSession, _CannedLLM],
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""Phase 119 (A1) × phase 118 (A3): a related-tier doc (rank 6+) that
the agent ``read``s is a CHIP (in ``done.sources``) and is EXCLUDED
from ``done.related`` — a "nearby doc" that was actually used must
not read as nearby (unchanged intent, keyed on read docs since
phase 119). The suggested docs, never read, are absent from the
frame; the other related doc stays in the tier."""
session, _llm = gate_env
docs = [_doc(f"Doc {i}", f"CONTENT_{i}") for i in range(7)]
chunks = [
_chunk(d, 0.9 - 0.1 * i, cosine=0.8 - 0.05 * i) for i, d in enumerate(docs)
]
monkeypatch.setattr(chat_api, "retrieve", _fake_retriever(chunks))
read_doc = docs[5] # rank 6 — the related tier (rank 7 = docs[6])
llm = _CannedLLM(read_paths=[f"Homelab/{read_doc.path}"])
fastapi_app.dependency_overrides[chat_api.get_llm] = lambda: llm
monkeypatch.setattr(
rag_agent,
"find_document",
_find_docs({(read_doc.source, read_doc.path): read_doc}),
)
frames = _ask(client, "What does the sixth ranked document say?")
done = frames[-1]
assert done["deflected"] is False
# A1: read docs only — the read related doc is the ONLY chip (the
# never-read suggested docs are nowhere in the frame).
assert [(s["source"], s["path"]) for s in done["sources"]] == [
(read_doc.source, read_doc.path)
]
# The read related doc is NOT "nearby"; the other related doc is.
related = [(s["source"], s["path"]) for s in done["related"]]
assert related == [(docs[6].source, docs[6].path)]
assert not any((d.source, d.path) in related for d in docs[:5])
# The durable record keeps the full retrieval (118-A3 stands).
(row,) = session.added
assert isinstance(row, QueryLog)
assert row.deflected is False
for d in docs:
assert f"Homelab/{d.path}" in row.sources
def test_endpoint_suggested_folder_lines_reach_the_high_prompt(
client: TestClient,
gate_env: tuple[_FakeSession, _CannedLLM],
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""Phase 119 (D3, LOCKED A4): the endpoint computes the
suggested-folder lines BEFORE plan_turn (the same deterministic
suggested walk, over the short-lived step session) and they ride
the grounded HIGH prompt — after ``SUGGEST_INTRO``, before the
first ``<document>`` block."""
_session, llm = gate_env
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_SENT")
monkeypatch.setattr(chat_api, "retrieve", _fake_retriever([_chunk(doc, 0.30)]))
line = "Homelab/: kubernetes-homelab-cluster.md (2 docs)"
monkeypatch.setattr(
chat_api,
"suggested_folder_lines",
lambda db, suggested: [line],
)
frames = _ask(client, "How is my Kubernetes cluster set up?")
assert frames[-1]["deflected"] is False
(system, _user) = llm.seen[0][0], llm.seen[0][1]
i_open = system["content"].index("<documents>")
i_block = system["content"].index("<document ")
assert (
system["content"][i_open:i_block]
== f"<documents>\n{SUGGEST_INTRO}\n{line}\n\n"
)
# ---------- endpoint: KB overview row (phase 31) ----------
def test_endpoint_stored_kb_row_injected_into_system_prompt(
client: TestClient,
gate_env: tuple[_FakeSession, _CannedLLM],
monkeypatch: pytest.MonkeyPatch,
caplog: pytest.LogCaptureFixture,
) -> None:
"""A non-empty ``kb_overview`` row (read via one PK lookup) reaches
the LLM's system prompt in both modes, and the per-turn log line
records ``kb_chars=N`` (PLAN §9)."""
session, llm = gate_env
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_SENT")
monkeypatch.setattr(chat_api, "retrieve", _fake_retriever([_chunk(doc, 0.30)]))
session.kb_overview = f" {KB_OVERVIEW} " # the loader trims it
with caplog.at_level("INFO", logger="app.chat"):
_ask(client, "How is my Kubernetes cluster set up?")
(system, _user) = llm.seen[0][0], llm.seen[0][1]
assert "<knowledge_base>" in system["content"]
assert KB_OVERVIEW in system["content"]
assert system["content"].index("<relevance>HIGH</relevance>") < system["content"].index(
"<knowledge_base>"
) < system["content"].index("<documents>")
log_lines = [r.getMessage() for r in caplog.records if "question=" in r.getMessage()]
assert log_lines and f"kb_chars={len(KB_OVERVIEW)}" in log_lines[-1]
def test_endpoint_no_kb_row_prompt_unchanged(
client: TestClient,
gate_env: tuple[_FakeSession, _CannedLLM],
monkeypatch: pytest.MonkeyPatch,
caplog: pytest.LogCaptureFixture,
) -> None:
"""No ``kb_overview`` row → the section is absent (byte-identical to
the pre-phase prompt) and the log line records ``kb_chars=0``."""
session, llm = gate_env
assert session.kb_overview == "" # fixture default: no stored row
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_SENT")
monkeypatch.setattr(chat_api, "retrieve", _fake_retriever([_chunk(doc, 0.30)]))
with caplog.at_level("INFO", logger="app.chat"):
_ask(client, "How is my Kubernetes cluster set up?")
(system, _user) = llm.seen[0][0], llm.seen[0][1]
assert "<knowledge_base>" not in system["content"]
log_lines = [r.getMessage() for r in caplog.records if "question=" in r.getMessage()]
assert log_lines and "kb_chars=0" in log_lines[-1]
def test_endpoint_log_line_records_suggested_after_summary_hits(
client: TestClient,
gate_env: tuple[_FakeSession, _CannedLLM],
monkeypatch: pytest.MonkeyPatch,
caplog: pytest.LogCaptureFixture,
) -> None:
"""Phase 118 (PLAN §9 extension): the per-turn log line gains
``suggested=N`` immediately AFTER ``summary_hits=N`` — the field
order of every existing field is untouched (the phase-114
``retries=N scaffold_stripped=N`` tail stays last). The value is the
seeded suggestion tier's size (``len(plan.suggested_docs)``)."""
_session, _llm = gate_env
doc = _doc("Kubernetes Homelab Cluster", "TALOS_DOC_SENT")
chunks = [_chunk(doc, 0.30), _chunk(doc, 0.20, is_summary=True)]
monkeypatch.setattr(chat_api, "retrieve", _fake_retriever(chunks))
with caplog.at_level("INFO", logger="app.chat"):
_ask(client, "How is my Kubernetes cluster set up?")
log_lines = [r.getMessage() for r in caplog.records if "question=" in r.getMessage()]
assert log_lines
line = log_lines[-1]
# The new slot: suggested=N right after summary_hits=N (one suggested
# doc — the single fixture doc — and one summary hit on it).
assert "summary_hits=1 suggested=1" in line
# The full field order (the phase-114 tail stays last).
order = (
"question=",
"embed_ms=",
"top_score=",
"fts_hits=",
"summary_hits=",
"suggested=",
"tuning=",
"kb_chars=",
"history_msgs=",
"threshold=",
"deflected=",
"sources=",
"thinking_chars=",
"tool_calls=",
"total_ms=",
"retries=",
"scaffold_stripped=",
)
idx = -1
for field in order:
pos = line.find(field)
assert pos > idx, f"{field} out of order in the per-turn log line"
idx = pos