"""Unit: retriever — ordering and dedup, no context cap (A7 revised) (fake rows). The SQL side of :func:`app.rag.retriever.retrieve` is exercised by the chat integration tests against real Postgres; the pure mapping logic in :func:`select_documents` is tested here with in-memory rows. """ from __future__ import annotations import uuid from types import SimpleNamespace import pytest from app.config import Settings from app.models import Document from app.rag import retriever from app.rag.retriever import ( TRUNCATION_MARKER, RetrievedChunk, select_documents, select_documents_tiered, select_related, select_suggested, ) def _doc(path: str, content: str, source: str = "Homelab", title: str | None = None) -> Document: return Document( id=uuid.uuid4(), source=source, path=path, full_path=f"/tmp/{path}", title=title or path, content=content, content_hash="0" * 64, ) def _chunk(doc: Document, score: float, position: int = 0) -> RetrievedChunk: return RetrievedChunk( chunk_id=uuid.uuid4(), position=position, content=doc.content[:40], score=score, document=doc, ) def test_ranks_by_best_chunk_score_not_first_hit() -> None: """A doc whose *later* chunk scores highest must still rank first.""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) c = _doc("c.md", "C" * 50) chunks = [ _chunk(a, 0.4, position=0), # a's weak chunk comes first _chunk(b, 0.8), _chunk(a, 0.9, position=2), # a's best chunk comes last _chunk(c, 0.5), ] docs = select_documents(chunks, n=3) assert [d.path for d in docs] == ["a.md", "b.md", "c.md"] def test_dedups_to_one_document_per_hit_set() -> None: a = _doc("a.md", "A" * 50) chunks = [_chunk(a, 0.2), _chunk(a, 0.7), _chunk(a, 0.5)] docs = select_documents(chunks, n=2) assert len(docs) == 1 assert docs[0] is a def test_caps_at_n_documents() -> None: docs_in = [_doc(f"d{i}.md", "X" * 20) for i in range(4)] chunks = [_chunk(d, 0.5 - 0.1 * i) for i, d in enumerate(docs_in)] out = select_documents(chunks, n=2) assert [d.path for d in out] == ["d0.md", "d1.md"] def test_content_never_truncated_even_past_old_budget() -> None: """Whole documents, never truncated (A7 revised, owner permission 2026-08-24). Two documents of 20 000 + 15 000 chars — 35 000 combined, well past the old 24 000 context budget — come back with content **byte-identical** to the originals, and the truncation marker is absent from both. """ big = _doc("big.md", "B" * 20_000) small = _doc("small.md", "S" * 15_000) chunks = [_chunk(big, 0.9), _chunk(small, 0.6)] out = select_documents(chunks, n=2) assert [d.path for d in out] == ["big.md", "small.md"] assert out[0].content == "B" * 20_000 assert out[1].content == "S" * 15_000 assert TRUNCATION_MARKER not in out[0].content assert TRUNCATION_MARKER not in out[1].content def test_empty_hits_yield_no_documents() -> None: assert select_documents([], n=2) == [] # --------------------------------------------------------------------------- # Phase 113 — the usefulness bar: cited vs related tiers # --------------------------------------------------------------------------- def _cos_chunk(doc: Document, score: float, cosine: float, position: int = 0) -> RetrievedChunk: """A candidate with *score* (fused rank key) and *cosine* (gate input) decoupled.""" return RetrievedChunk( chunk_id=uuid.uuid4(), position=position, content=doc.content[:40], score=score, document=doc, cosine=cosine, ) def test_tiered_both_clear_floor_both_cited() -> None: """Both docs clear the bar → both cited, nothing related.""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) chunks = [_cos_chunk(a, 0.9, 0.50), _cos_chunk(b, 0.8, 0.40)] cited, related = select_documents_tiered(chunks, n=2, floor=0.35, related_cap=2) assert [d.path for d in cited] == ["a.md", "b.md"] assert related == [] def test_tiered_strong_plus_weak_one_cited_one_related() -> None: """The recurring incident shape: a strong 1st doc and a weak 2nd — the weak doc loses its citation slot and lands in the related tier.""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) chunks = [_cos_chunk(a, 0.9, 0.50), _cos_chunk(b, 0.8, 0.10)] cited, related = select_documents_tiered(chunks, n=2, floor=0.35, related_cap=2) assert [d.path for d in cited] == ["a.md"] assert [d.path for d in related] == ["b.md"] def test_tiered_both_weak_zero_cited_all_related() -> None: """Neither doc clears the bar → no citation slot at all (the bar filters, it never backfills), the weak hits become the related tier.""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) chunks = [_cos_chunk(a, 0.9, 0.20), _cos_chunk(b, 0.8, 0.15)] cited, related = select_documents_tiered(chunks, n=2, floor=0.35, related_cap=2) assert cited == [] assert [d.path for d in related] == ["a.md", "b.md"] # rank order kept def test_tiered_related_cap_respected() -> None: """The related tier is capped (related_max_docs) in rank order.""" docs_in = [_doc(f"d{i}.md", "X" * 20) for i in range(3)] chunks = [_cos_chunk(d, 0.9 - 0.1 * i, 0.10 - 0.02 * i) for i, d in enumerate(docs_in)] cited, related = select_documents_tiered(chunks, n=2, floor=0.35, related_cap=2) assert cited == [] assert [d.path for d in related] == ["d0.md", "d1.md"] def test_tiered_n_is_ceiling_not_quota() -> None: """A single strong doc yields ONE cited doc — the bar never pads the cited tier up to ``n`` (LOCKED A2). And docs that clear the bar but exceed the ceiling fall through to the related tier (the next docs in rank order, never overlapping cited).""" only = _doc("only.md", "O" * 50) cited, related = select_documents_tiered( [_cos_chunk(only, 0.9, 0.80)], n=2, floor=0.35, related_cap=2 ) assert [d.path for d in cited] == ["only.md"] assert related == [] docs_in = [_doc(f"d{i}.md", "X" * 20) for i in range(4)] chunks = [_cos_chunk(d, 0.9 - 0.1 * i, 0.8 - 0.05 * i) for i, d in enumerate(docs_in)] cited, related = select_documents_tiered(chunks, n=2, floor=0.35, related_cap=2) assert [d.path for d in cited] == ["d0.md", "d1.md"] # the ceiling assert [d.path for d in related] == ["d2.md", "d3.md"] # the next in rank order def test_tiered_bar_skips_weak_rank_one() -> None: """A weak rank-1 doc does not consume a citation slot: the next-ranked bar-clearing docs take it (the bar filters, it does not backfill).""" w = _doc("w.md", "W" * 50) s1 = _doc("s1.md", "1" * 50) s2 = _doc("s2.md", "2" * 50) chunks = [ _cos_chunk(w, 0.9, 0.20), # rank 1 — below the bar _cos_chunk(s1, 0.8, 0.90), _cos_chunk(s2, 0.7, 0.80), ] cited, related = select_documents_tiered(chunks, n=2, floor=0.5, related_cap=2) assert [d.path for d in cited] == ["s1.md", "s2.md"] assert [d.path for d in related] == ["w.md"] def test_tiered_tracks_best_chunk_cosine_across_a_docs_chunks() -> None: """The bar is on the doc's BEST hit-chunk cosine — a weak first chunk (rank 1) does not sink a doc whose later chunk is vector-strong.""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) chunks = [ _cos_chunk(a, 0.9, 0.10, position=0), # a's weak chunk ranks first _cos_chunk(a, 0.5, 0.90, position=2), # a's strong chunk _cos_chunk(b, 0.4, 0.0), # lexical-only b ] cited, related = select_documents_tiered(chunks, n=2, floor=0.35, related_cap=2) assert [d.path for d in cited] == ["a.md"] assert [d.path for d in related] == ["b.md"] def test_tiered_lexical_only_hit_goes_to_related_above_zero_floor() -> None: """A lexical-only doc (cosine 0.0 by construction) is vector-unsupported by definition: above a zero floor it never earns a cited slot (LOCKED A2).""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) lexical_only = RetrievedChunk( chunk_id=uuid.uuid4(), position=0, content="X" * 10, score=0.9, # top fused rank (the FTS hit) document=a, cosine=0.0, # no vector rank — lexical-only fts_hit=True, ) chunks = [lexical_only, _cos_chunk(b, 0.8, 0.5)] cited, related = select_documents_tiered(chunks, n=2, floor=0.35, related_cap=2) assert [d.path for d in cited] == ["b.md"] assert [d.path for d in related] == ["a.md"] def test_tiered_floor_zero_is_no_bar() -> None: """A zero floor admits every scored document — lexical-only (cosine 0.0) and weak alike — so the bar can be disabled per deployment.""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) chunks = [_cos_chunk(a, 0.9, 0.0), _cos_chunk(b, 0.8, 0.1)] cited, related = select_documents_tiered(chunks, n=2, floor=0.0, related_cap=2) assert [d.path for d in cited] == ["a.md", "b.md"] assert related == [] def test_tiered_empty_chunks_yield_empty_tiers() -> None: assert select_documents_tiered([], n=2, floor=0.35, related_cap=2) == ([], []) def test_select_documents_wrapper_is_legacy_tiering() -> None: """The wrapper (floor 0.0, cap 0) is the legacy "any score, top-N" selection — byte-identical for the shapes the existing callers see: rank order, dedupe, multi-chunk docs, ties, lexical-only hits.""" docs_in = [_doc(f"d{i}.md", "X" * 20) for i in range(5)] chunks = [ _chunk(docs_in[0], 0.5), # cosine 0.0 (lexical-only) _cos_chunk(docs_in[0], 0.9, 0.4, position=1), _chunk(docs_in[1], 0.8), _cos_chunk(docs_in[1], 0.8, 0.2, position=1), # fused tie across docs _cos_chunk(docs_in[2], 0.7, 0.0), _cos_chunk(docs_in[3], 0.6, 0.1), _chunk(docs_in[4], 0.1), ] for n in (1, 2, 3, 10): assert select_documents(chunks, n=n) == select_documents_tiered( chunks, n, 0.0, 0 )[0] # --------------------------------------------------------------------------- # Phase 118 — select_suggested: the top-N "start here" tier, NO floor (A3) # --------------------------------------------------------------------------- def test_suggested_rank_order_by_first_seen_chunk() -> None: """The SAME stable walk as ``select_documents_tiered``: a document's rank is fixed by its FIRST seen chunk in score-descending order — a doc whose best chunk appears later in the input list still ranks where that chunk falls.""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) c = _doc("c.md", "C" * 50) chunks = [ _chunk(a, 0.4, position=0), # a's weak chunk comes first _chunk(b, 0.8), _chunk(a, 0.9, position=2), # a's best chunk comes last _chunk(c, 0.5), ] out = select_suggested(chunks, n=5) assert [d.path for d in out] == ["a.md", "b.md", "c.md"] # The rows carry the full content byte-identical (A6: the content is # what ``read`` serves later — never truncated). assert out[0].content == "A" * 50 assert TRUNCATION_MARKER not in out[0].content def test_suggested_dedupes_multiple_chunks_to_one_row() -> None: """Multiple hit chunks of one document collapse to a single row.""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) chunks = [ _chunk(a, 0.2), _chunk(b, 0.7), _chunk(a, 0.9, position=2), _chunk(a, 0.5), ] out = select_suggested(chunks, n=5) assert [d.path for d in out] == ["a.md", "b.md"] # one row per document assert out[0] is a def test_suggested_caps_at_n_in_rank_order() -> None: docs_in = [_doc(f"d{i}.md", "X" * 20) for i in range(7)] chunks = [_chunk(d, 0.9 - 0.1 * i) for i, d in enumerate(docs_in)] out = select_suggested(chunks, n=3) assert [d.path for d in out] == ["d0.md", "d1.md", "d2.md"] def test_suggested_default_cap_is_the_settings_value( monkeypatch: pytest.MonkeyPatch, ) -> None: """``n`` omitted → ``BOR_SUGGESTED_DOCS`` caps the walk — default 5 (LOCKED A3, the top-5 "start here" directive, TODO L3) — and the cap is the setting's LIVE value, not a frozen constant.""" docs_in = [_doc(f"d{i}.md", "X" * 20) for i in range(7)] chunks = [_chunk(d, 0.9 - 0.1 * i) for i, d in enumerate(docs_in)] settings = Settings(_env_file=None) # pyright: ignore[reportCallIssue] assert settings.suggested_docs == 5 # the production default monkeypatch.setattr(retriever, "get_settings", lambda: settings) assert [d.path for d in select_suggested(chunks)] == [ f"d{i}.md" for i in range(5) ] small = Settings(_env_file=None, suggested_docs=2) # pyright: ignore[reportCallIssue] monkeypatch.setattr(retriever, "get_settings", lambda: small) assert [d.path for d in select_suggested(chunks)] == ["d0.md", "d1.md"] def test_suggested_never_filters_on_cosine_floor() -> None: """NO floor (LOCKED A3): a lexical-only hit (cosine 0.0 by construction) is a suggestion when it ranks — the contrast pin against ``select_documents_tiered``'s floored cited tier on the SAME input, which demotes it to the related tier.""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) lexical_only = RetrievedChunk( chunk_id=uuid.uuid4(), position=0, content="X" * 10, score=0.9, # top fused rank (the FTS hit) document=a, cosine=0.0, # no vector rank — lexical-only fts_hit=True, ) chunks = [lexical_only, _cos_chunk(b, 0.8, 0.5)] # Suggested: the floor never filters — a leads, b follows in rank order. assert [d.path for d in select_suggested(chunks, n=5)] == ["a.md", "b.md"] # Contrast: the same input through the phase-113 cited tier — the # 0.35 usefulness bar demotes the lexical-only doc to related. cited, related = select_documents_tiered(chunks, n=2, floor=0.35, related_cap=2) assert [d.path for d in cited] == ["b.md"] assert [d.path for d in related] == ["a.md"] def test_suggested_tie_break_inherited_from_fused_order() -> None: """Equal fused scores keep the input (fused) order — the stable score-only walk inherits ``fuse()``'s (-score, -cosine, path, position) tie-break; the selector never re-sorts it away.""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) chunks = [_cos_chunk(a, 0.7, 0.5), _cos_chunk(b, 0.7, 0.4)] # a wins on cosine assert [d.path for d in select_suggested(chunks, n=5)] == ["a.md", "b.md"] # A FULL tie (score AND cosine): the input position — ``fuse()``'s # path/position tie-break already applied — decides. ``m.md`` sorts # AFTER ``b2.md`` alphabetically, so any re-sort by path would flip # the order; the fused input order must win. m = _doc("m.md", "M" * 50) b2 = _doc("b2.md", "B" * 50) chunks = [_cos_chunk(m, 0.7, 0.4), _cos_chunk(b2, 0.7, 0.4)] assert [d.path for d in select_suggested(chunks, n=5)] == ["m.md", "b2.md"] def test_suggested_empty_chunks_yield_no_documents() -> None: assert select_suggested([], n=5) == [] # --------------------------------------------------------------------------- # Phase 118, task 05 — select_related: the rank-6+ tier after the suggested set # --------------------------------------------------------------------------- def test_related_walk_order_after_excluded_set() -> None: """The SAME stable score-descending walk as ``select_suggested``: a document's rank is fixed by its FIRST seen chunk; documents in *excluded_ids* (the suggested set) are skipped and the rest come back in rank order.""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) c = _doc("c.md", "C" * 50) d = _doc("d.md", "D" * 50) chunks = [ _chunk(a, 0.4, position=0), # a's weak chunk comes first _chunk(b, 0.8), _chunk(a, 0.9, position=2), # a's best chunk last — a ranks first _chunk(c, 0.5), _chunk(d, 0.3), ] out = select_related(chunks, {a.id, b.id}, cap=2) assert [x.path for x in out] == ["c.md", "d.md"] def test_related_skips_excluded_documents() -> None: """Every document in *excluded_ids* is skipped, even when it would rank inside the cap — the suggested set never rides the related row.""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) c = _doc("c.md", "C" * 50) chunks = [_chunk(a, 0.9), _chunk(b, 0.8), _chunk(c, 0.5)] out = select_related(chunks, {a.id, b.id}, cap=5) assert out == [c] # Every doc excluded → empty, even with room left in the cap. assert select_related(chunks, {a.id, b.id, c.id}, cap=5) == [] def test_related_caps_at_cap_in_rank_order() -> None: """The phase-118 turn wiring on 9 docs: suggested = the top 5, related = rank 6–7 (capped at 2), disjoint from the suggested set.""" docs_in = [_doc(f"d{i}.md", "X" * 20) for i in range(9)] chunks = [_chunk(d, 0.9 - 0.1 * i) for i, d in enumerate(docs_in)] suggested = select_suggested(chunks, n=5) out = select_related(chunks, {d.id for d in suggested}, cap=2) assert [x.path for x in out] == ["d5.md", "d6.md"] # rank 6–7, capped suggested_paths = {d.path for d in suggested} assert suggested_paths.isdisjoint({x.path for x in out}) def test_related_cap_zero_yields_empty() -> None: """cap=0 is the kill switch (related_max_docs=0): no related docs, the pre-phase-113 visibility.""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) c = _doc("c.md", "C" * 50) chunks = [_chunk(a, 0.9), _chunk(b, 0.8), _chunk(c, 0.5)] assert select_related(chunks, {a.id}, cap=0) == [] def test_related_never_filters_on_cosine_floor() -> None: """NO floor: a lexical-only (cosine 0.0) doc is related when it ranks after the excluded set — the related tier is visibility, not citation (phase 118 applies no cosine floor to it).""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) lexical_only = RetrievedChunk( chunk_id=uuid.uuid4(), position=0, content="X" * 10, score=0.9, # top fused rank (the FTS hit) document=a, cosine=0.0, # no vector rank — lexical-only fts_hit=True, ) chunks = [lexical_only, _cos_chunk(b, 0.8, 0.5)] out = select_related(chunks, set(), cap=5) assert [x.path for x in out] == ["a.md", "b.md"] def test_related_dedupes_multiple_chunks_to_one_row() -> None: """Multiple hit chunks of one document collapse to a single row (first-seen-chunk rank, dedupe by document.id — the shared walk).""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) chunks = [_chunk(a, 0.2), _chunk(b, 0.7), _chunk(a, 0.9, position=2)] out = select_related(chunks, set(), cap=5) assert [x.path for x in out] == ["a.md", "b.md"] # one row per document assert out[0] is a def test_related_tie_break_inherited_from_fused_order() -> None: """Equal fused scores keep the input (fused) order — the stable score-only walk inherits ``fuse()``'s (-score, -cosine, path, position) tie-break; the selector never re-sorts it away.""" a = _doc("a.md", "A" * 50) b = _doc("b.md", "B" * 50) chunks = [_cos_chunk(a, 0.7, 0.5), _cos_chunk(b, 0.7, 0.4)] # a wins on cosine assert [x.path for x in select_related(chunks, set(), cap=5)] == ["a.md", "b.md"] m = _doc("m.md", "M" * 50) b2 = _doc("b2.md", "B" * 50) chunks = [_cos_chunk(m, 0.7, 0.4), _cos_chunk(b2, 0.7, 0.4)] # full tie assert [x.path for x in select_related(chunks, set(), cap=5)] == ["m.md", "b2.md"] def test_related_empty_chunks_yield_no_documents() -> None: assert select_related([], set(), cap=5) == [] # --------------------------------------------------------------------------- # Hybrid retrieval (A7): RRF fusion + lexical tsquery # --------------------------------------------------------------------------- from app.rag.retriever import fuse, lexical_tsquery # noqa: E402 def _rc( doc_path: str, cosine: float = 0.0, fts_hit: bool = False, position: int = 0, is_summary: bool = False, ) -> RetrievedChunk: return RetrievedChunk( chunk_id=uuid.uuid4(), position=position, content="x" * 20, score=0.0, document=_doc(doc_path, "x" * 20), cosine=cosine, fts_hit=fts_hit, is_summary=is_summary, ) def test_lexical_tsquery_tokens_lowercased_deduped_in_order() -> None: assert lexical_tsquery("How did I Install GITLAB gitlab?") == "how | did | i | install | gitlab" def test_lexical_tsquery_punctuation_and_umlauts_ignored() -> None: assert lexical_tsquery("c3-r00t? -- what's up!") == "c3 | r00t | what | s | up" def test_lexical_tsquery_pure_symbols_return_none() -> None: assert lexical_tsquery("??? ???") is None assert lexical_tsquery("") is None def test_lexical_tsquery_stopwords_left_to_postgres() -> None: # lexical_tsquery passes raw tokens through; Postgres's to_tsquery # lexing drops the stopwords (verified against real PG in # test_retrieve_empty_kb / integration tests). assert lexical_tsquery("how do i") == "how | do | i" def test_lexical_tsquery_dotted_tokens_kept_whole() -> None: """The 2026-09-05 incident: the default parser lexes dotted words as ONE lexeme ("llama.cpp" → 'llama.cpp', "Qwen 3.8" → '3.8'), so the query carries them whole — split tokens (llama | cpp) can never match the document side.""" assert lexical_tsquery( "What are the correct llama.cpp arguments for Qwen 3.8?" ) == "what | are | the | correct | llama.cpp | arguments | for | qwen | 3.8" # The dash still splits (only dots group): ai | internal.network. assert lexical_tsquery("how did I set up ai-internal.network?") == ( "how | did | i | set | up | ai | internal.network" ) def test_fuse_combines_both_lists_for_double_hits() -> None: v1 = _rc("a.md", cosine=0.9) v2 = _rc("b.md", cosine=0.5) l1 = _rc("a.md", cosine=0.1) # same chunk id -> matched in place a_id = v1.chunk_id l1.chunk_id = a_id out = fuse([v1, v2], [l1], k=60) by_id = {rc.chunk_id: rc for rc in out} # a: 1/61 (vector rank 1) + 1/61 (lexical rank 1); b: 1/62 only. assert by_id[a_id].score == pytest.approx(2 / 61) assert by_id[a_id].fts_hit is True assert by_id[v2.chunk_id].score == pytest.approx(1 / 62) assert by_id[v2.chunk_id].fts_hit is False assert [rc.chunk_id for rc in out] == [a_id, v2.chunk_id] def test_fuse_lexical_only_chunks_enter_with_zero_cosine() -> None: vector = [_rc("a.md", cosine=0.8)] lexical = [_rc("b.md", cosine=0.0, fts_hit=True)] out = fuse(vector, lexical, k=60) assert len(out) == 2 b = next(rc for rc in out if rc.document.path == "b.md") assert b.cosine == 0.0 assert b.fts_hit is True # Still ranked by its (only) RRF term. assert b.score == pytest.approx(1 / 61) def test_fuse_orders_by_score_then_cosine_then_path() -> None: # Two chunks share an RRF score (both rank 1 in different lists): # the higher-cosine one must sort first. hi = _rc("z.md", cosine=0.9) lo = _rc("a.md", cosine=0.2) out = fuse([hi], [lo], k=60) assert [rc.document.path for rc in out] == ["z.md", "a.md"] # Equal score AND cosine -> path order. p1 = _rc("b.md", cosine=0.5) p2 = _rc("a.md", cosine=0.5) out = fuse([p1], [p2], k=60) assert [rc.document.path for rc in out] == ["a.md", "b.md"] # Equal score, cosine, path -> position order. s1 = _rc("a.md", cosine=0.5, position=1) s2 = _rc("a.md", cosine=0.5, position=0) out = fuse([s1], [s2], k=60) assert [rc.position for rc in out] == [0, 1] def test_fuse_rejects_nonpositive_k() -> None: with pytest.raises(ValueError): fuse([], [], k=0) with pytest.raises(ValueError): fuse([], [], k=-1) def test_fuse_empty_lists() -> None: assert fuse([], [], k=60) == [] # --------------------------------------------------------------------------- # Phase 30: is_summary survives both candidate lists and the fusion # --------------------------------------------------------------------------- from app.models import Chunk # noqa: E402 from app.rag.retriever import ( # noqa: E402 NAME_HIT_LIMIT, _lexical_candidates, _name_hit_chunks, _normalize_name, _vector_candidates, name_hit_tokens, ) class _FakeResult: """Stands in for SQLAlchemy's RowMapping result (``.all()`` only).""" def __init__(self, rows: list) -> None: self._rows = rows def all(self) -> list: return self._rows class _FakeSession: """Returns canned rows from ``execute`` without touching Postgres. One list of rows (legacy form) is returned for EVERY call; several lists (one per successive ``execute``) model a query sequence — the name-hit lexical path (2026-09-05) issues the document-projection query and, when hits exist, the LATERAL chunk query, BEFORE the FTS query. """ def __init__(self, *rowsets: list) -> None: if len(rowsets) == 1 and not ( rowsets[0] and isinstance(rowsets[0][0], list) ): rowsets = (rowsets[0],) # the single-rowset legacy form self._rowsets = rowsets self._call = 0 self.statements: list = [] def execute(self, stmt, params: dict | None = None) -> _FakeResult: self.statements.append((stmt, params)) rows = self._rowsets[min(self._call, len(self._rowsets) - 1)] self._call += 1 return _FakeResult(rows) def _chunk_row(is_summary: bool) -> Chunk: doc = _doc("summary-src.yaml", "RAW_YAML_CONTENT") return Chunk( id=uuid.uuid4(), document_id=doc.id, position=-1, # the summary chunk's position (phase 30) content="Summary text", is_summary=is_summary, ) def test_vector_candidates_carry_is_summary_flag() -> None: """The vector list copies ``Chunk.is_summary`` onto each candidate.""" doc = _doc("summary-src.yaml", "RAW_YAML_CONTENT") summary = _chunk_row(is_summary=True) ordinary = _chunk_row(is_summary=False) ordinary.position = 0 ordinary.content = "ordinary content" rows = [ (summary, 0.123456, doc), (ordinary, 0.2, doc), ] out = _vector_candidates(_FakeSession(rows), [0.0] * 768, limit=5) # pyright: ignore[reportArgumentType] assert len(out) == 2 by_pos = {rc.position: rc for rc in out} assert by_pos[-1].is_summary is True # the summary chunk (position −1) assert by_pos[0].is_summary is False # ordinary content chunk assert by_pos[-1].cosine == pytest.approx(0.876544) # 1 − distance, still rounded def test_vector_candidates_default_is_summary_false_for_legacy_chunks() -> None: """Pre-phase-30 rows have ``is_summary=false`` — candidates stay False.""" doc = _doc("legacy.md", "LEGACY") legacy = Chunk( id=uuid.uuid4(), document_id=doc.id, position=0, content="legacy content", is_summary=False, ) out = _vector_candidates(_FakeSession([(legacy, 0.5, doc)]), [0.0] * 768, limit=5) # pyright: ignore[reportArgumentType] assert out[0].is_summary is False def _lexical_row(is_summary: bool, doc_path: str) -> object: """One row of ``_LEXICAL_SQL`` (attribute access, as SQLAlchemy returns).""" doc = _doc(doc_path, "DOC_BODY") return SimpleNamespace( chunk_id=uuid.uuid4(), position=-1 if is_summary else 0, content="summary chunk text" if is_summary else "content chunk text", doc_id=doc.id, source=doc.source, path=doc.path, full_path=doc.full_path, title=doc.title, doc_content=doc.content, content_hash=doc.content_hash, indexed_at=None, created_at=None, is_summary=is_summary, rank=0.33, ) def test_lexical_candidates_carry_is_summary_flag() -> None: """The lexical list reads ``c.is_summary`` from the raw row. The question carries no digit-bearing name token (no bare, no numeric-join), so the name-hit path issues NO queries at all — the single FTS rowset answers the only (FTS) call, and the list is the plain FTS rows: the pre-name-hit behavior, unchanged. """ rows = [_lexical_row(True, "summary-src.yaml"), _lexical_row(False, "other.md")] out = _lexical_candidates( _FakeSession(rows), "how do i configure the thing", limit=10 # pyright: ignore[reportArgumentType] ) assert len(out) == 2 by_path = {rc.document.path: rc for rc in out} assert by_path["summary-src.yaml"].is_summary is True assert by_path["summary-src.yaml"].position == -1 assert by_path["other.md"].is_summary is False assert all(rc.fts_hit is True for rc in out) def test_fuse_keeps_is_summary_on_double_hit() -> None: """A summary chunk in both lists keeps the flag after fusion.""" v1 = _rc("s.yaml", cosine=0.9, is_summary=True) l1 = _rc("s.yaml", cosine=0.9, is_summary=True) # lexical copy of the same chunk l1.chunk_id = v1.chunk_id out = fuse([v1], [l1], k=60) assert len(out) == 1 assert out[0].is_summary is True assert out[0].fts_hit is True assert out[0].score == pytest.approx(2 / 61) def test_fuse_keeps_is_summary_on_lexical_only_hit() -> None: """A summary-only lexical hit (no vector rank) keeps the flag.""" out = fuse([], [_rc("s.yaml", is_summary=True)], k=60) assert len(out) == 1 assert out[0].is_summary is True assert out[0].fts_hit is True assert out[0].cosine == 0.0 def test_fuse_default_is_summary_stays_false_for_legacy_chunks() -> None: """Neither list flagged ⇒ fusion never invents a summary flag.""" out = fuse([_rc("a.md", cosine=0.8)], [_rc("b.md")], k=60) assert len(out) == 2 assert all(rc.is_summary is False for rc in out) # --------------------------------------------------------------------------- # Name-hit lexical signal (the 2026-09-05 incident — the versioned-name # case the default parser lexes incompatibly: "Qwen 3.8" → qwen/3/8 can # never match a document's qwen3/8/27b tokens) # --------------------------------------------------------------------------- INCIDENT_QUESTION = "What are the correct llama.cpp arguments for Qwen 3.8?" def test_normalize_name() -> None: assert _normalize_name("Qwen 3.8") == "qwen38" assert _normalize_name("qwen3.8-27b-juggernaut-vulkan") == "qwen3827bjuggernautvulkan" assert _normalize_name("Mixed CASE-99") == "mixedcase99" assert _normalize_name("!!!") == "" def test_name_hit_tokens_incident_question() -> None: """The incident question yields EXACTLY the versioned join ``qwen38`` — the token the document names actually carry. Plain prose words (``what``, ``llamacpp``, ``arguments``, ``server`` — no digit) never name-match (the precision guard); the single digits ("3", "8") and the bare "38" are < 4 chars; the digit-leading ``38show`` boundary artifact is dropped.""" tokens = name_hit_tokens(INCIDENT_QUESTION) assert tokens == ["qwen38"] for absent in ("what", "qwen", "llamacpp", "arguments", "3", "8", "38", "38show", "server"): assert absent not in tokens def test_name_hit_tokens_no_digit_question_returns_empty() -> None: """A question with no digit-bearing token (bare or joined) yields no name candidates — prose joins like ``correctllama`` never count.""" assert name_hit_tokens("what is the correct caddy config") == [] assert name_hit_tokens("a e i o u 3 8") == [] def test_name_hit_tokens_bare_digit_bearing_token() -> None: """A single written token that carries a digit (``1panel``) is a name candidate on its own — no join needed.""" tokens = name_hit_tokens("what is my 1panel dashboard setup") assert tokens == ["1panel"] def _name_row(doc: Document) -> tuple: """One row of the name-hit document projection (catalog order).""" return (doc.id, doc.source, doc.path, doc.title) def _name_hit_lateral_row(doc: Document, is_summary: bool = False) -> SimpleNamespace: """One row of the name-hit LATERAL chunk query.""" return SimpleNamespace( doc_id=doc.id, source=doc.source, path=doc.path, full_path=doc.full_path, title=doc.title, doc_content=doc.content, content_hash=doc.content_hash, indexed_at=None, created_at=None, chunk_id=uuid.uuid4(), position=-1 if is_summary else 0, content="summary chunk" if is_summary else "content chunk", is_summary=is_summary, ) def test_name_hit_chunks_no_tokens_skips_all_queries() -> None: """A question with no name tokens issues no queries at all.""" session = _FakeSession([]) # any call would surface a statement assert _name_hit_chunks(session, "a e i o u 3 8") == [] # pyright: ignore[reportArgumentType] assert session.statements == [] def test_name_hit_chunks_no_matching_doc_returns_empty() -> None: """Name tokens exist but no document name carries one: the projection runs, the LATERAL fetch does not.""" doc = _doc("quadlets/other.container", "body") name_rows = [_name_row(doc)] session = _FakeSession(name_rows, []) assert _name_hit_chunks(session, INCIDENT_QUESTION) == [] # pyright: ignore[reportArgumentType] assert len(session.statements) == 1 # projection only — no LATERAL fetch def test_name_hit_chunks_ranked_by_count_length_catalog() -> None: """A two-candidate question (``qwen38`` + ``1panel``): the document whose name carries BOTH (2 matches, 12 total chars) leads; the two single-match documents tie on (1, 6) and fall to catalog order (``dashboards/1panel-notes.md`` before ``quadlets/qwen3.8…``). Hits carry ``fts_hit=True`` (the A8 gate answers), ``cosine=0.0``, and the summary flag of their representative chunk.""" both = _doc("dashboards/1panel-qwen3.8.md", "body", title="1Panel Qwen 3.8") panel = _doc("dashboards/1panel-notes.md", "body") q38 = _doc("quadlets/qwen3.8-27b-juggernaut-vulkan.container", "body") name_rows = [_name_row(d) for d in (panel, both, q38)] # catalog order question = "what are the correct llama.cpp arguments for qwen 3.8 and the 1panel dashboard?" lateral_rows = [ _name_hit_lateral_row(q38, is_summary=True), # LATERAL may return any order _name_hit_lateral_row(both), _name_hit_lateral_row(panel), ] session = _FakeSession(name_rows, lateral_rows) out = _name_hit_chunks(session, question) # pyright: ignore[reportArgumentType] assert [rc.document.path for rc in out] == [ "dashboards/1panel-qwen3.8.md", # 2 matched tokens — leads "dashboards/1panel-notes.md", # (1, 6) — catalog order "quadlets/qwen3.8-27b-juggernaut-vulkan.container", # (1, 6) — after ] assert all(rc.fts_hit is True for rc in out) # the lexical signal assert all(rc.cosine == 0.0 for rc in out) # no vector rank assert all(rc.score == 0.0 for rc in out) # fuse fills the score by_path = {rc.document.path: rc for rc in out} assert by_path["quadlets/qwen3.8-27b-juggernaut-vulkan.container"].is_summary is True assert by_path["quadlets/qwen3.8-27b-juggernaut-vulkan.container"].position == -1 assert by_path["dashboards/1panel-notes.md"].is_summary is False def test_name_hit_chunks_capped_at_limit() -> None: """Twelve tied name hits (one matched token each) yield exactly ``NAME_HIT_LIMIT`` of them — catalog order (the deterministic tie-break).""" docs = [_doc(f"quadlets/m{i:02d}.container", "body") for i in range(12)] for d in docs: # give every document a name that carries the token d.title = "qwen38 model i" name_rows = [_name_row(d) for d in docs] # Only the ten winners (catalog order — the deterministic tie-break # of the twelve identical scores) reach the LATERAL fetch; the fake # answers with exactly those rows. lateral_rows = [_name_hit_lateral_row(d) for d in docs[:NAME_HIT_LIMIT]] session = _FakeSession(name_rows, lateral_rows) out = _name_hit_chunks( session, "tell me about the qwen 3.8 models" # pyright: ignore[reportArgumentType] ) assert len(out) == NAME_HIT_LIMIT assert [rc.document.path for rc in out] == [f"quadlets/m{i:02d}.container" for i in range(10)] def test_lexical_candidates_name_hits_lead_and_dedupe_with_fts() -> None: """The full lexical list: name hits LEAD (their representative chunks), the FTS rows follow, and an FTS row sharing the name hit's chunk id appears exactly once (deduped).""" q38 = _doc("quadlets/qwen3.8-27b-juggernaut-vulkan.container", "body") other = _doc("quadlets/qwen38-other.container", "body") # 1 matched token name_rows = [_name_row(q38), _name_row(other)] q38_chunk = uuid.uuid4() def _lateral(doc: Document) -> SimpleNamespace: row = _name_hit_lateral_row(doc) if doc is q38: row.chunk_id = q38_chunk return row lateral_rows = [_lateral(q38), _lateral(other)] fts_rows = [ # an FTS hit on the SAME chunk as the q38 name hit (deduped away) SimpleNamespace( chunk_id=q38_chunk, position=1, content="c", doc_id=q38.id, source=q38.source, path=q38.path, full_path=q38.full_path, title=q38.title, doc_content=q38.content, content_hash=q38.content_hash, indexed_at=None, created_at=None, is_summary=False, rank=0.1, ), # an FTS hit on a different chunk of the OTHER doc (kept) _lexical_row(False, "quadlets/qwen38-other.container"), ] session = _FakeSession(name_rows, lateral_rows, fts_rows) out = _lexical_candidates(session, INCIDENT_QUESTION, limit=10) # pyright: ignore[reportArgumentType] assert len(out) == 3 # q38 (once), other (name hit), other (FTS chunk) # Both name hits tie on (1, 6) — catalog order: "qwen3." (ASCII 46) # sorts before "qwen38" (ASCII 56). assert out[0].document.path == "quadlets/qwen3.8-27b-juggernaut-vulkan.container" assert out[1].document.path == "quadlets/qwen38-other.container" assert out[0].chunk_id == q38_chunk # the name-hit representative row assert { rc.chunk_id for rc in out } == {q38_chunk, fts_rows[1].chunk_id, lateral_rows[1].chunk_id} assert all(rc.fts_hit is True for rc in out)