feat(rag): hybrid FTS+vector retrieval and multi-format ingestion — name-your-tool questions find the right document
This commit is contained in:
@@ -5,6 +5,7 @@ import json
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
from pydantic import ValidationError
|
||||
from pydantic_settings import SettingsError
|
||||
|
||||
from app.config import Settings
|
||||
@@ -16,16 +17,28 @@ def _settings(**kwargs: Any) -> Settings:
|
||||
return Settings(**kwargs) # pyright: ignore[reportCallIssue] (kwarg exists at runtime)
|
||||
|
||||
|
||||
def test_defaults_match_locked_decisions() -> None:
|
||||
def test_defaults_match_locked_decisions(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
# The test process sets BOR_RELEVANCE_THRESHOLD=0.30 for the mock-
|
||||
# calibrated in-process suites (see tests/conftest.py) — the *default*
|
||||
# under test is the production one.
|
||||
monkeypatch.delenv("BOR_RELEVANCE_THRESHOLD", raising=False)
|
||||
s = _settings()
|
||||
assert s.llm_chat_model == "turbo"
|
||||
assert s.llm_embed_model == "embed"
|
||||
assert s.embedding_dim == 768
|
||||
assert s.llm_base_url.endswith("/v1")
|
||||
assert 0 < s.relevance_threshold < 1
|
||||
assert s.top_k_chunks >= 1
|
||||
# A8 (revised): the honesty gate input is the best cosine, default 0.62.
|
||||
assert s.relevance_threshold == 0.62
|
||||
# A7 (revised): hybrid retrieval — cosine top-N ∪ FTS top-N, RRF-fused.
|
||||
assert s.hybrid_vector_candidates >= 1
|
||||
assert s.hybrid_lexical_candidates >= 1
|
||||
assert s.rrf_k >= 1
|
||||
assert s.top_n_docs >= 1
|
||||
assert len(s.suggestions) >= 3
|
||||
# A9 (revised): the import scope covers the seven A9 formats.
|
||||
assert s.import_extension_set == {
|
||||
".md", ".markdown", ".txt", ".yaml", ".yml", ".json", ".py"
|
||||
}
|
||||
|
||||
|
||||
def test_env_override(monkeypatch) -> None:
|
||||
@@ -36,6 +49,26 @@ def test_env_override(monkeypatch) -> None:
|
||||
assert s.llm_chat_model == "juggernaut"
|
||||
|
||||
|
||||
def test_import_extensions_env_override_is_a_csv_list(monkeypatch) -> None:
|
||||
monkeypatch.setenv("BOR_IMPORT_EXTENSIONS", "md,yml")
|
||||
s = _settings()
|
||||
assert s.import_extension_set == {".md", ".yml"}
|
||||
|
||||
|
||||
def test_import_extensions_rejects_unknown_format(monkeypatch) -> None:
|
||||
"""A typo in the CSV fails at startup (loudly), not by silently
|
||||
walking zero files."""
|
||||
monkeypatch.setenv("BOR_IMPORT_EXTENSIONS", "md,docx")
|
||||
with pytest.raises(ValidationError, match="docx"):
|
||||
_settings()
|
||||
|
||||
|
||||
def test_import_extensions_rejects_empty(monkeypatch) -> None:
|
||||
monkeypatch.setenv("BOR_IMPORT_EXTENSIONS", " ")
|
||||
with pytest.raises(ValidationError):
|
||||
_settings()
|
||||
|
||||
|
||||
def test_suggestions_default_is_three_plus_real_questions() -> None:
|
||||
s = _settings()
|
||||
assert len(s.suggestions) >= 3
|
||||
|
||||
Reference in New Issue
Block a user