Files
brain-of-reese/tests/unit/test_config.py
T
ducoterra 5d679f5184 feat(import): index quadlet unit files and jinja templates (A9 revision)
Phase 47 (owner permission 2026-08-27, TODO.md L10–11, roadmap R1): the
full Podman quadlet family (.container, .network, .volume, .image,
.pod, .kube, .swap, .os, .endpoint) and .j2 Jinja templates join the
allowed + default A9 import formats, chunked as plain text (owner
decision — no TOML/Jinja-aware splitter). No env configuration needed:
a default import now indexes them.

- app/config.py: _ALLOWED_IMPORT_EXTENSIONS + the default
  import_extensions CSV gain the ten names (the original seven first);
  the never-widen BOR_IMPORT_EXTENSIONS validator is untouched and
  still rejects truly unknown extensions.
- app/rag/chunker.py: ten _FORMAT_CHUNKERS entries -> chunk_text
  (HARD_MAX_CHARS 1200 honored, unknown-suffix fallback unchanged);
  docstring/comments cite the A9 revision 2026-08-27.
- tests/fixtures/docs/homelab/: quadlet/compose.container (realistic
  quadlet TOML, >1500 chars, [Unit]/[Service]/[Container] sections,
  RESE-QUADLET-SENTINEL-77aa), quadlet/lan.network,
  quadlet/cache.volume, templates/deploy.j2 (for/set/if Jinja
  constructs + RESE-JINJA-SENTINEL-33dd). Every suite that seeds the
  fixture tree updates its 9 -> 13 document-count constants.
- tests/unit/test_config.py: allowed set carries all seventeen formats,
  default CSV + dotted import_extension_set include the ten, the
  validator accepts the new names and still rejects unknowns.
- tests/unit/test_chunker.py: dispatch parity with chunk_text for every
  new suffix (parametrized), the .container fixture chunks >=2 under
  the cap with the sentinel surviving, the .j2 fixture keeps {{ }}
  verbatim, the unknown-suffix fallback is unchanged.
- tests/unit/test_importer.py: a default-extensions walk over a temp
  tree indexes exactly the ten new files (unknown/hidden/excluded
  filtered), the original seven still walk, stem-title fallback holds.
- tests/integration/test_import_quadlet_jinja.py (new): import_sources
  over a temp tree with .container/.volume/.j2 -> documents + chunks
  rows with stem titles; delta re-import updates only the changed .j2
  doc; prune drops the deleted .volume doc with cascade.
- tests/e2e/test_quadlet_jinja_import.py (new, story suite, mock-only,
  isolation): GET /api/docs (admin session) lists the four new-format
  docs with non-zero chunk counts and stem titles; the Sources table
  renders a row + .doc-link per file; the phase-26 modal shows the
  .container TOML ([Container] section + sentinel) with stem title and
  the container format badge; a RESE-JINJA-SENTINEL-33dd question
  FTS-matches the .j2 chunk -> honest-positive (A8: LOW requires zero
  FTS hits) — the bubble is not .is-deflected and a source chip names
  templates/deploy.j2.
- README.md + .env.example: the extended default format set (A9
  revised 2026-08-27, plain-text chunking, narrow-only rule intact).
- .agent/PLAN.md: the A9 revision (owner-locked R1) — A9 row status,
  the revision note under the anchors table, and the §5 chunking-policy
  + §11 workflow lines. The only PLAN edit this phase.

Gates: uv run pytest 795 passed; app/ coverage TOTAL 99% (>90%);
ruff check + pyright clean; story E2E 4/4 in isolation (DB up);
regression E2E suites test_import_documents (3) / test_sync_button
(3) / test_git_sources_admin (6) green in isolation.

Also records the 47_quadlet_jinja_import task-file moves (01–03)
todo/ -> complete/.
2026-08-28 07:02:24 -04:00

265 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Unit tests: settings defaults & env overrides."""
from __future__ import annotations
import json
from typing import Any
import pytest
from pydantic import ValidationError
from pydantic_settings import SettingsError
from app.config import _ALLOWED_IMPORT_EXTENSIONS, Settings # pyright: ignore[reportPrivateUsage]
def _settings(**kwargs: Any) -> Settings:
"""Build Settings without reading a .env file (deterministic tests)."""
kwargs.setdefault("_env_file", None)
return Settings(**kwargs) # pyright: ignore[reportCallIssue] (kwarg exists at runtime)
def test_defaults_match_locked_decisions(monkeypatch: pytest.MonkeyPatch) -> None:
# The test process sets BOR_RELEVANCE_THRESHOLD=0.30 for the mock-
# calibrated in-process suites (see tests/conftest.py) — the *default*
# under test is the production one.
monkeypatch.delenv("BOR_RELEVANCE_THRESHOLD", raising=False)
s = _settings()
assert s.llm_chat_model == "turbo"
assert s.llm_embed_model == "embed"
# A5 extended (phase 30): one-shot completions default to the ``lite``
# model on the same endpoint.
assert s.llm_summary_model == "lite"
assert s.embedding_dim == 768
assert s.llm_base_url.endswith("/v1")
# A8 (revised): the honesty gate input is the best cosine, default 0.62.
assert s.relevance_threshold == 0.62
# A7 (revised): hybrid retrieval — cosine top-N ∪ FTS top-N, RRF-fused.
assert s.hybrid_vector_candidates >= 1
assert s.hybrid_lexical_candidates >= 1
assert s.rrf_k >= 1
assert s.top_n_docs >= 1
# Owner instruction 2026-08-22: answers may run up to 32 768 tokens.
assert s.max_output_tokens == 32_768
# Phase 17: the model's thinking streams by default (kill-switch off).
assert s.stream_thinking is True
assert len(s.suggestions) >= 3
# A9 (revised 2026-08-27): the import scope covers all seventeen
# A9 formats (original seven + quadlet family + jinja).
assert s.import_extension_set == {
".md", ".markdown", ".txt", ".yaml", ".yml", ".json", ".py",
".container", ".network", ".volume", ".image", ".pod",
".kube", ".swap", ".os", ".endpoint", ".j2",
}
NEW_A9_FORMATS = (
"container", "network", "volume", "image", "pod",
"kube", "swap", "os", "endpoint", "j2",
)
def test_allowed_import_extensions_contains_all_seventeen_formats() -> None:
"""The validator's base set is the full A9 set: the original seven
plus the ten added 2026-08-27 (quadlet family + ``j2``). The
never-widen contract bounds :py:data:`import_extensions` against
exactly this set."""
assert {
"md", "markdown", "txt", "yaml", "yml", "json", "py",
*NEW_A9_FORMATS,
} == _ALLOWED_IMPORT_EXTENSIONS
def test_default_import_extensions_include_the_ten_new_formats() -> None:
"""A9 revised 2026-08-27 (owner permission): the quadlet family +
``j2`` are imported by default — no env configuration needed — with
the original seven first (order is cosmetic, the set is what
matters)."""
s = _settings()
for ext in NEW_A9_FORMATS:
assert ext in s.import_extensions
assert s.import_extension_set == (
{".md", ".markdown", ".txt", ".yaml", ".yml", ".json", ".py"}
| {f".{ext}" for ext in NEW_A9_FORMATS}
)
def test_env_override(monkeypatch) -> None:
monkeypatch.setenv("BOR_RELEVANCE_THRESHOLD", "0.42")
monkeypatch.setenv("BOR_LLM_CHAT_MODEL", "juggernaut")
s = _settings()
assert s.relevance_threshold == 0.42
assert s.llm_chat_model == "juggernaut"
def test_llm_summary_model_env_override(monkeypatch) -> None:
"""Phase 30: ``BOR_LLM_SUMMARY_MODEL`` overrides the ``lite`` default"""
monkeypatch.setenv("BOR_LLM_SUMMARY_MODEL", "mini")
s = _settings()
assert s.llm_summary_model == "mini"
def test_summary_max_chars_default_and_env_override(monkeypatch: pytest.MonkeyPatch) -> None:
"""Phase 30: document content sent to the ``lite`` model is capped at
``BOR_SUMMARY_MAX_CHARS`` (default 12 000 chars per call)."""
monkeypatch.delenv("BOR_SUMMARY_MAX_CHARS", raising=False)
assert _settings().summary_max_chars == 12_000
monkeypatch.setenv("BOR_SUMMARY_MAX_CHARS", "5000")
assert _settings().summary_max_chars == 5000
def test_max_output_tokens_env_override(monkeypatch) -> None:
monkeypatch.setenv("BOR_MAX_OUTPUT_TOKENS", "1234")
s = _settings()
assert s.max_output_tokens == 1234
def test_agent_max_rounds_default_and_env_override(monkeypatch: pytest.MonkeyPatch) -> None:
"""Phase 45: the per-tool budgets are gone — ``BOR_AGENT_MAX_ROUNDS``
(default 10) is the single agent-loop knob; ``0`` is the no-tools
kill switch."""
monkeypatch.delenv("BOR_AGENT_MAX_ROUNDS", raising=False)
assert _settings().agent_max_rounds == 10
monkeypatch.setenv("BOR_AGENT_MAX_ROUNDS", "5")
assert _settings().agent_max_rounds == 5
monkeypatch.setenv("BOR_AGENT_MAX_ROUNDS", "0")
assert _settings().agent_max_rounds == 0
def test_agent_max_rounds_rejects_negative(monkeypatch: pytest.MonkeyPatch) -> None:
"""``0`` is the kill switch — a negative value is a typo, so the
validator fails loudly at startup."""
monkeypatch.setenv("BOR_AGENT_MAX_ROUNDS", "-1")
with pytest.raises(ValidationError, match="agent_max_rounds"):
_settings()
def test_stream_thinking_default_true_and_env_parse(monkeypatch: pytest.MonkeyPatch) -> None:
"""Phase 17 kill-switch (``BOR_STREAM_THINKING``): on by default,
``0``/``false`` turn the ``thinking`` SSE frames off."""
assert _settings().stream_thinking is True
assert _settings(stream_thinking=False).stream_thinking is False
monkeypatch.setenv("BOR_STREAM_THINKING", "0")
assert _settings().stream_thinking is False
monkeypatch.setenv("BOR_STREAM_THINKING", "false")
assert _settings().stream_thinking is False
monkeypatch.setenv("BOR_STREAM_THINKING", "1")
assert _settings().stream_thinking is True
def test_import_extensions_env_override_is_a_csv_list(monkeypatch) -> None:
monkeypatch.setenv("BOR_IMPORT_EXTENSIONS", "md,yml")
s = _settings()
assert s.import_extension_set == {".md", ".yml"}
def test_import_extensions_rejects_unknown_format(monkeypatch) -> None:
"""A typo in the CSV fails at startup (loudly), not by silently
walking zero files."""
monkeypatch.setenv("BOR_IMPORT_EXTENSIONS", "md,docx")
with pytest.raises(ValidationError, match="docx"):
_settings()
def test_import_extensions_rejects_empty(monkeypatch) -> None:
monkeypatch.setenv("BOR_IMPORT_EXTENSIONS", " ")
with pytest.raises(ValidationError):
_settings()
def test_import_extensions_validator_accepts_new_a9_formats(monkeypatch) -> None:
"""A9 revised 2026-08-27: the new names are first-class — the
never-widen contract now holds against the widened base set, so a
narrowing CSV with quadlet/jinja names is accepted."""
monkeypatch.setenv("BOR_IMPORT_EXTENSIONS", "md,container,j2")
s = _settings()
assert s.import_extension_set == {".md", ".container", ".j2"}
def test_import_extensions_validator_still_rejects_unknown(monkeypatch) -> None:
"""Truly unknown extensions still fail loudly at startup (the
validator is intact — only the allowed base set widened)."""
monkeypatch.setenv("BOR_IMPORT_EXTENSIONS", "md,xyz")
with pytest.raises(ValidationError, match="xyz"):
_settings()
def test_git_sources_default_empty_and_sources_dir_default() -> None:
"""Phase 28: no git sources by default (backwards-compatible with the
``--source`` / ``DEFAULT_SOURCES`` fallback); the clone location stays
a raw string (``~`` is expanded by the import script, not the setting)."""
s = _settings()
assert s.git_sources == ""
assert s.git_source_list == []
assert s.sources_dir == "~/bor-sources"
def test_git_sources_env_override_parses_comma_separated_list(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""``BOR_GIT_SOURCES`` is a raw CSV: entries are trimmed and empty
entries dropped; URLs are stored untouched (no scheme parsing here)."""
monkeypatch.setenv(
"BOR_GIT_SOURCES",
"https://github.com/user/homelab.git, "
" git@github.com:user/deployments.git ,, https://git.reeseapps.com/x/y.git ",
)
s = _settings()
# The raw CSV string is preserved untouched (no parsing in the setting).
assert s.git_sources == (
"https://github.com/user/homelab.git, "
" git@github.com:user/deployments.git ,, https://git.reeseapps.com/x/y.git "
)
assert s.git_source_list == [
"https://github.com/user/homelab.git",
"git@github.com:user/deployments.git",
"https://git.reeseapps.com/x/y.git",
]
def test_git_sources_whitespace_only_yields_empty_list(monkeypatch) -> None:
"""A configured-but-blank value behaves the same as unset: no git
sources, so the script falls back to its legacy local defaults."""
monkeypatch.setenv("BOR_GIT_SOURCES", " , , ")
s = _settings()
assert s.git_source_list == []
def test_sources_dir_env_override_is_raw_string(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setenv("BOR_SOURCES_DIR", "/data/bor/sources")
s = _settings()
assert s.sources_dir == "/data/bor/sources"
def test_suggestions_default_is_three_plus_real_questions() -> None:
s = _settings()
assert len(s.suggestions) >= 3
assert all(isinstance(q, str) and q.strip() for q in s.suggestions)
# Distinct chips only — duplicates in the onboarding row are noise.
assert len({q.strip().lower() for q in s.suggestions}) == len(s.suggestions)
def test_suggestions_env_override_is_json_list(monkeypatch) -> None:
override = [
"How do I back up with Borg?",
"How is my K3S cluster set up?",
"How do I deploy a service?",
]
monkeypatch.setenv("BOR_SUGGESTIONS", json.dumps(override))
s = _settings()
assert s.suggestions == override
def test_suggestions_malformed_json_fails_loudly(monkeypatch) -> None:
monkeypatch.setenv("BOR_SUGGESTIONS", "[not valid json")
with pytest.raises(SettingsError):
_settings()
def test_effective_api_key_fallback(monkeypatch) -> None:
monkeypatch.delenv("AIPI_KEY", raising=False)
s = _settings()
assert s.effective_api_key == "not-needed"
monkeypatch.setenv("AIPI_KEY", "sk-from-env")
s2 = _settings()
assert s2.effective_api_key == "sk-from-env"