Phase 47 (owner permission 2026-08-27, TODO.md L10–11, roadmap R1): the
full Podman quadlet family (.container, .network, .volume, .image,
.pod, .kube, .swap, .os, .endpoint) and .j2 Jinja templates join the
allowed + default A9 import formats, chunked as plain text (owner
decision — no TOML/Jinja-aware splitter). No env configuration needed:
a default import now indexes them.
- app/config.py: _ALLOWED_IMPORT_EXTENSIONS + the default
import_extensions CSV gain the ten names (the original seven first);
the never-widen BOR_IMPORT_EXTENSIONS validator is untouched and
still rejects truly unknown extensions.
- app/rag/chunker.py: ten _FORMAT_CHUNKERS entries -> chunk_text
(HARD_MAX_CHARS 1200 honored, unknown-suffix fallback unchanged);
docstring/comments cite the A9 revision 2026-08-27.
- tests/fixtures/docs/homelab/: quadlet/compose.container (realistic
quadlet TOML, >1500 chars, [Unit]/[Service]/[Container] sections,
RESE-QUADLET-SENTINEL-77aa), quadlet/lan.network,
quadlet/cache.volume, templates/deploy.j2 (for/set/if Jinja
constructs + RESE-JINJA-SENTINEL-33dd). Every suite that seeds the
fixture tree updates its 9 -> 13 document-count constants.
- tests/unit/test_config.py: allowed set carries all seventeen formats,
default CSV + dotted import_extension_set include the ten, the
validator accepts the new names and still rejects unknowns.
- tests/unit/test_chunker.py: dispatch parity with chunk_text for every
new suffix (parametrized), the .container fixture chunks >=2 under
the cap with the sentinel surviving, the .j2 fixture keeps {{ }}
verbatim, the unknown-suffix fallback is unchanged.
- tests/unit/test_importer.py: a default-extensions walk over a temp
tree indexes exactly the ten new files (unknown/hidden/excluded
filtered), the original seven still walk, stem-title fallback holds.
- tests/integration/test_import_quadlet_jinja.py (new): import_sources
over a temp tree with .container/.volume/.j2 -> documents + chunks
rows with stem titles; delta re-import updates only the changed .j2
doc; prune drops the deleted .volume doc with cascade.
- tests/e2e/test_quadlet_jinja_import.py (new, story suite, mock-only,
isolation): GET /api/docs (admin session) lists the four new-format
docs with non-zero chunk counts and stem titles; the Sources table
renders a row + .doc-link per file; the phase-26 modal shows the
.container TOML ([Container] section + sentinel) with stem title and
the container format badge; a RESE-JINJA-SENTINEL-33dd question
FTS-matches the .j2 chunk -> honest-positive (A8: LOW requires zero
FTS hits) — the bubble is not .is-deflected and a source chip names
templates/deploy.j2.
- README.md + .env.example: the extended default format set (A9
revised 2026-08-27, plain-text chunking, narrow-only rule intact).
- .agent/PLAN.md: the A9 revision (owner-locked R1) — A9 row status,
the revision note under the anchors table, and the §5 chunking-policy
+ §11 workflow lines. The only PLAN edit this phase.
Gates: uv run pytest 795 passed; app/ coverage TOTAL 99% (>90%);
ruff check + pyright clean; story E2E 4/4 in isolation (DB up);
regression E2E suites test_import_documents (3) / test_sync_button
(3) / test_git_sources_admin (6) green in isolation.
Also records the 47_quadlet_jinja_import task-file moves (01–03)
todo/ -> complete/.
220 lines
9.9 KiB
Python
220 lines
9.9 KiB
Python
"""Application settings.
|
||
|
||
Every setting can be overridden with an environment variable prefixed
|
||
``BOR_`` (or a local gitignored ``.env`` file — see ``.env.example``).
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import os
|
||
from functools import lru_cache
|
||
|
||
from pydantic import field_validator
|
||
from pydantic_settings import BaseSettings, SettingsConfigDict
|
||
|
||
#: The A9 import formats (PLAN anchor A9, revised 2026-08-21; revised
|
||
#: 2026-08-27, owner permission — the full Podman quadlet family
|
||
#: ``container, network, volume, image, pod, kube, swap, os, endpoint``
|
||
#: plus Jinja templates ``j2`` join the allowed set, chunked as plain
|
||
#: text). ``BOR_IMPORT_EXTENSIONS`` may narrow — but never widen — this
|
||
#: set.
|
||
_ALLOWED_IMPORT_EXTENSIONS: frozenset[str] = frozenset(
|
||
{
|
||
"md", "markdown", "txt", "yaml", "yml", "json", "py",
|
||
# A9 revised 2026-08-27 (owner permission): quadlet family + jinja.
|
||
"container", "network", "volume", "image", "pod",
|
||
"kube", "swap", "os", "endpoint", "j2",
|
||
}
|
||
)
|
||
|
||
|
||
class Settings(BaseSettings):
|
||
model_config = SettingsConfigDict(
|
||
env_file=".env",
|
||
env_file_encoding="utf-8",
|
||
env_prefix="BOR_",
|
||
extra="ignore",
|
||
)
|
||
|
||
# --- App ---
|
||
app_name: str = "Brain of Reese"
|
||
app_version: str = "0.1.0"
|
||
environment: str = "development"
|
||
log_level: str = "INFO"
|
||
static_dir: str = "frontend"
|
||
|
||
# --- Database (PostgreSQL 17 + pgvector) ---
|
||
database_url: str = "postgresql+psycopg://reese:reese@localhost:5432/brain_of_reese"
|
||
|
||
# --- LLM (self-hosted, OpenAI-compatible "aipi" endpoint) ---
|
||
llm_base_url: str = "https://aipi.reeseapps.com/v1"
|
||
llm_api_key: str = ""
|
||
llm_chat_model: str = "turbo"
|
||
llm_embed_model: str = "embed"
|
||
#: One-shot (non-streaming) completion model (A5 extended, phase 30):
|
||
#: document summaries at import time and the KB overview (phase 31).
|
||
#: Served by the same OpenAI-compatible endpoint — no new model
|
||
#: management. Called via ``LLMClient.chat()``.
|
||
llm_summary_model: str = "lite"
|
||
#: Operator kill-switch for the ``thinking`` SSE events (phase 17,
|
||
#: ``BOR_STREAM_THINKING``; ``0``/``false`` → off). When off, thinking
|
||
#: pieces are still counted for the per-turn log line but never
|
||
#: emitted — the answer stream itself is unchanged.
|
||
stream_thinking: bool = True
|
||
|
||
# --- RAG tuning ---
|
||
embedding_dim: int = 768 # verified against aipi /v1 (embed model)
|
||
top_n_docs: int = 2
|
||
# Honesty gate (A8, re-tuned 2026-08-21): the ``embed`` model's cosine
|
||
# scores compress into 0.41–0.84 on the real corpus, so the old 0.30
|
||
# default never discriminated. LOW only fires when best cosine < this
|
||
# AND no candidate chunk matches the question lexically (see A8).
|
||
relevance_threshold: float = 0.62
|
||
#: Maximum output tokens a chat answer may use (owner instruction
|
||
#: 2026-08-22: answers must run to their natural end — the old hard
|
||
#: 700-token cap cut long answers off mid-sentence).
|
||
max_output_tokens: int = 32_768
|
||
chunk_target_chars: int = 2_000
|
||
chunk_overlap_chars: int = 200
|
||
embed_batch_size: int = 16
|
||
#: Total char budget for the ``<tuning>`` section of the system prompt
|
||
#: (phase 15, steering notes). The newest-fitting notes are kept and the
|
||
#: overflow is replaced by the ``[…truncated…]`` marker.
|
||
steering_max_chars: int = 8_000
|
||
#: Cap on the document content sent to the ``lite`` summary model in one
|
||
#: call (phase 30, ``BOR_SUMMARY_MAX_CHARS``). Overflow is cut at the cap
|
||
#: and the shared ``[…truncated…]`` marker is appended (see
|
||
#: ``app.rag.summarizer``).
|
||
summary_max_chars: int = 12_000
|
||
#: Char budget for the ``<knowledge_base>`` section of the system prompt
|
||
#: (phase 31: lite-generated KB overview, ``app.rag.overview``). The
|
||
#: newest-fitting prefix of the stored outline is kept and the overflow
|
||
#: is replaced by the shared ``[…truncated…]`` marker (phase 15
|
||
#: convention — ``app.rag.prompts``).
|
||
kb_overview_max_chars: int = 4_000
|
||
#: Cap on the document list (source/path/title/first summary line per
|
||
#: row) sent to the ``lite`` overview model in one call (phase 31,
|
||
#: ``app.rag.overview``). Overflow is cut at the cap and the shared
|
||
#: ``[…truncated…]`` marker is appended (summarizer convention).
|
||
overview_input_max_chars: int = 40_000
|
||
#: Hard cap on the agent tool rounds per grounded turn (phase 45,
|
||
#: revising phase 37's per-tool budgets — owner permission
|
||
#: 2026-08-27, TODO L8: "allow the LLM to make as many tool calls
|
||
#: as it wants"). Every tool call the model emits consumes a
|
||
#: round; at the cap the loop forces one final no-tools answer.
|
||
#: ``0`` disables the tools entirely — the turn is a single
|
||
#: request with ``tools=None`` (the pre-phase-37 path — the kill
|
||
#: switch). Negative values are rejected at startup (validator).
|
||
agent_max_rounds: int = 10
|
||
|
||
# --- Hybrid retrieval (A7, revised 2026-08-21) ---
|
||
# cosine top-N ∪ Postgres FTS top-N, fused with Reciprocal Rank Fusion
|
||
# (score = Σ 1/(rrf_k + rank) over the lists a chunk appears in).
|
||
#
|
||
# The vector window is deliberately wider than the lexical one: a
|
||
# name-your-tool question's best *lexical* chunk (e.g. the "Install"
|
||
# section of gitlab.md) can sit far down the vector ranking because the
|
||
# question embeds close to generic templates. A 100-wide window is what
|
||
# lets such chunks double-hit (one RRF term per list) and outrank a
|
||
# template that owns vector rank 1 — measured 2026-08-22 against the
|
||
# live 2774-chunk KB for "How did I install gitlab?" (gitlab.md:1 at
|
||
# vrank 100 / lrank 3 → fused 0.0221 vs the template's 0.0164).
|
||
hybrid_vector_candidates: int = 100
|
||
hybrid_lexical_candidates: int = 30
|
||
rrf_k: int = 60
|
||
|
||
# --- Admin & sign-in (phase 16; A10 revised 2026-08-22) ---
|
||
# Single-admin auth via a signed session cookie (Starlette
|
||
# SessionMiddleware — no new services, no DB tables). Both secrets are
|
||
# REQUIRED at startup: ``create_app()`` refuses to boot when either is
|
||
# empty (``app.core.auth.ensure_admin_configured``). The password is
|
||
# plaintext on purpose (homelab scope, owner decision 2026-08-22);
|
||
# the session secret signs the cookie (``secrets.token_hex(32)``).
|
||
admin_password: str = ""
|
||
session_secret: str = ""
|
||
#: Signed-cookie lifetime in seconds (default 12 h, refreshed on
|
||
#: session writes — sliding for an active admin).
|
||
session_max_age: int = 43_200
|
||
session_cookie: str = "bor_session"
|
||
|
||
# --- Import scope (A9, revised 2026-08-21 and 2026-08-27) ---
|
||
# Comma-separated list of lowercased file extensions (no dot) imported
|
||
# by ``scripts/import_docs.py``. Hidden (dot) path components are always
|
||
# skipped, plus the importer's exclusion list.
|
||
# Stored as a raw CSV string (env-native — no JSON) and parsed on demand
|
||
# via :py:meth:`import_extension_set`. ``mode="after"`` validation runs
|
||
# against the raw string so a typo fails loudly at startup.
|
||
import_extensions: str = (
|
||
"md,markdown,txt,yaml,yml,json,py,"
|
||
"container,network,volume,image,pod,kube,swap,os,endpoint,j2"
|
||
)
|
||
#: List of git repo URLs to clone/pull into ``sources_dir`` before
|
||
#: indexing (phase 28); comma-separated, stored raw. Empty means no git
|
||
#: sources — ``import_docs`` then falls back to ``--source`` / the old
|
||
#: ``DEFAULT_SOURCES``.
|
||
git_sources: str = ""
|
||
#: Where ``import_docs`` clones/pulls the ``git_sources`` repos (phase
|
||
#: 28). Stored as a raw string — ``Path.expanduser()`` is applied in
|
||
#: the import script, not here.
|
||
sources_dir: str = "~/bor-sources"
|
||
|
||
@field_validator("import_extensions")
|
||
@classmethod
|
||
def _import_extensions_known(cls, v: str) -> str:
|
||
"""Reject unknown/empty formats loudly instead of silently importing
|
||
nothing (a typo like ``md,jsonn`` would otherwise walk zero files)."""
|
||
exts = {part.strip().lstrip(".").lower() for part in v.split(",") if part.strip()}
|
||
if not exts:
|
||
raise ValueError("import_extensions must name at least one format")
|
||
unknown = exts - _ALLOWED_IMPORT_EXTENSIONS
|
||
if unknown:
|
||
raise ValueError(
|
||
f"unknown import extension(s): {', '.join(sorted(unknown))} — "
|
||
f"allowed: {', '.join(sorted(_ALLOWED_IMPORT_EXTENSIONS))}"
|
||
)
|
||
return v
|
||
|
||
@field_validator("agent_max_rounds")
|
||
@classmethod
|
||
def _agent_max_rounds_non_negative(cls, v: int) -> int:
|
||
"""``0`` is the no-tools kill switch — a negative value is a typo."""
|
||
if v < 0:
|
||
raise ValueError("agent_max_rounds must be >= 0 (0 = no tools)")
|
||
return v
|
||
|
||
# Suggested questions (onboarding + empty state).
|
||
suggestions: list[str] = [
|
||
"How is my Kubernetes cluster set up?",
|
||
"What's my backup strategy?",
|
||
"How do I deploy a new service?",
|
||
"What's currently running in the homelab?",
|
||
]
|
||
|
||
@property
|
||
def import_extension_set(self) -> frozenset[str]:
|
||
"""Lowercased, dotted extension set (``.md``) for path filtering."""
|
||
return frozenset(
|
||
f".{part.strip().lstrip('.').lower()}"
|
||
for part in self.import_extensions.split(",")
|
||
if part.strip()
|
||
)
|
||
|
||
@property
|
||
def git_source_list(self) -> list[str]:
|
||
"""Non-empty, stripped git URLs from :py:attr:`git_sources` (phase 28).
|
||
|
||
Whitespace around each entry is trimmed and empty entries dropped;
|
||
an unset/empty value yields ``[]`` (the import script then uses its
|
||
legacy local-directory defaults).
|
||
"""
|
||
return [part.strip() for part in self.git_sources.split(",") if part.strip()]
|
||
|
||
@property
|
||
def effective_api_key(self) -> str:
|
||
"""API key for aipi: explicit setting, then $AIPI_KEY, then a placeholder."""
|
||
return self.llm_api_key or os.environ.get("AIPI_KEY", "") or "not-needed"
|
||
|
||
|
||
@lru_cache
|
||
def get_settings() -> Settings:
|
||
return Settings()
|