Files
brain-of-reese/app/config.py
T

147 lines
6.1 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Application settings.
Every setting can be overridden with an environment variable prefixed
``BOR_`` (or a local gitignored ``.env`` file — see ``.env.example``).
"""
from __future__ import annotations
import os
from functools import lru_cache
from pydantic import field_validator
from pydantic_settings import BaseSettings, SettingsConfigDict
#: The A9 import formats (PLAN anchor A9, revised 2026-08-21).
#: ``BOR_IMPORT_EXTENSIONS`` may narrow — but never widen — this set.
_ALLOWED_IMPORT_EXTENSIONS: frozenset[str] = frozenset(
{"md", "markdown", "txt", "yaml", "yml", "json", "py"}
)
class Settings(BaseSettings):
model_config = SettingsConfigDict(
env_file=".env",
env_file_encoding="utf-8",
env_prefix="BOR_",
extra="ignore",
)
# --- App ---
app_name: str = "Brain of Reese"
app_version: str = "0.1.0"
environment: str = "development"
log_level: str = "INFO"
static_dir: str = "frontend"
# --- Database (PostgreSQL 17 + pgvector) ---
database_url: str = "postgresql+psycopg://reese:reese@localhost:5432/brain_of_reese"
# --- LLM (self-hosted, OpenAI-compatible "aipi" endpoint) ---
llm_base_url: str = "https://aipi.reeseapps.com/v1"
llm_api_key: str = ""
llm_chat_model: str = "turbo"
llm_embed_model: str = "embed"
# --- RAG tuning ---
embedding_dim: int = 768 # verified against aipi /v1 (embed model)
top_n_docs: int = 2
# Honesty gate (A8, re-tuned 2026-08-21): the ``embed`` model's cosine
# scores compress into 0.41–0.84 on the real corpus, so the old 0.30
# default never discriminated. LOW only fires when best cosine < this
# AND no candidate chunk matches the question lexically (see A8).
relevance_threshold: float = 0.62
max_context_chars: int = 24_000
#: Maximum output tokens a chat answer may use (owner instruction
#: 2026-08-22: answers must run to their natural end — the old hard
#: 700-token cap cut long answers off mid-sentence).
max_output_tokens: int = 32_768
chunk_target_chars: int = 2_000
chunk_overlap_chars: int = 200
embed_batch_size: int = 16
#: Total char budget for the ``<tuning>`` section of the system prompt
#: (phase 15, steering notes). The newest-fitting notes are kept and the
#: overflow is replaced by the ``[…truncated…]`` marker.
steering_max_chars: int = 8_000
# --- Hybrid retrieval (A7, revised 2026-08-21) ---
# cosine top-N ∪ Postgres FTS top-N, fused with Reciprocal Rank Fusion
# (score = Σ 1/(rrf_k + rank) over the lists a chunk appears in).
#
# The vector window is deliberately wider than the lexical one: a
# name-your-tool question's best *lexical* chunk (e.g. the "Install"
# section of gitlab.md) can sit far down the vector ranking because the
# question embeds close to generic templates. A 100-wide window is what
# lets such chunks double-hit (one RRF term per list) and outrank a
# template that owns vector rank 1 — measured 2026-08-22 against the
# live 2774-chunk KB for "How did I install gitlab?" (gitlab.md:1 at
# vrank 100 / lrank 3 → fused 0.0221 vs the template's 0.0164).
hybrid_vector_candidates: int = 100
hybrid_lexical_candidates: int = 30
rrf_k: int = 60
# --- Admin & sign-in (phase 16; A10 revised 2026-08-22) ---
# Single-admin auth via a signed session cookie (Starlette
# SessionMiddleware — no new services, no DB tables). Both secrets are
# REQUIRED at startup: ``create_app()`` refuses to boot when either is
# empty (``app.core.auth.ensure_admin_configured``). The password is
# plaintext on purpose (homelab scope, owner decision 2026-08-22);
# the session secret signs the cookie (``secrets.token_hex(32)``).
admin_password: str = ""
session_secret: str = ""
#: Signed-cookie lifetime in seconds (default 12 h, refreshed on
#: session writes — sliding for an active admin).
session_max_age: int = 43_200
session_cookie: str = "bor_session"
# --- Import scope (A9, revised 2026-08-21) ---
# Comma-separated list of lowercased file extensions (no dot) imported
# by ``scripts/import_docs.py``. Hidden (dot) path components are always
# skipped, plus the importer's exclusion list.
# Stored as a raw CSV string (env-native — no JSON) and parsed on demand
# via :py:meth:`import_extension_set`. ``mode="after"`` validation runs
# against the raw string so a typo fails loudly at startup.
import_extensions: str = "md,markdown,txt,yaml,yml,json,py"
@field_validator("import_extensions")
@classmethod
def _import_extensions_known(cls, v: str) -> str:
"""Reject unknown/empty formats loudly instead of silently importing
nothing (a typo like ``md,jsonn`` would otherwise walk zero files)."""
exts = {part.strip().lstrip(".").lower() for part in v.split(",") if part.strip()}
if not exts:
raise ValueError("import_extensions must name at least one format")
unknown = exts - _ALLOWED_IMPORT_EXTENSIONS
if unknown:
raise ValueError(
f"unknown import extension(s): {', '.join(sorted(unknown))} — "
f"allowed: {', '.join(sorted(_ALLOWED_IMPORT_EXTENSIONS))}"
)
return v
# Suggested questions (onboarding + empty state).
suggestions: list[str] = [
"How is my Kubernetes cluster set up?",
"What's my backup strategy?",
"How do I deploy a new service?",
"What's currently running in the homelab?",
]
@property
def import_extension_set(self) -> frozenset[str]:
"""Lowercased, dotted extension set (``.md``) for path filtering."""
return frozenset(
f".{part.strip().lstrip('.').lower()}"
for part in self.import_extensions.split(",")
if part.strip()
)
@property
def effective_api_key(self) -> str:
"""API key for aipi: explicit setting, then $AIPI_KEY, then a placeholder."""
return self.llm_api_key or os.environ.get("AIPI_KEY", "") or "not-needed"
@lru_cache
def get_settings() -> Settings:
return Settings()