Grounded chat turns now run the agent loop (app/rag/agent.py) instead
of a bare chat_stream: while the per-turn budgets last
(BOR_AGENT_LIST_CALLS / BOR_AGENT_READ_CALLS, default 1 each) the model
gets list_documents (the indexed catalog, /api/docs order) and
read_document (full text, never truncated — A7-revised contract); once
both budgets are spent the tools key is dropped from the request and
the model must answer. Rejected calls (unknown tool, unknown/missing
path, document already in context, spent budget) consume no budget.
Budgets 0/0 make exactly one tools=None request — byte-identical to
the pre-phase path (budgets-as-kill-switch). Deflected turns keep the
direct chat_stream (A8 unchanged; the LOW prompt never carries the
<tools> section).
SSE contract gains {"type":"tool","name":...,"argument":
"source/path"|null} frames ahead of the answer deltas (PLAN §4
extension, owner permission 2026-08-26); done.sources, query_log.sources
and the per-turn log line (gains tool_calls=N) report the retrieval
docs + read docs, deduped. The UI shows a "calling tool"
button/label state and one visible .tool-call line per call above the
answer; the lines persist with the chat record and re-render on
reload. chat_stream passes tools through and accumulates streaming
tool_calls deltas into ToolCallPiece (tools=None stays byte-identical).
E2E: deterministic mock tool flow ("use your tools" + <tools> marker:
list -> read first catalog line -> quoted answer) plus the story suite
(marker flow, reload re-render, plain/deflected no-tool regressions).
Docs: .env.example + README (the two tools, the budgets, the SSE tool
frame, the "calling tool" UI state).
probe: turbo tool_calls=supported 2026-08-26 (uv run python -m
scripts.llm_probe --tools — non-streaming + streaming
finish_reason=tool_calls, indexed delta.tool_calls partials)
198 lines
8.9 KiB
Python
198 lines
8.9 KiB
Python
"""Application settings.
|
||
|
||
Every setting can be overridden with an environment variable prefixed
|
||
``BOR_`` (or a local gitignored ``.env`` file — see ``.env.example``).
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import os
|
||
from functools import lru_cache
|
||
|
||
from pydantic import field_validator
|
||
from pydantic_settings import BaseSettings, SettingsConfigDict
|
||
|
||
#: The A9 import formats (PLAN anchor A9, revised 2026-08-21).
|
||
#: ``BOR_IMPORT_EXTENSIONS`` may narrow — but never widen — this set.
|
||
_ALLOWED_IMPORT_EXTENSIONS: frozenset[str] = frozenset(
|
||
{"md", "markdown", "txt", "yaml", "yml", "json", "py"}
|
||
)
|
||
|
||
|
||
class Settings(BaseSettings):
|
||
model_config = SettingsConfigDict(
|
||
env_file=".env",
|
||
env_file_encoding="utf-8",
|
||
env_prefix="BOR_",
|
||
extra="ignore",
|
||
)
|
||
|
||
# --- App ---
|
||
app_name: str = "Brain of Reese"
|
||
app_version: str = "0.1.0"
|
||
environment: str = "development"
|
||
log_level: str = "INFO"
|
||
static_dir: str = "frontend"
|
||
|
||
# --- Database (PostgreSQL 17 + pgvector) ---
|
||
database_url: str = "postgresql+psycopg://reese:reese@localhost:5432/brain_of_reese"
|
||
|
||
# --- LLM (self-hosted, OpenAI-compatible "aipi" endpoint) ---
|
||
llm_base_url: str = "https://aipi.reeseapps.com/v1"
|
||
llm_api_key: str = ""
|
||
llm_chat_model: str = "turbo"
|
||
llm_embed_model: str = "embed"
|
||
#: One-shot (non-streaming) completion model (A5 extended, phase 30):
|
||
#: document summaries at import time and the KB overview (phase 31).
|
||
#: Served by the same OpenAI-compatible endpoint — no new model
|
||
#: management. Called via ``LLMClient.chat()``.
|
||
llm_summary_model: str = "lite"
|
||
#: Operator kill-switch for the ``thinking`` SSE events (phase 17,
|
||
#: ``BOR_STREAM_THINKING``; ``0``/``false`` → off). When off, thinking
|
||
#: pieces are still counted for the per-turn log line but never
|
||
#: emitted — the answer stream itself is unchanged.
|
||
stream_thinking: bool = True
|
||
|
||
# --- RAG tuning ---
|
||
embedding_dim: int = 768 # verified against aipi /v1 (embed model)
|
||
top_n_docs: int = 2
|
||
# Honesty gate (A8, re-tuned 2026-08-21): the ``embed`` model's cosine
|
||
# scores compress into 0.41–0.84 on the real corpus, so the old 0.30
|
||
# default never discriminated. LOW only fires when best cosine < this
|
||
# AND no candidate chunk matches the question lexically (see A8).
|
||
relevance_threshold: float = 0.62
|
||
#: Maximum output tokens a chat answer may use (owner instruction
|
||
#: 2026-08-22: answers must run to their natural end — the old hard
|
||
#: 700-token cap cut long answers off mid-sentence).
|
||
max_output_tokens: int = 32_768
|
||
chunk_target_chars: int = 2_000
|
||
chunk_overlap_chars: int = 200
|
||
embed_batch_size: int = 16
|
||
#: Total char budget for the ``<tuning>`` section of the system prompt
|
||
#: (phase 15, steering notes). The newest-fitting notes are kept and the
|
||
#: overflow is replaced by the ``[…truncated…]`` marker.
|
||
steering_max_chars: int = 8_000
|
||
#: Cap on the document content sent to the ``lite`` summary model in one
|
||
#: call (phase 30, ``BOR_SUMMARY_MAX_CHARS``). Overflow is cut at the cap
|
||
#: and the shared ``[…truncated…]`` marker is appended (see
|
||
#: ``app.rag.summarizer``).
|
||
summary_max_chars: int = 12_000
|
||
#: Char budget for the ``<knowledge_base>`` section of the system prompt
|
||
#: (phase 31: lite-generated KB overview, ``app.rag.overview``). The
|
||
#: newest-fitting prefix of the stored outline is kept and the overflow
|
||
#: is replaced by the shared ``[…truncated…]`` marker (phase 15
|
||
#: convention — ``app.rag.prompts``).
|
||
kb_overview_max_chars: int = 4_000
|
||
#: Cap on the document list (source/path/title/first summary line per
|
||
#: row) sent to the ``lite`` overview model in one call (phase 31,
|
||
#: ``app.rag.overview``). Overflow is cut at the cap and the shared
|
||
#: ``[…truncated…]`` marker is appended (summarizer convention).
|
||
overview_input_max_chars: int = 40_000
|
||
#: Per-turn opportunities to call the ``list_documents`` agent tool
|
||
#: (phase 37, ``app.rag.agent``); 0 disables the tool entirely
|
||
#: (pre-phase behavior with both budgets at 0).
|
||
agent_list_calls: int = 1
|
||
#: Per-turn opportunities to call the ``read_document`` agent tool
|
||
#: (phase 37, ``app.rag.agent``); 0 disables the tool entirely.
|
||
agent_read_calls: int = 1
|
||
|
||
# --- Hybrid retrieval (A7, revised 2026-08-21) ---
|
||
# cosine top-N ∪ Postgres FTS top-N, fused with Reciprocal Rank Fusion
|
||
# (score = Σ 1/(rrf_k + rank) over the lists a chunk appears in).
|
||
#
|
||
# The vector window is deliberately wider than the lexical one: a
|
||
# name-your-tool question's best *lexical* chunk (e.g. the "Install"
|
||
# section of gitlab.md) can sit far down the vector ranking because the
|
||
# question embeds close to generic templates. A 100-wide window is what
|
||
# lets such chunks double-hit (one RRF term per list) and outrank a
|
||
# template that owns vector rank 1 — measured 2026-08-22 against the
|
||
# live 2774-chunk KB for "How did I install gitlab?" (gitlab.md:1 at
|
||
# vrank 100 / lrank 3 → fused 0.0221 vs the template's 0.0164).
|
||
hybrid_vector_candidates: int = 100
|
||
hybrid_lexical_candidates: int = 30
|
||
rrf_k: int = 60
|
||
|
||
# --- Admin & sign-in (phase 16; A10 revised 2026-08-22) ---
|
||
# Single-admin auth via a signed session cookie (Starlette
|
||
# SessionMiddleware — no new services, no DB tables). Both secrets are
|
||
# REQUIRED at startup: ``create_app()`` refuses to boot when either is
|
||
# empty (``app.core.auth.ensure_admin_configured``). The password is
|
||
# plaintext on purpose (homelab scope, owner decision 2026-08-22);
|
||
# the session secret signs the cookie (``secrets.token_hex(32)``).
|
||
admin_password: str = ""
|
||
session_secret: str = ""
|
||
#: Signed-cookie lifetime in seconds (default 12 h, refreshed on
|
||
#: session writes — sliding for an active admin).
|
||
session_max_age: int = 43_200
|
||
session_cookie: str = "bor_session"
|
||
|
||
# --- Import scope (A9, revised 2026-08-21) ---
|
||
# Comma-separated list of lowercased file extensions (no dot) imported
|
||
# by ``scripts/import_docs.py``. Hidden (dot) path components are always
|
||
# skipped, plus the importer's exclusion list.
|
||
# Stored as a raw CSV string (env-native — no JSON) and parsed on demand
|
||
# via :py:meth:`import_extension_set`. ``mode="after"`` validation runs
|
||
# against the raw string so a typo fails loudly at startup.
|
||
import_extensions: str = "md,markdown,txt,yaml,yml,json,py"
|
||
#: List of git repo URLs to clone/pull into ``sources_dir`` before
|
||
#: indexing (phase 28); comma-separated, stored raw. Empty means no git
|
||
#: sources — ``import_docs`` then falls back to ``--source`` / the old
|
||
#: ``DEFAULT_SOURCES``.
|
||
git_sources: str = ""
|
||
#: Where ``import_docs`` clones/pulls the ``git_sources`` repos (phase
|
||
#: 28). Stored as a raw string — ``Path.expanduser()`` is applied in
|
||
#: the import script, not here.
|
||
sources_dir: str = "~/bor-sources"
|
||
|
||
@field_validator("import_extensions")
|
||
@classmethod
|
||
def _import_extensions_known(cls, v: str) -> str:
|
||
"""Reject unknown/empty formats loudly instead of silently importing
|
||
nothing (a typo like ``md,jsonn`` would otherwise walk zero files)."""
|
||
exts = {part.strip().lstrip(".").lower() for part in v.split(",") if part.strip()}
|
||
if not exts:
|
||
raise ValueError("import_extensions must name at least one format")
|
||
unknown = exts - _ALLOWED_IMPORT_EXTENSIONS
|
||
if unknown:
|
||
raise ValueError(
|
||
f"unknown import extension(s): {', '.join(sorted(unknown))} — "
|
||
f"allowed: {', '.join(sorted(_ALLOWED_IMPORT_EXTENSIONS))}"
|
||
)
|
||
return v
|
||
|
||
# Suggested questions (onboarding + empty state).
|
||
suggestions: list[str] = [
|
||
"How is my Kubernetes cluster set up?",
|
||
"What's my backup strategy?",
|
||
"How do I deploy a new service?",
|
||
"What's currently running in the homelab?",
|
||
]
|
||
|
||
@property
|
||
def import_extension_set(self) -> frozenset[str]:
|
||
"""Lowercased, dotted extension set (``.md``) for path filtering."""
|
||
return frozenset(
|
||
f".{part.strip().lstrip('.').lower()}"
|
||
for part in self.import_extensions.split(",")
|
||
if part.strip()
|
||
)
|
||
|
||
@property
|
||
def git_source_list(self) -> list[str]:
|
||
"""Non-empty, stripped git URLs from :py:attr:`git_sources` (phase 28).
|
||
|
||
Whitespace around each entry is trimmed and empty entries dropped;
|
||
an unset/empty value yields ``[]`` (the import script then uses its
|
||
legacy local-directory defaults).
|
||
"""
|
||
return [part.strip() for part in self.git_sources.split(",") if part.strip()]
|
||
|
||
@property
|
||
def effective_api_key(self) -> str:
|
||
"""API key for aipi: explicit setting, then $AIPI_KEY, then a placeholder."""
|
||
return self.llm_api_key or os.environ.get("AIPI_KEY", "") or "not-needed"
|
||
|
||
|
||
@lru_cache
|
||
def get_settings() -> Settings:
|
||
return Settings()
|