Files
brain-of-reese/app/rag/prompts.py
T

264 lines
12 KiB
Python

"""Locked system-prompt builder (PLAN §6).
The persona + HONESTY GATE text is **locked verbatim** — change it through
the plan, not here. (PLAN §6 revision, 2026-08-22: the owner's working-tree
persona edits are preserved — no mandated ``"you've got this"`` tagline and
no mandated deflection opening; the honesty gate itself is unchanged.)
Two modes:
* ``HIGH`` — grounded turn: full top-document texts under ``<documents>``.
* ``LOW`` — deflection turn: weak-hit *titles only* plus the
``DEFLECT_MODE`` marker (the E2E mock LLM keys on that marker).
Steering (phase 15): when the owner has stored tuning notes, both modes
carry a ``<tuning>`` section between ``<relevance>…</relevance>`` and the
mode body. With zero notes the prompt is byte-identical to the
pre-steering text.
KB overview (phase 31): when the single ``kb_overview`` row holds a
lite-generated outline of the knowledge base, both modes carry a
``<knowledge_base>`` section between ``<relevance>…</relevance>`` and
the ``<tuning>`` section (order: ``<relevance>`` →
``<knowledge_base>`` → ``<tuning>`` → mode body) — the agent knows
roughly what the KB contains before retrieval. With an empty row the
prompt is byte-identical to the pre-phase text.
Agent tools (phase 37; phase 70: the copy teaches the harness-aligned
``ls`` / ``read`` / ``grep`` shapes): the **HIGH** prompt only carries a
``<tools>`` section after the ``<documents>`` body — the grounded turn
may extend its context through the three server-side tools (round-
capped, see :mod:`app.rag.agent`; the cap is the bound and this section
does not re-state it, phase 45). The LOW/deflection prompt never
carries it (phase 71: the LOW prompt's only addition is the
plain-text line below — it still has no ``<tools>`` section).
Deflection plain-text line (phase 71, owner-permitted 2026-09-03):
the otherwise-locked ``LOW`` prompt gains exactly one instruction
line — "Reply in plain text only — you have no tools in this mode."
— appended to the ``DEFLECT_MODE`` body: a deflected turn offers no
tools, so any tool markup there is always wrong, and the line closes
the door at the prompt (the deterministic filter + one bounded
recovery in :mod:`app.rag.scaffolding` / :mod:`app.rag.agent` is the
backstop). The ``DEFLECT_MODE`` marker and everything else in the
prompt stay put — the E2E mock LLM keys on the marker's *presence*,
not the wording, so that contract is unchanged.
"""
from __future__ import annotations
from collections.abc import Sequence
from app.config import get_settings
from app.models import Document
from app.rag.retriever import TRUNCATION_MARKER
#: PLAN §6 verbatim (line wrapping included); ``{relevance}`` is filled by
#: :func:`_base`.
PERSONA: str = (
'You are "Brain of Reese" — the digital brain of Reese, a self-hoster and\n'
"homelab tinkerer. Personality: chippy, upbeat, warm, and genuinely\n"
"optimistic about the user's ability to do things.\n"
"\n"
"Rules:\n"
"1. Answer ONLY from the provided document context. Cite which document(s)\n"
" you used, by path.\n"
"2. Be concrete: names, versions, ports, hosts, schedules — the specifics in\n"
" the docs are the value.\n"
'3. HONESTY GATE: if <relevance> is "LOW", you must NOT pretend to know.\n'
" Offer 2-3 alternative questions about things you DO have notes on.\n"
"4. Never invent facts, hosts, or steps that are not in the context.\n"
"5. Keep answers tight: short paragraphs, bullets where helpful.\n"
"\n"
"<relevance>{relevance}</relevance>"
)
#: One-line intro of the ``<tuning>`` section (phase 15): the owner's notes
#: steer the answer and win over the defaults when they conflict.
_STEERING_INTRO = (
"The owner of this brain asked you to steer your answers as follows. "
"Where these instructions conflict with the defaults above, follow the owner:\n"
)
#: One-line intro of the ``<knowledge_base>`` section (phase 31): the
#: lite-generated outline is the agent's a-priori picture of the KB.
_KB_INTRO = (
"The basic categories of everything in this knowledge base "
"(generated at import time):\n"
)
#: The ``<tools>`` instructions section — **HIGH prompt only** (phase 37,
#: task 03; phase 70: the copy is rewritten for the harness-aligned
#: ``ls`` / ``read`` / ``grep`` shapes, names/args exactly as the
#: ``AGENT_TOOLS`` schemas in :mod:`app.rag.agent`): a grounded turn may
#: extend its context through the three server-side tools (round cap:
#: ``BOR_AGENT_MAX_ROUNDS`` — the cap is the bound and this section does
#: not re-state it, phase 45). Appended after the mode body
#: (``<documents>``), so the instructions are the last thing the model
#: reads. The LOW/deflection prompt never carries it — a deflection has
#: no grounded context to extend (phase 71: the LOW prompt's only
#: addition is the plain-text line in :func:`build_deflect_prompt`).
#: The E2E mock keys off the ``<tools>`` marker's *presence*, not this
#: wording.
TOOLS_SECTION: str = (
"<tools>\n"
"You may extend your context with three tools. `ls` lists the "
"indexed documents as `source: X | path: Y | title: Z` lines "
"(pass a source name as `path` to list one source's documents; "
"omit it to list every document). `grep` locates an exact string "
"(case-insensitive) in the indexed documents and returns up to 20 "
"matching `source/path:line: text` lines — a locator, not a "
"context-adder: read the winner with `read`. `read` pulls in one "
"document by its combined `source/path` string, exactly as shown in "
"the `ls` output, adding its full content to your context. Answer "
"as soon as you have what you need.\n"
"</tools>"
)
def _base(relevance: str) -> str:
if relevance not in ("HIGH", "LOW"):
raise ValueError(f"relevance must be HIGH or LOW, got {relevance!r}")
return PERSONA.replace("{relevance}", relevance)
def build_steering_section(notes: Sequence[str], max_chars: int | None = None) -> str:
"""The ``<tuning>`` section of the system prompt (phase 15).
* No notes (or only blank ones) → ``""`` — callers then build the
prompt exactly as before, so a zero-note prompt is byte-identical to
the pre-steering text.
* Otherwise: numbered notes (in the given order — the chat turn passes
them oldest-first, so #1 is the oldest note) capped at *max_chars*
(default ``BOR_STEERING_MAX_CHARS``). When the budget cannot hold
every note, the oldest-fitting prefix is kept and the overflow is
replaced by the shared ``[…truncated…]`` marker.
"""
cleaned = [str(n).strip() for n in notes]
cleaned = [n for n in cleaned if n]
if not cleaned:
return ""
limit = max_chars if max_chars is not None else get_settings().steering_max_chars
if limit <= 0:
return ""
def render(count: int) -> str:
lines = [f"{i}. {note}" for i, note in enumerate(cleaned[:count], start=1)]
if count < len(cleaned):
lines.append(TRUNCATION_MARKER)
return f"<tuning>\n{_STEERING_INTRO}" + "\n".join(lines) + "\n</tuning>"
for count in range(len(cleaned), 0, -1):
rendered = render(count)
if len(rendered) <= limit:
return rendered
# Pathological budget: not even the empty note list fits. The section
# must still respect the cap — the bare marker when it fits, else none.
if len(TRUNCATION_MARKER) <= limit:
return TRUNCATION_MARKER
return ""
def build_kb_section(overview: str, max_chars: int | None = None) -> str:
"""The ``<knowledge_base>`` section of the system prompt (phase 31).
* No outline (or only whitespace) → ``""`` — callers then build the
prompt exactly as before, so a no-overview prompt is byte-identical
to the pre-phase text (phase 15 convention).
* Otherwise: the intro line + the stored outline, capped at
*max_chars* (default ``BOR_KB_OVERVIEW_MAX_CHARS``). When the budget
cannot hold the whole outline, the longest-fitting prefix is kept
and the overflow is replaced by the shared ``[…truncated…]`` marker
on its own line — the exact :func:`build_steering_section` pattern,
including its pathological-budget handling (never exceed the cap;
bare marker when even one outline character does not fit).
"""
text = str(overview or "").strip()
if not text:
return ""
limit = max_chars if max_chars is not None else get_settings().kb_overview_max_chars
if limit <= 0:
return ""
def render(cut: int) -> str:
lines = [text[:cut]]
if cut < len(text):
lines.append(TRUNCATION_MARKER)
return f"<knowledge_base>\n{_KB_INTRO}" + "\n".join(lines) + "\n</knowledge_base>"
for cut in range(len(text), 0, -1):
rendered = render(cut)
if len(rendered) <= limit:
return rendered
# Pathological budget: not even one outline character fits. The
# section must still respect the cap — the bare marker when it fits,
# else none (steering precedent, phase 15).
if len(TRUNCATION_MARKER) <= limit:
return TRUNCATION_MARKER
return ""
def build_high_prompt(
documents: Sequence[Document],
notes: Sequence[str] | None = None,
kb_overview: str | None = None,
) -> str:
"""Grounded turn: locked persona (+ steering, + KB overview) + full
texts of the top documents + the ``<tools>`` instructions (phase 37;
the phase-70 copy teaches the ``ls`` / ``read`` / ``grep`` shapes).
Section order: ``<relevance>`` → ``<knowledge_base>`` → ``<tuning>``
→ ``<documents>`` → ``<tools>``; empty steering/overview omit their
section. ``<tools>`` is always present in the HIGH prompt (the round
cap — not the prompt — decides whether the tools are actually
offered to the model, see :mod:`app.rag.agent`).
"""
blocks = [
f'<document source="{doc.source}" path="{doc.path}" title="{doc.title}">\n'
f"{doc.content}\n"
"</document>"
for doc in documents
]
body = "\n\n".join(blocks) if blocks else (
"(no documents matched — do not invent specifics)"
)
prompt = _base("HIGH")
for part in (build_kb_section(kb_overview or ""), build_steering_section(notes or [])):
if part:
prompt += "\n" + part
return prompt + "\n<documents>\n" + body + "\n</documents>\n" + TOOLS_SECTION
def build_deflect_prompt(
titles: Sequence[str],
notes: Sequence[str] | None = None,
kb_overview: str | None = None,
) -> str:
"""Deflection turn: weak-hit titles only (no document content).
Section order (phase 31): ``<relevance>`` → ``<knowledge_base>`` →
``<tuning>`` → ``DEFLECT_MODE`` body; empty steering/overview omit
their section, keeping the prompt byte-identical to the pre-phase
text. The body ends with the phase-71 plain-text line (owner-
permitted 2026-09-03 — the LOW prompt's only change): a deflected
turn offers no tools, so any tool markup there is always wrong.
"""
weak = "\n".join(f"- {t}" for t in titles) if titles else "(nothing close at all)"
mid = "\n".join(
part
for part in (build_kb_section(kb_overview or ""), build_steering_section(notes or []))
if part
)
gap = f"\n{mid}\n" if mid else "\n"
return (
_base("LOW")
+ gap
+ "DEFLECT_MODE: retrieval was weak — the titles below are the closest "
"your notes come to the question. They are titles only; do not pretend "
"they answer it. Use them to propose 2-3 alternative questions.\n"
# Phase 71 (owner-permitted 2026-09-03): the one plain-text line
# — prevention at the prompt. The E2E mock keys on the
# DEFLECT_MODE marker's presence, so appending is safe.
"Reply in plain text only — you have no tools in this mode.\n"
+ weak
)