feat: scaffold Brain of Reese — FastAPI RAG chat over Postgres 17 + pgvector

Foundation (phase 01, verified):
- FastAPI app: /api/health, /api/suggestions, /api/chat (placeholder),
  static frontend served locally (no CDN)
- Postgres 17 + pgvector via db/Containerfile + compose.yaml
  (podman compose up -d db), Alembic initial migration (documents,
  chunks with vector(768), query_log)
- LLM client targeting https://aipi.reeseapps.com/v1 (turbo/embed);
  scripts/llm_probe.py verified models + 768-dim embeddings live
- Conditional debugpy: imported only when DEBUGPY=1 (attach on demand,
  :5678); logging config for clean single-line logs
- Frontend shell: mobile-first chat + Sources pages, tokens, a11y baselines
- Tests: 24 unit+integration (99% coverage on app/), ruff + pyright clean,
  Playwright smoke E2E (3 tests) against a deterministic mock LLM
- Planning: .agent/PLAN.md (architecture + LOCKED decisions), AGENTS.md,
  6 user stories, 7 phase files (one story / one phase / one Playwright
  suite each)
This commit is contained in:
2026-08-21 13:42:21 -04:00
commit 022da8e2bc
63 changed files with 5225 additions and 0 deletions
+1
View File
@@ -0,0 +1 @@
"""E2E test package."""
+126
View File
@@ -0,0 +1,126 @@
"""Playwright E2E fixtures (shared by every story's test file).
Each user story in ``.agent/user_stories/`` gets its own isolated E2E test
file; this conftest provides the shared environment:
* ``mock_llm`` — deterministic OpenAI-compatible server (see mock_llm.py).
Set ``E2E_REAL_LLM=1`` to point at the real aipi endpoint
instead (requires an imported knowledge base).
* ``app_server`` — the real FastAPI app under test (uvicorn subprocess).
* ``browser``/``page`` — headless Chromium pointed at the app.
Prerequisite for story tests that touch the database:
podman compose up -d db
"""
from __future__ import annotations
import os
import subprocess
import sys
import time
from collections.abc import Iterator
from pathlib import Path
import httpx
import pytest
from playwright.sync_api import Browser, Page, sync_playwright
REPO = Path(__file__).resolve().parents[2]
APP_PORT = int(os.environ.get("E2E_APP_PORT", "8123"))
MOCK_PORT = int(os.environ.get("E2E_MOCK_PORT", "8901"))
APP_URL = f"http://127.0.0.1:{APP_PORT}"
USE_REAL_LLM = os.environ.get("E2E_REAL_LLM") == "1"
def _wait_http(url: str, timeout: float = 40.0) -> None:
deadline = time.monotonic() + timeout
last_err = "unknown"
while time.monotonic() < deadline:
try:
httpx.get(url, timeout=2.0)
return
except Exception as e: # noqa: BLE001 — retry until deadline
last_err = str(e)
time.sleep(0.5)
raise RuntimeError(f"server at {url} did not come up: {last_err}")
@pytest.fixture(scope="session")
def mock_llm() -> Iterator[int]:
"""Deterministic OpenAI-compatible LLM (chat + embeddings)."""
if USE_REAL_LLM:
yield 0
return
env = dict(os.environ)
env.pop("DEBUGPY", None)
proc = subprocess.Popen(
[sys.executable, "-m", "uvicorn", "tests.e2e.mock_llm:app",
"--host", "127.0.0.1", "--port", str(MOCK_PORT), "--log-level", "warning"],
cwd=REPO,
env=env,
)
try:
_wait_http(f"http://127.0.0.1:{MOCK_PORT}/v1/models")
yield MOCK_PORT
finally:
proc.terminate()
try:
proc.wait(timeout=10)
except subprocess.TimeoutExpired:
proc.kill()
@pytest.fixture(scope="session")
def app_server(mock_llm: int) -> Iterator[str]:
"""The real app under test."""
env = dict(os.environ)
env.pop("DEBUGPY", None)
env["BOR_ENVIRONMENT"] = "e2e"
env["BOR_STATIC_DIR"] = str(REPO / "frontend")
env["BOR_LLM_BASE_URL"] = (
"https://aipi.reeseapps.com/v1"
if USE_REAL_LLM
else f"http://127.0.0.1:{mock_llm}/v1"
)
env.setdefault("BOR_DATABASE_URL", "postgresql+psycopg://reese:reese@localhost:5432/brain_of_reese")
proc = subprocess.Popen(
[sys.executable, "-m", "uvicorn", "app.main:app",
"--host", "127.0.0.1", "--port", str(APP_PORT), "--log-level", "warning"],
cwd=REPO,
env=env,
)
try:
_wait_http(f"{APP_URL}/api/health")
yield APP_URL
finally:
proc.terminate()
try:
proc.wait(timeout=10)
except subprocess.TimeoutExpired:
proc.kill()
@pytest.fixture(scope="session")
def app_url(app_server: str) -> str:
return app_server
@pytest.fixture()
def db_ready(app_url: str) -> None:
"""Skip a test with clear instructions when Postgres is not running."""
body = httpx.get(f"{app_url}/api/health", timeout=5).json()
if body["db"] != "up":
pytest.skip("Postgres not reachable — run `podman compose up -d db` first")
@pytest.fixture(scope="session")
def browser() -> Iterator[Browser]:
with sync_playwright() as p:
yield p.chromium.launch(headless=True)
@pytest.fixture()
def page(browser: Browser) -> Iterator[Page]:
pg = browser.new_page(viewport={"width": 1280, "height": 800})
yield pg
pg.close()
+174
View File
@@ -0,0 +1,174 @@
"""Deterministic OpenAI-compatible mock for E2E tests (aipi stand-in).
Implements just enough of the aipi surface:
* ``GET /v1/models``
* ``POST /v1/embeddings`` — real bag-of-words vectors (768-dim, L2-normed).
Because similarity is *genuine token overlap*, the relevance threshold
behaves the same way it will in production: related questions score high,
unrelated ones score low and trigger honest deflection.
* ``POST /v1/chat/completions`` — streaming (SSE) or not. The content keys
off markers in the system prompt:
- ``DEFLECT_MODE`` -> honest "I haven't done anything like that" answer
- otherwise -> upbeat answer quoting the provided document context
- user message containing ``pretend to think slowly`` -> 3s warm-up delay
(used by the loading-feedback story).
"""
from __future__ import annotations
import hashlib
import math
import re
import time
import uuid
from typing import Any
from fastapi import FastAPI
from fastapi.responses import StreamingResponse
app = FastAPI()
DIM = 768
TOKEN_RE = re.compile(r"[a-z0-9]+")
def embed_text(text: str) -> list[float]:
vec = [0.0] * DIM
for tok in TOKEN_RE.findall(text.lower()):
idx = int(hashlib.md5(tok.encode()).hexdigest(), 16) % DIM
vec[idx] += 1.0
norm = math.sqrt(sum(v * v for v in vec)) or 1.0
return [v / norm for v in vec]
def _messages(body: dict[str, Any]) -> list[dict[str, str]]:
return body.get("messages", [])
def _system(body: dict[str, Any]) -> str:
return " ".join(m.get("content", "") for m in _messages(body) if m.get("role") == "system")
def _user(body: dict[str, Any]) -> str:
parts = [m.get("content", "") for m in _messages(body) if m.get("role") == "user"]
return parts[-1] if parts else ""
def _context(body: dict[str, Any]) -> str:
"""The document context is the longest system/user message in practice."""
msgs = _messages(body)
return max((m.get("content", "") for m in msgs), key=len)
def compose_answer(body: dict[str, Any]) -> str:
system = _system(body)
user = _user(body)
if "DEFLECT_MODE" in system:
return (
"Ah — I haven't done anything like that, so I don't want to make stuff up! "
"You're thinking bigger than my notes for a second. Try asking about "
"kubernetes, backups, or deploying a new service — I know those inside out. "
"You've got this!"
)
ctx = _context(body)
snippet = ctx[:220].replace("\n", " ").strip()
return (
f"Great question — you've absolutely got this! Here's what my notes say about "
f"“{user.strip()[:80]}”: {snippet}… That's the gist from the docs; happy to "
"dig into any of it. (Deterministic mock answer for E2E.)"
)
@app.get("/v1/models")
def models() -> dict[str, Any]:
return {
"object": "list",
"data": [
{"id": "turbo", "object": "model"},
{"id": "embed", "object": "model"},
{"id": "lite", "object": "model"},
],
}
@app.post("/v1/embeddings")
def embeddings(body: dict[str, Any]) -> dict[str, Any]:
raw = body.get("input")
if isinstance(raw, str):
raw = [raw]
inputs: list[Any] = list(raw) if isinstance(raw, list) else []
data = [
{"object": "embedding", "index": i, "embedding": embed_text(t)}
for i, t in enumerate(inputs)
]
return {
"object": "list",
"data": data,
"model": body.get("model", "embed"),
"usage": {"prompt_tokens": 8, "total_tokens": 8},
}
def _sse_stream(answer: str, delay: float) -> Any:
model = "turbo"
chunk_id = f"chatcmpl-{uuid.uuid4()}"
if delay:
time.sleep(delay)
for piece in re.findall(r".{1,12}", answer, re.S):
payload = {
"id": chunk_id,
"object": "chat.completion.chunk",
"created": int(time.time()),
"model": model,
"choices": [{"index": 0, "delta": {"content": piece}, "finish_reason": None}],
}
yield f"data: {json_dumps(payload)}\n\n"
time.sleep(0.02)
yield (
"data: "
+ json_dumps(
{
"id": chunk_id,
"object": "chat.completion.chunk",
"created": int(time.time()),
"model": model,
"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}],
}
)
+ "\n\n"
)
yield "data: [DONE]\n\n"
def json_dumps(obj: dict[str, Any]) -> str:
import json
return json.dumps(obj)
@app.post("/v1/chat/completions")
def chat_completions(body: dict[str, Any]) -> Any:
answer = compose_answer(body)
delay = 3.0 if "pretend to think slowly" in _user(body) else 0.0
if not body.get("stream"):
return {
"id": f"chatcmpl-{uuid.uuid4()}",
"object": "chat.completion",
"created": int(time.time()),
"model": body.get("model", "turbo"),
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": answer},
"finish_reason": "stop",
}
],
"usage": {"prompt_tokens": 100, "completion_tokens": 50, "total_tokens": 150},
}
return StreamingResponse(
_sse_stream(answer, delay),
media_type="text/event-stream",
headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"},
)
+45
View File
@@ -0,0 +1,45 @@
"""Phase 01 smoke E2E: the app boots, serves the local frontend, and the
placeholder chat round-trips without a stale button.
Run: uv run pytest tests/e2e/test_smoke.py -v
"""
from __future__ import annotations
import re
import httpx
from playwright.sync_api import Page, expect
def test_health_endpoint(app_url: str) -> None:
r = httpx.get(f"{app_url}/api/health", timeout=5)
assert r.status_code == 200
assert r.json()["status"] == "ok"
def test_index_page_loads_locally(page: Page, app_url: str) -> None:
page.goto(app_url)
assert page.title() == "Brain of Reese"
assert page.locator(".brand").is_visible()
# No external (CDN) resources in the document.
html = page.content()
assert 'src="http' not in html
assert 'href="http' not in html.replace('href="http://www.w3.org', "")
def test_placeholder_chat_roundtrip(page: Page, app_url: str) -> None:
page.goto(app_url)
page.locator("#message-input").fill("hello brain")
page.locator("#send-btn").click()
# User bubble appears, then the Brain placeholder answer arrives.
page.locator(".msg.user .bubble").first.wait_for(state="visible", timeout=10_000)
brain_bubble = page.locator(".msg.brain .bubble").first
brain_bubble.wait_for(state="visible", timeout=10_000)
# to_have_text retries until the async fetch resolves (no stale read).
expect(brain_bubble).to_have_text(re.compile("neurons"), timeout=10_000)
# Button is never left stuck: back to "Send" and enabled.
btn = page.locator("#send-btn")
assert btn.is_enabled()
assert "Send" in btn.inner_text()