feat: scaffold Brain of Reese — FastAPI RAG chat over Postgres 17 + pgvector
Foundation (phase 01, verified): - FastAPI app: /api/health, /api/suggestions, /api/chat (placeholder), static frontend served locally (no CDN) - Postgres 17 + pgvector via db/Containerfile + compose.yaml (podman compose up -d db), Alembic initial migration (documents, chunks with vector(768), query_log) - LLM client targeting https://aipi.reeseapps.com/v1 (turbo/embed); scripts/llm_probe.py verified models + 768-dim embeddings live - Conditional debugpy: imported only when DEBUGPY=1 (attach on demand, :5678); logging config for clean single-line logs - Frontend shell: mobile-first chat + Sources pages, tokens, a11y baselines - Tests: 24 unit+integration (99% coverage on app/), ruff + pyright clean, Playwright smoke E2E (3 tests) against a deterministic mock LLM - Planning: .agent/PLAN.md (architecture + LOCKED decisions), AGENTS.md, 6 user stories, 7 phase files (one story / one phase / one Playwright suite each)
This commit is contained in:
@@ -0,0 +1 @@
|
||||
"""E2E test package."""
|
||||
@@ -0,0 +1,126 @@
|
||||
"""Playwright E2E fixtures (shared by every story's test file).
|
||||
|
||||
Each user story in ``.agent/user_stories/`` gets its own isolated E2E test
|
||||
file; this conftest provides the shared environment:
|
||||
|
||||
* ``mock_llm`` — deterministic OpenAI-compatible server (see mock_llm.py).
|
||||
Set ``E2E_REAL_LLM=1`` to point at the real aipi endpoint
|
||||
instead (requires an imported knowledge base).
|
||||
* ``app_server`` — the real FastAPI app under test (uvicorn subprocess).
|
||||
* ``browser``/``page`` — headless Chromium pointed at the app.
|
||||
|
||||
Prerequisite for story tests that touch the database:
|
||||
podman compose up -d db
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from collections.abc import Iterator
|
||||
from pathlib import Path
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
from playwright.sync_api import Browser, Page, sync_playwright
|
||||
|
||||
REPO = Path(__file__).resolve().parents[2]
|
||||
APP_PORT = int(os.environ.get("E2E_APP_PORT", "8123"))
|
||||
MOCK_PORT = int(os.environ.get("E2E_MOCK_PORT", "8901"))
|
||||
APP_URL = f"http://127.0.0.1:{APP_PORT}"
|
||||
USE_REAL_LLM = os.environ.get("E2E_REAL_LLM") == "1"
|
||||
|
||||
|
||||
def _wait_http(url: str, timeout: float = 40.0) -> None:
|
||||
deadline = time.monotonic() + timeout
|
||||
last_err = "unknown"
|
||||
while time.monotonic() < deadline:
|
||||
try:
|
||||
httpx.get(url, timeout=2.0)
|
||||
return
|
||||
except Exception as e: # noqa: BLE001 — retry until deadline
|
||||
last_err = str(e)
|
||||
time.sleep(0.5)
|
||||
raise RuntimeError(f"server at {url} did not come up: {last_err}")
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def mock_llm() -> Iterator[int]:
|
||||
"""Deterministic OpenAI-compatible LLM (chat + embeddings)."""
|
||||
if USE_REAL_LLM:
|
||||
yield 0
|
||||
return
|
||||
env = dict(os.environ)
|
||||
env.pop("DEBUGPY", None)
|
||||
proc = subprocess.Popen(
|
||||
[sys.executable, "-m", "uvicorn", "tests.e2e.mock_llm:app",
|
||||
"--host", "127.0.0.1", "--port", str(MOCK_PORT), "--log-level", "warning"],
|
||||
cwd=REPO,
|
||||
env=env,
|
||||
)
|
||||
try:
|
||||
_wait_http(f"http://127.0.0.1:{MOCK_PORT}/v1/models")
|
||||
yield MOCK_PORT
|
||||
finally:
|
||||
proc.terminate()
|
||||
try:
|
||||
proc.wait(timeout=10)
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def app_server(mock_llm: int) -> Iterator[str]:
|
||||
"""The real app under test."""
|
||||
env = dict(os.environ)
|
||||
env.pop("DEBUGPY", None)
|
||||
env["BOR_ENVIRONMENT"] = "e2e"
|
||||
env["BOR_STATIC_DIR"] = str(REPO / "frontend")
|
||||
env["BOR_LLM_BASE_URL"] = (
|
||||
"https://aipi.reeseapps.com/v1"
|
||||
if USE_REAL_LLM
|
||||
else f"http://127.0.0.1:{mock_llm}/v1"
|
||||
)
|
||||
env.setdefault("BOR_DATABASE_URL", "postgresql+psycopg://reese:reese@localhost:5432/brain_of_reese")
|
||||
proc = subprocess.Popen(
|
||||
[sys.executable, "-m", "uvicorn", "app.main:app",
|
||||
"--host", "127.0.0.1", "--port", str(APP_PORT), "--log-level", "warning"],
|
||||
cwd=REPO,
|
||||
env=env,
|
||||
)
|
||||
try:
|
||||
_wait_http(f"{APP_URL}/api/health")
|
||||
yield APP_URL
|
||||
finally:
|
||||
proc.terminate()
|
||||
try:
|
||||
proc.wait(timeout=10)
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def app_url(app_server: str) -> str:
|
||||
return app_server
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def db_ready(app_url: str) -> None:
|
||||
"""Skip a test with clear instructions when Postgres is not running."""
|
||||
body = httpx.get(f"{app_url}/api/health", timeout=5).json()
|
||||
if body["db"] != "up":
|
||||
pytest.skip("Postgres not reachable — run `podman compose up -d db` first")
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def browser() -> Iterator[Browser]:
|
||||
with sync_playwright() as p:
|
||||
yield p.chromium.launch(headless=True)
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def page(browser: Browser) -> Iterator[Page]:
|
||||
pg = browser.new_page(viewport={"width": 1280, "height": 800})
|
||||
yield pg
|
||||
pg.close()
|
||||
@@ -0,0 +1,174 @@
|
||||
"""Deterministic OpenAI-compatible mock for E2E tests (aipi stand-in).
|
||||
|
||||
Implements just enough of the aipi surface:
|
||||
|
||||
* ``GET /v1/models``
|
||||
* ``POST /v1/embeddings`` — real bag-of-words vectors (768-dim, L2-normed).
|
||||
Because similarity is *genuine token overlap*, the relevance threshold
|
||||
behaves the same way it will in production: related questions score high,
|
||||
unrelated ones score low and trigger honest deflection.
|
||||
* ``POST /v1/chat/completions`` — streaming (SSE) or not. The content keys
|
||||
off markers in the system prompt:
|
||||
- ``DEFLECT_MODE`` -> honest "I haven't done anything like that" answer
|
||||
- otherwise -> upbeat answer quoting the provided document context
|
||||
- user message containing ``pretend to think slowly`` -> 3s warm-up delay
|
||||
(used by the loading-feedback story).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import math
|
||||
import re
|
||||
import time
|
||||
import uuid
|
||||
from typing import Any
|
||||
|
||||
from fastapi import FastAPI
|
||||
from fastapi.responses import StreamingResponse
|
||||
|
||||
app = FastAPI()
|
||||
|
||||
DIM = 768
|
||||
TOKEN_RE = re.compile(r"[a-z0-9]+")
|
||||
|
||||
|
||||
def embed_text(text: str) -> list[float]:
|
||||
vec = [0.0] * DIM
|
||||
for tok in TOKEN_RE.findall(text.lower()):
|
||||
idx = int(hashlib.md5(tok.encode()).hexdigest(), 16) % DIM
|
||||
vec[idx] += 1.0
|
||||
norm = math.sqrt(sum(v * v for v in vec)) or 1.0
|
||||
return [v / norm for v in vec]
|
||||
|
||||
|
||||
def _messages(body: dict[str, Any]) -> list[dict[str, str]]:
|
||||
return body.get("messages", [])
|
||||
|
||||
|
||||
def _system(body: dict[str, Any]) -> str:
|
||||
return " ".join(m.get("content", "") for m in _messages(body) if m.get("role") == "system")
|
||||
|
||||
|
||||
def _user(body: dict[str, Any]) -> str:
|
||||
parts = [m.get("content", "") for m in _messages(body) if m.get("role") == "user"]
|
||||
return parts[-1] if parts else ""
|
||||
|
||||
|
||||
def _context(body: dict[str, Any]) -> str:
|
||||
"""The document context is the longest system/user message in practice."""
|
||||
msgs = _messages(body)
|
||||
return max((m.get("content", "") for m in msgs), key=len)
|
||||
|
||||
|
||||
def compose_answer(body: dict[str, Any]) -> str:
|
||||
system = _system(body)
|
||||
user = _user(body)
|
||||
if "DEFLECT_MODE" in system:
|
||||
return (
|
||||
"Ah — I haven't done anything like that, so I don't want to make stuff up! "
|
||||
"You're thinking bigger than my notes for a second. Try asking about "
|
||||
"kubernetes, backups, or deploying a new service — I know those inside out. "
|
||||
"You've got this!"
|
||||
)
|
||||
ctx = _context(body)
|
||||
snippet = ctx[:220].replace("\n", " ").strip()
|
||||
return (
|
||||
f"Great question — you've absolutely got this! Here's what my notes say about "
|
||||
f"“{user.strip()[:80]}”: {snippet}… That's the gist from the docs; happy to "
|
||||
"dig into any of it. (Deterministic mock answer for E2E.)"
|
||||
)
|
||||
|
||||
|
||||
@app.get("/v1/models")
|
||||
def models() -> dict[str, Any]:
|
||||
return {
|
||||
"object": "list",
|
||||
"data": [
|
||||
{"id": "turbo", "object": "model"},
|
||||
{"id": "embed", "object": "model"},
|
||||
{"id": "lite", "object": "model"},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
@app.post("/v1/embeddings")
|
||||
def embeddings(body: dict[str, Any]) -> dict[str, Any]:
|
||||
raw = body.get("input")
|
||||
if isinstance(raw, str):
|
||||
raw = [raw]
|
||||
inputs: list[Any] = list(raw) if isinstance(raw, list) else []
|
||||
data = [
|
||||
{"object": "embedding", "index": i, "embedding": embed_text(t)}
|
||||
for i, t in enumerate(inputs)
|
||||
]
|
||||
return {
|
||||
"object": "list",
|
||||
"data": data,
|
||||
"model": body.get("model", "embed"),
|
||||
"usage": {"prompt_tokens": 8, "total_tokens": 8},
|
||||
}
|
||||
|
||||
|
||||
def _sse_stream(answer: str, delay: float) -> Any:
|
||||
model = "turbo"
|
||||
chunk_id = f"chatcmpl-{uuid.uuid4()}"
|
||||
if delay:
|
||||
time.sleep(delay)
|
||||
for piece in re.findall(r".{1,12}", answer, re.S):
|
||||
payload = {
|
||||
"id": chunk_id,
|
||||
"object": "chat.completion.chunk",
|
||||
"created": int(time.time()),
|
||||
"model": model,
|
||||
"choices": [{"index": 0, "delta": {"content": piece}, "finish_reason": None}],
|
||||
}
|
||||
yield f"data: {json_dumps(payload)}\n\n"
|
||||
time.sleep(0.02)
|
||||
yield (
|
||||
"data: "
|
||||
+ json_dumps(
|
||||
{
|
||||
"id": chunk_id,
|
||||
"object": "chat.completion.chunk",
|
||||
"created": int(time.time()),
|
||||
"model": model,
|
||||
"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}],
|
||||
}
|
||||
)
|
||||
+ "\n\n"
|
||||
)
|
||||
yield "data: [DONE]\n\n"
|
||||
|
||||
|
||||
def json_dumps(obj: dict[str, Any]) -> str:
|
||||
import json
|
||||
|
||||
return json.dumps(obj)
|
||||
|
||||
|
||||
@app.post("/v1/chat/completions")
|
||||
def chat_completions(body: dict[str, Any]) -> Any:
|
||||
answer = compose_answer(body)
|
||||
delay = 3.0 if "pretend to think slowly" in _user(body) else 0.0
|
||||
|
||||
if not body.get("stream"):
|
||||
return {
|
||||
"id": f"chatcmpl-{uuid.uuid4()}",
|
||||
"object": "chat.completion",
|
||||
"created": int(time.time()),
|
||||
"model": body.get("model", "turbo"),
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"message": {"role": "assistant", "content": answer},
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
],
|
||||
"usage": {"prompt_tokens": 100, "completion_tokens": 50, "total_tokens": 150},
|
||||
}
|
||||
|
||||
return StreamingResponse(
|
||||
_sse_stream(answer, delay),
|
||||
media_type="text/event-stream",
|
||||
headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"},
|
||||
)
|
||||
@@ -0,0 +1,45 @@
|
||||
"""Phase 01 smoke E2E: the app boots, serves the local frontend, and the
|
||||
placeholder chat round-trips without a stale button.
|
||||
|
||||
Run: uv run pytest tests/e2e/test_smoke.py -v
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
import httpx
|
||||
from playwright.sync_api import Page, expect
|
||||
|
||||
|
||||
def test_health_endpoint(app_url: str) -> None:
|
||||
r = httpx.get(f"{app_url}/api/health", timeout=5)
|
||||
assert r.status_code == 200
|
||||
assert r.json()["status"] == "ok"
|
||||
|
||||
|
||||
def test_index_page_loads_locally(page: Page, app_url: str) -> None:
|
||||
page.goto(app_url)
|
||||
assert page.title() == "Brain of Reese"
|
||||
assert page.locator(".brand").is_visible()
|
||||
# No external (CDN) resources in the document.
|
||||
html = page.content()
|
||||
assert 'src="http' not in html
|
||||
assert 'href="http' not in html.replace('href="http://www.w3.org', "")
|
||||
|
||||
|
||||
def test_placeholder_chat_roundtrip(page: Page, app_url: str) -> None:
|
||||
page.goto(app_url)
|
||||
page.locator("#message-input").fill("hello brain")
|
||||
page.locator("#send-btn").click()
|
||||
|
||||
# User bubble appears, then the Brain placeholder answer arrives.
|
||||
page.locator(".msg.user .bubble").first.wait_for(state="visible", timeout=10_000)
|
||||
brain_bubble = page.locator(".msg.brain .bubble").first
|
||||
brain_bubble.wait_for(state="visible", timeout=10_000)
|
||||
# to_have_text retries until the async fetch resolves (no stale read).
|
||||
expect(brain_bubble).to_have_text(re.compile("neurons"), timeout=10_000)
|
||||
|
||||
# Button is never left stuck: back to "Send" and enabled.
|
||||
btn = page.locator("#send-btn")
|
||||
assert btn.is_enabled()
|
||||
assert "Send" in btn.inner_text()
|
||||
Reference in New Issue
Block a user