Files
brain-of-reese/tests/e2e/test_loading_feedback.py
ducoterra 2683128876
Build and Push Containers / build-and-push-app (push) Successful in 2m24s
Build and Push Containers / build-and-push-db (push) Successful in 14s
phase: 111_chat_banner_retry
## Phase 111 Completion Report

**Implemented/Verified:**
- `#kb-banner` contains a `<button type="button" class="banner-retry" id="banner-retry">` (hidden by default, Retry label + SVG)
- `showErrorBanner(detail, opts)` reveals the button only when `opts.retryable` is true AND `lastBrainWrap` exists
- Turn-error path passes `{ retryable: true }`; all non-turn callers (share, save-doc, stale) remain text-only
- `clearErrorBanner()` re-hides the button
- `ERROR_HINT` changed from "Try again — …" to "If this persists, check the LLM is reachable."
- `.banner-retry` CSS styled as a pill (matching `.stale-regenerate` family)
- 12 source-assertion unit tests in `tests/unit/test_frontend_banner_retry.py`

**Test / Lint / Coverage:**
- `uv run pytest tests/unit/test_frontend_banner_retry.py -v --no-cov` → 12 passed
- `uv run pytest --cov=app --cov-report=term-missing` → 2362 passed, 99% coverage
- `uv run ruff check .` → All checks passed
- `uv run pyright` → 0 errors
- `tests/e2e/test_llm_retry.py` → 4 passed (in isolation)
- `tests/e2e/test_smoke.py` → 3 passed (in isolation)

**Completion Criteria:**
- ✅ Retry button visible after failed chat turn, re-runs last question
- ✅ Non-turn callers show text-only banner (no button)
- ✅ pytest green, coverage >90%, ruff + pyright clean
- ✅ Phase dir to be moved by pipeline gate

**Next pending phase:** `112_honesty_gate_weak_hits`
2026-09-14 23:07:56 -04:00

321 lines
12 KiB
Python

"""Phase 06 E2E (Playwright): loading feedback & progress.
Story: ``.agents/user_stories/loading-feedback.md``
Run in isolation (DB must be up: ``podman compose up -d db``):
uv run pytest tests/e2e/test_loading_feedback.py -v --no-cov
Determinism comes from two mock-LLM behaviors (tests/e2e/mock_llm.py):
* ``pretend to think slowly`` in the user message → a 3s warm-up before
the first token, wide enough to assert the pre-token UI (typing dots +
the enabled Stop button — phase 48 revised the busy-button contract) at
a known timestamp;
* the ``POST /__shutdown__`` hook → the ``llm_down`` fixture stops the
shared mock to simulate an LLM outage, then restores a fresh instance
on the same port so later tests keep working.
The state machine under test lives in frontend/assets/app.js:
``idle → thinking → streaming → done | error → idle`` (PLAN §7.4).
"""
from __future__ import annotations
import asyncio
import os
import re
import subprocess
import sys
import time
from collections.abc import Iterator
from pathlib import Path
from threading import Thread
from typing import Any
import httpx
import pytest
from playwright.sync_api import Page, expect
from sqlalchemy import text
from app.config import Settings
from app.db import SessionLocal
from app.rag.importer import ImportSummary, import_sources
from app.rag.llm import LLMClient
from e2e.auth_helpers import login
REPO = Path(__file__).resolve().parents[2]
FIXTURES = REPO / "tests" / "fixtures" / "docs"
MOCK_PORT = int(os.environ.get("E2E_MOCK_PORT", "8901"))
# The mock warms up for 3s when the question contains this marker, so the
# pre-token window is observable. (Retrieval may answer or deflect — the
# feedback states are identical either way.)
SLOW_QUESTION = "pretend to think slowly then tell me about kubernetes"
ON_TOPIC = "How is my Kubernetes cluster set up?"
TYPING = "#typing-indicator"
# The typing indicator is itself a .msg.brain — exclude its bubble.
ANSWER = ".msg.brain .bubble:not(.typing)"
# --------------------------------------------------------------------------
# KB seeding (same pattern as the phase 02/03 story suites)
# --------------------------------------------------------------------------
async def _import_fixtures(mock_port: int) -> ImportSummary:
kwargs: dict[str, Any] = {"_env_file": None, "llm_base_url": f"http://127.0.0.1:{mock_port}/v1"}
settings = Settings(**kwargs) # pyright: ignore[reportCallIssue]
return await import_sources([FIXTURES], LLMClient(settings))
def _run_in_thread(coro: Any) -> Any:
"""Run a coroutine on a worker thread.
Playwright's sync API keeps an asyncio loop running on the test thread,
so ``asyncio.run`` cannot be called directly from a test body.
"""
box: dict[str, Any] = {}
def runner() -> None:
try:
box["value"] = asyncio.run(coro)
except BaseException as e: # noqa: BLE001 — re-raised on the test thread
box["error"] = e
t = Thread(target=runner)
t.start()
t.join()
if "error" in box:
raise box["error"]
return box["value"]
def _reset_db(mock_port: int, seed: bool) -> ImportSummary | None:
"""Truncate the KB (and query log), then optionally re-import fixtures."""
with SessionLocal() as db:
db.execute(text("TRUNCATE chunks, documents, query_log"))
db.commit()
if not seed:
return None
return _run_in_thread(_import_fixtures(mock_port))
# --------------------------------------------------------------------------
# LLM-down fixture: stop the shared mock, restore it for later tests
# --------------------------------------------------------------------------
_replacement_mocks: list[subprocess.Popen] = []
@pytest.fixture(scope="session")
def _cleanup_replacement_mocks() -> Iterator[None]:
"""Terminate mock instances spawned to replace a stopped one."""
yield
for proc in _replacement_mocks:
proc.terminate()
try:
proc.wait(timeout=10)
except subprocess.TimeoutExpired:
proc.kill()
def _wait_http(url: str, timeout: float = 40.0) -> None:
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
try:
httpx.get(url, timeout=2.0)
return
except Exception: # noqa: BLE001 — retry until deadline
time.sleep(0.5)
raise RuntimeError(f"server at {url} did not come up")
def _spawn_mock() -> subprocess.Popen:
env = dict(os.environ)
env.pop("DEBUGPY", None)
return subprocess.Popen(
[sys.executable, "-m", "uvicorn", "tests.e2e.mock_llm:app",
"--host", "127.0.0.1", "--port", str(MOCK_PORT), "--log-level", "warning"],
cwd=REPO,
env=env,
)
@pytest.fixture()
def llm_down(mock_llm: int, _cleanup_replacement_mocks: None) -> Iterator[None]:
"""Simulate the LLM going down for one test (stop the shared mock via
its ``__shutdown__`` hook), then restore a fresh instance on the same
port so the rest of the session keeps working."""
r = httpx.post(f"http://127.0.0.1:{MOCK_PORT}/__shutdown__", timeout=10)
assert r.status_code == 200
stopped = False
deadline = time.monotonic() + 15
while time.monotonic() < deadline:
try:
httpx.get(f"http://127.0.0.1:{MOCK_PORT}/v1/models", timeout=1.0)
except Exception: # noqa: BLE001 — connection refused == stopped
stopped = True
break
time.sleep(0.2)
assert stopped, "mock LLM did not stop in time"
yield
proc = _spawn_mock()
_replacement_mocks.append(proc)
_wait_http(f"http://127.0.0.1:{MOCK_PORT}/v1/models")
# --------------------------------------------------------------------------
# Tests (story → test mapping, .agents/user_stories/loading-feedback.md)
# --------------------------------------------------------------------------
def test_typing_indicator_during_slow_think(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
"""AC1/AC5: the 3s mock warm-up must show the typing indicator for
>=2s before any text appears, then it is gone once the answer lands."""
summary = _reset_db(mock_llm, seed=True)
assert summary is not None and summary.added == 13 # A9 formats (phase 47 added quadlet+j2)
page.set_default_timeout(30_000)
login(page, app_url, next="/")
page.fill("#message-input", SLOW_QUESTION)
page.click("#send-btn")
# Visible within 500ms of submit — the indicator is added synchronously
# by setUiState("thinking").
typing = page.locator(TYPING)
expect(typing).to_be_visible(timeout=500)
bubble = typing.locator(".bubble")
expect(bubble).to_have_attribute("role", "status")
expect(bubble).to_have_attribute("aria-label", "Brain of Reese is thinking")
# ~2s in: still pre-token (the mock is in its 3s warm-up).
page.wait_for_timeout(2000)
assert typing.is_visible(), "typing indicator must persist through the pre-token window"
# First token lands (~3s): dots gone, answer text present.
answer = page.locator(ANSWER)
answer.wait_for(state="visible", timeout=30_000)
expect(typing).to_be_hidden()
assert answer.inner_text().strip(), "answer bubble must contain text"
def test_button_state_machine(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
"""AC1/AC3 (phase 48 revised contract, owner-locked 2026-08-29):
in flight the button is the enabled Stop control — "Stop" label,
.is-stop class, spinner hidden; after done it is enabled + "Send"
(class removed) + the input is focused back."""
_reset_db(mock_llm, seed=True)
page.set_default_timeout(30_000)
login(page, app_url, next="/")
page.fill("#message-input", SLOW_QUESTION)
page.click("#send-btn")
btn = page.locator("#send-btn")
label = page.locator("#send-label")
expect(btn).to_be_enabled(timeout=500)
expect(label).to_have_text("Stop")
expect(btn).to_have_class(re.compile(r"is-stop"))
expect(btn.locator(".spinner")).to_be_hidden()
expect(page.locator("#send-status")).to_contain_text("thinking")
# Done: button recovers to the Send state and the input is focused back.
page.locator(ANSWER).wait_for(state="visible", timeout=30_000)
expect(label).to_have_text("Send", timeout=30_000)
expect(btn).to_be_enabled()
expect(btn).not_to_have_class(re.compile(r"is-stop"))
expect(btn.locator(".spinner")).to_be_hidden()
expect(page.locator("#message-input")).to_be_focused()
def test_streaming_appends_live(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
"""AC2: text appends live — a sample taken mid-stream must be strictly
shorter than a later one (no 'whole answer appears at once')."""
_reset_db(mock_llm, seed=True)
page.set_default_timeout(30_000)
login(page, app_url, next="/")
page.fill("#message-input", ON_TOPIC)
page.click("#send-btn")
answer = page.locator(ANSWER)
answer.wait_for(state="visible", timeout=30_000)
first = answer.inner_text()
assert first.strip(), "bubble should carry text as soon as it appears"
second = first
deadline = time.monotonic() + 15
while time.monotonic() < deadline:
second = answer.inner_text()
if len(second) > len(first):
break
time.sleep(0.05)
assert len(second) > len(first), (
"answer text never grew after the first sample — no live streaming visible"
)
# The turn settles: button back to 'Send', final text no shorter.
expect(page.locator("#send-label")).to_have_text("Send", timeout=30_000)
final = answer.inner_text()
assert len(final) >= len(second)
def test_reduced_motion_keeps_feedback(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
"""AC7: under prefers-reduced-motion the typing dots stay visible
(static) — feedback is never removed, only calmed. Phase 48: the
in-flight button is the enabled Stop control (the spinner is no
longer part of the in-flight feedback)."""
_reset_db(mock_llm, seed=True)
page.emulate_media(reduced_motion="reduce")
page.set_default_timeout(30_000)
login(page, app_url, next="/")
page.fill("#message-input", SLOW_QUESTION)
page.click("#send-btn")
expect(page.locator(TYPING)).to_be_visible(timeout=500)
expect(page.locator(TYPING).locator(".bubble span")).to_have_count(3)
expect(page.locator("#send-btn")).to_have_class(re.compile(r"is-stop"))
expect(page.locator("#send-label")).to_have_text("Stop")
expect(page.locator("#send-btn .spinner")).to_be_hidden()
# And the turn still completes and recovers.
expect(page.locator("#send-label")).to_have_text("Send", timeout=30_000)
expect(page.locator("#send-btn")).to_be_enabled()
expect(page.locator(TYPING)).to_be_hidden()
def test_error_banner_on_llm_down(
page: Page, app_url: str, mock_llm: int, db_ready: None, llm_down: None
) -> None:
"""AC4a: with the LLM down the turn ends in the red role=alert banner
with the actionable retry hint, and the button recovers (never a
zombie)."""
page.set_default_timeout(30_000)
login(page, app_url, next="/")
page.fill("#message-input", ON_TOPIC)
page.click("#send-btn")
banner = page.locator("#kb-banner")
expect(banner).to_be_visible(timeout=30_000)
expect(banner).to_have_attribute("role", "alert")
expect(banner).to_have_class(re.compile(r"kb-banner.*is-error"))
expect(banner).to_contain_text("check the LLM is reachable")
# Recovery: button enabled, label 'Send', no lingering indicator.
expect(page.locator("#send-btn")).to_be_enabled(timeout=30_000)
expect(page.locator("#send-label")).to_have_text("Send")
expect(page.locator(TYPING)).to_be_hidden()