321 lines
12 KiB
Python
321 lines
12 KiB
Python
"""Phase 06 E2E (Playwright): loading feedback & progress.
|
|
|
|
Story: ``.agent/user_stories/loading-feedback.md``
|
|
Run in isolation (DB must be up: ``podman compose up -d db``):
|
|
|
|
uv run pytest tests/e2e/test_loading_feedback.py -v --no-cov
|
|
|
|
Determinism comes from two mock-LLM behaviors (tests/e2e/mock_llm.py):
|
|
|
|
* ``pretend to think slowly`` in the user message → a 3s warm-up before
|
|
the first token, wide enough to assert the pre-token UI (typing dots +
|
|
the enabled Stop button — phase 48 revised the busy-button contract) at
|
|
a known timestamp;
|
|
* the ``POST /__shutdown__`` hook → the ``llm_down`` fixture stops the
|
|
shared mock to simulate an LLM outage, then restores a fresh instance
|
|
on the same port so later tests keep working.
|
|
|
|
The state machine under test lives in frontend/assets/app.js:
|
|
``idle → thinking → streaming → done | error → idle`` (PLAN §7.4).
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import os
|
|
import re
|
|
import subprocess
|
|
import sys
|
|
import time
|
|
from collections.abc import Iterator
|
|
from pathlib import Path
|
|
from threading import Thread
|
|
from typing import Any
|
|
|
|
import httpx
|
|
import pytest
|
|
from playwright.sync_api import Page, expect
|
|
from sqlalchemy import text
|
|
|
|
from app.config import Settings
|
|
from app.db import SessionLocal
|
|
from app.rag.importer import ImportSummary, import_sources
|
|
from app.rag.llm import LLMClient
|
|
|
|
REPO = Path(__file__).resolve().parents[2]
|
|
FIXTURES = REPO / "tests" / "fixtures" / "docs"
|
|
MOCK_PORT = int(os.environ.get("E2E_MOCK_PORT", "8901"))
|
|
|
|
# The mock warms up for 3s when the question contains this marker, so the
|
|
# pre-token window is observable. (Retrieval may answer or deflect — the
|
|
# feedback states are identical either way.)
|
|
SLOW_QUESTION = "pretend to think slowly then tell me about kubernetes"
|
|
ON_TOPIC = "How is my Kubernetes cluster set up?"
|
|
|
|
TYPING = "#typing-indicator"
|
|
# The typing indicator is itself a .msg.brain — exclude its bubble.
|
|
ANSWER = ".msg.brain .bubble:not(.typing)"
|
|
|
|
|
|
# --------------------------------------------------------------------------
|
|
# KB seeding (same pattern as the phase 02/03 story suites)
|
|
# --------------------------------------------------------------------------
|
|
|
|
|
|
async def _import_fixtures(mock_port: int) -> ImportSummary:
|
|
kwargs: dict[str, Any] = {"_env_file": None, "llm_base_url": f"http://127.0.0.1:{mock_port}/v1"}
|
|
settings = Settings(**kwargs) # pyright: ignore[reportCallIssue]
|
|
return await import_sources([FIXTURES], LLMClient(settings))
|
|
|
|
|
|
def _run_in_thread(coro: Any) -> Any:
|
|
"""Run a coroutine on a worker thread.
|
|
|
|
Playwright's sync API keeps an asyncio loop running on the test thread,
|
|
so ``asyncio.run`` cannot be called directly from a test body.
|
|
"""
|
|
box: dict[str, Any] = {}
|
|
|
|
def runner() -> None:
|
|
try:
|
|
box["value"] = asyncio.run(coro)
|
|
except BaseException as e: # noqa: BLE001 — re-raised on the test thread
|
|
box["error"] = e
|
|
|
|
t = Thread(target=runner)
|
|
t.start()
|
|
t.join()
|
|
if "error" in box:
|
|
raise box["error"]
|
|
return box["value"]
|
|
|
|
|
|
def _reset_db(mock_port: int, seed: bool) -> ImportSummary | None:
|
|
"""Truncate the KB (and query log), then optionally re-import fixtures."""
|
|
with SessionLocal() as db:
|
|
db.execute(text("TRUNCATE chunks, documents, query_log"))
|
|
db.commit()
|
|
if not seed:
|
|
return None
|
|
return _run_in_thread(_import_fixtures(mock_port))
|
|
|
|
|
|
# --------------------------------------------------------------------------
|
|
# LLM-down fixture: stop the shared mock, restore it for later tests
|
|
# --------------------------------------------------------------------------
|
|
|
|
_replacement_mocks: list[subprocess.Popen] = []
|
|
|
|
|
|
@pytest.fixture(scope="session")
|
|
def _cleanup_replacement_mocks() -> Iterator[None]:
|
|
"""Terminate mock instances spawned to replace a stopped one."""
|
|
yield
|
|
for proc in _replacement_mocks:
|
|
proc.terminate()
|
|
try:
|
|
proc.wait(timeout=10)
|
|
except subprocess.TimeoutExpired:
|
|
proc.kill()
|
|
|
|
|
|
def _wait_http(url: str, timeout: float = 40.0) -> None:
|
|
deadline = time.monotonic() + timeout
|
|
while time.monotonic() < deadline:
|
|
try:
|
|
httpx.get(url, timeout=2.0)
|
|
return
|
|
except Exception: # noqa: BLE001 — retry until deadline
|
|
time.sleep(0.5)
|
|
raise RuntimeError(f"server at {url} did not come up")
|
|
|
|
|
|
def _spawn_mock() -> subprocess.Popen:
|
|
env = dict(os.environ)
|
|
env.pop("DEBUGPY", None)
|
|
return subprocess.Popen(
|
|
[sys.executable, "-m", "uvicorn", "tests.e2e.mock_llm:app",
|
|
"--host", "127.0.0.1", "--port", str(MOCK_PORT), "--log-level", "warning"],
|
|
cwd=REPO,
|
|
env=env,
|
|
)
|
|
|
|
|
|
@pytest.fixture()
|
|
def llm_down(mock_llm: int, _cleanup_replacement_mocks: None) -> Iterator[None]:
|
|
"""Simulate the LLM going down for one test (stop the shared mock via
|
|
its ``__shutdown__`` hook), then restore a fresh instance on the same
|
|
port so the rest of the session keeps working."""
|
|
r = httpx.post(f"http://127.0.0.1:{MOCK_PORT}/__shutdown__", timeout=10)
|
|
assert r.status_code == 200
|
|
|
|
stopped = False
|
|
deadline = time.monotonic() + 15
|
|
while time.monotonic() < deadline:
|
|
try:
|
|
httpx.get(f"http://127.0.0.1:{MOCK_PORT}/v1/models", timeout=1.0)
|
|
except Exception: # noqa: BLE001 — connection refused == stopped
|
|
stopped = True
|
|
break
|
|
time.sleep(0.2)
|
|
assert stopped, "mock LLM did not stop in time"
|
|
|
|
yield
|
|
|
|
proc = _spawn_mock()
|
|
_replacement_mocks.append(proc)
|
|
_wait_http(f"http://127.0.0.1:{MOCK_PORT}/v1/models")
|
|
|
|
|
|
# --------------------------------------------------------------------------
|
|
# Tests (story → test mapping, .agent/user_stories/loading-feedback.md)
|
|
# --------------------------------------------------------------------------
|
|
|
|
|
|
def test_typing_indicator_during_slow_think(
|
|
page: Page, app_url: str, mock_llm: int, db_ready: None
|
|
) -> None:
|
|
"""AC1/AC5: the 3s mock warm-up must show the typing indicator for
|
|
>=2s before any text appears, then it is gone once the answer lands."""
|
|
summary = _reset_db(mock_llm, seed=True)
|
|
assert summary is not None and summary.added == 13 # A9 formats (phase 47 added quadlet+j2)
|
|
page.set_default_timeout(30_000)
|
|
page.goto(app_url)
|
|
|
|
page.fill("#message-input", SLOW_QUESTION)
|
|
page.click("#send-btn")
|
|
|
|
# Visible within 500ms of submit — the indicator is added synchronously
|
|
# by setUiState("thinking").
|
|
typing = page.locator(TYPING)
|
|
expect(typing).to_be_visible(timeout=500)
|
|
bubble = typing.locator(".bubble")
|
|
expect(bubble).to_have_attribute("role", "status")
|
|
expect(bubble).to_have_attribute("aria-label", "Brain of Reese is thinking")
|
|
|
|
# ~2s in: still pre-token (the mock is in its 3s warm-up).
|
|
page.wait_for_timeout(2000)
|
|
assert typing.is_visible(), "typing indicator must persist through the pre-token window"
|
|
|
|
# First token lands (~3s): dots gone, answer text present.
|
|
answer = page.locator(ANSWER)
|
|
answer.wait_for(state="visible", timeout=30_000)
|
|
expect(typing).to_be_hidden()
|
|
assert answer.inner_text().strip(), "answer bubble must contain text"
|
|
|
|
|
|
def test_button_state_machine(
|
|
page: Page, app_url: str, mock_llm: int, db_ready: None
|
|
) -> None:
|
|
"""AC1/AC3 (phase 48 revised contract, owner-locked 2026-08-29):
|
|
in flight the button is the enabled Stop control — "Stop" label,
|
|
.is-stop class, spinner hidden; after done it is enabled + "Send"
|
|
(class removed) + the input is focused back."""
|
|
_reset_db(mock_llm, seed=True)
|
|
page.set_default_timeout(30_000)
|
|
page.goto(app_url)
|
|
|
|
page.fill("#message-input", SLOW_QUESTION)
|
|
page.click("#send-btn")
|
|
|
|
btn = page.locator("#send-btn")
|
|
label = page.locator("#send-label")
|
|
expect(btn).to_be_enabled(timeout=500)
|
|
expect(label).to_have_text("Stop")
|
|
expect(btn).to_have_class(re.compile(r"is-stop"))
|
|
expect(btn.locator(".spinner")).to_be_hidden()
|
|
expect(page.locator("#send-status")).to_contain_text("thinking")
|
|
|
|
# Done: button recovers to the Send state and the input is focused back.
|
|
page.locator(ANSWER).wait_for(state="visible", timeout=30_000)
|
|
expect(label).to_have_text("Send", timeout=30_000)
|
|
expect(btn).to_be_enabled()
|
|
expect(btn).not_to_have_class(re.compile(r"is-stop"))
|
|
expect(btn.locator(".spinner")).to_be_hidden()
|
|
expect(page.locator("#message-input")).to_be_focused()
|
|
|
|
|
|
def test_streaming_appends_live(
|
|
page: Page, app_url: str, mock_llm: int, db_ready: None
|
|
) -> None:
|
|
"""AC2: text appends live — a sample taken mid-stream must be strictly
|
|
shorter than a later one (no 'whole answer appears at once')."""
|
|
_reset_db(mock_llm, seed=True)
|
|
page.set_default_timeout(30_000)
|
|
page.goto(app_url)
|
|
|
|
page.fill("#message-input", ON_TOPIC)
|
|
page.click("#send-btn")
|
|
|
|
answer = page.locator(ANSWER)
|
|
answer.wait_for(state="visible", timeout=30_000)
|
|
first = answer.inner_text()
|
|
assert first.strip(), "bubble should carry text as soon as it appears"
|
|
|
|
second = first
|
|
deadline = time.monotonic() + 15
|
|
while time.monotonic() < deadline:
|
|
second = answer.inner_text()
|
|
if len(second) > len(first):
|
|
break
|
|
time.sleep(0.05)
|
|
assert len(second) > len(first), (
|
|
"answer text never grew after the first sample — no live streaming visible"
|
|
)
|
|
|
|
# The turn settles: button back to 'Send', final text no shorter.
|
|
expect(page.locator("#send-label")).to_have_text("Send", timeout=30_000)
|
|
final = answer.inner_text()
|
|
assert len(final) >= len(second)
|
|
|
|
|
|
def test_reduced_motion_keeps_feedback(
|
|
page: Page, app_url: str, mock_llm: int, db_ready: None
|
|
) -> None:
|
|
"""AC7: under prefers-reduced-motion the typing dots stay visible
|
|
(static) — feedback is never removed, only calmed. Phase 48: the
|
|
in-flight button is the enabled Stop control (the spinner is no
|
|
longer part of the in-flight feedback)."""
|
|
_reset_db(mock_llm, seed=True)
|
|
page.emulate_media(reduced_motion="reduce")
|
|
page.set_default_timeout(30_000)
|
|
page.goto(app_url)
|
|
|
|
page.fill("#message-input", SLOW_QUESTION)
|
|
page.click("#send-btn")
|
|
|
|
expect(page.locator(TYPING)).to_be_visible(timeout=500)
|
|
expect(page.locator(TYPING).locator(".bubble span")).to_have_count(3)
|
|
expect(page.locator("#send-btn")).to_have_class(re.compile(r"is-stop"))
|
|
expect(page.locator("#send-label")).to_have_text("Stop")
|
|
expect(page.locator("#send-btn .spinner")).to_be_hidden()
|
|
|
|
# And the turn still completes and recovers.
|
|
expect(page.locator("#send-label")).to_have_text("Send", timeout=30_000)
|
|
expect(page.locator("#send-btn")).to_be_enabled()
|
|
expect(page.locator(TYPING)).to_be_hidden()
|
|
|
|
|
|
def test_error_banner_on_llm_down(
|
|
page: Page, app_url: str, mock_llm: int, db_ready: None, llm_down: None
|
|
) -> None:
|
|
"""AC4a: with the LLM down the turn ends in the red role=alert banner
|
|
with the actionable retry hint, and the button recovers (never a
|
|
zombie)."""
|
|
page.set_default_timeout(30_000)
|
|
page.goto(app_url)
|
|
|
|
page.fill("#message-input", ON_TOPIC)
|
|
page.click("#send-btn")
|
|
|
|
banner = page.locator("#kb-banner")
|
|
expect(banner).to_be_visible(timeout=30_000)
|
|
expect(banner).to_have_attribute("role", "alert")
|
|
expect(banner).to_have_class(re.compile(r"kb-banner.*is-error"))
|
|
expect(banner).to_contain_text("Try again")
|
|
expect(banner).to_contain_text("check the LLM is reachable")
|
|
|
|
# Recovery: button enabled, label 'Send', no lingering indicator.
|
|
expect(page.locator("#send-btn")).to_be_enabled(timeout=30_000)
|
|
expect(page.locator("#send-label")).to_have_text("Send")
|
|
expect(page.locator(TYPING)).to_be_hidden()
|