"""Phase 80 E2E (Playwright): onboarding chips = the last 3 questions asked. Story: ``.agents/user_stories/suggestion-chips.md`` (phase 05) — REWRITTEN in place for the phase-80 semantics (the phase-76 precedent: a semantic change rewrites the story suite in place). Source: ``TODO.md`` L6. The new contract (owner decision A6): the empty-state chip row is the 3 most recent user questions across ALL saved chats — chats walked newest-``updated_at`` first, each chat's messages newest-first, exact (case-sensitive) de-duplicated, cap 3. A fresh deployment — zero saved questions — gets the SEED list instead (``BOR_SUGGESTIONS`` / the built-in default). 1–2 saved questions → exactly those chips (NO mixing with the seed). The row refetches when the empty state comes back (New chat), so it is never stale. The deflection "Maybe try" chips are a separate contract (``derive_suggestions``) — untouched. The four states pinned here: * **seed** — fresh DB (no saved chats) → the chip texts equal the built-in default list EXACTLY (the ``SEED`` literal below is the pin for the exact seed list — ``tests/unit/test_config.py`` pins only the shape) — rendered as accessible buttons in the role=list group, exactly as the phase-05 component contract; * **last-3** — two saved chats with 5 user questions total (the older one saved FIRST — the API stamps ``updated_at``) → a fresh page load shows EXACTLY the 3 newest questions, newest-first; * **partial** — exactly 2 saved questions deployment-wide → exactly 2 chips (no seed top-up — the A6 contract, visible in the UI); * **refetch** — boot with the seed chips, save a chat whose newest question is Q via the API, click New chat (``#new-chat-btn``) → the chips now are Q, and the request log shows a SECOND ``GET /api/suggestions`` (the boot fetch was the first). Carried-over story behavior (unchanged semantics from the phase-05 suite): one-tap submit (chip click → composer filled → submitted → the mock-LLM brain bubble), Tab+Enter keyboard reachability of the chips (the keyboard-walk assertion), and the mobile single horizontal-scroll row. The endpoint is authed (phase 79, ``require_user``), so every test signs in as admin first (``auth_helpers.login``). ``saved_chats`` is global state on the shared e2e Postgres AND the state this contract reads — the autouse fixture truncates it before and after EVERY test (including the ones whose turns auto-save a row), so each test starts from — and leaves — an empty deployment. Run in isolation (DB must be up: ``podman compose up -d db``): uv run pytest tests/e2e/test_suggestion_chips.py -v --no-cov """ from __future__ import annotations import asyncio import json import time from collections.abc import Iterator from datetime import datetime from pathlib import Path from threading import Thread from typing import Any import pytest from playwright.sync_api import Browser, Page, Request, expect from sqlalchemy import text from app.config import Settings from app.db import SessionLocal from app.rag.importer import ImportSummary, import_sources from app.rag.llm import LLMClient from e2e.auth_helpers import login REPO = Path(__file__).resolve().parents[2] FIXTURES = REPO / "tests" / "fixtures" / "docs" MOCK_ANSWER_MARKER = "Deterministic mock answer for E2E" #: The EXACT built-in onboarding SEED (phase 80, TODO.md L6): the chip #: row of a brand-new deployment, shown only before any question has #: ever been saved. This literal is the E2E pin for the exact seed #: list — ``tests/unit/test_config.py`` pins only the SHAPE (>=3 #: non-blank distinct strings), and the e2e app under test is forced #: to the code default by conftest's leak guard — keep in sync with #: the ``Settings.suggestions`` default in ``app/config.py``. SEED: list[str] = [ "What documents are in the knowledge base?", "Which source does each answer come from?", "How do I add a new source?", "Summarize the most recent document.", ] @pytest.fixture(autouse=True) def clean_chats(db_ready: None) -> Iterator[None]: """``saved_chats`` is the state the phase-80 contract reads: truncate it before and after every test so each state test starts from (and leaves) an empty deployment. Unlike the KB tables, this reset is non-optional — the chips ARE these rows, and the carried-over submit tests auto-save a row per turn, which would otherwise leak into the later state tests.""" with SessionLocal() as db: db.execute(text("TRUNCATE saved_chats")) db.commit() yield with SessionLocal() as db: db.execute(text("TRUNCATE saved_chats")) db.commit() async def _import_fixtures(mock_port: int) -> ImportSummary: kwargs: dict[str, Any] = {"_env_file": None, "llm_base_url": f"http://127.0.0.1:{mock_port}/v1"} settings = Settings(**kwargs) # pyright: ignore[reportCallIssue] return await import_sources([FIXTURES], LLMClient(settings)) def _run_in_thread(coro: Any) -> Any: """Run a coroutine on a worker thread (Playwright owns the test loop).""" box: dict[str, Any] = {} def runner() -> None: try: box["value"] = asyncio.run(coro) except BaseException as e: # noqa: BLE001 — re-raised on the test thread box["error"] = e t = Thread(target=runner) t.start() t.join() if "error" in box: raise box["error"] return box["value"] def _seed_kb(mock_port: int) -> ImportSummary: """Deterministic KB: truncate the KB tables, import the fixture docs (needed by the carried-over submit tests' grounded answers). ``saved_chats`` is the autouse fixture's job.""" with SessionLocal() as db: db.execute(text("TRUNCATE chunks, documents, query_log")) db.commit() summary = _run_in_thread(_import_fixtures(mock_port)) assert summary is not None and summary.added == 13 # A9 formats (phase 47 added quadlet+j2) return summary def _user(q: str) -> dict[str, Any]: return {"who": "user", "text": q} def _brain(text: str = "Grounded mock brain reply.") -> dict[str, Any]: return {"who": "brain", "text": text} def _save_chat(page: Page, app_url: str, messages: list[dict[str, Any]]) -> dict[str, Any]: """Save one conversation as the signed-in admin (``POST /api/chats``) and return the 201 body. ``updated_at`` is the API's stamp (server ``now()`` at INSERT) — the save ORDER is what makes the chip-walk order deterministic in the state tests.""" r = page.request.post( f"{app_url}/api/chats", data=json.dumps({"messages": messages}), headers={"Content-Type": "application/json"}, timeout=10_000, ) assert r.status == 201, r.text return r.json() def _api_suggestions(page: Page, app_url: str) -> list[str]: # Phase 79: the endpoint is require_user-gated — the request rides # the page's signed-in context (each test signs in above). r = page.request.get(f"{app_url}/api/suggestions", timeout=10) assert r.status == 200, r.text return r.json()["suggestions"] def _chip_locator(page: Page) -> Any: return page.locator("#suggestions .suggestion-chip") def _chip_texts(page: Page) -> list[str]: return [c.strip() for c in _chip_locator(page).all_inner_texts()] def test_seed_state_chips_are_the_builtin_default( page: Page, app_url: str, db_ready: None ) -> None: """Fresh deployment (no saved chats) → the chip row is EXACTLY the built-in seed list — texts, count, and order — rendered in ``#suggestions`` (role="list") as accessible buttons, exactly as the phase-05 component contract.""" page.set_default_timeout(30_000) login(page, app_url, next="/") # Accessible group: role=list + a name screen readers can announce. group = page.locator("#suggestions") expect(group).to_have_attribute("role", "list") expect(group).to_have_attribute("aria-label", "Suggested questions") chips = _chip_locator(page) expect(chips.first).to_be_visible(timeout=30_000) # The EXACT seed list, in order (the E2E pin — see the SEED literal). assert _chip_texts(page) == SEED # ...drawn from the endpoint (the API returns the same exact list). assert _api_suggestions(page, app_url) == SEED # Real buttons, each with non-empty text, one per seed entry. assert chips.count() == len(SEED) for i in range(chips.count()): chip = chips.nth(i) expect(chip).to_be_visible() expect(chip).to_have_attribute("type", "button") expect(chip).to_have_attribute("role", "listitem") assert chip.inner_text().strip(), "every chip needs non-empty label text" # Chips live in the empty state, which is visible before any message. expect(page.locator("#empty-state")).to_be_visible() # Chip component contract: brand pill, >=44px touch target. style = chips.first.evaluate("el => getComputedStyle(el)") assert style["backgroundColor"] == "rgb(45, 10, 10)" # --brand-soft #2d0a0a (dark-red rebrand) assert style["color"] == "rgb(252, 165, 165)" # --brand-ink #fca5a5 (dark-red rebrand) assert style["borderRadius"] == "999px" box = chips.first.bounding_box() assert box is not None and box["height"] >= 44 def test_last_three_questions_state(page: Page, app_url: str, db_ready: None) -> None: """5 user questions across two saved chats (the older one saved FIRST — the API stamps ``updated_at`` at INSERT) → a fresh page load shows EXACTLY the 3 newest questions, newest-first: the newer chat is walked first, then the older chat newest-first.""" page.set_default_timeout(30_000) login(page, app_url, next="/") old_q = [ "How did I install the GitLab runner on the Proxmox node?", "Which disk holds the Borg backup archives?", "How is the nftables firewall rule set ordered?", ] new_q = [ "What TLS termination does Traefik do for homelab.local?", "Which provider is the primary DNS for reeseapps.com?", ] # The OLDER chat first: the API stamps ``updated_at`` (server # now()), so save order IS walk order. The short pause keeps the # two stamps strictly apart (and the assert below pins that the # order the walk sees is the order the test intended). older = _save_chat( page, app_url, [ _user(old_q[0]), _brain(), _user(old_q[1]), _brain(), _user(old_q[2]), _brain(), ], ) time.sleep(0.05) newer = _save_chat( page, app_url, [ _user(new_q[0]), _brain(), _user(new_q[1]), _brain(), ], ) assert datetime.fromisoformat(newer["updated_at"]) > datetime.fromisoformat( older["updated_at"] ), "the two API-stamped updated_at values must be strictly ordered" # A FRESH page load (a new boot fetch, not the pre-save boot): # the chips are exactly the 3 newest questions, newest first. page.goto(app_url + "/") chips = _chip_locator(page) expect(chips.first).to_be_visible(timeout=30_000) expected = [new_q[1], new_q[0], old_q[2]] assert _chip_texts(page) == expected assert _api_suggestions(page, app_url) == expected # The two older questions (and everything seed-shaped) are gone. assert old_q[0] not in _chip_texts(page) assert old_q[1] not in _chip_texts(page) def test_partial_state_no_seed_topup(page: Page, app_url: str, db_ready: None) -> None: """Exactly 2 saved questions deployment-wide → EXACTLY 2 chips (newest first) — NO mixing/top-up with the seed (the A6 contract, visible in the UI).""" page.set_default_timeout(30_000) login(page, app_url, next="/") a = "How do I rotate the WireGuard keys on the VPN node?" b = "What cron schedule runs the restic prune?" _save_chat(page, app_url, [_user(a), _brain()]) time.sleep(0.05) _save_chat(page, app_url, [_user(b), _brain()]) page.goto(app_url + "/") chips = _chip_locator(page) expect(chips.first).to_be_visible(timeout=30_000) assert chips.count() == 2, "exactly 2 chips — the row is never padded toward 3" texts = _chip_texts(page) assert texts == [b, a] assert not (set(texts) & set(SEED)), "no seed text may appear once a question is saved" def test_new_chat_refetches_the_chips(page: Page, app_url: str, db_ready: None) -> None: """The row is never stale: boot with the seed chips → save a chat whose newest question is Q via the API → click New chat (``#new-chat-btn``) → the empty state comes back with the REFETCHED row (exactly Q — the deployment now has one saved question), and the request log shows a SECOND ``GET /api/suggestions`` (the boot fetch was the first).""" page.set_default_timeout(30_000) sugg_gets: list[float] = [] def on_request(req: Request) -> None: if req.url.endswith("/api/suggestions"): sugg_gets.append(time.monotonic()) page.on("request", on_request) login(page, app_url, next="/") chips = _chip_locator(page) expect(chips.first).to_be_visible(timeout=30_000) assert _chip_texts(page) == SEED, "boot state: the seed row" assert len(sugg_gets) == 1, "exactly one GET /api/suggestions at boot" q = "Which service fronts the Pi-hole DNS on the network?" _save_chat(page, app_url, [_user(q), _brain()]) clicked_at = time.monotonic() page.click("#new-chat-btn") # The refetch re-renders #suggestions in place: the 4 seed chips # are replaced by exactly Q (the partial state, live). expect(chips).to_have_count(1, timeout=15_000) expect(chips.first).to_have_text(q, timeout=15_000) assert len(sugg_gets) == 2, "New chat triggered the refetch" assert sugg_gets[1] > clicked_at, "the second GET is AFTER the click — the refetch" def test_chip_click_submits( page: Page, app_url: str, mock_llm: int, db_ready: None ) -> None: """Carried over (phase 05, unchanged semantics): one tap = one question — the click fills AND submits; a grounded mock reply follows. The chip submitted is the SEED row's first entry (the autouse fixture guarantees the seed state).""" _seed_kb(mock_llm) page.set_default_timeout(30_000) login(page, app_url, next="/") first = _chip_locator(page).first expect(first).to_be_visible(timeout=30_000) chip_text = first.inner_text().strip() assert chip_text == SEED[0] # One tap = one question: the click fills AND submits. first.click() expect(page.locator("#empty-state")).to_be_hidden() expect(page.locator("#message-input")).to_have_value("") # submitted, not queued expect(page.locator(".msg.user .bubble")).to_have_count(1, timeout=30_000) expect(page.locator(".msg.user .bubble")).to_have_text(chip_text) # A grounded mock reply follows (the seeded KB answers this topic). brain = page.locator(".msg.brain .bubble").first expect(brain).to_contain_text(MOCK_ANSWER_MARKER, timeout=30_000) expect(brain).to_contain_text(chip_text) expect(page.locator(".msg.brain.is-deflected")).to_have_count(0) # Never stale: the send button recovers after the turn. expect(page.locator("#send-btn")).to_be_enabled() expect(page.locator("#send-label")).to_have_text("Send") def test_chips_keyboard_accessible( page: Page, app_url: str, mock_llm: int, db_ready: None ) -> None: """Carried over (phase 05, unchanged semantics): the first chip is keyboard-reachable BEFORE the composer input (skip-link + nav links come first), and Enter activates it — submitting.""" _seed_kb(mock_llm) page.set_default_timeout(30_000) login(page, app_url, next="/") first = _chip_locator(page).first expect(first).to_be_visible(timeout=30_000) chip_text = first.inner_text().strip() assert chip_text # Tab from the page start: the first chip must be reachable on the # keyboard, and before the composer input (skip-link + 2 nav links come # first). Track tab stops until we land on a chip. reached_chip_at: int | None = None for step in range(1, 11): page.keyboard.press("Tab") state = page.evaluate( """() => { const el = document.activeElement; return { id: el ? el.id : "", isChip: !!(el && el.classList && el.classList.contains("suggestion-chip") && el.closest("#suggestions")), }; }""" ) if state["id"] == "message-input": pytest.fail("the composer input was reached before the suggestion chips") if state["isChip"]: reached_chip_at = step break assert reached_chip_at is not None, "no suggestion chip is keyboard-reachable" expect(first).to_be_focused() # Enter activates the focused chip button → it submits. page.keyboard.press("Enter") expect(page.locator("#empty-state")).to_be_hidden() expect(page.locator("#message-input")).to_have_value("") expect(page.locator(".msg.user .bubble")).to_have_count(1, timeout=30_000) expect(page.locator(".msg.user .bubble")).to_have_text(chip_text) expect(page.locator(".msg.brain .bubble").first).to_contain_text( MOCK_ANSWER_MARKER, timeout=30_000 ) expect(page.locator("#send-btn")).to_be_enabled() def test_chips_mobile_row( browser: Browser, app_url: str, db_ready: None ) -> None: """Carried over (phase 05, unchanged semantics): on mobile (375px) the row is a single horizontally scrollable line — the seed state (4 chips) overflows into scroll, nothing wraps, chips stay >=44px tall on one line.""" page = browser.new_page(viewport={"width": 375, "height": 720}) try: page.set_default_timeout(30_000) login(page, app_url, next="/") row = page.locator("#suggestions") expect(row).to_be_visible(timeout=30_000) # The row is a single line that scrolls horizontally: content is # wider than the viewport, the container scrolls, nothing wraps. wrap = row.evaluate("el => getComputedStyle(el)") assert wrap["flexWrap"] == "nowrap" assert wrap["overflowX"] in {"auto", "scroll"} dims = row.evaluate( "el => ({ sw: el.scrollWidth, cw: el.clientWidth, h: el.clientHeight })" ) assert dims["sw"] > dims["cw"], "chips must overflow into a scroll row" assert row.evaluate("el => { el.scrollLeft = 24; return el.scrollLeft; }") > 0 # Exactly one line: every chip shares the same top edge, and the # line height fits a single 44px-tall chip (no vertical clipping). chips = _chip_locator(page) assert chips.count() >= 3 tops: list[float] = [] for i in range(chips.count()): box = chips.nth(i).bounding_box() assert box is not None assert box["height"] >= 44, "chips stay >=44px tall on mobile" tops.append(box["y"]) assert max(tops) - min(tops) < 0.5, "all chips sit on one horizontal line" row_box = row.bounding_box() assert row_box is not None assert row_box["height"] < 2 * 44, "the mobile row is a single line tall" finally: page.close()