"""Phase 42 E2E (Playwright, mock-only): the chat NEVER autoscrolls. Story: ``.agent/user_stories/no-reply-autoscroll.md`` — owner direction 2026-08-27 (TODO.md L5) revising the phase-18 follow-the-bottom choice: "Get rid of the chat reply autoscroll, it's breaking things like making it impossible for the user to scroll while a reply generates." Run in isolation (DB must be up: ``podman compose up -d db``): uv run pytest tests/e2e/test_no_reply_autoscroll.py -v --no-cov This is the INVERSE of the phase-18 contract: while a turn streams (thinking, tool, or answer frames), nothing moves the viewport — a user reading earlier content stays exactly where they put it for the rest of the turn. The only scrolls left in the app are user intent: the submit (the user's own message is revealed) and the phase-14 restore landing (one-shot, load-time). The phase-18 suite (``tests/e2e/test_follow_bottom_scroll.py``) is deleted with this one — its behavior was intentionally removed. MOCK-ONLY suite: the scenarios key off the deterministic mock's ``write a long answer`` trigger (~900 words ≈ 8–11s of streaming — a wide, reliable window to scroll away in) and the phase-17 ``think out loud`` trigger (~4.5s reasoning stream). ``E2E_REAL_LLM=1`` would make the scroll-away windows unpredictable, so it is not supported here. Measurement convention: the scroller is the DOCUMENT — there is no inner scroll container. ``window.scrollY`` is read via ``page.evaluate``; "stable" means every sample is within 1px of every other sample (the story's tolerance). The stylesheet sets no ``scroll-behavior``, so ``window.scrollTo(0, y)`` is instant — the recorded position is the exact position the stream must not move. Test → story mapping (Playwright Mapping Rule): 1. ``test_no_autoscroll_during_long_answer`` 2. ``test_no_autoscroll_during_thinking`` 3. ``test_submit_reveals_user_message`` 4. ``test_restore_landing_one_shot`` 5. ``test_answer_content_intact`` """ from __future__ import annotations import asyncio import json import time from collections.abc import Iterator from pathlib import Path from threading import Thread from typing import Any import pytest from playwright.sync_api import Page, expect from sqlalchemy import text from app.config import Settings from app.db import SessionLocal from app.rag.importer import ImportSummary, import_sources from app.rag.llm import LLMClient REPO = Path(__file__).resolve().parents[2] FIXTURES = REPO / "tests" / "fixtures" / "docs" #: Mock long-answer trigger (~900 words ≈ 8–11s of streaming at the mock's #: 0.02s/frame pace) — the wide, deterministic window to scroll away in. LONG_QUESTION = "write a long answer about my kubernetes cluster" #: Phase-17 thinking trigger: a ~4.5s reasoning stream, then a short #: grounded mock answer (both fire independently of the long trigger). THINKING_QUESTION = "think out loud about my kubernetes cluster" #: Short grounded question (phase-14 marker answer). SHORT_QUESTION = "How is my Kubernetes cluster set up?" #: The mock long answer's unique final line (mock_llm.LONG_ANSWER_END) — #: proves the whole stream landed even while the viewport was up. LONG_ANSWER_END = "LONG-ANSWER-END" #: Line fragment the mock's deterministic scratchpad carries #: (mock_llm.compose_thinking) — same key phase 17's suite uses. THINKING_FRAGMENT = "Step 2: Check my notes" MOCK_ANSWER_MARKER = "Deterministic mock answer for E2E" STORAGE_KEY = "bor.chat.v1" #: The story's stability tolerance: the viewport must not move more than #: 1px while a turn streams with the user scrolled away. STABLE_PX = 1 async def _import_fixtures(mock_port: int) -> ImportSummary: kwargs: dict[str, Any] = {"_env_file": None, "llm_base_url": f"http://127.0.0.1:{mock_port}/v1"} settings = Settings(**kwargs) # pyright: ignore[reportCallIssue] return await import_sources([FIXTURES], LLMClient(settings)) def _run_in_thread(coro: Any) -> Any: """Run a coroutine on a worker thread. Playwright's sync API keeps an asyncio loop running on the test thread, so ``asyncio.run`` cannot be called directly from a test body. """ box: dict[str, Any] = {} def runner() -> None: try: box["value"] = asyncio.run(coro) except BaseException as e: # noqa: BLE001 — re-raised on the test thread box["error"] = e t = Thread(target=runner) t.start() t.join() if "error" in box: raise box["error"] return box["value"] def _reset_db(mock_port: int, seed: bool) -> ImportSummary | None: """Truncate the KB (and query log), then optionally re-import fixtures.""" with SessionLocal() as db: db.execute(text("TRUNCATE chunks, documents, query_log")) db.commit() if not seed: return None return _run_in_thread(_import_fixtures(mock_port)) @pytest.fixture() def seeded_kb(mock_llm: int, db_ready: None) -> Iterator[None]: """A fresh KB seeded from ``tests/fixtures/docs`` (13 docs since phase 47, A9 formats), truncated again on teardown. ``db_ready`` (conftest) skips with clear instructions when Postgres is down.""" summary = _reset_db(mock_llm, seed=True) assert summary is not None and summary.added == 13 yield _reset_db(mock_llm, seed=False) # --------------------------------------------------------------------------- # Measurement + flow helpers # --------------------------------------------------------------------------- def scroll_state(page: Page) -> dict[str, float]: """The document scroller's state (there is no inner scroll container).""" return page.evaluate( "() => ({ y: window.scrollY, " "sh: document.documentElement.scrollHeight, " "ch: window.innerHeight })" ) def brain_bubble_longer_than(n: int, min_bubbles: int = 1) -> str: """JS predicate: there are at least ``min_bubbles`` brain bubbles and the LAST one's rendered text is > n chars (i.e. that far into the stream). ``min_bubbles=2`` guards a second turn: before its first delta, ``.last`` would still be the PREVIOUS turn's bubble.""" return ( "() => { const els = document.querySelectorAll('.msg.brain .bubble');" f" return els.length >= {min_bubbles} && els[els.length - 1].innerText.length > {n}; }}" ) def wait_settled(page: Page) -> None: """The turn is over: the never-stale contract re-enabled the button.""" expect(page.locator("#send-btn")).to_be_enabled(timeout=30_000) expect(page.locator("#send-label")).to_have_text("Send") def wait_scroll_still(page: Page, timeout: float = 10.0) -> float: """window.scrollY once the viewport has stopped moving (two consecutive reads within STABLE_PX). The submit's smooth reveal and the restore landing's smooth scroll are the only animations left — both settle this way before any stream measurement begins.""" deadline = time.monotonic() + timeout prev: float | None = None while True: y = scroll_state(page)["y"] if prev is not None and abs(y - prev) <= STABLE_PX: return y prev = y if time.monotonic() >= deadline: raise AssertionError("the viewport did not settle within timeout") time.sleep(0.25) def submit(page: Page, question: str) -> None: """Submit through the composer (the real-user flow). Playwright's fill/click scroll the composer into view first — a user-initiated move, never an app scroll.""" page.fill("#message-input", question) page.click("#send-btn") expect(page.locator(".msg.user .bubble").last).to_contain_text(question) def submit_from_top(page: Page, question: str) -> None: """Submit with the viewport where the user left it (the very top). ``page.fill``/``page.click`` would scroll the composer into view first — which IS the viewport move under test — so the send goes through the page's own DOM: set the value, fire ``input`` (autoGrow), click the submit button. A JS click never scrolls the page.""" page.evaluate( """(q) => { const input = document.querySelector('#message-input'); input.value = q; input.dispatchEvent(new Event('input', { bubbles: true })); document.querySelector('#send-btn').click(); }""", question, ) expect(page.locator(".msg.user .bubble").last).to_contain_text(question) def user_message_in_view(page: Page) -> bool: """The LAST user message's box is fully inside the viewport (the submit reveal aligns it to the bottom edge).""" box = page.locator(".msg.user").last.bounding_box() if box is None: return False ch = scroll_state(page)["ch"] return box["y"] >= -1 and box["y"] + box["height"] <= ch + 1 # --------------------------------------------------------------------------- # 1. No autoscroll: scrolled up mid-ANSWER — the viewport holds for the # rest of the turn (the answer finishes off-screen below, by design) # --------------------------------------------------------------------------- def test_no_autoscroll_during_long_answer( page: Page, app_url: str, seeded_kb: None ) -> None: page.set_default_timeout(30_000) page.goto(app_url) # Turn 1 (settled) makes the document overflow the 800px viewport. submit(page, LONG_QUESTION) wait_settled(page) state = scroll_state(page) assert state["sh"] > state["ch"], "a long answer must make the document scrollable" # Turn 2: the same long answer. The submit reveals the user's message # (the one kept app scroll) — let that smooth reveal settle first. submit(page, LONG_QUESTION) page.wait_for_function(brain_bubble_longer_than(200, min_bubbles=2), timeout=30_000) y0 = wait_scroll_still(page) # The user scrolls UP ~2× the answer's current height to read earlier # context while the stream is still running. box = page.locator(".msg.brain").last.bounding_box() assert box is not None target = max(0.0, y0 - 2 * box["height"]) assert target <= y0 - STABLE_PX, "the scroll-up must actually move the viewport" page.evaluate("y => window.scrollTo(0, y)", target) assert abs(scroll_state(page)["y"] - target) <= STABLE_PX # Sample the viewport across the rest of the stream ... samples: list[float] = [] mid_stream = 0 deadline = time.monotonic() + 40 while time.monotonic() < deadline: time.sleep(0.25) samples.append(scroll_state(page)["y"]) if not page.locator("#send-btn").is_enabled(): mid_stream += 1 if len(samples) >= 10 and page.locator("#send-btn").is_enabled(): # ... and a few more AFTER `done` (the turn is over; nothing # queued behind the stream may move the page either). for _ in range(3): time.sleep(0.3) samples.append(scroll_state(page)["y"]) break else: raise AssertionError("the long turn did not settle within 40s") assert mid_stream >= 8, "the samples must land while the stream is running" spread = max(samples) - min(samples) assert spread <= STABLE_PX, ( f"the viewport moved {spread:.1f}px while the user was scrolled up " "(no-reply-autoscroll contract)" ) # The whole answer still landed (off-screen below — by design). expect(page.locator(".msg.brain .bubble").last).to_contain_text(LONG_ANSWER_END) # --------------------------------------------------------------------------- # 2. No autoscroll: scrolled up during THINKING — the whole reasoning # stream plus the answer's start happen with the viewport held # --------------------------------------------------------------------------- def test_no_autoscroll_during_thinking( page: Page, app_url: str, seeded_kb: None ) -> None: page.set_default_timeout(30_000) page.goto(app_url) # One settled long turn so the document overflows (scrollable). submit(page, LONG_QUESTION) wait_settled(page) # The thinking turn: the submit reveals the user's message (the kept # app scroll) ... submit(page, THINKING_QUESTION) details = page.locator(".msg.brain").last.locator("details.thinking") details.wait_for(state="attached", timeout=10_000) expect(details).to_have_attribute("open", "") # created open (phase 17) # ... and, once the reasoning stream is clearly running ... page.wait_for_function( "() => { const el = document.querySelector('details.thinking .thinking-text');" " return !!el && el.innerText.length > 300; }", timeout=30_000, ) # ... let the submit's smooth reveal settle, then the user goes up. wait_scroll_still(page) page.evaluate("() => window.scrollTo(0, 0)") # Sample across the remaining thinking stream: no per-chunk follow. samples: list[float] = [] open_samples = 0 deadline = time.monotonic() + 25 while time.monotonic() < deadline: time.sleep(0.3) state = page.evaluate( """() => { const block = document.querySelector('details.thinking'); const wrap = block ? block.closest('.msg.brain') : null; const el = wrap ? wrap.querySelector('.bubble') : null; return { y: window.scrollY, open: !!(block && block.open), bubble: el ? el.innerText.length : 0 }; }""" ) samples.append(state["y"]) if state["open"]: open_samples += 1 if state["bubble"] > 0 and len(samples) >= 8: break else: raise AssertionError("the first answer token never arrived") assert open_samples >= 6, "the samples must land while the thinking stream is open" spread = max(samples) - min(samples) assert spread <= STABLE_PX, ( f"the viewport moved {spread:.1f}px during the thinking stream " "(no per-chunk page follow)" ) # Settled: still at the top, everything landed (off-screen, which is # the point of the story). wait_settled(page) assert abs(scroll_state(page)["y"]) <= STABLE_PX expect(details.locator(".thinking-text")).to_contain_text(THINKING_FRAGMENT) expect(page.locator(".msg.brain .bubble").last).to_contain_text(MOCK_ANSWER_MARKER) # --------------------------------------------------------------------------- # 3. Submit: scrolled to the very top, sending a question still reveals # the user's own message (the kept, user-initiated scroll) # --------------------------------------------------------------------------- def test_submit_reveals_user_message( page: Page, app_url: str, seeded_kb: None ) -> None: page.set_default_timeout(30_000) page.goto(app_url) # A populated conversation that overflows the viewport. submit(page, LONG_QUESTION) wait_settled(page) state = scroll_state(page) assert state["sh"] > state["ch"], "a long answer must make the document scrollable" # The user is reading at the very top ... page.evaluate("() => window.scrollTo(0, 0)") assert scroll_state(page)["y"] <= STABLE_PX # ... and sends a question without first scrolling down. submit_from_top(page, SHORT_QUESTION) # The submit's reveal is the kept app scroll: the user's own message # ends up in view (its box fully inside the viewport). deadline = time.monotonic() + 10 while time.monotonic() < deadline and not user_message_in_view(page): time.sleep(0.2) assert user_message_in_view(page), ( "the submit must reveal the user's message — its box is not in the viewport" ) # The turn completes; the answer lands off-screen below, but the user # message stays revealed (nothing re-positions it). wait_settled(page) assert user_message_in_view(page) expect(page.locator(".msg.brain .bubble").last).to_contain_text(MOCK_ANSWER_MARKER) # --------------------------------------------------------------------------- # 4. Restore landing (phase 14, owner-kept): a reload lands one-shot on # the latest message and stays there while idle # --------------------------------------------------------------------------- def test_restore_landing_one_shot( page: Page, app_url: str, seeded_kb: None ) -> None: page.set_default_timeout(30_000) page.goto(app_url) # Settle a conversation (phase-14 persistence): long + short turns. submit(page, LONG_QUESTION) wait_settled(page) submit(page, SHORT_QUESTION) wait_settled(page) time.sleep(0.5) # let the persistence writes land before the reload page.reload() # Restore re-renders from localStorage; wait until the last restored # brain answer (the short one — the long answer carries no mock # marker) is fully back. expect(page.locator(".msg.brain .bubble").last).to_contain_text( MOCK_ANSWER_MARKER, timeout=30_000 ) # The one-shot landing rides a smooth scroll — let it settle ... wait_scroll_still(page) state = scroll_state(page) assert state["sh"] > state["ch"] # ... and it lands on the latest message: its bubble is in view, in # the lower half of the viewport (the chips + composer sit just # below the fold — the landing predates them by design). box = page.locator(".msg.brain").last.locator(".bubble").bounding_box() assert box is not None assert box["y"] < state["ch"] and box["y"] + box["height"] >= state["ch"] * 0.6, ( "the restore landing must put the latest message in view" ) # ... and STAYS: idle samples (no stream active) never move. samples = [state["y"]] for _ in range(4): time.sleep(0.4) samples.append(scroll_state(page)["y"]) spread = max(samples) - min(samples) assert spread <= STABLE_PX, ( f"the restored page moved {spread:.1f}px while idle" ) # --------------------------------------------------------------------------- # 5. Content intact (regression): the long answer completes with sources; # a thinking turn persists + restores its collapsed block (phase 17) # --------------------------------------------------------------------------- def test_answer_content_intact(page: Page, app_url: str, seeded_kb: None) -> None: page.set_default_timeout(30_000) page.goto(app_url) # The long answer streams to completion with its sources ... submit(page, LONG_QUESTION) wait_settled(page) expect(page.locator(".msg.brain .bubble").last).to_contain_text(LONG_ANSWER_END) expect( page.locator(".msg.brain .source-chip", has_text="kubernetes.md") ).to_have_count(1) # ... and a thinking turn completes with its block auto-collapsed # (phase 17: open while streaming, closed from the first delta on). submit(page, THINKING_QUESTION) wait_settled(page) details = page.locator(".msg.brain").last.locator("details.thinking") expect(details).not_to_have_attribute("open") expect(details.locator(".thinking-text")).to_contain_text(THINKING_FRAGMENT) expect(page.locator(".msg.brain .bubble").last).to_contain_text(MOCK_ANSWER_MARKER) # Persistence: four messages, the thinking text + sources stored raw. raw = page.evaluate(f"() => localStorage.getItem('{STORAGE_KEY}')") stored = json.loads(raw) assert [m["who"] for m in stored["messages"]] == ["user", "brain", "user", "brain"] assert LONG_ANSWER_END in stored["messages"][1]["text"] assert THINKING_FRAGMENT in stored["messages"][3]["thinking"] assert any( s["path"] == "homelab/kubernetes.md" for s in stored["messages"][3]["sources"] ) # Restore: the long answer (with its chip) and the COLLAPSED thinking # block come back intact. page.reload() expect(page.locator(".msg.user .bubble")).to_have_count(2) expect(page.locator(".msg.brain .bubble")).to_have_count(2) expect(page.locator(".msg.brain .bubble").first).to_contain_text(LONG_ANSWER_END) expect( page.locator(".msg.brain .source-chip", has_text="kubernetes.md") ).to_have_count(2) restored = page.locator(".msg.brain").last.locator("details.thinking") expect(restored).not_to_have_attribute("open") expect(restored.locator(".thinking-text")).to_contain_text(THINKING_FRAGMENT) wait_settled(page)