fix(ui): thinking window no longer scrolls — live 320px view pinned to the stream tail
This commit is contained in:
@@ -0,0 +1,455 @@
|
||||
"""Phase 21 E2E (Playwright, mock-only): the Thinking window is a live tail.
|
||||
|
||||
Story: ``.agent/user_stories/thinking-no-scroll.md``
|
||||
Run in isolation (DB must be up: ``podman compose up -d db``):
|
||||
|
||||
uv run pytest tests/e2e/test_thinking_no_scroll.py -v --no-cov
|
||||
|
||||
MOCK-ONLY suite: ``E2E_REAL_LLM=1`` is not supported here — the real
|
||||
``turbo`` decides its own reasoning length and pacing, and this story's
|
||||
contract (fixed 320px clip, no user scroll back, programmatic pin to the
|
||||
live tail) needs the deterministic mock's long, paced scratchpad.
|
||||
|
||||
The whole functional change is one CSS property
|
||||
(``details.thinking .thinking-text``: ``overflow-y: auto`` → ``hidden``);
|
||||
``overflow: hidden`` still allows the phase-17 programmatic bottom-pin
|
||||
(``scrollTop = scrollHeight`` per thinking chunk), which is the sole
|
||||
scroller. This suite proves the browser behavior the unit pins
|
||||
(``tests/unit/test_thinking_no_scroll.py``) can only pin at source level.
|
||||
|
||||
Determinism note: phase 21 lengthened the mock's ``compose_thinking``
|
||||
body to ~2 700 chars (≈230 frames at the mock's 12-char/0.02s pacing ≈
|
||||
4.5s) so the rendered scratchpad overflows the 320px window by ~2x.
|
||||
Tests 1–2 key off the mock's ``think out loud then hesitate`` trigger:
|
||||
after the thinking stream ends there is a deterministic 4s pre-content
|
||||
pause with the block still OPEN and no further pin frames — a frozen
|
||||
live tail, the only state where a (regressed, working) user scroll would
|
||||
persist and be observable. During live streaming the per-chunk re-pin
|
||||
masks any user scroll within one frame (≈20ms), so that state is covered
|
||||
by the tail-tracking invariant instead (test 2, sampled mid-stream).
|
||||
|
||||
Test → story mapping (Playwright Mapping Rule):
|
||||
1. ``test_thinking_window_not_user_scrollable`` — wheel / drag / keyboard
|
||||
on the frozen live tail do not move the window.
|
||||
2. ``test_thinking_window_tracks_live_tail`` — after the 2nd-to-last and
|
||||
the last thinking chunk the window is pinned to the tail and the last
|
||||
chunk's text sits inside the visible rectangle.
|
||||
3. ``test_thinking_window_css_contract`` — computed ``overflow-y`` is
|
||||
``hidden``, ``max-height`` is 320px, and the clip is real.
|
||||
4. ``test_answer_bubble_still_scrollable`` — regression (phase 11): the
|
||||
answer bubble's overflow is untouched and the page still scrolls.
|
||||
5. ``test_restored_collapsed_thinking_unaffected`` — regression
|
||||
(phase 17): a stored thinking turn restores a collapsed block.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import re
|
||||
from collections.abc import Iterator
|
||||
from pathlib import Path
|
||||
from threading import Thread
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
from playwright.sync_api import Page, expect
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.config import Settings
|
||||
from app.db import SessionLocal
|
||||
from app.rag.importer import ImportSummary, import_sources
|
||||
from app.rag.llm import LLMClient
|
||||
from e2e.mock_llm import compose_thinking
|
||||
|
||||
REPO = Path(__file__).resolve().parents[2]
|
||||
FIXTURES = REPO / "tests" / "fixtures" / "docs"
|
||||
|
||||
#: Phase-17 trigger question (grounded turn, thinking + short answer).
|
||||
THINK_QUESTION = "think out loud — how is my kubernetes cluster set up?"
|
||||
#: Phase-20 hesitation trigger: the long phase-17/21 thinking stream, then
|
||||
#: a deterministic 4s pause before the first content frame — a frozen
|
||||
#: live tail with the block still open (the no-scroll test window).
|
||||
HESITATE_QUESTION = "think out loud then hesitate — how is my kubernetes cluster set up?"
|
||||
#: Phase-11 long-answer trigger (regression test 4).
|
||||
LONG_QUESTION = "How is my Kubernetes cluster set up? write a long answer"
|
||||
LONG_ANSWER_END = "LONG-ANSWER-END"
|
||||
MOCK_ANSWER_MARKER = "Deterministic mock answer for E2E"
|
||||
THINKING_FRAGMENT = "Step 2: Check my notes"
|
||||
STORAGE_KEY = "bor.chat.v1"
|
||||
|
||||
SELECTOR = ".msg.brain details.thinking .thinking-text"
|
||||
|
||||
|
||||
async def _import_fixtures(mock_port: int) -> ImportSummary:
|
||||
kwargs: dict[str, Any] = {"_env_file": None, "llm_base_url": f"http://127.0.0.1:{mock_port}/v1"}
|
||||
settings = Settings(**kwargs) # pyright: ignore[reportCallIssue]
|
||||
return await import_sources([FIXTURES], LLMClient(settings))
|
||||
|
||||
|
||||
def _run_in_thread(coro: Any) -> Any:
|
||||
"""Run a coroutine on a worker thread (Playwright owns the test loop)."""
|
||||
box: dict[str, Any] = {}
|
||||
|
||||
def runner() -> None:
|
||||
try:
|
||||
box["value"] = asyncio.run(coro)
|
||||
except BaseException as e: # noqa: BLE001 — re-raised on the test thread
|
||||
box["error"] = e
|
||||
|
||||
t = Thread(target=runner)
|
||||
t.start()
|
||||
t.join()
|
||||
if "error" in box:
|
||||
raise box["error"]
|
||||
return box["value"]
|
||||
|
||||
|
||||
def _reset_db(mock_port: int, seed: bool) -> ImportSummary | None:
|
||||
"""Truncate the KB (and query log), then optionally re-import fixtures."""
|
||||
with SessionLocal() as db:
|
||||
db.execute(text("TRUNCATE chunks, documents, query_log"))
|
||||
db.commit()
|
||||
if not seed:
|
||||
return None
|
||||
return _run_in_thread(_import_fixtures(mock_port))
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def seeded_kb(mock_llm: int, db_ready: None) -> Iterator[None]:
|
||||
"""A fresh KB seeded from ``tests/fixtures/docs`` (8 docs, A9 formats),
|
||||
truncated again on teardown (same fixture shape as the phase-17 suite)."""
|
||||
summary = _reset_db(mock_llm, seed=True)
|
||||
assert summary is not None and summary.added == 8
|
||||
yield
|
||||
_reset_db(mock_llm, seed=False)
|
||||
|
||||
|
||||
def send_and_wait(page: Page, question: str) -> None:
|
||||
"""Type into #message-input, submit via #composer, then wait until the
|
||||
last brain message settles (send button re-enabled, label "Send")."""
|
||||
page.fill("#message-input", question)
|
||||
page.evaluate("() => document.querySelector('#composer').requestSubmit()")
|
||||
expect(page.locator(".msg.user .bubble").last).to_contain_text(question)
|
||||
# Thinking (~4.5s, phase 21) + answer land in a few seconds.
|
||||
expect(page.locator(".msg.brain .bubble").last).not_to_have_text("", timeout=30_000)
|
||||
expect(page.locator("#send-btn")).to_be_enabled(timeout=30_000)
|
||||
expect(page.locator("#send-label")).to_have_text("Send")
|
||||
|
||||
|
||||
def _scroll_sample(page: Page) -> dict[str, float]:
|
||||
"""scrollTop / scrollHeight / clientHeight of the live Thinking window."""
|
||||
return page.evaluate(
|
||||
f"""() => {{ const el = document.querySelector('{SELECTOR}');
|
||||
return {{ top: el.scrollTop, height: el.scrollHeight, client: el.clientHeight }}; }}"""
|
||||
)
|
||||
|
||||
|
||||
def _at_tail(sample: dict[str, float]) -> bool:
|
||||
"""True when the window is pinned to the live tail: the bottom of the
|
||||
content is visible (scrollTop clamped at scrollHeight - clientHeight,
|
||||
within 1px — the phase-17 pin's effect)."""
|
||||
return abs(sample["top"] - (sample["height"] - sample["client"])) <= 1
|
||||
|
||||
|
||||
def _wait_text_stable(page: Page, timeout_ms: int = 30_000) -> None:
|
||||
"""Wait until the scratchpad text stops growing for 300ms.
|
||||
|
||||
The mock paces frames at 0.02s, so a 300ms still length means the
|
||||
thinking stream has ended — with the hesitation trigger, the 4s
|
||||
pre-content pause (block still open, no further pin frames) is then
|
||||
running."""
|
||||
page.wait_for_function(
|
||||
f"""() => {{ const el = document.querySelector('{SELECTOR}');
|
||||
if (!el) return false;
|
||||
const len = el.innerText.length;
|
||||
const now = performance.now();
|
||||
if (!window.__thinkProbe) window.__thinkProbe = {{ len, at: now }};
|
||||
const p = window.__thinkProbe;
|
||||
if (len !== p.len) {{ p.len = len; p.at = now; return false; }}
|
||||
return now - p.at >= 300; }}""",
|
||||
timeout=timeout_ms,
|
||||
)
|
||||
|
||||
|
||||
def _wait_text_contains(page: Page, marker: str, timeout_ms: int = 30_000) -> None:
|
||||
"""Wait until the rendered scratchpad (whitespace-insensitive) contains
|
||||
``marker`` — a deterministic probe for a given point in the stream."""
|
||||
page.wait_for_function(
|
||||
f"""(tail) => {{ const el = document.querySelector('{SELECTOR}');
|
||||
return !!el && el.innerText.replace(/\\s+/g, '').includes(tail); }}""",
|
||||
arg=marker,
|
||||
timeout=timeout_ms,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. No user scroll back: wheel, drag, and keyboard on the frozen live tail
|
||||
# do not move the window
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_thinking_window_not_user_scrollable(
|
||||
page: Page, app_url: str, seeded_kb: None
|
||||
) -> None:
|
||||
page.set_default_timeout(30_000)
|
||||
page.goto(app_url)
|
||||
page.fill("#message-input", HESITATE_QUESTION)
|
||||
page.evaluate("() => document.querySelector('#composer').requestSubmit()")
|
||||
expect(page.locator(".msg.user .bubble").last).to_contain_text(HESITATE_QUESTION)
|
||||
|
||||
details = page.locator(".msg.brain").last.locator("details.thinking")
|
||||
details.wait_for(state="attached", timeout=10_000)
|
||||
text_el = details.locator(".thinking-text")
|
||||
|
||||
# Premise: the long scratchpad overflows the 320px clip.
|
||||
page.wait_for_function(
|
||||
f"() => {{ const el = document.querySelector('{SELECTOR}');"
|
||||
" return !!el && el.scrollHeight > el.clientHeight; }",
|
||||
timeout=30_000,
|
||||
)
|
||||
|
||||
# The thinking stream has ENDED (4s hesitation pause running): no more
|
||||
# pin frames, the block is still open, and the tail is frozen — any
|
||||
# user scroll would persist and be observable here.
|
||||
_wait_text_stable(page)
|
||||
expect(details).to_have_attribute("open", "")
|
||||
expect(page.locator(".msg.brain .bubble").last).to_have_text("")
|
||||
|
||||
# Precondition: the phase-17 pin left the window at the live tail.
|
||||
before = _scroll_sample(page)
|
||||
assert before["height"] > before["client"], "the window must overflow"
|
||||
assert _at_tail(before), "the pin must have left the window at the tail"
|
||||
|
||||
# Focus the window (a plain div is not focusable — el.focus() is a
|
||||
# no-op; the keyboard presses below land on the page focus instead.
|
||||
# Neither path may move the window).
|
||||
text_el.evaluate("el => el.focus()")
|
||||
box = text_el.bounding_box()
|
||||
assert box
|
||||
cx, cy = box["x"] + box["width"] / 2, box["y"] + box["height"] / 2
|
||||
page.mouse.move(cx, cy)
|
||||
|
||||
# Wheel back (up) — must not scroll the window.
|
||||
page.mouse.wheel(0, -200)
|
||||
# Keyboard: Home + ArrowUp — must not scroll the window.
|
||||
page.keyboard.press("Home")
|
||||
page.keyboard.press("ArrowUp")
|
||||
page.keyboard.press("ArrowUp")
|
||||
# Mouse drag over the window — must not scroll the window
|
||||
# (there is no scrollbar to grab with overflow hidden).
|
||||
page.mouse.down()
|
||||
page.mouse.move(cx, cy - 100, steps=5)
|
||||
page.mouse.up()
|
||||
# Wheel again, then let a would-be (regressed) scroll settle.
|
||||
page.mouse.wheel(0, -200)
|
||||
page.wait_for_timeout(150)
|
||||
|
||||
# Still inside the pure-thinking window (the assertions below are only
|
||||
# meaningful while the block is open and no content frame has landed).
|
||||
expect(details).to_have_attribute("open", "")
|
||||
after = _scroll_sample(page)
|
||||
assert abs(after["top"] - before["top"]) <= 1, (
|
||||
f"user scroll moved the window: {before['top']} -> {after['top']}"
|
||||
)
|
||||
assert _at_tail(after), "the window must still show the live tail"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Live tail tracking: the per-chunk pin keeps the window glued to the
|
||||
# newest content (sampled at the 2nd-to-last and last chunks), and the
|
||||
# last chunk's text is inside the visible rectangle
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_thinking_window_tracks_live_tail(
|
||||
page: Page, app_url: str, seeded_kb: None
|
||||
) -> None:
|
||||
page.set_default_timeout(30_000)
|
||||
page.goto(app_url)
|
||||
page.fill("#message-input", HESITATE_QUESTION)
|
||||
page.evaluate("() => document.querySelector('#composer').requestSubmit()")
|
||||
expect(page.locator(".msg.user .bubble").last).to_contain_text(HESITATE_QUESTION)
|
||||
|
||||
details = page.locator(".msg.brain").last.locator("details.thinking")
|
||||
details.wait_for(state="attached", timeout=10_000)
|
||||
|
||||
# The exact deterministic scratchpad the mock will stream, sliced the
|
||||
# same way the mock's _sse_stream does (12-char chunks).
|
||||
expected = compose_thinking(
|
||||
{"messages": [{"role": "user", "content": HESITATE_QUESTION}]}
|
||||
)
|
||||
pieces = re.findall(r".{1,12}", expected, re.S)
|
||||
ws = re.sub(r"\s+", "", "".join(pieces))
|
||||
#: 12 rendered chars ending at the 2nd-to-last chunk.
|
||||
marker_second_last = re.sub(r"\s+", "", "".join(pieces[:-1]))[-12:]
|
||||
#: 12 rendered chars at the very end (the last chunk).
|
||||
marker_last = ws[-12:]
|
||||
|
||||
# During the stream: once the 2nd-to-last chunk has landed, the
|
||||
# window is pinned to the tail (the invariant holds at EVERY chunk;
|
||||
# the pin runs per chunk while the block is open).
|
||||
_wait_text_contains(page, marker_second_last)
|
||||
assert _at_tail(_scroll_sample(page)), "not at the tail after chunk N-1"
|
||||
|
||||
# After the last chunk: the 4s hesitation pause holds this state with
|
||||
# the block still open — sample the tail pin, then the geometry.
|
||||
_wait_text_contains(page, marker_last)
|
||||
_wait_text_stable(page)
|
||||
expect(details).to_have_attribute("open", "")
|
||||
sample = _scroll_sample(page)
|
||||
assert _at_tail(sample), f"not at the tail after the last chunk: {sample}"
|
||||
|
||||
# Geometry: the last chunk's text (the final text node of the
|
||||
# scratchpad) renders INSIDE the visible rectangle, and a hit-test at
|
||||
# the box's bottom lands inside .thinking-text.
|
||||
geo = page.evaluate(
|
||||
f"""() => {{ const el = document.querySelector('{SELECTOR}');
|
||||
const box = el.getBoundingClientRect();
|
||||
const walker = document.createTreeWalker(el, NodeFilter.SHOW_TEXT);
|
||||
let last = null;
|
||||
while (walker.nextNode()) last = walker.currentNode;
|
||||
const range = document.createRange();
|
||||
range.selectNodeContents(last);
|
||||
const r = range.getBoundingClientRect();
|
||||
const hit = document.elementFromPoint(box.left + 10, box.bottom - 5);
|
||||
return {{
|
||||
nodeVisible: r.bottom > box.top && r.top < box.bottom,
|
||||
nodeBottomInBox: r.bottom <= box.bottom + 1,
|
||||
hitInside: hit ? el.contains(hit) : false,
|
||||
}}; }}"""
|
||||
)
|
||||
assert geo["nodeVisible"], "the last chunk's text is outside the window"
|
||||
assert geo["nodeBottomInBox"], "the last chunk's text is clipped off the bottom"
|
||||
assert geo["hitInside"], "a hit-test at the box bottom missed .thinking-text"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. CSS contract: overflow-y hidden, 320px max-height, and the clip is real
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_thinking_window_css_contract(
|
||||
page: Page, app_url: str, seeded_kb: None
|
||||
) -> None:
|
||||
page.set_default_timeout(30_000)
|
||||
page.goto(app_url)
|
||||
send_and_wait(page, THINK_QUESTION)
|
||||
|
||||
details = page.locator(".msg.brain").last.locator("details.thinking")
|
||||
expect(details).not_to_have_attribute("open") # auto-collapsed
|
||||
details.locator("summary").click() # open for measurement
|
||||
expect(details).to_have_attribute("open", "")
|
||||
|
||||
style = page.evaluate(
|
||||
f"""() => {{ const el = document.querySelector('{SELECTOR}');
|
||||
const cs = getComputedStyle(el);
|
||||
return {{ overflowY: cs.overflowY, maxHeight: cs.maxHeight,
|
||||
scrollHeight: el.scrollHeight, clientHeight: el.clientHeight }}; }}"""
|
||||
)
|
||||
assert style["overflowY"] == "hidden", "the window must not be user-scrollable"
|
||||
assert style["maxHeight"] == "320px", "the 320px clip must stay"
|
||||
# The clip is real, not cosmetic: the long scratchpad overflows it.
|
||||
assert style["scrollHeight"] > style["clientHeight"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4. Regression (phase 11): the answer bubble is untouched — a long answer
|
||||
# still grows the page, and normal (user) scrolling of the answer works
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_answer_bubble_still_scrollable(
|
||||
page: Page, app_url: str, seeded_kb: None
|
||||
) -> None:
|
||||
page.set_default_timeout(45_000)
|
||||
page.goto(app_url)
|
||||
page.fill("#message-input", LONG_QUESTION)
|
||||
page.click("#send-btn")
|
||||
expect(page.locator(".msg.user .bubble").last).to_contain_text(LONG_QUESTION)
|
||||
|
||||
# The ~900-word answer streams to completion (phase-11 contract).
|
||||
bubble = page.locator(".msg.brain .bubble").last
|
||||
expect(bubble).to_contain_text(LONG_ANSWER_END, timeout=60_000)
|
||||
expect(page.locator("#send-btn")).to_be_enabled()
|
||||
|
||||
# The answer bubble keeps its existing overflow (phase 21 only touched
|
||||
# .thinking-text) — it is NOT the hidden clip.
|
||||
overflow_y = page.evaluate(
|
||||
"() => { const els = document.querySelectorAll('.msg.brain .bubble');"
|
||||
" return getComputedStyle(els[els.length - 1]).overflowY; }"
|
||||
)
|
||||
assert overflow_y != "hidden", "the answer bubble must keep its scroll behavior"
|
||||
|
||||
# A long answer grows the PAGE — and the page still scrolls normally.
|
||||
state0 = page.evaluate(
|
||||
"() => ({ y: window.scrollY, sh: document.documentElement.scrollHeight,"
|
||||
" ch: window.innerHeight })"
|
||||
)
|
||||
assert state0["sh"] > state0["ch"], "the long answer must make the page scrollable"
|
||||
|
||||
box = bubble.bounding_box()
|
||||
assert box
|
||||
viewport = page.viewport_size
|
||||
assert viewport # the conftest `page` fixture is fixed at 1280x800
|
||||
# A point in the visible lower part of the answer area (the page is
|
||||
# pinned at the bottom, so the bubble's lower edge is in the viewport).
|
||||
mx = box["x"] + box["width"] / 2
|
||||
my = max(50.0, min(box["y"] + box["height"] - 60.0, viewport["height"] - 100))
|
||||
hit = page.evaluate(
|
||||
"([x, y]) => { const e = document.elementFromPoint(x, y);"
|
||||
" return e ? e.tagName + '.' + String(e.className) : 'none'; }",
|
||||
[mx, my],
|
||||
)
|
||||
page.mouse.move(mx, my)
|
||||
# Headless Chromium applies wheel scrolling through an async momentum
|
||||
# pipeline — let each gesture settle before reading the position.
|
||||
page.mouse.wheel(0, -400) # wheel up: away from the newest content
|
||||
page.wait_for_timeout(500)
|
||||
y_up = page.evaluate("() => window.scrollY")
|
||||
page.mouse.wheel(0, 400) # wheel back down
|
||||
page.wait_for_timeout(500)
|
||||
y_down = page.evaluate("() => window.scrollY")
|
||||
assert y_up < state0["y"] - 100, (
|
||||
f"the page must scroll up on wheel (wheel over {hit!r}): "
|
||||
f"y {state0['y']} -> {y_up}"
|
||||
)
|
||||
assert y_down > y_up, f"the page must scroll back down on wheel: {y_up} -> {y_down}"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 5. Regression (phase 17): a stored thinking turn restores a COLLAPSED
|
||||
# block with its full text (replicates the phase-17 reload pin — the
|
||||
# restore path is untouched by phase 21, where overflow is moot)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_restored_collapsed_thinking_unaffected(
|
||||
page: Page, app_url: str, seeded_kb: None
|
||||
) -> None:
|
||||
page.set_default_timeout(30_000)
|
||||
page.goto(app_url)
|
||||
send_and_wait(page, THINK_QUESTION)
|
||||
|
||||
details = page.locator(".msg.brain").last.locator("details.thinking")
|
||||
expect(details).not_to_have_attribute("open") # auto-collapsed
|
||||
captured = details.locator(".thinking-text").text_content()
|
||||
assert captured
|
||||
|
||||
page.reload()
|
||||
expect(page.locator("#empty-state")).to_be_hidden()
|
||||
|
||||
restored = page.locator(".msg.brain").last.locator("details.thinking")
|
||||
expect(restored).to_have_count(1)
|
||||
expect(restored).not_to_have_attribute("open") # restored COLLAPSED
|
||||
expect(restored.locator(".thinking-text")).to_have_text(captured)
|
||||
|
||||
# Opening the restored block still shows the full scratchpad, and the
|
||||
# answer + persistence are intact.
|
||||
restored.locator("summary").click()
|
||||
expect(restored).to_have_attribute("open", "")
|
||||
expect(restored.locator(".thinking-text")).to_contain_text(THINKING_FRAGMENT)
|
||||
expect(page.locator(".msg.brain .bubble").last).to_contain_text(MOCK_ANSWER_MARKER)
|
||||
raw = json.loads(page.evaluate(f"() => localStorage.getItem('{STORAGE_KEY}')"))[
|
||||
"messages"
|
||||
][1]["thinking"]
|
||||
assert re.sub(r"\s+", "", raw) == re.sub(r"\s+", "", captured)
|
||||
Reference in New Issue
Block a user