fix(ui): thinking window no longer scrolls — live 320px view pinned to the stream tail

This commit is contained in:
2026-08-24 18:24:57 -04:00
parent 76c6a01199
commit 04a7f4c05d
11 changed files with 618 additions and 279 deletions
+455
View File
@@ -0,0 +1,455 @@
"""Phase 21 E2E (Playwright, mock-only): the Thinking window is a live tail.
Story: ``.agent/user_stories/thinking-no-scroll.md``
Run in isolation (DB must be up: ``podman compose up -d db``):
uv run pytest tests/e2e/test_thinking_no_scroll.py -v --no-cov
MOCK-ONLY suite: ``E2E_REAL_LLM=1`` is not supported here — the real
``turbo`` decides its own reasoning length and pacing, and this story's
contract (fixed 320px clip, no user scroll back, programmatic pin to the
live tail) needs the deterministic mock's long, paced scratchpad.
The whole functional change is one CSS property
(``details.thinking .thinking-text``: ``overflow-y: auto`` → ``hidden``);
``overflow: hidden`` still allows the phase-17 programmatic bottom-pin
(``scrollTop = scrollHeight`` per thinking chunk), which is the sole
scroller. This suite proves the browser behavior the unit pins
(``tests/unit/test_thinking_no_scroll.py``) can only pin at source level.
Determinism note: phase 21 lengthened the mock's ``compose_thinking``
body to ~2 700 chars (≈230 frames at the mock's 12-char/0.02s pacing ≈
4.5s) so the rendered scratchpad overflows the 320px window by ~2x.
Tests 1–2 key off the mock's ``think out loud then hesitate`` trigger:
after the thinking stream ends there is a deterministic 4s pre-content
pause with the block still OPEN and no further pin frames — a frozen
live tail, the only state where a (regressed, working) user scroll would
persist and be observable. During live streaming the per-chunk re-pin
masks any user scroll within one frame (≈20ms), so that state is covered
by the tail-tracking invariant instead (test 2, sampled mid-stream).
Test → story mapping (Playwright Mapping Rule):
1. ``test_thinking_window_not_user_scrollable`` — wheel / drag / keyboard
on the frozen live tail do not move the window.
2. ``test_thinking_window_tracks_live_tail`` — after the 2nd-to-last and
the last thinking chunk the window is pinned to the tail and the last
chunk's text sits inside the visible rectangle.
3. ``test_thinking_window_css_contract`` — computed ``overflow-y`` is
``hidden``, ``max-height`` is 320px, and the clip is real.
4. ``test_answer_bubble_still_scrollable`` — regression (phase 11): the
answer bubble's overflow is untouched and the page still scrolls.
5. ``test_restored_collapsed_thinking_unaffected`` — regression
(phase 17): a stored thinking turn restores a collapsed block.
"""
from __future__ import annotations
import asyncio
import json
import re
from collections.abc import Iterator
from pathlib import Path
from threading import Thread
from typing import Any
import pytest
from playwright.sync_api import Page, expect
from sqlalchemy import text
from app.config import Settings
from app.db import SessionLocal
from app.rag.importer import ImportSummary, import_sources
from app.rag.llm import LLMClient
from e2e.mock_llm import compose_thinking
REPO = Path(__file__).resolve().parents[2]
FIXTURES = REPO / "tests" / "fixtures" / "docs"
#: Phase-17 trigger question (grounded turn, thinking + short answer).
THINK_QUESTION = "think out loud — how is my kubernetes cluster set up?"
#: Phase-20 hesitation trigger: the long phase-17/21 thinking stream, then
#: a deterministic 4s pause before the first content frame — a frozen
#: live tail with the block still open (the no-scroll test window).
HESITATE_QUESTION = "think out loud then hesitate — how is my kubernetes cluster set up?"
#: Phase-11 long-answer trigger (regression test 4).
LONG_QUESTION = "How is my Kubernetes cluster set up? write a long answer"
LONG_ANSWER_END = "LONG-ANSWER-END"
MOCK_ANSWER_MARKER = "Deterministic mock answer for E2E"
THINKING_FRAGMENT = "Step 2: Check my notes"
STORAGE_KEY = "bor.chat.v1"
SELECTOR = ".msg.brain details.thinking .thinking-text"
async def _import_fixtures(mock_port: int) -> ImportSummary:
kwargs: dict[str, Any] = {"_env_file": None, "llm_base_url": f"http://127.0.0.1:{mock_port}/v1"}
settings = Settings(**kwargs) # pyright: ignore[reportCallIssue]
return await import_sources([FIXTURES], LLMClient(settings))
def _run_in_thread(coro: Any) -> Any:
"""Run a coroutine on a worker thread (Playwright owns the test loop)."""
box: dict[str, Any] = {}
def runner() -> None:
try:
box["value"] = asyncio.run(coro)
except BaseException as e: # noqa: BLE001 — re-raised on the test thread
box["error"] = e
t = Thread(target=runner)
t.start()
t.join()
if "error" in box:
raise box["error"]
return box["value"]
def _reset_db(mock_port: int, seed: bool) -> ImportSummary | None:
"""Truncate the KB (and query log), then optionally re-import fixtures."""
with SessionLocal() as db:
db.execute(text("TRUNCATE chunks, documents, query_log"))
db.commit()
if not seed:
return None
return _run_in_thread(_import_fixtures(mock_port))
@pytest.fixture()
def seeded_kb(mock_llm: int, db_ready: None) -> Iterator[None]:
"""A fresh KB seeded from ``tests/fixtures/docs`` (8 docs, A9 formats),
truncated again on teardown (same fixture shape as the phase-17 suite)."""
summary = _reset_db(mock_llm, seed=True)
assert summary is not None and summary.added == 8
yield
_reset_db(mock_llm, seed=False)
def send_and_wait(page: Page, question: str) -> None:
"""Type into #message-input, submit via #composer, then wait until the
last brain message settles (send button re-enabled, label "Send")."""
page.fill("#message-input", question)
page.evaluate("() => document.querySelector('#composer').requestSubmit()")
expect(page.locator(".msg.user .bubble").last).to_contain_text(question)
# Thinking (~4.5s, phase 21) + answer land in a few seconds.
expect(page.locator(".msg.brain .bubble").last).not_to_have_text("", timeout=30_000)
expect(page.locator("#send-btn")).to_be_enabled(timeout=30_000)
expect(page.locator("#send-label")).to_have_text("Send")
def _scroll_sample(page: Page) -> dict[str, float]:
"""scrollTop / scrollHeight / clientHeight of the live Thinking window."""
return page.evaluate(
f"""() => {{ const el = document.querySelector('{SELECTOR}');
return {{ top: el.scrollTop, height: el.scrollHeight, client: el.clientHeight }}; }}"""
)
def _at_tail(sample: dict[str, float]) -> bool:
"""True when the window is pinned to the live tail: the bottom of the
content is visible (scrollTop clamped at scrollHeight - clientHeight,
within 1px — the phase-17 pin's effect)."""
return abs(sample["top"] - (sample["height"] - sample["client"])) <= 1
def _wait_text_stable(page: Page, timeout_ms: int = 30_000) -> None:
"""Wait until the scratchpad text stops growing for 300ms.
The mock paces frames at 0.02s, so a 300ms still length means the
thinking stream has ended — with the hesitation trigger, the 4s
pre-content pause (block still open, no further pin frames) is then
running."""
page.wait_for_function(
f"""() => {{ const el = document.querySelector('{SELECTOR}');
if (!el) return false;
const len = el.innerText.length;
const now = performance.now();
if (!window.__thinkProbe) window.__thinkProbe = {{ len, at: now }};
const p = window.__thinkProbe;
if (len !== p.len) {{ p.len = len; p.at = now; return false; }}
return now - p.at >= 300; }}""",
timeout=timeout_ms,
)
def _wait_text_contains(page: Page, marker: str, timeout_ms: int = 30_000) -> None:
"""Wait until the rendered scratchpad (whitespace-insensitive) contains
``marker`` — a deterministic probe for a given point in the stream."""
page.wait_for_function(
f"""(tail) => {{ const el = document.querySelector('{SELECTOR}');
return !!el && el.innerText.replace(/\\s+/g, '').includes(tail); }}""",
arg=marker,
timeout=timeout_ms,
)
# ---------------------------------------------------------------------------
# 1. No user scroll back: wheel, drag, and keyboard on the frozen live tail
# do not move the window
# ---------------------------------------------------------------------------
def test_thinking_window_not_user_scrollable(
page: Page, app_url: str, seeded_kb: None
) -> None:
page.set_default_timeout(30_000)
page.goto(app_url)
page.fill("#message-input", HESITATE_QUESTION)
page.evaluate("() => document.querySelector('#composer').requestSubmit()")
expect(page.locator(".msg.user .bubble").last).to_contain_text(HESITATE_QUESTION)
details = page.locator(".msg.brain").last.locator("details.thinking")
details.wait_for(state="attached", timeout=10_000)
text_el = details.locator(".thinking-text")
# Premise: the long scratchpad overflows the 320px clip.
page.wait_for_function(
f"() => {{ const el = document.querySelector('{SELECTOR}');"
" return !!el && el.scrollHeight > el.clientHeight; }",
timeout=30_000,
)
# The thinking stream has ENDED (4s hesitation pause running): no more
# pin frames, the block is still open, and the tail is frozen — any
# user scroll would persist and be observable here.
_wait_text_stable(page)
expect(details).to_have_attribute("open", "")
expect(page.locator(".msg.brain .bubble").last).to_have_text("")
# Precondition: the phase-17 pin left the window at the live tail.
before = _scroll_sample(page)
assert before["height"] > before["client"], "the window must overflow"
assert _at_tail(before), "the pin must have left the window at the tail"
# Focus the window (a plain div is not focusable — el.focus() is a
# no-op; the keyboard presses below land on the page focus instead.
# Neither path may move the window).
text_el.evaluate("el => el.focus()")
box = text_el.bounding_box()
assert box
cx, cy = box["x"] + box["width"] / 2, box["y"] + box["height"] / 2
page.mouse.move(cx, cy)
# Wheel back (up) — must not scroll the window.
page.mouse.wheel(0, -200)
# Keyboard: Home + ArrowUp — must not scroll the window.
page.keyboard.press("Home")
page.keyboard.press("ArrowUp")
page.keyboard.press("ArrowUp")
# Mouse drag over the window — must not scroll the window
# (there is no scrollbar to grab with overflow hidden).
page.mouse.down()
page.mouse.move(cx, cy - 100, steps=5)
page.mouse.up()
# Wheel again, then let a would-be (regressed) scroll settle.
page.mouse.wheel(0, -200)
page.wait_for_timeout(150)
# Still inside the pure-thinking window (the assertions below are only
# meaningful while the block is open and no content frame has landed).
expect(details).to_have_attribute("open", "")
after = _scroll_sample(page)
assert abs(after["top"] - before["top"]) <= 1, (
f"user scroll moved the window: {before['top']} -> {after['top']}"
)
assert _at_tail(after), "the window must still show the live tail"
# ---------------------------------------------------------------------------
# 2. Live tail tracking: the per-chunk pin keeps the window glued to the
# newest content (sampled at the 2nd-to-last and last chunks), and the
# last chunk's text is inside the visible rectangle
# ---------------------------------------------------------------------------
def test_thinking_window_tracks_live_tail(
page: Page, app_url: str, seeded_kb: None
) -> None:
page.set_default_timeout(30_000)
page.goto(app_url)
page.fill("#message-input", HESITATE_QUESTION)
page.evaluate("() => document.querySelector('#composer').requestSubmit()")
expect(page.locator(".msg.user .bubble").last).to_contain_text(HESITATE_QUESTION)
details = page.locator(".msg.brain").last.locator("details.thinking")
details.wait_for(state="attached", timeout=10_000)
# The exact deterministic scratchpad the mock will stream, sliced the
# same way the mock's _sse_stream does (12-char chunks).
expected = compose_thinking(
{"messages": [{"role": "user", "content": HESITATE_QUESTION}]}
)
pieces = re.findall(r".{1,12}", expected, re.S)
ws = re.sub(r"\s+", "", "".join(pieces))
#: 12 rendered chars ending at the 2nd-to-last chunk.
marker_second_last = re.sub(r"\s+", "", "".join(pieces[:-1]))[-12:]
#: 12 rendered chars at the very end (the last chunk).
marker_last = ws[-12:]
# During the stream: once the 2nd-to-last chunk has landed, the
# window is pinned to the tail (the invariant holds at EVERY chunk;
# the pin runs per chunk while the block is open).
_wait_text_contains(page, marker_second_last)
assert _at_tail(_scroll_sample(page)), "not at the tail after chunk N-1"
# After the last chunk: the 4s hesitation pause holds this state with
# the block still open — sample the tail pin, then the geometry.
_wait_text_contains(page, marker_last)
_wait_text_stable(page)
expect(details).to_have_attribute("open", "")
sample = _scroll_sample(page)
assert _at_tail(sample), f"not at the tail after the last chunk: {sample}"
# Geometry: the last chunk's text (the final text node of the
# scratchpad) renders INSIDE the visible rectangle, and a hit-test at
# the box's bottom lands inside .thinking-text.
geo = page.evaluate(
f"""() => {{ const el = document.querySelector('{SELECTOR}');
const box = el.getBoundingClientRect();
const walker = document.createTreeWalker(el, NodeFilter.SHOW_TEXT);
let last = null;
while (walker.nextNode()) last = walker.currentNode;
const range = document.createRange();
range.selectNodeContents(last);
const r = range.getBoundingClientRect();
const hit = document.elementFromPoint(box.left + 10, box.bottom - 5);
return {{
nodeVisible: r.bottom > box.top && r.top < box.bottom,
nodeBottomInBox: r.bottom <= box.bottom + 1,
hitInside: hit ? el.contains(hit) : false,
}}; }}"""
)
assert geo["nodeVisible"], "the last chunk's text is outside the window"
assert geo["nodeBottomInBox"], "the last chunk's text is clipped off the bottom"
assert geo["hitInside"], "a hit-test at the box bottom missed .thinking-text"
# ---------------------------------------------------------------------------
# 3. CSS contract: overflow-y hidden, 320px max-height, and the clip is real
# ---------------------------------------------------------------------------
def test_thinking_window_css_contract(
page: Page, app_url: str, seeded_kb: None
) -> None:
page.set_default_timeout(30_000)
page.goto(app_url)
send_and_wait(page, THINK_QUESTION)
details = page.locator(".msg.brain").last.locator("details.thinking")
expect(details).not_to_have_attribute("open") # auto-collapsed
details.locator("summary").click() # open for measurement
expect(details).to_have_attribute("open", "")
style = page.evaluate(
f"""() => {{ const el = document.querySelector('{SELECTOR}');
const cs = getComputedStyle(el);
return {{ overflowY: cs.overflowY, maxHeight: cs.maxHeight,
scrollHeight: el.scrollHeight, clientHeight: el.clientHeight }}; }}"""
)
assert style["overflowY"] == "hidden", "the window must not be user-scrollable"
assert style["maxHeight"] == "320px", "the 320px clip must stay"
# The clip is real, not cosmetic: the long scratchpad overflows it.
assert style["scrollHeight"] > style["clientHeight"]
# ---------------------------------------------------------------------------
# 4. Regression (phase 11): the answer bubble is untouched — a long answer
# still grows the page, and normal (user) scrolling of the answer works
# ---------------------------------------------------------------------------
def test_answer_bubble_still_scrollable(
page: Page, app_url: str, seeded_kb: None
) -> None:
page.set_default_timeout(45_000)
page.goto(app_url)
page.fill("#message-input", LONG_QUESTION)
page.click("#send-btn")
expect(page.locator(".msg.user .bubble").last).to_contain_text(LONG_QUESTION)
# The ~900-word answer streams to completion (phase-11 contract).
bubble = page.locator(".msg.brain .bubble").last
expect(bubble).to_contain_text(LONG_ANSWER_END, timeout=60_000)
expect(page.locator("#send-btn")).to_be_enabled()
# The answer bubble keeps its existing overflow (phase 21 only touched
# .thinking-text) — it is NOT the hidden clip.
overflow_y = page.evaluate(
"() => { const els = document.querySelectorAll('.msg.brain .bubble');"
" return getComputedStyle(els[els.length - 1]).overflowY; }"
)
assert overflow_y != "hidden", "the answer bubble must keep its scroll behavior"
# A long answer grows the PAGE — and the page still scrolls normally.
state0 = page.evaluate(
"() => ({ y: window.scrollY, sh: document.documentElement.scrollHeight,"
" ch: window.innerHeight })"
)
assert state0["sh"] > state0["ch"], "the long answer must make the page scrollable"
box = bubble.bounding_box()
assert box
viewport = page.viewport_size
assert viewport # the conftest `page` fixture is fixed at 1280x800
# A point in the visible lower part of the answer area (the page is
# pinned at the bottom, so the bubble's lower edge is in the viewport).
mx = box["x"] + box["width"] / 2
my = max(50.0, min(box["y"] + box["height"] - 60.0, viewport["height"] - 100))
hit = page.evaluate(
"([x, y]) => { const e = document.elementFromPoint(x, y);"
" return e ? e.tagName + '.' + String(e.className) : 'none'; }",
[mx, my],
)
page.mouse.move(mx, my)
# Headless Chromium applies wheel scrolling through an async momentum
# pipeline — let each gesture settle before reading the position.
page.mouse.wheel(0, -400) # wheel up: away from the newest content
page.wait_for_timeout(500)
y_up = page.evaluate("() => window.scrollY")
page.mouse.wheel(0, 400) # wheel back down
page.wait_for_timeout(500)
y_down = page.evaluate("() => window.scrollY")
assert y_up < state0["y"] - 100, (
f"the page must scroll up on wheel (wheel over {hit!r}): "
f"y {state0['y']} -> {y_up}"
)
assert y_down > y_up, f"the page must scroll back down on wheel: {y_up} -> {y_down}"
# ---------------------------------------------------------------------------
# 5. Regression (phase 17): a stored thinking turn restores a COLLAPSED
# block with its full text (replicates the phase-17 reload pin — the
# restore path is untouched by phase 21, where overflow is moot)
# ---------------------------------------------------------------------------
def test_restored_collapsed_thinking_unaffected(
page: Page, app_url: str, seeded_kb: None
) -> None:
page.set_default_timeout(30_000)
page.goto(app_url)
send_and_wait(page, THINK_QUESTION)
details = page.locator(".msg.brain").last.locator("details.thinking")
expect(details).not_to_have_attribute("open") # auto-collapsed
captured = details.locator(".thinking-text").text_content()
assert captured
page.reload()
expect(page.locator("#empty-state")).to_be_hidden()
restored = page.locator(".msg.brain").last.locator("details.thinking")
expect(restored).to_have_count(1)
expect(restored).not_to_have_attribute("open") # restored COLLAPSED
expect(restored.locator(".thinking-text")).to_have_text(captured)
# Opening the restored block still shows the full scratchpad, and the
# answer + persistence are intact.
restored.locator("summary").click()
expect(restored).to_have_attribute("open", "")
expect(restored.locator(".thinking-text")).to_contain_text(THINKING_FRAGMENT)
expect(page.locator(".msg.brain .bubble").last).to_contain_text(MOCK_ANSWER_MARKER)
raw = json.loads(page.evaluate(f"() => localStorage.getItem('{STORAGE_KEY}')"))[
"messages"
][1]["thinking"]
assert re.sub(r"\s+", "", raw) == re.sub(r"\s+", "", captured)