fix(ui): thinking window no longer scrolls — live 320px view pinned to the stream tail

This commit is contained in:
2026-08-24 18:24:57 -04:00
parent 76c6a01199
commit 04a7f4c05d
11 changed files with 618 additions and 279 deletions
+44 -8
View File
@@ -16,8 +16,9 @@ Implements just enough of the aipi surface:
- user message containing ``pretend to think slowly`` -> 3s warm-up delay
(used by the loading-feedback story).
- user message containing ``think out loud`` -> the answer is preceded by
~800 chars of deterministic ``reasoning_content`` chunks (the
thinking-display story, phase 17).
~2 700 chars of deterministic ``reasoning_content`` chunks (the
thinking-display story, phase 17; lengthened in phase 21 so the
rendered scratchpad overflows the 320px ``.thinking-text`` window)
- user message containing ``think out loud then hesitate`` -> the
``think out loud`` stream, then a 4s pause before the first content
frame (the sources-midstream story, phase 20 — a deterministic
@@ -167,13 +168,18 @@ def compose_answer(body: dict[str, Any]) -> str:
def compose_thinking(body: dict[str, Any]) -> str:
"""Deterministic reasoning scratchpad (thinking-display story, phase 17).
"""Deterministic reasoning scratchpad (thinking-display story, phase 17;
lengthened in phase 21).
A fixed 4-line "Step 1… Step 4" template quoting the first ~60 chars
of the user question: unique per question, byte-stable across runs,
~700–900 chars total (≈ 60–75 frames at the mock's 12-char/0.02s
pacing). The ``Step 2: Check my notes`` line fragment is what the E2E
assertions key off.
A fixed "Step 1… Step 4" template interleaved with a "Scratch" deep-dive
block, quoting the first ~60 chars of the user question: unique per
question, byte-stable across runs, ~2 700 chars total (≈ 230 frames at
the mock's 12-char/0.02s pacing). The length is deliberate (phase 21,
thinking-no-scroll story): rendered in the 320px ``.thinking-text``
window it overflows by ~2x, so the live-tail clip and the no-user-scroll
contract are observable in E2E. The ``Step 2: Check my notes`` line
fragment (phase 17) and the ``nothing is invented`` tail (phase 20's
THINKING_TAIL) are what the E2E assertions key off — both are preserved.
"""
q = _user(body).strip()[:60]
return (
@@ -183,6 +189,36 @@ def compose_thinking(body: dict[str, Any]) -> str:
"Step 2: Check my notes for the closest match. The homelab kubernetes file "
"is the obvious candidate, but I should also consider whether a deployments "
"note covers the same ground better.\n"
"Scratch 1: the kubernetes file is organized by component — control plane, "
"worker nodes, ingress, storage — so I can map each part of the question to "
"a section instead of summarizing the whole file at once, and keep the "
"answer anchored to the structure the notes actually use.\n"
"Scratch 2: I should check whether the deployments note duplicates any of "
"that ground; if it does, I will prefer the homelab file because the "
"question is phrased around the cluster itself, and I will say which file "
"each fact came from so the citation is honest.\n"
"Scratch 3: versions and ports are the facts most likely to be stale in my "
"memory — the etcd backup schedule, the ingress controller port, the "
"registry mirror address — so I will re-read those lines verbatim before "
"writing a single one of them into the answer.\n"
"Scratch 4: if the answer needs a sequence, for example how a node joins the "
"cluster or how the load balancer fronts the control plane, I will keep the "
"order exactly as the notes write it rather than re-deriving it from general "
"kubernetes knowledge that may not match this setup.\n"
"Scratch 5: anything I cannot find in the notes — a host I do not recognize, "
"a version I am not sure about, a schedule I cannot place — gets left out of "
"the answer instead of guessed, because the honesty rule beats a longer "
"answer every single time.\n"
"Scratch 6: one more pass over the question wording to make sure I am "
"answering the cluster setup, not some other homelab topic that shares the "
"same vocabulary, and I will stay on the specific the question asked about.\n"
"Scratch 7: I will also verify that the file describes the current setup — "
"if the notes mention a migration from an older cluster, I should answer "
"from the post-migration section and not mix in the old host names or the "
"old port numbers that no longer apply.\n"
"Scratch 8: final shape check before I commit — short paragraphs, a few "
"bullets at most, the document path cited where the fact came from, and no "
"invented facts anywhere in the draft.\n"
"Step 3: Re-read the relevant sections top to bottom so every specific — "
"hosts, versions, ports, schedules — is exact as written rather than "
"remembered, and note which document each fact comes from.\n"
+2 -1
View File
@@ -76,7 +76,8 @@ NEAR_BOTTOM_PX = 200
#: 0.02s/frame pace) — the wide, deterministic window to scroll away in.
LONG_QUESTION = "write a long answer about my kubernetes cluster"
#: Phase-17 thinking prefix + the long-answer trigger: both mock triggers
#: fire independently (a ~1.3s reasoning stream, then the long answer).
#: fire independently (a ~4.5s reasoning stream — lengthened in phase 21 —
#: then the long answer).
THINK_LONG_QUESTION = "think out loud — write a long answer about my kubernetes cluster"
#: The mock long answer's unique final line (mock_llm.LONG_ANSWER_END) —
#: proves the whole stream landed even while the viewport was at the top.
+3 -3
View File
@@ -250,9 +250,9 @@ def test_no_orphan_brain_message_when_navigated_before_first_token(
page.click("#send-btn")
expect(page.locator(".msg.user .bubble").last).to_contain_text(HESITATE_QUESTION)
# The mock streamed the ~800-char scratchpad (phase-17 thinking
# frames); wait until its tail is rendered — the 4s pre-content pause
# (SLOW_PRETOKEN_TRIGGER) is now running, so the navigation below
# Wait until the scratchpad's tail is rendered (phase-17 thinking body,
# ~2 700 chars / ≈4.5s, lengthened in phase 21) — the 4s pre-content
# pause (SLOW_PRETOKEN_TRIGGER) is now running, so the navigation below
# lands inside pure thinking with a wide margin.
thinking = page.locator(".msg.brain").last.locator("details.thinking")
thinking.wait_for(state="attached", timeout=10_000)
+7 -6
View File
@@ -18,10 +18,11 @@ Test → story mapping (Playwright Mapping Rule):
5. ``test_thinking_with_deflection``
Determinism note: the mock paces every SSE frame at 0.02s and the thinking
text is ~700–900 chars (≈ 60–75 frames ≈ 1.2–1.5s) before the first
content frame, so "attach → assert open" runs well inside the open window
on headless Chromium; all other assertions are made after the send button
re-enables (fully settled state).
text is ~2 700 chars (≈ 230 frames ≈ 4.5s — lengthened in phase 21 so the
scratchpad overflows the 320px window) before the first content frame, so
"attach → assert open" runs well inside the open window on headless
Chromium; all other assertions are made after the send button re-enables
(fully settled state).
"""
from __future__ import annotations
@@ -139,8 +140,8 @@ def test_thinking_block_streams_open_then_collapses(
# token — and is created OPEN.
details = page.locator(".msg.brain").last.locator("details.thinking")
details.wait_for(state="attached", timeout=10_000)
# The ~800-char thinking stream (≈1.3s) keeps the block open right
# after attach — assert while it is still streaming.
# The ~2 700-char thinking stream (≈4.5s, phase 21) keeps the block
# open right after attach — assert while it is still streaming.
expect(details).to_have_attribute("open", "")
expect(details.locator(".thinking-text")).not_to_have_text("")
+455
View File
@@ -0,0 +1,455 @@
"""Phase 21 E2E (Playwright, mock-only): the Thinking window is a live tail.
Story: ``.agent/user_stories/thinking-no-scroll.md``
Run in isolation (DB must be up: ``podman compose up -d db``):
uv run pytest tests/e2e/test_thinking_no_scroll.py -v --no-cov
MOCK-ONLY suite: ``E2E_REAL_LLM=1`` is not supported here — the real
``turbo`` decides its own reasoning length and pacing, and this story's
contract (fixed 320px clip, no user scroll back, programmatic pin to the
live tail) needs the deterministic mock's long, paced scratchpad.
The whole functional change is one CSS property
(``details.thinking .thinking-text``: ``overflow-y: auto`` → ``hidden``);
``overflow: hidden`` still allows the phase-17 programmatic bottom-pin
(``scrollTop = scrollHeight`` per thinking chunk), which is the sole
scroller. This suite proves the browser behavior the unit pins
(``tests/unit/test_thinking_no_scroll.py``) can only pin at source level.
Determinism note: phase 21 lengthened the mock's ``compose_thinking``
body to ~2 700 chars (≈230 frames at the mock's 12-char/0.02s pacing ≈
4.5s) so the rendered scratchpad overflows the 320px window by ~2x.
Tests 1–2 key off the mock's ``think out loud then hesitate`` trigger:
after the thinking stream ends there is a deterministic 4s pre-content
pause with the block still OPEN and no further pin frames — a frozen
live tail, the only state where a (regressed, working) user scroll would
persist and be observable. During live streaming the per-chunk re-pin
masks any user scroll within one frame (≈20ms), so that state is covered
by the tail-tracking invariant instead (test 2, sampled mid-stream).
Test → story mapping (Playwright Mapping Rule):
1. ``test_thinking_window_not_user_scrollable`` — wheel / drag / keyboard
on the frozen live tail do not move the window.
2. ``test_thinking_window_tracks_live_tail`` — after the 2nd-to-last and
the last thinking chunk the window is pinned to the tail and the last
chunk's text sits inside the visible rectangle.
3. ``test_thinking_window_css_contract`` — computed ``overflow-y`` is
``hidden``, ``max-height`` is 320px, and the clip is real.
4. ``test_answer_bubble_still_scrollable`` — regression (phase 11): the
answer bubble's overflow is untouched and the page still scrolls.
5. ``test_restored_collapsed_thinking_unaffected`` — regression
(phase 17): a stored thinking turn restores a collapsed block.
"""
from __future__ import annotations
import asyncio
import json
import re
from collections.abc import Iterator
from pathlib import Path
from threading import Thread
from typing import Any
import pytest
from playwright.sync_api import Page, expect
from sqlalchemy import text
from app.config import Settings
from app.db import SessionLocal
from app.rag.importer import ImportSummary, import_sources
from app.rag.llm import LLMClient
from e2e.mock_llm import compose_thinking
REPO = Path(__file__).resolve().parents[2]
FIXTURES = REPO / "tests" / "fixtures" / "docs"
#: Phase-17 trigger question (grounded turn, thinking + short answer).
THINK_QUESTION = "think out loud — how is my kubernetes cluster set up?"
#: Phase-20 hesitation trigger: the long phase-17/21 thinking stream, then
#: a deterministic 4s pause before the first content frame — a frozen
#: live tail with the block still open (the no-scroll test window).
HESITATE_QUESTION = "think out loud then hesitate — how is my kubernetes cluster set up?"
#: Phase-11 long-answer trigger (regression test 4).
LONG_QUESTION = "How is my Kubernetes cluster set up? write a long answer"
LONG_ANSWER_END = "LONG-ANSWER-END"
MOCK_ANSWER_MARKER = "Deterministic mock answer for E2E"
THINKING_FRAGMENT = "Step 2: Check my notes"
STORAGE_KEY = "bor.chat.v1"
SELECTOR = ".msg.brain details.thinking .thinking-text"
async def _import_fixtures(mock_port: int) -> ImportSummary:
kwargs: dict[str, Any] = {"_env_file": None, "llm_base_url": f"http://127.0.0.1:{mock_port}/v1"}
settings = Settings(**kwargs) # pyright: ignore[reportCallIssue]
return await import_sources([FIXTURES], LLMClient(settings))
def _run_in_thread(coro: Any) -> Any:
"""Run a coroutine on a worker thread (Playwright owns the test loop)."""
box: dict[str, Any] = {}
def runner() -> None:
try:
box["value"] = asyncio.run(coro)
except BaseException as e: # noqa: BLE001 — re-raised on the test thread
box["error"] = e
t = Thread(target=runner)
t.start()
t.join()
if "error" in box:
raise box["error"]
return box["value"]
def _reset_db(mock_port: int, seed: bool) -> ImportSummary | None:
"""Truncate the KB (and query log), then optionally re-import fixtures."""
with SessionLocal() as db:
db.execute(text("TRUNCATE chunks, documents, query_log"))
db.commit()
if not seed:
return None
return _run_in_thread(_import_fixtures(mock_port))
@pytest.fixture()
def seeded_kb(mock_llm: int, db_ready: None) -> Iterator[None]:
"""A fresh KB seeded from ``tests/fixtures/docs`` (8 docs, A9 formats),
truncated again on teardown (same fixture shape as the phase-17 suite)."""
summary = _reset_db(mock_llm, seed=True)
assert summary is not None and summary.added == 8
yield
_reset_db(mock_llm, seed=False)
def send_and_wait(page: Page, question: str) -> None:
"""Type into #message-input, submit via #composer, then wait until the
last brain message settles (send button re-enabled, label "Send")."""
page.fill("#message-input", question)
page.evaluate("() => document.querySelector('#composer').requestSubmit()")
expect(page.locator(".msg.user .bubble").last).to_contain_text(question)
# Thinking (~4.5s, phase 21) + answer land in a few seconds.
expect(page.locator(".msg.brain .bubble").last).not_to_have_text("", timeout=30_000)
expect(page.locator("#send-btn")).to_be_enabled(timeout=30_000)
expect(page.locator("#send-label")).to_have_text("Send")
def _scroll_sample(page: Page) -> dict[str, float]:
"""scrollTop / scrollHeight / clientHeight of the live Thinking window."""
return page.evaluate(
f"""() => {{ const el = document.querySelector('{SELECTOR}');
return {{ top: el.scrollTop, height: el.scrollHeight, client: el.clientHeight }}; }}"""
)
def _at_tail(sample: dict[str, float]) -> bool:
"""True when the window is pinned to the live tail: the bottom of the
content is visible (scrollTop clamped at scrollHeight - clientHeight,
within 1px — the phase-17 pin's effect)."""
return abs(sample["top"] - (sample["height"] - sample["client"])) <= 1
def _wait_text_stable(page: Page, timeout_ms: int = 30_000) -> None:
"""Wait until the scratchpad text stops growing for 300ms.
The mock paces frames at 0.02s, so a 300ms still length means the
thinking stream has ended — with the hesitation trigger, the 4s
pre-content pause (block still open, no further pin frames) is then
running."""
page.wait_for_function(
f"""() => {{ const el = document.querySelector('{SELECTOR}');
if (!el) return false;
const len = el.innerText.length;
const now = performance.now();
if (!window.__thinkProbe) window.__thinkProbe = {{ len, at: now }};
const p = window.__thinkProbe;
if (len !== p.len) {{ p.len = len; p.at = now; return false; }}
return now - p.at >= 300; }}""",
timeout=timeout_ms,
)
def _wait_text_contains(page: Page, marker: str, timeout_ms: int = 30_000) -> None:
"""Wait until the rendered scratchpad (whitespace-insensitive) contains
``marker`` — a deterministic probe for a given point in the stream."""
page.wait_for_function(
f"""(tail) => {{ const el = document.querySelector('{SELECTOR}');
return !!el && el.innerText.replace(/\\s+/g, '').includes(tail); }}""",
arg=marker,
timeout=timeout_ms,
)
# ---------------------------------------------------------------------------
# 1. No user scroll back: wheel, drag, and keyboard on the frozen live tail
# do not move the window
# ---------------------------------------------------------------------------
def test_thinking_window_not_user_scrollable(
page: Page, app_url: str, seeded_kb: None
) -> None:
page.set_default_timeout(30_000)
page.goto(app_url)
page.fill("#message-input", HESITATE_QUESTION)
page.evaluate("() => document.querySelector('#composer').requestSubmit()")
expect(page.locator(".msg.user .bubble").last).to_contain_text(HESITATE_QUESTION)
details = page.locator(".msg.brain").last.locator("details.thinking")
details.wait_for(state="attached", timeout=10_000)
text_el = details.locator(".thinking-text")
# Premise: the long scratchpad overflows the 320px clip.
page.wait_for_function(
f"() => {{ const el = document.querySelector('{SELECTOR}');"
" return !!el && el.scrollHeight > el.clientHeight; }",
timeout=30_000,
)
# The thinking stream has ENDED (4s hesitation pause running): no more
# pin frames, the block is still open, and the tail is frozen — any
# user scroll would persist and be observable here.
_wait_text_stable(page)
expect(details).to_have_attribute("open", "")
expect(page.locator(".msg.brain .bubble").last).to_have_text("")
# Precondition: the phase-17 pin left the window at the live tail.
before = _scroll_sample(page)
assert before["height"] > before["client"], "the window must overflow"
assert _at_tail(before), "the pin must have left the window at the tail"
# Focus the window (a plain div is not focusable — el.focus() is a
# no-op; the keyboard presses below land on the page focus instead.
# Neither path may move the window).
text_el.evaluate("el => el.focus()")
box = text_el.bounding_box()
assert box
cx, cy = box["x"] + box["width"] / 2, box["y"] + box["height"] / 2
page.mouse.move(cx, cy)
# Wheel back (up) — must not scroll the window.
page.mouse.wheel(0, -200)
# Keyboard: Home + ArrowUp — must not scroll the window.
page.keyboard.press("Home")
page.keyboard.press("ArrowUp")
page.keyboard.press("ArrowUp")
# Mouse drag over the window — must not scroll the window
# (there is no scrollbar to grab with overflow hidden).
page.mouse.down()
page.mouse.move(cx, cy - 100, steps=5)
page.mouse.up()
# Wheel again, then let a would-be (regressed) scroll settle.
page.mouse.wheel(0, -200)
page.wait_for_timeout(150)
# Still inside the pure-thinking window (the assertions below are only
# meaningful while the block is open and no content frame has landed).
expect(details).to_have_attribute("open", "")
after = _scroll_sample(page)
assert abs(after["top"] - before["top"]) <= 1, (
f"user scroll moved the window: {before['top']} -> {after['top']}"
)
assert _at_tail(after), "the window must still show the live tail"
# ---------------------------------------------------------------------------
# 2. Live tail tracking: the per-chunk pin keeps the window glued to the
# newest content (sampled at the 2nd-to-last and last chunks), and the
# last chunk's text is inside the visible rectangle
# ---------------------------------------------------------------------------
def test_thinking_window_tracks_live_tail(
page: Page, app_url: str, seeded_kb: None
) -> None:
page.set_default_timeout(30_000)
page.goto(app_url)
page.fill("#message-input", HESITATE_QUESTION)
page.evaluate("() => document.querySelector('#composer').requestSubmit()")
expect(page.locator(".msg.user .bubble").last).to_contain_text(HESITATE_QUESTION)
details = page.locator(".msg.brain").last.locator("details.thinking")
details.wait_for(state="attached", timeout=10_000)
# The exact deterministic scratchpad the mock will stream, sliced the
# same way the mock's _sse_stream does (12-char chunks).
expected = compose_thinking(
{"messages": [{"role": "user", "content": HESITATE_QUESTION}]}
)
pieces = re.findall(r".{1,12}", expected, re.S)
ws = re.sub(r"\s+", "", "".join(pieces))
#: 12 rendered chars ending at the 2nd-to-last chunk.
marker_second_last = re.sub(r"\s+", "", "".join(pieces[:-1]))[-12:]
#: 12 rendered chars at the very end (the last chunk).
marker_last = ws[-12:]
# During the stream: once the 2nd-to-last chunk has landed, the
# window is pinned to the tail (the invariant holds at EVERY chunk;
# the pin runs per chunk while the block is open).
_wait_text_contains(page, marker_second_last)
assert _at_tail(_scroll_sample(page)), "not at the tail after chunk N-1"
# After the last chunk: the 4s hesitation pause holds this state with
# the block still open — sample the tail pin, then the geometry.
_wait_text_contains(page, marker_last)
_wait_text_stable(page)
expect(details).to_have_attribute("open", "")
sample = _scroll_sample(page)
assert _at_tail(sample), f"not at the tail after the last chunk: {sample}"
# Geometry: the last chunk's text (the final text node of the
# scratchpad) renders INSIDE the visible rectangle, and a hit-test at
# the box's bottom lands inside .thinking-text.
geo = page.evaluate(
f"""() => {{ const el = document.querySelector('{SELECTOR}');
const box = el.getBoundingClientRect();
const walker = document.createTreeWalker(el, NodeFilter.SHOW_TEXT);
let last = null;
while (walker.nextNode()) last = walker.currentNode;
const range = document.createRange();
range.selectNodeContents(last);
const r = range.getBoundingClientRect();
const hit = document.elementFromPoint(box.left + 10, box.bottom - 5);
return {{
nodeVisible: r.bottom > box.top && r.top < box.bottom,
nodeBottomInBox: r.bottom <= box.bottom + 1,
hitInside: hit ? el.contains(hit) : false,
}}; }}"""
)
assert geo["nodeVisible"], "the last chunk's text is outside the window"
assert geo["nodeBottomInBox"], "the last chunk's text is clipped off the bottom"
assert geo["hitInside"], "a hit-test at the box bottom missed .thinking-text"
# ---------------------------------------------------------------------------
# 3. CSS contract: overflow-y hidden, 320px max-height, and the clip is real
# ---------------------------------------------------------------------------
def test_thinking_window_css_contract(
page: Page, app_url: str, seeded_kb: None
) -> None:
page.set_default_timeout(30_000)
page.goto(app_url)
send_and_wait(page, THINK_QUESTION)
details = page.locator(".msg.brain").last.locator("details.thinking")
expect(details).not_to_have_attribute("open") # auto-collapsed
details.locator("summary").click() # open for measurement
expect(details).to_have_attribute("open", "")
style = page.evaluate(
f"""() => {{ const el = document.querySelector('{SELECTOR}');
const cs = getComputedStyle(el);
return {{ overflowY: cs.overflowY, maxHeight: cs.maxHeight,
scrollHeight: el.scrollHeight, clientHeight: el.clientHeight }}; }}"""
)
assert style["overflowY"] == "hidden", "the window must not be user-scrollable"
assert style["maxHeight"] == "320px", "the 320px clip must stay"
# The clip is real, not cosmetic: the long scratchpad overflows it.
assert style["scrollHeight"] > style["clientHeight"]
# ---------------------------------------------------------------------------
# 4. Regression (phase 11): the answer bubble is untouched — a long answer
# still grows the page, and normal (user) scrolling of the answer works
# ---------------------------------------------------------------------------
def test_answer_bubble_still_scrollable(
page: Page, app_url: str, seeded_kb: None
) -> None:
page.set_default_timeout(45_000)
page.goto(app_url)
page.fill("#message-input", LONG_QUESTION)
page.click("#send-btn")
expect(page.locator(".msg.user .bubble").last).to_contain_text(LONG_QUESTION)
# The ~900-word answer streams to completion (phase-11 contract).
bubble = page.locator(".msg.brain .bubble").last
expect(bubble).to_contain_text(LONG_ANSWER_END, timeout=60_000)
expect(page.locator("#send-btn")).to_be_enabled()
# The answer bubble keeps its existing overflow (phase 21 only touched
# .thinking-text) — it is NOT the hidden clip.
overflow_y = page.evaluate(
"() => { const els = document.querySelectorAll('.msg.brain .bubble');"
" return getComputedStyle(els[els.length - 1]).overflowY; }"
)
assert overflow_y != "hidden", "the answer bubble must keep its scroll behavior"
# A long answer grows the PAGE — and the page still scrolls normally.
state0 = page.evaluate(
"() => ({ y: window.scrollY, sh: document.documentElement.scrollHeight,"
" ch: window.innerHeight })"
)
assert state0["sh"] > state0["ch"], "the long answer must make the page scrollable"
box = bubble.bounding_box()
assert box
viewport = page.viewport_size
assert viewport # the conftest `page` fixture is fixed at 1280x800
# A point in the visible lower part of the answer area (the page is
# pinned at the bottom, so the bubble's lower edge is in the viewport).
mx = box["x"] + box["width"] / 2
my = max(50.0, min(box["y"] + box["height"] - 60.0, viewport["height"] - 100))
hit = page.evaluate(
"([x, y]) => { const e = document.elementFromPoint(x, y);"
" return e ? e.tagName + '.' + String(e.className) : 'none'; }",
[mx, my],
)
page.mouse.move(mx, my)
# Headless Chromium applies wheel scrolling through an async momentum
# pipeline — let each gesture settle before reading the position.
page.mouse.wheel(0, -400) # wheel up: away from the newest content
page.wait_for_timeout(500)
y_up = page.evaluate("() => window.scrollY")
page.mouse.wheel(0, 400) # wheel back down
page.wait_for_timeout(500)
y_down = page.evaluate("() => window.scrollY")
assert y_up < state0["y"] - 100, (
f"the page must scroll up on wheel (wheel over {hit!r}): "
f"y {state0['y']} -> {y_up}"
)
assert y_down > y_up, f"the page must scroll back down on wheel: {y_up} -> {y_down}"
# ---------------------------------------------------------------------------
# 5. Regression (phase 17): a stored thinking turn restores a COLLAPSED
# block with its full text (replicates the phase-17 reload pin — the
# restore path is untouched by phase 21, where overflow is moot)
# ---------------------------------------------------------------------------
def test_restored_collapsed_thinking_unaffected(
page: Page, app_url: str, seeded_kb: None
) -> None:
page.set_default_timeout(30_000)
page.goto(app_url)
send_and_wait(page, THINK_QUESTION)
details = page.locator(".msg.brain").last.locator("details.thinking")
expect(details).not_to_have_attribute("open") # auto-collapsed
captured = details.locator(".thinking-text").text_content()
assert captured
page.reload()
expect(page.locator("#empty-state")).to_be_hidden()
restored = page.locator(".msg.brain").last.locator("details.thinking")
expect(restored).to_have_count(1)
expect(restored).not_to_have_attribute("open") # restored COLLAPSED
expect(restored.locator(".thinking-text")).to_have_text(captured)
# Opening the restored block still shows the full scratchpad, and the
# answer + persistence are intact.
restored.locator("summary").click()
expect(restored).to_have_attribute("open", "")
expect(restored.locator(".thinking-text")).to_contain_text(THINKING_FRAGMENT)
expect(page.locator(".msg.brain .bubble").last).to_contain_text(MOCK_ANSWER_MARKER)
raw = json.loads(page.evaluate(f"() => localStorage.getItem('{STORAGE_KEY}')"))[
"messages"
][1]["thinking"]
assert re.sub(r"\s+", "", raw) == re.sub(r"\s+", "", captured)