feat(rag): retry a failed LLM request before the first token lands — BOR_LLM_RETRIES/BOR_LLM_RETRY_DELAY with a live 'retrying' status

This commit is contained in:
2026-09-02 10:52:38 -04:00
parent f04ddbe1f8
commit 88293ed02f
44 changed files with 2488 additions and 56 deletions
+9
View File
@@ -98,6 +98,15 @@ def app_server(mock_llm: int) -> Iterator[str]:
# The production default stays 0.62 (re-tuned against the real
# `embed` model's 0.41–0.84 cosine range, PLAN A8).
env["BOR_RELEVANCE_THRESHOLD"] = "0.30"
# Phase 67 (LLM retry): the e2e pins the retry MECHANISM with instant
# waits (BOR_LLM_RETRY_DELAY=0 — the 5 s default is unit-pinned via
# tests/unit/test_config.py). BOR_LLM_RETRIES is forced to the code
# default (derived from the class field, never drifts from
# app/config.py) so the exhaustion test relies on the REAL budget and
# an operator's local (gitignored) .env cannot leak a different one
# into the app under test (the phase-61 leak-guard pattern).
env["BOR_LLM_RETRY_DELAY"] = "0"
env["BOR_LLM_RETRIES"] = str(_Settings.model_fields["llm_retries"].default)
env.setdefault("BOR_DATABASE_URL", "postgresql+psycopg://reese:reese@localhost:5432/brain_of_reese")
# Phase 16: admin auth must be set or create_app() refuses to boot.
env["BOR_ADMIN_PASSWORD"] = ADMIN_PASSWORD
+123 -2
View File
@@ -112,6 +112,36 @@ Implements just enough of the aipi surface:
answer; the E2E asks it against an on-topic fixture (HIGH gate) and
asserts non-deflection.
Failure injection (phase 67, LLM retry, TODO.md L3) — deterministic
dead-endpoint behavior for the retry E2E suite (``tests/e2e/
test_llm_retry.py``). The mock is single-conversation per e2e server, so
the sequences are driven by module-level counters that reset per
trigger phrase after the success they guard (a second question with
the same trigger re-drives the sequence from zero):
- user message containing ``fail then answer`` (``RETRY_TRIGGER``):
the first ``RETRY_DEAD_ATTEMPTS`` (2) app-level streaming attempts
respond 500 (JSON body, like a dead proxy) and the third streams
the normal composed answer — 2 = 1 original attempt + 1 retry under
the default ``BOR_LLM_RETRIES=3``, so a suite exercises a real
retry without waiting for the 4-attempt exhaustion. Counted in
APP-LEVEL attempts, not raw HTTP POSTs: while the endpoint stays
dead, the openai SDK's default policy (max_retries=2 — the app's
``LLMClient`` keeps it) re-POSTs a 500'd streaming request twice
before surfacing the error, so each dead attempt costs exactly 3
POSTs (``_HTTPS_PER_DEAD_ATTEMPT``).
- user message containing ``always fail``
(``ALWAYS_FAIL_TRIGGER``): EVERY streaming chat/completions request
responds 500 — the retry-budget exhaustion path (the terminal
error banner in the UI).
- embeddings request whose input contains ``embed fail once``
(``EMBED_FAIL_TRIGGER``): the FIRST such request responds 500, the
next returns the normal bag-of-words vector — the endpoint's
pre-stream embedding retry loop. Raw httpx on the client side (no
SDK-level retries), so one POST per attempt: the counter is per
POST here, unlike the chat counter above.
Non-streaming requests (document summaries, KB overview) never 500 —
the retry scope is the chat turn only (owner-locked A1).
``max_tokens`` is honored deterministically (token ≈ whitespace word),
like a real endpoint: an answer longer than the cap is truncated. This
is what makes the phase-11 truncation regression observable.
@@ -129,7 +159,7 @@ import uuid
from typing import Any
from fastapi import FastAPI
from fastapi.responses import StreamingResponse
from fastapi.responses import JSONResponse, StreamingResponse
app = FastAPI()
@@ -265,6 +295,76 @@ TABLE_ANSWER = (
"| value-one | value-two | value-three | value-four | value-five |"
)
# ---------------------------------------------------------------------------
# Phase 67 (LLM retry, TODO.md L3): deterministic failure injection
# ---------------------------------------------------------------------------
#: A user message containing this substring (case-insensitive) gets
#: ``RETRY_DEAD_ATTEMPTS`` dead streaming attempts (500, JSON body) before
#: the normal composed answer streams — 1 original attempt + 1 retry under
#: the default ``BOR_LLM_RETRIES=3`` (see the module docstring).
RETRY_TRIGGER = "fail then answer"
#: App-level attempts the endpoint stays dead for before the answer.
RETRY_DEAD_ATTEMPTS = 2
#: A user message containing this substring (case-insensitive) makes
#: EVERY streaming chat/completions request respond 500 — the
#: retry-budget exhaustion path (the terminal error banner in the UI).
ALWAYS_FAIL_TRIGGER = "always fail"
#: An embeddings request whose input contains this substring
#: (case-insensitive) 500s on its FIRST POST; the next returns the normal
#: bag-of-words vector — the endpoint's pre-stream embedding retry loop.
EMBED_FAIL_TRIGGER = "embed fail once"
#: One DEAD app-level chat attempt costs exactly this many HTTP POSTs
#: while the endpoint stays down: the openai SDK's default policy
#: (max_retries=2 — the app's ``LLMClient`` keeps it) re-POSTs a 500'd
#: streaming request twice before surfacing the error to
#: ``chat_stream_retried``. The failure counters below therefore count
#: app-level attempts (groups of this size), not raw POSTs — the visible
#: sequence (one SSE ``retry`` frame after each dead attempt, the answer
#: on the third) stays deterministic regardless of the SDK's internal
#: backoff pacing.
_HTTPS_PER_DEAD_ATTEMPT = 3
#: Module-level failure counters — the mock is single-conversation per
#: e2e server. Keyed by trigger phrase (reset per trigger): the number
#: of matching POSTs served so far. Each sequence resets after the
#: success it guards, so a second question carrying the same trigger
#: re-drives the failure sequence from zero.
_fail_posts: dict[str, int] = {}
def _llm_500(why: str) -> JSONResponse:
"""A dead-proxy 500 with a JSON error body (phase 67 injection)."""
return JSONResponse(
status_code=500,
content={
"error": {
"message": f"upstream connection reset ({why})",
"type": "proxy_error",
}
},
)
def _bump_fail(key: str) -> int:
n = _fail_posts.get(key, 0) + 1
_fail_posts[key] = n
return n
def _chat_dead(key: str, dead_attempts: int) -> bool:
"""Bump *key*'s counter; True while the endpoint stays dead.
Counted in app-level attempts (see ``_HTTPS_PER_DEAD_ATTEMPT``): the
first ``dead_attempts * _HTTPS_PER_DEAD_ATTEMPT`` POSTs 500 and the
next attempt's first POST streams (the caller resets the counter on
the success).
"""
return _bump_fail(key) <= dead_attempts * _HTTPS_PER_DEAD_ATTEMPT
#: The agent's ``read_document`` tool-result prefix (app.rag.agent
#: ``_execute_tool``): ``"Document <source/path>:\n<content>"``.
@@ -656,11 +756,22 @@ def models() -> dict[str, Any]:
@app.post("/v1/embeddings")
def embeddings(body: dict[str, Any]) -> dict[str, Any]:
def embeddings(body: dict[str, Any]) -> Any: # dict, or a 500 (phase 67)
raw = body.get("input")
if isinstance(raw, str):
raw = [raw]
inputs: list[Any] = list(raw) if isinstance(raw, list) else []
# Phase 67 (embedding retry): the first embeddings request whose
# input carries the marker 500s; the next returns the normal
# bag-of-words vector (see the module docstring). Raw httpx on the
# client side — no SDK-level retries — so one POST per app attempt:
# the counter is per POST here (unlike the chat counter below).
joined = " ".join(str(t) for t in inputs if isinstance(t, str)).lower()
if EMBED_FAIL_TRIGGER in joined:
n = _bump_fail(EMBED_FAIL_TRIGGER)
if n == 1:
return _llm_500(EMBED_FAIL_TRIGGER)
_fail_posts[EMBED_FAIL_TRIGGER] = 0 # the vector went out — restart
data = [
{"object": "embedding", "index": i, "embedding": embed_text(t)}
for i, t in enumerate(inputs)
@@ -827,6 +938,16 @@ def chat_completions(body: dict[str, Any]) -> Any:
# the flow handles streaming requests; a non-streaming marker request
# (never issued by the app) falls through to the regular answer.
if body.get("stream"):
# Phase 67 (LLM retry): deterministic failure injection — see
# the module docstring. Checked before the marker tool flow: the
# injection markers never combine with the tool-flow markers in
# any suite, and a dead endpoint answers nothing (no flow).
if ALWAYS_FAIL_TRIGGER in user_lower:
return _llm_500(ALWAYS_FAIL_TRIGGER)
if RETRY_TRIGGER in user_lower:
if _chat_dead(RETRY_TRIGGER, RETRY_DEAD_ATTEMPTS):
return _llm_500(RETRY_TRIGGER)
_fail_posts[RETRY_TRIGGER] = 0 # the answer streamed — restart
flow = _tool_flow(body)
if flow is not None:
if flow[0] == "list":
+506
View File
@@ -0,0 +1,506 @@
"""Phase 67 E2E (Playwright): LLM retry with live "Trying Again" feedback.
Source: ``TODO.md`` L3 — "Add a .env configurable retry in case the LLM
server fails to respond. Allow 3 retries by default, with 5 seconds
between each retry. Update the user interface to show 'communication
interrupted, trying again' … if the LLM server stops communicating."
(TODO-derived phase — no story file.)
Run in isolation (DB must be up: ``podman compose up -d db``):
uv run pytest tests/e2e/test_llm_retry.py -v --no-cov
MOCK-ONLY suite: ``E2E_REAL_LLM=1`` is not supported — the retry gate is
the deterministic failure injection in ``tests/e2e/mock_llm.py``:
* ``fail then answer`` (``RETRY_TRIGGER``): the first 2 app-level
streaming attempts 500 (JSON body, like a dead proxy) and the third
streams the normal composed answer — 1 original attempt + 1 retry
under the default ``BOR_LLM_RETRIES=3``. (Each dead app attempt costs
3 HTTP POSTs — the openai SDK's default 1 + 2 internal retries — so
the mock counts attempts, not POSTs; see the mock docstring.)
* ``always fail`` (``ALWAYS_FAIL_TRIGGER``): every streaming request
500s — the retry-budget exhaustion path.
* ``embed fail once`` (``EMBED_FAIL_TRIGGER``): the first embeddings
request 500s, the next returns the normal vector — the endpoint's
pre-stream embedding retry loop.
The e2e app server boots with ``BOR_LLM_RETRY_DELAY=0`` (conftest) so
the retry waits are instant — the suite pins the MECHANISM; the 5 s
default is unit-pinned via ``tests/unit/test_config.py``.
``BOR_LLM_RETRIES`` is forced to its real default (3, conftest) — the
exhaustion test relies on the real budget: 4 attempts total, so the
last ``retry`` frame reads "(4 of 4)".
KB seed (phase-37 direct-seed pattern): ONE fixture document
(``homelab/kubernetes.md``), one chunk carrying the mock's own
bag-of-words embedding. The marker questions were verified against this
exact seed (``plan_turn``, E2E threshold 0.30):
* the "sourdough" questions (DEFLECT_Q / EXHAUST_Q) share no FTS token
with the document (cosine 0.000, fts 0) → LOW → the DEFLECTED path
(``chat_stream_retried`` directly, no agent round);
* the "kubernetes cluster" questions (GROUNDED_Q / EMBED_Q) FTS-hit the
document (cosine 0.171, fts 1) → HIGH → the GROUNDED path (the agent
loop's per-round retry).
Test → source mapping:
1. ``test_dead_then_recovered_deflected`` — LOW turn: the recorded
``#send-status`` values contain "Communication interrupted —
retrying (2 of 4)…" and "(3 of 4)…" (the owner-locked A4 copy), the
wire carries the two ``retry`` frames ahead of the first delta, the
deflected answer settles, and no error banner appears.
2. ``test_dead_then_recovered_grounded`` — HIGH turn: the same status +
wire assertions; the grounded answer completes with the source chip
(the agent round retried, the turn is intact).
3. ``test_embedding_retry_completes`` — ``embed fail once``: the
pre-stream embedding retry is visible to the UI (a ``retry`` status
before any answer frame) and the turn completes normally.
4. ``test_exhaustion_lands_on_the_error_banner`` — ``always fail``:
after the 4th dead attempt the EXISTING terminal error banner
appears (role=alert, the "dropped the connection" copy), the last
retrying status is the highest attempt — "(4 of 4)…" — and the send
button re-enables (the banner path settles the state machine).
"""
from __future__ import annotations
import hashlib
import json
import re
import time
from datetime import UTC, datetime
from pathlib import Path
from playwright.sync_api import Page, expect
from sqlalchemy import select, text
from sqlalchemy.orm import Session
from app.db import SessionLocal
from app.models import Chunk, Document, QueryLog
from tests.e2e.mock_llm import embed_text
REPO = Path(__file__).resolve().parents[2]
# --------------------------------------------------------------------------
# Seed + questions (see the module docstring for the gate verification)
# --------------------------------------------------------------------------
SEED_SOURCE = "docs"
SEED_PATH = "homelab/kubernetes.md"
SEED_SP = f"{SEED_SOURCE}/{SEED_PATH}"
KUB_CONTENT = (REPO / "tests" / "fixtures" / "docs" / SEED_PATH).read_text()
#: LOW turn (deflected path) + the retry trigger: no FTS overlap with the
#: seeded document, cosine 0.000 → the honesty gate deflects.
DEFLECT_Q = "How do I bake sourdough bread? fail then answer"
#: HIGH turn (grounded path) + the retry trigger: FTS-hits the seeded
#: document → the agent loop runs (and its single round is retried).
GROUNDED_Q = "How is my Kubernetes cluster set up? fail then answer"
#: HIGH turn + the embedding trigger: the pre-stream embedding 500s once.
EMBED_Q = "How is my Kubernetes cluster set up? embed fail once"
#: LOW turn + the exhaustion trigger: every streaming request 500s.
EXHAUST_Q = "How do I bake sourdough bread? always fail"
#: The conftest forces BOR_LLM_RETRIES to its default (3) → 4 attempts
#: total; the attempt math in the assertions is fixed by that budget.
MAX_ATTEMPTS = 4
def _retry_status(attempt: int) -> str:
"""The owner-locked A4 copy for *attempt* (1-based) of MAX_ATTEMPTS."""
return f"Communication interrupted — retrying ({attempt} of {MAX_ATTEMPTS})…"
MOCK_ANSWER_MARKER = "Deterministic mock answer for E2E"
DEFLECT_PHRASE = r"haven't done anything like that"
ERROR_COPY = "The chat model dropped the connection — try again?"
# --------------------------------------------------------------------------
# DB seeding (TRUNCATE-then-seed, cf. test_agent_document_tools.py)
# --------------------------------------------------------------------------
def _seed(db: Session) -> None:
"""The single fixture document (see the module docstring)."""
md = Document(
source=SEED_SOURCE,
path=SEED_PATH,
full_path=f"/tmp/{SEED_PATH}",
title="Kubernetes",
content=KUB_CONTENT,
content_hash=hashlib.sha256(KUB_CONTENT.encode()).hexdigest(),
indexed_at=datetime.now(UTC),
)
db.add(md)
db.flush()
# One chunk carrying the mock's own embedding → genuine token overlap
# for the grounded questions (the FTS path carries them to HIGH).
db.add(
Chunk(
document_id=md.id,
position=0,
content=KUB_CONTENT,
embedding=embed_text(KUB_CONTENT),
)
)
def _reset_db() -> None:
"""Truncate the KB (plus the prompt-shaping tables), then re-seed.
``steering_notes`` / ``kb_overview`` are truncated too, so the
prompts are exactly ``<relevance>`` + ``<documents>`` (+ ``<tools>``
on HIGH) regardless of leftovers from other suites — byte-stable
prompts, byte-stable answers.
"""
with SessionLocal() as db:
db.execute(
text("TRUNCATE chunks, documents, query_log, steering_notes, kb_overview")
)
db.commit()
_seed(db)
db.commit()
def _last_query_log() -> QueryLog:
with SessionLocal() as db:
rows = db.scalars(select(QueryLog)).all()
assert len(rows) == 1, f"expected exactly one query_log row, got {len(rows)}"
return rows[0]
# --------------------------------------------------------------------------
# Page hooks (the #send-status recorder + SSE capture, the phase-37
# pattern from test_agent_document_tools.py)
# --------------------------------------------------------------------------
#: Records every value #send-status takes during the turn (a
#: MutationObserver on the element), so the in-flight status sequence —
#: including the transient phase-67 "retrying" states — is captured
#: deterministically (no polling race).
STATUS_RECORDER = """
() => {
if (window.__statusesInstalled) return;
window.__statusesInstalled = true;
window.__statuses = [];
const el = document.querySelector('#send-status');
if (!el) return;
const rec = (v) => {
const l = window.__statuses;
if (!l.length || l[l.length - 1] !== v) l.push(v);
};
rec(el.textContent);
new MutationObserver(() => rec(el.textContent)).observe(el, {
childList: true,
subtree: true,
});
}
"""
#: Captures the raw SSE ``data:`` payloads of the /api/chat stream
#: (a response clone read in the background) — wire-level assertions
#: for the ``retry`` frames, independent of the UI rendering.
SSE_HOOK = """
() => {
if (window.__sseInstalled) return;
window.__sseInstalled = true;
window.__sseFrames = [];
const origFetch = window.fetch;
window.fetch = async function (...args) {
const res = await origFetch.apply(this, args);
try {
const url = typeof args[0] === 'string' ? args[0] : args[0].url;
if (url.includes('/api/chat')) {
res.clone().text().then((bodyText) => {
for (const block of bodyText.split('\\n\\n')) {
const line = block.trim();
if (line.startsWith('data: ')) {
window.__sseFrames.push(line.slice(6));
}
}
});
}
} catch (e) { /* non-clonable responses: ignored */ }
return res;
};
}
"""
def _install_page_hooks(page: Page) -> None:
"""Install both hooks on the loaded page (post-goto, pre-submit).
The fetch wrapper only needs to be in place before the turn's
``fetch("/api/chat")`` call; the observer needs the rendered
``#send-status``. (``add_init_script`` would not do — it binds to
the NEXT navigation, and the story page is navigated exactly once.)
"""
page.evaluate(SSE_HOOK)
page.evaluate(STATUS_RECORDER)
def _frames(page: Page, terminal: str = "done") -> list[dict]:
"""The captured SSE frames, once the *terminal* frame lands.
The hook reads ``res.clone().text()`` in a background promise that
resolves right after the stream closes — poll briefly until the
terminal (``done``, or ``error`` for the exhaustion test) lands.
"""
deadline = time.monotonic() + 30.0
while True:
raw = page.evaluate("() => window.__sseFrames || []")
parsed = [json.loads(line) for line in raw if line]
if any(f.get("type") == terminal for f in parsed):
return parsed
if time.monotonic() > deadline:
raise AssertionError(
f"SSE hook captured no `{terminal}` frame (frames so far: "
f"{len(parsed)}) — hook install failed?"
)
time.sleep(0.05)
def _retry_frames(frames: list[dict]) -> list[dict]:
return [f for f in frames if f.get("type") == "retry"]
def _assert_retries_before_first_delta(frames: list[dict], retries: list[dict]) -> None:
"""The wire contract (locked A2): every retry frame precedes the
first answer delta — a retry can only restart a request that never
streamed a frame."""
assert retries, "no retry frames on the wire"
first_delta = next(i for i, f in enumerate(frames) if f.get("type") == "delta")
assert all(
i < first_delta for i, f in enumerate(frames) if f.get("type") == "retry"
)
def _assert_no_error_frames(frames: list[dict]) -> None:
assert not [f for f in frames if f.get("type") == "error"]
def _submit(page: Page, question: str) -> None:
page.fill("#message-input", question)
page.click("#send-btn")
# The user bubble lands synchronously with the submit handler.
expect(page.locator(".msg.user .bubble").last).to_contain_text(question)
def _wait_settled(page: Page) -> None:
"""The turn is complete: answer text in the bubble, button recovered.
Phase 48: the label assertion carries the settle wait with an
explicit timeout — the in-flight button is the enabled Stop control
(never disabled), so ``to_be_enabled`` no longer blocks until the
turn settles, and Playwright expect's default (5s) does not inherit
the page default."""
expect(page.locator(".msg.brain .bubble").last).not_to_have_text("", timeout=30_000)
expect(page.locator("#send-btn")).to_be_enabled(timeout=30_000)
expect(page.locator("#send-label")).to_have_text("Send", timeout=30_000)
def _assert_no_error_banner(page: Page) -> None:
"""A retried turn settles through the normal done path — never the
red role=alert error banner (the KB-offline banner is a separate,
health-driven state the db_ready fixture keeps away)."""
banner = page.locator("#kb-banner")
expect(banner).to_be_hidden()
expect(banner).not_to_have_attribute("role", "alert")
expect(banner).not_to_have_class(re.compile(r"is-error"))
# --------------------------------------------------------------------------
# 1. Dead-then-recovered, DEFLECTED path (LOW turn): the status line
# shows the live "retrying" copy and the answer still completes
# --------------------------------------------------------------------------
def test_dead_then_recovered_deflected(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
_reset_db()
page.set_default_timeout(30_000)
page.goto(app_url)
_install_page_hooks(page)
_submit(page, DEFLECT_Q)
_wait_settled(page)
# The live status (owner-locked A4 copy) was recorded for BOTH
# restarts: attempt 2 (after the original attempt died) and attempt
# 3 (after the first retry died) — attempt 4 never needed to start.
statuses = page.evaluate("() => window.__statuses")
assert _retry_status(2) in statuses, statuses
assert _retry_status(3) in statuses, statuses
# Wire: exactly the two retry frames (attempt = the attempt about to
# be tried, 1-based; max_attempts = the forced-default budget of 4),
# both ahead of the first delta, and no error frame anywhere.
frames = _frames(page)
retries = _retry_frames(frames)
assert retries == [
{"type": "retry", "attempt": 2, "max_attempts": MAX_ATTEMPTS},
{"type": "retry", "attempt": 3, "max_attempts": MAX_ATTEMPTS},
], retries
_assert_retries_before_first_delta(frames, retries)
_assert_no_error_frames(frames)
done = next(f for f in frames if f.get("type") == "done")
assert done["deflected"] is True
# The deflected answer settled normally — no error banner.
last = page.locator(".msg.brain").last
expect(last).to_have_class(re.compile(r"is-deflected"))
expect(last.locator(".bubble")).to_contain_text(
re.compile(DEFLECT_PHRASE, re.IGNORECASE)
)
_assert_no_error_banner(page)
row = _last_query_log()
assert row.question == DEFLECT_Q
assert row.deflected is True
# --------------------------------------------------------------------------
# 2. Dead-then-recovered, GROUNDED path (HIGH turn): the agent round
# retried and the answer completes with its sources
# --------------------------------------------------------------------------
def test_dead_then_recovered_grounded(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
_reset_db()
page.set_default_timeout(30_000)
page.goto(app_url)
_install_page_hooks(page)
_submit(page, GROUNDED_Q)
_wait_settled(page)
# Same status sequence as the deflected path — the per-round retry
# (agent loop, task 03) surfaces through the same status line.
statuses = page.evaluate("() => window.__statuses")
assert _retry_status(2) in statuses, statuses
assert _retry_status(3) in statuses, statuses
frames = _frames(page)
retries = _retry_frames(frames)
assert retries == [
{"type": "retry", "attempt": 2, "max_attempts": MAX_ATTEMPTS},
{"type": "retry", "attempt": 3, "max_attempts": MAX_ATTEMPTS},
], retries
_assert_retries_before_first_delta(frames, retries)
_assert_no_error_frames(frames)
done = next(f for f in frames if f.get("type") == "done")
assert done["deflected"] is False
assert [(s["source"], s["path"]) for s in done["sources"]] == [
(SEED_SOURCE, SEED_PATH)
]
# The grounded answer completed with the source chip — the agent
# round retried and the turn is intact.
bubble = page.locator(".msg.brain .bubble").last
expect(bubble).to_contain_text(GROUNDED_Q)
expect(bubble).to_contain_text(MOCK_ANSWER_MARKER)
chips = page.locator(".msg.brain .source-chip")
expect(chips).to_have_count(1)
expect(chips.first).to_contain_text(SEED_SP)
_assert_no_error_banner(page)
row = _last_query_log()
assert row.question == GROUNDED_Q
assert row.deflected is False
assert row.sources == SEED_SP
# --------------------------------------------------------------------------
# 3. Embedding retry: the pre-stream embedding loop is visible to the
# UI (a retry status before any answer) and the turn completes
# --------------------------------------------------------------------------
def test_embedding_retry_completes(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
_reset_db()
page.set_default_timeout(30_000)
page.goto(app_url)
_install_page_hooks(page)
_submit(page, EMBED_Q)
_wait_settled(page)
# The embedding step runs BEFORE the honesty gate and the answer
# stream, so its single retry is the ONLY retry frame of the turn —
# and the UI saw it as the live status (no answer frame had landed).
statuses = page.evaluate("() => window.__statuses")
assert _retry_status(2) in statuses, statuses
frames = _frames(page)
retries = _retry_frames(frames)
assert retries == [
{"type": "retry", "attempt": 2, "max_attempts": MAX_ATTEMPTS},
], retries
_assert_retries_before_first_delta(frames, retries)
_assert_no_error_frames(frames)
done = next(f for f in frames if f.get("type") == "done")
assert done["deflected"] is False
# The turn completed normally with the grounded answer + chip.
bubble = page.locator(".msg.brain .bubble").last
expect(bubble).to_contain_text(MOCK_ANSWER_MARKER)
chips = page.locator(".msg.brain .source-chip")
expect(chips).to_have_count(1)
expect(chips.first).to_contain_text(SEED_SP)
_assert_no_error_banner(page)
# --------------------------------------------------------------------------
# 4. Exhaustion: a dead endpoint burns the whole budget, then the
# EXISTING terminal error banner lands and the composer recovers
# --------------------------------------------------------------------------
def test_exhaustion_lands_on_the_error_banner(
page: Page, app_url: str, mock_llm: int, db_ready: None
) -> None:
_reset_db()
page.set_default_timeout(30_000)
page.goto(app_url)
_install_page_hooks(page)
_submit(page, EXHAUST_Q)
# After the 4th dead attempt the turn dies: the existing terminal
# error banner (role=alert) with the existing copy, and the send
# button re-enabled (the banner path settles the state machine).
expect(page.locator("#kb-banner")).to_have_attribute(
"role", "alert", timeout=60_000
)
expect(page.locator("#kb-banner")).to_contain_text(ERROR_COPY)
expect(page.locator("#send-btn")).to_be_enabled(timeout=30_000)
expect(page.locator("#send-label")).to_have_text("Send", timeout=30_000)
# The live status climbed the whole budget: the LAST retrying status
# is the highest attempt — "(4 of 4)…" (no attempt 5 exists).
statuses = page.evaluate("() => window.__statuses")
retry_statuses = [s for s in statuses if "retrying" in s]
assert retry_statuses, statuses
assert retry_statuses[-1] == _retry_status(MAX_ATTEMPTS), statuses
# Wire: the three retry frames (attempts 2, 3, 4 of 4), then the
# terminal error frame as the LAST event — no done, no delta.
frames = _frames(page, terminal="error")
retries = _retry_frames(frames)
assert retries == [
{"type": "retry", "attempt": a, "max_attempts": MAX_ATTEMPTS}
for a in (2, 3, 4)
], retries
assert frames[-1]["type"] == "error"
assert ERROR_COPY in frames[-1]["detail"]
assert not [f for f in frames if f.get("type") == "done"]
assert not [f for f in frames if f.get("type") == "delta"]
# No answer bubble was ever rendered (no frame ever streamed a
# token) — the user bubble is the only message in the DOM.
expect(page.locator("#messages > .msg.brain")).to_have_count(0)
expect(page.locator(".msg.user .bubble")).to_have_count(1)