"""Phase 45 E2E (Playwright, mock-only): as many tool calls as the model wants. Story: ``.agents/user_stories/agent-unlimited-tools.md`` Run in isolation (DB must be up: ``podman compose up -d db``): uv run pytest tests/e2e/test_agent_unlimited_tools.py -v --no-cov MOCK-ONLY suite: ``E2E_REAL_LLM=1`` is not supported — the gate is the deterministic MULTI-READ marker flow in ``tests/e2e/mock_llm.py`` (user message contains BOTH ``use your tools`` (``TOOLS_TRIGGER``) and ``read two documents`` (``MULTI_READ_TRIGGER``) **and** the system prompt carries the ```` section of the HIGH prompt; phase 70: the flow emits the harness-aligned names — ``ls``, then ``read`` on the JOINED combined ``source/path`` of each file line; phase 94: the drill-down ``ls`` — the top level lists sources only, so the flow drills one level into the first source before the first file line exists): 1. request 1 (``tools`` offered, no tool results yet) → streams ONLY ``tool_calls`` deltas calling ``ls`` (id ``call_0``); 2. request 2 (the top-level source listing in the messages — no file lines yet) → the drill: ``ls`` scoped to the FIRST source of the listing (id ``call_1``); the seed registers ``Deployments`` first, and both read documents live in it — so the drill's folder listing carries BOTH file lines; 3. request 3 (a ``tool``-role folder listing with file lines) → ``read`` on the JOINED combined ``source/path`` of the FIRST file line (id ``call_2``); 4. request 4 (one ``tool``-role read result) → ``read`` on the JOINED combined ``source/path`` of the SECOND file line (id ``call_3``) — the pre-phase-45 per-tool budgets would have refused exactly this second read (``No reading budget left — answer with what you have.``); 5. request 5 (two read results) → the forced answer, byte-stable: the single-read shape quoting the FIRST read result, plus the line ``I read and .`` naming both read paths in read order. KB fixture (the ``test_agent_document_tools.py`` TRUNCATE-then-seed pattern, grown to three documents): * ``Deployments/aaa-record-shape.json`` — read #1: indexed (in the agent's catalog, readable) but seeded WITHOUT chunks, so retrieval never puts it in context; sorts FIRST in the catalog; * ``Deployments/bbb-zone-sync.yaml`` — read #2: same shape; sorts SECOND; * ``Homelab/route53-notes.md`` — the ONLY retrievable document: one chunk whose embedding is the mock's own bag-of-words vector (genuine token overlap: the marker questions cosine ≈0.65/≈0.71 against it, well past the E2E 0.30 threshold, and they FTS-match too) → the grounded seed context. Three documents (not two, as in phase 37) so BOTH reads land on documents outside the seed: with a two-document corpus the second read would be the already-in-context retrieval document and the agent would answer "Already in your context — …" (the in-context refusal) — a rejection, not the multi-read flow this story proves. Test → story mapping (Playwright Mapping Rule): 1. ``test_multi_read_turn`` — the turn streams FOUR ``tool`` frames / ``.tool-call`` lines in order (the top-level list, the drill list — phase 94 — "is listing documents", and two reads — "is reading " — the #send-status transition recorded deterministically via MutationObserver), then a final non-deflected answer containing the mock's byte-stable ``I read and .`` line; the round cap (default 10) bounds the turn, no budget refusal anywhere. 2. ``test_done_sources_include_reads`` — the source chips under the answer list the retrieval doc PLUS both read documents, deduped (the phase-37 ``done.sources`` extension contract, now with 2 reads); the same combined list lands in ``query_log.sources``. 3. ``test_relist_allowed`` — the listing tool ran (its line rendered) and no pre-phase-45 budget refusal ("… budget left") appears anywhere in the message bubble or tool lines: the old ``LIST_EXHAUSTED`` / ``READ_EXHAUSTED`` refusal strings are gone from the product (the source-level grep was task 01's job). 4. ``test_single_tool_flow_regression`` (phase 37) — the original marker WITHOUT the multi-read trigger still answers after exactly ONE read with its single tool pair (list + one read). The full phase-37 suite runs unmodified in the regression pass. """ from __future__ import annotations import hashlib import json import re import time from collections.abc import Callable from datetime import UTC, datetime from playwright.sync_api import Page, expect from sqlalchemy import select, text from sqlalchemy.orm import Session from app.db import SessionLocal from app.models import Chunk, Document, GitSource, QueryLog from e2e.auth_helpers import login from tests.e2e.mock_llm import embed_text # -------------------------------------------------------------------------- # Fixture documents (deterministic, token-controlled) # -------------------------------------------------------------------------- READ1_SOURCE = "Deployments" READ1_PATH = "aaa-record-shape.json" READ1_SP = f"{READ1_SOURCE}/{READ1_PATH}" READ2_SOURCE = "Deployments" READ2_PATH = "bbb-zone-sync.yaml" READ2_SP = f"{READ2_SOURCE}/{READ2_PATH}" SEED_SOURCE = "Homelab" SEED_PATH = "route53-notes.md" SEED_SP = f"{SEED_SOURCE}/{SEED_PATH}" #: The retrievable document: references the record shape "for the exact #: JSON shape of reeselink.json" (the TODO failure, same story as the #: phase-37 fixture). The repeated record-file lines carry the marker #: questions' key tokens (aws, route53, hosted, zone, reeselink, json, #: exact, shape) — verified ≈0.65 (multi question) / ≈0.71 (single #: question) cosine against the mock's embeddings (E2E threshold 0.30) #: plus FTS hits, so both turns are solidly grounded. ROUTE53_CONTENT = ( "# AWS Route 53 Notes\n\n" "## Record file\n\n" + ( "The aws route53 hosted zone for reeselink keeps every record in " "reseelink.json — the exact JSON shape of reeselink.json is " "documented in aaa-record-shape.json.\n" ) * 10 + "\n## Sync job\n\n" "A cron job pushes reeselink.json to the aws route53 hosted zone " "every fifteen minutes; the diff is applied through the route53 api.\n" ) #: Read #1: the JSON shape. Its FIRST line is longer than 80 chars, so #: the mock's first-80-chars quote content part is newline-free (the #: rendered-text assertion matches it after the phase-106 D5 date #: line). Pinned by the assert below. RECORD_CONTENT = ( '{"version": 4, "comment": "ReeseLink hosted zone records — the exact ' 'JSON shape of reeselink.json",\n' ' "hosted_zone_id": "Z0RESEELINK45",\n' ' "record_sets": [\n' ' { "name": "www.reeselink.example", "type": "A", "ttl": 300 }\n' ' ]\n' "}\n" ) assert "\n" not in RECORD_CONTENT[:63] # the quote's content part stays one line #: Read #2: the sync runbook. RUNBOOK_CONTENT = ( "sync:\n" " schedule: every fifteen minutes\n" " target: reeselink.json\n" " engine: aws route53 api\n" " note: the diff is applied through the route53 api\n" ) #: Carries BOTH markers — ``use your tools`` (phase 37) and ``read two #: documents`` (phase 45 ``MULTI_READ_TRIGGER``). MULTI_QUESTION = ( "Use your tools and read two documents: what is the exact JSON shape " "of reeselink.json for my aws route53 hosted zone?" ) #: The phase-37 marker WITHOUT the multi-read trigger — the original #: 3-step single-read flow (regression test 4). SINGLE_QUESTION = ( "Use your tools: what is the exact JSON shape of reeselink.json " "for my aws route53 hosted zone?" ) assert "use your tools" in MULTI_QUESTION.lower() assert "read two documents" in MULTI_QUESTION.lower() assert "read two documents" not in SINGLE_QUESTION.lower() #: The mock's byte-stable multi-read answer pieces (mock_llm #: ``_tool_flow``): the single-read shape quoting the FIRST read result, #: plus both read paths in read order. Phase 106 (D5): the read #: result's ``date:`` SECOND line rides into the first-80-chars quote — #: the date line (the fixture's fixed ``created_at`` UTC date part, #: 17 chars; its trailing newline renders as a markdown soft break — #: no text between the date and the content) + the first 63 content #: chars (80 − 17). ANSWER_PREFIX = f"Read {READ1_SP}." ANSWER_QUOTE = "date: 2024-06-15" + RECORD_CONTENT[:63] BOTH_READS_LINE = f"I read {READ1_SP} and {READ2_SP}." #: The pre-phase-45 budget refusals (phase 37 ``LIST_EXHAUSTED`` / #: ``READ_EXHAUSTED``) — gone from the app (task 01) and never rendered #: (test 3). The generic "budget left" fragment covers both exact #: strings. BUDGET_REFUSAL_FRAGMENTS = ( "No listing budget left — answer with what you have.", "No reading budget left — answer with what you have.", "budget left", ) # The combined source list the app reports (app/api/chat.py): retrieval # docs first, then the agent's read docs, deduped by (source, path). EXPECTED_SOURCES = [ (SEED_SOURCE, SEED_PATH), (READ1_SOURCE, READ1_PATH), (READ2_SOURCE, READ2_PATH), ] EXPECTED_SOURCES_LINE = ", ".join(f"{s}/{p}" for s, p in EXPECTED_SOURCES) # -------------------------------------------------------------------------- # DB seeding (TRUNCATE-then-seed, cf. test_agent_document_tools.py) # -------------------------------------------------------------------------- def _doc(source: str, path: str, title: str, content: str) -> Document: return Document( source=source, path=path, full_path=f"/tmp/{path}", title=title, content=content, content_hash=hashlib.sha256(content.encode()).hexdigest(), indexed_at=datetime.now(UTC), # Phase 106 (D5): explicit dates — byte-stable prompts/quotes # (the mock's first-80-chars read quote carries the date line). created_at=datetime(2024, 6, 15, tzinfo=UTC), ) def _seed(db: Session) -> None: """The three-document KB from the module docstring. Phase 94: the drill-down ``ls`` top level reads the registry — register BOTH sources (TRUNCATEd in ``_reset_db``), ``Deployments`` FIRST (registry order is ``(added_at, id)``): the mock's drill (first source of the listing) lands on the folder that carries BOTH file lines. A non-empty table also ignores the operator's ``BOR_GIT_SOURCES`` fallback — deterministic. """ # COMMIT between the inserts (not flush): ``added_at`` is # ``server_default now()`` — the transaction timestamp — and the # tie-break is the RANDOM uuid ``id``, so two rows in one # transaction order nondeterministically (the integration # ``registry`` fixture's pattern). db.add(GitSource(url=READ1_SOURCE, kind="local")) db.commit() db.add(GitSource(url=SEED_SOURCE, kind="local")) md = _doc(SEED_SOURCE, SEED_PATH, "AWS Route 53 Notes", ROUTE53_CONTENT) db.add(md) db.flush() # One chunk carrying the mock's own embedding → genuine token # overlap between the marker questions and this document (the only # retrievable document — the grounded seed context). db.add( Chunk( document_id=md.id, position=0, content=ROUTE53_CONTENT, embedding=embed_text(ROUTE53_CONTENT), ) ) # The two read documents: indexed, catalogued, readable — but NO # chunks, so retrieval never puts them in context. db.add(_doc(READ1_SOURCE, READ1_PATH, "Record Shape", RECORD_CONTENT)) db.add(_doc(READ2_SOURCE, READ2_PATH, "Zone Sync Runbook", RUNBOOK_CONTENT)) def _reset_db(seed: Callable[[Session], None] | None = None) -> None: """Truncate the KB (plus the prompt-shaping tables), then re-seed. ``steering_notes`` / ``kb_overview`` are truncated too, so the HIGH prompt is exactly ```` + ```` + ```` regardless of leftovers from other suites — byte-stable prompts, byte-stable answers. """ with SessionLocal() as db: db.execute( text( "TRUNCATE chunks, documents, query_log, steering_notes, " "kb_overview, git_sources" ) ) db.commit() if seed is not None: seed(db) db.commit() def _last_query_log() -> QueryLog: with SessionLocal() as db: rows = db.scalars(select(QueryLog)).all() assert len(rows) == 1, f"expected exactly one query_log row, got {len(rows)}" return rows[0] # -------------------------------------------------------------------------- # Page helpers # -------------------------------------------------------------------------- #: Records every value #send-label takes during the turn (a #: MutationObserver on the element), so the transient "Calling tool…" #: state is captured deterministically — no polling race (the phase-37 #: flake fix, phase 44 task 03). LABEL_RECORDER = """ () => { if (window.__labelsInstalled) return; window.__labelsInstalled = true; window.__labels = []; const el = document.querySelector('#send-label'); if (!el) return; const rec = (v) => { const l = window.__labels; if (!l.length || l[l.length - 1] !== v) l.push(v); }; rec(el.textContent); new MutationObserver(() => rec(el.textContent)).observe(el, { childList: true, subtree: true, }); } """ #: Records every value #send-status takes during the turn — the #: "… is listing documents" / "… is reading " tool states #: are transient (the first delta switches the status to the streaming #: state), so the pre-submit observer is the deterministic source of #: truth for their order. STATUS_RECORDER = """ () => { if (window.__statusesInstalled) return; window.__statusesInstalled = true; window.__statuses = []; const el = document.querySelector('#send-status'); if (!el) return; const rec = (v) => { const l = window.__statuses; if (!l.length || l[l.length - 1] !== v) l.push(v); }; rec(el.textContent); new MutationObserver(() => rec(el.textContent)).observe(el, { childList: true, subtree: true, }); } """ #: Captures the raw SSE ``data:`` payloads of the /api/chat stream #: (a response clone read in the background) — wire-level assertions for #: the ``tool`` frames, independent of the UI rendering. SSE_HOOK = """ () => { if (window.__sseInstalled) return; window.__sseInstalled = true; window.__sseFrames = []; const origFetch = window.fetch; window.fetch = async function (...args) { const res = await origFetch.apply(this, args); try { const url = typeof args[0] === 'string' ? args[0] : args[0].url; if (url.includes('/api/chat')) { res.clone().text().then((bodyText) => { for (const block of bodyText.split('\\n\\n')) { const line = block.trim(); if (line.startsWith('data: ')) { window.__sseFrames.push(line.slice(6)); } } }); } } catch (e) { /* non-clonable responses: ignored */ } return res; }; } """ def _install_page_hooks(page: Page) -> None: """Install all hooks on the loaded page (post-goto, pre-submit).""" page.evaluate(SSE_HOOK) page.evaluate(LABEL_RECORDER) page.evaluate(STATUS_RECORDER) def _frames(page: Page) -> list[dict]: """The captured SSE frames, once the hook's background read settles.""" deadline = time.monotonic() + 10.0 while True: raw = page.evaluate("() => window.__sseFrames || []") parsed = [json.loads(line) for line in raw if line] if any(f.get("type") == "done" for f in parsed): return parsed if time.monotonic() > deadline: raise AssertionError( f"SSE hook captured no `done` frame (frames so far: " f"{len(parsed)}) — hook install failed?" ) time.sleep(0.05) def _tool_frames(frames: list[dict]) -> list[dict]: return [f for f in frames if f.get("type") == "tool"] def _submit(page: Page, question: str) -> None: page.fill("#message-input", question) page.click("#send-btn") # The user bubble lands synchronously with the submit handler. expect(page.locator(".msg.user .bubble").last).to_contain_text(question) def _wait_settled(page: Page) -> None: """The turn is complete: answer text in the bubble, button recovered. Phase 48: the label assertion carries the settle wait with an explicit timeout — the in-flight button is the enabled Stop control (never disabled), so ``to_be_enabled`` no longer blocks until the turn settles, and Playwright expect's default (5s) does not inherit the page default.""" expect(page.locator(".msg.brain .bubble").last).not_to_have_text("", timeout=30_000) expect(page.locator("#send-btn")).to_be_enabled(timeout=30_000) expect(page.locator("#send-label")).to_have_text("Send", timeout=30_000) # -------------------------------------------------------------------------- # 1. The multi-read turn: list → read #1 → read #2 → both-named answer # -------------------------------------------------------------------------- def test_multi_read_turn( page: Page, app_url: str, mock_llm: int, db_ready: None ) -> None: page.set_default_timeout(30_000) _reset_db(_seed) login(page, app_url, next="/") _install_page_hooks(page) _submit(page, MULTI_QUESTION) _wait_settled(page) # Wire level: exactly FOUR `tool` frames — the top-level ls, the # drill ls scoped to the first source (phase 94), then read #1 and # read #2 (each read's argument is the JOINED combined # source/path), in order — and all ahead of the first `delta` # frame. The fourth frame is the one the pre-phase-45 read budget # refused. frames = _frames(page) assert _tool_frames(frames) == [ {"type": "tool", "name": "ls", "argument": None}, {"type": "tool", "name": "ls", "argument": READ1_SOURCE}, {"type": "tool", "name": "read", "argument": READ1_SP}, {"type": "tool", "name": "read", "argument": READ2_SP}, ] first_delta = next(i for i, f in enumerate(frames) if f.get("type") == "delta") assert all( i < first_delta for i, f in enumerate(frames) if f.get("type") == "tool" ) done = next(f for f in frames if f.get("type") == "done") assert done["deflected"] is False # The transient "calling tool" states, recorded deterministically. # Phase 48 (owner-locked): the in-flight button is the Stop control — # the label holds "Stop" for the whole turn (it no longer relabels # to "Calling tool…"), and #send-status walked through # "… is listing documents" then "… is reading " for BOTH reads, # in order. labels = page.evaluate("() => window.__labels") assert "Stop" in labels, labels statuses = page.evaluate("() => window.__statuses") i_list = next( (i for i, s in enumerate(statuses) if "is listing documents" in s), None ) i_read1 = next( (i for i, s in enumerate(statuses) if f"is reading {READ1_SP}" in s), None ) i_read2 = next( (i for i, s in enumerate(statuses) if f"is reading {READ2_SP}" in s), None ) assert ( i_list is not None and i_read1 is not None and i_read2 is not None ), statuses assert i_list < i_read1 < i_read2, statuses # Four visible tool lines, in order, above the answer (phase 94: # the drill line is "Listing documents in "). lines = page.locator(".msg.brain .tool-call") expect(lines).to_have_count(4) expect(lines.nth(0)).to_contain_text("Listing documents") expect(lines.nth(1)).to_contain_text("Listing documents in") expect(lines.nth(1)).to_contain_text(READ1_SOURCE) expect(lines.nth(2)).to_contain_text("Reading ") expect(lines.nth(2)).to_contain_text(READ1_SP) expect(lines.nth(3)).to_contain_text("Reading ") expect(lines.nth(3)).to_contain_text(READ2_SP) # The final answer is non-deflected, quotes the FIRST read result, # and names BOTH read paths (the mock's byte-stable line). last = page.locator(".msg.brain").last expect(last).not_to_have_class(re.compile(r"is-deflected")) bubble = last.locator(".bubble") expect(bubble).to_contain_text(ANSWER_PREFIX) expect(bubble).to_contain_text(ANSWER_QUOTE) expect(bubble).to_contain_text(BOTH_READS_LINE) # Durable record: grounded, combined sources (retrieval + both # reads). row = _last_query_log() assert row.question == MULTI_QUESTION assert row.deflected is False assert row.sources == EXPECTED_SOURCES_LINE # -------------------------------------------------------------------------- # 2. done.sources / source chips: retrieval doc + BOTH reads, deduped # -------------------------------------------------------------------------- def test_done_sources_include_reads( page: Page, app_url: str, mock_llm: int, db_ready: None ) -> None: page.set_default_timeout(30_000) _reset_db(_seed) login(page, app_url, next="/") _install_page_hooks(page) _submit(page, MULTI_QUESTION) _wait_settled(page) # Wire level: done.sources is the retrieval doc FIRST, then both # read documents — deduped (the retrieval doc was never read, the # reads are each read once; nothing appears twice). frames = _frames(page) done = next(f for f in frames if f.get("type") == "done") assert [(s["source"], s["path"]) for s in done["sources"]] == EXPECTED_SOURCES pairs = [(s["source"], s["path"]) for s in done["sources"]] assert len(pairs) == len(set(pairs)), "done.sources must be deduped" # UI: exactly three source chips under the answer, in the same # order, each a viewer link — no duplicated chip. chips = page.locator(".msg.brain .source-chip") expect(chips).to_have_count(3) expect(chips.nth(0)).to_contain_text(SEED_SP) expect(chips.nth(1)).to_contain_text(READ1_SP) expect(chips.nth(2)).to_contain_text(READ2_SP) for i, (source, path) in enumerate(EXPECTED_SOURCES): expect(chips.nth(i)).to_have_attribute( "href", f"/document.html?source={source}&path={path}&back=%2F" ) # -------------------------------------------------------------------------- # 3. No budget refusal: the listing ran, and the pre-phase-45 refusal # strings are nowhere in the rendered message # -------------------------------------------------------------------------- def test_relist_allowed( page: Page, app_url: str, mock_llm: int, db_ready: None ) -> None: page.set_default_timeout(30_000) _reset_db(_seed) login(page, app_url, next="/") _install_page_hooks(page) _submit(page, MULTI_QUESTION) _wait_settled(page) # The listing tool actually ran (its line rendered, its wire frame # present) — and the turn completed past the point where the old # per-tool budgets would have refused (list budget 1, read budget # 1 — this turn makes one list and TWO reads). frames = _frames(page) assert {"type": "tool", "name": "ls", "argument": None} in _tool_frames( frames ) line0 = page.locator(".msg.brain .tool-call").nth(0) expect(line0).to_contain_text("Listing documents") # No pre-phase-45 budget refusal anywhere in the message — neither # the exact old strings nor the generic fragment — not in the # bubble, not in any tool line. msg_text = page.locator(".msg.brain").last.text_content() or "" for fragment in BUDGET_REFUSAL_FRAGMENTS: assert fragment not in msg_text, ( f"budget refusal {fragment!r} rendered: {msg_text!r}" ) # And it answered (a refusal would have left the model stuck — the # turn settled with a non-deflected, both-named answer). bubble = page.locator(".msg.brain .bubble").last expect(bubble).to_contain_text(BOTH_READS_LINE) # -------------------------------------------------------------------------- # 4. Phase-37 regression: the single-read marker flow still answers # after exactly ONE read with its single tool pair # -------------------------------------------------------------------------- def test_single_tool_flow_regression( page: Page, app_url: str, mock_llm: int, db_ready: None ) -> None: page.set_default_timeout(30_000) _reset_db(_seed) login(page, app_url, next="/") _install_page_hooks(page) _submit(page, SINGLE_QUESTION) _wait_settled(page) # Exactly THREE tool frames — the top-level ls, the drill ls # (phase 94), then ONE read of the first file line (the JOINED # combined source/path) — no second read (the marker carries no # multi-read trigger). frames = _frames(page) assert _tool_frames(frames) == [ {"type": "tool", "name": "ls", "argument": None}, {"type": "tool", "name": "ls", "argument": READ1_SOURCE}, {"type": "tool", "name": "read", "argument": READ1_SP}, ] lines = page.locator(".msg.brain .tool-call") expect(lines).to_have_count(3) expect(lines.nth(0)).to_contain_text("Listing documents") expect(lines.nth(1)).to_contain_text("Listing documents in") expect(lines.nth(1)).to_contain_text(READ1_SOURCE) expect(lines.nth(2)).to_contain_text("Reading ") expect(lines.nth(2)).to_contain_text(READ1_SP) # The single-read answer shape: quotes the read document; it does # NOT carry the multi-read both-named line (READ2 was never read). bubble = page.locator(".msg.brain .bubble").last expect(bubble).to_contain_text(ANSWER_PREFIX) expect(bubble).to_contain_text(ANSWER_QUOTE) expect(bubble).not_to_contain_text(BOTH_READS_LINE) expect(bubble).not_to_contain_text(READ2_SP) # done: non-deflected; sources = retrieval doc + the single read # (READ2 absent — it was never read). done = next(f for f in frames if f.get("type") == "done") assert done["deflected"] is False assert [(s["source"], s["path"]) for s in done["sources"]] == [ (SEED_SOURCE, SEED_PATH), (READ1_SOURCE, READ1_PATH), ] row = _last_query_log() assert row.question == SINGLE_QUESTION assert row.deflected is False assert row.sources == f"{SEED_SP}, {READ1_SP}"