phase: 122_image_documents
Build and Push Containers / build-and-push-app (push) Successful in 1m57s
Build and Push Containers / build-and-push-db (push) Failing after 13s

**Phase 122 (image documents) — final verification pass: all green. No code changes were needed; defects found: none.**

**Verified (implementation already complete in working tree, reviewed end-to-end):**
- Toggle (`BOR_IMAGES`/`BOR_IMAGE_EXTENSIONS`/`BOR_IMAGE_DIR`, off by default) + `GET /api/config` `images` flag
- Ingest: bytes digest, `image_dir` persistent copy, `content = summary = vision description` (chat-model call; only text embedded), fail-soft skip + `images_failed` counter
- Serve/display: `/api/documents/{id}/image` route (404 matrix), viewer `<img>` + description, Sources 48px lazy thumbnails, chat inline source figure (alt = summary), agent `read` marker
- Prune guard: images-off syncs never prune `is_image` docs

**Test / lint / coverage (exact commands & outcomes):**
- `uv run pytest` → exit 0 (green; note: pytest 9.1.1 `-q` omits the final count line in output — exit code authoritative)
- `uv run pytest --cov=app --cov-report=term-missing` → **2715 passed, exit 0, TOTAL 99%** (>90% gate)
- `uv run ruff check . && uv run pyright` → "All checks passed!" / "0 errors, 0 warnings, 0 informations"
- `uv run pytest tests/e2e/test_image_documents.py -v --no-cov` → **4 passed, exit 0** (isolation)

**Completion criteria:** (1) images=true → described/embedded/displayed docs: ✅ (E2E + integration) · (2) images=false byte-identical + image docs survive sync: ✅ (E2E negative app + unit/integration) · (3) viewer + chat rendering with alt text; failed description skips + logs, sync completes: ✅ · (4) test/lint/coverage gates: ✅ · (5) commit + phase move: deferred to harness per this pass's rules (working tree left uncommitted).

**Notable deviation (pre-existing, documented in code):** image route uses `require_user` (phase-79 posture, same gate as the document content endpoint) rather than the phase text's "public" parenthetical — matches the endpoint it mirrors.

**Next pending phase:** `123_chat_image_questions`.
This commit is contained in:
2026-09-25 01:54:23 -04:00
parent 0f77e9a876
commit a19d78d284
63 changed files with 5484 additions and 111 deletions
+611
View File
@@ -7,22 +7,33 @@ Uses the real compose Postgres (``db`` fixture) and FastAPI's TestClient.
"""
from __future__ import annotations
import asyncio
import base64
import hashlib
import inspect
import itertools
import logging
import uuid
from datetime import UTC, datetime, timedelta
from pathlib import Path
from typing import Any
import pytest
from fastapi.testclient import TestClient
from sqlalchemy import delete, func, select, text
import app.api.docs as docs_api
import app.rag.importer as rag_importer
from app.config import Settings
from app.core import tokens as token_service
from app.main import app as fastapi_app
from app.models import Chunk, Document, FolderSummary, GitSource
from app.rag import git_sources as rag_git_sources
from app.rag.folder_summaries import missing_folder_summaries
from app.rag.importer import import_sources
from app.rag.llm import LLMError
from app.rag.summarizer import DESCRIBE_PROMPT
from tests.fakes import FakeEmbedder
_TREE_TABLES = "chunks, documents, folder_summaries, git_sources"
@@ -823,3 +834,603 @@ def test_docs_tree_stat_walk_equivalence_with_flat_list(admin_client, db) -> Non
)
_truncate_tree_tables(db)
# ---------------------------------------------------------------------------
# Phase 122, task 02 — image ingest end-to-end (the full ``import_sources``
# pipeline against the real DB; task 06 finalizes the phase-122 suite here
# with the image route + content-endpoint shapes).
# ---------------------------------------------------------------------------
#: A real 1×1 transparent PNG — the importer is content-agnostic (it
#: never parses the image), but a well-formed fixture keeps the tests
#: honest about what a real upload looks like.
PNG_1X1 = base64.b64decode(
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAAC0lEQVR4nGP4DwQACfsD/fteaysAAAAASUVORK5CYII="
)
def _image_llm[
ImageLLM: FakeEmbedder
](tmp_path, llm_cls: type[ImageLLM] = FakeEmbedder, **kwargs) -> ImageLLM:
"""A fake LLM with the phase-122 image knobs (toggle ON by default;
the image dir defaults under *tmp_path* unless overridden).
*llm_cls* (task 03) may be the mock vision client (``_MockVisionLLM``
below) for the no-seam-patch end-to-end path — the PEP 695 type
parameter keeps the helper's return type honest (``chat_models``).
"""
kwargs.setdefault("_env_file", None)
kwargs.setdefault("images", True)
kwargs.setdefault("image_dir", str(tmp_path / "images"))
llm = llm_cls()
llm.settings = Settings(**kwargs) # pyright: ignore[reportCallIssue]
return llm
def _patch_description(monkeypatch, description) -> None:
"""Pin the task-02 seam (``rag_importer._describe_or_skip``) to
*description*. Task 03 fills the seam with the CHAT model's vision
call — these end-to-end mechanics do not change with it."""
async def _fake(llm, *, data, source, rel, full_path):
return description
monkeypatch.setattr(rag_importer, "_describe_or_skip", _fake)
def _cleanup_source(db, source: str) -> None:
for doc in db.scalars(select(Document).where(Document.source == source)).all():
db.delete(doc)
db.commit()
def test_import_sources_images_on_indexes_image_docs(
db, tmp_path, monkeypatch: pytest.MonkeyPatch
) -> None:
"""``images=True`` end-to-end: a standalone image in a source becomes
a Document — bytes digested, the persistent copy in ``image_dir``
(``<doc-id>.png``), ``content`` = the (mock) description, and ONLY
that text is embedded (the content chunks + the phase-30 summary
chunk, 768 dims) — while a text doc in the same source stays a
plain text doc."""
source_root = tmp_path / "imgsource"
source_root.mkdir()
(source_root / "diagram.png").write_bytes(PNG_1X1)
(source_root / "notes.md").write_text("# Notes\n\nBody text.\n", encoding="utf-8")
llm = _image_llm(tmp_path)
_patch_description(monkeypatch, "A network diagram of the homelab VLANs.")
try:
summary = asyncio.run(
import_sources([source_root], llm, session=db, prune=True)
)
assert (summary.files, summary.added, summary.images_failed) == (2, 2, 0)
assert summary.formats == {"md": 1, "png": 1}
doc = db.scalar(
select(Document).where(
Document.source == source_root.name, Document.path == "diagram.png"
)
)
assert doc is not None, "the image file must become a document"
assert doc.is_image is True and doc.image_path is not None
assert doc.content == "A network diagram of the homelab VLANs."
assert doc.title == "diagram" # the non-markdown stem rule
copy = Path(doc.image_path)
assert copy.parent == Path(llm.settings.image_dir)
assert copy.name == f"{doc.id}.png"
assert copy.read_bytes() == PNG_1X1
# The ONLY embedded text is the description (the embedding model
# never sees pixels): every content chunk carries it, the
# phase-30 summary chunk exists + is embedded, 768 dims. Task
# 03: the description IS the summary (stored verbatim — no
# ``lite`` call, no pointer line), so the summary chunk mirrors
# ``doc.content`` exactly.
chunks = db.scalars(select(Chunk).where(Chunk.document_id == doc.id)).all()
content_chunks = [c for c in chunks if not c.is_summary]
assert [c.content for c in content_chunks] == [doc.content]
assert doc.summary == doc.content
summary_chunks = [c for c in chunks if c.is_summary]
assert len(summary_chunks) == 1 and summary_chunks[0].position == -1
assert summary_chunks[0].content == doc.content
for c in chunks:
assert c.embedding is not None and len(c.embedding) == 768
# The text doc is untouched by the image machinery.
md = db.scalar(
select(Document).where(
Document.source == source_root.name, Document.path == "notes.md"
)
)
assert md is not None and md.is_image is False and md.image_path is None
finally:
_cleanup_source(db, source_root.name)
class _MockVisionLLM(FakeEmbedder):
"""The phase-122 mock VISION client (task 03): the CHAT model
answers the multimodal describe call (the image bytes' data URL)
with a fixed, retrieval-oriented description; text (``lite``) calls
keep the ``FakeEmbedder`` behaviour. The REAL
``rag_importer._describe_or_skip`` → ``summarizer.describe_image``
chain runs end-to-end against it (no seam patch); every chat
call's model is recorded (``chat_models``)."""
DESCRIPTION = (
"A network diagram of the homelab VLANs: the core switch, the "
"router, and three labeled subnets."
)
def __init__(self) -> None:
super().__init__()
self.chat_models: list[str | None] = []
async def chat(self, messages, model=None):
self.chat_calls.append(list(messages))
self.chat_models.append(model)
user = next((m["content"] for m in messages if m.get("role") == "user"), "")
if isinstance(user, list):
# The phase-122 describe call — the multimodal message.
return self.DESCRIPTION
first = user.split()
return "Summary of " + (first[0] if first else "<empty>")
def test_import_sources_mock_vision_client_end_to_end(db, tmp_path) -> None:
"""Task 03 end-to-end (NO seam patch): a fixture PNG through the
mock vision client — the real ``_describe_or_skip`` →
``describe_image`` → CHAT-model call — yields a doc whose
``content`` == ``summary`` == the description, with its
``is_summary`` chunk embedded, and whose ONLY embedded text is that
description (the embedding model never sees pixels)."""
source_root = tmp_path / "visione2e"
source_root.mkdir()
(source_root / "diagram.png").write_bytes(PNG_1X1)
llm = _image_llm(tmp_path, _MockVisionLLM)
try:
summary = asyncio.run(
import_sources([source_root], llm, session=db)
)
assert (summary.files, summary.added, summary.images_failed) == (1, 1, 0)
# The describe call went to the CHAT model (LOCKED A3), once.
assert llm.chat_models == [llm.settings.llm_chat_model]
# The wire shape: the multimodal user message — the fixed
# prompt's text part + the image's data-URL part.
(message,) = llm.chat_calls[0]
assert message["role"] == "user"
content: Any = message["content"] # the multimodal part list
assert content[0] == {"type": "text", "text": DESCRIBE_PROMPT}
assert content[1]["type"] == "image_url"
assert content[1]["image_url"]["url"].startswith("data:image/png;base64,")
doc = db.scalar(
select(Document).where(
Document.source == source_root.name, Document.path == "diagram.png"
)
)
assert doc is not None, "the image file must become a document"
assert doc.is_image is True and doc.image_path is not None
assert doc.content == _MockVisionLLM.DESCRIPTION
assert doc.summary == _MockVisionLLM.DESCRIPTION # task 03: verbatim
# The ONLY embedded text of the doc is the description — twice
# (the one content chunk + the is_summary chunk), 768 dims.
chunks = db.scalars(select(Chunk).where(Chunk.document_id == doc.id)).all()
summary_chunks = [c for c in chunks if c.is_summary]
assert len(summary_chunks) == 1 and summary_chunks[0].position == -1
assert all(c.content == _MockVisionLLM.DESCRIPTION for c in chunks)
for c in chunks:
assert c.embedding is not None and len(c.embedding) == 768
embedded = [t for batch in llm.calls for t in batch]
assert embedded == [_MockVisionLLM.DESCRIPTION] * 2
finally:
_cleanup_source(db, source_root.name)
class _MockVisionFailsLLM(FakeEmbedder):
"""A NON-VISION chat model (LOCKED A3's honest failure): the
multimodal describe call raises (the SDK errors — a chat model
without vision rejects the ``image_url`` part), text (``lite``)
calls keep the ``FakeEmbedder`` behaviour."""
async def chat(self, messages, model=None):
self.chat_calls.append(list(messages))
user = next((m["content"] for m in messages if m.get("role") == "user"), "")
if isinstance(user, list):
raise LLMError("simulated non-vision chat model (test sentinel)")
first = user.split()
return "Summary of " + (first[0] if first else "<empty>")
def test_import_sources_failing_vision_skips_image_keeps_sync_green(
db, tmp_path, caplog: pytest.LogCaptureFixture
) -> None:
"""LOCKED A3 fail-soft end-to-end (the task-06 integration pin):
a fixture PNG through a NON-VISION chat model — the real seam, no
patch — skips the image doc (``images_failed == 1``, NO row, NO
orphan copy — not even the image dir) while the TEXT doc in the
same source is indexed as usual (the sync completes, no row
mutation anywhere for the failed image)."""
source_root = tmp_path / "visionfail"
source_root.mkdir()
(source_root / "diagram.png").write_bytes(PNG_1X1)
(source_root / "notes.md").write_text("# Notes\n\nBody.\n", encoding="utf-8")
llm = _image_llm(tmp_path, _MockVisionFailsLLM)
try:
with caplog.at_level(logging.INFO, logger="app.importer"):
summary = asyncio.run(
import_sources([source_root], llm, session=db)
)
assert (summary.files, summary.added, summary.images_failed) == (2, 1, 1)
# The image: no row, no copy (the dir itself was never created).
assert (
db.scalar(
select(Document).where(
Document.source == source_root.name, Document.path == "diagram.png"
)
)
is None
)
assert not Path(llm.settings.image_dir).expanduser().exists()
# The text doc indexed as usual (the sync stayed green).
md = db.scalar(
select(Document).where(
Document.source == source_root.name, Document.path == "notes.md"
)
)
assert md is not None and md.is_image is False
# The importer's warning names the document (the PLAN §9 signal).
warnings = [
r
for r in caplog.records
if r.name == "app.importer" and "image description failed" in r.getMessage()
]
assert len(warnings) == 1 and warnings[0].levelno == logging.WARNING
assert f"source={source_root.name} path=diagram.png" in warnings[0].getMessage()
finally:
_cleanup_source(db, source_root.name)
def test_import_sources_images_off_ignores_and_prune_guard_protects(
db, tmp_path, monkeypatch: pytest.MonkeyPatch
) -> None:
"""``images=False`` (the default) end-to-end: the walk ignores the
image file entirely (not counted, no row, no copy), and a
``prune=True`` run MUST NOT delete a pre-existing image doc — the
LOCKED prune guard (invisible to the walk ≠ deleted)."""
source_root = tmp_path / "imgsrc_off"
source_root.mkdir()
(source_root / "diagram.png").write_bytes(PNG_1X1)
_patch_description(monkeypatch, "A network diagram.")
try:
llm_on = _image_llm(tmp_path)
s_on = asyncio.run(import_sources([source_root], llm_on, session=db))
assert s_on.added == 1
doc = db.scalar(
select(Document).where(
Document.source == source_root.name, Document.path == "diagram.png"
)
)
assert doc is not None
copy = Path(doc.image_path)
assert copy.exists()
# Toggle OFF (a fresh fake on the code defaults): the walk is
# blind to the file, and prune protects the pre-existing image
# doc + its copy.
llm_off = FakeEmbedder() # Settings(_env_file=None) → images False
s_off = asyncio.run(
import_sources([source_root], llm_off, session=db, prune=True)
)
assert (s_off.files, s_off.added, s_off.pruned) == (0, 0, 0)
assert db.scalar(select(Document).where(Document.id == doc.id)) is not None
assert copy.exists(), "the copy survives with the doc"
finally:
_cleanup_source(db, source_root.name)
def test_import_sources_toggle_on_prunes_deleted_image_with_copy(
db, tmp_path, monkeypatch: pytest.MonkeyPatch
) -> None:
"""Toggle ON, the image file deleted: the normal prune runs — the
doc row AND its ``image_dir`` copy are removed (the copy's
lifecycle is tied to the row)."""
source_root = tmp_path / "imgsrc_prune"
source_root.mkdir()
(source_root / "diagram.png").write_bytes(PNG_1X1)
llm_on = _image_llm(tmp_path)
_patch_description(monkeypatch, "A network diagram.")
try:
asyncio.run(import_sources([source_root], llm_on, session=db))
doc = db.scalar(
select(Document).where(
Document.source == source_root.name, Document.path == "diagram.png"
)
)
assert doc is not None
copy = Path(doc.image_path)
assert copy.exists()
(source_root / "diagram.png").unlink()
s = asyncio.run(
import_sources([source_root], llm_on, session=db, prune=True)
)
assert (s.pruned, s.added, s.unchanged) == (1, 0, 0)
assert (
db.scalar(select(Document).where(Document.source == source_root.name))
is None
)
assert not copy.exists(), "the image_dir copy is deleted with the doc"
finally:
_cleanup_source(db, source_root.name)
# ---------------------------------------------------------------------------
# Phase 122 (task 04) — the image BYTES route + the content/tree wire
# affordance (the serve side of the image-document contract).
# ---------------------------------------------------------------------------
def _seed_image_doc(
db,
tmp_path: Path,
*,
source: str = "ImgSrc",
path: str = "pic.png",
data: bytes = PNG_1X1,
content: str = "A red square on a white background.",
image_path: str | None = "auto",
doc_id: str | None = None,
) -> Document:
"""One ``is_image`` document row with its persistent copy (the
importer's ``image_dir`` layout) under *tmp_path*; ``image_path``
``"auto"`` writes the copy, ``None`` leaves the row without a copy
(the lost-copy corner)."""
doc = Document(
id=uuid.UUID(doc_id) if doc_id else uuid.uuid4(),
source=source,
path=path,
full_path=f"/tmp/{path}",
title=path.rsplit(".", 1)[0],
content=content,
content_hash=hashlib.sha256(data).hexdigest(),
indexed_at=datetime.now(UTC),
created_at=datetime(2026, 1, 1, tzinfo=UTC),
summary=content,
is_image=True,
)
if image_path == "auto":
copy = tmp_path / f"{doc.id}{Path(path).suffix}"
copy.write_bytes(data)
doc.image_path = str(copy)
elif image_path is not None:
doc.image_path = image_path
db.add(doc)
db.flush()
db.add_all(
[
Chunk(
document_id=doc.id, position=0, content=content, embedding=[0.01] * 768
),
Chunk(
document_id=doc.id,
position=-1,
content=content,
embedding=[0.01] * 768,
is_summary=True,
),
]
)
db.commit()
return doc
def _cleanup_kb(db) -> None:
"""Truncate the KB tables. The settle commit matters: SQLAlchemy
does NOT autoflush pending ORM objects before a raw ``text()``
statement — a pending chunks INSERT flushed *after* the TRUNCATE
would FK-violate (the document row is already gone), so any
pending state is committed (then truncated) first."""
db.commit()
db.execute(text("TRUNCATE chunks, documents"))
db.commit()
@pytest.mark.parametrize(
("ext", "mime"),
[
(".png", "image/png"),
(".jpg", "image/jpeg"),
(".jpeg", "image/jpeg"),
(".webp", "image/webp"),
(".gif", "image/gif"),
(".bmp", "image/bmp"),
],
)
def test_image_route_serves_exact_bytes_with_content_type(
admin_client: TestClient, db, tmp_path: Path, ext: str, mime: str
) -> None:
"""The serve contract: the route streams the EXACT stored bytes
(a per-extension sentinel — a mix-up between the six formats is
caught) with the extension's ``Content-Type`` (the
``IMAGE_MIMES`` map — one map, one truth) and
``Cache-Control: private, max-age=3600`` (content-hashed bytes —
long enough, bustable by re-upload)."""
_cleanup_kb(db)
try:
data = PNG_1X1 + ext.encode("ascii") # per-extension sentinel bytes
doc = _seed_image_doc(db, tmp_path, path=f"pic{ext}", data=data)
r = admin_client.get(f"/api/documents/{doc.id}/image")
assert r.status_code == 200
assert r.headers["content-type"] == mime # exact type per extension
assert r.headers["cache-control"] == "private, max-age=3600"
assert r.content == data # the exact uploaded bytes, nothing else
finally:
_cleanup_kb(db)
def test_image_route_404_matrix(admin_client: TestClient, db, tmp_path: Path) -> None:
"""Every non-servable case is 404 ``document not found`` (the
router's unknown-document shape — the same detail string the
content endpoint uses): a missing id, a MALFORMED id (an unparseable
string maps here, not to a 422 — a guessed id is an unknown
document), a text doc, an image doc whose ``image_path`` is NULL,
and a row whose copy was lost on disk (defensive — the row exists,
the bytes don't)."""
_cleanup_kb(db)
try:
_seed_doc(db, "TextSrc", "note.md", "Note", 1, datetime.now(UTC))
db.commit() # the module's _seed_doc leaves the row uncommitted
text_doc_id = db.scalar(
select(Document.id).where(
Document.source == "TextSrc", Document.path == "note.md"
)
)
no_copy = _seed_image_doc(db, tmp_path, path="nopy.png", image_path=None)
lost = _seed_image_doc(db, tmp_path, path="lost.png")
assert lost.image_path is not None # the "auto" copy was written
Path(lost.image_path).unlink() # the copy is lost (the row remains)
for doc_id in (
str(uuid.uuid4()), # missing id
"not-a-uuid", # malformed id → 404, not 422
str(text_doc_id), # text doc
str(no_copy.id), # image doc, image_path NULL
str(lost.id), # image doc, copy lost
):
r = admin_client.get(f"/api/documents/{doc_id}/image")
assert r.status_code == 404, doc_id
assert r.json() == {"detail": "document not found"}, doc_id
finally:
_cleanup_kb(db)
def test_image_route_requires_user_like_the_content_endpoint(
admin_client: TestClient, db, tmp_path: Path
) -> None:
"""Phase 79 posture (the task's "PUBLIC, like the document content
endpoint" — the content endpoint has been user-gated since phase
79; the ONLY anonymous surface is the shared chats, PLAN A10): an
anonymous caller gets 401 ``authentication required`` before any
row is read (a FRESH client — the module's fixture ``client``
stays unsigned here), a signed-in caller gets the bytes."""
_cleanup_kb(db)
try:
doc = _seed_image_doc(db, tmp_path)
anonymous = TestClient(fastapi_app)
r = anonymous.get(f"/api/documents/{doc.id}/image")
assert r.status_code == 401
assert r.json() == {"detail": "authentication required"}
# The signed-in client (admin) passes.
assert admin_client.get(f"/api/documents/{doc.id}/image").status_code == 200
finally:
_cleanup_kb(db)
def test_content_endpoint_exposes_the_image_affordance(
admin_client: TestClient, db, tmp_path: Path
) -> None:
"""The content endpoint (the viewer's data source): ``is_image`` is
ALWAYS present (text doc: false — the one new key; the wire shape
gains nothing else), and ``image_url`` — the bytes route's path —
is present for an image doc and ABSENT for a text doc (never null,
the ``DocContent`` omission rule)."""
_cleanup_kb(db)
try:
_seed_doc(db, "TextSrc", "note.md", "Note", 1, datetime.now(UTC))
db.commit() # the app's endpoint session reads committed data only
r = admin_client.get(
"/api/documents/content", params={"source": "TextSrc", "path": "note.md"}
)
assert r.status_code == 200
body = r.json()
assert body["is_image"] is False
assert "image_url" not in body # absent — never null (text doc)
doc = _seed_image_doc(db, tmp_path, source="ImgSrc", path="pic.png")
r = admin_client.get(
"/api/documents/content", params={"source": "ImgSrc", "path": "pic.png"}
)
assert r.status_code == 200
body = r.json()
assert body["is_image"] is True
assert body["image_url"] == f"/api/documents/{doc.id}/image"
assert body["content"] == body["summary"] # the description (task 03)
finally:
_cleanup_kb(db)
def _tree_file_nodes_all(sources) -> list[dict]:
"""Every file node of a tree response, walked recursively (all
sources — env-registered 0-document sources may join the response
when the ``git_sources`` table is truncated, and they carry no
file nodes; the assertions below hold over whatever files exist).
"""
files: list[dict] = []
def _walk(node: dict) -> None:
for child in node.get("children", ()):
if child["kind"] == "file":
files.append(child)
else:
_walk(child)
for source in sources:
_walk(source)
return files
def test_tree_image_file_node_affordance_and_text_node_byte_identical(
admin_client: TestClient, db, tmp_path: Path
) -> None:
"""The tree (the RAG view's single fetch): an image doc's file node
carries the thumbnail affordance (``is_image`` true,
``image_url`` = the bytes route's path, ``summary`` verbatim — the
RAG view's thumbnail ``alt``); EVERY text file node keeps the
pre-phase wire shape byte-identically (the six keys — no
``is_image``/``image_url``/``summary`` — the phase's
byte-identical criterion: the fields are row-driven, so a KB with
no image rows serializes exactly as pre-phase)."""
_truncate_tree_tables(db)
try:
base = datetime.now(UTC)
_seed_doc(db, "MixedSrc", "a.md", "A", 1, base)
img = _seed_image_doc(db, tmp_path, source="MixedSrc", path="pic.png")
r = admin_client.get("/api/docs/tree") # both seeds committed (_seed_image_doc)
assert r.status_code == 200
files = {f["path"]: f for f in _tree_file_nodes_all(r.json()["sources"])}
# Text node: byte-identical pre-phase wire shape (no image keys).
assert set(files["a.md"]) == {
"kind", "path", "title", "chunks", "created_at", "indexed_at"
}
# Image node: the affordance rides the node.
pic = files["pic.png"]
assert pic["is_image"] is True
assert pic["image_url"] == f"/api/documents/{img.id}/image"
assert pic["summary"] == "A red square on a white background."
finally:
_truncate_tree_tables(db)
def test_tree_with_no_image_rows_is_byte_identical(admin_client: TestClient, db) -> None:
"""The pre-phase KB (no ``is_image`` rows): the image-docs map is
empty and EVERY file node serializes in the pre-phase shape — the
row-driven fields introduce no wire change at all (the phase's
byte-identical criterion, the toggle irrelevant)."""
_truncate_tree_tables(db)
try:
base = datetime.now(UTC)
_seed_doc(db, "PlainSrc", "x.md", "X", 2, base)
db.commit() # the app's endpoint session reads committed data only
r = admin_client.get("/api/docs/tree")
assert r.status_code == 200
files = _tree_file_nodes_all(r.json()["sources"])
assert len(files) == 1 # the seeded doc (env sources carry no files)
assert set(files[0]) == {
"kind", "path", "title", "chunks", "created_at", "indexed_at"
}
finally:
_truncate_tree_tables(db)