Files
brain-of-reese/tests/integration/test_importer_e2e.py
T

104 lines
4.5 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Integration test: importer end-to-end against ``tests/fixtures/docs/``.
Runs the real import pipeline (walk → chunk → embed → upsert) into the
local compose Postgres, then checks the DB state *and* the API shape a
browser would consume. Embeddings come from a deterministic in-process
fake, so no network is needed.
"""
from __future__ import annotations
import asyncio
from pathlib import Path
from sqlalchemy import func, select, text
from app.models import Chunk, Document
from app.rag.importer import import_sources
from tests.fakes import FakeEmbedder
FIXTURES = Path(__file__).resolve().parents[1] / "fixtures" / "docs"
EXPECTED_DOCS = {
("docs", "homelab/kubernetes.md"),
("docs", "homelab/backups.md"),
("docs", "deployments/new-service.md"),
("docs", "homelab/container_gitlab/gitlab.md"),
("docs", "homelab/container_gitlab/gitlab-compose.yaml"),
("docs", "homelab/networking/static-dns.json"),
("docs", "homelab/scripts/uptime_probe.py"),
("docs", "homelab/ssh/ssh_aliases.txt"),
}
def test_import_fixtures_end_to_end(admin_client, db) -> None:
db.execute(text("TRUNCATE chunks, documents, query_log"))
db.commit()
llm = FakeEmbedder()
summary = asyncio.run(import_sources([FIXTURES], llm, session=db))
# Eight A9-format files; .hidden/junk.md is out of scope (A9 revised).
assert (summary.files, summary.added, summary.unchanged) == (8, 8, 0)
assert summary.chunks >= 8
assert summary.formats == {"md": 4, "yaml": 1, "json": 1, "py": 1, "txt": 1}
# PLAN §9 per-format summary line: highest count first, then alpha.
assert summary.format_counts() == "md:4,json:1,py:1,txt:1,yaml:1"
docs = db.scalars(select(Document)).all()
assert {(d.source, d.path) for d in docs} == EXPECTED_DOCS
titles = {d.path: d.title for d in docs}
assert titles["homelab/kubernetes.md"] == "Kubernetes Homelab Cluster"
assert titles["deployments/new-service.md"] == "Deploying a New Service"
assert titles["homelab/container_gitlab/gitlab.md"] == "Gitlab"
# Non-markdown titles come from the file stem (a leading ``#`` or docstring
# line is a comment there, not a heading).
assert titles["homelab/container_gitlab/gitlab-compose.yaml"] == "gitlab-compose"
assert titles["homelab/scripts/uptime_probe.py"] == "uptime_probe"
assert titles["homelab/networking/static-dns.json"] == "static-dns"
assert titles["homelab/ssh/ssh_aliases.txt"] == "ssh_aliases"
# Hidden junk was never imported.
assert not any(".hidden" in d.path for d in docs)
# Full content is stored — that is what the RAG context will be.
k8s = next(d for d in docs if d.path == "homelab/kubernetes.md")
assert "Talos Linux" in k8s.content and k8s.content_hash
# Phase 30: the four non-markdown fixtures each gained one embedded
# ``is_summary`` chunk, so the DB holds content + summary chunks.
n_chunks = db.scalar(select(func.count()).select_from(Chunk))
assert n_chunks == summary.chunks + summary.summaries
for c in db.scalars(select(Chunk)).all():
assert c.embedding is not None and len(c.embedding) == 768
assert summary.summary_errors == 0
for d in docs:
non_md = Path(d.path).suffix.lower() not in (".md", ".markdown")
schunks = [c for c in d.chunks if c.is_summary]
if non_md:
# Lite summary stored + exactly one embedded summary chunk (−1).
assert d.summary is not None, f"{d.path} should have a summary"
assert len(schunks) == 1
assert schunks[0].position == -1
assert schunks[0].content == d.summary
assert schunks[0].embedding is not None
else:
# Markdown docs never get a summary (phase 30 scope).
assert d.summary is None and not schunks
assert summary.summaries == sum(
1 for d in docs if Path(d.path).suffix.lower() not in (".md", ".markdown")
)
# The Sources page consumes exactly this shape.
r = admin_client.get("/api/docs") # phase 16: the catalog is admin-only
assert r.status_code == 200
body = r.json()
assert len(body["documents"]) == 8
assert all(d["chunks"] >= 1 for d in body["documents"])
# Idempotent re-run: nothing re-embedded.
calls_before = len(llm.calls)
s2 = asyncio.run(import_sources([FIXTURES], llm, session=db))
assert s2.unchanged == 8 and s2.added == 0
assert len(llm.calls) == calls_before # unchanged → no embedding requests
db.execute(text("TRUNCATE chunks, documents, query_log"))
db.commit()