Files
brain-of-reese/tests/unit/test_importer.py
T
ducoterra d731169b8b
Build and Push Containers / build-and-push-app (push) Successful in 1m44s
Build and Push Containers / build-and-push-db (push) Successful in 13s
phase: 105_hidden_folders_toggle
All completion criteria verified. Everything is green.

**Phase 105 final verification pass — all criteria verified**

- Verified the full implementation in the working tree: `git_sources.include_hidden` column + alembic `0019` (dev DB at head, column present), `iter_importable_files`/`import_sources` flag support with `str(root)`-keyed map used by both walk and progress pre-walk, `GitSourcePatchIn` rename with optional fields, sync/CLI pipeline wiring (OR-collision), and the per-row "Hidden" checkbox + tag + error line on the Sources page
- Unit + integration: `uv run pytest` → exit 0 (2148 tests collected, all pass; this sandbox occasionally swallows pytest's final status line — exit codes verified)
- Coverage: `uv run pytest --cov=app --cov-report=term-missing` → **TOTAL 99%** (3879 stmts, 15 miss) — >90% gate ✓
- Dedicated E2E: `uv run pytest tests/e2e/test_hidden_folders_toggle.py -v --no-cov` → **6 passed in 23.20s** (DB up, in isolation)
- Regression E2E in isolation: `test_source_ignore_paths` 6 passed, `test_git_sources_admin` 6 passed, `test_local_directory_sources` 3 passed, `test_sync_button` 3 passed, `test_smoke` 3 passed
- Lint/types: `uv run ruff check .` + `uv run pyright` → clean (0 errors/warnings)

**Completion criteria:** (1) checkbox persists via PATCH 200 → "hidden on" tag + GET round-trips `include_hidden: true`; failure path reverts box + `role="alert"` canned message ✓; (2) flag OFF byte-identical (only `visible.md` indexed), ON indexes `.hidden/note.md` into the KB catalog, `EXCLUDED_DIRS` excluded both states ✓; (3) A2: flag OFF → `detail.pruned==1`, doc gone from catalog ✓; (4) PATCH bool-only/list-only/both/neither no-op, phase-89 fixed 422s unchanged, 404, anonymous 403 (incl. bool-only body) ✓; (5) env-fallback rows render no checkbox, WCAG-clean (aria-label, keyboard focus, visible label, text tag) ✓; (6) full gate green ✓; (7) commit left to the harness per instructions (no `git add`/`commit` run; phase files untouched).

**Deviations:** none — no defects found; no code changes were needed on this pass.

**Next pending phase:** `.agents/phases/todo/98_sync_summary_visibility`.
2026-09-12 21:26:30 -04:00

938 lines
38 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Unit tests: importer directory walk + sha256 delta logic + summaries.
The walk tests are pure filesystem (``tmp_path``); the delta and summary
tests run against the local compose Postgres (preferred — a real vector
table), skipping with clear instructions when the stack is not up.
Summaries (phase 30): non-markdown files get a ``lite``-model summary via
the fake's deterministic ``chat`` (``"Summary of <first token>"``); the
sentinel word ``SUMMARY-BLOWUP`` makes ``chat`` raise :class:`LLMError`
for the fail-soft path.
"""
from __future__ import annotations
import asyncio
import logging
from pathlib import Path
import pytest
from sqlalchemy import func, select
import app.rag.importer as importer
from app.config import Settings
from app.models import Chunk, Document
from app.rag.importer import (
EXCLUDED_DIRS,
ImportSummary,
_store_summary,
import_sources,
iter_importable_files,
match_extension,
)
from app.rag.llm import EmbeddingError
from tests.fakes import FakeEmbedder
#: The original seven A9 formats as dotted suffixes (pre-phase-47 default
#: set). The phase-47 walk test uses the live config default
#: (``Settings().import_extension_set`` — now seventeen formats).
DEFAULT_EXTS = frozenset({".md", ".markdown", ".txt", ".yaml", ".yml", ".json", ".py"})
class _PoisonEmbedder(FakeEmbedder):
"""Fails (like a real endpoint) on any text containing 'poison'."""
async def embed(self, texts: list[str]) -> list[list[float]]:
if any("poison" in t for t in texts):
raise EmbeddingError("embeddings endpoint refused the input (simulated)")
return await super().embed(texts)
class _CapEmbedder(FakeEmbedder):
"""Simulates the endpoint's ~1024-token input cap at ~1.1 chars/token:
any single text over 1000 chars is rejected (URL-dense worst case)."""
async def embed(self, texts: list[str]) -> list[list[float]]:
if any(len(t) > 1000 for t in texts):
raise EmbeddingError(
"a single 1100-char chunk exceeded the endpoint's per-request "
"input token cap — lower BOR_CHUNK_TARGET_CHARS and re-import"
)
return await super().embed(texts)
def _embedder_with_extensions(extensions: str) -> FakeEmbedder:
"""A :class:`FakeEmbedder` whose settings carry a custom
``BOR_IMPORT_EXTENSIONS`` CSV (phase 102 — the import scope the walk
and the ``formats`` counter read from ``llm.settings``)."""
llm = FakeEmbedder()
llm.settings = Settings(_env_file=None, import_extensions=extensions) # pyright: ignore[reportCallIssue]
return llm
def _cleanup_source(db, source: str) -> None:
for doc in db.scalars(select(Document).where(Document.source == source)).all():
db.delete(doc)
db.commit()
def test_iter_importable_files_excludes_noncontent_dirs_and_hidden(tmp_path: Path) -> None:
"""Well-known non-content dirs, hidden (dot-) dirs/files, and non-A9
extensions are all skipped; the A9 formats pass."""
root = tmp_path / "proj"
for d in (
"notes/sub",
".venv/lib",
"node_modules/x",
".git",
"__pycache__",
".pytest_cache",
"dist",
"build",
".esphome/.espressif", # vendored hidden cache — the real A9 case
):
(root / d).mkdir(parents=True)
files = {
# content that must be found:
"README.md": "readme",
"notes/sub/deep.md": "deep",
"compose.yaml": "services: {}",
"legacy.YML": "a: b", # case-insensitive suffix
"notes/sub/agent.py": "x = 1",
"config.json": "{}",
"README.txt": "plain",
"notes/sub/deep.markdown": "md2",
# must be skipped:
".venv/lib/junk.md": "junk",
"node_modules/x/j.md": "j",
".git/c.md": "g",
"__pycache__/c.md": "p",
".pytest_cache/c.md": "pc",
"dist/d.md": "d",
"build/b.md": "b",
".esphome/.espressif/secret.md": "vendor",
".secret.md": "hidden file", # dot-prefixed FILE, not just dir
"notes/sub/notes.csv": "a,b", # not an A9 format
"notes/sub/file.md.bak": "x",
}
for rel, text in files.items():
(root / rel).write_text(text)
found = {p.relative_to(root).as_posix() for p in iter_importable_files(root, DEFAULT_EXTS)}
assert found == {
"README.md",
"notes/sub/deep.md",
"compose.yaml",
"legacy.YML",
"notes/sub/agent.py",
"config.json",
"README.txt",
"notes/sub/deep.markdown",
}
def test_iter_importable_files_missing_dir_yields_nothing(tmp_path: Path) -> None:
assert iter_importable_files(tmp_path / "definitely-missing", DEFAULT_EXTS) == []
def test_iter_importable_files_walker_picks_up_new_formats_by_default(
tmp_path: Path,
) -> None:
"""Phase 47 (A9 revised 2026-08-27): the ten new formats — the full
quadlet family + j2 — walk under the *default* extension set (no env
configuration needed), the original seven still walk (regression), and
unknown/hidden-dir/excluded-dir filtering is unchanged for them."""
root = tmp_path / "quadproj"
(root / ".esphome").mkdir(parents=True)
(root / "node_modules").mkdir()
files = {
# the ten new formats (A9 revised 2026-08-27):
"web.container": "# RESE-QUADLET-SENTINEL-77aa\n[Container]\nImage=alpine\n",
"lan.network": "[Network]\nDriver=bridge\n",
"cache.volume": "[Volume]\nDriver=local\n",
"alpine.image": "[Image]\nImages=alpine\n",
"app.pod": "[Pod]\nPodName=app\n",
"k3s.kube": "apiVersion: v1\nkind: Pod\n",
"zram.swap": "[Swap]\nFile=/swapfile\n",
"fedora.os": "[OS]\nImage=fedora\n",
"edge.endpoint": "[Endpoint]\nPort=8080\n",
"deploy.j2": "{% for s in services %}\n[{{ s }}]\n{% endfor %}\n",
# the original seven — regression:
"note.md": "# n\n",
"note.markdown": "# m\n",
"note.txt": "t\n",
"cfg.yaml": "a: b\n",
"cfg.yml": "a: b\n",
"data.json": "{}\n",
"agent.py": "x = 1\n",
# must be skipped — filtering is unchanged for the new formats too:
"mystery.xyz": "unknown extension",
".esphome/x.container": "hidden dir (new-format file)",
"node_modules/y.container": "excluded dir (new-format file)",
}
for rel, text in files.items():
(root / rel).write_text(text)
# The live config default (Settings with no env override) — no
# BOR_IMPORT_EXTENSIONS needed for the new formats to walk.
exts = Settings(_env_file=None).import_extension_set # pyright: ignore[reportCallIssue]
found = {p.relative_to(root).as_posix() for p in iter_importable_files(root, exts)}
assert found == {
# all ten new formats:
"web.container",
"lan.network",
"cache.volume",
"alpine.image",
"app.pod",
"k3s.kube",
"zram.swap",
"fedora.os",
"edge.endpoint",
"deploy.j2",
# …and the original seven:
"note.md",
"note.markdown",
"note.txt",
"cfg.yaml",
"cfg.yml",
"data.json",
"agent.py",
}
def test_iter_importable_files_respects_custom_extension_filter(tmp_path: Path) -> None:
"""A narrower filter (e.g. md only) excludes the other A9 formats."""
root = tmp_path / "filtered"
root.mkdir()
(root / "a.md").write_text("a")
(root / "b.yaml").write_text("a: b")
(root / "c.py").write_text("x = 1")
found = {p.name for p in iter_importable_files(root, frozenset([".md"]))}
assert found == {"a.md"}
def test_match_extension_matrix() -> None:
"""Phase 102, D1 — the matching matrix: the A9 dotted-suffix rule first,
then the extensionless exact-name rule (case-insensitive), exact name
only — no partial matching, suffixed files governed by their suffix."""
exts = frozenset({".md", ".dockerfile"})
# Rule 1 — the dotted suffix (case-insensitive, as today):
assert match_extension(Path("kubernetes.md"), frozenset({".md"})) == "md"
assert match_extension(Path("Kubernetes.MD"), frozenset({".md"})) == "md"
# Rule 2 — extensionless files by exact lowercased FULL filename:
assert match_extension(Path("Dockerfile"), exts) == "dockerfile"
assert match_extension(Path("DOCKERFILE"), exts) == "dockerfile"
# …only when the token is actually in the set:
assert match_extension(Path("Dockerfile"), frozenset({".md"})) is None
# Exact name only — compound lookalikes never match the token:
assert match_extension(Path("mydockerfile"), exts) is None
# A suffixed file is governed by its suffix, never its name:
assert match_extension(Path("Dockerfile.dev"), exts) is None
assert match_extension(Path("Dockerfile.dev"), frozenset({".md", ".dev"})) == "dev"
# An out-of-scope suffix is out of scope, name be damned:
assert match_extension(Path("readme.rst"), frozenset({".md"})) is None
def test_iter_importable_files_walks_extensionless_name_tokens(tmp_path: Path) -> None:
"""Phase 102, D1 — an extensionless file walks iff its lowercased full
name is a token (``md,dockerfile,containerfile`` here): lookalikes and
the dot-prefixed hidden file stay skipped by the pre-existing rules."""
root = tmp_path / "build"
root.mkdir()
for name in (
"Dockerfile",
"Containerfile",
"mydockerfile", # compound name — never matches the `dockerfile` token
"Dockerfile.dev", # governed by its .dev suffix (not a token here)
".dockerfile", # dot-prefixed FILE — the hidden-component rule
"notes.md",
):
(root / name).write_text(f"# {name}\n\nbody {name}\n")
exts = frozenset({".md", ".dockerfile", ".containerfile"})
found = [p.name for p in iter_importable_files(root, exts)]
assert found == ["Containerfile", "Dockerfile", "notes.md"]
def test_excluded_dirs_match_plan_anchor_a9() -> None:
assert {
".venv", "node_modules", ".git", "__pycache__", ".pytest_cache", "dist", "build"
} == EXCLUDED_DIRS
def test_added_then_unchanged_then_updated_then_pruned(db, tmp_path: Path) -> None:
root = tmp_path / "src"
root.mkdir()
(root / "a.md").write_text("# A\n\nalpha\n\n## Sub\n\nmore alpha\n")
(root / "b.md").write_text("# B\n\nbeta\n")
llm = FakeEmbedder()
try:
s1 = asyncio.run(import_sources([root], llm, session=db))
assert (s1.files, s1.added, s1.unchanged, s1.updated, s1.pruned) == (2, 2, 0, 0, 0)
# a.md has two sections (2 chunks), b.md one (1 chunk).
assert s1.chunks == 3
# Embeddings are stored with the configured dimension.
n = db.scalar(
select(func.count())
.select_from(Chunk)
.join(Document, Document.id == Chunk.document_id)
.where(Document.source == root.name)
)
assert n == 3
for c in db.scalars(
select(Chunk)
.join(Document, Document.id == Chunk.document_id)
.where(Document.source == root.name)
).all():
assert c.embedding is not None and len(c.embedding) == 768
s2 = asyncio.run(import_sources([root], llm, session=db))
assert s2.added == 0 and s2.unchanged == 2
(root / "a.md").write_text("# A\n\nalpha CHANGED\n")
s3 = asyncio.run(import_sources([root], llm, session=db))
assert s3.updated == 1 and s3.unchanged == 1
doc = db.scalar(
select(Document).where(Document.source == root.name, Document.path == "a.md")
)
assert doc is not None and "CHANGED" in doc.content
(root / "a.md").unlink()
s4 = asyncio.run(import_sources([root], llm, session=db, prune=True))
assert s4.pruned == 1
assert db.scalar(
select(Document).where(Document.source == root.name, Document.path == "a.md")
) is None
# Chunks of the pruned document are gone (FK cascade).
n_after = db.scalar(
select(func.count())
.select_from(Chunk)
.join(Document, Document.id == Chunk.document_id)
.where(Document.source == root.name)
)
assert n_after == 1
finally:
_cleanup_source(db, root.name)
def test_embedding_failure_is_logged_and_import_continues(db, tmp_path: Path) -> None:
"""A file the embedding endpoint refuses must not abort the whole KB:
its rows are rolled back, the error is counted, and other files import."""
root = tmp_path / "mixed"
root.mkdir()
(root / "bad.md").write_text("# Bad\n\npoison content that the endpoint refuses\n")
(root / "good.md").write_text("# Good\n\nperfectly fine content\n")
try:
summary = asyncio.run(import_sources([root], _PoisonEmbedder(), session=db))
assert summary.files == 2
assert summary.errors == 1
assert summary.added == 1 # only good.md
# bad.md left no row and no orphan chunks behind (rolled back).
assert db.scalar(
select(Document).where(Document.source == root.name, Document.path == "bad.md")
) is None
assert db.scalar(
select(func.count())
.select_from(Chunk)
.join(Document, Document.id == Chunk.document_id)
.where(Document.source == root.name, Document.path == "bad.md")
) == 0
assert db.scalar(
select(Document).where(Document.source == root.name, Document.path == "good.md")
) is not None
finally:
_cleanup_source(db, root.name)
def test_oversized_chunk_triggers_adaptive_rechunk(db, tmp_path: Path) -> None:
"""A URL-dense paragraph the endpoint rejects must be re-chunked smaller
for that file only — the import still succeeds."""
root = tmp_path / "dense"
root.mkdir()
# One ~1165-char paragraph: under the 1200-char hard cap, over the
# simulated token cap. The retry at 600 chars must split it.
para = "see https://example.com/" + "a" * 1100
(root / "dense.md").write_text(f"# D\n\n{para}\n")
try:
summary = asyncio.run(import_sources([root], _CapEmbedder(), session=db))
assert summary.errors == 0
assert summary.added == 1
doc = db.scalar(
select(Document).where(Document.source == root.name, Document.path == "dense.md")
)
assert doc is not None
assert len(doc.chunks) >= 2 # re-chunked smaller than the hard cap
assert all(len(c.content) <= 1000 for c in doc.chunks)
assert all(c.embedding is not None for c in doc.chunks)
# The content survives the split.
assert "".join(c.content for c in doc.chunks).count("a" * 500) >= 1
finally:
_cleanup_source(db, root.name)
def test_missing_source_dir_is_skipped(db, tmp_path: Path) -> None:
llm = FakeEmbedder()
summary = asyncio.run(import_sources([tmp_path / "missing"], llm, session=db))
assert summary.files == 0 and summary.added == 0
def test_limit_caps_files_and_disables_prune(db, tmp_path: Path) -> None:
root = tmp_path / "limited"
root.mkdir()
for name in ("a.md", "b.md", "c.md"):
(root / name).write_text(f"# {name}\n\nbody {name}\n")
llm = FakeEmbedder()
try:
summary = asyncio.run(import_sources([root], llm, limit=2, session=db, prune=True))
assert summary.files == 2 and summary.added == 2
# c.md was never walked, so it must NOT be pruned (prune disabled
# under --limit) — and nothing else disappears either.
assert summary.pruned == 0
assert db.scalar(
select(func.count()).select_from(Document).where(Document.source == root.name)
) == 2
finally:
_cleanup_source(db, root.name)
def test_limit_must_be_positive(db, tmp_path: Path) -> None:
with pytest.raises(ValueError):
asyncio.run(import_sources([tmp_path], FakeEmbedder(), limit=0, session=db))
def test_prune_is_scoped_to_the_given_sources(db, tmp_path: Path) -> None:
src_x = tmp_path / "SourceX"
src_y = tmp_path / "SourceY"
src_x.mkdir()
src_y.mkdir()
(src_x / "x.md").write_text("# X\n\nx body\n")
(src_y / "y.md").write_text("# Y\n\ny body\n")
llm = FakeEmbedder()
try:
asyncio.run(import_sources([src_x, src_y], llm, session=db))
# Re-import ONLY source Y (y.md removed) with prune: source X's doc
# must survive — prune never touches sources not passed to this run.
(src_y / "y.md").unlink()
summary = asyncio.run(import_sources([src_y], llm, session=db, prune=True))
assert summary.pruned == 1
assert db.scalar(
select(Document).where(Document.source == "SourceX", Document.path == "x.md")
) is not None
finally:
_cleanup_source(db, "SourceX")
_cleanup_source(db, "SourceY")
def test_chunk_positions_and_titles(db, tmp_path: Path) -> None:
root = tmp_path / "titled"
root.mkdir()
(root / "multi.md").write_text("# Real Title\n\n## One\n\na\n\n## Two\n\nb\n")
(root / "noh1.md").write_text("## Only heading\n\nbody\n")
llm = FakeEmbedder()
try:
asyncio.run(import_sources([root], llm, session=db))
titles = {
d.path: d.title
for d in db.scalars(select(Document).where(Document.source == root.name)).all()
}
assert titles["multi.md"] == "Real Title" # H1 wins
assert titles["noh1.md"] == "noh1" # …else the file stem
doc = db.scalar(
select(Document).where(Document.source == root.name, Document.path == "multi.md")
)
assert doc is not None
positions = sorted(c.position for c in doc.chunks)
assert positions == list(range(len(doc.chunks))) and len(doc.chunks) >= 2
finally:
_cleanup_source(db, root.name)
def test_multi_format_import_counts_per_format_and_titles_stem(db, tmp_path: Path) -> None:
"""A9 formats all import; the summary records per-format counts, and
non-markdown titles come from the file stem (a ``#`` line is a comment
there, not a heading)."""
root = tmp_path / "multi"
(root / "svc").mkdir(parents=True)
(root / "guide.md").write_text("# Real Heading\n\nbody\n")
(root / "svc" / "compose.yaml").write_text("# a comment\nservices:\n gitlab: {}\n")
(root / "svc" / "agent.py").write_text("# docstring-like comment\ndef ping():\n return 1\n")
(root / "inventory.json").write_text('{"hosts": []}\n')
(root / "notes.txt").write_text("plain text notes\n")
llm = FakeEmbedder()
try:
summary = asyncio.run(import_sources([root], llm, session=db))
assert summary.files == 5
assert summary.added == 5
assert summary.formats == {"md": 1, "yaml": 1, "py": 1, "json": 1, "txt": 1}
# PLAN §9 summary line: counts, highest first, ext:name pairs.
assert summary.format_counts() == "json:1,md:1,py:1,txt:1,yaml:1"
titles = {
d.path: d.title
for d in db.scalars(select(Document).where(Document.source == root.name)).all()
}
assert titles["guide.md"] == "Real Heading" # markdown keeps the H1
assert titles["svc/compose.yaml"] == "compose" # …comment is not a heading
assert titles["svc/agent.py"] == "agent"
assert titles["inventory.json"] == "inventory"
assert titles["notes.txt"] == "notes"
finally:
_cleanup_source(db, root.name)
def test_quadlet_and_j2_files_get_stem_titles_and_per_format_counts(
db, tmp_path: Path
) -> None:
"""Phase 47: the ten new formats import like any other A9 format — the
per-format counts land in the summary, every file yields content
chunks, and a leading ``#`` line (a TOML comment in quadlet files, not
a heading) never becomes the title: the file stem wins."""
root = tmp_path / "quadsrc"
(root / "quadlet").mkdir(parents=True)
(root / "tpl").mkdir(parents=True)
(root / "quadlet" / "web.container").write_text(
"# RESE-QUADLET-SENTINEL-77aa\n\n[Container]\nImage=alpine\n"
)
(root / "quadlet" / "lan.network").write_text("# net\n[Network]\nDriver=bridge\n")
(root / "quadlet" / "cache.volume").write_text("[Volume]\nDriver=local\n")
(root / "quadlet" / "alpine.image").write_text("[Image]\nImages=alpine\n")
(root / "quadlet" / "app.pod").write_text("[Pod]\nPodName=app\n")
(root / "quadlet" / "k3s.kube").write_text("apiVersion: v1\nkind: Pod\n")
(root / "quadlet" / "zram.swap").write_text("[Swap]\nFile=/swapfile\n")
(root / "quadlet" / "fedora.os").write_text("[OS]\nImage=fedora\n")
(root / "quadlet" / "edge.endpoint").write_text("[Endpoint]\nPort=8080\n")
(root / "tpl" / "deploy.j2").write_text(
"{% for s in services %}\n[{{ s }}]\nport = {{ s.port }}\n{% endfor %}\n"
)
llm = FakeEmbedder()
try:
summary = asyncio.run(import_sources([root], llm, session=db))
assert summary.files == 10 and summary.added == 10
assert summary.formats == {
"container": 1, "network": 1, "volume": 1, "image": 1, "pod": 1,
"kube": 1, "swap": 1, "os": 1, "endpoint": 1, "j2": 1,
}
docs = {
d.path: d
for d in db.scalars(select(Document).where(Document.source == root.name)).all()
}
titles = {p: d.title for p, d in docs.items()}
assert titles == {
# the ``#`` line is a comment in these formats — stem titles:
"quadlet/web.container": "web",
"quadlet/lan.network": "lan",
"quadlet/cache.volume": "cache",
"quadlet/alpine.image": "alpine",
"quadlet/app.pod": "app",
"quadlet/k3s.kube": "k3s",
"quadlet/zram.swap": "zram",
"quadlet/fedora.os": "fedora",
"quadlet/edge.endpoint": "edge",
"tpl/deploy.j2": "deploy",
}
for doc in docs.values():
content = [c for c in doc.chunks if not c.is_summary]
assert content # every new-format file yields content chunks
assert all(c.embedding is not None and len(c.embedding) == 768 for c in content)
finally:
_cleanup_source(db, root.name)
# ---------- phase 30: lite-model summaries for non-markdown files ----------
def test_non_markdown_file_gets_stored_and_indexed_summary(db, tmp_path: Path) -> None:
"""A ``.yaml`` file is summarized: ``documents.summary`` is set and one
``is_summary`` chunk (position −1, embedded) is indexed alongside the
content chunks."""
root = tmp_path / "sumsrc"
root.mkdir()
(root / "svc.yaml").write_text("alpha services:\n gitlab:\n port: 8929\n")
llm = FakeEmbedder()
try:
summary = asyncio.run(import_sources([root], llm, session=db))
assert summary.added == 1
assert summary.summaries == 1
assert summary.summary_errors == 0
doc = db.scalar(
select(Document).where(Document.source == root.name, Document.path == "svc.yaml")
)
assert doc is not None
assert doc.summary is not None
# Deterministic fake reply + the code-appended pointer line.
assert doc.summary.startswith("Summary of alpha")
assert doc.summary.endswith(f"Source: {root.name}/svc.yaml")
schunks = [c for c in doc.chunks if c.is_summary]
assert len(schunks) == 1
assert schunks[0].position == -1
assert schunks[0].content == doc.summary
assert schunks[0].embedding is not None and len(schunks[0].embedding) == 768
# Content chunks stay 0-based and are never flagged as summaries.
content = [c for c in doc.chunks if not c.is_summary]
assert sorted(c.position for c in content) == list(range(len(content)))
finally:
_cleanup_source(db, root.name)
def test_markdown_file_never_gets_summary(db, tmp_path: Path) -> None:
"""Markdown is already natural language: no summary, no ``is_summary``
chunk, and the ``lite`` model is never called."""
root = tmp_path / "mdsrc"
root.mkdir()
(root / "note.md").write_text("# Note\n\nmarkdown body\n")
llm = FakeEmbedder()
try:
summary = asyncio.run(import_sources([root], llm, session=db))
assert summary.added == 1
assert summary.summaries == 0 and summary.summary_errors == 0
assert llm.chat_calls == [] # the model was never asked
doc = db.scalar(
select(Document).where(Document.source == root.name, Document.path == "note.md")
)
assert doc is not None
assert doc.summary is None
assert doc.chunks and all(not c.is_summary for c in doc.chunks)
finally:
_cleanup_source(db, root.name)
def test_summary_failure_is_fail_soft(db, tmp_path: Path) -> None:
"""A ``lite``-model failure must never lose the document: the file is
fully indexed (content chunks + embeddings), ``documents.summary`` stays
NULL, and the failure is counted in ``summary_errors``."""
root = tmp_path / "blowup"
root.mkdir()
(root / "bad.txt").write_text("SUMMARY-BLOWUP the lite model chokes on this\n")
llm = FakeEmbedder()
try:
summary = asyncio.run(import_sources([root], llm, session=db))
assert summary.errors == 0 # the document itself imported fine
assert summary.added == 1
assert summary.summaries == 0
assert summary.summary_errors == 1
doc = db.scalar(
select(Document).where(Document.source == root.name, Document.path == "bad.txt")
)
assert doc is not None
assert doc.summary is None
assert len(doc.chunks) == 1
assert doc.chunks[0].embedding is not None # content chunk embedded
assert all(not c.is_summary for c in doc.chunks)
finally:
_cleanup_source(db, root.name)
def test_summary_chunk_is_replaced_on_reimport(db, tmp_path: Path) -> None:
"""Re-importing a changed non-markdown file keeps exactly one
``is_summary`` chunk — the old one is gone, the new summary is stored
and embedded, and the content chunks stay 0-based."""
root = tmp_path / "repl"
root.mkdir()
path = root / "cfg.yaml"
path.write_text("alpha settings:\n host: one\n")
llm = FakeEmbedder()
try:
asyncio.run(import_sources([root], llm, session=db))
path.write_text("bravo settings:\n host: two\n")
summary = asyncio.run(import_sources([root], llm, session=db))
assert summary.updated == 1
assert summary.summaries == 1 and summary.summary_errors == 0
doc = db.scalar(
select(Document).where(Document.source == root.name, Document.path == "cfg.yaml")
)
assert doc is not None
assert doc.summary is not None and doc.summary.startswith("Summary of bravo")
schunks = [c for c in doc.chunks if c.is_summary]
assert len(schunks) == 1 # the old one was deleted
assert schunks[0].position == -1
assert schunks[0].content == doc.summary
assert schunks[0].embedding is not None
assert "alpha" not in schunks[0].content # no stale summary text
assert sorted(c.position for c in doc.chunks if not c.is_summary) == [0]
finally:
_cleanup_source(db, root.name)
def test_store_summary_replaces_an_existing_summary_chunk(db, tmp_path: Path) -> None:
"""Replacement unit, driven directly: with a pre-existing
``is_summary`` chunk in place, ``_store_summary`` deletes the old one
and leaves exactly one (new) summary chunk + updated
``documents.summary`` — the at-most-one-summary invariant."""
root = tmp_path / "direct"
root.mkdir()
(root / "a.yaml").write_text("alpha x\n")
llm = FakeEmbedder()
try:
asyncio.run(import_sources([root], llm, session=db))
doc = db.scalar(
select(Document).where(Document.source == root.name, Document.path == "a.yaml")
)
assert doc is not None and doc.summary is not None
assert any(c.is_summary for c in doc.chunks) # the first import's summary
counters = ImportSummary()
asyncio.run(
_store_summary(
session=db, doc=doc, source=root.name, rel="a.yaml",
content=doc.content, llm=llm, summary=counters,
)
)
assert counters.summaries == 1 and counters.summary_errors == 0
schunks = [c for c in doc.chunks if c.is_summary]
assert len(schunks) == 1 # the old one was deleted
assert schunks[0].position == -1
assert schunks[0].content == doc.summary
assert schunks[0].embedding is not None
finally:
_cleanup_source(db, root.name)
def test_store_summary_fail_soft_leaves_document_untouched(db, tmp_path: Path) -> None:
"""A ``lite`` failure inside ``_store_summary`` rolls back only the
summary rows: the previous summary (if any) and the document survive,
and the failure is counted."""
root = tmp_path / "directfail"
root.mkdir()
(root / "a.yaml").write_text("alpha x\n")
llm = FakeEmbedder()
try:
asyncio.run(import_sources([root], llm, session=db))
doc = db.scalar(
select(Document).where(Document.source == root.name, Document.path == "a.yaml")
)
assert doc is not None and doc.summary is not None
previous_summary = doc.summary
(root / "a.yaml").write_text("SUMMARY-BLOWUP now the lite model fails\n")
counters = ImportSummary()
asyncio.run(
_store_summary(
session=db, doc=doc, source=root.name, rel="a.yaml",
content="SUMMARY-BLOWUP now the lite model fails\n",
llm=llm, summary=counters,
)
)
assert counters.summaries == 0 and counters.summary_errors == 1
assert doc.summary == previous_summary # rolled back, not nulled
schunks = [c for c in doc.chunks if c.is_summary]
assert len(schunks) == 1 # the old one survived the rollback
assert schunks[0].content == previous_summary
finally:
_cleanup_source(db, root.name)
def test_import_summary_log_line_includes_summary_counters(
caplog: pytest.LogCaptureFixture,
) -> None:
"""PLAN §9 summary line: the phase-30 counters sit between
``embed_batches`` and ``formats``."""
s = ImportSummary()
s.files, s.added, s.chunks, s.embed_batches = 3, 3, 5, 4
s.summaries, s.summary_errors = 2, 1
s.formats = {"md": 1, "yaml": 2}
with caplog.at_level(logging.INFO, logger="app.importer"):
s.log()
line = caplog.records[-1].getMessage()
assert line == (
"import: summary files=3 added=3 updated=0 unchanged=0 pruned=0 errors=0 "
"chunks=5 embed_batches=4 summaries=2 summary_errors=1 formats=yaml:2,md:1"
)
def test_prune_removes_files_now_excluded_by_format_filter(db, tmp_path: Path) -> None:
"""Previously-imported junk leaves the index: a file that no longer
matches the A9 extension filter is pruned on the next ``prune=True`` run.
This is how dot-dir READMEs imported before the scope fix get cleaned up."""
root = tmp_path / "cleanup"
root.mkdir()
(root / "keep.md").write_text("# Keep\n\nkept\n")
(root / "junk.md.bak").write_text("old junk that was once imported\n")
llm = FakeEmbedder()
try:
# Seed: import both files as if they were valid at the time.
(root / "junk.md").write_text("old junk that was once imported\n")
(root / "junk.md.bak").unlink()
asyncio.run(import_sources([root], llm, session=db))
# Rename the junk out of the A9 formats, then prune.
(root / "junk.md").rename(root / "junk.md.bak")
summary = asyncio.run(import_sources([root], llm, session=db, prune=True))
assert summary.pruned == 1
assert summary.unchanged == 1 # keep.md survived
assert db.scalar(
select(Document).where(Document.source == root.name, Document.path == "junk.md")
) is None
assert db.scalar(
select(Document).where(Document.source == root.name, Document.path == "keep.md")
) is not None
finally:
_cleanup_source(db, root.name)
# ---------- phase 102: extensionless name-token import ----------
def test_formats_counter_counts_extensionless_name_token(
db, tmp_path: Path, caplog: pytest.LogCaptureFixture
) -> None:
"""Phase 102, D2 — an imported ``Dockerfile`` counts under
``dockerfile`` in ``summary.formats`` and the PLAN §9 line, never
``unknown``."""
root = tmp_path / "extless"
root.mkdir()
(root / "Dockerfile").write_text("FROM alpine\n\nCMD [\"/bin/sh\"]\n")
(root / "notes.md").write_text("# Notes\n\nbody\n")
llm = _embedder_with_extensions("md,dockerfile")
try:
with caplog.at_level(logging.INFO, logger="app.importer"):
summary = asyncio.run(import_sources([root], llm, session=db))
assert summary.files == 2 and summary.added == 2
assert summary.formats == {"dockerfile": 1, "md": 1}
line = next(
r.getMessage()
for r in caplog.records
if "import: summary files=" in r.getMessage()
)
assert line.endswith("formats=dockerfile:1,md:1")
assert "unknown" not in line
finally:
_cleanup_source(db, root.name)
# ---------- phase 64 (task 01): optional per-file progress hook ----------
def test_progress_hook_reports_every_file_in_order_across_roots(
db, tmp_path: Path
) -> None:
"""Multi-root, multi-file: the hook receives the exact
``(source, rel, done, total)`` sequence — roots in *sources* order,
``rel`` the same POSIX path the doc rows use, ``done`` the 1-based
index across **all** sources, ``total`` the combined count."""
root_a = tmp_path / "Alpha"
root_b = tmp_path / "Beta"
root_a.mkdir()
(root_b / "sub").mkdir(parents=True)
(root_a / "a1.md").write_text("# A1\n\na one\n")
(root_a / "a2.md").write_text("# A2\n\na two\n")
(root_a / "a1.md").write_text("# A1\n\na one\n")
(root_b / "sub" / "b1.md").write_text("# B1\n\nb one\n")
events: list[tuple[str, str, int, int]] = []
def progress(source: str, rel: str, done: int, total: int) -> None:
events.append((source, rel, done, total))
try:
summary = asyncio.run(
import_sources([root_a, root_b], FakeEmbedder(), session=db, progress=progress)
)
assert summary.files == 3 and summary.added == 3
assert events == [
("Alpha", "a1.md", 1, 3),
("Alpha", "a2.md", 2, 3),
("Beta", "sub/b1.md", 3, 3), # POSIX rel, sorted within the root
]
finally:
_cleanup_source(db, "Alpha")
_cleanup_source(db, "Beta")
def test_no_progress_means_no_prewalk(
db, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
"""``progress=None`` callers pay no extra pass: the real walker is hit
exactly once per source root (one pass — as before phase 64), proven
with a counting sentinel; with the hook it is hit twice (pre-walk for
``total`` + the processing pass)."""
root = tmp_path / "nowalk"
root.mkdir()
(root / "a.md").write_text("# A\n\na\n")
real_walker = importer.iter_importable_files
walk_calls = 0
def counting(
r: Path,
extensions: frozenset[str],
excluded: frozenset[str] = EXCLUDED_DIRS,
ignore: tuple[str, ...] = (),
include_hidden: bool = False,
) -> list[Path]:
# Phase 89: the walker gained the ``ignore`` keyword; phase 105:
# the ``include_hidden`` flag — the sentinel accepts (and forwards)
# both to stay a drop-in.
nonlocal walk_calls
walk_calls += 1
return real_walker(r, extensions, excluded, ignore, include_hidden)
monkeypatch.setattr(importer, "iter_importable_files", counting)
try:
summary = asyncio.run(import_sources([root], FakeEmbedder(), session=db))
assert summary.files == 1
assert walk_calls == 1 # exactly one pass — the pre-change behaviour
walk_calls = 0
events: list[tuple[str, str, int, int]] = []
summary2 = asyncio.run(
import_sources(
[root], FakeEmbedder(), session=db,
progress=lambda s, r, d, t: events.append((s, r, d, t)),
)
)
assert summary2.files == 1
assert walk_calls == 2 # pre-walk (total) + processing pass
assert events == [(root.name, "a.md", 1, 1)]
finally:
_cleanup_source(db, root.name)
def test_progress_hook_counts_unchanged_and_error_files(db, tmp_path: Path) -> None:
"""The hook fires *before* ``_index_file``: a file whose embedding
fails (and one that re-imports as unchanged) is still reported as the
current file — the sequence covers every importable file."""
root = tmp_path / "progress-mixed"
root.mkdir()
(root / "bad.md").write_text("# Bad\n\npoison content the endpoint refuses\n")
(root / "good.md").write_text("# Good\n\nperfectly fine content\n")
expected = [(root.name, "bad.md", 1, 2), (root.name, "good.md", 2, 2)]
try:
events: list[tuple[str, str, int, int]] = []
first = asyncio.run(
import_sources(
[root], _PoisonEmbedder(), session=db,
progress=lambda s, r, d, t: events.append((s, r, d, t)),
)
)
# bad.md was already reported (done=1) when its embed raised —
# no file silently disappears from the sequence.
assert events == expected
assert first.errors == 1 and first.added == 1
# Re-run: good.md is now unchanged, bad.md is retried and fails
# again — both still count in the sequence.
events.clear()
second = asyncio.run(
import_sources(
[root], _PoisonEmbedder(), session=db,
progress=lambda s, r, d, t: events.append((s, r, d, t)),
)
)
assert events == expected
assert second.errors == 1 and second.unchanged == 1
finally:
_cleanup_source(db, root.name)
def test_progress_hook_with_limit_keeps_full_total(db, tmp_path: Path) -> None:
"""The debug ``limit`` path is unchanged for the hook: it fires only
for processed files (``done`` never exceeds the limit), while
``total`` stays the FULL pre-walk count — an incomplete walk must not
misreport the denominator."""
root = tmp_path / "progress-limited"
root.mkdir()
for name in ("a.md", "b.md", "c.md"):
(root / name).write_text(f"# {name}\n\nbody {name}\n")
events: list[tuple[str, str, int, int]] = []
try:
summary = asyncio.run(
import_sources(
[root], FakeEmbedder(), limit=2, session=db,
progress=lambda s, r, d, t: events.append((s, r, d, t)),
)
)
assert summary.files == 2
assert events == [(root.name, "a.md", 1, 3), (root.name, "b.md", 2, 3)]
finally:
_cleanup_source(db, root.name)