Files
brain-of-reese/scripts/test_embed_model.py
T
ducoterra 7fce6572d0
Build and Push Containers / build-and-push-app (push) Successful in 1m45s
Build and Push Containers / build-and-push-db (push) Successful in 13s
feat: phases 77–80 — navbar view refresh, static background, API tokens, history suggestion chips
Single consolidated commit for four completed, validated phases (77, 78,
79, 80). The pipeline run left all work uncommitted because the harness
commits only with PHASE_COMMIT=1 while child executors are forbidden from
committing; the phases themselves all passed validation and moved to
.agents/phases/complete/.

Phase 77 — navbar view refresh
- router.js dispatches bor:view-refresh on re-show / active re-click /
  popstate (gated on wasMounted; first show and boot exempt)
- History / RAG / Sources / Tuning re-fetch on refresh (admin branch);
  Chat deliberately excluded (stream survival)
- History "Refresh" button (admin-only, in-flight disable + status line)
- New story suite tests/e2e/test_navbar_refresh.py (7 tests)

Phase 78 — static background
- Removed the animated glow layers; static 44px grid over the flat --bg
  canvas; default and reduced-motion renders byte-identical
- Updated background/theme E2E suites; removed bg-glow test pins

Phase 79 — API tokens
- api_tokens model + migration 0012; hash-only token service
- Admin tokens API + Tokens admin view; POST /api/token-auth;
  live-revoking require_user on chat / suggestions / document content
- Frontend token gate with localStorage cache; anonymous E2E suites
  migrated to token login
- New story suite tests/e2e/test_api_tokens.py (9 tests)

Phase 80 — history suggestion chips
- last_questions() endpoint with SEED fallback; startNewChat() refetch
- Seed-semantics docs (config.py, .env.example, README)
- Integration state matrix + E2E suite rewritten to the 4 chip states

Also included: phase-76 report artifacts and the repo restore-test-db
skill (previously untracked), scripts/* ruff fixes from phase 77.

Final gate state (phase 80 final pass, covers everything above):
- uv run pytest --cov=app → 1637 passed, 0 failed, app/ coverage 99%
- uv run ruff check . && uv run pyright → clean, 0 errors
- Per-phase story E2E suites green in isolation
2026-09-07 12:39:01 -04:00

317 lines
11 KiB
Python

"""Embedding-model quality and speed benchmark.
Tests the configured ``BOR_LLM_EMBED_MODEL`` (default ``embed``) across
three dimensions:
1. **Dimension check** — output vector length matches ``BOR_EMBEDDING_DIM``
2. **Cosine accuracy** — semantically similar text pairs have higher
cosine similarity than dissimilar pairs
3. **Speed** — vectors produced per second
4. **Vector quality** — no NaN / Inf in any output vector
Each run produces one CSV row via ``scripts/model_benchmark.bench_write``
and prints a ``gate:`` verdict line.
Usage::
uv run python -m scripts.test_embed_model # default model
uv run python -m scripts.test_embed_model --model embed
uv run python -m scripts.test_embed_model --runs 3
"""
from __future__ import annotations
import argparse
import asyncio
import math
import os
import sys
import time
from dataclasses import dataclass
import httpx
from dotenv import load_dotenv
from scripts.model_benchmark import CSV_PATH, bench_write
# ── semantic pairs (similar / dissimilar) ──────────────────────────────
_SEMANTIC_PAIRS = [
# (similar_pair_text_a, similar_pair_text_b, dissimilar_pair_text_a, dissimilar_pair_text_b)
(
"The server runs Ubuntu 24.04 with nginx as a reverse proxy",
"Ubuntu 24.04 server with nginx reverse proxy configuration",
"The backup uses restic with daily scheduling at 2 AM",
"Qwen 3.8 model inference on llama.cpp with GPU acceleration",
),
(
"PostgreSQL 17 with pgvector extension for semantic search",
"Postgres 17 database with vector embeddings for similarity",
"Docker containers deployed via Quadlet on Proxmox VE",
"The network bridge connects VLAN 130 to the physical port",
),
(
"GitLab CI runner with autoscaling and Docker executor",
"CI/CD pipeline runner that scales containers automatically",
"Restic backup with 7-day retention and S3 repository",
"Uptime Kuma monitoring dashboard with HTTP health checks",
),
(
"Valkey cache running on localhost port 6379 for Mimir",
"Redis-compatible Valkey instance for caching services",
"Ansible inventory with three Proxmox nodes and custom roles",
"The Qwen model file is 16 GB loaded into GPU VRAM",
),
(
"Proxmox VE 8.3.4 cluster with three nodes and 64 GB RAM",
"PVE cluster of three servers each with 64 gigabytes memory",
"Obsidian vault with markdown documents and semantic search",
"The ntfy topic reese-uptime-7 receives monitoring alerts",
),
]
# ── results ─────────────────────────────────────────────────────────────
@dataclass
class EmbedResult:
mode: str # "dimension" | "cosine" | "speed"
gate_status: str
wall_s: float
extra1_col: str
extra1_val: str
extra2_col: str
extra2_val: str
# ── helpers ─────────────────────────────────────────────────────────────
def _cosine(a: list[float], b: list[float]) -> float:
dot = sum(x * y for x, y in zip(a, b, strict=True))
na = math.sqrt(sum(x * x for x in a))
nb = math.sqrt(sum(x * x for x in b))
if na == 0 or nb == 0:
return 0.0
return dot / (na * nb)
def _has_nan_or_inf(vec: list[float]) -> bool:
return any(math.isnan(v) or math.isinf(v) for v in vec)
async def _embed(client: httpx.AsyncClient, model: str, texts: list[str]) -> list[list[float]]:
"""POST /embeddings and return list of vectors."""
base_url = os.environ.get("BOR_LLM_BASE_URL", "https://aipi.reeseapps.com/v1")
api_key = os.environ.get("BOR_LLM_API_KEY", "")
embed_model = os.environ.get("BOR_LLM_EMBED_MODEL", model)
resp = await client.post(
f"{base_url}/embeddings",
json={"model": embed_model, "input": texts},
headers={"Authorization": f"Bearer {api_key}"},
)
if resp.status_code >= 400:
raise RuntimeError(
f"embeddings endpoint HTTP {resp.status_code}: {resp.text[:300]}"
)
body = resp.json()
return [d["embedding"] for d in body["data"]]
# ── tests ───────────────────────────────────────────────────────────────
async def _test_dimension(model: str) -> EmbedResult:
"""Check that output vector length matches BOR_EMBEDDING_DIM."""
expected_dim = int(os.environ.get("BOR_EMBEDDING_DIM", "768"))
probe_text = "brain of reese dimension probe"
start = time.monotonic()
async with httpx.AsyncClient(timeout=30) as client:
vectors = await _embed(client, model, [probe_text])
wall = time.monotonic() - start
actual_dim = len(vectors[0])
ok = actual_dim == expected_dim
gate = "PASS" if ok else "FAIL"
print(f" dimension: expected={expected_dim} actual={actual_dim} {'✓' if ok else '✗'}")
return EmbedResult(
mode="dimension",
gate_status=gate,
wall_s=wall,
extra1_col="actual_dim",
extra1_val=str(actual_dim),
extra2_col="expected_dim",
extra2_val=str(expected_dim),
)
async def _test_cosine(model: str) -> EmbedResult:
"""Similar pairs should have higher cosine similarity than dissimilar pairs."""
start = time.monotonic()
async with httpx.AsyncClient(timeout=120) as client:
all_texts = []
for a, b, c, d in _SEMANTIC_PAIRS:
all_texts.extend([a, b, c, d])
vectors = await _embed(client, model, all_texts)
wall = time.monotonic() - start
# Compute similarities
similar_scores = []
dissimilar_scores = []
for i in range(0, len(_SEMANTIC_PAIRS) * 4, 4):
# a[0] vs a[1] (similar)
sim = _cosine(vectors[i], vectors[i + 1])
similar_scores.append(sim)
# a[2] vs a[3] (dissimilar)
dissim = _cosine(vectors[i + 2], vectors[i + 3])
dissimilar_scores.append(dissim)
avg_similar = sum(similar_scores) / len(similar_scores)
avg_dissimilar = sum(dissimilar_scores) / len(dissimilar_scores)
margin = avg_similar - avg_dissimilar
# Gate: similar > dissimilar (margin > 0) and margin >= 0.10
ok = margin >= 0.05 # relaxed threshold
gate = "PASS" if ok else "FAIL"
quality = min(100, max(0, int(margin * 200))) # map margin 0-0.5 → 0-100
print(f" cosine: similar={avg_similar:.3f} dissimilar={avg_dissimilar:.3f} "
f"margin={margin:.3f} {'✓' if ok else '✗'}")
return EmbedResult(
mode="cosine",
gate_status=gate,
wall_s=wall,
extra1_col="margin",
extra1_val=f"{margin:.3f}",
extra2_col="quality",
extra2_val=str(quality),
)
async def _test_speed(model: str) -> EmbedResult:
"""Measure embeddings per second."""
# 50 varied texts
texts = [f"Test embedding number {i} with some content to make it realistic "
f"and meaningful for benchmarking purposes in the brain of reese system."
for i in range(50)]
start = time.monotonic()
async with httpx.AsyncClient(timeout=120) as client:
vectors = await _embed(client, model, texts)
wall = time.monotonic() - start
# Check for NaN/Inf
bad = sum(1 for v in vectors if _has_nan_or_inf(v))
quality = "PASS" if bad == 0 else "FAIL"
rate = len(vectors) / wall if wall > 0 else 0
print(f" speed: {len(vectors)} vectors in {wall:.1f}s = {rate:.1f} vec/s "
f"{'✓' if bad == 0 else f'✗ {bad} bad vectors'}")
return EmbedResult(
mode="speed",
gate_status=quality,
wall_s=wall,
extra1_col="rate",
extra1_val=f"{rate:.1f} vec/s",
extra2_col="bad_vectors",
extra2_val=str(bad),
)
# ── main ───────────────────────────────────────────────────────────────
async def run_benchmark(model: str, runs: int = 1) -> list[EmbedResult]:
"""Run all embedding benchmark tests."""
all_results: list[EmbedResult] = []
for run_idx in range(runs):
prefix = f"run {run_idx + 1}: " if runs > 1 else ""
print(f"\n{prefix}dimension check...")
r_dim = await _test_dimension(model)
all_results.append(r_dim)
print(f"\n{prefix}cosine accuracy...")
r_cos = await _test_cosine(model)
all_results.append(r_cos)
print(f"\n{prefix}speed test...")
r_spd = await _test_speed(model)
all_results.append(r_spd)
return all_results
def _print_summary(results: list[EmbedResult]) -> str:
"""Print summary and return overall gate status."""
# Group by mode
by_mode: dict[str, list[EmbedResult]] = {}
for r in results:
by_mode.setdefault(r.mode, []).append(r)
overall = "PASS"
for mode in ("dimension", "cosine", "speed"):
runs = by_mode.get(mode, [])
statuses = [r.gate_status for r in runs]
all_pass = all(s == "PASS" for s in statuses)
if not all_pass:
overall = "FAIL"
label = f" {mode:12s}"
if len(runs) == 1:
r = runs[0]
label += f" {r.gate_status}"
if r.extra1_col:
label += f" ({r.extra1_col}={r.extra1_val})"
else:
label += f" {'/'.join(r.gate_status for r in runs)}"
print(label)
return overall
def main() -> None:
parser = argparse.ArgumentParser(description="Embedding model benchmark")
parser.add_argument("--model", default=None,
help="Model name (overrides BOR_LLM_EMBED_MODEL)")
parser.add_argument("--runs", type=int, default=1,
help="Number of full test passes")
args = parser.parse_args()
load_dotenv()
model = args.model or os.environ.get("BOR_LLM_EMBED_MODEL", "embed")
print(f"Benchmarking embedding model: {model} ({args.runs} run(s))")
results = asyncio.run(run_benchmark(model, runs=args.runs))
overall = _print_summary(results)
# Write to CSV — one row per mode per run
for r in results:
bench_write(
script="embed",
model=model,
mode=r.mode,
gate_status=r.gate_status,
turns=1,
answered=1,
caps=0,
wall_s=r.wall_s,
contract=1, # dimension test: 1=pass
extra1_col=r.extra1_col,
extra1_val=r.extra1_val,
extra2_col=r.extra2_col,
extra2_val=r.extra2_val,
)
print(f"\nBenchmarks recorded to {CSV_PATH}")
sys.exit(0 if overall == "PASS" else 1)
if __name__ == "__main__":
main()