Files
brain-of-reese/scripts/test_embed_model.py
T
ducoterra f221b40fce feat(agent): add CSV benchmark recorder + summary/embedding test scripts and skills
New files:
- scripts/model_benchmark.py — shared CSV recorder for all model tests
- scripts/test_summary_model.py — summary model quality benchmark (coherence, coverage, brevity, hallucination)
- scripts/test_embed_model.py — embedding model benchmark (dimension, cosine accuracy, speed)
- .agents/skills/test-summary-model/SKILL.md — skill for testing summary models
- .agents/skills/test-embed-model/SKILL.md — skill for testing embedding models
- benchmarks/README.md — schema documentation

Updated:
- .agents/skills/test-chat-model/SKILL.md — now also records to CSV

All three scripts write to benchmarks/model_benchmarks.csv with one row
per run per check. The CSV accumulates results across runs for comparison.
2026-09-06 21:58:35 -04:00

322 lines
11 KiB
Python

"""Embedding-model quality and speed benchmark.
Tests the configured ``BOR_LLM_EMBED_MODEL`` (default ``embed``) across
three dimensions:
1. **Dimension check** — output vector length matches ``BOR_EMBEDDING_DIM``
2. **Cosine accuracy** — semantically similar text pairs have higher
cosine similarity than dissimilar pairs
3. **Speed** — vectors produced per second
4. **Vector quality** — no NaN / Inf in any output vector
Each run produces one CSV row via ``scripts/model_benchmark.bench_write``
and prints a ``gate:`` verdict line.
Usage::
uv run python -m scripts.test_embed_model # default model
uv run python -m scripts.test_embed_model --model embed
uv run python -m scripts.test_embed_model --runs 3
"""
from __future__ import annotations
import argparse
import asyncio
import math
import os
import sys
import time
from dataclasses import dataclass
import httpx
from dotenv import load_dotenv
from scripts.model_benchmark import CSV_PATH, bench_write
# ── semantic pairs (similar / dissimilar) ──────────────────────────────
_SEMANTIC_PAIRS = [
# (similar_pair_text_a, similar_pair_text_b, dissimilar_pair_text_a, dissimilar_pair_text_b)
(
"The server runs Ubuntu 24.04 with nginx as a reverse proxy",
"Ubuntu 24.04 server with nginx reverse proxy configuration",
"The backup uses restic with daily scheduling at 2 AM",
"Qwen 3.8 model inference on llama.cpp with GPU acceleration",
),
(
"PostgreSQL 17 with pgvector extension for semantic search",
"Postgres 17 database with vector embeddings for similarity",
"Docker containers deployed via Quadlet on Proxmox VE",
"The network bridge connects VLAN 130 to the physical port",
),
(
"GitLab CI runner with autoscaling and Docker executor",
"CI/CD pipeline runner that scales containers automatically",
"Restic backup with 7-day retention and S3 repository",
"Uptime Kuma monitoring dashboard with HTTP health checks",
),
(
"Valkey cache running on localhost port 6379 for Mimir",
"Redis-compatible Valkey instance for caching services",
"Ansible inventory with three Proxmox nodes and custom roles",
"The Qwen model file is 16 GB loaded into GPU VRAM",
),
(
"Proxmox VE 8.3.4 cluster with three nodes and 64 GB RAM",
"PVE cluster of three servers each with 64 gigabytes memory",
"Obsidian vault with markdown documents and semantic search",
"The ntfy topic reese-uptime-7 receives monitoring alerts",
),
]
# ── results ─────────────────────────────────────────────────────────────
@dataclass
class EmbedResult:
mode: str # "dimension" | "cosine" | "speed"
gate_status: str
wall_s: float
extra1_col: str
extra1_val: str
extra2_col: str
extra2_val: str
# ── helpers ─────────────────────────────────────────────────────────────
def _cosine(a: list[float], b: list[float]) -> float:
dot = sum(x * y for x, y in zip(a, b))
na = math.sqrt(sum(x * x for x in a))
nb = math.sqrt(sum(x * x for x in b))
if na == 0 or nb == 0:
return 0.0
return dot / (na * nb)
def _has_nan_or_inf(vec: list[float]) -> bool:
for v in vec:
if math.isnan(v) or math.isinf(v):
return True
return False
async def _embed(client: httpx.AsyncClient, model: str, texts: list[str]) -> list[list[float]]:
"""POST /embeddings and return list of vectors."""
base_url = os.environ.get("BOR_LLM_BASE_URL", "https://aipi.reeseapps.com/v1")
api_key = os.environ.get("BOR_LLM_API_KEY", "")
embed_model = os.environ.get("BOR_LLM_EMBED_MODEL", model)
resp = await client.post(
f"{base_url}/embeddings",
json={"model": embed_model, "input": texts},
headers={"Authorization": f"Bearer {api_key}"},
)
if resp.status_code >= 400:
raise RuntimeError(
f"embeddings endpoint HTTP {resp.status_code}: {resp.text[:300]}"
)
body = resp.json()
return [d["embedding"] for d in body["data"]]
# ── tests ───────────────────────────────────────────────────────────────
async def _test_dimension(model: str) -> EmbedResult:
"""Check that output vector length matches BOR_EMBEDDING_DIM."""
expected_dim = int(os.environ.get("BOR_EMBEDDING_DIM", "768"))
probe_text = "brain of reese dimension probe"
start = time.monotonic()
async with httpx.AsyncClient(timeout=30) as client:
vectors = await _embed(client, model, [probe_text])
wall = time.monotonic() - start
actual_dim = len(vectors[0])
ok = actual_dim == expected_dim
gate = "PASS" if ok else "FAIL"
print(f" dimension: expected={expected_dim} actual={actual_dim} {'✓' if ok else '✗'}")
return EmbedResult(
mode="dimension",
gate_status=gate,
wall_s=wall,
extra1_col="actual_dim",
extra1_val=str(actual_dim),
extra2_col="expected_dim",
extra2_val=str(expected_dim),
)
async def _test_cosine(model: str) -> EmbedResult:
"""Similar pairs should have higher cosine similarity than dissimilar pairs."""
start = time.monotonic()
async with httpx.AsyncClient(timeout=120) as client:
all_texts = []
for a, b, c, d in _SEMANTIC_PAIRS:
all_texts.extend([a, b, c, d])
vectors = await _embed(client, model, all_texts)
wall = time.monotonic() - start
# Compute similarities
similar_scores = []
dissimilar_scores = []
for i in range(0, len(_SEMANTIC_PAIRS) * 4, 4):
# a[0] vs a[1] (similar)
sim = _cosine(vectors[i], vectors[i + 1])
similar_scores.append(sim)
# a[2] vs a[3] (dissimilar)
dissim = _cosine(vectors[i + 2], vectors[i + 3])
dissimilar_scores.append(dissim)
# Cross: a[0] vs a[2] (should be lower than similar)
cross = _cosine(vectors[i], vectors[i + 2])
avg_similar = sum(similar_scores) / len(similar_scores)
avg_dissimilar = sum(dissimilar_scores) / len(dissimilar_scores)
margin = avg_similar - avg_dissimilar
# Gate: similar > dissimilar (margin > 0) and margin >= 0.10
ok = margin >= 0.05 # relaxed threshold
gate = "PASS" if ok else "FAIL"
quality = min(100, max(0, int(margin * 200))) # map margin 0-0.5 → 0-100
print(f" cosine: similar={avg_similar:.3f} dissimilar={avg_dissimilar:.3f} "
f"margin={margin:.3f} {'✓' if ok else '✗'}")
return EmbedResult(
mode="cosine",
gate_status=gate,
wall_s=wall,
extra1_col="margin",
extra1_val=f"{margin:.3f}",
extra2_col="quality",
extra2_val=str(quality),
)
async def _test_speed(model: str) -> EmbedResult:
"""Measure embeddings per second."""
# 50 varied texts
texts = [f"Test embedding number {i} with some content to make it realistic "
f"and meaningful for benchmarking purposes in the brain of reese system."
for i in range(50)]
start = time.monotonic()
async with httpx.AsyncClient(timeout=120) as client:
vectors = await _embed(client, model, texts)
wall = time.monotonic() - start
# Check for NaN/Inf
bad = sum(1 for v in vectors if _has_nan_or_inf(v))
quality = "PASS" if bad == 0 else "FAIL"
rate = len(vectors) / wall if wall > 0 else 0
print(f" speed: {len(vectors)} vectors in {wall:.1f}s = {rate:.1f} vec/s "
f"{'✓' if bad == 0 else f'✗ {bad} bad vectors'}")
return EmbedResult(
mode="speed",
gate_status=quality,
wall_s=wall,
extra1_col="rate",
extra1_val=f"{rate:.1f} vec/s",
extra2_col="bad_vectors",
extra2_val=str(bad),
)
# ── main ───────────────────────────────────────────────────────────────
async def run_benchmark(model: str, runs: int = 1) -> list[EmbedResult]:
"""Run all embedding benchmark tests."""
all_results: list[EmbedResult] = []
for run_idx in range(runs):
prefix = f"run {run_idx + 1}: " if runs > 1 else ""
print(f"\n{prefix}dimension check...")
r_dim = await _test_dimension(model)
all_results.append(r_dim)
print(f"\n{prefix}cosine accuracy...")
r_cos = await _test_cosine(model)
all_results.append(r_cos)
print(f"\n{prefix}speed test...")
r_spd = await _test_speed(model)
all_results.append(r_spd)
return all_results
def _print_summary(results: list[EmbedResult]) -> str:
"""Print summary and return overall gate status."""
# Group by mode
by_mode: dict[str, list[EmbedResult]] = {}
for r in results:
by_mode.setdefault(r.mode, []).append(r)
overall = "PASS"
for mode in ("dimension", "cosine", "speed"):
runs = by_mode.get(mode, [])
statuses = [r.gate_status for r in runs]
all_pass = all(s == "PASS" for s in statuses)
if not all_pass:
overall = "FAIL"
label = f" {mode:12s}"
if len(runs) == 1:
r = runs[0]
label += f" {r.gate_status}"
if r.extra1_col:
label += f" ({r.extra1_col}={r.extra1_val})"
else:
label += f" {'/'.join(r.gate_status for r in runs)}"
print(label)
return overall
def main() -> None:
parser = argparse.ArgumentParser(description="Embedding model benchmark")
parser.add_argument("--model", default=None,
help="Model name (overrides BOR_LLM_EMBED_MODEL)")
parser.add_argument("--runs", type=int, default=1,
help="Number of full test passes")
args = parser.parse_args()
load_dotenv()
model = args.model or os.environ.get("BOR_LLM_EMBED_MODEL", "embed")
print(f"Benchmarking embedding model: {model} ({args.runs} run(s))")
results = asyncio.run(run_benchmark(model, runs=args.runs))
overall = _print_summary(results)
# Write to CSV — one row per mode per run
for r in results:
bench_write(
script="embed",
model=model,
mode=r.mode,
gate_status=r.gate_status,
turns=1,
answered=1,
caps=0,
wall_s=r.wall_s,
contract=1, # dimension test: 1=pass
extra1_col=r.extra1_col,
extra1_val=r.extra1_val,
extra2_col=r.extra2_col,
extra2_val=r.extra2_val,
)
print(f"\nBenchmarks recorded to {CSV_PATH}")
sys.exit(0 if overall == "PASS" else 1)
if __name__ == "__main__":
main()