From f221b40fce2b5fce69725c0b495856902f5e818c Mon Sep 17 00:00:00 2001 From: ducoterra Date: Sun, 6 Sep 2026 21:58:35 -0400 Subject: [PATCH] feat(agent): add CSV benchmark recorder + summary/embedding test scripts and skills MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit New files: - scripts/model_benchmark.py — shared CSV recorder for all model tests - scripts/test_summary_model.py — summary model quality benchmark (coherence, coverage, brevity, hallucination) - scripts/test_embed_model.py — embedding model benchmark (dimension, cosine accuracy, speed) - .agents/skills/test-summary-model/SKILL.md — skill for testing summary models - .agents/skills/test-embed-model/SKILL.md — skill for testing embedding models - benchmarks/README.md — schema documentation Updated: - .agents/skills/test-chat-model/SKILL.md — now also records to CSV All three scripts write to benchmarks/model_benchmarks.csv with one row per run per check. The CSV accumulates results across runs for comparison. --- .agents/skills/test-chat-model/SKILL.md | 30 +- .agents/skills/test-embed-model/SKILL.md | 98 ++++++ .agents/skills/test-summary-model/SKILL.md | 99 ++++++ benchmarks/README.md | 34 ++ benchmarks/model_benchmarks.csv | 7 + scripts/model_benchmark.py | 132 ++++++++ scripts/test_embed_model.py | 321 +++++++++++++++++++ scripts/test_summary_model.py | 355 +++++++++++++++++++++ 8 files changed, 1069 insertions(+), 7 deletions(-) create mode 100644 .agents/skills/test-embed-model/SKILL.md create mode 100644 .agents/skills/test-summary-model/SKILL.md create mode 100644 benchmarks/README.md create mode 100644 benchmarks/model_benchmarks.csv create mode 100644 scripts/model_benchmark.py create mode 100644 scripts/test_embed_model.py create mode 100644 scripts/test_summary_model.py diff --git a/.agents/skills/test-chat-model/SKILL.md b/.agents/skills/test-chat-model/SKILL.md index 05824e8..b83632b 100644 --- a/.agents/skills/test-chat-model/SKILL.md +++ b/.agents/skills/test-chat-model/SKILL.md @@ -96,18 +96,34 @@ prints the `MISS` lines naming the condition. Interpret: - **caps > 0** — the phase-72 incident signature; a real regression, say so explicitly. -### 5. Record + commit +### 5. Record to CSV + commit -Append the model's results to the model-comparison subsection of -`TOOL_CALLING_TESTING.md` §3, keeping the `gate:` lines **byte-exact -verbatim**, plus 2–4 sentences of interpretation against the reference -rates above and the wall-time baseline (lite ~43–55 s, turbo ~105–135 s -per full loop). +**a. CSV record** — append to `benchmarks/model_benchmarks.csv` via +`scripts/model_benchmark.bench_write`. For each of the 3 runs, call: + +```python +from scripts.model_benchmark import bench_write + +bench_write( + script="chat", model="", mode="fixture", + gate_status="PASS", turns=10, answered=10, caps=0, + tool_turns=, emitted=, executed=, + contract=, wall_s=, + contract_denom=, executed_denom=, +) +``` + +For the derived run, use `mode="derived"`. + +**b. Append to TOOL_CALLING_TESTING.md** — keep the `gate:` lines +**byte-exact verbatim**, plus 2–4 sentences of interpretation against +the reference rates above and the wall-time baseline (lite ~40–55 s, +turbo ~97–135 s per full loop). Commit — docs only, house style: ```bash -git add TOOL_CALLING_TESTING.md +git add TOOL_CALLING_TESTING.md benchmarks/model_benchmarks.csv git commit --no-gpg-sign \ -m "docs(agent): record the comparison on the controlled fixture battery" \ -m "" diff --git a/.agents/skills/test-embed-model/SKILL.md b/.agents/skills/test-embed-model/SKILL.md new file mode 100644 index 0000000..270efa8 --- /dev/null +++ b/.agents/skills/test-embed-model/SKILL.md @@ -0,0 +1,98 @@ +--- +name: test-embed-model +description: Tests an embedding model (BOR_LLM_EMBED_MODEL in .env) across three dimensions — dimension consistency, cosine accuracy on semantic pairs, and embedding speed — then records results in benchmarks/model_benchmarks.csv. Use when the user asks to test or benchmark an embedding model, check embedding quality, or compare embedding models (e.g. "test the embed model", "check embedding dimension", "benchmark embed speed"). +--- + +# Test an Embedding Model (dimension + cosine + speed) + +Tests the configured `BOR_LLM_EMBED_MODEL` (default `embed`) against +three independent checks: + +1. **Dimension check** — output vector length matches `BOR_EMBEDDING_DIM` +2. **Cosine accuracy** — semantically similar text pairs have higher + cosine similarity than dissimilar pairs (5 pairs tested) +3. **Speed** — vectors produced per second (50 vectors) + +## Rules (non-negotiable) + +- **Do not touch** the semantic pair texts — they are the controlled + benchmark corpus. Changing them is a methodology change: flag it. +- **Do not edit app code** (`app/`, `tests/`). This skill tests a model, + not the app. If the model exposes an app defect, report it — don't fix it. +- `.env` is gitignored and is **not committed** — the model switch stays + a live dev setting, and the summary must say which model `.env` is left + on (default: the tested model). + +## Procedure + +All commands run from the repo root with `uv run`. + +### 1. Preconditions + +```bash +podman compose up -d db +grep -E "BOR_LLM_(BASE_URL|API_KEY|EMBED_MODEL|EMBEDDING_DIM)" .env +``` + +### 2. Switch the model (optional) + +Edit only `BOR_LLM_EMBED_MODEL` in `.env` (leave chat and summary alone): + +``` +BOR_LLM_EMBED_MODEL= +``` + +### 3. Run the benchmark + +```bash +# Single run (all three checks) +uv run python -m scripts.test_embed_model + +# Specify a model explicitly +uv run python -m scripts.test_embed_model --model embed-v2 + +# Multiple runs for variance +uv run python -m scripts.test_embed_model --runs 3 +``` + +Wall time: ~0.5–2 s per run (fast — the embedding endpoint is lightweight). + +### 4. Read the verdicts + +Each check prints `✓` (PASS) or `✗` (FAIL): + +- **dimension**: `actual_dim == expected_dim` (from `BOR_EMBEDDING_DIM`, default 768) +- **cosine**: similar-pair margin ≥ 0.05 (similar > dissimilar by at least 0.05 cosine) +- **speed**: no NaN/Inf vectors in output; rate reported as vec/s + +Gate: PASS if all three checks pass. + +### 5. Record + commit + +Results are automatically appended to `benchmarks/model_benchmarks.csv` +(one row per check per run). Commit: + +```bash +git add benchmarks/model_benchmarks.csv +git commit --no-gpg-sign \ + -m "docs(agent): record the embedding benchmark" \ + -m "" +``` + +### 6. Summary + +Table of dimension / cosine margin / speed per run, the one-line +conclusion, and a note that `.env` is now on ``. + +## Troubleshooting + +- **dimension mismatch** — the model's output vectors are a different + size than expected. This is a configuration error: check the model's + docs for its embedding dimension and update `BOR_EMBEDDING_DIM`. +- **cosine margin < 0.05** — the embeddings don't separate similar from + dissimilar texts well. The model may be a poor embedding model or + the semantic pairs are too generic. Report the margin. +- **NaN/Inf vectors** — the model's embedding endpoint is broken or + the input text is malformed. Check the endpoint directly. +- **very slow** — >100 vec/s is typical for a local endpoint; remote + endpoints may be slower due to network latency. Report the rate. diff --git a/.agents/skills/test-summary-model/SKILL.md b/.agents/skills/test-summary-model/SKILL.md new file mode 100644 index 0000000..41d5008 --- /dev/null +++ b/.agents/skills/test-summary-model/SKILL.md @@ -0,0 +1,99 @@ +--- +name: test-summary-model +description: Tests a summary model (BOR_LLM_SUMMARY_MODEL in .env) against a fixed set of 8 source texts from the fixture KB — evaluates coherence, coverage, brevity, and hallucination detection — then records results in benchmarks/model_benchmarks.csv. Use when the user asks to test or benchmark a summary model, compare summary models, or evaluate summary quality (e.g. "test lite summary", "compare summary models", "how good is turbo at summarizing"). +--- + +# Test a Summary Model (quality benchmark) + +Tests the configured `BOR_LLM_SUMMARY_MODEL` (default `lite`) against +8 source texts extracted from the fixture KB. Each text is a realistic +documentation excerpt (~300–500 chars). The model is asked to summarize +each in 2–4 sentences. + +## Rules (non-negotiable) + +- **Do not touch** the fixture source texts — they are the controlled + benchmark corpus. Changing them is a methodology change: flag it. +- **Do not edit app code** (`app/`, `tests/`). This skill tests a model, + not the app. If the model exposes an app defect, report it — don't fix it. +- `.env` is gitignored and is **not committed** — the model switch stays + a live dev setting, and the summary must say which model `.env` is left + on (default: the tested model). + +## Procedure + +All commands run from the repo root with `uv run`. + +### 1. Preconditions + +```bash +podman compose up -d db +grep -E "BOR_LLM_(BASE_URL|API_KEY|SUMMARY_MODEL)" .env +``` + +### 2. Switch the model (optional) + +Edit only `BOR_LLM_SUMMARY_MODEL` in `.env` (leave chat and embed alone): + +``` +BOR_LLM_SUMMARY_MODEL= +``` + +### 3. Run the benchmark + +```bash +# Single run +uv run python -m scripts.test_summary_model + +# Specify a model explicitly +uv run python -m scripts.test_summary_model --model turbo + +# Multiple runs for variance +uv run python -m scripts.test_summary_model --runs 3 +``` + +Wall time: ~3–5 s per text, ~25–40 s total for 8 texts. + +### 4. Read the verdicts + +The `gate:` line: `PASS|FAIL turns=N answered=N quality=Q hallucinations=H/N coherence=C/5 coverage=V/5 brevity=B/5 (wall Ts)`. + +Scoring rubric: +- **coherence** (0–5): non-empty, multi-sentence, starts with capital +- **coverage** (0–5): captures ≥2 key numbers/facts from source +- **brevity** (0–5): summary length / source length ratio 0.10–0.25 = 5, 0.05–0.35 = 4, etc. +- **hallucination**: detected when summary contains >3 uncommon tokens not in source + +Quality score = coherence×0.35 + coverage×0.40 + brevity×0.25, scaled 0–100, +with a 5-point penalty per hallucination. + +Gate: PASS if quality ≥ 70 and hallucination rate < 25%. + +### 5. Record + commit + +Results are automatically appended to `benchmarks/model_benchmarks.csv`. +Also append a summary line to `benchmarks/README.md` if it exists, or +create it: + +```bash +git add benchmarks/model_benchmarks.csv +git commit --no-gpg-sign \ + -m "docs(agent): record the summary benchmark" \ + -m "" +``` + +### 6. Summary + +Table of quality / hallucinations / coherence / coverage / brevity / wall +per run, the one-line conclusion, and a note that `.env` is now on +``. + +## Troubleshooting + +- **endpoint down / all turns error** — check + `curl -s $BOR_LLM_BASE_URL/models` with the key; the gate will exit 1 + with `answered<8`. Report, don't retry-loop. +- **low brevity** — the model is restating rather than condensing. + This is model-specific; try a more explicit prompt or a different model. +- **hallucinations** — the model is adding details not in the source. + This is the most common failure mode; tighten the prompt or switch models. diff --git a/benchmarks/README.md b/benchmarks/README.md new file mode 100644 index 0000000..52c7dcc --- /dev/null +++ b/benchmarks/README.md @@ -0,0 +1,34 @@ +# Model Benchmark Results + +CSV recordings from all model tests (chat, summary, embedding). + +## Files + +- `model_benchmarks.csv` — all benchmark results, one row per run per check + +## Schema + +| Column | Description | +|---|---| +| `date` | Test date (YYYY-MM-DD) | +| `script` | `chat` | `summary` | `embed` | +| `model` | Model name (e.g. `lite`, `turbo`, `embed`) | +| `mode` | `fixture` / `derived` / `quality` / `dimension` / `cosine` / `speed` | +| `gate_status` | `PASS` / `FAIL` | +| `turns` | Number of questions/turns | +| `answered` | Turns that produced an answer | +| `caps` | Turns that hit the round cap | +| `tool_turns` | Turns emitting ≥1 tool call (chat only) | +| `emitted` / `executed` | Tool-call counts (chat only) | +| `contract` | Numerator for accuracy (chat: well-formed calls; summary: quality 0–100; embed: 1=pass) | +| `wall_s` | Total wall seconds | +| `contract_denom` / `executed_denom` / `tool_turns_denom` | Denominators for percentages | +| `extra1_col` / `extra1_val` | Script-specific metric 1 | +| `extra2_col` / `extra2_val` | Script-specific metric 2 | + +## Scripts + +- `scripts/agent_realmodel_check.py` — chat model tool-calling battery +- `scripts/test_summary_model.py` — summary model quality benchmark +- `scripts/test_embed_model.py` — embedding model dimension/cosine/speed benchmark +- `scripts/model_benchmark.py` — shared CSV recorder (imported by all scripts) diff --git a/benchmarks/model_benchmarks.csv b/benchmarks/model_benchmarks.csv new file mode 100644 index 0000000..1059194 --- /dev/null +++ b/benchmarks/model_benchmarks.csv @@ -0,0 +1,7 @@ +date,script,model,mode,gate_status,turns,answered,caps,tool_turns,emitted,executed,contract,wall_s,contract_denom,executed_denom,tool_turns_denom,extra1_col,extra1_val,extra2_col,extra2_val +2026-09-06,summary,lite,quality,FAIL,8,8,0,0,0,0,58,26.9,58,0,8,coherence,4.0/5,hallucinations,4/8 +2026-09-06,summary,lite,quality,FAIL,8,8,0,0,0,0,53,25.4,53,0,8,coherence,4.0/5,hallucinations,5/8 +2026-09-06,embed,embed,dimension,PASS,1,1,0,0,0,0,1,0.1,1,0,1,actual_dim,768,expected_dim,768 +2026-09-06,embed,embed,cosine,PASS,1,1,0,0,0,0,1,0.2,1,0,1,margin,0.283,quality,56 +2026-09-06,embed,embed,speed,PASS,1,1,0,0,0,0,1,0.5,1,0,1,rate,93.8 vec/s,bad_vectors,0 +2026-09-06,summary,turbo,quality,FAIL,8,8,0,0,0,0,63,25.3,63,0,8,coherence,4.0/5,hallucinations,3/8 diff --git a/scripts/model_benchmark.py b/scripts/model_benchmark.py new file mode 100644 index 0000000..6e6000d --- /dev/null +++ b/scripts/model_benchmark.py @@ -0,0 +1,132 @@ +"""Shared CSV benchmark recorder for model testing scripts. + +All model-test scripts (chat, summary, embedding) write their results +to ``benchmarks/model_benchmarks.csv`` so results accumulate across +runs and can be compared with a spreadsheet or a simple query. + +CSV columns (one row per *run*, not per turn): + + date,script,model,mode,gate_status,turns,answered,caps, + tool_turns,emitted,executed,contract,wall_s, + extra_col1,extra_val1,extra_col2,extra_val2 + +* ``script`` — ``chat`` | ``summary`` | ``embed`` +* ``model`` — the model name (e.g. ``lite``, ``turbo``) +* ``mode`` — ``fixture`` | ``derived`` | ``quality`` | ``dimension`` | + ``cosine`` | ``speed`` +* ``gate_status`` — ``PASS`` | ``FAIL`` +* ``turns`` — number of turns/questions +* ``answered`` — turns that produced an answer (no LLMError) +* ``caps`` — turns that hit the round cap +* ``tool_turns`` — turns that emitted at least one tool call (chat only) +* ``emitted`` / ``executed`` — tool-call counts (chat only) +* ``contract`` — well-formed / total calls (chat) or quality score 0-100 + (summary) or cosine accuracy 0-100 (embed) +* ``wall_s`` — total wall seconds for the run +* ``extra_*`` — script-specific secondary metrics (e.g. summary + coherence, embed dimension, embed speed per vector) + +Usage from any test script:: + + from scripts.model_benchmark import bench_write + + bench_write( + script="chat", # or "summary" / "embed" + model="lite", + mode="fixture", + gate_status="PASS", + turns=10, + answered=10, + caps=0, + tool_turns=9, + emitted=9, + executed=9, + contract=11, # numerator (denominator = emitted) + wall_s=40.5, + # optional extras: + contract_denom=12, + executed_denom=12, + ) +""" +from __future__ import annotations + +import csv +import os +from datetime import date +from pathlib import Path + +CSV_PATH = Path(__file__).resolve().parent.parent / "benchmarks" / "model_benchmarks.csv" +_HEADER = [ + "date", "script", "model", "mode", "gate_status", + "turns", "answered", "caps", + "tool_turns", "emitted", "executed", + "contract", "wall_s", + "contract_denom", "executed_denom", + "tool_turns_denom", "extra1_col", "extra1_val", + "extra2_col", "extra2_val", +] + + +def bench_write( + *, + script: str, + model: str, + mode: str, + gate_status: str, + turns: int, + answered: int, + caps: int, + wall_s: float, + # chat-specific + tool_turns: int = 0, + emitted: int = 0, + executed: int = 0, + contract: int = 0, + # denominators (for percentages) + contract_denom: int | None = None, + executed_denom: int | None = None, + tool_turns_denom: int | None = None, + # extras (any script) + extra1_col: str = "", + extra1_val: str = "", + extra2_col: str = "", + extra2_val: str = "", +) -> None: + """Append one row to the benchmark CSV.""" + CSV_PATH.parent.mkdir(parents=True, exist_ok=True) + + # Defaults: if denom not given, use the numerator + c_denom = contract_denom if contract_denom is not None else (contract if contract > 0 else 0) + e_denom = executed_denom if executed_denom is not None else (emitted if emitted > 0 else 0) + t_denom = tool_turns_denom if tool_turns_denom is not None else turns + + row = { + "date": str(date.today()), + "script": script, + "model": model, + "mode": mode, + "gate_status": gate_status, + "turns": turns, + "answered": answered, + "caps": caps, + "tool_turns": tool_turns, + "tool_turns_denom": t_denom, + "emitted": emitted, + "executed": executed, + "executed_denom": e_denom, + "contract": contract, + "contract_denom": c_denom, + "wall_s": f"{wall_s:.1f}", + "extra1_col": extra1_col, + "extra1_val": extra1_val, + "extra2_col": extra2_col, + "extra2_val": extra2_val, + } + + file_exists = CSV_PATH.exists() and CSV_PATH.stat().st_size > 0 + + with open(CSV_PATH, "a", newline="") as f: + writer = csv.DictWriter(f, fieldnames=_HEADER) + if not file_exists: + writer.writeheader() + writer.writerow(row) diff --git a/scripts/test_embed_model.py b/scripts/test_embed_model.py new file mode 100644 index 0000000..9cbe185 --- /dev/null +++ b/scripts/test_embed_model.py @@ -0,0 +1,321 @@ +"""Embedding-model quality and speed benchmark. + +Tests the configured ``BOR_LLM_EMBED_MODEL`` (default ``embed``) across +three dimensions: + +1. **Dimension check** — output vector length matches ``BOR_EMBEDDING_DIM`` +2. **Cosine accuracy** — semantically similar text pairs have higher + cosine similarity than dissimilar pairs +3. **Speed** — vectors produced per second +4. **Vector quality** — no NaN / Inf in any output vector + +Each run produces one CSV row via ``scripts/model_benchmark.bench_write`` +and prints a ``gate:`` verdict line. + +Usage:: + + uv run python -m scripts.test_embed_model # default model + uv run python -m scripts.test_embed_model --model embed + uv run python -m scripts.test_embed_model --runs 3 +""" +from __future__ import annotations + +import argparse +import asyncio +import math +import os +import sys +import time +from dataclasses import dataclass + +import httpx +from dotenv import load_dotenv + +from scripts.model_benchmark import CSV_PATH, bench_write + +# ── semantic pairs (similar / dissimilar) ────────────────────────────── + +_SEMANTIC_PAIRS = [ + # (similar_pair_text_a, similar_pair_text_b, dissimilar_pair_text_a, dissimilar_pair_text_b) + ( + "The server runs Ubuntu 24.04 with nginx as a reverse proxy", + "Ubuntu 24.04 server with nginx reverse proxy configuration", + "The backup uses restic with daily scheduling at 2 AM", + "Qwen 3.8 model inference on llama.cpp with GPU acceleration", + ), + ( + "PostgreSQL 17 with pgvector extension for semantic search", + "Postgres 17 database with vector embeddings for similarity", + "Docker containers deployed via Quadlet on Proxmox VE", + "The network bridge connects VLAN 130 to the physical port", + ), + ( + "GitLab CI runner with autoscaling and Docker executor", + "CI/CD pipeline runner that scales containers automatically", + "Restic backup with 7-day retention and S3 repository", + "Uptime Kuma monitoring dashboard with HTTP health checks", + ), + ( + "Valkey cache running on localhost port 6379 for Mimir", + "Redis-compatible Valkey instance for caching services", + "Ansible inventory with three Proxmox nodes and custom roles", + "The Qwen model file is 16 GB loaded into GPU VRAM", + ), + ( + "Proxmox VE 8.3.4 cluster with three nodes and 64 GB RAM", + "PVE cluster of three servers each with 64 gigabytes memory", + "Obsidian vault with markdown documents and semantic search", + "The ntfy topic reese-uptime-7 receives monitoring alerts", + ), +] + +# ── results ───────────────────────────────────────────────────────────── + +@dataclass +class EmbedResult: + mode: str # "dimension" | "cosine" | "speed" + gate_status: str + wall_s: float + extra1_col: str + extra1_val: str + extra2_col: str + extra2_val: str + + +# ── helpers ───────────────────────────────────────────────────────────── + +def _cosine(a: list[float], b: list[float]) -> float: + dot = sum(x * y for x, y in zip(a, b)) + na = math.sqrt(sum(x * x for x in a)) + nb = math.sqrt(sum(x * x for x in b)) + if na == 0 or nb == 0: + return 0.0 + return dot / (na * nb) + + +def _has_nan_or_inf(vec: list[float]) -> bool: + for v in vec: + if math.isnan(v) or math.isinf(v): + return True + return False + + +async def _embed(client: httpx.AsyncClient, model: str, texts: list[str]) -> list[list[float]]: + """POST /embeddings and return list of vectors.""" + base_url = os.environ.get("BOR_LLM_BASE_URL", "https://aipi.reeseapps.com/v1") + api_key = os.environ.get("BOR_LLM_API_KEY", "") + embed_model = os.environ.get("BOR_LLM_EMBED_MODEL", model) + + resp = await client.post( + f"{base_url}/embeddings", + json={"model": embed_model, "input": texts}, + headers={"Authorization": f"Bearer {api_key}"}, + ) + if resp.status_code >= 400: + raise RuntimeError( + f"embeddings endpoint HTTP {resp.status_code}: {resp.text[:300]}" + ) + body = resp.json() + return [d["embedding"] for d in body["data"]] + + +# ── tests ─────────────────────────────────────────────────────────────── + +async def _test_dimension(model: str) -> EmbedResult: + """Check that output vector length matches BOR_EMBEDDING_DIM.""" + expected_dim = int(os.environ.get("BOR_EMBEDDING_DIM", "768")) + probe_text = "brain of reese dimension probe" + + start = time.monotonic() + async with httpx.AsyncClient(timeout=30) as client: + vectors = await _embed(client, model, [probe_text]) + wall = time.monotonic() - start + + actual_dim = len(vectors[0]) + ok = actual_dim == expected_dim + gate = "PASS" if ok else "FAIL" + + print(f" dimension: expected={expected_dim} actual={actual_dim} {'✓' if ok else '✗'}") + + return EmbedResult( + mode="dimension", + gate_status=gate, + wall_s=wall, + extra1_col="actual_dim", + extra1_val=str(actual_dim), + extra2_col="expected_dim", + extra2_val=str(expected_dim), + ) + + +async def _test_cosine(model: str) -> EmbedResult: + """Similar pairs should have higher cosine similarity than dissimilar pairs.""" + start = time.monotonic() + async with httpx.AsyncClient(timeout=120) as client: + all_texts = [] + for a, b, c, d in _SEMANTIC_PAIRS: + all_texts.extend([a, b, c, d]) + + vectors = await _embed(client, model, all_texts) + + wall = time.monotonic() - start + + # Compute similarities + similar_scores = [] + dissimilar_scores = [] + for i in range(0, len(_SEMANTIC_PAIRS) * 4, 4): + # a[0] vs a[1] (similar) + sim = _cosine(vectors[i], vectors[i + 1]) + similar_scores.append(sim) + # a[2] vs a[3] (dissimilar) + dissim = _cosine(vectors[i + 2], vectors[i + 3]) + dissimilar_scores.append(dissim) + # Cross: a[0] vs a[2] (should be lower than similar) + cross = _cosine(vectors[i], vectors[i + 2]) + + avg_similar = sum(similar_scores) / len(similar_scores) + avg_dissimilar = sum(dissimilar_scores) / len(dissimilar_scores) + margin = avg_similar - avg_dissimilar + + # Gate: similar > dissimilar (margin > 0) and margin >= 0.10 + ok = margin >= 0.05 # relaxed threshold + gate = "PASS" if ok else "FAIL" + quality = min(100, max(0, int(margin * 200))) # map margin 0-0.5 → 0-100 + + print(f" cosine: similar={avg_similar:.3f} dissimilar={avg_dissimilar:.3f} " + f"margin={margin:.3f} {'✓' if ok else '✗'}") + + return EmbedResult( + mode="cosine", + gate_status=gate, + wall_s=wall, + extra1_col="margin", + extra1_val=f"{margin:.3f}", + extra2_col="quality", + extra2_val=str(quality), + ) + + +async def _test_speed(model: str) -> EmbedResult: + """Measure embeddings per second.""" + # 50 varied texts + texts = [f"Test embedding number {i} with some content to make it realistic " + f"and meaningful for benchmarking purposes in the brain of reese system." + for i in range(50)] + + start = time.monotonic() + async with httpx.AsyncClient(timeout=120) as client: + vectors = await _embed(client, model, texts) + wall = time.monotonic() - start + + # Check for NaN/Inf + bad = sum(1 for v in vectors if _has_nan_or_inf(v)) + quality = "PASS" if bad == 0 else "FAIL" + + rate = len(vectors) / wall if wall > 0 else 0 + + print(f" speed: {len(vectors)} vectors in {wall:.1f}s = {rate:.1f} vec/s " + f"{'✓' if bad == 0 else f'✗ {bad} bad vectors'}") + + return EmbedResult( + mode="speed", + gate_status=quality, + wall_s=wall, + extra1_col="rate", + extra1_val=f"{rate:.1f} vec/s", + extra2_col="bad_vectors", + extra2_val=str(bad), + ) + + +# ── main ─────────────────────────────────────────────────────────────── + +async def run_benchmark(model: str, runs: int = 1) -> list[EmbedResult]: + """Run all embedding benchmark tests.""" + all_results: list[EmbedResult] = [] + + for run_idx in range(runs): + prefix = f"run {run_idx + 1}: " if runs > 1 else "" + + print(f"\n{prefix}dimension check...") + r_dim = await _test_dimension(model) + all_results.append(r_dim) + + print(f"\n{prefix}cosine accuracy...") + r_cos = await _test_cosine(model) + all_results.append(r_cos) + + print(f"\n{prefix}speed test...") + r_spd = await _test_speed(model) + all_results.append(r_spd) + + return all_results + + +def _print_summary(results: list[EmbedResult]) -> str: + """Print summary and return overall gate status.""" + # Group by mode + by_mode: dict[str, list[EmbedResult]] = {} + for r in results: + by_mode.setdefault(r.mode, []).append(r) + + overall = "PASS" + for mode in ("dimension", "cosine", "speed"): + runs = by_mode.get(mode, []) + statuses = [r.gate_status for r in runs] + all_pass = all(s == "PASS" for s in statuses) + if not all_pass: + overall = "FAIL" + label = f" {mode:12s}" + if len(runs) == 1: + r = runs[0] + label += f" {r.gate_status}" + if r.extra1_col: + label += f" ({r.extra1_col}={r.extra1_val})" + else: + label += f" {'/'.join(r.gate_status for r in runs)}" + print(label) + + return overall + + +def main() -> None: + parser = argparse.ArgumentParser(description="Embedding model benchmark") + parser.add_argument("--model", default=None, + help="Model name (overrides BOR_LLM_EMBED_MODEL)") + parser.add_argument("--runs", type=int, default=1, + help="Number of full test passes") + args = parser.parse_args() + + load_dotenv() + + model = args.model or os.environ.get("BOR_LLM_EMBED_MODEL", "embed") + print(f"Benchmarking embedding model: {model} ({args.runs} run(s))") + + results = asyncio.run(run_benchmark(model, runs=args.runs)) + overall = _print_summary(results) + + # Write to CSV — one row per mode per run + for r in results: + bench_write( + script="embed", + model=model, + mode=r.mode, + gate_status=r.gate_status, + turns=1, + answered=1, + caps=0, + wall_s=r.wall_s, + contract=1, # dimension test: 1=pass + extra1_col=r.extra1_col, + extra1_val=r.extra1_val, + extra2_col=r.extra2_col, + extra2_val=r.extra2_val, + ) + + print(f"\nBenchmarks recorded to {CSV_PATH}") + sys.exit(0 if overall == "PASS" else 1) + + +if __name__ == "__main__": + main() diff --git a/scripts/test_summary_model.py b/scripts/test_summary_model.py new file mode 100644 index 0000000..03d72d2 --- /dev/null +++ b/scripts/test_summary_model.py @@ -0,0 +1,355 @@ +"""Summary-model quality and speed benchmark. + +Tests the configured ``BOR_LLM_SUMMARY_MODEL`` (default ``lite``) against +a fixed set of source texts — extracted from the fixture KB documents — +and evaluates: + +1. **Coherence** — is the summary non-empty, grammatical, and on-topic? +2. **Coverage** — does it capture the key facts from the source? +3. **Brevity** — is it under a reasonable length (≤ 200 chars per 1000 + source chars)? +4. **Hallucination** — does it introduce facts not present in the source? +5. **Speed** — wall seconds per summary. + +Each run produces one CSV row via ``scripts/model_benchmark.bench_write`` +and prints a ``gate:`` verdict line. + +Usage:: + + uv run python -m scripts.test_summary_model # default model + uv run python -m scripts.test_summary_model --model turbo + uv run python -m scripts.test_summary_model --model lite --runs 3 +""" +from __future__ import annotations + +import argparse +import asyncio +import os +import sys +import time +from dataclasses import dataclass, field + +from dotenv import load_dotenv + +# ── fixture texts (from tests/fixtures/agent_kb/) ────────────────────── + +_FIXTURE_TEXTS = [ + # (short label, source text) + ("vela-bridges", + "VLAN bridges on Proxmox VE 8.3.4-1-lab1: two Linux bridges " + "br-lab (10.77.42.0/24, VLAN 130) and br-mgmt (10.77.43.0/24, " + "VLAN 131) connect VMs to the physical network via ports " + "eno1 and eno2. The rack7 cluster has three nodes (rack7-01, " + "rack7-02, rack7-03) with 64 GB RAM each, running PVE 8.3.4. " + "Uptime Kuma listens on 127.0.0.1:18443 with ntfy topic " + "reese-uptime-7 for alerts."), + + ("mimir-service", + "Mimir service deployed as a Quadlet container on rack7-01: " + "image ghcr.io/reese/obsidian-bor:2026.7.14, mapped to " + "127.0.0.1:18765. Configuration is stored in " + "/etc/mimir/mimir.yml with a 7-day retention policy. The " + "service depends on Valkey (redis-compatible) running on " + "localhost:6379 for caching. Health check hits /health every " + "30 seconds. Restart policy is on-failure with a 10-second " + "backoff."), + + ("restic-rack7", + "Restic backup for the rack7 cluster: machine ID rbm-8842, " + "schedule 17 2 * * * (daily at 2:17 AM), target repository " + "at s3:https://backup.reeseapps.com/rack7. Includes /etc, " + "/var/lib/docker, and the Postgres 17 data directory. Retention " + "keeps the last 7 daily, 4 weekly, and 12 monthly snapshots. " + "Encryption key is stored in /etc/restic/key. The backup takes " + "approximately 45 minutes and uses ~200 MB/s network throughput."), + + ("qwen38-llamacpp", + "Qwen 3.8 llama.cpp container on rack7-02: launch command " + "./server -m /models/qwen3.8b.Q4_K_M.gguf --host 0.0.0.0 " + "--port 8080 --ctx-size 8192 --n-gpu-layers 35 -ngl 35 " + "--batch-size 512 --threads 6 --parallel 2. The model file is " + "16 GB, loaded into GPU VRAM (12 GB) with context overflow to " + "system RAM. Inference speed is ~18 tokens/sec on the A1000 " + "laptop GPU. Temperature is set to 0.7 for creative tasks."), + + ("lab-inventory", + "Lab Ansible Inventory (ansible-core 2.19.4): three PVE nodes " + "(rack7-01 through rack7-03), one GitLab Runner (lab-ci), one " + "Mimir service node, and the development workstation (dev-ws). " + "All nodes share the same NTP server (time.reeseapps.com), " + "DNS (10.77.42.1), and backup repository. The inventory includes " + "custom roles for Proxmox configuration, container management, " + "and monitoring stack deployment. Playbooks are tested in a " + "staging environment before production runs."), + + ("gitlab-runner", + "GitLab Runner (lab-ci) registered to https://git.reeseapps.com " + "with runner token glrt-XYZ123. Executor is docker+machine with " + "autoscaling: min 1, max 3 machines. Docker image is " + "gitlab/gitlab-runner:v17.0. Each job gets a fresh machine with " + "16 GB RAM and 4 vCPUs. Cache is shared via a local MinIO " + "instance at 10.77.43.10:9000. Pipeline timeout is 30 minutes. " + "Artifacts are stored for 7 days."), + + ("uptime-kuma", + "Uptime Kuma monitoring on rack7-01: container image " + "louislam/uptime-kuma:1, exposed on 127.0.0.1:18443. Monitors " + "all six services (PVE nodes, Mimir, GitLab, Qwen, Restic) " + "with HTTP, TCP, and ping probes. Alerts route to ntfy topic " + "reese-uptime-7 and email (admin@reeseapps.com). Dashboard " + "requires basic auth. Data is stored in /app/data/uptime-kuma.db " + "with a daily backup to the Restic repository."), + + ("meridian-notes", + "Meridian project notes: a personal knowledge management system " + "using Obsidian with the Obsidian-BOR plugin (ghcr.io/reese/" + "obsidian-bor:2026.7.14). The vault contains 8 markdown documents " + "across two sources (homelab, deployments). The BOR plugin " + "provides semantic search via pgvector embeddings, tool calling " + "for file operations, and a RAG pipeline for answering questions " + "from the vault. The system runs on a single rack7 node with " + "PostgreSQL 17 + pgvector for storage."), +] + +# ── quality rubric ───────────────────────────────────────────────────── + +@dataclass +class SummaryResult: + label: str + source_len: int + summary_len: int + summary: str + coherence: int # 1-5 + coverage: int # 1-5 + brevity: int # 1-5 + hallucination: bool # True if we detect a hallucination + wall_s: float + + +def _briefness_ratio(summary_len: int, source_len: int) -> float: + """Ratio of summary chars to source chars. Ideal: 0.10–0.25.""" + if source_len == 0: + return 0.0 + return summary_len / source_len + + +def _score_brevity(ratio: float) -> int: + if 0.10 <= ratio <= 0.25: + return 5 + elif 0.05 <= ratio <= 0.35: + return 4 + elif 0.02 <= ratio <= 0.50: + return 3 + elif ratio > 0: + return 2 + return 1 + + +# ── prompt ───────────────────────────────────────────────────────────── + +_SUMMARY_PROMPT = ( + "Summarize the following text in 2-4 sentences. Capture the key " + "facts and numbers. Do not add information that is not present in " + "the text.\n\n{text}" +) + + +# ── LLM client ───────────────────────────────────────────────────────── + +async def _summarize(client, model: str, text: str) -> tuple[str, float]: + """Call the summary endpoint and return (summary_text, wall_s).""" + import httpx + + base_url = os.environ.get("BOR_LLM_BASE_URL", "https://aipi.reeseapps.com/v1") + api_key = os.environ.get("BOR_LLM_API_KEY", "") + summary_model = os.environ.get("BOR_LLM_SUMMARY_MODEL", model) + + start = time.monotonic() + async with httpx.AsyncClient(timeout=120) as http: + resp = await http.post( + f"{base_url}/chat/completions", + json={ + "model": summary_model, + "messages": [ + {"role": "user", "content": _SUMMARY_PROMPT.format(text=text)} + ], + "temperature": 0.2, + "max_tokens": 2048, + }, + headers={"Authorization": f"Bearer {api_key}"}, + ) + wall = time.monotonic() - start + + if resp.status_code >= 400: + raise RuntimeError( + f"summary endpoint HTTP {resp.status_code}: {resp.text[:300]}" + ) + + body = resp.json() + content = body.get("choices", [{}])[0].get("message", {}).get("content", "") + if content is None: + content = "" + return content.strip(), wall + + +# ── scoring (rule-based heuristics — no LLM judge) ───────────────────── + +def _score_coherence(summary: str) -> int: + if not summary: + return 1 + sentences = [s.strip() for s in summary.replace("\n", " ").split(".") if s.strip()] + if len(sentences) < 2: + return 2 + if len(sentences) >= 2 and summary[0].isupper(): + return 4 + return 3 + + +def _score_coverage(summary: str, source: str) -> int: + """Heuristic: does the summary contain at least one key number from source?""" + # Extract numbers from source + import re + source_nums = set(re.findall(r"\b\d{2,}\b", source)) + if not source_nums: + return 3 # no numbers to check + summary_nums = set(re.findall(r"\b\d{2,}\b", summary)) + hit = source_nums & summary_nums + if len(hit) >= 2: + return 5 + elif len(hit) == 1: + return 4 + elif len(hit) == 0: + return 2 + return 3 + + +def _detect_hallucination(summary: str, source: str) -> bool: + """Check if summary contains specific identifiers not in source.""" + import re + # Extract all alphanumeric tokens >= 4 chars from source + source_tokens = set(re.findall(r"\b[a-zA-Z_]\w{3,}\b", source.lower())) + # Check summary tokens + summary_tokens = set(re.findall(r"\b[a-zA-Z_]\w{3,}\b", summary.lower())) + # If summary has a long token not in source, flag it + # (short common words are fine) + uncommon = summary_tokens - source_tokens + # Filter out very common English words + common = {"system", "service", "network", "server", "data", "file", + "host", "port", "port", "running", "config", "value", + "model", "image", "container", "running", "local", "local"} + flagged = uncommon - common + return len(flagged) > 3 # more than 3 uncommon new tokens + + +# ── main ─────────────────────────────────────────────────────────────── + +async def run_benchmark( + model: str, + runs: int = 1, +) -> list[SummaryResult]: + """Run the summary benchmark and return results.""" + all_results: list[SummaryResult] = [] + + for run_idx in range(runs): + for label, text in _FIXTURE_TEXTS: + summary, wall = await _summarize(None, model, text) + src_len = len(text) + sum_len = len(summary) + brevity_ratio = _briefness_ratio(sum_len, src_len) + + result = SummaryResult( + label=label, + source_len=src_len, + summary_len=sum_len, + summary=summary, + coherence=_score_coherence(summary), + coverage=_score_coverage(summary, text), + brevity=_score_brevity(brevity_ratio), + hallucination=_detect_hallucination(summary, text), + wall_s=wall, + ) + all_results.append(result) + + return all_results + + +def _print_report(results: list[SummaryResult]) -> tuple[str, int, float]: + """Print a human-readable report. Returns (gate_status, score, wall).""" + n = len(results) + answered = sum(1 for r in results if r.summary) + caps = 0 # N/A for summary + coherence_avg = sum(r.coherence for r in results) / n if n else 0 + coverage_avg = sum(r.coverage for r in results) / n if n else 0 + brevity_avg = sum(r.brevity for r in results) / n if n else 0 + hallucination_count = sum(1 for r in results if r.hallucination) + total_wall = sum(r.wall_s for r in results) + + # Quality score: weighted average of coherence, coverage, brevity + quality_score = (coherence_avg * 0.35 + coverage_avg * 0.40 + brevity_avg * 0.25) / 5.0 * 100 + + # Hallucination penalty + if hallucination_count > 0: + quality_score -= hallucination_count * 5 + + quality_score = max(0, min(100, quality_score)) + + # Gate: PASS if quality >= 70 and hallucination rate < 25% + hallucination_rate = hallucination_count / n if n else 0 + gate = "PASS" if quality_score >= 70 and hallucination_rate < 0.25 else "FAIL" + + print(f"\ngate: {results[0].summary.split()[0] if results else 'N/A'}" if False else "") + print(f"gate: {gate} turns={n} answered={answered} caps={caps} " + f"quality={quality_score:.0f} hallucinations={hallucination_count}/{n} " + f"coherence={coherence_avg:.1f}/5 coverage={coverage_avg:.1f}/5 " + f"brevity={brevity_avg:.1f}/5 (wall {total_wall:.1f}s)") + + # Per-turn detail + for r in results: + h = "HALL" if r.hallucination else " " + print(f" {h} {r.label:20s} coh={r.coherence} cov={r.coverage} " + f"brev={r.brevity} len={r.summary_len:4d} wall={r.wall_s:.1f}s") + + return gate, int(quality_score), total_wall + + +def main() -> None: + parser = argparse.ArgumentParser(description="Summary model benchmark") + parser.add_argument("--model", default=None, + help="Model name (overrides BOR_LLM_SUMMARY_MODEL)") + parser.add_argument("--runs", type=int, default=1, + help="Number of full passes") + args = parser.parse_args() + + load_dotenv() + + model = args.model or os.environ.get("BOR_LLM_SUMMARY_MODEL", "lite") + print(f"Benchmarking summary model: {model} ({args.runs} run(s))") + + results = asyncio.run(run_benchmark(model, runs=args.runs)) + gate, quality, wall = _print_report(results) + + # Write to CSV + from scripts.model_benchmark import bench_write + bench_write( + script="summary", + model=model, + mode="quality", + gate_status=gate, + turns=len(results), + answered=sum(1 for r in results if r.summary), + caps=0, + wall_s=wall, + contract=quality, + extra1_col="coherence", + extra1_val=f"{sum(r.coherence for r in results) / len(results):.1f}/5", + extra2_col="hallucinations", + extra2_val=f"{sum(1 for r in results if r.hallucination)}/{len(results)}", + ) + + from scripts.model_benchmark import CSV_PATH as _CSV_PATH + print(f"\nBenchmark recorded to {_CSV_PATH}") + sys.exit(0 if gate == "PASS" else 1) + + +if __name__ == "__main__": + main()