"""Probe the self-hosted LLM endpoint (aipi). Lists available models and verifies the embedding dimension of the configured ``embed`` model against ``BOR_EMBEDDING_DIM`` (default 768). With ``--tools`` it additionally probes the configured chat model's OpenAI-style tool-calling support (phase 37, task 01): a trivial no-parameter ``get_time`` function is offered in a non-streaming and a streaming ``chat/completions`` request, and a supported / not-supported verdict is printed for each — the agent loop (``app/rag/agent.py``) is built against that verdict. Run this before the first import if the LLM backend ever changes: uv run python -m scripts.llm_probe uv run python -m scripts.llm_probe --tools """ from __future__ import annotations import argparse import json import os import sys from collections.abc import Iterable from datetime import date import httpx from dotenv import load_dotenv #: Trivial probe function (phase 37, task 01): no parameters, so a #: compliant tool call carries arguments of exactly ``{}``. _PROBE_TOOL: dict = { "type": "function", "function": { "name": "get_time", "description": "Get the current time.", "parameters": {"type": "object", "properties": {}}, }, } #: A question for which calling ``get_time`` is the natural move. _PROBE_MESSAGES: list[dict] = [{"role": "user", "content": "What time is it right now?"}] def parse_tool_response_nonstreaming(payload: dict | None) -> dict: """Extract tool-call facts from one non-streaming chat.completions body. Returns ``{"finish_reason": str | None, "calls": [(name, arguments)]}`` — empty/None when the reply carries no tool calls (a plain content answer) or is malformed (including a non-dict body). """ choices = (payload or {}).get("choices") or [] if not isinstance(choices, list) or not choices: return {"finish_reason": None, "calls": []} first = choices[0] message = first.get("message") or {} calls: list[tuple[str, str]] = [] for tc in message.get("tool_calls") or []: fn = tc.get("function") or {} calls.append((str(fn.get("name") or ""), str(fn.get("arguments") or ""))) return {"finish_reason": first.get("finish_reason"), "calls": calls} def parse_tool_response_streaming(lines: Iterable[str]) -> dict: """Accumulate ``delta.tool_calls`` fragments across SSE ``data:`` lines. Wire convention (OpenAI): the first fragment of a call carries ``index`` + ``id`` + ``function.name`` and (possibly partial) ``function.arguments``; later fragments carry ``index`` + further ``arguments`` pieces; the final chunk carries ``finish_reason``. Parsing stops at ``data: [DONE]``; malformed ``data:`` lines are skipped (aipi sometimes interleaves keep-alive noise). Returns:: { "finish_reason": str | None, "calls": [(name, arguments)], # accumulated per index "delta_chunks": int, # chunks carrying tool_calls "indexed": bool, # every such chunk had int "index" "had_id": bool, # some chunk carried the call "id" "arguments_in_deltas": bool, # some fragment carried arguments } """ finish_reason: str | None = None by_index: dict[int, dict[str, str]] = {} delta_chunks = 0 indexed = True had_id = False arguments_in_deltas = False for raw in lines: line = raw.strip() if not line.startswith("data:"): continue data = line[5:].strip() if data == "[DONE]": break try: payload = json.loads(data) except json.JSONDecodeError: continue choices = payload.get("choices") or [] if not isinstance(choices, list): continue for choice in choices: fr = choice.get("finish_reason") if fr: finish_reason = fr delta = choice.get("delta") or {} for tc in delta.get("tool_calls") or []: idx = tc.get("index") if isinstance(idx, int): key = idx else: indexed = False key = 0 if not by_index else max(by_index) + 1 slot = by_index.setdefault(key, {"name": "", "arguments": ""}) delta_chunks += 1 if tc.get("id"): had_id = True fn = tc.get("function") or {} if fn.get("name"): slot["name"] += str(fn["name"]) if fn.get("arguments"): arguments_in_deltas = True slot["arguments"] += str(fn["arguments"]) calls = [(slot["name"], slot["arguments"]) for _, slot in sorted(by_index.items())] return { "finish_reason": finish_reason, "calls": calls, "delta_chunks": delta_chunks, "indexed": indexed, "had_id": had_id, "arguments_in_deltas": arguments_in_deltas, } def _called(result: dict, expected: str) -> bool: """True when the request finished with tool_calls invoking *expected*.""" return ( result["finish_reason"] == "tool_calls" and any(name == expected for name, _ in result["calls"]) ) def classify_tool_calling(nonstream: dict, stream: dict, expected: str = "get_time") -> str: """Phase-37 verdict: ``"supported"`` or ``"not-supported"``. Supported requires *both* requests to finish with ``finish_reason="tool_calls"`` calling *expected*, and the streaming request to deliver the calls as indexed ``delta.tool_calls`` chunks with a call ``id`` (the OpenAI wire convention). Anything less — including an intermittent split outcome — is ``"not-supported"`` (fail-loud house style); the phase then uses the documented prompt-based structured-call fallback. """ if ( _called(nonstream, expected) and _called(stream, expected) and stream["delta_chunks"] > 0 and stream["indexed"] and stream["had_id"] ): return "supported" return "not-supported" def _no_stream_result() -> dict: return { "finish_reason": None, "calls": [], "delta_chunks": 0, "indexed": False, "had_id": False, "arguments_in_deltas": False, } def probe_tools(client: httpx.Client, chat_model: str) -> int: """Run the ``--tools`` probe (phase 37, task 01) and print the verdicts. Returns 0 when both requests were made and classified (either verdict is a valid, recorded outcome) and 1 when the endpoint is unreachable (the probe is inconclusive, not a "not supported" signal). """ base_payload: dict = { "model": chat_model, "messages": _PROBE_MESSAGES, "tools": [_PROBE_TOOL], } # (a) non-streaming request ns_raw: dict | None = None try: resp = client.post("/chat/completions", json={**base_payload, "stream": False}) resp.raise_for_status() ns_raw = resp.json() nonstream = parse_tool_response_nonstreaming(ns_raw) except httpx.HTTPStatusError as e: print(f"[probe] tools(non-stream) HTTP {e.response.status_code}: {e.response.text[:200]}") nonstream = {"finish_reason": None, "calls": []} except httpx.TransportError as e: print(f"[probe] tools(non-stream) transport error: {e}") return 1 ns_ok = _called(nonstream, "get_time") print(f"[probe] tools(non-stream) verdict : {'supported' if ns_ok else 'not supported'}") print( f"[probe] tools(non-stream) finish_reason={nonstream['finish_reason']!r} " f"calls={nonstream['calls']!r}" ) if not ns_ok and ns_raw: content = ((ns_raw.get("choices") or [{}])[0].get("message") or {}).get("content") or "" if content: print(f"[probe] tools(non-stream) answered in content instead: {content[:160]!r}") # (b) streaming request lines: list[str] = [] try: with client.stream( "POST", "/chat/completions", json={**base_payload, "stream": True} ) as resp: resp.raise_for_status() for line in resp.iter_lines(): lines.append(line) if line.startswith("data:") and line[5:].strip() == "[DONE]": break stream = parse_tool_response_streaming(lines) except httpx.HTTPStatusError as e: print(f"[probe] tools(stream) HTTP {e.response.status_code}: {e.response.text[:200]}") stream = _no_stream_result() except httpx.TransportError as e: print(f"[probe] tools(stream) transport error: {e}") return 1 st_ok = ( _called(stream, "get_time") and stream["delta_chunks"] > 0 and stream["indexed"] and stream["had_id"] ) print(f"[probe] tools(stream) verdict : {'supported' if st_ok else 'not supported'}") print( f"[probe] tools(stream) finish_reason={stream['finish_reason']!r} " f"calls={stream['calls']!r}" ) print( f"[probe] tools(stream) delta.tool_calls: chunks={stream['delta_chunks']} " f"indexed={stream['indexed']} had_id={stream['had_id']} " f"arguments_in_deltas={stream['arguments_in_deltas']}" ) verdict = classify_tool_calling(nonstream, stream) today = date.today().isoformat() detail = ( "non-streaming + streaming tool_calls follow the OpenAI wire convention" if verdict == "supported" else "phase falls back to the documented prompt-based structured-call path" ) print(f"[probe] TOOLS VERDICT ({today}) : {verdict} — {detail}") print(f"[probe] summary : probe: {chat_model} tool_calls={verdict} {today}") return 0 def main(argv: list[str] | None = None) -> int: # CLI-only: pick up .env without side effects on import (the parse/ # classify functions above are imported from unit tests, where the # ambient environment must stay pristine). load_dotenv() parser = argparse.ArgumentParser( description=( "Probe the aipi endpoint: models, embedding dimension, and " "(with --tools) the chat model's tool-calling support." ) ) parser.add_argument( "--tools", action="store_true", help=( "also probe the chat model's OpenAI-style tools/tool_calls support " "(non-streaming + streaming) and print a supported / not-supported verdict" ), ) args = parser.parse_args(argv) base_url = os.environ.get("BOR_LLM_BASE_URL", "https://aipi.reeseapps.com/v1").rstrip("/") api_key = ( os.environ.get("BOR_LLM_API_KEY") or os.environ.get("AIPI_KEY") or "not-needed" ) embed_model = os.environ.get("BOR_LLM_EMBED_MODEL", "embed") chat_model = os.environ.get("BOR_LLM_CHAT_MODEL", "turbo") expected_dim = int(os.environ.get("BOR_EMBEDDING_DIM", "768")) headers = {"Authorization": f"Bearer {api_key}"} tools_rc = 0 with httpx.Client(base_url=base_url, headers=headers, timeout=60.0) as client: r = client.get("/models") r.raise_for_status() models = [m["id"] for m in r.json()["data"]] print(f"[probe] base_url : {base_url}") print(f"[probe] models : {', '.join(models)}") for needed in (chat_model, embed_model): if needed not in models: print(f"[probe] ERROR: required model '{needed}' not available") return 1 r = client.post( "/embeddings", json={"model": embed_model, "input": "brain of reese dimension probe"}, ) r.raise_for_status() dims = sorted({len(d["embedding"]) for d in r.json()["data"]}) print(f"[probe] dims({embed_model}): {dims}") if args.tools: tools_rc = probe_tools(client, chat_model) if dims != [expected_dim]: print( f"[probe] MISMATCH: expected {expected_dim}, got {dims}. " "Update BOR_EMBEDDING_DIM and recreate the chunks table (see README)." ) return 1 print("[probe] OK — models present, embedding dimension matches configuration.") return tools_rc if __name__ == "__main__": sys.exit(main())