complete restructure

This commit is contained in:
2026-08-19 13:16:43 -04:00
parent d7b6f28cbd
commit f87e1d51ef
60 changed files with 8176 additions and 5143 deletions
+126
View File
@@ -0,0 +1,126 @@
"""Async OpenAI-compatible LLM, image, and embedding clients.
Public API facade: chat completion and the tool registry live in the
``vibe_bot.llm`` subpackage; the embedding HTTP plumbing stays in this
module.
``image_edit`` (``!retcon``) requests a fixed 768x768 output rather than
matching the source image's aspect ratio. Matching it would require
decoding the downloaded image (Pillow is not a dependency) and most
OpenAI-compatible edit endpoints only accept a fixed set of sizes anyway;
the square output bounds request cost and is universally honored.
"""
from __future__ import annotations
import json
import logging
from typing import Any
import requests
from vibe_bot.llm.chat import (
ToolCallNotifier,
ToolExecutor,
chat_complete,
chat_completion_instruct,
chat_completion_with_history,
chat_completion_with_tools,
get_chat_client,
)
from vibe_bot.llm.images import (
get_image_edit_client,
get_image_gen_client,
image_edit,
image_generation,
)
from vibe_bot.llm.registry import ToolRegistry, get_tool_registry
__all__ = [
"ToolCallNotifier",
"ToolExecutor",
"ToolRegistry",
"chat_complete",
"chat_completion_instruct",
"chat_completion_with_history",
"chat_completion_with_tools",
"embedding",
"get_chat_client",
"get_embedding_session",
"get_image_edit_client",
"get_image_gen_client",
"get_tool_registry",
"image_edit",
"image_generation",
]
logger = logging.getLogger(__name__)
_embedding_session: requests.Session | None = None
def get_embedding_session() -> requests.Session:
"""Return the shared requests session for embedding HTTP calls."""
global _embedding_session
if _embedding_session is None:
_embedding_session = requests.Session()
return _embedding_session
def embedding(
text: str,
*,
url: str,
api_key: str,
model: str,
) -> list[float]:
"""Generate an embedding vector for the given text (synchronous).
Uses a raw HTTP request (shared session) to avoid the SDK injecting
unsupported parameters like encoding_format.
"""
endpoint = f"{url.rstrip('/')}/embeddings"
headers = {
"Authorization": f"Bearer {api_key}",
"Content-Type": "application/json",
}
payload = {"model": model, "input": [text]}
try:
resp = get_embedding_session().post(
endpoint, headers=headers, json=payload, timeout=30
)
resp.raise_for_status()
# A 2xx body can still be non-JSON (e.g. an HTML error page);
# resp.json() would raise JSONDecodeError (a ValueError).
data = resp.json()
except (requests.RequestException, ValueError):
return []
# Handle both OpenAI-style response ({"data": [...]}) and
# Ollama-style response ([{...}]) where the API returns a list directly
if isinstance(data, list):
first = data[0]
if not isinstance(first, dict):
return []
raw: Any = first.get("embedding")
elif isinstance(data, dict):
if not data.get("data"):
return []
raw = data["data"][0].get("embedding")
else:
return []
if raw is None:
return []
if isinstance(raw, str):
try:
raw = json.loads(raw)
except ValueError:
return []
if not isinstance(raw, list):
raw = list(raw)
if not raw:
return []
return list[float](raw)