Files
brain-of-reese/app/models.py
T

102 lines
4.2 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""SQLAlchemy models (PostgreSQL 17 + pgvector).
Data model — see ``.agent/PLAN.md`` §Data Model:
* ``documents`` — one row per imported A9 file (full content, path, sha256 hash).
* ``chunks`` — retrieval units; each chunk points at its parent document
via ``document_id``. This is how an embedding maps back to
a document path (the "feed the whole document" requirement).
* ``query_log`` — observability: every question, its retrieval score,
the deflection decision, and latency.
* ``steering_notes`` — owner tuning notes injected into the system prompt
of every chat turn (phase 15, ``<tuning>`` section).
"""
from __future__ import annotations
import uuid
from datetime import datetime
from pgvector.sqlalchemy import Vector
from sqlalchemy import (
Boolean,
DateTime,
Float,
ForeignKey,
Integer,
String,
Text,
UniqueConstraint,
func,
)
from sqlalchemy.dialects.postgresql import UUID
from sqlalchemy.orm import Mapped, mapped_column, relationship
from app.config import get_settings
from app.db import Base
# Single source of truth for the vector column size (see .agent/PLAN.md A6).
EMBEDDING_DIM: int = get_settings().embedding_dim
class Document(Base):
__tablename__ = "documents"
__table_args__ = (UniqueConstraint("source", "path", name="uq_documents_source_path"),)
id: Mapped[uuid.UUID] = mapped_column(UUID(as_uuid=True), primary_key=True, default=uuid.uuid4)
source: Mapped[str] = mapped_column(String(120), index=True) # e.g. "Homelab"
path: Mapped[str] = mapped_column(String(1000), index=True) # relative to source dir
full_path: Mapped[str] = mapped_column(String(2000)) # absolute path at import time
title: Mapped[str] = mapped_column(String(500))
content: Mapped[str] = mapped_column(Text) # full markdown — the RAG context
content_hash: Mapped[str] = mapped_column(String(64), index=True) # sha256 for change detection
indexed_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), server_default=func.now())
chunks: Mapped[list[Chunk]] = relationship(
back_populates="document", cascade="all, delete-orphan"
)
class Chunk(Base):
__tablename__ = "chunks"
id: Mapped[uuid.UUID] = mapped_column(UUID(as_uuid=True), primary_key=True, default=uuid.uuid4)
document_id: Mapped[uuid.UUID] = mapped_column(
UUID(as_uuid=True), ForeignKey("documents.id", ondelete="CASCADE"), index=True
)
position: Mapped[int] = mapped_column(Integer)
content: Mapped[str] = mapped_column(Text)
embedding: Mapped[list[float] | None] = mapped_column(Vector(EMBEDDING_DIM))
document: Mapped[Document] = relationship(back_populates="chunks")
class QueryLog(Base):
__tablename__ = "query_log"
id: Mapped[uuid.UUID] = mapped_column(UUID(as_uuid=True), primary_key=True, default=uuid.uuid4)
question: Mapped[str] = mapped_column(Text)
top_score: Mapped[float] = mapped_column(Float, default=0.0) # best cosine similarity
#: Lexical (FTS) candidates matched — the OR-tsquery hit count (A8). NULL
#: for pre-hybrid rows (migration 0002).
fts_hits: Mapped[int | None] = mapped_column(Integer)
chunk_hits: Mapped[int] = mapped_column(Integer, default=0)
deflected: Mapped[bool] = mapped_column(Boolean, default=False) # True = honest "no idea"
sources: Mapped[str] = mapped_column(Text, default="") # comma-joined source paths
latency_ms: Mapped[int] = mapped_column(Integer, default=0)
created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), server_default=func.now())
class SteeringNote(Base):
"""One owner tuning instruction (phase 15).
Notes are read into the system prompt of **every** chat turn as the
``<tuning>`` section (oldest first, char-budgeted — see
:func:`app.rag.prompts.build_steering_section`).
"""
__tablename__ = "steering_notes"
id: Mapped[uuid.UUID] = mapped_column(UUID(as_uuid=True), primary_key=True, default=uuid.uuid4)
note: Mapped[str] = mapped_column(Text) # trimmed, 1–2000 chars (API-enforced)
created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), server_default=func.now())