85 lines
3.4 KiB
Python
85 lines
3.4 KiB
Python
"""SQLAlchemy models (PostgreSQL 17 + pgvector).
|
|
|
|
Data model — see ``.agent/PLAN.md`` §Data Model:
|
|
|
|
* ``documents`` — one row per imported A9 file (full content, path, sha256 hash).
|
|
* ``chunks`` — retrieval units; each chunk points at its parent document
|
|
via ``document_id``. This is how an embedding maps back to
|
|
a document path (the "feed the whole document" requirement).
|
|
* ``query_log`` — observability: every question, its retrieval score, the
|
|
deflection decision, and latency.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import uuid
|
|
from datetime import datetime
|
|
|
|
from pgvector.sqlalchemy import Vector
|
|
from sqlalchemy import (
|
|
Boolean,
|
|
DateTime,
|
|
Float,
|
|
ForeignKey,
|
|
Integer,
|
|
String,
|
|
Text,
|
|
UniqueConstraint,
|
|
func,
|
|
)
|
|
from sqlalchemy.dialects.postgresql import UUID
|
|
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
|
|
|
from app.config import get_settings
|
|
from app.db import Base
|
|
|
|
# Single source of truth for the vector column size (see .agent/PLAN.md A6).
|
|
EMBEDDING_DIM: int = get_settings().embedding_dim
|
|
|
|
|
|
class Document(Base):
|
|
__tablename__ = "documents"
|
|
__table_args__ = (UniqueConstraint("source", "path", name="uq_documents_source_path"),)
|
|
|
|
id: Mapped[uuid.UUID] = mapped_column(UUID(as_uuid=True), primary_key=True, default=uuid.uuid4)
|
|
source: Mapped[str] = mapped_column(String(120), index=True) # e.g. "Homelab"
|
|
path: Mapped[str] = mapped_column(String(1000), index=True) # relative to source dir
|
|
full_path: Mapped[str] = mapped_column(String(2000)) # absolute path at import time
|
|
title: Mapped[str] = mapped_column(String(500))
|
|
content: Mapped[str] = mapped_column(Text) # full markdown — the RAG context
|
|
content_hash: Mapped[str] = mapped_column(String(64), index=True) # sha256 for change detection
|
|
indexed_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), server_default=func.now())
|
|
|
|
chunks: Mapped[list[Chunk]] = relationship(
|
|
back_populates="document", cascade="all, delete-orphan"
|
|
)
|
|
|
|
|
|
class Chunk(Base):
|
|
__tablename__ = "chunks"
|
|
|
|
id: Mapped[uuid.UUID] = mapped_column(UUID(as_uuid=True), primary_key=True, default=uuid.uuid4)
|
|
document_id: Mapped[uuid.UUID] = mapped_column(
|
|
UUID(as_uuid=True), ForeignKey("documents.id", ondelete="CASCADE"), index=True
|
|
)
|
|
position: Mapped[int] = mapped_column(Integer)
|
|
content: Mapped[str] = mapped_column(Text)
|
|
embedding: Mapped[list[float] | None] = mapped_column(Vector(EMBEDDING_DIM))
|
|
|
|
document: Mapped[Document] = relationship(back_populates="chunks")
|
|
|
|
|
|
class QueryLog(Base):
|
|
__tablename__ = "query_log"
|
|
|
|
id: Mapped[uuid.UUID] = mapped_column(UUID(as_uuid=True), primary_key=True, default=uuid.uuid4)
|
|
question: Mapped[str] = mapped_column(Text)
|
|
top_score: Mapped[float] = mapped_column(Float, default=0.0) # best cosine similarity
|
|
#: Lexical (FTS) candidates matched — the OR-tsquery hit count (A8). NULL
|
|
#: for pre-hybrid rows (migration 0002).
|
|
fts_hits: Mapped[int | None] = mapped_column(Integer)
|
|
chunk_hits: Mapped[int] = mapped_column(Integer, default=0)
|
|
deflected: Mapped[bool] = mapped_column(Boolean, default=False) # True = honest "no idea"
|
|
sources: Mapped[str] = mapped_column(Text, default="") # comma-joined source paths
|
|
latency_ms: Mapped[int] = mapped_column(Integer, default=0)
|
|
created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), server_default=func.now())
|