"""rule_embeddings — rules become findable by meaning (milestone 307 step 4, decision note 3026) Revision ID: 0089 Revises: 0088 Create Date: 2026-08-26 Rules were the only major record type with no vector, so `search` could never return one and a rule could only ever arrive by being preloaded. That single fact is what made every rule compete for the same always-on budget. A sibling table rather than a generalisation of note_embeddings: the row could have been made polymorphic, but the SEARCH could not — semantic_search_notes is Note-specific scoping end to end, and a rule shares none of it. See the model docstring for the full reasoning. The vectors are DERIVED data. Nothing is backfilled here: the startup backfill regenerates them, which is also how a chunker-version bump is handled. """ import sqlalchemy as sa from alembic import op revision = "0089" down_revision = "0088" branch_labels = None depends_on = None # Matches note_embeddings — bge-small-en-v1.5, 384-dim unit-normalized. _EMBEDDING_DIM = 384 def upgrade() -> None: op.create_table( "rule_embeddings", sa.Column("rule_id", sa.BigInteger(), sa.ForeignKey("rules.id", ondelete="CASCADE"), primary_key=True), sa.Column("chunk_index", sa.Integer(), primary_key=True), sa.Column("chunk_text", sa.Text(), nullable=False), sa.Column("chunker_version", sa.Integer(), nullable=False), sa.Column("updated_at", sa.DateTime(timezone=True), nullable=False, server_default=sa.text("now()")), ) # The vector column is added by raw DDL for the same reason 0067 did it: # the type comes from the pgvector extension, not from SQLAlchemy's # type system. op.execute(f"ALTER TABLE rule_embeddings ADD COLUMN embedding vector({_EMBEDDING_DIM}) NOT NULL") # HNSW for cosine distance — matches Vector.cosine_distance (`<=>`), so the # search is an indexed ORDER BY ... LIMIT k rather than a full scan. op.execute( """ CREATE INDEX ix_rule_embeddings_embedding_hnsw ON rule_embeddings USING hnsw (embedding vector_cosine_ops) """ ) def downgrade() -> None: op.execute("DROP INDEX IF EXISTS ix_rule_embeddings_embedding_hnsw") op.drop_table("rule_embeddings")