feat(embeddings): every vector records the model whose space it lives in (#4132)
CI & Build / Python lint (push) Successful in 3s
CI & Build / Plugin hooks (push) Successful in 14s
CI & Build / TypeScript typecheck (push) Successful in 54s
CI & Build / integration (push) Successful in 1m1s
CI & Build / Python tests (push) Successful in 1m40s
CI & Build / Build & push image (push) Canceled after 27s

The four embedding tables stamped chunker_version but not the model, and
vector(384) is a width, not an identity: a same-width model swap would
write a second geometry beside the first with no error.

- embedding_model on note/rule/milestone/system embeddings (0109; existing
  rows stamped with the only model any install has ever run).
- Every write stamps EMBEDDING_MODEL; every backfill's "current" test is
  is_current_stamp(), both halves of calibration_stamp().
- migrate_floor refuses while any row its surface searches is off the live
  model, before sampling: re-embed, then migrate.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-23 19:02:30 -04:00
co-authored by Claude Opus 5.5
parent 11b286d786
commit f4e9cd429b
10 changed files with 178 additions and 10 deletions
+37 -5
View File
@@ -18,7 +18,7 @@ from collections.abc import Sequence
from typing import TYPE_CHECKING
from sqlalchemy import delete, func, or_, select
from sqlalchemy import and_, delete, func, or_, select
from scribe.models import async_session
from scribe.models.embedding import NoteEmbedding, RuleEmbedding
@@ -367,6 +367,34 @@ def calibration_stamp() -> dict:
"""
return {"embedding_model": EMBEDDING_MODEL, "shape_version": CHUNKER_VERSION}
def is_current_stamp(table):
"""The rows written by THIS chunker in THIS model's space (#4132).
Every embedding table stores both halves of `calibration_stamp()` per row,
and a row is current only when both match. One predicate for all four
tables, because a backfill that checked the version alone is exactly how a
same-width model swap would have gone unnoticed.
"""
return and_(
table.chunker_version == CHUNKER_VERSION,
table.embedding_model == EMBEDDING_MODEL,
)
async def rows_off_the_live_model(table) -> int:
"""How many rows of an embedding table were NOT written in the live space.
Nonzero means the corpus is part-way through a model change: a search over
it compares vectors from two geometries, and a statistic computed from it
(`retrieval_migration.migrate_floor`) is a blend of both.
"""
async with async_session() as session:
return int((await session.execute(
select(func.count()).select_from(table)
.where(table.embedding_model != EMBEDDING_MODEL)
)).scalar_one())
# Character budget approximating the model window. Tokens-per-char varies by
# content — ~4 chars/token for prose, closer to 3 for code and tables — so 1400
# chars sits at roughly 350-470 tokens, leaving headroom for the title prefixed
@@ -633,6 +661,7 @@ async def upsert_note_embedding(
embedding=vector,
chunk_text=chunk,
chunker_version=CHUNKER_VERSION,
embedding_model=EMBEDDING_MODEL,
)
)
await session.commit()
@@ -1092,7 +1121,7 @@ async def backfill_note_embeddings() -> None:
for row in (
await session.execute(
select(NoteEmbedding.note_id).where(
NoteEmbedding.chunker_version == CHUNKER_VERSION
is_current_stamp(NoteEmbedding)
)
)
).fetchall()
@@ -1294,6 +1323,7 @@ async def upsert_rule_embedding(
embedding=vector,
chunk_text=chunk,
chunker_version=CHUNKER_VERSION,
embedding_model=EMBEDDING_MODEL,
)
)
await session.commit()
@@ -1464,7 +1494,7 @@ async def backfill_rule_embeddings() -> None:
try:
async with async_session() as session:
current = select(RuleEmbedding.rule_id).where(
RuleEmbedding.chunker_version == CHUNKER_VERSION
is_current_stamp(RuleEmbedding)
)
# IDS ONLY — the text is re-read per rule below (#4262).
by_version = {
@@ -1563,6 +1593,7 @@ async def upsert_milestone_embedding(
session.add(MilestoneEmbedding(
milestone_id=milestone_id, chunk_index=index, embedding=vector,
chunk_text=chunk, chunker_version=CHUNKER_VERSION,
embedding_model=EMBEDDING_MODEL,
))
await session.commit()
except Exception:
@@ -1725,6 +1756,7 @@ async def upsert_system_embedding(
session.add(SystemEmbedding(
system_id=system_id, chunk_index=index, embedding=vector,
chunk_text=chunk, chunker_version=CHUNKER_VERSION,
embedding_model=EMBEDDING_MODEL,
))
await session.commit()
except Exception:
@@ -1841,7 +1873,7 @@ async def backfill_system_embeddings() -> None:
try:
async with async_session() as session:
current = select(SystemEmbedding.system_id).where(
SystemEmbedding.chunker_version == CHUNKER_VERSION
is_current_stamp(SystemEmbedding)
)
# IDS ONLY — the charter is re-read per System below (#4262).
by_version = {
@@ -1884,7 +1916,7 @@ async def backfill_milestone_embeddings() -> None:
try:
async with async_session() as session:
current = select(MilestoneEmbedding.milestone_id).where(
MilestoneEmbedding.chunker_version == CHUNKER_VERSION
is_current_stamp(MilestoneEmbedding)
)
# IDS ONLY — the plan is re-read per milestone below (#4262).
by_version = {