feat(embeddings): best-chunk-per-note on every retrieval surface (#280 step 4)
CI & Build / Plugin hooks (push) Failing after 1s
CI & Build / Python lint (push) Failing after 3s
CI & Build / integration (push) Successful in 17s
CI & Build / TypeScript typecheck (push) Successful in 32s
CI & Build / Python tests (push) Successful in 46s
CI & Build / Build & push image (push) Skipped

A note's relevance is now its best chunk's similarity, everywhere:

- semantic_search_notes keeps the indexed raw-distance top-k and over-fetches
  chunk rows (x4, composing with the x3 supersession over-fetch), then
  collapses to first-appearance-per-note — rows arrive distance-ordered, so
  first is best. Every ranked consumer (MCP/REST search, Browse, auto-inject,
  write-path, gate) inherits through the one function.
- list_notes semantic q swaps its join for a correlated MIN-distance
  subquery — the join would have repeated a long note once per matching chunk
  and made total count chunks.
- the duplicate report groups its self-join by note pair on MIN(distance):
  pair similarity = closest chunk pair, and the < join now also drops
  cross-chunk self-pairs that would flag every long note against itself.
- the write gate queries once per chunk of the candidate (capped at 8), so a
  note duplicating an existing record in ONE SECTION is caught — the
  whole-document query diluted exactly the section that mattered.

Integration test now seeds a two-chunk note and pins the collapse against
real pgvector.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01UaYUaouG9jjhATyuxCKrQs
This commit is contained in:
2026-08-08 23:51:01 -04:00
co-authored by Claude Fable 5
parent 0e70a3896b
commit 041d8defbc
6 changed files with 164 additions and 51 deletions
+57 -39
View File
@@ -69,6 +69,12 @@ _SEMANTIC_THRESHOLD = 0.90
# structural signals cannot see.
_SNIPPET_SEMANTIC_THRESHOLD = 0.96
# The gate queries per CHUNK of the candidate (#280) — this caps how many
# searches one save may cost. Eight chunks ≈ five thousand words of candidate;
# a duplicate hiding past that is the duplicate report's job to find, not a
# reason to stall the write path.
_GATE_MAX_CHUNKS = 8
@dataclass
class DuplicateMatch:
@@ -258,41 +264,45 @@ async def find_duplicate_note(
# --- Signal 3: semantic similarity (only with a substantial body) ---
if body and len(body.strip()) >= _MIN_BODY_FOR_SEMANTIC:
# Built by the SAME function the corpus was embedded with. This one is
# the copy that mattered most and was easiest to miss: it is a QUERY
# document, compared against embedded ones. Shaped differently from the
# corpus it searches, the gate degrades silently — it still returns
# neighbours, just less apt ones, and no signal says the query and the
# index stopped agreeing (found by the guard in test_embedding_text).
query = embeddings_svc.embedding_text(title, body)
# Scope the semantic check the same way as the title check: a record in
# project P compares only to P; a project-less (orphan) record compares
# only to other orphans (orphan_only), NOT across every project — without
# this, semantic_search_notes applies no project filter when project_id
# is None and would match an orphan note against any project's notes.
hits = await embeddings_svc.semantic_search_notes(
user_id, query, project_id=project_id, is_task=is_task,
orphan_only=(project_id is None),
limit=3,
threshold=(_SNIPPET_SEMANTIC_THRESHOLD
if note_type == SNIPPET_NOTE_TYPE else _SEMANTIC_THRESHOLD),
# Owner-only, deliberately: this gate BLOCKS a create and tells the
# caller to update the match instead. Matching someone else's record
# would refuse their write and point them at something they may not
# be able to edit.
scope="own",
# NOT demoted by supersession (#278). A superseded record is still a
# duplicate of what you are about to write — the claim is that it is
# no longer CURRENT, not that it is gone. Demoting it here would let
# the same note be recorded a second time, and the second copy would
# be the one nothing warns about.
demote_superseded=False,
)
for score, note in hits:
# semantic_search_notes doesn't filter note_type — enforce it here so
# a note doesn't shadow a task of the same wording, etc.
if note.note_type == note_type:
return DuplicateMatch(note.id, note.title, round(score, 3), "semantic")
# Query with the SAME chunker the corpus was embedded with (#280). This
# was the copy that mattered most and was easiest to miss: these are
# QUERY documents, compared against embedded ones — shaped differently
# from the corpus, the gate degrades silently. Chunking also makes the
# gate see what the whole-document query diluted: a long candidate that
# duplicates an existing record IN ONE SECTION now matches on that
# section. Capped so one pathological paste can't turn a save into
# dozens of searches — a duplicate past the cap is the duplicate
# report's job, not the gate's.
for query in embeddings_svc.chunk_document(title, body)[:_GATE_MAX_CHUNKS]:
# Scope the semantic check the same way as the title check: a record
# in project P compares only to P; a project-less (orphan) record
# compares only to other orphans (orphan_only), NOT across every
# project — without this, semantic_search_notes applies no project
# filter when project_id is None and would match an orphan note
# against any project's notes.
hits = await embeddings_svc.semantic_search_notes(
user_id, query, project_id=project_id, is_task=is_task,
orphan_only=(project_id is None),
limit=3,
threshold=(_SNIPPET_SEMANTIC_THRESHOLD
if note_type == SNIPPET_NOTE_TYPE else _SEMANTIC_THRESHOLD),
# Owner-only, deliberately: this gate BLOCKS a create and tells
# the caller to update the match instead. Matching someone
# else's record would refuse their write and point them at
# something they may not be able to edit.
scope="own",
# NOT demoted by supersession (#278). A superseded record is
# still a duplicate of what you are about to write — the claim
# is that it is no longer CURRENT, not that it is gone. Demoting
# it here would let the same note be recorded a second time, and
# the second copy would be the one nothing warns about.
demote_superseded=False,
)
for score, note in hits:
# semantic_search_notes doesn't filter note_type — enforce it
# here so a note doesn't shadow a task of the same wording, etc.
if note.note_type == note_type:
return DuplicateMatch(note.id, note.title, round(score, 3), "semantic")
return None
@@ -502,15 +512,22 @@ async def find_duplicate_records(
left_note = aliased(Note, name="left_note")
right_note = aliased(Note, name="right_note")
distance = left.embedding.cosine_distance(right.embedding)
# Chunk grain (#280): a note-pair's similarity is its closest CHUNK pair —
# two records duplicate each other where their most similar sections do,
# which is the honest definition when one section of a long note restates
# another record. GROUP BY collapses the chunk cross-product to one row
# per note pair.
best = func.min(distance)
pairs: list[tuple[int, int, float]] = []
try:
async with async_session() as session:
stmt = (
select(left.note_id, right.note_id, distance.label("distance"))
select(left.note_id, right.note_id, best.label("distance"))
.select_from(left)
# `<` not `!=`: each unordered pair exactly once, and it drops
# the self-pair (distance 0) that would otherwise dominate.
# the self-pairs (including cross-chunk self-pairs, which would
# otherwise flag every multi-chunk note against itself).
.join(right, left.note_id < right.note_id)
.join(left_note, left_note.id == left.note_id)
.join(right_note, right_note.id == right.note_id)
@@ -523,9 +540,10 @@ async def find_duplicate_records(
# report is bounded by what merge can actually act on.
left_note.user_id == user_id,
right_note.user_id == user_id,
distance <= max_distance,
)
.order_by(distance.asc())
.group_by(left.note_id, right.note_id)
.having(best <= max_distance)
.order_by(best.asc())
.limit(max(1, limit))
)
rows = list((await session.execute(stmt)).all())