feat(embeddings): best-chunk-per-note on every retrieval surface (#280 step 4)
CI & Build / Plugin hooks (push) Failing after 1s
CI & Build / Python lint (push) Failing after 3s
CI & Build / integration (push) Successful in 17s
CI & Build / TypeScript typecheck (push) Successful in 32s
CI & Build / Python tests (push) Successful in 46s
CI & Build / Build & push image (push) Skipped
CI & Build / Plugin hooks (push) Failing after 1s
CI & Build / Python lint (push) Failing after 3s
CI & Build / integration (push) Successful in 17s
CI & Build / TypeScript typecheck (push) Successful in 32s
CI & Build / Python tests (push) Successful in 46s
CI & Build / Build & push image (push) Skipped
A note's relevance is now its best chunk's similarity, everywhere: - semantic_search_notes keeps the indexed raw-distance top-k and over-fetches chunk rows (x4, composing with the x3 supersession over-fetch), then collapses to first-appearance-per-note — rows arrive distance-ordered, so first is best. Every ranked consumer (MCP/REST search, Browse, auto-inject, write-path, gate) inherits through the one function. - list_notes semantic q swaps its join for a correlated MIN-distance subquery — the join would have repeated a long note once per matching chunk and made total count chunks. - the duplicate report groups its self-join by note pair on MIN(distance): pair similarity = closest chunk pair, and the < join now also drops cross-chunk self-pairs that would flag every long note against itself. - the write gate queries once per chunk of the candidate (capped at 8), so a note duplicating an existing record in ONE SECTION is caught — the whole-document query diluted exactly the section that mattered. Integration test now seeds a two-chunk note and pins the collapse against real pgvector. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UaYUaouG9jjhATyuxCKrQs
This commit is contained in:
@@ -136,6 +136,36 @@ def test_a_monster_single_paragraph_is_hard_split_not_dropped():
|
||||
assert total_words == 2000
|
||||
|
||||
|
||||
# --- the read path: best chunk wins (#280 step 4) ----------------------------
|
||||
|
||||
|
||||
async def test_search_collapses_chunk_rows_to_best_chunk_per_note():
|
||||
"""Rows arrive at CHUNK grain ordered by distance; a note appearing via
|
||||
several chunks must come back ONCE, scored by its best chunk — otherwise a
|
||||
long record fills the top-k with copies of itself."""
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
from scribe.services import embeddings as emb
|
||||
|
||||
note_a, note_b = MagicMock(id=1), MagicMock(id=2)
|
||||
rows = [(note_a, 0.10), (note_b, 0.20), (note_a, 0.25), (note_a, 0.30)]
|
||||
result = MagicMock()
|
||||
result.all.return_value = rows
|
||||
session, ctx = _session_ctx()
|
||||
session.execute = AsyncMock(return_value=result)
|
||||
|
||||
with (
|
||||
patch.object(emb, "async_session", return_value=ctx),
|
||||
patch.object(emb, "get_embedding", AsyncMock(return_value=[0.0] * 384)),
|
||||
):
|
||||
out = await emb.semantic_search_notes(
|
||||
1, "a query", limit=8, demote_superseded=False
|
||||
)
|
||||
|
||||
assert [note.id for _s, note in out] == [1, 2]
|
||||
assert out[0][0] == 1.0 - 0.10 # the BEST chunk's score, not a later one
|
||||
|
||||
|
||||
# --- the write path: one row per chunk (#280 step 3) -------------------------
|
||||
|
||||
|
||||
|
||||
@@ -68,7 +68,11 @@ async def seeded():
|
||||
await s.flush()
|
||||
# query vector will be [1,0,0,...]; near ~ identical (sim≈1.0),
|
||||
# far is orthogonal (sim≈0.0 -> filtered by the default threshold).
|
||||
# near gets a SECOND, weaker chunk (sim≈0.6) — the collapse to
|
||||
# best-chunk-per-note (#280) is under test: near must come back once,
|
||||
# at its best chunk's score, not twice.
|
||||
s.add(_emb(near.id, user.id, 0, _vec(1.0)))
|
||||
s.add(_emb(near.id, user.id, 1, _vec(0.6, 0.8)))
|
||||
s.add(_emb(far.id, user.id, 0, _vec(0.0, 1.0)))
|
||||
await s.commit()
|
||||
ids = (user.id, near.id, far.id)
|
||||
@@ -96,6 +100,9 @@ async def test_semantic_search_ranks_and_thresholds_via_pgvector(seeded):
|
||||
assert near_id in ids
|
||||
assert far_id not in ids
|
||||
assert ids[0] == near_id
|
||||
# Chunk collapse (#280): near has TWO chunk rows above the floor (sim≈1.0
|
||||
# and ≈0.6) and must appear exactly once, at its best chunk's score.
|
||||
assert ids.count(near_id) == 1
|
||||
top_score = results[0][0]
|
||||
assert top_score == pytest.approx(1.0, abs=1e-3)
|
||||
|
||||
|
||||
@@ -68,6 +68,33 @@ async def test_semantic_match_when_body_substantial():
|
||||
assert dup.similarity == 0.93
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_gate_catches_a_duplicate_hiding_in_a_later_chunk():
|
||||
"""The capability #280 adds to the gate: a long candidate that duplicates
|
||||
an existing record in ONE SECTION is caught, where the whole-document
|
||||
query this replaces diluted exactly the section that mattered. The gate
|
||||
queries once per chunk and any chunk's hit blocks."""
|
||||
para = ("This section restates an existing decision in enough words to be "
|
||||
"a real paragraph of content for the chunker to keep. ") * 4
|
||||
body = "\n\n".join(f"## Topic {i}\n\n{para} (t{i})" for i in range(8))
|
||||
|
||||
from scribe.services.embeddings import chunk_document
|
||||
n_chunks = len(chunk_document("Title", body))
|
||||
assert n_chunks > 1, "test body must actually chunk"
|
||||
|
||||
hit = _fake_note(id=30, title="The existing decision", note_type="note")
|
||||
# Every chunk misses except the LAST one the gate will ask about.
|
||||
sem = AsyncMock(side_effect=[[] for _ in range(n_chunks - 1)] + [[(0.94, hit)]])
|
||||
with patch("scribe.services.dedup.async_session",
|
||||
return_value=_session_returning(None)), \
|
||||
patch("scribe.services.dedup.embeddings_svc.semantic_search_notes", sem):
|
||||
dup = await find_duplicate_note(
|
||||
7, "Title", body=body, project_id=2, is_task=False, note_type="note",
|
||||
)
|
||||
assert dup is not None and dup.id == 30
|
||||
assert sem.await_count == n_chunks
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_semantic_match_of_other_note_type_is_ignored():
|
||||
other = _fake_note(id=21, title="X", note_type="process")
|
||||
|
||||
Reference in New Issue
Block a user