Files
FabledScribe/tests/test_integration_pgvector_search.py
T
bvandeusenandClaude Fable 5 bbee0d0db1
CI & Build / Python lint (push) Successful in 3s
CI & Build / Plugin hooks (push) Successful in 8s
CI & Build / integration (push) Successful in 24s
CI & Build / TypeScript typecheck (push) Successful in 32s
CI & Build / Python tests (push) Successful in 54s
CI & Build / Build & push image (push) Successful in 18s
refactor(tests): one definition each for the copied fixtures and fakes (#2825, milestone 296 area 1)
The shape ledger showed the same test scaffolding defined over and over:
_bind_user x12 (byte-identical), _dispose_engine x10 in three wordings,
_no_supersession x3, _make_mock_session x7 in three subsets, a get-or-create
User helper x2 (+3 inlined), and fifteen hand-rolled MagicMock note factories
each re-explaining the same "an auto-MagicMock attribute is truthy" hazard
(note 2109).

Now: conftest.py carries _bind_user / _dispose_engine / _no_supersession as
opt-in fixtures (pytestmark = usefixtures(...) per module, so unit tests pay
nothing), and tests/helpers.py carries make_mock_session(), ensure_user() and
fake_note(**attrs) — the hazard documented once, real values on every
attribute the product reads. Call sites were rewritten by AST so titles with
dashes and commas survived; the three SimpleNamespace _note stand-ins that
only feed a single function stay local.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-08-21 11:03:48 -04:00

116 lines
4.4 KiB
Python

"""Real-Postgres integration test for pgvector semantic search.
Runs only in the CI integration lane (real Postgres + `vector` extension +
schema built by `alembic upgrade head`, which includes migration 0067). This
exercises what the unit mocks cannot: the native `vector(384)` column, the
`<=>` cosine-distance operator behind `Vector.cosine_distance`, the HNSW index,
and the distance->similarity recovery in `semantic_search_notes`.
The embedder itself is stubbed (get_embedding is patched) so the test does not
depend on downloading the fastembed model — only the Postgres/pgvector path is
under test.
"""
from unittest.mock import AsyncMock, patch
import pytest
import pytest_asyncio
from sqlalchemy import delete
from scribe.models import async_session
from scribe.models.embedding import EMBEDDING_DIM, NoteEmbedding
from scribe.models.note import Note
from scribe.models.user import User
from scribe.services.embeddings import semantic_search_notes
pytestmark = [pytest.mark.integration, pytest.mark.usefixtures("_dispose_engine")]
def _vec(*nonzero_first):
"""A 384-dim vector with the given leading values, zero-padded."""
v = list(nonzero_first) + [0.0] * (EMBEDDING_DIM - len(nonzero_first))
return v[:EMBEDDING_DIM]
def _emb(note_id, user_id, chunk_index, vec):
"""A chunk row at the current chunker version (#280, migration 0077)."""
from scribe.services.embeddings import CHUNKER_VERSION
return NoteEmbedding(
note_id=note_id,
chunk_index=chunk_index,
user_id=user_id,
embedding=vec,
chunk_text=f"chunk {chunk_index} of note {note_id}",
chunker_version=CHUNKER_VERSION,
)
@pytest_asyncio.fixture
async def seeded():
"""Insert a user + a near and a far note with hand-crafted embeddings.
Returns (user_id, near_note_id, far_note_id). Cleaned up after the test.
"""
async with async_session() as s:
user = User(username="pgvec_itest")
s.add(user)
await s.flush()
near = Note(user_id=user.id, title="near", body="near body")
far = Note(user_id=user.id, title="far", body="far body")
s.add_all([near, far])
await s.flush()
# query vector will be [1,0,0,...]; near ~ identical (sim≈1.0),
# far is orthogonal (sim≈0.0 -> filtered by the default threshold).
# near gets a SECOND, weaker chunk (sim≈0.6) — the collapse to
# best-chunk-per-note (#280) is under test: near must come back once,
# at its best chunk's score, not twice.
s.add(_emb(near.id, user.id, 0, _vec(1.0)))
s.add(_emb(near.id, user.id, 1, _vec(0.6, 0.8)))
s.add(_emb(far.id, user.id, 0, _vec(0.0, 1.0)))
await s.commit()
ids = (user.id, near.id, far.id)
yield ids
user_id = ids[0]
async with async_session() as s:
await s.execute(delete(NoteEmbedding).where(NoteEmbedding.user_id == user_id))
await s.execute(delete(Note).where(Note.user_id == user_id))
await s.execute(delete(User).where(User.id == user_id))
await s.commit()
@pytest.mark.asyncio
async def test_semantic_search_ranks_and_thresholds_via_pgvector(seeded):
user_id, near_id, far_id = seeded
with patch(
"scribe.services.embeddings.get_embedding",
AsyncMock(return_value=_vec(1.0)),
):
results = await semantic_search_notes(user_id=user_id, query="anything", limit=10)
ids = [note.id for _score, note in results]
# Near note returned and ranked first; far (orthogonal, sim≈0) excluded by
# the default 0.45 similarity threshold.
assert near_id in ids
assert far_id not in ids
assert ids[0] == near_id
# Chunk collapse (#280): near has TWO chunk rows above the floor (sim≈1.0
# and ≈0.6) and must appear exactly once, at its best chunk's score.
assert ids.count(near_id) == 1
top_score = results[0][0]
assert top_score == pytest.approx(1.0, abs=1e-3)
@pytest.mark.asyncio
async def test_low_threshold_lets_orthogonal_through(seeded):
user_id, near_id, far_id = seeded
with patch(
"scribe.services.embeddings.get_embedding",
AsyncMock(return_value=_vec(1.0)),
):
results = await semantic_search_notes(
user_id=user_id, query="anything", limit=10, threshold=-1.0,
)
ids = [note.id for _score, note in results]
# With the floor dropped, both come back and near still ranks above far.
assert ids.index(near_id) < ids.index(far_id)