"""Real-Postgres crowd tests: an in-scope record is found however many nearer records sit outside the scope (#4961; the rule search's is #4958's, in test_integration_rule_scope). Ordered straight off a `*_embeddings` table, the planner can walk the HNSW index, which takes ~`hnsw.ef_search` (40) nearest chunks from every owner and project and only then applies the scope. Each test here puts 80 out-of-scope records nearer the query than the reader's one in-scope record, so the old shape could not reach it when the index served the order. What a test cannot force is the plan — the shape guard in test_search_scope_shape is what fails on the old query whatever Postgres picks. Rows are inserted directly with hand-made vectors and the embedder is stubbed, so the scope alone decides what comes back. """ import traceback import uuid from unittest.mock import AsyncMock, patch import pytest from sqlalchemy import delete from scribe.models import async_session from scribe.models.embedding import ( EMBEDDING_DIM, MilestoneEmbedding, NoteEmbedding, SystemEmbedding, ) from scribe.models.milestone import Milestone from scribe.models.note import Note from scribe.models.project import Project from scribe.models.system import System from scribe.models.user import User from scribe.services import embeddings as emb from scribe.services.embeddings import CHUNKER_VERSION, EMBEDDING_MODEL from tests.helpers import ensure_user pytestmark = [pytest.mark.integration, pytest.mark.usefixtures("_dispose_engine", "_no_embedding")] CROWD = 80 QUERY_VEC = [1.0] + [0.0] * (EMBEDDING_DIM - 1) # Cosine 0.8 to the query: comfortably above the bar, and behind every crowd row. MINE_VEC = [0.8, 0.6] + [0.0] * (EMBEDDING_DIM - 2) def _near(i: int) -> list[float]: """Cosine ~0.9996 to the query, each distinct so the graph is a graph.""" vec = [1.0] + [0.0] * (EMBEDDING_DIM - 1) vec[1 + i] = 0.02 return vec def _chunk(model, key: str, record_id: int, vec: list[float], **extra): return model(**{key: record_id}, chunk_index=0, embedding=vec, chunk_text=f"record {record_id}", chunker_version=CHUNKER_VERSION, embedding_model=EMBEDDING_MODEL, **extra) async def _found(search, user_id: int, **scope) -> set[int]: """The ids a search finds. Every search fails open, so a query that raised reads as an empty result — the swallowed traceback is captured and named in the failure instead (#4958).""" raised: list[str] = [] with patch("scribe.services.embeddings.get_embedding", AsyncMock(return_value=QUERY_VEC)), \ patch("scribe.services.embeddings.logger") as log: log.warning.side_effect = lambda *_a, **_k: raised.append(traceback.format_exc()) hits = await search(user_id, "anything", limit=10, threshold=0.5, **scope) assert not raised, "the search raised:\n" + "\n".join(raised) return {int(record.id) for _score, record in hits} async def _people(tag: str): """A reader and a crowd, each with a project of their own.""" async with async_session() as s: reader = await ensure_user(s, f"scope_reader_{tag}") crowd = await ensure_user(s, f"scope_crowd_{tag}") mine = Project(user_id=reader.id, title="Reader's project") other = Project(user_id=reader.id, title="Reader's other project") theirs = Project(user_id=crowd.id, title="Crowd's project") s.add_all([mine, other, theirs]) await s.commit() return reader.id, crowd.id, mine.id, other.id, theirs.id async def _cleanup(*user_ids: int) -> None: # These vectors sit right beside the query every other vector test uses. async with async_session() as s: for model in (Note, Milestone, System): await s.execute(delete(model).where(model.user_id.in_(user_ids))) await s.execute(delete(Project).where(Project.user_id.in_(user_ids))) await s.execute(delete(User).where(User.id.in_(user_ids))) await s.commit() @pytest.mark.asyncio async def test_a_note_is_found_behind_another_users_nearer_notes(): reader, crowd, mine_p, _other, theirs_p = await _people(uuid.uuid4().hex[:8]) try: async with async_session() as s: mine = Note(user_id=reader, project_id=mine_p, title="mine", body="mine") crowd_notes = [Note(user_id=crowd, project_id=theirs_p, title=f"crowd {i}", body="not the reader's") for i in range(CROWD)] s.add_all([mine, *crowd_notes]) await s.flush() s.add(_chunk(NoteEmbedding, "note_id", mine.id, MINE_VEC, user_id=reader)) s.add_all(_chunk(NoteEmbedding, "note_id", n.id, _near(i), user_id=crowd) for i, n in enumerate(crowd_notes)) await s.commit() mine_id = mine.id assert await _found(emb.semantic_search_notes, reader) == {mine_id} found = await _found(emb.semantic_search_notes, crowd) assert mine_id not in found and len(found) == 10 finally: await _cleanup(reader, crowd) @pytest.mark.asyncio async def test_a_note_is_found_behind_the_readers_own_other_project(): """The single-user case, and the one the hook arms live in: a project-scoped search where the reader's OTHER projects hold the nearer chunks.""" reader, crowd, mine_p, other_p, _theirs = await _people(uuid.uuid4().hex[:8]) try: async with async_session() as s: mine = Note(user_id=reader, project_id=mine_p, title="mine", body="mine") elsewhere = [Note(user_id=reader, project_id=other_p, title=f"elsewhere {i}", body="another project") for i in range(CROWD)] s.add_all([mine, *elsewhere]) await s.flush() s.add(_chunk(NoteEmbedding, "note_id", mine.id, MINE_VEC, user_id=reader)) s.add_all(_chunk(NoteEmbedding, "note_id", n.id, _near(i), user_id=reader) for i, n in enumerate(elsewhere)) await s.commit() mine_id = mine.id assert await _found(emb.semantic_search_notes, reader, project_id=mine_p) == {mine_id} finally: await _cleanup(reader, crowd) @pytest.mark.asyncio async def test_a_milestone_is_found_behind_another_users_nearer_plans(): reader, crowd, mine_p, _other, theirs_p = await _people(uuid.uuid4().hex[:8]) try: async with async_session() as s: mine = Milestone(user_id=reader, project_id=mine_p, title="my plan") plans = [Milestone(user_id=crowd, project_id=theirs_p, title=f"plan {i}") for i in range(CROWD)] s.add_all([mine, *plans]) await s.flush() s.add(_chunk(MilestoneEmbedding, "milestone_id", mine.id, MINE_VEC)) s.add_all(_chunk(MilestoneEmbedding, "milestone_id", m.id, _near(i)) for i, m in enumerate(plans)) await s.commit() mine_id = mine.id assert await _found(emb.semantic_search_milestones, reader) == {mine_id} assert await _found(emb.semantic_search_milestones, reader, project_id=mine_p) == {mine_id} found = await _found(emb.semantic_search_milestones, crowd) assert mine_id not in found and len(found) == 10 finally: await _cleanup(reader, crowd) @pytest.mark.asyncio async def test_a_system_is_found_behind_another_users_nearer_charters(): reader, crowd, mine_p, _other, theirs_p = await _people(uuid.uuid4().hex[:8]) try: async with async_session() as s: mine = System(user_id=reader, project_id=mine_p, name="mine", description="the reader's area") areas = [System(user_id=crowd, project_id=theirs_p, name=f"area {i}", description="someone else's area") for i in range(CROWD)] s.add_all([mine, *areas]) await s.flush() s.add(_chunk(SystemEmbedding, "system_id", mine.id, MINE_VEC)) s.add_all(_chunk(SystemEmbedding, "system_id", a.id, _near(i)) for i, a in enumerate(areas)) await s.commit() mine_id = mine.id assert await _found(emb.semantic_search_systems, reader) == {mine_id} assert await _found(emb.semantic_search_systems, reader, project_id=mine_p) == {mine_id} found = await _found(emb.semantic_search_systems, crowd) assert mine_id not in found and len(found) == 10 finally: await _cleanup(reader, crowd)