From 983fd2c4d15aecf37c77c609d25e50ef76526a38 Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Mon, 5 Oct 2026 19:30:05 -0400 Subject: [PATCH] fix(retrieval): scope the rule search before ranking it - an in-scope rule is no longer lost behind other owners' nearer rules (#4958) Ordered straight off rule_embeddings, the planner walked the HNSW index, which returns about hnsw.ef_search (40) nearest chunks across every owner and project and filters by home only afterwards. A reader's own rule ranked past the 40th chunk overall was silently dropped - on a shared install, other users' rules fill those 40. This is the likely cause of the intermittent test_integration_rule_scope failures, whose axis-vector fixtures sit far from every real embedding in the graph. The in-scope chunks now go through a MATERIALIZED CTE, which the index cannot order, so the ranking over them is exact. A rulebook is hundreds of chunks; exact is cheap. New test: 80 nearer rules belonging to someone else no longer hide the reader's one rule. Co-Authored-By: Claude Opus 5.5 --- src/scribe/services/embeddings.py | 52 ++++++++++++++++-------- tests/test_integration_rule_scope.py | 61 ++++++++++++++++++++++++++++ 2 files changed, 95 insertions(+), 18 deletions(-) diff --git a/src/scribe/services/embeddings.py b/src/scribe/services/embeddings.py index e4f7622c..b79d5695 100644 --- a/src/scribe/services/embeddings.py +++ b/src/scribe/services/embeddings.py @@ -1550,28 +1550,44 @@ async def semantic_search_rules( # function fails open. home = await rule_home(user_id, project_id, everywhere=everywhere) + # SCOPE FIRST, THEN RANK — exactly (#4958). Ordered straight off + # rule_embeddings, the planner walks the HNSW index, which hands back + # about `hnsw.ef_search` (40) nearest chunks from EVERY owner and + # project and only then applies the home filter: an in-scope rule + # ranked past the 40th chunk overall is silently gone, and on a + # shared install other users' rules fill those 40. A MATERIALIZED + # CTE cannot be ordered through the index, so the distance is + # computed for each in-scope chunk and the ranking is exact. A + # rulebook is hundreds of chunks, not millions — exact is cheap here. + scoped = ( + joined_to_homes( + select( + RuleEmbedding.rule_id.label("rule_id"), + distance.label("distance"), + RuleEmbedding.chunk_index.label("chunk_index"), + RuleEmbedding.chunk_text.label("chunk_text"), + ) + .select_from(RuleEmbedding) + .join(Rule, RuleEmbedding.rule_id == Rule.id) + ) + .where( + Rule.deleted_at.is_(None), + # No threshold predicate — see the note above + # semantic_search_notes. Applied below, after the collapse. + home, + *( [Rule.kind == kind] if kind else [] ), + ) + .cte("scoped_rule_chunks") + .prefix_with("MATERIALIZED") + ) + async with async_session() as session: rows = (await session.execute( - joined_to_homes( - select( - Rule, - distance.label("distance"), - RuleEmbedding.chunk_index, - RuleEmbedding.chunk_text, - ) - .select_from(RuleEmbedding) - .join(Rule, RuleEmbedding.rule_id == Rule.id) - ) - .where( - Rule.deleted_at.is_(None), - # No threshold predicate — see the note above - # semantic_search_notes. Applied below, after the collapse. - home, - *( [Rule.kind == kind] if kind else [] ), - ) + select(Rule, scoped.c.distance, scoped.c.chunk_index, scoped.c.chunk_text) + .join(scoped, scoped.c.rule_id == Rule.id) # Overfetch so collapsing chunks to their best row still fills # the page — the same reason the note search overfetches. - .order_by(distance) + .order_by(scoped.c.distance) .limit(limit * _CHUNK_OVERFETCH) )).all() except Exception: diff --git a/tests/test_integration_rule_scope.py b/tests/test_integration_rule_scope.py index 464993d2..8c551d3c 100644 --- a/tests/test_integration_rule_scope.py +++ b/tests/test_integration_rule_scope.py @@ -115,3 +115,64 @@ async def test_a_shared_project_brings_its_rules_to_a_collaborator(homes): collaborator = homes["collaborator"] assert await _found(collaborator, project_id=homes["a"]) == {homes["on_a"]} assert await _found(collaborator, project_id=homes["b"]) == set() + + +async def test_an_in_scope_rule_is_found_however_many_closer_rules_are_out_of_scope(): + """#4958. Ordered through the HNSW index, the search took the ~40 nearest + chunks from every owner and filtered them AFTERWARDS — so a reader's own + rule ranked past the 40th chunk overall was never returned. Here another + user's 80 rules all sit nearer the query than the reader's one rule; the + reader's search must still find it.""" + tag = uuid.uuid4().hex[:8] + async with async_session() as s: + reader = await ensure_user(s, f"rule_scope_reader_{tag}") + crowd = await ensure_user(s, f"rule_scope_crowd_{tag}") + await s.commit() + reader_id, crowd_id = reader.id, crowd.id + + def near(i: int) -> list[float]: + # Cosine ~0.9996 to the query, each distinct so the graph is a graph. + vec = [1.0] + [0.0] * (EMBEDDING_DIM - 1) + vec[1 + i] = 0.02 + return vec + + mine_vec = [0.8, 0.6] + [0.0] * (EMBEDDING_DIM - 2) # cosine 0.8 + books = [] + try: + with patch("scribe.services.rulebooks._refresh_rule_embedding", MagicMock()): + book = await rulebooks_svc.create_rulebook(reader_id, "Reader's rules") + books.append((book.id, reader_id)) + topic = await rulebooks_svc.create_topic(book.id, reader_id, "mine") + mine = await rulebooks_svc.create_rule( + topic.id, reader_id, "The reader's own rule", "Mine.", when_to_apply="always", + ) + crowd_book = await rulebooks_svc.create_rulebook(crowd_id, "Someone else's rules") + books.append((crowd_book.id, crowd_id)) + crowd_topic = await rulebooks_svc.create_topic(crowd_book.id, crowd_id, "theirs") + theirs = [ + await rulebooks_svc.create_rule( + crowd_topic.id, crowd_id, f"Crowd rule {i}", "Not the reader's.", + when_to_apply="always", + ) + for i in range(80) + ] + async with async_session() as s: + s.add(RuleEmbedding( + rule_id=mine.id, chunk_index=0, embedding=mine_vec, chunk_text=mine.title, + chunker_version=CHUNKER_VERSION, embedding_model=EMBEDDING_MODEL, + )) + for i, rule in enumerate(theirs): + s.add(RuleEmbedding( + rule_id=rule.id, chunk_index=0, embedding=near(i), chunk_text=rule.title, + chunker_version=CHUNKER_VERSION, embedding_model=EMBEDDING_MODEL, + )) + await s.commit() + + assert await _found(reader_id) == {mine.id} + # And the crowd still finds its own, not the reader's. + found = await _found(crowd_id) + assert mine.id not in found and len(found) == 10 + finally: + # These vectors sit right beside the query every other test here uses. + for book_id, owner in books: + await rulebooks_svc.delete_rulebook(book_id, owner)