fix(retrieval): scope the rule search before ranking it - an in-scope rule is no longer lost behind other owners' nearer rules (#4958)
CI & Build / Python lint (push) Successful in 5s
CI & Build / Plugin hooks (push) Successful in 20s
CI & Build / TypeScript typecheck (push) Successful in 55s
CI & Build / integration (push) Successful in 1m11s
CI & Build / Python tests (push) Successful in 2m0s
CI & Build / Build & push image (push) Successful in 29s
CI & Build / Python lint (push) Successful in 5s
CI & Build / Plugin hooks (push) Successful in 20s
CI & Build / TypeScript typecheck (push) Successful in 55s
CI & Build / integration (push) Successful in 1m11s
CI & Build / Python tests (push) Successful in 2m0s
CI & Build / Build & push image (push) Successful in 29s
Ordered straight off rule_embeddings, the planner walked the HNSW index, which returns about hnsw.ef_search (40) nearest chunks across every owner and project and filters by home only afterwards. A reader's own rule ranked past the 40th chunk overall was silently dropped - on a shared install, other users' rules fill those 40. This is the likely cause of the intermittent test_integration_rule_scope failures, whose axis-vector fixtures sit far from every real embedding in the graph. The in-scope chunks now go through a MATERIALIZED CTE, which the index cannot order, so the ranking over them is exact. A rulebook is hundreds of chunks; exact is cheap. New test: 80 nearer rules belonging to someone else no longer hide the reader's one rule. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -1550,14 +1550,22 @@ async def semantic_search_rules(
|
||||
# function fails open.
|
||||
home = await rule_home(user_id, project_id, everywhere=everywhere)
|
||||
|
||||
async with async_session() as session:
|
||||
rows = (await session.execute(
|
||||
# SCOPE FIRST, THEN RANK — exactly (#4958). Ordered straight off
|
||||
# rule_embeddings, the planner walks the HNSW index, which hands back
|
||||
# about `hnsw.ef_search` (40) nearest chunks from EVERY owner and
|
||||
# project and only then applies the home filter: an in-scope rule
|
||||
# ranked past the 40th chunk overall is silently gone, and on a
|
||||
# shared install other users' rules fill those 40. A MATERIALIZED
|
||||
# CTE cannot be ordered through the index, so the distance is
|
||||
# computed for each in-scope chunk and the ranking is exact. A
|
||||
# rulebook is hundreds of chunks, not millions — exact is cheap here.
|
||||
scoped = (
|
||||
joined_to_homes(
|
||||
select(
|
||||
Rule,
|
||||
RuleEmbedding.rule_id.label("rule_id"),
|
||||
distance.label("distance"),
|
||||
RuleEmbedding.chunk_index,
|
||||
RuleEmbedding.chunk_text,
|
||||
RuleEmbedding.chunk_index.label("chunk_index"),
|
||||
RuleEmbedding.chunk_text.label("chunk_text"),
|
||||
)
|
||||
.select_from(RuleEmbedding)
|
||||
.join(Rule, RuleEmbedding.rule_id == Rule.id)
|
||||
@@ -1569,9 +1577,17 @@ async def semantic_search_rules(
|
||||
home,
|
||||
*( [Rule.kind == kind] if kind else [] ),
|
||||
)
|
||||
.cte("scoped_rule_chunks")
|
||||
.prefix_with("MATERIALIZED")
|
||||
)
|
||||
|
||||
async with async_session() as session:
|
||||
rows = (await session.execute(
|
||||
select(Rule, scoped.c.distance, scoped.c.chunk_index, scoped.c.chunk_text)
|
||||
.join(scoped, scoped.c.rule_id == Rule.id)
|
||||
# Overfetch so collapsing chunks to their best row still fills
|
||||
# the page — the same reason the note search overfetches.
|
||||
.order_by(distance)
|
||||
.order_by(scoped.c.distance)
|
||||
.limit(limit * _CHUNK_OVERFETCH)
|
||||
)).all()
|
||||
except Exception:
|
||||
|
||||
@@ -115,3 +115,64 @@ async def test_a_shared_project_brings_its_rules_to_a_collaborator(homes):
|
||||
collaborator = homes["collaborator"]
|
||||
assert await _found(collaborator, project_id=homes["a"]) == {homes["on_a"]}
|
||||
assert await _found(collaborator, project_id=homes["b"]) == set()
|
||||
|
||||
|
||||
async def test_an_in_scope_rule_is_found_however_many_closer_rules_are_out_of_scope():
|
||||
"""#4958. Ordered through the HNSW index, the search took the ~40 nearest
|
||||
chunks from every owner and filtered them AFTERWARDS — so a reader's own
|
||||
rule ranked past the 40th chunk overall was never returned. Here another
|
||||
user's 80 rules all sit nearer the query than the reader's one rule; the
|
||||
reader's search must still find it."""
|
||||
tag = uuid.uuid4().hex[:8]
|
||||
async with async_session() as s:
|
||||
reader = await ensure_user(s, f"rule_scope_reader_{tag}")
|
||||
crowd = await ensure_user(s, f"rule_scope_crowd_{tag}")
|
||||
await s.commit()
|
||||
reader_id, crowd_id = reader.id, crowd.id
|
||||
|
||||
def near(i: int) -> list[float]:
|
||||
# Cosine ~0.9996 to the query, each distinct so the graph is a graph.
|
||||
vec = [1.0] + [0.0] * (EMBEDDING_DIM - 1)
|
||||
vec[1 + i] = 0.02
|
||||
return vec
|
||||
|
||||
mine_vec = [0.8, 0.6] + [0.0] * (EMBEDDING_DIM - 2) # cosine 0.8
|
||||
books = []
|
||||
try:
|
||||
with patch("scribe.services.rulebooks._refresh_rule_embedding", MagicMock()):
|
||||
book = await rulebooks_svc.create_rulebook(reader_id, "Reader's rules")
|
||||
books.append((book.id, reader_id))
|
||||
topic = await rulebooks_svc.create_topic(book.id, reader_id, "mine")
|
||||
mine = await rulebooks_svc.create_rule(
|
||||
topic.id, reader_id, "The reader's own rule", "Mine.", when_to_apply="always",
|
||||
)
|
||||
crowd_book = await rulebooks_svc.create_rulebook(crowd_id, "Someone else's rules")
|
||||
books.append((crowd_book.id, crowd_id))
|
||||
crowd_topic = await rulebooks_svc.create_topic(crowd_book.id, crowd_id, "theirs")
|
||||
theirs = [
|
||||
await rulebooks_svc.create_rule(
|
||||
crowd_topic.id, crowd_id, f"Crowd rule {i}", "Not the reader's.",
|
||||
when_to_apply="always",
|
||||
)
|
||||
for i in range(80)
|
||||
]
|
||||
async with async_session() as s:
|
||||
s.add(RuleEmbedding(
|
||||
rule_id=mine.id, chunk_index=0, embedding=mine_vec, chunk_text=mine.title,
|
||||
chunker_version=CHUNKER_VERSION, embedding_model=EMBEDDING_MODEL,
|
||||
))
|
||||
for i, rule in enumerate(theirs):
|
||||
s.add(RuleEmbedding(
|
||||
rule_id=rule.id, chunk_index=0, embedding=near(i), chunk_text=rule.title,
|
||||
chunker_version=CHUNKER_VERSION, embedding_model=EMBEDDING_MODEL,
|
||||
))
|
||||
await s.commit()
|
||||
|
||||
assert await _found(reader_id) == {mine.id}
|
||||
# And the crowd still finds its own, not the reader's.
|
||||
found = await _found(crowd_id)
|
||||
assert mine.id not in found and len(found) == 10
|
||||
finally:
|
||||
# These vectors sit right beside the query every other test here uses.
|
||||
for book_id, owner in books:
|
||||
await rulebooks_svc.delete_rulebook(book_id, owner)
|
||||
|
||||
Reference in New Issue
Block a user