CI & Build / Python lint (push) Successful in 5s
CI & Build / Plugin hooks (push) Successful in 20s
CI & Build / TypeScript typecheck (push) Successful in 55s
CI & Build / integration (push) Successful in 1m11s
CI & Build / Python tests (push) Successful in 2m0s
CI & Build / Build & push image (push) Successful in 29s
Ordered straight off rule_embeddings, the planner walked the HNSW index, which returns about hnsw.ef_search (40) nearest chunks across every owner and project and filters by home only afterwards. A reader's own rule ranked past the 40th chunk overall was silently dropped - on a shared install, other users' rules fill those 40. This is the likely cause of the intermittent test_integration_rule_scope failures, whose axis-vector fixtures sit far from every real embedding in the graph. The in-scope chunks now go through a MATERIALIZED CTE, which the index cannot order, so the ranking over them is exact. A rulebook is hundreds of chunks; exact is cheap. New test: 80 nearer rules belonging to someone else no longer hide the reader's one rule. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
179 lines
8.3 KiB
Python
179 lines
8.3 KiB
Python
"""Real-Postgres tests for WHERE a rule reaches (milestone 414, step 1).
|
|
|
|
A rule lives in a rulebook topic (global) or on one project. Retrieval used to
|
|
ignore that and search every rule the user owned, so each project's rules were
|
|
injected into every other project's sessions. What a mock cannot show is the
|
|
join doing the scoping: these seed real rules with hand-made vectors and stub
|
|
only the embedder, so every rule is an equally good match and the home alone
|
|
decides what comes back.
|
|
"""
|
|
import traceback
|
|
import uuid
|
|
from unittest.mock import AsyncMock, MagicMock, patch
|
|
|
|
import pytest
|
|
import pytest_asyncio
|
|
|
|
from scribe.models import async_session
|
|
from scribe.models.embedding import EMBEDDING_DIM, RuleEmbedding
|
|
from scribe.models.project import Project
|
|
from scribe.models.share import ProjectShare
|
|
from scribe.services import rulebooks as rulebooks_svc
|
|
from scribe.services.embeddings import CHUNKER_VERSION, EMBEDDING_MODEL, semantic_search_rules
|
|
from tests.helpers import ensure_user
|
|
|
|
pytestmark = [pytest.mark.integration, pytest.mark.usefixtures("_dispose_engine")]
|
|
|
|
QUERY_VEC = [1.0] + [0.0] * (EMBEDDING_DIM - 1)
|
|
|
|
|
|
@pytest_asyncio.fixture
|
|
async def homes():
|
|
"""One global rule, one rule on project A, one on project B, and a
|
|
collaborator A is shared with. Every rule embeds identically to the query."""
|
|
# Fresh users per test: every rule matches the query equally, so a rule
|
|
# left by another test would be indistinguishable from a scoping leak.
|
|
tag = uuid.uuid4().hex[:8]
|
|
async with async_session() as s:
|
|
owner = await ensure_user(s, f"rule_scope_owner_{tag}")
|
|
collaborator = await ensure_user(s, f"rule_scope_collaborator_{tag}")
|
|
a = Project(user_id=owner.id, title="Scope A")
|
|
b = Project(user_id=owner.id, title="Scope B")
|
|
s.add_all([a, b])
|
|
await s.flush()
|
|
s.add(ProjectShare(project_id=a.id, shared_with_user_id=collaborator.id,
|
|
permission="viewer", invited_by=owner.id))
|
|
ids = {"owner": owner.id, "collaborator": collaborator.id, "a": a.id, "b": b.id}
|
|
await s.commit()
|
|
|
|
with patch("scribe.services.rulebooks._refresh_rule_embedding", MagicMock()):
|
|
book = await rulebooks_svc.create_rulebook(ids["owner"], "Scope house style")
|
|
topic = await rulebooks_svc.create_topic(book.id, ids["owner"], "everywhere")
|
|
glob = await rulebooks_svc.create_rule(
|
|
topic.id, ids["owner"], "Global scope rule", "Applies in every project.",
|
|
when_to_apply="always",
|
|
)
|
|
on_a = await rulebooks_svc.create_project_rule(
|
|
ids["a"], ids["owner"], "Project A rule", "Applies to A only.",
|
|
when_to_apply="working on A",
|
|
)
|
|
on_b = await rulebooks_svc.create_project_rule(
|
|
ids["b"], ids["owner"], "Project B rule", "Applies to B only.",
|
|
when_to_apply="working on B",
|
|
)
|
|
|
|
async with async_session() as s:
|
|
for rule in (glob, on_a, on_b):
|
|
s.add(RuleEmbedding(
|
|
rule_id=rule.id, chunk_index=0, embedding=QUERY_VEC,
|
|
chunk_text=rule.title, chunker_version=CHUNKER_VERSION,
|
|
embedding_model=EMBEDDING_MODEL,
|
|
))
|
|
await s.commit()
|
|
ids.update(glob=glob.id, on_a=on_a.id, on_b=on_b.id)
|
|
return ids
|
|
|
|
|
|
async def _found(user_id: int, **scope) -> set[int]:
|
|
"""The rule ids a search finds — and a failure, not an empty set, when the
|
|
search never ran. `semantic_search_rules` fails open (an exception becomes
|
|
[]), so without `report["searched"]` an error and a scoping miss read the
|
|
same (#4958)."""
|
|
report: dict = {}
|
|
raised: list[str] = []
|
|
with patch("scribe.services.embeddings.get_embedding",
|
|
AsyncMock(return_value=QUERY_VEC)), \
|
|
patch("scribe.services.embeddings.logger") as log:
|
|
# Called inside the except block, so the traceback is still current.
|
|
log.warning.side_effect = lambda *_a, **_k: raised.append(traceback.format_exc())
|
|
hits = await semantic_search_rules(user_id, "anything", limit=10,
|
|
threshold=0.5, report=report, **scope)
|
|
assert report.get("searched"), "the rule search did not run:\n" + "\n".join(raised)
|
|
return {rule.id for _score, rule in hits}
|
|
|
|
|
|
async def test_retrieval_reaches_a_rule_only_from_its_home(homes):
|
|
owner = homes["owner"]
|
|
glob, on_a, on_b = homes["glob"], homes["on_a"], homes["on_b"]
|
|
|
|
# A session bound to A: the global rule and A's own, never B's.
|
|
assert await _found(owner, project_id=homes["a"]) == {glob, on_a}
|
|
assert await _found(owner, project_id=homes["b"]) == {glob, on_b}
|
|
|
|
# No bound project, and the default: global only. A caller that forgets
|
|
# to pass a scope surfaces less, not another project's rules.
|
|
assert await _found(owner) == {glob}
|
|
|
|
# The explicit whole-rulebook question still reaches everything.
|
|
assert await _found(owner, everywhere=True) == {glob, on_a, on_b}
|
|
|
|
|
|
async def test_a_shared_project_brings_its_rules_to_a_collaborator(homes):
|
|
"""Readability goes through access.can_read_project (rule 78): a viewer on
|
|
A gets A's rules. Not the owner's global rules — rulebooks are the
|
|
owner's — and not B's, which is not shared."""
|
|
collaborator = homes["collaborator"]
|
|
assert await _found(collaborator, project_id=homes["a"]) == {homes["on_a"]}
|
|
assert await _found(collaborator, project_id=homes["b"]) == set()
|
|
|
|
|
|
async def test_an_in_scope_rule_is_found_however_many_closer_rules_are_out_of_scope():
|
|
"""#4958. Ordered through the HNSW index, the search took the ~40 nearest
|
|
chunks from every owner and filtered them AFTERWARDS — so a reader's own
|
|
rule ranked past the 40th chunk overall was never returned. Here another
|
|
user's 80 rules all sit nearer the query than the reader's one rule; the
|
|
reader's search must still find it."""
|
|
tag = uuid.uuid4().hex[:8]
|
|
async with async_session() as s:
|
|
reader = await ensure_user(s, f"rule_scope_reader_{tag}")
|
|
crowd = await ensure_user(s, f"rule_scope_crowd_{tag}")
|
|
await s.commit()
|
|
reader_id, crowd_id = reader.id, crowd.id
|
|
|
|
def near(i: int) -> list[float]:
|
|
# Cosine ~0.9996 to the query, each distinct so the graph is a graph.
|
|
vec = [1.0] + [0.0] * (EMBEDDING_DIM - 1)
|
|
vec[1 + i] = 0.02
|
|
return vec
|
|
|
|
mine_vec = [0.8, 0.6] + [0.0] * (EMBEDDING_DIM - 2) # cosine 0.8
|
|
books = []
|
|
try:
|
|
with patch("scribe.services.rulebooks._refresh_rule_embedding", MagicMock()):
|
|
book = await rulebooks_svc.create_rulebook(reader_id, "Reader's rules")
|
|
books.append((book.id, reader_id))
|
|
topic = await rulebooks_svc.create_topic(book.id, reader_id, "mine")
|
|
mine = await rulebooks_svc.create_rule(
|
|
topic.id, reader_id, "The reader's own rule", "Mine.", when_to_apply="always",
|
|
)
|
|
crowd_book = await rulebooks_svc.create_rulebook(crowd_id, "Someone else's rules")
|
|
books.append((crowd_book.id, crowd_id))
|
|
crowd_topic = await rulebooks_svc.create_topic(crowd_book.id, crowd_id, "theirs")
|
|
theirs = [
|
|
await rulebooks_svc.create_rule(
|
|
crowd_topic.id, crowd_id, f"Crowd rule {i}", "Not the reader's.",
|
|
when_to_apply="always",
|
|
)
|
|
for i in range(80)
|
|
]
|
|
async with async_session() as s:
|
|
s.add(RuleEmbedding(
|
|
rule_id=mine.id, chunk_index=0, embedding=mine_vec, chunk_text=mine.title,
|
|
chunker_version=CHUNKER_VERSION, embedding_model=EMBEDDING_MODEL,
|
|
))
|
|
for i, rule in enumerate(theirs):
|
|
s.add(RuleEmbedding(
|
|
rule_id=rule.id, chunk_index=0, embedding=near(i), chunk_text=rule.title,
|
|
chunker_version=CHUNKER_VERSION, embedding_model=EMBEDDING_MODEL,
|
|
))
|
|
await s.commit()
|
|
|
|
assert await _found(reader_id) == {mine.id}
|
|
# And the crowd still finds its own, not the reader's.
|
|
found = await _found(crowd_id)
|
|
assert mine.id not in found and len(found) == 10
|
|
finally:
|
|
# These vectors sit right beside the query every other test here uses.
|
|
for book_id, owner in books:
|
|
await rulebooks_svc.delete_rulebook(book_id, owner)
|