Files
FabledScribe/tests/test_mcp_tool_search.py
T
bvandeusenandClaude Opus 5 5fb41af9b0
CI & Build / Python lint (push) Successful in 2s
CI & Build / Plugin hooks (push) Successful in 11s
CI & Build / TypeScript typecheck (push) Successful in 52s
CI & Build / integration (push) Successful in 58s
CI & Build / Python tests (push) Successful in 1m42s
CI & Build / Build & push image (push) Successful in 28s
feat(search): the agent's search can ask for every kind the corpus has (#4250)
The engine took `note_type` and `task_kind` all along. What was missing was a
way to say them: the MCP tool's `content_type` knew `note`, `task` and `all`,
and `/api/search` knew the same two — so an agent could not ask "has this
snippet already been recorded" or "what lessons apply here" without searching
everything and reading past the rest. Browse offered nine kinds from the same
data.

The cause is that each door kept its own map. `_FACETS` in services/knowledge
is where a kind is declared, and #3161 made adding one a single edit by
generating the SQL filter, the Python predicate and the door's validation from
it — but the two search doors were written before that and never joined. So
this adds the third dialect, `search_filters_for`, and one composition over it,
`content_type_filters`, and both doors now derive instead of listing.

Two names keep a meaning of their own, and the docstrings say so: `all` is no
filter, and `note` is BROAD — any non-task, snippets and lessons included —
where the browse facet of the same name is narrow (`note_type == 'note'`).
They are left different deliberately; narrowing this one would stop returning
snippets to every caller that already asks this way.

An unrecognised kind is now refused rather than answered. Both doors used to
fall through: the MCP tool into a filter matching no row, the route into no
filter at all, so `?content_type=snippets` returned the whole corpus while
looking like a narrowed search. An empty result set is a claim — "the corpus
holds nothing like this" — and an agent acts on that claim by building the
thing it could not find, so a typo must not be able to make it.

The docstring is the agent-facing contract (#2846), and a test now holds it to
the table: every kind `_FACETS` declares has to appear in it, because a filter
an agent has not been told about is unreachable however well it is wired.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01821k5B3Ysecp9fNYs92Kuy
2026-09-21 11:12:48 -04:00

332 lines
14 KiB
Python

"""search tool — proves the tool pattern (context + service call + dict shape).
Service call is mocked; no DB needed."""
from unittest.mock import AsyncMock, patch
import pytest
from scribe.mcp._context import _user_id_ctx
from scribe.mcp.tools.search import search
from tests.helpers import fake_note
@pytest.fixture(autouse=True)
def _reset_user_ctx():
"""Each test starts with no MCP context. Tests set it explicitly."""
token = _user_id_ctx.set(None)
yield
_user_id_ctx.reset(token)
@pytest.mark.asyncio
async def test_fable_search_raises_without_context():
with pytest.raises(RuntimeError, match="no MCP user context"):
await search(q="anything")
@pytest.mark.asyncio
async def test_fable_search_returns_repackaged_results():
_user_id_ctx.set(7)
fake = fake_note(id=1, title="kafka rebalance", is_task=False, body="HPA details", tags=["ops"])
with patch(
"scribe.mcp.tools.search.semantic_search_notes",
AsyncMock(return_value=[(0.93, fake)]),
):
out = await search(q="kafka")
assert out["total"] == 1
assert len(out["results"]) == 1
r = out["results"][0]
assert r["id"] == 1
assert r["title"] == "kafka rebalance"
# The whole body, because it fits — and named as the opening, not as the
# passage that matched, since this call patched the search and so carries
# no chunk.
assert r["excerpt"] == "HPA details"
assert r["excerpt_is"] == "body_opening"
assert r["body_length"] == len("HPA details")
assert r["is_task"] is False
assert r["tags"] == ["ops"]
assert r["similarity"] == pytest.approx(0.93)
@pytest.mark.asyncio
async def test_search_shows_the_passage_that_matched_not_the_opening():
"""The point of #4243. A record can rank on its sixth paragraph; showing
its first is showing the caller a span the search already judged less
relevant, and letting them decide from it."""
_user_id_ctx.set(7)
body = "Chapter one, about nothing. " * 40 + " THE ANSWER IS 04775c3."
fake = fake_note(id=1, title="t", body=body)
async def _search(*a, **kw):
kw["report"]["best_chunk"] = {
1: {"index": 3, "text": "THE ANSWER IS 04775c3."}
}
return [(0.5, fake)]
with patch("scribe.mcp.tools.search.semantic_search_notes", _search):
out = await search(q="answer")
r = out["results"][0]
assert r["excerpt"] == "THE ANSWER IS 04775c3."
assert r["excerpt_is"] == "matched_passage"
assert r["chunk_index"] == 3
# And the caller is told there is more record behind the passage.
assert r["body_length"] == len(body)
assert "read_full" in r
@pytest.mark.asyncio
async def test_a_result_says_whether_its_excerpt_is_the_match_or_the_opening():
"""A caller that cannot tell the two apart cannot judge whether looking
deeper is worth it, which is the only decision this field supports."""
_user_id_ctx.set(7)
fake = fake_note(id=1, title="t", body="short body")
with patch(
"scribe.mcp.tools.search.semantic_search_notes",
AsyncMock(return_value=[(0.5, fake)]),
):
out = await search(q="x")
assert out["results"][0]["excerpt_is"] == "body_opening"
@pytest.mark.asyncio
async def test_a_long_excerpt_keeps_both_ends_and_says_how_much_went():
"""The fallback is still an elision, and an elision that drops the tail
drops wherever the conclusion was."""
_user_id_ctx.set(7)
body = "OPENING. " + ("m" * 3000) + " CLOSING."
fake = fake_note(id=1, title="t", body=body)
with patch(
"scribe.mcp.tools.search.semantic_search_notes",
AsyncMock(return_value=[(0.5, fake)]),
):
out = await search(q="x")
excerpt = out["results"][0]["excerpt"]
assert excerpt.startswith("OPENING.")
assert excerpt.rstrip().endswith("CLOSING.")
assert "characters omitted" in excerpt
assert out["results"][0]["body_length"] == len(body)
@pytest.mark.asyncio
async def test_a_short_record_arrives_whole_and_unmarked():
"""Fragmenting a 200-character note serves nobody."""
_user_id_ctx.set(7)
fake = fake_note(id=1, title="t", body="all of it")
with patch(
"scribe.mcp.tools.search.semantic_search_notes",
AsyncMock(return_value=[(0.5, fake)]),
):
out = await search(q="x")
assert out["results"][0]["excerpt"] == "all of it"
assert "read_full" not in out["results"][0]
@pytest.mark.asyncio
async def test_fable_search_content_type_filters_at_service_layer():
"""The three historical values keep meaning exactly what they meant.
`note` is the one that could plausibly have been tightened when the
specific kinds arrived — it is BROAD here (any non-task, snippets and
lessons included) while the web facet of the same name is narrow. Pinning
it stops a later tidy-up from silently removing snippets from every caller
that already asks this way (#4250)."""
_user_id_ctx.set(7)
mock_search = AsyncMock(return_value=[])
with patch("scribe.mcp.tools.search.semantic_search_notes", mock_search):
await search(q="x", content_type="task")
assert mock_search.call_args.kwargs["is_task"] is True
assert mock_search.call_args.kwargs.get("task_kind") is None
mock_search.reset_mock()
await search(q="x", content_type="note")
assert mock_search.call_args.kwargs["is_task"] is False
assert mock_search.call_args.kwargs.get("note_type") is None
mock_search.reset_mock()
await search(q="x", content_type="all")
assert mock_search.call_args.kwargs["is_task"] is None
# --- the specific kinds: the engine already supported them (#4250) -----------
@pytest.mark.asyncio
@pytest.mark.parametrize(
"content_type,expected",
[
("snippet", {"is_task": False, "note_type": "snippet"}),
("lesson", {"is_task": False, "note_type": "lesson"}),
("process", {"is_task": False, "note_type": "process"}),
("issue", {"is_task": True, "task_kind": "issue"}),
("spike", {"is_task": True, "task_kind": "spike"}),
("work", {"is_task": True, "task_kind": "work"}),
("plan", {"is_task": True, "task_kind": "plan"}),
],
)
async def test_each_specific_kind_reaches_the_engine_filter_it_names(
content_type, expected
):
"""`note_type` and `task_kind` were parameters of the search all along —
what was missing was a way to ask for them from the agent's door. Asking
for a snippet must narrow to snippets, not merely to non-tasks."""
_user_id_ctx.set(7)
mock_search = AsyncMock(return_value=[])
with patch("scribe.mcp.tools.search.semantic_search_notes", mock_search):
await search(q="x", content_type=content_type)
kwargs = mock_search.call_args.kwargs
for key, value in expected.items():
assert kwargs[key] == value, f"{content_type}: {key}"
def test_the_vocabulary_is_derived_from_the_facet_table_not_recopied():
"""#3161's property, held at this door too: adding a kind to `_FACETS` is
one edit. A hand-kept list here is exactly how the agent's search came to
offer two kinds while the web's offered nine."""
from scribe.services.knowledge import FACET_TYPES, content_type_filters
for facet in FACET_TYPES:
filters = content_type_filters(facet) # raises if a kind is unreachable
assert "is_task" in filters, facet
# The guard can fail: a name absent from the table is refused, so this is
# membership in the table and not "every string works" (#167).
with pytest.raises(ValueError):
content_type_filters("a-kind-that-is-not-in-the-facet-table")
@pytest.mark.asyncio
async def test_an_unknown_content_type_is_refused_rather_than_returning_nothing():
"""An empty result set is a CLAIM — "the corpus holds nothing like this" —
and an agent acts on it by writing the thing it could not find. A typo must
not be able to make that claim, so the door raises with the vocabulary
instead of falling through to a filter that matches no row."""
_user_id_ctx.set(7)
mock_search = AsyncMock(return_value=[])
with patch("scribe.mcp.tools.search.semantic_search_notes", mock_search):
with pytest.raises(ValueError) as err:
await search(q="x", content_type="snippets") # plural typo
mock_search.assert_not_awaited()
message = str(err.value)
assert "snippets" in message
assert "snippet" in message and "lesson" in message # names the valid ones
@pytest.mark.asyncio
async def test_the_docstring_names_every_kind_the_tool_accepts():
"""The docstring IS the agent-facing contract (#2846) — a filter an agent
has not been told about is unreachable however well it is wired."""
from scribe.mcp.tools.search import search as search_tool
from scribe.services.knowledge import FACET_TYPES
doc = search_tool.__doc__ or ""
for facet in FACET_TYPES:
assert f"'{facet}'" in doc, f"{facet} is accepted but never documented"
assert "'rule'" in doc and "'milestone'" in doc and "'all'" in doc
@pytest.mark.asyncio
async def test_an_explicit_search_reaches_a_lesson_from_any_project():
"""The wiring half of milestone 385 step 3.
A lesson records an insight that transfers, so a project filter that hid it
would hide it precisely on the project that has not learned it yet. This is
the EXPLICIT search — the operator asked — so it opts in; the unasked-for
injection arms decide their own budget separately (step 5).
Asserted on the kwarg rather than on results, because what can regress here
is the wiring: the service grew the capability and a call site that never
passes it leaves the whole kind unreachable, with every unit test still
green.
"""
_user_id_ctx.set(7)
mock_search = AsyncMock(return_value=[])
with patch("scribe.mcp.tools.search.semantic_search_notes", mock_search):
await search(q="x", project_id=3)
assert mock_search.call_args.kwargs["include_global_kinds"] is True
@pytest.mark.asyncio
async def test_fable_search_limit_is_clamped():
_user_id_ctx.set(7)
mock_search = AsyncMock(return_value=[])
with patch("scribe.mcp.tools.search.semantic_search_notes", mock_search):
await search(q="x", limit=999)
assert mock_search.call_args.kwargs["limit"] == 50
mock_search.reset_mock()
await search(q="x", limit=0)
assert mock_search.call_args.kwargs["limit"] == 1
@pytest.mark.asyncio
@pytest.mark.parametrize("project_id, scope", [
(5, {"project_id": 5}),
# No project: the explicit question is asked of the whole rulebook.
(0, {"everywhere": True}),
])
async def test_rule_search_scopes_to_the_project_it_is_given(project_id, scope):
"""With a project: global rules plus that project's (milestone 414).
Without one: every rule, because an unscoped "is there a rule about this"
is asking the whole rulebook — unlike a hook, which speaks unasked."""
_user_id_ctx.set(7)
found = AsyncMock(return_value=[])
with patch("scribe.mcp.tools.search.semantic_search_rules", found):
await search(q="release tagging", content_type="rule", project_id=project_id)
kwargs = found.await_args.kwargs
assert {k: kwargs[k] for k in scope} == scope
assert set(kwargs) & {"project_id", "everywhere"} == set(scope)
def test_a_milestone_is_embedded_by_what_it_is_for_then_its_plan():
from scribe.services.embeddings import milestone_document
assert milestone_document("M3", "metadata providers", "## Goal\nx") == (
"M3 — metadata providers", "metadata providers\n\n## Goal\nx")
# A roadmap milestone written with no description is still findable by its plan.
assert milestone_document("M3", None, "the plan") == ("M3", "the plan")
assert milestone_document(None, None, None) == (None, None)
@pytest.mark.asyncio
async def test_milestone_search_is_its_own_shape_and_scopes_to_the_project():
"""milestone 415: 'is there already a plan for this?' has a tool."""
from unittest.mock import MagicMock
_user_id_ctx.set(7)
# `body` is a REAL string, not left to MagicMock's attribute autovivication:
# the result now reads it, and a mock body would make `matched` a mock
# object and `body_length` zero while the assertion still looked green
# (lesson #2833).
ms = MagicMock(id=339, title="M3 — Metadata", description="works, editions",
status="active", project_id=30,
body="Step 4 — the metadata editions table.")
summary = AsyncMock(return_value=[{"id": 339, "total": 0, "completed": 0}])
seen: dict = {}
async def _milestone_search(*_a, **kw):
seen.update(kw)
kw["report"]["best_chunk"] = {
339: {"index": 1, "text": "Step 4 — the metadata editions table."}
}
return [(0.81, ms)]
with patch("scribe.mcp.tools.search.semantic_search_milestones",
_milestone_search), \
patch("scribe.services.milestones.get_project_milestone_summary", summary):
out = await search(q="book metadata", content_type="milestone", project_id=30)
assert seen["project_id"] == 30
assert out["results"] == [{
"id": 339, "title": "M3 — Metadata", "description": "works, editions",
# The plan body stays out; the passage that matched comes along, and
# says which of the two it is (#4243).
"matched": "Step 4 — the metadata editions table.",
"matched_is": "matched_passage",
"body_length": len("Step 4 — the metadata editions table."),
"status": "active", "project_id": 30, "total": 0, "completed": 0,
"similarity": 0.81,
}]