CI & Build / Plugin hooks (push) Successful in 10s
CI & Build / Python lint (push) Successful in 3s
CI & Build / TypeScript typecheck (push) Successful in 53s
CI & Build / Python tests (push) Failing after 1m5s
CI & Build / Build & push image (push) Skipped
CI & Build / integration (push) Successful in 46s
Raised by the operator: are we limiting what comes back by character count,
and how do we verify the pertinent part is the part displayed?
We were not. mcp/tools/search.py sent (note.body or "")[:240] — a head cut,
with no marker that anything had been removed, so a 240-character preview of
a 4000-character record was indistinguishable from a complete short one.
The opening is the wrong span. The match is semantic and per chunk, and
semantic_search_notes collapses to best-chunk-per-note — its own comment at
the collapse says "the first appearance of a note is its best chunk". So the
system identified the passage that earned the hit and then discarded it:
select(Note, distance) kept no chunk column. A record could rank first on its
sixth paragraph, be previewed by its first, and be judged irrelevant on a
span the search had already scored lower. That biases against long records,
and it is self-concealing — the caller who does not open it never learns the
preview was misleading.
- embeddings: chunk_index/chunk_text ride along in the select, and the
collapse records the winner in report["best_chunk"]. Carried in `report`,
NOT by widening the return tuple: ten callers unpack (score, note) at
~18 sites and nothing would catch the misses (lesson #4207). `report` is
the side-channel this function already uses for best_available_score.
- search(): excerpt / excerpt_is / body_length, and read_full when there is
more. A caller that cannot tell a matched passage from a document opening
cannot judge whether to look deeper, which is the only decision the field
supports.
elide() moves to services/text.py so both callers share one copy, and it
keeps BOTH ends with a stated gap — it is the fallback for when nothing
identifies a better span than "all of it", not the goal.
Also fixes a guard that produced a false failure on the previous commit:
test_pull_telemetry checked `"project_id: int = 0" in body.split("\n")[0]`,
which sees only the first line, so wrapping get_task's signature over four
lines made it report a function that does take the project as one that does
not. Parsed with ast now, and proven to still reject an absent or
wrongly-typed parameter rather than being appeased by reflowing the code.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01821k5B3Ysecp9fNYs92Kuy
227 lines
8.8 KiB
Python
227 lines
8.8 KiB
Python
"""search tool — proves the tool pattern (context + service call + dict shape).
|
|
|
|
Service call is mocked; no DB needed."""
|
|
from unittest.mock import AsyncMock, patch
|
|
|
|
import pytest
|
|
|
|
from scribe.mcp._context import _user_id_ctx
|
|
from scribe.mcp.tools.search import search
|
|
from tests.helpers import fake_note
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def _reset_user_ctx():
|
|
"""Each test starts with no MCP context. Tests set it explicitly."""
|
|
token = _user_id_ctx.set(None)
|
|
yield
|
|
_user_id_ctx.reset(token)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_fable_search_raises_without_context():
|
|
with pytest.raises(RuntimeError, match="no MCP user context"):
|
|
await search(q="anything")
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_fable_search_returns_repackaged_results():
|
|
_user_id_ctx.set(7)
|
|
fake = fake_note(id=1, title="kafka rebalance", is_task=False, body="HPA details", tags=["ops"])
|
|
with patch(
|
|
"scribe.mcp.tools.search.semantic_search_notes",
|
|
AsyncMock(return_value=[(0.93, fake)]),
|
|
):
|
|
out = await search(q="kafka")
|
|
|
|
assert out["total"] == 1
|
|
assert len(out["results"]) == 1
|
|
r = out["results"][0]
|
|
assert r["id"] == 1
|
|
assert r["title"] == "kafka rebalance"
|
|
# The whole body, because it fits — and named as the opening, not as the
|
|
# passage that matched, since this call patched the search and so carries
|
|
# no chunk.
|
|
assert r["excerpt"] == "HPA details"
|
|
assert r["excerpt_is"] == "body_opening"
|
|
assert r["body_length"] == len("HPA details")
|
|
assert r["is_task"] is False
|
|
assert r["tags"] == ["ops"]
|
|
assert r["similarity"] == pytest.approx(0.93)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_search_shows_the_passage_that_matched_not_the_opening():
|
|
"""The point of #4243. A record can rank on its sixth paragraph; showing
|
|
its first is showing the caller a span the search already judged less
|
|
relevant, and letting them decide from it."""
|
|
_user_id_ctx.set(7)
|
|
body = "Chapter one, about nothing. " * 40 + " THE ANSWER IS 04775c3."
|
|
fake = fake_note(id=1, title="t", body=body)
|
|
|
|
async def _search(*a, **kw):
|
|
kw["report"]["best_chunk"] = {
|
|
1: {"index": 3, "text": "THE ANSWER IS 04775c3."}
|
|
}
|
|
return [(0.5, fake)]
|
|
|
|
with patch("scribe.mcp.tools.search.semantic_search_notes", _search):
|
|
out = await search(q="answer")
|
|
r = out["results"][0]
|
|
assert r["excerpt"] == "THE ANSWER IS 04775c3."
|
|
assert r["excerpt_is"] == "matched_passage"
|
|
assert r["chunk_index"] == 3
|
|
# And the caller is told there is more record behind the passage.
|
|
assert r["body_length"] == len(body)
|
|
assert "read_full" in r
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_a_result_says_whether_its_excerpt_is_the_match_or_the_opening():
|
|
"""A caller that cannot tell the two apart cannot judge whether looking
|
|
deeper is worth it, which is the only decision this field supports."""
|
|
_user_id_ctx.set(7)
|
|
fake = fake_note(id=1, title="t", body="short body")
|
|
with patch(
|
|
"scribe.mcp.tools.search.semantic_search_notes",
|
|
AsyncMock(return_value=[(0.5, fake)]),
|
|
):
|
|
out = await search(q="x")
|
|
assert out["results"][0]["excerpt_is"] == "body_opening"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_a_long_excerpt_keeps_both_ends_and_says_how_much_went():
|
|
"""The fallback is still an elision, and an elision that drops the tail
|
|
drops wherever the conclusion was."""
|
|
_user_id_ctx.set(7)
|
|
body = "OPENING. " + ("m" * 3000) + " CLOSING."
|
|
fake = fake_note(id=1, title="t", body=body)
|
|
with patch(
|
|
"scribe.mcp.tools.search.semantic_search_notes",
|
|
AsyncMock(return_value=[(0.5, fake)]),
|
|
):
|
|
out = await search(q="x")
|
|
excerpt = out["results"][0]["excerpt"]
|
|
assert excerpt.startswith("OPENING.")
|
|
assert excerpt.rstrip().endswith("CLOSING.")
|
|
assert "characters omitted" in excerpt
|
|
assert out["results"][0]["body_length"] == len(body)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_a_short_record_arrives_whole_and_unmarked():
|
|
"""Fragmenting a 200-character note serves nobody."""
|
|
_user_id_ctx.set(7)
|
|
fake = fake_note(id=1, title="t", body="all of it")
|
|
with patch(
|
|
"scribe.mcp.tools.search.semantic_search_notes",
|
|
AsyncMock(return_value=[(0.5, fake)]),
|
|
):
|
|
out = await search(q="x")
|
|
assert out["results"][0]["excerpt"] == "all of it"
|
|
assert "read_full" not in out["results"][0]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_fable_search_content_type_filters_at_service_layer():
|
|
"""content_type maps to the is_task kwarg passed to the service."""
|
|
_user_id_ctx.set(7)
|
|
mock_search = AsyncMock(return_value=[])
|
|
with patch("scribe.mcp.tools.search.semantic_search_notes", mock_search):
|
|
await search(q="x", content_type="task")
|
|
assert mock_search.call_args.kwargs["is_task"] is True
|
|
|
|
mock_search.reset_mock()
|
|
await search(q="x", content_type="note")
|
|
assert mock_search.call_args.kwargs["is_task"] is False
|
|
|
|
mock_search.reset_mock()
|
|
await search(q="x", content_type="all")
|
|
assert mock_search.call_args.kwargs["is_task"] is None
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_an_explicit_search_reaches_a_lesson_from_any_project():
|
|
"""The wiring half of milestone 385 step 3.
|
|
|
|
A lesson records an insight that transfers, so a project filter that hid it
|
|
would hide it precisely on the project that has not learned it yet. This is
|
|
the EXPLICIT search — the operator asked — so it opts in; the unasked-for
|
|
injection arms decide their own budget separately (step 5).
|
|
|
|
Asserted on the kwarg rather than on results, because what can regress here
|
|
is the wiring: the service grew the capability and a call site that never
|
|
passes it leaves the whole kind unreachable, with every unit test still
|
|
green.
|
|
"""
|
|
_user_id_ctx.set(7)
|
|
mock_search = AsyncMock(return_value=[])
|
|
with patch("scribe.mcp.tools.search.semantic_search_notes", mock_search):
|
|
await search(q="x", project_id=3)
|
|
|
|
assert mock_search.call_args.kwargs["include_global_kinds"] is True
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_fable_search_limit_is_clamped():
|
|
_user_id_ctx.set(7)
|
|
mock_search = AsyncMock(return_value=[])
|
|
with patch("scribe.mcp.tools.search.semantic_search_notes", mock_search):
|
|
await search(q="x", limit=999)
|
|
assert mock_search.call_args.kwargs["limit"] == 50
|
|
|
|
mock_search.reset_mock()
|
|
await search(q="x", limit=0)
|
|
assert mock_search.call_args.kwargs["limit"] == 1
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize("project_id, scope", [
|
|
(5, {"project_id": 5}),
|
|
# No project: the explicit question is asked of the whole rulebook.
|
|
(0, {"everywhere": True}),
|
|
])
|
|
async def test_rule_search_scopes_to_the_project_it_is_given(project_id, scope):
|
|
"""With a project: global rules plus that project's (milestone 414).
|
|
Without one: every rule, because an unscoped "is there a rule about this"
|
|
is asking the whole rulebook — unlike a hook, which speaks unasked."""
|
|
_user_id_ctx.set(7)
|
|
found = AsyncMock(return_value=[])
|
|
with patch("scribe.mcp.tools.search.semantic_search_rules", found):
|
|
await search(q="release tagging", content_type="rule", project_id=project_id)
|
|
kwargs = found.await_args.kwargs
|
|
assert {k: kwargs[k] for k in scope} == scope
|
|
assert set(kwargs) & {"project_id", "everywhere"} == set(scope)
|
|
|
|
|
|
def test_a_milestone_is_embedded_by_what_it_is_for_then_its_plan():
|
|
from scribe.services.embeddings import milestone_document
|
|
|
|
assert milestone_document("M3", "metadata providers", "## Goal\nx") == (
|
|
"M3 — metadata providers", "metadata providers\n\n## Goal\nx")
|
|
# A roadmap milestone written with no description is still findable by its plan.
|
|
assert milestone_document("M3", None, "the plan") == ("M3", "the plan")
|
|
assert milestone_document(None, None, None) == (None, None)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_milestone_search_is_its_own_shape_and_scopes_to_the_project():
|
|
"""milestone 415: 'is there already a plan for this?' has a tool."""
|
|
from unittest.mock import MagicMock
|
|
|
|
_user_id_ctx.set(7)
|
|
ms = MagicMock(id=339, title="M3 — Metadata", description="works, editions",
|
|
status="active", project_id=30)
|
|
found = AsyncMock(return_value=[(0.81, ms)])
|
|
summary = AsyncMock(return_value=[{"id": 339, "total": 0, "completed": 0}])
|
|
with patch("scribe.mcp.tools.search.semantic_search_milestones", found), \
|
|
patch("scribe.services.milestones.get_project_milestone_summary", summary):
|
|
out = await search(q="book metadata", content_type="milestone", project_id=30)
|
|
assert found.await_args.kwargs["project_id"] == 30
|
|
assert out["results"] == [{
|
|
"id": 339, "title": "M3 — Metadata", "description": "works, editions",
|
|
"status": "active", "project_id": 30, "total": 0, "completed": 0,
|
|
"similarity": 0.81,
|
|
}]
|