fix(search): show the passage that matched, not the opening of the body (#4243)
CI & Build / Plugin hooks (push) Successful in 10s
CI & Build / Python lint (push) Successful in 3s
CI & Build / TypeScript typecheck (push) Successful in 53s
CI & Build / Python tests (push) Failing after 1m5s
CI & Build / Build & push image (push) Skipped
CI & Build / integration (push) Successful in 46s

Raised by the operator: are we limiting what comes back by character count,
and how do we verify the pertinent part is the part displayed?

We were not. mcp/tools/search.py sent (note.body or "")[:240] — a head cut,
with no marker that anything had been removed, so a 240-character preview of
a 4000-character record was indistinguishable from a complete short one.

The opening is the wrong span. The match is semantic and per chunk, and
semantic_search_notes collapses to best-chunk-per-note — its own comment at
the collapse says "the first appearance of a note is its best chunk". So the
system identified the passage that earned the hit and then discarded it:
select(Note, distance) kept no chunk column. A record could rank first on its
sixth paragraph, be previewed by its first, and be judged irrelevant on a
span the search had already scored lower. That biases against long records,
and it is self-concealing — the caller who does not open it never learns the
preview was misleading.

  - embeddings: chunk_index/chunk_text ride along in the select, and the
    collapse records the winner in report["best_chunk"]. Carried in `report`,
    NOT by widening the return tuple: ten callers unpack (score, note) at
    ~18 sites and nothing would catch the misses (lesson #4207). `report` is
    the side-channel this function already uses for best_available_score.
  - search(): excerpt / excerpt_is / body_length, and read_full when there is
    more. A caller that cannot tell a matched passage from a document opening
    cannot judge whether to look deeper, which is the only decision the field
    supports.

elide() moves to services/text.py so both callers share one copy, and it
keeps BOTH ends with a stated gap — it is the fallback for when nothing
identifies a better span than "all of it", not the goal.

Also fixes a guard that produced a false failure on the previous commit:
test_pull_telemetry checked `"project_id: int = 0" in body.split("\n")[0]`,
which sees only the first line, so wrapping get_task's signature over four
lines made it report a function that does take the project as one that does
not. Parsed with ast now, and proven to still reject an absent or
wrongly-typed parameter rather than being appeased by reflowing the code.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01821k5B3Ysecp9fNYs92Kuy
This commit is contained in:
2026-09-21 08:50:16 -04:00
co-authored by Claude Opus 5
parent 4f2977b848
commit fdc07f2a2b
7 changed files with 245 additions and 50 deletions
+70 -5
View File
@@ -39,23 +39,88 @@ async def test_fable_search_returns_repackaged_results():
r = out["results"][0]
assert r["id"] == 1
assert r["title"] == "kafka rebalance"
assert r["body"] == "HPA details"
# The whole body, because it fits — and named as the opening, not as the
# passage that matched, since this call patched the search and so carries
# no chunk.
assert r["excerpt"] == "HPA details"
assert r["excerpt_is"] == "body_opening"
assert r["body_length"] == len("HPA details")
assert r["is_task"] is False
assert r["tags"] == ["ops"]
assert r["similarity"] == pytest.approx(0.93)
@pytest.mark.asyncio
async def test_fable_search_body_is_truncated_to_240_chars():
async def test_search_shows_the_passage_that_matched_not_the_opening():
"""The point of #4243. A record can rank on its sixth paragraph; showing
its first is showing the caller a span the search already judged less
relevant, and letting them decide from it."""
_user_id_ctx.set(7)
long_body = "x" * 500
fake = fake_note(id=1, title="t", body=long_body)
body = "Chapter one, about nothing. " * 40 + " THE ANSWER IS 04775c3."
fake = fake_note(id=1, title="t", body=body)
async def _search(*a, **kw):
kw["report"]["best_chunk"] = {
1: {"index": 3, "text": "THE ANSWER IS 04775c3."}
}
return [(0.5, fake)]
with patch("scribe.mcp.tools.search.semantic_search_notes", _search):
out = await search(q="answer")
r = out["results"][0]
assert r["excerpt"] == "THE ANSWER IS 04775c3."
assert r["excerpt_is"] == "matched_passage"
assert r["chunk_index"] == 3
# And the caller is told there is more record behind the passage.
assert r["body_length"] == len(body)
assert "read_full" in r
@pytest.mark.asyncio
async def test_a_result_says_whether_its_excerpt_is_the_match_or_the_opening():
"""A caller that cannot tell the two apart cannot judge whether looking
deeper is worth it, which is the only decision this field supports."""
_user_id_ctx.set(7)
fake = fake_note(id=1, title="t", body="short body")
with patch(
"scribe.mcp.tools.search.semantic_search_notes",
AsyncMock(return_value=[(0.5, fake)]),
):
out = await search(q="x")
assert len(out["results"][0]["body"]) == 240
assert out["results"][0]["excerpt_is"] == "body_opening"
@pytest.mark.asyncio
async def test_a_long_excerpt_keeps_both_ends_and_says_how_much_went():
"""The fallback is still an elision, and an elision that drops the tail
drops wherever the conclusion was."""
_user_id_ctx.set(7)
body = "OPENING. " + ("m" * 3000) + " CLOSING."
fake = fake_note(id=1, title="t", body=body)
with patch(
"scribe.mcp.tools.search.semantic_search_notes",
AsyncMock(return_value=[(0.5, fake)]),
):
out = await search(q="x")
excerpt = out["results"][0]["excerpt"]
assert excerpt.startswith("OPENING.")
assert excerpt.rstrip().endswith("CLOSING.")
assert "characters omitted" in excerpt
assert out["results"][0]["body_length"] == len(body)
@pytest.mark.asyncio
async def test_a_short_record_arrives_whole_and_unmarked():
"""Fragmenting a 200-character note serves nobody."""
_user_id_ctx.set(7)
fake = fake_note(id=1, title="t", body="all of it")
with patch(
"scribe.mcp.tools.search.semantic_search_notes",
AsyncMock(return_value=[(0.5, fake)]),
):
out = await search(q="x")
assert out["results"][0]["excerpt"] == "all of it"
assert "read_full" not in out["results"][0]
@pytest.mark.asyncio
+22 -1
View File
@@ -224,13 +224,34 @@ def _pulling_getters():
yield module, name, body
def _takes_reading_project(body: str) -> bool:
"""Does this function declare `project_id: int = 0`?
Parsed, not string-matched against the first line. The first version read
`body.split("\n")[0]`, which sees only as far as the first newline — so
adding a parameter to `get_task` wrapped its signature over four lines and
the guard reported a function that DOES take the project as one that does
not. A guard that fails on formatting is a guard that gets appeased by
reflowing the code it was meant to check.
"""
node = ast.parse(body).body[0]
args = node.args
params = list(args.posonlyargs) + list(args.args) + list(args.kwonlyargs)
for arg in params:
if arg.arg != "project_id":
continue
ann = getattr(arg, "annotation", None)
return isinstance(ann, ast.Name) and ann.id == "int"
return False
def test_every_getter_that_pulls_takes_the_reading_project():
"""Asserted on structure (rule 167): a behavioural test cannot see a
parameter that was never threaded through."""
missing = [
f"{module}.{name}"
for module, name, body in _pulling_getters()
if "project_id: int = 0" not in body.split("\n")[0]
if not _takes_reading_project(body)
]
assert not missing, (
f"these getters record a pull but cannot say where the reader was: "
+14 -13
View File
@@ -20,6 +20,15 @@ from scribe.mcp.tools.tasks import (
elide, get_task, list_tasks, work_log_payload,
)
from scribe.services.notes import brief_row
# Bound at IMPORT time, which is before any fixture runs — so these names
# keep pointing at the real implementations even though conftest's autouse
# _no_task_log_arm replaces the module attributes for every test. Reaching
# them as `task_logs.logs_for_task` would get the stub and test nothing.
from scribe.services.task_logs import (
count_logs_for_task,
log_counts_for_tasks,
logs_for_task,
)
from tests.helpers import fake_task, make_mock_session
pytestmark = pytest.mark.usefixtures("_bind_user")
@@ -304,8 +313,6 @@ async def test_logs_for_task_is_not_filtered_by_who_wrote_the_entry():
"""`list_logs` filters TaskLog.user_id == user_id, which hands a shared
collaborator an empty list that reads as "no work has been done". The work
log belongs to the task."""
from scribe.services import task_logs as svc
session = make_mock_session()
session.execute = AsyncMock(return_value=MagicMock(
scalars=MagicMock(return_value=MagicMock(all=MagicMock(return_value=[])))
@@ -314,7 +321,7 @@ async def test_logs_for_task_is_not_filtered_by_who_wrote_the_entry():
AsyncMock(return_value=True)), \
patch("scribe.services.task_logs.async_session",
MagicMock(return_value=session)):
await svc.logs_for_task(7, 42)
await logs_for_task(7, 42)
stmt = str(session.execute.await_args.args[0])
assert "task_logs.task_id" in stmt
assert "task_logs.user_id" not in stmt
@@ -324,26 +331,22 @@ async def test_logs_for_task_is_not_filtered_by_who_wrote_the_entry():
async def test_logs_for_task_refuses_a_task_the_caller_cannot_read():
"""Unscoped would have been the mirror-image hole: rule #78 is about
routing the question through the access layer, in both directions."""
from scribe.services import task_logs as svc
opened = MagicMock()
with patch("scribe.services.task_logs.can_read_note",
AsyncMock(return_value=False)), \
patch("scribe.services.task_logs.async_session", opened):
out = await svc.logs_for_task(7, 42)
out = await logs_for_task(7, 42)
assert out == []
assert opened.call_count == 0
@pytest.mark.asyncio
async def test_count_for_an_unreadable_task_is_zero_not_a_leak():
from scribe.services import task_logs as svc
opened = MagicMock()
with patch("scribe.services.task_logs.can_read_note",
AsyncMock(return_value=False)), \
patch("scribe.services.task_logs.async_session", opened):
assert await svc.count_logs_for_task(7, 42) == 0
assert await count_logs_for_task(7, 42) == 0
assert opened.call_count == 0
@@ -351,15 +354,13 @@ async def test_count_for_an_unreadable_task_is_zero_not_a_leak():
async def test_log_counts_for_tasks_scopes_by_readability_in_the_same_query():
"""A per-row can_read_note would be the N+1 this function exists to avoid,
so the permission goes in as set membership instead."""
from scribe.services import task_logs as svc
session = make_mock_session()
session.execute = AsyncMock(return_value=MagicMock(
all=MagicMock(return_value=[(1, 3)])
))
with patch("scribe.services.task_logs.async_session",
MagicMock(return_value=session)):
out = await svc.log_counts_for_tasks(7, [1, 2])
out = await log_counts_for_tasks(7, [1, 2])
assert out == {1: 3}
stmt = str(session.execute.await_args.args[0])
assert "notes" in stmt.lower()
@@ -371,5 +372,5 @@ async def test_no_ids_asks_the_database_nothing():
opened = MagicMock()
with patch("scribe.services.task_logs.async_session", opened):
assert await svc.log_counts_for_tasks(7, []) == {}
assert await log_counts_for_tasks(7, []) == {}
assert opened.call_count == 0