Files
FabledScribe/tests/test_note_usage.py
T
bvandeusen 2d1e26f38f
CI & Build / Python lint (push) Successful in 5s
CI & Build / Plugin hooks (push) Successful in 8s
CI & Build / integration (push) Successful in 16s
CI & Build / TypeScript typecheck (push) Successful in 32s
CI & Build / Python tests (push) Successful in 47s
CI & Build / Build & push image (push) Successful in 25s
feat(telemetry): ambient surfacings count, apart — enter_project and the skill sync emit
#2477, option (a) as decided, with the readout changed in the same commit.

## The two silent surfaces

enter_project returns open tasks + recent notes on every project entry —
probably the largest surfacing by volume — and emitted nothing, so the pulls
it caused floated unattributed and the surfaced:pulled ratio ran against a
denominator missing its biggest contributor. Now source "enter_project".

build_process_manifest installs every reachable Process as an auto-surfacing
skill on the operator's machine — its own docstring calls it the most
consequential passive surface Scribe has — and emitted nothing, so a Process
matched on every relevant turn and never opened was indistinguishable from
one never installed. Now source "process_skill_sync": the honest event is
"installed", which is a surfacing in effect since the description sits in
front of the model each session.

## The readout, same commit — the condition option (a) carried

Both surfaces are AMBIENT: top-N-by-recency and install-everything are not
ranked choices. Pooling them into surfaced_count would make a note's number
dominated by "recently updated in a project you opened", and dead-weight
detection would read that as popularity — the wrong number read confidently,
which is the corrupts-data tier the survey ranked above everything else.

So usage_for_notes splits: surfaced_count stays RANKED-ONLY (every existing
consumer's reading — "surfaced often, never pulled → dead weight" — keeps
meaning what it meant), and ambient_count is new. Classified in SQL via a
CASE on AMBIENT_SOURCES so the group count stays three rows per note, not one
per distinct source. Pulls stay pooled: "did anyone ever open this?" does not
depend on how it was found.

#1038 and #2085 read agent pulls and ranked surfacings; both are unaffected
by ambient volume, which is the point.

Refs #2477
2026-08-08 22:39:13 -04:00

252 lines
9.9 KiB
Python

"""Tests for the usage signal (#2085) — was a surfaced record ever pulled?
Covers the payload shaping and the two contracts that make this safe to leave in
the hot path: telemetry never raises, and telemetry never blocks. Plus the thing
the feature exists for — that the write-path PLACE arm is now recorded, since
before this it surfaced snippets while leaving no trace anywhere.
"""
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
@pytest.fixture(autouse=True)
def _no_supersession():
"""The auto-inject menu now asks which of its lines are superseded (#278).
That is a real database call on a path these tests exercise without one.
Stubbed to "nothing superseded" — the ordinary state — rather than hidden
behind a try/except in the product, which would make the code lie about
what it does. The label's own behaviour is covered in
tests/test_supersession_ranking.py.
"""
with patch("scribe.services.plugin_context.superseded_ids",
AsyncMock(return_value=set())):
yield
from scribe.services import note_usage
from scribe.services.note_usage import (
empty_usage,
record_pulled,
record_surfaced,
usage_for_notes,
)
def _note(nid, title, user_id=1):
n = MagicMock()
n.id, n.title, n.user_id = nid, title, user_id
n.note_type = "snippet"
# See the same note in test_write_path_trigger: an auto-created mock on
# `.data` is truthy and would leak into a rendered menu line (#2244).
n.data = None
# And on `.is_task`, which the kind marker reads FIRST — truthy there makes
# every one of these snippets read as a task (#2246).
n.is_task, n.task_kind = False, "work"
return n
# --- recording ------------------------------------------------------------
async def test_record_surfaced_writes_one_row_per_note():
with patch.object(note_usage, "_schedule") as sched:
record_surfaced(user_id=7, note_ids=[11, 12], source="auto_inject")
rows = sched.call_args[0][0]
assert [r["note_id"] for r in rows] == [11, 12]
assert {r["event"] for r in rows} == {"surfaced"}
assert {r["source"] for r in rows} == {"auto_inject"}
assert {r["user_id"] for r in rows} == {7}
async def test_record_pulled_writes_a_single_row():
with patch.object(note_usage, "_schedule") as sched:
record_pulled(user_id=7, note_id=11, source="mcp_get_snippet")
rows = sched.call_args[0][0]
assert rows == [
{"user_id": 7, "note_id": 11, "event": "pulled", "source": "mcp_get_snippet"}
]
async def test_empty_menu_schedules_nothing():
"""No notes surfaced is not an event — it must not cost a write."""
with patch.object(note_usage, "_insert_events") as ins:
record_surfaced(user_id=1, note_ids=[], source="auto_inject")
ins.assert_not_called()
async def test_recording_never_raises_on_bad_input():
"""Telemetry sits in the hot path of every retrieval. A malformed id must
cost a data point, never the operator's request."""
with patch.object(note_usage, "_schedule"):
record_surfaced(user_id=1, note_ids=["not-an-int"], source="auto_inject")
record_pulled(user_id=1, note_id=None, source="mcp_get_note")
async def test_recording_without_an_event_loop_is_skipped_not_raised():
"""Called from a sync context outside the app (a script, a test helper),
there is no loop to schedule on. Skip rather than blow up."""
with patch.object(
note_usage.asyncio, "get_running_loop", side_effect=RuntimeError
):
record_pulled(user_id=1, note_id=5, source="mcp_get_note")
# --- readout --------------------------------------------------------------
async def test_usage_for_notes_zero_fills_every_requested_id():
"""The caller renders this shape unconditionally, so a note with no events
must come back as zeroes, not as a missing key."""
session = MagicMock()
session.execute = AsyncMock(return_value=MagicMock(all=MagicMock(return_value=[])))
ctx = MagicMock()
ctx.__aenter__ = AsyncMock(return_value=session)
ctx.__aexit__ = AsyncMock(return_value=False)
with patch.object(note_usage, "async_session", return_value=ctx):
out = await usage_for_notes([3, 4])
assert out == {3: empty_usage(), 4: empty_usage()}
async def test_usage_for_notes_splits_counts_by_event():
from datetime import datetime, timezone
ts = datetime(2026, 7, 28, tzinfo=timezone.utc)
# Rows are (note_id, event, count, last_at, ambient) since #2477 split the
# readout. Ranked and ambient surfacings arrive as separate groups.
rows = [
(3, "surfaced", 9, ts, False),
(3, "surfaced", 40, ts, True),
(3, "pulled", 2, ts, False),
]
session = MagicMock()
session.execute = AsyncMock(
return_value=MagicMock(all=MagicMock(return_value=rows))
)
ctx = MagicMock()
ctx.__aenter__ = AsyncMock(return_value=session)
ctx.__aexit__ = AsyncMock(return_value=False)
with patch.object(note_usage, "async_session", return_value=ctx):
out = await usage_for_notes([3])
# The dead-weight reading ("surfaced often, never pulled") is only valid
# over surfacings that were CHOICES. 40 enter_project appearances must not
# make a record look popular — they sit in ambient_count (#2477).
assert out[3]["surfaced_count"] == 9
assert out[3]["ambient_count"] == 40
assert out[3]["pull_count"] == 2
assert out[3]["last_pulled_at"] == ts.isoformat()
async def test_usage_readout_failure_degrades_to_zeroes():
"""A telemetry readout must not be able to break the list it decorates."""
with patch.object(note_usage, "async_session", side_effect=RuntimeError("boom")):
out = await usage_for_notes([3])
assert out == {3: empty_usage()}
async def test_no_ids_short_circuits_without_a_query():
with patch.object(note_usage, "async_session") as sess:
assert await usage_for_notes([]) == {}
sess.assert_not_called()
# --- the gap this closes --------------------------------------------------
@pytest.mark.parametrize(
"marker, expected_source",
[("here", "write_path_place"), ("nearby", "write_path_place")],
)
async def test_write_path_place_arm_is_recorded(marker, expected_source):
"""The place arm carries no score, so it has no home in retrieval_logs — it
surfaced snippets while leaving no trace anywhere. That was the blocker
#2082 recorded against this task; this is the assertion that it's closed."""
from scribe.services import plugin_context
here = [{"id": 42, "title": "helper", "user_id": 1, "note_type": "snippet"}]
with (
patch.object(
plugin_context,
"get_writepath_config",
AsyncMock(return_value={"enabled": True, "threshold": 0.55, "top_k": 3}),
),
patch.object(
plugin_context.snippets_svc,
"list_snippets",
AsyncMock(side_effect=[(here, 1), ([], 0)]),
),
patch.object(
plugin_context, "semantic_search_notes", AsyncMock(return_value=[])
),
patch.object(plugin_context, "owner_names_for", AsyncMock(return_value={})),
patch.object(plugin_context, "record_retrieval"),
patch.object(plugin_context, "record_surfaced") as surfaced,
):
out = await plugin_context.build_write_path_hint(
1, "src/a.py", code="def f(): pass"
)
assert out["note_ids"] == [42]
sources = {c.kwargs["source"] for c in surfaced.call_args_list}
assert expected_source in sources
async def test_auto_inject_records_what_survived_the_margin_gate():
"""Not what the ranker returned — retrieval_logs already holds that. These
two numbers must not silently mean different things per surface."""
from scribe.services import plugin_context
hits = [(0.90, _note(1, "kept")), (0.40, _note(2, "cut by the margin gate"))]
with (
patch.object(
plugin_context,
"get_autoinject_config",
AsyncMock(return_value={"enabled": True, "threshold": 0.3, "top_k": 5}),
),
patch.object(
plugin_context, "semantic_search_notes", AsyncMock(return_value=hits)
),
patch.object(plugin_context, "owner_names_for", AsyncMock(return_value={})),
patch.object(plugin_context, "record_retrieval"),
patch.object(plugin_context, "record_surfaced") as surfaced,
):
await plugin_context.build_autoinject_hint(1, "a query")
assert surfaced.call_args.kwargs["note_ids"] == [1]
assert surfaced.call_args.kwargs["source"] == "auto_inject"
def test_every_getter_that_can_be_surfaced_also_records_a_pull():
"""Rule #33 contract check — and the one that would have caught #2245.
`surfaced` and `pulled` only mean something as a PAIR: the rate between them
is what #1038 and #2085 gate on. That pair is only closed if the tool which
OPENS a record reports it. `get_note` did, `get_snippet` did, `get_task` did
NOT — and auto-inject ranks kind-blind over a corpus that is overwhelmingly
tasks and issues, so the gap sat exactly where the volume is: every surfaced
task counted as never-pulled, dragging measured pull-through toward zero for
the menu's own dominant kind.
Asserted by source inspection rather than by calling the tools, because the
failure is a MISSING call — which no behavioural test of the tool's return
value can see.
"""
import inspect
from scribe.mcp.tools import notes as notes_tools
from scribe.mcp.tools import snippets as snippet_tools
from scribe.mcp.tools import tasks as task_tools
getters = (
(notes_tools, "get_note"),
(task_tools, "get_task"),
(snippet_tools, "get_snippet"),
)
for module, name in getters:
src = inspect.getsource(getattr(module, name))
assert "record_pulled(" in src, (
f"{name} can be surfaced in an auto-inject menu but records no pull — "
"its pull-through rate will read as zero regardless of real usage"
)