diff --git a/src/scribe/mcp/tools/search.py b/src/scribe/mcp/tools/search.py index 682d24ea..8487dae4 100644 --- a/src/scribe/mcp/tools/search.py +++ b/src/scribe/mcp/tools/search.py @@ -624,6 +624,8 @@ It is an UPPER BOUND per surface: a pull records the door it came corpus. Not a verdict on them: a line carries its matched passage, so an unopened record may have been unrelated, enough as shown, or already in context. Judge a sample (`menus_to_review`) before acting on it. + For RULES there is no judged sample: a rule set aside again and again + is a trigger to fix (`update_rule(when_to_apply=...)`), not a floor. - `read_and_unacted` — distinct rules OPENED in the window that recorded no outcome, against the ones that did. The failure milestone 419 was opened on, and the worse sibling of `surfaced_never_pulled` above: a diff --git a/src/scribe/services/retrieval_telemetry.py b/src/scribe/services/retrieval_telemetry.py index 60d269d3..59eb89f0 100644 --- a/src/scribe/services/retrieval_telemetry.py +++ b/src/scribe/services/retrieval_telemetry.py @@ -492,6 +492,34 @@ def _num(raw: str, fallback): return fallback +# What "surfaced and never opened" can and cannot mean, per corpus, and where +# to look next. ONE SENTENCE PER CORPUS because the two lines are built +# differently (#4798): a note line carries its matched passage, so an unopened +# note may have done its job, and the judged sample is the instrument that can +# tell. A rule line carries only its trigger and asks to be opened, and no +# judged sample covers the rule arms (`retrieval_review.REVIEWABLE`), so +# sending that reader to `menus_to_review` sent them to a refusal. +_NEVER_PULLED_READING = { + "notes": ( + "That is not a verdict on them: a line carries its matched passage, " + "so an unopened record may have been unrelated, enough as shown, or " + "already in context. Before acting on it — a title, a floor, a " + "budget — judge a sample with `menus_to_review` and read the " + "`judged` block." + ), + "rules": ( + "Not a verdict either, and read differently from notes: the count " + "includes rules that only arrived in a listing (the project " + "handshake, a planning read), and a rule line shows its trigger, so a " + "session can rightly set one aside on the trigger alone. There is no " + "judged sample for the rule arms. A rule set aside again and again " + "where it does not apply is a trigger to fix, not a floor to move — " + "`update_rule(when_to_apply=...)`, then `what_might_apply` with the " + "moment's own words to see where it ranks." + ), +} + + def _warn(code, detail, source=None, **numbers) -> dict: """One finding, carrying the numbers that produced it. @@ -767,11 +795,7 @@ def _compute_warnings(sources: dict, usage: dict, rule_usage: dict, out.append(_warn( "surfaced_never_pulled", f"{never} of {shown} distinct {label} were surfaced in this " - f"window and never opened. That is not a verdict on them: a " - f"line carries its matched passage, so an unopened record may " - f"have been unrelated, enough as shown, or already in context. " - f"Before acting on it — a title, a floor, a budget — judge a " - f"sample with `menus_to_review` and read the `judged` block.", + f"window and never opened. {_NEVER_PULLED_READING[label]}", source=None, corpus=label, surfaced=int(shown), pulled=int(pulled or 0), never_pulled=never, )) diff --git a/tests/test_retrieval_warnings.py b/tests/test_retrieval_warnings.py index 0bb5f5ab..2f2e36bf 100644 --- a/tests/test_retrieval_warnings.py +++ b/tests/test_retrieval_warnings.py @@ -182,6 +182,29 @@ def test_records_shown_and_never_opened_are_reported_per_corpus() -> None: assert found["rules"]["never_pulled"] == 58 +def test_each_corpus_is_sent_only_to_a_tool_that_takes_it() -> None: + """#4798: one remedy for both corpora sent the rules reader to + `menus_to_review`, which refuses every source but the notes menu. Tied to + `REVIEWABLE` itself, so it goes red in either direction: the rules text + naming the tool while no rule arm is reviewable, or a rule arm becoming + reviewable while the text still says there is no sample.""" + from scribe.services.retrieval_review import REVIEWABLE + + ws = warn( + {}, + usage={"distinct_notes_surfaced": 9, "distinct_notes_pulled": 1}, + rule_usage={"distinct_rules_surfaced": 9, "distinct_rules_pulled": 1}, + ) + detail = {w["numbers"]["corpus"]: w["detail"] + for w in ws if w["code"] == "surfaced_never_pulled"} + assert set(detail) == {"notes", "rules"} + assert "auto_inject" in REVIEWABLE + assert "menus_to_review" in detail["notes"] + rule_arms = {"prompt_rule", "pre_tool_rule", "write_path_rule"} + assert ("menus_to_review" in detail["rules"]) == bool(rule_arms & set(REVIEWABLE)) + assert "when_to_apply" in detail["rules"], "the rules text names no next step" + + def test_everything_opened_reports_nothing() -> None: ws = warn({}, usage={"distinct_notes_surfaced": 5, "distinct_notes_pulled": 5}) assert "surfaced_never_pulled" not in codes(ws)