From 9909cd245051bb3e61d6cd9b6a820d33e70f09af Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Sat, 3 Oct 2026 23:02:32 -0400 Subject: [PATCH] fix(retrieval): the rules never-opened warning stops sending the reader to a review tool that refuses rules (#4798) #4798 "The rules corpus's surfaced_never_pulled warning sends the reader to menus_to_review, which can only review auto_inject". #4772 gave the warning one remedy for both corpora: "judge a sample with menus_to_review". That is right for notes, where a line carries its passage. For rules it is a dead end: retrieval_review.REVIEWABLE holds only auto_inject, so the tool refuses every rule arm. - The reading is now per corpus (_NEVER_PULLED_READING). For rules, the text says: - the count includes rules that only arrived in a listing; - a rule can rightly be set aside on its trigger alone; - no judged sample exists for the rule arms; - a rule set aside again and again is a trigger to fix (update_rule when_to_apply, then what_might_apply), not a floor. - The tool docstring says the same. - The guard is tied to REVIEWABLE, so it can fail in both directions: if the rules text names menus_to_review while no rule arm is reviewable, or if a rule arm becomes reviewable and the text still says there is no sample. Co-Authored-By: Claude Opus 5.5 --- src/scribe/mcp/tools/search.py | 2 ++ src/scribe/services/retrieval_telemetry.py | 34 ++++++++++++++++++---- tests/test_retrieval_warnings.py | 23 +++++++++++++++ 3 files changed, 54 insertions(+), 5 deletions(-) diff --git a/src/scribe/mcp/tools/search.py b/src/scribe/mcp/tools/search.py index 682d24ea..8487dae4 100644 --- a/src/scribe/mcp/tools/search.py +++ b/src/scribe/mcp/tools/search.py @@ -624,6 +624,8 @@ It is an UPPER BOUND per surface: a pull records the door it came corpus. Not a verdict on them: a line carries its matched passage, so an unopened record may have been unrelated, enough as shown, or already in context. Judge a sample (`menus_to_review`) before acting on it. + For RULES there is no judged sample: a rule set aside again and again + is a trigger to fix (`update_rule(when_to_apply=...)`), not a floor. - `read_and_unacted` — distinct rules OPENED in the window that recorded no outcome, against the ones that did. The failure milestone 419 was opened on, and the worse sibling of `surfaced_never_pulled` above: a diff --git a/src/scribe/services/retrieval_telemetry.py b/src/scribe/services/retrieval_telemetry.py index 60d269d3..59eb89f0 100644 --- a/src/scribe/services/retrieval_telemetry.py +++ b/src/scribe/services/retrieval_telemetry.py @@ -492,6 +492,34 @@ def _num(raw: str, fallback): return fallback +# What "surfaced and never opened" can and cannot mean, per corpus, and where +# to look next. ONE SENTENCE PER CORPUS because the two lines are built +# differently (#4798): a note line carries its matched passage, so an unopened +# note may have done its job, and the judged sample is the instrument that can +# tell. A rule line carries only its trigger and asks to be opened, and no +# judged sample covers the rule arms (`retrieval_review.REVIEWABLE`), so +# sending that reader to `menus_to_review` sent them to a refusal. +_NEVER_PULLED_READING = { + "notes": ( + "That is not a verdict on them: a line carries its matched passage, " + "so an unopened record may have been unrelated, enough as shown, or " + "already in context. Before acting on it — a title, a floor, a " + "budget — judge a sample with `menus_to_review` and read the " + "`judged` block." + ), + "rules": ( + "Not a verdict either, and read differently from notes: the count " + "includes rules that only arrived in a listing (the project " + "handshake, a planning read), and a rule line shows its trigger, so a " + "session can rightly set one aside on the trigger alone. There is no " + "judged sample for the rule arms. A rule set aside again and again " + "where it does not apply is a trigger to fix, not a floor to move — " + "`update_rule(when_to_apply=...)`, then `what_might_apply` with the " + "moment's own words to see where it ranks." + ), +} + + def _warn(code, detail, source=None, **numbers) -> dict: """One finding, carrying the numbers that produced it. @@ -767,11 +795,7 @@ def _compute_warnings(sources: dict, usage: dict, rule_usage: dict, out.append(_warn( "surfaced_never_pulled", f"{never} of {shown} distinct {label} were surfaced in this " - f"window and never opened. That is not a verdict on them: a " - f"line carries its matched passage, so an unopened record may " - f"have been unrelated, enough as shown, or already in context. " - f"Before acting on it — a title, a floor, a budget — judge a " - f"sample with `menus_to_review` and read the `judged` block.", + f"window and never opened. {_NEVER_PULLED_READING[label]}", source=None, corpus=label, surfaced=int(shown), pulled=int(pulled or 0), never_pulled=never, )) diff --git a/tests/test_retrieval_warnings.py b/tests/test_retrieval_warnings.py index 0bb5f5ab..2f2e36bf 100644 --- a/tests/test_retrieval_warnings.py +++ b/tests/test_retrieval_warnings.py @@ -182,6 +182,29 @@ def test_records_shown_and_never_opened_are_reported_per_corpus() -> None: assert found["rules"]["never_pulled"] == 58 +def test_each_corpus_is_sent_only_to_a_tool_that_takes_it() -> None: + """#4798: one remedy for both corpora sent the rules reader to + `menus_to_review`, which refuses every source but the notes menu. Tied to + `REVIEWABLE` itself, so it goes red in either direction: the rules text + naming the tool while no rule arm is reviewable, or a rule arm becoming + reviewable while the text still says there is no sample.""" + from scribe.services.retrieval_review import REVIEWABLE + + ws = warn( + {}, + usage={"distinct_notes_surfaced": 9, "distinct_notes_pulled": 1}, + rule_usage={"distinct_rules_surfaced": 9, "distinct_rules_pulled": 1}, + ) + detail = {w["numbers"]["corpus"]: w["detail"] + for w in ws if w["code"] == "surfaced_never_pulled"} + assert set(detail) == {"notes", "rules"} + assert "auto_inject" in REVIEWABLE + assert "menus_to_review" in detail["notes"] + rule_arms = {"prompt_rule", "pre_tool_rule", "write_path_rule"} + assert ("menus_to_review" in detail["rules"]) == bool(rule_arms & set(REVIEWABLE)) + assert "when_to_apply" in detail["rules"], "the rules text names no next step" + + def test_everything_opened_reports_nothing() -> None: ws = warn({}, usage={"distinct_notes_surfaced": 5, "distinct_notes_pulled": 5}) assert "surfaced_never_pulled" not in codes(ws)