Rules never-opened advice per corpus (#4798); a lesson's name is one line (#4797) #200

Merged
bvandeusen merged 2 commits from dev into main 2026-10-03 23:11:19 -04:00
3 changed files with 54 additions and 5 deletions
Showing only changes of commit 9909cd2450 - Show all commits
+2
View File
@@ -624,6 +624,8 @@ It is an UPPER BOUND per surface: a pull records the door it came
corpus. Not a verdict on them: a line carries its matched passage, so corpus. Not a verdict on them: a line carries its matched passage, so
an unopened record may have been unrelated, enough as shown, or already an unopened record may have been unrelated, enough as shown, or already
in context. Judge a sample (`menus_to_review`) before acting on it. in context. Judge a sample (`menus_to_review`) before acting on it.
For RULES there is no judged sample: a rule set aside again and again
is a trigger to fix (`update_rule(when_to_apply=...)`), not a floor.
- `read_and_unacted` — distinct rules OPENED in the window that recorded - `read_and_unacted` — distinct rules OPENED in the window that recorded
no outcome, against the ones that did. The failure milestone 419 was no outcome, against the ones that did. The failure milestone 419 was
opened on, and the worse sibling of `surfaced_never_pulled` above: a opened on, and the worse sibling of `surfaced_never_pulled` above: a
+29 -5
View File
@@ -492,6 +492,34 @@ def _num(raw: str, fallback):
return fallback return fallback
# What "surfaced and never opened" can and cannot mean, per corpus, and where
# to look next. ONE SENTENCE PER CORPUS because the two lines are built
# differently (#4798): a note line carries its matched passage, so an unopened
# note may have done its job, and the judged sample is the instrument that can
# tell. A rule line carries only its trigger and asks to be opened, and no
# judged sample covers the rule arms (`retrieval_review.REVIEWABLE`), so
# sending that reader to `menus_to_review` sent them to a refusal.
_NEVER_PULLED_READING = {
"notes": (
"That is not a verdict on them: a line carries its matched passage, "
"so an unopened record may have been unrelated, enough as shown, or "
"already in context. Before acting on it — a title, a floor, a "
"budget — judge a sample with `menus_to_review` and read the "
"`judged` block."
),
"rules": (
"Not a verdict either, and read differently from notes: the count "
"includes rules that only arrived in a listing (the project "
"handshake, a planning read), and a rule line shows its trigger, so a "
"session can rightly set one aside on the trigger alone. There is no "
"judged sample for the rule arms. A rule set aside again and again "
"where it does not apply is a trigger to fix, not a floor to move — "
"`update_rule(when_to_apply=...)`, then `what_might_apply` with the "
"moment's own words to see where it ranks."
),
}
def _warn(code, detail, source=None, **numbers) -> dict: def _warn(code, detail, source=None, **numbers) -> dict:
"""One finding, carrying the numbers that produced it. """One finding, carrying the numbers that produced it.
@@ -767,11 +795,7 @@ def _compute_warnings(sources: dict, usage: dict, rule_usage: dict,
out.append(_warn( out.append(_warn(
"surfaced_never_pulled", "surfaced_never_pulled",
f"{never} of {shown} distinct {label} were surfaced in this " f"{never} of {shown} distinct {label} were surfaced in this "
f"window and never opened. That is not a verdict on them: a " f"window and never opened. {_NEVER_PULLED_READING[label]}",
f"line carries its matched passage, so an unopened record may "
f"have been unrelated, enough as shown, or already in context. "
f"Before acting on it — a title, a floor, a budget — judge a "
f"sample with `menus_to_review` and read the `judged` block.",
source=None, corpus=label, source=None, corpus=label,
surfaced=int(shown), pulled=int(pulled or 0), never_pulled=never, surfaced=int(shown), pulled=int(pulled or 0), never_pulled=never,
)) ))
+23
View File
@@ -182,6 +182,29 @@ def test_records_shown_and_never_opened_are_reported_per_corpus() -> None:
assert found["rules"]["never_pulled"] == 58 assert found["rules"]["never_pulled"] == 58
def test_each_corpus_is_sent_only_to_a_tool_that_takes_it() -> None:
"""#4798: one remedy for both corpora sent the rules reader to
`menus_to_review`, which refuses every source but the notes menu. Tied to
`REVIEWABLE` itself, so it goes red in either direction: the rules text
naming the tool while no rule arm is reviewable, or a rule arm becoming
reviewable while the text still says there is no sample."""
from scribe.services.retrieval_review import REVIEWABLE
ws = warn(
{},
usage={"distinct_notes_surfaced": 9, "distinct_notes_pulled": 1},
rule_usage={"distinct_rules_surfaced": 9, "distinct_rules_pulled": 1},
)
detail = {w["numbers"]["corpus"]: w["detail"]
for w in ws if w["code"] == "surfaced_never_pulled"}
assert set(detail) == {"notes", "rules"}
assert "auto_inject" in REVIEWABLE
assert "menus_to_review" in detail["notes"]
rule_arms = {"prompt_rule", "pre_tool_rule", "write_path_rule"}
assert ("menus_to_review" in detail["rules"]) == bool(rule_arms & set(REVIEWABLE))
assert "when_to_apply" in detail["rules"], "the rules text names no next step"
def test_everything_opened_reports_nothing() -> None: def test_everything_opened_reports_nothing() -> None:
ws = warn({}, usage={"distinct_notes_surfaced": 5, "distinct_notes_pulled": 5}) ws = warn({}, usage={"distinct_notes_surfaced": 5, "distinct_notes_pulled": 5})
assert "surfaced_never_pulled" not in codes(ws) assert "surfaced_never_pulled" not in codes(ws)