CI & Build / Python lint (push) Successful in 4s
CI & Build / Plugin hooks (push) Successful in 15s
CI & Build / TypeScript typecheck (push) Successful in 54s
CI & Build / integration (push) Successful in 1m7s
CI & Build / Python tests (push) Successful in 1m55s
CI & Build / Build & push image (push) Successful in 26s
#4798 "The rules corpus's surfaced_never_pulled warning sends the reader to menus_to_review, which can only review auto_inject". #4772 gave the warning one remedy for both corpora: "judge a sample with menus_to_review". That is right for notes, where a line carries its passage. For rules it is a dead end: retrieval_review.REVIEWABLE holds only auto_inject, so the tool refuses every rule arm. - The reading is now per corpus (_NEVER_PULLED_READING). For rules, the text says: - the count includes rules that only arrived in a listing; - a rule can rightly be set aside on its trigger alone; - no judged sample exists for the rule arms; - a rule set aside again and again is a trigger to fix (update_rule when_to_apply, then what_might_apply), not a floor. - The tool docstring says the same. - The guard is tied to REVIEWABLE, so it can fail in both directions: if the rules text names menus_to_review while no rule arm is reviewable, or if a rule arm becomes reviewable and the text still says there is no sample. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
673 lines
30 KiB
Python
673 lines
30 KiB
Python
"""`retrieval_telemetry` says what is wrong, rather than implying it (#3431).
|
|
|
|
WHY THIS IS TESTED AS ARITHMETIC AND NOT THROUGH THE DATABASE. The warnings
|
|
are a pure function of the blocks the readout already built — that is the
|
|
design, so a verdict can never disagree with the numbers printed beside it —
|
|
and the risk in them is not whether rows load. It is whether a rule fires on
|
|
the wrong shape. Every case below is a shape that once produced, or would
|
|
produce, a wrong reading:
|
|
|
|
* an arm that never declines, which is either a missing floor or a missing
|
|
LOG (#3497 — the rule arms recorded only their hits, so their decline
|
|
count was structurally zero and the obvious warning would have sent a
|
|
reader to move a threshold that was never involved);
|
|
* a search that never declines, which is a search working correctly;
|
|
* a quiet window, where firing every check would describe an empty database
|
|
rather than a broken one (rule 115);
|
|
* a floor the band is sitting on, which is the case where tuning changes
|
|
volume while looking like it changes quality.
|
|
|
|
The ε and N boundaries are tested from BOTH sides. A threshold asserted only
|
|
where it fires is half-tested: the expensive failure here is a false positive,
|
|
because a readout that cries wolf is one nobody reads.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import pytest
|
|
|
|
from scribe.services.retrieval_telemetry import (
|
|
WARN_FLOOR_EPSILON_DEFAULT, WARN_MIN_CALLS_DEFAULT,
|
|
_compute_warnings, _num, _silent_surfaces,
|
|
)
|
|
|
|
N = WARN_MIN_CALLS_DEFAULT
|
|
EPS = WARN_FLOOR_EPSILON_DEFAULT
|
|
|
|
|
|
def src(**kw) -> dict:
|
|
"""One source block, shaped like the readout builds it."""
|
|
b = {
|
|
"calls": kw.pop("calls", 0),
|
|
"zero_result_calls": kw.pop("zero_result_calls", 0),
|
|
"top_score": {"p10": kw.pop("p10", None), "p50": None, "p90": None,
|
|
"min": None, "max": None},
|
|
"avg_result_count": None,
|
|
"p90_duration_ms": kw.pop("p90_duration_ms", 12.0),
|
|
}
|
|
b.update(kw)
|
|
return b
|
|
|
|
|
|
def warn(sources, usage=None, rule_usage=None, floors=None,
|
|
min_calls=N, epsilon=EPS, floor_moves=None, floor_gaps=None) -> list[dict]:
|
|
return _compute_warnings(
|
|
sources, usage or {}, rule_usage or {}, floors or {}, min_calls, epsilon,
|
|
floor_moves or {}, floor_gaps or {},
|
|
)
|
|
|
|
|
|
def codes(ws, source=None) -> set[str]:
|
|
return {w["code"] for w in ws if source is None or w["source"] == source}
|
|
|
|
|
|
# ── cannot_decline ────────────────────────────────────────────────────────
|
|
|
|
def test_an_unbidden_arm_that_never_declines_is_flagged() -> None:
|
|
ws = warn({"auto_inject": src(calls=300, zero_result_calls=0)})
|
|
assert "cannot_decline" in codes(ws, "auto_inject")
|
|
|
|
|
|
def test_the_warning_carries_the_numbers_that_produced_it() -> None:
|
|
"""Not "check auto_inject" — that is an instruction to redo the analysis."""
|
|
w = next(w for w in warn({"auto_inject": src(calls=300)})
|
|
if w["code"] == "cannot_decline")
|
|
assert w["numbers"] == {"calls": 300, "zero_result_calls": 0}
|
|
assert "300" in w["detail"]
|
|
|
|
|
|
def test_a_quiet_window_flags_nothing() -> None:
|
|
"""Three calls returning nothing is an afternoon, not a defect (rule 115)."""
|
|
assert codes(warn({"auto_inject": src(calls=3, zero_result_calls=0)})) == set()
|
|
|
|
|
|
def test_the_call_count_boundary_holds_on_both_sides() -> None:
|
|
assert "cannot_decline" not in codes(warn({"auto_inject": src(calls=N - 1)}))
|
|
assert "cannot_decline" in codes(warn({"auto_inject": src(calls=N)}))
|
|
|
|
|
|
def test_an_asked_surface_is_never_flagged_for_not_declining() -> None:
|
|
"""A search returning a list every time is a search doing its job."""
|
|
for asked in ("mcp_search", "wide_net", "rest_search", "browse_search"):
|
|
assert "cannot_decline" not in codes(
|
|
warn({asked: src(calls=500, zero_result_calls=0)}), asked)
|
|
|
|
|
|
def test_an_arm_that_does_decline_is_not_flagged() -> None:
|
|
ws = warn({"auto_inject": src(calls=300, zero_result_calls=1)})
|
|
assert "cannot_decline" not in codes(ws)
|
|
|
|
|
|
def test_an_arm_not_known_to_log_unconditionally_is_exempt(monkeypatch) -> None:
|
|
"""#3497: a structurally-zero decline count is a logging bug, not a floor.
|
|
|
|
Flagging it would report a ranking problem and send the reader to a
|
|
threshold that was never involved.
|
|
"""
|
|
from dataclasses import replace
|
|
|
|
from scribe.services import retrieval_registry as reg
|
|
|
|
patched = dict(reg.POINTS)
|
|
patched["auto_inject"] = replace(patched["auto_inject"],
|
|
logs_unconditionally=False)
|
|
monkeypatch.setattr(reg, "POINTS", patched)
|
|
assert "cannot_decline" not in codes(
|
|
warn({"auto_inject": src(calls=300, zero_result_calls=0)}))
|
|
|
|
|
|
# ── band_hugs_floor ───────────────────────────────────────────────────────
|
|
|
|
def test_a_band_sitting_on_its_floor_is_flagged() -> None:
|
|
ws = warn({"auto_inject": src(calls=100, zero_result_calls=5, p10=0.705)},
|
|
floors={"auto_inject": 0.70})
|
|
assert "band_hugs_floor" in codes(ws, "auto_inject")
|
|
|
|
|
|
def test_a_band_clear_of_its_floor_is_not() -> None:
|
|
ws = warn({"auto_inject": src(calls=100, zero_result_calls=5, p10=0.80)},
|
|
floors={"auto_inject": 0.70})
|
|
assert "band_hugs_floor" not in codes(ws)
|
|
|
|
|
|
@pytest.mark.parametrize("gap,flagged", [
|
|
(EPS / 2, True), # inside
|
|
(EPS, False), # exactly at the boundary is NOT "hugging"
|
|
(EPS * 2, False), # clear
|
|
])
|
|
def test_the_epsilon_boundary_holds_on_both_sides(gap, flagged) -> None:
|
|
ws = warn({"auto_inject": src(calls=100, zero_result_calls=5,
|
|
p10=round(0.70 + gap, 6))},
|
|
floors={"auto_inject": 0.70})
|
|
assert ("band_hugs_floor" in codes(ws)) is flagged
|
|
|
|
|
|
def test_an_arm_with_no_readable_floor_is_not_guessed_at() -> None:
|
|
"""Reserved slots borrow a parent's floor; naming it here would attribute
|
|
the parent's setting to a child that has no dial of its own."""
|
|
ws = warn({"preference_slot": src(calls=100, zero_result_calls=5, p10=0.7001)},
|
|
floors={})
|
|
assert "band_hugs_floor" not in codes(ws)
|
|
|
|
|
|
# ── no_duration ───────────────────────────────────────────────────────────
|
|
|
|
def test_rows_without_timings_are_a_logging_gap() -> None:
|
|
ws = warn({"auto_inject": src(calls=1, zero_result_calls=1,
|
|
p90_duration_ms=None)})
|
|
assert "no_duration" in codes(ws)
|
|
|
|
|
|
def test_one_untimed_call_is_already_the_defect() -> None:
|
|
"""No minimum on this one — unlike the others, it is not about volume."""
|
|
ws = warn({"auto_inject": src(calls=1, p90_duration_ms=None)})
|
|
assert "no_duration" in codes(ws)
|
|
assert "cannot_decline" not in codes(ws), "volume rules still need volume"
|
|
|
|
|
|
def test_a_source_with_no_calls_reports_no_timing_gap() -> None:
|
|
assert "no_duration" not in codes(warn({"auto_inject": src(calls=0)}))
|
|
|
|
|
|
# ── surfaced_never_pulled ─────────────────────────────────────────────────
|
|
|
|
def test_records_shown_and_never_opened_are_reported_per_corpus() -> None:
|
|
ws = warn(
|
|
{},
|
|
usage={"distinct_notes_surfaced": 171, "distinct_notes_pulled": 40},
|
|
rule_usage={"distinct_rules_surfaced": 69, "distinct_rules_pulled": 11},
|
|
)
|
|
found = {w["numbers"]["corpus"]: w["numbers"]
|
|
for w in ws if w["code"] == "surfaced_never_pulled"}
|
|
assert found["notes"]["never_pulled"] == 131
|
|
assert found["rules"]["never_pulled"] == 58
|
|
|
|
|
|
def test_each_corpus_is_sent_only_to_a_tool_that_takes_it() -> None:
|
|
"""#4798: one remedy for both corpora sent the rules reader to
|
|
`menus_to_review`, which refuses every source but the notes menu. Tied to
|
|
`REVIEWABLE` itself, so it goes red in either direction: the rules text
|
|
naming the tool while no rule arm is reviewable, or a rule arm becoming
|
|
reviewable while the text still says there is no sample."""
|
|
from scribe.services.retrieval_review import REVIEWABLE
|
|
|
|
ws = warn(
|
|
{},
|
|
usage={"distinct_notes_surfaced": 9, "distinct_notes_pulled": 1},
|
|
rule_usage={"distinct_rules_surfaced": 9, "distinct_rules_pulled": 1},
|
|
)
|
|
detail = {w["numbers"]["corpus"]: w["detail"]
|
|
for w in ws if w["code"] == "surfaced_never_pulled"}
|
|
assert set(detail) == {"notes", "rules"}
|
|
assert "auto_inject" in REVIEWABLE
|
|
assert "menus_to_review" in detail["notes"]
|
|
rule_arms = {"prompt_rule", "pre_tool_rule", "write_path_rule"}
|
|
assert ("menus_to_review" in detail["rules"]) == bool(rule_arms & set(REVIEWABLE))
|
|
assert "when_to_apply" in detail["rules"], "the rules text names no next step"
|
|
|
|
|
|
def test_everything_opened_reports_nothing() -> None:
|
|
ws = warn({}, usage={"distinct_notes_surfaced": 5, "distinct_notes_pulled": 5})
|
|
assert "surfaced_never_pulled" not in codes(ws)
|
|
|
|
|
|
def test_an_empty_corpus_reports_nothing_rather_than_zero() -> None:
|
|
ws = warn({}, usage={"distinct_notes_surfaced": 0, "distinct_notes_pulled": 0})
|
|
assert "surfaced_never_pulled" not in codes(ws)
|
|
|
|
|
|
# ── unregistered_source ───────────────────────────────────────────────────
|
|
|
|
def test_a_source_missing_from_the_registry_is_reported() -> None:
|
|
"""The check that covers what the static test cannot reach — a source
|
|
arriving through a parameter or a dict key."""
|
|
ws = warn({"a_fourth_write_path_arm": src(calls=50)})
|
|
assert codes(ws, "a_fourth_write_path_arm") == {"unregistered_source"}
|
|
|
|
|
|
def test_an_unregistered_source_gets_no_other_verdict() -> None:
|
|
"""Its numbers are real, but nothing says whether it was asked."""
|
|
ws = warn({"mystery": src(calls=500, zero_result_calls=0)})
|
|
assert "cannot_decline" not in codes(ws)
|
|
|
|
|
|
def test_every_registered_source_stays_quiet_when_healthy() -> None:
|
|
"""The negative control for the whole suite (rule 167).
|
|
|
|
A guard that cannot pass cleanly is not a guard — if a healthy window
|
|
produced warnings, every assertion above would be meaningless.
|
|
"""
|
|
from scribe.services.retrieval_registry import POINTS
|
|
|
|
healthy = {s: src(calls=100, zero_result_calls=40, p10=0.90)
|
|
for s in POINTS}
|
|
assert warn(healthy, floors={s: 0.70 for s in POINTS}) == []
|
|
|
|
|
|
# ── silent surfaces ───────────────────────────────────────────────────────
|
|
|
|
def test_a_registered_point_with_no_rows_is_reported() -> None:
|
|
quiet = _silent_surfaces({"auto_inject": src(calls=100)}, {}, {}, active=True)
|
|
assert "prompt_rule" in {p["source"] for p in quiet}
|
|
|
|
|
|
def test_a_point_that_emitted_is_not_reported() -> None:
|
|
quiet = _silent_surfaces({"auto_inject": src(calls=100)}, {}, {}, active=True)
|
|
assert "auto_inject" not in {p["source"] for p in quiet}
|
|
|
|
|
|
def test_a_deliberately_quiet_point_is_never_reported() -> None:
|
|
"""A justified silence must not read as a gap (#2475)."""
|
|
quiet = {p["source"] for p in _silent_surfaces({}, {}, {}, active=True)}
|
|
for web_only in ("rest_search", "browse_search", "rest_note", "rest_rule"):
|
|
assert web_only not in quiet
|
|
|
|
|
|
def test_an_inactive_window_reports_no_silence_at_all() -> None:
|
|
"""On a fresh install every point is silent; the list would be the
|
|
registry printed back (rule 115)."""
|
|
assert _silent_surfaces({}, {}, {}, active=False) == []
|
|
|
|
|
|
def test_a_point_seen_only_in_usage_counts_as_having_emitted() -> None:
|
|
"""The write-path arms and the pull sources never reach `sources` — they
|
|
live in note_usage_events — so reading only retrieval_logs would report
|
|
every one of them as silent."""
|
|
quiet = {p["source"] for p in _silent_surfaces(
|
|
{}, {"by_source": {"write_path_place": {}}},
|
|
{"by_source": {"enter_project": {}}}, active=True)}
|
|
assert "write_path_place" not in quiet
|
|
assert "enter_project" not in quiet
|
|
|
|
|
|
# ── settings parsing ──────────────────────────────────────────────────────
|
|
|
|
@pytest.mark.parametrize("raw,fallback,want", [
|
|
("50", 30, 50),
|
|
("0.05", 0.02, 0.05),
|
|
("", 30, 30),
|
|
("banana", 30, 30), # a malformed setting must not take the readout down
|
|
(None, 0.02, 0.02),
|
|
])
|
|
def test_a_setting_falls_back_rather_than_raising(raw, fallback, want) -> None:
|
|
assert _num(raw, fallback) == want
|
|
|
|
|
|
# ── read_and_unacted / outcomes_never_recorded (#4213, milestone 419) ─────
|
|
#
|
|
# The pair exists because ZERO OUTCOMES IS AMBIGUOUS, and getting that wrong
|
|
# would have been this milestone's own failure mode in miniature: a window
|
|
# with no outcome rows cannot tell "every rule was ignored" from "nothing
|
|
# reports outcomes yet". Reporting the first when the truth is the second
|
|
# manufactures a finding out of an unwired feature — #3311, where a statistic
|
|
# that could not vary was read as a fact about the corpus.
|
|
|
|
def ru(**kw) -> dict:
|
|
base = {
|
|
"distinct_rules_surfaced": 0, "distinct_rules_pulled": 0,
|
|
"distinct_rules_acted": 0, "applied": 0, "departed": 0,
|
|
}
|
|
base.update(kw)
|
|
return base
|
|
|
|
|
|
def test_rules_opened_with_no_outcome_machinery_running_says_so() -> None:
|
|
"""The cold-instrument case, which is what an install looks like the day
|
|
this ships. It must NOT read as "47 rules ignored"."""
|
|
ws = warn({}, rule_usage=ru(distinct_rules_pulled=47))
|
|
assert "outcomes_never_recorded" in codes(ws)
|
|
assert "read_and_unacted" not in codes(ws)
|
|
[w] = [w for w in ws if w["code"] == "outcomes_never_recorded"]
|
|
assert w["numbers"]["opened"] == 47
|
|
# The distinction is in the prose, because the prose is what gets read.
|
|
assert "does NOT mean they were ignored" in w["detail"]
|
|
|
|
|
|
def test_once_outcomes_exist_the_unacted_rules_are_named() -> None:
|
|
"""The instrument is live — some rules recorded an outcome — so the ones
|
|
that did not are a real finding rather than an artefact."""
|
|
ws = warn({}, rule_usage=ru(
|
|
distinct_rules_pulled=20, distinct_rules_acted=6, applied=5, departed=2,
|
|
))
|
|
assert "read_and_unacted" in codes(ws)
|
|
assert "outcomes_never_recorded" not in codes(ws)
|
|
[w] = [w for w in ws if w["code"] == "read_and_unacted"]
|
|
assert w["numbers"]["unacted"] == 14
|
|
assert w["numbers"]["opened"] == 20 and w["numbers"]["acted"] == 6
|
|
assert w["numbers"]["applied"] == 5 and w["numbers"]["departed"] == 2
|
|
|
|
|
|
def test_a_departure_alone_is_enough_to_warm_the_instrument() -> None:
|
|
"""Departures count as outcomes. An install whose every recorded outcome
|
|
is a departure is saying something loudly, and must not be mistaken for
|
|
one that records nothing."""
|
|
ws = warn({}, rule_usage=ru(
|
|
distinct_rules_pulled=9, distinct_rules_acted=2, departed=3,
|
|
))
|
|
assert "read_and_unacted" in codes(ws)
|
|
assert "outcomes_never_recorded" not in codes(ws)
|
|
|
|
|
|
def test_every_opened_rule_acted_on_reports_nothing() -> None:
|
|
ws = warn({}, rule_usage=ru(
|
|
distinct_rules_pulled=4, distinct_rules_acted=4, applied=4,
|
|
))
|
|
assert "read_and_unacted" not in codes(ws)
|
|
assert "outcomes_never_recorded" not in codes(ws)
|
|
|
|
|
|
def test_no_rules_opened_at_all_reports_neither() -> None:
|
|
"""Silence is not a finding. A window where nothing was opened has nothing
|
|
to say about outcomes, and saying it anyway would put a warning on every
|
|
fresh install (rule 115)."""
|
|
ws = warn({}, rule_usage=ru(distinct_rules_surfaced=12))
|
|
assert "read_and_unacted" not in codes(ws)
|
|
assert "outcomes_never_recorded" not in codes(ws)
|
|
|
|
|
|
def test_an_absent_rule_usage_block_is_not_a_finding() -> None:
|
|
"""A failed rule-usage read leaves the keys missing or zero. Neither may
|
|
become a warning, because a warning computed over rows that could not be
|
|
loaded describes the outage, not the corpus (#2663)."""
|
|
assert "read_and_unacted" not in codes(warn({}, rule_usage={}))
|
|
assert "outcomes_never_recorded" not in codes(warn({}, rule_usage={}))
|
|
assert "outcomes_never_recorded" not in codes(
|
|
warn({}, rule_usage={"rule_usage_failed": True})
|
|
)
|
|
|
|
|
|
# ── floor_moved_mid_window (#4225) ────────────────────────────────────────
|
|
#
|
|
# WHY THE BAND CHECK IS SUSPENDED RATHER THAN SOFTENED.
|
|
#
|
|
# `band_hugs_floor` asks whether the scores are piled on the bar. That needs
|
|
# the scores and the bar to come from the same regime, and across a floor
|
|
# change they do not — the comparison silently becomes one between two
|
|
# populations.
|
|
#
|
|
# It announced itself when the change was a RAISE: on the instance this was
|
|
# found on, `write_path_rule` went 0.68 -> 0.72 as a shipped default inside
|
|
# the window, and p10 computed over calls made under the old bar came out
|
|
# BELOW the new floor. The readout printed a band "-0.0216 above" its floor.
|
|
#
|
|
# A LOWERED floor is the dangerous one, because it hides: the gap comes out
|
|
# comfortably positive and reads as a clean bill of health on a sample that
|
|
# half predates the bar being judged. Both directions are pinned below.
|
|
|
|
MOVED = "2026-09-02T00:00:00+00:00"
|
|
|
|
|
|
def test_a_floor_that_moved_in_the_window_suspends_the_band_check() -> None:
|
|
ws = warn({"auto_inject": src(calls=100, zero_result_calls=5, p10=0.705)},
|
|
floors={"auto_inject": 0.70}, floor_moves={"auto_inject": MOVED})
|
|
assert "floor_moved_mid_window" in codes(ws, "auto_inject")
|
|
assert "band_hugs_floor" not in codes(ws), (
|
|
"a suspended check must not also answer — the two never accompany "
|
|
"each other, or the reader gets a number and a warning about it"
|
|
)
|
|
|
|
|
|
def test_the_impossible_negative_gap_is_not_printed_at_all() -> None:
|
|
"""The symptom that exposed this: p10 BELOW the floor that gates the arm.
|
|
|
|
Arithmetically impossible inside one population, and the sentence built
|
|
from it ("only -0.0216 above") is not one anybody can act on.
|
|
"""
|
|
ws = warn({"auto_inject": src(calls=100, zero_result_calls=5, p10=0.6984)},
|
|
floors={"auto_inject": 0.72}, floor_moves={"auto_inject": MOVED})
|
|
assert "band_hugs_floor" not in codes(ws)
|
|
assert not any(w.get("numbers", {}).get("gap", 0) < 0 for w in ws)
|
|
|
|
|
|
def test_a_lowered_floor_is_suspended_too_though_its_gap_looks_healthy() -> None:
|
|
"""The direction that does NOT announce itself.
|
|
|
|
A gap of 0.10 reads as a comfortable margin. It is computed over calls
|
|
half of which were made under a different bar, so it is not a margin at
|
|
all — and nothing in the number says so.
|
|
"""
|
|
ws = warn({"auto_inject": src(calls=100, zero_result_calls=5, p10=0.80)},
|
|
floors={"auto_inject": 0.70}, floor_moves={"auto_inject": MOVED})
|
|
assert "floor_moved_mid_window" in codes(ws, "auto_inject")
|
|
|
|
|
|
def test_a_floor_that_did_not_move_still_gets_judged() -> None:
|
|
"""The mirror error, and the expensive one: suspending on nothing would
|
|
retire a working check."""
|
|
ws = warn({"auto_inject": src(calls=100, zero_result_calls=5, p10=0.705)},
|
|
floors={"auto_inject": 0.70}, floor_moves={})
|
|
assert "band_hugs_floor" in codes(ws, "auto_inject")
|
|
assert "floor_moved_mid_window" not in codes(ws)
|
|
|
|
|
|
def test_only_the_arm_that_moved_is_suspended() -> None:
|
|
"""Surfaces are judged independently; one arm's release change says
|
|
nothing about another's sample."""
|
|
ws = warn(
|
|
{"auto_inject": src(calls=100, zero_result_calls=5, p10=0.705),
|
|
"write_path": src(calls=100, zero_result_calls=5, p10=0.705)},
|
|
floors={"auto_inject": 0.70, "write_path": 0.70},
|
|
floor_moves={"auto_inject": MOVED},
|
|
)
|
|
assert "floor_moved_mid_window" in codes(ws, "auto_inject")
|
|
assert "band_hugs_floor" in codes(ws, "write_path")
|
|
|
|
|
|
def test_the_warning_says_when_and_what_to_do_about_it() -> None:
|
|
"""A finding with no remedy is a complaint. The reader needs the date, so
|
|
they can ask again with a window that starts after it."""
|
|
w = next(w for w in warn(
|
|
{"auto_inject": src(calls=100, zero_result_calls=5, p10=0.705)},
|
|
floors={"auto_inject": 0.70}, floor_moves={"auto_inject": MOVED},
|
|
) if w["code"] == "floor_moved_mid_window")
|
|
assert w["numbers"]["moved_at"] == MOVED
|
|
assert MOVED in w["detail"] and "days" in w["detail"]
|
|
|
|
|
|
def test_a_quiet_arm_is_not_suspended_either_way() -> None:
|
|
"""Below `min_calls` neither check runs — a moved floor does not promote
|
|
an arm nobody used into something worth a line."""
|
|
ws = warn({"auto_inject": src(calls=1, zero_result_calls=0, p10=0.705)},
|
|
floors={"auto_inject": 0.70}, floor_moves={"auto_inject": MOVED})
|
|
assert "floor_moved_mid_window" not in codes(ws)
|
|
|
|
|
|
# ── a fixed-query arm: decline rate is arithmetic, not evidence (#4232) ─────
|
|
#
|
|
# `report_preference` searches one constant string (COMPLETION_QUERY), so it
|
|
# scores against one number on every call. Its decline rate is therefore 0% or
|
|
# 100% and never in between, and which one depends only on where the bar sits
|
|
# relative to that constant.
|
|
#
|
|
# So the two warnings swap roles for these arms. "Never declined" stops being
|
|
# evidence about the floor — `cannot_decline`'s own remedy, "check that it
|
|
# applies its floor", is unanswerable from it. "Always declined" starts being
|
|
# evidence, because for a constant score it means the bar is above it and no
|
|
# further traffic will ever say otherwise.
|
|
|
|
|
|
def test_cannot_decline_is_silent_on_a_fixed_query_arm():
|
|
"""The live readout fired this on `report_preference` at 45 calls, 0
|
|
empty, with p10 = p50 = p90 = min = max = 0.791 — five identical
|
|
percentiles, which is one record at one score rather than a ranking."""
|
|
ws = warn({"report_preference": src(calls=45, zero_result_calls=0, p10=0.791)})
|
|
assert "cannot_decline" not in codes(ws, "report_preference")
|
|
|
|
|
|
def test_cannot_decline_still_fires_where_the_rate_means_something():
|
|
"""The falsifier for the case above (rule 167). If this passes only
|
|
because the check was disabled rather than narrowed, this fails."""
|
|
ws = warn({"auto_inject": src(calls=N, zero_result_calls=0)})
|
|
assert "cannot_decline" in codes(ws, "auto_inject")
|
|
|
|
|
|
def test_a_fixed_query_arm_that_never_clears_its_bar_is_named():
|
|
"""69 consecutive declines at 0.0006 under the bar is a state this arm has
|
|
actually been in. Nothing else in the readout would have said so: it looks
|
|
exactly like an arm with nothing to report."""
|
|
ws = warn({"report_preference": src(calls=45, zero_result_calls=45)})
|
|
assert "fixed_query_never_clears" in codes(ws, "report_preference")
|
|
|
|
detail = next(w["detail"] for w in ws if w["code"] == "fixed_query_never_clears")
|
|
assert "near_miss_samples" in detail, (
|
|
"the last time this fired, every percentile said lower the floor and "
|
|
"the refused record showed the refusal was right — so the warning has "
|
|
"to send the reader to the record, not to the dial"
|
|
)
|
|
|
|
|
|
def test_an_ordinary_arm_returning_nothing_all_window_is_not_dead():
|
|
"""For an arm whose score can vary, an empty window means nothing matched,
|
|
which is an answer rather than a fault."""
|
|
ws = warn({"auto_inject": src(calls=N, zero_result_calls=N)})
|
|
assert "fixed_query_never_clears" not in codes(ws, "auto_inject")
|
|
|
|
|
|
def test_a_fixed_query_arm_that_sometimes_clears_is_not_dead():
|
|
"""Only ALL-empty says the bar is above the constant. Anything in between
|
|
means the score is not actually constant, and the premise is wrong."""
|
|
ws = warn({"report_preference": src(calls=45, zero_result_calls=44)})
|
|
assert "fixed_query_never_clears" not in codes(ws, "report_preference")
|
|
|
|
|
|
def test_the_dead_arm_warning_still_needs_volume():
|
|
ws = warn({"report_preference": src(calls=N - 1, zero_result_calls=N - 1)})
|
|
assert "fixed_query_never_clears" not in codes(ws)
|
|
|
|
|
|
def test_the_registry_declares_which_arms_ask_a_fixed_question():
|
|
"""Asserted on structure (rule 167), and able to fail: if `fixed_query`
|
|
is dropped or defaults to True, one of these two halves breaks."""
|
|
from scribe.services.retrieval_registry import POINTS
|
|
|
|
assert POINTS["report_preference"].fixed_query is True, (
|
|
"services/reply_preferences.py::COMPLETION_QUERY is a module constant"
|
|
)
|
|
# An arm whose query is built from the prompt, the file or the command is
|
|
# not fixed, and marking one would silence a warning that works there.
|
|
for varying in ("auto_inject", "write_path", "pre_tool_rule", "prompt_rule"):
|
|
assert POINTS[varying].fixed_query is False, varying
|
|
|
|
|
|
# ── floor_history_unknown: the window reaches back past the ledger ─────────
|
|
#
|
|
# THE SAME SUSPENSION, FOR THE CASE THE ONE ABOVE CANNOT SEE.
|
|
#
|
|
# `floor_moved_mid_window` fires on a move the ledger recorded. The check
|
|
# overhead reads an EMPTY `floor_moves` as "the floor held steady" — and that
|
|
# reading is sound only where the ledger was watching. Before an arm's first
|
|
# floor row, "no move recorded" and "no move" are different statements, and
|
|
# the first was standing in for the second.
|
|
#
|
|
# It showed up the day this instance's ledger was seeded. Its earliest event
|
|
# of any kind is 2026-09-17; `write_path_rule`'s only floor row is a release
|
|
# baseline written 2026-09-21; a 30-day window opens 2026-08-22. The readout
|
|
# printed that arm's band as "-0.0208 above its floor" — the impossible
|
|
# negative the section above exists to prevent — with the suspension silent,
|
|
# because there was nothing on record for it to notice.
|
|
#
|
|
# The distinction is not academic for #4261: every window for the next month
|
|
# opens before that date, and measuring a change against an instrument that
|
|
# reports a number it cannot support is the #4225 trap one level up.
|
|
|
|
GAP_FROM = "2026-09-21T11:58:00+00:00"
|
|
|
|
|
|
def test_an_arm_whose_floor_history_starts_mid_window_is_suspended() -> None:
|
|
ws = warn({"write_path_rule": src(calls=100, zero_result_calls=5, p10=0.6992)},
|
|
floors={"write_path_rule": 0.72},
|
|
floor_gaps={"write_path_rule": GAP_FROM})
|
|
assert "floor_history_unknown" in codes(ws, "write_path_rule")
|
|
assert "band_hugs_floor" not in codes(ws), (
|
|
"a suspended check must not also answer"
|
|
)
|
|
|
|
|
|
def test_the_negative_gap_that_exposed_this_is_never_printed() -> None:
|
|
"""The observed symptom, reproduced: p10 0.6992 under a floor of 0.72 is
|
|
a band -0.0208 "above" its floor. Arithmetically impossible inside one
|
|
population, and no reader can act on it."""
|
|
ws = warn({"write_path_rule": src(calls=100, zero_result_calls=5, p10=0.6992)},
|
|
floors={"write_path_rule": 0.72},
|
|
floor_gaps={"write_path_rule": GAP_FROM})
|
|
assert not any(w.get("numbers", {}).get("gap", 0) < 0 for w in ws)
|
|
|
|
|
|
def test_a_healthy_looking_gap_is_suspended_too() -> None:
|
|
"""The direction that hides. A gap of 0.10 reads as a comfortable margin
|
|
and is computed against a bar nothing can vouch for over that window — the
|
|
same asymmetry as a lowered floor, and the reason this suspends rather
|
|
than only catching negatives."""
|
|
ws = warn({"write_path_rule": src(calls=100, zero_result_calls=5, p10=0.82)},
|
|
floors={"write_path_rule": 0.72},
|
|
floor_gaps={"write_path_rule": GAP_FROM})
|
|
assert "floor_history_unknown" in codes(ws, "write_path_rule")
|
|
|
|
|
|
def test_an_arm_with_no_history_at_all_says_how_to_start_one() -> None:
|
|
"""`None` rather than a date: there is nothing to wait for, so the remedy
|
|
is different and the sentence has to be too. A reader told to "wait for
|
|
2026-…" when no date exists would wait forever."""
|
|
w = next(w for w in warn(
|
|
{"write_path_rule": src(calls=100, zero_result_calls=5, p10=0.6992)},
|
|
floors={"write_path_rule": 0.72},
|
|
floor_gaps={"write_path_rule": None},
|
|
) if w["code"] == "floor_history_unknown")
|
|
assert w["numbers"]["known_from"] is None
|
|
assert "no floor history" in w["detail"]
|
|
assert "tune_retrieval" in w["detail"], "a finding with no remedy is a complaint"
|
|
|
|
|
|
def test_the_dated_case_says_when_the_check_comes_back() -> None:
|
|
w = next(w for w in warn(
|
|
{"write_path_rule": src(calls=100, zero_result_calls=5, p10=0.6992)},
|
|
floors={"write_path_rule": 0.72},
|
|
floor_gaps={"write_path_rule": GAP_FROM},
|
|
) if w["code"] == "floor_history_unknown")
|
|
assert w["numbers"]["known_from"] == GAP_FROM
|
|
assert GAP_FROM in w["detail"] and "days" in w["detail"]
|
|
|
|
|
|
def test_a_covered_arm_is_still_judged() -> None:
|
|
"""The mirror error, and the expensive one. An arm absent from the gap map
|
|
is fully covered; suspending it would retire a working check on the
|
|
strength of nothing."""
|
|
ws = warn({"write_path_rule": src(calls=100, zero_result_calls=5, p10=0.7205)},
|
|
floors={"write_path_rule": 0.72}, floor_gaps={})
|
|
assert "band_hugs_floor" in codes(ws, "write_path_rule")
|
|
assert "floor_history_unknown" not in codes(ws)
|
|
|
|
|
|
def test_a_known_move_wins_over_a_history_gap() -> None:
|
|
"""Both can apply — a ledger that starts mid-window may still have caught
|
|
a move inside it. The arm that has a date to give should give it, and
|
|
exactly one line should be printed either way."""
|
|
ws = warn({"write_path_rule": src(calls=100, zero_result_calls=5, p10=0.6992)},
|
|
floors={"write_path_rule": 0.72},
|
|
floor_moves={"write_path_rule": MOVED},
|
|
floor_gaps={"write_path_rule": GAP_FROM})
|
|
assert "floor_moved_mid_window" in codes(ws, "write_path_rule")
|
|
assert "floor_history_unknown" not in codes(ws)
|
|
assert "band_hugs_floor" not in codes(ws)
|
|
|
|
|
|
def test_only_the_uncovered_arm_is_suspended() -> None:
|
|
"""Coverage is per-arm: the ledger can start watching one arm before
|
|
another, and one arm's blind spot says nothing about the next."""
|
|
ws = warn(
|
|
{"write_path_rule": src(calls=100, zero_result_calls=5, p10=0.7205),
|
|
"auto_inject": src(calls=100, zero_result_calls=5, p10=0.705)},
|
|
floors={"write_path_rule": 0.72, "auto_inject": 0.70},
|
|
floor_gaps={"write_path_rule": GAP_FROM},
|
|
)
|
|
assert "floor_history_unknown" in codes(ws, "write_path_rule")
|
|
assert "band_hugs_floor" in codes(ws, "auto_inject")
|
|
|
|
|
|
def test_a_quiet_arm_is_not_suspended_either() -> None:
|
|
"""Below `min_calls` nothing in this block runs — a ledger gap does not
|
|
promote an arm nobody used into something worth a line."""
|
|
ws = warn({"write_path_rule": src(calls=1, zero_result_calls=0, p10=0.6992)},
|
|
floors={"write_path_rule": 0.72},
|
|
floor_gaps={"write_path_rule": GAP_FROM})
|
|
assert "floor_history_unknown" not in codes(ws)
|