Files
FabledScribe/tests/test_session_slippage_readout.py
T
bvandeusenandClaude Opus 5 ae773740b4
CI & Build / Python lint (push) Successful in 3s
CI & Build / Plugin hooks (push) Successful in 14s
CI & Build / TypeScript typecheck (push) Successful in 54s
CI & Build / integration (push) Successful in 55s
CI & Build / Python tests (push) Successful in 1m46s
CI & Build / Build & push image (push) Successful in 14s
feat(plugin): the seam that erases the evidence is where the unresolved rules get named (#4216)
Step 5 of milestone 419. The milestone's subject is that a rule read and
ignored is arithmetically identical to a rule read and followed, and the
compaction is where that identity becomes permanent — the turns holding the
evidence are summarised away, and the unjudged thing survives as nothing.

WHY THIS IS ASSEMBLED IN THE HOOK. `rule_usage_events` has no session column;
it is per user over a window. A session-scoped answer therefore cannot be
asked of the server, and has to be built where a session is a thing that
exists. Four ledgers four hooks already write:

  .rules.ids      an arm NAMED the rule
  .opened.ids     the session called get_rule      (#4100)
  .acted.ids      the session called rule_outcome  (new here)
  .checkpoint.ids the rule HELD an act             (#4214)

Every one is an observed tool call. Nothing asks the model what it followed —
milestone 386 ruled that out, because a model asked "did you apply rule 156?"
says yes. Two subtractions: named-minus-opened is the arm talking to nobody,
opened-minus-acted is the milestone's whole subject.

PreCompact stdout is the compaction's custom instructions (#3680), not a
message to the model, so the readout does not say "you slipped" — it says
which ids must be carried through, which is the one thing a summary can do
about an unjudged finding.

SILENT WHEN NOTHING HAPPENED, and the accusations are conditional on having
members. "0 rules unresolved" on every compaction is how a readout teaches
its reader to skip it. Traffic is still reported, because the static
instructions already ask for it in prose; these lines are the measured
version.

scribe_record_outcome.sh is the third ledger's writer, matched on
mcp__.*__rule_outcome and mirroring scribe_record_opened.sh: TMPDIR only,
silent, exit 0 on every path. A PostToolUse hook that spoke would put a line
after every rule_outcome call and give recording an outcome a cost.

Also: check_plugin.py skipped the new hook for want of a smoke event, which
would have left the newest of the three ledgers as the only one the plugin
lane never runs. Added, mirroring its sibling.

tests/test_precompact_hook.py now isolates TMPDIR — the hook reads session
ledgers from there, so without isolation a test would see whatever this real
session had accumulated.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01821k5B3Ysecp9fNYs92Kuy
2026-09-21 00:53:29 -04:00

256 lines
10 KiB
Python

"""Which rules fired this session, which changed an action, which did not (#4216).
WHY THIS CANNOT BE ASKED OF THE SERVER
`rule_usage_events` has no session column — it is per user over a window — so
a SESSION-scoped answer has to be assembled where a session is a thing that
exists. That is the plugin, from four ledgers four hooks already write:
.rules.ids an arm NAMED the rule (a teaser was shown)
.opened.ids the session called get_rule (#4100)
.acted.ids the session called rule_outcome (#4216, new here)
.checkpoint.ids the rule HELD an act (#4214)
EVERY LINE IS AN OBSERVED TOOL CALL. Nothing asks the model what it followed;
milestone 386 ruled that out because a model asked "did you apply rule 156?"
will say yes. These record what happened.
WHY THE COMPACTION SEAM
A rule read and left unresolved is invisible by construction — it looks
exactly like a rule that worked. The compaction is where that invisibility
becomes permanent: the turns holding the evidence are summarised away, and an
unjudged thing that survives as nothing is how a decision quietly becomes
nobody's. A PreCompact hook's stdout becomes the summariser's instructions
(#3680), so this does not say "you slipped" — it says which ids must be
carried through, which is the one thing a summary can do about it.
WHAT IS PINNED: the arithmetic, and which conditions produce which lines. NOT
pinned: the wording.
"""
from __future__ import annotations
import json
import os
import shutil
import subprocess
import time
from pathlib import Path
import pytest
ROOT = Path(__file__).resolve().parents[1]
HOOKS = ROOT / "plugin" / "hooks"
DEFS = HOOKS / "scribe_defs.sh"
PRECOMPACT = HOOKS / "scribe_precompact_preserve.sh"
RECORDER = HOOKS / "scribe_record_outcome.sh"
HOOKS_JSON = HOOKS / "hooks.json"
def _need(*tools):
for t in tools:
if shutil.which(t) is None:
pytest.skip(f"hook runtime tool {t!r} not installed")
def sh(script: str) -> str:
_need("bash", "awk")
r = subprocess.run(
["bash", "-c", f'set -uo pipefail\n. "{DEFS}"\n{script}'],
capture_output=True, text=True, timeout=30,
)
assert r.returncode == 0, f"exit {r.returncode}: {r.stderr}"
return r.stdout
def ledgers(tmp_path, *, named=(), opened=(), acted=(), held=()) -> Path:
"""The four ledgers, written the way the hooks write them."""
d = tmp_path / "scribe-priorart"
d.mkdir(parents=True, exist_ok=True)
now = int(time.time())
for name, ids in (("rules", named), ("opened", opened), ("acted", acted)):
if ids:
(d / f"s.{name}.ids").write_text(
"".join(f"{i}\t{now}\n" for i in ids)
)
if held:
# The checkpoint ledger is bare ids — it records that something
# HAPPENED rather than what the context still holds, so it carries no
# stamp and never ages (#4214).
(d / "s.checkpoint.ids").write_text("".join(f"{i}\n" for i in held))
return d
def readout(d: Path) -> str:
return sh(f'scribe_slippage_lines "{d}" "s"')
# ── The arithmetic ────────────────────────────────────────────────────────
def minus(a: str, b: str) -> str:
return sh(f'scribe_ids_minus "{a}" "{b}"').strip()
def test_the_difference_keeps_order_and_drops_members():
assert minus("1 9 34 156 173", "9 34 156") == "1 173"
def test_an_empty_difference_prints_nothing():
assert minus("9 34", "9 34") == ""
def test_an_empty_minuend_is_not_an_error():
assert minus("", "9") == ""
def test_a_repeated_id_is_counted_once():
"""The ledgers are append-only, so the same rule can appear many times."""
assert minus("9 9 34 9", "") == "9 34"
# ── What the readout says ─────────────────────────────────────────────────
def test_the_readout_names_what_was_read(tmp_path):
out = readout(ledgers(tmp_path, named=[1, 9], opened=[9]))
assert "read: 9" in out
def test_a_rule_named_and_never_opened_is_named_as_such(tmp_path):
"""The arm talking to nobody. Not an accusation — a teaser skimmed past
leaves nothing behind — but it is the number that says whether the arm is
earning its place."""
out = readout(ledgers(tmp_path, named=[1, 9, 173], opened=[9]))
assert "never opened" in out
line = next(ln for ln in out.splitlines() if "never opened" in ln)
assert "1" in line and "173" in line
assert " 9" not in line.split(":")[1], "an opened rule is not also unread"
def test_a_rule_read_with_no_outcome_is_the_headline(tmp_path):
"""THE MILESTONE'S WHOLE SUBJECT. Read and unresolved is arithmetically
identical to read and followed, and this is the only place that difference
gets carried across the seam."""
out = readout(ledgers(tmp_path, named=[9, 34], opened=[9, 34], acted=[34]))
assert "READ WITH NO OUTCOME RECORDED" in out
line = next(ln for ln in out.splitlines() if "NO OUTCOME" in ln)
assert "9" in line
assert "rule_outcome" in line, "a finding with no remedy is a complaint"
def test_a_rule_that_held_an_act_is_reported_separately(tmp_path):
"""The strongest evidence a rule changed something: it stopped a call
before it ran (#4214). Kept apart from `read` because reading a rule and
having it alter what you did are different claims."""
out = readout(ledgers(tmp_path, named=[156], opened=[156], held=[156]))
assert "held an act" in out and "156" in out
def test_a_session_that_resolved_everything_makes_no_accusation(tmp_path):
"""Traffic is always reported — it is what the static instructions above
already ask for in prose. The SUBTRACTIONS are conditional, so a clean
session gets no scolding. '0 rules unresolved' on every compaction is how
a readout teaches its reader to skip it."""
out = readout(ledgers(tmp_path, named=[7], opened=[7], acted=[7]))
assert "read: 7" in out
assert "NO OUTCOME" not in out
assert "never opened" not in out
def test_a_session_no_rule_touched_says_nothing_at_all(tmp_path):
"""Rule 115's reasoning: a fresh install must not be told something is
wrong when the truth is that nothing has happened yet."""
d = tmp_path / "scribe-priorart"
d.mkdir(parents=True)
assert readout(d).strip() == ""
# ── The hook that carries it ──────────────────────────────────────────────
def run_precompact(event: dict, tmpdir: Path) -> subprocess.CompletedProcess:
_need("bash")
env = dict(os.environ)
env["TMPDIR"] = str(tmpdir)
return subprocess.run(["bash", str(PRECOMPACT)], input=json.dumps(event),
capture_output=True, text=True, timeout=30, env=env)
def test_the_compaction_hook_appends_the_readout(tmp_path):
ledgers(tmp_path, named=[1, 9], opened=[9], acted=[])
r = run_precompact({"session_id": "s", "trigger": "manual"}, tmp_path)
assert r.returncode == 0
assert "Preserve the following literally" in r.stdout, "static half intact"
assert "READ WITH NO OUTCOME RECORDED" in r.stdout
def test_the_compaction_hook_still_exits_zero_with_no_session(tmp_path):
"""An event with no session_id has no ledgers to read. The static
instructions still go out — they are the part that matters most, and
losing them because a measurement was unavailable would be the worse
trade."""
r = run_precompact({"trigger": "auto"}, tmp_path)
assert r.returncode == 0
assert "Preserve the following literally" in r.stdout
def test_the_compaction_hook_never_emits_a_json_envelope(tmp_path):
"""Regression guard on the change that added the readout: for PreCompact
the envelope is pasted into the summariser's prompt rather than
unwrapped."""
ledgers(tmp_path, named=[1], opened=[1])
r = run_precompact({"session_id": "s", "trigger": "manual"}, tmp_path)
assert not r.stdout.lstrip().startswith("{")
# ── The recorder that makes `acted` mean anything ─────────────────────────
def run_recorder(event: dict, tmpdir: Path) -> subprocess.CompletedProcess:
_need("bash")
env = {"PATH": os.environ["PATH"], "HOME": str(tmpdir), "TMPDIR": str(tmpdir)}
return subprocess.run(["bash", str(RECORDER)], input=json.dumps(event),
capture_output=True, text=True, timeout=30, env=env)
def test_declaring_an_outcome_is_recorded(tmp_path):
r = run_recorder(
{"session_id": "s", "tool_input": {"rule_id": 156, "outcome": "applied"}},
tmp_path,
)
assert r.returncode == 0
assert (tmp_path / "scribe-priorart" / "s.acted.ids").read_text().startswith("156\t")
def test_the_entry_is_stamped_like_its_siblings(tmp_path):
"""One reader ages all three ledgers, so all three carry a timestamp."""
run_recorder({"session_id": "s", "tool_input": {"rule_id": 9}}, tmp_path)
line = (tmp_path / "scribe-priorart" / "s.acted.ids").read_text().strip()
assert "\t" in line and line.split("\t")[1].isdigit()
@pytest.mark.parametrize("event", [
{}, {"session_id": "s"}, {"tool_input": {"rule_id": 9}},
{"session_id": "s", "tool_input": {"rule_id": "not-a-number"}},
])
def test_an_unusable_event_records_nothing_and_still_exits_zero(event, tmp_path):
"""A bookkeeping failure must never turn a successful tool call into a
hook error."""
r = run_recorder(event, tmp_path)
assert r.returncode == 0
assert not (tmp_path / "scribe-priorart" / "s.acted.ids").exists()
def test_the_recorder_is_registered_on_the_rule_outcome_tool():
"""A ledger nothing writes to reads as 'nothing was acted on', which is
the exact false finding this milestone exists to avoid producing."""
cfg = json.loads(HOOKS_JSON.read_text())
entries = cfg["hooks"]["PostToolUse"]
assert any(
"rule_outcome" in e.get("matcher", "")
and any("scribe_record_outcome.sh" in h["command"] for h in e["hooks"])
for e in entries
)
def test_the_recorder_is_shell_valid():
_need("bash")
subprocess.run(["bash", "-n", str(RECORDER)], check=True)