Files
FabledScribe/tests/test_instruction_surfaces_agree.py
T
bvandeusenandClaude Fable 5 fb757fb4ba
CI & Build / Python lint (push) Successful in 3s
CI & Build / Plugin hooks (push) Successful in 13s
CI & Build / integration (push) Failing after 21s
CI & Build / TypeScript typecheck (push) Successful in 25s
CI & Build / Python tests (push) Failing after 31s
CI & Build / Build & push image (push) Skipped
feat(reuse): recording gets a seam — the prior-art hook asks for create_snippet when duplication is proven and unrecorded (#2664)
Zero snippets were ever recorded outside sessions already thinking about
snippets: the read side had a real seam (the PreToolUse hook) and the record
side had a trailing clause of a floor bullet. The trigger moment — 'I just
wrote the second copy' — is mid-Write/Edit, so the nudge now rides the same
hook: when the local arm proves the definition exists elsewhere in the repo
AND Scribe returned no record of it, the context block asks for
create_snippet. Both gates or silence, so a brand-new helper and an
already-recorded one stay nudge-free.

Floor bullet promoted to name the trigger moments (extract, hoist, second
copy); guarded by test the same way the Systems reflex is. Plugin 0.1.29 so
the cache picks up the hook (#2209).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-08-14 21:45:32 -04:00

213 lines
9.4 KiB
Python

"""The instruction surfaces must agree that the agent pulls the rules itself.
WHY THIS EXISTS
Rule #119 makes the instruction surfaces the SPECIFICATION for product
behaviour — there is no other place the "load the operator's rules" obligation
is written down, and no code path enforces it. So a surface that states it
differently isn't a documentation slip; it is the product behaving differently.
That happened (#2497). `_INSTRUCTIONS` said the SessionStart hook "is the
bridge" for getting rules into a session, while the `using-scribe` skill said to
pull them yourself and treat any push as a bonus. An agent weighting the first
would reasonably skip the pull.
#2198 is the case where that is wrong: every plugin hook was silently inert for
an extended period, and nothing announced it. An agent trusting the push would
have run with no binding rules and no signal — while those rules govern branch,
commit, push and other hard-to-reverse actions.
The asymmetry is the whole argument, and it is what these tests pin: pulling
when a push also arrived costs one redundant call; not pulling when the push
never came costs the operator's rules entirely.
WHAT THIS DOES NOT DO
It cannot tell whether two surfaces contradict each other in prose generally —
that needs a reader. It pins the one instruction whose absence is known to be
load-bearing, and the specific shape the #2497 defect took: naming the push
without also stating the pull.
"""
from __future__ import annotations
import pathlib
ROOT = pathlib.Path(__file__).resolve().parents[1]
# The pull instruction, however a surface phrases the surrounding prose.
PULL = "list_always_on_rules"
# Surfaces a session loads before substantive work. Hand-written because
# "is this a session-start surface?" is an editorial fact, not a derivable one —
# but each entry is asserted to EXIST, so a move or rename fails loudly here
# instead of quietly dropping that surface from the check.
SESSION_START_SURFACES = (
ROOT / "src" / "scribe" / "mcp" / "server.py",
ROOT / "plugin" / "hooks" / "scribe_static_context.md",
ROOT / "plugin" / "skills" / "using-scribe" / "SKILL.md",
)
def _all_surfaces() -> list[tuple[str, str]]:
"""(label, text) for every file a SESSION loads as instructions.
Deliberately not every markdown file under plugin/: `README.md` describes
the push channel to the operator installing the plugin, and telling a human
what the hook does is not the same act as telling an agent it need not pull.
The boundary is "does a session read this", which is skills (loaded by
description match), the hook-injected static context, and the MCP server's
own instructions.
"""
found = [(str(p.relative_to(ROOT)), p.read_text())
for p in (ROOT / "plugin" / "skills").rglob("SKILL.md")]
found += [(str(p.relative_to(ROOT)), p.read_text())
for p in (ROOT / "plugin" / "hooks").glob("*.md")]
server = ROOT / "src" / "scribe" / "mcp" / "server.py"
found.append((str(server.relative_to(ROOT)), server.read_text()))
return found
def test_every_session_start_surface_states_the_pull():
missing = []
for path in SESSION_START_SURFACES:
assert path.exists(), (
f"{path.relative_to(ROOT)} is gone — it was one of the surfaces "
f"carrying the load-the-rules instruction. If it moved, update "
f"SESSION_START_SURFACES; if it was retired, check the instruction "
f"still lives somewhere a fresh session reads."
)
if PULL not in path.read_text():
missing.append(str(path.relative_to(ROOT)))
assert not missing, (
f"these surfaces no longer tell the agent to call {PULL}(): {missing}. "
f"The rules are pull-only and the push is best-effort, so a surface "
f"that omits this leaves a session bound by nothing (#2198, #2497)."
)
def _instructions_text() -> str:
"""The _INSTRUCTIONS literal from server.py, as the client would see it."""
import re
src = (ROOT / "src" / "scribe" / "mcp" / "server.py").read_text()
match = re.search(r'_INSTRUCTIONS = """(.*?)"""', src, re.S)
assert match, "server.py no longer defines _INSTRUCTIONS as a triple-quoted literal"
return match.group(1).strip()
# Claude Code injects only the first ~2,048 characters of an MCP server's
# instructions and silently cuts the rest mid-word (#2562: observed live —
# the previous 20k-char version delivered ~10% of itself, and none of the
# Systems tagging guidance ever reached a session). 2,000 leaves margin.
INSTRUCTIONS_BUDGET = 2000
def test_instructions_fit_the_fold():
text = _instructions_text()
assert len(text) <= INSTRUCTIONS_BUDGET, (
f"_INSTRUCTIONS is {len(text)} chars; the client injects only ~2,048 "
f"and silently cuts the rest (#2562). This block is a MAP — move the "
f"detail to the tool's docstring (delivered at reach-for time), the "
f"plugin static context (always delivered), or a skill; see the "
f"comment above _INSTRUCTIONS."
)
def test_floor_states_the_systems_reflex():
"""Write-time tagging guidance must live on the surface that always arrives.
#2562's behavioral finding: with the guidance only in tool descriptions,
sessions filed records untagged. The static context is the delivery floor,
so the tag-as-you-write reflex has to be stated there.
"""
floor = (ROOT / "plugin" / "hooks" / "scribe_static_context.md").read_text()
for needle in ("system_ids", "create_system"):
assert needle in floor, (
f"plugin/hooks/scribe_static_context.md no longer mentions "
f"{needle} — the Systems tagging reflex must be stated on the "
f"floor, not only in tool descriptions (#2562)."
)
def test_floor_names_the_snippet_recording_triggers():
"""The recording half of reuse needs NAMED trigger moments on the floor.
#2664's behavioral finding: with recording guidance as a trailing clause of
the reuse bullet, zero snippets were ever recorded outside sessions already
thinking about snippets — extracting a shared component (Roundtable's
BaseModal) produced task prose and no record. The floor must name the
moments, not just the tool.
"""
floor = (ROOT / "plugin" / "hooks" / "scribe_static_context.md").read_text()
for needle in ("create_snippet", "second copy"):
assert needle in floor, (
f"plugin/hooks/scribe_static_context.md no longer states the "
f"snippet-recording trigger ({needle!r}) — the record-as-you-build "
f"reflex must be stated on the floor with its trigger moments "
f"(#2664)."
)
# Topics displaced from _INSTRUCTIONS when it was cut to fit the fold. Each
# must remain stated on at least one DELIVERED surface: a tool docstring
# (arrives with the tool schema), the plugin static context (always arrives),
# or a bundled skill (arrives on trigger match). Keyed by a phrase distinctive
# enough that its disappearance means the guidance is gone, not reworded —
# update the phrase alongside a deliberate rewording.
DISPLACED_TOPICS = {
"supersedes": "supersedes",
"trash is recoverable": "deleted_batch_id",
"duplicate gate": "duplicate",
"systems tag-as-you-write": "system_ids",
"reference note vs dev-log": "reference note",
"work-logs over body rewrites": "add_task_log",
"rule homes / altitude": "create_project_rule",
"rules vs other entities": "standing instruction",
"shared records are suggestions": "shared",
"processes run verbatim": "verbatim",
"snippet reuse reflex": "when_to_use",
"compaction at seams": "compact",
"plans are milestones": "start_planning",
"scope to the entered project": "cross-project",
"project bootstrap needs confirmation": "never guessing a project",
}
def test_displaced_topics_live_on_a_delivered_surface():
corpus = ""
for p in (ROOT / "src" / "scribe" / "mcp" / "tools").glob("*.py"):
corpus += p.read_text()
corpus += (ROOT / "plugin" / "hooks" / "scribe_static_context.md").read_text()
for p in (ROOT / "plugin" / "skills").rglob("SKILL.md"):
corpus += p.read_text()
corpus = corpus.lower()
missing = [
f"{topic} (phrase: {phrase!r})"
for topic, phrase in DISPLACED_TOPICS.items()
if phrase.lower() not in corpus
]
assert not missing, (
f"guidance displaced from _INSTRUCTIONS has fallen off every delivered "
f"surface (tool docstrings / static context / skills): {missing}. It "
f"was cut from _INSTRUCTIONS deliberately (#2562) on the premise it "
f"lives elsewhere — restore it somewhere that delivers."
)
def test_no_surface_names_the_push_without_stating_the_pull():
"""The exact shape #2497 took.
Mentioning the SessionStart hook is fine and often useful. Mentioning it
*instead of* the pull is the defect: it reads as "this is handled", and the
surface that says so is the one an agent has least reason to doubt.
"""
offenders = [
label for label, text in _all_surfaces()
if "SessionStart" in text and PULL not in text
]
assert not offenders, (
f"these surfaces describe the SessionStart push but never state the "
f"explicit pull: {offenders}. The push is a delivery optimisation, not "
f"the bridge — it can be absent without saying so. Name it if it helps, "
f"but say to call {PULL}() regardless."
)