Files
FabledScribe/tests/test_reply_shapes.py
T
bvandeusenandClaude Opus 5.5 eadb08c347
CI & Build / Python lint (push) Successful in 3s
CI & Build / Plugin hooks (push) Successful in 12s
CI & Build / TypeScript typecheck (push) Successful in 56s
CI & Build / integration (push) Successful in 1m12s
CI & Build / Python tests (push) Failing after 1m31s
CI & Build / Build & push image (push) Skipped
feat(500): reply shapes are delivered - the core every turn through the ledger, each slice at its moment, reply mounts before the reply (#5495)
- Every turn (/api/plugin/retrieve, UserPromptSubmit): the core reply shape
  leads the payload - in full the first time, as its one-line reminder
  after that - followed by whatever is mounted on reply.report, under the
  shared rule ledger. Fresh keys come back as shape_keys.
- The ledger is <sid>.shapes.ids in scribe-priorart (scribe_shapes_file /
  _seen / _append), so the compaction sweep that clears every .ids ledger
  is what brings the full core back after one.
- At a moment (/api/plugin/moment): the slice for that reply - completion
  on work.finish, asks on reply.ask, plan on work.plan - ahead of the
  mounted rules. reachable_tools now lists tools reaching a shaped moment
  even on an install with nothing mounted.
- Scribe's own tools (attach_moment_rules): reply_shape in the response,
  in full, since that door has no ledger. enter_project carries the core
  for clients with no prompt hook.
- Telemetry: one AppLog row per delivery (plugin / reply_shape), each
  shape with full or pointer and the door (turn, hook, mcp).

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-09 14:45:25 -04:00

243 lines
8.6 KiB
Python

"""The default reply shapes and the moments they ride (milestone 500 step 1).
What this pins is what delivery will depend on: every shape rides a moment that
exists, the core is small enough to pay for on every turn, no slice grows back
into the skill it replaced, the text speaks for any kind of work, and both
doors hand out the same shapes.
"""
import pytest
from scribe.services import moments, reply_shapes
from tests.helpers import FakeMCP, dev_only_hits
# Bound before conftest's autouse stub replaces the module attribute.
_REAL_RECORD = reply_shapes.record_delivery
def test_there_is_a_core_and_at_least_one_slice():
"""The sweeps below are vacuous over an empty catalog (rule 167)."""
assert reply_shapes.CORE_KEY in reply_shapes.SHAPES
assert len(reply_shapes.SHAPES) >= 2
def test_every_shape_is_keyed_by_its_own_key():
for key, shape in reply_shapes.SHAPES.items():
assert key == shape.key
@pytest.mark.parametrize("key", list(reply_shapes.SHAPES))
def test_every_shape_rides_a_catalog_moment(key):
"""A shape on a moment nothing reaches would never arrive — and an
operator's preference mounted beside it would sit on a dead moment too."""
assert reply_shapes.SHAPES[key].moment in moments.MOMENTS
def test_the_core_rides_the_reply_moment():
"""It is the reply's shape; a preference about every reply is mounted
on `reply.report`, so that is where the core says it lives."""
assert reply_shapes.core().moment == "reply.report"
def test_no_two_slices_share_a_moment():
"""One moment, one default. Two slices on a moment would arrive together
and leave the reader to reconcile them."""
slices = [s.moment for s in reply_shapes.SHAPES.values() if s.key != reply_shapes.CORE_KEY]
assert len(slices) == len(set(slices))
def test_the_core_fits_its_budget():
"""It is paid for in every session. The budget is the ceiling, so a
sentence added to the core has to replace one."""
assert len(reply_shapes.core().text) <= reply_shapes.CORE_BUDGET_CHARS
@pytest.mark.parametrize("key", [k for k in reply_shapes.SHAPES if k != reply_shapes.CORE_KEY])
def test_each_slice_fits_its_budget(key):
assert len(reply_shapes.SHAPES[key].text) <= reply_shapes.SLICE_BUDGET_CHARS
def test_the_budget_guard_can_fail():
long = "x" * (reply_shapes.CORE_BUDGET_CHARS + 1)
assert not len(long) <= reply_shapes.CORE_BUDGET_CHARS
@pytest.mark.parametrize("key", list(reply_shapes.SHAPES))
def test_no_shape_assumes_a_particular_domain(key):
shape = reply_shapes.SHAPES[key]
assert not dev_only_hits(shape.title + "\n" + shape.text), key
@pytest.mark.parametrize("key", list(reply_shapes.SHAPES))
def test_every_shape_says_what_it_is_and_when_it_arrives(key):
shape = reply_shapes.SHAPES[key]
assert shape.title.strip() and shape.delivered.strip() and shape.text.strip()
def test_the_core_carries_the_length_discipline():
"""Milestone 409's live read: replies that had every section were still
too long. Delivery cannot fix that; the core has to say it."""
text = reply_shapes.core().text.lower()
assert "shortest reply that carries the answer" in text
assert "what can go" in text
def test_who_decides_what_rides_every_reply_and_every_ask():
"""The operator's line (milestone 500 step 2): the agent settles what
only it can see, the operator decides direction. A one-liner every turn,
in full where a question is about to be handed back."""
core = reply_shapes.core().text.lower()
asks = reply_shapes.SHAPES["asks"].text.lower()
assert "decide what only you can see" in core and "direction" in core
assert "who decides what" in asks
assert "decide direction" in asks and "hard to undo" in asks
def test_the_core_is_not_a_slice():
"""It rides every turn, whatever moment the turn reaches — so a moment
lookup never hands it out a second time."""
every = [s.moment for s in reply_shapes.SHAPES.values()]
assert reply_shapes.CORE_KEY not in {s.key for s in reply_shapes.for_moments(every)}
def test_a_moment_brings_its_own_slice_and_nothing_else():
got = reply_shapes.for_moments(["work.finish", "work.run"])
assert [s.key for s in got] == ["completion"]
assert reply_shapes.for_moments([]) == []
def test_the_payload_carries_every_shape_with_its_moments_meaning():
data = reply_shapes.catalog()
assert [s["key"] for s in data["shapes"]] == list(reply_shapes.SHAPES)
assert data["total"] == len(reply_shapes.SHAPES)
for row in data["shapes"]:
assert row["means"] == moments.MOMENTS[row["moment"]].means
async def test_the_tool_returns_the_catalog():
from scribe.mcp.tools import moments as tool
assert await tool.list_reply_shapes() == reply_shapes.catalog()
def test_the_tool_is_registered_as_a_read():
from scribe.mcp.server import _READ_ONLY_TOOLS
from scribe.mcp.tools import moments as tool
mcp = FakeMCP()
tool.register(mcp)
assert "list_reply_shapes" in mcp.names
assert "list_reply_shapes" in _READ_ONLY_TOOLS
def test_both_doors_read_one_catalog():
"""Rule 33 parity: the session and the Settings view cannot show
different defaults."""
from scribe.mcp.tools import moments as tool
from scribe.routes import retrieval as routes
assert tool.shapes_svc is reply_shapes
assert routes.shapes_svc is reply_shapes
async def test_the_route_returns_the_catalog():
from types import SimpleNamespace
from quart import Quart, g
from scribe.routes import retrieval as routes
app = Quart(__name__)
async with app.test_request_context("/api/retrieval/reply-shapes"):
g.user = SimpleNamespace(id=7)
resp = await routes.reply_shapes_route.__wrapped__()
assert await resp.get_json() == reply_shapes.catalog()
# ── delivery (milestone 500 step 3) ─────────────────────────────────────
@pytest.mark.parametrize("key", list(reply_shapes.SHAPES))
def test_every_shape_has_a_one_line_reminder(key):
shape = reply_shapes.SHAPES[key]
assert shape.reminder.strip() and "\n" not in shape.reminder
assert not dev_only_hits(shape.reminder), key
def test_the_full_form_carries_the_text_and_says_a_preference_wins():
core = reply_shapes.core()
out = reply_shapes.render(core, full=True)
assert out.endswith(core.text)
assert core.title in out and "preference" in out
def test_the_pointer_form_is_one_line_carrying_the_reminder():
core = reply_shapes.core()
out = reply_shapes.render(core, full=False)
assert "\n" not in out
assert core.reminder in out and "list_reply_shapes" in out
assert core.text not in out
def test_a_shape_the_session_holds_goes_out_as_its_pointer():
shapes = [reply_shapes.core(), reply_shapes.SHAPES["completion"]]
blocks, forms = reply_shapes.deliver(shapes, frozenset({"core"}))
assert forms == {"core": reply_shapes.POINTER, "completion": reply_shapes.FULL}
assert reply_shapes.core().text not in blocks[0]
assert reply_shapes.SHAPES["completion"].text in blocks[1]
def test_a_door_with_no_ledger_sends_everything_in_full():
_blocks, forms = reply_shapes.deliver([reply_shapes.core()], frozenset())
assert forms == {"core": reply_shapes.FULL}
@pytest.mark.parametrize("raw,keys", [
("core", {"core"}),
("core, asks,core", {"core", "asks"}),
("", set()),
(None, set()),
("core,nonsense,../x", {"core"}),
])
def test_the_ledger_keeps_only_shapes_that_exist(raw, keys):
assert reply_shapes.parse_seen(raw) == frozenset(keys)
async def test_a_delivery_is_one_plugin_row_naming_each_shape_and_its_form():
from unittest.mock import MagicMock, patch
added = []
session = MagicMock()
session.add = added.append
async def _commit():
return None
session.commit = _commit
class _Ctx:
async def __aenter__(self):
return session
async def __aexit__(self, *exc):
return False
with patch("scribe.models.async_session", lambda: _Ctx()):
await _REAL_RECORD(7, {"core": "pointer", "asks": "full"}, via="turn")
assert len(added) == 1
row = added[0]
assert (row.category, row.action, row.user_id) == ("plugin", "reply_shape", 7)
import json
assert json.loads(row.details) == {"shapes": {"core": "pointer", "asks": "full"},
"via": "turn"}
async def test_the_delivery_row_fails_open_and_skips_an_empty_delivery():
from unittest.mock import patch
def _boom():
raise RuntimeError("db down")
with patch("scribe.models.async_session", _boom):
await _REAL_RECORD(7, {"core": "full"}, via="turn") # no raise
await _REAL_RECORD(7, {}, via="turn")