feat(500): the reply shapes become server product content - a compact core and per-kind slices, each tied to its moment (#5493)
CI & Build / Python lint (push) Successful in 2s
CI & Build / Plugin hooks (push) Successful in 14s
CI & Build / TypeScript typecheck (push) Successful in 55s
CI & Build / integration (push) Successful in 1m12s
CI & Build / Python tests (push) Failing after 1m29s
CI & Build / Build & push image (push) Skipped

services/reply_shapes.py is the single source of the default reply shapes:
the core (every reply, ~1,800 chars against a 2,200 budget) and three
slices - completion on work.finish, asks on reply.ask, plan on work.plan.
Read through list_reply_shapes (MCP, read-only) and GET
/api/retrieval/reply-shapes, one service behind both doors.

Nothing delivers them yet; that is step 3. The skill still carries its
copy until step 4 shrinks it to the long-form reference.

The software-only vocabulary guard moves into tests/helpers.py
(DEV_ONLY, dev_only_hits) rather than becoming a fourth copy.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-10-09 14:29:44 -04:00
co-authored by Claude Opus 5.5
parent acda59fc73
commit a0390d9da7
8 changed files with 393 additions and 13 deletions
+16
View File
@@ -498,6 +498,22 @@ async def rule_row(rule_id: int):
return await s.get(Rule, rule_id)
# Software-only vocabulary. Scribe's product text speaks for any work a person
# drives through an agent — configuration, infrastructure, documents — so a
# catalog or a shipped shape that reads as software-only is wrong on most of
# the installs it ships to. The actions particular to one kind of work belong
# in data (mappings, rulebooks, preferences), not in the product's words.
DEV_ONLY = (r"\bCI\b", r"\bcommit", r"\bpull request", r"\bcode\b", r"\btest",
r"\bgit\b", r"\brepo(s|sitor\w*)?\b", r"\bbranch", r"\bcompil", r"\bbuild\b")
def dev_only_hits(text: str) -> list[str]:
"""The DEV_ONLY patterns this text contains, case-insensitively."""
import re
return [w for w in DEV_ONLY if re.search(w, text, re.IGNORECASE)]
def skill_text(name: str) -> str:
"""Everything a bundled skill states: its SKILL.md, then each reference file.
+6 -13
View File
@@ -10,7 +10,7 @@ import re
import pytest
from scribe.services import moments
from tests.helpers import FakeMCP
from tests.helpers import FakeMCP, dev_only_hits
_NAME = re.compile(r"^[a-z]+\.[a-z]+$")
@@ -38,26 +38,18 @@ def test_the_skill_family_is_not_a_catalog_entry():
assert not any(k.startswith(moments.SKILL_PREFIX) for k in moments.MOMENTS)
# Software-only vocabulary, the same guard the completion query carries. The
# moments are named for any work a person drives through an agent; the actions
# particular to one kind of work belong in the mappings, not in the meaning.
_DEV_ONLY = (r"\bCI\b", r"\bcommit", r"\bpull request", r"\bcode\b", r"\btest",
r"\bgit\b", r"\brepo(s|sitor\w*)?\b", r"\bbranch", r"\bcompil", r"\bbuild\b")
@pytest.mark.parametrize("field", ["means", "reached_by"])
def test_the_catalog_assumes_no_particular_domain(field):
found = {
m.name: hits
for m in [*moments.MOMENTS.values(), moments.SKILL_FAMILY]
if (hits := [w for w in _DEV_ONLY
if re.search(w, getattr(m, field), re.IGNORECASE)])
if (hits := dev_only_hits(getattr(m, field)))
}
assert not found, f"software-only vocabulary in `{field}`: {found}"
def test_the_guard_can_fail():
assert re.search(_DEV_ONLY[1], "after the commit lands", re.IGNORECASE)
assert dev_only_hits("after the commit lands")
@pytest.mark.parametrize("name", list(moments.MOMENTS))
@@ -116,11 +108,12 @@ def test_the_tools_are_registered_and_classified():
mcp = FakeMCP()
tool.register(mcp)
assert mcp.names == [
"list_moments", "map_action", "unmap_action",
"list_moments", "list_reply_shapes", "map_action", "unmap_action",
"rules_to_mount", "propose_rule_moments", "rule_moment_proposals",
"judge_rule_moments", "rule_misfired",
]
assert {"list_moments", "rules_to_mount", "rule_moment_proposals"} <= _READ_ONLY_TOOLS
assert {"list_moments", "list_reply_shapes",
"rules_to_mount", "rule_moment_proposals"} <= _READ_ONLY_TOOLS
# A confirm mounts a rule, so a read key must not reach it.
assert {"map_action", "unmap_action",
"propose_rule_moments", "judge_rule_moments", "rule_misfired"} <= _WRITE_TOOLS
+139
View File
@@ -0,0 +1,139 @@
"""The default reply shapes and the moments they ride (milestone 500 step 1).
What this pins is what delivery will depend on: every shape rides a moment that
exists, the core is small enough to pay for on every turn, no slice grows back
into the skill it replaced, the text speaks for any kind of work, and both
doors hand out the same shapes.
"""
import pytest
from scribe.services import moments, reply_shapes
from tests.helpers import FakeMCP, dev_only_hits
def test_there_is_a_core_and_at_least_one_slice():
"""The sweeps below are vacuous over an empty catalog (rule 167)."""
assert reply_shapes.CORE_KEY in reply_shapes.SHAPES
assert len(reply_shapes.SHAPES) >= 2
def test_every_shape_is_keyed_by_its_own_key():
for key, shape in reply_shapes.SHAPES.items():
assert key == shape.key
@pytest.mark.parametrize("key", list(reply_shapes.SHAPES))
def test_every_shape_rides_a_catalog_moment(key):
"""A shape on a moment nothing reaches would never arrive — and an
operator's preference mounted beside it would sit on a dead moment too."""
assert reply_shapes.SHAPES[key].moment in moments.MOMENTS
def test_the_core_rides_the_reply_moment():
"""It is the reply's shape; a preference about every reply is mounted
on `reply.report`, so that is where the core says it lives."""
assert reply_shapes.core().moment == "reply.report"
def test_no_two_slices_share_a_moment():
"""One moment, one default. Two slices on a moment would arrive together
and leave the reader to reconcile them."""
slices = [s.moment for s in reply_shapes.SHAPES.values() if s.key != reply_shapes.CORE_KEY]
assert len(slices) == len(set(slices))
def test_the_core_fits_its_budget():
"""It is paid for in every session. The budget is the ceiling, so a
sentence added to the core has to replace one."""
assert len(reply_shapes.core().text) <= reply_shapes.CORE_BUDGET_CHARS
@pytest.mark.parametrize("key", [k for k in reply_shapes.SHAPES if k != reply_shapes.CORE_KEY])
def test_each_slice_fits_its_budget(key):
assert len(reply_shapes.SHAPES[key].text) <= reply_shapes.SLICE_BUDGET_CHARS
def test_the_budget_guard_can_fail():
long = "x" * (reply_shapes.CORE_BUDGET_CHARS + 1)
assert not len(long) <= reply_shapes.CORE_BUDGET_CHARS
@pytest.mark.parametrize("key", list(reply_shapes.SHAPES))
def test_no_shape_assumes_a_particular_domain(key):
shape = reply_shapes.SHAPES[key]
assert not dev_only_hits(shape.title + "\n" + shape.text), key
@pytest.mark.parametrize("key", list(reply_shapes.SHAPES))
def test_every_shape_says_what_it_is_and_when_it_arrives(key):
shape = reply_shapes.SHAPES[key]
assert shape.title.strip() and shape.delivered.strip() and shape.text.strip()
def test_the_core_carries_the_length_discipline():
"""Milestone 409's live read: replies that had every section were still
too long. Delivery cannot fix that; the core has to say it."""
text = reply_shapes.core().text.lower()
assert "shortest reply that carries the answer" in text
assert "what can go" in text
def test_the_core_is_not_a_slice():
"""It rides every turn, whatever moment the turn reaches — so a moment
lookup never hands it out a second time."""
every = [s.moment for s in reply_shapes.SHAPES.values()]
assert reply_shapes.CORE_KEY not in {s.key for s in reply_shapes.for_moments(every)}
def test_a_moment_brings_its_own_slice_and_nothing_else():
got = reply_shapes.for_moments(["work.finish", "work.run"])
assert [s.key for s in got] == ["completion"]
assert reply_shapes.for_moments([]) == []
def test_the_payload_carries_every_shape_with_its_moments_meaning():
data = reply_shapes.catalog()
assert [s["key"] for s in data["shapes"]] == list(reply_shapes.SHAPES)
assert data["total"] == len(reply_shapes.SHAPES)
for row in data["shapes"]:
assert row["means"] == moments.MOMENTS[row["moment"]].means
async def test_the_tool_returns_the_catalog():
from scribe.mcp.tools import moments as tool
assert await tool.list_reply_shapes() == reply_shapes.catalog()
def test_the_tool_is_registered_as_a_read():
from scribe.mcp.server import _READ_ONLY_TOOLS
from scribe.mcp.tools import moments as tool
mcp = FakeMCP()
tool.register(mcp)
assert "list_reply_shapes" in mcp.names
assert "list_reply_shapes" in _READ_ONLY_TOOLS
def test_both_doors_read_one_catalog():
"""Rule 33 parity: the session and the Settings view cannot show
different defaults."""
from scribe.mcp.tools import moments as tool
from scribe.routes import retrieval as routes
assert tool.shapes_svc is reply_shapes
assert routes.shapes_svc is reply_shapes
async def test_the_route_returns_the_catalog():
from types import SimpleNamespace
from quart import Quart, g
from scribe.routes import retrieval as routes
app = Quart(__name__)
async with app.test_request_context("/api/retrieval/reply-shapes"):
g.user = SimpleNamespace(id=7)
resp = await routes.reply_shapes_route.__wrapped__()
assert await resp.get_json() == reply_shapes.catalog()
+1
View File
@@ -46,6 +46,7 @@ def test_every_endpoint_is_reachable_on_the_app():
"/api/retrieval/moments/mappings",
"/api/retrieval/moments/proposals",
"/api/retrieval/moments/proposals/judge",
"/api/retrieval/reply-shapes",
}