From a0390d9da73d05a40d16c6dcc6c6a72f583df85f Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Fri, 9 Oct 2026 14:29:44 -0400 Subject: [PATCH] feat(500): the reply shapes become server product content - a compact core and per-kind slices, each tied to its moment (#5493) services/reply_shapes.py is the single source of the default reply shapes: the core (every reply, ~1,800 chars against a 2,200 budget) and three slices - completion on work.finish, asks on reply.ask, plan on work.plan. Read through list_reply_shapes (MCP, read-only) and GET /api/retrieval/reply-shapes, one service behind both doors. Nothing delivers them yet; that is step 3. The skill still carries its copy until step 4 shrinks it to the long-form reference. The software-only vocabulary guard moves into tests/helpers.py (DEV_ONLY, dev_only_hits) rather than becoming a fourth copy. Co-Authored-By: Claude Opus 5.5 --- src/scribe/mcp/server.py | 3 + src/scribe/mcp/tools/moments.py | 22 +++ src/scribe/routes/retrieval.py | 12 ++ src/scribe/services/reply_shapes.py | 194 ++++++++++++++++++++++++++ tests/helpers.py | 16 +++ tests/test_moments.py | 19 +-- tests/test_reply_shapes.py | 139 ++++++++++++++++++ tests/test_routes_retrieval_tuning.py | 1 + 8 files changed, 393 insertions(+), 13 deletions(-) create mode 100644 src/scribe/services/reply_shapes.py create mode 100644 tests/test_reply_shapes.py diff --git a/src/scribe/mcp/server.py b/src/scribe/mcp/server.py index 9edc3aa1..2f228c12 100644 --- a/src/scribe/mcp/server.py +++ b/src/scribe/mcp/server.py @@ -159,6 +159,9 @@ _READ_ONLY_TOOLS = frozenset({ # retrieval_telemetry's reason, and needed by a read key so that a line # naming a moment can be understood by whoever was shown it. "list_moments", + # The default reply shapes (milestone 500). A pure read of a constant, and + # one a read key needs: a session shown a shape may want the rest. + "list_reply_shapes", # The platform catalog and a project's answers (milestone 463). A pure # read; set_project_platforms is the write. "list_platforms", diff --git a/src/scribe/mcp/tools/moments.py b/src/scribe/mcp/tools/moments.py index d943384c..4d50a995 100644 --- a/src/scribe/mcp/tools/moments.py +++ b/src/scribe/mcp/tools/moments.py @@ -15,6 +15,7 @@ from __future__ import annotations from scribe.mcp._context import current_user_id from scribe.services import moment_actions as actions_svc from scribe.services import moments as moments_svc +from scribe.services import reply_shapes as shapes_svc from scribe.services import rule_moment_judgments as judgments_svc @@ -47,6 +48,26 @@ async def list_moments() -> dict: return out +async def list_reply_shapes() -> dict: + """The default shapes of a reply, and the moment each one arrives at. + + You do not need this to write a reply: the shapes come to you. The core + arrives with the turn, and each kind's shape arrives at the moment before + that kind of reply is written (closing a task, putting a question, opening + a plan). Read this to see them all at once, to answer the operator about + what the default is, or before writing a preference that changes one. + + An operator's own adjustment to a shape is a `preference` mounted on that + shape's `moment` (create_preference with `moments=[...]`); it arrives + beside the default, and where the two differ the preference is what the + operator asked for. + + Each shape carries `key`, `title`, `moment` (and what the moment `means`), + `delivered` (when it arrives) and `text`. + """ + return shapes_svc.catalog() + + async def map_action(tool: str, moment: str, match: str = "", reason: str = "") -> dict: """Make an action reach a moment on this install — the in-session fix for a missed moment. @@ -213,6 +234,7 @@ async def rule_misfired(rule_id: int, moment: str, why: str, def register(mcp) -> None: mcp.tool(name="list_moments")(list_moments) + mcp.tool(name="list_reply_shapes")(list_reply_shapes) mcp.tool(name="map_action")(map_action) mcp.tool(name="unmap_action")(unmap_action) mcp.tool(name="rules_to_mount")(rules_to_mount) diff --git a/src/scribe/routes/retrieval.py b/src/scribe/routes/retrieval.py index ad7ebee6..2310998f 100644 --- a/src/scribe/routes/retrieval.py +++ b/src/scribe/routes/retrieval.py @@ -20,6 +20,7 @@ from quart import Blueprint, jsonify, request from scribe.auth import get_current_user_id, login_required from scribe.services import moment_actions as moment_actions_svc from scribe.services import moments as moments_svc +from scribe.services import reply_shapes as shapes_svc from scribe.services import rule_moment_judgments as judgments_svc from scribe.services import rulebooks as rulebooks_svc from scribe.services.retrieval_telemetry import moment_usage @@ -143,6 +144,17 @@ async def moments_route(): return jsonify(out) +@retrieval_bp.route("/reply-shapes", methods=["GET"]) +@login_required +async def reply_shapes_route(): + """The default reply shapes and the moment each rides (milestone 500). + + The same payload `list_reply_shapes` returns, from the same service, so + the Settings view and the session cannot show different defaults. + """ + return jsonify(shapes_svc.catalog()) + + async def _mapping_change(change): """map/unmap from the browser: the MCP tools' service, recorded as human. diff --git a/src/scribe/services/reply_shapes.py b/src/scribe/services/reply_shapes.py new file mode 100644 index 00000000..a1ef77b4 --- /dev/null +++ b/src/scribe/services/reply_shapes.py @@ -0,0 +1,194 @@ +"""The default shapes of a reply, and the moment each one rides (milestone 500). + +WHY THIS EXISTS + +A reply is where the person the work is for finds out what happened, and they +were not there while it was done. Milestone 409 wrote the shapes that make a +reply readable to them — conclusion first, the shortest reply that carries the +answer, where the work stands — and put them in one bundled skill. A skill +arrives only when the agent decides to load it, so in a long session the shape +faded out of context exactly when replies were getting longer. + +So the shapes are product content held here, on the server, and delivered by +the moments they belong to rather than by the agent remembering to ask: + + - the CORE applies to every reply, so it rides every turn — in full once per + session and after a compaction, and as a one-line pointer otherwise; + - each SLICE belongs to one kind of reply, and arrives at the moment that + comes just before that kind is written: closing a task comes before a + completion report, a structured question before an ask, opening a plan + before a plan is put up for review. + +The moment already knows which kind of reply is coming, so there is no "pick +the type, then load its shape" step for an agent to skip. + +ONE COPY + +This module is the single source. The `reporting-back` skill keeps the +long-form reference (a worked example, the reasoning) and points here; the +hooks carry timing and transport and say nothing of their own about shape. An +operator's adjustments are `preference` records mounted on the same moments, +and arrive beside the default — where the two differ, the preference is what +the operator asked for. + +HOW TO WRITE ONE + +Domain-neutral (rule 115): the work may be software, configuration, +infrastructure or anything else a person drives through an agent. "Evidence" +is what the reader could open to check; "verified" is whatever checking means +for that work. The test beside this module refuses software-only vocabulary. + +Short, because it is paid for in every session's context. The core's budget is +asserted; a sentence added to it should replace one. +""" +from __future__ import annotations + +from dataclasses import dataclass + +from scribe.services import moments + + +@dataclass(frozen=True) +class ReplyShape: + """One piece of the default reply shape, and when it arrives.""" + + key: str + """Stable identifier: what a delivery records and the Settings view keys on.""" + + title: str + """What the piece is, as the Settings view names it.""" + + moment: str + """The catalog moment this piece rides — where an operator's own + preferences for it are mounted too.""" + + delivered: str + """When it arrives, said plainly for the Settings view.""" + + text: str + """The shape itself, as the agent reads it.""" + + +CORE_KEY = "core" + +# ~550 tokens at four characters a token. The core is paid for in every +# session, so this is the ceiling the test holds it to — not a target. It +# shipped at ~1,800, leaving room for the one-line judgement (milestone 500 +# step 2) and little else. +CORE_BUDGET_CHARS = 2200 +# A slice is paid for only at its moment, and once per session; it can afford +# more than the core and still has to stay a slice, not the old skill again. +SLICE_BUDGET_CHARS = 1400 + + +_CORE = """\ +Shape the reply around where the work stands, not the order you did things in. \ +The reader was not there while you worked. + +- **Conclusion first** — the result, the verdict or the question, then what supports it. +- **The shortest reply that carries the answer.** Length is work handed back to the \ +reader. It is earned by a comparison they asked for, options that need laying side by \ +side, or numbers that are the point. +- **One topic per section; bold the few things that matter.** +- **Plain words** — the reader's vocabulary, not names you coined while working. +- **Place the work** — the task or plan it belongs to, by id and title, read from the \ +record rather than recalled. Work with no record behind it says so. +- **Answer the standing questions, even with "nothing"**: does anything need them, and \ +what happens next. A section that explains earns its place only when it changes what \ +they do or decide; the rest belongs in the record's log. +- **A decision they have made is the input to the work.** Act on it; reopen it only \ +for new evidence, said once. +- **Holding an action until they say yes?** Put it near the top, under "Approval requested". +- **End with the one thing they need to decide or do, in bold** — or say there is none. + +By kind: +- **Answer** — the answer, then what they could open to check it. An evaluation is \ +verdict · what exists · the gaps · a recommendation. +- **Proposal** — two or three approaches, the trade-off of each, one recommendation. \ +A review is findings ranked by how much they matter. +- **Progress** — a line or two. **Blocked** — what stopped, what you tried, what you need. +- **Where are we** — the plan and its progress · done · open · needs you · next. + +Before sending, read it as someone who was not there, reading quickly; then once more \ +for what can go.""" + + +_COMPLETION = """\ +A completion report, from the `placement` the closing call returned: + +- **Where this sits** — the plan and the step, or the task alone when it has no plan. +- **What now works** — outcomes the reader would notice ("you can now…", "X no longer…"), \ +not the steps behind them; those belong in the task's log. +- **How / why** — only the decisions worth knowing, and how it was verified. Anything \ +that could not be verified is said here rather than left to read as passed. +- **Needs you** — an action, an approval, a decision, or "nothing". It takes two tests: \ +is it theirs to decide, and is work waiting on it? A question you could settle by \ +reading or measuring something is work not yet done — settle it, say which way you went, \ +and leave them free to overrule. +- **Next** — from `placement.next`, or say the plan is finished. An offer to fix \ +something you found goes here.""" + + +_ASKS = """\ +Asking them to decide or to act — pick the shape: + +- **Decision** — the question first · two to four options, each with what it changes · \ +your recommendation first among them. +- **Clarification** — "My reading is X · the gap is Y · unless you say otherwise I'll do Z." +- **Handoff** (only they can do it) — the action · why it needs them · what it unblocks · \ +what you will do after. +- **Approval** (ready, and holding for a yes) — under "Approval requested": exactly what \ +happens on yes, one numbered item per change so they can approve part · why it needs \ +them · how it is undone. +- **Conflict** (what you are about to do clashes with a rule, a plan or an earlier \ +decision) — what it says · what you were about to do · where they clash · A or B? + +Before asking, check whether you can find the answer yourself: something you can read \ +or measure is a fact to check, not a question to send. When an option rests on existing \ +behaviour, say whose call that was — theirs (cite it) or a past session's nobody confirmed.""" + + +_PLAN = """\ +A plan put up for review, before starting: + +- **Goal** — what done looks like, and why, in a sentence or two. +- **Steps** — in order, each small enough to check on its own. +- **How you will know it works** — something observable, not "it is written". +- **Open questions** — only the ones that are theirs to answer.""" + + +SHAPES: dict[str, ReplyShape] = {s.key: s for s in ( + ReplyShape(CORE_KEY, "Every reply", "reply.report", + "every turn — in full once per session and after a compaction, " + "otherwise as a one-line pointer", _CORE), + ReplyShape("completion", "Completion report", "work.finish", + "when a piece of work is closed", _COMPLETION), + ReplyShape("asks", "Asking them to decide or act", "reply.ask", + "when a structured question is put to them", _ASKS), + ReplyShape("plan", "A plan for review", "work.plan", + "when a plan is opened", _PLAN), +)} + + +def core() -> ReplyShape: + return SHAPES[CORE_KEY] + + +def for_moments(names: list[str]) -> list[ReplyShape]: + """The slices that ride these moments, in catalog order. The core is not a + slice: it rides every turn, whatever moment the turn reaches.""" + wanted = set(names) + return [s for s in SHAPES.values() if s.key != CORE_KEY and s.moment in wanted] + + +def catalog() -> dict: + """The shapes as plain data, for the MCP tool and the REST door alike.""" + return { + "shapes": [ + {"key": s.key, "title": s.title, "moment": s.moment, + "means": moments.MOMENTS[s.moment].means, + "delivered": s.delivered, "text": s.text} + for s in SHAPES.values() + ], + "total": len(SHAPES), + } diff --git a/tests/helpers.py b/tests/helpers.py index 21ace608..cbd032ad 100644 --- a/tests/helpers.py +++ b/tests/helpers.py @@ -498,6 +498,22 @@ async def rule_row(rule_id: int): return await s.get(Rule, rule_id) +# Software-only vocabulary. Scribe's product text speaks for any work a person +# drives through an agent — configuration, infrastructure, documents — so a +# catalog or a shipped shape that reads as software-only is wrong on most of +# the installs it ships to. The actions particular to one kind of work belong +# in data (mappings, rulebooks, preferences), not in the product's words. +DEV_ONLY = (r"\bCI\b", r"\bcommit", r"\bpull request", r"\bcode\b", r"\btest", + r"\bgit\b", r"\brepo(s|sitor\w*)?\b", r"\bbranch", r"\bcompil", r"\bbuild\b") + + +def dev_only_hits(text: str) -> list[str]: + """The DEV_ONLY patterns this text contains, case-insensitively.""" + import re + + return [w for w in DEV_ONLY if re.search(w, text, re.IGNORECASE)] + + def skill_text(name: str) -> str: """Everything a bundled skill states: its SKILL.md, then each reference file. diff --git a/tests/test_moments.py b/tests/test_moments.py index 505c26e8..92466e27 100644 --- a/tests/test_moments.py +++ b/tests/test_moments.py @@ -10,7 +10,7 @@ import re import pytest from scribe.services import moments -from tests.helpers import FakeMCP +from tests.helpers import FakeMCP, dev_only_hits _NAME = re.compile(r"^[a-z]+\.[a-z]+$") @@ -38,26 +38,18 @@ def test_the_skill_family_is_not_a_catalog_entry(): assert not any(k.startswith(moments.SKILL_PREFIX) for k in moments.MOMENTS) -# Software-only vocabulary, the same guard the completion query carries. The -# moments are named for any work a person drives through an agent; the actions -# particular to one kind of work belong in the mappings, not in the meaning. -_DEV_ONLY = (r"\bCI\b", r"\bcommit", r"\bpull request", r"\bcode\b", r"\btest", - r"\bgit\b", r"\brepo(s|sitor\w*)?\b", r"\bbranch", r"\bcompil", r"\bbuild\b") - - @pytest.mark.parametrize("field", ["means", "reached_by"]) def test_the_catalog_assumes_no_particular_domain(field): found = { m.name: hits for m in [*moments.MOMENTS.values(), moments.SKILL_FAMILY] - if (hits := [w for w in _DEV_ONLY - if re.search(w, getattr(m, field), re.IGNORECASE)]) + if (hits := dev_only_hits(getattr(m, field))) } assert not found, f"software-only vocabulary in `{field}`: {found}" def test_the_guard_can_fail(): - assert re.search(_DEV_ONLY[1], "after the commit lands", re.IGNORECASE) + assert dev_only_hits("after the commit lands") @pytest.mark.parametrize("name", list(moments.MOMENTS)) @@ -116,11 +108,12 @@ def test_the_tools_are_registered_and_classified(): mcp = FakeMCP() tool.register(mcp) assert mcp.names == [ - "list_moments", "map_action", "unmap_action", + "list_moments", "list_reply_shapes", "map_action", "unmap_action", "rules_to_mount", "propose_rule_moments", "rule_moment_proposals", "judge_rule_moments", "rule_misfired", ] - assert {"list_moments", "rules_to_mount", "rule_moment_proposals"} <= _READ_ONLY_TOOLS + assert {"list_moments", "list_reply_shapes", + "rules_to_mount", "rule_moment_proposals"} <= _READ_ONLY_TOOLS # A confirm mounts a rule, so a read key must not reach it. assert {"map_action", "unmap_action", "propose_rule_moments", "judge_rule_moments", "rule_misfired"} <= _WRITE_TOOLS diff --git a/tests/test_reply_shapes.py b/tests/test_reply_shapes.py new file mode 100644 index 00000000..0351e4b4 --- /dev/null +++ b/tests/test_reply_shapes.py @@ -0,0 +1,139 @@ +"""The default reply shapes and the moments they ride (milestone 500 step 1). + +What this pins is what delivery will depend on: every shape rides a moment that +exists, the core is small enough to pay for on every turn, no slice grows back +into the skill it replaced, the text speaks for any kind of work, and both +doors hand out the same shapes. +""" +import pytest + +from scribe.services import moments, reply_shapes +from tests.helpers import FakeMCP, dev_only_hits + + +def test_there_is_a_core_and_at_least_one_slice(): + """The sweeps below are vacuous over an empty catalog (rule 167).""" + assert reply_shapes.CORE_KEY in reply_shapes.SHAPES + assert len(reply_shapes.SHAPES) >= 2 + + +def test_every_shape_is_keyed_by_its_own_key(): + for key, shape in reply_shapes.SHAPES.items(): + assert key == shape.key + + +@pytest.mark.parametrize("key", list(reply_shapes.SHAPES)) +def test_every_shape_rides_a_catalog_moment(key): + """A shape on a moment nothing reaches would never arrive — and an + operator's preference mounted beside it would sit on a dead moment too.""" + assert reply_shapes.SHAPES[key].moment in moments.MOMENTS + + +def test_the_core_rides_the_reply_moment(): + """It is the reply's shape; a preference about every reply is mounted + on `reply.report`, so that is where the core says it lives.""" + assert reply_shapes.core().moment == "reply.report" + + +def test_no_two_slices_share_a_moment(): + """One moment, one default. Two slices on a moment would arrive together + and leave the reader to reconcile them.""" + slices = [s.moment for s in reply_shapes.SHAPES.values() if s.key != reply_shapes.CORE_KEY] + assert len(slices) == len(set(slices)) + + +def test_the_core_fits_its_budget(): + """It is paid for in every session. The budget is the ceiling, so a + sentence added to the core has to replace one.""" + assert len(reply_shapes.core().text) <= reply_shapes.CORE_BUDGET_CHARS + + +@pytest.mark.parametrize("key", [k for k in reply_shapes.SHAPES if k != reply_shapes.CORE_KEY]) +def test_each_slice_fits_its_budget(key): + assert len(reply_shapes.SHAPES[key].text) <= reply_shapes.SLICE_BUDGET_CHARS + + +def test_the_budget_guard_can_fail(): + long = "x" * (reply_shapes.CORE_BUDGET_CHARS + 1) + assert not len(long) <= reply_shapes.CORE_BUDGET_CHARS + + +@pytest.mark.parametrize("key", list(reply_shapes.SHAPES)) +def test_no_shape_assumes_a_particular_domain(key): + shape = reply_shapes.SHAPES[key] + assert not dev_only_hits(shape.title + "\n" + shape.text), key + + +@pytest.mark.parametrize("key", list(reply_shapes.SHAPES)) +def test_every_shape_says_what_it_is_and_when_it_arrives(key): + shape = reply_shapes.SHAPES[key] + assert shape.title.strip() and shape.delivered.strip() and shape.text.strip() + + +def test_the_core_carries_the_length_discipline(): + """Milestone 409's live read: replies that had every section were still + too long. Delivery cannot fix that; the core has to say it.""" + text = reply_shapes.core().text.lower() + assert "shortest reply that carries the answer" in text + assert "what can go" in text + + +def test_the_core_is_not_a_slice(): + """It rides every turn, whatever moment the turn reaches — so a moment + lookup never hands it out a second time.""" + every = [s.moment for s in reply_shapes.SHAPES.values()] + assert reply_shapes.CORE_KEY not in {s.key for s in reply_shapes.for_moments(every)} + + +def test_a_moment_brings_its_own_slice_and_nothing_else(): + got = reply_shapes.for_moments(["work.finish", "work.run"]) + assert [s.key for s in got] == ["completion"] + assert reply_shapes.for_moments([]) == [] + + +def test_the_payload_carries_every_shape_with_its_moments_meaning(): + data = reply_shapes.catalog() + assert [s["key"] for s in data["shapes"]] == list(reply_shapes.SHAPES) + assert data["total"] == len(reply_shapes.SHAPES) + for row in data["shapes"]: + assert row["means"] == moments.MOMENTS[row["moment"]].means + + +async def test_the_tool_returns_the_catalog(): + from scribe.mcp.tools import moments as tool + + assert await tool.list_reply_shapes() == reply_shapes.catalog() + + +def test_the_tool_is_registered_as_a_read(): + from scribe.mcp.server import _READ_ONLY_TOOLS + from scribe.mcp.tools import moments as tool + + mcp = FakeMCP() + tool.register(mcp) + assert "list_reply_shapes" in mcp.names + assert "list_reply_shapes" in _READ_ONLY_TOOLS + + +def test_both_doors_read_one_catalog(): + """Rule 33 parity: the session and the Settings view cannot show + different defaults.""" + from scribe.mcp.tools import moments as tool + from scribe.routes import retrieval as routes + + assert tool.shapes_svc is reply_shapes + assert routes.shapes_svc is reply_shapes + + +async def test_the_route_returns_the_catalog(): + from types import SimpleNamespace + + from quart import Quart, g + + from scribe.routes import retrieval as routes + + app = Quart(__name__) + async with app.test_request_context("/api/retrieval/reply-shapes"): + g.user = SimpleNamespace(id=7) + resp = await routes.reply_shapes_route.__wrapped__() + assert await resp.get_json() == reply_shapes.catalog() diff --git a/tests/test_routes_retrieval_tuning.py b/tests/test_routes_retrieval_tuning.py index b0b74b07..7ffb3a60 100644 --- a/tests/test_routes_retrieval_tuning.py +++ b/tests/test_routes_retrieval_tuning.py @@ -46,6 +46,7 @@ def test_every_endpoint_is_reachable_on_the_app(): "/api/retrieval/moments/mappings", "/api/retrieval/moments/proposals", "/api/retrieval/moments/proposals/judge", + "/api/retrieval/reply-shapes", }