Files
FabledScribe/src/scribe/services/moment_delivery.py
T
bvandeusenandClaude Opus 5.5 eadb08c347
CI & Build / Python lint (push) Successful in 3s
CI & Build / Plugin hooks (push) Successful in 12s
CI & Build / TypeScript typecheck (push) Successful in 56s
CI & Build / integration (push) Successful in 1m12s
CI & Build / Python tests (push) Failing after 1m31s
CI & Build / Build & push image (push) Skipped
feat(500): reply shapes are delivered - the core every turn through the ledger, each slice at its moment, reply mounts before the reply (#5495)
- Every turn (/api/plugin/retrieve, UserPromptSubmit): the core reply shape
  leads the payload - in full the first time, as its one-line reminder
  after that - followed by whatever is mounted on reply.report, under the
  shared rule ledger. Fresh keys come back as shape_keys.
- The ledger is <sid>.shapes.ids in scribe-priorart (scribe_shapes_file /
  _seen / _append), so the compaction sweep that clears every .ids ledger
  is what brings the full core back after one.
- At a moment (/api/plugin/moment): the slice for that reply - completion
  on work.finish, asks on reply.ask, plan on work.plan - ahead of the
  mounted rules. reachable_tools now lists tools reaching a shaped moment
  even on an install with nothing mounted.
- Scribe's own tools (attach_moment_rules): reply_shape in the response,
  in full, since that door has no ledger. enter_project carries the core
  for clients with no prompt hook.
- Telemetry: one AppLog row per delivery (plugin / reply_shape), each
  shape with full or pointer and the door (turn, hook, mcp).

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-09 14:45:25 -04:00

342 lines
15 KiB
Python

"""Delivering the rules mounted on a moment, through every door (milestone 458 step 4).
One function decides what an act reaches and what arrives with it; the doors
differ only in how they carry the answer:
- the plugin's catch-all PreToolUse hook → `/api/plugin/moment`, with the
session ledger, so a rule already named this session is cited not repeated;
- Scribe's own MCP tools → `attach_moment_rules`, in the tool's response, so
a client without the plugin still gets `work.finish` when a task closes;
- the Stop hook → `deliver_moments`, for the reply moments no tool call marks.
The pipeline stage is `retrieval_pipeline.run_moment_arm`; this module only
resolves the act to its moments and hands both doors the same result.
"""
from __future__ import annotations
import logging
from scribe.services import moment_actions, reply_shapes
from scribe.services import retrieval_pipeline as rp
logger = logging.getLogger(__name__)
def _io() -> rp.MomentIO:
"""Resolved at CALL time, so a test patching either module's name is seen."""
from scribe.services import rule_usage, rulebooks
return rp.MomentIO(
lookup=rulebooks.rules_on_moments,
record_rule_surfaced=rule_usage.record_rule_surfaced,
)
async def reachable_tools(user_id: int) -> list[str]:
"""The tools a call to which could deliver a mounted rule, as tool keys.
What the plugin's catch-all hook reads once per session window, so the
calls that cannot reach a mounted rule — most of them, and every one on
an install that has mounted nothing — never leave the machine. An action
counts when its moment carries a mount or a default reply shape. The
skill loader counts whenever anything is mounted: a stored process
declares its own moments, which only the load itself can resolve, and a
load is rare enough that asking costs nothing.
"""
from scribe.services import rulebooks
# A moment that carries a default reply shape (milestone 500) is worth a
# request whether or not anything is mounted on it: the shape is product,
# so every install has it.
shaped = {s.moment for s in reply_shapes.SHAPES.values() if s.key != reply_shapes.CORE_KEY}
mounted = set(await rulebooks.mounted_moments(user_id))
wanted = mounted | shaped
mappings = await moment_actions.list_mappings(user_id)
keys = {
moment_actions.tool_key(action.tool)
for action, _via in moment_actions.effective_actions(mappings)
if action.moment in wanted
}
if mounted:
keys.add(moment_actions.SKILL_TOOL)
return sorted(keys)
async def deliver_moments(
user_id: int, reached: list[dict], *, project_id: int | None = None,
exclude: frozenset[int] = frozenset(), held: frozenset[int] = frozenset(),
) -> rp.RuleResult:
"""The mounted rules for moments already known to have happened."""
return await rp.run_moment_arm(
reached,
rp.RuleMoment(user_id=user_id, query="", project_id=project_id,
exclude=exclude, held=held),
io=_io(),
)
async def deliver_for_act(
user_id: int, tool: str, tool_input: dict | None, *,
project_id: int | None = None,
exclude: frozenset[int] = frozenset(), held: frozenset[int] = frozenset(),
) -> tuple[list[dict], rp.RuleResult]:
"""Which moments this act reached, and the mounted rules they deliver."""
reached = await moment_actions.moments_for(user_id, tool, tool_input or {})
if not reached:
return [], rp.RuleResult()
result = await deliver_moments(
user_id, reached, project_id=project_id, exclude=exclude, held=held,
)
return reached, result
def shapes_for_act(reached: list[dict], seen: frozenset[str] = frozenset()) -> tuple[list[str], dict[str, str]]:
"""The default reply shapes riding the moments this act reached (milestone 500).
Closing a task comes just before a completion report, a structured
question is an ask, opening a plan comes before a plan is put up for
review — so the shape for that reply arrives at the act, before the reply
is written. In full the first time a session meets it, as its reminder
after that; a door with no ledger passes `seen` empty.
"""
return reply_shapes.deliver(
reply_shapes.for_moments([hit["moment"] for hit in reached]), seen,
)
# What the per-turn delivery says reached `reply.report`: the turn has not
# ended yet, but every turn ends in a reply, and this is the last point
# before it is written.
_REACHED_BY_TURN = "the reply this turn will end with"
async def deliver_for_turn(
user_id: int, *, seen: frozenset[str] = frozenset(), project_id: int | None = None,
exclude: frozenset[int] = frozenset(), held: frozenset[int] = frozenset(),
) -> dict:
"""What every turn carries before its reply is written (milestone 500 step 3).
The core reply shape — in full once per session and again after a
compaction (the ledger is swept then), otherwise its one-line reminder —
and the rules and preferences mounted on `reply.report`, under the same
rule ledger as every other arm. The Stop hook's reply moment fires after
the reply exists, which is too late to shape it and is kept as the
backstop; this is the point before.
Returns `{context, rule_ids, shape_forms}`. Fails open to the core alone,
and the core itself never fails: it is a constant.
"""
blocks, forms = reply_shapes.deliver([reply_shapes.core()], seen)
rule_ids: list[int] = []
try:
reached = [{"moment": "reply.report", "tool": "", "match": _REACHED_BY_TURN,
"via": moment_actions.DEFAULT}]
result = await deliver_moments(
user_id, reached, project_id=project_id, exclude=exclude, held=held,
)
blocks.extend(result.lines)
rule_ids = result.rule_ids
except Exception: # noqa: BLE001 - the mounted half never costs the core
logger.debug("reply.report mounts not delivered for the turn", exc_info=True)
await reply_shapes.record_delivery(user_id, forms, via="turn")
return {"context": "\n\n".join(blocks), "rule_ids": rule_ids, "shape_forms": forms}
async def attach_moment_rules(
user_id: int, tool: str, arguments: dict | None, data: dict,
) -> dict:
"""Carry the rules mounted on this tool call's moments in its own response.
The MCP door (#4286's attach shape): an in-band decoration on a payload
already being returned, fail-open, and absent when there is nothing to
say. The project is the record's own when it has one, else the caller's
argument — a project's mounted rules belong to work in that project.
No session ledger reaches this door, so a mounted rule arrives every time
its moment happens here. That is the right failure for the moments these
tools mark: closing a task is exactly when a rule about finishing applies,
however many tasks were closed before it.
"""
if not isinstance(data, dict):
return data
try:
project_id = data.get("project_id") or (arguments or {}).get("project_id") or None
_reached, result = await deliver_for_act(
user_id, tool, arguments or {}, project_id=project_id,
)
if result.lines:
data["moment_rules"] = {
"lines": result.lines,
"rule_ids": result.shown_rule_ids,
"open_with": "get_rule(id)",
}
# The reply shape for this moment, in full: no ledger reaches this
# door, and an act like closing a task is rare enough that the shape
# arriving each time costs less than one report written without it.
blocks, forms = shapes_for_act(_reached)
if blocks:
data["reply_shape"] = "\n\n".join(blocks)
await reply_shapes.record_delivery(user_id, forms, via="mcp")
except Exception: # noqa: BLE001 - a decoration never breaks the payload
logger.debug("moment rules for %s could not be attached", tool, exc_info=True)
return data
# ── The reply moment: the backstop at the end of a turn ─────────────────
#
# The Stop hook's door, folded in from milestone 456 step 8. A reply is the
# one act no tool call marks, and the moment where "please verify this on
# your end" gets said — so it is checked twice over: the rules MOUNTED on the
# reply moments (deterministic, somebody said they belong here), and the
# reply text against every rule's trigger (the backstop for whatever the
# earlier arms missed). An unopened rule from either half HOLDS the reply for
# one read, in these words; the hook never holds the rewrite.
# The reply's head and tail. The embedder reads ~512 tokens, and the part of
# a report that asks something of the reader — "let me know if it works" —
# is at the END, where a head-only query would cut it off.
_REPLY_HEAD_CHARS = 600
_REPLY_TAIL_CHARS = 1200
_REACHED_BY_REPLY = "the reply that ends this turn"
def reply_moments(reply: str) -> list[dict]:
"""The moments a finished reply reaches: always `reply.report`, and
`reply.ask` when a line of its prose ends in a question."""
reached = [{"moment": "reply.report", "tool": "", "match": _REACHED_BY_REPLY,
"via": moment_actions.DEFAULT}]
in_code = False
for line in (reply or "").splitlines():
stripped = line.strip()
if stripped.startswith("```"):
in_code = not in_code
continue
if not in_code and stripped.endswith("?"):
reached.append({"moment": "reply.ask", "tool": "", "match": "a question in the reply",
"via": moment_actions.DEFAULT})
break
return reached
def _reply_query(reply: str) -> str:
text = " ".join((reply or "").split())
if len(text) <= _REPLY_HEAD_CHARS + _REPLY_TAIL_CHARS:
return text
return text[:_REPLY_HEAD_CHARS] + " … " + text[-_REPLY_TAIL_CHARS:]
def reply_hold_reason(held: list[dict]) -> str:
"""The text the agent reads INSTEAD of its reply going out.
A practice, not a prohibition (rule 165), on the act checkpoint's model:
nothing here knows the reply is wrong. It names each rule with why it was
raised, gives the remedy as one call per rule, and says the reply may go
out unchanged — so a reader who finds the rules beside the point is out in
as many calls as there are rules, and is never held twice.
"""
if not held:
return ""
parts = []
for item in held:
why = (f"mounted on {item['moment']}" if item.get("moment")
else f"scores {item['score']} against this reply")
trigger = f"; it applies when {item['trigger']}" if item.get("trigger") else ""
parts.append(f"“{item['title']}” ({why}{trigger}) — get_rule({item['rule_id']})")
noun = "a standing rule" if len(held) == 1 else "standing rules"
# A mount that does not bear on this reply is a misfire worth one call:
# reports gather into an unmount proposal for the operator (step 7b).
mounted = [item for item in held if item.get("moment")]
misfire = (
" If one mounted on a moment has nothing to do with this reply, "
f"rule_misfired({mounted[0]['rule_id']}, \"{mounted[0]['moment']}\", why) "
"records that."
if mounted else ""
)
return (
f"Held for one read before this reply goes out: {noun} this session "
f"has not opened: " + "; ".join(parts) + ". Read "
+ ("it" if len(held) == 1 else "them")
+ ", then send the reply — unchanged if it already does what the rule "
"asks, which is a judgement only you can make." + misfire
+ " This check runs once per rule; the rewrite is never held."
)
async def reply_hold(
user_id: int, reply: str, *, project_id: int | None = None,
exclude: frozenset[int] = frozenset(), held: frozenset[int] = frozenset(),
stopped: frozenset[int] = frozenset(),
) -> dict:
"""Whether this finished reply is held, and in what words.
`held` is what the session OPENED (the act checkpoint's exemption);
`stopped` is what an earlier hold already put in front of it, so a rule
holds a session once. Past the act checkpoint's per-session cap nothing
holds — a mis-set bar degrades to a quiet session, never a stuck one.
Returns {} or {"reason", "rule_ids", "moments"}. Fails open to {}.
"""
from scribe.services import plugin_context as pc
from scribe.services import rule_usage, rulebooks
from scribe.services.retrieval_surfaces import budget_for, floor_for
try:
if not (reply or "").strip() or len(stopped) >= pc.CHECKPOINT_SESSION_CAP:
return {}
reached = reply_moments(reply)
names = [hit["moment"] for hit in reached]
skip = set(held) | set(stopped)
room = pc.CHECKPOINT_SESSION_CAP - len(stopped)
out: list[dict] = []
# The mounted half: deterministic, so every unopened RULE on these
# moments holds. A preference claims no such force.
for rule, at in await rulebooks.rules_on_moments(user_id, names, project_id or None):
if rule.id in skip or rule.kind == "preference":
continue
out.append({"rule_id": rule.id, "title": rule.title, "moment": at,
"trigger": (rule.when_to_apply or "").strip()})
# Trimmed BEFORE recording: a rule the cap cut was shown to nobody.
out = out[:room]
mounted = [item["rule_id"] for item in out]
fresh = [rid for rid in mounted if rid not in exclude]
if fresh:
try:
rule_usage.record_rule_surfaced(
user_id=user_id, rule_ids=fresh, source=rp.MOMENT_RULE_SOURCE,
detail={item["rule_id"]: item["moment"] for item in out
if item["rule_id"] in fresh},
)
except Exception: # noqa: BLE001 - observation never breaks the observed
logger.debug("reply mount surfacing not recorded", exc_info=True)
# The semantic half: the backstop. A mounted rule already holding is
# treated as held here, so one rule is never named twice.
floor = await floor_for(user_id, rp.REPLY_RULE.source)
result = await rp.run_rule_arm(
rp.REPLY_RULE,
rp.RuleMoment(
user_id=user_id, query=_reply_query(reply), project_id=project_id,
where="to this reply", checkpoint_where="this reply",
exclude=exclude, held=frozenset(skip | set(mounted)),
),
floor=floor, budget=await budget_for(user_id, rp.REPLY_RULE.source),
io=pc._rule_io(), checkpoint_floor=floor,
)
if result.checkpoint and len(out) < room:
cp = result.checkpoint
out.append({"rule_id": cp["rule_id"], "title": cp["title"],
"score": cp["score"], "trigger": cp.get("trigger", "")})
if not out:
return {}
return {
"reason": reply_hold_reason(out),
"rule_ids": [item["rule_id"] for item in out],
"moments": names,
}
except Exception: # noqa: BLE001 - a recall aid never stops a session by failing
logger.debug("reply hold failed", exc_info=True)
return {}