Rules mount on moments, and one retrieval pipeline (milestones 458 steps 1–4, 456 steps 1–3) #201

Merged
bvandeusen merged 11 commits from dev into main 2026-10-05 12:41:56 -04:00
57 changed files with 4733 additions and 866 deletions
+47
View File
@@ -0,0 +1,47 @@
"""moment_mappings — an install's own corrections to which actions reach which
moment (milestone 458 step 2)
Revision ID: 0116
Revises: 0115
Create Date: 2026-10-05
The moments and their default actions are code; this table holds what an
install adds ("make ship" is a deliver here) and the defaults it switches off.
User-owned configuration, so it cascades with the user. No CHECK on `effect`,
which the service validates (rule 36).
"""
import sqlalchemy as sa
from alembic import op
revision = "0116"
down_revision = "0115"
branch_labels = None
depends_on = None
def upgrade() -> None:
op.create_table(
"moment_mappings",
sa.Column("id", sa.BigInteger(), primary_key=True),
sa.Column(
"created_at", sa.DateTime(timezone=True), nullable=False,
server_default=sa.text("now()"),
),
sa.Column(
"user_id", sa.Integer(),
sa.ForeignKey("users.id", ondelete="CASCADE"), nullable=False,
),
sa.Column("tool", sa.Text(), nullable=False),
sa.Column("match", sa.Text(), nullable=False, server_default=""),
sa.Column("moment", sa.Text(), nullable=False),
sa.Column("effect", sa.Text(), nullable=False, server_default="add"),
sa.Column("reason", sa.Text(), nullable=False, server_default=""),
sa.Column("actor", sa.Text(), nullable=False, server_default="model"),
sa.UniqueConstraint(
"user_id", "tool", "match", "moment", name="uq_moment_mapping",
),
)
def downgrade() -> None:
op.drop_table("moment_mappings")
+38
View File
@@ -0,0 +1,38 @@
"""rule_moments — the moments of work a rule arrives at (milestone 458 step 3)
Revision ID: 0117
Revises: 0116
Create Date: 2026-10-05
A rule mounted on a moment is delivered whenever that moment happens, by
lookup rather than by score. The moment is a catalog name, validated by the
service; the catalog is code, so there is no table to key it to. Cascades
with the rule, like rule_systems beside it.
"""
import sqlalchemy as sa
from alembic import op
revision = "0117"
down_revision = "0116"
branch_labels = None
depends_on = None
def upgrade() -> None:
op.create_table(
"rule_moments",
sa.Column(
"rule_id", sa.BigInteger(),
sa.ForeignKey("rules.id", ondelete="CASCADE"), primary_key=True,
),
sa.Column("moment", sa.Text(), primary_key=True),
sa.Column("created_at", sa.DateTime(timezone=True), nullable=True),
)
# The delivery read (step 4) asks "which rules are on this moment?", the
# opposite direction from the primary key.
op.create_index("ix_rule_moments_moment", "rule_moments", ["moment"])
def downgrade() -> None:
op.drop_index("ix_rule_moments_moment", table_name="rule_moments")
op.drop_table("rule_moments")
+54
View File
@@ -98,6 +98,9 @@ const kbToolRuleThreshold = ref("0.68");
// prose rather than anything a tool produced (#3852). // prose rather than anything a tool produced (#3852).
const kbPromptRuleThreshold = ref("0.72"); const kbPromptRuleThreshold = ref("0.72");
const kbReportPrefThreshold = ref("0.72"); const kbReportPrefThreshold = ref("0.72");
// The reply arm (milestone 458): its bar is the bar at which a finished reply
// is HELD for one read, so it sits with the checkpoint, not with the hints.
const kbReplyRuleThreshold = ref("0.8");
// The BUDGETS, one per arm (#4102). Until this step only auto-inject had one // The BUDGETS, one per arm (#4102). Until this step only auto-inject had one
// and every other arm's ceiling was a module constant nobody could reach — so // and every other arm's ceiling was a module constant nobody could reach — so
// the only control an operator had over a noisy surface was to raise its bar, // the only control an operator had over a noisy surface was to raise its bar,
@@ -108,6 +111,7 @@ const kbRuleHintTopK = ref("5");
const kbToolRuleTopK = ref("5"); const kbToolRuleTopK = ref("5");
const kbPromptRuleTopK = ref("3"); const kbPromptRuleTopK = ref("3");
const kbReportPrefTopK = ref("3"); const kbReportPrefTopK = ref("3");
const kbReplyRuleTopK = ref("1");
// What has been changed about retrieval, newest first — the review surface for // What has been changed about retrieval, newest first — the review surface for
// changes the model made on the operator's behalf (#4102). // changes the model made on the operator's behalf (#4102).
const tuningEvents = ref<TuningEvent[]>([]); const tuningEvents = ref<TuningEvent[]>([]);
@@ -313,6 +317,8 @@ async function saveKbInject() {
// hold the first command of every session behind whatever ranked first. // hold the first command of every session behind whatever ranked first.
const cpT = asBar(kbCheckpointThreshold.value, 0.8); const cpT = asBar(kbCheckpointThreshold.value, 0.8);
const rpT = asBar(kbReportPrefThreshold.value, 0.72); const rpT = asBar(kbReportPrefThreshold.value, 0.72);
// A stop bar like the checkpoint's: a fallback of 0 would hold every reply.
const ryT = asBar(kbReplyRuleThreshold.value, 0.8);
// The budgets, clamped the way the server clamps them: a whole number in // The budgets, clamped the way the server clamps them: a whole number in
// [1, 10]. Never 0 — an arm turned off is turned off by its switch, and a // [1, 10]. Never 0 — an arm turned off is turned off by its switch, and a
// budget of zero would run the search, log the retrieval and render nothing, // budget of zero would run the search, log the retrieval and render nothing,
@@ -324,11 +330,13 @@ async function saveKbInject() {
const trK = asK(kbToolRuleTopK.value, 5); const trK = asK(kbToolRuleTopK.value, 5);
const prK = asK(kbPromptRuleTopK.value, 3); const prK = asK(kbPromptRuleTopK.value, 3);
const rpK = asK(kbReportPrefTopK.value, 3); const rpK = asK(kbReportPrefTopK.value, 3);
const ryK = asK(kbReplyRuleTopK.value, 1);
kbWritePathTopK.value = String(wpK); kbWritePathTopK.value = String(wpK);
kbRuleHintTopK.value = String(rhK); kbRuleHintTopK.value = String(rhK);
kbToolRuleTopK.value = String(trK); kbToolRuleTopK.value = String(trK);
kbPromptRuleTopK.value = String(prK); kbPromptRuleTopK.value = String(prK);
kbReportPrefTopK.value = String(rpK); kbReportPrefTopK.value = String(rpK);
kbReplyRuleTopK.value = String(ryK);
kbInjectThreshold.value = String(t); kbInjectThreshold.value = String(t);
kbInjectTopK.value = String(k); kbInjectTopK.value = String(k);
kbDupThresholdSnippet.value = String(dupSnip); kbDupThresholdSnippet.value = String(dupSnip);
@@ -346,6 +354,7 @@ async function saveKbInject() {
kbToolRuleThreshold.value = String(trT); kbToolRuleThreshold.value = String(trT);
kbPromptRuleThreshold.value = String(prT); kbPromptRuleThreshold.value = String(prT);
kbReportPrefThreshold.value = String(rpT); kbReportPrefThreshold.value = String(rpT);
kbReplyRuleThreshold.value = String(ryT);
savingKbInject.value = true; savingKbInject.value = true;
kbInjectSaved.value = false; kbInjectSaved.value = false;
try { try {
@@ -379,6 +388,9 @@ async function saveKbInject() {
// and a constant that lands under the bar is a dead arm, not a quiet // and a constant that lands under the bar is a dead arm, not a quiet
// one (#3860). // one (#3860).
kb_reportpref_threshold: String(rpT), kb_reportpref_threshold: String(rpT),
// The reply backstop's own bar: it HOLDS a reply, so like the
// checkpoint it is never derived from a hint bar.
kb_replyrule_threshold: String(ryT),
// The budgets. Every one of these keys is recognised by the server as a // The budgets. Every one of these keys is recognised by the server as a
// retrieval dial, so this save is recorded in the tuning history as a // retrieval dial, so this save is recorded in the tuning history as a
// change the OPERATOR made — which is the one entry the model must not // change the OPERATOR made — which is the one entry the model must not
@@ -388,6 +400,7 @@ async function saveKbInject() {
kb_toolrule_top_k: String(trK), kb_toolrule_top_k: String(trK),
kb_promptrule_top_k: String(prK), kb_promptrule_top_k: String(prK),
kb_reportpref_top_k: String(rpK), kb_reportpref_top_k: String(rpK),
kb_replyrule_top_k: String(ryK),
kb_duplicate_threshold_snippet: String(dupSnip), kb_duplicate_threshold_snippet: String(dupSnip),
kb_duplicate_threshold_note: String(dupNote), kb_duplicate_threshold_note: String(dupNote),
kb_duplicate_threshold_task: String(dupTask), kb_duplicate_threshold_task: String(dupTask),
@@ -859,6 +872,9 @@ onMounted(async () => {
if (allSettings.kb_reportpref_threshold !== undefined) { if (allSettings.kb_reportpref_threshold !== undefined) {
kbReportPrefThreshold.value = allSettings.kb_reportpref_threshold; kbReportPrefThreshold.value = allSettings.kb_reportpref_threshold;
} }
if (allSettings.kb_replyrule_threshold !== undefined) {
kbReplyRuleThreshold.value = allSettings.kb_replyrule_threshold;
}
if (allSettings.kb_writepath_threshold !== undefined) { if (allSettings.kb_writepath_threshold !== undefined) {
kbWritePathThreshold.value = allSettings.kb_writepath_threshold; kbWritePathThreshold.value = allSettings.kb_writepath_threshold;
} }
@@ -882,6 +898,9 @@ onMounted(async () => {
if (allSettings.kb_reportpref_top_k !== undefined) { if (allSettings.kb_reportpref_top_k !== undefined) {
kbReportPrefTopK.value = allSettings.kb_reportpref_top_k; kbReportPrefTopK.value = allSettings.kb_reportpref_top_k;
} }
if (allSettings.kb_replyrule_top_k !== undefined) {
kbReplyRuleTopK.value = allSettings.kb_replyrule_top_k;
}
await loadTuningHistory(); await loadTuningHistory();
await loadSurfaces(); await loadSurfaces();
if (allSettings.kb_duplicate_threshold_snippet !== undefined) { if (allSettings.kb_duplicate_threshold_snippet !== undefined) {
@@ -1919,6 +1938,41 @@ async function deleteUser(userId: number) {
/> />
<p class="field-hint">How many preferences a finished task may be shown (1–10).</p> <p class="field-hint">How many preferences a finished task may be shown (1–10).</p>
</div> </div>
<div class="field">
<label for="kb-replyrule-threshold">Reply check threshold (0–1)</label>
<input
id="kb-replyrule-threshold"
v-model="kbReplyRuleThreshold"
type="number"
min="0"
max="1"
step="0.01"
class="fs-input input"
style="max-width: 8rem"
/>
<p class="field-hint">
The bar at which a <em>finished reply</em> is held for one read: the
reply is searched against every rule's trigger, and a rule this
session has not opened that scores at or above this bar holds the
reply once, naming the rule. It is the backstop for whatever the
earlier checks missed, so it is a stop bar like the one above, not a
hint bar — keep it high.
</p>
</div>
<div class="field">
<label for="kb-replyrule-topk">Rules searched per reply</label>
<input
id="kb-replyrule-topk"
v-model="kbReplyRuleTopK"
type="number"
min="1"
max="10"
step="1"
class="fs-input input"
style="max-width: 8rem"
/>
<p class="field-hint">How many of the best-scoring rules a reply is checked against (1–10). Only the best can hold the reply; the rest are counted, which is what tells you how close the others came.</p>
</div>
<!-- THE REVIEW SURFACE (#4102). These numbers are maintained by the <!-- THE REVIEW SURFACE (#4102). These numbers are maintained by the
model that uses them: it reads which records each bar refused and model that uses them: it reads which records each bar refused and
moves the dial with the argument attached. This panel is the other moves the dial with the argument attached. This panel is the other
+1 -1
View File
@@ -1,7 +1,7 @@
{ {
"name": "scribe", "name": "scribe",
"description": "Scribe for Claude Code: connects the scribe MCP server, adds the hooks that deliver live project state and relevant records at the right moment, ships the shared client-neutral Scribe skills (using-scribe, writing-plans, reporting-back, systematic-debugging, verification, brainstorming, reusing-code, shape-accounting), and syncs your saved Scribe Processes as skills (/scribe:sync).", "description": "Scribe for Claude Code: connects the scribe MCP server, adds the hooks that deliver live project state and relevant records at the right moment, ships the shared client-neutral Scribe skills (using-scribe, writing-plans, reporting-back, systematic-debugging, verification, brainstorming, reusing-code, shape-accounting), and syncs your saved Scribe Processes as skills (/scribe:sync).",
"version": "2026.10.04.0125", "version": "2026.10.05.1630",
"author": { "author": {
"name": "Bryan Van Deusen" "name": "Bryan Van Deusen"
}, },
+2
View File
@@ -39,7 +39,9 @@ another one means adding files, not moving or rewriting any.
| `hooks/scribe_prior_art.sh` | PreToolUse on editor writes: `GET /api/plugin/prior-art`. | | `hooks/scribe_prior_art.sh` | PreToolUse on editor writes: `GET /api/plugin/prior-art`. |
| `hooks/scribe_after_write.sh` | PostToolUse on shell commands: the same check for code written through the shell. | | `hooks/scribe_after_write.sh` | PostToolUse on shell commands: the same check for code written through the shell. |
| `hooks/scribe_tool_rules.sh` | PreToolUse on shell commands: `GET /api/plugin/tool-rules`. | | `hooks/scribe_tool_rules.sh` | PreToolUse on shell commands: `GET /api/plugin/tool-rules`. |
| `hooks/scribe_moment.sh` | PreToolUse on every tool: the rules mounted on the moments the call reaches, `POST /api/plugin/moment`; skips tools `GET /api/plugin/moment-tools` says reach nothing mounted, and Scribe's own (their responses carry `moment_rules`). |
| `hooks/scribe_report_check.sh` | Stop: when the turn closed a task, checks the reply for the completion sections and reports to `GET /api/plugin/report-check`; blocks once, with the reason the server returns. | | `hooks/scribe_report_check.sh` | Stop: when the turn closed a task, checks the reply for the completion sections and reports to `GET /api/plugin/report-check`; blocks once, with the reason the server returns. |
| `hooks/scribe_reply_check.sh` | Stop: sends the finished reply to `POST /api/plugin/reply-rules` — the rules mounted on the reply moments, and the reply against every rule's trigger; holds once per rule, with the reason the server returns, and never holds the rewrite. |
| `hooks/scribe_shape_check.sh` | Stop: sends the definitions the turn wrote (the write hooks' `<sid>.written.ids` ledger) to `GET /api/plugin/shape-check`; blocks once, with the reason the server returns, so the agent judges what it built. | | `hooks/scribe_shape_check.sh` | Stop: sends the definitions the turn wrote (the write hooks' `<sid>.written.ids` ledger) to `GET /api/plugin/shape-check`; blocks once, with the reason the server returns, so the agent judges what it built. |
| `hooks/scribe_sync_processes.sh` + `commands/sync.md` | `GET /api/plugin/processes` → `~/.claude/skills/scribe-proc-*` stubs; `/scribe:sync` on demand. | | `hooks/scribe_sync_processes.sh` + `commands/sync.md` | `GET /api/plugin/processes` → `~/.claude/skills/scribe-proc-*` stubs; `/scribe:sync` on demand. |
| `hooks/scribe_defs.sh` | Shared shell helpers: config, dedup ledgers, outage line. | | `hooks/scribe_defs.sh` | Shared shell helpers: config, dedup ledgers, outage line. |
+13
View File
@@ -42,6 +42,15 @@
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_tool_rules.sh\"" "command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_tool_rules.sh\""
} }
] ]
},
{
"matcher": "*",
"hooks": [
{
"type": "command",
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_moment.sh\""
}
]
} }
], ],
"PostToolUse": [ "PostToolUse": [
@@ -99,6 +108,10 @@
"type": "command", "type": "command",
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_report_check.sh\"" "command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_report_check.sh\""
}, },
{
"type": "command",
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_reply_check.sh\""
},
{ {
"type": "command", "type": "command",
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_shape_check.sh\"" "command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_shape_check.sh\""
+68
View File
@@ -1226,6 +1226,74 @@ scribe_clear_session_ledgers() {
return 0 return 0
} }
# ---------------------------------------------------------------------------
# WHAT HAPPENED IN THE LAST TURN of a transcript (#4107): scribe_turn.awk's
# facts — bounded, closed, task_ids, reply — one `KEY<TAB>VALUE` per line.
# Shared by the two Stop hooks that read a turn: the report check and the
# reply check (milestone 458). Empty output when the turn cannot be bounded.
#
# The turn is parsed once. A window of recent lines, flattened record by
# record and then read by scribe_turn.awk, which carries the turn-bounding
# rules. A line that does not parse is dropped and the rest are still read —
# the first line of a `tail -n 3000` window is routinely half a record. If the
# window holds no prompt, the turn cannot be bounded, and the caller says
# nothing rather than guessing.
#
# WHERE THE TURN STARTS, FOUND BEFORE PARSING RATHER THAN AFTER. The window is
# 3000 lines and routinely 7MB, of which a turn is the last few hundred lines
# and about a sixth of the bytes — the rest is tool results these checks never
# look at. The predecessor parsed all of it and threw most away, which jq
# could afford and a parser written in awk cannot: measured at 6.5s for a 7MB
# window against 94ms, on a hook that runs at the end of every turn.
#
# So grep — C, and reading a FIXED string — narrows first. A prompt record is
# `"type":"user"` whose `content` is a STRING; a tool result is the same type
# with an ARRAY, and the two are told apart by the character after `"content":`.
#
# WHY A FIXED STRING IS EXACT HERE, and not the usual regex-over-JSON guess.
# Every quote inside a JSON string is backslash-escaped, so a needle carrying
# UNESCAPED quotes cannot occur inside any string value — it can only match at
# a record's own top level. `"message":{"role":"user","content":"` therefore
# matches real prompt records and nothing else. Measured over a 27MB transcript
# against a full JSON parse: 152 prompt records, 152 matches, no misses and no
# extras. The looser `"content":"` matched 1101 lines, because a tool_result
# block has a `content` key of its own — which is the trap this avoids.
#
# THE LAST MATCH, not a few before it, because the margin is not free: the
# lines between two prompts are mostly tool results, and backing off three
# matches took the window from 53KB to 1.3MB and the parse from 21ms to 2.6s.
# The fallback below is the safety net instead — it is exact where a margin is
# only approximate, and it costs nothing in the case that actually happens.
#
# The needle assumes a key ORDER that a future Claude Code could change. If it
# does, grep matches nothing, `start` stays 1, and the whole window is read the
# slow way — correct, and slow, which is the right way round for a check that
# can block a stop.
_scribe_turn_window() { tail -n 3000 "$1" 2>/dev/null; }
_scribe_turn_parse() {
_scribe_turn_window "$1" | tail -n +"${2:-1}" | scribe_json_flat_lines \
| awk -f "$SCRIBE_HOOK_DIR/scribe_turn.awk" 2>/dev/null
}
scribe_turn_facts() {
local transcript="$1" start facts
[ -n "$transcript" ] && [ -f "$transcript" ] || return 0
start=$(_scribe_turn_window "$transcript" \
| grep -n -F '"message":{"role":"user","content":"' 2>/dev/null \
| cut -d: -f1 | awk '{ last = $0 } END { if (NR) print last }')
case "$start" in ''|*[!0-9]*) start=1 ;; esac
facts=$(_scribe_turn_parse "$transcript" "$start")
if [ "$(scribe_turn_fact "$facts" bounded)" != "1" ] && [ "$start" != "1" ]; then
facts=$(_scribe_turn_parse "$transcript" 1)
fi
printf '%s\n' "$facts"
}
# $1 facts from scribe_turn_facts, $2 key → that fact's value, or "".
scribe_turn_fact() {
printf '%s\n' "$1" | awk -F'\t' -v k="$2" '$1 == k { print substr($0, index($0, "\t") + 1); exit }'
}
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# WHICH PROJECT IS THIS DIRECTORY'S? (#4085) # WHICH PROJECT IS THIS DIRECTORY'S? (#4085)
# #
+120
View File
@@ -0,0 +1,120 @@
#!/usr/bin/env bash
# Scribe — PreToolUse moment arm: rules MOUNTED on a moment arrive when the
# moment happens (milestone 458 step 4).
#
# The other PreToolUse arms search: they turn the act into a query and hope a
# rule's words resemble it. A mount needs no resemblance. The operator said
# "this rule applies when work is delivered", the install maps `git push` onto
# work.deliver, and so `git push` brings the rule — whatever the command's
# words happen to be. The server resolves the call to its moments and looks up
# what is mounted there; this hook only carries the event and the answer.
#
# REGISTERED ON EVERY TOOL, KEPT OFF THE WIRE FOR MOST OF THEM. Which calls can
# reach a mounted rule depends on this install's mappings and mounts, so the
# matcher cannot say. Instead the hook keeps the server's answer to "which
# tools can?" (`/moment-tools`) on disk for a few minutes, and a call to any
# other tool costs one file read. An install that has mounted nothing gets an
# empty list and never sends a moment request at all.
#
# SCRIBE'S OWN TOOLS ARE SKIPPED. Closing a task or recording a lesson reaches
# its moment through the tool's own response (`moment_rules`), which works for
# a client without this plugin too; delivering it here as well would say it
# twice.
#
# SILENT ON OUTAGE, like scribe_tool_rules.sh and for its reason: this fires
# before many calls, and an outage line before each is noise. A failed list
# fetch is cached as an empty list, so a down instance costs one timeout per
# window rather than one per call.
#
# Env:
# SCRIBE_URL / SCRIBE_TOKEN override for the settings.json dogfooding path.
command -v curl >/dev/null 2>&1 || exit 0
# shellcheck source=plugin/hooks/scribe_defs.sh
. "$(dirname "${BASH_SOURCE[0]}")/scribe_defs.sh"
event=$(cat 2>/dev/null || true)
event_flat=$(printf '%s' "$event" | scribe_json_flat)
tool_name=$(scribe_json_pick "$event_flat" '.tool_name')
session_id=$(scribe_json_pick "$event_flat" '.session_id')
event_cwd=$(scribe_json_pick "$event_flat" '.cwd')
[ -n "$tool_name" ] && [ -n "$session_id" ] || exit 0
case "$tool_name" in
mcp__*scribe*__*) exit 0 ;;
esac
scribe_config || exit 0
# The tool's KEY, as the server writes mappings: no MCP server prefix, lowercase.
tool_key=${tool_name##*__}
tool_key=$(printf '%s' "$tool_key" | tr '[:upper:]' '[:lower:]')
state_dir="${TMPDIR:-/tmp}/scribe-priorart"
mkdir -p "$state_dir" 2>/dev/null || true
safe_sid=$(printf '%s' "$session_id" | tr -c 'A-Za-z0-9._-' '_')
# ── Which tools can reach a mounted rule: cached, first line a timestamp ──
#
# Five minutes. A mount or a mapping made in this session reaches the hook
# within that window; the MCP door delivers Scribe's own moments at once.
# Its own directory, outside SCRIBE_LEDGER_DIRS, so a compaction does not
# sweep it: it describes the install, not what this context holds.
cache_dir="${TMPDIR:-/tmp}/scribe-moment"
mkdir -p "$cache_dir" 2>/dev/null || true
tools_file="$cache_dir/${safe_sid}.tools"
now=$(date +%s 2>/dev/null) || now=0
stamp=""
[ -f "$tools_file" ] && stamp=$(head -n 1 "$tools_file" 2>/dev/null)
case "$stamp" in ''|*[!0-9]*) stamp=0 ;; esac
if [ "$now" -eq 0 ] || [ $((now - stamp)) -gt 300 ]; then
listed=$(curl -fsS --max-time 3 \
-H "Authorization: Bearer ${token}" \
"${url%/}/api/plugin/moment-tools" 2>/dev/null) || listed=""
{
printf '%s\n' "$now"
scribe_json_list "$(printf '%s' "$listed" | scribe_json_flat)" '.tools'
} > "$tools_file" 2>/dev/null || true
fi
tail -n +2 "$tools_file" 2>/dev/null | grep -Fqx -- "$tool_key" || exit 0
# ── The moment request ────────────────────────────────────────────────────
repo_q=""
lookup_dir=${event_cwd:-${CLAUDE_PROJECT_DIR:-$PWD}}
scope=$(scribe_scope_query "$lookup_dir")
[ -n "$scope" ] && repo_q="&${scope}"
# The shared session ledger, as scribe_tool_rules.sh reads and writes it: a
# rule this session was already shown is cited, not quoted again.
rulefile="$state_dir/${safe_sid}.rules.ids"
ledger_q=""
rule_seen=$(scribe_rules_live "$rulefile")
[ -n "$rule_seen" ] && ledger_q="&exclude_rule_ids=${rule_seen}"
ledger_q="${ledger_q}$(scribe_held_query "$state_dir/${safe_sid}.opened.ids")"
# The event whole, since which field a mapping matches is the mapping's
# business. A write of a very large file is reduced to its name: the moments a
# write reaches never depend on its content.
if [ "${#event}" -gt 65536 ]; then
esc=$(printf '%s' "$tool_name" | scribe_json_escape) || exit 0
event="{\"tool_name\":\"${esc}\",\"tool_input\":{}}"
fi
query="${repo_q}${ledger_q}"
query=${query#&}
body=$(printf '%s' "$event" | curl -fsS --max-time 3 \
-H "Authorization: Bearer ${token}" \
-H "Content-Type: application/json" \
--data-binary @- \
"${url%/}/api/plugin/moment${query:+?$query}" 2>/dev/null) || exit 0
body_flat=$(printf '%s' "$body" | scribe_json_flat)
scribe_json_list "$body_flat" '.rule_ids' | scribe_rules_append "$rulefile"
context=$(scribe_json_pick "$body_flat" '.context')
[ -n "$context" ] || exit 0
scribe_json_out PreToolUse "$context"
exit 0
+91
View File
@@ -0,0 +1,91 @@
#!/usr/bin/env bash
# Scribe — Stop hook: the reply moment (milestone 458 step 4, folded in from
# milestone 456 step 8).
#
# A reply is the one act no tool call marks, and it is where "please check
# this on your end" or "it's done" gets said — so every arm that fires on a
# tool call misses it. This hook sends the finished reply to the server, which
# checks it twice over: the rules MOUNTED on the reply moments, and the reply
# text against every rule's trigger, as the backstop for whatever the earlier
# arms missed. An unopened rule from either half holds the reply for one read.
#
# THE SAME CONTRACT AS scribe_report_check.sh, and for its reasons:
# - the server decides and supplies the words; this hook blocks only on a
# reason it was given, so an unconfigured or unreachable instance never
# stops a session;
# - the rewrite is never held (`stop_hook_active`), so a hold costs one turn
# at most;
# - a rule holds a session once. The ledger is the act checkpoint's own
# (`<sid>.checkpoint.ids`), so a rule that already stopped a command does
# not stop the reply about it, and the per-session cap counts both doors.
#
# Env:
# SCRIBE_URL / SCRIBE_TOKEN override for the settings.json dogfooding path.
command -v curl >/dev/null 2>&1 || exit 0
# shellcheck source=plugin/hooks/scribe_defs.sh
. "$(dirname "${BASH_SOURCE[0]}")/scribe_defs.sh"
event=$(cat 2>/dev/null || true)
event_flat=$(printf '%s' "$event" | scribe_json_flat)
transcript=$(scribe_json_pick "$event_flat" '.transcript_path')
session_id=$(scribe_json_pick "$event_flat" '.session_id')
active=$(scribe_json_pick "$event_flat" '.stop_hook_active')
event_cwd=$(scribe_json_pick "$event_flat" '.cwd')
# The rewrite after a hold goes out as written.
[ "$active" = "true" ] && exit 0
[ -n "$transcript" ] && [ -f "$transcript" ] && [ -n "$session_id" ] || exit 0
scribe_config || exit 0
facts=$(scribe_turn_facts "$transcript")
[ "$(scribe_turn_fact "$facts" bounded)" = "1" ] || exit 0
reply=$(scribe_turn_fact "$facts" reply | scribe_json_unescape)
[ -n "$(printf '%s' "$reply" | tr -d '[:space:]')" ] || exit 0
# Bounded before encoding. The server reads the head and the tail — the part
# of a report that asks something of the reader is at its end — so a very
# long reply keeps both.
if [ "${#reply}" -gt 12000 ]; then
reply="${reply:0:4000} … ${reply: -8000}"
fi
reply_esc=$(printf '%s' "$reply" | scribe_json_escape) || exit 0
state_dir="${TMPDIR:-/tmp}/scribe-priorart"
mkdir -p "$state_dir" 2>/dev/null || true
safe_sid=$(printf '%s' "$session_id" | tr -c 'A-Za-z0-9._-' '_')
rulefile="$state_dir/${safe_sid}.rules.ids"
stopfile="$state_dir/${safe_sid}.checkpoint.ids"
query=""
scope=$(scribe_scope_query "${event_cwd:-${CLAUDE_PROJECT_DIR:-$PWD}}")
[ -n "$scope" ] && query="&${scope}"
rule_seen=$(scribe_rules_live "$rulefile")
[ -n "$rule_seen" ] && query="${query}&exclude_rule_ids=${rule_seen}"
query="${query}$(scribe_held_query "$state_dir/${safe_sid}.opened.ids")"
if [ -f "$stopfile" ]; then
stopped=$(grep -E '^[0-9]+$' "$stopfile" 2>/dev/null | paste -sd, -)
[ -n "$stopped" ] && query="${query}&stopped_rule_ids=${stopped}"
fi
query=${query#&}
answer=$(printf '{"reply":"%s"}' "$reply_esc" | curl -fsS --max-time 6 \
-H "Authorization: Bearer ${token}" \
-H "Content-Type: application/json" \
--data-binary @- \
"${url%/}/api/plugin/reply-rules${query:+?$query}" 2>/dev/null) || exit 0
answer_flat=$(printf '%s' "$answer" | scribe_json_flat)
reason=$(scribe_json_pick "$answer_flat" '.reason')
[ -n "$reason" ] || exit 0
# Recorded BEFORE the block is emitted, for the act checkpoint's reason: a
# hold that is shown and not recorded is one that can be shown again.
for id in $(scribe_json_list "$answer_flat" '.rule_ids'); do
scribe_checkpoint_allowed "$stopfile" "$id" || true
done
scribe_json_list "$answer_flat" '.rule_ids' | scribe_rules_append "$rulefile"
printf '{"decision":"block","reason":"%s"}\n' "$(printf '%s' "$reason" | scribe_json_escape)"
exit 0
+3 -50
View File
@@ -78,56 +78,9 @@ grep -q -E '"name":[[:space:]]*"([^"]*__)?(update|create)_task"' < <(tail -c 200
exit 0 exit 0
} }
# The turn, parsed once. A window of recent lines, flattened record by record # The turn, parsed once — scribe_turn_facts, shared with the reply check.
# and then read by scribe_turn.awk, which carries the turn-bounding rules. A facts=$(scribe_turn_facts "$transcript")
# line that does not parse is dropped and the rest are still read — the first fact() { scribe_turn_fact "$facts" "$1"; }
# line of a `tail -n 3000` window is routinely half a record. If the window
# holds no prompt, the turn cannot be bounded, so the hook reports nothing and
# stays out of the way.
window() { tail -n 3000 "$transcript" 2>/dev/null; }
turn_facts() { window | tail -n +"${1:-1}" | scribe_json_flat_lines \
| awk -f "$SCRIBE_HOOK_DIR/scribe_turn.awk" 2>/dev/null; }
# WHERE THE TURN STARTS, FOUND BEFORE PARSING RATHER THAN AFTER. The window is
# 3000 lines and routinely 7MB, of which a turn is the last few hundred lines
# and about a sixth of the bytes — the rest is tool results this check never
# looks at. The predecessor parsed all of it and threw most away, which jq
# could afford and a parser written in awk cannot: measured at 6.5s for a 7MB
# window against 94ms, on a hook that runs at the end of every turn.
#
# So grep — C, and reading a FIXED string — narrows first. A prompt record is
# `"type":"user"` whose `content` is a STRING; a tool result is the same type
# with an ARRAY, and the two are told apart by the character after `"content":`.
#
# WHY A FIXED STRING IS EXACT HERE, and not the usual regex-over-JSON guess.
# Every quote inside a JSON string is backslash-escaped, so a needle carrying
# UNESCAPED quotes cannot occur inside any string value — it can only match at
# a record's own top level. `"message":{"role":"user","content":"` therefore
# matches real prompt records and nothing else. Measured over a 27MB transcript
# against a full JSON parse: 152 prompt records, 152 matches, no misses and no
# extras. The looser `"content":"` matched 1101 lines, because a tool_result
# block has a `content` key of its own — which is the trap this avoids.
#
# THE LAST MATCH, not a few before it, because the margin is not free: the
# lines between two prompts are mostly tool results, and backing off three
# matches took the window from 53KB to 1.3MB and the parse from 21ms to 2.6s.
# The fallback below is the safety net instead — it is exact where a margin is
# only approximate, and it costs nothing in the case that actually happens.
#
# The needle assumes a key ORDER that a future Claude Code could change. If it
# does, grep matches nothing, `start` stays 1, and the whole window is read the
# slow way — correct, and slow, which is the right way round for a check that
# can block a stop.
start=$(window | grep -n -F '"message":{"role":"user","content":"' 2>/dev/null \
| cut -d: -f1 | awk '{ last = $0 } END { if (NR) print last }')
case "$start" in ''|*[!0-9]*) start=1 ;; esac
facts=$(turn_facts "$start")
fact() { printf '%s\n' "$facts" | awk -F'\t' -v k="$1" '$1 == k { print substr($0, index($0, "\t") + 1); exit }'; }
if [ "$(fact bounded)" != "1" ] && [ "$start" != "1" ]; then
facts=$(turn_facts 1)
fi
[ "$(fact bounded)" = "1" ] || exit 0 [ "$(fact bounded)" = "1" ] || exit 0
closed=$(fact closed) closed=$(fact closed)
+14
View File
@@ -341,6 +341,14 @@ SMOKE_EVENTS: dict[str, str] = {
{"session_id": "smoke", "cwd": ".", "tool_name": "Bash", {"session_id": "smoke", "cwd": ".", "tool_name": "Bash",
"tool_input": {"command": "curl -s https://example.invalid/api/v1/runs"}} "tool_input": {"command": "curl -s https://example.invalid/api/v1/runs"}}
), ),
# The moment arm (milestone 458). A command the shipped mappings send to
# work.deliver, so a configured run reaches the moment-tools fetch and the
# moment request; with no instance both fail and it must stay SILENT,
# since it fires before every call that can reach a mounted rule.
"scribe_moment.sh": json.dumps(
{"session_id": "smoke", "cwd": ".", "tool_name": "Bash",
"tool_input": {"command": "git push origin dev"}}
),
"scribe_sync_processes.sh": json.dumps({"source": "startup"}), "scribe_sync_processes.sh": json.dumps({"source": "startup"}),
"scribe_session_context.sh": json.dumps({"source": "startup"}), "scribe_session_context.sh": json.dumps({"source": "startup"}),
# The after-write hook (#2901) diffs the working tree; on CI's clean # The after-write hook (#2901) diffs the working tree; on CI's clean
@@ -358,6 +366,12 @@ SMOKE_EVENTS: dict[str, str] = {
{"session_id": "smoke", "transcript_path": "/nonexistent/smoke.jsonl", {"session_id": "smoke", "transcript_path": "/nonexistent/smoke.jsonl",
"cwd": ".", "hook_event_name": "Stop", "stop_hook_active": False} "cwd": ".", "hook_event_name": "Stop", "stop_hook_active": False}
), ),
# The reply moment (milestone 458). No transcript, so no reply: silent and
# never a block, configured or not.
"scribe_reply_check.sh": json.dumps(
{"session_id": "smoke", "transcript_path": "/nonexistent/smoke.jsonl",
"cwd": ".", "hook_event_name": "Stop", "stop_hook_active": False}
),
# The end-of-turn shape check (milestone 439). With no ledger for the # The end-of-turn shape check (milestone 439). With no ledger for the
# session there is nothing to ask about, and with no instance there is # session there is nothing to ask about, and with no instance there is
# nobody to ask: silent, and never a block, configured or not. # nobody to ask: silent, and never a block, configured or not.
+8
View File
@@ -154,6 +154,11 @@ _READ_ONLY_TOOLS = frozenset({
# most: it is the surface the "ask before acting" reflex calls, and a key # most: it is the surface the "ask before acting" reflex calls, and a key
# that could not reach it would be denied exactly the check it should run. # that could not reach it would be denied exactly the check it should run.
"what_might_apply", "what_might_apply",
# The moment catalog (milestone 458). A pure read of a constant; the
# mount and mapping writes that use it are separate tools. Spelled out for
# retrieval_telemetry's reason, and needed by a read key so that a line
# naming a moment can be understood by whoever was shown it.
"list_moments",
}) })
# Every tool that WRITES, by name. Nothing reads this set at runtime — a tool # Every tool that WRITES, by name. Nothing reads this set at runtime — a tool
@@ -202,6 +207,9 @@ _WRITE_TOOLS = frozenset({
# reads, and it appends the reason to the audit trail (#4102). # reads, and it appends the reason to the audit trail (#4102).
"tune_retrieval", "tune_retrieval",
"migrate_retrieval_floor", "migrate_retrieval_floor",
# Which actions reach which moment on this install (milestone 458). Each
# changes what fires for every later session, so a read key is refused.
"map_action", "unmap_action",
# A reviewer's verdicts on logged menu lines (#4772) — rows carrying free # A reviewer's verdicts on logged menu lines (#4772) — rows carrying free
# prose the agent authored, `rule_outcome`'s reason for being a write. # prose the agent authored, `rule_outcome`'s reason for being a write.
"judge_menu", "judge_menu",
+2 -1
View File
@@ -6,7 +6,7 @@ from `mcp.server.build_mcp_server`.
""" """
from scribe.mcp.tools import ( from scribe.mcp.tools import (
design_systems, lessons, milestones, notes, processes, projects, recent, repos, design_systems, lessons, milestones, notes, processes, projects, recent, repos,
retrieval_review, retrieval_tuning, moments, retrieval_review, retrieval_tuning,
wide_net, wide_net,
rulebooks, search, shapes, snippets, systems, tags, tasks, trash, rulebooks, search, shapes, snippets, systems, tags, tasks, trash,
) )
@@ -18,6 +18,7 @@ def register_all(mcp) -> None:
retrieval_tuning.register(mcp) retrieval_tuning.register(mcp)
retrieval_review.register(mcp) retrieval_review.register(mcp)
wide_net.register(mcp) wide_net.register(mcp)
moments.register(mcp)
notes.register(mcp) notes.register(mcp)
tasks.register(mcp) tasks.register(mcp)
projects.register(mcp) projects.register(mcp)
+2 -1
View File
@@ -19,6 +19,7 @@ from scribe.services import systems as systems_svc
from scribe.services import trash as trash_svc from scribe.services import trash as trash_svc
from scribe.mcp.tools import systems as systems_tools from scribe.mcp.tools import systems as systems_tools
from scribe.services.note_usage import attach_usage, record_pulled from scribe.services.note_usage import attach_usage, record_pulled
from scribe.services import moment_delivery
# The payload shape lives in the service (`lesson_to_dict`), shared with the # The payload shape lives in the service (`lesson_to_dict`), shared with the
@@ -245,7 +246,7 @@ async def create_lesson(
await _offer_candidates(uid, data, what, when_to_apply, project_id) await _offer_candidates(uid, data, what, when_to_apply, project_id)
elif no_rule.strip(): elif no_rule.strip():
await _name_convergence(uid, data, note.id) await _name_convergence(uid, data, note.id)
return data return await moment_delivery.attach_moment_rules(uid, "create_lesson", {"project_id": project_id}, data)
async def _name_convergence(uid: int, data: dict, lesson_id: int) -> None: async def _name_convergence(uid: int, data: dict, lesson_id: int) -> None:
+8 -2
View File
@@ -19,6 +19,7 @@ from scribe.services import task_logs as task_logs_svc
from scribe.services import rulebooks as rulebooks_svc from scribe.services import rulebooks as rulebooks_svc
from scribe.services import trash as trash_svc from scribe.services import trash as trash_svc
from scribe.services.record_refs import refuse_guessed_ids from scribe.services.record_refs import refuse_guessed_ids
from scribe.services import moment_delivery
async def list_milestones(project_id: int) -> dict: async def list_milestones(project_id: int) -> dict:
@@ -132,7 +133,9 @@ async def create_milestone(
body=body or None, body=body or None,
status=status, status=status,
) )
return milestone.to_dict() return await moment_delivery.attach_moment_rules(
uid, "create_milestone", {"project_id": project_id}, milestone.to_dict(),
)
async def update_milestone( async def update_milestone(
@@ -172,7 +175,10 @@ async def update_milestone(
milestone = await milestones_svc.update_milestone(uid, milestone_id, **fields) milestone = await milestones_svc.update_milestone(uid, milestone_id, **fields)
if milestone is None: if milestone is None:
raise ValueError(f"milestone {milestone_id} not found") raise ValueError(f"milestone {milestone_id} not found")
return milestone.to_dict() return await moment_delivery.attach_moment_rules(
uid, "update_milestone", {"status": status, "project_id": project_id},
milestone.to_dict(),
)
async def delete_milestone(milestone_id: int) -> dict: async def delete_milestone(milestone_id: int) -> dict:
+111
View File
@@ -0,0 +1,111 @@
"""Moments as MCP tools: the catalog, and the in-session corrections to it (milestone 458).
A session needs the vocabulary in hand to mount a rule, to read a line that
names a moment, and to fix a misfire — and the operator's ruling is that the
fix happens in the session, not on a settings page:
"making the user leave the session to fix a misfire is not desirable and
the llm session should be able to offer corrections"
So the mapping writes are tools, built to be offered mid-work and made on the
operator's yes.
"""
from __future__ import annotations
from scribe.mcp._context import current_user_id
from scribe.services import moment_actions as actions_svc
from scribe.services import moments as moments_svc
async def list_moments() -> dict:
"""The moments of work that rules mount on, and which actions reach each one here.
A rule mounted on a moment arrives whenever that moment happens, whatever
the words of the work look like. That is how a rule reaches you when it is
about WHEN something is done rather than WHAT it is about: a rule on
finishing work belongs at `work.finish` and `reply.report`, where nothing
said needs to resemble it.
Read this when you are about to mount a rule, when a line you were shown
names a moment and you want its meaning, or when an action reached the
wrong moment — or none — and you are about to correct it with
`map_action` / `unmap_action`.
Names are `<area>.<verb>`; a named procedure (a skill or a stored process)
is its own moment, `skill.<name>`, listed under `families`. Each moment
says what is happening at it (`means`) and the kinds of action that
typically reach it (`reached_by`).
`actions` maps each moment to the actions that reach it on this install:
`via: "default"` ships with the product, `via: "install"` is this
install's own mapping. `removed_defaults` are shipped defaults this install
switched off.
"""
out = moments_svc.catalog()
out.update(await actions_svc.actions_by_moment(current_user_id()))
return out
async def map_action(tool: str, moment: str, match: str = "", reason: str = "") -> dict:
"""Make an action reach a moment on this install — the in-session fix for a missed moment.
Use it when an action plainly happened at a moment and the moment did not
fire: the operator ships with `make ship`, and nothing mounted on
`work.deliver` arrived. Offer the mapping when you notice, in one line
("that `make ship` was a deliver and nothing fired — map it?"), and make it
on their yes. The correction lasts: every later session on this install
gets it.
Args:
tool: the tool as the harness names it — `Bash`, `Edit`,
`update_task`. An MCP server prefix is ignored.
moment: a moment from `list_moments`, or `skill.<name>`.
match: which calls of the tool. Empty = every call. For a tool that
runs a command, how the command starts (`make ship`, `./deploy.sh`)
— it is checked against each part of a compound command line. For
any other tool, its arguments as `field=value` pairs, comma-separated
(`status=done`).
reason: what misfired, or what this action is for here. Optional; it
is shown beside the mapping when someone reviews it later.
Returns the change made and `now_reaches`: every moment that action
reaches after the change, so you can confirm it in the same reply.
Mapping a shipped default this install had switched off switches it back
on.
"""
return await actions_svc.map_action(
current_user_id(), tool, match, moment, reason=reason, actor="model",
)
async def unmap_action(tool: str, moment: str, match: str = "", reason: str = "") -> dict:
"""Stop an action reaching a moment on this install — the fix for a moment that fires wrongly.
Use it when a moment fires on an action that is not that moment here — a
`curl` to a local test server reaching `env.reach`, say — and the rules it
brings are noise every time. Offer it when you notice, and make it on the
operator's yes.
The arguments name the mapping exactly as `list_moments` shows it. This
install's own mapping is removed; a shipped default is switched off for
this install only, and stays off across upgrades. `map_action` with the
same arguments switches a default back on.
Args:
tool: the tool as `list_moments` names it.
moment: the moment it should stop reaching.
match: the mapping's match, as listed (empty for a whole-tool mapping).
reason: why it misfires here — worth giving for a default, since it is
the record of why this install differs from the product.
Returns the change made and `now_reaches` for that action.
"""
return await actions_svc.unmap_action(
current_user_id(), tool, match, moment, reason=reason, actor="model",
)
def register(mcp) -> None:
mcp.tool(name="list_moments")(list_moments)
mcp.tool(name="map_action")(map_action)
mcp.tool(name="unmap_action")(unmap_action)
+2 -1
View File
@@ -23,6 +23,7 @@ from scribe.services import systems as systems_svc
from scribe.services import trash as trash_svc from scribe.services import trash as trash_svc
from scribe.services.note_usage import record_pulled from scribe.services.note_usage import record_pulled
from scribe.services.record_refs import refuse_guessed_ids from scribe.services.record_refs import refuse_guessed_ids
from scribe.services import moment_delivery
async def list_notes( async def list_notes(
@@ -236,7 +237,7 @@ async def create_note(
await systems_tools.attach_systems(uid, uid, data, note.id, project_id or None) await systems_tools.attach_systems(uid, uid, data, note.id, project_id or None)
await supersession_svc.attach_relations(uid, note.id, data, hint=True) await supersession_svc.attach_relations(uid, note.id, data, hint=True)
data.update(dedup_svc.note_overlap_response(overlaps, "note")) data.update(dedup_svc.note_overlap_response(overlaps, "note"))
return data return await moment_delivery.attach_moment_rules(uid, "create_note", {"project_id": project_id}, data)
async def update_note( async def update_note(
+2 -1
View File
@@ -14,6 +14,7 @@ from scribe.services import notes as notes_svc
from scribe.services import systems as systems_svc from scribe.services import systems as systems_svc
from scribe.services import trash as trash_svc from scribe.services import trash as trash_svc
from scribe.services.note_usage import record_pulled from scribe.services.note_usage import record_pulled
from scribe.services import moment_delivery
async def list_processes( async def list_processes(
@@ -106,7 +107,7 @@ async def create_process(
) )
if system_ids: if system_ids:
await systems_svc.set_record_systems(uid, note.id, system_ids) await systems_svc.set_record_systems(uid, note.id, system_ids)
return note.to_dict() return await moment_delivery.attach_moment_rules(uid, "create_process", {}, note.to_dict())
async def get_process(name_or_id: str, project_id: int = 0) -> dict: async def get_process(name_or_id: str, project_id: int = 0) -> dict:
+57 -15
View File
@@ -16,11 +16,13 @@ from __future__ import annotations
from scribe.mcp._context import current_user_id from scribe.mcp._context import current_user_id
from scribe.services import dedup as dedup_svc from scribe.services import dedup as dedup_svc
from scribe.services import moments as moments_svc
from scribe.services import rulebooks as rulebooks_svc from scribe.services import rulebooks as rulebooks_svc
from scribe.services import trash as trash_svc from scribe.services import trash as trash_svc
from scribe.services.rule_usage import ( from scribe.services.rule_usage import (
record_rule_outcome, record_rule_pulled, record_rule_outcome, record_rule_pulled,
) )
from scribe.services import moment_delivery
# ── Rulebook CRUD ─────────────────────────────────────────────────────── # ── Rulebook CRUD ───────────────────────────────────────────────────────
@@ -192,14 +194,16 @@ async def delete_topic(topic_id: int, confirmed: bool = False) -> dict:
# ── Rule CRUD ────────────────────────────────────────────────────────── # ── Rule CRUD ──────────────────────────────────────────────────────────
def _rule_summary(r) -> dict: def _rule_summary(r, moments: list[str] | None = None) -> dict:
"""The list-row shape for a rule: what an agent needs to APPLY it. The """The list-row shape for a rule: what an agent needs to APPLY it. The
full record (why, how_to_apply, timestamps) is get_rule's job. full record (why, how_to_apply, timestamps) is get_rule's job.
One line, because the shape itself lives in the service — this was one of One line, because the shape itself lives in the service — this was one of
three hand-written copies that had already drifted apart (note 3026). three hand-written copies that had already drifted apart (note 3026).
`moments` rides along only when the rule is mounted (rule_brief drops a
None), so an unmounted rule's row says nothing rather than "[]".
""" """
return rulebooks_svc.rule_brief(r) return rulebooks_svc.rule_brief(r, moments=moments or None)
async def list_rules( async def list_rules(
@@ -224,7 +228,12 @@ async def list_rules(
topic_id=topic_id or None, topic_id=topic_id or None,
project_id=project_id or None, project_id=project_id or None,
) )
return {"rules": [_rule_summary(r) for r in rows], "total": len(rows)} # One batched read for the page, not one per row.
mounted = await rulebooks_svc.list_rule_moments([r.id for r in rows])
return {
"rules": [_rule_summary(r, mounted.get(r.id)) for r in rows],
"total": len(rows),
}
async def get_rule(rule_id: int) -> dict: async def get_rule(rule_id: int) -> dict:
@@ -311,7 +320,8 @@ async def create_rule(
topic_id: int, title: str, statement: str, when_to_apply: str, topic_id: int, title: str, statement: str, when_to_apply: str,
why: str = "", how_to_apply: str = "", order_index: int = 0, why: str = "", how_to_apply: str = "", order_index: int = 0,
arose_from_id: int = 0, verify_with: str = "", expires_when: str = "", arose_from_id: int = 0, verify_with: str = "", expires_when: str = "",
system_ids: list[int] | None = None, force: bool = False, system_ids: list[int] | None = None,
moments: list[str] | None = None, force: bool = False,
) -> dict: ) -> dict:
"""Create a new rule in a rulebook (a SHARED rule — keep it general). """Create a new rule in a rulebook (a SHARED rule — keep it general).
@@ -472,6 +482,12 @@ async def create_rule(
system_ids: Ids from list_canonical_systems — the global AREAS this system_ids: Ids from list_canonical_systems — the global AREAS this
rule is about. This is what lets a rule reach a project that is rule is about. This is what lets a rule reach a project that is
working in that area, so a CI rule surfaces on a CI change. working in that area, so a CI rule surfaces on a CI change.
moments: Moments from list_moments this rule arrives at — whenever
one happens, the rule is delivered, whatever the words of the work
look like. Mount a rule here when it is about WHEN something is
done rather than what it is about: a rule on when work counts as
finished belongs at work.finish and reply.report, where nothing
said resembles it. Replaces the set; [] clears it.
arose_from_id: The note or task that CAUSED this rule (an incident, a arose_from_id: The note or task that CAUSED this rule (an incident, a
decision). Prefer this over naming the record inside `why`, which decision). Prefer this over naming the record inside `why`, which
cannot be followed and does not survive a rewording. cannot be followed and does not survive a rewording.
@@ -503,6 +519,7 @@ async def create_rule(
`overlaps` and `overlap_note` — read the top one and decide. `overlaps` and `overlap_note` — read the top one and decide.
""" """
uid = current_user_id() uid = current_user_id()
moments = moments_svc.require_moments(moments)
if not force: if not force:
dup = await dedup_svc.find_duplicate_rule(title, topic_id=topic_id) dup = await dedup_svc.find_duplicate_rule(title, topic_id=topic_id)
if dup is not None: if dup is not None:
@@ -518,16 +535,17 @@ async def create_rule(
why=why, how_to_apply=how_to_apply, order_index=order_index, why=why, how_to_apply=how_to_apply, order_index=order_index,
verify_with=verify_with, expires_when=expires_when, verify_with=verify_with, expires_when=expires_when,
) )
data = await rulebooks_svc.rule_detail(uid, rule, system_ids) data = await rulebooks_svc.rule_detail(uid, rule, system_ids, moments)
data.update(dedup_svc.overlap_response(overlaps, "rule")) data.update(dedup_svc.overlap_response(overlaps, "rule"))
return data return await moment_delivery.attach_moment_rules(uid, "create_rule", {}, data)
async def create_project_rule( async def create_project_rule(
project_id: int, statement: str, when_to_apply: str, title: str = "", project_id: int, statement: str, when_to_apply: str, title: str = "",
why: str = "", how_to_apply: str = "", order_index: int = 0, why: str = "", how_to_apply: str = "", order_index: int = 0,
arose_from_id: int = 0, verify_with: str = "", expires_when: str = "", arose_from_id: int = 0, verify_with: str = "", expires_when: str = "",
system_ids: list[int] | None = None, force: bool = False, system_ids: list[int] | None = None,
moments: list[str] | None = None, force: bool = False,
) -> dict: ) -> dict:
"""Create a rule scoped to a single project (no rulebook needed). """Create a rule scoped to a single project (no rulebook needed).
@@ -581,6 +599,12 @@ async def create_project_rule(
rule is about. Worth setting even on a project rule: it is what rule is about. Worth setting even on a project rule: it is what
lets a conditional one surface when the project is working in lets a conditional one surface when the project is working in
that area. that area.
moments: Moments from list_moments this rule arrives at — whenever
one happens, the rule is delivered, whatever the words of the work
look like. Mount a rule here when it is about WHEN something is
done rather than what it is about: a rule on when work counts as
finished belongs at work.finish and reply.report, where nothing
said resembles it. Replaces the set; [] clears it.
arose_from_id: The note or task that CAUSED this rule. Reach for it arose_from_id: The note or task that CAUSED this rule. Reach for it
harder here than on a rulebook rule — a project rule usually harder here than on a rulebook rule — a project rule usually
comes from one traceable incident in this repo, where a family comes from one traceable incident in this repo, where a family
@@ -604,6 +628,7 @@ async def create_project_rule(
`overlap_note` on the reply — see create_rule. `overlap_note` on the reply — see create_rule.
""" """
uid = current_user_id() uid = current_user_id()
moments = moments_svc.require_moments(moments)
derived_title = title.strip() or statement.strip().split(".")[0][:50] derived_title = title.strip() or statement.strip().split(".")[0][:50]
if not force: if not force:
dup = await dedup_svc.find_duplicate_rule(derived_title, project_id=project_id) dup = await dedup_svc.find_duplicate_rule(derived_title, project_id=project_id)
@@ -619,15 +644,16 @@ async def create_project_rule(
why=why, how_to_apply=how_to_apply, order_index=order_index, why=why, how_to_apply=how_to_apply, order_index=order_index,
verify_with=verify_with, expires_when=expires_when, verify_with=verify_with, expires_when=expires_when,
) )
data = await rulebooks_svc.rule_detail(uid, rule, system_ids) data = await rulebooks_svc.rule_detail(uid, rule, system_ids, moments)
data.update(dedup_svc.overlap_response(overlaps, "rule")) data.update(dedup_svc.overlap_response(overlaps, "rule"))
return data return await moment_delivery.attach_moment_rules(uid, "create_project_rule", {"project_id": project_id}, data)
async def update_rule( async def update_rule(
rule_id: int, title: str = "", statement: str = "", when_to_apply: str = "", rule_id: int, title: str = "", statement: str = "", when_to_apply: str = "",
why: str = "", how_to_apply: str = "", order_index: int = -1, why: str = "", how_to_apply: str = "", order_index: int = -1,
system_ids: list[int] | None = None, arose_from_id: int = 0, system_ids: list[int] | None = None,
moments: list[str] | None = None, arose_from_id: int = 0,
verify_with: str = "", expires_when: str = "", kind: str = "", verify_with: str = "", expires_when: str = "", kind: str = "",
clear_fields: list[str] | None = None, clear_fields: list[str] | None = None,
) -> dict: ) -> dict:
@@ -644,6 +670,9 @@ async def update_rule(
milestone 394, so a rule with no trigger is not a quiet rule — it is one milestone 394, so a rule with no trigger is not a quiet rule — it is one
no session will ever be shown. `system_ids` REPLACES the rule's areas no session will ever be shown. `system_ids` REPLACES the rule's areas
(pass [] to clear), and they decide which PROJECTS a rule binds by area. (pass [] to clear), and they decide which PROJECTS a rule binds by area.
`moments` REPLACES the moments it arrives at the same way — see
create_rule; mounting a rule whose trigger keeps missing on its moment is
often the better fix than rewriting the trigger.
RETROFITTING A TRIGGER HAS ITS OWN TRAP, and it is not the one create_rule RETROFITTING A TRIGGER HAS ITS OWN TRAP, and it is not the one create_rule
warns about. There the field is empty and the instruction is "write one". warns about. There the field is empty and the instruction is "write one".
@@ -690,6 +719,7 @@ async def update_rule(
clear_fields: Names of fields to empty, as above. clear_fields: Names of fields to empty, as above.
""" """
uid = current_user_id() uid = current_user_id()
moments = moments_svc.require_moments(moments)
fields: dict = {} fields: dict = {}
if title: if title:
fields["title"] = title fields["title"] = title
@@ -716,7 +746,7 @@ async def update_rule(
) )
if rule is None: if rule is None:
raise ValueError(f"rule {rule_id} not found") raise ValueError(f"rule {rule_id} not found")
return await rulebooks_svc.rule_detail(uid, rule, system_ids) return await rulebooks_svc.rule_detail(uid, rule, system_ids, moments)
# ── Preferences ───────────────────────────────────────────────────────── # ── Preferences ─────────────────────────────────────────────────────────
@@ -737,6 +767,7 @@ async def create_preference(
topic_id: int, title: str, statement: str, when_to_apply: str, topic_id: int, title: str, statement: str, when_to_apply: str,
arose_from_id: int, why: str = "", how_to_apply: str = "", arose_from_id: int, why: str = "", how_to_apply: str = "",
order_index: int = 0, system_ids: list[int] | None = None, order_index: int = 0, system_ids: list[int] | None = None,
moments: list[str] | None = None,
force: bool = False, force: bool = False,
) -> dict: ) -> dict:
"""Record how the operator wants work done. No approval loop — write it. """Record how the operator wants work done. No approval loop — write it.
@@ -806,6 +837,12 @@ async def create_preference(
preference is about, which is what lets it reach a session working preference is about, which is what lets it reach a session working
in that area. `update_preference` took this and create did not, so in that area. `update_preference` took this and create did not, so
a preference could only be filed after the fact (#4249). a preference could only be filed after the fact (#4249).
moments: Moments from list_moments this rule arrives at — whenever
one happens, the rule is delivered, whatever the words of the work
look like. Mount a rule here when it is about WHEN something is
done rather than what it is about: a rule on when work counts as
finished belongs at work.finish and reply.report, where nothing
said resembles it. Replaces the set; [] clears it.
force: Bypass the near-duplicate gate. For a genuinely distinct force: Bypass the near-duplicate gate. For a genuinely distinct
preference, not for one that is "mostly" different — a mostly preference, not for one that is "mostly" different — a mostly
different preference is an update. A RULE that already answers different preference is an update. A RULE that already answers
@@ -814,6 +851,7 @@ async def create_preference(
the weaker copy of it and should go. the weaker copy of it and should go.
""" """
uid = current_user_id() uid = current_user_id()
moments = moments_svc.require_moments(moments)
if not when_to_apply.strip(): if not when_to_apply.strip():
raise ValueError( raise ValueError(
"when_to_apply is required: a preference with no trigger never " "when_to_apply is required: a preference with no trigger never "
@@ -839,16 +877,17 @@ async def create_preference(
kind="preference", arose_from_id=arose_from_id, kind="preference", arose_from_id=arose_from_id,
why=why, how_to_apply=how_to_apply, order_index=order_index, why=why, how_to_apply=how_to_apply, order_index=order_index,
) )
data = await rulebooks_svc.rule_detail(uid, rule, system_ids) data = await rulebooks_svc.rule_detail(uid, rule, system_ids, moments)
data.update(dedup_svc.overlap_response(overlaps, "preference")) data.update(dedup_svc.overlap_response(overlaps, "preference"))
return data return await moment_delivery.attach_moment_rules(uid, "create_preference", {}, data)
async def update_preference( async def update_preference(
rule_id: int, arose_from_id: int, statement: str = "", rule_id: int, arose_from_id: int, statement: str = "",
when_to_apply: str = "", title: str = "", why: str = "", when_to_apply: str = "", title: str = "", why: str = "",
how_to_apply: str = "", order_index: int = -1, how_to_apply: str = "", order_index: int = -1,
system_ids: list[int] | None = None, clear_fields: list[str] | None = None, system_ids: list[int] | None = None,
moments: list[str] | None = None, clear_fields: list[str] | None = None,
) -> dict: ) -> dict:
"""Bring a preference up to date. Doing this mid-work is expected. """Bring a preference up to date. Doing this mid-work is expected.
@@ -900,9 +939,12 @@ async def update_preference(
Args: Args:
rule_id: The preference to update. rule_id: The preference to update.
arose_from_id: What taught this change. Required; see above. arose_from_id: What taught this change. Required; see above.
moments: Replaces the moments this preference arrives at (see
create_preference); omit to leave them alone.
when_to_apply: The moment it applies, in session vocabulary. See above. when_to_apply: The moment it applies, in session vocabulary. See above.
""" """
uid = current_user_id() uid = current_user_id()
moments = moments_svc.require_moments(moments)
if not arose_from_id: if not arose_from_id:
raise ValueError( raise ValueError(
"arose_from_id is required: this edit is the record of how the " "arose_from_id is required: this edit is the record of how the "
@@ -927,7 +969,7 @@ async def update_preference(
) )
if rule is None: if rule is None:
raise ValueError(f"rule {rule_id} not found") raise ValueError(f"rule {rule_id} not found")
return await rulebooks_svc.rule_detail(uid, rule, system_ids) return await rulebooks_svc.rule_detail(uid, rule, system_ids, moments)
async def rule_history(rule_id: int, version_id: int = 0) -> dict: async def rule_history(rule_id: int, version_id: int = 0) -> dict:
+2 -1
View File
@@ -17,6 +17,7 @@ from scribe.services import dedup as dedup_svc
from scribe.services import snippets as snippets_svc from scribe.services import snippets as snippets_svc
from scribe.services.note_usage import attach_usage, record_pulled from scribe.services.note_usage import attach_usage, record_pulled
from scribe.services import systems as systems_svc from scribe.services import systems as systems_svc
from scribe.services import moment_delivery
async def list_snippets( async def list_snippets(
@@ -223,7 +224,7 @@ async def create_snippet(
advice = snippets_svc.trigger_advice(when_to_use) advice = snippets_svc.trigger_advice(when_to_use)
if advice: if advice:
data["trigger_advice"] = advice data["trigger_advice"] = advice
return data return await moment_delivery.attach_moment_rules(uid, "create_snippet", {"project_id": project_id}, data)
async def get_snippet(snippet_id: int, project_id: int = 0) -> dict: async def get_snippet(snippet_id: int, project_id: int = 0) -> dict:
+5 -1
View File
@@ -46,6 +46,7 @@ from scribe.services import trash as trash_svc
from scribe.services.note_usage import record_pulled from scribe.services.note_usage import record_pulled
from scribe.services.record_refs import refuse_guessed_ids from scribe.services.record_refs import refuse_guessed_ids
from scribe.services.text import elide from scribe.services.text import elide
from scribe.services import moment_delivery
# A work log entry is prose, often long — the discipline asks for what was # A work log entry is prose, often long — the discipline asks for what was
@@ -466,7 +467,9 @@ async def update_task(
if prefs: if prefs:
data["reply_preferences"] = prefs data["reply_preferences"] = prefs
data["report_back"] = REPORT_BACK_CUE + " " + REPLY_PREFERENCES_CUE data["report_back"] = REPORT_BACK_CUE + " " + REPLY_PREFERENCES_CUE
return data return await moment_delivery.attach_moment_rules(
uid, "update_task", {"status": status, "project_id": project_id}, data,
)
async def add_task_log(task_id: int, content: str) -> dict: async def add_task_log(task_id: int, content: str) -> dict:
@@ -734,6 +737,7 @@ async def start_planning(
) )
if isinstance(result, dict): if isinstance(result, dict):
result.update(dedup_svc.batch_overlap_response(overlaps)) result.update(dedup_svc.batch_overlap_response(overlaps))
await moment_delivery.attach_moment_rules(uid, "start_planning", {"project_id": project_id}, result)
return result return result
+2 -1
View File
@@ -62,6 +62,7 @@ from scribe.models.note_usage import NoteUsageEvent # noqa: E402, F401
from scribe.models.rule_usage import RuleUsageEvent # noqa: E402, F401 from scribe.models.rule_usage import RuleUsageEvent # noqa: E402, F401
from scribe.models.system_usage import SystemUsageEvent # noqa: E402, F401 from scribe.models.system_usage import SystemUsageEvent # noqa: E402, F401
from scribe.models.retrieval_judgment import RetrievalJudgment # noqa: E402, F401 from scribe.models.retrieval_judgment import RetrievalJudgment # noqa: E402, F401
from scribe.models.moment_mapping import MomentMapping # noqa: E402, F401
from scribe.models.project import Project # noqa: E402, F401 from scribe.models.project import Project # noqa: E402, F401
from scribe.models.milestone import Milestone # noqa: E402, F401 from scribe.models.milestone import Milestone # noqa: E402, F401
from scribe.models.task_log import TaskLog # noqa: E402, F401 from scribe.models.task_log import TaskLog # noqa: E402, F401
@@ -77,7 +78,7 @@ from scribe.models.user_profile import UserProfile # noqa: E402, F401
# Imported before rulebook: rule_systems foreign-keys canonical_systems. # Imported before rulebook: rule_systems foreign-keys canonical_systems.
from scribe.models.canonical_system import CanonicalSystem # noqa: E402, F401 from scribe.models.canonical_system import CanonicalSystem # noqa: E402, F401
from scribe.models.rulebook import ( # noqa: E402, F401 from scribe.models.rulebook import ( # noqa: E402, F401
Rulebook, RulebookTopic, Rule, RuleRelation, rule_systems, Rulebook, RulebookTopic, Rule, RuleRelation, rule_moments, rule_systems,
) )
# After notes and rules: it foreign-keys both (milestone 440). # After notes and rules: it foreign-keys both (milestone 440).
from scribe.models.lesson_rule_link import LessonNoRule, LessonRuleLink # noqa: E402, F401 from scribe.models.lesson_rule_link import LessonNoRule, LessonRuleLink # noqa: E402, F401
+75
View File
@@ -0,0 +1,75 @@
from sqlalchemy import BigInteger, ForeignKey, Integer, Text, UniqueConstraint
from sqlalchemy.orm import Mapped, mapped_column
from scribe.models import Base
from scribe.models.base import CreatedAtMixin, iso
class MomentMapping(Base, CreatedAtMixin):
"""One install's correction to which actions reach which moment (milestone 458).
The moments are the product's vocabulary and the shipped defaults say how
the common actions reach them (`services/moment_actions.py`). What no
default can know is how THIS operator works: one delivers with a push,
another with `make ship`, a third by publishing a document. A row here is
that local knowledge, written in-session the moment a misfire is noticed.
`effect` is "add" (this action reaches this moment) or "remove" (a shipped
default that misfires here is switched off). Removal is a row rather than
an edit to the defaults because the defaults are code: an install can only
say "not here", and saying so must survive an upgrade that ships the same
default again.
Text rather than a CHECK on `effect`, for retrieval_tuning_events' reason
(rule 36): the service validates it, and a constraint would buy nothing but
a migration the day a third effect is wanted.
ONE ROW PER (user, tool, match, moment), so mapping the same thing twice
is an update of the reason rather than a duplicate, and an add and a remove
of the same mapping cannot both stand.
CASCADES with the user, unlike the telemetry tables: this is the user's own
configuration, not a history that should outlive them.
"""
__tablename__ = "moment_mappings"
id: Mapped[int] = mapped_column(BigInteger, primary_key=True)
user_id: Mapped[int] = mapped_column(
Integer, ForeignKey("users.id", ondelete="CASCADE"), nullable=False,
)
# The tool as the harness names it, with any MCP server prefix stripped
# (`mcp__plugin_x__update_task` → `update_task`), so a mapping does not
# depend on what an install called its server.
tool: Mapped[str] = mapped_column(Text, nullable=False)
# "" = every call of the tool. For a tool that runs a command, how the
# command starts ("make ship"); for any other tool, `field=value` pairs.
match: Mapped[str] = mapped_column(Text, nullable=False, default="")
moment: Mapped[str] = mapped_column(Text, nullable=False)
effect: Mapped[str] = mapped_column(Text, nullable=False, default="add")
# Why — what misfired, or what this action is for here. Optional at the
# boundary: the correction is usually self-explanatory, and a required
# reason would be friction on the exact in-session fix this table exists
# to make cheap.
reason: Mapped[str] = mapped_column(Text, nullable=False, default="")
# "model" | "human", for retrieval_tuning_events' reason: both act as the
# same user, and "did I do this or did the session?" is the first question.
actor: Mapped[str] = mapped_column(Text, nullable=False, default="model")
__table_args__ = (
UniqueConstraint(
"user_id", "tool", "match", "moment", name="uq_moment_mapping",
),
)
def to_dict(self) -> dict:
return {
"id": self.id,
"tool": self.tool,
"match": self.match,
"moment": self.moment,
"effect": self.effect,
"reason": self.reason,
"actor": self.actor,
"created_at": iso(self.created_at),
}
+15
View File
@@ -188,6 +188,21 @@ rule_systems = Table(
) )
# Which MOMENTS of work a rule arrives at (milestone 458). A rule mounted on a
# moment is delivered whenever that moment happens, by lookup and with no
# score — the route for a rule that is about WHEN something is done rather
# than what it is about, which semantic retrieval structurally misses. The
# moment is a catalog NAME (`services/moments.py`), not a key: the catalog is
# code, so there is nothing to point a foreign key at, and the service refuses
# an unknown name before it is stored.
rule_moments = Table(
"rule_moments",
Base.metadata,
Column("rule_id", BigInteger, ForeignKey("rules.id", ondelete="CASCADE"), primary_key=True),
Column("moment", Text, primary_key=True),
Column("created_at", DateTime(timezone=True), default=lambda: datetime.now(timezone.utc)),
)
class RuleRelation(Base, CreatedAtMixin): class RuleRelation(Base, CreatedAtMixin):
"""A typed edge between two rules. Each kind exists because its ABSENCE """A typed edge between two rules. Each kind exists because its ABSENCE
forced a workaround somewhere in the operator's rulebook (note 3026). forced a workaround somewhere in the operator's rulebook (note 3026).
+93
View File
@@ -13,11 +13,13 @@ from quart import Blueprint, g, jsonify, request
from scribe.auth import admin_required, get_current_user_id, login_required from scribe.auth import admin_required, get_current_user_id, login_required
from scribe.config import Config from scribe.config import Config
from scribe.services import lesson_rules as lesson_rules_svc from scribe.services import lesson_rules as lesson_rules_svc
from scribe.services import moment_delivery as moment_delivery_svc
from scribe.services import plugin_context as plugin_ctx_svc from scribe.services import plugin_context as plugin_ctx_svc
from scribe.services import repo_bindings as repo_bindings_svc from scribe.services import repo_bindings as repo_bindings_svc
from scribe.services import report_check as report_check_svc from scribe.services import report_check as report_check_svc
from scribe.services import shape_check as shape_check_svc from scribe.services import shape_check as shape_check_svc
from scribe.services import task_claims as task_claims_svc from scribe.services import task_claims as task_claims_svc
from scribe.services.embeddings import memoized_query_embeddings
from scribe.services.settings import get_admin_setting, set_setting from scribe.services.settings import get_admin_setting, set_setting
plugin_bp = Blueprint("plugin", __name__, url_prefix="/api/plugin") plugin_bp = Blueprint("plugin", __name__, url_prefix="/api/plugin")
@@ -88,6 +90,7 @@ async def session_context():
@plugin_bp.get("/retrieve") @plugin_bp.get("/retrieve")
@login_required @login_required
@memoized_query_embeddings
async def autoinject_retrieve(): async def autoinject_retrieve():
"""Title-first knowledge auto-inject for the plugin's UserPromptSubmit hook. """Title-first knowledge auto-inject for the plugin's UserPromptSubmit hook.
@@ -163,6 +166,7 @@ async def autoinject_retrieve():
@plugin_bp.get("/tool-rules") @plugin_bp.get("/tool-rules")
@login_required @login_required
@memoized_query_embeddings
async def pre_tool_rules(): async def pre_tool_rules():
"""Standing rules for the plugin's PreToolUse hook on ACTIONS (#3476). """Standing rules for the plugin's PreToolUse hook on ACTIONS (#3476).
@@ -238,8 +242,97 @@ async def pre_tool_rules():
return jsonify(result) return jsonify(result)
@plugin_bp.get("/moment-tools")
@login_required
async def moment_tools():
"""The tools whose calls can deliver a mounted rule (milestone 458 step 4).
Read by the plugin's catch-all PreToolUse hook once per session window and
kept on disk, so a call that cannot reach a mounted rule costs no request.
`tools` are tool KEYS — bare, lowercased, no MCP server prefix — the same
keys the moment mappings are written in. Empty on an install that has
mounted nothing, which keeps that hook entirely off the wire.
"""
return jsonify({"tools": await moment_delivery_svc.reachable_tools(g.user.id)})
@plugin_bp.post("/moment")
@login_required
async def moment():
"""The rules mounted on the moments this tool call reaches (milestone 458).
Body: the PreToolUse event as Claude Code delivers it — `tool_name` and
`tool_input` are read, nothing else. Sent whole because which input field
a mapping matches on is the mapping's business, not the hook's.
Query: `repo` / `project_id` for the scope, and the shared session ledger —
`exclude_rule_ids` (named this session: a repeat is cited, not quoted) and
`held_rule_ids` (opened) — exactly as /tool-rules takes them.
Returns `context` (the lines, each naming the moment and the action that
reached it, so a misfire is visible and can be unmapped in-session),
`rule_ids` (the FRESH ones, for the hook to append to the ledger) and
`moments` (the names reached). A lookup, not a ranked search: no
retrieval_logs row; each fresh rule is recorded surfaced with its moment.
"""
event = await request.get_json(silent=True) or {}
tool = str(event.get("tool_name") or "").strip()
tool_input = event.get("tool_input")
if not tool:
return jsonify({"context": "", "rule_ids": [], "moments": []})
project_id, _repo, _unbound = await _project_scope()
reached, result = await moment_delivery_svc.deliver_for_act(
g.user.id, tool, tool_input if isinstance(tool_input, dict) else {},
project_id=project_id,
exclude=frozenset(_int_list(request.args.get("exclude_rule_ids"))),
held=frozenset(_int_list(request.args.get("held_rule_ids"))),
)
return jsonify({
"context": "\n".join(result.lines),
"rule_ids": result.rule_ids,
"moments": [hit["moment"] for hit in reached],
})
@plugin_bp.post("/reply-rules")
@login_required
@memoized_query_embeddings
async def reply_rules():
"""Whether the reply that ends a turn is held for one read (milestone 458).
Called by the plugin's Stop hook with the finished reply. Two halves: the
rules MOUNTED on the reply moments (`reply.report`, and `reply.ask` when
the reply asks something), and the reply text against every rule's
trigger — the backstop for whatever the earlier arms missed.
Body: `{"reply": "<text>"}`. Query: `repo` / `project_id` for the scope;
`held_rule_ids` (opened this session — exempt, as at the act checkpoint);
`exclude_rule_ids` (named this session — not re-counted as surfaced);
`stopped_rule_ids` (already held a reply or an act this session — a rule
holds once, and the per-session cap counts these).
Returns `reason` — the words the hook blocks with, empty when nothing
holds — plus `rule_ids` for the hook's ledger and `moments` reached.
"""
data = await request.get_json(silent=True) or {}
reply = str(data.get("reply") or "")
project_id, _repo, _unbound = await _project_scope()
held = await moment_delivery_svc.reply_hold(
g.user.id, reply, project_id=project_id,
exclude=frozenset(_int_list(request.args.get("exclude_rule_ids"))),
held=frozenset(_int_list(request.args.get("held_rule_ids"))),
stopped=frozenset(_int_list(request.args.get("stopped_rule_ids"))),
)
return jsonify({
"reason": held.get("reason", ""),
"rule_ids": held.get("rule_ids", []),
"moments": held.get("moments", []),
})
@plugin_bp.get("/prior-art") @plugin_bp.get("/prior-art")
@login_required @login_required
@memoized_query_embeddings
async def write_path_prior_art(): async def write_path_prior_art():
"""Prior-art hint for the plugin's PreToolUse hook on Write/Edit. """Prior-art hint for the plugin's PreToolUse hook on Write/Edit.
+47
View File
@@ -17,6 +17,8 @@ import logging
from quart import Blueprint, jsonify, request from quart import Blueprint, jsonify, request
from scribe.auth import get_current_user_id, login_required from scribe.auth import get_current_user_id, login_required
from scribe.services import moment_actions as moment_actions_svc
from scribe.services import moments as moments_svc
from scribe.services.retrieval_tuning import set_dial, current_settings, tuning_history from scribe.services.retrieval_tuning import set_dial, current_settings, tuning_history
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@@ -104,3 +106,48 @@ async def tuning_history_route():
except ValueError as e: except ValueError as e:
return jsonify({"error": str(e)}), 400 return jsonify({"error": str(e)}), 400
return jsonify({"events": events, "total": len(events)}) return jsonify({"events": events, "total": len(events)})
@retrieval_bp.route("/moments", methods=["GET"])
@login_required
async def moments_route():
"""The moments of work that rules mount on (milestone 458), with the
actions that reach each one on this install.
The same payload `list_moments` returns, from the same services, so the
session and the Settings view cannot name different moments.
"""
out = moments_svc.catalog()
out.update(await moment_actions_svc.actions_by_moment(get_current_user_id()))
return jsonify(out)
async def _mapping_change(change):
"""map/unmap from the browser: the MCP tools' service, recorded as human."""
data = await request.get_json()
if not isinstance(data, dict):
return jsonify({"error": "Expected a JSON object"}), 400
if not data.get("tool") or not data.get("moment"):
return jsonify({"error": "tool and moment are required"}), 400
try:
result = await change(
get_current_user_id(), str(data["tool"]), str(data.get("match") or ""),
str(data["moment"]), reason=str(data.get("reason") or ""), actor="human",
)
except ValueError as e:
# Unknown moment, a match the tool cannot take, nothing to remove —
# the service says what would work, so pass it through.
return jsonify({"error": str(e)}), 400
return jsonify(result)
@retrieval_bp.route("/moments/mappings", methods=["POST"])
@login_required
async def map_action_route():
return await _mapping_change(moment_actions_svc.map_action)
@retrieval_bp.route("/moments/mappings", methods=["DELETE"])
@login_required
async def unmap_action_route():
return await _mapping_change(moment_actions_svc.unmap_action)
+28 -3
View File
@@ -8,6 +8,7 @@ from __future__ import annotations
from quart import Blueprint, jsonify, request from quart import Blueprint, jsonify, request
from scribe.auth import get_current_user_id, login_required from scribe.auth import get_current_user_id, login_required
import scribe.services.moments as moments_svc
import scribe.services.rulebooks as rulebooks_svc import scribe.services.rulebooks as rulebooks_svc
from scribe.services.trash import delete as trash_delete from scribe.services.trash import delete as trash_delete
from scribe.services.rule_usage import ( from scribe.services.rule_usage import (
@@ -17,6 +18,15 @@ from scribe.services.rule_usage import (
rulebooks_bp = Blueprint("rulebooks", __name__, url_prefix="/api") rulebooks_bp = Blueprint("rulebooks", __name__, url_prefix="/api")
def _moments_or_refusal(data: dict):
"""The request's `moments`, validated BEFORE any write — so an unknown
name is a 400 and not a rule created without the mounts it asked for.
Absent means "leave them alone" (None), as on the MCP door."""
try:
return moments_svc.require_moments(data.get("moments")), None
except ValueError as exc:
return None, (jsonify({"error": str(exc)}), 400)
# ── Rulebooks ─────────────────────────────────────────────────────────── # ── Rulebooks ───────────────────────────────────────────────────────────
@@ -154,8 +164,12 @@ async def list_rules():
# than for snippets: every rule on every install predates this table, so # than for snippets: every rule on every install predates this table, so
# for a while the zero-filled shape IS the common case. # for a while the zero-filled shape IS the common case.
usage = await usage_for_rules([int(it["id"]) for it in items]) usage = await usage_for_rules([int(it["id"]) for it in items])
# Zero-filled like `usage`, for the same reason: the UI renders "mounted
# nowhere" from an empty list rather than treating a missing key as a state.
mounted = await rulebooks_svc.list_rule_moments([int(it["id"]) for it in items])
for it in items: for it in items:
it["usage"] = usage.get(int(it["id"]), empty_rule_usage()) it["usage"] = usage.get(int(it["id"]), empty_rule_usage())
it["moments"] = mounted.get(int(it["id"]), [])
return jsonify({"rules": items}) return jsonify({"rules": items})
@@ -167,6 +181,9 @@ async def create_rule(topic_id: int):
statement = (data.get("statement") or "").strip() statement = (data.get("statement") or "").strip()
if not title or not statement: if not title or not statement:
return jsonify({"error": "title and statement are required"}), 400 return jsonify({"error": "title and statement are required"}), 400
moments, refused = _moments_or_refusal(data)
if refused:
return refused
try: try:
rule = await rulebooks_svc.create_rule( rule = await rulebooks_svc.create_rule(
topic_id=topic_id, topic_id=topic_id,
@@ -189,7 +206,7 @@ async def create_rule(topic_id: int):
except ValueError as exc: except ValueError as exc:
return jsonify({"error": str(exc)}), 404 return jsonify({"error": str(exc)}), 404
return jsonify(await rulebooks_svc.rule_detail( return jsonify(await rulebooks_svc.rule_detail(
get_current_user_id(), rule, data.get("system_ids"), get_current_user_id(), rule, data.get("system_ids"), moments,
)), 201 )), 201
@@ -223,10 +240,15 @@ async def update_rule(rule_id: int):
# service normalises "" to NULL for every nullable text column. The MCP # service normalises "" to NULL for every nullable text column. The MCP
# door needs the explicit list only because "" already means "unchanged" # door needs the explicit list only because "" already means "unchanged"
# there — two idioms, one outcome. # there — two idioms, one outcome.
moments, refused = _moments_or_refusal(data)
if refused:
return refused
rule = await rulebooks_svc.update_rule(rule_id, uid, **fields) rule = await rulebooks_svc.update_rule(rule_id, uid, **fields)
if rule is None: if rule is None:
return jsonify({"error": "rule not found"}), 404 return jsonify({"error": "rule not found"}), 404
return jsonify(await rulebooks_svc.rule_detail(uid, rule, data.get("system_ids"))) return jsonify(await rulebooks_svc.rule_detail(
uid, rule, data.get("system_ids"), moments,
))
@rulebooks_bp.get("/rules/<int:rule_id>/versions") @rulebooks_bp.get("/rules/<int:rule_id>/versions")
@@ -401,6 +423,9 @@ async def create_project_rule(project_id: int):
"surfaces at the moment it applies." "surfaces at the moment it applies."
}), 400 }), 400
title = (data.get("title") or "").strip() or statement.split(".")[0][:50] title = (data.get("title") or "").strip() or statement.split(".")[0][:50]
moments, refused = _moments_or_refusal(data)
if refused:
return refused
try: try:
rule = await rulebooks_svc.create_project_rule( rule = await rulebooks_svc.create_project_rule(
project_id=project_id, project_id=project_id,
@@ -423,7 +448,7 @@ async def create_project_rule(project_id: int):
except ValueError as exc: except ValueError as exc:
return jsonify({"error": str(exc)}), 404 return jsonify({"error": str(exc)}), 404
return jsonify(await rulebooks_svc.rule_detail( return jsonify(await rulebooks_svc.rule_detail(
get_current_user_id(), rule, data.get("system_ids"), get_current_user_id(), rule, data.get("system_ids"), moments,
)), 201 )), 201
+102 -4
View File
@@ -14,9 +14,12 @@ from scribe.models.design_system import DesignSystem, DesignToken
from scribe.models.note_usage import NoteUsageEvent from scribe.models.note_usage import NoteUsageEvent
from scribe.models.rule_usage import RuleUsageEvent from scribe.models.rule_usage import RuleUsageEvent
from scribe.models.system_usage import SystemUsageEvent from scribe.models.system_usage import SystemUsageEvent
from scribe.models.moment_mapping import MomentMapping
from scribe.models.retrieval_tuning import RetrievalTuningEvent from scribe.models.retrieval_tuning import RetrievalTuningEvent
from scribe.models.canonical_system import CanonicalSystem from scribe.models.canonical_system import CanonicalSystem
from scribe.models.rulebook import RuleRelation, rule_systems as rule_systems_t from scribe.models.rulebook import (
RuleRelation, rule_moments as rule_moments_t, rule_systems as rule_systems_t,
)
from scribe.models.lesson_rule_link import LessonNoRule, LessonRuleLink from scribe.models.lesson_rule_link import LessonNoRule, LessonRuleLink
from scribe.models.code_shape import CodeShape, CodeShapeEvent, CodeShapeUse from scribe.models.code_shape import CodeShape, CodeShapeEvent, CodeShapeUse
from scribe.models.project import Project from scribe.models.project import Project
@@ -93,8 +96,16 @@ logger = logging.getLogger(__name__)
# 444): the files that are each area, and whether an area's rulings were read # 444): the files that are each area, and whether an area's rulings were read
# once shown. The usage rows restore through the SYSTEM map, for the reason # once shown. The usage rows restore through the SYSTEM map, for the reason
# the rule twin restores through the rule map. # the rule twin restores through the rule map.
# v21 (2026-10) added moment_mappings (milestone 458): which of this install's
# actions reach which moment, and the shipped defaults it switched off. Each
# row is a correction an operator made in-session; a restore that dropped them
# would silently put every misfire back.
# v22 (2026-10) added rule_moments (milestone 458): the moments each rule
# arrives at. A mount is a judgment about when a rule applies that nothing in
# the rule's text records, and losing it would put back exactly the misses the
# mounts were made to fix.
# Bump when the serialized schema changes. # Bump when the serialized schema changes.
BACKUP_VERSION = 20 BACKUP_VERSION = 22
# Every table this backup carries, by its REAL name. Paired with _NOT_INCLUDED # Every table this backup carries, by its REAL name. Paired with _NOT_INCLUDED
# below, these two lists must together account for the entire schema — which is # below, these two lists must together account for the entire schema — which is
@@ -139,6 +150,10 @@ _BACKED_UP = [
# v20 (2026-10): System usage telemetry (milestone 444), for the reason # v20 (2026-10): System usage telemetry (milestone 444), for the reason
# its note and rule twins travel. # its note and rule twins travel.
"system_usage_events", "system_usage_events",
# v21 (2026-10): an install's action → moment corrections (milestone 458).
"moment_mappings",
# v22 (2026-10): the moments each rule arrives at (milestone 458).
"rule_moments",
] ]
# Tables intentionally NOT in the backup, surfaced in the payload so the gap is # Tables intentionally NOT in the backup, surfaced in the payload so the gap is
@@ -247,6 +262,8 @@ _COLUMN_EXCLUSIONS: dict[str, set[str]] = {
# Same again — and everything else travels, because each remaining column # Same again — and everything else travels, because each remaining column
# is part of the argument: what moved, from what, to what, by whom, why. # is part of the argument: what moved, from what, to what, by whom, why.
"retrieval_tuning_events": {"id"}, "retrieval_tuning_events": {"id"},
# Every other column is the correction itself, or who made it and why.
"moment_mappings": {"id"},
"design_systems": {"deleted_at", "deleted_batch_id", "created_at", "updated_at"}, "design_systems": {"deleted_at", "deleted_batch_id", "created_at", "updated_at"},
"design_tokens": {"deleted_at", "deleted_batch_id", "created_at", "updated_at"}, "design_tokens": {"deleted_at", "deleted_batch_id", "created_at", "updated_at"},
"repo_bindings": {"id", "created_at", "updated_at"}, "repo_bindings": {"id", "created_at", "updated_at"},
@@ -334,6 +351,7 @@ _IMPORT_COLUMN_EXCLUSIONS: dict[str, set[str]] = {
"rule_usage_events": {"id"}, "rule_usage_events": {"id"},
"system_usage_events": {"id"}, "system_usage_events": {"id"},
"retrieval_tuning_events": {"id"}, "retrieval_tuning_events": {"id"},
"moment_mappings": {"id"},
"design_systems": { "design_systems": {
"id", "deleted_at", "deleted_batch_id", "created_at", "updated_at", "id", "deleted_at", "deleted_batch_id", "created_at", "updated_at",
}, },
@@ -486,6 +504,21 @@ def _rule_usage_event_rows(rows) -> list[dict]:
] ]
def _moment_mapping_rows(rows) -> list[dict]:
"""An install's action → moment corrections (milestone 458). Not
`to_dict()`, for _retrieval_tuning_event_rows' reason: the restore needs
`user_id` to remap, and the MCP reader omits it."""
return [
{
"user_id": r.user_id, "tool": r.tool, "match": r.match,
"moment": r.moment, "effect": r.effect, "reason": r.reason,
"actor": r.actor,
"created_at": r.created_at.isoformat() if r.created_at else None,
}
for r in rows
]
def _retrieval_tuning_event_rows(rows) -> list[dict]: def _retrieval_tuning_event_rows(rows) -> list[dict]:
"""The record of why a retrieval dial is where it is (#4102). """The record of why a retrieval dial is where it is (#4102).
@@ -716,6 +749,12 @@ def _topic_rows(rows) -> list[dict]:
] ]
def _rule_moment_rows(rows) -> list[dict]:
"""A rule's mounts. The moment is a catalog NAME, the same on every
install, so only the rule id needs remapping on the way back in."""
return [{"rule_id": rule_id, "moment": moment} for rule_id, moment in rows]
def _rule_system_rows(rows) -> list[dict]: def _rule_system_rows(rows) -> list[dict]:
"""A rule's area tags, carried by canonical SLUG for the same reason the """A rule's area tags, carried by canonical SLUG for the same reason the
Systems are: the catalog is global and its ids are per-install.""" Systems are: the catalog is global and its ids are per-install."""
@@ -808,6 +847,9 @@ async def export_full_backup() -> dict:
select(rule_systems_t.c.rule_id, CanonicalSystem.slug) select(rule_systems_t.c.rule_id, CanonicalSystem.slug)
.join(CanonicalSystem, CanonicalSystem.id == rule_systems_t.c.canonical_id) .join(CanonicalSystem, CanonicalSystem.id == rule_systems_t.c.canonical_id)
)).all() )).all()
rule_moment_rows = (await session.execute(
select(rule_moments_t.c.rule_id, rule_moments_t.c.moment)
)).all()
rule_relations = (await session.execute(select(RuleRelation))).scalars().all() rule_relations = (await session.execute(select(RuleRelation))).scalars().all()
lesson_rule_links = (await session.execute(select(LessonRuleLink))).scalars().all() lesson_rule_links = (await session.execute(select(LessonRuleLink))).scalars().all()
lesson_no_rule = (await session.execute(select(LessonNoRule))).scalars().all() lesson_no_rule = (await session.execute(select(LessonNoRule))).scalars().all()
@@ -836,6 +878,9 @@ async def export_full_backup() -> dict:
retrieval_tuning_events = (await session.execute( retrieval_tuning_events = (await session.execute(
select(RetrievalTuningEvent).order_by(RetrievalTuningEvent.id) select(RetrievalTuningEvent).order_by(RetrievalTuningEvent.id)
)).scalars().all() )).scalars().all()
moment_mappings = (await session.execute(
select(MomentMapping).order_by(MomentMapping.id)
)).scalars().all()
repo_bindings = (await session.execute(select(RepoBinding))).scalars().all() repo_bindings = (await session.execute(select(RepoBinding))).scalars().all()
code_shapes = (await session.execute(select(CodeShape))).scalars().all() code_shapes = (await session.execute(select(CodeShape))).scalars().all()
code_shape_events = (await session.execute( code_shape_events = (await session.execute(
@@ -871,6 +916,7 @@ async def export_full_backup() -> dict:
"rules": _rule_rows(rules), "rules": _rule_rows(rules),
"canonical_systems": _canonical_system_rows(canonical_systems), "canonical_systems": _canonical_system_rows(canonical_systems),
"rule_systems": _rule_system_rows(rule_system_rows), "rule_systems": _rule_system_rows(rule_system_rows),
"rule_moments": _rule_moment_rows(rule_moment_rows),
"rule_relations": _rule_relation_rows(rule_relations), "rule_relations": _rule_relation_rows(rule_relations),
"lesson_rule_links": _lesson_rule_link_rows(lesson_rule_links), "lesson_rule_links": _lesson_rule_link_rows(lesson_rule_links),
"lesson_no_rule": _lesson_no_rule_rows(lesson_no_rule), "lesson_no_rule": _lesson_no_rule_rows(lesson_no_rule),
@@ -886,6 +932,7 @@ async def export_full_backup() -> dict:
"retrieval_tuning_events": _retrieval_tuning_event_rows( "retrieval_tuning_events": _retrieval_tuning_event_rows(
retrieval_tuning_events retrieval_tuning_events
), ),
"moment_mappings": _moment_mapping_rows(moment_mappings),
"repo_bindings": _repo_binding_rows(repo_bindings), "repo_bindings": _repo_binding_rows(repo_bindings),
"note_supersessions": _note_supersession_rows(supersessions), "note_supersessions": _note_supersession_rows(supersessions),
"code_shapes": _code_shape_rows(code_shapes), "code_shapes": _code_shape_rows(code_shapes),
@@ -1004,6 +1051,10 @@ async def export_user_backup(user_id: int) -> dict:
.join(CanonicalSystem, CanonicalSystem.id == rule_systems_t.c.canonical_id) .join(CanonicalSystem, CanonicalSystem.id == rule_systems_t.c.canonical_id)
.where(rule_systems_t.c.rule_id.in_(_rule_ids)) .where(rule_systems_t.c.rule_id.in_(_rule_ids))
)).all() if _rule_ids else [] )).all() if _rule_ids else []
rule_moment_rows = (await session.execute(
select(rule_moments_t.c.rule_id, rule_moments_t.c.moment)
.where(rule_moments_t.c.rule_id.in_(_rule_ids))
)).all() if _rule_ids else []
# Scoped through the RULE, not the version's user_id. That column is # Scoped through the RULE, not the version's user_id. That column is
# the ACTOR (milestone 323), so filtering on it would carry the # the ACTOR (milestone 323), so filtering on it would carry the
# versions this user wrote on someone ELSE's rule and drop the ones # versions this user wrote on someone ELSE's rule and drop the ones
@@ -1035,6 +1086,13 @@ async def export_user_backup(user_id: int) -> dict:
.where(RetrievalTuningEvent.user_id == user_id) .where(RetrievalTuningEvent.user_id == user_id)
.order_by(RetrievalTuningEvent.id) .order_by(RetrievalTuningEvent.id)
)).scalars().all() )).scalars().all()
# The user's own configuration: user_id is the owner, nothing to
# route around.
moment_mappings = (await session.execute(
select(MomentMapping)
.where(MomentMapping.user_id == user_id)
.order_by(MomentMapping.id)
)).scalars().all()
rule_relations = (await session.execute( rule_relations = (await session.execute(
select(RuleRelation).where( select(RuleRelation).where(
RuleRelation.from_rule_id.in_(_rule_ids), RuleRelation.from_rule_id.in_(_rule_ids),
@@ -1078,6 +1136,7 @@ async def export_user_backup(user_id: int) -> dict:
"rules": _rule_rows(rules), "rules": _rule_rows(rules),
"canonical_systems": _canonical_system_rows(canonical_systems), "canonical_systems": _canonical_system_rows(canonical_systems),
"rule_systems": _rule_system_rows(rule_system_rows), "rule_systems": _rule_system_rows(rule_system_rows),
"rule_moments": _rule_moment_rows(rule_moment_rows),
"rule_relations": _rule_relation_rows(rule_relations), "rule_relations": _rule_relation_rows(rule_relations),
"lesson_rule_links": _lesson_rule_link_rows(lesson_rule_links), "lesson_rule_links": _lesson_rule_link_rows(lesson_rule_links),
"lesson_no_rule": _lesson_no_rule_rows(lesson_no_rule), "lesson_no_rule": _lesson_no_rule_rows(lesson_no_rule),
@@ -1093,6 +1152,7 @@ async def export_user_backup(user_id: int) -> dict:
"retrieval_tuning_events": _retrieval_tuning_event_rows( "retrieval_tuning_events": _retrieval_tuning_event_rows(
retrieval_tuning_events retrieval_tuning_events
), ),
"moment_mappings": _moment_mapping_rows(moment_mappings),
"repo_bindings": _repo_binding_rows(repo_bindings), "repo_bindings": _repo_binding_rows(repo_bindings),
"note_supersessions": _note_supersession_rows(supersessions), "note_supersessions": _note_supersession_rows(supersessions),
"code_shapes": _code_shape_rows(code_shapes), "code_shapes": _code_shape_rows(code_shapes),
@@ -1301,6 +1361,23 @@ def _build_setting(row: dict, maps: _Maps) -> Setting | None:
return Setting(user_id=uid, key=row["key"], value=row.get("value", "")) return Setting(user_id=uid, key=row["key"], value=row.get("value", ""))
def _build_moment_mapping(row: dict, maps: _Maps) -> MomentMapping | None:
"""No remapping beyond the user: `tool` and `moment` are names, not keys."""
uid = maps.users.get(row.get("user_id") or 0)
if uid is None or not row.get("tool") or not row.get("moment"):
return None
return MomentMapping(
user_id=uid,
tool=row["tool"],
match=row.get("match") or "",
moment=row["moment"],
effect=row.get("effect") or "add",
reason=row.get("reason") or "",
actor=row.get("actor") or "model",
created_at=_dt(row.get("created_at")),
)
def _build_retrieval_tuning_event(row: dict, maps: _Maps) -> RetrievalTuningEvent | None: def _build_retrieval_tuning_event(row: dict, maps: _Maps) -> RetrievalTuningEvent | None:
"""No id remapping beyond the user: `surface` is a registry NAME, not a """No id remapping beyond the user: `surface` is a registry NAME, not a
foreign key, which is what lets this history survive a restore into an foreign key, which is what lets this history survive a restore into an
@@ -1872,9 +1949,10 @@ async def _restore_v2(data: dict) -> dict:
"repo_bindings": 0, "repo_bindings": 0,
"note_supersessions": 0, "code_shapes": 0, "code_shape_events": 0, "note_supersessions": 0, "code_shapes": 0, "code_shape_events": 0,
"code_shape_uses": 0, "canonical_systems": 0, "code_shape_uses": 0, "canonical_systems": 0,
"rule_systems": 0, "rule_relations": 0, "rule_versions": 0, "rule_systems": 0, "rule_moments": 0,
"rule_relations": 0, "rule_versions": 0,
"retrieval_tuning_events": 0, "lesson_rule_links": 0, "retrieval_tuning_events": 0, "lesson_rule_links": 0,
"lesson_no_rule": 0, "lesson_no_rule": 0, "moment_mappings": 0,
} }
async with async_session() as session: async with async_session() as session:
@@ -1984,6 +2062,15 @@ async def _restore_v2(data: dict) -> dict:
session.add(event) session.add(event)
stats["retrieval_tuning_events"] += 1 stats["retrieval_tuning_events"] += 1
# 8c. Moment mappings (v21) — the install's corrections to which
# actions reach which moment. Names only, so only the user remaps.
for mm_data in data.get("moment_mappings", []):
mapping = _build_moment_mapping(mm_data, maps)
if mapping is None:
continue
session.add(mapping)
stats["moment_mappings"] += 1
# 9. Rulebooks (v3) # 9. Rulebooks (v3)
for rb_data in data.get("rulebooks", []): for rb_data in data.get("rulebooks", []):
rb = _build_rulebook(rb_data, maps) rb = _build_rulebook(rb_data, maps)
@@ -2062,6 +2149,17 @@ async def _restore_v2(data: dict) -> dict:
)) ))
stats["rule_systems"] += 1 stats["rule_systems"] += 1
# The same shape for a rule's mounts (v22): a join table with no
# model, so no builder — the rule id is the only thing to remap.
for rm in data.get("rule_moments", []):
mapped_rule = maps.rules.get(rm.get("rule_id", 0))
if mapped_rule is None or not rm.get("moment"):
continue
await session.execute(rule_moments_t.insert().values(
rule_id=mapped_rule, moment=rm["moment"],
))
stats["rule_moments"] += 1
for rr in data.get("rule_relations", []): for rr in data.get("rule_relations", []):
relation = _build_rule_relation(rr, maps) relation = _build_rule_relation(rr, maps)
if relation is None: if relation is None:
+67 -16
View File
@@ -10,11 +10,14 @@ volume so subsequent boots are instant).
""" """
import asyncio import asyncio
import functools
import logging import logging
import math import math
import os import os
from collections.abc import Sequence from collections.abc import Sequence
from contextlib import contextmanager
from contextvars import ContextVar
from typing import TYPE_CHECKING from typing import TYPE_CHECKING
@@ -77,13 +80,68 @@ async def _get_model():
return _model return _model
# ONE EMBEDDING PER QUERY PER REQUEST (milestone 456 step 1).
#
# A single plugin request runs several ranked arms over the SAME string: the
# operator's message is searched by auto_inject, its reuse and lesson slots,
# prompt_rule, the preference slot and rule_via_lesson — each of which called
# `get_embedding` on its own, so one prompt cost up to six model calls for one
# vector. The arms are not wrong to search separately; they were wrong to pay
# for the vector separately.
#
# A memo, not a cache: it lives for one request and is opened explicitly by
# `query_embedding_memo()`. Outside a scope the default is None and nothing is
# remembered, so every other caller behaves exactly as before. Keyed on the
# exact text — a query that differs by a character is a different query.
#
# Only `get_embedding` (a QUERY) consults it. `get_embeddings` embeds a
# document's chunks on the write path, where a repeat inside one request does
# not happen and a memo would only hold memory.
_query_memo: ContextVar[dict[str, list[float]] | None] = ContextVar(
"_query_memo", default=None,
)
@contextmanager
def query_embedding_memo():
"""Within this block, each distinct query text is embedded once."""
token = _query_memo.set({})
try:
yield
finally:
_query_memo.reset(token)
def memoized_query_embeddings(fn):
"""Run an async handler inside `query_embedding_memo()`.
For the plugin routes, each of which fans one query out to several arms.
Innermost decorator, so the scope opens after auth has passed.
"""
@functools.wraps(fn)
async def wrapper(*args, **kwargs):
with query_embedding_memo():
return await fn(*args, **kwargs)
return wrapper
async def get_embedding(text: str) -> list[float]: async def get_embedding(text: str) -> list[float]:
"""Get an embedding vector for the given text. """Get an embedding vector for the given text.
Raises if the fastembed model fails to load. Callers should catch and Raises if the fastembed model fails to load. Callers should catch and
degrade to keyword search. degrade to keyword search.
Inside `query_embedding_memo()`, a text already embedded in this request
returns the same vector without a model call. A failure is not
remembered, so the next arm tries again and degrades on its own.
""" """
return (await get_embeddings([text]))[0] memo = _query_memo.get()
if memo is not None and text in memo:
return memo[text]
vector = (await get_embeddings([text]))[0]
if memo is not None:
memo[text] = vector
return vector
async def get_embeddings(texts: list[str]) -> list[list[float]]: async def get_embeddings(texts: list[str]) -> list[list[float]]:
@@ -1470,8 +1528,8 @@ async def semantic_search_rules(
Returns an empty list if the embedder is unavailable or on any error. Returns an empty list if the embedder is unavailable or on any error.
""" """
from scribe.models.project import Project from scribe.models.rulebook import Rule
from scribe.models.rulebook import Rule, Rulebook, RulebookTopic from scribe.services.rule_scope import joined_to_homes, rule_home
# See the sibling search: stamped before anything can return (#3765). # See the sibling search: stamped before anything can return (#3765).
if report is not None: if report is not None:
@@ -1487,19 +1545,14 @@ async def semantic_search_rules(
distance = RuleEmbedding.embedding.cosine_distance(query_vec) distance = RuleEmbedding.embedding.cosine_distance(query_vec)
try: try:
# topic_id XOR project_id (migration 0059), so a rule matches exactly # The home clause is shared with the moment lookup (rule_scope).
# one arm of whichever clause applies. Inside the try: the access # Inside the try: the access check reads the database too, and this
# check reads the database too, and this function fails open. # function fails open.
global_rule = Rulebook.owner_user_id == user_id home = await rule_home(user_id, project_id, everywhere=everywhere)
if everywhere:
home = or_(global_rule, Project.user_id == user_id)
elif project_id and await can_read_project(user_id, project_id):
home = or_(global_rule, Rule.project_id == project_id)
else:
home = global_rule
async with async_session() as session: async with async_session() as session:
rows = (await session.execute( rows = (await session.execute(
joined_to_homes(
select( select(
Rule, Rule,
distance.label("distance"), distance.label("distance"),
@@ -1508,9 +1561,7 @@ async def semantic_search_rules(
) )
.select_from(RuleEmbedding) .select_from(RuleEmbedding)
.join(Rule, RuleEmbedding.rule_id == Rule.id) .join(Rule, RuleEmbedding.rule_id == Rule.id)
.outerjoin(RulebookTopic, Rule.topic_id == RulebookTopic.id) )
.outerjoin(Rulebook, RulebookTopic.rulebook_id == Rulebook.id)
.outerjoin(Project, Rule.project_id == Project.id)
.where( .where(
Rule.deleted_at.is_(None), Rule.deleted_at.is_(None),
# No threshold predicate — see the note above # No threshold predicate — see the note above
+393
View File
@@ -0,0 +1,393 @@
"""Which actions reach which moment: shipped defaults plus each install's own (milestone 458 step 2).
The catalog (`services/moments.py`) says what the moments ARE. This module says
how the work gets there. One call can reach several moments: `kubectl apply`
is a run, a deliver and a reach outside the workspace all at once, and a rule
mounted on any of them should arrive.
AN ACTION is a tool plus an optional `match`:
- `match` empty — every call of that tool (`Edit` → work.change).
- A tool that runs a command (its input carries `command`) — how the command
starts, tested against each segment of a compound line, so
`cd app && make ship` reaches what `make ship` reaches. Leading `VAR=value`
assignments are skipped; a word boundary is required, so `git push` does
not match `git pushd`.
- Any other tool — `field=value` pairs, comma-separated, all of which must
hold (`update_task` with `status=done`).
Tool names are compared without any MCP server prefix and without case:
`mcp__plugin_x__update_task` and `update_task` are one tool, because what an
install called its server is not something a mapping should depend on.
WHY THE DEFAULTS ARE CODE AND THE CORRECTIONS ARE ROWS
The defaults cover the actions every install shares — the harness's own tools,
Scribe's own tools, the commonest command shapes. They ship, and improve, with
the product. What no default can know is how one operator works, so an install
ADDS mappings and REMOVES defaults that misfire for it, in-session, through
`map_action` / `unmap_action`. A removal is stored rather than applied to the
defaults so it survives an upgrade that ships the same default again.
The concrete commands below are examples of reaching a moment, which is why
they may name particular tools when the moments themselves may not.
"""
from __future__ import annotations
import logging
import re
from dataclasses import dataclass
from typing import Iterable
from sqlalchemy import select
from scribe.models import async_session
from scribe.models.moment_mapping import MomentMapping
from scribe.services import moments as catalog
logger = logging.getLogger(__name__)
ADD, REMOVE = "add", "remove"
EFFECTS = (ADD, REMOVE)
DEFAULT, INSTALL = "default", "install"
# The input field that makes a tool a command-runner, and the tools known to
# be one. The field drives MATCHING (any tool whose call carries a command is
# matched by prefix); the names drive VALIDATION, so a prefix-style match on a
# tool that takes no command is refused rather than stored to never fire.
COMMAND_FIELD = "command"
COMMAND_TOOLS = frozenset({"bash"})
# The harness's skill loader: loading a procedure reaches `skill.<its name>`,
# derived from the call rather than listed, since the names are the
# procedures' own.
SKILL_TOOL, SKILL_FIELD = "skill", "skill"
_SEGMENT_SPLIT = re.compile(r"&&|\|\||[;|\n]")
_ENV_ASSIGN = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*=\S*\s+")
@dataclass(frozen=True)
class Action:
tool: str
match: str
moment: str
def _defaults() -> tuple[Action, ...]:
out: list[Action] = []
def on(moment: str, tool: str, *matches: str) -> None:
for m in matches or ("",):
out.append(Action(tool, m, moment))
# The harness's own tools.
on("work.change", "Edit")
on("work.change", "Write")
on("work.change", "MultiEdit")
on("work.change", "NotebookEdit")
on("work.run", "Bash")
on("work.delegate", "Task")
on("work.delegate", "Agent")
on("work.plan", "EnterPlanMode")
on("work.plan", "ExitPlanMode")
on("reply.ask", "AskUserQuestion")
# Scribe's own tools — every install has these, so their moments ship.
on("work.start", "update_task", "status=in_progress")
on("work.finish", "update_task", "status=done", "status=cancelled")
on("work.finish", "update_milestone", "status=done")
on("work.plan", "start_planning")
on("work.plan", "create_milestone")
for tool in ("create_note", "create_lesson", "create_rule",
"create_project_rule", "create_preference", "create_snippet",
"create_process"):
on("work.record", tool)
# The commonest command shapes. An install's own (`make ship`,
# `./deploy.sh`) are what map_action is for.
on("work.deliver", "Bash",
"git push", "git merge", "gh pr merge", "gh release create",
"docker push", "kubectl apply", "helm install", "helm upgrade",
"terraform apply", "npm publish", "twine upload", "cargo publish")
on("work.verify", "Bash",
"pytest", "python -m pytest", "npm test", "npm run test", "yarn test",
"pnpm test", "go test", "go vet", "cargo test", "make test",
"make check", "ruff check", "mypy", "terraform plan",
"terraform validate")
on("env.reach", "Bash",
"ssh", "scp", "curl", "wget", "kubectl", "helm")
return tuple(out)
DEFAULT_ACTIONS: tuple[Action, ...] = _defaults()
# ── matching ──────────────────────────────────────────────────────────────
def tool_key(name: str) -> str:
"""The tool's bare, lowercased name: no MCP server prefix."""
clean = (name or "").strip()
if clean.startswith("mcp__"):
clean = clean.rsplit("__", 1)[-1]
return clean.lower()
def _squash(text: str) -> str:
return " ".join((text or "").split())
def _segments(command: str) -> list[str]:
out = []
for part in _SEGMENT_SPLIT.split(command or ""):
seg = part.strip().lstrip("(").strip()
while _ENV_ASSIGN.match(seg):
seg = _ENV_ASSIGN.sub("", seg, count=1)
seg = _squash(seg)
if seg:
out.append(seg)
return out
def _pairs(match: str) -> dict[str, str] | None:
"""`a=b, c=d` → {"a": "b", "c": "d"}; None when it is not that shape."""
out: dict[str, str] = {}
for part in (match or "").split(","):
key, sep, value = part.partition("=")
if not sep or not key.strip():
return None
out[key.strip().lower()] = value.strip().lower()
return out or None
def action_matches(action: Action, tool: str, tool_input: dict | None) -> bool:
if tool_key(action.tool) != tool_key(tool):
return False
if not action.match:
return True
tool_input = tool_input or {}
command = tool_input.get(COMMAND_FIELD)
if isinstance(command, str):
want = _squash(action.match)
return any(seg == want or seg.startswith(want + " ")
for seg in _segments(command))
pairs = _pairs(action.match)
if pairs is None:
return False
given = {str(k).lower(): v for k, v in tool_input.items()}
return all(
str(given.get(k, "")).strip().lower() == v for k, v in pairs.items()
)
def effective_actions(mappings: Iterable) -> list[tuple[Action, str]]:
"""The defaults this install has not removed, then its own additions.
`mappings` are MomentMapping rows, or anything with the same four fields.
"""
removed: set[tuple[str, str, str]] = set()
added: list[Action] = []
for m in mappings:
key = (tool_key(m.tool), _squash(m.match), m.moment)
if m.effect == REMOVE:
removed.add(key)
elif m.effect == ADD:
added.append(Action(m.tool, _squash(m.match), m.moment))
out = [
(a, DEFAULT) for a in DEFAULT_ACTIONS
if (tool_key(a.tool), a.match, a.moment) not in removed
]
out.extend((a, INSTALL) for a in added)
return out
def resolve(tool: str, tool_input: dict | None, mappings: Iterable = ()) -> list[dict]:
"""Every moment this call reaches, each with the action that reached it.
One entry per moment, in catalog order. When two actions reach the same
moment the more specific one — the longer match — is the one named, since
"reached by `git push`" says more than "reached by Bash".
"""
best: dict[str, dict] = {}
for action, via in effective_actions(mappings):
if not action_matches(action, tool, tool_input):
continue
held = best.get(action.moment)
if held is None or len(action.match) > len(held["match"]):
best[action.moment] = {
"moment": action.moment, "tool": action.tool,
"match": action.match, "via": via,
}
if tool_key(tool) == SKILL_TOOL:
name = str((tool_input or {}).get(SKILL_FIELD) or "").strip().lower()
moment = catalog.SKILL_PREFIX + name
if name and catalog.is_moment(moment):
best[moment] = {"moment": moment, "tool": tool, "match": f"skill={name}",
"via": DEFAULT}
order = {name: i for i, name in enumerate(catalog.MOMENTS)}
return sorted(best.values(), key=lambda h: (order.get(h["moment"], len(order)), h["moment"]))
def _sample_input(tool: str, match: str) -> dict:
"""A call that this action would describe — for reporting what it reaches."""
if not match:
return {}
if tool_key(tool) in COMMAND_TOOLS:
return {COMMAND_FIELD: match}
return dict(_pairs(match) or {})
def _clean(tool: str, match: str, moment: str) -> tuple[str, str, str]:
"""Validate a mapping's three parts, or refuse with what would work."""
bare = (tool or "").strip()
if bare.startswith("mcp__"):
bare = bare.rsplit("__", 1)[-1]
if not bare:
raise ValueError("tool is required — the tool as the harness names it, e.g. Bash or update_task")
clean_match = _squash(match)
if clean_match and tool_key(bare) not in COMMAND_TOOLS and _pairs(clean_match) is None:
raise ValueError(
f"{bare} does not run a command, so `match` names its arguments as "
f"field=value pairs (e.g. status=done), not {clean_match!r}. Leave "
f"it empty to map every {bare} call."
)
return bare, clean_match, catalog.require_moment(moment)
def _is_default(tool: str, match: str, moment: str) -> bool:
return any(
tool_key(a.tool) == tool_key(tool) and a.match == match and a.moment == moment
for a in DEFAULT_ACTIONS
)
# ── the install's own ─────────────────────────────────────────────────────
async def list_mappings(user_id: int) -> list[MomentMapping]:
async with async_session() as session:
rows = await session.execute(
select(MomentMapping)
.where(MomentMapping.user_id == user_id)
.order_by(MomentMapping.id)
)
return list(rows.scalars().all())
async def moments_for(user_id: int, tool: str, tool_input: dict | None) -> list[dict]:
"""The moments a call reaches for this user.
Fails OPEN to the defaults: an unreadable mapping table must cost the
install its corrections, not every moment.
"""
try:
mappings = await list_mappings(user_id)
except Exception:
logger.warning("moment mappings unreadable; using the defaults", exc_info=True)
mappings = []
return resolve(tool, tool_input, mappings)
async def _find(session, user_id: int, tool: str, match: str, moment: str):
rows = await session.execute(
select(MomentMapping).where(
MomentMapping.user_id == user_id,
MomentMapping.moment == moment,
MomentMapping.match == match,
)
)
return next((r for r in rows.scalars().all() if tool_key(r.tool) == tool_key(tool)), None)
async def _reaches(user_id: int, tool: str, match: str) -> list[dict]:
return resolve(tool, _sample_input(tool, match), await list_mappings(user_id))
async def map_action(
user_id: int, tool: str, match: str, moment: str,
*, reason: str = "", actor: str = "model",
) -> dict:
"""Make an action reach a moment for this install.
Mapping a shipped default this install had removed restores it; mapping
one already in force changes nothing and says so.
"""
tool, match, moment = _clean(tool, match, moment)
reason = (reason or "").strip()
async with async_session() as session:
row = await _find(session, user_id, tool, match, moment)
if _is_default(tool, match, moment):
if row is not None and row.effect == REMOVE:
await session.delete(row)
await session.commit()
change = "restored the shipped default"
else:
change = "already a shipped default — nothing to add"
elif row is not None:
row.reason = reason or row.reason
row.actor = actor
await session.commit()
change = "already mapped — reason updated" if reason else "already mapped"
else:
session.add(MomentMapping(
user_id=user_id, tool=tool, match=match, moment=moment,
effect=ADD, reason=reason, actor=actor,
))
await session.commit()
change = "mapped"
return {
"change": change, "tool": tool, "match": match, "moment": moment,
"now_reaches": await _reaches(user_id, tool, match),
}
async def unmap_action(
user_id: int, tool: str, match: str, moment: str,
*, reason: str = "", actor: str = "model",
) -> dict:
"""Stop an action reaching a moment for this install.
An install's own mapping is deleted. A shipped default is switched off by
a stored removal, so the next release does not switch it back on.
"""
tool, match, moment = _clean(tool, match, moment)
reason = (reason or "").strip()
async with async_session() as session:
row = await _find(session, user_id, tool, match, moment)
if row is not None and row.effect == ADD:
await session.delete(row)
await session.commit()
change = "removed this install's mapping"
elif _is_default(tool, match, moment):
if row is None:
session.add(MomentMapping(
user_id=user_id, tool=tool, match=match, moment=moment,
effect=REMOVE, reason=reason, actor=actor,
))
await session.commit()
change = "switched off the shipped default"
else:
change = "the shipped default was already off"
else:
raise ValueError(
f"nothing maps {tool}{' ' + repr(match) if match else ''} onto "
f"{moment}, so there is nothing to remove. list_moments shows "
f"which actions reach each moment."
)
return {
"change": change, "tool": tool, "match": match, "moment": moment,
"now_reaches": await _reaches(user_id, tool, match),
}
async def actions_by_moment(user_id: int) -> dict:
"""Each moment's actions in force, and the defaults this install removed."""
mappings = await list_mappings(user_id)
by_moment: dict[str, list[dict]] = {}
for action, via in effective_actions(mappings):
by_moment.setdefault(action.moment, []).append(
{"tool": action.tool, "match": action.match, "via": via}
)
removed = [m.to_dict() for m in mappings if m.effect == REMOVE]
return {"actions": by_moment, "removed_defaults": removed}
+269
View File
@@ -0,0 +1,269 @@
"""Delivering the rules mounted on a moment, through every door (milestone 458 step 4).
One function decides what an act reaches and what arrives with it; the doors
differ only in how they carry the answer:
- the plugin's catch-all PreToolUse hook → `/api/plugin/moment`, with the
session ledger, so a rule already named this session is cited not repeated;
- Scribe's own MCP tools → `attach_moment_rules`, in the tool's response, so
a client without the plugin still gets `work.finish` when a task closes;
- the Stop hook → `deliver_moments`, for the reply moments no tool call marks.
The pipeline stage is `retrieval_pipeline.run_moment_arm`; this module only
resolves the act to its moments and hands both doors the same result.
"""
from __future__ import annotations
import logging
from scribe.services import moment_actions
from scribe.services import moments as catalog
from scribe.services import retrieval_pipeline as rp
logger = logging.getLogger(__name__)
def _io() -> rp.MomentIO:
"""Resolved at CALL time, so a test patching either module's name is seen."""
from scribe.services import rule_usage, rulebooks
return rp.MomentIO(
lookup=rulebooks.rules_on_moments,
record_rule_surfaced=rule_usage.record_rule_surfaced,
)
async def reachable_tools(user_id: int) -> list[str]:
"""The tools a call to which could deliver a mounted rule, as tool keys.
What the plugin's catch-all hook reads once per session window, so the
calls that cannot reach a mounted rule — most of them, and every one on
an install that has mounted nothing — never leave the machine. An action
counts only when its moment carries a mount; the skill loader counts when
any `skill.<name>` moment does.
"""
from scribe.services import rulebooks
mounted = await rulebooks.mounted_moments(user_id)
if not mounted:
return []
mappings = await moment_actions.list_mappings(user_id)
keys = {
moment_actions.tool_key(action.tool)
for action, _via in moment_actions.effective_actions(mappings)
if action.moment in mounted
}
if any(name.startswith(catalog.SKILL_PREFIX) for name in mounted):
keys.add(moment_actions.SKILL_TOOL)
return sorted(keys)
async def deliver_moments(
user_id: int, reached: list[dict], *, project_id: int | None = None,
exclude: frozenset[int] = frozenset(), held: frozenset[int] = frozenset(),
) -> rp.RuleResult:
"""The mounted rules for moments already known to have happened."""
return await rp.run_moment_arm(
reached,
rp.RuleMoment(user_id=user_id, query="", project_id=project_id,
exclude=exclude, held=held),
io=_io(),
)
async def deliver_for_act(
user_id: int, tool: str, tool_input: dict | None, *,
project_id: int | None = None,
exclude: frozenset[int] = frozenset(), held: frozenset[int] = frozenset(),
) -> tuple[list[dict], rp.RuleResult]:
"""Which moments this act reached, and the mounted rules they deliver."""
reached = await moment_actions.moments_for(user_id, tool, tool_input or {})
if not reached:
return [], rp.RuleResult()
result = await deliver_moments(
user_id, reached, project_id=project_id, exclude=exclude, held=held,
)
return reached, result
async def attach_moment_rules(
user_id: int, tool: str, arguments: dict | None, data: dict,
) -> dict:
"""Carry the rules mounted on this tool call's moments in its own response.
The MCP door (#4286's attach shape): an in-band decoration on a payload
already being returned, fail-open, and absent when there is nothing to
say. The project is the record's own when it has one, else the caller's
argument — a project's mounted rules belong to work in that project.
No session ledger reaches this door, so a mounted rule arrives every time
its moment happens here. That is the right failure for the moments these
tools mark: closing a task is exactly when a rule about finishing applies,
however many tasks were closed before it.
"""
if not isinstance(data, dict):
return data
try:
project_id = data.get("project_id") or (arguments or {}).get("project_id") or None
_reached, result = await deliver_for_act(
user_id, tool, arguments or {}, project_id=project_id,
)
if result.lines:
data["moment_rules"] = {
"lines": result.lines,
"rule_ids": result.shown_rule_ids,
"open_with": "get_rule(id)",
}
except Exception: # noqa: BLE001 - a decoration never breaks the payload
logger.debug("moment rules for %s could not be attached", tool, exc_info=True)
return data
# ── The reply moment: the backstop at the end of a turn ─────────────────
#
# The Stop hook's door, folded in from milestone 456 step 8. A reply is the
# one act no tool call marks, and the moment where "please verify this on
# your end" gets said — so it is checked twice over: the rules MOUNTED on the
# reply moments (deterministic, somebody said they belong here), and the
# reply text against every rule's trigger (the backstop for whatever the
# earlier arms missed). An unopened rule from either half HOLDS the reply for
# one read, in these words; the hook never holds the rewrite.
# The reply's head and tail. The embedder reads ~512 tokens, and the part of
# a report that asks something of the reader — "let me know if it works" —
# is at the END, where a head-only query would cut it off.
_REPLY_HEAD_CHARS = 600
_REPLY_TAIL_CHARS = 1200
_REACHED_BY_REPLY = "the reply that ends this turn"
def reply_moments(reply: str) -> list[dict]:
"""The moments a finished reply reaches: always `reply.report`, and
`reply.ask` when a line of its prose ends in a question."""
reached = [{"moment": "reply.report", "tool": "", "match": _REACHED_BY_REPLY,
"via": moment_actions.DEFAULT}]
in_code = False
for line in (reply or "").splitlines():
stripped = line.strip()
if stripped.startswith("```"):
in_code = not in_code
continue
if not in_code and stripped.endswith("?"):
reached.append({"moment": "reply.ask", "tool": "", "match": "a question in the reply",
"via": moment_actions.DEFAULT})
break
return reached
def _reply_query(reply: str) -> str:
text = " ".join((reply or "").split())
if len(text) <= _REPLY_HEAD_CHARS + _REPLY_TAIL_CHARS:
return text
return text[:_REPLY_HEAD_CHARS] + " … " + text[-_REPLY_TAIL_CHARS:]
def reply_hold_reason(held: list[dict]) -> str:
"""The text the agent reads INSTEAD of its reply going out.
A practice, not a prohibition (rule 165), on the act checkpoint's model:
nothing here knows the reply is wrong. It names each rule with why it was
raised, gives the remedy as one call per rule, and says the reply may go
out unchanged — so a reader who finds the rules beside the point is out in
as many calls as there are rules, and is never held twice.
"""
if not held:
return ""
parts = []
for item in held:
why = (f"mounted on {item['moment']}" if item.get("moment")
else f"scores {item['score']} against this reply")
trigger = f"; it applies when {item['trigger']}" if item.get("trigger") else ""
parts.append(f"“{item['title']}” ({why}{trigger}) — get_rule({item['rule_id']})")
noun = "a standing rule" if len(held) == 1 else "standing rules"
return (
f"Held for one read before this reply goes out: {noun} this session "
f"has not opened: " + "; ".join(parts) + ". Read "
+ ("it" if len(held) == 1 else "them")
+ ", then send the reply — unchanged if it already does what the rule "
"asks, which is a judgement only you can make. This check runs once "
"per rule; the rewrite is never held."
)
async def reply_hold(
user_id: int, reply: str, *, project_id: int | None = None,
exclude: frozenset[int] = frozenset(), held: frozenset[int] = frozenset(),
stopped: frozenset[int] = frozenset(),
) -> dict:
"""Whether this finished reply is held, and in what words.
`held` is what the session OPENED (the act checkpoint's exemption);
`stopped` is what an earlier hold already put in front of it, so a rule
holds a session once. Past the act checkpoint's per-session cap nothing
holds — a mis-set bar degrades to a quiet session, never a stuck one.
Returns {} or {"reason", "rule_ids", "moments"}. Fails open to {}.
"""
from scribe.services import plugin_context as pc
from scribe.services import rule_usage, rulebooks
from scribe.services.retrieval_surfaces import budget_for, floor_for
try:
if not (reply or "").strip() or len(stopped) >= pc.CHECKPOINT_SESSION_CAP:
return {}
reached = reply_moments(reply)
names = [hit["moment"] for hit in reached]
skip = set(held) | set(stopped)
room = pc.CHECKPOINT_SESSION_CAP - len(stopped)
out: list[dict] = []
# The mounted half: deterministic, so every unopened RULE on these
# moments holds. A preference claims no such force.
for rule, at in await rulebooks.rules_on_moments(user_id, names, project_id or None):
if rule.id in skip or rule.kind == "preference":
continue
out.append({"rule_id": rule.id, "title": rule.title, "moment": at,
"trigger": (rule.when_to_apply or "").strip()})
# Trimmed BEFORE recording: a rule the cap cut was shown to nobody.
out = out[:room]
mounted = [item["rule_id"] for item in out]
fresh = [rid for rid in mounted if rid not in exclude]
if fresh:
try:
rule_usage.record_rule_surfaced(
user_id=user_id, rule_ids=fresh, source=rp.MOMENT_RULE_SOURCE,
detail={item["rule_id"]: item["moment"] for item in out
if item["rule_id"] in fresh},
)
except Exception: # noqa: BLE001 - observation never breaks the observed
logger.debug("reply mount surfacing not recorded", exc_info=True)
# The semantic half: the backstop. A mounted rule already holding is
# treated as held here, so one rule is never named twice.
floor = await floor_for(user_id, rp.REPLY_RULE.source)
result = await rp.run_rule_arm(
rp.REPLY_RULE,
rp.RuleMoment(
user_id=user_id, query=_reply_query(reply), project_id=project_id,
where="to this reply", checkpoint_where="this reply",
exclude=exclude, held=frozenset(skip | set(mounted)),
),
floor=floor, budget=await budget_for(user_id, rp.REPLY_RULE.source),
io=pc._rule_io(), checkpoint_floor=floor,
)
if result.checkpoint and len(out) < room:
cp = result.checkpoint
out.append({"rule_id": cp["rule_id"], "title": cp["title"],
"score": cp["score"], "trigger": cp.get("trigger", "")})
if not out:
return {}
return {
"reason": reply_hold_reason(out),
"rule_ids": [item["rule_id"] for item in out],
"moments": names,
}
except Exception: # noqa: BLE001 - a recall aid never stops a session by failing
logger.debug("reply hold failed", exc_info=True)
return {}
+171
View File
@@ -0,0 +1,171 @@
"""The moments of work that rules mount on (milestone 458).
WHY THIS EXISTS
Semantic retrieval answers "what does this text resemble?", and most rules are
not ABOUT a topic — they belong to a MOMENT in the work. A rule about when work
counts as finished governs the moment work is sent out, checked, closed and
reported, and nothing said at those moments needs to resemble the rule. Matched
on topic, it scores below every bar; mounted on its moments, it cannot miss.
So a rule names the moments it belongs to, and when one happens every rule
mounted on it arrives — by lookup, with no embedding and no score. Semantic
retrieval stays the second net, for whatever nobody mounted.
WHY THE CATALOG IS CODE, NOT DATA
The moments are the product's vocabulary, not an install's. Every install needs
the same mount points, or a rule written on one could not mean the same thing on
another, and the plugin's hooks, the skills that declare moments and the
instruction surfaces all name them. What varies per install is which ACTIONS
reach a moment — one operator delivers with a push, another with a deploy
script, a third by sending a document — and that mapping is data (step 2). The
moment is generic; the action that reaches it is local.
HOW TO WRITE ONE
Domain-neutral (rule 115): Scribe's work may be software, configuration,
infrastructure or anything else a person drives through an agent, so a
definition names the moment in words every one of those recognises. The test
beside this module refuses software-only vocabulary in both fields. Concrete
actions belong in the default mappings, where they are examples of reaching a
moment rather than the meaning of it.
Names are expensive to change once rules point at them. Adding one is cheap;
renaming one is a migration of every mount.
"""
from __future__ import annotations
import re
from dataclasses import dataclass
@dataclass(frozen=True)
class Moment:
"""One point in the work that rules can mount on."""
name: str
"""`<area>.<verb>` — the string a mount stores and a hook reports."""
means: str
"""What is happening at this moment, in one line, for any kind of work."""
reached_by: str
"""The kinds of action that typically reach it, said generically."""
def _m(name: str, means: str, reached_by: str) -> tuple[str, Moment]:
return name, Moment(name=name, means=means, reached_by=reached_by)
MOMENTS: dict[str, Moment] = dict([
_m("session.start",
"a working session begins, or resumes after its context was summarized",
"opening a session; resuming after the earlier conversation was summarized"),
_m("work.start",
"taking up a piece of work",
"marking a task as in progress; beginning a step of a plan"),
_m("work.plan",
"designing an approach before carrying it out",
"opening a plan; entering a planning mode; loading a planning procedure"),
_m("work.change",
"altering the thing being worked on",
"editing or creating a file; changing a setting or a document"),
_m("work.run",
"carrying out an action in the environment",
"running a command or a script"),
_m("work.verify",
"checking that the work does what it should",
"running a check; reading the result of an automated check; "
"loading a verification procedure"),
_m("work.deliver",
"sending work beyond the place it was made",
"publishing, merging, releasing, deploying or sending it to someone"),
_m("work.finish",
"declaring a piece of work done, or abandoning it",
"closing a task or an issue; marking it done or cancelled"),
_m("work.record",
"writing down what was learned or decided",
"creating a note, a lesson, a rule or a reusable example"),
_m("work.debug",
"working out why something does not behave as expected",
"reading a failure; loading a diagnosis procedure"),
_m("work.delegate",
"handing part of the work to another agent",
"dispatching a helper agent"),
_m("env.reach",
"touching a machine or a service outside the workspace",
"connecting to a host; calling a remote service; inspecting what runs there"),
_m("reply.report",
"reporting results to the person the work is for",
"the reply that ends a turn; loading a reporting procedure"),
_m("reply.ask",
"asking the person the work is for to decide or to act",
"a question in the reply; a structured question to the person"),
])
# A FAMILY rather than entries: one moment per named procedure, reached when it
# is loaded. The names are the procedures' own (bundled skills, stored
# processes), so they cannot be listed here — only their shape can.
SKILL_PREFIX = "skill."
SKILL_FAMILY = Moment(
name=f"{SKILL_PREFIX}<name>",
means="a named procedure is loaded",
reached_by="loading a skill or a stored process by its name",
)
_SKILL_NAME = re.compile(r"^[a-z0-9][a-z0-9_:-]*$")
def is_moment(name: str) -> bool:
"""Is this a name a rule can mount on — a catalog moment, or skill.<name>?"""
if name in MOMENTS:
return True
if name.startswith(SKILL_PREFIX):
return bool(_SKILL_NAME.match(name[len(SKILL_PREFIX):]))
return False
def require_moment(name: str) -> str:
"""The name, normalised, or a refusal that lists what is available.
A typo'd moment must not be mountable: the mount would be stored, read
back, and never fire — a rule that looks attached and is not, which is the
silent failure moments exist to end.
"""
clean = (name or "").strip().lower()
if is_moment(clean):
return clean
raise ValueError(
f"unknown moment {name!r}. Moments are: "
+ ", ".join(MOMENTS)
+ f", or {SKILL_FAMILY.name} for a named procedure"
)
def require_moments(names: list[str] | None) -> list[str] | None:
"""A rule's moments, normalised and de-duplicated — or the first refusal.
None passes through as None, which every rule door reads as "leave the
mounts alone"; a list, [] included, is the whole set after the write.
All are checked before any is stored, so a refusal leaves nothing
half-mounted.
"""
if names is None:
return None
if isinstance(names, str):
names = [names]
return list(dict.fromkeys(require_moment(n) for n in names))
def catalog() -> dict:
"""The catalog as plain data, for the MCP tool and the REST door alike."""
return {
"moments": [
{"name": m.name, "means": m.means, "reached_by": m.reached_by}
for m in MOMENTS.values()
],
"families": [
{"name": SKILL_FAMILY.name, "prefix": SKILL_PREFIX,
"means": SKILL_FAMILY.means, "reached_by": SKILL_FAMILY.reached_by},
],
"total": len(MOMENTS),
}
+74 -652
View File
@@ -45,6 +45,10 @@ from scribe.services.retrieval_surfaces import (
budget_for, budget_for,
floor_for, floor_for,
) )
from scribe.services import retrieval_pipeline as rp
from scribe.services.retrieval_pipeline import ( # noqa: F401 - re-exported
_RULEHINT_BAND, _rule_band, _rule_hint_line, checkpoint_for, checkpoint_reason,
)
from scribe.services.retrieval_telemetry import record_retrieval from scribe.services.retrieval_telemetry import record_retrieval
from scribe.services.settings import get_setting from scribe.services.settings import get_setting
from scribe.services.systems import system_names_for from scribe.services.systems import system_names_for
@@ -351,33 +355,6 @@ TOOLRULE_DEFAULT_THRESHOLD = SURFACES["pre_tool_rule"].floor_default
# the value it is stuck with. The reasoning below is why 5 is where it starts. # the value it is stuck with. The reasoning below is why 5 is where it starts.
RULEHINT_LIMIT = SURFACES["write_path_rule"].budget_default RULEHINT_LIMIT = SURFACES["write_path_rule"].budget_default
# MEASURED, NOT REASONED — and the reasoning it replaced was wrong (#3851).
#
# The prediction was that rules would rank SHARPLY, because `rule_document()`
# shapes them the way snippets are shaped — trigger in the embedded title and
# again above the body — and note 2485 measured snippets separating their top
# hit by 0.153 while every other kind managed 0.010–0.023.
#
# They do not. Three probes against real act queries, scored by the same
# embedding the arms use:
#
# `git push origin dev` top 0.757, gap to second 0.022
# `docker compose up -d` top 0.685, gap to second 0.016
# a bare-owner-filter query top 0.656, gap to second 0.020
#
# That is dev-log territory, not snippet territory: rules arrive as a
# tightly-packed block. Shaping alone did not buy separation, which is worth
# recording because the opposite was the natural inference from 2485.
#
# So the band is narrow BECAUSE the corpus is flat. At 0.10 — the notes
# menu's value — every one of the top eight on the push probe falls inside,
# including a CI-registry rule and another project's branch policy. At 0.05
# it admits roughly three ranks, which is the span where the scores are still
# saying something. 2485's Finding 3 is the standing caveat: no band value
# fixes a tie, and if rules ever rank as flat as dev-logs did this control
# stops working and the answer is a reranker (#1038), not a smaller number.
_RULEHINT_BAND = 0.05
# ── The pre-act checkpoint (#4214, milestone 419) ─────────────────────── # ── The pre-act checkpoint (#4214, milestone 419) ───────────────────────
# #
# WHAT A CHECKPOINT IS, AND WHY IT IS NOT A LOUDER HINT. Every arm above # WHAT A CHECKPOINT IS, AND WHY IT IS NOT A LOUDER HINT. Every arm above
@@ -1392,109 +1369,6 @@ async def build_autoinject_hint(
} }
async def _reserve_slot_for_preference(
user_id: int,
query: str,
hits: list,
*,
threshold: float,
project_id: int,
already: set[int],
) -> tuple[list, int | None]:
"""Guarantee a preference one slot, if one clears the bar (#3894).
THE ASYMMETRY THIS EXISTS FOR. A rule and a preference are not equally
served by a shared score contest, because their losses are not equal:
- a RULE crowded out here can still fire at the act arm. A `git push`
reaches `pre_tool_rule`, a file write reaches `write_path_rule`. The
prompt hit is a preview of a second chance.
- a PREFERENCE about how to answer has no second chance. There is no
later act — the response IS the act — so crowded out here it is never
delivered at all.
A straight ranking therefore favours the record whose loss is recoverable
over the one whose loss is total, and it does so INVISIBLY: the rule that
won is a legitimate hit, the telemetry looks healthy, and the only symptom
is a preference that quietly never arrives. `reuse_slot` exists for the
same shape one corpus over (#2463), where snippets kept losing to project
records that merely resembled the query.
THE SLOT BUYS POSITION, NOT A LOWER BAR — same as `reuse_slot`, which also
reserves at `cfg["threshold"]`. A weak preference cannot buy the slot, so
silence stays the default and the reserved line is never worse than the
ones it sits beside. If `preference_slot` later shows a stream of
near-misses, `best_available_id` (#3807) names which preference was
refused and a separate bar becomes an argument with evidence behind it
rather than a knob added on a guess.
LEDGER REPEATS STILL COUNT AS REPRESENTED. A preference already on the
session's ledger occupies the slot rather than being skipped for a fresh
one: it is still rendered (#3750), just with the tail that says so, and a
preference is the kind of record where being reminded is the point.
Returns the possibly-extended hit list, and the id the slot spent — the
caller needs that to keep each source's surfaced set matching its own log
row (#3668), since the slot logs under its own name.
"""
if any(rule.kind == "preference" for _s, rule in hits):
return hits, None
_t0 = time.perf_counter()
_rep: dict = {}
# KIND-FILTERED, so the query can only answer with what the slot is for.
# Verifying the kind afterwards would be weaker: an unfiltered search that
# happened to return a rule would spend the slot on it, and the line would
# be indistinguishable from one that earned its place.
found = await semantic_search_rules(
user_id, query, limit=1, threshold=threshold,
kind="preference", report=_rep, project_id=project_id or None,
)
fresh = [(s, r) for s, r in found if r.id not in already]
# ITS OWN SOURCE, and both sides of the trade logged. #2463's own finding
# is the warning rather than the precedent here: the hit that slot pushed
# OUT was in retrieval_logs while the query that pushed it out was not, so
# the slot could never be judged against what it displaced. `results` is
# fresh-only, matching what gets recorded as surfaced below (#3752/#3668).
record_retrieval(
user_id=user_id, source="preference_slot", query=query,
threshold=threshold, limit=1, project_id=project_id,
is_task=None, results=fresh,
best_available=_rep.get("best_available_score"),
best_available_id=_rep.get("best_available_id"),
searched=bool(_rep.get("searched", True)),
suppressed=len(found) - len(fresh),
duration_ms=(time.perf_counter() - _t0) * 1000.0,
)
seen = {rule.id for _s, rule in hits}
slot = [(s, r) for s, r in found
if r.kind == "preference" and r.id not in seen][:1]
if not slot:
return hits, None
slot_id = int(slot[0][1].id)
if slot_id not in already:
record_rule_surfaced(
user_id=user_id, rule_ids=[slot_id], source="preference_slot",
)
# IT EXTENDS, IT NEVER DISPLACES — and here it parts company with
# `reuse_slot`, which evicts its menu's weakest hit. The reason is the
# ledger rather than taste. A displaced hit was RETURNED by the general
# search and is sitting in that call's `retrieval_logs` row, but would not
# have been shown — so `prompt_rule`'s surfaced set would stop matching
# its own log row, and #3668's identity would break for a reason nothing
# in the data explains. That identity is the cheapest true statement
# available about this pair of tables, and milestone #379 is what it costs
# to lose it: five steps planned against a gap that was two counters
# disagreeing, not a write path dropping rows.
#
# The price is one extra line, only when the general search already filled
# the limit AND a preference cleared the bar without placing. Cheap, and
# it buys a surface whose two tables can always be checked against each
# other.
return hits + slot, slot_id
# ── A rule reached through its lessons (milestone 440, #4633) ────────────── # ── A rule reached through its lessons (milestone 440, #4633) ──────────────
# #
# A rule's own document is written in the rule's words, which are general by # A rule's own document is written in the rule's words, which are general by
@@ -1688,95 +1562,48 @@ async def _prompt_rule_hint(
out["_via_query"] = q out["_via_query"] = q
try: try:
threshold = await floor_for(user_id, "prompt_rule")
limit = await budget_for(user_id, "prompt_rule")
t0 = time.perf_counter()
_rep: dict = {}
# SCOPED TO THIS SESSION'S PROJECT (milestone 414): global rules plus # SCOPED TO THIS SESSION'S PROJECT (milestone 414): global rules plus
# the bound project's own. An unbound session (project_id 0) gets # the bound project's own. An unbound session (project_id 0) gets
# global rules only. This arm used to search every rule the user owned, # global rules only — this surface speaks unasked, and a whole-rulebook
# so each project's rules were injected into every other project's # answer is only right for someone who asked the whole rulebook.
# sessions — this surface speaks unasked, and a whole-rulebook answer result = await rp.run_rule_arm(
# is only right for someone who asked the whole rulebook. rp.PROMPT_RULE,
hits = await semantic_search_rules( rp.RuleMoment(
user_id, q, limit=limit, threshold=threshold, user_id=user_id, query=q, project_id=project_id,
report=_rep, project_id=project_id or None, where="to this request",
exclude=frozenset(exclude_rule_ids or []),
held=frozenset(held_rule_ids or []),
),
floor=await floor_for(user_id, "prompt_rule"),
budget=await budget_for(user_id, "prompt_rule"),
io=_rule_io(),
) )
duration_ms = (time.perf_counter() - t0) * 1000.0 if result.lines:
out["context"] = "\n".join(result.lines)
already = set(exclude_rule_ids or []) out["rule_ids"] = result.rule_ids
held = set(held_rule_ids or []) # Every rule LINE, repeats and the reserved slot included — for
fresh = [(score, rule) for score, rule in hits if rule.id not in already] # the soft-link recorder (#4637). Distinct from `rule_ids`, which
# is telemetry's fresh-only cut.
# BEFORE the early return, for the reason both sibling arms spell out out["shown_rule_ids"] = result.shown_rule_ids
# at length: a call that found nothing is the only evidence a bar is
# too high, and an arm that logs only the calls it liked reports a
# flawless clear-rate however badly it is tuned. This bar is inherited
# and unverified for this corpus, so the zero rows are the point.
record_retrieval(
user_id=user_id, source="prompt_rule", query=q,
threshold=threshold, limit=limit,
project_id=project_id,
is_task=None, results=fresh, duration_ms=duration_ms,
best_available=_rep.get("best_available_score"),
best_available_id=_rep.get("best_available_id"),
searched=bool(_rep.get("searched", True)),
suppressed=len(hits) - len(fresh),
)
# THE RESERVED SLOT RUNS BEFORE THE BAIL-OUT, and that ordering is
# load-bearing rather than tidy. An empty general result is not proof
# that no preference qualifies: the general search overfetches by
# distance and then collapses, so a preference ranked below that
# window is invisible to it while a kind-filtered query finds it at
# once. Bailing first would make the slot dead in exactly the corpus
# it exists for — one where rules outnumber preferences.
hits, slot_id = await _reserve_slot_for_preference(
user_id, q, hits, threshold=threshold,
project_id=project_id, already=already,
)
# `hits`, not `fresh` (#3750): a call whose only hit is a repeat still
# has something to say, it just says it differently.
if not hits:
return out
lines = [
_rule_hint_line(
rule, where="to this request",
seen=rule.id in already, held=rule.id in held,
)
for _score, rule in hits
]
# FRESH-ONLY (#3752). A reference is a rendering decision, not a
# retrieval outcome, and counting one here would inflate the
# denominator pull_through is read from.
#
# `fresh` is the PRE-SLOT list on purpose: it is exactly what this
# call's own `retrieval_logs` row recorded, so the two stay equal
# (#3668). The slot's hit is surfaced under `preference_slot` by the
# helper, against that source's own row.
rule_ids = [rule.id for _score, rule in fresh]
# Every rule LINE, repeats and the reserved slot included — what the
# reader actually had in front of them, for the soft-link recorder
# (#4637). Distinct from `rule_ids`, which is telemetry's fresh-only cut.
out["shown_rule_ids"] = [rule.id for _score, rule in hits]
# RANKED, not ambient: this arm chose what it showed. The name is also
# in `rule_usage.RANKED_SOURCES`, and it has to be — a ranked source
# missing from that tuple is counted as a bulk delivery nobody decided
# on, which silently moves it out of the pull-through denominator.
if rule_ids:
record_rule_surfaced(
user_id=user_id, rule_ids=rule_ids, source="prompt_rule",
)
out["context"] = "\n".join(lines)
out["rule_ids"] = rule_ids
except Exception: except Exception:
logger.debug("prompt rule arm failed", exc_info=True) logger.debug("prompt rule arm failed", exc_info=True)
return out return out
def _rule_io() -> rp.RuleIO:
"""The ranker and recorders the rule arms report to, read at call time.
Resolved from this module's names on every call rather than bound once, so
the pipeline reports to whatever this module's `semantic_search_rules`,
`record_retrieval` and `record_rule_surfaced` are at the moment it runs.
"""
return rp.RuleIO(
search=semantic_search_rules,
record_retrieval=record_retrieval,
record_rule_surfaced=record_rule_surfaced,
)
# --- Write-path trigger (#2082): prior art at the moment code is written ------ # --- Write-path trigger (#2082): prior art at the moment code is written ------
# Auto-inject above fires on the operator's prompt. The moment reuse is actually # Auto-inject above fires on the operator's prompt. The moment reuse is actually
# lost is later — when the AGENT decides mid-task to write a helper — and nothing # lost is later — when the AGENT decides mid-task to write a helper — and nothing
@@ -2025,264 +1852,6 @@ async def _checkpoint_floor(user_id: int) -> float:
return _CHECKPOINT_DEFAULT return _CHECKPOINT_DEFAULT
return min(1.0, max(0.0, value)) return min(1.0, max(0.0, value))
def _rule_band(hits: list) -> list:
"""The top hit, plus every hit within `_RULEHINT_BAND` of it (#3851).
The instrument that lets an act surface a SET without inventing one: a
fixed k fills its slots whether or not anything deserves them, while this
keeps only what the scores say is close, so one clearly-relevant rule
still shows one and four competing rules show four.
Sync and pure, and deliberately its own function rather than a comparison
written twice — the two act arms are the pair #3497 records drifting apart
by being modelled on each other instead of sharing.
Takes `(score, rule)` pairs already ordered best-first, as both
`semantic_search_*` helpers return them.
"""
if not hits:
return []
top = hits[0][0]
return [(s, r) for s, r in hits if s >= top - _RULEHINT_BAND]
def checkpoint_for(
kept: list, *, held: set[int], floor: float, where: str,
) -> dict:
"""The one rule, if any, that should STOP this act rather than annotate it.
Sync and pure so it can be read against a fixed list of hits without a
database — the two arms share it for the reason `_rule_band` is shared:
#3497 is the record of these two drifting apart by being modelled on each
other instead of sharing one function.
FOUR CONDITIONS, AND EACH IS A DIFFERENT KIND OF WRONG IT PREVENTS.
1. THE SCORE CLEARS `floor`. Not the arm's own floor — a much higher bar,
measured at `_CHECKPOINT_DEFAULT`. The hint arms keep nudging at their
floor; only a hit the corpus is confident about is allowed to stop
anything.
2. IT IS A RULE, NEVER A PREFERENCE. A preference says how something has
been done before and following it is what keeps work consistent; a rule
says what happens if you do not. Stopping an act over a preference
would assert a force the record explicitly does not claim, and
`_rule_hint_line` already keeps that distinction in the one word that
names it.
3. THE SESSION HAS NOT OPENED IT. `held` is observable — a PostToolUse
hook watches for the `get_rule` call (#4100) — so this is a recorded
event and not a model's self-report about its own context. A session
that read the rule has already had the thing the checkpoint exists to
produce, and stopping it again would be punishing the behaviour being
asked for.
4. IT IS THE TOP HIT. `kept` is a band, and a band's tail is there to let
an act surface a SET; the ranker's confidence claim attaches to its
first element only. A checkpoint raised on the fourth line of a band is
a stop justified by a score nobody claimed.
Returns a dict rather than a rule or a tuple. Widening a tuple is an
interface change to every unpack site that the compiler does not report
(#4207, learned the expensive way in this same milestone's first week), and
this value crosses a JSON boundary into a shell script where a missing
field is a silently empty variable.
"""
if not kept or floor <= 0:
return {}
score, rule = kept[0]
if score < floor:
return {}
if getattr(rule, "kind", "") == "preference":
return {}
if rule.id in held:
return {}
found = {
"rule_id": rule.id,
"title": rule.title,
"trigger": (rule.when_to_apply or "").strip(),
"score": round(float(score), 4),
"where": where,
}
# RENDERED HERE, at the one call site, rather than by each arm. Two arms
# that each remember to render it are two arms that can stop agreeing on
# what a stop says — which is #3497's history for this exact pair. The
# text stays a separate function so it can be read and tested without a
# rule object, but nothing outside this line decides whether to call it.
found["reason"] = checkpoint_reason(found)
return found
def checkpoint_reason(checkpoint: dict) -> str:
"""The text the agent reads INSTEAD of running the act.
Written as a practice rather than a prohibition (rule 165): it says what
to do and why it is worth doing, not what is forbidden. The act is not
wrong — nothing here knows whether it is — and saying so plainly is what
keeps the stop from reading as an accusation the system is in no position
to make.
It names the remedy as ONE call, because a stop whose remedy is vague
costs more than the miss it prevents. And it says the act may simply be
re-submitted afterwards, so a reader who finds the rule irrelevant is out
in two calls rather than negotiating with a hook.
"""
if not checkpoint:
return ""
trigger = checkpoint.get("trigger") or ""
return (
f"Held for one read. “{checkpoint['title']}” is a standing rule "
f"this session has not opened, and it scores {checkpoint['score']} "
f"against what you are about to do"
+ (f" ({trigger})" if trigger else "")
+ f". Read it with get_rule({checkpoint['rule_id']}), then go ahead — "
f"re-submit this call unchanged if the rule does not apply, which is a "
f"judgement only you can make. Nothing here has decided the act is "
f"wrong; the rule is being put in front of it rather than beside it, "
f"because a rule delivered alongside a result arrives after the "
f"decision it was meant to inform."
)
def _rule_hint_line(
rule, *, where: str, seen: bool, held: bool = False, compact: bool = False,
) -> str:
"""One rule hint line — both arms, both tails, both kinds (#3750, #3849).
THREE INDEPENDENT AXES SINCE #3851. `compact` joins `kind` and `seen`, and
like them it reads none of the others: it says how much ROOM this line
gets, which is a fact about its rank among today's hits rather than about
the rule. A compact line is still a full claim that the rule may apply —
it simply cites the rule instead of quoting its trigger. Crucially it
still carries the `seen` tail, so the three axes stay genuinely
independent: shortening a line must not decide what it says about whether
the session is holding the rule.
WHY THE LATER LINES ARE QUIETER. Measured at #3851: a full line runs ~143
tokens once the trigger is rendered, and #3855 tripled trigger lengths
across the corpus, so five full lines cost ~568 tokens before every Bash
call. Top-full-plus-references costs ~299 — about 2x the old single line,
for four more rules. The budget argument that once justified a single slot
was real; what it actually forbids is four voices at full volume, not four
voices.
The top hit keeps the full rendering because it is the one the ranker is
most confident about, and a reader who acts on exactly one line should
have acted on that one.
TWO INDEPENDENT AXES. `kind` decides the head, `seen` decides the tail,
and neither reads the other. A preference and a rule differ in force; a
repeat and a first surfacing differ in whether the session already holds
the line. Those are unrelated facts, and keeping them unrelated in the
code is what stopped the second kind from reopening the repeat question.
ONE FUNCTION BECAUSE THE TAILS MUST NOT DRIFT. The two arms phrase their
heads differently ("may apply here" vs "may apply to this Bash call") and
that difference is deliberate. Everything after it must not differ, and
#3497's history is that the pre-tool arm inherited a defect from its
sibling by being modelled on it rather than sharing with it. Two copies of
a two-branch string is how one branch gets fixed and the other does not.
WHY A REPEAT GETS A LINE AT ALL. Both arms used to drop a hit whose id was
already on the session's exclusion ledger and emit nothing. That is correct
only while the session still HOLDS what it was told, and a compaction
breaks exactly that: the earlier injection is summarized away while the id
stays on the ledger, so the rule is absent from context AND unreachable for
the rest of the session (#3749 closes the compaction half; this closes the
ordinary half, where a session simply stops holding a line it read an hour
ago).
Only ONE CLAUSE of the original line is false on a repeat — the claim that
the rule is not in the session's loaded set. So only that clause changes.
Title, trigger and pull pointer are identical either way, the statement is
never injected either way, and a repeat therefore costs the same ~40 tokens
as a first surfacing and no more.
DELIBERATELY NOT ASKING THE SESSION WHETHER IT HOLDS THE RULE. A model
asked "do you still hold rule 156?" will say yes, and the claim is
unverifiable self-report about its own context. The answer is also not
needed: the line is cheap enough to always emit and carries its own remedy
in both branches. Removing the question removes the fragility rather than
managing it.
"""
trigger = (rule.when_to_apply or "").strip()
preference = rule.kind == "preference"
# KIND CHANGES THE HEAD; `seen` CHANGES THE TAIL. The two axes are
# independent and stay that way, which is what lets the repeat logic above
# survive a second kind without being reasoned about again: whether a
# record is already on the ledger has nothing to do with how much force it
# carries, so the seen-branch is shared verbatim.
#
# The noun is the whole of the visual difference, and that is deliberate.
# A reader skimming an injected block gets one word to place the register \u2014
# so it is the SECOND word that moves, and it is the word naming force.
# Everything structural after it is identical, so the three kinds read as
# one set rather than three formats (milestone 385's step 5 writes the
# lesson voice against these two; they are one paragraph, not three).
noun = "Preference" if preference else "Standing rule"
# The only other place force is asserted. A rule's line tells the reader
# not to dismiss it unread, because dismissing a rule unread is how the
# thing it prevents happens. A preference makes no such claim: it says
# where to find how this has been done, and following it is what keeps
# things consistent rather than what keeps them correct.
reason = (
"for how this has been done before" if preference
else "before deciding it does not apply"
)
# THREE STATES, BECAUSE TWO OF THEM WERE BEING TOLD THE SAME LIE (#4100).
#
# `seen` means an arm NAMED this rule earlier. It does not mean the session
# read it — the line is a teaser, and a teaser skimmed past leaves nothing
# behind, least of all after a compaction summarises the turn it arrived
# in. "You saw it earlier this session" asserted something about the
# reader's context that the server had no way to know.
#
# `held` is the observable half: a PostToolUse hook watches for the
# `get_rule` call itself, so this is a recorded EVENT rather than a claim.
# That distinction is what keeps the non-goal above intact — the objection
# was to asking a model about its own context, not to noticing what it did.
#
# The middle state is the honest one and the one that was missing: named,
# not opened. It gets the full invitation, because a session that skipped
# the teaser is in almost the same position as one that never saw it.
if held:
tail = (
f"You opened it earlier this session; pull it with "
f"get_rule({rule.id}) again if you no longer hold it."
)
elif seen:
tail = (
f"Mentioned earlier this session but not opened — read it with "
f"get_rule({rule.id}) {reason}."
)
else:
tail = (
f"Read it with get_rule({rule.id}) {reason}; it is not in this "
"session's loaded set."
)
if compact:
# THE TRIGGER GOES; THE TAIL STAYS. Only one of the two is expensive \u2014
# a trigger runs 300-400 characters after #3855, the tail about 100 \u2014
# so dropping the trigger is nearly the whole saving and dropping the
# tail would be mostly sacrifice.
#
# It would also destroy the one thing #3750 exists to say. The tail is
# what tells a reader whether they were already told this and may no
# longer be holding it, and that is the entire difference they can act
# on; a reference with no tail reads as a first surfacing whether it is
# one or not. This branch shipped without it for one commit and
# test_a_rule_the_session_already_holds_is_referenced_not_re_offered
# caught it, which is the guard working exactly as #3750 intended.
return (
f"Also \u2014 {noun.lower()} \u201c{rule.title}\u201d. {tail}"
)
return (
f"{noun} that may apply {where} \u2014 \u201c{rule.title}\u201d"
+ (f" ({trigger})" if trigger else "")
+ f". {tail}"
)
async def build_write_path_hint( async def build_write_path_hint(
user_id: int, user_id: int,
path: str, path: str,
@@ -2862,117 +2431,35 @@ async def build_write_path_hint(
# session was told about an hour ago is not a rule in front of the reader # session was told about an hour ago is not a rule in front of the reader
# now. # now.
# #
# The arm itself is `retrieval_pipeline.run_rule_arm` (milestone 456): the
# band, the fresh/repeat split, the unconditional call row (#3497) and the
# fresh-only surfacing rows (#3752) are written there once for all three
# rule arms. What is decided HERE is only what makes this moment this
# moment — the query is the code being written (or the path, for an edit
# with no body), and a line says the rule may apply "here".
#
# Fails open like every other arm: a rule hint must never break a write. # Fails open like every other arm: a rule hint must never break a write.
rule_ids: list[int] = [] rule_ids: list[int] = []
shown_rule_ids: list[int] = [] shown_rule_ids: list[int] = []
checkpoint: dict = {} checkpoint: dict = {}
try: try:
already = set(exclude_rule_ids or []) result = await rp.run_rule_arm(
held = set(held_rule_ids or []) rp.WRITE_PATH_RULE,
# Timed like the notes arm above. Without this the rule row was the one rp.RuleMoment(
# source in the whole readout reporting a null p90_duration_ms (#3311) user_id=user_id, query=code or path, project_id=project_id,
# — a gap that reads as "this surface is somehow not measurable" rather where="here", checkpoint_where="here",
# than "nobody passed the number". exclude=frozenset(exclude_rule_ids or []),
rule_t0 = time.perf_counter() held=frozenset(held_rule_ids or []),
_rep_wpr: dict = {} ),
hits = await semantic_search_rules( floor=cfg["rule_threshold"], budget=cfg["rule_top_k"],
user_id, code or path, limit=cfg["rule_top_k"], io=_rule_io(), checkpoint_floor=cfg["checkpoint_threshold"],
threshold=cfg["rule_threshold"],
report=_rep_wpr, project_id=project_id or None,
) )
rule_ms = (time.perf_counter() - rule_t0) * 1000.0 lines.extend(result.lines)
# BAND FIRST, dedup second, and the order is the whole point (#3851). rule_ids.extend(result.rule_ids)
# The band is a statement about the SCORES — what the ranker thinks is
# close to the best match — so letting the ledger reorder it would let
# "you were told this already" change what counts as relevant. Those
# are the independent axes the renderer keeps apart.
kept = _rule_band(hits)
fresh = [(score, rule) for score, rule in kept if rule.id not in already]
# EVERY kept hit gets a line; `already` only changes the tail (#3750),
# and rank only changes how much room it gets (#3851).
for idx, (_score, rule) in enumerate(kept):
lines.append(
_rule_hint_line(
rule, where="here", seen=rule.id in already,
held=rule.id in held,
compact=idx > 0,
)
)
# `rule_ids` stays FRESH-ONLY, and that is the whole telemetry story of
# this change (#3752). It is what the hook writes to the exclusion
# ledger and what `record_rule_surfaced` counts; a referenced rule is
# already on the ledger by definition, and counting it as a surfacing
# would inflate pull_through's denominator with a choice this arm never
# made. A reference is a RENDERING decision, not a retrieval outcome.
rule_ids.extend(rule.id for _score, rule in fresh)
# Every rule line, repeats included — what the soft-link recorder # Every rule line, repeats included — what the soft-link recorder
# pairs with this response's lessons (#4637). # pairs with this response's lessons (#4637).
shown_rule_ids = [rule.id for _score, rule in kept] shown_rule_ids = result.shown_rule_ids
# The stop, beside the lines rather than instead of them — see checkpoint = result.checkpoint
# `checkpoint_for`. `kept` is passed, not `fresh`: whether a rule
# was named earlier this session says nothing about whether this
# act should wait for it to be READ, and those are the two axes
# #3750 exists to keep apart.
checkpoint = checkpoint_for(
kept, held=held, floor=cfg["checkpoint_threshold"], where="here",
)
# TWO tables, and the split is not arbitrary. retrieval_logs is one
# row per CALL, keyed on the score distribution a threshold is tuned
# from. rule_usage_events is one row per RULE per event, which is the
# grain "was this hint ever acted on" needs and the grain a JSONB
# result_ids array cannot be indexed at.
#
# This comment used to say rule ids had nowhere to go — that
# note_usage_events remaps ids on restore, so a rule id there would
# return attached to whatever note took that number. That is still
# true of the NOTE table, and it is exactly why rule_usage_events is
# its own (milestone 333 step 1). The gap it described is closed.
#
# THE CALL LOG IS UNCONDITIONAL; THE SURFACING LOG IS NOT, and the
# asymmetry is the correction #3497 exists to make. Both used to sit
# inside an `if fresh:`, which is how this arm came to report
# `zero_result_calls: 0` and `cleared_threshold: 133/133` — not a
# perfectly tuned surface but one structurally unable to record its
# own misses. #3311 read that artifact as a measurement and a whole
# milestone was scoped on it. A call that found nothing is the ONLY
# evidence a threshold is set too high, and it is the row every note
# surface has always written (write_path: 421 zeroes of 613 calls;
# auto_inject: 114 of 326). A SURFACING is different in kind: nothing
# was shown, so no such event occurred, and its log stays guarded.
#
# `results=fresh`, not `hits`: the note arms pass their exclusions
# INTO semantic_search_notes, so what they log is already
# post-exclusion. semantic_search_rules takes no such parameter and
# this filter is where the equivalent happens — logging `hits` would
# quietly make this row mean something other than every other row in
# the same readout.
record_retrieval(
user_id=user_id, source="write_path_rule", query=code or path,
threshold=cfg["rule_threshold"], limit=cfg["rule_top_k"],
project_id=project_id,
is_task=None, results=fresh, duration_ms=rule_ms,
best_available=_rep_wpr.get("best_available_score"),
best_available_id=_rep_wpr.get("best_available_id"),
searched=bool(_rep_wpr.get("searched", True)),
# FOUND BUT NOT SHOWN, which since #3851 has TWO causes: the
# session had already been told (the ledger), or the score fell
# outside `_RULEHINT_BAND` of the top hit. Both are counted here
# because the question this answers is unchanged — a zero row must
# be able to say whether the bar was too high or whether the arm
# simply chose not to speak, and only the first is a reason to
# move the threshold. Splitting the two causes needs its own
# column and is worth doing only if the band turns out to be
# dropping rules anyone wanted.
suppressed=len(hits) - len(fresh),
)
if fresh:
# `rule_ids` is `fresh`, i.e. AFTER exclude_rule_ids. A rule the
# session already holds was considered and not shown, and counting
# it would inflate the denominator with claims the agent never saw
# — which reads as a precision problem this arm does not have.
record_rule_surfaced(
user_id=user_id, rule_ids=rule_ids, source="write_path_rule",
)
except Exception: except Exception:
logger.debug("write-path rule arm failed", exc_info=True) logger.debug("write-path rule arm failed", exc_info=True)
@@ -3128,87 +2615,22 @@ async def _tool_rule_hint(
# For the via-lesson step (#4633) — set only once the arm is enabled. # For the via-lesson step (#4633) — set only once the arm is enabled.
out["_via_query"] = query out["_via_query"] = query
t0 = time.perf_counter() result = await rp.run_rule_arm(
_rep_ptr: dict = {} rp.PRE_TOOL_RULE,
hits = await semantic_search_rules( rp.RuleMoment(
user_id, query, limit=cfg["tool_rule_top_k"], user_id=user_id, query=query, project_id=project_id,
threshold=cfg["tool_rule_threshold"], where=f"to this {tool_name} call",
report=_rep_ptr, project_id=project_id or None, checkpoint_where=f"this {tool_name} call",
) exclude=frozenset(exclude_rule_ids or []),
duration_ms = (time.perf_counter() - t0) * 1000.0 held=frozenset(held_rule_ids or []),
),
already = set(exclude_rule_ids or []) floor=cfg["tool_rule_threshold"], budget=cfg["tool_rule_top_k"],
held = set(held_rule_ids or []) io=_rule_io(), checkpoint_floor=cfg["checkpoint_threshold"],
# Band first, dedup second — see the sibling arm for why that order is
# load-bearing rather than incidental.
kept = _rule_band(hits)
fresh = [(score, rule) for score, rule in kept if rule.id not in already]
# Logged BEFORE the early return, for the reason spelled out at length
# on the write-path arm above: a call that found nothing is the only
# evidence a threshold is too high, and an arm that logs only the calls
# it liked reports a flawless clear-rate however badly it is tuned.
# This arm shipped with the same defect inherited from its sibling, and
# it mattered more here — a surface with no rows at all cannot be told
# apart from a hook that never fired, which is precisely the silent
# failure the arm was built to stop.
record_retrieval(
user_id=user_id, source="pre_tool_rule", query=query,
threshold=cfg["tool_rule_threshold"], limit=cfg["tool_rule_top_k"],
project_id=project_id,
is_task=None, results=fresh, duration_ms=duration_ms,
best_available=_rep_ptr.get("best_available_score"),
best_available_id=_rep_ptr.get("best_available_id"),
searched=bool(_rep_ptr.get("searched", True)),
# See the sibling arm, including why this now counts BOTH the
# ledger and the band. It matters more here: this arm fires on
# every Bash call, so a long session excludes its way to an
# all-zero row and the threshold looks wrong when nothing about it
# is.
suppressed=len(hits) - len(fresh),
)
# `kept`, not `fresh` (#3750). A call whose only hit is a repeat still
# has something to say — the arm just says it differently.
if not kept:
return out
lines = [
_rule_hint_line(
rule, where=f"to this {tool_name} call",
seen=rule.id in already,
held=rule.id in held,
# Rank decides volume (#3851): the ranker's best guess gets the
# trigger, the rest get cited.
compact=idx > 0,
)
for idx, (_score, rule) in enumerate(kept)
]
# FRESH-ONLY, for the reason given on the sibling arm: a reference is a
# rendering decision, not a retrieval outcome, and counting it here
# would inflate the denominator pull_through is read from.
rule_ids = [rule.id for _score, rule in fresh]
# RANKED, not ambient: this arm chose what it showed, so a pull can
# settle whether the choice was any good. `rule_usage.RANKED_SOURCES`
# carries the same name.
#
# GUARDED, which it did not need to be before #3750: `fresh` can now be
# empty on a call that still emitted a line, and recording a surfacing
# of nothing would write an event with no rules in it.
if rule_ids:
record_rule_surfaced(
user_id=user_id, rule_ids=rule_ids, source="pre_tool_rule",
)
out["context"] = "\n".join(lines)
out["rule_ids"] = rule_ids
# AFTER the lines, never instead of them. A checkpoint stops the act;
# it does not decide what the act should be told, and a reader who
# reads the rule and re-submits must find the same hint waiting. The
# two are independent renderings of one retrieval.
out["checkpoint"] = checkpoint_for(
kept, held=held, floor=cfg["checkpoint_threshold"],
where=f"this {tool_name} call",
) )
if result.lines:
out["context"] = "\n".join(result.lines)
out["rule_ids"] = result.rule_ids
out["checkpoint"] = result.checkpoint
except Exception: except Exception:
logger.debug("pre-tool rule arm failed", exc_info=True) logger.debug("pre-tool rule arm failed", exc_info=True)
return out return out
+18 -25
View File
@@ -58,8 +58,8 @@ bar can be wrong forever without a single call looking unusual.
from __future__ import annotations from __future__ import annotations
import logging import logging
import time
from scribe.services import retrieval_pipeline as rp
from scribe.services.embeddings import semantic_search_rules from scribe.services.embeddings import semantic_search_rules
from scribe.services.retrieval_surfaces import SURFACES, budget_for, floor_for from scribe.services.retrieval_surfaces import SURFACES, budget_for, floor_for
from scribe.services.retrieval_telemetry import record_retrieval from scribe.services.retrieval_telemetry import record_retrieval
@@ -67,7 +67,7 @@ from scribe.services.rule_usage import record_rule_surfaced
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
SOURCE = "report_preference" SOURCE = rp.REPORT_PREFERENCE.source
# Written in the vocabulary of the MOMENT, because that is what a trigger is # Written in the vocabulary of the MOMENT, because that is what a trigger is
# written in and what this query is scored against. Domain-neutral on purpose # written in and what this query is scored against. Domain-neutral on purpose
@@ -114,34 +114,27 @@ async def completion_preferences(user_id: int, *, project_id: int | None = None)
happened, and a lookup that errors must not turn it into a failure. happened, and a lookup that errors must not turn it into a failure.
""" """
try: try:
threshold = await _threshold(user_id) # The stages — the kind filter, the unconditional call row, the
limit = await _limit(user_id) # fresh-only surfacing rows — are the one pipeline's (milestone 456).
report: dict = {}
t0 = time.perf_counter()
hits = await semantic_search_rules(
user_id, COMPLETION_QUERY, limit=limit, threshold=threshold,
kind="preference", report=report, project_id=project_id,
)
hits = [(score, rule) for score, rule in hits if rule.kind == "preference"]
record_retrieval(
user_id=user_id, source=SOURCE, query=COMPLETION_QUERY,
threshold=threshold, limit=limit, project_id=project_id,
is_task=None, results=hits,
best_available=report.get("best_available_score"),
best_available_id=report.get("best_available_id"),
searched=bool(report.get("searched", True)),
duration_ms=(time.perf_counter() - t0) * 1000.0,
)
if not hits:
return []
# RANKED: this surface chose what it showed, so the name is in # RANKED: this surface chose what it showed, so the name is in
# rule_usage.RANKED_SOURCES and its hits count toward pull-through. # rule_usage.RANKED_SOURCES and its hits count toward pull-through.
record_rule_surfaced( # There is no session ledger here: a completion report is written
user_id=user_id, rule_ids=[rule.id for _s, rule in hits], source=SOURCE, # once, so nothing it could repeat has been shown before.
result = await rp.run_rule_arm(
rp.REPORT_PREFERENCE,
rp.RuleMoment(
user_id=user_id, query=COMPLETION_QUERY, project_id=project_id,
),
floor=await _threshold(user_id), budget=await _limit(user_id),
io=rp.RuleIO(
search=semantic_search_rules,
record_retrieval=record_retrieval,
record_rule_surfaced=record_rule_surfaced,
),
) )
return [ return [
{"id": rule.id, "title": rule.title, "statement": rule.statement, "kind": "preference"} {"id": rule.id, "title": rule.title, "statement": rule.statement, "kind": "preference"}
for _score, rule in hits for _score, rule in result.shown
] ]
except Exception: # noqa: BLE001 - a decoration never breaks its payload except Exception: # noqa: BLE001 - a decoration never breaks its payload
logger.warning("completion preference lookup failed", exc_info=True) logger.warning("completion preference lookup failed", exc_info=True)
@@ -115,6 +115,7 @@ _RESCORERS = {
), ),
"write_path_rule": lambda u, q, p: _rescore_rules(u, q, p, None), "write_path_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
"pre_tool_rule": lambda u, q, p: _rescore_rules(u, q, p, None), "pre_tool_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
"reply_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
"prompt_rule": lambda u, q, p: _rescore_rules(u, q, p, None), "prompt_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
"report_preference": lambda u, q, p: _rescore_rules(u, q, p, "preference"), "report_preference": lambda u, q, p: _rescore_rules(u, q, p, "preference"),
} }
@@ -128,6 +129,7 @@ _CORPUS = {
"write_path": NoteEmbedding, "write_path": NoteEmbedding,
"write_path_rule": RuleEmbedding, "write_path_rule": RuleEmbedding,
"pre_tool_rule": RuleEmbedding, "pre_tool_rule": RuleEmbedding,
"reply_rule": RuleEmbedding,
"prompt_rule": RuleEmbedding, "prompt_rule": RuleEmbedding,
"report_preference": RuleEmbedding, "report_preference": RuleEmbedding,
} }
+778
View File
@@ -0,0 +1,778 @@
"""One retrieval pipeline: a ranked surface is a spec, not a hand-copied arm.
Milestone 456. Every ranked rule surface used to write out the same sequence
by hand — search, band, split fresh from repeats, log the call before any early
return, reserve a slot, render, record what was surfaced — once per arm, each
modelled on the nearest sibling. The copies drifted, and the defects that
mattered were fixed one copy at a time (#3497, #3750, #3752): "this arm shipped
with the same defect inherited from its sibling" was a comment in the code.
The invariants were held together by a parametrised test over three copies
rather than by there being one implementation.
Here the sequence is written ONCE (`run_rule_arm`), and an arm is a `RuleArm`:
which source it records under, and which of the stages it takes. The
differences between arms that used to be buried in their copies are now four
booleans on one screen — and the ones that are accidents rather than
decisions are milestone 456 step 7's to settle, one at a time, on evidence.
THE INVARIANTS, each written once:
- the call row is written before any early return (#3497), because a call
that found nothing is the only evidence a bar is too high;
- the call row's results and the surfacing rows are both the FRESH cut, so
the two tables describe the same delivery (#3668, #3752);
- the session ledger decides how a repeat RENDERS, never whether it is
retrieved (#3750, #4101);
- band before dedup (#3851), so "you were shown this" never changes what
counts as relevant;
- telemetry never costs the reader a line, and the arm as a whole fails
open: a recall aid may never break the prompt, the command or the write.
THE I/O IS PASSED IN (`RuleIO`), not imported. The pipeline is the policy;
which ranker it asks and which recorders it reports to belong to the caller.
The callers resolve them from their own module at call time.
"""
from __future__ import annotations
import logging
import time
from dataclasses import dataclass, field
from typing import Any, Callable
logger = logging.getLogger(__name__)
# ── Shared rendering and ranking helpers (moved from plugin_context) ──────
# MEASURED, NOT REASONED — and the reasoning it replaced was wrong (#3851).
#
# The prediction was that rules would rank SHARPLY, because `rule_document()`
# shapes them the way snippets are shaped — trigger in the embedded title and
# again above the body — and note 2485 measured snippets separating their top
# hit by 0.153 while every other kind managed 0.010–0.023.
#
# They do not. Three probes against real act queries, scored by the same
# embedding the arms use:
#
# `git push origin dev` top 0.757, gap to second 0.022
# `docker compose up -d` top 0.685, gap to second 0.016
# a bare-owner-filter query top 0.656, gap to second 0.020
#
# That is dev-log territory, not snippet territory: rules arrive as a
# tightly-packed block. Shaping alone did not buy separation, which is worth
# recording because the opposite was the natural inference from 2485.
#
# So the band is narrow BECAUSE the corpus is flat. At 0.10 — the notes
# menu's value — every one of the top eight on the push probe falls inside,
# including a CI-registry rule and another project's branch policy. At 0.05
# it admits roughly three ranks, which is the span where the scores are still
# saying something. 2485's Finding 3 is the standing caveat: no band value
# fixes a tie, and if rules ever rank as flat as dev-logs did this control
# stops working and the answer is a reranker (#1038), not a smaller number.
_RULEHINT_BAND = 0.05
def _rule_band(hits: list) -> list:
"""The top hit, plus every hit within `_RULEHINT_BAND` of it (#3851).
The instrument that lets an act surface a SET without inventing one: a
fixed k fills its slots whether or not anything deserves them, while this
keeps only what the scores say is close, so one clearly-relevant rule
still shows one and four competing rules show four.
Sync and pure, and deliberately its own function rather than a comparison
written twice — the two act arms are the pair #3497 records drifting apart
by being modelled on each other instead of sharing.
Takes `(score, rule)` pairs already ordered best-first, as both
`semantic_search_*` helpers return them.
"""
if not hits:
return []
top = hits[0][0]
return [(s, r) for s, r in hits if s >= top - _RULEHINT_BAND]
def checkpoint_for(
kept: list, *, held: set[int], floor: float, where: str,
) -> dict:
"""The one rule, if any, that should STOP this act rather than annotate it.
Sync and pure so it can be read against a fixed list of hits without a
database — the two arms share it for the reason `_rule_band` is shared:
#3497 is the record of these two drifting apart by being modelled on each
other instead of sharing one function.
FOUR CONDITIONS, AND EACH IS A DIFFERENT KIND OF WRONG IT PREVENTS.
1. THE SCORE CLEARS `floor`. Not the arm's own floor — a much higher bar,
measured at `_CHECKPOINT_DEFAULT`. The hint arms keep nudging at their
floor; only a hit the corpus is confident about is allowed to stop
anything.
2. IT IS A RULE, NEVER A PREFERENCE. A preference says how something has
been done before and following it is what keeps work consistent; a rule
says what happens if you do not. Stopping an act over a preference
would assert a force the record explicitly does not claim, and
`_rule_hint_line` already keeps that distinction in the one word that
names it.
3. THE SESSION HAS NOT OPENED IT. `held` is observable — a PostToolUse
hook watches for the `get_rule` call (#4100) — so this is a recorded
event and not a model's self-report about its own context. A session
that read the rule has already had the thing the checkpoint exists to
produce, and stopping it again would be punishing the behaviour being
asked for.
4. IT IS THE TOP HIT. `kept` is a band, and a band's tail is there to let
an act surface a SET; the ranker's confidence claim attaches to its
first element only. A checkpoint raised on the fourth line of a band is
a stop justified by a score nobody claimed.
Returns a dict rather than a rule or a tuple. Widening a tuple is an
interface change to every unpack site that the compiler does not report
(#4207, learned the expensive way in this same milestone's first week), and
this value crosses a JSON boundary into a shell script where a missing
field is a silently empty variable.
"""
if not kept or floor <= 0:
return {}
score, rule = kept[0]
if score < floor:
return {}
if getattr(rule, "kind", "") == "preference":
return {}
if rule.id in held:
return {}
found = {
"rule_id": rule.id,
"title": rule.title,
"trigger": (rule.when_to_apply or "").strip(),
"score": round(float(score), 4),
"where": where,
}
# RENDERED HERE, at the one call site, rather than by each arm. Two arms
# that each remember to render it are two arms that can stop agreeing on
# what a stop says — which is #3497's history for this exact pair. The
# text stays a separate function so it can be read and tested without a
# rule object, but nothing outside this line decides whether to call it.
found["reason"] = checkpoint_reason(found)
return found
def checkpoint_reason(checkpoint: dict) -> str:
"""The text the agent reads INSTEAD of running the act.
Written as a practice rather than a prohibition (rule 165): it says what
to do and why it is worth doing, not what is forbidden. The act is not
wrong — nothing here knows whether it is — and saying so plainly is what
keeps the stop from reading as an accusation the system is in no position
to make.
It names the remedy as ONE call, because a stop whose remedy is vague
costs more than the miss it prevents. And it says the act may simply be
re-submitted afterwards, so a reader who finds the rule irrelevant is out
in two calls rather than negotiating with a hook.
"""
if not checkpoint:
return ""
trigger = checkpoint.get("trigger") or ""
return (
f"Held for one read. “{checkpoint['title']}” is a standing rule "
f"this session has not opened, and it scores {checkpoint['score']} "
f"against what you are about to do"
+ (f" ({trigger})" if trigger else "")
+ f". Read it with get_rule({checkpoint['rule_id']}), then go ahead — "
f"re-submit this call unchanged if the rule does not apply, which is a "
f"judgement only you can make. Nothing here has decided the act is "
f"wrong; the rule is being put in front of it rather than beside it, "
f"because a rule delivered alongside a result arrives after the "
f"decision it was meant to inform."
)
def _rule_hint_line(
rule, *, where: str, seen: bool, held: bool = False, compact: bool = False,
) -> str:
"""One rule hint line — both arms, both tails, both kinds (#3750, #3849).
THREE INDEPENDENT AXES SINCE #3851. `compact` joins `kind` and `seen`, and
like them it reads none of the others: it says how much ROOM this line
gets, which is a fact about its rank among today's hits rather than about
the rule. A compact line is still a full claim that the rule may apply —
it simply cites the rule instead of quoting its trigger. Crucially it
still carries the `seen` tail, so the three axes stay genuinely
independent: shortening a line must not decide what it says about whether
the session is holding the rule.
WHY THE LATER LINES ARE QUIETER. Measured at #3851: a full line runs ~143
tokens once the trigger is rendered, and #3855 tripled trigger lengths
across the corpus, so five full lines cost ~568 tokens before every Bash
call. Top-full-plus-references costs ~299 — about 2x the old single line,
for four more rules. The budget argument that once justified a single slot
was real; what it actually forbids is four voices at full volume, not four
voices.
The top hit keeps the full rendering because it is the one the ranker is
most confident about, and a reader who acts on exactly one line should
have acted on that one.
TWO INDEPENDENT AXES. `kind` decides the head, `seen` decides the tail,
and neither reads the other. A preference and a rule differ in force; a
repeat and a first surfacing differ in whether the session already holds
the line. Those are unrelated facts, and keeping them unrelated in the
code is what stopped the second kind from reopening the repeat question.
ONE FUNCTION BECAUSE THE TAILS MUST NOT DRIFT. The two arms phrase their
heads differently ("may apply here" vs "may apply to this Bash call") and
that difference is deliberate. Everything after it must not differ, and
#3497's history is that the pre-tool arm inherited a defect from its
sibling by being modelled on it rather than sharing with it. Two copies of
a two-branch string is how one branch gets fixed and the other does not.
WHY A REPEAT GETS A LINE AT ALL. Both arms used to drop a hit whose id was
already on the session's exclusion ledger and emit nothing. That is correct
only while the session still HOLDS what it was told, and a compaction
breaks exactly that: the earlier injection is summarized away while the id
stays on the ledger, so the rule is absent from context AND unreachable for
the rest of the session (#3749 closes the compaction half; this closes the
ordinary half, where a session simply stops holding a line it read an hour
ago).
Only ONE CLAUSE of the original line is false on a repeat — the claim that
the rule is not in the session's loaded set. So only that clause changes.
Title, trigger and pull pointer are identical either way, the statement is
never injected either way, and a repeat therefore costs the same ~40 tokens
as a first surfacing and no more.
DELIBERATELY NOT ASKING THE SESSION WHETHER IT HOLDS THE RULE. A model
asked "do you still hold rule 156?" will say yes, and the claim is
unverifiable self-report about its own context. The answer is also not
needed: the line is cheap enough to always emit and carries its own remedy
in both branches. Removing the question removes the fragility rather than
managing it.
"""
trigger = (rule.when_to_apply or "").strip()
preference = rule.kind == "preference"
# KIND CHANGES THE HEAD; `seen` CHANGES THE TAIL. The two axes are
# independent and stay that way, which is what lets the repeat logic above
# survive a second kind without being reasoned about again: whether a
# record is already on the ledger has nothing to do with how much force it
# carries, so the seen-branch is shared verbatim.
#
# The noun is the whole of the visual difference, and that is deliberate.
# A reader skimming an injected block gets one word to place the register \u2014
# so it is the SECOND word that moves, and it is the word naming force.
# Everything structural after it is identical, so the three kinds read as
# one set rather than three formats (milestone 385's step 5 writes the
# lesson voice against these two; they are one paragraph, not three).
noun = "Preference" if preference else "Standing rule"
# The only other place force is asserted. A rule's line tells the reader
# not to dismiss it unread, because dismissing a rule unread is how the
# thing it prevents happens. A preference makes no such claim: it says
# where to find how this has been done, and following it is what keeps
# things consistent rather than what keeps them correct.
reason = (
"for how this has been done before" if preference
else "before deciding it does not apply"
)
# THREE STATES, BECAUSE TWO OF THEM WERE BEING TOLD THE SAME LIE (#4100).
#
# `seen` means an arm NAMED this rule earlier. It does not mean the session
# read it — the line is a teaser, and a teaser skimmed past leaves nothing
# behind, least of all after a compaction summarises the turn it arrived
# in. "You saw it earlier this session" asserted something about the
# reader's context that the server had no way to know.
#
# `held` is the observable half: a PostToolUse hook watches for the
# `get_rule` call itself, so this is a recorded EVENT rather than a claim.
# That distinction is what keeps the non-goal above intact — the objection
# was to asking a model about its own context, not to noticing what it did.
#
# The middle state is the honest one and the one that was missing: named,
# not opened. It gets the full invitation, because a session that skipped
# the teaser is in almost the same position as one that never saw it.
if held:
tail = (
f"You opened it earlier this session; pull it with "
f"get_rule({rule.id}) again if you no longer hold it."
)
elif seen:
tail = (
f"Mentioned earlier this session but not opened — read it with "
f"get_rule({rule.id}) {reason}."
)
else:
tail = (
f"Read it with get_rule({rule.id}) {reason}; it is not in this "
"session's loaded set."
)
if compact:
# THE TRIGGER GOES; THE TAIL STAYS. Only one of the two is expensive \u2014
# a trigger runs 300-400 characters after #3855, the tail about 100 \u2014
# so dropping the trigger is nearly the whole saving and dropping the
# tail would be mostly sacrifice.
#
# It would also destroy the one thing #3750 exists to say. The tail is
# what tells a reader whether they were already told this and may no
# longer be holding it, and that is the entire difference they can act
# on; a reference with no tail reads as a first surfacing whether it is
# one or not. This branch shipped without it for one commit and
# test_a_rule_the_session_already_holds_is_referenced_not_re_offered
# caught it, which is the guard working exactly as #3750 intended.
return (
f"Also \u2014 {noun.lower()} \u201c{rule.title}\u201d. {tail}"
)
return (
f"{noun} that may apply {where} \u2014 \u201c{rule.title}\u201d"
+ (f" ({trigger})" if trigger else "")
+ f". {tail}"
)
# ── The specs ────────────────────────────────────────────────────────────
@dataclass(frozen=True)
class RuleArm:
"""One ranked rule surface: where it records, and which stages it takes."""
source: str
"""The telemetry `source`. MUST equal the registry and SURFACES key."""
band: bool
"""Keep only the hits within `_RULEHINT_BAND` of the top one (#3851)."""
compact_tail: bool
"""Lines after the first cite the rule instead of quoting its trigger."""
checkpoint: bool
"""The top hit may hold the act for one read (`checkpoint_for`)."""
preference_slot: bool
"""Reserve one line for a preference that lost the ranking (#3894)."""
kind: str | None = None
"""Ask the ranker for one record kind only ("preference"), or every kind."""
stop_only: bool = False
"""The door shows nothing but the stop (milestone 458's reply moment).
At the end of a turn there is no context to annotate: a hit either holds
the reply for one read or reaches nobody. So only the checkpoint's rule is
recorded as surfaced — the rest were ranked, and the call row counts them,
but nobody was shown them."""
# Today's differences, reproduced exactly (milestone 456 step 2). Whether the
# prompt arm should band, and whether the act arms should reserve a
# preference, are step 7's questions — not settled by this refactor.
PROMPT_RULE = RuleArm(
"prompt_rule", band=False, compact_tail=False, checkpoint=False,
preference_slot=True,
)
PRE_TOOL_RULE = RuleArm(
"pre_tool_rule", band=True, compact_tail=True, checkpoint=True,
preference_slot=False,
)
WRITE_PATH_RULE = RuleArm(
"write_path_rule", band=True, compact_tail=True, checkpoint=True,
preference_slot=False,
)
# The completion report's preferences (milestone 409 step 4): a FIXED query,
# preferences only, read by update_task as records rather than as lines. Its
# query is `reply_preferences.COMPLETION_QUERY`.
REPORT_PREFERENCE = RuleArm(
"report_preference", band=False, compact_tail=False, checkpoint=False,
preference_slot=False, kind="preference",
)
# The backstop for every arm that ran earlier in the turn and missed: the
# finished reply against every rule's trigger. Its floor IS its stop bar
# (the reply_rule surface), so it is passed as both.
REPLY_RULE = RuleArm(
"reply_rule", band=False, compact_tail=False, checkpoint=True,
preference_slot=False, stop_only=True,
)
RULE_ARMS: tuple[RuleArm, ...] = (
WRITE_PATH_RULE, PRE_TOOL_RULE, PROMPT_RULE, REPORT_PREFERENCE, REPLY_RULE,
)
PREFERENCE_SLOT_SOURCE = "preference_slot"
# Every source the shared stages below can record under — what the registry
# declares for this module's fan-out sites, since `source` reaches the
# recorders here as a value rather than a literal.
RULE_SOURCES: tuple[str, ...] = (
*(arm.source for arm in RULE_ARMS), PREFERENCE_SLOT_SOURCE,
)
@dataclass(frozen=True)
class RuleIO:
"""The ranker and the two recorders an arm reports to."""
search: Callable[..., Any]
"""`semantic_search_rules`, or a stand-in with its signature."""
record_retrieval: Callable[..., Any]
record_rule_surfaced: Callable[..., Any]
@dataclass(frozen=True)
class RuleMoment:
"""What is happening: the query an arm searches with, and the session."""
user_id: int
query: str
project_id: int | None
"""The bound project, 0 or None when unbound. Logged as given, searched as
`project_id or None` — global rules plus that project's own (milestone 414)."""
where: str = ""
"""How a line names the moment: "here", "to this Bash call", …"""
checkpoint_where: str = ""
exclude: frozenset[int] = frozenset()
"""Rules the session has been shown — rendered as repeats, not hidden."""
held: frozenset[int] = frozenset()
"""Rules the session has OPENED (#4100)."""
@dataclass
class RuleResult:
"""What an arm put in front of the reader, and what it counted."""
lines: list[str] = field(default_factory=list)
rule_ids: list[int] = field(default_factory=list)
"""The FRESH cut — telemetry's denominator, never a repeat."""
shown_rule_ids: list[int] = field(default_factory=list)
"""Every rule LINE, repeats and the reserved slot included."""
shown: list = field(default_factory=list)
"""The (score, rule) pairs behind those lines, best first — for a caller
that hands back records rather than lines (the completion report)."""
checkpoint: dict = field(default_factory=dict)
# ── The stages ───────────────────────────────────────────────────────────
def _telemetry_failed(source: str) -> None:
"""A recorder raised. Telemetry observes the surface and must never take it
down — or take a rendered line away from the reader, which an exception
reaching the arm's own fail-open would do. So it costs only its row.
The recorders are called DIRECTLY at each site, inside their own try,
rather than through a wrapper taking the recorder as an argument: the
registry test finds recorder call sites by their name, and a wrapper would
hide every site in this module from it (#3191's narrowing).
"""
logger.debug("retrieval telemetry failed for %s", source, exc_info=True)
async def _ranked(
io: RuleIO, moment: RuleMoment, *, source: str, floor: float, limit: int,
band: bool = False, kind: str | None = None,
) -> tuple[list, list, list]:
"""Search, band, split fresh from repeats, and log the call — every call.
Returns (hits, kept, fresh): what the ranker returned, what the band kept,
and the kept hits this session has not been shown. The row is written
HERE, before any caller can return early (#3497), with `results` the fresh
cut and `suppressed` counting both the band and the ledger, so a long
session's all-repeat zero reads apart from a bar that turned everything
away.
"""
t0 = time.perf_counter()
report: dict = {}
kwargs: dict = {"limit": limit, "threshold": floor, "report": report,
"project_id": moment.project_id or None}
if kind:
# KIND-FILTERED, so the query can only answer with what was asked
# for. Verifying the kind afterwards would be weaker: an unfiltered
# search that happened to return a rule would spend a preference's
# slot on it, indistinguishable from a line that earned its place.
kwargs["kind"] = kind
hits = await io.search(moment.user_id, moment.query, **kwargs)
if kind:
# And checked on the way out: a record of another kind that slipped
# through must not be logged, shown or counted under a source whose
# name claims the kind — a rule handed back as "how the operator likes
# this done" asserts a force the record does not have.
hits = [(score, rule) for score, rule in hits
if getattr(rule, "kind", None) == kind]
kept = _rule_band(hits) if band else hits
fresh = [(score, rule) for score, rule in kept if rule.id not in moment.exclude]
try:
io.record_retrieval(
user_id=moment.user_id, source=source, query=moment.query,
threshold=floor, limit=limit, project_id=moment.project_id,
is_task=None, results=fresh,
duration_ms=(time.perf_counter() - t0) * 1000.0,
best_available=report.get("best_available_score"),
best_available_id=report.get("best_available_id"),
searched=bool(report.get("searched", True)),
suppressed=len(hits) - len(fresh),
)
except Exception: # noqa: BLE001 - observation never breaks the observed
_telemetry_failed(source)
return hits, kept, fresh
async def _reserve_slot_for_preference(
io: RuleIO, moment: RuleMoment, hits: list, *, floor: float,
) -> tuple[list, int | None]:
"""Guarantee a preference one slot, if one clears the bar (#3894).
THE ASYMMETRY THIS EXISTS FOR. A rule and a preference are not equally
served by a shared score contest, because their losses are not equal:
- a RULE crowded out here can still fire at the act arm. A `git push`
reaches `pre_tool_rule`, a file write reaches `write_path_rule`. The
prompt hit is a preview of a second chance.
- a PREFERENCE about how to answer has no second chance. There is no
later act — the response IS the act — so crowded out here it is never
delivered at all.
A straight ranking therefore favours the record whose loss is recoverable
over the one whose loss is total, and it does so INVISIBLY: the rule that
won is a legitimate hit, the telemetry looks healthy, and the only symptom
is a preference that quietly never arrives. `reuse_slot` exists for the
same shape one corpus over (#2463), where snippets kept losing to project
records that merely resembled the query.
THE SLOT BUYS POSITION, NOT A LOWER BAR — same as `reuse_slot`, which also
reserves at `cfg["threshold"]`. A weak preference cannot buy the slot, so
silence stays the default and the reserved line is never worse than the
ones it sits beside. If `preference_slot` later shows a stream of
near-misses, `best_available_id` (#3807) names which preference was
refused and a separate bar becomes an argument with evidence behind it
rather than a knob added on a guess.
LEDGER REPEATS STILL COUNT AS REPRESENTED. A preference already on the
session's ledger occupies the slot rather than being skipped for a fresh
one: it is still rendered (#3750), just with the tail that says so, and a
preference is the kind of record where being reminded is the point.
Returns the possibly-extended hit list, and the id the slot spent — the
caller needs that to keep each source's surfaced set matching its own log
row (#3668), since the slot logs under its own name.
"""
if any(rule.kind == "preference" for _s, rule in hits):
return hits, None
# ITS OWN SOURCE, and both sides of the trade logged. #2463's own finding
# is the warning rather than the precedent here: the hit that slot pushed
# OUT was in retrieval_logs while the query that pushed it out was not, so
# the slot could never be judged against what it displaced.
found, _kept, _fresh = await _ranked(
io, moment, source=PREFERENCE_SLOT_SOURCE, floor=floor, limit=1,
kind="preference",
)
seen = {rule.id for _s, rule in hits}
slot = [(s, r) for s, r in found
if r.kind == "preference" and r.id not in seen][:1]
if not slot:
return hits, None
slot_id = int(slot[0][1].id)
if slot_id not in moment.exclude:
try:
io.record_rule_surfaced(
user_id=moment.user_id, rule_ids=[slot_id],
source=PREFERENCE_SLOT_SOURCE,
)
except Exception: # noqa: BLE001 - observation never breaks the observed
_telemetry_failed(PREFERENCE_SLOT_SOURCE)
# IT EXTENDS, IT NEVER DISPLACES — and here it parts company with
# `reuse_slot`, which evicts its menu's weakest hit. The reason is the
# ledger rather than taste. A displaced hit was RETURNED by the general
# search and is sitting in that call's `retrieval_logs` row, but would not
# have been shown — so `prompt_rule`'s surfaced set would stop matching
# its own log row, and #3668's identity would break for a reason nothing
# in the data explains. That identity is the cheapest true statement
# available about this pair of tables, and milestone #379 is what it costs
# to lose it: five steps planned against a gap that was two counters
# disagreeing, not a write path dropping rows.
#
# The price is one extra line, only when the general search already filled
# the limit AND a preference cleared the bar without placing. Cheap, and
# it buys a surface whose two tables can always be checked against each
# other.
return hits + slot, slot_id
# ── The pipeline ─────────────────────────────────────────────────────────
async def run_rule_arm(
arm: RuleArm, moment: RuleMoment, *, floor: float, budget: int, io: RuleIO,
checkpoint_floor: float = 0.0,
) -> RuleResult:
"""Run one rule arm through every stage it takes, in the one order.
search → band → fresh/repeat split → call row → reserved slot → render →
surfacing rows → checkpoint. Fails open to an empty result.
"""
try:
_hits, kept, fresh = await _ranked(
io, moment, source=arm.source, floor=floor, limit=budget,
band=arm.band, kind=arm.kind,
)
shown = kept
if arm.preference_slot:
# BEFORE the bail-out, and the ordering is load-bearing. An empty
# general result is not proof that no preference qualifies: the
# general search overfetches by distance and then collapses, so a
# preference ranked below that window is invisible to it while a
# kind-filtered query finds it at once.
shown, _slot_id = await _reserve_slot_for_preference(
io, moment, shown, floor=floor,
)
# `shown`, not `fresh` (#3750): a call whose only hit is a repeat
# still has something to say, it just says it differently.
if not shown:
return RuleResult()
lines = [
_rule_hint_line(
rule, where=moment.where,
seen=rule.id in moment.exclude,
held=rule.id in moment.held,
# Rank decides volume (#3851): the ranker's best guess gets
# the trigger, the rest get cited.
compact=arm.compact_tail and idx > 0,
)
for idx, (_score, rule) in enumerate(shown)
]
# FRESH-ONLY (#3752): a repeat is a rendering decision, not a
# retrieval outcome, and counting it would inflate the denominator
# pull_through is read from. It is also exactly what this call's own
# row recorded, so the two tables stay equal (#3668). RANKED, not
# ambient: the source is in `rule_usage.RANKED_SOURCES`.
rule_ids = [rule.id for _score, rule in fresh]
source = arm.source
checkpoint = (
checkpoint_for(
kept, held=set(moment.held), floor=checkpoint_floor,
where=moment.checkpoint_where,
)
if arm.checkpoint else {}
)
if arm.stop_only:
rule_ids = [rid for rid in rule_ids if rid == checkpoint.get("rule_id")]
if rule_ids:
try:
io.record_rule_surfaced(
user_id=moment.user_id, rule_ids=rule_ids, source=source,
)
except Exception: # noqa: BLE001 - observation never breaks the observed
_telemetry_failed(source)
return RuleResult(
lines=lines, rule_ids=rule_ids,
shown_rule_ids=[rule.id for _score, rule in shown],
checkpoint=checkpoint,
shown=list(shown),
)
except Exception: # noqa: BLE001 - a recall aid never breaks its act
logger.debug("%s arm failed", arm.source, exc_info=True)
return RuleResult()
# ── The moment arm (milestone 458) ───────────────────────────────────────
#
# A LOOKUP beside the ranked arms, not one of them. A rule mounted on a moment
# arrives because somebody said it belongs there, not because the work's words
# resemble it — which is the whole point: a rule about WHEN work is finished
# never scores against what is said while finishing it. So there is no score,
# no floor and no band, and it writes no retrieval_logs row (the rulings arm's
# precedent, #4769): every score-shaped warning would misread a lookup. What it
# records is the surfacing, with the moment beside each row, so a mount's
# pull-through can be read per moment.
#
# It shares everything else with the ranked arms: the line renderer, the
# session ledger (a repeat renders as a citation, never hidden — #3750) and
# the fresh-only surfacing rows (#3752).
MOMENT_RULE_SOURCE = "moment_rule"
# A cap against a misconfiguration, not a ranking budget. Mounts are
# deliberate, so the ordinary case is a handful; a rulebook that mounted
# dozens of rules on one moment would otherwise bury the act it annotates.
# The remedy for a moment that hits this is fewer mounts (unmount through
# update_rule), which is why it is not a tuning dial.
MOMENT_RULE_LIMIT = 8
@dataclass(frozen=True)
class MomentIO:
"""The mounted-rule lookup and the recorder the moment arm reports to."""
lookup: Callable[..., Any]
"""`rulebooks.rules_on_moments`, or a stand-in with its signature."""
record_rule_surfaced: Callable[..., Any]
def _reached_by(hit: dict) -> str:
"""How the line names what reached a moment: the action, as mapped."""
return f"`{hit.get('match') or hit.get('tool') or '?'}`"
async def run_moment_arm(
reached: list[dict], moment: RuleMoment, *, io: MomentIO,
limit: int = MOMENT_RULE_LIMIT,
) -> RuleResult:
"""Deliver the rules mounted on the moments this act reached.
`reached` is `moment_actions.resolve`'s answer — each moment with the
action that reached it — and every line says both ("at work.deliver,
reached by `git push`"), so a moment that fired on the wrong act is
visible in the line itself and can be corrected in the session with
`unmap_action` (step 2's ruling). Fresh rules come first and carry their
trigger; a rule the session was already shown is cited, not repeated in
full. Fails open to an empty result.
"""
try:
names = [hit["moment"] for hit in reached]
if not names:
return RuleResult()
pairs = await io.lookup(moment.user_id, names, moment.project_id or None)
if not pairs:
return RuleResult()
by_moment = {hit["moment"]: hit for hit in reached}
# Stable: catalog order within each half, fresh half first.
shown = sorted(pairs, key=lambda pair: pair[0].id in moment.exclude)[:limit]
lines = []
for rule, at in shown:
seen = rule.id in moment.exclude
lines.append(_rule_hint_line(
rule,
where=f"at {at}, reached by {_reached_by(by_moment.get(at, {}))}",
seen=seen, held=rule.id in moment.held,
# A repeat is cited rather than quoted: the session was told
# which moment brought it the first time.
compact=seen,
))
fresh = [(rule, at) for rule, at in shown if rule.id not in moment.exclude]
rule_ids = [rule.id for rule, _at in fresh]
if rule_ids:
try:
io.record_rule_surfaced(
user_id=moment.user_id, rule_ids=rule_ids,
source=MOMENT_RULE_SOURCE,
detail={rule.id: at for rule, at in fresh},
)
except Exception: # noqa: BLE001 - observation never breaks the observed
_telemetry_failed(MOMENT_RULE_SOURCE)
return RuleResult(
lines=lines, rule_ids=rule_ids,
shown_rule_ids=[rule.id for rule, _at in shown],
shown=[(None, rule) for rule, _at in shown],
)
except Exception: # noqa: BLE001 - a recall aid never breaks its act
logger.debug("%s arm failed", MOMENT_RULE_SOURCE, exc_info=True)
return RuleResult()
+32 -2
View File
@@ -34,14 +34,16 @@ every tunable surface also appears here, so the two cannot drift apart.
ADDING AN ARM MEANS ADDING A ROW HERE. `tests/test_retrieval_registry.py` ADDING AN ARM MEANS ADDING A ROW HERE. `tests/test_retrieval_registry.py`
walks the call sites with the ast module and fails on a source it cannot find walks the call sites with the ast module and fails on a source it cannot find
below — deliberately not a grep, because two of the sources in this file below — deliberately not a grep, because two of the sources in this file
(`wide_net`, `report_preference`) reach their recorder as `source=SOURCE` (`wide_net`, `preference_slot`) reach their recorder through a
through a module constant and a grep for `source="` misses both. That is the module constant and a grep for `source="` misses both. That is the
narrowing #3191 warns about, caught here in the act. narrowing #3191 warns about, caught here in the act.
""" """
from __future__ import annotations from __future__ import annotations
from dataclasses import dataclass from dataclasses import dataclass
from scribe.services.retrieval_pipeline import MOMENT_RULE_SOURCE, RULE_ARMS, RULE_SOURCES
# ── How a point is reached ──────────────────────────────────────────────── # ── How a point is reached ────────────────────────────────────────────────
# #
# UNBIDDEN: fires on its own, against a query the agent did not write — a # UNBIDDEN: fires on its own, against a query the agent did not write — a
@@ -133,6 +135,9 @@ POINTS: dict[str, Point] = dict([
_p("write_path_rule", UNBIDDEN, "rules that may govern the file being written"), _p("write_path_rule", UNBIDDEN, "rules that may govern the file being written"),
_p("pre_tool_rule", UNBIDDEN, "rules that may govern a command about to run"), _p("pre_tool_rule", UNBIDDEN, "rules that may govern a command about to run"),
_p("prompt_rule", UNBIDDEN, "rules that may govern what the operator just asked"), _p("prompt_rule", UNBIDDEN, "rules that may govern what the operator just asked"),
_p("reply_rule", UNBIDDEN,
"a rule that holds the finished reply for one read — the backstop "
"for whatever the earlier arms missed"),
_p("preference_slot", UNBIDDEN, _p("preference_slot", UNBIDDEN,
"the one line reserved for a preference at the prompt boundary"), "the one line reserved for a preference at the prompt boundary"),
_p("reuse_slot", UNBIDDEN, "the one line reserved for a reusable snippet"), _p("reuse_slot", UNBIDDEN, "the one line reserved for a reusable snippet"),
@@ -161,6 +166,15 @@ POINTS: dict[str, Point] = dict([
"the fixed question asked when a task finishes: how should this report read", "the fixed question asked when a task finishes: how should this report read",
fixed_query=True), fixed_query=True),
# The moment arm (milestone 458). A LOOKUP, not a ranker: a rule mounted on
# a moment arrives when that moment happens, with no score, so it writes no
# retrieval_logs row and no score-shaped warning can apply. Its rows are in
# `rule_usage_events`, each carrying the moment in `detail`.
_p("moment_rule", UNBIDDEN,
"rules mounted on a moment of work, delivered when that moment happens",
expects_traffic=False,
quiet_because="speaks only once some rule is mounted on a moment; an "
"install where none is mounted is correctly silent here"),
# The rulings arm (milestone 444). A LOOKUP, not a ranker: a path falls # The rulings arm (milestone 444). A LOOKUP, not a ranker: a path falls
# under a System's patterns or it does not, so nothing here writes to # under a System's patterns or it does not, so nothing here writes to
# retrieval_logs and no score-shaped warning can apply. Its rows are in # retrieval_logs and no score-shaped warning can apply. Its rows are in
@@ -258,6 +272,22 @@ FAN_OUT_SITES: dict[str, tuple[str, ...]] = {
"enter_project", "start_planning", "get_task", "get_project", "enter_project", "start_planning", "get_task", "get_project",
"get_milestone", "get_milestone",
), ),
# The one retrieval pipeline (milestone 456): every rule arm's call row and
# surfacing rows are written by the same two sites, so `source` arrives as
# a value. The values are the pipeline's own specs, read rather than
# restated, so an arm added there is declared here by construction.
"scribe/services/retrieval_pipeline.py::record_retrieval(source=source)": (
RULE_SOURCES
),
"scribe/services/retrieval_pipeline.py::record_rule_surfaced(source=source)": (
tuple(arm.source for arm in RULE_ARMS)
),
# The reply moment's mounted half (milestone 458): it records under the
# moment arm's source, read from the pipeline's constant so the two doors
# cannot drift apart on the name.
"scribe/services/moment_delivery.py::record_rule_surfaced(source=<expression>)": (
MOMENT_RULE_SOURCE,
),
} }
+15
View File
@@ -207,6 +207,21 @@ SURFACES: dict[str, Surface] = {
over="preferences", over="preferences",
fires="when a task finishes", fires="when a task finishes",
), ),
# THE REPLY BACKSTOP (milestone 458, folded in from 456 step 8). Its floor
# is a STOP bar, not a hint bar: at the end of a turn nothing can be shown
# beside the reply, so a hit either holds the reply for one read or says
# nothing. Hence a default at the checkpoint's level and a budget of one —
# the call row's results are then exactly the rule that would hold.
"reply_rule": Surface(
name="reply_rule",
floor_key="kb_replyrule_threshold",
floor_default=0.80,
budget_key="kb_replyrule_top_k",
budget_default=1,
asks="the reply that ends a turn, against rule triggers",
over="global rules plus the bound project's own",
fires="once per turn, when the reply is finished",
),
} }
# Reserved slots are deliberately absent. `preference_slot`, `reuse_slot` and # Reserved slots are deliberately absent. `preference_slot`, `reuse_slot` and
+55
View File
@@ -0,0 +1,55 @@
"""Which rules a caller may be shown — a rule's HOME, stated once (milestone 414).
A rule lives in a rulebook topic — GLOBAL, it applies wherever its owner
works — or on one project, where it applies to that project and nowhere else.
Every read that hands rules to a session answers the same question about that
home, so it is answered here and nowhere else:
- `project_id` unset: global rules only. A caller that does not say gets the
safe failure, which is surfacing less rather than another project's rules.
- `project_id=N`: global rules plus project N's own, and N's only when the
caller can read that project (access.can_read_project, so a shared project's
rules reach its collaborators too).
- `everywhere=True`: every rule the caller owns, in any home — the explicit
whole-rulebook question.
It was written inline in `semantic_search_rules` until the moment lookup
(milestone 458) needed the same answer. Two copies of a scope clause is how
one of them starts surfacing another project's rules, so it moved here first.
"""
from __future__ import annotations
from sqlalchemy import or_
from scribe.services.access import can_read_project
async def rule_home(user_id: int, project_id: int | None = None, *, everywhere: bool = False):
"""The WHERE clause for the rules this caller may be shown here.
Needs the statement joined through `joined_to_homes`, which outer-joins
the three tables the clause reads. A rule is in a topic XOR on a project
(migration 0059), so it matches exactly one arm of whichever clause
applies.
"""
from scribe.models.project import Project
from scribe.models.rulebook import Rule, Rulebook
global_rule = Rulebook.owner_user_id == user_id
if everywhere:
return or_(global_rule, Project.user_id == user_id)
if project_id and await can_read_project(user_id, project_id):
return or_(global_rule, Rule.project_id == project_id)
return global_rule
def joined_to_homes(stmt):
"""Outer-join a statement over `Rule` to the tables `rule_home` reads."""
from scribe.models.project import Project
from scribe.models.rulebook import Rule, Rulebook, RulebookTopic
return (
stmt.outerjoin(RulebookTopic, Rule.topic_id == RulebookTopic.id)
.outerjoin(Rulebook, RulebookTopic.rulebook_id == Rulebook.id)
.outerjoin(Project, Rule.project_id == Project.id)
)
+16 -1
View File
@@ -101,6 +101,15 @@ logger = logging.getLogger(__name__)
# Add a source here only when a ranker picked it. # Add a source here only when a ranker picked it.
RANKED_SOURCES = ( RANKED_SOURCES = (
"write_path_rule", "pre_tool_rule", "prompt_rule", "write_path_rule", "pre_tool_rule", "prompt_rule",
# The reply backstop records only the rule that HELD the reply — a claim
# put in front of the reader as plainly as any line, and one a pull
# confirms or refutes the same way.
"reply_rule",
# A rule mounted on a moment (milestone 458). Not a ranker's pick, but
# not bulk either: somebody decided this rule applies at this moment, and
# that is exactly the claim a pull can confirm or refute. Its
# pull-through is the evidence for whether a mount earns its line.
"moment_rule",
# A reserved slot is a ranker's choice twice over — it ran a query AND # A reserved slot is a ranker's choice twice over — it ran a query AND
# decided a kind was worth guaranteeing a place. Left out, its line would # decided a kind was worth guaranteeing a place. Left out, its line would
# be counted as bulk delivery and drop out of the denominator, so the one # be counted as bulk delivery and drop out of the denominator, so the one
@@ -154,7 +163,8 @@ def _schedule(rows: list[dict]) -> None:
def record_rule_surfaced( def record_rule_surfaced(
*, user_id: int | None, rule_ids: list[int] | set[int], source: str *, user_id: int | None, rule_ids: list[int] | set[int], source: str,
detail: dict[int, str] | None = None,
) -> None: ) -> None:
"""Fire-and-forget: record that these rules were shown to the agent. """Fire-and-forget: record that these rules were shown to the agent.
@@ -173,6 +183,10 @@ def record_rule_surfaced(
from the other end — everything in a preload IS shown. `source` is what from the other end — everything in a preload IS shown. `source` is what
separates the two afterwards (see `RANKED_SOURCES`); this function does not separates the two afterwards (see `RANKED_SOURCES`); this function does not
care which kind it is recording. care which kind it is recording.
`detail` maps a rule id to what to record beside its row — the moment a
mounted rule arrived at (milestone 458), so a mount's pull-through can be
read per moment rather than only per arm.
""" """
try: try:
rows = [ rows = [
@@ -181,6 +195,7 @@ def record_rule_surfaced(
"rule_id": int(rid), "rule_id": int(rid),
"event": SURFACED, "event": SURFACED,
"source": source, "source": source,
**({"detail": detail[rid]} if detail and rid in detail else {}),
} }
for rid in rule_ids for rid in rule_ids
] ]
+128 -4
View File
@@ -433,26 +433,36 @@ async def co_surfaced_partners(user_id: int, rule_ids: list[int]) -> list[Rule]:
return out return out
async def rule_detail(user_id: int, rule: Rule, system_ids: list[int] | None = None) -> dict: async def rule_detail(
"""The full record, with its areas and edges attached. user_id: int, rule: Rule, system_ids: list[int] | None = None,
moments: list[str] | None = None,
) -> dict:
"""The full record, with its areas, moments and edges attached.
ONE seam for both doors and every write path, so create, update and get ONE seam for both doors and every write path, so create, update and get
cannot disagree about what a rule looks like coming back — the same cannot disagree about what a rule looks like coming back — the same
reasoning as attach_relations for notes (#2859), and the same reasoning reasoning as attach_relations for notes (#2859), and the same reasoning
rule_brief exists for one level down. rule_brief exists for one level down.
`system_ids=None` means "leave the tags alone"; a list (including []) `system_ids=None` / `moments=None` mean "leave them alone"; a list
REPLACES them. (including []) REPLACES them. Doors validate `moments` with
`moments.require_moments` BEFORE their create, so an unknown name is
refused without leaving a rule behind.
""" """
if system_ids is not None: if system_ids is not None:
await set_rule_systems(rule.id, user_id, system_ids) await set_rule_systems(rule.id, user_id, system_ids)
if moments is not None:
await set_rule_moments(rule.id, user_id, moments)
data = rule.to_dict() data = rule.to_dict()
systems = (await list_rule_systems([rule.id])).get(rule.id, []) systems = (await list_rule_systems([rule.id])).get(rule.id, [])
mounted = (await list_rule_moments([rule.id])).get(rule.id, [])
relations = (await list_rule_relations([rule.id])).get(rule.id, []) relations = (await list_rule_relations([rule.id])).get(rule.id, [])
# Attached only when present (#2483): an empty key reads as a capability # Attached only when present (#2483): an empty key reads as a capability
# the record has and isn't using, which is a different claim. # the record has and isn't using, which is a different claim.
if systems: if systems:
data["systems"] = systems data["systems"] = systems
if mounted:
data["moments"] = mounted
if relations: if relations:
data["relations"] = relations data["relations"] = relations
# The concrete situations judged (or proposed) to be instances of this # The concrete situations judged (or proposed) to be instances of this
@@ -1100,6 +1110,120 @@ async def set_rule_systems(
return sorted(wanted) return sorted(wanted)
async def set_rule_moments(
rule_id: int, user_id: int, moments: list[str],
) -> list[str] | None:
"""Replace which MOMENTS a rule arrives at (milestone 458). None if not owned.
Set-semantics like set_rule_systems: the list given IS the state after.
Every name is checked against the catalog first and the whole write is
refused on the first unknown one — a typo stored as a mount would read
back as attached and never fire, the silent miss moments exist to end.
"""
from scribe.models.rulebook import rule_moments as rule_moments_t
from scribe.services.moments import require_moments
wanted = require_moments(moments) or []
async with async_session() as session:
rule = await _fetch_owned_rule(session, rule_id, user_id)
if rule is None:
return None
await session.execute(
sql_delete(rule_moments_t).where(rule_moments_t.c.rule_id == rule_id)
)
for moment in wanted:
await session.execute(
insert(rule_moments_t).values(rule_id=rule_id, moment=moment)
)
await session.commit()
return wanted
async def list_rule_moments(rule_ids: list[int]) -> dict[int, list[str]]:
"""The moments each of a batch of rules is mounted on, in catalog order."""
from scribe.models.rulebook import rule_moments as rule_moments_t
from scribe.services.moments import MOMENTS
if not rule_ids:
return {}
async with async_session() as session:
rows = (await session.execute(
select(rule_moments_t.c.rule_id, rule_moments_t.c.moment)
.where(rule_moments_t.c.rule_id.in_(rule_ids))
)).all()
order = {name: i for i, name in enumerate(MOMENTS)}
out: dict[int, list[str]] = {}
for rule_id, moment in rows:
out.setdefault(rule_id, []).append(moment)
for names in out.values():
names.sort(key=lambda n: (order.get(n, len(order)), n))
return out
async def rules_on_moments(
user_id: int, moments: list[str], project_id: int | None = None,
) -> list[tuple[Rule, str]]:
"""The rules mounted on any of these moments that this caller may be shown.
The delivery read (milestone 458 step 4), and a LOOKUP: no score, no bar.
Scoped by the same home clause the semantic search uses (rule_scope), so
a mount never delivers another project's rule. Each rule appears once,
under the first of `moments` it is mounted on — the caller passes them in
catalog order, so that is the most specific moment the work reached.
"""
from scribe.models.rulebook import rule_moments as rule_moments_t
from scribe.services.rule_scope import joined_to_homes, rule_home
if not moments:
return []
home = await rule_home(user_id, project_id)
async with async_session() as session:
rows = (await session.execute(
joined_to_homes(
select(Rule, rule_moments_t.c.moment)
.select_from(rule_moments_t)
.join(Rule, Rule.id == rule_moments_t.c.rule_id)
)
.where(
rule_moments_t.c.moment.in_(list(moments)),
Rule.deleted_at.is_(None),
home,
)
.order_by(Rule.id)
)).all()
rank = {m: i for i, m in enumerate(moments)}
first: dict[int, tuple[Rule, str]] = {}
for rule, moment in rows:
held = first.get(rule.id)
if held is None or rank[moment] < rank[held[1]]:
first[rule.id] = (rule, moment)
return sorted(first.values(), key=lambda pair: (rank[pair[1]], pair[0].id))
async def mounted_moments(user_id: int) -> set[str]:
"""Every moment at least one of this caller's live rules is mounted on.
In ANY home (rule_scope's `everywhere`): this answers "can a call reach
anything at all", which is what lets the plugin stay off the wire for a
tool whose moments carry nothing. A superset is the safe direction — the
delivery read still scopes by project.
"""
from scribe.models.rulebook import rule_moments as rule_moments_t
from scribe.services.rule_scope import joined_to_homes, rule_home
home = await rule_home(user_id, everywhere=True)
async with async_session() as session:
rows = (await session.execute(
joined_to_homes(
select(rule_moments_t.c.moment).distinct()
.select_from(rule_moments_t)
.join(Rule, Rule.id == rule_moments_t.c.rule_id)
)
.where(Rule.deleted_at.is_(None), home)
)).scalars().all()
return set(rows)
async def list_rule_systems(rule_ids: list[int]) -> dict[int, list[dict]]: async def list_rule_systems(rule_ids: list[int]) -> dict[int, list[dict]]:
"""The canon tags for a batch of rules, keyed by rule id. """The canon tags for a batch of rules, keyed by rule id.
+21
View File
@@ -215,6 +215,27 @@ def _no_lesson_rule_links(request):
yield yield
@pytest.fixture(autouse=True)
def _no_moment_delivery(request):
"""Stub the moment lookup the mapped MCP tools attach (milestone 458).
update_task, the create_* record tools and the milestone tools now hand
back the rules mounted on the moment they mark, and finding those is two
database reads. Patched BENEATH the attach, at `deliver_for_act`, so the
attach itself — its fail-open and its non-dict pass-through — still runs.
Skipped for integration tests; tests/test_moment_delivery.py binds the
real one.
"""
if request.node.get_closest_marker("integration"):
yield
return
from scribe.services.retrieval_pipeline import RuleResult
with patch("scribe.services.moment_delivery.deliver_for_act",
AsyncMock(return_value=([], RuleResult()))):
yield
@pytest.fixture(autouse=True) @pytest.fixture(autouse=True)
def _no_rule_arm(): def _no_rule_arm():
"""Stub the write-path hint's standing-RULES arm (milestone 307). """Stub the write-path hint's standing-RULES arm (milestone 307).
+26 -5
View File
@@ -315,7 +315,7 @@ def plain_rule_detail():
stubbing the seam slightly differently is a test asserting something stubbing the seam slightly differently is a test asserting something
slightly different than it appears to. slightly different than it appears to.
""" """
async def _detail(_uid, rule, _system_ids=None): async def _detail(_uid, rule, _system_ids=None, _moments=None):
return rule.to_dict() return rule.to_dict()
return patch("scribe.mcp.tools.rulebooks.rulebooks_svc.rule_detail", _detail) return patch("scribe.mcp.tools.rulebooks.rulebooks_svc.rule_detail", _detail)
@@ -349,7 +349,8 @@ def design_token_stub(name, value_by_mode, group_name=None, purpose=None,
@contextmanager @contextmanager
def http_sink(reply: bytes = b'{"context":"","note_ids":[]}'): def http_sink(reply: bytes = b'{"context":"","note_ids":[]}',
by_path: dict[str, bytes] | None = None):
"""A throwaway local HTTP listener for hook end-to-end tests: yields """A throwaway local HTTP listener for hook end-to-end tests: yields
``(port, seen)`` where ``seen`` collects every GET's parsed query string ``(port, seen)`` where ``seen`` collects every GET's parsed query string
(one dict per request, in order). Lets the shell be tested end to end — (one dict per request, in order). Lets the shell be tested end to end —
@@ -357,6 +358,11 @@ def http_sink(reply: bytes = b'{"context":"","note_ids":[]}'):
Three test modules each carried their own ``_Sink`` handler before #2904 Three test modules each carried their own ``_Sink`` handler before #2904
consolidated them here; pass ``reply`` for the body the hook should see. consolidated them here; pass ``reply`` for the body the hook should see.
``by_path`` is for a hook that calls more than one route: each request is
answered with the reply for its path (``reply`` when none is listed), and
every entry in ``seen`` then also carries ``_path`` and, for a POST,
``_body`` — the raw request body, decoded.
""" """
import http.server import http.server
import threading import threading
@@ -365,12 +371,27 @@ def http_sink(reply: bytes = b'{"context":"","note_ids":[]}'):
seen: list[dict] = [] seen: list[dict] = []
class _Sink(http.server.BaseHTTPRequestHandler): class _Sink(http.server.BaseHTTPRequestHandler):
def do_GET(self): def _answer(self, body: str | None):
seen.append(urllib.parse.parse_qs(urllib.parse.urlparse(self.path).query)) parsed = urllib.parse.urlparse(self.path)
entry = urllib.parse.parse_qs(parsed.query)
out = reply
if by_path is not None:
entry["_path"] = parsed.path
out = by_path.get(parsed.path, reply)
if body is not None:
entry["_body"] = body
seen.append(entry)
self.send_response(200) self.send_response(200)
self.send_header("Content-Type", "application/json") self.send_header("Content-Type", "application/json")
self.end_headers() self.end_headers()
self.wfile.write(reply) self.wfile.write(out)
def do_GET(self):
self._answer(None)
def do_POST(self):
size = int(self.headers.get("Content-Length") or 0)
self._answer(self.rfile.read(size).decode("utf-8", "replace"))
def log_message(self, *a): def log_message(self, *a):
pass pass
+99
View File
@@ -72,3 +72,102 @@ async def test_get_embedding_propagates_model_load_failures():
): ):
with pytest.raises(RuntimeError, match="model load failed"): with pytest.raises(RuntimeError, match="model load failed"):
await get_embedding("x") await get_embedding("x")
# ─── query_embedding_memo (milestone 456 step 1) ─────────────────────────────
def _counting_embeddings():
"""A stand-in for get_embeddings that counts model calls and returns a
vector derived from the text, so a wrong memo key returns a wrong vector."""
calls: list[list[str]] = []
async def fake(texts):
calls.append(list(texts))
return [[float(len(t)), 1.0] for t in texts]
return fake, calls
@pytest.mark.asyncio
async def test_memo_embeds_a_repeated_query_once_per_scope():
from scribe.services import embeddings as emb
fake, calls = _counting_embeddings()
with patch.object(emb, "get_embeddings", fake):
with emb.query_embedding_memo():
first = await emb.get_embedding("push to dev")
again = await emb.get_embedding("push to dev")
other = await emb.get_embedding("merge to main")
assert first == again == [11.0, 1.0]
assert other == [13.0, 1.0]
assert calls == [["push to dev"], ["merge to main"]]
@pytest.mark.asyncio
async def test_without_a_scope_every_call_reaches_the_model():
"""The default is None: outside a request scope nothing is remembered, so
every caller other than the plugin routes behaves exactly as before."""
from scribe.services import embeddings as emb
fake, calls = _counting_embeddings()
with patch.object(emb, "get_embeddings", fake):
await emb.get_embedding("q")
await emb.get_embedding("q")
assert calls == [["q"], ["q"]]
@pytest.mark.asyncio
async def test_a_scope_does_not_leak_into_the_next_one():
from scribe.services import embeddings as emb
fake, calls = _counting_embeddings()
with patch.object(emb, "get_embeddings", fake):
with emb.query_embedding_memo():
await emb.get_embedding("q")
with emb.query_embedding_memo():
await emb.get_embedding("q")
await emb.get_embedding("q")
assert calls == [["q"], ["q"], ["q"]]
assert emb._query_memo.get() is None
@pytest.mark.asyncio
async def test_a_failed_embedding_is_not_remembered():
"""The next arm in the request must try again and degrade on its own,
not inherit a failure as if it were a vector."""
from scribe.services import embeddings as emb
attempts = {"n": 0}
async def flaky(texts):
attempts["n"] += 1
if attempts["n"] == 1:
raise RuntimeError("model busy")
return [[0.5, 0.5] for _ in texts]
with patch.object(emb, "get_embeddings", flaky):
with emb.query_embedding_memo():
with pytest.raises(RuntimeError):
await emb.get_embedding("q")
assert await emb.get_embedding("q") == [0.5, 0.5]
assert attempts["n"] == 2
@pytest.mark.asyncio
async def test_the_route_decorator_opens_one_scope_per_call():
from scribe.services import embeddings as emb
fake, calls = _counting_embeddings()
@emb.memoized_query_embeddings
async def handler(q):
await emb.get_embedding(q)
await emb.get_embedding(q)
return emb._query_memo.get() is not None
with patch.object(emb, "get_embeddings", fake):
assert await handler("a") is True
assert await handler("a") is True
assert calls == [["a"], ["a"]]
assert handler.__name__ == "handler"
+119
View File
@@ -0,0 +1,119 @@
"""Real-Postgres tests for an install's moment mappings (milestone 458 step 2).
What these pin is what the in-session correction promises the operator: a
mapping made is in force on the next call, a default switched off stays off,
switching it back on leaves no residue, one user's corrections are not
another's, and asking to remove what nothing maps is refused rather than
recorded. Every one of those is a claim about rows, so none of it is mocked.
"""
import pytest
import pytest_asyncio
from sqlalchemy import delete, select
from scribe.models import async_session
from scribe.models.moment_mapping import MomentMapping
from scribe.services import moment_actions as ma
from tests.helpers import ensure_user
pytestmark = [pytest.mark.integration, pytest.mark.usefixtures("_dispose_engine")]
OWNER_USERNAME = "moment_mapping_owner"
STRANGER_USERNAME = "moment_mapping_stranger"
SHIP = {"command": "make ship"}
CURL = {"command": "curl localhost:8000"}
@pytest_asyncio.fixture
async def users():
async with async_session() as s:
owner = await ensure_user(s, OWNER_USERNAME)
stranger = await ensure_user(s, STRANGER_USERNAME)
await s.commit()
ids = (owner.id, stranger.id)
# At SETUP: the lane shares one database, so a previous run's rows
# are cleared before this one reads anything.
await s.execute(delete(MomentMapping).where(MomentMapping.user_id.in_(ids)))
await s.commit()
return ids
async def _reached(uid, tool, tool_input):
return [h["moment"] for h in await ma.moments_for(uid, tool, tool_input)]
async def _rows(uid):
async with async_session() as s:
return (await s.execute(
select(MomentMapping).where(MomentMapping.user_id == uid)
)).scalars().all()
async def test_a_mapping_is_in_force_on_the_next_call(users):
uid, _ = users
assert "work.deliver" not in await _reached(uid, "Bash", SHIP)
out = await ma.map_action(uid, "Bash", "make ship", "work.deliver",
reason="this install ships with make")
assert out["change"] == "mapped"
assert "work.deliver" in [h["moment"] for h in out["now_reaches"]]
assert "work.deliver" in await _reached(uid, "Bash", SHIP)
async def test_mapping_twice_is_one_row(users):
uid, _ = users
await ma.map_action(uid, "Bash", "make ship", "work.deliver")
again = await ma.map_action(uid, "Bash", "make ship", "work.deliver", reason="why")
assert again["change"].startswith("already mapped")
rows = await _rows(uid)
assert len(rows) == 1 and rows[0].reason == "why"
async def test_one_users_corrections_are_not_anothers(users):
uid, other = users
await ma.map_action(uid, "Bash", "make ship", "work.deliver")
assert "work.deliver" not in await _reached(other, "Bash", SHIP)
async def test_unmapping_an_installs_mapping_deletes_it(users):
uid, _ = users
await ma.map_action(uid, "Bash", "make ship", "work.deliver")
out = await ma.unmap_action(uid, "Bash", "make ship", "work.deliver")
assert out["change"] == "removed this install's mapping"
assert await _rows(uid) == []
assert "work.deliver" not in await _reached(uid, "Bash", SHIP)
async def test_a_default_switched_off_stays_off_and_switches_back_cleanly(users):
uid, _ = users
assert "env.reach" in await _reached(uid, "Bash", CURL)
off = await ma.unmap_action(uid, "Bash", "curl", "env.reach",
reason="curl here only hits the local dev server")
assert off["change"] == "switched off the shipped default"
assert await _reached(uid, "Bash", CURL) == ["work.run"]
[row] = await _rows(uid)
assert row.effect == ma.REMOVE
by_moment = await ma.actions_by_moment(uid)
assert [r["match"] for r in by_moment["removed_defaults"]] == ["curl"]
assert {"tool": "Bash", "match": "curl", "via": ma.DEFAULT} not in by_moment["actions"]["env.reach"]
on = await ma.map_action(uid, "Bash", "curl", "env.reach")
assert on["change"] == "restored the shipped default"
assert await _rows(uid) == []
assert "env.reach" in await _reached(uid, "Bash", CURL)
async def test_mapping_a_default_already_in_force_writes_nothing(users):
uid, _ = users
out = await ma.map_action(uid, "update_task", "status=done", "work.finish")
assert out["change"].startswith("already a shipped default")
assert await _rows(uid) == []
async def test_removing_what_nothing_maps_is_refused(users):
uid, _ = users
with pytest.raises(ValueError, match="nothing to remove"):
await ma.unmap_action(uid, "Bash", "make ship", "work.deliver")
assert await _rows(uid) == []
+201
View File
@@ -0,0 +1,201 @@
"""Real-Postgres tests for rule mounts (milestone 458 step 3).
A mount is the claim "this rule arrives at this moment". These pin what that
claim depends on:
- the set given is the set stored, and a refused name stores nothing;
- only the owner can mount;
- the rule's detail reads its mounts back;
- a backup carries them, attached to the RESTORED rule's new id rather than
the old number;
- the delivery lookup (step 4) answers by home: a project's mounted rule
reaches only work bound to that project, and nobody else's mount arrives.
"""
import pytest
import pytest_asyncio
from sqlalchemy import select
from scribe.models import async_session
from scribe.models.project import Project
from scribe.models.rulebook import Rule, Rulebook, RulebookTopic
from scribe.models.user import User
from scribe.services import backup
from scribe.services import projects as projects_svc
from scribe.services import rulebooks as rulebooks_svc
from tests.helpers import ensure_user
pytestmark = [pytest.mark.integration, pytest.mark.usefixtures("_dispose_engine")]
OWNER_USERNAME = "rule_moments_owner"
STRANGER_USERNAME = "rule_moments_stranger"
RESTORED_USERNAME = "rule_moments_restored"
TRIGGER = "about to tell the operator a piece of work is finished"
async def _purge_books(username: str) -> None:
"""At SETUP: rule writes fire a detached embedding refresh, and a teardown
delete races it (test_integration_rule_versions records why)."""
async with async_session() as s:
for user in (await s.execute(
select(User).where(User.username == username)
)).scalars().all():
for book in (await s.execute(
select(Rulebook).where(Rulebook.owner_user_id == user.id)
)).scalars().all():
await s.delete(book)
await s.commit()
@pytest_asyncio.fixture
async def world():
for name in (OWNER_USERNAME, STRANGER_USERNAME, RESTORED_USERNAME):
await _purge_books(name)
# The restore below creates a user under a unique username; a previous
# run's must be gone first, as in the sibling roundtrip files.
async with async_session() as s:
for user in (await s.execute(
select(User).where(User.username == RESTORED_USERNAME)
)).scalars().all():
await s.delete(user)
await s.commit()
async with async_session() as s:
owner = await ensure_user(s, OWNER_USERNAME)
stranger = await ensure_user(s, STRANGER_USERNAME)
await s.commit()
uid, sid = owner.id, stranger.id
book = await rulebooks_svc.create_rulebook(uid, "Moment fixtures")
topic = await rulebooks_svc.create_topic(book.id, uid, "verification")
rule = await rulebooks_svc.create_rule(
topic.id, uid, "Done means delivered and checked",
"Work is done when it is delivered and its automated checks pass.",
when_to_apply=TRIGGER,
)
return {"uid": uid, "sid": sid, "book_id": book.id, "rule": rule}
async def test_the_set_given_is_the_set_stored(world):
uid, rule = world["uid"], world["rule"]
assert await rulebooks_svc.set_rule_moments(
rule.id, uid, ["reply.report", " Work.Finish"],
) == ["reply.report", "work.finish"]
# Read back in CATALOG order, whatever order they were given in.
assert (await rulebooks_svc.list_rule_moments([rule.id]))[rule.id] == [
"work.finish", "reply.report",
]
await rulebooks_svc.set_rule_moments(rule.id, uid, ["work.deliver"])
assert (await rulebooks_svc.list_rule_moments([rule.id]))[rule.id] == ["work.deliver"]
await rulebooks_svc.set_rule_moments(rule.id, uid, [])
assert rule.id not in await rulebooks_svc.list_rule_moments([rule.id])
async def test_a_refused_name_stores_nothing(world):
uid, rule = world["uid"], world["rule"]
await rulebooks_svc.set_rule_moments(rule.id, uid, ["work.finish"])
with pytest.raises(ValueError, match="unknown moment"):
await rulebooks_svc.set_rule_moments(rule.id, uid, ["work.deliver", "work.shipped"])
assert (await rulebooks_svc.list_rule_moments([rule.id]))[rule.id] == ["work.finish"]
async def test_only_the_owner_can_mount(world):
rule = world["rule"]
assert await rulebooks_svc.set_rule_moments(rule.id, world["sid"], ["work.finish"]) is None
assert rule.id not in await rulebooks_svc.list_rule_moments([rule.id])
async def test_the_detail_carries_the_mounts_and_omits_an_empty_set(world):
uid, rule = world["uid"], world["rule"]
bare = await rulebooks_svc.rule_detail(uid, rule)
assert "moments" not in bare
mounted = await rulebooks_svc.rule_detail(uid, rule, moments=["work.finish"])
assert mounted["moments"] == ["work.finish"]
# None leaves them alone: a later read without the argument still has them.
assert (await rulebooks_svc.rule_detail(uid, rule))["moments"] == ["work.finish"]
async def test_a_backup_carries_the_mounts_to_the_restored_rule(world):
uid, rule = world["uid"], world["rule"]
await rulebooks_svc.set_rule_moments(rule.id, uid, ["work.finish", "reply.report"])
async with async_session() as s:
payload = {
"version": backup.BACKUP_VERSION,
"users": backup._user_rows(
[(await s.execute(select(User).where(User.id == uid))).scalars().one()]
),
"rulebooks": backup._rulebook_rows(
[(await s.execute(select(Rulebook).where(
Rulebook.id == world["book_id"]))).scalars().one()]
),
"rulebook_topics": backup._topic_rows((await s.execute(
select(RulebookTopic).where(RulebookTopic.rulebook_id == world["book_id"])
)).scalars().all()),
"rules": backup._rule_rows(
[(await s.execute(select(Rule).where(Rule.id == rule.id))).scalars().one()]
),
}
exported = await backup.export_user_backup(uid)
payload["rule_moments"] = [
row for row in exported["rule_moments"] if row["rule_id"] == rule.id
]
assert sorted(r["moment"] for r in payload["rule_moments"]) == [
"reply.report", "work.finish",
]
payload["users"][0]["username"] = RESTORED_USERNAME
stats = await backup.restore_full_backup(payload)
assert stats["rule_moments"] == 2
async with async_session() as s:
restored_user = (await s.execute(
select(User).where(User.username == RESTORED_USERNAME)
)).scalars().one()
restored_rule = (await s.execute(
select(Rule).join(RulebookTopic, RulebookTopic.id == Rule.topic_id)
.join(Rulebook, Rulebook.id == RulebookTopic.rulebook_id)
.where(Rulebook.owner_user_id == restored_user.id)
)).scalars().one()
assert restored_rule.id != rule.id
assert (await rulebooks_svc.list_rule_moments([restored_rule.id]))[restored_rule.id] == [
"work.finish", "reply.report",
]
async def _purge_projects(username: str) -> None:
async with async_session() as s:
for user in (await s.execute(
select(User).where(User.username == username)
)).scalars().all():
for project in (await s.execute(
select(Project).where(Project.user_id == user.id)
)).scalars().all():
await s.delete(project)
await s.commit()
async def test_the_delivery_lookup_answers_by_home(world):
uid, sid, rule = world["uid"], world["sid"], world["rule"]
await _purge_projects(OWNER_USERNAME)
project = await projects_svc.create_project(uid, "Moment delivery fixture")
local = await rulebooks_svc.create_project_rule(
project.id, uid, "Ship notes with the release",
"A release carries its notes.", when_to_apply="publishing a release",
)
await rulebooks_svc.set_rule_moments(rule.id, uid, ["work.finish", "reply.report"])
await rulebooks_svc.set_rule_moments(local.id, uid, ["work.deliver"])
# Unbound: the global rule, under the FIRST of the given moments it is on.
got = await rulebooks_svc.rules_on_moments(uid, ["reply.report", "work.finish"])
assert [(r.id, m) for r, m in got] == [(rule.id, "reply.report")]
assert await rulebooks_svc.rules_on_moments(uid, ["work.deliver"]) == []
# Bound to the project: its own rule arrives too.
got = await rulebooks_svc.rules_on_moments(uid, ["work.deliver"], project.id)
assert [(r.id, m) for r, m in got] == [(local.id, "work.deliver")]
# Somebody else's mount never arrives — not even naming their project.
assert await rulebooks_svc.rules_on_moments(sid, ["work.finish", "work.deliver"], project.id) == []
# The plugin's "can anything arrive" answer spans every home.
assert await rulebooks_svc.mounted_moments(uid) >= {"work.finish", "reply.report", "work.deliver"}
assert not (await rulebooks_svc.mounted_moments(sid)) & {"work.finish", "work.deliver"}
+181
View File
@@ -0,0 +1,181 @@
"""Which actions reach which moment (milestone 458 step 2) — the pure half.
Matching and resolution take no database: the defaults are code and an
install's mappings arrive as rows, so every case below hands `resolve` the
rows it needs. The writes that store those rows are exercised against real
Postgres in `test_integration_moment_mappings.py`.
"""
from types import SimpleNamespace
from unittest.mock import AsyncMock, patch
import pytest
from scribe.services import moment_actions as ma
from scribe.services import moments
def _row(tool, match, moment, effect=ma.ADD):
return SimpleNamespace(tool=tool, match=match, moment=moment, effect=effect)
def _reached(tool, tool_input=None, mappings=()):
return [h["moment"] for h in ma.resolve(tool, tool_input, mappings)]
# ── the defaults ─────────────────────────────────────────────────────────
def test_every_default_reaches_a_real_moment():
"""A default naming a moment that is not in the catalog would never be
mounted on — the typo require_moment refuses at the door."""
bad = [a for a in ma.DEFAULT_ACTIONS if not moments.is_moment(a.moment)]
assert not bad, bad
def test_no_default_is_listed_twice():
keys = [(ma.tool_key(a.tool), a.match, a.moment) for a in ma.DEFAULT_ACTIONS]
assert len(keys) == len(set(keys))
def test_every_default_is_a_mapping_the_door_would_accept():
"""The defaults pass the same validation an install's mapping does, so a
default and an install's copy of it can never disagree about its shape."""
for a in ma.DEFAULT_ACTIONS:
assert ma._clean(a.tool, a.match, a.moment) == (a.tool, a.match, a.moment)
# ── matching ─────────────────────────────────────────────────────────────
@pytest.mark.parametrize("command", [
"git push origin dev",
"git push",
"cd app && git push",
"make lint; git push --tags",
"GIT_TRACE=1 git push",
" git push ",
])
def test_a_command_prefix_matches_any_segment_of_the_line(command):
assert "work.deliver" in _reached("Bash", {"command": command})
@pytest.mark.parametrize("command", ["git pushd", "echo git push", "git status"])
def test_a_command_prefix_needs_a_word_boundary_and_the_head_of_a_segment(command):
assert "work.deliver" not in _reached("Bash", {"command": command})
def test_one_action_reaches_every_moment_it_is():
"""`kubectl apply` is a run, a deliver and a reach outside the workspace."""
assert _reached("Bash", {"command": "kubectl apply -f deploy.yaml"}) == [
"work.run", "work.deliver", "env.reach",
]
def test_the_most_specific_action_is_the_one_named():
hits = ma.resolve("Bash", {"command": "git push"})
deliver = next(h for h in hits if h["moment"] == "work.deliver")
assert deliver["match"] == "git push"
run = next(h for h in hits if h["moment"] == "work.run")
assert run["match"] == ""
def test_an_ordinary_command_is_only_a_run():
assert _reached("Bash", {"command": "ls -la"}) == ["work.run"]
def test_an_unknown_tool_reaches_nothing():
assert _reached("SomeOtherTool", {"x": 1}) == []
@pytest.mark.parametrize("name", [
"mcp__plugin_scribe_scribe__update_task", "mcp__scribe__update_task",
"update_task", "Update_Task",
])
def test_the_server_prefix_and_case_do_not_matter(name):
assert _reached(name, {"task_id": 1, "status": "done"}) == ["work.finish"]
@pytest.mark.parametrize("status,expected", [
("done", ["work.finish"]),
("cancelled", ["work.finish"]),
("in_progress", ["work.start"]),
("todo", []),
("", []),
])
def test_arguments_select_the_moment(status, expected):
assert _reached("update_task", {"task_id": 1, "status": status}) == expected
def test_a_whole_tool_mapping_ignores_its_arguments():
assert _reached("Edit", {"file_path": "x", "old_string": "a"}) == ["work.change"]
def test_loading_a_procedure_reaches_its_own_moment():
assert _reached("Skill", {"skill": "scribe:writing-plans"}) == [
"skill.scribe:writing-plans",
]
def test_a_procedure_name_that_is_not_a_moment_reaches_nothing():
assert _reached("Skill", {"skill": "two words"}) == []
assert _reached("Skill", {}) == []
# ── an install's own ─────────────────────────────────────────────────────
def test_an_install_mapping_adds_a_moment():
rows = [_row("Bash", "make ship", "work.deliver")]
assert "work.deliver" in _reached("Bash", {"command": "make ship"}, rows)
assert "work.deliver" not in _reached("Bash", {"command": "make ship"})
def test_removing_a_default_removes_only_that_default():
"""`curl` stops reaching env.reach here; it is still a run, and `ssh`
still reaches out — a removal overrides nothing it did not name."""
rows = [_row("Bash", "curl", "env.reach", ma.REMOVE)]
assert _reached("Bash", {"command": "curl localhost:8000"}, rows) == ["work.run"]
assert "env.reach" in _reached("Bash", {"command": "ssh box"}, rows)
def test_a_removal_matches_the_default_whatever_the_tool_is_called():
rows = [_row("mcp__x__update_task", "status=done", "work.finish", ma.REMOVE)]
assert _reached("update_task", {"status": "done"}, rows) == []
def test_an_install_mapping_is_named_as_the_installs():
rows = [_row("Bash", "make ship", "work.deliver")]
hit = next(h for h in ma.resolve("Bash", {"command": "make ship"}, rows)
if h["moment"] == "work.deliver")
assert (hit["via"], hit["match"]) == (ma.INSTALL, "make ship")
# ── validation ───────────────────────────────────────────────────────────
def test_a_prefix_on_a_tool_that_runs_no_command_is_refused():
"""It would be stored and never match — a mapping that looks made and is
not, which is the misfire this table exists to fix."""
with pytest.raises(ValueError, match="field=value"):
ma._clean("update_task", "done", "work.finish")
def test_an_unknown_moment_is_refused():
with pytest.raises(ValueError, match="unknown moment"):
ma._clean("Bash", "make ship", "work.ship")
def test_a_mapping_is_normalised_before_it_is_stored():
assert ma._clean("mcp__x__update_task", " status=done ", " Work.Finish") == (
"update_task", "status=done", "work.finish",
)
assert ma._clean("Bash", "make ship", "work.deliver")[1] == "make ship"
def test_a_tool_is_required():
with pytest.raises(ValueError, match="tool is required"):
ma._clean(" ", "", "work.run")
# ── failing open ─────────────────────────────────────────────────────────
async def test_an_unreadable_mapping_table_costs_the_corrections_not_the_moments():
with patch.object(ma, "list_mappings", AsyncMock(side_effect=RuntimeError("db down"))):
hits = await ma.moments_for(7, "Bash", {"command": "git push"})
assert [h["moment"] for h in hits] == ["work.run", "work.deliver"]
+375
View File
@@ -0,0 +1,375 @@
"""Delivering mounted rules when their moment happens (milestone 458 step 4).
The database is stubbed: the lookup's scoping is pinned against Postgres in
tests/test_integration_rule_moments.py. These pin what is decided above it:
- the line names the moment AND the act that reached it, so a misfire is
visible where it lands and can be unmapped in the session;
- a rule the session was already shown is cited, not repeated, and only the
fresh ones are recorded, each with its moment;
- every door fails open;
- the plugin is told which tools can reach anything, and nothing more;
- every Scribe tool the shipped mappings name carries the attach, and the
hook and the route agree on every name between them.
"""
from __future__ import annotations
import ast
import inspect
import json
import os
import subprocess
from pathlib import Path
from types import SimpleNamespace
from unittest.mock import AsyncMock, MagicMock, patch
from quart import Quart, g
from scribe.services import moment_actions
from scribe.services import moment_delivery as md
from scribe.services import retrieval_pipeline as rp
from scribe.services import rule_usage, rulebooks
from tests.helpers import fake_rule, http_sink, need_tools
# Bound before conftest's autouse stub replaces the module attribute.
_REAL_DELIVER = md.deliver_for_act
ROOT = Path(__file__).resolve().parents[1]
HOOK = ROOT / "plugin" / "hooks" / "scribe_moment.sh"
def _rule(rid, title="Done means delivered and checked"):
return fake_rule(id=rid, title=title, kind="rule",
when_to_apply="about to say a piece of work is finished")
def _moment(**kw):
return rp.RuleMoment(user_id=1, query="", project_id=kw.pop("project_id", 2), **kw)
PUSH = [{"moment": "work.deliver", "tool": "Bash", "match": "git push", "via": "default"}]
# ── the arm ─────────────────────────────────────────────────────────────
async def test_a_line_names_the_moment_and_the_act_that_reached_it():
lookup = AsyncMock(return_value=[(_rule(5), "work.deliver")])
surfaced = MagicMock()
result = await rp.run_moment_arm(
PUSH, _moment(), io=rp.MomentIO(lookup=lookup, record_rule_surfaced=surfaced),
)
assert len(result.lines) == 1
assert "at work.deliver, reached by `git push`" in result.lines[0]
assert "get_rule(5)" in result.lines[0]
assert result.rule_ids == [5]
lookup.assert_awaited_once_with(1, ["work.deliver"], 2)
surfaced.assert_called_once()
kw = surfaced.call_args.kwargs
assert kw["source"] == rp.MOMENT_RULE_SOURCE
assert kw["rule_ids"] == [5]
assert kw["detail"] == {5: "work.deliver"}
async def test_a_rule_already_shown_is_cited_after_the_fresh_ones_and_not_recorded():
lookup = AsyncMock(return_value=[(_rule(5), "work.deliver"), (_rule(9, "Other"), "work.deliver")])
surfaced = MagicMock()
result = await rp.run_moment_arm(
PUSH, _moment(exclude=frozenset({5})),
io=rp.MomentIO(lookup=lookup, record_rule_surfaced=surfaced),
)
assert result.shown_rule_ids == [9, 5]
assert result.rule_ids == [9]
assert result.lines[1].startswith("Also")
assert surfaced.call_args.kwargs["rule_ids"] == [9]
async def test_an_unbound_session_asks_for_global_rules_only():
lookup = AsyncMock(return_value=[])
await rp.run_moment_arm(PUSH, _moment(project_id=0),
io=rp.MomentIO(lookup=lookup, record_rule_surfaced=MagicMock()))
assert lookup.await_args.args[2] is None
async def test_the_arm_fails_open():
broken = AsyncMock(side_effect=RuntimeError("db down"))
result = await rp.run_moment_arm(
PUSH, _moment(), io=rp.MomentIO(lookup=broken, record_rule_surfaced=MagicMock()),
)
assert result.lines == [] and result.rule_ids == []
# A recorder that fails costs the telemetry, never the lines.
result = await rp.run_moment_arm(
PUSH, _moment(),
io=rp.MomentIO(lookup=AsyncMock(return_value=[(_rule(5), "work.deliver")]),
record_rule_surfaced=MagicMock(side_effect=RuntimeError("x"))),
)
assert len(result.lines) == 1
async def test_nothing_reached_asks_nothing():
lookup = AsyncMock()
result = await rp.run_moment_arm([], _moment(),
io=rp.MomentIO(lookup=lookup, record_rule_surfaced=MagicMock()))
assert result.lines == []
lookup.assert_not_awaited()
# ── the act → the moments → the rules ───────────────────────────────────
def _stub_db(pairs):
async def _moments_for(user_id, tool, tool_input):
return moment_actions.resolve(tool, tool_input, [])
return [
patch.object(md, "deliver_for_act", _REAL_DELIVER),
patch.object(moment_actions, "moments_for", _moments_for),
patch.object(rulebooks, "rules_on_moments", AsyncMock(return_value=pairs)),
patch.object(rule_usage, "record_rule_surfaced", MagicMock()),
]
async def _with(stack, coro_fn):
for p in stack:
p.start()
try:
return await coro_fn()
finally:
for p in reversed(stack):
p.stop()
async def test_a_task_closed_through_the_tool_carries_its_mounted_rules():
data = {"id": 40, "status": "done", "project_id": 2}
out = await _with(_stub_db([(_rule(11), "work.finish")]), lambda: md.attach_moment_rules(
1, "update_task", {"status": "done", "project_id": 7}, data,
))
assert out is data
assert out["moment_rules"]["rule_ids"] == [11]
assert "at work.finish, reached by `status=done`" in out["moment_rules"]["lines"][0]
assert out["moment_rules"]["open_with"] == "get_rule(id)"
async def test_the_records_project_wins_over_the_callers_argument():
lookup = AsyncMock(return_value=[])
stack = _stub_db([])
stack[2] = patch.object(rulebooks, "rules_on_moments", lookup)
await _with(stack, lambda: md.attach_moment_rules(
1, "update_task", {"status": "done", "project_id": 7}, {"project_id": 2},
))
assert lookup.await_args.args == (1, ["work.finish"], 2)
async def test_an_act_that_reaches_nothing_mounted_adds_nothing():
data = {"id": 40}
out = await _with(_stub_db([]), lambda: md.attach_moment_rules(
1, "update_task", {"status": "todo"}, data,
))
assert "moment_rules" not in out
async def test_the_attach_fails_open_and_passes_other_payloads_through():
data = {"id": 1}
with patch.object(md, "deliver_for_act", AsyncMock(side_effect=RuntimeError("x"))):
assert await md.attach_moment_rules(1, "create_note", {}, data) == {"id": 1}
marker = object()
assert await md.attach_moment_rules(1, "start_planning", {}, marker) is marker
# ── which tools the plugin needs to ask about ───────────────────────────
async def _reachable(mounted, mappings=()):
with patch.object(rulebooks, "mounted_moments", AsyncMock(return_value=set(mounted))), \
patch.object(moment_actions, "list_mappings", AsyncMock(return_value=list(mappings))):
return await md.reachable_tools(1)
async def test_an_install_with_nothing_mounted_keeps_the_hook_off_the_wire():
listing = AsyncMock()
with patch.object(rulebooks, "mounted_moments", AsyncMock(return_value=set())), \
patch.object(moment_actions, "list_mappings", listing):
assert await md.reachable_tools(1) == []
listing.assert_not_awaited()
async def test_only_the_tools_whose_moments_carry_a_mount_are_listed():
assert await _reachable({"work.deliver"}) == ["bash"]
assert await _reachable({"work.finish"}) == ["update_milestone", "update_task"]
assert await _reachable({"skill.release"}) == ["skill"]
async def test_an_installs_own_mapping_and_removal_both_count():
own = SimpleNamespace(tool="mcp__deploy__ship", match="", moment="work.deliver", effect="add")
assert "ship" in await _reachable({"work.deliver"}, [own])
gone = SimpleNamespace(tool="AskUserQuestion", match="", moment="reply.ask", effect="remove")
assert await _reachable({"reply.ask"}, [gone]) == []
# ── the plugin routes ───────────────────────────────────────────────────
async def test_the_route_reads_the_event_and_the_ledger():
from scribe.routes import plugin as routes
deliver = AsyncMock(return_value=(PUSH, rp.RuleResult(
lines=["a line"], rule_ids=[5], shown_rule_ids=[5, 9],
)))
app = Quart(__name__)
event = {"session_id": "s", "tool_name": "Bash", "tool_input": {"command": "git push"}}
async with app.test_request_context(
"/api/plugin/moment", method="POST", json=event,
query_string={"project_id": "3", "exclude_rule_ids": "9", "held_rule_ids": "4"},
):
g.user = SimpleNamespace(id=7)
with patch.object(routes.moment_delivery_svc, "deliver_for_act", deliver):
resp = await routes.moment.__wrapped__()
body = await resp.get_json()
assert body == {"context": "a line", "rule_ids": [5], "moments": ["work.deliver"]}
args, kw = deliver.await_args
assert args == (7, "Bash", {"command": "git push"})
assert kw == {"project_id": 3, "exclude": frozenset({9}), "held": frozenset({4})}
async def test_the_route_answers_an_empty_event_with_nothing():
from scribe.routes import plugin as routes
app = Quart(__name__)
async with app.test_request_context("/api/plugin/moment", method="POST", json={}):
g.user = SimpleNamespace(id=7)
resp = await routes.moment.__wrapped__()
assert await resp.get_json() == {"context": "", "rule_ids": [], "moments": []}
def test_both_routes_are_on_the_app():
from scribe.app import create_app
rules = {str(r.rule) for r in create_app().url_map.iter_rules()}
assert {"/api/plugin/moment", "/api/plugin/moment-tools"} <= rules
# ── the hook ────────────────────────────────────────────────────────────
def test_the_hook_is_registered_on_every_tool():
manifest = json.loads((ROOT / "plugin" / "hooks" / "hooks.json").read_text())
entries = {m.get("matcher"): [h["command"] for h in m["hooks"]]
for m in manifest["hooks"]["PreToolUse"]}
assert any("scribe_moment.sh" in c for c in entries.get("*", []))
def test_the_hook_and_the_routes_agree_on_every_name():
"""Rule 33 across the shell/Python seam: a renamed arg fails silently."""
src = HOOK.read_text()
assert "/api/plugin/moment-tools" in src and "/api/plugin/moment" in src
assert "exclude_rule_ids" in src and "scribe_held_query" in src
assert "scribe_rules_live" in src and "scribe_rules_append" in src
# The SHARED ledger, not one of its own.
assert '"${TMPDIR:-/tmp}/scribe-priorart"' in src and ".rules.ids" in src
for field in (".tools", ".rule_ids", ".context"):
assert f"'{field}'" in src
from scribe.routes import plugin as routes
route = inspect.getsource(routes.moment)
for name in ("exclude_rule_ids", "held_rule_ids", "tool_name", "tool_input"):
assert name in route
def test_the_hook_leaves_scribes_own_tools_to_their_responses():
assert "mcp__*scribe*__*) exit 0" in HOOK.read_text()
def test_the_hook_never_returns_a_permission_decision():
"""A mount is a recall aid: it informs the act and never holds it."""
code = [ln for ln in HOOK.read_text().splitlines() if not ln.lstrip().startswith("#")]
assert not any("permissionDecision" in ln or "scribe_json_deny" in ln for ln in code)
def _hook(tmp_path, port, tool="Bash", command="git push origin dev", session="s-moment"):
need_tools("bash", "curl", "awk")
env = {"PATH": os.environ["PATH"], "SCRIBE_URL": f"http://127.0.0.1:{port}",
"SCRIBE_TOKEN": "t", "TMPDIR": str(tmp_path), "HOME": str(tmp_path)}
out = subprocess.run(
["bash", str(HOOK)],
input=json.dumps({"session_id": session, "cwd": str(tmp_path), "tool_name": tool,
"tool_input": {"command": command}}),
capture_output=True, text=True, env=env, timeout=30,
)
assert out.returncode == 0, out.stderr
return out.stdout
TOOLS = {"/api/plugin/moment-tools": b'{"tools":["bash"]}'}
def test_a_listed_tool_is_asked_about_and_its_rules_injected(tmp_path):
replies = dict(TOOLS)
replies["/api/plugin/moment"] = json.dumps(
{"context": "Standing rule that may apply at work.deliver", "rule_ids": [5],
"moments": ["work.deliver"]}).encode()
with http_sink(by_path=replies) as (port, seen):
out = _hook(tmp_path, port)
# The list is read once per window, not once per call.
_hook(tmp_path, port)
paths = [e["_path"] for e in seen]
assert paths == ["/api/plugin/moment-tools", "/api/plugin/moment", "/api/plugin/moment"]
sent = json.loads(seen[1]["_body"])
assert sent["tool_input"] == {"command": "git push origin dev"}
# The first call's fresh rule is on the shared ledger for the second.
assert seen[2]["exclude_rule_ids"] == ["5"]
ctx = json.loads(out)["hookSpecificOutput"]["additionalContext"]
assert "work.deliver" in ctx
def test_an_unlisted_tool_never_leaves_the_machine(tmp_path):
with http_sink(by_path=TOOLS) as (port, seen):
assert _hook(tmp_path, port, tool="Read") == ""
assert _hook(tmp_path, port, tool="mcp__plugin_scribe_scribe__update_task") == ""
assert [e["_path"] for e in seen] == ["/api/plugin/moment-tools"]
def test_an_install_with_nothing_mounted_sends_one_request_per_window(tmp_path):
with http_sink(by_path={"/api/plugin/moment-tools": b'{"tools":[]}'}) as (port, seen):
for _ in range(3):
assert _hook(tmp_path, port) == ""
assert len(seen) == 1
# ── the MCP door: every shipped Scribe action carries the attach ────────
def _attached_names(path: Path) -> dict[str, set[str]]:
"""{function name: {tool names passed to attach_moment_rules in it}}."""
out: dict[str, set[str]] = {}
for node in ast.parse(path.read_text()).body:
if not isinstance(node, ast.AsyncFunctionDef):
continue
for call in ast.walk(node):
func = call.func if isinstance(call, ast.Call) else None
name = (func.id if isinstance(func, ast.Name)
else func.attr if isinstance(func, ast.Attribute) else "")
if (name == "attach_moment_rules" and len(call.args) > 1
and isinstance(call.args[1], ast.Constant)):
out.setdefault(node.name, set()).add(call.args[1].value)
return out
def test_every_scribe_tool_the_defaults_name_attaches_its_moment_rules():
tools_dir = ROOT / "src" / "scribe" / "mcp" / "tools"
defined: dict[str, dict[str, set[str]]] = {}
scribe_tools: set[str] = set()
for path in tools_dir.glob("*.py"):
tree = ast.parse(path.read_text())
scribe_tools |= {n.name for n in tree.body if isinstance(n, ast.AsyncFunctionDef)}
defined[path.name] = _attached_names(path)
shipped = {a.tool for a in moment_actions.DEFAULT_ACTIONS} & scribe_tools
assert {"update_task", "create_rule", "update_milestone"} <= shipped
for tool in sorted(shipped):
named = set().union(*(per_file.get(tool, set()) for per_file in defined.values()))
assert tool in named, (
f"{tool} is mapped onto a moment by the shipped defaults but does not "
f"attach its mounted rules — a client without the plugin would never "
f"receive them"
)
+151
View File
@@ -0,0 +1,151 @@
"""The moment catalog — the vocabulary rules mount on (milestone 458 step 1).
What this pins is what a mount depends on: every name is well-formed and
resolvable, an unknown name is refused rather than stored, the catalog speaks
for any kind of work rather than only software, and both doors — the MCP tool
and the REST read — hand out the same catalog.
"""
import re
import pytest
from scribe.services import moments
from tests.helpers import FakeMCP
_NAME = re.compile(r"^[a-z]+\.[a-z]+$")
def test_the_catalog_is_not_empty():
"""The sweeps below are vacuous over an empty catalog (rule 167)."""
assert len(moments.MOMENTS) >= 10
def test_every_moment_is_keyed_by_its_own_well_formed_name():
for key, m in moments.MOMENTS.items():
assert key == m.name, f"{key!r} is filed under a name it does not carry"
assert _NAME.match(key), f"{key!r} is not <area>.<verb>"
def test_every_moment_says_what_it_means_and_how_it_is_reached():
for m in moments.MOMENTS.values():
assert m.means.strip() and m.reached_by.strip(), m.name
assert "\n" not in m.means, f"{m.name}: `means` is one line"
def test_the_skill_family_is_not_a_catalog_entry():
"""A family is a shape, not a moment — no rule mounts on `skill.<name>`
literally, and a catalog key starting with the prefix would shadow it."""
assert not any(k.startswith(moments.SKILL_PREFIX) for k in moments.MOMENTS)
# Software-only vocabulary, the same guard the completion query carries. The
# moments are named for any work a person drives through an agent; the actions
# particular to one kind of work belong in the mappings, not in the meaning.
_DEV_ONLY = (r"\bCI\b", r"\bcommit", r"\bpull request", r"\bcode\b", r"\btest",
r"\bgit\b", r"\brepo(s|sitor\w*)?\b", r"\bbranch", r"\bcompil", r"\bbuild\b")
@pytest.mark.parametrize("field", ["means", "reached_by"])
def test_the_catalog_assumes_no_particular_domain(field):
found = {
m.name: hits
for m in [*moments.MOMENTS.values(), moments.SKILL_FAMILY]
if (hits := [w for w in _DEV_ONLY
if re.search(w, getattr(m, field), re.IGNORECASE)])
}
assert not found, f"software-only vocabulary in `{field}`: {found}"
def test_the_guard_can_fail():
assert re.search(_DEV_ONLY[1], "after the commit lands", re.IGNORECASE)
@pytest.mark.parametrize("name", list(moments.MOMENTS))
def test_every_catalog_moment_is_mountable(name):
assert moments.require_moment(name) == name
@pytest.mark.parametrize("raw,clean", [
(" Work.Finish ", "work.finish"),
("skill.verification", "skill.verification"),
("skill.scribe:writing-plans", "skill.scribe:writing-plans"),
("skill.scribe-proc-release_notes", "skill.scribe-proc-release_notes"),
])
def test_a_name_is_normalised_before_it_is_judged(raw, clean):
assert moments.require_moment(raw) == clean
@pytest.mark.parametrize("bad", [
"work.finished", "finish", "", "skill.", "skill.two words", "work.",
])
def test_an_unknown_moment_is_refused_with_the_catalog(bad):
"""A typo stored as a mount never fires — the silent failure moments end."""
with pytest.raises(ValueError) as exc:
moments.require_moment(bad)
assert "work.finish" in str(exc.value)
assert moments.SKILL_FAMILY.name in str(exc.value)
def test_the_catalog_payload_carries_every_moment_and_the_family():
data = moments.catalog()
assert [m["name"] for m in data["moments"]] == list(moments.MOMENTS)
assert data["total"] == len(moments.MOMENTS)
assert data["families"][0]["prefix"] == moments.SKILL_PREFIX
async def test_the_tool_returns_the_catalog_with_this_installs_actions():
from unittest.mock import AsyncMock, patch
from scribe.mcp.tools import moments as tool
by_moment = {"actions": {"work.change": []}, "removed_defaults": []}
with patch.object(tool, "current_user_id", lambda: 7), \
patch.object(tool.actions_svc, "actions_by_moment",
AsyncMock(return_value=by_moment)) as read:
out = await tool.list_moments()
read.assert_awaited_once_with(7)
assert out["moments"] == moments.catalog()["moments"]
assert out["actions"] == by_moment["actions"]
assert out["removed_defaults"] == []
def test_the_tools_are_registered_and_classified():
from scribe.mcp.server import _READ_ONLY_TOOLS, _WRITE_TOOLS
from scribe.mcp.tools import moments as tool
mcp = FakeMCP()
tool.register(mcp)
assert mcp.names == ["list_moments", "map_action", "unmap_action"]
assert "list_moments" in _READ_ONLY_TOOLS
assert {"map_action", "unmap_action"} <= _WRITE_TOOLS
def test_both_doors_read_one_catalog():
"""Rule 33 parity: the tool and the routes call the same services, so the
session and the Settings view cannot name different moments or disagree
about what a mapping does."""
from scribe.mcp.tools import moments as tool
from scribe.routes import retrieval as routes
from scribe.services import moment_actions
assert tool.moments_svc is moments
assert routes.moments_svc is moments
assert tool.actions_svc is moment_actions
assert routes.moment_actions_svc is moment_actions
def test_a_rules_moments_are_normalised_and_deduplicated():
assert moments.require_moments([" Work.Finish", "reply.report", "work.finish"]) == [
"work.finish", "reply.report",
]
assert moments.require_moments("work.finish") == ["work.finish"]
assert moments.require_moments([]) == []
def test_absent_moments_mean_leave_them_alone():
assert moments.require_moments(None) is None
def test_one_unknown_moment_refuses_the_whole_set():
with pytest.raises(ValueError, match="work.finished"):
moments.require_moments(["work.finish", "work.finished"])
+276
View File
@@ -0,0 +1,276 @@
"""The reply moment: the finished reply, checked before it goes out (milestone
458 step 4, folded in from milestone 456 step 8).
The operator's ruling: an unopened rule above the stop bar HOLDS the reply
once, in the server's words — the backstop for whatever the earlier arms
missed — and the rewrite is never held. These pin, with the database stubbed:
- which moments a reply reaches;
- what holds: a mounted RULE the session has not opened, or the reply's top
semantic hit above the reply surface's bar — never a preference, never a
rule already opened or already held once, never past the session cap;
- that the stop-only arm records surfacing for the held rule alone;
- the hook's contract: blocks only on the server's reason, records the hold
before emitting it, and never holds the rewrite.
"""
from __future__ import annotations
import json
import os
import subprocess
from pathlib import Path
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
from quart import Quart, g
from scribe.services import moment_delivery as md
from scribe.services import plugin_context as pc
from scribe.services import retrieval_pipeline as rp
from scribe.services import retrieval_surfaces as surfaces
from scribe.services import rule_usage, rulebooks
from tests.helpers import fake_rule, http_sink, need_tools
ROOT = Path(__file__).resolve().parents[1]
HOOK = ROOT / "plugin" / "hooks" / "scribe_reply_check.sh"
DONE = fake_rule(id=11, title="Definition of done", kind="rule",
when_to_apply="about to tell the operator work is finished")
# ── which moments a reply reaches ───────────────────────────────────────
def test_every_reply_reports_and_a_question_also_asks():
assert [h["moment"] for h in md.reply_moments("Shipped it.")] == ["reply.report"]
assert [h["moment"] for h in md.reply_moments("Shipped.\nShall I merge it?")] == [
"reply.report", "reply.ask",
]
def test_a_question_inside_code_is_not_the_reply_asking():
reply = "Done.\n```python\nok = x if y else z?\n```\n"
assert [h["moment"] for h in md.reply_moments(reply)] == ["reply.report"]
def test_a_long_reply_keeps_its_end():
"""The part of a report that asks something of the reader is at its end."""
reply = "head " * 400 + "please confirm it works on your end"
query = md._reply_query(reply)
assert query.startswith("head")
assert query.endswith("please confirm it works on your end")
assert len(query) < len(reply)
# ── what holds ──────────────────────────────────────────────────────────
def _stubs(*, mounted=(), checkpoint=None):
arm = AsyncMock(return_value=rp.RuleResult(checkpoint=checkpoint or {}))
surfaced = MagicMock()
lookup = AsyncMock(return_value=list(mounted))
return arm, surfaced, lookup, [
patch.object(rulebooks, "rules_on_moments", lookup),
patch.object(rule_usage, "record_rule_surfaced", surfaced),
patch.object(rp, "run_rule_arm", arm),
patch.object(surfaces, "floor_for", AsyncMock(return_value=0.8)),
patch.object(surfaces, "budget_for", AsyncMock(return_value=1)),
patch.object(pc, "_rule_io", MagicMock()),
]
async def _hold(stack, reply="All done — let me know if it works.", **kw):
for p in stack:
p.start()
try:
return await md.reply_hold(1, reply, project_id=2, **kw)
finally:
for p in reversed(stack):
p.stop()
async def test_a_mounted_rule_the_session_has_not_opened_holds_the_reply():
_arm, surfaced, lookup, stack = _stubs(mounted=[(DONE, "reply.report")])
out = await _hold(stack)
assert out["rule_ids"] == [11]
assert "Definition of done" in out["reason"] and "get_rule(11)" in out["reason"]
assert "mounted on reply.report" in out["reason"]
assert lookup.await_args.args == (1, ["reply.report"], 2)
kw = surfaced.call_args.kwargs
assert kw["source"] == rp.MOMENT_RULE_SOURCE and kw["detail"] == {11: "reply.report"}
async def test_an_opened_rule_a_preference_and_an_earlier_hold_do_not_hold():
pref = fake_rule(id=12, title="Lead with the outcome", kind="preference")
other = fake_rule(id=13, title="Name the commit", kind="rule")
mounted = [(DONE, "reply.report"), (pref, "reply.report"), (other, "reply.report")]
_arm, surfaced, _lookup, stack = _stubs(mounted=mounted)
assert await _hold(stack, held=frozenset({11}), stopped=frozenset({13})) == {}
surfaced.assert_not_called()
async def test_the_semantic_backstop_holds_and_never_names_a_mounted_rule_twice():
cp = {"rule_id": 40, "title": "Read the job log first", "score": 0.84,
"trigger": "a CI run overran"}
arm, _surfaced, _lookup, stack = _stubs(mounted=[(DONE, "reply.report")], checkpoint=cp)
out = await _hold(stack)
assert out["rule_ids"] == [11, 40]
assert "scores 0.84 against this reply" in out["reason"]
spec, moment = arm.await_args.args
assert spec is rp.REPLY_RULE
assert 11 in moment.held, "a rule the mounted half holds must not be raised again"
# Its floor IS its stop bar.
assert arm.await_args.kwargs["floor"] == arm.await_args.kwargs["checkpoint_floor"] == 0.8
async def test_past_the_session_cap_nothing_holds_and_nothing_is_asked():
arm, _surfaced, lookup, stack = _stubs(mounted=[(DONE, "reply.report")])
stopped = frozenset(range(100, 100 + pc.CHECKPOINT_SESSION_CAP))
assert await _hold(stack, stopped=stopped) == {}
lookup.assert_not_awaited()
arm.assert_not_awaited()
async def test_the_cap_trims_before_anything_is_recorded():
rules = [(fake_rule(id=i, title=f"rule {i}", kind="rule"), "reply.report") for i in range(1, 9)]
_arm, surfaced, _lookup, stack = _stubs(mounted=rules)
out = await _hold(stack, stopped=frozenset({50, 51}))
room = pc.CHECKPOINT_SESSION_CAP - 2
assert len(out["rule_ids"]) == room
assert surfaced.call_args.kwargs["rule_ids"] == out["rule_ids"]
async def test_the_reply_check_fails_open():
_arm, _surfaced, _lookup, stack = _stubs()
stack[0] = patch.object(rulebooks, "rules_on_moments", AsyncMock(side_effect=RuntimeError("x")))
assert await _hold(stack) == {}
assert await md.reply_hold(1, " ") == {}
# ── the stop-only arm ───────────────────────────────────────────────────
async def test_the_stop_only_arm_records_only_the_rule_that_holds():
hits = [(0.86, fake_rule(id=40, title="a", kind="rule", when_to_apply="t")),
(0.82, fake_rule(id=41, title="b", kind="rule", when_to_apply="t"))]
io = rp.RuleIO(search=AsyncMock(return_value=hits), record_retrieval=MagicMock(),
record_rule_surfaced=MagicMock())
moment = rp.RuleMoment(user_id=1, query="the reply", project_id=2,
checkpoint_where="this reply")
result = await rp.run_rule_arm(rp.REPLY_RULE, moment, floor=0.8, budget=3, io=io,
checkpoint_floor=0.8)
assert result.checkpoint["rule_id"] == 40
assert io.record_rule_surfaced.call_args.kwargs["rule_ids"] == [40]
# The top hit already opened: nothing holds, so nothing was shown.
io.record_rule_surfaced.reset_mock()
held = rp.RuleMoment(user_id=1, query="the reply", project_id=2, held=frozenset({40}))
result = await rp.run_rule_arm(rp.REPLY_RULE, held, floor=0.8, budget=3, io=io,
checkpoint_floor=0.8)
assert result.checkpoint == {}
io.record_rule_surfaced.assert_not_called()
def test_the_reply_arm_is_a_registered_ranked_surface():
from scribe.services.retrieval_registry import POINTS
assert rp.REPLY_RULE in rp.RULE_ARMS
assert "reply_rule" in surfaces.SURFACES and "reply_rule" in POINTS
assert "reply_rule" in rule_usage.RANKED_SOURCES
# A stop bar, not a hint bar: it sits with the checkpoint's default.
assert surfaces.SURFACES["reply_rule"].floor_default >= pc._CHECKPOINT_DEFAULT
# ── the route ───────────────────────────────────────────────────────────
async def test_the_route_passes_the_reply_and_all_three_ledgers():
from scribe.routes import plugin as routes
hold = AsyncMock(return_value={"reason": "Held.", "rule_ids": [11], "moments": ["reply.report"]})
app = Quart(__name__)
async with app.test_request_context(
"/api/plugin/reply-rules", method="POST", json={"reply": "done"},
query_string={"project_id": "3", "held_rule_ids": "4", "exclude_rule_ids": "5",
"stopped_rule_ids": "6,7"},
):
g.user = type("U", (), {"id": 7})()
with patch.object(routes.moment_delivery_svc, "reply_hold", hold):
resp = await routes.reply_rules.__wrapped__()
body = await resp.get_json()
assert body == {"reason": "Held.", "rule_ids": [11], "moments": ["reply.report"]}
assert hold.await_args.args == (7, "done")
assert hold.await_args.kwargs == {
"project_id": 3, "exclude": frozenset({5}), "held": frozenset({4}),
"stopped": frozenset({6, 7}),
}
# ── the hook ────────────────────────────────────────────────────────────
def _transcript(tmp_path, reply):
lines = [
{"type": "user", "message": {"role": "user", "content": "is it finished?"}},
{"type": "assistant", "message": {"content": [{"type": "text", "text": reply}]}},
]
path = tmp_path / "t.jsonl"
path.write_text("\n".join(json.dumps(x, separators=(",", ":")) for x in lines) + "\n")
return path
def _run(tmp_path, port, transcript, active=False):
need_tools("bash", "curl", "awk")
env = {"PATH": os.environ["PATH"], "SCRIBE_URL": f"http://127.0.0.1:{port}",
"SCRIBE_TOKEN": "t", "TMPDIR": str(tmp_path), "HOME": str(tmp_path)}
out = subprocess.run(
["bash", str(HOOK)],
input=json.dumps({"session_id": "s-reply", "transcript_path": str(transcript),
"cwd": str(tmp_path), "hook_event_name": "Stop",
"stop_hook_active": active}),
capture_output=True, text=True, env=env, timeout=30,
)
assert out.returncode == 0, out.stderr
return out.stdout.strip()
HELD = json.dumps({"reason": "Held for one read: get_rule(11).", "rule_ids": [11],
"moments": ["reply.report"]}).encode()
QUIET = b'{"reason":"","rule_ids":[],"moments":["reply.report"]}'
def test_the_hook_blocks_in_the_servers_words_and_records_the_hold(tmp_path):
t = _transcript(tmp_path, "All done.\nLet me know if it works on your end?")
with http_sink(by_path={"/api/plugin/reply-rules": HELD}) as (port, seen):
out = json.loads(_run(tmp_path, port, t))
_run(tmp_path, port, t)
assert out == {"decision": "block", "reason": "Held for one read: get_rule(11)."}
sent = json.loads(seen[0]["_body"])
assert sent["reply"] == "All done.\nLet me know if it works on your end?"
# The second turn tells the server what already held.
assert seen[1]["stopped_rule_ids"] == ["11"]
assert seen[1]["exclude_rule_ids"] == ["11"]
def test_the_hook_says_nothing_when_nothing_holds(tmp_path):
t = _transcript(tmp_path, "Done.")
with http_sink(by_path={"/api/plugin/reply-rules": QUIET}) as (port, _seen):
assert _run(tmp_path, port, t) == ""
def test_the_rewrite_is_never_held(tmp_path):
t = _transcript(tmp_path, "Done.")
with http_sink(by_path={"/api/plugin/reply-rules": HELD}) as (port, seen):
assert _run(tmp_path, port, t, active=True) == ""
assert seen == []
def test_an_unreachable_instance_never_holds_a_reply(tmp_path):
t = _transcript(tmp_path, "Done.")
assert _run(tmp_path, 9, t) == ""
@pytest.mark.parametrize("name", ["scribe_reply_check.sh"])
def test_the_hook_is_registered_on_stop(name):
manifest = json.loads((ROOT / "plugin" / "hooks" / "hooks.json").read_text())
commands = [h["command"] for m in manifest["hooks"]["Stop"] for h in m["hooks"]]
assert any(name in c for c in commands)
+9 -3
View File
@@ -2,7 +2,7 @@
WHY AST AND NOT GREP, demonstrated rather than asserted. Two of the sources in WHY AST AND NOT GREP, demonstrated rather than asserted. Two of the sources in
this system reach their recorder as `source=SOURCE` through a module-level this system reach their recorder as `source=SOURCE` through a module-level
constant — `wide_net` and `report_preference` — so a grep for `source="` finds constant — `wide_net` and `preference_slot` — so a grep for `source="` finds
neither. A registry test built on that grep would pass while being blind to neither. A registry test built on that grep would pass while being blind to
two arms, which is the narrowing #3191 warns about: a check that looks two arms, which is the narrowing #3191 warns about: a check that looks
thorough and quietly covers less than it claims. thorough and quietly covers less than it claims.
@@ -101,9 +101,15 @@ def test_the_extractor_finds_something() -> None:
def test_the_extractor_resolves_the_constant_sources() -> None: def test_the_extractor_resolves_the_constant_sources() -> None:
"""The specific capability a grep would lose. See the module docstring.""" """The specific capability a grep would lose. See the module docstring.
`report_preference` was the second example until milestone 456 moved it
into the retrieval pipeline, where its source is a spec field; the
pipeline's reserved slot still records through a module constant, so it
carries the case instead.
"""
found, _ = call_sites() found, _ = call_sites()
for via_constant in ("wide_net", "report_preference"): for via_constant in ("wide_net", "preference_slot"):
assert via_constant in found, ( assert via_constant in found, (
f"{via_constant} reaches its recorder through a module constant; " f"{via_constant} reaches its recorder through a module constant; "
f"an extractor that cannot resolve one is blind to it" f"an extractor that cannot resolve one is blind to it"
+8 -1
View File
@@ -67,9 +67,16 @@ def test_every_surface_name_is_a_real_telemetry_source():
# a literal, so that one name is satisfied by the constant holding it. # a literal, so that one name is satisfied by the constant holding it.
from scribe.services.reply_preferences import SOURCE from scribe.services.reply_preferences import SOURCE
# The rule arms record through the one pipeline (milestone 456), where
# `source` is the spec's own field — so for them the spec IS the string
# `record_retrieval` receives, and the join key is checked against it.
from scribe.services.retrieval_pipeline import RULE_ARMS
via_pipeline = {arm.source for arm in RULE_ARMS}
missing = [ missing = [
s.name for s in rs.SURFACES.values() s.name for s in rs.SURFACES.values()
if f'source="{s.name}"' not in blob and s.name != SOURCE if f'source="{s.name}"' not in blob
and s.name != SOURCE and s.name not in via_pipeline
] ]
assert not missing, ( assert not missing, (
f"these surfaces can be tuned but never measured: {missing}. The " f"these surfaces can be tuned but never measured: {missing}. The "
+2
View File
@@ -42,6 +42,8 @@ def test_every_endpoint_is_reachable_on_the_app():
"/api/retrieval/surfaces", "/api/retrieval/surfaces",
"/api/retrieval/surfaces/<surface>", "/api/retrieval/surfaces/<surface>",
"/api/retrieval/tuning-history", "/api/retrieval/tuning-history",
"/api/retrieval/moments",
"/api/retrieval/moments/mappings",
} }
+132
View File
@@ -0,0 +1,132 @@
"""Every door that writes a rule can mount it on moments (milestone 458 step 3).
THE SHAPE, NOT ONE FUNCTION. test_system_tagging_door_parity records how a
capability goes missing: whichever door nobody exercised for a kind never
grows the parameter. So this file asserts the whole table — every rule and
preference write, on both doors, takes `moments` — and that each one checks
the names BEFORE its create or update. A refusal after the write would leave
a rule created without the mounts it asked for, and the caller reading the
error would reasonably believe nothing happened.
"""
from __future__ import annotations
import ast
import pathlib
from unittest.mock import AsyncMock, patch
import pytest
from tests.helpers import fake_rule
from tests.helpers import plain_rule_detail as _plain_detail
ROOT = pathlib.Path(__file__).resolve().parents[1] / "src" / "scribe"
# (tool, the service write it must validate before)
MCP_DOORS = [
("create_rule", "create_rule"),
("create_project_rule", "create_project_rule"),
("update_rule", "update_rule"),
("create_preference", "create_rule"),
("update_preference", "update_rule"),
]
# (route handler, the service write it must validate before)
REST_DOORS = [
("create_rule", "create_rule"),
("update_rule", "update_rule"),
("create_project_rule", "create_project_rule"),
]
def _fn(path: pathlib.Path, name: str):
for node in ast.parse(path.read_text()).body:
if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)) and node.name == name:
return node
raise AssertionError(f"{path.name} has no {name}")
def _first_line_calling(fn, attr: str) -> int | None:
lines = [
n.lineno for n in ast.walk(fn)
if isinstance(n, ast.Call) and (
(isinstance(n.func, ast.Attribute) and n.func.attr == attr)
or (isinstance(n.func, ast.Name) and n.func.id == attr)
)
]
return min(lines) if lines else None
@pytest.mark.parametrize("tool,write", MCP_DOORS)
def test_every_mcp_rule_door_takes_moments_and_checks_them_first(tool, write):
fn = _fn(ROOT / "mcp" / "tools" / "rulebooks.py", tool)
params = {a.arg for a in fn.args.args + fn.args.kwonlyargs}
assert "moments" in params, f"{tool} cannot mount a rule"
check = _first_line_calling(fn, "require_moments")
wrote = _first_line_calling(fn, write)
assert check is not None and wrote is not None, (tool, check, wrote)
assert check < wrote, f"{tool} validates moments after it has already written"
detail = [n for n in ast.walk(fn) if isinstance(n, ast.Call)
and isinstance(n.func, ast.Attribute) and n.func.attr == "rule_detail"]
assert any(any(isinstance(a, ast.Name) and a.id == "moments" for a in c.args)
for c in detail), f"{tool} never hands its moments to rule_detail"
@pytest.mark.parametrize("handler,write", REST_DOORS)
def test_every_rest_rule_door_takes_moments_and_checks_them_first(handler, write):
fn = _fn(ROOT / "routes" / "rulebooks.py", handler)
check = _first_line_calling(fn, "_moments_or_refusal")
wrote = _first_line_calling(fn, write)
assert check is not None and wrote is not None, (handler, check, wrote)
assert check < wrote, f"{handler} validates moments after it has already written"
def test_the_guard_can_fail():
"""A door with no check at all must be reported, not skipped (rule 167)."""
fn = ast.parse("async def f():\n await svc.create_rule()\n").body[0]
assert _first_line_calling(fn, "require_moments") is None
async def test_an_unknown_moment_is_refused_before_anything_is_written():
create = AsyncMock(return_value=fake_rule(id=5))
with patch("scribe.mcp.tools.rulebooks.rulebooks_svc.create_rule", create), \
patch("scribe.mcp.tools.rulebooks.dedup_svc.find_duplicate_rule", AsyncMock()), \
_plain_detail(), patch("scribe.mcp.tools.rulebooks.current_user_id", lambda: 7):
from scribe.mcp.tools.rulebooks import create_rule
with pytest.raises(ValueError, match="unknown moment"):
await create_rule(
topic_id=10, title="t", statement="s",
when_to_apply="when the moment this fixture stands in for arises",
moments=["work.finished"],
)
create.assert_not_awaited()
async def test_the_door_hands_rule_detail_the_normalised_moments():
seen = {}
async def detail(_uid, rule, _system_ids=None, moments=None):
seen["moments"] = moments
return rule.to_dict()
with patch("scribe.mcp.tools.rulebooks.rulebooks_svc.update_rule",
AsyncMock(return_value=fake_rule(id=5))), \
patch("scribe.mcp.tools.rulebooks.rulebooks_svc.rule_detail", detail), \
patch("scribe.mcp.tools.rulebooks.current_user_id", lambda: 7):
from scribe.mcp.tools.rulebooks import update_rule
await update_rule(rule_id=5, moments=[" Work.Finish", "reply.report", "work.finish"])
assert seen["moments"] == ["work.finish", "reply.report"]
async def test_omitting_moments_leaves_the_mounts_alone():
seen = {}
async def detail(_uid, rule, _system_ids=None, moments="unset"):
seen["moments"] = moments
return rule.to_dict()
with patch("scribe.mcp.tools.rulebooks.rulebooks_svc.update_rule",
AsyncMock(return_value=fake_rule(id=5))), \
patch("scribe.mcp.tools.rulebooks.rulebooks_svc.rule_detail", detail), \
patch("scribe.mcp.tools.rulebooks.current_user_id", lambda: 7):
from scribe.mcp.tools.rulebooks import update_rule
await update_rule(rule_id=5, title="renamed")
assert seen["moments"] is None
+88 -64
View File
@@ -887,78 +887,74 @@ async def test_a_command_the_arm_never_searched_writes_no_row_at_all():
log.assert_not_called() log.assert_not_called()
def test_neither_rule_arm_logs_its_call_behind_a_results_guard(): def test_no_rule_arm_logs_its_call_behind_a_results_guard():
"""Structural, on top of the behavioural pair above, because the defect was """Structural, on top of the behavioural tests above, because the defect
one level of indentation and it appeared INDEPENDENTLY in two places — the was one level of indentation and it appeared INDEPENDENTLY in two arms —
pre-tool arm inherited it by being modelled on its sibling. The third arm the pre-tool arm inherited it by being modelled on its sibling (#3497).
modelled on either of them is the one this catches.
Since milestone 456 there is one implementation to guard rather than a
copy per arm, so the property is checked where it is written: the
pipeline's ranked-search stage writes the call row unconditionally, the
arm runner calls that stage before it can return, and the surfacing row
alone stays behind `if rule_ids:`. And no arm writes a rule-source row of
its own any more — a fourth copy would be the defect's way back in.
""" """
pc_src = Path("src/scribe/services/plugin_context.py").read_text() from scribe.services import retrieval_pipeline as rp
# Write-path arm: what remains inside `if fresh:` is the SURFACING log only. src = Path("src/scribe/services/retrieval_pipeline.py").read_text()
guarded = pc_src.split('source="write_path_rule", query=code or path')[1] tree = ast.parse(src)
guarded = guarded.split("if fresh:")[1].split("except Exception:")[0] fns = {n.name: n for n in ast.walk(tree) if isinstance(n, ast.AsyncFunctionDef)}
assert "record_rule_surfaced" in guarded, "the surfacing log must stay guarded"
assert "record_retrieval" not in guarded, (
"the call log is back inside the results guard — a call that found "
"nothing is the only evidence a threshold is set too high"
)
# Pre-tool arm: the call log comes BEFORE the early return. def _calls(fn, name):
# return [
# WALKED, NOT SUBSTRING-MATCHED (rule 167). This assertion used to read
# `body.index("if not fresh:")`, which pinned the name of a local variable
# rather than the property. #3750 changed that guard to `if not hits:` —
# the arm still logs before returning, so the property held perfectly, and
# a name-matching assertion would have raised ValueError and reported the
# #3497 defect as back. A guard that cries regression when the thing it
# protects is intact is the failure mode rule 167 names.
#
# The property is positional: between the search and the first guard that
# can return early, the call row has already been written.
# The arm's BODY, which since #4633 lives in `_tool_rule_hint`; the public
# `build_tool_rule_hint` is a wrapper that adds the via-lesson step and
# searches nothing itself. Located by what it does — the function that
# calls the rule search under the pre_tool_rule source — so the next
# rename moves the guard with it instead of emptying it.
fn = next(
n for n in ast.walk(ast.parse(pc_src))
if isinstance(n, ast.AsyncFunctionDef)
and any(
isinstance(c, ast.Call) and getattr(c.func, "id", None) == "semantic_search_rules"
for c in ast.walk(n)
)
and any(
isinstance(k, ast.keyword) and k.arg == "source"
and isinstance(k.value, ast.Constant) and k.value.value == "pre_tool_rule"
for k in ast.walk(n)
)
)
search_at = min(
n.lineno for n in ast.walk(fn) n.lineno for n in ast.walk(fn)
if isinstance(n, ast.Call) if isinstance(n, ast.Call)
and getattr(n.func, "id", None) == "semantic_search_rules" and (getattr(n.func, "attr", None) == name or getattr(n.func, "id", None) == name)
]
# 1. The stage logs, and nothing in it can return before it does.
ranked = fns["_ranked"]
logged_at = _calls(ranked, "record_retrieval")
assert logged_at, "the ranked-search stage no longer writes the call row"
early = [n.lineno for n in ast.walk(ranked)
if isinstance(n, ast.Return) and n.lineno < min(logged_at)]
assert not early, (
f"the ranked-search stage can return at line {early[0]} before logging "
f"its call — a call that found nothing is the only evidence a bar is "
f"too high (#3497)"
) )
logged_at = min(
n.lineno for n in ast.walk(fn) # 2. The runner reaches the stage before its first early return.
if isinstance(n, ast.Call) run = fns["run_rule_arm"]
and getattr(n.func, "id", None) == "record_retrieval" staged_at = min(_calls(run, "_ranked"))
)
# Every `if <cond>: return ...` after the search — whatever it tests.
bailouts = [ bailouts = [
n.lineno for n in ast.walk(fn) n.lineno for n in ast.walk(run)
if isinstance(n, ast.If) and n.lineno > search_at if isinstance(n, ast.If) and any(isinstance(b, ast.Return) for b in n.body)
and any(isinstance(b, ast.Return) for b in n.body)
] ]
assert bailouts, ( assert bailouts, (
"no early return found after the search in the pre-tool arm — the " "no early return found in run_rule_arm — this check has nothing left "
"guard has nothing left to protect, which means this test is now " "to protect and would pass vacuously"
"passing vacuously rather than the arm being correct"
) )
assert logged_at < min(bailouts), ( assert staged_at < min(bailouts), (
f"the pre-tool arm returns at line {min(bailouts)} before logging its " f"run_rule_arm returns at line {min(bailouts)} before the call row is "
f"call at line {logged_at} — a surface with no rows at all cannot be " f"written at line {staged_at}"
f"told apart from a hook that never fired (#3497)" )
# 3. The surfacing row is the guarded one.
guards = [n for n in ast.walk(run) if isinstance(n, ast.If)
and isinstance(n.test, ast.Name) and n.test.id == "rule_ids"]
assert guards and any(_calls(g, "record_rule_surfaced") for g in guards), (
"the surfacing row must stay behind `if rule_ids:` — nothing was shown, "
"so no surfacing occurred"
)
# 4. No arm keeps a private copy: outside the pipeline, nothing writes a
# call row under a rule arm's source.
pc_src = Path("src/scribe/services/plugin_context.py").read_text()
for arm in rp.RULE_ARMS:
assert f'source="{arm.source}"' not in pc_src, (
f"plugin_context writes a {arm.source} row of its own again — the "
f"arm is meant to be a spec over the one pipeline"
) )
@@ -1702,8 +1698,11 @@ def test_every_hook_rule_search_says_which_project_it_is_for():
`everywhere` is not an acceptable answer in a hook, which speaks unasked. `everywhere` is not an acceptable answer in a hook, which speaks unasked.
""" """
sources = { sources = {
"src/scribe/services/plugin_context.py": 4, # Since milestone 456 the hook arms search through the pipeline's one
"src/scribe/services/reply_preferences.py": 1, # ranked-search stage, checked below — so a direct search appearing
# here again is a copy of the arm coming back, and fails the count.
"src/scribe/services/plugin_context.py": 0,
"src/scribe/services/reply_preferences.py": 0,
} }
for path, expected in sources.items(): for path, expected in sources.items():
calls = [ calls = [
@@ -1728,6 +1727,31 @@ def test_every_hook_rule_search_says_which_project_it_is_for():
f"{path}:{call.lineno} searches every project's rules from a hook" f"{path}:{call.lineno} searches every project's rules from a hook"
) )
# The pipeline: exactly one search, in `_ranked`, whose keyword set is
# built as a dict literal — so the keys are read from that literal.
pipeline = ast.parse(Path("src/scribe/services/retrieval_pipeline.py").read_text())
searches = [
n for n in ast.walk(pipeline)
if isinstance(n, ast.Call) and getattr(n.func, "attr", None) == "search"
]
assert len(searches) == 1, (
f"the pipeline has {len(searches)} rule searches, expected 1 — every "
f"arm is meant to reach the ranker through `_ranked`"
)
ranked = next(n for n in ast.walk(pipeline)
if isinstance(n, ast.AsyncFunctionDef) and n.name == "_ranked")
keys = {
k.value for d in ast.walk(ranked) if isinstance(d, ast.Dict)
for k in d.keys if isinstance(k, ast.Constant)
}
assert "project_id" in keys, (
"the pipeline's rule search no longer passes project_id, so every hook "
"arm gets global rules only and never its own project's"
)
assert "everywhere" not in keys and not any(
isinstance(n, ast.keyword) and n.arg == "everywhere" for n in ast.walk(ranked)
), "the pipeline searches every project's rules from a hook"
@pytest.mark.asyncio @pytest.mark.asyncio
@pytest.mark.parametrize("bound, scope", [(7, 7), (0, None)]) @pytest.mark.parametrize("bound, scope", [(7, 7), (0, None)])
+5 -2
View File
@@ -27,7 +27,7 @@ def test_backup_version_is_current():
(Named for the number it asserted until v10, which is exactly the drift a (Named for the number it asserted until v10, which is exactly the drift a
name-carrying-a-value invites; it now says what it checks.)""" name-carrying-a-value invites; it now says what it checks.)"""
assert backup.BACKUP_VERSION == 20 assert backup.BACKUP_VERSION == 22
def _exportable_note(**over): def _exportable_note(**over):
@@ -141,6 +141,7 @@ def _column_guard_targets():
from scribe.models.rule_usage import RuleUsageEvent from scribe.models.rule_usage import RuleUsageEvent
from scribe.models.system_usage import SystemUsageEvent from scribe.models.system_usage import SystemUsageEvent
from scribe.models.retrieval_tuning import RetrievalTuningEvent from scribe.models.retrieval_tuning import RetrievalTuningEvent
from scribe.models.moment_mapping import MomentMapping
from scribe.models.note_version import NoteVersion from scribe.models.note_version import NoteVersion
from scribe.models.rule_version import RuleVersion from scribe.models.rule_version import RuleVersion
from scribe.models.project import Project from scribe.models.project import Project
@@ -177,6 +178,7 @@ def _column_guard_targets():
"retrieval_tuning_events": ( "retrieval_tuning_events": (
RetrievalTuningEvent, backup._retrieval_tuning_event_rows, RetrievalTuningEvent, backup._retrieval_tuning_event_rows,
), ),
"moment_mappings": (MomentMapping, backup._moment_mapping_rows),
"design_systems": (DesignSystem, backup._design_system_rows), "design_systems": (DesignSystem, backup._design_system_rows),
"design_tokens": (DesignToken, backup._design_token_rows), "design_tokens": (DesignToken, backup._design_token_rows),
"repo_bindings": (RepoBinding, backup._repo_binding_rows), "repo_bindings": (RepoBinding, backup._repo_binding_rows),
@@ -294,6 +296,7 @@ def _import_guard_targets():
"rule_usage_events": backup._build_rule_usage_event, "rule_usage_events": backup._build_rule_usage_event,
"system_usage_events": backup._build_system_usage_event, "system_usage_events": backup._build_system_usage_event,
"retrieval_tuning_events": backup._build_retrieval_tuning_event, "retrieval_tuning_events": backup._build_retrieval_tuning_event,
"moment_mappings": backup._build_moment_mapping,
"design_systems": backup._build_design_system, "design_systems": backup._build_design_system,
"design_tokens": backup._build_design_token, "design_tokens": backup._build_design_token,
"repo_bindings": backup._build_repo_binding, "repo_bindings": backup._build_repo_binding,
@@ -449,7 +452,7 @@ def test_the_column_guard_covers_every_table_with_a_row_helper():
# REAL table names, as _BACKED_UP holds them — not the shorter keys the # REAL table names, as _BACKED_UP holds them — not the shorter keys the
# payload uses for the same sections. Getting this wrong is what the guard # payload uses for the same sections. Getting this wrong is what the guard
# caught on its own first run. # caught on its own first run.
join_tables = {"rule_systems"} join_tables = {"rule_systems", "rule_moments"}
covered = set(_column_guard_targets()) | join_tables covered = set(_column_guard_targets()) | join_tables
assert set(backup._BACKED_UP) - covered == set() assert set(backup._BACKED_UP) - covered == set()
# And no stale entries: every declaration must name a real target. # And no stale entries: every declaration must name a real target.
+1
View File
@@ -54,6 +54,7 @@ _SURFACE_PAIRS = (
("pre_tool_rule", "kbToolRuleThreshold"), ("pre_tool_rule", "kbToolRuleThreshold"),
("prompt_rule", "kbPromptRuleThreshold"), ("prompt_rule", "kbPromptRuleThreshold"),
("report_preference", "kbReportPrefThreshold"), ("report_preference", "kbReportPrefThreshold"),
("reply_rule", "kbReplyRuleThreshold"),
) )
# (services module, python constant, vue ref) for the defaults that are NOT # (services module, python constant, vue ref) for the defaults that are NOT