feat(moments): the reply moment holds a finished reply for one read (milestone 458 step 4b, #4922)
CI & Build / Python lint (push) Successful in 2s
CI & Build / Plugin hooks (push) Successful in 14s
CI & Build / TypeScript typecheck (push) Successful in 55s
CI & Build / integration (push) Successful in 1m1s
CI & Build / Python tests (push) Failing after 1m25s
CI & Build / Build & push image (push) Skipped
CI & Build / Python lint (push) Successful in 2s
CI & Build / Plugin hooks (push) Successful in 14s
CI & Build / TypeScript typecheck (push) Successful in 55s
CI & Build / integration (push) Successful in 1m1s
CI & Build / Python tests (push) Failing after 1m25s
CI & Build / Build & push image (push) Skipped
The reply is the one act no tool call marks, and it is where "let me know if it works" gets said. A new Stop hook (scribe_reply_check.sh) sends the finished reply to POST /api/plugin/reply-rules, which checks it twice: - mounted: every unopened RULE on reply.report, plus reply.ask when the reply asks a question. Deterministic. - semantic: the reply's head and tail against every rule's trigger, on a new ranked surface, reply_rule. It is the backstop for whatever the earlier arms missed. Its floor is its stop bar (default 0.80, budget 1), with its own Settings dials. The new stop_only stage records surfacing for the rule that holds and nothing else, because nothing else reached anyone. Following the operator's ruling from 456 step 8, a rule that holds blocks once, in the server's words. The hook blocks only on a reason it was given, so an unreachable instance never stops a session, and it never holds the rewrite. The ledger is the act checkpoint's own, so a rule holds a session once across both doors and the per-session cap counts both. The turn reader moved from the report check into scribe_defs.sh (scribe_turn_facts / scribe_turn_fact), so the two Stop hooks read a turn the same way. The output was checked identical on a real transcript. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -98,6 +98,9 @@ const kbToolRuleThreshold = ref("0.68");
|
|||||||
// prose rather than anything a tool produced (#3852).
|
// prose rather than anything a tool produced (#3852).
|
||||||
const kbPromptRuleThreshold = ref("0.72");
|
const kbPromptRuleThreshold = ref("0.72");
|
||||||
const kbReportPrefThreshold = ref("0.72");
|
const kbReportPrefThreshold = ref("0.72");
|
||||||
|
// The reply arm (milestone 458): its bar is the bar at which a finished reply
|
||||||
|
// is HELD for one read, so it sits with the checkpoint, not with the hints.
|
||||||
|
const kbReplyRuleThreshold = ref("0.8");
|
||||||
// The BUDGETS, one per arm (#4102). Until this step only auto-inject had one
|
// The BUDGETS, one per arm (#4102). Until this step only auto-inject had one
|
||||||
// and every other arm's ceiling was a module constant nobody could reach — so
|
// and every other arm's ceiling was a module constant nobody could reach — so
|
||||||
// the only control an operator had over a noisy surface was to raise its bar,
|
// the only control an operator had over a noisy surface was to raise its bar,
|
||||||
@@ -108,6 +111,7 @@ const kbRuleHintTopK = ref("5");
|
|||||||
const kbToolRuleTopK = ref("5");
|
const kbToolRuleTopK = ref("5");
|
||||||
const kbPromptRuleTopK = ref("3");
|
const kbPromptRuleTopK = ref("3");
|
||||||
const kbReportPrefTopK = ref("3");
|
const kbReportPrefTopK = ref("3");
|
||||||
|
const kbReplyRuleTopK = ref("1");
|
||||||
// What has been changed about retrieval, newest first — the review surface for
|
// What has been changed about retrieval, newest first — the review surface for
|
||||||
// changes the model made on the operator's behalf (#4102).
|
// changes the model made on the operator's behalf (#4102).
|
||||||
const tuningEvents = ref<TuningEvent[]>([]);
|
const tuningEvents = ref<TuningEvent[]>([]);
|
||||||
@@ -313,6 +317,8 @@ async function saveKbInject() {
|
|||||||
// hold the first command of every session behind whatever ranked first.
|
// hold the first command of every session behind whatever ranked first.
|
||||||
const cpT = asBar(kbCheckpointThreshold.value, 0.8);
|
const cpT = asBar(kbCheckpointThreshold.value, 0.8);
|
||||||
const rpT = asBar(kbReportPrefThreshold.value, 0.72);
|
const rpT = asBar(kbReportPrefThreshold.value, 0.72);
|
||||||
|
// A stop bar like the checkpoint's: a fallback of 0 would hold every reply.
|
||||||
|
const ryT = asBar(kbReplyRuleThreshold.value, 0.8);
|
||||||
// The budgets, clamped the way the server clamps them: a whole number in
|
// The budgets, clamped the way the server clamps them: a whole number in
|
||||||
// [1, 10]. Never 0 — an arm turned off is turned off by its switch, and a
|
// [1, 10]. Never 0 — an arm turned off is turned off by its switch, and a
|
||||||
// budget of zero would run the search, log the retrieval and render nothing,
|
// budget of zero would run the search, log the retrieval and render nothing,
|
||||||
@@ -324,11 +330,13 @@ async function saveKbInject() {
|
|||||||
const trK = asK(kbToolRuleTopK.value, 5);
|
const trK = asK(kbToolRuleTopK.value, 5);
|
||||||
const prK = asK(kbPromptRuleTopK.value, 3);
|
const prK = asK(kbPromptRuleTopK.value, 3);
|
||||||
const rpK = asK(kbReportPrefTopK.value, 3);
|
const rpK = asK(kbReportPrefTopK.value, 3);
|
||||||
|
const ryK = asK(kbReplyRuleTopK.value, 1);
|
||||||
kbWritePathTopK.value = String(wpK);
|
kbWritePathTopK.value = String(wpK);
|
||||||
kbRuleHintTopK.value = String(rhK);
|
kbRuleHintTopK.value = String(rhK);
|
||||||
kbToolRuleTopK.value = String(trK);
|
kbToolRuleTopK.value = String(trK);
|
||||||
kbPromptRuleTopK.value = String(prK);
|
kbPromptRuleTopK.value = String(prK);
|
||||||
kbReportPrefTopK.value = String(rpK);
|
kbReportPrefTopK.value = String(rpK);
|
||||||
|
kbReplyRuleTopK.value = String(ryK);
|
||||||
kbInjectThreshold.value = String(t);
|
kbInjectThreshold.value = String(t);
|
||||||
kbInjectTopK.value = String(k);
|
kbInjectTopK.value = String(k);
|
||||||
kbDupThresholdSnippet.value = String(dupSnip);
|
kbDupThresholdSnippet.value = String(dupSnip);
|
||||||
@@ -346,6 +354,7 @@ async function saveKbInject() {
|
|||||||
kbToolRuleThreshold.value = String(trT);
|
kbToolRuleThreshold.value = String(trT);
|
||||||
kbPromptRuleThreshold.value = String(prT);
|
kbPromptRuleThreshold.value = String(prT);
|
||||||
kbReportPrefThreshold.value = String(rpT);
|
kbReportPrefThreshold.value = String(rpT);
|
||||||
|
kbReplyRuleThreshold.value = String(ryT);
|
||||||
savingKbInject.value = true;
|
savingKbInject.value = true;
|
||||||
kbInjectSaved.value = false;
|
kbInjectSaved.value = false;
|
||||||
try {
|
try {
|
||||||
@@ -379,6 +388,9 @@ async function saveKbInject() {
|
|||||||
// and a constant that lands under the bar is a dead arm, not a quiet
|
// and a constant that lands under the bar is a dead arm, not a quiet
|
||||||
// one (#3860).
|
// one (#3860).
|
||||||
kb_reportpref_threshold: String(rpT),
|
kb_reportpref_threshold: String(rpT),
|
||||||
|
// The reply backstop's own bar: it HOLDS a reply, so like the
|
||||||
|
// checkpoint it is never derived from a hint bar.
|
||||||
|
kb_replyrule_threshold: String(ryT),
|
||||||
// The budgets. Every one of these keys is recognised by the server as a
|
// The budgets. Every one of these keys is recognised by the server as a
|
||||||
// retrieval dial, so this save is recorded in the tuning history as a
|
// retrieval dial, so this save is recorded in the tuning history as a
|
||||||
// change the OPERATOR made — which is the one entry the model must not
|
// change the OPERATOR made — which is the one entry the model must not
|
||||||
@@ -388,6 +400,7 @@ async function saveKbInject() {
|
|||||||
kb_toolrule_top_k: String(trK),
|
kb_toolrule_top_k: String(trK),
|
||||||
kb_promptrule_top_k: String(prK),
|
kb_promptrule_top_k: String(prK),
|
||||||
kb_reportpref_top_k: String(rpK),
|
kb_reportpref_top_k: String(rpK),
|
||||||
|
kb_replyrule_top_k: String(ryK),
|
||||||
kb_duplicate_threshold_snippet: String(dupSnip),
|
kb_duplicate_threshold_snippet: String(dupSnip),
|
||||||
kb_duplicate_threshold_note: String(dupNote),
|
kb_duplicate_threshold_note: String(dupNote),
|
||||||
kb_duplicate_threshold_task: String(dupTask),
|
kb_duplicate_threshold_task: String(dupTask),
|
||||||
@@ -859,6 +872,9 @@ onMounted(async () => {
|
|||||||
if (allSettings.kb_reportpref_threshold !== undefined) {
|
if (allSettings.kb_reportpref_threshold !== undefined) {
|
||||||
kbReportPrefThreshold.value = allSettings.kb_reportpref_threshold;
|
kbReportPrefThreshold.value = allSettings.kb_reportpref_threshold;
|
||||||
}
|
}
|
||||||
|
if (allSettings.kb_replyrule_threshold !== undefined) {
|
||||||
|
kbReplyRuleThreshold.value = allSettings.kb_replyrule_threshold;
|
||||||
|
}
|
||||||
if (allSettings.kb_writepath_threshold !== undefined) {
|
if (allSettings.kb_writepath_threshold !== undefined) {
|
||||||
kbWritePathThreshold.value = allSettings.kb_writepath_threshold;
|
kbWritePathThreshold.value = allSettings.kb_writepath_threshold;
|
||||||
}
|
}
|
||||||
@@ -882,6 +898,9 @@ onMounted(async () => {
|
|||||||
if (allSettings.kb_reportpref_top_k !== undefined) {
|
if (allSettings.kb_reportpref_top_k !== undefined) {
|
||||||
kbReportPrefTopK.value = allSettings.kb_reportpref_top_k;
|
kbReportPrefTopK.value = allSettings.kb_reportpref_top_k;
|
||||||
}
|
}
|
||||||
|
if (allSettings.kb_replyrule_top_k !== undefined) {
|
||||||
|
kbReplyRuleTopK.value = allSettings.kb_replyrule_top_k;
|
||||||
|
}
|
||||||
await loadTuningHistory();
|
await loadTuningHistory();
|
||||||
await loadSurfaces();
|
await loadSurfaces();
|
||||||
if (allSettings.kb_duplicate_threshold_snippet !== undefined) {
|
if (allSettings.kb_duplicate_threshold_snippet !== undefined) {
|
||||||
@@ -1919,6 +1938,41 @@ async function deleteUser(userId: number) {
|
|||||||
/>
|
/>
|
||||||
<p class="field-hint">How many preferences a finished task may be shown (1–10).</p>
|
<p class="field-hint">How many preferences a finished task may be shown (1–10).</p>
|
||||||
</div>
|
</div>
|
||||||
|
<div class="field">
|
||||||
|
<label for="kb-replyrule-threshold">Reply check threshold (0–1)</label>
|
||||||
|
<input
|
||||||
|
id="kb-replyrule-threshold"
|
||||||
|
v-model="kbReplyRuleThreshold"
|
||||||
|
type="number"
|
||||||
|
min="0"
|
||||||
|
max="1"
|
||||||
|
step="0.01"
|
||||||
|
class="fs-input input"
|
||||||
|
style="max-width: 8rem"
|
||||||
|
/>
|
||||||
|
<p class="field-hint">
|
||||||
|
The bar at which a <em>finished reply</em> is held for one read: the
|
||||||
|
reply is searched against every rule's trigger, and a rule this
|
||||||
|
session has not opened that scores at or above this bar holds the
|
||||||
|
reply once, naming the rule. It is the backstop for whatever the
|
||||||
|
earlier checks missed, so it is a stop bar like the one above, not a
|
||||||
|
hint bar — keep it high.
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
<div class="field">
|
||||||
|
<label for="kb-replyrule-topk">Rules searched per reply</label>
|
||||||
|
<input
|
||||||
|
id="kb-replyrule-topk"
|
||||||
|
v-model="kbReplyRuleTopK"
|
||||||
|
type="number"
|
||||||
|
min="1"
|
||||||
|
max="10"
|
||||||
|
step="1"
|
||||||
|
class="fs-input input"
|
||||||
|
style="max-width: 8rem"
|
||||||
|
/>
|
||||||
|
<p class="field-hint">How many of the best-scoring rules a reply is checked against (1–10). Only the best can hold the reply; the rest are counted, which is what tells you how close the others came.</p>
|
||||||
|
</div>
|
||||||
<!-- THE REVIEW SURFACE (#4102). These numbers are maintained by the
|
<!-- THE REVIEW SURFACE (#4102). These numbers are maintained by the
|
||||||
model that uses them: it reads which records each bar refused and
|
model that uses them: it reads which records each bar refused and
|
||||||
moves the dial with the argument attached. This panel is the other
|
moves the dial with the argument attached. This panel is the other
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"name": "scribe",
|
"name": "scribe",
|
||||||
"description": "Scribe for Claude Code: connects the scribe MCP server, adds the hooks that deliver live project state and relevant records at the right moment, ships the shared client-neutral Scribe skills (using-scribe, writing-plans, reporting-back, systematic-debugging, verification, brainstorming, reusing-code, shape-accounting), and syncs your saved Scribe Processes as skills (/scribe:sync).",
|
"description": "Scribe for Claude Code: connects the scribe MCP server, adds the hooks that deliver live project state and relevant records at the right moment, ships the shared client-neutral Scribe skills (using-scribe, writing-plans, reporting-back, systematic-debugging, verification, brainstorming, reusing-code, shape-accounting), and syncs your saved Scribe Processes as skills (/scribe:sync).",
|
||||||
"version": "2026.10.05.1624",
|
"version": "2026.10.05.1630",
|
||||||
"author": {
|
"author": {
|
||||||
"name": "Bryan Van Deusen"
|
"name": "Bryan Van Deusen"
|
||||||
},
|
},
|
||||||
|
|||||||
@@ -41,6 +41,7 @@ another one means adding files, not moving or rewriting any.
|
|||||||
| `hooks/scribe_tool_rules.sh` | PreToolUse on shell commands: `GET /api/plugin/tool-rules`. |
|
| `hooks/scribe_tool_rules.sh` | PreToolUse on shell commands: `GET /api/plugin/tool-rules`. |
|
||||||
| `hooks/scribe_moment.sh` | PreToolUse on every tool: the rules mounted on the moments the call reaches, `POST /api/plugin/moment`; skips tools `GET /api/plugin/moment-tools` says reach nothing mounted, and Scribe's own (their responses carry `moment_rules`). |
|
| `hooks/scribe_moment.sh` | PreToolUse on every tool: the rules mounted on the moments the call reaches, `POST /api/plugin/moment`; skips tools `GET /api/plugin/moment-tools` says reach nothing mounted, and Scribe's own (their responses carry `moment_rules`). |
|
||||||
| `hooks/scribe_report_check.sh` | Stop: when the turn closed a task, checks the reply for the completion sections and reports to `GET /api/plugin/report-check`; blocks once, with the reason the server returns. |
|
| `hooks/scribe_report_check.sh` | Stop: when the turn closed a task, checks the reply for the completion sections and reports to `GET /api/plugin/report-check`; blocks once, with the reason the server returns. |
|
||||||
|
| `hooks/scribe_reply_check.sh` | Stop: sends the finished reply to `POST /api/plugin/reply-rules` — the rules mounted on the reply moments, and the reply against every rule's trigger; holds once per rule, with the reason the server returns, and never holds the rewrite. |
|
||||||
| `hooks/scribe_shape_check.sh` | Stop: sends the definitions the turn wrote (the write hooks' `<sid>.written.ids` ledger) to `GET /api/plugin/shape-check`; blocks once, with the reason the server returns, so the agent judges what it built. |
|
| `hooks/scribe_shape_check.sh` | Stop: sends the definitions the turn wrote (the write hooks' `<sid>.written.ids` ledger) to `GET /api/plugin/shape-check`; blocks once, with the reason the server returns, so the agent judges what it built. |
|
||||||
| `hooks/scribe_sync_processes.sh` + `commands/sync.md` | `GET /api/plugin/processes` → `~/.claude/skills/scribe-proc-*` stubs; `/scribe:sync` on demand. |
|
| `hooks/scribe_sync_processes.sh` + `commands/sync.md` | `GET /api/plugin/processes` → `~/.claude/skills/scribe-proc-*` stubs; `/scribe:sync` on demand. |
|
||||||
| `hooks/scribe_defs.sh` | Shared shell helpers: config, dedup ledgers, outage line. |
|
| `hooks/scribe_defs.sh` | Shared shell helpers: config, dedup ledgers, outage line. |
|
||||||
|
|||||||
@@ -108,6 +108,10 @@
|
|||||||
"type": "command",
|
"type": "command",
|
||||||
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_report_check.sh\""
|
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_report_check.sh\""
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
"type": "command",
|
||||||
|
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_reply_check.sh\""
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"type": "command",
|
"type": "command",
|
||||||
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_shape_check.sh\""
|
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_shape_check.sh\""
|
||||||
|
|||||||
@@ -1226,6 +1226,74 @@ scribe_clear_session_ledgers() {
|
|||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# WHAT HAPPENED IN THE LAST TURN of a transcript (#4107): scribe_turn.awk's
|
||||||
|
# facts — bounded, closed, task_ids, reply — one `KEY<TAB>VALUE` per line.
|
||||||
|
# Shared by the two Stop hooks that read a turn: the report check and the
|
||||||
|
# reply check (milestone 458). Empty output when the turn cannot be bounded.
|
||||||
|
#
|
||||||
|
# The turn is parsed once. A window of recent lines, flattened record by
|
||||||
|
# record and then read by scribe_turn.awk, which carries the turn-bounding
|
||||||
|
# rules. A line that does not parse is dropped and the rest are still read —
|
||||||
|
# the first line of a `tail -n 3000` window is routinely half a record. If the
|
||||||
|
# window holds no prompt, the turn cannot be bounded, and the caller says
|
||||||
|
# nothing rather than guessing.
|
||||||
|
#
|
||||||
|
# WHERE THE TURN STARTS, FOUND BEFORE PARSING RATHER THAN AFTER. The window is
|
||||||
|
# 3000 lines and routinely 7MB, of which a turn is the last few hundred lines
|
||||||
|
# and about a sixth of the bytes — the rest is tool results these checks never
|
||||||
|
# look at. The predecessor parsed all of it and threw most away, which jq
|
||||||
|
# could afford and a parser written in awk cannot: measured at 6.5s for a 7MB
|
||||||
|
# window against 94ms, on a hook that runs at the end of every turn.
|
||||||
|
#
|
||||||
|
# So grep — C, and reading a FIXED string — narrows first. A prompt record is
|
||||||
|
# `"type":"user"` whose `content` is a STRING; a tool result is the same type
|
||||||
|
# with an ARRAY, and the two are told apart by the character after `"content":`.
|
||||||
|
#
|
||||||
|
# WHY A FIXED STRING IS EXACT HERE, and not the usual regex-over-JSON guess.
|
||||||
|
# Every quote inside a JSON string is backslash-escaped, so a needle carrying
|
||||||
|
# UNESCAPED quotes cannot occur inside any string value — it can only match at
|
||||||
|
# a record's own top level. `"message":{"role":"user","content":"` therefore
|
||||||
|
# matches real prompt records and nothing else. Measured over a 27MB transcript
|
||||||
|
# against a full JSON parse: 152 prompt records, 152 matches, no misses and no
|
||||||
|
# extras. The looser `"content":"` matched 1101 lines, because a tool_result
|
||||||
|
# block has a `content` key of its own — which is the trap this avoids.
|
||||||
|
#
|
||||||
|
# THE LAST MATCH, not a few before it, because the margin is not free: the
|
||||||
|
# lines between two prompts are mostly tool results, and backing off three
|
||||||
|
# matches took the window from 53KB to 1.3MB and the parse from 21ms to 2.6s.
|
||||||
|
# The fallback below is the safety net instead — it is exact where a margin is
|
||||||
|
# only approximate, and it costs nothing in the case that actually happens.
|
||||||
|
#
|
||||||
|
# The needle assumes a key ORDER that a future Claude Code could change. If it
|
||||||
|
# does, grep matches nothing, `start` stays 1, and the whole window is read the
|
||||||
|
# slow way — correct, and slow, which is the right way round for a check that
|
||||||
|
# can block a stop.
|
||||||
|
_scribe_turn_window() { tail -n 3000 "$1" 2>/dev/null; }
|
||||||
|
_scribe_turn_parse() {
|
||||||
|
_scribe_turn_window "$1" | tail -n +"${2:-1}" | scribe_json_flat_lines \
|
||||||
|
| awk -f "$SCRIBE_HOOK_DIR/scribe_turn.awk" 2>/dev/null
|
||||||
|
}
|
||||||
|
|
||||||
|
scribe_turn_facts() {
|
||||||
|
local transcript="$1" start facts
|
||||||
|
[ -n "$transcript" ] && [ -f "$transcript" ] || return 0
|
||||||
|
start=$(_scribe_turn_window "$transcript" \
|
||||||
|
| grep -n -F '"message":{"role":"user","content":"' 2>/dev/null \
|
||||||
|
| cut -d: -f1 | awk '{ last = $0 } END { if (NR) print last }')
|
||||||
|
case "$start" in ''|*[!0-9]*) start=1 ;; esac
|
||||||
|
facts=$(_scribe_turn_parse "$transcript" "$start")
|
||||||
|
if [ "$(scribe_turn_fact "$facts" bounded)" != "1" ] && [ "$start" != "1" ]; then
|
||||||
|
facts=$(_scribe_turn_parse "$transcript" 1)
|
||||||
|
fi
|
||||||
|
printf '%s\n' "$facts"
|
||||||
|
}
|
||||||
|
|
||||||
|
# $1 facts from scribe_turn_facts, $2 key → that fact's value, or "".
|
||||||
|
scribe_turn_fact() {
|
||||||
|
printf '%s\n' "$1" | awk -F'\t' -v k="$2" '$1 == k { print substr($0, index($0, "\t") + 1); exit }'
|
||||||
|
}
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# WHICH PROJECT IS THIS DIRECTORY'S? (#4085)
|
# WHICH PROJECT IS THIS DIRECTORY'S? (#4085)
|
||||||
#
|
#
|
||||||
|
|||||||
@@ -0,0 +1,91 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# Scribe — Stop hook: the reply moment (milestone 458 step 4, folded in from
|
||||||
|
# milestone 456 step 8).
|
||||||
|
#
|
||||||
|
# A reply is the one act no tool call marks, and it is where "please check
|
||||||
|
# this on your end" or "it's done" gets said — so every arm that fires on a
|
||||||
|
# tool call misses it. This hook sends the finished reply to the server, which
|
||||||
|
# checks it twice over: the rules MOUNTED on the reply moments, and the reply
|
||||||
|
# text against every rule's trigger, as the backstop for whatever the earlier
|
||||||
|
# arms missed. An unopened rule from either half holds the reply for one read.
|
||||||
|
#
|
||||||
|
# THE SAME CONTRACT AS scribe_report_check.sh, and for its reasons:
|
||||||
|
# - the server decides and supplies the words; this hook blocks only on a
|
||||||
|
# reason it was given, so an unconfigured or unreachable instance never
|
||||||
|
# stops a session;
|
||||||
|
# - the rewrite is never held (`stop_hook_active`), so a hold costs one turn
|
||||||
|
# at most;
|
||||||
|
# - a rule holds a session once. The ledger is the act checkpoint's own
|
||||||
|
# (`<sid>.checkpoint.ids`), so a rule that already stopped a command does
|
||||||
|
# not stop the reply about it, and the per-session cap counts both doors.
|
||||||
|
#
|
||||||
|
# Env:
|
||||||
|
# SCRIBE_URL / SCRIBE_TOKEN override for the settings.json dogfooding path.
|
||||||
|
|
||||||
|
command -v curl >/dev/null 2>&1 || exit 0
|
||||||
|
|
||||||
|
# shellcheck source=plugin/hooks/scribe_defs.sh
|
||||||
|
. "$(dirname "${BASH_SOURCE[0]}")/scribe_defs.sh"
|
||||||
|
|
||||||
|
event=$(cat 2>/dev/null || true)
|
||||||
|
event_flat=$(printf '%s' "$event" | scribe_json_flat)
|
||||||
|
transcript=$(scribe_json_pick "$event_flat" '.transcript_path')
|
||||||
|
session_id=$(scribe_json_pick "$event_flat" '.session_id')
|
||||||
|
active=$(scribe_json_pick "$event_flat" '.stop_hook_active')
|
||||||
|
event_cwd=$(scribe_json_pick "$event_flat" '.cwd')
|
||||||
|
|
||||||
|
# The rewrite after a hold goes out as written.
|
||||||
|
[ "$active" = "true" ] && exit 0
|
||||||
|
[ -n "$transcript" ] && [ -f "$transcript" ] && [ -n "$session_id" ] || exit 0
|
||||||
|
scribe_config || exit 0
|
||||||
|
|
||||||
|
facts=$(scribe_turn_facts "$transcript")
|
||||||
|
[ "$(scribe_turn_fact "$facts" bounded)" = "1" ] || exit 0
|
||||||
|
reply=$(scribe_turn_fact "$facts" reply | scribe_json_unescape)
|
||||||
|
[ -n "$(printf '%s' "$reply" | tr -d '[:space:]')" ] || exit 0
|
||||||
|
|
||||||
|
# Bounded before encoding. The server reads the head and the tail — the part
|
||||||
|
# of a report that asks something of the reader is at its end — so a very
|
||||||
|
# long reply keeps both.
|
||||||
|
if [ "${#reply}" -gt 12000 ]; then
|
||||||
|
reply="${reply:0:4000} … ${reply: -8000}"
|
||||||
|
fi
|
||||||
|
reply_esc=$(printf '%s' "$reply" | scribe_json_escape) || exit 0
|
||||||
|
|
||||||
|
state_dir="${TMPDIR:-/tmp}/scribe-priorart"
|
||||||
|
mkdir -p "$state_dir" 2>/dev/null || true
|
||||||
|
safe_sid=$(printf '%s' "$session_id" | tr -c 'A-Za-z0-9._-' '_')
|
||||||
|
rulefile="$state_dir/${safe_sid}.rules.ids"
|
||||||
|
stopfile="$state_dir/${safe_sid}.checkpoint.ids"
|
||||||
|
|
||||||
|
query=""
|
||||||
|
scope=$(scribe_scope_query "${event_cwd:-${CLAUDE_PROJECT_DIR:-$PWD}}")
|
||||||
|
[ -n "$scope" ] && query="&${scope}"
|
||||||
|
rule_seen=$(scribe_rules_live "$rulefile")
|
||||||
|
[ -n "$rule_seen" ] && query="${query}&exclude_rule_ids=${rule_seen}"
|
||||||
|
query="${query}$(scribe_held_query "$state_dir/${safe_sid}.opened.ids")"
|
||||||
|
if [ -f "$stopfile" ]; then
|
||||||
|
stopped=$(grep -E '^[0-9]+$' "$stopfile" 2>/dev/null | paste -sd, -)
|
||||||
|
[ -n "$stopped" ] && query="${query}&stopped_rule_ids=${stopped}"
|
||||||
|
fi
|
||||||
|
query=${query#&}
|
||||||
|
|
||||||
|
answer=$(printf '{"reply":"%s"}' "$reply_esc" | curl -fsS --max-time 6 \
|
||||||
|
-H "Authorization: Bearer ${token}" \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
--data-binary @- \
|
||||||
|
"${url%/}/api/plugin/reply-rules${query:+?$query}" 2>/dev/null) || exit 0
|
||||||
|
|
||||||
|
answer_flat=$(printf '%s' "$answer" | scribe_json_flat)
|
||||||
|
reason=$(scribe_json_pick "$answer_flat" '.reason')
|
||||||
|
[ -n "$reason" ] || exit 0
|
||||||
|
|
||||||
|
# Recorded BEFORE the block is emitted, for the act checkpoint's reason: a
|
||||||
|
# hold that is shown and not recorded is one that can be shown again.
|
||||||
|
for id in $(scribe_json_list "$answer_flat" '.rule_ids'); do
|
||||||
|
scribe_checkpoint_allowed "$stopfile" "$id" || true
|
||||||
|
done
|
||||||
|
scribe_json_list "$answer_flat" '.rule_ids' | scribe_rules_append "$rulefile"
|
||||||
|
|
||||||
|
printf '{"decision":"block","reason":"%s"}\n' "$(printf '%s' "$reason" | scribe_json_escape)"
|
||||||
|
exit 0
|
||||||
@@ -78,56 +78,9 @@ grep -q -E '"name":[[:space:]]*"([^"]*__)?(update|create)_task"' < <(tail -c 200
|
|||||||
exit 0
|
exit 0
|
||||||
}
|
}
|
||||||
|
|
||||||
# The turn, parsed once. A window of recent lines, flattened record by record
|
# The turn, parsed once — scribe_turn_facts, shared with the reply check.
|
||||||
# and then read by scribe_turn.awk, which carries the turn-bounding rules. A
|
facts=$(scribe_turn_facts "$transcript")
|
||||||
# line that does not parse is dropped and the rest are still read — the first
|
fact() { scribe_turn_fact "$facts" "$1"; }
|
||||||
# line of a `tail -n 3000` window is routinely half a record. If the window
|
|
||||||
# holds no prompt, the turn cannot be bounded, so the hook reports nothing and
|
|
||||||
# stays out of the way.
|
|
||||||
window() { tail -n 3000 "$transcript" 2>/dev/null; }
|
|
||||||
turn_facts() { window | tail -n +"${1:-1}" | scribe_json_flat_lines \
|
|
||||||
| awk -f "$SCRIBE_HOOK_DIR/scribe_turn.awk" 2>/dev/null; }
|
|
||||||
|
|
||||||
# WHERE THE TURN STARTS, FOUND BEFORE PARSING RATHER THAN AFTER. The window is
|
|
||||||
# 3000 lines and routinely 7MB, of which a turn is the last few hundred lines
|
|
||||||
# and about a sixth of the bytes — the rest is tool results this check never
|
|
||||||
# looks at. The predecessor parsed all of it and threw most away, which jq
|
|
||||||
# could afford and a parser written in awk cannot: measured at 6.5s for a 7MB
|
|
||||||
# window against 94ms, on a hook that runs at the end of every turn.
|
|
||||||
#
|
|
||||||
# So grep — C, and reading a FIXED string — narrows first. A prompt record is
|
|
||||||
# `"type":"user"` whose `content` is a STRING; a tool result is the same type
|
|
||||||
# with an ARRAY, and the two are told apart by the character after `"content":`.
|
|
||||||
#
|
|
||||||
# WHY A FIXED STRING IS EXACT HERE, and not the usual regex-over-JSON guess.
|
|
||||||
# Every quote inside a JSON string is backslash-escaped, so a needle carrying
|
|
||||||
# UNESCAPED quotes cannot occur inside any string value — it can only match at
|
|
||||||
# a record's own top level. `"message":{"role":"user","content":"` therefore
|
|
||||||
# matches real prompt records and nothing else. Measured over a 27MB transcript
|
|
||||||
# against a full JSON parse: 152 prompt records, 152 matches, no misses and no
|
|
||||||
# extras. The looser `"content":"` matched 1101 lines, because a tool_result
|
|
||||||
# block has a `content` key of its own — which is the trap this avoids.
|
|
||||||
#
|
|
||||||
# THE LAST MATCH, not a few before it, because the margin is not free: the
|
|
||||||
# lines between two prompts are mostly tool results, and backing off three
|
|
||||||
# matches took the window from 53KB to 1.3MB and the parse from 21ms to 2.6s.
|
|
||||||
# The fallback below is the safety net instead — it is exact where a margin is
|
|
||||||
# only approximate, and it costs nothing in the case that actually happens.
|
|
||||||
#
|
|
||||||
# The needle assumes a key ORDER that a future Claude Code could change. If it
|
|
||||||
# does, grep matches nothing, `start` stays 1, and the whole window is read the
|
|
||||||
# slow way — correct, and slow, which is the right way round for a check that
|
|
||||||
# can block a stop.
|
|
||||||
start=$(window | grep -n -F '"message":{"role":"user","content":"' 2>/dev/null \
|
|
||||||
| cut -d: -f1 | awk '{ last = $0 } END { if (NR) print last }')
|
|
||||||
case "$start" in ''|*[!0-9]*) start=1 ;; esac
|
|
||||||
|
|
||||||
facts=$(turn_facts "$start")
|
|
||||||
fact() { printf '%s\n' "$facts" | awk -F'\t' -v k="$1" '$1 == k { print substr($0, index($0, "\t") + 1); exit }'; }
|
|
||||||
|
|
||||||
if [ "$(fact bounded)" != "1" ] && [ "$start" != "1" ]; then
|
|
||||||
facts=$(turn_facts 1)
|
|
||||||
fi
|
|
||||||
|
|
||||||
[ "$(fact bounded)" = "1" ] || exit 0
|
[ "$(fact bounded)" = "1" ] || exit 0
|
||||||
closed=$(fact closed)
|
closed=$(fact closed)
|
||||||
|
|||||||
@@ -366,6 +366,12 @@ SMOKE_EVENTS: dict[str, str] = {
|
|||||||
{"session_id": "smoke", "transcript_path": "/nonexistent/smoke.jsonl",
|
{"session_id": "smoke", "transcript_path": "/nonexistent/smoke.jsonl",
|
||||||
"cwd": ".", "hook_event_name": "Stop", "stop_hook_active": False}
|
"cwd": ".", "hook_event_name": "Stop", "stop_hook_active": False}
|
||||||
),
|
),
|
||||||
|
# The reply moment (milestone 458). No transcript, so no reply: silent and
|
||||||
|
# never a block, configured or not.
|
||||||
|
"scribe_reply_check.sh": json.dumps(
|
||||||
|
{"session_id": "smoke", "transcript_path": "/nonexistent/smoke.jsonl",
|
||||||
|
"cwd": ".", "hook_event_name": "Stop", "stop_hook_active": False}
|
||||||
|
),
|
||||||
# The end-of-turn shape check (milestone 439). With no ledger for the
|
# The end-of-turn shape check (milestone 439). With no ledger for the
|
||||||
# session there is nothing to ask about, and with no instance there is
|
# session there is nothing to ask about, and with no instance there is
|
||||||
# nobody to ask: silent, and never a block, configured or not.
|
# nobody to ask: silent, and never a block, configured or not.
|
||||||
|
|||||||
@@ -294,6 +294,42 @@ async def moment():
|
|||||||
})
|
})
|
||||||
|
|
||||||
|
|
||||||
|
@plugin_bp.post("/reply-rules")
|
||||||
|
@login_required
|
||||||
|
@memoized_query_embeddings
|
||||||
|
async def reply_rules():
|
||||||
|
"""Whether the reply that ends a turn is held for one read (milestone 458).
|
||||||
|
|
||||||
|
Called by the plugin's Stop hook with the finished reply. Two halves: the
|
||||||
|
rules MOUNTED on the reply moments (`reply.report`, and `reply.ask` when
|
||||||
|
the reply asks something), and the reply text against every rule's
|
||||||
|
trigger — the backstop for whatever the earlier arms missed.
|
||||||
|
|
||||||
|
Body: `{"reply": "<text>"}`. Query: `repo` / `project_id` for the scope;
|
||||||
|
`held_rule_ids` (opened this session — exempt, as at the act checkpoint);
|
||||||
|
`exclude_rule_ids` (named this session — not re-counted as surfaced);
|
||||||
|
`stopped_rule_ids` (already held a reply or an act this session — a rule
|
||||||
|
holds once, and the per-session cap counts these).
|
||||||
|
|
||||||
|
Returns `reason` — the words the hook blocks with, empty when nothing
|
||||||
|
holds — plus `rule_ids` for the hook's ledger and `moments` reached.
|
||||||
|
"""
|
||||||
|
data = await request.get_json(silent=True) or {}
|
||||||
|
reply = str(data.get("reply") or "")
|
||||||
|
project_id, _repo, _unbound = await _project_scope()
|
||||||
|
held = await moment_delivery_svc.reply_hold(
|
||||||
|
g.user.id, reply, project_id=project_id,
|
||||||
|
exclude=frozenset(_int_list(request.args.get("exclude_rule_ids"))),
|
||||||
|
held=frozenset(_int_list(request.args.get("held_rule_ids"))),
|
||||||
|
stopped=frozenset(_int_list(request.args.get("stopped_rule_ids"))),
|
||||||
|
)
|
||||||
|
return jsonify({
|
||||||
|
"reason": held.get("reason", ""),
|
||||||
|
"rule_ids": held.get("rule_ids", []),
|
||||||
|
"moments": held.get("moments", []),
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
@plugin_bp.get("/prior-art")
|
@plugin_bp.get("/prior-art")
|
||||||
@login_required
|
@login_required
|
||||||
@memoized_query_embeddings
|
@memoized_query_embeddings
|
||||||
|
|||||||
@@ -117,3 +117,153 @@ async def attach_moment_rules(
|
|||||||
except Exception: # noqa: BLE001 - a decoration never breaks the payload
|
except Exception: # noqa: BLE001 - a decoration never breaks the payload
|
||||||
logger.debug("moment rules for %s could not be attached", tool, exc_info=True)
|
logger.debug("moment rules for %s could not be attached", tool, exc_info=True)
|
||||||
return data
|
return data
|
||||||
|
|
||||||
|
|
||||||
|
# ── The reply moment: the backstop at the end of a turn ─────────────────
|
||||||
|
#
|
||||||
|
# The Stop hook's door, folded in from milestone 456 step 8. A reply is the
|
||||||
|
# one act no tool call marks, and the moment where "please verify this on
|
||||||
|
# your end" gets said — so it is checked twice over: the rules MOUNTED on the
|
||||||
|
# reply moments (deterministic, somebody said they belong here), and the
|
||||||
|
# reply text against every rule's trigger (the backstop for whatever the
|
||||||
|
# earlier arms missed). An unopened rule from either half HOLDS the reply for
|
||||||
|
# one read, in these words; the hook never holds the rewrite.
|
||||||
|
|
||||||
|
# The reply's head and tail. The embedder reads ~512 tokens, and the part of
|
||||||
|
# a report that asks something of the reader — "let me know if it works" —
|
||||||
|
# is at the END, where a head-only query would cut it off.
|
||||||
|
_REPLY_HEAD_CHARS = 600
|
||||||
|
_REPLY_TAIL_CHARS = 1200
|
||||||
|
|
||||||
|
_REACHED_BY_REPLY = "the reply that ends this turn"
|
||||||
|
|
||||||
|
|
||||||
|
def reply_moments(reply: str) -> list[dict]:
|
||||||
|
"""The moments a finished reply reaches: always `reply.report`, and
|
||||||
|
`reply.ask` when a line of its prose ends in a question."""
|
||||||
|
reached = [{"moment": "reply.report", "tool": "", "match": _REACHED_BY_REPLY,
|
||||||
|
"via": moment_actions.DEFAULT}]
|
||||||
|
in_code = False
|
||||||
|
for line in (reply or "").splitlines():
|
||||||
|
stripped = line.strip()
|
||||||
|
if stripped.startswith("```"):
|
||||||
|
in_code = not in_code
|
||||||
|
continue
|
||||||
|
if not in_code and stripped.endswith("?"):
|
||||||
|
reached.append({"moment": "reply.ask", "tool": "", "match": "a question in the reply",
|
||||||
|
"via": moment_actions.DEFAULT})
|
||||||
|
break
|
||||||
|
return reached
|
||||||
|
|
||||||
|
|
||||||
|
def _reply_query(reply: str) -> str:
|
||||||
|
text = " ".join((reply or "").split())
|
||||||
|
if len(text) <= _REPLY_HEAD_CHARS + _REPLY_TAIL_CHARS:
|
||||||
|
return text
|
||||||
|
return text[:_REPLY_HEAD_CHARS] + " … " + text[-_REPLY_TAIL_CHARS:]
|
||||||
|
|
||||||
|
|
||||||
|
def reply_hold_reason(held: list[dict]) -> str:
|
||||||
|
"""The text the agent reads INSTEAD of its reply going out.
|
||||||
|
|
||||||
|
A practice, not a prohibition (rule 165), on the act checkpoint's model:
|
||||||
|
nothing here knows the reply is wrong. It names each rule with why it was
|
||||||
|
raised, gives the remedy as one call per rule, and says the reply may go
|
||||||
|
out unchanged — so a reader who finds the rules beside the point is out in
|
||||||
|
as many calls as there are rules, and is never held twice.
|
||||||
|
"""
|
||||||
|
if not held:
|
||||||
|
return ""
|
||||||
|
parts = []
|
||||||
|
for item in held:
|
||||||
|
why = (f"mounted on {item['moment']}" if item.get("moment")
|
||||||
|
else f"scores {item['score']} against this reply")
|
||||||
|
trigger = f"; it applies when {item['trigger']}" if item.get("trigger") else ""
|
||||||
|
parts.append(f"“{item['title']}” ({why}{trigger}) — get_rule({item['rule_id']})")
|
||||||
|
noun = "a standing rule" if len(held) == 1 else "standing rules"
|
||||||
|
return (
|
||||||
|
f"Held for one read before this reply goes out: {noun} this session "
|
||||||
|
f"has not opened: " + "; ".join(parts) + ". Read "
|
||||||
|
+ ("it" if len(held) == 1 else "them")
|
||||||
|
+ ", then send the reply — unchanged if it already does what the rule "
|
||||||
|
"asks, which is a judgement only you can make. This check runs once "
|
||||||
|
"per rule; the rewrite is never held."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def reply_hold(
|
||||||
|
user_id: int, reply: str, *, project_id: int | None = None,
|
||||||
|
exclude: frozenset[int] = frozenset(), held: frozenset[int] = frozenset(),
|
||||||
|
stopped: frozenset[int] = frozenset(),
|
||||||
|
) -> dict:
|
||||||
|
"""Whether this finished reply is held, and in what words.
|
||||||
|
|
||||||
|
`held` is what the session OPENED (the act checkpoint's exemption);
|
||||||
|
`stopped` is what an earlier hold already put in front of it, so a rule
|
||||||
|
holds a session once. Past the act checkpoint's per-session cap nothing
|
||||||
|
holds — a mis-set bar degrades to a quiet session, never a stuck one.
|
||||||
|
|
||||||
|
Returns {} or {"reason", "rule_ids", "moments"}. Fails open to {}.
|
||||||
|
"""
|
||||||
|
from scribe.services import plugin_context as pc
|
||||||
|
from scribe.services import rule_usage, rulebooks
|
||||||
|
from scribe.services.retrieval_surfaces import budget_for, floor_for
|
||||||
|
|
||||||
|
try:
|
||||||
|
if not (reply or "").strip() or len(stopped) >= pc.CHECKPOINT_SESSION_CAP:
|
||||||
|
return {}
|
||||||
|
reached = reply_moments(reply)
|
||||||
|
names = [hit["moment"] for hit in reached]
|
||||||
|
skip = set(held) | set(stopped)
|
||||||
|
room = pc.CHECKPOINT_SESSION_CAP - len(stopped)
|
||||||
|
out: list[dict] = []
|
||||||
|
|
||||||
|
# The mounted half: deterministic, so every unopened RULE on these
|
||||||
|
# moments holds. A preference claims no such force.
|
||||||
|
for rule, at in await rulebooks.rules_on_moments(user_id, names, project_id or None):
|
||||||
|
if rule.id in skip or rule.kind == "preference":
|
||||||
|
continue
|
||||||
|
out.append({"rule_id": rule.id, "title": rule.title, "moment": at,
|
||||||
|
"trigger": (rule.when_to_apply or "").strip()})
|
||||||
|
# Trimmed BEFORE recording: a rule the cap cut was shown to nobody.
|
||||||
|
out = out[:room]
|
||||||
|
mounted = [item["rule_id"] for item in out]
|
||||||
|
fresh = [rid for rid in mounted if rid not in exclude]
|
||||||
|
if fresh:
|
||||||
|
try:
|
||||||
|
rule_usage.record_rule_surfaced(
|
||||||
|
user_id=user_id, rule_ids=fresh, source=rp.MOMENT_RULE_SOURCE,
|
||||||
|
detail={item["rule_id"]: item["moment"] for item in out
|
||||||
|
if item["rule_id"] in fresh},
|
||||||
|
)
|
||||||
|
except Exception: # noqa: BLE001 - observation never breaks the observed
|
||||||
|
logger.debug("reply mount surfacing not recorded", exc_info=True)
|
||||||
|
|
||||||
|
# The semantic half: the backstop. A mounted rule already holding is
|
||||||
|
# treated as held here, so one rule is never named twice.
|
||||||
|
floor = await floor_for(user_id, rp.REPLY_RULE.source)
|
||||||
|
result = await rp.run_rule_arm(
|
||||||
|
rp.REPLY_RULE,
|
||||||
|
rp.RuleMoment(
|
||||||
|
user_id=user_id, query=_reply_query(reply), project_id=project_id,
|
||||||
|
where="to this reply", checkpoint_where="this reply",
|
||||||
|
exclude=exclude, held=frozenset(skip | set(mounted)),
|
||||||
|
),
|
||||||
|
floor=floor, budget=await budget_for(user_id, rp.REPLY_RULE.source),
|
||||||
|
io=pc._rule_io(), checkpoint_floor=floor,
|
||||||
|
)
|
||||||
|
if result.checkpoint and len(out) < room:
|
||||||
|
cp = result.checkpoint
|
||||||
|
out.append({"rule_id": cp["rule_id"], "title": cp["title"],
|
||||||
|
"score": cp["score"], "trigger": cp.get("trigger", "")})
|
||||||
|
|
||||||
|
if not out:
|
||||||
|
return {}
|
||||||
|
return {
|
||||||
|
"reason": reply_hold_reason(out),
|
||||||
|
"rule_ids": [item["rule_id"] for item in out],
|
||||||
|
"moments": names,
|
||||||
|
}
|
||||||
|
except Exception: # noqa: BLE001 - a recall aid never stops a session by failing
|
||||||
|
logger.debug("reply hold failed", exc_info=True)
|
||||||
|
return {}
|
||||||
|
|||||||
@@ -115,6 +115,7 @@ _RESCORERS = {
|
|||||||
),
|
),
|
||||||
"write_path_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
|
"write_path_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
|
||||||
"pre_tool_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
|
"pre_tool_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
|
||||||
|
"reply_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
|
||||||
"prompt_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
|
"prompt_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
|
||||||
"report_preference": lambda u, q, p: _rescore_rules(u, q, p, "preference"),
|
"report_preference": lambda u, q, p: _rescore_rules(u, q, p, "preference"),
|
||||||
}
|
}
|
||||||
@@ -128,6 +129,7 @@ _CORPUS = {
|
|||||||
"write_path": NoteEmbedding,
|
"write_path": NoteEmbedding,
|
||||||
"write_path_rule": RuleEmbedding,
|
"write_path_rule": RuleEmbedding,
|
||||||
"pre_tool_rule": RuleEmbedding,
|
"pre_tool_rule": RuleEmbedding,
|
||||||
|
"reply_rule": RuleEmbedding,
|
||||||
"prompt_rule": RuleEmbedding,
|
"prompt_rule": RuleEmbedding,
|
||||||
"report_preference": RuleEmbedding,
|
"report_preference": RuleEmbedding,
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -354,6 +354,14 @@ class RuleArm:
|
|||||||
kind: str | None = None
|
kind: str | None = None
|
||||||
"""Ask the ranker for one record kind only ("preference"), or every kind."""
|
"""Ask the ranker for one record kind only ("preference"), or every kind."""
|
||||||
|
|
||||||
|
stop_only: bool = False
|
||||||
|
"""The door shows nothing but the stop (milestone 458's reply moment).
|
||||||
|
|
||||||
|
At the end of a turn there is no context to annotate: a hit either holds
|
||||||
|
the reply for one read or reaches nobody. So only the checkpoint's rule is
|
||||||
|
recorded as surfaced — the rest were ranked, and the call row counts them,
|
||||||
|
but nobody was shown them."""
|
||||||
|
|
||||||
|
|
||||||
# Today's differences, reproduced exactly (milestone 456 step 2). Whether the
|
# Today's differences, reproduced exactly (milestone 456 step 2). Whether the
|
||||||
# prompt arm should band, and whether the act arms should reserve a
|
# prompt arm should band, and whether the act arms should reserve a
|
||||||
@@ -377,8 +385,15 @@ REPORT_PREFERENCE = RuleArm(
|
|||||||
"report_preference", band=False, compact_tail=False, checkpoint=False,
|
"report_preference", band=False, compact_tail=False, checkpoint=False,
|
||||||
preference_slot=False, kind="preference",
|
preference_slot=False, kind="preference",
|
||||||
)
|
)
|
||||||
|
# The backstop for every arm that ran earlier in the turn and missed: the
|
||||||
|
# finished reply against every rule's trigger. Its floor IS its stop bar
|
||||||
|
# (the reply_rule surface), so it is passed as both.
|
||||||
|
REPLY_RULE = RuleArm(
|
||||||
|
"reply_rule", band=False, compact_tail=False, checkpoint=True,
|
||||||
|
preference_slot=False, stop_only=True,
|
||||||
|
)
|
||||||
RULE_ARMS: tuple[RuleArm, ...] = (
|
RULE_ARMS: tuple[RuleArm, ...] = (
|
||||||
WRITE_PATH_RULE, PRE_TOOL_RULE, PROMPT_RULE, REPORT_PREFERENCE,
|
WRITE_PATH_RULE, PRE_TOOL_RULE, PROMPT_RULE, REPORT_PREFERENCE, REPLY_RULE,
|
||||||
)
|
)
|
||||||
|
|
||||||
PREFERENCE_SLOT_SOURCE = "preference_slot"
|
PREFERENCE_SLOT_SOURCE = "preference_slot"
|
||||||
@@ -639,13 +654,6 @@ async def run_rule_arm(
|
|||||||
# ambient: the source is in `rule_usage.RANKED_SOURCES`.
|
# ambient: the source is in `rule_usage.RANKED_SOURCES`.
|
||||||
rule_ids = [rule.id for _score, rule in fresh]
|
rule_ids = [rule.id for _score, rule in fresh]
|
||||||
source = arm.source
|
source = arm.source
|
||||||
if rule_ids:
|
|
||||||
try:
|
|
||||||
io.record_rule_surfaced(
|
|
||||||
user_id=moment.user_id, rule_ids=rule_ids, source=source,
|
|
||||||
)
|
|
||||||
except Exception: # noqa: BLE001 - observation never breaks the observed
|
|
||||||
_telemetry_failed(source)
|
|
||||||
checkpoint = (
|
checkpoint = (
|
||||||
checkpoint_for(
|
checkpoint_for(
|
||||||
kept, held=set(moment.held), floor=checkpoint_floor,
|
kept, held=set(moment.held), floor=checkpoint_floor,
|
||||||
@@ -653,6 +661,15 @@ async def run_rule_arm(
|
|||||||
)
|
)
|
||||||
if arm.checkpoint else {}
|
if arm.checkpoint else {}
|
||||||
)
|
)
|
||||||
|
if arm.stop_only:
|
||||||
|
rule_ids = [rid for rid in rule_ids if rid == checkpoint.get("rule_id")]
|
||||||
|
if rule_ids:
|
||||||
|
try:
|
||||||
|
io.record_rule_surfaced(
|
||||||
|
user_id=moment.user_id, rule_ids=rule_ids, source=source,
|
||||||
|
)
|
||||||
|
except Exception: # noqa: BLE001 - observation never breaks the observed
|
||||||
|
_telemetry_failed(source)
|
||||||
return RuleResult(
|
return RuleResult(
|
||||||
lines=lines, rule_ids=rule_ids,
|
lines=lines, rule_ids=rule_ids,
|
||||||
shown_rule_ids=[rule.id for _score, rule in shown],
|
shown_rule_ids=[rule.id for _score, rule in shown],
|
||||||
|
|||||||
@@ -135,6 +135,9 @@ POINTS: dict[str, Point] = dict([
|
|||||||
_p("write_path_rule", UNBIDDEN, "rules that may govern the file being written"),
|
_p("write_path_rule", UNBIDDEN, "rules that may govern the file being written"),
|
||||||
_p("pre_tool_rule", UNBIDDEN, "rules that may govern a command about to run"),
|
_p("pre_tool_rule", UNBIDDEN, "rules that may govern a command about to run"),
|
||||||
_p("prompt_rule", UNBIDDEN, "rules that may govern what the operator just asked"),
|
_p("prompt_rule", UNBIDDEN, "rules that may govern what the operator just asked"),
|
||||||
|
_p("reply_rule", UNBIDDEN,
|
||||||
|
"a rule that holds the finished reply for one read — the backstop "
|
||||||
|
"for whatever the earlier arms missed"),
|
||||||
_p("preference_slot", UNBIDDEN,
|
_p("preference_slot", UNBIDDEN,
|
||||||
"the one line reserved for a preference at the prompt boundary"),
|
"the one line reserved for a preference at the prompt boundary"),
|
||||||
_p("reuse_slot", UNBIDDEN, "the one line reserved for a reusable snippet"),
|
_p("reuse_slot", UNBIDDEN, "the one line reserved for a reusable snippet"),
|
||||||
|
|||||||
@@ -207,6 +207,21 @@ SURFACES: dict[str, Surface] = {
|
|||||||
over="preferences",
|
over="preferences",
|
||||||
fires="when a task finishes",
|
fires="when a task finishes",
|
||||||
),
|
),
|
||||||
|
# THE REPLY BACKSTOP (milestone 458, folded in from 456 step 8). Its floor
|
||||||
|
# is a STOP bar, not a hint bar: at the end of a turn nothing can be shown
|
||||||
|
# beside the reply, so a hit either holds the reply for one read or says
|
||||||
|
# nothing. Hence a default at the checkpoint's level and a budget of one —
|
||||||
|
# the call row's results are then exactly the rule that would hold.
|
||||||
|
"reply_rule": Surface(
|
||||||
|
name="reply_rule",
|
||||||
|
floor_key="kb_replyrule_threshold",
|
||||||
|
floor_default=0.80,
|
||||||
|
budget_key="kb_replyrule_top_k",
|
||||||
|
budget_default=1,
|
||||||
|
asks="the reply that ends a turn, against rule triggers",
|
||||||
|
over="global rules plus the bound project's own",
|
||||||
|
fires="once per turn, when the reply is finished",
|
||||||
|
),
|
||||||
}
|
}
|
||||||
|
|
||||||
# Reserved slots are deliberately absent. `preference_slot`, `reuse_slot` and
|
# Reserved slots are deliberately absent. `preference_slot`, `reuse_slot` and
|
||||||
|
|||||||
@@ -101,6 +101,10 @@ logger = logging.getLogger(__name__)
|
|||||||
# Add a source here only when a ranker picked it.
|
# Add a source here only when a ranker picked it.
|
||||||
RANKED_SOURCES = (
|
RANKED_SOURCES = (
|
||||||
"write_path_rule", "pre_tool_rule", "prompt_rule",
|
"write_path_rule", "pre_tool_rule", "prompt_rule",
|
||||||
|
# The reply backstop records only the rule that HELD the reply — a claim
|
||||||
|
# put in front of the reader as plainly as any line, and one a pull
|
||||||
|
# confirms or refutes the same way.
|
||||||
|
"reply_rule",
|
||||||
# A rule mounted on a moment (milestone 458). Not a ranker's pick, but
|
# A rule mounted on a moment (milestone 458). Not a ranker's pick, but
|
||||||
# not bulk either: somebody decided this rule applies at this moment, and
|
# not bulk either: somebody decided this rule applies at this moment, and
|
||||||
# that is exactly the claim a pull can confirm or refute. Its
|
# that is exactly the claim a pull can confirm or refute. Its
|
||||||
|
|||||||
@@ -0,0 +1,276 @@
|
|||||||
|
"""The reply moment: the finished reply, checked before it goes out (milestone
|
||||||
|
458 step 4, folded in from milestone 456 step 8).
|
||||||
|
|
||||||
|
The operator's ruling: an unopened rule above the stop bar HOLDS the reply
|
||||||
|
once, in the server's words — the backstop for whatever the earlier arms
|
||||||
|
missed — and the rewrite is never held. These pin, with the database stubbed:
|
||||||
|
- which moments a reply reaches;
|
||||||
|
- what holds: a mounted RULE the session has not opened, or the reply's top
|
||||||
|
semantic hit above the reply surface's bar — never a preference, never a
|
||||||
|
rule already opened or already held once, never past the session cap;
|
||||||
|
- that the stop-only arm records surfacing for the held rule alone;
|
||||||
|
- the hook's contract: blocks only on the server's reason, records the hold
|
||||||
|
before emitting it, and never holds the rewrite.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
from pathlib import Path
|
||||||
|
from unittest.mock import AsyncMock, MagicMock, patch
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from quart import Quart, g
|
||||||
|
|
||||||
|
from scribe.services import moment_delivery as md
|
||||||
|
from scribe.services import plugin_context as pc
|
||||||
|
from scribe.services import retrieval_pipeline as rp
|
||||||
|
from scribe.services import retrieval_surfaces as surfaces
|
||||||
|
from scribe.services import rule_usage, rulebooks
|
||||||
|
from tests.helpers import fake_rule, http_sink, need_tools
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parents[1]
|
||||||
|
HOOK = ROOT / "plugin" / "hooks" / "scribe_reply_check.sh"
|
||||||
|
|
||||||
|
DONE = fake_rule(id=11, title="Definition of done", kind="rule",
|
||||||
|
when_to_apply="about to tell the operator work is finished")
|
||||||
|
|
||||||
|
|
||||||
|
# ── which moments a reply reaches ───────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_every_reply_reports_and_a_question_also_asks():
|
||||||
|
assert [h["moment"] for h in md.reply_moments("Shipped it.")] == ["reply.report"]
|
||||||
|
assert [h["moment"] for h in md.reply_moments("Shipped.\nShall I merge it?")] == [
|
||||||
|
"reply.report", "reply.ask",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_question_inside_code_is_not_the_reply_asking():
|
||||||
|
reply = "Done.\n```python\nok = x if y else z?\n```\n"
|
||||||
|
assert [h["moment"] for h in md.reply_moments(reply)] == ["reply.report"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_long_reply_keeps_its_end():
|
||||||
|
"""The part of a report that asks something of the reader is at its end."""
|
||||||
|
reply = "head " * 400 + "please confirm it works on your end"
|
||||||
|
query = md._reply_query(reply)
|
||||||
|
assert query.startswith("head")
|
||||||
|
assert query.endswith("please confirm it works on your end")
|
||||||
|
assert len(query) < len(reply)
|
||||||
|
|
||||||
|
|
||||||
|
# ── what holds ──────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def _stubs(*, mounted=(), checkpoint=None):
|
||||||
|
arm = AsyncMock(return_value=rp.RuleResult(checkpoint=checkpoint or {}))
|
||||||
|
surfaced = MagicMock()
|
||||||
|
lookup = AsyncMock(return_value=list(mounted))
|
||||||
|
return arm, surfaced, lookup, [
|
||||||
|
patch.object(rulebooks, "rules_on_moments", lookup),
|
||||||
|
patch.object(rule_usage, "record_rule_surfaced", surfaced),
|
||||||
|
patch.object(rp, "run_rule_arm", arm),
|
||||||
|
patch.object(surfaces, "floor_for", AsyncMock(return_value=0.8)),
|
||||||
|
patch.object(surfaces, "budget_for", AsyncMock(return_value=1)),
|
||||||
|
patch.object(pc, "_rule_io", MagicMock()),
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
async def _hold(stack, reply="All done — let me know if it works.", **kw):
|
||||||
|
for p in stack:
|
||||||
|
p.start()
|
||||||
|
try:
|
||||||
|
return await md.reply_hold(1, reply, project_id=2, **kw)
|
||||||
|
finally:
|
||||||
|
for p in reversed(stack):
|
||||||
|
p.stop()
|
||||||
|
|
||||||
|
|
||||||
|
async def test_a_mounted_rule_the_session_has_not_opened_holds_the_reply():
|
||||||
|
_arm, surfaced, lookup, stack = _stubs(mounted=[(DONE, "reply.report")])
|
||||||
|
out = await _hold(stack)
|
||||||
|
assert out["rule_ids"] == [11]
|
||||||
|
assert "Definition of done" in out["reason"] and "get_rule(11)" in out["reason"]
|
||||||
|
assert "mounted on reply.report" in out["reason"]
|
||||||
|
assert lookup.await_args.args == (1, ["reply.report"], 2)
|
||||||
|
kw = surfaced.call_args.kwargs
|
||||||
|
assert kw["source"] == rp.MOMENT_RULE_SOURCE and kw["detail"] == {11: "reply.report"}
|
||||||
|
|
||||||
|
|
||||||
|
async def test_an_opened_rule_a_preference_and_an_earlier_hold_do_not_hold():
|
||||||
|
pref = fake_rule(id=12, title="Lead with the outcome", kind="preference")
|
||||||
|
other = fake_rule(id=13, title="Name the commit", kind="rule")
|
||||||
|
mounted = [(DONE, "reply.report"), (pref, "reply.report"), (other, "reply.report")]
|
||||||
|
_arm, surfaced, _lookup, stack = _stubs(mounted=mounted)
|
||||||
|
assert await _hold(stack, held=frozenset({11}), stopped=frozenset({13})) == {}
|
||||||
|
surfaced.assert_not_called()
|
||||||
|
|
||||||
|
|
||||||
|
async def test_the_semantic_backstop_holds_and_never_names_a_mounted_rule_twice():
|
||||||
|
cp = {"rule_id": 40, "title": "Read the job log first", "score": 0.84,
|
||||||
|
"trigger": "a CI run overran"}
|
||||||
|
arm, _surfaced, _lookup, stack = _stubs(mounted=[(DONE, "reply.report")], checkpoint=cp)
|
||||||
|
out = await _hold(stack)
|
||||||
|
assert out["rule_ids"] == [11, 40]
|
||||||
|
assert "scores 0.84 against this reply" in out["reason"]
|
||||||
|
spec, moment = arm.await_args.args
|
||||||
|
assert spec is rp.REPLY_RULE
|
||||||
|
assert 11 in moment.held, "a rule the mounted half holds must not be raised again"
|
||||||
|
# Its floor IS its stop bar.
|
||||||
|
assert arm.await_args.kwargs["floor"] == arm.await_args.kwargs["checkpoint_floor"] == 0.8
|
||||||
|
|
||||||
|
|
||||||
|
async def test_past_the_session_cap_nothing_holds_and_nothing_is_asked():
|
||||||
|
arm, _surfaced, lookup, stack = _stubs(mounted=[(DONE, "reply.report")])
|
||||||
|
stopped = frozenset(range(100, 100 + pc.CHECKPOINT_SESSION_CAP))
|
||||||
|
assert await _hold(stack, stopped=stopped) == {}
|
||||||
|
lookup.assert_not_awaited()
|
||||||
|
arm.assert_not_awaited()
|
||||||
|
|
||||||
|
|
||||||
|
async def test_the_cap_trims_before_anything_is_recorded():
|
||||||
|
rules = [(fake_rule(id=i, title=f"rule {i}", kind="rule"), "reply.report") for i in range(1, 9)]
|
||||||
|
_arm, surfaced, _lookup, stack = _stubs(mounted=rules)
|
||||||
|
out = await _hold(stack, stopped=frozenset({50, 51}))
|
||||||
|
room = pc.CHECKPOINT_SESSION_CAP - 2
|
||||||
|
assert len(out["rule_ids"]) == room
|
||||||
|
assert surfaced.call_args.kwargs["rule_ids"] == out["rule_ids"]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_the_reply_check_fails_open():
|
||||||
|
_arm, _surfaced, _lookup, stack = _stubs()
|
||||||
|
stack[0] = patch.object(rulebooks, "rules_on_moments", AsyncMock(side_effect=RuntimeError("x")))
|
||||||
|
assert await _hold(stack) == {}
|
||||||
|
assert await md.reply_hold(1, " ") == {}
|
||||||
|
|
||||||
|
|
||||||
|
# ── the stop-only arm ───────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
async def test_the_stop_only_arm_records_only_the_rule_that_holds():
|
||||||
|
hits = [(0.86, fake_rule(id=40, title="a", kind="rule", when_to_apply="t")),
|
||||||
|
(0.82, fake_rule(id=41, title="b", kind="rule", when_to_apply="t"))]
|
||||||
|
io = rp.RuleIO(search=AsyncMock(return_value=hits), record_retrieval=MagicMock(),
|
||||||
|
record_rule_surfaced=MagicMock())
|
||||||
|
moment = rp.RuleMoment(user_id=1, query="the reply", project_id=2,
|
||||||
|
checkpoint_where="this reply")
|
||||||
|
result = await rp.run_rule_arm(rp.REPLY_RULE, moment, floor=0.8, budget=3, io=io,
|
||||||
|
checkpoint_floor=0.8)
|
||||||
|
assert result.checkpoint["rule_id"] == 40
|
||||||
|
assert io.record_rule_surfaced.call_args.kwargs["rule_ids"] == [40]
|
||||||
|
|
||||||
|
# The top hit already opened: nothing holds, so nothing was shown.
|
||||||
|
io.record_rule_surfaced.reset_mock()
|
||||||
|
held = rp.RuleMoment(user_id=1, query="the reply", project_id=2, held=frozenset({40}))
|
||||||
|
result = await rp.run_rule_arm(rp.REPLY_RULE, held, floor=0.8, budget=3, io=io,
|
||||||
|
checkpoint_floor=0.8)
|
||||||
|
assert result.checkpoint == {}
|
||||||
|
io.record_rule_surfaced.assert_not_called()
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_reply_arm_is_a_registered_ranked_surface():
|
||||||
|
from scribe.services.retrieval_registry import POINTS
|
||||||
|
|
||||||
|
assert rp.REPLY_RULE in rp.RULE_ARMS
|
||||||
|
assert "reply_rule" in surfaces.SURFACES and "reply_rule" in POINTS
|
||||||
|
assert "reply_rule" in rule_usage.RANKED_SOURCES
|
||||||
|
# A stop bar, not a hint bar: it sits with the checkpoint's default.
|
||||||
|
assert surfaces.SURFACES["reply_rule"].floor_default >= pc._CHECKPOINT_DEFAULT
|
||||||
|
|
||||||
|
|
||||||
|
# ── the route ───────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
async def test_the_route_passes_the_reply_and_all_three_ledgers():
|
||||||
|
from scribe.routes import plugin as routes
|
||||||
|
|
||||||
|
hold = AsyncMock(return_value={"reason": "Held.", "rule_ids": [11], "moments": ["reply.report"]})
|
||||||
|
app = Quart(__name__)
|
||||||
|
async with app.test_request_context(
|
||||||
|
"/api/plugin/reply-rules", method="POST", json={"reply": "done"},
|
||||||
|
query_string={"project_id": "3", "held_rule_ids": "4", "exclude_rule_ids": "5",
|
||||||
|
"stopped_rule_ids": "6,7"},
|
||||||
|
):
|
||||||
|
g.user = type("U", (), {"id": 7})()
|
||||||
|
with patch.object(routes.moment_delivery_svc, "reply_hold", hold):
|
||||||
|
resp = await routes.reply_rules.__wrapped__()
|
||||||
|
body = await resp.get_json()
|
||||||
|
assert body == {"reason": "Held.", "rule_ids": [11], "moments": ["reply.report"]}
|
||||||
|
assert hold.await_args.args == (7, "done")
|
||||||
|
assert hold.await_args.kwargs == {
|
||||||
|
"project_id": 3, "exclude": frozenset({5}), "held": frozenset({4}),
|
||||||
|
"stopped": frozenset({6, 7}),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# ── the hook ────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def _transcript(tmp_path, reply):
|
||||||
|
lines = [
|
||||||
|
{"type": "user", "message": {"role": "user", "content": "is it finished?"}},
|
||||||
|
{"type": "assistant", "message": {"content": [{"type": "text", "text": reply}]}},
|
||||||
|
]
|
||||||
|
path = tmp_path / "t.jsonl"
|
||||||
|
path.write_text("\n".join(json.dumps(x, separators=(",", ":")) for x in lines) + "\n")
|
||||||
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
def _run(tmp_path, port, transcript, active=False):
|
||||||
|
need_tools("bash", "curl", "awk")
|
||||||
|
env = {"PATH": os.environ["PATH"], "SCRIBE_URL": f"http://127.0.0.1:{port}",
|
||||||
|
"SCRIBE_TOKEN": "t", "TMPDIR": str(tmp_path), "HOME": str(tmp_path)}
|
||||||
|
out = subprocess.run(
|
||||||
|
["bash", str(HOOK)],
|
||||||
|
input=json.dumps({"session_id": "s-reply", "transcript_path": str(transcript),
|
||||||
|
"cwd": str(tmp_path), "hook_event_name": "Stop",
|
||||||
|
"stop_hook_active": active}),
|
||||||
|
capture_output=True, text=True, env=env, timeout=30,
|
||||||
|
)
|
||||||
|
assert out.returncode == 0, out.stderr
|
||||||
|
return out.stdout.strip()
|
||||||
|
|
||||||
|
|
||||||
|
HELD = json.dumps({"reason": "Held for one read: get_rule(11).", "rule_ids": [11],
|
||||||
|
"moments": ["reply.report"]}).encode()
|
||||||
|
QUIET = b'{"reason":"","rule_ids":[],"moments":["reply.report"]}'
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_hook_blocks_in_the_servers_words_and_records_the_hold(tmp_path):
|
||||||
|
t = _transcript(tmp_path, "All done.\nLet me know if it works on your end?")
|
||||||
|
with http_sink(by_path={"/api/plugin/reply-rules": HELD}) as (port, seen):
|
||||||
|
out = json.loads(_run(tmp_path, port, t))
|
||||||
|
_run(tmp_path, port, t)
|
||||||
|
assert out == {"decision": "block", "reason": "Held for one read: get_rule(11)."}
|
||||||
|
sent = json.loads(seen[0]["_body"])
|
||||||
|
assert sent["reply"] == "All done.\nLet me know if it works on your end?"
|
||||||
|
# The second turn tells the server what already held.
|
||||||
|
assert seen[1]["stopped_rule_ids"] == ["11"]
|
||||||
|
assert seen[1]["exclude_rule_ids"] == ["11"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_hook_says_nothing_when_nothing_holds(tmp_path):
|
||||||
|
t = _transcript(tmp_path, "Done.")
|
||||||
|
with http_sink(by_path={"/api/plugin/reply-rules": QUIET}) as (port, _seen):
|
||||||
|
assert _run(tmp_path, port, t) == ""
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_rewrite_is_never_held(tmp_path):
|
||||||
|
t = _transcript(tmp_path, "Done.")
|
||||||
|
with http_sink(by_path={"/api/plugin/reply-rules": HELD}) as (port, seen):
|
||||||
|
assert _run(tmp_path, port, t, active=True) == ""
|
||||||
|
assert seen == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_an_unreachable_instance_never_holds_a_reply(tmp_path):
|
||||||
|
t = _transcript(tmp_path, "Done.")
|
||||||
|
assert _run(tmp_path, 9, t) == ""
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("name", ["scribe_reply_check.sh"])
|
||||||
|
def test_the_hook_is_registered_on_stop(name):
|
||||||
|
manifest = json.loads((ROOT / "plugin" / "hooks" / "hooks.json").read_text())
|
||||||
|
commands = [h["command"] for m in manifest["hooks"]["Stop"] for h in m["hooks"]]
|
||||||
|
assert any(name in c for c in commands)
|
||||||
@@ -54,6 +54,7 @@ _SURFACE_PAIRS = (
|
|||||||
("pre_tool_rule", "kbToolRuleThreshold"),
|
("pre_tool_rule", "kbToolRuleThreshold"),
|
||||||
("prompt_rule", "kbPromptRuleThreshold"),
|
("prompt_rule", "kbPromptRuleThreshold"),
|
||||||
("report_preference", "kbReportPrefThreshold"),
|
("report_preference", "kbReportPrefThreshold"),
|
||||||
|
("reply_rule", "kbReplyRuleThreshold"),
|
||||||
)
|
)
|
||||||
|
|
||||||
# (services module, python constant, vue ref) for the defaults that are NOT
|
# (services module, python constant, vue ref) for the defaults that are NOT
|
||||||
|
|||||||
Reference in New Issue
Block a user