Reply shapes ride the moments (milestone 500, steps 1–5) #209

Merged
bvandeusen merged 13 commits from dev into main 2026-10-09 16:40:04 -04:00
65 changed files with 2733 additions and 1243 deletions
+12 -2
View File
@@ -296,8 +296,18 @@ jobs:
set -eux
echo "=== container landscape (diagnostic for the name filter) ==="
docker ps -a --format '{{.ID}} {{.Image}} -> {{.Names}}'
PG=$(docker ps --filter "name=integration" --filter "ancestor=pgvector/pgvector:pg17" -q | head -n1)
test -n "$PG"
# Only THIS job's services. The runner's docker daemon is shared, so another
# repo's job can be running beside this one, and a job-name filter matches its
# containers too (Steward run 8358 and Inkwell run 8653 each took another repo's
# Postgres). act names every container of a task GITEA-ACTIONS-TASK-<n>-...: read
# <n> from our own container (its id is in the /etc/hostname bind mount) and
# require exactly one match.
SELF=$(grep -o '/containers/[0-9a-f]\{64\}/' /proc/self/mountinfo | head -n1 | cut -d/ -f3 || true)
SELF=${SELF:-$(hostname)}
TASK=$(docker inspect -f '{{.Name}}' "$SELF" | grep -o 'GITEA-ACTIONS-TASK-[0-9]*-')
test -n "$TASK"
PG=$(docker ps --filter "name=$TASK" --filter "ancestor=pgvector/pgvector:pg17" -q)
test "$(printf '%s\n' "$PG" | grep -c .)" -eq 1
PG_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$PG")
test -n "$PG_IP"
export DATABASE_URL="postgresql+asyncpg://scribe:ci_integration@${PG_IP}:5432/scribe_test"
@@ -0,0 +1,59 @@
"""retire the report_preference arm's rows (milestone 500 step 4)
Revision ID: 0121
Revises: 0120
Create Date: 2026-10-09
The fixed-question arm that searched for completion-report preferences when a
task closed is gone: the reply shapes and the preferences mounted beside them
now arrive on the moments (milestone 500 step 3). Its rows go with it (rule
22 — no legacy to preserve), because each one would otherwise mean something
different once the arm is no longer registered:
- `retrieval_logs` / `retrieval_judgments` — a source missing from the
registry reads as `unregistered_source`, whose advice is "add it to the
registry": the opposite of what happened.
- `rule_usage_events` — a source missing from `RANKED_SOURCES` reads as
ambient, so its surfacings would silently move into the other column of
every rule's pull-through.
- `retrieval_tuning_events` — the tuning history of a surface that no longer
exists.
- `settings` — the arm's floor and budget keys, which nothing reads.
COST IS BOUNDED BY AN INDEX, NOT BY THE TABLE. A migration runs at boot
against the live tables, which CI never has (lesson #5221). `retrieval_logs`
and `retrieval_judgments` are indexed on `source`. `rule_usage_events` is not,
so its delete is also bounded by `created_at` (indexed): the arm shipped with
milestone 409 step 4, decided 2026-09-14, so no row of it predates
`ARM_BORN`, and the scan covers weeks rather than the table's whole history.
Not reversible: the arm is not coming back, so neither are its rows.
"""
from alembic import op
revision = "0121"
down_revision = "0120"
branch_labels = None
depends_on = None
SOURCE = "report_preference"
# A safe floor under the arm's first row (it shipped after 2026-09-14).
ARM_BORN = "2026-09-01"
def upgrade() -> None:
op.execute(f"DELETE FROM retrieval_logs WHERE source = '{SOURCE}'")
op.execute(f"DELETE FROM retrieval_judgments WHERE source = '{SOURCE}'")
op.execute(
f"DELETE FROM rule_usage_events WHERE created_at >= '{ARM_BORN}' "
f"AND source = '{SOURCE}'"
)
op.execute(f"DELETE FROM retrieval_tuning_events WHERE surface = '{SOURCE}'")
op.execute(
"DELETE FROM settings WHERE key IN "
"('kb_reportpref_threshold', 'kb_reportpref_top_k')"
)
def downgrade() -> None:
pass
+47
View File
@@ -0,0 +1,47 @@
import { apiGet } from "@/api/client";
/**
* The reply shapes Scribe ships (milestone 500) — `GET /api/retrieval/reply-shapes`.
* The same catalog the `list_reply_shapes` MCP tool returns, plus what only
* the operator's view needs: the preferences mounted on each shape's moment,
* and how often each shape was delivered.
*/
export interface ShapePreference {
id: number;
title: string;
statement: string;
}
/** How often a shape went out: whole, or as its one-line reminder. */
export interface ShapeDeliveries {
full: number;
pointer: number;
}
export interface ReplyShape {
key: string;
title: string;
/** The moment it rides — a preference mounted here adjusts it. */
moment: string;
/** What is happening at that moment. */
means: string;
/** When the shape arrives, in words. */
delivered: string;
/** The shipped text. Product, so read-only. */
text: string;
preferences: ShapePreference[];
/** null when the counts could not be read — unknown, not zero. */
deliveries: ShapeDeliveries | null;
}
export interface ReplyShapesOverview {
shapes: ReplyShape[];
total: number;
days: number;
deliveries_failed: boolean;
}
export async function getReplyShapes(days?: number): Promise<ReplyShapesOverview> {
const q = days ? `?days=${days}` : "";
return apiGet<ReplyShapesOverview>(`/api/retrieval/reply-shapes${q}`);
}
+9
View File
@@ -54,6 +54,15 @@ export interface RulebookTopic {
*/
export type RuleKind = "rule" | "preference";
/** A new record opened from somewhere that already knows what it is for —
* "Adjust this" on a reply shape opens a preference mounted on that shape's
* moment (milestone 500 step 5). The editor applies it once, when creating. */
export interface RulePreset {
kind?: RuleKind;
moments?: string[];
whenToApply?: string;
}
export interface Rule {
id: number;
topic_id: number | null;
@@ -0,0 +1,226 @@
<script setup lang="ts">
import { computed, onMounted, ref } from "vue";
import { apiErrorMessage } from "@/api/client";
import { getReplyShapes, type ReplyShape, type ReplyShapesOverview } from "@/api/replyShapes";
import type { RulePreset } from "@/api/rulebooks";
import RuleEditorSlideOver from "@/components/rules/RuleEditorSlideOver.vue";
/**
* The reply shapes Scribe ships, and the operator's adjustments to each
* (milestone 500 step 5). A shape is product, so its text is read-only here;
* what the operator owns is the preferences mounted on the shape's moment,
* which arrive beside it and win where they differ. "Adjust this" opens the
* preference editor already mounted there.
*/
const data = ref<ReplyShapesOverview | null>(null);
const loading = ref(true);
const error = ref("");
const open = ref<string | null>(null);
async function load() {
loading.value = true;
error.value = "";
try {
data.value = await getReplyShapes();
} catch (e: unknown) {
error.value = apiErrorMessage(e, "Could not load the reply shapes");
} finally {
loading.value = false;
}
}
const shapes = computed(() => data.value?.shapes ?? []);
const adjusted = computed(() => shapes.value.filter((s) => s.preferences.length).length);
const delivered = computed(() =>
shapes.value.reduce((n, s) => n + (s.deliveries ? s.deliveries.full + s.deliveries.pointer : 0), 0),
);
// One line answering "what do my replies get shaped by?".
const verdict = computed(() => {
if (!data.value) return "";
const n = shapes.value.length;
const parts = [`${n} ${n === 1 ? "shape ships" : "shapes ship"} with Scribe`];
parts.push(adjusted.value
? `${adjusted.value} adjusted by your preferences`
: "none adjusted — the defaults stand");
if (!data.value.deliveries_failed) parts.push(`delivered ${delivered.value} times in ${data.value.days} days`);
return parts.join(" · ");
});
function deliveryLabel(s: ReplyShape): string {
if (!s.deliveries) return "—";
const { full, pointer } = s.deliveries;
if (!full && !pointer) return "not delivered";
return `${full} in full · ${pointer} as reminder`;
}
function toggle(key: string) {
open.value = open.value === key ? null : key;
}
// ── The editor: an existing preference, or a new one mounted here ─────────
const editingId = ref<number | null>(null);
const preset = ref<RulePreset | null>(null);
const editorOpen = computed(() => editingId.value !== null || preset.value !== null);
function editPreference(id: number) {
preset.value = null;
editingId.value = id;
}
function adjust(s: ReplyShape) {
editingId.value = null;
preset.value = {
kind: "preference",
moments: [s.moment],
// A starting trigger in the moment's own words; the operator sharpens it.
whenToApply: `When ${s.means}.`,
};
}
function closeEditor() {
editingId.value = null;
preset.value = null;
void load();
}
onMounted(load);
</script>
<template>
<div class="reply-shapes">
<p v-if="loading && !data" class="state">Loading…</p>
<p v-else-if="error" class="state state-error" role="alert">
{{ error }}
<button type="button" class="btn-text" @click="load">Retry</button>
</p>
<template v-else-if="data">
<p class="verdict">{{ verdict }}</p>
<p v-if="data.deliveries_failed" class="state state-error" role="alert">
Delivery counts could not be read, so they show as “—”.
<button type="button" class="btn-text" @click="load">Retry</button>
</p>
<ul class="shape-list">
<li v-for="s in shapes" :key="s.key" :class="['shape-row', { open: open === s.key }]">
<button
type="button"
class="shape-head"
:aria-expanded="open === s.key"
:aria-controls="`shape-${s.key}`"
@click="toggle(s.key)"
>
<span class="shape-name">
<strong>{{ s.title }}</strong>
<span class="shape-when">{{ s.delivered }}</span>
</span>
<span class="shape-stats">
<span :class="{ 'stat-none': !s.preferences.length }">
{{ s.preferences.length
? `${s.preferences.length} ${s.preferences.length === 1 ? "preference" : "preferences"}`
: "default" }}
</span>
<span :class="{ 'stat-none': deliveryLabel(s) === 'not delivered' }">{{ deliveryLabel(s) }}</span>
<span class="chevron" aria-hidden="true">{{ open === s.key ? "▾" : "▸" }}</span>
</span>
</button>
<div v-if="open === s.key" :id="`shape-${s.key}`" class="shape-body">
<h4 class="body-label">Ships with Scribe · at <code>{{ s.moment }}</code></h4>
<div class="shape-text">{{ s.text }}</div>
<h4 class="body-label">Your adjustments</h4>
<ul v-if="s.preferences.length" class="pref-list">
<li v-for="p in s.preferences" :key="p.id">
<button type="button" class="pref-row" @click="editPreference(p.id)">
<span class="pref-title">{{ p.title }}</span>
<span class="chevron" aria-hidden="true">›</span>
</button>
</li>
</ul>
<p v-else class="stat-none">None — this shape arrives as shipped.</p>
<button type="button" class="btn-primary adjust" @click="adjust(s)">Adjust this</button>
</div>
</li>
</ul>
</template>
<RuleEditorSlideOver
v-if="editorOpen"
:rule-id="editingId"
:topic-id="null"
:preset="preset"
@close="closeEditor"
/>
</div>
</template>
<style scoped>
.reply-shapes { display: flex; flex-direction: column; gap: var(--fs-space-2); }
.verdict { margin: 0; font-size: var(--fs-size-body-sm); color: var(--fs-text-primary); }
.state { margin: 0; font-size: var(--fs-size-body-sm); color: var(--fs-text-secondary); }
.state-error { color: var(--fs-error); }
.shape-list { list-style: none; margin: 0; padding: 0; display: flex; flex-direction: column; gap: var(--fs-space-2); }
.shape-row {
background: var(--fs-surface-page);
border: 1px solid var(--fs-border-color);
border-radius: var(--fs-radius-md);
}
/* The whole row is the tap target (a settings row, not prose with a link). */
.shape-head {
width: 100%;
display: flex; align-items: baseline; gap: var(--fs-space-3);
padding: var(--fs-space-3);
background: none; border: 0; border-radius: var(--fs-radius-md);
color: inherit; font: inherit; text-align: left; cursor: pointer;
}
.shape-head:focus-visible { outline: 2px solid var(--fs-focus-ring); outline-offset: 2px; }
.shape-name { display: flex; flex-direction: column; gap: var(--fs-space-1); min-width: 0; }
.shape-name strong { font-size: var(--fs-size-body-sm); color: var(--fs-text-primary); }
.shape-when {
font-size: var(--fs-size-tiny); color: var(--fs-text-tertiary);
overflow: hidden; text-overflow: ellipsis; white-space: nowrap;
}
.shape-stats {
margin-left: auto;
display: flex; align-items: baseline; gap: var(--fs-space-3);
font-size: var(--fs-size-tiny); color: var(--fs-text-secondary);
font-variant-numeric: tabular-nums; white-space: nowrap;
}
.stat-none { color: var(--fs-text-tertiary); font-style: italic; }
.chevron { color: var(--fs-text-tertiary); }
.shape-body {
display: flex; flex-direction: column; align-items: flex-start; gap: var(--fs-space-2);
padding: 0 var(--fs-space-3) var(--fs-space-3);
}
.body-label {
margin: var(--fs-space-2) 0 0;
font-size: var(--fs-size-tiny); font-weight: normal;
text-transform: uppercase; letter-spacing: var(--fs-tracking-tiny);
color: var(--fs-text-tertiary);
}
.body-label code { font-family: var(--fs-font-mono); text-transform: none; }
/* Product text, shown as it reaches a session. Read-only by design. */
.shape-text {
width: 100%;
padding: var(--fs-space-2) var(--fs-space-3);
border-left: 2px solid var(--fs-border-color);
font-size: var(--fs-size-body-sm); line-height: var(--fs-leading-body);
color: var(--fs-text-secondary); white-space: pre-wrap;
}
.pref-list { list-style: none; margin: 0; padding: 0; width: 100%; display: flex; flex-direction: column; gap: var(--fs-space-1); }
.pref-row {
width: 100%;
display: flex; align-items: baseline; gap: var(--fs-space-2);
padding: var(--fs-space-1) var(--fs-space-2);
background: none; border: 1px solid var(--fs-border-color); border-radius: var(--fs-radius-sm);
color: var(--fs-text-primary); font: inherit; font-size: var(--fs-size-body-sm);
text-align: left; cursor: pointer;
}
.pref-row:focus-visible { outline: 2px solid var(--fs-focus-ring); outline-offset: 2px; }
.pref-title { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
.pref-row .chevron { margin-left: auto; }
.adjust { margin-top: var(--fs-space-1); }
</style>
@@ -5,9 +5,14 @@ import { useCanonicalSystemsStore } from "@/stores/canonicalSystems";
import { useMomentsStore } from "@/stores/moments";
import RuleHistoryPanel from "@/components/rules/RuleHistoryPanel.vue";
import RuleHomePicker from "@/components/rules/RuleHomePicker.vue";
import type { Rule, RuleKind } from "@/api/rulebooks";
import { apiErrorMessage } from "@/api/client";
import { listRulebooks, listTopics, type Rule, type RuleKind, type RulePreset } from "@/api/rulebooks";
const props = defineProps<{ ruleId: number | null; topicId: number | null }>();
const props = defineProps<{
ruleId: number | null;
topicId: number | null;
preset?: RulePreset | null;
}>();
const emit = defineEmits<{ close: [] }>();
const store = useRulebooksStore();
@@ -128,10 +133,10 @@ async function load() {
} else {
title.value = "";
statement.value = "";
whenToApply.value = "";
kind.value = "rule";
whenToApply.value = props.preset?.whenToApply ?? "";
kind.value = props.preset?.kind ?? "rule";
systemIds.value = [];
ruleMoments.value = [];
ruleMoments.value = [...(props.preset?.moments ?? [])];
why.value = "";
howToApply.value = "";
verifyWith.value = "";
@@ -139,7 +144,31 @@ async function load() {
}
procedureDraft.value = "";
procedureError.value = "";
await Promise.all([canon.fetchCatalog(), momentsStore.load()]);
await Promise.all([canon.fetchCatalog(), momentsStore.load(), loadHomes()]);
}
// ── Where a new record lives, when the opener did not say ──────────────────
// The rules view opens the editor inside a topic; Settings does not, so a new
// record opened there asks for its home. A rulebook topic is global — it
// applies in every project, as a reply shape does.
const homeChoices = ref<{ id: number; label: string }[]>([]);
const chosenTopic = ref<number | null>(null);
const homeError = ref("");
const needsHome = computed(() => isCreating.value && props.topicId === null);
async function loadHomes() {
if (!needsHome.value || homeChoices.value.length) return;
try {
const books = await listRulebooks();
const perBook = await Promise.all(books.map(async (rb) => {
const ts = await listTopics(rb.id);
return ts.map((t) => ({ id: t.id, label: `${rb.title} › ${t.title}` }));
}));
homeChoices.value = perBook.flat();
if (homeChoices.value.length === 1) chosenTopic.value = homeChoices.value[0].id;
} catch (e: unknown) {
homeError.value = apiErrorMessage(e, "Could not load where this could live");
}
}
async function save() {
@@ -166,8 +195,14 @@ async function save() {
verify_with: verifyWith.value,
expires_when: expiresWhen.value,
};
if (isCreating.value && props.topicId !== null) {
await store.createRule(props.topicId, fields);
const home = props.topicId ?? chosenTopic.value;
if (isCreating.value && home === null) {
// Written but homeless: keep the draft open rather than lose it on close.
homeError.value = "Choose where it lives before saving.";
return;
}
if (isCreating.value && home !== null) {
await store.createRule(home, fields);
} else if (props.ruleId !== null) {
await store.updateRule(props.ruleId, fields);
}
@@ -201,6 +236,14 @@ watch(() => props.ruleId, load);
<button v-if="!isCreating" class="trash" @click="remove" aria-label="Delete">🗑</button>
<button class="close" @click="save" aria-label="Close">×</button>
</header>
<label v-if="needsHome">
Where it lives
<select v-model="chosenTopic" class="fs-input">
<option :value="null" disabled>Choose a rulebook topic</option>
<option v-for="h in homeChoices" :key="h.id" :value="h.id">{{ h.label }}</option>
</select>
</label>
<p v-if="homeError" class="home-error" role="alert">{{ homeError }}</p>
<label>
Title
<input v-model="title" placeholder="e.g. dev is home" />
@@ -421,6 +464,7 @@ watch(() => props.ruleId, load);
one index everywhere (tests/test_frontend_shared_styles.py). -->
<style src="@/assets/rules-shared.css" />
<style scoped>
.home-error { margin: 0 0 var(--fs-space-2); color: var(--fs-error); font-size: var(--fs-size-body-sm); }
.kind { border: 1px solid var(--fs-border-color); border-radius: var(--fs-radius-md); padding: var(--fs-space-3); margin: var(--fs-space-3) 0; }
.kind legend { font-size: var(--fs-size-tiny); text-transform: uppercase; letter-spacing: var(--fs-tracking-tiny); color: var(--fs-text-tertiary); padding: 0 var(--fs-space-2); }
.kind-opt { display: flex; align-items: flex-start; gap: var(--fs-space-2); margin-bottom: var(--fs-space-2); font-size: var(--fs-size-body-sm); line-height: var(--fs-leading-body); }
+10 -55
View File
@@ -10,6 +10,7 @@ import type { User } from "@/types/auth";
import PaginationBar from "@/components/PaginationBar.vue";
import TagInput from "@/components/TagInput.vue";
import MomentsSettings from "@/components/MomentsSettings.vue";
import ReplyShapesSettings from "@/components/ReplyShapesSettings.vue";
import { fmtDate, fmtLogStamp } from "@/utils/dateFormat";
import { fetchVersion, type VersionPayload } from "@/api/version";
@@ -168,7 +169,6 @@ const kbToolRuleThreshold = ref("0.68");
// And the prompt boundary is a third query shape again — the operator's own
// prose rather than anything a tool produced (#3852).
const kbPromptRuleThreshold = ref("0.72");
const kbReportPrefThreshold = ref("0.72");
// The reply arm (milestone 458): its bar is the bar at which a finished reply
// is HELD for one read, so it sits with the checkpoint, not with the hints.
const kbReplyRuleThreshold = ref("0.8");
@@ -181,7 +181,6 @@ const kbWritePathTopK = ref("3");
const kbRuleHintTopK = ref("5");
const kbToolRuleTopK = ref("5");
const kbPromptRuleTopK = ref("3");
const kbReportPrefTopK = ref("3");
const kbReplyRuleTopK = ref("1");
// What has been changed about retrieval, newest first — the review surface for
// changes the model made on the operator's behalf (#4102).
@@ -387,7 +386,6 @@ async function saveKbInject() {
// this is the only number that can STOP a call, so a fallback of 0 would
// hold the first command of every session behind whatever ranked first.
const cpT = asBar(kbCheckpointThreshold.value, 0.8);
const rpT = asBar(kbReportPrefThreshold.value, 0.72);
// A stop bar like the checkpoint's: a fallback of 0 would hold every reply.
const ryT = asBar(kbReplyRuleThreshold.value, 0.8);
// The budgets, clamped the way the server clamps them: a whole number in
@@ -400,13 +398,11 @@ async function saveKbInject() {
const rhK = asK(kbRuleHintTopK.value, 5);
const trK = asK(kbToolRuleTopK.value, 5);
const prK = asK(kbPromptRuleTopK.value, 3);
const rpK = asK(kbReportPrefTopK.value, 3);
const ryK = asK(kbReplyRuleTopK.value, 1);
kbWritePathTopK.value = String(wpK);
kbRuleHintTopK.value = String(rhK);
kbToolRuleTopK.value = String(trK);
kbPromptRuleTopK.value = String(prK);
kbReportPrefTopK.value = String(rpK);
kbReplyRuleTopK.value = String(ryK);
kbInjectThreshold.value = String(t);
kbInjectTopK.value = String(k);
@@ -424,7 +420,6 @@ async function saveKbInject() {
kbCheckpointThreshold.value = String(cpT);
kbToolRuleThreshold.value = String(trT);
kbPromptRuleThreshold.value = String(prT);
kbReportPrefThreshold.value = String(rpT);
kbReplyRuleThreshold.value = String(ryT);
savingKbInject.value = true;
kbInjectSaved.value = false;
@@ -453,12 +448,6 @@ async function saveKbInject() {
// queries are different shapes. Moving one must not move the others.
kb_toolrule_threshold: String(trT),
kb_promptrule_threshold: String(prT),
// A SIXTH, and the one that most needed its own key: this arm's query
// is a fixed string, so its score is a constant for a given corpus.
// While it borrowed the prompt bar, tuning prose silently retuned it —
// and a constant that lands under the bar is a dead arm, not a quiet
// one (#3860).
kb_reportpref_threshold: String(rpT),
// The reply backstop's own bar: it HOLDS a reply, so like the
// checkpoint it is never derived from a hint bar.
kb_replyrule_threshold: String(ryT),
@@ -470,7 +459,6 @@ async function saveKbInject() {
kb_rulehint_top_k: String(rhK),
kb_toolrule_top_k: String(trK),
kb_promptrule_top_k: String(prK),
kb_reportpref_top_k: String(rpK),
kb_replyrule_top_k: String(ryK),
kb_duplicate_threshold_snippet: String(dupSnip),
kb_duplicate_threshold_note: String(dupNote),
@@ -941,9 +929,6 @@ onMounted(async () => {
if (allSettings.kb_promptrule_threshold !== undefined) {
kbPromptRuleThreshold.value = allSettings.kb_promptrule_threshold;
}
if (allSettings.kb_reportpref_threshold !== undefined) {
kbReportPrefThreshold.value = allSettings.kb_reportpref_threshold;
}
if (allSettings.kb_replyrule_threshold !== undefined) {
kbReplyRuleThreshold.value = allSettings.kb_replyrule_threshold;
}
@@ -967,9 +952,6 @@ onMounted(async () => {
if (allSettings.kb_promptrule_top_k !== undefined) {
kbPromptRuleTopK.value = allSettings.kb_promptrule_top_k;
}
if (allSettings.kb_reportpref_top_k !== undefined) {
kbReportPrefTopK.value = allSettings.kb_reportpref_top_k;
}
if (allSettings.kb_replyrule_top_k !== undefined) {
kbReplyRuleTopK.value = allSettings.kb_replyrule_top_k;
}
@@ -1974,42 +1956,6 @@ async function deleteUser(userId: number) {
/>
<p class="field-hint">How many rules or preferences one message may be shown (1–10).</p>
</div>
<div class="field">
<label for="kb-reportpref-threshold">Completion-report confidence threshold (0–1)</label>
<input
id="kb-reportpref-threshold"
v-model="kbReportPrefThreshold"
type="number"
min="0"
max="1"
step="0.01"
class="fs-input input"
style="max-width: 8rem"
/>
<p class="field-hint">
The bar for a <em>preference about how a completion report should be
written</em>, looked up when a task closes. Unlike every other bar
here, the question this arm asks never changes — so its score is
fixed by your preferences alone, and it will either always find one
or never find one, and no run of calls will reveal a dead one on its
own. That is why this arm is worth looking up in the panel below when
a report preference never seems to arrive.
</p>
</div>
<div class="field">
<label for="kb-reportpref-topk">Max report preferences</label>
<input
id="kb-reportpref-topk"
v-model="kbReportPrefTopK"
type="number"
min="1"
max="10"
step="1"
class="fs-input input"
style="max-width: 8rem"
/>
<p class="field-hint">How many preferences a finished task may be shown (1–10).</p>
</div>
<div class="field">
<label for="kb-replyrule-threshold">Reply check threshold (0–1)</label>
<input
@@ -2305,6 +2251,15 @@ async function deleteUser(userId: number) {
<MomentsSettings />
</section>
<section class="settings-section full-width">
<h2>Reply shapes</h2>
<p class="section-desc">
The default shape of each kind of reply, delivered to a session when it is due.
Adjust one with a preference; it arrives beside the shape and wins.
</p>
<ReplyShapesSettings />
</section>
</div>
<!-- ── Account ── -->
+1 -1
View File
@@ -1,7 +1,7 @@
{
"name": "scribe",
"description": "Scribe for Claude Code: connects the scribe MCP server, adds the hooks that deliver live project state and relevant records at the right moment, ships the shared client-neutral Scribe skills (using-scribe, writing-plans, reporting-back, systematic-debugging, verification, brainstorming, reusing-code, shape-accounting, family-canon), and syncs your saved Scribe Processes as skills (/scribe:sync).",
"version": "2026.10.06.1827",
"version": "2026.10.09.1955",
"author": {
"name": "Bryan Van Deusen"
},
+2 -3
View File
@@ -15,7 +15,7 @@ another one means adding files, not moving or rewriting any.
|---|---|---|
| **The skills** | `plugin/skills/*/SKILL.md` | Agent Skills (the open SKILL.md format). They state every Scribe reflex in full and name no client. `tests/test_guidance_ownership.py` fails if a skill names a particular client, or references anything outside its own folder. Every client package ships this folder verbatim. |
| **The MCP server** | `<base URL>/mcp` | HTTP, `Authorization: Bearer <fmcp_ key>`. Its `_INSTRUCTIONS` is a client-neutral orientation — the workflow across tools (≤1,600 chars, #4389); each tool's description carries its contract; in-band responses (`placement`, `report_back`, `systems_hint`, the duplicate gate, the guessed-id refusal) fire in every client. |
| **The adapter API** | `<base URL>/api/plugin/*` | Plain `GET` endpoints any client's hooks can call with the same key (read scope is enough): `context` (live session state), `retrieve` (rules, preferences and notes for a message), `prior-art` (records and shape-ledger hints for code being written), `tool-rules` (rules for a command about to run), `report-check` (records a completion-report check and returns the reason for a block), `processes` (stored Processes to expose as skills). |
| **The adapter API** | `<base URL>/api/plugin/*` | Plain `GET` endpoints any client's hooks can call with the same key (read scope is enough): `context` (live session state), `retrieve` (rules, preferences and notes for a message), `prior-art` (records and shape-ledger hints for code being written), `tool-rules` (rules for a command about to run), `reply-rules` (a `POST`: the finished reply, held for an unopened rule or — when the turn closed a task — for missing completion sections), `processes` (stored Processes to expose as skills). |
| **The API key** | Scribe → Settings → API Keys | One `fmcp_` key per install. Read scope for hooks; write scope for the MCP tools. |
## Added by each client
@@ -40,8 +40,7 @@ another one means adding files, not moving or rewriting any.
| `hooks/scribe_after_write.sh` | PostToolUse on shell commands: the same check for code written through the shell. |
| `hooks/scribe_tool_rules.sh` | PreToolUse on shell commands: `GET /api/plugin/tool-rules`. |
| `hooks/scribe_moment.sh` | PreToolUse on every tool: the rules mounted on the moments the call reaches, `POST /api/plugin/moment`; skips tools `GET /api/plugin/moment-tools` says reach nothing mounted, and Scribe's own (their responses carry `moment_rules`). |
| `hooks/scribe_report_check.sh` | Stop: when the turn closed a task, checks the reply for the completion sections and reports to `GET /api/plugin/report-check`; blocks once, with the reason the server returns. |
| `hooks/scribe_reply_check.sh` | Stop: sends the finished reply to `POST /api/plugin/reply-rules` — the rules mounted on the reply moments, and the reply against every rule's trigger; holds once per rule, with the reason the server returns, and never holds the rewrite. |
| `hooks/scribe_reply_check.sh` | Stop: sends the finished reply to `POST /api/plugin/reply-rules` — the rules mounted on the reply moments, and the reply against every rule's trigger, and — when the turn closed a task — the completion-section check; holds once per rule, with the reason the server returns, and never holds the rewrite. |
| `hooks/scribe_shape_check.sh` | Stop: sends the definitions the turn wrote (the write hooks' `<sid>.written.ids` ledger) to `GET /api/plugin/shape-check`; blocks once, with the reason the server returns, so the agent judges what it built. |
| `hooks/scribe_sync_processes.sh` + `commands/sync.md` | `GET /api/plugin/processes` → `~/.claude/skills/scribe-proc-*` stubs; `/scribe:sync` on demand. |
| `hooks/scribe_defs.sh` | Shared shell helpers: config, dedup ledgers, outage line. |
+10 -9
View File
@@ -82,15 +82,16 @@ On install you'll be asked for:
answer" line (8 s budget here — it runs after the tool, so it gates
nothing). The extractor, the prose/data skip list, the local by-name
duplicate arm and the outage line are shared in `hooks/scribe_defs.sh`.
- `hooks/hooks.json` → Stop hook (`hooks/scribe_report_check.sh`): when the
turn closed a Scribe task (`update_task`/`create_task` with status done),
checks the reply that ends it for the completion sections — where the work
sits, what needs you, what comes next — and reports the outcome to
`GET /api/plugin/report-check`. If sections are missing it blocks once with
the reason the server returns, and records how the rewrite came out; it
never blocks twice, and never blocks when the instance did not record the
check (unconfigured or unreachable). Outcomes land in the admin logs under
category `plugin`, action `report_check`.
- `hooks/hooks.json` → Stop hook (`hooks/scribe_reply_check.sh`): sends the
reply that ends the turn to `POST /api/plugin/reply-rules`, which holds it
once for an unopened rule mounted on the reply moments or matching the reply.
When the turn closed a Scribe task (`update_task`/`create_task` with status
done), the same request has the server check the reply for the completion
sections — where the work sits, what needs you, what comes next. If they are
missing it blocks once with the reason the server returns, and records how
the rewrite came out; it never blocks twice, and never blocks when the
instance did not record the check (unconfigured or unreachable). Outcomes
land in the admin logs under category `plugin`, action `report_check`.
- `hooks/hooks.json` → a second Stop hook (`hooks/scribe_shape_check.sh`): the
write hooks note every definition a write names (and every new file) in a
session ledger; at the end of the turn this sends them to
-4
View File
@@ -104,10 +104,6 @@
"Stop": [
{
"hooks": [
{
"type": "command",
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_report_check.sh\""
},
{
"type": "command",
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_reply_check.sh\""
+9
View File
@@ -110,6 +110,7 @@ rule_state_dir="${TMPDIR:-/tmp}/scribe-priorart"
mkdir -p "$rule_state_dir" 2>/dev/null || true
idfile=""
rulefile=""
shapefile=""
exclude_q=""
if [ -n "$session_id" ]; then
# session_id is an opaque token from Claude Code; keep only filename-safe chars.
@@ -127,6 +128,11 @@ if [ -n "$session_id" ]; then
[ -n "$rule_seen" ] && exclude_q="${exclude_q}&exclude_rule_ids=${rule_seen}"
# What the session actually OPENED, as against what it was shown (#4100).
exclude_q="${exclude_q}$(scribe_held_query "$rule_state_dir/${safe_sid}.opened.ids")"
# Which reply shapes it already holds in full (milestone 500): the core is
# sent whole once, and as its one-line reminder on every later turn.
shapefile=$(scribe_shapes_file "$safe_sid")
shapes_seen=$(scribe_shapes_seen "$shapefile")
[ -n "$shapes_seen" ] && exclude_q="${exclude_q}&shapes_seen=${shapes_seen}"
fi
body=$(curl -fsS --max-time 5 \
@@ -150,6 +156,9 @@ if [ -n "$rulefile" ]; then
scribe_json_list "$body_flat" '.rule_ids' \
| scribe_rules_append "$rulefile"
fi
if [ -n "$shapefile" ]; then
scribe_json_list "$body_flat" '.shape_keys' | scribe_shapes_append "$shapefile"
fi
scribe_json_out UserPromptSubmit "$context"
exit 0
+51 -4
View File
@@ -135,8 +135,8 @@ scribe_json_flat_lines() {
# Cheap by construction, because it runs on every prompt: the last 512 KB only,
# and grep narrows to assistant records carrying a text block BEFORE anything is
# parsed — a transcript is mostly tool results, and parsing those in awk is the
# multi-second cost scribe_report_check.sh already measured. Fixed strings with
# UNESCAPED quotes can only match at a record's own top level (see that hook).
# multi-second cost the completion-report check measured (#4107). Fixed strings
# with UNESCAPED quotes can only match at a record's own top level.
# A sidechain is a subagent talking, not this session, so it is skipped.
scribe_recent_context() {
[ -n "${1:-}" ] && [ -f "$1" ] || return 0
@@ -205,8 +205,26 @@ scribe_json_len() {
}
# Raw JSON string bodies on stdin → text. One line in, one value out.
#
# A `\uXXXX` ESCAPE IS ENCODED AS UTF-8 HERE, BYTE BY BYTE, UNDER LC_ALL=C.
# The server escapes every non-ASCII character this way (the JSON default),
# and `sprintf("%c", n)` for n > 127 means a different thing in every awk and
# locale: gawk in a UTF-8 locale writes the character, gawk in the C locale
# and mawk write the single byte n % 256. So "·" (U+00B7) arrived as a lone
# 0xB7 and "—" (U+2014) as 0x14 wherever the hook ran without a UTF-8 locale —
# found by a test that ran the hook with a bare environment (milestone 500).
# Under LC_ALL=C every awk writes exactly the byte asked for, so the encoding
# below is the only one that happens.
scribe_json_unescape() {
awk '
LC_ALL=C awk '
function utf8(n) {
if (n < 128) return sprintf("%c", n)
if (n < 2048) return sprintf("%c%c", 192 + int(n / 64), 128 + n % 64)
if (n < 65536) return sprintf("%c%c%c", 224 + int(n / 4096),
128 + int(n / 64) % 64, 128 + n % 64)
return sprintf("%c%c%c%c", 240 + int(n / 262144), 128 + int(n / 4096) % 64,
128 + int(n / 64) % 64, 128 + n % 64)
}
function hex4(h, i, c, d, v) {
v = 0
for (i = 1; i <= 4; i++) {
@@ -246,7 +264,7 @@ scribe_json_unescape() {
i += 6
}
}
o = o sprintf("%c", hi)
o = o utf8(hi)
}
else o = o d
}
@@ -1208,6 +1226,35 @@ scribe_written_append() {
return 0
}
# THE REPLY-SHAPE LEDGER (milestone 500). Which default reply shapes this
# session has been shown IN FULL — `core`, `completion`, `asks`, `plan` — one
# key per line in <sid>.shapes.ids under the prior-art directory. The server
# sends a shape in full when its key is absent and as a one-line reminder when
# present, so the core arrives whole once and is a pointer on later turns.
#
# Swept with every other `.ids` ledger on compact and clear, and that is the
# point rather than a side effect: after a compaction the session no longer
# holds the shape, so the next turn gets it in full again. Not aged — the
# per-turn reminder is what keeps it in view between compactions.
# scribe_shapes_file <sid> the ledger's path
# scribe_shapes_seen <file> its keys, comma-joined, for `shapes_seen=`
# scribe_shapes_append <file> keys on stdin, appended
scribe_shapes_file() {
printf '%s/scribe-priorart/%s.shapes.ids' "${TMPDIR:-/tmp}" "$1"
}
scribe_shapes_seen() {
[ -n "$1" ] && [ -f "$1" ] || return 0
awk '/^[a-z]+$/ && !seen[$0]++ { out = out (out == "" ? "" : ",") $0 } END { print out }' \
"$1" 2>/dev/null || true
}
scribe_shapes_append() {
[ -n "$1" ] || { cat >/dev/null; return 0; }
mkdir -p "$(dirname "$1")" 2>/dev/null || true
awk '/^[a-z]+$/' >> "$1" 2>/dev/null || true
}
scribe_clear_session_ledgers() {
# SPARES `*.keep.ids`, which are evidence rather than exclusions — see
# `scribe_rules_append`. Everything else still goes: the convention is
+1 -1
View File
@@ -39,7 +39,7 @@
#
# mode=lines the input is JSONL — one JSON value per line — and a line that
# does not parse is DROPPED, the rest still read. This is the
# transcript in scribe_report_check.sh, where the window starts
# transcript the Stop hooks read, where the window starts
# mid-record by construction: `tail -n 3000` cuts wherever it
# cuts, and the first line is routinely half a record. It is
# exactly the `map(try fromjson catch empty)` the jq program it
+6
View File
@@ -122,6 +122,11 @@ ledger_q=""
rule_seen=$(scribe_rules_live "$rulefile")
[ -n "$rule_seen" ] && ledger_q="&exclude_rule_ids=${rule_seen}"
ledger_q="${ledger_q}$(scribe_held_query "$state_dir/${safe_sid}.opened.ids")"
# The reply-shape ledger (milestone 500): a shape this session already holds in
# full comes back as its reminder.
shapefile=$(scribe_shapes_file "$safe_sid")
shapes_seen=$(scribe_shapes_seen "$shapefile")
[ -n "$shapes_seen" ] && ledger_q="${ledger_q}&shapes_seen=${shapes_seen}"
# The event whole, since which field a mapping matches is the mapping's
# business. A write of a very large file is reduced to its name: the moments a
@@ -141,6 +146,7 @@ body=$(printf '%s' "$event" | curl -fsS --max-time 3 \
body_flat=$(printf '%s' "$body" | scribe_json_flat)
scribe_json_list "$body_flat" '.rule_ids' | scribe_rules_append "$rulefile"
scribe_json_list "$body_flat" '.shape_keys' | scribe_shapes_append "$shapefile"
context=$(scribe_json_pick "$body_flat" '.context')
[ -n "$context" ] || exit 0
+47 -10
View File
@@ -9,7 +9,15 @@
# text against every rule's trigger, as the backstop for whatever the earlier
# arms missed. An unopened rule from either half holds the reply for one read.
#
# THE SAME CONTRACT AS scribe_report_check.sh, and for its reasons:
# ONE END-OF-TURN REQUEST (milestone 500 step 4). When the turn closed a task,
# the same request carries the task ids and the server also checks the reply
# for the completion sections — once a Stop hook of its own
# (scribe_report_check.sh), now the same call. The section check's outcome is
# recorded (`report_check`, milestone 409's adherence number), and when it
# held the reply, the rewrite is sent back once with `rewrite: true` so the
# server can record how it came out. A rewrite is never held.
#
# THE CONTRACT, and the reasons for it:
# - the server decides and supplies the words; this hook blocks only on a
# reason it was given, so an unconfigured or unreachable instance never
# stops a session;
@@ -34,15 +42,44 @@ session_id=$(scribe_json_pick "$event_flat" '.session_id')
active=$(scribe_json_pick "$event_flat" '.stop_hook_active')
event_cwd=$(scribe_json_pick "$event_flat" '.cwd')
# The rewrite after a hold goes out as written.
[ "$active" = "true" ] && exit 0
[ -n "$transcript" ] && [ -f "$transcript" ] && [ -n "$session_id" ] || exit 0
state_dir="${TMPDIR:-/tmp}/scribe-priorart"
mkdir -p "$state_dir" 2>/dev/null || true
safe_sid=$(printf '%s' "$session_id" | tr -c 'A-Za-z0-9._-' '_')
rulefile="$state_dir/${safe_sid}.rules.ids"
stopfile="$state_dir/${safe_sid}.checkpoint.ids"
# Set when the section check held the reply: the next stop is its rewrite.
# A ledger by name (`.ids`), so a compaction sweeps it with the rest.
reportfile="$state_dir/${safe_sid}.reportcheck.ids"
# The rewrite after a hold goes out as written. Only a rewrite of a SECTION
# hold is sent at all — to record how it came out; one after a rule hold, or
# another plugin's block, has nothing to report.
if [ "$active" = "true" ]; then
[ -f "$reportfile" ] || exit 0
rm -f "$reportfile" 2>/dev/null || true
rewrite=true
else
rm -f "$reportfile" 2>/dev/null || true
rewrite=false
fi
scribe_config || exit 0
facts=$(scribe_turn_facts "$transcript")
[ "$(scribe_turn_fact "$facts" bounded)" = "1" ] || exit 0
reply=$(scribe_turn_fact "$facts" reply | scribe_json_unescape)
# The reply may not be in the transcript yet when the hook fires: an empty
# reply is "cannot tell", never "missing everything".
[ -n "$(printf '%s' "$reply" | tr -d '[:space:]')" ] || exit 0
# How many tasks this turn closed, and which — successful closes only, this
# session only (scribe_turn.awk). A task created already done closes with no
# id, so the count is what says a check is due. Digits and commas only, so
# both drop straight into the JSON.
closed=$(scribe_turn_fact "$facts" closed | tr -cd '0-9')
closed=${closed:-0}
task_ids=$(scribe_turn_fact "$facts" task_ids | tr -cd '0-9,' | sed 's/,,*/,/g; s/^,//; s/,$//')
[ "$rewrite" = "true" ] && [ "$closed" = "0" ] && exit 0
# Bounded before encoding. The server reads the head and the tail — the part
# of a report that asks something of the reader is at its end — so a very
@@ -51,12 +88,10 @@ if [ "${#reply}" -gt 12000 ]; then
reply="${reply:0:4000} … ${reply: -8000}"
fi
reply_esc=$(printf '%s' "$reply" | scribe_json_escape) || exit 0
state_dir="${TMPDIR:-/tmp}/scribe-priorart"
mkdir -p "$state_dir" 2>/dev/null || true
safe_sid=$(printf '%s' "$session_id" | tr -c 'A-Za-z0-9._-' '_')
rulefile="$state_dir/${safe_sid}.rules.ids"
stopfile="$state_dir/${safe_sid}.checkpoint.ids"
closing=""
if [ "$closed" != "0" ]; then
closing=$(printf ',"closed":%d,"closed_task_ids":[%s],"rewrite":%s' "$closed" "$task_ids" "$rewrite")
fi
query=""
scope=$(scribe_scope_query "${event_cwd:-${CLAUDE_PROJECT_DIR:-$PWD}}")
@@ -70,15 +105,17 @@ if [ -f "$stopfile" ]; then
fi
query=${query#&}
answer=$(printf '{"reply":"%s"}' "$reply_esc" | curl -fsS --max-time 6 \
answer=$(printf '{"reply":"%s"%s}' "$reply_esc" "$closing" | curl -fsS --max-time 6 \
-H "Authorization: Bearer ${token}" \
-H "Content-Type: application/json" \
--data-binary @- \
"${url%/}/api/plugin/reply-rules${query:+?$query}" 2>/dev/null) || exit 0
[ "$rewrite" = "true" ] && exit 0
answer_flat=$(printf '%s' "$answer" | scribe_json_flat)
reason=$(scribe_json_pick "$answer_flat" '.reason')
[ -n "$reason" ] || exit 0
[ "$(scribe_json_pick "$answer_flat" '.report_check')" = "blocked" ] && : > "$reportfile" 2>/dev/null
# Recorded BEFORE the block is emitted, for the act checkpoint's reason: a
# hold that is shown and not recorded is one that can be shown again.
-155
View File
@@ -1,155 +0,0 @@
#!/usr/bin/env bash
# Scribe plugin — Stop hook: a reply that closes a task carries the completion
# sections (milestone 409 step 5).
#
# Everything else the plugin does happens BEFORE the agent writes: context,
# retrieval, the reporting-back skill. This is the one moment the finished
# reply exists, so it is both the last chance to fix a report the operator
# cannot read and the only place adherence to the shape can be measured.
#
# DETERMINISTIC, NO MODEL CALL. Three questions, cheapest first:
#
# 1. Did this turn close a Scribe task? An `update_task` / `create_task` tool
# call with status "done" since the turn's prompt, whose result was not an
# error. Most turns stop here, silently.
# 2. Does the reply that ends the turn have the completion sections? Loosely:
# where the work sits (a record named by id and title, or step N of M),
# what needs the operator, and what comes next. Matched on the words that
# carry the meaning rather than exact headings, so the skill's wording can
# change without breaking this.
# 3. If sections are missing, block once with a reason naming them. The agent
# rewrites; the rewrite is checked and recorded, and never blocked again.
#
# MEASURED FROM THE FIRST CALL. Each checked reply is reported to the instance
# (`/api/plugin/report-check`): passed, blocked, and after a rewrite either
# passed_after_rewrite or missing_after_rewrite. Turns that closed no task are
# not reported — they would cost a request on every turn and add nothing to
# the rate step 6 reads (blocked among checked replies).
#
# IT BLOCKS ONLY WHEN THE BLOCK IS RECORDED, AND ONLY IN THE SERVER'S WORDS.
# The report goes out first; the instance answers a recorded `blocked` with
# the reason to send the agent back with, and the hook blocks only on that
# reason. An unconfigured or unreachable instance therefore never stops a
# session, every intervention is one the numbers can see, and the guidance
# text lives on the server (plugin/PACKAGING.md: hooks carry timing and
# transport).
#
# THE TRANSCRIPT FORMAT IS OBSERVED, NOT DOCUMENTED. Claude Code documents
# `transcript_path` and `stop_hook_active` for Stop, not the JSONL inside. As
# read from real transcripts (2026-09-14): one content block per line;
# `type: "assistant"` lines carry `message.content[]` blocks of `text` /
# `tool_use` ({id, name, input}); tool results arrive as `type: "user"` lines
# whose content is a `tool_result` array ({tool_use_id, is_error}); a turn's
# prompt — typed, or a background-task notification — is a `user` line whose
# content is a plain string and which is not `isMeta`. Anything that does not
# parse that way makes the hook stay out of the way rather than guess.
#
# Config (same as the other hooks):
# CLAUDE_PLUGIN_OPTION_API_ENDPOINT base URL, no trailing slash
# CLAUDE_PLUGIN_OPTION_API_TOKEN fmcp_ API key (sensitive)
# SCRIBE_URL / SCRIBE_TOKEN override for the settings.json dogfooding path.
set -uo pipefail
command -v curl >/dev/null 2>&1 || exit 0
# shellcheck source=plugin/hooks/scribe_defs.sh
. "$(dirname "${BASH_SOURCE[0]}")/scribe_defs.sh"
# Stop delivers { session_id, transcript_path, cwd, hook_event_name, stop_hook_active }.
event=$(cat 2>/dev/null || true)
event_flat=$(printf '%s' "$event" | scribe_json_flat)
transcript=$(scribe_json_pick "$event_flat" '.transcript_path')
[ -n "$transcript" ] && [ -f "$transcript" ] || exit 0
session_id=$(scribe_json_pick "$event_flat" '.session_id')
active=$(scribe_json_pick "$event_flat" '.stop_hook_active')
event_cwd=$(scribe_json_pick "$event_flat" '.cwd')
safe_sid=$(printf '%s' "${session_id:-nosession}" | tr -c 'A-Za-z0-9._-' '_')
state_dir="${TMPDIR:-/tmp}/scribe-reportcheck"
mkdir -p "$state_dir" 2>/dev/null || true
marker="$state_dir/${safe_sid}.blocked"
# Cheap prefilter: no task tool anywhere in the recent transcript → nothing to
# check. Keeps the ordinary turn at one grep. Process substitution, NOT a pipe:
# under `pipefail`, `grep -q` exiting on the first match kills `tail` with
# SIGPIPE, and the pipeline then reports failure precisely when it matched.
grep -q -E '"name":[[:space:]]*"([^"]*__)?(update|create)_task"' < <(tail -c 2000000 "$transcript" 2>/dev/null) || {
rm -f "$marker" 2>/dev/null || true
exit 0
}
# The turn, parsed once — scribe_turn_facts, shared with the reply check.
facts=$(scribe_turn_facts "$transcript")
fact() { scribe_turn_fact "$facts" "$1"; }
[ "$(fact bounded)" = "1" ] || exit 0
closed=$(fact closed)
if [ "${closed:-0}" = "0" ]; then
rm -f "$marker" 2>/dev/null || true
exit 0
fi
# Escaped on one line coming out of awk, so the format survives a multi-line
# reply; decoded here, once, where it is about to be read as text.
reply=$(fact reply | scribe_json_unescape)
task_ids=$(fact task_ids)
# The reply may not be written to the transcript yet when the hook fires. An
# empty reply is "cannot tell", not "missing everything" — stay out of the way.
[ -n "$(printf '%s' "$reply" | tr -d '[:space:]')" ] || exit 0
missing=()
# Where the work sits: a record named by id AND title (#12 "…", milestone 3 "…"),
# or a step position. A bare id is exactly the homework this shape removes.
# shellcheck disable=SC2016 # backticks here are literal markdown, not an expansion
grep -q -i -E '(#[0-9]+|milestone [0-9]+|task [0-9]+)[*_`]*[[:space:]]*[*_`]*["“]|step [0-9]+ of [0-9]+' <<< "$reply" \
|| missing+=("where it sits")
# What needs the operator — "needs you: nothing" counts; it is an answer.
grep -q -i -E 'needs? (from )?you|nothing (is )?needed from you|your (call|decision)' <<< "$reply" \
|| missing+=("needs you")
# What comes next.
grep -q -i -E '\bnext\b' <<< "$reply" \
|| missing+=("next")
# Reports the outcome; prints the instance's reply and returns 0 only if the
# instance recorded it.
report() {
scribe_config || return 1
local q repo enc m
q="outcome=$1&task_ids=${task_ids}"
m=$(IFS=,; printf '%s' "${missing[*]:-}")
if [ -n "$m" ]; then
enc=$(printf '%s' "$m" | scribe_urlenc) || enc=""
q="${q}&missing=${enc}"
fi
scope=$(scribe_scope_query "${event_cwd:-${CLAUDE_PROJECT_DIR:-$PWD}}")
[ -n "$scope" ] && q="${q}&${scope}"
curl -fsS --max-time 4 \
-H "Authorization: Bearer ${token}" \
"${url%/}/api/plugin/report-check?${q}" 2>/dev/null
}
if [ "$active" = "true" ]; then
# A Stop hook already blocked this stop. If it was this one, the reply is
# the rewrite: record how it came out, and let the session stop whatever
# the answer. If it was another plugin's block, this hook has nothing to add.
[ -f "$marker" ] || exit 0
rm -f "$marker" 2>/dev/null || true
if [ ${#missing[@]} -eq 0 ]; then report passed_after_rewrite >/dev/null; else report missing_after_rewrite >/dev/null; fi
exit 0
fi
rm -f "$marker" 2>/dev/null || true
if [ ${#missing[@]} -eq 0 ]; then
report passed >/dev/null
exit 0
fi
# The words the agent is sent back with are the server's (plugin/PACKAGING.md:
# a hook carries timing and transport). No reason back → nothing recorded →
# no block.
answer=$(report blocked) || exit 0
reason=$(scribe_json_pick "$(printf '%s' "$answer" | scribe_json_flat)" '.reason')
[ -n "$reason" ] || exit 0
: > "$marker" 2>/dev/null || true
printf '{"decision":"block","reason":"%s"}\n' "$(printf '%s' "$reason" | scribe_json_escape)"
exit 0
+1 -1
View File
@@ -10,7 +10,7 @@
# words: the agent that built the code is the one participant who knows what
# it is, and the end of the turn is the last moment that is still true.
#
# THE SAME DISCIPLINE AS scribe_report_check.sh, deliberately:
# THE SAME DISCIPLINE AS the completion-section check in scribe_reply_check.sh:
# - it blocks only on a `reason` the instance returned, which it returns
# only for a block it RECORDED — an unconfigured or unreachable instance
# never stops a session, and every intervention is one the numbers see;
+1 -1
View File
@@ -2,7 +2,7 @@
#
# Reads the flat `IDX<TAB>PATH<TAB>VALUE` stream that scribe_json.awk produces
# in `mode=lines` from a Claude Code transcript, and answers the four questions
# scribe_report_check.sh asks. It replaces a thirty-line jq program; the shape
# the Stop hooks ask (scribe_reply_check.sh). It replaces a thirty-line jq program; the shape
# of the answer is unchanged, so the hook around it reads the same.
#
# bounded 1 if a user prompt was found in the window, else empty. A window
+42 -86
View File
@@ -1,6 +1,6 @@
---
name: reporting-back
description: Use when you are about to write the reply the operator will read — work finished, a task marked done, stopping on a blocker, asking them to decide or to do something, answering "where are we" / "what's next", or proposing an approach. Shapes the reply around where the work stands (which task, what changed, what needs them, what comes next) instead of the order you did things in. Triggers on reporting completion, handing off, asking a question, or summarising progress.
description: Use when you are about to write the reply the operator will read and the delivered reply shape is not enough — work finished, stopping on a blocker, asking them to decide or to do something, answering "where are we", or proposing an approach. The long-form reference behind the reply shapes Scribe delivers: why sections are chosen rather than filled, who decides what, taking placement from the record, and a completion report worked in full. Triggers on reporting completion, handing off, asking a question, or summarising progress.
metadata:
moments: reply.report
---
@@ -12,7 +12,23 @@ while you worked: they don't hold the files you read, the names you used or
the order you did things in. A reply that follows *your* path is accurate and
still unreadable to them. Shape it around **where the work stands**.
Pick the kind of reply first (the tables below). Its sections are **what to
## The shapes arrive on their own
Scribe delivers the default shape for each kind of reply as product, at the
moment it is due: the core ("Reply shape · Every reply") with each turn, the
completion report when a task closes, the asks when you put a question to the
operator, the plan when you plan. A shape this session has already seen comes
back as its one-line reminder, and `list_reply_shapes` has every one in full.
The operator's own preferences are mounted on the same moments and arrive
beside the shape; where one differs, the preference is what they asked for.
This skill does not restate those shapes. It holds the reasoning behind them
and the completion report worked in full — read it when a shape's lines are
not enough to decide what a reply should say.
## Sections are chosen, not filled
Pick the kind of reply first (the delivered shape). Its sections are **what to
consider including, not a form to complete.**
Two kinds of section, and they behave differently:
@@ -32,32 +48,15 @@ lines naming the two things that changed their position. Extra length has to be
earned: a comparison they asked for, options that need laying side by side, a
measurement whose numbers are the point.
## Every reply
## A decision already made
- **Conclusion first.** The verdict, the result, or the question — before the
reasoning that supports it.
- **One topic per section.** Two things the operator raised get two sections.
- **Make priority visible.** Bold the few things that matter; let the rest be
plain.
- **End with the ask, in bold** — the one thing they need to decide or do. If
there is nothing, say so.
- **A request for approval gets its own section, headed "Approval requested".**
Whenever you are holding an action until the operator says yes — a permission
prompt their client raised, something hard to undo, a change you have
prepared and not made — put it there, near the top, whatever kind of reply
this is. Inside a progress or completion list it reads as a status line, and
the operator does not see that you are waiting on them.
- **Plain words.** Use the operator's vocabulary, not the names you coined while
working. If a term has to appear, explain it once.
- **Place the work in Scribe.** Name the task, issue or milestone it belongs to,
by id *and* title (using-scribe: "Name the record, never just its number").
- **A decision already made gets acted on, and the reply says what you did with
it.** Once the operator has chosen, that is the input to the work, not a topic
to revisit. If something you have since learned genuinely overturns the
choice, say so once and plainly — name the new evidence and what it changes —
and otherwise let the decision stand. Laying out the trade-offs of a settled
question again reads as contradicting yourself rather than as being careful,
and it costs the operator the decision twice.
**A decision already made gets acted on, and the reply says what you did with
it.** Once the operator has chosen, that is the input to the work, not a topic
to revisit. If something you have since learned genuinely overturns the
choice, say so once and plainly — name the new evidence and what it changes —
and otherwise let the decision stand. Laying out the trade-offs of a settled
question again reads as contradicting yourself rather than as being careful,
and it costs the operator the decision twice.
## Take the placement from the record
@@ -80,39 +79,14 @@ it is wrong. So take placement from Scribe:
- Work with no task behind it: say so plainly — "this wasn't tracked as a
task" — and offer to record it. An honest "untracked" is a placement too.
## The operator's own shapes come first
The shapes below are defaults. An operator may have changed some of them — a
section they always want, an order they read faster, a kind of reply they want
shorter — and those changes are `preference` records. Where a preference and a
default differ, the preference is what they asked for.
- **A completion report brings its preferences with it.** Closing a task with
`update_task` returns them as **`reply_preferences`** when the operator has
any; the `report_back` line says so. Nothing to search for.
- **Every other reply, ask before writing it.** A finding, a decision, a
handoff, a "where are we" — no tool call comes before these, so nothing
hands their preferences over. Once you know which kind of reply you are
writing, `search(content_type="rule")` for it in the words of that moment —
"writing a decision for the operator", "handing off to the operator" — and
follow any preference that comes back. Nothing coming back means the default
shape stands.
## Reports — work happened
| Kind | Sections |
|---|---|
| **Completion** | Where this sits · What now works · How / why · Needs you · Next |
| **Finding** (a problem you found) | Symptom · Cause · **What you decided and did** — judge it and act; an offer to fix it is the judgment not made (see *You are the judge*) |
| **Blocked / failed** | What stopped · What you tried · What you need from them |
| **Progress** (mid-work) | One or two lines: where things are, what's next, any blocker |
| **Where are we** | The milestone and its progress · Done · Open · Needs you · Next |
## You are the judge
You are the judge of record for the work itself: what a shape is, whether a
finding holds, whether a record is right, whether something is done. The
operator reads the decision — they do not make it.
finding holds, whether a record is right, whether something is done — the
questions only you can see the evidence for. Decide them, and report what you
decided and why so the operator can overrule it. **Direction is theirs:** what
matters most, what the thing should be, a trade-off only they can price,
anything they will live with afterwards.
**A finding surfaced and not judged is a finding dropped, not deferred.** This
is the failure that hides inside a good report: the symptom named, the cause
@@ -125,11 +99,12 @@ So: decide, act, and report what you decided and why. If the evidence is
genuinely balanced, say which way you went and what would change your mind —
that is still a decision.
**What DOES go to them**, and the distinction is the act, not the difficulty:
spending their money, reaching their infrastructure, merging to a protected
branch, anything hard to reverse or facing outward — the **Handoff** and
**Approval** shapes below. Judging a record is never one of these. A hard call
is still yours; an irreversible act is still theirs.
**What DOES go to them** is direction, and every act that is hard to reverse
or faces outward: spending their money, reaching their infrastructure, merging
to a protected branch — the **Decision**, **Handoff** and **Approval** kinds of
the asks shape. The test is who can see the evidence, not how hard the call
is: a hard question of fact is still yours, and an easy question of direction
is still theirs.
**Judging is attended, not automatic.** You judge by reading the evidence and
recording why. A threshold, a sweep or a rule that reclassifies in bulk with
@@ -142,15 +117,11 @@ writes with another unattended write, stop.
evidence needed to decide and a way to record the decision under their own
name.
## Asks — the operator needs to act or decide
## Asking them
| Kind | Sections |
|---|---|
| **Decision** | The question first · 2–4 options, each with what it changes · recommendation first |
| **Clarification** | "My reading is X · the gap is Y · unless you say otherwise I'll do Z" |
| **Handoff** (only they can do it) | The action · why it needs them · what it unblocks · what you'll do after · any way to skip it |
| **Approval** (you are ready to act and holding for a yes) | **Approval requested:** exactly what happens once they approve, one numbered item per change so they can approve part · why it needs their yes · how it can be undone · what you'll do after |
| **Conflict** (what you're about to do clashes with a rule, a plan or an earlier decision) | What it says · what you were about to do · where they clash · A or B? |
The asks shape arrives when you put a question to the operator; it lists the
kinds — a decision, a clarification, a handoff, an approval you are holding
for, a conflict with a rule or an earlier decision. Two things behind it:
Before asking, check whether you can find the answer yourself — something that
can be read or looked up is a fact to check, not a question to send.
@@ -164,21 +135,6 @@ between options is not a decision about the assumptions they share, and an
unlabelled one gets approved as if it had been asked. An assumption that
contradicts a recorded ruling is a **Conflict**, not an option's fine print.
## Answers — the operator asked something
| Kind | Sections |
|---|---|
| **Explanation** | The answer first · then the evidence, pointing at what they could open to check it |
| **Evaluation** ("can we / should we") | Verdict · What exists · The gaps · Recommendation |
## Proposals — shaping future work
| Kind | Sections |
|---|---|
| **Options** | 2–3 approaches · the trade-off of each · one recommendation |
| **Plan** | Goal · Steps · Open questions — for review before starting (writing-plans) |
| **Review** | Findings ranked by how much they matter, one per item |
## The completion report, in full
The most common reply, and the one most often written in the order the work
@@ -235,7 +191,7 @@ happens next** without asking a follow-up? If not, the sections are what's
missing — not more detail.
Then read it once more for **what can go**. A section filled because it was in
the table, reasoning supporting a conclusion nobody is going to dispute, a
the shape, reasoning supporting a conclusion nobody is going to dispute, a
finding already written to the record — none of it changes what the operator
does, so none of it belongs in the reply. Cutting is not hiding: the log holds
it, and the reply stays readable. A reply that has been cut twice is the one
+5 -3
View File
@@ -263,9 +263,11 @@ Two constraints on *how* that's achieved:
order you did things in: which task or milestone it belongs to, what now
works, what needs them, and what comes next. Take the placement from the
`placement` block that `create_task` / `update_task` return — the milestone,
step N of M, the next open step — rather than from memory. The
`reporting-back` skill holds the shape for each kind of reply: completions,
findings, decisions, handoffs, "where are we".
step N of M, the next open step — rather than from memory. Scribe delivers
the shape for each kind of reply when it is due — the core with each turn,
the completion report when a task closes, the asks when you put a question —
and `list_reply_shapes` has them all; the `reporting-back` skill holds the
reasoning behind them and a completion report worked in full.
## Stay inside the active project's scope
+272
View File
@@ -0,0 +1,272 @@
#!/usr/bin/env python3
"""Measure near-verbatim duplication in a tree, and what a change added to it.
WHY THIS EXISTS
A DRY audit made of several passes ends with a question no single pass can
answer: did the audit as a whole reduce duplication, and did it create any? The
second half is the one that matters. A pass that shortens two statements can
leave them byte-identical to a third, and the new copy is invisible to that
pass's own scan because it did not exist when the scan ran (#4745: a retro over
a 17-pass audit found exactly this, and folded it into a view). This script is
the close-out measure the DRY Pass process names, so the measure is a tool and
not something each audit improvises.
WHAT IT MEASURES
Each file is reduced to its significant lines: blank lines, comment-only lines
and lines of bare punctuation (`}`, `});`, `],`) are dropped; whitespace is
collapsed; string literals are folded to "S", so two copies that differ only in
a message or a key still match. A *window* is N consecutive significant lines
(default 6). A window whose text occurs at two or more places in the same group
is duplicated, and every line it covers is a duplicated line.
Per group it reports significant lines, duplicated windows (distinct texts),
duplicated lines and their share. With `--before REV` it measures that revision
too and lists the duplicated windows that exist ONLY after — not duplicated
before, either because the text was absent or because it occurred once. Those
are grouped by the set of files they appear in, which is the list to read.
WHAT IT DOES NOT SEE
Structural or semantic copies — two components with the same shape and
different names, the same predicate written two ways. It catches near-verbatim
text only, so a fold of structural copies shows here less than it counts. Read
the numbers as a floor and the after-only list as the finding.
USAGE
measure_duplication.py [--repo DIR] [--before REV] [--after REV]
[--group NAME=GLOB[,GLOB...]]... [--exclude GLOB]...
[--window N] [--show N] [--json]
The after tree is the working tree's tracked files unless `--after REV` names a
revision. Revisions are read through `git archive`, so nothing is checked out.
Globs are fnmatch patterns over repo-relative paths, where `*` crosses `/`.
Groups are tried in order and the first match claims a file, so list
`--group 'go tests=*_test.go'` before `--group 'go=*.go'`. With no `--group`,
files are grouped by extension among common source types.
Stdlib only, so it runs from a scratch copy in any project:
`get_snippet` the recorded snippet, write its code to a file, run it there.
"""
from __future__ import annotations
import argparse
import fnmatch
import hashlib
import io
import json
import re
import subprocess
import sys
import tarfile
from collections import defaultdict
from pathlib import Path
DEFAULT_EXTS = {
"c", "cc", "cpp", "cs", "css", "dart", "go", "h", "hpp", "java", "js",
"jsx", "kt", "kts", "php", "py", "rb", "rs", "scss", "sh", "sql",
"svelte", "swift", "ts", "tsx", "vue",
}
_COMMENT = re.compile(r"^(//|/\*|\*|--|<!--|#(\s|!|$))")
_PUNCT_ONLY = re.compile(r"^[\s{}()\[\];,]*$")
_STRING = re.compile(r'"(?:\\.|[^"\\])*"|\'(?:\\.|[^\'\\])*\'|`(?:\\.|[^`\\])*`')
_SPACE = re.compile(r"\s+")
def significant_lines(text: str) -> list[tuple[int, str]]:
"""(1-based line number, normalized text) for each line that carries code."""
out = []
for number, raw in enumerate(text.splitlines(), 1):
line = raw.strip()
if not line or _COMMENT.match(line) or _PUNCT_ONLY.match(line):
continue
out.append((number, _SPACE.sub(" ", _STRING.sub('"S"', line))))
return out
def _git(repo: Path, *args: str) -> bytes:
return subprocess.run(
["git", "-C", str(repo), *args], check=True, capture_output=True,
).stdout
def read_working_tree(repo: Path):
"""Tracked files as they are on disk — what the next commit would hold."""
for rel in _git(repo, "ls-files", "-z").decode().split("\0"):
path = repo / rel
if rel and path.is_file():
yield rel, path.read_bytes()
def read_revision(repo: Path, rev: str):
"""Every file at `rev`, streamed out of `git archive` without a checkout."""
stream = io.BytesIO(_git(repo, "archive", "--format=tar", rev))
with tarfile.open(fileobj=stream, mode="r:") as archive:
for member in archive:
if member.isfile():
yield member.name, archive.extractfile(member).read()
def group_of(rel: str, groups: list[tuple[str, list[str]]]) -> str | None:
if not groups:
ext = rel.rsplit(".", 1)[-1] if "." in Path(rel).name else ""
return ext if ext in DEFAULT_EXTS else None
for name, patterns in groups:
if any(fnmatch.fnmatch(rel, p) for p in patterns):
return name
return None
def measure(files, groups, excludes, window: int) -> dict:
"""Per group: line counts, and every duplicated window with where it occurs."""
by_group: dict[str, dict] = defaultdict(
lambda: {"files": {}, "index": defaultdict(list)},
)
for rel, data in files:
if any(fnmatch.fnmatch(rel, p) for p in excludes):
continue
name = group_of(rel, groups)
if name is None:
continue
try:
text = data.decode("utf-8")
except UnicodeDecodeError:
continue
lines = significant_lines(text)
g = by_group[name]
g["files"][rel] = lines
for i in range(len(lines) - window + 1):
body = "\n".join(t for _, t in lines[i:i + window])
key = hashlib.sha1(body.encode()).hexdigest()
g["index"][key].append((rel, i))
result = {}
for name, g in sorted(by_group.items()):
dup = {k: occ for k, occ in g["index"].items() if len(occ) > 1}
covered = set()
for occ in dup.values():
for rel, i in occ:
covered.update((rel, j) for j in range(i, i + window))
total = sum(len(lines) for lines in g["files"].values())
result[name] = {
"lines": total,
"dup_windows": len(dup),
"dup_lines": len(covered),
"share": len(covered) / total if total else 0.0,
"_dup": dup,
"_files": g["files"],
}
return result
def only_after(before: dict, after: dict) -> dict[str, list[dict]]:
"""Duplicated windows in `after` that were not duplicated in `before`,
grouped by the set of files they occur in."""
out = {}
for name, a in after.items():
was = before.get(name, {}).get("_dup", {})
sets: dict[tuple, dict] = {}
for key, occ in a["_dup"].items():
if key in was:
continue
fileset = tuple(sorted({rel for rel, _ in occ}))
entry = sets.setdefault(fileset, {"files": list(fileset), "windows": 0, "at": []})
entry["windows"] += 1
if not entry["at"]:
entry["at"] = [
f"{rel}:{a['_files'][rel][i][0]}" for rel, i in sorted(occ)
]
if sets:
out[name] = sorted(sets.values(), key=lambda e: -e["windows"])
return out
def _public(result: dict) -> dict:
return {
name: {k: v for k, v in g.items() if not k.startswith("_")}
for name, g in result.items()
}
def _pct(share: float) -> str:
return f"{share * 100:.1f}%"
def report(before: dict | None, after: dict, fresh: dict, show: int) -> str:
rows = []
for name in sorted(set(after) | set(before or {})):
a = after.get(name, {"lines": 0, "dup_windows": 0, "dup_lines": 0, "share": 0.0})
if before is None:
rows.append((name, f"{a['lines']:,}", str(a["dup_windows"]),
str(a["dup_lines"]), _pct(a["share"])))
continue
b = before.get(name, {"lines": 0, "dup_windows": 0, "dup_lines": 0, "share": 0.0})
rows.append((
name,
f"{b['lines']:,} → {a['lines']:,}",
f"{b['dup_windows']} → {a['dup_windows']}",
f"{b['dup_lines']} → {a['dup_lines']}",
f"{_pct(b['share'])} → {_pct(a['share'])}",
))
header = ("group", "lines", "dup windows", "dup lines", "share")
widths = [max(len(r[c]) for r in [header, *rows]) for c in range(len(header))]
lines = [" ".join(cell.ljust(w) for cell, w in zip(r, widths)).rstrip()
for r in [header, *rows]]
if before is not None:
total = sum(len(v) for v in fresh.values())
lines += ["", f"Duplicated only after: {total} file set(s). Read each one —"
" a copy a pass's own shortening exposed, or a deliberate repeat."]
for name, sets in fresh.items():
for entry in sets[:show]:
lines.append(f" [{name}] {entry['windows']} window(s): "
+ ", ".join(entry["at"]))
if len(sets) > show:
lines.append(f" [{name}] … {len(sets) - show} more (--show)")
return "\n".join(lines)
def _parse_group(spec: str) -> tuple[str, list[str]]:
name, sep, globs = spec.partition("=")
if not sep or not name or not globs:
raise argparse.ArgumentTypeError(f"--group wants NAME=GLOB[,GLOB...], got {spec!r}")
return name, [g for g in globs.split(",") if g]
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
parser.add_argument("--repo", type=Path, default=Path("."))
parser.add_argument("--before", help="revision to compare against (e.g. the sha before the first pass)")
parser.add_argument("--after", help="revision to measure (default: the working tree's tracked files)")
parser.add_argument("--group", action="append", type=_parse_group, default=[])
parser.add_argument("--exclude", action="append", default=[])
parser.add_argument("--window", type=int, default=6)
parser.add_argument("--show", type=int, default=20)
parser.add_argument("--json", action="store_true")
args = parser.parse_args(argv)
repo = args.repo.resolve()
after_files = read_revision(repo, args.after) if args.after else read_working_tree(repo)
after = measure(after_files, args.group, args.exclude, args.window)
before = fresh = None
if args.before:
before = measure(read_revision(repo, args.before), args.group, args.exclude, args.window)
fresh = only_after(before, after)
if args.json:
json.dump({
"window": args.window,
"before": _public(before) if before is not None else None,
"after": _public(after),
"only_after": fresh,
}, sys.stdout, indent=2)
print()
else:
print(report(before, after, fresh or {}, args.show))
return 0
if __name__ == "__main__":
sys.exit(main())
+3
View File
@@ -159,6 +159,9 @@ _READ_ONLY_TOOLS = frozenset({
# retrieval_telemetry's reason, and needed by a read key so that a line
# naming a moment can be understood by whoever was shown it.
"list_moments",
# The default reply shapes (milestone 500). A pure read of a constant, and
# one a read key needs: a session shown a shape may want the rest.
"list_reply_shapes",
# The platform catalog and a project's answers (milestone 463). A pure
# read; set_project_platforms is the write.
"list_platforms",
+22
View File
@@ -15,6 +15,7 @@ from __future__ import annotations
from scribe.mcp._context import current_user_id
from scribe.services import moment_actions as actions_svc
from scribe.services import moments as moments_svc
from scribe.services import reply_shapes as shapes_svc
from scribe.services import rule_moment_judgments as judgments_svc
@@ -47,6 +48,26 @@ async def list_moments() -> dict:
return out
async def list_reply_shapes() -> dict:
"""The default shapes of a reply, and the moment each one arrives at.
You do not need this to write a reply: the shapes come to you. The core
arrives with the turn, and each kind's shape arrives at the moment before
that kind of reply is written (closing a task, putting a question, opening
a plan). Read this to see them all at once, to answer the operator about
what the default is, or before writing a preference that changes one.
An operator's own adjustment to a shape is a `preference` mounted on that
shape's `moment` (create_preference with `moments=[...]`); it arrives
beside the default, and where the two differ the preference is what the
operator asked for.
Each shape carries `key`, `title`, `moment` (and what the moment `means`),
`delivered` (when it arrives) and `text`.
"""
return shapes_svc.catalog()
async def map_action(tool: str, moment: str, match: str = "", reason: str = "") -> dict:
"""Make an action reach a moment on this install — the in-session fix for a missed moment.
@@ -213,6 +234,7 @@ async def rule_misfired(rule_id: int, moment: str, why: str,
def register(mcp) -> None:
mcp.tool(name="list_moments")(list_moments)
mcp.tool(name="list_reply_shapes")(list_reply_shapes)
mcp.tool(name="map_action")(map_action)
mcp.tool(name="unmap_action")(unmap_action)
mcp.tool(name="rules_to_mount")(rules_to_mount)
+12 -1
View File
@@ -28,6 +28,7 @@ from scribe.services import milestones as milestones_svc
from scribe.services import notes as notes_svc
from scribe.services import platforms as platforms_svc
from scribe.services import projects as projects_svc
from scribe.services import reply_shapes as reply_shapes_svc
from scribe.services import rulebooks as rulebooks_svc
from scribe.services import systems as systems_svc
from scribe.services import trash as trash_svc
@@ -73,11 +74,16 @@ async def enter_project(project_id: int) -> dict:
project_id: The project to enter.
Returns a dict with keys: project, milestone_summary, open_tasks, systems,
design_system, project_rules, pattern_coverage —
design_system, project_rules, pattern_coverage, reply_shape —
plus unplanned_milestones, milestone_summary_omitted,
unplanned_milestones_omitted, family, inception and systems_bootstrap,
each present only when it applies (see below).
`reply_shape` is the default shape of every reply you write to the
operator: conclusion first, the shortest reply that carries the answer,
where the work stands, and the one thing they need to do. Write to it;
an operator preference about replies that differs from it wins.
`project` is id, title, status and the full goal. get_project has the
whole record.
@@ -323,6 +329,11 @@ async def enter_project(project_id: int) -> dict:
out["family"] = family
if inception_ask:
out["inception"] = inception_ask
# The core reply shape (milestone 500): for a client with no prompt hook,
# entering the project is the one point every session passes before it
# writes a reply. A plugin session gets it with its first turn as well;
# one repeat per session is the price of reaching every client.
out["reply_shape"] = reply_shapes_svc.render(reply_shapes_svc.core(), full=True)
return out
+1 -13
View File
@@ -596,19 +596,7 @@ It is an UPPER BOUND per surface: a pull records the door it came
Only ever raised for unbidden arms known to log unconditionally: a
search returning a list every time is doing its job, and an arm whose
zeros were never written would flag a LOGGING bug while pointing you at
a threshold, which is #3497 exactly. Nor for an arm whose query never
changes — see the next entry.
- `fixed_query_never_clears` — an arm that always searches the SAME query
returned nothing on every call. Its score is one constant, so this is
not a quiet window: the bar sits above that constant and no amount of
further traffic will produce a different result. The arm is off rather
than silent, and nothing else here would say so. The same property is
why `cannot_decline` is not raised for these arms: with a constant
score the decline rate is 0% or 100% by construction, so "never
declined" is arithmetic and not evidence about the floor. Read the
refused record (`near_miss_samples`) BEFORE moving the dial — the last
time an arm sat here, every percentile said lower it and the refused
record showed the refusal was right.
a threshold, which is #3497 exactly.
- `band_hugs_floor` — the weakest tenth of what an arm returns sits on
its floor. The bar is doing the selecting and the score is not, so
moving that floor changes how MUCH you get, not how good it is.
+12 -28
View File
@@ -40,7 +40,6 @@ from scribe.services.notes import minted_kind
from scribe.services import placement as placement_svc
from scribe.services import planning as planning_svc
from scribe.services import record_batch as batch_svc
from scribe.services import reply_preferences as reply_prefs_svc
from scribe.services import rulebooks as rulebooks_svc
from scribe.services import systems as systems_svc
from scribe.services import task_logs as task_logs_svc
@@ -416,14 +415,13 @@ async def update_task(
reconstructing them, because a remembered milestone title or "next step"
reads exactly like a real one when it is wrong.
Closing a task (done or cancelled) also returns `report_back`: a one-line
reminder of what the reply to the operator should cover. When the
operator has preferences for how a completion report is written, they
come back as `reply_preferences` ({id, title, statement, kind}) — found
by their `when_to_apply`, so a preference whose trigger is writing the
report after finishing a task is the one that arrives here. Where one
differs from the default shape, the preference is what the operator
asked for. When family adoptions were answered `owed` while the task was
Closing a task (done or cancelled) also returns `reply_shape`, the
default shape of the completion report you are about to write, and
`report_back`: a one-line reminder of what that reply should cover. The
operator's own preferences for that report arrive beside it in
`moment_rules` — the ones mounted on finishing work. Where one differs
from the default shape, the preference is what the operator asked for.
When family adoptions were answered `owed` while the task was
open, they come back as `family_owed` ({idea_id, idea_title, project_id,
project_title, owed_task_id}): name each in the report — it is work
filed into that project.
@@ -469,14 +467,6 @@ async def update_task(
await family_svc.attach_family_hint(uid, data, note, created=False)
if status in _CLOSING_STATUSES:
data["report_back"] = REPORT_BACK_CUE
# The operator's own adjustments to the completion report, retrieved
# at the one moment a server can see that report coming (milestone
# 409 step 4). Omitted rather than sent empty, like every decoration.
prefs = await reply_prefs_svc.completion_preferences(
uid, project_id=getattr(note, "project_id", None))
if prefs:
data["reply_preferences"] = prefs
data["report_back"] = REPORT_BACK_CUE + " " + REPLY_PREFERENCES_CUE
# Owed family adoptions filed while the task was open (milestone 463
# step 6) — work now waiting in some project, which the report names.
await family_adoption_svc.attach_owed_adoptions(uid, data, note)
@@ -530,22 +520,16 @@ async def add_task_log(task_id: int, content: str) -> dict:
return data
# The in-band half of milestone 409 step 3. The reporting-back skill and the
# static context carry the full shapes, but both live only in the Claude Code
# plugin; a tool response reaches every MCP client, at the moment a piece of
# work closes, which is exactly when the report is about to be written. One
# line on purpose: a template here would be read as the reply itself.
# The in-band half of milestone 409 step 3. The completion shape itself rides
# the closing response as `reply_shape` (milestone 500); this is the one line
# that says what the reply covers, for every MCP client, at the moment a piece
# of work closes. One line on purpose: a template here would be read as the
# reply itself.
_CLOSING_STATUSES = ("done", "cancelled")
REPORT_BACK_CUE = (
"Reporting this to the operator? Say where it sits (from `placement`), "
"what now works, what needs them, and what comes next."
)
# Appended only when `reply_preferences` is present, so the key never arrives
# unexplained and a session with no preferences reads exactly what it did.
REPLY_PREFERENCES_CUE = (
"The operator has preferences for how this report is written — "
"follow `reply_preferences` over the default shape where they differ."
)
_ITEM_KEYS = {"title", "body", "type", "status", "priority", "kind", "tags", "system_ids"}
+76 -53
View File
@@ -15,6 +15,7 @@ from scribe.config import Config
from scribe.services import lesson_rules as lesson_rules_svc
from scribe.services import moment_delivery as moment_delivery_svc
from scribe.services import plugin_context as plugin_ctx_svc
from scribe.services import reply_shapes as reply_shapes_svc
from scribe.services import repo_bindings as repo_bindings_svc
from scribe.services import report_check as report_check_svc
from scribe.services import rule_moment_judgments as judgments_svc
@@ -126,6 +127,17 @@ async def autoinject_retrieve():
different things about what the reader holds —
so they get different lines (#4100). Shares the
ledger directory and the same clear-on-compact.
shapes_seen (opt) — comma-separated reply-shape keys this session was
already shown in full (milestone 500). The core
goes out in full when `core` is absent, as its
one-line reminder when present; the hook's
ledger is swept on compaction, which is how the
full core comes back after one.
THE TURN'S REPLY SHAPE COMES FIRST (milestone 500 step 3). Every turn ends
in a reply and this is the last point before it is written, so the core
shape and whatever is mounted on `reply.report` lead the payload. Fresh
shape keys come back as `shape_keys` for the ledger.
TWO ARMS, TWO SETS OF GATES. Rules ride the same hook and the same query
but nothing else: the notes menu can be disabled, thresholded and top-k'd
@@ -159,9 +171,19 @@ async def autoinject_retrieve():
g.user.id, result.get("lesson_ids") or [], rules.get("shown_rule_ids") or [],
arm="p", situation=q, project_id=project_id or None,
)
blocks = [b for b in (rules["context"], result["context"], proposal) if b]
# The reply mounts go through the same rule ledger; a rule the prompt arm
# just named is not quoted a second time by the turn's delivery.
turn = await moment_delivery_svc.deliver_for_turn(
g.user.id, seen=reply_shapes_svc.parse_seen(request.args.get("shapes_seen")),
project_id=project_id,
exclude=frozenset(exclude_rule_ids) | frozenset(rules["rule_ids"]),
held=frozenset(held_rule_ids),
)
blocks = [b for b in (turn["context"], rules["context"], result["context"], proposal) if b]
result["context"] = "\n\n".join(blocks)
result["rule_ids"] = rules["rule_ids"]
result["rule_ids"] = list(dict.fromkeys([*rules["rule_ids"], *turn["rule_ids"]]))
result["shape_keys"] = [k for k, form in turn["shape_forms"].items()
if form == reply_shapes_svc.FULL]
return jsonify(result)
@@ -275,12 +297,16 @@ async def moment():
`rule_ids` (the FRESH ones, for the hook to append to the ledger) and
`moments` (the names reached). A lookup, not a ranked search: no
retrieval_logs row; each fresh rule is recorded surfaced with its moment.
A moment that carries a default reply shape (milestone 500) brings it too,
ahead of the rules: in full unless `shapes_seen` names it, then as its
reminder. `shape_keys` are the ones sent in full, for the hook's ledger.
"""
event = await request.get_json(silent=True) or {}
tool = str(event.get("tool_name") or "").strip()
tool_input = event.get("tool_input")
if not tool:
return jsonify({"context": "", "rule_ids": [], "moments": []})
return jsonify({"context": "", "rule_ids": [], "moments": [], "shape_keys": []})
project_id, _repo, _unbound = await _project_scope()
reached, result = await moment_delivery_svc.deliver_for_act(
g.user.id, tool, tool_input if isinstance(tool_input, dict) else {},
@@ -288,10 +314,16 @@ async def moment():
exclude=frozenset(_int_list(request.args.get("exclude_rule_ids"))),
held=frozenset(_int_list(request.args.get("held_rule_ids"))),
)
shapes, forms = moment_delivery_svc.shapes_for_act(
reached, reply_shapes_svc.parse_seen(request.args.get("shapes_seen")),
)
await reply_shapes_svc.record_delivery(g.user.id, forms, via="hook")
context = "\n\n".join(b for b in ("\n\n".join(shapes), "\n".join(result.lines)) if b)
return jsonify({
"context": "\n".join(result.lines),
"context": context,
"rule_ids": result.rule_ids,
"moments": [hit["moment"] for hit in reached],
"shape_keys": [k for k, form in forms.items() if form == reply_shapes_svc.FULL],
})
@@ -335,28 +367,58 @@ async def reply_rules():
the reply asks something), and the reply text against every rule's
trigger — the backstop for whatever the earlier arms missed.
Body: `{"reply": "<text>"}`. Query: `repo` / `project_id` for the scope;
ONE END-OF-TURN REQUEST (milestone 500 step 4): a reply that closed tasks
is also checked here for the completion sections (services/report_check),
which used to be a Stop hook and an endpoint of its own. Both holds fold
into one `reason`; `report_check` says what the section check recorded, so
the hook knows the rewrite is one to report.
Body: `{"reply": "<text>", "closed": 1, "closed_task_ids": [41],
"rewrite": false}` — the last three only when the turn closed a task.
`closed` is the count and decides whether the check runs (a task created
already done has no id to list); `rewrite` marks the reply written after a
section hold, which is recorded and never held.
Query: `repo` / `project_id` for the scope;
`held_rule_ids` (opened this session — exempt, as at the act checkpoint);
`exclude_rule_ids` (named this session — not re-counted as surfaced);
`stopped_rule_ids` (already held a reply or an act this session — a rule
holds once, and the per-session cap counts these).
Returns `reason` — the words the hook blocks with, empty when nothing
holds — plus `rule_ids` for the hook's ledger and `moments` reached.
holds — plus `rule_ids` for the hook's ledger, `moments` reached and
`report_check` (the recorded outcome, or "" when nothing was checked).
"""
data = await request.get_json(silent=True) or {}
if not isinstance(data, dict):
data = {}
reply = str(data.get("reply") or "")
# `type(...) is int`, not isinstance: JSON `true` would otherwise read as 1.
closed = data.get("closed") if type(data.get("closed")) is int else 0
ids = data.get("closed_task_ids")
task_ids = [i for i in ids if type(i) is int and i > 0][:20] if isinstance(ids, list) else []
rewrite = data.get("rewrite") is True
project_id, _repo, _unbound = await _project_scope()
held = await moment_delivery_svc.reply_hold(
g.user.id, reply, project_id=project_id,
exclude=frozenset(_int_list(request.args.get("exclude_rule_ids"))),
held=frozenset(_int_list(request.args.get("held_rule_ids"))),
stopped=frozenset(_int_list(request.args.get("stopped_rule_ids"))),
)
checked: dict = {}
if closed > 0:
checked = await report_check_svc.check_reply(
g.user.id, reply, task_ids=task_ids, rewrite=rewrite, project_id=project_id or None,
)
# The rewrite after a hold goes out as written: recorded above, held by nothing.
held: dict = {}
if not rewrite:
held = await moment_delivery_svc.reply_hold(
g.user.id, reply, project_id=project_id,
exclude=frozenset(_int_list(request.args.get("exclude_rule_ids"))),
held=frozenset(_int_list(request.args.get("held_rule_ids"))),
stopped=frozenset(_int_list(request.args.get("stopped_rule_ids"))),
)
reasons = [r for r in (checked.get("reason", ""), held.get("reason", "")) if r]
return jsonify({
"reason": held.get("reason", ""),
"reason": "\n\n".join(reasons),
"rule_ids": held.get("rule_ids", []),
"moments": held.get("moments", []),
"report_check": checked.get("outcome", ""),
})
@@ -472,45 +534,6 @@ def _parse_shapes(raw: str) -> list[tuple[str, str]]:
return out
@plugin_bp.get("/report-check")
@login_required
async def report_check():
"""Record what a Stop hook found in a reply that closed a task (milestone 409 step 5).
The hook decides which completion sections the reply lacks — a local check
of text it can read — and reports the outcome here. For `blocked` the
response carries the `reason` to send the agent back with: the words are
the server's, so every client's hook says the same thing (plugin/PACKAGING.md).
A hook blocks only on a `reason` it received, which means only on a block
that was recorded.
A GET for the reason every plugin endpoint is one: a read-scoped key must
be enough to run the plugin, and this records telemetry the way /retrieve
records a retrieval log.
Query:
outcome (str) — passed | blocked | passed_after_rewrite |
missing_after_rewrite. Anything else is a 400.
missing (opt) — comma-separated sections the reply lacked:
"where it sits", "needs you", "next".
task_ids (opt) — comma-separated ids of the tasks the turn closed.
repo (opt) — working repo remote, resolved like the other arms.
"""
outcome = (request.args.get("outcome") or "").strip()
if outcome not in report_check_svc.OUTCOMES:
return jsonify({"error": f"outcome must be one of {list(report_check_svc.OUTCOMES)}"}), 400
missing = [m for m in (request.args.get("missing") or "").split(",") if m.strip()]
task_ids = _int_list(request.args.get("task_ids"))[:20]
project_id, _repo, _unbound = await _project_scope()
await report_check_svc.record_report_check(
g.user.id, outcome, missing=missing, task_ids=task_ids, project_id=project_id or None,
)
body: dict = {"status": "ok"}
if outcome == "blocked":
body["reason"] = report_check_svc.block_reason(missing)
return jsonify(body)
@plugin_bp.get("/shape-check")
@login_required
async def shape_check():
@@ -525,7 +548,7 @@ async def shape_check():
recorded, and never on the stop that follows one (`phase=after`).
A GET like every plugin endpoint: a read-scoped key runs the plugin, and
this records telemetry the way /report-check does. It changes no ledger
this records telemetry the way /reply-rules records its section check. It changes no ledger
row — the agent's own classify_shapes call does that.
Query:
+18
View File
@@ -20,6 +20,7 @@ from quart import Blueprint, jsonify, request
from scribe.auth import get_current_user_id, login_required
from scribe.services import moment_actions as moment_actions_svc
from scribe.services import moments as moments_svc
from scribe.services import reply_shapes as shapes_svc
from scribe.services import rule_moment_judgments as judgments_svc
from scribe.services import rulebooks as rulebooks_svc
from scribe.services.retrieval_telemetry import moment_usage
@@ -143,6 +144,23 @@ async def moments_route():
return jsonify(out)
@retrieval_bp.route("/reply-shapes", methods=["GET"])
@login_required
async def reply_shapes_route():
"""The default reply shapes and the moment each rides (milestone 500).
`list_reply_shapes`' payload from the same service, so the Settings view
and the session cannot show different defaults — plus what only the
operator's view needs (step 5): per shape, the preferences mounted on its
moment and its deliveries over `days` (default 30, 1–90).
"""
try:
days = min(90, max(1, int(request.args.get("days") or shapes_svc.OVERVIEW_DAYS)))
except ValueError:
days = shapes_svc.OVERVIEW_DAYS
return jsonify(await shapes_svc.overview(get_current_user_id(), days=days))
async def _mapping_change(change):
"""map/unmap from the browser: the MCP tools' service, recorded as human.
+101 -19
View File
@@ -37,6 +37,7 @@ from __future__ import annotations
import hashlib
import logging
import re
import statistics
from datetime import datetime, timedelta, timezone
from sqlalchemy import and_, delete, exists, func, not_, or_, select
@@ -735,17 +736,47 @@ async def confirmed_rules_in_scope(
# answered, and never by a sweep or a timer (#4183): the reader is in the
# situation then, and a nudge arriving anywhere else is one nobody acts on.
#
# WHAT "RESEMBLE" MEANS (#5193). Not a fixed similarity. Lessons are written
# in one register — a sentence of advice and the moment it applies — so an
# embedder scores any two of them well above two unrelated texts, and a
# constant bar sits inside that band: every no-rule lesson then "resembles"
# most of the others, and the group grows with the pool rather than with any
# situation recurring. The bar is instead relative to each lesson's own
# background: how far a neighbour stands above the typical score that lesson
# gets against every lesson in its register, measured in that register's own
# spread (median and MAD, so the near neighbours being judged do not move the
# yardstick). An install with a different embedder, or lessons in a different
# style, gets a different band and the same bar.
#
# And resemblance must be MUTUAL, among every member, not just to the lesson
# being answered. A lesson broad enough to stand near many others is a hub,
# and hub-and-spoke is how three unrelated lessons used to make a "group"
# through one general one. A clique is the shape of one situation met again.
#
# DEFAULTS, stated as defaults (rules 32, 115). Three lessons — the new one
# and two it resembles — is the smallest group that is a pattern rather than
# a pair. The similarity bar sits above the notes menu's ("worth showing")
# and below the duplicate gate's ("the same record"): these lessons should be
# about one situation without being one lesson written twice, which the
# duplicate gate already catches.
# a pair. One spread above the background is a modest bar for one direction
# alone; the strictness comes from requiring it of every pair in both
# directions, which unrelated lessons rarely all clear. Raise it if groups
# name lessons a reader would not call one situation; lower it if lessons a
# reader would group never meet.
CONVERGENCE_LESSONS = 3
CONVERGENCE_THRESHOLD = 0.65
# Candidates fetched before keeping the no-rule ones; most lessons near a
# situation may well have rules.
_CONVERGENCE_FETCH = 20
CONVERGENCE_STANDOUT = 1.0
# Below this many other lessons a median and its spread describe too little
# to say what stands out, so nothing is named.
CONVERGENCE_MIN_BACKGROUND = 8
# How many of a lesson's register the background is read from. A register
# larger than this is measured on its nearest part, which reads the
# background HIGH — the bar errs strict, toward naming nothing, never toward
# a group that is only the pool's size.
_CONVERGENCE_BACKGROUND = 200
# Neighbours checked for mutual resemblance, nearest first. Each costs one
# more search at the write, and a group large enough to need more is named
# just as well by its nearest members.
_CONVERGENCE_CANDIDATES = 8
# The consistency constant that makes a MAD estimate a standard deviation's
# scale on normal data, so CONVERGENCE_STANDOUT reads as "spreads".
_MAD_SCALE = 1.4826
def convergence_group(members: list[dict]) -> dict | None:
@@ -812,6 +843,38 @@ async def convergence_for(user_id: int, lesson_id: int) -> dict | None:
return None
def standout(scores: dict[int, float]) -> dict[int, float]:
"""How far each lesson stands above this one's background, in spreads.
`scores` is one lesson's similarity to every other lesson in its register,
itself left out. Pure, so the bar is testable. Empty when there is too
little background to measure, or none of it varies.
"""
if len(scores) < CONVERGENCE_MIN_BACKGROUND:
return {}
middle = statistics.median(scores.values())
spread = statistics.median(abs(v - middle) for v in scores.values()) * _MAD_SCALE
if spread <= 0:
return {}
return {i: (v - middle) / spread for i, v in scores.items()}
def converging(lesson_id: int, standouts: dict[int, dict[int, float]],
candidates: list[int]) -> list[int]:
"""The lesson, then each candidate that stands out to EVERY member so far
and every member to it. Candidates are taken nearest first, so the clique
grows around the closest situation rather than the first id. Pure."""
def near(a: int, b: int) -> bool:
return (standouts.get(a, {}).get(b, float("-inf")) >= CONVERGENCE_STANDOUT
and standouts.get(b, {}).get(a, float("-inf")) >= CONVERGENCE_STANDOUT)
group = [lesson_id]
for c in candidates:
if c not in group and all(near(c, m) for m in group):
group.append(c)
return group
async def _convergence_for(user_id: int, lesson_id: int) -> dict | None:
from scribe.services import lessons as lessons_svc
from scribe.services.embeddings import semantic_search_notes, trigger_title
@@ -819,14 +882,34 @@ async def _convergence_for(user_id: int, lesson_id: int) -> dict | None:
lesson = await lessons_svc.get_lesson(user_id, lesson_id)
if lesson is None:
return None
data = lesson.data if isinstance(lesson.data, dict) else {}
query = trigger_title(data.get("what") or lesson.title, lessons_svc.lesson_trigger(lesson))
found = await semantic_search_notes(
user_id, query, exclude_ids={int(lesson.id)}, limit=_CONVERGENCE_FETCH,
threshold=CONVERGENCE_THRESHOLD, note_type=(lessons_svc.LESSON_NOTE_TYPE,),
include_global_kinds=True, scope="browse",
)
answered = await _no_rule_ids([int(n.id) for _s, n in found])
async def background_of(note) -> list[tuple[float, Note]]:
"""`note`'s similarity to every other lesson it can reach — the
background it is measured against. No threshold: the bar is relative,
and needs the scores a fixed bar would have turned away. No
supersession penalty either: it would shift some scores and not
others, and the background is a measure of resemblance alone."""
data = note.data if isinstance(note.data, dict) else {}
query = trigger_title(data.get("what") or note.title, lessons_svc.lesson_trigger(note))
return await semantic_search_notes(
user_id, query, exclude_ids={int(note.id)}, limit=_CONVERGENCE_BACKGROUND,
threshold=-1.0, note_type=(lessons_svc.LESSON_NOTE_TYPE,),
include_global_kinds=True, scope="browse", demote_superseded=False,
)
lid = int(lesson.id)
found = await background_of(lesson)
notes = {int(n.id): n for _s, n in found}
standouts = {lid: standout({int(n.id): s for s, n in found})}
standing = [i for i, z in standouts[lid].items() if z >= CONVERGENCE_STANDOUT]
answered = await _no_rule_ids(standing)
candidates = sorted(answered, key=lambda i: (-standouts[lid][i], i))[:_CONVERGENCE_CANDIDATES]
# Too few stand out from here for any clique to reach the bar — skip the
# searches the mutual check would cost.
if len(candidates) < CONVERGENCE_LESSONS - 1:
return None
for c in candidates:
standouts[c] = standout({int(n.id): s for s, n in await background_of(notes[c])})
def member(note) -> dict:
return {
@@ -834,6 +917,5 @@ async def _convergence_for(user_id: int, lesson_id: int) -> dict | None:
"sources": lessons_svc.lesson_sources(note), "project_id": note.project_id,
}
return convergence_group(
[member(lesson)] + [member(n) for _s, n in found if int(n.id) in answered]
)
group = converging(lid, standouts, candidates)
return convergence_group([member(lesson)] + [member(notes[i]) for i in group[1:]])
+73 -10
View File
@@ -16,7 +16,7 @@ from __future__ import annotations
import logging
from scribe.services import moment_actions
from scribe.services import moment_actions, reply_shapes
from scribe.services import retrieval_pipeline as rp
logger = logging.getLogger(__name__)
@@ -38,23 +38,27 @@ async def reachable_tools(user_id: int) -> list[str]:
What the plugin's catch-all hook reads once per session window, so the
calls that cannot reach a mounted rule — most of them, and every one on
an install that has mounted nothing — never leave the machine. An action
counts only when its moment carries a mount. The skill loader counts
whenever anything is mounted: a stored process declares its own moments,
which only the load itself can resolve, and a load is rare enough that
asking costs nothing.
counts when its moment carries a mount or a default reply shape. The
skill loader counts whenever anything is mounted: a stored process
declares its own moments, which only the load itself can resolve, and a
load is rare enough that asking costs nothing.
"""
from scribe.services import rulebooks
mounted = await rulebooks.mounted_moments(user_id)
if not mounted:
return []
# A moment that carries a default reply shape (milestone 500) is worth a
# request whether or not anything is mounted on it: the shape is product,
# so every install has it.
shaped = {s.moment for s in reply_shapes.SHAPES.values() if s.key != reply_shapes.CORE_KEY}
mounted = set(await rulebooks.mounted_moments(user_id))
wanted = mounted | shaped
mappings = await moment_actions.list_mappings(user_id)
keys = {
moment_actions.tool_key(action.tool)
for action, _via in moment_actions.effective_actions(mappings)
if action.moment in mounted
if action.moment in wanted
}
keys.add(moment_actions.SKILL_TOOL)
if mounted:
keys.add(moment_actions.SKILL_TOOL)
return sorted(keys)
@@ -86,6 +90,58 @@ async def deliver_for_act(
return reached, result
def shapes_for_act(reached: list[dict], seen: frozenset[str] = frozenset()) -> tuple[list[str], dict[str, str]]:
"""The default reply shapes riding the moments this act reached (milestone 500).
Closing a task comes just before a completion report, a structured
question is an ask, opening a plan comes before a plan is put up for
review — so the shape for that reply arrives at the act, before the reply
is written. In full the first time a session meets it, as its reminder
after that; a door with no ledger passes `seen` empty.
"""
return reply_shapes.deliver(
reply_shapes.for_moments([hit["moment"] for hit in reached]), seen,
)
# What the per-turn delivery says reached `reply.report`: the turn has not
# ended yet, but every turn ends in a reply, and this is the last point
# before it is written.
_REACHED_BY_TURN = "the reply this turn will end with"
async def deliver_for_turn(
user_id: int, *, seen: frozenset[str] = frozenset(), project_id: int | None = None,
exclude: frozenset[int] = frozenset(), held: frozenset[int] = frozenset(),
) -> dict:
"""What every turn carries before its reply is written (milestone 500 step 3).
The core reply shape — in full once per session and again after a
compaction (the ledger is swept then), otherwise its one-line reminder —
and the rules and preferences mounted on `reply.report`, under the same
rule ledger as every other arm. The Stop hook's reply moment fires after
the reply exists, which is too late to shape it and is kept as the
backstop; this is the point before.
Returns `{context, rule_ids, shape_forms}`. Fails open to the core alone,
and the core itself never fails: it is a constant.
"""
blocks, forms = reply_shapes.deliver([reply_shapes.core()], seen)
rule_ids: list[int] = []
try:
reached = [{"moment": "reply.report", "tool": "", "match": _REACHED_BY_TURN,
"via": moment_actions.DEFAULT}]
result = await deliver_moments(
user_id, reached, project_id=project_id, exclude=exclude, held=held,
)
blocks.extend(result.lines)
rule_ids = result.rule_ids
except Exception: # noqa: BLE001 - the mounted half never costs the core
logger.debug("reply.report mounts not delivered for the turn", exc_info=True)
await reply_shapes.record_delivery(user_id, forms, via="turn")
return {"context": "\n\n".join(blocks), "rule_ids": rule_ids, "shape_forms": forms}
async def attach_moment_rules(
user_id: int, tool: str, arguments: dict | None, data: dict,
) -> dict:
@@ -114,6 +170,13 @@ async def attach_moment_rules(
"rule_ids": result.shown_rule_ids,
"open_with": "get_rule(id)",
}
# The reply shape for this moment, in full: no ledger reaches this
# door, and an act like closing a task is rare enough that the shape
# arriving each time costs less than one report written without it.
blocks, forms = shapes_for_act(_reached)
if blocks:
data["reply_shape"] = "\n\n".join(blocks)
await reply_shapes.record_delivery(user_id, forms, via="mcp")
except Exception: # noqa: BLE001 - a decoration never breaks the payload
logger.debug("moment rules for %s could not be attached", tool, exc_info=True)
return data
+2 -20
View File
@@ -601,25 +601,6 @@ PROMPTRULE_DEFAULT_THRESHOLD = SURFACES["prompt_rule"].floor_default
# at all, where the act arms never face it because a command is one thing.
PROMPTRULE_LIMIT = SURFACES["prompt_rule"].budget_default
# THE COMPLETION-REPORT ARM'S OWN BAR (services/reply_preferences.py).
#
# It borrowed PROMPTRULE_THRESHOLD_KEY when it shipped, which made the two
# arms one dial: an operator lowering the bar for their own prose moved this
# one with it, silently. That contradicts the rule every other bar here
# follows — one number cannot serve arms whose queries are different shapes —
# and this arm's query is the most different of all. The others score an
# operator's prose or a session's code, both of which vary per call; this one
# scores a FIXED string (`COMPLETION_QUERY`) against rule triggers, so its
# score for a given corpus is a constant. A constant that lands under the bar
# is not a quiet arm, it is a dead one, and nothing about the prose arm's
# traffic would ever reveal it.
#
# Kept at the prose arm's starting value rather than tuned: the split is what
# makes the two independently movable, and a default is a product decision
# that this install's corpus cannot settle (rule 115).
REPORTPREF_THRESHOLD_KEY = SURFACES["report_preference"].floor_key
REPORTPREF_DEFAULT_THRESHOLD = SURFACES["report_preference"].floor_default
def _slugify(text: str) -> str:
"""kebab-case slug for a skill directory name (a-z0-9 + single hyphens)."""
@@ -726,7 +707,8 @@ async def get_autoinject_config(user_id: int) -> dict:
THE TWO NUMBERS COME FROM THE REGISTRY NOW (#4102). They used to be read and
clamped here, and identically again in `get_writepath_config`, and again in
three rule arms, and once more in `reply_preferences`. That was
three rule arms, and once more in the completion-report arm (since
retired, milestone 500). That was
tolerable while the values were shipped constants. It stops being tolerable
once a tool is expected to MOVE them, because a tuning surface cannot be
consistent across arms that each spell their configuration differently.
-141
View File
@@ -1,141 +0,0 @@
"""The operator's own preferences for a completion report, at the moment it is written.
WHY THIS EXISTS (milestone 409 step 4)
The reporting-back skill ships DEFAULT shapes. An operator will want some of
them different ("my completion reports also say how it was tested", "decisions
as a numbered list"), and those adjustments are `preference` records. The gap
is the query: prompt-time retrieval matches the OPERATOR'S MESSAGE, and a shape
preference is about the REPLY. "Fix the flaky test" never retrieves "completion
reports should say how it was tested", so the preference is on file and never
arrives.
THE DECISION (operator, 2026-09-14, logged on the step): two deliveries, split
by reply kind.
- A COMPLETION REPORT has a moment the server can see — a task closing — so
the server retrieves for it and hands the matches back beside
`report_back`. That is this module.
- EVERY OTHER REPLY KIND (a finding, a decision, a handoff…) has no tool call
in front of it, so the reporting-back skill asks: it tells the agent to
`search(content_type="rule")` for that kind before writing.
Loading reply-shape preferences at session start was the rejected third option:
it is a small copy of the preloading milestone 394 retired, and just as
unmeasurable.
HOW A PREFERENCE SAYS IT IS ABOUT COMPLETION REPORTS
By its trigger, which is what it already has — no tag, no new column. A
preference's `when_to_apply` dominates its embedded document, so one written
for this moment ("writing the report after finishing a task") resembles
COMPLETION_QUERY below, and one about anything else does not. That keeps
delivery entirely in retrieval, as 394 decided, and leaves an operator nothing
new to learn: a preference reaches the completion report the same way every
other record reaches its moment.
THE BAR IS ITS OWN, AND THE FIRST READING EARNED IT (#3860)
This borrowed the prompt arm's key when it shipped, on the argument that a
fixed query against triggers is a different score distribution from an
operator's message against the same documents — true, and the reason the two
could not stay one dial. Five days of traffic settled it.
What the readout said: 69 calls, 69 declines, every one naming the SAME record
at the SAME score (rule 77 at 0.7194 against a 0.72 bar). That constancy is
the signature of this arm — `COMPLETION_QUERY` never varies, so for a given
corpus its best score is a constant, and a constant sitting under the bar is a
dead arm rather than a quiet one. The record it kept declining was about
reading a REQUEST, not about the shape of a report, so the decline was right
and the arm is healthy: this install simply has no completion-report
preference on file.
The bar stayed at 0.72, and the key moved out (REPORTPREF_THRESHOLD_KEY) so
that staying is a decision rather than a side effect of what the prose arm is
set to. A surface whose score cannot vary is the one surface where a borrowed
bar can be wrong forever without a single call looking unusual.
"""
from __future__ import annotations
import logging
from scribe.services import retrieval_pipeline as rp
from scribe.services.embeddings import semantic_search_rules
from scribe.services.retrieval_surfaces import SURFACES, budget_for, floor_for
from scribe.services.retrieval_telemetry import record_retrieval
from scribe.services.rule_usage import record_rule_surfaced
logger = logging.getLogger(__name__)
SOURCE = rp.REPORT_PREFERENCE.source
# Written in the vocabulary of the MOMENT, because that is what a trigger is
# written in and what this query is scored against. Domain-neutral on purpose
# (rule #115): a writing project or a home-infrastructure project closes tasks
# too, and its operator's preferences must match as well as a developer's.
COMPLETION_QUERY = (
"writing the completion report to the operator after finishing a task — "
"how that reply should be laid out and what it should include"
)
# A handful, not a menu. More than a few shape preferences for ONE kind of
# reply would contradict each other before they helped; the limit is here to
# keep one noisy corpus from turning a status change into a wall of text.
# The STARTING budget, not the budget (#4102). Both numbers this arm runs on
# now come from the surface registry, so the model that reads this arm's
# telemetry can move either — which matters more here than anywhere else,
# because a fixed query makes this arm's score a constant and a floor a hair
# above it produces a dead arm no amount of traffic will ever reveal.
LIMIT = SURFACES["report_preference"].budget_default
async def _threshold(user_id: int) -> float:
return await floor_for(user_id, SOURCE)
async def _limit(user_id: int) -> int:
return await budget_for(user_id, SOURCE)
async def completion_preferences(user_id: int, *, project_id: int | None = None) -> list[dict]:
"""The operator's preferences for a completion report, best match first.
KIND-FILTERED, for the reason `_reserve_slot_for_preference` gives: a
binding rule that happened to resemble the query would otherwise ride out
under a key that says "how the operator likes this written", which is a
claim about force the record does not make.
Every call is logged, the empty ones included: a surface that records only
the calls it liked reports a flawless clear-rate however badly its bar is
set. An install with no preferences at all logs nothing, because no search
ran (`searched` stays False) — that is not a decline.
Fails open to an empty list: this decorates a write that has already
happened, and a lookup that errors must not turn it into a failure.
"""
try:
# The stages — the kind filter, the unconditional call row, the
# fresh-only surfacing rows — are the one pipeline's (milestone 456).
# RANKED: this surface chose what it showed, so the name is in
# rule_usage.RANKED_SOURCES and its hits count toward pull-through.
# There is no session ledger here: a completion report is written
# once, so nothing it could repeat has been shown before.
result = await rp.run_rule_arm(
rp.REPORT_PREFERENCE,
rp.RuleMoment(
user_id=user_id, query=COMPLETION_QUERY, project_id=project_id,
),
floor=await _threshold(user_id), budget=await _limit(user_id),
io=rp.RuleIO(
search=semantic_search_rules,
record_retrieval=record_retrieval,
record_rule_surfaced=record_rule_surfaced,
),
)
return [
{"id": rule.id, "title": rule.title, "statement": rule.statement, "kind": "preference"}
for _score, rule in result.shown
]
except Exception: # noqa: BLE001 - a decoration never breaks its payload
logger.warning("completion preference lookup failed", exc_info=True)
return []
+363
View File
@@ -0,0 +1,363 @@
"""The default shapes of a reply, and the moment each one rides (milestone 500).
WHY THIS EXISTS
A reply is where the person the work is for finds out what happened, and they
were not there while it was done. Milestone 409 wrote the shapes that make a
reply readable to them — conclusion first, the shortest reply that carries the
answer, where the work stands — and put them in one bundled skill. A skill
arrives only when the agent decides to load it, so in a long session the shape
faded out of context exactly when replies were getting longer.
So the shapes are product content held here, on the server, and delivered by
the moments they belong to rather than by the agent remembering to ask:
- the CORE applies to every reply, so it rides every turn — in full once per
session and after a compaction, and as a one-line pointer otherwise;
- each SLICE belongs to one kind of reply, and arrives at the moment that
comes just before that kind is written: closing a task comes before a
completion report, a structured question before an ask, opening a plan
before a plan is put up for review.
The moment already knows which kind of reply is coming, so there is no "pick
the type, then load its shape" step for an agent to skip.
ONE COPY
This module is the single source. The `reporting-back` skill keeps the
long-form reference (a worked example, the reasoning) and points here; the
hooks carry timing and transport and say nothing of their own about shape. An
operator's adjustments are `preference` records mounted on the same moments,
and arrive beside the default — where the two differ, the preference is what
the operator asked for.
HOW TO WRITE ONE
Domain-neutral (rule 115): the work may be software, configuration,
infrastructure or anything else a person drives through an agent. "Evidence"
is what the reader could open to check; "verified" is whatever checking means
for that work. The test beside this module refuses software-only vocabulary.
Short, because it is paid for in every session's context. The core's budget is
asserted; a sentence added to it should replace one.
"""
from __future__ import annotations
import json
import logging
from dataclasses import dataclass
from scribe.services import moments
logger = logging.getLogger(__name__)
@dataclass(frozen=True)
class ReplyShape:
"""One piece of the default reply shape, and when it arrives."""
key: str
"""Stable identifier: what a delivery records and the Settings view keys on."""
title: str
"""What the piece is, as the Settings view names it."""
moment: str
"""The catalog moment this piece rides — where an operator's own
preferences for it are mounted too."""
delivered: str
"""When it arrives, said plainly for the Settings view."""
text: str
"""The shape itself, as the agent reads it."""
reminder: str
"""One line standing in for `text` once the session has been shown it:
the pointer form, which also serves as the reminder on later turns."""
CORE_KEY = "core"
# ~550 tokens at four characters a token. The core is paid for in every
# session, so this is the ceiling the test holds it to — not a target. It sits
# at ~2,050 with the judgement line (milestone 500 step 2): close to full, so a
# sentence added now has to replace one.
CORE_BUDGET_CHARS = 2200
# A slice is paid for only at its moment, and once per session; it can afford
# more than the core and still has to stay a slice, not the old skill again.
# The asks slice is the largest (~1,530) because it carries who decides what in
# full: handing a question back is what an ask is tempted to do.
SLICE_BUDGET_CHARS = 1600
_CORE = """\
Shape the reply around where the work stands, not the order you did things in. \
The reader was not there while you worked.
- **Conclusion first** — the result, the verdict or the question, then what supports it.
- **The shortest reply that carries the answer.** Length is work handed back to the \
reader. It is earned by a comparison they asked for, options that need laying side by \
side, or numbers that are the point.
- **One topic per section; bold the few things that matter.**
- **Plain words** — the reader's vocabulary, not names you coined while working.
- **Place the work** — the task or plan it belongs to, by id and title, read from the \
record rather than recalled. Work with no record behind it says so.
- **Answer the standing questions, even with "nothing"**: does anything need them, and \
what happens next. A section that explains earns its place only when it changes what \
they do or decide; the rest belongs in the record's log.
- **Decide what only you can see; ask about direction.** Whether a finding holds, what a \
measurement says, whether the work is done: settle it, act, and say why so they can \
overrule it. Priorities, choices they will live with, and anything hard to undo are theirs.
- **A decision they have made is the input to the work.** Act on it; reopen it only \
for new evidence, said once.
- **Holding an action until they say yes?** Put it near the top, under "Approval requested".
- **End with the one thing they need to decide or do, in bold** — or say there is none.
By kind:
- **Answer** — the answer, then what they could open to check it. An evaluation is \
verdict · what exists · the gaps · a recommendation.
- **Proposal** — two or three approaches, the trade-off of each, one recommendation. \
A review is findings ranked by how much they matter.
- **Finding** — what you found, why it happens, and what you decided and did about it.
- **Progress** — a line or two. **Blocked** — what stopped, what you tried, what you need.
- **Where are we** — the plan and its progress · done · open · needs you · next.
Before sending, read it as someone who was not there, reading quickly; then once more \
for what can go."""
_COMPLETION = """\
A completion report, from the `placement` the closing call returned:
- **Where this sits** — the plan and the step, or the task alone when it has no plan.
- **What now works** — outcomes the reader would notice ("you can now…", "X no longer…"), \
not the steps behind them; those belong in the task's log.
- **How / why** — only the decisions worth knowing, and how it was verified. Anything \
that could not be verified is said here rather than left to read as passed.
- **Needs you** — an action, an approval, a decision, or "nothing". It has to answer two \
questions yes: is it theirs to decide, and is work waiting on it? A question you could settle by \
reading or measuring something is work not yet done — settle it, say which way you went, \
and leave them free to overrule.
- **Next** — from `placement.next`, or say the plan is finished. An offer to fix \
something you found goes here."""
_ASKS = """\
Asking them to decide or to act — pick the shape:
- **Decision** — the question first · two to four options, each with what it changes · \
your recommendation first among them.
- **Clarification** — "My reading is X · the gap is Y · unless you say otherwise I'll do Z."
- **Handoff** (only they can do it) — the action · why it needs them · what it unblocks · \
what you will do after.
- **Approval** (ready, and holding for a yes) — under "Approval requested": exactly what \
happens on yes, one numbered item per change so they can approve part · why it needs \
them · how it is undone.
- **Conflict** (what you are about to do clashes with a rule, a plan or an earlier \
decision) — what it says · what you were about to do · where they clash · A or B?
**Who decides what.** You decide what only you can see the evidence for: whether a \
finding holds, what a measurement says, whether a record is right, whether the work is \
done, and anything you can settle by reading or measuring. Decide, act, and say why, so \
they can overrule it; "your call" on one of these hands them a question they cannot see \
into. Evenly balanced evidence is still yours — say which way you went and what would \
change your mind. They decide direction: what matters most, what the thing should be, a \
trade-off only they can price, anything they will live with afterwards — and every act \
that is hard to undo or reaches outside the work.
When an option rests on existing behaviour, say whose call that was — theirs (cite it) \
or a past session's nobody confirmed."""
_PLAN = """\
A plan put up for review, before starting:
- **Goal** — what done looks like, and why, in a sentence or two.
- **Steps** — in order, each small enough to check on its own.
- **How you will know it works** — something observable, not "it is written".
- **Open questions** — only the ones that are theirs to answer."""
SHAPES: dict[str, ReplyShape] = {s.key: s for s in (
ReplyShape(CORE_KEY, "Every reply", "reply.report",
"every turn — in full once per session and after a compaction, "
"otherwise as a one-line pointer", _CORE,
"conclusion first; the shortest reply that carries the answer; place the "
"work; settle what only you can see and ask about direction; end with the "
"one thing they need to do, or say there is none."),
ReplyShape("completion", "Completion report", "work.finish",
"when a piece of work is closed", _COMPLETION,
"where this sits · what now works · how / why, and how it was verified · "
"needs you (theirs to decide, and blocking) · next."),
ReplyShape("asks", "Asking them to decide or act", "reply.ask",
"when a structured question is put to them", _ASKS,
"the question first and your recommendation first among the options; "
"settle what you can see, and ask only about direction."),
ReplyShape("plan", "A plan for review", "work.plan",
"when a plan is opened", _PLAN,
"goal · steps · how you will know it works · open questions."),
)}
def core() -> ReplyShape:
return SHAPES[CORE_KEY]
def for_moments(names: list[str]) -> list[ReplyShape]:
"""The slices that ride these moments, in catalog order. The core is not a
slice: it rides every turn, whatever moment the turn reaches."""
wanted = set(names)
return [s for s in SHAPES.values() if s.key != CORE_KEY and s.moment in wanted]
def catalog() -> dict:
"""The shapes as plain data, for the MCP tool and the REST door alike."""
return {
"shapes": [
{"key": s.key, "title": s.title, "moment": s.moment,
"means": moments.MOMENTS[s.moment].means,
"delivered": s.delivered, "text": s.text}
for s in SHAPES.values()
],
"total": len(SHAPES),
}
# ── Delivery: what a door puts in front of the agent ────────────────────
# Said once, in the full form: the default is a floor an operator can raise.
_FULL_HEAD = ("Reply shape · {title} — the default shape for {what}. Where an operator "
"preference shown beside it differs, the preference is what they asked for.")
_POINTER = ("Reply shape · {title}, shown in full earlier this session "
"(list_reply_shapes has it): {reminder}")
FULL = "full"
POINTER = "pointer"
def render(shape: ReplyShape, *, full: bool) -> str:
"""The shape as a door delivers it: in full the first time a session sees
it, as its one-line reminder after that."""
if not full:
return _POINTER.format(title=shape.title, reminder=shape.reminder)
what = moments.MOMENTS[shape.moment].means
return _FULL_HEAD.format(title=shape.title, what=what) + "\n\n" + shape.text
def deliver(shapes: list[ReplyShape], seen: set[str] | frozenset[str]) -> tuple[list[str], dict[str, str]]:
"""Render these shapes against what the session was already shown.
Returns the blocks and `{key: form}` — every shape delivered and whether
it went out in full or as a pointer. The keys sent in full are the ones a
door's ledger appends; a door with no ledger passes `seen` empty and
every shape goes out in full.
"""
blocks, forms = [], {}
for shape in shapes:
full = shape.key not in seen
blocks.append(render(shape, full=full))
forms[shape.key] = FULL if full else POINTER
return blocks, forms
def parse_seen(raw: str | None) -> frozenset[str]:
"""A ledger's comma-separated keys, keeping only shapes that exist."""
return frozenset(k for k in (p.strip() for p in (raw or "").split(",")) if k in SHAPES)
async def record_delivery(user_id: int | None, forms: dict[str, str], *, via: str) -> None:
"""One `reply_shape` row per delivery: which shapes, in which form, by which
door. Fire-and-forget — telemetry never takes a shape away from the reader.
In AppLog beside 409's `report_check`, so the two readings of one reply —
was the shape delivered, did the reply carry it — sit in one place.
"""
if not forms:
return
try:
from scribe.models import async_session
from scribe.models.app_log import AppLog
async with async_session() as session:
session.add(AppLog(
category="plugin", user_id=user_id, action="reply_shape",
details=json.dumps({"shapes": forms, "via": via}),
))
await session.commit()
except Exception: # noqa: BLE001 - observation never breaks the observed
logger.debug("reply shape delivery not recorded", exc_info=True)
# ── The operator's view (milestone 500 step 5) ──────────────────────────
# The window the Settings view counts deliveries over — the same 30 days the
# moments panel beside it reads, so the two count the same stretch of work.
OVERVIEW_DAYS = 30
async def delivery_counts(user_id: int, *, days: int = OVERVIEW_DAYS) -> dict[str, dict[str, int]]:
"""How often each shape went out, in each form, over the last `days`.
Read from the `reply_shape` rows `record_delivery` writes — one per
delivery, naming each shape and whether it went whole or as its reminder.
A row whose details do not parse is skipped rather than guessed at.
"""
from datetime import datetime, timedelta, timezone
from sqlalchemy import select
from scribe.models import async_session
from scribe.models.app_log import AppLog
since = datetime.now(timezone.utc) - timedelta(days=days)
counts: dict[str, dict[str, int]] = {k: {FULL: 0, POINTER: 0} for k in SHAPES}
async with async_session() as session:
rows = (await session.execute(
select(AppLog.details).where(
AppLog.category == "plugin", AppLog.action == "reply_shape",
AppLog.user_id == user_id, AppLog.created_at >= since,
)
)).scalars().all()
for raw in rows:
try:
forms = json.loads(raw or "{}").get("shapes") or {}
except (ValueError, AttributeError):
continue
for key, form in forms.items() if isinstance(forms, dict) else ():
if key in counts and form in counts[key]:
counts[key][form] += 1
return counts
async def overview(user_id: int, *, days: int = OVERVIEW_DAYS) -> dict:
"""The Settings view's read: every shape, the operator's preferences that
adjust it, and how often it was delivered.
A preference adjusts a shape by being MOUNTED on the shape's moment — the
same lookup that delivers it beside the shape — so what this lists is
exactly what arrives with it. Global homes only: a project's own
preferences are that project's, and its rules tab lists them.
The counts are telemetry and fail open to `deliveries_failed` with "—" in
the view; the preference lookup is the substance and is allowed to fail
the request, so the view shows an error instead of "no preferences".
"""
from scribe.services import rulebooks
data = catalog()
try:
counts: dict[str, dict[str, int]] | None = await delivery_counts(user_id, days=days)
except Exception: # noqa: BLE001 - a missing count is shown as unknown
logger.warning("reply shape delivery counts failed", exc_info=True)
counts = None
for row in data["shapes"]:
mounted = await rulebooks.rules_on_moments(user_id, [row["moment"]])
row["preferences"] = [
{"id": rule.id, "title": rule.title, "statement": rule.statement}
for rule, _at in mounted if rule.kind == "preference"
]
row["deliveries"] = counts.get(row["key"]) if counts is not None else None
data["days"] = days
data["deliveries_failed"] = counts is None
return data
+74 -22
View File
@@ -1,38 +1,60 @@
"""The report-shape check: what the plugin's Stop hook found, and what it says.
"""The report-shape check: does a reply that closed a task carry the completion
sections, and what is it sent back with when it does not.
WHY THIS EXISTS (milestone 409 step 5)
Everything that helps an agent write a readable completion report arrives
BEFORE the reply is written. A client's Stop hook is the one moment the
finished reply exists, so it checks that a reply closing a task carries the
completion sections (where the work sits, what needs the operator, what comes
next), and reports what it found here. Two jobs live on this side:
BEFORE the reply is written. The Stop hook is the one moment the finished reply
exists, so a reply that closed a task is checked for the completion sections
(where the work sits, what needs the operator, what comes next). Two jobs:
- RECORDING the outcome, so the rate of `blocked` among checked replies is a
number milestone 409's last step can read rather than an impression.
- OWNING THE WORDS the agent is sent back with. A hook carries timing and
transport only (plugin/PACKAGING.md); guidance text comes from the server,
so a second client's hook gets the same instruction by calling the same
endpoint, and the wording changes in one place.
number rather than an impression (`category=plugin, action=report_check`).
- OWNING THE WORDS the agent is sent back with, so every client's hook says
the same thing (plugin/PACKAGING.md: hooks carry timing and transport).
ONE END-OF-TURN REQUEST (milestone 500 step 4). This used to be a Stop hook of
its own that checked the sections in shell and reported here; the reply-moment
check sent the same reply to `/reply-rules` a moment later. The hook now sends
the reply once, with the ids of the tasks the turn closed, and the section
check runs here beside the reply hold. The patterns are the ones the shell
used, matched on the words that carry the meaning rather than exact headings,
so a shape's wording can change without breaking this.
app_logs rather than a table of its own: one small event with a JSON detail is
what that table holds, it already has retention and an admin viewer, and
nothing here needs a join. If the numbers earn a readout, that is the moment to
decide whether they earn a table.
nothing here needs a join.
"""
from __future__ import annotations
import json
import logging
import re
from scribe.models import async_session
from scribe.models.app_log import AppLog
from scribe.services import reply_shapes
logger = logging.getLogger(__name__)
OUTCOMES = ("passed", "blocked", "passed_after_rewrite", "missing_after_rewrite")
# The sections a hook may name as missing, in the order the reason lists them.
# Anything else a client sends is dropped rather than echoed into an
# instruction the agent will follow.
SECTIONS = ("where it sits", "needs you", "next")
# The sections, in the order the reason lists them, each with what counts as
# having it. A bare id ("closed #41") does not place the work: naming a record
# by id AND title, or a step position, is the shape — the title is what spares
# the reader a lookup. "Needs you: nothing" counts; it is an answer.
SECTION_PATTERNS: dict[str, re.Pattern] = {
"where it sits": re.compile(
r'(#[0-9]+|milestone [0-9]+|task [0-9]+)[*_`]*\s*[*_`]*["“]|step [0-9]+ of [0-9]+', re.I),
"needs you": re.compile(
r"needs? (from )?you|nothing (is )?needed from you|your (call|decision)", re.I),
"next": re.compile(r"\bnext\b", re.I),
}
SECTIONS = tuple(SECTION_PATTERNS)
def missing_sections(reply: str) -> list[str]:
return [name for name, pat in SECTION_PATTERNS.items() if not pat.search(reply or "")]
def known_sections(missing: list[str]) -> list[str]:
@@ -43,16 +65,17 @@ def known_sections(missing: list[str]) -> list[str]:
def block_reason(missing: list[str]) -> str:
"""What the agent is told when its completion report is sent back.
Names what is missing and points at the reporting-back skill for the shape
rather than restating it — the skill owns the shape (decision #4027).
Names what is missing and carries the completion shape's one line rather
than the whole shape: the shape itself was delivered when the task closed,
and `list_reply_shapes` has it in full.
"""
listed = ", ".join(known_sections(missing)) or "the completion sections"
shape = reply_shapes.SHAPES["completion"]
return (
f"This turn closed a Scribe task, and the reply that ends it is missing: {listed}. "
"The operator reads this reply to find out where the work stands. Rewrite it as a "
"completion report (the reporting-back skill has the shape): where it sits — the task "
"or milestone by id and title, from `placement` — what now works, what needs them "
"(or \"nothing\"), and what comes next."
"The person the work is for reads this reply to find out where it stands. Rewrite it "
f"as a completion report — {shape.reminder} Take where it sits from `placement`, the "
"task or plan by id and title (`list_reply_shapes` has the shape in full)."
)
@@ -78,3 +101,32 @@ async def record_report_check(
details=json.dumps(details),
))
await session.commit()
async def check_reply(
user_id: int, reply: str, *, task_ids: list[int], rewrite: bool,
project_id: int | None = None,
) -> dict:
"""Check a reply that closed tasks, record the outcome, and say whether it is held.
`rewrite` is the reply written after an earlier hold: it is recorded and
never held again, so a hold costs one turn at most.
Returns {"outcome", "reason"} — `reason` empty unless the reply is held.
A hold happens only when its record was written: an outcome the numbers
cannot see is not one this check may act on, so a recording failure
degrades to no hold rather than to an unmeasured one.
"""
missing = missing_sections(reply)
if rewrite:
outcome = "missing_after_rewrite" if missing else "passed_after_rewrite"
else:
outcome = "blocked" if missing else "passed"
try:
await record_report_check(user_id, outcome, missing=missing, task_ids=task_ids,
project_id=project_id)
except Exception: # noqa: BLE001 - an unrecorded check never holds a reply
logger.warning("report check not recorded", exc_info=True)
return {"outcome": "", "reason": ""}
return {"outcome": outcome,
"reason": block_reason(missing) if outcome == "blocked" else ""}
@@ -117,7 +117,6 @@ _RESCORERS = {
"pre_tool_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
"reply_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
"prompt_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
"report_preference": lambda u, q, p: _rescore_rules(u, q, p, "preference"),
}
# The embedding table each surface's re-scorer reads. A migration from a corpus
@@ -131,7 +130,6 @@ _CORPUS = {
"pre_tool_rule": RuleEmbedding,
"reply_rule": RuleEmbedding,
"prompt_rule": RuleEmbedding,
"report_preference": RuleEmbedding,
}
+1 -36
View File
@@ -421,10 +421,6 @@ class Declared:
quiet_because: str = ""
"""Set when silence over an active window is correct, saying why (#2475)."""
fixed_query: bool = False
"""The arm always searches the same string, so its decline rate is 0% or
100% and `cannot_decline` says nothing about it."""
@dataclass(frozen=True)
class RankedSource:
@@ -523,37 +519,6 @@ WRITE_PATH_RULE = RuleArm(
),
declared=Declared("rules that may govern the file being written"),
)
# The completion report's preferences (milestone 409 step 4): a FIXED query,
# preferences only, read by update_task as records rather than as lines. Its
# query is `reply_preferences.COMPLETION_QUERY`.
REPORT_PREFERENCE = RuleArm(
"report_preference", band=False, compact_tail=False, checkpoint=False,
preference_slot=False, kind="preference",
tuning=Surface(
name="report_preference",
floor_key="kb_reportpref_threshold",
floor_default=0.72,
budget_key="kb_reportpref_top_k",
budget_default=3,
# THE ONE FIXED QUERY, and the reason this arm behaves unlike the rest.
# The others score something that varies per call; this one scores a
# constant string, so its top score for a given corpus is also a
# constant. A floor a hair above that constant is not a quiet arm, it
# is a dead one, and no amount of traffic will ever reveal it — which
# is precisely how this arm spent 69 calls declining the same record.
asks="a fixed question about how to lay out a completion report",
over="preferences",
fires="when a task finishes",
),
# `fixed_query`: COMPLETION_QUERY is a module constant, so this arm's top
# score is the same number on every call — measured at 0.791 across 45
# consecutive calls, with p10, p50, p90, min and max all identical. Five
# equal percentiles is the tell.
declared=Declared(
"the fixed question asked when a task finishes: how should this report read",
fixed_query=True,
),
)
# The backstop for every arm that ran earlier in the turn and missed: the
# finished reply against every rule's trigger. Its floor IS its stop bar
# (the reply_rule surface), so it is passed as both.
@@ -581,7 +546,7 @@ REPLY_RULE = RuleArm(
),
)
RULE_ARMS: tuple[RuleArm, ...] = (
WRITE_PATH_RULE, PRE_TOOL_RULE, PROMPT_RULE, REPORT_PREFERENCE, REPLY_RULE,
WRITE_PATH_RULE, PRE_TOOL_RULE, PROMPT_RULE, REPLY_RULE,
)
PREFERENCE_SLOT_SOURCE = "preference_slot"
+1 -25
View File
@@ -94,29 +94,6 @@ class Point:
quiet_because: str = ""
fixed_query: bool = False
"""Whether this arm always searches the SAME query string.
THE #3497 GUARD, ONE STEP OVER. `logs_unconditionally` below exists
because a warning computed over a LOGGING property read as a ranking
problem. This field exists because a warning computed over a QUERY-SHAPE
property does the same thing.
An arm with a fixed query scores against one constant. Its decline rate is
therefore 0% or 100% and nothing in between — which of the two depends
only on whether the bar sits below or above that single number. So "never
returned nothing" says nothing at all about whether a floor is applied,
and `cannot_decline` — whose whole remedy is "check that it applies its
floor" — is uninformative here and skips these arms.
What IS informative for them is the mirror image, and `reply_preferences`
names it in its own docstring: every call returning nothing means the bar
sits above the constant, no traffic will ever move it, and the arm is
dead. That has happened — 69 consecutive declines at 0.0006 under the bar
(see `retrieval_surfaces`) — so it gets its own warning rather than
inheriting one written for arms whose score can vary.
"""
logs_unconditionally: bool = True
"""Whether this arm writes a row even when it returns NOTHING.
@@ -144,8 +121,7 @@ POINTS: dict[str, Point] = dict([
# ambient or pulled.
*(_p(spec.source, UNBIDDEN, spec.declared.what,
expects_traffic=not spec.declared.quiet_because,
quiet_because=spec.declared.quiet_because,
fixed_query=spec.declared.fixed_query)
quiet_because=spec.declared.quiet_because)
for spec in RANKED),
# A LOOKUP, not a ranker (#4796): a record the operator named by number
# in the message, fetched by id. No score and no bar, so it writes no
+3 -3
View File
@@ -5,9 +5,9 @@ WHY THIS EXISTS
Six push arms each carried their own loose copy of the same shape: a settings
key, a default, and a limit that was usually a module constant nobody could
change. The read-and-clamp was written out separately in `plugin_context` (twice
over, for auto-inject and the write path), in three rule arms, and again as
`reply_preferences._threshold`. That was survivable while the numbers were
shipped constants an operator occasionally edited.
over, for auto-inject and the write path), in three rule arms, and again in
the completion-report arm (since retired, milestone 500). That was survivable
while the numbers were shipped constants an operator occasionally edited.
It stops being survivable once the numbers are meant to MOVE. The operator's
decision for this step:
@@ -591,22 +591,12 @@ def _compute_warnings(sources: dict, usage: dict, rule_usage: dict,
# arms once recorded only their hits, so their decline count was
# structurally zero and this warning would have fired on a LOGGING
# defect while pointing the reader at the threshold.
#
# FOUR NOW. A FIXED-QUERY arm is exempt for the same reason one step
# over (#4232): it scores against one constant, so its decline rate is
# 0% or 100% and never in between, and which one it is depends only on
# where the bar sits relative to that single number. "Never returned
# nothing" is then not evidence about the floor — it is arithmetic —
# and this warning's own remedy, "check that it applies its floor",
# cannot be answered from it. Those arms get `fixed_query_never_clears`
# below, which asks the question that IS answerable for them.
if (
calls >= min_calls
and (b.get("zero_result_calls") or 0) == 0
and point is not None
and point.kind == UNBIDDEN
and point.logs_unconditionally
and not point.fixed_query
):
out.append(_warn(
"cannot_decline",
@@ -617,46 +607,6 @@ def _compute_warnings(sources: dict, usage: dict, rule_usage: dict,
source=name, calls=calls, zero_result_calls=0,
))
# ── A fixed-query arm that never clears its bar ──────────────────
#
# The mirror image of `cannot_decline`, and the state that actually
# threatens these arms. `reply_preferences` names it in its own
# docstring: "a fixed query makes this arm's score a constant and a
# floor a hair above it produces a dead arm no amount of traffic will
# ever reveal."
#
# For an arm whose score can vary, a window of all-empty calls is
# ordinary — it means nothing matched, which is an answer. For one
# whose score is a constant it means the bar is above that constant,
# and no volume of further calls will ever produce a different result.
# The arm is not quiet; it is switched off, and nothing else in this
# readout would say so.
#
# It has happened: `report_preference` logged 69 consecutive declines
# at 0.0006 under the bar. Note what that incident also proves — the
# fix is NOT automatically to lower the floor. Reading the refused
# record showed the refusal was correct, so this warning sends the
# reader to `near_miss_samples` rather than to the dial.
if (
calls >= min_calls
and point is not None
and point.fixed_query
and point.logs_unconditionally
and (b.get("zero_result_calls") or 0) == calls
):
out.append(_warn(
"fixed_query_never_clears",
f"{calls} calls, every one of them empty — and this arm always "
f"searches the same query, so its score is a constant. That "
f"means the bar sits above it and no amount of further traffic "
f"will change the result: the arm is off, not quiet. Read the "
f"record it refused (`near_miss_samples`) before touching the "
f"floor — the last time this arm sat here, every percentile "
f"said lower it and the refused record showed the refusal was "
f"right.",
source=name, calls=calls, zero_result_calls=calls,
))
# ── Band hugs its floor ──────────────────────────────────────────
#
# Read on p10, the WEAKEST tenth of what the arm returned. If even
+14
View File
@@ -102,6 +102,20 @@ def _no_supersession():
yield
@pytest.fixture(autouse=True)
def _no_reply_shape_log():
"""Stub the reply-shape delivery row (milestone 500 step 3).
Autouse for _no_system_labels' reason: every turn's retrieval, every
moment request and every Scribe tool that reaches a shaped moment writes
one AppLog row, and those paths are tested without a database.
tests/test_reply_shapes.py binds the real function at import time,
before this patch runs.
"""
with patch("scribe.services.reply_shapes.record_delivery", AsyncMock()):
yield
@pytest.fixture(autouse=True)
def _no_system_labels():
"""Stub the menu's "which System is each line about?" lookup (#4364).
+16
View File
@@ -498,6 +498,22 @@ async def rule_row(rule_id: int):
return await s.get(Rule, rule_id)
# Software-only vocabulary. Scribe's product text speaks for any work a person
# drives through an agent — configuration, infrastructure, documents — so a
# catalog or a shipped shape that reads as software-only is wrong on most of
# the installs it ships to. The actions particular to one kind of work belong
# in data (mappings, rulebooks, preferences), not in the product's words.
DEV_ONLY = (r"\bCI\b", r"\bcommit", r"\bpull request", r"\bcode\b", r"\btest",
r"\bgit\b", r"\brepo(s|sitor\w*)?\b", r"\bbranch", r"\bcompil", r"\bbuild\b")
def dev_only_hits(text: str) -> list[str]:
"""The DEV_ONLY patterns this text contains, case-insensitively."""
import re
return [w for w in DEV_ONLY if re.search(w, text, re.IGNORECASE)]
def skill_text(name: str) -> str:
"""Everything a bundled skill states: its SKILL.md, then each reference file.
+20 -3
View File
@@ -71,6 +71,14 @@ def _live_session_context_source() -> str:
return src[start:start + 1 + nxt.start()] if nxt else src[start:]
def _delivered_shapes() -> str:
"""Every reply shape as a session receives it in full. Rendered rather than
read as source: the header a shape arrives under is guidance too."""
from scribe.services import reply_shapes
return "\n\n".join(reply_shapes.render(s, full=True) for s in reply_shapes.SHAPES.values())
def delivered_surfaces() -> dict[str, str]:
"""Every surface a session receives as guidance, by label.
@@ -81,6 +89,8 @@ def delivered_surfaces() -> dict[str, str]:
- `static` — the Claude Code adapter's static session context
- `commands` — the Claude Code adapter's slash commands
- `live` — the live session context the server builds
- `shapes` — the reply shapes the server delivers at their moments,
in the full form a session first sees (milestone 500)
"""
server = (ROOT / "src/scribe/mcp/server.py").read_text()
match = re.search(r'_INSTRUCTIONS = """(.*?)"""', server, re.S)
@@ -91,6 +101,7 @@ def delivered_surfaces() -> dict[str, str]:
"static": (ROOT / "plugin/hooks/scribe_static_context.md").read_text(),
"commands": "".join(p.read_text() for p in sorted((ROOT / "plugin/commands").glob("*.md"))),
"live": _live_session_context_source(),
"shapes": _delivered_shapes(),
}
# A skill is SKILL.md plus the reference files it links (#4398): one
# owner, however many files it is split across.
@@ -243,9 +254,15 @@ TOPICS: tuple[Topic, ...] = (
"you are the one who knows what the code you just wrote is"),
Topic("report back where the work stands", "skill:reporting-back", ("reporting-back", "placement"),
"take the placement from the record"),
Topic("the operator's own reply shapes come first", "skill:reporting-back",
("reply_preferences", 'content_type="rule"'),
"the operator's own shapes come first"),
# Milestone 500: the default shapes are server product, delivered at their
# moments with the operator's preferences mounted beside them — so the
# rule that a preference wins is stated where the shape arrives.
Topic("a preference beside a delivered shape is what the operator asked for", "shapes",
("preference", "reply shape"),
"where an operator preference shown beside it differs"),
Topic("every reply leads with its conclusion and stays short", "shapes",
("conclusion first", "the shortest reply that carries the answer", "what can go"),
"length is work handed back to the reader"),
# Milestone 409 step 7: the sections were being FILLED rather than chosen,
# so a reply could satisfy every heading and still be unreadable. These
# three are the discipline around the scaffold, not the scaffold itself.
+28
View File
@@ -21,6 +21,7 @@ guard here can fail (rule 167).
from __future__ import annotations
import json
import os
import re
import subprocess
from pathlib import Path
@@ -156,6 +157,33 @@ def test_a_value_survives_the_round_trip_the_hooks_actually_make(value):
assert got.rstrip("\n") == value.rstrip("\n")
ESCAPED = ["a · b", "an em dash — here", "a face 😀 here", "plain ascii"]
@pytest.mark.parametrize("locale", [None, "C", "C.UTF-8"])
@pytest.mark.parametrize("value", ESCAPED)
def test_an_escaped_character_decodes_to_utf8_in_any_locale(value, locale):
"""The server's JSON escapes every non-ASCII character (`\\u00b7`), and a
hook may run with no locale at all. The round trip above sends raw UTF-8
and never reaches the escape path; this one sends what the server sends,
from an environment with nothing but PATH."""
need_tools("bash", "awk")
env = {"PATH": os.environ["PATH"]}
if locale:
env["LC_ALL"] = locale
body = json.dumps(value)[1:-1]
r = subprocess.run(["bash", "-c", f'. "{DEFS}"; scribe_json_unescape'],
input=body.encode(), capture_output=True, env=env)
assert r.returncode == 0, r.stderr.decode()
assert r.stdout == (value + "\n").encode()
def test_the_utf8_guard_can_fail():
"""A lone byte where the encoded character belongs is what the old
decoder wrote; the comparison above has to tell them apart."""
assert "·\n".encode() != b"\xb7\n"
# --------------------------------------------------------------------------
# Reading: the shell side.
+88 -14
View File
@@ -53,30 +53,104 @@ def test_one_incident_written_up_three_times_is_not_a_recurring_situation():
assert links_svc.convergence_group(same) is None
def _note(lid, *, title="t", project=None, sources=()):
data = {"what": title, "when_to_apply": "a CI run overran"}
# ── The bar is relative to each lesson's own register (#5193) ──────────────
def test_too_little_background_says_nothing_stands_out():
few = {i: 0.6 for i in range(links_svc.CONVERGENCE_MIN_BACKGROUND - 1)}
assert links_svc.standout(few) == {}
def test_a_background_that_never_varies_says_nothing_stands_out():
assert links_svc.standout({i: 0.7 for i in range(20)}) == {}
def test_standing_out_is_measured_in_the_registers_own_spread():
"""The same neighbour score stands out in a tight register and not in a
loose one — the bar moves with the corpus, not with a constant."""
tight = {i: 0.70 + 0.01 * (i % 5) for i in range(20)} | {99: 0.80}
loose = {i: 0.60 + 0.06 * (i % 5) for i in range(20)} | {99: 0.80}
assert links_svc.standout(tight)[99] >= links_svc.CONVERGENCE_STANDOUT
assert links_svc.standout(loose)[99] < links_svc.CONVERGENCE_STANDOUT
def test_a_hub_near_two_lessons_that_are_not_near_each_other_is_no_clique():
up = links_svc.CONVERGENCE_STANDOUT + 1
standouts = {1: {2: up, 3: up}, 2: {1: up, 3: 0.0}, 3: {1: up, 2: 0.0}}
assert links_svc.converging(1, standouts, [2, 3]) == [1, 2]
def test_resemblance_must_run_both_ways():
up = links_svc.CONVERGENCE_STANDOUT + 1
standouts = {1: {2: up, 3: up}, 2: {1: 0.0, 3: up}, 3: {1: up, 2: up}}
assert links_svc.converging(1, standouts, [2, 3]) == [1, 3]
def _note(lid, *, title=None, project=None, sources=()):
data = {"what": title or f"lesson {lid}", "when_to_apply": "a CI run overran"}
if sources:
data["taught_by"] = list(sources)
return SimpleNamespace(
id=lid, title=title, note_type="lesson", body="", data=data,
id=lid, title=title or f"lesson {lid}", note_type="lesson", body="", data=data,
project_id=project, arose_from_id=None, tags=[],
created_at=None, updated_at=None,
)
# A register: lessons 1-3 resemble each other well above a background of
# filler lessons 10-29, which all sit in one flat band — the shape every
# lesson-to-lesson score has, because lessons share one register.
_TRIO = {1, 2, 3}
_FILLER = range(10, 30)
_NOTES = {i: _note(i, sources=[100 + i]) for i in (*_TRIO, *_FILLER)}
def _sim(a, b):
if a in _TRIO and b in _TRIO:
return 0.82
return 0.66 + 0.01 * ((a + b) % 5)
def _search_over(sim=_sim):
async def search(user_id, query, *, exclude_ids, **_kw):
(me,) = exclude_ids
return sorted(((sim(me, i), n) for i, n in _NOTES.items() if i != me),
key=lambda pair: -pair[0])
return search
def _run(no_rule, sim=_sim):
return (
patch.object(lessons_svc, "get_lesson", AsyncMock(return_value=_NOTES[1])),
patch("scribe.services.embeddings.semantic_search_notes", _search_over(sim)),
patch.object(links_svc, "_no_rule_ids",
AsyncMock(side_effect=lambda ids: {i for i in ids if i in no_rule})),
)
@pytest.mark.asyncio
async def test_lessons_that_stand_out_together_are_named():
a, b, c = _run(no_rule=set(_NOTES))
with a, b, c:
group = await _REAL(7, 1)
assert [m["id"] for m in group["lessons"]] == [1, 2, 3]
@pytest.mark.asyncio
async def test_a_large_no_rule_pool_in_one_flat_band_names_nothing():
"""The #5193 failure: every lesson answered "no rule", every score above
a fixed bar, and a group the size of the pool. Nothing stands out of a
flat band, so nothing is named however large the pool grows."""
a, b, c = _run(no_rule=set(_NOTES), sim=lambda x, y: 0.66 + 0.01 * ((x + y) % 5))
with a, b, c:
assert await _REAL(7, 1) is None
@pytest.mark.asyncio
async def test_only_lessons_answered_no_rule_join_the_group():
found = [(0.8, _note(2, sources=[11])), (0.7, _note(3, sources=[12])),
(0.7, _note(4, sources=[13]))]
with patch.object(lessons_svc, "get_lesson", AsyncMock(return_value=_note(1, sources=[10]))), \
patch("scribe.services.embeddings.semantic_search_notes", AsyncMock(return_value=found)), \
patch.object(links_svc, "_no_rule_ids", AsyncMock(return_value={2})):
assert await _REAL(7, 1) is None # only #2 answered: a pair
with patch.object(lessons_svc, "get_lesson", AsyncMock(return_value=_note(1, sources=[10]))), \
patch("scribe.services.embeddings.semantic_search_notes", AsyncMock(return_value=found)), \
patch.object(links_svc, "_no_rule_ids", AsyncMock(return_value={2, 4})):
group = await _REAL(7, 1)
assert [m["id"] for m in group["lessons"]] == [1, 2, 4]
a, b, c = _run(no_rule={1, 2}) # #3 has a rule: a pair is no group
with a, b, c:
assert await _REAL(7, 1) is None
@pytest.mark.asyncio
+15 -25
View File
@@ -4,8 +4,10 @@ The in-band half of milestone 409 step 3: the skill and static context only
exist in the Claude Code plugin, and a tool response reaches every MCP client
at the moment a piece of work closes. Pinned on the response, not the wording.
Step 4 adds the operator's own completion-report preferences beside the cue,
retrieved at that same moment — and only then, and only when there are any.
The operator's own completion-report preferences used to ride here too, from
a fixed-question search (409 step 4). Milestone 500 retired that: they are
mounted on finishing work and arrive in `moment_rules` beside the delivered
reply shape, so this response carries no key of its own for them.
"""
from unittest.mock import AsyncMock, MagicMock, patch
@@ -13,49 +15,37 @@ import pytest
pytestmark = pytest.mark.usefixtures("_bind_user")
_PREF = {"id": 41, "title": "Say how it was checked", "statement": "…", "kind": "preference"}
async def _update(prefs=None, owed=None, **kwargs):
async def _update(owed=None, **kwargs):
from scribe.mcp.tools.tasks import update_task
note = MagicMock(id=5, user_id=7, project_id=3, started_at="2026-10-06T09:00:00+00:00")
note.to_dict.return_value = {"id": 5}
lookup = AsyncMock(return_value=list(prefs or []))
with patch("scribe.mcp.tools.tasks.notes_svc.update_note", AsyncMock(return_value=note)), \
patch("scribe.mcp.tools.tasks.systems_tools.attach_systems", AsyncMock()), \
patch("scribe.mcp.tools.tasks.placement_svc.attach_placement", AsyncMock()), \
patch("scribe.mcp.tools.tasks.family_adoption_svc.owed_since",
AsyncMock(return_value=list(owed or []))), \
patch("scribe.mcp.tools.tasks.reply_prefs_svc.completion_preferences", lookup):
return await update_task(task_id=5, **kwargs), lookup
AsyncMock(return_value=list(owed or []))):
return await update_task(task_id=5, **kwargs)
@pytest.mark.parametrize("status", ["done", "cancelled"])
async def test_closing_a_task_carries_the_cue(status):
out, _ = await _update(status=status)
out = await _update(status=status)
assert "placement" in out["report_back"] and "needs them" in out["report_back"]
@pytest.mark.parametrize("kwargs", [{"status": "in_progress"}, {"status": "todo"}, {"body": "more notes"}])
async def test_other_updates_do_not(kwargs):
out, lookup = await _update(prefs=[_PREF], **kwargs)
assert "report_back" not in out and "reply_preferences" not in out
lookup.assert_not_awaited()
out = await _update(**kwargs)
assert "report_back" not in out
async def test_closing_hands_back_the_operators_report_preferences():
out, lookup = await _update(prefs=[_PREF], status="done")
assert out["reply_preferences"] == [_PREF]
# The cue says the key is there, so it never arrives unexplained.
assert "reply_preferences" in out["report_back"]
lookup.assert_awaited_once_with(7, project_id=3)
async def test_no_preferences_means_no_key_and_the_plain_cue():
async def test_closing_carries_the_plain_cue_and_no_preference_key_of_its_own():
"""The retired key does not come back beside the moment delivery."""
from scribe.mcp.tools.tasks import REPORT_BACK_CUE
out, _ = await _update(prefs=[], status="done")
out = await _update(status="done")
assert "reply_preferences" not in out
assert out["report_back"] == REPORT_BACK_CUE
@@ -69,11 +59,11 @@ async def test_closing_names_the_owed_adoptions_filed_while_the_task_was_open():
into a project during this task comes back for the report to name."""
from scribe.services.family_adoption import OWED_CUE
out, _ = await _update(owed=[_OWED], status="done")
out = await _update(owed=[_OWED], status="done")
assert out["family_owed"] == [_OWED]
assert out["report_back"].endswith(OWED_CUE) and "family_owed" in out["report_back"]
async def test_no_owed_adoptions_means_no_key():
out, _ = await _update(owed=[], status="done")
out = await _update(owed=[], status="done")
assert "family_owed" not in out
+128
View File
@@ -0,0 +1,128 @@
"""The DRY close-out measure (#4745).
Stdlib-only and kept in scripts/ so any project can run a scratch copy of it,
so these tests import it by path rather than as a package.
What matters is the after-only list: it is the part of the measure that finds
something a pass could not, so it must name a copy the change CREATED and stay
quiet about one that was already there. A list that repeats old duplication
every time is one nobody reads.
"""
import importlib.util
import pathlib
import shutil
import subprocess
import pytest
_PATH = pathlib.Path(__file__).resolve().parents[1] / "scripts" / "measure_duplication.py"
_spec = importlib.util.spec_from_file_location("measure_duplication", _PATH)
dup = importlib.util.module_from_spec(_spec)
_spec.loader.exec_module(dup)
BODY = """
def load(user_id, note_id):
# a comment the measure ignores
if not allowed(user_id, note_id):
raise ValueError("note {} not found".format(note_id))
with session() as s:
row = s.get(Row, note_id)
if row is None:
raise ValueError("not a row")
return row
"""
def _files(**texts):
return [(name, text.encode()) for name, text in texts.items()]
def test_comments_blanks_and_bare_punctuation_are_not_code():
lines = dup.significant_lines('x = 1\n\n// note\n# note\n });\n-- sql note\ny = "two"\n')
assert [t for _, t in lines] == ["x = 1", 'y = "S"']
assert [n for n, _ in lines] == [1, 7]
def test_a_preprocessor_line_is_code_and_a_hash_comment_is_not():
lines = dup.significant_lines("#include <x.h>\n# a comment\n#!/bin/sh\n")
assert [t for _, t in lines] == ["#include <x.h>"]
def test_copies_that_differ_only_in_their_strings_are_one_copy():
other = BODY.replace('"not a row"', '"no such row"')
result = dup.measure(_files(**{"a.py": BODY, "b.py": other}), [], [], window=6)
py = result["py"]
assert py["dup_windows"] > 0
assert py["dup_lines"] == py["lines"]
assert py["share"] == 1.0
def test_unrelated_files_measure_zero():
other = "\n".join(f"v{i} = compute({i})" for i in range(20))
result = dup.measure(_files(**{"a.py": BODY, "b.py": other}), [], [], window=6)
assert result["py"]["dup_lines"] == 0
def test_the_first_matching_group_claims_a_file():
groups = [("tests", ["tests/*"]), ("py", ["*.py"])]
assert dup.group_of("tests/test_x.py", groups) == "tests"
assert dup.group_of("src/x.py", groups) == "py"
assert dup.group_of("README.md", groups) is None
def test_without_groups_only_source_extensions_are_measured():
assert dup.group_of("src/x.go", []) == "go"
assert dup.group_of("docs/x.md", []) is None
assert dup.group_of("Makefile", []) is None
def test_excluded_paths_are_not_measured():
result = dup.measure(
_files(**{"a.py": BODY, "gen/b.py": BODY}), [], ["gen/*"], window=6,
)
assert result["py"]["dup_lines"] == 0
def test_a_copy_the_change_created_is_listed_after_only():
before = dup.measure(_files(**{"a.py": BODY, "c.py": "z = 1\n"}), [], [], window=6)
after = dup.measure(_files(**{"a.py": BODY, "c.py": BODY}), [], [], window=6)
fresh = dup.only_after(before, after)
assert [e["files"] for e in fresh["py"]] == [["a.py", "c.py"]]
assert fresh["py"][0]["at"] == ["a.py:2", "c.py:2"]
def test_a_copy_that_was_already_there_is_not_listed():
both = _files(**{"a.py": BODY, "b.py": BODY})
before = dup.measure(both, [], [], window=6)
after = dup.measure(both, [], [], window=6)
assert dup.only_after(before, after) == {}
def test_the_report_shows_before_and_after_side_by_side():
before = dup.measure(_files(**{"a.py": BODY}), [], [], window=6)
after = dup.measure(_files(**{"a.py": BODY, "b.py": BODY}), [], [], window=6)
text = dup.report(before, after, dup.only_after(before, after), show=5)
assert "0.0% → 100.0%" in text
assert "a.py:2, b.py:2" in text
@pytest.mark.skipif(shutil.which("git") is None, reason="needs git")
def test_a_revision_is_read_without_a_checkout(tmp_path, capsys):
def git(*args):
subprocess.run(
["git", "-C", str(tmp_path), "-c", "user.name=t", "-c", "user.email=t@t", *args],
check=True, capture_output=True,
)
git("init", "-q")
(tmp_path / "a.py").write_text(BODY)
git("add", "a.py")
git("commit", "-qm", "one copy")
(tmp_path / "b.py").write_text(BODY)
git("add", "b.py")
assert dup.main(["--repo", str(tmp_path), "--before", "HEAD"]) == 0
out = capsys.readouterr().out
assert "Duplicated only after: 1 file set(s)" in out
assert "a.py:2, b.py:2" in out
+130 -14
View File
@@ -185,18 +185,23 @@ async def _reachable(mounted, mappings=()):
return await md.reachable_tools(1)
async def test_an_install_with_nothing_mounted_keeps_the_hook_off_the_wire():
listing = AsyncMock()
with patch.object(rulebooks, "mounted_moments", AsyncMock(return_value=set())), \
patch.object(moment_actions, "list_mappings", listing):
assert await md.reachable_tools(1) == []
listing.assert_not_awaited()
# The tools the shipped defaults map onto a moment that carries a default reply
# shape (milestone 500): closing work, a structured question, opening a plan.
SHAPED = ["askuserquestion", "create_milestone", "enterplanmode", "exitplanmode",
"skill", "start_planning", "update_milestone", "update_task"]
async def test_only_the_tools_whose_moments_carry_a_mount_are_listed():
assert await _reachable({"work.deliver"}) == ["bash", "skill"]
assert await _reachable({"work.finish"}) == ["skill", "update_milestone", "update_task"]
assert await _reachable({"skill.release"}) == ["skill"]
async def test_an_install_with_nothing_mounted_still_asks_about_the_shaped_moments():
"""The reply shapes are product, so every install has them: the hook asks
about the acts that bring one, and about nothing else."""
assert await _reachable(set()) == SHAPED
async def test_only_the_tools_whose_moments_carry_a_mount_or_a_shape_are_listed():
assert await _reachable({"work.deliver"}) == sorted([*SHAPED, "bash"])
assert await _reachable({"work.finish"}) == SHAPED
assert await _reachable({"skill.release"}) == SHAPED
assert "bash" not in await _reachable(set())
async def test_the_skill_loader_counts_whenever_anything_is_mounted():
@@ -232,7 +237,8 @@ async def test_the_route_reads_the_event_and_the_ledger():
with patch.object(routes.moment_delivery_svc, "deliver_for_act", deliver):
resp = await routes.moment.__wrapped__()
body = await resp.get_json()
assert body == {"context": "a line", "rule_ids": [5], "moments": ["work.deliver"]}
assert body == {"context": "a line", "rule_ids": [5], "moments": ["work.deliver"],
"shape_keys": []}
args, kw = deliver.await_args
assert args == (7, "Bash", {"command": "git push"})
assert kw == {"project_id": 3, "exclude": frozenset({9}), "held": frozenset({4})}
@@ -245,7 +251,8 @@ async def test_the_route_answers_an_empty_event_with_nothing():
async with app.test_request_context("/api/plugin/moment", method="POST", json={}):
g.user = SimpleNamespace(id=7)
resp = await routes.moment.__wrapped__()
assert await resp.get_json() == {"context": "", "rule_ids": [], "moments": []}
assert await resp.get_json() == {"context": "", "rule_ids": [], "moments": [],
"shape_keys": []}
def test_both_routes_are_on_the_app():
@@ -273,13 +280,15 @@ def test_the_hook_and_the_routes_agree_on_every_name():
assert "scribe_rules_live" in src and "scribe_rules_append" in src
# The SHARED ledger, not one of its own.
assert '"${TMPDIR:-/tmp}/scribe-priorart"' in src and ".rules.ids" in src
for field in (".tools", ".rule_ids", ".context"):
for field in (".tools", ".rule_ids", ".context", ".shape_keys"):
assert f"'{field}'" in src
assert "shapes_seen" in src and "scribe_shapes_append" in src
from scribe.routes import plugin as routes
route = inspect.getsource(routes.moment)
for name in ("exclude_rule_ids", "held_rule_ids", "tool_name", "tool_input"):
for name in ("exclude_rule_ids", "held_rule_ids", "tool_name", "tool_input",
"shapes_seen", "shape_keys"):
assert name in route
@@ -380,3 +389,110 @@ def test_every_scribe_tool_the_defaults_name_attaches_its_moment_rules():
f"attach its mounted rules — a client without the plugin would never "
f"receive them"
)
# ── reply shapes ride the moments (milestone 500 step 3) ────────────────
from scribe.services import reply_shapes # noqa: E402
FINISH = [{"moment": "work.finish", "tool": "update_task", "match": "status=done", "via": "default"}]
ASK = [{"moment": "reply.ask", "tool": "AskUserQuestion", "match": "", "via": "default"}]
def test_an_act_brings_the_shape_for_the_reply_it_comes_before():
blocks, forms = md.shapes_for_act(FINISH)
assert forms == {"completion": reply_shapes.FULL}
assert reply_shapes.SHAPES["completion"].text in blocks[0]
_blocks, forms = md.shapes_for_act(ASK, frozenset({"asks"}))
assert forms == {"asks": reply_shapes.POINTER}
def test_an_act_at_an_unshaped_moment_brings_no_shape():
assert md.shapes_for_act(PUSH) == ([], {})
async def test_a_turn_carries_the_core_in_full_until_the_session_holds_it():
with patch.object(md, "deliver_moments", AsyncMock(return_value=rp.RuleResult())):
first = await md.deliver_for_turn(1)
later = await md.deliver_for_turn(1, seen=frozenset({"core"}))
assert first["shape_forms"] == {"core": reply_shapes.FULL}
assert reply_shapes.core().text in first["context"]
assert later["shape_forms"] == {"core": reply_shapes.POINTER}
assert reply_shapes.core().text not in later["context"]
assert reply_shapes.core().reminder in later["context"]
async def test_a_turn_brings_what_is_mounted_on_reply_report_after_the_core():
deliver = AsyncMock(return_value=rp.RuleResult(lines=["Preference that may apply"], rule_ids=[173]))
with patch.object(md, "deliver_moments", deliver):
out = await md.deliver_for_turn(1, project_id=2, exclude=frozenset({9}), held=frozenset({4}))
assert out["context"].index(reply_shapes.core().title) < out["context"].index("Preference")
assert out["rule_ids"] == [173]
reached = deliver.await_args.args[1]
assert [hit["moment"] for hit in reached] == ["reply.report"]
assert deliver.await_args.kwargs == {"project_id": 2, "exclude": frozenset({9}),
"held": frozenset({4})}
async def test_a_failed_mount_lookup_still_delivers_the_core():
with patch.object(md, "deliver_moments", AsyncMock(side_effect=RuntimeError("db down"))):
out = await md.deliver_for_turn(1)
assert reply_shapes.core().text in out["context"] and out["rule_ids"] == []
async def test_a_turn_records_its_delivery():
with patch.object(md, "deliver_moments", AsyncMock(return_value=rp.RuleResult())):
await md.deliver_for_turn(1, seen=frozenset({"core"}))
reply_shapes.record_delivery.assert_awaited_with(1, {"core": reply_shapes.POINTER}, via="turn")
async def test_a_task_closed_through_the_tool_carries_the_completion_shape():
data = {"id": 40, "status": "done", "project_id": 2}
out = await _with(_stub_db([]), lambda: md.attach_moment_rules(
1, "update_task", {"status": "done"}, data,
))
assert reply_shapes.SHAPES["completion"].text in out["reply_shape"]
assert "moment_rules" not in out # nothing mounted, still shaped
async def test_an_unshaped_act_through_a_tool_carries_no_shape():
out = await _with(_stub_db([]), lambda: md.attach_moment_rules(
1, "create_note", {"project_id": 2}, {"id": 3},
))
assert "reply_shape" not in out
async def test_the_route_puts_the_shape_ahead_of_the_rules_and_returns_its_key():
from scribe.routes import plugin as routes
deliver = AsyncMock(return_value=(ASK, rp.RuleResult(lines=["a line"], rule_ids=[5])))
app = Quart(__name__)
event = {"tool_name": "AskUserQuestion", "tool_input": {}}
async with app.test_request_context("/api/plugin/moment", method="POST", json=event):
g.user = SimpleNamespace(id=7)
with patch.object(routes.moment_delivery_svc, "deliver_for_act", deliver):
body = await (await routes.moment.__wrapped__()).get_json()
assert body["shape_keys"] == ["asks"]
assert body["context"].index("Who decides what") < body["context"].index("a line")
async with app.test_request_context("/api/plugin/moment", method="POST", json=event,
query_string={"shapes_seen": "asks"}):
g.user = SimpleNamespace(id=7)
with patch.object(routes.moment_delivery_svc, "deliver_for_act", deliver):
body = await (await routes.moment.__wrapped__()).get_json()
assert body["shape_keys"] == []
assert "Who decides what" not in body["context"]
def test_the_hook_carries_the_shape_ledger_between_calls(tmp_path):
replies = {"/api/plugin/moment-tools": b'{"tools":["askuserquestion"]}',
"/api/plugin/moment": json.dumps(
{"context": "Reply shape", "rule_ids": [], "moments": ["reply.ask"],
"shape_keys": ["asks"]}).encode()}
with http_sink(by_path=replies) as (port, seen):
_hook(tmp_path, port, tool="AskUserQuestion")
_hook(tmp_path, port, tool="AskUserQuestion")
moment_calls = [e for e in seen if e["_path"] == "/api/plugin/moment"]
assert "shapes_seen" not in moment_calls[0]
assert moment_calls[1]["shapes_seen"] == ["asks"]
+6 -13
View File
@@ -10,7 +10,7 @@ import re
import pytest
from scribe.services import moments
from tests.helpers import FakeMCP
from tests.helpers import FakeMCP, dev_only_hits
_NAME = re.compile(r"^[a-z]+\.[a-z]+$")
@@ -38,26 +38,18 @@ def test_the_skill_family_is_not_a_catalog_entry():
assert not any(k.startswith(moments.SKILL_PREFIX) for k in moments.MOMENTS)
# Software-only vocabulary, the same guard the completion query carries. The
# moments are named for any work a person drives through an agent; the actions
# particular to one kind of work belong in the mappings, not in the meaning.
_DEV_ONLY = (r"\bCI\b", r"\bcommit", r"\bpull request", r"\bcode\b", r"\btest",
r"\bgit\b", r"\brepo(s|sitor\w*)?\b", r"\bbranch", r"\bcompil", r"\bbuild\b")
@pytest.mark.parametrize("field", ["means", "reached_by"])
def test_the_catalog_assumes_no_particular_domain(field):
found = {
m.name: hits
for m in [*moments.MOMENTS.values(), moments.SKILL_FAMILY]
if (hits := [w for w in _DEV_ONLY
if re.search(w, getattr(m, field), re.IGNORECASE)])
if (hits := dev_only_hits(getattr(m, field)))
}
assert not found, f"software-only vocabulary in `{field}`: {found}"
def test_the_guard_can_fail():
assert re.search(_DEV_ONLY[1], "after the commit lands", re.IGNORECASE)
assert dev_only_hits("after the commit lands")
@pytest.mark.parametrize("name", list(moments.MOMENTS))
@@ -116,11 +108,12 @@ def test_the_tools_are_registered_and_classified():
mcp = FakeMCP()
tool.register(mcp)
assert mcp.names == [
"list_moments", "map_action", "unmap_action",
"list_moments", "list_reply_shapes", "map_action", "unmap_action",
"rules_to_mount", "propose_rule_moments", "rule_moment_proposals",
"judge_rule_moments", "rule_misfired",
]
assert {"list_moments", "rules_to_mount", "rule_moment_proposals"} <= _READ_ONLY_TOOLS
assert {"list_moments", "list_reply_shapes",
"rules_to_mount", "rule_moment_proposals"} <= _READ_ONLY_TOOLS
# A confirm mounts a rule, so a read key must not reach it.
assert {"map_action", "unmap_action",
"propose_rule_moments", "judge_rule_moments", "rule_misfired"} <= _WRITE_TOOLS
+7 -2
View File
@@ -132,6 +132,11 @@ def test_the_scaffold_itself_is_untouched():
for section in ("where this sits", "what now works", "how / why",
"needs you", "next"):
assert section in text, f"completion-report section {section!r} is gone"
for kind in ("completion", "finding", "blocked / failed", "progress",
# The kinds live in the delivered shapes since milestone 500 step 4; the
# skill keeps the reasoning and the worked completion report.
from scribe.services import reply_shapes
shapes = " ".join(s.title + " " + s.text for s in reply_shapes.SHAPES.values()).lower()
for kind in ("completion", "finding", "blocked", "progress",
"decision", "clarification", "handoff", "approval", "conflict"):
assert kind in text, f"reply kind {kind!r} is gone"
assert kind in shapes, f"reply kind {kind!r} is gone"
+103 -1
View File
@@ -197,7 +197,8 @@ async def test_the_route_passes_the_reply_and_all_three_ledgers():
with patch.object(routes.moment_delivery_svc, "reply_hold", hold):
resp = await routes.reply_rules.__wrapped__()
body = await resp.get_json()
assert body == {"reason": "Held.", "rule_ids": [11], "moments": ["reply.report"]}
assert body == {"reason": "Held.", "rule_ids": [11], "moments": ["reply.report"],
"report_check": ""}
assert hold.await_args.args == (7, "done")
assert hold.await_args.kwargs == {
"project_id": 3, "exclude": frozenset({5}), "held": frozenset({4}),
@@ -205,6 +206,55 @@ async def test_the_route_passes_the_reply_and_all_three_ledgers():
}
async def _reply_route(body, *, hold_reason="", checked=None):
from scribe.routes import plugin as routes
hold = AsyncMock(return_value={"reason": hold_reason, "rule_ids": [11] if hold_reason else [],
"moments": ["reply.report"]})
check = AsyncMock(return_value=checked or {"outcome": "passed", "reason": ""})
app = Quart(__name__)
async with app.test_request_context("/api/plugin/reply-rules", method="POST", json=body,
query_string={"project_id": "3"}):
g.user = type("U", (), {"id": 7})()
with patch.object(routes.moment_delivery_svc, "reply_hold", hold), \
patch.object(routes.report_check_svc, "check_reply", check):
resp = await routes.reply_rules.__wrapped__()
return await resp.get_json(), hold, check
async def test_a_turn_that_closed_nothing_is_not_section_checked():
_body, _hold, check = await _reply_route({"reply": "done"})
check.assert_not_awaited()
async def test_a_closing_turn_is_section_checked_and_both_holds_fold_into_one_reason():
"""One end-of-turn request (milestone 500 step 4): the section check and
the rule hold answer together, in one `reason`."""
body, _hold, check = await _reply_route(
{"reply": "done", "closed": 1, "closed_task_ids": [41, True, "x"], "rewrite": False},
hold_reason="RULE HOLD", checked={"outcome": "blocked", "reason": "SECTIONS"},
)
assert body["reason"] == "SECTIONS\n\nRULE HOLD"
assert body["report_check"] == "blocked"
# JSON `true` is not task 1, and a string is not an id.
assert check.await_args.kwargs == {"task_ids": [41], "rewrite": False, "project_id": 3}
async def test_a_task_created_already_done_is_checked_without_an_id():
_body, _hold, check = await _reply_route({"reply": "done", "closed": 1, "closed_task_ids": []})
assert check.await_args.kwargs["task_ids"] == []
async def test_the_rewrite_is_recorded_and_held_by_nothing():
body, hold, check = await _reply_route(
{"reply": "done", "closed": 1, "closed_task_ids": [41], "rewrite": True},
checked={"outcome": "passed_after_rewrite", "reason": ""},
)
assert check.await_args.kwargs["rewrite"] is True
hold.assert_not_awaited()
assert body["reason"] == "" and body["report_check"] == "passed_after_rewrite"
# ── the hook ────────────────────────────────────────────────────────────
@@ -251,6 +301,58 @@ def test_the_hook_blocks_in_the_servers_words_and_records_the_hold(tmp_path):
assert seen[1]["exclude_rule_ids"] == ["11"]
def _closing_transcript(tmp_path, *replies):
lines = [
{"type": "user", "message": {"role": "user", "content": "please finish it"}},
{"type": "assistant", "message": {"content": [
{"type": "tool_use", "id": "toolu_1", "name": "mcp__plugin_scribe_scribe__update_task",
"input": {"task_id": 41, "status": "done"}}]}},
{"type": "user", "message": {"content": [
{"type": "tool_result", "tool_use_id": "toolu_1", "is_error": False, "content": "{}"}]}},
] + [{"type": "assistant", "message": {"content": [{"type": "text", "text": r}]}} for r in replies]
path = tmp_path / "t.jsonl"
path.write_text("\n".join(json.dumps(x, separators=(",", ":")) for x in lines) + "\n")
return path
SECTION_HOLD = json.dumps({"reason": "Rewrite as a completion report.", "rule_ids": [],
"moments": ["reply.report"], "report_check": "blocked"}).encode()
def test_a_turn_that_closed_nothing_sends_no_close(tmp_path):
t = _transcript(tmp_path, "Done.")
with http_sink(by_path={"/api/plugin/reply-rules": QUIET}) as (port, seen):
_run(tmp_path, port, t)
assert "closed" not in json.loads(seen[0]["_body"])
def test_a_section_hold_blocks_once_then_the_rewrite_is_reported_and_never_held(tmp_path):
"""The completion-section check, folded into the one Stop request: the
first stop is held in the server's words, the rewrite goes back once with
`rewrite: true` so its outcome is recorded, and nothing after that is sent."""
t = _closing_transcript(tmp_path, "All done, pushed it.")
with http_sink(by_path={"/api/plugin/reply-rules": SECTION_HOLD}) as (port, seen):
out = json.loads(_run(tmp_path, port, t))
assert out == {"decision": "block", "reason": "Rewrite as a completion report."}
first = json.loads(seen[0]["_body"])
assert (first["closed"], first["closed_task_ids"], first["rewrite"]) == (1, [41], False)
rewritten = _closing_transcript(tmp_path, "All done, pushed it.", "Where this sits: …")
assert _run(tmp_path, port, rewritten, active=True) == ""
assert json.loads(seen[1]["_body"])["rewrite"] is True
# The marker is spent: a further stop in the loop sends nothing.
assert _run(tmp_path, port, rewritten, active=True) == ""
assert len(seen) == 2
def test_a_rule_hold_on_a_closing_turn_does_not_mark_a_rewrite(tmp_path):
t = _closing_transcript(tmp_path, "Done.")
with http_sink(by_path={"/api/plugin/reply-rules": HELD}) as (port, seen):
_run(tmp_path, port, t)
assert _run(tmp_path, port, t, active=True) == ""
assert len(seen) == 1
def test_the_hook_says_nothing_when_nothing_holds(tmp_path):
t = _transcript(tmp_path, "Done.")
with http_sink(by_path={"/api/plugin/reply-rules": QUIET}) as (port, _seen):
+129
View File
@@ -0,0 +1,129 @@
"""The core reply shape rides every turn (milestone 500 step 3).
Every turn ends in a reply, and the operator's prompt is the last point before
it is written, so the per-turn retrieval carries the core shape ahead of
everything else: in full the first time a session sees it, as its one-line
reminder after that, and in full again once a compaction has swept the
ledger. What is pinned here is that round trip across the shell/Python seam —
the route's order and keys, and the hook's ledger.
"""
from __future__ import annotations
import inspect
import json
import os
import subprocess
from pathlib import Path
from types import SimpleNamespace
from unittest.mock import AsyncMock, patch
from quart import Quart, g
from scribe.services import reply_shapes
from tests.helpers import http_sink, need_tools
ROOT = Path(__file__).resolve().parents[1]
HOOK = ROOT / "plugin" / "hooks" / "scribe_autoinject.sh"
SESSION_START = ROOT / "plugin" / "hooks" / "scribe_session_context.sh"
# ── the route ────────────────────────────────────────────────────────────
async def _retrieve(query_string, turn_context="CORE SHAPE"):
from scribe.routes import plugin as routes
turn = AsyncMock(return_value={"context": turn_context, "rule_ids": [173],
"shape_forms": {"core": reply_shapes.FULL}})
app = Quart(__name__)
async with app.test_request_context("/api/plugin/retrieve", query_string=query_string):
g.user = SimpleNamespace(id=7)
with patch.object(routes.plugin_ctx_svc, "build_prompt_rule_hint",
AsyncMock(return_value={"context": "RULE LINE", "rule_ids": [11],
"shown_rule_ids": [11]})), \
patch.object(routes.plugin_ctx_svc, "build_autoinject_hint",
AsyncMock(return_value={"context": "NOTES MENU", "note_ids": []})), \
patch.object(routes.lesson_rules_svc, "co_surfaced", AsyncMock(return_value="")), \
patch.object(routes.moment_delivery_svc, "deliver_for_turn", turn):
resp = await inspect.unwrap(routes.autoinject_retrieve)()
return await resp.get_json(), turn
async def test_the_turns_shape_leads_the_payload_and_its_key_comes_back():
body, _turn = await _retrieve({"q": "fix the flaky thing"})
ctx = body["context"]
assert ctx.index("CORE SHAPE") < ctx.index("RULE LINE") < ctx.index("NOTES MENU")
assert body["shape_keys"] == ["core"]
# Both arms' fresh rules reach the shared ledger.
assert body["rule_ids"] == [11, 173]
async def test_the_route_hands_the_ledger_and_the_rules_just_named_to_the_turn():
_body, turn = await _retrieve({"q": "x", "shapes_seen": "core,bogus",
"exclude_rule_ids": "9", "held_rule_ids": "4"})
kw = turn.await_args.kwargs
assert kw["seen"] == frozenset({"core"})
# A rule the prompt arm just named is not quoted again by the turn.
assert kw["exclude"] == frozenset({9, 11})
assert kw["held"] == frozenset({4})
# ── the hook ─────────────────────────────────────────────────────────────
def _env(tmp_path, port):
return {"PATH": os.environ["PATH"], "SCRIBE_URL": f"http://127.0.0.1:{port}",
"SCRIBE_TOKEN": "t", "TMPDIR": str(tmp_path), "HOME": str(tmp_path)}
def _prompt(tmp_path, port, session="s-shape"):
need_tools("bash", "curl", "awk")
out = subprocess.run(
["bash", str(HOOK)],
input=json.dumps({"session_id": session, "cwd": str(tmp_path),
"prompt": "please fix the importer"}),
capture_output=True, text=True, env=_env(tmp_path, port), timeout=30,
)
assert out.returncode == 0, out.stderr
return out.stdout
def _compact(tmp_path, session="s-shape"):
out = subprocess.run(
["bash", str(SESSION_START)],
input=json.dumps({"session_id": session, "source": "compact"}),
capture_output=True, text=True,
env={"PATH": os.environ["PATH"], "HOME": str(tmp_path), "TMPDIR": str(tmp_path)},
timeout=60,
)
assert out.returncode == 0, out.stderr
REPLY = json.dumps({"context": "Reply shape · Every reply", "note_ids": [], "rule_ids": [],
"shape_keys": ["core"]}).encode()
def test_the_core_goes_whole_once_then_as_a_reminder_then_whole_after_a_compaction(tmp_path):
with http_sink(by_path={"/api/plugin/retrieve": REPLY}) as (port, seen):
out = _prompt(tmp_path, port)
_prompt(tmp_path, port)
_compact(tmp_path)
_prompt(tmp_path, port)
asks = [e for e in seen if e["_path"] == "/api/plugin/retrieve"]
assert len(asks) == 3
assert "shapes_seen" not in asks[0]
assert asks[1]["shapes_seen"] == ["core"]
# The compaction swept the ledger: the session no longer holds the shape.
assert "shapes_seen" not in asks[2]
assert "Every reply" in json.loads(out)["hookSpecificOutput"]["additionalContext"]
def test_the_hook_and_the_route_agree_on_the_ledgers_names():
"""Rule 33 across the seam: a renamed field fails silently."""
from scribe.routes import plugin as routes
hook = HOOK.read_text()
assert "shapes_seen" in hook and "'.shape_keys'" in hook
assert "scribe_shapes_file" in hook and "scribe_shapes_append" in hook
route = inspect.getsource(routes.autoinject_retrieve)
assert 'request.args.get("shapes_seen")' in route and '"shape_keys"' in route
+315
View File
@@ -0,0 +1,315 @@
"""The default reply shapes and the moments they ride (milestone 500 step 1).
What this pins is what delivery will depend on: every shape rides a moment that
exists, the core is small enough to pay for on every turn, no slice grows back
into the skill it replaced, the text speaks for any kind of work, and both
doors hand out the same shapes.
"""
import pytest
from scribe.services import moments, reply_shapes
from tests.helpers import FakeMCP, dev_only_hits
# Bound before conftest's autouse stub replaces the module attribute.
_REAL_RECORD = reply_shapes.record_delivery
def test_there_is_a_core_and_at_least_one_slice():
"""The sweeps below are vacuous over an empty catalog (rule 167)."""
assert reply_shapes.CORE_KEY in reply_shapes.SHAPES
assert len(reply_shapes.SHAPES) >= 2
def test_every_shape_is_keyed_by_its_own_key():
for key, shape in reply_shapes.SHAPES.items():
assert key == shape.key
@pytest.mark.parametrize("key", list(reply_shapes.SHAPES))
def test_every_shape_rides_a_catalog_moment(key):
"""A shape on a moment nothing reaches would never arrive — and an
operator's preference mounted beside it would sit on a dead moment too."""
assert reply_shapes.SHAPES[key].moment in moments.MOMENTS
def test_the_core_rides_the_reply_moment():
"""It is the reply's shape; a preference about every reply is mounted
on `reply.report`, so that is where the core says it lives."""
assert reply_shapes.core().moment == "reply.report"
def test_no_two_slices_share_a_moment():
"""One moment, one default. Two slices on a moment would arrive together
and leave the reader to reconcile them."""
slices = [s.moment for s in reply_shapes.SHAPES.values() if s.key != reply_shapes.CORE_KEY]
assert len(slices) == len(set(slices))
def test_the_core_fits_its_budget():
"""It is paid for in every session. The budget is the ceiling, so a
sentence added to the core has to replace one."""
assert len(reply_shapes.core().text) <= reply_shapes.CORE_BUDGET_CHARS
@pytest.mark.parametrize("key", [k for k in reply_shapes.SHAPES if k != reply_shapes.CORE_KEY])
def test_each_slice_fits_its_budget(key):
assert len(reply_shapes.SHAPES[key].text) <= reply_shapes.SLICE_BUDGET_CHARS
def test_the_budget_guard_can_fail():
long = "x" * (reply_shapes.CORE_BUDGET_CHARS + 1)
assert not len(long) <= reply_shapes.CORE_BUDGET_CHARS
@pytest.mark.parametrize("key", list(reply_shapes.SHAPES))
def test_no_shape_assumes_a_particular_domain(key):
shape = reply_shapes.SHAPES[key]
assert not dev_only_hits(shape.title + "\n" + shape.text), key
@pytest.mark.parametrize("key", list(reply_shapes.SHAPES))
def test_every_shape_says_what_it_is_and_when_it_arrives(key):
shape = reply_shapes.SHAPES[key]
assert shape.title.strip() and shape.delivered.strip() and shape.text.strip()
def test_the_core_carries_the_length_discipline():
"""Milestone 409's live read: replies that had every section were still
too long. Delivery cannot fix that; the core has to say it."""
text = reply_shapes.core().text.lower()
assert "shortest reply that carries the answer" in text
assert "what can go" in text
def test_who_decides_what_rides_every_reply_and_every_ask():
"""The operator's line (milestone 500 step 2): the agent settles what
only it can see, the operator decides direction. A one-liner every turn,
in full where a question is about to be handed back."""
core = reply_shapes.core().text.lower()
asks = reply_shapes.SHAPES["asks"].text.lower()
assert "decide what only you can see" in core and "direction" in core
assert "who decides what" in asks
assert "decide direction" in asks and "hard to undo" in asks
def test_the_core_is_not_a_slice():
"""It rides every turn, whatever moment the turn reaches — so a moment
lookup never hands it out a second time."""
every = [s.moment for s in reply_shapes.SHAPES.values()]
assert reply_shapes.CORE_KEY not in {s.key for s in reply_shapes.for_moments(every)}
def test_a_moment_brings_its_own_slice_and_nothing_else():
got = reply_shapes.for_moments(["work.finish", "work.run"])
assert [s.key for s in got] == ["completion"]
assert reply_shapes.for_moments([]) == []
def test_the_payload_carries_every_shape_with_its_moments_meaning():
data = reply_shapes.catalog()
assert [s["key"] for s in data["shapes"]] == list(reply_shapes.SHAPES)
assert data["total"] == len(reply_shapes.SHAPES)
for row in data["shapes"]:
assert row["means"] == moments.MOMENTS[row["moment"]].means
async def test_the_tool_returns_the_catalog():
from scribe.mcp.tools import moments as tool
assert await tool.list_reply_shapes() == reply_shapes.catalog()
def test_the_tool_is_registered_as_a_read():
from scribe.mcp.server import _READ_ONLY_TOOLS
from scribe.mcp.tools import moments as tool
mcp = FakeMCP()
tool.register(mcp)
assert "list_reply_shapes" in mcp.names
assert "list_reply_shapes" in _READ_ONLY_TOOLS
def test_both_doors_read_one_catalog():
"""Rule 33 parity: the session and the Settings view cannot show
different defaults."""
from scribe.mcp.tools import moments as tool
from scribe.routes import retrieval as routes
assert tool.shapes_svc is reply_shapes
assert routes.shapes_svc is reply_shapes
async def test_the_route_returns_the_operators_overview_with_a_bounded_window():
from types import SimpleNamespace
from unittest.mock import AsyncMock, patch
from quart import Quart, g
from scribe.routes import retrieval as routes
view = AsyncMock(return_value={"shapes": [], "total": 0})
app = Quart(__name__)
for raw, days in (("", 30), ("7", 7), ("900", 90), ("x", 30)):
async with app.test_request_context("/api/retrieval/reply-shapes", query_string={"days": raw}):
g.user = SimpleNamespace(id=7)
with patch.object(routes.shapes_svc, "overview", view):
resp = await routes.reply_shapes_route.__wrapped__()
assert await resp.get_json() == {"shapes": [], "total": 0}
assert view.await_args.kwargs == {"days": days}
# ── the operator's view (milestone 500 step 5) ──────────────────────────
async def test_the_overview_lists_the_preferences_mounted_on_each_shapes_moment():
"""What adjusts a shape is what is mounted beside it — and only
preferences: a rule on the same moment binds, it does not reshape."""
from types import SimpleNamespace
from unittest.mock import AsyncMock, patch
pref = SimpleNamespace(id=173, title="Problem first", statement="…", kind="preference")
rule = SimpleNamespace(id=11, title="Definition of done", statement="…", kind="rule")
async def mounted(user_id, moments, project_id=None):
return [(pref, "reply.report"), (rule, "reply.report")] if moments == ["reply.report"] else []
counts = {k: {"full": 0, "pointer": 0} for k in reply_shapes.SHAPES}
counts["core"] = {"full": 2, "pointer": 9}
with patch("scribe.services.rulebooks.rules_on_moments", mounted), \
patch.object(reply_shapes, "delivery_counts", AsyncMock(return_value=counts)):
data = await reply_shapes.overview(7, days=30)
rows = {r["key"]: r for r in data["shapes"]}
assert rows["core"]["preferences"] == [{"id": 173, "title": "Problem first", "statement": "…"}]
assert rows["core"]["deliveries"] == {"full": 2, "pointer": 9}
assert rows["completion"]["preferences"] == []
assert data["days"] == 30 and data["deliveries_failed"] is False
async def test_counts_that_fail_read_as_unknown_not_zero():
from unittest.mock import AsyncMock, patch
async def mounted(user_id, moments, project_id=None):
return []
with patch("scribe.services.rulebooks.rules_on_moments", mounted), \
patch.object(reply_shapes, "delivery_counts", AsyncMock(side_effect=RuntimeError("db"))):
data = await reply_shapes.overview(7)
assert data["deliveries_failed"] is True
assert all(r["deliveries"] is None for r in data["shapes"])
async def test_delivery_counts_tally_each_shape_by_form_and_skip_what_does_not_parse():
import json
from unittest.mock import MagicMock, patch
rows = [json.dumps({"shapes": {"core": "full"}, "via": "turn"}),
json.dumps({"shapes": {"core": "pointer", "completion": "full"}, "via": "hook"}),
json.dumps({"shapes": {"core": "pointer", "bogus": "full"}, "via": "turn"}),
"not json", None]
result = MagicMock()
result.scalars.return_value.all.return_value = rows
class _Session:
async def __aenter__(self):
return self
async def __aexit__(self, *exc):
return False
async def execute(self, _stmt):
return result
with patch("scribe.models.async_session", lambda: _Session()):
counts = await reply_shapes.delivery_counts(7, days=30)
assert counts["core"] == {"full": 1, "pointer": 2}
assert counts["completion"] == {"full": 1, "pointer": 0}
assert "bogus" not in counts
# ── delivery (milestone 500 step 3) ─────────────────────────────────────
@pytest.mark.parametrize("key", list(reply_shapes.SHAPES))
def test_every_shape_has_a_one_line_reminder(key):
shape = reply_shapes.SHAPES[key]
assert shape.reminder.strip() and "\n" not in shape.reminder
assert not dev_only_hits(shape.reminder), key
def test_the_full_form_carries_the_text_and_says_a_preference_wins():
core = reply_shapes.core()
out = reply_shapes.render(core, full=True)
assert out.endswith(core.text)
assert core.title in out and "preference" in out
def test_the_pointer_form_is_one_line_carrying_the_reminder():
core = reply_shapes.core()
out = reply_shapes.render(core, full=False)
assert "\n" not in out
assert core.reminder in out and "list_reply_shapes" in out
assert core.text not in out
def test_a_shape_the_session_holds_goes_out_as_its_pointer():
shapes = [reply_shapes.core(), reply_shapes.SHAPES["completion"]]
blocks, forms = reply_shapes.deliver(shapes, frozenset({"core"}))
assert forms == {"core": reply_shapes.POINTER, "completion": reply_shapes.FULL}
assert reply_shapes.core().text not in blocks[0]
assert reply_shapes.SHAPES["completion"].text in blocks[1]
def test_a_door_with_no_ledger_sends_everything_in_full():
_blocks, forms = reply_shapes.deliver([reply_shapes.core()], frozenset())
assert forms == {"core": reply_shapes.FULL}
@pytest.mark.parametrize("raw,keys", [
("core", {"core"}),
("core, asks,core", {"core", "asks"}),
("", set()),
(None, set()),
("core,nonsense,../x", {"core"}),
])
def test_the_ledger_keeps_only_shapes_that_exist(raw, keys):
assert reply_shapes.parse_seen(raw) == frozenset(keys)
async def test_a_delivery_is_one_plugin_row_naming_each_shape_and_its_form():
from unittest.mock import MagicMock, patch
added = []
session = MagicMock()
session.add = added.append
async def _commit():
return None
session.commit = _commit
class _Ctx:
async def __aenter__(self):
return session
async def __aexit__(self, *exc):
return False
with patch("scribe.models.async_session", lambda: _Ctx()):
await _REAL_RECORD(7, {"core": "pointer", "asks": "full"}, via="turn")
assert len(added) == 1
row = added[0]
assert (row.category, row.action, row.user_id) == ("plugin", "reply_shape", 7)
import json
assert json.loads(row.details) == {"shapes": {"core": "pointer", "asks": "full"},
"via": "turn"}
async def test_the_delivery_row_fails_open_and_skips_an_empty_delivery():
from unittest.mock import patch
def _boom():
raise RuntimeError("db down")
with patch("scribe.models.async_session", _boom):
await _REAL_RECORD(7, {"core": "full"}, via="turn") # no raise
await _REAL_RECORD(7, {}, via="turn")
-169
View File
@@ -1,169 +0,0 @@
"""The Stop hook that checks a task-closing reply for the completion sections
(milestone 409 step 5).
Runs the real shell against synthetic transcripts in the shape Claude Code
writes (one content block per JSONL line) and the shared HTTP sink. What it
pins: silence on every turn that closed nothing; a block only when the
instance recorded it, in the words the instance returned; one rewrite at most,
recorded; and no block from another plugin's loop or a failed task write.
"""
from __future__ import annotations
import json
import os
import shutil
import subprocess
from pathlib import Path
import pytest
from tests.helpers import http_sink
HOOK = Path(__file__).resolve().parents[1] / "plugin" / "hooks" / "scribe_report_check.sh"
TOOL = "mcp__plugin_scribe_scribe__update_task"
GOOD = ('**Where this sits:** milestone 12 "Move the backups offsite", step 3 of 5.\n'
"**What now works:** the sync runs nightly.\n**Needs you:** nothing.\n**Next:** alerts.")
BAD = "All done, pushed it."
REASON = "SERVER REASON: rewrite as a completion report"
def _env(tmp_path, url="http://127.0.0.1:9"):
for tool in ("curl", "bash"):
if shutil.which(tool) is None:
pytest.skip(f"hook runtime tool {tool!r} not installed")
return {"PATH": os.environ["PATH"], "SCRIBE_URL": url, "SCRIBE_TOKEN": "t",
"TMPDIR": str(tmp_path), "HOME": str(tmp_path)}
def _prompt(text="please finish it"):
return {"type": "user", "message": {"role": "user", "content": text}}
def _tool_use(tid="toolu_1", status="done", name=TOOL, task_id=41):
return {"type": "assistant", "message": {"content": [
{"type": "tool_use", "id": tid, "name": name, "input": {"task_id": task_id, "status": status}}]}}
def _result(tid="toolu_1", is_error=False):
return {"type": "user", "message": {"content": [
{"type": "tool_result", "tool_use_id": tid, "is_error": is_error, "content": "{}"}]}}
def _text(text):
return {"type": "assistant", "message": {"content": [{"type": "text", "text": text}]}}
def _transcript(tmp_path, lines):
path = tmp_path / "t.jsonl"
# Compact, like the file Claude Code writes ({"name":"…"}, no spaces).
path.write_text("\n".join(json.dumps(line, separators=(",", ":")) for line in lines) + "\n")
return path
def _run(env, transcript, active=False, session="s1"):
out = subprocess.run(
["bash", str(HOOK)],
input=json.dumps({"session_id": session, "transcript_path": str(transcript),
"cwd": str(transcript.parent), "hook_event_name": "Stop",
"stop_hook_active": active}),
capture_output=True, text=True, env=env, timeout=30,
)
assert out.returncode == 0, out.stderr
return out.stdout.strip()
def _closing_turn(reply):
return [_prompt(), _text("On it."), _tool_use(), _result(), _text(reply)]
def test_a_turn_that_closed_nothing_is_silent_and_reports_nothing(tmp_path):
with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
t = _transcript(tmp_path, [_prompt(), _tool_use(status="in_progress"), _result(), _text(BAD)])
assert _run(env, t) == ""
assert seen == []
def test_a_complete_report_passes_silently_and_is_recorded(tmp_path):
with http_sink(b'{"status":"ok"}') as (port, seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
assert _run(env, _transcript(tmp_path, _closing_turn(GOOD))) == ""
assert [q["outcome"] for q in seen] == [["passed"]]
assert seen[0]["task_ids"] == ["41"]
def test_a_missing_section_blocks_once_in_the_servers_words_then_records_the_rewrite(tmp_path):
reply = json.dumps({"status": "ok", "reason": REASON}).encode()
with http_sink(reply) as (port, seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
out = json.loads(_run(env, _transcript(tmp_path, _closing_turn(BAD))))
assert out == {"decision": "block", "reason": REASON}
assert seen[0]["outcome"] == ["blocked"]
assert seen[0]["missing"] == ["where it sits,needs you,next"]
# The rewrite: Claude Code sets stop_hook_active; the hook records and never blocks again.
rewritten = _transcript(tmp_path, _closing_turn(BAD) + [_text(GOOD)])
assert _run(env, rewritten, active=True) == ""
assert seen[1]["outcome"] == ["passed_after_rewrite"]
assert _run(env, rewritten, active=True) == ""
assert len(seen) == 2
def test_a_rewrite_that_still_misses_is_recorded_and_not_blocked(tmp_path):
reply = json.dumps({"status": "ok", "reason": REASON}).encode()
with http_sink(reply) as (port, seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
t = _transcript(tmp_path, _closing_turn(BAD))
_run(env, t)
assert _run(env, t, active=True) == ""
assert [q["outcome"][0] for q in seen] == ["blocked", "missing_after_rewrite"]
def test_another_hooks_block_loop_is_left_alone(tmp_path):
with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
assert _run(env, _transcript(tmp_path, _closing_turn(BAD)), active=True) == ""
assert seen == []
def test_a_task_write_that_failed_closed_nothing(tmp_path):
with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
t = _transcript(tmp_path, [_prompt(), _tool_use(), _result(is_error=True), _text(BAD)])
assert _run(env, t) == ""
assert seen == []
def test_a_task_closed_in_an_earlier_turn_does_not_count(tmp_path):
with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
t = _transcript(tmp_path, _closing_turn(GOOD) + [_prompt("thanks, what else?"), _text(BAD)])
assert _run(env, t) == ""
assert seen == []
def test_a_reply_not_yet_written_is_not_judged(tmp_path):
with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
t = _transcript(tmp_path, [_prompt(), _tool_use(), _result()])
assert _run(env, t) == ""
assert seen == []
def test_no_block_without_a_recorded_check(tmp_path):
t = _transcript(tmp_path, _closing_turn(BAD))
# Unreachable instance.
assert _run(_env(tmp_path), t) == ""
# An instance that answered but returned no reason.
with http_sink(b'{"status":"ok"}') as (port, seen):
assert _run(_env(tmp_path, f"http://127.0.0.1:{port}"), t, session="s2") == ""
assert seen[0]["outcome"] == ["blocked"]
def test_a_bare_id_does_not_count_as_placing_the_work(tmp_path):
reply = json.dumps({"status": "ok", "reason": REASON}).encode()
with http_sink(reply) as (port, seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
bare = "Closed #41.\n**Needs you:** nothing.\n**Next:** #42."
assert json.loads(_run(env, _transcript(tmp_path, _closing_turn(bare))))["decision"] == "block"
assert seen[0]["missing"] == ["where it sits"]
+33 -3
View File
@@ -43,9 +43,13 @@ def test_a_request_for_approval_has_its_own_named_section():
a completion list as "Blocked by the permission check", and read as a
fault rather than a question waiting on them. The heading is what they
scan for, so it is pinned by name."""
text = _text()
assert "Approval requested" in text
assert "| **Approval**" in text, "the Asks table lost its Approval row"
from scribe.services import reply_shapes
assert "Approval requested" in _text()
# The kinds moved to the delivered shapes (milestone 500): the asks shape
# carries the Approval kind, and the core sends a held action there.
assert "**Approval**" in reply_shapes.SHAPES["asks"].text
assert "Approval requested" in reply_shapes.core().text
def test_placement_comes_from_the_record():
@@ -63,3 +67,29 @@ def test_the_shipped_shapes_assume_no_particular_domain():
dev_only = [w for w in (r"\bCI\b", r"\bcommit", r"\bpull request", r"file:line", r"\bpytest\b")
if re.search(w, text, re.IGNORECASE)]
assert not dev_only, f"software-only vocabulary in a product-wide shape: {dev_only}"
def test_the_skill_points_at_the_delivered_shapes_instead_of_restating_them():
"""Milestone 500 step 4: the shapes are server product, delivered at their
moments. The skill says where they come from and holds the reasoning."""
text = _text()
assert "list_reply_shapes" in text and "Reply shape · Every reply" in text
def test_the_worked_completion_report_agrees_with_the_delivered_completion_shape():
"""The two copies that could drift: the worked example here, and the
completion shape the server sends when a task closes. Same sections."""
from scribe.services import reply_shapes
shape = reply_shapes.SHAPES["completion"].text
for section in ("Where this sits", "What now works", "How / why", "Needs you", "Next"):
assert f"**{section}**" in shape, f"the completion shape lost {section!r}"
assert f"**{section}" in _text(), f"the worked example lost {section!r}"
def test_the_core_names_the_header_the_skill_quotes():
"""The skill tells the reader what the delivered core looks like; if the
header changes, the pointer would describe something that never arrives."""
from scribe.services import reply_shapes
assert reply_shapes.render(reply_shapes.core(), full=True).startswith("Reply shape · Every reply")
+2 -4
View File
@@ -103,10 +103,8 @@ def test_the_extractor_finds_something() -> None:
def test_the_extractor_resolves_the_constant_sources() -> None:
"""The specific capability a grep would lose. See the module docstring.
`report_preference` was the second example until milestone 456 moved it
into the retrieval pipeline, where its source is a spec field; the
pipeline's reserved slot still records through a module constant, so it
carries the case instead.
The pipeline's reserved slot records through a module constant, so it
carries the case.
"""
found, _ = call_sites()
for via_constant in ("wide_net", "preference_slot"):
-1
View File
@@ -34,7 +34,6 @@ def test_every_ranked_source_is_measured_as_its_spec_declares():
point = POINTS[spec.source]
assert point.kind == UNBIDDEN
assert point.what == d.what
assert point.fixed_query == d.fixed_query
# A quiet source must say why, and only a quiet source may (#2475).
assert point.expects_traffic == (not d.quiet_because)
assert point.quiet_because == d.quiet_because
+5 -9
View File
@@ -24,9 +24,9 @@ WHAT THIS PINS
different file. Renaming one without the other produces a surface that can
be tuned and cannot be measured, or measured and not tuned, and both fail
silently.
2. **Keys are unique.** Two surfaces sharing a settings key is how
`report_preference` spent its first release moving whenever the prompt arm
was tuned (#3860) — one dial wearing two labels.
2. **Keys are unique.** Two surfaces sharing a settings key is how the
completion-report arm (since retired) spent its first release moving
whenever the prompt arm was tuned (#3860) — one dial wearing two labels.
3. **An unknown surface is refused.** Settings keys are free-form strings in
a generic table, so a typo'd name would write a key nothing reads: a
change that appears to succeed, reports a new value, and alters nothing.
@@ -44,7 +44,7 @@ import pytest
from scribe.services import retrieval_surfaces as rs
SERVICES = ("plugin_context", "reply_preferences")
SERVICES = ("plugin_context",)
def _service_source(name: str) -> str:
@@ -63,10 +63,6 @@ def test_every_surface_name_is_a_real_telemetry_source():
to justify a change describes a different arm from the one the change hits.
"""
blob = "\n".join(_service_source(n) for n in SERVICES)
# `report_preference` passes its name through a module constant rather than
# a literal, so that one name is satisfied by the constant holding it.
from scribe.services.reply_preferences import SOURCE
# The rule arms record through the one pipeline (milestone 456), where
# `source` is the spec's own field — so for them the spec IS the string
# `record_retrieval` receives, and the join key is checked against it.
@@ -76,7 +72,7 @@ def test_every_surface_name_is_a_real_telemetry_source():
missing = [
s.name for s in rs.SURFACES.values()
if f'source="{s.name}"' not in blob
and s.name != SOURCE and s.name not in via_pipeline
and s.name not in via_pipeline
]
assert not missing, (
f"these surfaces can be tuned but never measured: {missing}. The "
-77
View File
@@ -471,83 +471,6 @@ def test_a_quiet_arm_is_not_suspended_either_way() -> None:
assert "floor_moved_mid_window" not in codes(ws)
# ── a fixed-query arm: decline rate is arithmetic, not evidence (#4232) ─────
#
# `report_preference` searches one constant string (COMPLETION_QUERY), so it
# scores against one number on every call. Its decline rate is therefore 0% or
# 100% and never in between, and which one depends only on where the bar sits
# relative to that constant.
#
# So the two warnings swap roles for these arms. "Never declined" stops being
# evidence about the floor — `cannot_decline`'s own remedy, "check that it
# applies its floor", is unanswerable from it. "Always declined" starts being
# evidence, because for a constant score it means the bar is above it and no
# further traffic will ever say otherwise.
def test_cannot_decline_is_silent_on_a_fixed_query_arm():
"""The live readout fired this on `report_preference` at 45 calls, 0
empty, with p10 = p50 = p90 = min = max = 0.791 — five identical
percentiles, which is one record at one score rather than a ranking."""
ws = warn({"report_preference": src(calls=45, zero_result_calls=0, p10=0.791)})
assert "cannot_decline" not in codes(ws, "report_preference")
def test_cannot_decline_still_fires_where_the_rate_means_something():
"""The falsifier for the case above (rule 167). If this passes only
because the check was disabled rather than narrowed, this fails."""
ws = warn({"auto_inject": src(calls=N, zero_result_calls=0)})
assert "cannot_decline" in codes(ws, "auto_inject")
def test_a_fixed_query_arm_that_never_clears_its_bar_is_named():
"""69 consecutive declines at 0.0006 under the bar is a state this arm has
actually been in. Nothing else in the readout would have said so: it looks
exactly like an arm with nothing to report."""
ws = warn({"report_preference": src(calls=45, zero_result_calls=45)})
assert "fixed_query_never_clears" in codes(ws, "report_preference")
detail = next(w["detail"] for w in ws if w["code"] == "fixed_query_never_clears")
assert "near_miss_samples" in detail, (
"the last time this fired, every percentile said lower the floor and "
"the refused record showed the refusal was right — so the warning has "
"to send the reader to the record, not to the dial"
)
def test_an_ordinary_arm_returning_nothing_all_window_is_not_dead():
"""For an arm whose score can vary, an empty window means nothing matched,
which is an answer rather than a fault."""
ws = warn({"auto_inject": src(calls=N, zero_result_calls=N)})
assert "fixed_query_never_clears" not in codes(ws, "auto_inject")
def test_a_fixed_query_arm_that_sometimes_clears_is_not_dead():
"""Only ALL-empty says the bar is above the constant. Anything in between
means the score is not actually constant, and the premise is wrong."""
ws = warn({"report_preference": src(calls=45, zero_result_calls=44)})
assert "fixed_query_never_clears" not in codes(ws, "report_preference")
def test_the_dead_arm_warning_still_needs_volume():
ws = warn({"report_preference": src(calls=N - 1, zero_result_calls=N - 1)})
assert "fixed_query_never_clears" not in codes(ws)
def test_the_registry_declares_which_arms_ask_a_fixed_question():
"""Asserted on structure (rule 167), and able to fail: if `fixed_query`
is dropped or defaults to True, one of these two halves breaks."""
from scribe.services.retrieval_registry import POINTS
assert POINTS["report_preference"].fixed_query is True, (
"services/reply_preferences.py::COMPLETION_QUERY is a module constant"
)
# An arm whose query is built from the prompt, the file or the command is
# not fixed, and marking one would silence a warning that works there.
for varying in ("auto_inject", "write_path", "pre_tool_rule", "prompt_rule"):
assert POINTS[varying].fixed_query is False, varying
# ── floor_history_unknown: the window reaches back past the ledger ─────────
#
# THE SAME SUSPENSION, FOR THE CASE THE ONE ABOVE CANNOT SEE.
+1
View File
@@ -46,6 +46,7 @@ def test_every_endpoint_is_reachable_on_the_app():
"/api/retrieval/moments/mappings",
"/api/retrieval/moments/proposals",
"/api/retrieval/moments/proposals/judge",
"/api/retrieval/reply-shapes",
}
-1
View File
@@ -1702,7 +1702,6 @@ def test_every_hook_rule_search_says_which_project_it_is_for():
# ranked-search stage, checked below — so a direct search appearing
# here again is a copy of the arm coming back, and fails the count.
"src/scribe/services/plugin_context.py": 0,
"src/scribe/services/reply_preferences.py": 0,
}
for path, expected in sources.items():
calls = [
-134
View File
@@ -1,134 +0,0 @@
"""The completion-report preference lookup (milestone 409 step 4).
What it pins: the lookup asks for PREFERENCES only, logs every call under its
own source (the empty ones too), counts only what it showed as surfaced, and
fails open. The query stays domain-neutral, because every kind of project
closes tasks.
"""
import re
from unittest.mock import AsyncMock, MagicMock, patch
M = "scribe.services.reply_preferences"
# Its two numbers are resolved through the registry now (#4102), so the
# settings reader to patch lives there rather than in the arm's own module.
RS = "scribe.services.retrieval_surfaces"
def _rule(rid, kind="preference"):
return MagicMock(id=rid, kind=kind, title=f"t{rid}", statement=f"s{rid}")
async def _run(hits, *, searched=True, raises=None):
from scribe.services.reply_preferences import completion_preferences
async def search(user_id, query, **kw):
if raises:
raise raises
kw["report"].update({"searched": searched, "best_available_score": 0.7})
return hits
search_mock = AsyncMock(side_effect=search)
with patch(f"{M}.semantic_search_rules", search_mock), \
patch(f"{RS}.get_setting", AsyncMock(return_value="0.72")), \
patch(f"{M}.record_retrieval") as logged, \
patch(f"{M}.record_rule_surfaced") as surfaced:
out = await completion_preferences(7, project_id=3)
return out, search_mock, logged, surfaced
async def test_asks_for_preferences_only_and_returns_them_best_first():
out, search, logged, surfaced = await _run([(0.9, _rule(1)), (0.8, _rule(2))])
assert search.await_args.kwargs["kind"] == "preference"
assert [p["id"] for p in out] == [1, 2]
assert all(p["kind"] == "preference" for p in out)
assert logged.call_args.kwargs["source"] == "report_preference"
assert surfaced.call_args.kwargs == {"user_id": 7, "rule_ids": [1, 2], "source": "report_preference"}
async def test_a_rule_that_slips_through_is_not_handed_back_as_a_preference():
out, _, _, surfaced = await _run([(0.9, _rule(1, kind="rule")), (0.8, _rule(2))])
assert [p["id"] for p in out] == [2]
assert surfaced.call_args.kwargs["rule_ids"] == [2]
async def test_an_empty_call_is_still_logged_and_surfaces_nothing():
out, _, logged, surfaced = await _run([])
assert out == []
logged.assert_called_once()
assert logged.call_args.kwargs["results"] == []
surfaced.assert_not_called()
async def test_a_search_that_never_ran_says_so_to_the_log():
_, _, logged, _ = await _run([], searched=False)
assert logged.call_args.kwargs["searched"] is False
async def test_fails_open():
out, _, _, surfaced = await _run([], raises=RuntimeError("embedder down"))
assert out == []
surfaced.assert_not_called()
def test_it_is_a_ranked_source():
from scribe.services.rule_usage import is_ambient
assert not is_ambient("report_preference")
def test_its_bar_is_its_own_key_not_the_prompt_arm_s(monkeypatch):
"""Regression on the coupling #3860 found (and on the fix being real).
This arm shipped reading PROMPTRULE_THRESHOLD_KEY, so an operator tuning
the bar for their own prose moved this one with it and was never told.
That is worse here than anywhere else: every other arm scores a query that
varies per call, while COMPLETION_QUERY is fixed — so this arm's score for
a given corpus is a CONSTANT, and a constant that lands under the bar is a
dead arm rather than a quiet one. No amount of traffic reveals it.
Asserted on the key the lookup actually asks for, which is the thing that
broke, rather than on the constant being defined somewhere.
"""
import asyncio
from scribe.services import plugin_context
from scribe.services.reply_preferences import completion_preferences
asked: list[str] = []
async def get_setting(user_id, key, default=""):
asked.append(key)
return default
async def search(user_id, query, **kw):
kw["report"].update({"searched": True})
return []
with patch(f"{RS}.get_setting", AsyncMock(side_effect=get_setting)), \
patch(f"{M}.semantic_search_rules", AsyncMock(side_effect=search)), \
patch(f"{M}.record_retrieval"):
asyncio.run(completion_preferences(7))
# Both numbers are its own since #4102 — a floor AND a budget — so the arm
# asks for two keys, and neither may be the prompt arm's.
from scribe.services.retrieval_surfaces import get_surface
mine = get_surface("report_preference")
assert asked == [mine.floor_key, mine.budget_key]
theirs = get_surface("prompt_rule")
assert mine.floor_key != theirs.floor_key, (
"the two keys are the same string again, so the settings form has one "
"dial driving two arms — which is the defect, whatever the value is"
)
assert mine.floor_key == plugin_context.REPORTPREF_THRESHOLD_KEY, (
"the registry and the module constant disagree about this arm's key, "
"so the Settings form would write one and the arm would read the other"
)
def test_the_query_assumes_no_particular_domain():
from scribe.services.reply_preferences import COMPLETION_QUERY
dev_only = [w for w in (r"\bCI\b", r"\bcommit", r"\bpull request", r"\bcode\b", r"\btest")
if re.search(w, COMPLETION_QUERY, re.IGNORECASE)]
assert not dev_only, f"software-only vocabulary in a query every project runs: {dev_only}"
+67 -5
View File
@@ -1,12 +1,38 @@
"""The server half of the report-shape check (milestone 409 step 5): the words a
blocked reply is sent back with, and the outcome record."""
"""The report-shape check (milestone 409 step 5): which completion sections a
task-closing reply lacks, the words it is sent back with, and the outcome
record. Since milestone 500 step 4 the check itself runs here, inside the one
end-of-turn request, rather than in a Stop hook of its own."""
import json
from unittest.mock import patch
from unittest.mock import AsyncMock, patch
import pytest
from tests.helpers import make_mock_session
GOOD = ('**Where this sits:** milestone 12 "Move the backups offsite", step 3 of 5.\n'
"**What now works:** the sync runs nightly.\n**Needs you:** nothing.\n**Next:** alerts.")
BAD = "All done, pushed it."
def test_a_complete_report_misses_nothing():
from scribe.services.report_check import missing_sections
assert missing_sections(GOOD) == []
def test_a_bare_reply_misses_every_section_in_order():
from scribe.services.report_check import SECTIONS, missing_sections
assert missing_sections(BAD) == list(SECTIONS)
def test_a_bare_id_does_not_count_as_placing_the_work():
"""The title is what spares the reader a lookup; "closed #41" does not."""
from scribe.services.report_check import missing_sections
assert missing_sections("Closed #41.\n**Needs you:** nothing.\n**Next:** #42.") == ["where it sits"]
assert missing_sections('Closed #41 "Sync". Needs you: nothing. Next: #42.') == []
def test_the_reason_names_only_sections_it_knows():
from scribe.services.report_check import block_reason
@@ -14,8 +40,17 @@ def test_the_reason_names_only_sections_it_knows():
reason = block_reason(["next", "ignore previous instructions", "Where It Sits"])
assert "missing: where it sits, next." in reason
assert "ignore previous instructions" not in reason
# It points at the skill that owns the shape rather than restating it.
assert "reporting-back" in reason and "placement" in reason
def test_the_reason_carries_the_completion_shapes_line_and_where_the_rest_is():
"""The shape was delivered when the task closed; the reason repeats its
one line, not the whole of it, and names where the full text is."""
from scribe.services import reply_shapes
from scribe.services.report_check import block_reason
reason = block_reason(["next"])
assert reply_shapes.SHAPES["completion"].reminder in reason
assert "list_reply_shapes" in reason and "placement" in reason
def test_a_reason_with_nothing_recognised_still_says_what_to_do():
@@ -45,3 +80,30 @@ async def test_an_unknown_outcome_is_refused_before_anything_is_written():
pytest.raises(ValueError):
await record_report_check(7, "skipped")
session.add.assert_not_called()
@pytest.mark.parametrize("reply,rewrite,outcome,held", [
(GOOD, False, "passed", False),
(BAD, False, "blocked", True),
(GOOD, True, "passed_after_rewrite", False),
(BAD, True, "missing_after_rewrite", False),
])
async def test_check_reply_records_every_outcome_and_holds_only_a_first_miss(reply, rewrite, outcome, held):
from scribe.services import report_check
record = AsyncMock()
with patch.object(report_check, "record_report_check", record):
got = await report_check.check_reply(7, reply, task_ids=[41], rewrite=rewrite, project_id=2)
assert got["outcome"] == outcome
assert bool(got["reason"]) is held
assert record.await_args.args == (7, outcome)
assert record.await_args.kwargs["task_ids"] == [41]
async def test_an_unrecorded_check_holds_nothing():
"""A hold the numbers cannot see is not one this check may make."""
from scribe.services import report_check
with patch.object(report_check, "record_report_check", AsyncMock(side_effect=RuntimeError("db"))):
got = await report_check.check_reply(7, BAD, task_ids=[41], rewrite=False)
assert got == {"outcome": "", "reason": ""}
+1 -1
View File
@@ -58,7 +58,7 @@ DEFS = HOOKS / "scribe_defs.sh"
# the convention tests below are what keep this roster honest as it grows.
LEDGERS = {
"scribe-priorart": (".ids", ".rules.ids", ".opened.ids", ".sync.ids",
".derive.ids"),
".derive.ids", ".shapes.ids", ".reportcheck.ids"),
"scribe-autoinject": (".ids",),
}
-1
View File
@@ -53,7 +53,6 @@ _SURFACE_PAIRS = (
("write_path_rule", "kbRuleHintThreshold"),
("pre_tool_rule", "kbToolRuleThreshold"),
("prompt_rule", "kbPromptRuleThreshold"),
("report_preference", "kbReportPrefThreshold"),
("reply_rule", "kbReplyRuleThreshold"),
)