Merge pull request 'Reply shapes ride the moments (milestone 500, steps 1–5)' (#209) from dev into main
CI & Build / Python lint (push) Successful in 3s
CI & Build / Plugin hooks (push) Successful in 13s
CI & Build / TypeScript typecheck (push) Successful in 53s
CI & Build / integration (push) Successful in 1m8s
CI & Build / Python tests (push) Successful in 1m55s
CI & Build / Build & push image (push) Successful in 17s
CI & Build / Python lint (push) Successful in 3s
CI & Build / Plugin hooks (push) Successful in 13s
CI & Build / TypeScript typecheck (push) Successful in 53s
CI & Build / integration (push) Successful in 1m8s
CI & Build / Python tests (push) Successful in 1m55s
CI & Build / Build & push image (push) Successful in 17s
This commit was merged in pull request #209.
This commit is contained in:
@@ -296,8 +296,18 @@ jobs:
|
||||
set -eux
|
||||
echo "=== container landscape (diagnostic for the name filter) ==="
|
||||
docker ps -a --format '{{.ID}} {{.Image}} -> {{.Names}}'
|
||||
PG=$(docker ps --filter "name=integration" --filter "ancestor=pgvector/pgvector:pg17" -q | head -n1)
|
||||
test -n "$PG"
|
||||
# Only THIS job's services. The runner's docker daemon is shared, so another
|
||||
# repo's job can be running beside this one, and a job-name filter matches its
|
||||
# containers too (Steward run 8358 and Inkwell run 8653 each took another repo's
|
||||
# Postgres). act names every container of a task GITEA-ACTIONS-TASK-<n>-...: read
|
||||
# <n> from our own container (its id is in the /etc/hostname bind mount) and
|
||||
# require exactly one match.
|
||||
SELF=$(grep -o '/containers/[0-9a-f]\{64\}/' /proc/self/mountinfo | head -n1 | cut -d/ -f3 || true)
|
||||
SELF=${SELF:-$(hostname)}
|
||||
TASK=$(docker inspect -f '{{.Name}}' "$SELF" | grep -o 'GITEA-ACTIONS-TASK-[0-9]*-')
|
||||
test -n "$TASK"
|
||||
PG=$(docker ps --filter "name=$TASK" --filter "ancestor=pgvector/pgvector:pg17" -q)
|
||||
test "$(printf '%s\n' "$PG" | grep -c .)" -eq 1
|
||||
PG_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$PG")
|
||||
test -n "$PG_IP"
|
||||
export DATABASE_URL="postgresql+asyncpg://scribe:ci_integration@${PG_IP}:5432/scribe_test"
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
"""retire the report_preference arm's rows (milestone 500 step 4)
|
||||
|
||||
Revision ID: 0121
|
||||
Revises: 0120
|
||||
Create Date: 2026-10-09
|
||||
|
||||
The fixed-question arm that searched for completion-report preferences when a
|
||||
task closed is gone: the reply shapes and the preferences mounted beside them
|
||||
now arrive on the moments (milestone 500 step 3). Its rows go with it (rule
|
||||
22 — no legacy to preserve), because each one would otherwise mean something
|
||||
different once the arm is no longer registered:
|
||||
|
||||
- `retrieval_logs` / `retrieval_judgments` — a source missing from the
|
||||
registry reads as `unregistered_source`, whose advice is "add it to the
|
||||
registry": the opposite of what happened.
|
||||
- `rule_usage_events` — a source missing from `RANKED_SOURCES` reads as
|
||||
ambient, so its surfacings would silently move into the other column of
|
||||
every rule's pull-through.
|
||||
- `retrieval_tuning_events` — the tuning history of a surface that no longer
|
||||
exists.
|
||||
- `settings` — the arm's floor and budget keys, which nothing reads.
|
||||
|
||||
COST IS BOUNDED BY AN INDEX, NOT BY THE TABLE. A migration runs at boot
|
||||
against the live tables, which CI never has (lesson #5221). `retrieval_logs`
|
||||
and `retrieval_judgments` are indexed on `source`. `rule_usage_events` is not,
|
||||
so its delete is also bounded by `created_at` (indexed): the arm shipped with
|
||||
milestone 409 step 4, decided 2026-09-14, so no row of it predates
|
||||
`ARM_BORN`, and the scan covers weeks rather than the table's whole history.
|
||||
|
||||
Not reversible: the arm is not coming back, so neither are its rows.
|
||||
"""
|
||||
from alembic import op
|
||||
|
||||
revision = "0121"
|
||||
down_revision = "0120"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
SOURCE = "report_preference"
|
||||
# A safe floor under the arm's first row (it shipped after 2026-09-14).
|
||||
ARM_BORN = "2026-09-01"
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.execute(f"DELETE FROM retrieval_logs WHERE source = '{SOURCE}'")
|
||||
op.execute(f"DELETE FROM retrieval_judgments WHERE source = '{SOURCE}'")
|
||||
op.execute(
|
||||
f"DELETE FROM rule_usage_events WHERE created_at >= '{ARM_BORN}' "
|
||||
f"AND source = '{SOURCE}'"
|
||||
)
|
||||
op.execute(f"DELETE FROM retrieval_tuning_events WHERE surface = '{SOURCE}'")
|
||||
op.execute(
|
||||
"DELETE FROM settings WHERE key IN "
|
||||
"('kb_reportpref_threshold', 'kb_reportpref_top_k')"
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
pass
|
||||
@@ -0,0 +1,47 @@
|
||||
import { apiGet } from "@/api/client";
|
||||
|
||||
/**
|
||||
* The reply shapes Scribe ships (milestone 500) — `GET /api/retrieval/reply-shapes`.
|
||||
* The same catalog the `list_reply_shapes` MCP tool returns, plus what only
|
||||
* the operator's view needs: the preferences mounted on each shape's moment,
|
||||
* and how often each shape was delivered.
|
||||
*/
|
||||
export interface ShapePreference {
|
||||
id: number;
|
||||
title: string;
|
||||
statement: string;
|
||||
}
|
||||
|
||||
/** How often a shape went out: whole, or as its one-line reminder. */
|
||||
export interface ShapeDeliveries {
|
||||
full: number;
|
||||
pointer: number;
|
||||
}
|
||||
|
||||
export interface ReplyShape {
|
||||
key: string;
|
||||
title: string;
|
||||
/** The moment it rides — a preference mounted here adjusts it. */
|
||||
moment: string;
|
||||
/** What is happening at that moment. */
|
||||
means: string;
|
||||
/** When the shape arrives, in words. */
|
||||
delivered: string;
|
||||
/** The shipped text. Product, so read-only. */
|
||||
text: string;
|
||||
preferences: ShapePreference[];
|
||||
/** null when the counts could not be read — unknown, not zero. */
|
||||
deliveries: ShapeDeliveries | null;
|
||||
}
|
||||
|
||||
export interface ReplyShapesOverview {
|
||||
shapes: ReplyShape[];
|
||||
total: number;
|
||||
days: number;
|
||||
deliveries_failed: boolean;
|
||||
}
|
||||
|
||||
export async function getReplyShapes(days?: number): Promise<ReplyShapesOverview> {
|
||||
const q = days ? `?days=${days}` : "";
|
||||
return apiGet<ReplyShapesOverview>(`/api/retrieval/reply-shapes${q}`);
|
||||
}
|
||||
@@ -54,6 +54,15 @@ export interface RulebookTopic {
|
||||
*/
|
||||
export type RuleKind = "rule" | "preference";
|
||||
|
||||
/** A new record opened from somewhere that already knows what it is for —
|
||||
* "Adjust this" on a reply shape opens a preference mounted on that shape's
|
||||
* moment (milestone 500 step 5). The editor applies it once, when creating. */
|
||||
export interface RulePreset {
|
||||
kind?: RuleKind;
|
||||
moments?: string[];
|
||||
whenToApply?: string;
|
||||
}
|
||||
|
||||
export interface Rule {
|
||||
id: number;
|
||||
topic_id: number | null;
|
||||
|
||||
@@ -0,0 +1,226 @@
|
||||
<script setup lang="ts">
|
||||
import { computed, onMounted, ref } from "vue";
|
||||
import { apiErrorMessage } from "@/api/client";
|
||||
import { getReplyShapes, type ReplyShape, type ReplyShapesOverview } from "@/api/replyShapes";
|
||||
import type { RulePreset } from "@/api/rulebooks";
|
||||
import RuleEditorSlideOver from "@/components/rules/RuleEditorSlideOver.vue";
|
||||
|
||||
/**
|
||||
* The reply shapes Scribe ships, and the operator's adjustments to each
|
||||
* (milestone 500 step 5). A shape is product, so its text is read-only here;
|
||||
* what the operator owns is the preferences mounted on the shape's moment,
|
||||
* which arrive beside it and win where they differ. "Adjust this" opens the
|
||||
* preference editor already mounted there.
|
||||
*/
|
||||
const data = ref<ReplyShapesOverview | null>(null);
|
||||
const loading = ref(true);
|
||||
const error = ref("");
|
||||
const open = ref<string | null>(null);
|
||||
|
||||
async function load() {
|
||||
loading.value = true;
|
||||
error.value = "";
|
||||
try {
|
||||
data.value = await getReplyShapes();
|
||||
} catch (e: unknown) {
|
||||
error.value = apiErrorMessage(e, "Could not load the reply shapes");
|
||||
} finally {
|
||||
loading.value = false;
|
||||
}
|
||||
}
|
||||
|
||||
const shapes = computed(() => data.value?.shapes ?? []);
|
||||
const adjusted = computed(() => shapes.value.filter((s) => s.preferences.length).length);
|
||||
const delivered = computed(() =>
|
||||
shapes.value.reduce((n, s) => n + (s.deliveries ? s.deliveries.full + s.deliveries.pointer : 0), 0),
|
||||
);
|
||||
|
||||
// One line answering "what do my replies get shaped by?".
|
||||
const verdict = computed(() => {
|
||||
if (!data.value) return "";
|
||||
const n = shapes.value.length;
|
||||
const parts = [`${n} ${n === 1 ? "shape ships" : "shapes ship"} with Scribe`];
|
||||
parts.push(adjusted.value
|
||||
? `${adjusted.value} adjusted by your preferences`
|
||||
: "none adjusted — the defaults stand");
|
||||
if (!data.value.deliveries_failed) parts.push(`delivered ${delivered.value} times in ${data.value.days} days`);
|
||||
return parts.join(" · ");
|
||||
});
|
||||
|
||||
function deliveryLabel(s: ReplyShape): string {
|
||||
if (!s.deliveries) return "—";
|
||||
const { full, pointer } = s.deliveries;
|
||||
if (!full && !pointer) return "not delivered";
|
||||
return `${full} in full · ${pointer} as reminder`;
|
||||
}
|
||||
|
||||
function toggle(key: string) {
|
||||
open.value = open.value === key ? null : key;
|
||||
}
|
||||
|
||||
// ── The editor: an existing preference, or a new one mounted here ─────────
|
||||
const editingId = ref<number | null>(null);
|
||||
const preset = ref<RulePreset | null>(null);
|
||||
const editorOpen = computed(() => editingId.value !== null || preset.value !== null);
|
||||
|
||||
function editPreference(id: number) {
|
||||
preset.value = null;
|
||||
editingId.value = id;
|
||||
}
|
||||
|
||||
function adjust(s: ReplyShape) {
|
||||
editingId.value = null;
|
||||
preset.value = {
|
||||
kind: "preference",
|
||||
moments: [s.moment],
|
||||
// A starting trigger in the moment's own words; the operator sharpens it.
|
||||
whenToApply: `When ${s.means}.`,
|
||||
};
|
||||
}
|
||||
|
||||
function closeEditor() {
|
||||
editingId.value = null;
|
||||
preset.value = null;
|
||||
void load();
|
||||
}
|
||||
|
||||
onMounted(load);
|
||||
</script>
|
||||
|
||||
<template>
|
||||
<div class="reply-shapes">
|
||||
<p v-if="loading && !data" class="state">Loading…</p>
|
||||
<p v-else-if="error" class="state state-error" role="alert">
|
||||
{{ error }}
|
||||
<button type="button" class="btn-text" @click="load">Retry</button>
|
||||
</p>
|
||||
<template v-else-if="data">
|
||||
<p class="verdict">{{ verdict }}</p>
|
||||
<p v-if="data.deliveries_failed" class="state state-error" role="alert">
|
||||
Delivery counts could not be read, so they show as “—”.
|
||||
<button type="button" class="btn-text" @click="load">Retry</button>
|
||||
</p>
|
||||
|
||||
<ul class="shape-list">
|
||||
<li v-for="s in shapes" :key="s.key" :class="['shape-row', { open: open === s.key }]">
|
||||
<button
|
||||
type="button"
|
||||
class="shape-head"
|
||||
:aria-expanded="open === s.key"
|
||||
:aria-controls="`shape-${s.key}`"
|
||||
@click="toggle(s.key)"
|
||||
>
|
||||
<span class="shape-name">
|
||||
<strong>{{ s.title }}</strong>
|
||||
<span class="shape-when">{{ s.delivered }}</span>
|
||||
</span>
|
||||
<span class="shape-stats">
|
||||
<span :class="{ 'stat-none': !s.preferences.length }">
|
||||
{{ s.preferences.length
|
||||
? `${s.preferences.length} ${s.preferences.length === 1 ? "preference" : "preferences"}`
|
||||
: "default" }}
|
||||
</span>
|
||||
<span :class="{ 'stat-none': deliveryLabel(s) === 'not delivered' }">{{ deliveryLabel(s) }}</span>
|
||||
<span class="chevron" aria-hidden="true">{{ open === s.key ? "▾" : "▸" }}</span>
|
||||
</span>
|
||||
</button>
|
||||
|
||||
<div v-if="open === s.key" :id="`shape-${s.key}`" class="shape-body">
|
||||
<h4 class="body-label">Ships with Scribe · at <code>{{ s.moment }}</code></h4>
|
||||
<div class="shape-text">{{ s.text }}</div>
|
||||
|
||||
<h4 class="body-label">Your adjustments</h4>
|
||||
<ul v-if="s.preferences.length" class="pref-list">
|
||||
<li v-for="p in s.preferences" :key="p.id">
|
||||
<button type="button" class="pref-row" @click="editPreference(p.id)">
|
||||
<span class="pref-title">{{ p.title }}</span>
|
||||
<span class="chevron" aria-hidden="true">›</span>
|
||||
</button>
|
||||
</li>
|
||||
</ul>
|
||||
<p v-else class="stat-none">None — this shape arrives as shipped.</p>
|
||||
<button type="button" class="btn-primary adjust" @click="adjust(s)">Adjust this</button>
|
||||
</div>
|
||||
</li>
|
||||
</ul>
|
||||
</template>
|
||||
|
||||
<RuleEditorSlideOver
|
||||
v-if="editorOpen"
|
||||
:rule-id="editingId"
|
||||
:topic-id="null"
|
||||
:preset="preset"
|
||||
@close="closeEditor"
|
||||
/>
|
||||
</div>
|
||||
</template>
|
||||
|
||||
<style scoped>
|
||||
.reply-shapes { display: flex; flex-direction: column; gap: var(--fs-space-2); }
|
||||
.verdict { margin: 0; font-size: var(--fs-size-body-sm); color: var(--fs-text-primary); }
|
||||
.state { margin: 0; font-size: var(--fs-size-body-sm); color: var(--fs-text-secondary); }
|
||||
.state-error { color: var(--fs-error); }
|
||||
|
||||
.shape-list { list-style: none; margin: 0; padding: 0; display: flex; flex-direction: column; gap: var(--fs-space-2); }
|
||||
.shape-row {
|
||||
background: var(--fs-surface-page);
|
||||
border: 1px solid var(--fs-border-color);
|
||||
border-radius: var(--fs-radius-md);
|
||||
}
|
||||
/* The whole row is the tap target (a settings row, not prose with a link). */
|
||||
.shape-head {
|
||||
width: 100%;
|
||||
display: flex; align-items: baseline; gap: var(--fs-space-3);
|
||||
padding: var(--fs-space-3);
|
||||
background: none; border: 0; border-radius: var(--fs-radius-md);
|
||||
color: inherit; font: inherit; text-align: left; cursor: pointer;
|
||||
}
|
||||
.shape-head:focus-visible { outline: 2px solid var(--fs-focus-ring); outline-offset: 2px; }
|
||||
.shape-name { display: flex; flex-direction: column; gap: var(--fs-space-1); min-width: 0; }
|
||||
.shape-name strong { font-size: var(--fs-size-body-sm); color: var(--fs-text-primary); }
|
||||
.shape-when {
|
||||
font-size: var(--fs-size-tiny); color: var(--fs-text-tertiary);
|
||||
overflow: hidden; text-overflow: ellipsis; white-space: nowrap;
|
||||
}
|
||||
.shape-stats {
|
||||
margin-left: auto;
|
||||
display: flex; align-items: baseline; gap: var(--fs-space-3);
|
||||
font-size: var(--fs-size-tiny); color: var(--fs-text-secondary);
|
||||
font-variant-numeric: tabular-nums; white-space: nowrap;
|
||||
}
|
||||
.stat-none { color: var(--fs-text-tertiary); font-style: italic; }
|
||||
.chevron { color: var(--fs-text-tertiary); }
|
||||
|
||||
.shape-body {
|
||||
display: flex; flex-direction: column; align-items: flex-start; gap: var(--fs-space-2);
|
||||
padding: 0 var(--fs-space-3) var(--fs-space-3);
|
||||
}
|
||||
.body-label {
|
||||
margin: var(--fs-space-2) 0 0;
|
||||
font-size: var(--fs-size-tiny); font-weight: normal;
|
||||
text-transform: uppercase; letter-spacing: var(--fs-tracking-tiny);
|
||||
color: var(--fs-text-tertiary);
|
||||
}
|
||||
.body-label code { font-family: var(--fs-font-mono); text-transform: none; }
|
||||
/* Product text, shown as it reaches a session. Read-only by design. */
|
||||
.shape-text {
|
||||
width: 100%;
|
||||
padding: var(--fs-space-2) var(--fs-space-3);
|
||||
border-left: 2px solid var(--fs-border-color);
|
||||
font-size: var(--fs-size-body-sm); line-height: var(--fs-leading-body);
|
||||
color: var(--fs-text-secondary); white-space: pre-wrap;
|
||||
}
|
||||
.pref-list { list-style: none; margin: 0; padding: 0; width: 100%; display: flex; flex-direction: column; gap: var(--fs-space-1); }
|
||||
.pref-row {
|
||||
width: 100%;
|
||||
display: flex; align-items: baseline; gap: var(--fs-space-2);
|
||||
padding: var(--fs-space-1) var(--fs-space-2);
|
||||
background: none; border: 1px solid var(--fs-border-color); border-radius: var(--fs-radius-sm);
|
||||
color: var(--fs-text-primary); font: inherit; font-size: var(--fs-size-body-sm);
|
||||
text-align: left; cursor: pointer;
|
||||
}
|
||||
.pref-row:focus-visible { outline: 2px solid var(--fs-focus-ring); outline-offset: 2px; }
|
||||
.pref-title { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
|
||||
.pref-row .chevron { margin-left: auto; }
|
||||
.adjust { margin-top: var(--fs-space-1); }
|
||||
</style>
|
||||
@@ -5,9 +5,14 @@ import { useCanonicalSystemsStore } from "@/stores/canonicalSystems";
|
||||
import { useMomentsStore } from "@/stores/moments";
|
||||
import RuleHistoryPanel from "@/components/rules/RuleHistoryPanel.vue";
|
||||
import RuleHomePicker from "@/components/rules/RuleHomePicker.vue";
|
||||
import type { Rule, RuleKind } from "@/api/rulebooks";
|
||||
import { apiErrorMessage } from "@/api/client";
|
||||
import { listRulebooks, listTopics, type Rule, type RuleKind, type RulePreset } from "@/api/rulebooks";
|
||||
|
||||
const props = defineProps<{ ruleId: number | null; topicId: number | null }>();
|
||||
const props = defineProps<{
|
||||
ruleId: number | null;
|
||||
topicId: number | null;
|
||||
preset?: RulePreset | null;
|
||||
}>();
|
||||
const emit = defineEmits<{ close: [] }>();
|
||||
|
||||
const store = useRulebooksStore();
|
||||
@@ -128,10 +133,10 @@ async function load() {
|
||||
} else {
|
||||
title.value = "";
|
||||
statement.value = "";
|
||||
whenToApply.value = "";
|
||||
kind.value = "rule";
|
||||
whenToApply.value = props.preset?.whenToApply ?? "";
|
||||
kind.value = props.preset?.kind ?? "rule";
|
||||
systemIds.value = [];
|
||||
ruleMoments.value = [];
|
||||
ruleMoments.value = [...(props.preset?.moments ?? [])];
|
||||
why.value = "";
|
||||
howToApply.value = "";
|
||||
verifyWith.value = "";
|
||||
@@ -139,7 +144,31 @@ async function load() {
|
||||
}
|
||||
procedureDraft.value = "";
|
||||
procedureError.value = "";
|
||||
await Promise.all([canon.fetchCatalog(), momentsStore.load()]);
|
||||
await Promise.all([canon.fetchCatalog(), momentsStore.load(), loadHomes()]);
|
||||
}
|
||||
|
||||
// ── Where a new record lives, when the opener did not say ──────────────────
|
||||
// The rules view opens the editor inside a topic; Settings does not, so a new
|
||||
// record opened there asks for its home. A rulebook topic is global — it
|
||||
// applies in every project, as a reply shape does.
|
||||
const homeChoices = ref<{ id: number; label: string }[]>([]);
|
||||
const chosenTopic = ref<number | null>(null);
|
||||
const homeError = ref("");
|
||||
const needsHome = computed(() => isCreating.value && props.topicId === null);
|
||||
|
||||
async function loadHomes() {
|
||||
if (!needsHome.value || homeChoices.value.length) return;
|
||||
try {
|
||||
const books = await listRulebooks();
|
||||
const perBook = await Promise.all(books.map(async (rb) => {
|
||||
const ts = await listTopics(rb.id);
|
||||
return ts.map((t) => ({ id: t.id, label: `${rb.title} › ${t.title}` }));
|
||||
}));
|
||||
homeChoices.value = perBook.flat();
|
||||
if (homeChoices.value.length === 1) chosenTopic.value = homeChoices.value[0].id;
|
||||
} catch (e: unknown) {
|
||||
homeError.value = apiErrorMessage(e, "Could not load where this could live");
|
||||
}
|
||||
}
|
||||
|
||||
async function save() {
|
||||
@@ -166,8 +195,14 @@ async function save() {
|
||||
verify_with: verifyWith.value,
|
||||
expires_when: expiresWhen.value,
|
||||
};
|
||||
if (isCreating.value && props.topicId !== null) {
|
||||
await store.createRule(props.topicId, fields);
|
||||
const home = props.topicId ?? chosenTopic.value;
|
||||
if (isCreating.value && home === null) {
|
||||
// Written but homeless: keep the draft open rather than lose it on close.
|
||||
homeError.value = "Choose where it lives before saving.";
|
||||
return;
|
||||
}
|
||||
if (isCreating.value && home !== null) {
|
||||
await store.createRule(home, fields);
|
||||
} else if (props.ruleId !== null) {
|
||||
await store.updateRule(props.ruleId, fields);
|
||||
}
|
||||
@@ -201,6 +236,14 @@ watch(() => props.ruleId, load);
|
||||
<button v-if="!isCreating" class="trash" @click="remove" aria-label="Delete">🗑</button>
|
||||
<button class="close" @click="save" aria-label="Close">×</button>
|
||||
</header>
|
||||
<label v-if="needsHome">
|
||||
Where it lives
|
||||
<select v-model="chosenTopic" class="fs-input">
|
||||
<option :value="null" disabled>Choose a rulebook topic</option>
|
||||
<option v-for="h in homeChoices" :key="h.id" :value="h.id">{{ h.label }}</option>
|
||||
</select>
|
||||
</label>
|
||||
<p v-if="homeError" class="home-error" role="alert">{{ homeError }}</p>
|
||||
<label>
|
||||
Title
|
||||
<input v-model="title" placeholder="e.g. dev is home" />
|
||||
@@ -421,6 +464,7 @@ watch(() => props.ruleId, load);
|
||||
one index everywhere (tests/test_frontend_shared_styles.py). -->
|
||||
<style src="@/assets/rules-shared.css" />
|
||||
<style scoped>
|
||||
.home-error { margin: 0 0 var(--fs-space-2); color: var(--fs-error); font-size: var(--fs-size-body-sm); }
|
||||
.kind { border: 1px solid var(--fs-border-color); border-radius: var(--fs-radius-md); padding: var(--fs-space-3); margin: var(--fs-space-3) 0; }
|
||||
.kind legend { font-size: var(--fs-size-tiny); text-transform: uppercase; letter-spacing: var(--fs-tracking-tiny); color: var(--fs-text-tertiary); padding: 0 var(--fs-space-2); }
|
||||
.kind-opt { display: flex; align-items: flex-start; gap: var(--fs-space-2); margin-bottom: var(--fs-space-2); font-size: var(--fs-size-body-sm); line-height: var(--fs-leading-body); }
|
||||
|
||||
@@ -10,6 +10,7 @@ import type { User } from "@/types/auth";
|
||||
import PaginationBar from "@/components/PaginationBar.vue";
|
||||
import TagInput from "@/components/TagInput.vue";
|
||||
import MomentsSettings from "@/components/MomentsSettings.vue";
|
||||
import ReplyShapesSettings from "@/components/ReplyShapesSettings.vue";
|
||||
import { fmtDate, fmtLogStamp } from "@/utils/dateFormat";
|
||||
import { fetchVersion, type VersionPayload } from "@/api/version";
|
||||
|
||||
@@ -168,7 +169,6 @@ const kbToolRuleThreshold = ref("0.68");
|
||||
// And the prompt boundary is a third query shape again — the operator's own
|
||||
// prose rather than anything a tool produced (#3852).
|
||||
const kbPromptRuleThreshold = ref("0.72");
|
||||
const kbReportPrefThreshold = ref("0.72");
|
||||
// The reply arm (milestone 458): its bar is the bar at which a finished reply
|
||||
// is HELD for one read, so it sits with the checkpoint, not with the hints.
|
||||
const kbReplyRuleThreshold = ref("0.8");
|
||||
@@ -181,7 +181,6 @@ const kbWritePathTopK = ref("3");
|
||||
const kbRuleHintTopK = ref("5");
|
||||
const kbToolRuleTopK = ref("5");
|
||||
const kbPromptRuleTopK = ref("3");
|
||||
const kbReportPrefTopK = ref("3");
|
||||
const kbReplyRuleTopK = ref("1");
|
||||
// What has been changed about retrieval, newest first — the review surface for
|
||||
// changes the model made on the operator's behalf (#4102).
|
||||
@@ -387,7 +386,6 @@ async function saveKbInject() {
|
||||
// this is the only number that can STOP a call, so a fallback of 0 would
|
||||
// hold the first command of every session behind whatever ranked first.
|
||||
const cpT = asBar(kbCheckpointThreshold.value, 0.8);
|
||||
const rpT = asBar(kbReportPrefThreshold.value, 0.72);
|
||||
// A stop bar like the checkpoint's: a fallback of 0 would hold every reply.
|
||||
const ryT = asBar(kbReplyRuleThreshold.value, 0.8);
|
||||
// The budgets, clamped the way the server clamps them: a whole number in
|
||||
@@ -400,13 +398,11 @@ async function saveKbInject() {
|
||||
const rhK = asK(kbRuleHintTopK.value, 5);
|
||||
const trK = asK(kbToolRuleTopK.value, 5);
|
||||
const prK = asK(kbPromptRuleTopK.value, 3);
|
||||
const rpK = asK(kbReportPrefTopK.value, 3);
|
||||
const ryK = asK(kbReplyRuleTopK.value, 1);
|
||||
kbWritePathTopK.value = String(wpK);
|
||||
kbRuleHintTopK.value = String(rhK);
|
||||
kbToolRuleTopK.value = String(trK);
|
||||
kbPromptRuleTopK.value = String(prK);
|
||||
kbReportPrefTopK.value = String(rpK);
|
||||
kbReplyRuleTopK.value = String(ryK);
|
||||
kbInjectThreshold.value = String(t);
|
||||
kbInjectTopK.value = String(k);
|
||||
@@ -424,7 +420,6 @@ async function saveKbInject() {
|
||||
kbCheckpointThreshold.value = String(cpT);
|
||||
kbToolRuleThreshold.value = String(trT);
|
||||
kbPromptRuleThreshold.value = String(prT);
|
||||
kbReportPrefThreshold.value = String(rpT);
|
||||
kbReplyRuleThreshold.value = String(ryT);
|
||||
savingKbInject.value = true;
|
||||
kbInjectSaved.value = false;
|
||||
@@ -453,12 +448,6 @@ async function saveKbInject() {
|
||||
// queries are different shapes. Moving one must not move the others.
|
||||
kb_toolrule_threshold: String(trT),
|
||||
kb_promptrule_threshold: String(prT),
|
||||
// A SIXTH, and the one that most needed its own key: this arm's query
|
||||
// is a fixed string, so its score is a constant for a given corpus.
|
||||
// While it borrowed the prompt bar, tuning prose silently retuned it —
|
||||
// and a constant that lands under the bar is a dead arm, not a quiet
|
||||
// one (#3860).
|
||||
kb_reportpref_threshold: String(rpT),
|
||||
// The reply backstop's own bar: it HOLDS a reply, so like the
|
||||
// checkpoint it is never derived from a hint bar.
|
||||
kb_replyrule_threshold: String(ryT),
|
||||
@@ -470,7 +459,6 @@ async function saveKbInject() {
|
||||
kb_rulehint_top_k: String(rhK),
|
||||
kb_toolrule_top_k: String(trK),
|
||||
kb_promptrule_top_k: String(prK),
|
||||
kb_reportpref_top_k: String(rpK),
|
||||
kb_replyrule_top_k: String(ryK),
|
||||
kb_duplicate_threshold_snippet: String(dupSnip),
|
||||
kb_duplicate_threshold_note: String(dupNote),
|
||||
@@ -941,9 +929,6 @@ onMounted(async () => {
|
||||
if (allSettings.kb_promptrule_threshold !== undefined) {
|
||||
kbPromptRuleThreshold.value = allSettings.kb_promptrule_threshold;
|
||||
}
|
||||
if (allSettings.kb_reportpref_threshold !== undefined) {
|
||||
kbReportPrefThreshold.value = allSettings.kb_reportpref_threshold;
|
||||
}
|
||||
if (allSettings.kb_replyrule_threshold !== undefined) {
|
||||
kbReplyRuleThreshold.value = allSettings.kb_replyrule_threshold;
|
||||
}
|
||||
@@ -967,9 +952,6 @@ onMounted(async () => {
|
||||
if (allSettings.kb_promptrule_top_k !== undefined) {
|
||||
kbPromptRuleTopK.value = allSettings.kb_promptrule_top_k;
|
||||
}
|
||||
if (allSettings.kb_reportpref_top_k !== undefined) {
|
||||
kbReportPrefTopK.value = allSettings.kb_reportpref_top_k;
|
||||
}
|
||||
if (allSettings.kb_replyrule_top_k !== undefined) {
|
||||
kbReplyRuleTopK.value = allSettings.kb_replyrule_top_k;
|
||||
}
|
||||
@@ -1974,42 +1956,6 @@ async function deleteUser(userId: number) {
|
||||
/>
|
||||
<p class="field-hint">How many rules or preferences one message may be shown (1–10).</p>
|
||||
</div>
|
||||
<div class="field">
|
||||
<label for="kb-reportpref-threshold">Completion-report confidence threshold (0–1)</label>
|
||||
<input
|
||||
id="kb-reportpref-threshold"
|
||||
v-model="kbReportPrefThreshold"
|
||||
type="number"
|
||||
min="0"
|
||||
max="1"
|
||||
step="0.01"
|
||||
class="fs-input input"
|
||||
style="max-width: 8rem"
|
||||
/>
|
||||
<p class="field-hint">
|
||||
The bar for a <em>preference about how a completion report should be
|
||||
written</em>, looked up when a task closes. Unlike every other bar
|
||||
here, the question this arm asks never changes — so its score is
|
||||
fixed by your preferences alone, and it will either always find one
|
||||
or never find one, and no run of calls will reveal a dead one on its
|
||||
own. That is why this arm is worth looking up in the panel below when
|
||||
a report preference never seems to arrive.
|
||||
</p>
|
||||
</div>
|
||||
<div class="field">
|
||||
<label for="kb-reportpref-topk">Max report preferences</label>
|
||||
<input
|
||||
id="kb-reportpref-topk"
|
||||
v-model="kbReportPrefTopK"
|
||||
type="number"
|
||||
min="1"
|
||||
max="10"
|
||||
step="1"
|
||||
class="fs-input input"
|
||||
style="max-width: 8rem"
|
||||
/>
|
||||
<p class="field-hint">How many preferences a finished task may be shown (1–10).</p>
|
||||
</div>
|
||||
<div class="field">
|
||||
<label for="kb-replyrule-threshold">Reply check threshold (0–1)</label>
|
||||
<input
|
||||
@@ -2305,6 +2251,15 @@ async function deleteUser(userId: number) {
|
||||
<MomentsSettings />
|
||||
</section>
|
||||
|
||||
<section class="settings-section full-width">
|
||||
<h2>Reply shapes</h2>
|
||||
<p class="section-desc">
|
||||
The default shape of each kind of reply, delivered to a session when it is due.
|
||||
Adjust one with a preference; it arrives beside the shape and wins.
|
||||
</p>
|
||||
<ReplyShapesSettings />
|
||||
</section>
|
||||
|
||||
</div>
|
||||
|
||||
<!-- ── Account ── -->
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "scribe",
|
||||
"description": "Scribe for Claude Code: connects the scribe MCP server, adds the hooks that deliver live project state and relevant records at the right moment, ships the shared client-neutral Scribe skills (using-scribe, writing-plans, reporting-back, systematic-debugging, verification, brainstorming, reusing-code, shape-accounting, family-canon), and syncs your saved Scribe Processes as skills (/scribe:sync).",
|
||||
"version": "2026.10.06.1827",
|
||||
"version": "2026.10.09.1955",
|
||||
"author": {
|
||||
"name": "Bryan Van Deusen"
|
||||
},
|
||||
|
||||
+2
-3
@@ -15,7 +15,7 @@ another one means adding files, not moving or rewriting any.
|
||||
|---|---|---|
|
||||
| **The skills** | `plugin/skills/*/SKILL.md` | Agent Skills (the open SKILL.md format). They state every Scribe reflex in full and name no client. `tests/test_guidance_ownership.py` fails if a skill names a particular client, or references anything outside its own folder. Every client package ships this folder verbatim. |
|
||||
| **The MCP server** | `<base URL>/mcp` | HTTP, `Authorization: Bearer <fmcp_ key>`. Its `_INSTRUCTIONS` is a client-neutral orientation — the workflow across tools (≤1,600 chars, #4389); each tool's description carries its contract; in-band responses (`placement`, `report_back`, `systems_hint`, the duplicate gate, the guessed-id refusal) fire in every client. |
|
||||
| **The adapter API** | `<base URL>/api/plugin/*` | Plain `GET` endpoints any client's hooks can call with the same key (read scope is enough): `context` (live session state), `retrieve` (rules, preferences and notes for a message), `prior-art` (records and shape-ledger hints for code being written), `tool-rules` (rules for a command about to run), `report-check` (records a completion-report check and returns the reason for a block), `processes` (stored Processes to expose as skills). |
|
||||
| **The adapter API** | `<base URL>/api/plugin/*` | Plain `GET` endpoints any client's hooks can call with the same key (read scope is enough): `context` (live session state), `retrieve` (rules, preferences and notes for a message), `prior-art` (records and shape-ledger hints for code being written), `tool-rules` (rules for a command about to run), `reply-rules` (a `POST`: the finished reply, held for an unopened rule or — when the turn closed a task — for missing completion sections), `processes` (stored Processes to expose as skills). |
|
||||
| **The API key** | Scribe → Settings → API Keys | One `fmcp_` key per install. Read scope for hooks; write scope for the MCP tools. |
|
||||
|
||||
## Added by each client
|
||||
@@ -40,8 +40,7 @@ another one means adding files, not moving or rewriting any.
|
||||
| `hooks/scribe_after_write.sh` | PostToolUse on shell commands: the same check for code written through the shell. |
|
||||
| `hooks/scribe_tool_rules.sh` | PreToolUse on shell commands: `GET /api/plugin/tool-rules`. |
|
||||
| `hooks/scribe_moment.sh` | PreToolUse on every tool: the rules mounted on the moments the call reaches, `POST /api/plugin/moment`; skips tools `GET /api/plugin/moment-tools` says reach nothing mounted, and Scribe's own (their responses carry `moment_rules`). |
|
||||
| `hooks/scribe_report_check.sh` | Stop: when the turn closed a task, checks the reply for the completion sections and reports to `GET /api/plugin/report-check`; blocks once, with the reason the server returns. |
|
||||
| `hooks/scribe_reply_check.sh` | Stop: sends the finished reply to `POST /api/plugin/reply-rules` — the rules mounted on the reply moments, and the reply against every rule's trigger; holds once per rule, with the reason the server returns, and never holds the rewrite. |
|
||||
| `hooks/scribe_reply_check.sh` | Stop: sends the finished reply to `POST /api/plugin/reply-rules` — the rules mounted on the reply moments, and the reply against every rule's trigger, and — when the turn closed a task — the completion-section check; holds once per rule, with the reason the server returns, and never holds the rewrite. |
|
||||
| `hooks/scribe_shape_check.sh` | Stop: sends the definitions the turn wrote (the write hooks' `<sid>.written.ids` ledger) to `GET /api/plugin/shape-check`; blocks once, with the reason the server returns, so the agent judges what it built. |
|
||||
| `hooks/scribe_sync_processes.sh` + `commands/sync.md` | `GET /api/plugin/processes` → `~/.claude/skills/scribe-proc-*` stubs; `/scribe:sync` on demand. |
|
||||
| `hooks/scribe_defs.sh` | Shared shell helpers: config, dedup ledgers, outage line. |
|
||||
|
||||
+10
-9
@@ -82,15 +82,16 @@ On install you'll be asked for:
|
||||
answer" line (8 s budget here — it runs after the tool, so it gates
|
||||
nothing). The extractor, the prose/data skip list, the local by-name
|
||||
duplicate arm and the outage line are shared in `hooks/scribe_defs.sh`.
|
||||
- `hooks/hooks.json` → Stop hook (`hooks/scribe_report_check.sh`): when the
|
||||
turn closed a Scribe task (`update_task`/`create_task` with status done),
|
||||
checks the reply that ends it for the completion sections — where the work
|
||||
sits, what needs you, what comes next — and reports the outcome to
|
||||
`GET /api/plugin/report-check`. If sections are missing it blocks once with
|
||||
the reason the server returns, and records how the rewrite came out; it
|
||||
never blocks twice, and never blocks when the instance did not record the
|
||||
check (unconfigured or unreachable). Outcomes land in the admin logs under
|
||||
category `plugin`, action `report_check`.
|
||||
- `hooks/hooks.json` → Stop hook (`hooks/scribe_reply_check.sh`): sends the
|
||||
reply that ends the turn to `POST /api/plugin/reply-rules`, which holds it
|
||||
once for an unopened rule mounted on the reply moments or matching the reply.
|
||||
When the turn closed a Scribe task (`update_task`/`create_task` with status
|
||||
done), the same request has the server check the reply for the completion
|
||||
sections — where the work sits, what needs you, what comes next. If they are
|
||||
missing it blocks once with the reason the server returns, and records how
|
||||
the rewrite came out; it never blocks twice, and never blocks when the
|
||||
instance did not record the check (unconfigured or unreachable). Outcomes
|
||||
land in the admin logs under category `plugin`, action `report_check`.
|
||||
- `hooks/hooks.json` → a second Stop hook (`hooks/scribe_shape_check.sh`): the
|
||||
write hooks note every definition a write names (and every new file) in a
|
||||
session ledger; at the end of the turn this sends them to
|
||||
|
||||
@@ -104,10 +104,6 @@
|
||||
"Stop": [
|
||||
{
|
||||
"hooks": [
|
||||
{
|
||||
"type": "command",
|
||||
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_report_check.sh\""
|
||||
},
|
||||
{
|
||||
"type": "command",
|
||||
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_reply_check.sh\""
|
||||
|
||||
@@ -110,6 +110,7 @@ rule_state_dir="${TMPDIR:-/tmp}/scribe-priorart"
|
||||
mkdir -p "$rule_state_dir" 2>/dev/null || true
|
||||
idfile=""
|
||||
rulefile=""
|
||||
shapefile=""
|
||||
exclude_q=""
|
||||
if [ -n "$session_id" ]; then
|
||||
# session_id is an opaque token from Claude Code; keep only filename-safe chars.
|
||||
@@ -127,6 +128,11 @@ if [ -n "$session_id" ]; then
|
||||
[ -n "$rule_seen" ] && exclude_q="${exclude_q}&exclude_rule_ids=${rule_seen}"
|
||||
# What the session actually OPENED, as against what it was shown (#4100).
|
||||
exclude_q="${exclude_q}$(scribe_held_query "$rule_state_dir/${safe_sid}.opened.ids")"
|
||||
# Which reply shapes it already holds in full (milestone 500): the core is
|
||||
# sent whole once, and as its one-line reminder on every later turn.
|
||||
shapefile=$(scribe_shapes_file "$safe_sid")
|
||||
shapes_seen=$(scribe_shapes_seen "$shapefile")
|
||||
[ -n "$shapes_seen" ] && exclude_q="${exclude_q}&shapes_seen=${shapes_seen}"
|
||||
fi
|
||||
|
||||
body=$(curl -fsS --max-time 5 \
|
||||
@@ -150,6 +156,9 @@ if [ -n "$rulefile" ]; then
|
||||
scribe_json_list "$body_flat" '.rule_ids' \
|
||||
| scribe_rules_append "$rulefile"
|
||||
fi
|
||||
if [ -n "$shapefile" ]; then
|
||||
scribe_json_list "$body_flat" '.shape_keys' | scribe_shapes_append "$shapefile"
|
||||
fi
|
||||
|
||||
scribe_json_out UserPromptSubmit "$context"
|
||||
exit 0
|
||||
|
||||
@@ -135,8 +135,8 @@ scribe_json_flat_lines() {
|
||||
# Cheap by construction, because it runs on every prompt: the last 512 KB only,
|
||||
# and grep narrows to assistant records carrying a text block BEFORE anything is
|
||||
# parsed — a transcript is mostly tool results, and parsing those in awk is the
|
||||
# multi-second cost scribe_report_check.sh already measured. Fixed strings with
|
||||
# UNESCAPED quotes can only match at a record's own top level (see that hook).
|
||||
# multi-second cost the completion-report check measured (#4107). Fixed strings
|
||||
# with UNESCAPED quotes can only match at a record's own top level.
|
||||
# A sidechain is a subagent talking, not this session, so it is skipped.
|
||||
scribe_recent_context() {
|
||||
[ -n "${1:-}" ] && [ -f "$1" ] || return 0
|
||||
@@ -205,8 +205,26 @@ scribe_json_len() {
|
||||
}
|
||||
|
||||
# Raw JSON string bodies on stdin → text. One line in, one value out.
|
||||
#
|
||||
# A `\uXXXX` ESCAPE IS ENCODED AS UTF-8 HERE, BYTE BY BYTE, UNDER LC_ALL=C.
|
||||
# The server escapes every non-ASCII character this way (the JSON default),
|
||||
# and `sprintf("%c", n)` for n > 127 means a different thing in every awk and
|
||||
# locale: gawk in a UTF-8 locale writes the character, gawk in the C locale
|
||||
# and mawk write the single byte n % 256. So "·" (U+00B7) arrived as a lone
|
||||
# 0xB7 and "—" (U+2014) as 0x14 wherever the hook ran without a UTF-8 locale —
|
||||
# found by a test that ran the hook with a bare environment (milestone 500).
|
||||
# Under LC_ALL=C every awk writes exactly the byte asked for, so the encoding
|
||||
# below is the only one that happens.
|
||||
scribe_json_unescape() {
|
||||
awk '
|
||||
LC_ALL=C awk '
|
||||
function utf8(n) {
|
||||
if (n < 128) return sprintf("%c", n)
|
||||
if (n < 2048) return sprintf("%c%c", 192 + int(n / 64), 128 + n % 64)
|
||||
if (n < 65536) return sprintf("%c%c%c", 224 + int(n / 4096),
|
||||
128 + int(n / 64) % 64, 128 + n % 64)
|
||||
return sprintf("%c%c%c%c", 240 + int(n / 262144), 128 + int(n / 4096) % 64,
|
||||
128 + int(n / 64) % 64, 128 + n % 64)
|
||||
}
|
||||
function hex4(h, i, c, d, v) {
|
||||
v = 0
|
||||
for (i = 1; i <= 4; i++) {
|
||||
@@ -246,7 +264,7 @@ scribe_json_unescape() {
|
||||
i += 6
|
||||
}
|
||||
}
|
||||
o = o sprintf("%c", hi)
|
||||
o = o utf8(hi)
|
||||
}
|
||||
else o = o d
|
||||
}
|
||||
@@ -1208,6 +1226,35 @@ scribe_written_append() {
|
||||
return 0
|
||||
}
|
||||
|
||||
# THE REPLY-SHAPE LEDGER (milestone 500). Which default reply shapes this
|
||||
# session has been shown IN FULL — `core`, `completion`, `asks`, `plan` — one
|
||||
# key per line in <sid>.shapes.ids under the prior-art directory. The server
|
||||
# sends a shape in full when its key is absent and as a one-line reminder when
|
||||
# present, so the core arrives whole once and is a pointer on later turns.
|
||||
#
|
||||
# Swept with every other `.ids` ledger on compact and clear, and that is the
|
||||
# point rather than a side effect: after a compaction the session no longer
|
||||
# holds the shape, so the next turn gets it in full again. Not aged — the
|
||||
# per-turn reminder is what keeps it in view between compactions.
|
||||
# scribe_shapes_file <sid> the ledger's path
|
||||
# scribe_shapes_seen <file> its keys, comma-joined, for `shapes_seen=`
|
||||
# scribe_shapes_append <file> keys on stdin, appended
|
||||
scribe_shapes_file() {
|
||||
printf '%s/scribe-priorart/%s.shapes.ids' "${TMPDIR:-/tmp}" "$1"
|
||||
}
|
||||
|
||||
scribe_shapes_seen() {
|
||||
[ -n "$1" ] && [ -f "$1" ] || return 0
|
||||
awk '/^[a-z]+$/ && !seen[$0]++ { out = out (out == "" ? "" : ",") $0 } END { print out }' \
|
||||
"$1" 2>/dev/null || true
|
||||
}
|
||||
|
||||
scribe_shapes_append() {
|
||||
[ -n "$1" ] || { cat >/dev/null; return 0; }
|
||||
mkdir -p "$(dirname "$1")" 2>/dev/null || true
|
||||
awk '/^[a-z]+$/' >> "$1" 2>/dev/null || true
|
||||
}
|
||||
|
||||
scribe_clear_session_ledgers() {
|
||||
# SPARES `*.keep.ids`, which are evidence rather than exclusions — see
|
||||
# `scribe_rules_append`. Everything else still goes: the convention is
|
||||
|
||||
@@ -39,7 +39,7 @@
|
||||
#
|
||||
# mode=lines the input is JSONL — one JSON value per line — and a line that
|
||||
# does not parse is DROPPED, the rest still read. This is the
|
||||
# transcript in scribe_report_check.sh, where the window starts
|
||||
# transcript the Stop hooks read, where the window starts
|
||||
# mid-record by construction: `tail -n 3000` cuts wherever it
|
||||
# cuts, and the first line is routinely half a record. It is
|
||||
# exactly the `map(try fromjson catch empty)` the jq program it
|
||||
|
||||
@@ -122,6 +122,11 @@ ledger_q=""
|
||||
rule_seen=$(scribe_rules_live "$rulefile")
|
||||
[ -n "$rule_seen" ] && ledger_q="&exclude_rule_ids=${rule_seen}"
|
||||
ledger_q="${ledger_q}$(scribe_held_query "$state_dir/${safe_sid}.opened.ids")"
|
||||
# The reply-shape ledger (milestone 500): a shape this session already holds in
|
||||
# full comes back as its reminder.
|
||||
shapefile=$(scribe_shapes_file "$safe_sid")
|
||||
shapes_seen=$(scribe_shapes_seen "$shapefile")
|
||||
[ -n "$shapes_seen" ] && ledger_q="${ledger_q}&shapes_seen=${shapes_seen}"
|
||||
|
||||
# The event whole, since which field a mapping matches is the mapping's
|
||||
# business. A write of a very large file is reduced to its name: the moments a
|
||||
@@ -141,6 +146,7 @@ body=$(printf '%s' "$event" | curl -fsS --max-time 3 \
|
||||
|
||||
body_flat=$(printf '%s' "$body" | scribe_json_flat)
|
||||
scribe_json_list "$body_flat" '.rule_ids' | scribe_rules_append "$rulefile"
|
||||
scribe_json_list "$body_flat" '.shape_keys' | scribe_shapes_append "$shapefile"
|
||||
|
||||
context=$(scribe_json_pick "$body_flat" '.context')
|
||||
[ -n "$context" ] || exit 0
|
||||
|
||||
@@ -9,7 +9,15 @@
|
||||
# text against every rule's trigger, as the backstop for whatever the earlier
|
||||
# arms missed. An unopened rule from either half holds the reply for one read.
|
||||
#
|
||||
# THE SAME CONTRACT AS scribe_report_check.sh, and for its reasons:
|
||||
# ONE END-OF-TURN REQUEST (milestone 500 step 4). When the turn closed a task,
|
||||
# the same request carries the task ids and the server also checks the reply
|
||||
# for the completion sections — once a Stop hook of its own
|
||||
# (scribe_report_check.sh), now the same call. The section check's outcome is
|
||||
# recorded (`report_check`, milestone 409's adherence number), and when it
|
||||
# held the reply, the rewrite is sent back once with `rewrite: true` so the
|
||||
# server can record how it came out. A rewrite is never held.
|
||||
#
|
||||
# THE CONTRACT, and the reasons for it:
|
||||
# - the server decides and supplies the words; this hook blocks only on a
|
||||
# reason it was given, so an unconfigured or unreachable instance never
|
||||
# stops a session;
|
||||
@@ -34,15 +42,44 @@ session_id=$(scribe_json_pick "$event_flat" '.session_id')
|
||||
active=$(scribe_json_pick "$event_flat" '.stop_hook_active')
|
||||
event_cwd=$(scribe_json_pick "$event_flat" '.cwd')
|
||||
|
||||
# The rewrite after a hold goes out as written.
|
||||
[ "$active" = "true" ] && exit 0
|
||||
[ -n "$transcript" ] && [ -f "$transcript" ] && [ -n "$session_id" ] || exit 0
|
||||
|
||||
state_dir="${TMPDIR:-/tmp}/scribe-priorart"
|
||||
mkdir -p "$state_dir" 2>/dev/null || true
|
||||
safe_sid=$(printf '%s' "$session_id" | tr -c 'A-Za-z0-9._-' '_')
|
||||
rulefile="$state_dir/${safe_sid}.rules.ids"
|
||||
stopfile="$state_dir/${safe_sid}.checkpoint.ids"
|
||||
# Set when the section check held the reply: the next stop is its rewrite.
|
||||
# A ledger by name (`.ids`), so a compaction sweeps it with the rest.
|
||||
reportfile="$state_dir/${safe_sid}.reportcheck.ids"
|
||||
|
||||
# The rewrite after a hold goes out as written. Only a rewrite of a SECTION
|
||||
# hold is sent at all — to record how it came out; one after a rule hold, or
|
||||
# another plugin's block, has nothing to report.
|
||||
if [ "$active" = "true" ]; then
|
||||
[ -f "$reportfile" ] || exit 0
|
||||
rm -f "$reportfile" 2>/dev/null || true
|
||||
rewrite=true
|
||||
else
|
||||
rm -f "$reportfile" 2>/dev/null || true
|
||||
rewrite=false
|
||||
fi
|
||||
scribe_config || exit 0
|
||||
|
||||
facts=$(scribe_turn_facts "$transcript")
|
||||
[ "$(scribe_turn_fact "$facts" bounded)" = "1" ] || exit 0
|
||||
reply=$(scribe_turn_fact "$facts" reply | scribe_json_unescape)
|
||||
# The reply may not be in the transcript yet when the hook fires: an empty
|
||||
# reply is "cannot tell", never "missing everything".
|
||||
[ -n "$(printf '%s' "$reply" | tr -d '[:space:]')" ] || exit 0
|
||||
# How many tasks this turn closed, and which — successful closes only, this
|
||||
# session only (scribe_turn.awk). A task created already done closes with no
|
||||
# id, so the count is what says a check is due. Digits and commas only, so
|
||||
# both drop straight into the JSON.
|
||||
closed=$(scribe_turn_fact "$facts" closed | tr -cd '0-9')
|
||||
closed=${closed:-0}
|
||||
task_ids=$(scribe_turn_fact "$facts" task_ids | tr -cd '0-9,' | sed 's/,,*/,/g; s/^,//; s/,$//')
|
||||
[ "$rewrite" = "true" ] && [ "$closed" = "0" ] && exit 0
|
||||
|
||||
# Bounded before encoding. The server reads the head and the tail — the part
|
||||
# of a report that asks something of the reader is at its end — so a very
|
||||
@@ -51,12 +88,10 @@ if [ "${#reply}" -gt 12000 ]; then
|
||||
reply="${reply:0:4000} … ${reply: -8000}"
|
||||
fi
|
||||
reply_esc=$(printf '%s' "$reply" | scribe_json_escape) || exit 0
|
||||
|
||||
state_dir="${TMPDIR:-/tmp}/scribe-priorart"
|
||||
mkdir -p "$state_dir" 2>/dev/null || true
|
||||
safe_sid=$(printf '%s' "$session_id" | tr -c 'A-Za-z0-9._-' '_')
|
||||
rulefile="$state_dir/${safe_sid}.rules.ids"
|
||||
stopfile="$state_dir/${safe_sid}.checkpoint.ids"
|
||||
closing=""
|
||||
if [ "$closed" != "0" ]; then
|
||||
closing=$(printf ',"closed":%d,"closed_task_ids":[%s],"rewrite":%s' "$closed" "$task_ids" "$rewrite")
|
||||
fi
|
||||
|
||||
query=""
|
||||
scope=$(scribe_scope_query "${event_cwd:-${CLAUDE_PROJECT_DIR:-$PWD}}")
|
||||
@@ -70,15 +105,17 @@ if [ -f "$stopfile" ]; then
|
||||
fi
|
||||
query=${query#&}
|
||||
|
||||
answer=$(printf '{"reply":"%s"}' "$reply_esc" | curl -fsS --max-time 6 \
|
||||
answer=$(printf '{"reply":"%s"%s}' "$reply_esc" "$closing" | curl -fsS --max-time 6 \
|
||||
-H "Authorization: Bearer ${token}" \
|
||||
-H "Content-Type: application/json" \
|
||||
--data-binary @- \
|
||||
"${url%/}/api/plugin/reply-rules${query:+?$query}" 2>/dev/null) || exit 0
|
||||
|
||||
[ "$rewrite" = "true" ] && exit 0
|
||||
answer_flat=$(printf '%s' "$answer" | scribe_json_flat)
|
||||
reason=$(scribe_json_pick "$answer_flat" '.reason')
|
||||
[ -n "$reason" ] || exit 0
|
||||
[ "$(scribe_json_pick "$answer_flat" '.report_check')" = "blocked" ] && : > "$reportfile" 2>/dev/null
|
||||
|
||||
# Recorded BEFORE the block is emitted, for the act checkpoint's reason: a
|
||||
# hold that is shown and not recorded is one that can be shown again.
|
||||
|
||||
@@ -1,155 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Scribe plugin — Stop hook: a reply that closes a task carries the completion
|
||||
# sections (milestone 409 step 5).
|
||||
#
|
||||
# Everything else the plugin does happens BEFORE the agent writes: context,
|
||||
# retrieval, the reporting-back skill. This is the one moment the finished
|
||||
# reply exists, so it is both the last chance to fix a report the operator
|
||||
# cannot read and the only place adherence to the shape can be measured.
|
||||
#
|
||||
# DETERMINISTIC, NO MODEL CALL. Three questions, cheapest first:
|
||||
#
|
||||
# 1. Did this turn close a Scribe task? An `update_task` / `create_task` tool
|
||||
# call with status "done" since the turn's prompt, whose result was not an
|
||||
# error. Most turns stop here, silently.
|
||||
# 2. Does the reply that ends the turn have the completion sections? Loosely:
|
||||
# where the work sits (a record named by id and title, or step N of M),
|
||||
# what needs the operator, and what comes next. Matched on the words that
|
||||
# carry the meaning rather than exact headings, so the skill's wording can
|
||||
# change without breaking this.
|
||||
# 3. If sections are missing, block once with a reason naming them. The agent
|
||||
# rewrites; the rewrite is checked and recorded, and never blocked again.
|
||||
#
|
||||
# MEASURED FROM THE FIRST CALL. Each checked reply is reported to the instance
|
||||
# (`/api/plugin/report-check`): passed, blocked, and after a rewrite either
|
||||
# passed_after_rewrite or missing_after_rewrite. Turns that closed no task are
|
||||
# not reported — they would cost a request on every turn and add nothing to
|
||||
# the rate step 6 reads (blocked among checked replies).
|
||||
#
|
||||
# IT BLOCKS ONLY WHEN THE BLOCK IS RECORDED, AND ONLY IN THE SERVER'S WORDS.
|
||||
# The report goes out first; the instance answers a recorded `blocked` with
|
||||
# the reason to send the agent back with, and the hook blocks only on that
|
||||
# reason. An unconfigured or unreachable instance therefore never stops a
|
||||
# session, every intervention is one the numbers can see, and the guidance
|
||||
# text lives on the server (plugin/PACKAGING.md: hooks carry timing and
|
||||
# transport).
|
||||
#
|
||||
# THE TRANSCRIPT FORMAT IS OBSERVED, NOT DOCUMENTED. Claude Code documents
|
||||
# `transcript_path` and `stop_hook_active` for Stop, not the JSONL inside. As
|
||||
# read from real transcripts (2026-09-14): one content block per line;
|
||||
# `type: "assistant"` lines carry `message.content[]` blocks of `text` /
|
||||
# `tool_use` ({id, name, input}); tool results arrive as `type: "user"` lines
|
||||
# whose content is a `tool_result` array ({tool_use_id, is_error}); a turn's
|
||||
# prompt — typed, or a background-task notification — is a `user` line whose
|
||||
# content is a plain string and which is not `isMeta`. Anything that does not
|
||||
# parse that way makes the hook stay out of the way rather than guess.
|
||||
#
|
||||
# Config (same as the other hooks):
|
||||
# CLAUDE_PLUGIN_OPTION_API_ENDPOINT base URL, no trailing slash
|
||||
# CLAUDE_PLUGIN_OPTION_API_TOKEN fmcp_ API key (sensitive)
|
||||
# SCRIBE_URL / SCRIBE_TOKEN override for the settings.json dogfooding path.
|
||||
set -uo pipefail
|
||||
|
||||
command -v curl >/dev/null 2>&1 || exit 0
|
||||
|
||||
# shellcheck source=plugin/hooks/scribe_defs.sh
|
||||
. "$(dirname "${BASH_SOURCE[0]}")/scribe_defs.sh"
|
||||
|
||||
# Stop delivers { session_id, transcript_path, cwd, hook_event_name, stop_hook_active }.
|
||||
event=$(cat 2>/dev/null || true)
|
||||
event_flat=$(printf '%s' "$event" | scribe_json_flat)
|
||||
transcript=$(scribe_json_pick "$event_flat" '.transcript_path')
|
||||
[ -n "$transcript" ] && [ -f "$transcript" ] || exit 0
|
||||
session_id=$(scribe_json_pick "$event_flat" '.session_id')
|
||||
active=$(scribe_json_pick "$event_flat" '.stop_hook_active')
|
||||
event_cwd=$(scribe_json_pick "$event_flat" '.cwd')
|
||||
|
||||
safe_sid=$(printf '%s' "${session_id:-nosession}" | tr -c 'A-Za-z0-9._-' '_')
|
||||
state_dir="${TMPDIR:-/tmp}/scribe-reportcheck"
|
||||
mkdir -p "$state_dir" 2>/dev/null || true
|
||||
marker="$state_dir/${safe_sid}.blocked"
|
||||
|
||||
# Cheap prefilter: no task tool anywhere in the recent transcript → nothing to
|
||||
# check. Keeps the ordinary turn at one grep. Process substitution, NOT a pipe:
|
||||
# under `pipefail`, `grep -q` exiting on the first match kills `tail` with
|
||||
# SIGPIPE, and the pipeline then reports failure precisely when it matched.
|
||||
grep -q -E '"name":[[:space:]]*"([^"]*__)?(update|create)_task"' < <(tail -c 2000000 "$transcript" 2>/dev/null) || {
|
||||
rm -f "$marker" 2>/dev/null || true
|
||||
exit 0
|
||||
}
|
||||
|
||||
# The turn, parsed once — scribe_turn_facts, shared with the reply check.
|
||||
facts=$(scribe_turn_facts "$transcript")
|
||||
fact() { scribe_turn_fact "$facts" "$1"; }
|
||||
|
||||
[ "$(fact bounded)" = "1" ] || exit 0
|
||||
closed=$(fact closed)
|
||||
if [ "${closed:-0}" = "0" ]; then
|
||||
rm -f "$marker" 2>/dev/null || true
|
||||
exit 0
|
||||
fi
|
||||
# Escaped on one line coming out of awk, so the format survives a multi-line
|
||||
# reply; decoded here, once, where it is about to be read as text.
|
||||
reply=$(fact reply | scribe_json_unescape)
|
||||
task_ids=$(fact task_ids)
|
||||
|
||||
# The reply may not be written to the transcript yet when the hook fires. An
|
||||
# empty reply is "cannot tell", not "missing everything" — stay out of the way.
|
||||
[ -n "$(printf '%s' "$reply" | tr -d '[:space:]')" ] || exit 0
|
||||
|
||||
missing=()
|
||||
# Where the work sits: a record named by id AND title (#12 "…", milestone 3 "…"),
|
||||
# or a step position. A bare id is exactly the homework this shape removes.
|
||||
# shellcheck disable=SC2016 # backticks here are literal markdown, not an expansion
|
||||
grep -q -i -E '(#[0-9]+|milestone [0-9]+|task [0-9]+)[*_`]*[[:space:]]*[*_`]*["“]|step [0-9]+ of [0-9]+' <<< "$reply" \
|
||||
|| missing+=("where it sits")
|
||||
# What needs the operator — "needs you: nothing" counts; it is an answer.
|
||||
grep -q -i -E 'needs? (from )?you|nothing (is )?needed from you|your (call|decision)' <<< "$reply" \
|
||||
|| missing+=("needs you")
|
||||
# What comes next.
|
||||
grep -q -i -E '\bnext\b' <<< "$reply" \
|
||||
|| missing+=("next")
|
||||
|
||||
# Reports the outcome; prints the instance's reply and returns 0 only if the
|
||||
# instance recorded it.
|
||||
report() {
|
||||
scribe_config || return 1
|
||||
local q repo enc m
|
||||
q="outcome=$1&task_ids=${task_ids}"
|
||||
m=$(IFS=,; printf '%s' "${missing[*]:-}")
|
||||
if [ -n "$m" ]; then
|
||||
enc=$(printf '%s' "$m" | scribe_urlenc) || enc=""
|
||||
q="${q}&missing=${enc}"
|
||||
fi
|
||||
scope=$(scribe_scope_query "${event_cwd:-${CLAUDE_PROJECT_DIR:-$PWD}}")
|
||||
[ -n "$scope" ] && q="${q}&${scope}"
|
||||
curl -fsS --max-time 4 \
|
||||
-H "Authorization: Bearer ${token}" \
|
||||
"${url%/}/api/plugin/report-check?${q}" 2>/dev/null
|
||||
}
|
||||
|
||||
if [ "$active" = "true" ]; then
|
||||
# A Stop hook already blocked this stop. If it was this one, the reply is
|
||||
# the rewrite: record how it came out, and let the session stop whatever
|
||||
# the answer. If it was another plugin's block, this hook has nothing to add.
|
||||
[ -f "$marker" ] || exit 0
|
||||
rm -f "$marker" 2>/dev/null || true
|
||||
if [ ${#missing[@]} -eq 0 ]; then report passed_after_rewrite >/dev/null; else report missing_after_rewrite >/dev/null; fi
|
||||
exit 0
|
||||
fi
|
||||
rm -f "$marker" 2>/dev/null || true
|
||||
|
||||
if [ ${#missing[@]} -eq 0 ]; then
|
||||
report passed >/dev/null
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# The words the agent is sent back with are the server's (plugin/PACKAGING.md:
|
||||
# a hook carries timing and transport). No reason back → nothing recorded →
|
||||
# no block.
|
||||
answer=$(report blocked) || exit 0
|
||||
reason=$(scribe_json_pick "$(printf '%s' "$answer" | scribe_json_flat)" '.reason')
|
||||
[ -n "$reason" ] || exit 0
|
||||
: > "$marker" 2>/dev/null || true
|
||||
printf '{"decision":"block","reason":"%s"}\n' "$(printf '%s' "$reason" | scribe_json_escape)"
|
||||
exit 0
|
||||
@@ -10,7 +10,7 @@
|
||||
# words: the agent that built the code is the one participant who knows what
|
||||
# it is, and the end of the turn is the last moment that is still true.
|
||||
#
|
||||
# THE SAME DISCIPLINE AS scribe_report_check.sh, deliberately:
|
||||
# THE SAME DISCIPLINE AS the completion-section check in scribe_reply_check.sh:
|
||||
# - it blocks only on a `reason` the instance returned, which it returns
|
||||
# only for a block it RECORDED — an unconfigured or unreachable instance
|
||||
# never stops a session, and every intervention is one the numbers see;
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
#
|
||||
# Reads the flat `IDX<TAB>PATH<TAB>VALUE` stream that scribe_json.awk produces
|
||||
# in `mode=lines` from a Claude Code transcript, and answers the four questions
|
||||
# scribe_report_check.sh asks. It replaces a thirty-line jq program; the shape
|
||||
# the Stop hooks ask (scribe_reply_check.sh). It replaces a thirty-line jq program; the shape
|
||||
# of the answer is unchanged, so the hook around it reads the same.
|
||||
#
|
||||
# bounded 1 if a user prompt was found in the window, else empty. A window
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
name: reporting-back
|
||||
description: Use when you are about to write the reply the operator will read — work finished, a task marked done, stopping on a blocker, asking them to decide or to do something, answering "where are we" / "what's next", or proposing an approach. Shapes the reply around where the work stands (which task, what changed, what needs them, what comes next) instead of the order you did things in. Triggers on reporting completion, handing off, asking a question, or summarising progress.
|
||||
description: Use when you are about to write the reply the operator will read and the delivered reply shape is not enough — work finished, stopping on a blocker, asking them to decide or to do something, answering "where are we", or proposing an approach. The long-form reference behind the reply shapes Scribe delivers: why sections are chosen rather than filled, who decides what, taking placement from the record, and a completion report worked in full. Triggers on reporting completion, handing off, asking a question, or summarising progress.
|
||||
metadata:
|
||||
moments: reply.report
|
||||
---
|
||||
@@ -12,7 +12,23 @@ while you worked: they don't hold the files you read, the names you used or
|
||||
the order you did things in. A reply that follows *your* path is accurate and
|
||||
still unreadable to them. Shape it around **where the work stands**.
|
||||
|
||||
Pick the kind of reply first (the tables below). Its sections are **what to
|
||||
## The shapes arrive on their own
|
||||
|
||||
Scribe delivers the default shape for each kind of reply as product, at the
|
||||
moment it is due: the core ("Reply shape · Every reply") with each turn, the
|
||||
completion report when a task closes, the asks when you put a question to the
|
||||
operator, the plan when you plan. A shape this session has already seen comes
|
||||
back as its one-line reminder, and `list_reply_shapes` has every one in full.
|
||||
The operator's own preferences are mounted on the same moments and arrive
|
||||
beside the shape; where one differs, the preference is what they asked for.
|
||||
|
||||
This skill does not restate those shapes. It holds the reasoning behind them
|
||||
and the completion report worked in full — read it when a shape's lines are
|
||||
not enough to decide what a reply should say.
|
||||
|
||||
## Sections are chosen, not filled
|
||||
|
||||
Pick the kind of reply first (the delivered shape). Its sections are **what to
|
||||
consider including, not a form to complete.**
|
||||
|
||||
Two kinds of section, and they behave differently:
|
||||
@@ -32,32 +48,15 @@ lines naming the two things that changed their position. Extra length has to be
|
||||
earned: a comparison they asked for, options that need laying side by side, a
|
||||
measurement whose numbers are the point.
|
||||
|
||||
## Every reply
|
||||
## A decision already made
|
||||
|
||||
- **Conclusion first.** The verdict, the result, or the question — before the
|
||||
reasoning that supports it.
|
||||
- **One topic per section.** Two things the operator raised get two sections.
|
||||
- **Make priority visible.** Bold the few things that matter; let the rest be
|
||||
plain.
|
||||
- **End with the ask, in bold** — the one thing they need to decide or do. If
|
||||
there is nothing, say so.
|
||||
- **A request for approval gets its own section, headed "Approval requested".**
|
||||
Whenever you are holding an action until the operator says yes — a permission
|
||||
prompt their client raised, something hard to undo, a change you have
|
||||
prepared and not made — put it there, near the top, whatever kind of reply
|
||||
this is. Inside a progress or completion list it reads as a status line, and
|
||||
the operator does not see that you are waiting on them.
|
||||
- **Plain words.** Use the operator's vocabulary, not the names you coined while
|
||||
working. If a term has to appear, explain it once.
|
||||
- **Place the work in Scribe.** Name the task, issue or milestone it belongs to,
|
||||
by id *and* title (using-scribe: "Name the record, never just its number").
|
||||
- **A decision already made gets acted on, and the reply says what you did with
|
||||
it.** Once the operator has chosen, that is the input to the work, not a topic
|
||||
to revisit. If something you have since learned genuinely overturns the
|
||||
choice, say so once and plainly — name the new evidence and what it changes —
|
||||
and otherwise let the decision stand. Laying out the trade-offs of a settled
|
||||
question again reads as contradicting yourself rather than as being careful,
|
||||
and it costs the operator the decision twice.
|
||||
**A decision already made gets acted on, and the reply says what you did with
|
||||
it.** Once the operator has chosen, that is the input to the work, not a topic
|
||||
to revisit. If something you have since learned genuinely overturns the
|
||||
choice, say so once and plainly — name the new evidence and what it changes —
|
||||
and otherwise let the decision stand. Laying out the trade-offs of a settled
|
||||
question again reads as contradicting yourself rather than as being careful,
|
||||
and it costs the operator the decision twice.
|
||||
|
||||
## Take the placement from the record
|
||||
|
||||
@@ -80,39 +79,14 @@ it is wrong. So take placement from Scribe:
|
||||
- Work with no task behind it: say so plainly — "this wasn't tracked as a
|
||||
task" — and offer to record it. An honest "untracked" is a placement too.
|
||||
|
||||
## The operator's own shapes come first
|
||||
|
||||
The shapes below are defaults. An operator may have changed some of them — a
|
||||
section they always want, an order they read faster, a kind of reply they want
|
||||
shorter — and those changes are `preference` records. Where a preference and a
|
||||
default differ, the preference is what they asked for.
|
||||
|
||||
- **A completion report brings its preferences with it.** Closing a task with
|
||||
`update_task` returns them as **`reply_preferences`** when the operator has
|
||||
any; the `report_back` line says so. Nothing to search for.
|
||||
- **Every other reply, ask before writing it.** A finding, a decision, a
|
||||
handoff, a "where are we" — no tool call comes before these, so nothing
|
||||
hands their preferences over. Once you know which kind of reply you are
|
||||
writing, `search(content_type="rule")` for it in the words of that moment —
|
||||
"writing a decision for the operator", "handing off to the operator" — and
|
||||
follow any preference that comes back. Nothing coming back means the default
|
||||
shape stands.
|
||||
|
||||
## Reports — work happened
|
||||
|
||||
| Kind | Sections |
|
||||
|---|---|
|
||||
| **Completion** | Where this sits · What now works · How / why · Needs you · Next |
|
||||
| **Finding** (a problem you found) | Symptom · Cause · **What you decided and did** — judge it and act; an offer to fix it is the judgment not made (see *You are the judge*) |
|
||||
| **Blocked / failed** | What stopped · What you tried · What you need from them |
|
||||
| **Progress** (mid-work) | One or two lines: where things are, what's next, any blocker |
|
||||
| **Where are we** | The milestone and its progress · Done · Open · Needs you · Next |
|
||||
|
||||
## You are the judge
|
||||
|
||||
You are the judge of record for the work itself: what a shape is, whether a
|
||||
finding holds, whether a record is right, whether something is done. The
|
||||
operator reads the decision — they do not make it.
|
||||
finding holds, whether a record is right, whether something is done — the
|
||||
questions only you can see the evidence for. Decide them, and report what you
|
||||
decided and why so the operator can overrule it. **Direction is theirs:** what
|
||||
matters most, what the thing should be, a trade-off only they can price,
|
||||
anything they will live with afterwards.
|
||||
|
||||
**A finding surfaced and not judged is a finding dropped, not deferred.** This
|
||||
is the failure that hides inside a good report: the symptom named, the cause
|
||||
@@ -125,11 +99,12 @@ So: decide, act, and report what you decided and why. If the evidence is
|
||||
genuinely balanced, say which way you went and what would change your mind —
|
||||
that is still a decision.
|
||||
|
||||
**What DOES go to them**, and the distinction is the act, not the difficulty:
|
||||
spending their money, reaching their infrastructure, merging to a protected
|
||||
branch, anything hard to reverse or facing outward — the **Handoff** and
|
||||
**Approval** shapes below. Judging a record is never one of these. A hard call
|
||||
is still yours; an irreversible act is still theirs.
|
||||
**What DOES go to them** is direction, and every act that is hard to reverse
|
||||
or faces outward: spending their money, reaching their infrastructure, merging
|
||||
to a protected branch — the **Decision**, **Handoff** and **Approval** kinds of
|
||||
the asks shape. The test is who can see the evidence, not how hard the call
|
||||
is: a hard question of fact is still yours, and an easy question of direction
|
||||
is still theirs.
|
||||
|
||||
**Judging is attended, not automatic.** You judge by reading the evidence and
|
||||
recording why. A threshold, a sweep or a rule that reclassifies in bulk with
|
||||
@@ -142,15 +117,11 @@ writes with another unattended write, stop.
|
||||
evidence needed to decide and a way to record the decision under their own
|
||||
name.
|
||||
|
||||
## Asks — the operator needs to act or decide
|
||||
## Asking them
|
||||
|
||||
| Kind | Sections |
|
||||
|---|---|
|
||||
| **Decision** | The question first · 2–4 options, each with what it changes · recommendation first |
|
||||
| **Clarification** | "My reading is X · the gap is Y · unless you say otherwise I'll do Z" |
|
||||
| **Handoff** (only they can do it) | The action · why it needs them · what it unblocks · what you'll do after · any way to skip it |
|
||||
| **Approval** (you are ready to act and holding for a yes) | **Approval requested:** exactly what happens once they approve, one numbered item per change so they can approve part · why it needs their yes · how it can be undone · what you'll do after |
|
||||
| **Conflict** (what you're about to do clashes with a rule, a plan or an earlier decision) | What it says · what you were about to do · where they clash · A or B? |
|
||||
The asks shape arrives when you put a question to the operator; it lists the
|
||||
kinds — a decision, a clarification, a handoff, an approval you are holding
|
||||
for, a conflict with a rule or an earlier decision. Two things behind it:
|
||||
|
||||
Before asking, check whether you can find the answer yourself — something that
|
||||
can be read or looked up is a fact to check, not a question to send.
|
||||
@@ -164,21 +135,6 @@ between options is not a decision about the assumptions they share, and an
|
||||
unlabelled one gets approved as if it had been asked. An assumption that
|
||||
contradicts a recorded ruling is a **Conflict**, not an option's fine print.
|
||||
|
||||
## Answers — the operator asked something
|
||||
|
||||
| Kind | Sections |
|
||||
|---|---|
|
||||
| **Explanation** | The answer first · then the evidence, pointing at what they could open to check it |
|
||||
| **Evaluation** ("can we / should we") | Verdict · What exists · The gaps · Recommendation |
|
||||
|
||||
## Proposals — shaping future work
|
||||
|
||||
| Kind | Sections |
|
||||
|---|---|
|
||||
| **Options** | 2–3 approaches · the trade-off of each · one recommendation |
|
||||
| **Plan** | Goal · Steps · Open questions — for review before starting (writing-plans) |
|
||||
| **Review** | Findings ranked by how much they matter, one per item |
|
||||
|
||||
## The completion report, in full
|
||||
|
||||
The most common reply, and the one most often written in the order the work
|
||||
@@ -235,7 +191,7 @@ happens next** without asking a follow-up? If not, the sections are what's
|
||||
missing — not more detail.
|
||||
|
||||
Then read it once more for **what can go**. A section filled because it was in
|
||||
the table, reasoning supporting a conclusion nobody is going to dispute, a
|
||||
the shape, reasoning supporting a conclusion nobody is going to dispute, a
|
||||
finding already written to the record — none of it changes what the operator
|
||||
does, so none of it belongs in the reply. Cutting is not hiding: the log holds
|
||||
it, and the reply stays readable. A reply that has been cut twice is the one
|
||||
|
||||
@@ -263,9 +263,11 @@ Two constraints on *how* that's achieved:
|
||||
order you did things in: which task or milestone it belongs to, what now
|
||||
works, what needs them, and what comes next. Take the placement from the
|
||||
`placement` block that `create_task` / `update_task` return — the milestone,
|
||||
step N of M, the next open step — rather than from memory. The
|
||||
`reporting-back` skill holds the shape for each kind of reply: completions,
|
||||
findings, decisions, handoffs, "where are we".
|
||||
step N of M, the next open step — rather than from memory. Scribe delivers
|
||||
the shape for each kind of reply when it is due — the core with each turn,
|
||||
the completion report when a task closes, the asks when you put a question —
|
||||
and `list_reply_shapes` has them all; the `reporting-back` skill holds the
|
||||
reasoning behind them and a completion report worked in full.
|
||||
|
||||
## Stay inside the active project's scope
|
||||
|
||||
|
||||
@@ -0,0 +1,272 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Measure near-verbatim duplication in a tree, and what a change added to it.
|
||||
|
||||
WHY THIS EXISTS
|
||||
|
||||
A DRY audit made of several passes ends with a question no single pass can
|
||||
answer: did the audit as a whole reduce duplication, and did it create any? The
|
||||
second half is the one that matters. A pass that shortens two statements can
|
||||
leave them byte-identical to a third, and the new copy is invisible to that
|
||||
pass's own scan because it did not exist when the scan ran (#4745: a retro over
|
||||
a 17-pass audit found exactly this, and folded it into a view). This script is
|
||||
the close-out measure the DRY Pass process names, so the measure is a tool and
|
||||
not something each audit improvises.
|
||||
|
||||
WHAT IT MEASURES
|
||||
|
||||
Each file is reduced to its significant lines: blank lines, comment-only lines
|
||||
and lines of bare punctuation (`}`, `});`, `],`) are dropped; whitespace is
|
||||
collapsed; string literals are folded to "S", so two copies that differ only in
|
||||
a message or a key still match. A *window* is N consecutive significant lines
|
||||
(default 6). A window whose text occurs at two or more places in the same group
|
||||
is duplicated, and every line it covers is a duplicated line.
|
||||
|
||||
Per group it reports significant lines, duplicated windows (distinct texts),
|
||||
duplicated lines and their share. With `--before REV` it measures that revision
|
||||
too and lists the duplicated windows that exist ONLY after — not duplicated
|
||||
before, either because the text was absent or because it occurred once. Those
|
||||
are grouped by the set of files they appear in, which is the list to read.
|
||||
|
||||
WHAT IT DOES NOT SEE
|
||||
|
||||
Structural or semantic copies — two components with the same shape and
|
||||
different names, the same predicate written two ways. It catches near-verbatim
|
||||
text only, so a fold of structural copies shows here less than it counts. Read
|
||||
the numbers as a floor and the after-only list as the finding.
|
||||
|
||||
USAGE
|
||||
|
||||
measure_duplication.py [--repo DIR] [--before REV] [--after REV]
|
||||
[--group NAME=GLOB[,GLOB...]]... [--exclude GLOB]...
|
||||
[--window N] [--show N] [--json]
|
||||
|
||||
The after tree is the working tree's tracked files unless `--after REV` names a
|
||||
revision. Revisions are read through `git archive`, so nothing is checked out.
|
||||
Globs are fnmatch patterns over repo-relative paths, where `*` crosses `/`.
|
||||
Groups are tried in order and the first match claims a file, so list
|
||||
`--group 'go tests=*_test.go'` before `--group 'go=*.go'`. With no `--group`,
|
||||
files are grouped by extension among common source types.
|
||||
|
||||
Stdlib only, so it runs from a scratch copy in any project:
|
||||
`get_snippet` the recorded snippet, write its code to a file, run it there.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import fnmatch
|
||||
import hashlib
|
||||
import io
|
||||
import json
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import tarfile
|
||||
from collections import defaultdict
|
||||
from pathlib import Path
|
||||
|
||||
DEFAULT_EXTS = {
|
||||
"c", "cc", "cpp", "cs", "css", "dart", "go", "h", "hpp", "java", "js",
|
||||
"jsx", "kt", "kts", "php", "py", "rb", "rs", "scss", "sh", "sql",
|
||||
"svelte", "swift", "ts", "tsx", "vue",
|
||||
}
|
||||
|
||||
_COMMENT = re.compile(r"^(//|/\*|\*|--|<!--|#(\s|!|$))")
|
||||
_PUNCT_ONLY = re.compile(r"^[\s{}()\[\];,]*$")
|
||||
_STRING = re.compile(r'"(?:\\.|[^"\\])*"|\'(?:\\.|[^\'\\])*\'|`(?:\\.|[^`\\])*`')
|
||||
_SPACE = re.compile(r"\s+")
|
||||
|
||||
|
||||
def significant_lines(text: str) -> list[tuple[int, str]]:
|
||||
"""(1-based line number, normalized text) for each line that carries code."""
|
||||
out = []
|
||||
for number, raw in enumerate(text.splitlines(), 1):
|
||||
line = raw.strip()
|
||||
if not line or _COMMENT.match(line) or _PUNCT_ONLY.match(line):
|
||||
continue
|
||||
out.append((number, _SPACE.sub(" ", _STRING.sub('"S"', line))))
|
||||
return out
|
||||
|
||||
|
||||
def _git(repo: Path, *args: str) -> bytes:
|
||||
return subprocess.run(
|
||||
["git", "-C", str(repo), *args], check=True, capture_output=True,
|
||||
).stdout
|
||||
|
||||
|
||||
def read_working_tree(repo: Path):
|
||||
"""Tracked files as they are on disk — what the next commit would hold."""
|
||||
for rel in _git(repo, "ls-files", "-z").decode().split("\0"):
|
||||
path = repo / rel
|
||||
if rel and path.is_file():
|
||||
yield rel, path.read_bytes()
|
||||
|
||||
|
||||
def read_revision(repo: Path, rev: str):
|
||||
"""Every file at `rev`, streamed out of `git archive` without a checkout."""
|
||||
stream = io.BytesIO(_git(repo, "archive", "--format=tar", rev))
|
||||
with tarfile.open(fileobj=stream, mode="r:") as archive:
|
||||
for member in archive:
|
||||
if member.isfile():
|
||||
yield member.name, archive.extractfile(member).read()
|
||||
|
||||
|
||||
def group_of(rel: str, groups: list[tuple[str, list[str]]]) -> str | None:
|
||||
if not groups:
|
||||
ext = rel.rsplit(".", 1)[-1] if "." in Path(rel).name else ""
|
||||
return ext if ext in DEFAULT_EXTS else None
|
||||
for name, patterns in groups:
|
||||
if any(fnmatch.fnmatch(rel, p) for p in patterns):
|
||||
return name
|
||||
return None
|
||||
|
||||
|
||||
def measure(files, groups, excludes, window: int) -> dict:
|
||||
"""Per group: line counts, and every duplicated window with where it occurs."""
|
||||
by_group: dict[str, dict] = defaultdict(
|
||||
lambda: {"files": {}, "index": defaultdict(list)},
|
||||
)
|
||||
for rel, data in files:
|
||||
if any(fnmatch.fnmatch(rel, p) for p in excludes):
|
||||
continue
|
||||
name = group_of(rel, groups)
|
||||
if name is None:
|
||||
continue
|
||||
try:
|
||||
text = data.decode("utf-8")
|
||||
except UnicodeDecodeError:
|
||||
continue
|
||||
lines = significant_lines(text)
|
||||
g = by_group[name]
|
||||
g["files"][rel] = lines
|
||||
for i in range(len(lines) - window + 1):
|
||||
body = "\n".join(t for _, t in lines[i:i + window])
|
||||
key = hashlib.sha1(body.encode()).hexdigest()
|
||||
g["index"][key].append((rel, i))
|
||||
|
||||
result = {}
|
||||
for name, g in sorted(by_group.items()):
|
||||
dup = {k: occ for k, occ in g["index"].items() if len(occ) > 1}
|
||||
covered = set()
|
||||
for occ in dup.values():
|
||||
for rel, i in occ:
|
||||
covered.update((rel, j) for j in range(i, i + window))
|
||||
total = sum(len(lines) for lines in g["files"].values())
|
||||
result[name] = {
|
||||
"lines": total,
|
||||
"dup_windows": len(dup),
|
||||
"dup_lines": len(covered),
|
||||
"share": len(covered) / total if total else 0.0,
|
||||
"_dup": dup,
|
||||
"_files": g["files"],
|
||||
}
|
||||
return result
|
||||
|
||||
|
||||
def only_after(before: dict, after: dict) -> dict[str, list[dict]]:
|
||||
"""Duplicated windows in `after` that were not duplicated in `before`,
|
||||
grouped by the set of files they occur in."""
|
||||
out = {}
|
||||
for name, a in after.items():
|
||||
was = before.get(name, {}).get("_dup", {})
|
||||
sets: dict[tuple, dict] = {}
|
||||
for key, occ in a["_dup"].items():
|
||||
if key in was:
|
||||
continue
|
||||
fileset = tuple(sorted({rel for rel, _ in occ}))
|
||||
entry = sets.setdefault(fileset, {"files": list(fileset), "windows": 0, "at": []})
|
||||
entry["windows"] += 1
|
||||
if not entry["at"]:
|
||||
entry["at"] = [
|
||||
f"{rel}:{a['_files'][rel][i][0]}" for rel, i in sorted(occ)
|
||||
]
|
||||
if sets:
|
||||
out[name] = sorted(sets.values(), key=lambda e: -e["windows"])
|
||||
return out
|
||||
|
||||
|
||||
def _public(result: dict) -> dict:
|
||||
return {
|
||||
name: {k: v for k, v in g.items() if not k.startswith("_")}
|
||||
for name, g in result.items()
|
||||
}
|
||||
|
||||
|
||||
def _pct(share: float) -> str:
|
||||
return f"{share * 100:.1f}%"
|
||||
|
||||
|
||||
def report(before: dict | None, after: dict, fresh: dict, show: int) -> str:
|
||||
rows = []
|
||||
for name in sorted(set(after) | set(before or {})):
|
||||
a = after.get(name, {"lines": 0, "dup_windows": 0, "dup_lines": 0, "share": 0.0})
|
||||
if before is None:
|
||||
rows.append((name, f"{a['lines']:,}", str(a["dup_windows"]),
|
||||
str(a["dup_lines"]), _pct(a["share"])))
|
||||
continue
|
||||
b = before.get(name, {"lines": 0, "dup_windows": 0, "dup_lines": 0, "share": 0.0})
|
||||
rows.append((
|
||||
name,
|
||||
f"{b['lines']:,} → {a['lines']:,}",
|
||||
f"{b['dup_windows']} → {a['dup_windows']}",
|
||||
f"{b['dup_lines']} → {a['dup_lines']}",
|
||||
f"{_pct(b['share'])} → {_pct(a['share'])}",
|
||||
))
|
||||
header = ("group", "lines", "dup windows", "dup lines", "share")
|
||||
widths = [max(len(r[c]) for r in [header, *rows]) for c in range(len(header))]
|
||||
lines = [" ".join(cell.ljust(w) for cell, w in zip(r, widths)).rstrip()
|
||||
for r in [header, *rows]]
|
||||
if before is not None:
|
||||
total = sum(len(v) for v in fresh.values())
|
||||
lines += ["", f"Duplicated only after: {total} file set(s). Read each one —"
|
||||
" a copy a pass's own shortening exposed, or a deliberate repeat."]
|
||||
for name, sets in fresh.items():
|
||||
for entry in sets[:show]:
|
||||
lines.append(f" [{name}] {entry['windows']} window(s): "
|
||||
+ ", ".join(entry["at"]))
|
||||
if len(sets) > show:
|
||||
lines.append(f" [{name}] … {len(sets) - show} more (--show)")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def _parse_group(spec: str) -> tuple[str, list[str]]:
|
||||
name, sep, globs = spec.partition("=")
|
||||
if not sep or not name or not globs:
|
||||
raise argparse.ArgumentTypeError(f"--group wants NAME=GLOB[,GLOB...], got {spec!r}")
|
||||
return name, [g for g in globs.split(",") if g]
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
|
||||
parser.add_argument("--repo", type=Path, default=Path("."))
|
||||
parser.add_argument("--before", help="revision to compare against (e.g. the sha before the first pass)")
|
||||
parser.add_argument("--after", help="revision to measure (default: the working tree's tracked files)")
|
||||
parser.add_argument("--group", action="append", type=_parse_group, default=[])
|
||||
parser.add_argument("--exclude", action="append", default=[])
|
||||
parser.add_argument("--window", type=int, default=6)
|
||||
parser.add_argument("--show", type=int, default=20)
|
||||
parser.add_argument("--json", action="store_true")
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
repo = args.repo.resolve()
|
||||
after_files = read_revision(repo, args.after) if args.after else read_working_tree(repo)
|
||||
after = measure(after_files, args.group, args.exclude, args.window)
|
||||
before = fresh = None
|
||||
if args.before:
|
||||
before = measure(read_revision(repo, args.before), args.group, args.exclude, args.window)
|
||||
fresh = only_after(before, after)
|
||||
|
||||
if args.json:
|
||||
json.dump({
|
||||
"window": args.window,
|
||||
"before": _public(before) if before is not None else None,
|
||||
"after": _public(after),
|
||||
"only_after": fresh,
|
||||
}, sys.stdout, indent=2)
|
||||
print()
|
||||
else:
|
||||
print(report(before, after, fresh or {}, args.show))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -159,6 +159,9 @@ _READ_ONLY_TOOLS = frozenset({
|
||||
# retrieval_telemetry's reason, and needed by a read key so that a line
|
||||
# naming a moment can be understood by whoever was shown it.
|
||||
"list_moments",
|
||||
# The default reply shapes (milestone 500). A pure read of a constant, and
|
||||
# one a read key needs: a session shown a shape may want the rest.
|
||||
"list_reply_shapes",
|
||||
# The platform catalog and a project's answers (milestone 463). A pure
|
||||
# read; set_project_platforms is the write.
|
||||
"list_platforms",
|
||||
|
||||
@@ -15,6 +15,7 @@ from __future__ import annotations
|
||||
from scribe.mcp._context import current_user_id
|
||||
from scribe.services import moment_actions as actions_svc
|
||||
from scribe.services import moments as moments_svc
|
||||
from scribe.services import reply_shapes as shapes_svc
|
||||
from scribe.services import rule_moment_judgments as judgments_svc
|
||||
|
||||
|
||||
@@ -47,6 +48,26 @@ async def list_moments() -> dict:
|
||||
return out
|
||||
|
||||
|
||||
async def list_reply_shapes() -> dict:
|
||||
"""The default shapes of a reply, and the moment each one arrives at.
|
||||
|
||||
You do not need this to write a reply: the shapes come to you. The core
|
||||
arrives with the turn, and each kind's shape arrives at the moment before
|
||||
that kind of reply is written (closing a task, putting a question, opening
|
||||
a plan). Read this to see them all at once, to answer the operator about
|
||||
what the default is, or before writing a preference that changes one.
|
||||
|
||||
An operator's own adjustment to a shape is a `preference` mounted on that
|
||||
shape's `moment` (create_preference with `moments=[...]`); it arrives
|
||||
beside the default, and where the two differ the preference is what the
|
||||
operator asked for.
|
||||
|
||||
Each shape carries `key`, `title`, `moment` (and what the moment `means`),
|
||||
`delivered` (when it arrives) and `text`.
|
||||
"""
|
||||
return shapes_svc.catalog()
|
||||
|
||||
|
||||
async def map_action(tool: str, moment: str, match: str = "", reason: str = "") -> dict:
|
||||
"""Make an action reach a moment on this install — the in-session fix for a missed moment.
|
||||
|
||||
@@ -213,6 +234,7 @@ async def rule_misfired(rule_id: int, moment: str, why: str,
|
||||
|
||||
def register(mcp) -> None:
|
||||
mcp.tool(name="list_moments")(list_moments)
|
||||
mcp.tool(name="list_reply_shapes")(list_reply_shapes)
|
||||
mcp.tool(name="map_action")(map_action)
|
||||
mcp.tool(name="unmap_action")(unmap_action)
|
||||
mcp.tool(name="rules_to_mount")(rules_to_mount)
|
||||
|
||||
@@ -28,6 +28,7 @@ from scribe.services import milestones as milestones_svc
|
||||
from scribe.services import notes as notes_svc
|
||||
from scribe.services import platforms as platforms_svc
|
||||
from scribe.services import projects as projects_svc
|
||||
from scribe.services import reply_shapes as reply_shapes_svc
|
||||
from scribe.services import rulebooks as rulebooks_svc
|
||||
from scribe.services import systems as systems_svc
|
||||
from scribe.services import trash as trash_svc
|
||||
@@ -73,11 +74,16 @@ async def enter_project(project_id: int) -> dict:
|
||||
project_id: The project to enter.
|
||||
|
||||
Returns a dict with keys: project, milestone_summary, open_tasks, systems,
|
||||
design_system, project_rules, pattern_coverage —
|
||||
design_system, project_rules, pattern_coverage, reply_shape —
|
||||
plus unplanned_milestones, milestone_summary_omitted,
|
||||
unplanned_milestones_omitted, family, inception and systems_bootstrap,
|
||||
each present only when it applies (see below).
|
||||
|
||||
`reply_shape` is the default shape of every reply you write to the
|
||||
operator: conclusion first, the shortest reply that carries the answer,
|
||||
where the work stands, and the one thing they need to do. Write to it;
|
||||
an operator preference about replies that differs from it wins.
|
||||
|
||||
`project` is id, title, status and the full goal. get_project has the
|
||||
whole record.
|
||||
|
||||
@@ -323,6 +329,11 @@ async def enter_project(project_id: int) -> dict:
|
||||
out["family"] = family
|
||||
if inception_ask:
|
||||
out["inception"] = inception_ask
|
||||
# The core reply shape (milestone 500): for a client with no prompt hook,
|
||||
# entering the project is the one point every session passes before it
|
||||
# writes a reply. A plugin session gets it with its first turn as well;
|
||||
# one repeat per session is the price of reaching every client.
|
||||
out["reply_shape"] = reply_shapes_svc.render(reply_shapes_svc.core(), full=True)
|
||||
return out
|
||||
|
||||
|
||||
|
||||
@@ -596,19 +596,7 @@ It is an UPPER BOUND per surface: a pull records the door it came
|
||||
Only ever raised for unbidden arms known to log unconditionally: a
|
||||
search returning a list every time is doing its job, and an arm whose
|
||||
zeros were never written would flag a LOGGING bug while pointing you at
|
||||
a threshold, which is #3497 exactly. Nor for an arm whose query never
|
||||
changes — see the next entry.
|
||||
- `fixed_query_never_clears` — an arm that always searches the SAME query
|
||||
returned nothing on every call. Its score is one constant, so this is
|
||||
not a quiet window: the bar sits above that constant and no amount of
|
||||
further traffic will produce a different result. The arm is off rather
|
||||
than silent, and nothing else here would say so. The same property is
|
||||
why `cannot_decline` is not raised for these arms: with a constant
|
||||
score the decline rate is 0% or 100% by construction, so "never
|
||||
declined" is arithmetic and not evidence about the floor. Read the
|
||||
refused record (`near_miss_samples`) BEFORE moving the dial — the last
|
||||
time an arm sat here, every percentile said lower it and the refused
|
||||
record showed the refusal was right.
|
||||
a threshold, which is #3497 exactly.
|
||||
- `band_hugs_floor` — the weakest tenth of what an arm returns sits on
|
||||
its floor. The bar is doing the selecting and the score is not, so
|
||||
moving that floor changes how MUCH you get, not how good it is.
|
||||
|
||||
@@ -40,7 +40,6 @@ from scribe.services.notes import minted_kind
|
||||
from scribe.services import placement as placement_svc
|
||||
from scribe.services import planning as planning_svc
|
||||
from scribe.services import record_batch as batch_svc
|
||||
from scribe.services import reply_preferences as reply_prefs_svc
|
||||
from scribe.services import rulebooks as rulebooks_svc
|
||||
from scribe.services import systems as systems_svc
|
||||
from scribe.services import task_logs as task_logs_svc
|
||||
@@ -416,14 +415,13 @@ async def update_task(
|
||||
reconstructing them, because a remembered milestone title or "next step"
|
||||
reads exactly like a real one when it is wrong.
|
||||
|
||||
Closing a task (done or cancelled) also returns `report_back`: a one-line
|
||||
reminder of what the reply to the operator should cover. When the
|
||||
operator has preferences for how a completion report is written, they
|
||||
come back as `reply_preferences` ({id, title, statement, kind}) — found
|
||||
by their `when_to_apply`, so a preference whose trigger is writing the
|
||||
report after finishing a task is the one that arrives here. Where one
|
||||
differs from the default shape, the preference is what the operator
|
||||
asked for. When family adoptions were answered `owed` while the task was
|
||||
Closing a task (done or cancelled) also returns `reply_shape`, the
|
||||
default shape of the completion report you are about to write, and
|
||||
`report_back`: a one-line reminder of what that reply should cover. The
|
||||
operator's own preferences for that report arrive beside it in
|
||||
`moment_rules` — the ones mounted on finishing work. Where one differs
|
||||
from the default shape, the preference is what the operator asked for.
|
||||
When family adoptions were answered `owed` while the task was
|
||||
open, they come back as `family_owed` ({idea_id, idea_title, project_id,
|
||||
project_title, owed_task_id}): name each in the report — it is work
|
||||
filed into that project.
|
||||
@@ -469,14 +467,6 @@ async def update_task(
|
||||
await family_svc.attach_family_hint(uid, data, note, created=False)
|
||||
if status in _CLOSING_STATUSES:
|
||||
data["report_back"] = REPORT_BACK_CUE
|
||||
# The operator's own adjustments to the completion report, retrieved
|
||||
# at the one moment a server can see that report coming (milestone
|
||||
# 409 step 4). Omitted rather than sent empty, like every decoration.
|
||||
prefs = await reply_prefs_svc.completion_preferences(
|
||||
uid, project_id=getattr(note, "project_id", None))
|
||||
if prefs:
|
||||
data["reply_preferences"] = prefs
|
||||
data["report_back"] = REPORT_BACK_CUE + " " + REPLY_PREFERENCES_CUE
|
||||
# Owed family adoptions filed while the task was open (milestone 463
|
||||
# step 6) — work now waiting in some project, which the report names.
|
||||
await family_adoption_svc.attach_owed_adoptions(uid, data, note)
|
||||
@@ -530,22 +520,16 @@ async def add_task_log(task_id: int, content: str) -> dict:
|
||||
return data
|
||||
|
||||
|
||||
# The in-band half of milestone 409 step 3. The reporting-back skill and the
|
||||
# static context carry the full shapes, but both live only in the Claude Code
|
||||
# plugin; a tool response reaches every MCP client, at the moment a piece of
|
||||
# work closes, which is exactly when the report is about to be written. One
|
||||
# line on purpose: a template here would be read as the reply itself.
|
||||
# The in-band half of milestone 409 step 3. The completion shape itself rides
|
||||
# the closing response as `reply_shape` (milestone 500); this is the one line
|
||||
# that says what the reply covers, for every MCP client, at the moment a piece
|
||||
# of work closes. One line on purpose: a template here would be read as the
|
||||
# reply itself.
|
||||
_CLOSING_STATUSES = ("done", "cancelled")
|
||||
REPORT_BACK_CUE = (
|
||||
"Reporting this to the operator? Say where it sits (from `placement`), "
|
||||
"what now works, what needs them, and what comes next."
|
||||
)
|
||||
# Appended only when `reply_preferences` is present, so the key never arrives
|
||||
# unexplained and a session with no preferences reads exactly what it did.
|
||||
REPLY_PREFERENCES_CUE = (
|
||||
"The operator has preferences for how this report is written — "
|
||||
"follow `reply_preferences` over the default shape where they differ."
|
||||
)
|
||||
|
||||
_ITEM_KEYS = {"title", "body", "type", "status", "priority", "kind", "tags", "system_ids"}
|
||||
|
||||
|
||||
+76
-53
@@ -15,6 +15,7 @@ from scribe.config import Config
|
||||
from scribe.services import lesson_rules as lesson_rules_svc
|
||||
from scribe.services import moment_delivery as moment_delivery_svc
|
||||
from scribe.services import plugin_context as plugin_ctx_svc
|
||||
from scribe.services import reply_shapes as reply_shapes_svc
|
||||
from scribe.services import repo_bindings as repo_bindings_svc
|
||||
from scribe.services import report_check as report_check_svc
|
||||
from scribe.services import rule_moment_judgments as judgments_svc
|
||||
@@ -126,6 +127,17 @@ async def autoinject_retrieve():
|
||||
different things about what the reader holds —
|
||||
so they get different lines (#4100). Shares the
|
||||
ledger directory and the same clear-on-compact.
|
||||
shapes_seen (opt) — comma-separated reply-shape keys this session was
|
||||
already shown in full (milestone 500). The core
|
||||
goes out in full when `core` is absent, as its
|
||||
one-line reminder when present; the hook's
|
||||
ledger is swept on compaction, which is how the
|
||||
full core comes back after one.
|
||||
|
||||
THE TURN'S REPLY SHAPE COMES FIRST (milestone 500 step 3). Every turn ends
|
||||
in a reply and this is the last point before it is written, so the core
|
||||
shape and whatever is mounted on `reply.report` lead the payload. Fresh
|
||||
shape keys come back as `shape_keys` for the ledger.
|
||||
|
||||
TWO ARMS, TWO SETS OF GATES. Rules ride the same hook and the same query
|
||||
but nothing else: the notes menu can be disabled, thresholded and top-k'd
|
||||
@@ -159,9 +171,19 @@ async def autoinject_retrieve():
|
||||
g.user.id, result.get("lesson_ids") or [], rules.get("shown_rule_ids") or [],
|
||||
arm="p", situation=q, project_id=project_id or None,
|
||||
)
|
||||
blocks = [b for b in (rules["context"], result["context"], proposal) if b]
|
||||
# The reply mounts go through the same rule ledger; a rule the prompt arm
|
||||
# just named is not quoted a second time by the turn's delivery.
|
||||
turn = await moment_delivery_svc.deliver_for_turn(
|
||||
g.user.id, seen=reply_shapes_svc.parse_seen(request.args.get("shapes_seen")),
|
||||
project_id=project_id,
|
||||
exclude=frozenset(exclude_rule_ids) | frozenset(rules["rule_ids"]),
|
||||
held=frozenset(held_rule_ids),
|
||||
)
|
||||
blocks = [b for b in (turn["context"], rules["context"], result["context"], proposal) if b]
|
||||
result["context"] = "\n\n".join(blocks)
|
||||
result["rule_ids"] = rules["rule_ids"]
|
||||
result["rule_ids"] = list(dict.fromkeys([*rules["rule_ids"], *turn["rule_ids"]]))
|
||||
result["shape_keys"] = [k for k, form in turn["shape_forms"].items()
|
||||
if form == reply_shapes_svc.FULL]
|
||||
return jsonify(result)
|
||||
|
||||
|
||||
@@ -275,12 +297,16 @@ async def moment():
|
||||
`rule_ids` (the FRESH ones, for the hook to append to the ledger) and
|
||||
`moments` (the names reached). A lookup, not a ranked search: no
|
||||
retrieval_logs row; each fresh rule is recorded surfaced with its moment.
|
||||
|
||||
A moment that carries a default reply shape (milestone 500) brings it too,
|
||||
ahead of the rules: in full unless `shapes_seen` names it, then as its
|
||||
reminder. `shape_keys` are the ones sent in full, for the hook's ledger.
|
||||
"""
|
||||
event = await request.get_json(silent=True) or {}
|
||||
tool = str(event.get("tool_name") or "").strip()
|
||||
tool_input = event.get("tool_input")
|
||||
if not tool:
|
||||
return jsonify({"context": "", "rule_ids": [], "moments": []})
|
||||
return jsonify({"context": "", "rule_ids": [], "moments": [], "shape_keys": []})
|
||||
project_id, _repo, _unbound = await _project_scope()
|
||||
reached, result = await moment_delivery_svc.deliver_for_act(
|
||||
g.user.id, tool, tool_input if isinstance(tool_input, dict) else {},
|
||||
@@ -288,10 +314,16 @@ async def moment():
|
||||
exclude=frozenset(_int_list(request.args.get("exclude_rule_ids"))),
|
||||
held=frozenset(_int_list(request.args.get("held_rule_ids"))),
|
||||
)
|
||||
shapes, forms = moment_delivery_svc.shapes_for_act(
|
||||
reached, reply_shapes_svc.parse_seen(request.args.get("shapes_seen")),
|
||||
)
|
||||
await reply_shapes_svc.record_delivery(g.user.id, forms, via="hook")
|
||||
context = "\n\n".join(b for b in ("\n\n".join(shapes), "\n".join(result.lines)) if b)
|
||||
return jsonify({
|
||||
"context": "\n".join(result.lines),
|
||||
"context": context,
|
||||
"rule_ids": result.rule_ids,
|
||||
"moments": [hit["moment"] for hit in reached],
|
||||
"shape_keys": [k for k, form in forms.items() if form == reply_shapes_svc.FULL],
|
||||
})
|
||||
|
||||
|
||||
@@ -335,28 +367,58 @@ async def reply_rules():
|
||||
the reply asks something), and the reply text against every rule's
|
||||
trigger — the backstop for whatever the earlier arms missed.
|
||||
|
||||
Body: `{"reply": "<text>"}`. Query: `repo` / `project_id` for the scope;
|
||||
ONE END-OF-TURN REQUEST (milestone 500 step 4): a reply that closed tasks
|
||||
is also checked here for the completion sections (services/report_check),
|
||||
which used to be a Stop hook and an endpoint of its own. Both holds fold
|
||||
into one `reason`; `report_check` says what the section check recorded, so
|
||||
the hook knows the rewrite is one to report.
|
||||
|
||||
Body: `{"reply": "<text>", "closed": 1, "closed_task_ids": [41],
|
||||
"rewrite": false}` — the last three only when the turn closed a task.
|
||||
`closed` is the count and decides whether the check runs (a task created
|
||||
already done has no id to list); `rewrite` marks the reply written after a
|
||||
section hold, which is recorded and never held.
|
||||
Query: `repo` / `project_id` for the scope;
|
||||
`held_rule_ids` (opened this session — exempt, as at the act checkpoint);
|
||||
`exclude_rule_ids` (named this session — not re-counted as surfaced);
|
||||
`stopped_rule_ids` (already held a reply or an act this session — a rule
|
||||
holds once, and the per-session cap counts these).
|
||||
|
||||
Returns `reason` — the words the hook blocks with, empty when nothing
|
||||
holds — plus `rule_ids` for the hook's ledger and `moments` reached.
|
||||
holds — plus `rule_ids` for the hook's ledger, `moments` reached and
|
||||
`report_check` (the recorded outcome, or "" when nothing was checked).
|
||||
"""
|
||||
data = await request.get_json(silent=True) or {}
|
||||
if not isinstance(data, dict):
|
||||
data = {}
|
||||
reply = str(data.get("reply") or "")
|
||||
# `type(...) is int`, not isinstance: JSON `true` would otherwise read as 1.
|
||||
closed = data.get("closed") if type(data.get("closed")) is int else 0
|
||||
ids = data.get("closed_task_ids")
|
||||
task_ids = [i for i in ids if type(i) is int and i > 0][:20] if isinstance(ids, list) else []
|
||||
rewrite = data.get("rewrite") is True
|
||||
project_id, _repo, _unbound = await _project_scope()
|
||||
held = await moment_delivery_svc.reply_hold(
|
||||
g.user.id, reply, project_id=project_id,
|
||||
exclude=frozenset(_int_list(request.args.get("exclude_rule_ids"))),
|
||||
held=frozenset(_int_list(request.args.get("held_rule_ids"))),
|
||||
stopped=frozenset(_int_list(request.args.get("stopped_rule_ids"))),
|
||||
)
|
||||
|
||||
checked: dict = {}
|
||||
if closed > 0:
|
||||
checked = await report_check_svc.check_reply(
|
||||
g.user.id, reply, task_ids=task_ids, rewrite=rewrite, project_id=project_id or None,
|
||||
)
|
||||
# The rewrite after a hold goes out as written: recorded above, held by nothing.
|
||||
held: dict = {}
|
||||
if not rewrite:
|
||||
held = await moment_delivery_svc.reply_hold(
|
||||
g.user.id, reply, project_id=project_id,
|
||||
exclude=frozenset(_int_list(request.args.get("exclude_rule_ids"))),
|
||||
held=frozenset(_int_list(request.args.get("held_rule_ids"))),
|
||||
stopped=frozenset(_int_list(request.args.get("stopped_rule_ids"))),
|
||||
)
|
||||
reasons = [r for r in (checked.get("reason", ""), held.get("reason", "")) if r]
|
||||
return jsonify({
|
||||
"reason": held.get("reason", ""),
|
||||
"reason": "\n\n".join(reasons),
|
||||
"rule_ids": held.get("rule_ids", []),
|
||||
"moments": held.get("moments", []),
|
||||
"report_check": checked.get("outcome", ""),
|
||||
})
|
||||
|
||||
|
||||
@@ -472,45 +534,6 @@ def _parse_shapes(raw: str) -> list[tuple[str, str]]:
|
||||
return out
|
||||
|
||||
|
||||
@plugin_bp.get("/report-check")
|
||||
@login_required
|
||||
async def report_check():
|
||||
"""Record what a Stop hook found in a reply that closed a task (milestone 409 step 5).
|
||||
|
||||
The hook decides which completion sections the reply lacks — a local check
|
||||
of text it can read — and reports the outcome here. For `blocked` the
|
||||
response carries the `reason` to send the agent back with: the words are
|
||||
the server's, so every client's hook says the same thing (plugin/PACKAGING.md).
|
||||
A hook blocks only on a `reason` it received, which means only on a block
|
||||
that was recorded.
|
||||
|
||||
A GET for the reason every plugin endpoint is one: a read-scoped key must
|
||||
be enough to run the plugin, and this records telemetry the way /retrieve
|
||||
records a retrieval log.
|
||||
|
||||
Query:
|
||||
outcome (str) — passed | blocked | passed_after_rewrite |
|
||||
missing_after_rewrite. Anything else is a 400.
|
||||
missing (opt) — comma-separated sections the reply lacked:
|
||||
"where it sits", "needs you", "next".
|
||||
task_ids (opt) — comma-separated ids of the tasks the turn closed.
|
||||
repo (opt) — working repo remote, resolved like the other arms.
|
||||
"""
|
||||
outcome = (request.args.get("outcome") or "").strip()
|
||||
if outcome not in report_check_svc.OUTCOMES:
|
||||
return jsonify({"error": f"outcome must be one of {list(report_check_svc.OUTCOMES)}"}), 400
|
||||
missing = [m for m in (request.args.get("missing") or "").split(",") if m.strip()]
|
||||
task_ids = _int_list(request.args.get("task_ids"))[:20]
|
||||
project_id, _repo, _unbound = await _project_scope()
|
||||
await report_check_svc.record_report_check(
|
||||
g.user.id, outcome, missing=missing, task_ids=task_ids, project_id=project_id or None,
|
||||
)
|
||||
body: dict = {"status": "ok"}
|
||||
if outcome == "blocked":
|
||||
body["reason"] = report_check_svc.block_reason(missing)
|
||||
return jsonify(body)
|
||||
|
||||
|
||||
@plugin_bp.get("/shape-check")
|
||||
@login_required
|
||||
async def shape_check():
|
||||
@@ -525,7 +548,7 @@ async def shape_check():
|
||||
recorded, and never on the stop that follows one (`phase=after`).
|
||||
|
||||
A GET like every plugin endpoint: a read-scoped key runs the plugin, and
|
||||
this records telemetry the way /report-check does. It changes no ledger
|
||||
this records telemetry the way /reply-rules records its section check. It changes no ledger
|
||||
row — the agent's own classify_shapes call does that.
|
||||
|
||||
Query:
|
||||
|
||||
@@ -20,6 +20,7 @@ from quart import Blueprint, jsonify, request
|
||||
from scribe.auth import get_current_user_id, login_required
|
||||
from scribe.services import moment_actions as moment_actions_svc
|
||||
from scribe.services import moments as moments_svc
|
||||
from scribe.services import reply_shapes as shapes_svc
|
||||
from scribe.services import rule_moment_judgments as judgments_svc
|
||||
from scribe.services import rulebooks as rulebooks_svc
|
||||
from scribe.services.retrieval_telemetry import moment_usage
|
||||
@@ -143,6 +144,23 @@ async def moments_route():
|
||||
return jsonify(out)
|
||||
|
||||
|
||||
@retrieval_bp.route("/reply-shapes", methods=["GET"])
|
||||
@login_required
|
||||
async def reply_shapes_route():
|
||||
"""The default reply shapes and the moment each rides (milestone 500).
|
||||
|
||||
`list_reply_shapes`' payload from the same service, so the Settings view
|
||||
and the session cannot show different defaults — plus what only the
|
||||
operator's view needs (step 5): per shape, the preferences mounted on its
|
||||
moment and its deliveries over `days` (default 30, 1–90).
|
||||
"""
|
||||
try:
|
||||
days = min(90, max(1, int(request.args.get("days") or shapes_svc.OVERVIEW_DAYS)))
|
||||
except ValueError:
|
||||
days = shapes_svc.OVERVIEW_DAYS
|
||||
return jsonify(await shapes_svc.overview(get_current_user_id(), days=days))
|
||||
|
||||
|
||||
async def _mapping_change(change):
|
||||
"""map/unmap from the browser: the MCP tools' service, recorded as human.
|
||||
|
||||
|
||||
@@ -37,6 +37,7 @@ from __future__ import annotations
|
||||
import hashlib
|
||||
import logging
|
||||
import re
|
||||
import statistics
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
from sqlalchemy import and_, delete, exists, func, not_, or_, select
|
||||
@@ -735,17 +736,47 @@ async def confirmed_rules_in_scope(
|
||||
# answered, and never by a sweep or a timer (#4183): the reader is in the
|
||||
# situation then, and a nudge arriving anywhere else is one nobody acts on.
|
||||
#
|
||||
# WHAT "RESEMBLE" MEANS (#5193). Not a fixed similarity. Lessons are written
|
||||
# in one register — a sentence of advice and the moment it applies — so an
|
||||
# embedder scores any two of them well above two unrelated texts, and a
|
||||
# constant bar sits inside that band: every no-rule lesson then "resembles"
|
||||
# most of the others, and the group grows with the pool rather than with any
|
||||
# situation recurring. The bar is instead relative to each lesson's own
|
||||
# background: how far a neighbour stands above the typical score that lesson
|
||||
# gets against every lesson in its register, measured in that register's own
|
||||
# spread (median and MAD, so the near neighbours being judged do not move the
|
||||
# yardstick). An install with a different embedder, or lessons in a different
|
||||
# style, gets a different band and the same bar.
|
||||
#
|
||||
# And resemblance must be MUTUAL, among every member, not just to the lesson
|
||||
# being answered. A lesson broad enough to stand near many others is a hub,
|
||||
# and hub-and-spoke is how three unrelated lessons used to make a "group"
|
||||
# through one general one. A clique is the shape of one situation met again.
|
||||
#
|
||||
# DEFAULTS, stated as defaults (rules 32, 115). Three lessons — the new one
|
||||
# and two it resembles — is the smallest group that is a pattern rather than
|
||||
# a pair. The similarity bar sits above the notes menu's ("worth showing")
|
||||
# and below the duplicate gate's ("the same record"): these lessons should be
|
||||
# about one situation without being one lesson written twice, which the
|
||||
# duplicate gate already catches.
|
||||
# a pair. One spread above the background is a modest bar for one direction
|
||||
# alone; the strictness comes from requiring it of every pair in both
|
||||
# directions, which unrelated lessons rarely all clear. Raise it if groups
|
||||
# name lessons a reader would not call one situation; lower it if lessons a
|
||||
# reader would group never meet.
|
||||
CONVERGENCE_LESSONS = 3
|
||||
CONVERGENCE_THRESHOLD = 0.65
|
||||
# Candidates fetched before keeping the no-rule ones; most lessons near a
|
||||
# situation may well have rules.
|
||||
_CONVERGENCE_FETCH = 20
|
||||
CONVERGENCE_STANDOUT = 1.0
|
||||
# Below this many other lessons a median and its spread describe too little
|
||||
# to say what stands out, so nothing is named.
|
||||
CONVERGENCE_MIN_BACKGROUND = 8
|
||||
# How many of a lesson's register the background is read from. A register
|
||||
# larger than this is measured on its nearest part, which reads the
|
||||
# background HIGH — the bar errs strict, toward naming nothing, never toward
|
||||
# a group that is only the pool's size.
|
||||
_CONVERGENCE_BACKGROUND = 200
|
||||
# Neighbours checked for mutual resemblance, nearest first. Each costs one
|
||||
# more search at the write, and a group large enough to need more is named
|
||||
# just as well by its nearest members.
|
||||
_CONVERGENCE_CANDIDATES = 8
|
||||
# The consistency constant that makes a MAD estimate a standard deviation's
|
||||
# scale on normal data, so CONVERGENCE_STANDOUT reads as "spreads".
|
||||
_MAD_SCALE = 1.4826
|
||||
|
||||
|
||||
def convergence_group(members: list[dict]) -> dict | None:
|
||||
@@ -812,6 +843,38 @@ async def convergence_for(user_id: int, lesson_id: int) -> dict | None:
|
||||
return None
|
||||
|
||||
|
||||
def standout(scores: dict[int, float]) -> dict[int, float]:
|
||||
"""How far each lesson stands above this one's background, in spreads.
|
||||
|
||||
`scores` is one lesson's similarity to every other lesson in its register,
|
||||
itself left out. Pure, so the bar is testable. Empty when there is too
|
||||
little background to measure, or none of it varies.
|
||||
"""
|
||||
if len(scores) < CONVERGENCE_MIN_BACKGROUND:
|
||||
return {}
|
||||
middle = statistics.median(scores.values())
|
||||
spread = statistics.median(abs(v - middle) for v in scores.values()) * _MAD_SCALE
|
||||
if spread <= 0:
|
||||
return {}
|
||||
return {i: (v - middle) / spread for i, v in scores.items()}
|
||||
|
||||
|
||||
def converging(lesson_id: int, standouts: dict[int, dict[int, float]],
|
||||
candidates: list[int]) -> list[int]:
|
||||
"""The lesson, then each candidate that stands out to EVERY member so far
|
||||
and every member to it. Candidates are taken nearest first, so the clique
|
||||
grows around the closest situation rather than the first id. Pure."""
|
||||
def near(a: int, b: int) -> bool:
|
||||
return (standouts.get(a, {}).get(b, float("-inf")) >= CONVERGENCE_STANDOUT
|
||||
and standouts.get(b, {}).get(a, float("-inf")) >= CONVERGENCE_STANDOUT)
|
||||
|
||||
group = [lesson_id]
|
||||
for c in candidates:
|
||||
if c not in group and all(near(c, m) for m in group):
|
||||
group.append(c)
|
||||
return group
|
||||
|
||||
|
||||
async def _convergence_for(user_id: int, lesson_id: int) -> dict | None:
|
||||
from scribe.services import lessons as lessons_svc
|
||||
from scribe.services.embeddings import semantic_search_notes, trigger_title
|
||||
@@ -819,14 +882,34 @@ async def _convergence_for(user_id: int, lesson_id: int) -> dict | None:
|
||||
lesson = await lessons_svc.get_lesson(user_id, lesson_id)
|
||||
if lesson is None:
|
||||
return None
|
||||
data = lesson.data if isinstance(lesson.data, dict) else {}
|
||||
query = trigger_title(data.get("what") or lesson.title, lessons_svc.lesson_trigger(lesson))
|
||||
found = await semantic_search_notes(
|
||||
user_id, query, exclude_ids={int(lesson.id)}, limit=_CONVERGENCE_FETCH,
|
||||
threshold=CONVERGENCE_THRESHOLD, note_type=(lessons_svc.LESSON_NOTE_TYPE,),
|
||||
include_global_kinds=True, scope="browse",
|
||||
)
|
||||
answered = await _no_rule_ids([int(n.id) for _s, n in found])
|
||||
|
||||
async def background_of(note) -> list[tuple[float, Note]]:
|
||||
"""`note`'s similarity to every other lesson it can reach — the
|
||||
background it is measured against. No threshold: the bar is relative,
|
||||
and needs the scores a fixed bar would have turned away. No
|
||||
supersession penalty either: it would shift some scores and not
|
||||
others, and the background is a measure of resemblance alone."""
|
||||
data = note.data if isinstance(note.data, dict) else {}
|
||||
query = trigger_title(data.get("what") or note.title, lessons_svc.lesson_trigger(note))
|
||||
return await semantic_search_notes(
|
||||
user_id, query, exclude_ids={int(note.id)}, limit=_CONVERGENCE_BACKGROUND,
|
||||
threshold=-1.0, note_type=(lessons_svc.LESSON_NOTE_TYPE,),
|
||||
include_global_kinds=True, scope="browse", demote_superseded=False,
|
||||
)
|
||||
|
||||
lid = int(lesson.id)
|
||||
found = await background_of(lesson)
|
||||
notes = {int(n.id): n for _s, n in found}
|
||||
standouts = {lid: standout({int(n.id): s for s, n in found})}
|
||||
standing = [i for i, z in standouts[lid].items() if z >= CONVERGENCE_STANDOUT]
|
||||
answered = await _no_rule_ids(standing)
|
||||
candidates = sorted(answered, key=lambda i: (-standouts[lid][i], i))[:_CONVERGENCE_CANDIDATES]
|
||||
# Too few stand out from here for any clique to reach the bar — skip the
|
||||
# searches the mutual check would cost.
|
||||
if len(candidates) < CONVERGENCE_LESSONS - 1:
|
||||
return None
|
||||
for c in candidates:
|
||||
standouts[c] = standout({int(n.id): s for s, n in await background_of(notes[c])})
|
||||
|
||||
def member(note) -> dict:
|
||||
return {
|
||||
@@ -834,6 +917,5 @@ async def _convergence_for(user_id: int, lesson_id: int) -> dict | None:
|
||||
"sources": lessons_svc.lesson_sources(note), "project_id": note.project_id,
|
||||
}
|
||||
|
||||
return convergence_group(
|
||||
[member(lesson)] + [member(n) for _s, n in found if int(n.id) in answered]
|
||||
)
|
||||
group = converging(lid, standouts, candidates)
|
||||
return convergence_group([member(lesson)] + [member(notes[i]) for i in group[1:]])
|
||||
|
||||
@@ -16,7 +16,7 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from scribe.services import moment_actions
|
||||
from scribe.services import moment_actions, reply_shapes
|
||||
from scribe.services import retrieval_pipeline as rp
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -38,23 +38,27 @@ async def reachable_tools(user_id: int) -> list[str]:
|
||||
What the plugin's catch-all hook reads once per session window, so the
|
||||
calls that cannot reach a mounted rule — most of them, and every one on
|
||||
an install that has mounted nothing — never leave the machine. An action
|
||||
counts only when its moment carries a mount. The skill loader counts
|
||||
whenever anything is mounted: a stored process declares its own moments,
|
||||
which only the load itself can resolve, and a load is rare enough that
|
||||
asking costs nothing.
|
||||
counts when its moment carries a mount or a default reply shape. The
|
||||
skill loader counts whenever anything is mounted: a stored process
|
||||
declares its own moments, which only the load itself can resolve, and a
|
||||
load is rare enough that asking costs nothing.
|
||||
"""
|
||||
from scribe.services import rulebooks
|
||||
|
||||
mounted = await rulebooks.mounted_moments(user_id)
|
||||
if not mounted:
|
||||
return []
|
||||
# A moment that carries a default reply shape (milestone 500) is worth a
|
||||
# request whether or not anything is mounted on it: the shape is product,
|
||||
# so every install has it.
|
||||
shaped = {s.moment for s in reply_shapes.SHAPES.values() if s.key != reply_shapes.CORE_KEY}
|
||||
mounted = set(await rulebooks.mounted_moments(user_id))
|
||||
wanted = mounted | shaped
|
||||
mappings = await moment_actions.list_mappings(user_id)
|
||||
keys = {
|
||||
moment_actions.tool_key(action.tool)
|
||||
for action, _via in moment_actions.effective_actions(mappings)
|
||||
if action.moment in mounted
|
||||
if action.moment in wanted
|
||||
}
|
||||
keys.add(moment_actions.SKILL_TOOL)
|
||||
if mounted:
|
||||
keys.add(moment_actions.SKILL_TOOL)
|
||||
return sorted(keys)
|
||||
|
||||
|
||||
@@ -86,6 +90,58 @@ async def deliver_for_act(
|
||||
return reached, result
|
||||
|
||||
|
||||
def shapes_for_act(reached: list[dict], seen: frozenset[str] = frozenset()) -> tuple[list[str], dict[str, str]]:
|
||||
"""The default reply shapes riding the moments this act reached (milestone 500).
|
||||
|
||||
Closing a task comes just before a completion report, a structured
|
||||
question is an ask, opening a plan comes before a plan is put up for
|
||||
review — so the shape for that reply arrives at the act, before the reply
|
||||
is written. In full the first time a session meets it, as its reminder
|
||||
after that; a door with no ledger passes `seen` empty.
|
||||
"""
|
||||
return reply_shapes.deliver(
|
||||
reply_shapes.for_moments([hit["moment"] for hit in reached]), seen,
|
||||
)
|
||||
|
||||
|
||||
# What the per-turn delivery says reached `reply.report`: the turn has not
|
||||
# ended yet, but every turn ends in a reply, and this is the last point
|
||||
# before it is written.
|
||||
_REACHED_BY_TURN = "the reply this turn will end with"
|
||||
|
||||
|
||||
async def deliver_for_turn(
|
||||
user_id: int, *, seen: frozenset[str] = frozenset(), project_id: int | None = None,
|
||||
exclude: frozenset[int] = frozenset(), held: frozenset[int] = frozenset(),
|
||||
) -> dict:
|
||||
"""What every turn carries before its reply is written (milestone 500 step 3).
|
||||
|
||||
The core reply shape — in full once per session and again after a
|
||||
compaction (the ledger is swept then), otherwise its one-line reminder —
|
||||
and the rules and preferences mounted on `reply.report`, under the same
|
||||
rule ledger as every other arm. The Stop hook's reply moment fires after
|
||||
the reply exists, which is too late to shape it and is kept as the
|
||||
backstop; this is the point before.
|
||||
|
||||
Returns `{context, rule_ids, shape_forms}`. Fails open to the core alone,
|
||||
and the core itself never fails: it is a constant.
|
||||
"""
|
||||
blocks, forms = reply_shapes.deliver([reply_shapes.core()], seen)
|
||||
rule_ids: list[int] = []
|
||||
try:
|
||||
reached = [{"moment": "reply.report", "tool": "", "match": _REACHED_BY_TURN,
|
||||
"via": moment_actions.DEFAULT}]
|
||||
result = await deliver_moments(
|
||||
user_id, reached, project_id=project_id, exclude=exclude, held=held,
|
||||
)
|
||||
blocks.extend(result.lines)
|
||||
rule_ids = result.rule_ids
|
||||
except Exception: # noqa: BLE001 - the mounted half never costs the core
|
||||
logger.debug("reply.report mounts not delivered for the turn", exc_info=True)
|
||||
await reply_shapes.record_delivery(user_id, forms, via="turn")
|
||||
return {"context": "\n\n".join(blocks), "rule_ids": rule_ids, "shape_forms": forms}
|
||||
|
||||
|
||||
async def attach_moment_rules(
|
||||
user_id: int, tool: str, arguments: dict | None, data: dict,
|
||||
) -> dict:
|
||||
@@ -114,6 +170,13 @@ async def attach_moment_rules(
|
||||
"rule_ids": result.shown_rule_ids,
|
||||
"open_with": "get_rule(id)",
|
||||
}
|
||||
# The reply shape for this moment, in full: no ledger reaches this
|
||||
# door, and an act like closing a task is rare enough that the shape
|
||||
# arriving each time costs less than one report written without it.
|
||||
blocks, forms = shapes_for_act(_reached)
|
||||
if blocks:
|
||||
data["reply_shape"] = "\n\n".join(blocks)
|
||||
await reply_shapes.record_delivery(user_id, forms, via="mcp")
|
||||
except Exception: # noqa: BLE001 - a decoration never breaks the payload
|
||||
logger.debug("moment rules for %s could not be attached", tool, exc_info=True)
|
||||
return data
|
||||
|
||||
@@ -601,25 +601,6 @@ PROMPTRULE_DEFAULT_THRESHOLD = SURFACES["prompt_rule"].floor_default
|
||||
# at all, where the act arms never face it because a command is one thing.
|
||||
PROMPTRULE_LIMIT = SURFACES["prompt_rule"].budget_default
|
||||
|
||||
# THE COMPLETION-REPORT ARM'S OWN BAR (services/reply_preferences.py).
|
||||
#
|
||||
# It borrowed PROMPTRULE_THRESHOLD_KEY when it shipped, which made the two
|
||||
# arms one dial: an operator lowering the bar for their own prose moved this
|
||||
# one with it, silently. That contradicts the rule every other bar here
|
||||
# follows — one number cannot serve arms whose queries are different shapes —
|
||||
# and this arm's query is the most different of all. The others score an
|
||||
# operator's prose or a session's code, both of which vary per call; this one
|
||||
# scores a FIXED string (`COMPLETION_QUERY`) against rule triggers, so its
|
||||
# score for a given corpus is a constant. A constant that lands under the bar
|
||||
# is not a quiet arm, it is a dead one, and nothing about the prose arm's
|
||||
# traffic would ever reveal it.
|
||||
#
|
||||
# Kept at the prose arm's starting value rather than tuned: the split is what
|
||||
# makes the two independently movable, and a default is a product decision
|
||||
# that this install's corpus cannot settle (rule 115).
|
||||
REPORTPREF_THRESHOLD_KEY = SURFACES["report_preference"].floor_key
|
||||
REPORTPREF_DEFAULT_THRESHOLD = SURFACES["report_preference"].floor_default
|
||||
|
||||
|
||||
def _slugify(text: str) -> str:
|
||||
"""kebab-case slug for a skill directory name (a-z0-9 + single hyphens)."""
|
||||
@@ -726,7 +707,8 @@ async def get_autoinject_config(user_id: int) -> dict:
|
||||
|
||||
THE TWO NUMBERS COME FROM THE REGISTRY NOW (#4102). They used to be read and
|
||||
clamped here, and identically again in `get_writepath_config`, and again in
|
||||
three rule arms, and once more in `reply_preferences`. That was
|
||||
three rule arms, and once more in the completion-report arm (since
|
||||
retired, milestone 500). That was
|
||||
tolerable while the values were shipped constants. It stops being tolerable
|
||||
once a tool is expected to MOVE them, because a tuning surface cannot be
|
||||
consistent across arms that each spell their configuration differently.
|
||||
|
||||
@@ -1,141 +0,0 @@
|
||||
"""The operator's own preferences for a completion report, at the moment it is written.
|
||||
|
||||
WHY THIS EXISTS (milestone 409 step 4)
|
||||
|
||||
The reporting-back skill ships DEFAULT shapes. An operator will want some of
|
||||
them different ("my completion reports also say how it was tested", "decisions
|
||||
as a numbered list"), and those adjustments are `preference` records. The gap
|
||||
is the query: prompt-time retrieval matches the OPERATOR'S MESSAGE, and a shape
|
||||
preference is about the REPLY. "Fix the flaky test" never retrieves "completion
|
||||
reports should say how it was tested", so the preference is on file and never
|
||||
arrives.
|
||||
|
||||
THE DECISION (operator, 2026-09-14, logged on the step): two deliveries, split
|
||||
by reply kind.
|
||||
|
||||
- A COMPLETION REPORT has a moment the server can see — a task closing — so
|
||||
the server retrieves for it and hands the matches back beside
|
||||
`report_back`. That is this module.
|
||||
- EVERY OTHER REPLY KIND (a finding, a decision, a handoff…) has no tool call
|
||||
in front of it, so the reporting-back skill asks: it tells the agent to
|
||||
`search(content_type="rule")` for that kind before writing.
|
||||
|
||||
Loading reply-shape preferences at session start was the rejected third option:
|
||||
it is a small copy of the preloading milestone 394 retired, and just as
|
||||
unmeasurable.
|
||||
|
||||
HOW A PREFERENCE SAYS IT IS ABOUT COMPLETION REPORTS
|
||||
|
||||
By its trigger, which is what it already has — no tag, no new column. A
|
||||
preference's `when_to_apply` dominates its embedded document, so one written
|
||||
for this moment ("writing the report after finishing a task") resembles
|
||||
COMPLETION_QUERY below, and one about anything else does not. That keeps
|
||||
delivery entirely in retrieval, as 394 decided, and leaves an operator nothing
|
||||
new to learn: a preference reaches the completion report the same way every
|
||||
other record reaches its moment.
|
||||
|
||||
THE BAR IS ITS OWN, AND THE FIRST READING EARNED IT (#3860)
|
||||
|
||||
This borrowed the prompt arm's key when it shipped, on the argument that a
|
||||
fixed query against triggers is a different score distribution from an
|
||||
operator's message against the same documents — true, and the reason the two
|
||||
could not stay one dial. Five days of traffic settled it.
|
||||
|
||||
What the readout said: 69 calls, 69 declines, every one naming the SAME record
|
||||
at the SAME score (rule 77 at 0.7194 against a 0.72 bar). That constancy is
|
||||
the signature of this arm — `COMPLETION_QUERY` never varies, so for a given
|
||||
corpus its best score is a constant, and a constant sitting under the bar is a
|
||||
dead arm rather than a quiet one. The record it kept declining was about
|
||||
reading a REQUEST, not about the shape of a report, so the decline was right
|
||||
and the arm is healthy: this install simply has no completion-report
|
||||
preference on file.
|
||||
|
||||
The bar stayed at 0.72, and the key moved out (REPORTPREF_THRESHOLD_KEY) so
|
||||
that staying is a decision rather than a side effect of what the prose arm is
|
||||
set to. A surface whose score cannot vary is the one surface where a borrowed
|
||||
bar can be wrong forever without a single call looking unusual.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from scribe.services import retrieval_pipeline as rp
|
||||
from scribe.services.embeddings import semantic_search_rules
|
||||
from scribe.services.retrieval_surfaces import SURFACES, budget_for, floor_for
|
||||
from scribe.services.retrieval_telemetry import record_retrieval
|
||||
from scribe.services.rule_usage import record_rule_surfaced
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
SOURCE = rp.REPORT_PREFERENCE.source
|
||||
|
||||
# Written in the vocabulary of the MOMENT, because that is what a trigger is
|
||||
# written in and what this query is scored against. Domain-neutral on purpose
|
||||
# (rule #115): a writing project or a home-infrastructure project closes tasks
|
||||
# too, and its operator's preferences must match as well as a developer's.
|
||||
COMPLETION_QUERY = (
|
||||
"writing the completion report to the operator after finishing a task — "
|
||||
"how that reply should be laid out and what it should include"
|
||||
)
|
||||
|
||||
# A handful, not a menu. More than a few shape preferences for ONE kind of
|
||||
# reply would contradict each other before they helped; the limit is here to
|
||||
# keep one noisy corpus from turning a status change into a wall of text.
|
||||
# The STARTING budget, not the budget (#4102). Both numbers this arm runs on
|
||||
# now come from the surface registry, so the model that reads this arm's
|
||||
# telemetry can move either — which matters more here than anywhere else,
|
||||
# because a fixed query makes this arm's score a constant and a floor a hair
|
||||
# above it produces a dead arm no amount of traffic will ever reveal.
|
||||
LIMIT = SURFACES["report_preference"].budget_default
|
||||
|
||||
|
||||
async def _threshold(user_id: int) -> float:
|
||||
return await floor_for(user_id, SOURCE)
|
||||
|
||||
|
||||
async def _limit(user_id: int) -> int:
|
||||
return await budget_for(user_id, SOURCE)
|
||||
|
||||
|
||||
async def completion_preferences(user_id: int, *, project_id: int | None = None) -> list[dict]:
|
||||
"""The operator's preferences for a completion report, best match first.
|
||||
|
||||
KIND-FILTERED, for the reason `_reserve_slot_for_preference` gives: a
|
||||
binding rule that happened to resemble the query would otherwise ride out
|
||||
under a key that says "how the operator likes this written", which is a
|
||||
claim about force the record does not make.
|
||||
|
||||
Every call is logged, the empty ones included: a surface that records only
|
||||
the calls it liked reports a flawless clear-rate however badly its bar is
|
||||
set. An install with no preferences at all logs nothing, because no search
|
||||
ran (`searched` stays False) — that is not a decline.
|
||||
|
||||
Fails open to an empty list: this decorates a write that has already
|
||||
happened, and a lookup that errors must not turn it into a failure.
|
||||
"""
|
||||
try:
|
||||
# The stages — the kind filter, the unconditional call row, the
|
||||
# fresh-only surfacing rows — are the one pipeline's (milestone 456).
|
||||
# RANKED: this surface chose what it showed, so the name is in
|
||||
# rule_usage.RANKED_SOURCES and its hits count toward pull-through.
|
||||
# There is no session ledger here: a completion report is written
|
||||
# once, so nothing it could repeat has been shown before.
|
||||
result = await rp.run_rule_arm(
|
||||
rp.REPORT_PREFERENCE,
|
||||
rp.RuleMoment(
|
||||
user_id=user_id, query=COMPLETION_QUERY, project_id=project_id,
|
||||
),
|
||||
floor=await _threshold(user_id), budget=await _limit(user_id),
|
||||
io=rp.RuleIO(
|
||||
search=semantic_search_rules,
|
||||
record_retrieval=record_retrieval,
|
||||
record_rule_surfaced=record_rule_surfaced,
|
||||
),
|
||||
)
|
||||
return [
|
||||
{"id": rule.id, "title": rule.title, "statement": rule.statement, "kind": "preference"}
|
||||
for _score, rule in result.shown
|
||||
]
|
||||
except Exception: # noqa: BLE001 - a decoration never breaks its payload
|
||||
logger.warning("completion preference lookup failed", exc_info=True)
|
||||
return []
|
||||
@@ -0,0 +1,363 @@
|
||||
"""The default shapes of a reply, and the moment each one rides (milestone 500).
|
||||
|
||||
WHY THIS EXISTS
|
||||
|
||||
A reply is where the person the work is for finds out what happened, and they
|
||||
were not there while it was done. Milestone 409 wrote the shapes that make a
|
||||
reply readable to them — conclusion first, the shortest reply that carries the
|
||||
answer, where the work stands — and put them in one bundled skill. A skill
|
||||
arrives only when the agent decides to load it, so in a long session the shape
|
||||
faded out of context exactly when replies were getting longer.
|
||||
|
||||
So the shapes are product content held here, on the server, and delivered by
|
||||
the moments they belong to rather than by the agent remembering to ask:
|
||||
|
||||
- the CORE applies to every reply, so it rides every turn — in full once per
|
||||
session and after a compaction, and as a one-line pointer otherwise;
|
||||
- each SLICE belongs to one kind of reply, and arrives at the moment that
|
||||
comes just before that kind is written: closing a task comes before a
|
||||
completion report, a structured question before an ask, opening a plan
|
||||
before a plan is put up for review.
|
||||
|
||||
The moment already knows which kind of reply is coming, so there is no "pick
|
||||
the type, then load its shape" step for an agent to skip.
|
||||
|
||||
ONE COPY
|
||||
|
||||
This module is the single source. The `reporting-back` skill keeps the
|
||||
long-form reference (a worked example, the reasoning) and points here; the
|
||||
hooks carry timing and transport and say nothing of their own about shape. An
|
||||
operator's adjustments are `preference` records mounted on the same moments,
|
||||
and arrive beside the default — where the two differ, the preference is what
|
||||
the operator asked for.
|
||||
|
||||
HOW TO WRITE ONE
|
||||
|
||||
Domain-neutral (rule 115): the work may be software, configuration,
|
||||
infrastructure or anything else a person drives through an agent. "Evidence"
|
||||
is what the reader could open to check; "verified" is whatever checking means
|
||||
for that work. The test beside this module refuses software-only vocabulary.
|
||||
|
||||
Short, because it is paid for in every session's context. The core's budget is
|
||||
asserted; a sentence added to it should replace one.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
|
||||
from scribe.services import moments
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ReplyShape:
|
||||
"""One piece of the default reply shape, and when it arrives."""
|
||||
|
||||
key: str
|
||||
"""Stable identifier: what a delivery records and the Settings view keys on."""
|
||||
|
||||
title: str
|
||||
"""What the piece is, as the Settings view names it."""
|
||||
|
||||
moment: str
|
||||
"""The catalog moment this piece rides — where an operator's own
|
||||
preferences for it are mounted too."""
|
||||
|
||||
delivered: str
|
||||
"""When it arrives, said plainly for the Settings view."""
|
||||
|
||||
text: str
|
||||
"""The shape itself, as the agent reads it."""
|
||||
|
||||
reminder: str
|
||||
"""One line standing in for `text` once the session has been shown it:
|
||||
the pointer form, which also serves as the reminder on later turns."""
|
||||
|
||||
|
||||
CORE_KEY = "core"
|
||||
|
||||
# ~550 tokens at four characters a token. The core is paid for in every
|
||||
# session, so this is the ceiling the test holds it to — not a target. It sits
|
||||
# at ~2,050 with the judgement line (milestone 500 step 2): close to full, so a
|
||||
# sentence added now has to replace one.
|
||||
CORE_BUDGET_CHARS = 2200
|
||||
# A slice is paid for only at its moment, and once per session; it can afford
|
||||
# more than the core and still has to stay a slice, not the old skill again.
|
||||
# The asks slice is the largest (~1,530) because it carries who decides what in
|
||||
# full: handing a question back is what an ask is tempted to do.
|
||||
SLICE_BUDGET_CHARS = 1600
|
||||
|
||||
|
||||
_CORE = """\
|
||||
Shape the reply around where the work stands, not the order you did things in. \
|
||||
The reader was not there while you worked.
|
||||
|
||||
- **Conclusion first** — the result, the verdict or the question, then what supports it.
|
||||
- **The shortest reply that carries the answer.** Length is work handed back to the \
|
||||
reader. It is earned by a comparison they asked for, options that need laying side by \
|
||||
side, or numbers that are the point.
|
||||
- **One topic per section; bold the few things that matter.**
|
||||
- **Plain words** — the reader's vocabulary, not names you coined while working.
|
||||
- **Place the work** — the task or plan it belongs to, by id and title, read from the \
|
||||
record rather than recalled. Work with no record behind it says so.
|
||||
- **Answer the standing questions, even with "nothing"**: does anything need them, and \
|
||||
what happens next. A section that explains earns its place only when it changes what \
|
||||
they do or decide; the rest belongs in the record's log.
|
||||
- **Decide what only you can see; ask about direction.** Whether a finding holds, what a \
|
||||
measurement says, whether the work is done: settle it, act, and say why so they can \
|
||||
overrule it. Priorities, choices they will live with, and anything hard to undo are theirs.
|
||||
- **A decision they have made is the input to the work.** Act on it; reopen it only \
|
||||
for new evidence, said once.
|
||||
- **Holding an action until they say yes?** Put it near the top, under "Approval requested".
|
||||
- **End with the one thing they need to decide or do, in bold** — or say there is none.
|
||||
|
||||
By kind:
|
||||
- **Answer** — the answer, then what they could open to check it. An evaluation is \
|
||||
verdict · what exists · the gaps · a recommendation.
|
||||
- **Proposal** — two or three approaches, the trade-off of each, one recommendation. \
|
||||
A review is findings ranked by how much they matter.
|
||||
- **Finding** — what you found, why it happens, and what you decided and did about it.
|
||||
- **Progress** — a line or two. **Blocked** — what stopped, what you tried, what you need.
|
||||
- **Where are we** — the plan and its progress · done · open · needs you · next.
|
||||
|
||||
Before sending, read it as someone who was not there, reading quickly; then once more \
|
||||
for what can go."""
|
||||
|
||||
|
||||
_COMPLETION = """\
|
||||
A completion report, from the `placement` the closing call returned:
|
||||
|
||||
- **Where this sits** — the plan and the step, or the task alone when it has no plan.
|
||||
- **What now works** — outcomes the reader would notice ("you can now…", "X no longer…"), \
|
||||
not the steps behind them; those belong in the task's log.
|
||||
- **How / why** — only the decisions worth knowing, and how it was verified. Anything \
|
||||
that could not be verified is said here rather than left to read as passed.
|
||||
- **Needs you** — an action, an approval, a decision, or "nothing". It has to answer two \
|
||||
questions yes: is it theirs to decide, and is work waiting on it? A question you could settle by \
|
||||
reading or measuring something is work not yet done — settle it, say which way you went, \
|
||||
and leave them free to overrule.
|
||||
- **Next** — from `placement.next`, or say the plan is finished. An offer to fix \
|
||||
something you found goes here."""
|
||||
|
||||
|
||||
_ASKS = """\
|
||||
Asking them to decide or to act — pick the shape:
|
||||
|
||||
- **Decision** — the question first · two to four options, each with what it changes · \
|
||||
your recommendation first among them.
|
||||
- **Clarification** — "My reading is X · the gap is Y · unless you say otherwise I'll do Z."
|
||||
- **Handoff** (only they can do it) — the action · why it needs them · what it unblocks · \
|
||||
what you will do after.
|
||||
- **Approval** (ready, and holding for a yes) — under "Approval requested": exactly what \
|
||||
happens on yes, one numbered item per change so they can approve part · why it needs \
|
||||
them · how it is undone.
|
||||
- **Conflict** (what you are about to do clashes with a rule, a plan or an earlier \
|
||||
decision) — what it says · what you were about to do · where they clash · A or B?
|
||||
|
||||
**Who decides what.** You decide what only you can see the evidence for: whether a \
|
||||
finding holds, what a measurement says, whether a record is right, whether the work is \
|
||||
done, and anything you can settle by reading or measuring. Decide, act, and say why, so \
|
||||
they can overrule it; "your call" on one of these hands them a question they cannot see \
|
||||
into. Evenly balanced evidence is still yours — say which way you went and what would \
|
||||
change your mind. They decide direction: what matters most, what the thing should be, a \
|
||||
trade-off only they can price, anything they will live with afterwards — and every act \
|
||||
that is hard to undo or reaches outside the work.
|
||||
|
||||
When an option rests on existing behaviour, say whose call that was — theirs (cite it) \
|
||||
or a past session's nobody confirmed."""
|
||||
|
||||
|
||||
_PLAN = """\
|
||||
A plan put up for review, before starting:
|
||||
|
||||
- **Goal** — what done looks like, and why, in a sentence or two.
|
||||
- **Steps** — in order, each small enough to check on its own.
|
||||
- **How you will know it works** — something observable, not "it is written".
|
||||
- **Open questions** — only the ones that are theirs to answer."""
|
||||
|
||||
|
||||
SHAPES: dict[str, ReplyShape] = {s.key: s for s in (
|
||||
ReplyShape(CORE_KEY, "Every reply", "reply.report",
|
||||
"every turn — in full once per session and after a compaction, "
|
||||
"otherwise as a one-line pointer", _CORE,
|
||||
"conclusion first; the shortest reply that carries the answer; place the "
|
||||
"work; settle what only you can see and ask about direction; end with the "
|
||||
"one thing they need to do, or say there is none."),
|
||||
ReplyShape("completion", "Completion report", "work.finish",
|
||||
"when a piece of work is closed", _COMPLETION,
|
||||
"where this sits · what now works · how / why, and how it was verified · "
|
||||
"needs you (theirs to decide, and blocking) · next."),
|
||||
ReplyShape("asks", "Asking them to decide or act", "reply.ask",
|
||||
"when a structured question is put to them", _ASKS,
|
||||
"the question first and your recommendation first among the options; "
|
||||
"settle what you can see, and ask only about direction."),
|
||||
ReplyShape("plan", "A plan for review", "work.plan",
|
||||
"when a plan is opened", _PLAN,
|
||||
"goal · steps · how you will know it works · open questions."),
|
||||
)}
|
||||
|
||||
|
||||
def core() -> ReplyShape:
|
||||
return SHAPES[CORE_KEY]
|
||||
|
||||
|
||||
def for_moments(names: list[str]) -> list[ReplyShape]:
|
||||
"""The slices that ride these moments, in catalog order. The core is not a
|
||||
slice: it rides every turn, whatever moment the turn reaches."""
|
||||
wanted = set(names)
|
||||
return [s for s in SHAPES.values() if s.key != CORE_KEY and s.moment in wanted]
|
||||
|
||||
|
||||
def catalog() -> dict:
|
||||
"""The shapes as plain data, for the MCP tool and the REST door alike."""
|
||||
return {
|
||||
"shapes": [
|
||||
{"key": s.key, "title": s.title, "moment": s.moment,
|
||||
"means": moments.MOMENTS[s.moment].means,
|
||||
"delivered": s.delivered, "text": s.text}
|
||||
for s in SHAPES.values()
|
||||
],
|
||||
"total": len(SHAPES),
|
||||
}
|
||||
|
||||
|
||||
# ── Delivery: what a door puts in front of the agent ────────────────────
|
||||
|
||||
# Said once, in the full form: the default is a floor an operator can raise.
|
||||
_FULL_HEAD = ("Reply shape · {title} — the default shape for {what}. Where an operator "
|
||||
"preference shown beside it differs, the preference is what they asked for.")
|
||||
_POINTER = ("Reply shape · {title}, shown in full earlier this session "
|
||||
"(list_reply_shapes has it): {reminder}")
|
||||
|
||||
FULL = "full"
|
||||
POINTER = "pointer"
|
||||
|
||||
|
||||
def render(shape: ReplyShape, *, full: bool) -> str:
|
||||
"""The shape as a door delivers it: in full the first time a session sees
|
||||
it, as its one-line reminder after that."""
|
||||
if not full:
|
||||
return _POINTER.format(title=shape.title, reminder=shape.reminder)
|
||||
what = moments.MOMENTS[shape.moment].means
|
||||
return _FULL_HEAD.format(title=shape.title, what=what) + "\n\n" + shape.text
|
||||
|
||||
|
||||
def deliver(shapes: list[ReplyShape], seen: set[str] | frozenset[str]) -> tuple[list[str], dict[str, str]]:
|
||||
"""Render these shapes against what the session was already shown.
|
||||
|
||||
Returns the blocks and `{key: form}` — every shape delivered and whether
|
||||
it went out in full or as a pointer. The keys sent in full are the ones a
|
||||
door's ledger appends; a door with no ledger passes `seen` empty and
|
||||
every shape goes out in full.
|
||||
"""
|
||||
blocks, forms = [], {}
|
||||
for shape in shapes:
|
||||
full = shape.key not in seen
|
||||
blocks.append(render(shape, full=full))
|
||||
forms[shape.key] = FULL if full else POINTER
|
||||
return blocks, forms
|
||||
|
||||
|
||||
def parse_seen(raw: str | None) -> frozenset[str]:
|
||||
"""A ledger's comma-separated keys, keeping only shapes that exist."""
|
||||
return frozenset(k for k in (p.strip() for p in (raw or "").split(",")) if k in SHAPES)
|
||||
|
||||
|
||||
async def record_delivery(user_id: int | None, forms: dict[str, str], *, via: str) -> None:
|
||||
"""One `reply_shape` row per delivery: which shapes, in which form, by which
|
||||
door. Fire-and-forget — telemetry never takes a shape away from the reader.
|
||||
|
||||
In AppLog beside 409's `report_check`, so the two readings of one reply —
|
||||
was the shape delivered, did the reply carry it — sit in one place.
|
||||
"""
|
||||
if not forms:
|
||||
return
|
||||
try:
|
||||
from scribe.models import async_session
|
||||
from scribe.models.app_log import AppLog
|
||||
|
||||
async with async_session() as session:
|
||||
session.add(AppLog(
|
||||
category="plugin", user_id=user_id, action="reply_shape",
|
||||
details=json.dumps({"shapes": forms, "via": via}),
|
||||
))
|
||||
await session.commit()
|
||||
except Exception: # noqa: BLE001 - observation never breaks the observed
|
||||
logger.debug("reply shape delivery not recorded", exc_info=True)
|
||||
|
||||
|
||||
# ── The operator's view (milestone 500 step 5) ──────────────────────────
|
||||
|
||||
# The window the Settings view counts deliveries over — the same 30 days the
|
||||
# moments panel beside it reads, so the two count the same stretch of work.
|
||||
OVERVIEW_DAYS = 30
|
||||
|
||||
|
||||
async def delivery_counts(user_id: int, *, days: int = OVERVIEW_DAYS) -> dict[str, dict[str, int]]:
|
||||
"""How often each shape went out, in each form, over the last `days`.
|
||||
|
||||
Read from the `reply_shape` rows `record_delivery` writes — one per
|
||||
delivery, naming each shape and whether it went whole or as its reminder.
|
||||
A row whose details do not parse is skipped rather than guessed at.
|
||||
"""
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
from sqlalchemy import select
|
||||
|
||||
from scribe.models import async_session
|
||||
from scribe.models.app_log import AppLog
|
||||
|
||||
since = datetime.now(timezone.utc) - timedelta(days=days)
|
||||
counts: dict[str, dict[str, int]] = {k: {FULL: 0, POINTER: 0} for k in SHAPES}
|
||||
async with async_session() as session:
|
||||
rows = (await session.execute(
|
||||
select(AppLog.details).where(
|
||||
AppLog.category == "plugin", AppLog.action == "reply_shape",
|
||||
AppLog.user_id == user_id, AppLog.created_at >= since,
|
||||
)
|
||||
)).scalars().all()
|
||||
for raw in rows:
|
||||
try:
|
||||
forms = json.loads(raw or "{}").get("shapes") or {}
|
||||
except (ValueError, AttributeError):
|
||||
continue
|
||||
for key, form in forms.items() if isinstance(forms, dict) else ():
|
||||
if key in counts and form in counts[key]:
|
||||
counts[key][form] += 1
|
||||
return counts
|
||||
|
||||
|
||||
async def overview(user_id: int, *, days: int = OVERVIEW_DAYS) -> dict:
|
||||
"""The Settings view's read: every shape, the operator's preferences that
|
||||
adjust it, and how often it was delivered.
|
||||
|
||||
A preference adjusts a shape by being MOUNTED on the shape's moment — the
|
||||
same lookup that delivers it beside the shape — so what this lists is
|
||||
exactly what arrives with it. Global homes only: a project's own
|
||||
preferences are that project's, and its rules tab lists them.
|
||||
|
||||
The counts are telemetry and fail open to `deliveries_failed` with "—" in
|
||||
the view; the preference lookup is the substance and is allowed to fail
|
||||
the request, so the view shows an error instead of "no preferences".
|
||||
"""
|
||||
from scribe.services import rulebooks
|
||||
|
||||
data = catalog()
|
||||
try:
|
||||
counts: dict[str, dict[str, int]] | None = await delivery_counts(user_id, days=days)
|
||||
except Exception: # noqa: BLE001 - a missing count is shown as unknown
|
||||
logger.warning("reply shape delivery counts failed", exc_info=True)
|
||||
counts = None
|
||||
for row in data["shapes"]:
|
||||
mounted = await rulebooks.rules_on_moments(user_id, [row["moment"]])
|
||||
row["preferences"] = [
|
||||
{"id": rule.id, "title": rule.title, "statement": rule.statement}
|
||||
for rule, _at in mounted if rule.kind == "preference"
|
||||
]
|
||||
row["deliveries"] = counts.get(row["key"]) if counts is not None else None
|
||||
data["days"] = days
|
||||
data["deliveries_failed"] = counts is None
|
||||
return data
|
||||
@@ -1,38 +1,60 @@
|
||||
"""The report-shape check: what the plugin's Stop hook found, and what it says.
|
||||
"""The report-shape check: does a reply that closed a task carry the completion
|
||||
sections, and what is it sent back with when it does not.
|
||||
|
||||
WHY THIS EXISTS (milestone 409 step 5)
|
||||
|
||||
Everything that helps an agent write a readable completion report arrives
|
||||
BEFORE the reply is written. A client's Stop hook is the one moment the
|
||||
finished reply exists, so it checks that a reply closing a task carries the
|
||||
completion sections (where the work sits, what needs the operator, what comes
|
||||
next), and reports what it found here. Two jobs live on this side:
|
||||
BEFORE the reply is written. The Stop hook is the one moment the finished reply
|
||||
exists, so a reply that closed a task is checked for the completion sections
|
||||
(where the work sits, what needs the operator, what comes next). Two jobs:
|
||||
|
||||
- RECORDING the outcome, so the rate of `blocked` among checked replies is a
|
||||
number milestone 409's last step can read rather than an impression.
|
||||
- OWNING THE WORDS the agent is sent back with. A hook carries timing and
|
||||
transport only (plugin/PACKAGING.md); guidance text comes from the server,
|
||||
so a second client's hook gets the same instruction by calling the same
|
||||
endpoint, and the wording changes in one place.
|
||||
number rather than an impression (`category=plugin, action=report_check`).
|
||||
- OWNING THE WORDS the agent is sent back with, so every client's hook says
|
||||
the same thing (plugin/PACKAGING.md: hooks carry timing and transport).
|
||||
|
||||
ONE END-OF-TURN REQUEST (milestone 500 step 4). This used to be a Stop hook of
|
||||
its own that checked the sections in shell and reported here; the reply-moment
|
||||
check sent the same reply to `/reply-rules` a moment later. The hook now sends
|
||||
the reply once, with the ids of the tasks the turn closed, and the section
|
||||
check runs here beside the reply hold. The patterns are the ones the shell
|
||||
used, matched on the words that carry the meaning rather than exact headings,
|
||||
so a shape's wording can change without breaking this.
|
||||
|
||||
app_logs rather than a table of its own: one small event with a JSON detail is
|
||||
what that table holds, it already has retention and an admin viewer, and
|
||||
nothing here needs a join. If the numbers earn a readout, that is the moment to
|
||||
decide whether they earn a table.
|
||||
nothing here needs a join.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
|
||||
from scribe.models import async_session
|
||||
from scribe.models.app_log import AppLog
|
||||
from scribe.services import reply_shapes
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
OUTCOMES = ("passed", "blocked", "passed_after_rewrite", "missing_after_rewrite")
|
||||
|
||||
# The sections a hook may name as missing, in the order the reason lists them.
|
||||
# Anything else a client sends is dropped rather than echoed into an
|
||||
# instruction the agent will follow.
|
||||
SECTIONS = ("where it sits", "needs you", "next")
|
||||
# The sections, in the order the reason lists them, each with what counts as
|
||||
# having it. A bare id ("closed #41") does not place the work: naming a record
|
||||
# by id AND title, or a step position, is the shape — the title is what spares
|
||||
# the reader a lookup. "Needs you: nothing" counts; it is an answer.
|
||||
SECTION_PATTERNS: dict[str, re.Pattern] = {
|
||||
"where it sits": re.compile(
|
||||
r'(#[0-9]+|milestone [0-9]+|task [0-9]+)[*_`]*\s*[*_`]*["“]|step [0-9]+ of [0-9]+', re.I),
|
||||
"needs you": re.compile(
|
||||
r"needs? (from )?you|nothing (is )?needed from you|your (call|decision)", re.I),
|
||||
"next": re.compile(r"\bnext\b", re.I),
|
||||
}
|
||||
SECTIONS = tuple(SECTION_PATTERNS)
|
||||
|
||||
|
||||
def missing_sections(reply: str) -> list[str]:
|
||||
return [name for name, pat in SECTION_PATTERNS.items() if not pat.search(reply or "")]
|
||||
|
||||
|
||||
def known_sections(missing: list[str]) -> list[str]:
|
||||
@@ -43,16 +65,17 @@ def known_sections(missing: list[str]) -> list[str]:
|
||||
def block_reason(missing: list[str]) -> str:
|
||||
"""What the agent is told when its completion report is sent back.
|
||||
|
||||
Names what is missing and points at the reporting-back skill for the shape
|
||||
rather than restating it — the skill owns the shape (decision #4027).
|
||||
Names what is missing and carries the completion shape's one line rather
|
||||
than the whole shape: the shape itself was delivered when the task closed,
|
||||
and `list_reply_shapes` has it in full.
|
||||
"""
|
||||
listed = ", ".join(known_sections(missing)) or "the completion sections"
|
||||
shape = reply_shapes.SHAPES["completion"]
|
||||
return (
|
||||
f"This turn closed a Scribe task, and the reply that ends it is missing: {listed}. "
|
||||
"The operator reads this reply to find out where the work stands. Rewrite it as a "
|
||||
"completion report (the reporting-back skill has the shape): where it sits — the task "
|
||||
"or milestone by id and title, from `placement` — what now works, what needs them "
|
||||
"(or \"nothing\"), and what comes next."
|
||||
"The person the work is for reads this reply to find out where it stands. Rewrite it "
|
||||
f"as a completion report — {shape.reminder} Take where it sits from `placement`, the "
|
||||
"task or plan by id and title (`list_reply_shapes` has the shape in full)."
|
||||
)
|
||||
|
||||
|
||||
@@ -78,3 +101,32 @@ async def record_report_check(
|
||||
details=json.dumps(details),
|
||||
))
|
||||
await session.commit()
|
||||
|
||||
|
||||
async def check_reply(
|
||||
user_id: int, reply: str, *, task_ids: list[int], rewrite: bool,
|
||||
project_id: int | None = None,
|
||||
) -> dict:
|
||||
"""Check a reply that closed tasks, record the outcome, and say whether it is held.
|
||||
|
||||
`rewrite` is the reply written after an earlier hold: it is recorded and
|
||||
never held again, so a hold costs one turn at most.
|
||||
|
||||
Returns {"outcome", "reason"} — `reason` empty unless the reply is held.
|
||||
A hold happens only when its record was written: an outcome the numbers
|
||||
cannot see is not one this check may act on, so a recording failure
|
||||
degrades to no hold rather than to an unmeasured one.
|
||||
"""
|
||||
missing = missing_sections(reply)
|
||||
if rewrite:
|
||||
outcome = "missing_after_rewrite" if missing else "passed_after_rewrite"
|
||||
else:
|
||||
outcome = "blocked" if missing else "passed"
|
||||
try:
|
||||
await record_report_check(user_id, outcome, missing=missing, task_ids=task_ids,
|
||||
project_id=project_id)
|
||||
except Exception: # noqa: BLE001 - an unrecorded check never holds a reply
|
||||
logger.warning("report check not recorded", exc_info=True)
|
||||
return {"outcome": "", "reason": ""}
|
||||
return {"outcome": outcome,
|
||||
"reason": block_reason(missing) if outcome == "blocked" else ""}
|
||||
|
||||
@@ -117,7 +117,6 @@ _RESCORERS = {
|
||||
"pre_tool_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
|
||||
"reply_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
|
||||
"prompt_rule": lambda u, q, p: _rescore_rules(u, q, p, None),
|
||||
"report_preference": lambda u, q, p: _rescore_rules(u, q, p, "preference"),
|
||||
}
|
||||
|
||||
# The embedding table each surface's re-scorer reads. A migration from a corpus
|
||||
@@ -131,7 +130,6 @@ _CORPUS = {
|
||||
"pre_tool_rule": RuleEmbedding,
|
||||
"reply_rule": RuleEmbedding,
|
||||
"prompt_rule": RuleEmbedding,
|
||||
"report_preference": RuleEmbedding,
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -421,10 +421,6 @@ class Declared:
|
||||
quiet_because: str = ""
|
||||
"""Set when silence over an active window is correct, saying why (#2475)."""
|
||||
|
||||
fixed_query: bool = False
|
||||
"""The arm always searches the same string, so its decline rate is 0% or
|
||||
100% and `cannot_decline` says nothing about it."""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RankedSource:
|
||||
@@ -523,37 +519,6 @@ WRITE_PATH_RULE = RuleArm(
|
||||
),
|
||||
declared=Declared("rules that may govern the file being written"),
|
||||
)
|
||||
# The completion report's preferences (milestone 409 step 4): a FIXED query,
|
||||
# preferences only, read by update_task as records rather than as lines. Its
|
||||
# query is `reply_preferences.COMPLETION_QUERY`.
|
||||
REPORT_PREFERENCE = RuleArm(
|
||||
"report_preference", band=False, compact_tail=False, checkpoint=False,
|
||||
preference_slot=False, kind="preference",
|
||||
tuning=Surface(
|
||||
name="report_preference",
|
||||
floor_key="kb_reportpref_threshold",
|
||||
floor_default=0.72,
|
||||
budget_key="kb_reportpref_top_k",
|
||||
budget_default=3,
|
||||
# THE ONE FIXED QUERY, and the reason this arm behaves unlike the rest.
|
||||
# The others score something that varies per call; this one scores a
|
||||
# constant string, so its top score for a given corpus is also a
|
||||
# constant. A floor a hair above that constant is not a quiet arm, it
|
||||
# is a dead one, and no amount of traffic will ever reveal it — which
|
||||
# is precisely how this arm spent 69 calls declining the same record.
|
||||
asks="a fixed question about how to lay out a completion report",
|
||||
over="preferences",
|
||||
fires="when a task finishes",
|
||||
),
|
||||
# `fixed_query`: COMPLETION_QUERY is a module constant, so this arm's top
|
||||
# score is the same number on every call — measured at 0.791 across 45
|
||||
# consecutive calls, with p10, p50, p90, min and max all identical. Five
|
||||
# equal percentiles is the tell.
|
||||
declared=Declared(
|
||||
"the fixed question asked when a task finishes: how should this report read",
|
||||
fixed_query=True,
|
||||
),
|
||||
)
|
||||
# The backstop for every arm that ran earlier in the turn and missed: the
|
||||
# finished reply against every rule's trigger. Its floor IS its stop bar
|
||||
# (the reply_rule surface), so it is passed as both.
|
||||
@@ -581,7 +546,7 @@ REPLY_RULE = RuleArm(
|
||||
),
|
||||
)
|
||||
RULE_ARMS: tuple[RuleArm, ...] = (
|
||||
WRITE_PATH_RULE, PRE_TOOL_RULE, PROMPT_RULE, REPORT_PREFERENCE, REPLY_RULE,
|
||||
WRITE_PATH_RULE, PRE_TOOL_RULE, PROMPT_RULE, REPLY_RULE,
|
||||
)
|
||||
|
||||
PREFERENCE_SLOT_SOURCE = "preference_slot"
|
||||
|
||||
@@ -94,29 +94,6 @@ class Point:
|
||||
|
||||
quiet_because: str = ""
|
||||
|
||||
fixed_query: bool = False
|
||||
"""Whether this arm always searches the SAME query string.
|
||||
|
||||
THE #3497 GUARD, ONE STEP OVER. `logs_unconditionally` below exists
|
||||
because a warning computed over a LOGGING property read as a ranking
|
||||
problem. This field exists because a warning computed over a QUERY-SHAPE
|
||||
property does the same thing.
|
||||
|
||||
An arm with a fixed query scores against one constant. Its decline rate is
|
||||
therefore 0% or 100% and nothing in between — which of the two depends
|
||||
only on whether the bar sits below or above that single number. So "never
|
||||
returned nothing" says nothing at all about whether a floor is applied,
|
||||
and `cannot_decline` — whose whole remedy is "check that it applies its
|
||||
floor" — is uninformative here and skips these arms.
|
||||
|
||||
What IS informative for them is the mirror image, and `reply_preferences`
|
||||
names it in its own docstring: every call returning nothing means the bar
|
||||
sits above the constant, no traffic will ever move it, and the arm is
|
||||
dead. That has happened — 69 consecutive declines at 0.0006 under the bar
|
||||
(see `retrieval_surfaces`) — so it gets its own warning rather than
|
||||
inheriting one written for arms whose score can vary.
|
||||
"""
|
||||
|
||||
logs_unconditionally: bool = True
|
||||
"""Whether this arm writes a row even when it returns NOTHING.
|
||||
|
||||
@@ -144,8 +121,7 @@ POINTS: dict[str, Point] = dict([
|
||||
# ambient or pulled.
|
||||
*(_p(spec.source, UNBIDDEN, spec.declared.what,
|
||||
expects_traffic=not spec.declared.quiet_because,
|
||||
quiet_because=spec.declared.quiet_because,
|
||||
fixed_query=spec.declared.fixed_query)
|
||||
quiet_because=spec.declared.quiet_because)
|
||||
for spec in RANKED),
|
||||
# A LOOKUP, not a ranker (#4796): a record the operator named by number
|
||||
# in the message, fetched by id. No score and no bar, so it writes no
|
||||
|
||||
@@ -5,9 +5,9 @@ WHY THIS EXISTS
|
||||
Six push arms each carried their own loose copy of the same shape: a settings
|
||||
key, a default, and a limit that was usually a module constant nobody could
|
||||
change. The read-and-clamp was written out separately in `plugin_context` (twice
|
||||
over, for auto-inject and the write path), in three rule arms, and again as
|
||||
`reply_preferences._threshold`. That was survivable while the numbers were
|
||||
shipped constants an operator occasionally edited.
|
||||
over, for auto-inject and the write path), in three rule arms, and again in
|
||||
the completion-report arm (since retired, milestone 500). That was survivable
|
||||
while the numbers were shipped constants an operator occasionally edited.
|
||||
|
||||
It stops being survivable once the numbers are meant to MOVE. The operator's
|
||||
decision for this step:
|
||||
|
||||
@@ -591,22 +591,12 @@ def _compute_warnings(sources: dict, usage: dict, rule_usage: dict,
|
||||
# arms once recorded only their hits, so their decline count was
|
||||
# structurally zero and this warning would have fired on a LOGGING
|
||||
# defect while pointing the reader at the threshold.
|
||||
#
|
||||
# FOUR NOW. A FIXED-QUERY arm is exempt for the same reason one step
|
||||
# over (#4232): it scores against one constant, so its decline rate is
|
||||
# 0% or 100% and never in between, and which one it is depends only on
|
||||
# where the bar sits relative to that single number. "Never returned
|
||||
# nothing" is then not evidence about the floor — it is arithmetic —
|
||||
# and this warning's own remedy, "check that it applies its floor",
|
||||
# cannot be answered from it. Those arms get `fixed_query_never_clears`
|
||||
# below, which asks the question that IS answerable for them.
|
||||
if (
|
||||
calls >= min_calls
|
||||
and (b.get("zero_result_calls") or 0) == 0
|
||||
and point is not None
|
||||
and point.kind == UNBIDDEN
|
||||
and point.logs_unconditionally
|
||||
and not point.fixed_query
|
||||
):
|
||||
out.append(_warn(
|
||||
"cannot_decline",
|
||||
@@ -617,46 +607,6 @@ def _compute_warnings(sources: dict, usage: dict, rule_usage: dict,
|
||||
source=name, calls=calls, zero_result_calls=0,
|
||||
))
|
||||
|
||||
# ── A fixed-query arm that never clears its bar ──────────────────
|
||||
#
|
||||
# The mirror image of `cannot_decline`, and the state that actually
|
||||
# threatens these arms. `reply_preferences` names it in its own
|
||||
# docstring: "a fixed query makes this arm's score a constant and a
|
||||
# floor a hair above it produces a dead arm no amount of traffic will
|
||||
# ever reveal."
|
||||
#
|
||||
# For an arm whose score can vary, a window of all-empty calls is
|
||||
# ordinary — it means nothing matched, which is an answer. For one
|
||||
# whose score is a constant it means the bar is above that constant,
|
||||
# and no volume of further calls will ever produce a different result.
|
||||
# The arm is not quiet; it is switched off, and nothing else in this
|
||||
# readout would say so.
|
||||
#
|
||||
# It has happened: `report_preference` logged 69 consecutive declines
|
||||
# at 0.0006 under the bar. Note what that incident also proves — the
|
||||
# fix is NOT automatically to lower the floor. Reading the refused
|
||||
# record showed the refusal was correct, so this warning sends the
|
||||
# reader to `near_miss_samples` rather than to the dial.
|
||||
if (
|
||||
calls >= min_calls
|
||||
and point is not None
|
||||
and point.fixed_query
|
||||
and point.logs_unconditionally
|
||||
and (b.get("zero_result_calls") or 0) == calls
|
||||
):
|
||||
out.append(_warn(
|
||||
"fixed_query_never_clears",
|
||||
f"{calls} calls, every one of them empty — and this arm always "
|
||||
f"searches the same query, so its score is a constant. That "
|
||||
f"means the bar sits above it and no amount of further traffic "
|
||||
f"will change the result: the arm is off, not quiet. Read the "
|
||||
f"record it refused (`near_miss_samples`) before touching the "
|
||||
f"floor — the last time this arm sat here, every percentile "
|
||||
f"said lower it and the refused record showed the refusal was "
|
||||
f"right.",
|
||||
source=name, calls=calls, zero_result_calls=calls,
|
||||
))
|
||||
|
||||
# ── Band hugs its floor ──────────────────────────────────────────
|
||||
#
|
||||
# Read on p10, the WEAKEST tenth of what the arm returned. If even
|
||||
|
||||
@@ -102,6 +102,20 @@ def _no_supersession():
|
||||
yield
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _no_reply_shape_log():
|
||||
"""Stub the reply-shape delivery row (milestone 500 step 3).
|
||||
|
||||
Autouse for _no_system_labels' reason: every turn's retrieval, every
|
||||
moment request and every Scribe tool that reaches a shaped moment writes
|
||||
one AppLog row, and those paths are tested without a database.
|
||||
tests/test_reply_shapes.py binds the real function at import time,
|
||||
before this patch runs.
|
||||
"""
|
||||
with patch("scribe.services.reply_shapes.record_delivery", AsyncMock()):
|
||||
yield
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _no_system_labels():
|
||||
"""Stub the menu's "which System is each line about?" lookup (#4364).
|
||||
|
||||
@@ -498,6 +498,22 @@ async def rule_row(rule_id: int):
|
||||
return await s.get(Rule, rule_id)
|
||||
|
||||
|
||||
# Software-only vocabulary. Scribe's product text speaks for any work a person
|
||||
# drives through an agent — configuration, infrastructure, documents — so a
|
||||
# catalog or a shipped shape that reads as software-only is wrong on most of
|
||||
# the installs it ships to. The actions particular to one kind of work belong
|
||||
# in data (mappings, rulebooks, preferences), not in the product's words.
|
||||
DEV_ONLY = (r"\bCI\b", r"\bcommit", r"\bpull request", r"\bcode\b", r"\btest",
|
||||
r"\bgit\b", r"\brepo(s|sitor\w*)?\b", r"\bbranch", r"\bcompil", r"\bbuild\b")
|
||||
|
||||
|
||||
def dev_only_hits(text: str) -> list[str]:
|
||||
"""The DEV_ONLY patterns this text contains, case-insensitively."""
|
||||
import re
|
||||
|
||||
return [w for w in DEV_ONLY if re.search(w, text, re.IGNORECASE)]
|
||||
|
||||
|
||||
def skill_text(name: str) -> str:
|
||||
"""Everything a bundled skill states: its SKILL.md, then each reference file.
|
||||
|
||||
|
||||
@@ -71,6 +71,14 @@ def _live_session_context_source() -> str:
|
||||
return src[start:start + 1 + nxt.start()] if nxt else src[start:]
|
||||
|
||||
|
||||
def _delivered_shapes() -> str:
|
||||
"""Every reply shape as a session receives it in full. Rendered rather than
|
||||
read as source: the header a shape arrives under is guidance too."""
|
||||
from scribe.services import reply_shapes
|
||||
|
||||
return "\n\n".join(reply_shapes.render(s, full=True) for s in reply_shapes.SHAPES.values())
|
||||
|
||||
|
||||
def delivered_surfaces() -> dict[str, str]:
|
||||
"""Every surface a session receives as guidance, by label.
|
||||
|
||||
@@ -81,6 +89,8 @@ def delivered_surfaces() -> dict[str, str]:
|
||||
- `static` — the Claude Code adapter's static session context
|
||||
- `commands` — the Claude Code adapter's slash commands
|
||||
- `live` — the live session context the server builds
|
||||
- `shapes` — the reply shapes the server delivers at their moments,
|
||||
in the full form a session first sees (milestone 500)
|
||||
"""
|
||||
server = (ROOT / "src/scribe/mcp/server.py").read_text()
|
||||
match = re.search(r'_INSTRUCTIONS = """(.*?)"""', server, re.S)
|
||||
@@ -91,6 +101,7 @@ def delivered_surfaces() -> dict[str, str]:
|
||||
"static": (ROOT / "plugin/hooks/scribe_static_context.md").read_text(),
|
||||
"commands": "".join(p.read_text() for p in sorted((ROOT / "plugin/commands").glob("*.md"))),
|
||||
"live": _live_session_context_source(),
|
||||
"shapes": _delivered_shapes(),
|
||||
}
|
||||
# A skill is SKILL.md plus the reference files it links (#4398): one
|
||||
# owner, however many files it is split across.
|
||||
@@ -243,9 +254,15 @@ TOPICS: tuple[Topic, ...] = (
|
||||
"you are the one who knows what the code you just wrote is"),
|
||||
Topic("report back where the work stands", "skill:reporting-back", ("reporting-back", "placement"),
|
||||
"take the placement from the record"),
|
||||
Topic("the operator's own reply shapes come first", "skill:reporting-back",
|
||||
("reply_preferences", 'content_type="rule"'),
|
||||
"the operator's own shapes come first"),
|
||||
# Milestone 500: the default shapes are server product, delivered at their
|
||||
# moments with the operator's preferences mounted beside them — so the
|
||||
# rule that a preference wins is stated where the shape arrives.
|
||||
Topic("a preference beside a delivered shape is what the operator asked for", "shapes",
|
||||
("preference", "reply shape"),
|
||||
"where an operator preference shown beside it differs"),
|
||||
Topic("every reply leads with its conclusion and stays short", "shapes",
|
||||
("conclusion first", "the shortest reply that carries the answer", "what can go"),
|
||||
"length is work handed back to the reader"),
|
||||
# Milestone 409 step 7: the sections were being FILLED rather than chosen,
|
||||
# so a reply could satisfy every heading and still be unreadable. These
|
||||
# three are the discipline around the scaffold, not the scaffold itself.
|
||||
|
||||
@@ -21,6 +21,7 @@ guard here can fail (rule 167).
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
@@ -156,6 +157,33 @@ def test_a_value_survives_the_round_trip_the_hooks_actually_make(value):
|
||||
assert got.rstrip("\n") == value.rstrip("\n")
|
||||
|
||||
|
||||
ESCAPED = ["a · b", "an em dash — here", "a face 😀 here", "plain ascii"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("locale", [None, "C", "C.UTF-8"])
|
||||
@pytest.mark.parametrize("value", ESCAPED)
|
||||
def test_an_escaped_character_decodes_to_utf8_in_any_locale(value, locale):
|
||||
"""The server's JSON escapes every non-ASCII character (`\\u00b7`), and a
|
||||
hook may run with no locale at all. The round trip above sends raw UTF-8
|
||||
and never reaches the escape path; this one sends what the server sends,
|
||||
from an environment with nothing but PATH."""
|
||||
need_tools("bash", "awk")
|
||||
env = {"PATH": os.environ["PATH"]}
|
||||
if locale:
|
||||
env["LC_ALL"] = locale
|
||||
body = json.dumps(value)[1:-1]
|
||||
r = subprocess.run(["bash", "-c", f'. "{DEFS}"; scribe_json_unescape'],
|
||||
input=body.encode(), capture_output=True, env=env)
|
||||
assert r.returncode == 0, r.stderr.decode()
|
||||
assert r.stdout == (value + "\n").encode()
|
||||
|
||||
|
||||
def test_the_utf8_guard_can_fail():
|
||||
"""A lone byte where the encoded character belongs is what the old
|
||||
decoder wrote; the comparison above has to tell them apart."""
|
||||
assert "·\n".encode() != b"\xb7\n"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Reading: the shell side.
|
||||
|
||||
|
||||
@@ -53,30 +53,104 @@ def test_one_incident_written_up_three_times_is_not_a_recurring_situation():
|
||||
assert links_svc.convergence_group(same) is None
|
||||
|
||||
|
||||
def _note(lid, *, title="t", project=None, sources=()):
|
||||
data = {"what": title, "when_to_apply": "a CI run overran"}
|
||||
# ── The bar is relative to each lesson's own register (#5193) ──────────────
|
||||
|
||||
|
||||
def test_too_little_background_says_nothing_stands_out():
|
||||
few = {i: 0.6 for i in range(links_svc.CONVERGENCE_MIN_BACKGROUND - 1)}
|
||||
assert links_svc.standout(few) == {}
|
||||
|
||||
|
||||
def test_a_background_that_never_varies_says_nothing_stands_out():
|
||||
assert links_svc.standout({i: 0.7 for i in range(20)}) == {}
|
||||
|
||||
|
||||
def test_standing_out_is_measured_in_the_registers_own_spread():
|
||||
"""The same neighbour score stands out in a tight register and not in a
|
||||
loose one — the bar moves with the corpus, not with a constant."""
|
||||
tight = {i: 0.70 + 0.01 * (i % 5) for i in range(20)} | {99: 0.80}
|
||||
loose = {i: 0.60 + 0.06 * (i % 5) for i in range(20)} | {99: 0.80}
|
||||
assert links_svc.standout(tight)[99] >= links_svc.CONVERGENCE_STANDOUT
|
||||
assert links_svc.standout(loose)[99] < links_svc.CONVERGENCE_STANDOUT
|
||||
|
||||
|
||||
def test_a_hub_near_two_lessons_that_are_not_near_each_other_is_no_clique():
|
||||
up = links_svc.CONVERGENCE_STANDOUT + 1
|
||||
standouts = {1: {2: up, 3: up}, 2: {1: up, 3: 0.0}, 3: {1: up, 2: 0.0}}
|
||||
assert links_svc.converging(1, standouts, [2, 3]) == [1, 2]
|
||||
|
||||
|
||||
def test_resemblance_must_run_both_ways():
|
||||
up = links_svc.CONVERGENCE_STANDOUT + 1
|
||||
standouts = {1: {2: up, 3: up}, 2: {1: 0.0, 3: up}, 3: {1: up, 2: up}}
|
||||
assert links_svc.converging(1, standouts, [2, 3]) == [1, 3]
|
||||
|
||||
|
||||
def _note(lid, *, title=None, project=None, sources=()):
|
||||
data = {"what": title or f"lesson {lid}", "when_to_apply": "a CI run overran"}
|
||||
if sources:
|
||||
data["taught_by"] = list(sources)
|
||||
return SimpleNamespace(
|
||||
id=lid, title=title, note_type="lesson", body="", data=data,
|
||||
id=lid, title=title or f"lesson {lid}", note_type="lesson", body="", data=data,
|
||||
project_id=project, arose_from_id=None, tags=[],
|
||||
created_at=None, updated_at=None,
|
||||
)
|
||||
|
||||
|
||||
# A register: lessons 1-3 resemble each other well above a background of
|
||||
# filler lessons 10-29, which all sit in one flat band — the shape every
|
||||
# lesson-to-lesson score has, because lessons share one register.
|
||||
_TRIO = {1, 2, 3}
|
||||
_FILLER = range(10, 30)
|
||||
_NOTES = {i: _note(i, sources=[100 + i]) for i in (*_TRIO, *_FILLER)}
|
||||
|
||||
|
||||
def _sim(a, b):
|
||||
if a in _TRIO and b in _TRIO:
|
||||
return 0.82
|
||||
return 0.66 + 0.01 * ((a + b) % 5)
|
||||
|
||||
|
||||
def _search_over(sim=_sim):
|
||||
async def search(user_id, query, *, exclude_ids, **_kw):
|
||||
(me,) = exclude_ids
|
||||
return sorted(((sim(me, i), n) for i, n in _NOTES.items() if i != me),
|
||||
key=lambda pair: -pair[0])
|
||||
return search
|
||||
|
||||
|
||||
def _run(no_rule, sim=_sim):
|
||||
return (
|
||||
patch.object(lessons_svc, "get_lesson", AsyncMock(return_value=_NOTES[1])),
|
||||
patch("scribe.services.embeddings.semantic_search_notes", _search_over(sim)),
|
||||
patch.object(links_svc, "_no_rule_ids",
|
||||
AsyncMock(side_effect=lambda ids: {i for i in ids if i in no_rule})),
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_lessons_that_stand_out_together_are_named():
|
||||
a, b, c = _run(no_rule=set(_NOTES))
|
||||
with a, b, c:
|
||||
group = await _REAL(7, 1)
|
||||
assert [m["id"] for m in group["lessons"]] == [1, 2, 3]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_large_no_rule_pool_in_one_flat_band_names_nothing():
|
||||
"""The #5193 failure: every lesson answered "no rule", every score above
|
||||
a fixed bar, and a group the size of the pool. Nothing stands out of a
|
||||
flat band, so nothing is named however large the pool grows."""
|
||||
a, b, c = _run(no_rule=set(_NOTES), sim=lambda x, y: 0.66 + 0.01 * ((x + y) % 5))
|
||||
with a, b, c:
|
||||
assert await _REAL(7, 1) is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_only_lessons_answered_no_rule_join_the_group():
|
||||
found = [(0.8, _note(2, sources=[11])), (0.7, _note(3, sources=[12])),
|
||||
(0.7, _note(4, sources=[13]))]
|
||||
with patch.object(lessons_svc, "get_lesson", AsyncMock(return_value=_note(1, sources=[10]))), \
|
||||
patch("scribe.services.embeddings.semantic_search_notes", AsyncMock(return_value=found)), \
|
||||
patch.object(links_svc, "_no_rule_ids", AsyncMock(return_value={2})):
|
||||
assert await _REAL(7, 1) is None # only #2 answered: a pair
|
||||
with patch.object(lessons_svc, "get_lesson", AsyncMock(return_value=_note(1, sources=[10]))), \
|
||||
patch("scribe.services.embeddings.semantic_search_notes", AsyncMock(return_value=found)), \
|
||||
patch.object(links_svc, "_no_rule_ids", AsyncMock(return_value={2, 4})):
|
||||
group = await _REAL(7, 1)
|
||||
assert [m["id"] for m in group["lessons"]] == [1, 2, 4]
|
||||
a, b, c = _run(no_rule={1, 2}) # #3 has a rule: a pair is no group
|
||||
with a, b, c:
|
||||
assert await _REAL(7, 1) is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
||||
@@ -4,8 +4,10 @@ The in-band half of milestone 409 step 3: the skill and static context only
|
||||
exist in the Claude Code plugin, and a tool response reaches every MCP client
|
||||
at the moment a piece of work closes. Pinned on the response, not the wording.
|
||||
|
||||
Step 4 adds the operator's own completion-report preferences beside the cue,
|
||||
retrieved at that same moment — and only then, and only when there are any.
|
||||
The operator's own completion-report preferences used to ride here too, from
|
||||
a fixed-question search (409 step 4). Milestone 500 retired that: they are
|
||||
mounted on finishing work and arrive in `moment_rules` beside the delivered
|
||||
reply shape, so this response carries no key of its own for them.
|
||||
"""
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
@@ -13,49 +15,37 @@ import pytest
|
||||
|
||||
pytestmark = pytest.mark.usefixtures("_bind_user")
|
||||
|
||||
_PREF = {"id": 41, "title": "Say how it was checked", "statement": "…", "kind": "preference"}
|
||||
|
||||
|
||||
async def _update(prefs=None, owed=None, **kwargs):
|
||||
async def _update(owed=None, **kwargs):
|
||||
from scribe.mcp.tools.tasks import update_task
|
||||
|
||||
note = MagicMock(id=5, user_id=7, project_id=3, started_at="2026-10-06T09:00:00+00:00")
|
||||
note.to_dict.return_value = {"id": 5}
|
||||
lookup = AsyncMock(return_value=list(prefs or []))
|
||||
with patch("scribe.mcp.tools.tasks.notes_svc.update_note", AsyncMock(return_value=note)), \
|
||||
patch("scribe.mcp.tools.tasks.systems_tools.attach_systems", AsyncMock()), \
|
||||
patch("scribe.mcp.tools.tasks.placement_svc.attach_placement", AsyncMock()), \
|
||||
patch("scribe.mcp.tools.tasks.family_adoption_svc.owed_since",
|
||||
AsyncMock(return_value=list(owed or []))), \
|
||||
patch("scribe.mcp.tools.tasks.reply_prefs_svc.completion_preferences", lookup):
|
||||
return await update_task(task_id=5, **kwargs), lookup
|
||||
AsyncMock(return_value=list(owed or []))):
|
||||
return await update_task(task_id=5, **kwargs)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("status", ["done", "cancelled"])
|
||||
async def test_closing_a_task_carries_the_cue(status):
|
||||
out, _ = await _update(status=status)
|
||||
out = await _update(status=status)
|
||||
assert "placement" in out["report_back"] and "needs them" in out["report_back"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("kwargs", [{"status": "in_progress"}, {"status": "todo"}, {"body": "more notes"}])
|
||||
async def test_other_updates_do_not(kwargs):
|
||||
out, lookup = await _update(prefs=[_PREF], **kwargs)
|
||||
assert "report_back" not in out and "reply_preferences" not in out
|
||||
lookup.assert_not_awaited()
|
||||
out = await _update(**kwargs)
|
||||
assert "report_back" not in out
|
||||
|
||||
|
||||
async def test_closing_hands_back_the_operators_report_preferences():
|
||||
out, lookup = await _update(prefs=[_PREF], status="done")
|
||||
assert out["reply_preferences"] == [_PREF]
|
||||
# The cue says the key is there, so it never arrives unexplained.
|
||||
assert "reply_preferences" in out["report_back"]
|
||||
lookup.assert_awaited_once_with(7, project_id=3)
|
||||
|
||||
|
||||
async def test_no_preferences_means_no_key_and_the_plain_cue():
|
||||
async def test_closing_carries_the_plain_cue_and_no_preference_key_of_its_own():
|
||||
"""The retired key does not come back beside the moment delivery."""
|
||||
from scribe.mcp.tools.tasks import REPORT_BACK_CUE
|
||||
|
||||
out, _ = await _update(prefs=[], status="done")
|
||||
out = await _update(status="done")
|
||||
assert "reply_preferences" not in out
|
||||
assert out["report_back"] == REPORT_BACK_CUE
|
||||
|
||||
@@ -69,11 +59,11 @@ async def test_closing_names_the_owed_adoptions_filed_while_the_task_was_open():
|
||||
into a project during this task comes back for the report to name."""
|
||||
from scribe.services.family_adoption import OWED_CUE
|
||||
|
||||
out, _ = await _update(owed=[_OWED], status="done")
|
||||
out = await _update(owed=[_OWED], status="done")
|
||||
assert out["family_owed"] == [_OWED]
|
||||
assert out["report_back"].endswith(OWED_CUE) and "family_owed" in out["report_back"]
|
||||
|
||||
|
||||
async def test_no_owed_adoptions_means_no_key():
|
||||
out, _ = await _update(owed=[], status="done")
|
||||
out = await _update(owed=[], status="done")
|
||||
assert "family_owed" not in out
|
||||
|
||||
@@ -0,0 +1,128 @@
|
||||
"""The DRY close-out measure (#4745).
|
||||
|
||||
Stdlib-only and kept in scripts/ so any project can run a scratch copy of it,
|
||||
so these tests import it by path rather than as a package.
|
||||
|
||||
What matters is the after-only list: it is the part of the measure that finds
|
||||
something a pass could not, so it must name a copy the change CREATED and stay
|
||||
quiet about one that was already there. A list that repeats old duplication
|
||||
every time is one nobody reads.
|
||||
"""
|
||||
import importlib.util
|
||||
import pathlib
|
||||
import shutil
|
||||
import subprocess
|
||||
|
||||
import pytest
|
||||
|
||||
_PATH = pathlib.Path(__file__).resolve().parents[1] / "scripts" / "measure_duplication.py"
|
||||
_spec = importlib.util.spec_from_file_location("measure_duplication", _PATH)
|
||||
dup = importlib.util.module_from_spec(_spec)
|
||||
_spec.loader.exec_module(dup)
|
||||
|
||||
|
||||
BODY = """
|
||||
def load(user_id, note_id):
|
||||
# a comment the measure ignores
|
||||
if not allowed(user_id, note_id):
|
||||
raise ValueError("note {} not found".format(note_id))
|
||||
with session() as s:
|
||||
row = s.get(Row, note_id)
|
||||
if row is None:
|
||||
raise ValueError("not a row")
|
||||
return row
|
||||
"""
|
||||
|
||||
|
||||
def _files(**texts):
|
||||
return [(name, text.encode()) for name, text in texts.items()]
|
||||
|
||||
|
||||
def test_comments_blanks_and_bare_punctuation_are_not_code():
|
||||
lines = dup.significant_lines('x = 1\n\n// note\n# note\n });\n-- sql note\ny = "two"\n')
|
||||
assert [t for _, t in lines] == ["x = 1", 'y = "S"']
|
||||
assert [n for n, _ in lines] == [1, 7]
|
||||
|
||||
|
||||
def test_a_preprocessor_line_is_code_and_a_hash_comment_is_not():
|
||||
lines = dup.significant_lines("#include <x.h>\n# a comment\n#!/bin/sh\n")
|
||||
assert [t for _, t in lines] == ["#include <x.h>"]
|
||||
|
||||
|
||||
def test_copies_that_differ_only_in_their_strings_are_one_copy():
|
||||
other = BODY.replace('"not a row"', '"no such row"')
|
||||
result = dup.measure(_files(**{"a.py": BODY, "b.py": other}), [], [], window=6)
|
||||
py = result["py"]
|
||||
assert py["dup_windows"] > 0
|
||||
assert py["dup_lines"] == py["lines"]
|
||||
assert py["share"] == 1.0
|
||||
|
||||
|
||||
def test_unrelated_files_measure_zero():
|
||||
other = "\n".join(f"v{i} = compute({i})" for i in range(20))
|
||||
result = dup.measure(_files(**{"a.py": BODY, "b.py": other}), [], [], window=6)
|
||||
assert result["py"]["dup_lines"] == 0
|
||||
|
||||
|
||||
def test_the_first_matching_group_claims_a_file():
|
||||
groups = [("tests", ["tests/*"]), ("py", ["*.py"])]
|
||||
assert dup.group_of("tests/test_x.py", groups) == "tests"
|
||||
assert dup.group_of("src/x.py", groups) == "py"
|
||||
assert dup.group_of("README.md", groups) is None
|
||||
|
||||
|
||||
def test_without_groups_only_source_extensions_are_measured():
|
||||
assert dup.group_of("src/x.go", []) == "go"
|
||||
assert dup.group_of("docs/x.md", []) is None
|
||||
assert dup.group_of("Makefile", []) is None
|
||||
|
||||
|
||||
def test_excluded_paths_are_not_measured():
|
||||
result = dup.measure(
|
||||
_files(**{"a.py": BODY, "gen/b.py": BODY}), [], ["gen/*"], window=6,
|
||||
)
|
||||
assert result["py"]["dup_lines"] == 0
|
||||
|
||||
|
||||
def test_a_copy_the_change_created_is_listed_after_only():
|
||||
before = dup.measure(_files(**{"a.py": BODY, "c.py": "z = 1\n"}), [], [], window=6)
|
||||
after = dup.measure(_files(**{"a.py": BODY, "c.py": BODY}), [], [], window=6)
|
||||
fresh = dup.only_after(before, after)
|
||||
assert [e["files"] for e in fresh["py"]] == [["a.py", "c.py"]]
|
||||
assert fresh["py"][0]["at"] == ["a.py:2", "c.py:2"]
|
||||
|
||||
|
||||
def test_a_copy_that_was_already_there_is_not_listed():
|
||||
both = _files(**{"a.py": BODY, "b.py": BODY})
|
||||
before = dup.measure(both, [], [], window=6)
|
||||
after = dup.measure(both, [], [], window=6)
|
||||
assert dup.only_after(before, after) == {}
|
||||
|
||||
|
||||
def test_the_report_shows_before_and_after_side_by_side():
|
||||
before = dup.measure(_files(**{"a.py": BODY}), [], [], window=6)
|
||||
after = dup.measure(_files(**{"a.py": BODY, "b.py": BODY}), [], [], window=6)
|
||||
text = dup.report(before, after, dup.only_after(before, after), show=5)
|
||||
assert "0.0% → 100.0%" in text
|
||||
assert "a.py:2, b.py:2" in text
|
||||
|
||||
|
||||
@pytest.mark.skipif(shutil.which("git") is None, reason="needs git")
|
||||
def test_a_revision_is_read_without_a_checkout(tmp_path, capsys):
|
||||
def git(*args):
|
||||
subprocess.run(
|
||||
["git", "-C", str(tmp_path), "-c", "user.name=t", "-c", "user.email=t@t", *args],
|
||||
check=True, capture_output=True,
|
||||
)
|
||||
|
||||
git("init", "-q")
|
||||
(tmp_path / "a.py").write_text(BODY)
|
||||
git("add", "a.py")
|
||||
git("commit", "-qm", "one copy")
|
||||
(tmp_path / "b.py").write_text(BODY)
|
||||
git("add", "b.py")
|
||||
|
||||
assert dup.main(["--repo", str(tmp_path), "--before", "HEAD"]) == 0
|
||||
out = capsys.readouterr().out
|
||||
assert "Duplicated only after: 1 file set(s)" in out
|
||||
assert "a.py:2, b.py:2" in out
|
||||
+130
-14
@@ -185,18 +185,23 @@ async def _reachable(mounted, mappings=()):
|
||||
return await md.reachable_tools(1)
|
||||
|
||||
|
||||
async def test_an_install_with_nothing_mounted_keeps_the_hook_off_the_wire():
|
||||
listing = AsyncMock()
|
||||
with patch.object(rulebooks, "mounted_moments", AsyncMock(return_value=set())), \
|
||||
patch.object(moment_actions, "list_mappings", listing):
|
||||
assert await md.reachable_tools(1) == []
|
||||
listing.assert_not_awaited()
|
||||
# The tools the shipped defaults map onto a moment that carries a default reply
|
||||
# shape (milestone 500): closing work, a structured question, opening a plan.
|
||||
SHAPED = ["askuserquestion", "create_milestone", "enterplanmode", "exitplanmode",
|
||||
"skill", "start_planning", "update_milestone", "update_task"]
|
||||
|
||||
|
||||
async def test_only_the_tools_whose_moments_carry_a_mount_are_listed():
|
||||
assert await _reachable({"work.deliver"}) == ["bash", "skill"]
|
||||
assert await _reachable({"work.finish"}) == ["skill", "update_milestone", "update_task"]
|
||||
assert await _reachable({"skill.release"}) == ["skill"]
|
||||
async def test_an_install_with_nothing_mounted_still_asks_about_the_shaped_moments():
|
||||
"""The reply shapes are product, so every install has them: the hook asks
|
||||
about the acts that bring one, and about nothing else."""
|
||||
assert await _reachable(set()) == SHAPED
|
||||
|
||||
|
||||
async def test_only_the_tools_whose_moments_carry_a_mount_or_a_shape_are_listed():
|
||||
assert await _reachable({"work.deliver"}) == sorted([*SHAPED, "bash"])
|
||||
assert await _reachable({"work.finish"}) == SHAPED
|
||||
assert await _reachable({"skill.release"}) == SHAPED
|
||||
assert "bash" not in await _reachable(set())
|
||||
|
||||
|
||||
async def test_the_skill_loader_counts_whenever_anything_is_mounted():
|
||||
@@ -232,7 +237,8 @@ async def test_the_route_reads_the_event_and_the_ledger():
|
||||
with patch.object(routes.moment_delivery_svc, "deliver_for_act", deliver):
|
||||
resp = await routes.moment.__wrapped__()
|
||||
body = await resp.get_json()
|
||||
assert body == {"context": "a line", "rule_ids": [5], "moments": ["work.deliver"]}
|
||||
assert body == {"context": "a line", "rule_ids": [5], "moments": ["work.deliver"],
|
||||
"shape_keys": []}
|
||||
args, kw = deliver.await_args
|
||||
assert args == (7, "Bash", {"command": "git push"})
|
||||
assert kw == {"project_id": 3, "exclude": frozenset({9}), "held": frozenset({4})}
|
||||
@@ -245,7 +251,8 @@ async def test_the_route_answers_an_empty_event_with_nothing():
|
||||
async with app.test_request_context("/api/plugin/moment", method="POST", json={}):
|
||||
g.user = SimpleNamespace(id=7)
|
||||
resp = await routes.moment.__wrapped__()
|
||||
assert await resp.get_json() == {"context": "", "rule_ids": [], "moments": []}
|
||||
assert await resp.get_json() == {"context": "", "rule_ids": [], "moments": [],
|
||||
"shape_keys": []}
|
||||
|
||||
|
||||
def test_both_routes_are_on_the_app():
|
||||
@@ -273,13 +280,15 @@ def test_the_hook_and_the_routes_agree_on_every_name():
|
||||
assert "scribe_rules_live" in src and "scribe_rules_append" in src
|
||||
# The SHARED ledger, not one of its own.
|
||||
assert '"${TMPDIR:-/tmp}/scribe-priorart"' in src and ".rules.ids" in src
|
||||
for field in (".tools", ".rule_ids", ".context"):
|
||||
for field in (".tools", ".rule_ids", ".context", ".shape_keys"):
|
||||
assert f"'{field}'" in src
|
||||
assert "shapes_seen" in src and "scribe_shapes_append" in src
|
||||
|
||||
from scribe.routes import plugin as routes
|
||||
|
||||
route = inspect.getsource(routes.moment)
|
||||
for name in ("exclude_rule_ids", "held_rule_ids", "tool_name", "tool_input"):
|
||||
for name in ("exclude_rule_ids", "held_rule_ids", "tool_name", "tool_input",
|
||||
"shapes_seen", "shape_keys"):
|
||||
assert name in route
|
||||
|
||||
|
||||
@@ -380,3 +389,110 @@ def test_every_scribe_tool_the_defaults_name_attaches_its_moment_rules():
|
||||
f"attach its mounted rules — a client without the plugin would never "
|
||||
f"receive them"
|
||||
)
|
||||
|
||||
|
||||
# ── reply shapes ride the moments (milestone 500 step 3) ────────────────
|
||||
|
||||
from scribe.services import reply_shapes # noqa: E402
|
||||
|
||||
FINISH = [{"moment": "work.finish", "tool": "update_task", "match": "status=done", "via": "default"}]
|
||||
ASK = [{"moment": "reply.ask", "tool": "AskUserQuestion", "match": "", "via": "default"}]
|
||||
|
||||
|
||||
def test_an_act_brings_the_shape_for_the_reply_it_comes_before():
|
||||
blocks, forms = md.shapes_for_act(FINISH)
|
||||
assert forms == {"completion": reply_shapes.FULL}
|
||||
assert reply_shapes.SHAPES["completion"].text in blocks[0]
|
||||
_blocks, forms = md.shapes_for_act(ASK, frozenset({"asks"}))
|
||||
assert forms == {"asks": reply_shapes.POINTER}
|
||||
|
||||
|
||||
def test_an_act_at_an_unshaped_moment_brings_no_shape():
|
||||
assert md.shapes_for_act(PUSH) == ([], {})
|
||||
|
||||
|
||||
async def test_a_turn_carries_the_core_in_full_until_the_session_holds_it():
|
||||
with patch.object(md, "deliver_moments", AsyncMock(return_value=rp.RuleResult())):
|
||||
first = await md.deliver_for_turn(1)
|
||||
later = await md.deliver_for_turn(1, seen=frozenset({"core"}))
|
||||
assert first["shape_forms"] == {"core": reply_shapes.FULL}
|
||||
assert reply_shapes.core().text in first["context"]
|
||||
assert later["shape_forms"] == {"core": reply_shapes.POINTER}
|
||||
assert reply_shapes.core().text not in later["context"]
|
||||
assert reply_shapes.core().reminder in later["context"]
|
||||
|
||||
|
||||
async def test_a_turn_brings_what_is_mounted_on_reply_report_after_the_core():
|
||||
deliver = AsyncMock(return_value=rp.RuleResult(lines=["Preference that may apply"], rule_ids=[173]))
|
||||
with patch.object(md, "deliver_moments", deliver):
|
||||
out = await md.deliver_for_turn(1, project_id=2, exclude=frozenset({9}), held=frozenset({4}))
|
||||
assert out["context"].index(reply_shapes.core().title) < out["context"].index("Preference")
|
||||
assert out["rule_ids"] == [173]
|
||||
reached = deliver.await_args.args[1]
|
||||
assert [hit["moment"] for hit in reached] == ["reply.report"]
|
||||
assert deliver.await_args.kwargs == {"project_id": 2, "exclude": frozenset({9}),
|
||||
"held": frozenset({4})}
|
||||
|
||||
|
||||
async def test_a_failed_mount_lookup_still_delivers_the_core():
|
||||
with patch.object(md, "deliver_moments", AsyncMock(side_effect=RuntimeError("db down"))):
|
||||
out = await md.deliver_for_turn(1)
|
||||
assert reply_shapes.core().text in out["context"] and out["rule_ids"] == []
|
||||
|
||||
|
||||
async def test_a_turn_records_its_delivery():
|
||||
with patch.object(md, "deliver_moments", AsyncMock(return_value=rp.RuleResult())):
|
||||
await md.deliver_for_turn(1, seen=frozenset({"core"}))
|
||||
reply_shapes.record_delivery.assert_awaited_with(1, {"core": reply_shapes.POINTER}, via="turn")
|
||||
|
||||
|
||||
async def test_a_task_closed_through_the_tool_carries_the_completion_shape():
|
||||
data = {"id": 40, "status": "done", "project_id": 2}
|
||||
out = await _with(_stub_db([]), lambda: md.attach_moment_rules(
|
||||
1, "update_task", {"status": "done"}, data,
|
||||
))
|
||||
assert reply_shapes.SHAPES["completion"].text in out["reply_shape"]
|
||||
assert "moment_rules" not in out # nothing mounted, still shaped
|
||||
|
||||
|
||||
async def test_an_unshaped_act_through_a_tool_carries_no_shape():
|
||||
out = await _with(_stub_db([]), lambda: md.attach_moment_rules(
|
||||
1, "create_note", {"project_id": 2}, {"id": 3},
|
||||
))
|
||||
assert "reply_shape" not in out
|
||||
|
||||
|
||||
async def test_the_route_puts_the_shape_ahead_of_the_rules_and_returns_its_key():
|
||||
from scribe.routes import plugin as routes
|
||||
|
||||
deliver = AsyncMock(return_value=(ASK, rp.RuleResult(lines=["a line"], rule_ids=[5])))
|
||||
app = Quart(__name__)
|
||||
event = {"tool_name": "AskUserQuestion", "tool_input": {}}
|
||||
async with app.test_request_context("/api/plugin/moment", method="POST", json=event):
|
||||
g.user = SimpleNamespace(id=7)
|
||||
with patch.object(routes.moment_delivery_svc, "deliver_for_act", deliver):
|
||||
body = await (await routes.moment.__wrapped__()).get_json()
|
||||
assert body["shape_keys"] == ["asks"]
|
||||
assert body["context"].index("Who decides what") < body["context"].index("a line")
|
||||
|
||||
async with app.test_request_context("/api/plugin/moment", method="POST", json=event,
|
||||
query_string={"shapes_seen": "asks"}):
|
||||
g.user = SimpleNamespace(id=7)
|
||||
with patch.object(routes.moment_delivery_svc, "deliver_for_act", deliver):
|
||||
body = await (await routes.moment.__wrapped__()).get_json()
|
||||
assert body["shape_keys"] == []
|
||||
assert "Who decides what" not in body["context"]
|
||||
|
||||
|
||||
def test_the_hook_carries_the_shape_ledger_between_calls(tmp_path):
|
||||
replies = {"/api/plugin/moment-tools": b'{"tools":["askuserquestion"]}',
|
||||
"/api/plugin/moment": json.dumps(
|
||||
{"context": "Reply shape", "rule_ids": [], "moments": ["reply.ask"],
|
||||
"shape_keys": ["asks"]}).encode()}
|
||||
with http_sink(by_path=replies) as (port, seen):
|
||||
_hook(tmp_path, port, tool="AskUserQuestion")
|
||||
_hook(tmp_path, port, tool="AskUserQuestion")
|
||||
moment_calls = [e for e in seen if e["_path"] == "/api/plugin/moment"]
|
||||
assert "shapes_seen" not in moment_calls[0]
|
||||
assert moment_calls[1]["shapes_seen"] == ["asks"]
|
||||
|
||||
|
||||
+6
-13
@@ -10,7 +10,7 @@ import re
|
||||
import pytest
|
||||
|
||||
from scribe.services import moments
|
||||
from tests.helpers import FakeMCP
|
||||
from tests.helpers import FakeMCP, dev_only_hits
|
||||
|
||||
_NAME = re.compile(r"^[a-z]+\.[a-z]+$")
|
||||
|
||||
@@ -38,26 +38,18 @@ def test_the_skill_family_is_not_a_catalog_entry():
|
||||
assert not any(k.startswith(moments.SKILL_PREFIX) for k in moments.MOMENTS)
|
||||
|
||||
|
||||
# Software-only vocabulary, the same guard the completion query carries. The
|
||||
# moments are named for any work a person drives through an agent; the actions
|
||||
# particular to one kind of work belong in the mappings, not in the meaning.
|
||||
_DEV_ONLY = (r"\bCI\b", r"\bcommit", r"\bpull request", r"\bcode\b", r"\btest",
|
||||
r"\bgit\b", r"\brepo(s|sitor\w*)?\b", r"\bbranch", r"\bcompil", r"\bbuild\b")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("field", ["means", "reached_by"])
|
||||
def test_the_catalog_assumes_no_particular_domain(field):
|
||||
found = {
|
||||
m.name: hits
|
||||
for m in [*moments.MOMENTS.values(), moments.SKILL_FAMILY]
|
||||
if (hits := [w for w in _DEV_ONLY
|
||||
if re.search(w, getattr(m, field), re.IGNORECASE)])
|
||||
if (hits := dev_only_hits(getattr(m, field)))
|
||||
}
|
||||
assert not found, f"software-only vocabulary in `{field}`: {found}"
|
||||
|
||||
|
||||
def test_the_guard_can_fail():
|
||||
assert re.search(_DEV_ONLY[1], "after the commit lands", re.IGNORECASE)
|
||||
assert dev_only_hits("after the commit lands")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name", list(moments.MOMENTS))
|
||||
@@ -116,11 +108,12 @@ def test_the_tools_are_registered_and_classified():
|
||||
mcp = FakeMCP()
|
||||
tool.register(mcp)
|
||||
assert mcp.names == [
|
||||
"list_moments", "map_action", "unmap_action",
|
||||
"list_moments", "list_reply_shapes", "map_action", "unmap_action",
|
||||
"rules_to_mount", "propose_rule_moments", "rule_moment_proposals",
|
||||
"judge_rule_moments", "rule_misfired",
|
||||
]
|
||||
assert {"list_moments", "rules_to_mount", "rule_moment_proposals"} <= _READ_ONLY_TOOLS
|
||||
assert {"list_moments", "list_reply_shapes",
|
||||
"rules_to_mount", "rule_moment_proposals"} <= _READ_ONLY_TOOLS
|
||||
# A confirm mounts a rule, so a read key must not reach it.
|
||||
assert {"map_action", "unmap_action",
|
||||
"propose_rule_moments", "judge_rule_moments", "rule_misfired"} <= _WRITE_TOOLS
|
||||
|
||||
@@ -132,6 +132,11 @@ def test_the_scaffold_itself_is_untouched():
|
||||
for section in ("where this sits", "what now works", "how / why",
|
||||
"needs you", "next"):
|
||||
assert section in text, f"completion-report section {section!r} is gone"
|
||||
for kind in ("completion", "finding", "blocked / failed", "progress",
|
||||
# The kinds live in the delivered shapes since milestone 500 step 4; the
|
||||
# skill keeps the reasoning and the worked completion report.
|
||||
from scribe.services import reply_shapes
|
||||
|
||||
shapes = " ".join(s.title + " " + s.text for s in reply_shapes.SHAPES.values()).lower()
|
||||
for kind in ("completion", "finding", "blocked", "progress",
|
||||
"decision", "clarification", "handoff", "approval", "conflict"):
|
||||
assert kind in text, f"reply kind {kind!r} is gone"
|
||||
assert kind in shapes, f"reply kind {kind!r} is gone"
|
||||
|
||||
+103
-1
@@ -197,7 +197,8 @@ async def test_the_route_passes_the_reply_and_all_three_ledgers():
|
||||
with patch.object(routes.moment_delivery_svc, "reply_hold", hold):
|
||||
resp = await routes.reply_rules.__wrapped__()
|
||||
body = await resp.get_json()
|
||||
assert body == {"reason": "Held.", "rule_ids": [11], "moments": ["reply.report"]}
|
||||
assert body == {"reason": "Held.", "rule_ids": [11], "moments": ["reply.report"],
|
||||
"report_check": ""}
|
||||
assert hold.await_args.args == (7, "done")
|
||||
assert hold.await_args.kwargs == {
|
||||
"project_id": 3, "exclude": frozenset({5}), "held": frozenset({4}),
|
||||
@@ -205,6 +206,55 @@ async def test_the_route_passes_the_reply_and_all_three_ledgers():
|
||||
}
|
||||
|
||||
|
||||
async def _reply_route(body, *, hold_reason="", checked=None):
|
||||
from scribe.routes import plugin as routes
|
||||
|
||||
hold = AsyncMock(return_value={"reason": hold_reason, "rule_ids": [11] if hold_reason else [],
|
||||
"moments": ["reply.report"]})
|
||||
check = AsyncMock(return_value=checked or {"outcome": "passed", "reason": ""})
|
||||
app = Quart(__name__)
|
||||
async with app.test_request_context("/api/plugin/reply-rules", method="POST", json=body,
|
||||
query_string={"project_id": "3"}):
|
||||
g.user = type("U", (), {"id": 7})()
|
||||
with patch.object(routes.moment_delivery_svc, "reply_hold", hold), \
|
||||
patch.object(routes.report_check_svc, "check_reply", check):
|
||||
resp = await routes.reply_rules.__wrapped__()
|
||||
return await resp.get_json(), hold, check
|
||||
|
||||
|
||||
async def test_a_turn_that_closed_nothing_is_not_section_checked():
|
||||
_body, _hold, check = await _reply_route({"reply": "done"})
|
||||
check.assert_not_awaited()
|
||||
|
||||
|
||||
async def test_a_closing_turn_is_section_checked_and_both_holds_fold_into_one_reason():
|
||||
"""One end-of-turn request (milestone 500 step 4): the section check and
|
||||
the rule hold answer together, in one `reason`."""
|
||||
body, _hold, check = await _reply_route(
|
||||
{"reply": "done", "closed": 1, "closed_task_ids": [41, True, "x"], "rewrite": False},
|
||||
hold_reason="RULE HOLD", checked={"outcome": "blocked", "reason": "SECTIONS"},
|
||||
)
|
||||
assert body["reason"] == "SECTIONS\n\nRULE HOLD"
|
||||
assert body["report_check"] == "blocked"
|
||||
# JSON `true` is not task 1, and a string is not an id.
|
||||
assert check.await_args.kwargs == {"task_ids": [41], "rewrite": False, "project_id": 3}
|
||||
|
||||
|
||||
async def test_a_task_created_already_done_is_checked_without_an_id():
|
||||
_body, _hold, check = await _reply_route({"reply": "done", "closed": 1, "closed_task_ids": []})
|
||||
assert check.await_args.kwargs["task_ids"] == []
|
||||
|
||||
|
||||
async def test_the_rewrite_is_recorded_and_held_by_nothing():
|
||||
body, hold, check = await _reply_route(
|
||||
{"reply": "done", "closed": 1, "closed_task_ids": [41], "rewrite": True},
|
||||
checked={"outcome": "passed_after_rewrite", "reason": ""},
|
||||
)
|
||||
assert check.await_args.kwargs["rewrite"] is True
|
||||
hold.assert_not_awaited()
|
||||
assert body["reason"] == "" and body["report_check"] == "passed_after_rewrite"
|
||||
|
||||
|
||||
# ── the hook ────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
@@ -251,6 +301,58 @@ def test_the_hook_blocks_in_the_servers_words_and_records_the_hold(tmp_path):
|
||||
assert seen[1]["exclude_rule_ids"] == ["11"]
|
||||
|
||||
|
||||
def _closing_transcript(tmp_path, *replies):
|
||||
lines = [
|
||||
{"type": "user", "message": {"role": "user", "content": "please finish it"}},
|
||||
{"type": "assistant", "message": {"content": [
|
||||
{"type": "tool_use", "id": "toolu_1", "name": "mcp__plugin_scribe_scribe__update_task",
|
||||
"input": {"task_id": 41, "status": "done"}}]}},
|
||||
{"type": "user", "message": {"content": [
|
||||
{"type": "tool_result", "tool_use_id": "toolu_1", "is_error": False, "content": "{}"}]}},
|
||||
] + [{"type": "assistant", "message": {"content": [{"type": "text", "text": r}]}} for r in replies]
|
||||
path = tmp_path / "t.jsonl"
|
||||
path.write_text("\n".join(json.dumps(x, separators=(",", ":")) for x in lines) + "\n")
|
||||
return path
|
||||
|
||||
|
||||
SECTION_HOLD = json.dumps({"reason": "Rewrite as a completion report.", "rule_ids": [],
|
||||
"moments": ["reply.report"], "report_check": "blocked"}).encode()
|
||||
|
||||
|
||||
def test_a_turn_that_closed_nothing_sends_no_close(tmp_path):
|
||||
t = _transcript(tmp_path, "Done.")
|
||||
with http_sink(by_path={"/api/plugin/reply-rules": QUIET}) as (port, seen):
|
||||
_run(tmp_path, port, t)
|
||||
assert "closed" not in json.loads(seen[0]["_body"])
|
||||
|
||||
|
||||
def test_a_section_hold_blocks_once_then_the_rewrite_is_reported_and_never_held(tmp_path):
|
||||
"""The completion-section check, folded into the one Stop request: the
|
||||
first stop is held in the server's words, the rewrite goes back once with
|
||||
`rewrite: true` so its outcome is recorded, and nothing after that is sent."""
|
||||
t = _closing_transcript(tmp_path, "All done, pushed it.")
|
||||
with http_sink(by_path={"/api/plugin/reply-rules": SECTION_HOLD}) as (port, seen):
|
||||
out = json.loads(_run(tmp_path, port, t))
|
||||
assert out == {"decision": "block", "reason": "Rewrite as a completion report."}
|
||||
first = json.loads(seen[0]["_body"])
|
||||
assert (first["closed"], first["closed_task_ids"], first["rewrite"]) == (1, [41], False)
|
||||
|
||||
rewritten = _closing_transcript(tmp_path, "All done, pushed it.", "Where this sits: …")
|
||||
assert _run(tmp_path, port, rewritten, active=True) == ""
|
||||
assert json.loads(seen[1]["_body"])["rewrite"] is True
|
||||
# The marker is spent: a further stop in the loop sends nothing.
|
||||
assert _run(tmp_path, port, rewritten, active=True) == ""
|
||||
assert len(seen) == 2
|
||||
|
||||
|
||||
def test_a_rule_hold_on_a_closing_turn_does_not_mark_a_rewrite(tmp_path):
|
||||
t = _closing_transcript(tmp_path, "Done.")
|
||||
with http_sink(by_path={"/api/plugin/reply-rules": HELD}) as (port, seen):
|
||||
_run(tmp_path, port, t)
|
||||
assert _run(tmp_path, port, t, active=True) == ""
|
||||
assert len(seen) == 1
|
||||
|
||||
|
||||
def test_the_hook_says_nothing_when_nothing_holds(tmp_path):
|
||||
t = _transcript(tmp_path, "Done.")
|
||||
with http_sink(by_path={"/api/plugin/reply-rules": QUIET}) as (port, _seen):
|
||||
|
||||
@@ -0,0 +1,129 @@
|
||||
"""The core reply shape rides every turn (milestone 500 step 3).
|
||||
|
||||
Every turn ends in a reply, and the operator's prompt is the last point before
|
||||
it is written, so the per-turn retrieval carries the core shape ahead of
|
||||
everything else: in full the first time a session sees it, as its one-line
|
||||
reminder after that, and in full again once a compaction has swept the
|
||||
ledger. What is pinned here is that round trip across the shell/Python seam —
|
||||
the route's order and keys, and the hook's ledger.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import inspect
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
from quart import Quart, g
|
||||
|
||||
from scribe.services import reply_shapes
|
||||
from tests.helpers import http_sink, need_tools
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
HOOK = ROOT / "plugin" / "hooks" / "scribe_autoinject.sh"
|
||||
SESSION_START = ROOT / "plugin" / "hooks" / "scribe_session_context.sh"
|
||||
|
||||
|
||||
# ── the route ────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
async def _retrieve(query_string, turn_context="CORE SHAPE"):
|
||||
from scribe.routes import plugin as routes
|
||||
|
||||
turn = AsyncMock(return_value={"context": turn_context, "rule_ids": [173],
|
||||
"shape_forms": {"core": reply_shapes.FULL}})
|
||||
app = Quart(__name__)
|
||||
async with app.test_request_context("/api/plugin/retrieve", query_string=query_string):
|
||||
g.user = SimpleNamespace(id=7)
|
||||
with patch.object(routes.plugin_ctx_svc, "build_prompt_rule_hint",
|
||||
AsyncMock(return_value={"context": "RULE LINE", "rule_ids": [11],
|
||||
"shown_rule_ids": [11]})), \
|
||||
patch.object(routes.plugin_ctx_svc, "build_autoinject_hint",
|
||||
AsyncMock(return_value={"context": "NOTES MENU", "note_ids": []})), \
|
||||
patch.object(routes.lesson_rules_svc, "co_surfaced", AsyncMock(return_value="")), \
|
||||
patch.object(routes.moment_delivery_svc, "deliver_for_turn", turn):
|
||||
resp = await inspect.unwrap(routes.autoinject_retrieve)()
|
||||
return await resp.get_json(), turn
|
||||
|
||||
|
||||
async def test_the_turns_shape_leads_the_payload_and_its_key_comes_back():
|
||||
body, _turn = await _retrieve({"q": "fix the flaky thing"})
|
||||
ctx = body["context"]
|
||||
assert ctx.index("CORE SHAPE") < ctx.index("RULE LINE") < ctx.index("NOTES MENU")
|
||||
assert body["shape_keys"] == ["core"]
|
||||
# Both arms' fresh rules reach the shared ledger.
|
||||
assert body["rule_ids"] == [11, 173]
|
||||
|
||||
|
||||
async def test_the_route_hands_the_ledger_and_the_rules_just_named_to_the_turn():
|
||||
_body, turn = await _retrieve({"q": "x", "shapes_seen": "core,bogus",
|
||||
"exclude_rule_ids": "9", "held_rule_ids": "4"})
|
||||
kw = turn.await_args.kwargs
|
||||
assert kw["seen"] == frozenset({"core"})
|
||||
# A rule the prompt arm just named is not quoted again by the turn.
|
||||
assert kw["exclude"] == frozenset({9, 11})
|
||||
assert kw["held"] == frozenset({4})
|
||||
|
||||
|
||||
# ── the hook ─────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _env(tmp_path, port):
|
||||
return {"PATH": os.environ["PATH"], "SCRIBE_URL": f"http://127.0.0.1:{port}",
|
||||
"SCRIBE_TOKEN": "t", "TMPDIR": str(tmp_path), "HOME": str(tmp_path)}
|
||||
|
||||
|
||||
def _prompt(tmp_path, port, session="s-shape"):
|
||||
need_tools("bash", "curl", "awk")
|
||||
out = subprocess.run(
|
||||
["bash", str(HOOK)],
|
||||
input=json.dumps({"session_id": session, "cwd": str(tmp_path),
|
||||
"prompt": "please fix the importer"}),
|
||||
capture_output=True, text=True, env=_env(tmp_path, port), timeout=30,
|
||||
)
|
||||
assert out.returncode == 0, out.stderr
|
||||
return out.stdout
|
||||
|
||||
|
||||
def _compact(tmp_path, session="s-shape"):
|
||||
out = subprocess.run(
|
||||
["bash", str(SESSION_START)],
|
||||
input=json.dumps({"session_id": session, "source": "compact"}),
|
||||
capture_output=True, text=True,
|
||||
env={"PATH": os.environ["PATH"], "HOME": str(tmp_path), "TMPDIR": str(tmp_path)},
|
||||
timeout=60,
|
||||
)
|
||||
assert out.returncode == 0, out.stderr
|
||||
|
||||
|
||||
REPLY = json.dumps({"context": "Reply shape · Every reply", "note_ids": [], "rule_ids": [],
|
||||
"shape_keys": ["core"]}).encode()
|
||||
|
||||
|
||||
def test_the_core_goes_whole_once_then_as_a_reminder_then_whole_after_a_compaction(tmp_path):
|
||||
with http_sink(by_path={"/api/plugin/retrieve": REPLY}) as (port, seen):
|
||||
out = _prompt(tmp_path, port)
|
||||
_prompt(tmp_path, port)
|
||||
_compact(tmp_path)
|
||||
_prompt(tmp_path, port)
|
||||
asks = [e for e in seen if e["_path"] == "/api/plugin/retrieve"]
|
||||
assert len(asks) == 3
|
||||
assert "shapes_seen" not in asks[0]
|
||||
assert asks[1]["shapes_seen"] == ["core"]
|
||||
# The compaction swept the ledger: the session no longer holds the shape.
|
||||
assert "shapes_seen" not in asks[2]
|
||||
assert "Every reply" in json.loads(out)["hookSpecificOutput"]["additionalContext"]
|
||||
|
||||
|
||||
def test_the_hook_and_the_route_agree_on_the_ledgers_names():
|
||||
"""Rule 33 across the seam: a renamed field fails silently."""
|
||||
from scribe.routes import plugin as routes
|
||||
|
||||
hook = HOOK.read_text()
|
||||
assert "shapes_seen" in hook and "'.shape_keys'" in hook
|
||||
assert "scribe_shapes_file" in hook and "scribe_shapes_append" in hook
|
||||
route = inspect.getsource(routes.autoinject_retrieve)
|
||||
assert 'request.args.get("shapes_seen")' in route and '"shape_keys"' in route
|
||||
@@ -0,0 +1,315 @@
|
||||
"""The default reply shapes and the moments they ride (milestone 500 step 1).
|
||||
|
||||
What this pins is what delivery will depend on: every shape rides a moment that
|
||||
exists, the core is small enough to pay for on every turn, no slice grows back
|
||||
into the skill it replaced, the text speaks for any kind of work, and both
|
||||
doors hand out the same shapes.
|
||||
"""
|
||||
import pytest
|
||||
|
||||
from scribe.services import moments, reply_shapes
|
||||
from tests.helpers import FakeMCP, dev_only_hits
|
||||
|
||||
# Bound before conftest's autouse stub replaces the module attribute.
|
||||
_REAL_RECORD = reply_shapes.record_delivery
|
||||
|
||||
|
||||
def test_there_is_a_core_and_at_least_one_slice():
|
||||
"""The sweeps below are vacuous over an empty catalog (rule 167)."""
|
||||
assert reply_shapes.CORE_KEY in reply_shapes.SHAPES
|
||||
assert len(reply_shapes.SHAPES) >= 2
|
||||
|
||||
|
||||
def test_every_shape_is_keyed_by_its_own_key():
|
||||
for key, shape in reply_shapes.SHAPES.items():
|
||||
assert key == shape.key
|
||||
|
||||
|
||||
@pytest.mark.parametrize("key", list(reply_shapes.SHAPES))
|
||||
def test_every_shape_rides_a_catalog_moment(key):
|
||||
"""A shape on a moment nothing reaches would never arrive — and an
|
||||
operator's preference mounted beside it would sit on a dead moment too."""
|
||||
assert reply_shapes.SHAPES[key].moment in moments.MOMENTS
|
||||
|
||||
|
||||
def test_the_core_rides_the_reply_moment():
|
||||
"""It is the reply's shape; a preference about every reply is mounted
|
||||
on `reply.report`, so that is where the core says it lives."""
|
||||
assert reply_shapes.core().moment == "reply.report"
|
||||
|
||||
|
||||
def test_no_two_slices_share_a_moment():
|
||||
"""One moment, one default. Two slices on a moment would arrive together
|
||||
and leave the reader to reconcile them."""
|
||||
slices = [s.moment for s in reply_shapes.SHAPES.values() if s.key != reply_shapes.CORE_KEY]
|
||||
assert len(slices) == len(set(slices))
|
||||
|
||||
|
||||
def test_the_core_fits_its_budget():
|
||||
"""It is paid for in every session. The budget is the ceiling, so a
|
||||
sentence added to the core has to replace one."""
|
||||
assert len(reply_shapes.core().text) <= reply_shapes.CORE_BUDGET_CHARS
|
||||
|
||||
|
||||
@pytest.mark.parametrize("key", [k for k in reply_shapes.SHAPES if k != reply_shapes.CORE_KEY])
|
||||
def test_each_slice_fits_its_budget(key):
|
||||
assert len(reply_shapes.SHAPES[key].text) <= reply_shapes.SLICE_BUDGET_CHARS
|
||||
|
||||
|
||||
def test_the_budget_guard_can_fail():
|
||||
long = "x" * (reply_shapes.CORE_BUDGET_CHARS + 1)
|
||||
assert not len(long) <= reply_shapes.CORE_BUDGET_CHARS
|
||||
|
||||
|
||||
@pytest.mark.parametrize("key", list(reply_shapes.SHAPES))
|
||||
def test_no_shape_assumes_a_particular_domain(key):
|
||||
shape = reply_shapes.SHAPES[key]
|
||||
assert not dev_only_hits(shape.title + "\n" + shape.text), key
|
||||
|
||||
|
||||
@pytest.mark.parametrize("key", list(reply_shapes.SHAPES))
|
||||
def test_every_shape_says_what_it_is_and_when_it_arrives(key):
|
||||
shape = reply_shapes.SHAPES[key]
|
||||
assert shape.title.strip() and shape.delivered.strip() and shape.text.strip()
|
||||
|
||||
|
||||
def test_the_core_carries_the_length_discipline():
|
||||
"""Milestone 409's live read: replies that had every section were still
|
||||
too long. Delivery cannot fix that; the core has to say it."""
|
||||
text = reply_shapes.core().text.lower()
|
||||
assert "shortest reply that carries the answer" in text
|
||||
assert "what can go" in text
|
||||
|
||||
|
||||
def test_who_decides_what_rides_every_reply_and_every_ask():
|
||||
"""The operator's line (milestone 500 step 2): the agent settles what
|
||||
only it can see, the operator decides direction. A one-liner every turn,
|
||||
in full where a question is about to be handed back."""
|
||||
core = reply_shapes.core().text.lower()
|
||||
asks = reply_shapes.SHAPES["asks"].text.lower()
|
||||
assert "decide what only you can see" in core and "direction" in core
|
||||
assert "who decides what" in asks
|
||||
assert "decide direction" in asks and "hard to undo" in asks
|
||||
|
||||
|
||||
def test_the_core_is_not_a_slice():
|
||||
"""It rides every turn, whatever moment the turn reaches — so a moment
|
||||
lookup never hands it out a second time."""
|
||||
every = [s.moment for s in reply_shapes.SHAPES.values()]
|
||||
assert reply_shapes.CORE_KEY not in {s.key for s in reply_shapes.for_moments(every)}
|
||||
|
||||
|
||||
def test_a_moment_brings_its_own_slice_and_nothing_else():
|
||||
got = reply_shapes.for_moments(["work.finish", "work.run"])
|
||||
assert [s.key for s in got] == ["completion"]
|
||||
assert reply_shapes.for_moments([]) == []
|
||||
|
||||
|
||||
def test_the_payload_carries_every_shape_with_its_moments_meaning():
|
||||
data = reply_shapes.catalog()
|
||||
assert [s["key"] for s in data["shapes"]] == list(reply_shapes.SHAPES)
|
||||
assert data["total"] == len(reply_shapes.SHAPES)
|
||||
for row in data["shapes"]:
|
||||
assert row["means"] == moments.MOMENTS[row["moment"]].means
|
||||
|
||||
|
||||
async def test_the_tool_returns_the_catalog():
|
||||
from scribe.mcp.tools import moments as tool
|
||||
|
||||
assert await tool.list_reply_shapes() == reply_shapes.catalog()
|
||||
|
||||
|
||||
def test_the_tool_is_registered_as_a_read():
|
||||
from scribe.mcp.server import _READ_ONLY_TOOLS
|
||||
from scribe.mcp.tools import moments as tool
|
||||
|
||||
mcp = FakeMCP()
|
||||
tool.register(mcp)
|
||||
assert "list_reply_shapes" in mcp.names
|
||||
assert "list_reply_shapes" in _READ_ONLY_TOOLS
|
||||
|
||||
|
||||
def test_both_doors_read_one_catalog():
|
||||
"""Rule 33 parity: the session and the Settings view cannot show
|
||||
different defaults."""
|
||||
from scribe.mcp.tools import moments as tool
|
||||
from scribe.routes import retrieval as routes
|
||||
|
||||
assert tool.shapes_svc is reply_shapes
|
||||
assert routes.shapes_svc is reply_shapes
|
||||
|
||||
|
||||
async def test_the_route_returns_the_operators_overview_with_a_bounded_window():
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
from quart import Quart, g
|
||||
|
||||
from scribe.routes import retrieval as routes
|
||||
|
||||
view = AsyncMock(return_value={"shapes": [], "total": 0})
|
||||
app = Quart(__name__)
|
||||
for raw, days in (("", 30), ("7", 7), ("900", 90), ("x", 30)):
|
||||
async with app.test_request_context("/api/retrieval/reply-shapes", query_string={"days": raw}):
|
||||
g.user = SimpleNamespace(id=7)
|
||||
with patch.object(routes.shapes_svc, "overview", view):
|
||||
resp = await routes.reply_shapes_route.__wrapped__()
|
||||
assert await resp.get_json() == {"shapes": [], "total": 0}
|
||||
assert view.await_args.kwargs == {"days": days}
|
||||
|
||||
|
||||
# ── the operator's view (milestone 500 step 5) ──────────────────────────
|
||||
|
||||
|
||||
async def test_the_overview_lists_the_preferences_mounted_on_each_shapes_moment():
|
||||
"""What adjusts a shape is what is mounted beside it — and only
|
||||
preferences: a rule on the same moment binds, it does not reshape."""
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
pref = SimpleNamespace(id=173, title="Problem first", statement="…", kind="preference")
|
||||
rule = SimpleNamespace(id=11, title="Definition of done", statement="…", kind="rule")
|
||||
|
||||
async def mounted(user_id, moments, project_id=None):
|
||||
return [(pref, "reply.report"), (rule, "reply.report")] if moments == ["reply.report"] else []
|
||||
|
||||
counts = {k: {"full": 0, "pointer": 0} for k in reply_shapes.SHAPES}
|
||||
counts["core"] = {"full": 2, "pointer": 9}
|
||||
with patch("scribe.services.rulebooks.rules_on_moments", mounted), \
|
||||
patch.object(reply_shapes, "delivery_counts", AsyncMock(return_value=counts)):
|
||||
data = await reply_shapes.overview(7, days=30)
|
||||
rows = {r["key"]: r for r in data["shapes"]}
|
||||
assert rows["core"]["preferences"] == [{"id": 173, "title": "Problem first", "statement": "…"}]
|
||||
assert rows["core"]["deliveries"] == {"full": 2, "pointer": 9}
|
||||
assert rows["completion"]["preferences"] == []
|
||||
assert data["days"] == 30 and data["deliveries_failed"] is False
|
||||
|
||||
|
||||
async def test_counts_that_fail_read_as_unknown_not_zero():
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
async def mounted(user_id, moments, project_id=None):
|
||||
return []
|
||||
|
||||
with patch("scribe.services.rulebooks.rules_on_moments", mounted), \
|
||||
patch.object(reply_shapes, "delivery_counts", AsyncMock(side_effect=RuntimeError("db"))):
|
||||
data = await reply_shapes.overview(7)
|
||||
assert data["deliveries_failed"] is True
|
||||
assert all(r["deliveries"] is None for r in data["shapes"])
|
||||
|
||||
|
||||
async def test_delivery_counts_tally_each_shape_by_form_and_skip_what_does_not_parse():
|
||||
import json
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
rows = [json.dumps({"shapes": {"core": "full"}, "via": "turn"}),
|
||||
json.dumps({"shapes": {"core": "pointer", "completion": "full"}, "via": "hook"}),
|
||||
json.dumps({"shapes": {"core": "pointer", "bogus": "full"}, "via": "turn"}),
|
||||
"not json", None]
|
||||
result = MagicMock()
|
||||
result.scalars.return_value.all.return_value = rows
|
||||
|
||||
class _Session:
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *exc):
|
||||
return False
|
||||
|
||||
async def execute(self, _stmt):
|
||||
return result
|
||||
|
||||
with patch("scribe.models.async_session", lambda: _Session()):
|
||||
counts = await reply_shapes.delivery_counts(7, days=30)
|
||||
assert counts["core"] == {"full": 1, "pointer": 2}
|
||||
assert counts["completion"] == {"full": 1, "pointer": 0}
|
||||
assert "bogus" not in counts
|
||||
|
||||
|
||||
# ── delivery (milestone 500 step 3) ─────────────────────────────────────
|
||||
|
||||
|
||||
@pytest.mark.parametrize("key", list(reply_shapes.SHAPES))
|
||||
def test_every_shape_has_a_one_line_reminder(key):
|
||||
shape = reply_shapes.SHAPES[key]
|
||||
assert shape.reminder.strip() and "\n" not in shape.reminder
|
||||
assert not dev_only_hits(shape.reminder), key
|
||||
|
||||
|
||||
def test_the_full_form_carries_the_text_and_says_a_preference_wins():
|
||||
core = reply_shapes.core()
|
||||
out = reply_shapes.render(core, full=True)
|
||||
assert out.endswith(core.text)
|
||||
assert core.title in out and "preference" in out
|
||||
|
||||
|
||||
def test_the_pointer_form_is_one_line_carrying_the_reminder():
|
||||
core = reply_shapes.core()
|
||||
out = reply_shapes.render(core, full=False)
|
||||
assert "\n" not in out
|
||||
assert core.reminder in out and "list_reply_shapes" in out
|
||||
assert core.text not in out
|
||||
|
||||
|
||||
def test_a_shape_the_session_holds_goes_out_as_its_pointer():
|
||||
shapes = [reply_shapes.core(), reply_shapes.SHAPES["completion"]]
|
||||
blocks, forms = reply_shapes.deliver(shapes, frozenset({"core"}))
|
||||
assert forms == {"core": reply_shapes.POINTER, "completion": reply_shapes.FULL}
|
||||
assert reply_shapes.core().text not in blocks[0]
|
||||
assert reply_shapes.SHAPES["completion"].text in blocks[1]
|
||||
|
||||
|
||||
def test_a_door_with_no_ledger_sends_everything_in_full():
|
||||
_blocks, forms = reply_shapes.deliver([reply_shapes.core()], frozenset())
|
||||
assert forms == {"core": reply_shapes.FULL}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("raw,keys", [
|
||||
("core", {"core"}),
|
||||
("core, asks,core", {"core", "asks"}),
|
||||
("", set()),
|
||||
(None, set()),
|
||||
("core,nonsense,../x", {"core"}),
|
||||
])
|
||||
def test_the_ledger_keeps_only_shapes_that_exist(raw, keys):
|
||||
assert reply_shapes.parse_seen(raw) == frozenset(keys)
|
||||
|
||||
|
||||
async def test_a_delivery_is_one_plugin_row_naming_each_shape_and_its_form():
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
added = []
|
||||
session = MagicMock()
|
||||
session.add = added.append
|
||||
|
||||
async def _commit():
|
||||
return None
|
||||
session.commit = _commit
|
||||
|
||||
class _Ctx:
|
||||
async def __aenter__(self):
|
||||
return session
|
||||
|
||||
async def __aexit__(self, *exc):
|
||||
return False
|
||||
|
||||
with patch("scribe.models.async_session", lambda: _Ctx()):
|
||||
await _REAL_RECORD(7, {"core": "pointer", "asks": "full"}, via="turn")
|
||||
assert len(added) == 1
|
||||
row = added[0]
|
||||
assert (row.category, row.action, row.user_id) == ("plugin", "reply_shape", 7)
|
||||
import json
|
||||
assert json.loads(row.details) == {"shapes": {"core": "pointer", "asks": "full"},
|
||||
"via": "turn"}
|
||||
|
||||
|
||||
async def test_the_delivery_row_fails_open_and_skips_an_empty_delivery():
|
||||
from unittest.mock import patch
|
||||
|
||||
def _boom():
|
||||
raise RuntimeError("db down")
|
||||
|
||||
with patch("scribe.models.async_session", _boom):
|
||||
await _REAL_RECORD(7, {"core": "full"}, via="turn") # no raise
|
||||
await _REAL_RECORD(7, {}, via="turn")
|
||||
|
||||
@@ -1,169 +0,0 @@
|
||||
"""The Stop hook that checks a task-closing reply for the completion sections
|
||||
(milestone 409 step 5).
|
||||
|
||||
Runs the real shell against synthetic transcripts in the shape Claude Code
|
||||
writes (one content block per JSONL line) and the shared HTTP sink. What it
|
||||
pins: silence on every turn that closed nothing; a block only when the
|
||||
instance recorded it, in the words the instance returned; one rewrite at most,
|
||||
recorded; and no block from another plugin's loop or a failed task write.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.helpers import http_sink
|
||||
|
||||
HOOK = Path(__file__).resolve().parents[1] / "plugin" / "hooks" / "scribe_report_check.sh"
|
||||
TOOL = "mcp__plugin_scribe_scribe__update_task"
|
||||
GOOD = ('**Where this sits:** milestone 12 "Move the backups offsite", step 3 of 5.\n'
|
||||
"**What now works:** the sync runs nightly.\n**Needs you:** nothing.\n**Next:** alerts.")
|
||||
BAD = "All done, pushed it."
|
||||
REASON = "SERVER REASON: rewrite as a completion report"
|
||||
|
||||
|
||||
def _env(tmp_path, url="http://127.0.0.1:9"):
|
||||
for tool in ("curl", "bash"):
|
||||
if shutil.which(tool) is None:
|
||||
pytest.skip(f"hook runtime tool {tool!r} not installed")
|
||||
return {"PATH": os.environ["PATH"], "SCRIBE_URL": url, "SCRIBE_TOKEN": "t",
|
||||
"TMPDIR": str(tmp_path), "HOME": str(tmp_path)}
|
||||
|
||||
|
||||
def _prompt(text="please finish it"):
|
||||
return {"type": "user", "message": {"role": "user", "content": text}}
|
||||
|
||||
|
||||
def _tool_use(tid="toolu_1", status="done", name=TOOL, task_id=41):
|
||||
return {"type": "assistant", "message": {"content": [
|
||||
{"type": "tool_use", "id": tid, "name": name, "input": {"task_id": task_id, "status": status}}]}}
|
||||
|
||||
|
||||
def _result(tid="toolu_1", is_error=False):
|
||||
return {"type": "user", "message": {"content": [
|
||||
{"type": "tool_result", "tool_use_id": tid, "is_error": is_error, "content": "{}"}]}}
|
||||
|
||||
|
||||
def _text(text):
|
||||
return {"type": "assistant", "message": {"content": [{"type": "text", "text": text}]}}
|
||||
|
||||
|
||||
def _transcript(tmp_path, lines):
|
||||
path = tmp_path / "t.jsonl"
|
||||
# Compact, like the file Claude Code writes ({"name":"…"}, no spaces).
|
||||
path.write_text("\n".join(json.dumps(line, separators=(",", ":")) for line in lines) + "\n")
|
||||
return path
|
||||
|
||||
|
||||
def _run(env, transcript, active=False, session="s1"):
|
||||
out = subprocess.run(
|
||||
["bash", str(HOOK)],
|
||||
input=json.dumps({"session_id": session, "transcript_path": str(transcript),
|
||||
"cwd": str(transcript.parent), "hook_event_name": "Stop",
|
||||
"stop_hook_active": active}),
|
||||
capture_output=True, text=True, env=env, timeout=30,
|
||||
)
|
||||
assert out.returncode == 0, out.stderr
|
||||
return out.stdout.strip()
|
||||
|
||||
|
||||
def _closing_turn(reply):
|
||||
return [_prompt(), _text("On it."), _tool_use(), _result(), _text(reply)]
|
||||
|
||||
|
||||
def test_a_turn_that_closed_nothing_is_silent_and_reports_nothing(tmp_path):
|
||||
with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
|
||||
env = _env(tmp_path, f"http://127.0.0.1:{port}")
|
||||
t = _transcript(tmp_path, [_prompt(), _tool_use(status="in_progress"), _result(), _text(BAD)])
|
||||
assert _run(env, t) == ""
|
||||
assert seen == []
|
||||
|
||||
|
||||
def test_a_complete_report_passes_silently_and_is_recorded(tmp_path):
|
||||
with http_sink(b'{"status":"ok"}') as (port, seen):
|
||||
env = _env(tmp_path, f"http://127.0.0.1:{port}")
|
||||
assert _run(env, _transcript(tmp_path, _closing_turn(GOOD))) == ""
|
||||
assert [q["outcome"] for q in seen] == [["passed"]]
|
||||
assert seen[0]["task_ids"] == ["41"]
|
||||
|
||||
|
||||
def test_a_missing_section_blocks_once_in_the_servers_words_then_records_the_rewrite(tmp_path):
|
||||
reply = json.dumps({"status": "ok", "reason": REASON}).encode()
|
||||
with http_sink(reply) as (port, seen):
|
||||
env = _env(tmp_path, f"http://127.0.0.1:{port}")
|
||||
out = json.loads(_run(env, _transcript(tmp_path, _closing_turn(BAD))))
|
||||
assert out == {"decision": "block", "reason": REASON}
|
||||
assert seen[0]["outcome"] == ["blocked"]
|
||||
assert seen[0]["missing"] == ["where it sits,needs you,next"]
|
||||
|
||||
# The rewrite: Claude Code sets stop_hook_active; the hook records and never blocks again.
|
||||
rewritten = _transcript(tmp_path, _closing_turn(BAD) + [_text(GOOD)])
|
||||
assert _run(env, rewritten, active=True) == ""
|
||||
assert seen[1]["outcome"] == ["passed_after_rewrite"]
|
||||
assert _run(env, rewritten, active=True) == ""
|
||||
assert len(seen) == 2
|
||||
|
||||
|
||||
def test_a_rewrite_that_still_misses_is_recorded_and_not_blocked(tmp_path):
|
||||
reply = json.dumps({"status": "ok", "reason": REASON}).encode()
|
||||
with http_sink(reply) as (port, seen):
|
||||
env = _env(tmp_path, f"http://127.0.0.1:{port}")
|
||||
t = _transcript(tmp_path, _closing_turn(BAD))
|
||||
_run(env, t)
|
||||
assert _run(env, t, active=True) == ""
|
||||
assert [q["outcome"][0] for q in seen] == ["blocked", "missing_after_rewrite"]
|
||||
|
||||
|
||||
def test_another_hooks_block_loop_is_left_alone(tmp_path):
|
||||
with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
|
||||
env = _env(tmp_path, f"http://127.0.0.1:{port}")
|
||||
assert _run(env, _transcript(tmp_path, _closing_turn(BAD)), active=True) == ""
|
||||
assert seen == []
|
||||
|
||||
|
||||
def test_a_task_write_that_failed_closed_nothing(tmp_path):
|
||||
with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
|
||||
env = _env(tmp_path, f"http://127.0.0.1:{port}")
|
||||
t = _transcript(tmp_path, [_prompt(), _tool_use(), _result(is_error=True), _text(BAD)])
|
||||
assert _run(env, t) == ""
|
||||
assert seen == []
|
||||
|
||||
|
||||
def test_a_task_closed_in_an_earlier_turn_does_not_count(tmp_path):
|
||||
with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
|
||||
env = _env(tmp_path, f"http://127.0.0.1:{port}")
|
||||
t = _transcript(tmp_path, _closing_turn(GOOD) + [_prompt("thanks, what else?"), _text(BAD)])
|
||||
assert _run(env, t) == ""
|
||||
assert seen == []
|
||||
|
||||
|
||||
def test_a_reply_not_yet_written_is_not_judged(tmp_path):
|
||||
with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
|
||||
env = _env(tmp_path, f"http://127.0.0.1:{port}")
|
||||
t = _transcript(tmp_path, [_prompt(), _tool_use(), _result()])
|
||||
assert _run(env, t) == ""
|
||||
assert seen == []
|
||||
|
||||
|
||||
def test_no_block_without_a_recorded_check(tmp_path):
|
||||
t = _transcript(tmp_path, _closing_turn(BAD))
|
||||
# Unreachable instance.
|
||||
assert _run(_env(tmp_path), t) == ""
|
||||
# An instance that answered but returned no reason.
|
||||
with http_sink(b'{"status":"ok"}') as (port, seen):
|
||||
assert _run(_env(tmp_path, f"http://127.0.0.1:{port}"), t, session="s2") == ""
|
||||
assert seen[0]["outcome"] == ["blocked"]
|
||||
|
||||
|
||||
def test_a_bare_id_does_not_count_as_placing_the_work(tmp_path):
|
||||
reply = json.dumps({"status": "ok", "reason": REASON}).encode()
|
||||
with http_sink(reply) as (port, seen):
|
||||
env = _env(tmp_path, f"http://127.0.0.1:{port}")
|
||||
bare = "Closed #41.\n**Needs you:** nothing.\n**Next:** #42."
|
||||
assert json.loads(_run(env, _transcript(tmp_path, _closing_turn(bare))))["decision"] == "block"
|
||||
assert seen[0]["missing"] == ["where it sits"]
|
||||
@@ -43,9 +43,13 @@ def test_a_request_for_approval_has_its_own_named_section():
|
||||
a completion list as "Blocked by the permission check", and read as a
|
||||
fault rather than a question waiting on them. The heading is what they
|
||||
scan for, so it is pinned by name."""
|
||||
text = _text()
|
||||
assert "Approval requested" in text
|
||||
assert "| **Approval**" in text, "the Asks table lost its Approval row"
|
||||
from scribe.services import reply_shapes
|
||||
|
||||
assert "Approval requested" in _text()
|
||||
# The kinds moved to the delivered shapes (milestone 500): the asks shape
|
||||
# carries the Approval kind, and the core sends a held action there.
|
||||
assert "**Approval**" in reply_shapes.SHAPES["asks"].text
|
||||
assert "Approval requested" in reply_shapes.core().text
|
||||
|
||||
|
||||
def test_placement_comes_from_the_record():
|
||||
@@ -63,3 +67,29 @@ def test_the_shipped_shapes_assume_no_particular_domain():
|
||||
dev_only = [w for w in (r"\bCI\b", r"\bcommit", r"\bpull request", r"file:line", r"\bpytest\b")
|
||||
if re.search(w, text, re.IGNORECASE)]
|
||||
assert not dev_only, f"software-only vocabulary in a product-wide shape: {dev_only}"
|
||||
|
||||
|
||||
def test_the_skill_points_at_the_delivered_shapes_instead_of_restating_them():
|
||||
"""Milestone 500 step 4: the shapes are server product, delivered at their
|
||||
moments. The skill says where they come from and holds the reasoning."""
|
||||
text = _text()
|
||||
assert "list_reply_shapes" in text and "Reply shape · Every reply" in text
|
||||
|
||||
|
||||
def test_the_worked_completion_report_agrees_with_the_delivered_completion_shape():
|
||||
"""The two copies that could drift: the worked example here, and the
|
||||
completion shape the server sends when a task closes. Same sections."""
|
||||
from scribe.services import reply_shapes
|
||||
|
||||
shape = reply_shapes.SHAPES["completion"].text
|
||||
for section in ("Where this sits", "What now works", "How / why", "Needs you", "Next"):
|
||||
assert f"**{section}**" in shape, f"the completion shape lost {section!r}"
|
||||
assert f"**{section}" in _text(), f"the worked example lost {section!r}"
|
||||
|
||||
|
||||
def test_the_core_names_the_header_the_skill_quotes():
|
||||
"""The skill tells the reader what the delivered core looks like; if the
|
||||
header changes, the pointer would describe something that never arrives."""
|
||||
from scribe.services import reply_shapes
|
||||
|
||||
assert reply_shapes.render(reply_shapes.core(), full=True).startswith("Reply shape · Every reply")
|
||||
|
||||
@@ -103,10 +103,8 @@ def test_the_extractor_finds_something() -> None:
|
||||
def test_the_extractor_resolves_the_constant_sources() -> None:
|
||||
"""The specific capability a grep would lose. See the module docstring.
|
||||
|
||||
`report_preference` was the second example until milestone 456 moved it
|
||||
into the retrieval pipeline, where its source is a spec field; the
|
||||
pipeline's reserved slot still records through a module constant, so it
|
||||
carries the case instead.
|
||||
The pipeline's reserved slot records through a module constant, so it
|
||||
carries the case.
|
||||
"""
|
||||
found, _ = call_sites()
|
||||
for via_constant in ("wide_net", "preference_slot"):
|
||||
|
||||
@@ -34,7 +34,6 @@ def test_every_ranked_source_is_measured_as_its_spec_declares():
|
||||
point = POINTS[spec.source]
|
||||
assert point.kind == UNBIDDEN
|
||||
assert point.what == d.what
|
||||
assert point.fixed_query == d.fixed_query
|
||||
# A quiet source must say why, and only a quiet source may (#2475).
|
||||
assert point.expects_traffic == (not d.quiet_because)
|
||||
assert point.quiet_because == d.quiet_because
|
||||
|
||||
@@ -24,9 +24,9 @@ WHAT THIS PINS
|
||||
different file. Renaming one without the other produces a surface that can
|
||||
be tuned and cannot be measured, or measured and not tuned, and both fail
|
||||
silently.
|
||||
2. **Keys are unique.** Two surfaces sharing a settings key is how
|
||||
`report_preference` spent its first release moving whenever the prompt arm
|
||||
was tuned (#3860) — one dial wearing two labels.
|
||||
2. **Keys are unique.** Two surfaces sharing a settings key is how the
|
||||
completion-report arm (since retired) spent its first release moving
|
||||
whenever the prompt arm was tuned (#3860) — one dial wearing two labels.
|
||||
3. **An unknown surface is refused.** Settings keys are free-form strings in
|
||||
a generic table, so a typo'd name would write a key nothing reads: a
|
||||
change that appears to succeed, reports a new value, and alters nothing.
|
||||
@@ -44,7 +44,7 @@ import pytest
|
||||
|
||||
from scribe.services import retrieval_surfaces as rs
|
||||
|
||||
SERVICES = ("plugin_context", "reply_preferences")
|
||||
SERVICES = ("plugin_context",)
|
||||
|
||||
|
||||
def _service_source(name: str) -> str:
|
||||
@@ -63,10 +63,6 @@ def test_every_surface_name_is_a_real_telemetry_source():
|
||||
to justify a change describes a different arm from the one the change hits.
|
||||
"""
|
||||
blob = "\n".join(_service_source(n) for n in SERVICES)
|
||||
# `report_preference` passes its name through a module constant rather than
|
||||
# a literal, so that one name is satisfied by the constant holding it.
|
||||
from scribe.services.reply_preferences import SOURCE
|
||||
|
||||
# The rule arms record through the one pipeline (milestone 456), where
|
||||
# `source` is the spec's own field — so for them the spec IS the string
|
||||
# `record_retrieval` receives, and the join key is checked against it.
|
||||
@@ -76,7 +72,7 @@ def test_every_surface_name_is_a_real_telemetry_source():
|
||||
missing = [
|
||||
s.name for s in rs.SURFACES.values()
|
||||
if f'source="{s.name}"' not in blob
|
||||
and s.name != SOURCE and s.name not in via_pipeline
|
||||
and s.name not in via_pipeline
|
||||
]
|
||||
assert not missing, (
|
||||
f"these surfaces can be tuned but never measured: {missing}. The "
|
||||
|
||||
@@ -471,83 +471,6 @@ def test_a_quiet_arm_is_not_suspended_either_way() -> None:
|
||||
assert "floor_moved_mid_window" not in codes(ws)
|
||||
|
||||
|
||||
# ── a fixed-query arm: decline rate is arithmetic, not evidence (#4232) ─────
|
||||
#
|
||||
# `report_preference` searches one constant string (COMPLETION_QUERY), so it
|
||||
# scores against one number on every call. Its decline rate is therefore 0% or
|
||||
# 100% and never in between, and which one depends only on where the bar sits
|
||||
# relative to that constant.
|
||||
#
|
||||
# So the two warnings swap roles for these arms. "Never declined" stops being
|
||||
# evidence about the floor — `cannot_decline`'s own remedy, "check that it
|
||||
# applies its floor", is unanswerable from it. "Always declined" starts being
|
||||
# evidence, because for a constant score it means the bar is above it and no
|
||||
# further traffic will ever say otherwise.
|
||||
|
||||
|
||||
def test_cannot_decline_is_silent_on_a_fixed_query_arm():
|
||||
"""The live readout fired this on `report_preference` at 45 calls, 0
|
||||
empty, with p10 = p50 = p90 = min = max = 0.791 — five identical
|
||||
percentiles, which is one record at one score rather than a ranking."""
|
||||
ws = warn({"report_preference": src(calls=45, zero_result_calls=0, p10=0.791)})
|
||||
assert "cannot_decline" not in codes(ws, "report_preference")
|
||||
|
||||
|
||||
def test_cannot_decline_still_fires_where_the_rate_means_something():
|
||||
"""The falsifier for the case above (rule 167). If this passes only
|
||||
because the check was disabled rather than narrowed, this fails."""
|
||||
ws = warn({"auto_inject": src(calls=N, zero_result_calls=0)})
|
||||
assert "cannot_decline" in codes(ws, "auto_inject")
|
||||
|
||||
|
||||
def test_a_fixed_query_arm_that_never_clears_its_bar_is_named():
|
||||
"""69 consecutive declines at 0.0006 under the bar is a state this arm has
|
||||
actually been in. Nothing else in the readout would have said so: it looks
|
||||
exactly like an arm with nothing to report."""
|
||||
ws = warn({"report_preference": src(calls=45, zero_result_calls=45)})
|
||||
assert "fixed_query_never_clears" in codes(ws, "report_preference")
|
||||
|
||||
detail = next(w["detail"] for w in ws if w["code"] == "fixed_query_never_clears")
|
||||
assert "near_miss_samples" in detail, (
|
||||
"the last time this fired, every percentile said lower the floor and "
|
||||
"the refused record showed the refusal was right — so the warning has "
|
||||
"to send the reader to the record, not to the dial"
|
||||
)
|
||||
|
||||
|
||||
def test_an_ordinary_arm_returning_nothing_all_window_is_not_dead():
|
||||
"""For an arm whose score can vary, an empty window means nothing matched,
|
||||
which is an answer rather than a fault."""
|
||||
ws = warn({"auto_inject": src(calls=N, zero_result_calls=N)})
|
||||
assert "fixed_query_never_clears" not in codes(ws, "auto_inject")
|
||||
|
||||
|
||||
def test_a_fixed_query_arm_that_sometimes_clears_is_not_dead():
|
||||
"""Only ALL-empty says the bar is above the constant. Anything in between
|
||||
means the score is not actually constant, and the premise is wrong."""
|
||||
ws = warn({"report_preference": src(calls=45, zero_result_calls=44)})
|
||||
assert "fixed_query_never_clears" not in codes(ws, "report_preference")
|
||||
|
||||
|
||||
def test_the_dead_arm_warning_still_needs_volume():
|
||||
ws = warn({"report_preference": src(calls=N - 1, zero_result_calls=N - 1)})
|
||||
assert "fixed_query_never_clears" not in codes(ws)
|
||||
|
||||
|
||||
def test_the_registry_declares_which_arms_ask_a_fixed_question():
|
||||
"""Asserted on structure (rule 167), and able to fail: if `fixed_query`
|
||||
is dropped or defaults to True, one of these two halves breaks."""
|
||||
from scribe.services.retrieval_registry import POINTS
|
||||
|
||||
assert POINTS["report_preference"].fixed_query is True, (
|
||||
"services/reply_preferences.py::COMPLETION_QUERY is a module constant"
|
||||
)
|
||||
# An arm whose query is built from the prompt, the file or the command is
|
||||
# not fixed, and marking one would silence a warning that works there.
|
||||
for varying in ("auto_inject", "write_path", "pre_tool_rule", "prompt_rule"):
|
||||
assert POINTS[varying].fixed_query is False, varying
|
||||
|
||||
|
||||
# ── floor_history_unknown: the window reaches back past the ledger ─────────
|
||||
#
|
||||
# THE SAME SUSPENSION, FOR THE CASE THE ONE ABOVE CANNOT SEE.
|
||||
|
||||
@@ -46,6 +46,7 @@ def test_every_endpoint_is_reachable_on_the_app():
|
||||
"/api/retrieval/moments/mappings",
|
||||
"/api/retrieval/moments/proposals",
|
||||
"/api/retrieval/moments/proposals/judge",
|
||||
"/api/retrieval/reply-shapes",
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -1702,7 +1702,6 @@ def test_every_hook_rule_search_says_which_project_it_is_for():
|
||||
# ranked-search stage, checked below — so a direct search appearing
|
||||
# here again is a copy of the arm coming back, and fails the count.
|
||||
"src/scribe/services/plugin_context.py": 0,
|
||||
"src/scribe/services/reply_preferences.py": 0,
|
||||
}
|
||||
for path, expected in sources.items():
|
||||
calls = [
|
||||
|
||||
@@ -1,134 +0,0 @@
|
||||
"""The completion-report preference lookup (milestone 409 step 4).
|
||||
|
||||
What it pins: the lookup asks for PREFERENCES only, logs every call under its
|
||||
own source (the empty ones too), counts only what it showed as surfaced, and
|
||||
fails open. The query stays domain-neutral, because every kind of project
|
||||
closes tasks.
|
||||
"""
|
||||
import re
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
M = "scribe.services.reply_preferences"
|
||||
# Its two numbers are resolved through the registry now (#4102), so the
|
||||
# settings reader to patch lives there rather than in the arm's own module.
|
||||
RS = "scribe.services.retrieval_surfaces"
|
||||
|
||||
|
||||
def _rule(rid, kind="preference"):
|
||||
return MagicMock(id=rid, kind=kind, title=f"t{rid}", statement=f"s{rid}")
|
||||
|
||||
|
||||
async def _run(hits, *, searched=True, raises=None):
|
||||
from scribe.services.reply_preferences import completion_preferences
|
||||
|
||||
async def search(user_id, query, **kw):
|
||||
if raises:
|
||||
raise raises
|
||||
kw["report"].update({"searched": searched, "best_available_score": 0.7})
|
||||
return hits
|
||||
|
||||
search_mock = AsyncMock(side_effect=search)
|
||||
with patch(f"{M}.semantic_search_rules", search_mock), \
|
||||
patch(f"{RS}.get_setting", AsyncMock(return_value="0.72")), \
|
||||
patch(f"{M}.record_retrieval") as logged, \
|
||||
patch(f"{M}.record_rule_surfaced") as surfaced:
|
||||
out = await completion_preferences(7, project_id=3)
|
||||
return out, search_mock, logged, surfaced
|
||||
|
||||
|
||||
async def test_asks_for_preferences_only_and_returns_them_best_first():
|
||||
out, search, logged, surfaced = await _run([(0.9, _rule(1)), (0.8, _rule(2))])
|
||||
assert search.await_args.kwargs["kind"] == "preference"
|
||||
assert [p["id"] for p in out] == [1, 2]
|
||||
assert all(p["kind"] == "preference" for p in out)
|
||||
assert logged.call_args.kwargs["source"] == "report_preference"
|
||||
assert surfaced.call_args.kwargs == {"user_id": 7, "rule_ids": [1, 2], "source": "report_preference"}
|
||||
|
||||
|
||||
async def test_a_rule_that_slips_through_is_not_handed_back_as_a_preference():
|
||||
out, _, _, surfaced = await _run([(0.9, _rule(1, kind="rule")), (0.8, _rule(2))])
|
||||
assert [p["id"] for p in out] == [2]
|
||||
assert surfaced.call_args.kwargs["rule_ids"] == [2]
|
||||
|
||||
|
||||
async def test_an_empty_call_is_still_logged_and_surfaces_nothing():
|
||||
out, _, logged, surfaced = await _run([])
|
||||
assert out == []
|
||||
logged.assert_called_once()
|
||||
assert logged.call_args.kwargs["results"] == []
|
||||
surfaced.assert_not_called()
|
||||
|
||||
|
||||
async def test_a_search_that_never_ran_says_so_to_the_log():
|
||||
_, _, logged, _ = await _run([], searched=False)
|
||||
assert logged.call_args.kwargs["searched"] is False
|
||||
|
||||
|
||||
async def test_fails_open():
|
||||
out, _, _, surfaced = await _run([], raises=RuntimeError("embedder down"))
|
||||
assert out == []
|
||||
surfaced.assert_not_called()
|
||||
|
||||
|
||||
def test_it_is_a_ranked_source():
|
||||
from scribe.services.rule_usage import is_ambient
|
||||
|
||||
assert not is_ambient("report_preference")
|
||||
|
||||
|
||||
def test_its_bar_is_its_own_key_not_the_prompt_arm_s(monkeypatch):
|
||||
"""Regression on the coupling #3860 found (and on the fix being real).
|
||||
|
||||
This arm shipped reading PROMPTRULE_THRESHOLD_KEY, so an operator tuning
|
||||
the bar for their own prose moved this one with it and was never told.
|
||||
That is worse here than anywhere else: every other arm scores a query that
|
||||
varies per call, while COMPLETION_QUERY is fixed — so this arm's score for
|
||||
a given corpus is a CONSTANT, and a constant that lands under the bar is a
|
||||
dead arm rather than a quiet one. No amount of traffic reveals it.
|
||||
|
||||
Asserted on the key the lookup actually asks for, which is the thing that
|
||||
broke, rather than on the constant being defined somewhere.
|
||||
"""
|
||||
import asyncio
|
||||
|
||||
from scribe.services import plugin_context
|
||||
from scribe.services.reply_preferences import completion_preferences
|
||||
|
||||
asked: list[str] = []
|
||||
|
||||
async def get_setting(user_id, key, default=""):
|
||||
asked.append(key)
|
||||
return default
|
||||
|
||||
async def search(user_id, query, **kw):
|
||||
kw["report"].update({"searched": True})
|
||||
return []
|
||||
|
||||
with patch(f"{RS}.get_setting", AsyncMock(side_effect=get_setting)), \
|
||||
patch(f"{M}.semantic_search_rules", AsyncMock(side_effect=search)), \
|
||||
patch(f"{M}.record_retrieval"):
|
||||
asyncio.run(completion_preferences(7))
|
||||
|
||||
# Both numbers are its own since #4102 — a floor AND a budget — so the arm
|
||||
# asks for two keys, and neither may be the prompt arm's.
|
||||
from scribe.services.retrieval_surfaces import get_surface
|
||||
|
||||
mine = get_surface("report_preference")
|
||||
assert asked == [mine.floor_key, mine.budget_key]
|
||||
theirs = get_surface("prompt_rule")
|
||||
assert mine.floor_key != theirs.floor_key, (
|
||||
"the two keys are the same string again, so the settings form has one "
|
||||
"dial driving two arms — which is the defect, whatever the value is"
|
||||
)
|
||||
assert mine.floor_key == plugin_context.REPORTPREF_THRESHOLD_KEY, (
|
||||
"the registry and the module constant disagree about this arm's key, "
|
||||
"so the Settings form would write one and the arm would read the other"
|
||||
)
|
||||
|
||||
|
||||
def test_the_query_assumes_no_particular_domain():
|
||||
from scribe.services.reply_preferences import COMPLETION_QUERY
|
||||
|
||||
dev_only = [w for w in (r"\bCI\b", r"\bcommit", r"\bpull request", r"\bcode\b", r"\btest")
|
||||
if re.search(w, COMPLETION_QUERY, re.IGNORECASE)]
|
||||
assert not dev_only, f"software-only vocabulary in a query every project runs: {dev_only}"
|
||||
@@ -1,12 +1,38 @@
|
||||
"""The server half of the report-shape check (milestone 409 step 5): the words a
|
||||
blocked reply is sent back with, and the outcome record."""
|
||||
"""The report-shape check (milestone 409 step 5): which completion sections a
|
||||
task-closing reply lacks, the words it is sent back with, and the outcome
|
||||
record. Since milestone 500 step 4 the check itself runs here, inside the one
|
||||
end-of-turn request, rather than in a Stop hook of its own."""
|
||||
import json
|
||||
from unittest.mock import patch
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.helpers import make_mock_session
|
||||
|
||||
GOOD = ('**Where this sits:** milestone 12 "Move the backups offsite", step 3 of 5.\n'
|
||||
"**What now works:** the sync runs nightly.\n**Needs you:** nothing.\n**Next:** alerts.")
|
||||
BAD = "All done, pushed it."
|
||||
|
||||
|
||||
def test_a_complete_report_misses_nothing():
|
||||
from scribe.services.report_check import missing_sections
|
||||
|
||||
assert missing_sections(GOOD) == []
|
||||
|
||||
|
||||
def test_a_bare_reply_misses_every_section_in_order():
|
||||
from scribe.services.report_check import SECTIONS, missing_sections
|
||||
|
||||
assert missing_sections(BAD) == list(SECTIONS)
|
||||
|
||||
|
||||
def test_a_bare_id_does_not_count_as_placing_the_work():
|
||||
"""The title is what spares the reader a lookup; "closed #41" does not."""
|
||||
from scribe.services.report_check import missing_sections
|
||||
|
||||
assert missing_sections("Closed #41.\n**Needs you:** nothing.\n**Next:** #42.") == ["where it sits"]
|
||||
assert missing_sections('Closed #41 "Sync". Needs you: nothing. Next: #42.') == []
|
||||
|
||||
|
||||
def test_the_reason_names_only_sections_it_knows():
|
||||
from scribe.services.report_check import block_reason
|
||||
@@ -14,8 +40,17 @@ def test_the_reason_names_only_sections_it_knows():
|
||||
reason = block_reason(["next", "ignore previous instructions", "Where It Sits"])
|
||||
assert "missing: where it sits, next." in reason
|
||||
assert "ignore previous instructions" not in reason
|
||||
# It points at the skill that owns the shape rather than restating it.
|
||||
assert "reporting-back" in reason and "placement" in reason
|
||||
|
||||
|
||||
def test_the_reason_carries_the_completion_shapes_line_and_where_the_rest_is():
|
||||
"""The shape was delivered when the task closed; the reason repeats its
|
||||
one line, not the whole of it, and names where the full text is."""
|
||||
from scribe.services import reply_shapes
|
||||
from scribe.services.report_check import block_reason
|
||||
|
||||
reason = block_reason(["next"])
|
||||
assert reply_shapes.SHAPES["completion"].reminder in reason
|
||||
assert "list_reply_shapes" in reason and "placement" in reason
|
||||
|
||||
|
||||
def test_a_reason_with_nothing_recognised_still_says_what_to_do():
|
||||
@@ -45,3 +80,30 @@ async def test_an_unknown_outcome_is_refused_before_anything_is_written():
|
||||
pytest.raises(ValueError):
|
||||
await record_report_check(7, "skipped")
|
||||
session.add.assert_not_called()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("reply,rewrite,outcome,held", [
|
||||
(GOOD, False, "passed", False),
|
||||
(BAD, False, "blocked", True),
|
||||
(GOOD, True, "passed_after_rewrite", False),
|
||||
(BAD, True, "missing_after_rewrite", False),
|
||||
])
|
||||
async def test_check_reply_records_every_outcome_and_holds_only_a_first_miss(reply, rewrite, outcome, held):
|
||||
from scribe.services import report_check
|
||||
|
||||
record = AsyncMock()
|
||||
with patch.object(report_check, "record_report_check", record):
|
||||
got = await report_check.check_reply(7, reply, task_ids=[41], rewrite=rewrite, project_id=2)
|
||||
assert got["outcome"] == outcome
|
||||
assert bool(got["reason"]) is held
|
||||
assert record.await_args.args == (7, outcome)
|
||||
assert record.await_args.kwargs["task_ids"] == [41]
|
||||
|
||||
|
||||
async def test_an_unrecorded_check_holds_nothing():
|
||||
"""A hold the numbers cannot see is not one this check may make."""
|
||||
from scribe.services import report_check
|
||||
|
||||
with patch.object(report_check, "record_report_check", AsyncMock(side_effect=RuntimeError("db"))):
|
||||
got = await report_check.check_reply(7, BAD, task_ids=[41], rewrite=False)
|
||||
assert got == {"outcome": "", "reason": ""}
|
||||
|
||||
@@ -58,7 +58,7 @@ DEFS = HOOKS / "scribe_defs.sh"
|
||||
# the convention tests below are what keep this roster honest as it grows.
|
||||
LEDGERS = {
|
||||
"scribe-priorart": (".ids", ".rules.ids", ".opened.ids", ".sync.ids",
|
||||
".derive.ids"),
|
||||
".derive.ids", ".shapes.ids", ".reportcheck.ids"),
|
||||
"scribe-autoinject": (".ids",),
|
||||
}
|
||||
|
||||
|
||||
@@ -53,7 +53,6 @@ _SURFACE_PAIRS = (
|
||||
("write_path_rule", "kbRuleHintThreshold"),
|
||||
("pre_tool_rule", "kbToolRuleThreshold"),
|
||||
("prompt_rule", "kbPromptRuleThreshold"),
|
||||
("report_preference", "kbReportPrefThreshold"),
|
||||
("reply_rule", "kbReplyRuleThreshold"),
|
||||
)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user