Merge pull request 'Shapes are judged when they are written (milestone 439 steps 1–6) + #4608 fixes' (#188) from dev into main
CI & Build / Python lint (push) Successful in 3s
CI & Build / Plugin hooks (push) Successful in 12s
CI & Build / integration (push) Successful in 56s
CI & Build / TypeScript typecheck (push) Successful in 57s
CI & Build / Python tests (push) Successful in 1m45s
CI & Build / Build & push image (push) Successful in 17s

This commit was merged in pull request #188.
This commit is contained in:
2026-10-01 08:55:10 -04:00
30 changed files with 1613 additions and 193 deletions
+20 -17
View File
@@ -275,41 +275,44 @@ function actorLabel(actor: string): string {
}
async function saveKbInject() {
const t = Math.min(1, Math.max(0, Number(kbInjectThreshold.value) || 0));
// Every similarity bar, clamped the way the server's `bounded_float` clamps
// it: an unparseable value falls back to its default, never to 0, and `lo`
// is the floor a bar that BLOCKS must not go under (0.80 block, 0.70 overlap).
const asBar = (v: string, d: number, lo = 0) => Math.min(1, Math.max(lo, Number(v) || d));
const t = asBar(kbInjectThreshold.value, 0);
const k = Math.min(10, Math.max(1, Math.floor(Number(kbInjectTopK.value) || 1)));
// `|| default` not `|| 0`: an unparseable value here should fall back to the
// per-kind default, not to 0 — a 0 floor would report every record as a
// duplicate of every other one.
const dupSnip = Math.min(1, Math.max(0, Number(kbDupThresholdSnippet.value) || 0.82));
const dupNote = Math.min(1, Math.max(0, Number(kbDupThresholdNote.value) || 0.93));
const dupTask = Math.min(1, Math.max(0, Number(kbDupThresholdTask.value) || 0.93));
const dupSnip = asBar(kbDupThresholdSnippet.value, 0.82);
const dupNote = asBar(kbDupThresholdNote.value, 0.93);
const dupTask = asBar(kbDupThresholdTask.value, 0.93);
// The gate BLOCKS a write, so its bars have a floor the server enforces too:
// 0.80 for a block, 0.70 for the overlap list.
const gateAt = (v: string, d: number, lo: number) => Math.min(1, Math.max(lo, Number(v) || d));
const gate = gateAt(kbGateThreshold.value, 0.9, 0.8);
const gateSnip = gateAt(kbGateThresholdSnippet.value, 0.96, 0.8);
const gateLesson = gateAt(kbGateThresholdLesson.value, 0.96, 0.8);
const gateCopy = gateAt(kbGateThresholdNoteCopy.value, 0.98, 0.8);
const gateOverlap = gateAt(kbGateNoteOverlapFloor.value, 0.87, 0.7);
const gate = asBar(kbGateThreshold.value, 0.9, 0.8);
const gateSnip = asBar(kbGateThresholdSnippet.value, 0.96, 0.8);
const gateLesson = asBar(kbGateThresholdLesson.value, 0.96, 0.8);
const gateCopy = asBar(kbGateThresholdNoteCopy.value, 0.98, 0.8);
const gateOverlap = asBar(kbGateNoteOverlapFloor.value, 0.87, 0.7);
// Same `|| default` guard: a floor of 0 would hand back an existing plan
// for every new one, and no plan could be started without force.
const planT = Math.min(1, Math.max(0, Number(kbPlanMatchThreshold.value) || 0.8));
const planT = asBar(kbPlanMatchThreshold.value, 0.8);
// Same `|| default` reasoning: falling back to 0 would surface every
// snippet in the corpus on every edit, which is the failure this knob fixes.
const wpT = Math.min(1, Math.max(0, Number(kbWritePathThreshold.value) || 0.68));
const wpT = asBar(kbWritePathThreshold.value, 0.68);
// Same `|| default` reasoning again, and it bites harder here: a rule hint
// fires on every write, so a fallback of 0 would attach a standing rule to
// every edit in the session.
const rhT = Math.min(1, Math.max(0, Number(kbRuleHintThreshold.value) || 0.72));
const rhT = asBar(kbRuleHintThreshold.value, 0.72);
// Same `|| default` guard, and the same reason: this arm fires before every
// Bash call, so a fallback of 0 would put a rule in front of every command.
const trT = Math.min(1, Math.max(0, Number(kbToolRuleThreshold.value) || 0.68));
const prT = Math.min(1, Math.max(0, Number(kbPromptRuleThreshold.value) || 0.72));
const trT = asBar(kbToolRuleThreshold.value, 0.68);
const prT = asBar(kbPromptRuleThreshold.value, 0.72);
// The checkpoint bar, and the `|| default` guard matters most here of all:
// this is the only number that can STOP a call, so a fallback of 0 would
// hold the first command of every session behind whatever ranked first.
const cpT = Math.min(1, Math.max(0, Number(kbCheckpointThreshold.value) || 0.8));
const rpT = Math.min(1, Math.max(0, Number(kbReportPrefThreshold.value) || 0.72));
const cpT = asBar(kbCheckpointThreshold.value, 0.8);
const rpT = asBar(kbReportPrefThreshold.value, 0.72);
// The budgets, clamped the way the server clamps them: a whole number in
// [1, 10]. Never 0 — an arm turned off is turned off by its switch, and a
// budget of zero would run the search, log the retrieval and render nothing,
+1 -1
View File
@@ -1,7 +1,7 @@
{
"name": "scribe",
"description": "Scribe for Claude Code: connects the scribe MCP server, adds the hooks that deliver live project state and relevant records at the right moment, ships the shared client-neutral Scribe skills (using-scribe, writing-plans, reporting-back, systematic-debugging, verification, brainstorming, reusing-code, shape-accounting), and syncs your saved Scribe Processes as skills (/scribe:sync).",
"version": "2026.09.24.1358",
"version": "2026.10.01.1246",
"author": {
"name": "Bryan Van Deusen"
},
+1
View File
@@ -40,6 +40,7 @@ another one means adding files, not moving or rewriting any.
| `hooks/scribe_after_write.sh` | PostToolUse on shell commands: the same check for code written through the shell. |
| `hooks/scribe_tool_rules.sh` | PreToolUse on shell commands: `GET /api/plugin/tool-rules`. |
| `hooks/scribe_report_check.sh` | Stop: when the turn closed a task, checks the reply for the completion sections and reports to `GET /api/plugin/report-check`; blocks once, with the reason the server returns. |
| `hooks/scribe_shape_check.sh` | Stop: sends the definitions the turn wrote (the write hooks' `<sid>.written.ids` ledger) to `GET /api/plugin/shape-check`; blocks once, with the reason the server returns, so the agent judges what it built. |
| `hooks/scribe_sync_processes.sh` + `commands/sync.md` | `GET /api/plugin/processes` → `~/.claude/skills/scribe-proc-*` stubs; `/scribe:sync` on demand. |
| `hooks/scribe_defs.sh` | Shared shell helpers: config, dedup ledgers, outage line. |
| `hooks/scribe_static_context.md` | The adapter static text. |
+9
View File
@@ -91,6 +91,15 @@ On install you'll be asked for:
never blocks twice, and never blocks when the instance did not record the
check (unconfigured or unreachable). Outcomes land in the admin logs under
category `plugin`, action `report_check`.
- `hooks/hooks.json` → a second Stop hook (`hooks/scribe_shape_check.sh`): the
write hooks note every definition a write names (and every new file) in a
session ledger; at the end of the turn this sends them to
`GET /api/plugin/shape-check`, which answers with the ones nobody has judged.
If any are, it blocks once with the server's reason, asking the agent that
wrote them to say what each is — reusable (record a snippet), an instance or
variant of one, or a one-off — in one `classify_shapes` call. Same discipline
as the report check: never twice, never without a recorded check. Outcomes
land under category `plugin`, action `shape_check`.
- `skills/` → the universal process-skills, surfaced by description match.
- `hooks/scribe_sync_processes.sh` (a 2nd SessionStart hook) + the `/scribe:sync`
command → generate `~/.claude/skills/scribe-proc-*` stubs from your Scribe
+4
View File
@@ -98,6 +98,10 @@
{
"type": "command",
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_report_check.sh\""
},
{
"type": "command",
"command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_shape_check.sh\""
}
]
}
+5
View File
@@ -129,14 +129,19 @@ while IFS= read -r rel_path; do
# The code just written: the ADDED lines of the uncommitted diff for a
# tracked file (sed, not cut: this strips one marker char per line, it is
# not a payload cap), the whole file when untracked.
fresh=""
if git -C "$repo_root" ls-files --error-unmatch -- "$rel_path" >/dev/null 2>&1; then
code=$(git -C "$repo_root" diff -U0 -- "$rel_path" 2>/dev/null | grep '^+' | grep -v '^+++' | sed 's/^+//') || code=""
else
code=$(cat "$file_path" 2>/dev/null) || code=""
fresh="new"
fi
[ -n "$code" ] || continue
# `|| true` inside: an early-exiting `head` must not void its own output (#4042).
names=$(printf '%s' "$code" | scribe_defs | sort -u | head -12 || true)
# What this write defined, for the end-of-turn question (milestone 439) —
# recorded before the arms below, which skip a write that defines nothing.
printf '%s\n' "$names" | scribe_written_append "$safe_sid" "$rel_path" "$fresh" "$code"
# Nothing DEFINED in what was written (prose, data, a call-site edit) →
# nothing to say; the arms are about shapes.
[ -n "$names" ] || continue
+35
View File
@@ -1158,6 +1158,41 @@ scribe_held_query() {
# the tests read THIS string rather than a copy of it.
SCRIBE_LEDGER_DIRS="scribe-priorart scribe-autoinject"
# THE WRITTEN-SHAPES LEDGER (milestone 439). Every definition a write names,
# appended as `path<TAB>kind<TAB>name`, for the Stop hook to ask about at the
# end of the turn (scribe_shape_check.sh). Its own file — one file, one
# question (lesson #4226) — under the prior-art directory, so the session
# clear above already sweeps it. Read `kind<TAB>name` lines on stdin (the
# scribe_defs format).
#
# A NEW file that is a unit in its own right adds one `file` line named by its
# stem — the same test the server's sync applies (coverage.is_file_unit): it
# RENDERS (markup names classes, or it opens a <script>/<template>/<style>
# block) and nothing inside it is named after it. That is a component in any
# framework, without a list of frameworks.
# $1 session id (already made filename-safe) $2 repo-relative path
# $3 "new" when the file did not exist before this write
# $4 the code written (to test whether the file renders)
scribe_written_append() {
local sid="$1" rel="$2" fresh="${3:-}" code="${4:-}" dir stem names
if [ -z "$sid" ] || [ -z "$rel" ]; then cat >/dev/null; return 0; fi
dir="${TMPDIR:-/tmp}/scribe-priorart"
mkdir -p "$dir" 2>/dev/null || true
names=$(cat)
{
if [ "$fresh" = "new" ]; then
stem=${rel##*/}; stem=${stem%.*}
if [ -n "$stem" ] \
&& ! printf '%s\n' "$names" | awk -F'\t' -v s="$stem" '$1 == "sym" && $2 == s { f = 1 } END { exit !f }' \
&& printf '%s\n' "$code" | grep -qE '^[[:space:]]*<(script|template|style)([[:space:]>]|$)|(class|className)[[:space:]]*='; then
printf '%s\tfile\t%s\n' "$rel" "$stem"
fi
fi
printf '%s\n' "$names" | awk -F'\t' -v p="$rel" 'NF >= 2 && ($1 == "sym" || $1 == "css") && $2 != "" { printf "%s\t%s\t%s\n", p, $1, $2 }'
} >> "$dir/${sid}.written.ids" 2>/dev/null || true
return 0
}
scribe_clear_session_ledgers() {
# SPARES `*.keep.ids`, which are evidence rather than exclusions — see
# `scribe_rules_append`. Everything else still goes: the convention is
+13 -4
View File
@@ -17,8 +17,9 @@
# It is also the shape ledger's write-path feed (#2791): it names the
# definitions being written (`shapes=`), and the server — only when the
# session has PULLED a snippet this code references or resembles — records
# them as instance rows, classified_by=hook. Evidence, not judgment; the
# context line says what landed so a wrong stamp is corrected in the moment.
# it as a PROPOSAL on them ("looks like #N"), never a status (milestone 439).
# The agent's own end-of-turn judgment is the verdict; the same names go to
# the written-shapes ledger the Stop hook asks about.
#
# NEVER BLOCKS. It returns `additionalContext` with no `permissionDecision`, so
# the write proceeds untouched and Claude sees the note beside the tool result.
@@ -103,8 +104,8 @@ fi
# changes the inside of a function rather than its signature — the definition
# enclosing the edit, found by walking the target file upward from the edited
# lines. The server decides whether evidence exists (the session pulled a
# snippet this code references or resembles) and stamps instance rows; with
# no pulled canon in play, nothing is recorded. Titles only still — this sends
# snippet this code references or resembles) and records it as a proposal;
# with no pulled canon in play, nothing is recorded. Titles only still — this sends
# names, not bodies.
# ---------------------------------------------------------------------------
#
@@ -201,6 +202,14 @@ derive_exclude_q=""
rule_exclude_q=""
if [ -n "$session_id" ]; then
safe_sid=$(printf '%s' "$session_id" | tr -c 'A-Za-z0-9._-' '_')
# What this write defined, for the end-of-turn question (milestone 439). A
# Write onto a path that does not exist yet creates a file, which may be a
# unit in its own right (see scribe_written_append). Written before any
# server call: the ask at the end of the turn must not depend on this
# write's hint having been answered.
fresh=""
[ -e "$file_path" ] || fresh="new"
printf '%s\n' "$shapes" | scribe_written_append "$safe_sid" "$rel_path" "$fresh" "$code"
idfile="$state_dir/${safe_sid}.ids"
syncfile="$state_dir/${safe_sid}.sync.ids"
derivefile="$state_dir/${safe_sid}.derive.ids"
+105
View File
@@ -0,0 +1,105 @@
#!/usr/bin/env bash
# Scribe plugin — Stop hook: the turn's new shapes are judged by the agent
# that wrote them (milestone 439).
#
# The write hooks (scribe_prior_art.sh, scribe_after_write.sh) append every
# definition a write names to a session ledger, `<sid>.written.ids`
# (`path<TAB>kind<TAB>name`, see scribe_written_append). At the end of the
# turn this hook sends that ledger to the instance, which answers with what
# nobody has judged yet. If anything is, the hook blocks ONCE, in the server's
# words: the agent that built the code is the one participant who knows what
# it is, and the end of the turn is the last moment that is still true.
#
# THE SAME DISCIPLINE AS scribe_report_check.sh, deliberately:
# - it blocks only on a `reason` the instance returned, which it returns
# only for a block it RECORDED — an unconfigured or unreachable instance
# never stops a session, and every intervention is one the numbers see;
# - never twice for one turn: the stop that follows its own block
# (`stop_hook_active` with this hook's marker present) is recorded as
# judged_after_block / left_after_block and let through;
# - another plugin's block loop is left alone (active, no marker).
#
# THE LEDGER IS CONSUMED, NOT TURN-STAMPED. Everything written since the last
# stop this hook answered is "this turn". It is emptied once the instance has
# answered — after a pass, or after the post-block stop — and KEPT when the
# instance could not be reached, so the question is asked at the next stop
# that can be answered rather than silently dropped.
#
# Config (same as the other hooks):
# CLAUDE_PLUGIN_OPTION_API_ENDPOINT base URL, no trailing slash
# CLAUDE_PLUGIN_OPTION_API_TOKEN fmcp_ API key (sensitive)
# SCRIBE_URL / SCRIBE_TOKEN override for the settings.json dogfooding path.
set -uo pipefail
command -v curl >/dev/null 2>&1 || exit 0
# shellcheck source=plugin/hooks/scribe_defs.sh
. "$(dirname "${BASH_SOURCE[0]}")/scribe_defs.sh"
# Stop delivers { session_id, transcript_path, cwd, hook_event_name, stop_hook_active }.
event=$(cat 2>/dev/null || true)
event_flat=$(printf '%s' "$event" | scribe_json_flat)
session_id=$(scribe_json_pick "$event_flat" '.session_id')
active=$(scribe_json_pick "$event_flat" '.stop_hook_active')
event_cwd=$(scribe_json_pick "$event_flat" '.cwd')
[ -n "$session_id" ] || exit 0
safe_sid=$(printf '%s' "$session_id" | tr -c 'A-Za-z0-9._-' '_')
ledger="${TMPDIR:-/tmp}/scribe-priorart/${safe_sid}.written.ids"
state_dir="${TMPDIR:-/tmp}/scribe-shapecheck"
mkdir -p "$state_dir" 2>/dev/null || true
marker="$state_dir/${safe_sid}.blocked"
if [ "$active" = "true" ]; then
# A Stop hook already blocked this stop. Another plugin's → nothing to add.
[ -f "$marker" ] || exit 0
phase=after
else
rm -f "$marker" 2>/dev/null || true
phase=check
fi
if [ ! -s "$ledger" ]; then
rm -f "$marker" 2>/dev/null || true
exit 0
fi
scribe_config || exit 0
# Unique lines, oldest first, capped: the request stays a GET (a read-scoped
# key runs the plugin), and a turn that wrote more than this generated code.
written=$(awk '!seen[$0]++' "$ledger" 2>/dev/null | head -n 80)
written_enc=$(printf '%s' "$written" | scribe_urlenc) || exit 0
dir="${event_cwd:-${CLAUDE_PROJECT_DIR:-$PWD}}"
q="phase=${phase}&written=${written_enc}"
scope=$(scribe_scope_query "$dir")
[ -n "$scope" ] && q="${q}&${scope}"
# The repo rides along even when a marker names the project: the instance
# needs it to tell the agent which repo a just-written shape belongs to.
case "$scope" in
repo=*) ;;
*)
remote=$(git -C "$dir" remote get-url origin 2>/dev/null || true)
if [ -n "$remote" ]; then
remote_enc=$(printf '%s' "$remote" | scribe_urlenc) || remote_enc=""
[ -n "$remote_enc" ] && q="${q}&repo=${remote_enc}"
fi
;;
esac
answer=$(curl -fsS --max-time 4 \
-H "Authorization: Bearer ${token}" \
"${url%/}/api/plugin/shape-check?${q}" 2>/dev/null) || exit 0 # unreachable: keep the ledger
if [ "$phase" = "after" ]; then
rm -f "$marker" "$ledger" 2>/dev/null || true
exit 0
fi
reason=$(scribe_json_pick "$(printf '%s' "$answer" | scribe_json_flat)" '.reason')
if [ -z "$reason" ]; then
rm -f "$ledger" 2>/dev/null || true
exit 0
fi
: > "$marker" 2>/dev/null || true
printf '{"decision":"block","reason":"%s"}\n' "$(printf '%s' "$reason" | scribe_json_escape)"
exit 0
+30 -4
View File
@@ -1,6 +1,6 @@
---
name: reusing-code
description: Use when you're about to build ANY shape — a component, control, route handler, service class, helper, test scaffold — search recorded snippets FIRST and start from the recorded shape instead of re-solving it. And the FIRST time a shape is built, record it as a snippet so every later instance starts from it. Triggers on "write a util/helper", "I need a function that…", "let me add a component/button/field/route", or having just built the first instance of anything.
description: Use when you're about to build ANY shape — a component, control, route handler, service class, helper, test scaffold — search recorded snippets FIRST and start from the recorded shape instead of re-solving it. And before the turn that built it ends, say what it is — record the reusable as a snippet, classify the rest — so every later instance starts from it. Triggers on "write a util/helper", "I need a function that…", "let me add a component/button/field/route", having just built the first instance of anything, or the end-of-turn list of shapes to judge.
---
# Reusing code — the pattern library
@@ -34,9 +34,9 @@ through recall/auto-inject; this skill is the active reflex around that.
to ask both at once.
- If a snippet fits, pull it in full with `get_snippet(id)` and reuse it — its
`location` points at the reference implementation. Adapt, don't re-derive.
The pull also does the accounting: the code you then write that references
or resembles it is stamped an `instance` of that canon in the shape ledger
(classified_by=hook) — reuse from memory leaves no row.
The pull also feeds the accounting: the code you then write that references
or resembles it is offered back to you as "looks like #N" when you judge the
turn's shapes — reuse from memory leaves no such evidence.
- If auto-inject already surfaced a snippet title that looks relevant, that's
your cue to `get_snippet` it rather than start from scratch.
- **Prior art offered beside a write is not noise — read it.** When Scribe notes
@@ -85,6 +85,32 @@ through recall/auto-inject; this skill is the active reflex around that.
per reusable thing. If it already exists, `update_snippet` it instead of
recording a second copy (the create gate will flag a near-duplicate anyway).
## Before the turn ends — say what you built
You are the one who knows what the code you just wrote is. The shape ledger
records it from your answer, not from a guess made later, so give the answer
in the turn that built it — while it is still true. When a turn wrote
definitions nobody has judged, the plugin's Stop hook lists them once, with
whatever evidence the machinery holds ("looks like #N", "#M is canon in this
directory", "new"); other clients reach the same moment through
`list_shapes(project_id, path=…)`. For each one, decide:
- **Something another part of the code should reuse** → record it with
`create_snippet` (the fields above). Its own row becomes the canon.
- **Built from a recorded snippet** → `instance` of it.
- **A deliberate departure from one** → `variant`, with the why.
- **A genuine one-off** → `exempt`, with a reason (`reason_code` such as
`one-off-handler`, `test-helper`, `scoped-css` or `pure-helper` indexes it).
One call records them all: `classify_shapes(project_id, repo="<the repo's
remote>", classifications=[{path, symbol, kind, status, snippet_id?, reason?}])`.
`repo` lets a verdict land on a shape the ledger has not synced yet — the one
you wrote a minute ago. A component file is a shape too (`kind: "file"`, named
by its stem): when a card, chip or dialog is the reusable thing, the file is
what you record. An unsure call is still yours to make — the reason you write
is what the next reader judges it by — and a verdict is permanent, so each
shape is asked about once.
## A shared snippet is a suggestion, not a standard
Scribe is multi-user, so a search can return snippets other people own. Those
+31 -15
View File
@@ -18,8 +18,8 @@ row carries a status:
- `exempt` — judged genuinely one-off. **Reason required.** A recorded
judgment, not silence — it stops the next pass re-litigating it.
- `scoped` — one-off **by construction**, stamped by the coverage sync
(a Vue component's scoped `<style>` rules and its `<script setup>`
functions — unreachable from any other file). Accounted for without a
(a Vue or Svelte component's own styles and instance-script functions —
unreachable from any other file). Accounted for without a
judgment; still proposed against, grouped and flagged; any judgment you
make overrides it. Not the todo.
- `unclassified` — nobody has judged it yet. **This is the todo list.**
@@ -39,22 +39,32 @@ row carries a status:
nothing. Rows, never prose — a consumer list in a note or verification
detail cannot be sorted, queried, or diffed.
## Rows that arrive on their own
## The writer judges; audits check that it happened
Two feeds keep the ledger current between your batches, so most shapes never
need a hand judgment:
The ledger is filled from your answers, given while you still know them. When
a turn wrote definitions nobody has judged, the plugin's Stop hook lists them
once — with whatever the machinery holds as evidence — and you classify them
in one `classify_shapes(project_id, repo=…, …)` call (the `reusing-code` skill
says what each verdict means). `repo` lets a verdict land on a shape the sync
has not read yet. So a project worked this way stays accounted as it grows,
and an audit pass is a CHECK that the reflex held — the coverage line names
what slipped — not the way rows get classified.
What still arrives without you:
- **The sync** stamps a snippet's own reference location `canonical`
(`classified_by: mechanical`).
- **The write path** stamps instances as you work: when you `get_snippet` a
canon and then write code that references or resembles it, the
definitions being written land as `instance` rows (`classified_by: hook`,
the evidence in `reason`), and the prior-art hint tells you what landed
("Shape accounting: recorded at … → instance of #N"). Offered-but-unopened
snippets stamp nothing — so *pull the canon you are instantiating*; that
pull is what turns your reuse into accounting. A `hook` row is evidence, not
judgment: it never overrides a classification you made, and a
`classify_shapes` call overrides it.
(`classified_by: mechanical`) — including a component FILE row when the
snippet's location names the file and no symbol.
- **The write path suggests**: when you `get_snippet` a canon and then write
code that references or resembles it, the shapes being written carry it as
a proposal, and the prior-art hint says "→ looks like #N". It is evidence
for your verdict, never a verdict — so *pull the canon you are
instantiating*; that pull is what puts it in front of you at the end of the
turn. Rows the hook stamped `instance` before it stopped doing so are listed
by `stamps_to_review` for judgment.
- **A component is a row of its own** (`kind: "file"`, named by its stem): a
file that renders and defines nothing named after itself. It is judged like
any other shape.
## The machine proposes, judgment classifies
@@ -152,6 +162,12 @@ Three questions the ledger answers mechanically (#2793):
- **Recheck** — `list_shapes(project_id, flag="recheck")`: judged
instances/variants whose body changed since judged. The judgment stands;
re-confirm it (classify again with the same status) or re-judge.
- **Weak stamps** — the coverage line's "N weak stamps … to judge —
stamps_to_review": rows the write-path hook asserted, unattended, on a
resemblance below today's floor. They no longer elect a directory's canon
or raise a divergence, but they still claim to be instances. Read each
one and record the verdict with `classify_shapes` — `instance` if it is,
otherwise what it actually is.
## What this buys
+7
View File
@@ -358,6 +358,13 @@ SMOKE_EVENTS: dict[str, str] = {
{"session_id": "smoke", "transcript_path": "/nonexistent/smoke.jsonl",
"cwd": ".", "hook_event_name": "Stop", "stop_hook_active": False}
),
# The end-of-turn shape check (milestone 439). With no ledger for the
# session there is nothing to ask about, and with no instance there is
# nobody to ask: silent, and never a block, configured or not.
"scribe_shape_check.sh": json.dumps(
{"session_id": "smoke", "transcript_path": "/nonexistent/smoke.jsonl",
"cwd": ".", "hook_event_name": "Stop", "stop_hook_active": False}
),
# The PreCompact preserver (#3680). Its whole contract is the inverse of
# every other hook's: it must exit 0 AND print, because stdout is what
# becomes the summarizer's custom instructions. A silent success here is
+2 -1
View File
@@ -56,7 +56,8 @@ reads Agent Skills) and in each tool's description.
matched, not none. Rules bind; preferences guide and you keep them current;
lessons inform.
- Search before acting or building, scoped with the active project_id; start
from a recorded snippet, and create_snippet what you build.
from a recorded snippet; before the turn ends, say what you built
(create_snippet the reusable, classify_shapes the rest).
- Work is tasks (a fix is kind="issue"): in_progress on start, add_task_log as
you go, done on finish; tag system_ids.
- A plan is a milestone: find the existing one
+15 -7
View File
@@ -4,7 +4,8 @@ The accounting model (note 2786): the snippet library records CANON (small);
the ledger accounts for EVERY extracted shape (total). These tools are how
agents move shapes out of `unclassified` — the todo state — and how they read
what still needs judgment. The ledger rows themselves are fed by the coverage
refresh; these tools only ever judge what the sync has seen.
refresh; a judgment given with `repo` may also land on a shape written this
turn that the sync has not seen yet (milestone 439).
"""
from __future__ import annotations
@@ -14,7 +15,7 @@ from scribe.services import shape_ledger as shape_ledger_svc
async def classify_shapes(
project_id: int, classifications: list[dict], via: str = "agent"
project_id: int, classifications: list[dict], via: str = "agent", repo: str = ""
) -> dict:
"""Record judgments for a project's code shapes — in batch, as rows.
@@ -52,15 +53,22 @@ async def classify_shapes(
and get_snippet's `uses` read them.
via: Who is judging — "agent" (default), "audit" (a sweep), or
"import" (carrying maps recorded elsewhere).
repo: The repo the shapes live in (its remote URL or bound key), for
judging what you wrote THIS turn — shapes the ledger has not
synced yet. With it, a shape no row matches is recorded as a
provisional row under that repo and judged; the next sync
confirms it. Must be bound to the project. Omit it to judge only
synced rows.
All-or-nothing: a structural error, a missing snippet target, or no write
access applies NOTHING. Returns {"classified": N, "unmatched": [...]} —
unmatched names shapes no live ledger row matches (the tree may have
moved since you listed; re-run the project's coverage refresh to re-sync).
All-or-nothing: a structural error, a missing snippet target, an unbound
repo, or no write access applies NOTHING. Returns {"classified": N,
"unmatched": [...], "provisional": N?} — unmatched names shapes no live
ledger row matches (the tree may have moved since you listed; pass `repo`
for a shape you just wrote, or re-run the project's coverage refresh).
"""
uid = current_user_id()
return await shape_ledger_svc.classify_shapes(
uid, project_id, classifications, via=via
uid, project_id, classifications, via=via, repo=repo
)
+72 -4
View File
@@ -15,6 +15,7 @@ from scribe.config import Config
from scribe.services import plugin_context as plugin_ctx_svc
from scribe.services import repo_bindings as repo_bindings_svc
from scribe.services import report_check as report_check_svc
from scribe.services import shape_check as shape_check_svc
from scribe.services import task_claims as task_claims_svc
from scribe.services.settings import get_admin_setting, set_setting
@@ -269,10 +270,12 @@ async def write_path_prior_art():
css|sym. The shape ledger's write-path feed
(#2791): when the session recently PULLED a
snippet this payload references or resembles,
these land as instance rows (classified_by=hook).
Honoured only for a caller allowed to write — a
read-scoped key still gets the hint, and never
changes accounting on a GET.
these carry it as a proposal — evidence for the
agent's own end-of-turn judgment, never a
verdict (milestone 439). Honoured only for a
caller allowed to write — a read-scoped key still
gets the hint, and never changes the ledger on a
GET.
"""
path = (request.args.get("path") or "").strip()
code = request.args.get("code") or ""
@@ -359,6 +362,71 @@ async def report_check():
return jsonify(body)
@plugin_bp.get("/shape-check")
@login_required
async def shape_check():
"""The end-of-turn question (milestone 439): what did this turn write that
nobody has judged?
The client's write hooks keep a ledger of the definitions each turn wrote;
its Stop hook sends them here. The server decides which carry no judgment
— against the ledger and the project's recorded snippets — records the
outcome, and for a block returns the `reason` the agent is sent back with.
A hook blocks only on a reason it received, so only on a block that was
recorded, and never on the stop that follows one (`phase=after`).
A GET like every plugin endpoint: a read-scoped key runs the plugin, and
this records telemetry the way /report-check does. It changes no ledger
row — the agent's own classify_shapes call does that.
Query:
written (str) — newline-separated `path<TAB>kind<TAB>name` lines,
repo-relative, kind css|sym|file.
phase (str) — check | after. Anything else is a 400.
repo (opt) — working repo remote, resolved like the other arms.
project_id (opt) — explicit scope override.
"""
phase = (request.args.get("phase") or "check").strip()
if phase not in shape_check_svc.PHASES:
return jsonify({"error": f"phase must be one of {list(shape_check_svc.PHASES)}"}), 400
written = shape_check_svc.parse_written(request.args.get("written") or "")
project_id, repo, _unbound = await _project_scope()
body: dict = {"status": "ok", "unjudged": []}
if not project_id or not written:
return jsonify(body)
from scribe.services import access
from scribe.services import coverage as coverage_svc
from scribe.services import shape_ledger as shape_ledger_svc
if not await access.can_read_project(g.user.id, project_id):
return jsonify(body)
repo_key = repo_bindings_svc.normalize_repo_key(repo) if repo else ""
unjudged = await shape_ledger_svc.unjudged_shapes(project_id, written, repo_key=repo_key)
# A snippet recorded AT a shape is the strongest verdict there is — the
# next refresh stamps that row canonical. Until then the ledger cannot
# know, so the recorded locations answer for it: an agent who just
# recorded what it built is not asked again.
if unjudged:
recorded = {
(path, symbol)
for _sid, path, symbol in await coverage_svc._recorded_locations(g.user.id, project_id)
}
unjudged = [u for u in unjudged if (u["path"], u["symbol"]) not in recorded
and not (u["kind"] == "file" and (u["path"], "") in recorded)]
outcome = shape_check_svc.outcome_for(phase, unjudged)
await shape_check_svc.record_shape_check(
g.user.id, outcome, written=len(written), unjudged=len(unjudged),
project_id=project_id,
)
body["outcome"] = outcome
body["unjudged"] = unjudged
if outcome == "blocked":
body["reason"] = shape_check_svc.block_reason(
unjudged, project_id=project_id, repo=repo_key or repo,
)
return jsonify(body)
@plugin_bp.get("/claim-session")
@login_required
async def claim_session():
+138 -4
View File
@@ -369,16 +369,62 @@ def extract_definitions(text: str) -> list[Definition]:
# the human todo holds only shapes a person should look at, while the bodies
# stay in play for the proposer, derive grouping and divergence — the five
# auth views' identical rules were found exactly there. Unscoped `<style>` in
# a .vue and every non-.vue file stay ordinary.
# a .vue and every file that is neither .vue nor .svelte stay ordinary.
_STYLE_OPEN_RE = re.compile(r"^\s*<style\b[^>]*\bscoped\b", re.IGNORECASE)
_STYLE_CLOSE_RE = re.compile(r"^\s*</style\s*>", re.IGNORECASE)
# Svelte scopes the other way round (#4608): every `<style>` is component-
# scoped unless it opts out — `<style global>` (svelte-preprocess) for a whole
# block, `:global(...)` for one selector. Instance `<script>` code is the
# component's own; `<script context="module">` / `<script module>` is the one
# block other files can import from, so its definitions stay ordinary.
_SVELTE_STYLE_OPEN_RE = re.compile(r"^\s*<style\b(?![^>]*\bglobal\b)[^>]*>", re.IGNORECASE)
_SVELTE_SCRIPT_OPEN_RE = re.compile(r"^\s*<script\b([^>]*)>", re.IGNORECASE)
_SVELTE_MODULE_ATTR_RE = re.compile(r"""\bcontext\s*=\s*["']module["']|\bmodule\b""", re.IGNORECASE)
_SCRIPT_CLOSE_RE = re.compile(r"</script\s*>", re.IGNORECASE)
def _svelte_scoped(text: str, defs: list[Definition]) -> set[tuple[str, str]]:
"""`scoped_definitions` for a .svelte file: syms outside a module script,
css inside a non-global `<style>` that is not itself `:global(...)`."""
lines = text.splitlines()
styles: list[tuple[int, int]] = []
modules: list[tuple[int, int]] = []
style_at: int | None = None
module_at: int | None = None
for i, ln in enumerate(lines):
if style_at is None and _SVELTE_STYLE_OPEN_RE.match(ln):
style_at = i
elif style_at is not None and _STYLE_CLOSE_RE.match(ln):
styles.append((style_at, i))
style_at = None
m = _SVELTE_SCRIPT_OPEN_RE.match(ln)
if module_at is None and m and _SVELTE_MODULE_ATTR_RE.search(m.group(1)):
module_at = i
if module_at is not None and _SCRIPT_CLOSE_RE.search(ln):
modules.append((module_at, i))
module_at = None
out: set[tuple[str, str]] = set()
for d in defs:
if d.kind == "sym":
if not any(a <= d.line <= b for a, b in modules):
out.add((d.kind, d.name))
elif any(a <= d.line <= b for a, b in styles):
line = lines[d.line] if 0 <= d.line < len(lines) else ""
if ":global" not in line:
out.add((d.kind, d.name))
return out
def scoped_definitions(path: str, text: str, defs: list[Definition]) -> set[tuple[str, str]]:
"""The (kind, name) pairs among ``defs`` that are one-offs by
construction in this file: every sym in a .vue, and every css rule
that starts inside a `<style scoped>` block. Empty for other files."""
if not (path or "").lower().endswith(".vue"):
that starts inside a `<style scoped>` block; for a .svelte, see
`_svelte_scoped`. Empty for other files."""
lowered = (path or "").lower()
if lowered.endswith(".svelte"):
return _svelte_scoped(text, defs)
if not lowered.endswith(".vue"):
return set()
ranges: list[tuple[int, int]] = []
open_at: int | None = None
@@ -551,6 +597,42 @@ def scannable(path: str) -> bool:
return not path.lower().endswith(_SKIP_SUFFIXES)
# --- the file as a unit (milestone 439) ---------------------------------------
#
# A component is the thing people reuse in a UI, and in a single-file
# component nothing inside is named after it: CatalogueBookCard.svelte defines
# `.card`, `.title` and a `Props`, never `CatalogueBookCard`. A ledger of
# definitions alone therefore could never say "a new card was built where
# CatalogueBookCard is canon". So such a file gets a row of its own.
#
# WHICH FILES, decided by structure rather than by a framework list: a file
# that RENDERS — its markup names classes, or it opens with a <script>,
# <template> or <style> block — and that DEFINES NOTHING NAMED AFTER ITSELF.
# A TSX `function CatalogueBookCard` already has its row; a module of helpers
# is accounted for by its helpers. Emitting a row for every file would put
# every module in every project into the todo at once, which is a flood, not
# a question. No body fingerprint either: every edit would otherwise ask for
# the file's judgment to be re-confirmed.
_RENDER_BLOCK_RE = re.compile(r"^\s*<(?:script|template|style)\b", re.IGNORECASE | re.MULTILINE)
def file_stem(path: str) -> str:
"""`web/src/lib/StatusChip.svelte` → `StatusChip`."""
return path.rsplit("/", 1)[-1].rsplit(".", 1)[0]
def is_file_unit(
path: str, text: str, defs: list[Definition], refs: dict[str, int] | None = None,
) -> bool:
"""Is this file a candidate shape in its own right? See the block above."""
stem = file_stem(path)
if not stem:
return False
if any(d.name == stem for d in defs if d.kind == "sym"):
return False
return bool(refs) or _RENDER_BLOCK_RE.search(text) is not None
class ArchiveShape(NamedTuple):
"""A definition located in a repo archive — what the sync upserts and
the proposer matches. The leading (path, kind, name) triple is the
@@ -612,6 +694,11 @@ def scan_archive(blob: bytes) -> ArchiveScan:
continue
defs = extract_definitions(text)
scoped = scoped_definitions(path, text, defs)
refs = class_references(path, text)
if is_file_unit(path, text, defs, refs):
shapes.append(ArchiveShape(
path, "file", file_stem(path), f"file {path}", "", "",
))
shapes.extend(
ArchiveShape(
path, d.kind, d.name, d.signature, d.body_sha, d.body,
@@ -619,7 +706,6 @@ def scan_archive(blob: bytes) -> ArchiveScan:
)
for d in defs
)
refs = class_references(path, text)
if refs:
references[path] = refs
return ArchiveScan(shapes, references)
@@ -827,6 +913,25 @@ async def compute_coverage(
proposals = shape_ledger.proposal_summary(rows, consumer_paths=consumer_paths)
divergence = shape_ledger.divergence_summary(rows)
derive_new = shape_ledger.derive_new_summary(rows, since=since)
# What `stamps_to_review` would list (#4608). The review tool existed and
# nobody on two projects knew to call it: 132 and 488 weak stamps sat
# electing canons for weeks. The arrival line is the one place a
# project's own session is sure to look, so the count goes there.
try:
review = shape_ledger.stamps_to_review(rows, top=0)
except Exception:
logger.warning("stamp review read failed", exc_info=True)
review = {}
# The write-time figure (milestone 439): how often the end-of-turn
# question was put on this project, and how often it was walked past.
# An audit reads the ledger; this reads whether the reflex held.
try:
from scribe.services import shape_check as shape_check_svc
write_time = await shape_check_svc.window_summary(project_id)
except Exception:
logger.warning("write-time figure read failed", exc_info=True)
write_time = None
return {
"total": len(rows),
"accounted": len(rows) - unclassified,
@@ -850,6 +955,13 @@ async def compute_coverage(
"divergent": divergence["divergent"],
"divergence": divergence["divergence"],
"recheck": divergence["recheck"],
# Unattended stamps below today's floor, and canons whose members no
# longer agree what they are — the judgment queue `stamps_to_review`
# holds (#4608).
"weak_stamps": review.get("weak_count", 0),
"incoherent_canons": review.get("incoherent_count", 0),
# {checked, asked, left, days} from the Stop hook's recorded outcomes.
"write_time": write_time,
# Honesty flag, not decoration: every surface that shows the number
# is expected to carry it through.
"estimate": True,
@@ -1042,4 +1154,26 @@ def coverage_line(coverage: dict) -> str:
line += f"; standing: {', '.join(standing)}"
if coverage.get("recheck"):
line += f"; {coverage['recheck']} judged shape{'s' if coverage['recheck'] != 1 else ''} changed since judged — recheck"
# Rows the hook asserted on evidence the ledger no longer accepts (#4608).
# They no longer elect a canon, but they still claim to be instances
# until someone reads them — the tool to call is named, not implied.
review = []
if coverage.get("weak_stamps"):
n_w = coverage["weak_stamps"]
review.append(f"{n_w} weak stamp{'s' if n_w != 1 else ''}")
if coverage.get("incoherent_canons"):
n_i = coverage["incoherent_canons"]
review.append(f"{n_i} incoherent canon{'s' if n_i != 1 else ''}")
if review:
line += f"; {' · '.join(review)} to judge — stamps_to_review"
# Whether the writers judged what they wrote (milestone 439). Silent until
# the question has been put at least once: a project nobody has written to
# through the plugin has no figure, not a perfect one.
wt = coverage.get("write_time") or {}
if wt.get("checked"):
line += (
f"; written-shape check ({wt.get('days', 7)}d): {wt['checked']} "
f"turn{'s' if wt['checked'] != 1 else ''} checked, {wt.get('asked', 0)} asked, "
f"{wt.get('left', 0)} left unjudged"
)
return line
+12 -20
View File
@@ -287,16 +287,16 @@ def _gate_key(note_type: str) -> str:
async def _gate_setting(user_id: int, key: str, lo: float) -> float:
from scribe.services.settings import get_setting
from scribe.services.settings import bounded_float, get_setting
default = GATE_DEFAULT_THRESHOLDS[key]
try:
value = float(await get_setting(user_id, GATE_THRESHOLD_KEYS[key], str(default)))
raw = await get_setting(user_id, GATE_THRESHOLD_KEYS[key], str(default))
except Exception:
# Fail-open like the rest of the gate: an unreadable setting falls back
# to the measured default rather than blocking or waving through.
value = default
return min(1.0, max(lo, value))
raw = None
return bounded_float(raw, default, lo)
async def gate_bars(user_id: int, note_type: str) -> tuple[float, float]:
@@ -557,16 +557,11 @@ _MAX_DUPLICATE_PAIRS = 200
async def get_duplicate_threshold(user_id: int, kind: str = "snippet") -> float:
"""The user's near-duplicate similarity floor for `kind`, clamped to [0, 1]."""
from scribe.services.settings import get_setting
from scribe.services.settings import bounded_float, get_setting
default = DUPLICATE_DEFAULT_THRESHOLDS[kind]
try:
value = float(await get_setting(
user_id, DUPLICATE_THRESHOLD_KEYS[kind], str(default)
))
except (TypeError, ValueError):
value = default
return min(1.0, max(0.0, value))
raw = await get_setting(user_id, DUPLICATE_THRESHOLD_KEYS[kind], str(default))
return bounded_float(raw, default)
def group_pairs(pairs: list[tuple[int, int, float]]) -> list[list[int]]:
@@ -1120,15 +1115,12 @@ PLAN_MATCH_DEFAULT_THRESHOLD = 0.80
async def get_plan_match_threshold(user_id: int) -> float:
"""The user's plan-gate similarity floor, clamped to [0, 1]."""
from scribe.services.settings import get_setting
from scribe.services.settings import bounded_float, get_setting
try:
value = float(await get_setting(
user_id, PLAN_MATCH_THRESHOLD_KEY, str(PLAN_MATCH_DEFAULT_THRESHOLD)
))
except (TypeError, ValueError):
value = PLAN_MATCH_DEFAULT_THRESHOLD
return min(1.0, max(0.0, value))
raw = await get_setting(
user_id, PLAN_MATCH_THRESHOLD_KEY, str(PLAN_MATCH_DEFAULT_THRESHOLD)
)
return bounded_float(raw, PLAN_MATCH_DEFAULT_THRESHOLD)
def plan_candidate_text(
+29 -26
View File
@@ -2083,16 +2083,18 @@ async def build_write_path_hint(
``stamp_shapes`` turns the same request into the ledger's write-path feed
(#2791): the (kind, name) definitions the hook saw in — or enclosing —
the payload. When the session has PULLED a snippet recently and this
payload references or resembles it, those shapes land as `instance` rows
(classified_by=hook, see shape_ledger.stamp_write_path_instances) and
the result's ``stamped`` lists them. The route passes it only for a
caller allowed to write — a read-scoped key gets the hint, never the
stamp. ``repo_key`` (the hook's remote, normalised) homes a provisional
row for a shape the ledger has not synced yet.
payload references or resembles it, those shapes carry it as a PROPOSAL
(see shape_ledger.suggest_write_path_instances) and the result's
``suggested`` lists them. Never a verdict since milestone 439: the agent
that wrote the code judges it at the end of the turn, shown this as
evidence. The route passes it only for a caller allowed to write — a
read-scoped key gets the hint, never a ledger write. ``repo_key`` (the
hook's remote, normalised) homes a provisional row for a shape the
ledger has not synced yet.
"""
cfg = await get_writepath_config(user_id)
empty = {"context": "", "note_ids": [], "sync_note_ids": [], "config": cfg,
"stamped": [], "divergence": [], "derive": [], "derive_keys": [],
"suggested": [], "divergence": [], "derive": [], "derive_keys": [],
"rule_ids": []}
path = (path or "").strip()
if not cfg["enabled"] or not path:
@@ -2356,18 +2358,18 @@ async def build_write_path_hint(
menu = (placed + scored)[:max(0, top_k - len(synced))]
# The stamp runs whether or not anything is rendered — after dedup, the
# common case is a silent hint and a pulled canon being instantiated.
stamped: list[dict] = []
# The suggestion runs whether or not anything is rendered — after dedup,
# the common case is a silent hint and a pulled canon being instantiated.
suggested: list[dict] = []
if stamp_shapes and pulled:
try:
stamped = await shape_ledger_svc.stamp_write_path_instances(
suggested = await shape_ledger_svc.suggest_write_path_instances(
user_id, project_id, path=path, shapes=stamp_shapes,
code=code or "", pulled=pulled, resembles=resembles,
repo_key=repo_key,
)
except Exception:
logger.warning("Write-path ledger stamping failed", exc_info=True)
logger.warning("Write-path ledger suggestion failed", exc_info=True)
# The in-band button-B check (#2793): the hook named the shapes being
# written; if this directory+kind is canon-dense and a named shape isn't
# (about to be) an instance of that canon, say so NOW — at the write,
@@ -2376,7 +2378,7 @@ async def build_write_path_hint(
if stamp_shapes and project_id:
try:
divergence = await shape_ledger_svc.write_time_divergence(
project_id, path, stamp_shapes, stamped, code or "",
project_id, path, stamp_shapes, suggested, code or "",
)
except Exception:
logger.warning("write-time divergence check failed", exc_info=True)
@@ -2430,7 +2432,7 @@ async def build_write_path_hint(
design_text, design_dedup = await _design_arm(
user_id, project_id, path, set(exclude_derive or []),
)
if not staleness and not synced and not menu and not stamped and not divergence and not derive:
if not staleness and not synced and not menu and not suggested and not divergence and not derive:
if design_text:
return {**empty, "context": design_text, "derive_keys": [design_dedup]}
return empty
@@ -2531,8 +2533,8 @@ async def build_write_path_hint(
if passage:
lines.append(f"> ↳ {passage}")
if stamped:
lines.append(_stamp_line(path, stamped))
if suggested:
lines.append(_suggest_line(path, suggested))
if divergence:
lines.append(_divergence_line(path, divergence))
if derive:
@@ -2694,7 +2696,7 @@ async def build_write_path_hint(
"note_ids": note_ids,
"sync_note_ids": sync_note_ids,
"config": cfg,
"stamped": stamped,
"suggested": suggested,
"divergence": divergence,
"derive": derive,
"derive_keys": [d["key"] for d in derive] + (
@@ -3017,22 +3019,23 @@ def _divergence_line(path: str, divergence: list[dict]) -> str:
)
def _stamp_line(path: str, stamped: list[dict]) -> str:
"""One line saying what the ledger just recorded, so the session can
correct a wrong stamp in the moment rather than an audit finding it."""
def _suggest_line(path: str, suggested: list[dict]) -> str:
"""One line naming the recorded shape this code looks like, as evidence
for the judgment the agent gives at the end of the turn (milestone 439)
— not a record of one already made."""
by_snippet: dict[int, list[str]] = {}
for row in stamped:
for row in suggested:
label = f".{row['symbol']}" if row["kind"] == "css" else row["symbol"]
by_snippet.setdefault(int(row["snippet_id"]), []).append(f"`{label}`")
parts = [
f"{', '.join(names)} → instance of #{sid}"
f"{', '.join(names)} → looks like #{sid}"
for sid, names in by_snippet.items()
]
return (
f"> Shape accounting: recorded at `{path}` — {'; '.join(parts)} "
"(classified_by=hook: you pulled that snippet this session and this "
"code references/resembles it). Not an instance? `classify_shapes` "
"overrides a hook stamp."
f"> Shape accounting: at `{path}` — {'; '.join(parts)} "
"(you pulled that snippet this session and this code names or resembles "
"it). When you judge this turn's shapes, say whether it is: `instance` "
"of it, `variant` with the why, or something else."
)
+18
View File
@@ -122,6 +122,24 @@ async def bindings_for_project(user_id: int, project_id: int) -> list[RepoBindin
return list(rows.scalars().all())
async def is_bound(project_id: int, raw_repo: str) -> str:
"""The normalized key when ``raw_repo`` is bound to ``project_id`` by
anyone, else "". By anyone, not by the caller: a collaborator judging a
shared project's shapes holds no binding of their own, and the question
is whether the repo belongs to the project, not who bound it."""
key = normalize_repo_key(raw_repo or "")
if not key or not project_id:
return ""
async with async_session() as session:
found = await session.execute(
select(RepoBinding.id).where(
RepoBinding.project_id == project_id,
RepoBinding.repo_key == key,
).limit(1)
)
return key if found.first() is not None else ""
async def keys_for_project(user_id: int, project_id: int) -> list[str]:
"""Every repo key bound to a project — the snippet→forge join (#2691).
+3 -6
View File
@@ -66,7 +66,7 @@ from __future__ import annotations
from dataclasses import dataclass
from scribe.services.settings import get_setting
from scribe.services.settings import bounded_float, get_setting
# A budget nobody should be able to set past. Not a tuning value — a guard on
# the worst case, so a mistyped setting cannot turn a menu into a wall of text.
@@ -270,11 +270,8 @@ def dial_for_key(key: str) -> tuple[str, str] | None:
async def floor_for(user_id: int, name: str) -> float:
"""This install's current floor for a surface, clamped to [0, 1]."""
s = get_surface(name)
try:
value = float(await get_setting(user_id, s.floor_key, str(s.floor_default)))
except (TypeError, ValueError):
value = s.floor_default
return min(1.0, max(0.0, value))
raw = await get_setting(user_id, s.floor_key, str(s.floor_default))
return bounded_float(raw, s.floor_default)
async def budget_for(user_id: int, name: str) -> int:
+16
View File
@@ -16,6 +16,22 @@ logger = logging.getLogger(__name__)
SECRET_MASK = "********"
def bounded_float(raw: str | None, default: float, lo: float = 0.0, hi: float = 1.0) -> float:
"""A numeric setting as stored text, parsed and clamped to [lo, hi].
Every similarity bar and retrieval floor is kept as a string and read the
same way: an unparseable value falls back to the DEFAULT, never to 0 (a
floor of 0 admits everything), and a value out of range is pulled back
into it. Pure on purpose: each caller keeps its own `get_setting` read, so
what can fail there — and whether that fails open — stays the caller's.
"""
try:
value = float(raw)
except (TypeError, ValueError):
value = default
return min(hi, max(lo, value))
async def get_admin_setting(key: str, default: str = "") -> str:
"""Read an instance-global setting (one stored on an admin account).
+192
View File
@@ -0,0 +1,192 @@
"""The end-of-turn shape check: what a turn wrote that nobody has judged, and what the agent is asked.
WHY THIS EXISTS (milestone 439)
The shape ledger used to be filled by machinery: extraction found the
definitions, framework rules decided which were one-offs, and the write-path
hook stamped "instance of #N" with nobody reading. The agent that WROTE the
code — the one participant who knows what it is — was asked only in an audit,
weeks later, if anyone ran one. A first instance of a reusable piece was never
asked about at all, so it was never recorded, and the next session rebuilt it.
So the question moves to the moment the knowledge exists. The client's write
hooks keep a ledger of the definitions each turn wrote; its Stop hook sends
them here at the end of the turn. Two jobs live on this side, as with the
report check (services/report_check.py):
- DECIDING what is unjudged, against the ledger and the project's recorded
snippets — something only the server can see.
- OWNING THE WORDS. A hook carries timing and transport (plugin/PACKAGING.md);
the instruction comes from here, so every client asks the same question
and the wording changes in one place.
And it RECORDS every outcome, so "how much did agents leave unjudged after
being asked" is a number rather than an impression. app_logs, for the reason
report_check gives.
"""
from __future__ import annotations
import json
from datetime import datetime, timedelta, timezone
from sqlalchemy import select
from scribe.models import async_session
from scribe.models.app_log import AppLog
# check — the turn's first stop: block if anything is unjudged.
# after — the stop that follows a block: never block again, only record
# whether the agent answered.
PHASES = ("check", "after")
OUTCOMES = ("passed", "blocked", "judged_after_block", "left_after_block")
# How many shapes the reason names before "+N more". Enough to judge a normal
# turn in one call; a turn that wrote more than this generated something.
_LISTED = 25
def _evidence_note(evidence: dict) -> str:
"""One parenthesis of what the machinery holds — offered, never asserted."""
parts = []
looks = evidence.get("looks_like") or {}
if looks.get("snippet_id"):
score = looks.get("score")
at = f" {score:.2f}" if isinstance(score, (int, float)) else ""
parts.append(f"looks like #{looks['snippet_id']}{at}")
stamped = evidence.get("stamped") or {}
if stamped.get("snippet_id"):
parts.append(f"the hook guessed #{stamped['snippet_id']}")
if evidence.get("copies"):
parts.append("other copies of it exist")
if evidence.get("diverges_from"):
parts.append(f"#{evidence['diverges_from']} is canon in this directory")
return f" ({'; '.join(parts)})" if parts else ""
def block_reason(unjudged: list[dict], *, project_id: int, repo: str) -> str:
"""What the agent is told at the end of a turn that wrote unjudged shapes.
Phrased as the practice wanted (note #3565): what to decide and the one
call that records it, not a prohibition. The reusing-code skill owns the
longer version of what each verdict means.
"""
lines = []
for item in unjudged[:_LISTED]:
new = " — new" if item.get("status") == "new" else ""
lines.append(
f"- {item['path']} · {item['symbol']} ({item['kind']}){new}"
f"{_evidence_note(item.get('evidence') or {})}"
)
more = len(unjudged) - _LISTED
if more > 0:
lines.append(f"- … and {more} more (list_shapes(project_id={project_id}, path=…) shows them)")
repo_arg = f', repo="{repo}"' if repo else ""
return (
f"This turn wrote {len(unjudged)} definition{'s' if len(unjudged) != 1 else ''} "
"that nobody has judged yet. You built them, so you know what each one is — "
"record it now, while that is still true, so the next session starts from your "
"answer instead of rebuilding it:\n"
+ "\n".join(lines)
+ "\n\nFor each: something another part of the code should reuse → record it "
"with create_snippet (name, code, when to reach for it, location); built from a "
"recorded snippet → `instance` of it; a deliberate departure from one → "
"`variant` with the why; a genuine one-off → `exempt` with a reason "
"(reason_code one-off-handler, test-helper, scoped-css, pure-helper … indexes it). "
"One call records them all: "
f"classify_shapes(project_id={project_id}{repo_arg}, classifications=[{{path, "
"symbol, kind, status, snippet_id?, reason?, reason_code?}}, …]). "
"`repo` lets a verdict land on a shape the ledger has not synced yet."
)
async def record_shape_check(
user_id: int | None,
outcome: str,
*,
written: int,
unjudged: int,
project_id: int | None = None,
) -> None:
if outcome not in OUTCOMES:
raise ValueError(f"unknown shape-check outcome {outcome!r}")
details: dict = {"outcome": outcome, "written": written, "unjudged": unjudged}
if project_id:
details["project_id"] = project_id
async with async_session() as session:
session.add(AppLog(
category="plugin",
user_id=user_id,
action="shape_check",
details=json.dumps(details),
))
await session.commit()
def outcome_for(phase: str, unjudged: list[dict]) -> str:
"""The recorded outcome: a first stop blocks or passes; the stop after a
block records whether the agent answered, and never blocks."""
if phase == "after":
return "left_after_block" if unjudged else "judged_after_block"
return "blocked" if unjudged else "passed"
def parse_written(raw: str, *, cap: int = 200) -> list[tuple[str, str, str]]:
"""The hook's ledger lines, `path<TAB>kind<TAB>name`, as triples.
Kinds outside the ledger's vocabulary and malformed lines are dropped
rather than echoed into an instruction; duplicates collapse; capped."""
out: list[tuple[str, str, str]] = []
for line in (raw or "").splitlines():
parts = line.split("\t")
if len(parts) != 3:
continue
path, kind, name = (p.strip() for p in parts)
if kind not in ("css", "sym", "file") or not path or not name:
continue
if (path, kind, name) not in out:
out.append((path, kind, name))
if len(out) >= cap:
break
return out
# The window the coverage line reports the write-time figure over. A week is
# long enough to hold a few working sessions on a project and short enough
# that a reflex that started slipping shows up while it is still news.
WINDOW_DAYS = 7
def summarise(outcomes: list[str]) -> dict:
"""{checked, asked, left} from a list of recorded outcomes.
`checked` is turns the question was put to (a first stop that wrote
something): passed + blocked. `asked` is the blocked ones. `left` is
asks the agent walked past — the stop after a block that still had
unjudged shapes. That last number is the slip the milestone exists to
make visible."""
checked = sum(1 for o in outcomes if o in ("passed", "blocked"))
asked = sum(1 for o in outcomes if o == "blocked")
left = sum(1 for o in outcomes if o == "left_after_block")
return {"checked": checked, "asked": asked, "left": left}
async def window_summary(project_id: int, *, days: int = WINDOW_DAYS) -> dict:
"""The write-time figure for one project over the last ``days``."""
since = datetime.now(timezone.utc) - timedelta(days=days)
async with async_session() as session:
rows = (await session.execute(
select(AppLog.details).where(
AppLog.category == "plugin",
AppLog.action == "shape_check",
AppLog.created_at >= since,
)
)).scalars().all()
outcomes: list[str] = []
for raw in rows:
try:
details = json.loads(raw or "{}")
except (TypeError, ValueError):
continue
if details.get("project_id") == project_id:
outcomes.append(details.get("outcome") or "")
return {**summarise(outcomes), "days": days}
+214 -34
View File
@@ -408,6 +408,11 @@ async def mark_canonicals(
fall back to unclassified. Agent judgments are never overwritten.
"""
usable = [(nid, p, s) for nid, p, s in recorded if (s or "").strip()]
# A location naming a FILE and no symbol is a whole-file record: it makes
# no claim about a definition inside (`location_covers`), but it is
# exactly the claim a `file` row asks about (milestone 439).
whole_files = {(p or "").strip(): nid for nid, p, s in recorded
if (p or "").strip() and not (s or "").strip()}
now = datetime.now(timezone.utc)
async with async_session() as session:
rows = (
@@ -426,6 +431,8 @@ async def mark_canonicals(
),
None,
)
if covering is None and row.kind == "file":
covering = whole_files.get(row.path)
if covering is not None and row.status in _MECHANICAL_TODO:
await _judge(session, row, status="canonical", snippet_id=covering,
by="mechanical", reason=None, at=now)
@@ -535,6 +542,7 @@ async def classify_shapes(
classifications: list[dict],
*,
via: str = "agent",
repo: str = "",
) -> dict:
"""Apply a batch of judgments to a project's live ledger rows.
@@ -544,7 +552,19 @@ async def classify_shapes(
item carries one — and a target no live row matches is reported in
``unmatched``, not an error: the tree may simply have moved since the
caller listed. Idempotent by construction.
``repo`` — a repo bound to the project — lets a judgment land on a shape
the ledger has not synced yet (milestone 439): the shape the agent wrote
this turn. Such an item becomes a PROVISIONAL row under that repo, the
same row the write-path hook has always created, with no seen commit;
the next sync confirms it, or marks it vanished until the code reaches
the bound ref, and either way the judgment rides along (a vanished row
that reappears keeps its status). Without ``repo`` an unknown shape is
`unmatched`, exactly as before. An unbound ``repo`` is refused rather
than ignored: a verdict silently dropped reads as one recorded.
"""
from scribe.services import repo_bindings as repo_bindings_svc
from scribe.services import access
from scribe.services import snippets as snippets_svc
@@ -569,9 +589,18 @@ async def classify_shapes(
for sid in sorted(target_ids):
if await snippets_svc.get_snippet(user_id, sid) is None:
raise ValueError(f"snippet {sid} not found (or not readable)")
repo_key = ""
if (repo or "").strip():
repo_key = await repo_bindings_svc.is_bound(project_id, repo)
if not repo_key:
raise ValueError(
f"repo {repo!r} is not bound to project {project_id} — "
"bind_repo it, or omit repo to judge synced shapes only"
)
now = datetime.now(timezone.utc)
classified = 0
provisional = 0
unmatched: list[dict] = []
async with async_session() as session:
rows = (
@@ -592,6 +621,17 @@ async def classify_shapes(
kind = (item.get("kind") or "").strip()
if kind:
matches = [r for r in matches if r.kind == kind]
if not matches and repo_key and item.get("status") != "unclassified":
path = (item.get("path") or "").strip()
symbol = (item.get("symbol") or "").strip()
row = CodeShape(
project_id=project_id, repo_key=repo_key, path=path,
symbol=symbol, kind=kind or snippet_kind(symbol, ""),
)
session.add(row)
by_key.setdefault((path, symbol), []).append(row)
matches = [row]
provisional += 1
if not matches:
unmatched.append({
"path": item.get("path"), "symbol": item.get("symbol"),
@@ -610,7 +650,10 @@ async def classify_shapes(
evidence=item.get("reason"))
classified += 1
await session.commit()
return {"classified": classified, "unmatched": unmatched}
out = {"classified": classified, "unmatched": unmatched}
if provisional:
out["provisional"] = provisional
return out
def rule_matches(row: CodeShape, *, path: str, pattern: str, kind: str) -> bool:
@@ -872,11 +915,13 @@ async def snippet_consumers(user_id: int, note_id: int) -> dict:
# semantic arm scored it above the write-path threshold for this
# very payload. Either is evidence; the pull alone is not.
# Both hold → every shape the hook named at that path, of the snippet's kind,
# becomes instance-of-N with classified_by="hook" and the evidence as reason.
# carries N as a PROPOSAL (milestone 439). Until then it became instance-of-N
# with classified_by="hook" — a verdict nobody read, and the rows written that
# way are what `stamps_to_review` still lists.
#
# A hook row is EVIDENCE, not judgment: it only ever lands on rows nobody has
# judged (unclassified) or rows an earlier hook stamped, never on a canonical
# row or an agent/audit/import judgment. Re-judge with classify_shapes.
# A proposal is EVIDENCE, not judgment: it lands only on rows nobody has
# judged, never touches a status, and is what the end-of-turn question shows
# the agent as "looks like #N". The agent's classify_shapes is the verdict.
# "The write path actually pulled it": a working session's reach. The
# precision comes from the in-play test above, not from this window.
@@ -953,6 +998,10 @@ def shape_form(signature: str, kind: str = "sym") -> str:
"""
if kind == "css":
return "css"
if kind == "file":
# A whole file (milestone 439): its own family, comparable only with
# other files — a component is never told to build from a helper.
return "file"
sig = (signature or "").strip()
if not sig:
return FORM_UNKNOWN
@@ -1047,6 +1096,8 @@ def shape_family(form: str) -> str:
return "value"
if form == "css":
return "css"
if form == "file":
return "file"
return ""
@@ -1066,6 +1117,8 @@ def canon_form(rows: Iterable, snippet_id: int) -> str:
for r in rows:
if r.snippet_id != snippet_id or r.status not in ("canonical", "instance"):
continue
if is_weak_stamp(r):
continue
f = shape_form(getattr(r, "signature", "") or "", r.kind)
if f:
forms[f] = forms.get(f, 0) + 1
@@ -1316,7 +1369,7 @@ def _stamp_allowed(rank: int, mine: str, canon: str) -> bool:
return forms_agree(mine, canon)
async def stamp_write_path_instances(
async def suggest_write_path_instances(
user_id: int,
project_id: int,
*,
@@ -1327,22 +1380,35 @@ async def stamp_write_path_instances(
resembles: dict[int, float] | None = None,
repo_key: str = "",
) -> list[dict]:
"""Land hook evidence as `instance` rows for the shapes being written.
"""Offer hook evidence as a PROPOSAL on the shapes being written.
This used to land the evidence as `instance` rows, classified_by=hook —
a permanent classification written with nobody reading it. That is what
poisoned two ledgers (132 and 488 rows below today's floor, #4608), and
what milestone 439 retires: the agent that wrote the code says what it is,
at the end of the turn, and this evidence is what it is shown when it
does. So the row keeps its status; the candidate canon goes into the
proposal fields (`proposed_snippet_id`, `proposal_basis` "reference" for a
by-name hit or "semantic" for a resemblance, `proposal_score`), where the
end-of-turn question reads it as "looks like #N". A judged row is left
alone entirely.
``shapes`` is the hook's (kind, name) list for ``path``; ``pulled`` is
recent_pulls(); ``resembles`` maps snippet ids the semantic arm scored
for this payload to their score. Returns the rows stamped, each
for this payload to their score. Returns the suggestions, each
{path, symbol, kind, snippet_id, reason} — empty in the common case.
A shape the ledger has no live row for yet (it is being written right
now) gets a PROVISIONAL row under ``repo_key`` — first/last-seen empty —
so the stamp is not lost to the next sync, which either confirms the
shape (sets its seen marker) or stamps it vanished. No repo key → only
existing rows are stamped.
now) gets a PROVISIONAL unclassified row under ``repo_key`` — first/
last-seen empty — so the suggestion survives to the end of the turn and
the next sync either confirms the shape or stamps it vanished. No repo
key → only existing rows carry a suggestion.
Every pulled canon the payload NAMES is still recorded as a `uses` edge:
that is a fact about the code, not a verdict about what it is.
When more than one pulled snippet is in play for a shape, a by-name
reference beats resemblance and the most recent pull breaks ties: a row
holds one canon (the known model limit logged on #2790).
reference beats resemblance and the most recent pull breaks ties.
"""
from scribe.services import access
from scribe.services import snippets as snippets_svc
@@ -1395,8 +1461,7 @@ async def stamp_write_path_instances(
for bucket in in_play.values():
bucket.sort(key=lambda t: (t[0], t[1]), reverse=True)
now = datetime.now(timezone.utc)
stamped: list[dict] = []
suggested: list[dict] = []
async with async_session() as session:
rows = (
await session.execute(
@@ -1448,9 +1513,12 @@ async def stamp_write_path_instances(
by_key[(name, kind)] = row
elif not (row.status in _MECHANICAL_TODO or row.classified_by == "hook"):
continue # a judgment — or the canon itself — stands
await _judge(session, row, status="instance", snippet_id=sid, by="hook",
reason=why, at=now)
stamped.append({
# A proposal, never a status: the writer decides (milestone 439).
row.proposed_snippet_id = sid
row.proposal_basis = "reference" if _rank >= 2 else "semantic"
row.proposal_score = resembles.get(sid) if _rank < 2 else 1.0
row.proposal_group = None
suggested.append({
"path": path, "symbol": name, "kind": kind,
"snippet_id": sid, "reason": why,
})
@@ -1466,9 +1534,9 @@ async def stamp_write_path_instances(
[t[2] for t in bucket if t[0] == 2],
basis="hook", evidence="write path: pulled the snippet, payload names its symbol",
)
if stamped:
if suggested:
await session.commit()
return stamped
return suggested
# --- the mechanical proposer (#2792): the machine proposes, judgment classifies
@@ -1985,6 +2053,10 @@ async def apply_derive_groups(project_id: int) -> int:
CodeShape.status.in_(_MECHANICAL_TODO),
CodeShape.vanished_at.is_(None),
CodeShape.proposed_snippet_id.is_(None),
# A file row is named by its stem, and `index`, `+page`
# and `__init__` repeat by convention, not by copying —
# a derive family of them would be pure noise.
CodeShape.kind != "file",
)
)
).scalars().all()
@@ -2172,8 +2244,11 @@ async def confirm_proposals(
# A canon dominates a directory+kind when at least this many siblings are
# judged (canonical/instance) and this share of them answer to one snippet.
# How much a payload must resemble a canon before the hook may assert, with
# nobody watching, that a shape IS an instance of it.
# How much a payload must resemble a canon before the hook SUGGESTS it as the
# shape's canon — and, for the rows stamped before milestone 439, the bar below
# which a stored stamp is weak (`is_weak_stamp`). Until milestone 439 this was
# the bar for the hook to ASSERT, unattended, that a shape IS an instance; the
# hook no longer asserts anything, and the history below is why.
#
# WHY A FLOOR AT ALL. There was none: any score the semantic arm produced
# counted, so 0.69 asserted as confidently as 0.95, and the score went into
@@ -2190,6 +2265,25 @@ async def confirm_proposals(
# suggestion, not the same.
_RESEMBLE_MIN = 0.80
def is_weak_stamp(row) -> bool:
"""Was this row written by the hook, unattended, on a resemblance below
the floor it would have to clear today?
ONE PREDICATE FOR "WEAK", read by the review surface that lists these rows
and by every reader that would otherwise take them as evidence. The floor
guards new writes only; the rows stamped before it (#4204) are still
stored, and until an agent re-judges them they must not speak. They are
not withdrawn here — `stamps_to_review` still lists them for judgment —
they are simply not counted, which is the module's rule for evidence it
cannot trust: quieter, never more confident (#4608).
"""
if getattr(row, "classified_by", None) not in _UNATTENDED_BY:
return False
score = stamp_score(getattr(row, "reason", None))
return score is not None and score < _RESEMBLE_MIN
_DENSITY_MIN_JUDGED = 3
_DENSITY_SHARE = 0.6
@@ -2229,6 +2323,11 @@ def dominant_canon(rows: Iterable[CodeShape]) -> tuple[int, int, int] | None:
counts: dict[int, int] = {}
judged = 0
for r in rows:
# A weak stamp is not a judgment: counted, it let a YAML CI snippet
# "dominate" a Svelte directory and every write there was told it
# diverged from it (#4608).
if is_weak_stamp(r):
continue
if r.status in ("canonical", "instance") and r.snippet_id is not None:
judged += 1
counts[r.snippet_id] = counts.get(r.snippet_id, 0) + 1
@@ -2302,15 +2401,15 @@ async def canon_density(
async def write_time_divergence(
project_id: int, path: str, shapes: list[tuple[str, str]], stamped: list[dict],
project_id: int, path: str, shapes: list[tuple[str, str]], suggested: list[dict],
code: str = "",
) -> list[dict]:
"""The in-band check for the shapes the hook named at ``path``: for each
kind whose directory has a dominant canon, the named shapes that are
not (already or just now) that canon's instance/canonical — new or
not that canon's instance/canonical (judged, or just suggested as it) — new or
unclassified rows only; a judged shape is not re-litigated at every
edit. Returns [{symbol, kind, canon_snippet_id, instances, judged}]."""
just_stamped = {(s["symbol"], s["kind"]): s["snippet_id"] for s in stamped}
just_stamped = {(s["symbol"], s["kind"]): s["snippet_id"] for s in suggested}
out: list[dict] = []
if not shapes:
return out
@@ -2488,6 +2587,16 @@ async def flag_divergence(project_id: int, *, since: datetime | None) -> int:
if r.status not in _MECHANICAL_TODO:
continue
if r.diverges_from is not None:
# A flag is a mechanical prompt resting on one premise —
# "this canon dominates here". When the premise no longer
# holds (the rows that elected it were weak stamps, #4608,
# or were re-judged away) the prompt is false, and a false
# prompt on every write is how the flag stopped being
# read. Withdrawing it judges nothing: the row stays
# unclassified, and a judgment is never touched here.
if dom is None or dom[0] != r.diverges_from:
r.diverges_from = None
continue
flagged += 1
continue
if dom is None or r.created_at is None or r.created_at <= since:
@@ -2514,6 +2623,83 @@ async def flag_divergence(project_id: int, *, since: datetime | None) -> int:
return flagged
# --- the end-of-turn question (milestone 439): what did this turn write that
# nobody has judged? ---------------------------------------------------------
#
# The write hooks keep a session ledger of the definitions each turn wrote; at
# the end of the turn the client asks this, and the agent that wrote them
# answers. "Judged" means a verdict SOMEONE READ: an agent's, an audit's, an
# import's. The sync's `scoped` stamp, `unclassified`, a hook stamp and no row
# at all are all unjudged — each is a machine's guess or nothing, and the
# point of the question is that the writer, who knows, says.
def _is_judged(row) -> bool:
return (
row.status not in _MECHANICAL_TODO
and row.classified_by not in _UNATTENDED_BY
)
async def unjudged_shapes(
project_id: int, written: list[tuple[str, str, str]], *, repo_key: str = "",
) -> list[dict]:
"""The (path, kind, name) shapes in ``written`` that carry no judgment.
Each comes back as {path, kind, symbol, status, evidence} — status is the
row's, or "new" when the ledger has none yet; evidence is what the
machinery holds that the writer may want to weigh: the proposer's
candidate canon, a hook's stamp, a divergence flag. Evidence, never a
verdict — the writer is asked because the machine's guesses were the
thing that poisoned two ledgers. Order follows ``written``; duplicates
collapse. The caller has already resolved the project and checked access.
"""
wanted: list[tuple[str, str, str]] = []
for path, kind, name in written:
key = ((path or "").strip(), (kind or "").strip(), (name or "").strip())
if all(key) and key not in wanted:
wanted.append(key)
if not project_id or not wanted:
return []
paths = sorted({p for p, _k, _n in wanted})
query = select(CodeShape).where(
CodeShape.project_id == project_id,
CodeShape.vanished_at.is_(None),
CodeShape.path.in_(paths),
)
if repo_key:
query = query.where(CodeShape.repo_key == repo_key)
async with async_session() as session:
rows = (await session.execute(query)).scalars().all()
by_key = {(r.path, r.kind, r.symbol): r for r in rows}
out: list[dict] = []
for path, kind, name in wanted:
row = by_key.get((path, kind, name))
if row is not None and _is_judged(row):
continue
evidence: dict = {}
if row is not None:
if row.proposed_snippet_id:
evidence["looks_like"] = {
"snippet_id": row.proposed_snippet_id,
"basis": row.proposal_basis,
"score": row.proposal_score,
}
if row.classified_by in _UNATTENDED_BY and row.snippet_id:
evidence["stamped"] = {
"snippet_id": row.snippet_id, "score": stamp_score(row.reason),
}
if row.proposal_group:
evidence["copies"] = row.proposal_group
if row.diverges_from:
evidence["diverges_from"] = row.diverges_from
out.append({
"path": path, "kind": kind, "symbol": name,
"status": row.status if row is not None else "new",
"evidence": evidence,
})
return out
# How many of a canon's rows a review listing shows before "…" — enough to
# judge from, not the whole table.
_REVIEW_ROWS_SHOWN = 12
@@ -2565,11 +2751,9 @@ def stamps_to_review(rows: Iterable[CodeShape], *, top: int = 10) -> dict:
live = [r for r in rows if r.vanished_at is None]
weak = []
for r in live:
if r.status not in _NEEDS_TARGET or r.classified_by != "hook":
if r.status not in _NEEDS_TARGET or not is_weak_stamp(r):
continue
score = stamp_score(r.reason)
if score is None or score >= _RESEMBLE_MIN:
continue
weak.append({
"path": r.path, "symbol": r.symbol, "kind": r.kind,
"status": r.status, "snippet_id": r.snippet_id,
@@ -2602,11 +2786,7 @@ def stamps_to_review(rows: Iterable[CodeShape], *, top: int = 10) -> dict:
# The listing below is capped; without this you cannot tell a
# canon with twelve strangers from one with three hundred.
"stranger_count": len(coh["strangers"]),
"weak_rows": sum(
1 for r in members
if r.classified_by in _UNATTENDED_BY
and (stamp_score(r.reason) or 1.0) < _RESEMBLE_MIN
),
"weak_rows": sum(1 for r in members if is_weak_stamp(r)),
# The rows that do NOT fit — not the first dozen members. The
# reader's question is "which ones are wrong", and a sample of
# the majority cannot answer it.
+5
View File
@@ -198,6 +198,11 @@ TOPICS: tuple[Topic, ...] = (
Topic("reuse recorded shapes; record at first build", "skill:reusing-code",
("create_snippet", "when_to_use", "first build", "second copy"),
"prior art offered beside a write is not noise", index=("create_snippet",)),
# Milestone 439: the writer judges what it wrote, at the end of the turn,
# asked by the plugin's Stop hook — the moment the knowledge exists.
Topic("judge what you wrote before the turn ends", "skill:reusing-code",
("before the turn ends", "classify_shapes", 'kind: "file"'),
"you are the one who knows what the code you just wrote is"),
Topic("report back where the work stands", "skill:reporting-back", ("reporting-back", "placement"),
"take the placement from the record"),
Topic("the operator's own reply shapes come first", "skill:reporting-back",
+235 -36
View File
@@ -322,15 +322,16 @@ async def test_sync_refiles_rows_whose_snippet_was_purged(seeded):
@pytest.mark.integration
async def test_write_path_stamp_is_evidence_that_yields_to_judgment(seeded):
"""Pulled + referenced → every named shape of the snippet's kind becomes
an instance row, classified_by=hook, carrying the evidence as reason. A
later agent judgment on one of them stands against a re-stamp; the hook
may only overwrite nobody's judgment or its own. The outsider stamps
nothing (write-gated like every other ledger write)."""
async def test_write_path_evidence_is_a_proposal_never_a_verdict(seeded):
"""Milestone 439. Pulled + referenced → every named shape of the
snippet's kind CARRIES the snippet as a proposal (basis "reference"),
and its status is untouched: the agent that wrote it says what it is.
A judgment is never touched, not even by a proposal. The outsider
writes nothing (write-gated like every other ledger write). The uses
edge — a fact about the code — is still recorded."""
from datetime import datetime, timezone
from scribe.services.shape_ledger import stamp_write_path_instances
from scribe.services.shape_ledger import suggest_write_path_instances
owner, other, pid, sid = (
seeded["owner"], seeded["other"], seeded["pid"], seeded["snippet"]
@@ -338,84 +339,98 @@ async def test_write_path_stamp_is_evidence_that_yields_to_judgment(seeded):
pulled = {sid: datetime.now(timezone.utc)}
code = "app = factory()\nreturn app\n" # references the snippet's symbol
assert await stamp_write_path_instances(
assert await suggest_write_path_instances(
other, pid, path="src/app.py", shapes=[("sym", "make_app")],
code=code, pulled=pulled,
) == []
stamped = await stamp_write_path_instances(
suggested = await suggest_write_path_instances(
owner, pid, path="src/app.py",
shapes=[("sym", "make_app"), ("sym", "Config"), ("css", "nope")],
code=code, pulled=pulled,
)
assert {s["symbol"] for s in stamped} == {"make_app", "Config"} # css skipped: no css canon
rows, _ = await list_project_shapes(owner, pid, snippet_id=sid)
assert {s["symbol"] for s in suggested} == {"make_app", "Config"} # css skipped: no css canon
rows, _ = await list_project_shapes(owner, pid, path="src/app.py")
by_symbol = {r.symbol: r for r in rows}
assert by_symbol["make_app"].status == "instance"
assert by_symbol["make_app"].classified_by == "hook"
assert by_symbol["make_app"].reason == f"hook: pulled #{sid}; payload references `factory`"
for name in ("make_app", "Config"):
row = by_symbol[name]
assert row.status == "unclassified" and row.classified_by is None, name
assert (row.proposed_snippet_id, row.proposal_basis) == (sid, "reference"), name
# The by-name reference is still recorded as a call-site fact.
users, _ = await list_project_shapes(owner, pid, uses=sid)
assert {"make_app", "Config"} <= {r.symbol for r in users}
# A judgment lands; the next stamp must leave it alone but may re-stamp
# its own earlier row.
# A judgment stands: the next suggestion leaves the judged row alone.
await classify_shapes(owner, pid, [
{"path": "src/app.py", "symbol": "make_app", "status": "exempt",
"reason": "the app factory is its own thing"},
])
again = await stamp_write_path_instances(
again = await suggest_write_path_instances(
owner, pid, path="src/app.py",
shapes=[("sym", "make_app"), ("sym", "Config")], code=code, pulled=pulled,
)
assert {s["symbol"] for s in again} == {"Config"}
rows, _ = await list_project_shapes(owner, pid, path="src/app.py")
by_symbol = {r.symbol: r for r in rows}
assert by_symbol["make_app"].status == "exempt"
assert by_symbol["Config"].status == "instance"
assert {r.symbol: r.status for r in rows}["make_app"] == "exempt"
# Neither pulled nor in play → nothing, even with shapes named.
assert await stamp_write_path_instances(
assert await suggest_write_path_instances(
owner, pid, path="src/util.py", shapes=[("sym", "helper")],
code="print('unrelated')", pulled=pulled,
) == []
# And nothing the hook does writes classified_by="hook" any more.
async with async_session() as s:
hook_rows = (await s.execute(select(CodeShape).where(
CodeShape.project_id == pid, CodeShape.classified_by == "hook",
))).scalars().all()
assert hook_rows == []
@pytest.mark.integration
async def test_a_brand_new_shape_gets_a_provisional_row_the_sync_settles(seeded):
async def test_a_brand_new_shape_gets_a_provisional_row_carrying_the_suggestion(seeded):
"""The shape being written right now has no ledger row yet. With the
hook's repo key it gets a provisional one — seen markers empty — so the
stamp survives until the next sync, which confirms it (sets the marker)
or stamps it vanished. Without a repo key only existing rows are touched."""
hook's repo key it gets a provisional, UNCLASSIFIED one — seen markers
empty — carrying the suggestion to the end-of-turn question; the sync
confirms it or stamps it vanished. Without a repo key only existing rows
carry a suggestion."""
from datetime import datetime, timezone
from scribe.services.shape_ledger import stamp_write_path_instances
from scribe.services.shape_ledger import suggest_write_path_instances, unjudged_shapes
owner, pid, sid = seeded["owner"], seeded["pid"], seeded["snippet"]
pulled = {sid: datetime.now(timezone.utc)}
code = "def build():\n return factory()\n"
assert await stamp_write_path_instances(
assert await suggest_write_path_instances(
owner, pid, path="src/new.py", shapes=[("sym", "build")],
code=code, pulled=pulled, # no repo_key
) == []
stamped = await stamp_write_path_instances(
suggested = await suggest_write_path_instances(
owner, pid, path="src/new.py", shapes=[("sym", "build")],
code=code, pulled=pulled, repo_key=REPO,
)
assert [s["symbol"] for s in stamped] == ["build"]
assert [s["symbol"] for s in suggested] == ["build"]
async with async_session() as s:
row = (await s.execute(select(CodeShape).where(
CodeShape.project_id == pid, CodeShape.path == "src/new.py",
))).scalar_one()
assert row.status == "instance" and row.classified_by == "hook"
assert row.status == "unclassified" and row.classified_by is None
assert row.proposed_snippet_id == sid
# "Unset" is the column's empty default — the markers are non-null
# Text, and the sync is what first fills them.
assert row.first_seen_commit == "" and row.last_seen_commit == ""
# The sync sees the shape in the tree → confirmed, stamp intact.
# The end-of-turn question shows the suggestion as evidence.
asked = await unjudged_shapes(pid, [("src/new.py", "sym", "build")], repo_key=REPO)
assert asked[0]["evidence"]["looks_like"]["snippet_id"] == sid
# The sync sees the shape in the tree → confirmed, still unjudged.
await sync_repo_shapes(
pid, REPO, SHAPES + [("src/new.py", "sym", "build")], seen_marker="abc123",
)
rows, _ = await list_project_shapes(owner, pid, path="src/new.py")
assert rows[0].status == "instance" and rows[0].last_seen_commit == "abc123"
assert rows[0].status == "unclassified" and rows[0].last_seen_commit == "abc123"
# The sync no longer sees it → vanished, out of the live accounting.
await sync_repo_shapes(pid, REPO, SHAPES, seen_marker="def456")
@@ -866,19 +881,19 @@ async def test_a_second_confirm_dialog_is_detected_and_named(seeded):
# In-band: the hook names the shape at write time → the check names the canon.
named = await write_time_divergence(
pid, f"{comp}/Danger.vue", [("sym", "confirmDanger")], stamped=[]
pid, f"{comp}/Danger.vue", [("sym", "confirmDanger")], suggested=[]
)
assert named == [{"symbol": "confirmDanger", "kind": "sym", "canon_snippet_id": sid,
"instances": 4, "judged": 4}]
# ...but an already-judged shape, or one just stamped as the canon's
# instance, is not re-litigated.
assert await write_time_divergence(pid, f"{comp}/Trash.vue", [("sym", "onTrash")], stamped=[]) == []
assert await write_time_divergence(pid, f"{comp}/Trash.vue", [("sym", "onTrash")], suggested=[]) == []
assert await write_time_divergence(
pid, f"{comp}/New.vue", [("sym", "onNew")],
stamped=[{"symbol": "onNew", "kind": "sym", "snippet_id": sid}],
suggested=[{"symbol": "onNew", "kind": "sym", "snippet_id": sid}],
) == []
# A directory with no dominant canon is silent.
assert await write_time_divergence(pid, "src/other.py", [("sym", "thing")], stamped=[]) == []
assert await write_time_divergence(pid, "src/other.py", [("sym", "thing")], suggested=[]) == []
# The judgment answers the question and clears the flag.
await classify_shapes(owner, pid, [
@@ -889,6 +904,84 @@ async def test_a_second_confirm_dialog_is_detected_and_named(seeded):
assert total == 0
@pytest.mark.integration
async def test_weak_stamps_elect_no_canon_and_their_flags_are_withdrawn(seeded):
"""#4608, the poisoned-ledger case as it stood on two projects: a
directory's "dominant canon" was elected entirely by hook stamps written
at 0.68-0.72, before the 0.80 floor, so every new shape there was told it
diverged from a YAML CI snippet. The same fixture as the acceptance case
above, with one difference — the four instances are weak hook stamps
rather than an audit — and the prompt must fall silent: at refresh (the
standing flag is withdrawn) and at write time. The stamps themselves are
left exactly as they were, for `stamps_to_review` to put before a judge."""
from datetime import datetime, timedelta, timezone
from sqlalchemy import update
from scribe.services import snippets as snippets_svc
from scribe.services.shape_ledger import (
_RESEMBLE_REASON, flag_divergence, propose_for_repo, write_time_divergence,
)
owner, pid = seeded["owner"], seeded["pid"]
canon = await snippets_svc.create_snippet(
owner, name="cls_confirm_factory",
code="export async function factory(): Promise<boolean> {\n return true;\n}\n",
language="typescript", repo="Widget",
path="frontend/src/composables/useConfirm.ts", symbol="factory",
project_id=pid,
)
sid = int(canon.id)
comp = "frontend/src/components"
names = ("Trash", "Delete", "Remove", "Restore")
base = _defs(
*[(f"{comp}/{n}.vue", "sym", f"on{n}", f"async function on{n}() {{",
f"async function on{n}() {{\n const ok = await factory();\n if (!ok) return;\n}}")
for n in names],
)
await sync_repo_shapes(pid, REPO, base, seen_marker="aaa111")
await classify_shapes(owner, pid, [
{"path": f"{comp}/{n}.vue", "symbol": f"on{n}", "status": "instance", "snippet_id": sid}
for n in names
], via="audit")
previous = datetime.now(timezone.utc)
later = base + _defs(
(f"{comp}/Danger.vue", "sym", "confirmDanger", "function confirmDanger() {",
"function confirmDanger() {\n return window.confirm('Really?');\n}"),
)
await sync_repo_shapes(pid, REPO, later, seen_marker="bbb222")
with _quiet_semantic():
await propose_for_repo(owner, pid, REPO, later)
since = previous - timedelta(seconds=1)
# Judged by an audit, the canon dominates and the new shape is flagged.
assert await flag_divergence(pid, since=since) == 1
# Now make the four what the poisoned ledgers held: unattended stamps
# below today's floor.
weak = _RESEMBLE_REASON.format(sid=sid, score=0.72)
async with async_session() as session:
await session.execute(
update(CodeShape)
.where(CodeShape.project_id == pid, CodeShape.snippet_id == sid)
.values(classified_by="hook", reason=weak)
)
await session.commit()
assert await flag_divergence(pid, since=since) == 0
rows, total = await list_project_shapes(owner, pid, flag="divergence")
assert total == 0, "a flag whose canon no longer dominates is withdrawn"
assert await write_time_divergence(
pid, f"{comp}/Danger.vue", [("sym", "confirmDanger")], suggested=[]
) == []
# Nothing was un-judged: the stamps stand until someone reads them.
async with async_session() as session:
kept = (await session.execute(
select(CodeShape).where(CodeShape.project_id == pid, CodeShape.snippet_id == sid)
)).scalars().all()
assert len(kept) == 4
assert all(r.status == "instance" and r.classified_by == "hook" for r in kept)
@pytest.mark.integration
async def test_a_semantic_miss_does_not_silence_the_prompt(seeded):
"""#4208, reversed on measurement: a semantic miss is not evidence.
@@ -1060,3 +1153,109 @@ async def test_history_records_what_was_used_when_and_drift_asks_for_a_recheck(s
# reads nothing.
assert len((await shape_history(owner, pid, "src"))["events"]) == 6
assert await shape_history(other, pid, "src/app.py") == {}
# --- milestone 439: the writer judges what it wrote, before the sync has seen it
@pytest.mark.integration
async def test_a_shape_written_this_turn_is_judged_before_the_sync_sees_it(seeded):
"""The end-of-turn verdict lands on a shape no sync has read yet: with a
bound `repo` it becomes a provisional row and is judged, the sync that
then reads the shape keeps the judgment, and a sync that does NOT have it
yet (the code has not reached the bound ref) keeps it through the vanish
and the reappearance. Without `repo` the old contract holds."""
from scribe.services import repo_bindings as repo_bindings_svc
owner, pid = seeded["owner"], seeded["pid"]
await repo_bindings_svc.set_binding(owner, REPO, pid)
verdict = [{"path": "web/src/lib/StatusChip.svelte", "symbol": "tone",
"kind": "sym", "status": "exempt",
"reason": "the chip's own colour switch", "reason_code": "pure-helper"}]
out = await classify_shapes(owner, pid, verdict)
assert out["unmatched"] == [{"path": "web/src/lib/StatusChip.svelte", "symbol": "tone"}]
with pytest.raises(ValueError, match="not bound"):
await classify_shapes(owner, pid, verdict, repo="git.example.com/alice/elsewhere")
out = await classify_shapes(owner, pid, verdict, repo=REPO)
assert out == {"classified": 1, "unmatched": [], "provisional": 1}
# A sync that has not got the code yet: the row vanishes, judgment intact…
await sync_repo_shapes(pid, REPO, SHAPES, seen_marker="c1")
# …and the sync that has it brings the row back with the verdict.
await sync_repo_shapes(
pid, REPO, SHAPES + [("web/src/lib/StatusChip.svelte", "sym", "tone")], seen_marker="c2",
)
rows, _ = await list_project_shapes(owner, pid, path="web/src/lib/StatusChip.svelte")
assert len(rows) == 1
row = rows[0]
assert (row.status, row.classified_by, row.reason_code) == ("exempt", "agent", "pure-helper")
assert row.vanished_at is None and row.last_seen_commit == "c2"
@pytest.mark.integration
async def test_unjudged_is_what_nobody_read(seeded):
"""The end-of-turn question: a shape with no row, an unclassified one, a
sync-scoped one and a hook stamp are all asked about; an agent's or an
audit's verdict is not. Evidence travels; a verdict is never inferred."""
from sqlalchemy import update
from scribe.services.shape_ledger import _RESEMBLE_REASON, unjudged_shapes
owner, pid, sid = seeded["owner"], seeded["pid"], seeded["snippet"]
await classify_shapes(owner, pid, [
{"path": "src/app.py", "symbol": "make_app", "status": "instance", "snippet_id": sid},
{"path": "src/util.py", "symbol": "helper", "status": "exempt", "reason": "x"},
])
async with async_session() as session:
await session.execute(
update(CodeShape)
.where(CodeShape.project_id == pid, CodeShape.symbol == "helper")
.values(classified_by="hook", status="instance", snippet_id=sid,
reason=_RESEMBLE_REASON.format(sid=sid, score=0.7))
)
await session.commit()
written = [
("src/app.py", "sym", "make_app"), # judged by an agent → not asked
("src/app.py", "sym", "Config"), # unclassified → asked
("src/util.py", "sym", "helper"), # a hook stamp → asked, with its evidence
("src/new.py", "sym", "fresh"), # no row → asked as new
("src/new.py", "sym", "fresh"), # duplicates collapse
]
got = await unjudged_shapes(pid, written, repo_key=REPO)
assert [(g["symbol"], g["status"]) for g in got] == [
("Config", "unclassified"), ("helper", "instance"), ("fresh", "new"),
]
assert got[1]["evidence"]["stamped"] == {"snippet_id": sid, "score": 0.7}
assert got[2]["evidence"] == {}
assert await unjudged_shapes(pid, []) == []
@pytest.mark.integration
async def test_a_whole_file_snippet_makes_its_file_row_canonical(seeded):
"""Milestone 439: a snippet recorded at a component's path, with no
symbol, is the canon for that file row — the claim a whole-file record
makes and a definition row never could. It still covers no definition
inside the file."""
from scribe.services import snippets as snippets_svc
from scribe.services.coverage import _recorded_locations
from scribe.services.shape_ledger import mark_canonicals
owner, pid = seeded["owner"], seeded["pid"]
await sync_repo_shapes(pid, REPO, SHAPES + [
("web/lib/StatusChip.svelte", "file", "StatusChip"),
("web/lib/StatusChip.svelte", "css", "chip"),
], seen_marker="c1")
chip = await snippets_svc.create_snippet(
owner, name="StatusChip — the one status pill",
code="<span class=\"chip\">{label}</span>", language="svelte",
repo="Widget", path="web/lib/StatusChip.svelte", project_id=pid,
)
await mark_canonicals(pid, await _recorded_locations(owner, pid))
rows, _ = await list_project_shapes(owner, pid, path="web/lib/StatusChip.svelte")
by = {(r.kind, r.symbol): r for r in rows}
assert by[("file", "StatusChip")].status == "canonical"
assert by[("file", "StatusChip")].snippet_id == int(chip.id)
assert by[("css", "chip")].status == "unclassified"
+77
View File
@@ -310,6 +310,7 @@ def test_scan_archive_returns_definitions_and_references_from_one_walk():
scan = scan_archive(_tarball(tree))
assert isinstance(scan, ArchiveScan)
assert [(d.path, d.kind, d.name) for d in scan.definitions] == TREE_SHAPES + [
("web/Card.vue", "file", "Card"), # a component is a unit (milestone 439)
("web/Card.vue", "css", "card"),
]
# Only files whose markup names a class appear; the .py/.css files don't.
@@ -823,6 +824,49 @@ def test_scoped_definitions_are_vue_script_setup_and_scoped_style_only():
assert scoped_definitions("src/a.py", "def load():\n pass\n", extract_definitions("def load():\n pass\n")) == set()
def test_svelte_scopes_by_default_and_its_module_script_stays_importable():
"""#4608: Svelte scopes the other way round from Vue — every <style> is
the component's own unless it opts out, and only <script module> /
<script context="module"> can be imported elsewhere. Without this a
Svelte project's every `.card` and `.title` sat in the human todo."""
from scribe.services.coverage import extract_definitions, scoped_definitions
svelte = (
"<script module lang=\"ts\">\n"
"export function stageOf(x: number) {\n return x;\n}\n"
"</script>\n\n"
"<script lang=\"ts\">\n"
"function toggle() {\n open = !open;\n}\n"
"</script>\n\n"
"<div class=\"card\"></div>\n\n"
"<style>\n.card {\n padding: 1rem;\n}\n:global(.toast) {\n color: red;\n}\n</style>\n"
)
defs = extract_definitions(svelte)
names = {(d.kind, d.name) for d in defs}
assert {("sym", "stageOf"), ("sym", "toggle"), ("css", "card")} <= names
scoped = scoped_definitions("web/src/lib/A.svelte", svelte, defs)
assert ("sym", "toggle") in scoped and ("css", "card") in scoped
assert ("sym", "stageOf") not in scoped, "a module script's export is importable"
assert not any(n == "toast" for _, n in scoped), ":global(...) opts a selector out"
# The legacy spelling of the module block, and a whole global style block.
legacy = (
"<script context=\"module\">\nexport function shared() {\n return 1;\n}\n</script>\n"
"<style global>\n.page {\n margin: 0;\n}\n</style>\n"
)
assert scoped_definitions("web/src/lib/B.svelte", legacy, extract_definitions(legacy)) == set()
def test_the_line_names_the_review_queue_and_the_tool_that_holds_it():
"""#4608: `stamps_to_review` existed and two projects never called it,
because nothing on arrival said there was anything to review."""
from scribe.services.coverage import coverage_line
base = {"accounted": 10, "total": 10, "counts": {"exempt": 10},
"unclassified": 0, "computed_at": "2026-09-30T00:00:00"}
assert "stamps_to_review" not in coverage_line(base)
line = coverage_line({**base, "weak_stamps": 132, "incoherent_canons": 3})
assert line.endswith("; 132 weak stamps · 3 incoherent canons to judge — stamps_to_review")
assert coverage_line({**base, "weak_stamps": 1}).endswith("; 1 weak stamp to judge — stamps_to_review")
@pytest.mark.integration
async def test_binding_ref_is_the_branch_the_ledger_follows(seeded):
"""#2873: a binding that names a ref is read at that ref (not the forge's
@@ -843,3 +887,36 @@ async def test_binding_ref_is_the_branch_the_ledger_follows(seeded):
coverage = await compute_coverage(uid, pid, selector=_selector(_tarball(TREE)))
assert coverage["repos"][0]["ref"] == "main"
@pytest.mark.parametrize("path,text,unit", [
# A single-file component: nothing inside carries its name.
("web/lib/StatusChip.svelte", "<script>\nfunction tone() {\n return 1;\n}\n</script>\n<span class=\"chip\"></span>\n", True),
("web/views/Card.vue", "<template><div class=\"card\"/></template>\n", True),
# A TSX component already has its row: the function named after the file.
("web/Card.tsx", "export function Card() {\n return <div className=\"card\" />;\n}\n", False),
# A JSX file whose markup names classes but whose component is named otherwise.
("web/index.jsx", "function App() {\n return <div className=\"app\" />;\n}\n", True),
# Modules render nothing: their definitions account for them.
("internal/api/upnext.go", "func stageOf(x int) int {\n\treturn x\n}\n", False),
("src/app/util.py", "def helper():\n pass\n", False),
("web/button.css", ".btn {\n color: red;\n}\n", False),
])
def test_a_file_is_a_unit_when_it_renders_and_nothing_inside_carries_its_name(path, text, unit):
"""Milestone 439: one structural test, not a framework list."""
from scribe.services.coverage import class_references, extract_definitions, is_file_unit
assert is_file_unit(path, text, extract_definitions(text), class_references(path, text)) is unit
def test_the_line_says_whether_writers_judged_what_they_wrote():
"""Milestone 439: the write-time figure — silent until the question has
been put, then turns checked, asked, and asks walked past."""
from scribe.services.coverage import coverage_line
base = {"accounted": 10, "total": 10, "counts": {"exempt": 10},
"unclassified": 0, "computed_at": "2026-10-01T00:00:00"}
assert "written-shape check" not in coverage_line(base)
assert "written-shape check" not in coverage_line(
{**base, "write_time": {"checked": 0, "asked": 0, "left": 0, "days": 7}})
line = coverage_line({**base, "write_time": {"checked": 12, "asked": 4, "left": 1, "days": 7}})
assert line.endswith("; written-shape check (7d): 12 turns checked, 4 asked, 1 left unjudged")
+89
View File
@@ -0,0 +1,89 @@
"""The server half of the end-of-turn shape check (milestone 439): what the
agent is asked, what the hook's ledger may say, and the outcome record."""
import json
from unittest.mock import patch
import pytest
from tests.helpers import make_mock_session
def test_the_ledger_lines_parse_and_garbage_is_dropped():
from scribe.services.shape_check import parse_written
raw = ("a.py\tsym\thelper\n"
"a.py\tsym\thelper\n" # duplicate
"web/A.svelte\tfile\tA\n"
"x.py\tbogus\tthing\n" # unknown kind
"only-two\tsym\n" # malformed
"\tsym\tnameless-path\n")
assert parse_written(raw) == [("a.py", "sym", "helper"), ("web/A.svelte", "file", "A")]
assert len(parse_written("".join(f"p\tsym\tn{i}\n" for i in range(500)), cap=7)) == 7
def test_a_first_stop_blocks_or_passes_and_the_next_never_blocks():
from scribe.services.shape_check import outcome_for
one = [{"path": "a", "kind": "sym", "symbol": "b"}]
assert outcome_for("check", one) == "blocked"
assert outcome_for("check", []) == "passed"
assert outcome_for("after", one) == "left_after_block"
assert outcome_for("after", []) == "judged_after_block"
def test_the_reason_names_each_shape_its_evidence_and_the_one_call():
from scribe.services.shape_check import block_reason
unjudged = [
{"path": "web/A.svelte", "kind": "file", "symbol": "A", "status": "new", "evidence": {}},
{"path": "src/x.py", "kind": "sym", "symbol": "load", "status": "unclassified",
"evidence": {"looks_like": {"snippet_id": 12, "basis": "semantic", "score": 0.84},
"diverges_from": 9}},
]
reason = block_reason(unjudged, project_id=3, repo="git.example.com/a/w")
assert "2 definitions" in reason
assert "web/A.svelte · A (file) — new" in reason
assert "looks like #12 0.84" in reason and "#9 is canon in this directory" in reason
assert 'classify_shapes(project_id=3, repo="git.example.com/a/w"' in reason
for verdict in ("create_snippet", "`instance`", "`variant`", "`exempt`"):
assert verdict in reason, verdict
def test_a_long_turn_is_listed_with_a_tail_not_dropped():
from scribe.services.shape_check import _LISTED, block_reason
many = [{"path": "p.py", "kind": "sym", "symbol": f"f{i}", "status": "new", "evidence": {}}
for i in range(_LISTED + 3)]
reason = block_reason(many, project_id=1, repo="")
assert "… and 3 more" in reason
assert "repo=" not in reason
async def test_the_outcome_is_recorded_as_a_plugin_event():
from scribe.services.shape_check import record_shape_check
session = make_mock_session()
with patch("scribe.services.shape_check.async_session", return_value=session):
await record_shape_check(7, "blocked", written=5, unjudged=2, project_id=2)
row = session.add.call_args.args[0]
assert (row.category, row.action, row.user_id) == ("plugin", "shape_check", 7)
assert json.loads(row.details) == {"outcome": "blocked", "written": 5,
"unjudged": 2, "project_id": 2}
async def test_an_unknown_outcome_is_refused_before_anything_is_written():
from scribe.services.shape_check import record_shape_check
session = make_mock_session()
with patch("scribe.services.shape_check.async_session", return_value=session), \
pytest.raises(ValueError):
await record_shape_check(7, "skipped", written=1, unjudged=1)
session.add.assert_not_called()
def test_the_window_counts_turns_asks_and_asks_walked_past():
from scribe.services.shape_check import summarise
assert summarise([]) == {"checked": 0, "asked": 0, "left": 0}
got = summarise(["passed", "blocked", "judged_after_block", "blocked", "left_after_block"])
assert got == {"checked": 3, "asked": 2, "left": 1}
+171
View File
@@ -0,0 +1,171 @@
"""The end-of-turn shape check (milestone 439): the write hooks note what each
write defined, and the Stop hook asks the instance which of those nobody has
judged, blocking once in the server's words.
Runs the real shell against the shared HTTP sink. What it pins: the write
hooks' ledger (a Write of a new file adds a `file` line; a skipped path adds
nothing); silence with nothing written; a block only on a reason the instance
returned; the post-block stop recorded and let through; another hook's block
loop left alone; the ledger kept when the instance could not be reached.
"""
from __future__ import annotations
import json
import os
import shutil
import subprocess
from pathlib import Path
import pytest
from tests.helpers import http_sink
HOOKS = Path(__file__).resolve().parents[1] / "plugin" / "hooks"
CHECK = HOOKS / "scribe_shape_check.sh"
PRIOR_ART = HOOKS / "scribe_prior_art.sh"
REASON = "SERVER REASON: judge what you wrote"
def _env(tmp_path, url="http://127.0.0.1:9"):
for tool in ("curl", "bash", "git"):
if shutil.which(tool) is None:
pytest.skip(f"hook runtime tool {tool!r} not installed")
return {"PATH": os.environ["PATH"], "SCRIBE_URL": url, "SCRIBE_TOKEN": "t",
"TMPDIR": str(tmp_path), "HOME": str(tmp_path)}
def _ledger(tmp_path, session="s1"):
return tmp_path / "scribe-priorart" / f"{session}.written.ids"
def _write_ledger(tmp_path, lines, session="s1"):
path = _ledger(tmp_path, session)
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text("".join(f"{line}\n" for line in lines))
return path
def _stop(env, cwd, active=False, session="s1"):
out = subprocess.run(
["bash", str(CHECK)],
input=json.dumps({"session_id": session, "transcript_path": "/nonexistent",
"cwd": str(cwd), "hook_event_name": "Stop",
"stop_hook_active": active}),
capture_output=True, text=True, env=env, timeout=30,
)
assert out.returncode == 0, out.stderr
return out.stdout.strip()
def _repo(tmp_path):
repo = tmp_path / "repo"
repo.mkdir()
subprocess.run(["git", "init", "-q", str(repo)], check=True)
subprocess.run(["git", "-C", str(repo), "remote", "add", "origin",
"https://git.example.com/alice/widget.git"], check=True)
return repo
# ── the write hooks keep the ledger ───────────────────────────────────────
def _pre_write(env, repo, rel, content, tool="Write"):
field = "content" if tool == "Write" else "new_string"
out = subprocess.run(
["bash", str(PRIOR_ART)],
input=json.dumps({"session_id": "s1", "cwd": str(repo), "tool_name": tool,
"tool_input": {"file_path": str(repo / rel), field: content}}),
capture_output=True, text=True, env=env, timeout=30,
)
assert out.returncode == 0, out.stderr
def test_a_write_notes_what_it_defined_and_a_new_file_is_a_candidate(tmp_path):
with http_sink(b'{"note_ids":[]}') as (port, _seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
repo = _repo(tmp_path)
(repo / "web").mkdir()
_pre_write(env, repo, "web/StatusChip.svelte",
"<script>\nfunction tone(s) {\n return s;\n}\n</script>\n"
"<style>\n.chip {\n color: red;\n}\n</style>\n")
lines = _ledger(tmp_path).read_text().splitlines()
assert "web/StatusChip.svelte\tfile\tStatusChip" in lines
assert "web/StatusChip.svelte\tsym\ttone" in lines
assert "web/StatusChip.svelte\tcss\tchip" in lines
def test_an_edit_to_an_existing_file_adds_no_file_line(tmp_path):
with http_sink(b'{"note_ids":[]}') as (port, _seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
repo = _repo(tmp_path)
(repo / "lib.py").write_text("x = 1\n")
_pre_write(env, repo, "lib.py", "def helper():\n return 1\n", tool="Edit")
lines = _ledger(tmp_path).read_text().splitlines()
assert lines == ["lib.py\tsym\thelper"]
def test_a_skipped_path_notes_nothing(tmp_path):
with http_sink(b'{"note_ids":[]}') as (port, _seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
repo = _repo(tmp_path)
_pre_write(env, repo, "README.md", "# def helper():\n")
assert not _ledger(tmp_path).exists()
# ── the Stop hook asks ────────────────────────────────────────────────────
def test_nothing_written_is_silent_and_asks_nothing(tmp_path):
with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
assert _stop(env, _repo(tmp_path)) == ""
assert seen == []
def test_all_judged_passes_silently_and_empties_the_ledger(tmp_path):
with http_sink(b'{"status":"ok","outcome":"passed","unjudged":[]}') as (port, seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
ledger = _write_ledger(tmp_path, ["a.py\tsym\thelper", "a.py\tsym\thelper"])
assert _stop(env, _repo(tmp_path)) == ""
assert seen[0]["phase"] == ["check"]
assert seen[0]["written"] == ["a.py\tsym\thelper"], "duplicates collapse"
assert seen[0]["repo"] == ["https://git.example.com/alice/widget.git"]
assert not ledger.exists()
def test_unjudged_blocks_once_in_the_servers_words_then_records_the_answer(tmp_path):
reply = json.dumps({"status": "ok", "outcome": "blocked", "reason": REASON}).encode()
with http_sink(reply) as (port, seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
repo = _repo(tmp_path)
ledger = _write_ledger(tmp_path, ["web/A.svelte\tfile\tA"])
assert json.loads(_stop(env, repo)) == {"decision": "block", "reason": REASON}
assert ledger.exists(), "kept, so the stop after the block asks about the same set"
# The stop after the block: recorded, never blocked, ledger consumed.
assert _stop(env, repo, active=True) == ""
assert [q["phase"] for q in seen] == [["check"], ["after"]]
assert not ledger.exists()
assert _stop(env, repo, active=True) == ""
assert len(seen) == 2
def test_no_block_without_a_reason_from_the_instance(tmp_path):
with http_sink(b'{"status":"ok"}') as (port, _seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
_write_ledger(tmp_path, ["a.py\tsym\thelper"])
assert _stop(env, _repo(tmp_path)) == ""
def test_another_hooks_block_loop_is_left_alone(tmp_path):
with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
env = _env(tmp_path, f"http://127.0.0.1:{port}")
ledger = _write_ledger(tmp_path, ["a.py\tsym\thelper"])
assert _stop(env, _repo(tmp_path), active=True) == ""
assert seen == []
assert ledger.exists(), "this hook's own question is still owed"
def test_an_unreachable_instance_keeps_the_question_for_later(tmp_path):
env = _env(tmp_path) # port 9: nothing listens
ledger = _write_ledger(tmp_path, ["a.py\tsym\thelper"])
assert _stop(env, _repo(tmp_path)) == ""
assert ledger.exists()
+50
View File
@@ -350,3 +350,53 @@ def test_the_coherence_verdict_reads_rows_and_writes_nothing() -> None:
canon_coherence(rows)
assert [(r.path, r.symbol, r.status, r.snippet_id, r.signature)
for r in rows] == before
# ── a weak stamp is listed for judgment and counted by nobody (#4608) ─────
#
# The floor guarded new writes; the rows written before it went on electing
# canons. On two projects a YAML CI snippet "dominated" a Svelte directory on
# the strength of 0.68-0.72 stamps, and every write there was told it
# diverged from it. The readers below must agree with the review surface
# about what "weak" means — one predicate, so they cannot drift apart.
def test_one_predicate_says_what_weak_is() -> None:
from scribe.services.shape_ledger import is_weak_stamp
assert is_weak_stamp(_stamped(0.72))
assert not is_weak_stamp(_stamped(_RESEMBLE_MIN))
# A by-name stamp carries no score: it is not resemblance evidence at all.
assert not is_weak_stamp(_Row(reason="hook: pulled #1; payload references `f`"))
for by in ("agent", "audit", "mechanical"):
assert not is_weak_stamp(_stamped(0.5, classified_by=by)), by
def test_weak_stamps_do_not_elect_a_dominant_canon() -> None:
from scribe.services.shape_ledger import dominant_canon
weak = [_stamped(0.72, path=f"web/src/lib/{i}.ts", symbol=f"f{i}") for i in range(5)]
assert dominant_canon(weak) is None
# The same rows judged by an agent do — the kind of evidence is the point.
judged = [_stamped(0.72, path=f"web/src/lib/{i}.ts", symbol=f"f{i}", classified_by="agent")
for i in range(5)]
assert dominant_canon(judged) == (1, 5, 5)
# And weak rows do not dilute a real majority either: they are not counted
# as judged at all.
mixed = judged[:3] + [_stamped(0.7, snippet_id=2, symbol=f"g{i}") for i in range(4)]
assert dominant_canon(mixed) == (1, 3, 3)
def test_weak_stamps_do_not_say_what_form_a_canon_is() -> None:
from scribe.services.shape_ledger import FORM_UNKNOWN, canon_form
weak = [_stamped(0.7, symbol=f"T{i}", signature=f"class T{i}:") for i in range(5)]
assert canon_form(weak, 1) == FORM_UNKNOWN
judged = [_stamped(0.7, symbol=f"T{i}", signature=f"class T{i}:", classified_by="audit")
for i in range(5)]
assert canon_form(judged + weak, 1) == canon_form(judged, 1) != FORM_UNKNOWN
def test_counting_them_out_leaves_them_listed() -> None:
"""Not counted is not withdrawn: the review still lists every one."""
rows = [_stamped(0.7, symbol=f"f{i}") for i in range(3)]
assert stamps_to_review(rows)["weak_count"] == 3
+14 -14
View File
@@ -1291,9 +1291,9 @@ async def test_stamping_needs_named_shapes_and_a_recent_pull():
patch.object(pc, "semantic_search_notes", AsyncMock(return_value=[])), \
patch.object(pc, "record_retrieval", MagicMock()), \
patch.object(pc.shape_ledger_svc, "recent_pulls", pulls), \
patch.object(pc.shape_ledger_svc, "stamp_write_path_instances", stamp):
patch.object(pc.shape_ledger_svc, "suggest_write_path_instances", stamp):
out = await pc.build_write_path_hint(1, "src/x.py", code=REAL_CODE, project_id=4)
assert out["stamped"] == []
assert out["suggested"] == []
pulls.assert_not_awaited() # no shapes → no read
out = await pc.build_write_path_hint(
1, "src/x.py", code=REAL_CODE, project_id=4,
@@ -1301,7 +1301,7 @@ async def test_stamping_needs_named_shapes_and_a_recent_pull():
)
pulls.assert_awaited_once()
stamp.assert_not_awaited() # shapes, but no pull
assert out["stamped"] == []
assert out["suggested"] == []
@pytest.mark.asyncio
@@ -1330,7 +1330,7 @@ async def test_a_pulled_snippet_already_seen_is_evidence_not_menu():
patch.object(pc, "record_retrieval", MagicMock()), \
patch.object(pc, "owner_names_for", AsyncMock(return_value={})), \
patch.object(pc.shape_ledger_svc, "recent_pulls", AsyncMock(return_value={7: _ts()})), \
patch.object(pc.shape_ledger_svc, "stamp_write_path_instances", stamp):
patch.object(pc.shape_ledger_svc, "suggest_write_path_instances", stamp):
out = await pc.build_write_path_hint(
1, "src/x.py", code=REAL_CODE, project_id=4, exclude_ids=[7],
stamp_shapes=[("sym", "debounce")], repo_key="git.example.com/a/b",
@@ -1355,11 +1355,11 @@ async def test_a_pulled_snippet_already_seen_is_evidence_not_menu():
assert skw["resembles"] == {7: 0.91}
assert skw["shapes"] == [("sym", "debounce")]
assert skw["repo_key"] == "git.example.com/a/b"
assert out["stamped"][0]["snippet_id"] == 7
# And the session is told what landed, with the way to correct it.
assert out["suggested"][0]["snippet_id"] == 7
# And the session is shown it as evidence for its own end-of-turn judgment.
assert "Shape accounting" in out["context"]
assert "`debounce` → instance of #7" in out["context"]
assert "classify_shapes" in out["context"]
assert "`debounce` → looks like #7" in out["context"]
assert "When you judge this turn's shapes" in out["context"]
@pytest.mark.asyncio
@@ -1376,15 +1376,15 @@ async def test_a_stamp_renders_even_when_the_hint_is_otherwise_silent():
patch.object(pc, "record_retrieval", MagicMock()), \
patch.object(pc, "owner_names_for", AsyncMock(return_value={})), \
patch.object(pc.shape_ledger_svc, "recent_pulls", AsyncMock(return_value={5: _ts()})), \
patch.object(pc.shape_ledger_svc, "stamp_write_path_instances",
patch.object(pc.shape_ledger_svc, "suggest_write_path_instances",
AsyncMock(return_value=stamped)):
out = await pc.build_write_path_hint(
1, "web/b.css", code=".btn-primary { color: red; }" * 4, project_id=4,
stamp_shapes=[("css", "btn-primary")],
)
assert out["note_ids"] == []
assert out["stamped"] == stamped
assert "`.btn-primary` → instance of #5" in out["context"]
assert out["suggested"] == stamped
assert "`.btn-primary` → looks like #5" in out["context"]
@pytest.mark.asyncio
@@ -1397,13 +1397,13 @@ async def test_a_failing_stamp_does_not_sink_the_hint():
patch.object(pc, "record_retrieval", MagicMock()), \
patch.object(pc, "owner_names_for", AsyncMock(return_value={})), \
patch.object(pc.shape_ledger_svc, "recent_pulls", AsyncMock(return_value={12: _ts()})), \
patch.object(pc.shape_ledger_svc, "stamp_write_path_instances",
patch.object(pc.shape_ledger_svc, "suggest_write_path_instances",
AsyncMock(side_effect=RuntimeError("ledger down"))):
out = await pc.build_write_path_hint(
1, "src/x.py", code=REAL_CODE, project_id=4, stamp_shapes=[("sym", "f")],
)
assert out["sync_note_ids"] == [12]
assert out["stamped"] == []
assert out["suggested"] == []
def test_route_stamps_only_for_a_caller_allowed_to_write():
@@ -1760,7 +1760,7 @@ async def test_a_record_this_arm_withheld_itself_is_not_a_near_miss():
patch.object(pc, "owner_names_for", AsyncMock(return_value={})), \
patch.object(pc.shape_ledger_svc, "recent_pulls",
AsyncMock(return_value={7: _ts()})), \
patch.object(pc.shape_ledger_svc, "stamp_write_path_instances",
patch.object(pc.shape_ledger_svc, "suggest_write_path_instances",
AsyncMock(return_value=[])), \
patch.object(pc, "semantic_search_notes",
_search_reporting(0.9, fake_note(