From 3270fe90c1728fddeea250f735620bd5c83044b7 Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Thu, 24 Sep 2026 12:30:58 -0400 Subject: [PATCH 1/6] refactor(settings): one bounded_float for every numeric bar read from settings The shape ledger's divergence readout flagged _gate_setting (#4385). Reading it turned up a family with no canon: floor_for, get_duplicate_threshold, get_plan_match_threshold and _gate_setting each parsed a stored string, fell back to the default (never 0) and clamped into [lo, 1]. - services/settings.bounded_float(raw, default, lo=0, hi=1) is the pure parse, fallback and clamp. Each caller keeps its own get_setting read, so tests patching get_setting per module still take effect, and _gate_setting still fails open on an unreadable setting. - SettingsView.saveKbInject: eleven inline Math.min/Math.max clamps and the local gateAt become one asBar(v, d, lo = 0), mirroring the server. No behaviour change. Co-Authored-By: Claude Opus 5.5 --- frontend/src/views/SettingsView.vue | 37 ++++++++++++----------- src/scribe/services/dedup.py | 32 ++++++++------------ src/scribe/services/retrieval_surfaces.py | 9 ++---- src/scribe/services/settings.py | 16 ++++++++++ 4 files changed, 51 insertions(+), 43 deletions(-) diff --git a/frontend/src/views/SettingsView.vue b/frontend/src/views/SettingsView.vue index 102af45..25c590e 100644 --- a/frontend/src/views/SettingsView.vue +++ b/frontend/src/views/SettingsView.vue @@ -275,41 +275,44 @@ function actorLabel(actor: string): string { } async function saveKbInject() { - const t = Math.min(1, Math.max(0, Number(kbInjectThreshold.value) || 0)); + // Every similarity bar, clamped the way the server's `bounded_float` clamps + // it: an unparseable value falls back to its default, never to 0, and `lo` + // is the floor a bar that BLOCKS must not go under (0.80 block, 0.70 overlap). + const asBar = (v: string, d: number, lo = 0) => Math.min(1, Math.max(lo, Number(v) || d)); + const t = asBar(kbInjectThreshold.value, 0); const k = Math.min(10, Math.max(1, Math.floor(Number(kbInjectTopK.value) || 1))); // `|| default` not `|| 0`: an unparseable value here should fall back to the // per-kind default, not to 0 — a 0 floor would report every record as a // duplicate of every other one. - const dupSnip = Math.min(1, Math.max(0, Number(kbDupThresholdSnippet.value) || 0.82)); - const dupNote = Math.min(1, Math.max(0, Number(kbDupThresholdNote.value) || 0.93)); - const dupTask = Math.min(1, Math.max(0, Number(kbDupThresholdTask.value) || 0.93)); + const dupSnip = asBar(kbDupThresholdSnippet.value, 0.82); + const dupNote = asBar(kbDupThresholdNote.value, 0.93); + const dupTask = asBar(kbDupThresholdTask.value, 0.93); // The gate BLOCKS a write, so its bars have a floor the server enforces too: // 0.80 for a block, 0.70 for the overlap list. - const gateAt = (v: string, d: number, lo: number) => Math.min(1, Math.max(lo, Number(v) || d)); - const gate = gateAt(kbGateThreshold.value, 0.9, 0.8); - const gateSnip = gateAt(kbGateThresholdSnippet.value, 0.96, 0.8); - const gateLesson = gateAt(kbGateThresholdLesson.value, 0.96, 0.8); - const gateCopy = gateAt(kbGateThresholdNoteCopy.value, 0.98, 0.8); - const gateOverlap = gateAt(kbGateNoteOverlapFloor.value, 0.87, 0.7); + const gate = asBar(kbGateThreshold.value, 0.9, 0.8); + const gateSnip = asBar(kbGateThresholdSnippet.value, 0.96, 0.8); + const gateLesson = asBar(kbGateThresholdLesson.value, 0.96, 0.8); + const gateCopy = asBar(kbGateThresholdNoteCopy.value, 0.98, 0.8); + const gateOverlap = asBar(kbGateNoteOverlapFloor.value, 0.87, 0.7); // Same `|| default` guard: a floor of 0 would hand back an existing plan // for every new one, and no plan could be started without force. - const planT = Math.min(1, Math.max(0, Number(kbPlanMatchThreshold.value) || 0.8)); + const planT = asBar(kbPlanMatchThreshold.value, 0.8); // Same `|| default` reasoning: falling back to 0 would surface every // snippet in the corpus on every edit, which is the failure this knob fixes. - const wpT = Math.min(1, Math.max(0, Number(kbWritePathThreshold.value) || 0.68)); + const wpT = asBar(kbWritePathThreshold.value, 0.68); // Same `|| default` reasoning again, and it bites harder here: a rule hint // fires on every write, so a fallback of 0 would attach a standing rule to // every edit in the session. - const rhT = Math.min(1, Math.max(0, Number(kbRuleHintThreshold.value) || 0.72)); + const rhT = asBar(kbRuleHintThreshold.value, 0.72); // Same `|| default` guard, and the same reason: this arm fires before every // Bash call, so a fallback of 0 would put a rule in front of every command. - const trT = Math.min(1, Math.max(0, Number(kbToolRuleThreshold.value) || 0.68)); - const prT = Math.min(1, Math.max(0, Number(kbPromptRuleThreshold.value) || 0.72)); + const trT = asBar(kbToolRuleThreshold.value, 0.68); + const prT = asBar(kbPromptRuleThreshold.value, 0.72); // The checkpoint bar, and the `|| default` guard matters most here of all: // this is the only number that can STOP a call, so a fallback of 0 would // hold the first command of every session behind whatever ranked first. - const cpT = Math.min(1, Math.max(0, Number(kbCheckpointThreshold.value) || 0.8)); - const rpT = Math.min(1, Math.max(0, Number(kbReportPrefThreshold.value) || 0.72)); + const cpT = asBar(kbCheckpointThreshold.value, 0.8); + const rpT = asBar(kbReportPrefThreshold.value, 0.72); // The budgets, clamped the way the server clamps them: a whole number in // [1, 10]. Never 0 — an arm turned off is turned off by its switch, and a // budget of zero would run the search, log the retrieval and render nothing, diff --git a/src/scribe/services/dedup.py b/src/scribe/services/dedup.py index d15e18b..208f086 100644 --- a/src/scribe/services/dedup.py +++ b/src/scribe/services/dedup.py @@ -287,16 +287,16 @@ def _gate_key(note_type: str) -> str: async def _gate_setting(user_id: int, key: str, lo: float) -> float: - from scribe.services.settings import get_setting + from scribe.services.settings import bounded_float, get_setting default = GATE_DEFAULT_THRESHOLDS[key] try: - value = float(await get_setting(user_id, GATE_THRESHOLD_KEYS[key], str(default))) + raw = await get_setting(user_id, GATE_THRESHOLD_KEYS[key], str(default)) except Exception: # Fail-open like the rest of the gate: an unreadable setting falls back # to the measured default rather than blocking or waving through. - value = default - return min(1.0, max(lo, value)) + raw = None + return bounded_float(raw, default, lo) async def gate_bars(user_id: int, note_type: str) -> tuple[float, float]: @@ -557,16 +557,11 @@ _MAX_DUPLICATE_PAIRS = 200 async def get_duplicate_threshold(user_id: int, kind: str = "snippet") -> float: """The user's near-duplicate similarity floor for `kind`, clamped to [0, 1].""" - from scribe.services.settings import get_setting + from scribe.services.settings import bounded_float, get_setting default = DUPLICATE_DEFAULT_THRESHOLDS[kind] - try: - value = float(await get_setting( - user_id, DUPLICATE_THRESHOLD_KEYS[kind], str(default) - )) - except (TypeError, ValueError): - value = default - return min(1.0, max(0.0, value)) + raw = await get_setting(user_id, DUPLICATE_THRESHOLD_KEYS[kind], str(default)) + return bounded_float(raw, default) def group_pairs(pairs: list[tuple[int, int, float]]) -> list[list[int]]: @@ -1120,15 +1115,12 @@ PLAN_MATCH_DEFAULT_THRESHOLD = 0.80 async def get_plan_match_threshold(user_id: int) -> float: """The user's plan-gate similarity floor, clamped to [0, 1].""" - from scribe.services.settings import get_setting + from scribe.services.settings import bounded_float, get_setting - try: - value = float(await get_setting( - user_id, PLAN_MATCH_THRESHOLD_KEY, str(PLAN_MATCH_DEFAULT_THRESHOLD) - )) - except (TypeError, ValueError): - value = PLAN_MATCH_DEFAULT_THRESHOLD - return min(1.0, max(0.0, value)) + raw = await get_setting( + user_id, PLAN_MATCH_THRESHOLD_KEY, str(PLAN_MATCH_DEFAULT_THRESHOLD) + ) + return bounded_float(raw, PLAN_MATCH_DEFAULT_THRESHOLD) def plan_candidate_text( diff --git a/src/scribe/services/retrieval_surfaces.py b/src/scribe/services/retrieval_surfaces.py index cecdf7e..fb104d5 100644 --- a/src/scribe/services/retrieval_surfaces.py +++ b/src/scribe/services/retrieval_surfaces.py @@ -66,7 +66,7 @@ from __future__ import annotations from dataclasses import dataclass -from scribe.services.settings import get_setting +from scribe.services.settings import bounded_float, get_setting # A budget nobody should be able to set past. Not a tuning value — a guard on # the worst case, so a mistyped setting cannot turn a menu into a wall of text. @@ -270,11 +270,8 @@ def dial_for_key(key: str) -> tuple[str, str] | None: async def floor_for(user_id: int, name: str) -> float: """This install's current floor for a surface, clamped to [0, 1].""" s = get_surface(name) - try: - value = float(await get_setting(user_id, s.floor_key, str(s.floor_default))) - except (TypeError, ValueError): - value = s.floor_default - return min(1.0, max(0.0, value)) + raw = await get_setting(user_id, s.floor_key, str(s.floor_default)) + return bounded_float(raw, s.floor_default) async def budget_for(user_id: int, name: str) -> int: diff --git a/src/scribe/services/settings.py b/src/scribe/services/settings.py index fb044aa..ae13359 100644 --- a/src/scribe/services/settings.py +++ b/src/scribe/services/settings.py @@ -16,6 +16,22 @@ logger = logging.getLogger(__name__) SECRET_MASK = "********" +def bounded_float(raw: str | None, default: float, lo: float = 0.0, hi: float = 1.0) -> float: + """A numeric setting as stored text, parsed and clamped to [lo, hi]. + + Every similarity bar and retrieval floor is kept as a string and read the + same way: an unparseable value falls back to the DEFAULT, never to 0 (a + floor of 0 admits everything), and a value out of range is pulled back + into it. Pure on purpose: each caller keeps its own `get_setting` read, so + what can fail there — and whether that fails open — stays the caller's. + """ + try: + value = float(raw) + except (TypeError, ValueError): + value = default + return min(hi, max(lo, value)) + + async def get_admin_setting(key: str, default: str = "") -> str: """Read an instance-global setting (one stored on an admin account). -- 2.54.0 From eb5cc6d3a76ae0c055659e6d15bc55f47b42921d Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Wed, 30 Sep 2026 23:27:07 -0400 Subject: [PATCH 2/6] fix(shapes): weak stamps elect no canon; the arrival line names the review; Svelte scopes by default (#4608) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Librarian and Stash had 132 and 488 write-path stamps written at 0.68-0.72, before the 0.80 floor (#4204). Nothing re-judged them, and dominant_canon counted them, so a YAML CI snippet (#3410) "dominated" Librarian's web directory and every write there was told it diverged from it — 1408 flags on one project, 902 on the other, all skimmed past. - is_weak_stamp: one predicate for "hook-stamped below today's floor". dominant_canon and canon_form skip such rows; stamps_to_review lists them by the same predicate. No stored row changes — they stay for judgment. - flag_divergence withdraws a standing flag whose canon no longer dominates its directory. A flag is a mechanical prompt, not a judgment. - The coverage line ends "N weak stamps · M incoherent canons to judge — stamps_to_review" when either is nonzero, so a project's own session sees the queue on arrival. The shape-accounting skill says what to do with it. - scoped_definitions handles .svelte: every \n" + ) + defs = extract_definitions(svelte) + names = {(d.kind, d.name) for d in defs} + assert {("sym", "stageOf"), ("sym", "toggle"), ("css", "card")} <= names + scoped = scoped_definitions("web/src/lib/A.svelte", svelte, defs) + assert ("sym", "toggle") in scoped and ("css", "card") in scoped + assert ("sym", "stageOf") not in scoped, "a module script's export is importable" + assert not any(n == "toast" for _, n in scoped), ":global(...) opts a selector out" + # The legacy spelling of the module block, and a whole global style block. + legacy = ( + "\n" + "\n" + ) + assert scoped_definitions("web/src/lib/B.svelte", legacy, extract_definitions(legacy)) == set() + + +def test_the_line_names_the_review_queue_and_the_tool_that_holds_it(): + """#4608: `stamps_to_review` existed and two projects never called it, + because nothing on arrival said there was anything to review.""" + from scribe.services.coverage import coverage_line + base = {"accounted": 10, "total": 10, "counts": {"exempt": 10}, + "unclassified": 0, "computed_at": "2026-09-30T00:00:00"} + assert "stamps_to_review" not in coverage_line(base) + line = coverage_line({**base, "weak_stamps": 132, "incoherent_canons": 3}) + assert line.endswith("; 132 weak stamps · 3 incoherent canons to judge — stamps_to_review") + assert coverage_line({**base, "weak_stamps": 1}).endswith("; 1 weak stamp to judge — stamps_to_review") + + @pytest.mark.integration async def test_binding_ref_is_the_branch_the_ledger_follows(seeded): """#2873: a binding that names a ref is read at that ref (not the forge's diff --git a/tests/test_stamps_to_review.py b/tests/test_stamps_to_review.py index ac839b6..b4d7066 100644 --- a/tests/test_stamps_to_review.py +++ b/tests/test_stamps_to_review.py @@ -350,3 +350,53 @@ def test_the_coherence_verdict_reads_rows_and_writes_nothing() -> None: canon_coherence(rows) assert [(r.path, r.symbol, r.status, r.snippet_id, r.signature) for r in rows] == before + + +# ── a weak stamp is listed for judgment and counted by nobody (#4608) ───── +# +# The floor guarded new writes; the rows written before it went on electing +# canons. On two projects a YAML CI snippet "dominated" a Svelte directory on +# the strength of 0.68-0.72 stamps, and every write there was told it +# diverged from it. The readers below must agree with the review surface +# about what "weak" means — one predicate, so they cannot drift apart. + +def test_one_predicate_says_what_weak_is() -> None: + from scribe.services.shape_ledger import is_weak_stamp + + assert is_weak_stamp(_stamped(0.72)) + assert not is_weak_stamp(_stamped(_RESEMBLE_MIN)) + # A by-name stamp carries no score: it is not resemblance evidence at all. + assert not is_weak_stamp(_Row(reason="hook: pulled #1; payload references `f`")) + for by in ("agent", "audit", "mechanical"): + assert not is_weak_stamp(_stamped(0.5, classified_by=by)), by + + +def test_weak_stamps_do_not_elect_a_dominant_canon() -> None: + from scribe.services.shape_ledger import dominant_canon + + weak = [_stamped(0.72, path=f"web/src/lib/{i}.ts", symbol=f"f{i}") for i in range(5)] + assert dominant_canon(weak) is None + # The same rows judged by an agent do — the kind of evidence is the point. + judged = [_stamped(0.72, path=f"web/src/lib/{i}.ts", symbol=f"f{i}", classified_by="agent") + for i in range(5)] + assert dominant_canon(judged) == (1, 5, 5) + # And weak rows do not dilute a real majority either: they are not counted + # as judged at all. + mixed = judged[:3] + [_stamped(0.7, snippet_id=2, symbol=f"g{i}") for i in range(4)] + assert dominant_canon(mixed) == (1, 3, 3) + + +def test_weak_stamps_do_not_say_what_form_a_canon_is() -> None: + from scribe.services.shape_ledger import FORM_UNKNOWN, canon_form + + weak = [_stamped(0.7, symbol=f"T{i}", signature=f"class T{i}:") for i in range(5)] + assert canon_form(weak, 1) == FORM_UNKNOWN + judged = [_stamped(0.7, symbol=f"T{i}", signature=f"class T{i}:", classified_by="audit") + for i in range(5)] + assert canon_form(judged + weak, 1) == canon_form(judged, 1) != FORM_UNKNOWN + + +def test_counting_them_out_leaves_them_listed() -> None: + """Not counted is not withdrawn: the review still lists every one.""" + rows = [_stamped(0.7, symbol=f"f{i}") for i in range(3)] + assert stamps_to_review(rows)["weak_count"] == 3 -- 2.54.0 From f1fbdf746ad358cb1618c0c62f6d2e54b1a47b5d Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Thu, 1 Oct 2026 08:39:36 -0400 Subject: [PATCH 3/6] feat(shapes): the agent judges what it wrote, at the end of the turn (milestone 439 steps 1-3) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Recording used to be decided by machinery — the only "record it" prompt fired when a same-named copy already existed (#2664), so a first instance of a reusable piece was never asked about, and judgment arrived only through audits. Now the question is asked where the knowledge is: the end of the turn that wrote the code, of the agent that wrote it. - Write hooks keep `.written.ids` (path, kind, name) for every definition a write names; a new file adds a `file` line for its stem — a candidate in any language without a framework rule (scribe_written_append). - Stop hook scribe_shape_check.sh sends the ledger to GET /api/plugin/shape-check and blocks once, in the server's words, when anything is unjudged. Same discipline as the report check: never twice, never without a recorded check, another hook's loop left alone; the ledger is kept when the instance cannot be reached. - shape_ledger.unjudged_shapes: no row, unclassified, scoped and hook stamps are unjudged; an agent/audit/import verdict is not. A snippet recorded at the shape answers for it until the refresh stamps it canonical. - services/shape_check owns the reason text and records every outcome in app_logs (passed / blocked / judged_after_block / left_after_block). - classify_shapes(repo=…) judges a shape the ledger has not synced yet via a provisional row under a bound repo; the sync confirms it, or vanishes and revives it with the verdict intact. An unbound repo is refused. Plugin version minted. Co-Authored-By: Claude Opus 5.5 --- plugin/.claude-plugin/plugin.json | 2 +- plugin/PACKAGING.md | 1 + plugin/README.md | 9 ++ plugin/hooks/hooks.json | 4 + plugin/hooks/scribe_after_write.sh | 5 + plugin/hooks/scribe_defs.sh | 25 ++++ plugin/hooks/scribe_prior_art.sh | 7 + plugin/hooks/scribe_shape_check.sh | 105 ++++++++++++++ scripts/check_plugin.py | 7 + src/scribe/mcp/tools/shapes.py | 22 ++- src/scribe/routes/plugin.py | 66 +++++++++ src/scribe/services/repo_bindings.py | 18 +++ src/scribe/services/shape_check.py | 147 +++++++++++++++++++ src/scribe/services/shape_ledger.py | 115 ++++++++++++++- tests/test_integration_shape_classify.py | 78 +++++++++++ tests/test_services_shape_check.py | 81 +++++++++++ tests/test_shape_check_hook.py | 171 +++++++++++++++++++++++ 17 files changed, 854 insertions(+), 9 deletions(-) create mode 100644 plugin/hooks/scribe_shape_check.sh create mode 100644 src/scribe/services/shape_check.py create mode 100644 tests/test_services_shape_check.py create mode 100644 tests/test_shape_check_hook.py diff --git a/plugin/.claude-plugin/plugin.json b/plugin/.claude-plugin/plugin.json index a33665e..f969db7 100644 --- a/plugin/.claude-plugin/plugin.json +++ b/plugin/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "scribe", "description": "Scribe for Claude Code: connects the scribe MCP server, adds the hooks that deliver live project state and relevant records at the right moment, ships the shared client-neutral Scribe skills (using-scribe, writing-plans, reporting-back, systematic-debugging, verification, brainstorming, reusing-code, shape-accounting), and syncs your saved Scribe Processes as skills (/scribe:sync).", - "version": "2026.10.01.0326", + "version": "2026.10.01.1239", "author": { "name": "Bryan Van Deusen" }, diff --git a/plugin/PACKAGING.md b/plugin/PACKAGING.md index 780825e..72376a8 100644 --- a/plugin/PACKAGING.md +++ b/plugin/PACKAGING.md @@ -40,6 +40,7 @@ another one means adding files, not moving or rewriting any. | `hooks/scribe_after_write.sh` | PostToolUse on shell commands: the same check for code written through the shell. | | `hooks/scribe_tool_rules.sh` | PreToolUse on shell commands: `GET /api/plugin/tool-rules`. | | `hooks/scribe_report_check.sh` | Stop: when the turn closed a task, checks the reply for the completion sections and reports to `GET /api/plugin/report-check`; blocks once, with the reason the server returns. | +| `hooks/scribe_shape_check.sh` | Stop: sends the definitions the turn wrote (the write hooks' `.written.ids` ledger) to `GET /api/plugin/shape-check`; blocks once, with the reason the server returns, so the agent judges what it built. | | `hooks/scribe_sync_processes.sh` + `commands/sync.md` | `GET /api/plugin/processes` → `~/.claude/skills/scribe-proc-*` stubs; `/scribe:sync` on demand. | | `hooks/scribe_defs.sh` | Shared shell helpers: config, dedup ledgers, outage line. | | `hooks/scribe_static_context.md` | The adapter static text. | diff --git a/plugin/README.md b/plugin/README.md index cfd79e9..859e151 100644 --- a/plugin/README.md +++ b/plugin/README.md @@ -91,6 +91,15 @@ On install you'll be asked for: never blocks twice, and never blocks when the instance did not record the check (unconfigured or unreachable). Outcomes land in the admin logs under category `plugin`, action `report_check`. +- `hooks/hooks.json` → a second Stop hook (`hooks/scribe_shape_check.sh`): the + write hooks note every definition a write names (and every new file) in a + session ledger; at the end of the turn this sends them to + `GET /api/plugin/shape-check`, which answers with the ones nobody has judged. + If any are, it blocks once with the server's reason, asking the agent that + wrote them to say what each is — reusable (record a snippet), an instance or + variant of one, or a one-off — in one `classify_shapes` call. Same discipline + as the report check: never twice, never without a recorded check. Outcomes + land under category `plugin`, action `shape_check`. - `skills/` → the universal process-skills, surfaced by description match. - `hooks/scribe_sync_processes.sh` (a 2nd SessionStart hook) + the `/scribe:sync` command → generate `~/.claude/skills/scribe-proc-*` stubs from your Scribe diff --git a/plugin/hooks/hooks.json b/plugin/hooks/hooks.json index 5c6b7ef..1bdf248 100644 --- a/plugin/hooks/hooks.json +++ b/plugin/hooks/hooks.json @@ -98,6 +98,10 @@ { "type": "command", "command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_report_check.sh\"" + }, + { + "type": "command", + "command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_shape_check.sh\"" } ] } diff --git a/plugin/hooks/scribe_after_write.sh b/plugin/hooks/scribe_after_write.sh index 689bead..d9db77f 100644 --- a/plugin/hooks/scribe_after_write.sh +++ b/plugin/hooks/scribe_after_write.sh @@ -129,14 +129,19 @@ while IFS= read -r rel_path; do # The code just written: the ADDED lines of the uncommitted diff for a # tracked file (sed, not cut: this strips one marker char per line, it is # not a payload cap), the whole file when untracked. + fresh="" if git -C "$repo_root" ls-files --error-unmatch -- "$rel_path" >/dev/null 2>&1; then code=$(git -C "$repo_root" diff -U0 -- "$rel_path" 2>/dev/null | grep '^+' | grep -v '^+++' | sed 's/^+//') || code="" else code=$(cat "$file_path" 2>/dev/null) || code="" + fresh="new" fi [ -n "$code" ] || continue # `|| true` inside: an early-exiting `head` must not void its own output (#4042). names=$(printf '%s' "$code" | scribe_defs | sort -u | head -12 || true) + # What this write defined, for the end-of-turn question (milestone 439) — + # recorded before the arms below, which skip a write that defines nothing. + printf '%s\n' "$names" | scribe_written_append "$safe_sid" "$rel_path" "$fresh" # Nothing DEFINED in what was written (prose, data, a call-site edit) → # nothing to say; the arms are about shapes. [ -n "$names" ] || continue diff --git a/plugin/hooks/scribe_defs.sh b/plugin/hooks/scribe_defs.sh index 675cfab..a88b97a 100644 --- a/plugin/hooks/scribe_defs.sh +++ b/plugin/hooks/scribe_defs.sh @@ -1158,6 +1158,31 @@ scribe_held_query() { # the tests read THIS string rather than a copy of it. SCRIBE_LEDGER_DIRS="scribe-priorart scribe-autoinject" +# THE WRITTEN-SHAPES LEDGER (milestone 439). Every definition a write names, +# appended as `pathkindname`, for the Stop hook to ask about at the +# end of the turn (scribe_shape_check.sh). Its own file — one file, one +# question (lesson #4226) — under the prior-art directory, so the session +# clear above already sweeps it. Read `kindname` lines on stdin (the +# scribe_defs format). A NEW file adds one `file` line named by its stem: a +# component, module or package file is a candidate in its own right, in any +# language, without a framework rule saying so. +# $1 session id (already made filename-safe) $2 repo-relative path +# $3 "new" when the file did not exist before this write +scribe_written_append() { + local sid="$1" rel="$2" fresh="${3:-}" dir stem + if [ -z "$sid" ] || [ -z "$rel" ]; then cat >/dev/null; return 0; fi + dir="${TMPDIR:-/tmp}/scribe-priorart" + mkdir -p "$dir" 2>/dev/null || true + { + if [ "$fresh" = "new" ]; then + stem=${rel##*/}; stem=${stem%.*} + [ -n "$stem" ] && printf '%s\tfile\t%s\n' "$rel" "$stem" + fi + awk -F'\t' -v p="$rel" 'NF >= 2 && ($1 == "sym" || $1 == "css") && $2 != "" { printf "%s\t%s\t%s\n", p, $1, $2 }' + } >> "$dir/${sid}.written.ids" 2>/dev/null || true + return 0 +} + scribe_clear_session_ledgers() { # SPARES `*.keep.ids`, which are evidence rather than exclusions — see # `scribe_rules_append`. Everything else still goes: the convention is diff --git a/plugin/hooks/scribe_prior_art.sh b/plugin/hooks/scribe_prior_art.sh index 3813b7a..72c8111 100755 --- a/plugin/hooks/scribe_prior_art.sh +++ b/plugin/hooks/scribe_prior_art.sh @@ -201,6 +201,13 @@ derive_exclude_q="" rule_exclude_q="" if [ -n "$session_id" ]; then safe_sid=$(printf '%s' "$session_id" | tr -c 'A-Za-z0-9._-' '_') + # What this write defined, for the end-of-turn question (milestone 439). A + # Write onto a path that does not exist yet creates a file — a candidate in + # its own right. Written before any server call: the ask at the end of the + # turn must not depend on this write's hint having been answered. + fresh="" + [ -e "$file_path" ] || fresh="new" + printf '%s\n' "$shapes" | scribe_written_append "$safe_sid" "$rel_path" "$fresh" idfile="$state_dir/${safe_sid}.ids" syncfile="$state_dir/${safe_sid}.sync.ids" derivefile="$state_dir/${safe_sid}.derive.ids" diff --git a/plugin/hooks/scribe_shape_check.sh b/plugin/hooks/scribe_shape_check.sh new file mode 100644 index 0000000..a63097a --- /dev/null +++ b/plugin/hooks/scribe_shape_check.sh @@ -0,0 +1,105 @@ +#!/usr/bin/env bash +# Scribe plugin — Stop hook: the turn's new shapes are judged by the agent +# that wrote them (milestone 439). +# +# The write hooks (scribe_prior_art.sh, scribe_after_write.sh) append every +# definition a write names to a session ledger, `.written.ids` +# (`pathkindname`, see scribe_written_append). At the end of the +# turn this hook sends that ledger to the instance, which answers with what +# nobody has judged yet. If anything is, the hook blocks ONCE, in the server's +# words: the agent that built the code is the one participant who knows what +# it is, and the end of the turn is the last moment that is still true. +# +# THE SAME DISCIPLINE AS scribe_report_check.sh, deliberately: +# - it blocks only on a `reason` the instance returned, which it returns +# only for a block it RECORDED — an unconfigured or unreachable instance +# never stops a session, and every intervention is one the numbers see; +# - never twice for one turn: the stop that follows its own block +# (`stop_hook_active` with this hook's marker present) is recorded as +# judged_after_block / left_after_block and let through; +# - another plugin's block loop is left alone (active, no marker). +# +# THE LEDGER IS CONSUMED, NOT TURN-STAMPED. Everything written since the last +# stop this hook answered is "this turn". It is emptied once the instance has +# answered — after a pass, or after the post-block stop — and KEPT when the +# instance could not be reached, so the question is asked at the next stop +# that can be answered rather than silently dropped. +# +# Config (same as the other hooks): +# CLAUDE_PLUGIN_OPTION_API_ENDPOINT base URL, no trailing slash +# CLAUDE_PLUGIN_OPTION_API_TOKEN fmcp_ API key (sensitive) +# SCRIBE_URL / SCRIBE_TOKEN override for the settings.json dogfooding path. +set -uo pipefail + +command -v curl >/dev/null 2>&1 || exit 0 + +# shellcheck source=plugin/hooks/scribe_defs.sh +. "$(dirname "${BASH_SOURCE[0]}")/scribe_defs.sh" + +# Stop delivers { session_id, transcript_path, cwd, hook_event_name, stop_hook_active }. +event=$(cat 2>/dev/null || true) +event_flat=$(printf '%s' "$event" | scribe_json_flat) +session_id=$(scribe_json_pick "$event_flat" '.session_id') +active=$(scribe_json_pick "$event_flat" '.stop_hook_active') +event_cwd=$(scribe_json_pick "$event_flat" '.cwd') +[ -n "$session_id" ] || exit 0 + +safe_sid=$(printf '%s' "$session_id" | tr -c 'A-Za-z0-9._-' '_') +ledger="${TMPDIR:-/tmp}/scribe-priorart/${safe_sid}.written.ids" +state_dir="${TMPDIR:-/tmp}/scribe-shapecheck" +mkdir -p "$state_dir" 2>/dev/null || true +marker="$state_dir/${safe_sid}.blocked" + +if [ "$active" = "true" ]; then + # A Stop hook already blocked this stop. Another plugin's → nothing to add. + [ -f "$marker" ] || exit 0 + phase=after +else + rm -f "$marker" 2>/dev/null || true + phase=check +fi + +if [ ! -s "$ledger" ]; then + rm -f "$marker" 2>/dev/null || true + exit 0 +fi +scribe_config || exit 0 + +# Unique lines, oldest first, capped: the request stays a GET (a read-scoped +# key runs the plugin), and a turn that wrote more than this generated code. +written=$(awk '!seen[$0]++' "$ledger" 2>/dev/null | head -n 80) +written_enc=$(printf '%s' "$written" | scribe_urlenc) || exit 0 +dir="${event_cwd:-${CLAUDE_PROJECT_DIR:-$PWD}}" +q="phase=${phase}&written=${written_enc}" +scope=$(scribe_scope_query "$dir") +[ -n "$scope" ] && q="${q}&${scope}" +# The repo rides along even when a marker names the project: the instance +# needs it to tell the agent which repo a just-written shape belongs to. +case "$scope" in + repo=*) ;; + *) + remote=$(git -C "$dir" remote get-url origin 2>/dev/null || true) + if [ -n "$remote" ]; then + remote_enc=$(printf '%s' "$remote" | scribe_urlenc) || remote_enc="" + [ -n "$remote_enc" ] && q="${q}&repo=${remote_enc}" + fi + ;; +esac + +answer=$(curl -fsS --max-time 4 \ + -H "Authorization: Bearer ${token}" \ + "${url%/}/api/plugin/shape-check?${q}" 2>/dev/null) || exit 0 # unreachable: keep the ledger + +if [ "$phase" = "after" ]; then + rm -f "$marker" "$ledger" 2>/dev/null || true + exit 0 +fi + +reason=$(scribe_json_pick "$(printf '%s' "$answer" | scribe_json_flat)" '.reason') +if [ -z "$reason" ]; then + rm -f "$ledger" 2>/dev/null || true + exit 0 +fi +: > "$marker" 2>/dev/null || true +printf '{"decision":"block","reason":"%s"}\n' "$(printf '%s' "$reason" | scribe_json_escape)" +exit 0 diff --git a/scripts/check_plugin.py b/scripts/check_plugin.py index 33bac05..57b6677 100755 --- a/scripts/check_plugin.py +++ b/scripts/check_plugin.py @@ -358,6 +358,13 @@ SMOKE_EVENTS: dict[str, str] = { {"session_id": "smoke", "transcript_path": "/nonexistent/smoke.jsonl", "cwd": ".", "hook_event_name": "Stop", "stop_hook_active": False} ), + # The end-of-turn shape check (milestone 439). With no ledger for the + # session there is nothing to ask about, and with no instance there is + # nobody to ask: silent, and never a block, configured or not. + "scribe_shape_check.sh": json.dumps( + {"session_id": "smoke", "transcript_path": "/nonexistent/smoke.jsonl", + "cwd": ".", "hook_event_name": "Stop", "stop_hook_active": False} + ), # The PreCompact preserver (#3680). Its whole contract is the inverse of # every other hook's: it must exit 0 AND print, because stdout is what # becomes the summarizer's custom instructions. A silent success here is diff --git a/src/scribe/mcp/tools/shapes.py b/src/scribe/mcp/tools/shapes.py index 7765b8a..cab1158 100644 --- a/src/scribe/mcp/tools/shapes.py +++ b/src/scribe/mcp/tools/shapes.py @@ -4,7 +4,8 @@ The accounting model (note 2786): the snippet library records CANON (small); the ledger accounts for EVERY extracted shape (total). These tools are how agents move shapes out of `unclassified` — the todo state — and how they read what still needs judgment. The ledger rows themselves are fed by the coverage -refresh; these tools only ever judge what the sync has seen. +refresh; a judgment given with `repo` may also land on a shape written this +turn that the sync has not seen yet (milestone 439). """ from __future__ import annotations @@ -14,7 +15,7 @@ from scribe.services import shape_ledger as shape_ledger_svc async def classify_shapes( - project_id: int, classifications: list[dict], via: str = "agent" + project_id: int, classifications: list[dict], via: str = "agent", repo: str = "" ) -> dict: """Record judgments for a project's code shapes — in batch, as rows. @@ -52,15 +53,22 @@ async def classify_shapes( and get_snippet's `uses` read them. via: Who is judging — "agent" (default), "audit" (a sweep), or "import" (carrying maps recorded elsewhere). + repo: The repo the shapes live in (its remote URL or bound key), for + judging what you wrote THIS turn — shapes the ledger has not + synced yet. With it, a shape no row matches is recorded as a + provisional row under that repo and judged; the next sync + confirms it. Must be bound to the project. Omit it to judge only + synced rows. - All-or-nothing: a structural error, a missing snippet target, or no write - access applies NOTHING. Returns {"classified": N, "unmatched": [...]} — - unmatched names shapes no live ledger row matches (the tree may have - moved since you listed; re-run the project's coverage refresh to re-sync). + All-or-nothing: a structural error, a missing snippet target, an unbound + repo, or no write access applies NOTHING. Returns {"classified": N, + "unmatched": [...], "provisional": N?} — unmatched names shapes no live + ledger row matches (the tree may have moved since you listed; pass `repo` + for a shape you just wrote, or re-run the project's coverage refresh). """ uid = current_user_id() return await shape_ledger_svc.classify_shapes( - uid, project_id, classifications, via=via + uid, project_id, classifications, via=via, repo=repo ) diff --git a/src/scribe/routes/plugin.py b/src/scribe/routes/plugin.py index a9cd7de..da71341 100644 --- a/src/scribe/routes/plugin.py +++ b/src/scribe/routes/plugin.py @@ -15,6 +15,7 @@ from scribe.config import Config from scribe.services import plugin_context as plugin_ctx_svc from scribe.services import repo_bindings as repo_bindings_svc from scribe.services import report_check as report_check_svc +from scribe.services import shape_check as shape_check_svc from scribe.services import task_claims as task_claims_svc from scribe.services.settings import get_admin_setting, set_setting @@ -359,6 +360,71 @@ async def report_check(): return jsonify(body) +@plugin_bp.get("/shape-check") +@login_required +async def shape_check(): + """The end-of-turn question (milestone 439): what did this turn write that + nobody has judged? + + The client's write hooks keep a ledger of the definitions each turn wrote; + its Stop hook sends them here. The server decides which carry no judgment + — against the ledger and the project's recorded snippets — records the + outcome, and for a block returns the `reason` the agent is sent back with. + A hook blocks only on a reason it received, so only on a block that was + recorded, and never on the stop that follows one (`phase=after`). + + A GET like every plugin endpoint: a read-scoped key runs the plugin, and + this records telemetry the way /report-check does. It changes no ledger + row — the agent's own classify_shapes call does that. + + Query: + written (str) — newline-separated `pathkindname` lines, + repo-relative, kind css|sym|file. + phase (str) — check | after. Anything else is a 400. + repo (opt) — working repo remote, resolved like the other arms. + project_id (opt) — explicit scope override. + """ + phase = (request.args.get("phase") or "check").strip() + if phase not in shape_check_svc.PHASES: + return jsonify({"error": f"phase must be one of {list(shape_check_svc.PHASES)}"}), 400 + written = shape_check_svc.parse_written(request.args.get("written") or "") + project_id, repo, _unbound = await _project_scope() + body: dict = {"status": "ok", "unjudged": []} + if not project_id or not written: + return jsonify(body) + from scribe.services import access + from scribe.services import coverage as coverage_svc + from scribe.services import shape_ledger as shape_ledger_svc + + if not await access.can_read_project(g.user.id, project_id): + return jsonify(body) + repo_key = repo_bindings_svc.normalize_repo_key(repo) if repo else "" + unjudged = await shape_ledger_svc.unjudged_shapes(project_id, written, repo_key=repo_key) + # A snippet recorded AT a shape is the strongest verdict there is — the + # next refresh stamps that row canonical. Until then the ledger cannot + # know, so the recorded locations answer for it: an agent who just + # recorded what it built is not asked again. + if unjudged: + recorded = { + (path, symbol) + for _sid, path, symbol in await coverage_svc._recorded_locations(g.user.id, project_id) + } + unjudged = [u for u in unjudged if (u["path"], u["symbol"]) not in recorded + and not (u["kind"] == "file" and (u["path"], "") in recorded)] + outcome = shape_check_svc.outcome_for(phase, unjudged) + await shape_check_svc.record_shape_check( + g.user.id, outcome, written=len(written), unjudged=len(unjudged), + project_id=project_id, + ) + body["outcome"] = outcome + body["unjudged"] = unjudged + if outcome == "blocked": + body["reason"] = shape_check_svc.block_reason( + unjudged, project_id=project_id, repo=repo_key or repo, + ) + return jsonify(body) + + @plugin_bp.get("/claim-session") @login_required async def claim_session(): diff --git a/src/scribe/services/repo_bindings.py b/src/scribe/services/repo_bindings.py index 5d0b85b..d8a57eb 100644 --- a/src/scribe/services/repo_bindings.py +++ b/src/scribe/services/repo_bindings.py @@ -122,6 +122,24 @@ async def bindings_for_project(user_id: int, project_id: int) -> list[RepoBindin return list(rows.scalars().all()) +async def is_bound(project_id: int, raw_repo: str) -> str: + """The normalized key when ``raw_repo`` is bound to ``project_id`` by + anyone, else "". By anyone, not by the caller: a collaborator judging a + shared project's shapes holds no binding of their own, and the question + is whether the repo belongs to the project, not who bound it.""" + key = normalize_repo_key(raw_repo or "") + if not key or not project_id: + return "" + async with async_session() as session: + found = await session.execute( + select(RepoBinding.id).where( + RepoBinding.project_id == project_id, + RepoBinding.repo_key == key, + ).limit(1) + ) + return key if found.first() is not None else "" + + async def keys_for_project(user_id: int, project_id: int) -> list[str]: """Every repo key bound to a project — the snippet→forge join (#2691). diff --git a/src/scribe/services/shape_check.py b/src/scribe/services/shape_check.py new file mode 100644 index 0000000..c92ca7f --- /dev/null +++ b/src/scribe/services/shape_check.py @@ -0,0 +1,147 @@ +"""The end-of-turn shape check: what a turn wrote that nobody has judged, and what the agent is asked. + +WHY THIS EXISTS (milestone 439) + +The shape ledger used to be filled by machinery: extraction found the +definitions, framework rules decided which were one-offs, and the write-path +hook stamped "instance of #N" with nobody reading. The agent that WROTE the +code — the one participant who knows what it is — was asked only in an audit, +weeks later, if anyone ran one. A first instance of a reusable piece was never +asked about at all, so it was never recorded, and the next session rebuilt it. + +So the question moves to the moment the knowledge exists. The client's write +hooks keep a ledger of the definitions each turn wrote; its Stop hook sends +them here at the end of the turn. Two jobs live on this side, as with the +report check (services/report_check.py): + + - DECIDING what is unjudged, against the ledger and the project's recorded + snippets — something only the server can see. + - OWNING THE WORDS. A hook carries timing and transport (plugin/PACKAGING.md); + the instruction comes from here, so every client asks the same question + and the wording changes in one place. + +And it RECORDS every outcome, so "how much did agents leave unjudged after +being asked" is a number rather than an impression. app_logs, for the reason +report_check gives. +""" +from __future__ import annotations + +import json + +from scribe.models import async_session +from scribe.models.app_log import AppLog + +# check — the turn's first stop: block if anything is unjudged. +# after — the stop that follows a block: never block again, only record +# whether the agent answered. +PHASES = ("check", "after") +OUTCOMES = ("passed", "blocked", "judged_after_block", "left_after_block") + +# How many shapes the reason names before "+N more". Enough to judge a normal +# turn in one call; a turn that wrote more than this generated something. +_LISTED = 25 + + +def _evidence_note(evidence: dict) -> str: + """One parenthesis of what the machinery holds — offered, never asserted.""" + parts = [] + looks = evidence.get("looks_like") or {} + if looks.get("snippet_id"): + score = looks.get("score") + at = f" {score:.2f}" if isinstance(score, (int, float)) else "" + parts.append(f"looks like #{looks['snippet_id']}{at}") + stamped = evidence.get("stamped") or {} + if stamped.get("snippet_id"): + parts.append(f"the hook guessed #{stamped['snippet_id']}") + if evidence.get("copies"): + parts.append("other copies of it exist") + if evidence.get("diverges_from"): + parts.append(f"#{evidence['diverges_from']} is canon in this directory") + return f" ({'; '.join(parts)})" if parts else "" + + +def block_reason(unjudged: list[dict], *, project_id: int, repo: str) -> str: + """What the agent is told at the end of a turn that wrote unjudged shapes. + + Phrased as the practice wanted (note #3565): what to decide and the one + call that records it, not a prohibition. The reusing-code skill owns the + longer version of what each verdict means. + """ + lines = [] + for item in unjudged[:_LISTED]: + new = " — new" if item.get("status") == "new" else "" + lines.append( + f"- {item['path']} · {item['symbol']} ({item['kind']}){new}" + f"{_evidence_note(item.get('evidence') or {})}" + ) + more = len(unjudged) - _LISTED + if more > 0: + lines.append(f"- … and {more} more (list_shapes(project_id={project_id}, path=…) shows them)") + repo_arg = f', repo="{repo}"' if repo else "" + return ( + f"This turn wrote {len(unjudged)} definition{'s' if len(unjudged) != 1 else ''} " + "that nobody has judged yet. You built them, so you know what each one is — " + "record it now, while that is still true, so the next session starts from your " + "answer instead of rebuilding it:\n" + + "\n".join(lines) + + "\n\nFor each: something another part of the code should reuse → record it " + "with create_snippet (name, code, when to reach for it, location); built from a " + "recorded snippet → `instance` of it; a deliberate departure from one → " + "`variant` with the why; a genuine one-off → `exempt` with a reason " + "(reason_code one-off-handler, test-helper, scoped-css, pure-helper … indexes it). " + "One call records them all: " + f"classify_shapes(project_id={project_id}{repo_arg}, classifications=[{{path, " + "symbol, kind, status, snippet_id?, reason?, reason_code?}}, …]). " + "`repo` lets a verdict land on a shape the ledger has not synced yet." + ) + + +async def record_shape_check( + user_id: int | None, + outcome: str, + *, + written: int, + unjudged: int, + project_id: int | None = None, +) -> None: + if outcome not in OUTCOMES: + raise ValueError(f"unknown shape-check outcome {outcome!r}") + details: dict = {"outcome": outcome, "written": written, "unjudged": unjudged} + if project_id: + details["project_id"] = project_id + async with async_session() as session: + session.add(AppLog( + category="plugin", + user_id=user_id, + action="shape_check", + details=json.dumps(details), + )) + await session.commit() + + +def outcome_for(phase: str, unjudged: list[dict]) -> str: + """The recorded outcome: a first stop blocks or passes; the stop after a + block records whether the agent answered, and never blocks.""" + if phase == "after": + return "left_after_block" if unjudged else "judged_after_block" + return "blocked" if unjudged else "passed" + + +def parse_written(raw: str, *, cap: int = 200) -> list[tuple[str, str, str]]: + """The hook's ledger lines, `pathkindname`, as triples. + + Kinds outside the ledger's vocabulary and malformed lines are dropped + rather than echoed into an instruction; duplicates collapse; capped.""" + out: list[tuple[str, str, str]] = [] + for line in (raw or "").splitlines(): + parts = line.split("\t") + if len(parts) != 3: + continue + path, kind, name = (p.strip() for p in parts) + if kind not in ("css", "sym", "file") or not path or not name: + continue + if (path, kind, name) not in out: + out.append((path, kind, name)) + if len(out) >= cap: + break + return out diff --git a/src/scribe/services/shape_ledger.py b/src/scribe/services/shape_ledger.py index d671099..f0511ae 100644 --- a/src/scribe/services/shape_ledger.py +++ b/src/scribe/services/shape_ledger.py @@ -535,6 +535,7 @@ async def classify_shapes( classifications: list[dict], *, via: str = "agent", + repo: str = "", ) -> dict: """Apply a batch of judgments to a project's live ledger rows. @@ -544,7 +545,19 @@ async def classify_shapes( item carries one — and a target no live row matches is reported in ``unmatched``, not an error: the tree may simply have moved since the caller listed. Idempotent by construction. + + ``repo`` — a repo bound to the project — lets a judgment land on a shape + the ledger has not synced yet (milestone 439): the shape the agent wrote + this turn. Such an item becomes a PROVISIONAL row under that repo, the + same row the write-path hook has always created, with no seen commit; + the next sync confirms it, or marks it vanished until the code reaches + the bound ref, and either way the judgment rides along (a vanished row + that reappears keeps its status). Without ``repo`` an unknown shape is + `unmatched`, exactly as before. An unbound ``repo`` is refused rather + than ignored: a verdict silently dropped reads as one recorded. """ + from scribe.services import repo_bindings as repo_bindings_svc + from scribe.services import access from scribe.services import snippets as snippets_svc @@ -569,9 +582,18 @@ async def classify_shapes( for sid in sorted(target_ids): if await snippets_svc.get_snippet(user_id, sid) is None: raise ValueError(f"snippet {sid} not found (or not readable)") + repo_key = "" + if (repo or "").strip(): + repo_key = await repo_bindings_svc.is_bound(project_id, repo) + if not repo_key: + raise ValueError( + f"repo {repo!r} is not bound to project {project_id} — " + "bind_repo it, or omit repo to judge synced shapes only" + ) now = datetime.now(timezone.utc) classified = 0 + provisional = 0 unmatched: list[dict] = [] async with async_session() as session: rows = ( @@ -592,6 +614,17 @@ async def classify_shapes( kind = (item.get("kind") or "").strip() if kind: matches = [r for r in matches if r.kind == kind] + if not matches and repo_key and item.get("status") != "unclassified": + path = (item.get("path") or "").strip() + symbol = (item.get("symbol") or "").strip() + row = CodeShape( + project_id=project_id, repo_key=repo_key, path=path, + symbol=symbol, kind=kind or snippet_kind(symbol, ""), + ) + session.add(row) + by_key.setdefault((path, symbol), []).append(row) + matches = [row] + provisional += 1 if not matches: unmatched.append({ "path": item.get("path"), "symbol": item.get("symbol"), @@ -610,7 +643,10 @@ async def classify_shapes( evidence=item.get("reason")) classified += 1 await session.commit() - return {"classified": classified, "unmatched": unmatched} + out = {"classified": classified, "unmatched": unmatched} + if provisional: + out["provisional"] = provisional + return out def rule_matches(row: CodeShape, *, path: str, pattern: str, kind: str) -> bool: @@ -2550,6 +2586,83 @@ async def flag_divergence(project_id: int, *, since: datetime | None) -> int: return flagged +# --- the end-of-turn question (milestone 439): what did this turn write that +# nobody has judged? --------------------------------------------------------- +# +# The write hooks keep a session ledger of the definitions each turn wrote; at +# the end of the turn the client asks this, and the agent that wrote them +# answers. "Judged" means a verdict SOMEONE READ: an agent's, an audit's, an +# import's. The sync's `scoped` stamp, `unclassified`, a hook stamp and no row +# at all are all unjudged — each is a machine's guess or nothing, and the +# point of the question is that the writer, who knows, says. + +def _is_judged(row) -> bool: + return ( + row.status not in _MECHANICAL_TODO + and row.classified_by not in _UNATTENDED_BY + ) + + +async def unjudged_shapes( + project_id: int, written: list[tuple[str, str, str]], *, repo_key: str = "", +) -> list[dict]: + """The (path, kind, name) shapes in ``written`` that carry no judgment. + + Each comes back as {path, kind, symbol, status, evidence} — status is the + row's, or "new" when the ledger has none yet; evidence is what the + machinery holds that the writer may want to weigh: the proposer's + candidate canon, a hook's stamp, a divergence flag. Evidence, never a + verdict — the writer is asked because the machine's guesses were the + thing that poisoned two ledgers. Order follows ``written``; duplicates + collapse. The caller has already resolved the project and checked access. + """ + wanted: list[tuple[str, str, str]] = [] + for path, kind, name in written: + key = ((path or "").strip(), (kind or "").strip(), (name or "").strip()) + if all(key) and key not in wanted: + wanted.append(key) + if not project_id or not wanted: + return [] + paths = sorted({p for p, _k, _n in wanted}) + query = select(CodeShape).where( + CodeShape.project_id == project_id, + CodeShape.vanished_at.is_(None), + CodeShape.path.in_(paths), + ) + if repo_key: + query = query.where(CodeShape.repo_key == repo_key) + async with async_session() as session: + rows = (await session.execute(query)).scalars().all() + by_key = {(r.path, r.kind, r.symbol): r for r in rows} + out: list[dict] = [] + for path, kind, name in wanted: + row = by_key.get((path, kind, name)) + if row is not None and _is_judged(row): + continue + evidence: dict = {} + if row is not None: + if row.proposed_snippet_id: + evidence["looks_like"] = { + "snippet_id": row.proposed_snippet_id, + "basis": row.proposal_basis, + "score": row.proposal_score, + } + if row.classified_by in _UNATTENDED_BY and row.snippet_id: + evidence["stamped"] = { + "snippet_id": row.snippet_id, "score": stamp_score(row.reason), + } + if row.proposal_group: + evidence["copies"] = row.proposal_group + if row.diverges_from: + evidence["diverges_from"] = row.diverges_from + out.append({ + "path": path, "kind": kind, "symbol": name, + "status": row.status if row is not None else "new", + "evidence": evidence, + }) + return out + + # How many of a canon's rows a review listing shows before "…" — enough to # judge from, not the whole table. _REVIEW_ROWS_SHOWN = 12 diff --git a/tests/test_integration_shape_classify.py b/tests/test_integration_shape_classify.py index b8a2343..3029e2a 100644 --- a/tests/test_integration_shape_classify.py +++ b/tests/test_integration_shape_classify.py @@ -1138,3 +1138,81 @@ async def test_history_records_what_was_used_when_and_drift_asks_for_a_recheck(s # reads nothing. assert len((await shape_history(owner, pid, "src"))["events"]) == 6 assert await shape_history(other, pid, "src/app.py") == {} + + +# --- milestone 439: the writer judges what it wrote, before the sync has seen it + +@pytest.mark.integration +async def test_a_shape_written_this_turn_is_judged_before_the_sync_sees_it(seeded): + """The end-of-turn verdict lands on a shape no sync has read yet: with a + bound `repo` it becomes a provisional row and is judged, the sync that + then reads the shape keeps the judgment, and a sync that does NOT have it + yet (the code has not reached the bound ref) keeps it through the vanish + and the reappearance. Without `repo` the old contract holds.""" + from scribe.services import repo_bindings as repo_bindings_svc + + owner, pid = seeded["owner"], seeded["pid"] + await repo_bindings_svc.set_binding(owner, REPO, pid) + verdict = [{"path": "web/src/lib/StatusChip.svelte", "symbol": "tone", + "kind": "sym", "status": "exempt", + "reason": "the chip's own colour switch", "reason_code": "pure-helper"}] + + out = await classify_shapes(owner, pid, verdict) + assert out["unmatched"] == [{"path": "web/src/lib/StatusChip.svelte", "symbol": "tone"}] + + with pytest.raises(ValueError, match="not bound"): + await classify_shapes(owner, pid, verdict, repo="git.example.com/alice/elsewhere") + + out = await classify_shapes(owner, pid, verdict, repo=REPO) + assert out == {"classified": 1, "unmatched": [], "provisional": 1} + + # A sync that has not got the code yet: the row vanishes, judgment intact… + await sync_repo_shapes(pid, REPO, SHAPES, seen_marker="c1") + # …and the sync that has it brings the row back with the verdict. + await sync_repo_shapes( + pid, REPO, SHAPES + [("web/src/lib/StatusChip.svelte", "sym", "tone")], seen_marker="c2", + ) + rows, _ = await list_project_shapes(owner, pid, path="web/src/lib/StatusChip.svelte") + assert len(rows) == 1 + row = rows[0] + assert (row.status, row.classified_by, row.reason_code) == ("exempt", "agent", "pure-helper") + assert row.vanished_at is None and row.last_seen_commit == "c2" + + +@pytest.mark.integration +async def test_unjudged_is_what_nobody_read(seeded): + """The end-of-turn question: a shape with no row, an unclassified one, a + sync-scoped one and a hook stamp are all asked about; an agent's or an + audit's verdict is not. Evidence travels; a verdict is never inferred.""" + from sqlalchemy import update + + from scribe.services.shape_ledger import _RESEMBLE_REASON, unjudged_shapes + + owner, pid, sid = seeded["owner"], seeded["pid"], seeded["snippet"] + await classify_shapes(owner, pid, [ + {"path": "src/app.py", "symbol": "make_app", "status": "instance", "snippet_id": sid}, + {"path": "src/util.py", "symbol": "helper", "status": "exempt", "reason": "x"}, + ]) + async with async_session() as session: + await session.execute( + update(CodeShape) + .where(CodeShape.project_id == pid, CodeShape.symbol == "helper") + .values(classified_by="hook", status="instance", snippet_id=sid, + reason=_RESEMBLE_REASON.format(sid=sid, score=0.7)) + ) + await session.commit() + + written = [ + ("src/app.py", "sym", "make_app"), # judged by an agent → not asked + ("src/app.py", "sym", "Config"), # unclassified → asked + ("src/util.py", "sym", "helper"), # a hook stamp → asked, with its evidence + ("src/new.py", "sym", "fresh"), # no row → asked as new + ("src/new.py", "sym", "fresh"), # duplicates collapse + ] + got = await unjudged_shapes(pid, written, repo_key=REPO) + assert [(g["symbol"], g["status"]) for g in got] == [ + ("Config", "unclassified"), ("helper", "instance"), ("fresh", "new"), + ] + assert got[1]["evidence"]["stamped"] == {"snippet_id": sid, "score": 0.7} + assert got[2]["evidence"] == {} + assert await unjudged_shapes(pid, []) == [] diff --git a/tests/test_services_shape_check.py b/tests/test_services_shape_check.py new file mode 100644 index 0000000..191db3b --- /dev/null +++ b/tests/test_services_shape_check.py @@ -0,0 +1,81 @@ +"""The server half of the end-of-turn shape check (milestone 439): what the +agent is asked, what the hook's ledger may say, and the outcome record.""" +import json +from unittest.mock import patch + +import pytest + +from tests.helpers import make_mock_session + + +def test_the_ledger_lines_parse_and_garbage_is_dropped(): + from scribe.services.shape_check import parse_written + + raw = ("a.py\tsym\thelper\n" + "a.py\tsym\thelper\n" # duplicate + "web/A.svelte\tfile\tA\n" + "x.py\tbogus\tthing\n" # unknown kind + "only-two\tsym\n" # malformed + "\tsym\tnameless-path\n") + assert parse_written(raw) == [("a.py", "sym", "helper"), ("web/A.svelte", "file", "A")] + assert len(parse_written("".join(f"p\tsym\tn{i}\n" for i in range(500)), cap=7)) == 7 + + +def test_a_first_stop_blocks_or_passes_and_the_next_never_blocks(): + from scribe.services.shape_check import outcome_for + + one = [{"path": "a", "kind": "sym", "symbol": "b"}] + assert outcome_for("check", one) == "blocked" + assert outcome_for("check", []) == "passed" + assert outcome_for("after", one) == "left_after_block" + assert outcome_for("after", []) == "judged_after_block" + + +def test_the_reason_names_each_shape_its_evidence_and_the_one_call(): + from scribe.services.shape_check import block_reason + + unjudged = [ + {"path": "web/A.svelte", "kind": "file", "symbol": "A", "status": "new", "evidence": {}}, + {"path": "src/x.py", "kind": "sym", "symbol": "load", "status": "unclassified", + "evidence": {"looks_like": {"snippet_id": 12, "basis": "semantic", "score": 0.84}, + "diverges_from": 9}}, + ] + reason = block_reason(unjudged, project_id=3, repo="git.example.com/a/w") + assert "2 definitions" in reason + assert "web/A.svelte · A (file) — new" in reason + assert "looks like #12 0.84" in reason and "#9 is canon in this directory" in reason + assert 'classify_shapes(project_id=3, repo="git.example.com/a/w"' in reason + for verdict in ("create_snippet", "`instance`", "`variant`", "`exempt`"): + assert verdict in reason, verdict + + +def test_a_long_turn_is_listed_with_a_tail_not_dropped(): + from scribe.services.shape_check import _LISTED, block_reason + + many = [{"path": "p.py", "kind": "sym", "symbol": f"f{i}", "status": "new", "evidence": {}} + for i in range(_LISTED + 3)] + reason = block_reason(many, project_id=1, repo="") + assert "… and 3 more" in reason + assert "repo=" not in reason + + +async def test_the_outcome_is_recorded_as_a_plugin_event(): + from scribe.services.shape_check import record_shape_check + + session = make_mock_session() + with patch("scribe.services.shape_check.async_session", return_value=session): + await record_shape_check(7, "blocked", written=5, unjudged=2, project_id=2) + row = session.add.call_args.args[0] + assert (row.category, row.action, row.user_id) == ("plugin", "shape_check", 7) + assert json.loads(row.details) == {"outcome": "blocked", "written": 5, + "unjudged": 2, "project_id": 2} + + +async def test_an_unknown_outcome_is_refused_before_anything_is_written(): + from scribe.services.shape_check import record_shape_check + + session = make_mock_session() + with patch("scribe.services.shape_check.async_session", return_value=session), \ + pytest.raises(ValueError): + await record_shape_check(7, "skipped", written=1, unjudged=1) + session.add.assert_not_called() diff --git a/tests/test_shape_check_hook.py b/tests/test_shape_check_hook.py new file mode 100644 index 0000000..249fb8f --- /dev/null +++ b/tests/test_shape_check_hook.py @@ -0,0 +1,171 @@ +"""The end-of-turn shape check (milestone 439): the write hooks note what each +write defined, and the Stop hook asks the instance which of those nobody has +judged, blocking once in the server's words. + +Runs the real shell against the shared HTTP sink. What it pins: the write +hooks' ledger (a Write of a new file adds a `file` line; a skipped path adds +nothing); silence with nothing written; a block only on a reason the instance +returned; the post-block stop recorded and let through; another hook's block +loop left alone; the ledger kept when the instance could not be reached. +""" +from __future__ import annotations + +import json +import os +import shutil +import subprocess +from pathlib import Path + +import pytest + +from tests.helpers import http_sink + +HOOKS = Path(__file__).resolve().parents[1] / "plugin" / "hooks" +CHECK = HOOKS / "scribe_shape_check.sh" +PRIOR_ART = HOOKS / "scribe_prior_art.sh" +REASON = "SERVER REASON: judge what you wrote" + + +def _env(tmp_path, url="http://127.0.0.1:9"): + for tool in ("curl", "bash", "git"): + if shutil.which(tool) is None: + pytest.skip(f"hook runtime tool {tool!r} not installed") + return {"PATH": os.environ["PATH"], "SCRIBE_URL": url, "SCRIBE_TOKEN": "t", + "TMPDIR": str(tmp_path), "HOME": str(tmp_path)} + + +def _ledger(tmp_path, session="s1"): + return tmp_path / "scribe-priorart" / f"{session}.written.ids" + + +def _write_ledger(tmp_path, lines, session="s1"): + path = _ledger(tmp_path, session) + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text("".join(f"{line}\n" for line in lines)) + return path + + +def _stop(env, cwd, active=False, session="s1"): + out = subprocess.run( + ["bash", str(CHECK)], + input=json.dumps({"session_id": session, "transcript_path": "/nonexistent", + "cwd": str(cwd), "hook_event_name": "Stop", + "stop_hook_active": active}), + capture_output=True, text=True, env=env, timeout=30, + ) + assert out.returncode == 0, out.stderr + return out.stdout.strip() + + +def _repo(tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + subprocess.run(["git", "init", "-q", str(repo)], check=True) + subprocess.run(["git", "-C", str(repo), "remote", "add", "origin", + "https://git.example.com/alice/widget.git"], check=True) + return repo + + +# ── the write hooks keep the ledger ─────────────────────────────────────── + +def _pre_write(env, repo, rel, content, tool="Write"): + field = "content" if tool == "Write" else "new_string" + out = subprocess.run( + ["bash", str(PRIOR_ART)], + input=json.dumps({"session_id": "s1", "cwd": str(repo), "tool_name": tool, + "tool_input": {"file_path": str(repo / rel), field: content}}), + capture_output=True, text=True, env=env, timeout=30, + ) + assert out.returncode == 0, out.stderr + + +def test_a_write_notes_what_it_defined_and_a_new_file_is_a_candidate(tmp_path): + with http_sink(b'{"note_ids":[]}') as (port, _seen): + env = _env(tmp_path, f"http://127.0.0.1:{port}") + repo = _repo(tmp_path) + (repo / "web").mkdir() + _pre_write(env, repo, "web/StatusChip.svelte", + "\n" + "\n") + lines = _ledger(tmp_path).read_text().splitlines() + assert "web/StatusChip.svelte\tfile\tStatusChip" in lines + assert "web/StatusChip.svelte\tsym\ttone" in lines + assert "web/StatusChip.svelte\tcss\tchip" in lines + + +def test_an_edit_to_an_existing_file_adds_no_file_line(tmp_path): + with http_sink(b'{"note_ids":[]}') as (port, _seen): + env = _env(tmp_path, f"http://127.0.0.1:{port}") + repo = _repo(tmp_path) + (repo / "lib.py").write_text("x = 1\n") + _pre_write(env, repo, "lib.py", "def helper():\n return 1\n", tool="Edit") + lines = _ledger(tmp_path).read_text().splitlines() + assert lines == ["lib.py\tsym\thelper"] + + +def test_a_skipped_path_notes_nothing(tmp_path): + with http_sink(b'{"note_ids":[]}') as (port, _seen): + env = _env(tmp_path, f"http://127.0.0.1:{port}") + repo = _repo(tmp_path) + _pre_write(env, repo, "README.md", "# def helper():\n") + assert not _ledger(tmp_path).exists() + + +# ── the Stop hook asks ──────────────────────────────────────────────────── + +def test_nothing_written_is_silent_and_asks_nothing(tmp_path): + with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen): + env = _env(tmp_path, f"http://127.0.0.1:{port}") + assert _stop(env, _repo(tmp_path)) == "" + assert seen == [] + + +def test_all_judged_passes_silently_and_empties_the_ledger(tmp_path): + with http_sink(b'{"status":"ok","outcome":"passed","unjudged":[]}') as (port, seen): + env = _env(tmp_path, f"http://127.0.0.1:{port}") + ledger = _write_ledger(tmp_path, ["a.py\tsym\thelper", "a.py\tsym\thelper"]) + assert _stop(env, _repo(tmp_path)) == "" + assert seen[0]["phase"] == ["check"] + assert seen[0]["written"] == ["a.py\tsym\thelper"], "duplicates collapse" + assert seen[0]["repo"] == ["https://git.example.com/alice/widget.git"] + assert not ledger.exists() + + +def test_unjudged_blocks_once_in_the_servers_words_then_records_the_answer(tmp_path): + reply = json.dumps({"status": "ok", "outcome": "blocked", "reason": REASON}).encode() + with http_sink(reply) as (port, seen): + env = _env(tmp_path, f"http://127.0.0.1:{port}") + repo = _repo(tmp_path) + ledger = _write_ledger(tmp_path, ["web/A.svelte\tfile\tA"]) + assert json.loads(_stop(env, repo)) == {"decision": "block", "reason": REASON} + assert ledger.exists(), "kept, so the stop after the block asks about the same set" + + # The stop after the block: recorded, never blocked, ledger consumed. + assert _stop(env, repo, active=True) == "" + assert [q["phase"] for q in seen] == [["check"], ["after"]] + assert not ledger.exists() + assert _stop(env, repo, active=True) == "" + assert len(seen) == 2 + + +def test_no_block_without_a_reason_from_the_instance(tmp_path): + with http_sink(b'{"status":"ok"}') as (port, _seen): + env = _env(tmp_path, f"http://127.0.0.1:{port}") + _write_ledger(tmp_path, ["a.py\tsym\thelper"]) + assert _stop(env, _repo(tmp_path)) == "" + + +def test_another_hooks_block_loop_is_left_alone(tmp_path): + with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen): + env = _env(tmp_path, f"http://127.0.0.1:{port}") + ledger = _write_ledger(tmp_path, ["a.py\tsym\thelper"]) + assert _stop(env, _repo(tmp_path), active=True) == "" + assert seen == [] + assert ledger.exists(), "this hook's own question is still owed" + + +def test_an_unreachable_instance_keeps_the_question_for_later(tmp_path): + env = _env(tmp_path) # port 9: nothing listens + ledger = _write_ledger(tmp_path, ["a.py\tsym\thelper"]) + assert _stop(env, _repo(tmp_path)) == "" + assert ledger.exists() -- 2.54.0 From 4cf1c6f042fb0079a1b9780495dd3668c7dc3032 Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Thu, 1 Oct 2026 08:42:10 -0400 Subject: [PATCH 4/6] feat(shapes): the write-path hook suggests and no longer stamps (milestone 439 step 4) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The hook used to land its evidence as `instance` rows, classified_by=hook — a permanent verdict nobody read, and the source of the weak stamps that poisoned two ledgers (#4608). Now the agent that wrote the code judges it at the end of the turn, and the hook's evidence is what it is shown. - shape_ledger.suggest_write_path_instances (was stamp_write_path_instances): same evidence and form gate, but it writes a PROPOSAL — proposed_snippet_id, basis "reference" (named) or "semantic" (resembles), score — and never a status. A judged row is left alone; a new shape gets an unclassified provisional row so the suggestion reaches the end-of-turn question as "looks like #N". By-name uses edges stay: a fact, not a verdict. - The prior-art line says "looks like #N … when you judge this turn's shapes, say whether it is", and the result key is `suggested`. - write_time_divergence takes `suggested`; _RESEMBLE_MIN re-documented as the suggestion bar and the weak-stamp line. - reusing-code skill no longer says the pull stamps an instance. - Tests moved to the new contract, plus a guard that nothing the hook does writes classified_by="hook". Plugin version minted. Co-Authored-By: Claude Opus 5.5 --- plugin/.claude-plugin/plugin.json | 2 +- plugin/skills/reusing-code/SKILL.md | 6 +- src/scribe/routes/plugin.py | 10 +-- src/scribe/services/plugin_context.py | 55 ++++++++------- src/scribe/services/shape_ledger.py | 70 ++++++++++++------- tests/test_integration_shape_classify.py | 89 ++++++++++++++---------- tests/test_write_path_trigger.py | 28 ++++---- 7 files changed, 150 insertions(+), 110 deletions(-) diff --git a/plugin/.claude-plugin/plugin.json b/plugin/.claude-plugin/plugin.json index f969db7..0715aba 100644 --- a/plugin/.claude-plugin/plugin.json +++ b/plugin/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "scribe", "description": "Scribe for Claude Code: connects the scribe MCP server, adds the hooks that deliver live project state and relevant records at the right moment, ships the shared client-neutral Scribe skills (using-scribe, writing-plans, reporting-back, systematic-debugging, verification, brainstorming, reusing-code, shape-accounting), and syncs your saved Scribe Processes as skills (/scribe:sync).", - "version": "2026.10.01.1239", + "version": "2026.10.01.1242", "author": { "name": "Bryan Van Deusen" }, diff --git a/plugin/skills/reusing-code/SKILL.md b/plugin/skills/reusing-code/SKILL.md index 86b8388..5d4fe98 100644 --- a/plugin/skills/reusing-code/SKILL.md +++ b/plugin/skills/reusing-code/SKILL.md @@ -34,9 +34,9 @@ through recall/auto-inject; this skill is the active reflex around that. to ask both at once. - If a snippet fits, pull it in full with `get_snippet(id)` and reuse it — its `location` points at the reference implementation. Adapt, don't re-derive. - The pull also does the accounting: the code you then write that references - or resembles it is stamped an `instance` of that canon in the shape ledger - (classified_by=hook) — reuse from memory leaves no row. + The pull also feeds the accounting: the code you then write that references + or resembles it is offered back to you as "looks like #N" when you judge the + turn's shapes — reuse from memory leaves no such evidence. - If auto-inject already surfaced a snippet title that looks relevant, that's your cue to `get_snippet` it rather than start from scratch. - **Prior art offered beside a write is not noise — read it.** When Scribe notes diff --git a/src/scribe/routes/plugin.py b/src/scribe/routes/plugin.py index da71341..90f42b1 100644 --- a/src/scribe/routes/plugin.py +++ b/src/scribe/routes/plugin.py @@ -270,10 +270,12 @@ async def write_path_prior_art(): css|sym. The shape ledger's write-path feed (#2791): when the session recently PULLED a snippet this payload references or resembles, - these land as instance rows (classified_by=hook). - Honoured only for a caller allowed to write — a - read-scoped key still gets the hint, and never - changes accounting on a GET. + these carry it as a proposal — evidence for the + agent's own end-of-turn judgment, never a + verdict (milestone 439). Honoured only for a + caller allowed to write — a read-scoped key still + gets the hint, and never changes the ledger on a + GET. """ path = (request.args.get("path") or "").strip() code = request.args.get("code") or "" diff --git a/src/scribe/services/plugin_context.py b/src/scribe/services/plugin_context.py index dccf733..3ae2c43 100644 --- a/src/scribe/services/plugin_context.py +++ b/src/scribe/services/plugin_context.py @@ -2083,16 +2083,18 @@ async def build_write_path_hint( ``stamp_shapes`` turns the same request into the ledger's write-path feed (#2791): the (kind, name) definitions the hook saw in — or enclosing — the payload. When the session has PULLED a snippet recently and this - payload references or resembles it, those shapes land as `instance` rows - (classified_by=hook, see shape_ledger.stamp_write_path_instances) and - the result's ``stamped`` lists them. The route passes it only for a - caller allowed to write — a read-scoped key gets the hint, never the - stamp. ``repo_key`` (the hook's remote, normalised) homes a provisional - row for a shape the ledger has not synced yet. + payload references or resembles it, those shapes carry it as a PROPOSAL + (see shape_ledger.suggest_write_path_instances) and the result's + ``suggested`` lists them. Never a verdict since milestone 439: the agent + that wrote the code judges it at the end of the turn, shown this as + evidence. The route passes it only for a caller allowed to write — a + read-scoped key gets the hint, never a ledger write. ``repo_key`` (the + hook's remote, normalised) homes a provisional row for a shape the + ledger has not synced yet. """ cfg = await get_writepath_config(user_id) empty = {"context": "", "note_ids": [], "sync_note_ids": [], "config": cfg, - "stamped": [], "divergence": [], "derive": [], "derive_keys": [], + "suggested": [], "divergence": [], "derive": [], "derive_keys": [], "rule_ids": []} path = (path or "").strip() if not cfg["enabled"] or not path: @@ -2356,18 +2358,18 @@ async def build_write_path_hint( menu = (placed + scored)[:max(0, top_k - len(synced))] - # The stamp runs whether or not anything is rendered — after dedup, the - # common case is a silent hint and a pulled canon being instantiated. - stamped: list[dict] = [] + # The suggestion runs whether or not anything is rendered — after dedup, + # the common case is a silent hint and a pulled canon being instantiated. + suggested: list[dict] = [] if stamp_shapes and pulled: try: - stamped = await shape_ledger_svc.stamp_write_path_instances( + suggested = await shape_ledger_svc.suggest_write_path_instances( user_id, project_id, path=path, shapes=stamp_shapes, code=code or "", pulled=pulled, resembles=resembles, repo_key=repo_key, ) except Exception: - logger.warning("Write-path ledger stamping failed", exc_info=True) + logger.warning("Write-path ledger suggestion failed", exc_info=True) # The in-band button-B check (#2793): the hook named the shapes being # written; if this directory+kind is canon-dense and a named shape isn't # (about to be) an instance of that canon, say so NOW — at the write, @@ -2376,7 +2378,7 @@ async def build_write_path_hint( if stamp_shapes and project_id: try: divergence = await shape_ledger_svc.write_time_divergence( - project_id, path, stamp_shapes, stamped, code or "", + project_id, path, stamp_shapes, suggested, code or "", ) except Exception: logger.warning("write-time divergence check failed", exc_info=True) @@ -2430,7 +2432,7 @@ async def build_write_path_hint( design_text, design_dedup = await _design_arm( user_id, project_id, path, set(exclude_derive or []), ) - if not staleness and not synced and not menu and not stamped and not divergence and not derive: + if not staleness and not synced and not menu and not suggested and not divergence and not derive: if design_text: return {**empty, "context": design_text, "derive_keys": [design_dedup]} return empty @@ -2531,8 +2533,8 @@ async def build_write_path_hint( if passage: lines.append(f"> ↳ {passage}") - if stamped: - lines.append(_stamp_line(path, stamped)) + if suggested: + lines.append(_suggest_line(path, suggested)) if divergence: lines.append(_divergence_line(path, divergence)) if derive: @@ -2694,7 +2696,7 @@ async def build_write_path_hint( "note_ids": note_ids, "sync_note_ids": sync_note_ids, "config": cfg, - "stamped": stamped, + "suggested": suggested, "divergence": divergence, "derive": derive, "derive_keys": [d["key"] for d in derive] + ( @@ -3017,22 +3019,23 @@ def _divergence_line(path: str, divergence: list[dict]) -> str: ) -def _stamp_line(path: str, stamped: list[dict]) -> str: - """One line saying what the ledger just recorded, so the session can - correct a wrong stamp in the moment rather than an audit finding it.""" +def _suggest_line(path: str, suggested: list[dict]) -> str: + """One line naming the recorded shape this code looks like, as evidence + for the judgment the agent gives at the end of the turn (milestone 439) + — not a record of one already made.""" by_snippet: dict[int, list[str]] = {} - for row in stamped: + for row in suggested: label = f".{row['symbol']}" if row["kind"] == "css" else row["symbol"] by_snippet.setdefault(int(row["snippet_id"]), []).append(f"`{label}`") parts = [ - f"{', '.join(names)} → instance of #{sid}" + f"{', '.join(names)} → looks like #{sid}" for sid, names in by_snippet.items() ] return ( - f"> Shape accounting: recorded at `{path}` — {'; '.join(parts)} " - "(classified_by=hook: you pulled that snippet this session and this " - "code references/resembles it). Not an instance? `classify_shapes` " - "overrides a hook stamp." + f"> Shape accounting: at `{path}` — {'; '.join(parts)} " + "(you pulled that snippet this session and this code names or resembles " + "it). When you judge this turn's shapes, say whether it is: `instance` " + "of it, `variant` with the why, or something else." ) diff --git a/src/scribe/services/shape_ledger.py b/src/scribe/services/shape_ledger.py index f0511ae..e1d69c5 100644 --- a/src/scribe/services/shape_ledger.py +++ b/src/scribe/services/shape_ledger.py @@ -908,11 +908,13 @@ async def snippet_consumers(user_id: int, note_id: int) -> dict: # semantic arm scored it above the write-path threshold for this # very payload. Either is evidence; the pull alone is not. # Both hold → every shape the hook named at that path, of the snippet's kind, -# becomes instance-of-N with classified_by="hook" and the evidence as reason. +# carries N as a PROPOSAL (milestone 439). Until then it became instance-of-N +# with classified_by="hook" — a verdict nobody read, and the rows written that +# way are what `stamps_to_review` still lists. # -# A hook row is EVIDENCE, not judgment: it only ever lands on rows nobody has -# judged (unclassified) or rows an earlier hook stamped, never on a canonical -# row or an agent/audit/import judgment. Re-judge with classify_shapes. +# A proposal is EVIDENCE, not judgment: it lands only on rows nobody has +# judged, never touches a status, and is what the end-of-turn question shows +# the agent as "looks like #N". The agent's classify_shapes is the verdict. # "The write path actually pulled it": a working session's reach. The # precision comes from the in-play test above, not from this window. @@ -1354,7 +1356,7 @@ def _stamp_allowed(rank: int, mine: str, canon: str) -> bool: return forms_agree(mine, canon) -async def stamp_write_path_instances( +async def suggest_write_path_instances( user_id: int, project_id: int, *, @@ -1365,22 +1367,35 @@ async def stamp_write_path_instances( resembles: dict[int, float] | None = None, repo_key: str = "", ) -> list[dict]: - """Land hook evidence as `instance` rows for the shapes being written. + """Offer hook evidence as a PROPOSAL on the shapes being written. + + This used to land the evidence as `instance` rows, classified_by=hook — + a permanent classification written with nobody reading it. That is what + poisoned two ledgers (132 and 488 rows below today's floor, #4608), and + what milestone 439 retires: the agent that wrote the code says what it is, + at the end of the turn, and this evidence is what it is shown when it + does. So the row keeps its status; the candidate canon goes into the + proposal fields (`proposed_snippet_id`, `proposal_basis` "reference" for a + by-name hit or "semantic" for a resemblance, `proposal_score`), where the + end-of-turn question reads it as "looks like #N". A judged row is left + alone entirely. ``shapes`` is the hook's (kind, name) list for ``path``; ``pulled`` is recent_pulls(); ``resembles`` maps snippet ids the semantic arm scored - for this payload to their score. Returns the rows stamped, each + for this payload to their score. Returns the suggestions, each {path, symbol, kind, snippet_id, reason} — empty in the common case. A shape the ledger has no live row for yet (it is being written right - now) gets a PROVISIONAL row under ``repo_key`` — first/last-seen empty — - so the stamp is not lost to the next sync, which either confirms the - shape (sets its seen marker) or stamps it vanished. No repo key → only - existing rows are stamped. + now) gets a PROVISIONAL unclassified row under ``repo_key`` — first/ + last-seen empty — so the suggestion survives to the end of the turn and + the next sync either confirms the shape or stamps it vanished. No repo + key → only existing rows carry a suggestion. + + Every pulled canon the payload NAMES is still recorded as a `uses` edge: + that is a fact about the code, not a verdict about what it is. When more than one pulled snippet is in play for a shape, a by-name - reference beats resemblance and the most recent pull breaks ties: a row - holds one canon (the known model limit logged on #2790). + reference beats resemblance and the most recent pull breaks ties. """ from scribe.services import access from scribe.services import snippets as snippets_svc @@ -1433,8 +1448,7 @@ async def stamp_write_path_instances( for bucket in in_play.values(): bucket.sort(key=lambda t: (t[0], t[1]), reverse=True) - now = datetime.now(timezone.utc) - stamped: list[dict] = [] + suggested: list[dict] = [] async with async_session() as session: rows = ( await session.execute( @@ -1486,9 +1500,12 @@ async def stamp_write_path_instances( by_key[(name, kind)] = row elif not (row.status in _MECHANICAL_TODO or row.classified_by == "hook"): continue # a judgment — or the canon itself — stands - await _judge(session, row, status="instance", snippet_id=sid, by="hook", - reason=why, at=now) - stamped.append({ + # A proposal, never a status: the writer decides (milestone 439). + row.proposed_snippet_id = sid + row.proposal_basis = "reference" if _rank >= 2 else "semantic" + row.proposal_score = resembles.get(sid) if _rank < 2 else 1.0 + row.proposal_group = None + suggested.append({ "path": path, "symbol": name, "kind": kind, "snippet_id": sid, "reason": why, }) @@ -1504,9 +1521,9 @@ async def stamp_write_path_instances( [t[2] for t in bucket if t[0] == 2], basis="hook", evidence="write path: pulled the snippet, payload names its symbol", ) - if stamped: + if suggested: await session.commit() - return stamped + return suggested # --- the mechanical proposer (#2792): the machine proposes, judgment classifies @@ -2210,8 +2227,11 @@ async def confirm_proposals( # A canon dominates a directory+kind when at least this many siblings are # judged (canonical/instance) and this share of them answer to one snippet. -# How much a payload must resemble a canon before the hook may assert, with -# nobody watching, that a shape IS an instance of it. +# How much a payload must resemble a canon before the hook SUGGESTS it as the +# shape's canon — and, for the rows stamped before milestone 439, the bar below +# which a stored stamp is weak (`is_weak_stamp`). Until milestone 439 this was +# the bar for the hook to ASSERT, unattended, that a shape IS an instance; the +# hook no longer asserts anything, and the history below is why. # # WHY A FLOOR AT ALL. There was none: any score the semantic arm produced # counted, so 0.69 asserted as confidently as 0.95, and the score went into @@ -2364,15 +2384,15 @@ async def canon_density( async def write_time_divergence( - project_id: int, path: str, shapes: list[tuple[str, str]], stamped: list[dict], + project_id: int, path: str, shapes: list[tuple[str, str]], suggested: list[dict], code: str = "", ) -> list[dict]: """The in-band check for the shapes the hook named at ``path``: for each kind whose directory has a dominant canon, the named shapes that are - not (already or just now) that canon's instance/canonical — new or + not that canon's instance/canonical (judged, or just suggested as it) — new or unclassified rows only; a judged shape is not re-litigated at every edit. Returns [{symbol, kind, canon_snippet_id, instances, judged}].""" - just_stamped = {(s["symbol"], s["kind"]): s["snippet_id"] for s in stamped} + just_stamped = {(s["symbol"], s["kind"]): s["snippet_id"] for s in suggested} out: list[dict] = [] if not shapes: return out diff --git a/tests/test_integration_shape_classify.py b/tests/test_integration_shape_classify.py index 3029e2a..8336248 100644 --- a/tests/test_integration_shape_classify.py +++ b/tests/test_integration_shape_classify.py @@ -322,15 +322,16 @@ async def test_sync_refiles_rows_whose_snippet_was_purged(seeded): @pytest.mark.integration -async def test_write_path_stamp_is_evidence_that_yields_to_judgment(seeded): - """Pulled + referenced → every named shape of the snippet's kind becomes - an instance row, classified_by=hook, carrying the evidence as reason. A - later agent judgment on one of them stands against a re-stamp; the hook - may only overwrite nobody's judgment or its own. The outsider stamps - nothing (write-gated like every other ledger write).""" +async def test_write_path_evidence_is_a_proposal_never_a_verdict(seeded): + """Milestone 439. Pulled + referenced → every named shape of the + snippet's kind CARRIES the snippet as a proposal (basis "reference"), + and its status is untouched: the agent that wrote it says what it is. + A judgment is never touched, not even by a proposal. The outsider + writes nothing (write-gated like every other ledger write). The uses + edge — a fact about the code — is still recorded.""" from datetime import datetime, timezone - from scribe.services.shape_ledger import stamp_write_path_instances + from scribe.services.shape_ledger import suggest_write_path_instances owner, other, pid, sid = ( seeded["owner"], seeded["other"], seeded["pid"], seeded["snippet"] @@ -338,84 +339,98 @@ async def test_write_path_stamp_is_evidence_that_yields_to_judgment(seeded): pulled = {sid: datetime.now(timezone.utc)} code = "app = factory()\nreturn app\n" # references the snippet's symbol - assert await stamp_write_path_instances( + assert await suggest_write_path_instances( other, pid, path="src/app.py", shapes=[("sym", "make_app")], code=code, pulled=pulled, ) == [] - stamped = await stamp_write_path_instances( + suggested = await suggest_write_path_instances( owner, pid, path="src/app.py", shapes=[("sym", "make_app"), ("sym", "Config"), ("css", "nope")], code=code, pulled=pulled, ) - assert {s["symbol"] for s in stamped} == {"make_app", "Config"} # css skipped: no css canon - rows, _ = await list_project_shapes(owner, pid, snippet_id=sid) + assert {s["symbol"] for s in suggested} == {"make_app", "Config"} # css skipped: no css canon + rows, _ = await list_project_shapes(owner, pid, path="src/app.py") by_symbol = {r.symbol: r for r in rows} - assert by_symbol["make_app"].status == "instance" - assert by_symbol["make_app"].classified_by == "hook" - assert by_symbol["make_app"].reason == f"hook: pulled #{sid}; payload references `factory`" + for name in ("make_app", "Config"): + row = by_symbol[name] + assert row.status == "unclassified" and row.classified_by is None, name + assert (row.proposed_snippet_id, row.proposal_basis) == (sid, "reference"), name + # The by-name reference is still recorded as a call-site fact. + users, _ = await list_project_shapes(owner, pid, uses=sid) + assert {"make_app", "Config"} <= {r.symbol for r in users} - # A judgment lands; the next stamp must leave it alone but may re-stamp - # its own earlier row. + # A judgment stands: the next suggestion leaves the judged row alone. await classify_shapes(owner, pid, [ {"path": "src/app.py", "symbol": "make_app", "status": "exempt", "reason": "the app factory is its own thing"}, ]) - again = await stamp_write_path_instances( + again = await suggest_write_path_instances( owner, pid, path="src/app.py", shapes=[("sym", "make_app"), ("sym", "Config")], code=code, pulled=pulled, ) assert {s["symbol"] for s in again} == {"Config"} rows, _ = await list_project_shapes(owner, pid, path="src/app.py") - by_symbol = {r.symbol: r for r in rows} - assert by_symbol["make_app"].status == "exempt" - assert by_symbol["Config"].status == "instance" + assert {r.symbol: r.status for r in rows}["make_app"] == "exempt" # Neither pulled nor in play → nothing, even with shapes named. - assert await stamp_write_path_instances( + assert await suggest_write_path_instances( owner, pid, path="src/util.py", shapes=[("sym", "helper")], code="print('unrelated')", pulled=pulled, ) == [] + # And nothing the hook does writes classified_by="hook" any more. + async with async_session() as s: + hook_rows = (await s.execute(select(CodeShape).where( + CodeShape.project_id == pid, CodeShape.classified_by == "hook", + ))).scalars().all() + assert hook_rows == [] + @pytest.mark.integration -async def test_a_brand_new_shape_gets_a_provisional_row_the_sync_settles(seeded): +async def test_a_brand_new_shape_gets_a_provisional_row_carrying_the_suggestion(seeded): """The shape being written right now has no ledger row yet. With the - hook's repo key it gets a provisional one — seen markers empty — so the - stamp survives until the next sync, which confirms it (sets the marker) - or stamps it vanished. Without a repo key only existing rows are touched.""" + hook's repo key it gets a provisional, UNCLASSIFIED one — seen markers + empty — carrying the suggestion to the end-of-turn question; the sync + confirms it or stamps it vanished. Without a repo key only existing rows + carry a suggestion.""" from datetime import datetime, timezone - from scribe.services.shape_ledger import stamp_write_path_instances + from scribe.services.shape_ledger import suggest_write_path_instances, unjudged_shapes owner, pid, sid = seeded["owner"], seeded["pid"], seeded["snippet"] pulled = {sid: datetime.now(timezone.utc)} code = "def build():\n return factory()\n" - assert await stamp_write_path_instances( + assert await suggest_write_path_instances( owner, pid, path="src/new.py", shapes=[("sym", "build")], code=code, pulled=pulled, # no repo_key ) == [] - stamped = await stamp_write_path_instances( + suggested = await suggest_write_path_instances( owner, pid, path="src/new.py", shapes=[("sym", "build")], code=code, pulled=pulled, repo_key=REPO, ) - assert [s["symbol"] for s in stamped] == ["build"] + assert [s["symbol"] for s in suggested] == ["build"] async with async_session() as s: row = (await s.execute(select(CodeShape).where( CodeShape.project_id == pid, CodeShape.path == "src/new.py", ))).scalar_one() - assert row.status == "instance" and row.classified_by == "hook" + assert row.status == "unclassified" and row.classified_by is None + assert row.proposed_snippet_id == sid # "Unset" is the column's empty default — the markers are non-null # Text, and the sync is what first fills them. assert row.first_seen_commit == "" and row.last_seen_commit == "" - # The sync sees the shape in the tree → confirmed, stamp intact. + # The end-of-turn question shows the suggestion as evidence. + asked = await unjudged_shapes(pid, [("src/new.py", "sym", "build")], repo_key=REPO) + assert asked[0]["evidence"]["looks_like"]["snippet_id"] == sid + + # The sync sees the shape in the tree → confirmed, still unjudged. await sync_repo_shapes( pid, REPO, SHAPES + [("src/new.py", "sym", "build")], seen_marker="abc123", ) rows, _ = await list_project_shapes(owner, pid, path="src/new.py") - assert rows[0].status == "instance" and rows[0].last_seen_commit == "abc123" + assert rows[0].status == "unclassified" and rows[0].last_seen_commit == "abc123" # The sync no longer sees it → vanished, out of the live accounting. await sync_repo_shapes(pid, REPO, SHAPES, seen_marker="def456") @@ -866,19 +881,19 @@ async def test_a_second_confirm_dialog_is_detected_and_named(seeded): # In-band: the hook names the shape at write time → the check names the canon. named = await write_time_divergence( - pid, f"{comp}/Danger.vue", [("sym", "confirmDanger")], stamped=[] + pid, f"{comp}/Danger.vue", [("sym", "confirmDanger")], suggested=[] ) assert named == [{"symbol": "confirmDanger", "kind": "sym", "canon_snippet_id": sid, "instances": 4, "judged": 4}] # ...but an already-judged shape, or one just stamped as the canon's # instance, is not re-litigated. - assert await write_time_divergence(pid, f"{comp}/Trash.vue", [("sym", "onTrash")], stamped=[]) == [] + assert await write_time_divergence(pid, f"{comp}/Trash.vue", [("sym", "onTrash")], suggested=[]) == [] assert await write_time_divergence( pid, f"{comp}/New.vue", [("sym", "onNew")], - stamped=[{"symbol": "onNew", "kind": "sym", "snippet_id": sid}], + suggested=[{"symbol": "onNew", "kind": "sym", "snippet_id": sid}], ) == [] # A directory with no dominant canon is silent. - assert await write_time_divergence(pid, "src/other.py", [("sym", "thing")], stamped=[]) == [] + assert await write_time_divergence(pid, "src/other.py", [("sym", "thing")], suggested=[]) == [] # The judgment answers the question and clears the flag. await classify_shapes(owner, pid, [ @@ -956,7 +971,7 @@ async def test_weak_stamps_elect_no_canon_and_their_flags_are_withdrawn(seeded): rows, total = await list_project_shapes(owner, pid, flag="divergence") assert total == 0, "a flag whose canon no longer dominates is withdrawn" assert await write_time_divergence( - pid, f"{comp}/Danger.vue", [("sym", "confirmDanger")], stamped=[] + pid, f"{comp}/Danger.vue", [("sym", "confirmDanger")], suggested=[] ) == [] # Nothing was un-judged: the stamps stand until someone reads them. async with async_session() as session: diff --git a/tests/test_write_path_trigger.py b/tests/test_write_path_trigger.py index 02c2f9b..8800f5f 100644 --- a/tests/test_write_path_trigger.py +++ b/tests/test_write_path_trigger.py @@ -1291,9 +1291,9 @@ async def test_stamping_needs_named_shapes_and_a_recent_pull(): patch.object(pc, "semantic_search_notes", AsyncMock(return_value=[])), \ patch.object(pc, "record_retrieval", MagicMock()), \ patch.object(pc.shape_ledger_svc, "recent_pulls", pulls), \ - patch.object(pc.shape_ledger_svc, "stamp_write_path_instances", stamp): + patch.object(pc.shape_ledger_svc, "suggest_write_path_instances", stamp): out = await pc.build_write_path_hint(1, "src/x.py", code=REAL_CODE, project_id=4) - assert out["stamped"] == [] + assert out["suggested"] == [] pulls.assert_not_awaited() # no shapes → no read out = await pc.build_write_path_hint( 1, "src/x.py", code=REAL_CODE, project_id=4, @@ -1301,7 +1301,7 @@ async def test_stamping_needs_named_shapes_and_a_recent_pull(): ) pulls.assert_awaited_once() stamp.assert_not_awaited() # shapes, but no pull - assert out["stamped"] == [] + assert out["suggested"] == [] @pytest.mark.asyncio @@ -1330,7 +1330,7 @@ async def test_a_pulled_snippet_already_seen_is_evidence_not_menu(): patch.object(pc, "record_retrieval", MagicMock()), \ patch.object(pc, "owner_names_for", AsyncMock(return_value={})), \ patch.object(pc.shape_ledger_svc, "recent_pulls", AsyncMock(return_value={7: _ts()})), \ - patch.object(pc.shape_ledger_svc, "stamp_write_path_instances", stamp): + patch.object(pc.shape_ledger_svc, "suggest_write_path_instances", stamp): out = await pc.build_write_path_hint( 1, "src/x.py", code=REAL_CODE, project_id=4, exclude_ids=[7], stamp_shapes=[("sym", "debounce")], repo_key="git.example.com/a/b", @@ -1355,11 +1355,11 @@ async def test_a_pulled_snippet_already_seen_is_evidence_not_menu(): assert skw["resembles"] == {7: 0.91} assert skw["shapes"] == [("sym", "debounce")] assert skw["repo_key"] == "git.example.com/a/b" - assert out["stamped"][0]["snippet_id"] == 7 - # And the session is told what landed, with the way to correct it. + assert out["suggested"][0]["snippet_id"] == 7 + # And the session is shown it as evidence for its own end-of-turn judgment. assert "Shape accounting" in out["context"] - assert "`debounce` → instance of #7" in out["context"] - assert "classify_shapes" in out["context"] + assert "`debounce` → looks like #7" in out["context"] + assert "When you judge this turn's shapes" in out["context"] @pytest.mark.asyncio @@ -1376,15 +1376,15 @@ async def test_a_stamp_renders_even_when_the_hint_is_otherwise_silent(): patch.object(pc, "record_retrieval", MagicMock()), \ patch.object(pc, "owner_names_for", AsyncMock(return_value={})), \ patch.object(pc.shape_ledger_svc, "recent_pulls", AsyncMock(return_value={5: _ts()})), \ - patch.object(pc.shape_ledger_svc, "stamp_write_path_instances", + patch.object(pc.shape_ledger_svc, "suggest_write_path_instances", AsyncMock(return_value=stamped)): out = await pc.build_write_path_hint( 1, "web/b.css", code=".btn-primary { color: red; }" * 4, project_id=4, stamp_shapes=[("css", "btn-primary")], ) assert out["note_ids"] == [] - assert out["stamped"] == stamped - assert "`.btn-primary` → instance of #5" in out["context"] + assert out["suggested"] == stamped + assert "`.btn-primary` → looks like #5" in out["context"] @pytest.mark.asyncio @@ -1397,13 +1397,13 @@ async def test_a_failing_stamp_does_not_sink_the_hint(): patch.object(pc, "record_retrieval", MagicMock()), \ patch.object(pc, "owner_names_for", AsyncMock(return_value={})), \ patch.object(pc.shape_ledger_svc, "recent_pulls", AsyncMock(return_value={12: _ts()})), \ - patch.object(pc.shape_ledger_svc, "stamp_write_path_instances", + patch.object(pc.shape_ledger_svc, "suggest_write_path_instances", AsyncMock(side_effect=RuntimeError("ledger down"))): out = await pc.build_write_path_hint( 1, "src/x.py", code=REAL_CODE, project_id=4, stamp_shapes=[("sym", "f")], ) assert out["sync_note_ids"] == [12] - assert out["stamped"] == [] + assert out["suggested"] == [] def test_route_stamps_only_for_a_caller_allowed_to_write(): @@ -1760,7 +1760,7 @@ async def test_a_record_this_arm_withheld_itself_is_not_a_near_miss(): patch.object(pc, "owner_names_for", AsyncMock(return_value={})), \ patch.object(pc.shape_ledger_svc, "recent_pulls", AsyncMock(return_value={7: _ts()})), \ - patch.object(pc.shape_ledger_svc, "stamp_write_path_instances", + patch.object(pc.shape_ledger_svc, "suggest_write_path_instances", AsyncMock(return_value=[])), \ patch.object(pc, "semantic_search_notes", _search_reporting(0.9, fake_note( -- 2.54.0 From 8b1567cf2cd5997da217b258a18aba753b698454 Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Thu, 1 Oct 2026 08:45:14 -0400 Subject: [PATCH 5/6] feat(shapes): a component file is a candidate shape in its own right (milestone 439 step 5) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A single-file component defines `.card`, `.title` and a `Props`, never anything named after itself — so the ledger could not hold "StatusChip is canon" and could not say a new card was built where one already existed. - coverage.is_file_unit: a file that RENDERS (its markup names classes, or it opens a \n\n", True), + ("web/views/Card.vue", "\n", True), + # A TSX component already has its row: the function named after the file. + ("web/Card.tsx", "export function Card() {\n return
;\n}\n", False), + # A JSX file whose markup names classes but whose component is named otherwise. + ("web/index.jsx", "function App() {\n return
;\n}\n", True), + # Modules render nothing: their definitions account for them. + ("internal/api/upnext.go", "func stageOf(x int) int {\n\treturn x\n}\n", False), + ("src/app/util.py", "def helper():\n pass\n", False), + ("web/button.css", ".btn {\n color: red;\n}\n", False), +]) +def test_a_file_is_a_unit_when_it_renders_and_nothing_inside_carries_its_name(path, text, unit): + """Milestone 439: one structural test, not a framework list.""" + from scribe.services.coverage import class_references, extract_definitions, is_file_unit + + assert is_file_unit(path, text, extract_definitions(text), class_references(path, text)) is unit -- 2.54.0 From 582a5a4f48bc0ff88f05bf05d0227ffd15830bc6 Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Thu, 1 Oct 2026 08:46:58 -0400 Subject: [PATCH 6/6] feat(shapes): the practice is written where it is read, and the coverage line measures the slip (milestone 439 step 6) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - reusing-code: "Before the turn ends — say what you built" — the four verdicts, the one classify_shapes(repo=…) call, and the component file as a shape. Description names the end-of-turn moment. - shape-accounting: the writer judges; audits are the check that it held. The write path SUGGESTS (no more hook instances); component file rows and whole-file canon described; scoped covers Svelte too. - _INSTRUCTIONS reuse line: "before the turn ends, say what you built (create_snippet the reusable, classify_shapes the rest)" — 1570/1600. - Coverage line: "written-shape check (7d): N turns checked, M asked, K left unjudged", from the Stop hook's recorded outcomes; silent until the question has been put. - test_guidance_ownership pins the new topic on reusing-code. - prior-art hook header no longer says it stamps instance rows. Plugin version minted. Co-Authored-By: Claude Opus 5.5 --- plugin/.claude-plugin/plugin.json | 2 +- plugin/hooks/scribe_prior_art.sh | 9 ++--- plugin/skills/reusing-code/SKILL.md | 28 ++++++++++++++- plugin/skills/shape-accounting/SKILL.md | 40 +++++++++++++--------- src/scribe/mcp/server.py | 3 +- src/scribe/services/coverage.py | 22 ++++++++++++ src/scribe/services/shape_check.py | 45 +++++++++++++++++++++++++ tests/test_guidance_ownership.py | 5 +++ tests/test_pattern_coverage.py | 13 +++++++ tests/test_services_shape_check.py | 8 +++++ 10 files changed, 153 insertions(+), 22 deletions(-) diff --git a/plugin/.claude-plugin/plugin.json b/plugin/.claude-plugin/plugin.json index 1874341..45eb2b3 100644 --- a/plugin/.claude-plugin/plugin.json +++ b/plugin/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "scribe", "description": "Scribe for Claude Code: connects the scribe MCP server, adds the hooks that deliver live project state and relevant records at the right moment, ships the shared client-neutral Scribe skills (using-scribe, writing-plans, reporting-back, systematic-debugging, verification, brainstorming, reusing-code, shape-accounting), and syncs your saved Scribe Processes as skills (/scribe:sync).", - "version": "2026.10.01.1245", + "version": "2026.10.01.1246", "author": { "name": "Bryan Van Deusen" }, diff --git a/plugin/hooks/scribe_prior_art.sh b/plugin/hooks/scribe_prior_art.sh index c3a5715..5a741b9 100755 --- a/plugin/hooks/scribe_prior_art.sh +++ b/plugin/hooks/scribe_prior_art.sh @@ -17,8 +17,9 @@ # It is also the shape ledger's write-path feed (#2791): it names the # definitions being written (`shapes=`), and the server — only when the # session has PULLED a snippet this code references or resembles — records -# them as instance rows, classified_by=hook. Evidence, not judgment; the -# context line says what landed so a wrong stamp is corrected in the moment. +# it as a PROPOSAL on them ("looks like #N"), never a status (milestone 439). +# The agent's own end-of-turn judgment is the verdict; the same names go to +# the written-shapes ledger the Stop hook asks about. # # NEVER BLOCKS. It returns `additionalContext` with no `permissionDecision`, so # the write proceeds untouched and Claude sees the note beside the tool result. @@ -103,8 +104,8 @@ fi # changes the inside of a function rather than its signature — the definition # enclosing the edit, found by walking the target file upward from the edited # lines. The server decides whether evidence exists (the session pulled a -# snippet this code references or resembles) and stamps instance rows; with -# no pulled canon in play, nothing is recorded. Titles only still — this sends +# snippet this code references or resembles) and records it as a proposal; +# with no pulled canon in play, nothing is recorded. Titles only still — this sends # names, not bodies. # --------------------------------------------------------------------------- # diff --git a/plugin/skills/reusing-code/SKILL.md b/plugin/skills/reusing-code/SKILL.md index 5d4fe98..0aeab71 100644 --- a/plugin/skills/reusing-code/SKILL.md +++ b/plugin/skills/reusing-code/SKILL.md @@ -1,6 +1,6 @@ --- name: reusing-code -description: Use when you're about to build ANY shape — a component, control, route handler, service class, helper, test scaffold — search recorded snippets FIRST and start from the recorded shape instead of re-solving it. And the FIRST time a shape is built, record it as a snippet so every later instance starts from it. Triggers on "write a util/helper", "I need a function that…", "let me add a component/button/field/route", or having just built the first instance of anything. +description: Use when you're about to build ANY shape — a component, control, route handler, service class, helper, test scaffold — search recorded snippets FIRST and start from the recorded shape instead of re-solving it. And before the turn that built it ends, say what it is — record the reusable as a snippet, classify the rest — so every later instance starts from it. Triggers on "write a util/helper", "I need a function that…", "let me add a component/button/field/route", having just built the first instance of anything, or the end-of-turn list of shapes to judge. --- # Reusing code — the pattern library @@ -85,6 +85,32 @@ through recall/auto-inject; this skill is the active reflex around that. per reusable thing. If it already exists, `update_snippet` it instead of recording a second copy (the create gate will flag a near-duplicate anyway). +## Before the turn ends — say what you built + +You are the one who knows what the code you just wrote is. The shape ledger +records it from your answer, not from a guess made later, so give the answer +in the turn that built it — while it is still true. When a turn wrote +definitions nobody has judged, the plugin's Stop hook lists them once, with +whatever evidence the machinery holds ("looks like #N", "#M is canon in this +directory", "new"); other clients reach the same moment through +`list_shapes(project_id, path=…)`. For each one, decide: + +- **Something another part of the code should reuse** → record it with + `create_snippet` (the fields above). Its own row becomes the canon. +- **Built from a recorded snippet** → `instance` of it. +- **A deliberate departure from one** → `variant`, with the why. +- **A genuine one-off** → `exempt`, with a reason (`reason_code` such as + `one-off-handler`, `test-helper`, `scoped-css` or `pure-helper` indexes it). + +One call records them all: `classify_shapes(project_id, repo="", classifications=[{path, symbol, kind, status, snippet_id?, reason?}])`. +`repo` lets a verdict land on a shape the ledger has not synced yet — the one +you wrote a minute ago. A component file is a shape too (`kind: "file"`, named +by its stem): when a card, chip or dialog is the reusable thing, the file is +what you record. An unsure call is still yours to make — the reason you write +is what the next reader judges it by — and a verdict is permanent, so each +shape is asked about once. + ## A shared snippet is a suggestion, not a standard Scribe is multi-user, so a search can return snippets other people own. Those diff --git a/plugin/skills/shape-accounting/SKILL.md b/plugin/skills/shape-accounting/SKILL.md index 740ebe7..dff2eff 100644 --- a/plugin/skills/shape-accounting/SKILL.md +++ b/plugin/skills/shape-accounting/SKILL.md @@ -18,8 +18,8 @@ row carries a status: - `exempt` — judged genuinely one-off. **Reason required.** A recorded judgment, not silence — it stops the next pass re-litigating it. - `scoped` — one-off **by construction**, stamped by the coverage sync - (a Vue component's scoped `