diff --git a/plugin/.claude-plugin/plugin.json b/plugin/.claude-plugin/plugin.json
index ae281d1..6017ce9 100644
--- a/plugin/.claude-plugin/plugin.json
+++ b/plugin/.claude-plugin/plugin.json
@@ -1,7 +1,7 @@
{
"name": "scribe",
"description": "Scribe for Claude Code: connects the scribe MCP server, adds the hooks that deliver live project state and relevant records at the right moment, ships the shared client-neutral Scribe skills (using-scribe, writing-plans, reporting-back, systematic-debugging, verification, brainstorming, reusing-code, shape-accounting), and syncs your saved Scribe Processes as skills (/scribe:sync).",
- "version": "2026.09.14.2143",
+ "version": "2026.09.14.2234",
"author": {
"name": "Bryan Van Deusen"
},
diff --git a/plugin/PACKAGING.md b/plugin/PACKAGING.md
index a0919f6..fd96a0e 100644
--- a/plugin/PACKAGING.md
+++ b/plugin/PACKAGING.md
@@ -15,7 +15,7 @@ another one means adding files, not moving or rewriting any.
|---|---|---|
| **The skills** | `plugin/skills/*/SKILL.md` | Agent Skills (the open SKILL.md format). They state every Scribe reflex in full and name no client. `tests/test_guidance_ownership.py` fails if a skill names a particular client, or references anything outside its own folder. Every client package ships this folder verbatim. |
| **The MCP server** | `/mcp` | HTTP, `Authorization: Bearer `. Its `_INSTRUCTIONS` is a client-neutral index (≤2,000 chars); each tool's description carries its contract; in-band responses (`placement`, `report_back`, `systems_hint`, the duplicate gate, the guessed-id refusal) fire in every client. |
-| **The adapter API** | `/api/plugin/*` | Plain `GET` endpoints any client's hooks can call with the same key (read scope is enough): `context` (live session state), `retrieve` (rules, preferences and notes for a message), `prior-art` (records and shape-ledger hints for code being written), `tool-rules` (rules for a command about to run), `processes` (stored Processes to expose as skills). |
+| **The adapter API** | `/api/plugin/*` | Plain `GET` endpoints any client's hooks can call with the same key (read scope is enough): `context` (live session state), `retrieve` (rules, preferences and notes for a message), `prior-art` (records and shape-ledger hints for code being written), `tool-rules` (rules for a command about to run), `report-check` (records a completion-report check and returns the reason for a block), `processes` (stored Processes to expose as skills). |
| **The API key** | Scribe → Settings → API Keys | One `fmcp_` key per install. Read scope for hooks; write scope for the MCP tools. |
## Added by each client
@@ -39,6 +39,7 @@ another one means adding files, not moving or rewriting any.
| `hooks/scribe_prior_art.sh` | PreToolUse on editor writes: `GET /api/plugin/prior-art`. |
| `hooks/scribe_after_write.sh` | PostToolUse on shell commands: the same check for code written through the shell. |
| `hooks/scribe_tool_rules.sh` | PreToolUse on shell commands: `GET /api/plugin/tool-rules`. |
+| `hooks/scribe_report_check.sh` | Stop: when the turn closed a task, checks the reply for the completion sections and reports to `GET /api/plugin/report-check`; blocks once, with the reason the server returns. |
| `hooks/scribe_sync_processes.sh` + `commands/sync.md` | `GET /api/plugin/processes` → `~/.claude/skills/scribe-proc-*` stubs; `/scribe:sync` on demand. |
| `hooks/scribe_defs.sh` | Shared shell helpers: config, dedup ledgers, outage line. |
| `hooks/scribe_static_context.md` | The adapter static text. |
diff --git a/plugin/README.md b/plugin/README.md
index dea320e..cfd79e9 100644
--- a/plugin/README.md
+++ b/plugin/README.md
@@ -82,6 +82,15 @@ On install you'll be asked for:
answer" line (8 s budget here — it runs after the tool, so it gates
nothing). The extractor, the prose/data skip list, the local by-name
duplicate arm and the outage line are shared in `hooks/scribe_defs.sh`.
+- `hooks/hooks.json` → Stop hook (`hooks/scribe_report_check.sh`): when the
+ turn closed a Scribe task (`update_task`/`create_task` with status done),
+ checks the reply that ends it for the completion sections — where the work
+ sits, what needs you, what comes next — and reports the outcome to
+ `GET /api/plugin/report-check`. If sections are missing it blocks once with
+ the reason the server returns, and records how the rewrite came out; it
+ never blocks twice, and never blocks when the instance did not record the
+ check (unconfigured or unreachable). Outcomes land in the admin logs under
+ category `plugin`, action `report_check`.
- `skills/` → the universal process-skills, surfaced by description match.
- `hooks/scribe_sync_processes.sh` (a 2nd SessionStart hook) + the `/scribe:sync`
command → generate `~/.claude/skills/scribe-proc-*` stubs from your Scribe
diff --git a/plugin/hooks/hooks.json b/plugin/hooks/hooks.json
index 0bec8ea..c55be67 100644
--- a/plugin/hooks/hooks.json
+++ b/plugin/hooks/hooks.json
@@ -54,6 +54,16 @@
}
]
}
+ ],
+ "Stop": [
+ {
+ "hooks": [
+ {
+ "type": "command",
+ "command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_report_check.sh\""
+ }
+ ]
+ }
]
}
}
diff --git a/plugin/hooks/scribe_report_check.sh b/plugin/hooks/scribe_report_check.sh
new file mode 100644
index 0000000..72f8a5d
--- /dev/null
+++ b/plugin/hooks/scribe_report_check.sh
@@ -0,0 +1,186 @@
+#!/usr/bin/env bash
+# Scribe plugin — Stop hook: a reply that closes a task carries the completion
+# sections (milestone 409 step 5).
+#
+# Everything else the plugin does happens BEFORE the agent writes: context,
+# retrieval, the reporting-back skill. This is the one moment the finished
+# reply exists, so it is both the last chance to fix a report the operator
+# cannot read and the only place adherence to the shape can be measured.
+#
+# DETERMINISTIC, NO MODEL CALL. Three questions, cheapest first:
+#
+# 1. Did this turn close a Scribe task? An `update_task` / `create_task` tool
+# call with status "done" since the turn's prompt, whose result was not an
+# error. Most turns stop here, silently.
+# 2. Does the reply that ends the turn have the completion sections? Loosely:
+# where the work sits (a record named by id and title, or step N of M),
+# what needs the operator, and what comes next. Matched on the words that
+# carry the meaning rather than exact headings, so the skill's wording can
+# change without breaking this.
+# 3. If sections are missing, block once with a reason naming them. The agent
+# rewrites; the rewrite is checked and recorded, and never blocked again.
+#
+# MEASURED FROM THE FIRST CALL. Each checked reply is reported to the instance
+# (`/api/plugin/report-check`): passed, blocked, and after a rewrite either
+# passed_after_rewrite or missing_after_rewrite. Turns that closed no task are
+# not reported — they would cost a request on every turn and add nothing to
+# the rate step 6 reads (blocked among checked replies).
+#
+# IT BLOCKS ONLY WHEN THE BLOCK IS RECORDED, AND ONLY IN THE SERVER'S WORDS.
+# The report goes out first; the instance answers a recorded `blocked` with
+# the reason to send the agent back with, and the hook blocks only on that
+# reason. An unconfigured or unreachable instance therefore never stops a
+# session, every intervention is one the numbers can see, and the guidance
+# text lives on the server (plugin/PACKAGING.md: hooks carry timing and
+# transport).
+#
+# THE TRANSCRIPT FORMAT IS OBSERVED, NOT DOCUMENTED. Claude Code documents
+# `transcript_path` and `stop_hook_active` for Stop, not the JSONL inside. As
+# read from real transcripts (2026-09-14): one content block per line;
+# `type: "assistant"` lines carry `message.content[]` blocks of `text` /
+# `tool_use` ({id, name, input}); tool results arrive as `type: "user"` lines
+# whose content is a `tool_result` array ({tool_use_id, is_error}); a turn's
+# prompt — typed, or a background-task notification — is a `user` line whose
+# content is a plain string and which is not `isMeta`. Anything that does not
+# parse that way makes the hook stay out of the way rather than guess.
+#
+# Config (same as the other hooks):
+# CLAUDE_PLUGIN_OPTION_API_ENDPOINT base URL, no trailing slash
+# CLAUDE_PLUGIN_OPTION_API_TOKEN fmcp_ API key (sensitive)
+# SCRIBE_URL / SCRIBE_TOKEN override for the settings.json dogfooding path.
+set -uo pipefail
+
+command -v jq >/dev/null 2>&1 || exit 0
+command -v curl >/dev/null 2>&1 || exit 0
+
+# shellcheck source=plugin/hooks/scribe_defs.sh
+. "$(dirname "${BASH_SOURCE[0]}")/scribe_defs.sh"
+
+# Stop delivers { session_id, transcript_path, cwd, hook_event_name, stop_hook_active }.
+event=$(cat 2>/dev/null || true)
+transcript=$(printf '%s' "$event" | jq -r '.transcript_path // empty' 2>/dev/null) || exit 0
+[ -n "$transcript" ] && [ -f "$transcript" ] || exit 0
+session_id=$(printf '%s' "$event" | jq -r '.session_id // empty' 2>/dev/null) || session_id=""
+active=$(printf '%s' "$event" | jq -r '.stop_hook_active // false' 2>/dev/null) || active="false"
+event_cwd=$(printf '%s' "$event" | jq -r '.cwd // empty' 2>/dev/null) || event_cwd=""
+
+safe_sid=$(printf '%s' "${session_id:-nosession}" | tr -c 'A-Za-z0-9._-' '_')
+state_dir="${TMPDIR:-/tmp}/scribe-reportcheck"
+mkdir -p "$state_dir" 2>/dev/null || true
+marker="$state_dir/${safe_sid}.blocked"
+
+# Cheap prefilter: no task tool anywhere in the recent transcript → nothing to
+# check. Keeps the ordinary turn at one grep. Process substitution, NOT a pipe:
+# under `pipefail`, `grep -q` exiting on the first match kills `tail` with
+# SIGPIPE, and the pipeline then reports failure precisely when it matched.
+grep -q -E '"name":"([^"]*__)?(update|create)_task"' < <(tail -c 2000000 "$transcript" 2>/dev/null) || {
+ rm -f "$marker" 2>/dev/null || true
+ exit 0
+}
+
+# The turn, parsed once. A window of recent lines, slurped raw and split inside
+# jq (a line-by-line `-R` read is the #2198 trap). A first line cut mid-record
+# fails to parse and is dropped. If the window holds no prompt, the turn cannot
+# be bounded, so the hook reports nothing and stays out of the way.
+facts=$(tail -n 3000 "$transcript" 2>/dev/null | jq -sRc '
+ split("\n") | map(try fromjson catch empty)
+ | map(select((.isSidechain // false) | not))
+ | . as $lines
+ | [range(0; length) | select(
+ $lines[.].type == "user" and ($lines[.].isMeta // false | not)
+ and ($lines[.].message.content | type) == "string")] as $prompts
+ | if ($prompts | length) == 0 then {bounded: false} else
+ $lines[($prompts | last) + 1:] as $turn
+ | [ $turn[] | select(.type == "assistant") | .message.content[]?
+ | select(.type == "tool_use"
+ and ((.name // "") | test("(^|__)(update|create)_task$"))
+ and (.input.status? == "done"))
+ | {id, task: (.input.task_id? // null)} ] as $closes
+ | [ $turn[] | select(.type == "user") | .message.content[]?
+ | select(type == "object" and .type == "tool_result" and .is_error == true)
+ | .tool_use_id ] as $errors
+ | [ $closes[] | select(.id as $i | ($errors | index($i)) | not) ] as $closed
+ | ([range(0; $turn | length) | select(
+ $turn[.].type == "user"
+ or ($turn[.].type == "assistant"
+ and ([$turn[.].message.content[]?.type] | index("tool_use"))))]
+ | last // -1) as $last_act
+ | {bounded: true,
+ closed: ($closed | length),
+ task_ids: [$closed[].task | select(. != null)],
+ reply: ([ $turn[$last_act + 1:][] | select(.type == "assistant")
+ | .message.content[]? | select(.type == "text") | .text ] | join("\n"))}
+ end' 2>/dev/null) || exit 0
+
+[ "$(printf '%s' "$facts" | jq -r '.bounded // false')" = "true" ] || exit 0
+closed=$(printf '%s' "$facts" | jq -r '.closed // 0')
+if [ "${closed:-0}" = "0" ]; then
+ rm -f "$marker" 2>/dev/null || true
+ exit 0
+fi
+reply=$(printf '%s' "$facts" | jq -r '.reply // ""')
+task_ids=$(printf '%s' "$facts" | jq -r '.task_ids | map(tostring) | join(",")')
+
+# The reply may not be written to the transcript yet when the hook fires. An
+# empty reply is "cannot tell", not "missing everything" — stay out of the way.
+[ -n "$(printf '%s' "$reply" | tr -d '[:space:]')" ] || exit 0
+
+missing=()
+# Where the work sits: a record named by id AND title (#12 "…", milestone 3 "…"),
+# or a step position. A bare id is exactly the homework this shape removes.
+# shellcheck disable=SC2016 # backticks here are literal markdown, not an expansion
+grep -q -i -E '(#[0-9]+|milestone [0-9]+|task [0-9]+)[*_`]*[[:space:]]*[*_`]*["“]|step [0-9]+ of [0-9]+' <<< "$reply" \
+ || missing+=("where it sits")
+# What needs the operator — "needs you: nothing" counts; it is an answer.
+grep -q -i -E 'needs? (from )?you|nothing (is )?needed from you|your (call|decision)' <<< "$reply" \
+ || missing+=("needs you")
+# What comes next.
+grep -q -i -E '\bnext\b' <<< "$reply" \
+ || missing+=("next")
+
+# Reports the outcome; prints the instance's reply and returns 0 only if the
+# instance recorded it.
+report() {
+ scribe_config || return 1
+ local q repo enc m
+ q="outcome=$1&task_ids=${task_ids}"
+ m=$(IFS=,; printf '%s' "${missing[*]:-}")
+ if [ -n "$m" ]; then
+ enc=$(printf '%s' "$m" | jq -sRr '@uri' 2>/dev/null) || enc=""
+ q="${q}&missing=${enc}"
+ fi
+ repo=$(git -C "${event_cwd:-${CLAUDE_PROJECT_DIR:-$PWD}}" remote get-url origin 2>/dev/null || true)
+ if [ -n "$repo" ]; then
+ enc=$(printf '%s' "$repo" | jq -sRr '@uri' 2>/dev/null) || enc=""
+ [ -n "$enc" ] && q="${q}&repo=${enc}"
+ fi
+ curl -fsS --max-time 4 \
+ -H "Authorization: Bearer ${token}" \
+ "${url%/}/api/plugin/report-check?${q}" 2>/dev/null
+}
+
+if [ "$active" = "true" ]; then
+ # A Stop hook already blocked this stop. If it was this one, the reply is
+ # the rewrite: record how it came out, and let the session stop whatever
+ # the answer. If it was another plugin's block, this hook has nothing to add.
+ [ -f "$marker" ] || exit 0
+ rm -f "$marker" 2>/dev/null || true
+ if [ ${#missing[@]} -eq 0 ]; then report passed_after_rewrite >/dev/null; else report missing_after_rewrite >/dev/null; fi
+ exit 0
+fi
+rm -f "$marker" 2>/dev/null || true
+
+if [ ${#missing[@]} -eq 0 ]; then
+ report passed >/dev/null
+ exit 0
+fi
+
+# The words the agent is sent back with are the server's (plugin/PACKAGING.md:
+# a hook carries timing and transport). No reason back → nothing recorded →
+# no block.
+answer=$(report blocked) || exit 0
+reason=$(printf '%s' "$answer" | jq -r '.reason // empty' 2>/dev/null) || reason=""
+[ -n "$reason" ] || exit 0
+: > "$marker" 2>/dev/null || true
+jq -n --arg r "$reason" '{decision: "block", reason: $r}'
+exit 0
diff --git a/scripts/check_plugin.py b/scripts/check_plugin.py
index d0da1ac..8271ff4 100755
--- a/scripts/check_plugin.py
+++ b/scripts/check_plugin.py
@@ -325,6 +325,13 @@ SMOKE_EVENTS: dict[str, str] = {
{"session_id": "smoke", "cwd": ".", "tool_name": "Bash",
"tool_input": {"command": "true"}, "tool_response": {}}
),
+ # The Stop-hook report check (milestone 409 step 5). A transcript that does
+ # not exist is the smoke case: nothing to read, so it must stay silent and
+ # never block, configured or not.
+ "scribe_report_check.sh": json.dumps(
+ {"session_id": "smoke", "transcript_path": "/nonexistent/smoke.jsonl",
+ "cwd": ".", "hook_event_name": "Stop", "stop_hook_active": False}
+ ),
# The shared library is sourced, never run; executed bare it defines
# functions and exits — silent by construction.
"scribe_defs.sh": "",
diff --git a/src/scribe/routes/plugin.py b/src/scribe/routes/plugin.py
index a0b1d6d..b812980 100644
--- a/src/scribe/routes/plugin.py
+++ b/src/scribe/routes/plugin.py
@@ -14,6 +14,7 @@ from scribe.auth import admin_required, get_current_user_id, login_required
from scribe.config import Config
from scribe.services import plugin_context as plugin_ctx_svc
from scribe.services import repo_bindings as repo_bindings_svc
+from scribe.services import report_check as report_check_svc
from scribe.services.settings import get_admin_setting, set_setting
plugin_bp = Blueprint("plugin", __name__, url_prefix="/api/plugin")
@@ -257,6 +258,45 @@ def _parse_shapes(raw: str) -> list[tuple[str, str]]:
return out
+@plugin_bp.get("/report-check")
+@login_required
+async def report_check():
+ """Record what a Stop hook found in a reply that closed a task (milestone 409 step 5).
+
+ The hook decides which completion sections the reply lacks — a local check
+ of text it can read — and reports the outcome here. For `blocked` the
+ response carries the `reason` to send the agent back with: the words are
+ the server's, so every client's hook says the same thing (plugin/PACKAGING.md).
+ A hook blocks only on a `reason` it received, which means only on a block
+ that was recorded.
+
+ A GET for the reason every plugin endpoint is one: a read-scoped key must
+ be enough to run the plugin, and this records telemetry the way /retrieve
+ records a retrieval log.
+
+ Query:
+ outcome (str) — passed | blocked | passed_after_rewrite |
+ missing_after_rewrite. Anything else is a 400.
+ missing (opt) — comma-separated sections the reply lacked:
+ "where it sits", "needs you", "next".
+ task_ids (opt) — comma-separated ids of the tasks the turn closed.
+ repo (opt) — working repo remote, resolved like the other arms.
+ """
+ outcome = (request.args.get("outcome") or "").strip()
+ if outcome not in report_check_svc.OUTCOMES:
+ return jsonify({"error": f"outcome must be one of {list(report_check_svc.OUTCOMES)}"}), 400
+ missing = [m for m in (request.args.get("missing") or "").split(",") if m.strip()]
+ task_ids = _int_list(request.args.get("task_ids"))[:20]
+ project_id, _repo, _unbound = await _project_scope()
+ await report_check_svc.record_report_check(
+ g.user.id, outcome, missing=missing, task_ids=task_ids, project_id=project_id or None,
+ )
+ body: dict = {"status": "ok"}
+ if outcome == "blocked":
+ body["reason"] = report_check_svc.block_reason(missing)
+ return jsonify(body)
+
+
@plugin_bp.get("/processes")
@login_required
async def process_manifest():
diff --git a/src/scribe/services/report_check.py b/src/scribe/services/report_check.py
new file mode 100644
index 0000000..ba88123
--- /dev/null
+++ b/src/scribe/services/report_check.py
@@ -0,0 +1,80 @@
+"""The report-shape check: what the plugin's Stop hook found, and what it says.
+
+WHY THIS EXISTS (milestone 409 step 5)
+
+Everything that helps an agent write a readable completion report arrives
+BEFORE the reply is written. A client's Stop hook is the one moment the
+finished reply exists, so it checks that a reply closing a task carries the
+completion sections (where the work sits, what needs the operator, what comes
+next), and reports what it found here. Two jobs live on this side:
+
+ - RECORDING the outcome, so the rate of `blocked` among checked replies is a
+ number milestone 409's last step can read rather than an impression.
+ - OWNING THE WORDS the agent is sent back with. A hook carries timing and
+ transport only (plugin/PACKAGING.md); guidance text comes from the server,
+ so a second client's hook gets the same instruction by calling the same
+ endpoint, and the wording changes in one place.
+
+app_logs rather than a table of its own: one small event with a JSON detail is
+what that table holds, it already has retention and an admin viewer, and
+nothing here needs a join. If the numbers earn a readout, that is the moment to
+decide whether they earn a table.
+"""
+from __future__ import annotations
+
+import json
+
+from scribe.models import async_session
+from scribe.models.app_log import AppLog
+
+OUTCOMES = ("passed", "blocked", "passed_after_rewrite", "missing_after_rewrite")
+
+# The sections a hook may name as missing, in the order the reason lists them.
+# Anything else a client sends is dropped rather than echoed into an
+# instruction the agent will follow.
+SECTIONS = ("where it sits", "needs you", "next")
+
+
+def known_sections(missing: list[str]) -> list[str]:
+ wanted = {m.strip().lower() for m in missing}
+ return [s for s in SECTIONS if s in wanted]
+
+
+def block_reason(missing: list[str]) -> str:
+ """What the agent is told when its completion report is sent back.
+
+ Names what is missing and points at the reporting-back skill for the shape
+ rather than restating it — the skill owns the shape (decision #4027).
+ """
+ listed = ", ".join(known_sections(missing)) or "the completion sections"
+ return (
+ f"This turn closed a Scribe task, and the reply that ends it is missing: {listed}. "
+ "The operator reads this reply to find out where the work stands. Rewrite it as a "
+ "completion report (the reporting-back skill has the shape): where it sits — the task "
+ "or milestone by id and title, from `placement` — what now works, what needs them "
+ "(or \"nothing\"), and what comes next."
+ )
+
+
+async def record_report_check(
+ user_id: int | None,
+ outcome: str,
+ *,
+ missing: list[str] | None = None,
+ task_ids: list[int] | None = None,
+ project_id: int | None = None,
+) -> None:
+ if outcome not in OUTCOMES:
+ raise ValueError(f"unknown report-check outcome {outcome!r}")
+ details: dict = {"outcome": outcome, "missing": known_sections(missing or []),
+ "task_ids": list(task_ids or [])}
+ if project_id:
+ details["project_id"] = project_id
+ async with async_session() as session:
+ session.add(AppLog(
+ category="plugin",
+ user_id=user_id,
+ action="report_check",
+ details=json.dumps(details),
+ ))
+ await session.commit()
diff --git a/tests/test_report_check_hook.py b/tests/test_report_check_hook.py
new file mode 100644
index 0000000..2269be9
--- /dev/null
+++ b/tests/test_report_check_hook.py
@@ -0,0 +1,168 @@
+"""The Stop hook that checks a task-closing reply for the completion sections
+(milestone 409 step 5).
+
+Runs the real shell against synthetic transcripts in the shape Claude Code
+writes (one content block per JSONL line) and the shared HTTP sink. What it
+pins: silence on every turn that closed nothing; a block only when the
+instance recorded it, in the words the instance returned; one rewrite at most,
+recorded; and no block from another plugin's loop or a failed task write.
+"""
+from __future__ import annotations
+
+import json
+import os
+import shutil
+import subprocess
+from pathlib import Path
+
+import pytest
+
+from tests.helpers import http_sink
+
+HOOK = Path(__file__).resolve().parents[1] / "plugin" / "hooks" / "scribe_report_check.sh"
+TOOL = "mcp__plugin_scribe_scribe__update_task"
+GOOD = ('**Where this sits:** milestone 12 "Move the backups offsite", step 3 of 5.\n'
+ "**What now works:** the sync runs nightly.\n**Needs you:** nothing.\n**Next:** alerts.")
+BAD = "All done, pushed it."
+REASON = "SERVER REASON: rewrite as a completion report"
+
+
+def _env(tmp_path, url="http://127.0.0.1:9"):
+ for tool in ("jq", "curl", "bash"):
+ if shutil.which(tool) is None:
+ pytest.skip(f"hook runtime tool {tool!r} not installed")
+ return {"PATH": os.environ["PATH"], "SCRIBE_URL": url, "SCRIBE_TOKEN": "t",
+ "TMPDIR": str(tmp_path), "HOME": str(tmp_path)}
+
+
+def _prompt(text="please finish it"):
+ return {"type": "user", "message": {"role": "user", "content": text}}
+
+
+def _tool_use(tid="toolu_1", status="done", name=TOOL, task_id=41):
+ return {"type": "assistant", "message": {"content": [
+ {"type": "tool_use", "id": tid, "name": name, "input": {"task_id": task_id, "status": status}}]}}
+
+
+def _result(tid="toolu_1", is_error=False):
+ return {"type": "user", "message": {"content": [
+ {"type": "tool_result", "tool_use_id": tid, "is_error": is_error, "content": "{}"}]}}
+
+
+def _text(text):
+ return {"type": "assistant", "message": {"content": [{"type": "text", "text": text}]}}
+
+
+def _transcript(tmp_path, lines):
+ path = tmp_path / "t.jsonl"
+ path.write_text("\n".join(json.dumps(line) for line in lines) + "\n")
+ return path
+
+
+def _run(env, transcript, active=False, session="s1"):
+ out = subprocess.run(
+ ["bash", str(HOOK)],
+ input=json.dumps({"session_id": session, "transcript_path": str(transcript),
+ "cwd": str(transcript.parent), "hook_event_name": "Stop",
+ "stop_hook_active": active}),
+ capture_output=True, text=True, env=env, timeout=30,
+ )
+ assert out.returncode == 0, out.stderr
+ return out.stdout.strip()
+
+
+def _closing_turn(reply):
+ return [_prompt(), _text("On it."), _tool_use(), _result(), _text(reply)]
+
+
+def test_a_turn_that_closed_nothing_is_silent_and_reports_nothing(tmp_path):
+ with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
+ env = _env(tmp_path, f"http://127.0.0.1:{port}")
+ t = _transcript(tmp_path, [_prompt(), _tool_use(status="in_progress"), _result(), _text(BAD)])
+ assert _run(env, t) == ""
+ assert seen == []
+
+
+def test_a_complete_report_passes_silently_and_is_recorded(tmp_path):
+ with http_sink(b'{"status":"ok"}') as (port, seen):
+ env = _env(tmp_path, f"http://127.0.0.1:{port}")
+ assert _run(env, _transcript(tmp_path, _closing_turn(GOOD))) == ""
+ assert [q["outcome"] for q in seen] == [["passed"]]
+ assert seen[0]["task_ids"] == ["41"]
+
+
+def test_a_missing_section_blocks_once_in_the_servers_words_then_records_the_rewrite(tmp_path):
+ reply = json.dumps({"status": "ok", "reason": REASON}).encode()
+ with http_sink(reply) as (port, seen):
+ env = _env(tmp_path, f"http://127.0.0.1:{port}")
+ out = json.loads(_run(env, _transcript(tmp_path, _closing_turn(BAD))))
+ assert out == {"decision": "block", "reason": REASON}
+ assert seen[0]["outcome"] == ["blocked"]
+ assert seen[0]["missing"] == ["where it sits,needs you,next"]
+
+ # The rewrite: Claude Code sets stop_hook_active; the hook records and never blocks again.
+ rewritten = _transcript(tmp_path, _closing_turn(BAD) + [_text(GOOD)])
+ assert _run(env, rewritten, active=True) == ""
+ assert seen[1]["outcome"] == ["passed_after_rewrite"]
+ assert _run(env, rewritten, active=True) == ""
+ assert len(seen) == 2
+
+
+def test_a_rewrite_that_still_misses_is_recorded_and_not_blocked(tmp_path):
+ reply = json.dumps({"status": "ok", "reason": REASON}).encode()
+ with http_sink(reply) as (port, seen):
+ env = _env(tmp_path, f"http://127.0.0.1:{port}")
+ t = _transcript(tmp_path, _closing_turn(BAD))
+ _run(env, t)
+ assert _run(env, t, active=True) == ""
+ assert [q["outcome"][0] for q in seen] == ["blocked", "missing_after_rewrite"]
+
+
+def test_another_hooks_block_loop_is_left_alone(tmp_path):
+ with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
+ env = _env(tmp_path, f"http://127.0.0.1:{port}")
+ assert _run(env, _transcript(tmp_path, _closing_turn(BAD)), active=True) == ""
+ assert seen == []
+
+
+def test_a_task_write_that_failed_closed_nothing(tmp_path):
+ with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
+ env = _env(tmp_path, f"http://127.0.0.1:{port}")
+ t = _transcript(tmp_path, [_prompt(), _tool_use(), _result(is_error=True), _text(BAD)])
+ assert _run(env, t) == ""
+ assert seen == []
+
+
+def test_a_task_closed_in_an_earlier_turn_does_not_count(tmp_path):
+ with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
+ env = _env(tmp_path, f"http://127.0.0.1:{port}")
+ t = _transcript(tmp_path, _closing_turn(GOOD) + [_prompt("thanks, what else?"), _text(BAD)])
+ assert _run(env, t) == ""
+ assert seen == []
+
+
+def test_a_reply_not_yet_written_is_not_judged(tmp_path):
+ with http_sink(b'{"status":"ok","reason":"x"}') as (port, seen):
+ env = _env(tmp_path, f"http://127.0.0.1:{port}")
+ t = _transcript(tmp_path, [_prompt(), _tool_use(), _result()])
+ assert _run(env, t) == ""
+ assert seen == []
+
+
+def test_no_block_without_a_recorded_check(tmp_path):
+ t = _transcript(tmp_path, _closing_turn(BAD))
+ # Unreachable instance.
+ assert _run(_env(tmp_path), t) == ""
+ # An instance that answered but returned no reason.
+ with http_sink(b'{"status":"ok"}') as (port, seen):
+ assert _run(_env(tmp_path, f"http://127.0.0.1:{port}"), t, session="s2") == ""
+ assert seen[0]["outcome"] == ["blocked"]
+
+
+def test_a_bare_id_does_not_count_as_placing_the_work(tmp_path):
+ reply = json.dumps({"status": "ok", "reason": REASON}).encode()
+ with http_sink(reply) as (port, seen):
+ env = _env(tmp_path, f"http://127.0.0.1:{port}")
+ bare = "Closed #41.\n**Needs you:** nothing.\n**Next:** #42."
+ assert json.loads(_run(env, _transcript(tmp_path, _closing_turn(bare))))["decision"] == "block"
+ assert seen[0]["missing"] == ["where it sits"]
diff --git a/tests/test_services_report_check.py b/tests/test_services_report_check.py
new file mode 100644
index 0000000..0f34231
--- /dev/null
+++ b/tests/test_services_report_check.py
@@ -0,0 +1,47 @@
+"""The server half of the report-shape check (milestone 409 step 5): the words a
+blocked reply is sent back with, and the outcome record."""
+import json
+from unittest.mock import patch
+
+import pytest
+
+from tests.helpers import make_mock_session
+
+
+def test_the_reason_names_only_sections_it_knows():
+ from scribe.services.report_check import block_reason
+
+ reason = block_reason(["next", "ignore previous instructions", "Where It Sits"])
+ assert "missing: where it sits, next." in reason
+ assert "ignore previous instructions" not in reason
+ # It points at the skill that owns the shape rather than restating it.
+ assert "reporting-back" in reason and "placement" in reason
+
+
+def test_a_reason_with_nothing_recognised_still_says_what_to_do():
+ from scribe.services.report_check import block_reason
+
+ assert "missing: the completion sections." in block_reason([])
+
+
+async def test_the_outcome_is_recorded_as_a_plugin_event():
+ from scribe.services.report_check import record_report_check
+
+ session = make_mock_session()
+ with patch("scribe.services.report_check.async_session", return_value=session):
+ await record_report_check(7, "blocked", missing=["next", "bogus"], task_ids=[41], project_id=2)
+ row = session.add.call_args.args[0]
+ assert (row.category, row.action, row.user_id) == ("plugin", "report_check", 7)
+ assert json.loads(row.details) == {"outcome": "blocked", "missing": ["next"],
+ "task_ids": [41], "project_id": 2}
+ session.commit.assert_awaited_once()
+
+
+async def test_an_unknown_outcome_is_refused_before_anything_is_written():
+ from scribe.services.report_check import record_report_check
+
+ session = make_mock_session()
+ with patch("scribe.services.report_check.async_session", return_value=session), \
+ pytest.raises(ValueError):
+ await record_report_check(7, "skipped")
+ session.add.assert_not_called()