Files
FabledScribe/src/scribe/services/shape_check.py
T
bvandeusenandClaude Opus 5.5 582a5a4f48
CI & Build / Python lint (push) Successful in 2s
CI & Build / Plugin hooks (push) Successful in 16s
CI & Build / TypeScript typecheck (push) Successful in 54s
CI & Build / integration (push) Successful in 47s
CI & Build / Python tests (push) Successful in 1m39s
CI & Build / Build & push image (push) Successful in 29s
feat(shapes): the practice is written where it is read, and the coverage line measures the slip (milestone 439 step 6)
- reusing-code: "Before the turn ends — say what you built" — the four
  verdicts, the one classify_shapes(repo=…) call, and the component file as a
  shape. Description names the end-of-turn moment.
- shape-accounting: the writer judges; audits are the check that it held.
  The write path SUGGESTS (no more hook instances); component file rows and
  whole-file canon described; scoped covers Svelte too.
- _INSTRUCTIONS reuse line: "before the turn ends, say what you built
  (create_snippet the reusable, classify_shapes the rest)" — 1570/1600.
- Coverage line: "written-shape check (7d): N turns checked, M asked, K left
  unjudged", from the Stop hook's recorded outcomes; silent until the
  question has been put.
- test_guidance_ownership pins the new topic on reusing-code.
- prior-art hook header no longer says it stamps instance rows.

Plugin version minted.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-01 08:46:58 -04:00

193 lines
8.1 KiB
Python

"""The end-of-turn shape check: what a turn wrote that nobody has judged, and what the agent is asked.
WHY THIS EXISTS (milestone 439)
The shape ledger used to be filled by machinery: extraction found the
definitions, framework rules decided which were one-offs, and the write-path
hook stamped "instance of #N" with nobody reading. The agent that WROTE the
code — the one participant who knows what it is — was asked only in an audit,
weeks later, if anyone ran one. A first instance of a reusable piece was never
asked about at all, so it was never recorded, and the next session rebuilt it.
So the question moves to the moment the knowledge exists. The client's write
hooks keep a ledger of the definitions each turn wrote; its Stop hook sends
them here at the end of the turn. Two jobs live on this side, as with the
report check (services/report_check.py):
- DECIDING what is unjudged, against the ledger and the project's recorded
snippets — something only the server can see.
- OWNING THE WORDS. A hook carries timing and transport (plugin/PACKAGING.md);
the instruction comes from here, so every client asks the same question
and the wording changes in one place.
And it RECORDS every outcome, so "how much did agents leave unjudged after
being asked" is a number rather than an impression. app_logs, for the reason
report_check gives.
"""
from __future__ import annotations
import json
from datetime import datetime, timedelta, timezone
from sqlalchemy import select
from scribe.models import async_session
from scribe.models.app_log import AppLog
# check — the turn's first stop: block if anything is unjudged.
# after — the stop that follows a block: never block again, only record
# whether the agent answered.
PHASES = ("check", "after")
OUTCOMES = ("passed", "blocked", "judged_after_block", "left_after_block")
# How many shapes the reason names before "+N more". Enough to judge a normal
# turn in one call; a turn that wrote more than this generated something.
_LISTED = 25
def _evidence_note(evidence: dict) -> str:
"""One parenthesis of what the machinery holds — offered, never asserted."""
parts = []
looks = evidence.get("looks_like") or {}
if looks.get("snippet_id"):
score = looks.get("score")
at = f" {score:.2f}" if isinstance(score, (int, float)) else ""
parts.append(f"looks like #{looks['snippet_id']}{at}")
stamped = evidence.get("stamped") or {}
if stamped.get("snippet_id"):
parts.append(f"the hook guessed #{stamped['snippet_id']}")
if evidence.get("copies"):
parts.append("other copies of it exist")
if evidence.get("diverges_from"):
parts.append(f"#{evidence['diverges_from']} is canon in this directory")
return f" ({'; '.join(parts)})" if parts else ""
def block_reason(unjudged: list[dict], *, project_id: int, repo: str) -> str:
"""What the agent is told at the end of a turn that wrote unjudged shapes.
Phrased as the practice wanted (note #3565): what to decide and the one
call that records it, not a prohibition. The reusing-code skill owns the
longer version of what each verdict means.
"""
lines = []
for item in unjudged[:_LISTED]:
new = " — new" if item.get("status") == "new" else ""
lines.append(
f"- {item['path']} · {item['symbol']} ({item['kind']}){new}"
f"{_evidence_note(item.get('evidence') or {})}"
)
more = len(unjudged) - _LISTED
if more > 0:
lines.append(f"- … and {more} more (list_shapes(project_id={project_id}, path=…) shows them)")
repo_arg = f', repo="{repo}"' if repo else ""
return (
f"This turn wrote {len(unjudged)} definition{'s' if len(unjudged) != 1 else ''} "
"that nobody has judged yet. You built them, so you know what each one is — "
"record it now, while that is still true, so the next session starts from your "
"answer instead of rebuilding it:\n"
+ "\n".join(lines)
+ "\n\nFor each: something another part of the code should reuse → record it "
"with create_snippet (name, code, when to reach for it, location); built from a "
"recorded snippet → `instance` of it; a deliberate departure from one → "
"`variant` with the why; a genuine one-off → `exempt` with a reason "
"(reason_code one-off-handler, test-helper, scoped-css, pure-helper … indexes it). "
"One call records them all: "
f"classify_shapes(project_id={project_id}{repo_arg}, classifications=[{{path, "
"symbol, kind, status, snippet_id?, reason?, reason_code?}}, …]). "
"`repo` lets a verdict land on a shape the ledger has not synced yet."
)
async def record_shape_check(
user_id: int | None,
outcome: str,
*,
written: int,
unjudged: int,
project_id: int | None = None,
) -> None:
if outcome not in OUTCOMES:
raise ValueError(f"unknown shape-check outcome {outcome!r}")
details: dict = {"outcome": outcome, "written": written, "unjudged": unjudged}
if project_id:
details["project_id"] = project_id
async with async_session() as session:
session.add(AppLog(
category="plugin",
user_id=user_id,
action="shape_check",
details=json.dumps(details),
))
await session.commit()
def outcome_for(phase: str, unjudged: list[dict]) -> str:
"""The recorded outcome: a first stop blocks or passes; the stop after a
block records whether the agent answered, and never blocks."""
if phase == "after":
return "left_after_block" if unjudged else "judged_after_block"
return "blocked" if unjudged else "passed"
def parse_written(raw: str, *, cap: int = 200) -> list[tuple[str, str, str]]:
"""The hook's ledger lines, `path<TAB>kind<TAB>name`, as triples.
Kinds outside the ledger's vocabulary and malformed lines are dropped
rather than echoed into an instruction; duplicates collapse; capped."""
out: list[tuple[str, str, str]] = []
for line in (raw or "").splitlines():
parts = line.split("\t")
if len(parts) != 3:
continue
path, kind, name = (p.strip() for p in parts)
if kind not in ("css", "sym", "file") or not path or not name:
continue
if (path, kind, name) not in out:
out.append((path, kind, name))
if len(out) >= cap:
break
return out
# The window the coverage line reports the write-time figure over. A week is
# long enough to hold a few working sessions on a project and short enough
# that a reflex that started slipping shows up while it is still news.
WINDOW_DAYS = 7
def summarise(outcomes: list[str]) -> dict:
"""{checked, asked, left} from a list of recorded outcomes.
`checked` is turns the question was put to (a first stop that wrote
something): passed + blocked. `asked` is the blocked ones. `left` is
asks the agent walked past — the stop after a block that still had
unjudged shapes. That last number is the slip the milestone exists to
make visible."""
checked = sum(1 for o in outcomes if o in ("passed", "blocked"))
asked = sum(1 for o in outcomes if o == "blocked")
left = sum(1 for o in outcomes if o == "left_after_block")
return {"checked": checked, "asked": asked, "left": left}
async def window_summary(project_id: int, *, days: int = WINDOW_DAYS) -> dict:
"""The write-time figure for one project over the last ``days``."""
since = datetime.now(timezone.utc) - timedelta(days=days)
async with async_session() as session:
rows = (await session.execute(
select(AppLog.details).where(
AppLog.category == "plugin",
AppLog.action == "shape_check",
AppLog.created_at >= since,
)
)).scalars().all()
outcomes: list[str] = []
for raw in rows:
try:
details = json.loads(raw or "{}")
except (TypeError, ValueError):
continue
if details.get("project_id") == project_id:
outcomes.append(details.get("outcome") or "")
return {**summarise(outcomes), "days": days}