"""The end-of-turn shape check: what a turn wrote that nobody has judged, and what the agent is asked. WHY THIS EXISTS (milestone 439) The shape ledger used to be filled by machinery: extraction found the definitions, framework rules decided which were one-offs, and the write-path hook stamped "instance of #N" with nobody reading. The agent that WROTE the code — the one participant who knows what it is — was asked only in an audit, weeks later, if anyone ran one. A first instance of a reusable piece was never asked about at all, so it was never recorded, and the next session rebuilt it. So the question moves to the moment the knowledge exists. The client's write hooks keep a ledger of the definitions each turn wrote; its Stop hook sends them here at the end of the turn. Two jobs live on this side, as with the report check (services/report_check.py): - DECIDING what is unjudged, against the ledger and the project's recorded snippets — something only the server can see. - OWNING THE WORDS. A hook carries timing and transport (plugin/PACKAGING.md); the instruction comes from here, so every client asks the same question and the wording changes in one place. And it RECORDS every outcome, so "how much did agents leave unjudged after being asked" is a number rather than an impression. app_logs, for the reason report_check gives. """ from __future__ import annotations import json from datetime import datetime, timedelta, timezone from sqlalchemy import select from scribe.models import async_session from scribe.models.app_log import AppLog # check — the turn's first stop: block if anything is unjudged. # after — the stop that follows a block: never block again, only record # whether the agent answered. PHASES = ("check", "after") OUTCOMES = ("passed", "blocked", "judged_after_block", "left_after_block") # How many shapes the reason names before "+N more". Enough to judge a normal # turn in one call; a turn that wrote more than this generated something. _LISTED = 25 def _evidence_note(evidence: dict) -> str: """One parenthesis of what the machinery holds — offered, never asserted.""" parts = [] looks = evidence.get("looks_like") or {} if looks.get("snippet_id"): score = looks.get("score") at = f" {score:.2f}" if isinstance(score, (int, float)) else "" parts.append(f"looks like #{looks['snippet_id']}{at}") stamped = evidence.get("stamped") or {} if stamped.get("snippet_id"): parts.append(f"the hook guessed #{stamped['snippet_id']}") if evidence.get("copies"): parts.append("other copies of it exist") if evidence.get("diverges_from"): parts.append(f"#{evidence['diverges_from']} is canon in this directory") return f" ({'; '.join(parts)})" if parts else "" def block_reason(unjudged: list[dict], *, project_id: int, repo: str) -> str: """What the agent is told at the end of a turn that wrote unjudged shapes. Phrased as the practice wanted (note #3565): what to decide and the one call that records it, not a prohibition. The reusing-code skill owns the longer version of what each verdict means. """ lines = [] for item in unjudged[:_LISTED]: new = " — new" if item.get("status") == "new" else "" lines.append( f"- {item['path']} · {item['symbol']} ({item['kind']}){new}" f"{_evidence_note(item.get('evidence') or {})}" ) more = len(unjudged) - _LISTED if more > 0: lines.append(f"- … and {more} more (list_shapes(project_id={project_id}, path=…) shows them)") repo_arg = f', repo="{repo}"' if repo else "" return ( f"This turn wrote {len(unjudged)} definition{'s' if len(unjudged) != 1 else ''} " "that nobody has judged yet. You built them, so you know what each one is — " "record it now, while that is still true, so the next session starts from your " "answer instead of rebuilding it:\n" + "\n".join(lines) + "\n\nFor each: something another part of the code should reuse → record it " "with create_snippet (name, code, when to reach for it, location); built from a " "recorded snippet → `instance` of it; a deliberate departure from one → " "`variant` with the why; a genuine one-off → `exempt` with a reason " "(reason_code one-off-handler, test-helper, scoped-css, pure-helper … indexes it). " "One call records them all: " f"classify_shapes(project_id={project_id}{repo_arg}, classifications=[{{path, " "symbol, kind, status, snippet_id?, reason?, reason_code?}}, …]). " "`repo` lets a verdict land on a shape the ledger has not synced yet." ) async def record_shape_check( user_id: int | None, outcome: str, *, written: int, unjudged: int, project_id: int | None = None, ) -> None: if outcome not in OUTCOMES: raise ValueError(f"unknown shape-check outcome {outcome!r}") details: dict = {"outcome": outcome, "written": written, "unjudged": unjudged} if project_id: details["project_id"] = project_id async with async_session() as session: session.add(AppLog( category="plugin", user_id=user_id, action="shape_check", details=json.dumps(details), )) await session.commit() def outcome_for(phase: str, unjudged: list[dict]) -> str: """The recorded outcome: a first stop blocks or passes; the stop after a block records whether the agent answered, and never blocks.""" if phase == "after": return "left_after_block" if unjudged else "judged_after_block" return "blocked" if unjudged else "passed" def parse_written(raw: str, *, cap: int = 200) -> list[tuple[str, str, str]]: """The hook's ledger lines, `pathkindname`, as triples. Kinds outside the ledger's vocabulary and malformed lines are dropped rather than echoed into an instruction; duplicates collapse; capped.""" out: list[tuple[str, str, str]] = [] for line in (raw or "").splitlines(): parts = line.split("\t") if len(parts) != 3: continue path, kind, name = (p.strip() for p in parts) if kind not in ("css", "sym", "file") or not path or not name: continue if (path, kind, name) not in out: out.append((path, kind, name)) if len(out) >= cap: break return out # The window the coverage line reports the write-time figure over. A week is # long enough to hold a few working sessions on a project and short enough # that a reflex that started slipping shows up while it is still news. WINDOW_DAYS = 7 def summarise(outcomes: list[str]) -> dict: """{checked, asked, left} from a list of recorded outcomes. `checked` is turns the question was put to (a first stop that wrote something): passed + blocked. `asked` is the blocked ones. `left` is asks the agent walked past — the stop after a block that still had unjudged shapes. That last number is the slip the milestone exists to make visible.""" checked = sum(1 for o in outcomes if o in ("passed", "blocked")) asked = sum(1 for o in outcomes if o == "blocked") left = sum(1 for o in outcomes if o == "left_after_block") return {"checked": checked, "asked": asked, "left": left} async def window_summary(project_id: int, *, days: int = WINDOW_DAYS) -> dict: """The write-time figure for one project over the last ``days``.""" since = datetime.now(timezone.utc) - timedelta(days=days) async with async_session() as session: rows = (await session.execute( select(AppLog.details).where( AppLog.category == "plugin", AppLog.action == "shape_check", AppLog.created_at >= since, ) )).scalars().all() outcomes: list[str] = [] for raw in rows: try: details = json.loads(raw or "{}") except (TypeError, ValueError): continue if details.get("project_id") == project_id: outcomes.append(details.get("outcome") or "") return {**summarise(outcomes), "days": days}