Merge pull request 'Ledger follow-ups from the shape audit (milestone 294: #2868–#2874)' (#122) from dev into main
CI & Build / Build & push image (push) Successful in 15s
CI & Build / TypeScript typecheck (push) Successful in 23s
CI & Build / Plugin hooks (push) Successful in 8s
CI & Build / Python lint (push) Successful in 3s
CI & Build / Python tests (push) Successful in 57s
CI & Build / integration (push) Successful in 21s

This commit was merged in pull request #122.
This commit is contained in:
2026-08-21 15:24:03 -04:00
22 changed files with 1202 additions and 75 deletions
+26
View File
@@ -0,0 +1,26 @@
"""Per-binding ref — the branch a project's ledger follows (#2873, milestone 294)
Revision ID: 0082
Revises: 0081
Create Date: 2026-08-21
A repo binding used to imply the repo's default branch; the shape ledger
therefore only saw work after a merge to main, while the operator's work
lands on dev (rule 1). `ref` names the branch the coverage refresh reads —
NULL keeps today's behaviour (the forge's default branch).
"""
import sqlalchemy as sa
from alembic import op
revision = "0082"
down_revision = "0081"
branch_labels = None
depends_on = None
def upgrade() -> None:
op.add_column("repo_bindings", sa.Column("ref", sa.Text(), nullable=True))
def downgrade() -> None:
op.drop_column("repo_bindings", "ref")
@@ -0,0 +1,26 @@
"""Exempt/variant reason codes — a small fixed catalogue beside the prose (#2874, milestone 294)
Revision ID: 0083
Revises: 0082
Create Date: 2026-08-21
The 2026-08 audit wrote the same free-text reason thousands of times
("scoped rule — styles one element of this view"); a judgment's WHY stays
prose, but an optional code from a fixed catalogue makes the ledger
filterable and aggregable ("how many pure helpers, how many test helpers").
"""
import sqlalchemy as sa
from alembic import op
revision = "0083"
down_revision = "0082"
branch_labels = None
depends_on = None
def upgrade() -> None:
op.add_column("code_shapes", sa.Column("reason_code", sa.Text(), nullable=True))
def downgrade() -> None:
op.drop_column("code_shapes", "reason_code")
+40
View File
@@ -0,0 +1,40 @@
"""code_shape_uses — consumption edges, separate from conformance (#2870, milestone 294)
Revision ID: 0084
Revises: 0083
Create Date: 2026-08-21
A ledger row carries ONE snippet_id: what shape this is (instance/variant of
a canon). But a shape can also CALL several canonical helpers — a service
function that is an instance of the service-function convention and a
consumer of hash_token. The 2026-08 audit had to pick one; hook evidence
("pulled #N then wrote code referencing it") was stamped as instance when it
is a uses fact. This table holds the many-valued relation: shape → snippet,
with the basis and the evidence. Cascades with the shape and the snippet.
"""
import sqlalchemy as sa
from alembic import op
revision = "0084"
down_revision = "0083"
branch_labels = None
depends_on = None
def upgrade() -> None:
op.create_table(
"code_shape_uses",
sa.Column("id", sa.Integer(), primary_key=True),
sa.Column("shape_id", sa.Integer(), sa.ForeignKey("code_shapes.id", ondelete="CASCADE"), nullable=False),
sa.Column("snippet_id", sa.Integer(), sa.ForeignKey("notes.id", ondelete="CASCADE"), nullable=False),
sa.Column("basis", sa.Text(), nullable=False),
sa.Column("evidence", sa.Text(), nullable=True),
sa.Column("created_at", sa.DateTime(timezone=True), nullable=False, server_default=sa.text("now()")),
sa.UniqueConstraint("shape_id", "snippet_id", name="uq_code_shape_uses_shape_snippet"),
)
op.create_index("ix_code_shape_uses_snippet", "code_shape_uses", ["snippet_id"])
def downgrade() -> None:
op.drop_index("ix_code_shape_uses_snippet", table_name="code_shape_uses")
op.drop_table("code_shape_uses")
+1 -1
View File
@@ -754,7 +754,7 @@ async function confirmDelete() {
</div> </div>
<div v-if="coverage.counts" class="coverage-gaps"> <div v-if="coverage.counts" class="coverage-gaps">
<span <span
v-for="k in ['canonical', 'instance', 'variant', 'exempt']" v-for="k in ['canonical', 'instance', 'variant', 'exempt', 'scoped']"
:key="k" :key="k"
> >
<span v-if="coverage.counts[k]" class="coverage-gap-chip"> <span v-if="coverage.counts[k]" class="coverage-gap-chip">
+1 -1
View File
@@ -1,7 +1,7 @@
{ {
"name": "scribe", "name": "scribe",
"description": "Scribe system-of-record for Claude Code: MCP tools over your notes/tasks/projects/rules, a session-start push channel that surfaces your always-on rules + active-project context, process-skills (writing-plans, systematic-debugging, verification, brainstorming, reusing-code), and your saved Scribe Processes auto-surfaced as skills (/scribe:sync). Replaces superpowers + file-memory with one app-backed plugin.", "description": "Scribe system-of-record for Claude Code: MCP tools over your notes/tasks/projects/rules, a session-start push channel that surfaces your always-on rules + active-project context, process-skills (writing-plans, systematic-debugging, verification, brainstorming, reusing-code), and your saved Scribe Processes auto-surfaced as skills (/scribe:sync). Replaces superpowers + file-memory with one app-backed plugin.",
"version": "0.1.36", "version": "0.1.37",
"author": { "name": "Bryan Van Deusen" }, "author": { "name": "Bryan Van Deusen" },
"mcpServers": { "mcpServers": {
"scribe": { "scribe": {
+5
View File
@@ -17,6 +17,11 @@ row carries a status:
— the why IS the record. — the why IS the record.
- `exempt` — judged genuinely one-off. **Reason required.** A recorded - `exempt` — judged genuinely one-off. **Reason required.** A recorded
judgment, not silence — it stops the next pass re-litigating it. judgment, not silence — it stops the next pass re-litigating it.
- `scoped` — one-off **by construction**, stamped by the coverage sync
(a Vue component's scoped `<style>` rules and its `<script setup>`
functions — unreachable from any other file). Accounted for without a
judgment; still proposed against, grouped and flagged; any judgment you
make overrides it. Not the todo.
- `unclassified` — nobody has judged it yet. **This is the todo list.** - `unclassified` — nobody has judged it yet. **This is the todo list.**
## The loop ## The loop
+18 -3
View File
@@ -13,28 +13,43 @@ from scribe.services import projects as projects_svc
from scribe.services import repo_bindings as repo_bindings_svc from scribe.services import repo_bindings as repo_bindings_svc
async def bind_repo(repo_url: str, project_id: int) -> dict: async def bind_repo(repo_url: str, project_id: int, ref: str = "") -> dict:
"""Bind a git repository to a Scribe project for session-start context. """Bind a git repository to a Scribe project for session-start context.
After this, any session started in that repo auto-loads the project's After this, any session started in that repo auto-loads the project's
context (the SessionStart hook sends the repo's remote; the server resolves context (the SessionStart hook sends the repo's remote; the server resolves
it here). Idempotent — re-binding the same repo updates the target project. it here). Idempotent — re-binding the same repo updates the target project.
The binding is also what the shape ledger reads (refresh_pattern_coverage):
`ref` names the branch it follows. Default (""): the repo's default branch
— which means the ledger only sees work after a merge. A dev-first project
(rule 1: dev is home) should bind with ref="dev" so classification follows
the push, not the merge. Re-binding with ref="" keeps the standing ref;
pass ref="-" to clear it back to the default branch.
Args: Args:
repo_url: the repo's git remote (e.g. the output of repo_url: the repo's git remote (e.g. the output of
`git remote get-url origin` — ssh or https form, both work). `git remote get-url origin` — ssh or https form, both work).
project_id: the Scribe project this repo represents. project_id: the Scribe project this repo represents.
ref: branch the ledger follows ("" = leave as is / default branch on
a new binding; "-" = clear to the default branch).
""" """
uid = current_user_id() uid = current_user_id()
project = await projects_svc.get_project(uid, project_id) project = await projects_svc.get_project(uid, project_id)
if project is None: if project is None:
raise ValueError(f"project {project_id} not found") raise ValueError(f"project {project_id} not found")
binding = await repo_bindings_svc.set_binding(uid, repo_url, project_id) ref_arg = None if not ref else ("" if ref.strip() == "-" else ref)
binding = await repo_bindings_svc.set_binding(uid, repo_url, project_id, ref_arg)
follows = binding.ref or "the default branch"
return { return {
"repo_key": binding.repo_key, "repo_key": binding.repo_key,
"project_id": binding.project_id, "project_id": binding.project_id,
"project_title": project.title, "project_title": project.title,
"message": f"Bound `{binding.repo_key}` -> {project.title} (id {project.id}).", "ref": binding.ref,
"message": (
f"Bound `{binding.repo_key}` -> {project.title} (id {project.id}); "
f"the ledger follows {follows}."
),
} }
+101 -10
View File
@@ -36,10 +36,20 @@ async def classify_shapes(
Args: Args:
project_id: The project whose ledger is being judged. project_id: The project whose ledger is being judged.
classifications: Objects of {path, symbol, status, kind?, snippet_id?, classifications: Objects of {path, symbol, status, kind?, snippet_id?,
reason?}. path+symbol name the shape exactly as list_shapes shows reason?, reason_code?}. path+symbol name the shape exactly as
it; kind ("sym"/"css") narrows when one file defines both. list_shapes shows it; kind ("sym"/"css") narrows when one file
snippet_id is required for canonical/instance/variant; reason is defines both. snippet_id is required for canonical/instance/
required for variant/exempt. variant; reason is required for variant/exempt. reason_code is
an OPTIONAL index beside the prose (one of: scoped-css,
one-off-handler, test-helper, convention-plumbing, pure-helper,
generated, script, typed-record) so the ledger can be filtered
and aggregated by kind of one-off — the prose stays the record.
uses is an OPTIONAL list of snippet ids this shape CALLS (#2870):
conformance (status + snippet_id) says what shape it is, uses
says which canonical helpers it consumes — a service function
can be an instance of the service-function convention AND use
hash_token. Consumer maps are uses edges; list_shapes(uses=N)
and get_snippet's `uses` read them.
via: Who is judging — "agent" (default), "audit" (a sweep), or via: Who is judging — "agent" (default), "audit" (a sweep), or
"import" (carrying maps recorded elsewhere). "import" (carrying maps recorded elsewhere).
@@ -64,6 +74,8 @@ async def list_shapes(
offset: int = 0, offset: int = 0,
proposal: str = "", proposal: str = "",
flag: str = "", flag: str = "",
compact: bool = False,
uses: int = 0,
) -> dict: ) -> dict:
"""Read a project's shape ledger — `status="unclassified"` IS the todo. """Read a project's shape ledger — `status="unclassified"` IS the todo.
@@ -71,12 +83,27 @@ async def list_shapes(
(fed by the coverage refresh). Filters compose: (fed by the coverage refresh). Filters compose:
Args: Args:
status: canonical | instance | variant | exempt | unclassified. status: canonical | instance | variant | exempt | scoped | unclassified.
`scoped` (#2869) is the sync's mechanical stamp on one-offs by
construction (a Vue component's scoped <style> rules and its
<script setup> functions): accounted for, not judged, still
proposed against / grouped / flagged, and overridable by any
classify_shapes judgment. The human todo is `unclassified`.
path: exact file, or a directory — matches everything beneath it path: exact file, or a directory — matches everything beneath it
(the coverage line's "largest" dirs go straight in here). (the coverage line's "largest" dirs go straight in here).
snippet_id: rows classified against this snippet — a consumer map. snippet_id: rows classified against this snippet (instance/variant
of it — conformance).
uses: rows that CALL this snippet (#2870) — the consumer map proper,
whatever shape each row is itself; edges come from judgments
(classify_shapes uses=), the write-path hook, and the proposer's
by-name reference hits.
include_vanished: include shapes no longer in the tree (history). include_vanished: include shapes no longer in the tree (history).
limit/offset: page through big ledgers (limit caps at 500). limit/offset: page through big ledgers (limit caps at 500).
compact: rows as `path · symbol · kind · status · signature` plus
snippet_id / by / proposal / diverges_from / recheck only when
set — no commits, shas or timestamps. THE form for an audit:
a full 500-row page fits the tool budget. The default rows carry
everything (shape_history-grade bookkeeping).
proposal: the proposer's queue (#2792) — "any", "canon" (rows the proposal: the proposer's queue (#2792) — "any", "canon" (rows the
machine thinks are an instance of a snippet: `proposal` carries machine thinks are an instance of a snippet: `proposal` carries
snippet_id, basis, score), "derive" (rows that repeat with NO snippet_id, basis, score), "derive" (rows that repeat with NO
@@ -116,9 +143,73 @@ async def list_shapes(
uid, project_id, uid, project_id,
status=status, path=path, snippet_id=snippet_id, status=status, path=path, snippet_id=snippet_id,
include_vanished=include_vanished, limit=limit, offset=offset, include_vanished=include_vanished, limit=limit, offset=offset,
proposal=proposal, flag=flag, proposal=proposal, flag=flag, uses=uses,
) )
return {"shapes": [r.to_dict() for r in rows], "total": total} return {
"shapes": [r.to_compact() if compact else r.to_dict() for r in rows],
"total": total,
}
async def classify_shapes_by_rule(
project_id: int,
path: str,
status: str,
pattern: str = "",
kind: str = "",
snippet_id: int = 0,
reason: str = "",
via: str = "agent",
include_judged: bool = False,
reason_code: str = "",
uses: list[int] | None = None,
) -> dict:
"""The sweep form of classify_shapes: ONE judgment applied to every
unclassified shape under a directory whose symbol matches a glob.
For the long tail an audit judges by family, not by row — "every scoped
rule under frontend/src/views is exempt: styles one element of its view",
"every `*_scheduler.py` symbol is an instance of ScheduledJob" — where
listing 900 rows and sending them back is the whole cost. The row form
stays the precise tool; reach for it when each row gets its own reason.
Args:
project_id: The project whose ledger is being judged.
path: A file, or a directory and everything beneath it. Required —
a sweep names what it judges.
status: instance | variant | exempt | unclassified (canonical is the
sync's stamp, not a sweep's).
pattern: Shell glob on the symbol (`*_rows`, `_*`, `modal-*`, `*`);
"" = every symbol under path.
kind: "sym" or "css" to narrow; "" = both.
snippet_id: Required for instance/variant — the canon judged against.
reason: Required for variant/exempt — the why, recorded on every row.
via: "agent" (default) | "audit" | "import".
reason_code: Optional catalogue code beside the reason (see
classify_shapes) — a sweep is exactly where one applies.
uses: Optional snippet ids every matched shape CALLS (#2870) — e.g.
"every *_scheduler.py symbol uses ScheduledJob".
include_judged: By default only unjudged rows are touched —
`unclassified` and the sync's mechanical `scoped` stamp — a
sweep never silently overwrites a judgment. True re-judges every
matching live row (use to re-confirm after a recheck, or to
revise a family you judged earlier).
One transaction: applies whole or not at all. Returns
{"classified": N, "sample": ["path::symbol", ...]} (first 12, sorted)
so you can see what the rule reached; N = 0 means the rule matched
nothing live and unclassified — widen the pattern or refresh coverage.
"""
uid = current_user_id()
try:
return await shape_ledger_svc.classify_shapes_where(
uid, project_id, path=path, status=status, pattern=pattern,
kind=kind, snippet_id=snippet_id or None, reason=reason or None,
via=via, include_judged=include_judged,
reason_code=reason_code or None, uses=uses or None,
)
except ValueError as exc:
return {"error": str(exc)}
async def shape_history( async def shape_history(
@@ -217,7 +308,7 @@ async def refresh_pattern_coverage(project_id: int) -> dict:
def register(mcp) -> None: def register(mcp) -> None:
for fn in ( for fn in (
classify_shapes, list_shapes, refresh_pattern_coverage, classify_shapes, classify_shapes_by_rule, list_shapes,
confirm_shape_proposals, shape_history, refresh_pattern_coverage, confirm_shape_proposals, shape_history,
): ):
mcp.tool(name=fn.__name__)(fn) mcp.tool(name=fn.__name__)(fn)
+4
View File
@@ -245,6 +245,10 @@ async def get_snippet(snippet_id: int) -> dict:
data["instances"] = consumers["instances"] data["instances"] = consumers["instances"]
if consumers["variants"]: if consumers["variants"]:
data["variants"] = consumers["variants"] data["variants"] = consumers["variants"]
if consumers.get("uses"):
# The call sites (#2870): shapes that use this snippet, whatever
# shape they are themselves.
data["uses"] = consumers["uses"]
return data return data
+1 -1
View File
@@ -44,6 +44,6 @@ from scribe.models.rulebook import ( # noqa: E402, F401
) )
from scribe.models.repo_binding import RepoBinding # noqa: E402, F401 from scribe.models.repo_binding import RepoBinding # noqa: E402, F401
from scribe.models.forge_connection import ForgeConnection # noqa: E402, F401 from scribe.models.forge_connection import ForgeConnection # noqa: E402, F401
from scribe.models.code_shape import CodeShape, CodeShapeEvent # noqa: E402, F401 from scribe.models.code_shape import CodeShape, CodeShapeEvent, CodeShapeUse # noqa: E402, F401
from scribe.models.system import System, RecordSystem # noqa: E402, F401 from scribe.models.system import System, RecordSystem # noqa: E402, F401
from scribe.models.design_system import DesignSystem, DesignToken # noqa: E402, F401 from scribe.models.design_system import DesignSystem, DesignToken # noqa: E402, F401
+100 -3
View File
@@ -1,4 +1,4 @@
from datetime import datetime from datetime import datetime, timezone
from sqlalchemy import ( from sqlalchemy import (
BigInteger, BigInteger,
@@ -16,10 +16,30 @@ from scribe.models import Base
from scribe.models.base import TimestampMixin, iso from scribe.models.base import TimestampMixin, iso
# The classification vocabulary (note 2786). `unclassified` is the default and # The classification vocabulary (note 2786). `unclassified` is the default and
# THE todo state; every other status is a judgment, stamped with who made it. # THE todo state; every other status is a judgment, stamped with who made it
SHAPE_STATUSES = ("canonical", "instance", "variant", "exempt", "unclassified") # except `scoped` (#2869): the coverage sync's mechanical stamp on shapes that
# are one-offs BY CONSTRUCTION (a Vue component's scoped <style> rules and its
# <script setup> functions — unreachable from any other file). Scoped rows are
# accounted for without a human judging them, so `exempt` keeps meaning "a
# person looked"; the proposer, derive grouping and divergence still see them,
# and any judgment (instance/variant/exempt) overrides the stamp.
SHAPE_STATUSES = ("canonical", "instance", "variant", "exempt", "scoped", "unclassified")
SHAPE_CLASSIFIERS = ("agent", "audit", "hook", "mechanical", "import") SHAPE_CLASSIFIERS = ("agent", "audit", "hook", "mechanical", "import")
# The reason catalogue (#2874): an OPTIONAL code beside the prose reason on
# variant/exempt rows, so the ledger can be filtered and aggregated by kind
# of one-off. The prose remains the record; the code is the index.
REASON_CODES = (
"scoped-css", # a scoped rule styling one element (pre-#2869 rows)
"one-off-handler", # a view/component handler or loader, one per surface
"test-helper", # a test module's stub, driver or fixture data
"convention-plumbing", # registration, wiring, app factory — one of each
"pure-helper", # a sync module-private helper with no session
"generated", # generated source (theme.css, protos, bundles)
"script", # a standalone dev/CI script
"typed-record", # a NamedTuple / dataclass / error class — one each
)
# How the mechanical proposer (#2792) arrived at a proposal, strongest first. # How the mechanical proposer (#2792) arrived at a proposal, strongest first.
# `derive` is the odd one out: not "this is an instance of #N" but "this # `derive` is the odd one out: not "this is an instance of #N" but "this
# shape repeats with NO canon — derive one first" (note 2786's derive-first # shape repeats with NO canon — derive one first" (note 2786's derive-first
@@ -99,6 +119,7 @@ class CodeShape(Base, TimestampMixin):
BigInteger, ForeignKey("notes.id", ondelete="SET NULL"), nullable=True BigInteger, ForeignKey("notes.id", ondelete="SET NULL"), nullable=True
) )
reason: Mapped[str | None] = mapped_column(Text, nullable=True) reason: Mapped[str | None] = mapped_column(Text, nullable=True)
reason_code: Mapped[str | None] = mapped_column(Text, nullable=True) # REASON_CODES (#2874)
classified_by: Mapped[str | None] = mapped_column(Text, nullable=True) classified_by: Mapped[str | None] = mapped_column(Text, nullable=True)
classified_at: Mapped[datetime | None] = mapped_column( classified_at: Mapped[datetime | None] = mapped_column(
DateTime(timezone=True), nullable=True DateTime(timezone=True), nullable=True
@@ -152,6 +173,7 @@ class CodeShape(Base, TimestampMixin):
"status": self.status, "status": self.status,
"snippet_id": self.snippet_id, "snippet_id": self.snippet_id,
"reason": self.reason, "reason": self.reason,
"reason_code": self.reason_code,
"classified_by": self.classified_by, "classified_by": self.classified_by,
"classified_at": iso(self.classified_at), "classified_at": iso(self.classified_at),
"first_seen_commit": self.first_seen_commit, "first_seen_commit": self.first_seen_commit,
@@ -167,6 +189,81 @@ class CodeShape(Base, TimestampMixin):
"updated_at": iso(self.updated_at), "updated_at": iso(self.updated_at),
} }
def to_compact(self) -> dict:
"""The row as an audit reads it (#2868): identity, standing, the
definition line and the proposer's word — none of the bookkeeping
(commits, shas, timestamps). A 500-row page of these fits the tool
budget; a page of to_dict() does not."""
out = {
"path": self.path,
"symbol": self.symbol,
"kind": self.kind,
"status": self.status,
"signature": self.signature,
}
if self.snippet_id is not None:
out["snippet_id"] = self.snippet_id
if self.classified_by:
out["by"] = self.classified_by
if self.reason_code:
out["reason_code"] = self.reason_code
proposal = self.proposal
if proposal:
out["proposal"] = proposal
if self.diverges_from is not None:
out["diverges_from"] = self.diverges_from
if self.recheck_at is not None:
out["recheck"] = True
return out
# How a uses edge was established (#2870): who/what said "this shape calls
# that canon". `reference` is the proposer's mechanical by-name hit on the
# body (language-gated, #2871); `hook` is write-path evidence (pulled the
# snippet, then wrote code naming its symbol); agent/audit/import are
# judgments carried on classify_shapes(..., uses=[...]).
USE_BASES = ("reference", "hook", "agent", "audit", "import")
class CodeShapeUse(Base):
"""One consumption edge: shape → canonical snippet it calls/uses (#2870).
Conformance (CodeShape.status/snippet_id) answers "what shape is this";
this table answers "what does it use" — many per shape. A service function
that is an instance of the service-function convention AND a consumer of
hash_token has one snippet_id and one uses edge. Cascades with both ends:
a use of a deleted snippet is no longer a fact worth keeping.
"""
__tablename__ = "code_shape_uses"
__table_args__ = (
UniqueConstraint("shape_id", "snippet_id", name="uq_code_shape_uses_shape_snippet"),
Index("ix_code_shape_uses_snippet", "snippet_id"),
)
id: Mapped[int] = mapped_column(primary_key=True)
shape_id: Mapped[int] = mapped_column(
Integer, ForeignKey("code_shapes.id", ondelete="CASCADE"), nullable=False
)
snippet_id: Mapped[int] = mapped_column(
Integer, ForeignKey("notes.id", ondelete="CASCADE"), nullable=False
)
basis: Mapped[str] = mapped_column(Text, nullable=False)
evidence: Mapped[str | None] = mapped_column(Text, nullable=True)
created_at: Mapped[datetime] = mapped_column(
DateTime(timezone=True), nullable=False, default=lambda: datetime.now(timezone.utc)
)
def to_dict(self) -> dict:
return {
"id": self.id,
"shape_id": self.shape_id,
"snippet_id": self.snippet_id,
"basis": self.basis,
"evidence": self.evidence,
"created_at": iso(self.created_at),
}
# What a shape's history records (#2793). Not "appeared" — first_seen and # What a shape's history records (#2793). Not "appeared" — first_seen and
# created_at already say that on the row; history is for what CHANGED: # created_at already say that on the row; history is for what CHANGED:
+5
View File
@@ -28,6 +28,10 @@ class RepoBinding(Base, TimestampMixin):
Integer, ForeignKey("projects.id", ondelete="CASCADE"), nullable=False Integer, ForeignKey("projects.id", ondelete="CASCADE"), nullable=False
) )
repo_key: Mapped[str] = mapped_column(Text, nullable=False) repo_key: Mapped[str] = mapped_column(Text, nullable=False)
# The branch the coverage refresh reads for this binding (#2873); NULL =
# the forge's default branch. Chosen at bind time so a dev-first project
# can have its ledger follow dev instead of waiting for the merge.
ref: Mapped[str | None] = mapped_column(Text, nullable=True)
def to_dict(self) -> dict: def to_dict(self) -> dict:
return { return {
@@ -35,6 +39,7 @@ class RepoBinding(Base, TimestampMixin):
"user_id": self.user_id, "user_id": self.user_id,
"project_id": self.project_id, "project_id": self.project_id,
"repo_key": self.repo_key, "repo_key": self.repo_key,
"ref": self.ref,
"created_at": iso(self.created_at), "created_at": iso(self.created_at),
"updated_at": iso(self.updated_at), "updated_at": iso(self.updated_at),
} }
+35 -3
View File
@@ -11,7 +11,7 @@ from scribe.models.note_supersession import NoteSupersession
from scribe.models.note_version import NoteVersion from scribe.models.note_version import NoteVersion
from scribe.models.design_system import DesignSystem, DesignToken from scribe.models.design_system import DesignSystem, DesignToken
from scribe.models.note_usage import NoteUsageEvent from scribe.models.note_usage import NoteUsageEvent
from scribe.models.code_shape import CodeShape, CodeShapeEvent from scribe.models.code_shape import CodeShape, CodeShapeEvent, CodeShapeUse
from scribe.models.project import Project from scribe.models.project import Project
from scribe.models.repo_binding import RepoBinding from scribe.models.repo_binding import RepoBinding
from scribe.models.rulebook import ( from scribe.models.rulebook import (
@@ -42,8 +42,11 @@ logger = logging.getLogger(__name__)
# snippet target survives the id re-mapping, else the row rejoins the todo. # snippet target survives the id re-mapping, else the row rejoins the todo.
# v8 (2026-08) added code_shape_events — the ledger's history (#2793): what # v8 (2026-08) added code_shape_events — the ledger's history (#2793): what
# was used where, when, and why is not recomputable, so it travels. # was used where, when, and why is not recomputable, so it travels.
# v9 (2026-08) added code_shape_uses — the ledger's consumption edges (#2870):
# judgment-grade edges (agent/audit/import) are operator records; mechanical
# ones (reference/hook) travel too, cheaply, and the next refresh refreshes them.
# Bump when the serialized schema changes. # Bump when the serialized schema changes.
BACKUP_VERSION = 8 BACKUP_VERSION = 9
# Every table this backup carries, by its REAL name. Paired with _NOT_INCLUDED # Every table this backup carries, by its REAL name. Paired with _NOT_INCLUDED
# below, these two lists must together account for the entire schema — which is # below, these two lists must together account for the entire schema — which is
@@ -62,7 +65,7 @@ _BACKED_UP = [
"systems", "record_systems", "design_systems", "design_tokens", "systems", "record_systems", "design_systems", "design_tokens",
"note_usage_events", "repo_bindings", "note_supersessions", "note_usage_events", "repo_bindings", "note_supersessions",
# v7 (2026-08): the shape ledger (#2787); v8: its history (#2793). # v7 (2026-08): the shape ledger (#2787); v8: its history (#2793).
"code_shapes", "code_shape_events", "code_shapes", "code_shape_events", "code_shape_uses",
] ]
# Tables intentionally NOT in the backup, surfaced in the payload so the gap is # Tables intentionally NOT in the backup, surfaced in the payload so the gap is
@@ -182,6 +185,10 @@ def _code_shape_event_rows(rows) -> list[dict]:
return [r.to_dict() for r in rows] return [r.to_dict() for r in rows]
def _code_shape_use_rows(rows) -> list[dict]:
return [r.to_dict() for r in rows]
def _repo_binding_rows(rows) -> list[dict]: def _repo_binding_rows(rows) -> list[dict]:
return [ return [
{"user_id": r.user_id, "project_id": r.project_id, "repo_key": r.repo_key} {"user_id": r.user_id, "project_id": r.project_id, "repo_key": r.repo_key}
@@ -361,6 +368,9 @@ async def export_full_backup() -> dict:
code_shape_events = (await session.execute( code_shape_events = (await session.execute(
select(CodeShapeEvent).order_by(CodeShapeEvent.at, CodeShapeEvent.id) select(CodeShapeEvent).order_by(CodeShapeEvent.at, CodeShapeEvent.id)
)).scalars().all() )).scalars().all()
code_shape_uses = (await session.execute(
select(CodeShapeUse).order_by(CodeShapeUse.shape_id, CodeShapeUse.snippet_id)
)).scalars().all()
rulebooks = (await session.execute(select(Rulebook))).scalars().all() rulebooks = (await session.execute(select(Rulebook))).scalars().all()
topics = (await session.execute(select(RulebookTopic))).scalars().all() topics = (await session.execute(select(RulebookTopic))).scalars().all()
rules = (await session.execute(select(Rule))).scalars().all() rules = (await session.execute(select(Rule))).scalars().all()
@@ -406,6 +416,7 @@ async def export_full_backup() -> dict:
"note_supersessions": _note_supersession_rows(supersessions), "note_supersessions": _note_supersession_rows(supersessions),
"code_shapes": _code_shape_rows(code_shapes), "code_shapes": _code_shape_rows(code_shapes),
"code_shape_events": _code_shape_event_rows(code_shape_events), "code_shape_events": _code_shape_event_rows(code_shape_events),
"code_shape_uses": _code_shape_use_rows(code_shape_uses),
} }
@@ -482,6 +493,11 @@ async def export_user_backup(user_id: int) -> dict:
select(CodeShapeEvent).where(CodeShapeEvent.project_id.in_(project_ids)) select(CodeShapeEvent).where(CodeShapeEvent.project_id.in_(project_ids))
.order_by(CodeShapeEvent.at, CodeShapeEvent.id) .order_by(CodeShapeEvent.at, CodeShapeEvent.id)
)).scalars().all() if project_ids else [] )).scalars().all() if project_ids else []
code_shape_uses = (await session.execute(
select(CodeShapeUse).join(CodeShape, CodeShape.id == CodeShapeUse.shape_id)
.where(CodeShape.project_id.in_(project_ids))
.order_by(CodeShapeUse.shape_id, CodeShapeUse.snippet_id)
)).scalars().all() if project_ids else []
rulebooks = (await session.execute( rulebooks = (await session.execute(
select(Rulebook).where(Rulebook.owner_user_id == user_id) select(Rulebook).where(Rulebook.owner_user_id == user_id)
)).scalars().all() )).scalars().all()
@@ -553,6 +569,7 @@ async def export_user_backup(user_id: int) -> dict:
"note_supersessions": _note_supersession_rows(supersessions), "note_supersessions": _note_supersession_rows(supersessions),
"code_shapes": _code_shape_rows(code_shapes), "code_shapes": _code_shape_rows(code_shapes),
"code_shape_events": _code_shape_event_rows(code_shape_events), "code_shape_events": _code_shape_event_rows(code_shape_events),
"code_shape_uses": _code_shape_use_rows(code_shape_uses),
} }
@@ -657,6 +674,7 @@ async def _restore_v2(data: dict) -> dict:
"systems": 0, "record_systems": 0, "design_systems": 0, "systems": 0, "record_systems": 0, "design_systems": 0,
"design_tokens": 0, "note_usage_events": 0, "repo_bindings": 0, "design_tokens": 0, "note_usage_events": 0, "repo_bindings": 0,
"note_supersessions": 0, "code_shapes": 0, "code_shape_events": 0, "note_supersessions": 0, "code_shapes": 0, "code_shape_events": 0,
"code_shape_uses": 0,
} }
async with async_session() as session: async with async_session() as session:
@@ -1105,6 +1123,20 @@ async def _restore_v2(data: dict) -> dict:
)) ))
stats["code_shape_events"] += 1 stats["code_shape_events"] += 1
# v9: consumption edges (#2870) ride their shape AND their snippet —
# both ends must have survived, or the edge is no longer a fact.
for use in data.get("code_shape_uses", []):
new_shape_id = shape_id_map.get(use.get("shape_id") or 0)
new_sid = note_id_map.get(use.get("snippet_id") or 0)
if new_shape_id is None or new_sid is None:
continue
session.add(CodeShapeUse(
shape_id=new_shape_id, snippet_id=new_sid,
basis=use.get("basis", "import"), evidence=use.get("evidence"),
created_at=_dt(use.get("created_at")),
))
stats["code_shape_uses"] += 1
await session.commit() await session.commit()
logger.info("Restored v2/v3 backup: %s", stats) logger.info("Restored v2/v3 backup: %s", stats)
+71 -9
View File
@@ -36,7 +36,7 @@ from typing import NamedTuple
from datetime import datetime, timedelta, timezone from datetime import datetime, timedelta, timezone
from scribe.services.forge import ForgeSelector, get_forges from scribe.services.forge import ForgeSelector, get_forges
from scribe.services.repo_bindings import keys_for_project from scribe.services.repo_bindings import bindings_for_project
from scribe.services.settings import get_setting, set_setting from scribe.services.settings import get_setting, set_setting
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@@ -107,6 +107,7 @@ class Definition(NamedTuple):
signature: str signature: str
body_sha: str body_sha: str
body: str body: str
line: int = -1 # 0-based line the definition starts on (#2869)
def _definition_on(raw: str) -> tuple[str, str] | None: def _definition_on(raw: str) -> tuple[str, str] | None:
@@ -184,13 +185,56 @@ def extract_definitions(text: str) -> list[Definition]:
end = j end = j
break break
block = lines[i:end] block = lines[i:end]
# A CSS rule's fingerprint is its DECLARATIONS, not its selector
# (#2872): the row's identity already carries the selector, and the
# question the fingerprint answers for derive grouping is "is this the
# same rule under another name?" — .closed-msg / .error-block /
# .success-msg with identical bodies are one dup group, not three
# lonely rows. Sym blocks keep their signature line in the hash.
hashed = block[1:] if kind == "css" and len(block) > 1 else block
out.append(Definition( out.append(Definition(
kind, name, lines[i].strip()[:_SIGNATURE_CAP], _block_sha(block), kind, name, lines[i].strip()[:_SIGNATURE_CAP], _block_sha(hashed),
"\n".join(block), "\n".join(block), i,
)) ))
return out return out
# --- by-construction scope (#2869) -------------------------------------------
#
# A Vue single-file component's `<style scoped>` rules and its `<script setup>`
# functions cannot be reached from any other file: they are one-offs by
# construction, not by judgment. The sync stamps them `scoped` (mechanical) so
# the human todo holds only shapes a person should look at, while the bodies
# stay in play for the proposer, derive grouping and divergence — the five
# auth views' identical rules were found exactly there. Unscoped `<style>` in
# a .vue and every non-.vue file stay ordinary.
_STYLE_OPEN_RE = re.compile(r"^\s*<style\b[^>]*\bscoped\b", re.IGNORECASE)
_STYLE_CLOSE_RE = re.compile(r"^\s*</style\s*>", re.IGNORECASE)
def scoped_definitions(path: str, text: str, defs: list[Definition]) -> set[tuple[str, str]]:
"""The (kind, name) pairs among ``defs`` that are one-offs by
construction in this file: every sym in a .vue, and every css rule
that starts inside a `<style scoped>` block. Empty for other files."""
if not (path or "").lower().endswith(".vue"):
return set()
ranges: list[tuple[int, int]] = []
open_at: int | None = None
for i, ln in enumerate(text.splitlines()):
if open_at is None and _STYLE_OPEN_RE.match(ln):
open_at = i
elif open_at is not None and _STYLE_CLOSE_RE.match(ln):
ranges.append((open_at, i))
open_at = None
out: set[tuple[str, str]] = set()
for d in defs:
if d.kind == "sym":
out.add((d.kind, d.name))
elif any(a <= d.line <= b for a, b in ranges):
out.add((d.kind, d.name))
return out
def extract_shapes(text: str) -> list[tuple[str, str]]: def extract_shapes(text: str) -> list[tuple[str, str]]:
"""Every (kind, name) this text DEFINES — kind is "css" or "sym". """Every (kind, name) this text DEFINES — kind is "css" or "sym".
@@ -220,6 +264,7 @@ class ArchiveShape(NamedTuple):
signature: str signature: str
body_sha: str body_sha: str
body: str body: str
scoped: bool = False # one-off by construction (#2869)
def shapes_from_archive(blob: bytes) -> list[tuple[str, str, str]]: def shapes_from_archive(blob: bytes) -> list[tuple[str, str, str]]:
@@ -250,9 +295,14 @@ def definitions_from_archive(blob: bytes) -> list[ArchiveShape]:
text = handle.read().decode("utf-8") text = handle.read().decode("utf-8")
except UnicodeDecodeError: except UnicodeDecodeError:
continue continue
defs = extract_definitions(text)
scoped = scoped_definitions(path, text, defs)
shapes.extend( shapes.extend(
ArchiveShape(path, d.kind, d.name, d.signature, d.body_sha, d.body) ArchiveShape(
for d in extract_definitions(text) path, d.kind, d.name, d.signature, d.body_sha, d.body,
(d.kind, d.name) in scoped,
)
for d in defs
) )
return shapes return shapes
@@ -357,12 +407,15 @@ async def compute_coverage(
# the project's repos (#2792). # the project's repos (#2792).
canons = None canons = None
proposer_stats = {"examined": 0, "proposed": 0, "semantic_checked": 0} proposer_stats = {"examined": 0, "proposed": 0, "semantic_checked": 0}
for key in await keys_for_project(user_id, project_id): for binding in await bindings_for_project(user_id, project_id):
key = binding.repo_key
hit = selector.resolve(key) hit = selector.resolve(key)
if hit is None: if hit is None:
continue # bound to a host no connection serves continue # bound to a host no connection serves
forge, api_repo = hit forge, api_repo = hit
ref = await forge.default_branch(api_repo) # The binding's own ref when it names one (#2873: a dev-first project
# has its ledger follow dev), else the forge's default branch.
ref = binding.ref or await forge.default_branch(api_repo)
definitions = definitions_from_archive(await forge.archive(api_repo, ref)) definitions = definitions_from_archive(await forge.archive(api_repo, ref))
# The head commit is provenance sugar on the ledger rows; failing to # The head commit is provenance sugar on the ledger rows; failing to
# learn it must not fail the sync — the ref names the point well # learn it must not fail the sync — the ref names the point well
@@ -414,7 +467,7 @@ async def compute_coverage(
# repo that was unreachable today still has live rows, and they count. # repo that was unreachable today still has live rows, and they count.
rows = await shape_ledger.live_rows(project_id) rows = await shape_ledger.live_rows(project_id)
counts = {"canonical": 0, "instance": 0, "variant": 0, "exempt": 0, counts = {"canonical": 0, "instance": 0, "variant": 0, "exempt": 0,
"unclassified": 0} "scoped": 0, "unclassified": 0}
for row in rows: for row in rows:
counts[row.status] = counts.get(row.status, 0) + 1 counts[row.status] = counts.get(row.status, 0) + 1
by_repo: dict[str, dict[str, int]] = {} by_repo: dict[str, dict[str, int]] = {}
@@ -435,6 +488,7 @@ async def compute_coverage(
# confirm, the largest derive-first groups, and what this refresh did. # confirm, the largest derive-first groups, and what this refresh did.
"proposed": proposals["proposed"], "proposed": proposals["proposed"],
"derive_groups": proposals["derive_groups"], "derive_groups": proposals["derive_groups"],
"top_canon": proposals.get("top_canon"),
"proposer": proposer_stats, "proposer": proposer_stats,
# The divergence readout (#2793): button B where button A is canon, # The divergence readout (#2793): button B where button A is canon,
# and judged shapes whose bodies moved since they were judged. # and judged shapes whose bodies moved since they were judged.
@@ -571,7 +625,7 @@ def coverage_line(coverage: dict) -> str:
counts = coverage.get("counts") or {} counts = coverage.get("counts") or {}
breakdown = " · ".join( breakdown = " · ".join(
f"{counts[k]} {k}" f"{counts[k]} {k}"
for k in ("canonical", "instance", "variant", "exempt") for k in ("canonical", "instance", "variant", "exempt", "scoped")
if counts.get(k) if counts.get(k)
) )
line = ( line = (
@@ -592,6 +646,14 @@ def coverage_line(coverage: dict) -> str:
standing.append(f"{n_groups} derive group{'s' if n_groups != 1 else ''}") standing.append(f"{n_groups} derive group{'s' if n_groups != 1 else ''}")
if coverage.get("divergent"): if coverage.get("divergent"):
standing.append(f"{coverage['divergent']} DIVERGENT") standing.append(f"{coverage['divergent']} DIVERGENT")
# The next action, on the line (#2874): the canon with the biggest
# queue to confirm, and the widest body-identical copy to consolidate.
top = coverage.get("top_canon") or {}
if top.get("snippet_id"):
standing.append(f"top canon #{top['snippet_id']} ×{top.get('count', 0)}")
first = (coverage.get("derive_groups") or [{}])[0]
if first.get("label") and first.get("files"):
standing.append(f"top copy {first['label']} ×{first['files']} files")
if standing: if standing:
line += f" ({', '.join(standing)})" line += f" ({', '.join(standing)})"
gaps = [g["dir"] for g in coverage.get("largest_gaps") or []] gaps = [g["dir"] for g in coverage.get("largest_gaps") or []]
+10
View File
@@ -16,6 +16,7 @@ endorsed, so a one-off direct share has to be searched for rather than arriving
in your ambient lists. in your ambient lists.
""" """
import json import json
import re
import logging import logging
from sqlalchemy import and_, func, or_, select from sqlalchemy import and_, func, or_, select
@@ -76,6 +77,10 @@ def location_matches(data: dict | None, parts: dict[str, str]) -> bool:
if all( if all(
_path_matches((loc.get(key) or "").strip(), want) _path_matches((loc.get(key) or "").strip(), want)
if key == "path" if key == "path"
# Repo names are recorded free-form ("Scribe" / "FabledScribe" /
# "fabledscribe") — case is never the distinguishing thing (#2874).
else (loc.get(key) or "").strip().lower() == want.lower()
if key == "repo"
else (loc.get(key) or "").strip() == want else (loc.get(key) or "").strip() == want
for key, want in parts.items() for key, want in parts.items()
): ):
@@ -99,6 +104,11 @@ def location_jsonpath(parts: dict[str, str]) -> str:
if key == "path": if key == "path":
prefix = json.dumps(want.rstrip("/") + "/") prefix = json.dumps(want.rstrip("/") + "/")
filters.append(f"(@.path == {literal} || @.path starts with {prefix})") filters.append(f"(@.path == {literal} || @.path starts with {prefix})")
elif key == "repo":
# Case-insensitive, anchored, regex-escaped (#2874) — mirrors the
# Python dialect's .lower() compare.
pattern = json.dumps("^" + re.escape(want) + "$")
filters.append(f'(@.repo like_regex {pattern} flag "i")')
else: else:
filters.append(f"@.{key} == {literal}") filters.append(f"@.{key} == {literal}")
return f"$.locations[*] ? ({' && '.join(filters)})" return f"$.locations[*] ? ({' && '.join(filters)})"
+24 -2
View File
@@ -68,8 +68,16 @@ async def resolve_project(user_id: int, raw_repo: str) -> int | None:
return row.scalar_one_or_none() return row.scalar_one_or_none()
async def set_binding(user_id: int, raw_repo: str, project_id: int) -> RepoBinding: async def set_binding(
"""Create or update the binding for a repo. Idempotent on (user, repo_key).""" user_id: int, raw_repo: str, project_id: int, ref: str | None = None,
) -> RepoBinding:
"""Create or update the binding for a repo. Idempotent on (user, repo_key).
``ref`` (#2873) is the branch the coverage refresh reads for this
binding: a name sets it, ``""`` clears it back to the forge's default
branch, ``None`` leaves whatever stands (a re-bind that only moves the
project keeps the ref it had).
"""
key = normalize_repo_key(raw_repo) key = normalize_repo_key(raw_repo)
if not key: if not key:
raise ValueError("repo remote is empty or unparseable") raise ValueError("repo remote is empty or unparseable")
@@ -85,6 +93,8 @@ async def set_binding(user_id: int, raw_repo: str, project_id: int) -> RepoBindi
session.add(binding) session.add(binding)
else: else:
binding.project_id = project_id binding.project_id = project_id
if ref is not None:
binding.ref = ref.strip() or None
await session.commit() await session.commit()
await session.refresh(binding) await session.refresh(binding)
return binding return binding
@@ -100,6 +110,18 @@ async def list_bindings(user_id: int) -> list[RepoBinding]:
return list(rows.scalars().all()) return list(rows.scalars().all())
async def bindings_for_project(user_id: int, project_id: int) -> list[RepoBinding]:
"""Every binding of a project — key AND the ref its ledger follows (#2873)."""
async with async_session() as session:
rows = await session.execute(
select(RepoBinding).where(
RepoBinding.user_id == user_id,
RepoBinding.project_id == project_id,
).order_by(RepoBinding.repo_key)
)
return list(rows.scalars().all())
async def keys_for_project(user_id: int, project_id: int) -> list[str]: async def keys_for_project(user_id: int, project_id: int) -> list[str]:
"""Every repo key bound to a project — the snippet→forge join (#2691). """Every repo key bound to a project — the snippet→forge join (#2691).
+362 -30
View File
@@ -30,7 +30,7 @@ from typing import Iterable, NamedTuple
from sqlalchemy import select from sqlalchemy import select
from scribe.models import async_session from scribe.models import async_session
from scribe.models.code_shape import CodeShape, CodeShapeEvent from scribe.models.code_shape import REASON_CODES, CodeShape, CodeShapeEvent, CodeShapeUse
from scribe.models.base import iso from scribe.models.base import iso
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@@ -65,6 +65,17 @@ def location_covers(loc_path: str, loc_symbol: str, path: str, name: str) -> boo
return _path_touches(loc_path, path) return _path_touches(loc_path, path)
# The rows the machine may still speak about: nobody's judgment stands on
# them. `scoped` (#2869) is the sync's own by-construction stamp — the
# proposer, derive grouping, divergence, hook evidence and sweeps treat it
# like the todo; only the human todo (`unclassified`) excludes it.
_MECHANICAL_TODO = ("unclassified", "scoped")
_SCOPED_REASON = (
"by construction: a Vue component's scoped <style> rule / <script setup> "
"function — unreachable from any other file (stamped by the coverage sync)"
)
async def sync_repo_shapes( async def sync_repo_shapes(
project_id: int, project_id: int,
repo_key: str, repo_key: str,
@@ -76,7 +87,9 @@ async def sync_repo_shapes(
``shapes`` are (path, kind, name) triples, or the richer ArchiveShape ``shapes`` are (path, kind, name) triples, or the richer ArchiveShape
records (#2792) whose 4th/5th fields — signature, body_sha — refresh the records (#2792) whose 4th/5th fields — signature, body_sha — refresh the
row's content fingerprint. ``seen_marker`` is the commit the archive was row's content fingerprint, and whose 7th (#2869) says the shape is a
one-off by construction: such rows are stamped `scoped` (mechanical)
while unjudged, and un-stamped if a later tree makes them reachable. ``seen_marker`` is the commit the archive was
read at when the forge can say, else the ref name — provenance sugar; read at when the forge can say, else the ref name — provenance sugar;
the row timestamps carry the when. the row timestamps carry the when.
""" """
@@ -96,20 +109,32 @@ async def sync_repo_shapes(
path, kind, name = shape[0], shape[1], shape[2] path, kind, name = shape[0], shape[1], shape[2]
signature = shape[3] if len(shape) > 3 else "" signature = shape[3] if len(shape) > 3 else ""
body_sha = shape[4] if len(shape) > 4 else "" body_sha = shape[4] if len(shape) > 4 else ""
scoped = bool(shape[6]) if len(shape) > 6 else False
key = (path, name, kind) key = (path, name, kind)
if key in seen: if key in seen:
continue continue
seen.add(key) seen.add(key)
row = by_key.get(key) row = by_key.get(key)
if row is None: if row is None:
session.add(CodeShape( row = CodeShape(
project_id=project_id, repo_key=repo_key, project_id=project_id, repo_key=repo_key,
path=path, symbol=name, kind=kind, path=path, symbol=name, kind=kind,
first_seen_commit=seen_marker, last_seen_commit=seen_marker, first_seen_commit=seen_marker, last_seen_commit=seen_marker,
signature=signature, body_sha=body_sha, signature=signature, body_sha=body_sha,
)) )
if scoped:
await _judge(session, row, status="scoped", snippet_id=None,
by="mechanical", reason=_SCOPED_REASON, at=now)
else:
session.add(row)
continue continue
row.last_seen_commit = seen_marker row.last_seen_commit = seen_marker
if scoped and row.status == "unclassified":
await _judge(session, row, status="scoped", snippet_id=None,
by="mechanical", reason=_SCOPED_REASON, at=now)
elif not scoped and row.status == "scoped":
await _judge(session, row, status="unclassified", snippet_id=None,
by=None, reason=None, at=now)
if signature: if signature:
row.signature = signature row.signature = signature
if body_sha and body_sha != row.body_sha: if body_sha and body_sha != row.body_sha:
@@ -158,7 +183,7 @@ def _event(row: CodeShape, event: str, at: datetime, *, commit: str = "") -> Cod
async def _judge( async def _judge(
session, row: CodeShape, *, status: str, snippet_id: int | None, session, row: CodeShape, *, status: str, snippet_id: int | None,
by: str | None, reason: str | None, at: datetime, by: str | None, reason: str | None, at: datetime, reason_code: str | None = None,
) -> None: ) -> None:
"""Apply a judgment to a row — the ONE place a status is set — and write """Apply a judgment to a row — the ONE place a status is set — and write
its history. Clears what a judgment settles: the standing proposal, the its history. Clears what a judgment settles: the standing proposal, the
@@ -169,6 +194,7 @@ async def _judge(
row.status = status row.status = status
row.snippet_id = snippet_id if status in _NEEDS_TARGET else None row.snippet_id = snippet_id if status in _NEEDS_TARGET else None
row.reason = (reason or "").strip() or None row.reason = (reason or "").strip() or None
row.reason_code = (reason_code or "").strip() or None
row.classified_by = by if status != "unclassified" else None row.classified_by = by if status != "unclassified" else None
row.classified_at = at if status != "unclassified" else None row.classified_at = at if status != "unclassified" else None
row.classified_sha = row.body_sha if status != "unclassified" else "" row.classified_sha = row.body_sha if status != "unclassified" else ""
@@ -183,6 +209,59 @@ async def _judge(
session.add(_event(row, "classified", at)) session.add(_event(row, "classified", at))
async def record_uses(
session, row: CodeShape, snippet_ids, *, basis: str, evidence: str | None = None,
) -> int:
"""Upsert consumption edges shape → snippet (#2870). A judgment-grade
basis (agent/audit/import) overwrites a mechanical one (reference/hook)
on the same edge; mechanical never overwrites a judgment. Returns the
number of edges written or refreshed. The row must be persisted (flushed)
so it has an id."""
wanted = {int(x) for x in (snippet_ids or []) if x}
if not wanted:
return 0
if row.id is None:
session.add(row)
await session.flush()
existing = {
e.snippet_id: e
for e in (
await session.execute(
select(CodeShapeUse).where(CodeShapeUse.shape_id == row.id)
)
).scalars().all()
}
judged = basis in _CALLER_VIAS
n = 0
for sid in wanted:
edge = existing.get(sid)
if edge is None:
session.add(CodeShapeUse(shape_id=row.id, snippet_id=sid, basis=basis, evidence=evidence))
n += 1
elif judged or edge.basis not in _CALLER_VIAS:
edge.basis, edge.evidence = basis, evidence
n += 1
return n
async def uses_of(shape_ids) -> dict[int, list[CodeShapeUse]]:
"""{shape_id: [edges]} for a set of rows — the read side of record_uses."""
ids = [int(x) for x in shape_ids if x]
if not ids:
return {}
async with async_session() as session:
edges = (
await session.execute(
select(CodeShapeUse).where(CodeShapeUse.shape_id.in_(ids))
.order_by(CodeShapeUse.shape_id, CodeShapeUse.snippet_id)
)
).scalars().all()
out: dict[int, list[CodeShapeUse]] = {}
for e in edges:
out.setdefault(e.shape_id, []).append(e)
return out
async def mark_canonicals( async def mark_canonicals(
project_id: int, recorded: list[tuple[int, str, str]] project_id: int, recorded: list[tuple[int, str, str]]
) -> None: ) -> None:
@@ -214,7 +293,7 @@ async def mark_canonicals(
), ),
None, None,
) )
if covering is not None and row.status == "unclassified": if covering is not None and row.status in _MECHANICAL_TODO:
await _judge(session, row, status="canonical", snippet_id=covering, await _judge(session, row, status="canonical", snippet_id=covering,
by="mechanical", reason=None, at=now) by="mechanical", reason=None, at=now)
elif ( elif (
@@ -289,6 +368,17 @@ def validate_classifications(items: list[dict]) -> str | None:
f"classifications[{i}]: status {status!r} needs a reason — " f"classifications[{i}]: status {status!r} needs a reason — "
"the WHY is the record (note 2786)" "the WHY is the record (note 2786)"
) )
code = (item.get("reason_code") or "").strip()
if code and code not in REASON_CODES:
return (
f"classifications[{i}]: unknown reason_code {code!r} "
f"(one of: {', '.join(REASON_CODES)})"
)
uses = item.get("uses")
if uses is not None and (
not isinstance(uses, list) or not all(isinstance(u, int) and u > 0 for u in uses)
):
return f"classifications[{i}]: uses must be a list of snippet ids"
return None return None
@@ -327,6 +417,8 @@ async def classify_shapes(
for item in classifications for item in classifications
if item.get("status") in _NEEDS_TARGET if item.get("status") in _NEEDS_TARGET
} }
for item in classifications:
target_ids.update(int(u) for u in (item.get("uses") or []))
for sid in sorted(target_ids): for sid in sorted(target_ids):
if await snippets_svc.get_snippet(user_id, sid) is None: if await snippets_svc.get_snippet(user_id, sid) is None:
raise ValueError(f"snippet {sid} not found (or not readable)") raise ValueError(f"snippet {sid} not found (or not readable)")
@@ -364,12 +456,101 @@ async def classify_shapes(
session, row, status=status, session, row, status=status,
snippet_id=int(item["snippet_id"]) if status in _NEEDS_TARGET else None, snippet_id=int(item["snippet_id"]) if status in _NEEDS_TARGET else None,
by=via, reason=item.get("reason"), at=now, by=via, reason=item.get("reason"), at=now,
reason_code=item.get("reason_code"),
) )
if item.get("uses"):
await record_uses(session, row, item["uses"], basis=via,
evidence=item.get("reason"))
classified += 1 classified += 1
await session.commit() await session.commit()
return {"classified": classified, "unmatched": unmatched} return {"classified": classified, "unmatched": unmatched}
def rule_matches(row: CodeShape, *, path: str, pattern: str, kind: str) -> bool:
"""Does a ledger row fall under a rule-form classification (#2868)?
``path`` is a file or a directory (everything beneath it), ``pattern``
a shell glob on the symbol (``""`` = every symbol), ``kind`` narrows to
sym/css. Pure, so the sweep's reach can be tested without a database."""
import fnmatch
clean = (path or "").strip().strip("/")
if clean and not (row.path == clean or row.path.startswith(clean + "/")):
return False
if kind and row.kind != kind:
return False
if pattern and not fnmatch.fnmatchcase(_norm_symbol(row.symbol), pattern):
return False
return True
async def classify_shapes_where(
user_id: int,
project_id: int,
*,
path: str,
status: str,
pattern: str = "",
kind: str = "",
snippet_id: int | None = None,
reason: str | None = None,
via: str = "agent",
include_judged: bool = False,
reason_code: str | None = None,
uses: list[int] | None = None,
) -> dict:
"""The sweep form of classify_shapes (#2868): one judgment applied to
every live row under ``path`` whose symbol matches ``pattern`` (and
``kind``). By default only unjudged rows are touched — `unclassified`
and the sync's mechanical `scoped` stamp — a sweep must never silently
overwrite a judgment; ``include_judged`` opts in.
Same gates as the row form (status vocabulary, snippet target, reason
for variant/exempt, write access); one transaction, so it applies whole
or not at all. Returns the count and a sample of what it judged."""
from scribe.services import access
from scribe.services import snippets as snippets_svc
if via not in _CALLER_VIAS:
raise ValueError(f"via must be one of: {', '.join(_CALLER_VIAS)}")
if not (path or "").strip():
raise ValueError("path is required — a sweep names the directory it judges")
if status == "canonical":
raise ValueError("canonical is the sync's stamp on a snippet's own location — a sweep cannot set it")
probe = {"path": path, "symbol": "*", "status": status,
"snippet_id": snippet_id or 0, "reason": reason or "",
"reason_code": reason_code or "", "uses": uses}
error = validate_classifications([probe])
if error:
raise ValueError(error.replace("classifications[0]", "rule"))
if not await access.can_write_project(user_id, project_id):
raise ValueError(f"project {project_id} not found or no write access")
if status in _NEEDS_TARGET and await snippets_svc.get_snippet(user_id, int(snippet_id)) is None:
raise ValueError(f"snippet {snippet_id} not found (or not readable)")
for sid in sorted({int(u) for u in (uses or [])}):
if await snippets_svc.get_snippet(user_id, sid) is None:
raise ValueError(f"snippet {sid} not found (or not readable)")
now = datetime.now(timezone.utc)
judged: list[str] = []
async with async_session() as session:
conds = [CodeShape.project_id == project_id, CodeShape.vanished_at.is_(None)]
if not include_judged:
conds.append(CodeShape.status.in_(_MECHANICAL_TODO))
rows = (await session.execute(select(CodeShape).where(*conds))).scalars().all()
for row in rows:
if not rule_matches(row, path=path, pattern=pattern, kind=kind):
continue
await _judge(
session, row, status=status,
snippet_id=int(snippet_id) if status in _NEEDS_TARGET else None,
by=via, reason=reason, at=now, reason_code=reason_code,
)
if uses:
await record_uses(session, row, uses, basis=via, evidence=reason)
judged.append(f"{row.path}::{row.symbol}")
await session.commit()
return {"classified": len(judged), "sample": sorted(judged)[:12]}
async def list_project_shapes( async def list_project_shapes(
user_id: int, user_id: int,
project_id: int, project_id: int,
@@ -382,6 +563,7 @@ async def list_project_shapes(
offset: int = 0, offset: int = 0,
proposal: str = "", proposal: str = "",
flag: str = "", flag: str = "",
uses: int = 0,
) -> tuple[list[CodeShape], int]: ) -> tuple[list[CodeShape], int]:
"""A filtered page of a project's ledger, with the unfiltered-match total. """A filtered page of a project's ledger, with the unfiltered-match total.
@@ -427,6 +609,12 @@ async def list_project_shapes(
conds.append(CodeShape.diverges_from.isnot(None)) conds.append(CodeShape.diverges_from.isnot(None))
elif flag == "recheck": elif flag == "recheck":
conds.append(CodeShape.recheck_at.isnot(None)) conds.append(CodeShape.recheck_at.isnot(None))
if uses:
# Consumers of a canon (#2870): rows with a uses edge to it, whatever
# shape they themselves are.
conds.append(CodeShape.id.in_(
select(CodeShapeUse.shape_id).where(CodeShapeUse.snippet_id == uses)
))
async with async_session() as session: async with async_session() as session:
total = ( total = (
await session.execute( await session.execute(
@@ -477,18 +665,38 @@ async def snippet_consumers(user_id: int, note_id: int) -> dict:
) )
) )
).scalars().all() ).scalars().all()
readable: dict[int, bool] = {} # The consumption edges (#2870): rows that USE this canon, whatever
out: dict[str, list[dict]] = {"instances": [], "variants": []} # shape they are themselves — the call-site map.
for row in rows: using = (
if row.project_id not in readable: await session.execute(
readable[row.project_id] = await access.can_read_project( select(CodeShape, CodeShapeUse.basis, CodeShapeUse.evidence)
user_id, row.project_id .join(CodeShapeUse, CodeShapeUse.shape_id == CodeShape.id)
.where(CodeShapeUse.snippet_id == note_id, CodeShape.vanished_at.is_(None))
) )
if not readable[row.project_id]: ).all()
readable: dict[int, bool] = {}
async def can_read(pid: int) -> bool:
if pid not in readable:
readable[pid] = await access.can_read_project(user_id, pid)
return readable[pid]
out: dict[str, list[dict]] = {"instances": [], "variants": [], "uses": []}
for row in rows:
if not await can_read(row.project_id):
continue continue
out["instances" if row.status == "instance" else "variants"].append( out["instances" if row.status == "instance" else "variants"].append(
_consumer_dict(row) _consumer_dict(row)
) )
for row, basis, evidence in using:
if not await can_read(row.project_id):
continue
d = _consumer_dict(row)
d["basis"] = basis
if evidence:
d["evidence"] = evidence
d.pop("reason", None)
out["uses"].append(d)
return out return out
@@ -660,7 +868,7 @@ async def stamp_write_path_instances(
) )
session.add(row) session.add(row)
by_key[(name, kind)] = row by_key[(name, kind)] = row
elif not (row.status == "unclassified" or row.classified_by == "hook"): elif not (row.status in _MECHANICAL_TODO or row.classified_by == "hook"):
continue # a judgment — or the canon itself — stands continue # a judgment — or the canon itself — stands
await _judge(session, row, status="instance", snippet_id=sid, by="hook", await _judge(session, row, status="instance", snippet_id=sid, by="hook",
reason=why, at=now) reason=why, at=now)
@@ -668,6 +876,13 @@ async def stamp_write_path_instances(
"path": path, "symbol": name, "kind": kind, "path": path, "symbol": name, "kind": kind,
"snippet_id": sid, "reason": why, "snippet_id": sid, "reason": why,
}) })
# Every pulled canon the payload NAMES is a uses edge (#2870) — the
# call-site fact, independent of which one the row is judged to be.
await record_uses(
session, row,
[s_id for rank, _at, s_id, _why in bucket if rank == 2],
basis="hook", evidence="write path: pulled the snippet, payload names its symbol",
)
if stamped: if stamped:
await session.commit() await session.commit()
return stamped return stamped
@@ -718,7 +933,9 @@ _SEMANTIC_CAP = 150
_SEMANTIC_FLOOR = 0.8 _SEMANTIC_FLOOR = 0.8
# Bump when a basis's rule changes: rows remember the (body, ruleset) they # Bump when a basis's rule changes: rows remember the (body, ruleset) they
# were examined under, so a tightened rule re-examines everything once. # were examined under, so a tightened rule re-examines everything once.
_PROPOSER_VERSION = 2 # v3: language-family gate on the sym bases, reference stoplist, semantic
# restricted to the shape's own project (#2871).
_PROPOSER_VERSION = 3
# Signature resemblance floor, name blanked (difflib ratio) — and a length # Signature resemblance floor, name blanked (difflib ratio) — and a length
# floor, because `def NAME():` resembles `def NAME(x):` at 0.95 while saying # floor, because `def NAME():` resembles `def NAME(x):` at 0.95 while saying
# nothing; a family shape has parameters to resemble. # nothing; a family shape has parameters to resemble.
@@ -737,6 +954,64 @@ class Canon(NamedTuple):
signature: str signature: str
code_norm: str code_norm: str
project_id: int = 0 project_id: int = 0
language: str = "" # the snippet's recorded language; "" = unknown, no gate
# Language families: the sym bases only propose within one. The 2026-08
# audit (#2871) found every cross-language hit wrong — a Python tool-module
# canon named `register` proposed for Vue `handleSubmit`s that call
# `authStore.register()`, and a TS store's `register` matched it by symbol;
# Minstrel/Forge TS canon proposed for Python bodies by resemblance. CSS is
# its own kind and is not gated here.
_FAMILY_BY_LANG = {
"python": "py", "py": "py",
"typescript": "js", "ts": "js", "tsx": "js", "javascript": "js", "js": "js",
"jsx": "js", "vue": "js", "mjs": "js", "cjs": "js",
"css": "css", "scss": "css", "sass": "css", "less": "css",
"bash": "sh", "sh": "sh", "shell": "sh", "zsh": "sh",
"sql": "sql",
}
_FAMILY_BY_EXT = {
".py": "py", ".pyi": "py",
".ts": "js", ".tsx": "js", ".js": "js", ".jsx": "js", ".vue": "js", ".mjs": "js", ".cjs": "js",
".css": "css", ".scss": "css", ".sass": "css", ".less": "css",
".sh": "sh", ".bash": "sh", ".zsh": "sh",
".sql": "sql",
}
def language_family(language: str) -> str:
"""The family a recorded snippet language belongs to ("" when unknown)."""
return _FAMILY_BY_LANG.get((language or "").strip().lower(), "")
def path_family(path: str) -> str:
"""The family a file path belongs to, by extension ("" when unknown)."""
p = (path or "").lower()
for ext, fam in _FAMILY_BY_EXT.items():
if p.endswith(ext):
return fam
return ""
def same_family(path: str, canon_language: str) -> bool:
"""A sym basis may propose this canon for this path: both families known
and equal, or either unknown (no evidence either way → no gate)."""
a = path_family(path)
b = language_family(canon_language)
return not a or not b or a == b
# Reference basis: generic verbs name too many unrelated things to count a
# bare mention as a call site of THIS canon (`register`, `load`, `save` …).
# The symbol basis still catches a second definition of such a name; the
# call-site relation for these becomes a `uses` edge once #2870 lands.
_REFERENCE_STOPLIST = frozenset({
"get", "set", "put", "post", "load", "save", "run", "main", "init", "setup",
"register", "restore", "reset", "toggle", "close", "open", "submit", "handler",
"update", "create", "delete", "remove", "add", "start", "stop", "send",
"receive", "render", "mount", "dispatch", "call", "apply", "execute",
})
def _norm_text(text: str) -> str: def _norm_text(text: str) -> str:
@@ -768,6 +1043,26 @@ def text_contains(body: str, code: str) -> bool:
return a in b or b in a return a in b or b in a
def reference_canons(kind: str, path: str, symbol: str, body: str, canons: Iterable[Canon]) -> list[int]:
"""Every canon this body NAMES (#2870) — the uses edges the proposer can
write mechanically: same kind, same language family, symbol not in the
generic-verb stoplist, and not the shape's own name."""
norm_sym = _norm_symbol(symbol)
out: list[int] = []
for c in canons:
if c.kind != kind or not c.symbol:
continue
if kind == "sym" and not same_family(path, c.language):
continue
if _norm_symbol(c.symbol) == norm_sym:
continue
if _norm_symbol(c.symbol).lower() in _REFERENCE_STOPLIST:
continue
if references_symbol(body, c.symbol, kind):
out.append(c.snippet_id)
return out
def match_canon( def match_canon(
kind: str, path: str, symbol: str, signature: str, body: str, kind: str, path: str, symbol: str, signature: str, body: str,
canons: Iterable[Canon], *, project_id: int = 0, canons: Iterable[Canon], *, project_id: int = 0,
@@ -789,11 +1084,17 @@ def match_canon(
for c in canons: for c in canons:
if c.kind != kind: if c.kind != kind:
continue continue
if kind == "sym" and not same_family(path, c.language):
continue # a Python canon says nothing about a Vue body, and vice versa
if c.symbol and _norm_symbol(c.symbol) == norm_sym: if c.symbol and _norm_symbol(c.symbol) == norm_sym:
if not any(location_covers(lp, ls, path, symbol) for lp, ls in c.locations): if not any(location_covers(lp, ls, path, symbol) for lp, ls in c.locations):
offer("symbol", 1.0, c) offer("symbol", 1.0, c)
continue # its own location is canonical territory, not a proposal continue # its own location is canonical territory, not a proposal
if c.symbol and references_symbol(body, c.symbol, kind): if (
c.symbol
and _norm_symbol(c.symbol).lower() not in _REFERENCE_STOPLIST
and references_symbol(body, c.symbol, kind)
):
offer("reference", 0.9, c) offer("reference", 0.9, c)
if c.code_norm and text_contains(body, c.code_norm): if c.code_norm and text_contains(body, c.code_norm):
offer("text", 0.95, c) offer("text", 0.95, c)
@@ -849,6 +1150,7 @@ async def canon_catalog(user_id: int) -> list[Canon]:
for loc in fields.get("locations") or [] for loc in fields.get("locations") or []
), ),
signature, _norm_text(code), int(note.project_id or 0), signature, _norm_text(code), int(note.project_id or 0),
(fields.get("language") or "").strip().lower(),
)) ))
return out return out
@@ -907,7 +1209,14 @@ async def propose_for_repo(
if canons is None: if canons is None:
canons = await canon_catalog(user_id) canons = await canon_catalog(user_id)
by_key = {(d[0], d[1], d[2]): d for d in definitions} by_key = {(d[0], d[1], d[2]): d for d in definitions}
sym_canon_ids = {c.snippet_id for c in canons if c.kind == "sym"} # The semantic arm is the widest net and, across projects, was pure noise
# in the 2026-08 audit (#2871): it is held to the shape's own project and
# language family. The precise bases (symbol/text) still reach family
# canon in other projects (note 2786).
sym_canons = [c for c in canons if c.kind == "sym" and c.project_id == project_id]
def semantic_allowed(path: str) -> set[int]:
return {c.snippet_id for c in sym_canons if same_family(path, c.language)}
now = datetime.now(timezone.utc) now = datetime.now(timezone.utc)
examined = proposed = checked = 0 examined = proposed = checked = 0
async with async_session() as session: async with async_session() as session:
@@ -916,7 +1225,7 @@ async def propose_for_repo(
select(CodeShape).where( select(CodeShape).where(
CodeShape.project_id == project_id, CodeShape.project_id == project_id,
CodeShape.repo_key == repo_key, CodeShape.repo_key == repo_key,
CodeShape.status == "unclassified", CodeShape.status.in_(_MECHANICAL_TODO),
CodeShape.vanished_at.is_(None), CodeShape.vanished_at.is_(None),
) )
) )
@@ -940,6 +1249,12 @@ async def propose_for_repo(
row.proposal_group = group row.proposal_group = group
row.proposed_at = now row.proposed_at = now
row.proposed_sha = examined_as row.proposed_sha = examined_as
# Consumption is recorded for every canon the body names (#2870),
# whatever the row is then judged to be.
used = reference_canons(row.kind, row.path, row.symbol, body, canons)
if used:
await record_uses(session, row, used, basis="reference",
evidence="proposer: body names the canon's symbol")
if hit: if hit:
row.proposed_snippet_id, row.proposal_basis, row.proposal_score = hit row.proposed_snippet_id, row.proposal_basis, row.proposal_score = hit
row.proposal_group = None row.proposal_group = None
@@ -955,7 +1270,7 @@ async def propose_for_repo(
continue continue
checked += 1 checked += 1
try: try:
found = await _semantic_canon(user_id, d[5], sym_canon_ids) found = await _semantic_canon(user_id, d[5], semantic_allowed(row.path))
except Exception: except Exception:
logger.warning("semantic proposal failed", exc_info=True) logger.warning("semantic proposal failed", exc_info=True)
found = None found = None
@@ -1005,7 +1320,7 @@ async def apply_derive_groups(project_id: int) -> int:
await session.execute( await session.execute(
select(CodeShape).where( select(CodeShape).where(
CodeShape.project_id == project_id, CodeShape.project_id == project_id,
CodeShape.status == "unclassified", CodeShape.status.in_(_MECHANICAL_TODO),
CodeShape.vanished_at.is_(None), CodeShape.vanished_at.is_(None),
CodeShape.proposed_snippet_id.is_(None), CodeShape.proposed_snippet_id.is_(None),
) )
@@ -1038,27 +1353,44 @@ def proposal_summary(rows: Iterable[CodeShape], *, top: int = 8) -> dict:
"""The readout's view of the proposer's standing: how many canon """The readout's view of the proposer's standing: how many canon
proposals await confirmation, and the largest derive-first groups.""" proposals await confirmation, and the largest derive-first groups."""
proposed = 0 proposed = 0
by_canon: dict[int, int] = {}
groups: dict[str, dict] = {} groups: dict[str, dict] = {}
files: dict[str, set[str]] = {}
for row in rows: for row in rows:
if row.status != "unclassified": if row.status not in _MECHANICAL_TODO:
continue continue
if row.proposed_snippet_id is not None: if row.proposed_snippet_id is not None:
proposed += 1 proposed += 1
by_canon[row.proposed_snippet_id] = by_canon.get(row.proposed_snippet_id, 0) + 1
elif row.proposal_group: elif row.proposal_group:
dup = not row.proposal_group.startswith("name:")
g = groups.setdefault(row.proposal_group, { g = groups.setdefault(row.proposal_group, {
"group": row.proposal_group, "kind": row.kind, "group": row.proposal_group, "kind": row.kind,
"label": ( "label": (
("." if row.kind == "css" else "") + row.symbol f"{row.symbol} (identical body)" if dup
if row.proposal_group.startswith("name:") else ("." if row.kind == "css" else "") + row.symbol
else f"{row.symbol} (identical body)"
), ),
"size": 0, "paths": [], "size": 0, "files": 0, "paths": [],
}) })
g["size"] += 1 g["size"] += 1
files.setdefault(row.proposal_group, set()).add(row.path)
if len(g["paths"]) < 3: if len(g["paths"]) < 3:
g["paths"].append(row.path) g["paths"].append(row.path)
ranked = sorted(groups.values(), key=lambda g: (-g["size"], g["group"])) for key, g in groups.items():
return {"proposed": proposed, "derive_groups": ranked[:top]} g["files"] = len(files[key])
# Body-identical groups first (#2872): the things an audit actually
# consolidated were identical bodies under different names/files; a
# name repeated across modules is usually convention. Within a tier,
# the group spread over more files is the bigger copy.
ranked = sorted(
groups.values(),
key=lambda g: (g["group"].startswith("name:"), -g["files"], -g["size"], g["group"]),
)
top_canon = None
if by_canon:
sid, n = max(by_canon.items(), key=lambda kv: (kv[1], -kv[0]))
top_canon = {"snippet_id": sid, "count": n}
return {"proposed": proposed, "derive_groups": ranked[:top], "top_canon": top_canon}
async def confirm_proposals( async def confirm_proposals(
@@ -1089,7 +1421,7 @@ async def confirm_proposals(
conds = [ conds = [
CodeShape.project_id == project_id, CodeShape.project_id == project_id,
CodeShape.status == "unclassified", CodeShape.status.in_(_MECHANICAL_TODO),
CodeShape.vanished_at.is_(None), CodeShape.vanished_at.is_(None),
CodeShape.proposed_snippet_id.isnot(None), CodeShape.proposed_snippet_id.isnot(None),
] ]
@@ -1246,7 +1578,7 @@ async def flag_divergence(project_id: int, *, since: datetime | None) -> int:
for siblings in by_dir.values(): for siblings in by_dir.values():
dom = dominant_canon(siblings) dom = dominant_canon(siblings)
for r in siblings: for r in siblings:
if r.status != "unclassified": if r.status not in _MECHANICAL_TODO:
continue continue
if r.diverges_from is not None: if r.diverges_from is not None:
flagged += 1 flagged += 1
@@ -1263,7 +1595,7 @@ async def flag_divergence(project_id: int, *, since: datetime | None) -> int:
def divergence_summary(rows: Iterable[CodeShape], *, top: int = 10) -> dict: def divergence_summary(rows: Iterable[CodeShape], *, top: int = 10) -> dict:
"""Readout view: flagged shapes (newest first) and the recheck count.""" """Readout view: flagged shapes (newest first) and the recheck count."""
flagged = [r for r in rows if r.diverges_from is not None and r.status == "unclassified"] flagged = [r for r in rows if r.diverges_from is not None and r.status in _MECHANICAL_TODO]
flagged.sort(key=lambda r: (r.created_at or datetime.min.replace(tzinfo=timezone.utc)), reverse=True) flagged.sort(key=lambda r: (r.created_at or datetime.min.replace(tzinfo=timezone.utc)), reverse=True)
recheck = sum(1 for r in rows if r.recheck_at is not None and r.vanished_at is None) recheck = sum(1 for r in rows if r.recheck_at is not None and r.vanished_at is None)
return { return {
+131 -5
View File
@@ -15,6 +15,7 @@ from scribe.models.project import Project
from scribe.models.user import User from scribe.models.user import User
from scribe.services.shape_ledger import ( from scribe.services.shape_ledger import (
classify_shapes, classify_shapes,
classify_shapes_where,
list_project_shapes, list_project_shapes,
snippet_consumers, snippet_consumers,
sync_repo_shapes, sync_repo_shapes,
@@ -128,6 +129,113 @@ async def test_classification_is_write_gated_and_listing_read_gated(seeded):
assert await list_project_shapes(other, pid) == ([], 0) assert await list_project_shapes(other, pid) == ([], 0)
@pytest.mark.integration
async def test_rule_form_sweeps_unclassified_rows_only_and_applies_whole(seeded):
"""#2868: one judgment over a directory + glob; judged rows are left
alone unless include_judged; the same gates as the row form."""
owner, other, pid, sid = seeded["owner"], seeded["other"], seeded["pid"], seeded["snippet"]
await classify_shapes(owner, pid, [
{"path": "src/app.py", "symbol": "Config", "status": "exempt", "reason": "settings holder"},
])
out = await classify_shapes_where(
owner, pid, path="src", status="instance", snippet_id=sid, via="audit",
)
# make_app + helper swept; Config (already judged) untouched; css not under src/.
assert out["classified"] == 2
assert out["sample"] == ["src/app.py::make_app", "src/util.py::helper"]
rows, _ = await list_project_shapes(owner, pid)
by_symbol = {r.symbol: r for r in rows}
assert by_symbol["make_app"].status == "instance" and by_symbol["make_app"].classified_by == "audit"
assert by_symbol["Config"].status == "exempt" and by_symbol["Config"].reason == "settings holder"
assert by_symbol["btn"].status == "unclassified"
# Glob + kind narrow; include_judged re-judges.
out = await classify_shapes_where(
owner, pid, path="web", status="exempt", pattern="btn*", kind="css",
reason="one toolbar button", include_judged=True,
)
assert out["classified"] == 1
out = await classify_shapes_where(
owner, pid, path="src", status="unclassified", include_judged=True,
)
assert out["classified"] == 3 # withdrawal sweeps judged rows when asked
# Gates: reason for exempt, snippet for instance, write access, a path.
with pytest.raises(ValueError):
await classify_shapes_where(owner, pid, path="src", status="exempt")
with pytest.raises(ValueError):
await classify_shapes_where(owner, pid, path="src", status="instance")
with pytest.raises(ValueError):
await classify_shapes_where(owner, pid, path="", status="exempt", reason="x")
with pytest.raises(ValueError):
await classify_shapes_where(other, pid, path="src", status="exempt", reason="x")
@pytest.mark.integration
async def test_sync_stamps_scoped_rows_and_unstamps_when_they_become_reachable(seeded):
"""#2869: by-construction one-offs arrive `scoped` (mechanical), count as
accounted, are reached by the sweep's default, and go back to the todo
if a later tree makes them ordinary. A judgment overrides the stamp."""
from scribe.services.coverage import ArchiveShape
owner, pid, sid = seeded["owner"], seeded["pid"], seeded["snippet"]
scoped_shapes = [
ArchiveShape("web/Card.vue", "css", "card", ".card {", "s1", ".card { x: 1 }", True),
ArchiveShape("web/Card.vue", "sym", "load", "function load() {", "s2", "function load() {}", True),
ArchiveShape("src/util.py", "sym", "helper", "def helper():", "s3", "def helper(): pass", False),
]
await sync_repo_shapes(pid, REPO, scoped_shapes, seen_marker="main")
rows, _ = await list_project_shapes(owner, pid, path="web/Card.vue")
assert {r.status for r in rows} == {"scoped"}
assert all(r.classified_by == "mechanical" and "by construction" in (r.reason or "") for r in rows)
# The human todo excludes them; the sweep's default still reaches them.
assert (await list_project_shapes(owner, pid, status="unclassified", path="web/Card.vue"))[1] == 0
out = await classify_shapes_where(
owner, pid, path="web/Card.vue", status="instance", snippet_id=sid, kind="css",
)
assert out["classified"] == 1
# Re-synced as ordinary: the stamped sym returns to the todo; the
# judged css keeps its judgment.
plain = [ArchiveShape(s.path, s.kind, s.name, s.signature, s.body_sha, s.body, False) for s in scoped_shapes]
await sync_repo_shapes(pid, REPO, plain, seen_marker="main")
rows, _ = await list_project_shapes(owner, pid, path="web/Card.vue")
by_symbol = {r.symbol: r for r in rows}
assert by_symbol["load"].status == "unclassified" and by_symbol["load"].classified_by is None
assert by_symbol["card"].status == "instance" and by_symbol["card"].snippet_id == sid
@pytest.mark.integration
async def test_uses_edges_are_the_consumer_map(seeded):
"""#2870: a shape keeps ONE snippet_id (what it is) and any number of
uses edges (what it calls); the snippet's consumer map lists them,
list_shapes(uses=N) finds them, and a sweep can write them."""
from scribe.services import snippets as snippets_svc
owner, pid, sid = seeded["owner"], seeded["pid"], seeded["snippet"]
helper = await snippets_svc.create_snippet(
owner, name="cls_hash_helper", code="def hash_token(raw):\n return raw\n",
language="python", repo="Widget", path="src/hash.py", symbol="hash_token",
project_id=pid,
)
hid = int(helper.id)
out = await classify_shapes(owner, pid, [
{"path": "src/app.py", "symbol": "make_app", "status": "instance",
"snippet_id": sid, "uses": [hid]},
], via="audit")
assert out["classified"] == 1
rows, total = await list_project_shapes(owner, pid, uses=hid)
assert total == 1 and rows[0].symbol == "make_app" and rows[0].snippet_id == sid
consumers = await snippet_consumers(owner, hid)
assert consumers["instances"] == [] and len(consumers["uses"]) == 1
assert consumers["uses"][0]["symbol"] == "make_app" and consumers["uses"][0]["basis"] == "audit"
# A sweep writes uses too; an unknown snippet in uses applies nothing.
out = await classify_shapes_where(
owner, pid, path="src/util.py", status="exempt", reason="local", uses=[hid],
)
assert out["classified"] == 1
assert (await list_project_shapes(owner, pid, uses=hid))[1] == 2
with pytest.raises(ValueError):
await classify_shapes(owner, pid, [
{"path": "src/app.py", "symbol": "Config", "status": "exempt", "reason": "x", "uses": [999999]},
])
@pytest.mark.integration @pytest.mark.integration
async def test_list_filters_compose(seeded): async def test_list_filters_compose(seeded):
owner, pid, sid = seeded["owner"], seeded["pid"], seeded["snippet"] owner, pid, sid = seeded["owner"], seeded["pid"], seeded["snippet"]
@@ -170,7 +278,7 @@ async def test_get_snippet_carries_the_structured_consumer_map(seeded):
# The map is caller-scoped: an outsider asking the service directly gets # The map is caller-scoped: an outsider asking the service directly gets
# silence, not another project's file layout. # silence, not another project's file layout.
consumers = await snippet_consumers(seeded["other"], sid) consumers = await snippet_consumers(seeded["other"], sid)
assert consumers == {"instances": [], "variants": []} assert consumers == {"instances": [], "variants": [], "uses": []}
@pytest.mark.integration @pytest.mark.integration
@@ -458,9 +566,14 @@ async def test_derive_groups_land_on_rows_and_in_the_summary(seeded):
summary = proposal_summary(await live_rows(pid)) summary = proposal_summary(await live_rows(pid))
assert summary["proposed"] == 1 assert summary["proposed"] == 1
assert [g["group"] for g in summary["derive_groups"]][0] == "name:css:card" # #2872: the body-identical group (a real copy) outranks the bigger name
assert summary["derive_groups"][0]["label"] == ".card" # group (usually convention), even at size 2 vs 3.
assert summary["derive_groups"][0]["size"] == 3 order = [g["group"] for g in summary["derive_groups"]]
assert order[0].startswith("dup:") and order[1] == "name:css:card"
assert summary["derive_groups"][0]["files"] == 2
assert summary["derive_groups"][0]["label"] == "slug (identical body)"
assert summary["derive_groups"][1]["label"] == ".card"
assert summary["derive_groups"][0]["size"] == 2 and summary["derive_groups"][1]["size"] == 3
# One of the css copies gets judged → the group shrinks on the next pass. # One of the css copies gets judged → the group shrinks on the next pass.
await classify_shapes(owner, pid, [ await classify_shapes(owner, pid, [
@@ -489,7 +602,20 @@ async def test_a_second_confirm_dialog_is_detected_and_named(seeded):
write_time_divergence, write_time_divergence,
) )
owner, pid, sid = seeded["owner"], seeded["pid"], seeded["snippet"] from scribe.services import snippets as snippets_svc
owner, pid = seeded["owner"], seeded["pid"]
# The confirm helper is TS canon: since #2871 a sym basis only proposes
# within the shape's language family, so the fixture's Python snippet
# says nothing about these Vue bodies — the canon must be one of theirs.
canon = await snippets_svc.create_snippet(
owner, name="cls_confirm_factory",
code="export async function factory(): Promise<boolean> {\n return true;\n}\n",
language="typescript", repo="Widget",
path="frontend/src/composables/useConfirm.ts", symbol="factory",
project_id=pid,
)
sid = int(canon.id)
comp = "frontend/src/components" comp = "frontend/src/components"
base = _defs( base = _defs(
*[(f"{comp}/{n}.vue", "sym", f"on{n}", f"async function on{n}() {{", *[(f"{comp}/{n}.vue", "sym", f"on{n}", f"async function on{n}() {{",
+70 -2
View File
@@ -168,6 +168,13 @@ def test_coverage_line_is_evidence_carrying_and_labeled_estimate():
assert "internal/api, web/src/components" in line assert "internal/api, web/src/components" in line
def test_bind_repo_tool_takes_a_ref():
"""#2873: the binding names the branch the ledger follows."""
from scribe.mcp.server import build_mcp_server
tool = build_mcp_server()._tool_manager.get_tool("bind_repo")
assert "ref" in tool.parameters.get("properties", {})
def test_coverage_routes_are_registered(): def test_coverage_routes_are_registered():
from scribe.app import create_app from scribe.app import create_app
@@ -188,7 +195,8 @@ def _forge(tar_bytes: bytes):
path = request.url.path path = request.url.path
if path == "/api/v1/repos/alice/widget": if path == "/api/v1/repos/alice/widget":
return httpx.Response(200, json={"default_branch": "main"}) return httpx.Response(200, json={"default_branch": "main"})
if path == "/api/v1/repos/alice/widget/archive/main.tar.gz": if path in ("/api/v1/repos/alice/widget/archive/main.tar.gz",
"/api/v1/repos/alice/widget/archive/dev.tar.gz"):
return httpx.Response(200, content=tar_bytes) return httpx.Response(200, content=tar_bytes)
return httpx.Response(404, json={"message": "not found"}) return httpx.Response(404, json={"message": "not found"})
@@ -258,7 +266,7 @@ async def test_coverage_measures_the_tree_exactly_and_caches(seeded):
assert coverage["accounted"] == 2 assert coverage["accounted"] == 2
assert coverage["unclassified"] == 2 assert coverage["unclassified"] == 2
assert coverage["counts"] == { assert coverage["counts"] == {
"canonical": 2, "instance": 0, "variant": 0, "exempt": 0, "canonical": 2, "instance": 0, "variant": 0, "exempt": 0, "scoped": 0,
} }
assert coverage["estimate"] is True assert coverage["estimate"] is True
assert coverage["repos"] == [{ assert coverage["repos"] == [{
@@ -468,6 +476,11 @@ def test_extract_definitions_fingerprints_each_block():
# And the identity view is unchanged for the hook mirror. # And the identity view is unchanged for the hook mirror.
from scribe.services.coverage import extract_shapes from scribe.services.coverage import extract_shapes
assert extract_shapes(text) == [(d.kind, d.name) for d in extract_definitions(text)] assert extract_shapes(text) == [(d.kind, d.name) for d in extract_definitions(text)]
# A CSS rule's fingerprint is its declarations (#2872): the same body
# under another selector is the same shape to the derive grouping.
css = ".closed-msg {\n text-align: center;\n padding: 0.5rem 0;\n}\n.error-block {\n text-align: center;\n padding: 0.5rem 0;\n}\n.other {\n text-align: left;\n}\n"
d = {x.name: x for x in extract_definitions(css)}
assert d["closed-msg"].body_sha == d["error-block"].body_sha != d["other"].body_sha
def test_coverage_line_names_the_proposers_standing(): def test_coverage_line_names_the_proposers_standing():
@@ -484,6 +497,12 @@ def test_coverage_line_names_the_proposers_standing():
assert "; 90 unclassified (40 proposed, 2 derive groups), largest: src" in line assert "; 90 unclassified (40 proposed, 2 derive groups), largest: src" in line
line = coverage_line({**base, "proposed": 0, "derive_groups": [{"group": "a"}]}) line = coverage_line({**base, "proposed": 0, "derive_groups": [{"group": "a"}]})
assert "(1 derive group)" in line assert "(1 derive group)" in line
# #2874: the next action on the line — biggest canon queue, widest copy.
line = coverage_line({
**base, "proposed": 40, "top_canon": {"snippet_id": 2844, "count": 78},
"derive_groups": [{"group": "dup:abc", "label": "closed-msg (identical body)", "files": 3}],
})
assert "top canon #2844 ×78" in line and "top copy closed-msg (identical body) ×3 files" in line
def test_coverage_line_names_divergence_and_recheck(): def test_coverage_line_names_divergence_and_recheck():
@@ -500,3 +519,52 @@ def test_coverage_line_names_divergence_and_recheck():
assert line.endswith("; 1 judged shape changed since judged — recheck") assert line.endswith("; 1 judged shape changed since judged — recheck")
assert "DIVERGENT" not in coverage_line(base) assert "DIVERGENT" not in coverage_line(base)
assert "recheck" not in coverage_line(base) assert "recheck" not in coverage_line(base)
def test_scoped_definitions_are_vue_script_setup_and_scoped_style_only():
"""#2869: one-offs by construction — every sym in a .vue and every css
rule inside <style scoped>; an unscoped <style> block and non-.vue files
stay ordinary."""
from scribe.services.coverage import extract_definitions, scoped_definitions
vue = (
"<script setup lang=\"ts\">\n"
"function load() {\n return 1;\n}\n"
"const save = async () => {\n return 2;\n};\n"
"</script>\n\n"
"<template><div class=\"card\"/></template>\n\n"
"<style scoped>\n.card {\n padding: 1rem;\n}\n.title {\n margin: 0;\n}\n</style>\n"
"<style>\n.global-toast {\n color: red;\n}\n</style>\n"
)
defs = extract_definitions(vue)
names = {(d.kind, d.name) for d in defs}
assert {("sym", "load"), ("sym", "save"), ("css", "card"), ("css", "title"), ("css", "global-toast")} <= names
scoped = scoped_definitions("frontend/src/views/A.vue", vue, defs)
assert scoped == {("sym", "load"), ("sym", "save"), ("css", "card"), ("css", "title")}
# Definitions know their line, which is what the scoped-style range uses.
assert next(d for d in defs if d.name == "card").line > next(d for d in defs if d.name == "save").line
# Not a .vue: nothing is scoped, whatever it contains.
assert scoped_definitions("frontend/src/assets/components.css", ".card {\n x: 1;\n}\n",
extract_definitions(".card {\n x: 1;\n}\n")) == set()
assert scoped_definitions("src/a.py", "def load():\n pass\n", extract_definitions("def load():\n pass\n")) == set()
@pytest.mark.integration
async def test_binding_ref_is_the_branch_the_ledger_follows(seeded):
"""#2873: a binding that names a ref is read at that ref (not the forge's
default branch); "" clears it; None on a re-bind leaves it standing."""
from scribe.services.coverage import compute_coverage
from scribe.services.repo_bindings import bindings_for_project, set_binding
uid, pid = seeded["uid"], seeded["pid"]
b = await set_binding(uid, "https://git.example.com/alice/widget.git", pid, "dev")
assert b.ref == "dev"
coverage = await compute_coverage(uid, pid, selector=_selector(_tarball(TREE)))
assert coverage["repos"][0]["ref"] == "dev"
# A re-bind without a ref keeps it; "" clears it back to the default branch.
b = await set_binding(uid, "https://git.example.com/alice/widget.git", pid)
assert b.ref == "dev"
b = await set_binding(uid, "https://git.example.com/alice/widget.git", pid, "")
assert b.ref is None
assert [x.ref for x in await bindings_for_project(uid, pid)] == [None]
coverage = await compute_coverage(uid, pid, selector=_selector(_tarball(TREE)))
assert coverage["repos"][0]["ref"] == "main"
+3 -2
View File
@@ -18,7 +18,7 @@ def test_backup_version_is_v8():
point of the test — a payload section added without moving the version point of the test — a payload section added without moving the version
produces backups that are structurally different and indistinguishable produces backups that are structurally different and indistinguishable
by inspection.""" by inspection."""
assert backup.BACKUP_VERSION == 8 assert backup.BACKUP_VERSION == 9
def test_not_included_lists_the_known_gaps(): def test_not_included_lists_the_known_gaps():
@@ -115,7 +115,8 @@ async def test_export_full_backup_contains_every_declared_section():
"topic_suppressions", "topic_suppressions",
"systems", "record_systems", "design_systems", "systems", "record_systems", "design_systems",
"design_tokens", "note_usage_events", "repo_bindings", "design_tokens", "note_usage_events", "repo_bindings",
"note_supersessions", "code_shapes", "code_shape_events"): "note_supersessions", "code_shapes", "code_shape_events",
"code_shape_uses"):
assert key in out, f"missing export section: {key}" assert key in out, f"missing export section: {key}"
assert out[key] == [] assert out[key] == []
+163 -1
View File
@@ -30,7 +30,7 @@ def test_the_todo_state_is_the_default():
assert CodeShape.__table__.c.status.default.arg == "unclassified" assert CodeShape.__table__.c.status.default.arg == "unclassified"
assert "unclassified" in SHAPE_STATUSES assert "unclassified" in SHAPE_STATUSES
assert set(SHAPE_STATUSES) == { assert set(SHAPE_STATUSES) == {
"canonical", "instance", "variant", "exempt", "unclassified", "canonical", "instance", "variant", "exempt", "scoped", "unclassified",
} }
assert set(SHAPE_CLASSIFIERS) == { assert set(SHAPE_CLASSIFIERS) == {
"agent", "audit", "hook", "mechanical", "import", "agent", "audit", "hook", "mechanical", "import",
@@ -248,6 +248,49 @@ def test_match_canon_orders_bases_strongest_first_and_respects_kind():
assert match_canon("sym", "x.py", "unrelated", "def unrelated(a, b, c, d, e):", "return 1", canons) is None assert match_canon("sym", "x.py", "unrelated", "def unrelated(a, b, c, d, e):", "return 1", canons) is None
def test_match_canon_gates_sym_bases_by_language_family():
"""A Python canon says nothing about a Vue body (and vice versa): the
2026-08 audit's worst proposals were `register` (MCP tool module, python)
offered for every auth view's handleSubmit that calls authStore.register()
and for a TS store's own `register`. Unknown language on either side →
no gate (the canons recorded without a language keep proposing)."""
from scribe.services.shape_ledger import Canon, _norm_text, match_canon, same_family
py_register = Canon(46, "sym", "register", (("src/scribe/mcp/tools/notes.py", "register"),),
"def register(mcp) -> None:", _norm_text("def register(mcp) -> None: ..."), 2, "python")
ts_helper = Canon(53, "sym", "apiErrorMessage", (("frontend/src/api/client.ts", "apiErrorMessage"),),
"export function apiErrorMessage(e: unknown, fallback: string): string {",
_norm_text("export function apiErrorMessage(e, fallback) { return fallback }"), 2, "typescript")
canons = [py_register, ts_helper]
vue_body = "async function handleSubmit() {\n await authStore.register(username.value);\n error.value = apiErrorMessage(e, 'x');\n}"
# The Vue handler references the TS helper, never the Python canon.
assert match_canon("sym", "frontend/src/views/RegisterView.vue", "handleSubmit",
"async function handleSubmit() {", vue_body, canons) == (53, "reference", 0.9)
# A TS store's own `register` is not a second definition of the Python one.
assert match_canon("sym", "frontend/src/stores/auth.ts", "register",
"async function register(u: string) {", "return apiPost('/api/auth/register', {u})",
[py_register]) is None
# Same family still proposes by symbol; unknown language still proposes.
assert match_canon("sym", "src/scribe/mcp/tools/other.py", "register",
"def register(mcp) -> None:", "pass", [py_register]) == (46, "symbol", 1.0)
unknown = py_register._replace(language="")
assert match_canon("sym", "frontend/src/stores/auth.ts", "register",
"async function register(u: string) {", "", [unknown]) == (46, "symbol", 1.0)
assert same_family("a.py", "python") and same_family("a.vue", "typescript")
assert same_family("a.py", "") and same_family("", "python")
assert not same_family("a.py", "vue")
def test_match_canon_reference_skips_generic_verbs():
"""A bare mention of `load`/`save`/`register` is not a call site of THIS
canon; the symbol basis still catches a second definition of the name."""
from scribe.services.shape_ledger import Canon, _norm_text, match_canon
loader = Canon(70, "sym", "load", (("frontend/src/components/A.vue", "load"),),
"async function load() {", _norm_text("async function load() { await fetch() }"), 2, "vue")
body = "async function refresh() {\n await load();\n}"
assert match_canon("sym", "frontend/src/components/B.vue", "refresh", "async function refresh() {", body, [loader]) is None
assert match_canon("sym", "frontend/src/components/B.vue", "load", "async function load() {", "", [loader]) == (70, "symbol", 1.0)
def test_match_canon_symbol_beats_everything_including_css_copies(): def test_match_canon_symbol_beats_everything_including_css_copies():
"""The previous test's css `btn-primary`-elsewhere case, stated plainly: """The previous test's css `btn-primary`-elsewhere case, stated plainly:
a second definition of the canon's own name is the symbol basis.""" a second definition of the canon's own name is the symbol basis."""
@@ -274,6 +317,32 @@ def test_derive_groups_copy_before_name_with_floors():
assert ("i.py", "sym", "one") not in g assert ("i.py", "sym", "one") not in g
def test_proposal_summary_ranks_body_identical_groups_first_and_sees_scoped_rows():
"""#2872: dup groups (the real copies) outrank name groups (usually
convention), wider spread first; #2869: scoped rows are in the readout."""
from scribe.models.code_shape import CodeShape
from scribe.services.shape_ledger import proposal_summary
def row(path, symbol, group, kind="css", status="scoped"):
return CodeShape(project_id=2, repo_key="r", path=path, symbol=symbol, kind=kind,
status=status, proposal_basis="derive", proposal_group=group)
rows = [
# a name group of 6 across 6 files
*[row(f"v/{i}.vue", "status-badge", "name:css:status-badge") for i in range(6)],
# a dup group of 3 across 3 files (different selector names, one body)
row("v/Login.vue", "closed-msg", "dup:abc"), row("v/Reset.vue", "error-block", "dup:abc"),
row("v/Forgot.vue", "success-msg", "dup:abc"),
# a dup group of 2 in ONE file — a copy, but not across files
row("v/A.vue", "x", "dup:def"), row("v/A.vue", "y", "dup:def"),
# judged rows never count
row("v/J.vue", "closed-msg", "dup:abc", status="exempt"),
]
out = proposal_summary(rows)
assert [g["group"] for g in out["derive_groups"]] == ["dup:abc", "dup:def", "name:css:status-badge"]
assert out["derive_groups"][0]["files"] == 3 and out["derive_groups"][0]["size"] == 3
assert out["derive_groups"][0]["label"] == "closed-msg (identical body)"
def test_confirm_requires_a_named_scope(): def test_confirm_requires_a_named_scope():
import asyncio import asyncio
@@ -291,6 +360,99 @@ def test_proposer_tools_are_mounted():
assert mcp._tool_manager.get_tool("confirm_shape_proposals") is not None assert mcp._tool_manager.get_tool("confirm_shape_proposals") is not None
tool = mcp._tool_manager.get_tool("list_shapes") tool = mcp._tool_manager.get_tool("list_shapes")
assert "proposal" in tool.parameters.get("properties", {}) assert "proposal" in tool.parameters.get("properties", {})
# #2868: the audit surfaces — compact pages and the sweep form.
assert "compact" in tool.parameters.get("properties", {})
rule = mcp._tool_manager.get_tool("classify_shapes_by_rule")
assert rule is not None
for name in ("path", "status", "pattern", "kind", "snippet_id", "reason", "include_judged"):
assert name in rule.parameters.get("properties", {}), name
# --- #2868: the bulk surfaces (pure) -----------------------------------------
def test_compact_row_carries_identity_standing_and_the_proposers_word_only():
"""A 500-row compact page must fit the tool budget: no commits, shas or
timestamps; optional fields only when set."""
from scribe.models.code_shape import CodeShape
row = CodeShape(project_id=2, repo_key="r", path="src/a.py", symbol="f", kind="sym",
status="unclassified", signature="def f(x):", body_sha="abc",
first_seen_commit="c1", last_seen_commit="c2")
assert row.to_compact() == {
"path": "src/a.py", "symbol": "f", "kind": "sym",
"status": "unclassified", "signature": "def f(x):",
}
row.status, row.snippet_id, row.classified_by = "instance", 9, "audit"
row.proposed_snippet_id, row.proposal_basis, row.proposal_score = 9, "symbol", 1.0
compact = row.to_compact()
assert compact["snippet_id"] == 9 and compact["by"] == "audit"
assert compact["proposal"]["basis"] == "symbol"
for noisy in ("first_seen_commit", "last_seen_commit", "body_sha", "created_at", "classified_at"):
assert noisy not in compact
def test_uses_edges_table_and_validation():
"""#2870: consumption is its own relation — a table that cascades with
both ends, and `uses` on a classification must be a list of ids."""
from scribe.models import Base
from scribe.models.code_shape import USE_BASES, CodeShapeUse
from scribe.services.shape_ledger import validate_classifications
assert "code_shape_uses" in Base.metadata.tables
cols = CodeShapeUse.__table__.c
assert next(iter(cols.shape_id.foreign_keys)).ondelete == "CASCADE"
assert next(iter(cols.snippet_id.foreign_keys)).ondelete == "CASCADE"
assert set(USE_BASES) == {"reference", "hook", "agent", "audit", "import"}
ok = [{"path": "a.py", "symbol": "f", "status": "instance", "snippet_id": 9, "uses": [3, 4]}]
assert validate_classifications(ok) is None
bad = [{"path": "a.py", "symbol": "f", "status": "instance", "snippet_id": 9, "uses": "3"}]
assert "uses must be a list" in validate_classifications(bad)
def test_reference_canons_names_every_used_canon_not_just_the_best():
from scribe.services.shape_ledger import Canon, _norm_text, reference_canons
a = Canon(1, "sym", "hash_token", (("src/x.py", "hash_token"),), "def hash_token(raw):", _norm_text("x"), 2, "python")
b = Canon(2, "sym", "rules_payload", (("src/y.py", "rules_payload"),), "def rules_payload(r):", _norm_text("y"), 2, "python")
ts = Canon(3, "sym", "fmtDate", (("f/d.ts", "fmtDate"),), "export function fmtDate(iso: string): string {", _norm_text("z"), 2, "typescript")
body = "def create_invitation(email):\n h = hash_token(raw)\n return rules_payload(h)\n"
assert reference_canons("sym", "src/scribe/services/auth.py", "create_invitation", body, [a, b, ts]) == [1, 2]
# the shape's own name and the other language family are never "uses"
assert reference_canons("sym", "src/x.py", "hash_token", body, [a]) == []
assert reference_canons("sym", "f/v.vue", "show", "fmtDate(x); hash_token(y)", [a, ts]) == [3]
def test_reason_codes_are_a_fixed_catalogue_and_validated():
"""#2874: an optional index beside the prose reason; unknown codes are a
structural error (the batch applies nothing)."""
from scribe.models.code_shape import REASON_CODES
from scribe.services.shape_ledger import validate_classifications
assert set(REASON_CODES) == {
"scoped-css", "one-off-handler", "test-helper", "convention-plumbing",
"pure-helper", "generated", "script", "typed-record",
}
assert "reason_code" in CodeShape.__table__.c
ok = [{"path": "a.py", "symbol": "f", "status": "exempt", "reason": "x", "reason_code": "pure-helper"}]
assert validate_classifications(ok) is None
bad = [{"path": "a.py", "symbol": "f", "status": "exempt", "reason": "x", "reason_code": "nope"}]
assert "unknown reason_code" in validate_classifications(bad)
def test_rule_matches_is_directory_glob_and_kind_aware():
from scribe.models.code_shape import CodeShape
from scribe.services.shape_ledger import rule_matches
def row(path, symbol, kind="sym"):
return CodeShape(project_id=2, repo_key="r", path=path, symbol=symbol, kind=kind, status="unclassified")
r = row("frontend/src/views/LoginView.vue", "auth-card", "css")
assert rule_matches(r, path="frontend/src/views", pattern="", kind="")
assert rule_matches(r, path="frontend/src/views", pattern="auth-*", kind="css")
assert not rule_matches(r, path="frontend/src/views", pattern="auth-*", kind="sym")
assert not rule_matches(r, path="frontend/src/view", pattern="", kind="") # directory, not prefix
assert rule_matches(r, path="frontend/src/views/LoginView.vue", pattern="", kind="")
# CSS symbols compare without the leading dot, like everywhere else.
assert rule_matches(row("w/a.css", ".btn-primary", "css"), path="w", pattern="btn-*", kind="css")
assert rule_matches(row("src/scribe/services/backup.py", "_note_rows"), path="src/scribe/services", pattern="_*_rows", kind="")
assert not rule_matches(row("src/scribe/services/backup.py", "export_full_backup"), path="src/scribe/services", pattern="_*_rows", kind="")
# --- step 7: the divergence readout (pure) ---------------------------------- # --- step 7: the divergence readout (pure) ----------------------------------
+5 -2
View File
@@ -56,6 +56,7 @@ def test_no_parts_matches_everything():
def test_matches_exact_repo_path_and_symbol(): def test_matches_exact_repo_path_and_symbol():
data = _data({"repo": "Scribe", "path": "src/scribe/x.py", "symbol": "helper"}) data = _data({"repo": "Scribe", "path": "src/scribe/x.py", "symbol": "helper"})
assert location_matches(data, {"repo": "Scribe"}) assert location_matches(data, {"repo": "Scribe"})
assert location_matches(data, {"repo": "scribe"}) # repo names: case never distinguishes (#2874)
assert location_matches(data, {"path": "src/scribe/x.py"}) assert location_matches(data, {"path": "src/scribe/x.py"})
assert location_matches(data, {"symbol": "helper"}) assert location_matches(data, {"symbol": "helper"})
assert location_matches(data, {"repo": "Scribe", "symbol": "helper"}) assert location_matches(data, {"repo": "Scribe", "symbol": "helper"})
@@ -101,7 +102,8 @@ def test_blank_recorded_part_does_not_match_a_requested_one():
def test_jsonpath_filters_within_one_locations_entry(): def test_jsonpath_filters_within_one_locations_entry():
expr = location_jsonpath({"repo": "Scribe", "symbol": "helper"}) expr = location_jsonpath({"repo": "Scribe", "symbol": "helper"})
assert expr.startswith("$.locations[*] ? (") assert expr.startswith("$.locations[*] ? (")
assert '@.repo == "Scribe"' in expr # repo: anchored, case-insensitive (#2874) — mirrors location_matches.
assert '@.repo like_regex "^Scribe$" flag "i"' in expr
assert '@.symbol == "helper"' in expr assert '@.symbol == "helper"' in expr
assert " && " in expr assert " && " in expr
@@ -121,8 +123,9 @@ def test_jsonpath_quotes_values_as_json_literals():
"""A quote in a repo name must stay inside the literal, not end it.""" """A quote in a repo name must stay inside the literal, not end it."""
nasty = 'we"ird' nasty = 'we"ird'
expr = location_jsonpath({"repo": nasty}) expr = location_jsonpath({"repo": nasty})
assert json.dumps(nasty) in expr assert json.dumps("^" + nasty + "$") in expr # the regex is a JSON literal too
assert '\\"' in expr assert '\\"' in expr
assert json.dumps(nasty) in location_jsonpath({"symbol": nasty})
def test_jsonpath_emits_parts_in_a_fixed_key_order(): def test_jsonpath_emits_parts_in_a_fixed_key_order():