Merge pull request 'Ledger follow-ups from the shape audit (milestone 294: #2868–#2874)' (#122) from dev into main
CI & Build / Build & push image (push) Successful in 15s
CI & Build / TypeScript typecheck (push) Successful in 23s
CI & Build / Plugin hooks (push) Successful in 8s
CI & Build / Python lint (push) Successful in 3s
CI & Build / Python tests (push) Successful in 57s
CI & Build / integration (push) Successful in 21s
CI & Build / Build & push image (push) Successful in 15s
CI & Build / TypeScript typecheck (push) Successful in 23s
CI & Build / Plugin hooks (push) Successful in 8s
CI & Build / Python lint (push) Successful in 3s
CI & Build / Python tests (push) Successful in 57s
CI & Build / integration (push) Successful in 21s
This commit was merged in pull request #122.
This commit is contained in:
@@ -0,0 +1,26 @@
|
|||||||
|
"""Per-binding ref — the branch a project's ledger follows (#2873, milestone 294)
|
||||||
|
|
||||||
|
Revision ID: 0082
|
||||||
|
Revises: 0081
|
||||||
|
Create Date: 2026-08-21
|
||||||
|
|
||||||
|
A repo binding used to imply the repo's default branch; the shape ledger
|
||||||
|
therefore only saw work after a merge to main, while the operator's work
|
||||||
|
lands on dev (rule 1). `ref` names the branch the coverage refresh reads —
|
||||||
|
NULL keeps today's behaviour (the forge's default branch).
|
||||||
|
"""
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision = "0082"
|
||||||
|
down_revision = "0081"
|
||||||
|
branch_labels = None
|
||||||
|
depends_on = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column("repo_bindings", sa.Column("ref", sa.Text(), nullable=True))
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("repo_bindings", "ref")
|
||||||
@@ -0,0 +1,26 @@
|
|||||||
|
"""Exempt/variant reason codes — a small fixed catalogue beside the prose (#2874, milestone 294)
|
||||||
|
|
||||||
|
Revision ID: 0083
|
||||||
|
Revises: 0082
|
||||||
|
Create Date: 2026-08-21
|
||||||
|
|
||||||
|
The 2026-08 audit wrote the same free-text reason thousands of times
|
||||||
|
("scoped rule — styles one element of this view"); a judgment's WHY stays
|
||||||
|
prose, but an optional code from a fixed catalogue makes the ledger
|
||||||
|
filterable and aggregable ("how many pure helpers, how many test helpers").
|
||||||
|
"""
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision = "0083"
|
||||||
|
down_revision = "0082"
|
||||||
|
branch_labels = None
|
||||||
|
depends_on = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column("code_shapes", sa.Column("reason_code", sa.Text(), nullable=True))
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("code_shapes", "reason_code")
|
||||||
@@ -0,0 +1,40 @@
|
|||||||
|
"""code_shape_uses — consumption edges, separate from conformance (#2870, milestone 294)
|
||||||
|
|
||||||
|
Revision ID: 0084
|
||||||
|
Revises: 0083
|
||||||
|
Create Date: 2026-08-21
|
||||||
|
|
||||||
|
A ledger row carries ONE snippet_id: what shape this is (instance/variant of
|
||||||
|
a canon). But a shape can also CALL several canonical helpers — a service
|
||||||
|
function that is an instance of the service-function convention and a
|
||||||
|
consumer of hash_token. The 2026-08 audit had to pick one; hook evidence
|
||||||
|
("pulled #N then wrote code referencing it") was stamped as instance when it
|
||||||
|
is a uses fact. This table holds the many-valued relation: shape → snippet,
|
||||||
|
with the basis and the evidence. Cascades with the shape and the snippet.
|
||||||
|
"""
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision = "0084"
|
||||||
|
down_revision = "0083"
|
||||||
|
branch_labels = None
|
||||||
|
depends_on = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"code_shape_uses",
|
||||||
|
sa.Column("id", sa.Integer(), primary_key=True),
|
||||||
|
sa.Column("shape_id", sa.Integer(), sa.ForeignKey("code_shapes.id", ondelete="CASCADE"), nullable=False),
|
||||||
|
sa.Column("snippet_id", sa.Integer(), sa.ForeignKey("notes.id", ondelete="CASCADE"), nullable=False),
|
||||||
|
sa.Column("basis", sa.Text(), nullable=False),
|
||||||
|
sa.Column("evidence", sa.Text(), nullable=True),
|
||||||
|
sa.Column("created_at", sa.DateTime(timezone=True), nullable=False, server_default=sa.text("now()")),
|
||||||
|
sa.UniqueConstraint("shape_id", "snippet_id", name="uq_code_shape_uses_shape_snippet"),
|
||||||
|
)
|
||||||
|
op.create_index("ix_code_shape_uses_snippet", "code_shape_uses", ["snippet_id"])
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_index("ix_code_shape_uses_snippet", table_name="code_shape_uses")
|
||||||
|
op.drop_table("code_shape_uses")
|
||||||
@@ -754,7 +754,7 @@ async function confirmDelete() {
|
|||||||
</div>
|
</div>
|
||||||
<div v-if="coverage.counts" class="coverage-gaps">
|
<div v-if="coverage.counts" class="coverage-gaps">
|
||||||
<span
|
<span
|
||||||
v-for="k in ['canonical', 'instance', 'variant', 'exempt']"
|
v-for="k in ['canonical', 'instance', 'variant', 'exempt', 'scoped']"
|
||||||
:key="k"
|
:key="k"
|
||||||
>
|
>
|
||||||
<span v-if="coverage.counts[k]" class="coverage-gap-chip">
|
<span v-if="coverage.counts[k]" class="coverage-gap-chip">
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"name": "scribe",
|
"name": "scribe",
|
||||||
"description": "Scribe system-of-record for Claude Code: MCP tools over your notes/tasks/projects/rules, a session-start push channel that surfaces your always-on rules + active-project context, process-skills (writing-plans, systematic-debugging, verification, brainstorming, reusing-code), and your saved Scribe Processes auto-surfaced as skills (/scribe:sync). Replaces superpowers + file-memory with one app-backed plugin.",
|
"description": "Scribe system-of-record for Claude Code: MCP tools over your notes/tasks/projects/rules, a session-start push channel that surfaces your always-on rules + active-project context, process-skills (writing-plans, systematic-debugging, verification, brainstorming, reusing-code), and your saved Scribe Processes auto-surfaced as skills (/scribe:sync). Replaces superpowers + file-memory with one app-backed plugin.",
|
||||||
"version": "0.1.36",
|
"version": "0.1.37",
|
||||||
"author": { "name": "Bryan Van Deusen" },
|
"author": { "name": "Bryan Van Deusen" },
|
||||||
"mcpServers": {
|
"mcpServers": {
|
||||||
"scribe": {
|
"scribe": {
|
||||||
|
|||||||
@@ -17,6 +17,11 @@ row carries a status:
|
|||||||
— the why IS the record.
|
— the why IS the record.
|
||||||
- `exempt` — judged genuinely one-off. **Reason required.** A recorded
|
- `exempt` — judged genuinely one-off. **Reason required.** A recorded
|
||||||
judgment, not silence — it stops the next pass re-litigating it.
|
judgment, not silence — it stops the next pass re-litigating it.
|
||||||
|
- `scoped` — one-off **by construction**, stamped by the coverage sync
|
||||||
|
(a Vue component's scoped `<style>` rules and its `<script setup>`
|
||||||
|
functions — unreachable from any other file). Accounted for without a
|
||||||
|
judgment; still proposed against, grouped and flagged; any judgment you
|
||||||
|
make overrides it. Not the todo.
|
||||||
- `unclassified` — nobody has judged it yet. **This is the todo list.**
|
- `unclassified` — nobody has judged it yet. **This is the todo list.**
|
||||||
|
|
||||||
## The loop
|
## The loop
|
||||||
|
|||||||
@@ -13,28 +13,43 @@ from scribe.services import projects as projects_svc
|
|||||||
from scribe.services import repo_bindings as repo_bindings_svc
|
from scribe.services import repo_bindings as repo_bindings_svc
|
||||||
|
|
||||||
|
|
||||||
async def bind_repo(repo_url: str, project_id: int) -> dict:
|
async def bind_repo(repo_url: str, project_id: int, ref: str = "") -> dict:
|
||||||
"""Bind a git repository to a Scribe project for session-start context.
|
"""Bind a git repository to a Scribe project for session-start context.
|
||||||
|
|
||||||
After this, any session started in that repo auto-loads the project's
|
After this, any session started in that repo auto-loads the project's
|
||||||
context (the SessionStart hook sends the repo's remote; the server resolves
|
context (the SessionStart hook sends the repo's remote; the server resolves
|
||||||
it here). Idempotent — re-binding the same repo updates the target project.
|
it here). Idempotent — re-binding the same repo updates the target project.
|
||||||
|
|
||||||
|
The binding is also what the shape ledger reads (refresh_pattern_coverage):
|
||||||
|
`ref` names the branch it follows. Default (""): the repo's default branch
|
||||||
|
— which means the ledger only sees work after a merge. A dev-first project
|
||||||
|
(rule 1: dev is home) should bind with ref="dev" so classification follows
|
||||||
|
the push, not the merge. Re-binding with ref="" keeps the standing ref;
|
||||||
|
pass ref="-" to clear it back to the default branch.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
repo_url: the repo's git remote (e.g. the output of
|
repo_url: the repo's git remote (e.g. the output of
|
||||||
`git remote get-url origin` — ssh or https form, both work).
|
`git remote get-url origin` — ssh or https form, both work).
|
||||||
project_id: the Scribe project this repo represents.
|
project_id: the Scribe project this repo represents.
|
||||||
|
ref: branch the ledger follows ("" = leave as is / default branch on
|
||||||
|
a new binding; "-" = clear to the default branch).
|
||||||
"""
|
"""
|
||||||
uid = current_user_id()
|
uid = current_user_id()
|
||||||
project = await projects_svc.get_project(uid, project_id)
|
project = await projects_svc.get_project(uid, project_id)
|
||||||
if project is None:
|
if project is None:
|
||||||
raise ValueError(f"project {project_id} not found")
|
raise ValueError(f"project {project_id} not found")
|
||||||
binding = await repo_bindings_svc.set_binding(uid, repo_url, project_id)
|
ref_arg = None if not ref else ("" if ref.strip() == "-" else ref)
|
||||||
|
binding = await repo_bindings_svc.set_binding(uid, repo_url, project_id, ref_arg)
|
||||||
|
follows = binding.ref or "the default branch"
|
||||||
return {
|
return {
|
||||||
"repo_key": binding.repo_key,
|
"repo_key": binding.repo_key,
|
||||||
"project_id": binding.project_id,
|
"project_id": binding.project_id,
|
||||||
"project_title": project.title,
|
"project_title": project.title,
|
||||||
"message": f"Bound `{binding.repo_key}` -> {project.title} (id {project.id}).",
|
"ref": binding.ref,
|
||||||
|
"message": (
|
||||||
|
f"Bound `{binding.repo_key}` -> {project.title} (id {project.id}); "
|
||||||
|
f"the ledger follows {follows}."
|
||||||
|
),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+101
-10
@@ -36,10 +36,20 @@ async def classify_shapes(
|
|||||||
Args:
|
Args:
|
||||||
project_id: The project whose ledger is being judged.
|
project_id: The project whose ledger is being judged.
|
||||||
classifications: Objects of {path, symbol, status, kind?, snippet_id?,
|
classifications: Objects of {path, symbol, status, kind?, snippet_id?,
|
||||||
reason?}. path+symbol name the shape exactly as list_shapes shows
|
reason?, reason_code?}. path+symbol name the shape exactly as
|
||||||
it; kind ("sym"/"css") narrows when one file defines both.
|
list_shapes shows it; kind ("sym"/"css") narrows when one file
|
||||||
snippet_id is required for canonical/instance/variant; reason is
|
defines both. snippet_id is required for canonical/instance/
|
||||||
required for variant/exempt.
|
variant; reason is required for variant/exempt. reason_code is
|
||||||
|
an OPTIONAL index beside the prose (one of: scoped-css,
|
||||||
|
one-off-handler, test-helper, convention-plumbing, pure-helper,
|
||||||
|
generated, script, typed-record) so the ledger can be filtered
|
||||||
|
and aggregated by kind of one-off — the prose stays the record.
|
||||||
|
uses is an OPTIONAL list of snippet ids this shape CALLS (#2870):
|
||||||
|
conformance (status + snippet_id) says what shape it is, uses
|
||||||
|
says which canonical helpers it consumes — a service function
|
||||||
|
can be an instance of the service-function convention AND use
|
||||||
|
hash_token. Consumer maps are uses edges; list_shapes(uses=N)
|
||||||
|
and get_snippet's `uses` read them.
|
||||||
via: Who is judging — "agent" (default), "audit" (a sweep), or
|
via: Who is judging — "agent" (default), "audit" (a sweep), or
|
||||||
"import" (carrying maps recorded elsewhere).
|
"import" (carrying maps recorded elsewhere).
|
||||||
|
|
||||||
@@ -64,6 +74,8 @@ async def list_shapes(
|
|||||||
offset: int = 0,
|
offset: int = 0,
|
||||||
proposal: str = "",
|
proposal: str = "",
|
||||||
flag: str = "",
|
flag: str = "",
|
||||||
|
compact: bool = False,
|
||||||
|
uses: int = 0,
|
||||||
) -> dict:
|
) -> dict:
|
||||||
"""Read a project's shape ledger — `status="unclassified"` IS the todo.
|
"""Read a project's shape ledger — `status="unclassified"` IS the todo.
|
||||||
|
|
||||||
@@ -71,12 +83,27 @@ async def list_shapes(
|
|||||||
(fed by the coverage refresh). Filters compose:
|
(fed by the coverage refresh). Filters compose:
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
status: canonical | instance | variant | exempt | unclassified.
|
status: canonical | instance | variant | exempt | scoped | unclassified.
|
||||||
|
`scoped` (#2869) is the sync's mechanical stamp on one-offs by
|
||||||
|
construction (a Vue component's scoped <style> rules and its
|
||||||
|
<script setup> functions): accounted for, not judged, still
|
||||||
|
proposed against / grouped / flagged, and overridable by any
|
||||||
|
classify_shapes judgment. The human todo is `unclassified`.
|
||||||
path: exact file, or a directory — matches everything beneath it
|
path: exact file, or a directory — matches everything beneath it
|
||||||
(the coverage line's "largest" dirs go straight in here).
|
(the coverage line's "largest" dirs go straight in here).
|
||||||
snippet_id: rows classified against this snippet — a consumer map.
|
snippet_id: rows classified against this snippet (instance/variant
|
||||||
|
of it — conformance).
|
||||||
|
uses: rows that CALL this snippet (#2870) — the consumer map proper,
|
||||||
|
whatever shape each row is itself; edges come from judgments
|
||||||
|
(classify_shapes uses=), the write-path hook, and the proposer's
|
||||||
|
by-name reference hits.
|
||||||
include_vanished: include shapes no longer in the tree (history).
|
include_vanished: include shapes no longer in the tree (history).
|
||||||
limit/offset: page through big ledgers (limit caps at 500).
|
limit/offset: page through big ledgers (limit caps at 500).
|
||||||
|
compact: rows as `path · symbol · kind · status · signature` plus
|
||||||
|
snippet_id / by / proposal / diverges_from / recheck only when
|
||||||
|
set — no commits, shas or timestamps. THE form for an audit:
|
||||||
|
a full 500-row page fits the tool budget. The default rows carry
|
||||||
|
everything (shape_history-grade bookkeeping).
|
||||||
proposal: the proposer's queue (#2792) — "any", "canon" (rows the
|
proposal: the proposer's queue (#2792) — "any", "canon" (rows the
|
||||||
machine thinks are an instance of a snippet: `proposal` carries
|
machine thinks are an instance of a snippet: `proposal` carries
|
||||||
snippet_id, basis, score), "derive" (rows that repeat with NO
|
snippet_id, basis, score), "derive" (rows that repeat with NO
|
||||||
@@ -116,9 +143,73 @@ async def list_shapes(
|
|||||||
uid, project_id,
|
uid, project_id,
|
||||||
status=status, path=path, snippet_id=snippet_id,
|
status=status, path=path, snippet_id=snippet_id,
|
||||||
include_vanished=include_vanished, limit=limit, offset=offset,
|
include_vanished=include_vanished, limit=limit, offset=offset,
|
||||||
proposal=proposal, flag=flag,
|
proposal=proposal, flag=flag, uses=uses,
|
||||||
)
|
)
|
||||||
return {"shapes": [r.to_dict() for r in rows], "total": total}
|
return {
|
||||||
|
"shapes": [r.to_compact() if compact else r.to_dict() for r in rows],
|
||||||
|
"total": total,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def classify_shapes_by_rule(
|
||||||
|
project_id: int,
|
||||||
|
path: str,
|
||||||
|
status: str,
|
||||||
|
pattern: str = "",
|
||||||
|
kind: str = "",
|
||||||
|
snippet_id: int = 0,
|
||||||
|
reason: str = "",
|
||||||
|
via: str = "agent",
|
||||||
|
include_judged: bool = False,
|
||||||
|
reason_code: str = "",
|
||||||
|
uses: list[int] | None = None,
|
||||||
|
) -> dict:
|
||||||
|
"""The sweep form of classify_shapes: ONE judgment applied to every
|
||||||
|
unclassified shape under a directory whose symbol matches a glob.
|
||||||
|
|
||||||
|
For the long tail an audit judges by family, not by row — "every scoped
|
||||||
|
rule under frontend/src/views is exempt: styles one element of its view",
|
||||||
|
"every `*_scheduler.py` symbol is an instance of ScheduledJob" — where
|
||||||
|
listing 900 rows and sending them back is the whole cost. The row form
|
||||||
|
stays the precise tool; reach for it when each row gets its own reason.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
project_id: The project whose ledger is being judged.
|
||||||
|
path: A file, or a directory and everything beneath it. Required —
|
||||||
|
a sweep names what it judges.
|
||||||
|
status: instance | variant | exempt | unclassified (canonical is the
|
||||||
|
sync's stamp, not a sweep's).
|
||||||
|
pattern: Shell glob on the symbol (`*_rows`, `_*`, `modal-*`, `*`);
|
||||||
|
"" = every symbol under path.
|
||||||
|
kind: "sym" or "css" to narrow; "" = both.
|
||||||
|
snippet_id: Required for instance/variant — the canon judged against.
|
||||||
|
reason: Required for variant/exempt — the why, recorded on every row.
|
||||||
|
via: "agent" (default) | "audit" | "import".
|
||||||
|
reason_code: Optional catalogue code beside the reason (see
|
||||||
|
classify_shapes) — a sweep is exactly where one applies.
|
||||||
|
uses: Optional snippet ids every matched shape CALLS (#2870) — e.g.
|
||||||
|
"every *_scheduler.py symbol uses ScheduledJob".
|
||||||
|
include_judged: By default only unjudged rows are touched —
|
||||||
|
`unclassified` and the sync's mechanical `scoped` stamp — a
|
||||||
|
sweep never silently overwrites a judgment. True re-judges every
|
||||||
|
matching live row (use to re-confirm after a recheck, or to
|
||||||
|
revise a family you judged earlier).
|
||||||
|
|
||||||
|
One transaction: applies whole or not at all. Returns
|
||||||
|
{"classified": N, "sample": ["path::symbol", ...]} (first 12, sorted)
|
||||||
|
so you can see what the rule reached; N = 0 means the rule matched
|
||||||
|
nothing live and unclassified — widen the pattern or refresh coverage.
|
||||||
|
"""
|
||||||
|
uid = current_user_id()
|
||||||
|
try:
|
||||||
|
return await shape_ledger_svc.classify_shapes_where(
|
||||||
|
uid, project_id, path=path, status=status, pattern=pattern,
|
||||||
|
kind=kind, snippet_id=snippet_id or None, reason=reason or None,
|
||||||
|
via=via, include_judged=include_judged,
|
||||||
|
reason_code=reason_code or None, uses=uses or None,
|
||||||
|
)
|
||||||
|
except ValueError as exc:
|
||||||
|
return {"error": str(exc)}
|
||||||
|
|
||||||
|
|
||||||
async def shape_history(
|
async def shape_history(
|
||||||
@@ -217,7 +308,7 @@ async def refresh_pattern_coverage(project_id: int) -> dict:
|
|||||||
|
|
||||||
def register(mcp) -> None:
|
def register(mcp) -> None:
|
||||||
for fn in (
|
for fn in (
|
||||||
classify_shapes, list_shapes, refresh_pattern_coverage,
|
classify_shapes, classify_shapes_by_rule, list_shapes,
|
||||||
confirm_shape_proposals, shape_history,
|
refresh_pattern_coverage, confirm_shape_proposals, shape_history,
|
||||||
):
|
):
|
||||||
mcp.tool(name=fn.__name__)(fn)
|
mcp.tool(name=fn.__name__)(fn)
|
||||||
|
|||||||
@@ -245,6 +245,10 @@ async def get_snippet(snippet_id: int) -> dict:
|
|||||||
data["instances"] = consumers["instances"]
|
data["instances"] = consumers["instances"]
|
||||||
if consumers["variants"]:
|
if consumers["variants"]:
|
||||||
data["variants"] = consumers["variants"]
|
data["variants"] = consumers["variants"]
|
||||||
|
if consumers.get("uses"):
|
||||||
|
# The call sites (#2870): shapes that use this snippet, whatever
|
||||||
|
# shape they are themselves.
|
||||||
|
data["uses"] = consumers["uses"]
|
||||||
return data
|
return data
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -44,6 +44,6 @@ from scribe.models.rulebook import ( # noqa: E402, F401
|
|||||||
)
|
)
|
||||||
from scribe.models.repo_binding import RepoBinding # noqa: E402, F401
|
from scribe.models.repo_binding import RepoBinding # noqa: E402, F401
|
||||||
from scribe.models.forge_connection import ForgeConnection # noqa: E402, F401
|
from scribe.models.forge_connection import ForgeConnection # noqa: E402, F401
|
||||||
from scribe.models.code_shape import CodeShape, CodeShapeEvent # noqa: E402, F401
|
from scribe.models.code_shape import CodeShape, CodeShapeEvent, CodeShapeUse # noqa: E402, F401
|
||||||
from scribe.models.system import System, RecordSystem # noqa: E402, F401
|
from scribe.models.system import System, RecordSystem # noqa: E402, F401
|
||||||
from scribe.models.design_system import DesignSystem, DesignToken # noqa: E402, F401
|
from scribe.models.design_system import DesignSystem, DesignToken # noqa: E402, F401
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
from datetime import datetime
|
from datetime import datetime, timezone
|
||||||
|
|
||||||
from sqlalchemy import (
|
from sqlalchemy import (
|
||||||
BigInteger,
|
BigInteger,
|
||||||
@@ -16,10 +16,30 @@ from scribe.models import Base
|
|||||||
from scribe.models.base import TimestampMixin, iso
|
from scribe.models.base import TimestampMixin, iso
|
||||||
|
|
||||||
# The classification vocabulary (note 2786). `unclassified` is the default and
|
# The classification vocabulary (note 2786). `unclassified` is the default and
|
||||||
# THE todo state; every other status is a judgment, stamped with who made it.
|
# THE todo state; every other status is a judgment, stamped with who made it —
|
||||||
SHAPE_STATUSES = ("canonical", "instance", "variant", "exempt", "unclassified")
|
# except `scoped` (#2869): the coverage sync's mechanical stamp on shapes that
|
||||||
|
# are one-offs BY CONSTRUCTION (a Vue component's scoped <style> rules and its
|
||||||
|
# <script setup> functions — unreachable from any other file). Scoped rows are
|
||||||
|
# accounted for without a human judging them, so `exempt` keeps meaning "a
|
||||||
|
# person looked"; the proposer, derive grouping and divergence still see them,
|
||||||
|
# and any judgment (instance/variant/exempt) overrides the stamp.
|
||||||
|
SHAPE_STATUSES = ("canonical", "instance", "variant", "exempt", "scoped", "unclassified")
|
||||||
SHAPE_CLASSIFIERS = ("agent", "audit", "hook", "mechanical", "import")
|
SHAPE_CLASSIFIERS = ("agent", "audit", "hook", "mechanical", "import")
|
||||||
|
|
||||||
|
# The reason catalogue (#2874): an OPTIONAL code beside the prose reason on
|
||||||
|
# variant/exempt rows, so the ledger can be filtered and aggregated by kind
|
||||||
|
# of one-off. The prose remains the record; the code is the index.
|
||||||
|
REASON_CODES = (
|
||||||
|
"scoped-css", # a scoped rule styling one element (pre-#2869 rows)
|
||||||
|
"one-off-handler", # a view/component handler or loader, one per surface
|
||||||
|
"test-helper", # a test module's stub, driver or fixture data
|
||||||
|
"convention-plumbing", # registration, wiring, app factory — one of each
|
||||||
|
"pure-helper", # a sync module-private helper with no session
|
||||||
|
"generated", # generated source (theme.css, protos, bundles)
|
||||||
|
"script", # a standalone dev/CI script
|
||||||
|
"typed-record", # a NamedTuple / dataclass / error class — one each
|
||||||
|
)
|
||||||
|
|
||||||
# How the mechanical proposer (#2792) arrived at a proposal, strongest first.
|
# How the mechanical proposer (#2792) arrived at a proposal, strongest first.
|
||||||
# `derive` is the odd one out: not "this is an instance of #N" but "this
|
# `derive` is the odd one out: not "this is an instance of #N" but "this
|
||||||
# shape repeats with NO canon — derive one first" (note 2786's derive-first
|
# shape repeats with NO canon — derive one first" (note 2786's derive-first
|
||||||
@@ -99,6 +119,7 @@ class CodeShape(Base, TimestampMixin):
|
|||||||
BigInteger, ForeignKey("notes.id", ondelete="SET NULL"), nullable=True
|
BigInteger, ForeignKey("notes.id", ondelete="SET NULL"), nullable=True
|
||||||
)
|
)
|
||||||
reason: Mapped[str | None] = mapped_column(Text, nullable=True)
|
reason: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
reason_code: Mapped[str | None] = mapped_column(Text, nullable=True) # REASON_CODES (#2874)
|
||||||
classified_by: Mapped[str | None] = mapped_column(Text, nullable=True)
|
classified_by: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
classified_at: Mapped[datetime | None] = mapped_column(
|
classified_at: Mapped[datetime | None] = mapped_column(
|
||||||
DateTime(timezone=True), nullable=True
|
DateTime(timezone=True), nullable=True
|
||||||
@@ -152,6 +173,7 @@ class CodeShape(Base, TimestampMixin):
|
|||||||
"status": self.status,
|
"status": self.status,
|
||||||
"snippet_id": self.snippet_id,
|
"snippet_id": self.snippet_id,
|
||||||
"reason": self.reason,
|
"reason": self.reason,
|
||||||
|
"reason_code": self.reason_code,
|
||||||
"classified_by": self.classified_by,
|
"classified_by": self.classified_by,
|
||||||
"classified_at": iso(self.classified_at),
|
"classified_at": iso(self.classified_at),
|
||||||
"first_seen_commit": self.first_seen_commit,
|
"first_seen_commit": self.first_seen_commit,
|
||||||
@@ -167,6 +189,81 @@ class CodeShape(Base, TimestampMixin):
|
|||||||
"updated_at": iso(self.updated_at),
|
"updated_at": iso(self.updated_at),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
def to_compact(self) -> dict:
|
||||||
|
"""The row as an audit reads it (#2868): identity, standing, the
|
||||||
|
definition line and the proposer's word — none of the bookkeeping
|
||||||
|
(commits, shas, timestamps). A 500-row page of these fits the tool
|
||||||
|
budget; a page of to_dict() does not."""
|
||||||
|
out = {
|
||||||
|
"path": self.path,
|
||||||
|
"symbol": self.symbol,
|
||||||
|
"kind": self.kind,
|
||||||
|
"status": self.status,
|
||||||
|
"signature": self.signature,
|
||||||
|
}
|
||||||
|
if self.snippet_id is not None:
|
||||||
|
out["snippet_id"] = self.snippet_id
|
||||||
|
if self.classified_by:
|
||||||
|
out["by"] = self.classified_by
|
||||||
|
if self.reason_code:
|
||||||
|
out["reason_code"] = self.reason_code
|
||||||
|
proposal = self.proposal
|
||||||
|
if proposal:
|
||||||
|
out["proposal"] = proposal
|
||||||
|
if self.diverges_from is not None:
|
||||||
|
out["diverges_from"] = self.diverges_from
|
||||||
|
if self.recheck_at is not None:
|
||||||
|
out["recheck"] = True
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
# How a uses edge was established (#2870): who/what said "this shape calls
|
||||||
|
# that canon". `reference` is the proposer's mechanical by-name hit on the
|
||||||
|
# body (language-gated, #2871); `hook` is write-path evidence (pulled the
|
||||||
|
# snippet, then wrote code naming its symbol); agent/audit/import are
|
||||||
|
# judgments carried on classify_shapes(..., uses=[...]).
|
||||||
|
USE_BASES = ("reference", "hook", "agent", "audit", "import")
|
||||||
|
|
||||||
|
|
||||||
|
class CodeShapeUse(Base):
|
||||||
|
"""One consumption edge: shape → canonical snippet it calls/uses (#2870).
|
||||||
|
|
||||||
|
Conformance (CodeShape.status/snippet_id) answers "what shape is this";
|
||||||
|
this table answers "what does it use" — many per shape. A service function
|
||||||
|
that is an instance of the service-function convention AND a consumer of
|
||||||
|
hash_token has one snippet_id and one uses edge. Cascades with both ends:
|
||||||
|
a use of a deleted snippet is no longer a fact worth keeping.
|
||||||
|
"""
|
||||||
|
|
||||||
|
__tablename__ = "code_shape_uses"
|
||||||
|
__table_args__ = (
|
||||||
|
UniqueConstraint("shape_id", "snippet_id", name="uq_code_shape_uses_shape_snippet"),
|
||||||
|
Index("ix_code_shape_uses_snippet", "snippet_id"),
|
||||||
|
)
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
|
shape_id: Mapped[int] = mapped_column(
|
||||||
|
Integer, ForeignKey("code_shapes.id", ondelete="CASCADE"), nullable=False
|
||||||
|
)
|
||||||
|
snippet_id: Mapped[int] = mapped_column(
|
||||||
|
Integer, ForeignKey("notes.id", ondelete="CASCADE"), nullable=False
|
||||||
|
)
|
||||||
|
basis: Mapped[str] = mapped_column(Text, nullable=False)
|
||||||
|
evidence: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
created_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=False, default=lambda: datetime.now(timezone.utc)
|
||||||
|
)
|
||||||
|
|
||||||
|
def to_dict(self) -> dict:
|
||||||
|
return {
|
||||||
|
"id": self.id,
|
||||||
|
"shape_id": self.shape_id,
|
||||||
|
"snippet_id": self.snippet_id,
|
||||||
|
"basis": self.basis,
|
||||||
|
"evidence": self.evidence,
|
||||||
|
"created_at": iso(self.created_at),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
# What a shape's history records (#2793). Not "appeared" — first_seen and
|
# What a shape's history records (#2793). Not "appeared" — first_seen and
|
||||||
# created_at already say that on the row; history is for what CHANGED:
|
# created_at already say that on the row; history is for what CHANGED:
|
||||||
|
|||||||
@@ -28,6 +28,10 @@ class RepoBinding(Base, TimestampMixin):
|
|||||||
Integer, ForeignKey("projects.id", ondelete="CASCADE"), nullable=False
|
Integer, ForeignKey("projects.id", ondelete="CASCADE"), nullable=False
|
||||||
)
|
)
|
||||||
repo_key: Mapped[str] = mapped_column(Text, nullable=False)
|
repo_key: Mapped[str] = mapped_column(Text, nullable=False)
|
||||||
|
# The branch the coverage refresh reads for this binding (#2873); NULL =
|
||||||
|
# the forge's default branch. Chosen at bind time so a dev-first project
|
||||||
|
# can have its ledger follow dev instead of waiting for the merge.
|
||||||
|
ref: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
|
||||||
def to_dict(self) -> dict:
|
def to_dict(self) -> dict:
|
||||||
return {
|
return {
|
||||||
@@ -35,6 +39,7 @@ class RepoBinding(Base, TimestampMixin):
|
|||||||
"user_id": self.user_id,
|
"user_id": self.user_id,
|
||||||
"project_id": self.project_id,
|
"project_id": self.project_id,
|
||||||
"repo_key": self.repo_key,
|
"repo_key": self.repo_key,
|
||||||
|
"ref": self.ref,
|
||||||
"created_at": iso(self.created_at),
|
"created_at": iso(self.created_at),
|
||||||
"updated_at": iso(self.updated_at),
|
"updated_at": iso(self.updated_at),
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ from scribe.models.note_supersession import NoteSupersession
|
|||||||
from scribe.models.note_version import NoteVersion
|
from scribe.models.note_version import NoteVersion
|
||||||
from scribe.models.design_system import DesignSystem, DesignToken
|
from scribe.models.design_system import DesignSystem, DesignToken
|
||||||
from scribe.models.note_usage import NoteUsageEvent
|
from scribe.models.note_usage import NoteUsageEvent
|
||||||
from scribe.models.code_shape import CodeShape, CodeShapeEvent
|
from scribe.models.code_shape import CodeShape, CodeShapeEvent, CodeShapeUse
|
||||||
from scribe.models.project import Project
|
from scribe.models.project import Project
|
||||||
from scribe.models.repo_binding import RepoBinding
|
from scribe.models.repo_binding import RepoBinding
|
||||||
from scribe.models.rulebook import (
|
from scribe.models.rulebook import (
|
||||||
@@ -42,8 +42,11 @@ logger = logging.getLogger(__name__)
|
|||||||
# snippet target survives the id re-mapping, else the row rejoins the todo.
|
# snippet target survives the id re-mapping, else the row rejoins the todo.
|
||||||
# v8 (2026-08) added code_shape_events — the ledger's history (#2793): what
|
# v8 (2026-08) added code_shape_events — the ledger's history (#2793): what
|
||||||
# was used where, when, and why is not recomputable, so it travels.
|
# was used where, when, and why is not recomputable, so it travels.
|
||||||
|
# v9 (2026-08) added code_shape_uses — the ledger's consumption edges (#2870):
|
||||||
|
# judgment-grade edges (agent/audit/import) are operator records; mechanical
|
||||||
|
# ones (reference/hook) travel too, cheaply, and the next refresh refreshes them.
|
||||||
# Bump when the serialized schema changes.
|
# Bump when the serialized schema changes.
|
||||||
BACKUP_VERSION = 8
|
BACKUP_VERSION = 9
|
||||||
|
|
||||||
# Every table this backup carries, by its REAL name. Paired with _NOT_INCLUDED
|
# Every table this backup carries, by its REAL name. Paired with _NOT_INCLUDED
|
||||||
# below, these two lists must together account for the entire schema — which is
|
# below, these two lists must together account for the entire schema — which is
|
||||||
@@ -62,7 +65,7 @@ _BACKED_UP = [
|
|||||||
"systems", "record_systems", "design_systems", "design_tokens",
|
"systems", "record_systems", "design_systems", "design_tokens",
|
||||||
"note_usage_events", "repo_bindings", "note_supersessions",
|
"note_usage_events", "repo_bindings", "note_supersessions",
|
||||||
# v7 (2026-08): the shape ledger (#2787); v8: its history (#2793).
|
# v7 (2026-08): the shape ledger (#2787); v8: its history (#2793).
|
||||||
"code_shapes", "code_shape_events",
|
"code_shapes", "code_shape_events", "code_shape_uses",
|
||||||
]
|
]
|
||||||
|
|
||||||
# Tables intentionally NOT in the backup, surfaced in the payload so the gap is
|
# Tables intentionally NOT in the backup, surfaced in the payload so the gap is
|
||||||
@@ -182,6 +185,10 @@ def _code_shape_event_rows(rows) -> list[dict]:
|
|||||||
return [r.to_dict() for r in rows]
|
return [r.to_dict() for r in rows]
|
||||||
|
|
||||||
|
|
||||||
|
def _code_shape_use_rows(rows) -> list[dict]:
|
||||||
|
return [r.to_dict() for r in rows]
|
||||||
|
|
||||||
|
|
||||||
def _repo_binding_rows(rows) -> list[dict]:
|
def _repo_binding_rows(rows) -> list[dict]:
|
||||||
return [
|
return [
|
||||||
{"user_id": r.user_id, "project_id": r.project_id, "repo_key": r.repo_key}
|
{"user_id": r.user_id, "project_id": r.project_id, "repo_key": r.repo_key}
|
||||||
@@ -361,6 +368,9 @@ async def export_full_backup() -> dict:
|
|||||||
code_shape_events = (await session.execute(
|
code_shape_events = (await session.execute(
|
||||||
select(CodeShapeEvent).order_by(CodeShapeEvent.at, CodeShapeEvent.id)
|
select(CodeShapeEvent).order_by(CodeShapeEvent.at, CodeShapeEvent.id)
|
||||||
)).scalars().all()
|
)).scalars().all()
|
||||||
|
code_shape_uses = (await session.execute(
|
||||||
|
select(CodeShapeUse).order_by(CodeShapeUse.shape_id, CodeShapeUse.snippet_id)
|
||||||
|
)).scalars().all()
|
||||||
rulebooks = (await session.execute(select(Rulebook))).scalars().all()
|
rulebooks = (await session.execute(select(Rulebook))).scalars().all()
|
||||||
topics = (await session.execute(select(RulebookTopic))).scalars().all()
|
topics = (await session.execute(select(RulebookTopic))).scalars().all()
|
||||||
rules = (await session.execute(select(Rule))).scalars().all()
|
rules = (await session.execute(select(Rule))).scalars().all()
|
||||||
@@ -406,6 +416,7 @@ async def export_full_backup() -> dict:
|
|||||||
"note_supersessions": _note_supersession_rows(supersessions),
|
"note_supersessions": _note_supersession_rows(supersessions),
|
||||||
"code_shapes": _code_shape_rows(code_shapes),
|
"code_shapes": _code_shape_rows(code_shapes),
|
||||||
"code_shape_events": _code_shape_event_rows(code_shape_events),
|
"code_shape_events": _code_shape_event_rows(code_shape_events),
|
||||||
|
"code_shape_uses": _code_shape_use_rows(code_shape_uses),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -482,6 +493,11 @@ async def export_user_backup(user_id: int) -> dict:
|
|||||||
select(CodeShapeEvent).where(CodeShapeEvent.project_id.in_(project_ids))
|
select(CodeShapeEvent).where(CodeShapeEvent.project_id.in_(project_ids))
|
||||||
.order_by(CodeShapeEvent.at, CodeShapeEvent.id)
|
.order_by(CodeShapeEvent.at, CodeShapeEvent.id)
|
||||||
)).scalars().all() if project_ids else []
|
)).scalars().all() if project_ids else []
|
||||||
|
code_shape_uses = (await session.execute(
|
||||||
|
select(CodeShapeUse).join(CodeShape, CodeShape.id == CodeShapeUse.shape_id)
|
||||||
|
.where(CodeShape.project_id.in_(project_ids))
|
||||||
|
.order_by(CodeShapeUse.shape_id, CodeShapeUse.snippet_id)
|
||||||
|
)).scalars().all() if project_ids else []
|
||||||
rulebooks = (await session.execute(
|
rulebooks = (await session.execute(
|
||||||
select(Rulebook).where(Rulebook.owner_user_id == user_id)
|
select(Rulebook).where(Rulebook.owner_user_id == user_id)
|
||||||
)).scalars().all()
|
)).scalars().all()
|
||||||
@@ -553,6 +569,7 @@ async def export_user_backup(user_id: int) -> dict:
|
|||||||
"note_supersessions": _note_supersession_rows(supersessions),
|
"note_supersessions": _note_supersession_rows(supersessions),
|
||||||
"code_shapes": _code_shape_rows(code_shapes),
|
"code_shapes": _code_shape_rows(code_shapes),
|
||||||
"code_shape_events": _code_shape_event_rows(code_shape_events),
|
"code_shape_events": _code_shape_event_rows(code_shape_events),
|
||||||
|
"code_shape_uses": _code_shape_use_rows(code_shape_uses),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -657,6 +674,7 @@ async def _restore_v2(data: dict) -> dict:
|
|||||||
"systems": 0, "record_systems": 0, "design_systems": 0,
|
"systems": 0, "record_systems": 0, "design_systems": 0,
|
||||||
"design_tokens": 0, "note_usage_events": 0, "repo_bindings": 0,
|
"design_tokens": 0, "note_usage_events": 0, "repo_bindings": 0,
|
||||||
"note_supersessions": 0, "code_shapes": 0, "code_shape_events": 0,
|
"note_supersessions": 0, "code_shapes": 0, "code_shape_events": 0,
|
||||||
|
"code_shape_uses": 0,
|
||||||
}
|
}
|
||||||
|
|
||||||
async with async_session() as session:
|
async with async_session() as session:
|
||||||
@@ -1105,6 +1123,20 @@ async def _restore_v2(data: dict) -> dict:
|
|||||||
))
|
))
|
||||||
stats["code_shape_events"] += 1
|
stats["code_shape_events"] += 1
|
||||||
|
|
||||||
|
# v9: consumption edges (#2870) ride their shape AND their snippet —
|
||||||
|
# both ends must have survived, or the edge is no longer a fact.
|
||||||
|
for use in data.get("code_shape_uses", []):
|
||||||
|
new_shape_id = shape_id_map.get(use.get("shape_id") or 0)
|
||||||
|
new_sid = note_id_map.get(use.get("snippet_id") or 0)
|
||||||
|
if new_shape_id is None or new_sid is None:
|
||||||
|
continue
|
||||||
|
session.add(CodeShapeUse(
|
||||||
|
shape_id=new_shape_id, snippet_id=new_sid,
|
||||||
|
basis=use.get("basis", "import"), evidence=use.get("evidence"),
|
||||||
|
created_at=_dt(use.get("created_at")),
|
||||||
|
))
|
||||||
|
stats["code_shape_uses"] += 1
|
||||||
|
|
||||||
await session.commit()
|
await session.commit()
|
||||||
|
|
||||||
logger.info("Restored v2/v3 backup: %s", stats)
|
logger.info("Restored v2/v3 backup: %s", stats)
|
||||||
|
|||||||
@@ -36,7 +36,7 @@ from typing import NamedTuple
|
|||||||
from datetime import datetime, timedelta, timezone
|
from datetime import datetime, timedelta, timezone
|
||||||
|
|
||||||
from scribe.services.forge import ForgeSelector, get_forges
|
from scribe.services.forge import ForgeSelector, get_forges
|
||||||
from scribe.services.repo_bindings import keys_for_project
|
from scribe.services.repo_bindings import bindings_for_project
|
||||||
from scribe.services.settings import get_setting, set_setting
|
from scribe.services.settings import get_setting, set_setting
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
@@ -107,6 +107,7 @@ class Definition(NamedTuple):
|
|||||||
signature: str
|
signature: str
|
||||||
body_sha: str
|
body_sha: str
|
||||||
body: str
|
body: str
|
||||||
|
line: int = -1 # 0-based line the definition starts on (#2869)
|
||||||
|
|
||||||
|
|
||||||
def _definition_on(raw: str) -> tuple[str, str] | None:
|
def _definition_on(raw: str) -> tuple[str, str] | None:
|
||||||
@@ -184,13 +185,56 @@ def extract_definitions(text: str) -> list[Definition]:
|
|||||||
end = j
|
end = j
|
||||||
break
|
break
|
||||||
block = lines[i:end]
|
block = lines[i:end]
|
||||||
|
# A CSS rule's fingerprint is its DECLARATIONS, not its selector
|
||||||
|
# (#2872): the row's identity already carries the selector, and the
|
||||||
|
# question the fingerprint answers for derive grouping is "is this the
|
||||||
|
# same rule under another name?" — .closed-msg / .error-block /
|
||||||
|
# .success-msg with identical bodies are one dup group, not three
|
||||||
|
# lonely rows. Sym blocks keep their signature line in the hash.
|
||||||
|
hashed = block[1:] if kind == "css" and len(block) > 1 else block
|
||||||
out.append(Definition(
|
out.append(Definition(
|
||||||
kind, name, lines[i].strip()[:_SIGNATURE_CAP], _block_sha(block),
|
kind, name, lines[i].strip()[:_SIGNATURE_CAP], _block_sha(hashed),
|
||||||
"\n".join(block),
|
"\n".join(block), i,
|
||||||
))
|
))
|
||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
# --- by-construction scope (#2869) -------------------------------------------
|
||||||
|
#
|
||||||
|
# A Vue single-file component's `<style scoped>` rules and its `<script setup>`
|
||||||
|
# functions cannot be reached from any other file: they are one-offs by
|
||||||
|
# construction, not by judgment. The sync stamps them `scoped` (mechanical) so
|
||||||
|
# the human todo holds only shapes a person should look at, while the bodies
|
||||||
|
# stay in play for the proposer, derive grouping and divergence — the five
|
||||||
|
# auth views' identical rules were found exactly there. Unscoped `<style>` in
|
||||||
|
# a .vue and every non-.vue file stay ordinary.
|
||||||
|
_STYLE_OPEN_RE = re.compile(r"^\s*<style\b[^>]*\bscoped\b", re.IGNORECASE)
|
||||||
|
_STYLE_CLOSE_RE = re.compile(r"^\s*</style\s*>", re.IGNORECASE)
|
||||||
|
|
||||||
|
|
||||||
|
def scoped_definitions(path: str, text: str, defs: list[Definition]) -> set[tuple[str, str]]:
|
||||||
|
"""The (kind, name) pairs among ``defs`` that are one-offs by
|
||||||
|
construction in this file: every sym in a .vue, and every css rule
|
||||||
|
that starts inside a `<style scoped>` block. Empty for other files."""
|
||||||
|
if not (path or "").lower().endswith(".vue"):
|
||||||
|
return set()
|
||||||
|
ranges: list[tuple[int, int]] = []
|
||||||
|
open_at: int | None = None
|
||||||
|
for i, ln in enumerate(text.splitlines()):
|
||||||
|
if open_at is None and _STYLE_OPEN_RE.match(ln):
|
||||||
|
open_at = i
|
||||||
|
elif open_at is not None and _STYLE_CLOSE_RE.match(ln):
|
||||||
|
ranges.append((open_at, i))
|
||||||
|
open_at = None
|
||||||
|
out: set[tuple[str, str]] = set()
|
||||||
|
for d in defs:
|
||||||
|
if d.kind == "sym":
|
||||||
|
out.add((d.kind, d.name))
|
||||||
|
elif any(a <= d.line <= b for a, b in ranges):
|
||||||
|
out.add((d.kind, d.name))
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
def extract_shapes(text: str) -> list[tuple[str, str]]:
|
def extract_shapes(text: str) -> list[tuple[str, str]]:
|
||||||
"""Every (kind, name) this text DEFINES — kind is "css" or "sym".
|
"""Every (kind, name) this text DEFINES — kind is "css" or "sym".
|
||||||
|
|
||||||
@@ -220,6 +264,7 @@ class ArchiveShape(NamedTuple):
|
|||||||
signature: str
|
signature: str
|
||||||
body_sha: str
|
body_sha: str
|
||||||
body: str
|
body: str
|
||||||
|
scoped: bool = False # one-off by construction (#2869)
|
||||||
|
|
||||||
|
|
||||||
def shapes_from_archive(blob: bytes) -> list[tuple[str, str, str]]:
|
def shapes_from_archive(blob: bytes) -> list[tuple[str, str, str]]:
|
||||||
@@ -250,9 +295,14 @@ def definitions_from_archive(blob: bytes) -> list[ArchiveShape]:
|
|||||||
text = handle.read().decode("utf-8")
|
text = handle.read().decode("utf-8")
|
||||||
except UnicodeDecodeError:
|
except UnicodeDecodeError:
|
||||||
continue
|
continue
|
||||||
|
defs = extract_definitions(text)
|
||||||
|
scoped = scoped_definitions(path, text, defs)
|
||||||
shapes.extend(
|
shapes.extend(
|
||||||
ArchiveShape(path, d.kind, d.name, d.signature, d.body_sha, d.body)
|
ArchiveShape(
|
||||||
for d in extract_definitions(text)
|
path, d.kind, d.name, d.signature, d.body_sha, d.body,
|
||||||
|
(d.kind, d.name) in scoped,
|
||||||
|
)
|
||||||
|
for d in defs
|
||||||
)
|
)
|
||||||
return shapes
|
return shapes
|
||||||
|
|
||||||
@@ -357,12 +407,15 @@ async def compute_coverage(
|
|||||||
# the project's repos (#2792).
|
# the project's repos (#2792).
|
||||||
canons = None
|
canons = None
|
||||||
proposer_stats = {"examined": 0, "proposed": 0, "semantic_checked": 0}
|
proposer_stats = {"examined": 0, "proposed": 0, "semantic_checked": 0}
|
||||||
for key in await keys_for_project(user_id, project_id):
|
for binding in await bindings_for_project(user_id, project_id):
|
||||||
|
key = binding.repo_key
|
||||||
hit = selector.resolve(key)
|
hit = selector.resolve(key)
|
||||||
if hit is None:
|
if hit is None:
|
||||||
continue # bound to a host no connection serves
|
continue # bound to a host no connection serves
|
||||||
forge, api_repo = hit
|
forge, api_repo = hit
|
||||||
ref = await forge.default_branch(api_repo)
|
# The binding's own ref when it names one (#2873: a dev-first project
|
||||||
|
# has its ledger follow dev), else the forge's default branch.
|
||||||
|
ref = binding.ref or await forge.default_branch(api_repo)
|
||||||
definitions = definitions_from_archive(await forge.archive(api_repo, ref))
|
definitions = definitions_from_archive(await forge.archive(api_repo, ref))
|
||||||
# The head commit is provenance sugar on the ledger rows; failing to
|
# The head commit is provenance sugar on the ledger rows; failing to
|
||||||
# learn it must not fail the sync — the ref names the point well
|
# learn it must not fail the sync — the ref names the point well
|
||||||
@@ -414,7 +467,7 @@ async def compute_coverage(
|
|||||||
# repo that was unreachable today still has live rows, and they count.
|
# repo that was unreachable today still has live rows, and they count.
|
||||||
rows = await shape_ledger.live_rows(project_id)
|
rows = await shape_ledger.live_rows(project_id)
|
||||||
counts = {"canonical": 0, "instance": 0, "variant": 0, "exempt": 0,
|
counts = {"canonical": 0, "instance": 0, "variant": 0, "exempt": 0,
|
||||||
"unclassified": 0}
|
"scoped": 0, "unclassified": 0}
|
||||||
for row in rows:
|
for row in rows:
|
||||||
counts[row.status] = counts.get(row.status, 0) + 1
|
counts[row.status] = counts.get(row.status, 0) + 1
|
||||||
by_repo: dict[str, dict[str, int]] = {}
|
by_repo: dict[str, dict[str, int]] = {}
|
||||||
@@ -435,6 +488,7 @@ async def compute_coverage(
|
|||||||
# confirm, the largest derive-first groups, and what this refresh did.
|
# confirm, the largest derive-first groups, and what this refresh did.
|
||||||
"proposed": proposals["proposed"],
|
"proposed": proposals["proposed"],
|
||||||
"derive_groups": proposals["derive_groups"],
|
"derive_groups": proposals["derive_groups"],
|
||||||
|
"top_canon": proposals.get("top_canon"),
|
||||||
"proposer": proposer_stats,
|
"proposer": proposer_stats,
|
||||||
# The divergence readout (#2793): button B where button A is canon,
|
# The divergence readout (#2793): button B where button A is canon,
|
||||||
# and judged shapes whose bodies moved since they were judged.
|
# and judged shapes whose bodies moved since they were judged.
|
||||||
@@ -571,7 +625,7 @@ def coverage_line(coverage: dict) -> str:
|
|||||||
counts = coverage.get("counts") or {}
|
counts = coverage.get("counts") or {}
|
||||||
breakdown = " · ".join(
|
breakdown = " · ".join(
|
||||||
f"{counts[k]} {k}"
|
f"{counts[k]} {k}"
|
||||||
for k in ("canonical", "instance", "variant", "exempt")
|
for k in ("canonical", "instance", "variant", "exempt", "scoped")
|
||||||
if counts.get(k)
|
if counts.get(k)
|
||||||
)
|
)
|
||||||
line = (
|
line = (
|
||||||
@@ -592,6 +646,14 @@ def coverage_line(coverage: dict) -> str:
|
|||||||
standing.append(f"{n_groups} derive group{'s' if n_groups != 1 else ''}")
|
standing.append(f"{n_groups} derive group{'s' if n_groups != 1 else ''}")
|
||||||
if coverage.get("divergent"):
|
if coverage.get("divergent"):
|
||||||
standing.append(f"{coverage['divergent']} DIVERGENT")
|
standing.append(f"{coverage['divergent']} DIVERGENT")
|
||||||
|
# The next action, on the line (#2874): the canon with the biggest
|
||||||
|
# queue to confirm, and the widest body-identical copy to consolidate.
|
||||||
|
top = coverage.get("top_canon") or {}
|
||||||
|
if top.get("snippet_id"):
|
||||||
|
standing.append(f"top canon #{top['snippet_id']} ×{top.get('count', 0)}")
|
||||||
|
first = (coverage.get("derive_groups") or [{}])[0]
|
||||||
|
if first.get("label") and first.get("files"):
|
||||||
|
standing.append(f"top copy {first['label']} ×{first['files']} files")
|
||||||
if standing:
|
if standing:
|
||||||
line += f" ({', '.join(standing)})"
|
line += f" ({', '.join(standing)})"
|
||||||
gaps = [g["dir"] for g in coverage.get("largest_gaps") or []]
|
gaps = [g["dir"] for g in coverage.get("largest_gaps") or []]
|
||||||
|
|||||||
@@ -16,6 +16,7 @@ endorsed, so a one-off direct share has to be searched for rather than arriving
|
|||||||
in your ambient lists.
|
in your ambient lists.
|
||||||
"""
|
"""
|
||||||
import json
|
import json
|
||||||
|
import re
|
||||||
import logging
|
import logging
|
||||||
|
|
||||||
from sqlalchemy import and_, func, or_, select
|
from sqlalchemy import and_, func, or_, select
|
||||||
@@ -76,6 +77,10 @@ def location_matches(data: dict | None, parts: dict[str, str]) -> bool:
|
|||||||
if all(
|
if all(
|
||||||
_path_matches((loc.get(key) or "").strip(), want)
|
_path_matches((loc.get(key) or "").strip(), want)
|
||||||
if key == "path"
|
if key == "path"
|
||||||
|
# Repo names are recorded free-form ("Scribe" / "FabledScribe" /
|
||||||
|
# "fabledscribe") — case is never the distinguishing thing (#2874).
|
||||||
|
else (loc.get(key) or "").strip().lower() == want.lower()
|
||||||
|
if key == "repo"
|
||||||
else (loc.get(key) or "").strip() == want
|
else (loc.get(key) or "").strip() == want
|
||||||
for key, want in parts.items()
|
for key, want in parts.items()
|
||||||
):
|
):
|
||||||
@@ -99,6 +104,11 @@ def location_jsonpath(parts: dict[str, str]) -> str:
|
|||||||
if key == "path":
|
if key == "path":
|
||||||
prefix = json.dumps(want.rstrip("/") + "/")
|
prefix = json.dumps(want.rstrip("/") + "/")
|
||||||
filters.append(f"(@.path == {literal} || @.path starts with {prefix})")
|
filters.append(f"(@.path == {literal} || @.path starts with {prefix})")
|
||||||
|
elif key == "repo":
|
||||||
|
# Case-insensitive, anchored, regex-escaped (#2874) — mirrors the
|
||||||
|
# Python dialect's .lower() compare.
|
||||||
|
pattern = json.dumps("^" + re.escape(want) + "$")
|
||||||
|
filters.append(f'(@.repo like_regex {pattern} flag "i")')
|
||||||
else:
|
else:
|
||||||
filters.append(f"@.{key} == {literal}")
|
filters.append(f"@.{key} == {literal}")
|
||||||
return f"$.locations[*] ? ({' && '.join(filters)})"
|
return f"$.locations[*] ? ({' && '.join(filters)})"
|
||||||
|
|||||||
@@ -68,8 +68,16 @@ async def resolve_project(user_id: int, raw_repo: str) -> int | None:
|
|||||||
return row.scalar_one_or_none()
|
return row.scalar_one_or_none()
|
||||||
|
|
||||||
|
|
||||||
async def set_binding(user_id: int, raw_repo: str, project_id: int) -> RepoBinding:
|
async def set_binding(
|
||||||
"""Create or update the binding for a repo. Idempotent on (user, repo_key)."""
|
user_id: int, raw_repo: str, project_id: int, ref: str | None = None,
|
||||||
|
) -> RepoBinding:
|
||||||
|
"""Create or update the binding for a repo. Idempotent on (user, repo_key).
|
||||||
|
|
||||||
|
``ref`` (#2873) is the branch the coverage refresh reads for this
|
||||||
|
binding: a name sets it, ``""`` clears it back to the forge's default
|
||||||
|
branch, ``None`` leaves whatever stands (a re-bind that only moves the
|
||||||
|
project keeps the ref it had).
|
||||||
|
"""
|
||||||
key = normalize_repo_key(raw_repo)
|
key = normalize_repo_key(raw_repo)
|
||||||
if not key:
|
if not key:
|
||||||
raise ValueError("repo remote is empty or unparseable")
|
raise ValueError("repo remote is empty or unparseable")
|
||||||
@@ -85,6 +93,8 @@ async def set_binding(user_id: int, raw_repo: str, project_id: int) -> RepoBindi
|
|||||||
session.add(binding)
|
session.add(binding)
|
||||||
else:
|
else:
|
||||||
binding.project_id = project_id
|
binding.project_id = project_id
|
||||||
|
if ref is not None:
|
||||||
|
binding.ref = ref.strip() or None
|
||||||
await session.commit()
|
await session.commit()
|
||||||
await session.refresh(binding)
|
await session.refresh(binding)
|
||||||
return binding
|
return binding
|
||||||
@@ -100,6 +110,18 @@ async def list_bindings(user_id: int) -> list[RepoBinding]:
|
|||||||
return list(rows.scalars().all())
|
return list(rows.scalars().all())
|
||||||
|
|
||||||
|
|
||||||
|
async def bindings_for_project(user_id: int, project_id: int) -> list[RepoBinding]:
|
||||||
|
"""Every binding of a project — key AND the ref its ledger follows (#2873)."""
|
||||||
|
async with async_session() as session:
|
||||||
|
rows = await session.execute(
|
||||||
|
select(RepoBinding).where(
|
||||||
|
RepoBinding.user_id == user_id,
|
||||||
|
RepoBinding.project_id == project_id,
|
||||||
|
).order_by(RepoBinding.repo_key)
|
||||||
|
)
|
||||||
|
return list(rows.scalars().all())
|
||||||
|
|
||||||
|
|
||||||
async def keys_for_project(user_id: int, project_id: int) -> list[str]:
|
async def keys_for_project(user_id: int, project_id: int) -> list[str]:
|
||||||
"""Every repo key bound to a project — the snippet→forge join (#2691).
|
"""Every repo key bound to a project — the snippet→forge join (#2691).
|
||||||
|
|
||||||
|
|||||||
@@ -30,7 +30,7 @@ from typing import Iterable, NamedTuple
|
|||||||
from sqlalchemy import select
|
from sqlalchemy import select
|
||||||
|
|
||||||
from scribe.models import async_session
|
from scribe.models import async_session
|
||||||
from scribe.models.code_shape import CodeShape, CodeShapeEvent
|
from scribe.models.code_shape import REASON_CODES, CodeShape, CodeShapeEvent, CodeShapeUse
|
||||||
from scribe.models.base import iso
|
from scribe.models.base import iso
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
@@ -65,6 +65,17 @@ def location_covers(loc_path: str, loc_symbol: str, path: str, name: str) -> boo
|
|||||||
return _path_touches(loc_path, path)
|
return _path_touches(loc_path, path)
|
||||||
|
|
||||||
|
|
||||||
|
# The rows the machine may still speak about: nobody's judgment stands on
|
||||||
|
# them. `scoped` (#2869) is the sync's own by-construction stamp — the
|
||||||
|
# proposer, derive grouping, divergence, hook evidence and sweeps treat it
|
||||||
|
# like the todo; only the human todo (`unclassified`) excludes it.
|
||||||
|
_MECHANICAL_TODO = ("unclassified", "scoped")
|
||||||
|
_SCOPED_REASON = (
|
||||||
|
"by construction: a Vue component's scoped <style> rule / <script setup> "
|
||||||
|
"function — unreachable from any other file (stamped by the coverage sync)"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
async def sync_repo_shapes(
|
async def sync_repo_shapes(
|
||||||
project_id: int,
|
project_id: int,
|
||||||
repo_key: str,
|
repo_key: str,
|
||||||
@@ -76,7 +87,9 @@ async def sync_repo_shapes(
|
|||||||
|
|
||||||
``shapes`` are (path, kind, name) triples, or the richer ArchiveShape
|
``shapes`` are (path, kind, name) triples, or the richer ArchiveShape
|
||||||
records (#2792) whose 4th/5th fields — signature, body_sha — refresh the
|
records (#2792) whose 4th/5th fields — signature, body_sha — refresh the
|
||||||
row's content fingerprint. ``seen_marker`` is the commit the archive was
|
row's content fingerprint, and whose 7th (#2869) says the shape is a
|
||||||
|
one-off by construction: such rows are stamped `scoped` (mechanical)
|
||||||
|
while unjudged, and un-stamped if a later tree makes them reachable. ``seen_marker`` is the commit the archive was
|
||||||
read at when the forge can say, else the ref name — provenance sugar;
|
read at when the forge can say, else the ref name — provenance sugar;
|
||||||
the row timestamps carry the when.
|
the row timestamps carry the when.
|
||||||
"""
|
"""
|
||||||
@@ -96,20 +109,32 @@ async def sync_repo_shapes(
|
|||||||
path, kind, name = shape[0], shape[1], shape[2]
|
path, kind, name = shape[0], shape[1], shape[2]
|
||||||
signature = shape[3] if len(shape) > 3 else ""
|
signature = shape[3] if len(shape) > 3 else ""
|
||||||
body_sha = shape[4] if len(shape) > 4 else ""
|
body_sha = shape[4] if len(shape) > 4 else ""
|
||||||
|
scoped = bool(shape[6]) if len(shape) > 6 else False
|
||||||
key = (path, name, kind)
|
key = (path, name, kind)
|
||||||
if key in seen:
|
if key in seen:
|
||||||
continue
|
continue
|
||||||
seen.add(key)
|
seen.add(key)
|
||||||
row = by_key.get(key)
|
row = by_key.get(key)
|
||||||
if row is None:
|
if row is None:
|
||||||
session.add(CodeShape(
|
row = CodeShape(
|
||||||
project_id=project_id, repo_key=repo_key,
|
project_id=project_id, repo_key=repo_key,
|
||||||
path=path, symbol=name, kind=kind,
|
path=path, symbol=name, kind=kind,
|
||||||
first_seen_commit=seen_marker, last_seen_commit=seen_marker,
|
first_seen_commit=seen_marker, last_seen_commit=seen_marker,
|
||||||
signature=signature, body_sha=body_sha,
|
signature=signature, body_sha=body_sha,
|
||||||
))
|
)
|
||||||
|
if scoped:
|
||||||
|
await _judge(session, row, status="scoped", snippet_id=None,
|
||||||
|
by="mechanical", reason=_SCOPED_REASON, at=now)
|
||||||
|
else:
|
||||||
|
session.add(row)
|
||||||
continue
|
continue
|
||||||
row.last_seen_commit = seen_marker
|
row.last_seen_commit = seen_marker
|
||||||
|
if scoped and row.status == "unclassified":
|
||||||
|
await _judge(session, row, status="scoped", snippet_id=None,
|
||||||
|
by="mechanical", reason=_SCOPED_REASON, at=now)
|
||||||
|
elif not scoped and row.status == "scoped":
|
||||||
|
await _judge(session, row, status="unclassified", snippet_id=None,
|
||||||
|
by=None, reason=None, at=now)
|
||||||
if signature:
|
if signature:
|
||||||
row.signature = signature
|
row.signature = signature
|
||||||
if body_sha and body_sha != row.body_sha:
|
if body_sha and body_sha != row.body_sha:
|
||||||
@@ -158,7 +183,7 @@ def _event(row: CodeShape, event: str, at: datetime, *, commit: str = "") -> Cod
|
|||||||
|
|
||||||
async def _judge(
|
async def _judge(
|
||||||
session, row: CodeShape, *, status: str, snippet_id: int | None,
|
session, row: CodeShape, *, status: str, snippet_id: int | None,
|
||||||
by: str | None, reason: str | None, at: datetime,
|
by: str | None, reason: str | None, at: datetime, reason_code: str | None = None,
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Apply a judgment to a row — the ONE place a status is set — and write
|
"""Apply a judgment to a row — the ONE place a status is set — and write
|
||||||
its history. Clears what a judgment settles: the standing proposal, the
|
its history. Clears what a judgment settles: the standing proposal, the
|
||||||
@@ -169,6 +194,7 @@ async def _judge(
|
|||||||
row.status = status
|
row.status = status
|
||||||
row.snippet_id = snippet_id if status in _NEEDS_TARGET else None
|
row.snippet_id = snippet_id if status in _NEEDS_TARGET else None
|
||||||
row.reason = (reason or "").strip() or None
|
row.reason = (reason or "").strip() or None
|
||||||
|
row.reason_code = (reason_code or "").strip() or None
|
||||||
row.classified_by = by if status != "unclassified" else None
|
row.classified_by = by if status != "unclassified" else None
|
||||||
row.classified_at = at if status != "unclassified" else None
|
row.classified_at = at if status != "unclassified" else None
|
||||||
row.classified_sha = row.body_sha if status != "unclassified" else ""
|
row.classified_sha = row.body_sha if status != "unclassified" else ""
|
||||||
@@ -183,6 +209,59 @@ async def _judge(
|
|||||||
session.add(_event(row, "classified", at))
|
session.add(_event(row, "classified", at))
|
||||||
|
|
||||||
|
|
||||||
|
async def record_uses(
|
||||||
|
session, row: CodeShape, snippet_ids, *, basis: str, evidence: str | None = None,
|
||||||
|
) -> int:
|
||||||
|
"""Upsert consumption edges shape → snippet (#2870). A judgment-grade
|
||||||
|
basis (agent/audit/import) overwrites a mechanical one (reference/hook)
|
||||||
|
on the same edge; mechanical never overwrites a judgment. Returns the
|
||||||
|
number of edges written or refreshed. The row must be persisted (flushed)
|
||||||
|
so it has an id."""
|
||||||
|
wanted = {int(x) for x in (snippet_ids or []) if x}
|
||||||
|
if not wanted:
|
||||||
|
return 0
|
||||||
|
if row.id is None:
|
||||||
|
session.add(row)
|
||||||
|
await session.flush()
|
||||||
|
existing = {
|
||||||
|
e.snippet_id: e
|
||||||
|
for e in (
|
||||||
|
await session.execute(
|
||||||
|
select(CodeShapeUse).where(CodeShapeUse.shape_id == row.id)
|
||||||
|
)
|
||||||
|
).scalars().all()
|
||||||
|
}
|
||||||
|
judged = basis in _CALLER_VIAS
|
||||||
|
n = 0
|
||||||
|
for sid in wanted:
|
||||||
|
edge = existing.get(sid)
|
||||||
|
if edge is None:
|
||||||
|
session.add(CodeShapeUse(shape_id=row.id, snippet_id=sid, basis=basis, evidence=evidence))
|
||||||
|
n += 1
|
||||||
|
elif judged or edge.basis not in _CALLER_VIAS:
|
||||||
|
edge.basis, edge.evidence = basis, evidence
|
||||||
|
n += 1
|
||||||
|
return n
|
||||||
|
|
||||||
|
|
||||||
|
async def uses_of(shape_ids) -> dict[int, list[CodeShapeUse]]:
|
||||||
|
"""{shape_id: [edges]} for a set of rows — the read side of record_uses."""
|
||||||
|
ids = [int(x) for x in shape_ids if x]
|
||||||
|
if not ids:
|
||||||
|
return {}
|
||||||
|
async with async_session() as session:
|
||||||
|
edges = (
|
||||||
|
await session.execute(
|
||||||
|
select(CodeShapeUse).where(CodeShapeUse.shape_id.in_(ids))
|
||||||
|
.order_by(CodeShapeUse.shape_id, CodeShapeUse.snippet_id)
|
||||||
|
)
|
||||||
|
).scalars().all()
|
||||||
|
out: dict[int, list[CodeShapeUse]] = {}
|
||||||
|
for e in edges:
|
||||||
|
out.setdefault(e.shape_id, []).append(e)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
async def mark_canonicals(
|
async def mark_canonicals(
|
||||||
project_id: int, recorded: list[tuple[int, str, str]]
|
project_id: int, recorded: list[tuple[int, str, str]]
|
||||||
) -> None:
|
) -> None:
|
||||||
@@ -214,7 +293,7 @@ async def mark_canonicals(
|
|||||||
),
|
),
|
||||||
None,
|
None,
|
||||||
)
|
)
|
||||||
if covering is not None and row.status == "unclassified":
|
if covering is not None and row.status in _MECHANICAL_TODO:
|
||||||
await _judge(session, row, status="canonical", snippet_id=covering,
|
await _judge(session, row, status="canonical", snippet_id=covering,
|
||||||
by="mechanical", reason=None, at=now)
|
by="mechanical", reason=None, at=now)
|
||||||
elif (
|
elif (
|
||||||
@@ -289,6 +368,17 @@ def validate_classifications(items: list[dict]) -> str | None:
|
|||||||
f"classifications[{i}]: status {status!r} needs a reason — "
|
f"classifications[{i}]: status {status!r} needs a reason — "
|
||||||
"the WHY is the record (note 2786)"
|
"the WHY is the record (note 2786)"
|
||||||
)
|
)
|
||||||
|
code = (item.get("reason_code") or "").strip()
|
||||||
|
if code and code not in REASON_CODES:
|
||||||
|
return (
|
||||||
|
f"classifications[{i}]: unknown reason_code {code!r} "
|
||||||
|
f"(one of: {', '.join(REASON_CODES)})"
|
||||||
|
)
|
||||||
|
uses = item.get("uses")
|
||||||
|
if uses is not None and (
|
||||||
|
not isinstance(uses, list) or not all(isinstance(u, int) and u > 0 for u in uses)
|
||||||
|
):
|
||||||
|
return f"classifications[{i}]: uses must be a list of snippet ids"
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
@@ -327,6 +417,8 @@ async def classify_shapes(
|
|||||||
for item in classifications
|
for item in classifications
|
||||||
if item.get("status") in _NEEDS_TARGET
|
if item.get("status") in _NEEDS_TARGET
|
||||||
}
|
}
|
||||||
|
for item in classifications:
|
||||||
|
target_ids.update(int(u) for u in (item.get("uses") or []))
|
||||||
for sid in sorted(target_ids):
|
for sid in sorted(target_ids):
|
||||||
if await snippets_svc.get_snippet(user_id, sid) is None:
|
if await snippets_svc.get_snippet(user_id, sid) is None:
|
||||||
raise ValueError(f"snippet {sid} not found (or not readable)")
|
raise ValueError(f"snippet {sid} not found (or not readable)")
|
||||||
@@ -364,12 +456,101 @@ async def classify_shapes(
|
|||||||
session, row, status=status,
|
session, row, status=status,
|
||||||
snippet_id=int(item["snippet_id"]) if status in _NEEDS_TARGET else None,
|
snippet_id=int(item["snippet_id"]) if status in _NEEDS_TARGET else None,
|
||||||
by=via, reason=item.get("reason"), at=now,
|
by=via, reason=item.get("reason"), at=now,
|
||||||
|
reason_code=item.get("reason_code"),
|
||||||
)
|
)
|
||||||
|
if item.get("uses"):
|
||||||
|
await record_uses(session, row, item["uses"], basis=via,
|
||||||
|
evidence=item.get("reason"))
|
||||||
classified += 1
|
classified += 1
|
||||||
await session.commit()
|
await session.commit()
|
||||||
return {"classified": classified, "unmatched": unmatched}
|
return {"classified": classified, "unmatched": unmatched}
|
||||||
|
|
||||||
|
|
||||||
|
def rule_matches(row: CodeShape, *, path: str, pattern: str, kind: str) -> bool:
|
||||||
|
"""Does a ledger row fall under a rule-form classification (#2868)?
|
||||||
|
``path`` is a file or a directory (everything beneath it), ``pattern``
|
||||||
|
a shell glob on the symbol (``""`` = every symbol), ``kind`` narrows to
|
||||||
|
sym/css. Pure, so the sweep's reach can be tested without a database."""
|
||||||
|
import fnmatch
|
||||||
|
|
||||||
|
clean = (path or "").strip().strip("/")
|
||||||
|
if clean and not (row.path == clean or row.path.startswith(clean + "/")):
|
||||||
|
return False
|
||||||
|
if kind and row.kind != kind:
|
||||||
|
return False
|
||||||
|
if pattern and not fnmatch.fnmatchcase(_norm_symbol(row.symbol), pattern):
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
async def classify_shapes_where(
|
||||||
|
user_id: int,
|
||||||
|
project_id: int,
|
||||||
|
*,
|
||||||
|
path: str,
|
||||||
|
status: str,
|
||||||
|
pattern: str = "",
|
||||||
|
kind: str = "",
|
||||||
|
snippet_id: int | None = None,
|
||||||
|
reason: str | None = None,
|
||||||
|
via: str = "agent",
|
||||||
|
include_judged: bool = False,
|
||||||
|
reason_code: str | None = None,
|
||||||
|
uses: list[int] | None = None,
|
||||||
|
) -> dict:
|
||||||
|
"""The sweep form of classify_shapes (#2868): one judgment applied to
|
||||||
|
every live row under ``path`` whose symbol matches ``pattern`` (and
|
||||||
|
``kind``). By default only unjudged rows are touched — `unclassified`
|
||||||
|
and the sync's mechanical `scoped` stamp — a sweep must never silently
|
||||||
|
overwrite a judgment; ``include_judged`` opts in.
|
||||||
|
Same gates as the row form (status vocabulary, snippet target, reason
|
||||||
|
for variant/exempt, write access); one transaction, so it applies whole
|
||||||
|
or not at all. Returns the count and a sample of what it judged."""
|
||||||
|
from scribe.services import access
|
||||||
|
from scribe.services import snippets as snippets_svc
|
||||||
|
|
||||||
|
if via not in _CALLER_VIAS:
|
||||||
|
raise ValueError(f"via must be one of: {', '.join(_CALLER_VIAS)}")
|
||||||
|
if not (path or "").strip():
|
||||||
|
raise ValueError("path is required — a sweep names the directory it judges")
|
||||||
|
if status == "canonical":
|
||||||
|
raise ValueError("canonical is the sync's stamp on a snippet's own location — a sweep cannot set it")
|
||||||
|
probe = {"path": path, "symbol": "*", "status": status,
|
||||||
|
"snippet_id": snippet_id or 0, "reason": reason or "",
|
||||||
|
"reason_code": reason_code or "", "uses": uses}
|
||||||
|
error = validate_classifications([probe])
|
||||||
|
if error:
|
||||||
|
raise ValueError(error.replace("classifications[0]", "rule"))
|
||||||
|
if not await access.can_write_project(user_id, project_id):
|
||||||
|
raise ValueError(f"project {project_id} not found or no write access")
|
||||||
|
if status in _NEEDS_TARGET and await snippets_svc.get_snippet(user_id, int(snippet_id)) is None:
|
||||||
|
raise ValueError(f"snippet {snippet_id} not found (or not readable)")
|
||||||
|
for sid in sorted({int(u) for u in (uses or [])}):
|
||||||
|
if await snippets_svc.get_snippet(user_id, sid) is None:
|
||||||
|
raise ValueError(f"snippet {sid} not found (or not readable)")
|
||||||
|
|
||||||
|
now = datetime.now(timezone.utc)
|
||||||
|
judged: list[str] = []
|
||||||
|
async with async_session() as session:
|
||||||
|
conds = [CodeShape.project_id == project_id, CodeShape.vanished_at.is_(None)]
|
||||||
|
if not include_judged:
|
||||||
|
conds.append(CodeShape.status.in_(_MECHANICAL_TODO))
|
||||||
|
rows = (await session.execute(select(CodeShape).where(*conds))).scalars().all()
|
||||||
|
for row in rows:
|
||||||
|
if not rule_matches(row, path=path, pattern=pattern, kind=kind):
|
||||||
|
continue
|
||||||
|
await _judge(
|
||||||
|
session, row, status=status,
|
||||||
|
snippet_id=int(snippet_id) if status in _NEEDS_TARGET else None,
|
||||||
|
by=via, reason=reason, at=now, reason_code=reason_code,
|
||||||
|
)
|
||||||
|
if uses:
|
||||||
|
await record_uses(session, row, uses, basis=via, evidence=reason)
|
||||||
|
judged.append(f"{row.path}::{row.symbol}")
|
||||||
|
await session.commit()
|
||||||
|
return {"classified": len(judged), "sample": sorted(judged)[:12]}
|
||||||
|
|
||||||
|
|
||||||
async def list_project_shapes(
|
async def list_project_shapes(
|
||||||
user_id: int,
|
user_id: int,
|
||||||
project_id: int,
|
project_id: int,
|
||||||
@@ -382,6 +563,7 @@ async def list_project_shapes(
|
|||||||
offset: int = 0,
|
offset: int = 0,
|
||||||
proposal: str = "",
|
proposal: str = "",
|
||||||
flag: str = "",
|
flag: str = "",
|
||||||
|
uses: int = 0,
|
||||||
) -> tuple[list[CodeShape], int]:
|
) -> tuple[list[CodeShape], int]:
|
||||||
"""A filtered page of a project's ledger, with the unfiltered-match total.
|
"""A filtered page of a project's ledger, with the unfiltered-match total.
|
||||||
|
|
||||||
@@ -427,6 +609,12 @@ async def list_project_shapes(
|
|||||||
conds.append(CodeShape.diverges_from.isnot(None))
|
conds.append(CodeShape.diverges_from.isnot(None))
|
||||||
elif flag == "recheck":
|
elif flag == "recheck":
|
||||||
conds.append(CodeShape.recheck_at.isnot(None))
|
conds.append(CodeShape.recheck_at.isnot(None))
|
||||||
|
if uses:
|
||||||
|
# Consumers of a canon (#2870): rows with a uses edge to it, whatever
|
||||||
|
# shape they themselves are.
|
||||||
|
conds.append(CodeShape.id.in_(
|
||||||
|
select(CodeShapeUse.shape_id).where(CodeShapeUse.snippet_id == uses)
|
||||||
|
))
|
||||||
async with async_session() as session:
|
async with async_session() as session:
|
||||||
total = (
|
total = (
|
||||||
await session.execute(
|
await session.execute(
|
||||||
@@ -477,18 +665,38 @@ async def snippet_consumers(user_id: int, note_id: int) -> dict:
|
|||||||
)
|
)
|
||||||
)
|
)
|
||||||
).scalars().all()
|
).scalars().all()
|
||||||
readable: dict[int, bool] = {}
|
# The consumption edges (#2870): rows that USE this canon, whatever
|
||||||
out: dict[str, list[dict]] = {"instances": [], "variants": []}
|
# shape they are themselves — the call-site map.
|
||||||
for row in rows:
|
using = (
|
||||||
if row.project_id not in readable:
|
await session.execute(
|
||||||
readable[row.project_id] = await access.can_read_project(
|
select(CodeShape, CodeShapeUse.basis, CodeShapeUse.evidence)
|
||||||
user_id, row.project_id
|
.join(CodeShapeUse, CodeShapeUse.shape_id == CodeShape.id)
|
||||||
|
.where(CodeShapeUse.snippet_id == note_id, CodeShape.vanished_at.is_(None))
|
||||||
)
|
)
|
||||||
if not readable[row.project_id]:
|
).all()
|
||||||
|
readable: dict[int, bool] = {}
|
||||||
|
|
||||||
|
async def can_read(pid: int) -> bool:
|
||||||
|
if pid not in readable:
|
||||||
|
readable[pid] = await access.can_read_project(user_id, pid)
|
||||||
|
return readable[pid]
|
||||||
|
|
||||||
|
out: dict[str, list[dict]] = {"instances": [], "variants": [], "uses": []}
|
||||||
|
for row in rows:
|
||||||
|
if not await can_read(row.project_id):
|
||||||
continue
|
continue
|
||||||
out["instances" if row.status == "instance" else "variants"].append(
|
out["instances" if row.status == "instance" else "variants"].append(
|
||||||
_consumer_dict(row)
|
_consumer_dict(row)
|
||||||
)
|
)
|
||||||
|
for row, basis, evidence in using:
|
||||||
|
if not await can_read(row.project_id):
|
||||||
|
continue
|
||||||
|
d = _consumer_dict(row)
|
||||||
|
d["basis"] = basis
|
||||||
|
if evidence:
|
||||||
|
d["evidence"] = evidence
|
||||||
|
d.pop("reason", None)
|
||||||
|
out["uses"].append(d)
|
||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
@@ -660,7 +868,7 @@ async def stamp_write_path_instances(
|
|||||||
)
|
)
|
||||||
session.add(row)
|
session.add(row)
|
||||||
by_key[(name, kind)] = row
|
by_key[(name, kind)] = row
|
||||||
elif not (row.status == "unclassified" or row.classified_by == "hook"):
|
elif not (row.status in _MECHANICAL_TODO or row.classified_by == "hook"):
|
||||||
continue # a judgment — or the canon itself — stands
|
continue # a judgment — or the canon itself — stands
|
||||||
await _judge(session, row, status="instance", snippet_id=sid, by="hook",
|
await _judge(session, row, status="instance", snippet_id=sid, by="hook",
|
||||||
reason=why, at=now)
|
reason=why, at=now)
|
||||||
@@ -668,6 +876,13 @@ async def stamp_write_path_instances(
|
|||||||
"path": path, "symbol": name, "kind": kind,
|
"path": path, "symbol": name, "kind": kind,
|
||||||
"snippet_id": sid, "reason": why,
|
"snippet_id": sid, "reason": why,
|
||||||
})
|
})
|
||||||
|
# Every pulled canon the payload NAMES is a uses edge (#2870) — the
|
||||||
|
# call-site fact, independent of which one the row is judged to be.
|
||||||
|
await record_uses(
|
||||||
|
session, row,
|
||||||
|
[s_id for rank, _at, s_id, _why in bucket if rank == 2],
|
||||||
|
basis="hook", evidence="write path: pulled the snippet, payload names its symbol",
|
||||||
|
)
|
||||||
if stamped:
|
if stamped:
|
||||||
await session.commit()
|
await session.commit()
|
||||||
return stamped
|
return stamped
|
||||||
@@ -718,7 +933,9 @@ _SEMANTIC_CAP = 150
|
|||||||
_SEMANTIC_FLOOR = 0.8
|
_SEMANTIC_FLOOR = 0.8
|
||||||
# Bump when a basis's rule changes: rows remember the (body, ruleset) they
|
# Bump when a basis's rule changes: rows remember the (body, ruleset) they
|
||||||
# were examined under, so a tightened rule re-examines everything once.
|
# were examined under, so a tightened rule re-examines everything once.
|
||||||
_PROPOSER_VERSION = 2
|
# v3: language-family gate on the sym bases, reference stoplist, semantic
|
||||||
|
# restricted to the shape's own project (#2871).
|
||||||
|
_PROPOSER_VERSION = 3
|
||||||
# Signature resemblance floor, name blanked (difflib ratio) — and a length
|
# Signature resemblance floor, name blanked (difflib ratio) — and a length
|
||||||
# floor, because `def NAME():` resembles `def NAME(x):` at 0.95 while saying
|
# floor, because `def NAME():` resembles `def NAME(x):` at 0.95 while saying
|
||||||
# nothing; a family shape has parameters to resemble.
|
# nothing; a family shape has parameters to resemble.
|
||||||
@@ -737,6 +954,64 @@ class Canon(NamedTuple):
|
|||||||
signature: str
|
signature: str
|
||||||
code_norm: str
|
code_norm: str
|
||||||
project_id: int = 0
|
project_id: int = 0
|
||||||
|
language: str = "" # the snippet's recorded language; "" = unknown, no gate
|
||||||
|
|
||||||
|
|
||||||
|
# Language families: the sym bases only propose within one. The 2026-08
|
||||||
|
# audit (#2871) found every cross-language hit wrong — a Python tool-module
|
||||||
|
# canon named `register` proposed for Vue `handleSubmit`s that call
|
||||||
|
# `authStore.register()`, and a TS store's `register` matched it by symbol;
|
||||||
|
# Minstrel/Forge TS canon proposed for Python bodies by resemblance. CSS is
|
||||||
|
# its own kind and is not gated here.
|
||||||
|
_FAMILY_BY_LANG = {
|
||||||
|
"python": "py", "py": "py",
|
||||||
|
"typescript": "js", "ts": "js", "tsx": "js", "javascript": "js", "js": "js",
|
||||||
|
"jsx": "js", "vue": "js", "mjs": "js", "cjs": "js",
|
||||||
|
"css": "css", "scss": "css", "sass": "css", "less": "css",
|
||||||
|
"bash": "sh", "sh": "sh", "shell": "sh", "zsh": "sh",
|
||||||
|
"sql": "sql",
|
||||||
|
}
|
||||||
|
_FAMILY_BY_EXT = {
|
||||||
|
".py": "py", ".pyi": "py",
|
||||||
|
".ts": "js", ".tsx": "js", ".js": "js", ".jsx": "js", ".vue": "js", ".mjs": "js", ".cjs": "js",
|
||||||
|
".css": "css", ".scss": "css", ".sass": "css", ".less": "css",
|
||||||
|
".sh": "sh", ".bash": "sh", ".zsh": "sh",
|
||||||
|
".sql": "sql",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def language_family(language: str) -> str:
|
||||||
|
"""The family a recorded snippet language belongs to ("" when unknown)."""
|
||||||
|
return _FAMILY_BY_LANG.get((language or "").strip().lower(), "")
|
||||||
|
|
||||||
|
|
||||||
|
def path_family(path: str) -> str:
|
||||||
|
"""The family a file path belongs to, by extension ("" when unknown)."""
|
||||||
|
p = (path or "").lower()
|
||||||
|
for ext, fam in _FAMILY_BY_EXT.items():
|
||||||
|
if p.endswith(ext):
|
||||||
|
return fam
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def same_family(path: str, canon_language: str) -> bool:
|
||||||
|
"""A sym basis may propose this canon for this path: both families known
|
||||||
|
and equal, or either unknown (no evidence either way → no gate)."""
|
||||||
|
a = path_family(path)
|
||||||
|
b = language_family(canon_language)
|
||||||
|
return not a or not b or a == b
|
||||||
|
|
||||||
|
|
||||||
|
# Reference basis: generic verbs name too many unrelated things to count a
|
||||||
|
# bare mention as a call site of THIS canon (`register`, `load`, `save` …).
|
||||||
|
# The symbol basis still catches a second definition of such a name; the
|
||||||
|
# call-site relation for these becomes a `uses` edge once #2870 lands.
|
||||||
|
_REFERENCE_STOPLIST = frozenset({
|
||||||
|
"get", "set", "put", "post", "load", "save", "run", "main", "init", "setup",
|
||||||
|
"register", "restore", "reset", "toggle", "close", "open", "submit", "handler",
|
||||||
|
"update", "create", "delete", "remove", "add", "start", "stop", "send",
|
||||||
|
"receive", "render", "mount", "dispatch", "call", "apply", "execute",
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
def _norm_text(text: str) -> str:
|
def _norm_text(text: str) -> str:
|
||||||
@@ -768,6 +1043,26 @@ def text_contains(body: str, code: str) -> bool:
|
|||||||
return a in b or b in a
|
return a in b or b in a
|
||||||
|
|
||||||
|
|
||||||
|
def reference_canons(kind: str, path: str, symbol: str, body: str, canons: Iterable[Canon]) -> list[int]:
|
||||||
|
"""Every canon this body NAMES (#2870) — the uses edges the proposer can
|
||||||
|
write mechanically: same kind, same language family, symbol not in the
|
||||||
|
generic-verb stoplist, and not the shape's own name."""
|
||||||
|
norm_sym = _norm_symbol(symbol)
|
||||||
|
out: list[int] = []
|
||||||
|
for c in canons:
|
||||||
|
if c.kind != kind or not c.symbol:
|
||||||
|
continue
|
||||||
|
if kind == "sym" and not same_family(path, c.language):
|
||||||
|
continue
|
||||||
|
if _norm_symbol(c.symbol) == norm_sym:
|
||||||
|
continue
|
||||||
|
if _norm_symbol(c.symbol).lower() in _REFERENCE_STOPLIST:
|
||||||
|
continue
|
||||||
|
if references_symbol(body, c.symbol, kind):
|
||||||
|
out.append(c.snippet_id)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
def match_canon(
|
def match_canon(
|
||||||
kind: str, path: str, symbol: str, signature: str, body: str,
|
kind: str, path: str, symbol: str, signature: str, body: str,
|
||||||
canons: Iterable[Canon], *, project_id: int = 0,
|
canons: Iterable[Canon], *, project_id: int = 0,
|
||||||
@@ -789,11 +1084,17 @@ def match_canon(
|
|||||||
for c in canons:
|
for c in canons:
|
||||||
if c.kind != kind:
|
if c.kind != kind:
|
||||||
continue
|
continue
|
||||||
|
if kind == "sym" and not same_family(path, c.language):
|
||||||
|
continue # a Python canon says nothing about a Vue body, and vice versa
|
||||||
if c.symbol and _norm_symbol(c.symbol) == norm_sym:
|
if c.symbol and _norm_symbol(c.symbol) == norm_sym:
|
||||||
if not any(location_covers(lp, ls, path, symbol) for lp, ls in c.locations):
|
if not any(location_covers(lp, ls, path, symbol) for lp, ls in c.locations):
|
||||||
offer("symbol", 1.0, c)
|
offer("symbol", 1.0, c)
|
||||||
continue # its own location is canonical territory, not a proposal
|
continue # its own location is canonical territory, not a proposal
|
||||||
if c.symbol and references_symbol(body, c.symbol, kind):
|
if (
|
||||||
|
c.symbol
|
||||||
|
and _norm_symbol(c.symbol).lower() not in _REFERENCE_STOPLIST
|
||||||
|
and references_symbol(body, c.symbol, kind)
|
||||||
|
):
|
||||||
offer("reference", 0.9, c)
|
offer("reference", 0.9, c)
|
||||||
if c.code_norm and text_contains(body, c.code_norm):
|
if c.code_norm and text_contains(body, c.code_norm):
|
||||||
offer("text", 0.95, c)
|
offer("text", 0.95, c)
|
||||||
@@ -849,6 +1150,7 @@ async def canon_catalog(user_id: int) -> list[Canon]:
|
|||||||
for loc in fields.get("locations") or []
|
for loc in fields.get("locations") or []
|
||||||
),
|
),
|
||||||
signature, _norm_text(code), int(note.project_id or 0),
|
signature, _norm_text(code), int(note.project_id or 0),
|
||||||
|
(fields.get("language") or "").strip().lower(),
|
||||||
))
|
))
|
||||||
return out
|
return out
|
||||||
|
|
||||||
@@ -907,7 +1209,14 @@ async def propose_for_repo(
|
|||||||
if canons is None:
|
if canons is None:
|
||||||
canons = await canon_catalog(user_id)
|
canons = await canon_catalog(user_id)
|
||||||
by_key = {(d[0], d[1], d[2]): d for d in definitions}
|
by_key = {(d[0], d[1], d[2]): d for d in definitions}
|
||||||
sym_canon_ids = {c.snippet_id for c in canons if c.kind == "sym"}
|
# The semantic arm is the widest net and, across projects, was pure noise
|
||||||
|
# in the 2026-08 audit (#2871): it is held to the shape's own project and
|
||||||
|
# language family. The precise bases (symbol/text) still reach family
|
||||||
|
# canon in other projects (note 2786).
|
||||||
|
sym_canons = [c for c in canons if c.kind == "sym" and c.project_id == project_id]
|
||||||
|
|
||||||
|
def semantic_allowed(path: str) -> set[int]:
|
||||||
|
return {c.snippet_id for c in sym_canons if same_family(path, c.language)}
|
||||||
now = datetime.now(timezone.utc)
|
now = datetime.now(timezone.utc)
|
||||||
examined = proposed = checked = 0
|
examined = proposed = checked = 0
|
||||||
async with async_session() as session:
|
async with async_session() as session:
|
||||||
@@ -916,7 +1225,7 @@ async def propose_for_repo(
|
|||||||
select(CodeShape).where(
|
select(CodeShape).where(
|
||||||
CodeShape.project_id == project_id,
|
CodeShape.project_id == project_id,
|
||||||
CodeShape.repo_key == repo_key,
|
CodeShape.repo_key == repo_key,
|
||||||
CodeShape.status == "unclassified",
|
CodeShape.status.in_(_MECHANICAL_TODO),
|
||||||
CodeShape.vanished_at.is_(None),
|
CodeShape.vanished_at.is_(None),
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
@@ -940,6 +1249,12 @@ async def propose_for_repo(
|
|||||||
row.proposal_group = group
|
row.proposal_group = group
|
||||||
row.proposed_at = now
|
row.proposed_at = now
|
||||||
row.proposed_sha = examined_as
|
row.proposed_sha = examined_as
|
||||||
|
# Consumption is recorded for every canon the body names (#2870),
|
||||||
|
# whatever the row is then judged to be.
|
||||||
|
used = reference_canons(row.kind, row.path, row.symbol, body, canons)
|
||||||
|
if used:
|
||||||
|
await record_uses(session, row, used, basis="reference",
|
||||||
|
evidence="proposer: body names the canon's symbol")
|
||||||
if hit:
|
if hit:
|
||||||
row.proposed_snippet_id, row.proposal_basis, row.proposal_score = hit
|
row.proposed_snippet_id, row.proposal_basis, row.proposal_score = hit
|
||||||
row.proposal_group = None
|
row.proposal_group = None
|
||||||
@@ -955,7 +1270,7 @@ async def propose_for_repo(
|
|||||||
continue
|
continue
|
||||||
checked += 1
|
checked += 1
|
||||||
try:
|
try:
|
||||||
found = await _semantic_canon(user_id, d[5], sym_canon_ids)
|
found = await _semantic_canon(user_id, d[5], semantic_allowed(row.path))
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.warning("semantic proposal failed", exc_info=True)
|
logger.warning("semantic proposal failed", exc_info=True)
|
||||||
found = None
|
found = None
|
||||||
@@ -1005,7 +1320,7 @@ async def apply_derive_groups(project_id: int) -> int:
|
|||||||
await session.execute(
|
await session.execute(
|
||||||
select(CodeShape).where(
|
select(CodeShape).where(
|
||||||
CodeShape.project_id == project_id,
|
CodeShape.project_id == project_id,
|
||||||
CodeShape.status == "unclassified",
|
CodeShape.status.in_(_MECHANICAL_TODO),
|
||||||
CodeShape.vanished_at.is_(None),
|
CodeShape.vanished_at.is_(None),
|
||||||
CodeShape.proposed_snippet_id.is_(None),
|
CodeShape.proposed_snippet_id.is_(None),
|
||||||
)
|
)
|
||||||
@@ -1038,27 +1353,44 @@ def proposal_summary(rows: Iterable[CodeShape], *, top: int = 8) -> dict:
|
|||||||
"""The readout's view of the proposer's standing: how many canon
|
"""The readout's view of the proposer's standing: how many canon
|
||||||
proposals await confirmation, and the largest derive-first groups."""
|
proposals await confirmation, and the largest derive-first groups."""
|
||||||
proposed = 0
|
proposed = 0
|
||||||
|
by_canon: dict[int, int] = {}
|
||||||
groups: dict[str, dict] = {}
|
groups: dict[str, dict] = {}
|
||||||
|
files: dict[str, set[str]] = {}
|
||||||
for row in rows:
|
for row in rows:
|
||||||
if row.status != "unclassified":
|
if row.status not in _MECHANICAL_TODO:
|
||||||
continue
|
continue
|
||||||
if row.proposed_snippet_id is not None:
|
if row.proposed_snippet_id is not None:
|
||||||
proposed += 1
|
proposed += 1
|
||||||
|
by_canon[row.proposed_snippet_id] = by_canon.get(row.proposed_snippet_id, 0) + 1
|
||||||
elif row.proposal_group:
|
elif row.proposal_group:
|
||||||
|
dup = not row.proposal_group.startswith("name:")
|
||||||
g = groups.setdefault(row.proposal_group, {
|
g = groups.setdefault(row.proposal_group, {
|
||||||
"group": row.proposal_group, "kind": row.kind,
|
"group": row.proposal_group, "kind": row.kind,
|
||||||
"label": (
|
"label": (
|
||||||
("." if row.kind == "css" else "") + row.symbol
|
f"{row.symbol} (identical body)" if dup
|
||||||
if row.proposal_group.startswith("name:")
|
else ("." if row.kind == "css" else "") + row.symbol
|
||||||
else f"{row.symbol} (identical body)"
|
|
||||||
),
|
),
|
||||||
"size": 0, "paths": [],
|
"size": 0, "files": 0, "paths": [],
|
||||||
})
|
})
|
||||||
g["size"] += 1
|
g["size"] += 1
|
||||||
|
files.setdefault(row.proposal_group, set()).add(row.path)
|
||||||
if len(g["paths"]) < 3:
|
if len(g["paths"]) < 3:
|
||||||
g["paths"].append(row.path)
|
g["paths"].append(row.path)
|
||||||
ranked = sorted(groups.values(), key=lambda g: (-g["size"], g["group"]))
|
for key, g in groups.items():
|
||||||
return {"proposed": proposed, "derive_groups": ranked[:top]}
|
g["files"] = len(files[key])
|
||||||
|
# Body-identical groups first (#2872): the things an audit actually
|
||||||
|
# consolidated were identical bodies under different names/files; a
|
||||||
|
# name repeated across modules is usually convention. Within a tier,
|
||||||
|
# the group spread over more files is the bigger copy.
|
||||||
|
ranked = sorted(
|
||||||
|
groups.values(),
|
||||||
|
key=lambda g: (g["group"].startswith("name:"), -g["files"], -g["size"], g["group"]),
|
||||||
|
)
|
||||||
|
top_canon = None
|
||||||
|
if by_canon:
|
||||||
|
sid, n = max(by_canon.items(), key=lambda kv: (kv[1], -kv[0]))
|
||||||
|
top_canon = {"snippet_id": sid, "count": n}
|
||||||
|
return {"proposed": proposed, "derive_groups": ranked[:top], "top_canon": top_canon}
|
||||||
|
|
||||||
|
|
||||||
async def confirm_proposals(
|
async def confirm_proposals(
|
||||||
@@ -1089,7 +1421,7 @@ async def confirm_proposals(
|
|||||||
|
|
||||||
conds = [
|
conds = [
|
||||||
CodeShape.project_id == project_id,
|
CodeShape.project_id == project_id,
|
||||||
CodeShape.status == "unclassified",
|
CodeShape.status.in_(_MECHANICAL_TODO),
|
||||||
CodeShape.vanished_at.is_(None),
|
CodeShape.vanished_at.is_(None),
|
||||||
CodeShape.proposed_snippet_id.isnot(None),
|
CodeShape.proposed_snippet_id.isnot(None),
|
||||||
]
|
]
|
||||||
@@ -1246,7 +1578,7 @@ async def flag_divergence(project_id: int, *, since: datetime | None) -> int:
|
|||||||
for siblings in by_dir.values():
|
for siblings in by_dir.values():
|
||||||
dom = dominant_canon(siblings)
|
dom = dominant_canon(siblings)
|
||||||
for r in siblings:
|
for r in siblings:
|
||||||
if r.status != "unclassified":
|
if r.status not in _MECHANICAL_TODO:
|
||||||
continue
|
continue
|
||||||
if r.diverges_from is not None:
|
if r.diverges_from is not None:
|
||||||
flagged += 1
|
flagged += 1
|
||||||
@@ -1263,7 +1595,7 @@ async def flag_divergence(project_id: int, *, since: datetime | None) -> int:
|
|||||||
|
|
||||||
def divergence_summary(rows: Iterable[CodeShape], *, top: int = 10) -> dict:
|
def divergence_summary(rows: Iterable[CodeShape], *, top: int = 10) -> dict:
|
||||||
"""Readout view: flagged shapes (newest first) and the recheck count."""
|
"""Readout view: flagged shapes (newest first) and the recheck count."""
|
||||||
flagged = [r for r in rows if r.diverges_from is not None and r.status == "unclassified"]
|
flagged = [r for r in rows if r.diverges_from is not None and r.status in _MECHANICAL_TODO]
|
||||||
flagged.sort(key=lambda r: (r.created_at or datetime.min.replace(tzinfo=timezone.utc)), reverse=True)
|
flagged.sort(key=lambda r: (r.created_at or datetime.min.replace(tzinfo=timezone.utc)), reverse=True)
|
||||||
recheck = sum(1 for r in rows if r.recheck_at is not None and r.vanished_at is None)
|
recheck = sum(1 for r in rows if r.recheck_at is not None and r.vanished_at is None)
|
||||||
return {
|
return {
|
||||||
|
|||||||
@@ -15,6 +15,7 @@ from scribe.models.project import Project
|
|||||||
from scribe.models.user import User
|
from scribe.models.user import User
|
||||||
from scribe.services.shape_ledger import (
|
from scribe.services.shape_ledger import (
|
||||||
classify_shapes,
|
classify_shapes,
|
||||||
|
classify_shapes_where,
|
||||||
list_project_shapes,
|
list_project_shapes,
|
||||||
snippet_consumers,
|
snippet_consumers,
|
||||||
sync_repo_shapes,
|
sync_repo_shapes,
|
||||||
@@ -128,6 +129,113 @@ async def test_classification_is_write_gated_and_listing_read_gated(seeded):
|
|||||||
assert await list_project_shapes(other, pid) == ([], 0)
|
assert await list_project_shapes(other, pid) == ([], 0)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.integration
|
||||||
|
async def test_rule_form_sweeps_unclassified_rows_only_and_applies_whole(seeded):
|
||||||
|
"""#2868: one judgment over a directory + glob; judged rows are left
|
||||||
|
alone unless include_judged; the same gates as the row form."""
|
||||||
|
owner, other, pid, sid = seeded["owner"], seeded["other"], seeded["pid"], seeded["snippet"]
|
||||||
|
await classify_shapes(owner, pid, [
|
||||||
|
{"path": "src/app.py", "symbol": "Config", "status": "exempt", "reason": "settings holder"},
|
||||||
|
])
|
||||||
|
out = await classify_shapes_where(
|
||||||
|
owner, pid, path="src", status="instance", snippet_id=sid, via="audit",
|
||||||
|
)
|
||||||
|
# make_app + helper swept; Config (already judged) untouched; css not under src/.
|
||||||
|
assert out["classified"] == 2
|
||||||
|
assert out["sample"] == ["src/app.py::make_app", "src/util.py::helper"]
|
||||||
|
rows, _ = await list_project_shapes(owner, pid)
|
||||||
|
by_symbol = {r.symbol: r for r in rows}
|
||||||
|
assert by_symbol["make_app"].status == "instance" and by_symbol["make_app"].classified_by == "audit"
|
||||||
|
assert by_symbol["Config"].status == "exempt" and by_symbol["Config"].reason == "settings holder"
|
||||||
|
assert by_symbol["btn"].status == "unclassified"
|
||||||
|
# Glob + kind narrow; include_judged re-judges.
|
||||||
|
out = await classify_shapes_where(
|
||||||
|
owner, pid, path="web", status="exempt", pattern="btn*", kind="css",
|
||||||
|
reason="one toolbar button", include_judged=True,
|
||||||
|
)
|
||||||
|
assert out["classified"] == 1
|
||||||
|
out = await classify_shapes_where(
|
||||||
|
owner, pid, path="src", status="unclassified", include_judged=True,
|
||||||
|
)
|
||||||
|
assert out["classified"] == 3 # withdrawal sweeps judged rows when asked
|
||||||
|
# Gates: reason for exempt, snippet for instance, write access, a path.
|
||||||
|
with pytest.raises(ValueError):
|
||||||
|
await classify_shapes_where(owner, pid, path="src", status="exempt")
|
||||||
|
with pytest.raises(ValueError):
|
||||||
|
await classify_shapes_where(owner, pid, path="src", status="instance")
|
||||||
|
with pytest.raises(ValueError):
|
||||||
|
await classify_shapes_where(owner, pid, path="", status="exempt", reason="x")
|
||||||
|
with pytest.raises(ValueError):
|
||||||
|
await classify_shapes_where(other, pid, path="src", status="exempt", reason="x")
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.integration
|
||||||
|
async def test_sync_stamps_scoped_rows_and_unstamps_when_they_become_reachable(seeded):
|
||||||
|
"""#2869: by-construction one-offs arrive `scoped` (mechanical), count as
|
||||||
|
accounted, are reached by the sweep's default, and go back to the todo
|
||||||
|
if a later tree makes them ordinary. A judgment overrides the stamp."""
|
||||||
|
from scribe.services.coverage import ArchiveShape
|
||||||
|
owner, pid, sid = seeded["owner"], seeded["pid"], seeded["snippet"]
|
||||||
|
scoped_shapes = [
|
||||||
|
ArchiveShape("web/Card.vue", "css", "card", ".card {", "s1", ".card { x: 1 }", True),
|
||||||
|
ArchiveShape("web/Card.vue", "sym", "load", "function load() {", "s2", "function load() {}", True),
|
||||||
|
ArchiveShape("src/util.py", "sym", "helper", "def helper():", "s3", "def helper(): pass", False),
|
||||||
|
]
|
||||||
|
await sync_repo_shapes(pid, REPO, scoped_shapes, seen_marker="main")
|
||||||
|
rows, _ = await list_project_shapes(owner, pid, path="web/Card.vue")
|
||||||
|
assert {r.status for r in rows} == {"scoped"}
|
||||||
|
assert all(r.classified_by == "mechanical" and "by construction" in (r.reason or "") for r in rows)
|
||||||
|
# The human todo excludes them; the sweep's default still reaches them.
|
||||||
|
assert (await list_project_shapes(owner, pid, status="unclassified", path="web/Card.vue"))[1] == 0
|
||||||
|
out = await classify_shapes_where(
|
||||||
|
owner, pid, path="web/Card.vue", status="instance", snippet_id=sid, kind="css",
|
||||||
|
)
|
||||||
|
assert out["classified"] == 1
|
||||||
|
# Re-synced as ordinary: the stamped sym returns to the todo; the
|
||||||
|
# judged css keeps its judgment.
|
||||||
|
plain = [ArchiveShape(s.path, s.kind, s.name, s.signature, s.body_sha, s.body, False) for s in scoped_shapes]
|
||||||
|
await sync_repo_shapes(pid, REPO, plain, seen_marker="main")
|
||||||
|
rows, _ = await list_project_shapes(owner, pid, path="web/Card.vue")
|
||||||
|
by_symbol = {r.symbol: r for r in rows}
|
||||||
|
assert by_symbol["load"].status == "unclassified" and by_symbol["load"].classified_by is None
|
||||||
|
assert by_symbol["card"].status == "instance" and by_symbol["card"].snippet_id == sid
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.integration
|
||||||
|
async def test_uses_edges_are_the_consumer_map(seeded):
|
||||||
|
"""#2870: a shape keeps ONE snippet_id (what it is) and any number of
|
||||||
|
uses edges (what it calls); the snippet's consumer map lists them,
|
||||||
|
list_shapes(uses=N) finds them, and a sweep can write them."""
|
||||||
|
from scribe.services import snippets as snippets_svc
|
||||||
|
owner, pid, sid = seeded["owner"], seeded["pid"], seeded["snippet"]
|
||||||
|
helper = await snippets_svc.create_snippet(
|
||||||
|
owner, name="cls_hash_helper", code="def hash_token(raw):\n return raw\n",
|
||||||
|
language="python", repo="Widget", path="src/hash.py", symbol="hash_token",
|
||||||
|
project_id=pid,
|
||||||
|
)
|
||||||
|
hid = int(helper.id)
|
||||||
|
out = await classify_shapes(owner, pid, [
|
||||||
|
{"path": "src/app.py", "symbol": "make_app", "status": "instance",
|
||||||
|
"snippet_id": sid, "uses": [hid]},
|
||||||
|
], via="audit")
|
||||||
|
assert out["classified"] == 1
|
||||||
|
rows, total = await list_project_shapes(owner, pid, uses=hid)
|
||||||
|
assert total == 1 and rows[0].symbol == "make_app" and rows[0].snippet_id == sid
|
||||||
|
consumers = await snippet_consumers(owner, hid)
|
||||||
|
assert consumers["instances"] == [] and len(consumers["uses"]) == 1
|
||||||
|
assert consumers["uses"][0]["symbol"] == "make_app" and consumers["uses"][0]["basis"] == "audit"
|
||||||
|
# A sweep writes uses too; an unknown snippet in uses applies nothing.
|
||||||
|
out = await classify_shapes_where(
|
||||||
|
owner, pid, path="src/util.py", status="exempt", reason="local", uses=[hid],
|
||||||
|
)
|
||||||
|
assert out["classified"] == 1
|
||||||
|
assert (await list_project_shapes(owner, pid, uses=hid))[1] == 2
|
||||||
|
with pytest.raises(ValueError):
|
||||||
|
await classify_shapes(owner, pid, [
|
||||||
|
{"path": "src/app.py", "symbol": "Config", "status": "exempt", "reason": "x", "uses": [999999]},
|
||||||
|
])
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
@pytest.mark.integration
|
||||||
async def test_list_filters_compose(seeded):
|
async def test_list_filters_compose(seeded):
|
||||||
owner, pid, sid = seeded["owner"], seeded["pid"], seeded["snippet"]
|
owner, pid, sid = seeded["owner"], seeded["pid"], seeded["snippet"]
|
||||||
@@ -170,7 +278,7 @@ async def test_get_snippet_carries_the_structured_consumer_map(seeded):
|
|||||||
# The map is caller-scoped: an outsider asking the service directly gets
|
# The map is caller-scoped: an outsider asking the service directly gets
|
||||||
# silence, not another project's file layout.
|
# silence, not another project's file layout.
|
||||||
consumers = await snippet_consumers(seeded["other"], sid)
|
consumers = await snippet_consumers(seeded["other"], sid)
|
||||||
assert consumers == {"instances": [], "variants": []}
|
assert consumers == {"instances": [], "variants": [], "uses": []}
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.integration
|
@pytest.mark.integration
|
||||||
@@ -458,9 +566,14 @@ async def test_derive_groups_land_on_rows_and_in_the_summary(seeded):
|
|||||||
|
|
||||||
summary = proposal_summary(await live_rows(pid))
|
summary = proposal_summary(await live_rows(pid))
|
||||||
assert summary["proposed"] == 1
|
assert summary["proposed"] == 1
|
||||||
assert [g["group"] for g in summary["derive_groups"]][0] == "name:css:card"
|
# #2872: the body-identical group (a real copy) outranks the bigger name
|
||||||
assert summary["derive_groups"][0]["label"] == ".card"
|
# group (usually convention), even at size 2 vs 3.
|
||||||
assert summary["derive_groups"][0]["size"] == 3
|
order = [g["group"] for g in summary["derive_groups"]]
|
||||||
|
assert order[0].startswith("dup:") and order[1] == "name:css:card"
|
||||||
|
assert summary["derive_groups"][0]["files"] == 2
|
||||||
|
assert summary["derive_groups"][0]["label"] == "slug (identical body)"
|
||||||
|
assert summary["derive_groups"][1]["label"] == ".card"
|
||||||
|
assert summary["derive_groups"][0]["size"] == 2 and summary["derive_groups"][1]["size"] == 3
|
||||||
|
|
||||||
# One of the css copies gets judged → the group shrinks on the next pass.
|
# One of the css copies gets judged → the group shrinks on the next pass.
|
||||||
await classify_shapes(owner, pid, [
|
await classify_shapes(owner, pid, [
|
||||||
@@ -489,7 +602,20 @@ async def test_a_second_confirm_dialog_is_detected_and_named(seeded):
|
|||||||
write_time_divergence,
|
write_time_divergence,
|
||||||
)
|
)
|
||||||
|
|
||||||
owner, pid, sid = seeded["owner"], seeded["pid"], seeded["snippet"]
|
from scribe.services import snippets as snippets_svc
|
||||||
|
|
||||||
|
owner, pid = seeded["owner"], seeded["pid"]
|
||||||
|
# The confirm helper is TS canon: since #2871 a sym basis only proposes
|
||||||
|
# within the shape's language family, so the fixture's Python snippet
|
||||||
|
# says nothing about these Vue bodies — the canon must be one of theirs.
|
||||||
|
canon = await snippets_svc.create_snippet(
|
||||||
|
owner, name="cls_confirm_factory",
|
||||||
|
code="export async function factory(): Promise<boolean> {\n return true;\n}\n",
|
||||||
|
language="typescript", repo="Widget",
|
||||||
|
path="frontend/src/composables/useConfirm.ts", symbol="factory",
|
||||||
|
project_id=pid,
|
||||||
|
)
|
||||||
|
sid = int(canon.id)
|
||||||
comp = "frontend/src/components"
|
comp = "frontend/src/components"
|
||||||
base = _defs(
|
base = _defs(
|
||||||
*[(f"{comp}/{n}.vue", "sym", f"on{n}", f"async function on{n}() {{",
|
*[(f"{comp}/{n}.vue", "sym", f"on{n}", f"async function on{n}() {{",
|
||||||
|
|||||||
@@ -168,6 +168,13 @@ def test_coverage_line_is_evidence_carrying_and_labeled_estimate():
|
|||||||
assert "internal/api, web/src/components" in line
|
assert "internal/api, web/src/components" in line
|
||||||
|
|
||||||
|
|
||||||
|
def test_bind_repo_tool_takes_a_ref():
|
||||||
|
"""#2873: the binding names the branch the ledger follows."""
|
||||||
|
from scribe.mcp.server import build_mcp_server
|
||||||
|
tool = build_mcp_server()._tool_manager.get_tool("bind_repo")
|
||||||
|
assert "ref" in tool.parameters.get("properties", {})
|
||||||
|
|
||||||
|
|
||||||
def test_coverage_routes_are_registered():
|
def test_coverage_routes_are_registered():
|
||||||
from scribe.app import create_app
|
from scribe.app import create_app
|
||||||
|
|
||||||
@@ -188,7 +195,8 @@ def _forge(tar_bytes: bytes):
|
|||||||
path = request.url.path
|
path = request.url.path
|
||||||
if path == "/api/v1/repos/alice/widget":
|
if path == "/api/v1/repos/alice/widget":
|
||||||
return httpx.Response(200, json={"default_branch": "main"})
|
return httpx.Response(200, json={"default_branch": "main"})
|
||||||
if path == "/api/v1/repos/alice/widget/archive/main.tar.gz":
|
if path in ("/api/v1/repos/alice/widget/archive/main.tar.gz",
|
||||||
|
"/api/v1/repos/alice/widget/archive/dev.tar.gz"):
|
||||||
return httpx.Response(200, content=tar_bytes)
|
return httpx.Response(200, content=tar_bytes)
|
||||||
return httpx.Response(404, json={"message": "not found"})
|
return httpx.Response(404, json={"message": "not found"})
|
||||||
|
|
||||||
@@ -258,7 +266,7 @@ async def test_coverage_measures_the_tree_exactly_and_caches(seeded):
|
|||||||
assert coverage["accounted"] == 2
|
assert coverage["accounted"] == 2
|
||||||
assert coverage["unclassified"] == 2
|
assert coverage["unclassified"] == 2
|
||||||
assert coverage["counts"] == {
|
assert coverage["counts"] == {
|
||||||
"canonical": 2, "instance": 0, "variant": 0, "exempt": 0,
|
"canonical": 2, "instance": 0, "variant": 0, "exempt": 0, "scoped": 0,
|
||||||
}
|
}
|
||||||
assert coverage["estimate"] is True
|
assert coverage["estimate"] is True
|
||||||
assert coverage["repos"] == [{
|
assert coverage["repos"] == [{
|
||||||
@@ -468,6 +476,11 @@ def test_extract_definitions_fingerprints_each_block():
|
|||||||
# And the identity view is unchanged for the hook mirror.
|
# And the identity view is unchanged for the hook mirror.
|
||||||
from scribe.services.coverage import extract_shapes
|
from scribe.services.coverage import extract_shapes
|
||||||
assert extract_shapes(text) == [(d.kind, d.name) for d in extract_definitions(text)]
|
assert extract_shapes(text) == [(d.kind, d.name) for d in extract_definitions(text)]
|
||||||
|
# A CSS rule's fingerprint is its declarations (#2872): the same body
|
||||||
|
# under another selector is the same shape to the derive grouping.
|
||||||
|
css = ".closed-msg {\n text-align: center;\n padding: 0.5rem 0;\n}\n.error-block {\n text-align: center;\n padding: 0.5rem 0;\n}\n.other {\n text-align: left;\n}\n"
|
||||||
|
d = {x.name: x for x in extract_definitions(css)}
|
||||||
|
assert d["closed-msg"].body_sha == d["error-block"].body_sha != d["other"].body_sha
|
||||||
|
|
||||||
|
|
||||||
def test_coverage_line_names_the_proposers_standing():
|
def test_coverage_line_names_the_proposers_standing():
|
||||||
@@ -484,6 +497,12 @@ def test_coverage_line_names_the_proposers_standing():
|
|||||||
assert "; 90 unclassified (40 proposed, 2 derive groups), largest: src" in line
|
assert "; 90 unclassified (40 proposed, 2 derive groups), largest: src" in line
|
||||||
line = coverage_line({**base, "proposed": 0, "derive_groups": [{"group": "a"}]})
|
line = coverage_line({**base, "proposed": 0, "derive_groups": [{"group": "a"}]})
|
||||||
assert "(1 derive group)" in line
|
assert "(1 derive group)" in line
|
||||||
|
# #2874: the next action on the line — biggest canon queue, widest copy.
|
||||||
|
line = coverage_line({
|
||||||
|
**base, "proposed": 40, "top_canon": {"snippet_id": 2844, "count": 78},
|
||||||
|
"derive_groups": [{"group": "dup:abc", "label": "closed-msg (identical body)", "files": 3}],
|
||||||
|
})
|
||||||
|
assert "top canon #2844 ×78" in line and "top copy closed-msg (identical body) ×3 files" in line
|
||||||
|
|
||||||
|
|
||||||
def test_coverage_line_names_divergence_and_recheck():
|
def test_coverage_line_names_divergence_and_recheck():
|
||||||
@@ -500,3 +519,52 @@ def test_coverage_line_names_divergence_and_recheck():
|
|||||||
assert line.endswith("; 1 judged shape changed since judged — recheck")
|
assert line.endswith("; 1 judged shape changed since judged — recheck")
|
||||||
assert "DIVERGENT" not in coverage_line(base)
|
assert "DIVERGENT" not in coverage_line(base)
|
||||||
assert "recheck" not in coverage_line(base)
|
assert "recheck" not in coverage_line(base)
|
||||||
|
|
||||||
|
|
||||||
|
def test_scoped_definitions_are_vue_script_setup_and_scoped_style_only():
|
||||||
|
"""#2869: one-offs by construction — every sym in a .vue and every css
|
||||||
|
rule inside <style scoped>; an unscoped <style> block and non-.vue files
|
||||||
|
stay ordinary."""
|
||||||
|
from scribe.services.coverage import extract_definitions, scoped_definitions
|
||||||
|
vue = (
|
||||||
|
"<script setup lang=\"ts\">\n"
|
||||||
|
"function load() {\n return 1;\n}\n"
|
||||||
|
"const save = async () => {\n return 2;\n};\n"
|
||||||
|
"</script>\n\n"
|
||||||
|
"<template><div class=\"card\"/></template>\n\n"
|
||||||
|
"<style scoped>\n.card {\n padding: 1rem;\n}\n.title {\n margin: 0;\n}\n</style>\n"
|
||||||
|
"<style>\n.global-toast {\n color: red;\n}\n</style>\n"
|
||||||
|
)
|
||||||
|
defs = extract_definitions(vue)
|
||||||
|
names = {(d.kind, d.name) for d in defs}
|
||||||
|
assert {("sym", "load"), ("sym", "save"), ("css", "card"), ("css", "title"), ("css", "global-toast")} <= names
|
||||||
|
scoped = scoped_definitions("frontend/src/views/A.vue", vue, defs)
|
||||||
|
assert scoped == {("sym", "load"), ("sym", "save"), ("css", "card"), ("css", "title")}
|
||||||
|
# Definitions know their line, which is what the scoped-style range uses.
|
||||||
|
assert next(d for d in defs if d.name == "card").line > next(d for d in defs if d.name == "save").line
|
||||||
|
# Not a .vue: nothing is scoped, whatever it contains.
|
||||||
|
assert scoped_definitions("frontend/src/assets/components.css", ".card {\n x: 1;\n}\n",
|
||||||
|
extract_definitions(".card {\n x: 1;\n}\n")) == set()
|
||||||
|
assert scoped_definitions("src/a.py", "def load():\n pass\n", extract_definitions("def load():\n pass\n")) == set()
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.integration
|
||||||
|
async def test_binding_ref_is_the_branch_the_ledger_follows(seeded):
|
||||||
|
"""#2873: a binding that names a ref is read at that ref (not the forge's
|
||||||
|
default branch); "" clears it; None on a re-bind leaves it standing."""
|
||||||
|
from scribe.services.coverage import compute_coverage
|
||||||
|
from scribe.services.repo_bindings import bindings_for_project, set_binding
|
||||||
|
uid, pid = seeded["uid"], seeded["pid"]
|
||||||
|
b = await set_binding(uid, "https://git.example.com/alice/widget.git", pid, "dev")
|
||||||
|
assert b.ref == "dev"
|
||||||
|
coverage = await compute_coverage(uid, pid, selector=_selector(_tarball(TREE)))
|
||||||
|
assert coverage["repos"][0]["ref"] == "dev"
|
||||||
|
# A re-bind without a ref keeps it; "" clears it back to the default branch.
|
||||||
|
b = await set_binding(uid, "https://git.example.com/alice/widget.git", pid)
|
||||||
|
assert b.ref == "dev"
|
||||||
|
b = await set_binding(uid, "https://git.example.com/alice/widget.git", pid, "")
|
||||||
|
assert b.ref is None
|
||||||
|
assert [x.ref for x in await bindings_for_project(uid, pid)] == [None]
|
||||||
|
coverage = await compute_coverage(uid, pid, selector=_selector(_tarball(TREE)))
|
||||||
|
assert coverage["repos"][0]["ref"] == "main"
|
||||||
|
|
||||||
|
|||||||
@@ -18,7 +18,7 @@ def test_backup_version_is_v8():
|
|||||||
point of the test — a payload section added without moving the version
|
point of the test — a payload section added without moving the version
|
||||||
produces backups that are structurally different and indistinguishable
|
produces backups that are structurally different and indistinguishable
|
||||||
by inspection."""
|
by inspection."""
|
||||||
assert backup.BACKUP_VERSION == 8
|
assert backup.BACKUP_VERSION == 9
|
||||||
|
|
||||||
|
|
||||||
def test_not_included_lists_the_known_gaps():
|
def test_not_included_lists_the_known_gaps():
|
||||||
@@ -115,7 +115,8 @@ async def test_export_full_backup_contains_every_declared_section():
|
|||||||
"topic_suppressions",
|
"topic_suppressions",
|
||||||
"systems", "record_systems", "design_systems",
|
"systems", "record_systems", "design_systems",
|
||||||
"design_tokens", "note_usage_events", "repo_bindings",
|
"design_tokens", "note_usage_events", "repo_bindings",
|
||||||
"note_supersessions", "code_shapes", "code_shape_events"):
|
"note_supersessions", "code_shapes", "code_shape_events",
|
||||||
|
"code_shape_uses"):
|
||||||
assert key in out, f"missing export section: {key}"
|
assert key in out, f"missing export section: {key}"
|
||||||
assert out[key] == []
|
assert out[key] == []
|
||||||
|
|
||||||
|
|||||||
+163
-1
@@ -30,7 +30,7 @@ def test_the_todo_state_is_the_default():
|
|||||||
assert CodeShape.__table__.c.status.default.arg == "unclassified"
|
assert CodeShape.__table__.c.status.default.arg == "unclassified"
|
||||||
assert "unclassified" in SHAPE_STATUSES
|
assert "unclassified" in SHAPE_STATUSES
|
||||||
assert set(SHAPE_STATUSES) == {
|
assert set(SHAPE_STATUSES) == {
|
||||||
"canonical", "instance", "variant", "exempt", "unclassified",
|
"canonical", "instance", "variant", "exempt", "scoped", "unclassified",
|
||||||
}
|
}
|
||||||
assert set(SHAPE_CLASSIFIERS) == {
|
assert set(SHAPE_CLASSIFIERS) == {
|
||||||
"agent", "audit", "hook", "mechanical", "import",
|
"agent", "audit", "hook", "mechanical", "import",
|
||||||
@@ -248,6 +248,49 @@ def test_match_canon_orders_bases_strongest_first_and_respects_kind():
|
|||||||
assert match_canon("sym", "x.py", "unrelated", "def unrelated(a, b, c, d, e):", "return 1", canons) is None
|
assert match_canon("sym", "x.py", "unrelated", "def unrelated(a, b, c, d, e):", "return 1", canons) is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_match_canon_gates_sym_bases_by_language_family():
|
||||||
|
"""A Python canon says nothing about a Vue body (and vice versa): the
|
||||||
|
2026-08 audit's worst proposals were `register` (MCP tool module, python)
|
||||||
|
offered for every auth view's handleSubmit that calls authStore.register()
|
||||||
|
and for a TS store's own `register`. Unknown language on either side →
|
||||||
|
no gate (the canons recorded without a language keep proposing)."""
|
||||||
|
from scribe.services.shape_ledger import Canon, _norm_text, match_canon, same_family
|
||||||
|
py_register = Canon(46, "sym", "register", (("src/scribe/mcp/tools/notes.py", "register"),),
|
||||||
|
"def register(mcp) -> None:", _norm_text("def register(mcp) -> None: ..."), 2, "python")
|
||||||
|
ts_helper = Canon(53, "sym", "apiErrorMessage", (("frontend/src/api/client.ts", "apiErrorMessage"),),
|
||||||
|
"export function apiErrorMessage(e: unknown, fallback: string): string {",
|
||||||
|
_norm_text("export function apiErrorMessage(e, fallback) { return fallback }"), 2, "typescript")
|
||||||
|
canons = [py_register, ts_helper]
|
||||||
|
vue_body = "async function handleSubmit() {\n await authStore.register(username.value);\n error.value = apiErrorMessage(e, 'x');\n}"
|
||||||
|
# The Vue handler references the TS helper, never the Python canon.
|
||||||
|
assert match_canon("sym", "frontend/src/views/RegisterView.vue", "handleSubmit",
|
||||||
|
"async function handleSubmit() {", vue_body, canons) == (53, "reference", 0.9)
|
||||||
|
# A TS store's own `register` is not a second definition of the Python one.
|
||||||
|
assert match_canon("sym", "frontend/src/stores/auth.ts", "register",
|
||||||
|
"async function register(u: string) {", "return apiPost('/api/auth/register', {u})",
|
||||||
|
[py_register]) is None
|
||||||
|
# Same family still proposes by symbol; unknown language still proposes.
|
||||||
|
assert match_canon("sym", "src/scribe/mcp/tools/other.py", "register",
|
||||||
|
"def register(mcp) -> None:", "pass", [py_register]) == (46, "symbol", 1.0)
|
||||||
|
unknown = py_register._replace(language="")
|
||||||
|
assert match_canon("sym", "frontend/src/stores/auth.ts", "register",
|
||||||
|
"async function register(u: string) {", "", [unknown]) == (46, "symbol", 1.0)
|
||||||
|
assert same_family("a.py", "python") and same_family("a.vue", "typescript")
|
||||||
|
assert same_family("a.py", "") and same_family("", "python")
|
||||||
|
assert not same_family("a.py", "vue")
|
||||||
|
|
||||||
|
|
||||||
|
def test_match_canon_reference_skips_generic_verbs():
|
||||||
|
"""A bare mention of `load`/`save`/`register` is not a call site of THIS
|
||||||
|
canon; the symbol basis still catches a second definition of the name."""
|
||||||
|
from scribe.services.shape_ledger import Canon, _norm_text, match_canon
|
||||||
|
loader = Canon(70, "sym", "load", (("frontend/src/components/A.vue", "load"),),
|
||||||
|
"async function load() {", _norm_text("async function load() { await fetch() }"), 2, "vue")
|
||||||
|
body = "async function refresh() {\n await load();\n}"
|
||||||
|
assert match_canon("sym", "frontend/src/components/B.vue", "refresh", "async function refresh() {", body, [loader]) is None
|
||||||
|
assert match_canon("sym", "frontend/src/components/B.vue", "load", "async function load() {", "", [loader]) == (70, "symbol", 1.0)
|
||||||
|
|
||||||
|
|
||||||
def test_match_canon_symbol_beats_everything_including_css_copies():
|
def test_match_canon_symbol_beats_everything_including_css_copies():
|
||||||
"""The previous test's css `btn-primary`-elsewhere case, stated plainly:
|
"""The previous test's css `btn-primary`-elsewhere case, stated plainly:
|
||||||
a second definition of the canon's own name is the symbol basis."""
|
a second definition of the canon's own name is the symbol basis."""
|
||||||
@@ -274,6 +317,32 @@ def test_derive_groups_copy_before_name_with_floors():
|
|||||||
assert ("i.py", "sym", "one") not in g
|
assert ("i.py", "sym", "one") not in g
|
||||||
|
|
||||||
|
|
||||||
|
def test_proposal_summary_ranks_body_identical_groups_first_and_sees_scoped_rows():
|
||||||
|
"""#2872: dup groups (the real copies) outrank name groups (usually
|
||||||
|
convention), wider spread first; #2869: scoped rows are in the readout."""
|
||||||
|
from scribe.models.code_shape import CodeShape
|
||||||
|
from scribe.services.shape_ledger import proposal_summary
|
||||||
|
|
||||||
|
def row(path, symbol, group, kind="css", status="scoped"):
|
||||||
|
return CodeShape(project_id=2, repo_key="r", path=path, symbol=symbol, kind=kind,
|
||||||
|
status=status, proposal_basis="derive", proposal_group=group)
|
||||||
|
rows = [
|
||||||
|
# a name group of 6 across 6 files
|
||||||
|
*[row(f"v/{i}.vue", "status-badge", "name:css:status-badge") for i in range(6)],
|
||||||
|
# a dup group of 3 across 3 files (different selector names, one body)
|
||||||
|
row("v/Login.vue", "closed-msg", "dup:abc"), row("v/Reset.vue", "error-block", "dup:abc"),
|
||||||
|
row("v/Forgot.vue", "success-msg", "dup:abc"),
|
||||||
|
# a dup group of 2 in ONE file — a copy, but not across files
|
||||||
|
row("v/A.vue", "x", "dup:def"), row("v/A.vue", "y", "dup:def"),
|
||||||
|
# judged rows never count
|
||||||
|
row("v/J.vue", "closed-msg", "dup:abc", status="exempt"),
|
||||||
|
]
|
||||||
|
out = proposal_summary(rows)
|
||||||
|
assert [g["group"] for g in out["derive_groups"]] == ["dup:abc", "dup:def", "name:css:status-badge"]
|
||||||
|
assert out["derive_groups"][0]["files"] == 3 and out["derive_groups"][0]["size"] == 3
|
||||||
|
assert out["derive_groups"][0]["label"] == "closed-msg (identical body)"
|
||||||
|
|
||||||
|
|
||||||
def test_confirm_requires_a_named_scope():
|
def test_confirm_requires_a_named_scope():
|
||||||
import asyncio
|
import asyncio
|
||||||
|
|
||||||
@@ -291,6 +360,99 @@ def test_proposer_tools_are_mounted():
|
|||||||
assert mcp._tool_manager.get_tool("confirm_shape_proposals") is not None
|
assert mcp._tool_manager.get_tool("confirm_shape_proposals") is not None
|
||||||
tool = mcp._tool_manager.get_tool("list_shapes")
|
tool = mcp._tool_manager.get_tool("list_shapes")
|
||||||
assert "proposal" in tool.parameters.get("properties", {})
|
assert "proposal" in tool.parameters.get("properties", {})
|
||||||
|
# #2868: the audit surfaces — compact pages and the sweep form.
|
||||||
|
assert "compact" in tool.parameters.get("properties", {})
|
||||||
|
rule = mcp._tool_manager.get_tool("classify_shapes_by_rule")
|
||||||
|
assert rule is not None
|
||||||
|
for name in ("path", "status", "pattern", "kind", "snippet_id", "reason", "include_judged"):
|
||||||
|
assert name in rule.parameters.get("properties", {}), name
|
||||||
|
|
||||||
|
|
||||||
|
# --- #2868: the bulk surfaces (pure) -----------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_compact_row_carries_identity_standing_and_the_proposers_word_only():
|
||||||
|
"""A 500-row compact page must fit the tool budget: no commits, shas or
|
||||||
|
timestamps; optional fields only when set."""
|
||||||
|
from scribe.models.code_shape import CodeShape
|
||||||
|
row = CodeShape(project_id=2, repo_key="r", path="src/a.py", symbol="f", kind="sym",
|
||||||
|
status="unclassified", signature="def f(x):", body_sha="abc",
|
||||||
|
first_seen_commit="c1", last_seen_commit="c2")
|
||||||
|
assert row.to_compact() == {
|
||||||
|
"path": "src/a.py", "symbol": "f", "kind": "sym",
|
||||||
|
"status": "unclassified", "signature": "def f(x):",
|
||||||
|
}
|
||||||
|
row.status, row.snippet_id, row.classified_by = "instance", 9, "audit"
|
||||||
|
row.proposed_snippet_id, row.proposal_basis, row.proposal_score = 9, "symbol", 1.0
|
||||||
|
compact = row.to_compact()
|
||||||
|
assert compact["snippet_id"] == 9 and compact["by"] == "audit"
|
||||||
|
assert compact["proposal"]["basis"] == "symbol"
|
||||||
|
for noisy in ("first_seen_commit", "last_seen_commit", "body_sha", "created_at", "classified_at"):
|
||||||
|
assert noisy not in compact
|
||||||
|
|
||||||
|
|
||||||
|
def test_uses_edges_table_and_validation():
|
||||||
|
"""#2870: consumption is its own relation — a table that cascades with
|
||||||
|
both ends, and `uses` on a classification must be a list of ids."""
|
||||||
|
from scribe.models import Base
|
||||||
|
from scribe.models.code_shape import USE_BASES, CodeShapeUse
|
||||||
|
from scribe.services.shape_ledger import validate_classifications
|
||||||
|
assert "code_shape_uses" in Base.metadata.tables
|
||||||
|
cols = CodeShapeUse.__table__.c
|
||||||
|
assert next(iter(cols.shape_id.foreign_keys)).ondelete == "CASCADE"
|
||||||
|
assert next(iter(cols.snippet_id.foreign_keys)).ondelete == "CASCADE"
|
||||||
|
assert set(USE_BASES) == {"reference", "hook", "agent", "audit", "import"}
|
||||||
|
ok = [{"path": "a.py", "symbol": "f", "status": "instance", "snippet_id": 9, "uses": [3, 4]}]
|
||||||
|
assert validate_classifications(ok) is None
|
||||||
|
bad = [{"path": "a.py", "symbol": "f", "status": "instance", "snippet_id": 9, "uses": "3"}]
|
||||||
|
assert "uses must be a list" in validate_classifications(bad)
|
||||||
|
|
||||||
|
|
||||||
|
def test_reference_canons_names_every_used_canon_not_just_the_best():
|
||||||
|
from scribe.services.shape_ledger import Canon, _norm_text, reference_canons
|
||||||
|
a = Canon(1, "sym", "hash_token", (("src/x.py", "hash_token"),), "def hash_token(raw):", _norm_text("x"), 2, "python")
|
||||||
|
b = Canon(2, "sym", "rules_payload", (("src/y.py", "rules_payload"),), "def rules_payload(r):", _norm_text("y"), 2, "python")
|
||||||
|
ts = Canon(3, "sym", "fmtDate", (("f/d.ts", "fmtDate"),), "export function fmtDate(iso: string): string {", _norm_text("z"), 2, "typescript")
|
||||||
|
body = "def create_invitation(email):\n h = hash_token(raw)\n return rules_payload(h)\n"
|
||||||
|
assert reference_canons("sym", "src/scribe/services/auth.py", "create_invitation", body, [a, b, ts]) == [1, 2]
|
||||||
|
# the shape's own name and the other language family are never "uses"
|
||||||
|
assert reference_canons("sym", "src/x.py", "hash_token", body, [a]) == []
|
||||||
|
assert reference_canons("sym", "f/v.vue", "show", "fmtDate(x); hash_token(y)", [a, ts]) == [3]
|
||||||
|
|
||||||
|
|
||||||
|
def test_reason_codes_are_a_fixed_catalogue_and_validated():
|
||||||
|
"""#2874: an optional index beside the prose reason; unknown codes are a
|
||||||
|
structural error (the batch applies nothing)."""
|
||||||
|
from scribe.models.code_shape import REASON_CODES
|
||||||
|
from scribe.services.shape_ledger import validate_classifications
|
||||||
|
assert set(REASON_CODES) == {
|
||||||
|
"scoped-css", "one-off-handler", "test-helper", "convention-plumbing",
|
||||||
|
"pure-helper", "generated", "script", "typed-record",
|
||||||
|
}
|
||||||
|
assert "reason_code" in CodeShape.__table__.c
|
||||||
|
ok = [{"path": "a.py", "symbol": "f", "status": "exempt", "reason": "x", "reason_code": "pure-helper"}]
|
||||||
|
assert validate_classifications(ok) is None
|
||||||
|
bad = [{"path": "a.py", "symbol": "f", "status": "exempt", "reason": "x", "reason_code": "nope"}]
|
||||||
|
assert "unknown reason_code" in validate_classifications(bad)
|
||||||
|
|
||||||
|
|
||||||
|
def test_rule_matches_is_directory_glob_and_kind_aware():
|
||||||
|
from scribe.models.code_shape import CodeShape
|
||||||
|
from scribe.services.shape_ledger import rule_matches
|
||||||
|
|
||||||
|
def row(path, symbol, kind="sym"):
|
||||||
|
return CodeShape(project_id=2, repo_key="r", path=path, symbol=symbol, kind=kind, status="unclassified")
|
||||||
|
|
||||||
|
r = row("frontend/src/views/LoginView.vue", "auth-card", "css")
|
||||||
|
assert rule_matches(r, path="frontend/src/views", pattern="", kind="")
|
||||||
|
assert rule_matches(r, path="frontend/src/views", pattern="auth-*", kind="css")
|
||||||
|
assert not rule_matches(r, path="frontend/src/views", pattern="auth-*", kind="sym")
|
||||||
|
assert not rule_matches(r, path="frontend/src/view", pattern="", kind="") # directory, not prefix
|
||||||
|
assert rule_matches(r, path="frontend/src/views/LoginView.vue", pattern="", kind="")
|
||||||
|
# CSS symbols compare without the leading dot, like everywhere else.
|
||||||
|
assert rule_matches(row("w/a.css", ".btn-primary", "css"), path="w", pattern="btn-*", kind="css")
|
||||||
|
assert rule_matches(row("src/scribe/services/backup.py", "_note_rows"), path="src/scribe/services", pattern="_*_rows", kind="")
|
||||||
|
assert not rule_matches(row("src/scribe/services/backup.py", "export_full_backup"), path="src/scribe/services", pattern="_*_rows", kind="")
|
||||||
|
|
||||||
|
|
||||||
# --- step 7: the divergence readout (pure) ----------------------------------
|
# --- step 7: the divergence readout (pure) ----------------------------------
|
||||||
|
|||||||
@@ -56,6 +56,7 @@ def test_no_parts_matches_everything():
|
|||||||
def test_matches_exact_repo_path_and_symbol():
|
def test_matches_exact_repo_path_and_symbol():
|
||||||
data = _data({"repo": "Scribe", "path": "src/scribe/x.py", "symbol": "helper"})
|
data = _data({"repo": "Scribe", "path": "src/scribe/x.py", "symbol": "helper"})
|
||||||
assert location_matches(data, {"repo": "Scribe"})
|
assert location_matches(data, {"repo": "Scribe"})
|
||||||
|
assert location_matches(data, {"repo": "scribe"}) # repo names: case never distinguishes (#2874)
|
||||||
assert location_matches(data, {"path": "src/scribe/x.py"})
|
assert location_matches(data, {"path": "src/scribe/x.py"})
|
||||||
assert location_matches(data, {"symbol": "helper"})
|
assert location_matches(data, {"symbol": "helper"})
|
||||||
assert location_matches(data, {"repo": "Scribe", "symbol": "helper"})
|
assert location_matches(data, {"repo": "Scribe", "symbol": "helper"})
|
||||||
@@ -101,7 +102,8 @@ def test_blank_recorded_part_does_not_match_a_requested_one():
|
|||||||
def test_jsonpath_filters_within_one_locations_entry():
|
def test_jsonpath_filters_within_one_locations_entry():
|
||||||
expr = location_jsonpath({"repo": "Scribe", "symbol": "helper"})
|
expr = location_jsonpath({"repo": "Scribe", "symbol": "helper"})
|
||||||
assert expr.startswith("$.locations[*] ? (")
|
assert expr.startswith("$.locations[*] ? (")
|
||||||
assert '@.repo == "Scribe"' in expr
|
# repo: anchored, case-insensitive (#2874) — mirrors location_matches.
|
||||||
|
assert '@.repo like_regex "^Scribe$" flag "i"' in expr
|
||||||
assert '@.symbol == "helper"' in expr
|
assert '@.symbol == "helper"' in expr
|
||||||
assert " && " in expr
|
assert " && " in expr
|
||||||
|
|
||||||
@@ -121,8 +123,9 @@ def test_jsonpath_quotes_values_as_json_literals():
|
|||||||
"""A quote in a repo name must stay inside the literal, not end it."""
|
"""A quote in a repo name must stay inside the literal, not end it."""
|
||||||
nasty = 'we"ird'
|
nasty = 'we"ird'
|
||||||
expr = location_jsonpath({"repo": nasty})
|
expr = location_jsonpath({"repo": nasty})
|
||||||
assert json.dumps(nasty) in expr
|
assert json.dumps("^" + nasty + "$") in expr # the regex is a JSON literal too
|
||||||
assert '\\"' in expr
|
assert '\\"' in expr
|
||||||
|
assert json.dumps(nasty) in location_jsonpath({"symbol": nasty})
|
||||||
|
|
||||||
|
|
||||||
def test_jsonpath_emits_parts_in_a_fixed_key_order():
|
def test_jsonpath_emits_parts_in_a_fixed_key_order():
|
||||||
|
|||||||
Reference in New Issue
Block a user