"""Pattern-library coverage — what fraction of a bound repo's shapes have a
recorded snippet (#2692, forge job 3 of decision #2686).
The all-shapes doctrine says every shape gets recorded at first build. This
module is the hoping→knowing move: it enumerates the definitions that exist in
a project's bound repos (via the forge, one archive download per repo) and
compares them against recorded snippet locations, so "record everything"
becomes a watched number instead of an aspiration.
The definition extractor MIRRORS the write-path hook's awk rules
(plugin/hooks/scribe_prior_art.sh, ARM 1) — one shared notion of "a
definition" between the hook and the server, so the metric and the backstop
agree on what counts. The two are pinned together by shared test vectors in
tests/test_pattern_coverage.py; change one, change both.
The number is an ESTIMATE and every surface must say so: keyword extraction
over-counts (private one-offs, generated code that slips the dir filter) and
under-counts (keyword-less declaration syntax — C/Java/Dart — needs a real
parser and is out of scope, exactly as it is for the hook). The trend carries
the meaning, like the usage counters; the raw number is not a grade.
Compute is on demand + cached with a freshness stamp — recomputed on webhook
push and explicit refresh, NEVER in the request path of enter_project, which
only ever reads the cache.
"""
from __future__ import annotations
import hashlib
import io
import json
import logging
import posixpath
import re
import tarfile
from typing import NamedTuple
from datetime import datetime, timedelta, timezone
from scribe.services.forge import ForgeSelector, get_forges
from scribe.services.repo_bindings import bindings_for_project
from scribe.services.settings import get_setting, set_setting
logger = logging.getLogger(__name__)
# Cache key in the settings KV, on the project OWNER's user_id — the same
# channel the scheduler's last-run summary uses for machine-written state.
# v2 suffix with #2788: the payload shape inverted (accounted/unclassified);
# pre-ledger blobs under the old key simply stop being found, so the card
# honestly reads "not measured yet" until the first ledger-era refresh.
_CACHE_KEY_PREFIX = "pattern_coverage_v2_"
# Files whose content can't hold definitions — the hook's skip list, verbatim,
# plus sourcemaps (which are JSON in a trenchcoat).
_SKIP_SUFFIXES = (
".md", ".mdx", ".txt", ".rst", ".json", ".lock", ".log", ".csv", ".tsv",
".svg", ".png", ".jpg", ".jpeg", ".gif", ".ico", ".pdf", ".map",
)
# Vendored/generated trees would swamp the metric with shapes nobody should
# record — the dunder-skip lesson at directory scale: guaranteed noise teaches
# people to ignore the number.
_SKIP_DIRS = frozenset({
"node_modules", "vendor", "dist", "build", "target",
"__pycache__", ".git", ".venv", "venv",
})
# A single source file bigger than this is almost certainly generated or
# vendored (bundles, lockstep protos) — skipped, and part of why the number
# is labeled an estimate.
_MAX_FILE_BYTES = 1_000_000
# --- the definition extractor (mirror of scribe_prior_art.sh ARM 1) ----------
_CSS_RE = re.compile(r"^\s*\.([A-Za-z][A-Za-z0-9_-]*)\s*[,{]")
# Leading declaration modifiers, so the definition keyword is the first word
# regardless of language (export/pub/private/suspend/...).
_MODIFIERS_RE = re.compile(
r"^(?:(?:pub(?:\([a-z]+\))?|export|default|private|internal|protected"
r"|public|static|suspend|async|open|sealed|data|abstract|final|inline"
r"|unsafe|extern|override)\s+)*"
)
# Go method with receiver: func (r *T) Name(
_GO_METHOD_RE = re.compile(r"^func\s*\([^)]*\)\s*([A-Za-z_][A-Za-z0-9_]*)")
# Keyword-announced definitions, functions and named types alike. `impl` is
# excluded on purpose — several per type is normal Rust, not duplication.
_KEYWORD_RE = re.compile(
r"^(?:function|def|class|func|fun|fn|sub|struct|trait|interface|enum"
r"|object|protocol|type)\s+([A-Za-z_$][A-Za-z0-9_$]*)"
)
# Arrow/expression assignment: const name = (…) / let name = async (
_ARROW_RE = re.compile(
r"^(?:const|let)\s+([A-Za-z_$][A-Za-z0-9_$]*)\s*=\s*(?:async\s*)?[(<]"
)
# --- comment and string spans (#4222) ----------------------------------------
#
# The matchers above are line-oriented and know nothing about what a line is
# INSIDE. A wrapped docstring whose line happens to begin "class AND the …"
# reads as a definition of a shape called `AND`, and that phantom reaches the
# agent mid-edit as a divergence prompt about a symbol that does not exist.
# This repo's own source minted twenty-two of them, measured by running the
# extractor over every scannable file with and without this scan. Two —
# `with` and `nobody`, out of one module docstring in
# scripts/check_dangling_styles.py — are persisted, JUDGED `code_shapes`
# rows, so the cost was never only noise in the moment. Those clear
# themselves: sync_shapes marks a row it no longer extracts as vanished.
#
# The honest tool for .py would be `ast`, which cannot be fooled by prose at
# all. It is not what this uses, because this extractor is mirrored rule for
# rule by an awk program in plugin/hooks/scribe_defs.sh, awk cannot parse
# Python, and a fix only one of the pair can run is the drift the mirror
# exists to prevent. What both can do is blank the SPANS: a triple-quoted
# string or a /* … */ comment is replaced by its own newlines before any
# matcher sees a line, so every line index still lines up and the signature,
# body and fingerprint keep reading the untouched original.
#
# ONLY CLOSED SPANS ARE BLANKED, and an unterminated opener is stepped over
# rather than bailed on, so a stray opener costs one mishandled span and never
# every definition below it in the file.
_SPANS = (('"""', '"""'), ("'''", "'''"), ("/*", "*/"))
def _string_end(text: str, at: int) -> int:
"""Offset just past the one-line string opening at ``at``.
``at`` + 1 when it does not close before the newline, so an apostrophe in
prose — `don't` in a Vue template, outside any comment — costs one
character rather than everything up to the next quote in the file.
"""
quote = text[at]
i = at + 1
while i < len(text):
c = text[i]
if c == "\\":
i += 2
elif c == "\n":
return at + 1
elif c == quote:
return i + 1
else:
i += 1
return at + 1
def _hash_comment(text: str, at: int) -> bool:
"""Does the `#` at ``at`` open a comment, or is it a CSS colour or id?
`#` is the one marker whose meaning depends on the language, and the
extractor is handed text with no path. An alphanumeric straight after it
is `#fff` or `#app`; anything else — a space, a `!`, a `-` — is a comment
in every language that has one.
"""
nxt = text[at + 1:at + 2]
return not nxt.isalnum()
def _blank_spans(text: str) -> str:
"""``text`` with comment and string spans replaced by their own newlines.
A single left-to-right pass, because the alternative — matching markers
wherever they appear — cannot tell a comment from a string that QUOTES
one. Both of those are in this repo: the docstring-matching regex in
plugin_context.py holds a triple quote inside a single-quoted literal,
and test_design_stylesheet.py asserts on the text `red /*` inside a
double-quoted one. Each cost every definition below it in its file before
the scan was written this way.
Line COUNT is preserved, column positions are not — the result is only
ever fed to the line matchers, which lstrip anyway.
"""
out: list[str] = []
i = cut = 0
n = len(text)
while i < n:
opener = closer = ""
for op, cl in _SPANS:
if text.startswith(op, i):
opener, closer = op, cl
break
if opener:
end = text.find(closer, i + len(opener))
if end < 0:
# Unterminated: step over the opener rather than bail, so a
# stray marker costs one span and not the rest of the file.
i += len(opener)
continue
end += len(closer)
out.append(text[cut:i])
out.append("\n" * text.count("\n", i, end))
i = cut = end
continue
if text.startswith("//", i) or (text[i] == "#" and _hash_comment(text, i)):
# A line comment is COPIED, not blanked: its continuation lines
# carry their own marker, so none of them can read as a
# definition on their own.
nl = text.find("\n", i)
i = n if nl < 0 else nl
continue
if text[i] in "\"'":
i = _string_end(text, i)
continue
i += 1
out.append(text[cut:])
return "".join(out)
def _masked_lines(text: str, count: int) -> list[str]:
"""``text`` blanked and split, reconciled to ``count`` lines.
Blanking preserves every ``\n``, but ``splitlines`` also breaks on a bare
``\r`` and on the vertical-tab family, which a blanked span drops. The
reconciliation is what keeps a span containing one of those from shifting
suppression onto the wrong lines — padding is short by a line, never
misaligned by one.
"""
masked = _blank_spans(text).splitlines()
if len(masked) < count:
masked += [""] * (count - len(masked))
return masked[:count]
class Definition(NamedTuple):
"""One extracted definition with its content fingerprint (#2792).
`signature` is the definition line itself; `body_sha` hashes the block
whitespace- and comment-insensitively; `body` is the block's text, held
only for the duration of a refresh (the proposer matches on it) and
never stored.
"""
kind: str
name: str
signature: str
body_sha: str
body: str
line: int = -1 # 0-based line the definition starts on (#2869)
def _definition_on(raw: str) -> tuple[str, str] | None:
"""The (kind, name) this one line defines, or None. First match wins —
the same order the hook's awk program tries."""
m = _CSS_RE.match(raw)
if m:
return ("css", m.group(1))
line = _MODIFIERS_RE.sub("", raw.lstrip())
if m := _GO_METHOD_RE.match(line):
return ("sym", m.group(1))
if m := _KEYWORD_RE.match(line):
name = m.group(1)
if name.startswith("__") and name.endswith("__"):
return None
# `type` announces a definition only when something is declared after
# the name (`type Foo = …`, `type Foo struct {`); an import specifier
# (`import { type Foo, bar }`) is the same two words and defines
# nothing — it showed up as a two-file "identical body" family (#2904).
if line.startswith("type") and not re.search(r"[={]", line[m.end():]):
return None
return ("sym", name)
if m := _ARROW_RE.match(line):
return ("sym", m.group(1))
return None
# A definition's block runs from its line until the next non-blank line at
# its own indentation or shallower that is not a closer — so a Python def ends
# at the next top-level statement, a braces block keeps its `}`, a CSS rule
# keeps its `}`. Capped so a generated monolith can't make one shape's
# fingerprint cover the file.
_BLOCK_CAP = 120
_CLOSERS = ("}", ")", "]", "end", "};", "});", ");", "})", "]);")
# Lines that don't change what a shape IS: comments and decorators. Dropped
# from the fingerprint so touching a comment above the next function doesn't
# read as this one's body changing.
_NOISE_PREFIXES = ("#", "//", "/*", "*", "*/", "@", "")
_SIGNATURE_CAP = 300
def _indent(line: str) -> int:
return len(line) - len(line.lstrip())
def _block_sha(lines: list[str]) -> str:
kept = [
" ".join(ln.split())
for ln in lines
if ln.strip() and not ln.lstrip().startswith(_NOISE_PREFIXES)
]
return hashlib.sha1("\n".join(kept).encode("utf-8")).hexdigest()[:16]
def _declaration_count(lines: list[str]) -> int:
"""How many `prop: value` declarations a CSS block body carries."""
body = " ".join(lines)
return sum(1 for part in body.replace("}", "").split(";") if ":" in part)
def extract_definitions(text: str) -> list[Definition]:
"""Every definition this text makes, with signature + fingerprint.
Duplicate (kind, name) within one text collapse to the first — the
ledger's identity is per file, so a second definition of the same name
(an overload, a re-declaration) is the same shape to it.
"""
lines = text.splitlines()
masked = _masked_lines(text, len(lines))
starts: list[tuple[int, str, str]] = []
for i, raw in enumerate(masked):
hit = _definition_on(raw)
if hit:
starts.append((i, hit[0], hit[1]))
seen: set[tuple[str, str]] = set()
out: list[Definition] = []
for i, kind, name in starts:
if (kind, name) in seen:
continue
seen.add((kind, name))
base = _indent(lines[i])
end = min(len(lines), i + _BLOCK_CAP)
for j in range(i + 1, min(len(lines), i + _BLOCK_CAP)):
ln = lines[j]
if not ln.strip():
continue
if _indent(ln) <= base and ln.strip() not in _CLOSERS:
end = j
break
block = lines[i:end]
# A CSS rule's fingerprint is its DECLARATIONS, not its selector
# (#2872): the row's identity already carries the selector. Since
# note 2917 the derive grouping no longer reads CSS bodies at all (a
# class is grouped by name only), so for CSS the fingerprint is the
# recheck identity — "did this rule's body change since it was
# judged?" — and nothing more. The shape of the hash is kept as-is on
# purpose: changing it would flip every judged CSS row to recheck on
# the next sync. Sym blocks keep their signature line in the hash.
if kind == "css":
# One-line rules (`.x { color: red; }`) carry their declarations on
# the selector line itself; a block that is only the selector plus
# trailing blanks must not hash to the empty string (which grouped
# 68 unrelated one-liners as one "copy" on first deploy, #2872).
first = lines[i]
brace = first.find("{")
head = [first[brace + 1:]] if brace >= 0 and first[brace + 1:].strip() else []
hashed = head + block[1:]
if not any(x.strip() for x in hashed):
hashed = block
# A SINGLE declaration is not a shape (#2903): `color: var(--fs-
# text-tertiary)` under .text-muted, .task-mark and .pin-badge-auto
# is three meanings sharing one line, not three copies of one
# rule. Keep the selector in the hash for one-liners; two
# declarations and up stay selector-agnostic. (Moot for grouping
# since note 2917, kept for fingerprint stability — see above.)
elif _declaration_count(hashed) < 2:
hashed = block
else:
hashed = block
out.append(Definition(
kind, name, lines[i].strip()[:_SIGNATURE_CAP], _block_sha(hashed),
"\n".join(block), i,
))
return out
# --- by-construction scope (#2869) -------------------------------------------
#
# A Vue single-file component's `", re.IGNORECASE)
# Svelte scopes the other way round (#4608): every `