Rule outcomes, the contract hint, and four extractor/backup fixes #174

Merged
bvandeusen merged 18 commits from dev into main 2026-09-21 06:31:10 -04:00
3 changed files with 320 additions and 11 deletions
Showing only changes of commit 84476d7ecf - Show all commits
+120 -10
View File
@@ -303,15 +303,118 @@ scribe_urlenc() {
# or the one enclosing an Edit). Rule-for-rule mirrored by the server's
# services/coverage.py extract_shapes — ledger rows are keyed by what THAT
# sees, so the two must agree on what counts as a definition.
#
# THE WHOLE INPUT IS BUFFERED (#4222) so the span scan below can look ahead.
# The matchers are line-oriented and know nothing about what a line is INSIDE:
# a wrapped docstring beginning "class AND the …" announces a shape called
# `AND`, which reaches the session mid-edit as a divergence prompt about a
# symbol that does not exist. blank_spans() replaces every comment and string
# span with its own newlines before a matcher sees a line — the same scan, in
# the same order, as coverage.py::_blank_spans. Change one, change both.
scribe_defs() {
awk '
{
# CSS class definition: .name { or .name,
if (match($0, /^[[:space:]]*\.[A-Za-z][A-Za-z0-9_-]*[[:space:]]*[,{]/)) {
t = $0; sub(/^[[:space:]]*\./, "", t); sub(/[[:space:]]*[,{].*$/, "", t)
if (t != "") print "css\t" t; next
BEGIN {
SQ = sprintf("%c", 39)
SQ3 = SQ SQ SQ
DQ = "\""
DQ3 = DQ DQ DQ
# The only characters that can begin a span, a line comment or a
# string. Everything between two of them is copied in one go rather
# than a character at a time.
MARKERS = "[" DQ SQ "/#]"
}
# Offset just past the one-line string opening at c, or c+1 when it does
# not close before the end of the line — so an apostrophe in prose costs
# one character rather than everything up to the next quote.
function string_end(L, c, q, i, n, ch) {
n = length(L); i = c + 1
while (i <= n) {
ch = substr(L, i, 1)
if (ch == "\\") { i = i + 2; continue }
if (ch == q) return i + 1
i++
}
line = $0; sub(/^[[:space:]]+/, "", line)
return c + 1
}
# Does the "#" at c open a comment, or is it a CSS colour or id? An
# alphanumeric straight after it is #fff or #app; anything else is a
# comment in every language that has one.
function hash_comment(L, c) {
return substr(L, c + 1, 1) !~ /^[A-Za-z0-9]$/
}
# raw[1..n] -> msk[1..n] with comment and string spans emptied. Line
# COUNT is preserved and column positions are not; the matchers lstrip.
# ONLY CLOSED SPANS ARE BLANKED: an opener with no closer is rewound past
# and scanning resumes, so a stray marker costs one span rather than
# every definition below it.
function blank_spans(raw, n, msk,
i, c, L, len, state, closer, oplen, sl, sc, sprefix,
t3, t2, ch, e, k, rest, m) {
for (i = 1; i <= n; i++) msk[i] = ""
i = 1; c = 1; state = 0; closer = ""
while (1) {
while (i <= n) {
L = raw[i]; len = length(L)
if (c > len) { i++; c = 1; continue }
if (state) {
e = index(substr(L, c), closer)
if (e == 0) { i++; c = 1; continue }
c = c + e - 1 + length(closer)
state = 0; closer = ""
continue
}
rest = substr(L, c)
m = match(rest, MARKERS)
if (m == 0) { msk[i] = msk[i] rest; i++; c = 1; continue }
if (m > 1) {
msk[i] = msk[i] substr(rest, 1, m - 1)
c = c + m - 1
continue
}
t3 = substr(L, c, 3); t2 = substr(L, c, 2); ch = substr(L, c, 1)
if (t3 == DQ3 || t3 == SQ3) {
sl = i; sc = c; sprefix = msk[i]
state = 1; closer = t3; oplen = 3; c = c + 3
continue
}
if (t2 == "/*") {
sl = i; sc = c; sprefix = msk[i]
state = 1; closer = "*/"; oplen = 2; c = c + 2
continue
}
if (t2 == "//" || (ch == "#" && hash_comment(L, c))) {
# A line comment is COPIED, not blanked: its continuation lines
# carry their own marker, so none can read as a definition alone.
msk[i] = msk[i] substr(L, c)
i++; c = 1
continue
}
if (ch == DQ || ch == SQ) {
k = string_end(L, c, ch)
msk[i] = msk[i] substr(L, c, k - c)
c = k
continue
}
msk[i] = msk[i] ch
c++
}
if (!state) return
for (k = sl; k <= n; k++) msk[k] = ""
msk[sl] = sprefix
i = sl; c = sc + oplen; state = 0; closer = ""
}
}
function emit(line, t, rest) {
# CSS class definition: .name { or .name,
if (match(line, /^[[:space:]]*\.[A-Za-z][A-Za-z0-9_-]*[[:space:]]*[,{]/)) {
t = line; sub(/^[[:space:]]*\./, "", t); sub(/[[:space:]]*[,{].*$/, "", t)
if (t != "") print "css\t" t; return
}
sub(/^[[:space:]]+/, "", line)
# Strip leading declaration modifiers so the definition keyword is the
# first word regardless of language (export/pub/private/suspend/...).
sub(/^((pub(\([a-z]+\))?|export|default|private|internal|protected|public|static|suspend|async|open|sealed|data|abstract|final|inline|unsafe|extern|override)[[:space:]]+)*/, "", line)
@@ -319,7 +422,7 @@ scribe_defs() {
if (match(line, /^func[[:space:]]*\([^)]*\)[[:space:]]*[A-Za-z_]/)) {
t = line; sub(/^func[[:space:]]*\([^)]*\)[[:space:]]*/, "", t)
sub(/[^A-Za-z0-9_].*$/, "", t)
if (t != "") print "sym\t" t; next
if (t != "") print "sym\t" t; return
}
# Keyword-announced definitions, functions and named types alike.
# Dunders are skipped: every class defines __init__, so "already defined
@@ -333,17 +436,24 @@ scribe_defs() {
# nothing (mirror of coverage.py, #2904).
if (line ~ /^type[[:space:]]/) {
rest = line; sub(/^type[[:space:]]+[A-Za-z_$][A-Za-z0-9_$]*/, "", rest)
if (rest !~ /[={]/) next
if (rest !~ /[={]/) return
}
if (t != "" && t !~ /^__.*__$/) print "sym\t" t; next
if (t != "" && t !~ /^__.*__$/) print "sym\t" t; return
}
# Arrow/expression assignment: const name = (…) / let name = async (
if (match(line, /^(const|let)[[:space:]]+[A-Za-z_$][A-Za-z0-9_$]*[[:space:]]*=[[:space:]]*(async[[:space:]]*)?[(<]/)) {
t = line; sub(/^(const|let)[[:space:]]+/, "", t)
sub(/[^A-Za-z0-9_$].*$/, "", t)
if (t != "") print "sym\t" t; next
if (t != "") print "sym\t" t; return
}
}
{ raw[NR] = $0 }
END {
blank_spans(raw, NR, msk)
for (r = 1; r <= NR; r++) emit(msk[r])
}
' 2>/dev/null
}
+129 -1
View File
@@ -93,6 +93,133 @@ _ARROW_RE = re.compile(
)
# --- comment and string spans (#4222) ----------------------------------------
#
# The matchers above are line-oriented and know nothing about what a line is
# INSIDE. A wrapped docstring whose line happens to begin "class AND the …"
# reads as a definition of a shape called `AND`, and that phantom reaches the
# agent mid-edit as a divergence prompt about a symbol that does not exist.
# This repo's own source minted twenty-two of them, measured by running the
# extractor over every scannable file with and without this scan. Two —
# `with` and `nobody`, out of one module docstring in
# scripts/check_dangling_styles.py — are persisted, JUDGED `code_shapes`
# rows, so the cost was never only noise in the moment. Those clear
# themselves: sync_shapes marks a row it no longer extracts as vanished.
#
# The honest tool for .py would be `ast`, which cannot be fooled by prose at
# all. It is not what this uses, because this extractor is mirrored rule for
# rule by an awk program in plugin/hooks/scribe_defs.sh, awk cannot parse
# Python, and a fix only one of the pair can run is the drift the mirror
# exists to prevent. What both can do is blank the SPANS: a triple-quoted
# string or a /* … */ comment is replaced by its own newlines before any
# matcher sees a line, so every line index still lines up and the signature,
# body and fingerprint keep reading the untouched original.
#
# ONLY CLOSED SPANS ARE BLANKED, and an unterminated opener is stepped over
# rather than bailed on, so a stray opener costs one mishandled span and never
# every definition below it in the file.
_SPANS = (('"""', '"""'), ("'''", "'''"), ("/*", "*/"))
def _string_end(text: str, at: int) -> int:
"""Offset just past the one-line string opening at ``at``.
``at`` + 1 when it does not close before the newline, so an apostrophe in
prose — `don't` in a Vue template, outside any comment — costs one
character rather than everything up to the next quote in the file.
"""
quote = text[at]
i = at + 1
while i < len(text):
c = text[i]
if c == "\\":
i += 2
elif c == "\n":
return at + 1
elif c == quote:
return i + 1
else:
i += 1
return at + 1
def _hash_comment(text: str, at: int) -> bool:
"""Does the `#` at ``at`` open a comment, or is it a CSS colour or id?
`#` is the one marker whose meaning depends on the language, and the
extractor is handed text with no path. An alphanumeric straight after it
is `#fff` or `#app`; anything else — a space, a `!`, a `-` — is a comment
in every language that has one.
"""
nxt = text[at + 1:at + 2]
return not nxt.isalnum()
def _blank_spans(text: str) -> str:
"""``text`` with comment and string spans replaced by their own newlines.
A single left-to-right pass, because the alternative — matching markers
wherever they appear — cannot tell a comment from a string that QUOTES
one. Both of those are in this repo: the docstring-matching regex in
plugin_context.py holds a triple quote inside a single-quoted literal,
and test_design_stylesheet.py asserts on the text `red /*` inside a
double-quoted one. Each cost every definition below it in its file before
the scan was written this way.
Line COUNT is preserved, column positions are not — the result is only
ever fed to the line matchers, which lstrip anyway.
"""
out: list[str] = []
i = cut = 0
n = len(text)
while i < n:
opener = closer = ""
for op, cl in _SPANS:
if text.startswith(op, i):
opener, closer = op, cl
break
if opener:
end = text.find(closer, i + len(opener))
if end < 0:
# Unterminated: step over the opener rather than bail, so a
# stray marker costs one span and not the rest of the file.
i += len(opener)
continue
end += len(closer)
out.append(text[cut:i])
out.append("\n" * text.count("\n", i, end))
i = cut = end
continue
if text.startswith("//", i) or (text[i] == "#" and _hash_comment(text, i)):
# A line comment is COPIED, not blanked: its continuation lines
# carry their own marker, so none of them can read as a
# definition on their own.
nl = text.find("\n", i)
i = n if nl < 0 else nl
continue
if text[i] in "\"'":
i = _string_end(text, i)
continue
i += 1
out.append(text[cut:])
return "".join(out)
def _masked_lines(text: str, count: int) -> list[str]:
"""``text`` blanked and split, reconciled to ``count`` lines.
Blanking preserves every ``\n``, but ``splitlines`` also breaks on a bare
``\r`` and on the vertical-tab family, which a blanked span drops. The
reconciliation is what keeps a span containing one of those from shifting
suppression onto the wrong lines — padding is short by a line, never
misaligned by one.
"""
masked = _blank_spans(text).splitlines()
if len(masked) < count:
masked += [""] * (count - len(masked))
return masked[:count]
class Definition(NamedTuple):
"""One extracted definition with its content fingerprint (#2792).
@@ -176,8 +303,9 @@ def extract_definitions(text: str) -> list[Definition]:
(an overload, a re-declaration) is the same shape to it.
"""
lines = text.splitlines()
masked = _masked_lines(text, len(lines))
starts: list[tuple[int, str, str]] = []
for i, raw in enumerate(lines):
for i, raw in enumerate(masked):
hit = _definition_on(raw)
if hit:
starts.append((i, hit[0], hit[1]))
+71
View File
@@ -10,7 +10,10 @@ tripwire.
"""
import io
import json
import shutil
import subprocess
import tarfile
from pathlib import Path
import pytest
import pytest_asyncio
@@ -28,6 +31,8 @@ from scribe.services.coverage import (
from scribe.services.shape_ledger import location_covers
from tests.helpers import ensure_user
PLUGIN = Path(__file__).resolve().parents[1] / "plugin"
# --- unit: the definition extractor (shared vectors with the hook) -----------
EXTRACTION_VECTORS = [
@@ -63,6 +68,41 @@ EXTRACTION_VECTORS = [
[]),
("dedup-within-file", "def f():\n pass\ndef f():\n pass\n",
[("sym", "f")]),
# --- comment and string spans (#4222) ------------------------------------
# Prose is not code. A wrapped docstring line beginning "class AND the"
# announced a shape called `AND` to a live session; `with` and `nobody`
# out of one module docstring in scripts/check_dangling_styles.py reached
# persisted, judged `code_shapes` rows.
("docstring-prose",
'def real_one():\n """Its own text is about the\n'
' class AND the to_dict, and is a def bar():\n'
' class Foo: lives here too.\n """\n pass\n',
[("sym", "real_one")]),
# An opener with no closer blanks NOTHING: the scan rewinds past it, so a
# stray marker costs one span rather than the rest of the file.
("unterminated-docstring",
'def before():\n """oops, never closed\n\ndef after():\n pass\n',
[("sym", "before"), ("sym", "after")]),
# The CSS half of the same defect (#2990): a wrapped comment line that
# happens to begin with a dotted token reads as a selector.
("css-comment-selector",
"/* A real base rule, not just descendants: the check reads a\n"
" .ghost, class that only ever appears as an ancestor */\n.check { }\n",
[("css", "check")]),
# A string that HOLDS a comment marker is not a comment — the case that
# made the first draft of the scan eat 70 lines of live code.
("string-holds-a-marker",
'SAMPLE = "red /* "\ndef after_the_string():\n pass\n',
[("sym", "after_the_string")]),
# `#` is the one marker whose meaning is the language\'s: a colour here,
# a comment two lines down, and the extractor is handed no path.
("hash-is-a-colour-not-a-comment",
".a { color: #fff; } /* .ghost,\n class Phantom: */\n.b { }\n",
[("css", "a"), ("css", "b")]),
("line-comment-mentioning-a-docstring",
'def kept():\n pass\n# a stray """ in a comment\n'
'def also_kept():\n pass\n',
[("sym", "kept"), ("sym", "also_kept")]),
]
@@ -75,6 +115,37 @@ def test_extractor_agrees_with_the_hook_on_what_defines(text, expected):
assert extract_shapes(text) == expected
@pytest.mark.parametrize(
"text",
[t for _i, t, _e in EXTRACTION_VECTORS],
ids=[i for i, _t, _e in EXTRACTION_VECTORS],
)
def test_the_hook_extractor_runs_and_agrees_line_for_line(text):
"""The comment at the top of this module has been the only thing holding
the two extractors together, and a comment cannot fail. This RUNS the
hook's awk program over the same vectors.
The hook emits every definition in source order with no dedup — identity
there is per payload, not per file — so the comparison de-dupes its
output before matching, which is the one difference between the two that
is by design.
"""
if shutil.which("awk") is None: # pragma: no cover - env guard
pytest.skip("awk not available")
lib = PLUGIN / "hooks" / "scribe_defs.sh"
out = subprocess.run(
["bash", "-c", f'. "{lib}"; scribe_defs'],
input=text, capture_output=True, text=True,
)
assert out.returncode == 0, out.stderr
seen: list[tuple[str, str]] = []
for line in out.stdout.splitlines():
kind, _, name = line.partition("\t")
if name and (kind, name) not in seen:
seen.append((kind, name))
assert seen == extract_shapes(text)
def test_scannable_gates_prose_vendored_and_sourcemaps():
assert scannable("src/app.py")
assert scannable("web/button.css")