diff --git a/plugin/hooks/scribe_defs.sh b/plugin/hooks/scribe_defs.sh index 02ed033..18a62a8 100644 --- a/plugin/hooks/scribe_defs.sh +++ b/plugin/hooks/scribe_defs.sh @@ -303,15 +303,118 @@ scribe_urlenc() { # or the one enclosing an Edit). Rule-for-rule mirrored by the server's # services/coverage.py extract_shapes — ledger rows are keyed by what THAT # sees, so the two must agree on what counts as a definition. +# +# THE WHOLE INPUT IS BUFFERED (#4222) so the span scan below can look ahead. +# The matchers are line-oriented and know nothing about what a line is INSIDE: +# a wrapped docstring beginning "class AND the …" announces a shape called +# `AND`, which reaches the session mid-edit as a divergence prompt about a +# symbol that does not exist. blank_spans() replaces every comment and string +# span with its own newlines before a matcher sees a line — the same scan, in +# the same order, as coverage.py::_blank_spans. Change one, change both. scribe_defs() { awk ' - { - # CSS class definition: .name { or .name, - if (match($0, /^[[:space:]]*\.[A-Za-z][A-Za-z0-9_-]*[[:space:]]*[,{]/)) { - t = $0; sub(/^[[:space:]]*\./, "", t); sub(/[[:space:]]*[,{].*$/, "", t) - if (t != "") print "css\t" t; next + BEGIN { + SQ = sprintf("%c", 39) + SQ3 = SQ SQ SQ + DQ = "\"" + DQ3 = DQ DQ DQ + # The only characters that can begin a span, a line comment or a + # string. Everything between two of them is copied in one go rather + # than a character at a time. + MARKERS = "[" DQ SQ "/#]" + } + + # Offset just past the one-line string opening at c, or c+1 when it does + # not close before the end of the line — so an apostrophe in prose costs + # one character rather than everything up to the next quote. + function string_end(L, c, q, i, n, ch) { + n = length(L); i = c + 1 + while (i <= n) { + ch = substr(L, i, 1) + if (ch == "\\") { i = i + 2; continue } + if (ch == q) return i + 1 + i++ } - line = $0; sub(/^[[:space:]]+/, "", line) + return c + 1 + } + + # Does the "#" at c open a comment, or is it a CSS colour or id? An + # alphanumeric straight after it is #fff or #app; anything else is a + # comment in every language that has one. + function hash_comment(L, c) { + return substr(L, c + 1, 1) !~ /^[A-Za-z0-9]$/ + } + + # raw[1..n] -> msk[1..n] with comment and string spans emptied. Line + # COUNT is preserved and column positions are not; the matchers lstrip. + # ONLY CLOSED SPANS ARE BLANKED: an opener with no closer is rewound past + # and scanning resumes, so a stray marker costs one span rather than + # every definition below it. + function blank_spans(raw, n, msk, + i, c, L, len, state, closer, oplen, sl, sc, sprefix, + t3, t2, ch, e, k, rest, m) { + for (i = 1; i <= n; i++) msk[i] = "" + i = 1; c = 1; state = 0; closer = "" + while (1) { + while (i <= n) { + L = raw[i]; len = length(L) + if (c > len) { i++; c = 1; continue } + if (state) { + e = index(substr(L, c), closer) + if (e == 0) { i++; c = 1; continue } + c = c + e - 1 + length(closer) + state = 0; closer = "" + continue + } + rest = substr(L, c) + m = match(rest, MARKERS) + if (m == 0) { msk[i] = msk[i] rest; i++; c = 1; continue } + if (m > 1) { + msk[i] = msk[i] substr(rest, 1, m - 1) + c = c + m - 1 + continue + } + t3 = substr(L, c, 3); t2 = substr(L, c, 2); ch = substr(L, c, 1) + if (t3 == DQ3 || t3 == SQ3) { + sl = i; sc = c; sprefix = msk[i] + state = 1; closer = t3; oplen = 3; c = c + 3 + continue + } + if (t2 == "/*") { + sl = i; sc = c; sprefix = msk[i] + state = 1; closer = "*/"; oplen = 2; c = c + 2 + continue + } + if (t2 == "//" || (ch == "#" && hash_comment(L, c))) { + # A line comment is COPIED, not blanked: its continuation lines + # carry their own marker, so none can read as a definition alone. + msk[i] = msk[i] substr(L, c) + i++; c = 1 + continue + } + if (ch == DQ || ch == SQ) { + k = string_end(L, c, ch) + msk[i] = msk[i] substr(L, c, k - c) + c = k + continue + } + msk[i] = msk[i] ch + c++ + } + if (!state) return + for (k = sl; k <= n; k++) msk[k] = "" + msk[sl] = sprefix + i = sl; c = sc + oplen; state = 0; closer = "" + } + } + + function emit(line, t, rest) { + # CSS class definition: .name { or .name, + if (match(line, /^[[:space:]]*\.[A-Za-z][A-Za-z0-9_-]*[[:space:]]*[,{]/)) { + t = line; sub(/^[[:space:]]*\./, "", t); sub(/[[:space:]]*[,{].*$/, "", t) + if (t != "") print "css\t" t; return + } + sub(/^[[:space:]]+/, "", line) # Strip leading declaration modifiers so the definition keyword is the # first word regardless of language (export/pub/private/suspend/...). sub(/^((pub(\([a-z]+\))?|export|default|private|internal|protected|public|static|suspend|async|open|sealed|data|abstract|final|inline|unsafe|extern|override)[[:space:]]+)*/, "", line) @@ -319,7 +422,7 @@ scribe_defs() { if (match(line, /^func[[:space:]]*\([^)]*\)[[:space:]]*[A-Za-z_]/)) { t = line; sub(/^func[[:space:]]*\([^)]*\)[[:space:]]*/, "", t) sub(/[^A-Za-z0-9_].*$/, "", t) - if (t != "") print "sym\t" t; next + if (t != "") print "sym\t" t; return } # Keyword-announced definitions, functions and named types alike. # Dunders are skipped: every class defines __init__, so "already defined @@ -333,17 +436,24 @@ scribe_defs() { # nothing (mirror of coverage.py, #2904). if (line ~ /^type[[:space:]]/) { rest = line; sub(/^type[[:space:]]+[A-Za-z_$][A-Za-z0-9_$]*/, "", rest) - if (rest !~ /[={]/) next + if (rest !~ /[={]/) return } - if (t != "" && t !~ /^__.*__$/) print "sym\t" t; next + if (t != "" && t !~ /^__.*__$/) print "sym\t" t; return } # Arrow/expression assignment: const name = (…) / let name = async ( if (match(line, /^(const|let)[[:space:]]+[A-Za-z_$][A-Za-z0-9_$]*[[:space:]]*=[[:space:]]*(async[[:space:]]*)?[(<]/)) { t = line; sub(/^(const|let)[[:space:]]+/, "", t) sub(/[^A-Za-z0-9_$].*$/, "", t) - if (t != "") print "sym\t" t; next + if (t != "") print "sym\t" t; return } } + + { raw[NR] = $0 } + + END { + blank_spans(raw, NR, msk) + for (r = 1; r <= NR; r++) emit(msk[r]) + } ' 2>/dev/null } diff --git a/src/scribe/services/coverage.py b/src/scribe/services/coverage.py index a16ff7a..d104d45 100644 --- a/src/scribe/services/coverage.py +++ b/src/scribe/services/coverage.py @@ -93,6 +93,133 @@ _ARROW_RE = re.compile( ) +# --- comment and string spans (#4222) ---------------------------------------- +# +# The matchers above are line-oriented and know nothing about what a line is +# INSIDE. A wrapped docstring whose line happens to begin "class AND the …" +# reads as a definition of a shape called `AND`, and that phantom reaches the +# agent mid-edit as a divergence prompt about a symbol that does not exist. +# This repo's own source minted twenty-two of them, measured by running the +# extractor over every scannable file with and without this scan. Two — +# `with` and `nobody`, out of one module docstring in +# scripts/check_dangling_styles.py — are persisted, JUDGED `code_shapes` +# rows, so the cost was never only noise in the moment. Those clear +# themselves: sync_shapes marks a row it no longer extracts as vanished. +# +# The honest tool for .py would be `ast`, which cannot be fooled by prose at +# all. It is not what this uses, because this extractor is mirrored rule for +# rule by an awk program in plugin/hooks/scribe_defs.sh, awk cannot parse +# Python, and a fix only one of the pair can run is the drift the mirror +# exists to prevent. What both can do is blank the SPANS: a triple-quoted +# string or a /* … */ comment is replaced by its own newlines before any +# matcher sees a line, so every line index still lines up and the signature, +# body and fingerprint keep reading the untouched original. +# +# ONLY CLOSED SPANS ARE BLANKED, and an unterminated opener is stepped over +# rather than bailed on, so a stray opener costs one mishandled span and never +# every definition below it in the file. +_SPANS = (('"""', '"""'), ("'''", "'''"), ("/*", "*/")) + + +def _string_end(text: str, at: int) -> int: + """Offset just past the one-line string opening at ``at``. + + ``at`` + 1 when it does not close before the newline, so an apostrophe in + prose — `don't` in a Vue template, outside any comment — costs one + character rather than everything up to the next quote in the file. + """ + quote = text[at] + i = at + 1 + while i < len(text): + c = text[i] + if c == "\\": + i += 2 + elif c == "\n": + return at + 1 + elif c == quote: + return i + 1 + else: + i += 1 + return at + 1 + + +def _hash_comment(text: str, at: int) -> bool: + """Does the `#` at ``at`` open a comment, or is it a CSS colour or id? + + `#` is the one marker whose meaning depends on the language, and the + extractor is handed text with no path. An alphanumeric straight after it + is `#fff` or `#app`; anything else — a space, a `!`, a `-` — is a comment + in every language that has one. + """ + nxt = text[at + 1:at + 2] + return not nxt.isalnum() + + +def _blank_spans(text: str) -> str: + """``text`` with comment and string spans replaced by their own newlines. + + A single left-to-right pass, because the alternative — matching markers + wherever they appear — cannot tell a comment from a string that QUOTES + one. Both of those are in this repo: the docstring-matching regex in + plugin_context.py holds a triple quote inside a single-quoted literal, + and test_design_stylesheet.py asserts on the text `red /*` inside a + double-quoted one. Each cost every definition below it in its file before + the scan was written this way. + + Line COUNT is preserved, column positions are not — the result is only + ever fed to the line matchers, which lstrip anyway. + """ + out: list[str] = [] + i = cut = 0 + n = len(text) + while i < n: + opener = closer = "" + for op, cl in _SPANS: + if text.startswith(op, i): + opener, closer = op, cl + break + if opener: + end = text.find(closer, i + len(opener)) + if end < 0: + # Unterminated: step over the opener rather than bail, so a + # stray marker costs one span and not the rest of the file. + i += len(opener) + continue + end += len(closer) + out.append(text[cut:i]) + out.append("\n" * text.count("\n", i, end)) + i = cut = end + continue + if text.startswith("//", i) or (text[i] == "#" and _hash_comment(text, i)): + # A line comment is COPIED, not blanked: its continuation lines + # carry their own marker, so none of them can read as a + # definition on their own. + nl = text.find("\n", i) + i = n if nl < 0 else nl + continue + if text[i] in "\"'": + i = _string_end(text, i) + continue + i += 1 + out.append(text[cut:]) + return "".join(out) + + +def _masked_lines(text: str, count: int) -> list[str]: + """``text`` blanked and split, reconciled to ``count`` lines. + + Blanking preserves every ``\n``, but ``splitlines`` also breaks on a bare + ``\r`` and on the vertical-tab family, which a blanked span drops. The + reconciliation is what keeps a span containing one of those from shifting + suppression onto the wrong lines — padding is short by a line, never + misaligned by one. + """ + masked = _blank_spans(text).splitlines() + if len(masked) < count: + masked += [""] * (count - len(masked)) + return masked[:count] + + class Definition(NamedTuple): """One extracted definition with its content fingerprint (#2792). @@ -176,8 +303,9 @@ def extract_definitions(text: str) -> list[Definition]: (an overload, a re-declaration) is the same shape to it. """ lines = text.splitlines() + masked = _masked_lines(text, len(lines)) starts: list[tuple[int, str, str]] = [] - for i, raw in enumerate(lines): + for i, raw in enumerate(masked): hit = _definition_on(raw) if hit: starts.append((i, hit[0], hit[1])) diff --git a/tests/test_pattern_coverage.py b/tests/test_pattern_coverage.py index d1cec30..2acaba0 100644 --- a/tests/test_pattern_coverage.py +++ b/tests/test_pattern_coverage.py @@ -10,7 +10,10 @@ tripwire. """ import io import json +import shutil +import subprocess import tarfile +from pathlib import Path import pytest import pytest_asyncio @@ -28,6 +31,8 @@ from scribe.services.coverage import ( from scribe.services.shape_ledger import location_covers from tests.helpers import ensure_user +PLUGIN = Path(__file__).resolve().parents[1] / "plugin" + # --- unit: the definition extractor (shared vectors with the hook) ----------- EXTRACTION_VECTORS = [ @@ -63,6 +68,41 @@ EXTRACTION_VECTORS = [ []), ("dedup-within-file", "def f():\n pass\ndef f():\n pass\n", [("sym", "f")]), + # --- comment and string spans (#4222) ------------------------------------ + # Prose is not code. A wrapped docstring line beginning "class AND the" + # announced a shape called `AND` to a live session; `with` and `nobody` + # out of one module docstring in scripts/check_dangling_styles.py reached + # persisted, judged `code_shapes` rows. + ("docstring-prose", + 'def real_one():\n """Its own text is about the\n' + ' class AND the to_dict, and is a def bar():\n' + ' class Foo: lives here too.\n """\n pass\n', + [("sym", "real_one")]), + # An opener with no closer blanks NOTHING: the scan rewinds past it, so a + # stray marker costs one span rather than the rest of the file. + ("unterminated-docstring", + 'def before():\n """oops, never closed\n\ndef after():\n pass\n', + [("sym", "before"), ("sym", "after")]), + # The CSS half of the same defect (#2990): a wrapped comment line that + # happens to begin with a dotted token reads as a selector. + ("css-comment-selector", + "/* A real base rule, not just descendants: the check reads a\n" + " .ghost, class that only ever appears as an ancestor */\n.check { }\n", + [("css", "check")]), + # A string that HOLDS a comment marker is not a comment — the case that + # made the first draft of the scan eat 70 lines of live code. + ("string-holds-a-marker", + 'SAMPLE = "red /* "\ndef after_the_string():\n pass\n', + [("sym", "after_the_string")]), + # `#` is the one marker whose meaning is the language\'s: a colour here, + # a comment two lines down, and the extractor is handed no path. + ("hash-is-a-colour-not-a-comment", + ".a { color: #fff; } /* .ghost,\n class Phantom: */\n.b { }\n", + [("css", "a"), ("css", "b")]), + ("line-comment-mentioning-a-docstring", + 'def kept():\n pass\n# a stray """ in a comment\n' + 'def also_kept():\n pass\n', + [("sym", "kept"), ("sym", "also_kept")]), ] @@ -75,6 +115,37 @@ def test_extractor_agrees_with_the_hook_on_what_defines(text, expected): assert extract_shapes(text) == expected +@pytest.mark.parametrize( + "text", + [t for _i, t, _e in EXTRACTION_VECTORS], + ids=[i for i, _t, _e in EXTRACTION_VECTORS], +) +def test_the_hook_extractor_runs_and_agrees_line_for_line(text): + """The comment at the top of this module has been the only thing holding + the two extractors together, and a comment cannot fail. This RUNS the + hook's awk program over the same vectors. + + The hook emits every definition in source order with no dedup — identity + there is per payload, not per file — so the comparison de-dupes its + output before matching, which is the one difference between the two that + is by design. + """ + if shutil.which("awk") is None: # pragma: no cover - env guard + pytest.skip("awk not available") + lib = PLUGIN / "hooks" / "scribe_defs.sh" + out = subprocess.run( + ["bash", "-c", f'. "{lib}"; scribe_defs'], + input=text, capture_output=True, text=True, + ) + assert out.returncode == 0, out.stderr + seen: list[tuple[str, str]] = [] + for line in out.stdout.splitlines(): + kind, _, name = line.partition("\t") + if name and (kind, name) not in seen: + seen.append((kind, name)) + assert seen == extract_shapes(text) + + def test_scannable_gates_prose_vendored_and_sourcemaps(): assert scannable("src/app.py") assert scannable("web/button.css")