dev → main: rule overlap check, design-guidance write arm, usage chip seam, divergence meaning gate #180

Merged
bvandeusen merged 7 commits from dev into main 2026-09-22 07:48:54 -04:00
3 changed files with 353 additions and 4 deletions
Showing only changes of commit 91cde6c3e4 - Show all commits
+86 -4
View File
@@ -1521,11 +1521,37 @@ _SEMANTIC_CAP = 150
# "both are about migrations"; first live run paired every alembic
# upgrade()/downgrade() with an unrelated canon at exactly that band.
_SEMANTIC_FLOOR = 0.8
# How many above-floor hits the semantic arm asks for. Named because the
# NUMBER is load-bearing twice over: it caps the work, and a result set that
# came back short of it is a complete picture of what cleared the floor —
# which is what lets a miss be read as evidence rather than as a cut-off
# (`BASIS_NO_SEMANTIC_MATCH`).
_SEMANTIC_LIMIT = 3
# The proposer looked at this body, compared it against every canon in its
# language family, and matched none of them above `_SEMANTIC_FLOOR` (#4208).
#
# This is a NEGATIVE RESULT, and it is stored because it is the only evidence
# in the ledger that speaks to what a shape MEANS rather than what it looks
# like. `proposal_basis` otherwise names how a proposal was arrived at; here
# it records that the arm ran and came back empty, with `proposed_snippet_id`
# left NULL. Every reader keys "is there a proposal" on `proposed_snippet_id`
# or `proposal_group`, never on the basis, so this cannot be mistaken for one:
# `list_shapes(proposal=...)` and `confirm_shape_proposals` both filter on the
# id, and the latter requires it non-NULL before it will confirm anything.
#
# It is deliberately NOT written for the two cases that merely look the same:
# a body too thin to compare (`_substance` below the write-path minimum), and
# a row the per-refresh cap never reached. Those are "I cannot tell", and the
# ledger's standing discipline — the one `FORM_UNKNOWN` enforces everywhere
# else — is that not knowing must make a check quieter, never more confident.
BASIS_NO_SEMANTIC_MATCH = "no-semantic-match"
# Bump when a basis's rule changes: rows remember the (body, ruleset) they
# were examined under, so a tightened rule re-examines everything once.
# v3: language-family gate on the sym bases, reference stoplist, semantic
# restricted to the shape's own project (#2871).
_PROPOSER_VERSION = 3
# v4: the semantic arm records its misses as well as its hits (#4208), so
# every already-examined row must be looked at once more to acquire one.
_PROPOSER_VERSION = 4
# Signature resemblance floor, name blanked (difflib ratio) — and a length
# floor, because `def NAME():` resembles `def NAME(x):` at 0.95 while saying
# nothing; a family shape has parameters to resemble.
@@ -1760,8 +1786,29 @@ def _substance(text: str) -> int:
async def _semantic_canon(
user_id: int, body: str, allowed: set[int]
user_id: int, body: str, allowed: set[int], *, report: dict | None = None
) -> tuple[int, float] | None:
"""The canon this body MEANS, or None.
`report` is an out-param in the style `semantic_search_notes` already
uses, and it carries the one thing the return value cannot: whether a
None is EVIDENCE. `report["conclusive"] = True` says the arm really
compared this body against the allowed canons and none cleared the floor.
It is left unset whenever the arm could not form an opinion — a body with
too little substance to embed, no allowed canon to compare against, or a
result set that came back full and may therefore have been truncated.
The truncation case is why `_SEMANTIC_LIMIT` is named. The search returns
the top N above the floor; if it returns fewer than N, N was not binding
and we have seen everything that cleared the floor, so "no allowed canon
among them" is a fact about the corpus. If it returns exactly N, an
allowed canon could be sitting at N+1 and the same silence means nothing.
Reading the second case as the first is how a cut-off becomes a finding.
Callers must treat a missing key as "cannot tell", never as "no match"
which is also what makes the existing test double, an `AsyncMock` that
returns None and touches no report, stay correct by default.
"""
from scribe.services.embeddings import semantic_search_notes
from scribe.services.plugin_context import (
WRITEPATH_DEFAULT_THRESHOLD, WRITEPATH_MIN_CODE_CHARS, concept_query,
@@ -1771,13 +1818,15 @@ async def _semantic_canon(
return None
query = concept_query(body) or body
hits = await semantic_search_notes(
user_id, query, limit=3,
user_id, query, limit=_SEMANTIC_LIMIT,
threshold=max(WRITEPATH_DEFAULT_THRESHOLD, _SEMANTIC_FLOOR),
note_type="snippet", scope="browse",
)
for score, note in hits:
if int(note.id) in allowed:
return int(note.id), round(float(score), 3)
if report is not None and len(hits) < _SEMANTIC_LIMIT:
report["conclusive"] = True
return None
@@ -1870,16 +1919,29 @@ async def propose_for_repo(
row.proposed_sha = ""
continue
checked += 1
verdict: dict = {}
try:
found = await _semantic_canon(user_id, d[5], semantic_allowed(row.path))
found = await _semantic_canon(
user_id, d[5], semantic_allowed(row.path), report=verdict,
)
except Exception:
logger.warning("semantic proposal failed", exc_info=True)
found = None
# An arm that threw formed no opinion. Clearing this is not
# belt-and-braces: a partially-filled report would record a
# failure as a finding about the code.
verdict = {}
if found:
row.proposed_snippet_id, row.proposal_score = found
row.proposal_basis = "semantic"
row.proposal_group = None
proposed += 1
elif verdict.get("conclusive"):
# No canon, and the arm is sure of it. Kept as the row's basis
# with `proposed_snippet_id` still NULL, so it reads as "asked
# and answered" rather than "not asked" — the distinction
# `flag_divergence` needs and could not previously make.
row.proposal_basis = BASIS_NO_SEMANTIC_MATCH
await session.commit()
return {"examined": examined, "proposed": proposed, "semantic_checked": checked}
@@ -2435,6 +2497,26 @@ async def flag_divergence(project_id: int, *, since: datetime | None) -> int:
continue
if r.proposed_snippet_id == dom[0]:
continue # the proposer already says "instance of the canon"
# ...and the converse, which is the only evidence here that
# is about MEANING rather than shape (#4208). The four false
# prompts #4204 left standing are callables in a directory of
# callables: at the signature level they are indistinguishable
# from #2793's acceptance case, a sync `confirmDanger` beside
# an async confirm canon, and no refinement of `shape_form`
# ever separates them — a registry accessor and a service unit
# differ by the JOB they do, which a signature does not carry.
#
# The proposer does read bodies, and when its semantic arm
# compared this one against every canon in its language family
# and matched none of them, that is a positive finding that
# this shape is not the canon's work. Urging the canon anyway
# would be asserting over a measurement we already hold.
#
# Only the conclusive miss is stored, so an unexamined row and
# a body too thin to embed still ask the question rather than
# being quietly excused.
if r.proposal_basis == BASIS_NO_SEMANTIC_MATCH:
continue
# The same structural test the write-time check applies
# (#4204). The sweep and the hook must agree about what counts
# as divergence, or an audit contradicts the line the writer
+182
View File
@@ -0,0 +1,182 @@
"""A divergence prompt may be silenced by MEANING, never by silence (#4208).
WHAT THIS IS ABOUT. #4204 gave the divergence check a structural gate: a canon
is only urged on a shape whose form could plausibly BE it. That silenced one of
the five false prompts it was filed for. The other four are `def` helpers in a
directory whose canon is an `async def` service unit — callables beside a
callable — and no refinement of `shape_form` ever separates them, because they
differ from #2793's acceptance case (a hand-rolled sync `confirmDanger` where
an async confirm helper is canon) only by the JOB they do. A signature does not
carry a job.
So the lever has to be meaning, and the ledger already holds one reading of it:
the proposer's semantic arm embeds each definition's own BODY against canon.
What it did not do was record its misses. A hit became `proposal_basis =
"semantic"`; a miss left the row indistinguishable from a row nobody had looked
at yet. `flag_divergence` could therefore ask the proposer "do you agree this is
the canon?" but never "did you check, and did you find it is not?".
THE WHOLE RISK IS IN THE NEGATIVE. A miss is only evidence if the arm actually
formed an opinion, and there are three ways for it to come back empty that look
identical from the outside:
body too thin to embed -> no opinion
no allowed canon to test -> no opinion
result set was truncated -> no opinion (the canon may be at N+1)
compared, nothing above the floor -> EVIDENCE
Only the last may silence a prompt. Reading any of the others as a negative is
how "I cannot tell" turns into "I checked" — the exact failure #4204 was opened
on, and the one `FORM_UNKNOWN` already guards against everywhere else in this
module: not knowing must make a check QUIETER, never more confident.
These tests pin the report contract that carries that distinction. The
end-to-end behaviour — a conclusive miss silencing a real prompt while #2793's
acceptance case still raises — is in
tests/test_integration_shape_classify.py, because it needs real rows.
"""
from __future__ import annotations
from unittest.mock import AsyncMock, patch
import pytest
from scribe.services.shape_ledger import (
_SEMANTIC_LIMIT, BASIS_NO_SEMANTIC_MATCH, _semantic_canon,
)
# Comfortably over WRITEPATH_MIN_CODE_CHARS (48 non-whitespace characters), so
# these tests exercise the comparison rather than the substance guard. One of
# #4204's four survivors, quoted rather than invented.
BODY = (
"def is_registered(source: str) -> bool:\n"
" return source in _REGISTRY and _REGISTRY[source].enabled\n"
)
TOO_THIN = "def f():\n pass\n"
CANON = 2860 # the allowed canon, as a caller would pass it
OTHER = 9999 # a snippet that is not in the allowed set
class _FakeNote:
"""Only `.id` is read off a hit."""
def __init__(self, note_id: int) -> None:
self.id = note_id
def _hits(*hits: tuple[float, int]) -> AsyncMock:
return AsyncMock(return_value=[(score, _FakeNote(nid)) for score, nid in hits])
def _patch(mock: AsyncMock):
return patch("scribe.services.embeddings.semantic_search_notes", mock)
# ── the miss that IS evidence ────────────────────────────────────────────
async def test_a_short_result_set_is_a_conclusive_miss() -> None:
"""Fewer hits than asked for means the limit was not binding: everything
above the floor came back, and the canon was not among it. That is a fact
about the corpus, not an artefact of where the list was cut."""
mock = _hits((0.91, OTHER))
report: dict = {}
with _patch(mock):
found = await _semantic_canon(1, BODY, {CANON}, report=report)
assert found is None
assert report.get("conclusive") is True
async def test_an_empty_result_set_is_also_conclusive() -> None:
"""Nothing cleared the floor at all — the strongest form of the miss."""
report: dict = {}
with _patch(_hits()):
assert await _semantic_canon(1, BODY, {CANON}, report=report) is None
assert report.get("conclusive") is True
# ── the three misses that are NOT ────────────────────────────────────────
async def test_a_full_result_set_may_have_been_truncated() -> None:
"""The case that makes `_SEMANTIC_LIMIT` load-bearing rather than a tuning
knob. The search returns the top N above the floor; when it returns
exactly N, an allowed canon can be sitting at N+1 and this same silence
would mean nothing. Reading it as a negative would silence real
divergences in direct proportion to how many snippets the operator has —
a check that quietly weakens as the corpus grows, which is the worst
possible failure mode for a guard nobody is watching."""
mock = _hits(*[(0.9, OTHER + i) for i in range(_SEMANTIC_LIMIT)])
report: dict = {}
with _patch(mock):
assert await _semantic_canon(1, BODY, {CANON}, report=report) is None
assert "conclusive" not in report
async def test_a_body_too_thin_to_embed_forms_no_opinion() -> None:
"""And does not spend an embedding finding that out."""
mock = _hits()
report: dict = {}
with _patch(mock):
assert await _semantic_canon(1, TOO_THIN, {CANON}, report=report) is None
assert "conclusive" not in report
mock.assert_not_awaited()
async def test_no_allowed_canon_means_nothing_was_compared() -> None:
"""An empty allowed set is not "the canons all missed" — there were none
to miss. Distinct because the language-family gate (#2871) empties this
set routinely: a Vue body simply has no Python canon to be compared to."""
mock = _hits()
report: dict = {}
with _patch(mock):
assert await _semantic_canon(1, BODY, set(), report=report) is None
assert "conclusive" not in report
mock.assert_not_awaited()
# ── a hit is a proposal, not a miss ──────────────────────────────────────
async def test_a_hit_returns_the_canon_and_claims_no_miss() -> None:
mock = _hits((0.88, CANON))
report: dict = {}
with _patch(mock):
found = await _semantic_canon(1, BODY, {CANON}, report=report)
assert found == (CANON, 0.88)
assert "conclusive" not in report
async def test_an_allowed_canon_below_the_top_hit_still_wins() -> None:
"""The scan is over the whole result set, so a disallowed snippet ranking
first does not hide an allowed one behind it. Pinned because if it did,
the short-list case above would start reporting conclusive misses for
bodies that DO have a canon."""
mock = _hits((0.95, OTHER), (0.83, CANON))
report: dict = {}
with _patch(mock):
found = await _semantic_canon(1, BODY, {CANON}, report=report)
assert found == (CANON, 0.83)
assert "conclusive" not in report
# ── the contract callers depend on ───────────────────────────────────────
async def test_a_caller_that_passes_no_report_still_gets_an_answer() -> None:
"""The existing test double is an `AsyncMock(return_value=None)` that
never touches a report. Absence of the key must therefore mean "cannot
tell" at every call site — so a stub, an older caller, or an arm that
threw all default to asking the question rather than excusing it."""
with _patch(_hits()):
assert await _semantic_canon(1, BODY, {CANON}) is None
@pytest.mark.parametrize("value", ["semantic", "symbol", "reference", "derive"])
def test_the_miss_basis_is_not_one_of_the_proposal_bases(value: str) -> None:
"""It shares a column with them and must not collide: every reader keys
"is there a proposal" on `proposed_snippet_id`, but `confirm_shape_proposals`
filters BY basis, and a collision there would mean confirming a miss as
though it were a match."""
assert BASIS_NO_SEMANTIC_MATCH != value
+85
View File
@@ -889,6 +889,91 @@ async def test_a_second_confirm_dialog_is_detected_and_named(seeded):
assert total == 0
@pytest.mark.integration
async def test_a_conclusive_meaning_miss_silences_what_the_signature_cannot(seeded):
"""#4208: the four false prompts #4204's form gate provably cannot reach.
THE FIXTURE IS THE ACCEPTANCE CASE ABOVE, DELIBERATELY. That is the whole
difficulty of this issue: a hand-rolled `confirmDanger` beside an async
confirm canon is structurally IDENTICAL to a registry helper beside an
async service canon — same family, same form contradiction, same directory
density. The form gate has to keep asking about both, so nothing derived
from a signature can separate them. The only difference is whether the
shape does the canon's JOB, and the only reading of that the ledger holds
is the proposer's per-symbol body comparison.
So the two runs differ in exactly one thing. In the test above the semantic
arm is quiet — it answers "nothing" without claiming to have looked — and
the prompt is RAISED, which is what milestone #2793 exists to produce. Here
it answers "I compared this body against the canons in its family and it is
none of them", and the prompt is WITHHELD. Holding the fixture identical is
what makes this a test of the meaning gate rather than of the setup.
Asserted on the stored basis as well as the outcome, so that a future
change which silences the prompt for some other reason fails here instead
of reading as a pass.
"""
from datetime import datetime, timedelta, timezone
from unittest.mock import AsyncMock, patch
from scribe.services import shape_ledger
from scribe.services import snippets as snippets_svc
from scribe.services.shape_ledger import (
BASIS_NO_SEMANTIC_MATCH, flag_divergence, live_rows, propose_for_repo,
)
owner, pid = seeded["owner"], seeded["pid"]
canon = await snippets_svc.create_snippet(
owner, name="cls_confirm_factory_meaning",
code="export async function factory(): Promise<boolean> {\n return true;\n}\n",
language="typescript", repo="Widget",
path="frontend/src/composables/useConfirm.ts", symbol="factory",
project_id=pid,
)
sid = int(canon.id)
comp = "frontend/src/components"
base = _defs(
*[(f"{comp}/{n}.vue", "sym", f"on{n}", f"async function on{n}() {{",
f"async function on{n}() {{\n const ok = await factory();\n if (!ok) return;\n}}")
for n in ("Trash", "Delete", "Remove", "Restore")],
)
await sync_repo_shapes(pid, REPO, base, seen_marker="aaa111")
await classify_shapes(owner, pid, [
{"path": f"{comp}/{n}.vue", "symbol": f"on{n}", "status": "instance", "snippet_id": sid}
for n in ("Trash", "Delete", "Remove", "Restore")
], via="audit")
previous = datetime.now(timezone.utc)
later = base + _defs(
(f"{comp}/Danger.vue", "sym", "confirmDanger", "function confirmDanger() {",
"function confirmDanger() {\n return window.confirm('Really?');\n}"),
)
await sync_repo_shapes(pid, REPO, later, seen_marker="bbb222")
def _conclusive_miss(*_args, report=None, **_kw):
"""The arm ran, compared, and found no canon — the one empty answer
that is evidence. `_semantic_canon` itself decides when it may say
this (a result set shorter than the limit); the unit tests for that
judgment are in tests/test_divergence_meaning_gate.py."""
if report is not None:
report["conclusive"] = True
return None
with patch.object(shape_ledger, "_semantic_canon",
AsyncMock(side_effect=_conclusive_miss)):
await propose_for_repo(owner, pid, REPO, later)
rows = await live_rows(pid)
danger = next(r for r in rows if r.symbol == "confirmDanger")
assert danger.proposal_basis == BASIS_NO_SEMANTIC_MATCH
# The miss is not a proposal: nothing may read it as one.
assert danger.proposed_snippet_id is None
assert await flag_divergence(pid, since=previous - timedelta(seconds=1)) == 0
_, total = await list_project_shapes(owner, pid, flag="divergence")
assert total == 0, "a shape the proposer measured as unrelated must not be urged"
@pytest.mark.integration
async def test_history_records_what_was_used_when_and_drift_asks_for_a_recheck(seeded):
from scribe.services.shape_ledger import shape_history