From a875a1b2ee8b1b68febd8d1348812233caf44179 Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Tue, 22 Sep 2026 08:27:15 -0400 Subject: [PATCH 1/3] =?UTF-8?q?fix(shapes):=20a=20semantic=20miss=20is=20n?= =?UTF-8?q?ot=20evidence=20=E2=80=94=20the=20divergence=20check=20stops=20?= =?UTF-8?q?reading=20it=20(#4208)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Measured live: true instances of the service-unit canon score 0.68-0.71 against it at best, helpers 0.66-0.75 against unrelated snippets, and nothing reaches the 0.8 floor. The "conclusive miss" fired for nearly every body and silenced real divergences exactly as it silenced helpers. The stored miss basis, its flag withdrawal and the report plumbing are removed; the floor stays at 0.8 and the new-shapes-first ordering stays. False prompts are answered by judgment (exempt with a reason). Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01821k5B3Ysecp9fNYs92Kuy --- src/scribe/services/shape_ledger.py | 139 ++++----------- tests/test_divergence_meaning_gate.py | 211 ++++++++--------------- tests/test_integration_shape_classify.py | 117 +++---------- 3 files changed, 128 insertions(+), 339 deletions(-) diff --git a/src/scribe/services/shape_ledger.py b/src/scribe/services/shape_ledger.py index fb2c05d..1c96620 100644 --- a/src/scribe/services/shape_ledger.py +++ b/src/scribe/services/shape_ledger.py @@ -1521,43 +1521,33 @@ _SEMANTIC_CAP = 150 # "both are about migrations"; first live run paired every alembic # upgrade()/downgrade() with an unrelated canon at exactly that band. _SEMANTIC_FLOOR = 0.8 -# How many above-floor hits the semantic arm asks for. Named because the -# NUMBER is load-bearing twice over: it caps the work, and a result set that -# came back short of it is a complete picture of what cleared the floor — -# which is what lets a miss be read as evidence rather than as a cut-off -# (`BASIS_NO_SEMANTIC_MATCH`). +# How many above-floor hits the semantic arm asks for. _SEMANTIC_LIMIT = 3 -# The proposer looked at this body, compared it against every canon in its -# language family, and matched none of them above `_SEMANTIC_FLOOR` (#4208). +# A MISS FROM THIS ARM IS NOT EVIDENCE, and nothing may read it as one (#4208). # -# This is a NEGATIVE RESULT, and it is stored because it is the only evidence -# in the ledger that speaks to what a shape MEANS rather than what it looks -# like. `proposal_basis` otherwise names how a proposal was arrived at; here -# it records that the arm ran and came back empty, with `proposed_snippet_id` -# left NULL. Every reader keys "is there a proposal" on `proposed_snippet_id` -# or `proposal_group`, never on the basis, so this cannot be mistaken for one: -# `list_shapes(proposal=...)` and `confirm_shape_proposals` both filter on the -# id, and the latter requires it non-NULL before it will confirm anything. -# -# It is deliberately NOT written for the two cases that merely look the same: -# a body too thin to compare (`_substance` below the write-path minimum), and -# a row the per-refresh cap never reached. Those are "I cannot tell", and the -# ledger's standing discipline — the one `FORM_UNKNOWN` enforces everywhere -# else — is that not knowing must make a check quieter, never more confident. -BASIS_NO_SEMANTIC_MATCH = "no-semantic-match" +# #4208 briefly stored "compared, nothing cleared the floor" and let it +# withdraw divergence prompts. Measured live on 2026-09-22, with the arm's own +# query shape: judged instances of the service-unit canon (#2860) score 0.68– +# 0.71 against it at best, and most fall below 0.66; helpers beside it score +# 0.66–0.75 against snippets they have nothing to do with. Nothing reaches +# 0.8, so the arm answers "no canon" for almost every body, true instance or +# not — a real divergence was silenced exactly as a helper was. A code body +# against a prose-forward snippet document measures the wrong field (#2518), +# and no floor separates the two bands. The arm PROPOSES on a hit; its silence +# says nothing. def _semantic_priority(row) -> tuple: - """Order for the capped semantic pass: the rows its verdict can still act on - come first (#4208, measured on the first live refresh after it shipped). + """Order for the capped semantic pass: new shapes first (#4208, measured + on the first live refresh after the proposer version was bumped). The pass is capped (`_SEMANTIC_CAP`), and `flag_divergence` acts only on - shapes NEW since the previous refresh. In plain row order those are the - highest ids — the back of the queue — so whenever the backlog exceeds the - cap (every `_PROPOSER_VERSION` bump queues the whole todo at once), the - cap is spent on old rows and the new ones are flagged as "cannot tell" - before the arm ever reads them. That is what happened: 1,533 rows queued, - 150 checked, and all six new shapes flagged unexamined. + shapes NEW since the previous refresh — and skips one the proposer matched + to the dominant canon. In plain row order the new shapes are the highest + ids, the back of the queue, so whenever the backlog exceeds the cap (every + `_PROPOSER_VERSION` bump queues the whole todo at once) they were flagged + before the arm could propose for them: 1,533 rows queued, 150 checked, and + every new shape left unread. So: the human todo (`unclassified`) before `scoped`, newest first within each. A row with no creation time sorts last in its group. @@ -1568,8 +1558,9 @@ def _semantic_priority(row) -> tuple: # were examined under, so a tightened rule re-examines everything once. # v3: language-family gate on the sym bases, reference stoplist, semantic # restricted to the shape's own project (#2871). -# v4: the semantic arm records its misses as well as its hits (#4208), so -# every already-examined row must be looked at once more to acquire one. +# v4: the semantic arm recorded its misses too (#4208). Retired — a miss is +# not evidence (see _SEMANTIC_LIMIT) — but the proposals themselves did not +# change, so no re-examination is owed; a stale miss basis is inert. _PROPOSER_VERSION = 4 # Signature resemblance floor, name blanked (difflib ratio) — and a length # floor, because `def NAME():` resembles `def NAME(x):` at 0.95 while saying @@ -1805,29 +1796,10 @@ def _substance(text: str) -> int: async def _semantic_canon( - user_id: int, body: str, allowed: set[int], *, report: dict | None = None + user_id: int, body: str, allowed: set[int], ) -> tuple[int, float] | None: - """The canon this body MEANS, or None. - - `report` is an out-param in the style `semantic_search_notes` already - uses, and it carries the one thing the return value cannot: whether a - None is EVIDENCE. `report["conclusive"] = True` says the arm really - compared this body against the allowed canons and none cleared the floor. - It is left unset whenever the arm could not form an opinion — a body with - too little substance to embed, no allowed canon to compare against, or a - result set that came back full and may therefore have been truncated. - - The truncation case is why `_SEMANTIC_LIMIT` is named. The search returns - the top N above the floor; if it returns fewer than N, N was not binding - and we have seen everything that cleared the floor, so "no allowed canon - among them" is a fact about the corpus. If it returns exactly N, an - allowed canon could be sitting at N+1 and the same silence means nothing. - Reading the second case as the first is how a cut-off becomes a finding. - - Callers must treat a missing key as "cannot tell", never as "no match" — - which is also what makes the existing test double, an `AsyncMock` that - returns None and touches no report, stay correct by default. - """ + """The canon this body MEANS, or None. None is "no proposal", never "not + the canon" — see the note above `_SEMANTIC_LIMIT`.""" from scribe.services.embeddings import semantic_search_notes from scribe.services.plugin_context import ( WRITEPATH_DEFAULT_THRESHOLD, WRITEPATH_MIN_CODE_CHARS, concept_query, @@ -1844,8 +1816,6 @@ async def _semantic_canon( for score, note in hits: if int(note.id) in allowed: return int(note.id), round(float(score), 3) - if report is not None and len(hits) < _SEMANTIC_LIMIT: - report["conclusive"] = True return None @@ -1939,29 +1909,16 @@ async def propose_for_repo( row.proposed_sha = "" continue checked += 1 - verdict: dict = {} try: - found = await _semantic_canon( - user_id, d[5], semantic_allowed(row.path), report=verdict, - ) + found = await _semantic_canon(user_id, d[5], semantic_allowed(row.path)) except Exception: logger.warning("semantic proposal failed", exc_info=True) found = None - # An arm that threw formed no opinion. Clearing this is not - # belt-and-braces: a partially-filled report would record a - # failure as a finding about the code. - verdict = {} if found: row.proposed_snippet_id, row.proposal_score = found row.proposal_basis = "semantic" row.proposal_group = None proposed += 1 - elif verdict.get("conclusive"): - # No canon, and the arm is sure of it. Kept as the row's basis - # with `proposed_snippet_id` still NULL, so it reads as "asked - # and answered" rather than "not asked" — the distinction - # `flag_divergence` needs and could not previously make. - row.proposal_basis = BASIS_NO_SEMANTIC_MATCH await session.commit() return {"examined": examined, "proposed": proposed, "semantic_checked": checked} @@ -2486,9 +2443,7 @@ async def flag_divergence(project_id: int, *, since: datetime | None) -> int: """Flag shapes created after ``since`` (the previous refresh) that sit where a canon dominates and were not proposed as that canon. With no previous refresh (first seed) nothing is new, nothing is flagged. - Standing flags persist until judged — or until the proposer's semantic - arm reports a conclusive miss for the row (#4208). Returns how many are - flagged.""" + Standing flags persist until judged. Returns how many are flagged.""" if since is None: return 0 async with async_session() as session: @@ -2512,17 +2467,6 @@ async def flag_divergence(project_id: int, *, since: datetime | None) -> int: for r in siblings: if r.status not in _MECHANICAL_TODO: continue - # A conclusive miss outranks a standing flag, and is checked - # before it. A flag raised on a row the semantic arm had not - # yet reached was raised on "cannot tell" — the arm is capped, - # so that is routine — and once the arm has read the body and - # found it is none of the canons, keeping the prompt would be - # asserting over a measurement we now hold. Only this evidence - # withdraws a flag; nothing else changes "persist until judged". - if r.proposal_basis == BASIS_NO_SEMANTIC_MATCH: - if r.diverges_from is not None: - r.diverges_from = None - continue if r.diverges_from is not None: flagged += 1 continue @@ -2530,25 +2474,14 @@ async def flag_divergence(project_id: int, *, since: datetime | None) -> int: continue if r.proposed_snippet_id == dom[0]: continue # the proposer already says "instance of the canon" - # ...and the converse, which is the only evidence here that - # is about MEANING rather than shape (#4208). The four false - # prompts #4204 left standing are callables in a directory of - # callables: at the signature level they are indistinguishable - # from #2793's acceptance case, a sync `confirmDanger` beside - # an async confirm canon, and no refinement of `shape_form` - # ever separates them — a registry accessor and a service unit - # differ by the JOB they do, which a signature does not carry. - # - # The proposer does read bodies, and when its semantic arm - # compared this one against every canon in its language family - # and matched none of them, that is a positive finding that - # this shape is not the canon's work. Urging the canon anyway - # would be asserting over a measurement we already hold. - # - # Only the conclusive miss is stored, so an unexamined row and - # a body too thin to embed still ask the question rather than - # being quietly excused. (Tested at the top of this loop, where - # it also withdraws a flag raised before the arm got there.) + # NOT the converse (#4208). A semantic MISS does not mean "not + # the canon": measured, true instances of a canon rarely clear + # the arm's floor either (see the note above _SEMANTIC_LIMIT). + # A helper beside a canon and a genuine divergence — #2793's + # sync `confirmDanger` beside an async confirm canon — are the + # same case to every signal this ledger holds, so the prompt + # asks, and the judge answers it (`exempt` with a reason + # retires a false one for good). # The same structural test the write-time check applies # (#4204). The sweep and the hook must agree about what counts # as divergence, or an audit contradicts the line the writer diff --git a/tests/test_divergence_meaning_gate.py b/tests/test_divergence_meaning_gate.py index 4f172a5..49967a5 100644 --- a/tests/test_divergence_meaning_gate.py +++ b/tests/test_divergence_meaning_gate.py @@ -1,49 +1,33 @@ -"""A divergence prompt may be silenced by MEANING, never by silence (#4208). +"""A semantic MISS is not evidence, and the divergence check never reads one (#4208). -WHAT THIS IS ABOUT. #4204 gave the divergence check a structural gate: a canon -is only urged on a shape whose form could plausibly BE it. That silenced one of -the five false prompts it was filed for. The other four are `def` helpers in a -directory whose canon is an `async def` service unit — callables beside a -callable — and no refinement of `shape_form` ever separates them, because they -differ from #2793's acceptance case (a hand-rolled sync `confirmDanger` where -an async confirm helper is canon) only by the JOB they do. A signature does not -carry a job. +WHAT THIS IS ABOUT. #4204 gave the divergence check a structural gate, which +silenced one of the five false prompts it was filed for. The other four are +`def` helpers beside an `async def` service canon — callables beside a +callable — and differ from #2793's acceptance case (a hand-rolled sync +`confirmDanger` where an async confirm helper is canon) only by the JOB they +do. A signature does not carry a job. -So the lever has to be meaning, and the ledger already holds one reading of it: -the proposer's semantic arm embeds each definition's own BODY against canon. -What it did not do was record its misses. A hit became `proposal_basis = -"semantic"`; a miss left the row indistinguishable from a row nobody had looked -at yet. `flag_divergence` could therefore ask the proposer "do you agree this is -the canon?" but never "did you check, and did you find it is not?". +#4208 tried meaning: store the proposer's semantic "compared, nothing cleared +the floor" and let it withdraw the prompt. Measured live on 2026-09-22 it +cannot discriminate. Judged instances of the service-unit canon score 0.68– +0.71 against it at best, most below 0.66; helpers score 0.66–0.75 against +snippets unrelated to them. Nothing reaches the 0.8 floor, so the "miss" +fires for nearly every body — the real divergence silenced exactly as the +helper was. The gate was removed; the prompt asks and the judge answers. -THE WHOLE RISK IS IN THE NEGATIVE. A miss is only evidence if the arm actually -formed an opinion, and there are three ways for it to come back empty that look -identical from the outside: - - body too thin to embed -> no opinion - no allowed canon to test -> no opinion - result set was truncated -> no opinion (the canon may be at N+1) - compared, nothing above the floor -> EVIDENCE - -Only the last may silence a prompt. Reading any of the others as a negative is -how "I cannot tell" turns into "I checked" — the exact failure #4204 was opened -on, and the one `FORM_UNKNOWN` already guards against everywhere else in this -module: not knowing must make a check QUIETER, never more confident. - -These tests pin the report contract that carries that distinction. The -end-to-end behaviour — a conclusive miss silencing a real prompt while #2793's -acceptance case still raises — is in -tests/test_integration_shape_classify.py, because it needs real rows. +These tests pin what stayed: the arm proposes on a hit and says nothing on a +miss, the divergence check does not consult it, and the capped pass reads new +shapes first so the proposals it CAN make land where the check looks. """ from __future__ import annotations +import ast +import inspect +import textwrap from unittest.mock import AsyncMock, patch -import pytest - -from scribe.services.shape_ledger import ( - _SEMANTIC_LIMIT, BASIS_NO_SEMANTIC_MATCH, _semantic_canon, -) +from scribe.services import shape_ledger +from scribe.services.shape_ledger import _semantic_canon # Comfortably over WRITEPATH_MIN_CODE_CHARS (48 non-whitespace characters), so # these tests exercise the comparison rather than the substance guard. One of @@ -73,116 +57,64 @@ def _patch(mock: AsyncMock): return patch("scribe.services.embeddings.semantic_search_notes", mock) -# ── the miss that IS evidence ──────────────────────────────────────────── +# ── the arm proposes on a hit ──────────────────────────────────────────── -async def test_a_short_result_set_is_a_conclusive_miss() -> None: - """Fewer hits than asked for means the limit was not binding: everything - above the floor came back, and the canon was not among it. That is a fact - about the corpus, not an artefact of where the list was cut.""" - mock = _hits((0.91, OTHER)) - report: dict = {} - with _patch(mock): - found = await _semantic_canon(1, BODY, {CANON}, report=report) - assert found is None - assert report.get("conclusive") is True - - -async def test_an_empty_result_set_is_also_conclusive() -> None: - """Nothing cleared the floor at all — the strongest form of the miss.""" - report: dict = {} - with _patch(_hits()): - assert await _semantic_canon(1, BODY, {CANON}, report=report) is None - assert report.get("conclusive") is True - - -# ── the three misses that are NOT ──────────────────────────────────────── - - -async def test_a_full_result_set_may_have_been_truncated() -> None: - """The case that makes `_SEMANTIC_LIMIT` load-bearing rather than a tuning - knob. The search returns the top N above the floor; when it returns - exactly N, an allowed canon can be sitting at N+1 and this same silence - would mean nothing. Reading it as a negative would silence real - divergences in direct proportion to how many snippets the operator has — - a check that quietly weakens as the corpus grows, which is the worst - possible failure mode for a guard nobody is watching.""" - mock = _hits(*[(0.9, OTHER + i) for i in range(_SEMANTIC_LIMIT)]) - report: dict = {} - with _patch(mock): - assert await _semantic_canon(1, BODY, {CANON}, report=report) is None - assert "conclusive" not in report - - -async def test_a_body_too_thin_to_embed_forms_no_opinion() -> None: - """And does not spend an embedding finding that out.""" - mock = _hits() - report: dict = {} - with _patch(mock): - assert await _semantic_canon(1, TOO_THIN, {CANON}, report=report) is None - assert "conclusive" not in report - mock.assert_not_awaited() - - -async def test_no_allowed_canon_means_nothing_was_compared() -> None: - """An empty allowed set is not "the canons all missed" — there were none - to miss. Distinct because the language-family gate (#2871) empties this - set routinely: a Vue body simply has no Python canon to be compared to.""" - mock = _hits() - report: dict = {} - with _patch(mock): - assert await _semantic_canon(1, BODY, set(), report=report) is None - assert "conclusive" not in report - mock.assert_not_awaited() - - -# ── a hit is a proposal, not a miss ────────────────────────────────────── - - -async def test_a_hit_returns_the_canon_and_claims_no_miss() -> None: - mock = _hits((0.88, CANON)) - report: dict = {} - with _patch(mock): - found = await _semantic_canon(1, BODY, {CANON}, report=report) - assert found == (CANON, 0.88) - assert "conclusive" not in report +async def test_a_hit_returns_the_canon() -> None: + with _patch(_hits((0.88, CANON))): + assert await _semantic_canon(1, BODY, {CANON}) == (CANON, 0.88) async def test_an_allowed_canon_below_the_top_hit_still_wins() -> None: """The scan is over the whole result set, so a disallowed snippet ranking - first does not hide an allowed one behind it. Pinned because if it did, - the short-list case above would start reporting conclusive misses for - bodies that DO have a canon.""" - mock = _hits((0.95, OTHER), (0.83, CANON)) - report: dict = {} - with _patch(mock): - found = await _semantic_canon(1, BODY, {CANON}, report=report) - assert found == (CANON, 0.83) - assert "conclusive" not in report + first does not hide an allowed one behind it.""" + with _patch(_hits((0.95, OTHER), (0.83, CANON))): + assert await _semantic_canon(1, BODY, {CANON}) == (CANON, 0.83) -# ── the contract callers depend on ─────────────────────────────────────── - - -async def test_a_caller_that_passes_no_report_still_gets_an_answer() -> None: - """The existing test double is an `AsyncMock(return_value=None)` that - never touches a report. Absence of the key must therefore mean "cannot - tell" at every call site — so a stub, an older caller, or an arm that - threw all default to asking the question rather than excusing it.""" - with _patch(_hits()): +async def test_a_miss_is_none_and_nothing_more() -> None: + with _patch(_hits((0.91, OTHER))): assert await _semantic_canon(1, BODY, {CANON}) is None -@pytest.mark.parametrize("value", ["semantic", "symbol", "reference", "derive"]) -def test_the_miss_basis_is_not_one_of_the_proposal_bases(value: str) -> None: - """It shares a column with them and must not collide: every reader keys - "is there a proposal" on `proposed_snippet_id`, but `confirm_shape_proposals` - filters BY basis, and a collision there would mean confirming a miss as - though it were a match.""" - assert BASIS_NO_SEMANTIC_MATCH != value +async def test_a_body_too_thin_to_embed_spends_no_embedding() -> None: + mock = _hits() + with _patch(mock): + assert await _semantic_canon(1, TOO_THIN, {CANON}) is None + mock.assert_not_awaited() -# ── the cap spends itself on the rows its verdict can act on ───────────── +async def test_no_allowed_canon_spends_no_search() -> None: + """The language-family gate (#2871) empties the allowed set routinely.""" + mock = _hits() + with _patch(mock): + assert await _semantic_canon(1, BODY, set()) is None + mock.assert_not_awaited() + + +# ── ...and its silence reaches nothing ─────────────────────────────────── + + +def test_the_divergence_check_does_not_read_the_proposal_basis() -> None: + """The retired gate keyed on `proposal_basis`. Asserted on the function's + structure (rule 167): any attribute read of it inside `flag_divergence` is + the miss being consulted again, whatever the constant is called.""" + tree = ast.parse(textwrap.dedent(inspect.getsource(shape_ledger.flag_divergence))) + reads = [n for n in ast.walk(tree) + if isinstance(n, ast.Attribute) and n.attr == "proposal_basis"] + assert not reads, ( + "flag_divergence reads proposal_basis — a semantic miss is 'cannot " + "tell' at the arm's floor (#4208), never 'not the canon'" + ) + + +def test_no_miss_basis_is_stored() -> None: + """A miss basis sharing the column with real bases is a thing + `confirm_shape_proposals(basis=…)` could filter on and confirm.""" + assert not hasattr(shape_ledger, "BASIS_NO_SEMANTIC_MATCH") + + +# ── the cap spends itself where proposals matter ───────────────────────── class _AgedRow: @@ -191,11 +123,10 @@ class _AgedRow: def test_the_capped_pass_reads_new_shapes_first() -> None: - """Measured on the first live refresh after the gate shipped: a version - bump queued 1,533 rows, the cap read 150 of them in row order, and every - shape NEW since the previous refresh — the only rows `flag_divergence` - acts on, and the highest ids — was flagged before the arm reached it. The - gate silenced nothing because it never got to look.""" + """Measured on the first live refresh after a version bump: 1,533 rows + queued, the cap read 150 of them in row order, and every shape NEW since + the previous refresh — the only rows `flag_divergence` acts on, and the + highest ids — was flagged before the arm could propose for it.""" from datetime import datetime, timedelta, timezone from scribe.services.shape_ledger import _semantic_priority diff --git a/tests/test_integration_shape_classify.py b/tests/test_integration_shape_classify.py index 74e2578..2b8e329 100644 --- a/tests/test_integration_shape_classify.py +++ b/tests/test_integration_shape_classify.py @@ -890,37 +890,26 @@ async def test_a_second_confirm_dialog_is_detected_and_named(seeded): @pytest.mark.integration -async def test_a_conclusive_meaning_miss_silences_what_the_signature_cannot(seeded): - """#4208: the four false prompts #4204's form gate provably cannot reach. +async def test_a_semantic_miss_does_not_silence_the_prompt(seeded): + """#4208, reversed on measurement: a semantic miss is not evidence. - THE FIXTURE IS THE ACCEPTANCE CASE ABOVE, DELIBERATELY. That is the whole - difficulty of this issue: a hand-rolled `confirmDanger` beside an async - confirm canon is structurally IDENTICAL to a registry helper beside an - async service canon — same family, same form contradiction, same directory - density. The form gate has to keep asking about both, so nothing derived - from a signature can separate them. The only difference is whether the - shape does the canon's JOB, and the only reading of that the ledger holds - is the proposer's per-symbol body comparison. + THE FIXTURE IS THE ACCEPTANCE CASE ABOVE, DELIBERATELY. A hand-rolled + `confirmDanger` beside an async confirm canon is structurally identical to + a registry helper beside an async service canon, and the arm's miss was + meant to tell them apart. Measured live, true instances of a canon rarely + clear the arm's 0.8 floor either, so a miss fires on both — silencing on + it silenced #2793's own case. Here the arm has read the body, found + nothing, and the prompt is still RAISED. - So the two runs differ in exactly one thing. In the test above the semantic - arm is quiet — it answers "nothing" without claiming to have looked — and - the prompt is RAISED, which is what milestone #2793 exists to produce. Here - it answers "I compared this body against the canons in its family and it is - none of them", and the prompt is WITHHELD. Holding the fixture identical is - what makes this a test of the meaning gate rather than of the setup. - - Asserted on the stored basis as well as the outcome, so that a future - change which silences the prompt for some other reason fails here instead - of reading as a pass. + Asserted on the row, not a flag count: `since` sits a second early to + absorb clock skew, which also admits other fixture rows as "new". """ from datetime import datetime, timedelta, timezone from unittest.mock import AsyncMock, patch from scribe.services import shape_ledger from scribe.services import snippets as snippets_svc - from scribe.services.shape_ledger import ( - BASIS_NO_SEMANTIC_MATCH, flag_divergence, live_rows, propose_for_repo, - ) + from scribe.services.shape_ledger import flag_divergence, live_rows, propose_for_repo owner, pid = seeded["owner"], seeded["pid"] canon = await snippets_svc.create_snippet( @@ -950,28 +939,15 @@ async def test_a_conclusive_meaning_miss_silences_what_the_signature_cannot(seed ) await sync_repo_shapes(pid, REPO, later, seen_marker="bbb222") - def _conclusive_miss(*_args, report=None, **_kw): - """The arm ran, compared, and found no canon — the one empty answer - that is evidence. `_semantic_canon` itself decides when it may say - this (a result set shorter than the limit); the unit tests for that - judgment are in tests/test_divergence_meaning_gate.py.""" - if report is not None: - report["conclusive"] = True - return None - - with patch.object(shape_ledger, "_semantic_canon", - AsyncMock(side_effect=_conclusive_miss)): + miss = AsyncMock(return_value=None) + with patch.object(shape_ledger, "_semantic_canon", miss): await propose_for_repo(owner, pid, REPO, later) + assert miss.await_count >= 1, "the arm must actually have read a body" - rows = await live_rows(pid) - danger = next(r for r in rows if r.symbol == "confirmDanger") - assert danger.proposal_basis == BASIS_NO_SEMANTIC_MATCH - # The miss is not a proposal: nothing may read it as one. + await flag_divergence(pid, since=previous - timedelta(seconds=1)) + danger = next(r for r in await live_rows(pid) if r.symbol == "confirmDanger") assert danger.proposed_snippet_id is None - - assert await flag_divergence(pid, since=previous - timedelta(seconds=1)) == 0 - _, total = await list_project_shapes(owner, pid, flag="divergence") - assert total == 0, "a shape the proposer measured as unrelated must not be urged" + assert danger.diverges_from == sid, "a miss at the arm's floor is 'cannot tell'" async def _two_generations(seeded, tag: str): @@ -1016,12 +992,6 @@ async def _two_generations(seeded, tag: str): return owner, pid, later, previous -def _conclusive(*_a, report=None, **_k): - if report is not None: - report["conclusive"] = True - return None - - @pytest.mark.integration async def test_the_capped_semantic_pass_reaches_the_new_shape_first(seeded): """Measured live: after the #4208 version bump, 1,533 rows competed for 150 @@ -1031,64 +1001,19 @@ async def test_the_capped_semantic_pass_reaches_the_new_shape_first(seeded): from unittest.mock import AsyncMock, patch from scribe.services import shape_ledger - from scribe.services.shape_ledger import ( - BASIS_NO_SEMANTIC_MATCH, live_rows, propose_for_repo, - ) + from scribe.services.shape_ledger import live_rows, propose_for_repo owner, pid, later, _previous = await _two_generations(seeded, "cap") - with patch.object(shape_ledger, "_semantic_canon", - AsyncMock(side_effect=_conclusive)): + with patch.object(shape_ledger, "_semantic_canon", AsyncMock(return_value=None)): stats = await propose_for_repo(owner, pid, REPO, later, semantic_cap=1) assert stats["semantic_checked"] == 1 rows = {r.symbol: r for r in await live_rows(pid)} - assert rows["confirmDanger"].proposal_basis == BASIS_NO_SEMANTIC_MATCH + assert rows["confirmDanger"].proposed_sha != "", "the new shape got the one check" # The old row is left for the next refresh, unexamined — never stamped. - assert rows["oldHelper"].proposal_basis is None assert rows["oldHelper"].proposed_sha == "" -@pytest.mark.integration -async def test_a_later_conclusive_miss_withdraws_a_flag_raised_before_it(seeded): - """A flag raised while the arm had not yet read the row was raised on - "cannot tell". When a later refresh reads the body and finds it is none of - the canons, the flag goes — and only that evidence withdraws one.""" - from datetime import timedelta - from unittest.mock import AsyncMock, patch - - from scribe.services import shape_ledger - from scribe.services.shape_ledger import ( - BASIS_NO_SEMANTIC_MATCH, flag_divergence, live_rows, propose_for_repo, - ) - - owner, pid, later, previous = await _two_generations(seeded, "withdraw") - since = previous - timedelta(seconds=1) - - async def _danger(): - return next(r for r in await live_rows(pid) if r.symbol == "confirmDanger") - - # Asserted on the ROW, not on a flag count. `since` sits a second before - # the second sync to absorb clock skew, which also admits the fixture's - # old helper as "new" — how many rows get flagged is a property of that - # margin, and this test is about one row's flag coming off. - - # First refresh: the arm is quiet (formed no opinion), so the prompt is raised. - with patch.object(shape_ledger, "_semantic_canon", AsyncMock(return_value=None)): - await propose_for_repo(owner, pid, REPO, later) - await flag_divergence(pid, since=since) - assert (await _danger()).diverges_from is not None - - # A later refresh re-reads the body (a ruleset bump forces it) and is sure. - with patch.object(shape_ledger, "_PROPOSER_VERSION", 10_000), \ - patch.object(shape_ledger, "_semantic_canon", - AsyncMock(side_effect=_conclusive)): - await propose_for_repo(owner, pid, REPO, later) - await flag_divergence(pid, since=since) - danger = await _danger() - assert danger.proposal_basis == BASIS_NO_SEMANTIC_MATCH - assert danger.diverges_from is None - - @pytest.mark.integration async def test_history_records_what_was_used_when_and_drift_asks_for_a_recheck(seeded): from scribe.services.shape_ledger import shape_history -- 2.54.0 From 312dc9f6f77812cff7fe90c4734210cfafd7b6b4 Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Tue, 22 Sep 2026 08:33:01 -0400 Subject: [PATCH 2/3] fix(shapes): the semantic arm proposes at the write-path floor, not a private 0.8 (#4208) 0.8 was sized for proposals nobody reads; every semantic proposal is read by a judge before it is confirmed. Measured, it proposed 0 of 150 while judged instances of a canon score 0.68-0.71 - the write-path hint's own floor asks the same question of the same documents at 0.68. The arm now uses that floor, scans 8 hits instead of 3 (a true instance ranked 4th behind snippets of other language families), and the proposer version bumps to 5 so rows examined under the old floor are read again. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01821k5B3Ysecp9fNYs92Kuy --- src/scribe/services/shape_ledger.py | 46 ++++++++++++++---------- tests/test_divergence_meaning_gate.py | 20 +++++++++-- tests/test_integration_shape_classify.py | 2 +- 3 files changed, 46 insertions(+), 22 deletions(-) diff --git a/src/scribe/services/shape_ledger.py b/src/scribe/services/shape_ledger.py index 1c96620..fc657d8 100644 --- a/src/scribe/services/shape_ledger.py +++ b/src/scribe/services/shape_ledger.py @@ -1516,24 +1516,32 @@ _DERIVE_MIN_NAME_CSS = 2 # Semantic checks per repo per refresh — an embedding each (local fastembed), # bounded so a 4,000-row ledger is worked through over refreshes, not in one. _SEMANTIC_CAP = 150 -# The semantic basis's own floor — stricter than the write-path arm's, because -# a code BODY against a prose-forward snippet document scores 0.68–0.75 for -# "both are about migrations"; first live run paired every alembic -# upgrade()/downgrade() with an unrelated canon at exactly that band. -_SEMANTIC_FLOOR = 0.8 -# How many above-floor hits the semantic arm asks for. -_SEMANTIC_LIMIT = 3 -# A MISS FROM THIS ARM IS NOT EVIDENCE, and nothing may read it as one (#4208). +# The semantic arm has NO floor of its own: it uses the write-path surface's, +# because it asks the write-path hint's question — "which canon does this body +# resemble?" — of the same snippet documents (#4208). # +# It had one, 0.8, set after the first live run paired alembic +# upgrade()/downgrade() with unrelated canons at 0.68–0.75. That floor was +# sized for a proposal nobody reads, and nobody-reads is not how proposals are +# consumed: a judge reviews each semantic one (confirm_proposals refuses an +# unnamed batch, and the tool says "semantic deserves a look"). Measured on +# 2026-09-22 it proposed 0 of 150, while judged instances of the service-unit +# canon score 0.68–0.71 against it. A wrong proposal costs the judge a look; a +# floor that yields nothing costs the signal. Noise in this band is expected — +# the score rides on the proposal so the judge can weigh it. +# +# How many hits the arm scans for an ALLOWED canon. Wider than the hint's +# budget because disallowed snippets (other language families) share the +# ranking: one measured true instance sat 4th, behind three it could not use. +_SEMANTIC_LIMIT = 8 +# A MISS FROM THIS ARM IS NOT EVIDENCE, and nothing may read it as one (#4208). # #4208 briefly stored "compared, nothing cleared the floor" and let it -# withdraw divergence prompts. Measured live on 2026-09-22, with the arm's own -# query shape: judged instances of the service-unit canon (#2860) score 0.68– -# 0.71 against it at best, and most fall below 0.66; helpers beside it score -# 0.66–0.75 against snippets they have nothing to do with. Nothing reaches -# 0.8, so the arm answers "no canon" for almost every body, true instance or -# not — a real divergence was silenced exactly as a helper was. A code body -# against a prose-forward snippet document measures the wrong field (#2518), -# and no floor separates the two bands. The arm PROPOSES on a hit; its silence +# withdraw divergence prompts. It silenced real divergences exactly as it +# silenced helpers: most judged instances of a canon score below any floor +# that keeps noise out (0.66 and under, in the measurement above), and helpers +# score 0.66–0.75 against snippets they have nothing to do with. A code body +# against a prose-forward snippet document measures the wrong field (#2518); +# no floor separates the two bands. The arm PROPOSES on a hit; its silence # says nothing. @@ -1561,7 +1569,9 @@ def _semantic_priority(row) -> tuple: # v4: the semantic arm recorded its misses too (#4208). Retired — a miss is # not evidence (see _SEMANTIC_LIMIT) — but the proposals themselves did not # change, so no re-examination is owed; a stale miss basis is inert. -_PROPOSER_VERSION = 4 +# v5: the semantic arm's floor dropped from 0.8 to the write-path floor (#4208), +# so rows it examined and found nothing for must be read once more. +_PROPOSER_VERSION = 5 # Signature resemblance floor, name blanked (difflib ratio) — and a length # floor, because `def NAME():` resembles `def NAME(x):` at 0.95 while saying # nothing; a family shape has parameters to resemble. @@ -1810,7 +1820,7 @@ async def _semantic_canon( query = concept_query(body) or body hits = await semantic_search_notes( user_id, query, limit=_SEMANTIC_LIMIT, - threshold=max(WRITEPATH_DEFAULT_THRESHOLD, _SEMANTIC_FLOOR), + threshold=WRITEPATH_DEFAULT_THRESHOLD, note_type="snippet", scope="browse", ) for score, note in hits: diff --git a/tests/test_divergence_meaning_gate.py b/tests/test_divergence_meaning_gate.py index 49967a5..864d00c 100644 --- a/tests/test_divergence_meaning_gate.py +++ b/tests/test_divergence_meaning_gate.py @@ -11,9 +11,11 @@ do. A signature does not carry a job. the floor" and let it withdraw the prompt. Measured live on 2026-09-22 it cannot discriminate. Judged instances of the service-unit canon score 0.68– 0.71 against it at best, most below 0.66; helpers score 0.66–0.75 against -snippets unrelated to them. Nothing reaches the 0.8 floor, so the "miss" -fires for nearly every body — the real divergence silenced exactly as the -helper was. The gate was removed; the prompt asks and the judge answers. +snippets unrelated to them. The then-floor of 0.8 was reached by nothing, so +the "miss" fired for nearly every body — the real divergence silenced exactly +as the helper was. The gate was removed; the prompt asks and the judge +answers. The same measurement moved the arm's floor down to the write-path +hint's: 0.8 was sized for proposals nobody reads, and it proposed 0 of 150. These tests pin what stayed: the arm proposes on a hit and says nothing on a miss, the divergence check does not consult it, and the capped pass reads new @@ -92,6 +94,18 @@ async def test_no_allowed_canon_spends_no_search() -> None: mock.assert_not_awaited() +async def test_the_arm_proposes_at_the_write_path_floor() -> None: + """It asks the write-path hint's question of the same documents, so it + uses that surface's floor — not a stricter private one that yields no + proposals for the judge to weigh (#4208).""" + from scribe.services.plugin_context import WRITEPATH_DEFAULT_THRESHOLD + + mock = _hits() + with _patch(mock): + await _semantic_canon(1, BODY, {CANON}) + assert mock.await_args.kwargs["threshold"] == WRITEPATH_DEFAULT_THRESHOLD + + # ── ...and its silence reaches nothing ─────────────────────────────────── diff --git a/tests/test_integration_shape_classify.py b/tests/test_integration_shape_classify.py index 2b8e329..a961e2d 100644 --- a/tests/test_integration_shape_classify.py +++ b/tests/test_integration_shape_classify.py @@ -897,7 +897,7 @@ async def test_a_semantic_miss_does_not_silence_the_prompt(seeded): `confirmDanger` beside an async confirm canon is structurally identical to a registry helper beside an async service canon, and the arm's miss was meant to tell them apart. Measured live, true instances of a canon rarely - clear the arm's 0.8 floor either, so a miss fires on both — silencing on + clear the arm's floor either, so a miss fires on both — silencing on it silenced #2793's own case. Here the arm has read the body, found nothing, and the prompt is still RAISED. -- 2.54.0 From a01deeb851db9daba1dccbcc5434c9c84e91632d Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Tue, 22 Sep 2026 08:44:58 -0400 Subject: [PATCH 3/3] fix(shapes): the signature basis admits the band true instances score in (#4306) Measured over 459 judged instance rows against every same-family canon: true instances score 1.0 or ~0.795 (a model whose mixin list differs from the canon's), and 0.8 cut the second band entirely. The proposal is read before it is confirmed, so the floor admits it: 0.75 takes every measured true row, at 35 wrong-canon pairings instead of 14. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01821k5B3Ysecp9fNYs92Kuy --- src/scribe/services/shape_ledger.py | 14 ++++++++++++-- tests/test_shape_ledger.py | 16 ++++++++++++++++ 2 files changed, 28 insertions(+), 2 deletions(-) diff --git a/src/scribe/services/shape_ledger.py b/src/scribe/services/shape_ledger.py index fc657d8..8556e68 100644 --- a/src/scribe/services/shape_ledger.py +++ b/src/scribe/services/shape_ledger.py @@ -1570,12 +1570,22 @@ def _semantic_priority(row) -> tuple: # not evidence (see _SEMANTIC_LIMIT) — but the proposals themselves did not # change, so no re-examination is owed; a stale miss basis is inert. # v5: the semantic arm's floor dropped from 0.8 to the write-path floor (#4208), -# so rows it examined and found nothing for must be read once more. +# and the signature floor from 0.8 to 0.75 (#4306), so rows examined and +# found nothing for must be read once more. _PROPOSER_VERSION = 5 # Signature resemblance floor, name blanked (difflib ratio) — and a length # floor, because `def NAME():` resembles `def NAME(x):` at 0.95 while saying # nothing; a family shape has parameters to resemble. -_SIGNATURE_FLOOR = 0.8 +# +# A proposal is read before it is confirmed, so this admits the band where +# true instances live rather than the band where nothing is wrong (#4306). +# Measured 2026-09-22 over 459 judged instance rows against every same-family +# canon: true instances score 1.0 or ~0.795 (a model class whose mixin list +# differs from the canon's), and 0.8 cut the whole second band. At 0.75 all +# of them clear; the price is `async def NAME(user_id: int, …) -> dict` +# look-alikes (35 wrong-canon pairings vs 14 at 0.8), which read as wrong at +# a glance. Below 0.75 the next true row is at 0.6 behind 150+ wrong ones. +_SIGNATURE_FLOOR = 0.75 _SIGNATURE_MIN_LEN = 30 # Textual containment needs enough substance to mean anything. _TEXT_FLOOR = 40 diff --git a/tests/test_shape_ledger.py b/tests/test_shape_ledger.py index 70fc648..61a3ef3 100644 --- a/tests/test_shape_ledger.py +++ b/tests/test_shape_ledger.py @@ -180,6 +180,22 @@ def test_signature_similarity_blanks_the_names(): assert sim("def helper(x):", "helper", "def make_app():", "make_app") == 0.0 +def test_a_model_with_a_different_mixin_list_is_proposed_against_the_model_canon(): + """The band 0.8 cut (#4306). Measured over judged instances: a model class + whose mixins differ from the canon's scores ~0.795 against it — a true + instance, and the proposal is read before it is confirmed, so the floor + must admit it.""" + from scribe.services.shape_ledger import Canon, _norm_text, match_canon + + model = Canon(44, "sym", "Gadget", (("src/app/models/gadget.py", "Gadget"),), + "class Gadget(Base, TimestampMixin, SoftDeleteMixin):", + _norm_text(""), 0, "python") + hit = match_canon("sym", "src/app/models/widget.py", "Widget", + "class Widget(Base, TimestampMixin):", " pass", [model]) + assert hit is not None and hit[:2] == (44, "signature") + assert 0.75 <= hit[2] < 0.8 + + def test_text_containment_is_whitespace_insensitive_with_a_floor(): from scribe.services.shape_ledger import text_contains -- 2.54.0