Compare commits
4
Commits
aea7b63b62
...
dev
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c3ecdf0972 | ||
|
|
1a34363059 | ||
|
|
30d87e461a | ||
|
|
8be555d6dd |
@@ -0,0 +1,52 @@
|
||||
"""add retrieval_logs.suppressed_count — tell a ranker decline from a repeat (#3497)
|
||||
|
||||
Revision ID: 0095
|
||||
Revises: 0094
|
||||
Create Date: 2026-09-03
|
||||
|
||||
`result_count == 0` has always meant "this surface said nothing", which is the
|
||||
right number for "was the hint any use" and the wrong one for tuning a
|
||||
threshold. It folds together two unrelated events:
|
||||
|
||||
- the ranker found nothing above the bar — the ONLY evidence a threshold is
|
||||
set too high; and
|
||||
- the ranker found something the session had already been shown — a decline
|
||||
that says nothing whatever about the bar.
|
||||
|
||||
The rule arms filter in Python after the search, so they can count the second
|
||||
kind exactly. The note arms pass `exclude_ids` INTO semantic_search_notes, so
|
||||
the dropped rows never come back and there is nothing to count.
|
||||
|
||||
NULLABLE, AND THE NULL IS THE POINT. A surface that does not measure
|
||||
suppression stores NULL, not 0, and the readout renders it as "not measured"
|
||||
rather than "none". Defaulting to 0 would make an unmeasured surface look like
|
||||
a perfectly clean one — the exact substitution of an artifact for a
|
||||
measurement that #3311 made and that #3497 exists to correct. Doing it again,
|
||||
in the migration that fixes it, would be its own small joke.
|
||||
|
||||
No backfill for the same reason: existing rows genuinely do not know, and
|
||||
saying so is the honest state. `retrieval_logs` is not restored from backup,
|
||||
so no importer changes.
|
||||
|
||||
Downgrade drops the column. Purely observational — nothing reads it for
|
||||
correctness.
|
||||
"""
|
||||
from alembic import op
|
||||
import sqlalchemy as sa
|
||||
|
||||
|
||||
revision = "0095"
|
||||
down_revision = "0094"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"retrieval_logs",
|
||||
sa.Column("suppressed_count", sa.Integer(), nullable=True),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("retrieval_logs", "suppressed_count")
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "scribe",
|
||||
"description": "Scribe system-of-record for Claude Code: MCP tools over your notes/tasks/projects/rules, a session-start push channel that surfaces your always-on rules + active-project context, process-skills (writing-plans, systematic-debugging, verification, brainstorming, reusing-code), and your saved Scribe Processes auto-surfaced as skills (/scribe:sync). Replaces superpowers + file-memory with one app-backed plugin.",
|
||||
"version": "2026.09.03.0329",
|
||||
"version": "2026.09.04.0140",
|
||||
"author": {
|
||||
"name": "Bryan Van Deusen"
|
||||
},
|
||||
|
||||
@@ -21,6 +21,16 @@ for the operator's work, and as your own working memory across sessions.
|
||||
compaction — call `list_always_on_rules()` (and `enter_project()` when a
|
||||
project is in scope) BEFORE acting. When a loaded rule and a default habit
|
||||
disagree, the rule wins; if no rule speaks to it, ask rather than assume.
|
||||
- **What you loaded is not all of the rules.** Only the always-on tier arrives
|
||||
that way; conditional rules are RETRIEVED, and one you were never handed
|
||||
binds exactly as hard. So before a consequential act, `search` for a rule
|
||||
about it (`content_type="rule"`) rather than concluding from an empty
|
||||
loaded set that nothing applies. "I was not told" is not the same as "there
|
||||
is no rule," and only one of those is checkable.
|
||||
This bites hardest on which TOOL to reach for — curling an API that has an
|
||||
MCP client, standing up a local stack, running a suite CI owns. Those feel
|
||||
like mechanics rather than decisions, so they raise no doubt and generate no
|
||||
query; the moment you are most confident is the moment to look.
|
||||
- **Recall before acting** — before you answer anything about the operator's
|
||||
work or start a task, `search` Scribe first; assume a related note, task, or
|
||||
decision already exists. Concretely, reach for recall whenever a request
|
||||
|
||||
@@ -56,10 +56,31 @@ Two constraints on *how* that's achieved:
|
||||
re-deriving it or opening a duplicate. When a project is in scope, pass its
|
||||
`project_id` so results stay scoped.
|
||||
|
||||
2. **Standing rules are binding.** Load them via `list_always_on_rules()` at
|
||||
session start (see "Do this first"); treat every one as binding. Pull a
|
||||
rule's full statement with `get_rule(id)` when it's about to bite. When a
|
||||
project is in scope, `enter_project(id)` also returns its applicable rules.
|
||||
2. **Standing rules are binding — and the ones you were handed are not all of
|
||||
them.** Load the resident set via `list_always_on_rules()` at session start
|
||||
(see "Do this first"); treat every one as binding. Pull a rule's full
|
||||
statement with `get_rule(id)` when it's about to bite. When a project is in
|
||||
scope, `enter_project(id)` also returns its applicable rules.
|
||||
|
||||
Rules come in two tiers. **Always-on** rules are delivered — they arrive
|
||||
whether or not you ask. **Conditional** rules are RETRIEVED, and one binds
|
||||
just as hard for never having been handed to you. So before a consequential
|
||||
act, `search(content_type="rule")` on what you are about to do. An empty
|
||||
loaded set is not evidence that no rule applies; it is only evidence that
|
||||
none was pushed, and those are different claims.
|
||||
|
||||
The tier split exists because delivery does not scale: every resident rule
|
||||
costs tokens in every session forever, so a rulebook that grows past a few
|
||||
dozen either stops growing or stops fitting. Retrieval is what lets the
|
||||
rulebook keep growing — but retrieval only fires if something asks.
|
||||
|
||||
**Ask hardest where you feel most certain.** Rules about which TOOL to reach
|
||||
for — use the forge's MCP client rather than curling its API, don't stand up
|
||||
a local stack, don't run the suite CI owns — govern moves that feel like
|
||||
mechanics rather than decisions. A reflex raises no doubt, so it generates
|
||||
no query, so the rule that would have stopped it is never retrieved. That is
|
||||
the failure this instruction exists to prevent, and confidence is its only
|
||||
warning sign.
|
||||
|
||||
3. **Update over duplicate.** When recording, prefer updating an existing
|
||||
note/rule/task over creating a new one. Search first; revise what's there.
|
||||
|
||||
@@ -45,6 +45,31 @@ from quart import Quart
|
||||
# The accepted cost: an agent that never opens create_note's docstring never
|
||||
# learns the field exists. Guidance lives in the create_note / update_note
|
||||
# docstrings and the using-scribe skill instead.
|
||||
#
|
||||
# Milestone 333 step 3 (2026-09-04) bought the HOW bullet's second clause —
|
||||
# search(content_type="rule") before a consequential act — by TRADING OUT
|
||||
# "Processes are saved procedures (follow verbatim)" and "Deletes are
|
||||
# trash-recoverable". Recorded so the trade is not silently reversed:
|
||||
# - Both were already in test_instruction_surfaces_agree's DISPLACED_TOPICS
|
||||
# and already stated on a delivered surface, so nothing fell off: the
|
||||
# process reflex is in every scribe-proc-* skill listing (each says the
|
||||
# process governs and is followed verbatim), and trash recovery is in the
|
||||
# delete_*/list_trash/restore docstrings, which is where per-tool guidance
|
||||
# belongs by this block's own doctrine.
|
||||
# - What it bought is not per-tool guidance and has nowhere else to live at
|
||||
# session-start altitude. Rules were retrievable only by RESIDENCY: the
|
||||
# always-on preload put them in front of the agent, and nothing told a
|
||||
# session to go looking for one it had not been handed. The tier split is
|
||||
# therefore load-bearing on ANY install (rule 115): a delivered rule costs
|
||||
# tokens in every session forever, so a rulebook that only delivers cannot
|
||||
# grow past what one session can hold, and every rule worth keeping has to
|
||||
# become resident to bind at all. Retrieval is what lets it keep growing —
|
||||
# and retrieval fires only if something asks, which nothing told a session
|
||||
# to do. A tool-choice reflex asks least of all (#3476, #161).
|
||||
# - This states the PULL for conditional rules, exactly as the surrounding
|
||||
# line states it for always-on ones. Rule 119 makes these surfaces the
|
||||
# specification, so the same sentence lands on all three session-start
|
||||
# surfaces, and test_instruction_surfaces_agree pins it.
|
||||
_INSTRUCTIONS = """
|
||||
Scribe is the operator's self-hosted second brain and system of record — and
|
||||
yours: recall from it before acting, record as you go. Keep no parallel copy
|
||||
@@ -63,13 +88,13 @@ Hierarchy: Project -> Milestone -> Task/Note. The map, by purpose:
|
||||
active project_id to stay in scope.
|
||||
- WHERE work happens: Systems. Tag records with system_ids as you write;
|
||||
create_system when the area is unmodelled.
|
||||
- HOW: rules are binding — list_always_on_rules() at session start.
|
||||
- HOW: rules bind. list_always_on_rules() at start; before a consequential
|
||||
act, search(content_type="rule") — the resident set is not all of them.
|
||||
- UI: the project's design system is binding — resolve_design_system /
|
||||
get_design_system_stylesheet before hand-writing a value.
|
||||
- REUSE: search snippets before writing a helper; record what you build with
|
||||
create_snippet; classify shapes against canon (classify_shapes) — a
|
||||
consumer map is rows, never prose. Processes are saved procedures (follow
|
||||
verbatim). Deletes are trash-recoverable.
|
||||
consumer map is rows, never prose.
|
||||
|
||||
A task is a note with status (*_note vs *_task tools).
|
||||
Creates are duplicate-gated: a near-match BLOCKS and returns the existing
|
||||
|
||||
@@ -315,6 +315,61 @@ async def create_rule(
|
||||
) -> dict:
|
||||
"""Create a new rule in a rulebook (a SHARED rule — keep it general).
|
||||
|
||||
PROPOSE RULES READILY, AND WRITE ONE WHEN THE OPERATOR SAYS YES. Noticing
|
||||
that something has hardened into a standing instruction is valuable work,
|
||||
and a session that notices it and says nothing has thrown the observation
|
||||
away. So raise it whenever you see one. The single step that belongs
|
||||
between noticing and writing is the operator's yes: a rule binds every
|
||||
future session, and they are the person it binds.
|
||||
|
||||
Their yes is also the only moment the rule is reliably IN FRONT of them.
|
||||
After the write it may not be again for months — a conditional rule is not
|
||||
read aloud at session start, and a project-scoped one does not appear in
|
||||
an unfiltered list_rules() at all. So the proposal is the review.
|
||||
|
||||
When the operator asks for a rule in so many words, that IS the yes —
|
||||
write it and move on. The loop below is for the rule you thought of.
|
||||
|
||||
A PROPOSAL CARRIES FOUR THINGS, and the fourth is the one that decides it:
|
||||
|
||||
1. WHAT it would require — the statement, in the words it would carry,
|
||||
not a gloss of them. The operator is agreeing to text.
|
||||
2. INTENT — what it changes about how work gets done, and what goes
|
||||
wrong today without it. "Be careful about X" is not an intent; the
|
||||
behaviour that would differ tomorrow is.
|
||||
3. WHY NOW — the incident, observation or decision behind it. Pass that
|
||||
record as arose_from_id, and say it in the conversation too: the
|
||||
field is for the reader six months out, the sentence is for the
|
||||
person deciding.
|
||||
4. HOW IT WOULD BE ENFORCED — a test, a CI check, a hook, a schema
|
||||
constraint, a duplicate gate, a review step... or nothing, in which
|
||||
case say so plainly: "nothing — this is prose a session has to
|
||||
remember." Answer this one honestly and it will sometimes dissolve
|
||||
the rule, which is the point rather than a side effect. What a test
|
||||
can assert should BE that test; a rule is what remains when nothing
|
||||
mechanical can hold the thing. A rulebook grows by default and
|
||||
shrinks only on purpose, so a question that prevents a rule is worth
|
||||
more than any question that improves one's wording.
|
||||
|
||||
THEN CLOSE WITH A QUESTION THEY CAN ANSWER IN ONE WORD. Offer three
|
||||
answers, and make the middle one the easy one:
|
||||
|
||||
* "Approve it AS WRITTEN" — you create it with the statement exactly as
|
||||
shown. This is what makes element 1 load-bearing: they approved TEXT,
|
||||
so that text is what gets stored, verbatim.
|
||||
* "LET'S TALK ABOUT IT" — the wording, the scope, the tier, whether it
|
||||
wants to be a rule at all. Most good rules arrive this way, so treat
|
||||
this answer as the expected one rather than a setback.
|
||||
* "NO" — let it go. If the observation is still worth keeping, it is a
|
||||
note (create_note): recorded, findable, and binding on nobody.
|
||||
|
||||
Where the interface offers structured choices, ask it that way — a
|
||||
question with named options is answered in a click, while the same
|
||||
question inside a paragraph is answered by scrolling past. Where it does
|
||||
not, write the three options out as three options. Either way ask once
|
||||
and let the answer stand; re-raising a declined proposal argues a rule
|
||||
into existence, which is the thing this whole loop exists to prevent.
|
||||
|
||||
A rulebook rule is shared by every project that gets the rulebook: an
|
||||
always_on rulebook binds ALL your projects; a subscribed rulebook binds the
|
||||
projects that opt in. So a rulebook rule must read as a general standard —
|
||||
@@ -433,6 +488,16 @@ async def create_project_rule(
|
||||
the rule is returned in get_project's applicable_rules (under
|
||||
project_rules) and in list_rules(project_id=...).
|
||||
|
||||
PROPOSE, THEN WRITE ON A YES — create_rule's opening carries the whole
|
||||
loop: the four things a proposal states (what it would require, its
|
||||
intent, why now, and how it would be enforced) and the one-word question
|
||||
that closes it (approve as written / talk about it / no). All of it
|
||||
applies here unchanged. Reach for that loop MORE readily on this surface,
|
||||
not less: a project rule stays out of an unfiltered list_rules(), and a
|
||||
conditional one stays out of session start too, so the operator's yes is
|
||||
the one moment this rule is certain to have been seen by the person it
|
||||
binds.
|
||||
|
||||
Check first whether a rule is the right shape at all — create_rule's
|
||||
opening asks that question and it applies identically here. A visual
|
||||
standard is a design system; a procedure is a process (create_process);
|
||||
|
||||
@@ -168,6 +168,20 @@ async def retrieval_telemetry(days: int = 30) -> dict:
|
||||
against `calls`, with the spread beside it: a surface that clears its bar
|
||||
on nearly every call is either well-tuned or too loose, and p10 says which.
|
||||
|
||||
READ `cleared_threshold` AND `zero_result_calls` TOGETHER, and check
|
||||
`suppression` before concluding anything from either. A zero-result call is
|
||||
two different events wearing one number: the ranker found nothing above the
|
||||
bar, or it found only what this session had already been shown. Just the
|
||||
first is evidence the bar is too high. `suppression` splits them where the
|
||||
surface can tell — `zero_because_already_shown` comes off
|
||||
`zero_result_calls` to leave the true ranker declines.
|
||||
|
||||
`suppression` is `null` when NO row in the window reported it, and that is
|
||||
"not measured here", NOT "none suppressed". Surfaces that pass their
|
||||
exclusions into the search never see what was dropped, so they cannot say.
|
||||
Do not read a null as a zero: reading an artifact as a measurement is how
|
||||
this surface got mis-scoped once already (#3311, #3497).
|
||||
|
||||
`usage` — NOTES ONLY, from `note_usage_events`, at the per-note grain
|
||||
`retrieval_logs` cannot be indexed at: `surfaced` (ranked surfacings — a
|
||||
scored surface CHOSE the record), `ambient` (the rest), `pulled` split into
|
||||
|
||||
@@ -42,6 +42,16 @@ class RetrievalLog(Base):
|
||||
# False=notes, NULL=any.
|
||||
is_task: Mapped[bool | None] = mapped_column(Boolean, nullable=True)
|
||||
result_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
# How many scored hits this call DROPPED because the session had already
|
||||
# been shown them. NULLABLE, and the null is load-bearing: it means "this
|
||||
# surface does not report suppression", which must not read as "nothing was
|
||||
# suppressed". `result_count == 0` alone conflates two different events —
|
||||
# the ranker found nothing above threshold, and the ranker found something
|
||||
# the reader already had — and only the first says a threshold is too high.
|
||||
# Reading a zero as a ranker decline is how #3311 mis-scoped a milestone;
|
||||
# an unmeasured value that renders as 0 is the same mistake with a nicer
|
||||
# face, so surfaces that filter INSIDE the search leave this null.
|
||||
suppressed_count: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
||||
top_score: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||
min_score: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||
# [{"id": int, "score": float, "rank": int}, ...], highest-first.
|
||||
@@ -67,6 +77,7 @@ class RetrievalLog(Base):
|
||||
"project_id": self.project_id,
|
||||
"is_task": self.is_task,
|
||||
"result_count": self.result_count,
|
||||
"suppressed_count": self.suppressed_count,
|
||||
"top_score": self.top_score,
|
||||
"min_score": self.min_score,
|
||||
"result_ids": self.result_ids,
|
||||
|
||||
@@ -1253,6 +1253,11 @@ async def build_write_path_hint(
|
||||
threshold=cfg["rule_threshold"], limit=RULEHINT_LIMIT,
|
||||
project_id=project_id,
|
||||
is_task=None, results=fresh, duration_ms=rule_ms,
|
||||
# What the ranker found and this session had already been told.
|
||||
# Without it a zero row cannot say whether the bar was too high or
|
||||
# the reader was simply ahead of it — and only the first is a
|
||||
# reason to move the threshold.
|
||||
suppressed=len(hits) - len(fresh),
|
||||
)
|
||||
if fresh:
|
||||
# `rule_ids` is `fresh`, i.e. AFTER exclude_rule_ids. A rule the
|
||||
@@ -1352,6 +1357,10 @@ async def build_tool_rule_hint(
|
||||
threshold=cfg["rule_threshold"], limit=RULEHINT_LIMIT,
|
||||
project_id=project_id,
|
||||
is_task=None, results=fresh, duration_ms=duration_ms,
|
||||
# See the sibling arm. It matters more here: this arm fires on every
|
||||
# Bash call, so a long session excludes its way to an all-zero row
|
||||
# and the threshold looks wrong when nothing about it is.
|
||||
suppressed=len(hits) - len(fresh),
|
||||
)
|
||||
if not fresh:
|
||||
return out
|
||||
|
||||
@@ -55,12 +55,18 @@ def _build_payload(
|
||||
is_task: bool | None,
|
||||
results: list[tuple[float, Note]],
|
||||
duration_ms: float | None,
|
||||
suppressed: int | None = None,
|
||||
) -> dict:
|
||||
"""Reduce a retrieval call to a flat, JSON-safe RetrievalLog payload.
|
||||
|
||||
Pure and synchronous (no DB, no event loop) so it is unit-testable and safe
|
||||
to run inline before scheduling the write. `results` is the
|
||||
`(score, Note)` list from semantic_search_notes, already highest-first.
|
||||
|
||||
`suppressed` is how many scored hits the caller dropped because the session
|
||||
had already been shown them, and it stays None for callers that cannot
|
||||
know. See the column's comment: None means "not measured here", which is a
|
||||
different fact from 0 and must never render as one.
|
||||
"""
|
||||
items = [
|
||||
{"id": int(note.id), "score": round(float(score), 5), "rank": rank}
|
||||
@@ -76,6 +82,7 @@ def _build_payload(
|
||||
"project_id": project_id,
|
||||
"is_task": is_task,
|
||||
"result_count": len(items),
|
||||
"suppressed_count": (None if suppressed is None else int(suppressed)),
|
||||
"top_score": (scores[0] if scores else None),
|
||||
"min_score": (scores[-1] if scores else None),
|
||||
"result_ids": items,
|
||||
@@ -115,6 +122,7 @@ def record_retrieval(
|
||||
is_task: bool | None,
|
||||
results: list[tuple[float, Any]],
|
||||
duration_ms: float | None = None,
|
||||
suppressed: int | None = None,
|
||||
) -> None:
|
||||
"""Fire-and-forget: record one retrieval call.
|
||||
|
||||
@@ -140,6 +148,7 @@ def record_retrieval(
|
||||
is_task=is_task,
|
||||
results=results,
|
||||
duration_ms=duration_ms,
|
||||
suppressed=suppressed,
|
||||
)
|
||||
except Exception:
|
||||
logger.debug("retrieval telemetry payload build failed", exc_info=True)
|
||||
@@ -166,7 +175,8 @@ def record_retrieval(
|
||||
|
||||
def _bucket(rows: list) -> dict:
|
||||
"""A score readout a human can act on, from one aggregate row."""
|
||||
calls, zero, cleared, p10, p50, p90, lo, hi, avg_n, dur = rows
|
||||
(calls, zero, cleared, p10, p50, p90, lo, hi, avg_n, dur,
|
||||
measured, supp_calls, supp_zero) = rows
|
||||
return {
|
||||
"calls": int(calls or 0),
|
||||
# A call that returned nothing is not a low-scoring call — it is a
|
||||
@@ -178,6 +188,22 @@ def _bucket(rows: list) -> dict:
|
||||
# bar on almost every call is either well-tuned or too loose, and the
|
||||
# score spread below says which.
|
||||
"cleared_threshold": int(cleared or 0),
|
||||
# Of the zeros above, which were the RANKER declining and which were
|
||||
# the reader having seen it already? `zero_result_calls` cannot say,
|
||||
# and only the first kind is evidence about the threshold.
|
||||
#
|
||||
# None — not a zeroed dict — when no row in the window reported it. A
|
||||
# surface that filters inside the search genuinely does not know, and
|
||||
# rendering that as `{"calls": 0}` would state a measurement nobody
|
||||
# made. That substitution is the whole of #3311.
|
||||
"suppression": (
|
||||
None if not int(measured or 0) else {
|
||||
"measured_calls": int(measured or 0),
|
||||
"calls_with_suppression": int(supp_calls or 0),
|
||||
# Subtract from zero_result_calls for the true ranker declines.
|
||||
"zero_because_already_shown": int(supp_zero or 0),
|
||||
}
|
||||
),
|
||||
"top_score": {
|
||||
"p10": _round(p10), "p50": _round(p50), "p90": _round(p90),
|
||||
"min": _round(lo), "max": _round(hi),
|
||||
@@ -240,6 +266,14 @@ async def retrieval_summary(user_id: int | None, *, days: int = 30) -> dict:
|
||||
else_=0,
|
||||
)
|
||||
zero = case((RetrievalLog.result_count == 0, 1), else_=0)
|
||||
# Three sums rather than one, because "not measured" and "measured as zero"
|
||||
# are different answers and a single counter cannot hold both.
|
||||
measured = case((RetrievalLog.suppressed_count.isnot(None), 1), else_=0)
|
||||
supp_calls = case((RetrievalLog.suppressed_count > 0, 1), else_=0)
|
||||
supp_zero = case(
|
||||
((RetrievalLog.result_count == 0) & (RetrievalLog.suppressed_count > 0), 1),
|
||||
else_=0,
|
||||
)
|
||||
|
||||
def pct(p: float):
|
||||
return func.percentile_cont(p).within_group(RetrievalLog.top_score.asc())
|
||||
@@ -266,6 +300,9 @@ async def retrieval_summary(user_id: int | None, *, days: int = 30) -> dict:
|
||||
func.percentile_cont(0.9).within_group(
|
||||
RetrievalLog.duration_ms.asc()
|
||||
),
|
||||
func.sum(measured).label("measured"),
|
||||
func.sum(supp_calls).label("supp_calls"),
|
||||
func.sum(supp_zero).label("supp_zero"),
|
||||
)
|
||||
.where(
|
||||
RetrievalLog.created_at >= since,
|
||||
|
||||
@@ -195,6 +195,46 @@ def test_displaced_topics_live_on_a_delivered_surface():
|
||||
)
|
||||
|
||||
|
||||
# The SECOND pull (milestone 333 step 3). `list_always_on_rules()` fetches the
|
||||
# resident tier; this one says that tier is not all of them, and that a
|
||||
# conditional rule has to be gone looking for. However a surface words the
|
||||
# surrounding prose, it names the call.
|
||||
RETRIEVE = 'content_type="rule"'
|
||||
|
||||
|
||||
def test_every_session_start_surface_states_the_conditional_retrieval():
|
||||
"""The push/pull asymmetry, one level in.
|
||||
|
||||
The tests above pin that a session PULLS the resident rules rather than
|
||||
trusting the SessionStart push. This pins the same shape between the two
|
||||
TIERS: an always-on rule is delivered, a conditional one is retrieved, and
|
||||
a surface that states only the first leaves a session reading its loaded
|
||||
set as the whole rulebook.
|
||||
|
||||
That reading is wrong in the direction that costs something. "Nothing was
|
||||
pushed" and "no rule applies" are different claims, and only one of them
|
||||
has been checked — the same asymmetry as #2198, now between tiers instead
|
||||
of between channels.
|
||||
|
||||
It is also what made the always-on tier the only one that worked, on any
|
||||
install rather than this one (rule 115). A rule nothing retrieves has to be
|
||||
resident to bind at all, so every rule worth keeping becomes resident; and
|
||||
a resident rule costs tokens in every session forever, so a rulebook that
|
||||
only delivers cannot grow past what one session can hold. Retrieval is what
|
||||
lifts that ceiling — and it only fires if something asks.
|
||||
"""
|
||||
missing = []
|
||||
for path in SESSION_START_SURFACES:
|
||||
if RETRIEVE not in path.read_text():
|
||||
missing.append(str(path.relative_to(ROOT)))
|
||||
assert not missing, (
|
||||
f"these surfaces state the always-on pull but never tell the agent to "
|
||||
f"retrieve a conditional rule ({RETRIEVE}): {missing}. A session that "
|
||||
f"reads its loaded set as the whole rulebook will act on \"I was not "
|
||||
f"told\" as if it meant \"there is no rule\" (milestone 333 step 3)."
|
||||
)
|
||||
|
||||
|
||||
def test_no_surface_names_the_push_without_stating_the_pull():
|
||||
"""The exact shape #2497 took.
|
||||
|
||||
|
||||
@@ -0,0 +1,120 @@
|
||||
"""Both rule-creation tools run the propose-then-approve loop (#3557).
|
||||
|
||||
WHY THIS EXISTS
|
||||
|
||||
Every other gate on `create_rule` and `create_project_rule` is about SHAPE:
|
||||
is this a rule or a process, is it one thing you could violate, is it general
|
||||
enough for a rulebook, is it a near-duplicate. All of those improve a rule
|
||||
someone has already decided to write. None of them asks the prior question —
|
||||
whether the person the rule will bind has agreed to be bound by it.
|
||||
|
||||
That question belongs at the tool, because the tool is the last surface a
|
||||
caller reads before the write, and because the write is less reversible than
|
||||
it looks. The operator's yes is not merely consent; it is the one moment the
|
||||
rule is certainly IN FRONT of them. Afterwards it may not be again for
|
||||
months: a conditional rule is not read aloud at session start, and a
|
||||
project-scoped rule does not appear in an unfiltered `list_rules()` at all.
|
||||
The proposal IS the review, so there had better be one.
|
||||
|
||||
WHY IT IS PHRASED AS A PRACTICE AND NOT A PROHIBITION
|
||||
|
||||
The first cut of this guidance opened "NOT YOURS TO CALL UNPROMPTED." That is
|
||||
the wrong instrument, and the failure it invites is worse than the one it
|
||||
prevents: a caller reading a prohibition stops NOTICING rule-shaped things,
|
||||
rather than noticing them and asking. The wanted behaviour is more proposals,
|
||||
not fewer — spotting that something has hardened into a standing instruction
|
||||
is valuable work, and the only step that was ever missing came after it.
|
||||
|
||||
So the docstrings describe what to DO: propose readily, state four things,
|
||||
close with a question the operator answers in one word. This test is written
|
||||
the same way — it asserts the parts of the loop are present, and has nothing
|
||||
to say about any wording that forbids.
|
||||
|
||||
The fourth element — how the rule would be ENFORCED — is not ceremony. It is
|
||||
the part that sometimes dissolves the rule: a thing a test can assert should
|
||||
be that test, and a rule is what is left when nothing mechanical can hold it.
|
||||
A rulebook grows by default and shrinks only on purpose, so the question that
|
||||
prevents a rule earns more than any question that improves one's wording.
|
||||
|
||||
WHAT THIS PINS, AND WHAT IT DOES NOT
|
||||
|
||||
STRUCTURE, never wording — the same bargain the disambiguator guard (#3123)
|
||||
strikes next door. Each element matches a family of synonyms, so the prose
|
||||
stays free to be rewritten, reordered or sharpened; only DELETING one fails.
|
||||
Pinning phrasing would make every improvement a red build, and a test that
|
||||
punishes editing is a test someone deletes.
|
||||
|
||||
It cannot tell whether an agent actually proposes. Nothing in a docstring
|
||||
can. It catches the regression that really happens: guidance tidied away in
|
||||
a later pass by someone who read it as throat-clearing in front of the Args.
|
||||
"""
|
||||
import pytest
|
||||
|
||||
from tests.helpers import tool_doc as _doc
|
||||
|
||||
# Both surfaces, because the one that needs it most is the one that looks
|
||||
# minor. A project rule is the least visible record the system can hold —
|
||||
# absent from an unfiltered list_rules(), and absent from session start too
|
||||
# whenever it is conditional — so the surface that writes one carries the
|
||||
# larger risk while reading as the smaller act.
|
||||
_SURFACES = [
|
||||
("scribe.mcp.tools.rulebooks", "create_rule"),
|
||||
("scribe.mcp.tools.rulebooks", "create_project_rule"),
|
||||
]
|
||||
|
||||
# The loop, element by element, each as a family of ways to say it. A
|
||||
# docstring satisfies an element by containing ANY member — that is the room
|
||||
# left for rewriting. The families deliberately exclude bare words a
|
||||
# docstring would hold by accident ("why", "how", "reason", "rule"), which
|
||||
# would let the assertion pass on prose that says nothing of the kind.
|
||||
_ELEMENTS = {
|
||||
"the invitation to propose": ("propose", "proposal"),
|
||||
"the operator's approval": ("approve", "approval", "says yes", "a yes"),
|
||||
"the rule's intent": ("intent", "what it changes about how work"),
|
||||
"why it is being proposed now": (
|
||||
"why now", "arose_from_id", "the incident", "prompted it",
|
||||
),
|
||||
"how it would be enforced": ("enforc",),
|
||||
"the answers offered back": ("as written", "talk about it", "discuss"),
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(("module", "name"), _SURFACES)
|
||||
@pytest.mark.parametrize("element", sorted(_ELEMENTS))
|
||||
def test_a_rule_surface_carries_every_part_of_the_proposal_loop(
|
||||
module, name, element
|
||||
):
|
||||
"""Each element of propose → state four things → ask survives."""
|
||||
doc = _doc(module, name).lower()
|
||||
assert any(token in doc for token in _ELEMENTS[element]), (
|
||||
f"{name}'s docstring no longer mentions {element}. A caller reads "
|
||||
f"this immediately before writing a rule that will bind every future "
|
||||
f"session, and the proposal is the one moment that rule is certain to "
|
||||
f"be seen by the operator. Say it in whatever words you like; this "
|
||||
f"guard only checks it is still said. See create_rule's opening."
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(("module", "name"), _SURFACES)
|
||||
def test_the_proposal_loop_comes_before_the_parameter_contract(module, name):
|
||||
"""It has to be read to work, and the Args: block is where reading stops.
|
||||
|
||||
A caller who has decided to make the call skims down to the parameters.
|
||||
Guidance parked below them — or folded into one argument's description —
|
||||
arrives after the decision it was meant to inform, which is the same as
|
||||
not being there.
|
||||
"""
|
||||
doc = _doc(module, name).lower()
|
||||
args_at = doc.find("args:")
|
||||
assert args_at > 0, f"{name}'s docstring has no Args: block"
|
||||
loop_at = min(
|
||||
(doc.find(t) for t in _ELEMENTS["the invitation to propose"]
|
||||
if doc.find(t) >= 0),
|
||||
default=-1,
|
||||
)
|
||||
assert 0 <= loop_at < args_at, (
|
||||
f"{name} introduces the proposal loop at or after its Args: block "
|
||||
f"(loop {loop_at}, args {args_at}). Move it to the opening — a "
|
||||
f"caller who has already decided to write the rule reads the "
|
||||
f"parameters, not the prose under them."
|
||||
)
|
||||
@@ -702,3 +702,65 @@ def test_neither_rule_arm_logs_its_call_behind_a_results_guard():
|
||||
"the pre-tool arm returns before logging its call — a surface with no "
|
||||
"rows at all cannot be told apart from a hook that never fired"
|
||||
)
|
||||
|
||||
|
||||
# ── Suppression: which zeros were the ranker, which were repeats (#3497) ──
|
||||
#
|
||||
# Making the call log unconditional exposed a second ambiguity in the same row.
|
||||
# A zero-result rule call is two unrelated events: the ranker found nothing
|
||||
# above the bar, or it found only what this session already held. Only the
|
||||
# first says anything about the threshold, and a long session excludes its way
|
||||
# into the second — so without the split, the arm looks worse the longer it
|
||||
# runs correctly.
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_the_write_path_arm_reports_what_the_session_already_held():
|
||||
log, rec = MagicMock(), MagicMock()
|
||||
hits = [(0.71, fake_rule(id=156, title="A wait with no deadline is a bug")),
|
||||
(0.70, fake_rule(id=157, title="A loop re-arms in a finally"))]
|
||||
await _run_arm(hits, rec, retrieval_log=log, exclude_rule_ids=[156, 157])
|
||||
|
||||
row = next(c for c in log.call_args_list
|
||||
if c.kwargs.get("source") == "write_path_rule")
|
||||
assert row.kwargs["results"] == []
|
||||
assert row.kwargs["suppressed"] == 2, (
|
||||
"both hits were repeats, so this zero is not evidence about the bar"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_genuine_ranker_decline_reports_zero_suppression():
|
||||
"""Zero, not None. The arm filters in Python, so it always knows — and
|
||||
'measured none' has to stay distinguishable from 'cannot measure'."""
|
||||
log, rec = MagicMock(), MagicMock()
|
||||
await _run_arm([], rec, retrieval_log=log)
|
||||
|
||||
row = next(c for c in log.call_args_list
|
||||
if c.kwargs.get("source") == "write_path_rule")
|
||||
assert row.kwargs["suppressed"] == 0
|
||||
assert row.kwargs["suppressed"] is not None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_the_tool_arm_reports_suppression_too():
|
||||
log, rec = MagicMock(), MagicMock()
|
||||
hits = [(0.75, fake_rule(id=161, title="Reach the forge through its MCP tools"))]
|
||||
await _run_tool_arm(hits, rec, retrieval_log=log, exclude_rule_ids=[161])
|
||||
|
||||
assert log.call_args.kwargs["results"] == []
|
||||
assert log.call_args.kwargs["suppressed"] == 1
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_shown_hit_is_not_counted_as_suppressed():
|
||||
"""The obvious inverse, worth pinning: `suppressed` counts what was DROPPED,
|
||||
not what came back. Off by one here and every zero row reads as a repeat."""
|
||||
log, rec = MagicMock(), MagicMock()
|
||||
hits = [(0.75, fake_rule(id=161, title="Reach the forge through its MCP tools")),
|
||||
(0.70, fake_rule(id=12, title="Don't run a local stack unless asked"))]
|
||||
out = await _run_tool_arm(hits, rec, retrieval_log=log, exclude_rule_ids=[12])
|
||||
|
||||
assert out["rule_ids"] == [161]
|
||||
assert log.call_args.kwargs["suppressed"] == 1
|
||||
assert len(log.call_args.kwargs["results"]) == 1
|
||||
|
||||
@@ -57,6 +57,68 @@ def test_build_payload_rounds_scores_to_5dp():
|
||||
assert p["result_ids"][0]["score"] == 0.12346
|
||||
|
||||
|
||||
# ─── suppression: "not measured" is not "none" (#3497) ───────────────────────
|
||||
|
||||
|
||||
def test_a_caller_that_cannot_measure_suppression_stores_null():
|
||||
"""The distinction the whole column exists for.
|
||||
|
||||
A surface that passes its exclusions into the search never sees what was
|
||||
dropped. Storing 0 would assert a clean run nobody observed — reading an
|
||||
artifact as a measurement, which is exactly #3311's mistake.
|
||||
"""
|
||||
p = _build_payload(
|
||||
user_id=1, source="auto_inject", query="q", threshold=0.6,
|
||||
limit=3, project_id=None, is_task=None, results=[], duration_ms=None,
|
||||
)
|
||||
assert p["suppressed_count"] is None, "unmeasured must not render as zero"
|
||||
|
||||
|
||||
def test_a_caller_that_measured_no_suppression_stores_zero():
|
||||
"""The other side of it. Zero is a real observation and must survive."""
|
||||
p = _build_payload(
|
||||
user_id=1, source="pre_tool_rule", query="git status", threshold=0.6,
|
||||
limit=1, project_id=None, is_task=None, results=[], duration_ms=None,
|
||||
suppressed=0,
|
||||
)
|
||||
assert p["suppressed_count"] == 0
|
||||
|
||||
|
||||
def test_the_count_of_hits_the_reader_already_held_is_carried():
|
||||
p = _build_payload(
|
||||
user_id=1, source="write_path_rule", query="code", threshold=0.6,
|
||||
limit=2, project_id=None, is_task=None, results=[], duration_ms=None,
|
||||
suppressed=2,
|
||||
)
|
||||
assert p["result_count"] == 0
|
||||
assert p["suppressed_count"] == 2, (
|
||||
"a zero row that was really two repeats must be distinguishable from "
|
||||
"a zero row where the ranker found nothing"
|
||||
)
|
||||
|
||||
|
||||
def test_the_readout_reports_unmeasured_suppression_as_none():
|
||||
"""`_bucket` renders the aggregate. No row reporting it → null, never a
|
||||
zeroed dict: a zeroed dict states a measurement nobody made."""
|
||||
from scribe.services.retrieval_telemetry import _bucket
|
||||
|
||||
# calls, zero, cleared, p10, p50, p90, min, max, avg_n, dur,
|
||||
# measured, supp_calls, supp_zero
|
||||
unmeasured = _bucket([326, 114, 212, 0.6, 0.68, 0.77, 0.55, 0.85, 1.7, 130.9,
|
||||
0, 0, 0])
|
||||
assert unmeasured["suppression"] is None
|
||||
|
||||
measured = _bucket([35, 34, 1, 0.75, 0.75, 0.75, 0.75, 0.75, 0.03, 51.9,
|
||||
35, 9, 9])
|
||||
assert measured["suppression"] == {
|
||||
"measured_calls": 35,
|
||||
"calls_with_suppression": 9,
|
||||
"zero_because_already_shown": 9,
|
||||
}
|
||||
# The number the threshold is actually tuned from.
|
||||
assert measured["zero_result_calls"] - 9 == 25
|
||||
|
||||
|
||||
def test_record_retrieval_without_event_loop_is_safe():
|
||||
"""Called from a sync context (no running loop) it must swallow and return,
|
||||
never raise — telemetry can't be allowed to break a caller."""
|
||||
|
||||
Reference in New Issue
Block a user