feat(rulings): system_usage_events is read back — per-System counts and a telemetry block (#4769)
CI & Build / Python lint (push) Successful in 3s
CI & Build / Plugin hooks (push) Successful in 13s
CI & Build / integration (push) Successful in 50s
CI & Build / TypeScript typecheck (push) Successful in 52s
CI & Build / Python tests (push) Successful in 1m44s
CI & Build / Build & push image (push) Successful in 36s
CI & Build / Python lint (push) Successful in 3s
CI & Build / Plugin hooks (push) Successful in 13s
CI & Build / integration (push) Successful in 50s
CI & Build / TypeScript typecheck (push) Successful in 52s
CI & Build / Python tests (push) Successful in 1m44s
CI & Build / Build & push image (push) Successful in 36s
#4769 "Rulings are counted where someone will read them": milestone 444 step 4 wrote system_usage_events and nothing read it. - retrieval_telemetry gains a `system_usage` block: surfacings and opens by source, distinct counts, and `by_system` naming the areas most shown. There is deliberately no pull-through ratio, because rulings travel in full in the line and opens are the exception. - usage_for_systems (one GROUP BY) adds `usage` to the REST Systems list and detail, and to MCP get_system. MCP list_systems is unchanged. - The Systems UI shows a "rulings shown N×" chip. - rulings_pre_tool, rulings_write_path and mcp_get_system are now declared registry points; the registry guard covers their recorders. - The Systems store merges a PATCH reply instead of replacing the row. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -380,7 +380,7 @@ async def retrieval_telemetry(
|
||||
— so `near_miss_samples=5` and opening the ids it returns is the step that
|
||||
separates a real miss from a bar doing its job.
|
||||
|
||||
Three readouts, from the three tables built for them:
|
||||
Four readouts, from the four tables built for them:
|
||||
|
||||
`sources` — per retrieval surface (`auto_inject`, `write_path`,
|
||||
`mcp_search`, …), from `retrieval_logs`: `calls`, `zero_result_calls`,
|
||||
@@ -529,6 +529,16 @@ It is an UPPER BOUND per surface: a pull records the door it came
|
||||
the zeros were missing rather than absent (#3497). Measured since, it
|
||||
declines the large majority of its calls like any other surface.
|
||||
|
||||
`system_usage` — the RULINGS arm (#4769), from `system_usage_events`: how
|
||||
often an area's rulings were shown because a command (`rulings_pre_tool`)
|
||||
or an edit (`rulings_write_path`) touched its files, how often a System
|
||||
was then opened (`mcp_get_system`), and `by_system` naming the areas, most
|
||||
shown first. A lookup, not a ranker, so it has no row in `sources` and no
|
||||
floor to tune. And NO `pull_through`, on purpose: the rulings travel in
|
||||
full in the line, so a session reads them without opening anything, and
|
||||
a ratio of opens would read near zero on an arm that is working.
|
||||
`system_usage_failed: true` means its read broke and the zeros mean nothing.
|
||||
|
||||
EVERY COUNTER BLOCK CARRIES ITS OWN COVERAGE — `complete_from` and
|
||||
`covers_window`. `complete_from` is when the number became trustworthy:
|
||||
for one source, its first recorded row; for a section that sums several,
|
||||
|
||||
@@ -18,6 +18,7 @@ from scribe.services import canonical_systems as canonical_systems_svc
|
||||
from scribe.services import milestones as milestones_svc
|
||||
from scribe.services import notes as notes_svc
|
||||
from scribe.services import systems as systems_svc
|
||||
from scribe.services import system_usage as system_usage_svc
|
||||
from scribe.services.system_usage import record_system_pulled
|
||||
|
||||
# Below this, a project is young enough that the mild "which area is this
|
||||
@@ -288,7 +289,9 @@ async def get_system(system_id: int) -> dict:
|
||||
|
||||
Returns the system, plus its associated records split into `issues`,
|
||||
`tasks` (work/plan), and `notes` — each saying what the record is and where
|
||||
it sits, not what it says (open one with get_note / get_task).
|
||||
it sits, not what it says (open one with get_note / get_task) — and
|
||||
`usage`: how many times its rulings were shown to a session because its
|
||||
files were touched (`surfaced_count`), and how many times it was opened.
|
||||
"""
|
||||
uid = current_user_id()
|
||||
system = await systems_svc.get_system(uid, system_id)
|
||||
@@ -312,6 +315,10 @@ async def get_system(system_id: int) -> dict:
|
||||
data["issues"] = issues
|
||||
data["tasks"] = tasks
|
||||
data["notes"] = notes
|
||||
# How often this area's rulings were shown to a session and the System
|
||||
# opened (#4769). Here and not on list_systems: one lookup on a read that
|
||||
# asked for the System, rather than a column on every project handshake.
|
||||
data["usage"] = (await system_usage_svc.usage_for_systems([system.id]))[system.id]
|
||||
return data
|
||||
|
||||
|
||||
|
||||
@@ -10,6 +10,7 @@ from scribe.routes.utils import not_found
|
||||
from scribe.services import systems as systems_svc
|
||||
from scribe.services.access import can_write_project
|
||||
from scribe.services.projects import get_project_for_user
|
||||
from scribe.services import system_usage as system_usage_svc
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -43,10 +44,14 @@ async def list_systems_route(project_id: int):
|
||||
uid, project_id, include_archived=_truthy(request.args.get("include_archived")),
|
||||
)
|
||||
counts = await systems_svc.open_issue_counts_by_system(uid, project_id)
|
||||
# How often each area's rulings reached a session (#4769) — one GROUP BY
|
||||
# for the page, and a zero shape for an area nothing has touched.
|
||||
usage = await system_usage_svc.usage_for_systems([s.id for s in systems])
|
||||
out = []
|
||||
for s in systems:
|
||||
d = s.to_dict()
|
||||
d["open_issue_count"] = counts.get(s.id, 0)
|
||||
d["usage"] = usage.get(s.id, system_usage_svc.empty_system_usage())
|
||||
out.append(d)
|
||||
return jsonify({"systems": out})
|
||||
|
||||
@@ -114,6 +119,7 @@ async def get_system_route(project_id: int, system_id: int):
|
||||
)
|
||||
data = system.to_dict()
|
||||
data["issues"], data["tasks"], data["notes"] = issues, tasks, notes
|
||||
data["usage"] = (await system_usage_svc.usage_for_systems([system.id]))[system.id]
|
||||
return jsonify(data)
|
||||
|
||||
|
||||
|
||||
@@ -151,6 +151,23 @@ POINTS: dict[str, Point] = dict([
|
||||
"the fixed question asked when a task finishes: how should this report read",
|
||||
fixed_query=True),
|
||||
|
||||
# The rulings arm (milestone 444). A LOOKUP, not a ranker: a path falls
|
||||
# under a System's patterns or it does not, so nothing here writes to
|
||||
# retrieval_logs and no score-shaped warning can apply. Its rows are in
|
||||
# `system_usage_events`, read back as `retrieval_telemetry`'s
|
||||
# `system_usage` block (#4769).
|
||||
_p("rulings_pre_tool", UNBIDDEN,
|
||||
"an area's rulings, shown because a command touched its files",
|
||||
expects_traffic=False,
|
||||
quiet_because="speaks only once a System has both path_patterns and a "
|
||||
"Rulings section; an install where none does is "
|
||||
"correctly silent here"),
|
||||
_p("rulings_write_path", UNBIDDEN,
|
||||
"an area's rulings, shown because an edit touched its files",
|
||||
expects_traffic=False,
|
||||
quiet_because="same as rulings_pre_tool — no System with patterns "
|
||||
"and rulings, nothing to show"),
|
||||
|
||||
# ── Asked: a caller wanted a ranked list ─────────────────────────────
|
||||
_p("mcp_search", ASKED, "an agent called search"),
|
||||
_p("wide_net", ASKED, "an agent asked what might apply, with no bar"),
|
||||
@@ -186,6 +203,10 @@ POINTS: dict[str, Point] = dict([
|
||||
_p("mcp_get_process", PULL, "an agent opened a process"),
|
||||
_p("mcp_get_lesson", PULL, "an agent opened a lesson"),
|
||||
_p("mcp_get_rule", PULL, "an agent opened a rule"),
|
||||
_p("mcp_get_system", PULL, "an agent opened a System", expects_traffic=False,
|
||||
quiet_because="a System's rulings arrive in full in the rulings line, "
|
||||
"so opening one is the exception rather than the "
|
||||
"measure; a window without it is not a gap"),
|
||||
_p("rest_note", PULL, "a person opened a note", expects_traffic=False,
|
||||
quiet_because="web UI only"),
|
||||
_p("rest_task", PULL, "a person opened a task", expects_traffic=False,
|
||||
@@ -220,6 +241,9 @@ FAN_OUT_SITES: dict[str, tuple[str, ...]] = {
|
||||
"scribe/services/plugin_context.py::record_surfaced(source=arm)": (
|
||||
"write_path_place", "write_path_semantic", "write_path_sync",
|
||||
),
|
||||
"scribe/services/system_rulings.py::record_system_surfaced(source=source)": (
|
||||
"rulings_pre_tool", "rulings_write_path",
|
||||
),
|
||||
"scribe/services/rulebooks.py::record_rule_surfaced(source=source)": (
|
||||
"enter_project", "start_planning", "get_task", "get_project",
|
||||
"get_milestone",
|
||||
|
||||
@@ -34,6 +34,10 @@ from scribe.models.rule_usage import DEPARTED as RULE_DEPARTED
|
||||
from scribe.models.rule_usage import PULLED as RULE_PULLED
|
||||
from scribe.models.rule_usage import SURFACED as RULE_SURFACED
|
||||
from scribe.models.rule_usage import RuleUsageEvent
|
||||
from scribe.models.system import System
|
||||
from scribe.models.system_usage import PULLED as SYSTEM_PULLED
|
||||
from scribe.models.system_usage import SURFACED as SYSTEM_SURFACED
|
||||
from scribe.models.system_usage import SystemUsageEvent
|
||||
from scribe.services.rule_usage import is_ambient
|
||||
from scribe.models.retrieval_log import RetrievalLog
|
||||
from scribe.services.retrieval_registry import (
|
||||
@@ -817,7 +821,7 @@ def _compute_warnings(sources: dict, usage: dict, rule_usage: dict,
|
||||
|
||||
|
||||
def _silent_surfaces(sources: dict, usage: dict, rule_usage: dict,
|
||||
active: bool) -> list[dict]:
|
||||
active: bool, system_usage: dict | None = None) -> list[dict]:
|
||||
"""Registered points that emitted nothing at all in the window.
|
||||
|
||||
THE HALF THE ROWS CANNOT SEE. Every check above reads rows, so an arm that
|
||||
@@ -834,12 +838,124 @@ def _silent_surfaces(sources: dict, usage: dict, rule_usage: dict,
|
||||
return []
|
||||
seen = set(sources) | set((usage.get("by_source") or {}))
|
||||
seen |= set((rule_usage.get("by_source") or {}))
|
||||
seen |= set(((system_usage or {}).get("by_source") or {}))
|
||||
return [
|
||||
{"source": s, "kind": POINTS[s].kind, "what": POINTS[s].what}
|
||||
for s in sources_expected_to_emit() if s not in seen
|
||||
]
|
||||
|
||||
|
||||
# How many Systems `by_system` lists. A readout of which AREAS' rulings are
|
||||
# reaching sessions, not an export: an install with more ruled areas than this
|
||||
# reads the rest from the per-System counts on the Systems list.
|
||||
SYSTEM_USAGE_TOP = 20
|
||||
|
||||
|
||||
async def _system_usage(user_id: int | None, since) -> dict:
|
||||
"""The rulings arm's own block (#4769): were an area's rulings shown, and
|
||||
was the area then opened.
|
||||
|
||||
FROM `system_usage_events`, IN ITS OWN SESSION AND ITS OWN GUARD, for the
|
||||
reason the rule block is guarded apart: a failure here must not take down
|
||||
the readouts beside it, and must say so rather than print zeros (#2663) —
|
||||
the keys are kept and `system_usage_failed` is added.
|
||||
|
||||
NO PULL-THROUGH, and the omission is the design. The note and rule ratios
|
||||
ask "was the hint any use" of a line that carries a TITLE, where opening
|
||||
the record is how it gets read. A rulings line carries the rulings
|
||||
themselves; the agent reads them without opening anything, and
|
||||
`get_system` is reached for only when a long list was cut ("+N more").
|
||||
A ratio here would read near zero on an arm that is working, and invite
|
||||
the dead-weight verdict the per-record badges give — the wrong conclusion,
|
||||
built from a true number. The counts are printed side by side instead.
|
||||
|
||||
`by_system` names the areas, because "rulings were shown 40 times" says
|
||||
nothing a reader can act on, and "Billing, 38 of them" does.
|
||||
Ordered by surfacings, then by name, so a stable window reads stably.
|
||||
"""
|
||||
block: dict = {
|
||||
"surfaced": 0,
|
||||
"pulled": 0,
|
||||
"distinct_systems_surfaced": 0,
|
||||
"distinct_systems_pulled": 0,
|
||||
"by_source": {},
|
||||
"by_system": [],
|
||||
}
|
||||
try:
|
||||
async with async_session() as session:
|
||||
where = (
|
||||
SystemUsageEvent.created_at >= since,
|
||||
SystemUsageEvent.user_id == user_id,
|
||||
)
|
||||
per_source = (
|
||||
await session.execute(
|
||||
select(
|
||||
SystemUsageEvent.event,
|
||||
SystemUsageEvent.source,
|
||||
func.count().label("n"),
|
||||
)
|
||||
.where(*where)
|
||||
.group_by(SystemUsageEvent.event, SystemUsageEvent.source)
|
||||
)
|
||||
).all()
|
||||
per_system = (
|
||||
await session.execute(
|
||||
select(
|
||||
SystemUsageEvent.system_id,
|
||||
System.name,
|
||||
SystemUsageEvent.event,
|
||||
func.count().label("n"),
|
||||
func.max(SystemUsageEvent.created_at).label("last_at"),
|
||||
)
|
||||
# Outer: telemetry outlives the row it describes, and a
|
||||
# deleted System's surfacings still happened.
|
||||
.outerjoin(System, System.id == SystemUsageEvent.system_id)
|
||||
.where(*where)
|
||||
.group_by(
|
||||
SystemUsageEvent.system_id, System.name,
|
||||
SystemUsageEvent.event,
|
||||
)
|
||||
)
|
||||
).all()
|
||||
complete = await _complete_from(session, SystemUsageEvent, user_id)
|
||||
except Exception:
|
||||
logger.warning("system usage read failed", exc_info=True)
|
||||
block["system_usage_failed"] = True
|
||||
return block
|
||||
|
||||
for event, source, n in per_source:
|
||||
n = int(n)
|
||||
key = "surfaced" if event == SYSTEM_SURFACED else (
|
||||
"pulled" if event == SYSTEM_PULLED else None)
|
||||
if key is None:
|
||||
continue
|
||||
block[key] += n
|
||||
block["by_source"][source] = block["by_source"].get(source, 0) + n
|
||||
|
||||
systems: dict[int, dict] = {}
|
||||
for system_id, name, event, n, last_at in per_system:
|
||||
row = systems.setdefault(int(system_id), {
|
||||
"system_id": int(system_id),
|
||||
"name": name,
|
||||
"surfaced": 0,
|
||||
"pulled": 0,
|
||||
"last_surfaced_at": None,
|
||||
})
|
||||
if event == SYSTEM_SURFACED:
|
||||
row["surfaced"] += int(n)
|
||||
row["last_surfaced_at"] = iso(last_at)
|
||||
elif event == SYSTEM_PULLED:
|
||||
row["pulled"] += int(n)
|
||||
block["distinct_systems_surfaced"] = sum(1 for r in systems.values() if r["surfaced"])
|
||||
block["distinct_systems_pulled"] = sum(1 for r in systems.values() if r["pulled"])
|
||||
block["by_system"] = sorted(
|
||||
(r for r in systems.values() if r["surfaced"]),
|
||||
key=lambda r: (-r["surfaced"], r["name"] or ""),
|
||||
)[:SYSTEM_USAGE_TOP]
|
||||
block.update(_coverage(complete.get("*"), since))
|
||||
return block
|
||||
|
||||
|
||||
async def retrieval_summary(
|
||||
user_id: int | None, *, days: int = 30, near_miss_samples: int = 0,
|
||||
) -> dict:
|
||||
@@ -878,6 +994,7 @@ async def retrieval_summary(
|
||||
"sources": {},
|
||||
"usage": {},
|
||||
"rule_usage": {},
|
||||
"system_usage": {},
|
||||
"read_failed": False,
|
||||
}
|
||||
|
||||
@@ -1435,6 +1552,7 @@ async def retrieval_summary(
|
||||
)
|
||||
rule_usage.update(_coverage((rule_complete or {}).get("*"), since))
|
||||
out["rule_usage"] = rule_usage
|
||||
out["system_usage"] = await _system_usage(user_id, since)
|
||||
|
||||
# ── What is wrong (#3431) ────────────────────────────────────────────
|
||||
#
|
||||
@@ -1514,6 +1632,7 @@ async def retrieval_summary(
|
||||
)
|
||||
out["silent_surfaces"] = _silent_surfaces(
|
||||
out["sources"], usage, rule_usage, active,
|
||||
system_usage=out["system_usage"],
|
||||
)
|
||||
|
||||
return out
|
||||
|
||||
@@ -13,7 +13,10 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from sqlalchemy import func, select
|
||||
|
||||
from scribe.models import async_session
|
||||
from scribe.models.base import iso
|
||||
from scribe.models.system_usage import PULLED, SURFACED, SystemUsageEvent
|
||||
from scribe.services.background import report_telemetry_failure, spawn
|
||||
|
||||
@@ -65,3 +68,67 @@ def record_system_pulled(
|
||||
logger.debug("system usage payload build failed", exc_info=True)
|
||||
return
|
||||
spawn(_insert_events(rows), site="system_usage_write")
|
||||
|
||||
|
||||
def empty_system_usage() -> dict:
|
||||
"""The zero readout — a System no event has touched.
|
||||
|
||||
Rendered unconditionally, like `empty_rule_usage`, so a System predating
|
||||
the table reads as "never shown, never opened" rather than as a missing
|
||||
key. Every System in an install predates it, so for a while this is the
|
||||
normal state and must not look like a broken readout.
|
||||
|
||||
The same four keys as the client's `RecordUsage`, but read differently: a
|
||||
System's rulings are delivered IN FULL in the injected line, so a
|
||||
surfacing with no pull is the arm working, not dead weight. See
|
||||
`retrieval_telemetry`'s `system_usage` block for why no ratio is offered.
|
||||
"""
|
||||
return {
|
||||
"surfaced_count": 0,
|
||||
"pull_count": 0,
|
||||
"last_surfaced_at": None,
|
||||
"last_pulled_at": None,
|
||||
}
|
||||
|
||||
|
||||
async def usage_for_systems(system_ids: list[int]) -> dict[int, dict]:
|
||||
"""Aggregate usage for a set of Systems: {system_id: {counts + timestamps}}.
|
||||
|
||||
One GROUP BY for the whole list, never a query per row — this decorates a
|
||||
list view. Systems with no events come back as `empty_system_usage()`.
|
||||
Fails open: a readout that could break the Systems list is worse than a
|
||||
zero, but the failure is reported so the zero is not mistaken for silence
|
||||
(#2663).
|
||||
"""
|
||||
ids = [int(s) for s in system_ids]
|
||||
out: dict[int, dict] = {sid: empty_system_usage() for sid in ids}
|
||||
if not ids:
|
||||
return out
|
||||
try:
|
||||
async with async_session() as session:
|
||||
rows = (
|
||||
await session.execute(
|
||||
select(
|
||||
SystemUsageEvent.system_id,
|
||||
SystemUsageEvent.event,
|
||||
func.count().label("n"),
|
||||
func.max(SystemUsageEvent.created_at).label("last_at"),
|
||||
)
|
||||
.where(SystemUsageEvent.system_id.in_(ids))
|
||||
.group_by(SystemUsageEvent.system_id, SystemUsageEvent.event)
|
||||
)
|
||||
).all()
|
||||
except Exception:
|
||||
await report_telemetry_failure("system_usage", "readout")
|
||||
return out
|
||||
for system_id, event, n, last_at in rows:
|
||||
slot = out.get(int(system_id))
|
||||
if slot is None:
|
||||
continue
|
||||
if event == SURFACED:
|
||||
slot["surfaced_count"] = int(n)
|
||||
slot["last_surfaced_at"] = iso(last_at)
|
||||
elif event == PULLED:
|
||||
slot["pull_count"] = int(n)
|
||||
slot["last_pulled_at"] = iso(last_at)
|
||||
return out
|
||||
|
||||
Reference in New Issue
Block a user