From 3f1523b19f85a28b604e79da99ba3a83e8ffe8a6 Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Sat, 8 Aug 2026 18:19:58 -0400 Subject: [PATCH] =?UTF-8?q?feat(systems):=20read-side=20teeth=20=E2=80=94?= =?UTF-8?q?=20the=20vocabulary=20at=20session=20start,=20a=20search=20filt?= =?UTF-8?q?er,=20and=20the=20state/chronicle=20instructions?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Step 4 of #278, product half. The audit that motivated it: one System in project 2, thirty records tagged, nothing since July 28 — three days after the feature landed. Not a discipline failure; retrieval was completely blind to the association (zero references in embeddings, knowledge, search, auto-inject, or enter_project), so tagging was a write-side label with no read-side payoff, and labels nobody reads don't get maintained. Three changes, ordered by what makes the others workable: 1. enter_project returns the project's Systems (id, name, first line of the charter). Load-bearing for the tagging instruction: you cannot ask an agent to check a record against a vocabulary it never sees. Trimmed because it rides on every session start; the full charter stays get_system's job. Present-and-empty rather than absent when a project has none — "no named areas yet" is information the create-the-System instruction acts on. 2. search accepts system_id, MCP and REST (#33). Implemented once in semantic_search_notes as an EXISTS against record_systems — an association filter deciding candidate-set membership before scoring, like project_id, not a ranking signal. The REST route's missing project filter stays #2463's: it carries a default-scope UI decision this change must not preempt. 3. The instructions (#119, _INSTRUCTIONS + using-scribe skill; plugin 0.1.25 for the cache): - Tag as you write, with an executable test — "would someone investigating that subsystem want this in the pile list_system_records returns?" — rather than "tag appropriately", which is what died. - Create the System when the area has no record: the two-or-more test snippets use, plus "don't wait to be asked to name an area that plainly exists", because the agent's default was leaving un-modelled areas un-modelled forever. - State vs chronicle: dev-logs are written once and never rewritten; durable findings live in the System's reference note, updated in place — safe because note versions are the changelog, which has existed since the feature shipped and was never named as one. list_system_records' docstring now sells it as the way to READ a subsystem, reference note first. No auto-inject boost by System — vocabulary and filter first, measure before adding ranking behaviour (the #2486 lesson). Refs #278, #2546 --- plugin/.claude-plugin/plugin.json | 2 +- plugin/skills/using-scribe/SKILL.md | 19 ++++++++++ src/scribe/mcp/server.py | 27 ++++++++++++--- src/scribe/mcp/tools/projects.py | 24 ++++++++++++- src/scribe/mcp/tools/search.py | 7 ++++ src/scribe/mcp/tools/systems.py | 9 ++++- src/scribe/routes/search.py | 5 +++ src/scribe/services/embeddings.py | 14 ++++++++ tests/test_mcp_tool_projects.py | 54 +++++++++++++++++++++++++++++ 9 files changed, 154 insertions(+), 7 deletions(-) diff --git a/plugin/.claude-plugin/plugin.json b/plugin/.claude-plugin/plugin.json index 18b1602..c4d53f2 100644 --- a/plugin/.claude-plugin/plugin.json +++ b/plugin/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "scribe", "description": "Scribe system-of-record for Claude Code: MCP tools over your notes/tasks/projects/rules, a session-start push channel that surfaces your always-on rules + active-project context, process-skills (writing-plans, systematic-debugging, verification, brainstorming, reusing-code), and your saved Scribe Processes auto-surfaced as skills (/scribe:sync). Replaces superpowers + file-memory with one app-backed plugin.", - "version": "0.1.24", + "version": "0.1.25", "author": { "name": "Bryan Van Deusen" }, "mcpServers": { "scribe": { diff --git a/plugin/skills/using-scribe/SKILL.md b/plugin/skills/using-scribe/SKILL.md index 8562d58..fbad010 100644 --- a/plugin/skills/using-scribe/SKILL.md +++ b/plugin/skills/using-scribe/SKILL.md @@ -81,6 +81,25 @@ Two constraints on *how* that's achieved: (`arose_from_id`) and the subsystem it touches (`system_ids`). Don't bury a fix as a work-log line on whatever task happened to be open. +7. **Tag records to Systems.** `enter_project` lists the project's Systems — + its named subsystems/areas. When you create or meaningfully update a record, + ask which areas it is *about* and pass `system_ids`. The test: would someone + investigating that subsystem want this record in the pile + `list_system_records` returns? If the area has no System yet, create one + (`create_system`: name + a one-paragraph charter) — an area that plainly + exists deserves naming the moment two records would share it; don't wait to + be asked. A record about no particular area takes none. + +8. **State updates in place; chronicles don't.** A dev-log records what + *happened* — write it once, never rewrite it. A durable finding (how a + subsystem works, a measured number) lives in that System's **reference + note** ("«System» — reference"), which you UPDATE as facts change — safe, + because every meaningful edit is snapshotted and the version history is the + changelog. The dev-log then `[[links]]` the reference note instead of + restating state. When a new record outright *corrects* an older one (a + re-measurement, a reversed decision), pass the old id in `supersedes` so the + stale record is demoted and labelled rather than left competing. + ## Stay inside the active project's scope Once a project is in scope — you called `enter_project`, or the working repo is diff --git a/src/scribe/mcp/server.py b/src/scribe/mcp/server.py index ec9fe16..05e2f91 100644 --- a/src/scribe/mcp/server.py +++ b/src/scribe/mcp/server.py @@ -54,10 +54,29 @@ What each part is for, and when to reach for it: system as a rulebook — rules are for behaviour, and tokens kept as prose cannot be resolved, inherited, rendered to a stylesheet, or checked against code. -- System: a per-project, reusable, self-describing subsystem/area. Associate any - record (note, task, issue) with it via system_ids so research, build-work, and - fixes for the same area line up, and recurring problem-spots surface. Manage - with create_system / list_systems / get_system. +- System: a per-project, reusable, self-describing subsystem/area — the + project's vocabulary for WHERE work happens. enter_project returns the list. + TAG AS YOU WRITE: when you create or meaningfully update a note, task, or + snippet, ask which of those areas it is about and pass system_ids. The test: + would someone investigating that subsystem want this record in the pile + list_system_records returns? Cross-cutting records take several; a record + about no particular area takes none — don't force it. If the area a record + describes has no System yet, CREATE it (create_system: name + a one-paragraph + charter) and tag the record — a subsystem that exists in the code deserves a + System the moment two records would share it, the same two-or-more test + snippets use; don't wait to be asked to name an area that plainly exists. + Read a subsystem back with list_system_records, or search(system_id=...) for + a ranked cut. +- Reference note vs dev-log — STATE vs CHRONICLE. A dev-log records what + HAPPENED: write it once, never rewrite it. A durable finding — how a + subsystem works, a measured number, an architecture fact — belongs in that + System's REFERENCE NOTE ("«System name» — reference", tagged to the System), + which is UPDATED IN PLACE as the facts change. Updating loses nothing: every + meaningful edit is snapshotted (note versions are the changelog). Create the + reference note if the System lacks one; update it if it exists; have the + dev-log [[link]] it rather than restating state. State smeared across dated + logs is unreachable by search — sixteen near-identical dev-logs tie, and no + ranking can pick the right one, because no right one exists. Mechanics: - Notes and Tasks share a model; tasks are notes with is_task=True. diff --git a/src/scribe/mcp/tools/projects.py b/src/scribe/mcp/tools/projects.py index 3d26ab1..5e4ee07 100644 --- a/src/scribe/mcp/tools/projects.py +++ b/src/scribe/mcp/tools/projects.py @@ -22,6 +22,7 @@ from scribe.services import milestones as milestones_svc from scribe.services import notes as notes_svc from scribe.services import projects as projects_svc from scribe.services import rulebooks as rulebooks_svc +from scribe.services import systems as systems_svc from scribe.services import trash as trash_svc @@ -54,7 +55,14 @@ async def enter_project(project_id: int) -> dict: Returns a dict with keys: project, milestone_summary, applicable_rules, project_rules, subscribed_rulebooks, applicable_rules_truncated, - open_tasks, recent_notes, design_system. + open_tasks, recent_notes, design_system, systems. + + `systems` is the project's vocabulary of named subsystems/areas. It is + returned here so you can TAG as you write: when creating or meaningfully + updating a record, ask which of these areas it is about and pass their ids + as `system_ids`. If the area a record describes is missing from this list, + create it with create_system rather than leaving the area unmodelled. Read + a subsystem's accumulated records with list_system_records. `design_system` is null unless the project points at one. When present it carries the chain-merged guidance (the house style AND this project's @@ -81,6 +89,11 @@ async def enter_project(project_id: int) -> dict: uid, is_task=False, project_id=project_id, sort="updated_at", limit=5, ) + # The tagging vocabulary. Surfaced HERE because an instruction to "tag + # records to Systems" is only executable if the list is in front of the + # agent when it writes — which it never was, and tagging stopped within + # three days of the feature landing (#2546's audit). + systems = await systems_svc.list_systems(uid, project_id) # A project need not have one, and most installs won't — null is ordinary # here, not a missing prerequisite. design_system = None @@ -91,6 +104,15 @@ async def enter_project(project_id: int) -> dict: return { "project": project.to_dict(), + # Trimmed to what tagging needs. The full charter is get_system's job — + # this list rides along on every session start, so it stays lean. + "systems": [ + { + "id": s.id, "name": s.name, + "description": (s.description or "").split("\n")[0][:200], + } + for s in systems + ], "design_system": design_system, "milestone_summary": milestone_summary, "applicable_rules": applicable["rules"], diff --git a/src/scribe/mcp/tools/search.py b/src/scribe/mcp/tools/search.py index 0c317a4..dbe2f28 100644 --- a/src/scribe/mcp/tools/search.py +++ b/src/scribe/mcp/tools/search.py @@ -20,6 +20,7 @@ async def search( content_type: str = "all", limit: int = 10, project_id: int = 0, + system_id: int = 0, ) -> dict: """Semantic search over the user's existing notes and tasks — Scribe's recall. @@ -39,6 +40,11 @@ async def search( enter_project) — otherwise this searches across ALL projects and bleeds unrelated work into the result set. 0 = search everything (use only when you genuinely want a cross-project sweep). + system_id: Narrow to records tagged to one System (a named + subsystem/area — enter_project lists them). Use when investigating + a specific subsystem: it cuts the candidates to records someone + deliberately filed under that area. 0 = no system filter. + list_system_records gives the same slice unranked. Returns: {"results": [{"id", "title", "body", "is_task", "tags", "similarity"}], @@ -55,6 +61,7 @@ async def search( raw = await semantic_search_notes( uid, q, limit=limit, is_task=is_task, project_id=project_id or None, + system_id=system_id or None, # An explicit search reaches everything the operator may read, including # records shared with them one-to-one. scope="read", diff --git a/src/scribe/mcp/tools/systems.py b/src/scribe/mcp/tools/systems.py index e518acf..af1cdd8 100644 --- a/src/scribe/mcp/tools/systems.py +++ b/src/scribe/mcp/tools/systems.py @@ -116,7 +116,14 @@ async def update_system( async def list_system_records( system_id: int, kind: str = "", open_only: bool = False ) -> dict: - """List records associated with a System. + """Everything filed under one System — the way to READ a subsystem. + + Reach for this when investigating a specific area: it returns the notes, + tasks, issues and snippets someone deliberately tagged to it — the + subsystem's accumulated record, unranked. Start with its reference note if + one exists (titled "«System» — reference"); that is the living state, and + the rest is history and open work around it. For a ranked cut of the same + slice, search(system_id=...) filters semantic search to this association. Args: kind: filter by task_kind — 'issue', 'work', or 'plan'. Omit for all. diff --git a/src/scribe/routes/search.py b/src/scribe/routes/search.py index 8d56583..bdfe6b5 100644 --- a/src/scribe/routes/search.py +++ b/src/scribe/routes/search.py @@ -34,10 +34,15 @@ async def search_route(): content_type = request.args.get("content_type", "all") limit = min(request.args.get("limit", 10, type=int), 50) is_task = _content_type_to_is_task(content_type) + # Same association filter the MCP tool takes (#33). The project filter this + # route is still missing is #2463's — it carries a default-scope UI decision + # this change must not preempt. + system_id = request.args.get("system_id", type=int) t0 = time.perf_counter() results = await semantic_search_notes( uid, q, limit=limit, is_task=is_task, threshold=_REST_SEARCH_THRESHOLD, + system_id=system_id, # The user typed this, so it reaches everything they may read. scope="read", ) diff --git a/src/scribe/services/embeddings.py b/src/scribe/services/embeddings.py index de20b09..13b576a 100644 --- a/src/scribe/services/embeddings.py +++ b/src/scribe/services/embeddings.py @@ -209,6 +209,7 @@ async def semantic_search_notes( orphan_only: bool = False, scope: str = "own", demote_superseded: bool = True, + system_id: int | None = None, ) -> list[tuple[float, Note]]: """Return up to *limit* (score, note) pairs most relevant to *query*. @@ -281,6 +282,19 @@ async def semantic_search_notes( stmt = stmt.where(Note.project_id.is_(None)) elif project_id is not None: stmt = stmt.where(Note.project_id == project_id) + # Narrow to records tagged to one System (subsystem/area). An + # association filter, not a ranking signal — membership in the + # candidate set, decided before scoring, like project_id above. + if system_id is not None: + from scribe.models.system import RecordSystem + stmt = stmt.where( + select(RecordSystem.id) + .where( + RecordSystem.note_id == Note.id, + RecordSystem.system_id == system_id, + ) + .exists() + ) if is_task is True: stmt = stmt.where(Note.status.isnot(None)) elif is_task is False: diff --git a/tests/test_mcp_tool_projects.py b/tests/test_mcp_tool_projects.py index 3b52c4e..f6e545c 100644 --- a/tests/test_mcp_tool_projects.py +++ b/tests/test_mcp_tool_projects.py @@ -17,6 +17,18 @@ def _bind_user(): _user_id_ctx.reset(token) +@pytest.fixture(autouse=True) +def _no_systems(): + """enter_project now surfaces the project's Systems as the tagging + vocabulary (#2546). These are tool-layer unit tests with no database, so + the lookup is stubbed to the common case — a project with none. The + populated shape is asserted in its own test below. + """ + with patch("scribe.mcp.tools.projects.systems_svc.list_systems", + AsyncMock(return_value=[])): + yield + + def _fake_project(design_system_id=None, **overrides) -> MagicMock: p = MagicMock() base = {"id": 1, "title": "P", "description": "", "goal": "", @@ -184,6 +196,48 @@ async def test_enter_project_composes_full_context(): # absent. A caller that has to distinguish "no key" from "no system" will # eventually get it wrong. assert out["design_system"] is None + # No Systems -> present-and-empty, NOT absent: this key is the tagging + # vocabulary, and "this project has no named areas yet" is information the + # create-the-System instruction acts on. + assert out["systems"] == [] + + +@pytest.mark.asyncio +async def test_enter_project_surfaces_the_systems_vocabulary(): + """The tagging instruction is only executable if the vocabulary is in + front of the agent when it writes. It never was, and tagging stopped three + days after the feature landed — one System, nothing tagged since July 28 + (#2546's audit). Trimmed to id/name/first-line: it rides on every session + start, and the full charter is get_system's job.""" + p = _fake_project(id=5) + sys1 = MagicMock() + sys1.id = 3 + sys1.name = "retrieval" + sys1.description = "Embeddings, ranking, auto-inject.\nLong detail below." + + with patch( + "scribe.mcp.tools.projects.projects_svc.get_project", + AsyncMock(return_value=p), + ), patch( + "scribe.mcp.tools.projects.rulebooks_svc.get_applicable_rules", + AsyncMock(return_value={"rules": [], "truncated": False, + "subscribed_rulebooks": []}), + ), patch( + "scribe.mcp.tools.projects.milestones_svc.get_project_milestone_summary", + AsyncMock(return_value=[]), + ), patch( + "scribe.mcp.tools.projects.notes_svc.list_notes", + AsyncMock(side_effect=[([], 0), ([], 0)]), + ), patch( + "scribe.mcp.tools.projects.systems_svc.list_systems", + AsyncMock(return_value=[sys1]), + ): + out = await enter_project(project_id=5) + + assert out["systems"] == [ + {"id": 3, "name": "retrieval", + "description": "Embeddings, ranking, auto-inject."} + ] @pytest.mark.asyncio