feat(mcp): enter_project becomes a small primer: goal, recent work, open work, vocabulary (#4045)
CI & Build / Python lint (push) Successful in 3s
CI & Build / Plugin hooks (push) Successful in 9s
CI & Build / integration (push) Successful in 46s
CI & Build / TypeScript typecheck (push) Successful in 53s
CI & Build / Python tests (push) Successful in 1m44s
CI & Build / Build & push image (push) Successful in 23s

The handshake carried the whole project record, every milestone's plan, full
rule text, the notes most recently edited and ~9k of design guidance. For
project 2 that was ~222k characters, past what an MCP client accepts as a tool
result. Each category was walked through with the operator and sized to what a
session needs on arrival; each names the call that has the rest.

- project: id, title, status and the full goal (session start's "full goal"
  pointer still lands here). get_project keeps the whole record.
- milestone_summary: the 5 most recently touched milestones, any status, most
  recent first, without plans. Summaries gain last_touched_at: the later of
  the milestone's own edit and its newest step update, from the query that
  already counts steps. milestone_summary_omitted counts the rest and points
  to list_milestones. get_project and list_milestones list every milestone,
  also without plans.
- open_tasks: the 10 most recently touched open tasks, with or without a
  milestone, each naming its milestone. list_notes gains sort="touched"
  (the later of updated_at and the newest work-log), because a log doesn't
  bump updated_at.
- recent_notes: dropped. Retrieval surfaces notes by relevance, and
  get_recent covers recency.
- systems: id and name.
- design_system: summary plus guidance_call. get_design_system gains
  resolved_guidance, the chain-merged prose; its own guidance field is only
  the departures, so session start's old pointer to it led to a fragment.
  The session start pointer and using-scribe's "Building UI" section now
  name resolved_guidance.
- rules: rules_payload(brief=True) gives project_rules as id and title plus
  subscribed_rulebooks, and records only what it shows. Retrieval delivers
  rules in full and ignores subscriptions (#4052). Other callers unchanged.
- pattern_coverage, inception and systems_bootstrap: unchanged.

Clients: the plugin's using-scribe skill, the compaction notice and session
start are updated here; the REST project summary only gains last_touched_at.
Plugin version minted.

Tests: a size ceiling on the handshake for a large project; milestone and
task selection and naming; brief rules; resolved_guidance; the session
start pointer; and a real-Postgres test that a work-log touches its task and
a step update touches its milestone.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01821k5B3Ysecp9fNYs92Kuy
This commit is contained in:
2026-09-14 22:04:38 -04:00
co-authored by Claude Opus 5
parent 9b2de3552f
commit 7f974d9749
17 changed files with 445 additions and 182 deletions
+114 -57
View File
@@ -1,12 +1,13 @@
"""Milestone listings stay brief and bounded (#4045).
"""The enter_project handshake stays a small primer (#4045).
enter_project once carried every milestone's full plan body. On a project with
39 milestones that came to ~222k characters, past what an MCP client accepts
as a tool result, so the session handshake arrived as a file to page through.
enter_project once carried every milestone's full plan body, the whole project
record, full rule text and ~9k of design guidance. On a project with 39
milestones that came to ~222k characters, past what an MCP client accepts as
a tool result, so the session handshake arrived as a file to page through.
"""
import contextlib
import json
from unittest.mock import AsyncMock, patch
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
@@ -19,28 +20,32 @@ from tests.helpers import fake_project
pytestmark = pytest.mark.usefixtures("_bind_user")
PLAN = "A plan paragraph long enough to matter. " * 125 # ~5k chars
GOAL = "What the project is for. " * 50 # ~1.2k chars
def _row(mid: int, status: str, updated: str) -> dict:
def _milestone(mid: int, status: str, touched_day: int) -> dict:
"""A summary row as get_project_milestone_summary returns it."""
touched = f"2026-08-{touched_day:02d}T00:00:00+00:00"
return {
"id": mid, "user_id": 7, "project_id": 5, "title": f"M{mid}",
"description": f"what M{mid} is for", "body": PLAN, "status": status,
"order_index": mid, "created_at": "2026-01-01T00:00:00+00:00",
"updated_at": updated, "total": 4, "completed": 2, "pct": 50.0,
"updated_at": "2026-01-01T00:00:00+00:00", "last_touched_at": touched,
"total": 4, "completed": 2, "pct": 50.0,
"status_counts": {"todo": 2, "in_progress": 0, "done": 2, "cancelled": 0},
}
def _history(done: int, active: int) -> list[dict]:
"""`done` done milestones (higher id = more recently updated), then `active` open ones."""
rows = [_row(i, "done", f"2026-08-{i + 1:02d}T00:00:00+00:00") for i in range(done)]
rows += [_row(done + i, "active", "2026-01-01T00:00:00+00:00") for i in range(active)]
return rows
def _history(count: int) -> list[dict]:
"""`count` milestones, alternating done/active; higher id = touched later."""
return [
_milestone(i, "done" if i % 2 else "active", 1 + i % 28)
for i in range(count)
]
def test_brief_rows_leave_out_the_plan_and_what_the_caller_already_knows():
brief, omitted = brief_milestone_summary([_row(1, "active", "2026-09-01")])
brief, omitted = brief_milestone_summary([_milestone(1, "active", 3)])
assert omitted == 0
assert brief == [{
"id": 1, "title": "M1", "description": "what M1 is for", "status": "active",
@@ -49,93 +54,145 @@ def test_brief_rows_leave_out_the_plan_and_what_the_caller_already_knows():
}]
def test_cap_keeps_every_open_milestone_and_the_most_recent_done_ones_in_order():
rows = _history(done=8, active=3)
brief, omitted = brief_milestone_summary(rows, done_kept=2)
assert omitted == 6
# The two most recently updated done ones (ids 6, 7) and all open ones,
# still in order_index order.
assert [r["id"] for r in brief] == [6, 7, 8, 9, 10]
def test_limit_keeps_the_most_recently_touched_whatever_their_status():
"""A plan can sit "active" for months; recency is what says it's current."""
rows = [_milestone(1, "active", 1), _milestone(2, "done", 9),
_milestone(3, "active", 5), _milestone(4, "done", 7)]
brief, omitted = brief_milestone_summary(rows, limit=2)
assert omitted == 2
assert [r["id"] for r in brief] == [2, 4] # most recent first
def test_no_cap_keeps_every_row():
brief, omitted = brief_milestone_summary(_history(done=8, active=3))
def test_no_limit_keeps_every_row_in_order():
brief, omitted = brief_milestone_summary(_history(8))
assert omitted == 0
assert len(brief) == 11
assert [r["id"] for r in brief] == list(range(8))
def _enter_stubs(rows: list[dict]):
def _task(tid: int, milestone_id: int | None) -> MagicMock:
t = MagicMock()
t.id = tid
t.title = f"T{tid}"
t.status = "todo"
t.milestone_id = milestone_id
return t
def _enter_stubs(project, milestones: list[dict], tasks: list, *, rules=None, systems=None,
design=None):
applicable = rules or {"rules": [], "project_rules": [], "truncated": False,
"subscribed_rulebooks": []}
return [
patch("scribe.mcp.tools.projects.projects_svc.get_project",
AsyncMock(return_value=fake_project(id=5))),
AsyncMock(return_value=project)),
patch("scribe.mcp.tools.projects.rulebooks_svc.get_applicable_rules",
AsyncMock(return_value={"rules": [], "truncated": False,
"subscribed_rulebooks": []})),
AsyncMock(return_value=applicable)),
patch("scribe.mcp.tools.projects.milestones_svc.get_project_milestone_summary",
AsyncMock(return_value=rows)),
AsyncMock(return_value=milestones)),
patch("scribe.mcp.tools.projects.notes_svc.list_notes",
AsyncMock(side_effect=[([], 0), ([], 0)])),
AsyncMock(return_value=(tasks, len(tasks)))),
patch("scribe.mcp.tools.projects.systems_svc.list_systems",
AsyncMock(return_value=[])),
AsyncMock(return_value=systems or [])),
patch("scribe.mcp.tools.projects.systems_tools.bootstrap_systems_ask",
AsyncMock(return_value=None)),
patch("scribe.mcp.tools.projects.design_systems_svc.design_context",
AsyncMock(return_value=design)),
patch("scribe.mcp.tools.projects.coverage_svc.cached_coverage",
AsyncMock(return_value=None)),
patch("scribe.mcp.tools.projects.spawn"),
]
@pytest.mark.asyncio
async def test_enter_project_stays_small_however_long_the_history():
"""The ceiling is the point: a long-lived project's handshake must not
grow with its history. 200 milestones with 5k-character plans would be
~1M characters if bodies rode along."""
async def _enter(*stubs):
with contextlib.ExitStack() as stack:
for cm in _enter_stubs(_history(done=190, active=10)):
stack.enter_context(cm)
mocks = [stack.enter_context(cm) for cm in stubs]
out = await enter_project(project_id=5)
return out, mocks
summary = out["milestone_summary"]
assert all("body" not in m for m in summary)
assert sum(m["status"] == "done" for m in summary) == 5
assert sum(m["status"] == "active" for m in summary) == 10
assert out["milestone_summary_omitted"].startswith("185 older done milestone(s)")
assert "list_milestones(5)" in out["milestone_summary_omitted"]
assert len(json.dumps(out["milestone_summary"], indent=2)) < 10_000
@pytest.mark.asyncio
async def test_enter_project_stays_small_however_large_the_project():
"""The ceiling is the point: a long-lived project's handshake must not
grow with its history. 200 milestones with 5k-character plans, 60 project
rules and 40 Systems would be well over 1M characters in the old shape."""
project = fake_project(id=5, design_system_id=9, goal=GOAL,
description="Background. " * 300)
rules = {
"rules": [{"id": i, "title": f"r{i}", "statement": PLAN} for i in range(50)],
"project_rules": [{"id": 100 + i, "title": f"pr{i}", "statement": PLAN,
"when_to_apply": PLAN} for i in range(60)],
"truncated": True, "subscribed_rulebooks": [{"id": 1, "title": "Family"}],
"suppressed_rules": [], "suppressed_topics": [],
}
systems = []
for i in range(40):
s = MagicMock()
s.id, s.name, s.description = i, f"Area {i}", PLAN
systems.append(s)
design = {"id": 9, "title": "Kit", "description": "", "inherits_from": ["House"],
"guidance": [{"design_system_id": 9, "title": "Kit", "guidance": PLAN * 2}],
"token_count": 111, "token_groups": ["accent", "surface"]}
tasks = [_task(1000 + i, i % 200) for i in range(10)]
out, _ = await _enter(*_enter_stubs(project, _history(200), tasks, rules=rules,
systems=systems, design=design))
assert len(out["milestone_summary"]) == 5
assert out["milestone_summary_omitted"].startswith("195 other milestone(s)")
assert len(out["open_tasks"]) == 10
assert "applicable_rules" not in out and "recent_notes" not in out
assert "guidance" not in out["design_system"]
# The fixed parts are bounded by their caps; what's left to grow is the
# goal, the rule and System titles, and the ask keys when they apply.
size = len(json.dumps(out, indent=2))
assert size < 16_000, size
@pytest.mark.asyncio
async def test_open_tasks_name_their_milestone_even_when_it_is_not_listed():
"""The milestone list is capped at 5; a task's milestone can fall outside
it, and its id must not arrive without a name."""
milestones = _history(20)
tasks = [_task(1, 0), _task(2, None)] # milestone 0 is the least recent
out, mocks = await _enter(*_enter_stubs(fake_project(id=5), milestones, tasks))
assert 0 not in [m["id"] for m in out["milestone_summary"]]
assert out["open_tasks"] == [
{"id": 1, "title": "T1", "status": "todo", "milestone_id": 0, "milestone_title": "M0"},
{"id": 2, "title": "T2", "status": "todo", "milestone_id": None, "milestone_title": None},
]
list_notes = mocks[3]
assert list_notes.await_args.kwargs["sort"] == "touched"
assert list_notes.await_args.kwargs["limit"] == 10
@pytest.mark.asyncio
async def test_omitted_key_is_absent_when_nothing_was_left_out():
"""Attached only when it applies (#2483)."""
with contextlib.ExitStack() as stack:
for cm in _enter_stubs(_history(done=3, active=2)):
stack.enter_context(cm)
out = await enter_project(project_id=5)
assert len(out["milestone_summary"]) == 5
out, _ = await _enter(*_enter_stubs(fake_project(id=5), _history(3), []))
assert len(out["milestone_summary"]) == 3
assert "milestone_summary_omitted" not in out
@pytest.mark.asyncio
async def test_get_project_uses_the_same_brief_block():
async def test_get_project_lists_every_milestone_without_plans():
with patch("scribe.mcp.tools.projects.projects_svc.get_project",
AsyncMock(return_value=fake_project(id=5))), \
patch("scribe.mcp.tools.projects.milestones_svc.get_project_milestone_summary",
AsyncMock(return_value=_history(done=9, active=1))), \
AsyncMock(return_value=_history(10))), \
patch("scribe.mcp.tools.projects.rulebooks_svc.get_applicable_rules",
AsyncMock(return_value={"rules": [], "truncated": False,
"subscribed_rulebooks": []})):
out = await get_project(project_id=5)
assert len(out["milestone_summary"]) == 6
assert len(out["milestone_summary"]) == 10
assert all("body" not in m for m in out["milestone_summary"])
assert out["milestone_summary_omitted"].startswith("4 older done milestone(s)")
@pytest.mark.asyncio
async def test_list_milestones_lists_every_milestone_without_plans():
"""The call milestone_summary_omitted points to: every milestone, done
ones included, and no bodies (get_milestone has the plan)."""
"""The call milestone_summary_omitted points to."""
with patch("scribe.mcp.tools.milestones.milestones_svc.get_project_milestone_summary",
AsyncMock(return_value=_history(done=30, active=2))):
AsyncMock(return_value=_history(30))):
out = await list_milestones(project_id=5)
assert len(out["milestones"]) == 32
assert len(out["milestones"]) == 30
assert all("body" not in m for m in out["milestones"])