feat(planning): start_planning hands back the active plan that already covers the work (#4079)
CI & Build / Python lint (push) Successful in 5s
CI & Build / Plugin hooks (push) Successful in 13s
CI & Build / integration (push) Successful in 1m10s
CI & Build / TypeScript typecheck (push) Successful in 1m15s
CI & Build / Python tests (push) Failing after 1m22s
CI & Build / Build & push image (push) Skipped

Step 4 of milestone 415 "An existing plan is found before a new one is made".
A session that could not see an existing plan made a second one beside it.
start_planning and create_milestone now ask first: an ACTIVE milestone in the
project with the same title, or one that reads as the same plan (title, design
and steps against milestone embeddings), is returned with its progress and a
pointer to create_records(milestone_id=...). Nothing is created; force=true
bypasses.

- dedup.find_matching_plan / plan_gate / plan_match_response; access-checked
  before either arm (rule 78), fail-open like the other gates.
- Done milestones never block; the semantic arm needs 200+ chars of candidate.
- kb_plan_match_threshold (default 0.90) is a setting, in the Settings view,
  and pinned against the Python default by test_settings_defaults_agree.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01821k5B3Ysecp9fNYs92Kuy
This commit is contained in:
2026-09-15 13:43:18 -04:00
co-authored by Claude Opus 5
parent 3a501c2cac
commit 59407728e6
11 changed files with 468 additions and 22 deletions
+4
View File
@@ -45,6 +45,10 @@ Tasks carry status (`todo` → `in_progress` → `done`/`cancelled`), priority
- **Milestones** — Ordered stages within a project. A milestone is also the home of a - **Milestones** — Ordered stages within a project. A milestone is also the home of a
**plan** — its body holds the design (Goal/Approach/Verification) and its child **plan** — its body holds the design (Goal/Approach/Verification) and its child
tasks are the steps. Completion percentage is shown on the project page. tasks are the steps. Completion percentage is shown on the project page.
Milestones are searchable by meaning (`search(content_type="milestone")`), and an
agent starting a plan (`start_planning`, `create_milestone`) is handed the active
milestone that already has its title or reads as the same plan, so steps are added
there rather than to a parallel plan. The match threshold is in Settings.
- **Kanban view** — `/projects/:id` groups tasks by milestone in a column layout with - **Kanban view** — `/projects/:id` groups tasks by milestone in a column layout with
status-advance buttons on the cards. status-advance buttons on the cards.
+31
View File
@@ -104,6 +104,8 @@ const kbPromptRuleThreshold = ref("0.72");
const kbDupThresholdSnippet = ref("0.82"); const kbDupThresholdSnippet = ref("0.82");
const kbDupThresholdNote = ref("0.93"); const kbDupThresholdNote = ref("0.93");
const kbDupThresholdTask = ref("0.93"); const kbDupThresholdTask = ref("0.93");
// PLAN_MATCH_DEFAULT_THRESHOLD in services/dedup.py.
const kbPlanMatchThreshold = ref("0.90");
const savingKbInject = ref(false); const savingKbInject = ref(false);
const kbInjectSaved = ref(false); const kbInjectSaved = ref(false);
@@ -155,6 +157,9 @@ async function saveKbInject() {
const dupSnip = Math.min(1, Math.max(0, Number(kbDupThresholdSnippet.value) || 0.82)); const dupSnip = Math.min(1, Math.max(0, Number(kbDupThresholdSnippet.value) || 0.82));
const dupNote = Math.min(1, Math.max(0, Number(kbDupThresholdNote.value) || 0.93)); const dupNote = Math.min(1, Math.max(0, Number(kbDupThresholdNote.value) || 0.93));
const dupTask = Math.min(1, Math.max(0, Number(kbDupThresholdTask.value) || 0.93)); const dupTask = Math.min(1, Math.max(0, Number(kbDupThresholdTask.value) || 0.93));
// Same `|| default` guard: a floor of 0 would hand back an existing plan
// for every new one, and no plan could be started without force.
const planT = Math.min(1, Math.max(0, Number(kbPlanMatchThreshold.value) || 0.9));
// Same `|| default` reasoning: falling back to 0 would surface every // Same `|| default` reasoning: falling back to 0 would surface every
// snippet in the corpus on every edit, which is the failure this knob fixes. // snippet in the corpus on every edit, which is the failure this knob fixes.
const wpT = Math.min(1, Math.max(0, Number(kbWritePathThreshold.value) || 0.68)); const wpT = Math.min(1, Math.max(0, Number(kbWritePathThreshold.value) || 0.68));
@@ -171,6 +176,7 @@ async function saveKbInject() {
kbDupThresholdSnippet.value = String(dupSnip); kbDupThresholdSnippet.value = String(dupSnip);
kbDupThresholdNote.value = String(dupNote); kbDupThresholdNote.value = String(dupNote);
kbDupThresholdTask.value = String(dupTask); kbDupThresholdTask.value = String(dupTask);
kbPlanMatchThreshold.value = String(planT);
kbWritePathThreshold.value = String(wpT); kbWritePathThreshold.value = String(wpT);
kbRuleHintThreshold.value = String(rhT); kbRuleHintThreshold.value = String(rhT);
kbToolRuleThreshold.value = String(trT); kbToolRuleThreshold.value = String(trT);
@@ -199,6 +205,7 @@ async function saveKbInject() {
kb_duplicate_threshold_snippet: String(dupSnip), kb_duplicate_threshold_snippet: String(dupSnip),
kb_duplicate_threshold_note: String(dupNote), kb_duplicate_threshold_note: String(dupNote),
kb_duplicate_threshold_task: String(dupTask), kb_duplicate_threshold_task: String(dupTask),
kb_plan_match_threshold: String(planT),
}); });
kbInjectSaved.value = true; kbInjectSaved.value = true;
setTimeout(() => (kbInjectSaved.value = false), 2000); setTimeout(() => (kbInjectSaved.value = false), 2000);
@@ -662,6 +669,9 @@ onMounted(async () => {
if (allSettings.kb_duplicate_threshold_task !== undefined) { if (allSettings.kb_duplicate_threshold_task !== undefined) {
kbDupThresholdTask.value = allSettings.kb_duplicate_threshold_task; kbDupThresholdTask.value = allSettings.kb_duplicate_threshold_task;
} }
if (allSettings.kb_plan_match_threshold !== undefined) {
kbPlanMatchThreshold.value = allSettings.kb_plan_match_threshold;
}
if (allSettings.notify_task_reminders !== undefined) { if (allSettings.notify_task_reminders !== undefined) {
notifyTaskReminders.value = allSettings.notify_task_reminders !== "false"; notifyTaskReminders.value = allSettings.notify_task_reminders !== "false";
} }
@@ -1623,6 +1633,27 @@ async function deleteUser(userId: number) {
stays strict to keep the report pointed at work opened twice. stays strict to keep the report pointed at work opened twice.
</p> </p>
</div> </div>
<div class="field">
<label for="kb-plan-match-threshold">Existing-plan match threshold</label>
<input
id="kb-plan-match-threshold"
v-model="kbPlanMatchThreshold"
type="number"
min="0"
max="1"
step="0.01"
class="fs-input input"
style="max-width: 8rem"
/>
<p class="field-hint">
How alike a new plan must be to an active milestone in the same project
before an agent is handed that milestone instead of creating a second
one. A plan with the same title always matches. Lower it if sessions
still open parallel plans for work that already has one; raise it if
they are sent to plans that are only related.
</p>
</div>
<div class="actions"> <div class="actions">
<button class="btn-primary" @click="saveKbInject" :disabled="savingKbInject"> <button class="btn-primary" @click="saveKbInject" :disabled="savingKbInject">
{{ savingKbInject ? 'Saving' : 'Save' }} {{ savingKbInject ? 'Saving' : 'Save' }}
+14
View File
@@ -12,6 +12,7 @@ Sentinels:
from __future__ import annotations from __future__ import annotations
from scribe.mcp._context import current_user_id from scribe.mcp._context import current_user_id
from scribe.services import dedup as dedup_svc
from scribe.services import milestones as milestones_svc from scribe.services import milestones as milestones_svc
from scribe.services import notes as notes_svc from scribe.services import notes as notes_svc
from scribe.services import rulebooks as rulebooks_svc from scribe.services import rulebooks as rulebooks_svc
@@ -71,6 +72,7 @@ async def create_milestone(
description: str = "", description: str = "",
body: str = "", body: str = "",
status: str = "active", status: str = "active",
force: bool = False,
) -> dict: ) -> dict:
"""Create a milestone within a Scribe project. """Create a milestone within a Scribe project.
@@ -85,9 +87,21 @@ async def create_milestone(
description: Optional one-line summary of what this milestone covers. description: Optional one-line summary of what this milestone covers.
body: Optional plan/design (markdown) — the milestone's full plan text. body: Optional plan/design (markdown) — the milestone's full plan text.
status: active (default) or done. status: active (default) or done.
force: Bypass the plan gate. By default an active milestone that
already has this title, or reads as the same plan, is returned
(`existing_milestone`) and nothing is created.
""" """
uid = current_user_id() uid = current_user_id()
await refuse_guessed_ids(title, description, body) await refuse_guessed_ids(title, description, body)
# A done milestone is a record of past work, not a competing plan, so
# only an active create is gated — the same line find_matching_plan draws.
if status == "active" and not force:
match = await dedup_svc.plan_gate(
uid, project_id, title,
dedup_svc.plan_candidate_text(description=description, body=body),
)
if match is not None:
return match
milestone = await milestones_svc.create_milestone( milestone = await milestones_svc.create_milestone(
uid, uid,
project_id=project_id, project_id=project_id,
+20 -4
View File
@@ -531,17 +531,33 @@ async def start_planning(
steps: The plan's step-tasks, in order — each an object with `title` steps: The plan's step-tasks, in order — each an object with `title`
(required) and optionally `body`, `status`, `priority`, `kind` (required) and optionally `body`, `status`, `priority`, `kind`
('work' | 'issue' | 'spike'), `tags`, `system_ids`. ('work' | 'issue' | 'spike'), `tags`, `system_ids`.
force: Bypass the near-duplicate gate on the steps. By default a step force: Bypass both duplicate gates. By default nothing is created —
that near-duplicates an existing task blocks the whole plan, and milestone included — when an ACTIVE plan in the project already
nothing — milestone included — is created. has this title or reads as the same plan, or when a step
near-duplicates an existing task. Pass it once you have read the
match and know this is separate work.
Returns the milestone, the project's applicable rules and brief context, Returns the milestone, the project's applicable rules and brief context,
plus `steps` (the created tasks, in order) when steps were given — OR a plus `steps` (the created tasks, in order) when steps were given — OR a
duplicate payload naming the `record` that matched, with nothing created. duplicate payload with nothing created: `existing_milestone` when a plan
already covers this (add your steps to it with create_records(
milestone_id=...)), or `record` naming the step that matched a task.
""" """
uid = current_user_id() uid = current_user_id()
items = _batch_items(steps or [], what="step") items = _batch_items(steps or [], what="step")
await refuse_guessed_ids(body, *[t for item in items for t in (item.title, item.body)]) await refuse_guessed_ids(body, *[t for item in items for t in (item.title, item.body)])
if not force:
# The plan before its steps: when a plan already covers this, the
# answer is to add these steps THERE, and a step-level match would
# only name one symptom of that.
match = await dedup_svc.plan_gate(
uid, project_id, title,
dedup_svc.plan_candidate_text(
body=body, step_texts=[f"{i.title}\n{i.body or ''}" for i in items],
),
)
if match is not None:
return match
if items and not force: if items and not force:
dup = await _first_duplicate(uid, items, project_id or None) dup = await _first_duplicate(uid, items, project_id or None)
if dup is not None: if dup is not None:
+173
View File
@@ -33,6 +33,7 @@ from scribe.models.embedding import NoteEmbedding
from scribe.models.note import Note from scribe.models.note import Note
from scribe.models.rulebook import Rule from scribe.models.rulebook import Rule
from scribe.models.base import iso from scribe.models.base import iso
from scribe.services.access import can_read_project
from scribe.services import embeddings as embeddings_svc from scribe.services import embeddings as embeddings_svc
# Imported rather than redeclared: no service imports this module (the create # Imported rather than redeclared: no service imports this module (the create
# gate is called from the routes/tools layer), so there is no cycle to dodge, # gate is called from the routes/tools layer), so there is no cycle to dodge,
@@ -720,3 +721,175 @@ async def find_duplicate_rule(
except Exception: except Exception:
logger.debug("dedup rule title check skipped — query failed", exc_info=True) logger.debug("dedup rule title check skipped — query failed", exc_info=True)
return None return None
# --- the plan gate (milestone 415) -------------------------------------------
# A session asked "what work is open?" that cannot see an existing plan makes a
# second one: a new milestone beside the one that already covers the work, or
# loose tasks beside it. Each copy then collects its own steps, and neither
# shows the whole. This gate asks the question before start_planning (or
# create_milestone) writes: is there an ACTIVE plan in this project for this?
#
# Active only: a done milestone is history, and planning the next round of
# the same area is legitimate work rather than a copy of it.
#
# Project-scoped, not owner-scoped like the note gate. The note gate refuses to
# point at someone else's record because they may not be able to edit it; a
# plan is different. Creating one needs write on the project, and write on the
# project is exactly what adding steps to its existing plan needs, so a caller
# who reaches this gate can act on whatever it returns.
#
# Its own threshold, as a setting (rule 25). The note gate's 0.90 is where it
# starts, because it asks the same question ("the same thing, reworded") with
# the same embedder. Plan documents are shaped differently, though: a milestone
# is embedded as title, description and plan (embeddings.milestone_document),
# while the candidate usually has no description and carries its steps
# instead. That difference has not been measured, and a gate that blocks on
# noise teaches sessions to pass force=true every time, which is worse than no
# gate. So the default is conservative and the operator can lower it.
PLAN_MATCH_THRESHOLD_KEY = "kb_plan_match_threshold"
PLAN_MATCH_DEFAULT_THRESHOLD = 0.90
async def get_plan_match_threshold(user_id: int) -> float:
"""The user's plan-gate similarity floor, clamped to [0, 1]."""
from scribe.services.settings import get_setting
try:
value = float(await get_setting(
user_id, PLAN_MATCH_THRESHOLD_KEY, str(PLAN_MATCH_DEFAULT_THRESHOLD)
))
except (TypeError, ValueError):
value = PLAN_MATCH_DEFAULT_THRESHOLD
return min(1.0, max(0.0, value))
def plan_candidate_text(
description: str | None = None,
body: str | None = None,
step_texts: list[str] | None = None,
) -> str:
"""What a plan that doesn't exist yet says about itself, for the gate.
The steps belong in it: a plan passed with steps and no design is still
recognisable by them, and what its steps say is most of what makes two
plans the same plan.
"""
parts = [(description or "").strip(), (body or "").strip()]
parts += [t.strip() for t in (step_texts or [])]
return "\n\n".join(p for p in parts if p)
async def find_matching_plan(
user_id: int,
project_id: int,
title: str,
text: str = "",
) -> DuplicateMatch | None:
"""An ACTIVE milestone in the project that already is this plan, or None.
Normalized-title match first, then semantic when `text` (from
plan_candidate_text) is long enough to mean something, the same floor the
note gate uses and for the same reason: a title-only embedding sits in a
tight neighbourhood and false-positives. Never raises; a failed check lets
the plan through, because a create must not depend on a recall aid.
"""
from scribe.models.milestone import Milestone
if not project_id:
return None
# Rule 78, before either arm: a match names a milestone, and a caller who
# cannot read the project must not learn its plans by guessing titles.
try:
if not await can_read_project(user_id, project_id):
return None
except Exception:
logger.debug("plan gate access check failed — letting the plan through", exc_info=True)
return None
norm = " ".join((title or "").split()).lower()
if norm:
try:
async with async_session() as session:
existing = (await session.execute(
select(Milestone).where(
Milestone.project_id == project_id,
Milestone.deleted_at.is_(None),
Milestone.status == "active",
func.lower(func.trim(Milestone.title)) == norm,
).limit(1)
)).scalars().first()
if existing is not None:
return DuplicateMatch(existing.id, existing.title, 1.0, "title")
except Exception:
logger.debug("plan gate title check skipped — query failed", exc_info=True)
return None
if len((text or "").strip()) < _MIN_BODY_FOR_SEMANTIC:
return None
doc_title, doc_body = embeddings_svc.milestone_document(title, None, text)
query = "\n\n".join(p for p in (doc_title, doc_body) if p)
try:
hits = await embeddings_svc.semantic_search_milestones(
user_id, query, project_id=project_id, status="active", limit=1,
threshold=await get_plan_match_threshold(user_id),
)
except Exception:
logger.debug("plan gate semantic check skipped", exc_info=True)
return None
if hits:
score, milestone = hits[0]
return DuplicateMatch(milestone.id, milestone.title, round(score, 3), "semantic")
return None
def plan_match_response(dup: DuplicateMatch, progress: dict | None = None) -> dict:
"""The payload start_planning / create_milestone return instead of a second
plan: the existing one, how far along it is, and how to add to it."""
progress = progress or {}
total, completed = progress.get("total", 0), progress.get("completed", 0)
how = "has the same title" if dup.reason == "title" else "reads as the same plan"
return {
"duplicate": True,
"existing_id": dup.id,
"existing_title": dup.title,
"existing_milestone": {
"id": dup.id,
"title": dup.title,
"description": progress.get("description") or "",
"total": total,
"completed": completed,
},
"similarity": dup.similarity,
"match": dup.reason,
"message": (
f'An active plan in this project {how}: milestone {dup.id} '
f'"{dup.title}" ({completed} of {total} steps done). Nothing was '
f"created. Add your steps to it with create_records(milestone_id="
f"{dup.id}, ...), and revise its design with update_milestone if the "
f"scope has grown. Read it first with get_milestone({dup.id}). If this "
f"really is a separate plan, retry with force=true."
),
}
async def plan_gate(
user_id: int,
project_id: int,
title: str,
text: str = "",
) -> dict | None:
"""find_matching_plan, answered: the plan_match_response to return in
place of a new plan, or None to go ahead and create it."""
from scribe.services import milestones as milestones_svc
dup = await find_matching_plan(user_id, project_id, title, text)
if dup is None:
return None
try:
rows = await milestones_svc.get_project_milestone_summary(user_id, project_id)
progress = next((r for r in rows if r.get("id") == dup.id), None)
except Exception:
# The match stands without its progress; losing the count must not
# turn a found plan into a second one.
progress = None
return plan_match_response(dup, progress)
+45 -1
View File
@@ -1,4 +1,4 @@
"""Real-Postgres tests for finding a plan by meaning (milestone 415, step 3). """Real-Postgres tests for finding a plan by meaning (milestone 415, steps 3 and 4).
A project's roadmap written as milestones was invisible to recall: `search` A project's roadmap written as milestones was invisible to recall: `search`
covered notes, tasks and rules, so "is there already a plan for this?" had no covered notes, tasks and rules, so "is there already a plan for this?" had no
@@ -16,6 +16,7 @@ from scribe.models import async_session
from scribe.models.embedding import EMBEDDING_DIM, MilestoneEmbedding from scribe.models.embedding import EMBEDDING_DIM, MilestoneEmbedding
from scribe.models.milestone import Milestone from scribe.models.milestone import Milestone
from scribe.models.project import Project from scribe.models.project import Project
from scribe.services import dedup as dedup_svc
from scribe.services.embeddings import CHUNKER_VERSION, semantic_search_milestones from scribe.services.embeddings import CHUNKER_VERSION, semantic_search_milestones
from tests.helpers import ensure_user from tests.helpers import ensure_user
@@ -80,3 +81,46 @@ async def test_without_a_project_it_searches_the_callers_own(roadmap):
async def test_a_project_the_caller_cannot_read_returns_nothing(roadmap): async def test_a_project_the_caller_cannot_read_returns_nothing(roadmap):
assert await _found(roadmap["stranger"], project_id=roadmap["mine"]) == [] assert await _found(roadmap["stranger"], project_id=roadmap["mine"]) == []
assert await _found(roadmap["stranger"]) == [] assert await _found(roadmap["stranger"]) == []
# ── the plan gate (step 4): start_planning finds the plan before making another ──
LONG = "Resolve works and editions against the metadata providers. " * 5
async def _gate(roadmap, title, text="", project=None):
with patch("scribe.services.embeddings.get_embedding", AsyncMock(return_value=NEAR)):
return await dedup_svc.plan_gate(
roadmap["owner"], project or roadmap["mine"], title, text,
)
async def test_the_gate_returns_an_active_plan_with_the_same_title(roadmap):
out = await _gate(roadmap, " m3 — METADATA ")
assert out["duplicate"] is True and out["match"] == "title"
assert out["existing_milestone"]["id"] == roadmap["m3"]
assert f"create_records(milestone_id={roadmap['m3']}" in out["message"]
async def test_the_gate_finds_a_plan_by_meaning_and_never_a_done_one(roadmap):
"""NEAR matches m3 (active), `done` (done) and `foreign` (another project).
Only m3 is a plan this project is still working through."""
out = await _gate(roadmap, "Book metadata", LONG)
assert out["match"] == "semantic"
assert out["existing_id"] == roadmap["m3"]
async def test_a_done_plan_and_another_projects_plan_do_not_block(roadmap):
assert await _gate(roadmap, "Covers") is None
assert await _gate(roadmap, "Metadata elsewhere") is None
async with async_session() as s:
m3 = await s.get(Milestone, roadmap["m3"])
m3.status = "done"
await s.commit()
assert await _gate(roadmap, "Book metadata", LONG) is None
async def test_a_short_candidate_is_judged_by_title_alone(roadmap):
"""A title-only embedding sits in a tight neighbourhood and matches
anything nearby, so a bare title never takes the semantic arm."""
assert await _gate(roadmap, "Book metadata", "just a title") is None
+1
View File
@@ -120,6 +120,7 @@ async def test_start_planning_hands_its_steps_to_the_service():
from scribe.mcp.tools.tasks import start_planning from scribe.mcp.tools.tasks import start_planning
with patch("scribe.mcp.tools.tasks.dedup_svc.find_duplicate_note", AsyncMock(return_value=None)), \ with patch("scribe.mcp.tools.tasks.dedup_svc.find_duplicate_note", AsyncMock(return_value=None)), \
patch("scribe.mcp.tools.tasks.dedup_svc.plan_gate", AsyncMock(return_value=None)), \
patch("scribe.mcp.tools.tasks.planning_svc.start_planning", patch("scribe.mcp.tools.tasks.planning_svc.start_planning",
AsyncMock(return_value={"milestone": {"id": 1}})) as svc: AsyncMock(return_value={"milestone": {"id": 1}})) as svc:
await start_planning(project_id=3, title="Plan", body="see {{ref:1}}", await start_planning(project_id=3, title="Plan", body="see {{ref:1}}",
+36
View File
@@ -12,6 +12,15 @@ from tests.helpers import fake_milestone
pytestmark = pytest.mark.usefixtures("_bind_user") pytestmark = pytest.mark.usefixtures("_bind_user")
@pytest.fixture(autouse=True)
def _no_plan_gate():
"""The plan gate reads the database; these tests are about what reaches
the service. The gate's own tests re-patch it."""
with patch("scribe.mcp.tools.milestones.dedup_svc.plan_gate",
AsyncMock(return_value=None)):
yield
@pytest.mark.asyncio @pytest.mark.asyncio
async def test_list_milestones_returns_dict_with_progress(): async def test_list_milestones_returns_dict_with_progress():
rows = [{"id": 1, "title": "MS1", "status": "active", "total": 2}] rows = [{"id": 1, "title": "MS1", "status": "active", "total": 2}]
@@ -63,6 +72,33 @@ async def test_create_milestone_empty_body_becomes_none():
assert mock.call_args.kwargs["body"] is None assert mock.call_args.kwargs["body"] is None
@pytest.mark.asyncio
async def test_create_milestone_returns_the_active_plan_that_already_covers_it():
match = {"duplicate": True, "existing_id": 4}
create = AsyncMock()
with patch("scribe.mcp.tools.milestones.dedup_svc.plan_gate",
AsyncMock(return_value=match)) as gate, \
patch("scribe.mcp.tools.milestones.milestones_svc.create_milestone", create):
out = await create_milestone(project_id=1, title="t", description="d", body="b")
assert out is match
create.assert_not_awaited()
assert gate.call_args.args[:3] == (7, 1, "t")
assert gate.call_args.args[3] == "d\n\nb"
@pytest.mark.asyncio
async def test_create_milestone_skips_the_gate_when_forced_or_done():
"""force: the caller read the match. done: a record of past work is not a
plan competing with an open one."""
gate = AsyncMock(return_value={"duplicate": True})
create = AsyncMock(return_value=fake_milestone(id=6))
with patch("scribe.mcp.tools.milestones.dedup_svc.plan_gate", gate), \
patch("scribe.mcp.tools.milestones.milestones_svc.create_milestone", create):
assert (await create_milestone(project_id=1, title="t", force=True))["id"] == 6
assert (await create_milestone(project_id=1, title="t", status="done"))["id"] == 6
gate.assert_not_awaited()
@pytest.mark.asyncio @pytest.mark.asyncio
async def test_update_milestone_sends_body(): async def test_update_milestone_sends_body():
m = fake_milestone() m = fake_milestone()
+43
View File
@@ -7,6 +7,13 @@ from tests.helpers import fake_task
pytestmark = pytest.mark.usefixtures("_bind_user") pytestmark = pytest.mark.usefixtures("_bind_user")
@pytest.fixture(autouse=True)
def _no_plan_gate():
"""The plan gate reads the database; the tests that are about it re-patch it."""
with patch("scribe.mcp.tools.tasks.dedup_svc.plan_gate", AsyncMock(return_value=None)):
yield
@pytest.mark.asyncio @pytest.mark.asyncio
async def test_start_planning_tool_delegates_to_service(): async def test_start_planning_tool_delegates_to_service():
payload = {"milestone": {"id": 5}, "applicable_rules": [], "project_rules": [], payload = {"milestone": {"id": 5}, "applicable_rules": [], "project_rules": [],
@@ -23,6 +30,42 @@ async def test_start_planning_tool_delegates_to_service():
} }
@pytest.mark.asyncio
async def test_start_planning_returns_the_plan_that_already_covers_it():
"""The FabledLibrarian failure (milestone 415): a session that could not see
the existing plan made a second one. Now it is handed the first, and
nothing is created."""
match = {"duplicate": True, "existing_id": 12}
svc = AsyncMock()
with patch("scribe.mcp.tools.tasks.dedup_svc.plan_gate",
AsyncMock(return_value=match)) as gate, \
patch("scribe.mcp.tools.tasks.dedup_svc.find_duplicate_note", AsyncMock()) as steps, \
patch("scribe.mcp.tools.tasks.planning_svc.start_planning", svc):
from scribe.mcp.tools.tasks import start_planning
out = await start_planning(project_id=3, title="Metadata", body="design",
steps=[{"title": "Resolve editions", "body": "via providers"}])
assert out is match
svc.assert_not_awaited()
steps.assert_not_awaited()
# The candidate is judged by its steps as well as its design.
assert gate.call_args.args[:3] == (7, 3, "Metadata")
assert "design" in gate.call_args.args[3]
assert "Resolve editions\nvia providers" in gate.call_args.args[3]
@pytest.mark.asyncio
async def test_force_creates_the_plan_without_asking_the_gate():
gate = AsyncMock(return_value={"duplicate": True})
with patch("scribe.mcp.tools.tasks.dedup_svc.plan_gate", gate), \
patch("scribe.mcp.tools.tasks.planning_svc.start_planning",
AsyncMock(return_value={"milestone": {"id": 5}})) as svc:
from scribe.mcp.tools.tasks import start_planning
out = await start_planning(project_id=3, title="Metadata", force=True)
assert out["milestone"]["id"] == 5
svc.assert_awaited_once()
gate.assert_not_awaited()
@pytest.mark.asyncio @pytest.mark.asyncio
async def test_get_task_augments_plan_with_rules(): async def test_get_task_augments_plan_with_rules():
applicable = {"rules": [{"id": 1, "title": "r"}], "truncated": False, applicable = {"rules": [{"id": 1, "title": "r"}], "truncated": False,
+80
View File
@@ -4,10 +4,15 @@ from unittest.mock import AsyncMock, MagicMock, patch
import pytest import pytest
from scribe.services.dedup import ( from scribe.services.dedup import (
PLAN_MATCH_DEFAULT_THRESHOLD,
DuplicateMatch, DuplicateMatch,
duplicate_response, duplicate_response,
find_duplicate_note, find_duplicate_note,
find_duplicate_rule, find_duplicate_rule,
find_matching_plan,
get_plan_match_threshold,
plan_candidate_text,
plan_match_response,
) )
from tests.helpers import fake_note, make_mock_session from tests.helpers import fake_note, make_mock_session
@@ -342,3 +347,78 @@ def test_every_kind_has_a_suggestion_and_none_proposes_merging_notes():
assert "merge" in _KIND_SUGGESTION["snippet"] assert "merge" in _KIND_SUGGESTION["snippet"]
assert "NOT merge" in _KIND_SUGGESTION["note"] assert "NOT merge" in _KIND_SUGGESTION["note"]
assert "supersedes" in _KIND_SUGGESTION["note"] assert "supersedes" in _KIND_SUGGESTION["note"]
# ── the plan gate (milestone 415, step 4) ─────────────────────────────────────
@pytest.mark.asyncio
async def test_plan_title_match_short_circuits_the_semantic_arm():
ms = MagicMock(id=415, title="Plan gate")
sem = AsyncMock()
with patch("scribe.services.dedup.can_read_project", AsyncMock(return_value=True)), \
patch("scribe.services.dedup.async_session", return_value=_session_returning(ms)), \
patch("scribe.services.dedup.embeddings_svc.semantic_search_milestones", sem):
dup = await find_matching_plan(7, 2, " plan GATE", "x" * 300)
assert (dup.id, dup.reason, dup.similarity) == (415, "title", 1.0)
sem.assert_not_awaited()
@pytest.mark.asyncio
async def test_plan_semantic_arm_asks_for_active_plans_in_the_project_at_the_setting():
ms = MagicMock(id=9, title="Metadata")
sem = AsyncMock(return_value=[(0.912345, ms)])
with patch("scribe.services.dedup.can_read_project", AsyncMock(return_value=True)), \
patch("scribe.services.dedup.async_session", return_value=_session_returning(None)), \
patch("scribe.services.dedup.embeddings_svc.semantic_search_milestones", sem), \
patch("scribe.services.settings.get_setting", AsyncMock(return_value="0.8")):
dup = await find_matching_plan(7, 2, "Book metadata", "x" * 300)
assert (dup.id, dup.reason, dup.similarity) == (9, "semantic", 0.912)
kw = sem.call_args.kwargs
assert (kw["project_id"], kw["status"], kw["threshold"]) == (2, "active", 0.8)
@pytest.mark.asyncio
async def test_plan_gate_fails_open():
boom = MagicMock(side_effect=RuntimeError("db down"))
with patch("scribe.services.dedup.can_read_project", AsyncMock(return_value=True)), \
patch("scribe.services.dedup.async_session", boom):
assert await find_matching_plan(7, 2, "Anything", "x" * 300) is None
with patch("scribe.services.dedup.can_read_project", AsyncMock(side_effect=RuntimeError)):
assert await find_matching_plan(7, 2, "Anything", "x" * 300) is None
assert await find_matching_plan(7, 0, "No project") is None
@pytest.mark.asyncio
async def test_plan_gate_says_nothing_about_a_project_the_caller_cannot_read():
ms = MagicMock(id=415, title="Their plan")
session = MagicMock(return_value=_session_returning(ms))
with patch("scribe.services.dedup.can_read_project", AsyncMock(return_value=False)), \
patch("scribe.services.dedup.async_session", session):
assert await find_matching_plan(8, 2, "Their plan") is None
session.assert_not_called()
@pytest.mark.asyncio
async def test_a_bad_threshold_setting_falls_back_to_the_default():
with patch("scribe.services.settings.get_setting", AsyncMock(return_value="lots")):
assert await get_plan_match_threshold(7) == PLAN_MATCH_DEFAULT_THRESHOLD
with patch("scribe.services.settings.get_setting", AsyncMock(return_value="7")):
assert await get_plan_match_threshold(7) == 1.0
def test_plan_candidate_text_carries_the_steps():
text = plan_candidate_text(description=None, body=" design ", step_texts=["Step one\n", ""])
assert text == "design\n\nStep one"
def test_plan_match_response_points_at_adding_steps_not_a_second_plan():
out = plan_match_response(DuplicateMatch(12, "Metadata", 0.93, "semantic"),
{"total": 5, "completed": 2, "description": "providers"})
assert out["duplicate"] is True and out["existing_id"] == 12
assert out["existing_milestone"] == {
"id": 12, "title": "Metadata", "description": "providers", "total": 5, "completed": 2,
}
for phrase in ("create_records(milestone_id=12", "get_milestone(12)", "force=true",
"2 of 5 steps"):
assert phrase in out["message"]
+21 -17
View File
@@ -36,28 +36,32 @@ import re
import pytest import pytest
ROOT = pathlib.Path(__file__).resolve().parents[1] ROOT = pathlib.Path(__file__).resolve().parents[1]
_PY = ROOT / "src" / "scribe" / "services" / "plugin_context.py" _SERVICES = ROOT / "src" / "scribe" / "services"
_VUE = ROOT / "frontend" / "src" / "views" / "SettingsView.vue" _VUE = ROOT / "frontend" / "src" / "views" / "SettingsView.vue"
# (python constant, vue ref). Hand-written because the pairing is an editorial # (services module, python constant, vue ref). Hand-written because the
# fact — the names do not share a convention either side could derive — but # pairing is an editorial fact — the names do not share a convention either
# every entry is asserted to EXIST on both sides, so a rename fails loudly # side could derive — but every entry is asserted to EXIST on both sides, so a
# here rather than silently dropping that threshold from the check. # rename fails loudly here rather than silently dropping that threshold from
# the check.
_PAIRS = ( _PAIRS = (
("AUTOINJECT_DEFAULT_THRESHOLD", "kbInjectThreshold"), ("plugin_context.py", "AUTOINJECT_DEFAULT_THRESHOLD", "kbInjectThreshold"),
("WRITEPATH_DEFAULT_THRESHOLD", "kbWritePathThreshold"), ("plugin_context.py", "WRITEPATH_DEFAULT_THRESHOLD", "kbWritePathThreshold"),
("RULEHINT_DEFAULT_THRESHOLD", "kbRuleHintThreshold"), ("plugin_context.py", "RULEHINT_DEFAULT_THRESHOLD", "kbRuleHintThreshold"),
("TOOLRULE_DEFAULT_THRESHOLD", "kbToolRuleThreshold"), ("plugin_context.py", "TOOLRULE_DEFAULT_THRESHOLD", "kbToolRuleThreshold"),
("PROMPTRULE_DEFAULT_THRESHOLD", "kbPromptRuleThreshold"), ("plugin_context.py", "PROMPTRULE_DEFAULT_THRESHOLD", "kbPromptRuleThreshold"),
# The plan gate (milestone 415): it blocks a create, so a form showing a
# looser bar than the one in force would be the more misleading drift.
("dedup.py", "PLAN_MATCH_DEFAULT_THRESHOLD", "kbPlanMatchThreshold"),
) )
def _python_default(name: str) -> float: def _python_default(module: str, name: str) -> float:
m = re.search(rf"^{re.escape(name)}\s*=\s*([0-9.]+)\s*$", m = re.search(rf"^{re.escape(name)}\s*=\s*([0-9.]+)\s*$",
_PY.read_text(), re.M) (_SERVICES / module).read_text(), re.M)
assert m, ( assert m, (
f"{name} is no longer a bare module-level float in " f"{name} is no longer a bare module-level float in "
f"services/plugin_context.py. If it moved or was renamed, update " f"services/{module}. If it moved or was renamed, update "
f"_PAIRS; if it was retired, drop its row — leaving it here checks " f"_PAIRS; if it was retired, drop its row — leaving it here checks "
f"nothing while looking like coverage." f"nothing while looking like coverage."
) )
@@ -75,11 +79,11 @@ def _vue_default(ref_name: str) -> float:
return float(m.group(1)) return float(m.group(1))
@pytest.mark.parametrize(("constant", "ref_name"), _PAIRS, @pytest.mark.parametrize(("module", "constant", "ref_name"), _PAIRS,
ids=[p[0] for p in _PAIRS]) ids=[p[1] for p in _PAIRS])
def test_the_form_shows_the_default_the_server_uses(constant, ref_name): def test_the_form_shows_the_default_the_server_uses(module, constant, ref_name):
"""An untouched control must render the bar actually in force.""" """An untouched control must render the bar actually in force."""
server, form = _python_default(constant), _vue_default(ref_name) server, form = _python_default(module, constant), _vue_default(ref_name)
assert form == server, ( assert form == server, (
f"SettingsView shows {form} for {ref_name} while the server defaults " f"SettingsView shows {form} for {ref_name} while the server defaults "
f"to {server} ({constant}). An operator who has never set this reads " f"to {server} ({constant}). An operator who has never set this reads "