feat(mcp): Phase 5 — write-time near-duplicate gate (update-over-create)
#755 Phase 5. create_note / create_task now BLOCK a near-duplicate instead of silently inserting: they return {"duplicate": true, "existing_id", message} pointing at the record to UPDATE. Fights store bloat and stale competing copies that semantic search (RAG) would otherwise resurface for reconciliation. A force=true override creates anyway for genuinely-distinct records. - services/dedup.py: find_duplicate_note — two signals, scoped to owner + same project + same kind: (1) normalized-title exact match (cheap, always); (2) semantic cosine ≥ 0.90 but ONLY when body ≥ 200 chars (short/title-only embeddings false-positive — the pre-pivot lesson). Project-less (orphan) records compare only to other orphans on BOTH signals (orphan_only on the semantic call) — they're not matched across every project. - Gate wired into the MCP create_note/create_task tools (the LLM write path) with force override; _INSTRUCTIONS documents the duplicate response + force. - Opt-in by design: the service helper is only called from the interactive create tools. Internal/programmatic creates (recurrence spawn, imports) go straight through services.create_note and are NOT gated — a recurring task spawning its next same-titled instance must not be blocked. - Scope v1: MCP tools only. REST/web (human CRUD, needs a UI affordance) and create_rule (not a RAG surface; _INSTRUCTIONS already steer it) are follow-ups. - tests: dedup service (title/semantic/body-gate/type-filter) + tool gate (blocks, force bypasses) for notes and tasks. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,124 @@
|
||||
"""Write-time near-duplicate detection — the update-over-create gate.
|
||||
|
||||
Goal: stop a second near-identical row from being created when an existing one
|
||||
should be UPDATED instead. Duplicates bloat the store and, worse, get surfaced
|
||||
by semantic search (RAG) later as competing/stale copies that then have to be
|
||||
reconciled. This is the enforcement half of the instruction-level "prefer
|
||||
updating over creating" reflex.
|
||||
|
||||
OPT-IN by design: the interactive create paths (MCP create tools + REST create
|
||||
routes) run this gate; internal/programmatic creates do NOT (e.g. a recurring
|
||||
task spawning its next instance, or a bulk import — those legitimately repeat a
|
||||
title and must not be blocked). Callers that want the gate call find_duplicate_*
|
||||
themselves and act on a hit; nothing here mutates.
|
||||
|
||||
Two signals, both scoped to the same owner + project + kind:
|
||||
1. Normalized-title exact match — cheap, always checked.
|
||||
2. Semantic similarity (cosine ≥ _SEMANTIC_THRESHOLD) — only when the incoming
|
||||
body is substantial. Short/title-only embeddings sit in a tight neighborhood
|
||||
and false-positive (the pre-pivot lesson: "Lore: Shell 0" vs
|
||||
"Lore: Reinitialization 0" matched at 0.91 with no body), so we gate it on
|
||||
a minimum body length.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
from sqlalchemy import func, select
|
||||
|
||||
from scribe.models import async_session
|
||||
from scribe.models.note import Note
|
||||
from scribe.services import embeddings as embeddings_svc
|
||||
|
||||
# Run the semantic check only when the incoming body has at least this many
|
||||
# characters — below it, embeddings are dominated by the title and false-positive.
|
||||
_MIN_BODY_FOR_SEMANTIC = 200
|
||||
# Cosine threshold for "this is the same thing, reworded." Deliberately high to
|
||||
# keep false positives rare (a hard block with a force-override is unforgiving of
|
||||
# noise). Matches the 0.90 the pre-pivot dedup settled on.
|
||||
_SEMANTIC_THRESHOLD = 0.90
|
||||
|
||||
|
||||
@dataclass
|
||||
class DuplicateMatch:
|
||||
"""An existing record judged a near-duplicate of an incoming create."""
|
||||
id: int
|
||||
title: str
|
||||
similarity: float # 1.0 for an exact normalized-title match
|
||||
reason: str # "title" | "semantic"
|
||||
|
||||
|
||||
def duplicate_response(dup: "DuplicateMatch", kind: str) -> dict:
|
||||
"""Standard 'blocked — update instead' payload returned by a create tool
|
||||
when the gate finds a near-duplicate. `kind` is 'note' or 'task' (drives the
|
||||
update_<kind> hint)."""
|
||||
return {
|
||||
"duplicate": True,
|
||||
"existing_id": dup.id,
|
||||
"existing_title": dup.title,
|
||||
"similarity": dup.similarity,
|
||||
"match": dup.reason,
|
||||
"message": (
|
||||
f'A {dup.reason}-similar {kind} already exists (id {dup.id}: '
|
||||
f'"{dup.title}"). Prefer UPDATING it (update_{kind}) over creating a '
|
||||
f"near-duplicate. If this really is a distinct {kind}, retry with "
|
||||
f"force=true."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
async def find_duplicate_note(
|
||||
user_id: int,
|
||||
title: str,
|
||||
body: str = "",
|
||||
project_id: int | None = None,
|
||||
is_task: bool | None = None,
|
||||
note_type: str = "note",
|
||||
) -> DuplicateMatch | None:
|
||||
"""Best near-duplicate of (title, body) within the same owner + project +
|
||||
kind, or None. Title match first (cheap, exact), then semantic when the body
|
||||
is long enough to be meaningful. Never raises — embedder failure degrades to
|
||||
title-only (callers should still be able to create)."""
|
||||
norm = " ".join((title or "").split()).lower()
|
||||
|
||||
# --- Signal 1: normalized-title exact match (same scope) ---
|
||||
if norm:
|
||||
async with async_session() as session:
|
||||
stmt = select(Note).where(
|
||||
Note.user_id == user_id,
|
||||
Note.deleted_at.is_(None),
|
||||
Note.note_type == note_type,
|
||||
func.lower(func.trim(Note.title)) == norm,
|
||||
)
|
||||
if project_id is not None:
|
||||
stmt = stmt.where(Note.project_id == project_id)
|
||||
else:
|
||||
stmt = stmt.where(Note.project_id.is_(None))
|
||||
if is_task is True:
|
||||
stmt = stmt.where(Note.status.isnot(None))
|
||||
elif is_task is False:
|
||||
stmt = stmt.where(Note.status.is_(None))
|
||||
existing = (await session.execute(stmt.limit(1))).scalars().first()
|
||||
if existing is not None:
|
||||
return DuplicateMatch(existing.id, existing.title, 1.0, "title")
|
||||
|
||||
# --- Signal 2: semantic similarity (only with a substantial body) ---
|
||||
if body and len(body.strip()) >= _MIN_BODY_FOR_SEMANTIC:
|
||||
query = f"{title}\n{body}".strip()
|
||||
# Scope the semantic check the same way as the title check: a record in
|
||||
# project P compares only to P; a project-less (orphan) record compares
|
||||
# only to other orphans (orphan_only), NOT across every project — without
|
||||
# this, semantic_search_notes applies no project filter when project_id
|
||||
# is None and would match an orphan note against any project's notes.
|
||||
hits = await embeddings_svc.semantic_search_notes(
|
||||
user_id, query, project_id=project_id, is_task=is_task,
|
||||
orphan_only=(project_id is None),
|
||||
limit=3, threshold=_SEMANTIC_THRESHOLD,
|
||||
)
|
||||
for score, note in hits:
|
||||
# semantic_search_notes doesn't filter note_type — enforce it here so
|
||||
# a note doesn't shadow a task of the same wording, etc.
|
||||
if note.note_type == note_type:
|
||||
return DuplicateMatch(note.id, note.title, round(score, 3), "semantic")
|
||||
|
||||
return None
|
||||
Reference in New Issue
Block a user