diff --git a/alembic/versions/0070_notes_data_jsonb.py b/alembic/versions/0070_notes_data_jsonb.py
new file mode 100644
index 0000000..5518c36
--- /dev/null
+++ b/alembic/versions/0070_notes_data_jsonb.py
@@ -0,0 +1,60 @@
+"""add notes.data JSONB — queryable structured fields for typed records
+
+Revision ID: 0070
+Revises: 0069
+Create Date: 2026-07-26
+
+Snippets (note_type='snippet') carry structured fields — name, language,
+signature, and a list of canonical locations (repo · path · symbol). Those were
+stored as a markdown body-convention, which reads well and feeds the embedding
+but cannot be QUERIED: answering "which snippets live in this file?" meant
+scanning every snippet and regexing its body.
+
+This adds a general `data` JSONB column plus a GIN index, so those fields become
+indexable. The body stays exactly as it was — it is still the human-readable
+form and still what gets embedded. `data` is a queryable mirror of the same
+facts, not a replacement, and the code itself is deliberately NOT copied into it
+(the body already holds it; duplicating a blob to index fields around it would
+be waste).
+
+Relationship to 0069: that migration DROPPED `notes.metadata`, a JSONB column
+which only ever held person/place/list entity fields, when those surfaces were
+removed. This is not a revival of it — different name, different purpose, and
+nothing reads the old shape. The column is named `data` rather than `metadata`
+because `metadata` collides with SQLAlchemy's declarative `Base.metadata`, which
+is why the old model had to map an awkward `entity_metadata` attribute onto it.
+
+Nullable with no backfill, deliberately: rows written before this migration keep
+working because the service falls back to parsing the body when `data` is
+absent. That means no migration deadline and no risk of a backfill mangling a
+hand-edited body.
+
+Downgrade drops the index and the column. Any structured fields it held remain
+recoverable from the body convention, which is the same source they mirror.
+"""
+from alembic import op
+import sqlalchemy as sa
+from sqlalchemy.dialects.postgresql import JSONB
+
+
+revision = "0070"
+down_revision = "0069"
+branch_labels = None
+depends_on = None
+
+
+def upgrade() -> None:
+ op.add_column("notes", sa.Column("data", JSONB, nullable=True))
+ # GIN supports containment (`data @> '{"locations":[{"repo":"x"}]}'`), which
+ # covers exact repo/path/symbol and language lookups. Path PREFIX matching
+ # ("everything under frontend/src/") is not an index-served operation here
+ # and still filters after the fact — acceptable while snippet counts are
+ # small, and a generated column is the escape hatch if that changes.
+ op.create_index(
+ "ix_notes_data_gin", "notes", ["data"], postgresql_using="gin",
+ )
+
+
+def downgrade() -> None:
+ op.drop_index("ix_notes_data_gin", table_name="notes")
+ op.drop_column("notes", "data")
diff --git a/frontend/src/api/snippets.ts b/frontend/src/api/snippets.ts
index 6b4a025..fa7ef03 100644
--- a/frontend/src/api/snippets.ts
+++ b/frontend/src/api/snippets.ts
@@ -20,6 +20,9 @@ export interface SnippetFields {
path: string;
symbol: string;
locations: SnippetLocation[];
+ /** Ids of the snippets folded into this one by merge, oldest first. Read-only:
+ * merge is the only thing that adds to it, and an edit carries it forward. */
+ merged_from: number[];
code: string;
}
@@ -77,13 +80,26 @@ export interface SnippetInput {
force?: boolean;
}
+/** `repo`/`path`/`symbol` are the reverse lookup — "what already lives here?"
+ * They must all match the same recorded location; `path` also matches anything
+ * beneath it. Combinable with `q`. */
export async function listSnippets(
- params: { q?: string; tag?: string; projectId?: number | null } = {},
+ params: {
+ q?: string;
+ tag?: string;
+ projectId?: number | null;
+ repo?: string;
+ path?: string;
+ symbol?: string;
+ } = {},
): Promise<{ snippets: SnippetListItem[]; total: number }> {
const qs = new URLSearchParams();
if (params.q) qs.set("q", params.q);
if (params.tag) qs.set("tag", params.tag);
if (params.projectId) qs.set("project_id", String(params.projectId));
+ if (params.repo) qs.set("repo", params.repo);
+ if (params.path) qs.set("path", params.path);
+ if (params.symbol) qs.set("symbol", params.symbol);
const query = qs.toString();
return apiGet(`/api/snippets${query ? `?${query}` : ""}`);
}
diff --git a/frontend/src/views/SettingsView.vue b/frontend/src/views/SettingsView.vue
index 4d3b3b0..1321cc4 100644
--- a/frontend/src/views/SettingsView.vue
+++ b/frontend/src/views/SettingsView.vue
@@ -22,6 +22,7 @@ const retentionSaved = ref(false);
const kbInjectEnabled = ref(true);
const kbInjectThreshold = ref("0.55");
const kbInjectTopK = ref("3");
+const kbWritePathEnabled = ref(true);
const savingKbInject = ref(false);
const kbInjectSaved = ref(false);
@@ -76,6 +77,9 @@ async function saveKbInject() {
kb_autoinject_enabled: kbInjectEnabled.value ? 'true' : 'false',
kb_autoinject_threshold: String(t),
kb_autoinject_top_k: String(k),
+ // Its own switch, but deliberately the same threshold/ceiling — see
+ // WRITEPATH_ENABLED_KEY in services/plugin_context.py.
+ kb_writepath_enabled: kbWritePathEnabled.value ? 'true' : 'false',
});
kbInjectSaved.value = true;
setTimeout(() => (kbInjectSaved.value = false), 2000);
@@ -459,6 +463,7 @@ onMounted(async () => {
if (allSettings.kb_autoinject_top_k !== undefined) {
kbInjectTopK.value = allSettings.kb_autoinject_top_k;
}
+ kbWritePathEnabled.value = allSettings.kb_writepath_enabled !== "false";
if (allSettings.notify_task_reminders !== undefined) {
notifyTaskReminders.value = allSettings.notify_task_reminders !== "false";
}
@@ -1193,6 +1198,19 @@ function formatUserDate(iso: string): string {
/>
Ceiling on titles surfaced at once (1–10).
+
+
+
+ Also surface prior art when Claude writes code
+
+
+ Checks the file Claude is about to write or edit against your recorded
+ snippets — what's already kept at that path, and what resembles the code
+ being written — so a helper you already have is offered before it's
+ rewritten. Uses the same threshold and ceiling above, and never blocks the
+ edit. Off = prior art surfaces only on your own prompts.
+
+
{{ savingKbInject ? 'Saving…' : 'Save' }}
diff --git a/frontend/src/views/SnippetDetailView.vue b/frontend/src/views/SnippetDetailView.vue
index 6718ac5..3d72f92 100644
--- a/frontend/src/views/SnippetDetailView.vue
+++ b/frontend/src/views/SnippetDetailView.vue
@@ -23,6 +23,10 @@ const canWrite = computed(() => {
const locations = computed(() => snippet.value?.snippet.locations ?? []);
+// What this record absorbed. Merge keeps the target's fields and trashes the
+// sources, so this line is the only visible trace that the variants existed.
+const mergedFrom = computed(() => snippet.value?.snippet.merged_from ?? []);
+
function locParts(loc: { repo: string; path: string; symbol: string }): string[] {
return [loc.repo, loc.path, loc.symbol].filter((p) => p && p.trim());
}
@@ -110,6 +114,15 @@ async function confirmDelete() {
Language
{{ snippet.snippet.language }}
+
+ Merged from
+
+ #{{ mid }}
+
+ folded in here — the originals are in the trash
+
+
+
@@ -274,6 +287,16 @@ async function confirmDelete() {
gap: 0.35rem;
align-items: center;
}
+.merged-from {
+ display: flex;
+ flex-wrap: wrap;
+ gap: 0.35rem;
+ align-items: baseline;
+}
+.merged-hint {
+ color: var(--color-text-muted);
+ font-size: 0.8rem;
+}
.code-block {
border: 1px solid var(--color-border);
diff --git a/frontend/src/views/SnippetListView.vue b/frontend/src/views/SnippetListView.vue
index c5849cb..8fbed78 100644
--- a/frontend/src/views/SnippetListView.vue
+++ b/frontend/src/views/SnippetListView.vue
@@ -13,6 +13,27 @@ const error = ref(null);
const search = ref("");
let searchTimer: ReturnType | undefined;
+// Reverse lookup — "what already lives here?" All three must match the same
+// recorded location; path also matches anything beneath it.
+const showLocationFilter = ref(false);
+const locRepo = ref("");
+const locPath = ref("");
+const locSymbol = ref("");
+const locationActive = computed(
+ () => !!(locRepo.value.trim() || locPath.value.trim() || locSymbol.value.trim()),
+);
+function clearLocation() {
+ locRepo.value = "";
+ locPath.value = "";
+ locSymbol.value = "";
+ loadSnippets();
+}
+function toggleLocationFilter() {
+ showLocationFilter.value = !showLocationFilter.value;
+ // Collapsing while filtered would hide why the list is short.
+ if (!showLocationFilter.value && locationActive.value) clearLocation();
+}
+
// Multi-select → merge
const selectMode = ref(false);
const selectedIds = ref>(new Set());
@@ -70,7 +91,12 @@ async function loadSnippets() {
loading.value = true;
error.value = null;
try {
- const data = await listSnippets({ q: search.value.trim() || undefined });
+ const data = await listSnippets({
+ q: search.value.trim() || undefined,
+ repo: locRepo.value.trim() || undefined,
+ path: locPath.value.trim() || undefined,
+ symbol: locSymbol.value.trim() || undefined,
+ });
snippets.value = data.snippets;
} catch {
error.value = "Couldn't load your snippets.";
@@ -84,6 +110,9 @@ function onSearchInput() {
clearTimeout(searchTimer);
searchTimer = setTimeout(loadSnippets, 300);
}
+// Same debounce for the location fields — typing a path shouldn't fire a query
+// per keystroke.
+const onLocationInput = onSearchInput;
onMounted(loadSnippets);
@@ -127,6 +156,48 @@ function languageOf(tags: string[]): string {
@input="onSearchInput"
@keydown.enter="loadSnippets"
/>
+
+
+ {{ locationActive ? "Location · filtering" : "Location" }}
+
+
+
+
+
+
+
+
+
+ Clear
+
@@ -138,15 +209,24 @@ function languageOf(tags: string[]): string {
❭_
- {{ search.trim() ? "No snippets match your search" : "No snippets kept yet" }}
+ {{ locationActive
+ ? "Nothing kept at that location yet"
+ : search.trim()
+ ? "No snippets match your search"
+ : "No snippets kept yet" }}
- {{ search.trim()
- ? "Try a different term, or clear the search."
- : "Record a reusable function or component and it will be offered back to you later." }}
+ {{ locationActive
+ ? "No recorded snippet lives there — so whatever you're about to write is new. Widen the path, or clear the filter."
+ : search.trim()
+ ? "Try a different term, or clear the search."
+ : "Record a reusable function or component and it will be offered back to you later." }}
+
+ Clear location filter
+
@@ -278,6 +358,62 @@ function languageOf(tags: string[]): string {
.search-row {
margin-bottom: 1.25rem;
+ display: flex;
+ gap: 0.5rem;
+ align-items: center;
+ flex-wrap: wrap;
+}
+/* Filter is engaged — the accent marks "you are here", per the design system. */
+.filter-on {
+ border-color: var(--color-primary);
+ color: var(--color-primary);
+}
+
+.location-row {
+ display: flex;
+ gap: 0.5rem;
+ align-items: center;
+ flex-wrap: wrap;
+ margin: -0.5rem 0 1.25rem;
+}
+.loc-input {
+ flex: 1 1 8rem;
+ min-width: 0;
+ max-width: 12rem;
+ padding: 0.4rem 0.65rem;
+ border: 1px solid var(--color-border);
+ border-radius: var(--radius-sm);
+ background: var(--color-bg);
+ color: var(--color-text);
+ font-size: 0.85rem;
+ font-family: var(--font-mono, ui-monospace, "JetBrains Mono", monospace);
+ box-sizing: border-box;
+}
+.loc-input-wide {
+ flex: 2 1 16rem;
+ max-width: 26rem;
+}
+.loc-input::placeholder {
+ font-family: inherit;
+ color: var(--color-text-muted);
+}
+.loc-input:focus {
+ outline: none;
+ border-color: var(--color-primary);
+ box-shadow: var(--focus-ring);
+}
+.loc-clear {
+ padding: 0.4rem 0.65rem;
+ border: none;
+ background: transparent;
+ color: var(--color-text-secondary);
+ font-size: 0.85rem;
+ font-family: inherit;
+ cursor: pointer;
+ text-decoration: underline;
+}
+.loc-clear:hover {
+ color: var(--color-text);
}
.search-input {
width: 100%;
diff --git a/plugin/.claude-plugin/plugin.json b/plugin/.claude-plugin/plugin.json
index 69161cc..6342cbc 100644
--- a/plugin/.claude-plugin/plugin.json
+++ b/plugin/.claude-plugin/plugin.json
@@ -1,7 +1,7 @@
{
"name": "scribe",
"description": "Scribe system-of-record for Claude Code: MCP tools over your notes/tasks/projects/rules, a session-start push channel that surfaces your always-on rules + active-project context, process-skills (writing-plans, systematic-debugging, verification, brainstorming, reusing-code), and your saved Scribe Processes auto-surfaced as skills (/scribe:sync). Replaces superpowers + file-memory with one app-backed plugin.",
- "version": "0.1.16",
+ "version": "0.1.18",
"author": { "name": "Bryan Van Deusen" },
"mcpServers": {
"scribe": {
diff --git a/plugin/README.md b/plugin/README.md
index 9a76a44..6b98c20 100644
--- a/plugin/README.md
+++ b/plugin/README.md
@@ -7,6 +7,10 @@ instance into a first-class Claude Code extension:
rulebook (the `scribe` server).
- **Session-start push channel** — a `SessionStart` hook injects your always-on
rules + active-project context so Scribe surfaces *without being asked*.
+- **Prior-art recall on writes** — a `PreToolUse` hook on Write/Edit checks the
+ file about to be written against your recorded snippets (what's kept at that
+ path, and what resembles the code) and offers them before the helper is
+ rewritten. Titles only, never blocks the edit.
- **Universal process-skills** — using-scribe, writing-plans,
systematic-debugging, verification, brainstorming, reusing-code (record and
recall reusable code as snippets). Replaces superpowers.
@@ -42,6 +46,11 @@ On install you'll be asked for:
- `hooks/hooks.json` → SessionStart hook (`hooks/scribe_session_context.sh`),
**fail-open**: if Scribe is unreachable it injects nothing and never blocks
the session.
+- `hooks/hooks.json` → PreToolUse hook on `Write|Edit`
+ (`hooks/scribe_prior_art.sh`) → `GET /api/plugin/prior-art`. Returns
+ `additionalContext` with **no** permission decision, so it can inform the write
+ but never stop it; silent when nothing is recorded, which is most of the time.
+ Toggle in **Settings → Knowledge auto-inject**.
- `skills/` → the universal process-skills, surfaced by description match.
- `hooks/scribe_sync_processes.sh` (a 2nd SessionStart hook) + the `/scribe:sync`
command → generate `~/.claude/skills/scribe-proc-*` stubs from your Scribe
@@ -52,5 +61,6 @@ On install you'll be asked for:
- Set a `version` bump in `.claude-plugin/plugin.json` per release so clients
pick up changes.
-- The session-start hook needs only a **read**-scoped key; the MCP tools need
- **write** scope to create/update.
+- The session-start, auto-inject and prior-art hooks need only a **read**-scoped
+ key; the MCP tools need **write** scope to create/update. Every hook is a GET
+ for that reason — a read key cannot POST.
diff --git a/plugin/hooks/hooks.json b/plugin/hooks/hooks.json
index 3b1b802..a9cde62 100644
--- a/plugin/hooks/hooks.json
+++ b/plugin/hooks/hooks.json
@@ -23,6 +23,17 @@
}
]
}
+ ],
+ "PreToolUse": [
+ {
+ "matcher": "Write|Edit",
+ "hooks": [
+ {
+ "type": "command",
+ "command": "bash \"${CLAUDE_PLUGIN_ROOT}/hooks/scribe_prior_art.sh\""
+ }
+ ]
+ }
]
}
}
diff --git a/plugin/hooks/scribe_prior_art.sh b/plugin/hooks/scribe_prior_art.sh
new file mode 100755
index 0000000..8ae9bd1
--- /dev/null
+++ b/plugin/hooks/scribe_prior_art.sh
@@ -0,0 +1,119 @@
+#!/usr/bin/env bash
+# Scribe plugin — PreToolUse write-path trigger (prior-art recall).
+#
+# Auto-inject (scribe_autoinject.sh) fires on the operator's prompt. The moment
+# reuse is actually lost is later: when the AGENT decides mid-task to write a
+# helper. This hook fires there — on Write/Edit — and asks the operator's Scribe
+# instance what prior art is already recorded for the target file: a snippet at
+# that path or in its directory, plus snippets resembling the code about to be
+# written. Titles + ids only, never bodies.
+#
+# NEVER BLOCKS. It returns `additionalContext` with no `permissionDecision`, so
+# the write proceeds untouched and Claude sees the note beside the tool result.
+# Any failure — unconfigured, unreachable, malformed — exits 0 in silence. A
+# recall aid must not be able to stop the operator's work.
+#
+# Config (same as the other hooks), exported to the hook by Claude Code:
+# CLAUDE_PLUGIN_OPTION_api_endpoint base URL, no trailing slash
+# CLAUDE_PLUGIN_OPTION_api_token fmcp_ API key (sensitive)
+# SCRIBE_URL / SCRIBE_TOKEN override for the settings.json dogfooding path.
+set -uo pipefail
+
+command -v jq >/dev/null 2>&1 || exit 0
+command -v curl >/dev/null 2>&1 || exit 0
+
+# PreToolUse delivers { session_id, cwd, tool_name, tool_input: {...}, ... }
+event=$(cat 2>/dev/null || true)
+file_path=$(printf '%s' "$event" | jq -r '.tool_input.file_path // empty' 2>/dev/null) || exit 0
+session_id=$(printf '%s' "$event" | jq -r '.session_id // empty' 2>/dev/null) || session_id=""
+event_cwd=$(printf '%s' "$event" | jq -r '.cwd // empty' 2>/dev/null) || event_cwd=""
+
+[ -n "$file_path" ] || exit 0
+
+# The code about to be written. Write and Edit name this field differently, and
+# the names have changed across Claude Code versions — take whichever is present
+# rather than betting on one shape.
+code=$(printf '%s' "$event" | jq -r '
+ .tool_input.content // .tool_input.file_content //
+ .tool_input.new_string // .tool_input.new_str // empty' 2>/dev/null) || code=""
+
+# Skip formats that hold prose or data rather than reusable code. Purely to
+# avoid a pointless round-trip — the server would return nothing for these
+# anyway. Config formats are NOT skipped: a CI workflow or a compose file is
+# often exactly the thing worth reusing.
+case "$file_path" in
+ *.md|*.mdx|*.txt|*.rst|*.json|*.lock|*.log|*.csv|*.tsv|*.svg|*.png|*.jpg|*.jpeg|*.gif|*.ico|*.pdf)
+ exit 0 ;;
+esac
+
+url=${SCRIBE_URL:-${CLAUDE_PLUGIN_OPTION_api_endpoint:-}}
+token=${SCRIBE_TOKEN:-${CLAUDE_PLUGIN_OPTION_api_token:-}}
+# Guard against an unexpanded ${...} placeholder arriving as a literal.
+case "$url" in *'${'*) url="" ;; esac
+case "$token" in *'${'*) token="" ;; esac
+# Unconfigured install → silent. Prior-art recall is pure enrichment.
+[ -n "$url" ] && [ -n "$token" ] || exit 0
+
+# Snippet locations are recorded repo-relative, so send a repo-relative path —
+# an absolute one would simply match nothing.
+lookup_dir=$(dirname -- "$file_path" 2>/dev/null || true)
+[ -d "$lookup_dir" ] || lookup_dir=${event_cwd:-${CLAUDE_PROJECT_DIR:-$PWD}}
+repo_root=$(git -C "$lookup_dir" rev-parse --show-toplevel 2>/dev/null || true)
+rel_path="$file_path"
+if [ -n "$repo_root" ]; then
+ case "$file_path" in
+ "$repo_root"/*) rel_path="${file_path#"$repo_root"/}" ;;
+ esac
+fi
+
+# Cap the code sent as the semantic query. The embedder truncates at its own
+# token limit well before this, so a bigger slice buys no extra signal — and the
+# payload has to stay a GET (a read-scoped API key cannot POST, and every other
+# plugin hook works with a read key).
+q=$(printf '%s' "$code" | cut -c1-1200)
+
+path_enc=$(printf '%s' "$rel_path" | jq -rR '@uri' 2>/dev/null) || exit 0
+code_enc=$(printf '%s' "$q" | jq -rR '@uri' 2>/dev/null) || code_enc=""
+
+# Resolve the working repo's remote so the server can scope to the bound project.
+repo=$(git -C "$lookup_dir" remote get-url origin 2>/dev/null || true)
+repo_q=""
+if [ -n "$repo" ]; then
+ enc=$(printf '%s' "$repo" | jq -rR '@uri' 2>/dev/null) || enc=""
+ [ -n "$enc" ] && repo_q="&repo=${enc}"
+fi
+
+# Per-session dedup, in its own file rather than sharing auto-inject's. Each
+# surface shows a given snippet at most once per session, but they don't silence
+# each other: a title that flew past in a prompt menu twenty turns ago is
+# exactly what should reappear at the moment the duplicate is being written.
+state_dir="${TMPDIR:-/tmp}/scribe-priorart"
+mkdir -p "$state_dir" 2>/dev/null || true
+idfile=""
+exclude_q=""
+if [ -n "$session_id" ]; then
+ safe_sid=$(printf '%s' "$session_id" | tr -c 'A-Za-z0-9._-' '_')
+ idfile="$state_dir/${safe_sid}.ids"
+ if [ -f "$idfile" ]; then
+ seen=$(tr '\n' ',' < "$idfile" 2>/dev/null | sed 's/,$//')
+ [ -n "$seen" ] && exclude_q="&exclude_ids=${seen}"
+ fi
+fi
+
+body=$(curl -fsS --max-time 5 \
+ -H "Authorization: Bearer ${token}" \
+ "${url%/}/api/plugin/prior-art?path=${path_enc}&code=${code_enc}${repo_q}${exclude_q}" 2>/dev/null) || exit 0
+[ -n "$body" ] || exit 0
+
+context=$(printf '%s' "$body" | jq -r '.context // empty' 2>/dev/null) || exit 0
+[ -n "$context" ] || exit 0
+
+# Remember what was surfaced so it isn't shown again this session.
+if [ -n "$idfile" ]; then
+ printf '%s' "$body" | jq -r '.note_ids[]? // empty' 2>/dev/null >> "$idfile" || true
+fi
+
+# No permissionDecision: this is a nudge, not a gate. The write goes ahead.
+jq -n --arg c "$context" \
+ '{hookSpecificOutput: {hookEventName: "PreToolUse", additionalContext: $c}}'
+exit 0
diff --git a/plugin/skills/reusing-code/SKILL.md b/plugin/skills/reusing-code/SKILL.md
index afb097a..8767e94 100644
--- a/plugin/skills/reusing-code/SKILL.md
+++ b/plugin/skills/reusing-code/SKILL.md
@@ -20,10 +20,23 @@ through recall/auto-inject; this skill is the active reflex around that.
`list_snippets` searches every project by default; that's deliberate, since a
helper you need here was quite possibly written somewhere else. Narrow with
`project_id` only when you specifically want this project's own.
+- **About to edit an existing file? Ask by place, not just by meaning.**
+ `list_snippets(repo="…", path="…")` answers "what canonical helpers are already
+ recorded here?" — `path` matches the exact file or anything beneath it, so
+ `path="frontend/src"` covers the whole tree. Cheaper and sharper than a
+ wording search when you already know where the code is going, and it catches
+ the helper you'd otherwise duplicate a few lines down. `symbol="…"` narrows
+ further, and all three must match the same recorded location. Combine with `q`
+ to ask both at once.
- If a snippet fits, pull it in full with `get_snippet(id)` and reuse it — its
`location` points at the reference implementation. Adapt, don't re-derive.
- If auto-inject already surfaced a snippet title that looks relevant, that's
your cue to `get_snippet` it rather than start from scratch.
+- **Prior art offered beside a write is not noise — read it.** When Scribe notes
+ that a snippet is already recorded for the file you just wrote or edited, open
+ it before you go any further. Either it's the helper you were about to
+ duplicate — reuse it and drop yours — or it isn't, and the record needs the new
+ location adding. Both are cheaper now than after the duplicate settles in.
## The moment you build something reusable — record it
diff --git a/src/scribe/app.py b/src/scribe/app.py
index 1c67cc9..abd7c2a 100644
--- a/src/scribe/app.py
+++ b/src/scribe/app.py
@@ -170,6 +170,15 @@ def create_app() -> Quart:
await backfill_note_embeddings()
except Exception:
logger.warning("Embedding backfill failed", exc_info=True)
+ # Snippets written before migration 0070 have no `notes.data` mirror,
+ # and the location reverse lookup queries that column — an unfilled
+ # row would read as "no snippet here" rather than as a gap. Separate
+ # try block so neither backfill can skip the other.
+ try:
+ from scribe.services.snippets import backfill_snippet_data
+ await backfill_snippet_data()
+ except Exception:
+ logger.warning("Snippet data backfill failed", exc_info=True)
asyncio.create_task(_delayed_backfill())
diff --git a/src/scribe/mcp/tools/notes.py b/src/scribe/mcp/tools/notes.py
index d8a5b8a..58f696a 100644
--- a/src/scribe/mcp/tools/notes.py
+++ b/src/scribe/mcp/tools/notes.py
@@ -14,6 +14,7 @@ Sentinel conventions (inherited from existing fable-mcp tools):
from __future__ import annotations
from scribe.mcp._context import current_user_id
+from scribe.services import access as access_svc
from scribe.services import dedup as dedup_svc
from scribe.services import notes as notes_svc
from scribe.services import systems as systems_svc
@@ -57,12 +58,17 @@ async def get_note(note_id: int) -> dict:
"""Fetch the full content of a single Scribe note by its ID.
Returns id, title, body (markdown), tags, project_id, created_at, updated_at.
+ A note another user shared with you also carries `shared`, `owner` and
+ `permission` — read it as their suggestion, not as settled practice you set.
"""
uid = current_user_id()
- note = await notes_svc.get_note(uid, note_id)
- if note is None:
+ loaded = await notes_svc.get_note_for_user(uid, note_id)
+ note = loaded[0] if loaded else None
+ if note is None or note.deleted_at is not None:
raise ValueError(f"note {note_id} not found")
- return note.to_dict()
+ out = note.to_dict()
+ out.update(await access_svc.describe_provenance(uid, note))
+ return out
async def create_note(
diff --git a/src/scribe/mcp/tools/snippets.py b/src/scribe/mcp/tools/snippets.py
index f2a93fe..f0ec970 100644
--- a/src/scribe/mcp/tools/snippets.py
+++ b/src/scribe/mcp/tools/snippets.py
@@ -19,9 +19,16 @@ from scribe.services import systems as systems_svc
async def list_snippets(
q: str = "", tag: str = "", limit: int = 50, project_id: int = 0,
+ repo: str = "", path: str = "", symbol: str = "",
) -> dict:
"""List recorded snippets (reusable functions/components).
+ Two ways to ask, usable together: by MEANING (`q` — "what do I need this code
+ to do?") and by PLACE (`repo`/`path`/`symbol` — "what canonical helpers
+ already live in the file I'm about to edit?"). Reach for the place form
+ before you write or change code in a file: it is the cheap way to find prior
+ art you'd otherwise duplicate a few lines down.
+
Args:
q: Free-text search across name + body (optional). Matches on meaning as
well as wording, so describe what you need the code to DO.
@@ -30,6 +37,16 @@ async def list_snippets(
project_id: Narrow to one project. 0 (default) searches every project —
usually what you want, since a helper you need here may well have
been written somewhere else.
+ repo: Narrow to snippets recorded in this repo, matched exactly — the
+ same repo string used when recording, e.g. "Scribe".
+ path: Narrow to snippets recorded at this path. Matches the exact file
+ OR anything beneath it, so "frontend/src" finds
+ "frontend/src/lib/x.ts" as well as itself.
+ symbol: Narrow to snippets recorded under this symbol name, exactly.
+
+ `repo`/`path`/`symbol` must all match the SAME recorded location, so a
+ snippet that lives in repo A and, separately, at path B in another repo is
+ not returned for repo=A + path=B.
Returns {"snippets": [{id, title, tags, preview}], "total": int}. The title
reads "name — when to reach for it"; open one in full with get_snippet(id).
@@ -44,6 +61,7 @@ async def list_snippets(
items, total = await snippets_svc.list_snippets(
uid, q=q or None, tag=tag, limit=max(1, min(limit, 100)),
project_id=project_id or None,
+ repo=repo, path=path, symbol=symbol,
)
return {
"snippets": await access_svc.label_shared_items(uid, items),
@@ -243,6 +261,10 @@ async def merge_snippets(target_id: int, source_ids: list[int]) -> dict:
trash (recoverable). The survivor is re-embedded so recall stops surfacing the
now-merged duplicates.
+ The survivor records what it absorbed as `merged_from` (in its `snippet`
+ fields and as a "Merged from: #ids" line in the body), so a variant that got
+ folded in leaves a trace outside the trash. It accumulates across merges.
+
Args:
target_id: The snippet to keep (the canonical record).
source_ids: Snippet ids to fold into the target and retire. Ids that
diff --git a/src/scribe/mcp/tools/tasks.py b/src/scribe/mcp/tools/tasks.py
index 1cb6eef..143f670 100644
--- a/src/scribe/mcp/tools/tasks.py
+++ b/src/scribe/mcp/tools/tasks.py
@@ -19,6 +19,7 @@ Sentinels (preserved from existing fable-mcp):
from __future__ import annotations
from scribe.mcp._context import current_user_id
+from scribe.services import access as access_svc
from scribe.services import dedup as dedup_svc
from scribe.services import notes as notes_svc
from scribe.services import planning as planning_svc
@@ -68,17 +69,21 @@ async def get_task(task_id: int) -> dict:
kind=plan tasks, the response also includes applicable_rules +
subscribed_rulebooks from the task's project's rulebook subscriptions (new
plans are milestones — use get_milestone for those).
+
+ A task another user shared with you also carries `shared`, `owner` and
+ `permission` — it's their work item, not one you took on.
"""
uid = current_user_id()
- note = await notes_svc.get_note(uid, task_id)
- if note is None:
+ loaded = await notes_svc.get_note_for_user(uid, task_id)
+ note = loaded[0] if loaded else None
+ if note is None or note.deleted_at is not None:
raise ValueError(f"task {task_id} not found")
data = note.to_dict()
parent_title = None
if note.parent_id:
- parent = await notes_svc.get_note(uid, note.parent_id)
- if parent is not None:
- parent_title = parent.title
+ parent_loaded = await notes_svc.get_note_for_user(uid, note.parent_id)
+ if parent_loaded is not None:
+ parent_title = parent_loaded[0].title
data["parent_title"] = parent_title
# Legacy kind=plan tasks predate milestone-as-plan; still surface their
@@ -93,6 +98,7 @@ async def get_task(task_id: int) -> dict:
data["project_rules"] = applicable.get("project_rules", [])
data["suppressed_rules"] = applicable.get("suppressed_rules", [])
data["suppressed_topics"] = applicable.get("suppressed_topics", [])
+ data.update(await access_svc.describe_provenance(uid, note))
return data
diff --git a/src/scribe/models/note.py b/src/scribe/models/note.py
index 6682c6d..1b85701 100644
--- a/src/scribe/models/note.py
+++ b/src/scribe/models/note.py
@@ -69,6 +69,13 @@ class Note(Base, TimestampMixin, SoftDeleteMixin):
# notes keep the 'work' default and ignore it. Orthogonal to note_type
# (which is the note/entity axis).
task_kind: Mapped[str] = mapped_column(Text, default="work", server_default="work")
+ # Queryable structured fields for typed records — currently snippets, whose
+ # name/language/signature/locations live here so they can be INDEXED. The
+ # body keeps the same facts in readable markdown and remains what gets
+ # embedded; this is a mirror for querying, not the source of truth for
+ # display. NULL on every row written before migration 0070, so readers fall
+ # back to parsing the body (see services/snippets.snippet_fields).
+ data: Mapped[dict | None] = mapped_column(JSONB, nullable=True)
__table_args__ = (
Index("ix_notes_tags", "tags", postgresql_using="gin"),
@@ -79,6 +86,9 @@ class Note(Base, TimestampMixin, SoftDeleteMixin):
Index("ix_notes_milestone_id", "milestone_id"),
Index("ix_notes_note_type", "note_type"),
Index("ix_notes_arose_from_id", "arose_from_id"),
+ # Containment queries into `data` — e.g. which snippets name a given
+ # repo/path in their locations. See migration 0070.
+ Index("ix_notes_data_gin", "data", postgresql_using="gin"),
)
@property
diff --git a/src/scribe/routes/plugin.py b/src/scribe/routes/plugin.py
index 1dc4498..339a9ba 100644
--- a/src/scribe/routes/plugin.py
+++ b/src/scribe/routes/plugin.py
@@ -99,6 +99,53 @@ async def autoinject_retrieve():
return jsonify(result)
+@plugin_bp.get("/prior-art")
+@login_required
+async def write_path_prior_art():
+ """Prior-art hint for the plugin's PreToolUse hook on Write/Edit.
+
+ Answers "what is already recorded for the file about to be written?" — by
+ place (a snippet at this path or in its directory) and by meaning (snippets
+ resembling the code about to be written). Titles only, gated, and empty most
+ of the time. See services.plugin_context.build_write_path_hint.
+
+ Query:
+ path (str) — the target file, REPO-RELATIVE, matching the
+ convention snippet locations are recorded in. An
+ absolute path simply won't match anything.
+ code (optional) — the code about to be written; the semantic query.
+ Omit it and only the location arm runs.
+ repo (optional) — working repo remote; resolved to the bound project
+ to scope the search, exactly as /retrieve does. It
+ is NOT used as the location `repo` filter — see the
+ service docstring for why those two differ.
+ project_id (opt) — explicit project scope override (ad-hoc/testing).
+ exclude_ids (opt) — comma-separated ids already surfaced this session.
+ """
+ path = (request.args.get("path") or "").strip()
+ code = request.args.get("code") or ""
+ try:
+ project_id = int(request.args.get("project_id", 0) or 0)
+ except (TypeError, ValueError):
+ project_id = 0
+
+ repo = (request.args.get("repo") or "").strip()
+ if repo and not project_id:
+ resolved = await repo_bindings_svc.resolve_project(g.user.id, repo)
+ if resolved:
+ project_id = resolved
+
+ exclude_ids = [
+ int(p) for p in (request.args.get("exclude_ids") or "").split(",")
+ if p.strip().isdigit()
+ ]
+
+ result = await plugin_ctx_svc.build_write_path_hint(
+ g.user.id, path, code=code, project_id=project_id, exclude_ids=exclude_ids
+ )
+ return jsonify(result)
+
+
@plugin_bp.get("/processes")
@login_required
async def process_manifest():
diff --git a/src/scribe/routes/snippets.py b/src/scribe/routes/snippets.py
index 0074eb9..a7f5cc0 100644
--- a/src/scribe/routes/snippets.py
+++ b/src/scribe/routes/snippets.py
@@ -56,9 +56,15 @@ async def list_snippets_route():
project_id = int(request.args.get("project_id", 0) or 0) or None
except (TypeError, ValueError):
project_id = None
+ # Reverse lookup — "what already lives here?" Empty args are ignored, so the
+ # plain list is unchanged.
+ repo = request.args.get("repo", "")
+ path = request.args.get("path", "")
+ symbol = request.args.get("symbol", "")
limit, offset = parse_pagination()
items, total = await snippets_svc.list_snippets(
uid, q=q, tag=tag, limit=limit, offset=offset, project_id=project_id,
+ repo=repo, path=path, symbol=symbol,
)
# Mark rows owned by someone else so the UI can show whose they are — an
# unmarked row in your own list reads as one you recorded and vetted.
diff --git a/src/scribe/routes/tasks.py b/src/scribe/routes/tasks.py
index 9f0ee5f..c76f4f5 100644
--- a/src/scribe/routes/tasks.py
+++ b/src/scribe/routes/tasks.py
@@ -11,7 +11,6 @@ from scribe.services import systems as systems_svc
from scribe.services.embeddings import upsert_note_embedding
from scribe.services.notes import (
create_note,
- get_note,
get_note_for_user,
list_notes,
update_note,
@@ -187,8 +186,10 @@ async def get_task_route(task_id: int):
data = task.to_dict()
data["permission"] = permission
if task.parent_id:
- parent = await get_note(uid, task.parent_id)
- data["parent_title"] = parent.title if parent else None
+ # Share-aware like the task itself: a shared subtask whose parent is also
+ # shared would otherwise render as an orphan with no parent title.
+ parent = await get_note_for_user(uid, task.parent_id)
+ data["parent_title"] = parent[0].title if parent else None
data["systems"] = [s.to_dict() for s in await systems_svc.list_record_systems(uid, task_id)]
return jsonify(data)
diff --git a/src/scribe/services/embeddings.py b/src/scribe/services/embeddings.py
index b88e2a4..b870433 100644
--- a/src/scribe/services/embeddings.py
+++ b/src/scribe/services/embeddings.py
@@ -114,6 +114,7 @@ async def semantic_search_notes(
threshold: float = _SIMILARITY_THRESHOLD,
project_id: int | None = None,
is_task: bool | None = None,
+ note_type: str | None = None,
orphan_only: bool = False,
scope: str = "own",
) -> list[tuple[float, Note]]:
@@ -122,6 +123,9 @@ async def semantic_search_notes(
Scores are cosine similarities in [-1, 1]; only notes at or above
*threshold* are returned, sorted highest-first.
+ `note_type` narrows to a single record kind (e.g. "snippet"), for callers
+ that want prior art rather than everything embedded.
+
`scope` ("own" | "browse" | "read", see access.notes_visibility_clause)
decides how far this may see. It exists because this one function serves
three different kinds of act: an explicit search, which should reach
@@ -175,6 +179,11 @@ async def semantic_search_notes(
stmt = stmt.where(Note.status.isnot(None))
elif is_task is False:
stmt = stmt.where(Note.status.is_(None))
+ # Narrow to one kind of record. Composes with is_task rather than
+ # replacing it — 'snippet' is a non-task note_type, so a caller asking
+ # for prior art gets snippets and not the dev-log that mentions them.
+ if note_type:
+ stmt = stmt.where(Note.note_type == note_type)
if exclude_ids:
stmt = stmt.where(NoteEmbedding.note_id.notin_(exclude_ids))
stmt = stmt.where(distance <= max_distance).order_by(distance.asc()).limit(limit)
diff --git a/src/scribe/services/knowledge.py b/src/scribe/services/knowledge.py
index 71d987f..c3a5e4c 100644
--- a/src/scribe/services/knowledge.py
+++ b/src/scribe/services/knowledge.py
@@ -15,6 +15,7 @@ The asymmetry is the point. A record that appears unasked reads as one you
endorsed, so a one-off direct share has to be searched for rather than arriving
in your ambient lists.
"""
+import json
import logging
from sqlalchemy import func, select
@@ -28,6 +29,85 @@ logger = logging.getLogger(__name__)
_SNIPPET_LEN = 200
+# --- the location filter (reverse lookup, #2083) ------------------------------
+#
+# ONE predicate in two dialects: "this record carries a location matching every
+# part asked for." The SQL form runs in the browse arm and the keyword arm,
+# before the count and the page slice, so totals stay honest; the Python form
+# runs in the semantic arm, which post-filters candidates it already holds in
+# memory. THE TWO MUST CHANGE TOGETHER — a filter only one arm applies makes a
+# snippet findable by wording and invisible by location, which is exactly the
+# split tests/test_retrieval_scopes.py exists to prevent.
+#
+# Semantics, both dialects:
+# - parts are ANDed WITHIN a single location entry, never across the list. A
+# snippet whose first location is in repo A and whose second is at path B
+# does not answer repo=A + path=B — it was never in that place.
+# - `path` matches exactly OR as a directory prefix: "frontend/src" finds
+# "frontend/src/lib/x.ts". Prefix is the one part a GIN containment lookup
+# can't serve (migration 0070), which is why this is a jsonpath `@?` rather
+# than `@>` — jsonpath `starts with` is index-served by the same GIN index.
+#
+# The filter reads `notes.data`, so it only sees rows carrying that mirror.
+# Rows written before migration 0070 are populated once at startup by
+# snippets.backfill_snippet_data — without that, this query would answer "no
+# snippets here" for an old snippet and the caller would write the helper again.
+
+LOCATION_KEYS = ("repo", "path", "symbol")
+
+
+def location_parts(repo: str = "", path: str = "", symbol: str = "") -> dict[str, str]:
+ """The non-empty, stripped location parts asked for; {} when none were."""
+ given = {"repo": repo, "path": path, "symbol": symbol}
+ return {k: (v or "").strip() for k, v in given.items() if (v or "").strip()}
+
+
+def _path_matches(have: str, want: str) -> bool:
+ """Exact, or `have` sits somewhere under the `want` directory."""
+ return have == want or have.startswith(want.rstrip("/") + "/")
+
+
+def location_matches(data: dict | None, parts: dict[str, str]) -> bool:
+ """Python dialect of the location predicate. Keep in step with _location_clause."""
+ if not parts:
+ return True
+ for loc in (data or {}).get("locations") or []:
+ if all(
+ _path_matches((loc.get(key) or "").strip(), want)
+ if key == "path"
+ else (loc.get(key) or "").strip() == want
+ for key, want in parts.items()
+ ):
+ return True
+ return False
+
+
+def location_jsonpath(parts: dict[str, str]) -> str:
+ """The jsonpath behind the SQL dialect — one `locations` entry matching all parts.
+
+ Values are embedded as JSON string literals (jsonpath uses JSON quoting), so
+ a repo or path carrying a quote can't break out of the expression. The keys
+ are our own fixed set, never caller input.
+ """
+ filters = []
+ for key in LOCATION_KEYS:
+ if key not in parts:
+ continue
+ want = parts[key]
+ literal = json.dumps(want)
+ if key == "path":
+ prefix = json.dumps(want.rstrip("/") + "/")
+ filters.append(f"(@.path == {literal} || @.path starts with {prefix})")
+ else:
+ filters.append(f"@.{key} == {literal}")
+ return f"$.locations[*] ? ({' && '.join(filters)})"
+
+
+def _location_clause(parts: dict[str, str]):
+ """SQL dialect of the location predicate. Keep in step with location_matches."""
+ return Note.data.path_exists(location_jsonpath(parts))
+
+
def _note_to_item(note: Note) -> dict:
item: dict = {
"id": note.id,
@@ -80,18 +160,24 @@ async def query_knowledge(
limit: int,
offset: int,
project_id: int | None = None,
+ locations: dict[str, str] | None = None,
) -> tuple[list[dict], int]:
"""Query knowledge objects (non-task notes) with filters.
`project_id` narrows to one project (None = every project).
+ `locations` narrows to records whose `data.locations` holds an entry matching
+ every part given — build it with `location_parts(repo=…, path=…, symbol=…)`.
+ Today only snippets carry locations, but the column is general, so the filter
+ lives here with the query rather than in one type's service.
+
Returns (items, total_count).
"""
# Semantic search path — scores take priority over sort
if q:
return await _semantic_knowledge_search(
user_id, q, note_type=note_type, tags=tags, limit=limit,
- offset=offset, project_id=project_id,
+ offset=offset, project_id=project_id, locations=locations,
)
# No query = browsing. Narrower scope: a record shared directly with the
@@ -107,6 +193,9 @@ async def query_knowledge(
for tag in tags:
base = base.where(Note.tags.contains([tag]))
+ if locations:
+ base = base.where(_location_clause(locations))
+
# Count before pagination
count_stmt = select(func.count()).select_from(base.subquery())
total: int = (await session.execute(count_stmt)).scalar_one()
@@ -134,6 +223,7 @@ async def _semantic_knowledge_search(
limit: int,
offset: int,
project_id: int | None = None,
+ locations: dict[str, str] | None = None,
) -> tuple[list[dict], int]:
"""Hybrid search: keyword matches first (title/body ILIKE), then semantic results.
@@ -167,6 +257,8 @@ async def _semantic_knowledge_search(
base = base.where(Note.project_id == project_id)
for tag in tags:
base = base.where(Note.tags.contains([tag]))
+ if locations:
+ base = base.where(_location_clause(locations))
# Title matches first, then body-only matches, newest first within each
base = base.order_by(
Note.title.ilike(pattern).desc(),
@@ -204,6 +296,11 @@ async def _semantic_knowledge_search(
continue
if tags and not all(t in (note.tags or []) for t in tags):
continue
+ # The Python dialect of the same predicate the SQL arms apply above —
+ # these candidates arrive already fetched, so there's no query to
+ # narrow. See the comment on location_matches.
+ if locations and not location_matches(note.data, locations):
+ continue
semantic_notes.append(note)
except Exception:
logger.warning("Semantic search unavailable, using keyword results only", exc_info=True)
diff --git a/src/scribe/services/notes.py b/src/scribe/services/notes.py
index 3df0486..f36d08b 100644
--- a/src/scribe/services/notes.py
+++ b/src/scribe/services/notes.py
@@ -65,6 +65,7 @@ async def create_note(
note_type: str = "note",
task_kind: str = "work",
arose_from_id: int | None = None,
+ data: dict | None = None,
) -> Note:
# Validate status/priority here so the MCP create_task path (which passes
# them straight through) can't persist an out-of-enum value that the REST
@@ -108,6 +109,7 @@ async def create_note(
note_type=note_type,
task_kind=task_kind,
arose_from_id=arose_from_id,
+ data=data,
)
session.add(note)
await session.commit()
diff --git a/src/scribe/services/plugin_context.py b/src/scribe/services/plugin_context.py
index 7241838..3924337 100644
--- a/src/scribe/services/plugin_context.py
+++ b/src/scribe/services/plugin_context.py
@@ -14,6 +14,7 @@ index alone already steers behavior.
"""
from __future__ import annotations
+import logging
import re
import time
@@ -25,11 +26,14 @@ from scribe.services import knowledge as knowledge_svc
from scribe.services import notes as notes_svc
from scribe.services import projects as projects_svc
from scribe.services import rulebooks as rulebooks_svc
+from scribe.services import snippets as snippets_svc
from scribe.services.access import label_shared_items, owner_names_for
from scribe.services.embeddings import semantic_search_notes
from scribe.services.retrieval_telemetry import record_retrieval
from scribe.services.settings import get_setting
+logger = logging.getLogger(__name__)
+
# Defensive cap below Claude Code's 10k additionalContext limit.
_MAX_CHARS = 9000
@@ -51,6 +55,17 @@ AUTOINJECT_DEFAULT_ENABLED = True
AUTOINJECT_DEFAULT_THRESHOLD = 0.55
AUTOINJECT_DEFAULT_TOP_K = 3
+# The write-path trigger (#2082) gets its own on/off switch but SHARES the
+# threshold and top-k above. One knob for "how loud may Scribe be" is easier to
+# reason about than two that drift; the switch is separate because wanting prior
+# art on writes while keeping prompts quiet (or the reverse) is a real
+# preference, and it's the only part of the gate an operator can judge without
+# data. The two surfaces log under different `source` values, so if the
+# telemetry ever shows they want different thresholds, splitting them is a
+# data-backed change rather than a guess made up front.
+WRITEPATH_ENABLED_KEY = "kb_writepath_enabled"
+WRITEPATH_DEFAULT_ENABLED = True
+
# Margin gate: drop any hit more than this far below the top hit's score, so a
# single strong match doesn't drag in a wall of barely-passing neighbours.
_AUTOINJECT_BAND = 0.10
@@ -163,6 +178,22 @@ async def get_autoinject_config(user_id: int) -> dict:
return {"enabled": enabled, "threshold": threshold, "top_k": top_k}
+def _record_kind(note) -> str:
+ """The one-word kind marker for an injected menu line.
+
+ The menu is drawn from every record that carries an embedding, so a snippet,
+ a stored process, an issue and a stray dev-log all arrive looking identical.
+ Recorded prior art only stands out if the line says what it is — and the kind
+ is also what tells the reader which tool opens it.
+
+ Task-ness wins over `note_type` because it's the more useful distinction at a
+ glance: "there's an open issue about this" beats "there's a note about this".
+ """
+ if note.is_task:
+ return "issue" if note.task_kind == "issue" else "task"
+ return note.note_type or "note"
+
+
async def build_autoinject_hint(
user_id: int,
query: str,
@@ -175,7 +206,7 @@ async def build_autoinject_hint(
1. high-confidence threshold (stricter than pull) — set per-user;
2. margin gate — keep only hits within _AUTOINJECT_BAND of the top score;
3. session dedup — caller passes already-injected ids as `exclude_ids`;
- 4. title-first payload — id + title + score only, never bodies.
+ 4. title-first payload — id + kind + title + score only, never bodies.
Disabled, blank-query, or nothing-clears-the-gates all return empty context,
so most turns inject nothing.
@@ -222,15 +253,19 @@ async def build_autoinject_hint(
int(n.user_id) for _s, n in kept if n.user_id != user_id
})
+ # "records", not "notes" — the menu can hold snippets, processes and tasks
+ # too, and the kind marker on each line is only legible if the header doesn't
+ # already claim they're all one thing.
lines = [
- "> Possibly relevant from your Scribe notes — call `get_note(id)` to "
- "open any in full (titles only; injected once per session):",
+ "> Possibly relevant from your Scribe records — open any in full with "
+ "`get_note(id)`, or `get_snippet` / `get_process` for those kinds "
+ "(titles only; injected once per session):",
]
note_ids: list[int] = []
for score, note in kept:
note_ids.append(int(note.id))
title = (note.title or "(untitled)").replace("\n", " ").strip()
- line = f"> - #{note.id} \"{title}\" ({score:.2f})"
+ line = f"> - #{note.id} [{_record_kind(note)}] \"{title}\" ({score:.2f})"
if note.user_id != user_id:
who = owners.get(int(note.user_id)) or "another user"
line += f" — shared by {who}, treat as a suggestion"
@@ -239,6 +274,174 @@ async def build_autoinject_hint(
return {"context": "\n".join(lines), "note_ids": note_ids, "config": cfg}
+# --- Write-path trigger (#2082): prior art at the moment code is written ------
+# Auto-inject above fires on the operator's prompt. The moment reuse is actually
+# lost is later — when the AGENT decides mid-task to write a helper — and nothing
+# fired there. This is that trigger: the plugin's PreToolUse hook on Write/Edit
+# asks what prior art is already recorded for the file being written.
+#
+# Two arms, deliberately different in kind:
+# - BY PLACE — a snippet recorded at this path (or in its directory) is prior
+# art by definition, not by resemblance, so it isn't scored or thresholded.
+# This is what the reverse lookup (#2083) was built to answer.
+# - BY MEANING — semantic search over snippets only, using the code about to be
+# written, under the same gates as auto-inject.
+# Place beats meaning in the menu because "there is already a canonical helper in
+# this exact file" is a stronger claim than "this resembles something".
+
+
+def _prior_art_line(item: dict, marker: str, owner: str | None) -> str:
+ """One menu line: `- #12 [here] "title"`, attributed when it isn't yours."""
+ title = (item.get("title") or "(untitled)").replace("\n", " ").strip()
+ line = f"> - #{item['id']} [{marker}] \"{title}\""
+ if owner:
+ line += f" — shared by {owner}, treat as a suggestion"
+ return line
+
+
+async def get_writepath_config(user_id: int) -> dict:
+ """Write-path trigger settings: its own `enabled`, auto-inject's gates."""
+ cfg = await get_autoinject_config(user_id)
+ enabled_raw = await get_setting(
+ user_id, WRITEPATH_ENABLED_KEY,
+ "true" if WRITEPATH_DEFAULT_ENABLED else "false",
+ )
+ return {
+ **cfg,
+ "enabled": enabled_raw.strip().lower() in ("true", "1", "yes", "on"),
+ }
+
+
+async def build_write_path_hint(
+ user_id: int,
+ path: str,
+ code: str = "",
+ project_id: int = 0,
+ exclude_ids: list[int] | None = None,
+) -> dict:
+ """Prior-art hint for the plugin's PreToolUse hook on Write/Edit.
+
+ `path` is the file about to be written, REPO-RELATIVE — matching the
+ convention snippet locations are recorded in. `code` is what's about to be
+ written, used only as the semantic query.
+
+ Same four anti-bloat gates as auto-inject (threshold, margin, session dedup
+ via `exclude_ids`, titles-never-bodies) plus the shared top-k cap across BOTH
+ arms — so a file with a lot of recorded history can't turn one edit into a
+ wall of text. Returns empty context when disabled, when there's no path, or
+ when nothing is recorded — which is the common case, and the point.
+
+ Note the repo↔project mapping is deliberately one-way: the hook sends a git
+ remote, which the ROUTE resolves to `project_id` through the repo bindings.
+ It is never used as the location `repo` filter — a snippet's `repo` is a
+ free-text label the operator typed ("Scribe"), not a remote URL, and matching
+ one against the other would silently return nothing.
+
+ Returns {"context": str, "note_ids": list[int], "config": dict}. The semantic
+ arm is logged to retrieval_logs as source='write_path' — its own source, so
+ its precision is tunable separately from auto-inject's.
+
+ KNOWN GAP: only the semantic arm is logged, matching auto-inject's convention
+ of recording the candidate set for threshold tuning. Location hits carry no
+ score, so folding them in would corrupt the score distribution the log exists
+ to capture — but it does mean a snippet surfaced BY PLACE leaves no trace. The
+ usage signal (#2085) correlates retrieval_logs against later pulls, so it will
+ need a home for un-scored surfacing before it can measure this arm.
+ """
+ cfg = await get_writepath_config(user_id)
+ empty = {"context": "", "note_ids": [], "config": cfg}
+ path = (path or "").strip()
+ if not cfg["enabled"] or not path:
+ return empty
+
+ top_k = cfg["top_k"]
+ excluded = set(exclude_ids or [])
+ scope_project = project_id or None
+
+ # --- arm 1: by place ---
+ here: list[dict] = []
+ nearby: list[dict] = []
+ try:
+ here, _ = await snippets_svc.list_snippets(
+ user_id, path=path, limit=top_k, project_id=scope_project,
+ )
+ directory = path.rsplit("/", 1)[0] if "/" in path else ""
+ if directory and len(here) < top_k:
+ nearby, _ = await snippets_svc.list_snippets(
+ user_id, path=directory, limit=top_k, project_id=scope_project,
+ )
+ except Exception:
+ logger.warning("Write-path location lookup failed", exc_info=True)
+
+ seen: set[int] = set(excluded)
+ placed: list[tuple[str, dict]] = []
+ for marker, items in (("here", here), ("nearby", nearby)):
+ for item in items:
+ nid = int(item["id"])
+ if nid in seen:
+ continue
+ seen.add(nid)
+ placed.append((marker, item))
+
+ # --- arm 2: by meaning ---
+ scored: list[tuple[str, dict]] = []
+ remaining = top_k - len(placed)
+ query = (code or "").strip()
+ if remaining > 0 and query:
+ t0 = time.perf_counter()
+ hits = await semantic_search_notes(
+ user_id, query,
+ limit=remaining,
+ threshold=cfg["threshold"],
+ project_id=scope_project,
+ exclude_ids=seen,
+ note_type="snippet",
+ # Same reasoning as auto-inject: nobody asked for this, so it takes
+ # the browse scope and never surfaces a one-to-one direct share.
+ scope="browse",
+ )
+ record_retrieval(
+ user_id=user_id, source="write_path", query=query,
+ threshold=cfg["threshold"], limit=remaining,
+ project_id=scope_project, is_task=False, results=hits,
+ duration_ms=(time.perf_counter() - t0) * 1000.0,
+ )
+ if hits:
+ top_score = hits[0][0]
+ for score, note in hits:
+ if score < top_score - _AUTOINJECT_BAND:
+ continue
+ scored.append((
+ f"similar {score:.2f}",
+ {"id": int(note.id), "title": note.title, "user_id": note.user_id},
+ ))
+
+ menu = (placed + scored)[:top_k]
+ if not menu:
+ return empty
+
+ owners = await owner_names_for({
+ int(it["user_id"]) for _m, it in menu
+ if it.get("user_id") is not None and int(it["user_id"]) != user_id
+ })
+
+ lines = [
+ f"> Prior art already recorded in Scribe for `{path}` — open one with "
+ "`get_snippet(id)` and reuse it rather than writing a fresh one-off "
+ "(titles only; shown once per session):",
+ ]
+ note_ids: list[int] = []
+ for marker, item in menu:
+ note_ids.append(int(item["id"]))
+ owner_id = item.get("user_id")
+ owner = None
+ if owner_id is not None and int(owner_id) != user_id:
+ owner = owners.get(int(owner_id)) or "another user"
+ lines.append(_prior_art_line(item, marker, owner))
+
+ return {"context": "\n".join(lines), "note_ids": note_ids, "config": cfg}
+
+
async def _topic_titles(topic_ids: set[int]) -> dict[int, str]:
"""Map topic_id -> title for the given ids (live topics only)."""
if not topic_ids:
diff --git a/src/scribe/services/snippets.py b/src/scribe/services/snippets.py
index 0c4636e..30d1b4b 100644
--- a/src/scribe/services/snippets.py
+++ b/src/scribe/services/snippets.py
@@ -5,10 +5,10 @@ component recorded once so a later session can recall it before writing a
one-off. It carries a name, language, signature, canonical location
(repo · path · symbol), a one-line "when to reach for it", and the code itself.
-Structured fields are stored as a **body-convention** — no dedicated column, so
-a snippet inherits everything a note has (embeddings, ACL, project/System
-association, dedup) and, crucially, becomes eligible for semantic recall the
-moment it's embedded:
+Structured fields are stored as a **body-convention** — the body is the readable
+form and the thing that gets embedded, so a snippet inherits everything a note
+has (embeddings, ACL, project/System association, dedup) and, crucially, becomes
+eligible for semantic recall the moment it's embedded:
- ``title`` = ``"{name} — {when_to_use}"``. The title is exactly what the
title-first auto-inject surfaces, so this one line self-describes the snippet
@@ -17,18 +17,30 @@ moment it's embedded:
- ``body`` = templated markdown (When to use / Signature / Location, then a
fenced code block).
+Since migration 0070 the same fields are ALSO written to ``notes.data`` (see
+``compose_data``) — a queryable mirror, not a second source of truth: it carries
+no code, and it exists so location/language can be indexed rather than regexed
+out of every body. Reads prefer it and fall back to the body parse.
+
The public field API (name/language/signature/location/when_to_use/code) lives
-here, so the underlying storage could later move to a structured column without
-changing any caller (MCP tool, REST route, UI).
+here, so callers (MCP tool, REST route, UI) never see which of the two a field
+came from.
"""
from __future__ import annotations
import asyncio
+import logging
import re
+from sqlalchemy import func, or_, select
+
+from scribe.models import async_session
+from scribe.models.note import Note
from scribe.services import knowledge as knowledge_svc
from scribe.services import notes as notes_svc
+logger = logging.getLogger(__name__)
+
SNIPPET_NOTE_TYPE = "snippet"
SNIPPET_TAG = "snippet"
@@ -120,6 +132,23 @@ def _render_location_block(locations: list[dict]) -> str | None:
return f"**Locations:**\n{lines}"
+def _normalize_merged_from(ids: list[int] | None) -> list[int]:
+ """Merge provenance as a clean id list: ints only, no dups, order kept.
+
+ Order is history, not sorting — earlier merges stay first, so the list reads
+ as the sequence of things folded in.
+ """
+ out: list[int] = []
+ for raw in ids or []:
+ try:
+ i = int(raw)
+ except (TypeError, ValueError):
+ continue
+ if i > 0 and i not in out:
+ out.append(i)
+ return out
+
+
def compose_body(
*,
code: str,
@@ -130,6 +159,7 @@ def compose_body(
path: str = "",
symbol: str = "",
locations: list[dict] | None = None,
+ merged_from: list[int] | None = None,
) -> str:
"""Render structured fields into the snippet body markdown. Empty fields are
omitted so the body stays clean.
@@ -151,6 +181,14 @@ def compose_body(
loc_block = _render_location_block(locs)
if loc_block:
header.append(loc_block)
+ merged = _normalize_merged_from(merged_from)
+ if merged:
+ # Human-readable mirror of data["merged_from"]. Without it, a merge folds
+ # variants in and the record of what was absorbed lives only in the trash
+ # — recoverable only by someone who already knows to go looking.
+ header.append(
+ "**Merged from:** " + ", ".join(f"#{i}" for i in merged)
+ )
fence_lang = (language or "").strip().lower()
code_block = f"```{fence_lang}\n{(code or '').rstrip()}\n```"
if header:
@@ -166,7 +204,9 @@ _LOC_RE = re.compile(r"^\*\*Location:\*\*\s*(.+?)\s*$", re.MULTILINE)
_LOCS_RE = re.compile(
r"^\*\*Locations:\*\*[ \t]*\n((?:[ \t]*-[ \t]*.+\n?)+)", re.MULTILINE
)
+_MERGED_RE = re.compile(r"^\*\*Merged from:\*\*\s*(.+?)\s*$", re.MULTILINE)
_CODE_RE = re.compile(r"```([\w+.#-]*)\n(.*?)\n```", re.DOTALL)
+_ID_RE = re.compile(r"#(\d+)")
def _parse_location_str(s: str) -> dict | None:
@@ -202,6 +242,7 @@ def parse_snippet_fields(
"path": "",
"symbol": "",
"locations": [],
+ "merged_from": [],
"code": "",
}
@@ -235,6 +276,12 @@ def parse_snippet_fields(
fields["path"] = locations[0]["path"]
fields["symbol"] = locations[0]["symbol"]
+ m = _MERGED_RE.search(body)
+ if m:
+ fields["merged_from"] = _normalize_merged_from(
+ [int(i) for i in _ID_RE.findall(m.group(1))]
+ )
+
m = _CODE_RE.search(body)
if m:
fields["language"] = m.group(1).strip()
@@ -250,11 +297,149 @@ def parse_snippet_fields(
return fields
+# --- the queryable mirror (notes.data, migration 0070) -----------------------
+
+# Fields kept in `data`. Code is deliberately absent: the body already holds it,
+# and copying a blob into the column we index *around* would be pure weight.
+_DATA_FIELDS = (
+ "name", "when_to_use", "signature", "language", "locations", "merged_from",
+)
+
+
+def compose_data(
+ *,
+ name: str = "",
+ when_to_use: str = "",
+ signature: str = "",
+ language: str = "",
+ locations: list[dict] | None = None,
+ merged_from: list[int] | None = None,
+) -> dict:
+ """Build the `notes.data` mirror of a snippet's structured fields.
+
+ Same facts as the body convention, in a shape Postgres can index — so
+ "which snippets live in this path?" is a containment query rather than a
+ regex over every body. Empty values are omitted so the column stays sparse
+ and containment matches don't trip over blanks.
+ """
+ out: dict = {}
+ for key, value in (
+ ("name", (name or "").strip()),
+ ("when_to_use", (when_to_use or "").strip()),
+ ("signature", (signature or "").strip()),
+ ("language", (language or "").strip().lower()),
+ ):
+ if value:
+ out[key] = value
+ locs = _normalize_locations(locations)
+ if locs:
+ out["locations"] = locs
+ merged = _normalize_merged_from(merged_from)
+ if merged:
+ out["merged_from"] = merged
+ return out
+
+
+def snippet_fields(note) -> dict:
+ """Structured fields for a snippet, preferring the indexed `data` column and
+ falling back to parsing the body.
+
+ Both paths must agree, because rows written before migration 0070 have no
+ `data` and are never backfilled — a hand-edited body is the authority for
+ those, and there is no deadline by which they must be converted. `code` only
+ ever comes from the body, since `data` doesn't carry it.
+ """
+ parsed = parse_snippet_fields(note.title, note.body, note.tags)
+ stored = getattr(note, "data", None)
+ if not stored:
+ return parsed
+ merged = dict(parsed)
+ for key in _DATA_FIELDS:
+ if stored.get(key):
+ merged[key] = stored[key]
+ # Keep the single-location back-compat mirror consistent with whichever
+ # location list won.
+ locs = merged.get("locations") or []
+ merged["repo"] = locs[0]["repo"] if locs else ""
+ merged["path"] = locs[0]["path"] if locs else ""
+ merged["symbol"] = locs[0]["symbol"] if locs else ""
+ return merged
+
+
+async def backfill_snippet_data(*, batch: int = 500) -> int:
+ """Populate `notes.data` for snippet rows that predate migration 0070.
+
+ Runs once at startup (see app.py). Returns how many rows were filled.
+
+ Migration 0070 deliberately left `data` NULL on existing rows, because
+ *reading* a snippet never needed it — `snippet_fields` falls back to parsing
+ the body. Querying does: the location reverse lookup (#2083) is a jsonpath
+ over this column, so a NULL-`data` snippet would be reported as "nothing
+ recorded here" and the caller would write the helper again — silently worse
+ than having no reverse lookup at all. Every consumer of the column
+ (reverse lookup, and the write-path trigger built on it) can then assume the
+ mirror is present instead of carrying a second body-regex arm.
+
+ This does not reverse 0070's actual caution, which was about mangling a
+ hand-edited *body*: the body is never touched here, only the mirror derived
+ from it, by the same parser the read path already trusts. Idempotent — a
+ filled row is skipped forever after, and a snippet with no structured fields
+ at all settles at `{}` rather than staying NULL and being re-scanned. Trashed
+ rows are included so a later restore comes back queryable.
+
+ "Unfilled" means SQL NULL *or* JSON `null` — two different states in a JSONB
+ column, and only the first is what migration 0070 left behind. SQLAlchemy's
+ JSON types default to ``none_as_null=False``, so assigning Python ``None`` to
+ this column persists the JSON encoding of null rather than SQL NULL; an
+ ``IS NULL`` test alone walks straight past such a row and reports nothing to
+ do. Both mean "no usable mirror", so both are filled.
+ """
+ filled = 0
+ unfilled = or_(Note.data.is_(None), func.jsonb_typeof(Note.data) == "null")
+ async with async_session() as session:
+ while True:
+ rows = list(
+ (
+ await session.execute(
+ select(Note)
+ .where(Note.note_type == SNIPPET_NOTE_TYPE)
+ .where(unfilled)
+ .limit(batch)
+ )
+ )
+ .scalars()
+ .all()
+ )
+ if not rows:
+ break
+ for note in rows:
+ fields = parse_snippet_fields(note.title, note.body, note.tags)
+ note.data = compose_data(
+ name=fields["name"],
+ when_to_use=fields["when_to_use"],
+ signature=fields["signature"],
+ language=fields["language"],
+ locations=fields["locations"],
+ merged_from=fields["merged_from"],
+ )
+ await session.commit()
+ filled += len(rows)
+ if len(rows) < batch:
+ break
+ if filled:
+ logger.info("Snippet data backfill: populated `data` for %d snippet(s)", filled)
+ return filled
+
+
def snippet_to_dict(note) -> dict:
"""Note serialization plus a parsed ``snippet`` sub-object of structured
- fields, so callers get both the raw record and the typed view."""
+ fields, so callers get both the raw record and the typed view.
+
+ The raw `data` column is intentionally NOT exposed: it mirrors what
+ ``snippet`` already reports, and shipping both would give API consumers two
+ sources of truth for the same facts."""
data = note.to_dict()
- data["snippet"] = parse_snippet_fields(note.title, note.body, note.tags)
+ data["snippet"] = snippet_fields(note)
return data
@@ -278,17 +463,24 @@ async def create_snippet(
"""Create a snippet note (embedded on create for immediate recall). Returns
the created Note. Pass ``locations`` for the multi-location case; the single
``repo``/``path``/``symbol`` are the one-location shorthand."""
+ if locations is None:
+ locations = [{"repo": repo, "path": path, "symbol": symbol}]
note = await notes_svc.create_note(
user_id,
title=compose_title(name, when_to_use),
body=compose_body(
code=code, language=language, signature=signature,
- when_to_use=when_to_use, repo=repo, path=path, symbol=symbol,
- locations=locations,
+ when_to_use=when_to_use, locations=locations,
),
note_type=SNIPPET_NOTE_TYPE,
tags=compose_tags(language, tags),
project_id=project_id,
+ # The indexed mirror of the same fields (0070). Written together with the
+ # body so the two can never describe different things.
+ data=compose_data(
+ name=name, when_to_use=when_to_use, signature=signature,
+ language=language, locations=locations,
+ ),
)
_embed_snippet(note)
return note
@@ -318,12 +510,20 @@ async def list_snippets(
limit: int = 50,
offset: int = 0,
project_id: int | None = None,
+ repo: str = "",
+ path: str = "",
+ symbol: str = "",
) -> tuple[list[dict], int]:
"""List snippets (id/title/tags/preview dicts), most-recently-updated first.
``project_id`` narrows to one project; omit it to reach across every project
— which is the point when the thing you're about to write was already solved
- somewhere else."""
+ somewhere else.
+
+ ``repo`` / ``path`` / ``symbol`` are the reverse lookup: "what canonical
+ helpers already live here?" They narrow to snippets recorded at a matching
+ location, ANDed within one location entry, with ``path`` also matching as a
+ directory prefix. Combinable with ``q`` — search *and* place."""
return await knowledge_svc.query_knowledge(
user_id=user_id,
note_type=SNIPPET_NOTE_TYPE,
@@ -333,6 +533,8 @@ async def list_snippets(
limit=max(1, min(limit, 100)),
offset=max(0, offset),
project_id=project_id,
+ locations=knowledge_svc.location_parts(repo=repo, path=path, symbol=symbol)
+ or None,
)
@@ -379,7 +581,7 @@ async def update_snippet(
f"for edit access, or record your own version"
)
- cur = parse_snippet_fields(note.title, note.body, note.tags)
+ cur = snippet_fields(note)
overlay = {
"name": name, "code": code, "language": language,
"signature": signature, "when_to_use": when_to_use,
@@ -404,6 +606,17 @@ async def update_snippet(
code=merged["code"], language=merged["language"],
signature=merged["signature"], when_to_use=merged["when_to_use"],
locations=merged_locations,
+ # Carried, never set here: an ordinary edit must not erase the record
+ # of what was folded in, and only a merge may add to it.
+ merged_from=merged.get("merged_from"),
+ ),
+ # Re-derived from the same merged field set as the body, so an edit can't
+ # leave the indexed mirror describing the previous version.
+ "data": compose_data(
+ name=merged["name"], when_to_use=merged["when_to_use"],
+ signature=merged["signature"], language=merged["language"],
+ locations=merged_locations,
+ merged_from=merged.get("merged_from"),
),
}
# Recompute tags: keep any non-language, non-marker tags the note already had
@@ -485,8 +698,16 @@ async def merge_snippets(user_id: int, target_id: int, source_ids: list[int]):
cross-owner merge stays out of scope (#231) — and must itself be writable;
sources failing either test are skipped rather than silently half-merged.
+ The survivor records what it absorbed as `merged_from` — in the `data` mirror
+ and as a `**Merged from:** #ids` line in the body, written from one value like
+ every other field. It accumulates across merges and ordinary edits carry it
+ forward; only a merge adds to it.
+
Returns (merged_target_note, merged_source_ids), or None if the target isn't
- a snippet the caller can see."""
+ a snippet the caller can see. Note the returned ids are the sources actually
+ TRASHED, while `merged_from` is everything folded in — they differ only if a
+ source vanished between the field union and the trash call, which leaves the
+ source visible rather than losing anything."""
from scribe.services.access import can_write_note
target = await get_snippet(user_id, target_id)
if target is None:
@@ -511,21 +732,36 @@ async def merge_snippets(user_id: int, target_id: int, source_ids: list[int]):
continue
sources.append(s)
- tgt_fields = parse_snippet_fields(target.title, target.body, target.tags)
- parsed_sources = [
- (parse_snippet_fields(s.title, s.body, s.tags), s.tags) for s in sources
- ]
+ tgt_fields = snippet_fields(target)
+ parsed_sources = [(snippet_fields(s), s.tags) for s in sources]
locations, extra_tags = merge_snippet_fields(tgt_fields, target.tags, parsed_sources)
+ # Provenance: what this record absorbed, and when it absorbed it, in order.
+ # Merge keeps the target's scalar fields and trashes the sources, so without
+ # this the fact that a variant ever existed survives only in the trash — and
+ # only for someone who already knew to go looking. Accumulated, not replaced:
+ # a target merged twice keeps both histories.
+ merged_from = _normalize_merged_from(
+ list(tgt_fields.get("merged_from") or []) + [s.id for s in sources]
+ )
+
# Owner-scoped write, authorised above — same reason as update_snippet.
updated = await notes_svc.update_note(
owner_id, target_id,
body=compose_body(
code=tgt_fields["code"], language=tgt_fields["language"],
signature=tgt_fields["signature"], when_to_use=tgt_fields["when_to_use"],
- locations=locations,
+ locations=locations, merged_from=merged_from,
),
tags=compose_tags(tgt_fields["language"], extra_tags),
+ # The survivor's location set grew, so its mirror has to grow with it —
+ # otherwise a merged snippet would be unfindable at the very call sites
+ # the merge just recorded.
+ data=compose_data(
+ name=tgt_fields["name"], when_to_use=tgt_fields["when_to_use"],
+ signature=tgt_fields["signature"], language=tgt_fields["language"],
+ locations=locations, merged_from=merged_from,
+ ),
)
if updated is None:
return None
diff --git a/tests/test_integration_snippet_locations.py b/tests/test_integration_snippet_locations.py
new file mode 100644
index 0000000..d430538
--- /dev/null
+++ b/tests/test_integration_snippet_locations.py
@@ -0,0 +1,316 @@
+"""Real-Postgres integration test for the snippet location reverse lookup (#2083).
+
+Runs only in the CI integration lane (real Postgres, schema built by
+`alembic upgrade head`, which includes migration 0070's `notes.data` + GIN index).
+This exercises what the unit mocks cannot:
+
+ - the jsonpath `@?` containment behind `_location_clause` — including
+ `starts with` for a path prefix, and the fact that parts must match within a
+ SINGLE `locations` entry;
+ - that the SQL dialect and the Python dialect (`location_matches`, used by the
+ semantic arm) return the SAME rows. Those two must never drift, or a snippet
+ is findable by wording and invisible by location;
+ - `backfill_snippet_data`, which is what stops a pre-0070 snippet from being
+ silently reported as "nothing recorded here."
+"""
+from unittest.mock import AsyncMock, patch
+
+import pytest
+import pytest_asyncio
+from sqlalchemy import delete, func, select
+
+from scribe.models import async_session, engine
+from scribe.models.note import Note
+from scribe.models.user import User
+from scribe.services.knowledge import location_matches, location_parts
+from scribe.services.snippets import (
+ SNIPPET_NOTE_TYPE,
+ backfill_snippet_data,
+ compose_body,
+ compose_data,
+ list_snippets,
+ snippet_fields,
+)
+
+pytestmark = pytest.mark.integration
+
+
+@pytest_asyncio.fixture(autouse=True)
+async def _dispose_engine():
+ # Per-loop pool: dispose after each test (see test_integration_db_maintenance).
+ yield
+ await engine.dispose()
+
+
+def _loc(repo="", path="", symbol=""):
+ return {"repo": repo, "path": path, "symbol": symbol}
+
+
+@pytest_asyncio.fixture
+async def seeded():
+ """A user with four snippets covering the cases the filter has to separate.
+
+ Returns (user_id, {label: note_id}).
+ """
+ async with async_session() as s:
+ user = User(username="snippet_loc_itest")
+ s.add(user)
+ await s.flush()
+
+ def _snippet(name, locations, code="return 1"):
+ return Note(
+ user_id=user.id,
+ note_type=SNIPPET_NOTE_TYPE,
+ title=f"{name} — does a thing",
+ body=compose_body(code=code, language="python", locations=locations),
+ tags=["python", "snippet"],
+ data=compose_data(name=name, language="python", locations=locations),
+ )
+
+ nested = _snippet("nested", [_loc("Scribe", "frontend/src/lib/x.ts", "helper")])
+ sibling = _snippet("sibling", [_loc("Scribe", "frontend/srcmap.ts", "other")])
+ # Two locations, deliberately crossed: repo Scribe at src/a.py and repo
+ # Portal at src/b.py. repo=Scribe + path=src/b.py must NOT match it.
+ multi = _snippet(
+ "multi",
+ [_loc("Scribe", "src/a.py", "alpha"), _loc("Portal", "src/b.py", "beta")],
+ )
+ # No structured location at all — must never satisfy a location filter,
+ # and must not error the query either.
+ placeless = _snippet("placeless", [])
+ # A plain note (not a snippet) with no `data` — proves the jsonpath
+ # operator tolerates NULL rather than raising.
+ plain = Note(user_id=user.id, title="plain", body="no data here")
+
+ s.add_all([nested, sibling, multi, placeless, plain])
+ await s.commit()
+ ids = (
+ user.id,
+ {
+ "nested": nested.id,
+ "sibling": sibling.id,
+ "multi": multi.id,
+ "placeless": placeless.id,
+ "plain": plain.id,
+ },
+ )
+ yield ids
+ user_id = ids[0]
+ async with async_session() as s:
+ await s.execute(delete(Note).where(Note.user_id == user_id))
+ await s.execute(delete(User).where(User.id == user_id))
+ await s.commit()
+
+
+async def _ids_for(user_id: int, **parts) -> set[int]:
+ items, total = await list_snippets(user_id, **parts)
+ assert total == len(items), "total must match the filtered page, not the whole list"
+ return {i["id"] for i in items}
+
+
+@pytest.mark.asyncio
+async def test_repo_filter_finds_every_snippet_recorded_in_it(seeded):
+ user_id, ids = seeded
+ found = await _ids_for(user_id, repo="Scribe")
+ assert found == {ids["nested"], ids["sibling"], ids["multi"]}
+ assert ids["placeless"] not in found
+
+
+@pytest.mark.asyncio
+async def test_exact_path_finds_only_that_file(seeded):
+ user_id, ids = seeded
+ assert await _ids_for(user_id, path="frontend/src/lib/x.ts") == {ids["nested"]}
+
+
+@pytest.mark.asyncio
+async def test_path_prefix_reaches_into_subdirectories(seeded):
+ user_id, ids = seeded
+ # ...and stops at the directory boundary: srcmap.ts is a sibling, not a child.
+ assert await _ids_for(user_id, path="frontend/src") == {ids["nested"]}
+ assert await _ids_for(user_id, path="frontend/src/") == {ids["nested"]}
+ assert await _ids_for(user_id, path="frontend") == {ids["nested"], ids["sibling"]}
+
+
+@pytest.mark.asyncio
+async def test_parts_must_match_within_one_location(seeded):
+ user_id, ids = seeded
+ assert await _ids_for(user_id, repo="Scribe", path="src/a.py") == {ids["multi"]}
+ # Same snippet, but repo and path belong to two different entries.
+ assert await _ids_for(user_id, repo="Scribe", path="src/b.py") == set()
+ assert await _ids_for(user_id, repo="Portal", path="src/b.py") == {ids["multi"]}
+
+
+@pytest.mark.asyncio
+async def test_symbol_filter_and_an_unrecorded_location(seeded):
+ user_id, ids = seeded
+ assert await _ids_for(user_id, symbol="helper") == {ids["nested"]}
+ assert await _ids_for(user_id, repo="NoSuchRepo") == set()
+
+
+@pytest.mark.asyncio
+async def test_no_location_filter_still_lists_everything(seeded):
+ """The filter is opt-in — a plain browse must not lose the placeless snippet."""
+ found = await _ids_for(user_id=seeded[0])
+ ids = seeded[1]
+ assert {ids["nested"], ids["sibling"], ids["multi"], ids["placeless"]} <= found
+ assert ids["plain"] not in found # not a snippet
+
+
+@pytest.mark.asyncio
+async def test_keyword_arm_applies_the_location_filter(seeded):
+ """Search AND place compose. The embedder is stubbed out so this isolates the
+ keyword (ILIKE) arm, which is the second place the SQL clause has to appear."""
+ user_id, ids = seeded
+ with patch(
+ "scribe.services.embeddings.semantic_search_notes",
+ AsyncMock(return_value=[]),
+ ):
+ assert await _ids_for(user_id, q="nested", repo="Scribe") == {ids["nested"]}
+ # Matches the query but not the place.
+ assert await _ids_for(user_id, q="nested", repo="Portal") == set()
+
+
+@pytest.mark.asyncio
+async def test_semantic_arm_applies_the_location_filter(seeded):
+ """The semantic arm holds its candidates in memory, so it filters in Python.
+ A query that matches no title or body leaves only that arm in play."""
+ user_id, ids = seeded
+ async with async_session() as s:
+ rows = list(
+ (
+ await s.execute(
+ select(Note)
+ .where(Note.user_id == user_id)
+ .where(Note.note_type == SNIPPET_NOTE_TYPE)
+ )
+ )
+ .scalars()
+ .all()
+ )
+ candidates = [(0.9, n) for n in rows]
+
+ with patch(
+ "scribe.services.embeddings.semantic_search_notes",
+ AsyncMock(return_value=candidates),
+ ):
+ # No keyword can match this, so every hit below came from the stub.
+ assert await _ids_for(user_id, q="zzzznomatch") == {
+ ids["nested"], ids["sibling"], ids["multi"], ids["placeless"],
+ }
+ assert await _ids_for(user_id, q="zzzznomatch", path="frontend/src") == {
+ ids["nested"],
+ }
+ assert await _ids_for(user_id, q="zzzznomatch", repo="Scribe", path="src/b.py") == set()
+
+
+@pytest.mark.asyncio
+async def test_sql_and_python_dialects_agree_on_the_same_rows(seeded):
+ """The drift guard: whatever the SQL arm returns, the Python arm must too."""
+ user_id, _ids = seeded
+ async with async_session() as s:
+ rows = list(
+ (
+ await s.execute(
+ select(Note)
+ .where(Note.user_id == user_id)
+ .where(Note.note_type == SNIPPET_NOTE_TYPE)
+ )
+ )
+ .scalars()
+ .all()
+ )
+ by_id = {n.id: n.data for n in rows}
+
+ cases = [
+ {"repo": "Scribe"},
+ {"path": "frontend/src"},
+ {"path": "frontend/src/lib/x.ts"},
+ {"repo": "Scribe", "path": "src/a.py"},
+ {"repo": "Scribe", "path": "src/b.py"},
+ {"symbol": "beta"},
+ {"repo": "Scribe", "path": "frontend", "symbol": "helper"},
+ ]
+ for case in cases:
+ sql_ids = await _ids_for(user_id, **case)
+ parts = location_parts(**case)
+ python_ids = {nid for nid, data in by_id.items() if location_matches(data, parts)}
+ assert sql_ids == python_ids, f"dialects disagree on {case}"
+
+
+@pytest.mark.asyncio
+async def test_backfill_makes_a_pre_0070_snippet_findable_by_location(seeded):
+ """A snippet stored before migration 0070 has no `data`, so the location query
+ can't see it. Unfilled, it reads as "nothing recorded here" — the failure the
+ backfill exists to prevent.
+
+ `data` is OMITTED from the constructor on purpose: that leaves a genuine SQL
+ NULL, which is the shape migration 0070 actually left behind. Passing
+ `data=None` would store the JSON encoding of null instead (SQLAlchemy JSON
+ defaults to none_as_null=False) — a different state, covered by the next test.
+ """
+ user_id, _ids = seeded
+ locations = [_loc("Legacy", "old/path/y.py", "legacy_helper")]
+ async with async_session() as s:
+ old = Note(
+ user_id=user_id,
+ note_type=SNIPPET_NOTE_TYPE,
+ title="legacy — recorded before the data column existed",
+ body=compose_body(code="return 2", language="python", locations=locations),
+ tags=["python", "snippet"],
+ )
+ s.add(old)
+ await s.commit()
+ old_id = old.id
+
+ assert await _ids_for(user_id, repo="Legacy") == set()
+
+ filled = await backfill_snippet_data()
+ assert filled >= 1
+
+ assert await _ids_for(user_id, repo="Legacy") == {old_id}
+ assert await _ids_for(user_id, path="old/path") == {old_id}
+
+ # The mirror agrees with the body it was derived from, and the body is untouched.
+ async with async_session() as s:
+ row = (await s.execute(select(Note).where(Note.id == old_id))).scalar_one()
+ assert row.data["locations"] == locations
+ assert snippet_fields(row)["symbol"] == "legacy_helper"
+ assert "old/path/y.py" in row.body
+
+
+@pytest.mark.asyncio
+async def test_backfill_fills_a_json_null_and_is_idempotent(seeded):
+ """The OTHER unfilled shape: `data=None` persists JSON `null`, not SQL NULL
+ (SQLAlchemy JSON defaults to none_as_null=False), so an `IS NULL` test alone
+ would walk past this row. Also pins idempotence — a fieldless snippet settles
+ at `{}` instead of staying unfilled and being rescanned on every boot."""
+ user_id, _ids = seeded
+ async with async_session() as s:
+ bare = Note(
+ user_id=user_id,
+ note_type=SNIPPET_NOTE_TYPE,
+ title="",
+ body="",
+ tags=[],
+ data=None,
+ )
+ s.add(bare)
+ await s.commit()
+ bare_id = bare.id
+
+ # Confirm the premise rather than assuming it: this row really is JSON null,
+ # so the assertion below is testing the second arm of the predicate.
+ async with async_session() as s:
+ kind = (
+ await s.execute(
+ select(func.jsonb_typeof(Note.data)).where(Note.id == bare_id)
+ )
+ ).scalar_one()
+ assert kind == "null"
+
+ assert await backfill_snippet_data() >= 1
+ assert await backfill_snippet_data() == 0
+
+ async with async_session() as s:
+ row = (await s.execute(select(Note).where(Note.id == bare_id))).scalar_one()
+ assert row.data == {}
diff --git a/tests/test_mcp_tool_notes.py b/tests/test_mcp_tool_notes.py
index ccb5144..6836f20 100644
--- a/tests/test_mcp_tool_notes.py
+++ b/tests/test_mcp_tool_notes.py
@@ -17,11 +17,18 @@ def _bind_user():
_user_id_ctx.reset(token)
-def _fake_note(**overrides) -> MagicMock:
+def _fake_note(*, user_id: int = 7, **overrides) -> MagicMock:
note = MagicMock()
base = {"id": 1, "title": "t", "body": "b", "tags": [], "is_task": False}
base.update(overrides)
note.to_dict.return_value = base
+ # Real values, not auto-attributes: get_note reads deleted_at, and compares
+ # user_id against the bound caller to decide whether to attach a shared/owner
+ # marker. A MagicMock is truthy on both, so it would read as another user's
+ # trashed note and reach for the DB (note 2109).
+ note.id = base["id"]
+ note.user_id = user_id
+ note.deleted_at = None
return note
@@ -108,24 +115,61 @@ async def test_list_notes_limit_clamped():
async def test_get_note_returns_dict():
fake = _fake_note(id=5, title="found")
with patch(
- "scribe.mcp.tools.notes.notes_svc.get_note",
- AsyncMock(return_value=fake),
+ "scribe.mcp.tools.notes.notes_svc.get_note_for_user",
+ AsyncMock(return_value=(fake, "owner")),
):
out = await get_note(note_id=5)
assert out["id"] == 5
assert out["title"] == "found"
+ # Own record: no provenance noise.
+ assert "shared" not in out
+
+
+@pytest.mark.asyncio
+async def test_get_note_opens_a_shared_record_and_says_whose_it_is():
+ """The injected menu tells the agent to open any hit with get_note(id), and
+ that menu can list a collaborator's note reached through a shared project.
+ An owner-only fetch would answer "not found" for a record the agent was just
+ handed — and the web UI opens the same note fine."""
+ theirs = _fake_note(id=5, title="Their note", user_id=9)
+ with patch(
+ "scribe.mcp.tools.notes.notes_svc.get_note_for_user",
+ AsyncMock(return_value=(theirs, "viewer")),
+ ), patch(
+ "scribe.services.access.describe_provenance",
+ AsyncMock(return_value={"shared": True, "owner": "alex",
+ "permission": "viewer"}),
+ ):
+ out = await get_note(note_id=5)
+ assert out["shared"] is True
+ assert out["owner"] == "alex"
+ assert out["permission"] == "viewer"
@pytest.mark.asyncio
async def test_get_note_raises_when_not_found():
with patch(
- "scribe.mcp.tools.notes.notes_svc.get_note",
+ "scribe.mcp.tools.notes.notes_svc.get_note_for_user",
AsyncMock(return_value=None),
):
with pytest.raises(ValueError, match="note 999 not found"):
await get_note(note_id=999)
+@pytest.mark.asyncio
+async def test_get_note_treats_a_trashed_note_as_missing():
+ """get_note_for_user resolves permission, not liveness — the trash filter is
+ the caller's to apply."""
+ trashed = _fake_note(id=5)
+ trashed.deleted_at = "2026-07-01T00:00:00Z"
+ with patch(
+ "scribe.mcp.tools.notes.notes_svc.get_note_for_user",
+ AsyncMock(return_value=(trashed, "owner")),
+ ):
+ with pytest.raises(ValueError, match="note 5 not found"):
+ await get_note(note_id=5)
+
+
@pytest.mark.asyncio
async def test_create_note_passes_through():
fake = _fake_note(id=10, title="new")
diff --git a/tests/test_mcp_tool_planning.py b/tests/test_mcp_tool_planning.py
index d079e5d..e4d3ba3 100644
--- a/tests/test_mcp_tool_planning.py
+++ b/tests/test_mcp_tool_planning.py
@@ -24,16 +24,25 @@ async def test_start_planning_tool_delegates_to_service():
assert mock.call_args.kwargs == {"user_id": 7, "project_id": 3, "title": "Plan it"}
-@pytest.mark.asyncio
-async def test_get_task_augments_plan_with_rules():
+def _plan_note(task_kind: str):
note = MagicMock()
note.parent_id = None
note.project_id = 3
- note.to_dict.return_value = {"id": 9, "task_kind": "plan", "project_id": 3}
+ note.id = 9
+ # Real values — get_task reads deleted_at and compares user_id to the bound
+ # caller, and a MagicMock is truthy on both (note 2109).
+ note.user_id = 7
+ note.deleted_at = None
+ note.to_dict.return_value = {"id": 9, "task_kind": task_kind, "project_id": 3}
+ return note
+
+
+@pytest.mark.asyncio
+async def test_get_task_augments_plan_with_rules():
applicable = {"rules": [{"id": 1, "title": "r"}], "truncated": False,
"subscribed_rulebooks": [{"id": 2, "title": "rb"}]}
- with patch("scribe.mcp.tools.tasks.notes_svc.get_note",
- AsyncMock(return_value=note)), \
+ with patch("scribe.mcp.tools.tasks.notes_svc.get_note_for_user",
+ AsyncMock(return_value=(_plan_note("plan"), "owner"))), \
patch("scribe.mcp.tools.tasks.rulebooks_svc.get_applicable_rules",
AsyncMock(return_value=applicable)):
from scribe.mcp.tools.tasks import get_task
@@ -45,12 +54,8 @@ async def test_get_task_augments_plan_with_rules():
@pytest.mark.asyncio
async def test_get_task_work_kind_has_no_rules():
- note = MagicMock()
- note.parent_id = None
- note.project_id = 3
- note.to_dict.return_value = {"id": 9, "task_kind": "work", "project_id": 3}
- with patch("scribe.mcp.tools.tasks.notes_svc.get_note",
- AsyncMock(return_value=note)), \
+ with patch("scribe.mcp.tools.tasks.notes_svc.get_note_for_user",
+ AsyncMock(return_value=(_plan_note("work"), "owner"))), \
patch("scribe.mcp.tools.tasks.rulebooks_svc.get_applicable_rules",
AsyncMock()) as mock_rules:
from scribe.mcp.tools.tasks import get_task
diff --git a/tests/test_mcp_tool_snippets.py b/tests/test_mcp_tool_snippets.py
index adadc80..4c10f95 100644
--- a/tests/test_mcp_tool_snippets.py
+++ b/tests/test_mcp_tool_snippets.py
@@ -24,6 +24,10 @@ def _fake_snippet(user_id: int = 7):
# decide whether to attach a shared/owner marker; an auto-MagicMock would read
# as another user's record and send them off to look up a username.
n.user_id = user_id
+ # Explicitly None, not an auto-attribute: snippet_fields prefers `data` when
+ # truthy, and a MagicMock is truthy — every parsed field would come back as a
+ # MagicMock instead of a string.
+ n.data = None
n.to_dict.return_value = {
"id": 1, "title": n.title, "note_type": "snippet", "tags": n.tags,
}
diff --git a/tests/test_mcp_tool_tasks.py b/tests/test_mcp_tool_tasks.py
index 3841eb7..30d62d6 100644
--- a/tests/test_mcp_tool_tasks.py
+++ b/tests/test_mcp_tool_tasks.py
@@ -18,7 +18,8 @@ def _bind_user():
_user_id_ctx.reset(token)
-def _fake_task(*, parent_id: int | None = None, **overrides) -> MagicMock:
+def _fake_task(*, parent_id: int | None = None, user_id: int = 7,
+ **overrides) -> MagicMock:
n = MagicMock()
n.parent_id = parent_id
base = {
@@ -29,6 +30,12 @@ def _fake_task(*, parent_id: int | None = None, **overrides) -> MagicMock:
base.update(overrides)
n.to_dict.return_value = base
n.title = base["title"]
+ n.id = base["id"]
+ # Real values, not auto-attributes: get_task reads deleted_at and compares
+ # user_id against the bound caller for the shared/owner marker — a MagicMock
+ # is truthy on both (note 2109).
+ n.user_id = user_id
+ n.deleted_at = None
return n
@@ -63,12 +70,13 @@ async def test_list_tasks_empty_status_means_no_filter():
async def test_get_task_with_no_parent_returns_null_parent_title():
fake = _fake_task(id=5, title="solo", parent_id=None)
with patch(
- "scribe.mcp.tools.tasks.notes_svc.get_note",
- AsyncMock(return_value=fake),
+ "scribe.mcp.tools.tasks.notes_svc.get_note_for_user",
+ AsyncMock(return_value=(fake, "owner")),
):
out = await get_task(task_id=5)
assert out["parent_title"] is None
assert out["title"] == "solo"
+ assert "shared" not in out
@pytest.mark.asyncio
@@ -76,9 +84,9 @@ async def test_get_task_enriches_with_parent_title():
"""When parent_id is set, get_task fetches the parent and adds parent_title."""
child = _fake_task(id=10, title="child", parent_id=5)
parent = _fake_task(id=5, title="parent of 10", parent_id=None)
- # get_note is called twice: once for child, once for parent
- mock_get = AsyncMock(side_effect=[child, parent])
- with patch("scribe.mcp.tools.tasks.notes_svc.get_note", mock_get):
+ # fetched twice: once for the child, once for the parent
+ mock_get = AsyncMock(side_effect=[(child, "owner"), (parent, "owner")])
+ with patch("scribe.mcp.tools.tasks.notes_svc.get_note_for_user", mock_get):
out = await get_task(task_id=10)
assert out["parent_title"] == "parent of 10"
assert mock_get.await_count == 2
@@ -88,16 +96,34 @@ async def test_get_task_enriches_with_parent_title():
async def test_get_task_parent_missing_returns_null():
"""If parent_id is set but the parent is gone (orphaned), parent_title is None."""
child = _fake_task(id=10, parent_id=5)
- mock_get = AsyncMock(side_effect=[child, None])
- with patch("scribe.mcp.tools.tasks.notes_svc.get_note", mock_get):
+ mock_get = AsyncMock(side_effect=[(child, "owner"), None])
+ with patch("scribe.mcp.tools.tasks.notes_svc.get_note_for_user", mock_get):
out = await get_task(task_id=10)
assert out["parent_title"] is None
+@pytest.mark.asyncio
+async def test_get_task_opens_a_shared_task_and_says_whose_it_is():
+ """A task in a shared project opens in the web UI, so the agent path must not
+ answer "not found" for the same id — and must say it isn't the caller's."""
+ theirs = _fake_task(id=5, title="Their task", user_id=9)
+ with patch(
+ "scribe.mcp.tools.tasks.notes_svc.get_note_for_user",
+ AsyncMock(return_value=(theirs, "viewer")),
+ ), patch(
+ "scribe.services.access.describe_provenance",
+ AsyncMock(return_value={"shared": True, "owner": "alex",
+ "permission": "viewer"}),
+ ):
+ out = await get_task(task_id=5)
+ assert out["shared"] is True
+ assert out["owner"] == "alex"
+
+
@pytest.mark.asyncio
async def test_get_task_raises_when_not_found():
with patch(
- "scribe.mcp.tools.tasks.notes_svc.get_note",
+ "scribe.mcp.tools.tasks.notes_svc.get_note_for_user",
AsyncMock(return_value=None),
):
with pytest.raises(ValueError, match="task 999 not found"):
diff --git a/tests/test_retrieval_scopes.py b/tests/test_retrieval_scopes.py
index 5eeaf29..478edce 100644
--- a/tests/test_retrieval_scopes.py
+++ b/tests/test_retrieval_scopes.py
@@ -17,15 +17,19 @@ from unittest.mock import AsyncMock, MagicMock, patch
import pytest
-def _note(id=1, user_id=7, title="A note"):
+def _note(id=1, user_id=7, title="A note", note_type="note",
+ is_task=False, task_kind="work"):
n = MagicMock()
n.id = id
n.user_id = user_id
n.title = title
n.body = "body"
n.tags = []
- n.is_task = False
- n.note_type = "note"
+ # Real values, not auto-attributes: the menu reads these to label each line,
+ # and a MagicMock is truthy — every record would render as a task (note 2109).
+ n.is_task = is_task
+ n.task_kind = task_kind
+ n.note_type = note_type
return n
@@ -146,3 +150,39 @@ async def test_injected_menu_attributes_another_users_note():
assert "suggestion" in theirs_line
# The operator's own note stays unadorned — absence of a marker is the signal.
assert "shared by" not in mine_line
+
+
+@pytest.mark.asyncio
+async def test_injected_menu_labels_the_record_kind():
+ """Every injected line renders the same shape, so without a kind marker a
+ recorded snippet is indistinguishable from a stray dev-log — exactly where
+ prior art most needs to stand out."""
+ from scribe.services import plugin_context
+
+ hits = [
+ (0.92, _note(1, title="debounce — rate-limit a callback", note_type="snippet")),
+ (0.91, _note(2, title="Release checklist", note_type="process")),
+ (0.90, _note(3, title="Auth token expiry", is_task=True, task_kind="issue")),
+ (0.89, _note(4, title="Ship the drafter", is_task=True)),
+ (0.88, _note(5, title="Why we dropped CalDAV")),
+ ]
+ with patch.object(plugin_context, "semantic_search_notes",
+ AsyncMock(return_value=hits)), \
+ patch.object(plugin_context, "get_autoinject_config",
+ AsyncMock(return_value={"enabled": True, "threshold": 0.5,
+ "top_k": 5})), \
+ patch.object(plugin_context, "record_retrieval", MagicMock()):
+ out = await plugin_context.build_autoinject_hint(7, "anything")
+
+ lines = out["context"].splitlines()
+ by_id = {n: next(ln for ln in lines if f"#{n} " in ln) for n in (1, 2, 3, 4, 5)}
+ assert "[snippet]" in by_id[1]
+ assert "[process]" in by_id[2]
+ # Task-ness wins over note_type, and an issue says so rather than "task".
+ assert "[issue]" in by_id[3]
+ assert "[task]" in by_id[4]
+ assert "[note]" in by_id[5]
+ # Still title-first: the marker is metadata, not an excuse to carry bodies.
+ assert "body" not in out["context"]
+ # The header can't claim they're all notes when the markers say otherwise.
+ assert "Scribe records" in lines[0]
diff --git a/tests/test_routes_snippets.py b/tests/test_routes_snippets.py
index c011be2..7b0497f 100644
--- a/tests/test_routes_snippets.py
+++ b/tests/test_routes_snippets.py
@@ -72,6 +72,19 @@ def test_project_scoping_reaches_every_caller():
assert "project_id" in inspect.getsource(routes.list_snippets_route)
+def test_location_lookup_reaches_every_caller():
+ """The reverse lookup — "what already lives here?" — has to be askable from
+ both surfaces, or an agent and a human get different answers about the same
+ file (rule #33)."""
+ from scribe.mcp.tools import snippets as tools
+ from scribe.routes import snippets as routes
+ from scribe.services import snippets as svc
+ for key in ("repo", "path", "symbol"):
+ assert key in inspect.signature(svc.list_snippets).parameters
+ assert key in inspect.signature(tools.list_snippets).parameters
+ assert f'request.args.get("{key}"' in inspect.getsource(routes)
+
+
def test_update_field_map_matches_service_kwargs():
"""Every field the PATCH route forwards must be a real update_snippet kwarg
(rule #33 interface-contract parity)."""
diff --git a/tests/test_services_plugin_context.py b/tests/test_services_plugin_context.py
index 6281a47..42c19b6 100644
--- a/tests/test_services_plugin_context.py
+++ b/tests/test_services_plugin_context.py
@@ -10,13 +10,15 @@ def _rule(rid, title, topic_id):
return r
-def _note(nid, title, user_id=1):
+def _note(nid, title, user_id=1, note_type="note", is_task=False, task_kind="work"):
n = MagicMock()
n.id, n.title = nid, title
- # Real int, defaulting to the caller used in these tests: the injected menu
- # compares it to decide whether the line needs a "shared by …" attribution,
- # and an auto-MagicMock would read as another user's note.
+ # Real values, defaulting to the caller used in these tests: the injected menu
+ # compares user_id to decide whether the line needs a "shared by …"
+ # attribution, and reads is_task/task_kind/note_type for the kind marker. An
+ # auto-MagicMock is truthy, so every line would read as another user's task.
n.user_id = user_id
+ n.note_type, n.is_task, n.task_kind = note_type, is_task, task_kind
return n
@@ -79,7 +81,7 @@ async def test_build_autoinject_hint_titles_only_with_margin_gate():
exclude_ids=[99])
# Margin gate kept the top two, dropped the straggler.
assert out["note_ids"] == [11, 22]
- assert '#11 "Pool sizing decision" (0.80)' in out["context"]
+ assert '#11 [note] "Pool sizing decision" (0.80)' in out["context"]
assert "#33" not in out["context"]
# Title-first: no body text, ever.
assert "get_note(id)" in out["context"]
diff --git a/tests/test_services_snippets.py b/tests/test_services_snippets.py
index 55487a9..a6246e2 100644
--- a/tests/test_services_snippets.py
+++ b/tests/test_services_snippets.py
@@ -158,6 +158,134 @@ def test_merge_snippet_fields_unions_locations_and_tags():
assert extra == ["core", "helper"]
+# --- the queryable mirror (notes.data, migration 0070) -----------------------
+
+def test_compose_data_keeps_only_populated_fields_and_no_code():
+ got = s.compose_data(
+ name="formatDuration", when_to_use="humanize a ms count",
+ signature="f(ms) -> string", language="TS",
+ locations=[{"repo": "web", "path": "a.ts", "symbol": "f"},
+ {"repo": "", "path": "", "symbol": ""}],
+ )
+ assert got == {
+ "name": "formatDuration",
+ "when_to_use": "humanize a ms count",
+ "signature": "f(ms) -> string",
+ "language": "ts",
+ "locations": [{"repo": "web", "path": "a.ts", "symbol": "f"}],
+ }
+ # Code stays in the body — duplicating a blob into the column we index
+ # around would be pure weight.
+ assert "code" not in got
+
+
+def test_compose_data_omits_blanks_entirely():
+ """A sparse column keeps containment matches from tripping over empties."""
+ assert s.compose_data(name="x") == {"name": "x"}
+ assert s.compose_data() == {}
+
+
+class _Note:
+ """Minimal stand-in — snippet_fields only reads title/body/tags/data."""
+
+ def __init__(self, title="", body="", tags=None, data=None):
+ self.title, self.body, self.tags, self.data = title, body, tags or [], data
+
+
+def test_snippet_fields_prefers_the_data_column():
+ body = s.compose_body(code="x = 1", language="py", signature="old()",
+ locations=[{"repo": "old", "path": "o.py", "symbol": "o"}])
+ note = _Note(
+ title="thing — old blurb", body=body, tags=["py", "snippet"],
+ data=s.compose_data(name="thing", when_to_use="new blurb",
+ signature="new()", language="py",
+ locations=[{"repo": "new", "path": "n.py", "symbol": "n"}]),
+ )
+ got = s.snippet_fields(note)
+ assert got["when_to_use"] == "new blurb"
+ assert got["signature"] == "new()"
+ assert got["locations"] == [{"repo": "new", "path": "n.py", "symbol": "n"}]
+ # The back-compat single-location mirror follows whichever list won.
+ assert (got["repo"], got["path"], got["symbol"]) == ("new", "n.py", "n")
+ # Code has no home in `data`, so it still comes from the body.
+ assert got["code"] == "x = 1"
+
+
+def test_snippet_fields_falls_back_to_the_body_when_data_is_absent():
+ """Rows written before 0070 are never backfilled, so the body stays
+ authoritative for them — with no deadline to convert."""
+ body = s.compose_body(code="y = 2", language="rb", signature="g()",
+ when_to_use="do a thing",
+ locations=[{"repo": "r", "path": "p.rb", "symbol": "g"}])
+ got = s.snippet_fields(_Note(title="g — do a thing", body=body,
+ tags=["rb", "snippet"], data=None))
+ assert got["signature"] == "g()"
+ assert got["language"] == "rb"
+ assert got["locations"] == [{"repo": "r", "path": "p.rb", "symbol": "g"}]
+ assert got["code"] == "y = 2"
+
+
+def test_data_and_body_round_trip_to_the_same_fields():
+ """The two representations must agree — they're written together, and a
+ disagreement would make a snippet read one way and query another."""
+ name, when, sig, lang = ("debounce", "rate-limit a callback",
+ "debounce(fn, ms)", "ts")
+ locs = [{"repo": "web", "path": "src/util.ts", "symbol": "debounce"}]
+ # compose_body takes no `name` — the name lives in the title — so the two
+ # serializers get their own argument lists rather than a shared spread.
+ title = s.compose_title(name, when)
+ body = s.compose_body(code="const x = 1", language=lang, signature=sig,
+ when_to_use=when, locations=locs, merged_from=[41, 42])
+ tags = s.compose_tags(lang)
+
+ from_body = s.snippet_fields(_Note(title=title, body=body, tags=tags))
+ from_data = s.snippet_fields(_Note(
+ title=title, body=body, tags=tags,
+ data=s.compose_data(name=name, when_to_use=when, signature=sig,
+ language=lang, locations=locs, merged_from=[41, 42]),
+ ))
+ for key in ("name", "when_to_use", "signature", "language", "locations",
+ "merged_from", "repo", "path", "symbol", "code"):
+ assert from_body[key] == from_data[key], key
+
+
+# --- merge provenance (#2087) ------------------------------------------------
+
+def test_normalize_merged_from_keeps_history_order_and_drops_junk():
+ """Order is history, not sorting — earlier merges stay first."""
+ assert s._normalize_merged_from([9, 3, 9, "4", None, 0, -2, "x"]) == [9, 3, 4]
+ assert s._normalize_merged_from(None) == []
+
+
+def test_compose_body_renders_merged_from_and_parse_reads_it_back():
+ body = s.compose_body(code="x = 1", merged_from=[12, 13])
+ assert "**Merged from:** #12, #13" in body
+ got = s.parse_snippet_fields("n — u", body, None)
+ assert got["merged_from"] == [12, 13]
+ # The code fence still comes last — provenance is header metadata.
+ assert body.rstrip().endswith("```")
+
+
+def test_compose_body_omits_merged_from_when_there_is_none():
+ """A snippet that was never merged says nothing about merging."""
+ assert "Merged from" not in s.compose_body(code="x = 1")
+ assert s.parse_snippet_fields("n — u", "```\nx = 1\n```", None)["merged_from"] == []
+
+
+def test_compose_data_mirrors_merged_from():
+ assert s.compose_data(name="f", merged_from=[3, 3, 2])["merged_from"] == [3, 2]
+ assert "merged_from" not in s.compose_data(name="f")
+
+
+def test_snippet_fields_prefers_the_data_columns_merged_from():
+ """Same rule as every other mirrored field: `data` wins when it's populated,
+ so a hand-edited body can't silently rewrite the merge history."""
+ body = s.compose_body(code="x = 1", merged_from=[12])
+ note = _Note(title="n — u", body=body, tags=["snippet"],
+ data=s.compose_data(name="n", merged_from=[12, 13]))
+ assert s.snippet_fields(note)["merged_from"] == [12, 13]
+
+
def test_snippet_to_dict_includes_parsed_fields():
class FakeNote:
title = "debounce — rate-limit"
diff --git a/tests/test_shared_write_access.py b/tests/test_shared_write_access.py
index c68abb7..99e498a 100644
--- a/tests/test_shared_write_access.py
+++ b/tests/test_shared_write_access.py
@@ -24,6 +24,9 @@ def _snippet(id=1, owner=9):
n.tags = ["ts", "snippet"]
n.note_type = "snippet"
n.deleted_at = None
+ # Explicitly None — snippet_fields prefers `data` when truthy, and an
+ # auto-MagicMock attribute is truthy (see note 2109).
+ n.data = None
return n
diff --git a/tests/test_snippet_location_filter.py b/tests/test_snippet_location_filter.py
new file mode 100644
index 0000000..882c4dc
--- /dev/null
+++ b/tests/test_snippet_location_filter.py
@@ -0,0 +1,205 @@
+"""Tests for the snippet location reverse lookup (#2083).
+
+Covers the pure half of the predicate — `location_parts`, `location_matches`,
+`location_jsonpath` — plus the parameter plumbing across the three surfaces
+(service / REST route / MCP tool), which rule #33 requires to agree on names and
+defaults.
+
+The SQL half can only be proved against real Postgres — see
+tests/test_integration_snippet_locations.py, which also pins the two dialects to
+the same answers on the same rows.
+"""
+import inspect
+import json
+from unittest.mock import AsyncMock, patch
+
+import pytest
+
+from scribe.services.knowledge import (
+ LOCATION_KEYS,
+ location_jsonpath,
+ location_matches,
+ location_parts,
+)
+
+
+def _data(*locations: dict) -> dict:
+ return {"locations": list(locations)}
+
+
+# --- location_parts -----------------------------------------------------------
+
+def test_location_parts_keeps_only_what_was_asked_for():
+ assert location_parts() == {}
+ assert location_parts(repo="Scribe") == {"repo": "Scribe"}
+ assert location_parts(repo=" Scribe ", path=" src/x.py ") == {
+ "repo": "Scribe", "path": "src/x.py",
+ }
+
+
+def test_location_parts_treats_whitespace_as_absent():
+ """An empty query-string arg must not filter the list down to nothing."""
+ assert location_parts(repo=" ", path="", symbol=None or "") == {}
+
+
+def test_location_parts_covers_every_location_key():
+ assert set(location_parts(repo="a", path="b", symbol="c")) == set(LOCATION_KEYS)
+
+
+# --- location_matches ---------------------------------------------------------
+
+def test_no_parts_matches_everything():
+ assert location_matches(None, {}) is True
+ assert location_matches(_data({"repo": "x", "path": "y", "symbol": "z"}), {}) is True
+
+
+def test_matches_exact_repo_path_and_symbol():
+ data = _data({"repo": "Scribe", "path": "src/scribe/x.py", "symbol": "helper"})
+ assert location_matches(data, {"repo": "Scribe"})
+ assert location_matches(data, {"path": "src/scribe/x.py"})
+ assert location_matches(data, {"symbol": "helper"})
+ assert location_matches(data, {"repo": "Scribe", "symbol": "helper"})
+
+
+def test_path_matches_as_a_directory_prefix():
+ data = _data({"repo": "Scribe", "path": "frontend/src/lib/x.ts", "symbol": ""})
+ assert location_matches(data, {"path": "frontend/src"})
+ assert location_matches(data, {"path": "frontend/src/"})
+ assert location_matches(data, {"path": "frontend/src/lib/x.ts"})
+
+
+def test_path_prefix_stops_at_a_directory_boundary():
+ """`frontend/src` must not match `frontend/srcmap.ts` — a sibling, not a child."""
+ data = _data({"repo": "Scribe", "path": "frontend/srcmap.ts", "symbol": ""})
+ assert not location_matches(data, {"path": "frontend/src"})
+
+
+def test_parts_must_all_match_the_same_location():
+ """Repo A in one entry and path B in another is not "recorded at A/B"."""
+ data = _data(
+ {"repo": "Scribe", "path": "src/a.py", "symbol": ""},
+ {"repo": "Portal", "path": "src/b.py", "symbol": ""},
+ )
+ assert location_matches(data, {"repo": "Scribe", "path": "src/a.py"})
+ assert not location_matches(data, {"repo": "Scribe", "path": "src/b.py"})
+
+
+def test_missing_or_empty_data_never_matches():
+ assert not location_matches(None, {"repo": "Scribe"})
+ assert not location_matches({}, {"repo": "Scribe"})
+ assert not location_matches({"locations": []}, {"repo": "Scribe"})
+ assert not location_matches({"name": "x"}, {"repo": "Scribe"})
+
+
+def test_blank_recorded_part_does_not_match_a_requested_one():
+ data = _data({"repo": "Scribe", "path": "", "symbol": ""})
+ assert not location_matches(data, {"repo": "Scribe", "symbol": "helper"})
+
+
+# --- location_jsonpath --------------------------------------------------------
+
+def test_jsonpath_filters_within_one_locations_entry():
+ expr = location_jsonpath({"repo": "Scribe", "symbol": "helper"})
+ assert expr.startswith("$.locations[*] ? (")
+ assert '@.repo == "Scribe"' in expr
+ assert '@.symbol == "helper"' in expr
+ assert " && " in expr
+
+
+def test_jsonpath_path_asks_for_exact_or_prefix():
+ expr = location_jsonpath({"path": "frontend/src"})
+ assert '@.path == "frontend/src"' in expr
+ assert '@.path starts with "frontend/src/"' in expr
+ assert " || " in expr
+
+
+def test_jsonpath_does_not_double_the_prefix_slash():
+ assert 'starts with "frontend/src/"' in location_jsonpath({"path": "frontend/src/"})
+
+
+def test_jsonpath_quotes_values_as_json_literals():
+ """A quote in a repo name must stay inside the literal, not end it."""
+ nasty = 'we"ird'
+ expr = location_jsonpath({"repo": nasty})
+ assert json.dumps(nasty) in expr
+ assert '\\"' in expr
+
+
+def test_jsonpath_emits_parts_in_a_fixed_key_order():
+ """Stable text keeps the query plan cacheable and the tests readable."""
+ a = location_jsonpath({"symbol": "s", "repo": "r", "path": "p"})
+ b = location_jsonpath({"repo": "r", "path": "p", "symbol": "s"})
+ assert a == b
+
+
+# --- the SQL dialect renders (shape only; behaviour is the integration test) ---
+
+def test_location_clause_renders_a_jsonpath_containment():
+ """Catches an API slip in the fast lane rather than waiting for Postgres —
+ the clause has to compile to `data @? :jsonpath` against the pg dialect."""
+ from sqlalchemy.dialects import postgresql
+
+ from scribe.services.knowledge import _location_clause
+
+ sql = str(_location_clause({"repo": "Scribe"}).compile(dialect=postgresql.dialect()))
+ assert "@?" in sql
+ assert "data" in sql
+
+
+# --- surface plumbing (rule #33) ---------------------------------------------
+
+def test_agent_and_service_surfaces_take_the_same_location_params():
+ from scribe.mcp.tools.snippets import list_snippets as mcp_list
+ from scribe.services.snippets import list_snippets as svc_list
+
+ for fn in (mcp_list, svc_list):
+ params = inspect.signature(fn).parameters
+ for key in LOCATION_KEYS:
+ assert key in params, f"{fn.__qualname__} is missing {key}"
+ assert params[key].default == "", f"{fn.__qualname__}.{key} default drifted"
+
+
+def test_route_forwards_the_location_args_by_name():
+ """The REST surface is the third caller; it must use the same arg names."""
+ from scribe.routes import snippets as routes
+ src = inspect.getsource(routes)
+ for key in LOCATION_KEYS:
+ assert f'request.args.get("{key}"' in src
+ assert f"{key}={key}" in src
+
+
+@pytest.mark.asyncio
+async def test_mcp_tool_passes_locations_through_to_the_service():
+ with patch("scribe.services.snippets.list_snippets",
+ new=AsyncMock(return_value=([], 0))) as m, \
+ patch("scribe.services.access.label_shared_items",
+ new=AsyncMock(return_value=[])):
+ from scribe.mcp.tools.snippets import list_snippets
+ await list_snippets(repo="Scribe", path="src/scribe", symbol="helper")
+ kwargs = m.await_args.kwargs
+ assert kwargs["repo"] == "Scribe"
+ assert kwargs["path"] == "src/scribe"
+ assert kwargs["symbol"] == "helper"
+
+
+@pytest.mark.asyncio
+async def test_service_sends_no_location_filter_when_none_was_asked_for():
+ """A plain list must not carry an empty filter — that would exclude every
+ row whose `data` has no locations at all."""
+ with patch("scribe.services.knowledge.query_knowledge",
+ new=AsyncMock(return_value=([], 0))) as m:
+ from scribe.services.snippets import list_snippets
+ await list_snippets(1)
+ assert m.await_args.kwargs["locations"] is None
+
+
+@pytest.mark.asyncio
+async def test_service_builds_the_filter_from_the_parts_given():
+ with patch("scribe.services.knowledge.query_knowledge",
+ new=AsyncMock(return_value=([], 0))) as m:
+ from scribe.services.snippets import list_snippets
+ await list_snippets(1, q="debounce", repo="Scribe", path=" src/ ")
+ kwargs = m.await_args.kwargs
+ assert kwargs["locations"] == {"repo": "Scribe", "path": "src/"}
+ # Search and place compose — the location filter must not drop the query.
+ assert kwargs["q"] == "debounce"
diff --git a/tests/test_snippet_merge_provenance.py b/tests/test_snippet_merge_provenance.py
new file mode 100644
index 0000000..9fc6bb4
--- /dev/null
+++ b/tests/test_snippet_merge_provenance.py
@@ -0,0 +1,113 @@
+"""Merge records what it absorbed (#2087).
+
+Merging keeps the target's scalar fields, unions locations + tags, and trashes
+the sources. Without provenance the fact that a variant ever existed survives
+only in the trash — and only for someone who already knew to go looking. So the
+survivor carries `merged_from`, written to the body and the indexed mirror from
+one value, and ordinary edits must carry it forward rather than erase it.
+"""
+from unittest.mock import AsyncMock, MagicMock, patch
+
+import pytest
+
+from scribe.services import snippets as s
+
+
+def _snippet(id=1, owner=9, body=None, data=None, tags=None):
+ n = MagicMock()
+ n.id = id
+ n.user_id = owner
+ n.title = "formatDuration — humanize a millisecond count"
+ n.body = body if body is not None else s.compose_body(code="x = 1", language="ts")
+ n.tags = tags if tags is not None else ["ts", "snippet"]
+ n.note_type = "snippet"
+ n.deleted_at = None
+ # Explicitly None — snippet_fields prefers `data` when truthy, and an
+ # auto-MagicMock attribute is truthy (note 2109).
+ n.data = data
+ return n
+
+
+async def _run_merge(target, sources, source_ids):
+ """Drive merge_snippets with the DB stubbed out; return update_note's kwargs."""
+ by_id = {target.id: target, **{x.id: x for x in sources}}
+
+ async def fake_get(_uid, sid):
+ return by_id.get(sid)
+
+ with patch.object(s, "get_snippet", AsyncMock(side_effect=fake_get)), \
+ patch("scribe.services.access.can_write_note", AsyncMock(return_value=True)), \
+ patch.object(s.notes_svc, "update_note",
+ AsyncMock(return_value=target)) as mock_update, \
+ patch("scribe.services.trash.delete", AsyncMock(return_value=object())), \
+ patch.object(s, "_embed_snippet", MagicMock()):
+ await s.merge_snippets(7, target.id, source_ids)
+ return mock_update.await_args.kwargs
+
+
+@pytest.mark.asyncio
+async def test_merge_records_what_it_absorbed_in_body_and_mirror():
+ kwargs = await _run_merge(
+ _snippet(id=1), [_snippet(id=2), _snippet(id=3)], [2, 3],
+ )
+ assert "**Merged from:** #2, #3" in kwargs["body"]
+ assert kwargs["data"]["merged_from"] == [2, 3]
+
+
+@pytest.mark.asyncio
+async def test_merge_accumulates_across_merges():
+ """A target merged twice keeps both histories — the second merge must not
+ replace the record of the first."""
+ target = _snippet(id=1, data=s.compose_data(name="f", merged_from=[2]))
+ kwargs = await _run_merge(target, [_snippet(id=3)], [3])
+ assert kwargs["data"]["merged_from"] == [2, 3]
+ assert "**Merged from:** #2, #3" in kwargs["body"]
+
+
+@pytest.mark.asyncio
+async def test_merge_does_not_record_a_skipped_cross_owner_source():
+ """A source owned by someone else is skipped, not folded in (#231) — so it
+ must not appear in the provenance either, or the record would claim to
+ contain something it never absorbed."""
+ kwargs = await _run_merge(
+ _snippet(id=1, owner=9),
+ [_snippet(id=2, owner=9), _snippet(id=3, owner=42)],
+ [2, 3],
+ )
+ assert kwargs["data"]["merged_from"] == [2]
+ assert "#3" not in kwargs["body"]
+
+
+@pytest.mark.asyncio
+async def test_an_ordinary_edit_carries_provenance_forward():
+ """Only a merge may add to it, but nothing else may drop it — update
+ recomposes body and mirror from scratch, so an omission here would silently
+ erase the history on the next unrelated edit."""
+ note = _snippet(id=1, data=s.compose_data(name="f", language="ts",
+ merged_from=[2, 3]))
+ with patch.object(s, "get_snippet", AsyncMock(return_value=note)), \
+ patch("scribe.services.access.can_write_note", AsyncMock(return_value=True)), \
+ patch.object(s.notes_svc, "update_note",
+ AsyncMock(return_value=note)) as mock_update, \
+ patch.object(s, "_embed_snippet", MagicMock()):
+ await s.update_snippet(7, 1, signature="f(ms) -> string")
+
+ kwargs = mock_update.await_args.kwargs
+ assert kwargs["data"]["merged_from"] == [2, 3]
+ assert "**Merged from:** #2, #3" in kwargs["body"]
+
+
+@pytest.mark.asyncio
+async def test_a_pre_0070_row_keeps_provenance_through_the_body():
+ """Rows written before the `data` column are never backfilled, so the body
+ line is their only copy — and it has to survive an edit too."""
+ body = s.compose_body(code="x = 1", language="ts", merged_from=[5])
+ note = _snippet(id=1, body=body, data=None)
+ with patch.object(s, "get_snippet", AsyncMock(return_value=note)), \
+ patch("scribe.services.access.can_write_note", AsyncMock(return_value=True)), \
+ patch.object(s.notes_svc, "update_note",
+ AsyncMock(return_value=note)) as mock_update, \
+ patch.object(s, "_embed_snippet", MagicMock()):
+ await s.update_snippet(7, 1, when_to_use="humanize a ms count")
+
+ assert "**Merged from:** #5" in mock_update.await_args.kwargs["body"]
diff --git a/tests/test_write_path_trigger.py b/tests/test_write_path_trigger.py
new file mode 100644
index 0000000..3fff4bf
--- /dev/null
+++ b/tests/test_write_path_trigger.py
@@ -0,0 +1,405 @@
+"""Tests for the write-path trigger (#2082) — prior art at the moment code is written.
+
+Covers the service (`build_write_path_hint`): the two arms and their precedence,
+the anti-bloat gates carried over from milestone 93, telemetry under its own
+source, and the two ways this must stay silent. Plus the plugin hook contract —
+a PreToolUse hook that returns a permission decision would be able to block the
+operator's edit, which this feature must never do.
+"""
+import json
+import re
+import subprocess
+from pathlib import Path
+from unittest.mock import AsyncMock, MagicMock, patch
+
+import pytest
+
+PLUGIN = Path(__file__).resolve().parents[1] / "plugin"
+HOOK = PLUGIN / "hooks" / "scribe_prior_art.sh"
+
+
+def _snippet_item(nid, title, user_id=1):
+ """A row as services.snippets.list_snippets returns it (a dict, not a Note)."""
+ return {"id": nid, "title": title, "user_id": user_id, "note_type": "snippet"}
+
+
+def _note(nid, title, user_id=1):
+ n = MagicMock()
+ n.id, n.title, n.user_id = nid, title, user_id
+ return n
+
+
+def _cfg(**over):
+ base = {"enabled": True, "threshold": 0.55, "top_k": 3}
+ base.update(over)
+ return base
+
+
+# --- silence: the common case ------------------------------------------------
+
+@pytest.mark.asyncio
+async def test_disabled_returns_empty_and_touches_nothing():
+ from scribe.services import plugin_context as pc
+ search, listing = AsyncMock(), AsyncMock()
+ with patch.object(pc, "get_writepath_config", AsyncMock(return_value=_cfg(enabled=False))), \
+ patch.object(pc.snippets_svc, "list_snippets", listing), \
+ patch.object(pc, "semantic_search_notes", search):
+ out = await pc.build_write_path_hint(1, "src/x.py", code="def f(): ...")
+ assert out["context"] == "" and out["note_ids"] == []
+ listing.assert_not_called()
+ search.assert_not_called()
+
+
+@pytest.mark.asyncio
+async def test_no_path_returns_empty():
+ """Nothing to look up by place, and the semantic arm alone isn't this feature."""
+ from scribe.services import plugin_context as pc
+ search = AsyncMock()
+ with patch.object(pc, "get_writepath_config", AsyncMock(return_value=_cfg())), \
+ patch.object(pc, "semantic_search_notes", search):
+ out = await pc.build_write_path_hint(1, " ", code="def f(): ...")
+ assert out["context"] == ""
+ search.assert_not_called()
+
+
+@pytest.mark.asyncio
+async def test_nothing_recorded_is_silent():
+ from scribe.services import plugin_context as pc
+ with patch.object(pc, "get_writepath_config", AsyncMock(return_value=_cfg())), \
+ patch.object(pc.snippets_svc, "list_snippets", AsyncMock(return_value=([], 0))), \
+ patch.object(pc, "semantic_search_notes", AsyncMock(return_value=[])), \
+ patch.object(pc, "record_retrieval", MagicMock()):
+ out = await pc.build_write_path_hint(1, "src/x.py", code="def f(): ...")
+ assert out["context"] == "" and out["note_ids"] == []
+
+
+# --- arm 1: by place ---------------------------------------------------------
+
+@pytest.mark.asyncio
+async def test_snippet_at_this_path_is_surfaced_without_a_score():
+ """A snippet recorded HERE is prior art by definition, not by resemblance —
+ it must not be subject to the similarity threshold."""
+ from scribe.services import plugin_context as pc
+ with patch.object(pc, "get_writepath_config", AsyncMock(return_value=_cfg())), \
+ patch.object(pc.snippets_svc, "list_snippets",
+ AsyncMock(return_value=([_snippet_item(12, "debounce — rate-limit a callback")], 1))), \
+ patch.object(pc, "semantic_search_notes", AsyncMock(return_value=[])), \
+ patch.object(pc, "record_retrieval", MagicMock()), \
+ patch.object(pc, "owner_names_for", AsyncMock(return_value={})):
+ out = await pc.build_write_path_hint(1, "src/x.py", code="def f(): ...")
+ assert out["note_ids"] == [12]
+ assert '#12 [here] "debounce — rate-limit a callback"' in out["context"]
+ assert "get_snippet(id)" in out["context"]
+ assert "src/x.py" in out["context"]
+
+
+@pytest.mark.asyncio
+async def test_directory_is_only_consulted_when_the_file_leaves_room():
+ from scribe.services import plugin_context as pc
+ calls = []
+
+ async def _listing(uid, **kw):
+ calls.append(kw["path"])
+ if kw["path"] == "src/lib/x.py":
+ return ([_snippet_item(i, f"s{i}") for i in (1, 2, 3)], 3)
+ return ([_snippet_item(9, "nearby")], 1)
+
+ with patch.object(pc, "get_writepath_config", AsyncMock(return_value=_cfg(top_k=3))), \
+ patch.object(pc.snippets_svc, "list_snippets", _listing), \
+ patch.object(pc, "semantic_search_notes", AsyncMock(return_value=[])), \
+ patch.object(pc, "record_retrieval", MagicMock()), \
+ patch.object(pc, "owner_names_for", AsyncMock(return_value={})):
+ out = await pc.build_write_path_hint(1, "src/lib/x.py", code="x")
+ # The file alone filled the budget, so the directory was never queried.
+ assert calls == ["src/lib/x.py"]
+ assert out["note_ids"] == [1, 2, 3]
+
+
+@pytest.mark.asyncio
+async def test_directory_hits_are_marked_nearby_not_here():
+ from scribe.services import plugin_context as pc
+
+ async def _listing(uid, **kw):
+ if kw["path"] == "src/lib":
+ return ([_snippet_item(9, "sibling helper")], 1)
+ return ([], 0)
+
+ with patch.object(pc, "get_writepath_config", AsyncMock(return_value=_cfg())), \
+ patch.object(pc.snippets_svc, "list_snippets", _listing), \
+ patch.object(pc, "semantic_search_notes", AsyncMock(return_value=[])), \
+ patch.object(pc, "record_retrieval", MagicMock()), \
+ patch.object(pc, "owner_names_for", AsyncMock(return_value={})):
+ out = await pc.build_write_path_hint(1, "src/lib/x.py", code="x")
+ assert '#9 [nearby] "sibling helper"' in out["context"]
+
+
+@pytest.mark.asyncio
+async def test_a_failing_location_lookup_does_not_sink_the_hint():
+ """The place arm is best-effort — a broken query must not cost the semantic one."""
+ from scribe.services import plugin_context as pc
+ with patch.object(pc, "get_writepath_config", AsyncMock(return_value=_cfg())), \
+ patch.object(pc.snippets_svc, "list_snippets", AsyncMock(side_effect=RuntimeError("boom"))), \
+ patch.object(pc, "semantic_search_notes",
+ AsyncMock(return_value=[(0.80, _note(5, "throttle — …"))])), \
+ patch.object(pc, "record_retrieval", MagicMock()), \
+ patch.object(pc, "owner_names_for", AsyncMock(return_value={})):
+ out = await pc.build_write_path_hint(1, "src/x.py", code="def f(): ...")
+ assert out["note_ids"] == [5]
+
+
+# --- arm 2: by meaning, under the milestone-93 gates -------------------------
+
+@pytest.mark.asyncio
+async def test_semantic_arm_is_snippet_only_and_browse_scoped():
+ from scribe.services import plugin_context as pc
+ search = AsyncMock(return_value=[])
+ with patch.object(pc, "get_writepath_config", AsyncMock(return_value=_cfg())), \
+ patch.object(pc.snippets_svc, "list_snippets", AsyncMock(return_value=([], 0))), \
+ patch.object(pc, "semantic_search_notes", search), \
+ patch.object(pc, "record_retrieval", MagicMock()):
+ await pc.build_write_path_hint(1, "src/x.py", code="def f(): ...")
+ kwargs = search.await_args.kwargs
+ # Prior art means snippets — not the dev-log that happens to mention one.
+ assert kwargs["note_type"] == "snippet"
+ # Nobody asked for this, so it must not reach a one-to-one direct share.
+ assert kwargs["scope"] == "browse"
+
+
+@pytest.mark.asyncio
+async def test_margin_gate_applies_to_the_semantic_arm():
+ from scribe.services import plugin_context as pc
+ hits = [(0.80, _note(11, "near")), (0.74, _note(22, "alsoNear")), (0.61, _note(33, "far"))]
+ with patch.object(pc, "get_writepath_config", AsyncMock(return_value=_cfg(top_k=5))), \
+ patch.object(pc.snippets_svc, "list_snippets", AsyncMock(return_value=([], 0))), \
+ patch.object(pc, "semantic_search_notes", AsyncMock(return_value=hits)), \
+ patch.object(pc, "record_retrieval", MagicMock()), \
+ patch.object(pc, "owner_names_for", AsyncMock(return_value={})):
+ out = await pc.build_write_path_hint(1, "src/x.py", code="def f(): ...")
+ assert out["note_ids"] == [11, 22]
+ assert "#33" not in out["context"]
+
+
+@pytest.mark.asyncio
+async def test_no_code_means_no_semantic_arm():
+ """An Edit whose payload we couldn't read still gets the location answer."""
+ from scribe.services import plugin_context as pc
+ search = AsyncMock()
+ with patch.object(pc, "get_writepath_config", AsyncMock(return_value=_cfg())), \
+ patch.object(pc.snippets_svc, "list_snippets",
+ AsyncMock(return_value=([_snippet_item(12, "here helper")], 1))), \
+ patch.object(pc, "semantic_search_notes", search), \
+ patch.object(pc, "record_retrieval", MagicMock()), \
+ patch.object(pc, "owner_names_for", AsyncMock(return_value={})):
+ out = await pc.build_write_path_hint(1, "src/x.py", code="")
+ assert out["note_ids"] == [12]
+ search.assert_not_called()
+
+
+@pytest.mark.asyncio
+async def test_place_beats_meaning_and_the_cap_covers_both_arms():
+ from scribe.services import plugin_context as pc
+
+ async def _listing(uid, **kw):
+ if kw["path"] == "src/x.py":
+ return ([_snippet_item(1, "placed")], 1)
+ return ([], 0)
+
+ with patch.object(pc, "get_writepath_config", AsyncMock(return_value=_cfg(top_k=2))), \
+ patch.object(pc.snippets_svc, "list_snippets", _listing), \
+ patch.object(pc, "semantic_search_notes",
+ AsyncMock(return_value=[(0.9, _note(7, "scored"))])), \
+ patch.object(pc, "record_retrieval", MagicMock()), \
+ patch.object(pc, "owner_names_for", AsyncMock(return_value={})):
+ out = await pc.build_write_path_hint(1, "src/x.py", code="x")
+ # Location hit first; the whole menu stays within top_k.
+ assert out["note_ids"] == [1, 7]
+ assert len(out["note_ids"]) <= 2
+ assert out["context"].index("#1") < out["context"].index("#7")
+
+
+@pytest.mark.asyncio
+async def test_semantic_arm_only_asks_for_the_budget_the_place_arm_left():
+ from scribe.services import plugin_context as pc
+ search = AsyncMock(return_value=[])
+
+ async def _listing(uid, **kw):
+ if kw["path"] == "src/x.py":
+ return ([_snippet_item(1, "placed")], 1)
+ return ([], 0)
+
+ with patch.object(pc, "get_writepath_config", AsyncMock(return_value=_cfg(top_k=3))), \
+ patch.object(pc.snippets_svc, "list_snippets", _listing), \
+ patch.object(pc, "semantic_search_notes", search), \
+ patch.object(pc, "record_retrieval", MagicMock()):
+ await pc.build_write_path_hint(1, "src/x.py", code="x")
+ assert search.await_args.kwargs["limit"] == 2
+
+
+@pytest.mark.asyncio
+async def test_session_dedup_excludes_ids_from_both_arms():
+ from scribe.services import plugin_context as pc
+ search = AsyncMock(return_value=[])
+ with patch.object(pc, "get_writepath_config", AsyncMock(return_value=_cfg())), \
+ patch.object(pc.snippets_svc, "list_snippets",
+ AsyncMock(return_value=([_snippet_item(12, "already shown")], 1))), \
+ patch.object(pc, "semantic_search_notes", search), \
+ patch.object(pc, "record_retrieval", MagicMock()):
+ out = await pc.build_write_path_hint(1, "src/x.py", code="x", exclude_ids=[12])
+ # The location hit was already surfaced this session → dropped, not repeated.
+ assert out["note_ids"] == []
+ assert 12 in search.await_args.kwargs["exclude_ids"]
+
+
+# --- attribution + telemetry -------------------------------------------------
+
+@pytest.mark.asyncio
+async def test_another_users_snippet_is_attributed():
+ """Rule #47 / the #2084 lesson: an unattributed line reads as your own record."""
+ from scribe.services import plugin_context as pc
+ with patch.object(pc, "get_writepath_config", AsyncMock(return_value=_cfg())), \
+ patch.object(pc.snippets_svc, "list_snippets",
+ AsyncMock(return_value=([_snippet_item(12, "theirs", user_id=42)], 1))), \
+ patch.object(pc, "semantic_search_notes", AsyncMock(return_value=[])), \
+ patch.object(pc, "record_retrieval", MagicMock()), \
+ patch.object(pc, "owner_names_for", AsyncMock(return_value={42: "alex"})):
+ out = await pc.build_write_path_hint(1, "src/x.py", code="x")
+ assert "shared by alex, treat as a suggestion" in out["context"]
+
+
+@pytest.mark.asyncio
+async def test_telemetry_uses_its_own_source():
+ """Separate source is what lets this surface be tuned apart from auto_inject."""
+ from scribe.services import plugin_context as pc
+ rec = MagicMock()
+ with patch.object(pc, "get_writepath_config", AsyncMock(return_value=_cfg())), \
+ patch.object(pc.snippets_svc, "list_snippets", AsyncMock(return_value=([], 0))), \
+ patch.object(pc, "semantic_search_notes",
+ AsyncMock(return_value=[(0.9, _note(7, "scored"))])), \
+ patch.object(pc, "record_retrieval", rec), \
+ patch.object(pc, "owner_names_for", AsyncMock(return_value={})):
+ await pc.build_write_path_hint(1, "src/x.py", code="x", project_id=4)
+ rec.assert_called_once()
+ assert rec.call_args.kwargs["source"] == "write_path"
+ assert rec.call_args.kwargs["source"] != "auto_inject"
+ assert rec.call_args.kwargs["project_id"] == 4
+
+
+@pytest.mark.asyncio
+async def test_config_has_its_own_switch_but_shares_the_gates():
+ from scribe.services import plugin_context as pc
+ stored = {pc.WRITEPATH_ENABLED_KEY: "false"}
+ with patch.object(pc, "get_setting",
+ AsyncMock(side_effect=lambda uid, k, d: stored.get(k, d))):
+ cfg = await pc.get_writepath_config(1)
+ # Its own switch is off while auto-inject stays on...
+ assert cfg["enabled"] is False
+ # ...and the gates are auto-inject's, not a second set that could drift.
+ assert cfg["threshold"] == pc.AUTOINJECT_DEFAULT_THRESHOLD
+ assert cfg["top_k"] == pc.AUTOINJECT_DEFAULT_TOP_K
+
+
+# --- the route between them --------------------------------------------------
+
+def test_route_reads_every_arg_the_hook_sends():
+ """Rule #33: the hook's query string and the route's reader are one contract,
+ and the hook is the only caller — a rename on either side is silent."""
+ import inspect
+
+ from scribe.routes import plugin as routes
+ src = inspect.getsource(routes.write_path_prior_art)
+ for arg in ("path", "code", "repo", "project_id", "exclude_ids"):
+ assert f'request.args.get("{arg}"' in src, f"route ignores {arg}"
+
+ hook = HOOK.read_text()
+ for arg in ("path=", "code=", "repo=", "exclude_ids="):
+ assert arg in hook, f"hook never sends {arg}"
+
+
+def test_route_resolves_repo_to_a_project_not_to_a_location_filter():
+ """A snippet's `repo` is a label the operator typed; the hook sends a git
+ remote. Matching one against the other would silently return nothing, so the
+ remote may only reach resolve_project."""
+ import inspect
+
+ from scribe.routes import plugin as routes
+ src = inspect.getsource(routes.write_path_prior_art)
+ assert "resolve_project" in src
+ assert "repo=repo" not in src
+
+
+# --- the plugin surface ------------------------------------------------------
+
+def test_hook_is_registered_for_write_and_edit():
+ cfg = json.loads((PLUGIN / "hooks" / "hooks.json").read_text())
+ entries = cfg["hooks"]["PreToolUse"]
+ assert any(
+ "Write" in e.get("matcher", "") and "Edit" in e.get("matcher", "")
+ and any("scribe_prior_art.sh" in h["command"] for h in e["hooks"])
+ for e in entries
+ )
+
+
+def test_hook_never_returns_a_permission_decision():
+ """The load-bearing property: this hook informs a write, it cannot stop one.
+ A permissionDecision of deny/ask would put a recall aid in the way of the
+ operator's work."""
+ src = HOOK.read_text()
+ assert "additionalContext" in src
+ code_lines = [ln for ln in src.splitlines() if not ln.lstrip().startswith("#")]
+ assert not any("permissionDecision" in ln for ln in code_lines)
+
+
+def test_hook_is_executable_and_shell_valid():
+ assert HOOK.stat().st_mode & 0o111, "hook must be executable"
+ subprocess.run(["bash", "-n", str(HOOK)], check=True)
+
+
+def test_hook_reads_both_write_and_edit_payload_shapes():
+ """Write and Edit name the payload differently, and the names have changed
+ across Claude Code versions — read whichever is present."""
+ src = HOOK.read_text()
+ for field in ("content", "file_content", "new_string", "new_str"):
+ assert f".tool_input.{field}" in src
+ assert ".tool_input.file_path" in src
+
+
+def test_hook_stays_a_get_so_a_read_scoped_key_works():
+ """Every other plugin hook works with a read-scoped key; a POST would demand
+ write scope (see auth.py) and silently break those installs."""
+ src = HOOK.read_text()
+ assert "-X POST" not in src and "--data" not in src
+ assert "/api/plugin/prior-art?" in src
+
+
+def test_hook_exits_silently_when_unconfigured():
+ """An install with no Scribe URL/token must produce no output at all."""
+ out = subprocess.run(
+ ["bash", str(HOOK)],
+ input=json.dumps({
+ "session_id": "s1", "cwd": "/tmp", "tool_name": "Write",
+ "tool_input": {"file_path": "/tmp/x.py", "content": "def f(): ..."},
+ }),
+ capture_output=True, text=True,
+ env={"PATH": "/usr/bin:/bin", "SCRIBE_URL": "", "SCRIBE_TOKEN": ""},
+ )
+ assert out.returncode == 0
+ assert out.stdout.strip() == ""
+
+
+def test_hook_skips_prose_and_data_files():
+ """No round-trip for a markdown edit — the server would return nothing anyway."""
+ src = HOOK.read_text()
+ skip = re.search(r"case \"\$file_path\" in\n(.*?)esac", src, re.S)
+ assert skip, "expected an extension skip list"
+ for ext in ("*.md", "*.json", "*.lock", "*.png"):
+ assert ext in skip.group(1)
+ # Config formats are deliberately NOT skipped — a workflow file is reusable.
+ assert "*.yml" not in skip.group(1)
+
+
+def test_plugin_version_bumped_with_the_hook():
+ """The #1040 lesson: a plugin change clients can't see is a change that didn't
+ ship."""
+ manifest = json.loads((PLUGIN / ".claude-plugin" / "plugin.json").read_text())
+ version = tuple(int(p) for p in manifest["version"].split("."))
+ assert version >= (0, 1, 18)