Files
FabledScribe/tests/test_services_snippets.py
T
bvandeusenandClaude Opus 5.5 66e21a6c60
CI & Build / Python lint (push) Successful in 3s
CI & Build / Plugin hooks (push) Successful in 14s
CI & Build / integration (push) Successful in 52s
CI & Build / TypeScript typecheck (push) Successful in 53s
CI & Build / Python tests (push) Successful in 1m35s
CI & Build / Build & push image (push) Successful in 32s
refactor(notes): a snippet's and lesson's stored title is its name; the trigger joins it only in the embedded document (milestone 427)
The title was `subject — trigger` because the stored title WAS the
embedded one, and the join is what makes these kinds rank on the
situation they apply to (#2485). Every surface that shows a title then
showed the trigger too -- menus, lists and search rows ran to kilobytes.

- embeddings.document_title(title, note_type, data, body) joins the
  trigger from `data` (body fallback) at embed time. Idempotent: an
  un-migrated composed title comes out the same, never doubled. The
  embed path, the startup backfill and the dedup gate's semantic signal
  all use it, so the embedded text -- and every vector -- is unchanged.
- Writers store the subject: snippet create/update (service, REST, MCP)
  and lesson_document. Both compose_title helpers are removed.
- Readers: dedup takes `data`; the menus strip the embedded title from a
  passage; list rows project `when_to_use`, which SnippetListView reads.
- 0108 rewrites existing rows on an exact `' — ' || <own trigger>`
  suffix with raw SQL, leaving updated_at alone so the backfill does not
  re-embed the corpus for identical vectors. Downgrade recomposes.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-23 16:48:28 -04:00

346 lines
15 KiB
Python

"""Unit tests for the snippet serialize/parse helpers (pure functions, no DB)."""
from scribe.services import snippets as s
def test_the_embedded_title_joins_the_trigger_the_stored_one_does_not_carry():
"""Milestone 427: stored title = name; the trigger joins it at embed time,
idempotently, so an old composed title comes out the same."""
from scribe.services.embeddings import document_title
data = {"name": "debounce", "when_to_use": "rate-limit a callback"}
assert document_title("debounce", "snippet", data) == "debounce — rate-limit a callback"
assert document_title("debounce — rate-limit a callback", "snippet", data) == (
"debounce — rate-limit a callback"
)
assert document_title("debounce", "snippet", {"name": "debounce"}) == "debounce"
assert document_title("a — note", "note", data) == "a — note"
def test_compose_tags_lowercases_language_and_dedups():
assert s.compose_tags("Python", ["util", "python"]) == ["python", "snippet", "util"]
assert s.compose_tags("", None) == ["snippet"]
assert s.compose_tags("vue", ["snippet"]) == ["vue", "snippet"]
def test_compose_body_includes_fields_and_fence():
body = s.compose_body(
code="return 1", language="python", signature="f() -> int",
when_to_use="always", repo="scribe", path="a.py", symbol="f",
)
assert "**When to use:** always" in body
assert "**Signature:** `f() -> int`" in body
assert "`scribe` · `a.py` · `f`" in body
assert "```python\nreturn 1\n```" in body
assert body.rstrip().endswith("```")
def test_compose_body_bare_code_only():
body = s.compose_body(code="x = 1")
assert body.strip() == "```\nx = 1\n```"
def test_parse_round_trips_a_composed_snippet():
title = "useDebouncedRef"
body = s.compose_body(
code="const x = 1", language="ts", signature="useDebouncedRef(v, ms)",
when_to_use="debounce a reactive ref", repo="scribe",
path="frontend/src/composables/x.ts", symbol="useDebouncedRef",
)
got = s.parse_snippet_fields(title, body, ["ts", "snippet"])
assert got["name"] == "useDebouncedRef"
assert got["when_to_use"] == "debounce a reactive ref"
assert got["signature"] == "useDebouncedRef(v, ms)"
assert got["language"] == "ts"
assert got["code"] == "const x = 1"
assert got["repo"] == "scribe"
assert got["path"] == "frontend/src/composables/x.ts"
assert got["symbol"] == "useDebouncedRef"
def test_parse_is_tolerant_of_plain_body():
got = s.parse_snippet_fields("just a name", "no structure here", None)
assert got["name"] == "just a name"
assert got["when_to_use"] == ""
assert got["signature"] == ""
assert got["code"] == "" # never raises on an unstructured body
def test_parse_falls_back_to_tag_for_language():
got = s.parse_snippet_fields("n — u", "```\ncode\n```", ["ruby", "snippet"])
assert got["language"] == "ruby"
def test_parse_does_not_promote_a_caller_tag_to_language():
# compose_tags puts the language FIRST, so a leading "snippet" marker means
# no language was recorded — the tags after it are the caller's own and must
# not be mistaken for one (which would also corrupt the code fence on the
# next update).
tags = s.compose_tags("", ["auth"])
assert tags == ["snippet", "auth"]
got = s.parse_snippet_fields("n — u", s.compose_body(code="x = 1"), tags)
assert got["language"] == ""
def test_caller_tag_survives_an_update_round_trip_without_a_language():
# Regression: the tag used to be read back as the language, then dropped
# from the extra-tag set on re-compose — so it silently disappeared.
tags = s.compose_tags("", ["auth"])
fields = s.parse_snippet_fields("n — u", s.compose_body(code="x = 1"), tags)
extra = [t for t in tags if t not in (s.SNIPPET_TAG, fields["language"])]
assert s.compose_tags(fields["language"], extra) == ["snippet", "auth"]
def test_compose_body_multi_location_renders_bullet_list():
body = s.compose_body(
code="x = 1",
locations=[
{"repo": "scribe", "path": "a.py", "symbol": "f"},
{"repo": "web", "path": "b.ts", "symbol": "g"},
],
)
assert "**Locations:**" in body
assert "- `scribe` · `a.py` · `f`" in body
assert "- `web` · `b.ts` · `g`" in body
assert "**Location:**" not in body.replace("**Locations:**", "")
def test_compose_body_single_location_via_list_uses_singular_label():
body = s.compose_body(
code="x = 1", locations=[{"repo": "scribe", "path": "a.py", "symbol": "f"}],
)
assert "**Location:** `scribe` · `a.py` · `f`" in body
assert "**Locations:**" not in body
def test_normalize_locations_drops_empty_and_dedups():
got = s._normalize_locations([
{"repo": "scribe", "path": "a.py", "symbol": "f"},
{"repo": "", "path": "", "symbol": ""}, # dropped (empty)
{"repo": "scribe", "path": "a.py", "symbol": "f"}, # dropped (dup)
{"repo": "web", "path": "", "symbol": ""},
])
assert got == [
{"repo": "scribe", "path": "a.py", "symbol": "f"},
{"repo": "web", "path": "", "symbol": ""},
]
def test_parse_round_trips_multi_location():
body = s.compose_body(
code="const x = 1", language="ts",
locations=[
{"repo": "scribe", "path": "a.ts", "symbol": "f"},
{"repo": "web", "path": "b.ts", "symbol": "g"},
],
)
got = s.parse_snippet_fields("n — u", body, ["ts", "snippet"])
assert got["locations"] == [
{"repo": "scribe", "path": "a.ts", "symbol": "f"},
{"repo": "web", "path": "b.ts", "symbol": "g"},
]
# repo/path/symbol mirror the first location for back-compat.
assert (got["repo"], got["path"], got["symbol"]) == ("scribe", "a.ts", "f")
def test_parse_legacy_single_location_line_still_works():
# A body written by the pre-multi-location serializer.
body = "**Location:** `scribe` · `a.py` · `f`\n\n```py\nx = 1\n```\n"
got = s.parse_snippet_fields("n — u", body, None)
assert got["locations"] == [{"repo": "scribe", "path": "a.py", "symbol": "f"}]
assert (got["repo"], got["path"], got["symbol"]) == ("scribe", "a.py", "f")
def test_merge_snippet_fields_unions_locations_and_tags():
target_fields = {"language": "py", "locations": [{"repo": "a", "path": "a.py", "symbol": "f"}]}
src1 = ({"language": "py", "locations": [{"repo": "b", "path": "b.py", "symbol": "g"}]},
["py", "snippet", "helper"])
src2 = ({"language": "py", "locations": [{"repo": "a", "path": "a.py", "symbol": "f"}]}, # dup loc
["snippet"])
locs, extra, contributions = s.merge_snippet_fields(
target_fields, ["py", "snippet", "core"], [src1, src2]
)
# target location first, then unique source locations; dup dropped.
assert locs == [
{"repo": "a", "path": "a.py", "symbol": "f"},
{"repo": "b", "path": "b.py", "symbol": "g"},
]
# extra tags unioned, language + "snippet" markers excluded.
assert extra == ["core", "helper"]
# Attribution (#2165): only what each source ACTUALLY added. src2's location
# is one the target already had, so it contributed nothing — un-merging it
# must not strip a call site the survivor owns in its own right.
assert contributions[0]["locations"] == [{"repo": "b", "path": "b.py", "symbol": "g"}]
assert contributions[0]["tags"] == ["helper"]
assert contributions[1]["locations"] == []
assert contributions[1]["tags"] == []
# --- the queryable mirror (notes.data, migration 0070) -----------------------
def test_compose_data_keeps_only_populated_fields_and_no_code():
got = s.compose_data(
name="formatDuration", when_to_use="humanize a ms count",
signature="f(ms) -> string", language="TS",
locations=[{"repo": "web", "path": "a.ts", "symbol": "f"},
{"repo": "", "path": "", "symbol": ""}],
)
assert got == {
"name": "formatDuration",
"when_to_use": "humanize a ms count",
"signature": "f(ms) -> string",
"language": "ts",
"locations": [{"repo": "web", "path": "a.ts", "symbol": "f"}],
}
# Code stays in the body — duplicating a blob into the column we index
# around would be pure weight.
assert "code" not in got
def test_compose_data_omits_blanks_entirely():
"""A sparse column keeps containment matches from tripping over empties."""
assert s.compose_data(name="x") == {"name": "x"}
assert s.compose_data() == {}
class _Note:
"""Minimal stand-in — snippet_fields only reads title/body/tags/data."""
def __init__(self, title="", body="", tags=None, data=None):
self.title, self.body, self.tags, self.data = title, body, tags or [], data
def test_snippet_fields_prefers_the_data_column():
body = s.compose_body(code="x = 1", language="py", signature="old()",
locations=[{"repo": "old", "path": "o.py", "symbol": "o"}])
note = _Note(
title="thing — old blurb", body=body, tags=["py", "snippet"],
data=s.compose_data(name="thing", when_to_use="new blurb",
signature="new()", language="py",
locations=[{"repo": "new", "path": "n.py", "symbol": "n"}]),
)
got = s.snippet_fields(note)
assert got["when_to_use"] == "new blurb"
assert got["signature"] == "new()"
assert got["locations"] == [{"repo": "new", "path": "n.py", "symbol": "n"}]
# The back-compat single-location mirror follows whichever list won.
assert (got["repo"], got["path"], got["symbol"]) == ("new", "n.py", "n")
# Code has no home in `data`, so it still comes from the body.
assert got["code"] == "x = 1"
def test_snippet_fields_falls_back_to_the_body_when_data_is_absent():
"""Rows written before 0070 are never backfilled, so the body stays
authoritative for them — with no deadline to convert."""
body = s.compose_body(code="y = 2", language="rb", signature="g()",
when_to_use="do a thing",
locations=[{"repo": "r", "path": "p.rb", "symbol": "g"}])
got = s.snippet_fields(_Note(title="g — do a thing", body=body,
tags=["rb", "snippet"], data=None))
assert got["signature"] == "g()"
assert got["language"] == "rb"
assert got["locations"] == [{"repo": "r", "path": "p.rb", "symbol": "g"}]
assert got["code"] == "y = 2"
def test_data_and_body_round_trip_to_the_same_fields():
"""The two representations must agree — they're written together, and a
disagreement would make a snippet read one way and query another."""
name, when, sig, lang = ("debounce", "rate-limit a callback",
"debounce(fn, ms)", "ts")
locs = [{"repo": "web", "path": "src/util.ts", "symbol": "debounce"}]
# compose_body takes no `name` — the name IS the title (milestone 427) — so
# the two serializers get their own argument lists rather than a shared spread.
title = name
body = s.compose_body(code="const x = 1", language=lang, signature=sig,
when_to_use=when, locations=locs, merged_from=[41, 42])
tags = s.compose_tags(lang)
from_body = s.snippet_fields(_Note(title=title, body=body, tags=tags))
from_data = s.snippet_fields(_Note(
title=title, body=body, tags=tags,
data=s.compose_data(name=name, when_to_use=when, signature=sig,
language=lang, locations=locs, merged_from=[41, 42]),
))
for key in ("name", "when_to_use", "signature", "language", "locations",
"merged_from", "repo", "path", "symbol", "code"):
assert from_body[key] == from_data[key], key
# --- merge provenance (#2087) ------------------------------------------------
def test_normalize_merged_from_keeps_history_order_and_drops_junk():
"""Order is history, not sorting — earlier merges stay first."""
assert s.merged_from_ids([9, 3, 9, "4", None, 0, -2, "x"]) == [9, 3, 4]
assert s._normalize_merged_from(None) == []
def test_normalize_merged_from_carries_per_source_attribution():
"""The shape that makes un-merge exact (#2165): each entry holds what THAT
source contributed, so reversing one can't strip what the survivor owns."""
loc = {"repo": "a", "path": "a.py", "symbol": "f"}
out = s._normalize_merged_from([{"id": 7, "locations": [loc], "tags": ["x"]}])
assert out == [{"id": 7, "locations": [loc], "tags": ["x"]}]
def test_a_bare_id_normalizes_to_an_entry_with_no_attribution():
"""Not legacy tolerance: snippet_fields falls back to PARSING THE BODY when a
row has no `data`, and the body's provenance line can only carry ids. Such an
entry still shows history; un-merge refuses it rather than guessing."""
assert s._normalize_merged_from([5]) == [{"id": 5}]
def test_compose_body_renders_merged_from_and_parse_reads_it_back():
body = s.compose_body(code="x = 1", merged_from=[12, 13])
assert "**Merged from:** #12, #13" in body
got = s.parse_snippet_fields("n — u", body, None)
# Ids survive the body round-trip; attribution cannot, since the body only
# ever renders ids — which is exactly why `data` is the authority for it.
assert s.merged_from_ids(got["merged_from"]) == [12, 13]
# The code fence still comes last — provenance is header metadata.
assert body.rstrip().endswith("```")
def test_compose_body_renders_ids_from_rich_entries():
body = s.compose_body(
code="x = 1",
merged_from=[{"id": 12, "locations": [{"repo": "", "path": "a.py", "symbol": ""}]}],
)
assert "**Merged from:** #12" in body
def test_compose_body_omits_merged_from_when_there_is_none():
"""A snippet that was never merged says nothing about merging."""
assert "Merged from" not in s.compose_body(code="x = 1")
assert s.parse_snippet_fields("n — u", "```\nx = 1\n```", None)["merged_from"] == []
def test_compose_data_mirrors_merged_from():
got = s.compose_data(name="f", merged_from=[3, 3, 2])["merged_from"]
assert s.merged_from_ids(got) == [3, 2]
assert "merged_from" not in s.compose_data(name="f")
def test_snippet_fields_prefers_the_data_columns_merged_from():
"""Same rule as every other mirrored field: `data` wins when it's populated,
so a hand-edited body can't silently rewrite the merge history."""
body = s.compose_body(code="x = 1", merged_from=[12])
note = _Note(title="n — u", body=body, tags=["snippet"],
data=s.compose_data(name="n", merged_from=[12, 13]))
assert s.merged_from_ids(s.snippet_fields(note)["merged_from"]) == [12, 13]
def test_snippet_to_dict_includes_parsed_fields():
class FakeNote:
title = "debounce — rate-limit"
body = "```js\ncode\n```\n"
tags = ["js", "snippet"]
def to_dict(self):
return {"id": 1, "title": self.title, "note_type": "snippet", "tags": self.tags}
data = s.snippet_to_dict(FakeNote())
assert data["snippet"]["name"] == "debounce"
assert data["snippet"]["language"] == "js"
assert data["snippet"]["code"] == "code"