cca40affe4
CI & Build / Python lint (push) Successful in 2s
CI & Build / TypeScript typecheck (push) Successful in 12s
CI & Build / integration (push) Successful in 25s
CI & Build / Python tests (push) Failing after 29s
CI & Build / Build & push image (push) Has been skipped
Milestone #232 step 1 (task #2081). Takes the enabler first rather than the write-path trigger: reverse lookup, drift checks and the duplicate finder all need to QUERY structured fields, and building them on body-regex first means writing them twice. #227 deferred this bag "unless body-convention ergonomics prove insufficient" — answering "which snippets live in this file?" by scanning every snippet and regexing its body is that condition being met. Migration 0070 adds `notes.data` (nullable JSONB) + a GIN index. The body is UNCHANGED and still what gets embedded and read by humans; `data` mirrors the same facts in a shape Postgres can index. Code is deliberately not copied into it — the body holds it, and duplicating a blob into the column we index around would be waste. - compose_data() builds the mirror, omitting empties so the column stays sparse - snippet_fields() prefers `data`, falling back to parsing the body. Rows written before 0070 have no `data` and are never backfilled, so a hand-edited body stays authoritative for them with no conversion deadline - create / update / merge all write body and mirror from the same merged field set, so the two can't drift; merge in particular has to grow the mirror with the survivor's location set or a merged snippet would be unfindable at the very call sites the merge just recorded Named `data`, not `metadata`, because that collides with SQLAlchemy's declarative Base.metadata — which is why the pre-0069 model had to map an awkward `entity_metadata` attribute. Not a revival of the column 0069 dropped: different name, different purpose, nothing reads the old shape. Two test fakes needed an explicit `data = None`: snippet_fields prefers `data` when truthy and an auto-MagicMock attribute is truthy, so every parsed field would have come back a MagicMock. Checked every fake reaching snippet code this time rather than waiting for CI (note 2109). Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01RLwAaV4DQEmVyn496HnEvt
263 lines
11 KiB
Python
263 lines
11 KiB
Python
"""Unit tests for the snippet serialize/parse helpers (pure functions, no DB)."""
|
|
from scribe.services import snippets as s
|
|
|
|
|
|
def test_compose_title_with_and_without_usage():
|
|
assert s.compose_title("debounce", "rate-limit a callback") == "debounce — rate-limit a callback"
|
|
assert s.compose_title(" debounce ", "") == "debounce"
|
|
assert s.compose_title("debounce") == "debounce"
|
|
|
|
|
|
def test_compose_tags_lowercases_language_and_dedups():
|
|
assert s.compose_tags("Python", ["util", "python"]) == ["python", "snippet", "util"]
|
|
assert s.compose_tags("", None) == ["snippet"]
|
|
assert s.compose_tags("vue", ["snippet"]) == ["vue", "snippet"]
|
|
|
|
|
|
def test_compose_body_includes_fields_and_fence():
|
|
body = s.compose_body(
|
|
code="return 1", language="python", signature="f() -> int",
|
|
when_to_use="always", repo="scribe", path="a.py", symbol="f",
|
|
)
|
|
assert "**When to use:** always" in body
|
|
assert "**Signature:** `f() -> int`" in body
|
|
assert "`scribe` · `a.py` · `f`" in body
|
|
assert "```python\nreturn 1\n```" in body
|
|
assert body.rstrip().endswith("```")
|
|
|
|
|
|
def test_compose_body_bare_code_only():
|
|
body = s.compose_body(code="x = 1")
|
|
assert body.strip() == "```\nx = 1\n```"
|
|
|
|
|
|
def test_parse_round_trips_a_composed_snippet():
|
|
title = s.compose_title("useDebouncedRef", "debounce a reactive ref")
|
|
body = s.compose_body(
|
|
code="const x = 1", language="ts", signature="useDebouncedRef(v, ms)",
|
|
when_to_use="debounce a reactive ref", repo="scribe",
|
|
path="frontend/src/composables/x.ts", symbol="useDebouncedRef",
|
|
)
|
|
got = s.parse_snippet_fields(title, body, ["ts", "snippet"])
|
|
assert got["name"] == "useDebouncedRef"
|
|
assert got["when_to_use"] == "debounce a reactive ref"
|
|
assert got["signature"] == "useDebouncedRef(v, ms)"
|
|
assert got["language"] == "ts"
|
|
assert got["code"] == "const x = 1"
|
|
assert got["repo"] == "scribe"
|
|
assert got["path"] == "frontend/src/composables/x.ts"
|
|
assert got["symbol"] == "useDebouncedRef"
|
|
|
|
|
|
def test_parse_is_tolerant_of_plain_body():
|
|
got = s.parse_snippet_fields("just a name", "no structure here", None)
|
|
assert got["name"] == "just a name"
|
|
assert got["when_to_use"] == ""
|
|
assert got["signature"] == ""
|
|
assert got["code"] == "" # never raises on an unstructured body
|
|
|
|
|
|
def test_parse_falls_back_to_tag_for_language():
|
|
got = s.parse_snippet_fields("n — u", "```\ncode\n```", ["ruby", "snippet"])
|
|
assert got["language"] == "ruby"
|
|
|
|
|
|
def test_parse_does_not_promote_a_caller_tag_to_language():
|
|
# compose_tags puts the language FIRST, so a leading "snippet" marker means
|
|
# no language was recorded — the tags after it are the caller's own and must
|
|
# not be mistaken for one (which would also corrupt the code fence on the
|
|
# next update).
|
|
tags = s.compose_tags("", ["auth"])
|
|
assert tags == ["snippet", "auth"]
|
|
got = s.parse_snippet_fields("n — u", s.compose_body(code="x = 1"), tags)
|
|
assert got["language"] == ""
|
|
|
|
|
|
def test_caller_tag_survives_an_update_round_trip_without_a_language():
|
|
# Regression: the tag used to be read back as the language, then dropped
|
|
# from the extra-tag set on re-compose — so it silently disappeared.
|
|
tags = s.compose_tags("", ["auth"])
|
|
fields = s.parse_snippet_fields("n — u", s.compose_body(code="x = 1"), tags)
|
|
extra = [t for t in tags if t not in (s.SNIPPET_TAG, fields["language"])]
|
|
assert s.compose_tags(fields["language"], extra) == ["snippet", "auth"]
|
|
|
|
|
|
def test_compose_body_multi_location_renders_bullet_list():
|
|
body = s.compose_body(
|
|
code="x = 1",
|
|
locations=[
|
|
{"repo": "scribe", "path": "a.py", "symbol": "f"},
|
|
{"repo": "web", "path": "b.ts", "symbol": "g"},
|
|
],
|
|
)
|
|
assert "**Locations:**" in body
|
|
assert "- `scribe` · `a.py` · `f`" in body
|
|
assert "- `web` · `b.ts` · `g`" in body
|
|
assert "**Location:**" not in body.replace("**Locations:**", "")
|
|
|
|
|
|
def test_compose_body_single_location_via_list_uses_singular_label():
|
|
body = s.compose_body(
|
|
code="x = 1", locations=[{"repo": "scribe", "path": "a.py", "symbol": "f"}],
|
|
)
|
|
assert "**Location:** `scribe` · `a.py` · `f`" in body
|
|
assert "**Locations:**" not in body
|
|
|
|
|
|
def test_normalize_locations_drops_empty_and_dedups():
|
|
got = s._normalize_locations([
|
|
{"repo": "scribe", "path": "a.py", "symbol": "f"},
|
|
{"repo": "", "path": "", "symbol": ""}, # dropped (empty)
|
|
{"repo": "scribe", "path": "a.py", "symbol": "f"}, # dropped (dup)
|
|
{"repo": "web", "path": "", "symbol": ""},
|
|
])
|
|
assert got == [
|
|
{"repo": "scribe", "path": "a.py", "symbol": "f"},
|
|
{"repo": "web", "path": "", "symbol": ""},
|
|
]
|
|
|
|
|
|
def test_parse_round_trips_multi_location():
|
|
body = s.compose_body(
|
|
code="const x = 1", language="ts",
|
|
locations=[
|
|
{"repo": "scribe", "path": "a.ts", "symbol": "f"},
|
|
{"repo": "web", "path": "b.ts", "symbol": "g"},
|
|
],
|
|
)
|
|
got = s.parse_snippet_fields("n — u", body, ["ts", "snippet"])
|
|
assert got["locations"] == [
|
|
{"repo": "scribe", "path": "a.ts", "symbol": "f"},
|
|
{"repo": "web", "path": "b.ts", "symbol": "g"},
|
|
]
|
|
# repo/path/symbol mirror the first location for back-compat.
|
|
assert (got["repo"], got["path"], got["symbol"]) == ("scribe", "a.ts", "f")
|
|
|
|
|
|
def test_parse_legacy_single_location_line_still_works():
|
|
# A body written by the pre-multi-location serializer.
|
|
body = "**Location:** `scribe` · `a.py` · `f`\n\n```py\nx = 1\n```\n"
|
|
got = s.parse_snippet_fields("n — u", body, None)
|
|
assert got["locations"] == [{"repo": "scribe", "path": "a.py", "symbol": "f"}]
|
|
assert (got["repo"], got["path"], got["symbol"]) == ("scribe", "a.py", "f")
|
|
|
|
|
|
def test_merge_snippet_fields_unions_locations_and_tags():
|
|
target_fields = {"language": "py", "locations": [{"repo": "a", "path": "a.py", "symbol": "f"}]}
|
|
src1 = ({"language": "py", "locations": [{"repo": "b", "path": "b.py", "symbol": "g"}]},
|
|
["py", "snippet", "helper"])
|
|
src2 = ({"language": "py", "locations": [{"repo": "a", "path": "a.py", "symbol": "f"}]}, # dup loc
|
|
["snippet"])
|
|
locs, extra = s.merge_snippet_fields(target_fields, ["py", "snippet", "core"], [src1, src2])
|
|
# target location first, then unique source locations; dup dropped.
|
|
assert locs == [
|
|
{"repo": "a", "path": "a.py", "symbol": "f"},
|
|
{"repo": "b", "path": "b.py", "symbol": "g"},
|
|
]
|
|
# extra tags unioned, language + "snippet" markers excluded.
|
|
assert extra == ["core", "helper"]
|
|
|
|
|
|
# --- the queryable mirror (notes.data, migration 0070) -----------------------
|
|
|
|
def test_compose_data_keeps_only_populated_fields_and_no_code():
|
|
got = s.compose_data(
|
|
name="formatDuration", when_to_use="humanize a ms count",
|
|
signature="f(ms) -> string", language="TS",
|
|
locations=[{"repo": "web", "path": "a.ts", "symbol": "f"},
|
|
{"repo": "", "path": "", "symbol": ""}],
|
|
)
|
|
assert got == {
|
|
"name": "formatDuration",
|
|
"when_to_use": "humanize a ms count",
|
|
"signature": "f(ms) -> string",
|
|
"language": "ts",
|
|
"locations": [{"repo": "web", "path": "a.ts", "symbol": "f"}],
|
|
}
|
|
# Code stays in the body — duplicating a blob into the column we index
|
|
# around would be pure weight.
|
|
assert "code" not in got
|
|
|
|
|
|
def test_compose_data_omits_blanks_entirely():
|
|
"""A sparse column keeps containment matches from tripping over empties."""
|
|
assert s.compose_data(name="x") == {"name": "x"}
|
|
assert s.compose_data() == {}
|
|
|
|
|
|
class _Note:
|
|
"""Minimal stand-in — snippet_fields only reads title/body/tags/data."""
|
|
|
|
def __init__(self, title="", body="", tags=None, data=None):
|
|
self.title, self.body, self.tags, self.data = title, body, tags or [], data
|
|
|
|
|
|
def test_snippet_fields_prefers_the_data_column():
|
|
body = s.compose_body(code="x = 1", language="py", signature="old()",
|
|
locations=[{"repo": "old", "path": "o.py", "symbol": "o"}])
|
|
note = _Note(
|
|
title="thing — old blurb", body=body, tags=["py", "snippet"],
|
|
data=s.compose_data(name="thing", when_to_use="new blurb",
|
|
signature="new()", language="py",
|
|
locations=[{"repo": "new", "path": "n.py", "symbol": "n"}]),
|
|
)
|
|
got = s.snippet_fields(note)
|
|
assert got["when_to_use"] == "new blurb"
|
|
assert got["signature"] == "new()"
|
|
assert got["locations"] == [{"repo": "new", "path": "n.py", "symbol": "n"}]
|
|
# The back-compat single-location mirror follows whichever list won.
|
|
assert (got["repo"], got["path"], got["symbol"]) == ("new", "n.py", "n")
|
|
# Code has no home in `data`, so it still comes from the body.
|
|
assert got["code"] == "x = 1"
|
|
|
|
|
|
def test_snippet_fields_falls_back_to_the_body_when_data_is_absent():
|
|
"""Rows written before 0070 are never backfilled, so the body stays
|
|
authoritative for them — with no deadline to convert."""
|
|
body = s.compose_body(code="y = 2", language="rb", signature="g()",
|
|
when_to_use="do a thing",
|
|
locations=[{"repo": "r", "path": "p.rb", "symbol": "g"}])
|
|
got = s.snippet_fields(_Note(title="g — do a thing", body=body,
|
|
tags=["rb", "snippet"], data=None))
|
|
assert got["signature"] == "g()"
|
|
assert got["language"] == "rb"
|
|
assert got["locations"] == [{"repo": "r", "path": "p.rb", "symbol": "g"}]
|
|
assert got["code"] == "y = 2"
|
|
|
|
|
|
def test_data_and_body_round_trip_to_the_same_fields():
|
|
"""The two representations must agree — they're written together, and a
|
|
disagreement would make a snippet read one way and query another."""
|
|
fields = dict(name="debounce", when_to_use="rate-limit a callback",
|
|
signature="debounce(fn, ms)", language="ts")
|
|
locs = [{"repo": "web", "path": "src/util.ts", "symbol": "debounce"}]
|
|
from_body = s.snippet_fields(_Note(
|
|
title=s.compose_title(fields["name"], fields["when_to_use"]),
|
|
body=s.compose_body(code="const x = 1", locations=locs, **fields),
|
|
tags=s.compose_tags(fields["language"]),
|
|
))
|
|
from_data = s.snippet_fields(_Note(
|
|
title=s.compose_title(fields["name"], fields["when_to_use"]),
|
|
body=s.compose_body(code="const x = 1", locations=locs, **fields),
|
|
tags=s.compose_tags(fields["language"]),
|
|
data=s.compose_data(locations=locs, **fields),
|
|
))
|
|
for key in ("name", "when_to_use", "signature", "language", "locations",
|
|
"repo", "path", "symbol", "code"):
|
|
assert from_body[key] == from_data[key], key
|
|
|
|
|
|
def test_snippet_to_dict_includes_parsed_fields():
|
|
class FakeNote:
|
|
title = "debounce — rate-limit"
|
|
body = "```js\ncode\n```\n"
|
|
tags = ["js", "snippet"]
|
|
|
|
def to_dict(self):
|
|
return {"id": 1, "title": self.title, "note_type": "snippet", "tags": self.tags}
|
|
|
|
data = s.snippet_to_dict(FakeNote())
|
|
assert data["snippet"]["name"] == "debounce"
|
|
assert data["snippet"]["language"] == "js"
|
|
assert data["snippet"]["code"] == "code"
|