Files
thoughtsync/tests/test_notes.py
T
bvandeusen de72d27bd4
CI & Build / Build now, or wait for Android? (push) Successful in 3s
CI & Build / Python lint (push) Successful in 2s
CI & Build / TypeScript typecheck (push) Successful in 7s
CI & Build / Python tests (push) Successful in 9s
CI & Build / integration (push) Successful in 14s
CI & Build / Build & push image (push) Successful in 37s
Desktop (Tauri) / Windows installer (cross-compiled) (push) Successful in 2m6s
Desktop (Tauri) / Tauri desktop (Linux) (push) Successful in 4m16s
Desktop (Tauri) / Update manifest (push) Successful in 6s
URLs unfurl on their own, and a lone link becomes the note
Operator: *"I'd like for URLs to unfurl. To be the whole note when the note is a
single URL, and to be a compact slot on the bottom of the note when the URL is
inline. We also need to support multiple URLs in a single note."*

Less new machinery than it sounds: `unfurl.py` already fetched and parsed OG
tags, SSRF-hardened, and `note_link_previews` was already `UNIQUE(note_id, url)`
— so several URLs per note has worked at the storage layer all along. What was
missing was that it needed a button, had one size, and drew that size in the
wrong place.

**Automatic, and never in the way.** New `unfurl_queue.py` detects a body's URLs
and fetches them on a background task AFTER the note is committed. Capture speed
is the product: an unfurl is a five-second timeout against a host nobody
controls, and a note has to persist the instant someone stops typing. Scheduled
from create, from a body edit, and from a synced push — so a linked desktop or
Android client gets previews too, on its next pull. An unlinked one has no server
to ask and simply has none, which is the honest consequence of being offline.

Safe to call on every save: it re-reads what's cached and does nothing when
nothing is new. Capped at five URLs per note, silent on every failure (a link
that won't fetch isn't an error the person needs — the note is fine, the link is
still there), and it re-checks before storing, so a slow fetch can't resurrect a
preview for a URL that was deleted while it was in flight.

**Two presentations.** A note whose body is nothing but a URL renders as its
preview and nothing else — printing the raw URL under a card that already says
where it goes is saying the same thing twice, badly. Until the fetch lands, or if
it never does, the URL stands in, so the card is never blank. Anything else gets
a compact strip.

**And the strip moved.** Previews were rendered ABOVE the body, which put a
stranger's headline where the note's own first line should be — worse now that
the first line IS the note's name. They sit at the foot of the card now, under
the note's own words.

The editor's "Preview example.com" button is gone with the manual path; removing
an unwanted preview stays, and stays editor-only.

Nine tests: three on detection (order, dedupe, sentence-punctuation trimming,
non-http rejection) in the unit lane, and three in the integration lane for what
only a real database shows — the upsert landing on the right row, a second pass
fetching nothing, and a preview NOT being stored for a URL that left the body.
2026-08-23 01:05:37 -04:00

409 lines
15 KiB
Python

from datetime import datetime, timezone
import pytest
from thoughtsync.app import create_app
from thoughtsync.common import coerce_bool, parse_dt
from thoughtsync.models.note import NOTE_COLORS, Note
from thoughtsync.unfurl_queue import detect_urls
from thoughtsync.notes import (
_attachment_ext,
_header_filename,
_keep_spec,
_native_spec,
_safe_filename,
_slugify,
_usec_to_dt,
derive_display_title,
is_empty_note,
next_occurrence,
normalize_color,
normalize_recurrence,
parse_list_items,
parse_tags,
)
@pytest.fixture
def app():
return create_app()
def test_all_note_routes_registered(app):
# Guards the notes package split: every route handler must still be attached to the
# blueprint. A route whose module isn't imported by notes/__init__ would silently
# 404 at runtime, and most routes have no auth-guard test to otherwise catch it.
registered = {r.endpoint for r in app.url_map.iter_rules()}
expected = {
f"notes.{name}"
for name in (
"list_notes", "search_notes", "list_reminders", "complete_reminder",
"snooze_reminder", "export_notes", "import_notes", "list_titles",
"reorder_notes", "create_note",
"get_note", "update_note", "list_revisions", "restore_revision",
"set_note_labels", "add_item", "update_item", "delete_item",
"reorder_items", "upload_attachment", "get_attachment",
"delete_attachment", "unfurl_link", "delete_preview", "trash_note",
"restore_note", "delete_note",
)
}
assert expected <= registered, f"unregistered note routes: {expected - registered}"
def test_is_empty_note():
assert is_empty_note(None, None)
assert is_empty_note(" ", [])
assert not is_empty_note("body")
# A note that is only a checklist is not empty — it just has nothing in its body.
assert not is_empty_note("", ["milk"])
def test_normalize_color():
assert normalize_color("blue") == "blue"
assert normalize_color("chartreuse") == "default"
assert normalize_color(None) == "default"
assert normalize_color(123) == "default"
def test_palette_has_core_colors():
for c in ("default", "red", "orange", "yellow", "green", "teal", "blue", "purple", "pink", "gray"):
assert c in NOTE_COLORS
def test_serialize_shape():
n = Note(body="b", color="blue", pinned=True, archived=False)
s = n.serialize()
assert "title" not in s # there is no title field any more (M13 step 3)
assert s["body"] == "b"
assert s["color"] == "blue"
assert s["pinned"] is True
assert s["archived"] is False
assert s["trashed"] is False
async def test_notes_list_requires_auth(app):
client = app.test_client()
resp = await client.get("/api/notes")
assert resp.status_code == 401
async def test_notes_create_requires_auth(app):
client = app.test_client()
resp = await client.post("/api/notes", json={"body": "hi"})
assert resp.status_code == 401
async def test_search_requires_auth(app):
client = app.test_client()
resp = await client.get("/api/notes/search?q=hello")
assert resp.status_code == 401
async def test_add_item_requires_auth(app):
client = app.test_client()
resp = await client.post("/api/notes/00000000-0000-0000-0000-000000000000/items", json={"text": "x"})
assert resp.status_code == 401
async def test_upload_attachment_requires_auth(app):
client = app.test_client()
resp = await client.post("/api/notes/00000000-0000-0000-0000-000000000000/attachments")
assert resp.status_code == 401
async def test_reorder_requires_auth(app):
client = app.test_client()
resp = await client.post("/api/notes/reorder", json={"ids": []})
assert resp.status_code == 401
# [[wiki-links]] are gone entirely (note 2897), and with them backlinks, the graph,
# the name index and the `[[` autocomplete. So are the two helpers that used to keep
# links alive across a rename, and the id-binding that briefly replaced them. Nothing
# here asserts their absence — `test_all_note_routes_registered` below is what would
# notice a route coming back, and the removal is one commit rather than a fossil.
def test_derive_display_title_is_the_first_body_line():
assert derive_display_title("first line\nsecond line") == "first line"
assert derive_display_title(" spaced first \nnext") == "spaced first"
# leading blank/whitespace lines are skipped to the first line with content
assert derive_display_title("\n \nreal line\nmore") == "real line"
def test_derive_display_title_falls_back_to_the_first_item():
# What step 2 bought: a note that is only a checklist still has a name. Without
# this it would have none at all, which is why the title could not go first.
assert derive_display_title("", "milk") == "milk"
assert derive_display_title(" \n ", " eggs ") == "eggs"
# The body still wins when it has anything to say.
assert derive_display_title("shopping", "milk") == "shopping"
def test_derive_display_title_empty():
assert derive_display_title(None) == ""
assert derive_display_title("") == ""
assert derive_display_title(" \n ", None) == ""
assert derive_display_title(" \n ", " ") == ""
def test_derive_display_title_caps_length():
long = "x" * 300
assert derive_display_title(long) == "x" * 200
# the item fallback is capped on the same rule
assert derive_display_title("", long) == "x" * 200
def test_parse_tags():
assert parse_tags("buy milk #groceries and #to-do now") == ["groceries", "to-do"]
# case-insensitive dedup, first spelling wins
assert parse_tags("#Work then #work") == ["Work"]
# url fragments, mid-word #, purely-numeric, and a bare # are not tags
assert parse_tags("frag http://x/#nope mid#word #2024 #") == []
assert parse_tags(None) == []
assert parse_tags("#a #b #a") == ["a", "b"]
def test_parse_list_items():
assert parse_list_items(["milk", " eggs ", "", " ", "bread"]) == ["milk", "eggs", "bread"]
assert parse_list_items("not a list") == []
assert parse_list_items(None) == []
assert parse_list_items([1, "x", None, {"a": 1}]) == ["x"]
def test_parse_dt():
# A full ISO instant round-trips (used to validate the Timeline date range).
d = parse_dt("2026-07-19T12:30:00+00:00")
assert (d.year, d.month, d.day, d.hour, d.minute) == (2026, 7, 19, 12, 30)
assert d.tzinfo is not None
# a trailing Z is accepted as UTC
assert parse_dt("2026-07-19T00:00:00Z").tzinfo is not None
# a plain calendar date parses to midnight
assert parse_dt("2026-07-19").hour == 0
# garbage / non-strings return None (the endpoint turns this into a 400)
assert parse_dt("not-a-date") is None
assert parse_dt("") is None
assert parse_dt(None) is None
async def test_titles_requires_auth(app):
client = app.test_client()
resp = await client.get("/api/notes/titles")
assert resp.status_code == 401
async def test_reminders_requires_auth(app):
client = app.test_client()
resp = await client.get("/api/notes/reminders")
assert resp.status_code == 401
def test_slugify():
assert _slugify("My Great Note!") == "my-great-note"
assert _slugify(" spaced / weird __name ") == "spaced-weird-name"
assert _slugify("") == "note" # empty falls back
assert _slugify("!!!") == "note" # all punctuation strips to empty → fallback
assert len(_slugify("x" * 100)) == 60
async def test_export_requires_auth(app):
client = app.test_client()
resp = await client.get("/api/notes/export")
assert resp.status_code == 401
async def test_list_revisions_requires_auth(app):
client = app.test_client()
resp = await client.get("/api/notes/00000000-0000-0000-0000-000000000000/revisions")
assert resp.status_code == 401
async def test_restore_revision_requires_auth(app):
client = app.test_client()
resp = await client.post(
"/api/notes/00000000-0000-0000-0000-000000000000/revisions/00000000-0000-0000-0000-000000000001/restore"
)
assert resp.status_code == 401
async def test_import_requires_auth(app):
client = app.test_client()
resp = await client.post("/api/notes/import")
assert resp.status_code == 401
def test_safe_filename():
assert _safe_filename("report.pdf") == "report.pdf"
assert _safe_filename("/etc/passwd") == "passwd" # path components stripped
assert _safe_filename("a\\b\\c.doc") == "c.doc" # windows separators too
assert _safe_filename("") == "file" # fallback
assert _safe_filename(None) == "file"
def test_attachment_ext():
assert _attachment_ext("report.pdf", "application/pdf") == ".pdf"
assert _attachment_ext("memo.m4a", "audio/mp4") == ".m4a"
# no extension in the name → fall back to a known image mime, else empty
assert _attachment_ext("noext", "image/png") == ".png"
assert _attachment_ext("noext", "application/octet-stream") == ""
def test_header_filename():
# Quotes/newlines are stripped so the Content-Disposition header can't be broken.
assert _header_filename('a"b\r\n.pdf') == "ab.pdf"
assert _header_filename("") == "file"
def test_coerce_bool():
assert coerce_bool("true") and coerce_bool("1") and coerce_bool("yes") and coerce_bool("on")
assert coerce_bool(True)
assert not coerce_bool("false")
assert not coerce_bool(None)
assert not coerce_bool("")
assert not coerce_bool(False)
def test_normalize_recurrence():
for v in ("daily", "weekly", "monthly", "yearly"):
assert normalize_recurrence(v) == v
assert normalize_recurrence("none") is None
assert normalize_recurrence("") is None
assert normalize_recurrence(None) is None
assert normalize_recurrence("hourly") is None
def test_next_occurrence_daily_weekly():
base = datetime(2026, 7, 1, 9, 0, tzinfo=timezone.utc)
after = datetime(2026, 7, 1, 12, 0, tzinfo=timezone.utc) # same day, later
assert next_occurrence(base, "daily", after) == datetime(2026, 7, 2, 9, 0, tzinfo=timezone.utc)
assert next_occurrence(base, "weekly", after) == datetime(2026, 7, 8, 9, 0, tzinfo=timezone.utc)
def test_next_occurrence_skips_missed():
base = datetime(2026, 7, 1, 9, 0, tzinfo=timezone.utc)
after = datetime(2026, 7, 10, 12, 0, tzinfo=timezone.utc) # 9+ days later
# Rolls forward past every missed day to the first fire strictly after `after`.
assert next_occurrence(base, "daily", after) == datetime(2026, 7, 11, 9, 0, tzinfo=timezone.utc)
def test_next_occurrence_monthly_clamps_month_end():
base = datetime(2026, 1, 31, 8, 0, tzinfo=timezone.utc)
after = datetime(2026, 2, 1, 0, 0, tzinfo=timezone.utc)
# Jan 31 + 1 month → Feb 28 (clamped to the shorter month).
assert next_occurrence(base, "monthly", after) == datetime(2026, 2, 28, 8, 0, tzinfo=timezone.utc)
def test_next_occurrence_yearly_and_none():
base = datetime(2026, 3, 15, 7, 0, tzinfo=timezone.utc)
after = datetime(2026, 3, 16, tzinfo=timezone.utc)
assert next_occurrence(base, "yearly", after) == datetime(2027, 3, 15, 7, 0, tzinfo=timezone.utc)
assert next_occurrence(base, "none", after) is None
async def test_complete_reminder_requires_auth(app):
client = app.test_client()
resp = await client.post("/api/notes/00000000-0000-0000-0000-000000000000/reminder/complete")
assert resp.status_code == 401
async def test_snooze_reminder_requires_auth(app):
client = app.test_client()
resp = await client.post(
"/api/notes/00000000-0000-0000-0000-000000000000/reminder/snooze", json={"minutes": 10}
)
assert resp.status_code == 401
def test_usec_to_dt():
# Google Keep timestamps are microseconds since the epoch (UTC).
d = _usec_to_dt(1600000000000000)
assert d is not None and d.year == 2020 and d.tzinfo is not None
# garbage / missing → None (the note still imports, just without the timestamp)
assert _usec_to_dt("nope") is None
assert _usec_to_dt(None) is None
def test_keep_spec_list_note_keeps_its_text_too():
# Keep's own notes carry one or the other, but its textContent used to be
# DISCARDED whenever a note also had listContent, because a note could only be
# one kind. A note holds both now, so nothing is dropped on the way in.
kn = {
"title": "Groceries",
"textContent": "for the weekend",
"listContent": [{"text": "Milk", "isChecked": False}, {"text": "Eggs", "isChecked": True}],
"labels": [{"name": "shopping"}],
"color": "TEAL",
"isPinned": True,
"isArchived": False,
"isTrashed": False,
"createdTimestampUsec": 1600000000000000,
"userEditedTimestampUsec": 1600000100000000,
}
spec = _keep_spec(kn, "Takeout/Keep")
assert spec["body"] == "for the weekend"
assert spec["color"] == "teal"
assert spec["pinned"] is True
assert spec["archived"] is False
assert spec["trashed"] is False
assert spec["items"] == [{"text": "Milk", "checked": False}, {"text": "Eggs", "checked": True}]
assert spec["labels"] == ["shopping"]
assert spec["created_at"].year == 2020
def test_keep_spec_text_note_folds_annotation_urls_and_maps_color():
kn = {
"textContent": "Read this later",
"annotations": [{"url": "https://example.com"}],
"color": "BROWN", # no brown in our palette → nearest (orange)
"attachments": [{"filePath": "img.jpg", "mimetype": "image/jpeg"}],
}
spec = _keep_spec(kn, "Takeout/Keep")
assert "https://example.com" in spec["body"]
assert spec["color"] == "orange"
# attachment path is resolved relative to the note JSON's folder
assert spec["attachments"] == [{"file": "Takeout/Keep/img.jpg", "mime": "image/jpeg"}]
def test_native_spec_roundtrip_fields():
n = {
"title": "T",
"body": "b",
"color": "blue",
"pinned": True,
"archived": False,
"created_at": "2026-07-19T00:00:00+00:00",
"labels": ["x"],
"items": [],
"attachments": [{"file": "attachments/ab/img.png", "mime": "image/png"}],
}
spec = _native_spec(n)
# The spec still CARRIES a title — an export taken before M13 has one, and
# _create_imported_note folds it into the body rather than dropping it.
assert spec["title"] == "T"
assert spec["body"] == "b"
assert spec["color"] == "blue"
assert spec["pinned"] is True
assert spec["trashed"] is False # exports only carry live notes
assert spec["created_at"].year == 2026
assert spec["labels"] == ["x"]
assert spec["attachments"] == [{"file": "attachments/ab/img.png", "mime": "image/png"}]
def test_detect_urls_finds_each_link_once_in_order():
body = "see https://example.com/a and https://example.com/b\nand https://example.com/a again"
assert detect_urls(body) == ["https://example.com/a", "https://example.com/b"]
def test_detect_urls_trims_sentence_punctuation():
# A URL can end in most punctuation; a SENTENCE containing one usually doesn't.
assert detect_urls("read https://example.com/page.") == ["https://example.com/page"]
assert detect_urls("(see https://example.com/x)") == ["https://example.com/x"]
# …but a path that legitimately ends in a slash or a dash keeps it.
assert detect_urls("https://example.com/dir/") == ["https://example.com/dir/"]
def test_detect_urls_ignores_non_http():
assert detect_urls("ftp://example.com and mailto:a@b.c and bare example.com") == []
assert detect_urls(None) == []
assert detect_urls("") == []