Files
inkwell/src/thoughtsync/notes/import_export.py
T
bvandeusen 95aa10c2c3
CI & Build / Build now, or wait for Android? (push) Successful in 3s
CI & Build / Python lint (push) Successful in 3s
CI & Build / TypeScript typecheck (push) Failing after 7s
CI & Build / Build & push image (push) Skipped
Desktop (Tauri) / Tauri desktop (Linux) (push) Failing after 7s
CI & Build / Python tests (push) Successful in 11s
Desktop (Tauri) / Windows installer (cross-compiled) (push) Failing after 31s
Desktop (Tauri) / Update manifest (push) Skipped
Android / Kotlin + Rust (APK) (push) Successful in 6m45s
Remove the title field — a note is named by its first line
Operator (note 2897): "notes shouldn't have a title field." The concept of a NAME
stays — search results, export filenames and the command palette all need one —
but nothing is typed into it any more. `display_title` is now the first non-empty
line of the body, falling back to the first checklist item.

That fallback is what step 2 bought, and the reason this could not go first: a
checklist had no body to be named from, so the title was its only name. Now every
note has a body, and a note that is only a checklist is named by its first item.

Gone everywhere: the column and note_revisions.title (0026), the field on the
core's Note/NoteCreateInput/NoteRevision and its SQLite columns (user_version 7),
`normalize_title`, the wire field, the FFI record and `NoteEdit::Title` /
`ClearTitle`, the web editor's "Title (optional)" input and the card's <h3>, and
the Android title field in both the compose sheet and the editor.

**The search vector had to be rebuilt, not just left alone.** `notes.search_vector`
is a STORED GENERATED column whose expression names `title` — Postgres refuses to
drop a column another generated column depends on. It is dropped and recreated over
`display_title` at weight A, which keeps the original intent: a note's NAME ranks
above the rest of its body.

**An imported title becomes the note's first body line.** Keep notes carry one, and
so does any ThoughtSync export taken before this. Dropping it would silently lose
text someone wrote; folding it in puts it exactly where a name now lives, so the
note arrives named as it was. Skipped when the body already opens with that line,
so re-importing an export this code produced doesn't stack duplicates.

Two smaller things fell out. The Android editor loses its bold first field — one
weight throughout, because the first line is the note's name but not a different
KIND of text, which is most of step 4 arriving early. And `ClearTitle`'s
justification comment moved to `ClearRemindAt`, which is now the surviving example
of why NoteEdit is a list rather than a struct of options.

Protocol note corrected to say what actually shipped: v2 is "no kind, no title",
one bump for the pair.

Verified with the local Rust gate this time, not by CI: fmt, clippy and 116 tests
all green before pushing. It caught four things — orphaned serde attributes where
fields were removed, a `wire::Preview.title` I deleted by mistake (a link preview
still has one), nine retention fixtures inserting a dropped column, and four
rustfmt diffs.
2026-08-22 19:33:57 -04:00

345 lines
13 KiB
Python

"""Export/import helpers (the non-route logic). Export renders each note as Markdown;
import normalizes a ThoughtSync export OR a Google Keep Takeout zip into a common spec
and materializes notes — with a decompression-size budget so an import zip bomb can't
exhaust memory/disk."""
from __future__ import annotations
import hashlib
import json
import os
import posixpath
import uuid
import zipfile
from datetime import datetime, timezone
from sqlalchemy import select
from ..colors import normalize_color
from ..common import parse_dt
from ..config import Config
from ..models.label import NoteLabel
from ..models.note import Note
from ..models.note_attachment import NoteAttachment
from ..models.note_item import NoteItem
from .helpers import (
ALLOWED_IMAGE_MIMES,
_attachment_ext,
_safe_filename,
derive_display_title,
is_empty_note,
)
from .tags import _find_or_create_label, _reconcile_tags
from .recurrence import normalize_recurrence
def _note_markdown(note: Note, labels: list, items: list) -> str:
"""One note as a human-readable Markdown file with a small frontmatter block.
The authoritative machine format is notes.json; this is for reading/portability."""
fm = ["---"]
fm.append(f"display_name: {note.display_title}")
if labels:
fm.append("labels: [" + ", ".join(lb["name"] for lb in labels) + "]")
fm.append(f"color: {note.color}")
if note.pinned:
fm.append("pinned: true")
if note.archived:
fm.append("archived: true")
if note.remind_at:
fm.append(f"remind_at: {note.remind_at.isoformat()}")
fm.append(f"created: {note.created_at.isoformat() if note.created_at else ''}")
fm.append(f"updated: {note.updated_at.isoformat() if note.updated_at else ''}")
fm.append("---")
fm.append("")
# Body and checklist are no longer alternatives — a note can carry both, so both
# are written, body first, with a blank line between them when there is.
if note.body:
fm.append(note.body)
if items:
if note.body:
fm.append("")
for it in items:
fm.append(f"- [{'x' if it['checked'] else ' '}] {it['text']}")
return "\n".join(fm) + "\n"
# --- Import: ThoughtSync's own export (round-trip) OR a Google Keep Takeout zip ---
# Google Keep (Takeout) color enum → our palette. Keep has a few hues we don't
# (BROWN/DARKBLUE/CERULEAN); map each to the nearest. Unknowns fall back to default.
_KEEP_COLOR_MAP = {
"DEFAULT": "default",
"RED": "red",
"ORANGE": "orange",
"YELLOW": "yellow",
"GREEN": "green",
"TEAL": "teal",
"CERULEAN": "teal",
"BLUE": "blue",
"DARKBLUE": "blue",
"PURPLE": "purple",
"PINK": "pink",
"BROWN": "orange",
"GRAY": "gray",
}
# Reverse of ALLOWED_IMAGE_MIMES, for inferring an attachment's mime from its
# filename when the source didn't record one (Keep usually does; be defensive).
_EXT_MIME = {ext: mime for mime, ext in ALLOWED_IMAGE_MIMES.items()}
_EXT_MIME[".jpeg"] = "image/jpeg"
def _usec_to_dt(usec: object) -> datetime | None:
"""Google Keep timestamps are integer MICROseconds since the Unix epoch (UTC)."""
try:
return datetime.fromtimestamp(int(usec) / 1_000_000, tz=timezone.utc)
except (TypeError, ValueError, OverflowError, OSError):
return None
def _native_spec(n: dict) -> dict:
"""Normalize one note from a ThoughtSync export's notes.json into the common
import spec consumed by _create_imported_note."""
return {
"title": n.get("title"),
"body": n.get("body") or "",
"color": n.get("color"),
"pinned": bool(n.get("pinned")),
"archived": bool(n.get("archived")),
"trashed": False, # export only includes live notes
"remind_at": parse_dt(n.get("remind_at")),
"recurrence": normalize_recurrence(n.get("recurrence")),
"created_at": parse_dt(n.get("created_at")),
"updated_at": parse_dt(n.get("updated_at")),
"labels": [s for s in (n.get("labels") or []) if isinstance(s, str)],
"items": [
{"text": it.get("text"), "checked": bool(it.get("checked"))}
for it in (n.get("items") or [])
if isinstance(it, dict)
],
# export writes attachments[].file as the zip-internal path already.
"attachments": [
{"file": a.get("file"), "mime": a.get("mime")}
for a in (n.get("attachments") or [])
if isinstance(a, dict) and a.get("file")
],
}
def _keep_spec(kn: dict, keep_dir: str) -> dict:
"""Normalize one Google Keep note (Takeout <note>.json) into the common import
spec. `keep_dir` is the note JSON's folder, used to resolve attachment paths."""
list_content = kn.get("listContent") if isinstance(kn.get("listContent"), list) else []
# Keep's own notes are one or the other, but its text was being DISCARDED whenever
# a note also had list content, because the target model could only hold one.
# It can hold both now, so both are kept.
body = kn.get("textContent") or ""
# Keep stores link annotations (e.g. shared URLs) separately from the text —
# fold any URLs into the body so the content survives the move.
urls = [
ann.get("url")
for ann in (kn.get("annotations") or [])
if isinstance(ann, dict) and ann.get("url")
]
extra = "\n".join(u for u in urls if u and u not in body)
if extra:
body = f"{body}\n\n{extra}" if body.strip() else extra
attachments = []
for a in kn.get("attachments") or []:
if not isinstance(a, dict):
continue
fp = a.get("filePath")
if not fp:
continue
zpath = posixpath.join(keep_dir, fp) if keep_dir else fp
mime = a.get("mimetype") or _EXT_MIME.get(posixpath.splitext(fp)[1].lower())
attachments.append({"file": zpath, "mime": mime})
return {
"title": kn.get("title"),
"body": body,
"color": _KEEP_COLOR_MAP.get(str(kn.get("color") or "DEFAULT").upper(), "default"),
"pinned": bool(kn.get("isPinned")),
"archived": bool(kn.get("isArchived")),
"trashed": bool(kn.get("isTrashed")),
"remind_at": None, # Keep reminders aren't in Takeout note JSON
"created_at": _usec_to_dt(kn.get("createdTimestampUsec")),
"updated_at": _usec_to_dt(kn.get("userEditedTimestampUsec")),
"labels": [
lb.get("name")
for lb in (kn.get("labels") or [])
if isinstance(lb, dict) and lb.get("name")
],
"items": [
{"text": li.get("text"), "checked": bool(li.get("isChecked"))}
for li in list_content
if isinstance(li, dict)
],
"attachments": attachments,
}
IMPORT_MAX_ENTRIES = 10_000
IMPORT_MAX_ENTRY_BYTES = 64 * 1024 * 1024 # 64 MB decompressed per file
IMPORT_MAX_TOTAL_BYTES = 512 * 1024 * 1024 # 512 MB decompressed across the whole import
class _ImportTooLarge(Exception):
"""An import zip decompressed past the byte budget (a zip bomb, or just too big)."""
class _ImportBudget:
"""Caps DECOMPRESSED bytes pulled from an import zip — per entry and cumulatively.
zipfile inflates into memory on read, so an archive that's tiny on disk can expand
to gigabytes. We stream each entry and read at most the remaining budget + 1 byte,
so an oversized (or size-lying) entry is caught mid-read instead of after it has
already been fully inflated."""
def __init__(self) -> None:
self.remaining = IMPORT_MAX_TOTAL_BYTES
def read(self, zf: zipfile.ZipFile, name: str) -> bytes:
cap = min(IMPORT_MAX_ENTRY_BYTES, self.remaining)
with zf.open(name) as fh:
data = fh.read(cap + 1)
if len(data) > cap:
raise _ImportTooLarge()
self.remaining -= len(data)
return data
def _read_import_specs(zf: zipfile.ZipFile, budget: _ImportBudget) -> tuple[list[dict], str]:
"""Detect the archive format and return (specs, source). A ThoughtSync export
is recognized by its notes.json (app == thoughtsync); otherwise each Keep-shaped
<note>.json is imported. Returns ([], "") when nothing importable is found."""
names = zf.namelist()
for name in names:
if posixpath.basename(name) == "notes.json":
try:
doc = json.loads(budget.read(zf, name))
except (ValueError, KeyError):
continue
if isinstance(doc, dict) and doc.get("app") == "thoughtsync":
specs = [_native_spec(n) for n in (doc.get("notes") or []) if isinstance(n, dict)]
return specs, "thoughtsync"
keep_specs: list[dict] = []
keep_keys = ("textContent", "listContent", "isPinned", "isArchived", "isTrashed", "userEditedTimestampUsec")
for name in names:
if not name.lower().endswith(".json") or posixpath.basename(name) == "notes.json":
continue
try:
kn = json.loads(budget.read(zf, name))
except (ValueError, KeyError):
continue
if isinstance(kn, dict) and any(k in kn for k in keep_keys):
keep_specs.append(_keep_spec(kn, posixpath.dirname(name)))
return (keep_specs, "keep") if keep_specs else ([], "")
def _import_attachment(db, note: Note, zf: zipfile.ZipFile, att: dict, budget: _ImportBudget) -> bool:
"""Copy one attachment (any type — incl. Keep audio memos) out of the zip into
media storage and record it, preserving its filename + hash. Returns True if written."""
zpath = att.get("file")
if not zpath:
return False
try:
raw = budget.read(zf, zpath)
except KeyError:
return False
filename = _safe_filename(posixpath.basename(zpath))
mime = (att.get("mime") or "application/octet-stream").split(";")[0].strip().lower()
ext = _attachment_ext(filename, mime)
att_id = uuid.uuid4()
rel = os.path.join(str(note.id), f"{att_id}{ext}")
dest = Config.media_root() / rel
dest.parent.mkdir(parents=True, exist_ok=True)
dest.write_bytes(raw)
db.add(
NoteAttachment(
id=att_id,
note_id=note.id,
path=rel,
filename=filename,
mime=mime,
size=len(raw),
sha256=hashlib.sha256(raw).hexdigest(),
)
)
return True
async def _create_imported_note(
db, owner_id, spec: dict, zf: zipfile.ZipFile, position: int, budget: _ImportBudget
) -> bool:
"""Insert one imported note plus its items/labels/attachments, reusing the same
name derivation + tag reconciliation as create_note. Returns False (nothing
written) when the spec is empty."""
body = spec.get("body") or ""
# An imported title becomes the note's FIRST BODY LINE.
#
# ThoughtSync has no title field any more (M13 step 3), but the things people
# import from do — Keep notes carry one, and so does any export taken before this.
# Dropping it would silently lose text someone wrote; folding it into the body puts
# it exactly where a name now lives, so the note comes in named the way it was.
# Skipped when the body already opens with that line, so re-importing an export
# this code produced doesn't stack duplicates.
title = (spec.get("title") or "").strip()
if title and body.lstrip().split("\n", 1)[0].strip() != title:
body = f"{title}\n{body}" if body.strip() else title
items = spec.get("items") or []
item_texts = [t for t in ((it.get("text") or "").strip() for it in items) if t]
if is_empty_note(body, item_texts):
return False
note = Note(
owner_id=owner_id,
display_title=derive_display_title(body, item_texts[0] if item_texts else None),
body=body,
color=normalize_color(spec.get("color")),
pinned=bool(spec.get("pinned")),
archived=bool(spec.get("archived")),
position=position,
)
if spec.get("remind_at"):
note.remind_at = spec["remind_at"]
if spec.get("recurrence"):
note.recurrence = spec["recurrence"]
if spec.get("trashed"):
note.deleted_at = datetime.now(timezone.utc)
# Preserve source timestamps: set before flush so they land in the INSERT
# (updated_at's onupdate only fires on later UPDATEs, which we don't trigger).
if spec.get("created_at"):
note.created_at = spec["created_at"]
if spec.get("updated_at"):
note.updated_at = spec["updated_at"]
db.add(note)
await db.flush() # assign note.id before items/labels/attachments/links
for pos, it in enumerate(items):
text = (it.get("text") or "").strip()
if text:
db.add(NoteItem(note_id=note.id, text=text, checked=bool(it.get("checked")), position=pos))
# Explicit (picker-style) labels are manual — via_tag=False. Inline #tags in the
# body are handled by _reconcile_tags below, same as a normal create.
for name in spec.get("labels") or []:
name = (name or "").strip()
if not name:
continue
lid = await _find_or_create_label(db, owner_id, name)
exists = await db.scalar(
select(NoteLabel).where(NoteLabel.note_id == note.id, NoteLabel.label_id == lid)
)
if exists is None:
db.add(NoteLabel(note_id=note.id, label_id=lid, via_tag=False))
for att in spec.get("attachments") or []:
if isinstance(att, dict):
_import_attachment(db, note, zf, att, budget)
await _reconcile_tags(db, note)
return True