"""Export/import helpers (the non-route logic). Export renders each note as Markdown; import normalizes a ThoughtSync export OR a Google Keep Takeout zip into a common spec and materializes notes — with a decompression-size budget so an import zip bomb can't exhaust memory/disk.""" from __future__ import annotations import hashlib import json import os import posixpath import uuid import zipfile from datetime import datetime, timezone from sqlalchemy import select from ..colors import normalize_color from ..common import parse_dt from ..config import Config from ..models.label import NoteLabel from ..models.note import Note from ..models.note_attachment import NoteAttachment from ..models.note_item import NoteItem from .helpers import ( ALLOWED_IMAGE_MIMES, _attachment_ext, _safe_filename, derive_display_title, is_empty_note, ) from .tags import _find_or_create_label, _reconcile_tags from .recurrence import normalize_recurrence def _note_markdown(note: Note, labels: list, items: list) -> str: """One note as a human-readable Markdown file with a small frontmatter block. The authoritative machine format is notes.json; this is for reading/portability.""" fm = ["---"] fm.append(f"display_name: {note.display_title}") if labels: fm.append("labels: [" + ", ".join(lb["name"] for lb in labels) + "]") fm.append(f"color: {note.color}") if note.pinned: fm.append("pinned: true") if note.archived: fm.append("archived: true") if note.remind_at: fm.append(f"remind_at: {note.remind_at.isoformat()}") fm.append(f"created: {note.created_at.isoformat() if note.created_at else ''}") fm.append(f"updated: {note.updated_at.isoformat() if note.updated_at else ''}") fm.append("---") fm.append("") # Body and checklist are no longer alternatives — a note can carry both, so both # are written, body first, with a blank line between them when there is. if note.body: fm.append(note.body) if items: if note.body: fm.append("") for it in items: fm.append(f"- [{'x' if it['checked'] else ' '}] {it['text']}") return "\n".join(fm) + "\n" # --- Import: ThoughtSync's own export (round-trip) OR a Google Keep Takeout zip --- # Google Keep (Takeout) color enum → our palette. Keep has a few hues we don't # (BROWN/DARKBLUE/CERULEAN); map each to the nearest. Unknowns fall back to default. _KEEP_COLOR_MAP = { "DEFAULT": "default", "RED": "red", "ORANGE": "orange", "YELLOW": "yellow", "GREEN": "green", "TEAL": "teal", "CERULEAN": "teal", "BLUE": "blue", "DARKBLUE": "blue", "PURPLE": "purple", "PINK": "pink", "BROWN": "orange", "GRAY": "gray", } # Reverse of ALLOWED_IMAGE_MIMES, for inferring an attachment's mime from its # filename when the source didn't record one (Keep usually does; be defensive). _EXT_MIME = {ext: mime for mime, ext in ALLOWED_IMAGE_MIMES.items()} _EXT_MIME[".jpeg"] = "image/jpeg" def _usec_to_dt(usec: object) -> datetime | None: """Google Keep timestamps are integer MICROseconds since the Unix epoch (UTC).""" try: return datetime.fromtimestamp(int(usec) / 1_000_000, tz=timezone.utc) except (TypeError, ValueError, OverflowError, OSError): return None def _native_spec(n: dict) -> dict: """Normalize one note from a ThoughtSync export's notes.json into the common import spec consumed by _create_imported_note.""" return { "title": n.get("title"), "body": n.get("body") or "", "color": n.get("color"), "pinned": bool(n.get("pinned")), "archived": bool(n.get("archived")), "trashed": False, # export only includes live notes "remind_at": parse_dt(n.get("remind_at")), "recurrence": normalize_recurrence(n.get("recurrence")), "created_at": parse_dt(n.get("created_at")), "updated_at": parse_dt(n.get("updated_at")), "labels": [s for s in (n.get("labels") or []) if isinstance(s, str)], "items": [ {"text": it.get("text"), "checked": bool(it.get("checked"))} for it in (n.get("items") or []) if isinstance(it, dict) ], # export writes attachments[].file as the zip-internal path already. "attachments": [ {"file": a.get("file"), "mime": a.get("mime")} for a in (n.get("attachments") or []) if isinstance(a, dict) and a.get("file") ], } def _keep_spec(kn: dict, keep_dir: str) -> dict: """Normalize one Google Keep note (Takeout .json) into the common import spec. `keep_dir` is the note JSON's folder, used to resolve attachment paths.""" list_content = kn.get("listContent") if isinstance(kn.get("listContent"), list) else [] # Keep's own notes are one or the other, but its text was being DISCARDED whenever # a note also had list content, because the target model could only hold one. # It can hold both now, so both are kept. body = kn.get("textContent") or "" # Keep stores link annotations (e.g. shared URLs) separately from the text — # fold any URLs into the body so the content survives the move. urls = [ ann.get("url") for ann in (kn.get("annotations") or []) if isinstance(ann, dict) and ann.get("url") ] extra = "\n".join(u for u in urls if u and u not in body) if extra: body = f"{body}\n\n{extra}" if body.strip() else extra attachments = [] for a in kn.get("attachments") or []: if not isinstance(a, dict): continue fp = a.get("filePath") if not fp: continue zpath = posixpath.join(keep_dir, fp) if keep_dir else fp mime = a.get("mimetype") or _EXT_MIME.get(posixpath.splitext(fp)[1].lower()) attachments.append({"file": zpath, "mime": mime}) return { "title": kn.get("title"), "body": body, "color": _KEEP_COLOR_MAP.get(str(kn.get("color") or "DEFAULT").upper(), "default"), "pinned": bool(kn.get("isPinned")), "archived": bool(kn.get("isArchived")), "trashed": bool(kn.get("isTrashed")), "remind_at": None, # Keep reminders aren't in Takeout note JSON "created_at": _usec_to_dt(kn.get("createdTimestampUsec")), "updated_at": _usec_to_dt(kn.get("userEditedTimestampUsec")), "labels": [ lb.get("name") for lb in (kn.get("labels") or []) if isinstance(lb, dict) and lb.get("name") ], "items": [ {"text": li.get("text"), "checked": bool(li.get("isChecked"))} for li in list_content if isinstance(li, dict) ], "attachments": attachments, } IMPORT_MAX_ENTRIES = 10_000 IMPORT_MAX_ENTRY_BYTES = 64 * 1024 * 1024 # 64 MB decompressed per file IMPORT_MAX_TOTAL_BYTES = 512 * 1024 * 1024 # 512 MB decompressed across the whole import class _ImportTooLarge(Exception): """An import zip decompressed past the byte budget (a zip bomb, or just too big).""" class _ImportBudget: """Caps DECOMPRESSED bytes pulled from an import zip — per entry and cumulatively. zipfile inflates into memory on read, so an archive that's tiny on disk can expand to gigabytes. We stream each entry and read at most the remaining budget + 1 byte, so an oversized (or size-lying) entry is caught mid-read instead of after it has already been fully inflated.""" def __init__(self) -> None: self.remaining = IMPORT_MAX_TOTAL_BYTES def read(self, zf: zipfile.ZipFile, name: str) -> bytes: cap = min(IMPORT_MAX_ENTRY_BYTES, self.remaining) with zf.open(name) as fh: data = fh.read(cap + 1) if len(data) > cap: raise _ImportTooLarge() self.remaining -= len(data) return data def _read_import_specs(zf: zipfile.ZipFile, budget: _ImportBudget) -> tuple[list[dict], str]: """Detect the archive format and return (specs, source). A ThoughtSync export is recognized by its notes.json (app == thoughtsync); otherwise each Keep-shaped .json is imported. Returns ([], "") when nothing importable is found.""" names = zf.namelist() for name in names: if posixpath.basename(name) == "notes.json": try: doc = json.loads(budget.read(zf, name)) except (ValueError, KeyError): continue if isinstance(doc, dict) and doc.get("app") == "thoughtsync": specs = [_native_spec(n) for n in (doc.get("notes") or []) if isinstance(n, dict)] return specs, "thoughtsync" keep_specs: list[dict] = [] keep_keys = ("textContent", "listContent", "isPinned", "isArchived", "isTrashed", "userEditedTimestampUsec") for name in names: if not name.lower().endswith(".json") or posixpath.basename(name) == "notes.json": continue try: kn = json.loads(budget.read(zf, name)) except (ValueError, KeyError): continue if isinstance(kn, dict) and any(k in kn for k in keep_keys): keep_specs.append(_keep_spec(kn, posixpath.dirname(name))) return (keep_specs, "keep") if keep_specs else ([], "") def _import_attachment(db, note: Note, zf: zipfile.ZipFile, att: dict, budget: _ImportBudget) -> bool: """Copy one attachment (any type — incl. Keep audio memos) out of the zip into media storage and record it, preserving its filename + hash. Returns True if written.""" zpath = att.get("file") if not zpath: return False try: raw = budget.read(zf, zpath) except KeyError: return False filename = _safe_filename(posixpath.basename(zpath)) mime = (att.get("mime") or "application/octet-stream").split(";")[0].strip().lower() ext = _attachment_ext(filename, mime) att_id = uuid.uuid4() rel = os.path.join(str(note.id), f"{att_id}{ext}") dest = Config.media_root() / rel dest.parent.mkdir(parents=True, exist_ok=True) dest.write_bytes(raw) db.add( NoteAttachment( id=att_id, note_id=note.id, path=rel, filename=filename, mime=mime, size=len(raw), sha256=hashlib.sha256(raw).hexdigest(), ) ) return True async def _create_imported_note( db, owner_id, spec: dict, zf: zipfile.ZipFile, position: int, budget: _ImportBudget ) -> bool: """Insert one imported note plus its items/labels/attachments, reusing the same name derivation + tag reconciliation as create_note. Returns False (nothing written) when the spec is empty.""" body = spec.get("body") or "" # An imported title becomes the note's FIRST BODY LINE. # # ThoughtSync has no title field any more (M13 step 3), but the things people # import from do — Keep notes carry one, and so does any export taken before this. # Dropping it would silently lose text someone wrote; folding it into the body puts # it exactly where a name now lives, so the note comes in named the way it was. # Skipped when the body already opens with that line, so re-importing an export # this code produced doesn't stack duplicates. title = (spec.get("title") or "").strip() if title and body.lstrip().split("\n", 1)[0].strip() != title: body = f"{title}\n{body}" if body.strip() else title items = spec.get("items") or [] item_texts = [t for t in ((it.get("text") or "").strip() for it in items) if t] if is_empty_note(body, item_texts): return False note = Note( owner_id=owner_id, display_title=derive_display_title(body, item_texts[0] if item_texts else None), body=body, color=normalize_color(spec.get("color")), pinned=bool(spec.get("pinned")), archived=bool(spec.get("archived")), position=position, ) if spec.get("remind_at"): note.remind_at = spec["remind_at"] if spec.get("recurrence"): note.recurrence = spec["recurrence"] if spec.get("trashed"): note.deleted_at = datetime.now(timezone.utc) # Preserve source timestamps: set before flush so they land in the INSERT # (updated_at's onupdate only fires on later UPDATEs, which we don't trigger). if spec.get("created_at"): note.created_at = spec["created_at"] if spec.get("updated_at"): note.updated_at = spec["updated_at"] db.add(note) await db.flush() # assign note.id before items/labels/attachments/links for pos, it in enumerate(items): text = (it.get("text") or "").strip() if text: db.add(NoteItem(note_id=note.id, text=text, checked=bool(it.get("checked")), position=pos)) # Explicit (picker-style) labels are manual — via_tag=False. Inline #tags in the # body are handled by _reconcile_tags below, same as a normal create. for name in spec.get("labels") or []: name = (name or "").strip() if not name: continue lid = await _find_or_create_label(db, owner_id, name) exists = await db.scalar( select(NoteLabel).where(NoteLabel.note_id == note.id, NoteLabel.label_id == lid) ) if exists is None: db.add(NoteLabel(note_id=note.id, label_id=lid, via_tag=False)) for att in spec.get("attachments") or []: if isinstance(att, dict): _import_attachment(db, note, zf, att, budget) await _reconcile_tags(db, note) return True