"""Trash retention — what "permanently deleted" means, and when it happens by itself. Two things live here, deliberately together: **`purge_note`** — the single definition of destroying a note. Three callers reach permanent deletion by different routes (the user's Delete forever in the web UI, a client's `op=delete` over sync, and the sweeper below), and if each had its own idea of what to tear down they would drift — one would forget the files, another the revision history, and "permanently deleted" would quietly mean three different things depending on how you got there. **The sweeper** — trash that nobody empties is not free: a trashed note keeps its attachment BYTES on disk for as long as it sits there. So trash expires. The window is the `trash_retention_days` setting (default 30, `0` = keep forever), re-read on every pass so a change in admin Settings takes effect without a restart. A purged note is not a deleted ROW — it's a content-less tombstone. That's what lets an offline client that reappears next month learn the note is gone instead of faithfully resurrecting it on the next push. """ from __future__ import annotations import asyncio import logging from datetime import datetime, timedelta, timezone from sqlalchemy import delete as sa_delete from sqlalchemy import select from .config import Config from .db import session_scope from .models.label import NoteLabel from .models.note import Note from .models.note_attachment import NoteAttachment from .models.note_item import NoteItem from .models.note_link_preview import NoteLinkPreview from .models.note_revision import NoteRevision from .settings import get_setting logger = logging.getLogger(__name__) # How often the sweeper wakes. Retention is measured in days, so anything under # "a few times a day" buys nothing but load — a note trashed at 09:00 expiring at # 14:00 rather than 09:00 thirty days later is not a difference anyone can feel. SWEEP_INTERVAL_SECONDS = 6 * 60 * 60 # Let the app finish booting (migrations, first requests) before the first sweep. SWEEP_STARTUP_DELAY_SECONDS = 60 # Rows purged per transaction. A long-neglected install could have thousands of # expired notes on the first sweep; committing in batches keeps that from becoming # one enormous transaction holding locks while it deletes files. SWEEP_BATCH = 200 def expired_before(now: datetime, retention_days: int) -> datetime | None: """The cutoff: trash older than this has expired. `None` = retention is off. Kept separate from the query so the window arithmetic — including the two ways to say "never" (0 and negative, the latter reachable by typing a stray minus in Settings) — is testable without a database. """ if retention_days <= 0: return None return now - timedelta(days=retention_days) async def purge_note(db, note: Note, edited_at: datetime | None = None) -> None: """Turn a note into a content-less tombstone: delete its children (and the attachment files on disk), clear its content, stamp `purged_at`. The row survives on purpose — offline clients read it off the delta feed and learn the note is gone. Everything that carries the note's CONTENT goes, and that includes history: a revision row holds the full body, so leaving revisions behind would mean the text of a "permanently deleted" note is still on the server, recoverable by anyone who can read the table. """ atts = (await db.scalars(select(NoteAttachment).where(NoteAttachment.note_id == note.id))).all() for a in atts: try: (Config.media_root() / a.path).unlink(missing_ok=True) except OSError: # A missing or unreadable file must not strand the row: the DB record is # what the user asked us to destroy, and a failed unlink leaving it in # place would make the note reappear whole on the next sweep. logger.warning("couldn't remove attachment file %s during purge", a.path, exc_info=True) await db.execute(sa_delete(NoteAttachment).where(NoteAttachment.note_id == note.id)) await db.execute(sa_delete(NoteItem).where(NoteItem.note_id == note.id)) await db.execute(sa_delete(NoteLabel).where(NoteLabel.note_id == note.id)) await db.execute(sa_delete(NoteLinkPreview).where(NoteLinkPreview.note_id == note.id)) await db.execute(sa_delete(NoteRevision).where(NoteRevision.note_id == note.id)) note.body = "" note.display_title = "" # `deleted_at` deliberately SURVIVES. It's still true — that is when the note was # deleted — and keeping it means every ordinary query, present and future, that # says "not trashed" (`deleted_at IS NULL`) excludes tombstones for free. Clearing # it would leave a content-less row looking like a perfectly normal active note, # and it would surface on the board as a blank card. Only the Trash view, which # asks for `deleted_at IS NOT NULL`, has to name `purged_at` explicitly. note.remind_at = None note.purged_at = datetime.now(timezone.utc) if edited_at is not None: note.updated_at = edited_at async def sweep_expired_trash(db, retention_days: int, *, now: datetime | None = None) -> int: """Purge every note whose trash has expired. Returns how many were purged. Runs across ALL owners — it's a server-wide policy, not a per-user action, and the sweeper has no session to scope it by (rule 47 is about honoring the ACL on user-initiated reads, not about exempting rows from server maintenance). """ cutoff = expired_before(now or datetime.now(timezone.utc), retention_days) if cutoff is None: return 0 total = 0 while True: expired = ( await db.scalars( select(Note) .where( Note.deleted_at.is_not(None), Note.deleted_at < cutoff, # Already a tombstone. Without this the purge would re-run on # every sweep forever, bumping sync_revision each time and # handing clients an endless stream of "news" about one note. Note.purged_at.is_(None), ) .order_by(Note.deleted_at) .limit(SWEEP_BATCH) ) ).all() if not expired: return total for note in expired: await purge_note(db, note) await db.commit() total += len(expired) async def sweep_once() -> int: """One sweep against the live retention setting, in its own session.""" async with session_scope() as db: try: days = int(await get_setting(db, "trash_retention_days")) except (KeyError, TypeError, ValueError): return 0 return await sweep_expired_trash(db, days) async def run_sweeper() -> None: """The background loop. Started in `before_serving`, cancelled on shutdown. A sweep failure (DB blip, unreadable media directory) must never take the loop down with it — the next pass simply finds the same expired rows and tries again. """ await asyncio.sleep(SWEEP_STARTUP_DELAY_SECONDS) while True: try: purged = await sweep_once() if purged: logger.info("trash retention: purged %d expired note(s)", purged) except asyncio.CancelledError: raise except Exception: logger.exception("trash retention sweep failed; will retry next pass") await asyncio.sleep(SWEEP_INTERVAL_SECONDS)