"""Fire-and-forget background tasks that actually run. The event loop holds only a WEAK reference to a task, so a bare ``create_task`` with no other holder can be garbage-collected mid-flight — a write that never errors and never lands (the #2663 GC footgun). This module is the one place that gets the pattern right: strong references in ``_pending``, discarded on completion, with failures logged at WARNING instead of vanishing. ``retrieval_telemetry`` predates this module and keeps its own copy, because its canary is a genuinely different shape — one process-wide flag and no AppLog row. ``note_usage`` and ``rule_usage`` share ``report_telemetry_failure`` below. New fire-and-forget callers use ``spawn`` rather than writing another copy of the strong-reference dance. """ from __future__ import annotations import asyncio import logging import traceback from collections.abc import Coroutine logger = logging.getLogger(__name__) _pending: set[asyncio.Task] = set() # Sites that have already dropped their once-per-process AppLog row, keyed # ":". A readout can run on every list render — without this, # a broken table turns the error log into a firehose that buries the finding it # exists to surface. _reported: set[str] = set() async def report_telemetry_failure(subsystem: str, site: str) -> None: """Make a swallowed telemetry failure visible. Call from an except block. WARNING to the process log every time; one AppLog error row per process per (subsystem, site) so the admin UI shows the outage without host access. THIS IS NOT DECORATION. #2663 is the record of a telemetry subsystem running at zero for weeks — every counter reading empty, indistinguishable from "nobody uses this" — because every failure went to ``logger.debug``. A subsystem whose failures are all invisible cannot report its own death. The AppLog write is itself guarded: when the database is down it fails too, and that is fine. The WARNING already said so, and a canary must never take down the surface it watches. """ logger.warning("%s telemetry %s failed", subsystem, site, exc_info=True) key = f"{subsystem}:{site}" if key in _reported: return _reported.add(key) try: from scribe.services.logging import log_error await log_error( endpoint=subsystem, error_type=f"{subsystem}_{site}_failed", error_message=f"{subsystem} telemetry {site} is failing; " "usage counters will read zero until this is fixed", traceback=traceback.format_exc(), ) except Exception: logger.debug("%s canary write failed", subsystem, exc_info=True) def spawn(coro: Coroutine, *, site: str) -> None: """Schedule ``coro`` fire-and-forget; ``site`` names it in failure logs. No running loop (sync context outside the app) closes the coroutine and skips — every app path runs on the loop, and blocking would be worse. """ try: task = asyncio.get_running_loop().create_task(coro) except RuntimeError: coro.close() logger.debug("background task %s skipped — no running event loop", site) return _pending.add(task) def _done(t: asyncio.Task) -> None: _pending.discard(t) if not t.cancelled() and t.exception() is not None: logger.warning( "background task %s failed", site, exc_info=t.exception() ) task.add_done_callback(_done) def start_periodic(interval_s: float, work, *, label: str) -> asyncio.Task: """A forever loop that sleeps ``interval_s`` then awaits ``work()``, logging (never raising) when a tick fails — the one shape the hourly/daily retention sweeps share (log retention, notification sweep, auth-token purge). Sleeps FIRST so startup isn't a sweep; holds a strong reference like spawn() so the loop cannot be garbage-collected mid-flight.""" async def _loop() -> None: while True: await asyncio.sleep(interval_s) try: await work() except Exception: logger.exception("periodic task %s failed", label) task = asyncio.get_running_loop().create_task(_loop(), name=f"periodic-{label}") _pending.add(task) task.add_done_callback(_pending.discard) return task async def drain() -> None: """Await everything in flight — for tests that need the writes landed.""" while _pending: await asyncio.gather(*list(_pending), return_exceptions=True)