"""The learned roster: which of FabledCurator's parts have checked in, and when. Milestone 365. `celery inspect` answers "who is here"; this answers "who is missing", which nothing in the application could do before — see `models/service_seen.py` for why the identity is a queue set and not a worker hostname. ## Who does the observing, and why it is the web process Three candidates, and the choice matters more than the code: * **A celery beat sweep.** Rejected. If the scheduler dies, the sweep stops, every row goes stale, and the page reports that everything is down when one thing is. An alarm that cannot distinguish "one part died" from "the observer died" is worse than no alarm. * **A background task in web.** Rejected on a detail of how this deploys: hypercorn runs `--workers 4`, so a `before_serving` loop would be FOUR concurrent inspect loops hammering the broker, forever, per container. * **Refresh on demand, rate-limited by the data itself.** Taken. Whichever web process happens to serve a health request refreshes the roster if it is older than REFRESH_TTL, and otherwise reads what is already there. The third has the property the other two lack: **the observer is the thing serving the page.** If web is down you get a browser error rather than a confidently green page, which is the honest failure. It also self-limits without coordination — the TTL lives in the row everybody can see. """ from __future__ import annotations import asyncio import logging from sqlalchemy import func, select from sqlalchemy.dialects.postgresql import insert as pg_insert from sqlalchemy.ext.asyncio import AsyncSession from ..models import ServiceSeen log = logging.getLogger(__name__) # How stale the roster may be before a health request refreshes it. Comfortably # under the staleness thresholds that decide a service is missing, so the # verdict is never limited by how often anyone looked. REFRESH_TTL_SECONDS = 20.0 # celery inspect is a broker round trip and this sits on a request path, so it # gets a deadline (rule 156). A broker that has stopped answering must make the # roster stale — which is a true statement about the system — not hang the one # page that exists to explain it. INSPECT_TIMEOUT_SECONDS = 2.0 # Queue set -> the name an operator recognises. Sorted-tuple keys, because the # order celery reports them in is not guaranteed. # # A deployment that slices CELERY_QUEUES differently falls through to the raw # queue list rather than being given a name this table invented for it: a # wrong-but-confident label on a status page is worse than an ugly true one. ROLE_NAMES: dict[tuple[str, ...], str] = { ("default", "download", "import", "thumbnail"): "Worker", ("maintenance", "scan"): "Scheduler", ("ml",): "ML worker", } def role_display_name(queues: tuple[str, ...]) -> str: known = ROLE_NAMES.get(queues) if known: return known return "Worker (" + ", ".join(queues) + ")" def _inspect_celery_sync() -> dict[tuple[str, ...], dict]: """celery inspect, grouped by queue set rather than by worker. Returns {queue_set: {"hostnames": [...], "active": int}}. Two replicas of one role collapse into one entry on purpose — the question is whether the role is being served, not how many containers exist. """ from ..celery_app import celery as celery_app insp = celery_app.control.inspect(timeout=INSPECT_TIMEOUT_SECONDS) active_queues = insp.active_queues() or {} active_tasks = insp.active() or {} grouped: dict[tuple[str, ...], dict] = {} for hostname, queues in active_queues.items(): key = tuple(sorted({q["name"] for q in queues})) entry = grouped.setdefault(key, {"hostnames": [], "active": 0}) entry["hostnames"].append(hostname) entry["active"] += len(active_tasks.get(hostname, [])) for entry in grouped.values(): entry["hostnames"].sort() return grouped async def touch_service( session: AsyncSession, *, key: str, kind: str, display_name: str, details: dict ) -> None: """Record that a part checked in just now. Upsert rather than read-modify-write: several web processes and several agents can be doing this at once, and the last writer is simply the most recent sighting. `first_seen_at` is deliberately NOT updated — it is the one field that answers "has this ever run", which the learned-roster design depends on. """ stmt = pg_insert(ServiceSeen).values( key=key, kind=kind, display_name=display_name, details=details, ) stmt = stmt.on_conflict_do_update( index_elements=[ServiceSeen.key], set_={ "kind": stmt.excluded.kind, "display_name": stmt.excluded.display_name, "details": stmt.excluded.details, "last_seen_at": func.now(), }, ) await session.execute(stmt) async def refresh_celery_roster(session: AsyncSession) -> None: """Inspect the broker and record what answered. Never raises. A failure here means the roster does not advance, and the rows going stale is then a TRUE report about a broker nobody can reach. Letting the exception out would instead break the health endpoint, which is the one thing that must keep answering when the stack is unwell. """ try: grouped = await asyncio.wait_for( asyncio.to_thread(_inspect_celery_sync), timeout=INSPECT_TIMEOUT_SECONDS * 2, ) except Exception: log.warning("service roster: celery inspect failed; roster not refreshed", exc_info=True) return for queues, entry in grouped.items(): await touch_service( session, key="celery:" + ",".join(queues), kind="celery", display_name=role_display_name(queues), details={ "queues": list(queues), "hostnames": entry["hostnames"], "replicas": len(entry["hostnames"]), "active": entry["active"], }, ) async def refresh_if_stale(session: AsyncSession) -> None: """Refresh the celery roster if nobody has for REFRESH_TTL_SECONDS. Rate-limited by the data rather than by a lock: the gate is the newest last_seen_at across the celery rows, which every web process can see. Two processes racing through the gate costs one redundant inspect and writes the same values twice, so the benign outcome needs no coordination to prevent. """ newest = ( await session.execute( select(func.max(ServiceSeen.last_seen_at)).where(ServiceSeen.kind == "celery") ) ).scalar_one_or_none() if newest is not None: age = (await session.execute(select(func.now()))).scalar_one() - newest if age.total_seconds() < REFRESH_TTL_SECONDS: return await refresh_celery_roster(session)