fix: task runs record the lane Celery really routes them to, and no healthy long job is swept as stalled (#4432)
CI and images / lint (push) Successful in 3s
CI and images / extension-version (push) Successful in 4s
CI and images / frontend-build (push) Successful in 25s
CI and images / extension-test (push) Successful in 28s
CI and images / backend-lint-and-test (push) Successful in 34s
CI and images / integration (push) Failing after 2m24s
CI and images / sign-extension (push) Skipped
CI and images / build-web (push) Skipped
CI and images / smoke-web (push) Skipped
CI and images / promote (push) Skipped
CI and images / build-agent (push) Skipped

celery_signals._queue_for was a hand-kept copy of task_routes and had
drifted:

- backup, admin and library_audit jobs, and backfill_phash, run on
  maintenance_long but were recorded as `maintenance`;
- translation and gpu_queue jobs were recorded as `default`, where the
  5-minute stall sweep failed healthy 35-minute translation runs.

It now asks the router, cached per task name.

A new guard test checks every registered task's hard time limit against
the stall threshold the sweep would use for it. It also caught these
sweeps, which failed healthy runs mid-flight and are fixed here:

- ml's scheduled sweeps (35 min, previously swept at 25);
- train_heads and apply_head_tags (65 min);
- import_media_file (6 min, previously swept at 5);
- the long lane, which now has its own 45-minute threshold.

UI changes:

- the admin job poller follows a job by celery_task_id, via a new filter
  on /runs, instead of by lane;
- the archive re-extract and missing-file repair cards show the long
  lane's backlog, where their jobs actually wait;
- the queue table lists maintenance_long.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LVjrnpQjRgHdvq95rASoiR
This commit is contained in:
2026-09-25 09:46:39 -04:00
co-authored by Claude Opus 5.5
parent dfd28a0aa6
commit a360d69ee8
9 changed files with 112 additions and 50 deletions
+5
View File
@@ -152,6 +152,8 @@ async def list_runs():
queue=<name> filter to one queue queue=<name> filter to one queue
status=<status> filter to one status (running/ok/error/timeout/retry) status=<status> filter to one status (running/ok/error/timeout/retry)
task=<substr> case-insensitive substring match on task_name task=<substr> case-insensitive substring match on task_name
celery_task_id=<id> exactly one run — how a page follows a job it
started without having to know its lane
limit=<int> default 50, max 200 limit=<int> default 50, max 200
before_id=<int> cursor for keyset pagination before_id=<int> cursor for keyset pagination
@@ -167,6 +169,7 @@ async def list_runs():
queue = request.args.get("queue") queue = request.args.get("queue")
status = request.args.get("status") status = request.args.get("status")
task = request.args.get("task") task = request.args.get("task")
celery_task_id = request.args.get("celery_task_id")
before_id_raw = request.args.get("before_id") before_id_raw = request.args.get("before_id")
before_id = int(before_id_raw) if before_id_raw else None before_id = int(before_id_raw) if before_id_raw else None
@@ -176,6 +179,8 @@ async def list_runs():
stmt = stmt.where(TaskRun.queue == queue) stmt = stmt.where(TaskRun.queue == queue)
if status: if status:
stmt = stmt.where(TaskRun.status == status) stmt = stmt.where(TaskRun.status == status)
if celery_task_id:
stmt = stmt.where(TaskRun.celery_task_id == celery_task_id)
if task: if task:
# Task names contain literal underscores (download_source, # Task names contain literal underscores (download_source,
# vacuum_analyze) — escape LIKE wildcards so a search for # vacuum_analyze) — escape LIKE wildcards so a search for
+20 -32
View File
@@ -18,6 +18,7 @@ dark for that interval. Monitoring NEVER breaks the thing it's
monitoring. monitoring.
""" """
import functools
import logging import logging
from datetime import UTC, datetime from datetime import UTC, datetime
@@ -53,42 +54,29 @@ _INT32_MIN = -2_147_483_648
def _queue_for(task) -> str: def _queue_for(task) -> str:
"""Reverse the task→queue routing from celery_app.task_routes. """The queue Celery routes this task to — asked of the router itself.
Keep in sync if task_routes is reordered.
Audit 2026-06-02: backup/admin/library_audit prefixes were This was a hand-kept copy of `celery_app.task_routes`, and it drifted
missing here even though task_routes sent all three to twice (the 2026-06-02 audit, then #4432). Long-lane jobs were recorded as
'maintenance'. The TaskRun.queue column then lied for those `maintenance`, and translation and gpu_queue runs as `default`, where the
rows (claimed 'default') so per-queue dashboard filters and 5-minute stall sweep failed healthy 35-minute translation runs. The router
per-queue threshold overrides silently missed them. answers from the same table the broker uses, so the two cannot disagree.
""" """
name = getattr(task, "name", "") or "" name = getattr(task, "name", "") or ""
if name.startswith("backend.app.tasks.import_file."): app = getattr(task, "app", None)
return "import" if app is None:
if name.startswith("backend.app.tasks.ml."): from .celery_app import celery as app
return "ml" return _routed_queue(app, name)
if name.startswith("backend.app.tasks.thumbnail."):
return "thumbnail"
if name.startswith(( @functools.lru_cache(maxsize=1024)
"backend.app.tasks.download.", def _routed_queue(app, name: str) -> str:
# External file-host fetches share the download lane (celery_app try:
# routes external.* → download). Mirror it here or TaskRun.queue queue = app.amqp.router.route({}, name).get("queue")
# lies 'default' for them, so per-queue dashboard filters and the except Exception: # noqa: BLE001 — monitoring never breaks the task
# per-queue threshold override miss them — the same gap the log.warning("task_run: could not resolve the queue for %s", name)
# 2026-06-02 audit fixed for backup/admin/library_audit.
"backend.app.tasks.external.",
)):
return "download"
if name.startswith("backend.app.tasks.scan."):
return "scan"
if name.startswith((
"backend.app.tasks.maintenance.",
"backend.app.tasks.backup.",
"backend.app.tasks.admin.",
"backend.app.tasks.library_audit.",
)):
return "maintenance"
return "default" return "default"
return getattr(queue, "name", None) or (queue if isinstance(queue, str) else "default")
def _target_id_from_args(args) -> int | None: def _target_id_from_args(args) -> int | None:
+17 -1
View File
@@ -137,7 +137,13 @@ IMPORT_BATCH_KEEP_DAYS = 30
# (the import queue itself stays at the 5-min default for single # (the import queue itself stays at the 5-min default for single
# files); time_limit=2100. # files); time_limit=2100.
QUEUE_STUCK_THRESHOLD_MINUTES: dict[str, int] = { QUEUE_STUCK_THRESHOLD_MINUTES: dict[str, int] = {
"ml": 25, # ml: the scheduled auto-apply sweeps and refresh_character_prototypes run
# to a 35-min hard limit (2100s); 25 swept them mid-run (#4432). The two
# 65-min jobs have their own entries below.
"ml": 40,
# import: import_media_file's hard limit is 6 min (360s), one past the
# 5-min default this queue fell to (#4432).
"import": 10,
# download_source legitimately walks 5-25 min (Patreon/gallery-dl # download_source legitimately walks 5-25 min (Patreon/gallery-dl
# deep creators); its hard time_limit is DOWNLOAD_HARD_TIME_LIMIT # deep creators); its hard time_limit is DOWNLOAD_HARD_TIME_LIMIT
# (1500s = 25m). The 5-min default flagged healthy in-flight walks as # (1500s = 25m). The 5-min default flagged healthy in-flight walks as
@@ -154,6 +160,12 @@ QUEUE_STUCK_THRESHOLD_MINUTES: dict[str, int] = {
# overrides below cover the outliers (backups, library audit). # overrides below cover the outliers (backups, library audit).
"maintenance": 75, "maintenance": 75,
"scan": 75, "scan": 75,
# The long lane (#4432). Until TaskRun.queue asked the router, nothing was
# recorded here: these runs read as `maintenance` (75) or, for
# translation, `default` (5 — which failed healthy 35-min runs). The
# longest task without its own entry below is the admin family at a
# 40-min hard limit; 45 = 40 + 5.
"maintenance_long": 45,
} }
TASK_STUCK_THRESHOLD_MINUTES: dict[str, int] = { TASK_STUCK_THRESHOLD_MINUTES: dict[str, int] = {
"backend.app.tasks.import_file.import_archive_file": 40, "backend.app.tasks.import_file.import_archive_file": 40,
@@ -179,6 +191,10 @@ TASK_STUCK_THRESHOLD_MINUTES: dict[str, int] = {
# external-fetch entry above — without an override a healthy in-flight walk # external-fetch entry above — without an override a healthy in-flight walk
# is swept 'RecoverySweep' at the bare 5-min default. 30 = 25 + 5. # is swept 'RecoverySweep' at the bare 5-min default. 30 = 25 + 5.
"backend.app.tasks.admin.reclaim_orphaned_attachments_task": 30, "backend.app.tasks.admin.reclaim_orphaned_attachments_task": 30,
# Head training and the manual head apply run to 65 min (3900s) — past the
# ml queue's threshold (#4432). 70 = 65 + 5.
"backend.app.tasks.ml.train_heads": 70,
"backend.app.tasks.ml.apply_head_tags": 70,
} }
@@ -17,7 +17,7 @@
<v-icon start>mdi-folder-zip-outline</v-icon> Re-extract archives now <v-icon start>mdi-folder-zip-outline</v-icon> Re-extract archives now
</v-btn> </v-btn>
<span v-if="queued" class="ml-3 text-caption text-success">Queued ✓</span> <span v-if="queued" class="ml-3 text-caption text-success">Queued ✓</span>
<QueueStatusBar queue="maintenance" queue-label="Maintenance" /> <QueueStatusBar queue="maintenance_long" queue-label="Long maintenance" />
</MaintenanceTile> </MaintenanceTile>
</template> </template>
@@ -18,7 +18,7 @@
<v-icon start>mdi-file-remove-outline</v-icon> Repair missing-file records <v-icon start>mdi-file-remove-outline</v-icon> Repair missing-file records
</v-btn> </v-btn>
<span v-if="queued" class="ml-3 text-caption text-success">Queued ✓</span> <span v-if="queued" class="ml-3 text-caption text-success">Queued ✓</span>
<QueueStatusBar queue="maintenance" queue-label="Maintenance" /> <QueueStatusBar queue="maintenance_long" queue-label="Long maintenance" />
</MaintenanceTile> </MaintenanceTile>
</template> </template>
@@ -41,7 +41,7 @@ const props = defineProps({
const QUEUE_NAMES = [ const QUEUE_NAMES = [
'default', 'import', 'thumbnail', 'ml', 'default', 'import', 'thumbnail', 'ml',
'download', 'scan', 'maintenance', 'download', 'scan', 'maintenance', 'maintenance_long',
] ]
function formatDepth(name) { function formatDepth(name) {
+4 -2
View File
@@ -119,7 +119,7 @@ export const useAdminStore = defineStore('admin', () => {
// --- Task progress polling (taps FC-3i activity dashboard) -------- // --- Task progress polling (taps FC-3i activity dashboard) --------
/** /**
* Polls /api/system/activity/runs?queue=maintenance every 3s, * Polls /api/system/activity/runs?celery_task_id=<id> every 3s,
* resolves when a task_run row with the given celery task_id * resolves when a task_run row with the given celery task_id
* reaches a terminal status (ok / error / timeout). Returns the * reaches a terminal status (ok / error / timeout). Returns the
* row. Times out after 30 min by default. * row. Times out after 30 min by default.
@@ -129,7 +129,9 @@ export const useAdminStore = defineStore('admin', () => {
while (Date.now() < deadline) { while (Date.now() < deadline) {
const body = await api.get( const body = await api.get(
'/api/system/activity/runs', '/api/system/activity/runs',
{ params: { queue: 'maintenance', limit: 20 } }, // By id, not by lane: these jobs run on `maintenance_long`, and a
// lane filter here is one more copy of the routing table (#4432).
{ params: { celery_task_id: taskId, limit: 1 } },
) )
const row = (body.runs || []).find(r => r.celery_task_id === taskId) const row = (body.runs || []).find(r => r.celery_task_id === taskId)
if (row && ['ok', 'error', 'timeout'].includes(row.status)) { if (row && ['ok', 'error', 'timeout'].includes(row.status)) {
+8
View File
@@ -295,3 +295,11 @@ async def test_failures_only_within_24h_window(client, _seed_failures):
body = await resp.get_json() body = await resp.get_json()
ids = {r["error_type"] for r in body["recent"]} ids = {r["error_type"] for r in body["recent"]}
assert "OldError" not in ids assert "OldError" not in ids
@pytest.mark.asyncio
async def test_runs_filter_by_celery_task_id(client, _seed_runs):
# How a page follows a job it started, without knowing its lane (#4432).
resp = await client.get("/api/system/activity/runs?celery_task_id=tid-3")
body = await resp.get_json()
assert [r["celery_task_id"] for r in body["runs"]] == ["tid-3"]
+54 -11
View File
@@ -2,6 +2,8 @@
`maintenance_long` lane so a 30-min backup or a multi-chunk audit can't starve `maintenance_long` lane so a 30-min backup or a multi-chunk audit can't starve
the quick recovery sweeps / vacuum on the concurrency-1 `maintenance` lane the quick recovery sweeps / vacuum on the concurrency-1 `maintenance` lane
(operator-flagged 2026-06-07).""" (operator-flagged 2026-06-07)."""
import pytest
from backend.app.celery_app import celery from backend.app.celery_app import celery
@@ -20,23 +22,64 @@ def test_quick_maintenance_stays_on_maintenance():
assert routes["backend.app.tasks.maintenance.*"]["queue"] == "maintenance" assert routes["backend.app.tasks.maintenance.*"]["queue"] == "maintenance"
def test_queue_for_mirrors_external_to_download(): @pytest.mark.parametrize(("name", "queue"), [
"""celery_signals._queue_for is a hand-maintained mirror of task_routes ("backend.app.tasks.external.fetch_external_link", "download"),
that stamps TaskRun.queue. external.* routes to the download lane, so the # The rows #4432 found recorded on the wrong lane:
mirror must agree — else TaskRun.queue lies 'default' for external fetches ("backend.app.tasks.translation.translate_posts", "maintenance_long"),
and per-queue dashboard filters / threshold overrides miss them ("backend.app.tasks.gpu_queue.enqueue_gpu_backfill", "maintenance"),
(operator-flagged 2026-06-17).""" ("backend.app.tasks.maintenance.backfill_phash", "maintenance_long"),
("backend.app.tasks.admin.normalize_tags_task", "maintenance_long"),
("backend.app.tasks.backup.backup_db_task", "maintenance_long"),
("backend.app.tasks.maintenance.vacuum_analyze", "maintenance"),
("backend.app.tasks.not_routed.anything", "default"),
])
def test_task_run_records_the_queue_the_router_sends_to(name, queue):
"""TaskRun.queue comes from the router, not a copy of task_routes (#4432):
the stall sweep's per-queue thresholds and the System activity filters both
key off it."""
from backend.app.celery_signals import _queue_for from backend.app.celery_signals import _queue_for
class _T: class _T:
name = "backend.app.tasks.external.fetch_external_link" pass
assert _queue_for(_T()) == "download" t = _T()
assert ( t.name = name
celery.conf.task_routes["backend.app.tasks.external.*"]["queue"] t.app = celery
== "download" assert _queue_for(t) == queue
def test_no_task_outlives_its_stall_threshold():
"""A task whose hard time limit is longer than the stall sweep's threshold
for it gets failed 'RecoverySweep' while it is still healthy — the class
#4432 found on the ml, import and long-maintenance lanes. The threshold is
resolved the way recover_stalled_task_runs resolves it: task-name override,
then queue, then the default."""
from backend.app.celery_signals import _queue_for
from backend.app.tasks.maintenance import (
QUEUE_STUCK_THRESHOLD_MINUTES,
STUCK_THRESHOLD_MINUTES,
TASK_STUCK_THRESHOLD_MINUTES,
) )
celery.loader.import_default_modules()
checked = 0
too_short = []
for name, task in sorted(celery.tasks.items()):
if name.startswith("celery."):
continue
limit = getattr(task, "time_limit", None)
if not limit:
continue
threshold = TASK_STUCK_THRESHOLD_MINUTES.get(
name,
QUEUE_STUCK_THRESHOLD_MINUTES.get(_queue_for(task), STUCK_THRESHOLD_MINUTES),
)
checked += 1
if limit / 60 > threshold:
too_short.append(f"{name}: limit {limit / 60:.0f} min > sweep {threshold} min")
assert checked > 20, "the task registry did not load — this guard checked nothing"
assert not too_short, "\n".join(too_short)
def test_backfill_phash_runs_on_the_long_lane(): def test_backfill_phash_runs_on_the_long_lane():
"""It lives in maintenance.py, so the quick-lane glob matches it too — """It lives in maintenance.py, so the quick-lane glob matches it too —