#!/usr/bin/env bash set -euo pipefail ROLE="${1:-web}" shift || true case "$ROLE" in web) echo "[entrypoint] Running alembic upgrade head" alembic upgrade head echo "[entrypoint] Starting hypercorn on :8080" # create_app is a factory — the `()` tells hypercorn to call it once # and serve the returned Quart (ASGI) app, rather than treating the # function itself as the application (which it then mis-invokes as WSGI). # Default 4 workers (was 2): each worker is one asyncio loop, and a large # file download occupies its worker for the transfer — 2 was too few once the # GPU agent + the browser's thumbnail grid hit /images concurrently (they # queued behind each other). Env-tunable via HYPERCORN_WORKERS. exec hypercorn \ --bind 0.0.0.0:8080 \ --workers "${HYPERCORN_WORKERS:-4}" \ --access-logfile - \ "backend.app:create_app()" ;; worker) QUEUES="${CELERY_QUEUES:-default,import,thumbnail}" CONCURRENCY="${CELERY_CONCURRENCY:-2}" echo "[entrypoint] Starting Celery worker queues=$QUEUES concurrency=$CONCURRENCY" exec celery -A backend.app.celery_app:celery worker \ --loglevel=info \ -Q "$QUEUES" \ --concurrency="$CONCURRENCY" ;; scheduler) QUEUES="${CELERY_QUEUES:-maintenance,scan}" # Honours CELERY_CONCURRENCY like the `worker` role does. It was hardcoded # to 1, which was harmless while only compose started this lane and set no # concurrency for it — but the generated supervisord config (milestone 422 # step 5) passes one, and a value silently ignored at boot would leave the # lane at 1 until the reconcile sweep noticed, with nothing saying why. CONCURRENCY="${CELERY_CONCURRENCY:-1}" echo "[entrypoint] Starting Celery beat+worker queues=$QUEUES concurrency=$CONCURRENCY" exec celery -A backend.app.celery_app:celery worker \ --beat \ --loglevel=info \ -Q "$QUEUES" \ --concurrency="$CONCURRENCY" ;; ml-worker) # NO MODEL DOWNLOAD HERE (milestone 422 step 6). This used to run # download_models before celery started, which made every boot of this # role reach HuggingFace for ~3.5GB. Rule 164 permits a runtime fetch only # for a feature that is "optional and clearly off" — so the fetch moved to # the moment the operator ENABLES the lane, where it is visible, retryable # and attributable, instead of being a silent precondition of starting. # # The worker therefore starts with no model present, which is correct: it # is not consuming the ml queue until the lane is enabled, and enabling it # is what enqueues ensure_models. QUEUES="${CELERY_QUEUES:-ml}" CONCURRENCY="${CELERY_CONCURRENCY:-1}" echo "[entrypoint] Starting ML Celery worker queues=$QUEUES concurrency=$CONCURRENCY" exec celery -A backend.app.celery_app:celery worker \ --loglevel=info \ -Q "$QUEUES" \ --concurrency="$CONCURRENCY" ;; all) # The single-container layout (milestone 422 step 5): hypercorn plus one # celery process per lane, under supervisord, in one container beside # Postgres and Redis. # # The config is GENERATED from services/worker_lanes.LANES rather than # checked in, so the processes this container runs and the lanes the # application believes in cannot disagree — see the generator's docstring # for why a static .conf would have been a fifth copy of the queue names. # # supervisord is PID 1 here and never reads the database. Every lane boots # at its LANES default; the reconcile sweep raises it to whatever the # operator stored, within one tick. That ordering is deliberate: settings # adjust a baseline that already works, and can never prevent a boot. CONF="${SUPERVISOR_CONF:-/tmp/supervisord.conf}" echo "[entrypoint] Generating $CONF from the lane table" python -m backend.app.scripts.gen_supervisord > "$CONF" echo "[entrypoint] Starting supervisord (web + worker lanes)" exec supervisord -c "$CONF" ;; shell|bash) exec /bin/bash "$@" ;; alembic) exec alembic "$@" ;; *) echo "[entrypoint] Unknown role: $ROLE" >&2 echo "[entrypoint] Valid roles: all | web | worker | scheduler | maintenance_long | ml | ml-worker | shell | alembic" >&2 exit 1 ;; esac