# syntax=docker/dockerfile:1.25 FROM node:24-alpine AS frontend-builder WORKDIR /build COPY frontend/package.json frontend/package-lock.json* ./ # No package-lock.json is tracked yet (we don't run npm locally per # feedback-no-local-runs), so `npm install` instead of `npm ci`. Flip to # `npm ci` once a lockfile is committed. RUN npm install --no-audit --no-fund COPY frontend/ ./ RUN npm run build FROM python:3.14-slim AS runtime ENV PYTHONUNBUFFERED=1 \ PYTHONDONTWRITEBYTECODE=1 \ PIP_NO_CACHE_DIR=1 \ PIP_DISABLE_PIP_VERSION_CHECK=1 # System deps: ffmpeg (transcode + thumbnails, FC-2), unar (archives, FC-2), # libpq for psycopg, postgresql-client + zstd for FC-5 backup/restore # (pg_dump + tar --zstd), image libs, megatools (mega.nz public-link downloads # for off-platform file-host links, #830 — `megatools dl`; Debian-native, no # external MEGA apt repo needed). RUN apt-get update && apt-get install -y --no-install-recommends \ ffmpeg \ unar \ libpq5 \ postgresql-client \ zstd \ megatools \ libjpeg62-turbo \ libwebp7 \ libpng16-16 \ ca-certificates \ # opencv-python-headless (via requirements-ml.txt) links these even in its # headless build. Came from Dockerfile.ml when the images merged # (milestone 422 step 6). libgl1 \ libglib2.0-0 \ && rm -rf /var/lib/apt/lists/* WORKDIR /app COPY requirements.txt requirements-ml.txt ./ RUN pip install -r requirements.txt # --- ML, merged from Dockerfile.ml (milestone 422 step 6) -------------------- # # ONE image now serves every lane. It was two because the ML lane ran in its # own container; with the single-container layout (step 5) running every lane # in one process tree, a second image would mean the `ml` lane could never be # enabled from the UI — there would be no worker in this container to enable. # # THE COST, MEASURED from run 7273 rather than guessed — and it is far # smaller than the estimate this comment first carried, which said "everyone # pulls ~4GB": # # torch 2.12.1+cpu wheel 192.3 MB # torchvision 0.27.1+cpu 1.8 MB # transformers / onnxruntime / opencv / sklearn and friends # 62.0, 35.3, 23.6, 16.7, 12.3, 9.2, 6.9 MB # largest newly-pushed layer 222.07 MB # # So the ML code adds a few hundred MB to the pull, not gigabytes. The CPU # index is what makes that true: the default PyPI torch wheel bundles the # NVIDIA CUDA runtime and is ~2GB on its own. # # The GIGABYTES are in the MODEL — ~3.5GB of SigLIP weights — and those are # NOT in this image. They arrive only when the operator enables the lane, # which is what lets rule 164 permit a runtime fetch at all ("optional and # clearly off"). That also settles the trade this step was asked to weigh: # baking the weights in would add ~3.5GB to every pull for a feature many # adopters never enable, against ~350MB for the code that makes the switch # available. Off-by-default wins by an order of magnitude, which was NOT # obvious before measuring — the estimate had the two costs within 15% of # each other. # # `--index-url`, not `--extra-index-url`: the latter would let pip resolve a # +cu wheel anyway, and the whole saving above depends on it not doing that. # # CPU-only torch from the PyTorch CPU index. Nothing here uses a GPU — the # GPU agent is a separate service with its own image. RUN pip install --index-url https://download.pytorch.org/whl/cpu \ "torch>=2.12,<3.0" "torchvision>=0.27,<0.28" RUN pip install -r requirements-ml.txt # Where the model lands. Deliberately NOT a VOLUME instruction: that mints an # anonymous volume when nobody mounts one, which survives `docker rm` and # accumulates 3.5GB copies nobody can find. The compose files mount it # explicitly instead, so an unmounted run simply re-downloads — visible, and # recoverable. ENV HF_HOME=/models/.huggingface \ TRANSFORMERS_CACHE=/models/.huggingface \ ML_MODEL_DIR=/models COPY backend/ ./backend/ COPY alembic/ ./alembic/ COPY alembic.ini ./ COPY entrypoint.sh ./ RUN chmod +x entrypoint.sh COPY --from=frontend-builder /build/dist ./frontend/dist # Which channel this image belongs to — `dev` or `main` (milestone 271 step 7). # build.yml passes it; /api/extension/manifest reports it beside the version so # an operator can tell which channel an install came from without the channel # ever touching the version string. # # Empty by default, deliberately: a locally-built image then reports NO channel # rather than claiming to be one, and the manifest omits the field entirely — # indistinguishable from an image built before the field existed, which is # exactly the shape every reader already has to handle. # # Declared LAST on purpose. An ARG/ENV invalidates every layer below it, and # these are the values that differ between builds of otherwise identical # source — put them any earlier and the two channels could never share a # cached pip install. # # FC_VERSION is what the instance reports about itself in the UI. Since # milestone 318 stopped publishing version image tags, that self-report is # the only answer to "which build is this?" — nothing else names it. ARG FC_CHANNEL="" ENV FC_CHANNEL=${FC_CHANNEL} ARG FC_VERSION="" ENV FC_VERSION=${FC_VERSION} EXPOSE 8080 ENTRYPOINT ["./entrypoint.sh"] CMD ["web"]