"""The generated supervisord config (milestone 422 step 5). Asserts the config against the LANE TABLE rather than against a fixture of expected text. A fixture would have to be updated whenever a lane changes, which is the same hand-kept coupling generating the config exists to remove — and it would pass while describing a container that does not match the application's own idea of what it runs. """ from __future__ import annotations import configparser from backend.app.scripts import gen_supervisord as gen from backend.app.services.worker_lanes import LANES, LANES_BY_NAME # --- structure --------------------------------------------------------------- def _parse(**kwargs) -> configparser.ConfigParser: """supervisord's config is ini, so parse it rather than grepping strings. A substring assertion passes on a line that is present but malformed — inside a comment, in the wrong section, or with a typo'd key that supervisord silently ignores. """ cp = configparser.ConfigParser() cp.read_string(gen.render(**kwargs)) return cp def test_it_is_valid_ini_with_a_supervisord_section(): cp = _parse() assert cp.has_section("supervisord") # PID 1 in a container: daemonising would exit immediately and take the # container with it. assert cp.get("supervisord", "nodaemon") == "true" def test_every_lane_gets_a_program(): """One image carries every lane since step 6, so nothing is conditional. A lane in LANES with no program is a queue with no consumer. Compared over the `program:` sections ONLY, both directions: no lane without a program, and no program that is not a lane. It used to compare every section minus `[supervisord]`, which made it fail the moment the config grew non-program plumbing — the control socket did exactly that, and the property it exists for had not moved at all. A guard that fires on a correct change is one people learn to edit rather than read. """ cp = _parse() programs = {s for s in cp.sections() if s.startswith("program:")} expected = {"program:web"} | {f"program:{lane.name}" for lane in LANES} assert programs == expected def test_the_ml_lane_runs_even_though_it_ships_disabled(): """It holds a PROCESS and no model. `add_consumer` needs a running worker to reach, so without this the UI switch would have nothing to switch — and nothing is downloaded by starting it, which is what lets rule 164 permit the fetch at all.""" assert _parse().has_section("program:ml") assert LANES_BY_NAME["ml"].default_enabled is False # --- the coupling this generator exists to guarantee ------------------------- def test_each_program_serves_exactly_its_lane_s_queues(): """The whole point: the container's processes and the application's lane table are one list. A queue in LANES with no program means work that queues forever with nothing consuming it.""" cp = _parse() for lane in LANES: env = cp.get(f"program:{lane.name}", "environment") # The QUOTED form. supervisord splits `environment` on commas, so an # unquoted multi-queue value silently degrades to its first queue — # and an assertion on the bare string passes either way, which is how # that would have shipped. assert f'CELERY_QUEUES="{",".join(lane.queues)}"' in env def test_each_program_invokes_the_lane_s_entrypoint_role_not_its_name(): """`maintenance_long` is the plain `worker` role pointed at a different queue — exactly as docker-compose starts it today. Invoking `entrypoint.sh maintenance_long` would hit the unknown-role branch and exit 1 on every restart.""" cp = _parse() for lane in LANES: command = cp.get(f"program:{lane.name}", "command") assert f"entrypoint.sh {lane.entrypoint_role}" in command def test_every_entrypoint_role_a_lane_names_actually_exists(): """Reads entrypoint.sh itself. The generator can only emit a role name; whether the script handles it is a separate fact, and getting it wrong fails at container start rather than here.""" from pathlib import Path script = Path(__file__).resolve().parents[1] / "entrypoint.sh" text = script.read_text() for lane in LANES: # Roles are `case` arms: ` worker)` possibly in an alternation. assert f" {lane.entrypoint_role})" in text or \ f"|{lane.entrypoint_role})" in text, \ f"{lane.name} names entrypoint role {lane.entrypoint_role!r}, which does not exist" # --- shutdown ---------------------------------------------------------------- def test_every_program_signals_its_whole_process_group(): """Celery's prefork pool forks children. A TERM delivered only to the parent leaves them running and holding tasks — a 'graceful' shutdown that orphans workers. The `sh -c … | sed` wrapper makes this doubly necessary: without it the signal reaches the shell holding the pipeline, not celery.""" cp = _parse() for section in cp.sections(): if not section.startswith("program:"): continue assert cp.get(section, "stopasgroup") == "true", section assert cp.get(section, "killasgroup") == "true", section def test_the_long_maintenance_lane_keeps_its_180s_drain(): """The per-service stop_grace_period values from the multi-service stack are preserved per program. maintenance_long runs DB backups and library audits; cutting its drain turns a restart into a SIGKILL mid-backup.""" cp = _parse() assert cp.getint("program:maintenance_long", "stopwaitsecs") == 180 def test_no_program_waits_longer_than_the_compose_stop_grace_period(): """The container gets ONE timeout and the programs stop in parallel, so it must cover the slowest. If a lane's stopwaitsecs ever exceeds what docker-compose.single.yml allows, docker kills the container while that lane still believes it has time to drain.""" import re from pathlib import Path compose = (Path(__file__).resolve().parents[1] / "docker-compose.single.yml").read_text() m = re.search(r"stop_grace_period:\s*(\d+)s", compose) assert m, "docker-compose.single.yml has no stop_grace_period" grace = int(m.group(1)) cp = _parse() for section in cp.sections(): if section.startswith("program:"): assert cp.getint(section, "stopwaitsecs") <= grace, section # --- what runs, and how much ------------------------------------------------ def test_a_zero_slot_lane_still_gets_a_running_process(): """ML ships at 0 slots and disabled — but `add_consumer` needs something to reach. With no process there would be nothing for the UI switch to switch, and enabling tagging could not work at all.""" cp = _parse() assert LANES_BY_NAME["ml"].default_slots == 0 env = cp.get("program:ml", "environment") assert "CELERY_CONCURRENCY=1" in env def test_programs_restart_but_back_off_rather_than_looping(): """A lane that dies instantly and repeatedly is a broken image, not a transient fault. Unbounded restarts would burn a core forever and bury the original error under its own noise.""" cp = _parse() for section in cp.sections(): if section.startswith("program:"): assert cp.get(section, "autorestart") == "true", section assert cp.getint(section, "startretries") >= 1, section def test_web_starts_first_because_it_runs_the_migration(): """A worker booting against an un-migrated schema fails in a way that looks like application breakage rather than an ordering problem.""" cp = _parse() assert cp.getint("program:web", "priority") == 1 def test_every_program_writes_to_the_container_stdout_with_its_lane_named(): """Four celery workers and hypercorn on one stream are indistinguishable without this. Unbuffered (`maxbytes 0`) so `docker logs` is live rather than arriving in rotated chunks.""" cp = _parse() for section in cp.sections(): if not section.startswith("program:"): continue name = section.split(":", 1)[1] assert cp.get(section, "stdout_logfile") == "/dev/fd/1", section assert cp.getint(section, "stdout_logfile_maxbytes") == 0, section assert cp.get(section, "redirect_stderr") == "true", section assert f"[{name}] " in cp.get(section, "command"), section def test_every_lane_gets_its_own_celery_node_name(): """The bug that made the single-container layout unusable, found the first time CI actually booted it (run 7319). Celery's default node name is `celery@`, and these processes share one hostname — so all four lanes registered as the SAME node: DuplicateNodenameWarning: Received multiple replies from node name: celery@72adc5b706a7 `inspect` then collapses four replies into one dict key, last one wins. Three lanes read as absent, and WHICH three varied between calls: lanes not answering: maintenance_long, ml, worker lanes not answering: maintenance_long, scheduler, worker Fatal twice: the composite healthcheck can never pass, so the container is permanently unhealthy and Swarm restart-loops it; and `pool_grow` takes a `destination` of node names, so the UI dial and the autoscaler would have resized whichever lane happened to answer rather than the one asked for. Asserted as DISTINCTNESS across the whole table rather than as a fixed string per lane — the property is that no two lanes collide, which is what actually broke, and it keeps holding when a lane is added. """ cp = _parse() names = {} for section in cp.sections(): if not section.startswith("program:"): continue lane = section.split(":", 1)[1] if lane == "web": continue # hypercorn, not a celery node env = cp.get(section, "environment") pairs = dict( p.split("=", 1) for p in env.split(",") if "=" in p and not p.startswith('"') ) assert "CELERY_NODENAME" in pairs, f"{lane} has no CELERY_NODENAME: {env}" names[lane] = pairs["CELERY_NODENAME"] assert len(set(names.values())) == len(names), ( f"two lanes share a celery node name, which is the collapse itself: {names}" ) for lane, node in names.items(): assert node == lane, f"{lane} announces itself as {node!r}" def test_supervisorctl_can_reach_supervisord(): """Consolidation took away `docker ps` as the way to see the lanes, and this is what replaces it. Without the three control-socket sections supervisord runs perfectly and `supervisorctl` cannot talk to it at all: Error: .ini file does not include supervisorctl section That is the first thing anyone reaches for when a lane misbehaves inside the one container — `supervisorctl status` to see which processes are up, `restart ml` to bounce one without taking the application down. Shipping without it leaves an operator with five processes and no way to ask about any of them. Asserted as the three sections AGREEING on one socket path, not merely as three sections existing: a serverurl pointing somewhere the server does not listen fails exactly the same way, and reads as configured. """ cp = _parse() for section in ("unix_http_server", "supervisorctl", "rpcinterface:supervisor"): assert cp.has_section(section), f"no [{section}] — supervisorctl is blind" listening = cp.get("unix_http_server", "file") talking = cp.get("supervisorctl", "serverurl") assert talking == f"unix://{listening}", ( f"supervisorctl talks to {talking}, supervisord listens on {listening}" ) def test_each_program_starts_at_the_smallest_pool_the_control_path_allows(): """The two ends of the same floor, asserted together. `gen_supervisord` starts every lane at `max(1, default_slots)` because billiard will not run a pool of zero. `worker_control` has the same floor for the opposite reason: it cannot SHRINK to zero either — [ml] pidbox command error: ValueError("Can't shrink pool. All processes busy!") Live, 2026-09-23. ML starts at one process and stores zero, so the reconcile tried 1 -> 0 on every tick, billiard refused, and `set_lane_slots_sync` — which returns True on SENDING the message — reported the lane changed forever (lesson #4183, on the default configuration of every install). Two constants, in two files, that must agree or the container cannot settle. Asserted through `effective_slots` rather than against a literal 1, so raising the floor moves both ends at once. """ from backend.app.services.worker_control import effective_slots cp = _parse() for lane in LANES: env = cp.get(f"program:{lane.name}", "environment") want = effective_slots(lane.default_slots) assert f"CELERY_CONCURRENCY={want}," in env, ( f"{lane.name} starts at a size the control path cannot reach" )