diff --git a/.forgejo/workflows/baseline.yml b/.forgejo/workflows/baseline.yml new file mode 100644 index 0000000..7b49421 --- /dev/null +++ b/.forgejo/workflows/baseline.yml @@ -0,0 +1,370 @@ + +# TEMPORARY — milestone 328 steps 1-2. Delete once the baseline is stamped. +# +# Squashing 87 alembic revisions into one baseline has exactly one dangerous +# failure: the generated baseline does not reproduce the schema the chain +# produced, `alembic stamp` writes a version string anyway (it validates +# NOTHING), and the divergence surfaces on the next real migration against the +# operator's live data. +# +# So this workflow does the comparison in CI, where a pgvector Postgres already +# gets built from the chain on every integration run, and nothing is at risk. +# It answers one question: does `upgrade head` on the collapsed chain produce a +# byte-identical schema to `upgrade head` on the 87-revision chain? +# +# The chain is read from git rather than from the working tree, so this keeps +# working AFTER the old revisions are deleted — `chain_ref` names a commit that +# still has them. That is what makes this the proof for step 1 and the +# pre-flight for step 2, rather than a one-shot script. +# +# While the chain is still present it also autogenerates a candidate baseline +# from the models and prints it. That is a starting point, NOT the answer: +# autogenerate reads SQLAlchemy metadata, and three things here do not live +# there — +# * CREATE EXTENSION vector (0001) +# * CREATE EXTENSION tsm_system_rows (0004) +# * the HNSW index on image_record.siglip_embedding, which is raw SQL +# because alembic's create_index cannot express `USING hnsw (...)` (0036) +# plus any CHECK constraint or server_default that a migration added without +# the model declaring it. Those must be hand-added, and the diff below is what +# proves none were missed. +name: Alembic baseline + +on: + workflow_dispatch: + inputs: + chain_ref: + description: 'Commit/tag carrying the full chain; blank = this ref (use a pinned commit only AFTER the collapse)' + type: string + default: '' + mode: + description: 'chain = compare against this tree''s migrations; models = compare against a schema built from the MODELS' + type: string + default: 'chain' + +jobs: + compare: + runs-on: python-ci + container: + image: git.fabledsword.com/bvandeusen/ci-python:3.14 + env: + DB_USER: fabledcurator + DB_PASSWORD: ci_integration + DB_PORT: "5432" + DB_NAME: fabledcurator_test + SECRET_KEY: ci_integration_placeholder + services: + postgres: + image: pgvector/pgvector:pg16 + env: + POSTGRES_USER: fabledcurator + POSTGRES_PASSWORD: ci_integration + POSTGRES_DB: fabledcurator_test + options: >- + --health-cmd "pg_isready -U fabledcurator" + --health-interval 10s + --health-timeout 5s + --health-retries 10 + steps: + - uses: actions/checkout@v4 + with: + # Full history is the point: `chain_ref` is read out of git, so a + # shallow clone would not have the revisions to compare against. + fetch-depth: 0 + + - name: Resolve the Postgres service and install deps + run: | + set -eux + # Same service-IP dance as ci.yml's integration job; see the long + # comment there for why the job name must stay separator-free. + PG=$(docker ps --filter "name=compare" --filter "ancestor=pgvector/pgvector:pg16" -q | head -n1) + test -n "$PG" + PG_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$PG") + test -n "$PG_IP" + echo "PG_CONTAINER=$PG" >> "$GITHUB_ENV" + echo "DB_HOST=$PG_IP" >> "$GITHUB_ENV" + for i in $(seq 1 60); do + (echo > "/dev/tcp/$PG_IP/5432") >/dev/null 2>&1 && break + sleep 2 + done + if command -v uv >/dev/null 2>&1; then + uv pip install --system -r requirements.txt + else + pip install -r requirements.txt + fi + + # DB 1: the 87-revision chain, read out of git at `chain_ref`. + # + # A git worktree rather than a checkout, so the current tree — which is + # what we are testing — is left completely alone. + - name: Build the schema the OLD chain produces + env: + CHAIN_REF: ${{ github.event.inputs.chain_ref }} + THIS_SHA: ${{ github.sha }} + run: | + set -eux + docker exec "$PG_CONTAINER" createdb -U fabledcurator fc_chain + # Blank means "the chain in this ref", which is what you want while + # the chain is still intact — comparing the models against a PINNED + # older commit reports every migration written since as a difference. + # Pin it only after the collapse, when the tree no longer has them. + git worktree add /tmp/chain "${CHAIN_REF:-$THIS_SHA}" + ls /tmp/chain/alembic/versions/*.py | wc -l + cd /tmp/chain + DB_NAME=fc_chain alembic upgrade head + cd - + docker exec "$PG_CONTAINER" pg_dump -U fabledcurator --schema-only \ + --no-owner --no-privileges -d fc_chain > chain.sql + wc -l chain.sql + # Emit the dump itself, checksummed, for local analysis. Reconciling + # the models against the deployed schema (#3275) needs the ACTUAL + # schema, not an inference from a diff — parsing table context out of + # unified-diff hunks drops every table whose CREATE TABLE line falls + # outside a hunk, which silently under-reports. + # + # base64 + sha256 for the same reason as the candidate: a plain cat + # of a file this size was truncated mid-line by the runner with the + # step still green (run 4964). + set +x + B64=$(base64 -w 120 chain.sql) + echo "===== BEGIN CHAIN SCHEMA (base64) =====" + echo "$B64" + echo "===== END CHAIN SCHEMA =====" + echo "chain-sha256: $(sha256sum chain.sql | cut -d' ' -f1)" + echo "chain-bytes: $(wc -c < chain.sql)" + set -x + + # A candidate baseline, autogenerated from the models against an EMPTY + # database so every table shows up as a create. Printed for a human to + # finish — it will be missing the three raw-SQL items named at the top. + # + # Gated on the TREE, not on a workflow input. A `type: boolean` input + # read back as `github.event.inputs.generate == 'true'` silently + # evaluated false on this runner (run 4960 skipped this step entirely + # with no diagnostic) — the same `github.event.inputs` typing quirk + # build.yml already works around. The file count is the real question + # anyway: there is nothing to generate once the chain is collapsed. + - name: Autogenerate a candidate baseline + run: | + set -eux + if [ "$(ls alembic/versions/*.py | wc -l)" -le 1 ]; then + echo "already collapsed — nothing to generate" + exit 0 + fi + docker exec "$PG_CONTAINER" createdb -U fabledcurator fc_gen + # Hide the existing revisions so alembic sees an empty history and + # emits the whole schema rather than a delta. + mkdir -p /tmp/versions_held + mv alembic/versions/*.py /tmp/versions_held/ 2>/dev/null || true + DB_NAME=fc_gen alembic revision --autogenerate -m "baseline" || true + # Printed rather than uploaded: ci-requirements.md records that this + # runner cannot do actions/upload-artifact@v4+, and the repo dropped + # the action entirely in 2026-05, so the job log is the retrieval + # channel actually proven here. + # + # base64, not the raw file. A plain `cat` of the ~33KB candidate was + # TRUNCATED MID-LINE by the runner on run 4964 — it stopped inside + # `sa.Column('mime', sa.String(length=128)` and carried straight on + # to the next traced command, with the step still green. A silent + # cut in the middle of a schema definition is the worst possible + # failure here, because the truncated text still looks like a + # plausible file. + # + # base64 at a fixed narrow width gives many short lines instead of + # few long ones, and — the actual point — a checksum and a line + # count that make truncation DETECTABLE rather than invisible. + set +x + F=$(ls alembic/versions/*.py | head -1) + B64=$(base64 -w 120 "$F") + echo "===== BEGIN CANDIDATE BASELINE (base64) =====" + echo "$B64" + echo "===== END CANDIDATE BASELINE =====" + echo "candidate-sha256: $(sha256sum "$F" | cut -d' ' -f1)" + echo "candidate-bytes: $(wc -c < "$F")" + echo "candidate-b64-lines: $(echo "$B64" | wc -l)" + set -x + mkdir -p /tmp/candidate + cp alembic/versions/*.py /tmp/candidate/ + # Put the tree back exactly as it was; this job never mutates state. + rm -f alembic/versions/*.py + mv /tmp/versions_held/*.py alembic/versions/ 2>/dev/null || true + + # DB 2: what the CURRENT tree produces. + # + # `mode: models` applies the candidate autogenerated from the MODELS + # instead, which is what answers "do the models describe the schema?" — + # the question #3275 exists because nobody had ever asked it. Under that + # mode a clean diff means autogenerate is trustworthy again. + # + # The two extensions are created by hand first. They are database + # objects, not table metadata, so no model can carry them and their + # absence is not a model defect — it is simply outside what this + # comparison is asking about. + - name: Build the schema the CURRENT tree produces + env: + MODE: ${{ github.event.inputs.mode }} + run: | + set -eux + docker exec "$PG_CONTAINER" createdb -U fabledcurator fc_base + if [ "${MODE:-chain}" = "models" ]; then + docker exec "$PG_CONTAINER" psql -U fabledcurator -d fc_base \ + -c "CREATE EXTENSION IF NOT EXISTS vector" \ + -c "CREATE EXTENSION IF NOT EXISTS tsm_system_rows" + mkdir -p /tmp/held + mv alembic/versions/*.py /tmp/held/ + cp /tmp/candidate/*.py alembic/versions/ + # Autogenerate EMITS pgvector.sqlalchemy.vector.VECTOR(...) without + # importing it, so the file it writes cannot run: + # NameError: name 'pgvector' is not defined + # Observed on run 4988, which is the proof rather than the theory. + # This is a defect in the GENERATOR, not in the models, so it is + # repaired here rather than counted as a schema difference — the + # comparison is about whether the models describe the schema. + sed -i '0,/^import sqlalchemy as sa$/s//import sqlalchemy as sa\nimport pgvector.sqlalchemy.vector/' alembic/versions/*.py + grep -n 'import pgvector' alembic/versions/*.py + # Second generator defect, same class as the missing import. + # + # base.py's naming convention includes %(constraint_name)s for ck, + # which — unlike uq/fk/ix — means the convention is applied even to + # a CheckConstraint that HAS a name. So a model declaring + # name="singleton" correctly becomes ck_ml_settings_singleton in + # the metadata. Autogenerate then writes that RENDERED name into + # the migration, and running the migration applies the convention a + # SECOND time: ck_ml_settings_ck_ml_settings_singleton. + # + # That is round-tripping damage done by the generator, not a claim + # the models make, so it is repaired here rather than counted as a + # schema difference. Undone by removing the ck__ prefix the + # convention will re-add — the exact inverse, and it only fires on + # a name that actually carries its own table's prefix. + python3 - alembic/versions/*.py <<'PYEOF' + import re, sys + + table = None + for path in sys.argv[1:]: + out = [] + for line in open(path): + m = re.search(r"op\.create_table\(\s*[\"']([A-Za-z0-9_]+)[\"']", line) + if m: + table = m.group(1) + if table and "CheckConstraint" in line: + prefix = f"ck_{table}_" + line = re.sub( + r"(name=[\"'])" + re.escape(prefix), + r"\1", + line, + ) + out.append(line) + open(path, "w").writelines(out) + PYEOF + grep -n 'CheckConstraint' alembic/versions/*.py || true + ls alembic/versions/*.py + DB_NAME=fc_base alembic upgrade head + rm -f alembic/versions/*.py + mv /tmp/held/*.py alembic/versions/ + else + ls alembic/versions/*.py | wc -l + DB_NAME=fc_base alembic upgrade head + fi + docker exec "$PG_CONTAINER" pg_dump -U fabledcurator --schema-only \ + --no-owner --no-privileges -d fc_base > baseline.sql + wc -l baseline.sql + + # The verdict. + # + # pg_dump orders dumpable objects by name within type, not by creation + # order, so two schemas built by different routes are directly + # comparable. Normalisation is deliberately minimal, because a filter + # that hides a real difference is the one way this check passes when it + # should fail — blank lines, SQL comments, trailing whitespace, and: + # + # \restrict / \unrestrict — a per-invocation RANDOM NONCE that newer + # pg_dump emits to fence the dump against injection during restore. It + # differs on every run by construction, so it is noise by definition, + # not a schema difference. Measured on run 4960, the control: two dumps + # of the SAME schema came back 1123 lines each and differed on exactly + # these two lines and nothing else. That control is what licenses this + # filter — it was observed to be the only false positive, rather than + # assumed to be one. + # Column ORDER inside a CREATE TABLE is compared separately from column + # CONTENT, and only content is fatal. + # + # A table built by 87 migrations has its columns in ADD COLUMN order; the + # same table built in one shot has them in declaration order. That is a + # real and permanent difference which no baseline can erase — the + # operator's existing database keeps chain order forever, a fresh install + # gets model order — so a check that fails on it would never pass and + # would teach nothing. FC reaches every column through the ORM by name, + # and `SELECT *` ordering is not depended on anywhere. + # + # So the second pass SORTS the column lines within each CREATE TABLE + # rather than DELETING them. That distinction is the whole point: sorting + # cannot hide a column that exists on one side only, or one whose type, + # nullability or default differs — those still land in the diff. A filter + # could have hidden all three. + # + # Both diffs are reported. The ordered one is informational; the + # order-insensitive one is the verdict. + - name: Diff + run: | + set -eu + norm() { + grep -vE '^\s*(--|$)' "$1" \ + | grep -vE '^\\(un)?restrict ' \ + | sed 's/[[:space:]]*$//' + } + norm chain.sql > a.txt + norm baseline.sql > b.txt + echo "normalised: chain=$(wc -l < a.txt) lines, current=$(wc -l < b.txt) lines" + + sort_table_columns() { + python3 - "$1" <<'PYEOF' + import re, sys + + lines = open(sys.argv[1]).read().splitlines() + out, block = [], None + for line in lines: + if block is not None: + # ');' on its own closes the CREATE TABLE body. + if line.strip() == ");": + out.extend(sorted(block)) + out.append(line) + block = None + else: + # Drop the list comma before sorting. Only the LAST + # column lacks one, so keeping it would make every + # reordering look like a content change as well — the + # comma is punctuation, and carries no schema meaning. + block.append(line.rstrip().rstrip(",")) + continue + out.append(line) + if re.match(r"CREATE TABLE .*\($", line): + block = [] + if block is not None: # unterminated body: emit it rather than drop it + out.extend(block) + print("\n".join(out)) + PYEOF + } + sort_table_columns a.txt > a.sorted.txt + sort_table_columns b.txt > b.sorted.txt + test "$(wc -l < a.sorted.txt)" = "$(wc -l < a.txt)" + test "$(wc -l < b.sorted.txt)" = "$(wc -l < b.txt)" + + if diff -u a.txt b.txt > schema.diff; then + echo "ORDERED DIFF: identical, column order included." + else + echo "ORDERED DIFF: $(grep -cE '^[+-]' schema.diff) changed lines (informational):" + cat schema.diff + fi + echo + echo "================================================================" + echo + if diff -u a.sorted.txt b.sorted.txt > sorted.diff; then + echo "SCHEMAS MATCH — every difference above is column ORDER alone." + else + echo "SCHEMAS DIFFER — $(grep -cE '^[+-]' sorted.diff) changed lines that are NOT ordering:" + cat sorted.diff + echo + echo "The baseline is wrong, not the database. Do not stamp." + exit 1 + fi diff --git a/.forgejo/workflows/build.yml b/.forgejo/workflows/build.yml index eb4299b..7bd8066 100644 --- a/.forgejo/workflows/build.yml +++ b/.forgejo/workflows/build.yml @@ -45,6 +45,36 @@ on: type: boolean default: false + # The base-image refresh (milestone 326 step 4, #3154). + # + # Skip-if-exists is keyed on OUR source, so an artifact whose source stops + # moving stops picking up base-image updates. `agent/` last changed + # 2026-07-17; every push since has correctly declined to rebuild it, which + # also means it will serve that day's `nvidia/cuda` layers forever. Nothing + # is wrong until it has been unchanged for months, which is precisely why + # this is a calendar trigger and not a condition on the push path. + # + # Weekly, Sunday 06:00 UTC. Away from CI-runner's Monday security sweep so + # the two are never diagnosing each other, and on the quietest day so a + # surprise rebuild is not competing with a push. + schedule: + - cron: '0 6 * * 0' + +# Which branch a run BUILDS, as opposed to which one triggered it. +# +# They are the same thing on every trigger but `schedule`. Forgejo registers a +# cron from the DEFAULT branch — `dev` here — so a scheduled run arrives with +# `github.ref` pointing at dev, and a refresh that rebuilt `:dev` would be +# refreshing the one channel that gets rebuilt constantly anyway. Production is +# `main` (rule 147), and `:latest` is the tag that goes stale. +# +# So the ref is decided once, here, and every checkout in the file takes it. +# Deriving it per job invites the two halves to disagree: sign-extension would +# derive dev's extension version while build-web bundled main's, and the +# release download would 404 on a version that exists perfectly well. +env: + BUILD_REF: ${{ github.event_name == 'schedule' && 'main' || github.ref }} + # Requires repo secret RELEASE_TOKEN — a Forgejo PAT with scopes: # - write:package, read:package (for docker push to git.fabledsword.com) # - write:release (for ext- release asset cache) @@ -88,12 +118,43 @@ jobs: steps: - uses: actions/checkout@v4 with: + # Not the triggering ref — see the `env:` block at the top. On a + # scheduled refresh this is `main`; on everything else it is the ref + # that fired, so this is a no-op on every ordinary path. + ref: ${{ env.BUILD_REF }} # Full history is load-bearing, not a convenience: the version this # job signs is derived from the commit TIME of the newest packaged # extension change. A depth-1 clone sees one commit and derives a # wrong, too-low value rather than failing (ci-requirements.md). fetch-depth: 0 + # BUILD_REF is what makes a scheduled run build `main` rather than the + # branch its cron fired from — and it is read through the `env` context + # inside `with:`, which this runner is NOT known to evaluate. If it does + # not, checkout silently falls back to the triggering ref and the weekly + # refresh publishes DEV's source to `:latest`, which is production. + # Every lane would stay green; the first sign of it would be production + # running code that was never merged. + # + # So assert the checkout instead of trusting the expression. A red + # weekly job is a fine outcome. Shipping dev to production is not. + # + # `if:` reads the `github` context, which the runner demonstrably does + # evaluate — this file already gates steps on it — so the guard cannot + # be disabled by the same uncertainty it exists to cover. + - name: Guard — a scheduled run must have checked out main + if: github.event_name == 'schedule' + run: | + set -eu + BRANCH=$(git rev-parse --abbrev-ref HEAD) + echo "schedule: HEAD is $BRANCH ($(git rev-parse --short HEAD))" + if [ "$BRANCH" != "main" ]; then + echo "schedule: expected main, got '$BRANCH'." >&2 + echo "schedule: BUILD_REF was not honoured by the runner." >&2 + echo "schedule: refusing to publish a channel tag from it." >&2 + exit 1 + fi + # The version is DERIVED, not read from the repo (milestone 271 step 4, # cut over 2026-08-27). `packaging.sh version` returns `YYYY.M.D.HHMM` # UTC — the commit TIME of the newest change to a PACKAGED extension @@ -364,12 +425,30 @@ jobs: steps: - uses: actions/checkout@v4 with: + # Not the triggering ref — see the `env:` block at the top. On a + # scheduled refresh this is `main`; on everything else it is the ref + # that fired, so this is a no-op on every ordinary path. + ref: ${{ env.BUILD_REF }} # Full history: this job RE-DERIVES the extension version rather than # being handed it, and a depth-1 clone derives a wrong, too-low value # rather than failing — which would 404 the download of a release # that exists perfectly well under its real name. fetch-depth: 0 + # See sign-extension's copy for why this guard exists. + - name: Guard — a scheduled run must have checked out main + if: github.event_name == 'schedule' + run: | + set -eu + BRANCH=$(git rev-parse --abbrev-ref HEAD) + echo "schedule: HEAD is $BRANCH ($(git rev-parse --short HEAD))" + if [ "$BRANCH" != "main" ]; then + echo "schedule: expected main, got '$BRANCH'." >&2 + echo "schedule: BUILD_REF was not honoured by the runner." >&2 + echo "schedule: refusing to publish a channel tag from it." >&2 + exit 1 + fi + # --- derived values, one line (milestone 313) ------------------------ # These stopped being shadow output at step 3. `revision` decides # whether the build below runs at all and `version` is what the image @@ -429,8 +508,30 @@ jobs: # everywhere). Operator-flagged 2026-06-01 after the first :c- # main-push build failed at this step. SHORT_SHA=$(printf '%s' "$GITHUB_SHA" | cut -c1-7) - # Mirrors build-web's tag list; see the comment there. - if [ "${GITHUB_REF##*/}" = "main" ]; then + + # A scheduled refresh publishes the CHANNEL and nothing else + # (#3154). :c- for main's HEAD already exists and names the + # bytes that commit actually built; re-pushing it over refreshed + # base layers would break the one tag rule 145 makes immutable — + # and it is the rollback unit, so the breakage would surface on the + # day somebody needed it. + # + # The accepted consequence: between a refresh and the next main + # push, :latest and :c- point at different manifests. That is + # the design, not drift. They RE-CONVERGE on that push — it hits + # reuse (a refresh does not move fc.revision, because it does not + # touch the source), and the repoint step then writes the new + # :c- from the refreshed :latest. So the rollback unit ends up + # naming the bytes production is actually running, which is the + # property that matters. + # + # Checked BEFORE the ref test, not after: a scheduled run's + # GITHUB_REF is the default branch (dev), so the main test would + # never fire on it. + if [ "${GITHUB_EVENT_NAME:-}" = "schedule" ]; then + echo "tags=git.fabledsword.com/bvandeusen/fabledcurator:latest" >> "$GITHUB_OUTPUT" + echo "channel=main" >> "$GITHUB_OUTPUT" + elif [ "${GITHUB_REF##*/}" = "main" ]; then echo "tags=git.fabledsword.com/bvandeusen/fabledcurator:latest,git.fabledsword.com/bvandeusen/fabledcurator:c-${SHORT_SHA}" >> "$GITHUB_OUTPUT" echo "channel=main" >> "$GITHUB_OUTPUT" else @@ -524,6 +625,10 @@ jobs: # this runner is known to evaluate. Read through env rather than # interpolated into the run block, same rule as release.yml's TAG. FORCE: ${{ github.event.inputs.force_build }} + # A scheduled refresh has to bypass reuse by construction: it + # rebuilds the SAME source, so fc.revision always matches and the + # check would skip every refresh there has ever been. + EVENT: ${{ github.event_name }} run: | set -eu DERIVED=$(sh scripts/artifacts.sh revision web) @@ -570,6 +675,9 @@ jobs: if [ "${FORCE:-false}" = "true" ]; then echo "hit=false" >> "$GITHUB_OUTPUT" echo "reuse: force_build set — building regardless" + elif [ "${EVENT:-}" = "schedule" ]; then + echo "hit=false" >> "$GITHUB_OUTPUT" + echo "reuse: scheduled base refresh — building regardless" elif [ -n "$PUBLISHED" ] && [ "$PUBLISHED" = "$DERIVED" ]; then echo "hit=true" >> "$GITHUB_OUTPUT" echo "reuse: already published — skipping the build" @@ -661,6 +769,37 @@ jobs: context: . file: Dockerfile push: true + # Re-resolve the FROM references against the registry instead of + # trusting whatever digest the cache was built against. This is the + # whole mechanism of the scheduled refresh (#3154): if the base tag + # moved, the FROM layer's cache key changes, every layer above it + # invalidates, and the image genuinely rebuilds. + # + # MEASURED on the first real fire, run 4934 (#3265): when the base + # did NOT move, the build is ~13s and every content step reports + # CACHED — but the channel tag STILL gets a new manifest digest. + # buildkit mints a fresh image config each run, so identical layers + # are republished under a new config blob. All three images moved + # that way on 2026-08-30 with nothing whatsoever changed in them. + # + # So a refresh currently rewrites :latest every Sunday whether or + # not there is anything new in it, and :c- is handed a new + # manifest to diverge from on the same cadence. Layers are shared, + # so the storage cost is a config blob; the cost that matters is + # that a digest change no longer MEANS anything. Tracked in #3265 — + # the likely fix is a deterministic SOURCE_DATE_EPOCH, which would + # make "same source, same bytes" true and turn the no-op case into + # a genuine no-op. + # + # What `pull` does NOT catch either: a Debian package update inside + # the `apt-get install` layer while the base tag itself stands + # still. The official python/cuda images rebuild with those updates + # baked in, so this is a lag rather than a hole; closing it needs + # `no-cache: true`, which is a much larger version of the same + # churn #3265 is about. + # + # Only on the schedule. An ordinary push wants the cached base. + pull: ${{ github.event_name == 'schedule' }} # ONE tag, the channel's. Every other tag is written by the step # below, registry-side. buildx here pushes the first tag to the # registry and then re-pushes the rest through the DOCKER driver, @@ -783,6 +922,12 @@ jobs: ARGS="$ARGS -t $t" done unset IFS + # + # This is also the whole of the scheduled refresh's tag handling + # (#3154): a refresh's tag list is the channel tag alone, so SOURCE + # is the only entry, it gets excluded, and this step correctly does + # nothing. No `if:` on the step and no schedule special-case — + # excluding the source was already the right rule. if [ -z "$ARGS" ]; then echo "repoint: $SOURCE is the only tag for this channel and" echo "repoint: already holds this revision — nothing to write." @@ -799,6 +944,10 @@ jobs: steps: - uses: actions/checkout@v4 with: + # Not the triggering ref — see the `env:` block at the top. On a + # scheduled refresh this is `main`; on everything else it is the ref + # that fired, so this is a no-op on every ordinary path. + ref: ${{ env.BUILD_REF }} # Full history: this job derives its artifact's version from the # commit its shipped files last changed in (milestone 313). A # depth-1 clone cannot see that commit — it either derives a wrong, @@ -806,6 +955,20 @@ jobs: # the build would otherwise notice. fetch-depth: 0 + # See sign-extension's copy for why this guard exists. + - name: Guard — a scheduled run must have checked out main + if: github.event_name == 'schedule' + run: | + set -eu + BRANCH=$(git rev-parse --abbrev-ref HEAD) + echo "schedule: HEAD is $BRANCH ($(git rev-parse --short HEAD))" + if [ "$BRANCH" != "main" ]; then + echo "schedule: expected main, got '$BRANCH'." >&2 + echo "schedule: BUILD_REF was not honoured by the runner." >&2 + echo "schedule: refusing to publish a channel tag from it." >&2 + exit 1 + fi + # --- derived values, one line (milestone 313) ------------------------ # These stopped being shadow output at step 3. `revision` decides # whether the build below runs at all and `version` is what the image @@ -843,8 +1006,12 @@ jobs: # everywhere). Operator-flagged 2026-06-01 after first :c- # main-push build failed at this step. SHORT_SHA=$(printf '%s' "$GITHUB_SHA" | cut -c1-7) - # Mirrors build-web's tag list; see the comment there. - if [ "${GITHUB_REF##*/}" = "main" ]; then + # Mirrors build-web's tag list and its schedule handling; see + # the comments there. + if [ "${GITHUB_EVENT_NAME:-}" = "schedule" ]; then + echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-ml:latest" >> "$GITHUB_OUTPUT" + echo "channel=main" >> "$GITHUB_OUTPUT" + elif [ "${GITHUB_REF##*/}" = "main" ]; then echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-ml:latest,git.fabledsword.com/bvandeusen/fabledcurator-ml:c-${SHORT_SHA}" >> "$GITHUB_OUTPUT" echo "channel=main" >> "$GITHUB_OUTPUT" else @@ -921,6 +1088,10 @@ jobs: # this runner is known to evaluate. Read through env rather than # interpolated into the run block, same rule as release.yml's TAG. FORCE: ${{ github.event.inputs.force_build }} + # A scheduled refresh has to bypass reuse by construction: it + # rebuilds the SAME source, so fc.revision always matches and the + # check would skip every refresh there has ever been. + EVENT: ${{ github.event_name }} run: | set -eu DERIVED=$(sh scripts/artifacts.sh revision ml) @@ -963,6 +1134,9 @@ jobs: if [ "${FORCE:-false}" = "true" ]; then echo "hit=false" >> "$GITHUB_OUTPUT" echo "reuse: force_build set — building regardless" + elif [ "${EVENT:-}" = "schedule" ]; then + echo "hit=false" >> "$GITHUB_OUTPUT" + echo "reuse: scheduled base refresh — building regardless" elif [ -n "$PUBLISHED" ] && [ "$PUBLISHED" = "$DERIVED" ]; then echo "hit=true" >> "$GITHUB_OUTPUT" echo "reuse: already published — skipping the build" @@ -978,6 +1152,37 @@ jobs: context: . file: Dockerfile.ml push: true + # Re-resolve the FROM references against the registry instead of + # trusting whatever digest the cache was built against. This is the + # whole mechanism of the scheduled refresh (#3154): if the base tag + # moved, the FROM layer's cache key changes, every layer above it + # invalidates, and the image genuinely rebuilds. + # + # MEASURED on the first real fire, run 4934 (#3265): when the base + # did NOT move, the build is ~13s and every content step reports + # CACHED — but the channel tag STILL gets a new manifest digest. + # buildkit mints a fresh image config each run, so identical layers + # are republished under a new config blob. All three images moved + # that way on 2026-08-30 with nothing whatsoever changed in them. + # + # So a refresh currently rewrites :latest every Sunday whether or + # not there is anything new in it, and :c- is handed a new + # manifest to diverge from on the same cadence. Layers are shared, + # so the storage cost is a config blob; the cost that matters is + # that a digest change no longer MEANS anything. Tracked in #3265 — + # the likely fix is a deterministic SOURCE_DATE_EPOCH, which would + # make "same source, same bytes" true and turn the no-op case into + # a genuine no-op. + # + # What `pull` does NOT catch either: a Debian package update inside + # the `apt-get install` layer while the base tag itself stands + # still. The official python/cuda images rebuild with those updates + # baked in, so this is a lag rather than a hole; closing it needs + # `no-cache: true`, which is a much larger version of the same + # churn #3265 is about. + # + # Only on the schedule. An ordinary push wants the cached base. + pull: ${{ github.event_name == 'schedule' }} # ONE tag, the channel's. Every other tag is written by the step # below, registry-side. buildx here pushes the first tag to the # registry and then re-pushes the rest through the DOCKER driver, @@ -1113,6 +1318,10 @@ jobs: steps: - uses: actions/checkout@v4 with: + # Not the triggering ref — see the `env:` block at the top. On a + # scheduled refresh this is `main`; on everything else it is the ref + # that fired, so this is a no-op on every ordinary path. + ref: ${{ env.BUILD_REF }} # Full history: this job derives its artifact's version from the # commit its shipped files last changed in (milestone 313). A # depth-1 clone cannot see that commit — it either derives a wrong, @@ -1120,6 +1329,20 @@ jobs: # the build would otherwise notice. fetch-depth: 0 + # See sign-extension's copy for why this guard exists. + - name: Guard — a scheduled run must have checked out main + if: github.event_name == 'schedule' + run: | + set -eu + BRANCH=$(git rev-parse --abbrev-ref HEAD) + echo "schedule: HEAD is $BRANCH ($(git rev-parse --short HEAD))" + if [ "$BRANCH" != "main" ]; then + echo "schedule: expected main, got '$BRANCH'." >&2 + echo "schedule: BUILD_REF was not honoured by the runner." >&2 + echo "schedule: refusing to publish a channel tag from it." >&2 + exit 1 + fi + # --- derived values, one line (milestone 313) ------------------------ # These stopped being shadow output at step 3. `revision` decides # whether the build below runs at all and `version` is what the image @@ -1152,8 +1375,12 @@ jobs: id: tag run: | SHORT_SHA=$(printf '%s' "$GITHUB_SHA" | cut -c1-7) - # Mirrors build-web's tag list; see the comment there. - if [ "${GITHUB_REF##*/}" = "main" ]; then + # Mirrors build-web's tag list and its schedule handling; see + # the comments there. + if [ "${GITHUB_EVENT_NAME:-}" = "schedule" ]; then + echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-agent:latest" >> "$GITHUB_OUTPUT" + echo "channel=main" >> "$GITHUB_OUTPUT" + elif [ "${GITHUB_REF##*/}" = "main" ]; then echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-agent:latest,git.fabledsword.com/bvandeusen/fabledcurator-agent:c-${SHORT_SHA}" >> "$GITHUB_OUTPUT" echo "channel=main" >> "$GITHUB_OUTPUT" else @@ -1230,6 +1457,10 @@ jobs: # this runner is known to evaluate. Read through env rather than # interpolated into the run block, same rule as release.yml's TAG. FORCE: ${{ github.event.inputs.force_build }} + # A scheduled refresh has to bypass reuse by construction: it + # rebuilds the SAME source, so fc.revision always matches and the + # check would skip every refresh there has ever been. + EVENT: ${{ github.event_name }} run: | set -eu DERIVED=$(sh scripts/artifacts.sh revision agent) @@ -1272,6 +1503,9 @@ jobs: if [ "${FORCE:-false}" = "true" ]; then echo "hit=false" >> "$GITHUB_OUTPUT" echo "reuse: force_build set — building regardless" + elif [ "${EVENT:-}" = "schedule" ]; then + echo "hit=false" >> "$GITHUB_OUTPUT" + echo "reuse: scheduled base refresh — building regardless" elif [ -n "$PUBLISHED" ] && [ "$PUBLISHED" = "$DERIVED" ]; then echo "hit=true" >> "$GITHUB_OUTPUT" echo "reuse: already published — skipping the build" @@ -1287,6 +1521,37 @@ jobs: context: agent file: agent/Dockerfile push: true + # Re-resolve the FROM references against the registry instead of + # trusting whatever digest the cache was built against. This is the + # whole mechanism of the scheduled refresh (#3154): if the base tag + # moved, the FROM layer's cache key changes, every layer above it + # invalidates, and the image genuinely rebuilds. + # + # MEASURED on the first real fire, run 4934 (#3265): when the base + # did NOT move, the build is ~13s and every content step reports + # CACHED — but the channel tag STILL gets a new manifest digest. + # buildkit mints a fresh image config each run, so identical layers + # are republished under a new config blob. All three images moved + # that way on 2026-08-30 with nothing whatsoever changed in them. + # + # So a refresh currently rewrites :latest every Sunday whether or + # not there is anything new in it, and :c- is handed a new + # manifest to diverge from on the same cadence. Layers are shared, + # so the storage cost is a config blob; the cost that matters is + # that a digest change no longer MEANS anything. Tracked in #3265 — + # the likely fix is a deterministic SOURCE_DATE_EPOCH, which would + # make "same source, same bytes" true and turn the no-op case into + # a genuine no-op. + # + # What `pull` does NOT catch either: a Debian package update inside + # the `apt-get install` layer while the base tag itself stands + # still. The official python/cuda images rebuild with those updates + # baked in, so this is a lag rather than a hole; closing it needs + # `no-cache: true`, which is a much larger version of the same + # churn #3265 is about. + # + # Only on the schedule. An ordinary push wants the cached base. + pull: ${{ github.event_name == 'schedule' }} # ONE tag, the channel's. Every other tag is written by the step # below, registry-side. buildx here pushes the first tag to the # registry and then re-pushes the rest through the DOCKER driver, diff --git a/alembic/versions/0088_reconcile_models_with_schema.py b/alembic/versions/0088_reconcile_models_with_schema.py new file mode 100644 index 0000000..d5b84c2 --- /dev/null +++ b/alembic/versions/0088_reconcile_models_with_schema.py @@ -0,0 +1,128 @@ +"""Reconcile the database with what the models have always claimed (#3275). + +Milestone 328 discovered ~130 places where the ORM models and the deployed +schema disagreed. Almost all of them were the MODEL being wrong — missing +`server_default`s, indexes and CHECK constraints that only ever existed in a +migration — and those are fixed in the model files with no DDL at all, because +the database already had them. + +This migration carries the remainder — the two places where DDL is actually +needed, because the database is what is wrong. + +`tag.fandom_id` is declared `index=True` on the model, but no migration ever +created that index. Every autogenerate run since would have proposed adding +it; nobody ran one, so the model and the database simply drifted apart and +stayed that way. + +Deliberately NOT in this migration: anything about `image_record.sha256`. An +earlier draft of this file claimed sha256 was not unique in the database and +that duplicate rows were therefore possible. That was WRONG, and it was wrong +because it was read off `op.create_index("ix_image_record_sha256", ...)` at +0001 line 151 without reading line 149 two lines above it: + + sa.UniqueConstraint("sha256", name="uq_image_record_sha256"), + +Uniqueness has been enforced since the initial schema. The database simply +expresses it as a CONSTRAINT plus a separate non-unique lookup index, where +the model expressed it as one `unique=True, index=True` column — the same +guarantee built from different objects, which is why the two schemas did not +line up. The model now declares the constraint and the plain index separately, +so it describes what is actually there. No DDL is needed for it. + +Also here: six CHECK constraints whose names carry their table prefix TWICE. + +`base.py`'s naming convention is `ck_%(table_name)s_%(constraint_name)s`, and +unlike the uq/fk/ix entries it applies even to a constraint that already has a +name. Six migrations passed an already-prefixed name, so the convention +prefixed it again: + + ck_external_link_ck_external_link_host + ck_external_link_ck_external_link_status + ck_import_settings_ck_import_settings_singleton + ck_ml_settings_ck_ml_settings_singleton + ck_post_ck_post_translation_override + ck_tag_ck_tag_fandom_requires_character + +Nothing reads a CHECK constraint by name, so this has never done any harm — +but it is exactly the development-era residue the collapsed baseline exists to +leave behind, and a public schema should not ship it. The models now declare +bare names, which the convention renders into the single-prefix form; this +renames the deployed constraints to match. + +RENAME CONSTRAINT is a catalog-only operation: no table scan, no rewrite, no +validation of existing rows. It takes a brief ACCESS EXCLUSIVE lock and +returns. That is why this is safe to do on `post` and `tag`, which are the two +large tables in the schema. + +Revision ID: 0088 +Revises: 0087 +Create Date: 2026-08-30 + +""" +from typing import Sequence, Union + +from alembic import op + +revision: str = "0088" +down_revision: Union[str, None] = "0087" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + +# (table, doubled name, single-prefix name) +# +# Six, not the four a first read of the migrations turned up. The list that +# settles it is the one extracted from the chain's pg_dump by matching +# `ck_(\w+?)_ck_\1_` — reading the migrations by eye missed external_link +# twice over, in the same way an earlier pass missed a UNIQUE constraint two +# lines above the index it was looking at (see the sha256 note above). +DOUBLED_CHECKS = ( + ("external_link", "ck_external_link_ck_external_link_host", + "ck_external_link_host"), + ("external_link", "ck_external_link_ck_external_link_status", + "ck_external_link_status"), + ("import_settings", "ck_import_settings_ck_import_settings_singleton", + "ck_import_settings_singleton"), + ("ml_settings", "ck_ml_settings_ck_ml_settings_singleton", + "ck_ml_settings_singleton"), + ("post", "ck_post_ck_post_translation_override", + "ck_post_translation_override"), + ("tag", "ck_tag_ck_tag_fandom_requires_character", + "ck_tag_fandom_requires_character"), +) + + +def _rename_check(table: str, old: str, new: str) -> None: + # Guarded on pg_constraint rather than run bare: a database built from the + # models (a fresh install, or the CI integration schema) already has the + # single-prefix name, and this migration must be a no-op there rather than + # an error. Same reasoning as the CREATE INDEX IF NOT EXISTS below. + op.execute( + f""" + DO $$ + BEGIN + IF EXISTS ( + SELECT 1 FROM pg_constraint + WHERE conname = '{old}' AND conrelid = '{table}'::regclass + ) THEN + ALTER TABLE {table} RENAME CONSTRAINT {old} TO {new}; + END IF; + END $$; + """ + ) + + +def upgrade() -> None: + # IF NOT EXISTS because the index is what the model already asks for: any + # database built from metadata rather than from this chain will have it, + # and this migration must be a no-op there rather than an error. + op.execute("CREATE INDEX IF NOT EXISTS ix_tag_fandom_id ON tag (fandom_id)") + + for table, old, new in DOUBLED_CHECKS: + _rename_check(table, old, new) + + +def downgrade() -> None: + for table, old, new in DOUBLED_CHECKS: + _rename_check(table, new, old) + + op.execute("DROP INDEX IF EXISTS ix_tag_fandom_id") diff --git a/alembic/versions/0089_index_hygiene.py b/alembic/versions/0089_index_hygiene.py new file mode 100644 index 0000000..0b22b5c --- /dev/null +++ b/alembic/versions/0089_index_hygiene.py @@ -0,0 +1,120 @@ +"""Index the seven unindexed FKs; drop the seven redundant indexes (#3300, #3301). + +Found by a structural sweep of the deployed schema done AFTER 0088 brought the +models and the migration chain into exact agreement. That agreement is what +0088 achieved, and it is worth being precise about what it does NOT prove: a +models-vs-chain diff shows the two describe the same schema. It says nothing +about whether that schema is right. Everything here was wrong in BOTH, which is +exactly the class of problem the reconciliation could not see. + +## Added: seven FK indexes + +`image_tag.tag_id` is the one that matters. The table's only index is +PRIMARY KEY (image_record_id, tag_id), which leads with the wrong column for +the two hottest things done with it: + + * the gallery's tag filter — services/tag_query.py builds + `image_tag.c.tag_id == tid` (and `.in_(tids)`) on every tag-scoped browse; + * ON DELETE CASCADE from `tag` — deleting or merging a tag makes Postgres + find that tag's rows before it can remove them. + +Both had to scan the largest table in the schema. The other six are the same +shape on much smaller tables; `presentation_review.tag_id` is the notable one, +since it also CASCADEs. + +## Dropped: seven redundant indexes + +`ix_image_record_sha256` was an exact duplicate. A UNIQUE constraint builds its +own index, so `uq_image_record_sha256` already covered the column and +`image_record` carried two btrees on `sha256` — on the highest-insert-rate +table in the system. + +The other six are single-column indexes that a later composite superseded +without the narrow one being retired. A btree on (a, b) already serves lookups +on `a`, so each was pure write amplification. `task_run` and `backup_run` are +append-heavy operational logs, which is where that cost lands hardest. + +Note for anyone reading 0088 next to this: 0088 deliberately taught the models +to declare BOTH sha256 indexes, so they would describe reality. That was right. +This migration changes the reality instead, and the models change with it. + +## CONCURRENTLY, and why this migration has no transaction + +`CREATE INDEX` takes an ACCESS EXCLUSIVE lock for the whole build, which on +`image_tag` means stalling every write for as long as it takes. CONCURRENTLY +builds without blocking writers, at the cost of two table passes and an +inability to run inside a transaction — hence `autocommit_block()`. + +The consequence to know about: this migration is NOT atomic. If it fails +partway, the work already done stays done. Every statement is therefore written +IF NOT EXISTS / IF EXISTS so that re-running it after a failure is safe rather +than an error. + +A failed CONCURRENTLY build also leaves an INVALID index behind — it is not +used by the planner and not repaired automatically. Find them with: + + SELECT c.relname FROM pg_index i + JOIN pg_class c ON c.oid = i.indexrelid + WHERE NOT i.indisvalid; + +Drop what that returns and re-run; nothing else is needed. + +Revision ID: 0089 +Revises: 0088 +Create Date: 2026-08-31 + +""" +from typing import Sequence, Union + +from alembic import op + +revision: str = "0089" +down_revision: Union[str, None] = "0088" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + +# (index name, table, column) — names match what the models render under +# base.py's naming convention, so autogenerate stays quiet after this. +MISSING_FK_INDEXES = ( + ("ix_image_tag_tag_id", "image_tag", "tag_id"), + ("ix_presentation_review_tag_id", "presentation_review", "tag_id"), + ("ix_presentation_review_conflict_tag_id", "presentation_review", "conflict_tag_id"), + ("ix_import_task_result_image_id", "import_task", "result_image_id"), + ("ix_external_link_attachment_id", "external_link", "attachment_id"), + ("ix_character_prototype_region_id", "character_prototype", "region_id"), + ("ix_backup_run_restored_from_id", "backup_run", "restored_from_id"), +) + +# (index name, table, column) — redundant; the second element of each pair in +# the docstring is what still covers the column after the drop. +REDUNDANT_INDEXES = ( + ("ix_image_record_sha256", "image_record", "sha256"), + ("ix_backup_run_kind", "backup_run", "kind"), + ("ix_backup_run_status", "backup_run", "status"), + ("ix_task_run_queue", "task_run", "queue"), + ("ix_task_run_status", "task_run", "status"), + ("ix_task_run_task_name", "task_run", "task_name"), + ("ix_external_link_post_id", "external_link", "post_id"), +) + + +def upgrade() -> None: + with op.get_context().autocommit_block(): + for name, table, column in MISSING_FK_INDEXES: + op.execute( + f"CREATE INDEX CONCURRENTLY IF NOT EXISTS {name} " + f"ON {table} ({column})" + ) + for name, _table, _column in REDUNDANT_INDEXES: + op.execute(f"DROP INDEX CONCURRENTLY IF EXISTS {name}") + + +def downgrade() -> None: + with op.get_context().autocommit_block(): + for name, table, column in REDUNDANT_INDEXES: + op.execute( + f"CREATE INDEX CONCURRENTLY IF NOT EXISTS {name} " + f"ON {table} ({column})" + ) + for name, _table, _column in MISSING_FK_INDEXES: + op.execute(f"DROP INDEX CONCURRENTLY IF EXISTS {name}") diff --git a/backend/app/models/artist.py b/backend/app/models/artist.py index e7ca894..fb3401c 100644 --- a/backend/app/models/artist.py +++ b/backend/app/models/artist.py @@ -27,10 +27,10 @@ class Artist(Base): notes: Mapped[str | None] = mapped_column(Text, nullable=True) # True once a Source is attached; flips false if all sources removed. - is_subscription: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False) + is_subscription: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, server_default="false") # Per-artist scheduling overrides; null means "use global default". - auto_check: Mapped[bool] = mapped_column(Boolean, nullable=False, default=True) + auto_check: Mapped[bool] = mapped_column(Boolean, nullable=False, default=True, server_default="true") check_interval_seconds: Mapped[int | None] = mapped_column(Integer, nullable=True) created_at: Mapped[datetime] = mapped_column( diff --git a/backend/app/models/backup_run.py b/backend/app/models/backup_run.py index 1717aea..d5e54cd 100644 --- a/backend/app/models/backup_run.py +++ b/backend/app/models/backup_run.py @@ -20,7 +20,7 @@ feedback_check_existing_enums): from datetime import datetime -from sqlalchemy import JSON, BigInteger, DateTime, ForeignKey, Integer, String, Text +from sqlalchemy import JSON, BigInteger, DateTime, ForeignKey, Index, Integer, String, Text, text from sqlalchemy.orm import Mapped, mapped_column from .base import Base @@ -29,10 +29,21 @@ from .base import Base class BackupRun(Base): __tablename__ = "backup_run" + + __table_args__ = ( + # alembic 0017: reporting indexes, never declared on the model (#3275). + Index("ix_backup_run_kind_started", "kind", text("started_at DESC")), + Index("ix_backup_run_status_finished", "status", text("finished_at DESC")), + Index("ix_backup_run_tag_partial", "tag", postgresql_where=text("tag IS NOT NULL")), + ) id: Mapped[int] = mapped_column(Integer, primary_key=True) - kind: Mapped[str] = mapped_column(String(16), nullable=False, index=True) + # No index=True: ix_backup_run_kind_started (above) already leads with + # `kind`, so a single-column index on it was pure write cost (#3301). + kind: Mapped[str] = mapped_column(String(16), nullable=False) status: Mapped[str] = mapped_column( - String(16), nullable=False, default="pending", index=True, + # No index=True — ix_backup_run_status_finished leads with `status`. + String(16), nullable=False, default="pending", + server_default="pending", ) tag: Mapped[str | None] = mapped_column(String(64), nullable=True, index=True) triggered_by: Mapped[str] = mapped_column(String(32), nullable=False) @@ -49,7 +60,9 @@ class BackupRun(Base): manifest: Mapped[dict] = mapped_column( JSON, nullable=False, default=dict, server_default="{}", ) + # Self-referential FK, unindexed until 0089 (#3300): SET NULL has to find + # the rows pointing at a deleted run before it can null them. restored_from_id: Mapped[int | None] = mapped_column( ForeignKey("backup_run.id", ondelete="SET NULL"), - nullable=True, + nullable=True, index=True, ) diff --git a/backend/app/models/character_prototype.py b/backend/app/models/character_prototype.py index 191a29d..f281251 100644 --- a/backend/app/models/character_prototype.py +++ b/backend/app/models/character_prototype.py @@ -40,8 +40,10 @@ class CharacterPrototype(Base): ) # Provenance: the region this vector was copied from. SET NULL so pruning a # region doesn't delete the prototype mid-cycle (the next refresh reconciles). + # index=True added in 0089 — the FK was unindexed (#3300). region_id: Mapped[int | None] = mapped_column( - ForeignKey("image_region.id", ondelete="SET NULL"), nullable=True + ForeignKey("image_region.id", ondelete="SET NULL"), nullable=True, + index=True, ) diff --git a/backend/app/models/download_event.py b/backend/app/models/download_event.py index 7fe5b37..3ec00ae 100644 --- a/backend/app/models/download_event.py +++ b/backend/app/models/download_event.py @@ -25,8 +25,8 @@ class DownloadEvent(Base): DateTime(timezone=True), nullable=False, server_default=func.now() ) finished_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True) - bytes_downloaded: Mapped[int] = mapped_column(BigInteger, nullable=False, default=0) - files_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0) + bytes_downloaded: Mapped[int] = mapped_column(BigInteger, nullable=False, default=0, server_default="0") + files_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") error: Mapped[str | None] = mapped_column(Text, nullable=True) metadata_: Mapped[dict] = mapped_column( "metadata", JSONB, nullable=False, default=dict, diff --git a/backend/app/models/external_link.py b/backend/app/models/external_link.py index 0902e28..4882125 100644 --- a/backend/app/models/external_link.py +++ b/backend/app/models/external_link.py @@ -16,6 +16,7 @@ doesn't delete the link record). from datetime import datetime from sqlalchemy import ( + CheckConstraint, DateTime, Float, ForeignKey, @@ -38,15 +39,33 @@ STATUSES = ("pending", "downloading", "downloaded", "failed", "skipped", "dead") class ExternalLink(Base): __tablename__ = "external_link" __table_args__ = ( + # alembic 0028 enum CHECKs. Rule 36 territory: a new host or status value + # needs its constraint swapped in the same migration (#3275). + CheckConstraint( + "host IN ('mega', 'gdrive', 'mediafire', 'dropbox', 'pixeldrain')", + # Bare name: Base.metadata's naming convention prepends + # ck_
_. Pre-prefixing it here doubles the prefix — see + # alembic 0088, which renames the four constraints that shipped + # that way (#3275). + name="host", + ), + CheckConstraint( + "status IN ('pending', 'downloading', 'downloaded', 'failed', 'skipped', 'dead')", + name="status", + ), # One row per (post, url). The full url (incl. #fragment) is the identity # — the same file linked twice in a post collapses to one row. Index("uq_external_link_post_url", "post_id", "url", unique=True), Index("ix_external_link_status", "status"), + # Unindexed FK (#3300). + Index("ix_external_link_attachment_id", "attachment_id"), ) id: Mapped[int] = mapped_column(Integer, primary_key=True) + # No index=True: uq_external_link_post_url (post_id, url) already leads + # with post_id (#3301). post_id: Mapped[int] = mapped_column( - ForeignKey("post.id", ondelete="CASCADE"), nullable=False, index=True + ForeignKey("post.id", ondelete="CASCADE"), nullable=False ) artist_id: Mapped[int | None] = mapped_column( ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, index=True diff --git a/backend/app/models/gpu_job.py b/backend/app/models/gpu_job.py index dba5997..931b455 100644 --- a/backend/app/models/gpu_job.py +++ b/backend/app/models/gpu_job.py @@ -50,7 +50,8 @@ class GpuJob(Base): # What to compute, e.g. 'ccip' (detect figures + CCIP-embed) or 'siglip_region'. task: Mapped[str] = mapped_column(String(32), nullable=False) status: Mapped[str] = mapped_column( - String(16), nullable=False, default="pending", index=True + String(16), nullable=False, default="pending", index=True, + server_default="pending", ) # pending | leased | done | error lease_token: Mapped[str | None] = mapped_column(String(64), nullable=True) @@ -60,7 +61,7 @@ class GpuJob(Base): lease_expires_at: Mapped[datetime | None] = mapped_column( DateTime(timezone=True), nullable=True ) - attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=0) + attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") error: Mapped[str | None] = mapped_column(Text, nullable=True) # Triage verdict for an ERRORED job (#125): NULL = not yet probed; # 'defect' = the integrity probe says the FILE itself is bad (surfaced for diff --git a/backend/app/models/head_auto_apply_run.py b/backend/app/models/head_auto_apply_run.py index 08c359a..031109b 100644 --- a/backend/app/models/head_auto_apply_run.py +++ b/backend/app/models/head_auto_apply_run.py @@ -24,10 +24,11 @@ class HeadAutoApplyRun(Base): id: Mapped[int] = mapped_column(Integer, primary_key=True) # dry_run=True is a PREVIEW: scores + counts what WOULD apply, writes nothing # (preview/apply parity, rule 93). - dry_run: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False) + dry_run: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, server_default="false") params: Mapped[dict[str, Any]] = mapped_column(JSONB, nullable=False) status: Mapped[str] = mapped_column( - String(16), nullable=False, default="running", index=True + String(16), nullable=False, default="running", index=True, + server_default="running", ) # running | ready | error started_at: Mapped[datetime] = mapped_column( diff --git a/backend/app/models/head_metric.py b/backend/app/models/head_metric.py index a034e51..7afc104 100644 --- a/backend/app/models/head_metric.py +++ b/backend/app/models/head_metric.py @@ -24,9 +24,9 @@ class HeadMetric(Base): ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True ) # An auto-applied (source='head_auto') tag the operator later REMOVED. - n_misfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0) + n_misfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") # A tag with a head that the operator added by HAND (the head missed it). - n_underfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0) + n_underfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") updated_at: Mapped[datetime] = mapped_column( DateTime(timezone=True), nullable=False, server_default=func.now() ) diff --git a/backend/app/models/head_metrics_snapshot.py b/backend/app/models/head_metrics_snapshot.py index a9ec7ac..dfde05e 100644 --- a/backend/app/models/head_metrics_snapshot.py +++ b/backend/app/models/head_metrics_snapshot.py @@ -19,8 +19,14 @@ class HeadMetricsSnapshot(Base): __tablename__ = "head_metrics_snapshot" id: Mapped[int] = mapped_column(Integer, primary_key=True) - tag_id: Mapped[int] = mapped_column( - ForeignKey("tag.id", ondelete="CASCADE"), index=True + # Nullable, matching alembic 0060, which declared this column without + # `nullable=False`. The model had it as `Mapped[int]` — NOT NULL — which + # was simply never true of the database (#3275). Left nullable rather than + # tightened: a snapshot of a tag that is later hard-deleted is a row worth + # keeping, and the FK is ON DELETE CASCADE, so tightening it would only + # change behaviour, not correct a bug. + tag_id: Mapped[int | None] = mapped_column( + ForeignKey("tag.id", ondelete="CASCADE"), nullable=True, index=True ) # Denormalized so a snapshot stays readable even if the tag is later renamed. name: Mapped[str] = mapped_column(String(255), nullable=False) @@ -28,9 +34,9 @@ class HeadMetricsSnapshot(Base): DateTime(timezone=True), nullable=False, server_default=func.now(), index=True ) # Current count of source='head_auto' applications still standing. - n_auto_applied: Mapped[int] = mapped_column(Integer, nullable=False, default=0) - n_misfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0) - n_underfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0) + n_auto_applied: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") + n_misfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") + n_underfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") # The head's measured quality at snapshot time (null if no head exists). ap: Mapped[float | None] = mapped_column(Float, nullable=True) precision_cv: Mapped[float | None] = mapped_column(Float, nullable=True) diff --git a/backend/app/models/head_training_run.py b/backend/app/models/head_training_run.py index fd5858e..21c4ac7 100644 --- a/backend/app/models/head_training_run.py +++ b/backend/app/models/head_training_run.py @@ -24,7 +24,8 @@ class HeadTrainingRun(Base): # Training parameters: {min_positives, neg_ratio, precision_target, ...}. params: Mapped[dict[str, Any]] = mapped_column(JSONB, nullable=False) status: Mapped[str] = mapped_column( - String(16), nullable=False, default="running", index=True + String(16), nullable=False, default="running", index=True, + server_default="running", ) # running | ready | error started_at: Mapped[datetime] = mapped_column( diff --git a/backend/app/models/image_provenance.py b/backend/app/models/image_provenance.py index fb18178..2ce95de 100644 --- a/backend/app/models/image_provenance.py +++ b/backend/app/models/image_provenance.py @@ -47,8 +47,15 @@ class ImageProvenance(Base): # attachment on the post. NULL for loose downloads and pre-backfill rows. # SET NULL so deleting the archive attachment never destroys the (image, # post) edge — it just forgets which archive it came from. + # FK named explicitly: the convention renders this + # `fk_image_provenance_from_attachment_id_post_attachment`, but alembic + # 0055 created it as `fk_image_provenance_from_attachment` (#3275). from_attachment_id: Mapped[int | None] = mapped_column( - ForeignKey("post_attachment.id", ondelete="SET NULL"), + ForeignKey( + "post_attachment.id", + ondelete="SET NULL", + name="fk_image_provenance_from_attachment", + ), nullable=True, index=True, ) captured_metadata: Mapped[dict | None] = mapped_column(JSON, nullable=True) diff --git a/backend/app/models/image_record.py b/backend/app/models/image_record.py index 32d6eaa..f5f4050 100644 --- a/backend/app/models/image_record.py +++ b/backend/app/models/image_record.py @@ -14,10 +14,13 @@ from sqlalchemy import ( Enum, Float, ForeignKey, + Index, Integer, String, Text, + UniqueConstraint, func, + text, ) from sqlalchemy.orm import Mapped, mapped_column @@ -29,11 +32,38 @@ ORIGIN_CHOICES = ("downloaded", "imported_filesystem", "uploaded") class ImageRecord(Base): __tablename__ = "image_record" + + __table_args__ = ( + # alembic 0001. The database enforces sha256 uniqueness with a + # CONSTRAINT and carries a SEPARATE non-unique btree index; the model + # said `unique=True, index=True`, which collapses both into a single + # UNIQUE index under a different name. Same guarantee either way, but + # not the same objects, so autogenerate saw a drop and an add (#3275). + UniqueConstraint("sha256", name="uq_image_record_sha256"), + # alembic 0036, and the last thing in this schema that lived only in a + # migration. SQLAlchemy CAN express an hnsw index with an operator + # class, so there is no reason for it to be invisible to the models — + # and its absence was the quietest failure of the lot: everything + # works, similarity search just silently stops using an index. + Index( + "ix_image_record_siglip_hnsw", + "siglip_embedding", + postgresql_using="hnsw", + postgresql_ops={"siglip_embedding": "vector_cosine_ops"}, + ), + # alembic 0035/0071: the date-ordered browse indexes (#3275). + Index("ix_image_record_effective_date", text("effective_date DESC"), text("id DESC")), + Index("ix_image_record_earliest_post_date", text("earliest_post_date DESC"), text("id DESC")), + ) id: Mapped[int] = mapped_column(Integer, primary_key=True) # On-disk identity path: Mapped[str] = mapped_column(Text, nullable=False, unique=True) - sha256: Mapped[str] = mapped_column(String(64), nullable=False, unique=True, index=True) + # Neither unique= nor index=: uq_image_record_sha256 in __table_args__ + # above creates its own index, and the separate ix_image_record_sha256 + # that 0001 also built was an exact duplicate of it — dropped in 0089 + # (#3301). Lookups by sha256 use the constraint's index. + sha256: Mapped[str] = mapped_column(String(64), nullable=False) phash: Mapped[str | None] = mapped_column(String(32), nullable=True, index=True) size_bytes: Mapped[int] = mapped_column(BigInteger, nullable=False) mime: Mapped[str] = mapped_column(String(64), nullable=False) @@ -47,7 +77,8 @@ class ImageRecord(Base): # Integrity verification status. FC-2e populates this; FC-2a leaves rows at 'unknown'. # Values: 'unknown' (default), 'ok', 'corrupt', 'failed_verification'. integrity_status: Mapped[str] = mapped_column( - String(24), nullable=False, default="unknown", index=True + String(24), nullable=False, default="unknown", index=True, + server_default="unknown", ) # Thumbnail (populated by FC-2) @@ -72,8 +103,15 @@ class ImageRecord(Base): ) # FC-2d-vii-c: canonical per-image artist (the single source of truth # for attribution; provenance posts remain lineage detail). + # FK named explicitly: the naming convention renders this + # `fk_image_record_artist_id_artist`, but alembic 0008 created it as + # `fk_image_record_artist_id` (#3275). artist_id: Mapped[int | None] = mapped_column( - ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, index=True + ForeignKey( + "artist.id", ondelete="SET NULL", name="fk_image_record_artist_id" + ), + nullable=True, + index=True, ) # ML fields (populated by the ml-worker / GPU agent). 1152 = SigLIP-so400m diff --git a/backend/app/models/import_batch.py b/backend/app/models/import_batch.py index 474f111..8d8fa61 100644 --- a/backend/app/models/import_batch.py +++ b/backend/app/models/import_batch.py @@ -21,17 +21,17 @@ class ImportBatch(Base): ) finished_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True) - total_files: Mapped[int] = mapped_column(Integer, nullable=False, default=0) - imported: Mapped[int] = mapped_column(Integer, nullable=False, default=0) - skipped: Mapped[int] = mapped_column(Integer, nullable=False, default=0) - failed: Mapped[int] = mapped_column(Integer, nullable=False, default=0) - attachments: Mapped[int] = mapped_column(Integer, nullable=False, default=0) + total_files: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") + imported: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") + skipped: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") + failed: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") + attachments: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") # Deep-scan only: count of already-imported files whose sidecar metadata # got re-applied this run (post/source/provenance upsert). Stays 0 on # quick-scan batches. See `Importer.import_one(deep_scan=True)`. - refreshed: Mapped[int] = mapped_column(Integer, nullable=False, default=0) + refreshed: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") - status: Mapped[str] = mapped_column(String(16), nullable=False, default="running", index=True) + status: Mapped[str] = mapped_column(String(16), nullable=False, default="running", index=True, server_default="running") # running | complete | cancelled tasks = relationship("ImportTask", back_populates="batch", cascade="all, delete-orphan") diff --git a/backend/app/models/import_settings.py b/backend/app/models/import_settings.py index f4b8937..83ae9c2 100644 --- a/backend/app/models/import_settings.py +++ b/backend/app/models/import_settings.py @@ -4,7 +4,15 @@ Enforced as a single row via a CHECK (id = 1) constraint. The application always SELECTs id=1 and never inserts/deletes after the initial migration. """ -from sqlalchemy import Boolean, CheckConstraint, Float, Integer, Text, select +from sqlalchemy import ( + Boolean, + CheckConstraint, + Float, + Integer, + Text, + select, + text, +) from sqlalchemy.orm import Mapped, mapped_column from .base import Base @@ -14,63 +22,79 @@ class ImportSettings(Base): __tablename__ = "import_settings" # Bare constraint name — Base.metadata's naming convention applies the # ck_
_ prefix, producing the final ck_import_settings_singleton. + # Bare name — Base.metadata's naming convention prepends ck_
_, + # producing ck_import_settings_singleton. The chain shipped the DOUBLED + # ck_import_settings_ck_import_settings_singleton, because the migration + # pre-prefixed the name and the convention prefixed it again; alembic + # 0088 renames it to what this line has always produced (#3275). __table_args__ = (CheckConstraint("id = 1", name="singleton"),) id: Mapped[int] = mapped_column(Integer, primary_key=True) - import_scan_path: Mapped[str] = mapped_column(Text, nullable=False, default="/import") + import_scan_path: Mapped[str] = mapped_column(Text, nullable=False, default="/import", server_default="/import") - min_width: Mapped[int] = mapped_column(Integer, nullable=False, default=0) - min_height: Mapped[int] = mapped_column(Integer, nullable=False, default=0) + min_width: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") + min_height: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") - skip_transparent: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False) - transparency_threshold: Mapped[float] = mapped_column(Float, nullable=False, default=0.9) + skip_transparent: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, server_default="false") + transparency_threshold: Mapped[float] = mapped_column(Float, nullable=False, default=0.9, server_default="0.9") - skip_single_color: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False) - single_color_threshold: Mapped[float] = mapped_column(Float, nullable=False, default=0.95) - single_color_tolerance: Mapped[int] = mapped_column(Integer, nullable=False, default=30) + skip_single_color: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, server_default="false") + single_color_threshold: Mapped[float] = mapped_column(Float, nullable=False, default=0.95, server_default="0.95") + single_color_tolerance: Mapped[int] = mapped_column(Integer, nullable=False, default=30, server_default="30") - phash_threshold: Mapped[int] = mapped_column(Integer, nullable=False, default=10) + phash_threshold: Mapped[int] = mapped_column(Integer, nullable=False, default=10, server_default="10") # FC-3c downloader knobs download_rate_limit_seconds: Mapped[float] = mapped_column( - Float, nullable=False, default=3.0 + Float, nullable=False, default=3.0, + server_default="3", ) download_validate_files: Mapped[bool] = mapped_column( - Boolean, nullable=False, default=True + Boolean, nullable=False, default=True, + server_default="true", ) # FC-3d scheduling knobs download_schedule_default_seconds: Mapped[int] = mapped_column( - Integer, nullable=False, default=28800 + Integer, nullable=False, default=28800, + server_default="28800", ) download_event_retention_days: Mapped[int] = mapped_column( - Integer, nullable=False, default=90 + Integer, nullable=False, default=90, + server_default="90", ) download_failure_warning_threshold: Mapped[int] = mapped_column( - Integer, nullable=False, default=5 + Integer, nullable=False, default=5, + server_default="5", ) # FC-3h backup knobs. backup_db_nightly_enabled: Mapped[bool] = mapped_column( Boolean, nullable=False, default=False, + server_default="false", ) backup_db_nightly_hour_utc: Mapped[int] = mapped_column( Integer, nullable=False, default=3, + server_default="3", ) backup_db_keep_last_n: Mapped[int] = mapped_column( Integer, nullable=False, default=14, + server_default="14", ) backup_images_keep_last_n: Mapped[int] = mapped_column( Integer, nullable=False, default=3, + server_default="3", ) # FC-6.3 series continuation matcher. enabled gates the rescan; threshold is # the weighted-score cut-off (0..1) above which a pending suggestion is made. series_suggest_enabled: Mapped[bool] = mapped_column( Boolean, nullable=False, default=True, + server_default="true", ) series_suggest_threshold: Mapped[float] = mapped_column( Float, nullable=False, default=0.5, + server_default="0.5", ) # #830 off-platform file-host downloads — per-host enable lever (default on, @@ -113,7 +137,9 @@ class ImportSettings(Base): # English (e.g. "… WIP Part 1") as a European language at ~0.86. CJK stays # trusted regardless (script-detected). Per-post overrides handle the misses. translation_min_confidence: Mapped[float] = mapped_column( - Float, nullable=False, default=0.9, server_default="0.9", + # text() because alembic 0084 used sa.text(); see ml_settings for why + # the form matters and why it is per-column (#3275). + Float, nullable=False, default=0.9, server_default=text("0.9"), ) # Title-based WIP auto-tagging (task #1458). When a freshly-imported post's diff --git a/backend/app/models/import_task.py b/backend/app/models/import_task.py index 3c947c1..e7921e6 100644 --- a/backend/app/models/import_task.py +++ b/backend/app/models/import_task.py @@ -13,10 +13,12 @@ from sqlalchemy import ( Boolean, DateTime, ForeignKey, + Index, Integer, String, Text, func, + text, ) from sqlalchemy.orm import Mapped, mapped_column, relationship @@ -26,6 +28,12 @@ from .base import Base class ImportTask(Base): __tablename__ = "import_task" + + __table_args__ = ( + Index("ix_import_task_created_at_desc", text("created_at DESC")), + # Unindexed FK (#3300). + Index("ix_import_task_result_image_id", "result_image_id"), + ) id: Mapped[int] = mapped_column(Integer, primary_key=True) batch_id: Mapped[int] = mapped_column( ForeignKey("import_batch.id", ondelete="CASCADE"), nullable=False, index=True @@ -33,14 +41,14 @@ class ImportTask(Base): source_path: Mapped[str] = mapped_column(Text, nullable=False) task_type: Mapped[str] = mapped_column(String(16), nullable=False) # media|archive - status: Mapped[str] = mapped_column(String(16), nullable=False, default="pending", index=True) + status: Mapped[str] = mapped_column(String(16), nullable=False, default="pending", index=True, server_default="pending") # Poison-pill circuit breaker (alembic 0026). recovery_count tracks # how many times the stuck-task sweep has re-queued this row; after # the cap it's failed with a diagnostic instead of looping. refetched # bounds the one-shot re-download remediation to a single attempt. - recovery_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0) - refetched: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False) + recovery_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") + refetched: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, server_default="false") result_image_id: Mapped[int | None] = mapped_column( ForeignKey("image_record.id", ondelete="SET NULL"), nullable=True diff --git a/backend/app/models/library_audit_run.py b/backend/app/models/library_audit_run.py index a2d4bc2..6ac9bb2 100644 --- a/backend/app/models/library_audit_run.py +++ b/backend/app/models/library_audit_run.py @@ -8,7 +8,7 @@ reads it and routes through cleanup_service.delete_images. from datetime import datetime from typing import Any -from sqlalchemy import DateTime, Integer, String, Text, func +from sqlalchemy import DateTime, Integer, String, Text, func, text from sqlalchemy.dialects.postgresql import JSONB from sqlalchemy.orm import Mapped, mapped_column @@ -23,6 +23,7 @@ class LibraryAuditRun(Base): params: Mapped[dict[str, Any]] = mapped_column(JSONB, nullable=False) status: Mapped[str] = mapped_column( String(16), nullable=False, default="running", index=True, + server_default="running", ) # running | ready | applied | cancelled | error started_at: Mapped[datetime] = mapped_column( @@ -31,14 +32,16 @@ class LibraryAuditRun(Base): finished_at: Mapped[datetime | None] = mapped_column( DateTime(timezone=True), nullable=True, ) - scanned_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0) - matched_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0) - matched_ids: Mapped[list[int]] = mapped_column(JSONB, nullable=False, default=list) + scanned_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") + matched_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") + matched_ids: Mapped[list[int]] = mapped_column( + JSONB, nullable=False, default=list, server_default=text("'[]'::jsonb") + ) error: Mapped[str | None] = mapped_column(Text, nullable=True) # Chunked-scan state (alembic 0039): keyset cursor the next chunk resumes # from, and the last time a chunk made progress (so the recovery sweep can # tell a progressing multi-chunk audit from a stuck one). - resume_after_id: Mapped[int] = mapped_column(Integer, nullable=False, default=0) + resume_after_id: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") last_progress_at: Mapped[datetime | None] = mapped_column( DateTime(timezone=True), nullable=True, ) diff --git a/backend/app/models/ml_settings.py b/backend/app/models/ml_settings.py index 72da17b..4705d1a 100644 --- a/backend/app/models/ml_settings.py +++ b/backend/app/models/ml_settings.py @@ -11,6 +11,7 @@ from sqlalchemy import ( String, func, select, + text, ) from sqlalchemy.orm import Mapped, mapped_column @@ -20,7 +21,10 @@ from .base import Base class MLSettings(Base): __tablename__ = "ml_settings" # Bare name — Base.metadata's naming convention prepends ck_
_, - # producing the final ck_ml_settings_singleton (matches migration 0003). + # producing ck_ml_settings_singleton. The chain shipped the DOUBLED + # ck_ml_settings_ck_ml_settings_singleton, because the migration + # pre-prefixed the name and the convention prefixed it again; alembic + # 0088 renames it to what this line has always produced (#3275). __table_args__ = (CheckConstraint("id = 1", name="singleton"),) id: Mapped[int] = mapped_column(Integer, primary_key=True) @@ -31,17 +35,20 @@ class MLSettings(Base): # queueing embed work nothing will consume (the daily GPU 'embed' backfill # covers those images instead). cpu_embed_enabled: Mapped[bool] = mapped_column( - Boolean, nullable=False, default=True + Boolean, nullable=False, default=True, + server_default="true", ) # Video embedding (#747). Sample one frame every N seconds (fixed CADENCE, not # a fixed count) so coverage reflects real screen time regardless of length; # cap the total so a long video can't explode into hundreds of embeds. The # per-frame SigLIP embeddings are mean-pooled. Operator-tunable. video_frame_interval_seconds: Mapped[float] = mapped_column( - Float, nullable=False, default=4.0 + Float, nullable=False, default=4.0, + server_default="4", ) video_max_frames: Mapped[int] = mapped_column( - Integer, nullable=False, default=64 + Integer, nullable=False, default=64, + server_default="64", ) # Tagging-v2 head training (#114). The head is the suggestion source that # LEARNS from the operator's tags (replacing Camie + centroid). A concept @@ -49,10 +56,12 @@ class MLSettings(Base): # head_auto_apply_precision is the precision bar a head must clear (at some # operating point) to "graduate" into earned auto-apply. Operator-tunable. head_min_positives: Mapped[int] = mapped_column( - Integer, nullable=False, default=8 + Integer, nullable=False, default=8, + server_default="8", ) head_auto_apply_precision: Mapped[float] = mapped_column( - Float, nullable=False, default=0.97 + Float, nullable=False, default=0.97, + server_default="0.97", ) # Earned auto-apply (#114). A graduated head fires (tags images without a # human) when this master switch is on AND the head has at least @@ -61,29 +70,34 @@ class MLSettings(Base): # default (operator-asked 2026-06-29: opt-OUT, not opt-in); the support + # measured-precision gates keep it safe, and every auto-tag is reversible. head_auto_apply_enabled: Mapped[bool] = mapped_column( - Boolean, nullable=False, default=True + Boolean, nullable=False, default=True, + server_default="true", ) head_auto_apply_min_positives: Mapped[int] = mapped_column( # Support floor raised 30→50 (operator-asked 2026-07-06): a head needs # more human labels before it may fire without a human. - Integer, nullable=False, default=50 + Integer, nullable=False, default=50, + server_default="30", ) # CCIP character-match cosine cut (#114). 0.85 default — the v1 flat 0.75 # over-fired (high-reference characters matched a scatter of images); 0.85 # keeps the confident single-character matches. Tunable from the agent card. ccip_match_threshold: Mapped[float] = mapped_column( - Float, nullable=False, default=0.85 + Float, nullable=False, default=0.85, + server_default="0.85", ) # CCIP auto-apply (#114). Confident matches (>= ccip_auto_apply_threshold, # above the suggest cut) auto-tag on a daily sweep. ON by default (opt-out); # single-character references + the high bar keep it safe, every tag reversible. ccip_auto_apply_enabled: Mapped[bool] = mapped_column( - Boolean, nullable=False, default=True + Boolean, nullable=False, default=True, + server_default="true", ) ccip_auto_apply_threshold: Mapped[float] = mapped_column( # Raised 0.92→0.95 (operator-asked 2026-07-06) so only very confident # character matches auto-tag. - Float, nullable=False, default=0.95 + Float, nullable=False, default=0.95, + server_default="0.92", ) # -- Presentation chrome auto-hide (#141) ------------------------------- # `banner` (chrome — clusters on UI, not content) auto-applies on the sweep @@ -95,13 +109,21 @@ class MLSettings(Base): # (opt-out); every auto-tag is reversible. NOTE (#1464): `wip` + `editor # screenshot` are no longer chrome — they went to the PROCESS path below. presentation_auto_apply_enabled: Mapped[bool] = mapped_column( - Boolean, nullable=False, default=True + Boolean, nullable=False, default=True, + server_default="true", ) presentation_auto_apply_threshold: Mapped[float] = mapped_column( - Float, nullable=False, default=0.90 + Float, nullable=False, default=0.90, + # text(), not a string, because alembic 0082 used sa.text(): a bare + # string renders DEFAULT '0.90'::double precision while text() renders + # DEFAULT 0.90, and the chain is MIXED — some migrations used one, + # some the other. Same value, different stored expression, so each + # column here mirrors whichever form its own migration used (#3275). + server_default=text("0.90"), ) presentation_conflict_threshold: Mapped[float] = mapped_column( - Float, nullable=False, default=0.50 + Float, nullable=False, default=0.50, + server_default=text("0.50"), ) # -- Process auto-apply (#1464) ---------------------------------------- # `wip` / `editor screenshot` are PROCESS art — unfinished pieces + program @@ -115,24 +137,29 @@ class MLSettings(Base): # (PresentationReview, mode='process') rather than silently marked. OFF by # default — a new whole-library auto-tagger is opt-in; every auto-tag reversible. process_auto_apply_enabled: Mapped[bool] = mapped_column( - Boolean, nullable=False, default=False + Boolean, nullable=False, default=False, + server_default="false", ) process_auto_apply_threshold: Mapped[float] = mapped_column( - Float, nullable=False, default=0.90 + Float, nullable=False, default=0.90, + server_default="0.90", ) process_conflict_threshold: Mapped[float] = mapped_column( - Float, nullable=False, default=0.50 + Float, nullable=False, default=0.50, + server_default="0.50", ) # Default = SigLIP 2 (so400m, 512px) for new installs (migration 0069); # existing libraries keep their stored value until the operator re-embeds. embedder_model_version: Mapped[str] = mapped_column( - String(128), nullable=False, default="siglip2-so400m-patch16-512" + String(128), nullable=False, default="siglip2-so400m-patch16-512", + server_default="siglip2-so400m-patch16-512", ) # The HF model NAME the embedder loads (server CPU embed + announced to the # GPU agent in the lease). Operator-settable so the embedder is a choice, not # a hardcode (#1190): set name + version together, then re-embed + retrain. embedder_model_name: Mapped[str] = mapped_column( - String(128), nullable=False, default="google/siglip2-so400m-patch16-512" + String(128), nullable=False, default="google/siglip2-so400m-patch16-512", + server_default="google/siglip2-so400m-patch16-512", ) # -- Crop proposers / detectors (#1202, #134) -------------------------- # WHERE-to-crop YOLO detectors feeding the crop→SigLIP bag + CCIP. Config @@ -145,20 +172,24 @@ class MLSettings(Base): # person: general COCO figure detector for Western/realistic art the anime # person-detector misses → NMS-merged with imgutils → CCIP + concept. detector_person_enabled: Mapped[bool] = mapped_column( - Boolean, nullable=False, default=True + Boolean, nullable=False, default=True, + server_default="true", ) detector_person_weights: Mapped[str] = mapped_column( - String(512), nullable=False, default="yolo11n.pt" + String(512), nullable=False, default="yolo11n.pt", + server_default="yolo11n.pt", ) detector_person_conf: Mapped[float] = mapped_column( - Float, nullable=False, default=0.35 + Float, nullable=False, default=0.35, + server_default=text("0.35"), ) # anatomy: booru_yolo anime/furry/NSFW torso components → concept crops. # Default = yolov11m_aa22 (26 classes, best mAP50-95 0.96), committed in the # upstream repo so the URL resolves. License UNSTATED — fine for a private # homelab (operator accepted #1202). detector_anatomy_enabled: Mapped[bool] = mapped_column( - Boolean, nullable=False, default=True + Boolean, nullable=False, default=True, + server_default="true", ) detector_anatomy_weights: Mapped[str] = mapped_column( String(512), nullable=False, @@ -166,37 +197,47 @@ class MLSettings(Base): "https://github.com/aperveyev/booru_yolo/raw/main/models/" "yolov11m_aa22.pt" ), + server_default="https://github.com/aperveyev/booru_yolo/raw/main/models/yolov11m_aa22.pt", ) detector_anatomy_conf: Mapped[float] = mapped_column( - Float, nullable=False, default=0.30 + Float, nullable=False, default=0.30, + server_default=text("0.30"), ) # panel: comic page → panel regions → concept crops (Apache-2.0, YOLOv12x). detector_panel_enabled: Mapped[bool] = mapped_column( - Boolean, nullable=False, default=True + Boolean, nullable=False, default=True, + server_default="true", ) detector_panel_weights: Mapped[str] = mapped_column( String(512), nullable=False, default="mosesb/best-comic-panel-detection::best.pt", + server_default="mosesb/best-comic-panel-detection::best.pt", ) detector_panel_conf: Mapped[float] = mapped_column( - Float, nullable=False, default=0.30 + Float, nullable=False, default=0.30, + server_default=text("0.30"), ) # Per-frame caps bound the crop→embed explosion; max_regions is the hard # per-job backstop; dedupe_iou drops near-duplicate crops before the embed. detector_max_figures: Mapped[int] = mapped_column( - Integer, nullable=False, default=8 + Integer, nullable=False, default=8, + server_default="8", ) detector_max_components: Mapped[int] = mapped_column( - Integer, nullable=False, default=8 + Integer, nullable=False, default=8, + server_default="8", ) detector_max_panels: Mapped[int] = mapped_column( - Integer, nullable=False, default=8 + Integer, nullable=False, default=8, + server_default="8", ) detector_max_regions: Mapped[int] = mapped_column( - Integer, nullable=False, default=128 + Integer, nullable=False, default=128, + server_default="128", ) detector_dedupe_iou: Mapped[float] = mapped_column( - Float, nullable=False, default=0.85 + Float, nullable=False, default=0.85, + server_default=text("0.85"), ) # -- CCIP character prototypes (#1317) --------------------------------- # The per-character reference set is precomputed + refreshed INCREMENTALLY @@ -208,7 +249,8 @@ class MLSettings(Base): String(128), nullable=True ) ccip_prototype_cap: Mapped[int] = mapped_column( - Integer, nullable=False, default=64 + Integer, nullable=False, default=64, + server_default="64", ) updated_at: Mapped[datetime] = mapped_column( DateTime(timezone=True), nullable=False, server_default=func.now() diff --git a/backend/app/models/patreon_failed_media.py b/backend/app/models/patreon_failed_media.py index 79976ef..26557fe 100644 --- a/backend/app/models/patreon_failed_media.py +++ b/backend/app/models/patreon_failed_media.py @@ -35,7 +35,7 @@ class PatreonFailedMedia(Base): ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True ) filehash: Mapped[str] = mapped_column(String(128), nullable=False) - attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1) + attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1, server_default="1") last_error: Mapped[str | None] = mapped_column(Text, nullable=True) first_failed_at: Mapped[datetime] = mapped_column( DateTime(timezone=True), nullable=False, server_default=func.now() diff --git a/backend/app/models/pixiv_failed_media.py b/backend/app/models/pixiv_failed_media.py index a33d15e..7737594 100644 --- a/backend/app/models/pixiv_failed_media.py +++ b/backend/app/models/pixiv_failed_media.py @@ -35,7 +35,7 @@ class PixivFailedMedia(Base): ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True ) filehash: Mapped[str] = mapped_column(String(128), nullable=False) - attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1) + attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1, server_default="1") last_error: Mapped[str | None] = mapped_column(Text, nullable=True) first_failed_at: Mapped[datetime] = mapped_column( DateTime(timezone=True), nullable=False, server_default=func.now() diff --git a/backend/app/models/post.py b/backend/app/models/post.py index 4183330..e3f8e5e 100644 --- a/backend/app/models/post.py +++ b/backend/app/models/post.py @@ -13,11 +13,13 @@ from sqlalchemy import ( CheckConstraint, DateTime, ForeignKey, + Index, Integer, String, Text, UniqueConstraint, func, + text, ) from sqlalchemy.orm import Mapped, mapped_column @@ -27,6 +29,10 @@ from .base import Base class Post(Base): __tablename__ = "post" __table_args__ = ( + # alembic 0030. The comment above described this index; nothing declared + # it, so autogenerate proposed dropping it (#3275). + Index("uq_post_artist_external_id_null_source", "artist_id", "external_post_id", + unique=True, postgresql_where=text("source_id IS NULL")), # Source-bound dedup. Postgres treats NULL != NULL so rows # with source_id IS NULL aren't deduped by this constraint; # the partial unique index `uq_post_artist_external_id_null_source` @@ -35,7 +41,11 @@ class Post(Base): UniqueConstraint("source_id", "external_post_id", name="uq_post_source_external_id"), CheckConstraint( "translation_override IN ('auto', 'force', 'original')", - name="ck_post_translation_override", + # Bare name: Base.metadata's naming convention prepends + # ck_
_. Pre-prefixing it here doubles the prefix — see + # alembic 0088, which renames the four constraints that shipped + # that way (#3275). + name="translation_override", ), ) diff --git a/backend/app/models/presentation_review.py b/backend/app/models/presentation_review.py index 73da13f..3e83b21 100644 --- a/backend/app/models/presentation_review.py +++ b/backend/app/models/presentation_review.py @@ -11,7 +11,7 @@ are pruned by retention. from datetime import datetime -from sqlalchemy import DateTime, Float, ForeignKey, String, func +from sqlalchemy import DateTime, Float, ForeignKey, Index, String, func from sqlalchemy.orm import Mapped, mapped_column from .base import Base @@ -20,6 +20,14 @@ from .base import Base class PresentationReview(Base): __tablename__ = "presentation_review" + + __table_args__ = ( + Index("ix_presentation_review_resolved_at", "resolved_at"), + # Both FKs to tag were unindexed (#3300); tag_id CASCADEs, so a tag + # delete had to scan this table to find its rows. + Index("ix_presentation_review_tag_id", "tag_id"), + Index("ix_presentation_review_conflict_tag_id", "conflict_tag_id"), + ) image_record_id: Mapped[int] = mapped_column( ForeignKey("image_record.id", ondelete="CASCADE"), primary_key=True ) diff --git a/backend/app/models/series_chapter.py b/backend/app/models/series_chapter.py index ff7698a..87fc5c2 100644 --- a/backend/app/models/series_chapter.py +++ b/backend/app/models/series_chapter.py @@ -16,7 +16,14 @@ title is the optional chapter name; stated_part is the optional operator-facing from datetime import datetime -from sqlalchemy import DateTime, ForeignKey, Integer, Text, func +from sqlalchemy import ( + DateTime, + ForeignKey, + Integer, + Text, + UniqueConstraint, + func, +) from sqlalchemy.orm import Mapped, mapped_column from .base import Base @@ -25,14 +32,26 @@ from .base import Base class SeriesChapter(Base): __tablename__ = "series_chapter" + __table_args__ = ( + # alembic 0047 named the UNIQUE `uq_series_chapter_anchor_page`, not + # the `uq_series_chapter_anchor_page_id` a bare `unique=True` would + # render (#3275). + UniqueConstraint("anchor_page_id", name="uq_series_chapter_anchor_page"), + ) + id: Mapped[int] = mapped_column(Integer, primary_key=True) series_tag_id: Mapped[int] = mapped_column( ForeignKey("tag.id", ondelete="CASCADE"), nullable=False, index=True ) + # Both the UNIQUE (above) and the FK carry the names 0047 gave them; the + # convention would render the FK `fk_series_chapter_anchor_page_id_series_page`. anchor_page_id: Mapped[int] = mapped_column( - ForeignKey("series_page.id", ondelete="CASCADE"), + ForeignKey( + "series_page.id", + ondelete="CASCADE", + name="fk_series_chapter_anchor_page", + ), nullable=False, - unique=True, ) title: Mapped[str | None] = mapped_column(Text, nullable=True) stated_part: Mapped[int | None] = mapped_column(Integer, nullable=True) diff --git a/backend/app/models/series_page.py b/backend/app/models/series_page.py index 0bb0b40..22e19e1 100644 --- a/backend/app/models/series_page.py +++ b/backend/app/models/series_page.py @@ -14,7 +14,14 @@ number parsed from the source post, nullable when unknown. from datetime import datetime -from sqlalchemy import DateTime, ForeignKey, Integer, String, func +from sqlalchemy import ( + DateTime, + ForeignKey, + Integer, + String, + UniqueConstraint, + func, +) from sqlalchemy.orm import Mapped, mapped_column from .base import Base @@ -23,14 +30,22 @@ from .base import Base class SeriesPage(Base): __tablename__ = "series_page" + __table_args__ = ( + # alembic 0005 named this `uq_series_page_image`; a bare `unique=True` + # on the column renders `uq_series_page_image_id` under the naming + # convention, which is a different object from the one the database + # has (#3275). + UniqueConstraint("image_id", name="uq_series_page_image"), + ) + id: Mapped[int] = mapped_column(Integer, primary_key=True) series_tag_id: Mapped[int] = mapped_column( ForeignKey("tag.id", ondelete="CASCADE"), nullable=False, index=True ) + # UNIQUE lives in __table_args__ above, under the name 0005 gave it. image_id: Mapped[int] = mapped_column( ForeignKey("image_record.id", ondelete="CASCADE"), nullable=False, - unique=True, ) # 'placed' = in the series-global run (page_number set); 'pending' = staged # from a post awaiting the operator's sort (page_number NULL). (#789 P2) diff --git a/backend/app/models/source.py b/backend/app/models/source.py index 1bf6c67..3501505 100644 --- a/backend/app/models/source.py +++ b/backend/app/models/source.py @@ -5,7 +5,16 @@ Multiple sources per artist support creators with cross-platform presence. from datetime import datetime -from sqlalchemy import JSON, Boolean, DateTime, ForeignKey, Integer, String, Text +from sqlalchemy import ( + JSON, + Boolean, + DateTime, + ForeignKey, + Integer, + String, + Text, + UniqueConstraint, +) from sqlalchemy.orm import Mapped, mapped_column, relationship from .base import Base @@ -14,13 +23,27 @@ from .base import Base class Source(Base): __tablename__ = "source" + __table_args__ = ( + # alembic 0010. One row per (artist, platform, url): re-adding a source + # the artist already has is an update, not a second row. The model had + # never declared it (#3275), so autogenerate would have proposed + # DROPPING it — the guarantee existed only in the migration chain. + # + # Named explicitly because the naming convention would render this + # `uq_source_artist_id` (uq keys off column_0_name), which is both + # wrong about the shape and not what the database actually has. + UniqueConstraint( + "artist_id", "platform", "url", name="uq_source_artist_platform_url" + ), + ) + id: Mapped[int] = mapped_column(Integer, primary_key=True) artist_id: Mapped[int] = mapped_column( ForeignKey("artist.id", ondelete="CASCADE"), nullable=False, index=True ) platform: Mapped[str] = mapped_column(String(64), nullable=False) url: Mapped[str] = mapped_column(Text, nullable=False) - enabled: Mapped[bool] = mapped_column(Boolean, nullable=False, default=True) + enabled: Mapped[bool] = mapped_column(Boolean, nullable=False, default=True, server_default="true") config_overrides: Mapped[dict | None] = mapped_column(JSON, nullable=True) @@ -32,7 +55,7 @@ class Source(Base): # by _update_source_health alongside last_error; cleared on 'ok'. error_type: Mapped[str | None] = mapped_column(String(32), nullable=True, index=True) check_interval_override: Mapped[int | None] = mapped_column(Integer, nullable=True) - consecutive_failures: Mapped[int] = mapped_column(Integer, nullable=False, default=0) + consecutive_failures: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0") # alembic 0031: sticky deep-scan budget. When > 0, the next N download # runs use gallery-dl's full-walk config (skip: True + 1800s timeout); diff --git a/backend/app/models/subscribestar_failed_media.py b/backend/app/models/subscribestar_failed_media.py index 9201aa7..d12ff73 100644 --- a/backend/app/models/subscribestar_failed_media.py +++ b/backend/app/models/subscribestar_failed_media.py @@ -34,7 +34,7 @@ class SubscribeStarFailedMedia(Base): ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True ) filehash: Mapped[str] = mapped_column(String(128), nullable=False) - attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1) + attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1, server_default="1") last_error: Mapped[str | None] = mapped_column(Text, nullable=True) first_failed_at: Mapped[datetime] = mapped_column( DateTime(timezone=True), nullable=False, server_default=func.now() diff --git a/backend/app/models/tag.py b/backend/app/models/tag.py index 8d5a256..1b85936 100644 --- a/backend/app/models/tag.py +++ b/backend/app/models/tag.py @@ -15,11 +15,13 @@ from sqlalchemy import ( Column, DateTime, ForeignKey, + Index, Integer, String, Table, false, func, + text, ) from sqlalchemy import ( Enum as SQLEnum, @@ -67,17 +69,31 @@ image_tag = Table( primary_key=True, ), Column("tag_id", ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True), - Column("source", String(32), nullable=False, default="manual"), + Column("source", String(32), nullable=False, default="manual", server_default="manual"), Column("created_at", DateTime(timezone=True), nullable=False, server_default=func.now()), + # The PK is (image_record_id, tag_id), which leads with the WRONG column + # for the two things that matter most here (#3300): the gallery's tag + # filter (tag_query.py builds `image_tag.c.tag_id == tid`) and the + # ON DELETE CASCADE from tag, which has to find a tag's rows to remove + # them. Without this index both scan the largest table in the schema. + Index("ix_image_tag_tag_id", "tag_id"), ) class Tag(Base): __tablename__ = "tag" __table_args__ = ( + # alembic 0002. An EXPRESSION index — COALESCE cannot be expressed as a + # UniqueConstraint, which is why it only ever existed in a migration (#3275). + Index("uq_tag_name_kind_fandom", "name", "kind", text("COALESCE(fandom_id, 0)"), + unique=True), CheckConstraint( "(fandom_id IS NULL) OR (kind = 'character')", - name="ck_tag_fandom_requires_character", + # Bare name: Base.metadata's naming convention prepends + # ck_
_. Pre-prefixing it here doubles the prefix — see + # alembic 0088, which renames the four constraints that shipped + # that way (#3275). + name="fandom_requires_character", ), ) @@ -87,6 +103,7 @@ class Tag(Base): SQLEnum(TagKind, name="tag_kind", values_callable=lambda e: [m.value for m in e]), nullable=False, default=TagKind.general, + server_default="general", ) fandom_id: Mapped[int | None] = mapped_column( ForeignKey("tag.id", ondelete="SET NULL"), nullable=True, index=True diff --git a/backend/app/models/tag_alias.py b/backend/app/models/tag_alias.py index 93f4755..533cec3 100644 --- a/backend/app/models/tag_alias.py +++ b/backend/app/models/tag_alias.py @@ -5,7 +5,7 @@ in image_prediction stay unmolested. from datetime import datetime -from sqlalchemy import DateTime, ForeignKey, String, func +from sqlalchemy import DateTime, ForeignKey, Index, String, func from sqlalchemy.orm import Mapped, mapped_column from .base import Base @@ -14,10 +14,17 @@ from .base import Base class TagAlias(Base): __tablename__ = "tag_alias" + + __table_args__ = ( + # Named explicitly: the database calls this ix_tag_alias_canonical, while + # a bare index=True on the column would generate ix_tag_alias_canonical_tag_id + # and silently propose a drop+create on the next autogenerate (#3275). + Index("ix_tag_alias_canonical", "canonical_tag_id"), + ) alias_string: Mapped[str] = mapped_column(String(255), primary_key=True) alias_category: Mapped[str] = mapped_column(String(32), primary_key=True) canonical_tag_id: Mapped[int] = mapped_column( - ForeignKey("tag.id", ondelete="CASCADE"), nullable=False, index=True + ForeignKey("tag.id", ondelete="CASCADE"), nullable=False ) created_at: Mapped[datetime] = mapped_column( DateTime(timezone=True), nullable=False, server_default=func.now() diff --git a/backend/app/models/tag_suggestion_rejection.py b/backend/app/models/tag_suggestion_rejection.py index 8a834fe..341f67f 100644 --- a/backend/app/models/tag_suggestion_rejection.py +++ b/backend/app/models/tag_suggestion_rejection.py @@ -5,7 +5,7 @@ Prevents re-suggestion AND prevents allowlist auto-apply on that image. from datetime import datetime -from sqlalchemy import DateTime, ForeignKey, func +from sqlalchemy import DateTime, ForeignKey, Index, func from sqlalchemy.orm import Mapped, mapped_column from .base import Base @@ -14,11 +14,24 @@ from .base import Base class TagSuggestionRejection(Base): __tablename__ = "tag_suggestion_rejection" + + __table_args__ = ( + # Named explicitly; see tag_alias for why (#3275). + Index("ix_tag_suggestion_rejection_tag", "tag_id"), + ) + # Both FKs named explicitly. alembic 0003 used a hand-shortened `tsr` + # prefix; the convention would render the full table name (#3275). image_record_id: Mapped[int] = mapped_column( - ForeignKey("image_record.id", ondelete="CASCADE"), primary_key=True + ForeignKey( + "image_record.id", + ondelete="CASCADE", + name="fk_tsr_image_record_id_image_record", + ), + primary_key=True, ) tag_id: Mapped[int] = mapped_column( - ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True, index=True + ForeignKey("tag.id", ondelete="CASCADE", name="fk_tsr_tag_id_tag"), + primary_key=True, ) rejected_at: Mapped[datetime] = mapped_column( DateTime(timezone=True), nullable=False, server_default=func.now() diff --git a/backend/app/models/task_run.py b/backend/app/models/task_run.py index e8eac05..2cd8b70 100644 --- a/backend/app/models/task_run.py +++ b/backend/app/models/task_run.py @@ -15,7 +15,7 @@ backend.app.tasks.maintenance.recover_stalled_task_runs (Beat 5 min). from datetime import datetime -from sqlalchemy import DateTime, Integer, String, Text +from sqlalchemy import DateTime, Index, Integer, String, Text, text from sqlalchemy.orm import Mapped, mapped_column from .base import Base @@ -24,12 +24,21 @@ from .base import Base class TaskRun(Base): __tablename__ = "task_run" + + __table_args__ = ( + # alembic 0016: the three task-history indexes (#3275). + Index("ix_task_run_name_started", "task_name", text("started_at DESC")), + Index("ix_task_run_queue_started", "queue", text("started_at DESC")), + Index("ix_task_run_status_started", "status", text("started_at DESC")), + ) id: Mapped[int] = mapped_column(Integer, primary_key=True) celery_task_id: Mapped[str] = mapped_column( String(64), nullable=False, index=True, ) - queue: Mapped[str] = mapped_column(String(32), nullable=False, index=True) - task_name: Mapped[str] = mapped_column(String(128), nullable=False, index=True) + # Neither carries index=True: ix_task_run_queue_started and + # ix_task_run_name_started already lead with these columns (#3301). + queue: Mapped[str] = mapped_column(String(32), nullable=False) + task_name: Mapped[str] = mapped_column(String(128), nullable=False) target_id: Mapped[int | None] = mapped_column(Integer, nullable=True) started_at: Mapped[datetime] = mapped_column( DateTime(timezone=True), nullable=False, index=True, @@ -39,7 +48,9 @@ class TaskRun(Base): ) duration_ms: Mapped[int | None] = mapped_column(Integer, nullable=True) status: Mapped[str] = mapped_column( - String(16), nullable=False, default="running", index=True, + # No index=True — ix_task_run_status_started leads with `status`. + String(16), nullable=False, default="running", + server_default="running", ) error_type: Mapped[str | None] = mapped_column(String(128), nullable=True) error_message: Mapped[str | None] = mapped_column(Text, nullable=True) diff --git a/ci-requirements.md b/ci-requirements.md index 340e277..ccd3081 100644 --- a/ci-requirements.md +++ b/ci-requirements.md @@ -167,6 +167,48 @@ per `docs/process.md`'s "add deps to the image when used by >1 project". `github.event.inputs` into an env var rather than interpolated into a run block, and it is checked inside the reuse step so that one decision drives both the build and the repoint. +- **A weekly `schedule` rebuilds all three images against fresh base layers** + (Sunday 06:00 UTC, milestone 326 step 4, #3154). Skip-if-exists is keyed on + OUR source, so an artifact whose source stops moving stops picking up base + updates — `agent/` has not changed since 2026-07-17 and would otherwise serve + that day's `nvidia/cuda` layers forever. Four things make it work: + - It **builds `main`, not the branch that triggered it.** Forgejo registers a + cron from the DEFAULT branch (`dev` here), so a scheduled run arrives with + `github.ref` on dev. The ref is decided once in a top-level `env: + BUILD_REF` that every checkout in the file takes, rather than per job — + otherwise `sign-extension` would derive dev's extension version while + `build-web` bundled main's, and the release download would 404 on a version + that exists perfectly well. Every job then ASSERTS its checkout is `main` + before doing anything, because `env` inside `with:` is not a context this + runner is known to evaluate — if it silently resolved to empty, checkout + would fall back to the triggering ref and the refresh would publish dev's + source to `:latest` with every lane green. + - It **publishes only `:latest`.** `:c-` for main's HEAD already names + the bytes that commit built; re-pushing it over refreshed layers would + break the one tag rule 145 makes immutable, and it is the rollback unit. + The repoint step needs no schedule case for this — the tag list is the + channel tag alone, so SOURCE is the only entry, it is excluded as always, + and the step correctly does nothing. + - **`:latest` and `:c-` therefore diverge between a refresh and the next + `main` push, by design.** They re-converge on that push: it hits reuse (a + refresh does not move `fc.revision`, because it does not touch the source), + and the repoint writes the NEW `:c-` from the refreshed `:latest`. The + push path needed no change for this, because the repoint already excluded + the source tag — the same rule that keeps the label readable also keeps a + refresh from being undone. + - **`pull: true` on the scheduled path only** is the mechanism: a moved base + tag changes the `FROM` layer's cache key and everything above it rebuilds. + **It does not currently make the unmoved case free.** Measured on the first + real fire (run 4934, 2026-08-30): every content step reported `CACHED` and + the bases resolved to unchanged digests, yet all three `:latest` tags got a + NEW manifest digest, because buildkit mints a fresh image config per run and + republishes identical layers under it. So `:latest` is rewritten weekly + whether or not anything changed, and `:c-` is handed a new manifest to + diverge from on the same cadence — a digest change stops meaning anything. + Tracked as #3265; the likely fix is a deterministic `SOURCE_DATE_EPOCH`. + Separately not caught: a Debian package update inside the `apt-get install` + layer while the base tag stands still — a lag rather than a hole, since the + official python/cuda images rebuild with those updates baked in. - **`FC_CHANNEL` and `FC_VERSION` are build args, not runtime settings.** `build.yml` passes them to the web image only — the ml and agent images have nothing to report them to. `/api/health` returns both, the foot of Settings