Compare commits
33
Commits
2529b516e6
...
dev
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ad8392b790 | ||
|
|
5084ba666b | ||
|
|
fe4e0f2b71 | ||
|
|
dc8af8b1a7 | ||
|
|
131237143b | ||
|
|
59d27ef76e | ||
|
|
f630e50e75 | ||
|
|
86abaf0b94 | ||
|
|
4815040d74 | ||
|
|
81b7b6f308 | ||
|
|
bfa9fd678b | ||
|
|
24a2b70a5a | ||
|
|
2c88ad3efb | ||
|
|
bfc4f9cec9 | ||
|
|
b590d25f8f | ||
|
|
635138b0d1 | ||
|
|
c0370069e0 | ||
|
|
3590c478f5 | ||
|
|
8a4af589f1 | ||
|
|
aa71cbbdbf | ||
|
|
973db73221 | ||
|
|
bc4eba636d | ||
|
|
dbc4e8b0c6 | ||
|
|
08418d54a3 | ||
|
|
b1bd2531ad | ||
|
|
389afe2f7b | ||
|
|
b979062dd7 | ||
|
|
573228b9da | ||
|
|
d044e93bdb | ||
|
|
ed2b1adc2e | ||
|
|
5e1996e77f | ||
|
|
98b56330d0 | ||
|
|
6959e1220c |
+97
-18
@@ -1,24 +1,103 @@
|
||||
# Database
|
||||
DB_USER=fabledcurator
|
||||
DB_PASSWORD=changeme_use_a_real_password
|
||||
DB_HOST=postgres
|
||||
DB_PORT=5432
|
||||
DB_NAME=fabledcurator
|
||||
# FabledCurator configuration.
|
||||
#
|
||||
# Copy to `.env` and edit before your first production start:
|
||||
#
|
||||
# cp .env.example .env
|
||||
#
|
||||
# Only the two values under CHANGE THESE actually need your attention. The
|
||||
# rest have working defaults baked into docker-compose.yml and are listed
|
||||
# here so you know they exist, not because you have to set them.
|
||||
#
|
||||
# Almost nothing else lives here on purpose. FabledCurator is configured from
|
||||
# its own Settings UI, backed by the database — no restart, no YAML. If you
|
||||
# are looking for where to set an import path, a download schedule or an ML
|
||||
# threshold, it is in the app, not in this file.
|
||||
|
||||
# Redis / Celery
|
||||
CELERY_BROKER_URL=redis://redis:6379/0
|
||||
CELERY_RESULT_BACKEND=redis://redis:6379/0
|
||||
|
||||
# App
|
||||
# Generate with: openssl rand -hex 32
|
||||
SECRET_KEY=changeme_32_byte_hex_secret
|
||||
# ---------------------------------------------------------------------------
|
||||
# CHANGE THESE
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Extension API key — used in FC-3, lands later but reserved now
|
||||
# Generate with: openssl rand -hex 32
|
||||
EXTENSION_API_KEY=
|
||||
# The Postgres password. docker-compose.yml falls back to a published default
|
||||
# (`fabledcurator_dev`) so that `docker compose up` works with no config at
|
||||
# all — which is exactly why you must not leave it at that on a real install.
|
||||
# It is the credential protecting your stored platform session cookies.
|
||||
DB_PASSWORD=
|
||||
|
||||
# Logging
|
||||
# Sets Quart's app.secret_key. Today it signs nothing: FabledCurator has no
|
||||
# login and uses no session cookies, so no value here is protecting anything
|
||||
# right now. Set it anyway. It is required at boot rather than defaulted so
|
||||
# that the day something session-backed does land, no instance is already
|
||||
# running on a value published in this file.
|
||||
#
|
||||
# openssl rand -hex 32
|
||||
SECRET_KEY=
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# FIRST BOOT ONLY — then delete this line
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# FabledCurator encrypts your stored platform credentials with a Fernet key it
|
||||
# keeps at /images/secrets/credential_key.b64 — inside the ./images bind mount,
|
||||
# so it outlives the container. On a brand-new install that file does not exist
|
||||
# yet, and the app REFUSES TO START rather than quietly create one:
|
||||
#
|
||||
# MissingCredentialKey: Fernet key file not found at
|
||||
# /images/secrets/credential_key.b64
|
||||
#
|
||||
# That refusal is deliberate. Auto-creating a key is indistinguishable from the
|
||||
# disaster case — a restore that brought the database back but lost
|
||||
# ./images/secrets — and there it would mint a key that cannot decrypt anything,
|
||||
# leaving an instance that looks healthy while every paywalled download fails.
|
||||
# So the choice is yours to make explicitly, once.
|
||||
#
|
||||
# Set this for your first `up`, watch the container come up, then DELETE THE
|
||||
# LINE. Leaving it set disarms the protection permanently, on an instance that
|
||||
# by then has credentials worth protecting.
|
||||
#
|
||||
# BACK UP ./images/secrets/ ALONGSIDE YOUR DATABASE. The key is the only thing
|
||||
# that can read your stored credentials; a database restored without it needs
|
||||
# every credential re-entered by hand.
|
||||
CURATOR_BOOTSTRAP_NEW_KEY=1
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Optional — defaults are fine
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Host port the UI is published on. The container always listens on 8080;
|
||||
# this is only the left-hand side of the port mapping.
|
||||
PORT=8080
|
||||
|
||||
# DEBUG | INFO | WARNING | ERROR
|
||||
LOG_LEVEL=INFO
|
||||
|
||||
# Deployment posture: plain HTTP (no TLS in the app; reverse proxy if needed)
|
||||
# See docs/superpowers/specs/2026-05-13-fabledcurator-merge-design.md §2.1
|
||||
# Postgres identity. Change these only if you are pointing at a database you
|
||||
# manage yourself — the bundled postgres service is created with whatever is
|
||||
# set here, so changing them after the first start will not rename anything.
|
||||
DB_USER=fabledcurator
|
||||
DB_NAME=fabledcurator
|
||||
|
||||
# Set by docker-compose.yml to reach the bundled services. Override only when
|
||||
# running Postgres or Redis outside this stack.
|
||||
# DB_HOST=postgres
|
||||
# DB_PORT=5432
|
||||
# CELERY_BROKER_URL=redis://redis:6379/0
|
||||
# CELERY_RESULT_BACKEND=redis://redis:6379/0
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# There is no authentication variable here, and that is not an omission
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# FabledCurator has no login, no accounts and no permission model. Anything
|
||||
# that can reach PORT is an administrator and can read the platform session
|
||||
# cookies the app stores for Patreon, SubscribeStar and Pixiv.
|
||||
#
|
||||
# Bind it to a trusted network. See "Before you expose it" in README.md and
|
||||
# the deployment posture section of SECURITY.md.
|
||||
#
|
||||
# The Firefox extension's API key is NOT configured here — it is generated
|
||||
# automatically on first use and shown under Settings → Maintenance, where you
|
||||
# can also rotate it.
|
||||
|
||||
+171
-35
@@ -1,42 +1,50 @@
|
||||
|
||||
# TEMPORARY — milestone 328 steps 1-2. Delete once the baseline is stamped.
|
||||
# TEMPORARY — milestone 328. Delete once the baseline has shipped and settled.
|
||||
#
|
||||
# Squashing 87 alembic revisions into one baseline has exactly one dangerous
|
||||
# failure: the generated baseline does not reproduce the schema the chain
|
||||
# produced, `alembic stamp` writes a version string anyway (it validates
|
||||
# NOTHING), and the divergence surfaces on the next real migration against the
|
||||
# operator's live data.
|
||||
# Collapsing 89 alembic revisions into one baseline has exactly one dangerous
|
||||
# failure: the baseline does not reproduce the schema the chain produced, and
|
||||
# the divergence surfaces later, on the operator's live data, in whatever
|
||||
# migration comes next.
|
||||
#
|
||||
# So this workflow does the comparison in CI, where a pgvector Postgres already
|
||||
# gets built from the chain on every integration run, and nothing is at risk.
|
||||
# It answers one question: does `upgrade head` on the collapsed chain produce a
|
||||
# byte-identical schema to `upgrade head` on the 87-revision chain?
|
||||
# So the comparison happens in CI, against a throwaway pgvector Postgres, where
|
||||
# nothing is at risk. It answers one question: does `upgrade head` on the
|
||||
# collapsed tree produce the same schema as `upgrade head` on the full chain?
|
||||
#
|
||||
# The chain is read from git rather than from the working tree, so this keeps
|
||||
# working AFTER the old revisions are deleted — `chain_ref` names a commit that
|
||||
# still has them. That is what makes this the proof for step 1 and the
|
||||
# pre-flight for step 2, rather than a one-shot script.
|
||||
# The chain is read out of GIT, not the working tree, which is what lets this
|
||||
# keep working now that the revisions are deleted — `chain_ref` names a commit
|
||||
# that still carries 0001..0089. That is the whole reason this is a workflow
|
||||
# rather than a script someone ran once.
|
||||
#
|
||||
# While the chain is still present it also autogenerates a candidate baseline
|
||||
# from the models and prints it. That is a starting point, NOT the answer:
|
||||
# autogenerate reads SQLAlchemy metadata, and three things here do not live
|
||||
# there —
|
||||
# * CREATE EXTENSION vector (0001)
|
||||
# * CREATE EXTENSION tsm_system_rows (0004)
|
||||
# * the HNSW index on image_record.siglip_embedding, which is raw SQL
|
||||
# because alembic's create_index cannot express `USING hnsw (...)` (0036)
|
||||
# plus any CHECK constraint or server_default that a migration added without
|
||||
# the model declaring it. Those must be hand-added, and the diff below is what
|
||||
# proves none were missed.
|
||||
# WHAT THIS CANNOT SEE, and it matters: the comparison is of SCHEMA. Migrations
|
||||
# 0002 and 0003 also INSERTED rows (the import_settings and ml_settings
|
||||
# singletons), and the application reads those with scalar_one(), which raises
|
||||
# on an empty result. A baseline that omitted them would produce an identical
|
||||
# schema, pass this check with a perfect diff, and crash a fresh install on its
|
||||
# first settings access. Only running the app against a new database finds
|
||||
# that class of defect. Do not read a green run here as "the baseline is
|
||||
# correct" — read it as "the schema is correct".
|
||||
#
|
||||
# Autogenerate now emits nearly all of the baseline unaided, which was NOT true
|
||||
# before #3275 put the previously migration-only objects onto the models — the
|
||||
# HNSW index with its opclass, the COALESCE expression index, the partial
|
||||
# unique indexes, 107 server_defaults, the enum CHECKs. An earlier attempt at
|
||||
# this squash was reverted precisely because the generator dropped them all
|
||||
# silently. What still needs hand-adding is only what cannot live in a model:
|
||||
# the two CREATE EXTENSION statements, the two seed rows, and the pgvector
|
||||
# import the generator forgets to write.
|
||||
name: Alembic baseline
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
chain_ref:
|
||||
description: 'Commit/tag that still carries the full 0001..0087 chain'
|
||||
description: 'Commit/tag carrying the full 0001..0089 chain (pinned: the tree no longer has it)'
|
||||
type: string
|
||||
default: '0a5bbe8'
|
||||
default: '725bf15'
|
||||
mode:
|
||||
description: 'chain = compare against this tree''s migrations; models = compare against a schema built from the MODELS'
|
||||
type: string
|
||||
default: 'chain'
|
||||
|
||||
jobs:
|
||||
compare:
|
||||
@@ -79,10 +87,21 @@ jobs:
|
||||
test -n "$PG_IP"
|
||||
echo "PG_CONTAINER=$PG" >> "$GITHUB_ENV"
|
||||
echo "DB_HOST=$PG_IP" >> "$GITHUB_ENV"
|
||||
# Socket probe in python, not bash's /dev/tcp — these steps run under
|
||||
# `sh -e`, where that path does not exist. Same fix and same reasoning
|
||||
# as ci.yml's integration job; see the comment there.
|
||||
pg_ready=""
|
||||
for i in $(seq 1 60); do
|
||||
(echo > "/dev/tcp/$PG_IP/5432") >/dev/null 2>&1 && break
|
||||
if python -c "import socket,sys; s=socket.socket(); s.settimeout(2); sys.exit(0 if s.connect_ex(('$PG_IP', 5432)) == 0 else 1)"; then
|
||||
pg_ready=1
|
||||
break
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
if [ -z "$pg_ready" ]; then
|
||||
echo "postgres at $PG_IP:5432 did not accept a connection within 120s"
|
||||
exit 1
|
||||
fi
|
||||
if command -v uv >/dev/null 2>&1; then
|
||||
uv pip install --system -r requirements.txt
|
||||
else
|
||||
@@ -96,10 +115,15 @@ jobs:
|
||||
- name: Build the schema the OLD chain produces
|
||||
env:
|
||||
CHAIN_REF: ${{ github.event.inputs.chain_ref }}
|
||||
THIS_SHA: ${{ github.sha }}
|
||||
run: |
|
||||
set -eux
|
||||
docker exec "$PG_CONTAINER" createdb -U fabledcurator fc_chain
|
||||
git worktree add /tmp/chain "$CHAIN_REF"
|
||||
# Blank means "the chain in this ref", which is what you want while
|
||||
# the chain is still intact — comparing the models against a PINNED
|
||||
# older commit reports every migration written since as a difference.
|
||||
# Pin it only after the collapse, when the tree no longer has them.
|
||||
git worktree add /tmp/chain "${CHAIN_REF:-$THIS_SHA}"
|
||||
ls /tmp/chain/alembic/versions/*.py | wc -l
|
||||
cd /tmp/chain
|
||||
DB_NAME=fc_chain alembic upgrade head
|
||||
@@ -107,6 +131,23 @@ jobs:
|
||||
docker exec "$PG_CONTAINER" pg_dump -U fabledcurator --schema-only \
|
||||
--no-owner --no-privileges -d fc_chain > chain.sql
|
||||
wc -l chain.sql
|
||||
# Emit the dump itself, checksummed, for local analysis. Reconciling
|
||||
# the models against the deployed schema (#3275) needs the ACTUAL
|
||||
# schema, not an inference from a diff — parsing table context out of
|
||||
# unified-diff hunks drops every table whose CREATE TABLE line falls
|
||||
# outside a hunk, which silently under-reports.
|
||||
#
|
||||
# base64 + sha256 for the same reason as the candidate: a plain cat
|
||||
# of a file this size was truncated mid-line by the runner with the
|
||||
# step still green (run 4964).
|
||||
set +x
|
||||
B64=$(base64 -w 120 chain.sql)
|
||||
echo "===== BEGIN CHAIN SCHEMA (base64) ====="
|
||||
echo "$B64"
|
||||
echo "===== END CHAIN SCHEMA ====="
|
||||
echo "chain-sha256: $(sha256sum chain.sql | cut -d' ' -f1)"
|
||||
echo "chain-bytes: $(wc -c < chain.sql)"
|
||||
set -x
|
||||
|
||||
# A candidate baseline, autogenerated from the models against an EMPTY
|
||||
# database so every table shows up as a create. Printed for a human to
|
||||
@@ -157,20 +198,53 @@ jobs:
|
||||
echo "candidate-bytes: $(wc -c < "$F")"
|
||||
echo "candidate-b64-lines: $(echo "$B64" | wc -l)"
|
||||
set -x
|
||||
mkdir -p /tmp/candidate
|
||||
cp alembic/versions/*.py /tmp/candidate/
|
||||
# Put the tree back exactly as it was; this job never mutates state.
|
||||
rm -f alembic/versions/*.py
|
||||
mv /tmp/versions_held/*.py alembic/versions/ 2>/dev/null || true
|
||||
|
||||
# DB 2: whatever the CURRENT tree's alembic/versions produces. Before the
|
||||
# squash that is the same 87 revisions and the diff is trivially clean —
|
||||
# which is worth running once as a control, so a clean diff after the
|
||||
# squash means something.
|
||||
# DB 2: what the CURRENT tree produces.
|
||||
#
|
||||
# `mode: models` applies the candidate autogenerated from the MODELS
|
||||
# instead, which is what answers "do the models describe the schema?" —
|
||||
# the question #3275 exists because nobody had ever asked it. Under that
|
||||
# mode a clean diff means autogenerate is trustworthy again.
|
||||
#
|
||||
# The two extensions are created by hand first. They are database
|
||||
# objects, not table metadata, so no model can carry them and their
|
||||
# absence is not a model defect — it is simply outside what this
|
||||
# comparison is asking about.
|
||||
- name: Build the schema the CURRENT tree produces
|
||||
env:
|
||||
MODE: ${{ github.event.inputs.mode }}
|
||||
run: |
|
||||
set -eux
|
||||
docker exec "$PG_CONTAINER" createdb -U fabledcurator fc_base
|
||||
if [ "${MODE:-chain}" = "models" ]; then
|
||||
docker exec "$PG_CONTAINER" psql -U fabledcurator -d fc_base \
|
||||
-c "CREATE EXTENSION IF NOT EXISTS vector" \
|
||||
-c "CREATE EXTENSION IF NOT EXISTS tsm_system_rows"
|
||||
mkdir -p /tmp/held
|
||||
mv alembic/versions/*.py /tmp/held/
|
||||
cp /tmp/candidate/*.py alembic/versions/
|
||||
# Autogenerate EMITS pgvector.sqlalchemy.vector.VECTOR(...) without
|
||||
# importing it, so the file it writes cannot run:
|
||||
# NameError: name 'pgvector' is not defined
|
||||
# Observed on run 4988, which is the proof rather than the theory.
|
||||
# This is a defect in the GENERATOR, not in the models, so it is
|
||||
# repaired here rather than counted as a schema difference — the
|
||||
# comparison is about whether the models describe the schema.
|
||||
sed -i '0,/^import sqlalchemy as sa$/s//import sqlalchemy as sa\nimport pgvector.sqlalchemy.vector/' alembic/versions/*.py
|
||||
grep -n 'import pgvector' alembic/versions/*.py
|
||||
ls alembic/versions/*.py
|
||||
DB_NAME=fc_base alembic upgrade head
|
||||
rm -f alembic/versions/*.py
|
||||
mv /tmp/held/*.py alembic/versions/
|
||||
else
|
||||
ls alembic/versions/*.py | wc -l
|
||||
DB_NAME=fc_base alembic upgrade head
|
||||
fi
|
||||
docker exec "$PG_CONTAINER" pg_dump -U fabledcurator --schema-only \
|
||||
--no-owner --no-privileges -d fc_base > baseline.sql
|
||||
wc -l baseline.sql
|
||||
@@ -191,6 +265,25 @@ jobs:
|
||||
# these two lines and nothing else. That control is what licenses this
|
||||
# filter — it was observed to be the only false positive, rather than
|
||||
# assumed to be one.
|
||||
# Column ORDER inside a CREATE TABLE is compared separately from column
|
||||
# CONTENT, and only content is fatal.
|
||||
#
|
||||
# A table built by 87 migrations has its columns in ADD COLUMN order; the
|
||||
# same table built in one shot has them in declaration order. That is a
|
||||
# real and permanent difference which no baseline can erase — the
|
||||
# operator's existing database keeps chain order forever, a fresh install
|
||||
# gets model order — so a check that fails on it would never pass and
|
||||
# would teach nothing. FC reaches every column through the ORM by name,
|
||||
# and `SELECT *` ordering is not depended on anywhere.
|
||||
#
|
||||
# So the second pass SORTS the column lines within each CREATE TABLE
|
||||
# rather than DELETING them. That distinction is the whole point: sorting
|
||||
# cannot hide a column that exists on one side only, or one whose type,
|
||||
# nullability or default differs — those still land in the diff. A filter
|
||||
# could have hidden all three.
|
||||
#
|
||||
# Both diffs are reported. The ordered one is informational; the
|
||||
# order-insensitive one is the verdict.
|
||||
- name: Diff
|
||||
run: |
|
||||
set -eu
|
||||
@@ -202,11 +295,54 @@ jobs:
|
||||
norm chain.sql > a.txt
|
||||
norm baseline.sql > b.txt
|
||||
echo "normalised: chain=$(wc -l < a.txt) lines, current=$(wc -l < b.txt) lines"
|
||||
|
||||
sort_table_columns() {
|
||||
python3 - "$1" <<'PYEOF'
|
||||
import re, sys
|
||||
|
||||
lines = open(sys.argv[1]).read().splitlines()
|
||||
out, block = [], None
|
||||
for line in lines:
|
||||
if block is not None:
|
||||
# ');' on its own closes the CREATE TABLE body.
|
||||
if line.strip() == ");":
|
||||
out.extend(sorted(block))
|
||||
out.append(line)
|
||||
block = None
|
||||
else:
|
||||
# Drop the list comma before sorting. Only the LAST
|
||||
# column lacks one, so keeping it would make every
|
||||
# reordering look like a content change as well — the
|
||||
# comma is punctuation, and carries no schema meaning.
|
||||
block.append(line.rstrip().rstrip(","))
|
||||
continue
|
||||
out.append(line)
|
||||
if re.match(r"CREATE TABLE .*\($", line):
|
||||
block = []
|
||||
if block is not None: # unterminated body: emit it rather than drop it
|
||||
out.extend(block)
|
||||
print("\n".join(out))
|
||||
PYEOF
|
||||
}
|
||||
sort_table_columns a.txt > a.sorted.txt
|
||||
sort_table_columns b.txt > b.sorted.txt
|
||||
test "$(wc -l < a.sorted.txt)" = "$(wc -l < a.txt)"
|
||||
test "$(wc -l < b.sorted.txt)" = "$(wc -l < b.txt)"
|
||||
|
||||
if diff -u a.txt b.txt > schema.diff; then
|
||||
echo "SCHEMAS IDENTICAL — the collapsed chain reproduces the old one."
|
||||
echo "ORDERED DIFF: identical, column order included."
|
||||
else
|
||||
echo "SCHEMAS DIFFER — $(grep -cE '^[+-]' schema.diff) changed lines:"
|
||||
echo "ORDERED DIFF: $(grep -cE '^[+-]' schema.diff) changed lines (informational):"
|
||||
cat schema.diff
|
||||
fi
|
||||
echo
|
||||
echo "================================================================"
|
||||
echo
|
||||
if diff -u a.sorted.txt b.sorted.txt > sorted.diff; then
|
||||
echo "SCHEMAS MATCH — every difference above is column ORDER alone."
|
||||
else
|
||||
echo "SCHEMAS DIFFER — $(grep -cE '^[+-]' sorted.diff) changed lines that are NOT ordering:"
|
||||
cat sorted.diff
|
||||
echo
|
||||
echo "The baseline is wrong, not the database. Do not stamp."
|
||||
exit 1
|
||||
|
||||
+565
-59
@@ -44,6 +44,10 @@ on:
|
||||
description: 'Rebuild every image even if the published revision matches'
|
||||
type: boolean
|
||||
default: false
|
||||
refresh:
|
||||
description: 'Behave as the weekly base refresh: build main against fresh bases, publish through the candidate tag'
|
||||
type: boolean
|
||||
default: false
|
||||
|
||||
# The base-image refresh (milestone 326 step 4, #3154).
|
||||
#
|
||||
@@ -72,8 +76,43 @@ on:
|
||||
# Deriving it per job invites the two halves to disagree: sign-extension would
|
||||
# derive dev's extension version while build-web bundled main's, and the
|
||||
# release download would 404 on a version that exists perfectly well.
|
||||
# IS THIS A BASE REFRESH? Asked in five places and previously spelled five
|
||||
# ways — `github.event_name == 'schedule'` in an `if:`, `$GITHUB_EVENT_NAME` in
|
||||
# one shell, an `EVENT:` env passed into another, and a bare expression on
|
||||
# `pull:`. Five spellings of one fact is how half of them come to disagree
|
||||
# after somebody adds a sixth trigger.
|
||||
#
|
||||
# The `refresh` dispatch input is here so this path can be EXERCISED. A weekly
|
||||
# cron is otherwise testable once a week, which is not a cadence anything can
|
||||
# be developed against — the same reason `force_build` exists (#3252, added to
|
||||
# confirm #3190 was gone rather than wait for it to recur). It is also what
|
||||
# makes the milestone-362 gate verifiable at all: a gate has to be watched
|
||||
# rejecting something before anyone can believe it is wired up.
|
||||
#
|
||||
# The input is normalised through `format()` before it is compared, and that
|
||||
# is not defensive styling — the direct comparison is WRONG and fails silently.
|
||||
#
|
||||
# `type: boolean` delivers a real boolean, and GitHub expression semantics cast
|
||||
# operands to numbers when their types differ: `true == 'true'` compares 1
|
||||
# against NaN and is FALSE. Measured on run 5270, whose own log says it —
|
||||
#
|
||||
# expression '(github.event_name == 'schedule'
|
||||
# || github.event.inputs.refresh == 'true') && 'true' || 'false''
|
||||
# evaluated to '%!t(string=false)'
|
||||
# trigger: raw inputs refresh='true'
|
||||
#
|
||||
# — the input arrived as `true` and the expression still said false. The run
|
||||
# then went green with every step skipped, because a refresh that evaluates
|
||||
# false behaves exactly like an ordinary push. That is the whole hazard: the
|
||||
# failure has no symptom.
|
||||
#
|
||||
# `force_build` never hit this because it never compares in an expression. It
|
||||
# passes the raw value into an env var and tests it in the shell, where
|
||||
# everything is already a string. `format('{0}', x)` buys the same thing here,
|
||||
# where a step-level `if:` needs the answer before any shell runs.
|
||||
env:
|
||||
BUILD_REF: ${{ github.event_name == 'schedule' && 'main' || github.ref }}
|
||||
IS_REFRESH: ${{ (github.event_name == 'schedule' || format('{0}', github.event.inputs.refresh) == 'true') && 'true' || 'false' }}
|
||||
BUILD_REF: ${{ (github.event_name == 'schedule' || format('{0}', github.event.inputs.refresh) == 'true') && 'main' || github.ref }}
|
||||
|
||||
# Requires repo secret RELEASE_TOKEN — a Forgejo PAT with scopes:
|
||||
# - write:package, read:package (for docker push to git.fabledsword.com)
|
||||
@@ -143,7 +182,7 @@ jobs:
|
||||
# evaluate — this file already gates steps on it — so the guard cannot
|
||||
# be disabled by the same uncertainty it exists to cover.
|
||||
- name: Guard — a scheduled run must have checked out main
|
||||
if: github.event_name == 'schedule'
|
||||
if: env.IS_REFRESH == 'true'
|
||||
run: |
|
||||
set -eu
|
||||
BRANCH=$(git rev-parse --abbrev-ref HEAD)
|
||||
@@ -413,6 +452,18 @@ jobs:
|
||||
# to. Same source of truth; no double-store.
|
||||
|
||||
build-web:
|
||||
# Consumed by smoke-web's job-level `if:`. It cannot read `env` — the env
|
||||
# context is available to STEP `if:` and step bodies, never to a job's own
|
||||
# condition, and an unresolvable context there is empty rather than an
|
||||
# error. `smoke-web` skipped silently on run 5290 for exactly that reason.
|
||||
#
|
||||
# Keying off the reuse step's own output is better than re-deriving the
|
||||
# trigger anyway: it is the same single decision the build, the XPI
|
||||
# download and the promote all take (build.yml's "one decision drives
|
||||
# everything downstream"), and it says the thing smoke-web actually needs
|
||||
# to know — a candidate was published — rather than restating why.
|
||||
outputs:
|
||||
candidate: ${{ steps.reuse.outputs.promote }}
|
||||
# A plain `needs` — no `always()`. That expression existed to let a
|
||||
# SKIPPED sign-extension through on a tag push while still blocking a
|
||||
# FAILED one. With no tag trigger, sign-extension always runs, so the
|
||||
@@ -437,7 +488,7 @@ jobs:
|
||||
|
||||
# See sign-extension's copy for why this guard exists.
|
||||
- name: Guard — a scheduled run must have checked out main
|
||||
if: github.event_name == 'schedule'
|
||||
if: env.IS_REFRESH == 'true'
|
||||
run: |
|
||||
set -eu
|
||||
BRANCH=$(git rev-parse --abbrev-ref HEAD)
|
||||
@@ -470,8 +521,18 @@ jobs:
|
||||
# the PREVIOUS XPI while the freshly signed one is orphaned (#3156).
|
||||
# * dev and main derive the same values for the same source.
|
||||
- name: Report the derived artifact version
|
||||
env:
|
||||
# Diagnostic for the trigger normalisation. `refresh` is reported RAW
|
||||
# as well as normalised, because the two disagreeing is the whole
|
||||
# failure mode: a dispatch input whose type does not compare the way
|
||||
# the expression assumes evaluates to false silently, and the only
|
||||
# symptom is a refresh that quietly behaves like an ordinary push.
|
||||
RAW_REFRESH: ${{ github.event.inputs.refresh }}
|
||||
RAW_FORCE: ${{ github.event.inputs.force_build }}
|
||||
run: |
|
||||
set -u
|
||||
echo "trigger: event=$GITHUB_EVENT_NAME IS_REFRESH='${IS_REFRESH:-<unset>}' BUILD_REF='${BUILD_REF:-<unset>}'"
|
||||
echo "trigger: raw inputs refresh='${RAW_REFRESH:-<unset>}' force_build='${RAW_FORCE:-<unset>}'"
|
||||
A=web
|
||||
V=$(sh scripts/artifacts.sh version "$A" 2>&1 || echo UNAVAILABLE)
|
||||
R=$(sh scripts/artifacts.sh revision "$A" 2>&1 || echo UNAVAILABLE)
|
||||
@@ -528,7 +589,7 @@ jobs:
|
||||
# Checked BEFORE the ref test, not after: a scheduled run's
|
||||
# GITHUB_REF is the default branch (dev), so the main test would
|
||||
# never fire on it.
|
||||
if [ "${GITHUB_EVENT_NAME:-}" = "schedule" ]; then
|
||||
if [ "${IS_REFRESH:-}" = "true" ]; then
|
||||
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator:latest" >> "$GITHUB_OUTPUT"
|
||||
echo "channel=main" >> "$GITHUB_OUTPUT"
|
||||
elif [ "${GITHUB_REF##*/}" = "main" ]; then
|
||||
@@ -628,7 +689,6 @@ jobs:
|
||||
# A scheduled refresh has to bypass reuse by construction: it
|
||||
# rebuilds the SAME source, so fc.revision always matches and the
|
||||
# check would skip every refresh there has ever been.
|
||||
EVENT: ${{ github.event_name }}
|
||||
run: |
|
||||
set -eu
|
||||
DERIVED=$(sh scripts/artifacts.sh revision web)
|
||||
@@ -638,11 +698,60 @@ jobs:
|
||||
# adds no variability the reuse check would have to account for.
|
||||
echo "version=$(sh scripts/artifacts.sh version web)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# The build clock, pinned to the same commit (#3265). Without it
|
||||
# buildkit stamps the image config with the wall clock of the build,
|
||||
# so identical layers republish under a new config blob and the
|
||||
# channel tag gets a new manifest digest for no reason. Derived from
|
||||
# `newest()` like revision and version, so all three name one commit
|
||||
# and cannot drift apart.
|
||||
echo "epoch=$(sh scripts/artifacts.sh epoch web)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# The moving tag for this channel. Which tag we ask IS the channel —
|
||||
# that is why the revision needs no -main/-dev qualifier any more.
|
||||
if [ "$CHANNEL" = "main" ]; then T=latest; else T=dev; fi
|
||||
echo "channel_ref=$IMAGE:$T" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# WHERE THE BUILD PUBLISHES, which is not always the channel — and
|
||||
# whether the channel then has to be written separately.
|
||||
#
|
||||
# On a push the build writes the channel tag directly: the bytes came
|
||||
# from a commit, and a commit is the thing CI tests. Nothing to hold
|
||||
# it behind.
|
||||
#
|
||||
# On the scheduled refresh it writes a CANDIDATE tag instead. A
|
||||
# refresh rebuilds against freshly resolved base images, and the web
|
||||
# image's runtime is a line of UNPINNED Debian packages (ffmpeg,
|
||||
# libjpeg62-turbo, libpq5, megatools…) re-resolved on every build.
|
||||
# Nothing in ci.yml can see that: its lanes run on ci-python:3.14 and
|
||||
# install requirements.txt, and a base bump changes neither. So
|
||||
# refreshed bytes have to be proven before :latest names them, and
|
||||
# proving needs a moment between "built" and "published" to occupy.
|
||||
# This is that moment; :latest goes on naming the build that works
|
||||
# until something says otherwise.
|
||||
#
|
||||
# `:refresh-candidate` is one moving ref per image, overwritten in
|
||||
# place, holding a build nobody is told to pull — the shape rule 145
|
||||
# already allows for :buildcache, not the per-build tag family that
|
||||
# milestone 318 withdrew.
|
||||
#
|
||||
# Decided HERE, beside `hit`, for the reason the force/schedule
|
||||
# branch below gives: one step decides what this job does. A
|
||||
# condition derived independently could disagree with the tag the
|
||||
# build actually wrote.
|
||||
#
|
||||
# build-web additionally exposes this as `outputs.candidate`, which is
|
||||
# what gates the `promote` job — a job's `if:` cannot read `env`, and
|
||||
# one flag is enough because all three derive it from the same
|
||||
# IS_REFRESH. ml and agent do not re-emit it; a second copy nothing
|
||||
# reads is the kind of thing that later reads as load-bearing.
|
||||
if [ "${IS_REFRESH:-}" = "true" ]; then
|
||||
echo "build_ref=$IMAGE:refresh-candidate" >> "$GITHUB_OUTPUT"
|
||||
echo "promote=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "build_ref=$IMAGE:$T" >> "$GITHUB_OUTPUT"
|
||||
echo "promote=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
# Compare VALUES, never exit codes. Measured on buildx v0.36.1
|
||||
# (run 4732): a missing key returns an empty string and exits 0, so
|
||||
# branching on the exit code would read "no label yet" as success.
|
||||
@@ -675,7 +784,7 @@ jobs:
|
||||
if [ "${FORCE:-false}" = "true" ]; then
|
||||
echo "hit=false" >> "$GITHUB_OUTPUT"
|
||||
echo "reuse: force_build set — building regardless"
|
||||
elif [ "${EVENT:-}" = "schedule" ]; then
|
||||
elif [ "${IS_REFRESH:-}" = "true" ]; then
|
||||
echo "hit=false" >> "$GITHUB_OUTPUT"
|
||||
echo "reuse: scheduled base refresh — building regardless"
|
||||
elif [ -n "$PUBLISHED" ] && [ "$PUBLISHED" = "$DERIVED" ]; then
|
||||
@@ -764,6 +873,12 @@ jobs:
|
||||
|
||||
- name: Build and push web image
|
||||
if: steps.reuse.outputs.hit != 'true'
|
||||
# Read by buildx out of the ENVIRONMENT, not passed as a build-arg —
|
||||
# it normalises the image config's `created` field and the history
|
||||
# timestamps rather than being consumed by the Dockerfile. See #3265
|
||||
# and the reuse step's `epoch` output.
|
||||
env:
|
||||
SOURCE_DATE_EPOCH: ${{ steps.reuse.outputs.epoch }}
|
||||
uses: docker/build-push-action@v5
|
||||
with:
|
||||
context: .
|
||||
@@ -776,20 +891,17 @@ jobs:
|
||||
# invalidates, and the image genuinely rebuilds.
|
||||
#
|
||||
# MEASURED on the first real fire, run 4934 (#3265): when the base
|
||||
# did NOT move, the build is ~13s and every content step reports
|
||||
# CACHED — but the channel tag STILL gets a new manifest digest.
|
||||
# buildkit mints a fresh image config each run, so identical layers
|
||||
# are republished under a new config blob. All three images moved
|
||||
# that way on 2026-08-30 with nothing whatsoever changed in them.
|
||||
# did NOT move, the build was ~13s with every content step CACHED —
|
||||
# and the channel tag STILL got a new manifest digest, because
|
||||
# buildkit stamps a fresh image config per run and republishes the
|
||||
# identical layers under it. All three images moved that way on
|
||||
# 2026-08-30 with nothing whatsoever changed in them.
|
||||
#
|
||||
# So a refresh currently rewrites :latest every Sunday whether or
|
||||
# not there is anything new in it, and :c-<sha> is handed a new
|
||||
# manifest to diverge from on the same cadence. Layers are shared,
|
||||
# so the storage cost is a config blob; the cost that matters is
|
||||
# that a digest change no longer MEANS anything. Tracked in #3265 —
|
||||
# the likely fix is a deterministic SOURCE_DATE_EPOCH, which would
|
||||
# make "same source, same bytes" true and turn the no-op case into
|
||||
# a genuine no-op.
|
||||
# SOURCE_DATE_EPOCH (below) is the fix: pinned to the commit the
|
||||
# content came from, the config is byte-identical across runs, so
|
||||
# the manifest digest is too and the push is a registry no-op. A
|
||||
# digest change means the content changed again, which is the only
|
||||
# thing a digest is any use for.
|
||||
#
|
||||
# What `pull` does NOT catch either: a Debian package update inside
|
||||
# the `apt-get install` layer while the base tag itself stands
|
||||
@@ -799,14 +911,14 @@ jobs:
|
||||
# churn #3265 is about.
|
||||
#
|
||||
# Only on the schedule. An ordinary push wants the cached base.
|
||||
pull: ${{ github.event_name == 'schedule' }}
|
||||
pull: ${{ env.IS_REFRESH == 'true' }}
|
||||
# ONE tag, the channel's. Every other tag is written by the step
|
||||
# below, registry-side. buildx here pushes the first tag to the
|
||||
# registry and then re-pushes the rest through the DOCKER driver,
|
||||
# out of a local image store a registry-direct build never filled —
|
||||
# #3190, which cost `main` its :c-<sha> on 2026-08-29 while :latest
|
||||
# published perfectly well.
|
||||
tags: ${{ steps.reuse.outputs.channel_ref }}
|
||||
tags: ${{ steps.reuse.outputs.build_ref }}
|
||||
# The reuse key. Read back off the channel tag on the next push to
|
||||
# decide whether that push needs to build at all, so this is not
|
||||
# decoration — an unstamped image is one that will always rebuild.
|
||||
@@ -937,6 +1049,282 @@ jobs:
|
||||
docker buildx imagetools create $ARGS "$SOURCE"
|
||||
echo "repointed from $SOURCE:$ARGS"
|
||||
|
||||
# Does the image a refresh just built still work?
|
||||
#
|
||||
# This is the gate the base refresh never had. `ci.yml` cannot be it: its
|
||||
# lanes run on ci-python:3.14 and install requirements.txt, and a base bump
|
||||
# changes neither — all five stay green through a refresh that breaks the
|
||||
# product. What a refresh re-resolves is the Dockerfile's apt layer (ffmpeg,
|
||||
# unar, libpq5, postgresql-client, zstd, megatools, libjpeg62-turbo,
|
||||
# libwebp7, libpng16-16), unpinned, every build.
|
||||
#
|
||||
# So this runs the CANDIDATE IMAGE, against real Postgres and Redis. Not the
|
||||
# source tree, and not a static inspection: `ffmpeg -version` exiting 0 would
|
||||
# pass while a codec removal broke every thumbnail in the library.
|
||||
#
|
||||
# Refresh-only. On a push the bytes came from a commit, and a commit is what
|
||||
# ci.yml already tests.
|
||||
#
|
||||
# Reports a verdict; it does not yet gate the promote (milestone 362 step 4).
|
||||
# Landing the gate and the thing it gates in one change would mean the first
|
||||
# time anyone saw this job run would also be the first time it could stop a
|
||||
# publish.
|
||||
smoke-web:
|
||||
needs: [build-web]
|
||||
if: needs.build-web.outputs.candidate == 'true'
|
||||
runs-on: python-ci
|
||||
container:
|
||||
image: git.fabledsword.com/bvandeusen/ci-python:3.14
|
||||
env:
|
||||
DB_USER: fabledcurator
|
||||
DB_PASSWORD: ci_smoke
|
||||
DB_PORT: "5432"
|
||||
DB_NAME: fabledcurator_smoke
|
||||
SECRET_KEY: ci_smoke_placeholder
|
||||
IMAGE: git.fabledsword.com/bvandeusen/fabledcurator
|
||||
services:
|
||||
postgres:
|
||||
image: pgvector/pgvector:pg16
|
||||
env:
|
||||
POSTGRES_USER: fabledcurator
|
||||
POSTGRES_PASSWORD: ci_smoke
|
||||
POSTGRES_DB: fabledcurator_smoke
|
||||
options: >-
|
||||
--health-cmd "pg_isready -U fabledcurator"
|
||||
--health-interval 10s
|
||||
--health-timeout 5s
|
||||
--health-retries 10
|
||||
redis:
|
||||
image: redis:7-alpine
|
||||
options: >-
|
||||
--health-cmd "redis-cli ping"
|
||||
--health-interval 10s
|
||||
--health-timeout 5s
|
||||
--health-retries 10
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
# The same ref the image was built from, so the smoke script matches
|
||||
# the code inside the candidate.
|
||||
ref: ${{ env.BUILD_REF }}
|
||||
|
||||
- name: Smoke the candidate image
|
||||
env:
|
||||
TOKEN: ${{ secrets.RELEASE_TOKEN }}
|
||||
ACTOR: ${{ github.actor }}
|
||||
run: |
|
||||
set -eux
|
||||
# Service discovery mirrors ci.yml's integration lane: these jobs run
|
||||
# in a container against a mounted docker socket, so the services are
|
||||
# SIBLINGS reachable by IP, not by hostname.
|
||||
PG=$(docker ps --filter "name=smoke" --filter "ancestor=pgvector/pgvector:pg16" -q | head -n1)
|
||||
RD=$(docker ps --filter "name=smoke" --filter "ancestor=redis:7-alpine" -q | head -n1)
|
||||
test -n "$PG" && test -n "$RD"
|
||||
PG_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$PG")
|
||||
RD_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$RD")
|
||||
test -n "$PG_IP" && test -n "$RD_IP"
|
||||
|
||||
# Socket probe in python, not bash's /dev/tcp — these steps run under
|
||||
# `sh -e`, where that path does not exist. Same fix and reasoning as
|
||||
# ci.yml's integration job; see the comment there.
|
||||
pg_ready=""
|
||||
for i in $(seq 1 60); do
|
||||
if python -c "import socket,sys; s=socket.socket(); s.settimeout(2); sys.exit(0 if s.connect_ex(('$PG_IP', 5432)) == 0 else 1)"; then
|
||||
pg_ready=1
|
||||
break
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
if [ -z "$pg_ready" ]; then
|
||||
echo "postgres at $PG_IP:5432 did not accept a connection within 120s"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "$TOKEN" | docker login git.fabledsword.com -u "$ACTOR" --password-stdin
|
||||
CANDIDATE="$IMAGE:refresh-candidate"
|
||||
docker pull "$CANDIDATE"
|
||||
|
||||
ENVOPTS="-e DB_USER=$DB_USER -e DB_PASSWORD=$DB_PASSWORD -e DB_HOST=$PG_IP"
|
||||
ENVOPTS="$ENVOPTS -e DB_PORT=5432 -e DB_NAME=$DB_NAME -e SECRET_KEY=$SECRET_KEY"
|
||||
ENVOPTS="$ENVOPTS -e CELERY_BROKER_URL=redis://$RD_IP:6379/0"
|
||||
ENVOPTS="$ENVOPTS -e CELERY_RESULT_BACKEND=redis://$RD_IP:6379/0"
|
||||
# A throwaway CI instance IS first-time setup, which is the one case
|
||||
# credential_crypto allows a key to be minted in. Without it the web
|
||||
# role refuses to boot — deliberately, since silently generating a
|
||||
# key on a restored-DB-but-lost-secrets deployment would leave every
|
||||
# Credential row undecryptable (the 2026-06-02 audit). Discovered by
|
||||
# this job on its first real run; see #3422 for the fact that no
|
||||
# user-facing file mentions this variable at all.
|
||||
ENVOPTS="$ENVOPTS -e CURATOR_BOOTSTRAP_NEW_KEY=1"
|
||||
|
||||
# 1. The schema builds from empty, using the image's OWN libpq and
|
||||
# psycopg. This is the same call entrypoint.sh makes before it
|
||||
# serves anything, so a failure here is a failure to boot.
|
||||
echo "smoke: alembic upgrade head"
|
||||
docker run --rm $ENVOPTS "$CANDIDATE" alembic upgrade head
|
||||
|
||||
# 2. The apt layer's binaries and the app's own thumbnail path, run
|
||||
# inside the image. Piped over stdin rather than bind-mounted: the
|
||||
# workspace is a docker VOLUME belonging to this job's container,
|
||||
# so a host bind of $PWD would not resolve for a sibling.
|
||||
echo "smoke: image-internal checks"
|
||||
docker run --rm -i $ENVOPTS "$CANDIDATE" shell -c 'python3 -' < scripts/smoke_image.py
|
||||
|
||||
# 3. It actually serves. `docker run -d` then poll the container's own
|
||||
# IP — no port publishing, because the job container reaches
|
||||
# siblings directly and a published port would collide with
|
||||
# whatever else the runner is hosting.
|
||||
echo "smoke: web boots and answers /api/health"
|
||||
CID=$(docker run -d $ENVOPTS "$CANDIDATE" web)
|
||||
# Clean up the container however this ends, and dump its log ONLY
|
||||
# on failure — a boot that never answers must fail with the reason
|
||||
# visible rather than as a bare timeout (rule 156), while a green run
|
||||
# has nothing to say. `exit $rc` preserves the real status, which a
|
||||
# trap that ends on a successful `docker rm` would otherwise mask.
|
||||
trap 'rc=$?; [ $rc -eq 0 ] || docker logs "$CID" 2>&1 | tail -40; docker rm -f "$CID" >/dev/null 2>&1 || true; exit $rc' EXIT
|
||||
WEB_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$CID")
|
||||
test -n "$WEB_IP"
|
||||
healthy=""
|
||||
for i in $(seq 1 60); do
|
||||
if curl -fsS --max-time 5 "http://$WEB_IP:8080/api/health" >/dev/null 2>&1; then
|
||||
healthy=1
|
||||
break
|
||||
fi
|
||||
# A container that has EXITED will never answer, so stop asking.
|
||||
# Without this the loop spent 3m35s polling a dead container on
|
||||
# this job's first run, and — because docker recycles the IP — got
|
||||
# a confusing mix of connection-refused and 5s timeouts from
|
||||
# whatever took the address next. The trap's log dump had the real
|
||||
# answer the whole time; this just stops burying it.
|
||||
if [ "$(docker inspect -f '{{.State.Running}}' "$CID" 2>/dev/null)" != "true" ]; then
|
||||
echo "smoke: FAILED — the web container exited during boot." >&2
|
||||
echo "smoke: its log follows; entrypoint runs alembic BEFORE" >&2
|
||||
echo "smoke: serving, so a startup exception lands here." >&2
|
||||
exit 1
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
if [ -z "$healthy" ]; then
|
||||
# 60 iterations of (up to 5s connect + 2s sleep) — up to ~7min, not
|
||||
# the 120s an earlier version of this message claimed.
|
||||
echo "smoke: FAILED — web is running but never answered" >&2
|
||||
echo "smoke: /api/health. It is up, so look at hypercorn and the" >&2
|
||||
echo "smoke: python base rather than at startup." >&2
|
||||
exit 1
|
||||
fi
|
||||
curl -fsS --max-time 5 "http://$WEB_IP:8080/api/health"
|
||||
echo
|
||||
|
||||
echo "smoke: all checks passed against $CANDIDATE"
|
||||
|
||||
# Move the channel tags — the whole point of the gate.
|
||||
#
|
||||
# Lives in its own job because the verdict it depends on cannot exist until
|
||||
# after build-web has finished, and the promote used to run INSIDE build-web.
|
||||
#
|
||||
# `needs` on smoke-web is the gate. A failed smoke skips this job, so a
|
||||
# refresh that broke something leaves :latest naming the build that works —
|
||||
# "the refresh failed" and "production is broken" must not be the same event.
|
||||
# A SKIPPED smoke also skips this job, which is the behaviour that matters
|
||||
# most: on run 5290 the gate silently skipped itself, and a design where only
|
||||
# a FAILED gate blocks would have published unverified images while reporting
|
||||
# success. Not running is not the same as passing.
|
||||
#
|
||||
# All three images promote TOGETHER, or none do. They are one stack: build.yml
|
||||
# already refuses to publish a :dev web image beside a stale :dev ml, because
|
||||
# the mismatch only shows up as a runtime failure. A refresh that published ml
|
||||
# and withheld web would be that same trap, arrived at through the gate.
|
||||
#
|
||||
# The gate covers the web image only (milestone 362 step 3 scoped it there),
|
||||
# so ml and agent are being held to web's verdict rather than their own. That
|
||||
# is deliberate and it is the conservative direction — they ship together, so
|
||||
# the weakest evidence should govern all three — but it is not the same as
|
||||
# having smoked them, and it should not be read as if it were.
|
||||
promote:
|
||||
needs: [build-web, build-ml, build-agent, smoke-web]
|
||||
# Only a refresh publishes through a candidate; a push writes its channel
|
||||
# tag directly from the build. Reads the same reuse-step decision the build
|
||||
# took, via a job output — a job's `if:` cannot see the `env` context.
|
||||
if: needs.build-web.outputs.candidate == 'true'
|
||||
runs-on: python-ci
|
||||
container:
|
||||
image: git.fabledsword.com/bvandeusen/ci-python:3.14
|
||||
steps:
|
||||
- name: Point the channel tags at the smoked candidates
|
||||
env:
|
||||
TOKEN: ${{ secrets.RELEASE_TOKEN }}
|
||||
ACTOR: ${{ github.actor }}
|
||||
run: |
|
||||
set -eu
|
||||
# `latest` is not a guess: a refresh always builds `main` (BUILD_REF),
|
||||
# and the "must have checked out main" guard in every build job fails
|
||||
# the run if that did not hold. So the channel is main's.
|
||||
TAG=latest
|
||||
FAILED=""
|
||||
|
||||
for NAME in fabledcurator fabledcurator-ml fabledcurator-agent; do
|
||||
REPO="bvandeusen/$NAME"
|
||||
echo "promote: $REPO"
|
||||
|
||||
# Registry auth is its own token exchange — `docker login`
|
||||
# authenticates the docker client, not curl. Deadline on every call
|
||||
# (rule 156): a registry that stops answering must fail this step,
|
||||
# not hang the weekly refresh until the job times out.
|
||||
BEARER=$(curl -fsS --max-time 30 -u "$ACTOR:$TOKEN" \
|
||||
"https://git.fabledsword.com/v2/token?scope=repository:$REPO:pull,push&service=git.fabledsword.com" \
|
||||
| python3 -c 'import sys,json; print(json.load(sys.stdin)["token"])')
|
||||
|
||||
# Ask for the IMAGE manifest media types only. Offering the index
|
||||
# types too would let the registry hand back an index if one ever
|
||||
# existed at this tag, and we would faithfully copy the thing this
|
||||
# whole approach exists to avoid creating.
|
||||
ACCEPT='application/vnd.oci.image.manifest.v1+json, application/vnd.docker.distribution.manifest.v2+json'
|
||||
CT=$(curl -fsS --max-time 60 -o manifest.json -D headers.txt \
|
||||
-H "Authorization: Bearer $BEARER" -H "Accept: $ACCEPT" \
|
||||
"https://git.fabledsword.com/v2/$REPO/manifests/refresh-candidate" \
|
||||
&& tr -d '\r' < headers.txt | awk -F': ' '/^[Cc]ontent-[Tt]ype:/{print $2}')
|
||||
test -n "$CT"
|
||||
SRC=$(tr -d '\r' < headers.txt | awk -F': ' '/^[Dd]ocker-[Cc]ontent-[Dd]igest:/{print $2}')
|
||||
echo "promote: candidate $SRC ($CT)"
|
||||
|
||||
# NOT `imagetools create`. That wraps its source in an INDEX, and
|
||||
# `.Image.Config.Labels` does not resolve through one — the
|
||||
# fc.revision the reuse check reads off the channel tag would come
|
||||
# back empty, every later push would miss and rebuild, and nothing
|
||||
# would go red (#3183, run 4751). A manifest PUT is what "make this
|
||||
# tag name that image" means at the registry: same bytes, same media
|
||||
# type, same digest, no layer transfer.
|
||||
curl -fsS --max-time 120 -X PUT \
|
||||
-H "Authorization: Bearer $BEARER" -H "Content-Type: $CT" \
|
||||
--data-binary @manifest.json \
|
||||
"https://git.fabledsword.com/v2/$REPO/manifests/$TAG"
|
||||
|
||||
# Read it back. A PUT that returned 2xx but landed something else is
|
||||
# exactly the silent-and-plausible failure this pipeline keeps
|
||||
# producing, and the check costs one request.
|
||||
NOW=$(curl -fsS --max-time 30 -o /dev/null -D - \
|
||||
-H "Authorization: Bearer $BEARER" -H "Accept: $ACCEPT" \
|
||||
"https://git.fabledsword.com/v2/$REPO/manifests/$TAG" \
|
||||
| tr -d '\r' | awk -F': ' '/^[Dd]ocker-[Cc]ontent-[Dd]igest:/{print $2}')
|
||||
if [ "$NOW" != "$SRC" ]; then
|
||||
echo "promote: FAILED — $NAME:$TAG is $NOW, expected $SRC" >&2
|
||||
FAILED="$FAILED $NAME"
|
||||
continue
|
||||
fi
|
||||
echo "promote: $NAME:$TAG now names $NOW"
|
||||
done
|
||||
|
||||
if [ -n "$FAILED" ]; then
|
||||
echo "" >&2
|
||||
echo "promote: FAILED for:$FAILED" >&2
|
||||
echo "promote: the channel tags are now INCONSISTENT — some images" >&2
|
||||
echo "promote: moved and some did not. Re-run this refresh; the" >&2
|
||||
echo "promote: candidates are still published and the promote is" >&2
|
||||
echo "promote: idempotent." >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "promote: all three channel tags moved"
|
||||
|
||||
build-ml:
|
||||
runs-on: python-ci
|
||||
container:
|
||||
@@ -957,7 +1345,7 @@ jobs:
|
||||
|
||||
# See sign-extension's copy for why this guard exists.
|
||||
- name: Guard — a scheduled run must have checked out main
|
||||
if: github.event_name == 'schedule'
|
||||
if: env.IS_REFRESH == 'true'
|
||||
run: |
|
||||
set -eu
|
||||
BRANCH=$(git rev-parse --abbrev-ref HEAD)
|
||||
@@ -990,8 +1378,18 @@ jobs:
|
||||
# the PREVIOUS XPI while the freshly signed one is orphaned (#3156).
|
||||
# * dev and main derive the same values for the same source.
|
||||
- name: Report the derived artifact version
|
||||
env:
|
||||
# Diagnostic for the trigger normalisation. `refresh` is reported RAW
|
||||
# as well as normalised, because the two disagreeing is the whole
|
||||
# failure mode: a dispatch input whose type does not compare the way
|
||||
# the expression assumes evaluates to false silently, and the only
|
||||
# symptom is a refresh that quietly behaves like an ordinary push.
|
||||
RAW_REFRESH: ${{ github.event.inputs.refresh }}
|
||||
RAW_FORCE: ${{ github.event.inputs.force_build }}
|
||||
run: |
|
||||
set -u
|
||||
echo "trigger: event=$GITHUB_EVENT_NAME IS_REFRESH='${IS_REFRESH:-<unset>}' BUILD_REF='${BUILD_REF:-<unset>}'"
|
||||
echo "trigger: raw inputs refresh='${RAW_REFRESH:-<unset>}' force_build='${RAW_FORCE:-<unset>}'"
|
||||
A=ml
|
||||
V=$(sh scripts/artifacts.sh version "$A" 2>&1 || echo UNAVAILABLE)
|
||||
R=$(sh scripts/artifacts.sh revision "$A" 2>&1 || echo UNAVAILABLE)
|
||||
@@ -1008,7 +1406,7 @@ jobs:
|
||||
SHORT_SHA=$(printf '%s' "$GITHUB_SHA" | cut -c1-7)
|
||||
# Mirrors build-web's tag list and its schedule handling; see
|
||||
# the comments there.
|
||||
if [ "${GITHUB_EVENT_NAME:-}" = "schedule" ]; then
|
||||
if [ "${IS_REFRESH:-}" = "true" ]; then
|
||||
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-ml:latest" >> "$GITHUB_OUTPUT"
|
||||
echo "channel=main" >> "$GITHUB_OUTPUT"
|
||||
elif [ "${GITHUB_REF##*/}" = "main" ]; then
|
||||
@@ -1091,17 +1489,63 @@ jobs:
|
||||
# A scheduled refresh has to bypass reuse by construction: it
|
||||
# rebuilds the SAME source, so fc.revision always matches and the
|
||||
# check would skip every refresh there has ever been.
|
||||
EVENT: ${{ github.event_name }}
|
||||
run: |
|
||||
set -eu
|
||||
DERIVED=$(sh scripts/artifacts.sh revision ml)
|
||||
echo "revision=$DERIVED" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# The build clock, pinned to the same commit (#3265). Without it
|
||||
# buildkit stamps the image config with the wall clock of the build,
|
||||
# so identical layers republish under a new config blob and the
|
||||
# channel tag gets a new manifest digest for no reason. Derived from
|
||||
# `newest()` like revision and version, so all three name one commit
|
||||
# and cannot drift apart.
|
||||
echo "epoch=$(sh scripts/artifacts.sh epoch ml)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# The moving tag for this channel. Which tag we ask IS the channel —
|
||||
# that is why the revision needs no -main/-dev qualifier any more.
|
||||
if [ "$CHANNEL" = "main" ]; then T=latest; else T=dev; fi
|
||||
echo "channel_ref=$IMAGE:$T" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# WHERE THE BUILD PUBLISHES, which is not always the channel — and
|
||||
# whether the channel then has to be written separately.
|
||||
#
|
||||
# On a push the build writes the channel tag directly: the bytes came
|
||||
# from a commit, and a commit is the thing CI tests. Nothing to hold
|
||||
# it behind.
|
||||
#
|
||||
# On the scheduled refresh it writes a CANDIDATE tag instead. A
|
||||
# refresh rebuilds against freshly resolved base images, and the web
|
||||
# image's runtime is a line of UNPINNED Debian packages (ffmpeg,
|
||||
# libjpeg62-turbo, libpq5, megatools…) re-resolved on every build.
|
||||
# Nothing in ci.yml can see that: its lanes run on ci-python:3.14 and
|
||||
# install requirements.txt, and a base bump changes neither. So
|
||||
# refreshed bytes have to be proven before :latest names them, and
|
||||
# proving needs a moment between "built" and "published" to occupy.
|
||||
# This is that moment; :latest goes on naming the build that works
|
||||
# until something says otherwise.
|
||||
#
|
||||
# `:refresh-candidate` is one moving ref per image, overwritten in
|
||||
# place, holding a build nobody is told to pull — the shape rule 145
|
||||
# already allows for :buildcache, not the per-build tag family that
|
||||
# milestone 318 withdrew.
|
||||
#
|
||||
# Decided HERE, beside `hit`, for the reason the force/schedule
|
||||
# branch below gives: one step decides what this job does. A
|
||||
# condition derived independently could disagree with the tag the
|
||||
# build actually wrote.
|
||||
#
|
||||
# build-web additionally exposes this as `outputs.candidate`, which is
|
||||
# what gates the `promote` job — a job's `if:` cannot read `env`, and
|
||||
# one flag is enough because all three derive it from the same
|
||||
# IS_REFRESH. ml and agent do not re-emit it; a second copy nothing
|
||||
# reads is the kind of thing that later reads as load-bearing.
|
||||
if [ "${IS_REFRESH:-}" = "true" ]; then
|
||||
echo "build_ref=$IMAGE:refresh-candidate" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "build_ref=$IMAGE:$T" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
# Compare VALUES, never exit codes. Measured on buildx v0.36.1
|
||||
# (run 4732): a missing key returns an empty string and exits 0, so
|
||||
# branching on the exit code would read "no label yet" as success.
|
||||
@@ -1134,7 +1578,7 @@ jobs:
|
||||
if [ "${FORCE:-false}" = "true" ]; then
|
||||
echo "hit=false" >> "$GITHUB_OUTPUT"
|
||||
echo "reuse: force_build set — building regardless"
|
||||
elif [ "${EVENT:-}" = "schedule" ]; then
|
||||
elif [ "${IS_REFRESH:-}" = "true" ]; then
|
||||
echo "hit=false" >> "$GITHUB_OUTPUT"
|
||||
echo "reuse: scheduled base refresh — building regardless"
|
||||
elif [ -n "$PUBLISHED" ] && [ "$PUBLISHED" = "$DERIVED" ]; then
|
||||
@@ -1147,6 +1591,12 @@ jobs:
|
||||
|
||||
- name: Build and push ml image
|
||||
if: steps.reuse.outputs.hit != 'true'
|
||||
# Read by buildx out of the ENVIRONMENT, not passed as a build-arg —
|
||||
# it normalises the image config's `created` field and the history
|
||||
# timestamps rather than being consumed by the Dockerfile. See #3265
|
||||
# and the reuse step's `epoch` output.
|
||||
env:
|
||||
SOURCE_DATE_EPOCH: ${{ steps.reuse.outputs.epoch }}
|
||||
uses: docker/build-push-action@v5
|
||||
with:
|
||||
context: .
|
||||
@@ -1159,20 +1609,17 @@ jobs:
|
||||
# invalidates, and the image genuinely rebuilds.
|
||||
#
|
||||
# MEASURED on the first real fire, run 4934 (#3265): when the base
|
||||
# did NOT move, the build is ~13s and every content step reports
|
||||
# CACHED — but the channel tag STILL gets a new manifest digest.
|
||||
# buildkit mints a fresh image config each run, so identical layers
|
||||
# are republished under a new config blob. All three images moved
|
||||
# that way on 2026-08-30 with nothing whatsoever changed in them.
|
||||
# did NOT move, the build was ~13s with every content step CACHED —
|
||||
# and the channel tag STILL got a new manifest digest, because
|
||||
# buildkit stamps a fresh image config per run and republishes the
|
||||
# identical layers under it. All three images moved that way on
|
||||
# 2026-08-30 with nothing whatsoever changed in them.
|
||||
#
|
||||
# So a refresh currently rewrites :latest every Sunday whether or
|
||||
# not there is anything new in it, and :c-<sha> is handed a new
|
||||
# manifest to diverge from on the same cadence. Layers are shared,
|
||||
# so the storage cost is a config blob; the cost that matters is
|
||||
# that a digest change no longer MEANS anything. Tracked in #3265 —
|
||||
# the likely fix is a deterministic SOURCE_DATE_EPOCH, which would
|
||||
# make "same source, same bytes" true and turn the no-op case into
|
||||
# a genuine no-op.
|
||||
# SOURCE_DATE_EPOCH (below) is the fix: pinned to the commit the
|
||||
# content came from, the config is byte-identical across runs, so
|
||||
# the manifest digest is too and the push is a registry no-op. A
|
||||
# digest change means the content changed again, which is the only
|
||||
# thing a digest is any use for.
|
||||
#
|
||||
# What `pull` does NOT catch either: a Debian package update inside
|
||||
# the `apt-get install` layer while the base tag itself stands
|
||||
@@ -1182,14 +1629,14 @@ jobs:
|
||||
# churn #3265 is about.
|
||||
#
|
||||
# Only on the schedule. An ordinary push wants the cached base.
|
||||
pull: ${{ github.event_name == 'schedule' }}
|
||||
pull: ${{ env.IS_REFRESH == 'true' }}
|
||||
# ONE tag, the channel's. Every other tag is written by the step
|
||||
# below, registry-side. buildx here pushes the first tag to the
|
||||
# registry and then re-pushes the rest through the DOCKER driver,
|
||||
# out of a local image store a registry-direct build never filled —
|
||||
# #3190, which cost `main` its :c-<sha> on 2026-08-29 while :latest
|
||||
# published perfectly well.
|
||||
tags: ${{ steps.reuse.outputs.channel_ref }}
|
||||
tags: ${{ steps.reuse.outputs.build_ref }}
|
||||
# The reuse key. Read back off the channel tag on the next push to
|
||||
# decide whether that push needs to build at all, so this is not
|
||||
# decoration — an unstamped image is one that will always rebuild.
|
||||
@@ -1331,7 +1778,7 @@ jobs:
|
||||
|
||||
# See sign-extension's copy for why this guard exists.
|
||||
- name: Guard — a scheduled run must have checked out main
|
||||
if: github.event_name == 'schedule'
|
||||
if: env.IS_REFRESH == 'true'
|
||||
run: |
|
||||
set -eu
|
||||
BRANCH=$(git rev-parse --abbrev-ref HEAD)
|
||||
@@ -1364,8 +1811,18 @@ jobs:
|
||||
# the PREVIOUS XPI while the freshly signed one is orphaned (#3156).
|
||||
# * dev and main derive the same values for the same source.
|
||||
- name: Report the derived artifact version
|
||||
env:
|
||||
# Diagnostic for the trigger normalisation. `refresh` is reported RAW
|
||||
# as well as normalised, because the two disagreeing is the whole
|
||||
# failure mode: a dispatch input whose type does not compare the way
|
||||
# the expression assumes evaluates to false silently, and the only
|
||||
# symptom is a refresh that quietly behaves like an ordinary push.
|
||||
RAW_REFRESH: ${{ github.event.inputs.refresh }}
|
||||
RAW_FORCE: ${{ github.event.inputs.force_build }}
|
||||
run: |
|
||||
set -u
|
||||
echo "trigger: event=$GITHUB_EVENT_NAME IS_REFRESH='${IS_REFRESH:-<unset>}' BUILD_REF='${BUILD_REF:-<unset>}'"
|
||||
echo "trigger: raw inputs refresh='${RAW_REFRESH:-<unset>}' force_build='${RAW_FORCE:-<unset>}'"
|
||||
A=agent
|
||||
V=$(sh scripts/artifacts.sh version "$A" 2>&1 || echo UNAVAILABLE)
|
||||
R=$(sh scripts/artifacts.sh revision "$A" 2>&1 || echo UNAVAILABLE)
|
||||
@@ -1377,7 +1834,7 @@ jobs:
|
||||
SHORT_SHA=$(printf '%s' "$GITHUB_SHA" | cut -c1-7)
|
||||
# Mirrors build-web's tag list and its schedule handling; see
|
||||
# the comments there.
|
||||
if [ "${GITHUB_EVENT_NAME:-}" = "schedule" ]; then
|
||||
if [ "${IS_REFRESH:-}" = "true" ]; then
|
||||
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-agent:latest" >> "$GITHUB_OUTPUT"
|
||||
echo "channel=main" >> "$GITHUB_OUTPUT"
|
||||
elif [ "${GITHUB_REF##*/}" = "main" ]; then
|
||||
@@ -1460,17 +1917,63 @@ jobs:
|
||||
# A scheduled refresh has to bypass reuse by construction: it
|
||||
# rebuilds the SAME source, so fc.revision always matches and the
|
||||
# check would skip every refresh there has ever been.
|
||||
EVENT: ${{ github.event_name }}
|
||||
run: |
|
||||
set -eu
|
||||
DERIVED=$(sh scripts/artifacts.sh revision agent)
|
||||
echo "revision=$DERIVED" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# The build clock, pinned to the same commit (#3265). Without it
|
||||
# buildkit stamps the image config with the wall clock of the build,
|
||||
# so identical layers republish under a new config blob and the
|
||||
# channel tag gets a new manifest digest for no reason. Derived from
|
||||
# `newest()` like revision and version, so all three name one commit
|
||||
# and cannot drift apart.
|
||||
echo "epoch=$(sh scripts/artifacts.sh epoch agent)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# The moving tag for this channel. Which tag we ask IS the channel —
|
||||
# that is why the revision needs no -main/-dev qualifier any more.
|
||||
if [ "$CHANNEL" = "main" ]; then T=latest; else T=dev; fi
|
||||
echo "channel_ref=$IMAGE:$T" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# WHERE THE BUILD PUBLISHES, which is not always the channel — and
|
||||
# whether the channel then has to be written separately.
|
||||
#
|
||||
# On a push the build writes the channel tag directly: the bytes came
|
||||
# from a commit, and a commit is the thing CI tests. Nothing to hold
|
||||
# it behind.
|
||||
#
|
||||
# On the scheduled refresh it writes a CANDIDATE tag instead. A
|
||||
# refresh rebuilds against freshly resolved base images, and the web
|
||||
# image's runtime is a line of UNPINNED Debian packages (ffmpeg,
|
||||
# libjpeg62-turbo, libpq5, megatools…) re-resolved on every build.
|
||||
# Nothing in ci.yml can see that: its lanes run on ci-python:3.14 and
|
||||
# install requirements.txt, and a base bump changes neither. So
|
||||
# refreshed bytes have to be proven before :latest names them, and
|
||||
# proving needs a moment between "built" and "published" to occupy.
|
||||
# This is that moment; :latest goes on naming the build that works
|
||||
# until something says otherwise.
|
||||
#
|
||||
# `:refresh-candidate` is one moving ref per image, overwritten in
|
||||
# place, holding a build nobody is told to pull — the shape rule 145
|
||||
# already allows for :buildcache, not the per-build tag family that
|
||||
# milestone 318 withdrew.
|
||||
#
|
||||
# Decided HERE, beside `hit`, for the reason the force/schedule
|
||||
# branch below gives: one step decides what this job does. A
|
||||
# condition derived independently could disagree with the tag the
|
||||
# build actually wrote.
|
||||
#
|
||||
# build-web additionally exposes this as `outputs.candidate`, which is
|
||||
# what gates the `promote` job — a job's `if:` cannot read `env`, and
|
||||
# one flag is enough because all three derive it from the same
|
||||
# IS_REFRESH. ml and agent do not re-emit it; a second copy nothing
|
||||
# reads is the kind of thing that later reads as load-bearing.
|
||||
if [ "${IS_REFRESH:-}" = "true" ]; then
|
||||
echo "build_ref=$IMAGE:refresh-candidate" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "build_ref=$IMAGE:$T" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
# Compare VALUES, never exit codes. Measured on buildx v0.36.1
|
||||
# (run 4732): a missing key returns an empty string and exits 0, so
|
||||
# branching on the exit code would read "no label yet" as success.
|
||||
@@ -1503,7 +2006,7 @@ jobs:
|
||||
if [ "${FORCE:-false}" = "true" ]; then
|
||||
echo "hit=false" >> "$GITHUB_OUTPUT"
|
||||
echo "reuse: force_build set — building regardless"
|
||||
elif [ "${EVENT:-}" = "schedule" ]; then
|
||||
elif [ "${IS_REFRESH:-}" = "true" ]; then
|
||||
echo "hit=false" >> "$GITHUB_OUTPUT"
|
||||
echo "reuse: scheduled base refresh — building regardless"
|
||||
elif [ -n "$PUBLISHED" ] && [ "$PUBLISHED" = "$DERIVED" ]; then
|
||||
@@ -1516,6 +2019,12 @@ jobs:
|
||||
|
||||
- name: Build and push agent image
|
||||
if: steps.reuse.outputs.hit != 'true'
|
||||
# Read by buildx out of the ENVIRONMENT, not passed as a build-arg —
|
||||
# it normalises the image config's `created` field and the history
|
||||
# timestamps rather than being consumed by the Dockerfile. See #3265
|
||||
# and the reuse step's `epoch` output.
|
||||
env:
|
||||
SOURCE_DATE_EPOCH: ${{ steps.reuse.outputs.epoch }}
|
||||
uses: docker/build-push-action@v5
|
||||
with:
|
||||
context: agent
|
||||
@@ -1528,20 +2037,17 @@ jobs:
|
||||
# invalidates, and the image genuinely rebuilds.
|
||||
#
|
||||
# MEASURED on the first real fire, run 4934 (#3265): when the base
|
||||
# did NOT move, the build is ~13s and every content step reports
|
||||
# CACHED — but the channel tag STILL gets a new manifest digest.
|
||||
# buildkit mints a fresh image config each run, so identical layers
|
||||
# are republished under a new config blob. All three images moved
|
||||
# that way on 2026-08-30 with nothing whatsoever changed in them.
|
||||
# did NOT move, the build was ~13s with every content step CACHED —
|
||||
# and the channel tag STILL got a new manifest digest, because
|
||||
# buildkit stamps a fresh image config per run and republishes the
|
||||
# identical layers under it. All three images moved that way on
|
||||
# 2026-08-30 with nothing whatsoever changed in them.
|
||||
#
|
||||
# So a refresh currently rewrites :latest every Sunday whether or
|
||||
# not there is anything new in it, and :c-<sha> is handed a new
|
||||
# manifest to diverge from on the same cadence. Layers are shared,
|
||||
# so the storage cost is a config blob; the cost that matters is
|
||||
# that a digest change no longer MEANS anything. Tracked in #3265 —
|
||||
# the likely fix is a deterministic SOURCE_DATE_EPOCH, which would
|
||||
# make "same source, same bytes" true and turn the no-op case into
|
||||
# a genuine no-op.
|
||||
# SOURCE_DATE_EPOCH (below) is the fix: pinned to the commit the
|
||||
# content came from, the config is byte-identical across runs, so
|
||||
# the manifest digest is too and the push is a registry no-op. A
|
||||
# digest change means the content changed again, which is the only
|
||||
# thing a digest is any use for.
|
||||
#
|
||||
# What `pull` does NOT catch either: a Debian package update inside
|
||||
# the `apt-get install` layer while the base tag itself stands
|
||||
@@ -1551,14 +2057,14 @@ jobs:
|
||||
# churn #3265 is about.
|
||||
#
|
||||
# Only on the schedule. An ordinary push wants the cached base.
|
||||
pull: ${{ github.event_name == 'schedule' }}
|
||||
pull: ${{ env.IS_REFRESH == 'true' }}
|
||||
# ONE tag, the channel's. Every other tag is written by the step
|
||||
# below, registry-side. buildx here pushes the first tag to the
|
||||
# registry and then re-pushes the rest through the DOCKER driver,
|
||||
# out of a local image store a registry-direct build never filled —
|
||||
# #3190, which cost `main` its :c-<sha> on 2026-08-29 while :latest
|
||||
# published perfectly well.
|
||||
tags: ${{ steps.reuse.outputs.channel_ref }}
|
||||
tags: ${{ steps.reuse.outputs.build_ref }}
|
||||
# The reuse key. Read back off the channel tag on the next push to
|
||||
# decide whether that push needs to build at all, so this is not
|
||||
# decoration — an unstamped image is one that will always rebuild.
|
||||
|
||||
@@ -255,10 +255,27 @@ jobs:
|
||||
export DB_HOST="$PG_IP"
|
||||
export CELERY_BROKER_URL="redis://$RD_IP:6379/0"
|
||||
export CELERY_RESULT_BACKEND="redis://$RD_IP:6379/0"
|
||||
# These steps run under `sh -e`, not bash, so bash's /dev/tcp magic
|
||||
# path does not exist here — the probe this loop used to run could
|
||||
# never succeed and simply burned the full 120s on every run, green
|
||||
# or red, then continued without having established anything. Python
|
||||
# is in the image and needs no installed package for a socket
|
||||
# connect, so it is the probe. Exhausting the budget is now a named
|
||||
# failure rather than a silent fall-through (rule 156): if Postgres
|
||||
# is genuinely not up, that is what the log should say, instead of
|
||||
# whatever the first query happens to raise two minutes later.
|
||||
pg_ready=""
|
||||
for i in $(seq 1 60); do
|
||||
(echo > "/dev/tcp/$PG_IP/5432") >/dev/null 2>&1 && break
|
||||
if python -c "import socket,sys; s=socket.socket(); s.settimeout(2); sys.exit(0 if s.connect_ex(('$PG_IP', 5432)) == 0 else 1)"; then
|
||||
pg_ready=1
|
||||
break
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
if [ -z "$pg_ready" ]; then
|
||||
echo "postgres at $PG_IP:5432 did not accept a connection within 120s"
|
||||
exit 1
|
||||
fi
|
||||
if command -v uv >/dev/null 2>&1; then
|
||||
uv pip install --system -r requirements.txt pytest pytest-asyncio
|
||||
else
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
# Contributing
|
||||
|
||||
FabledCurator is developed by a single maintainer for their own use, and
|
||||
published because it may be useful to others. That shapes what contribution
|
||||
looks like here.
|
||||
|
||||
**Issues are welcome** — bug reports, and questions about running it, are
|
||||
genuinely useful and often the fastest way to find out that something is
|
||||
broken outside the one environment it was built in.
|
||||
|
||||
**Open an issue before writing a pull request.** Not as a formality: the
|
||||
project has opinions that are not obvious from the code, and it is unpleasant
|
||||
for everyone when a finished patch turns out to conflict with one. A short
|
||||
issue first costs you nothing and may save you an evening.
|
||||
|
||||
**Contributions are licensed under the AGPL-3.0**, like the rest of the
|
||||
project. By submitting one you agree it ships under that licence. There is no
|
||||
CLA and no copyright assignment.
|
||||
|
||||
## Running it for development
|
||||
|
||||
```bash
|
||||
docker compose up -d # UI on http://localhost:8080
|
||||
```
|
||||
|
||||
The dev override (`docker-compose.override.yml`) is auto-merged and builds the
|
||||
app images locally from source, so this needs no `.env` and no registry
|
||||
access. Postgres and Redis ports are exposed on the host.
|
||||
|
||||
## What CI checks
|
||||
|
||||
Every push runs these, and they are the definition of done for a change:
|
||||
|
||||
```bash
|
||||
ruff check backend/ tests/ alembic/ agent/ scripts/ # lint (and import order)
|
||||
pytest tests/ -m "not integration" # backend unit tests
|
||||
pytest tests/ -m integration # needs pgvector + redis
|
||||
cd frontend && npm run test:unit && npm run build # frontend
|
||||
```
|
||||
|
||||
The integration lane builds its schema by running the real migrations
|
||||
(`alembic upgrade head`), never from ORM metadata — so a migration that does
|
||||
not apply cleanly fails CI rather than being discovered later.
|
||||
|
||||
Note for the linter: ruff's isort runs with `order-by-type`, which sorts
|
||||
ALL-CAPS names ahead of CamelCase. `from sqlalchemy import JSON, DateTime, ...`
|
||||
is correct; putting `JSON` alphabetically between `Integer` and `String` is
|
||||
not. This catches people out.
|
||||
|
||||
## Database changes
|
||||
|
||||
The ORM models and the migration chain must agree. This is enforced, and it is
|
||||
enforced because they silently diverged for a long time and nobody noticed
|
||||
until they were compared: the models were missing indexes, defaults and
|
||||
uniqueness guarantees that only ever existed inside a migration, which made
|
||||
`alembic revision --autogenerate` actively unsafe to run.
|
||||
|
||||
So: if you change a model, write the migration; if you write a migration,
|
||||
change the model to match. Both, in the same commit.
|
||||
|
||||
Adding a value to a CHECK-constrained column means swapping the constraint in
|
||||
the same change — the constraint is not documentation, and a new value without
|
||||
it fails at insert time.
|
||||
|
||||
## Branch model
|
||||
|
||||
`dev` is where work happens. `main` is production and is only reached by a
|
||||
merge from `dev`, never pushed to directly. If you are sending a pull request,
|
||||
target `dev`.
|
||||
|
||||
## Style
|
||||
|
||||
Match the surrounding code. The one convention worth stating explicitly is
|
||||
that comments here explain *why*, especially where a choice looks wrong at a
|
||||
glance — a comment recording which migration a constraint came from, or why a
|
||||
default is a `text()` rather than a string, is the kind that has repeatedly
|
||||
turned out to be worth its space.
|
||||
@@ -0,0 +1,661 @@
|
||||
GNU AFFERO GENERAL PUBLIC LICENSE
|
||||
Version 3, 19 November 2007
|
||||
|
||||
Copyright (C) 2007 Free Software Foundation, Inc. <https://fsf.org/>
|
||||
Everyone is permitted to copy and distribute verbatim copies
|
||||
of this license document, but changing it is not allowed.
|
||||
|
||||
Preamble
|
||||
|
||||
The GNU Affero General Public License is a free, copyleft license for
|
||||
software and other kinds of works, specifically designed to ensure
|
||||
cooperation with the community in the case of network server software.
|
||||
|
||||
The licenses for most software and other practical works are designed
|
||||
to take away your freedom to share and change the works. By contrast,
|
||||
our General Public Licenses are intended to guarantee your freedom to
|
||||
share and change all versions of a program--to make sure it remains free
|
||||
software for all its users.
|
||||
|
||||
When we speak of free software, we are referring to freedom, not
|
||||
price. Our General Public Licenses are designed to make sure that you
|
||||
have the freedom to distribute copies of free software (and charge for
|
||||
them if you wish), that you receive source code or can get it if you
|
||||
want it, that you can change the software or use pieces of it in new
|
||||
free programs, and that you know you can do these things.
|
||||
|
||||
Developers that use our General Public Licenses protect your rights
|
||||
with two steps: (1) assert copyright on the software, and (2) offer
|
||||
you this License which gives you legal permission to copy, distribute
|
||||
and/or modify the software.
|
||||
|
||||
A secondary benefit of defending all users' freedom is that
|
||||
improvements made in alternate versions of the program, if they
|
||||
receive widespread use, become available for other developers to
|
||||
incorporate. Many developers of free software are heartened and
|
||||
encouraged by the resulting cooperation. However, in the case of
|
||||
software used on network servers, this result may fail to come about.
|
||||
The GNU General Public License permits making a modified version and
|
||||
letting the public access it on a server without ever releasing its
|
||||
source code to the public.
|
||||
|
||||
The GNU Affero General Public License is designed specifically to
|
||||
ensure that, in such cases, the modified source code becomes available
|
||||
to the community. It requires the operator of a network server to
|
||||
provide the source code of the modified version running there to the
|
||||
users of that server. Therefore, public use of a modified version, on
|
||||
a publicly accessible server, gives the public access to the source
|
||||
code of the modified version.
|
||||
|
||||
An older license, called the Affero General Public License and
|
||||
published by Affero, was designed to accomplish similar goals. This is
|
||||
a different license, not a version of the Affero GPL, but Affero has
|
||||
released a new version of the Affero GPL which permits relicensing under
|
||||
this license.
|
||||
|
||||
The precise terms and conditions for copying, distribution and
|
||||
modification follow.
|
||||
|
||||
TERMS AND CONDITIONS
|
||||
|
||||
0. Definitions.
|
||||
|
||||
"This License" refers to version 3 of the GNU Affero General Public License.
|
||||
|
||||
"Copyright" also means copyright-like laws that apply to other kinds of
|
||||
works, such as semiconductor masks.
|
||||
|
||||
"The Program" refers to any copyrightable work licensed under this
|
||||
License. Each licensee is addressed as "you". "Licensees" and
|
||||
"recipients" may be individuals or organizations.
|
||||
|
||||
To "modify" a work means to copy from or adapt all or part of the work
|
||||
in a fashion requiring copyright permission, other than the making of an
|
||||
exact copy. The resulting work is called a "modified version" of the
|
||||
earlier work or a work "based on" the earlier work.
|
||||
|
||||
A "covered work" means either the unmodified Program or a work based
|
||||
on the Program.
|
||||
|
||||
To "propagate" a work means to do anything with it that, without
|
||||
permission, would make you directly or secondarily liable for
|
||||
infringement under applicable copyright law, except executing it on a
|
||||
computer or modifying a private copy. Propagation includes copying,
|
||||
distribution (with or without modification), making available to the
|
||||
public, and in some countries other activities as well.
|
||||
|
||||
To "convey" a work means any kind of propagation that enables other
|
||||
parties to make or receive copies. Mere interaction with a user through
|
||||
a computer network, with no transfer of a copy, is not conveying.
|
||||
|
||||
An interactive user interface displays "Appropriate Legal Notices"
|
||||
to the extent that it includes a convenient and prominently visible
|
||||
feature that (1) displays an appropriate copyright notice, and (2)
|
||||
tells the user that there is no warranty for the work (except to the
|
||||
extent that warranties are provided), that licensees may convey the
|
||||
work under this License, and how to view a copy of this License. If
|
||||
the interface presents a list of user commands or options, such as a
|
||||
menu, a prominent item in the list meets this criterion.
|
||||
|
||||
1. Source Code.
|
||||
|
||||
The "source code" for a work means the preferred form of the work
|
||||
for making modifications to it. "Object code" means any non-source
|
||||
form of a work.
|
||||
|
||||
A "Standard Interface" means an interface that either is an official
|
||||
standard defined by a recognized standards body, or, in the case of
|
||||
interfaces specified for a particular programming language, one that
|
||||
is widely used among developers working in that language.
|
||||
|
||||
The "System Libraries" of an executable work include anything, other
|
||||
than the work as a whole, that (a) is included in the normal form of
|
||||
packaging a Major Component, but which is not part of that Major
|
||||
Component, and (b) serves only to enable use of the work with that
|
||||
Major Component, or to implement a Standard Interface for which an
|
||||
implementation is available to the public in source code form. A
|
||||
"Major Component", in this context, means a major essential component
|
||||
(kernel, window system, and so on) of the specific operating system
|
||||
(if any) on which the executable work runs, or a compiler used to
|
||||
produce the work, or an object code interpreter used to run it.
|
||||
|
||||
The "Corresponding Source" for a work in object code form means all
|
||||
the source code needed to generate, install, and (for an executable
|
||||
work) run the object code and to modify the work, including scripts to
|
||||
control those activities. However, it does not include the work's
|
||||
System Libraries, or general-purpose tools or generally available free
|
||||
programs which are used unmodified in performing those activities but
|
||||
which are not part of the work. For example, Corresponding Source
|
||||
includes interface definition files associated with source files for
|
||||
the work, and the source code for shared libraries and dynamically
|
||||
linked subprograms that the work is specifically designed to require,
|
||||
such as by intimate data communication or control flow between those
|
||||
subprograms and other parts of the work.
|
||||
|
||||
The Corresponding Source need not include anything that users
|
||||
can regenerate automatically from other parts of the Corresponding
|
||||
Source.
|
||||
|
||||
The Corresponding Source for a work in source code form is that
|
||||
same work.
|
||||
|
||||
2. Basic Permissions.
|
||||
|
||||
All rights granted under this License are granted for the term of
|
||||
copyright on the Program, and are irrevocable provided the stated
|
||||
conditions are met. This License explicitly affirms your unlimited
|
||||
permission to run the unmodified Program. The output from running a
|
||||
covered work is covered by this License only if the output, given its
|
||||
content, constitutes a covered work. This License acknowledges your
|
||||
rights of fair use or other equivalent, as provided by copyright law.
|
||||
|
||||
You may make, run and propagate covered works that you do not
|
||||
convey, without conditions so long as your license otherwise remains
|
||||
in force. You may convey covered works to others for the sole purpose
|
||||
of having them make modifications exclusively for you, or provide you
|
||||
with facilities for running those works, provided that you comply with
|
||||
the terms of this License in conveying all material for which you do
|
||||
not control copyright. Those thus making or running the covered works
|
||||
for you must do so exclusively on your behalf, under your direction
|
||||
and control, on terms that prohibit them from making any copies of
|
||||
your copyrighted material outside their relationship with you.
|
||||
|
||||
Conveying under any other circumstances is permitted solely under
|
||||
the conditions stated below. Sublicensing is not allowed; section 10
|
||||
makes it unnecessary.
|
||||
|
||||
3. Protecting Users' Legal Rights From Anti-Circumvention Law.
|
||||
|
||||
No covered work shall be deemed part of an effective technological
|
||||
measure under any applicable law fulfilling obligations under article
|
||||
11 of the WIPO copyright treaty adopted on 20 December 1996, or
|
||||
similar laws prohibiting or restricting circumvention of such
|
||||
measures.
|
||||
|
||||
When you convey a covered work, you waive any legal power to forbid
|
||||
circumvention of technological measures to the extent such circumvention
|
||||
is effected by exercising rights under this License with respect to
|
||||
the covered work, and you disclaim any intention to limit operation or
|
||||
modification of the work as a means of enforcing, against the work's
|
||||
users, your or third parties' legal rights to forbid circumvention of
|
||||
technological measures.
|
||||
|
||||
4. Conveying Verbatim Copies.
|
||||
|
||||
You may convey verbatim copies of the Program's source code as you
|
||||
receive it, in any medium, provided that you conspicuously and
|
||||
appropriately publish on each copy an appropriate copyright notice;
|
||||
keep intact all notices stating that this License and any
|
||||
non-permissive terms added in accord with section 7 apply to the code;
|
||||
keep intact all notices of the absence of any warranty; and give all
|
||||
recipients a copy of this License along with the Program.
|
||||
|
||||
You may charge any price or no price for each copy that you convey,
|
||||
and you may offer support or warranty protection for a fee.
|
||||
|
||||
5. Conveying Modified Source Versions.
|
||||
|
||||
You may convey a work based on the Program, or the modifications to
|
||||
produce it from the Program, in the form of source code under the
|
||||
terms of section 4, provided that you also meet all of these conditions:
|
||||
|
||||
a) The work must carry prominent notices stating that you modified
|
||||
it, and giving a relevant date.
|
||||
|
||||
b) The work must carry prominent notices stating that it is
|
||||
released under this License and any conditions added under section
|
||||
7. This requirement modifies the requirement in section 4 to
|
||||
"keep intact all notices".
|
||||
|
||||
c) You must license the entire work, as a whole, under this
|
||||
License to anyone who comes into possession of a copy. This
|
||||
License will therefore apply, along with any applicable section 7
|
||||
additional terms, to the whole of the work, and all its parts,
|
||||
regardless of how they are packaged. This License gives no
|
||||
permission to license the work in any other way, but it does not
|
||||
invalidate such permission if you have separately received it.
|
||||
|
||||
d) If the work has interactive user interfaces, each must display
|
||||
Appropriate Legal Notices; however, if the Program has interactive
|
||||
interfaces that do not display Appropriate Legal Notices, your
|
||||
work need not make them do so.
|
||||
|
||||
A compilation of a covered work with other separate and independent
|
||||
works, which are not by their nature extensions of the covered work,
|
||||
and which are not combined with it such as to form a larger program,
|
||||
in or on a volume of a storage or distribution medium, is called an
|
||||
"aggregate" if the compilation and its resulting copyright are not
|
||||
used to limit the access or legal rights of the compilation's users
|
||||
beyond what the individual works permit. Inclusion of a covered work
|
||||
in an aggregate does not cause this License to apply to the other
|
||||
parts of the aggregate.
|
||||
|
||||
6. Conveying Non-Source Forms.
|
||||
|
||||
You may convey a covered work in object code form under the terms
|
||||
of sections 4 and 5, provided that you also convey the
|
||||
machine-readable Corresponding Source under the terms of this License,
|
||||
in one of these ways:
|
||||
|
||||
a) Convey the object code in, or embodied in, a physical product
|
||||
(including a physical distribution medium), accompanied by the
|
||||
Corresponding Source fixed on a durable physical medium
|
||||
customarily used for software interchange.
|
||||
|
||||
b) Convey the object code in, or embodied in, a physical product
|
||||
(including a physical distribution medium), accompanied by a
|
||||
written offer, valid for at least three years and valid for as
|
||||
long as you offer spare parts or customer support for that product
|
||||
model, to give anyone who possesses the object code either (1) a
|
||||
copy of the Corresponding Source for all the software in the
|
||||
product that is covered by this License, on a durable physical
|
||||
medium customarily used for software interchange, for a price no
|
||||
more than your reasonable cost of physically performing this
|
||||
conveying of source, or (2) access to copy the
|
||||
Corresponding Source from a network server at no charge.
|
||||
|
||||
c) Convey individual copies of the object code with a copy of the
|
||||
written offer to provide the Corresponding Source. This
|
||||
alternative is allowed only occasionally and noncommercially, and
|
||||
only if you received the object code with such an offer, in accord
|
||||
with subsection 6b.
|
||||
|
||||
d) Convey the object code by offering access from a designated
|
||||
place (gratis or for a charge), and offer equivalent access to the
|
||||
Corresponding Source in the same way through the same place at no
|
||||
further charge. You need not require recipients to copy the
|
||||
Corresponding Source along with the object code. If the place to
|
||||
copy the object code is a network server, the Corresponding Source
|
||||
may be on a different server (operated by you or a third party)
|
||||
that supports equivalent copying facilities, provided you maintain
|
||||
clear directions next to the object code saying where to find the
|
||||
Corresponding Source. Regardless of what server hosts the
|
||||
Corresponding Source, you remain obligated to ensure that it is
|
||||
available for as long as needed to satisfy these requirements.
|
||||
|
||||
e) Convey the object code using peer-to-peer transmission, provided
|
||||
you inform other peers where the object code and Corresponding
|
||||
Source of the work are being offered to the general public at no
|
||||
charge under subsection 6d.
|
||||
|
||||
A separable portion of the object code, whose source code is excluded
|
||||
from the Corresponding Source as a System Library, need not be
|
||||
included in conveying the object code work.
|
||||
|
||||
A "User Product" is either (1) a "consumer product", which means any
|
||||
tangible personal property which is normally used for personal, family,
|
||||
or household purposes, or (2) anything designed or sold for incorporation
|
||||
into a dwelling. In determining whether a product is a consumer product,
|
||||
doubtful cases shall be resolved in favor of coverage. For a particular
|
||||
product received by a particular user, "normally used" refers to a
|
||||
typical or common use of that class of product, regardless of the status
|
||||
of the particular user or of the way in which the particular user
|
||||
actually uses, or expects or is expected to use, the product. A product
|
||||
is a consumer product regardless of whether the product has substantial
|
||||
commercial, industrial or non-consumer uses, unless such uses represent
|
||||
the only significant mode of use of the product.
|
||||
|
||||
"Installation Information" for a User Product means any methods,
|
||||
procedures, authorization keys, or other information required to install
|
||||
and execute modified versions of a covered work in that User Product from
|
||||
a modified version of its Corresponding Source. The information must
|
||||
suffice to ensure that the continued functioning of the modified object
|
||||
code is in no case prevented or interfered with solely because
|
||||
modification has been made.
|
||||
|
||||
If you convey an object code work under this section in, or with, or
|
||||
specifically for use in, a User Product, and the conveying occurs as
|
||||
part of a transaction in which the right of possession and use of the
|
||||
User Product is transferred to the recipient in perpetuity or for a
|
||||
fixed term (regardless of how the transaction is characterized), the
|
||||
Corresponding Source conveyed under this section must be accompanied
|
||||
by the Installation Information. But this requirement does not apply
|
||||
if neither you nor any third party retains the ability to install
|
||||
modified object code on the User Product (for example, the work has
|
||||
been installed in ROM).
|
||||
|
||||
The requirement to provide Installation Information does not include a
|
||||
requirement to continue to provide support service, warranty, or updates
|
||||
for a work that has been modified or installed by the recipient, or for
|
||||
the User Product in which it has been modified or installed. Access to a
|
||||
network may be denied when the modification itself materially and
|
||||
adversely affects the operation of the network or violates the rules and
|
||||
protocols for communication across the network.
|
||||
|
||||
Corresponding Source conveyed, and Installation Information provided,
|
||||
in accord with this section must be in a format that is publicly
|
||||
documented (and with an implementation available to the public in
|
||||
source code form), and must require no special password or key for
|
||||
unpacking, reading or copying.
|
||||
|
||||
7. Additional Terms.
|
||||
|
||||
"Additional permissions" are terms that supplement the terms of this
|
||||
License by making exceptions from one or more of its conditions.
|
||||
Additional permissions that are applicable to the entire Program shall
|
||||
be treated as though they were included in this License, to the extent
|
||||
that they are valid under applicable law. If additional permissions
|
||||
apply only to part of the Program, that part may be used separately
|
||||
under those permissions, but the entire Program remains governed by
|
||||
this License without regard to the additional permissions.
|
||||
|
||||
When you convey a copy of a covered work, you may at your option
|
||||
remove any additional permissions from that copy, or from any part of
|
||||
it. (Additional permissions may be written to require their own
|
||||
removal in certain cases when you modify the work.) You may place
|
||||
additional permissions on material, added by you to a covered work,
|
||||
for which you have or can give appropriate copyright permission.
|
||||
|
||||
Notwithstanding any other provision of this License, for material you
|
||||
add to a covered work, you may (if authorized by the copyright holders of
|
||||
that material) supplement the terms of this License with terms:
|
||||
|
||||
a) Disclaiming warranty or limiting liability differently from the
|
||||
terms of sections 15 and 16 of this License; or
|
||||
|
||||
b) Requiring preservation of specified reasonable legal notices or
|
||||
author attributions in that material or in the Appropriate Legal
|
||||
Notices displayed by works containing it; or
|
||||
|
||||
c) Prohibiting misrepresentation of the origin of that material, or
|
||||
requiring that modified versions of such material be marked in
|
||||
reasonable ways as different from the original version; or
|
||||
|
||||
d) Limiting the use for publicity purposes of names of licensors or
|
||||
authors of the material; or
|
||||
|
||||
e) Declining to grant rights under trademark law for use of some
|
||||
trade names, trademarks, or service marks; or
|
||||
|
||||
f) Requiring indemnification of licensors and authors of that
|
||||
material by anyone who conveys the material (or modified versions of
|
||||
it) with contractual assumptions of liability to the recipient, for
|
||||
any liability that these contractual assumptions directly impose on
|
||||
those licensors and authors.
|
||||
|
||||
All other non-permissive additional terms are considered "further
|
||||
restrictions" within the meaning of section 10. If the Program as you
|
||||
received it, or any part of it, contains a notice stating that it is
|
||||
governed by this License along with a term that is a further
|
||||
restriction, you may remove that term. If a license document contains
|
||||
a further restriction but permits relicensing or conveying under this
|
||||
License, you may add to a covered work material governed by the terms
|
||||
of that license document, provided that the further restriction does
|
||||
not survive such relicensing or conveying.
|
||||
|
||||
If you add terms to a covered work in accord with this section, you
|
||||
must place, in the relevant source files, a statement of the
|
||||
additional terms that apply to those files, or a notice indicating
|
||||
where to find the applicable terms.
|
||||
|
||||
Additional terms, permissive or non-permissive, may be stated in the
|
||||
form of a separately written license, or stated as exceptions;
|
||||
the above requirements apply either way.
|
||||
|
||||
8. Termination.
|
||||
|
||||
You may not propagate or modify a covered work except as expressly
|
||||
provided under this License. Any attempt otherwise to propagate or
|
||||
modify it is void, and will automatically terminate your rights under
|
||||
this License (including any patent licenses granted under the third
|
||||
paragraph of section 11).
|
||||
|
||||
However, if you cease all violation of this License, then your
|
||||
license from a particular copyright holder is reinstated (a)
|
||||
provisionally, unless and until the copyright holder explicitly and
|
||||
finally terminates your license, and (b) permanently, if the copyright
|
||||
holder fails to notify you of the violation by some reasonable means
|
||||
prior to 60 days after the cessation.
|
||||
|
||||
Moreover, your license from a particular copyright holder is
|
||||
reinstated permanently if the copyright holder notifies you of the
|
||||
violation by some reasonable means, this is the first time you have
|
||||
received notice of violation of this License (for any work) from that
|
||||
copyright holder, and you cure the violation prior to 30 days after
|
||||
your receipt of the notice.
|
||||
|
||||
Termination of your rights under this section does not terminate the
|
||||
licenses of parties who have received copies or rights from you under
|
||||
this License. If your rights have been terminated and not permanently
|
||||
reinstated, you do not qualify to receive new licenses for the same
|
||||
material under section 10.
|
||||
|
||||
9. Acceptance Not Required for Having Copies.
|
||||
|
||||
You are not required to accept this License in order to receive or
|
||||
run a copy of the Program. Ancillary propagation of a covered work
|
||||
occurring solely as a consequence of using peer-to-peer transmission
|
||||
to receive a copy likewise does not require acceptance. However,
|
||||
nothing other than this License grants you permission to propagate or
|
||||
modify any covered work. These actions infringe copyright if you do
|
||||
not accept this License. Therefore, by modifying or propagating a
|
||||
covered work, you indicate your acceptance of this License to do so.
|
||||
|
||||
10. Automatic Licensing of Downstream Recipients.
|
||||
|
||||
Each time you convey a covered work, the recipient automatically
|
||||
receives a license from the original licensors, to run, modify and
|
||||
propagate that work, subject to this License. You are not responsible
|
||||
for enforcing compliance by third parties with this License.
|
||||
|
||||
An "entity transaction" is a transaction transferring control of an
|
||||
organization, or substantially all assets of one, or subdividing an
|
||||
organization, or merging organizations. If propagation of a covered
|
||||
work results from an entity transaction, each party to that
|
||||
transaction who receives a copy of the work also receives whatever
|
||||
licenses to the work the party's predecessor in interest had or could
|
||||
give under the previous paragraph, plus a right to possession of the
|
||||
Corresponding Source of the work from the predecessor in interest, if
|
||||
the predecessor has it or can get it with reasonable efforts.
|
||||
|
||||
You may not impose any further restrictions on the exercise of the
|
||||
rights granted or affirmed under this License. For example, you may
|
||||
not impose a license fee, royalty, or other charge for exercise of
|
||||
rights granted under this License, and you may not initiate litigation
|
||||
(including a cross-claim or counterclaim in a lawsuit) alleging that
|
||||
any patent claim is infringed by making, using, selling, offering for
|
||||
sale, or importing the Program or any portion of it.
|
||||
|
||||
11. Patents.
|
||||
|
||||
A "contributor" is a copyright holder who authorizes use under this
|
||||
License of the Program or a work on which the Program is based. The
|
||||
work thus licensed is called the contributor's "contributor version".
|
||||
|
||||
A contributor's "essential patent claims" are all patent claims
|
||||
owned or controlled by the contributor, whether already acquired or
|
||||
hereafter acquired, that would be infringed by some manner, permitted
|
||||
by this License, of making, using, or selling its contributor version,
|
||||
but do not include claims that would be infringed only as a
|
||||
consequence of further modification of the contributor version. For
|
||||
purposes of this definition, "control" includes the right to grant
|
||||
patent sublicenses in a manner consistent with the requirements of
|
||||
this License.
|
||||
|
||||
Each contributor grants you a non-exclusive, worldwide, royalty-free
|
||||
patent license under the contributor's essential patent claims, to
|
||||
make, use, sell, offer for sale, import and otherwise run, modify and
|
||||
propagate the contents of its contributor version.
|
||||
|
||||
In the following three paragraphs, a "patent license" is any express
|
||||
agreement or commitment, however denominated, not to enforce a patent
|
||||
(such as an express permission to practice a patent or covenant not to
|
||||
sue for patent infringement). To "grant" such a patent license to a
|
||||
party means to make such an agreement or commitment not to enforce a
|
||||
patent against the party.
|
||||
|
||||
If you convey a covered work, knowingly relying on a patent license,
|
||||
and the Corresponding Source of the work is not available for anyone
|
||||
to copy, free of charge and under the terms of this License, through a
|
||||
publicly available network server or other readily accessible means,
|
||||
then you must either (1) cause the Corresponding Source to be so
|
||||
available, or (2) arrange to deprive yourself of the benefit of the
|
||||
patent license for this particular work, or (3) arrange, in a manner
|
||||
consistent with the requirements of this License, to extend the patent
|
||||
license to downstream recipients. "Knowingly relying" means you have
|
||||
actual knowledge that, but for the patent license, your conveying the
|
||||
covered work in a country, or your recipient's use of the covered work
|
||||
in a country, would infringe one or more identifiable patents in that
|
||||
country that you have reason to believe are valid.
|
||||
|
||||
If, pursuant to or in connection with a single transaction or
|
||||
arrangement, you convey, or propagate by procuring conveyance of, a
|
||||
covered work, and grant a patent license to some of the parties
|
||||
receiving the covered work authorizing them to use, propagate, modify
|
||||
or convey a specific copy of the covered work, then the patent license
|
||||
you grant is automatically extended to all recipients of the covered
|
||||
work and works based on it.
|
||||
|
||||
A patent license is "discriminatory" if it does not include within
|
||||
the scope of its coverage, prohibits the exercise of, or is
|
||||
conditioned on the non-exercise of one or more of the rights that are
|
||||
specifically granted under this License. You may not convey a covered
|
||||
work if you are a party to an arrangement with a third party that is
|
||||
in the business of distributing software, under which you make payment
|
||||
to the third party based on the extent of your activity of conveying
|
||||
the work, and under which the third party grants, to any of the
|
||||
parties who would receive the covered work from you, a discriminatory
|
||||
patent license (a) in connection with copies of the covered work
|
||||
conveyed by you (or copies made from those copies), or (b) primarily
|
||||
for and in connection with specific products or compilations that
|
||||
contain the covered work, unless you entered into that arrangement,
|
||||
or that patent license was granted, prior to 28 March 2007.
|
||||
|
||||
Nothing in this License shall be construed as excluding or limiting
|
||||
any implied license or other defenses to infringement that may
|
||||
otherwise be available to you under applicable patent law.
|
||||
|
||||
12. No Surrender of Others' Freedom.
|
||||
|
||||
If conditions are imposed on you (whether by court order, agreement or
|
||||
otherwise) that contradict the conditions of this License, they do not
|
||||
excuse you from the conditions of this License. If you cannot convey a
|
||||
covered work so as to satisfy simultaneously your obligations under this
|
||||
License and any other pertinent obligations, then as a consequence you may
|
||||
not convey it at all. For example, if you agree to terms that obligate you
|
||||
to collect a royalty for further conveying from those to whom you convey
|
||||
the Program, the only way you could satisfy both those terms and this
|
||||
License would be to refrain entirely from conveying the Program.
|
||||
|
||||
13. Remote Network Interaction; Use with the GNU General Public License.
|
||||
|
||||
Notwithstanding any other provision of this License, if you modify the
|
||||
Program, your modified version must prominently offer all users
|
||||
interacting with it remotely through a computer network (if your version
|
||||
supports such interaction) an opportunity to receive the Corresponding
|
||||
Source of your version by providing access to the Corresponding Source
|
||||
from a network server at no charge, through some standard or customary
|
||||
means of facilitating copying of software. This Corresponding Source
|
||||
shall include the Corresponding Source for any work covered by version 3
|
||||
of the GNU General Public License that is incorporated pursuant to the
|
||||
following paragraph.
|
||||
|
||||
Notwithstanding any other provision of this License, you have
|
||||
permission to link or combine any covered work with a work licensed
|
||||
under version 3 of the GNU General Public License into a single
|
||||
combined work, and to convey the resulting work. The terms of this
|
||||
License will continue to apply to the part which is the covered work,
|
||||
but the work with which it is combined will remain governed by version
|
||||
3 of the GNU General Public License.
|
||||
|
||||
14. Revised Versions of this License.
|
||||
|
||||
The Free Software Foundation may publish revised and/or new versions of
|
||||
the GNU Affero General Public License from time to time. Such new versions
|
||||
will be similar in spirit to the present version, but may differ in detail to
|
||||
address new problems or concerns.
|
||||
|
||||
Each version is given a distinguishing version number. If the
|
||||
Program specifies that a certain numbered version of the GNU Affero General
|
||||
Public License "or any later version" applies to it, you have the
|
||||
option of following the terms and conditions either of that numbered
|
||||
version or of any later version published by the Free Software
|
||||
Foundation. If the Program does not specify a version number of the
|
||||
GNU Affero General Public License, you may choose any version ever published
|
||||
by the Free Software Foundation.
|
||||
|
||||
If the Program specifies that a proxy can decide which future
|
||||
versions of the GNU Affero General Public License can be used, that proxy's
|
||||
public statement of acceptance of a version permanently authorizes you
|
||||
to choose that version for the Program.
|
||||
|
||||
Later license versions may give you additional or different
|
||||
permissions. However, no additional obligations are imposed on any
|
||||
author or copyright holder as a result of your choosing to follow a
|
||||
later version.
|
||||
|
||||
15. Disclaimer of Warranty.
|
||||
|
||||
THERE IS NO WARRANTY FOR THE PROGRAM, TO THE EXTENT PERMITTED BY
|
||||
APPLICABLE LAW. EXCEPT WHEN OTHERWISE STATED IN WRITING THE COPYRIGHT
|
||||
HOLDERS AND/OR OTHER PARTIES PROVIDE THE PROGRAM "AS IS" WITHOUT WARRANTY
|
||||
OF ANY KIND, EITHER EXPRESSED OR IMPLIED, INCLUDING, BUT NOT LIMITED TO,
|
||||
THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
PURPOSE. THE ENTIRE RISK AS TO THE QUALITY AND PERFORMANCE OF THE PROGRAM
|
||||
IS WITH YOU. SHOULD THE PROGRAM PROVE DEFECTIVE, YOU ASSUME THE COST OF
|
||||
ALL NECESSARY SERVICING, REPAIR OR CORRECTION.
|
||||
|
||||
16. Limitation of Liability.
|
||||
|
||||
IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING
|
||||
WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MODIFIES AND/OR CONVEYS
|
||||
THE PROGRAM AS PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY
|
||||
GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING OUT OF THE
|
||||
USE OR INABILITY TO USE THE PROGRAM (INCLUDING BUT NOT LIMITED TO LOSS OF
|
||||
DATA OR DATA BEING RENDERED INACCURATE OR LOSSES SUSTAINED BY YOU OR THIRD
|
||||
PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE WITH ANY OTHER PROGRAMS),
|
||||
EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE POSSIBILITY OF
|
||||
SUCH DAMAGES.
|
||||
|
||||
17. Interpretation of Sections 15 and 16.
|
||||
|
||||
If the disclaimer of warranty and limitation of liability provided
|
||||
above cannot be given local legal effect according to their terms,
|
||||
reviewing courts shall apply local law that most closely approximates
|
||||
an absolute waiver of all civil liability in connection with the
|
||||
Program, unless a warranty or assumption of liability accompanies a
|
||||
copy of the Program in return for a fee.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
How to Apply These Terms to Your New Programs
|
||||
|
||||
If you develop a new program, and you want it to be of the greatest
|
||||
possible use to the public, the best way to achieve this is to make it
|
||||
free software which everyone can redistribute and change under these terms.
|
||||
|
||||
To do so, attach the following notices to the program. It is safest
|
||||
to attach them to the start of each source file to most effectively
|
||||
state the exclusion of warranty; and each file should have at least
|
||||
the "copyright" line and a pointer to where the full notice is found.
|
||||
|
||||
<one line to give the program's name and a brief idea of what it does.>
|
||||
Copyright (C) <year> <name of author>
|
||||
|
||||
This program is free software: you can redistribute it and/or modify
|
||||
it under the terms of the GNU Affero General Public License as published by
|
||||
the Free Software Foundation, either version 3 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU Affero General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU Affero General Public License
|
||||
along with this program. If not, see <https://www.gnu.org/licenses/>.
|
||||
|
||||
Also add information on how to contact you by electronic and paper mail.
|
||||
|
||||
If your software can interact with users remotely through a computer
|
||||
network, you should also make sure that it provides a way for users to
|
||||
get its source. For example, if your program is a web application, its
|
||||
interface could display a "Source" link that leads users to an archive
|
||||
of the code. There are many ways you could offer source, and different
|
||||
solutions will be better for different programs; see section 13 for the
|
||||
specific requirements.
|
||||
|
||||
You should also get your employer (if you work as a programmer) or school,
|
||||
if any, to sign a "copyright disclaimer" for the program, if necessary.
|
||||
For more information on this, and how to apply and follow the GNU AGPL, see
|
||||
<https://www.gnu.org/licenses/>.
|
||||
@@ -1,14 +1,230 @@
|
||||
# FabledCurator
|
||||
|
||||
Self-hosted media curation — gallery, ML tagging, and subscription-driven downloading in one app. Part of the FabledSword family.
|
||||
<!-- overview:start -->
|
||||
Self-hosted media curation — a gallery, ML auto-tagging, and subscription-driven
|
||||
downloading in one application. Part of the FabledSword family.
|
||||
|
||||
Combines what was [ImageRepo](https://git.fabledsword.com/bvandeusen/ImageRepo) (gallery, ML, importer) and [GallerySubscriber](https://git.fabledsword.com/bvandeusen/GallerySubscriber) (gallery-dl wrapper, subscriptions, credential capture) into a single product.
|
||||
## What it does
|
||||
|
||||
## Status
|
||||
You point it at creators you follow. It downloads what they post, files it,
|
||||
tags it, and gives you something better than a folder full of images to look
|
||||
through afterwards.
|
||||
|
||||
In production. `main` is continuously deployed — every merge to `main` builds
|
||||
and publishes `:latest` images, so whatever is on `main` is what is running.
|
||||
Day-to-day work happens on `dev`, which publishes `:dev` images.
|
||||
- **Gallery and browsing.** Images, videos and multi-page works, organised by
|
||||
artist, tag, post and series. A Showcase front page, a filterable gallery, a
|
||||
similarity-driven Explore view, and a page-turning reader for series.
|
||||
- **Subscriptions.** Follows creators on Patreon, SubscribeStar, Pixiv and
|
||||
anything `gallery-dl` supports, on a schedule. Handles paywalled posts using
|
||||
your own logged-in session.
|
||||
- **ML tagging.** Runs image models in-container to suggest tags, group
|
||||
characters, find near-duplicates and power similarity search. Suggestions are
|
||||
reviewable — it proposes, you confirm, and it learns which proposals you keep
|
||||
rejecting.
|
||||
- **Deduplication and provenance.** Everything that arrives is hashed and
|
||||
deduplicated by content, metadata sidecars are read wherever the source
|
||||
writes them, and every file keeps a record of where it came from.
|
||||
- **Maintenance.** Backups, library audits, thumbnail and embedding backfills,
|
||||
orphan cleanup — all from the UI, all as background jobs you can watch.
|
||||
|
||||
Everything is configured from the Settings UI and stored in the database. There
|
||||
is no config file to edit beyond a handful of bootstrap environment variables.
|
||||
<!-- overview:end -->
|
||||
|
||||
## Before you expose it
|
||||
|
||||
**FabledCurator has no login.** There are no user accounts, no passwords and no
|
||||
permission model. Anything that can reach the port is an administrator.
|
||||
|
||||
That matters more here than it would in most self-hosted apps, because of what
|
||||
this one stores: **live platform session cookies for Patreon, SubscribeStar and
|
||||
Pixiv** — accounts that usually have a payment method attached. Whoever reaches
|
||||
the port can read them, alongside your entire library.
|
||||
|
||||
So:
|
||||
|
||||
- Bind it to a LAN, a VPN, or a tunnel you control.
|
||||
- Do not port-forward it. Do not put it on a public hostname.
|
||||
- A reverse proxy that adds TLS but no authentication **does not help**. If you
|
||||
want it reachable from outside, put an authenticating proxy in front of it —
|
||||
a forward-auth provider, HTTP basic auth, an identity-aware tunnel — and treat
|
||||
that layer as the only thing standing between the internet and your accounts.
|
||||
|
||||
This is a deliberate design decision for a single-operator tool on a trusted
|
||||
network, not a bug and not an oversight. It is stated here because it decides
|
||||
how you are allowed to deploy it. [SECURITY.md](SECURITY.md) covers the rest of
|
||||
the threat model.
|
||||
|
||||
## Requirements
|
||||
|
||||
- **Docker** with Compose v2.
|
||||
- **~4 GB RAM** for the app, plus whatever Postgres needs for your library size.
|
||||
- **Disk** for your media, plus several GB for ML model weights.
|
||||
- **No GPU required.** The ML worker runs on CPU — tagging and embedding are
|
||||
slower, and that is the whole difference. A GPU is only involved if you
|
||||
separately run the optional agent (below), which is a different machine's job.
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
git clone https://git.fabledsword.com/bvandeusen/FabledCurator.git
|
||||
cd FabledCurator
|
||||
|
||||
cp .env.example .env
|
||||
$EDITOR .env # set DB_PASSWORD and SECRET_KEY
|
||||
|
||||
docker compose -f docker-compose.yml up -d
|
||||
```
|
||||
|
||||
Then open <http://localhost:8080>.
|
||||
|
||||
**The `-f docker-compose.yml` is required, not decoration.** Compose
|
||||
auto-merges `docker-compose.override.yml` when you leave it off, and that
|
||||
override builds the images locally from source — the contributor path, not
|
||||
yours. Naming the file explicitly skips the override and pulls the published
|
||||
`:latest` images, which is the stable channel built from `main`.
|
||||
|
||||
If you forget it, the symptom is a long build instead of a quick pull.
|
||||
|
||||
## First run
|
||||
|
||||
The database schema is created automatically on first start — the web container
|
||||
runs its migrations before serving. Nothing to initialise by hand.
|
||||
|
||||
**One thing does need a deliberate act, and the app will not start without it.**
|
||||
FabledCurator encrypts your stored platform credentials with a key it keeps at
|
||||
`./images/secrets/credential_key.b64`. On a brand-new install that file does not
|
||||
exist, and rather than quietly creating one the app stops:
|
||||
|
||||
```
|
||||
MissingCredentialKey: Fernet key file not found at /images/secrets/credential_key.b64
|
||||
```
|
||||
|
||||
Set `CURATOR_BOOTSTRAP_NEW_KEY=1` in your `.env` for the first `up`, then delete
|
||||
the line once the container is running. `.env.example` ships it with that
|
||||
instruction attached.
|
||||
|
||||
The refusal is deliberate, and worth understanding rather than working around:
|
||||
auto-creating a key is indistinguishable from the disaster case — a restore that
|
||||
brought the database back but lost `./images/secrets` — where it would mint a key
|
||||
that cannot decrypt anything, leaving an instance that looks healthy while every
|
||||
paywalled download fails. Making you say so once, on an empty install, is the
|
||||
price of that not happening silently later.
|
||||
|
||||
**Which means: back up `./images/secrets/` alongside your database.** It is the
|
||||
only thing that can read your stored credentials. A database restored without it
|
||||
needs every credential entered again by hand.
|
||||
|
||||
A few other things are worth knowing about the first few minutes:
|
||||
|
||||
- **The ML worker downloads its model weights on first boot**, several GB from
|
||||
HuggingFace into `./models`. Until that finishes, tagging is queued rather
|
||||
than broken. It is idempotent — a restart resumes rather than refetches.
|
||||
- **The gallery starts empty**, and that is the expected state. Add a creator
|
||||
under **Subscriptions** and it fills as posts come down.
|
||||
- **If you already have a library on disk**, there is no screen that imports
|
||||
it, and there is not going to be one. Folder ingestion had a UI until July
|
||||
2026; it was retired once posts began arriving entirely through
|
||||
subscriptions and the browser extension, and the decision to leave it
|
||||
retired is deliberate — the folder path carries complexity the product does
|
||||
not need in order to do its job. The supported way to fill a new install is
|
||||
to add the creators you follow under **Subscriptions** and let it pull.
|
||||
|
||||
The `/api/import/trigger` endpoint is still wired up for anyone who wants to
|
||||
script a one-off against a folder mounted at `./import`, and its progress
|
||||
shows under **Settings → Activity**. Treat it as an unsupported escape
|
||||
hatch rather than a feature: nothing in the UI drives it and nothing else
|
||||
in this README depends on it.
|
||||
- **To download from a paywalled account**, FabledCurator needs that account's
|
||||
session — see the browser extension below. Without one it can still fetch
|
||||
public posts.
|
||||
- **Check Settings → Overview** to confirm the workers are alive. Every long
|
||||
operation in FabledCurator is a background job, so if the queues are not
|
||||
running, the UI will look like it is ignoring you rather than like it is
|
||||
broken.
|
||||
|
||||
## The browser extension
|
||||
|
||||
A Firefox extension does two jobs: it hands your logged-in platform sessions to
|
||||
FabledCurator so it can download on your behalf, and it adds a creator as a
|
||||
subscription in one click from their page.
|
||||
|
||||
It ships **inside the web image** — there is no add-on store listing to find.
|
||||
Go to **Subscriptions → Settings**, find the *Browser extension* card, and click
|
||||
**Install Firefox extension**. The XPI is Mozilla-signed, so Firefox installs it
|
||||
like any other add-on; the button serves it directly rather than making you
|
||||
download and side-load a file.
|
||||
|
||||
It pairs with your instance using an API key generated automatically on first
|
||||
use. The bar directly under that card shows the key and can rotate it.
|
||||
|
||||
See [extension/README.md](extension/README.md) for what it does in detail.
|
||||
|
||||
## The GPU agent
|
||||
|
||||
Optional, and separate. If you have a desktop with a graphics card, you can run
|
||||
an agent on it that leases ML jobs from FabledCurator over HTTP, does them on
|
||||
the GPU, and hands the results back. It never touches the database or Redis, so
|
||||
it is safe to run somewhere the rest of the stack is not.
|
||||
|
||||
Run it for a burst of tagging, stop it to get your card back. It deploys from
|
||||
`agent/docker-compose.yml`, not the main stack — see
|
||||
[agent/README.md](agent/README.md).
|
||||
|
||||
## Upgrading
|
||||
|
||||
```bash
|
||||
docker compose -f docker-compose.yml pull
|
||||
docker compose -f docker-compose.yml up -d
|
||||
```
|
||||
|
||||
Migrations run automatically on start. Take a database backup first — Settings →
|
||||
Maintenance has one — because the schema moves forward and does not move back.
|
||||
|
||||
## Deployment posture
|
||||
|
||||
FabledCurator is built to run inside a homelab over plain HTTP. It does not
|
||||
generate certificates, redirect to HTTPS, or set HSTS. If you want TLS,
|
||||
terminate it at your reverse proxy. See [Before you expose it](#before-you-expose-it)
|
||||
for why TLS alone is not enough.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
**The UI loads but nothing ever finishes.** The web container is up and the
|
||||
workers are not. `docker compose -f docker-compose.yml ps` — check `worker`,
|
||||
`scheduler` and `ml-worker` are healthy, not restarting.
|
||||
|
||||
**`docker compose up` started building instead of pulling.** You left off
|
||||
`-f docker-compose.yml`, so the dev override took over. See [Install](#install).
|
||||
|
||||
**Downloads fail with an auth error.** The stored session for that platform has
|
||||
expired. Re-capture it with the extension; sessions do not last forever.
|
||||
|
||||
**Which build am I running?** The foot of Settings shows a version and a
|
||||
channel, and `/api/health` returns the same two fields. There are no version
|
||||
tags on the images, so this is the authoritative answer.
|
||||
|
||||
---
|
||||
|
||||
# Developing FabledCurator
|
||||
|
||||
Everything below is about working on FabledCurator rather than running it. If
|
||||
you are installing it, you are done — see [CONTRIBUTING.md](CONTRIBUTING.md) if
|
||||
you want to send a patch.
|
||||
|
||||
## Status and channels
|
||||
|
||||
In production. `main` is continuously deployed — every merge builds and
|
||||
publishes `:latest`, so whatever is on `main` is what is running. Day-to-day
|
||||
work happens on `dev`, which publishes `:dev`.
|
||||
|
||||
For local development, the dev override handles everything:
|
||||
|
||||
```bash
|
||||
docker compose up -d # note: no -f, so the override applies
|
||||
```
|
||||
|
||||
That builds the images from source, turns on DEBUG logging, and exposes
|
||||
Postgres and Redis on the host. No `.env` required.
|
||||
|
||||
## Versions and tags
|
||||
|
||||
@@ -47,38 +263,10 @@ Five deployable pieces, built by `.forgejo/workflows/build.yml`:
|
||||
| --- | --- | --- | --- |
|
||||
| **Web / workers** | `Dockerfile` | `fabledcurator` | Quart API + the built Vue SPA in one image. `entrypoint.sh` picks the role: `web`, `worker`, `scheduler`. The `maintenance-long` service is a second `worker` pinned to the long-running maintenance queue. |
|
||||
| **ML worker** | `Dockerfile.ml` | `fabledcurator-ml` | Same app, plus `requirements-ml.txt` — tagging and embedding models that run in-container. |
|
||||
| **GPU agent** | `agent/Dockerfile` | `fabledcurator-agent` | Optional desktop-GPU worker (`agent/`). Leases jobs over **HTTP only** — never touches the database or Redis. Run it for a burst, stop it to reclaim the card. See `agent/README.md`. |
|
||||
| **GPU agent** | `agent/Dockerfile` | `fabledcurator-agent` | Optional desktop-GPU worker (`agent/`). Leases jobs over **HTTP only** — never touches the database or Redis. See `agent/README.md`. |
|
||||
| **Firefox extension** | `extension/` | signed XPI | MV3 extension: pushes platform session cookies into FC and adds a creator as a Source in one click. AMO-signed on both `dev` and `main` (one signature per extension change, shared by the two channels), bundled into that channel's web image and served from Settings → Maintenance. See `extension/README.md`. |
|
||||
| **Data** | — | `pgvector/pgvector:pg16`, `redis:7-alpine` | Postgres with pgvector for embeddings; Redis as the Celery broker. |
|
||||
|
||||
## Quick start
|
||||
|
||||
For local development and testing, just:
|
||||
|
||||
```bash
|
||||
docker compose up -d
|
||||
# UI: http://localhost:8080
|
||||
```
|
||||
|
||||
That uses sane dev defaults baked into `docker-compose.yml` and the dev
|
||||
override (`docker-compose.override.yml`, auto-merged) — local builds, DEBUG
|
||||
logging, exposed Postgres + Redis ports on the host. No `.env` required.
|
||||
|
||||
For a production-like deployment, override the dev defaults via shell env
|
||||
or a `.env` file (see `.env.example` for the variable names) and use:
|
||||
|
||||
```bash
|
||||
docker compose -f docker-compose.yml up -d
|
||||
# (skips the override so containers pull registry images)
|
||||
```
|
||||
|
||||
The GPU agent is deployed separately, on the machine with the card —
|
||||
`agent/docker-compose.yml`, not this stack.
|
||||
|
||||
## Deployment posture
|
||||
|
||||
FabledCurator is designed to run inside a self-hosted homelab environment over plain HTTP. If you want TLS, terminate it at your reverse proxy. The app does not generate certificates, redirect to HTTPS, or set HSTS.
|
||||
|
||||
## CI / Forgejo setup
|
||||
|
||||
Four workflows: `ci.yml` (lint, extension-version check, backend unit tests,
|
||||
@@ -108,6 +296,29 @@ source, so `main` finds `dev`'s signature already cached and makes no second AMO
|
||||
call. That cache is why signing must be one-shot — AMO rejects a re-signed
|
||||
version.
|
||||
|
||||
## History
|
||||
|
||||
FabledCurator combines what was
|
||||
[ImageRepo](https://git.fabledsword.com/bvandeusen/ImageRepo) (gallery, ML,
|
||||
importer) and
|
||||
[GallerySubscriber](https://git.fabledsword.com/bvandeusen/GallerySubscriber)
|
||||
(gallery-dl wrapper, subscriptions, credential capture) into a single product.
|
||||
Both are superseded; neither is maintained.
|
||||
|
||||
## License
|
||||
|
||||
Personal project; use at your own discretion.
|
||||
**GNU Affero General Public License v3.0** — see [LICENSE](LICENSE).
|
||||
|
||||
You may run, study, modify and redistribute this software. The condition is
|
||||
reciprocity: if you distribute a modified version, or **run one as a network
|
||||
service that other people use**, you must offer those users the corresponding
|
||||
source under the same licence. That second clause (AGPL §13) is the reason this
|
||||
licence rather than the GPL — for a self-hosted web application, "distribution"
|
||||
otherwise never happens, and the obligation would never bite.
|
||||
|
||||
Running an unmodified copy for yourself, your household or your organisation
|
||||
carries no obligation at all. Neither does modifying it privately. The licence
|
||||
asks something of you only when you hand your modified version to others.
|
||||
|
||||
Contributions ship under the same licence — see [CONTRIBUTING](CONTRIBUTING.md).
|
||||
Security reports: [SECURITY.md](SECURITY.md).
|
||||
|
||||
+77
@@ -0,0 +1,77 @@
|
||||
# Security Policy
|
||||
|
||||
## Reporting a vulnerability
|
||||
|
||||
**Please do not put vulnerability details in a public issue.**
|
||||
|
||||
This project has no private disclosure channel yet. Until it does, open an
|
||||
issue on the repository that says only that you have a security report — no
|
||||
reproduction steps, no affected endpoint, no payload — and a maintainer will
|
||||
reply with a private contact to send the details to.
|
||||
|
||||
That is a deliberately awkward first step, and it exists because the
|
||||
alternative is worse: an issue tracker is public the moment it is written to,
|
||||
and every self-hosted instance stays vulnerable until its operator has had a
|
||||
chance to update.
|
||||
|
||||
Please include, once you have a private channel:
|
||||
|
||||
- what an attacker can do, and what access they need to start
|
||||
- the version or commit you tested
|
||||
- reproduction steps
|
||||
|
||||
## Scope — what this software actually handles
|
||||
|
||||
FabledCurator is self-hosted and holds things worth stating plainly, because
|
||||
they shape what counts as a serious bug here:
|
||||
|
||||
- **Platform credentials.** The app captures and stores session cookies for
|
||||
third-party subscription sites (Patreon, SubscribeStar, Pixiv) so it can
|
||||
download on the operator's behalf. These are live credentials for accounts
|
||||
that usually carry a payment method. Anything that discloses them, decrypts
|
||||
them, or lets one user of a shared instance read another's is high severity.
|
||||
- **An extension API key.** The Firefox extension authenticates to the backend
|
||||
with a shared key. Anything that leaks it or lets it be bypassed is a way in.
|
||||
- **No authentication of its own.** This is the most important thing on this
|
||||
page. FabledCurator has no login, no user accounts and no permission model —
|
||||
there is no `User` table and no session auth anywhere in the backend. Every
|
||||
HTTP client that can reach the port is the administrator, with full read and
|
||||
write access to everything above, including the stored platform credentials.
|
||||
Access control is entirely the operator's job, done at the network layer.
|
||||
Reports that an unauthenticated caller can reach an endpoint are therefore
|
||||
describing the design; reports that something *crosses the network boundary
|
||||
the operator drew* — an SSRF, a request forgery that rides a browser the
|
||||
operator already has open, a path that leaks state to an origin the operator
|
||||
did not authorise — are in scope and are serious.
|
||||
- **Arbitrary media from the internet.** Downloaded files are decoded, hashed,
|
||||
thumbnailed and fed to ML models. Anything that turns a hostile file into
|
||||
code execution is in scope.
|
||||
|
||||
## Deployment posture — read this before reporting
|
||||
|
||||
FabledCurator is designed to run **inside a private network, over plain HTTP,
|
||||
reachable only by its operator**. It does not terminate TLS, redirect to
|
||||
HTTPS, or set HSTS; if you want transport security, terminate it at your
|
||||
reverse proxy. It also does not authenticate anyone — see above. These are
|
||||
documented design decisions, not oversights.
|
||||
|
||||
Putting this on the public internet, with or without TLS, hands whoever finds
|
||||
it your Patreon, SubscribeStar and Pixiv sessions. A reverse proxy that adds
|
||||
TLS but not an authentication layer does not change that.
|
||||
|
||||
Reports that reduce to "the application is served over HTTP", "there is no
|
||||
HSTS header", or "the API needs no credentials" describe those decisions
|
||||
rather than vulnerabilities. Reports that the operator can cause the software
|
||||
to do something destructive are usually also by design — the operator is the
|
||||
administrator of their own instance.
|
||||
|
||||
What remains in scope is everything that crosses a boundary the software is
|
||||
actually supposed to hold: between untrusted downloaded content and the host,
|
||||
between a third-party origin and an operator's open browser session, and
|
||||
between the credentials at rest and anything that is not the operator.
|
||||
|
||||
## Supported versions
|
||||
|
||||
Fixes land on the `main` branch and reach the `:latest` image. There are no
|
||||
maintained release branches — the supported version is the current one, and
|
||||
the remedy for a security issue is to update.
|
||||
@@ -1,78 +1,107 @@
|
||||
"""Collapsed baseline — the whole schema in one revision.
|
||||
"""The whole schema, in one migration.
|
||||
|
||||
Replaces revisions 0001..0087, which narrated the build-out of this project
|
||||
and were deleted in milestone 328 step 1. A new install creates the schema in
|
||||
one step instead of replaying that history.
|
||||
This replaces alembic revisions 0001..0089 — the entire build-out of the
|
||||
project, 89 files and ~6,000 lines that a new installation used to replay in
|
||||
order to arrive at a schema this file creates in one pass. Nothing about the
|
||||
resulting database changes; what goes away is the requirement that a stranger
|
||||
re-run our development history to get it.
|
||||
|
||||
WHY THE REVISION ID IS "0087" AND NOT "0001"
|
||||
--------------------------------------------
|
||||
It is deliberately the id of the LAST revision this baseline collapses, so an
|
||||
existing database needs no intervention at all:
|
||||
## Why the revision id is 0089
|
||||
|
||||
* a fresh install finds current=none, head=0087, runs this file once, and
|
||||
ends stamped at 0087.
|
||||
* an existing install is ALREADY at 0087, so `alembic upgrade head` finds
|
||||
current == head and does nothing.
|
||||
`revision = "0089"` and `down_revision = None` are both deliberate, and the
|
||||
combination is the entire migration strategy for existing installations.
|
||||
|
||||
The alternative — numbering this 0001 and stamping every existing database —
|
||||
means running `alembic stamp` against live data, and stamp VALIDATES NOTHING.
|
||||
It writes a version string whether or not the schema actually matches, so a
|
||||
wrong baseline would be discovered later, by the next real migration, with no
|
||||
clean way back. Keeping the id removes that operation instead of making it
|
||||
safe. Future revisions continue at 0088.
|
||||
An already-deployed database has `alembic_version = '0089'`, because it ran the
|
||||
real 0089. This file claims that same id, so alembic reads the version table,
|
||||
sees head already reached, and does nothing at all. No stamp is needed — which
|
||||
matters because `alembic stamp` writes a version string without validating
|
||||
anything about the schema it is writing it against, and a stamp that is wrong
|
||||
is indistinguishable from one that is right until the next migration fails.
|
||||
|
||||
The one case this makes worse, and it fails LOUDLY rather than silently: a
|
||||
database still sitting between 0001 and 0086 (i.e. never upgraded to head)
|
||||
cannot be located in this chain and errors out. Upgrade to 0087 on a
|
||||
pre-squash build first, then take this one.
|
||||
An empty database has no version row, so alembic runs this file and then
|
||||
records `0089`. Both paths converge on the same schema and the same version,
|
||||
and neither requires anyone to assert anything by hand.
|
||||
|
||||
WHAT IS HAND-WRITTEN HERE
|
||||
-------------------------
|
||||
Most of this file is `alembic revision --autogenerate` output, but four
|
||||
things are NOT in SQLAlchemy metadata and the generator cannot produce them.
|
||||
Each fails differently, and none of them fail at generation time:
|
||||
The next migration written after this one is `0090`, exactly as it would have
|
||||
been. The numbering is continuous across the collapse on purpose.
|
||||
|
||||
1. CREATE EXTENSION vector (was 0001) — without it the VECTOR
|
||||
columns below cannot be created at all.
|
||||
2. CREATE EXTENSION tsm_system_rows (was 0004) — used by the random-sample
|
||||
query path; its absence surfaces only when that query runs.
|
||||
3. The HNSW index on image_record.siglip_embedding (was 0036). Raw SQL
|
||||
because alembic's create_index cannot express `USING hnsw (...
|
||||
vector_cosine_ops)`. Its absence is the quietest failure of the four:
|
||||
everything works, similarity search just stops using an index.
|
||||
4. `import pgvector.sqlalchemy.vector`. Autogenerate EMITS references to
|
||||
pgvector.sqlalchemy.vector.VECTOR but does not add the import, so the
|
||||
generated file dies with NameError on first run.
|
||||
## What was added to the generated output, and why
|
||||
|
||||
The acceptance test for this file is not that it reads correctly — it is
|
||||
`.forgejo/workflows/baseline.yml`, which builds a database from the old
|
||||
0001..0087 chain (read out of git) and one from this file, and diffs
|
||||
pg_dump --schema-only output. That is what proves nothing was missed.
|
||||
`alembic revision --autogenerate` produced almost all of this from the models,
|
||||
which is only true because #3275 first made the models actually describe the
|
||||
schema. Before that reconciliation the generator silently omitted eleven
|
||||
indexes and three uniqueness guarantees, and an earlier attempt at this squash
|
||||
had to be reverted for exactly that reason.
|
||||
|
||||
Revision ID: 0087
|
||||
Four things still had to be added by hand, because they are not in the models:
|
||||
|
||||
1. **`CREATE EXTENSION vector`** (from 0001) and **`tsm_system_rows`** (0004).
|
||||
Extensions are database objects, not table metadata, so no model can carry
|
||||
them. `IF NOT EXISTS` because a re-run must not fail.
|
||||
|
||||
2. **Three seed inserts** — the two settings singletons (0002, 0003) and the
|
||||
three hygiene system tags (0075). Some migrations did not only build schema;
|
||||
they inserted rows the product needs in order to function, and nothing in
|
||||
the application ever creates them. Every consumer reads them with
|
||||
`scalar_one()`, which RAISES `NoResultFound` on an empty result rather than
|
||||
returning None, so their absence is a crash and not a degradation.
|
||||
|
||||
Distinguishing these from the other data statements in the chain is the
|
||||
whole trick, and the rule turns out to be mechanical:
|
||||
|
||||
* `INSERT ... VALUES (...)` with literal values is a SEED. It creates
|
||||
something the product ships. It must be carried.
|
||||
* `INSERT ... SELECT ... FROM <table>` is a BACKFILL. It derives rows
|
||||
from rows that already exist, so on an empty database it inserts
|
||||
nothing and carrying it would be pointless. 0034 (artist_visit), 0040
|
||||
and 0047 (series_chapter) are all of this shape and are correctly
|
||||
absent here.
|
||||
|
||||
This category is invisible to every automated check this project has:
|
||||
`baseline.yml` compares SCHEMA, and a baseline missing all three seeds still
|
||||
produces a byte-identical schema and a perfectly green diff. What caught the
|
||||
system tags was the integration suite — 36 tests failing on
|
||||
`NoResultFound` — after a first version of this file shipped with only the
|
||||
two settings rows. A first-run check against the real application is the
|
||||
only thing that finds this class of defect.
|
||||
|
||||
3. **The `pgvector` import.** Autogenerate emits qualified
|
||||
`pgvector.sqlalchemy.vector.VECTOR(...)` references without importing the
|
||||
package, so the file it writes cannot execute — `NameError: name 'pgvector'
|
||||
is not defined`, observed on run 4988.
|
||||
|
||||
The other data statements in the old chain were deliberately NOT carried over.
|
||||
0023's `DELETE FROM tag WHERE kind IN (...)`, and 0047's `series_page` /
|
||||
`series_chapter` deletes, are historical cleanups that operate on rows an empty
|
||||
database does not have.
|
||||
|
||||
## Downgrade
|
||||
|
||||
There is none. A baseline's downgrade would be "drop the entire schema", which
|
||||
is not a migration but a data-loss event wearing one as a disguise. Restore
|
||||
from a backup instead — that is what backup_run exists for.
|
||||
|
||||
Revision ID: 0089
|
||||
Revises:
|
||||
Create Date: 2026-08-30
|
||||
Create Date: 2026-09-01
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
import sqlalchemy as sa
|
||||
import pgvector.sqlalchemy.vector
|
||||
from sqlalchemy.dialects import postgresql
|
||||
|
||||
# Autogenerate references pgvector.sqlalchemy.vector.VECTOR without importing
|
||||
# it. Item 4 above.
|
||||
import pgvector.sqlalchemy.vector
|
||||
|
||||
revision: str = "0087"
|
||||
revision: str = "0089"
|
||||
down_revision: Union[str, None] = None
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
# Extensions FIRST: the VECTOR columns below cannot be created without
|
||||
# `vector`, so ordering here is load-bearing, not tidiness.
|
||||
# Extensions first: image_record.siglip_embedding is a vector column and
|
||||
# cannot be created before the type exists. From 0001 and 0004.
|
||||
op.execute("CREATE EXTENSION IF NOT EXISTS vector")
|
||||
op.execute("CREATE EXTENSION IF NOT EXISTS tsm_system_rows")
|
||||
|
||||
@@ -87,8 +116,8 @@ def upgrade() -> None:
|
||||
sa.Column('name', sa.String(length=255), nullable=False),
|
||||
sa.Column('slug', sa.String(length=255), nullable=False),
|
||||
sa.Column('notes', sa.Text(), nullable=True),
|
||||
sa.Column('is_subscription', sa.Boolean(), nullable=False),
|
||||
sa.Column('auto_check', sa.Boolean(), nullable=False),
|
||||
sa.Column('is_subscription', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('auto_check', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('check_interval_seconds', sa.Integer(), nullable=True),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_artist')),
|
||||
@@ -97,7 +126,7 @@ def upgrade() -> None:
|
||||
op.create_table('backup_run',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('kind', sa.String(length=16), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='pending', nullable=False),
|
||||
sa.Column('tag', sa.String(length=64), nullable=True),
|
||||
sa.Column('triggered_by', sa.String(length=32), nullable=False),
|
||||
sa.Column('started_at', sa.DateTime(timezone=True), nullable=False),
|
||||
@@ -112,10 +141,12 @@ def upgrade() -> None:
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_backup_run'))
|
||||
)
|
||||
op.create_index(op.f('ix_backup_run_finished_at'), 'backup_run', ['finished_at'], unique=False)
|
||||
op.create_index(op.f('ix_backup_run_kind'), 'backup_run', ['kind'], unique=False)
|
||||
op.create_index('ix_backup_run_kind_started', 'backup_run', ['kind', sa.literal_column('started_at DESC')], unique=False)
|
||||
op.create_index(op.f('ix_backup_run_restored_from_id'), 'backup_run', ['restored_from_id'], unique=False)
|
||||
op.create_index(op.f('ix_backup_run_started_at'), 'backup_run', ['started_at'], unique=False)
|
||||
op.create_index(op.f('ix_backup_run_status'), 'backup_run', ['status'], unique=False)
|
||||
op.create_index('ix_backup_run_status_finished', 'backup_run', ['status', sa.literal_column('finished_at DESC')], unique=False)
|
||||
op.create_index(op.f('ix_backup_run_tag'), 'backup_run', ['tag'], unique=False)
|
||||
op.create_index('ix_backup_run_tag_partial', 'backup_run', ['tag'], unique=False, postgresql_where=sa.text('tag IS NOT NULL'))
|
||||
op.create_table('credential',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('platform', sa.String(length=64), nullable=False),
|
||||
@@ -129,9 +160,9 @@ def upgrade() -> None:
|
||||
)
|
||||
op.create_table('head_auto_apply_run',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('dry_run', sa.Boolean(), nullable=False),
|
||||
sa.Column('dry_run', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='running', nullable=False),
|
||||
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('n_applied', sa.Integer(), nullable=True),
|
||||
@@ -144,7 +175,7 @@ def upgrade() -> None:
|
||||
op.create_table('head_training_run',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='running', nullable=False),
|
||||
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('n_trained', sa.Integer(), nullable=True),
|
||||
@@ -161,38 +192,38 @@ def upgrade() -> None:
|
||||
sa.Column('scan_mode', sa.String(length=16), nullable=False),
|
||||
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('total_files', sa.Integer(), nullable=False),
|
||||
sa.Column('imported', sa.Integer(), nullable=False),
|
||||
sa.Column('skipped', sa.Integer(), nullable=False),
|
||||
sa.Column('failed', sa.Integer(), nullable=False),
|
||||
sa.Column('attachments', sa.Integer(), nullable=False),
|
||||
sa.Column('refreshed', sa.Integer(), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), nullable=False),
|
||||
sa.Column('total_files', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('imported', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('skipped', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('failed', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('attachments', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('refreshed', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='running', nullable=False),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_import_batch'))
|
||||
)
|
||||
op.create_index(op.f('ix_import_batch_status'), 'import_batch', ['status'], unique=False)
|
||||
op.create_table('import_settings',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('import_scan_path', sa.Text(), nullable=False),
|
||||
sa.Column('min_width', sa.Integer(), nullable=False),
|
||||
sa.Column('min_height', sa.Integer(), nullable=False),
|
||||
sa.Column('skip_transparent', sa.Boolean(), nullable=False),
|
||||
sa.Column('transparency_threshold', sa.Float(), nullable=False),
|
||||
sa.Column('skip_single_color', sa.Boolean(), nullable=False),
|
||||
sa.Column('single_color_threshold', sa.Float(), nullable=False),
|
||||
sa.Column('single_color_tolerance', sa.Integer(), nullable=False),
|
||||
sa.Column('phash_threshold', sa.Integer(), nullable=False),
|
||||
sa.Column('download_rate_limit_seconds', sa.Float(), nullable=False),
|
||||
sa.Column('download_validate_files', sa.Boolean(), nullable=False),
|
||||
sa.Column('download_schedule_default_seconds', sa.Integer(), nullable=False),
|
||||
sa.Column('download_event_retention_days', sa.Integer(), nullable=False),
|
||||
sa.Column('download_failure_warning_threshold', sa.Integer(), nullable=False),
|
||||
sa.Column('backup_db_nightly_enabled', sa.Boolean(), nullable=False),
|
||||
sa.Column('backup_db_nightly_hour_utc', sa.Integer(), nullable=False),
|
||||
sa.Column('backup_db_keep_last_n', sa.Integer(), nullable=False),
|
||||
sa.Column('backup_images_keep_last_n', sa.Integer(), nullable=False),
|
||||
sa.Column('series_suggest_enabled', sa.Boolean(), nullable=False),
|
||||
sa.Column('series_suggest_threshold', sa.Float(), nullable=False),
|
||||
sa.Column('import_scan_path', sa.Text(), server_default='/import', nullable=False),
|
||||
sa.Column('min_width', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('min_height', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('skip_transparent', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('transparency_threshold', sa.Float(), server_default='0.9', nullable=False),
|
||||
sa.Column('skip_single_color', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('single_color_threshold', sa.Float(), server_default='0.95', nullable=False),
|
||||
sa.Column('single_color_tolerance', sa.Integer(), server_default='30', nullable=False),
|
||||
sa.Column('phash_threshold', sa.Integer(), server_default='10', nullable=False),
|
||||
sa.Column('download_rate_limit_seconds', sa.Float(), server_default='3', nullable=False),
|
||||
sa.Column('download_validate_files', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('download_schedule_default_seconds', sa.Integer(), server_default='28800', nullable=False),
|
||||
sa.Column('download_event_retention_days', sa.Integer(), server_default='90', nullable=False),
|
||||
sa.Column('download_failure_warning_threshold', sa.Integer(), server_default='5', nullable=False),
|
||||
sa.Column('backup_db_nightly_enabled', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('backup_db_nightly_hour_utc', sa.Integer(), server_default='3', nullable=False),
|
||||
sa.Column('backup_db_keep_last_n', sa.Integer(), server_default='14', nullable=False),
|
||||
sa.Column('backup_images_keep_last_n', sa.Integer(), server_default='3', nullable=False),
|
||||
sa.Column('series_suggest_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('series_suggest_threshold', sa.Float(), server_default='0.5', nullable=False),
|
||||
sa.Column('extdl_mega_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('extdl_gdrive_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('extdl_mediafire_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
@@ -201,7 +232,7 @@ def upgrade() -> None:
|
||||
sa.Column('translation_enabled', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('interpreter_base_url', sa.Text(), server_default='', nullable=False),
|
||||
sa.Column('translation_target_lang', sa.Text(), server_default='en', nullable=False),
|
||||
sa.Column('translation_min_confidence', sa.Float(), server_default='0.9', nullable=False),
|
||||
sa.Column('translation_min_confidence', sa.Float(), server_default=sa.text('0.9'), nullable=False),
|
||||
sa.Column('wip_title_tagging_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('wip_soft_title_tagging_enabled', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.CheckConstraint('id = 1', name=op.f('ck_import_settings_singleton')),
|
||||
@@ -211,14 +242,14 @@ def upgrade() -> None:
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('rule', sa.String(length=32), nullable=False),
|
||||
sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='running', nullable=False),
|
||||
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('scanned_count', sa.Integer(), nullable=False),
|
||||
sa.Column('matched_count', sa.Integer(), nullable=False),
|
||||
sa.Column('matched_ids', postgresql.JSONB(astext_type=sa.Text()), nullable=False),
|
||||
sa.Column('scanned_count', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('matched_count', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('matched_ids', postgresql.JSONB(astext_type=sa.Text()), server_default=sa.text("'[]'::jsonb"), nullable=False),
|
||||
sa.Column('error', sa.Text(), nullable=True),
|
||||
sa.Column('resume_after_id', sa.Integer(), nullable=False),
|
||||
sa.Column('resume_after_id', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('last_progress_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_library_audit_run'))
|
||||
)
|
||||
@@ -226,40 +257,40 @@ def upgrade() -> None:
|
||||
op.create_index(op.f('ix_library_audit_run_status'), 'library_audit_run', ['status'], unique=False)
|
||||
op.create_table('ml_settings',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('cpu_embed_enabled', sa.Boolean(), nullable=False),
|
||||
sa.Column('video_frame_interval_seconds', sa.Float(), nullable=False),
|
||||
sa.Column('video_max_frames', sa.Integer(), nullable=False),
|
||||
sa.Column('head_min_positives', sa.Integer(), nullable=False),
|
||||
sa.Column('head_auto_apply_precision', sa.Float(), nullable=False),
|
||||
sa.Column('head_auto_apply_enabled', sa.Boolean(), nullable=False),
|
||||
sa.Column('head_auto_apply_min_positives', sa.Integer(), nullable=False),
|
||||
sa.Column('ccip_match_threshold', sa.Float(), nullable=False),
|
||||
sa.Column('ccip_auto_apply_enabled', sa.Boolean(), nullable=False),
|
||||
sa.Column('ccip_auto_apply_threshold', sa.Float(), nullable=False),
|
||||
sa.Column('presentation_auto_apply_enabled', sa.Boolean(), nullable=False),
|
||||
sa.Column('presentation_auto_apply_threshold', sa.Float(), nullable=False),
|
||||
sa.Column('presentation_conflict_threshold', sa.Float(), nullable=False),
|
||||
sa.Column('process_auto_apply_enabled', sa.Boolean(), nullable=False),
|
||||
sa.Column('process_auto_apply_threshold', sa.Float(), nullable=False),
|
||||
sa.Column('process_conflict_threshold', sa.Float(), nullable=False),
|
||||
sa.Column('embedder_model_version', sa.String(length=128), nullable=False),
|
||||
sa.Column('embedder_model_name', sa.String(length=128), nullable=False),
|
||||
sa.Column('detector_person_enabled', sa.Boolean(), nullable=False),
|
||||
sa.Column('detector_person_weights', sa.String(length=512), nullable=False),
|
||||
sa.Column('detector_person_conf', sa.Float(), nullable=False),
|
||||
sa.Column('detector_anatomy_enabled', sa.Boolean(), nullable=False),
|
||||
sa.Column('detector_anatomy_weights', sa.String(length=512), nullable=False),
|
||||
sa.Column('detector_anatomy_conf', sa.Float(), nullable=False),
|
||||
sa.Column('detector_panel_enabled', sa.Boolean(), nullable=False),
|
||||
sa.Column('detector_panel_weights', sa.String(length=512), nullable=False),
|
||||
sa.Column('detector_panel_conf', sa.Float(), nullable=False),
|
||||
sa.Column('detector_max_figures', sa.Integer(), nullable=False),
|
||||
sa.Column('detector_max_components', sa.Integer(), nullable=False),
|
||||
sa.Column('detector_max_panels', sa.Integer(), nullable=False),
|
||||
sa.Column('detector_max_regions', sa.Integer(), nullable=False),
|
||||
sa.Column('detector_dedupe_iou', sa.Float(), nullable=False),
|
||||
sa.Column('cpu_embed_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('video_frame_interval_seconds', sa.Float(), server_default='4', nullable=False),
|
||||
sa.Column('video_max_frames', sa.Integer(), server_default='64', nullable=False),
|
||||
sa.Column('head_min_positives', sa.Integer(), server_default='8', nullable=False),
|
||||
sa.Column('head_auto_apply_precision', sa.Float(), server_default='0.97', nullable=False),
|
||||
sa.Column('head_auto_apply_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('head_auto_apply_min_positives', sa.Integer(), server_default='30', nullable=False),
|
||||
sa.Column('ccip_match_threshold', sa.Float(), server_default='0.85', nullable=False),
|
||||
sa.Column('ccip_auto_apply_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('ccip_auto_apply_threshold', sa.Float(), server_default='0.92', nullable=False),
|
||||
sa.Column('presentation_auto_apply_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('presentation_auto_apply_threshold', sa.Float(), server_default=sa.text('0.90'), nullable=False),
|
||||
sa.Column('presentation_conflict_threshold', sa.Float(), server_default=sa.text('0.50'), nullable=False),
|
||||
sa.Column('process_auto_apply_enabled', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('process_auto_apply_threshold', sa.Float(), server_default='0.90', nullable=False),
|
||||
sa.Column('process_conflict_threshold', sa.Float(), server_default='0.50', nullable=False),
|
||||
sa.Column('embedder_model_version', sa.String(length=128), server_default='siglip2-so400m-patch16-512', nullable=False),
|
||||
sa.Column('embedder_model_name', sa.String(length=128), server_default='google/siglip2-so400m-patch16-512', nullable=False),
|
||||
sa.Column('detector_person_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('detector_person_weights', sa.String(length=512), server_default='yolo11n.pt', nullable=False),
|
||||
sa.Column('detector_person_conf', sa.Float(), server_default=sa.text('0.35'), nullable=False),
|
||||
sa.Column('detector_anatomy_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('detector_anatomy_weights', sa.String(length=512), server_default='https://github.com/aperveyev/booru_yolo/raw/main/models/yolov11m_aa22.pt', nullable=False),
|
||||
sa.Column('detector_anatomy_conf', sa.Float(), server_default=sa.text('0.30'), nullable=False),
|
||||
sa.Column('detector_panel_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('detector_panel_weights', sa.String(length=512), server_default='mosesb/best-comic-panel-detection::best.pt', nullable=False),
|
||||
sa.Column('detector_panel_conf', sa.Float(), server_default=sa.text('0.30'), nullable=False),
|
||||
sa.Column('detector_max_figures', sa.Integer(), server_default='8', nullable=False),
|
||||
sa.Column('detector_max_components', sa.Integer(), server_default='8', nullable=False),
|
||||
sa.Column('detector_max_panels', sa.Integer(), server_default='8', nullable=False),
|
||||
sa.Column('detector_max_regions', sa.Integer(), server_default='128', nullable=False),
|
||||
sa.Column('detector_dedupe_iou', sa.Float(), server_default=sa.text('0.85'), nullable=False),
|
||||
sa.Column('ccip_ref_signature', sa.String(length=128), nullable=True),
|
||||
sa.Column('ccip_prototype_cap', sa.Integer(), nullable=False),
|
||||
sa.Column('ccip_prototype_cap', sa.Integer(), server_default='64', nullable=False),
|
||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.CheckConstraint('id = 1', name=op.f('ck_ml_settings_singleton')),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_ml_settings'))
|
||||
@@ -267,15 +298,16 @@ def upgrade() -> None:
|
||||
op.create_table('tag',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('name', sa.String(length=255), nullable=False),
|
||||
sa.Column('kind', sa.Enum('artist', 'character', 'fandom', 'general', 'series', 'archive', 'post', name='tag_kind'), nullable=False),
|
||||
sa.Column('kind', sa.Enum('artist', 'character', 'fandom', 'general', 'series', 'archive', 'post', name='tag_kind'), server_default='general', nullable=False),
|
||||
sa.Column('fandom_id', sa.Integer(), nullable=True),
|
||||
sa.Column('is_system', sa.Boolean(), server_default=sa.text('false'), nullable=False),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.CheckConstraint("(fandom_id IS NULL) OR (kind = 'character')", name=op.f('ck_tag_ck_tag_fandom_requires_character')),
|
||||
sa.CheckConstraint("(fandom_id IS NULL) OR (kind = 'character')", name=op.f('ck_tag_fandom_requires_character')),
|
||||
sa.ForeignKeyConstraint(['fandom_id'], ['tag.id'], name=op.f('fk_tag_fandom_id_tag'), ondelete='SET NULL'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_tag'))
|
||||
)
|
||||
op.create_index(op.f('ix_tag_fandom_id'), 'tag', ['fandom_id'], unique=False)
|
||||
op.create_index('uq_tag_name_kind_fandom', 'tag', ['name', 'kind', sa.literal_column('COALESCE(fandom_id, 0)')], unique=True)
|
||||
op.create_table('task_run',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('celery_task_id', sa.String(length=64), nullable=False),
|
||||
@@ -285,7 +317,7 @@ def upgrade() -> None:
|
||||
sa.Column('started_at', sa.DateTime(timezone=True), nullable=False),
|
||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('duration_ms', sa.Integer(), nullable=True),
|
||||
sa.Column('status', sa.String(length=16), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='running', nullable=False),
|
||||
sa.Column('error_type', sa.String(length=128), nullable=True),
|
||||
sa.Column('error_message', sa.Text(), nullable=True),
|
||||
sa.Column('retry_count', sa.Integer(), nullable=True),
|
||||
@@ -295,10 +327,10 @@ def upgrade() -> None:
|
||||
)
|
||||
op.create_index(op.f('ix_task_run_celery_task_id'), 'task_run', ['celery_task_id'], unique=False)
|
||||
op.create_index(op.f('ix_task_run_finished_at'), 'task_run', ['finished_at'], unique=False)
|
||||
op.create_index(op.f('ix_task_run_queue'), 'task_run', ['queue'], unique=False)
|
||||
op.create_index('ix_task_run_name_started', 'task_run', ['task_name', sa.literal_column('started_at DESC')], unique=False)
|
||||
op.create_index('ix_task_run_queue_started', 'task_run', ['queue', sa.literal_column('started_at DESC')], unique=False)
|
||||
op.create_index(op.f('ix_task_run_started_at'), 'task_run', ['started_at'], unique=False)
|
||||
op.create_index(op.f('ix_task_run_status'), 'task_run', ['status'], unique=False)
|
||||
op.create_index(op.f('ix_task_run_task_name'), 'task_run', ['task_name'], unique=False)
|
||||
op.create_index('ix_task_run_status_started', 'task_run', ['status', sa.literal_column('started_at DESC')], unique=False)
|
||||
op.create_table('artist_visit',
|
||||
sa.Column('artist_id', sa.Integer(), nullable=False),
|
||||
sa.Column('last_viewed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
@@ -314,20 +346,20 @@ def upgrade() -> None:
|
||||
)
|
||||
op.create_table('head_metric',
|
||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('n_misfires', sa.Integer(), nullable=False),
|
||||
sa.Column('n_underfires', sa.Integer(), nullable=False),
|
||||
sa.Column('n_misfires', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('n_underfires', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_head_metric_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('tag_id', name=op.f('pk_head_metric'))
|
||||
)
|
||||
op.create_table('head_metrics_snapshot',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('tag_id', sa.Integer(), nullable=True),
|
||||
sa.Column('name', sa.String(length=255), nullable=False),
|
||||
sa.Column('snapshot_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('n_auto_applied', sa.Integer(), nullable=False),
|
||||
sa.Column('n_misfires', sa.Integer(), nullable=False),
|
||||
sa.Column('n_underfires', sa.Integer(), nullable=False),
|
||||
sa.Column('n_auto_applied', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('n_misfires', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('n_underfires', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('ap', sa.Float(), nullable=True),
|
||||
sa.Column('precision_cv', sa.Float(), nullable=True),
|
||||
sa.Column('recall', sa.Float(), nullable=True),
|
||||
@@ -342,16 +374,17 @@ def upgrade() -> None:
|
||||
sa.Column('artist_id', sa.Integer(), nullable=False),
|
||||
sa.Column('platform', sa.String(length=64), nullable=False),
|
||||
sa.Column('url', sa.Text(), nullable=False),
|
||||
sa.Column('enabled', sa.Boolean(), nullable=False),
|
||||
sa.Column('enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('config_overrides', sa.JSON(), nullable=True),
|
||||
sa.Column('last_checked_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('last_error', sa.Text(), nullable=True),
|
||||
sa.Column('error_type', sa.String(length=32), nullable=True),
|
||||
sa.Column('check_interval_override', sa.Integer(), nullable=True),
|
||||
sa.Column('consecutive_failures', sa.Integer(), nullable=False),
|
||||
sa.Column('consecutive_failures', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('backfill_runs_remaining', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_source_artist_id_artist'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_source'))
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_source')),
|
||||
sa.UniqueConstraint('artist_id', 'platform', 'url', name='uq_source_artist_platform_url')
|
||||
)
|
||||
op.create_index(op.f('ix_source_artist_id'), 'source', ['artist_id'], unique=False)
|
||||
op.create_index(op.f('ix_source_error_type'), 'source', ['error_type'], unique=False)
|
||||
@@ -363,7 +396,7 @@ def upgrade() -> None:
|
||||
sa.ForeignKeyConstraint(['canonical_tag_id'], ['tag.id'], name=op.f('fk_tag_alias_canonical_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('alias_string', 'alias_category', name=op.f('pk_tag_alias'))
|
||||
)
|
||||
op.create_index(op.f('ix_tag_alias_canonical_tag_id'), 'tag_alias', ['canonical_tag_id'], unique=False)
|
||||
op.create_index('ix_tag_alias_canonical', 'tag_alias', ['canonical_tag_id'], unique=False)
|
||||
op.create_table('tag_head',
|
||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('embedding_version', sa.String(length=128), nullable=False),
|
||||
@@ -386,7 +419,7 @@ def upgrade() -> None:
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('source_id', sa.Integer(), nullable=False),
|
||||
sa.Column('filehash', sa.String(length=128), nullable=False),
|
||||
sa.Column('attempts', sa.Integer(), nullable=False),
|
||||
sa.Column('attempts', sa.Integer(), server_default='1', nullable=False),
|
||||
sa.Column('last_error', sa.Text(), nullable=True),
|
||||
sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
@@ -410,7 +443,7 @@ def upgrade() -> None:
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('source_id', sa.Integer(), nullable=False),
|
||||
sa.Column('filehash', sa.String(length=128), nullable=False),
|
||||
sa.Column('attempts', sa.Integer(), nullable=False),
|
||||
sa.Column('attempts', sa.Integer(), server_default='1', nullable=False),
|
||||
sa.Column('last_error', sa.Text(), nullable=True),
|
||||
sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
@@ -448,7 +481,7 @@ def upgrade() -> None:
|
||||
sa.Column('translated_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('translation_override', sa.String(length=16), server_default='auto', nullable=False),
|
||||
sa.Column('downloaded_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.CheckConstraint("translation_override IN ('auto', 'force', 'original')", name=op.f('ck_post_ck_post_translation_override')),
|
||||
sa.CheckConstraint("translation_override IN ('auto', 'force', 'original')", name=op.f('ck_post_translation_override')),
|
||||
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_post_artist_id_artist'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_post_source_id_source'), ondelete='SET NULL'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_post')),
|
||||
@@ -456,11 +489,12 @@ def upgrade() -> None:
|
||||
)
|
||||
op.create_index(op.f('ix_post_artist_id'), 'post', ['artist_id'], unique=False)
|
||||
op.create_index(op.f('ix_post_source_id'), 'post', ['source_id'], unique=False)
|
||||
op.create_index('uq_post_artist_external_id_null_source', 'post', ['artist_id', 'external_post_id'], unique=True, postgresql_where=sa.text('source_id IS NULL'))
|
||||
op.create_table('subscribestar_failed_media',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('source_id', sa.Integer(), nullable=False),
|
||||
sa.Column('filehash', sa.String(length=128), nullable=False),
|
||||
sa.Column('attempts', sa.Integer(), nullable=False),
|
||||
sa.Column('attempts', sa.Integer(), server_default='1', nullable=False),
|
||||
sa.Column('last_error', sa.Text(), nullable=True),
|
||||
sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
@@ -487,8 +521,8 @@ def upgrade() -> None:
|
||||
sa.Column('status', sa.String(length=32), nullable=False),
|
||||
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('bytes_downloaded', sa.BigInteger(), nullable=False),
|
||||
sa.Column('files_count', sa.Integer(), nullable=False),
|
||||
sa.Column('bytes_downloaded', sa.BigInteger(), server_default='0', nullable=False),
|
||||
sa.Column('files_count', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('error', sa.Text(), nullable=True),
|
||||
sa.Column('metadata', postgresql.JSONB(astext_type=sa.Text()), server_default=sa.text("'{}'::jsonb"), nullable=False),
|
||||
sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_download_event_post_id_post'), ondelete='SET NULL'),
|
||||
@@ -507,7 +541,7 @@ def upgrade() -> None:
|
||||
sa.Column('width', sa.Integer(), nullable=True),
|
||||
sa.Column('height', sa.Integer(), nullable=True),
|
||||
sa.Column('duration_seconds', sa.Float(), nullable=True),
|
||||
sa.Column('integrity_status', sa.String(length=24), nullable=False),
|
||||
sa.Column('integrity_status', sa.String(length=24), server_default='unknown', nullable=False),
|
||||
sa.Column('thumbnail_path', sa.Text(), nullable=True),
|
||||
sa.Column('source_url', sa.Text(), nullable=True),
|
||||
sa.Column('source_filehash', sa.String(length=32), nullable=True),
|
||||
@@ -520,16 +554,19 @@ def upgrade() -> None:
|
||||
sa.Column('effective_date', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('earliest_post_date', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_image_record_artist_id_artist'), ondelete='SET NULL'),
|
||||
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name='fk_image_record_artist_id', ondelete='SET NULL'),
|
||||
sa.ForeignKeyConstraint(['primary_post_id'], ['post.id'], name=op.f('fk_image_record_primary_post_id_post'), ondelete='SET NULL'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_image_record')),
|
||||
sa.UniqueConstraint('path', name=op.f('uq_image_record_path'))
|
||||
sa.UniqueConstraint('path', name=op.f('uq_image_record_path')),
|
||||
sa.UniqueConstraint('sha256', name='uq_image_record_sha256')
|
||||
)
|
||||
op.create_index(op.f('ix_image_record_artist_id'), 'image_record', ['artist_id'], unique=False)
|
||||
op.create_index('ix_image_record_earliest_post_date', 'image_record', [sa.literal_column('earliest_post_date DESC'), sa.literal_column('id DESC')], unique=False)
|
||||
op.create_index('ix_image_record_effective_date', 'image_record', [sa.literal_column('effective_date DESC'), sa.literal_column('id DESC')], unique=False)
|
||||
op.create_index(op.f('ix_image_record_integrity_status'), 'image_record', ['integrity_status'], unique=False)
|
||||
op.create_index(op.f('ix_image_record_phash'), 'image_record', ['phash'], unique=False)
|
||||
op.create_index(op.f('ix_image_record_primary_post_id'), 'image_record', ['primary_post_id'], unique=False)
|
||||
op.create_index(op.f('ix_image_record_sha256'), 'image_record', ['sha256'], unique=True)
|
||||
op.create_index('ix_image_record_siglip_hnsw', 'image_record', ['siglip_embedding'], unique=False, postgresql_using='hnsw', postgresql_ops={'siglip_embedding': 'vector_cosine_ops'})
|
||||
op.create_index(op.f('ix_image_record_source_filehash'), 'image_record', ['source_filehash'], unique=False)
|
||||
op.create_table('post_attachment',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
@@ -582,24 +619,26 @@ def upgrade() -> None:
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('completed_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('duration_seconds', sa.Float(), nullable=True),
|
||||
sa.CheckConstraint("host IN ('mega', 'gdrive', 'mediafire', 'dropbox', 'pixeldrain')", name=op.f('ck_external_link_host')),
|
||||
sa.CheckConstraint("status IN ('pending', 'downloading', 'downloaded', 'failed', 'skipped', 'dead')", name=op.f('ck_external_link_status')),
|
||||
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_external_link_artist_id_artist'), ondelete='SET NULL'),
|
||||
sa.ForeignKeyConstraint(['attachment_id'], ['post_attachment.id'], name=op.f('fk_external_link_attachment_id_post_attachment'), ondelete='SET NULL'),
|
||||
sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_external_link_post_id_post'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_external_link'))
|
||||
)
|
||||
op.create_index(op.f('ix_external_link_artist_id'), 'external_link', ['artist_id'], unique=False)
|
||||
op.create_index(op.f('ix_external_link_post_id'), 'external_link', ['post_id'], unique=False)
|
||||
op.create_index('ix_external_link_attachment_id', 'external_link', ['attachment_id'], unique=False)
|
||||
op.create_index('ix_external_link_status', 'external_link', ['status'], unique=False)
|
||||
op.create_index('uq_external_link_post_url', 'external_link', ['post_id', 'url'], unique=True)
|
||||
op.create_table('gpu_job',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
||||
sa.Column('task', sa.String(length=32), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='pending', nullable=False),
|
||||
sa.Column('lease_token', sa.String(length=64), nullable=True),
|
||||
sa.Column('leased_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('lease_expires_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('attempts', sa.Integer(), nullable=False),
|
||||
sa.Column('attempts', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('error', sa.Text(), nullable=True),
|
||||
sa.Column('triage_status', sa.String(length=16), nullable=True),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
@@ -619,7 +658,7 @@ def upgrade() -> None:
|
||||
sa.Column('from_attachment_id', sa.Integer(), nullable=True),
|
||||
sa.Column('captured_metadata', sa.JSON(), nullable=True),
|
||||
sa.Column('captured_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['from_attachment_id'], ['post_attachment.id'], name=op.f('fk_image_provenance_from_attachment_id_post_attachment'), ondelete='SET NULL'),
|
||||
sa.ForeignKeyConstraint(['from_attachment_id'], ['post_attachment.id'], name='fk_image_provenance_from_attachment', ondelete='SET NULL'),
|
||||
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_image_provenance_image_record_id_image_record'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_image_provenance_post_id_post'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_image_provenance_source_id_source'), ondelete='SET NULL'),
|
||||
@@ -653,20 +692,21 @@ def upgrade() -> None:
|
||||
op.create_table('image_tag',
|
||||
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('source', sa.String(length=32), nullable=False),
|
||||
sa.Column('source', sa.String(length=32), server_default='manual', nullable=False),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_image_tag_image_record_id_image_record'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_image_tag_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_image_tag'))
|
||||
)
|
||||
op.create_index('ix_image_tag_tag_id', 'image_tag', ['tag_id'], unique=False)
|
||||
op.create_table('import_task',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('batch_id', sa.Integer(), nullable=False),
|
||||
sa.Column('source_path', sa.Text(), nullable=False),
|
||||
sa.Column('task_type', sa.String(length=16), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), nullable=False),
|
||||
sa.Column('recovery_count', sa.Integer(), nullable=False),
|
||||
sa.Column('refetched', sa.Boolean(), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='pending', nullable=False),
|
||||
sa.Column('recovery_count', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('refetched', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('result_image_id', sa.Integer(), nullable=True),
|
||||
sa.Column('error', sa.Text(), nullable=True),
|
||||
sa.Column('size_bytes', sa.BigInteger(), nullable=True),
|
||||
@@ -678,6 +718,8 @@ def upgrade() -> None:
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_import_task'))
|
||||
)
|
||||
op.create_index(op.f('ix_import_task_batch_id'), 'import_task', ['batch_id'], unique=False)
|
||||
op.create_index('ix_import_task_created_at_desc', 'import_task', [sa.literal_column('created_at DESC')], unique=False)
|
||||
op.create_index('ix_import_task_result_image_id', 'import_task', ['result_image_id'], unique=False)
|
||||
op.create_index(op.f('ix_import_task_status'), 'import_task', ['status'], unique=False)
|
||||
op.create_table('presentation_review',
|
||||
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
||||
@@ -692,6 +734,9 @@ def upgrade() -> None:
|
||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_presentation_review_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_presentation_review'))
|
||||
)
|
||||
op.create_index('ix_presentation_review_conflict_tag_id', 'presentation_review', ['conflict_tag_id'], unique=False)
|
||||
op.create_index('ix_presentation_review_resolved_at', 'presentation_review', ['resolved_at'], unique=False)
|
||||
op.create_index('ix_presentation_review_tag_id', 'presentation_review', ['tag_id'], unique=False)
|
||||
op.create_table('series_page',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('series_tag_id', sa.Integer(), nullable=False),
|
||||
@@ -704,7 +749,7 @@ def upgrade() -> None:
|
||||
sa.ForeignKeyConstraint(['image_id'], ['image_record.id'], name=op.f('fk_series_page_image_id_image_record'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['series_tag_id'], ['tag.id'], name=op.f('fk_series_page_series_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_series_page')),
|
||||
sa.UniqueConstraint('image_id', name=op.f('uq_series_page_image_id'))
|
||||
sa.UniqueConstraint('image_id', name='uq_series_page_image')
|
||||
)
|
||||
op.create_index(op.f('ix_series_page_series_tag_id'), 'series_page', ['series_tag_id'], unique=False)
|
||||
op.create_table('tag_positive_confirmation',
|
||||
@@ -720,11 +765,11 @@ def upgrade() -> None:
|
||||
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('rejected_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_tag_suggestion_rejection_image_record_id_image_record'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_tag_suggestion_rejection_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name='fk_tsr_image_record_id_image_record', ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name='fk_tsr_tag_id_tag', ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_tag_suggestion_rejection'))
|
||||
)
|
||||
op.create_index(op.f('ix_tag_suggestion_rejection_tag_id'), 'tag_suggestion_rejection', ['tag_id'], unique=False)
|
||||
op.create_index('ix_tag_suggestion_rejection_tag', 'tag_suggestion_rejection', ['tag_id'], unique=False)
|
||||
op.create_table('character_prototype',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||
@@ -734,6 +779,7 @@ def upgrade() -> None:
|
||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_character_prototype_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_character_prototype'))
|
||||
)
|
||||
op.create_index(op.f('ix_character_prototype_region_id'), 'character_prototype', ['region_id'], unique=False)
|
||||
op.create_index(op.f('ix_character_prototype_tag_id'), 'character_prototype', ['tag_id'], unique=False)
|
||||
op.create_table('series_chapter',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
@@ -743,130 +789,48 @@ def upgrade() -> None:
|
||||
sa.Column('stated_part', sa.Integer(), nullable=True),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['anchor_page_id'], ['series_page.id'], name=op.f('fk_series_chapter_anchor_page_id_series_page'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['anchor_page_id'], ['series_page.id'], name='fk_series_chapter_anchor_page', ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['series_tag_id'], ['tag.id'], name=op.f('fk_series_chapter_series_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_series_chapter')),
|
||||
sa.UniqueConstraint('anchor_page_id', name=op.f('uq_series_chapter_anchor_page_id'))
|
||||
sa.UniqueConstraint('anchor_page_id', name='uq_series_chapter_anchor_page')
|
||||
)
|
||||
op.create_index(op.f('ix_series_chapter_series_tag_id'), 'series_chapter', ['series_tag_id'], unique=False)
|
||||
|
||||
# The HNSW index, item 3 above. Must match the query's cosine-distance
|
||||
# operator class or the planner will not use it.
|
||||
# The singleton settings rows. NOT schema — see the note above; the app
|
||||
# reads these with scalar_one() and never creates them, so a fresh
|
||||
# install without these two rows raises NoResultFound on first use.
|
||||
# From 0002 and 0003.
|
||||
op.execute("INSERT INTO import_settings (id) VALUES (1)")
|
||||
op.execute("INSERT INTO ml_settings (id) VALUES (1)")
|
||||
|
||||
# The three hygiene system tags, from 0075. These are PRODUCT data, not
|
||||
# operator configuration — 0075's own docstring says so: "the fix keys on
|
||||
# SYSTEM tags the product ships". The presentation and process auto-apply
|
||||
# sweeps look them up with scalar_one(), so without these rows those
|
||||
# features raise NoResultFound rather than degrading.
|
||||
#
|
||||
# 0075 adopted an existing same-name general tag before inserting, because
|
||||
# an operator might already have tagged `wip` by hand. That cannot happen
|
||||
# on the empty database this file runs against, but the guard is kept: it
|
||||
# costs nothing and makes the statement safe to re-run.
|
||||
for _name in ("wip", "banner", "editor screenshot"):
|
||||
op.execute(
|
||||
"CREATE INDEX ix_image_record_siglip_hnsw "
|
||||
"ON image_record USING hnsw (siglip_embedding vector_cosine_ops)"
|
||||
sa.text(
|
||||
"INSERT INTO tag (name, kind, is_system) "
|
||||
"SELECT :name, 'general', true WHERE NOT EXISTS ("
|
||||
" SELECT 1 FROM tag WHERE lower(name) = lower(:name)"
|
||||
")"
|
||||
).bindparams(name=_name)
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# Dropping image_record takes its indexes with it, so the HNSW index needs
|
||||
# no separate drop. The extensions are deliberately left in place: they are
|
||||
# database-scoped and something else may be using them.
|
||||
op.drop_index(op.f('ix_series_chapter_series_tag_id'), table_name='series_chapter')
|
||||
op.drop_table('series_chapter')
|
||||
op.drop_index(op.f('ix_character_prototype_tag_id'), table_name='character_prototype')
|
||||
op.drop_table('character_prototype')
|
||||
op.drop_index(op.f('ix_tag_suggestion_rejection_tag_id'), table_name='tag_suggestion_rejection')
|
||||
op.drop_table('tag_suggestion_rejection')
|
||||
op.drop_index(op.f('ix_tag_positive_confirmation_tag_id'), table_name='tag_positive_confirmation')
|
||||
op.drop_table('tag_positive_confirmation')
|
||||
op.drop_index(op.f('ix_series_page_series_tag_id'), table_name='series_page')
|
||||
op.drop_table('series_page')
|
||||
op.drop_table('presentation_review')
|
||||
op.drop_index(op.f('ix_import_task_status'), table_name='import_task')
|
||||
op.drop_index(op.f('ix_import_task_batch_id'), table_name='import_task')
|
||||
op.drop_table('import_task')
|
||||
op.drop_table('image_tag')
|
||||
op.drop_index(op.f('ix_image_region_image_record_id'), table_name='image_region')
|
||||
op.drop_table('image_region')
|
||||
op.drop_index(op.f('ix_image_provenance_source_id'), table_name='image_provenance')
|
||||
op.drop_index(op.f('ix_image_provenance_post_id'), table_name='image_provenance')
|
||||
op.drop_index(op.f('ix_image_provenance_image_record_id'), table_name='image_provenance')
|
||||
op.drop_index(op.f('ix_image_provenance_from_attachment_id'), table_name='image_provenance')
|
||||
op.drop_table('image_provenance')
|
||||
op.drop_index(op.f('ix_gpu_job_status'), table_name='gpu_job')
|
||||
op.drop_index('ix_gpu_job_pending', table_name='gpu_job', postgresql_where=sa.text("status = 'pending'"))
|
||||
op.drop_index('ix_gpu_job_leased_expires', table_name='gpu_job', postgresql_where=sa.text("status = 'leased'"))
|
||||
op.drop_index(op.f('ix_gpu_job_image_record_id'), table_name='gpu_job')
|
||||
op.drop_table('gpu_job')
|
||||
op.drop_index('uq_external_link_post_url', table_name='external_link')
|
||||
op.drop_index('ix_external_link_status', table_name='external_link')
|
||||
op.drop_index(op.f('ix_external_link_post_id'), table_name='external_link')
|
||||
op.drop_index(op.f('ix_external_link_artist_id'), table_name='external_link')
|
||||
op.drop_table('external_link')
|
||||
op.drop_index(op.f('ix_series_suggestion_status'), table_name='series_suggestion')
|
||||
op.drop_index(op.f('ix_series_suggestion_series_tag_id'), table_name='series_suggestion')
|
||||
op.drop_index(op.f('ix_series_suggestion_post_id'), table_name='series_suggestion')
|
||||
op.drop_table('series_suggestion')
|
||||
op.drop_index('uq_post_attachment_post_sha', table_name='post_attachment', postgresql_where=sa.text('post_id IS NOT NULL'))
|
||||
op.drop_index('uq_post_attachment_null_post_sha', table_name='post_attachment', postgresql_where=sa.text('post_id IS NULL'))
|
||||
op.drop_index(op.f('ix_post_attachment_sha256'), table_name='post_attachment')
|
||||
op.drop_index(op.f('ix_post_attachment_post_id'), table_name='post_attachment')
|
||||
op.drop_index(op.f('ix_post_attachment_artist_id'), table_name='post_attachment')
|
||||
op.drop_table('post_attachment')
|
||||
op.drop_index(op.f('ix_image_record_source_filehash'), table_name='image_record')
|
||||
op.drop_index(op.f('ix_image_record_sha256'), table_name='image_record')
|
||||
op.drop_index(op.f('ix_image_record_primary_post_id'), table_name='image_record')
|
||||
op.drop_index(op.f('ix_image_record_phash'), table_name='image_record')
|
||||
op.drop_index(op.f('ix_image_record_integrity_status'), table_name='image_record')
|
||||
op.drop_index(op.f('ix_image_record_artist_id'), table_name='image_record')
|
||||
op.drop_table('image_record')
|
||||
op.drop_index(op.f('ix_download_event_source_id'), table_name='download_event')
|
||||
op.drop_index(op.f('ix_download_event_post_id'), table_name='download_event')
|
||||
op.drop_table('download_event')
|
||||
op.drop_index(op.f('ix_subscribestar_seen_media_source_id'), table_name='subscribestar_seen_media')
|
||||
op.drop_table('subscribestar_seen_media')
|
||||
op.drop_index(op.f('ix_subscribestar_failed_media_source_id'), table_name='subscribestar_failed_media')
|
||||
op.drop_table('subscribestar_failed_media')
|
||||
op.drop_index(op.f('ix_post_source_id'), table_name='post')
|
||||
op.drop_index(op.f('ix_post_artist_id'), table_name='post')
|
||||
op.drop_table('post')
|
||||
op.drop_index(op.f('ix_pixiv_seen_media_source_id'), table_name='pixiv_seen_media')
|
||||
op.drop_table('pixiv_seen_media')
|
||||
op.drop_index(op.f('ix_pixiv_failed_media_source_id'), table_name='pixiv_failed_media')
|
||||
op.drop_table('pixiv_failed_media')
|
||||
op.drop_index(op.f('ix_patreon_seen_media_source_id'), table_name='patreon_seen_media')
|
||||
op.drop_table('patreon_seen_media')
|
||||
op.drop_index(op.f('ix_patreon_failed_media_source_id'), table_name='patreon_failed_media')
|
||||
op.drop_table('patreon_failed_media')
|
||||
op.drop_table('tag_head')
|
||||
op.drop_index(op.f('ix_tag_alias_canonical_tag_id'), table_name='tag_alias')
|
||||
op.drop_table('tag_alias')
|
||||
op.drop_index(op.f('ix_source_error_type'), table_name='source')
|
||||
op.drop_index(op.f('ix_source_artist_id'), table_name='source')
|
||||
op.drop_table('source')
|
||||
op.drop_index(op.f('ix_head_metrics_snapshot_tag_id'), table_name='head_metrics_snapshot')
|
||||
op.drop_index(op.f('ix_head_metrics_snapshot_snapshot_at'), table_name='head_metrics_snapshot')
|
||||
op.drop_table('head_metrics_snapshot')
|
||||
op.drop_table('head_metric')
|
||||
op.drop_table('ccip_prototype_state')
|
||||
op.drop_table('artist_visit')
|
||||
op.drop_index(op.f('ix_task_run_task_name'), table_name='task_run')
|
||||
op.drop_index(op.f('ix_task_run_status'), table_name='task_run')
|
||||
op.drop_index(op.f('ix_task_run_started_at'), table_name='task_run')
|
||||
op.drop_index(op.f('ix_task_run_queue'), table_name='task_run')
|
||||
op.drop_index(op.f('ix_task_run_finished_at'), table_name='task_run')
|
||||
op.drop_index(op.f('ix_task_run_celery_task_id'), table_name='task_run')
|
||||
op.drop_table('task_run')
|
||||
op.drop_index(op.f('ix_tag_fandom_id'), table_name='tag')
|
||||
op.drop_table('tag')
|
||||
op.drop_table('ml_settings')
|
||||
op.drop_index(op.f('ix_library_audit_run_status'), table_name='library_audit_run')
|
||||
op.drop_index(op.f('ix_library_audit_run_rule'), table_name='library_audit_run')
|
||||
op.drop_table('library_audit_run')
|
||||
op.drop_table('import_settings')
|
||||
op.drop_index(op.f('ix_import_batch_status'), table_name='import_batch')
|
||||
op.drop_table('import_batch')
|
||||
op.drop_index(op.f('ix_head_training_run_status'), table_name='head_training_run')
|
||||
op.drop_table('head_training_run')
|
||||
op.drop_index(op.f('ix_head_auto_apply_run_status'), table_name='head_auto_apply_run')
|
||||
op.drop_table('head_auto_apply_run')
|
||||
op.drop_table('credential')
|
||||
op.drop_index(op.f('ix_backup_run_tag'), table_name='backup_run')
|
||||
op.drop_index(op.f('ix_backup_run_status'), table_name='backup_run')
|
||||
op.drop_index(op.f('ix_backup_run_started_at'), table_name='backup_run')
|
||||
op.drop_index(op.f('ix_backup_run_kind'), table_name='backup_run')
|
||||
op.drop_index(op.f('ix_backup_run_finished_at'), table_name='backup_run')
|
||||
op.drop_table('backup_run')
|
||||
op.drop_table('artist')
|
||||
op.drop_table('app_setting')
|
||||
"""Deliberately not implemented.
|
||||
|
||||
Downgrading a baseline means dropping every table in the database. That is
|
||||
not a migration, and offering it as one invites someone to run it. Restore
|
||||
from a backup instead.
|
||||
"""
|
||||
raise NotImplementedError(
|
||||
"0089 is the baseline; there is nothing below it. Restore from a backup."
|
||||
)
|
||||
@@ -0,0 +1,64 @@
|
||||
"""service_seen — the learned roster that makes a stopped part observable.
|
||||
|
||||
Milestone 365. Nothing in FabledCurator knew what was SUPPOSED to be running:
|
||||
`celery inspect` reports the workers that answer, so a dead worker was a
|
||||
shorter list rather than a red light, and the only surface that could tell an
|
||||
operator otherwise was Portainer. This table is the memory that turns an
|
||||
absence into something the app can see.
|
||||
|
||||
Keyed on the queue set for a celery role and on agent_id for the GPU agent —
|
||||
NOT on the celery worker name, which here is `celery@<container id>` and is
|
||||
minted fresh on every deploy. See the model docstring for why that choice is
|
||||
the whole design.
|
||||
|
||||
## First migration on the collapsed baseline
|
||||
|
||||
0089 is the single generated baseline that replaced revisions 0001..0089
|
||||
(milestone 328). This is the first revision written on top of it, so it is
|
||||
also the first evidence that the chain steps forward from the collapse rather
|
||||
than merely reproducing the schema — which nothing had demonstrated yet.
|
||||
|
||||
An existing install is at 0089 because it ran the real 0089; a fresh one is at
|
||||
0089 because it ran the baseline. Both arrive here identically, which was the
|
||||
property the collapse was designed around.
|
||||
|
||||
Revision ID: 0090
|
||||
Revises: 0089
|
||||
Create Date: 2026-09-02
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0090"
|
||||
down_revision: Union[str, None] = "0089"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"service_seen",
|
||||
sa.Column("key", sa.String(length=128), nullable=False),
|
||||
sa.Column("kind", sa.String(length=16), nullable=False),
|
||||
sa.Column("display_name", sa.String(length=64), nullable=False),
|
||||
sa.Column(
|
||||
"first_seen_at", sa.DateTime(timezone=True),
|
||||
server_default=sa.text("now()"), nullable=False,
|
||||
),
|
||||
sa.Column(
|
||||
"last_seen_at", sa.DateTime(timezone=True),
|
||||
server_default=sa.text("now()"), nullable=False,
|
||||
),
|
||||
sa.Column("details", sa.JSON(), nullable=False),
|
||||
sa.PrimaryKeyConstraint("key", name=op.f("pk_service_seen")),
|
||||
)
|
||||
# No secondary indexes, deliberately: one row per moving part means every
|
||||
# read is a handful of rows and an index would be write cost buying
|
||||
# nothing (#3301 removed seven of exactly that shape).
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_table("service_seen")
|
||||
@@ -38,6 +38,7 @@ def all_blueprints() -> list[Blueprint]:
|
||||
from .suggestions import suggestions_bp
|
||||
from .system_activity import system_activity_bp
|
||||
from .system_backup import system_backup_bp
|
||||
from .system_health import system_health_bp
|
||||
from .tags import tags_bp
|
||||
from .thumbnails import thumbnails_bp
|
||||
return [
|
||||
@@ -51,6 +52,7 @@ def all_blueprints() -> list[Blueprint]:
|
||||
showcase_bp,
|
||||
settings_bp,
|
||||
system_activity_bp,
|
||||
system_health_bp,
|
||||
system_backup_bp,
|
||||
admin_bp,
|
||||
cleanup_bp,
|
||||
|
||||
@@ -21,6 +21,7 @@ from ..services.gallery_service import image_url
|
||||
from ..services.ml.gpu_jobs import GpuJobService, error_dedupe_statements
|
||||
from ..services.ml.gpu_triage import classify_reason, recover_defective_image
|
||||
from ..services.ml.regions import RegionService
|
||||
from ..services.service_roster import touch_service
|
||||
|
||||
gpu_bp = Blueprint("gpu", __name__, url_prefix="/api/gpu")
|
||||
|
||||
@@ -256,6 +257,18 @@ async def lease():
|
||||
if not await _agent_authed(session):
|
||||
return jsonify({"error": "unauthorized"}), 401
|
||||
jobs = await GpuJobService(session).lease(agent_id, batch_size=batch)
|
||||
# The agent cannot be polled — it is HTTP-only and pulls from here, so
|
||||
# web never dials it. A lease IS the check-in, and until milestone 365
|
||||
# it was thrown away: an agent sitting idle with nothing to lease left
|
||||
# no trace at all and was indistinguishable from one switched off a
|
||||
# week ago. Recorded on the call that was already happening.
|
||||
await touch_service(
|
||||
session,
|
||||
key=f"agent:{agent_id}",
|
||||
kind="agent",
|
||||
display_name="GPU agent" if agent_id == "agent" else f"GPU agent ({agent_id})",
|
||||
details={"agent_id": agent_id, "last_call": "lease", "leased": len(jobs)},
|
||||
)
|
||||
ml = await MLSettings.load(session)
|
||||
# image rows for url/mime in one shot
|
||||
ids = [j.image_record_id for j in jobs]
|
||||
@@ -329,6 +342,13 @@ async def heartbeat():
|
||||
if not await _agent_authed(session):
|
||||
return jsonify({"error": "unauthorized"}), 401
|
||||
n = await GpuJobService(session).heartbeat(agent_id, job_ids)
|
||||
await touch_service(
|
||||
session,
|
||||
key=f"agent:{agent_id}",
|
||||
kind="agent",
|
||||
display_name="GPU agent" if agent_id == "agent" else f"GPU agent ({agent_id})",
|
||||
details={"agent_id": agent_id, "last_call": "heartbeat", "extended": n},
|
||||
)
|
||||
await session.commit()
|
||||
return jsonify({"extended": n})
|
||||
|
||||
|
||||
@@ -0,0 +1,192 @@
|
||||
"""Is every part of FabledCurator running? One verdict, one endpoint.
|
||||
|
||||
Milestone 365. The nav indicator and the System page both read this and
|
||||
nothing else — composing a verdict is this module's job, not the UI's.
|
||||
|
||||
## Two kinds of part, answered two different ways
|
||||
|
||||
**Learned** — celery roles and the GPU agent, from `service_seen`. The
|
||||
question is "how long since it checked in", and these are the parts that can
|
||||
be ABSENT, which is the whole point: `celery inspect` alone reports presence,
|
||||
so a dead worker is a shorter list rather than a red light.
|
||||
|
||||
**Probed live** — Postgres and Redis. Always expected, never learned, and a
|
||||
last-seen for them would be actively misleading: that Redis answered thirty
|
||||
seconds ago says nothing about now.
|
||||
|
||||
## This endpoint must never fail because something it checks has failed
|
||||
|
||||
The inversion is easy to write by accident and it destroys the feature exactly
|
||||
when it is needed — a 500 when Redis is down, instead of `redis: down`. Every
|
||||
probe is wrapped, every wait has a deadline (rule 156), and the roster refresh
|
||||
swallows its own errors. The worst case is a part reported `unknown`, which is
|
||||
a true statement.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import time
|
||||
from datetime import UTC, datetime
|
||||
|
||||
from quart import Blueprint, jsonify
|
||||
from sqlalchemy import select, text
|
||||
|
||||
from ..config import get_config
|
||||
from ..extensions import get_session
|
||||
from ..models import ServiceSeen
|
||||
from ..services.service_roster import refresh_if_stale
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
system_health_bp = Blueprint("system_health", __name__, url_prefix="/api/system")
|
||||
|
||||
# How long a learned part may go quiet before it is doubted, then disbelieved.
|
||||
#
|
||||
# These are deliberately generous, and the reason is a deploy rather than a
|
||||
# worker: `docker compose up -d` rolls start-first, so a role is briefly served
|
||||
# by two containers and then by neither while the old one drains. Thresholds
|
||||
# tight enough to catch a crash in seconds would paint the page red every time
|
||||
# the stack is updated, and an alarm that cries wolf on every deploy is one
|
||||
# nobody reads. Tune down only after watching a real deploy pass through.
|
||||
STALE_AFTER_SECONDS = 90
|
||||
DOWN_AFTER_SECONDS = 300
|
||||
|
||||
# Probes cross a process boundary, so they carry deadlines. A hung Postgres
|
||||
# must make this endpoint say "postgres: down", not hang alongside it.
|
||||
PROBE_TIMEOUT_SECONDS = 2.0
|
||||
|
||||
_OK, _STALE, _DOWN, _UNKNOWN = "ok", "stale", "down", "unknown"
|
||||
|
||||
# Worst-first, so an overall verdict is just the max.
|
||||
_SEVERITY = {_OK: 0, _UNKNOWN: 1, _STALE: 2, _DOWN: 3}
|
||||
|
||||
|
||||
def _age_state(age_seconds: float) -> str:
|
||||
if age_seconds >= DOWN_AFTER_SECONDS:
|
||||
return _DOWN
|
||||
if age_seconds >= STALE_AFTER_SECONDS:
|
||||
return _STALE
|
||||
return _OK
|
||||
|
||||
|
||||
def _describe_learned(name: str, state: str, age: float, details: dict) -> str:
|
||||
"""Say what the state MEANS. A red chip tells an operator less than a
|
||||
sentence does at the moment they are deciding whether to go and look."""
|
||||
if state == _OK:
|
||||
replicas = details.get("replicas")
|
||||
if replicas and replicas > 1:
|
||||
return f"{name} is running ({replicas} replicas)"
|
||||
return f"{name} is running"
|
||||
mins = int(age // 60)
|
||||
ago = f"{mins} min" if mins else f"{int(age)}s"
|
||||
if state == _STALE:
|
||||
return f"{name} has not checked in for {ago}"
|
||||
return f"{name} has not checked in for {ago} — treat it as stopped"
|
||||
|
||||
|
||||
async def _probe_postgres(session) -> dict:
|
||||
started = time.monotonic()
|
||||
try:
|
||||
await asyncio.wait_for(
|
||||
session.execute(text("SELECT 1")), timeout=PROBE_TIMEOUT_SECONDS
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 — a probe reports, it never raises
|
||||
return {
|
||||
"key": "postgres", "kind": "datastore", "name": "PostgreSQL",
|
||||
"state": _DOWN, "detail": f"not answering: {type(exc).__name__}",
|
||||
}
|
||||
return {
|
||||
"key": "postgres", "kind": "datastore", "name": "PostgreSQL", "state": _OK,
|
||||
"detail": "answering", "latency_ms": round((time.monotonic() - started) * 1000, 1),
|
||||
}
|
||||
|
||||
|
||||
def _ping_redis_sync() -> None:
|
||||
import redis # local import; mirrors system_activity's pattern
|
||||
|
||||
client = redis.Redis.from_url(
|
||||
get_config().celery_broker_url,
|
||||
socket_connect_timeout=PROBE_TIMEOUT_SECONDS,
|
||||
socket_timeout=PROBE_TIMEOUT_SECONDS,
|
||||
)
|
||||
client.ping()
|
||||
|
||||
|
||||
async def _probe_redis() -> dict:
|
||||
started = time.monotonic()
|
||||
try:
|
||||
await asyncio.wait_for(
|
||||
asyncio.to_thread(_ping_redis_sync), timeout=PROBE_TIMEOUT_SECONDS * 2
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {
|
||||
"key": "redis", "kind": "datastore", "name": "Redis",
|
||||
"state": _DOWN,
|
||||
"detail": f"not answering: {type(exc).__name__} — queues and workers "
|
||||
f"cannot be reached either",
|
||||
}
|
||||
return {
|
||||
"key": "redis", "kind": "datastore", "name": "Redis", "state": _OK,
|
||||
"detail": "answering", "latency_ms": round((time.monotonic() - started) * 1000, 1),
|
||||
}
|
||||
|
||||
|
||||
@system_health_bp.route("/health", methods=["GET"])
|
||||
async def system_health():
|
||||
"""Every part, its state, and one overall verdict.
|
||||
|
||||
Response: {overall, parts: [{key, kind, name, state, detail, last_seen_at,
|
||||
…}], checked_at}
|
||||
"""
|
||||
parts: list[dict] = []
|
||||
now = datetime.now(UTC)
|
||||
|
||||
async with get_session() as session:
|
||||
# Postgres first, and if it is unreachable nothing else can be read —
|
||||
# say so rather than failing, because "the database is down" is the
|
||||
# single most useful thing this endpoint can ever report.
|
||||
pg = await _probe_postgres(session)
|
||||
parts.append(pg)
|
||||
|
||||
if pg["state"] == _OK:
|
||||
# Rate-limited inside; see service_roster on why the web process
|
||||
# is the right observer.
|
||||
try:
|
||||
await refresh_if_stale(session)
|
||||
await session.commit()
|
||||
except Exception: # noqa: BLE001
|
||||
log.warning("system health: roster refresh failed", exc_info=True)
|
||||
|
||||
rows = (
|
||||
await session.execute(select(ServiceSeen).order_by(ServiceSeen.display_name))
|
||||
).scalars().all()
|
||||
for row in rows:
|
||||
age = (now - row.last_seen_at).total_seconds()
|
||||
state = _age_state(age)
|
||||
parts.append({
|
||||
"key": row.key,
|
||||
"kind": row.kind,
|
||||
"name": row.display_name,
|
||||
"state": state,
|
||||
"detail": _describe_learned(row.display_name, state, age, row.details or {}),
|
||||
"last_seen_at": row.last_seen_at.isoformat(),
|
||||
"first_seen_at": row.first_seen_at.isoformat(),
|
||||
**{k: v for k, v in (row.details or {}).items() if k != "agent_id"},
|
||||
})
|
||||
|
||||
parts.append(await _probe_redis())
|
||||
|
||||
overall = max((p["state"] for p in parts), key=lambda s: _SEVERITY[s], default=_UNKNOWN)
|
||||
return jsonify({
|
||||
"overall": overall,
|
||||
"parts": sorted(parts, key=lambda p: (-_SEVERITY[p["state"]], p["name"])),
|
||||
"checked_at": now.isoformat(),
|
||||
# So the UI can explain a `stale` without hard-coding the same numbers
|
||||
# in a second place.
|
||||
"thresholds": {
|
||||
"stale_after_seconds": STALE_AFTER_SECONDS,
|
||||
"down_after_seconds": DOWN_AFTER_SECONDS,
|
||||
},
|
||||
})
|
||||
@@ -16,8 +16,11 @@ class Config:
|
||||
celery_broker_url: str
|
||||
celery_result_backend: str
|
||||
|
||||
# Sets Quart's app.secret_key. Nothing signs a cookie today (FC has no
|
||||
# login and no session use), so this currently protects nothing — it is
|
||||
# required rather than defaulted so that the day something session-backed
|
||||
# does land, no instance is already running on a value we published.
|
||||
secret_key: str
|
||||
extension_api_key: str # used by the Firefox extension; lands in FC-3 but read here
|
||||
log_level: str
|
||||
|
||||
@property
|
||||
@@ -47,6 +50,5 @@ def get_config() -> Config:
|
||||
celery_broker_url=os.environ.get("CELERY_BROKER_URL", "redis://redis:6379/0"),
|
||||
celery_result_backend=os.environ.get("CELERY_RESULT_BACKEND", "redis://redis:6379/0"),
|
||||
secret_key=os.environ["SECRET_KEY"],
|
||||
extension_api_key=os.environ.get("EXTENSION_API_KEY", ""),
|
||||
log_level=os.environ.get("LOG_LEVEL", "INFO"),
|
||||
)
|
||||
|
||||
@@ -32,6 +32,7 @@ from .presentation_review import PresentationReview
|
||||
from .series_chapter import SeriesChapter
|
||||
from .series_page import SeriesPage
|
||||
from .series_suggestion import SeriesSuggestion
|
||||
from .service_seen import ServiceSeen
|
||||
from .source import Source
|
||||
from .subscribestar_failed_media import SubscribeStarFailedMedia
|
||||
from .subscribestar_seen_media import SubscribeStarSeenMedia
|
||||
@@ -63,6 +64,7 @@ __all__ = [
|
||||
"SeriesChapter",
|
||||
"SeriesPage",
|
||||
"SeriesSuggestion",
|
||||
"ServiceSeen",
|
||||
"ImageRecord",
|
||||
"ImageProvenance",
|
||||
"ImageRegion",
|
||||
|
||||
@@ -27,10 +27,10 @@ class Artist(Base):
|
||||
notes: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||
|
||||
# True once a Source is attached; flips false if all sources removed.
|
||||
is_subscription: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False)
|
||||
is_subscription: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, server_default="false")
|
||||
|
||||
# Per-artist scheduling overrides; null means "use global default".
|
||||
auto_check: Mapped[bool] = mapped_column(Boolean, nullable=False, default=True)
|
||||
auto_check: Mapped[bool] = mapped_column(Boolean, nullable=False, default=True, server_default="true")
|
||||
check_interval_seconds: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
||||
|
||||
created_at: Mapped[datetime] = mapped_column(
|
||||
|
||||
@@ -20,7 +20,7 @@ feedback_check_existing_enums):
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
from sqlalchemy import JSON, BigInteger, DateTime, ForeignKey, Integer, String, Text
|
||||
from sqlalchemy import JSON, BigInteger, DateTime, ForeignKey, Index, Integer, String, Text, text
|
||||
from sqlalchemy.orm import Mapped, mapped_column
|
||||
|
||||
from .base import Base
|
||||
@@ -29,10 +29,21 @@ from .base import Base
|
||||
class BackupRun(Base):
|
||||
__tablename__ = "backup_run"
|
||||
|
||||
|
||||
__table_args__ = (
|
||||
# alembic 0017: reporting indexes, never declared on the model (#3275).
|
||||
Index("ix_backup_run_kind_started", "kind", text("started_at DESC")),
|
||||
Index("ix_backup_run_status_finished", "status", text("finished_at DESC")),
|
||||
Index("ix_backup_run_tag_partial", "tag", postgresql_where=text("tag IS NOT NULL")),
|
||||
)
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
kind: Mapped[str] = mapped_column(String(16), nullable=False, index=True)
|
||||
# No index=True: ix_backup_run_kind_started (above) already leads with
|
||||
# `kind`, so a single-column index on it was pure write cost (#3301).
|
||||
kind: Mapped[str] = mapped_column(String(16), nullable=False)
|
||||
status: Mapped[str] = mapped_column(
|
||||
String(16), nullable=False, default="pending", index=True,
|
||||
# No index=True — ix_backup_run_status_finished leads with `status`.
|
||||
String(16), nullable=False, default="pending",
|
||||
server_default="pending",
|
||||
)
|
||||
tag: Mapped[str | None] = mapped_column(String(64), nullable=True, index=True)
|
||||
triggered_by: Mapped[str] = mapped_column(String(32), nullable=False)
|
||||
@@ -49,7 +60,9 @@ class BackupRun(Base):
|
||||
manifest: Mapped[dict] = mapped_column(
|
||||
JSON, nullable=False, default=dict, server_default="{}",
|
||||
)
|
||||
# Self-referential FK, unindexed until 0089 (#3300): SET NULL has to find
|
||||
# the rows pointing at a deleted run before it can null them.
|
||||
restored_from_id: Mapped[int | None] = mapped_column(
|
||||
ForeignKey("backup_run.id", ondelete="SET NULL"),
|
||||
nullable=True,
|
||||
nullable=True, index=True,
|
||||
)
|
||||
|
||||
@@ -40,8 +40,10 @@ class CharacterPrototype(Base):
|
||||
)
|
||||
# Provenance: the region this vector was copied from. SET NULL so pruning a
|
||||
# region doesn't delete the prototype mid-cycle (the next refresh reconciles).
|
||||
# index=True added in 0089 — the FK was unindexed (#3300).
|
||||
region_id: Mapped[int | None] = mapped_column(
|
||||
ForeignKey("image_region.id", ondelete="SET NULL"), nullable=True
|
||||
ForeignKey("image_region.id", ondelete="SET NULL"), nullable=True,
|
||||
index=True,
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -25,8 +25,8 @@ class DownloadEvent(Base):
|
||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||
)
|
||||
finished_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
|
||||
bytes_downloaded: Mapped[int] = mapped_column(BigInteger, nullable=False, default=0)
|
||||
files_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
bytes_downloaded: Mapped[int] = mapped_column(BigInteger, nullable=False, default=0, server_default="0")
|
||||
files_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||
metadata_: Mapped[dict] = mapped_column(
|
||||
"metadata", JSONB, nullable=False, default=dict,
|
||||
|
||||
@@ -16,6 +16,7 @@ doesn't delete the link record).
|
||||
from datetime import datetime
|
||||
|
||||
from sqlalchemy import (
|
||||
CheckConstraint,
|
||||
DateTime,
|
||||
Float,
|
||||
ForeignKey,
|
||||
@@ -38,15 +39,33 @@ STATUSES = ("pending", "downloading", "downloaded", "failed", "skipped", "dead")
|
||||
class ExternalLink(Base):
|
||||
__tablename__ = "external_link"
|
||||
__table_args__ = (
|
||||
# alembic 0028 enum CHECKs. Rule 36 territory: a new host or status value
|
||||
# needs its constraint swapped in the same migration (#3275).
|
||||
CheckConstraint(
|
||||
"host IN ('mega', 'gdrive', 'mediafire', 'dropbox', 'pixeldrain')",
|
||||
# Bare name: Base.metadata's naming convention prepends
|
||||
# ck_<table>_. Pre-prefixing it here doubles the prefix — see
|
||||
# alembic 0088, which renames the four constraints that shipped
|
||||
# that way (#3275).
|
||||
name="host",
|
||||
),
|
||||
CheckConstraint(
|
||||
"status IN ('pending', 'downloading', 'downloaded', 'failed', 'skipped', 'dead')",
|
||||
name="status",
|
||||
),
|
||||
# One row per (post, url). The full url (incl. #fragment) is the identity
|
||||
# — the same file linked twice in a post collapses to one row.
|
||||
Index("uq_external_link_post_url", "post_id", "url", unique=True),
|
||||
Index("ix_external_link_status", "status"),
|
||||
# Unindexed FK (#3300).
|
||||
Index("ix_external_link_attachment_id", "attachment_id"),
|
||||
)
|
||||
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
# No index=True: uq_external_link_post_url (post_id, url) already leads
|
||||
# with post_id (#3301).
|
||||
post_id: Mapped[int] = mapped_column(
|
||||
ForeignKey("post.id", ondelete="CASCADE"), nullable=False, index=True
|
||||
ForeignKey("post.id", ondelete="CASCADE"), nullable=False
|
||||
)
|
||||
artist_id: Mapped[int | None] = mapped_column(
|
||||
ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, index=True
|
||||
|
||||
@@ -50,7 +50,8 @@ class GpuJob(Base):
|
||||
# What to compute, e.g. 'ccip' (detect figures + CCIP-embed) or 'siglip_region'.
|
||||
task: Mapped[str] = mapped_column(String(32), nullable=False)
|
||||
status: Mapped[str] = mapped_column(
|
||||
String(16), nullable=False, default="pending", index=True
|
||||
String(16), nullable=False, default="pending", index=True,
|
||||
server_default="pending",
|
||||
)
|
||||
# pending | leased | done | error
|
||||
lease_token: Mapped[str | None] = mapped_column(String(64), nullable=True)
|
||||
@@ -60,7 +61,7 @@ class GpuJob(Base):
|
||||
lease_expires_at: Mapped[datetime | None] = mapped_column(
|
||||
DateTime(timezone=True), nullable=True
|
||||
)
|
||||
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||
# Triage verdict for an ERRORED job (#125): NULL = not yet probed;
|
||||
# 'defect' = the integrity probe says the FILE itself is bad (surfaced for
|
||||
|
||||
@@ -24,10 +24,11 @@ class HeadAutoApplyRun(Base):
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
# dry_run=True is a PREVIEW: scores + counts what WOULD apply, writes nothing
|
||||
# (preview/apply parity, rule 93).
|
||||
dry_run: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False)
|
||||
dry_run: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, server_default="false")
|
||||
params: Mapped[dict[str, Any]] = mapped_column(JSONB, nullable=False)
|
||||
status: Mapped[str] = mapped_column(
|
||||
String(16), nullable=False, default="running", index=True
|
||||
String(16), nullable=False, default="running", index=True,
|
||||
server_default="running",
|
||||
)
|
||||
# running | ready | error
|
||||
started_at: Mapped[datetime] = mapped_column(
|
||||
|
||||
@@ -24,9 +24,9 @@ class HeadMetric(Base):
|
||||
ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True
|
||||
)
|
||||
# An auto-applied (source='head_auto') tag the operator later REMOVED.
|
||||
n_misfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
n_misfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
# A tag with a head that the operator added by HAND (the head missed it).
|
||||
n_underfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
n_underfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
updated_at: Mapped[datetime] = mapped_column(
|
||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||
)
|
||||
|
||||
@@ -19,8 +19,14 @@ class HeadMetricsSnapshot(Base):
|
||||
__tablename__ = "head_metrics_snapshot"
|
||||
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
tag_id: Mapped[int] = mapped_column(
|
||||
ForeignKey("tag.id", ondelete="CASCADE"), index=True
|
||||
# Nullable, matching alembic 0060, which declared this column without
|
||||
# `nullable=False`. The model had it as `Mapped[int]` — NOT NULL — which
|
||||
# was simply never true of the database (#3275). Left nullable rather than
|
||||
# tightened: a snapshot of a tag that is later hard-deleted is a row worth
|
||||
# keeping, and the FK is ON DELETE CASCADE, so tightening it would only
|
||||
# change behaviour, not correct a bug.
|
||||
tag_id: Mapped[int | None] = mapped_column(
|
||||
ForeignKey("tag.id", ondelete="CASCADE"), nullable=True, index=True
|
||||
)
|
||||
# Denormalized so a snapshot stays readable even if the tag is later renamed.
|
||||
name: Mapped[str] = mapped_column(String(255), nullable=False)
|
||||
@@ -28,9 +34,9 @@ class HeadMetricsSnapshot(Base):
|
||||
DateTime(timezone=True), nullable=False, server_default=func.now(), index=True
|
||||
)
|
||||
# Current count of source='head_auto' applications still standing.
|
||||
n_auto_applied: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
n_misfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
n_underfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
n_auto_applied: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
n_misfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
n_underfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
# The head's measured quality at snapshot time (null if no head exists).
|
||||
ap: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||
precision_cv: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||
|
||||
@@ -24,7 +24,8 @@ class HeadTrainingRun(Base):
|
||||
# Training parameters: {min_positives, neg_ratio, precision_target, ...}.
|
||||
params: Mapped[dict[str, Any]] = mapped_column(JSONB, nullable=False)
|
||||
status: Mapped[str] = mapped_column(
|
||||
String(16), nullable=False, default="running", index=True
|
||||
String(16), nullable=False, default="running", index=True,
|
||||
server_default="running",
|
||||
)
|
||||
# running | ready | error
|
||||
started_at: Mapped[datetime] = mapped_column(
|
||||
|
||||
@@ -47,8 +47,15 @@ class ImageProvenance(Base):
|
||||
# attachment on the post. NULL for loose downloads and pre-backfill rows.
|
||||
# SET NULL so deleting the archive attachment never destroys the (image,
|
||||
# post) edge — it just forgets which archive it came from.
|
||||
# FK named explicitly: the convention renders this
|
||||
# `fk_image_provenance_from_attachment_id_post_attachment`, but alembic
|
||||
# 0055 created it as `fk_image_provenance_from_attachment` (#3275).
|
||||
from_attachment_id: Mapped[int | None] = mapped_column(
|
||||
ForeignKey("post_attachment.id", ondelete="SET NULL"),
|
||||
ForeignKey(
|
||||
"post_attachment.id",
|
||||
ondelete="SET NULL",
|
||||
name="fk_image_provenance_from_attachment",
|
||||
),
|
||||
nullable=True, index=True,
|
||||
)
|
||||
captured_metadata: Mapped[dict | None] = mapped_column(JSON, nullable=True)
|
||||
|
||||
@@ -14,10 +14,13 @@ from sqlalchemy import (
|
||||
Enum,
|
||||
Float,
|
||||
ForeignKey,
|
||||
Index,
|
||||
Integer,
|
||||
String,
|
||||
Text,
|
||||
UniqueConstraint,
|
||||
func,
|
||||
text,
|
||||
)
|
||||
from sqlalchemy.orm import Mapped, mapped_column
|
||||
|
||||
@@ -29,11 +32,38 @@ ORIGIN_CHOICES = ("downloaded", "imported_filesystem", "uploaded")
|
||||
class ImageRecord(Base):
|
||||
__tablename__ = "image_record"
|
||||
|
||||
|
||||
__table_args__ = (
|
||||
# alembic 0001. The database enforces sha256 uniqueness with a
|
||||
# CONSTRAINT and carries a SEPARATE non-unique btree index; the model
|
||||
# said `unique=True, index=True`, which collapses both into a single
|
||||
# UNIQUE index under a different name. Same guarantee either way, but
|
||||
# not the same objects, so autogenerate saw a drop and an add (#3275).
|
||||
UniqueConstraint("sha256", name="uq_image_record_sha256"),
|
||||
# alembic 0036, and the last thing in this schema that lived only in a
|
||||
# migration. SQLAlchemy CAN express an hnsw index with an operator
|
||||
# class, so there is no reason for it to be invisible to the models —
|
||||
# and its absence was the quietest failure of the lot: everything
|
||||
# works, similarity search just silently stops using an index.
|
||||
Index(
|
||||
"ix_image_record_siglip_hnsw",
|
||||
"siglip_embedding",
|
||||
postgresql_using="hnsw",
|
||||
postgresql_ops={"siglip_embedding": "vector_cosine_ops"},
|
||||
),
|
||||
# alembic 0035/0071: the date-ordered browse indexes (#3275).
|
||||
Index("ix_image_record_effective_date", text("effective_date DESC"), text("id DESC")),
|
||||
Index("ix_image_record_earliest_post_date", text("earliest_post_date DESC"), text("id DESC")),
|
||||
)
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
|
||||
# On-disk identity
|
||||
path: Mapped[str] = mapped_column(Text, nullable=False, unique=True)
|
||||
sha256: Mapped[str] = mapped_column(String(64), nullable=False, unique=True, index=True)
|
||||
# Neither unique= nor index=: uq_image_record_sha256 in __table_args__
|
||||
# above creates its own index, and the separate ix_image_record_sha256
|
||||
# that 0001 also built was an exact duplicate of it — dropped in 0089
|
||||
# (#3301). Lookups by sha256 use the constraint's index.
|
||||
sha256: Mapped[str] = mapped_column(String(64), nullable=False)
|
||||
phash: Mapped[str | None] = mapped_column(String(32), nullable=True, index=True)
|
||||
size_bytes: Mapped[int] = mapped_column(BigInteger, nullable=False)
|
||||
mime: Mapped[str] = mapped_column(String(64), nullable=False)
|
||||
@@ -47,7 +77,8 @@ class ImageRecord(Base):
|
||||
# Integrity verification status. FC-2e populates this; FC-2a leaves rows at 'unknown'.
|
||||
# Values: 'unknown' (default), 'ok', 'corrupt', 'failed_verification'.
|
||||
integrity_status: Mapped[str] = mapped_column(
|
||||
String(24), nullable=False, default="unknown", index=True
|
||||
String(24), nullable=False, default="unknown", index=True,
|
||||
server_default="unknown",
|
||||
)
|
||||
|
||||
# Thumbnail (populated by FC-2)
|
||||
@@ -72,8 +103,15 @@ class ImageRecord(Base):
|
||||
)
|
||||
# FC-2d-vii-c: canonical per-image artist (the single source of truth
|
||||
# for attribution; provenance posts remain lineage detail).
|
||||
# FK named explicitly: the naming convention renders this
|
||||
# `fk_image_record_artist_id_artist`, but alembic 0008 created it as
|
||||
# `fk_image_record_artist_id` (#3275).
|
||||
artist_id: Mapped[int | None] = mapped_column(
|
||||
ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, index=True
|
||||
ForeignKey(
|
||||
"artist.id", ondelete="SET NULL", name="fk_image_record_artist_id"
|
||||
),
|
||||
nullable=True,
|
||||
index=True,
|
||||
)
|
||||
|
||||
# ML fields (populated by the ml-worker / GPU agent). 1152 = SigLIP-so400m
|
||||
|
||||
@@ -21,17 +21,17 @@ class ImportBatch(Base):
|
||||
)
|
||||
finished_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
|
||||
|
||||
total_files: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
imported: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
skipped: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
failed: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
attachments: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
total_files: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
imported: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
skipped: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
failed: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
attachments: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
# Deep-scan only: count of already-imported files whose sidecar metadata
|
||||
# got re-applied this run (post/source/provenance upsert). Stays 0 on
|
||||
# quick-scan batches. See `Importer.import_one(deep_scan=True)`.
|
||||
refreshed: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
refreshed: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
|
||||
status: Mapped[str] = mapped_column(String(16), nullable=False, default="running", index=True)
|
||||
status: Mapped[str] = mapped_column(String(16), nullable=False, default="running", index=True, server_default="running")
|
||||
# running | complete | cancelled
|
||||
|
||||
tasks = relationship("ImportTask", back_populates="batch", cascade="all, delete-orphan")
|
||||
|
||||
@@ -4,7 +4,15 @@ Enforced as a single row via a CHECK (id = 1) constraint. The application
|
||||
always SELECTs id=1 and never inserts/deletes after the initial migration.
|
||||
"""
|
||||
|
||||
from sqlalchemy import Boolean, CheckConstraint, Float, Integer, Text, select
|
||||
from sqlalchemy import (
|
||||
Boolean,
|
||||
CheckConstraint,
|
||||
Float,
|
||||
Integer,
|
||||
Text,
|
||||
select,
|
||||
text,
|
||||
)
|
||||
from sqlalchemy.orm import Mapped, mapped_column
|
||||
|
||||
from .base import Base
|
||||
@@ -14,63 +22,79 @@ class ImportSettings(Base):
|
||||
__tablename__ = "import_settings"
|
||||
# Bare constraint name — Base.metadata's naming convention applies the
|
||||
# ck_<table>_<name> prefix, producing the final ck_import_settings_singleton.
|
||||
# Bare name — Base.metadata's naming convention prepends ck_<table>_,
|
||||
# producing ck_import_settings_singleton. The chain shipped the DOUBLED
|
||||
# ck_import_settings_ck_import_settings_singleton, because the migration
|
||||
# pre-prefixed the name and the convention prefixed it again; alembic
|
||||
# 0088 renames it to what this line has always produced (#3275).
|
||||
__table_args__ = (CheckConstraint("id = 1", name="singleton"),)
|
||||
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
import_scan_path: Mapped[str] = mapped_column(Text, nullable=False, default="/import")
|
||||
import_scan_path: Mapped[str] = mapped_column(Text, nullable=False, default="/import", server_default="/import")
|
||||
|
||||
min_width: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
min_height: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
min_width: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
min_height: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
|
||||
skip_transparent: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False)
|
||||
transparency_threshold: Mapped[float] = mapped_column(Float, nullable=False, default=0.9)
|
||||
skip_transparent: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, server_default="false")
|
||||
transparency_threshold: Mapped[float] = mapped_column(Float, nullable=False, default=0.9, server_default="0.9")
|
||||
|
||||
skip_single_color: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False)
|
||||
single_color_threshold: Mapped[float] = mapped_column(Float, nullable=False, default=0.95)
|
||||
single_color_tolerance: Mapped[int] = mapped_column(Integer, nullable=False, default=30)
|
||||
skip_single_color: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, server_default="false")
|
||||
single_color_threshold: Mapped[float] = mapped_column(Float, nullable=False, default=0.95, server_default="0.95")
|
||||
single_color_tolerance: Mapped[int] = mapped_column(Integer, nullable=False, default=30, server_default="30")
|
||||
|
||||
phash_threshold: Mapped[int] = mapped_column(Integer, nullable=False, default=10)
|
||||
phash_threshold: Mapped[int] = mapped_column(Integer, nullable=False, default=10, server_default="10")
|
||||
|
||||
# FC-3c downloader knobs
|
||||
download_rate_limit_seconds: Mapped[float] = mapped_column(
|
||||
Float, nullable=False, default=3.0
|
||||
Float, nullable=False, default=3.0,
|
||||
server_default="3",
|
||||
)
|
||||
download_validate_files: Mapped[bool] = mapped_column(
|
||||
Boolean, nullable=False, default=True
|
||||
Boolean, nullable=False, default=True,
|
||||
server_default="true",
|
||||
)
|
||||
|
||||
# FC-3d scheduling knobs
|
||||
download_schedule_default_seconds: Mapped[int] = mapped_column(
|
||||
Integer, nullable=False, default=28800
|
||||
Integer, nullable=False, default=28800,
|
||||
server_default="28800",
|
||||
)
|
||||
download_event_retention_days: Mapped[int] = mapped_column(
|
||||
Integer, nullable=False, default=90
|
||||
Integer, nullable=False, default=90,
|
||||
server_default="90",
|
||||
)
|
||||
download_failure_warning_threshold: Mapped[int] = mapped_column(
|
||||
Integer, nullable=False, default=5
|
||||
Integer, nullable=False, default=5,
|
||||
server_default="5",
|
||||
)
|
||||
|
||||
# FC-3h backup knobs.
|
||||
backup_db_nightly_enabled: Mapped[bool] = mapped_column(
|
||||
Boolean, nullable=False, default=False,
|
||||
server_default="false",
|
||||
)
|
||||
backup_db_nightly_hour_utc: Mapped[int] = mapped_column(
|
||||
Integer, nullable=False, default=3,
|
||||
server_default="3",
|
||||
)
|
||||
backup_db_keep_last_n: Mapped[int] = mapped_column(
|
||||
Integer, nullable=False, default=14,
|
||||
server_default="14",
|
||||
)
|
||||
backup_images_keep_last_n: Mapped[int] = mapped_column(
|
||||
Integer, nullable=False, default=3,
|
||||
server_default="3",
|
||||
)
|
||||
|
||||
# FC-6.3 series continuation matcher. enabled gates the rescan; threshold is
|
||||
# the weighted-score cut-off (0..1) above which a pending suggestion is made.
|
||||
series_suggest_enabled: Mapped[bool] = mapped_column(
|
||||
Boolean, nullable=False, default=True,
|
||||
server_default="true",
|
||||
)
|
||||
series_suggest_threshold: Mapped[float] = mapped_column(
|
||||
Float, nullable=False, default=0.5,
|
||||
server_default="0.5",
|
||||
)
|
||||
|
||||
# #830 off-platform file-host downloads — per-host enable lever (default on,
|
||||
@@ -113,7 +137,9 @@ class ImportSettings(Base):
|
||||
# English (e.g. "… WIP Part 1") as a European language at ~0.86. CJK stays
|
||||
# trusted regardless (script-detected). Per-post overrides handle the misses.
|
||||
translation_min_confidence: Mapped[float] = mapped_column(
|
||||
Float, nullable=False, default=0.9, server_default="0.9",
|
||||
# text() because alembic 0084 used sa.text(); see ml_settings for why
|
||||
# the form matters and why it is per-column (#3275).
|
||||
Float, nullable=False, default=0.9, server_default=text("0.9"),
|
||||
)
|
||||
|
||||
# Title-based WIP auto-tagging (task #1458). When a freshly-imported post's
|
||||
|
||||
@@ -13,10 +13,12 @@ from sqlalchemy import (
|
||||
Boolean,
|
||||
DateTime,
|
||||
ForeignKey,
|
||||
Index,
|
||||
Integer,
|
||||
String,
|
||||
Text,
|
||||
func,
|
||||
text,
|
||||
)
|
||||
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
||||
|
||||
@@ -26,6 +28,12 @@ from .base import Base
|
||||
class ImportTask(Base):
|
||||
__tablename__ = "import_task"
|
||||
|
||||
|
||||
__table_args__ = (
|
||||
Index("ix_import_task_created_at_desc", text("created_at DESC")),
|
||||
# Unindexed FK (#3300).
|
||||
Index("ix_import_task_result_image_id", "result_image_id"),
|
||||
)
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
batch_id: Mapped[int] = mapped_column(
|
||||
ForeignKey("import_batch.id", ondelete="CASCADE"), nullable=False, index=True
|
||||
@@ -33,14 +41,14 @@ class ImportTask(Base):
|
||||
|
||||
source_path: Mapped[str] = mapped_column(Text, nullable=False)
|
||||
task_type: Mapped[str] = mapped_column(String(16), nullable=False) # media|archive
|
||||
status: Mapped[str] = mapped_column(String(16), nullable=False, default="pending", index=True)
|
||||
status: Mapped[str] = mapped_column(String(16), nullable=False, default="pending", index=True, server_default="pending")
|
||||
|
||||
# Poison-pill circuit breaker (alembic 0026). recovery_count tracks
|
||||
# how many times the stuck-task sweep has re-queued this row; after
|
||||
# the cap it's failed with a diagnostic instead of looping. refetched
|
||||
# bounds the one-shot re-download remediation to a single attempt.
|
||||
recovery_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
refetched: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False)
|
||||
recovery_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
refetched: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, server_default="false")
|
||||
|
||||
result_image_id: Mapped[int | None] = mapped_column(
|
||||
ForeignKey("image_record.id", ondelete="SET NULL"), nullable=True
|
||||
|
||||
@@ -8,7 +8,7 @@ reads it and routes through cleanup_service.delete_images.
|
||||
from datetime import datetime
|
||||
from typing import Any
|
||||
|
||||
from sqlalchemy import DateTime, Integer, String, Text, func
|
||||
from sqlalchemy import DateTime, Integer, String, Text, func, text
|
||||
from sqlalchemy.dialects.postgresql import JSONB
|
||||
from sqlalchemy.orm import Mapped, mapped_column
|
||||
|
||||
@@ -23,6 +23,7 @@ class LibraryAuditRun(Base):
|
||||
params: Mapped[dict[str, Any]] = mapped_column(JSONB, nullable=False)
|
||||
status: Mapped[str] = mapped_column(
|
||||
String(16), nullable=False, default="running", index=True,
|
||||
server_default="running",
|
||||
)
|
||||
# running | ready | applied | cancelled | error
|
||||
started_at: Mapped[datetime] = mapped_column(
|
||||
@@ -31,14 +32,16 @@ class LibraryAuditRun(Base):
|
||||
finished_at: Mapped[datetime | None] = mapped_column(
|
||||
DateTime(timezone=True), nullable=True,
|
||||
)
|
||||
scanned_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
matched_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
matched_ids: Mapped[list[int]] = mapped_column(JSONB, nullable=False, default=list)
|
||||
scanned_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
matched_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
matched_ids: Mapped[list[int]] = mapped_column(
|
||||
JSONB, nullable=False, default=list, server_default=text("'[]'::jsonb")
|
||||
)
|
||||
error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||
# Chunked-scan state (alembic 0039): keyset cursor the next chunk resumes
|
||||
# from, and the last time a chunk made progress (so the recovery sweep can
|
||||
# tell a progressing multi-chunk audit from a stuck one).
|
||||
resume_after_id: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
resume_after_id: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
last_progress_at: Mapped[datetime | None] = mapped_column(
|
||||
DateTime(timezone=True), nullable=True,
|
||||
)
|
||||
|
||||
@@ -11,6 +11,7 @@ from sqlalchemy import (
|
||||
String,
|
||||
func,
|
||||
select,
|
||||
text,
|
||||
)
|
||||
from sqlalchemy.orm import Mapped, mapped_column
|
||||
|
||||
@@ -20,7 +21,10 @@ from .base import Base
|
||||
class MLSettings(Base):
|
||||
__tablename__ = "ml_settings"
|
||||
# Bare name — Base.metadata's naming convention prepends ck_<table>_,
|
||||
# producing the final ck_ml_settings_singleton (matches migration 0003).
|
||||
# producing ck_ml_settings_singleton. The chain shipped the DOUBLED
|
||||
# ck_ml_settings_ck_ml_settings_singleton, because the migration
|
||||
# pre-prefixed the name and the convention prefixed it again; alembic
|
||||
# 0088 renames it to what this line has always produced (#3275).
|
||||
__table_args__ = (CheckConstraint("id = 1", name="singleton"),)
|
||||
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
@@ -31,17 +35,20 @@ class MLSettings(Base):
|
||||
# queueing embed work nothing will consume (the daily GPU 'embed' backfill
|
||||
# covers those images instead).
|
||||
cpu_embed_enabled: Mapped[bool] = mapped_column(
|
||||
Boolean, nullable=False, default=True
|
||||
Boolean, nullable=False, default=True,
|
||||
server_default="true",
|
||||
)
|
||||
# Video embedding (#747). Sample one frame every N seconds (fixed CADENCE, not
|
||||
# a fixed count) so coverage reflects real screen time regardless of length;
|
||||
# cap the total so a long video can't explode into hundreds of embeds. The
|
||||
# per-frame SigLIP embeddings are mean-pooled. Operator-tunable.
|
||||
video_frame_interval_seconds: Mapped[float] = mapped_column(
|
||||
Float, nullable=False, default=4.0
|
||||
Float, nullable=False, default=4.0,
|
||||
server_default="4",
|
||||
)
|
||||
video_max_frames: Mapped[int] = mapped_column(
|
||||
Integer, nullable=False, default=64
|
||||
Integer, nullable=False, default=64,
|
||||
server_default="64",
|
||||
)
|
||||
# Tagging-v2 head training (#114). The head is the suggestion source that
|
||||
# LEARNS from the operator's tags (replacing Camie + centroid). A concept
|
||||
@@ -49,10 +56,12 @@ class MLSettings(Base):
|
||||
# head_auto_apply_precision is the precision bar a head must clear (at some
|
||||
# operating point) to "graduate" into earned auto-apply. Operator-tunable.
|
||||
head_min_positives: Mapped[int] = mapped_column(
|
||||
Integer, nullable=False, default=8
|
||||
Integer, nullable=False, default=8,
|
||||
server_default="8",
|
||||
)
|
||||
head_auto_apply_precision: Mapped[float] = mapped_column(
|
||||
Float, nullable=False, default=0.97
|
||||
Float, nullable=False, default=0.97,
|
||||
server_default="0.97",
|
||||
)
|
||||
# Earned auto-apply (#114). A graduated head fires (tags images without a
|
||||
# human) when this master switch is on AND the head has at least
|
||||
@@ -61,29 +70,34 @@ class MLSettings(Base):
|
||||
# default (operator-asked 2026-06-29: opt-OUT, not opt-in); the support +
|
||||
# measured-precision gates keep it safe, and every auto-tag is reversible.
|
||||
head_auto_apply_enabled: Mapped[bool] = mapped_column(
|
||||
Boolean, nullable=False, default=True
|
||||
Boolean, nullable=False, default=True,
|
||||
server_default="true",
|
||||
)
|
||||
head_auto_apply_min_positives: Mapped[int] = mapped_column(
|
||||
# Support floor raised 30→50 (operator-asked 2026-07-06): a head needs
|
||||
# more human labels before it may fire without a human.
|
||||
Integer, nullable=False, default=50
|
||||
Integer, nullable=False, default=50,
|
||||
server_default="30",
|
||||
)
|
||||
# CCIP character-match cosine cut (#114). 0.85 default — the v1 flat 0.75
|
||||
# over-fired (high-reference characters matched a scatter of images); 0.85
|
||||
# keeps the confident single-character matches. Tunable from the agent card.
|
||||
ccip_match_threshold: Mapped[float] = mapped_column(
|
||||
Float, nullable=False, default=0.85
|
||||
Float, nullable=False, default=0.85,
|
||||
server_default="0.85",
|
||||
)
|
||||
# CCIP auto-apply (#114). Confident matches (>= ccip_auto_apply_threshold,
|
||||
# above the suggest cut) auto-tag on a daily sweep. ON by default (opt-out);
|
||||
# single-character references + the high bar keep it safe, every tag reversible.
|
||||
ccip_auto_apply_enabled: Mapped[bool] = mapped_column(
|
||||
Boolean, nullable=False, default=True
|
||||
Boolean, nullable=False, default=True,
|
||||
server_default="true",
|
||||
)
|
||||
ccip_auto_apply_threshold: Mapped[float] = mapped_column(
|
||||
# Raised 0.92→0.95 (operator-asked 2026-07-06) so only very confident
|
||||
# character matches auto-tag.
|
||||
Float, nullable=False, default=0.95
|
||||
Float, nullable=False, default=0.95,
|
||||
server_default="0.92",
|
||||
)
|
||||
# -- Presentation chrome auto-hide (#141) -------------------------------
|
||||
# `banner` (chrome — clusters on UI, not content) auto-applies on the sweep
|
||||
@@ -95,13 +109,21 @@ class MLSettings(Base):
|
||||
# (opt-out); every auto-tag is reversible. NOTE (#1464): `wip` + `editor
|
||||
# screenshot` are no longer chrome — they went to the PROCESS path below.
|
||||
presentation_auto_apply_enabled: Mapped[bool] = mapped_column(
|
||||
Boolean, nullable=False, default=True
|
||||
Boolean, nullable=False, default=True,
|
||||
server_default="true",
|
||||
)
|
||||
presentation_auto_apply_threshold: Mapped[float] = mapped_column(
|
||||
Float, nullable=False, default=0.90
|
||||
Float, nullable=False, default=0.90,
|
||||
# text(), not a string, because alembic 0082 used sa.text(): a bare
|
||||
# string renders DEFAULT '0.90'::double precision while text() renders
|
||||
# DEFAULT 0.90, and the chain is MIXED — some migrations used one,
|
||||
# some the other. Same value, different stored expression, so each
|
||||
# column here mirrors whichever form its own migration used (#3275).
|
||||
server_default=text("0.90"),
|
||||
)
|
||||
presentation_conflict_threshold: Mapped[float] = mapped_column(
|
||||
Float, nullable=False, default=0.50
|
||||
Float, nullable=False, default=0.50,
|
||||
server_default=text("0.50"),
|
||||
)
|
||||
# -- Process auto-apply (#1464) ----------------------------------------
|
||||
# `wip` / `editor screenshot` are PROCESS art — unfinished pieces + program
|
||||
@@ -115,24 +137,29 @@ class MLSettings(Base):
|
||||
# (PresentationReview, mode='process') rather than silently marked. OFF by
|
||||
# default — a new whole-library auto-tagger is opt-in; every auto-tag reversible.
|
||||
process_auto_apply_enabled: Mapped[bool] = mapped_column(
|
||||
Boolean, nullable=False, default=False
|
||||
Boolean, nullable=False, default=False,
|
||||
server_default="false",
|
||||
)
|
||||
process_auto_apply_threshold: Mapped[float] = mapped_column(
|
||||
Float, nullable=False, default=0.90
|
||||
Float, nullable=False, default=0.90,
|
||||
server_default="0.90",
|
||||
)
|
||||
process_conflict_threshold: Mapped[float] = mapped_column(
|
||||
Float, nullable=False, default=0.50
|
||||
Float, nullable=False, default=0.50,
|
||||
server_default="0.50",
|
||||
)
|
||||
# Default = SigLIP 2 (so400m, 512px) for new installs (migration 0069);
|
||||
# existing libraries keep their stored value until the operator re-embeds.
|
||||
embedder_model_version: Mapped[str] = mapped_column(
|
||||
String(128), nullable=False, default="siglip2-so400m-patch16-512"
|
||||
String(128), nullable=False, default="siglip2-so400m-patch16-512",
|
||||
server_default="siglip2-so400m-patch16-512",
|
||||
)
|
||||
# The HF model NAME the embedder loads (server CPU embed + announced to the
|
||||
# GPU agent in the lease). Operator-settable so the embedder is a choice, not
|
||||
# a hardcode (#1190): set name + version together, then re-embed + retrain.
|
||||
embedder_model_name: Mapped[str] = mapped_column(
|
||||
String(128), nullable=False, default="google/siglip2-so400m-patch16-512"
|
||||
String(128), nullable=False, default="google/siglip2-so400m-patch16-512",
|
||||
server_default="google/siglip2-so400m-patch16-512",
|
||||
)
|
||||
# -- Crop proposers / detectors (#1202, #134) --------------------------
|
||||
# WHERE-to-crop YOLO detectors feeding the crop→SigLIP bag + CCIP. Config
|
||||
@@ -145,20 +172,24 @@ class MLSettings(Base):
|
||||
# person: general COCO figure detector for Western/realistic art the anime
|
||||
# person-detector misses → NMS-merged with imgutils → CCIP + concept.
|
||||
detector_person_enabled: Mapped[bool] = mapped_column(
|
||||
Boolean, nullable=False, default=True
|
||||
Boolean, nullable=False, default=True,
|
||||
server_default="true",
|
||||
)
|
||||
detector_person_weights: Mapped[str] = mapped_column(
|
||||
String(512), nullable=False, default="yolo11n.pt"
|
||||
String(512), nullable=False, default="yolo11n.pt",
|
||||
server_default="yolo11n.pt",
|
||||
)
|
||||
detector_person_conf: Mapped[float] = mapped_column(
|
||||
Float, nullable=False, default=0.35
|
||||
Float, nullable=False, default=0.35,
|
||||
server_default=text("0.35"),
|
||||
)
|
||||
# anatomy: booru_yolo anime/furry/NSFW torso components → concept crops.
|
||||
# Default = yolov11m_aa22 (26 classes, best mAP50-95 0.96), committed in the
|
||||
# upstream repo so the URL resolves. License UNSTATED — fine for a private
|
||||
# homelab (operator accepted #1202).
|
||||
detector_anatomy_enabled: Mapped[bool] = mapped_column(
|
||||
Boolean, nullable=False, default=True
|
||||
Boolean, nullable=False, default=True,
|
||||
server_default="true",
|
||||
)
|
||||
detector_anatomy_weights: Mapped[str] = mapped_column(
|
||||
String(512), nullable=False,
|
||||
@@ -166,37 +197,47 @@ class MLSettings(Base):
|
||||
"https://github.com/aperveyev/booru_yolo/raw/main/models/"
|
||||
"yolov11m_aa22.pt"
|
||||
),
|
||||
server_default="https://github.com/aperveyev/booru_yolo/raw/main/models/yolov11m_aa22.pt",
|
||||
)
|
||||
detector_anatomy_conf: Mapped[float] = mapped_column(
|
||||
Float, nullable=False, default=0.30
|
||||
Float, nullable=False, default=0.30,
|
||||
server_default=text("0.30"),
|
||||
)
|
||||
# panel: comic page → panel regions → concept crops (Apache-2.0, YOLOv12x).
|
||||
detector_panel_enabled: Mapped[bool] = mapped_column(
|
||||
Boolean, nullable=False, default=True
|
||||
Boolean, nullable=False, default=True,
|
||||
server_default="true",
|
||||
)
|
||||
detector_panel_weights: Mapped[str] = mapped_column(
|
||||
String(512), nullable=False,
|
||||
default="mosesb/best-comic-panel-detection::best.pt",
|
||||
server_default="mosesb/best-comic-panel-detection::best.pt",
|
||||
)
|
||||
detector_panel_conf: Mapped[float] = mapped_column(
|
||||
Float, nullable=False, default=0.30
|
||||
Float, nullable=False, default=0.30,
|
||||
server_default=text("0.30"),
|
||||
)
|
||||
# Per-frame caps bound the crop→embed explosion; max_regions is the hard
|
||||
# per-job backstop; dedupe_iou drops near-duplicate crops before the embed.
|
||||
detector_max_figures: Mapped[int] = mapped_column(
|
||||
Integer, nullable=False, default=8
|
||||
Integer, nullable=False, default=8,
|
||||
server_default="8",
|
||||
)
|
||||
detector_max_components: Mapped[int] = mapped_column(
|
||||
Integer, nullable=False, default=8
|
||||
Integer, nullable=False, default=8,
|
||||
server_default="8",
|
||||
)
|
||||
detector_max_panels: Mapped[int] = mapped_column(
|
||||
Integer, nullable=False, default=8
|
||||
Integer, nullable=False, default=8,
|
||||
server_default="8",
|
||||
)
|
||||
detector_max_regions: Mapped[int] = mapped_column(
|
||||
Integer, nullable=False, default=128
|
||||
Integer, nullable=False, default=128,
|
||||
server_default="128",
|
||||
)
|
||||
detector_dedupe_iou: Mapped[float] = mapped_column(
|
||||
Float, nullable=False, default=0.85
|
||||
Float, nullable=False, default=0.85,
|
||||
server_default=text("0.85"),
|
||||
)
|
||||
# -- CCIP character prototypes (#1317) ---------------------------------
|
||||
# The per-character reference set is precomputed + refreshed INCREMENTALLY
|
||||
@@ -208,7 +249,8 @@ class MLSettings(Base):
|
||||
String(128), nullable=True
|
||||
)
|
||||
ccip_prototype_cap: Mapped[int] = mapped_column(
|
||||
Integer, nullable=False, default=64
|
||||
Integer, nullable=False, default=64,
|
||||
server_default="64",
|
||||
)
|
||||
updated_at: Mapped[datetime] = mapped_column(
|
||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||
|
||||
@@ -35,7 +35,7 @@ class PatreonFailedMedia(Base):
|
||||
ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True
|
||||
)
|
||||
filehash: Mapped[str] = mapped_column(String(128), nullable=False)
|
||||
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1)
|
||||
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1, server_default="1")
|
||||
last_error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||
first_failed_at: Mapped[datetime] = mapped_column(
|
||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||
|
||||
@@ -35,7 +35,7 @@ class PixivFailedMedia(Base):
|
||||
ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True
|
||||
)
|
||||
filehash: Mapped[str] = mapped_column(String(128), nullable=False)
|
||||
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1)
|
||||
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1, server_default="1")
|
||||
last_error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||
first_failed_at: Mapped[datetime] = mapped_column(
|
||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||
|
||||
@@ -13,11 +13,13 @@ from sqlalchemy import (
|
||||
CheckConstraint,
|
||||
DateTime,
|
||||
ForeignKey,
|
||||
Index,
|
||||
Integer,
|
||||
String,
|
||||
Text,
|
||||
UniqueConstraint,
|
||||
func,
|
||||
text,
|
||||
)
|
||||
from sqlalchemy.orm import Mapped, mapped_column
|
||||
|
||||
@@ -27,6 +29,10 @@ from .base import Base
|
||||
class Post(Base):
|
||||
__tablename__ = "post"
|
||||
__table_args__ = (
|
||||
# alembic 0030. The comment above described this index; nothing declared
|
||||
# it, so autogenerate proposed dropping it (#3275).
|
||||
Index("uq_post_artist_external_id_null_source", "artist_id", "external_post_id",
|
||||
unique=True, postgresql_where=text("source_id IS NULL")),
|
||||
# Source-bound dedup. Postgres treats NULL != NULL so rows
|
||||
# with source_id IS NULL aren't deduped by this constraint;
|
||||
# the partial unique index `uq_post_artist_external_id_null_source`
|
||||
@@ -35,7 +41,11 @@ class Post(Base):
|
||||
UniqueConstraint("source_id", "external_post_id", name="uq_post_source_external_id"),
|
||||
CheckConstraint(
|
||||
"translation_override IN ('auto', 'force', 'original')",
|
||||
name="ck_post_translation_override",
|
||||
# Bare name: Base.metadata's naming convention prepends
|
||||
# ck_<table>_. Pre-prefixing it here doubles the prefix — see
|
||||
# alembic 0088, which renames the four constraints that shipped
|
||||
# that way (#3275).
|
||||
name="translation_override",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@@ -11,7 +11,7 @@ are pruned by retention.
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
from sqlalchemy import DateTime, Float, ForeignKey, String, func
|
||||
from sqlalchemy import DateTime, Float, ForeignKey, Index, String, func
|
||||
from sqlalchemy.orm import Mapped, mapped_column
|
||||
|
||||
from .base import Base
|
||||
@@ -20,6 +20,14 @@ from .base import Base
|
||||
class PresentationReview(Base):
|
||||
__tablename__ = "presentation_review"
|
||||
|
||||
|
||||
__table_args__ = (
|
||||
Index("ix_presentation_review_resolved_at", "resolved_at"),
|
||||
# Both FKs to tag were unindexed (#3300); tag_id CASCADEs, so a tag
|
||||
# delete had to scan this table to find its rows.
|
||||
Index("ix_presentation_review_tag_id", "tag_id"),
|
||||
Index("ix_presentation_review_conflict_tag_id", "conflict_tag_id"),
|
||||
)
|
||||
image_record_id: Mapped[int] = mapped_column(
|
||||
ForeignKey("image_record.id", ondelete="CASCADE"), primary_key=True
|
||||
)
|
||||
|
||||
@@ -16,7 +16,14 @@ title is the optional chapter name; stated_part is the optional operator-facing
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
from sqlalchemy import DateTime, ForeignKey, Integer, Text, func
|
||||
from sqlalchemy import (
|
||||
DateTime,
|
||||
ForeignKey,
|
||||
Integer,
|
||||
Text,
|
||||
UniqueConstraint,
|
||||
func,
|
||||
)
|
||||
from sqlalchemy.orm import Mapped, mapped_column
|
||||
|
||||
from .base import Base
|
||||
@@ -25,14 +32,26 @@ from .base import Base
|
||||
class SeriesChapter(Base):
|
||||
__tablename__ = "series_chapter"
|
||||
|
||||
__table_args__ = (
|
||||
# alembic 0047 named the UNIQUE `uq_series_chapter_anchor_page`, not
|
||||
# the `uq_series_chapter_anchor_page_id` a bare `unique=True` would
|
||||
# render (#3275).
|
||||
UniqueConstraint("anchor_page_id", name="uq_series_chapter_anchor_page"),
|
||||
)
|
||||
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
series_tag_id: Mapped[int] = mapped_column(
|
||||
ForeignKey("tag.id", ondelete="CASCADE"), nullable=False, index=True
|
||||
)
|
||||
# Both the UNIQUE (above) and the FK carry the names 0047 gave them; the
|
||||
# convention would render the FK `fk_series_chapter_anchor_page_id_series_page`.
|
||||
anchor_page_id: Mapped[int] = mapped_column(
|
||||
ForeignKey("series_page.id", ondelete="CASCADE"),
|
||||
ForeignKey(
|
||||
"series_page.id",
|
||||
ondelete="CASCADE",
|
||||
name="fk_series_chapter_anchor_page",
|
||||
),
|
||||
nullable=False,
|
||||
unique=True,
|
||||
)
|
||||
title: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||
stated_part: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
||||
|
||||
@@ -14,7 +14,14 @@ number parsed from the source post, nullable when unknown.
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
from sqlalchemy import DateTime, ForeignKey, Integer, String, func
|
||||
from sqlalchemy import (
|
||||
DateTime,
|
||||
ForeignKey,
|
||||
Integer,
|
||||
String,
|
||||
UniqueConstraint,
|
||||
func,
|
||||
)
|
||||
from sqlalchemy.orm import Mapped, mapped_column
|
||||
|
||||
from .base import Base
|
||||
@@ -23,14 +30,22 @@ from .base import Base
|
||||
class SeriesPage(Base):
|
||||
__tablename__ = "series_page"
|
||||
|
||||
__table_args__ = (
|
||||
# alembic 0005 named this `uq_series_page_image`; a bare `unique=True`
|
||||
# on the column renders `uq_series_page_image_id` under the naming
|
||||
# convention, which is a different object from the one the database
|
||||
# has (#3275).
|
||||
UniqueConstraint("image_id", name="uq_series_page_image"),
|
||||
)
|
||||
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
series_tag_id: Mapped[int] = mapped_column(
|
||||
ForeignKey("tag.id", ondelete="CASCADE"), nullable=False, index=True
|
||||
)
|
||||
# UNIQUE lives in __table_args__ above, under the name 0005 gave it.
|
||||
image_id: Mapped[int] = mapped_column(
|
||||
ForeignKey("image_record.id", ondelete="CASCADE"),
|
||||
nullable=False,
|
||||
unique=True,
|
||||
)
|
||||
# 'placed' = in the series-global run (page_number set); 'pending' = staged
|
||||
# from a post awaiting the operator's sort (page_number NULL). (#789 P2)
|
||||
|
||||
@@ -0,0 +1,88 @@
|
||||
"""service_seen — the learned roster of FabledCurator's own moving parts.
|
||||
|
||||
Nothing else in this application knows what is SUPPOSED to be running.
|
||||
`celery inspect` reports the workers that answer, so a stopped worker is a
|
||||
shorter list rather than a red light, and Postgres and Redis have no
|
||||
representation at all. That is why the only place an operator could see a
|
||||
dead service was Portainer, which knows the intended set (milestone 365).
|
||||
|
||||
This table is the memory that makes an absence observable: every part that
|
||||
has ever checked in, and when it last did. A row that stops advancing is a
|
||||
part that stopped.
|
||||
|
||||
## Why the key is not the hostname
|
||||
|
||||
`_read_workers_sync()` returns celery's worker names, which here are
|
||||
`celery@<container id>`. Those are minted fresh on every deploy. Keyed on
|
||||
them, this table would record a death and a birth every time the stack is
|
||||
updated — and a status page that goes red on every deploy is a status page
|
||||
nobody reads, which is worse than not having one.
|
||||
|
||||
So a celery role is keyed on its **queue set**, which is assigned per role in
|
||||
docker-compose.yml (`CELERY_QUEUES`) and survives container replacement:
|
||||
|
||||
default,import,thumbnail,download -> worker
|
||||
maintenance,scan -> scheduler (celery worker --beat)
|
||||
ml -> ml-worker
|
||||
|
||||
Two replicas of one role share a queue set and are therefore ONE row — which
|
||||
is right, because the question being answered is "is that role being served",
|
||||
not "how many containers exist". The replica count and their hostnames go in
|
||||
`details`, where they can change without the identity changing.
|
||||
|
||||
The GPU agent is keyed on its `agent_id`, the identity its lease protocol
|
||||
already uses (`api/gpu.py`).
|
||||
|
||||
## What is NOT in here
|
||||
|
||||
Postgres and Redis. They are always expected and never learned, and a
|
||||
last-seen for them would be actively misleading — that one answered thirty
|
||||
seconds ago says nothing about now. They are probed live at request time.
|
||||
|
||||
## kind
|
||||
|
||||
Plain `String`, not a Postgres ENUM and not CHECK-gated, matching
|
||||
`gpu_job.status` and `backup_run.status`. The value set here is expected to
|
||||
grow as parts are added, and a constraint swap per new kind (rule 36) would
|
||||
be cost with no invariant behind it.
|
||||
|
||||
celery — a worker role, keyed on its queue set
|
||||
agent — a GPU agent, keyed on its agent_id
|
||||
"""
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
from sqlalchemy import JSON, DateTime, String, func
|
||||
from sqlalchemy.orm import Mapped, mapped_column
|
||||
|
||||
from .base import Base
|
||||
|
||||
|
||||
class ServiceSeen(Base):
|
||||
__tablename__ = "service_seen"
|
||||
|
||||
# No indexes beyond the primary key, deliberately. This table holds one row
|
||||
# per moving part — a handful, forever — so every query against it is a
|
||||
# full read of a few rows and an index would be write cost buying nothing
|
||||
# (the lesson of #3301, which removed seven redundant ones).
|
||||
key: Mapped[str] = mapped_column(String(128), primary_key=True)
|
||||
kind: Mapped[str] = mapped_column(String(16), nullable=False)
|
||||
|
||||
# What to call it in the UI. Derived from the queue set where it is
|
||||
# recognised, and falling back to the raw queue list where it is not — a
|
||||
# deployment that slices its queues differently should still show something
|
||||
# true rather than a name this code invented for it.
|
||||
display_name: Mapped[str] = mapped_column(String(64), nullable=False)
|
||||
|
||||
first_seen_at: Mapped[datetime] = mapped_column(
|
||||
DateTime(timezone=True), nullable=False, server_default=func.now(),
|
||||
)
|
||||
last_seen_at: Mapped[datetime] = mapped_column(
|
||||
DateTime(timezone=True), nullable=False, server_default=func.now(),
|
||||
)
|
||||
|
||||
# The parts that change without changing identity: replica hostnames,
|
||||
# active task counts, the queues actually being served. Kept as a blob
|
||||
# because it is displayed and never queried — giving it columns would
|
||||
# invite filtering on it, which is what the activity endpoints are for.
|
||||
details: Mapped[dict] = mapped_column(JSON, nullable=False, default=dict)
|
||||
@@ -5,7 +5,16 @@ Multiple sources per artist support creators with cross-platform presence.
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
from sqlalchemy import JSON, Boolean, DateTime, ForeignKey, Integer, String, Text
|
||||
from sqlalchemy import (
|
||||
JSON,
|
||||
Boolean,
|
||||
DateTime,
|
||||
ForeignKey,
|
||||
Integer,
|
||||
String,
|
||||
Text,
|
||||
UniqueConstraint,
|
||||
)
|
||||
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
||||
|
||||
from .base import Base
|
||||
@@ -14,13 +23,27 @@ from .base import Base
|
||||
class Source(Base):
|
||||
__tablename__ = "source"
|
||||
|
||||
__table_args__ = (
|
||||
# alembic 0010. One row per (artist, platform, url): re-adding a source
|
||||
# the artist already has is an update, not a second row. The model had
|
||||
# never declared it (#3275), so autogenerate would have proposed
|
||||
# DROPPING it — the guarantee existed only in the migration chain.
|
||||
#
|
||||
# Named explicitly because the naming convention would render this
|
||||
# `uq_source_artist_id` (uq keys off column_0_name), which is both
|
||||
# wrong about the shape and not what the database actually has.
|
||||
UniqueConstraint(
|
||||
"artist_id", "platform", "url", name="uq_source_artist_platform_url"
|
||||
),
|
||||
)
|
||||
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
artist_id: Mapped[int] = mapped_column(
|
||||
ForeignKey("artist.id", ondelete="CASCADE"), nullable=False, index=True
|
||||
)
|
||||
platform: Mapped[str] = mapped_column(String(64), nullable=False)
|
||||
url: Mapped[str] = mapped_column(Text, nullable=False)
|
||||
enabled: Mapped[bool] = mapped_column(Boolean, nullable=False, default=True)
|
||||
enabled: Mapped[bool] = mapped_column(Boolean, nullable=False, default=True, server_default="true")
|
||||
|
||||
config_overrides: Mapped[dict | None] = mapped_column(JSON, nullable=True)
|
||||
|
||||
@@ -32,7 +55,7 @@ class Source(Base):
|
||||
# by _update_source_health alongside last_error; cleared on 'ok'.
|
||||
error_type: Mapped[str | None] = mapped_column(String(32), nullable=True, index=True)
|
||||
check_interval_override: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
||||
consecutive_failures: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
||||
consecutive_failures: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||
|
||||
# alembic 0031: sticky deep-scan budget. When > 0, the next N download
|
||||
# runs use gallery-dl's full-walk config (skip: True + 1800s timeout);
|
||||
|
||||
@@ -34,7 +34,7 @@ class SubscribeStarFailedMedia(Base):
|
||||
ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True
|
||||
)
|
||||
filehash: Mapped[str] = mapped_column(String(128), nullable=False)
|
||||
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1)
|
||||
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1, server_default="1")
|
||||
last_error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||
first_failed_at: Mapped[datetime] = mapped_column(
|
||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||
|
||||
@@ -15,11 +15,13 @@ from sqlalchemy import (
|
||||
Column,
|
||||
DateTime,
|
||||
ForeignKey,
|
||||
Index,
|
||||
Integer,
|
||||
String,
|
||||
Table,
|
||||
false,
|
||||
func,
|
||||
text,
|
||||
)
|
||||
from sqlalchemy import (
|
||||
Enum as SQLEnum,
|
||||
@@ -67,17 +69,31 @@ image_tag = Table(
|
||||
primary_key=True,
|
||||
),
|
||||
Column("tag_id", ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True),
|
||||
Column("source", String(32), nullable=False, default="manual"),
|
||||
Column("source", String(32), nullable=False, default="manual", server_default="manual"),
|
||||
Column("created_at", DateTime(timezone=True), nullable=False, server_default=func.now()),
|
||||
# The PK is (image_record_id, tag_id), which leads with the WRONG column
|
||||
# for the two things that matter most here (#3300): the gallery's tag
|
||||
# filter (tag_query.py builds `image_tag.c.tag_id == tid`) and the
|
||||
# ON DELETE CASCADE from tag, which has to find a tag's rows to remove
|
||||
# them. Without this index both scan the largest table in the schema.
|
||||
Index("ix_image_tag_tag_id", "tag_id"),
|
||||
)
|
||||
|
||||
|
||||
class Tag(Base):
|
||||
__tablename__ = "tag"
|
||||
__table_args__ = (
|
||||
# alembic 0002. An EXPRESSION index — COALESCE cannot be expressed as a
|
||||
# UniqueConstraint, which is why it only ever existed in a migration (#3275).
|
||||
Index("uq_tag_name_kind_fandom", "name", "kind", text("COALESCE(fandom_id, 0)"),
|
||||
unique=True),
|
||||
CheckConstraint(
|
||||
"(fandom_id IS NULL) OR (kind = 'character')",
|
||||
name="ck_tag_fandom_requires_character",
|
||||
# Bare name: Base.metadata's naming convention prepends
|
||||
# ck_<table>_. Pre-prefixing it here doubles the prefix — see
|
||||
# alembic 0088, which renames the four constraints that shipped
|
||||
# that way (#3275).
|
||||
name="fandom_requires_character",
|
||||
),
|
||||
)
|
||||
|
||||
@@ -87,6 +103,7 @@ class Tag(Base):
|
||||
SQLEnum(TagKind, name="tag_kind", values_callable=lambda e: [m.value for m in e]),
|
||||
nullable=False,
|
||||
default=TagKind.general,
|
||||
server_default="general",
|
||||
)
|
||||
fandom_id: Mapped[int | None] = mapped_column(
|
||||
ForeignKey("tag.id", ondelete="SET NULL"), nullable=True, index=True
|
||||
|
||||
@@ -5,7 +5,7 @@ in image_prediction stay unmolested.
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
from sqlalchemy import DateTime, ForeignKey, String, func
|
||||
from sqlalchemy import DateTime, ForeignKey, Index, String, func
|
||||
from sqlalchemy.orm import Mapped, mapped_column
|
||||
|
||||
from .base import Base
|
||||
@@ -14,10 +14,17 @@ from .base import Base
|
||||
class TagAlias(Base):
|
||||
__tablename__ = "tag_alias"
|
||||
|
||||
|
||||
__table_args__ = (
|
||||
# Named explicitly: the database calls this ix_tag_alias_canonical, while
|
||||
# a bare index=True on the column would generate ix_tag_alias_canonical_tag_id
|
||||
# and silently propose a drop+create on the next autogenerate (#3275).
|
||||
Index("ix_tag_alias_canonical", "canonical_tag_id"),
|
||||
)
|
||||
alias_string: Mapped[str] = mapped_column(String(255), primary_key=True)
|
||||
alias_category: Mapped[str] = mapped_column(String(32), primary_key=True)
|
||||
canonical_tag_id: Mapped[int] = mapped_column(
|
||||
ForeignKey("tag.id", ondelete="CASCADE"), nullable=False, index=True
|
||||
ForeignKey("tag.id", ondelete="CASCADE"), nullable=False
|
||||
)
|
||||
created_at: Mapped[datetime] = mapped_column(
|
||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||
|
||||
@@ -5,7 +5,7 @@ Prevents re-suggestion AND prevents allowlist auto-apply on that image.
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
from sqlalchemy import DateTime, ForeignKey, func
|
||||
from sqlalchemy import DateTime, ForeignKey, Index, func
|
||||
from sqlalchemy.orm import Mapped, mapped_column
|
||||
|
||||
from .base import Base
|
||||
@@ -14,11 +14,24 @@ from .base import Base
|
||||
class TagSuggestionRejection(Base):
|
||||
__tablename__ = "tag_suggestion_rejection"
|
||||
|
||||
|
||||
__table_args__ = (
|
||||
# Named explicitly; see tag_alias for why (#3275).
|
||||
Index("ix_tag_suggestion_rejection_tag", "tag_id"),
|
||||
)
|
||||
# Both FKs named explicitly. alembic 0003 used a hand-shortened `tsr`
|
||||
# prefix; the convention would render the full table name (#3275).
|
||||
image_record_id: Mapped[int] = mapped_column(
|
||||
ForeignKey("image_record.id", ondelete="CASCADE"), primary_key=True
|
||||
ForeignKey(
|
||||
"image_record.id",
|
||||
ondelete="CASCADE",
|
||||
name="fk_tsr_image_record_id_image_record",
|
||||
),
|
||||
primary_key=True,
|
||||
)
|
||||
tag_id: Mapped[int] = mapped_column(
|
||||
ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True, index=True
|
||||
ForeignKey("tag.id", ondelete="CASCADE", name="fk_tsr_tag_id_tag"),
|
||||
primary_key=True,
|
||||
)
|
||||
rejected_at: Mapped[datetime] = mapped_column(
|
||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||
|
||||
@@ -15,7 +15,7 @@ backend.app.tasks.maintenance.recover_stalled_task_runs (Beat 5 min).
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
from sqlalchemy import DateTime, Integer, String, Text
|
||||
from sqlalchemy import DateTime, Index, Integer, String, Text, text
|
||||
from sqlalchemy.orm import Mapped, mapped_column
|
||||
|
||||
from .base import Base
|
||||
@@ -24,12 +24,21 @@ from .base import Base
|
||||
class TaskRun(Base):
|
||||
__tablename__ = "task_run"
|
||||
|
||||
|
||||
__table_args__ = (
|
||||
# alembic 0016: the three task-history indexes (#3275).
|
||||
Index("ix_task_run_name_started", "task_name", text("started_at DESC")),
|
||||
Index("ix_task_run_queue_started", "queue", text("started_at DESC")),
|
||||
Index("ix_task_run_status_started", "status", text("started_at DESC")),
|
||||
)
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
celery_task_id: Mapped[str] = mapped_column(
|
||||
String(64), nullable=False, index=True,
|
||||
)
|
||||
queue: Mapped[str] = mapped_column(String(32), nullable=False, index=True)
|
||||
task_name: Mapped[str] = mapped_column(String(128), nullable=False, index=True)
|
||||
# Neither carries index=True: ix_task_run_queue_started and
|
||||
# ix_task_run_name_started already lead with these columns (#3301).
|
||||
queue: Mapped[str] = mapped_column(String(32), nullable=False)
|
||||
task_name: Mapped[str] = mapped_column(String(128), nullable=False)
|
||||
target_id: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
||||
started_at: Mapped[datetime] = mapped_column(
|
||||
DateTime(timezone=True), nullable=False, index=True,
|
||||
@@ -39,7 +48,9 @@ class TaskRun(Base):
|
||||
)
|
||||
duration_ms: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
||||
status: Mapped[str] = mapped_column(
|
||||
String(16), nullable=False, default="running", index=True,
|
||||
# No index=True — ix_task_run_status_started leads with `status`.
|
||||
String(16), nullable=False, default="running",
|
||||
server_default="running",
|
||||
)
|
||||
error_type: Mapped[str | None] = mapped_column(String(128), nullable=True)
|
||||
error_message: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||
|
||||
@@ -0,0 +1,175 @@
|
||||
"""The learned roster: which of FabledCurator's parts have checked in, and when.
|
||||
|
||||
Milestone 365. `celery inspect` answers "who is here"; this answers "who is
|
||||
missing", which nothing in the application could do before — see
|
||||
`models/service_seen.py` for why the identity is a queue set and not a
|
||||
worker hostname.
|
||||
|
||||
## Who does the observing, and why it is the web process
|
||||
|
||||
Three candidates, and the choice matters more than the code:
|
||||
|
||||
* **A celery beat sweep.** Rejected. If the scheduler dies, the sweep stops,
|
||||
every row goes stale, and the page reports that everything is down when one
|
||||
thing is. An alarm that cannot distinguish "one part died" from "the
|
||||
observer died" is worse than no alarm.
|
||||
* **A background task in web.** Rejected on a detail of how this deploys:
|
||||
hypercorn runs `--workers 4`, so a `before_serving` loop would be FOUR
|
||||
concurrent inspect loops hammering the broker, forever, per container.
|
||||
* **Refresh on demand, rate-limited by the data itself.** Taken. Whichever web
|
||||
process happens to serve a health request refreshes the roster if it is
|
||||
older than REFRESH_TTL, and otherwise reads what is already there.
|
||||
|
||||
The third has the property the other two lack: **the observer is the thing
|
||||
serving the page.** If web is down you get a browser error rather than a
|
||||
confidently green page, which is the honest failure. It also self-limits
|
||||
without coordination — the TTL lives in the row everybody can see.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
|
||||
from sqlalchemy import func, select
|
||||
from sqlalchemy.dialects.postgresql import insert as pg_insert
|
||||
from sqlalchemy.ext.asyncio import AsyncSession
|
||||
|
||||
from ..models import ServiceSeen
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
# How stale the roster may be before a health request refreshes it. Comfortably
|
||||
# under the staleness thresholds that decide a service is missing, so the
|
||||
# verdict is never limited by how often anyone looked.
|
||||
REFRESH_TTL_SECONDS = 20.0
|
||||
|
||||
# celery inspect is a broker round trip and this sits on a request path, so it
|
||||
# gets a deadline (rule 156). A broker that has stopped answering must make the
|
||||
# roster stale — which is a true statement about the system — not hang the one
|
||||
# page that exists to explain it.
|
||||
INSPECT_TIMEOUT_SECONDS = 2.0
|
||||
|
||||
# Queue set -> the name an operator recognises. Sorted-tuple keys, because the
|
||||
# order celery reports them in is not guaranteed.
|
||||
#
|
||||
# A deployment that slices CELERY_QUEUES differently falls through to the raw
|
||||
# queue list rather than being given a name this table invented for it: a
|
||||
# wrong-but-confident label on a status page is worse than an ugly true one.
|
||||
ROLE_NAMES: dict[tuple[str, ...], str] = {
|
||||
("default", "download", "import", "thumbnail"): "Worker",
|
||||
("maintenance", "scan"): "Scheduler",
|
||||
("ml",): "ML worker",
|
||||
}
|
||||
|
||||
|
||||
def role_display_name(queues: tuple[str, ...]) -> str:
|
||||
known = ROLE_NAMES.get(queues)
|
||||
if known:
|
||||
return known
|
||||
return "Worker (" + ", ".join(queues) + ")"
|
||||
|
||||
|
||||
def _inspect_celery_sync() -> dict[tuple[str, ...], dict]:
|
||||
"""celery inspect, grouped by queue set rather than by worker.
|
||||
|
||||
Returns {queue_set: {"hostnames": [...], "active": int}}. Two replicas of
|
||||
one role collapse into one entry on purpose — the question is whether the
|
||||
role is being served, not how many containers exist.
|
||||
"""
|
||||
from ..celery_app import celery as celery_app
|
||||
|
||||
insp = celery_app.control.inspect(timeout=INSPECT_TIMEOUT_SECONDS)
|
||||
active_queues = insp.active_queues() or {}
|
||||
active_tasks = insp.active() or {}
|
||||
|
||||
grouped: dict[tuple[str, ...], dict] = {}
|
||||
for hostname, queues in active_queues.items():
|
||||
key = tuple(sorted({q["name"] for q in queues}))
|
||||
entry = grouped.setdefault(key, {"hostnames": [], "active": 0})
|
||||
entry["hostnames"].append(hostname)
|
||||
entry["active"] += len(active_tasks.get(hostname, []))
|
||||
for entry in grouped.values():
|
||||
entry["hostnames"].sort()
|
||||
return grouped
|
||||
|
||||
|
||||
async def touch_service(
|
||||
session: AsyncSession, *, key: str, kind: str, display_name: str, details: dict
|
||||
) -> None:
|
||||
"""Record that a part checked in just now.
|
||||
|
||||
Upsert rather than read-modify-write: several web processes and several
|
||||
agents can be doing this at once, and the last writer is simply the most
|
||||
recent sighting. `first_seen_at` is deliberately NOT updated — it is the
|
||||
one field that answers "has this ever run", which the learned-roster design
|
||||
depends on.
|
||||
"""
|
||||
stmt = pg_insert(ServiceSeen).values(
|
||||
key=key, kind=kind, display_name=display_name, details=details,
|
||||
)
|
||||
stmt = stmt.on_conflict_do_update(
|
||||
index_elements=[ServiceSeen.key],
|
||||
set_={
|
||||
"kind": stmt.excluded.kind,
|
||||
"display_name": stmt.excluded.display_name,
|
||||
"details": stmt.excluded.details,
|
||||
"last_seen_at": func.now(),
|
||||
},
|
||||
)
|
||||
await session.execute(stmt)
|
||||
|
||||
|
||||
async def refresh_celery_roster(session: AsyncSession) -> None:
|
||||
"""Inspect the broker and record what answered. Never raises.
|
||||
|
||||
A failure here means the roster does not advance, and the rows going stale
|
||||
is then a TRUE report about a broker nobody can reach. Letting the
|
||||
exception out would instead break the health endpoint, which is the one
|
||||
thing that must keep answering when the stack is unwell.
|
||||
"""
|
||||
try:
|
||||
grouped = await asyncio.wait_for(
|
||||
asyncio.to_thread(_inspect_celery_sync),
|
||||
timeout=INSPECT_TIMEOUT_SECONDS * 2,
|
||||
)
|
||||
except Exception:
|
||||
log.warning("service roster: celery inspect failed; roster not refreshed", exc_info=True)
|
||||
return
|
||||
|
||||
for queues, entry in grouped.items():
|
||||
await touch_service(
|
||||
session,
|
||||
key="celery:" + ",".join(queues),
|
||||
kind="celery",
|
||||
display_name=role_display_name(queues),
|
||||
details={
|
||||
"queues": list(queues),
|
||||
"hostnames": entry["hostnames"],
|
||||
"replicas": len(entry["hostnames"]),
|
||||
"active": entry["active"],
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
async def refresh_if_stale(session: AsyncSession) -> None:
|
||||
"""Refresh the celery roster if nobody has for REFRESH_TTL_SECONDS.
|
||||
|
||||
Rate-limited by the data rather than by a lock: the gate is the newest
|
||||
last_seen_at across the celery rows, which every web process can see. Two
|
||||
processes racing through the gate costs one redundant inspect and writes
|
||||
the same values twice, so the benign outcome needs no coordination to
|
||||
prevent.
|
||||
"""
|
||||
newest = (
|
||||
await session.execute(
|
||||
select(func.max(ServiceSeen.last_seen_at)).where(ServiceSeen.kind == "celery")
|
||||
)
|
||||
).scalar_one_or_none()
|
||||
|
||||
if newest is not None:
|
||||
age = (await session.execute(select(func.now()))).scalar_one() - newest
|
||||
if age.total_seconds() < REFRESH_TTL_SECONDS:
|
||||
return
|
||||
|
||||
await refresh_celery_roster(session)
|
||||
+15
-8
@@ -198,14 +198,21 @@ per `docs/process.md`'s "add deps to the image when used by >1 project".
|
||||
refresh from being undone.
|
||||
- **`pull: true` on the scheduled path only** is the mechanism: a moved base
|
||||
tag changes the `FROM` layer's cache key and everything above it rebuilds.
|
||||
**It does not currently make the unmoved case free.** Measured on the first
|
||||
real fire (run 4934, 2026-08-30): every content step reported `CACHED` and
|
||||
the bases resolved to unchanged digests, yet all three `:latest` tags got a
|
||||
NEW manifest digest, because buildkit mints a fresh image config per run and
|
||||
republishes identical layers under it. So `:latest` is rewritten weekly
|
||||
whether or not anything changed, and `:c-<sha>` is handed a new manifest to
|
||||
diverge from on the same cadence — a digest change stops meaning anything.
|
||||
Tracked as #3265; the likely fix is a deterministic `SOURCE_DATE_EPOCH`.
|
||||
It did not always make the unmoved case free. Measured on the first real
|
||||
fire (run 4934, 2026-08-30): every content step reported `CACHED` and the
|
||||
bases resolved to unchanged digests, yet all three `:latest` tags got a NEW
|
||||
manifest digest, because buildkit stamps a fresh image config per run and
|
||||
republishes identical layers under it — so `:latest` was rewritten weekly
|
||||
whether or not anything changed, and a digest change stopped meaning
|
||||
anything (#3265).
|
||||
- **`SOURCE_DATE_EPOCH` is what makes it free.** Set on each build step from
|
||||
`artifacts.sh epoch <artifact>` — the unix timestamp of the same commit
|
||||
`revision` and `version` name, so all three are views of one `newest()`
|
||||
lookup and cannot drift into disagreeing. With the config's `created` field
|
||||
and history timestamps pinned to the content rather than to the wall clock,
|
||||
identical source produces an identical manifest digest and the push is a
|
||||
registry no-op. That restores the property the whole scheme rests on: a
|
||||
channel tag's digest changes when, and only when, its content does.
|
||||
Separately not caught: a Debian package update inside the `apt-get install`
|
||||
layer while the base tag stands still — a lag rather than a hole, since the
|
||||
official python/cuda images rebuild with those updates baked in.
|
||||
|
||||
+69
-20
@@ -1,12 +1,21 @@
|
||||
# Base compose stack. Uses ${VAR:-default} interpolation throughout so the
|
||||
# stack boots with zero config — sane dev defaults baked in. For production
|
||||
# deployments, override the defaults via shell env vars or a .env file:
|
||||
# Base compose stack, and the install path. Uses ${VAR:-default} throughout so
|
||||
# the stack boots with zero config — but those defaults are DEV defaults, and
|
||||
# two of them (DB_PASSWORD, SECRET_KEY) are published in this file. Copy
|
||||
# .env.example to .env and set them before running this anywhere real.
|
||||
#
|
||||
# DB_PASSWORD=...real... SECRET_KEY=...real... docker compose up
|
||||
# To run FabledCurator:
|
||||
#
|
||||
# The dev override (docker-compose.override.yml) is auto-merged when you
|
||||
# run `docker compose up` from this directory and switches images to
|
||||
# local builds + DEBUG logging.
|
||||
# docker compose -f docker-compose.yml up -d
|
||||
#
|
||||
# The -f is load-bearing. Without it Compose auto-merges
|
||||
# docker-compose.override.yml, which replaces every image: with a local
|
||||
# build: and turns on DEBUG logging — the contributor path. Naming this file
|
||||
# explicitly skips the override and pulls the published :latest images.
|
||||
#
|
||||
# FabledCurator has no authentication. Whatever can reach ${PORT} is an
|
||||
# administrator, including over the stored Patreon/SubscribeStar/Pixiv session
|
||||
# cookies. Do not publish this port beyond a network you trust — see
|
||||
# "Before you expose it" in README.md.
|
||||
|
||||
# Rolling-deploy safety (Swarm / `docker stack deploy`): update one task at a
|
||||
# time, START the new task before stopping the old (zero-downtime via the ingress
|
||||
@@ -74,7 +83,25 @@ services:
|
||||
retries: 5
|
||||
|
||||
web:
|
||||
image: git.fabledsword.com/bvandeusen/fabledcurator:dev
|
||||
# :latest, NOT :dev — this file IS the install path.
|
||||
#
|
||||
# `docker compose up -d` merges docker-compose.override.yml, which sets
|
||||
# build: for all five app services, and a build: wins over image:. So a
|
||||
# contributor never pulls this tag and is unaffected by what it says.
|
||||
#
|
||||
# The tag is consulted only on `docker compose -f docker-compose.yml up -d`
|
||||
# — the documented production path, which skips the override. That is a
|
||||
# stranger installing the product, and they must land on the stable channel.
|
||||
#
|
||||
# :latest is main, which IS production (rule 147). :dev is the rolling
|
||||
# bleeding-edge channel we work out of, republished several times a day with
|
||||
# no stability promise. This file pinned :dev on all five services until
|
||||
# 2026-08-31 (#3270), so the documented install shipped development builds.
|
||||
# It went unnoticed because nobody who works on the project takes this path:
|
||||
# the operator deploys from a swarm stack file, contributors get the
|
||||
# override. Do not "fix" this back to :dev while debugging — use the
|
||||
# override, or -f with an explicit tag on the command line.
|
||||
image: git.fabledsword.com/bvandeusen/fabledcurator:latest
|
||||
command: ["web"]
|
||||
# Graceful shutdown: give the container time to drain in-flight work on a
|
||||
# deploy (docker SIGTERMs, then SIGKILLs after this window — default is only
|
||||
@@ -105,24 +132,46 @@ services:
|
||||
CELERY_BROKER_URL: redis://redis:6379/0
|
||||
CELERY_RESULT_BACKEND: redis://redis:6379/0
|
||||
SECRET_KEY: ${SECRET_KEY:-dev_secret_key_not_for_production_change_me}
|
||||
EXTENSION_API_KEY: ${EXTENSION_API_KEY:-}
|
||||
LOG_LEVEL: ${LOG_LEVEL:-INFO}
|
||||
# First boot only. FabledCurator refuses to start until the credential
|
||||
# encryption key at /images/secrets/credential_key.b64 exists, and
|
||||
# refuses to create one unless told to — auto-creating is
|
||||
# indistinguishable from a restore that lost ./images/secrets, where it
|
||||
# would mint a key that decrypts nothing and leave an instance that looks
|
||||
# healthy while every paywalled download fails.
|
||||
#
|
||||
# Passed through EXPLICITLY because a variable in `.env` is only used for
|
||||
# ${...} interpolation; it does not reach the container unless it is
|
||||
# named here. Defaulted to empty so the refusal stands for everyone who
|
||||
# has not opted in — the app tests for exactly "1".
|
||||
#
|
||||
# Set it in .env for one `up`, then remove it. See .env.example.
|
||||
CURATOR_BOOTSTRAP_NEW_KEY: ${CURATOR_BOOTSTRAP_NEW_KEY:-}
|
||||
volumes:
|
||||
- ./images:/images
|
||||
- ./import:/import
|
||||
# FC-5 legacy migration: bind-mount the host's ImageRepo images dir
|
||||
# under /import (FC's existing filesystem scan picks them up). Read-only
|
||||
# is sufficient — FC copies into /images during the scan. The worker +
|
||||
# scheduler services see the same /import via their own mounts below
|
||||
# because of /import volume reuse. Edit the host path to match your
|
||||
# install before running Settings → Maintenance → Legacy migration.
|
||||
# - /var/lib/imagerepo/images:/import/imagerepo:ro
|
||||
# /import is a staging area for scripting a one-off ingest of a library
|
||||
# you already have on disk. Drop files in ./import, or bind-mount an
|
||||
# existing directory under it as below, then trigger the scan:
|
||||
#
|
||||
# curl -X POST http://localhost:8080/api/import/trigger
|
||||
#
|
||||
# Read-only is sufficient — FC copies into /images during the scan. The
|
||||
# worker + scheduler services mount the same /import so the scan can run
|
||||
# on whichever lane picks it up.
|
||||
#
|
||||
# Deliberately has no UI. The manual-scan surface was retired 2026-07-02
|
||||
# once imports arrived via subscriptions + the extension, and the call
|
||||
# not to restore it stands (operator, 2026-09-02): folder ingestion
|
||||
# brings complexity the product does not need. The endpoint stays as an
|
||||
# unsupported escape hatch; the supported way in is Subscriptions.
|
||||
# - /srv/media/my-library:/import/my-library:ro
|
||||
depends_on:
|
||||
postgres: { condition: service_healthy }
|
||||
redis: { condition: service_healthy }
|
||||
|
||||
worker:
|
||||
image: git.fabledsword.com/bvandeusen/fabledcurator:dev
|
||||
image: git.fabledsword.com/bvandeusen/fabledcurator:latest
|
||||
command: ["worker"]
|
||||
# Drain in-flight import/thumbnail/download tasks before SIGKILL on deploy.
|
||||
stop_grace_period: 90s
|
||||
@@ -142,7 +191,7 @@ services:
|
||||
redis: { condition: service_healthy }
|
||||
|
||||
scheduler:
|
||||
image: git.fabledsword.com/bvandeusen/fabledcurator:dev
|
||||
image: git.fabledsword.com/bvandeusen/fabledcurator:latest
|
||||
command: ["scheduler"]
|
||||
# Quick maintenance/scan lane + beat — short tasks, modest drain window.
|
||||
stop_grace_period: 60s
|
||||
@@ -163,7 +212,7 @@ services:
|
||||
# 30-min backup or a multi-chunk audit can never starve the 5-min recovery
|
||||
# sweeps / vacuum (operator-flagged 2026-06-07). One slot — these are heavy.
|
||||
maintenance-long:
|
||||
image: git.fabledsword.com/bvandeusen/fabledcurator:dev
|
||||
image: git.fabledsword.com/bvandeusen/fabledcurator:latest
|
||||
command: ["worker"]
|
||||
# Longest lane (DB backups, library audits, translation backfill) — give it
|
||||
# the most room to finish a chunk gracefully. Chunked + idempotent, so a job
|
||||
@@ -184,7 +233,7 @@ services:
|
||||
redis: { condition: service_healthy }
|
||||
|
||||
ml-worker:
|
||||
image: git.fabledsword.com/bvandeusen/fabledcurator-ml:dev
|
||||
image: git.fabledsword.com/bvandeusen/fabledcurator-ml:latest
|
||||
command: ["ml-worker"]
|
||||
# A single GPU inference pass can run tens of seconds — let it finish.
|
||||
stop_grace_period: 120s
|
||||
|
||||
@@ -5,9 +5,13 @@
|
||||
<img src="/favicon.svg" alt="" class="fc-brand__glyph" width="22" height="22" />
|
||||
<span class="fc-brand__text">FabledCurator</span>
|
||||
</RouterLink>
|
||||
<span class="fc-health" :title="health.label">
|
||||
<RouterLink
|
||||
:to="{ name: 'settings', query: { tab: 'system' } }"
|
||||
class="fc-health" :title="health.label"
|
||||
:aria-label="`System health: ${health.label}`"
|
||||
>
|
||||
<v-icon size="x-small" :color="health.color">{{ health.icon }}</v-icon>
|
||||
</span>
|
||||
</RouterLink>
|
||||
<PipelineStatusChip />
|
||||
</div>
|
||||
|
||||
@@ -64,13 +68,15 @@
|
||||
</template>
|
||||
|
||||
<script setup>
|
||||
import { computed, onBeforeUnmount, onMounted, ref } from 'vue'
|
||||
import { computed, onBeforeUnmount, onMounted, onUnmounted, ref } from 'vue'
|
||||
import { useRoute } from 'vue-router'
|
||||
import router, { FRONT_DOOR } from '../router.js'
|
||||
import { useSystemStore } from '../stores/system.js'
|
||||
import { useSystemHealthStore } from '../stores/systemHealth.js'
|
||||
import PipelineStatusChip from './PipelineStatusChip.vue'
|
||||
|
||||
const system = useSystemStore()
|
||||
const healthStore = useSystemHealthStore()
|
||||
|
||||
// Publish the nav's REAL height as --fc-nav-h so full-height workspaces
|
||||
// (Explore/Subscriptions) and sticky sub-headers pin to it exactly instead of a
|
||||
@@ -116,15 +122,55 @@ const settingsRoute = computed(() =>
|
||||
navRoutes.value.find(r => r.name === 'settings') || null
|
||||
)
|
||||
|
||||
// The dot beside the brand, and the only ambient signal that something in the
|
||||
// stack has stopped (milestone 365).
|
||||
//
|
||||
// It used to read /api/health — a no-DB liveness check that proves the WEB
|
||||
// container is serving and nothing else. Green there while the worker was dead
|
||||
// is exactly what it looked like, and a green dot next to the product name is
|
||||
// read as "everything is fine". It now reflects the whole-stack verdict.
|
||||
//
|
||||
// Deliberately re-using this element rather than adding a second indicator:
|
||||
// there were already three partial surfaces (this, the pipeline chip, the
|
||||
// Settings Activity tab) and a fourth would have made the question harder to
|
||||
// answer, not easier. This is the one that already occupied the slot.
|
||||
const health = computed(() => {
|
||||
if (system.healthy === null) {
|
||||
const overall = healthStore.overall
|
||||
if (overall === null) {
|
||||
return { icon: 'mdi-circle-outline', color: 'on-surface', label: 'checking…' }
|
||||
}
|
||||
if (system.healthy === true) {
|
||||
return { icon: 'mdi-circle', color: 'success', label: 'healthy' }
|
||||
if (overall === 'ok') {
|
||||
return { icon: 'mdi-circle', color: 'success', label: 'All parts running' }
|
||||
}
|
||||
return { icon: 'mdi-alert-circle', color: 'error', label: 'unreachable' }
|
||||
// Name what is wrong in the tooltip. "Something is unhealthy" sends someone
|
||||
// hunting; "Scheduler has not checked in for 6 min" does not.
|
||||
const worst = healthStore.problems[0]
|
||||
const others = healthStore.problems.length - 1
|
||||
const suffix = others > 0 ? ` (+${others} more)` : ''
|
||||
if (overall === 'down') {
|
||||
return {
|
||||
icon: 'mdi-alert-circle', color: 'error',
|
||||
label: (worst?.detail || 'A part has stopped') + suffix,
|
||||
}
|
||||
}
|
||||
if (overall === 'stale') {
|
||||
return {
|
||||
icon: 'mdi-alert', color: 'warning',
|
||||
label: (worst?.detail || 'A part is quiet') + suffix,
|
||||
}
|
||||
}
|
||||
return { icon: 'mdi-help-circle-outline', color: 'on-surface', label: 'Health unknown' }
|
||||
})
|
||||
|
||||
const HEALTH_POLL_MS = 15_000
|
||||
let healthTimer = null
|
||||
onMounted(() => {
|
||||
healthStore.refresh()
|
||||
healthTimer = setInterval(() => {
|
||||
if (!document.hidden) healthStore.refresh()
|
||||
}, HEALTH_POLL_MS)
|
||||
})
|
||||
onUnmounted(() => { if (healthTimer) clearInterval(healthTimer) })
|
||||
</script>
|
||||
|
||||
<style scoped>
|
||||
@@ -237,7 +283,14 @@ const health = computed(() => {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
flex-shrink: 0;
|
||||
/* A RouterLink since milestone 365 — it is the path to the Settings System
|
||||
tab, not just an indicator. Reset the anchor so turning a span into a
|
||||
link changed nothing about how the nav reads. */
|
||||
text-decoration: none;
|
||||
color: inherit;
|
||||
border-radius: 50%;
|
||||
}
|
||||
.fc-health:hover { background: rgb(var(--v-theme-on-surface) / 0.12); }
|
||||
.fc-nav-right {
|
||||
flex: 1 1 0;
|
||||
min-width: 0;
|
||||
|
||||
@@ -0,0 +1,128 @@
|
||||
<template>
|
||||
<!-- A Settings tab, not a page of its own (operator 2026-09-02): the first
|
||||
cut hung this off the health dot alone, which is a target you have to
|
||||
already suspect something to look for. Settings is where someone goes
|
||||
to ask the instance about itself, so it lives beside Activity. -->
|
||||
<div>
|
||||
<div class="d-flex align-center mb-1">
|
||||
<v-spacer />
|
||||
<span class="fc-sys__checked">
|
||||
{{ store.checkedAt ? `checked ${formatRelative(store.checkedAt)}` : 'checking…' }}
|
||||
</span>
|
||||
</div>
|
||||
|
||||
<p class="fc-sys__lede text-body-2 mb-5">
|
||||
Every moving part of FabledCurator and whether it is still checking in.
|
||||
Parts are learned as they appear, so anything that has run at least once
|
||||
stays listed — that is what lets a stopped one be noticed rather than
|
||||
simply vanishing.
|
||||
</p>
|
||||
|
||||
<v-alert
|
||||
v-if="store.lastError" type="error" variant="tonal" density="compact" class="mb-4"
|
||||
>
|
||||
Could not reach FabledCurator: {{ store.lastError }}
|
||||
</v-alert>
|
||||
|
||||
<v-card v-else variant="flat" class="fc-sys__card">
|
||||
<div v-if="!store.parts.length" class="pa-6 text-center fc-sys__muted">
|
||||
Still gathering — this fills in on the first check.
|
||||
</div>
|
||||
|
||||
<div
|
||||
v-for="part in store.parts" :key="part.key"
|
||||
class="fc-sys__row" :class="`fc-sys__row--${part.state}`"
|
||||
>
|
||||
<span class="fc-sys__dot" :class="`fc-sys__dot--${part.state}`" />
|
||||
|
||||
<div class="fc-sys__body">
|
||||
<div class="fc-sys__name">
|
||||
{{ part.name }}
|
||||
<span class="fc-sys__kind">{{ kindLabel(part.kind) }}</span>
|
||||
</div>
|
||||
<!-- The sentence, not just a chip. At the moment someone is deciding
|
||||
whether to go and open Portainer, "has not checked in for 6 min"
|
||||
is the thing that answers them. -->
|
||||
<div class="fc-sys__detail">{{ part.detail }}</div>
|
||||
</div>
|
||||
|
||||
<div class="fc-sys__meta">
|
||||
<div v-if="part.last_seen_at" :title="part.last_seen_at">
|
||||
seen {{ formatRelative(part.last_seen_at) }}
|
||||
</div>
|
||||
<div v-if="part.latency_ms != null">{{ part.latency_ms }} ms</div>
|
||||
<div v-if="part.queues?.length" class="fc-sys__queues">{{ part.queues.join(', ') }}</div>
|
||||
</div>
|
||||
</div>
|
||||
</v-card>
|
||||
|
||||
<p v-if="store.thresholds" class="fc-sys__foot text-caption mt-4">
|
||||
A part is called stale after
|
||||
{{ Math.round(store.thresholds.stale_after_seconds / 60) }} min without a
|
||||
check-in and treated as stopped after
|
||||
{{ Math.round(store.thresholds.down_after_seconds / 60) }} min. The window
|
||||
is deliberately wide: a rolling deploy briefly runs two of a service and
|
||||
then neither, and an indicator that reddened on every update would stop
|
||||
being read.
|
||||
</p>
|
||||
</div>
|
||||
</template>
|
||||
|
||||
<script setup>
|
||||
import { onMounted } from 'vue'
|
||||
|
||||
import { useSystemHealthStore } from '../../stores/systemHealth.js'
|
||||
import { formatRelative } from '../../utils/date.js'
|
||||
|
||||
const store = useSystemHealthStore()
|
||||
|
||||
function kindLabel(kind) {
|
||||
if (kind === 'celery') return 'background worker'
|
||||
if (kind === 'agent') return 'GPU agent'
|
||||
if (kind === 'datastore') return 'datastore'
|
||||
return kind
|
||||
}
|
||||
|
||||
// No timer of its own. TopNav already polls this same pinia store every 15s
|
||||
// for the health dot, and it is mounted on every route this tab is reachable
|
||||
// from — a second interval here would just double the request rate for a 5s
|
||||
// freshness gain. v-window keeps a visited item MOUNTED (hidden, not
|
||||
// destroyed), so a local timer would also have kept firing behind Maintenance.
|
||||
// One refresh on open, so arriving at the tab doesn't wait out the nav's tick.
|
||||
onMounted(() => { store.refresh() })
|
||||
</script>
|
||||
|
||||
<style scoped>
|
||||
.fc-sys__lede, .fc-sys__muted, .fc-sys__checked, .fc-sys__foot {
|
||||
color: rgb(var(--v-theme-on-surface) / 0.66);
|
||||
}
|
||||
.fc-sys__checked { font-size: 0.78rem; }
|
||||
.fc-sys__card { background: rgb(var(--v-theme-on-surface) / 0.04); }
|
||||
|
||||
.fc-sys__row {
|
||||
display: flex; align-items: center; gap: 12px;
|
||||
padding: 12px 16px;
|
||||
border-bottom: 1px solid rgb(var(--v-theme-on-surface) / 0.08);
|
||||
}
|
||||
.fc-sys__row:last-child { border-bottom: 0; }
|
||||
|
||||
.fc-sys__dot { width: 9px; height: 9px; border-radius: 50%; flex: 0 0 auto; }
|
||||
.fc-sys__dot--ok { background: rgb(var(--v-theme-success)); }
|
||||
.fc-sys__dot--stale { background: rgb(var(--v-theme-warning)); }
|
||||
.fc-sys__dot--down { background: rgb(var(--v-theme-error)); }
|
||||
.fc-sys__dot--unknown { background: rgb(var(--v-theme-on-surface) / 0.35); }
|
||||
|
||||
.fc-sys__body { min-width: 0; flex: 1 1 auto; }
|
||||
.fc-sys__name { font-weight: 600; }
|
||||
.fc-sys__kind {
|
||||
margin-left: 8px; font-weight: 400; font-size: 0.72rem; text-transform: uppercase;
|
||||
letter-spacing: 0.04em; color: rgb(var(--v-theme-on-surface) / 0.5);
|
||||
}
|
||||
.fc-sys__detail { font-size: 0.82rem; color: rgb(var(--v-theme-on-surface) / 0.72); }
|
||||
|
||||
.fc-sys__meta {
|
||||
text-align: right; font-size: 0.75rem; flex: 0 0 auto;
|
||||
font-variant-numeric: tabular-nums; color: rgb(var(--v-theme-on-surface) / 0.6);
|
||||
}
|
||||
.fc-sys__queues { opacity: 0.75; }
|
||||
</style>
|
||||
@@ -45,6 +45,13 @@ const routes = [
|
||||
|
||||
// Settings — config, pinned to the right of the nav (TopNav special-cases it).
|
||||
{ path: '/settings', name: 'settings', component: SettingsView, meta: { title: 'Settings', stickyChrome: true } },
|
||||
// System health is a Settings TAB, not a route of its own (operator
|
||||
// 2026-09-02). It first shipped as /system reachable only from the health
|
||||
// dot, which is a target you have to already suspect something to go
|
||||
// looking for. Settings is where someone goes to ask the instance about
|
||||
// itself. The path stays as a redirect so the dot's old link, and any
|
||||
// bookmark from that build, still land somewhere real.
|
||||
{ path: '/system', name: 'system', redirect: () => ({ name: 'settings', query: { tab: 'system' } }) },
|
||||
|
||||
// The old standalone paths now redirect into the Browse hub, preserving any
|
||||
// deep-link query (e.g. /posts?post_id=N → /browse?tab=posts&post_id=N). The
|
||||
|
||||
@@ -4,7 +4,12 @@ import { useApi } from '../composables/useApi.js'
|
||||
|
||||
export const useSystemStore = defineStore('system', () => {
|
||||
const api = useApi()
|
||||
const healthy = ref(null) // null=unknown, true=ok, false=down
|
||||
// NOT what the nav dot reads any more (milestone 365): that is the
|
||||
// whole-stack verdict in systemHealth.js. /api/health only proves the web
|
||||
// container is serving, which is why a green dot here sat happily beside a
|
||||
// dead worker. refreshHealth() is still called — it is also how build/version
|
||||
// info arrives — so this stays as its by-product rather than its purpose.
|
||||
const healthy = ref(null)
|
||||
// What the instance says it is. Since milestone 318 stopped publishing
|
||||
// version image tags, this is the only answer to "which build is this?" —
|
||||
// there is no registry name left to check it against.
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
import { defineStore } from 'pinia'
|
||||
import { computed, ref } from 'vue'
|
||||
|
||||
import { useApi } from '../composables/useApi.js'
|
||||
|
||||
// Whole-stack health: is every part of FabledCurator running (milestone 365)?
|
||||
//
|
||||
// Distinct from `system.js`, which polls /api/health — a no-DB liveness check
|
||||
// that only proves the web container is serving. That endpoint answers "can I
|
||||
// reach the API"; this one answers "is anything broken", which is the question
|
||||
// a green dot beside the brand was already being read as answering.
|
||||
//
|
||||
// Also distinct from `systemActivity.js`, which is about what the pipeline is
|
||||
// DOING — queue depths, running tasks, failures. Running and alive are
|
||||
// different questions and they fail independently: a perfectly idle stack with
|
||||
// a dead worker looks identical to a healthy one on the activity surfaces.
|
||||
export const useSystemHealthStore = defineStore('systemHealth', () => {
|
||||
const api = useApi()
|
||||
|
||||
const overall = ref(null) // null until the first answer: unknown ≠ ok
|
||||
const parts = ref([])
|
||||
const checkedAt = ref(null)
|
||||
const thresholds = ref(null) // server-owned, so the UI keeps no second copy
|
||||
const lastError = ref(null)
|
||||
|
||||
async function refresh() {
|
||||
try {
|
||||
const body = await api.get('/api/system/health')
|
||||
overall.value = body.overall
|
||||
parts.value = body.parts || []
|
||||
checkedAt.value = body.checked_at
|
||||
thresholds.value = body.thresholds || null
|
||||
lastError.value = null
|
||||
} catch (e) {
|
||||
// The endpoint is built never to fail because a dependency failed, so a
|
||||
// throw here means the API itself is unreachable — which is its own kind
|
||||
// of unhealthy and must not be shown as "ok".
|
||||
lastError.value = e.message
|
||||
overall.value = 'unknown'
|
||||
}
|
||||
return overall.value
|
||||
}
|
||||
|
||||
// The parts worth naming in a tooltip — everything that is not ok, worst
|
||||
// first. The endpoint already sorts that way.
|
||||
const problems = computed(() => parts.value.filter(p => p.state !== 'ok'))
|
||||
|
||||
return { overall, parts, checkedAt, thresholds, lastError, problems, refresh }
|
||||
})
|
||||
@@ -14,6 +14,7 @@
|
||||
style="position: sticky; top: var(--fc-nav-h, 64px); z-index: 4;"
|
||||
>
|
||||
<v-tab value="overview">Overview</v-tab>
|
||||
<v-tab value="system">System</v-tab>
|
||||
<v-tab value="activity">Activity</v-tab>
|
||||
<v-tab value="cleanup">Cleanup</v-tab>
|
||||
<v-tab value="maintenance">Maintenance</v-tab>
|
||||
@@ -42,6 +43,13 @@
|
||||
</v-alert>
|
||||
</v-window-item>
|
||||
|
||||
<!-- Is every part of the stack still running (milestone 365). Sits
|
||||
beside Activity deliberately: Activity answers "what is the queue
|
||||
doing", this answers "is anything left to do it". -->
|
||||
<v-window-item value="system">
|
||||
<SystemHealthTab />
|
||||
</v-window-item>
|
||||
|
||||
<v-window-item value="activity">
|
||||
<SystemActivityTab @open-maintenance="tab = 'maintenance'" />
|
||||
</v-window-item>
|
||||
@@ -73,18 +81,24 @@
|
||||
</template>
|
||||
|
||||
<script setup>
|
||||
import { onMounted, onUnmounted, ref, watch } from 'vue'
|
||||
import { onMounted, onUnmounted, watch } from 'vue'
|
||||
import { useSystemStore } from '../stores/system.js'
|
||||
import SystemStatsCards from '../components/settings/SystemStatsCards.vue'
|
||||
import SystemActivitySummary from '../components/settings/SystemActivitySummary.vue'
|
||||
import SystemActivityTab from '../components/settings/SystemActivityTab.vue'
|
||||
import SystemHealthTab from '../components/settings/SystemHealthTab.vue'
|
||||
import GpuActivityPanel from '../components/settings/GpuActivityPanel.vue'
|
||||
import DownloadsActivityPanel from '../components/settings/DownloadsActivityPanel.vue'
|
||||
import MaintenancePanel from '../components/settings/MaintenancePanel.vue'
|
||||
import CleanupView from './CleanupView.vue'
|
||||
import { useTabQuery } from '../composables/useTabQuery.js'
|
||||
import { useMLStore } from '../stores/ml.js'
|
||||
|
||||
const tab = ref('overview')
|
||||
// ?tab= sync (the same composable Browse/Subscriptions use) so a tab can be
|
||||
// linked TO — the health dot beside the brand points at ?tab=system, and the
|
||||
// old /system path redirects there.
|
||||
const VALID_TABS = ['overview', 'system', 'activity', 'cleanup', 'maintenance']
|
||||
const { tab } = useTabQuery(VALID_TABS, 'overview')
|
||||
const system = useSystemStore()
|
||||
const mlStore = useMLStore()
|
||||
|
||||
|
||||
@@ -40,6 +40,14 @@ describe('router', () => {
|
||||
expect(router.currentRoute.value.query.post_id).toBe('7')
|
||||
})
|
||||
|
||||
it('/system redirects into the Settings System tab', async () => {
|
||||
// It shipped as a standalone page for one build; the health dot and any
|
||||
// bookmark from it must still land on the surface, which is now a tab.
|
||||
await router.push('/system')
|
||||
expect(router.currentRoute.value.name).toBe('settings')
|
||||
expect(router.currentRoute.value.query.tab).toBe('system')
|
||||
})
|
||||
|
||||
it('series-read is an immersive route', () => {
|
||||
const r = router.resolve('/series/5/read')
|
||||
expect(r.name).toBe('series-read')
|
||||
|
||||
+22
-1
@@ -90,7 +90,7 @@ DERIVER='scripts/artifacts.sh'
|
||||
|
||||
|
||||
usage() {
|
||||
echo "usage: artifacts.sh {paths|revision|version} {web|ml|agent|extension}" >&2
|
||||
echo "usage: artifacts.sh {paths|revision|version|epoch} {web|ml|agent|extension}" >&2
|
||||
exit 2
|
||||
}
|
||||
|
||||
@@ -149,6 +149,26 @@ cmd_revision() {
|
||||
echo "$(newest "$1")" | cut -d' ' -f2 | cut -c1-12
|
||||
}
|
||||
|
||||
# The BUILD CLOCK: the same commit's unix timestamp, for SOURCE_DATE_EPOCH.
|
||||
#
|
||||
# buildkit stamps the image config's `created` field and every history entry
|
||||
# with the wall clock of the build unless this is set, so two builds of
|
||||
# identical source produce different config blobs and therefore different
|
||||
# manifest digests. That is #3265: the weekly refresh republished all three
|
||||
# `:latest` tags on 2026-08-30 with every content step CACHED and the bases
|
||||
# resolved to unchanged digests — nothing was different, and the digest moved
|
||||
# anyway. A digest that changes on a calendar cannot also mean "the content
|
||||
# changed", which is the only thing anyone wants it for.
|
||||
#
|
||||
# It is the same commit `revision` and `version` name — deliberately, and this
|
||||
# is the point of routing it through `newest()` rather than taking git's word
|
||||
# separately. Three values derived from three lookups can disagree; three
|
||||
# views of one lookup cannot. Note #3127 §2 is the record of what a second
|
||||
# clock costs.
|
||||
cmd_epoch() {
|
||||
echo "$(newest "$1")" | cut -d' ' -f1
|
||||
}
|
||||
|
||||
# The VERSION: `YYYY.MM.DD.HHMM`, zero-padded, UTC. One shape across the whole
|
||||
# family (note #3127 §1, rule 148) — the number an instance reports about
|
||||
# itself, and, with a `v` in front, the release tag naming the same build.
|
||||
@@ -197,5 +217,6 @@ case "$1" in
|
||||
paths) cmd_paths "$2" ;;
|
||||
revision) cmd_revision "$2" ;;
|
||||
version) cmd_version "$2" ;;
|
||||
epoch) cmd_epoch "$2" ;;
|
||||
*) usage ;;
|
||||
esac
|
||||
|
||||
+112
-8
@@ -33,6 +33,32 @@ history. Ancestry is immune to the shape change, and it is also the more honest
|
||||
question: "what is in this that was not in the last one" IS a reachability
|
||||
question.
|
||||
|
||||
Ancestry alone is not enough, though, and milestone 328 is where that showed.
|
||||
The 28 `v26.*` tags are still in the repo — the operator kept them as history
|
||||
when their releases were deleted — so `--match v*` walks straight back to
|
||||
`v26.06.04.0` and reports 533 commits. That span is not a changelog: nobody has
|
||||
run `v26.06.04.0`, its release page no longer exists to compare against, and
|
||||
the 200 lines that survive truncation are precisely the internal build-out that
|
||||
milestone 328 exists to stop shipping. So the match is `v[0-9][0-9][0-9][0-9].*`
|
||||
— rule 148's four-digit-year shape — which is exactly the set of tags that name
|
||||
a release a reader could have been running. A pre-convention tag is history,
|
||||
not a predecessor.
|
||||
|
||||
## The first release has no changelog, and should not pretend to
|
||||
|
||||
Once the match is narrowed, the first rule-148 tag reaches no predecessor at
|
||||
all, and the old fallback — diff against the whole history — is worse than the
|
||||
problem it replaced. The honest content for a release nobody has a previous
|
||||
version of is what the thing IS.
|
||||
|
||||
So a release with no reachable predecessor renders the product overview instead
|
||||
of a commit list. It is read out of README.md between `<!-- overview:start -->`
|
||||
and `<!-- overview:end -->` rather than written here, for the same reason the
|
||||
changelog is derived: two hand-maintained descriptions of one product drift,
|
||||
and nothing ever catches it. The release page and the repo front page are one
|
||||
source. Every later release goes back to being a changelog, which is what §5 of
|
||||
note #3127 says a release is for.
|
||||
|
||||
## Re-runs update, they do not fall through
|
||||
|
||||
Note #3127 §6.7: a publisher that POSTs and recovers the id from a `409` never
|
||||
@@ -91,18 +117,45 @@ def git_ok(*args: str) -> str | None:
|
||||
|
||||
|
||||
def previous_tag(ref: str, tag: str | None) -> str | None:
|
||||
"""The most recent `v*` tag reachable from `ref`, excluding `tag` itself.
|
||||
"""The most recent rule-148 tag reachable from `ref`, excluding `tag` itself.
|
||||
|
||||
`--exclude` rather than `<ref>^` so this is the same call whether or not
|
||||
`ref` is the tag being released — and so it does not blow up on a root
|
||||
commit that has no parent to walk to.
|
||||
|
||||
The glob deliberately does NOT match the old `v26.*` tags. They are kept as
|
||||
history and their releases are gone, so naming one as the predecessor emits
|
||||
a span nobody can look up. See the module docstring.
|
||||
"""
|
||||
args = ["describe", "--tags", "--abbrev=0", "--match", "v*"]
|
||||
args = ["describe", "--tags", "--abbrev=0", "--match", "v[0-9][0-9][0-9][0-9].*"]
|
||||
if tag:
|
||||
args += ["--exclude", tag]
|
||||
return git_ok(*args, ref)
|
||||
|
||||
|
||||
def product_overview() -> str | None:
|
||||
"""The product description, lifted verbatim from README.md.
|
||||
|
||||
Returns None if the markers are absent or empty — a missing overview is
|
||||
reported as a note and the release still publishes, on the same reasoning
|
||||
as cross_checks(): the release is the useful object even when one part of
|
||||
the derivation could not run.
|
||||
"""
|
||||
root = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
try:
|
||||
with open(os.path.join(root, "README.md"), encoding="utf-8") as fh:
|
||||
readme = fh.read()
|
||||
except OSError:
|
||||
return None
|
||||
match = re.search(
|
||||
r"<!--\s*overview:start\s*-->(.*?)<!--\s*overview:end\s*-->",
|
||||
readme, re.S,
|
||||
)
|
||||
if not match:
|
||||
return None
|
||||
return match.group(1).strip() or None
|
||||
|
||||
|
||||
def commits(previous: str | None, ref: str) -> list[str]:
|
||||
"""The subjects between the previous release and this one.
|
||||
|
||||
@@ -125,7 +178,10 @@ def truncate(log: list[str]) -> tuple[list[str], str | None]:
|
||||
)
|
||||
|
||||
|
||||
def render(tag: str, sha: str, previous: str | None, log: list[str], notes: list[str]) -> str:
|
||||
def render(
|
||||
tag: str, sha: str, previous: str | None, log: list[str], notes: list[str],
|
||||
overview: str | None,
|
||||
) -> str:
|
||||
short = sha[:7]
|
||||
parts = []
|
||||
|
||||
@@ -135,6 +191,25 @@ def render(tag: str, sha: str, previous: str | None, log: list[str], notes: list
|
||||
# is the failure this whole milestone is about.
|
||||
parts.append("\n".join(f"> **Note:** {n}" for n in notes))
|
||||
|
||||
# No predecessor means nobody reading this has run an earlier one, so the
|
||||
# release describes the product rather than a diff. The overview is
|
||||
# README.md's own words — see the module docstring on why it is not
|
||||
# written here.
|
||||
if previous is None and overview:
|
||||
parts.append(overview)
|
||||
parts.append(
|
||||
"## Installing\n\n"
|
||||
"```\ncurl -O https://git.fabledsword.com/bvandeusen/FabledCurator/raw/"
|
||||
f"tag/{tag}/docker-compose.yml\ncurl -O https://git.fabledsword.com/"
|
||||
f"bvandeusen/FabledCurator/raw/tag/{tag}/.env.example\n"
|
||||
"mv .env.example .env # then set SECRET_KEY, DB_PASSWORD\n"
|
||||
"docker compose -f docker-compose.yml up -d\n```\n\n"
|
||||
"**Read \"Before you expose it\" in the README first.** FabledCurator "
|
||||
"has no login, and it stores live platform session cookies for "
|
||||
"accounts that usually have a payment method attached. Bind it to a "
|
||||
"network you trust."
|
||||
)
|
||||
|
||||
parts.append(
|
||||
f"Built from `{short}`. The rollback unit is the immutable `:c-` tag "
|
||||
f"(rule 145) — these three move together:\n\n```\n"
|
||||
@@ -142,7 +217,19 @@ def render(tag: str, sha: str, previous: str | None, log: list[str], notes: list
|
||||
+ "\n```"
|
||||
)
|
||||
|
||||
heading = f"## Changes since {previous}" if previous else "## Changes"
|
||||
if previous is None:
|
||||
# Deliberately NOT a commit list. The alternative is the whole history
|
||||
# truncated to MAX_COMMITS, which is 200 lines of internal build-out
|
||||
# presented to someone who has never seen this project.
|
||||
parts.append(
|
||||
"---\n\n_First release under rule 148's `vYYYY.MM.DD.HHMM` shape, so "
|
||||
"there is no predecessor to diff against and no changelog to derive. "
|
||||
"The description above is README.md's, quoted at publish time. Later "
|
||||
"releases carry the commits since the previous one._"
|
||||
)
|
||||
return "\n\n".join(parts)
|
||||
|
||||
heading = f"## Changes since {previous}"
|
||||
if log:
|
||||
parts.append(heading + "\n\n" + "\n".join(f"- {line}" for line in log))
|
||||
else:
|
||||
@@ -152,10 +239,9 @@ def render(tag: str, sha: str, previous: str | None, log: list[str], notes: list
|
||||
"names the same source under a new name._"
|
||||
)
|
||||
|
||||
span = f"{previous}..{tag}" if previous else tag
|
||||
parts.append(
|
||||
f"---\n\n_Derived at publish time from `git log --no-merges {span}`. "
|
||||
f"Nothing here is hand-maintained._"
|
||||
f"---\n\n_Derived at publish time from "
|
||||
f"`git log --no-merges {previous}..{tag}`. Nothing here is hand-maintained._"
|
||||
)
|
||||
return "\n\n".join(parts)
|
||||
|
||||
@@ -289,13 +375,31 @@ def main() -> None:
|
||||
for note in notes:
|
||||
print(f"release: NOTE {note}")
|
||||
|
||||
# A first release renders the overview instead of a changelog, so the
|
||||
# commit walk is skipped entirely rather than computed and discarded —
|
||||
# `commits(None, ref)` is the whole history and there is no reason to ask
|
||||
# for it.
|
||||
overview = None
|
||||
log: list[str] = []
|
||||
if previous is None:
|
||||
overview = product_overview()
|
||||
if overview is None:
|
||||
note = (
|
||||
"No `<!-- overview:start -->` block found in README.md, so this "
|
||||
"first release has no product description. Published anyway; add "
|
||||
"the markers and re-run the workflow to fill it in."
|
||||
)
|
||||
print(f"release: NOTE {note}")
|
||||
notes.append(note)
|
||||
print("release: no rule-148 predecessor — rendering the product overview")
|
||||
else:
|
||||
log = commits(previous, ref)
|
||||
print(f"release: {len(log)} non-merge commits in the span")
|
||||
log, overflow = truncate(log)
|
||||
if overflow:
|
||||
print(f"release: NOTE {overflow}")
|
||||
notes.append(overflow)
|
||||
body = render(tag or ref, sha, previous, log, notes)
|
||||
body = render(tag or ref, sha, previous, log, notes, overview)
|
||||
|
||||
if args.dry_run or not tag:
|
||||
print("--- body ---")
|
||||
|
||||
@@ -0,0 +1,155 @@
|
||||
"""Prove a freshly built image can still do the things its OS packages provide.
|
||||
|
||||
Run INSIDE the image, not against the source tree. That distinction is the
|
||||
entire reason this file exists.
|
||||
|
||||
`ci.yml`'s lanes run on `ci-python:3.14` and install `requirements.txt`. A base
|
||||
refresh changes neither, so all five lanes stay green through a base bump that
|
||||
breaks the product. What a refresh actually re-resolves is this, from the
|
||||
Dockerfile:
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ffmpeg unar libpq5 postgresql-client zstd megatools \
|
||||
libjpeg62-turbo libwebp7 libpng16-16 ca-certificates
|
||||
|
||||
Unpinned, every build. Nothing else in this repo looks at it.
|
||||
|
||||
So the checks below run the APPLICATION'S OWN code — `Thumbnailer`, which needs
|
||||
no database and no app context — against whatever Pillow and ffmpeg have
|
||||
become. `ffmpeg -version` exiting 0 would pass while a codec removal or an
|
||||
soname bump broke every thumbnail in the library; producing a thumbnail would
|
||||
not.
|
||||
|
||||
Every failure names the package it implicates. This fires on a Sunday,
|
||||
unattended, about a change nobody made deliberately — "assertion failed" a week
|
||||
later teaches nobody anything.
|
||||
|
||||
Usage: docker run --rm -i <image> shell -c 'python3 -' < scripts/smoke_image.py
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import shutil
|
||||
import subprocess
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
from PIL import Image
|
||||
|
||||
from backend.app.services.thumbnailer import Thumbnailer
|
||||
except Exception as exc: # noqa: BLE001 — a smoke test reports, it never raises
|
||||
print(f"smoke: FAILED — could not import the thumbnail path at all: {exc}")
|
||||
print(" Implicates Pillow or its shared libraries (libjpeg62-turbo,")
|
||||
print(" libpng16-16, libwebp7), or the python base image itself.")
|
||||
raise SystemExit(1) from exc
|
||||
|
||||
|
||||
# Binary → what stops working without it. Listed individually because
|
||||
# `--no-install-recommends` means any one of them can vanish on its own when a
|
||||
# dependency chain higher up changes.
|
||||
REQUIRED_BINARIES = {
|
||||
"ffmpeg": "video thumbnails and transcoding (Dockerfile: ffmpeg)",
|
||||
"unar": "archive import — cbz/zip/rar members (Dockerfile: unar)",
|
||||
"pg_dump": "database backup (Dockerfile: postgresql-client)",
|
||||
"zstd": "backup compression, pg_dump | tar --zstd (Dockerfile: zstd)",
|
||||
"megatools": "mega.nz public-link downloads, #830 (Dockerfile: megatools)",
|
||||
}
|
||||
|
||||
|
||||
def check_jpeg(thumbs: Thumbnailer, src: Path) -> None:
|
||||
path = src / "flat.jpg"
|
||||
Image.new("RGB", (900, 400), (30, 90, 160)).save(path, "JPEG")
|
||||
result = thumbs.generate_image_thumbnail(path, "a" * 64)
|
||||
assert result.mime == "image/jpeg", f"mime was {result.mime}"
|
||||
assert result.path.stat().st_size > 0, "no bytes written"
|
||||
# Re-open it. A file that writes but cannot be read back is the shape a
|
||||
# half-broken codec produces, and size alone would not catch it.
|
||||
with Image.open(result.path) as im:
|
||||
im.load()
|
||||
|
||||
|
||||
def check_png_alpha(thumbs: Thumbnailer, src: Path) -> None:
|
||||
path = src / "alpha.png"
|
||||
Image.new("RGBA", (400, 900), (200, 40, 40, 128)).save(path, "PNG")
|
||||
result = thumbs.generate_image_thumbnail(path, "b" * 64)
|
||||
assert result.mime == "image/png", f"mime was {result.mime}"
|
||||
with Image.open(result.path) as im:
|
||||
im.load()
|
||||
assert im.mode in ("RGBA", "LA", "P"), f"alpha lost, mode={im.mode}"
|
||||
|
||||
|
||||
def check_webp(thumbs: Thumbnailer, src: Path) -> None:
|
||||
path = src / "sample.webp"
|
||||
Image.new("RGB", (500, 500), (10, 140, 70)).save(path, "WEBP")
|
||||
result = thumbs.generate_image_thumbnail(path, "c" * 64)
|
||||
assert result.path.stat().st_size > 0, "no bytes written"
|
||||
|
||||
|
||||
def check_video(thumbs: Thumbnailer, src: Path) -> None:
|
||||
# Synthesised rather than committed as a fixture: a checked-in video is a
|
||||
# binary blob nobody can review, and lavfi ships with every ffmpeg build.
|
||||
#
|
||||
# 3 seconds, not 2. The seek lands at max(1.0, duration * 0.05) = 1.0s, and
|
||||
# a clip barely longer than its own seek is how #1231 produced zero frames.
|
||||
# This check exists to exercise ffmpeg, not to re-litigate that edge.
|
||||
clip = src / "clip.mp4"
|
||||
subprocess.run(
|
||||
["ffmpeg", "-nostdin", "-f", "lavfi", "-i", "testsrc=size=640x360:rate=10",
|
||||
"-t", "3", "-pix_fmt", "yuv420p", "-y", str(clip)],
|
||||
check=True, capture_output=True, timeout=120,
|
||||
)
|
||||
result = thumbs.generate_video_thumbnail(clip, "d" * 64, duration_seconds=3.0)
|
||||
assert result.path.stat().st_size > 0, "no bytes written"
|
||||
with Image.open(result.path) as im:
|
||||
im.load()
|
||||
|
||||
|
||||
CHECKS = (
|
||||
("JPEG thumbnail", "libjpeg62-turbo / Pillow", check_jpeg),
|
||||
("PNG thumbnail (alpha)", "libpng16-16 / Pillow", check_png_alpha),
|
||||
("WebP decode", "libwebp7 / Pillow", check_webp),
|
||||
("video thumbnail", "ffmpeg", check_video),
|
||||
)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
failures: list[str] = []
|
||||
|
||||
print("smoke: binaries the apt layer provides")
|
||||
for binary, purpose in REQUIRED_BINARIES.items():
|
||||
if shutil.which(binary) is None:
|
||||
print(f" FAIL {binary}: not on PATH")
|
||||
failures.append(f"{binary} — {purpose}")
|
||||
else:
|
||||
print(f" ok {binary}")
|
||||
|
||||
print("smoke: the application's own thumbnail path, against this image's libraries")
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
root = Path(tmp)
|
||||
src = root / "src"
|
||||
src.mkdir()
|
||||
thumbs = Thumbnailer(root)
|
||||
for name, implicates, fn in CHECKS:
|
||||
try:
|
||||
fn(thumbs, src)
|
||||
print(f" ok {name}")
|
||||
except Exception as exc: # noqa: BLE001 — report every check, then fail once
|
||||
print(f" FAIL {name}: {exc}")
|
||||
failures.append(f"{name} — {implicates}")
|
||||
|
||||
if failures:
|
||||
print(f"\nsmoke: FAILED — {len(failures)} check(s)")
|
||||
for failure in failures:
|
||||
print(f" - {failure}")
|
||||
print("\nThis image was built against freshly resolved base layers. The")
|
||||
print("named packages are where to look: compare this build's apt versions")
|
||||
print("against the previous :latest before assuming the app changed.")
|
||||
return 1
|
||||
|
||||
print("\nsmoke: all checks passed")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -234,3 +234,56 @@ def test_version_and_revision_describe_the_same_commit(artifact):
|
||||
f"in AMO_UNPADDED may differ here."
|
||||
)
|
||||
assert sha.startswith(revision(artifact))
|
||||
|
||||
|
||||
@pytest.mark.parametrize("artifact", ARTIFACTS)
|
||||
def test_epoch_is_the_same_commit_the_version_names(artifact):
|
||||
"""The build clock and the version must be one lookup, not two.
|
||||
|
||||
`epoch` feeds SOURCE_DATE_EPOCH, which decides the image config's bytes and
|
||||
therefore the manifest digest; `version` is what the instance reports about
|
||||
itself. If they could name different commits, an image would be stamped
|
||||
reproducibly against one commit while claiming to be another — and both
|
||||
values would look perfectly well-formed, exactly like the divergence the
|
||||
test above guards.
|
||||
|
||||
They cannot, because `cmd_epoch` and `cmd_version` are two fields of one
|
||||
`newest()` result. This pins that they stay that way: a future refactor
|
||||
that gave epoch its own `git log` would pass every other test here.
|
||||
"""
|
||||
epoch = artifacts("epoch", artifact).strip()
|
||||
assert epoch.isdigit(), f"{artifact} epoch is {epoch!r}, not a unix timestamp"
|
||||
|
||||
sha = newest_by_commit_time(artifact)
|
||||
committed = subprocess.run(
|
||||
["git", "show", "-s", "--format=%ct", sha],
|
||||
capture_output=True, text=True, check=True, cwd=ROOT,
|
||||
).stdout.strip()
|
||||
assert epoch == committed, (
|
||||
f"{artifact} derives epoch {epoch}, but its newest shipped commit "
|
||||
f"{sha[:12]} was committed at {committed}. SOURCE_DATE_EPOCH would "
|
||||
f"pin the image config to a commit the version does not name."
|
||||
)
|
||||
|
||||
# And the two renderings must agree, which is the property that actually
|
||||
# matters at build time: same commit in, same digest and same reported
|
||||
# version out.
|
||||
rendered = subprocess.run(
|
||||
["git", "show", "-s", "--format=%cd", "--date=format-local:%Y.%m.%d.%H%M", sha],
|
||||
capture_output=True, text=True, check=True, cwd=ROOT,
|
||||
env={"TZ": "UTC", "PATH": os.environ.get("PATH", "")},
|
||||
).stdout.strip()
|
||||
assert segments(artifacts("version", artifact).strip()) == segments(rendered)
|
||||
|
||||
|
||||
def test_epoch_is_stable_across_calls():
|
||||
"""SOURCE_DATE_EPOCH's entire job is to be the same on the next build.
|
||||
|
||||
A value that moved between two invocations on one unchanged checkout would
|
||||
reintroduce #3265 through the very mechanism meant to close it, and the
|
||||
symptom would be indistinguishable: a digest that changes for no reason.
|
||||
"""
|
||||
for artifact in ARTIFACTS:
|
||||
first = artifacts("epoch", artifact).strip()
|
||||
second = artifacts("epoch", artifact).strip()
|
||||
assert first == second, f"{artifact} epoch moved: {first} then {second}"
|
||||
|
||||
+101
-17
@@ -24,12 +24,36 @@ SCRIPT = ROOT / "scripts" / "release_notes.py"
|
||||
|
||||
|
||||
def notes(*args: str, cwd: Path | None = None) -> str:
|
||||
"""Run the script the way release.yml does.
|
||||
|
||||
A synthetic repo runs its OWN copy of the script, because the overview is
|
||||
read relative to `__file__` rather than to the cwd — which is right in
|
||||
production (release.yml checks out the tag, so the script IS the tagged
|
||||
tree's copy) and would otherwise make every synthetic repo silently quote
|
||||
FabledCurator's real README.
|
||||
"""
|
||||
root = cwd or ROOT
|
||||
script = root / "scripts" / "release_notes.py"
|
||||
return subprocess.run(
|
||||
["python3", str(SCRIPT), "--dry-run", *args],
|
||||
capture_output=True, text=True, check=True, cwd=cwd or ROOT,
|
||||
["python3", str(script if script.exists() else SCRIPT), "--dry-run", *args],
|
||||
capture_output=True, text=True, check=True, cwd=root,
|
||||
).stdout
|
||||
|
||||
|
||||
OVERVIEW_TEXT = "A synthetic product, described once."
|
||||
|
||||
|
||||
def install_script(repo: Path, *, overview: bool = True) -> None:
|
||||
"""Give a synthetic repo the script and a README to quote."""
|
||||
(repo / "scripts").mkdir(exist_ok=True)
|
||||
(repo / "scripts" / "release_notes.py").write_text(SCRIPT.read_text())
|
||||
(repo / "scripts" / "artifacts.sh").write_text("#!/bin/sh\nexit 1\n")
|
||||
readme = "# Synthetic\n\n"
|
||||
if overview:
|
||||
readme += f"<!-- overview:start -->\n{OVERVIEW_TEXT}\n<!-- overview:end -->\n"
|
||||
(repo / "README.md").write_text(readme)
|
||||
|
||||
|
||||
def body_of(out: str) -> str:
|
||||
assert "--- body ---" in out, f"no body was rendered:\n{out}"
|
||||
return out.split("--- body ---", 1)[1]
|
||||
@@ -56,9 +80,10 @@ def shaped_history(tmp_path: Path) -> Path:
|
||||
repo = tmp_path / "shaped"
|
||||
repo.mkdir()
|
||||
git(repo, "init", "-q", "-b", "main")
|
||||
install_script(repo)
|
||||
for i, tag in enumerate(("v26.06.04.0", "v2026.08.28.2208", "v2026.08.29.1000")):
|
||||
(repo / "f.txt").write_text(f"{i}\n")
|
||||
git(repo, "add", "f.txt")
|
||||
git(repo, "add", "-A")
|
||||
git(repo, "commit", "-q", "-m", f"work landing in {tag}")
|
||||
git(repo, "tag", tag)
|
||||
# One more commit and a merge, so the merge-exclusion test has something to
|
||||
@@ -109,12 +134,57 @@ def test_merges_are_excluded_so_the_list_is_the_work(shaped_history):
|
||||
assert "Merge pull request #999" not in body
|
||||
|
||||
|
||||
def test_the_first_release_still_renders_with_nothing_behind_it(shaped_history):
|
||||
"""No previous tag is reachable from the oldest one. That is a real state,
|
||||
not an error, and it must not take the release down with it."""
|
||||
out = notes("v26.06.04.0", cwd=shaped_history)
|
||||
def test_a_pre_convention_tag_is_history_not_a_predecessor(shaped_history):
|
||||
"""The defect milestone 328 hit, and the reason the match glob narrowed.
|
||||
|
||||
The 28 `v26.*` tags are kept as history while their releases were deleted.
|
||||
Ancestry alone happily names `v26.06.04.0` as the predecessor of the first
|
||||
rule-148 tag — and then the body offers "changes since" a release that no
|
||||
longer exists, over a span (533 commits in the real repo) that is the
|
||||
internal build-out this milestone exists to stop publishing.
|
||||
|
||||
Reachable is not the same as comparable. Only a `vYYYY.` tag names a
|
||||
release a reader could have been running.
|
||||
"""
|
||||
out = notes("v2026.08.28.2208", cwd=shaped_history)
|
||||
assert "previous=<none>" in out
|
||||
assert "## Changes" in body_of(out)
|
||||
assert "v26.06.04.0" not in body_of(out)
|
||||
|
||||
|
||||
def test_the_first_release_describes_the_product_instead_of_diffing(shaped_history):
|
||||
"""No predecessor means nobody reading has run an earlier version, so a
|
||||
changelog has no referent. The alternative the script used to take — diff
|
||||
against the whole history, truncated — puts 200 lines of internal build-out
|
||||
in front of someone meeting the project for the first time."""
|
||||
body = body_of(notes("v2026.08.28.2208", cwd=shaped_history))
|
||||
assert OVERVIEW_TEXT in body
|
||||
assert "## Changes" not in body
|
||||
assert not [ln for ln in body.split("\n") if ln.startswith("- work landing")]
|
||||
|
||||
|
||||
def test_the_overview_is_readmes_words_not_a_second_copy(shaped_history):
|
||||
"""Two hand-maintained descriptions of one product drift and nothing
|
||||
catches it. The release page quotes README.md so there is one source."""
|
||||
readme = (shaped_history / "README.md").read_text()
|
||||
assert OVERVIEW_TEXT in readme
|
||||
assert OVERVIEW_TEXT in body_of(notes("v2026.08.28.2208", cwd=shaped_history))
|
||||
|
||||
|
||||
def test_a_missing_overview_block_is_reported_and_still_publishes(tmp_path):
|
||||
"""Same reasoning as cross_checks(): the release is the useful object even
|
||||
when part of the derivation could not run. Say what is missing, publish
|
||||
anyway — do not leave the operator with a tag and no release."""
|
||||
repo = tmp_path / "no-markers"
|
||||
repo.mkdir()
|
||||
git(repo, "init", "-q", "-b", "main")
|
||||
install_script(repo, overview=False)
|
||||
git(repo, "add", "-A")
|
||||
git(repo, "commit", "-q", "-m", "first")
|
||||
git(repo, "tag", "v2026.09.01.1200")
|
||||
|
||||
out = notes("v2026.09.01.1200", cwd=repo)
|
||||
assert "No `<!-- overview:start -->` block found" in out
|
||||
assert "No `<!-- overview:start -->` block found" in body_of(out)
|
||||
|
||||
|
||||
def test_a_non_tag_ref_renders_but_refuses_to_claim_it_published():
|
||||
@@ -135,15 +205,29 @@ def test_the_rollback_refs_name_all_three_images():
|
||||
assert f"bvandeusen/{image}:c-" in body, f"{image} missing from the rollback refs"
|
||||
|
||||
|
||||
def test_an_unbounded_span_is_truncated_and_says_so():
|
||||
"""With no reachable previous tag the span is the whole history. Emitting
|
||||
eleven hundred lines would bury the one line explaining why there are
|
||||
eleven hundred of them, so the cap is part of the message, not a silent
|
||||
slice."""
|
||||
out = notes("HEAD")
|
||||
if "previous=<none>" not in out:
|
||||
pytest.skip("a previous tag is reachable from HEAD in this checkout")
|
||||
def test_a_long_span_between_two_releases_is_truncated_and_says_so(tmp_path):
|
||||
"""The cap is still reachable, just not by the route it used to be.
|
||||
|
||||
It no longer fires on "no predecessor" — that renders the overview now.
|
||||
What it still guards is two real releases far enough apart that the list
|
||||
stops being something anyone reads, which is the ordinary case for a
|
||||
project that cuts a bookmark twice a year. The cap is part of the message,
|
||||
not a silent slice.
|
||||
"""
|
||||
repo = tmp_path / "long"
|
||||
repo.mkdir()
|
||||
git(repo, "init", "-q", "-b", "main")
|
||||
install_script(repo)
|
||||
git(repo, "add", "-A")
|
||||
git(repo, "commit", "-q", "-m", "scaffold")
|
||||
git(repo, "tag", "v2026.01.01.0000")
|
||||
for i in range(205):
|
||||
git(repo, "commit", "-q", "--allow-empty", "-m", f"fix: change {i}")
|
||||
git(repo, "tag", "v2026.07.01.0000")
|
||||
|
||||
out = notes("v2026.07.01.0000", cwd=repo)
|
||||
assert "previous=v2026.01.01.0000" in out
|
||||
body = body_of(out)
|
||||
listed = [ln for ln in body.split("\n") if ln.startswith("- ")]
|
||||
assert len(listed) <= 200
|
||||
assert len(listed) == 200
|
||||
assert "more than a changelog is for" in body
|
||||
|
||||
Reference in New Issue
Block a user