Compare commits
33
Commits
2529b516e6
...
dev
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ad8392b790 | ||
|
|
5084ba666b | ||
|
|
fe4e0f2b71 | ||
|
|
dc8af8b1a7 | ||
|
|
131237143b | ||
|
|
59d27ef76e | ||
|
|
f630e50e75 | ||
|
|
86abaf0b94 | ||
|
|
4815040d74 | ||
|
|
81b7b6f308 | ||
|
|
bfa9fd678b | ||
|
|
24a2b70a5a | ||
|
|
2c88ad3efb | ||
|
|
bfc4f9cec9 | ||
|
|
b590d25f8f | ||
|
|
635138b0d1 | ||
|
|
c0370069e0 | ||
|
|
3590c478f5 | ||
|
|
8a4af589f1 | ||
|
|
aa71cbbdbf | ||
|
|
973db73221 | ||
|
|
bc4eba636d | ||
|
|
dbc4e8b0c6 | ||
|
|
08418d54a3 | ||
|
|
b1bd2531ad | ||
|
|
389afe2f7b | ||
|
|
b979062dd7 | ||
|
|
573228b9da | ||
|
|
d044e93bdb | ||
|
|
ed2b1adc2e | ||
|
|
5e1996e77f | ||
|
|
98b56330d0 | ||
|
|
6959e1220c |
+97
-18
@@ -1,24 +1,103 @@
|
|||||||
# Database
|
# FabledCurator configuration.
|
||||||
DB_USER=fabledcurator
|
#
|
||||||
DB_PASSWORD=changeme_use_a_real_password
|
# Copy to `.env` and edit before your first production start:
|
||||||
DB_HOST=postgres
|
#
|
||||||
DB_PORT=5432
|
# cp .env.example .env
|
||||||
DB_NAME=fabledcurator
|
#
|
||||||
|
# Only the two values under CHANGE THESE actually need your attention. The
|
||||||
|
# rest have working defaults baked into docker-compose.yml and are listed
|
||||||
|
# here so you know they exist, not because you have to set them.
|
||||||
|
#
|
||||||
|
# Almost nothing else lives here on purpose. FabledCurator is configured from
|
||||||
|
# its own Settings UI, backed by the database — no restart, no YAML. If you
|
||||||
|
# are looking for where to set an import path, a download schedule or an ML
|
||||||
|
# threshold, it is in the app, not in this file.
|
||||||
|
|
||||||
# Redis / Celery
|
|
||||||
CELERY_BROKER_URL=redis://redis:6379/0
|
|
||||||
CELERY_RESULT_BACKEND=redis://redis:6379/0
|
|
||||||
|
|
||||||
# App
|
# ---------------------------------------------------------------------------
|
||||||
# Generate with: openssl rand -hex 32
|
# CHANGE THESE
|
||||||
SECRET_KEY=changeme_32_byte_hex_secret
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
# Extension API key — used in FC-3, lands later but reserved now
|
# The Postgres password. docker-compose.yml falls back to a published default
|
||||||
# Generate with: openssl rand -hex 32
|
# (`fabledcurator_dev`) so that `docker compose up` works with no config at
|
||||||
EXTENSION_API_KEY=
|
# all — which is exactly why you must not leave it at that on a real install.
|
||||||
|
# It is the credential protecting your stored platform session cookies.
|
||||||
|
DB_PASSWORD=
|
||||||
|
|
||||||
# Logging
|
# Sets Quart's app.secret_key. Today it signs nothing: FabledCurator has no
|
||||||
|
# login and uses no session cookies, so no value here is protecting anything
|
||||||
|
# right now. Set it anyway. It is required at boot rather than defaulted so
|
||||||
|
# that the day something session-backed does land, no instance is already
|
||||||
|
# running on a value published in this file.
|
||||||
|
#
|
||||||
|
# openssl rand -hex 32
|
||||||
|
SECRET_KEY=
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# FIRST BOOT ONLY — then delete this line
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
# FabledCurator encrypts your stored platform credentials with a Fernet key it
|
||||||
|
# keeps at /images/secrets/credential_key.b64 — inside the ./images bind mount,
|
||||||
|
# so it outlives the container. On a brand-new install that file does not exist
|
||||||
|
# yet, and the app REFUSES TO START rather than quietly create one:
|
||||||
|
#
|
||||||
|
# MissingCredentialKey: Fernet key file not found at
|
||||||
|
# /images/secrets/credential_key.b64
|
||||||
|
#
|
||||||
|
# That refusal is deliberate. Auto-creating a key is indistinguishable from the
|
||||||
|
# disaster case — a restore that brought the database back but lost
|
||||||
|
# ./images/secrets — and there it would mint a key that cannot decrypt anything,
|
||||||
|
# leaving an instance that looks healthy while every paywalled download fails.
|
||||||
|
# So the choice is yours to make explicitly, once.
|
||||||
|
#
|
||||||
|
# Set this for your first `up`, watch the container come up, then DELETE THE
|
||||||
|
# LINE. Leaving it set disarms the protection permanently, on an instance that
|
||||||
|
# by then has credentials worth protecting.
|
||||||
|
#
|
||||||
|
# BACK UP ./images/secrets/ ALONGSIDE YOUR DATABASE. The key is the only thing
|
||||||
|
# that can read your stored credentials; a database restored without it needs
|
||||||
|
# every credential re-entered by hand.
|
||||||
|
CURATOR_BOOTSTRAP_NEW_KEY=1
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Optional — defaults are fine
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
# Host port the UI is published on. The container always listens on 8080;
|
||||||
|
# this is only the left-hand side of the port mapping.
|
||||||
|
PORT=8080
|
||||||
|
|
||||||
|
# DEBUG | INFO | WARNING | ERROR
|
||||||
LOG_LEVEL=INFO
|
LOG_LEVEL=INFO
|
||||||
|
|
||||||
# Deployment posture: plain HTTP (no TLS in the app; reverse proxy if needed)
|
# Postgres identity. Change these only if you are pointing at a database you
|
||||||
# See docs/superpowers/specs/2026-05-13-fabledcurator-merge-design.md §2.1
|
# manage yourself — the bundled postgres service is created with whatever is
|
||||||
|
# set here, so changing them after the first start will not rename anything.
|
||||||
|
DB_USER=fabledcurator
|
||||||
|
DB_NAME=fabledcurator
|
||||||
|
|
||||||
|
# Set by docker-compose.yml to reach the bundled services. Override only when
|
||||||
|
# running Postgres or Redis outside this stack.
|
||||||
|
# DB_HOST=postgres
|
||||||
|
# DB_PORT=5432
|
||||||
|
# CELERY_BROKER_URL=redis://redis:6379/0
|
||||||
|
# CELERY_RESULT_BACKEND=redis://redis:6379/0
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# There is no authentication variable here, and that is not an omission
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
#
|
||||||
|
# FabledCurator has no login, no accounts and no permission model. Anything
|
||||||
|
# that can reach PORT is an administrator and can read the platform session
|
||||||
|
# cookies the app stores for Patreon, SubscribeStar and Pixiv.
|
||||||
|
#
|
||||||
|
# Bind it to a trusted network. See "Before you expose it" in README.md and
|
||||||
|
# the deployment posture section of SECURITY.md.
|
||||||
|
#
|
||||||
|
# The Firefox extension's API key is NOT configured here — it is generated
|
||||||
|
# automatically on first use and shown under Settings → Maintenance, where you
|
||||||
|
# can also rotate it.
|
||||||
|
|||||||
+173
-37
@@ -1,42 +1,50 @@
|
|||||||
|
|
||||||
# TEMPORARY — milestone 328 steps 1-2. Delete once the baseline is stamped.
|
# TEMPORARY — milestone 328. Delete once the baseline has shipped and settled.
|
||||||
#
|
#
|
||||||
# Squashing 87 alembic revisions into one baseline has exactly one dangerous
|
# Collapsing 89 alembic revisions into one baseline has exactly one dangerous
|
||||||
# failure: the generated baseline does not reproduce the schema the chain
|
# failure: the baseline does not reproduce the schema the chain produced, and
|
||||||
# produced, `alembic stamp` writes a version string anyway (it validates
|
# the divergence surfaces later, on the operator's live data, in whatever
|
||||||
# NOTHING), and the divergence surfaces on the next real migration against the
|
# migration comes next.
|
||||||
# operator's live data.
|
|
||||||
#
|
#
|
||||||
# So this workflow does the comparison in CI, where a pgvector Postgres already
|
# So the comparison happens in CI, against a throwaway pgvector Postgres, where
|
||||||
# gets built from the chain on every integration run, and nothing is at risk.
|
# nothing is at risk. It answers one question: does `upgrade head` on the
|
||||||
# It answers one question: does `upgrade head` on the collapsed chain produce a
|
# collapsed tree produce the same schema as `upgrade head` on the full chain?
|
||||||
# byte-identical schema to `upgrade head` on the 87-revision chain?
|
|
||||||
#
|
#
|
||||||
# The chain is read from git rather than from the working tree, so this keeps
|
# The chain is read out of GIT, not the working tree, which is what lets this
|
||||||
# working AFTER the old revisions are deleted — `chain_ref` names a commit that
|
# keep working now that the revisions are deleted — `chain_ref` names a commit
|
||||||
# still has them. That is what makes this the proof for step 1 and the
|
# that still carries 0001..0089. That is the whole reason this is a workflow
|
||||||
# pre-flight for step 2, rather than a one-shot script.
|
# rather than a script someone ran once.
|
||||||
#
|
#
|
||||||
# While the chain is still present it also autogenerates a candidate baseline
|
# WHAT THIS CANNOT SEE, and it matters: the comparison is of SCHEMA. Migrations
|
||||||
# from the models and prints it. That is a starting point, NOT the answer:
|
# 0002 and 0003 also INSERTED rows (the import_settings and ml_settings
|
||||||
# autogenerate reads SQLAlchemy metadata, and three things here do not live
|
# singletons), and the application reads those with scalar_one(), which raises
|
||||||
# there —
|
# on an empty result. A baseline that omitted them would produce an identical
|
||||||
# * CREATE EXTENSION vector (0001)
|
# schema, pass this check with a perfect diff, and crash a fresh install on its
|
||||||
# * CREATE EXTENSION tsm_system_rows (0004)
|
# first settings access. Only running the app against a new database finds
|
||||||
# * the HNSW index on image_record.siglip_embedding, which is raw SQL
|
# that class of defect. Do not read a green run here as "the baseline is
|
||||||
# because alembic's create_index cannot express `USING hnsw (...)` (0036)
|
# correct" — read it as "the schema is correct".
|
||||||
# plus any CHECK constraint or server_default that a migration added without
|
#
|
||||||
# the model declaring it. Those must be hand-added, and the diff below is what
|
# Autogenerate now emits nearly all of the baseline unaided, which was NOT true
|
||||||
# proves none were missed.
|
# before #3275 put the previously migration-only objects onto the models — the
|
||||||
|
# HNSW index with its opclass, the COALESCE expression index, the partial
|
||||||
|
# unique indexes, 107 server_defaults, the enum CHECKs. An earlier attempt at
|
||||||
|
# this squash was reverted precisely because the generator dropped them all
|
||||||
|
# silently. What still needs hand-adding is only what cannot live in a model:
|
||||||
|
# the two CREATE EXTENSION statements, the two seed rows, and the pgvector
|
||||||
|
# import the generator forgets to write.
|
||||||
name: Alembic baseline
|
name: Alembic baseline
|
||||||
|
|
||||||
on:
|
on:
|
||||||
workflow_dispatch:
|
workflow_dispatch:
|
||||||
inputs:
|
inputs:
|
||||||
chain_ref:
|
chain_ref:
|
||||||
description: 'Commit/tag that still carries the full 0001..0087 chain'
|
description: 'Commit/tag carrying the full 0001..0089 chain (pinned: the tree no longer has it)'
|
||||||
type: string
|
type: string
|
||||||
default: '0a5bbe8'
|
default: '725bf15'
|
||||||
|
mode:
|
||||||
|
description: 'chain = compare against this tree''s migrations; models = compare against a schema built from the MODELS'
|
||||||
|
type: string
|
||||||
|
default: 'chain'
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
compare:
|
compare:
|
||||||
@@ -79,10 +87,21 @@ jobs:
|
|||||||
test -n "$PG_IP"
|
test -n "$PG_IP"
|
||||||
echo "PG_CONTAINER=$PG" >> "$GITHUB_ENV"
|
echo "PG_CONTAINER=$PG" >> "$GITHUB_ENV"
|
||||||
echo "DB_HOST=$PG_IP" >> "$GITHUB_ENV"
|
echo "DB_HOST=$PG_IP" >> "$GITHUB_ENV"
|
||||||
|
# Socket probe in python, not bash's /dev/tcp — these steps run under
|
||||||
|
# `sh -e`, where that path does not exist. Same fix and same reasoning
|
||||||
|
# as ci.yml's integration job; see the comment there.
|
||||||
|
pg_ready=""
|
||||||
for i in $(seq 1 60); do
|
for i in $(seq 1 60); do
|
||||||
(echo > "/dev/tcp/$PG_IP/5432") >/dev/null 2>&1 && break
|
if python -c "import socket,sys; s=socket.socket(); s.settimeout(2); sys.exit(0 if s.connect_ex(('$PG_IP', 5432)) == 0 else 1)"; then
|
||||||
|
pg_ready=1
|
||||||
|
break
|
||||||
|
fi
|
||||||
sleep 2
|
sleep 2
|
||||||
done
|
done
|
||||||
|
if [ -z "$pg_ready" ]; then
|
||||||
|
echo "postgres at $PG_IP:5432 did not accept a connection within 120s"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
if command -v uv >/dev/null 2>&1; then
|
if command -v uv >/dev/null 2>&1; then
|
||||||
uv pip install --system -r requirements.txt
|
uv pip install --system -r requirements.txt
|
||||||
else
|
else
|
||||||
@@ -96,10 +115,15 @@ jobs:
|
|||||||
- name: Build the schema the OLD chain produces
|
- name: Build the schema the OLD chain produces
|
||||||
env:
|
env:
|
||||||
CHAIN_REF: ${{ github.event.inputs.chain_ref }}
|
CHAIN_REF: ${{ github.event.inputs.chain_ref }}
|
||||||
|
THIS_SHA: ${{ github.sha }}
|
||||||
run: |
|
run: |
|
||||||
set -eux
|
set -eux
|
||||||
docker exec "$PG_CONTAINER" createdb -U fabledcurator fc_chain
|
docker exec "$PG_CONTAINER" createdb -U fabledcurator fc_chain
|
||||||
git worktree add /tmp/chain "$CHAIN_REF"
|
# Blank means "the chain in this ref", which is what you want while
|
||||||
|
# the chain is still intact — comparing the models against a PINNED
|
||||||
|
# older commit reports every migration written since as a difference.
|
||||||
|
# Pin it only after the collapse, when the tree no longer has them.
|
||||||
|
git worktree add /tmp/chain "${CHAIN_REF:-$THIS_SHA}"
|
||||||
ls /tmp/chain/alembic/versions/*.py | wc -l
|
ls /tmp/chain/alembic/versions/*.py | wc -l
|
||||||
cd /tmp/chain
|
cd /tmp/chain
|
||||||
DB_NAME=fc_chain alembic upgrade head
|
DB_NAME=fc_chain alembic upgrade head
|
||||||
@@ -107,6 +131,23 @@ jobs:
|
|||||||
docker exec "$PG_CONTAINER" pg_dump -U fabledcurator --schema-only \
|
docker exec "$PG_CONTAINER" pg_dump -U fabledcurator --schema-only \
|
||||||
--no-owner --no-privileges -d fc_chain > chain.sql
|
--no-owner --no-privileges -d fc_chain > chain.sql
|
||||||
wc -l chain.sql
|
wc -l chain.sql
|
||||||
|
# Emit the dump itself, checksummed, for local analysis. Reconciling
|
||||||
|
# the models against the deployed schema (#3275) needs the ACTUAL
|
||||||
|
# schema, not an inference from a diff — parsing table context out of
|
||||||
|
# unified-diff hunks drops every table whose CREATE TABLE line falls
|
||||||
|
# outside a hunk, which silently under-reports.
|
||||||
|
#
|
||||||
|
# base64 + sha256 for the same reason as the candidate: a plain cat
|
||||||
|
# of a file this size was truncated mid-line by the runner with the
|
||||||
|
# step still green (run 4964).
|
||||||
|
set +x
|
||||||
|
B64=$(base64 -w 120 chain.sql)
|
||||||
|
echo "===== BEGIN CHAIN SCHEMA (base64) ====="
|
||||||
|
echo "$B64"
|
||||||
|
echo "===== END CHAIN SCHEMA ====="
|
||||||
|
echo "chain-sha256: $(sha256sum chain.sql | cut -d' ' -f1)"
|
||||||
|
echo "chain-bytes: $(wc -c < chain.sql)"
|
||||||
|
set -x
|
||||||
|
|
||||||
# A candidate baseline, autogenerated from the models against an EMPTY
|
# A candidate baseline, autogenerated from the models against an EMPTY
|
||||||
# database so every table shows up as a create. Printed for a human to
|
# database so every table shows up as a create. Printed for a human to
|
||||||
@@ -157,20 +198,53 @@ jobs:
|
|||||||
echo "candidate-bytes: $(wc -c < "$F")"
|
echo "candidate-bytes: $(wc -c < "$F")"
|
||||||
echo "candidate-b64-lines: $(echo "$B64" | wc -l)"
|
echo "candidate-b64-lines: $(echo "$B64" | wc -l)"
|
||||||
set -x
|
set -x
|
||||||
|
mkdir -p /tmp/candidate
|
||||||
|
cp alembic/versions/*.py /tmp/candidate/
|
||||||
# Put the tree back exactly as it was; this job never mutates state.
|
# Put the tree back exactly as it was; this job never mutates state.
|
||||||
rm -f alembic/versions/*.py
|
rm -f alembic/versions/*.py
|
||||||
mv /tmp/versions_held/*.py alembic/versions/ 2>/dev/null || true
|
mv /tmp/versions_held/*.py alembic/versions/ 2>/dev/null || true
|
||||||
|
|
||||||
# DB 2: whatever the CURRENT tree's alembic/versions produces. Before the
|
# DB 2: what the CURRENT tree produces.
|
||||||
# squash that is the same 87 revisions and the diff is trivially clean —
|
#
|
||||||
# which is worth running once as a control, so a clean diff after the
|
# `mode: models` applies the candidate autogenerated from the MODELS
|
||||||
# squash means something.
|
# instead, which is what answers "do the models describe the schema?" —
|
||||||
|
# the question #3275 exists because nobody had ever asked it. Under that
|
||||||
|
# mode a clean diff means autogenerate is trustworthy again.
|
||||||
|
#
|
||||||
|
# The two extensions are created by hand first. They are database
|
||||||
|
# objects, not table metadata, so no model can carry them and their
|
||||||
|
# absence is not a model defect — it is simply outside what this
|
||||||
|
# comparison is asking about.
|
||||||
- name: Build the schema the CURRENT tree produces
|
- name: Build the schema the CURRENT tree produces
|
||||||
|
env:
|
||||||
|
MODE: ${{ github.event.inputs.mode }}
|
||||||
run: |
|
run: |
|
||||||
set -eux
|
set -eux
|
||||||
docker exec "$PG_CONTAINER" createdb -U fabledcurator fc_base
|
docker exec "$PG_CONTAINER" createdb -U fabledcurator fc_base
|
||||||
ls alembic/versions/*.py | wc -l
|
if [ "${MODE:-chain}" = "models" ]; then
|
||||||
DB_NAME=fc_base alembic upgrade head
|
docker exec "$PG_CONTAINER" psql -U fabledcurator -d fc_base \
|
||||||
|
-c "CREATE EXTENSION IF NOT EXISTS vector" \
|
||||||
|
-c "CREATE EXTENSION IF NOT EXISTS tsm_system_rows"
|
||||||
|
mkdir -p /tmp/held
|
||||||
|
mv alembic/versions/*.py /tmp/held/
|
||||||
|
cp /tmp/candidate/*.py alembic/versions/
|
||||||
|
# Autogenerate EMITS pgvector.sqlalchemy.vector.VECTOR(...) without
|
||||||
|
# importing it, so the file it writes cannot run:
|
||||||
|
# NameError: name 'pgvector' is not defined
|
||||||
|
# Observed on run 4988, which is the proof rather than the theory.
|
||||||
|
# This is a defect in the GENERATOR, not in the models, so it is
|
||||||
|
# repaired here rather than counted as a schema difference — the
|
||||||
|
# comparison is about whether the models describe the schema.
|
||||||
|
sed -i '0,/^import sqlalchemy as sa$/s//import sqlalchemy as sa\nimport pgvector.sqlalchemy.vector/' alembic/versions/*.py
|
||||||
|
grep -n 'import pgvector' alembic/versions/*.py
|
||||||
|
ls alembic/versions/*.py
|
||||||
|
DB_NAME=fc_base alembic upgrade head
|
||||||
|
rm -f alembic/versions/*.py
|
||||||
|
mv /tmp/held/*.py alembic/versions/
|
||||||
|
else
|
||||||
|
ls alembic/versions/*.py | wc -l
|
||||||
|
DB_NAME=fc_base alembic upgrade head
|
||||||
|
fi
|
||||||
docker exec "$PG_CONTAINER" pg_dump -U fabledcurator --schema-only \
|
docker exec "$PG_CONTAINER" pg_dump -U fabledcurator --schema-only \
|
||||||
--no-owner --no-privileges -d fc_base > baseline.sql
|
--no-owner --no-privileges -d fc_base > baseline.sql
|
||||||
wc -l baseline.sql
|
wc -l baseline.sql
|
||||||
@@ -191,6 +265,25 @@ jobs:
|
|||||||
# these two lines and nothing else. That control is what licenses this
|
# these two lines and nothing else. That control is what licenses this
|
||||||
# filter — it was observed to be the only false positive, rather than
|
# filter — it was observed to be the only false positive, rather than
|
||||||
# assumed to be one.
|
# assumed to be one.
|
||||||
|
# Column ORDER inside a CREATE TABLE is compared separately from column
|
||||||
|
# CONTENT, and only content is fatal.
|
||||||
|
#
|
||||||
|
# A table built by 87 migrations has its columns in ADD COLUMN order; the
|
||||||
|
# same table built in one shot has them in declaration order. That is a
|
||||||
|
# real and permanent difference which no baseline can erase — the
|
||||||
|
# operator's existing database keeps chain order forever, a fresh install
|
||||||
|
# gets model order — so a check that fails on it would never pass and
|
||||||
|
# would teach nothing. FC reaches every column through the ORM by name,
|
||||||
|
# and `SELECT *` ordering is not depended on anywhere.
|
||||||
|
#
|
||||||
|
# So the second pass SORTS the column lines within each CREATE TABLE
|
||||||
|
# rather than DELETING them. That distinction is the whole point: sorting
|
||||||
|
# cannot hide a column that exists on one side only, or one whose type,
|
||||||
|
# nullability or default differs — those still land in the diff. A filter
|
||||||
|
# could have hidden all three.
|
||||||
|
#
|
||||||
|
# Both diffs are reported. The ordered one is informational; the
|
||||||
|
# order-insensitive one is the verdict.
|
||||||
- name: Diff
|
- name: Diff
|
||||||
run: |
|
run: |
|
||||||
set -eu
|
set -eu
|
||||||
@@ -202,11 +295,54 @@ jobs:
|
|||||||
norm chain.sql > a.txt
|
norm chain.sql > a.txt
|
||||||
norm baseline.sql > b.txt
|
norm baseline.sql > b.txt
|
||||||
echo "normalised: chain=$(wc -l < a.txt) lines, current=$(wc -l < b.txt) lines"
|
echo "normalised: chain=$(wc -l < a.txt) lines, current=$(wc -l < b.txt) lines"
|
||||||
|
|
||||||
|
sort_table_columns() {
|
||||||
|
python3 - "$1" <<'PYEOF'
|
||||||
|
import re, sys
|
||||||
|
|
||||||
|
lines = open(sys.argv[1]).read().splitlines()
|
||||||
|
out, block = [], None
|
||||||
|
for line in lines:
|
||||||
|
if block is not None:
|
||||||
|
# ');' on its own closes the CREATE TABLE body.
|
||||||
|
if line.strip() == ");":
|
||||||
|
out.extend(sorted(block))
|
||||||
|
out.append(line)
|
||||||
|
block = None
|
||||||
|
else:
|
||||||
|
# Drop the list comma before sorting. Only the LAST
|
||||||
|
# column lacks one, so keeping it would make every
|
||||||
|
# reordering look like a content change as well — the
|
||||||
|
# comma is punctuation, and carries no schema meaning.
|
||||||
|
block.append(line.rstrip().rstrip(","))
|
||||||
|
continue
|
||||||
|
out.append(line)
|
||||||
|
if re.match(r"CREATE TABLE .*\($", line):
|
||||||
|
block = []
|
||||||
|
if block is not None: # unterminated body: emit it rather than drop it
|
||||||
|
out.extend(block)
|
||||||
|
print("\n".join(out))
|
||||||
|
PYEOF
|
||||||
|
}
|
||||||
|
sort_table_columns a.txt > a.sorted.txt
|
||||||
|
sort_table_columns b.txt > b.sorted.txt
|
||||||
|
test "$(wc -l < a.sorted.txt)" = "$(wc -l < a.txt)"
|
||||||
|
test "$(wc -l < b.sorted.txt)" = "$(wc -l < b.txt)"
|
||||||
|
|
||||||
if diff -u a.txt b.txt > schema.diff; then
|
if diff -u a.txt b.txt > schema.diff; then
|
||||||
echo "SCHEMAS IDENTICAL — the collapsed chain reproduces the old one."
|
echo "ORDERED DIFF: identical, column order included."
|
||||||
else
|
else
|
||||||
echo "SCHEMAS DIFFER — $(grep -cE '^[+-]' schema.diff) changed lines:"
|
echo "ORDERED DIFF: $(grep -cE '^[+-]' schema.diff) changed lines (informational):"
|
||||||
cat schema.diff
|
cat schema.diff
|
||||||
|
fi
|
||||||
|
echo
|
||||||
|
echo "================================================================"
|
||||||
|
echo
|
||||||
|
if diff -u a.sorted.txt b.sorted.txt > sorted.diff; then
|
||||||
|
echo "SCHEMAS MATCH — every difference above is column ORDER alone."
|
||||||
|
else
|
||||||
|
echo "SCHEMAS DIFFER — $(grep -cE '^[+-]' sorted.diff) changed lines that are NOT ordering:"
|
||||||
|
cat sorted.diff
|
||||||
echo
|
echo
|
||||||
echo "The baseline is wrong, not the database. Do not stamp."
|
echo "The baseline is wrong, not the database. Do not stamp."
|
||||||
exit 1
|
exit 1
|
||||||
|
|||||||
+565
-59
@@ -44,6 +44,10 @@ on:
|
|||||||
description: 'Rebuild every image even if the published revision matches'
|
description: 'Rebuild every image even if the published revision matches'
|
||||||
type: boolean
|
type: boolean
|
||||||
default: false
|
default: false
|
||||||
|
refresh:
|
||||||
|
description: 'Behave as the weekly base refresh: build main against fresh bases, publish through the candidate tag'
|
||||||
|
type: boolean
|
||||||
|
default: false
|
||||||
|
|
||||||
# The base-image refresh (milestone 326 step 4, #3154).
|
# The base-image refresh (milestone 326 step 4, #3154).
|
||||||
#
|
#
|
||||||
@@ -72,8 +76,43 @@ on:
|
|||||||
# Deriving it per job invites the two halves to disagree: sign-extension would
|
# Deriving it per job invites the two halves to disagree: sign-extension would
|
||||||
# derive dev's extension version while build-web bundled main's, and the
|
# derive dev's extension version while build-web bundled main's, and the
|
||||||
# release download would 404 on a version that exists perfectly well.
|
# release download would 404 on a version that exists perfectly well.
|
||||||
|
# IS THIS A BASE REFRESH? Asked in five places and previously spelled five
|
||||||
|
# ways — `github.event_name == 'schedule'` in an `if:`, `$GITHUB_EVENT_NAME` in
|
||||||
|
# one shell, an `EVENT:` env passed into another, and a bare expression on
|
||||||
|
# `pull:`. Five spellings of one fact is how half of them come to disagree
|
||||||
|
# after somebody adds a sixth trigger.
|
||||||
|
#
|
||||||
|
# The `refresh` dispatch input is here so this path can be EXERCISED. A weekly
|
||||||
|
# cron is otherwise testable once a week, which is not a cadence anything can
|
||||||
|
# be developed against — the same reason `force_build` exists (#3252, added to
|
||||||
|
# confirm #3190 was gone rather than wait for it to recur). It is also what
|
||||||
|
# makes the milestone-362 gate verifiable at all: a gate has to be watched
|
||||||
|
# rejecting something before anyone can believe it is wired up.
|
||||||
|
#
|
||||||
|
# The input is normalised through `format()` before it is compared, and that
|
||||||
|
# is not defensive styling — the direct comparison is WRONG and fails silently.
|
||||||
|
#
|
||||||
|
# `type: boolean` delivers a real boolean, and GitHub expression semantics cast
|
||||||
|
# operands to numbers when their types differ: `true == 'true'` compares 1
|
||||||
|
# against NaN and is FALSE. Measured on run 5270, whose own log says it —
|
||||||
|
#
|
||||||
|
# expression '(github.event_name == 'schedule'
|
||||||
|
# || github.event.inputs.refresh == 'true') && 'true' || 'false''
|
||||||
|
# evaluated to '%!t(string=false)'
|
||||||
|
# trigger: raw inputs refresh='true'
|
||||||
|
#
|
||||||
|
# — the input arrived as `true` and the expression still said false. The run
|
||||||
|
# then went green with every step skipped, because a refresh that evaluates
|
||||||
|
# false behaves exactly like an ordinary push. That is the whole hazard: the
|
||||||
|
# failure has no symptom.
|
||||||
|
#
|
||||||
|
# `force_build` never hit this because it never compares in an expression. It
|
||||||
|
# passes the raw value into an env var and tests it in the shell, where
|
||||||
|
# everything is already a string. `format('{0}', x)` buys the same thing here,
|
||||||
|
# where a step-level `if:` needs the answer before any shell runs.
|
||||||
env:
|
env:
|
||||||
BUILD_REF: ${{ github.event_name == 'schedule' && 'main' || github.ref }}
|
IS_REFRESH: ${{ (github.event_name == 'schedule' || format('{0}', github.event.inputs.refresh) == 'true') && 'true' || 'false' }}
|
||||||
|
BUILD_REF: ${{ (github.event_name == 'schedule' || format('{0}', github.event.inputs.refresh) == 'true') && 'main' || github.ref }}
|
||||||
|
|
||||||
# Requires repo secret RELEASE_TOKEN — a Forgejo PAT with scopes:
|
# Requires repo secret RELEASE_TOKEN — a Forgejo PAT with scopes:
|
||||||
# - write:package, read:package (for docker push to git.fabledsword.com)
|
# - write:package, read:package (for docker push to git.fabledsword.com)
|
||||||
@@ -143,7 +182,7 @@ jobs:
|
|||||||
# evaluate — this file already gates steps on it — so the guard cannot
|
# evaluate — this file already gates steps on it — so the guard cannot
|
||||||
# be disabled by the same uncertainty it exists to cover.
|
# be disabled by the same uncertainty it exists to cover.
|
||||||
- name: Guard — a scheduled run must have checked out main
|
- name: Guard — a scheduled run must have checked out main
|
||||||
if: github.event_name == 'schedule'
|
if: env.IS_REFRESH == 'true'
|
||||||
run: |
|
run: |
|
||||||
set -eu
|
set -eu
|
||||||
BRANCH=$(git rev-parse --abbrev-ref HEAD)
|
BRANCH=$(git rev-parse --abbrev-ref HEAD)
|
||||||
@@ -413,6 +452,18 @@ jobs:
|
|||||||
# to. Same source of truth; no double-store.
|
# to. Same source of truth; no double-store.
|
||||||
|
|
||||||
build-web:
|
build-web:
|
||||||
|
# Consumed by smoke-web's job-level `if:`. It cannot read `env` — the env
|
||||||
|
# context is available to STEP `if:` and step bodies, never to a job's own
|
||||||
|
# condition, and an unresolvable context there is empty rather than an
|
||||||
|
# error. `smoke-web` skipped silently on run 5290 for exactly that reason.
|
||||||
|
#
|
||||||
|
# Keying off the reuse step's own output is better than re-deriving the
|
||||||
|
# trigger anyway: it is the same single decision the build, the XPI
|
||||||
|
# download and the promote all take (build.yml's "one decision drives
|
||||||
|
# everything downstream"), and it says the thing smoke-web actually needs
|
||||||
|
# to know — a candidate was published — rather than restating why.
|
||||||
|
outputs:
|
||||||
|
candidate: ${{ steps.reuse.outputs.promote }}
|
||||||
# A plain `needs` — no `always()`. That expression existed to let a
|
# A plain `needs` — no `always()`. That expression existed to let a
|
||||||
# SKIPPED sign-extension through on a tag push while still blocking a
|
# SKIPPED sign-extension through on a tag push while still blocking a
|
||||||
# FAILED one. With no tag trigger, sign-extension always runs, so the
|
# FAILED one. With no tag trigger, sign-extension always runs, so the
|
||||||
@@ -437,7 +488,7 @@ jobs:
|
|||||||
|
|
||||||
# See sign-extension's copy for why this guard exists.
|
# See sign-extension's copy for why this guard exists.
|
||||||
- name: Guard — a scheduled run must have checked out main
|
- name: Guard — a scheduled run must have checked out main
|
||||||
if: github.event_name == 'schedule'
|
if: env.IS_REFRESH == 'true'
|
||||||
run: |
|
run: |
|
||||||
set -eu
|
set -eu
|
||||||
BRANCH=$(git rev-parse --abbrev-ref HEAD)
|
BRANCH=$(git rev-parse --abbrev-ref HEAD)
|
||||||
@@ -470,8 +521,18 @@ jobs:
|
|||||||
# the PREVIOUS XPI while the freshly signed one is orphaned (#3156).
|
# the PREVIOUS XPI while the freshly signed one is orphaned (#3156).
|
||||||
# * dev and main derive the same values for the same source.
|
# * dev and main derive the same values for the same source.
|
||||||
- name: Report the derived artifact version
|
- name: Report the derived artifact version
|
||||||
|
env:
|
||||||
|
# Diagnostic for the trigger normalisation. `refresh` is reported RAW
|
||||||
|
# as well as normalised, because the two disagreeing is the whole
|
||||||
|
# failure mode: a dispatch input whose type does not compare the way
|
||||||
|
# the expression assumes evaluates to false silently, and the only
|
||||||
|
# symptom is a refresh that quietly behaves like an ordinary push.
|
||||||
|
RAW_REFRESH: ${{ github.event.inputs.refresh }}
|
||||||
|
RAW_FORCE: ${{ github.event.inputs.force_build }}
|
||||||
run: |
|
run: |
|
||||||
set -u
|
set -u
|
||||||
|
echo "trigger: event=$GITHUB_EVENT_NAME IS_REFRESH='${IS_REFRESH:-<unset>}' BUILD_REF='${BUILD_REF:-<unset>}'"
|
||||||
|
echo "trigger: raw inputs refresh='${RAW_REFRESH:-<unset>}' force_build='${RAW_FORCE:-<unset>}'"
|
||||||
A=web
|
A=web
|
||||||
V=$(sh scripts/artifacts.sh version "$A" 2>&1 || echo UNAVAILABLE)
|
V=$(sh scripts/artifacts.sh version "$A" 2>&1 || echo UNAVAILABLE)
|
||||||
R=$(sh scripts/artifacts.sh revision "$A" 2>&1 || echo UNAVAILABLE)
|
R=$(sh scripts/artifacts.sh revision "$A" 2>&1 || echo UNAVAILABLE)
|
||||||
@@ -528,7 +589,7 @@ jobs:
|
|||||||
# Checked BEFORE the ref test, not after: a scheduled run's
|
# Checked BEFORE the ref test, not after: a scheduled run's
|
||||||
# GITHUB_REF is the default branch (dev), so the main test would
|
# GITHUB_REF is the default branch (dev), so the main test would
|
||||||
# never fire on it.
|
# never fire on it.
|
||||||
if [ "${GITHUB_EVENT_NAME:-}" = "schedule" ]; then
|
if [ "${IS_REFRESH:-}" = "true" ]; then
|
||||||
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator:latest" >> "$GITHUB_OUTPUT"
|
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator:latest" >> "$GITHUB_OUTPUT"
|
||||||
echo "channel=main" >> "$GITHUB_OUTPUT"
|
echo "channel=main" >> "$GITHUB_OUTPUT"
|
||||||
elif [ "${GITHUB_REF##*/}" = "main" ]; then
|
elif [ "${GITHUB_REF##*/}" = "main" ]; then
|
||||||
@@ -628,7 +689,6 @@ jobs:
|
|||||||
# A scheduled refresh has to bypass reuse by construction: it
|
# A scheduled refresh has to bypass reuse by construction: it
|
||||||
# rebuilds the SAME source, so fc.revision always matches and the
|
# rebuilds the SAME source, so fc.revision always matches and the
|
||||||
# check would skip every refresh there has ever been.
|
# check would skip every refresh there has ever been.
|
||||||
EVENT: ${{ github.event_name }}
|
|
||||||
run: |
|
run: |
|
||||||
set -eu
|
set -eu
|
||||||
DERIVED=$(sh scripts/artifacts.sh revision web)
|
DERIVED=$(sh scripts/artifacts.sh revision web)
|
||||||
@@ -638,11 +698,60 @@ jobs:
|
|||||||
# adds no variability the reuse check would have to account for.
|
# adds no variability the reuse check would have to account for.
|
||||||
echo "version=$(sh scripts/artifacts.sh version web)" >> "$GITHUB_OUTPUT"
|
echo "version=$(sh scripts/artifacts.sh version web)" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
|
# The build clock, pinned to the same commit (#3265). Without it
|
||||||
|
# buildkit stamps the image config with the wall clock of the build,
|
||||||
|
# so identical layers republish under a new config blob and the
|
||||||
|
# channel tag gets a new manifest digest for no reason. Derived from
|
||||||
|
# `newest()` like revision and version, so all three name one commit
|
||||||
|
# and cannot drift apart.
|
||||||
|
echo "epoch=$(sh scripts/artifacts.sh epoch web)" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
# The moving tag for this channel. Which tag we ask IS the channel —
|
# The moving tag for this channel. Which tag we ask IS the channel —
|
||||||
# that is why the revision needs no -main/-dev qualifier any more.
|
# that is why the revision needs no -main/-dev qualifier any more.
|
||||||
if [ "$CHANNEL" = "main" ]; then T=latest; else T=dev; fi
|
if [ "$CHANNEL" = "main" ]; then T=latest; else T=dev; fi
|
||||||
echo "channel_ref=$IMAGE:$T" >> "$GITHUB_OUTPUT"
|
echo "channel_ref=$IMAGE:$T" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
|
# WHERE THE BUILD PUBLISHES, which is not always the channel — and
|
||||||
|
# whether the channel then has to be written separately.
|
||||||
|
#
|
||||||
|
# On a push the build writes the channel tag directly: the bytes came
|
||||||
|
# from a commit, and a commit is the thing CI tests. Nothing to hold
|
||||||
|
# it behind.
|
||||||
|
#
|
||||||
|
# On the scheduled refresh it writes a CANDIDATE tag instead. A
|
||||||
|
# refresh rebuilds against freshly resolved base images, and the web
|
||||||
|
# image's runtime is a line of UNPINNED Debian packages (ffmpeg,
|
||||||
|
# libjpeg62-turbo, libpq5, megatools…) re-resolved on every build.
|
||||||
|
# Nothing in ci.yml can see that: its lanes run on ci-python:3.14 and
|
||||||
|
# install requirements.txt, and a base bump changes neither. So
|
||||||
|
# refreshed bytes have to be proven before :latest names them, and
|
||||||
|
# proving needs a moment between "built" and "published" to occupy.
|
||||||
|
# This is that moment; :latest goes on naming the build that works
|
||||||
|
# until something says otherwise.
|
||||||
|
#
|
||||||
|
# `:refresh-candidate` is one moving ref per image, overwritten in
|
||||||
|
# place, holding a build nobody is told to pull — the shape rule 145
|
||||||
|
# already allows for :buildcache, not the per-build tag family that
|
||||||
|
# milestone 318 withdrew.
|
||||||
|
#
|
||||||
|
# Decided HERE, beside `hit`, for the reason the force/schedule
|
||||||
|
# branch below gives: one step decides what this job does. A
|
||||||
|
# condition derived independently could disagree with the tag the
|
||||||
|
# build actually wrote.
|
||||||
|
#
|
||||||
|
# build-web additionally exposes this as `outputs.candidate`, which is
|
||||||
|
# what gates the `promote` job — a job's `if:` cannot read `env`, and
|
||||||
|
# one flag is enough because all three derive it from the same
|
||||||
|
# IS_REFRESH. ml and agent do not re-emit it; a second copy nothing
|
||||||
|
# reads is the kind of thing that later reads as load-bearing.
|
||||||
|
if [ "${IS_REFRESH:-}" = "true" ]; then
|
||||||
|
echo "build_ref=$IMAGE:refresh-candidate" >> "$GITHUB_OUTPUT"
|
||||||
|
echo "promote=true" >> "$GITHUB_OUTPUT"
|
||||||
|
else
|
||||||
|
echo "build_ref=$IMAGE:$T" >> "$GITHUB_OUTPUT"
|
||||||
|
echo "promote=false" >> "$GITHUB_OUTPUT"
|
||||||
|
fi
|
||||||
|
|
||||||
# Compare VALUES, never exit codes. Measured on buildx v0.36.1
|
# Compare VALUES, never exit codes. Measured on buildx v0.36.1
|
||||||
# (run 4732): a missing key returns an empty string and exits 0, so
|
# (run 4732): a missing key returns an empty string and exits 0, so
|
||||||
# branching on the exit code would read "no label yet" as success.
|
# branching on the exit code would read "no label yet" as success.
|
||||||
@@ -675,7 +784,7 @@ jobs:
|
|||||||
if [ "${FORCE:-false}" = "true" ]; then
|
if [ "${FORCE:-false}" = "true" ]; then
|
||||||
echo "hit=false" >> "$GITHUB_OUTPUT"
|
echo "hit=false" >> "$GITHUB_OUTPUT"
|
||||||
echo "reuse: force_build set — building regardless"
|
echo "reuse: force_build set — building regardless"
|
||||||
elif [ "${EVENT:-}" = "schedule" ]; then
|
elif [ "${IS_REFRESH:-}" = "true" ]; then
|
||||||
echo "hit=false" >> "$GITHUB_OUTPUT"
|
echo "hit=false" >> "$GITHUB_OUTPUT"
|
||||||
echo "reuse: scheduled base refresh — building regardless"
|
echo "reuse: scheduled base refresh — building regardless"
|
||||||
elif [ -n "$PUBLISHED" ] && [ "$PUBLISHED" = "$DERIVED" ]; then
|
elif [ -n "$PUBLISHED" ] && [ "$PUBLISHED" = "$DERIVED" ]; then
|
||||||
@@ -764,6 +873,12 @@ jobs:
|
|||||||
|
|
||||||
- name: Build and push web image
|
- name: Build and push web image
|
||||||
if: steps.reuse.outputs.hit != 'true'
|
if: steps.reuse.outputs.hit != 'true'
|
||||||
|
# Read by buildx out of the ENVIRONMENT, not passed as a build-arg —
|
||||||
|
# it normalises the image config's `created` field and the history
|
||||||
|
# timestamps rather than being consumed by the Dockerfile. See #3265
|
||||||
|
# and the reuse step's `epoch` output.
|
||||||
|
env:
|
||||||
|
SOURCE_DATE_EPOCH: ${{ steps.reuse.outputs.epoch }}
|
||||||
uses: docker/build-push-action@v5
|
uses: docker/build-push-action@v5
|
||||||
with:
|
with:
|
||||||
context: .
|
context: .
|
||||||
@@ -776,20 +891,17 @@ jobs:
|
|||||||
# invalidates, and the image genuinely rebuilds.
|
# invalidates, and the image genuinely rebuilds.
|
||||||
#
|
#
|
||||||
# MEASURED on the first real fire, run 4934 (#3265): when the base
|
# MEASURED on the first real fire, run 4934 (#3265): when the base
|
||||||
# did NOT move, the build is ~13s and every content step reports
|
# did NOT move, the build was ~13s with every content step CACHED —
|
||||||
# CACHED — but the channel tag STILL gets a new manifest digest.
|
# and the channel tag STILL got a new manifest digest, because
|
||||||
# buildkit mints a fresh image config each run, so identical layers
|
# buildkit stamps a fresh image config per run and republishes the
|
||||||
# are republished under a new config blob. All three images moved
|
# identical layers under it. All three images moved that way on
|
||||||
# that way on 2026-08-30 with nothing whatsoever changed in them.
|
# 2026-08-30 with nothing whatsoever changed in them.
|
||||||
#
|
#
|
||||||
# So a refresh currently rewrites :latest every Sunday whether or
|
# SOURCE_DATE_EPOCH (below) is the fix: pinned to the commit the
|
||||||
# not there is anything new in it, and :c-<sha> is handed a new
|
# content came from, the config is byte-identical across runs, so
|
||||||
# manifest to diverge from on the same cadence. Layers are shared,
|
# the manifest digest is too and the push is a registry no-op. A
|
||||||
# so the storage cost is a config blob; the cost that matters is
|
# digest change means the content changed again, which is the only
|
||||||
# that a digest change no longer MEANS anything. Tracked in #3265 —
|
# thing a digest is any use for.
|
||||||
# the likely fix is a deterministic SOURCE_DATE_EPOCH, which would
|
|
||||||
# make "same source, same bytes" true and turn the no-op case into
|
|
||||||
# a genuine no-op.
|
|
||||||
#
|
#
|
||||||
# What `pull` does NOT catch either: a Debian package update inside
|
# What `pull` does NOT catch either: a Debian package update inside
|
||||||
# the `apt-get install` layer while the base tag itself stands
|
# the `apt-get install` layer while the base tag itself stands
|
||||||
@@ -799,14 +911,14 @@ jobs:
|
|||||||
# churn #3265 is about.
|
# churn #3265 is about.
|
||||||
#
|
#
|
||||||
# Only on the schedule. An ordinary push wants the cached base.
|
# Only on the schedule. An ordinary push wants the cached base.
|
||||||
pull: ${{ github.event_name == 'schedule' }}
|
pull: ${{ env.IS_REFRESH == 'true' }}
|
||||||
# ONE tag, the channel's. Every other tag is written by the step
|
# ONE tag, the channel's. Every other tag is written by the step
|
||||||
# below, registry-side. buildx here pushes the first tag to the
|
# below, registry-side. buildx here pushes the first tag to the
|
||||||
# registry and then re-pushes the rest through the DOCKER driver,
|
# registry and then re-pushes the rest through the DOCKER driver,
|
||||||
# out of a local image store a registry-direct build never filled —
|
# out of a local image store a registry-direct build never filled —
|
||||||
# #3190, which cost `main` its :c-<sha> on 2026-08-29 while :latest
|
# #3190, which cost `main` its :c-<sha> on 2026-08-29 while :latest
|
||||||
# published perfectly well.
|
# published perfectly well.
|
||||||
tags: ${{ steps.reuse.outputs.channel_ref }}
|
tags: ${{ steps.reuse.outputs.build_ref }}
|
||||||
# The reuse key. Read back off the channel tag on the next push to
|
# The reuse key. Read back off the channel tag on the next push to
|
||||||
# decide whether that push needs to build at all, so this is not
|
# decide whether that push needs to build at all, so this is not
|
||||||
# decoration — an unstamped image is one that will always rebuild.
|
# decoration — an unstamped image is one that will always rebuild.
|
||||||
@@ -937,6 +1049,282 @@ jobs:
|
|||||||
docker buildx imagetools create $ARGS "$SOURCE"
|
docker buildx imagetools create $ARGS "$SOURCE"
|
||||||
echo "repointed from $SOURCE:$ARGS"
|
echo "repointed from $SOURCE:$ARGS"
|
||||||
|
|
||||||
|
# Does the image a refresh just built still work?
|
||||||
|
#
|
||||||
|
# This is the gate the base refresh never had. `ci.yml` cannot be it: its
|
||||||
|
# lanes run on ci-python:3.14 and install requirements.txt, and a base bump
|
||||||
|
# changes neither — all five stay green through a refresh that breaks the
|
||||||
|
# product. What a refresh re-resolves is the Dockerfile's apt layer (ffmpeg,
|
||||||
|
# unar, libpq5, postgresql-client, zstd, megatools, libjpeg62-turbo,
|
||||||
|
# libwebp7, libpng16-16), unpinned, every build.
|
||||||
|
#
|
||||||
|
# So this runs the CANDIDATE IMAGE, against real Postgres and Redis. Not the
|
||||||
|
# source tree, and not a static inspection: `ffmpeg -version` exiting 0 would
|
||||||
|
# pass while a codec removal broke every thumbnail in the library.
|
||||||
|
#
|
||||||
|
# Refresh-only. On a push the bytes came from a commit, and a commit is what
|
||||||
|
# ci.yml already tests.
|
||||||
|
#
|
||||||
|
# Reports a verdict; it does not yet gate the promote (milestone 362 step 4).
|
||||||
|
# Landing the gate and the thing it gates in one change would mean the first
|
||||||
|
# time anyone saw this job run would also be the first time it could stop a
|
||||||
|
# publish.
|
||||||
|
smoke-web:
|
||||||
|
needs: [build-web]
|
||||||
|
if: needs.build-web.outputs.candidate == 'true'
|
||||||
|
runs-on: python-ci
|
||||||
|
container:
|
||||||
|
image: git.fabledsword.com/bvandeusen/ci-python:3.14
|
||||||
|
env:
|
||||||
|
DB_USER: fabledcurator
|
||||||
|
DB_PASSWORD: ci_smoke
|
||||||
|
DB_PORT: "5432"
|
||||||
|
DB_NAME: fabledcurator_smoke
|
||||||
|
SECRET_KEY: ci_smoke_placeholder
|
||||||
|
IMAGE: git.fabledsword.com/bvandeusen/fabledcurator
|
||||||
|
services:
|
||||||
|
postgres:
|
||||||
|
image: pgvector/pgvector:pg16
|
||||||
|
env:
|
||||||
|
POSTGRES_USER: fabledcurator
|
||||||
|
POSTGRES_PASSWORD: ci_smoke
|
||||||
|
POSTGRES_DB: fabledcurator_smoke
|
||||||
|
options: >-
|
||||||
|
--health-cmd "pg_isready -U fabledcurator"
|
||||||
|
--health-interval 10s
|
||||||
|
--health-timeout 5s
|
||||||
|
--health-retries 10
|
||||||
|
redis:
|
||||||
|
image: redis:7-alpine
|
||||||
|
options: >-
|
||||||
|
--health-cmd "redis-cli ping"
|
||||||
|
--health-interval 10s
|
||||||
|
--health-timeout 5s
|
||||||
|
--health-retries 10
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
with:
|
||||||
|
# The same ref the image was built from, so the smoke script matches
|
||||||
|
# the code inside the candidate.
|
||||||
|
ref: ${{ env.BUILD_REF }}
|
||||||
|
|
||||||
|
- name: Smoke the candidate image
|
||||||
|
env:
|
||||||
|
TOKEN: ${{ secrets.RELEASE_TOKEN }}
|
||||||
|
ACTOR: ${{ github.actor }}
|
||||||
|
run: |
|
||||||
|
set -eux
|
||||||
|
# Service discovery mirrors ci.yml's integration lane: these jobs run
|
||||||
|
# in a container against a mounted docker socket, so the services are
|
||||||
|
# SIBLINGS reachable by IP, not by hostname.
|
||||||
|
PG=$(docker ps --filter "name=smoke" --filter "ancestor=pgvector/pgvector:pg16" -q | head -n1)
|
||||||
|
RD=$(docker ps --filter "name=smoke" --filter "ancestor=redis:7-alpine" -q | head -n1)
|
||||||
|
test -n "$PG" && test -n "$RD"
|
||||||
|
PG_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$PG")
|
||||||
|
RD_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$RD")
|
||||||
|
test -n "$PG_IP" && test -n "$RD_IP"
|
||||||
|
|
||||||
|
# Socket probe in python, not bash's /dev/tcp — these steps run under
|
||||||
|
# `sh -e`, where that path does not exist. Same fix and reasoning as
|
||||||
|
# ci.yml's integration job; see the comment there.
|
||||||
|
pg_ready=""
|
||||||
|
for i in $(seq 1 60); do
|
||||||
|
if python -c "import socket,sys; s=socket.socket(); s.settimeout(2); sys.exit(0 if s.connect_ex(('$PG_IP', 5432)) == 0 else 1)"; then
|
||||||
|
pg_ready=1
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
sleep 2
|
||||||
|
done
|
||||||
|
if [ -z "$pg_ready" ]; then
|
||||||
|
echo "postgres at $PG_IP:5432 did not accept a connection within 120s"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "$TOKEN" | docker login git.fabledsword.com -u "$ACTOR" --password-stdin
|
||||||
|
CANDIDATE="$IMAGE:refresh-candidate"
|
||||||
|
docker pull "$CANDIDATE"
|
||||||
|
|
||||||
|
ENVOPTS="-e DB_USER=$DB_USER -e DB_PASSWORD=$DB_PASSWORD -e DB_HOST=$PG_IP"
|
||||||
|
ENVOPTS="$ENVOPTS -e DB_PORT=5432 -e DB_NAME=$DB_NAME -e SECRET_KEY=$SECRET_KEY"
|
||||||
|
ENVOPTS="$ENVOPTS -e CELERY_BROKER_URL=redis://$RD_IP:6379/0"
|
||||||
|
ENVOPTS="$ENVOPTS -e CELERY_RESULT_BACKEND=redis://$RD_IP:6379/0"
|
||||||
|
# A throwaway CI instance IS first-time setup, which is the one case
|
||||||
|
# credential_crypto allows a key to be minted in. Without it the web
|
||||||
|
# role refuses to boot — deliberately, since silently generating a
|
||||||
|
# key on a restored-DB-but-lost-secrets deployment would leave every
|
||||||
|
# Credential row undecryptable (the 2026-06-02 audit). Discovered by
|
||||||
|
# this job on its first real run; see #3422 for the fact that no
|
||||||
|
# user-facing file mentions this variable at all.
|
||||||
|
ENVOPTS="$ENVOPTS -e CURATOR_BOOTSTRAP_NEW_KEY=1"
|
||||||
|
|
||||||
|
# 1. The schema builds from empty, using the image's OWN libpq and
|
||||||
|
# psycopg. This is the same call entrypoint.sh makes before it
|
||||||
|
# serves anything, so a failure here is a failure to boot.
|
||||||
|
echo "smoke: alembic upgrade head"
|
||||||
|
docker run --rm $ENVOPTS "$CANDIDATE" alembic upgrade head
|
||||||
|
|
||||||
|
# 2. The apt layer's binaries and the app's own thumbnail path, run
|
||||||
|
# inside the image. Piped over stdin rather than bind-mounted: the
|
||||||
|
# workspace is a docker VOLUME belonging to this job's container,
|
||||||
|
# so a host bind of $PWD would not resolve for a sibling.
|
||||||
|
echo "smoke: image-internal checks"
|
||||||
|
docker run --rm -i $ENVOPTS "$CANDIDATE" shell -c 'python3 -' < scripts/smoke_image.py
|
||||||
|
|
||||||
|
# 3. It actually serves. `docker run -d` then poll the container's own
|
||||||
|
# IP — no port publishing, because the job container reaches
|
||||||
|
# siblings directly and a published port would collide with
|
||||||
|
# whatever else the runner is hosting.
|
||||||
|
echo "smoke: web boots and answers /api/health"
|
||||||
|
CID=$(docker run -d $ENVOPTS "$CANDIDATE" web)
|
||||||
|
# Clean up the container however this ends, and dump its log ONLY
|
||||||
|
# on failure — a boot that never answers must fail with the reason
|
||||||
|
# visible rather than as a bare timeout (rule 156), while a green run
|
||||||
|
# has nothing to say. `exit $rc` preserves the real status, which a
|
||||||
|
# trap that ends on a successful `docker rm` would otherwise mask.
|
||||||
|
trap 'rc=$?; [ $rc -eq 0 ] || docker logs "$CID" 2>&1 | tail -40; docker rm -f "$CID" >/dev/null 2>&1 || true; exit $rc' EXIT
|
||||||
|
WEB_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$CID")
|
||||||
|
test -n "$WEB_IP"
|
||||||
|
healthy=""
|
||||||
|
for i in $(seq 1 60); do
|
||||||
|
if curl -fsS --max-time 5 "http://$WEB_IP:8080/api/health" >/dev/null 2>&1; then
|
||||||
|
healthy=1
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
# A container that has EXITED will never answer, so stop asking.
|
||||||
|
# Without this the loop spent 3m35s polling a dead container on
|
||||||
|
# this job's first run, and — because docker recycles the IP — got
|
||||||
|
# a confusing mix of connection-refused and 5s timeouts from
|
||||||
|
# whatever took the address next. The trap's log dump had the real
|
||||||
|
# answer the whole time; this just stops burying it.
|
||||||
|
if [ "$(docker inspect -f '{{.State.Running}}' "$CID" 2>/dev/null)" != "true" ]; then
|
||||||
|
echo "smoke: FAILED — the web container exited during boot." >&2
|
||||||
|
echo "smoke: its log follows; entrypoint runs alembic BEFORE" >&2
|
||||||
|
echo "smoke: serving, so a startup exception lands here." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
sleep 2
|
||||||
|
done
|
||||||
|
if [ -z "$healthy" ]; then
|
||||||
|
# 60 iterations of (up to 5s connect + 2s sleep) — up to ~7min, not
|
||||||
|
# the 120s an earlier version of this message claimed.
|
||||||
|
echo "smoke: FAILED — web is running but never answered" >&2
|
||||||
|
echo "smoke: /api/health. It is up, so look at hypercorn and the" >&2
|
||||||
|
echo "smoke: python base rather than at startup." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
curl -fsS --max-time 5 "http://$WEB_IP:8080/api/health"
|
||||||
|
echo
|
||||||
|
|
||||||
|
echo "smoke: all checks passed against $CANDIDATE"
|
||||||
|
|
||||||
|
# Move the channel tags — the whole point of the gate.
|
||||||
|
#
|
||||||
|
# Lives in its own job because the verdict it depends on cannot exist until
|
||||||
|
# after build-web has finished, and the promote used to run INSIDE build-web.
|
||||||
|
#
|
||||||
|
# `needs` on smoke-web is the gate. A failed smoke skips this job, so a
|
||||||
|
# refresh that broke something leaves :latest naming the build that works —
|
||||||
|
# "the refresh failed" and "production is broken" must not be the same event.
|
||||||
|
# A SKIPPED smoke also skips this job, which is the behaviour that matters
|
||||||
|
# most: on run 5290 the gate silently skipped itself, and a design where only
|
||||||
|
# a FAILED gate blocks would have published unverified images while reporting
|
||||||
|
# success. Not running is not the same as passing.
|
||||||
|
#
|
||||||
|
# All three images promote TOGETHER, or none do. They are one stack: build.yml
|
||||||
|
# already refuses to publish a :dev web image beside a stale :dev ml, because
|
||||||
|
# the mismatch only shows up as a runtime failure. A refresh that published ml
|
||||||
|
# and withheld web would be that same trap, arrived at through the gate.
|
||||||
|
#
|
||||||
|
# The gate covers the web image only (milestone 362 step 3 scoped it there),
|
||||||
|
# so ml and agent are being held to web's verdict rather than their own. That
|
||||||
|
# is deliberate and it is the conservative direction — they ship together, so
|
||||||
|
# the weakest evidence should govern all three — but it is not the same as
|
||||||
|
# having smoked them, and it should not be read as if it were.
|
||||||
|
promote:
|
||||||
|
needs: [build-web, build-ml, build-agent, smoke-web]
|
||||||
|
# Only a refresh publishes through a candidate; a push writes its channel
|
||||||
|
# tag directly from the build. Reads the same reuse-step decision the build
|
||||||
|
# took, via a job output — a job's `if:` cannot see the `env` context.
|
||||||
|
if: needs.build-web.outputs.candidate == 'true'
|
||||||
|
runs-on: python-ci
|
||||||
|
container:
|
||||||
|
image: git.fabledsword.com/bvandeusen/ci-python:3.14
|
||||||
|
steps:
|
||||||
|
- name: Point the channel tags at the smoked candidates
|
||||||
|
env:
|
||||||
|
TOKEN: ${{ secrets.RELEASE_TOKEN }}
|
||||||
|
ACTOR: ${{ github.actor }}
|
||||||
|
run: |
|
||||||
|
set -eu
|
||||||
|
# `latest` is not a guess: a refresh always builds `main` (BUILD_REF),
|
||||||
|
# and the "must have checked out main" guard in every build job fails
|
||||||
|
# the run if that did not hold. So the channel is main's.
|
||||||
|
TAG=latest
|
||||||
|
FAILED=""
|
||||||
|
|
||||||
|
for NAME in fabledcurator fabledcurator-ml fabledcurator-agent; do
|
||||||
|
REPO="bvandeusen/$NAME"
|
||||||
|
echo "promote: $REPO"
|
||||||
|
|
||||||
|
# Registry auth is its own token exchange — `docker login`
|
||||||
|
# authenticates the docker client, not curl. Deadline on every call
|
||||||
|
# (rule 156): a registry that stops answering must fail this step,
|
||||||
|
# not hang the weekly refresh until the job times out.
|
||||||
|
BEARER=$(curl -fsS --max-time 30 -u "$ACTOR:$TOKEN" \
|
||||||
|
"https://git.fabledsword.com/v2/token?scope=repository:$REPO:pull,push&service=git.fabledsword.com" \
|
||||||
|
| python3 -c 'import sys,json; print(json.load(sys.stdin)["token"])')
|
||||||
|
|
||||||
|
# Ask for the IMAGE manifest media types only. Offering the index
|
||||||
|
# types too would let the registry hand back an index if one ever
|
||||||
|
# existed at this tag, and we would faithfully copy the thing this
|
||||||
|
# whole approach exists to avoid creating.
|
||||||
|
ACCEPT='application/vnd.oci.image.manifest.v1+json, application/vnd.docker.distribution.manifest.v2+json'
|
||||||
|
CT=$(curl -fsS --max-time 60 -o manifest.json -D headers.txt \
|
||||||
|
-H "Authorization: Bearer $BEARER" -H "Accept: $ACCEPT" \
|
||||||
|
"https://git.fabledsword.com/v2/$REPO/manifests/refresh-candidate" \
|
||||||
|
&& tr -d '\r' < headers.txt | awk -F': ' '/^[Cc]ontent-[Tt]ype:/{print $2}')
|
||||||
|
test -n "$CT"
|
||||||
|
SRC=$(tr -d '\r' < headers.txt | awk -F': ' '/^[Dd]ocker-[Cc]ontent-[Dd]igest:/{print $2}')
|
||||||
|
echo "promote: candidate $SRC ($CT)"
|
||||||
|
|
||||||
|
# NOT `imagetools create`. That wraps its source in an INDEX, and
|
||||||
|
# `.Image.Config.Labels` does not resolve through one — the
|
||||||
|
# fc.revision the reuse check reads off the channel tag would come
|
||||||
|
# back empty, every later push would miss and rebuild, and nothing
|
||||||
|
# would go red (#3183, run 4751). A manifest PUT is what "make this
|
||||||
|
# tag name that image" means at the registry: same bytes, same media
|
||||||
|
# type, same digest, no layer transfer.
|
||||||
|
curl -fsS --max-time 120 -X PUT \
|
||||||
|
-H "Authorization: Bearer $BEARER" -H "Content-Type: $CT" \
|
||||||
|
--data-binary @manifest.json \
|
||||||
|
"https://git.fabledsword.com/v2/$REPO/manifests/$TAG"
|
||||||
|
|
||||||
|
# Read it back. A PUT that returned 2xx but landed something else is
|
||||||
|
# exactly the silent-and-plausible failure this pipeline keeps
|
||||||
|
# producing, and the check costs one request.
|
||||||
|
NOW=$(curl -fsS --max-time 30 -o /dev/null -D - \
|
||||||
|
-H "Authorization: Bearer $BEARER" -H "Accept: $ACCEPT" \
|
||||||
|
"https://git.fabledsword.com/v2/$REPO/manifests/$TAG" \
|
||||||
|
| tr -d '\r' | awk -F': ' '/^[Dd]ocker-[Cc]ontent-[Dd]igest:/{print $2}')
|
||||||
|
if [ "$NOW" != "$SRC" ]; then
|
||||||
|
echo "promote: FAILED — $NAME:$TAG is $NOW, expected $SRC" >&2
|
||||||
|
FAILED="$FAILED $NAME"
|
||||||
|
continue
|
||||||
|
fi
|
||||||
|
echo "promote: $NAME:$TAG now names $NOW"
|
||||||
|
done
|
||||||
|
|
||||||
|
if [ -n "$FAILED" ]; then
|
||||||
|
echo "" >&2
|
||||||
|
echo "promote: FAILED for:$FAILED" >&2
|
||||||
|
echo "promote: the channel tags are now INCONSISTENT — some images" >&2
|
||||||
|
echo "promote: moved and some did not. Re-run this refresh; the" >&2
|
||||||
|
echo "promote: candidates are still published and the promote is" >&2
|
||||||
|
echo "promote: idempotent." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
echo "promote: all three channel tags moved"
|
||||||
|
|
||||||
build-ml:
|
build-ml:
|
||||||
runs-on: python-ci
|
runs-on: python-ci
|
||||||
container:
|
container:
|
||||||
@@ -957,7 +1345,7 @@ jobs:
|
|||||||
|
|
||||||
# See sign-extension's copy for why this guard exists.
|
# See sign-extension's copy for why this guard exists.
|
||||||
- name: Guard — a scheduled run must have checked out main
|
- name: Guard — a scheduled run must have checked out main
|
||||||
if: github.event_name == 'schedule'
|
if: env.IS_REFRESH == 'true'
|
||||||
run: |
|
run: |
|
||||||
set -eu
|
set -eu
|
||||||
BRANCH=$(git rev-parse --abbrev-ref HEAD)
|
BRANCH=$(git rev-parse --abbrev-ref HEAD)
|
||||||
@@ -990,8 +1378,18 @@ jobs:
|
|||||||
# the PREVIOUS XPI while the freshly signed one is orphaned (#3156).
|
# the PREVIOUS XPI while the freshly signed one is orphaned (#3156).
|
||||||
# * dev and main derive the same values for the same source.
|
# * dev and main derive the same values for the same source.
|
||||||
- name: Report the derived artifact version
|
- name: Report the derived artifact version
|
||||||
|
env:
|
||||||
|
# Diagnostic for the trigger normalisation. `refresh` is reported RAW
|
||||||
|
# as well as normalised, because the two disagreeing is the whole
|
||||||
|
# failure mode: a dispatch input whose type does not compare the way
|
||||||
|
# the expression assumes evaluates to false silently, and the only
|
||||||
|
# symptom is a refresh that quietly behaves like an ordinary push.
|
||||||
|
RAW_REFRESH: ${{ github.event.inputs.refresh }}
|
||||||
|
RAW_FORCE: ${{ github.event.inputs.force_build }}
|
||||||
run: |
|
run: |
|
||||||
set -u
|
set -u
|
||||||
|
echo "trigger: event=$GITHUB_EVENT_NAME IS_REFRESH='${IS_REFRESH:-<unset>}' BUILD_REF='${BUILD_REF:-<unset>}'"
|
||||||
|
echo "trigger: raw inputs refresh='${RAW_REFRESH:-<unset>}' force_build='${RAW_FORCE:-<unset>}'"
|
||||||
A=ml
|
A=ml
|
||||||
V=$(sh scripts/artifacts.sh version "$A" 2>&1 || echo UNAVAILABLE)
|
V=$(sh scripts/artifacts.sh version "$A" 2>&1 || echo UNAVAILABLE)
|
||||||
R=$(sh scripts/artifacts.sh revision "$A" 2>&1 || echo UNAVAILABLE)
|
R=$(sh scripts/artifacts.sh revision "$A" 2>&1 || echo UNAVAILABLE)
|
||||||
@@ -1008,7 +1406,7 @@ jobs:
|
|||||||
SHORT_SHA=$(printf '%s' "$GITHUB_SHA" | cut -c1-7)
|
SHORT_SHA=$(printf '%s' "$GITHUB_SHA" | cut -c1-7)
|
||||||
# Mirrors build-web's tag list and its schedule handling; see
|
# Mirrors build-web's tag list and its schedule handling; see
|
||||||
# the comments there.
|
# the comments there.
|
||||||
if [ "${GITHUB_EVENT_NAME:-}" = "schedule" ]; then
|
if [ "${IS_REFRESH:-}" = "true" ]; then
|
||||||
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-ml:latest" >> "$GITHUB_OUTPUT"
|
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-ml:latest" >> "$GITHUB_OUTPUT"
|
||||||
echo "channel=main" >> "$GITHUB_OUTPUT"
|
echo "channel=main" >> "$GITHUB_OUTPUT"
|
||||||
elif [ "${GITHUB_REF##*/}" = "main" ]; then
|
elif [ "${GITHUB_REF##*/}" = "main" ]; then
|
||||||
@@ -1091,17 +1489,63 @@ jobs:
|
|||||||
# A scheduled refresh has to bypass reuse by construction: it
|
# A scheduled refresh has to bypass reuse by construction: it
|
||||||
# rebuilds the SAME source, so fc.revision always matches and the
|
# rebuilds the SAME source, so fc.revision always matches and the
|
||||||
# check would skip every refresh there has ever been.
|
# check would skip every refresh there has ever been.
|
||||||
EVENT: ${{ github.event_name }}
|
|
||||||
run: |
|
run: |
|
||||||
set -eu
|
set -eu
|
||||||
DERIVED=$(sh scripts/artifacts.sh revision ml)
|
DERIVED=$(sh scripts/artifacts.sh revision ml)
|
||||||
echo "revision=$DERIVED" >> "$GITHUB_OUTPUT"
|
echo "revision=$DERIVED" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
|
# The build clock, pinned to the same commit (#3265). Without it
|
||||||
|
# buildkit stamps the image config with the wall clock of the build,
|
||||||
|
# so identical layers republish under a new config blob and the
|
||||||
|
# channel tag gets a new manifest digest for no reason. Derived from
|
||||||
|
# `newest()` like revision and version, so all three name one commit
|
||||||
|
# and cannot drift apart.
|
||||||
|
echo "epoch=$(sh scripts/artifacts.sh epoch ml)" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
# The moving tag for this channel. Which tag we ask IS the channel —
|
# The moving tag for this channel. Which tag we ask IS the channel —
|
||||||
# that is why the revision needs no -main/-dev qualifier any more.
|
# that is why the revision needs no -main/-dev qualifier any more.
|
||||||
if [ "$CHANNEL" = "main" ]; then T=latest; else T=dev; fi
|
if [ "$CHANNEL" = "main" ]; then T=latest; else T=dev; fi
|
||||||
echo "channel_ref=$IMAGE:$T" >> "$GITHUB_OUTPUT"
|
echo "channel_ref=$IMAGE:$T" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
|
# WHERE THE BUILD PUBLISHES, which is not always the channel — and
|
||||||
|
# whether the channel then has to be written separately.
|
||||||
|
#
|
||||||
|
# On a push the build writes the channel tag directly: the bytes came
|
||||||
|
# from a commit, and a commit is the thing CI tests. Nothing to hold
|
||||||
|
# it behind.
|
||||||
|
#
|
||||||
|
# On the scheduled refresh it writes a CANDIDATE tag instead. A
|
||||||
|
# refresh rebuilds against freshly resolved base images, and the web
|
||||||
|
# image's runtime is a line of UNPINNED Debian packages (ffmpeg,
|
||||||
|
# libjpeg62-turbo, libpq5, megatools…) re-resolved on every build.
|
||||||
|
# Nothing in ci.yml can see that: its lanes run on ci-python:3.14 and
|
||||||
|
# install requirements.txt, and a base bump changes neither. So
|
||||||
|
# refreshed bytes have to be proven before :latest names them, and
|
||||||
|
# proving needs a moment between "built" and "published" to occupy.
|
||||||
|
# This is that moment; :latest goes on naming the build that works
|
||||||
|
# until something says otherwise.
|
||||||
|
#
|
||||||
|
# `:refresh-candidate` is one moving ref per image, overwritten in
|
||||||
|
# place, holding a build nobody is told to pull — the shape rule 145
|
||||||
|
# already allows for :buildcache, not the per-build tag family that
|
||||||
|
# milestone 318 withdrew.
|
||||||
|
#
|
||||||
|
# Decided HERE, beside `hit`, for the reason the force/schedule
|
||||||
|
# branch below gives: one step decides what this job does. A
|
||||||
|
# condition derived independently could disagree with the tag the
|
||||||
|
# build actually wrote.
|
||||||
|
#
|
||||||
|
# build-web additionally exposes this as `outputs.candidate`, which is
|
||||||
|
# what gates the `promote` job — a job's `if:` cannot read `env`, and
|
||||||
|
# one flag is enough because all three derive it from the same
|
||||||
|
# IS_REFRESH. ml and agent do not re-emit it; a second copy nothing
|
||||||
|
# reads is the kind of thing that later reads as load-bearing.
|
||||||
|
if [ "${IS_REFRESH:-}" = "true" ]; then
|
||||||
|
echo "build_ref=$IMAGE:refresh-candidate" >> "$GITHUB_OUTPUT"
|
||||||
|
else
|
||||||
|
echo "build_ref=$IMAGE:$T" >> "$GITHUB_OUTPUT"
|
||||||
|
fi
|
||||||
|
|
||||||
# Compare VALUES, never exit codes. Measured on buildx v0.36.1
|
# Compare VALUES, never exit codes. Measured on buildx v0.36.1
|
||||||
# (run 4732): a missing key returns an empty string and exits 0, so
|
# (run 4732): a missing key returns an empty string and exits 0, so
|
||||||
# branching on the exit code would read "no label yet" as success.
|
# branching on the exit code would read "no label yet" as success.
|
||||||
@@ -1134,7 +1578,7 @@ jobs:
|
|||||||
if [ "${FORCE:-false}" = "true" ]; then
|
if [ "${FORCE:-false}" = "true" ]; then
|
||||||
echo "hit=false" >> "$GITHUB_OUTPUT"
|
echo "hit=false" >> "$GITHUB_OUTPUT"
|
||||||
echo "reuse: force_build set — building regardless"
|
echo "reuse: force_build set — building regardless"
|
||||||
elif [ "${EVENT:-}" = "schedule" ]; then
|
elif [ "${IS_REFRESH:-}" = "true" ]; then
|
||||||
echo "hit=false" >> "$GITHUB_OUTPUT"
|
echo "hit=false" >> "$GITHUB_OUTPUT"
|
||||||
echo "reuse: scheduled base refresh — building regardless"
|
echo "reuse: scheduled base refresh — building regardless"
|
||||||
elif [ -n "$PUBLISHED" ] && [ "$PUBLISHED" = "$DERIVED" ]; then
|
elif [ -n "$PUBLISHED" ] && [ "$PUBLISHED" = "$DERIVED" ]; then
|
||||||
@@ -1147,6 +1591,12 @@ jobs:
|
|||||||
|
|
||||||
- name: Build and push ml image
|
- name: Build and push ml image
|
||||||
if: steps.reuse.outputs.hit != 'true'
|
if: steps.reuse.outputs.hit != 'true'
|
||||||
|
# Read by buildx out of the ENVIRONMENT, not passed as a build-arg —
|
||||||
|
# it normalises the image config's `created` field and the history
|
||||||
|
# timestamps rather than being consumed by the Dockerfile. See #3265
|
||||||
|
# and the reuse step's `epoch` output.
|
||||||
|
env:
|
||||||
|
SOURCE_DATE_EPOCH: ${{ steps.reuse.outputs.epoch }}
|
||||||
uses: docker/build-push-action@v5
|
uses: docker/build-push-action@v5
|
||||||
with:
|
with:
|
||||||
context: .
|
context: .
|
||||||
@@ -1159,20 +1609,17 @@ jobs:
|
|||||||
# invalidates, and the image genuinely rebuilds.
|
# invalidates, and the image genuinely rebuilds.
|
||||||
#
|
#
|
||||||
# MEASURED on the first real fire, run 4934 (#3265): when the base
|
# MEASURED on the first real fire, run 4934 (#3265): when the base
|
||||||
# did NOT move, the build is ~13s and every content step reports
|
# did NOT move, the build was ~13s with every content step CACHED —
|
||||||
# CACHED — but the channel tag STILL gets a new manifest digest.
|
# and the channel tag STILL got a new manifest digest, because
|
||||||
# buildkit mints a fresh image config each run, so identical layers
|
# buildkit stamps a fresh image config per run and republishes the
|
||||||
# are republished under a new config blob. All three images moved
|
# identical layers under it. All three images moved that way on
|
||||||
# that way on 2026-08-30 with nothing whatsoever changed in them.
|
# 2026-08-30 with nothing whatsoever changed in them.
|
||||||
#
|
#
|
||||||
# So a refresh currently rewrites :latest every Sunday whether or
|
# SOURCE_DATE_EPOCH (below) is the fix: pinned to the commit the
|
||||||
# not there is anything new in it, and :c-<sha> is handed a new
|
# content came from, the config is byte-identical across runs, so
|
||||||
# manifest to diverge from on the same cadence. Layers are shared,
|
# the manifest digest is too and the push is a registry no-op. A
|
||||||
# so the storage cost is a config blob; the cost that matters is
|
# digest change means the content changed again, which is the only
|
||||||
# that a digest change no longer MEANS anything. Tracked in #3265 —
|
# thing a digest is any use for.
|
||||||
# the likely fix is a deterministic SOURCE_DATE_EPOCH, which would
|
|
||||||
# make "same source, same bytes" true and turn the no-op case into
|
|
||||||
# a genuine no-op.
|
|
||||||
#
|
#
|
||||||
# What `pull` does NOT catch either: a Debian package update inside
|
# What `pull` does NOT catch either: a Debian package update inside
|
||||||
# the `apt-get install` layer while the base tag itself stands
|
# the `apt-get install` layer while the base tag itself stands
|
||||||
@@ -1182,14 +1629,14 @@ jobs:
|
|||||||
# churn #3265 is about.
|
# churn #3265 is about.
|
||||||
#
|
#
|
||||||
# Only on the schedule. An ordinary push wants the cached base.
|
# Only on the schedule. An ordinary push wants the cached base.
|
||||||
pull: ${{ github.event_name == 'schedule' }}
|
pull: ${{ env.IS_REFRESH == 'true' }}
|
||||||
# ONE tag, the channel's. Every other tag is written by the step
|
# ONE tag, the channel's. Every other tag is written by the step
|
||||||
# below, registry-side. buildx here pushes the first tag to the
|
# below, registry-side. buildx here pushes the first tag to the
|
||||||
# registry and then re-pushes the rest through the DOCKER driver,
|
# registry and then re-pushes the rest through the DOCKER driver,
|
||||||
# out of a local image store a registry-direct build never filled —
|
# out of a local image store a registry-direct build never filled —
|
||||||
# #3190, which cost `main` its :c-<sha> on 2026-08-29 while :latest
|
# #3190, which cost `main` its :c-<sha> on 2026-08-29 while :latest
|
||||||
# published perfectly well.
|
# published perfectly well.
|
||||||
tags: ${{ steps.reuse.outputs.channel_ref }}
|
tags: ${{ steps.reuse.outputs.build_ref }}
|
||||||
# The reuse key. Read back off the channel tag on the next push to
|
# The reuse key. Read back off the channel tag on the next push to
|
||||||
# decide whether that push needs to build at all, so this is not
|
# decide whether that push needs to build at all, so this is not
|
||||||
# decoration — an unstamped image is one that will always rebuild.
|
# decoration — an unstamped image is one that will always rebuild.
|
||||||
@@ -1331,7 +1778,7 @@ jobs:
|
|||||||
|
|
||||||
# See sign-extension's copy for why this guard exists.
|
# See sign-extension's copy for why this guard exists.
|
||||||
- name: Guard — a scheduled run must have checked out main
|
- name: Guard — a scheduled run must have checked out main
|
||||||
if: github.event_name == 'schedule'
|
if: env.IS_REFRESH == 'true'
|
||||||
run: |
|
run: |
|
||||||
set -eu
|
set -eu
|
||||||
BRANCH=$(git rev-parse --abbrev-ref HEAD)
|
BRANCH=$(git rev-parse --abbrev-ref HEAD)
|
||||||
@@ -1364,8 +1811,18 @@ jobs:
|
|||||||
# the PREVIOUS XPI while the freshly signed one is orphaned (#3156).
|
# the PREVIOUS XPI while the freshly signed one is orphaned (#3156).
|
||||||
# * dev and main derive the same values for the same source.
|
# * dev and main derive the same values for the same source.
|
||||||
- name: Report the derived artifact version
|
- name: Report the derived artifact version
|
||||||
|
env:
|
||||||
|
# Diagnostic for the trigger normalisation. `refresh` is reported RAW
|
||||||
|
# as well as normalised, because the two disagreeing is the whole
|
||||||
|
# failure mode: a dispatch input whose type does not compare the way
|
||||||
|
# the expression assumes evaluates to false silently, and the only
|
||||||
|
# symptom is a refresh that quietly behaves like an ordinary push.
|
||||||
|
RAW_REFRESH: ${{ github.event.inputs.refresh }}
|
||||||
|
RAW_FORCE: ${{ github.event.inputs.force_build }}
|
||||||
run: |
|
run: |
|
||||||
set -u
|
set -u
|
||||||
|
echo "trigger: event=$GITHUB_EVENT_NAME IS_REFRESH='${IS_REFRESH:-<unset>}' BUILD_REF='${BUILD_REF:-<unset>}'"
|
||||||
|
echo "trigger: raw inputs refresh='${RAW_REFRESH:-<unset>}' force_build='${RAW_FORCE:-<unset>}'"
|
||||||
A=agent
|
A=agent
|
||||||
V=$(sh scripts/artifacts.sh version "$A" 2>&1 || echo UNAVAILABLE)
|
V=$(sh scripts/artifacts.sh version "$A" 2>&1 || echo UNAVAILABLE)
|
||||||
R=$(sh scripts/artifacts.sh revision "$A" 2>&1 || echo UNAVAILABLE)
|
R=$(sh scripts/artifacts.sh revision "$A" 2>&1 || echo UNAVAILABLE)
|
||||||
@@ -1377,7 +1834,7 @@ jobs:
|
|||||||
SHORT_SHA=$(printf '%s' "$GITHUB_SHA" | cut -c1-7)
|
SHORT_SHA=$(printf '%s' "$GITHUB_SHA" | cut -c1-7)
|
||||||
# Mirrors build-web's tag list and its schedule handling; see
|
# Mirrors build-web's tag list and its schedule handling; see
|
||||||
# the comments there.
|
# the comments there.
|
||||||
if [ "${GITHUB_EVENT_NAME:-}" = "schedule" ]; then
|
if [ "${IS_REFRESH:-}" = "true" ]; then
|
||||||
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-agent:latest" >> "$GITHUB_OUTPUT"
|
echo "tags=git.fabledsword.com/bvandeusen/fabledcurator-agent:latest" >> "$GITHUB_OUTPUT"
|
||||||
echo "channel=main" >> "$GITHUB_OUTPUT"
|
echo "channel=main" >> "$GITHUB_OUTPUT"
|
||||||
elif [ "${GITHUB_REF##*/}" = "main" ]; then
|
elif [ "${GITHUB_REF##*/}" = "main" ]; then
|
||||||
@@ -1460,17 +1917,63 @@ jobs:
|
|||||||
# A scheduled refresh has to bypass reuse by construction: it
|
# A scheduled refresh has to bypass reuse by construction: it
|
||||||
# rebuilds the SAME source, so fc.revision always matches and the
|
# rebuilds the SAME source, so fc.revision always matches and the
|
||||||
# check would skip every refresh there has ever been.
|
# check would skip every refresh there has ever been.
|
||||||
EVENT: ${{ github.event_name }}
|
|
||||||
run: |
|
run: |
|
||||||
set -eu
|
set -eu
|
||||||
DERIVED=$(sh scripts/artifacts.sh revision agent)
|
DERIVED=$(sh scripts/artifacts.sh revision agent)
|
||||||
echo "revision=$DERIVED" >> "$GITHUB_OUTPUT"
|
echo "revision=$DERIVED" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
|
# The build clock, pinned to the same commit (#3265). Without it
|
||||||
|
# buildkit stamps the image config with the wall clock of the build,
|
||||||
|
# so identical layers republish under a new config blob and the
|
||||||
|
# channel tag gets a new manifest digest for no reason. Derived from
|
||||||
|
# `newest()` like revision and version, so all three name one commit
|
||||||
|
# and cannot drift apart.
|
||||||
|
echo "epoch=$(sh scripts/artifacts.sh epoch agent)" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
# The moving tag for this channel. Which tag we ask IS the channel —
|
# The moving tag for this channel. Which tag we ask IS the channel —
|
||||||
# that is why the revision needs no -main/-dev qualifier any more.
|
# that is why the revision needs no -main/-dev qualifier any more.
|
||||||
if [ "$CHANNEL" = "main" ]; then T=latest; else T=dev; fi
|
if [ "$CHANNEL" = "main" ]; then T=latest; else T=dev; fi
|
||||||
echo "channel_ref=$IMAGE:$T" >> "$GITHUB_OUTPUT"
|
echo "channel_ref=$IMAGE:$T" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
|
# WHERE THE BUILD PUBLISHES, which is not always the channel — and
|
||||||
|
# whether the channel then has to be written separately.
|
||||||
|
#
|
||||||
|
# On a push the build writes the channel tag directly: the bytes came
|
||||||
|
# from a commit, and a commit is the thing CI tests. Nothing to hold
|
||||||
|
# it behind.
|
||||||
|
#
|
||||||
|
# On the scheduled refresh it writes a CANDIDATE tag instead. A
|
||||||
|
# refresh rebuilds against freshly resolved base images, and the web
|
||||||
|
# image's runtime is a line of UNPINNED Debian packages (ffmpeg,
|
||||||
|
# libjpeg62-turbo, libpq5, megatools…) re-resolved on every build.
|
||||||
|
# Nothing in ci.yml can see that: its lanes run on ci-python:3.14 and
|
||||||
|
# install requirements.txt, and a base bump changes neither. So
|
||||||
|
# refreshed bytes have to be proven before :latest names them, and
|
||||||
|
# proving needs a moment between "built" and "published" to occupy.
|
||||||
|
# This is that moment; :latest goes on naming the build that works
|
||||||
|
# until something says otherwise.
|
||||||
|
#
|
||||||
|
# `:refresh-candidate` is one moving ref per image, overwritten in
|
||||||
|
# place, holding a build nobody is told to pull — the shape rule 145
|
||||||
|
# already allows for :buildcache, not the per-build tag family that
|
||||||
|
# milestone 318 withdrew.
|
||||||
|
#
|
||||||
|
# Decided HERE, beside `hit`, for the reason the force/schedule
|
||||||
|
# branch below gives: one step decides what this job does. A
|
||||||
|
# condition derived independently could disagree with the tag the
|
||||||
|
# build actually wrote.
|
||||||
|
#
|
||||||
|
# build-web additionally exposes this as `outputs.candidate`, which is
|
||||||
|
# what gates the `promote` job — a job's `if:` cannot read `env`, and
|
||||||
|
# one flag is enough because all three derive it from the same
|
||||||
|
# IS_REFRESH. ml and agent do not re-emit it; a second copy nothing
|
||||||
|
# reads is the kind of thing that later reads as load-bearing.
|
||||||
|
if [ "${IS_REFRESH:-}" = "true" ]; then
|
||||||
|
echo "build_ref=$IMAGE:refresh-candidate" >> "$GITHUB_OUTPUT"
|
||||||
|
else
|
||||||
|
echo "build_ref=$IMAGE:$T" >> "$GITHUB_OUTPUT"
|
||||||
|
fi
|
||||||
|
|
||||||
# Compare VALUES, never exit codes. Measured on buildx v0.36.1
|
# Compare VALUES, never exit codes. Measured on buildx v0.36.1
|
||||||
# (run 4732): a missing key returns an empty string and exits 0, so
|
# (run 4732): a missing key returns an empty string and exits 0, so
|
||||||
# branching on the exit code would read "no label yet" as success.
|
# branching on the exit code would read "no label yet" as success.
|
||||||
@@ -1503,7 +2006,7 @@ jobs:
|
|||||||
if [ "${FORCE:-false}" = "true" ]; then
|
if [ "${FORCE:-false}" = "true" ]; then
|
||||||
echo "hit=false" >> "$GITHUB_OUTPUT"
|
echo "hit=false" >> "$GITHUB_OUTPUT"
|
||||||
echo "reuse: force_build set — building regardless"
|
echo "reuse: force_build set — building regardless"
|
||||||
elif [ "${EVENT:-}" = "schedule" ]; then
|
elif [ "${IS_REFRESH:-}" = "true" ]; then
|
||||||
echo "hit=false" >> "$GITHUB_OUTPUT"
|
echo "hit=false" >> "$GITHUB_OUTPUT"
|
||||||
echo "reuse: scheduled base refresh — building regardless"
|
echo "reuse: scheduled base refresh — building regardless"
|
||||||
elif [ -n "$PUBLISHED" ] && [ "$PUBLISHED" = "$DERIVED" ]; then
|
elif [ -n "$PUBLISHED" ] && [ "$PUBLISHED" = "$DERIVED" ]; then
|
||||||
@@ -1516,6 +2019,12 @@ jobs:
|
|||||||
|
|
||||||
- name: Build and push agent image
|
- name: Build and push agent image
|
||||||
if: steps.reuse.outputs.hit != 'true'
|
if: steps.reuse.outputs.hit != 'true'
|
||||||
|
# Read by buildx out of the ENVIRONMENT, not passed as a build-arg —
|
||||||
|
# it normalises the image config's `created` field and the history
|
||||||
|
# timestamps rather than being consumed by the Dockerfile. See #3265
|
||||||
|
# and the reuse step's `epoch` output.
|
||||||
|
env:
|
||||||
|
SOURCE_DATE_EPOCH: ${{ steps.reuse.outputs.epoch }}
|
||||||
uses: docker/build-push-action@v5
|
uses: docker/build-push-action@v5
|
||||||
with:
|
with:
|
||||||
context: agent
|
context: agent
|
||||||
@@ -1528,20 +2037,17 @@ jobs:
|
|||||||
# invalidates, and the image genuinely rebuilds.
|
# invalidates, and the image genuinely rebuilds.
|
||||||
#
|
#
|
||||||
# MEASURED on the first real fire, run 4934 (#3265): when the base
|
# MEASURED on the first real fire, run 4934 (#3265): when the base
|
||||||
# did NOT move, the build is ~13s and every content step reports
|
# did NOT move, the build was ~13s with every content step CACHED —
|
||||||
# CACHED — but the channel tag STILL gets a new manifest digest.
|
# and the channel tag STILL got a new manifest digest, because
|
||||||
# buildkit mints a fresh image config each run, so identical layers
|
# buildkit stamps a fresh image config per run and republishes the
|
||||||
# are republished under a new config blob. All three images moved
|
# identical layers under it. All three images moved that way on
|
||||||
# that way on 2026-08-30 with nothing whatsoever changed in them.
|
# 2026-08-30 with nothing whatsoever changed in them.
|
||||||
#
|
#
|
||||||
# So a refresh currently rewrites :latest every Sunday whether or
|
# SOURCE_DATE_EPOCH (below) is the fix: pinned to the commit the
|
||||||
# not there is anything new in it, and :c-<sha> is handed a new
|
# content came from, the config is byte-identical across runs, so
|
||||||
# manifest to diverge from on the same cadence. Layers are shared,
|
# the manifest digest is too and the push is a registry no-op. A
|
||||||
# so the storage cost is a config blob; the cost that matters is
|
# digest change means the content changed again, which is the only
|
||||||
# that a digest change no longer MEANS anything. Tracked in #3265 —
|
# thing a digest is any use for.
|
||||||
# the likely fix is a deterministic SOURCE_DATE_EPOCH, which would
|
|
||||||
# make "same source, same bytes" true and turn the no-op case into
|
|
||||||
# a genuine no-op.
|
|
||||||
#
|
#
|
||||||
# What `pull` does NOT catch either: a Debian package update inside
|
# What `pull` does NOT catch either: a Debian package update inside
|
||||||
# the `apt-get install` layer while the base tag itself stands
|
# the `apt-get install` layer while the base tag itself stands
|
||||||
@@ -1551,14 +2057,14 @@ jobs:
|
|||||||
# churn #3265 is about.
|
# churn #3265 is about.
|
||||||
#
|
#
|
||||||
# Only on the schedule. An ordinary push wants the cached base.
|
# Only on the schedule. An ordinary push wants the cached base.
|
||||||
pull: ${{ github.event_name == 'schedule' }}
|
pull: ${{ env.IS_REFRESH == 'true' }}
|
||||||
# ONE tag, the channel's. Every other tag is written by the step
|
# ONE tag, the channel's. Every other tag is written by the step
|
||||||
# below, registry-side. buildx here pushes the first tag to the
|
# below, registry-side. buildx here pushes the first tag to the
|
||||||
# registry and then re-pushes the rest through the DOCKER driver,
|
# registry and then re-pushes the rest through the DOCKER driver,
|
||||||
# out of a local image store a registry-direct build never filled —
|
# out of a local image store a registry-direct build never filled —
|
||||||
# #3190, which cost `main` its :c-<sha> on 2026-08-29 while :latest
|
# #3190, which cost `main` its :c-<sha> on 2026-08-29 while :latest
|
||||||
# published perfectly well.
|
# published perfectly well.
|
||||||
tags: ${{ steps.reuse.outputs.channel_ref }}
|
tags: ${{ steps.reuse.outputs.build_ref }}
|
||||||
# The reuse key. Read back off the channel tag on the next push to
|
# The reuse key. Read back off the channel tag on the next push to
|
||||||
# decide whether that push needs to build at all, so this is not
|
# decide whether that push needs to build at all, so this is not
|
||||||
# decoration — an unstamped image is one that will always rebuild.
|
# decoration — an unstamped image is one that will always rebuild.
|
||||||
|
|||||||
@@ -255,10 +255,27 @@ jobs:
|
|||||||
export DB_HOST="$PG_IP"
|
export DB_HOST="$PG_IP"
|
||||||
export CELERY_BROKER_URL="redis://$RD_IP:6379/0"
|
export CELERY_BROKER_URL="redis://$RD_IP:6379/0"
|
||||||
export CELERY_RESULT_BACKEND="redis://$RD_IP:6379/0"
|
export CELERY_RESULT_BACKEND="redis://$RD_IP:6379/0"
|
||||||
|
# These steps run under `sh -e`, not bash, so bash's /dev/tcp magic
|
||||||
|
# path does not exist here — the probe this loop used to run could
|
||||||
|
# never succeed and simply burned the full 120s on every run, green
|
||||||
|
# or red, then continued without having established anything. Python
|
||||||
|
# is in the image and needs no installed package for a socket
|
||||||
|
# connect, so it is the probe. Exhausting the budget is now a named
|
||||||
|
# failure rather than a silent fall-through (rule 156): if Postgres
|
||||||
|
# is genuinely not up, that is what the log should say, instead of
|
||||||
|
# whatever the first query happens to raise two minutes later.
|
||||||
|
pg_ready=""
|
||||||
for i in $(seq 1 60); do
|
for i in $(seq 1 60); do
|
||||||
(echo > "/dev/tcp/$PG_IP/5432") >/dev/null 2>&1 && break
|
if python -c "import socket,sys; s=socket.socket(); s.settimeout(2); sys.exit(0 if s.connect_ex(('$PG_IP', 5432)) == 0 else 1)"; then
|
||||||
|
pg_ready=1
|
||||||
|
break
|
||||||
|
fi
|
||||||
sleep 2
|
sleep 2
|
||||||
done
|
done
|
||||||
|
if [ -z "$pg_ready" ]; then
|
||||||
|
echo "postgres at $PG_IP:5432 did not accept a connection within 120s"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
if command -v uv >/dev/null 2>&1; then
|
if command -v uv >/dev/null 2>&1; then
|
||||||
uv pip install --system -r requirements.txt pytest pytest-asyncio
|
uv pip install --system -r requirements.txt pytest pytest-asyncio
|
||||||
else
|
else
|
||||||
|
|||||||
@@ -0,0 +1,77 @@
|
|||||||
|
# Contributing
|
||||||
|
|
||||||
|
FabledCurator is developed by a single maintainer for their own use, and
|
||||||
|
published because it may be useful to others. That shapes what contribution
|
||||||
|
looks like here.
|
||||||
|
|
||||||
|
**Issues are welcome** — bug reports, and questions about running it, are
|
||||||
|
genuinely useful and often the fastest way to find out that something is
|
||||||
|
broken outside the one environment it was built in.
|
||||||
|
|
||||||
|
**Open an issue before writing a pull request.** Not as a formality: the
|
||||||
|
project has opinions that are not obvious from the code, and it is unpleasant
|
||||||
|
for everyone when a finished patch turns out to conflict with one. A short
|
||||||
|
issue first costs you nothing and may save you an evening.
|
||||||
|
|
||||||
|
**Contributions are licensed under the AGPL-3.0**, like the rest of the
|
||||||
|
project. By submitting one you agree it ships under that licence. There is no
|
||||||
|
CLA and no copyright assignment.
|
||||||
|
|
||||||
|
## Running it for development
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker compose up -d # UI on http://localhost:8080
|
||||||
|
```
|
||||||
|
|
||||||
|
The dev override (`docker-compose.override.yml`) is auto-merged and builds the
|
||||||
|
app images locally from source, so this needs no `.env` and no registry
|
||||||
|
access. Postgres and Redis ports are exposed on the host.
|
||||||
|
|
||||||
|
## What CI checks
|
||||||
|
|
||||||
|
Every push runs these, and they are the definition of done for a change:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ruff check backend/ tests/ alembic/ agent/ scripts/ # lint (and import order)
|
||||||
|
pytest tests/ -m "not integration" # backend unit tests
|
||||||
|
pytest tests/ -m integration # needs pgvector + redis
|
||||||
|
cd frontend && npm run test:unit && npm run build # frontend
|
||||||
|
```
|
||||||
|
|
||||||
|
The integration lane builds its schema by running the real migrations
|
||||||
|
(`alembic upgrade head`), never from ORM metadata — so a migration that does
|
||||||
|
not apply cleanly fails CI rather than being discovered later.
|
||||||
|
|
||||||
|
Note for the linter: ruff's isort runs with `order-by-type`, which sorts
|
||||||
|
ALL-CAPS names ahead of CamelCase. `from sqlalchemy import JSON, DateTime, ...`
|
||||||
|
is correct; putting `JSON` alphabetically between `Integer` and `String` is
|
||||||
|
not. This catches people out.
|
||||||
|
|
||||||
|
## Database changes
|
||||||
|
|
||||||
|
The ORM models and the migration chain must agree. This is enforced, and it is
|
||||||
|
enforced because they silently diverged for a long time and nobody noticed
|
||||||
|
until they were compared: the models were missing indexes, defaults and
|
||||||
|
uniqueness guarantees that only ever existed inside a migration, which made
|
||||||
|
`alembic revision --autogenerate` actively unsafe to run.
|
||||||
|
|
||||||
|
So: if you change a model, write the migration; if you write a migration,
|
||||||
|
change the model to match. Both, in the same commit.
|
||||||
|
|
||||||
|
Adding a value to a CHECK-constrained column means swapping the constraint in
|
||||||
|
the same change — the constraint is not documentation, and a new value without
|
||||||
|
it fails at insert time.
|
||||||
|
|
||||||
|
## Branch model
|
||||||
|
|
||||||
|
`dev` is where work happens. `main` is production and is only reached by a
|
||||||
|
merge from `dev`, never pushed to directly. If you are sending a pull request,
|
||||||
|
target `dev`.
|
||||||
|
|
||||||
|
## Style
|
||||||
|
|
||||||
|
Match the surrounding code. The one convention worth stating explicitly is
|
||||||
|
that comments here explain *why*, especially where a choice looks wrong at a
|
||||||
|
glance — a comment recording which migration a constraint came from, or why a
|
||||||
|
default is a `text()` rather than a string, is the kind that has repeatedly
|
||||||
|
turned out to be worth its space.
|
||||||
@@ -0,0 +1,661 @@
|
|||||||
|
GNU AFFERO GENERAL PUBLIC LICENSE
|
||||||
|
Version 3, 19 November 2007
|
||||||
|
|
||||||
|
Copyright (C) 2007 Free Software Foundation, Inc. <https://fsf.org/>
|
||||||
|
Everyone is permitted to copy and distribute verbatim copies
|
||||||
|
of this license document, but changing it is not allowed.
|
||||||
|
|
||||||
|
Preamble
|
||||||
|
|
||||||
|
The GNU Affero General Public License is a free, copyleft license for
|
||||||
|
software and other kinds of works, specifically designed to ensure
|
||||||
|
cooperation with the community in the case of network server software.
|
||||||
|
|
||||||
|
The licenses for most software and other practical works are designed
|
||||||
|
to take away your freedom to share and change the works. By contrast,
|
||||||
|
our General Public Licenses are intended to guarantee your freedom to
|
||||||
|
share and change all versions of a program--to make sure it remains free
|
||||||
|
software for all its users.
|
||||||
|
|
||||||
|
When we speak of free software, we are referring to freedom, not
|
||||||
|
price. Our General Public Licenses are designed to make sure that you
|
||||||
|
have the freedom to distribute copies of free software (and charge for
|
||||||
|
them if you wish), that you receive source code or can get it if you
|
||||||
|
want it, that you can change the software or use pieces of it in new
|
||||||
|
free programs, and that you know you can do these things.
|
||||||
|
|
||||||
|
Developers that use our General Public Licenses protect your rights
|
||||||
|
with two steps: (1) assert copyright on the software, and (2) offer
|
||||||
|
you this License which gives you legal permission to copy, distribute
|
||||||
|
and/or modify the software.
|
||||||
|
|
||||||
|
A secondary benefit of defending all users' freedom is that
|
||||||
|
improvements made in alternate versions of the program, if they
|
||||||
|
receive widespread use, become available for other developers to
|
||||||
|
incorporate. Many developers of free software are heartened and
|
||||||
|
encouraged by the resulting cooperation. However, in the case of
|
||||||
|
software used on network servers, this result may fail to come about.
|
||||||
|
The GNU General Public License permits making a modified version and
|
||||||
|
letting the public access it on a server without ever releasing its
|
||||||
|
source code to the public.
|
||||||
|
|
||||||
|
The GNU Affero General Public License is designed specifically to
|
||||||
|
ensure that, in such cases, the modified source code becomes available
|
||||||
|
to the community. It requires the operator of a network server to
|
||||||
|
provide the source code of the modified version running there to the
|
||||||
|
users of that server. Therefore, public use of a modified version, on
|
||||||
|
a publicly accessible server, gives the public access to the source
|
||||||
|
code of the modified version.
|
||||||
|
|
||||||
|
An older license, called the Affero General Public License and
|
||||||
|
published by Affero, was designed to accomplish similar goals. This is
|
||||||
|
a different license, not a version of the Affero GPL, but Affero has
|
||||||
|
released a new version of the Affero GPL which permits relicensing under
|
||||||
|
this license.
|
||||||
|
|
||||||
|
The precise terms and conditions for copying, distribution and
|
||||||
|
modification follow.
|
||||||
|
|
||||||
|
TERMS AND CONDITIONS
|
||||||
|
|
||||||
|
0. Definitions.
|
||||||
|
|
||||||
|
"This License" refers to version 3 of the GNU Affero General Public License.
|
||||||
|
|
||||||
|
"Copyright" also means copyright-like laws that apply to other kinds of
|
||||||
|
works, such as semiconductor masks.
|
||||||
|
|
||||||
|
"The Program" refers to any copyrightable work licensed under this
|
||||||
|
License. Each licensee is addressed as "you". "Licensees" and
|
||||||
|
"recipients" may be individuals or organizations.
|
||||||
|
|
||||||
|
To "modify" a work means to copy from or adapt all or part of the work
|
||||||
|
in a fashion requiring copyright permission, other than the making of an
|
||||||
|
exact copy. The resulting work is called a "modified version" of the
|
||||||
|
earlier work or a work "based on" the earlier work.
|
||||||
|
|
||||||
|
A "covered work" means either the unmodified Program or a work based
|
||||||
|
on the Program.
|
||||||
|
|
||||||
|
To "propagate" a work means to do anything with it that, without
|
||||||
|
permission, would make you directly or secondarily liable for
|
||||||
|
infringement under applicable copyright law, except executing it on a
|
||||||
|
computer or modifying a private copy. Propagation includes copying,
|
||||||
|
distribution (with or without modification), making available to the
|
||||||
|
public, and in some countries other activities as well.
|
||||||
|
|
||||||
|
To "convey" a work means any kind of propagation that enables other
|
||||||
|
parties to make or receive copies. Mere interaction with a user through
|
||||||
|
a computer network, with no transfer of a copy, is not conveying.
|
||||||
|
|
||||||
|
An interactive user interface displays "Appropriate Legal Notices"
|
||||||
|
to the extent that it includes a convenient and prominently visible
|
||||||
|
feature that (1) displays an appropriate copyright notice, and (2)
|
||||||
|
tells the user that there is no warranty for the work (except to the
|
||||||
|
extent that warranties are provided), that licensees may convey the
|
||||||
|
work under this License, and how to view a copy of this License. If
|
||||||
|
the interface presents a list of user commands or options, such as a
|
||||||
|
menu, a prominent item in the list meets this criterion.
|
||||||
|
|
||||||
|
1. Source Code.
|
||||||
|
|
||||||
|
The "source code" for a work means the preferred form of the work
|
||||||
|
for making modifications to it. "Object code" means any non-source
|
||||||
|
form of a work.
|
||||||
|
|
||||||
|
A "Standard Interface" means an interface that either is an official
|
||||||
|
standard defined by a recognized standards body, or, in the case of
|
||||||
|
interfaces specified for a particular programming language, one that
|
||||||
|
is widely used among developers working in that language.
|
||||||
|
|
||||||
|
The "System Libraries" of an executable work include anything, other
|
||||||
|
than the work as a whole, that (a) is included in the normal form of
|
||||||
|
packaging a Major Component, but which is not part of that Major
|
||||||
|
Component, and (b) serves only to enable use of the work with that
|
||||||
|
Major Component, or to implement a Standard Interface for which an
|
||||||
|
implementation is available to the public in source code form. A
|
||||||
|
"Major Component", in this context, means a major essential component
|
||||||
|
(kernel, window system, and so on) of the specific operating system
|
||||||
|
(if any) on which the executable work runs, or a compiler used to
|
||||||
|
produce the work, or an object code interpreter used to run it.
|
||||||
|
|
||||||
|
The "Corresponding Source" for a work in object code form means all
|
||||||
|
the source code needed to generate, install, and (for an executable
|
||||||
|
work) run the object code and to modify the work, including scripts to
|
||||||
|
control those activities. However, it does not include the work's
|
||||||
|
System Libraries, or general-purpose tools or generally available free
|
||||||
|
programs which are used unmodified in performing those activities but
|
||||||
|
which are not part of the work. For example, Corresponding Source
|
||||||
|
includes interface definition files associated with source files for
|
||||||
|
the work, and the source code for shared libraries and dynamically
|
||||||
|
linked subprograms that the work is specifically designed to require,
|
||||||
|
such as by intimate data communication or control flow between those
|
||||||
|
subprograms and other parts of the work.
|
||||||
|
|
||||||
|
The Corresponding Source need not include anything that users
|
||||||
|
can regenerate automatically from other parts of the Corresponding
|
||||||
|
Source.
|
||||||
|
|
||||||
|
The Corresponding Source for a work in source code form is that
|
||||||
|
same work.
|
||||||
|
|
||||||
|
2. Basic Permissions.
|
||||||
|
|
||||||
|
All rights granted under this License are granted for the term of
|
||||||
|
copyright on the Program, and are irrevocable provided the stated
|
||||||
|
conditions are met. This License explicitly affirms your unlimited
|
||||||
|
permission to run the unmodified Program. The output from running a
|
||||||
|
covered work is covered by this License only if the output, given its
|
||||||
|
content, constitutes a covered work. This License acknowledges your
|
||||||
|
rights of fair use or other equivalent, as provided by copyright law.
|
||||||
|
|
||||||
|
You may make, run and propagate covered works that you do not
|
||||||
|
convey, without conditions so long as your license otherwise remains
|
||||||
|
in force. You may convey covered works to others for the sole purpose
|
||||||
|
of having them make modifications exclusively for you, or provide you
|
||||||
|
with facilities for running those works, provided that you comply with
|
||||||
|
the terms of this License in conveying all material for which you do
|
||||||
|
not control copyright. Those thus making or running the covered works
|
||||||
|
for you must do so exclusively on your behalf, under your direction
|
||||||
|
and control, on terms that prohibit them from making any copies of
|
||||||
|
your copyrighted material outside their relationship with you.
|
||||||
|
|
||||||
|
Conveying under any other circumstances is permitted solely under
|
||||||
|
the conditions stated below. Sublicensing is not allowed; section 10
|
||||||
|
makes it unnecessary.
|
||||||
|
|
||||||
|
3. Protecting Users' Legal Rights From Anti-Circumvention Law.
|
||||||
|
|
||||||
|
No covered work shall be deemed part of an effective technological
|
||||||
|
measure under any applicable law fulfilling obligations under article
|
||||||
|
11 of the WIPO copyright treaty adopted on 20 December 1996, or
|
||||||
|
similar laws prohibiting or restricting circumvention of such
|
||||||
|
measures.
|
||||||
|
|
||||||
|
When you convey a covered work, you waive any legal power to forbid
|
||||||
|
circumvention of technological measures to the extent such circumvention
|
||||||
|
is effected by exercising rights under this License with respect to
|
||||||
|
the covered work, and you disclaim any intention to limit operation or
|
||||||
|
modification of the work as a means of enforcing, against the work's
|
||||||
|
users, your or third parties' legal rights to forbid circumvention of
|
||||||
|
technological measures.
|
||||||
|
|
||||||
|
4. Conveying Verbatim Copies.
|
||||||
|
|
||||||
|
You may convey verbatim copies of the Program's source code as you
|
||||||
|
receive it, in any medium, provided that you conspicuously and
|
||||||
|
appropriately publish on each copy an appropriate copyright notice;
|
||||||
|
keep intact all notices stating that this License and any
|
||||||
|
non-permissive terms added in accord with section 7 apply to the code;
|
||||||
|
keep intact all notices of the absence of any warranty; and give all
|
||||||
|
recipients a copy of this License along with the Program.
|
||||||
|
|
||||||
|
You may charge any price or no price for each copy that you convey,
|
||||||
|
and you may offer support or warranty protection for a fee.
|
||||||
|
|
||||||
|
5. Conveying Modified Source Versions.
|
||||||
|
|
||||||
|
You may convey a work based on the Program, or the modifications to
|
||||||
|
produce it from the Program, in the form of source code under the
|
||||||
|
terms of section 4, provided that you also meet all of these conditions:
|
||||||
|
|
||||||
|
a) The work must carry prominent notices stating that you modified
|
||||||
|
it, and giving a relevant date.
|
||||||
|
|
||||||
|
b) The work must carry prominent notices stating that it is
|
||||||
|
released under this License and any conditions added under section
|
||||||
|
7. This requirement modifies the requirement in section 4 to
|
||||||
|
"keep intact all notices".
|
||||||
|
|
||||||
|
c) You must license the entire work, as a whole, under this
|
||||||
|
License to anyone who comes into possession of a copy. This
|
||||||
|
License will therefore apply, along with any applicable section 7
|
||||||
|
additional terms, to the whole of the work, and all its parts,
|
||||||
|
regardless of how they are packaged. This License gives no
|
||||||
|
permission to license the work in any other way, but it does not
|
||||||
|
invalidate such permission if you have separately received it.
|
||||||
|
|
||||||
|
d) If the work has interactive user interfaces, each must display
|
||||||
|
Appropriate Legal Notices; however, if the Program has interactive
|
||||||
|
interfaces that do not display Appropriate Legal Notices, your
|
||||||
|
work need not make them do so.
|
||||||
|
|
||||||
|
A compilation of a covered work with other separate and independent
|
||||||
|
works, which are not by their nature extensions of the covered work,
|
||||||
|
and which are not combined with it such as to form a larger program,
|
||||||
|
in or on a volume of a storage or distribution medium, is called an
|
||||||
|
"aggregate" if the compilation and its resulting copyright are not
|
||||||
|
used to limit the access or legal rights of the compilation's users
|
||||||
|
beyond what the individual works permit. Inclusion of a covered work
|
||||||
|
in an aggregate does not cause this License to apply to the other
|
||||||
|
parts of the aggregate.
|
||||||
|
|
||||||
|
6. Conveying Non-Source Forms.
|
||||||
|
|
||||||
|
You may convey a covered work in object code form under the terms
|
||||||
|
of sections 4 and 5, provided that you also convey the
|
||||||
|
machine-readable Corresponding Source under the terms of this License,
|
||||||
|
in one of these ways:
|
||||||
|
|
||||||
|
a) Convey the object code in, or embodied in, a physical product
|
||||||
|
(including a physical distribution medium), accompanied by the
|
||||||
|
Corresponding Source fixed on a durable physical medium
|
||||||
|
customarily used for software interchange.
|
||||||
|
|
||||||
|
b) Convey the object code in, or embodied in, a physical product
|
||||||
|
(including a physical distribution medium), accompanied by a
|
||||||
|
written offer, valid for at least three years and valid for as
|
||||||
|
long as you offer spare parts or customer support for that product
|
||||||
|
model, to give anyone who possesses the object code either (1) a
|
||||||
|
copy of the Corresponding Source for all the software in the
|
||||||
|
product that is covered by this License, on a durable physical
|
||||||
|
medium customarily used for software interchange, for a price no
|
||||||
|
more than your reasonable cost of physically performing this
|
||||||
|
conveying of source, or (2) access to copy the
|
||||||
|
Corresponding Source from a network server at no charge.
|
||||||
|
|
||||||
|
c) Convey individual copies of the object code with a copy of the
|
||||||
|
written offer to provide the Corresponding Source. This
|
||||||
|
alternative is allowed only occasionally and noncommercially, and
|
||||||
|
only if you received the object code with such an offer, in accord
|
||||||
|
with subsection 6b.
|
||||||
|
|
||||||
|
d) Convey the object code by offering access from a designated
|
||||||
|
place (gratis or for a charge), and offer equivalent access to the
|
||||||
|
Corresponding Source in the same way through the same place at no
|
||||||
|
further charge. You need not require recipients to copy the
|
||||||
|
Corresponding Source along with the object code. If the place to
|
||||||
|
copy the object code is a network server, the Corresponding Source
|
||||||
|
may be on a different server (operated by you or a third party)
|
||||||
|
that supports equivalent copying facilities, provided you maintain
|
||||||
|
clear directions next to the object code saying where to find the
|
||||||
|
Corresponding Source. Regardless of what server hosts the
|
||||||
|
Corresponding Source, you remain obligated to ensure that it is
|
||||||
|
available for as long as needed to satisfy these requirements.
|
||||||
|
|
||||||
|
e) Convey the object code using peer-to-peer transmission, provided
|
||||||
|
you inform other peers where the object code and Corresponding
|
||||||
|
Source of the work are being offered to the general public at no
|
||||||
|
charge under subsection 6d.
|
||||||
|
|
||||||
|
A separable portion of the object code, whose source code is excluded
|
||||||
|
from the Corresponding Source as a System Library, need not be
|
||||||
|
included in conveying the object code work.
|
||||||
|
|
||||||
|
A "User Product" is either (1) a "consumer product", which means any
|
||||||
|
tangible personal property which is normally used for personal, family,
|
||||||
|
or household purposes, or (2) anything designed or sold for incorporation
|
||||||
|
into a dwelling. In determining whether a product is a consumer product,
|
||||||
|
doubtful cases shall be resolved in favor of coverage. For a particular
|
||||||
|
product received by a particular user, "normally used" refers to a
|
||||||
|
typical or common use of that class of product, regardless of the status
|
||||||
|
of the particular user or of the way in which the particular user
|
||||||
|
actually uses, or expects or is expected to use, the product. A product
|
||||||
|
is a consumer product regardless of whether the product has substantial
|
||||||
|
commercial, industrial or non-consumer uses, unless such uses represent
|
||||||
|
the only significant mode of use of the product.
|
||||||
|
|
||||||
|
"Installation Information" for a User Product means any methods,
|
||||||
|
procedures, authorization keys, or other information required to install
|
||||||
|
and execute modified versions of a covered work in that User Product from
|
||||||
|
a modified version of its Corresponding Source. The information must
|
||||||
|
suffice to ensure that the continued functioning of the modified object
|
||||||
|
code is in no case prevented or interfered with solely because
|
||||||
|
modification has been made.
|
||||||
|
|
||||||
|
If you convey an object code work under this section in, or with, or
|
||||||
|
specifically for use in, a User Product, and the conveying occurs as
|
||||||
|
part of a transaction in which the right of possession and use of the
|
||||||
|
User Product is transferred to the recipient in perpetuity or for a
|
||||||
|
fixed term (regardless of how the transaction is characterized), the
|
||||||
|
Corresponding Source conveyed under this section must be accompanied
|
||||||
|
by the Installation Information. But this requirement does not apply
|
||||||
|
if neither you nor any third party retains the ability to install
|
||||||
|
modified object code on the User Product (for example, the work has
|
||||||
|
been installed in ROM).
|
||||||
|
|
||||||
|
The requirement to provide Installation Information does not include a
|
||||||
|
requirement to continue to provide support service, warranty, or updates
|
||||||
|
for a work that has been modified or installed by the recipient, or for
|
||||||
|
the User Product in which it has been modified or installed. Access to a
|
||||||
|
network may be denied when the modification itself materially and
|
||||||
|
adversely affects the operation of the network or violates the rules and
|
||||||
|
protocols for communication across the network.
|
||||||
|
|
||||||
|
Corresponding Source conveyed, and Installation Information provided,
|
||||||
|
in accord with this section must be in a format that is publicly
|
||||||
|
documented (and with an implementation available to the public in
|
||||||
|
source code form), and must require no special password or key for
|
||||||
|
unpacking, reading or copying.
|
||||||
|
|
||||||
|
7. Additional Terms.
|
||||||
|
|
||||||
|
"Additional permissions" are terms that supplement the terms of this
|
||||||
|
License by making exceptions from one or more of its conditions.
|
||||||
|
Additional permissions that are applicable to the entire Program shall
|
||||||
|
be treated as though they were included in this License, to the extent
|
||||||
|
that they are valid under applicable law. If additional permissions
|
||||||
|
apply only to part of the Program, that part may be used separately
|
||||||
|
under those permissions, but the entire Program remains governed by
|
||||||
|
this License without regard to the additional permissions.
|
||||||
|
|
||||||
|
When you convey a copy of a covered work, you may at your option
|
||||||
|
remove any additional permissions from that copy, or from any part of
|
||||||
|
it. (Additional permissions may be written to require their own
|
||||||
|
removal in certain cases when you modify the work.) You may place
|
||||||
|
additional permissions on material, added by you to a covered work,
|
||||||
|
for which you have or can give appropriate copyright permission.
|
||||||
|
|
||||||
|
Notwithstanding any other provision of this License, for material you
|
||||||
|
add to a covered work, you may (if authorized by the copyright holders of
|
||||||
|
that material) supplement the terms of this License with terms:
|
||||||
|
|
||||||
|
a) Disclaiming warranty or limiting liability differently from the
|
||||||
|
terms of sections 15 and 16 of this License; or
|
||||||
|
|
||||||
|
b) Requiring preservation of specified reasonable legal notices or
|
||||||
|
author attributions in that material or in the Appropriate Legal
|
||||||
|
Notices displayed by works containing it; or
|
||||||
|
|
||||||
|
c) Prohibiting misrepresentation of the origin of that material, or
|
||||||
|
requiring that modified versions of such material be marked in
|
||||||
|
reasonable ways as different from the original version; or
|
||||||
|
|
||||||
|
d) Limiting the use for publicity purposes of names of licensors or
|
||||||
|
authors of the material; or
|
||||||
|
|
||||||
|
e) Declining to grant rights under trademark law for use of some
|
||||||
|
trade names, trademarks, or service marks; or
|
||||||
|
|
||||||
|
f) Requiring indemnification of licensors and authors of that
|
||||||
|
material by anyone who conveys the material (or modified versions of
|
||||||
|
it) with contractual assumptions of liability to the recipient, for
|
||||||
|
any liability that these contractual assumptions directly impose on
|
||||||
|
those licensors and authors.
|
||||||
|
|
||||||
|
All other non-permissive additional terms are considered "further
|
||||||
|
restrictions" within the meaning of section 10. If the Program as you
|
||||||
|
received it, or any part of it, contains a notice stating that it is
|
||||||
|
governed by this License along with a term that is a further
|
||||||
|
restriction, you may remove that term. If a license document contains
|
||||||
|
a further restriction but permits relicensing or conveying under this
|
||||||
|
License, you may add to a covered work material governed by the terms
|
||||||
|
of that license document, provided that the further restriction does
|
||||||
|
not survive such relicensing or conveying.
|
||||||
|
|
||||||
|
If you add terms to a covered work in accord with this section, you
|
||||||
|
must place, in the relevant source files, a statement of the
|
||||||
|
additional terms that apply to those files, or a notice indicating
|
||||||
|
where to find the applicable terms.
|
||||||
|
|
||||||
|
Additional terms, permissive or non-permissive, may be stated in the
|
||||||
|
form of a separately written license, or stated as exceptions;
|
||||||
|
the above requirements apply either way.
|
||||||
|
|
||||||
|
8. Termination.
|
||||||
|
|
||||||
|
You may not propagate or modify a covered work except as expressly
|
||||||
|
provided under this License. Any attempt otherwise to propagate or
|
||||||
|
modify it is void, and will automatically terminate your rights under
|
||||||
|
this License (including any patent licenses granted under the third
|
||||||
|
paragraph of section 11).
|
||||||
|
|
||||||
|
However, if you cease all violation of this License, then your
|
||||||
|
license from a particular copyright holder is reinstated (a)
|
||||||
|
provisionally, unless and until the copyright holder explicitly and
|
||||||
|
finally terminates your license, and (b) permanently, if the copyright
|
||||||
|
holder fails to notify you of the violation by some reasonable means
|
||||||
|
prior to 60 days after the cessation.
|
||||||
|
|
||||||
|
Moreover, your license from a particular copyright holder is
|
||||||
|
reinstated permanently if the copyright holder notifies you of the
|
||||||
|
violation by some reasonable means, this is the first time you have
|
||||||
|
received notice of violation of this License (for any work) from that
|
||||||
|
copyright holder, and you cure the violation prior to 30 days after
|
||||||
|
your receipt of the notice.
|
||||||
|
|
||||||
|
Termination of your rights under this section does not terminate the
|
||||||
|
licenses of parties who have received copies or rights from you under
|
||||||
|
this License. If your rights have been terminated and not permanently
|
||||||
|
reinstated, you do not qualify to receive new licenses for the same
|
||||||
|
material under section 10.
|
||||||
|
|
||||||
|
9. Acceptance Not Required for Having Copies.
|
||||||
|
|
||||||
|
You are not required to accept this License in order to receive or
|
||||||
|
run a copy of the Program. Ancillary propagation of a covered work
|
||||||
|
occurring solely as a consequence of using peer-to-peer transmission
|
||||||
|
to receive a copy likewise does not require acceptance. However,
|
||||||
|
nothing other than this License grants you permission to propagate or
|
||||||
|
modify any covered work. These actions infringe copyright if you do
|
||||||
|
not accept this License. Therefore, by modifying or propagating a
|
||||||
|
covered work, you indicate your acceptance of this License to do so.
|
||||||
|
|
||||||
|
10. Automatic Licensing of Downstream Recipients.
|
||||||
|
|
||||||
|
Each time you convey a covered work, the recipient automatically
|
||||||
|
receives a license from the original licensors, to run, modify and
|
||||||
|
propagate that work, subject to this License. You are not responsible
|
||||||
|
for enforcing compliance by third parties with this License.
|
||||||
|
|
||||||
|
An "entity transaction" is a transaction transferring control of an
|
||||||
|
organization, or substantially all assets of one, or subdividing an
|
||||||
|
organization, or merging organizations. If propagation of a covered
|
||||||
|
work results from an entity transaction, each party to that
|
||||||
|
transaction who receives a copy of the work also receives whatever
|
||||||
|
licenses to the work the party's predecessor in interest had or could
|
||||||
|
give under the previous paragraph, plus a right to possession of the
|
||||||
|
Corresponding Source of the work from the predecessor in interest, if
|
||||||
|
the predecessor has it or can get it with reasonable efforts.
|
||||||
|
|
||||||
|
You may not impose any further restrictions on the exercise of the
|
||||||
|
rights granted or affirmed under this License. For example, you may
|
||||||
|
not impose a license fee, royalty, or other charge for exercise of
|
||||||
|
rights granted under this License, and you may not initiate litigation
|
||||||
|
(including a cross-claim or counterclaim in a lawsuit) alleging that
|
||||||
|
any patent claim is infringed by making, using, selling, offering for
|
||||||
|
sale, or importing the Program or any portion of it.
|
||||||
|
|
||||||
|
11. Patents.
|
||||||
|
|
||||||
|
A "contributor" is a copyright holder who authorizes use under this
|
||||||
|
License of the Program or a work on which the Program is based. The
|
||||||
|
work thus licensed is called the contributor's "contributor version".
|
||||||
|
|
||||||
|
A contributor's "essential patent claims" are all patent claims
|
||||||
|
owned or controlled by the contributor, whether already acquired or
|
||||||
|
hereafter acquired, that would be infringed by some manner, permitted
|
||||||
|
by this License, of making, using, or selling its contributor version,
|
||||||
|
but do not include claims that would be infringed only as a
|
||||||
|
consequence of further modification of the contributor version. For
|
||||||
|
purposes of this definition, "control" includes the right to grant
|
||||||
|
patent sublicenses in a manner consistent with the requirements of
|
||||||
|
this License.
|
||||||
|
|
||||||
|
Each contributor grants you a non-exclusive, worldwide, royalty-free
|
||||||
|
patent license under the contributor's essential patent claims, to
|
||||||
|
make, use, sell, offer for sale, import and otherwise run, modify and
|
||||||
|
propagate the contents of its contributor version.
|
||||||
|
|
||||||
|
In the following three paragraphs, a "patent license" is any express
|
||||||
|
agreement or commitment, however denominated, not to enforce a patent
|
||||||
|
(such as an express permission to practice a patent or covenant not to
|
||||||
|
sue for patent infringement). To "grant" such a patent license to a
|
||||||
|
party means to make such an agreement or commitment not to enforce a
|
||||||
|
patent against the party.
|
||||||
|
|
||||||
|
If you convey a covered work, knowingly relying on a patent license,
|
||||||
|
and the Corresponding Source of the work is not available for anyone
|
||||||
|
to copy, free of charge and under the terms of this License, through a
|
||||||
|
publicly available network server or other readily accessible means,
|
||||||
|
then you must either (1) cause the Corresponding Source to be so
|
||||||
|
available, or (2) arrange to deprive yourself of the benefit of the
|
||||||
|
patent license for this particular work, or (3) arrange, in a manner
|
||||||
|
consistent with the requirements of this License, to extend the patent
|
||||||
|
license to downstream recipients. "Knowingly relying" means you have
|
||||||
|
actual knowledge that, but for the patent license, your conveying the
|
||||||
|
covered work in a country, or your recipient's use of the covered work
|
||||||
|
in a country, would infringe one or more identifiable patents in that
|
||||||
|
country that you have reason to believe are valid.
|
||||||
|
|
||||||
|
If, pursuant to or in connection with a single transaction or
|
||||||
|
arrangement, you convey, or propagate by procuring conveyance of, a
|
||||||
|
covered work, and grant a patent license to some of the parties
|
||||||
|
receiving the covered work authorizing them to use, propagate, modify
|
||||||
|
or convey a specific copy of the covered work, then the patent license
|
||||||
|
you grant is automatically extended to all recipients of the covered
|
||||||
|
work and works based on it.
|
||||||
|
|
||||||
|
A patent license is "discriminatory" if it does not include within
|
||||||
|
the scope of its coverage, prohibits the exercise of, or is
|
||||||
|
conditioned on the non-exercise of one or more of the rights that are
|
||||||
|
specifically granted under this License. You may not convey a covered
|
||||||
|
work if you are a party to an arrangement with a third party that is
|
||||||
|
in the business of distributing software, under which you make payment
|
||||||
|
to the third party based on the extent of your activity of conveying
|
||||||
|
the work, and under which the third party grants, to any of the
|
||||||
|
parties who would receive the covered work from you, a discriminatory
|
||||||
|
patent license (a) in connection with copies of the covered work
|
||||||
|
conveyed by you (or copies made from those copies), or (b) primarily
|
||||||
|
for and in connection with specific products or compilations that
|
||||||
|
contain the covered work, unless you entered into that arrangement,
|
||||||
|
or that patent license was granted, prior to 28 March 2007.
|
||||||
|
|
||||||
|
Nothing in this License shall be construed as excluding or limiting
|
||||||
|
any implied license or other defenses to infringement that may
|
||||||
|
otherwise be available to you under applicable patent law.
|
||||||
|
|
||||||
|
12. No Surrender of Others' Freedom.
|
||||||
|
|
||||||
|
If conditions are imposed on you (whether by court order, agreement or
|
||||||
|
otherwise) that contradict the conditions of this License, they do not
|
||||||
|
excuse you from the conditions of this License. If you cannot convey a
|
||||||
|
covered work so as to satisfy simultaneously your obligations under this
|
||||||
|
License and any other pertinent obligations, then as a consequence you may
|
||||||
|
not convey it at all. For example, if you agree to terms that obligate you
|
||||||
|
to collect a royalty for further conveying from those to whom you convey
|
||||||
|
the Program, the only way you could satisfy both those terms and this
|
||||||
|
License would be to refrain entirely from conveying the Program.
|
||||||
|
|
||||||
|
13. Remote Network Interaction; Use with the GNU General Public License.
|
||||||
|
|
||||||
|
Notwithstanding any other provision of this License, if you modify the
|
||||||
|
Program, your modified version must prominently offer all users
|
||||||
|
interacting with it remotely through a computer network (if your version
|
||||||
|
supports such interaction) an opportunity to receive the Corresponding
|
||||||
|
Source of your version by providing access to the Corresponding Source
|
||||||
|
from a network server at no charge, through some standard or customary
|
||||||
|
means of facilitating copying of software. This Corresponding Source
|
||||||
|
shall include the Corresponding Source for any work covered by version 3
|
||||||
|
of the GNU General Public License that is incorporated pursuant to the
|
||||||
|
following paragraph.
|
||||||
|
|
||||||
|
Notwithstanding any other provision of this License, you have
|
||||||
|
permission to link or combine any covered work with a work licensed
|
||||||
|
under version 3 of the GNU General Public License into a single
|
||||||
|
combined work, and to convey the resulting work. The terms of this
|
||||||
|
License will continue to apply to the part which is the covered work,
|
||||||
|
but the work with which it is combined will remain governed by version
|
||||||
|
3 of the GNU General Public License.
|
||||||
|
|
||||||
|
14. Revised Versions of this License.
|
||||||
|
|
||||||
|
The Free Software Foundation may publish revised and/or new versions of
|
||||||
|
the GNU Affero General Public License from time to time. Such new versions
|
||||||
|
will be similar in spirit to the present version, but may differ in detail to
|
||||||
|
address new problems or concerns.
|
||||||
|
|
||||||
|
Each version is given a distinguishing version number. If the
|
||||||
|
Program specifies that a certain numbered version of the GNU Affero General
|
||||||
|
Public License "or any later version" applies to it, you have the
|
||||||
|
option of following the terms and conditions either of that numbered
|
||||||
|
version or of any later version published by the Free Software
|
||||||
|
Foundation. If the Program does not specify a version number of the
|
||||||
|
GNU Affero General Public License, you may choose any version ever published
|
||||||
|
by the Free Software Foundation.
|
||||||
|
|
||||||
|
If the Program specifies that a proxy can decide which future
|
||||||
|
versions of the GNU Affero General Public License can be used, that proxy's
|
||||||
|
public statement of acceptance of a version permanently authorizes you
|
||||||
|
to choose that version for the Program.
|
||||||
|
|
||||||
|
Later license versions may give you additional or different
|
||||||
|
permissions. However, no additional obligations are imposed on any
|
||||||
|
author or copyright holder as a result of your choosing to follow a
|
||||||
|
later version.
|
||||||
|
|
||||||
|
15. Disclaimer of Warranty.
|
||||||
|
|
||||||
|
THERE IS NO WARRANTY FOR THE PROGRAM, TO THE EXTENT PERMITTED BY
|
||||||
|
APPLICABLE LAW. EXCEPT WHEN OTHERWISE STATED IN WRITING THE COPYRIGHT
|
||||||
|
HOLDERS AND/OR OTHER PARTIES PROVIDE THE PROGRAM "AS IS" WITHOUT WARRANTY
|
||||||
|
OF ANY KIND, EITHER EXPRESSED OR IMPLIED, INCLUDING, BUT NOT LIMITED TO,
|
||||||
|
THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||||
|
PURPOSE. THE ENTIRE RISK AS TO THE QUALITY AND PERFORMANCE OF THE PROGRAM
|
||||||
|
IS WITH YOU. SHOULD THE PROGRAM PROVE DEFECTIVE, YOU ASSUME THE COST OF
|
||||||
|
ALL NECESSARY SERVICING, REPAIR OR CORRECTION.
|
||||||
|
|
||||||
|
16. Limitation of Liability.
|
||||||
|
|
||||||
|
IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING
|
||||||
|
WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MODIFIES AND/OR CONVEYS
|
||||||
|
THE PROGRAM AS PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY
|
||||||
|
GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING OUT OF THE
|
||||||
|
USE OR INABILITY TO USE THE PROGRAM (INCLUDING BUT NOT LIMITED TO LOSS OF
|
||||||
|
DATA OR DATA BEING RENDERED INACCURATE OR LOSSES SUSTAINED BY YOU OR THIRD
|
||||||
|
PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE WITH ANY OTHER PROGRAMS),
|
||||||
|
EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE POSSIBILITY OF
|
||||||
|
SUCH DAMAGES.
|
||||||
|
|
||||||
|
17. Interpretation of Sections 15 and 16.
|
||||||
|
|
||||||
|
If the disclaimer of warranty and limitation of liability provided
|
||||||
|
above cannot be given local legal effect according to their terms,
|
||||||
|
reviewing courts shall apply local law that most closely approximates
|
||||||
|
an absolute waiver of all civil liability in connection with the
|
||||||
|
Program, unless a warranty or assumption of liability accompanies a
|
||||||
|
copy of the Program in return for a fee.
|
||||||
|
|
||||||
|
END OF TERMS AND CONDITIONS
|
||||||
|
|
||||||
|
How to Apply These Terms to Your New Programs
|
||||||
|
|
||||||
|
If you develop a new program, and you want it to be of the greatest
|
||||||
|
possible use to the public, the best way to achieve this is to make it
|
||||||
|
free software which everyone can redistribute and change under these terms.
|
||||||
|
|
||||||
|
To do so, attach the following notices to the program. It is safest
|
||||||
|
to attach them to the start of each source file to most effectively
|
||||||
|
state the exclusion of warranty; and each file should have at least
|
||||||
|
the "copyright" line and a pointer to where the full notice is found.
|
||||||
|
|
||||||
|
<one line to give the program's name and a brief idea of what it does.>
|
||||||
|
Copyright (C) <year> <name of author>
|
||||||
|
|
||||||
|
This program is free software: you can redistribute it and/or modify
|
||||||
|
it under the terms of the GNU Affero General Public License as published by
|
||||||
|
the Free Software Foundation, either version 3 of the License, or
|
||||||
|
(at your option) any later version.
|
||||||
|
|
||||||
|
This program is distributed in the hope that it will be useful,
|
||||||
|
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
GNU Affero General Public License for more details.
|
||||||
|
|
||||||
|
You should have received a copy of the GNU Affero General Public License
|
||||||
|
along with this program. If not, see <https://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
|
Also add information on how to contact you by electronic and paper mail.
|
||||||
|
|
||||||
|
If your software can interact with users remotely through a computer
|
||||||
|
network, you should also make sure that it provides a way for users to
|
||||||
|
get its source. For example, if your program is a web application, its
|
||||||
|
interface could display a "Source" link that leads users to an archive
|
||||||
|
of the code. There are many ways you could offer source, and different
|
||||||
|
solutions will be better for different programs; see section 13 for the
|
||||||
|
specific requirements.
|
||||||
|
|
||||||
|
You should also get your employer (if you work as a programmer) or school,
|
||||||
|
if any, to sign a "copyright disclaimer" for the program, if necessary.
|
||||||
|
For more information on this, and how to apply and follow the GNU AGPL, see
|
||||||
|
<https://www.gnu.org/licenses/>.
|
||||||
@@ -1,14 +1,230 @@
|
|||||||
# FabledCurator
|
# FabledCurator
|
||||||
|
|
||||||
Self-hosted media curation — gallery, ML tagging, and subscription-driven downloading in one app. Part of the FabledSword family.
|
<!-- overview:start -->
|
||||||
|
Self-hosted media curation — a gallery, ML auto-tagging, and subscription-driven
|
||||||
|
downloading in one application. Part of the FabledSword family.
|
||||||
|
|
||||||
Combines what was [ImageRepo](https://git.fabledsword.com/bvandeusen/ImageRepo) (gallery, ML, importer) and [GallerySubscriber](https://git.fabledsword.com/bvandeusen/GallerySubscriber) (gallery-dl wrapper, subscriptions, credential capture) into a single product.
|
## What it does
|
||||||
|
|
||||||
## Status
|
You point it at creators you follow. It downloads what they post, files it,
|
||||||
|
tags it, and gives you something better than a folder full of images to look
|
||||||
|
through afterwards.
|
||||||
|
|
||||||
In production. `main` is continuously deployed — every merge to `main` builds
|
- **Gallery and browsing.** Images, videos and multi-page works, organised by
|
||||||
and publishes `:latest` images, so whatever is on `main` is what is running.
|
artist, tag, post and series. A Showcase front page, a filterable gallery, a
|
||||||
Day-to-day work happens on `dev`, which publishes `:dev` images.
|
similarity-driven Explore view, and a page-turning reader for series.
|
||||||
|
- **Subscriptions.** Follows creators on Patreon, SubscribeStar, Pixiv and
|
||||||
|
anything `gallery-dl` supports, on a schedule. Handles paywalled posts using
|
||||||
|
your own logged-in session.
|
||||||
|
- **ML tagging.** Runs image models in-container to suggest tags, group
|
||||||
|
characters, find near-duplicates and power similarity search. Suggestions are
|
||||||
|
reviewable — it proposes, you confirm, and it learns which proposals you keep
|
||||||
|
rejecting.
|
||||||
|
- **Deduplication and provenance.** Everything that arrives is hashed and
|
||||||
|
deduplicated by content, metadata sidecars are read wherever the source
|
||||||
|
writes them, and every file keeps a record of where it came from.
|
||||||
|
- **Maintenance.** Backups, library audits, thumbnail and embedding backfills,
|
||||||
|
orphan cleanup — all from the UI, all as background jobs you can watch.
|
||||||
|
|
||||||
|
Everything is configured from the Settings UI and stored in the database. There
|
||||||
|
is no config file to edit beyond a handful of bootstrap environment variables.
|
||||||
|
<!-- overview:end -->
|
||||||
|
|
||||||
|
## Before you expose it
|
||||||
|
|
||||||
|
**FabledCurator has no login.** There are no user accounts, no passwords and no
|
||||||
|
permission model. Anything that can reach the port is an administrator.
|
||||||
|
|
||||||
|
That matters more here than it would in most self-hosted apps, because of what
|
||||||
|
this one stores: **live platform session cookies for Patreon, SubscribeStar and
|
||||||
|
Pixiv** — accounts that usually have a payment method attached. Whoever reaches
|
||||||
|
the port can read them, alongside your entire library.
|
||||||
|
|
||||||
|
So:
|
||||||
|
|
||||||
|
- Bind it to a LAN, a VPN, or a tunnel you control.
|
||||||
|
- Do not port-forward it. Do not put it on a public hostname.
|
||||||
|
- A reverse proxy that adds TLS but no authentication **does not help**. If you
|
||||||
|
want it reachable from outside, put an authenticating proxy in front of it —
|
||||||
|
a forward-auth provider, HTTP basic auth, an identity-aware tunnel — and treat
|
||||||
|
that layer as the only thing standing between the internet and your accounts.
|
||||||
|
|
||||||
|
This is a deliberate design decision for a single-operator tool on a trusted
|
||||||
|
network, not a bug and not an oversight. It is stated here because it decides
|
||||||
|
how you are allowed to deploy it. [SECURITY.md](SECURITY.md) covers the rest of
|
||||||
|
the threat model.
|
||||||
|
|
||||||
|
## Requirements
|
||||||
|
|
||||||
|
- **Docker** with Compose v2.
|
||||||
|
- **~4 GB RAM** for the app, plus whatever Postgres needs for your library size.
|
||||||
|
- **Disk** for your media, plus several GB for ML model weights.
|
||||||
|
- **No GPU required.** The ML worker runs on CPU — tagging and embedding are
|
||||||
|
slower, and that is the whole difference. A GPU is only involved if you
|
||||||
|
separately run the optional agent (below), which is a different machine's job.
|
||||||
|
|
||||||
|
## Install
|
||||||
|
|
||||||
|
```bash
|
||||||
|
git clone https://git.fabledsword.com/bvandeusen/FabledCurator.git
|
||||||
|
cd FabledCurator
|
||||||
|
|
||||||
|
cp .env.example .env
|
||||||
|
$EDITOR .env # set DB_PASSWORD and SECRET_KEY
|
||||||
|
|
||||||
|
docker compose -f docker-compose.yml up -d
|
||||||
|
```
|
||||||
|
|
||||||
|
Then open <http://localhost:8080>.
|
||||||
|
|
||||||
|
**The `-f docker-compose.yml` is required, not decoration.** Compose
|
||||||
|
auto-merges `docker-compose.override.yml` when you leave it off, and that
|
||||||
|
override builds the images locally from source — the contributor path, not
|
||||||
|
yours. Naming the file explicitly skips the override and pulls the published
|
||||||
|
`:latest` images, which is the stable channel built from `main`.
|
||||||
|
|
||||||
|
If you forget it, the symptom is a long build instead of a quick pull.
|
||||||
|
|
||||||
|
## First run
|
||||||
|
|
||||||
|
The database schema is created automatically on first start — the web container
|
||||||
|
runs its migrations before serving. Nothing to initialise by hand.
|
||||||
|
|
||||||
|
**One thing does need a deliberate act, and the app will not start without it.**
|
||||||
|
FabledCurator encrypts your stored platform credentials with a key it keeps at
|
||||||
|
`./images/secrets/credential_key.b64`. On a brand-new install that file does not
|
||||||
|
exist, and rather than quietly creating one the app stops:
|
||||||
|
|
||||||
|
```
|
||||||
|
MissingCredentialKey: Fernet key file not found at /images/secrets/credential_key.b64
|
||||||
|
```
|
||||||
|
|
||||||
|
Set `CURATOR_BOOTSTRAP_NEW_KEY=1` in your `.env` for the first `up`, then delete
|
||||||
|
the line once the container is running. `.env.example` ships it with that
|
||||||
|
instruction attached.
|
||||||
|
|
||||||
|
The refusal is deliberate, and worth understanding rather than working around:
|
||||||
|
auto-creating a key is indistinguishable from the disaster case — a restore that
|
||||||
|
brought the database back but lost `./images/secrets` — where it would mint a key
|
||||||
|
that cannot decrypt anything, leaving an instance that looks healthy while every
|
||||||
|
paywalled download fails. Making you say so once, on an empty install, is the
|
||||||
|
price of that not happening silently later.
|
||||||
|
|
||||||
|
**Which means: back up `./images/secrets/` alongside your database.** It is the
|
||||||
|
only thing that can read your stored credentials. A database restored without it
|
||||||
|
needs every credential entered again by hand.
|
||||||
|
|
||||||
|
A few other things are worth knowing about the first few minutes:
|
||||||
|
|
||||||
|
- **The ML worker downloads its model weights on first boot**, several GB from
|
||||||
|
HuggingFace into `./models`. Until that finishes, tagging is queued rather
|
||||||
|
than broken. It is idempotent — a restart resumes rather than refetches.
|
||||||
|
- **The gallery starts empty**, and that is the expected state. Add a creator
|
||||||
|
under **Subscriptions** and it fills as posts come down.
|
||||||
|
- **If you already have a library on disk**, there is no screen that imports
|
||||||
|
it, and there is not going to be one. Folder ingestion had a UI until July
|
||||||
|
2026; it was retired once posts began arriving entirely through
|
||||||
|
subscriptions and the browser extension, and the decision to leave it
|
||||||
|
retired is deliberate — the folder path carries complexity the product does
|
||||||
|
not need in order to do its job. The supported way to fill a new install is
|
||||||
|
to add the creators you follow under **Subscriptions** and let it pull.
|
||||||
|
|
||||||
|
The `/api/import/trigger` endpoint is still wired up for anyone who wants to
|
||||||
|
script a one-off against a folder mounted at `./import`, and its progress
|
||||||
|
shows under **Settings → Activity**. Treat it as an unsupported escape
|
||||||
|
hatch rather than a feature: nothing in the UI drives it and nothing else
|
||||||
|
in this README depends on it.
|
||||||
|
- **To download from a paywalled account**, FabledCurator needs that account's
|
||||||
|
session — see the browser extension below. Without one it can still fetch
|
||||||
|
public posts.
|
||||||
|
- **Check Settings → Overview** to confirm the workers are alive. Every long
|
||||||
|
operation in FabledCurator is a background job, so if the queues are not
|
||||||
|
running, the UI will look like it is ignoring you rather than like it is
|
||||||
|
broken.
|
||||||
|
|
||||||
|
## The browser extension
|
||||||
|
|
||||||
|
A Firefox extension does two jobs: it hands your logged-in platform sessions to
|
||||||
|
FabledCurator so it can download on your behalf, and it adds a creator as a
|
||||||
|
subscription in one click from their page.
|
||||||
|
|
||||||
|
It ships **inside the web image** — there is no add-on store listing to find.
|
||||||
|
Go to **Subscriptions → Settings**, find the *Browser extension* card, and click
|
||||||
|
**Install Firefox extension**. The XPI is Mozilla-signed, so Firefox installs it
|
||||||
|
like any other add-on; the button serves it directly rather than making you
|
||||||
|
download and side-load a file.
|
||||||
|
|
||||||
|
It pairs with your instance using an API key generated automatically on first
|
||||||
|
use. The bar directly under that card shows the key and can rotate it.
|
||||||
|
|
||||||
|
See [extension/README.md](extension/README.md) for what it does in detail.
|
||||||
|
|
||||||
|
## The GPU agent
|
||||||
|
|
||||||
|
Optional, and separate. If you have a desktop with a graphics card, you can run
|
||||||
|
an agent on it that leases ML jobs from FabledCurator over HTTP, does them on
|
||||||
|
the GPU, and hands the results back. It never touches the database or Redis, so
|
||||||
|
it is safe to run somewhere the rest of the stack is not.
|
||||||
|
|
||||||
|
Run it for a burst of tagging, stop it to get your card back. It deploys from
|
||||||
|
`agent/docker-compose.yml`, not the main stack — see
|
||||||
|
[agent/README.md](agent/README.md).
|
||||||
|
|
||||||
|
## Upgrading
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker compose -f docker-compose.yml pull
|
||||||
|
docker compose -f docker-compose.yml up -d
|
||||||
|
```
|
||||||
|
|
||||||
|
Migrations run automatically on start. Take a database backup first — Settings →
|
||||||
|
Maintenance has one — because the schema moves forward and does not move back.
|
||||||
|
|
||||||
|
## Deployment posture
|
||||||
|
|
||||||
|
FabledCurator is built to run inside a homelab over plain HTTP. It does not
|
||||||
|
generate certificates, redirect to HTTPS, or set HSTS. If you want TLS,
|
||||||
|
terminate it at your reverse proxy. See [Before you expose it](#before-you-expose-it)
|
||||||
|
for why TLS alone is not enough.
|
||||||
|
|
||||||
|
## Troubleshooting
|
||||||
|
|
||||||
|
**The UI loads but nothing ever finishes.** The web container is up and the
|
||||||
|
workers are not. `docker compose -f docker-compose.yml ps` — check `worker`,
|
||||||
|
`scheduler` and `ml-worker` are healthy, not restarting.
|
||||||
|
|
||||||
|
**`docker compose up` started building instead of pulling.** You left off
|
||||||
|
`-f docker-compose.yml`, so the dev override took over. See [Install](#install).
|
||||||
|
|
||||||
|
**Downloads fail with an auth error.** The stored session for that platform has
|
||||||
|
expired. Re-capture it with the extension; sessions do not last forever.
|
||||||
|
|
||||||
|
**Which build am I running?** The foot of Settings shows a version and a
|
||||||
|
channel, and `/api/health` returns the same two fields. There are no version
|
||||||
|
tags on the images, so this is the authoritative answer.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# Developing FabledCurator
|
||||||
|
|
||||||
|
Everything below is about working on FabledCurator rather than running it. If
|
||||||
|
you are installing it, you are done — see [CONTRIBUTING.md](CONTRIBUTING.md) if
|
||||||
|
you want to send a patch.
|
||||||
|
|
||||||
|
## Status and channels
|
||||||
|
|
||||||
|
In production. `main` is continuously deployed — every merge builds and
|
||||||
|
publishes `:latest`, so whatever is on `main` is what is running. Day-to-day
|
||||||
|
work happens on `dev`, which publishes `:dev`.
|
||||||
|
|
||||||
|
For local development, the dev override handles everything:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker compose up -d # note: no -f, so the override applies
|
||||||
|
```
|
||||||
|
|
||||||
|
That builds the images from source, turns on DEBUG logging, and exposes
|
||||||
|
Postgres and Redis on the host. No `.env` required.
|
||||||
|
|
||||||
## Versions and tags
|
## Versions and tags
|
||||||
|
|
||||||
@@ -47,38 +263,10 @@ Five deployable pieces, built by `.forgejo/workflows/build.yml`:
|
|||||||
| --- | --- | --- | --- |
|
| --- | --- | --- | --- |
|
||||||
| **Web / workers** | `Dockerfile` | `fabledcurator` | Quart API + the built Vue SPA in one image. `entrypoint.sh` picks the role: `web`, `worker`, `scheduler`. The `maintenance-long` service is a second `worker` pinned to the long-running maintenance queue. |
|
| **Web / workers** | `Dockerfile` | `fabledcurator` | Quart API + the built Vue SPA in one image. `entrypoint.sh` picks the role: `web`, `worker`, `scheduler`. The `maintenance-long` service is a second `worker` pinned to the long-running maintenance queue. |
|
||||||
| **ML worker** | `Dockerfile.ml` | `fabledcurator-ml` | Same app, plus `requirements-ml.txt` — tagging and embedding models that run in-container. |
|
| **ML worker** | `Dockerfile.ml` | `fabledcurator-ml` | Same app, plus `requirements-ml.txt` — tagging and embedding models that run in-container. |
|
||||||
| **GPU agent** | `agent/Dockerfile` | `fabledcurator-agent` | Optional desktop-GPU worker (`agent/`). Leases jobs over **HTTP only** — never touches the database or Redis. Run it for a burst, stop it to reclaim the card. See `agent/README.md`. |
|
| **GPU agent** | `agent/Dockerfile` | `fabledcurator-agent` | Optional desktop-GPU worker (`agent/`). Leases jobs over **HTTP only** — never touches the database or Redis. See `agent/README.md`. |
|
||||||
| **Firefox extension** | `extension/` | signed XPI | MV3 extension: pushes platform session cookies into FC and adds a creator as a Source in one click. AMO-signed on both `dev` and `main` (one signature per extension change, shared by the two channels), bundled into that channel's web image and served from Settings → Maintenance. See `extension/README.md`. |
|
| **Firefox extension** | `extension/` | signed XPI | MV3 extension: pushes platform session cookies into FC and adds a creator as a Source in one click. AMO-signed on both `dev` and `main` (one signature per extension change, shared by the two channels), bundled into that channel's web image and served from Settings → Maintenance. See `extension/README.md`. |
|
||||||
| **Data** | — | `pgvector/pgvector:pg16`, `redis:7-alpine` | Postgres with pgvector for embeddings; Redis as the Celery broker. |
|
| **Data** | — | `pgvector/pgvector:pg16`, `redis:7-alpine` | Postgres with pgvector for embeddings; Redis as the Celery broker. |
|
||||||
|
|
||||||
## Quick start
|
|
||||||
|
|
||||||
For local development and testing, just:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
docker compose up -d
|
|
||||||
# UI: http://localhost:8080
|
|
||||||
```
|
|
||||||
|
|
||||||
That uses sane dev defaults baked into `docker-compose.yml` and the dev
|
|
||||||
override (`docker-compose.override.yml`, auto-merged) — local builds, DEBUG
|
|
||||||
logging, exposed Postgres + Redis ports on the host. No `.env` required.
|
|
||||||
|
|
||||||
For a production-like deployment, override the dev defaults via shell env
|
|
||||||
or a `.env` file (see `.env.example` for the variable names) and use:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
docker compose -f docker-compose.yml up -d
|
|
||||||
# (skips the override so containers pull registry images)
|
|
||||||
```
|
|
||||||
|
|
||||||
The GPU agent is deployed separately, on the machine with the card —
|
|
||||||
`agent/docker-compose.yml`, not this stack.
|
|
||||||
|
|
||||||
## Deployment posture
|
|
||||||
|
|
||||||
FabledCurator is designed to run inside a self-hosted homelab environment over plain HTTP. If you want TLS, terminate it at your reverse proxy. The app does not generate certificates, redirect to HTTPS, or set HSTS.
|
|
||||||
|
|
||||||
## CI / Forgejo setup
|
## CI / Forgejo setup
|
||||||
|
|
||||||
Four workflows: `ci.yml` (lint, extension-version check, backend unit tests,
|
Four workflows: `ci.yml` (lint, extension-version check, backend unit tests,
|
||||||
@@ -108,6 +296,29 @@ source, so `main` finds `dev`'s signature already cached and makes no second AMO
|
|||||||
call. That cache is why signing must be one-shot — AMO rejects a re-signed
|
call. That cache is why signing must be one-shot — AMO rejects a re-signed
|
||||||
version.
|
version.
|
||||||
|
|
||||||
|
## History
|
||||||
|
|
||||||
|
FabledCurator combines what was
|
||||||
|
[ImageRepo](https://git.fabledsword.com/bvandeusen/ImageRepo) (gallery, ML,
|
||||||
|
importer) and
|
||||||
|
[GallerySubscriber](https://git.fabledsword.com/bvandeusen/GallerySubscriber)
|
||||||
|
(gallery-dl wrapper, subscriptions, credential capture) into a single product.
|
||||||
|
Both are superseded; neither is maintained.
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
Personal project; use at your own discretion.
|
**GNU Affero General Public License v3.0** — see [LICENSE](LICENSE).
|
||||||
|
|
||||||
|
You may run, study, modify and redistribute this software. The condition is
|
||||||
|
reciprocity: if you distribute a modified version, or **run one as a network
|
||||||
|
service that other people use**, you must offer those users the corresponding
|
||||||
|
source under the same licence. That second clause (AGPL §13) is the reason this
|
||||||
|
licence rather than the GPL — for a self-hosted web application, "distribution"
|
||||||
|
otherwise never happens, and the obligation would never bite.
|
||||||
|
|
||||||
|
Running an unmodified copy for yourself, your household or your organisation
|
||||||
|
carries no obligation at all. Neither does modifying it privately. The licence
|
||||||
|
asks something of you only when you hand your modified version to others.
|
||||||
|
|
||||||
|
Contributions ship under the same licence — see [CONTRIBUTING](CONTRIBUTING.md).
|
||||||
|
Security reports: [SECURITY.md](SECURITY.md).
|
||||||
|
|||||||
+77
@@ -0,0 +1,77 @@
|
|||||||
|
# Security Policy
|
||||||
|
|
||||||
|
## Reporting a vulnerability
|
||||||
|
|
||||||
|
**Please do not put vulnerability details in a public issue.**
|
||||||
|
|
||||||
|
This project has no private disclosure channel yet. Until it does, open an
|
||||||
|
issue on the repository that says only that you have a security report — no
|
||||||
|
reproduction steps, no affected endpoint, no payload — and a maintainer will
|
||||||
|
reply with a private contact to send the details to.
|
||||||
|
|
||||||
|
That is a deliberately awkward first step, and it exists because the
|
||||||
|
alternative is worse: an issue tracker is public the moment it is written to,
|
||||||
|
and every self-hosted instance stays vulnerable until its operator has had a
|
||||||
|
chance to update.
|
||||||
|
|
||||||
|
Please include, once you have a private channel:
|
||||||
|
|
||||||
|
- what an attacker can do, and what access they need to start
|
||||||
|
- the version or commit you tested
|
||||||
|
- reproduction steps
|
||||||
|
|
||||||
|
## Scope — what this software actually handles
|
||||||
|
|
||||||
|
FabledCurator is self-hosted and holds things worth stating plainly, because
|
||||||
|
they shape what counts as a serious bug here:
|
||||||
|
|
||||||
|
- **Platform credentials.** The app captures and stores session cookies for
|
||||||
|
third-party subscription sites (Patreon, SubscribeStar, Pixiv) so it can
|
||||||
|
download on the operator's behalf. These are live credentials for accounts
|
||||||
|
that usually carry a payment method. Anything that discloses them, decrypts
|
||||||
|
them, or lets one user of a shared instance read another's is high severity.
|
||||||
|
- **An extension API key.** The Firefox extension authenticates to the backend
|
||||||
|
with a shared key. Anything that leaks it or lets it be bypassed is a way in.
|
||||||
|
- **No authentication of its own.** This is the most important thing on this
|
||||||
|
page. FabledCurator has no login, no user accounts and no permission model —
|
||||||
|
there is no `User` table and no session auth anywhere in the backend. Every
|
||||||
|
HTTP client that can reach the port is the administrator, with full read and
|
||||||
|
write access to everything above, including the stored platform credentials.
|
||||||
|
Access control is entirely the operator's job, done at the network layer.
|
||||||
|
Reports that an unauthenticated caller can reach an endpoint are therefore
|
||||||
|
describing the design; reports that something *crosses the network boundary
|
||||||
|
the operator drew* — an SSRF, a request forgery that rides a browser the
|
||||||
|
operator already has open, a path that leaks state to an origin the operator
|
||||||
|
did not authorise — are in scope and are serious.
|
||||||
|
- **Arbitrary media from the internet.** Downloaded files are decoded, hashed,
|
||||||
|
thumbnailed and fed to ML models. Anything that turns a hostile file into
|
||||||
|
code execution is in scope.
|
||||||
|
|
||||||
|
## Deployment posture — read this before reporting
|
||||||
|
|
||||||
|
FabledCurator is designed to run **inside a private network, over plain HTTP,
|
||||||
|
reachable only by its operator**. It does not terminate TLS, redirect to
|
||||||
|
HTTPS, or set HSTS; if you want transport security, terminate it at your
|
||||||
|
reverse proxy. It also does not authenticate anyone — see above. These are
|
||||||
|
documented design decisions, not oversights.
|
||||||
|
|
||||||
|
Putting this on the public internet, with or without TLS, hands whoever finds
|
||||||
|
it your Patreon, SubscribeStar and Pixiv sessions. A reverse proxy that adds
|
||||||
|
TLS but not an authentication layer does not change that.
|
||||||
|
|
||||||
|
Reports that reduce to "the application is served over HTTP", "there is no
|
||||||
|
HSTS header", or "the API needs no credentials" describe those decisions
|
||||||
|
rather than vulnerabilities. Reports that the operator can cause the software
|
||||||
|
to do something destructive are usually also by design — the operator is the
|
||||||
|
administrator of their own instance.
|
||||||
|
|
||||||
|
What remains in scope is everything that crosses a boundary the software is
|
||||||
|
actually supposed to hold: between untrusted downloaded content and the host,
|
||||||
|
between a third-party origin and an operator's open browser session, and
|
||||||
|
between the credentials at rest and anything that is not the operator.
|
||||||
|
|
||||||
|
## Supported versions
|
||||||
|
|
||||||
|
Fixes land on the `main` branch and reach the `:latest` image. There are no
|
||||||
|
maintained release branches — the supported version is the current one, and
|
||||||
|
the remedy for a security issue is to update.
|
||||||
@@ -1,78 +1,107 @@
|
|||||||
"""Collapsed baseline — the whole schema in one revision.
|
"""The whole schema, in one migration.
|
||||||
|
|
||||||
Replaces revisions 0001..0087, which narrated the build-out of this project
|
This replaces alembic revisions 0001..0089 — the entire build-out of the
|
||||||
and were deleted in milestone 328 step 1. A new install creates the schema in
|
project, 89 files and ~6,000 lines that a new installation used to replay in
|
||||||
one step instead of replaying that history.
|
order to arrive at a schema this file creates in one pass. Nothing about the
|
||||||
|
resulting database changes; what goes away is the requirement that a stranger
|
||||||
|
re-run our development history to get it.
|
||||||
|
|
||||||
WHY THE REVISION ID IS "0087" AND NOT "0001"
|
## Why the revision id is 0089
|
||||||
--------------------------------------------
|
|
||||||
It is deliberately the id of the LAST revision this baseline collapses, so an
|
|
||||||
existing database needs no intervention at all:
|
|
||||||
|
|
||||||
* a fresh install finds current=none, head=0087, runs this file once, and
|
`revision = "0089"` and `down_revision = None` are both deliberate, and the
|
||||||
ends stamped at 0087.
|
combination is the entire migration strategy for existing installations.
|
||||||
* an existing install is ALREADY at 0087, so `alembic upgrade head` finds
|
|
||||||
current == head and does nothing.
|
|
||||||
|
|
||||||
The alternative — numbering this 0001 and stamping every existing database —
|
An already-deployed database has `alembic_version = '0089'`, because it ran the
|
||||||
means running `alembic stamp` against live data, and stamp VALIDATES NOTHING.
|
real 0089. This file claims that same id, so alembic reads the version table,
|
||||||
It writes a version string whether or not the schema actually matches, so a
|
sees head already reached, and does nothing at all. No stamp is needed — which
|
||||||
wrong baseline would be discovered later, by the next real migration, with no
|
matters because `alembic stamp` writes a version string without validating
|
||||||
clean way back. Keeping the id removes that operation instead of making it
|
anything about the schema it is writing it against, and a stamp that is wrong
|
||||||
safe. Future revisions continue at 0088.
|
is indistinguishable from one that is right until the next migration fails.
|
||||||
|
|
||||||
The one case this makes worse, and it fails LOUDLY rather than silently: a
|
An empty database has no version row, so alembic runs this file and then
|
||||||
database still sitting between 0001 and 0086 (i.e. never upgraded to head)
|
records `0089`. Both paths converge on the same schema and the same version,
|
||||||
cannot be located in this chain and errors out. Upgrade to 0087 on a
|
and neither requires anyone to assert anything by hand.
|
||||||
pre-squash build first, then take this one.
|
|
||||||
|
|
||||||
WHAT IS HAND-WRITTEN HERE
|
The next migration written after this one is `0090`, exactly as it would have
|
||||||
-------------------------
|
been. The numbering is continuous across the collapse on purpose.
|
||||||
Most of this file is `alembic revision --autogenerate` output, but four
|
|
||||||
things are NOT in SQLAlchemy metadata and the generator cannot produce them.
|
|
||||||
Each fails differently, and none of them fail at generation time:
|
|
||||||
|
|
||||||
1. CREATE EXTENSION vector (was 0001) — without it the VECTOR
|
## What was added to the generated output, and why
|
||||||
columns below cannot be created at all.
|
|
||||||
2. CREATE EXTENSION tsm_system_rows (was 0004) — used by the random-sample
|
|
||||||
query path; its absence surfaces only when that query runs.
|
|
||||||
3. The HNSW index on image_record.siglip_embedding (was 0036). Raw SQL
|
|
||||||
because alembic's create_index cannot express `USING hnsw (...
|
|
||||||
vector_cosine_ops)`. Its absence is the quietest failure of the four:
|
|
||||||
everything works, similarity search just stops using an index.
|
|
||||||
4. `import pgvector.sqlalchemy.vector`. Autogenerate EMITS references to
|
|
||||||
pgvector.sqlalchemy.vector.VECTOR but does not add the import, so the
|
|
||||||
generated file dies with NameError on first run.
|
|
||||||
|
|
||||||
The acceptance test for this file is not that it reads correctly — it is
|
`alembic revision --autogenerate` produced almost all of this from the models,
|
||||||
`.forgejo/workflows/baseline.yml`, which builds a database from the old
|
which is only true because #3275 first made the models actually describe the
|
||||||
0001..0087 chain (read out of git) and one from this file, and diffs
|
schema. Before that reconciliation the generator silently omitted eleven
|
||||||
pg_dump --schema-only output. That is what proves nothing was missed.
|
indexes and three uniqueness guarantees, and an earlier attempt at this squash
|
||||||
|
had to be reverted for exactly that reason.
|
||||||
|
|
||||||
Revision ID: 0087
|
Four things still had to be added by hand, because they are not in the models:
|
||||||
|
|
||||||
|
1. **`CREATE EXTENSION vector`** (from 0001) and **`tsm_system_rows`** (0004).
|
||||||
|
Extensions are database objects, not table metadata, so no model can carry
|
||||||
|
them. `IF NOT EXISTS` because a re-run must not fail.
|
||||||
|
|
||||||
|
2. **Three seed inserts** — the two settings singletons (0002, 0003) and the
|
||||||
|
three hygiene system tags (0075). Some migrations did not only build schema;
|
||||||
|
they inserted rows the product needs in order to function, and nothing in
|
||||||
|
the application ever creates them. Every consumer reads them with
|
||||||
|
`scalar_one()`, which RAISES `NoResultFound` on an empty result rather than
|
||||||
|
returning None, so their absence is a crash and not a degradation.
|
||||||
|
|
||||||
|
Distinguishing these from the other data statements in the chain is the
|
||||||
|
whole trick, and the rule turns out to be mechanical:
|
||||||
|
|
||||||
|
* `INSERT ... VALUES (...)` with literal values is a SEED. It creates
|
||||||
|
something the product ships. It must be carried.
|
||||||
|
* `INSERT ... SELECT ... FROM <table>` is a BACKFILL. It derives rows
|
||||||
|
from rows that already exist, so on an empty database it inserts
|
||||||
|
nothing and carrying it would be pointless. 0034 (artist_visit), 0040
|
||||||
|
and 0047 (series_chapter) are all of this shape and are correctly
|
||||||
|
absent here.
|
||||||
|
|
||||||
|
This category is invisible to every automated check this project has:
|
||||||
|
`baseline.yml` compares SCHEMA, and a baseline missing all three seeds still
|
||||||
|
produces a byte-identical schema and a perfectly green diff. What caught the
|
||||||
|
system tags was the integration suite — 36 tests failing on
|
||||||
|
`NoResultFound` — after a first version of this file shipped with only the
|
||||||
|
two settings rows. A first-run check against the real application is the
|
||||||
|
only thing that finds this class of defect.
|
||||||
|
|
||||||
|
3. **The `pgvector` import.** Autogenerate emits qualified
|
||||||
|
`pgvector.sqlalchemy.vector.VECTOR(...)` references without importing the
|
||||||
|
package, so the file it writes cannot execute — `NameError: name 'pgvector'
|
||||||
|
is not defined`, observed on run 4988.
|
||||||
|
|
||||||
|
The other data statements in the old chain were deliberately NOT carried over.
|
||||||
|
0023's `DELETE FROM tag WHERE kind IN (...)`, and 0047's `series_page` /
|
||||||
|
`series_chapter` deletes, are historical cleanups that operate on rows an empty
|
||||||
|
database does not have.
|
||||||
|
|
||||||
|
## Downgrade
|
||||||
|
|
||||||
|
There is none. A baseline's downgrade would be "drop the entire schema", which
|
||||||
|
is not a migration but a data-loss event wearing one as a disguise. Restore
|
||||||
|
from a backup instead — that is what backup_run exists for.
|
||||||
|
|
||||||
|
Revision ID: 0089
|
||||||
Revises:
|
Revises:
|
||||||
Create Date: 2026-08-30
|
Create Date: 2026-09-01
|
||||||
|
|
||||||
"""
|
"""
|
||||||
from typing import Sequence, Union
|
from typing import Sequence, Union
|
||||||
|
|
||||||
from alembic import op
|
from alembic import op
|
||||||
import sqlalchemy as sa
|
import sqlalchemy as sa
|
||||||
|
import pgvector.sqlalchemy.vector
|
||||||
from sqlalchemy.dialects import postgresql
|
from sqlalchemy.dialects import postgresql
|
||||||
|
|
||||||
# Autogenerate references pgvector.sqlalchemy.vector.VECTOR without importing
|
revision: str = "0089"
|
||||||
# it. Item 4 above.
|
|
||||||
import pgvector.sqlalchemy.vector
|
|
||||||
|
|
||||||
revision: str = "0087"
|
|
||||||
down_revision: Union[str, None] = None
|
down_revision: Union[str, None] = None
|
||||||
branch_labels: Union[str, Sequence[str], None] = None
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
depends_on: Union[str, Sequence[str], None] = None
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
def upgrade() -> None:
|
def upgrade() -> None:
|
||||||
# Extensions FIRST: the VECTOR columns below cannot be created without
|
# Extensions first: image_record.siglip_embedding is a vector column and
|
||||||
# `vector`, so ordering here is load-bearing, not tidiness.
|
# cannot be created before the type exists. From 0001 and 0004.
|
||||||
op.execute("CREATE EXTENSION IF NOT EXISTS vector")
|
op.execute("CREATE EXTENSION IF NOT EXISTS vector")
|
||||||
op.execute("CREATE EXTENSION IF NOT EXISTS tsm_system_rows")
|
op.execute("CREATE EXTENSION IF NOT EXISTS tsm_system_rows")
|
||||||
|
|
||||||
@@ -87,8 +116,8 @@ def upgrade() -> None:
|
|||||||
sa.Column('name', sa.String(length=255), nullable=False),
|
sa.Column('name', sa.String(length=255), nullable=False),
|
||||||
sa.Column('slug', sa.String(length=255), nullable=False),
|
sa.Column('slug', sa.String(length=255), nullable=False),
|
||||||
sa.Column('notes', sa.Text(), nullable=True),
|
sa.Column('notes', sa.Text(), nullable=True),
|
||||||
sa.Column('is_subscription', sa.Boolean(), nullable=False),
|
sa.Column('is_subscription', sa.Boolean(), server_default='false', nullable=False),
|
||||||
sa.Column('auto_check', sa.Boolean(), nullable=False),
|
sa.Column('auto_check', sa.Boolean(), server_default='true', nullable=False),
|
||||||
sa.Column('check_interval_seconds', sa.Integer(), nullable=True),
|
sa.Column('check_interval_seconds', sa.Integer(), nullable=True),
|
||||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_artist')),
|
sa.PrimaryKeyConstraint('id', name=op.f('pk_artist')),
|
||||||
@@ -97,7 +126,7 @@ def upgrade() -> None:
|
|||||||
op.create_table('backup_run',
|
op.create_table('backup_run',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('kind', sa.String(length=16), nullable=False),
|
sa.Column('kind', sa.String(length=16), nullable=False),
|
||||||
sa.Column('status', sa.String(length=16), nullable=False),
|
sa.Column('status', sa.String(length=16), server_default='pending', nullable=False),
|
||||||
sa.Column('tag', sa.String(length=64), nullable=True),
|
sa.Column('tag', sa.String(length=64), nullable=True),
|
||||||
sa.Column('triggered_by', sa.String(length=32), nullable=False),
|
sa.Column('triggered_by', sa.String(length=32), nullable=False),
|
||||||
sa.Column('started_at', sa.DateTime(timezone=True), nullable=False),
|
sa.Column('started_at', sa.DateTime(timezone=True), nullable=False),
|
||||||
@@ -112,10 +141,12 @@ def upgrade() -> None:
|
|||||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_backup_run'))
|
sa.PrimaryKeyConstraint('id', name=op.f('pk_backup_run'))
|
||||||
)
|
)
|
||||||
op.create_index(op.f('ix_backup_run_finished_at'), 'backup_run', ['finished_at'], unique=False)
|
op.create_index(op.f('ix_backup_run_finished_at'), 'backup_run', ['finished_at'], unique=False)
|
||||||
op.create_index(op.f('ix_backup_run_kind'), 'backup_run', ['kind'], unique=False)
|
op.create_index('ix_backup_run_kind_started', 'backup_run', ['kind', sa.literal_column('started_at DESC')], unique=False)
|
||||||
|
op.create_index(op.f('ix_backup_run_restored_from_id'), 'backup_run', ['restored_from_id'], unique=False)
|
||||||
op.create_index(op.f('ix_backup_run_started_at'), 'backup_run', ['started_at'], unique=False)
|
op.create_index(op.f('ix_backup_run_started_at'), 'backup_run', ['started_at'], unique=False)
|
||||||
op.create_index(op.f('ix_backup_run_status'), 'backup_run', ['status'], unique=False)
|
op.create_index('ix_backup_run_status_finished', 'backup_run', ['status', sa.literal_column('finished_at DESC')], unique=False)
|
||||||
op.create_index(op.f('ix_backup_run_tag'), 'backup_run', ['tag'], unique=False)
|
op.create_index(op.f('ix_backup_run_tag'), 'backup_run', ['tag'], unique=False)
|
||||||
|
op.create_index('ix_backup_run_tag_partial', 'backup_run', ['tag'], unique=False, postgresql_where=sa.text('tag IS NOT NULL'))
|
||||||
op.create_table('credential',
|
op.create_table('credential',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('platform', sa.String(length=64), nullable=False),
|
sa.Column('platform', sa.String(length=64), nullable=False),
|
||||||
@@ -129,9 +160,9 @@ def upgrade() -> None:
|
|||||||
)
|
)
|
||||||
op.create_table('head_auto_apply_run',
|
op.create_table('head_auto_apply_run',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('dry_run', sa.Boolean(), nullable=False),
|
sa.Column('dry_run', sa.Boolean(), server_default='false', nullable=False),
|
||||||
sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False),
|
sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False),
|
||||||
sa.Column('status', sa.String(length=16), nullable=False),
|
sa.Column('status', sa.String(length=16), server_default='running', nullable=False),
|
||||||
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||||
sa.Column('n_applied', sa.Integer(), nullable=True),
|
sa.Column('n_applied', sa.Integer(), nullable=True),
|
||||||
@@ -144,7 +175,7 @@ def upgrade() -> None:
|
|||||||
op.create_table('head_training_run',
|
op.create_table('head_training_run',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False),
|
sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False),
|
||||||
sa.Column('status', sa.String(length=16), nullable=False),
|
sa.Column('status', sa.String(length=16), server_default='running', nullable=False),
|
||||||
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||||
sa.Column('n_trained', sa.Integer(), nullable=True),
|
sa.Column('n_trained', sa.Integer(), nullable=True),
|
||||||
@@ -161,38 +192,38 @@ def upgrade() -> None:
|
|||||||
sa.Column('scan_mode', sa.String(length=16), nullable=False),
|
sa.Column('scan_mode', sa.String(length=16), nullable=False),
|
||||||
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||||
sa.Column('total_files', sa.Integer(), nullable=False),
|
sa.Column('total_files', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('imported', sa.Integer(), nullable=False),
|
sa.Column('imported', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('skipped', sa.Integer(), nullable=False),
|
sa.Column('skipped', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('failed', sa.Integer(), nullable=False),
|
sa.Column('failed', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('attachments', sa.Integer(), nullable=False),
|
sa.Column('attachments', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('refreshed', sa.Integer(), nullable=False),
|
sa.Column('refreshed', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('status', sa.String(length=16), nullable=False),
|
sa.Column('status', sa.String(length=16), server_default='running', nullable=False),
|
||||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_import_batch'))
|
sa.PrimaryKeyConstraint('id', name=op.f('pk_import_batch'))
|
||||||
)
|
)
|
||||||
op.create_index(op.f('ix_import_batch_status'), 'import_batch', ['status'], unique=False)
|
op.create_index(op.f('ix_import_batch_status'), 'import_batch', ['status'], unique=False)
|
||||||
op.create_table('import_settings',
|
op.create_table('import_settings',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('import_scan_path', sa.Text(), nullable=False),
|
sa.Column('import_scan_path', sa.Text(), server_default='/import', nullable=False),
|
||||||
sa.Column('min_width', sa.Integer(), nullable=False),
|
sa.Column('min_width', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('min_height', sa.Integer(), nullable=False),
|
sa.Column('min_height', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('skip_transparent', sa.Boolean(), nullable=False),
|
sa.Column('skip_transparent', sa.Boolean(), server_default='false', nullable=False),
|
||||||
sa.Column('transparency_threshold', sa.Float(), nullable=False),
|
sa.Column('transparency_threshold', sa.Float(), server_default='0.9', nullable=False),
|
||||||
sa.Column('skip_single_color', sa.Boolean(), nullable=False),
|
sa.Column('skip_single_color', sa.Boolean(), server_default='false', nullable=False),
|
||||||
sa.Column('single_color_threshold', sa.Float(), nullable=False),
|
sa.Column('single_color_threshold', sa.Float(), server_default='0.95', nullable=False),
|
||||||
sa.Column('single_color_tolerance', sa.Integer(), nullable=False),
|
sa.Column('single_color_tolerance', sa.Integer(), server_default='30', nullable=False),
|
||||||
sa.Column('phash_threshold', sa.Integer(), nullable=False),
|
sa.Column('phash_threshold', sa.Integer(), server_default='10', nullable=False),
|
||||||
sa.Column('download_rate_limit_seconds', sa.Float(), nullable=False),
|
sa.Column('download_rate_limit_seconds', sa.Float(), server_default='3', nullable=False),
|
||||||
sa.Column('download_validate_files', sa.Boolean(), nullable=False),
|
sa.Column('download_validate_files', sa.Boolean(), server_default='true', nullable=False),
|
||||||
sa.Column('download_schedule_default_seconds', sa.Integer(), nullable=False),
|
sa.Column('download_schedule_default_seconds', sa.Integer(), server_default='28800', nullable=False),
|
||||||
sa.Column('download_event_retention_days', sa.Integer(), nullable=False),
|
sa.Column('download_event_retention_days', sa.Integer(), server_default='90', nullable=False),
|
||||||
sa.Column('download_failure_warning_threshold', sa.Integer(), nullable=False),
|
sa.Column('download_failure_warning_threshold', sa.Integer(), server_default='5', nullable=False),
|
||||||
sa.Column('backup_db_nightly_enabled', sa.Boolean(), nullable=False),
|
sa.Column('backup_db_nightly_enabled', sa.Boolean(), server_default='false', nullable=False),
|
||||||
sa.Column('backup_db_nightly_hour_utc', sa.Integer(), nullable=False),
|
sa.Column('backup_db_nightly_hour_utc', sa.Integer(), server_default='3', nullable=False),
|
||||||
sa.Column('backup_db_keep_last_n', sa.Integer(), nullable=False),
|
sa.Column('backup_db_keep_last_n', sa.Integer(), server_default='14', nullable=False),
|
||||||
sa.Column('backup_images_keep_last_n', sa.Integer(), nullable=False),
|
sa.Column('backup_images_keep_last_n', sa.Integer(), server_default='3', nullable=False),
|
||||||
sa.Column('series_suggest_enabled', sa.Boolean(), nullable=False),
|
sa.Column('series_suggest_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||||
sa.Column('series_suggest_threshold', sa.Float(), nullable=False),
|
sa.Column('series_suggest_threshold', sa.Float(), server_default='0.5', nullable=False),
|
||||||
sa.Column('extdl_mega_enabled', sa.Boolean(), server_default='true', nullable=False),
|
sa.Column('extdl_mega_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||||
sa.Column('extdl_gdrive_enabled', sa.Boolean(), server_default='true', nullable=False),
|
sa.Column('extdl_gdrive_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||||
sa.Column('extdl_mediafire_enabled', sa.Boolean(), server_default='true', nullable=False),
|
sa.Column('extdl_mediafire_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||||
@@ -201,7 +232,7 @@ def upgrade() -> None:
|
|||||||
sa.Column('translation_enabled', sa.Boolean(), server_default='false', nullable=False),
|
sa.Column('translation_enabled', sa.Boolean(), server_default='false', nullable=False),
|
||||||
sa.Column('interpreter_base_url', sa.Text(), server_default='', nullable=False),
|
sa.Column('interpreter_base_url', sa.Text(), server_default='', nullable=False),
|
||||||
sa.Column('translation_target_lang', sa.Text(), server_default='en', nullable=False),
|
sa.Column('translation_target_lang', sa.Text(), server_default='en', nullable=False),
|
||||||
sa.Column('translation_min_confidence', sa.Float(), server_default='0.9', nullable=False),
|
sa.Column('translation_min_confidence', sa.Float(), server_default=sa.text('0.9'), nullable=False),
|
||||||
sa.Column('wip_title_tagging_enabled', sa.Boolean(), server_default='true', nullable=False),
|
sa.Column('wip_title_tagging_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||||
sa.Column('wip_soft_title_tagging_enabled', sa.Boolean(), server_default='false', nullable=False),
|
sa.Column('wip_soft_title_tagging_enabled', sa.Boolean(), server_default='false', nullable=False),
|
||||||
sa.CheckConstraint('id = 1', name=op.f('ck_import_settings_singleton')),
|
sa.CheckConstraint('id = 1', name=op.f('ck_import_settings_singleton')),
|
||||||
@@ -211,14 +242,14 @@ def upgrade() -> None:
|
|||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('rule', sa.String(length=32), nullable=False),
|
sa.Column('rule', sa.String(length=32), nullable=False),
|
||||||
sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False),
|
sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False),
|
||||||
sa.Column('status', sa.String(length=16), nullable=False),
|
sa.Column('status', sa.String(length=16), server_default='running', nullable=False),
|
||||||
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||||
sa.Column('scanned_count', sa.Integer(), nullable=False),
|
sa.Column('scanned_count', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('matched_count', sa.Integer(), nullable=False),
|
sa.Column('matched_count', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('matched_ids', postgresql.JSONB(astext_type=sa.Text()), nullable=False),
|
sa.Column('matched_ids', postgresql.JSONB(astext_type=sa.Text()), server_default=sa.text("'[]'::jsonb"), nullable=False),
|
||||||
sa.Column('error', sa.Text(), nullable=True),
|
sa.Column('error', sa.Text(), nullable=True),
|
||||||
sa.Column('resume_after_id', sa.Integer(), nullable=False),
|
sa.Column('resume_after_id', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('last_progress_at', sa.DateTime(timezone=True), nullable=True),
|
sa.Column('last_progress_at', sa.DateTime(timezone=True), nullable=True),
|
||||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_library_audit_run'))
|
sa.PrimaryKeyConstraint('id', name=op.f('pk_library_audit_run'))
|
||||||
)
|
)
|
||||||
@@ -226,40 +257,40 @@ def upgrade() -> None:
|
|||||||
op.create_index(op.f('ix_library_audit_run_status'), 'library_audit_run', ['status'], unique=False)
|
op.create_index(op.f('ix_library_audit_run_status'), 'library_audit_run', ['status'], unique=False)
|
||||||
op.create_table('ml_settings',
|
op.create_table('ml_settings',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('cpu_embed_enabled', sa.Boolean(), nullable=False),
|
sa.Column('cpu_embed_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||||
sa.Column('video_frame_interval_seconds', sa.Float(), nullable=False),
|
sa.Column('video_frame_interval_seconds', sa.Float(), server_default='4', nullable=False),
|
||||||
sa.Column('video_max_frames', sa.Integer(), nullable=False),
|
sa.Column('video_max_frames', sa.Integer(), server_default='64', nullable=False),
|
||||||
sa.Column('head_min_positives', sa.Integer(), nullable=False),
|
sa.Column('head_min_positives', sa.Integer(), server_default='8', nullable=False),
|
||||||
sa.Column('head_auto_apply_precision', sa.Float(), nullable=False),
|
sa.Column('head_auto_apply_precision', sa.Float(), server_default='0.97', nullable=False),
|
||||||
sa.Column('head_auto_apply_enabled', sa.Boolean(), nullable=False),
|
sa.Column('head_auto_apply_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||||
sa.Column('head_auto_apply_min_positives', sa.Integer(), nullable=False),
|
sa.Column('head_auto_apply_min_positives', sa.Integer(), server_default='30', nullable=False),
|
||||||
sa.Column('ccip_match_threshold', sa.Float(), nullable=False),
|
sa.Column('ccip_match_threshold', sa.Float(), server_default='0.85', nullable=False),
|
||||||
sa.Column('ccip_auto_apply_enabled', sa.Boolean(), nullable=False),
|
sa.Column('ccip_auto_apply_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||||
sa.Column('ccip_auto_apply_threshold', sa.Float(), nullable=False),
|
sa.Column('ccip_auto_apply_threshold', sa.Float(), server_default='0.92', nullable=False),
|
||||||
sa.Column('presentation_auto_apply_enabled', sa.Boolean(), nullable=False),
|
sa.Column('presentation_auto_apply_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||||
sa.Column('presentation_auto_apply_threshold', sa.Float(), nullable=False),
|
sa.Column('presentation_auto_apply_threshold', sa.Float(), server_default=sa.text('0.90'), nullable=False),
|
||||||
sa.Column('presentation_conflict_threshold', sa.Float(), nullable=False),
|
sa.Column('presentation_conflict_threshold', sa.Float(), server_default=sa.text('0.50'), nullable=False),
|
||||||
sa.Column('process_auto_apply_enabled', sa.Boolean(), nullable=False),
|
sa.Column('process_auto_apply_enabled', sa.Boolean(), server_default='false', nullable=False),
|
||||||
sa.Column('process_auto_apply_threshold', sa.Float(), nullable=False),
|
sa.Column('process_auto_apply_threshold', sa.Float(), server_default='0.90', nullable=False),
|
||||||
sa.Column('process_conflict_threshold', sa.Float(), nullable=False),
|
sa.Column('process_conflict_threshold', sa.Float(), server_default='0.50', nullable=False),
|
||||||
sa.Column('embedder_model_version', sa.String(length=128), nullable=False),
|
sa.Column('embedder_model_version', sa.String(length=128), server_default='siglip2-so400m-patch16-512', nullable=False),
|
||||||
sa.Column('embedder_model_name', sa.String(length=128), nullable=False),
|
sa.Column('embedder_model_name', sa.String(length=128), server_default='google/siglip2-so400m-patch16-512', nullable=False),
|
||||||
sa.Column('detector_person_enabled', sa.Boolean(), nullable=False),
|
sa.Column('detector_person_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||||
sa.Column('detector_person_weights', sa.String(length=512), nullable=False),
|
sa.Column('detector_person_weights', sa.String(length=512), server_default='yolo11n.pt', nullable=False),
|
||||||
sa.Column('detector_person_conf', sa.Float(), nullable=False),
|
sa.Column('detector_person_conf', sa.Float(), server_default=sa.text('0.35'), nullable=False),
|
||||||
sa.Column('detector_anatomy_enabled', sa.Boolean(), nullable=False),
|
sa.Column('detector_anatomy_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||||
sa.Column('detector_anatomy_weights', sa.String(length=512), nullable=False),
|
sa.Column('detector_anatomy_weights', sa.String(length=512), server_default='https://github.com/aperveyev/booru_yolo/raw/main/models/yolov11m_aa22.pt', nullable=False),
|
||||||
sa.Column('detector_anatomy_conf', sa.Float(), nullable=False),
|
sa.Column('detector_anatomy_conf', sa.Float(), server_default=sa.text('0.30'), nullable=False),
|
||||||
sa.Column('detector_panel_enabled', sa.Boolean(), nullable=False),
|
sa.Column('detector_panel_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||||
sa.Column('detector_panel_weights', sa.String(length=512), nullable=False),
|
sa.Column('detector_panel_weights', sa.String(length=512), server_default='mosesb/best-comic-panel-detection::best.pt', nullable=False),
|
||||||
sa.Column('detector_panel_conf', sa.Float(), nullable=False),
|
sa.Column('detector_panel_conf', sa.Float(), server_default=sa.text('0.30'), nullable=False),
|
||||||
sa.Column('detector_max_figures', sa.Integer(), nullable=False),
|
sa.Column('detector_max_figures', sa.Integer(), server_default='8', nullable=False),
|
||||||
sa.Column('detector_max_components', sa.Integer(), nullable=False),
|
sa.Column('detector_max_components', sa.Integer(), server_default='8', nullable=False),
|
||||||
sa.Column('detector_max_panels', sa.Integer(), nullable=False),
|
sa.Column('detector_max_panels', sa.Integer(), server_default='8', nullable=False),
|
||||||
sa.Column('detector_max_regions', sa.Integer(), nullable=False),
|
sa.Column('detector_max_regions', sa.Integer(), server_default='128', nullable=False),
|
||||||
sa.Column('detector_dedupe_iou', sa.Float(), nullable=False),
|
sa.Column('detector_dedupe_iou', sa.Float(), server_default=sa.text('0.85'), nullable=False),
|
||||||
sa.Column('ccip_ref_signature', sa.String(length=128), nullable=True),
|
sa.Column('ccip_ref_signature', sa.String(length=128), nullable=True),
|
||||||
sa.Column('ccip_prototype_cap', sa.Integer(), nullable=False),
|
sa.Column('ccip_prototype_cap', sa.Integer(), server_default='64', nullable=False),
|
||||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.CheckConstraint('id = 1', name=op.f('ck_ml_settings_singleton')),
|
sa.CheckConstraint('id = 1', name=op.f('ck_ml_settings_singleton')),
|
||||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_ml_settings'))
|
sa.PrimaryKeyConstraint('id', name=op.f('pk_ml_settings'))
|
||||||
@@ -267,15 +298,16 @@ def upgrade() -> None:
|
|||||||
op.create_table('tag',
|
op.create_table('tag',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('name', sa.String(length=255), nullable=False),
|
sa.Column('name', sa.String(length=255), nullable=False),
|
||||||
sa.Column('kind', sa.Enum('artist', 'character', 'fandom', 'general', 'series', 'archive', 'post', name='tag_kind'), nullable=False),
|
sa.Column('kind', sa.Enum('artist', 'character', 'fandom', 'general', 'series', 'archive', 'post', name='tag_kind'), server_default='general', nullable=False),
|
||||||
sa.Column('fandom_id', sa.Integer(), nullable=True),
|
sa.Column('fandom_id', sa.Integer(), nullable=True),
|
||||||
sa.Column('is_system', sa.Boolean(), server_default=sa.text('false'), nullable=False),
|
sa.Column('is_system', sa.Boolean(), server_default=sa.text('false'), nullable=False),
|
||||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.CheckConstraint("(fandom_id IS NULL) OR (kind = 'character')", name=op.f('ck_tag_ck_tag_fandom_requires_character')),
|
sa.CheckConstraint("(fandom_id IS NULL) OR (kind = 'character')", name=op.f('ck_tag_fandom_requires_character')),
|
||||||
sa.ForeignKeyConstraint(['fandom_id'], ['tag.id'], name=op.f('fk_tag_fandom_id_tag'), ondelete='SET NULL'),
|
sa.ForeignKeyConstraint(['fandom_id'], ['tag.id'], name=op.f('fk_tag_fandom_id_tag'), ondelete='SET NULL'),
|
||||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_tag'))
|
sa.PrimaryKeyConstraint('id', name=op.f('pk_tag'))
|
||||||
)
|
)
|
||||||
op.create_index(op.f('ix_tag_fandom_id'), 'tag', ['fandom_id'], unique=False)
|
op.create_index(op.f('ix_tag_fandom_id'), 'tag', ['fandom_id'], unique=False)
|
||||||
|
op.create_index('uq_tag_name_kind_fandom', 'tag', ['name', 'kind', sa.literal_column('COALESCE(fandom_id, 0)')], unique=True)
|
||||||
op.create_table('task_run',
|
op.create_table('task_run',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('celery_task_id', sa.String(length=64), nullable=False),
|
sa.Column('celery_task_id', sa.String(length=64), nullable=False),
|
||||||
@@ -285,7 +317,7 @@ def upgrade() -> None:
|
|||||||
sa.Column('started_at', sa.DateTime(timezone=True), nullable=False),
|
sa.Column('started_at', sa.DateTime(timezone=True), nullable=False),
|
||||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||||
sa.Column('duration_ms', sa.Integer(), nullable=True),
|
sa.Column('duration_ms', sa.Integer(), nullable=True),
|
||||||
sa.Column('status', sa.String(length=16), nullable=False),
|
sa.Column('status', sa.String(length=16), server_default='running', nullable=False),
|
||||||
sa.Column('error_type', sa.String(length=128), nullable=True),
|
sa.Column('error_type', sa.String(length=128), nullable=True),
|
||||||
sa.Column('error_message', sa.Text(), nullable=True),
|
sa.Column('error_message', sa.Text(), nullable=True),
|
||||||
sa.Column('retry_count', sa.Integer(), nullable=True),
|
sa.Column('retry_count', sa.Integer(), nullable=True),
|
||||||
@@ -295,10 +327,10 @@ def upgrade() -> None:
|
|||||||
)
|
)
|
||||||
op.create_index(op.f('ix_task_run_celery_task_id'), 'task_run', ['celery_task_id'], unique=False)
|
op.create_index(op.f('ix_task_run_celery_task_id'), 'task_run', ['celery_task_id'], unique=False)
|
||||||
op.create_index(op.f('ix_task_run_finished_at'), 'task_run', ['finished_at'], unique=False)
|
op.create_index(op.f('ix_task_run_finished_at'), 'task_run', ['finished_at'], unique=False)
|
||||||
op.create_index(op.f('ix_task_run_queue'), 'task_run', ['queue'], unique=False)
|
op.create_index('ix_task_run_name_started', 'task_run', ['task_name', sa.literal_column('started_at DESC')], unique=False)
|
||||||
|
op.create_index('ix_task_run_queue_started', 'task_run', ['queue', sa.literal_column('started_at DESC')], unique=False)
|
||||||
op.create_index(op.f('ix_task_run_started_at'), 'task_run', ['started_at'], unique=False)
|
op.create_index(op.f('ix_task_run_started_at'), 'task_run', ['started_at'], unique=False)
|
||||||
op.create_index(op.f('ix_task_run_status'), 'task_run', ['status'], unique=False)
|
op.create_index('ix_task_run_status_started', 'task_run', ['status', sa.literal_column('started_at DESC')], unique=False)
|
||||||
op.create_index(op.f('ix_task_run_task_name'), 'task_run', ['task_name'], unique=False)
|
|
||||||
op.create_table('artist_visit',
|
op.create_table('artist_visit',
|
||||||
sa.Column('artist_id', sa.Integer(), nullable=False),
|
sa.Column('artist_id', sa.Integer(), nullable=False),
|
||||||
sa.Column('last_viewed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('last_viewed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
@@ -314,20 +346,20 @@ def upgrade() -> None:
|
|||||||
)
|
)
|
||||||
op.create_table('head_metric',
|
op.create_table('head_metric',
|
||||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||||
sa.Column('n_misfires', sa.Integer(), nullable=False),
|
sa.Column('n_misfires', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('n_underfires', sa.Integer(), nullable=False),
|
sa.Column('n_underfires', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_head_metric_tag_id_tag'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_head_metric_tag_id_tag'), ondelete='CASCADE'),
|
||||||
sa.PrimaryKeyConstraint('tag_id', name=op.f('pk_head_metric'))
|
sa.PrimaryKeyConstraint('tag_id', name=op.f('pk_head_metric'))
|
||||||
)
|
)
|
||||||
op.create_table('head_metrics_snapshot',
|
op.create_table('head_metrics_snapshot',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
sa.Column('tag_id', sa.Integer(), nullable=True),
|
||||||
sa.Column('name', sa.String(length=255), nullable=False),
|
sa.Column('name', sa.String(length=255), nullable=False),
|
||||||
sa.Column('snapshot_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('snapshot_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.Column('n_auto_applied', sa.Integer(), nullable=False),
|
sa.Column('n_auto_applied', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('n_misfires', sa.Integer(), nullable=False),
|
sa.Column('n_misfires', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('n_underfires', sa.Integer(), nullable=False),
|
sa.Column('n_underfires', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('ap', sa.Float(), nullable=True),
|
sa.Column('ap', sa.Float(), nullable=True),
|
||||||
sa.Column('precision_cv', sa.Float(), nullable=True),
|
sa.Column('precision_cv', sa.Float(), nullable=True),
|
||||||
sa.Column('recall', sa.Float(), nullable=True),
|
sa.Column('recall', sa.Float(), nullable=True),
|
||||||
@@ -342,16 +374,17 @@ def upgrade() -> None:
|
|||||||
sa.Column('artist_id', sa.Integer(), nullable=False),
|
sa.Column('artist_id', sa.Integer(), nullable=False),
|
||||||
sa.Column('platform', sa.String(length=64), nullable=False),
|
sa.Column('platform', sa.String(length=64), nullable=False),
|
||||||
sa.Column('url', sa.Text(), nullable=False),
|
sa.Column('url', sa.Text(), nullable=False),
|
||||||
sa.Column('enabled', sa.Boolean(), nullable=False),
|
sa.Column('enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||||
sa.Column('config_overrides', sa.JSON(), nullable=True),
|
sa.Column('config_overrides', sa.JSON(), nullable=True),
|
||||||
sa.Column('last_checked_at', sa.DateTime(timezone=True), nullable=True),
|
sa.Column('last_checked_at', sa.DateTime(timezone=True), nullable=True),
|
||||||
sa.Column('last_error', sa.Text(), nullable=True),
|
sa.Column('last_error', sa.Text(), nullable=True),
|
||||||
sa.Column('error_type', sa.String(length=32), nullable=True),
|
sa.Column('error_type', sa.String(length=32), nullable=True),
|
||||||
sa.Column('check_interval_override', sa.Integer(), nullable=True),
|
sa.Column('check_interval_override', sa.Integer(), nullable=True),
|
||||||
sa.Column('consecutive_failures', sa.Integer(), nullable=False),
|
sa.Column('consecutive_failures', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('backfill_runs_remaining', sa.Integer(), server_default='0', nullable=False),
|
sa.Column('backfill_runs_remaining', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_source_artist_id_artist'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_source_artist_id_artist'), ondelete='CASCADE'),
|
||||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_source'))
|
sa.PrimaryKeyConstraint('id', name=op.f('pk_source')),
|
||||||
|
sa.UniqueConstraint('artist_id', 'platform', 'url', name='uq_source_artist_platform_url')
|
||||||
)
|
)
|
||||||
op.create_index(op.f('ix_source_artist_id'), 'source', ['artist_id'], unique=False)
|
op.create_index(op.f('ix_source_artist_id'), 'source', ['artist_id'], unique=False)
|
||||||
op.create_index(op.f('ix_source_error_type'), 'source', ['error_type'], unique=False)
|
op.create_index(op.f('ix_source_error_type'), 'source', ['error_type'], unique=False)
|
||||||
@@ -363,7 +396,7 @@ def upgrade() -> None:
|
|||||||
sa.ForeignKeyConstraint(['canonical_tag_id'], ['tag.id'], name=op.f('fk_tag_alias_canonical_tag_id_tag'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['canonical_tag_id'], ['tag.id'], name=op.f('fk_tag_alias_canonical_tag_id_tag'), ondelete='CASCADE'),
|
||||||
sa.PrimaryKeyConstraint('alias_string', 'alias_category', name=op.f('pk_tag_alias'))
|
sa.PrimaryKeyConstraint('alias_string', 'alias_category', name=op.f('pk_tag_alias'))
|
||||||
)
|
)
|
||||||
op.create_index(op.f('ix_tag_alias_canonical_tag_id'), 'tag_alias', ['canonical_tag_id'], unique=False)
|
op.create_index('ix_tag_alias_canonical', 'tag_alias', ['canonical_tag_id'], unique=False)
|
||||||
op.create_table('tag_head',
|
op.create_table('tag_head',
|
||||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||||
sa.Column('embedding_version', sa.String(length=128), nullable=False),
|
sa.Column('embedding_version', sa.String(length=128), nullable=False),
|
||||||
@@ -386,7 +419,7 @@ def upgrade() -> None:
|
|||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('source_id', sa.Integer(), nullable=False),
|
sa.Column('source_id', sa.Integer(), nullable=False),
|
||||||
sa.Column('filehash', sa.String(length=128), nullable=False),
|
sa.Column('filehash', sa.String(length=128), nullable=False),
|
||||||
sa.Column('attempts', sa.Integer(), nullable=False),
|
sa.Column('attempts', sa.Integer(), server_default='1', nullable=False),
|
||||||
sa.Column('last_error', sa.Text(), nullable=True),
|
sa.Column('last_error', sa.Text(), nullable=True),
|
||||||
sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
@@ -410,7 +443,7 @@ def upgrade() -> None:
|
|||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('source_id', sa.Integer(), nullable=False),
|
sa.Column('source_id', sa.Integer(), nullable=False),
|
||||||
sa.Column('filehash', sa.String(length=128), nullable=False),
|
sa.Column('filehash', sa.String(length=128), nullable=False),
|
||||||
sa.Column('attempts', sa.Integer(), nullable=False),
|
sa.Column('attempts', sa.Integer(), server_default='1', nullable=False),
|
||||||
sa.Column('last_error', sa.Text(), nullable=True),
|
sa.Column('last_error', sa.Text(), nullable=True),
|
||||||
sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
@@ -448,7 +481,7 @@ def upgrade() -> None:
|
|||||||
sa.Column('translated_at', sa.DateTime(timezone=True), nullable=True),
|
sa.Column('translated_at', sa.DateTime(timezone=True), nullable=True),
|
||||||
sa.Column('translation_override', sa.String(length=16), server_default='auto', nullable=False),
|
sa.Column('translation_override', sa.String(length=16), server_default='auto', nullable=False),
|
||||||
sa.Column('downloaded_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('downloaded_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.CheckConstraint("translation_override IN ('auto', 'force', 'original')", name=op.f('ck_post_ck_post_translation_override')),
|
sa.CheckConstraint("translation_override IN ('auto', 'force', 'original')", name=op.f('ck_post_translation_override')),
|
||||||
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_post_artist_id_artist'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_post_artist_id_artist'), ondelete='CASCADE'),
|
||||||
sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_post_source_id_source'), ondelete='SET NULL'),
|
sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_post_source_id_source'), ondelete='SET NULL'),
|
||||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_post')),
|
sa.PrimaryKeyConstraint('id', name=op.f('pk_post')),
|
||||||
@@ -456,11 +489,12 @@ def upgrade() -> None:
|
|||||||
)
|
)
|
||||||
op.create_index(op.f('ix_post_artist_id'), 'post', ['artist_id'], unique=False)
|
op.create_index(op.f('ix_post_artist_id'), 'post', ['artist_id'], unique=False)
|
||||||
op.create_index(op.f('ix_post_source_id'), 'post', ['source_id'], unique=False)
|
op.create_index(op.f('ix_post_source_id'), 'post', ['source_id'], unique=False)
|
||||||
|
op.create_index('uq_post_artist_external_id_null_source', 'post', ['artist_id', 'external_post_id'], unique=True, postgresql_where=sa.text('source_id IS NULL'))
|
||||||
op.create_table('subscribestar_failed_media',
|
op.create_table('subscribestar_failed_media',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('source_id', sa.Integer(), nullable=False),
|
sa.Column('source_id', sa.Integer(), nullable=False),
|
||||||
sa.Column('filehash', sa.String(length=128), nullable=False),
|
sa.Column('filehash', sa.String(length=128), nullable=False),
|
||||||
sa.Column('attempts', sa.Integer(), nullable=False),
|
sa.Column('attempts', sa.Integer(), server_default='1', nullable=False),
|
||||||
sa.Column('last_error', sa.Text(), nullable=True),
|
sa.Column('last_error', sa.Text(), nullable=True),
|
||||||
sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
@@ -487,8 +521,8 @@ def upgrade() -> None:
|
|||||||
sa.Column('status', sa.String(length=32), nullable=False),
|
sa.Column('status', sa.String(length=32), nullable=False),
|
||||||
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||||
sa.Column('bytes_downloaded', sa.BigInteger(), nullable=False),
|
sa.Column('bytes_downloaded', sa.BigInteger(), server_default='0', nullable=False),
|
||||||
sa.Column('files_count', sa.Integer(), nullable=False),
|
sa.Column('files_count', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('error', sa.Text(), nullable=True),
|
sa.Column('error', sa.Text(), nullable=True),
|
||||||
sa.Column('metadata', postgresql.JSONB(astext_type=sa.Text()), server_default=sa.text("'{}'::jsonb"), nullable=False),
|
sa.Column('metadata', postgresql.JSONB(astext_type=sa.Text()), server_default=sa.text("'{}'::jsonb"), nullable=False),
|
||||||
sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_download_event_post_id_post'), ondelete='SET NULL'),
|
sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_download_event_post_id_post'), ondelete='SET NULL'),
|
||||||
@@ -507,7 +541,7 @@ def upgrade() -> None:
|
|||||||
sa.Column('width', sa.Integer(), nullable=True),
|
sa.Column('width', sa.Integer(), nullable=True),
|
||||||
sa.Column('height', sa.Integer(), nullable=True),
|
sa.Column('height', sa.Integer(), nullable=True),
|
||||||
sa.Column('duration_seconds', sa.Float(), nullable=True),
|
sa.Column('duration_seconds', sa.Float(), nullable=True),
|
||||||
sa.Column('integrity_status', sa.String(length=24), nullable=False),
|
sa.Column('integrity_status', sa.String(length=24), server_default='unknown', nullable=False),
|
||||||
sa.Column('thumbnail_path', sa.Text(), nullable=True),
|
sa.Column('thumbnail_path', sa.Text(), nullable=True),
|
||||||
sa.Column('source_url', sa.Text(), nullable=True),
|
sa.Column('source_url', sa.Text(), nullable=True),
|
||||||
sa.Column('source_filehash', sa.String(length=32), nullable=True),
|
sa.Column('source_filehash', sa.String(length=32), nullable=True),
|
||||||
@@ -520,16 +554,19 @@ def upgrade() -> None:
|
|||||||
sa.Column('effective_date', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('effective_date', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.Column('earliest_post_date', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('earliest_post_date', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_image_record_artist_id_artist'), ondelete='SET NULL'),
|
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name='fk_image_record_artist_id', ondelete='SET NULL'),
|
||||||
sa.ForeignKeyConstraint(['primary_post_id'], ['post.id'], name=op.f('fk_image_record_primary_post_id_post'), ondelete='SET NULL'),
|
sa.ForeignKeyConstraint(['primary_post_id'], ['post.id'], name=op.f('fk_image_record_primary_post_id_post'), ondelete='SET NULL'),
|
||||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_image_record')),
|
sa.PrimaryKeyConstraint('id', name=op.f('pk_image_record')),
|
||||||
sa.UniqueConstraint('path', name=op.f('uq_image_record_path'))
|
sa.UniqueConstraint('path', name=op.f('uq_image_record_path')),
|
||||||
|
sa.UniqueConstraint('sha256', name='uq_image_record_sha256')
|
||||||
)
|
)
|
||||||
op.create_index(op.f('ix_image_record_artist_id'), 'image_record', ['artist_id'], unique=False)
|
op.create_index(op.f('ix_image_record_artist_id'), 'image_record', ['artist_id'], unique=False)
|
||||||
|
op.create_index('ix_image_record_earliest_post_date', 'image_record', [sa.literal_column('earliest_post_date DESC'), sa.literal_column('id DESC')], unique=False)
|
||||||
|
op.create_index('ix_image_record_effective_date', 'image_record', [sa.literal_column('effective_date DESC'), sa.literal_column('id DESC')], unique=False)
|
||||||
op.create_index(op.f('ix_image_record_integrity_status'), 'image_record', ['integrity_status'], unique=False)
|
op.create_index(op.f('ix_image_record_integrity_status'), 'image_record', ['integrity_status'], unique=False)
|
||||||
op.create_index(op.f('ix_image_record_phash'), 'image_record', ['phash'], unique=False)
|
op.create_index(op.f('ix_image_record_phash'), 'image_record', ['phash'], unique=False)
|
||||||
op.create_index(op.f('ix_image_record_primary_post_id'), 'image_record', ['primary_post_id'], unique=False)
|
op.create_index(op.f('ix_image_record_primary_post_id'), 'image_record', ['primary_post_id'], unique=False)
|
||||||
op.create_index(op.f('ix_image_record_sha256'), 'image_record', ['sha256'], unique=True)
|
op.create_index('ix_image_record_siglip_hnsw', 'image_record', ['siglip_embedding'], unique=False, postgresql_using='hnsw', postgresql_ops={'siglip_embedding': 'vector_cosine_ops'})
|
||||||
op.create_index(op.f('ix_image_record_source_filehash'), 'image_record', ['source_filehash'], unique=False)
|
op.create_index(op.f('ix_image_record_source_filehash'), 'image_record', ['source_filehash'], unique=False)
|
||||||
op.create_table('post_attachment',
|
op.create_table('post_attachment',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
@@ -582,24 +619,26 @@ def upgrade() -> None:
|
|||||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.Column('completed_at', sa.DateTime(timezone=True), nullable=True),
|
sa.Column('completed_at', sa.DateTime(timezone=True), nullable=True),
|
||||||
sa.Column('duration_seconds', sa.Float(), nullable=True),
|
sa.Column('duration_seconds', sa.Float(), nullable=True),
|
||||||
|
sa.CheckConstraint("host IN ('mega', 'gdrive', 'mediafire', 'dropbox', 'pixeldrain')", name=op.f('ck_external_link_host')),
|
||||||
|
sa.CheckConstraint("status IN ('pending', 'downloading', 'downloaded', 'failed', 'skipped', 'dead')", name=op.f('ck_external_link_status')),
|
||||||
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_external_link_artist_id_artist'), ondelete='SET NULL'),
|
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_external_link_artist_id_artist'), ondelete='SET NULL'),
|
||||||
sa.ForeignKeyConstraint(['attachment_id'], ['post_attachment.id'], name=op.f('fk_external_link_attachment_id_post_attachment'), ondelete='SET NULL'),
|
sa.ForeignKeyConstraint(['attachment_id'], ['post_attachment.id'], name=op.f('fk_external_link_attachment_id_post_attachment'), ondelete='SET NULL'),
|
||||||
sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_external_link_post_id_post'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_external_link_post_id_post'), ondelete='CASCADE'),
|
||||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_external_link'))
|
sa.PrimaryKeyConstraint('id', name=op.f('pk_external_link'))
|
||||||
)
|
)
|
||||||
op.create_index(op.f('ix_external_link_artist_id'), 'external_link', ['artist_id'], unique=False)
|
op.create_index(op.f('ix_external_link_artist_id'), 'external_link', ['artist_id'], unique=False)
|
||||||
op.create_index(op.f('ix_external_link_post_id'), 'external_link', ['post_id'], unique=False)
|
op.create_index('ix_external_link_attachment_id', 'external_link', ['attachment_id'], unique=False)
|
||||||
op.create_index('ix_external_link_status', 'external_link', ['status'], unique=False)
|
op.create_index('ix_external_link_status', 'external_link', ['status'], unique=False)
|
||||||
op.create_index('uq_external_link_post_url', 'external_link', ['post_id', 'url'], unique=True)
|
op.create_index('uq_external_link_post_url', 'external_link', ['post_id', 'url'], unique=True)
|
||||||
op.create_table('gpu_job',
|
op.create_table('gpu_job',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
||||||
sa.Column('task', sa.String(length=32), nullable=False),
|
sa.Column('task', sa.String(length=32), nullable=False),
|
||||||
sa.Column('status', sa.String(length=16), nullable=False),
|
sa.Column('status', sa.String(length=16), server_default='pending', nullable=False),
|
||||||
sa.Column('lease_token', sa.String(length=64), nullable=True),
|
sa.Column('lease_token', sa.String(length=64), nullable=True),
|
||||||
sa.Column('leased_at', sa.DateTime(timezone=True), nullable=True),
|
sa.Column('leased_at', sa.DateTime(timezone=True), nullable=True),
|
||||||
sa.Column('lease_expires_at', sa.DateTime(timezone=True), nullable=True),
|
sa.Column('lease_expires_at', sa.DateTime(timezone=True), nullable=True),
|
||||||
sa.Column('attempts', sa.Integer(), nullable=False),
|
sa.Column('attempts', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('error', sa.Text(), nullable=True),
|
sa.Column('error', sa.Text(), nullable=True),
|
||||||
sa.Column('triage_status', sa.String(length=16), nullable=True),
|
sa.Column('triage_status', sa.String(length=16), nullable=True),
|
||||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
@@ -619,7 +658,7 @@ def upgrade() -> None:
|
|||||||
sa.Column('from_attachment_id', sa.Integer(), nullable=True),
|
sa.Column('from_attachment_id', sa.Integer(), nullable=True),
|
||||||
sa.Column('captured_metadata', sa.JSON(), nullable=True),
|
sa.Column('captured_metadata', sa.JSON(), nullable=True),
|
||||||
sa.Column('captured_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('captured_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.ForeignKeyConstraint(['from_attachment_id'], ['post_attachment.id'], name=op.f('fk_image_provenance_from_attachment_id_post_attachment'), ondelete='SET NULL'),
|
sa.ForeignKeyConstraint(['from_attachment_id'], ['post_attachment.id'], name='fk_image_provenance_from_attachment', ondelete='SET NULL'),
|
||||||
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_image_provenance_image_record_id_image_record'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_image_provenance_image_record_id_image_record'), ondelete='CASCADE'),
|
||||||
sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_image_provenance_post_id_post'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_image_provenance_post_id_post'), ondelete='CASCADE'),
|
||||||
sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_image_provenance_source_id_source'), ondelete='SET NULL'),
|
sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_image_provenance_source_id_source'), ondelete='SET NULL'),
|
||||||
@@ -653,20 +692,21 @@ def upgrade() -> None:
|
|||||||
op.create_table('image_tag',
|
op.create_table('image_tag',
|
||||||
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
||||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||||
sa.Column('source', sa.String(length=32), nullable=False),
|
sa.Column('source', sa.String(length=32), server_default='manual', nullable=False),
|
||||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_image_tag_image_record_id_image_record'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_image_tag_image_record_id_image_record'), ondelete='CASCADE'),
|
||||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_image_tag_tag_id_tag'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_image_tag_tag_id_tag'), ondelete='CASCADE'),
|
||||||
sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_image_tag'))
|
sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_image_tag'))
|
||||||
)
|
)
|
||||||
|
op.create_index('ix_image_tag_tag_id', 'image_tag', ['tag_id'], unique=False)
|
||||||
op.create_table('import_task',
|
op.create_table('import_task',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('batch_id', sa.Integer(), nullable=False),
|
sa.Column('batch_id', sa.Integer(), nullable=False),
|
||||||
sa.Column('source_path', sa.Text(), nullable=False),
|
sa.Column('source_path', sa.Text(), nullable=False),
|
||||||
sa.Column('task_type', sa.String(length=16), nullable=False),
|
sa.Column('task_type', sa.String(length=16), nullable=False),
|
||||||
sa.Column('status', sa.String(length=16), nullable=False),
|
sa.Column('status', sa.String(length=16), server_default='pending', nullable=False),
|
||||||
sa.Column('recovery_count', sa.Integer(), nullable=False),
|
sa.Column('recovery_count', sa.Integer(), server_default='0', nullable=False),
|
||||||
sa.Column('refetched', sa.Boolean(), nullable=False),
|
sa.Column('refetched', sa.Boolean(), server_default='false', nullable=False),
|
||||||
sa.Column('result_image_id', sa.Integer(), nullable=True),
|
sa.Column('result_image_id', sa.Integer(), nullable=True),
|
||||||
sa.Column('error', sa.Text(), nullable=True),
|
sa.Column('error', sa.Text(), nullable=True),
|
||||||
sa.Column('size_bytes', sa.BigInteger(), nullable=True),
|
sa.Column('size_bytes', sa.BigInteger(), nullable=True),
|
||||||
@@ -678,6 +718,8 @@ def upgrade() -> None:
|
|||||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_import_task'))
|
sa.PrimaryKeyConstraint('id', name=op.f('pk_import_task'))
|
||||||
)
|
)
|
||||||
op.create_index(op.f('ix_import_task_batch_id'), 'import_task', ['batch_id'], unique=False)
|
op.create_index(op.f('ix_import_task_batch_id'), 'import_task', ['batch_id'], unique=False)
|
||||||
|
op.create_index('ix_import_task_created_at_desc', 'import_task', [sa.literal_column('created_at DESC')], unique=False)
|
||||||
|
op.create_index('ix_import_task_result_image_id', 'import_task', ['result_image_id'], unique=False)
|
||||||
op.create_index(op.f('ix_import_task_status'), 'import_task', ['status'], unique=False)
|
op.create_index(op.f('ix_import_task_status'), 'import_task', ['status'], unique=False)
|
||||||
op.create_table('presentation_review',
|
op.create_table('presentation_review',
|
||||||
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
||||||
@@ -692,6 +734,9 @@ def upgrade() -> None:
|
|||||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_presentation_review_tag_id_tag'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_presentation_review_tag_id_tag'), ondelete='CASCADE'),
|
||||||
sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_presentation_review'))
|
sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_presentation_review'))
|
||||||
)
|
)
|
||||||
|
op.create_index('ix_presentation_review_conflict_tag_id', 'presentation_review', ['conflict_tag_id'], unique=False)
|
||||||
|
op.create_index('ix_presentation_review_resolved_at', 'presentation_review', ['resolved_at'], unique=False)
|
||||||
|
op.create_index('ix_presentation_review_tag_id', 'presentation_review', ['tag_id'], unique=False)
|
||||||
op.create_table('series_page',
|
op.create_table('series_page',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('series_tag_id', sa.Integer(), nullable=False),
|
sa.Column('series_tag_id', sa.Integer(), nullable=False),
|
||||||
@@ -704,7 +749,7 @@ def upgrade() -> None:
|
|||||||
sa.ForeignKeyConstraint(['image_id'], ['image_record.id'], name=op.f('fk_series_page_image_id_image_record'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['image_id'], ['image_record.id'], name=op.f('fk_series_page_image_id_image_record'), ondelete='CASCADE'),
|
||||||
sa.ForeignKeyConstraint(['series_tag_id'], ['tag.id'], name=op.f('fk_series_page_series_tag_id_tag'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['series_tag_id'], ['tag.id'], name=op.f('fk_series_page_series_tag_id_tag'), ondelete='CASCADE'),
|
||||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_series_page')),
|
sa.PrimaryKeyConstraint('id', name=op.f('pk_series_page')),
|
||||||
sa.UniqueConstraint('image_id', name=op.f('uq_series_page_image_id'))
|
sa.UniqueConstraint('image_id', name='uq_series_page_image')
|
||||||
)
|
)
|
||||||
op.create_index(op.f('ix_series_page_series_tag_id'), 'series_page', ['series_tag_id'], unique=False)
|
op.create_index(op.f('ix_series_page_series_tag_id'), 'series_page', ['series_tag_id'], unique=False)
|
||||||
op.create_table('tag_positive_confirmation',
|
op.create_table('tag_positive_confirmation',
|
||||||
@@ -720,11 +765,11 @@ def upgrade() -> None:
|
|||||||
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
||||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||||
sa.Column('rejected_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('rejected_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_tag_suggestion_rejection_image_record_id_image_record'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name='fk_tsr_image_record_id_image_record', ondelete='CASCADE'),
|
||||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_tag_suggestion_rejection_tag_id_tag'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name='fk_tsr_tag_id_tag', ondelete='CASCADE'),
|
||||||
sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_tag_suggestion_rejection'))
|
sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_tag_suggestion_rejection'))
|
||||||
)
|
)
|
||||||
op.create_index(op.f('ix_tag_suggestion_rejection_tag_id'), 'tag_suggestion_rejection', ['tag_id'], unique=False)
|
op.create_index('ix_tag_suggestion_rejection_tag', 'tag_suggestion_rejection', ['tag_id'], unique=False)
|
||||||
op.create_table('character_prototype',
|
op.create_table('character_prototype',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||||
@@ -734,6 +779,7 @@ def upgrade() -> None:
|
|||||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_character_prototype_tag_id_tag'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_character_prototype_tag_id_tag'), ondelete='CASCADE'),
|
||||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_character_prototype'))
|
sa.PrimaryKeyConstraint('id', name=op.f('pk_character_prototype'))
|
||||||
)
|
)
|
||||||
|
op.create_index(op.f('ix_character_prototype_region_id'), 'character_prototype', ['region_id'], unique=False)
|
||||||
op.create_index(op.f('ix_character_prototype_tag_id'), 'character_prototype', ['tag_id'], unique=False)
|
op.create_index(op.f('ix_character_prototype_tag_id'), 'character_prototype', ['tag_id'], unique=False)
|
||||||
op.create_table('series_chapter',
|
op.create_table('series_chapter',
|
||||||
sa.Column('id', sa.Integer(), nullable=False),
|
sa.Column('id', sa.Integer(), nullable=False),
|
||||||
@@ -743,130 +789,48 @@ def upgrade() -> None:
|
|||||||
sa.Column('stated_part', sa.Integer(), nullable=True),
|
sa.Column('stated_part', sa.Integer(), nullable=True),
|
||||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||||
sa.ForeignKeyConstraint(['anchor_page_id'], ['series_page.id'], name=op.f('fk_series_chapter_anchor_page_id_series_page'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['anchor_page_id'], ['series_page.id'], name='fk_series_chapter_anchor_page', ondelete='CASCADE'),
|
||||||
sa.ForeignKeyConstraint(['series_tag_id'], ['tag.id'], name=op.f('fk_series_chapter_series_tag_id_tag'), ondelete='CASCADE'),
|
sa.ForeignKeyConstraint(['series_tag_id'], ['tag.id'], name=op.f('fk_series_chapter_series_tag_id_tag'), ondelete='CASCADE'),
|
||||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_series_chapter')),
|
sa.PrimaryKeyConstraint('id', name=op.f('pk_series_chapter')),
|
||||||
sa.UniqueConstraint('anchor_page_id', name=op.f('uq_series_chapter_anchor_page_id'))
|
sa.UniqueConstraint('anchor_page_id', name='uq_series_chapter_anchor_page')
|
||||||
)
|
)
|
||||||
op.create_index(op.f('ix_series_chapter_series_tag_id'), 'series_chapter', ['series_tag_id'], unique=False)
|
op.create_index(op.f('ix_series_chapter_series_tag_id'), 'series_chapter', ['series_tag_id'], unique=False)
|
||||||
|
|
||||||
# The HNSW index, item 3 above. Must match the query's cosine-distance
|
# The singleton settings rows. NOT schema — see the note above; the app
|
||||||
# operator class or the planner will not use it.
|
# reads these with scalar_one() and never creates them, so a fresh
|
||||||
op.execute(
|
# install without these two rows raises NoResultFound on first use.
|
||||||
"CREATE INDEX ix_image_record_siglip_hnsw "
|
# From 0002 and 0003.
|
||||||
"ON image_record USING hnsw (siglip_embedding vector_cosine_ops)"
|
op.execute("INSERT INTO import_settings (id) VALUES (1)")
|
||||||
)
|
op.execute("INSERT INTO ml_settings (id) VALUES (1)")
|
||||||
|
|
||||||
|
# The three hygiene system tags, from 0075. These are PRODUCT data, not
|
||||||
|
# operator configuration — 0075's own docstring says so: "the fix keys on
|
||||||
|
# SYSTEM tags the product ships". The presentation and process auto-apply
|
||||||
|
# sweeps look them up with scalar_one(), so without these rows those
|
||||||
|
# features raise NoResultFound rather than degrading.
|
||||||
|
#
|
||||||
|
# 0075 adopted an existing same-name general tag before inserting, because
|
||||||
|
# an operator might already have tagged `wip` by hand. That cannot happen
|
||||||
|
# on the empty database this file runs against, but the guard is kept: it
|
||||||
|
# costs nothing and makes the statement safe to re-run.
|
||||||
|
for _name in ("wip", "banner", "editor screenshot"):
|
||||||
|
op.execute(
|
||||||
|
sa.text(
|
||||||
|
"INSERT INTO tag (name, kind, is_system) "
|
||||||
|
"SELECT :name, 'general', true WHERE NOT EXISTS ("
|
||||||
|
" SELECT 1 FROM tag WHERE lower(name) = lower(:name)"
|
||||||
|
")"
|
||||||
|
).bindparams(name=_name)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def downgrade() -> None:
|
def downgrade() -> None:
|
||||||
# Dropping image_record takes its indexes with it, so the HNSW index needs
|
"""Deliberately not implemented.
|
||||||
# no separate drop. The extensions are deliberately left in place: they are
|
|
||||||
# database-scoped and something else may be using them.
|
Downgrading a baseline means dropping every table in the database. That is
|
||||||
op.drop_index(op.f('ix_series_chapter_series_tag_id'), table_name='series_chapter')
|
not a migration, and offering it as one invites someone to run it. Restore
|
||||||
op.drop_table('series_chapter')
|
from a backup instead.
|
||||||
op.drop_index(op.f('ix_character_prototype_tag_id'), table_name='character_prototype')
|
"""
|
||||||
op.drop_table('character_prototype')
|
raise NotImplementedError(
|
||||||
op.drop_index(op.f('ix_tag_suggestion_rejection_tag_id'), table_name='tag_suggestion_rejection')
|
"0089 is the baseline; there is nothing below it. Restore from a backup."
|
||||||
op.drop_table('tag_suggestion_rejection')
|
)
|
||||||
op.drop_index(op.f('ix_tag_positive_confirmation_tag_id'), table_name='tag_positive_confirmation')
|
|
||||||
op.drop_table('tag_positive_confirmation')
|
|
||||||
op.drop_index(op.f('ix_series_page_series_tag_id'), table_name='series_page')
|
|
||||||
op.drop_table('series_page')
|
|
||||||
op.drop_table('presentation_review')
|
|
||||||
op.drop_index(op.f('ix_import_task_status'), table_name='import_task')
|
|
||||||
op.drop_index(op.f('ix_import_task_batch_id'), table_name='import_task')
|
|
||||||
op.drop_table('import_task')
|
|
||||||
op.drop_table('image_tag')
|
|
||||||
op.drop_index(op.f('ix_image_region_image_record_id'), table_name='image_region')
|
|
||||||
op.drop_table('image_region')
|
|
||||||
op.drop_index(op.f('ix_image_provenance_source_id'), table_name='image_provenance')
|
|
||||||
op.drop_index(op.f('ix_image_provenance_post_id'), table_name='image_provenance')
|
|
||||||
op.drop_index(op.f('ix_image_provenance_image_record_id'), table_name='image_provenance')
|
|
||||||
op.drop_index(op.f('ix_image_provenance_from_attachment_id'), table_name='image_provenance')
|
|
||||||
op.drop_table('image_provenance')
|
|
||||||
op.drop_index(op.f('ix_gpu_job_status'), table_name='gpu_job')
|
|
||||||
op.drop_index('ix_gpu_job_pending', table_name='gpu_job', postgresql_where=sa.text("status = 'pending'"))
|
|
||||||
op.drop_index('ix_gpu_job_leased_expires', table_name='gpu_job', postgresql_where=sa.text("status = 'leased'"))
|
|
||||||
op.drop_index(op.f('ix_gpu_job_image_record_id'), table_name='gpu_job')
|
|
||||||
op.drop_table('gpu_job')
|
|
||||||
op.drop_index('uq_external_link_post_url', table_name='external_link')
|
|
||||||
op.drop_index('ix_external_link_status', table_name='external_link')
|
|
||||||
op.drop_index(op.f('ix_external_link_post_id'), table_name='external_link')
|
|
||||||
op.drop_index(op.f('ix_external_link_artist_id'), table_name='external_link')
|
|
||||||
op.drop_table('external_link')
|
|
||||||
op.drop_index(op.f('ix_series_suggestion_status'), table_name='series_suggestion')
|
|
||||||
op.drop_index(op.f('ix_series_suggestion_series_tag_id'), table_name='series_suggestion')
|
|
||||||
op.drop_index(op.f('ix_series_suggestion_post_id'), table_name='series_suggestion')
|
|
||||||
op.drop_table('series_suggestion')
|
|
||||||
op.drop_index('uq_post_attachment_post_sha', table_name='post_attachment', postgresql_where=sa.text('post_id IS NOT NULL'))
|
|
||||||
op.drop_index('uq_post_attachment_null_post_sha', table_name='post_attachment', postgresql_where=sa.text('post_id IS NULL'))
|
|
||||||
op.drop_index(op.f('ix_post_attachment_sha256'), table_name='post_attachment')
|
|
||||||
op.drop_index(op.f('ix_post_attachment_post_id'), table_name='post_attachment')
|
|
||||||
op.drop_index(op.f('ix_post_attachment_artist_id'), table_name='post_attachment')
|
|
||||||
op.drop_table('post_attachment')
|
|
||||||
op.drop_index(op.f('ix_image_record_source_filehash'), table_name='image_record')
|
|
||||||
op.drop_index(op.f('ix_image_record_sha256'), table_name='image_record')
|
|
||||||
op.drop_index(op.f('ix_image_record_primary_post_id'), table_name='image_record')
|
|
||||||
op.drop_index(op.f('ix_image_record_phash'), table_name='image_record')
|
|
||||||
op.drop_index(op.f('ix_image_record_integrity_status'), table_name='image_record')
|
|
||||||
op.drop_index(op.f('ix_image_record_artist_id'), table_name='image_record')
|
|
||||||
op.drop_table('image_record')
|
|
||||||
op.drop_index(op.f('ix_download_event_source_id'), table_name='download_event')
|
|
||||||
op.drop_index(op.f('ix_download_event_post_id'), table_name='download_event')
|
|
||||||
op.drop_table('download_event')
|
|
||||||
op.drop_index(op.f('ix_subscribestar_seen_media_source_id'), table_name='subscribestar_seen_media')
|
|
||||||
op.drop_table('subscribestar_seen_media')
|
|
||||||
op.drop_index(op.f('ix_subscribestar_failed_media_source_id'), table_name='subscribestar_failed_media')
|
|
||||||
op.drop_table('subscribestar_failed_media')
|
|
||||||
op.drop_index(op.f('ix_post_source_id'), table_name='post')
|
|
||||||
op.drop_index(op.f('ix_post_artist_id'), table_name='post')
|
|
||||||
op.drop_table('post')
|
|
||||||
op.drop_index(op.f('ix_pixiv_seen_media_source_id'), table_name='pixiv_seen_media')
|
|
||||||
op.drop_table('pixiv_seen_media')
|
|
||||||
op.drop_index(op.f('ix_pixiv_failed_media_source_id'), table_name='pixiv_failed_media')
|
|
||||||
op.drop_table('pixiv_failed_media')
|
|
||||||
op.drop_index(op.f('ix_patreon_seen_media_source_id'), table_name='patreon_seen_media')
|
|
||||||
op.drop_table('patreon_seen_media')
|
|
||||||
op.drop_index(op.f('ix_patreon_failed_media_source_id'), table_name='patreon_failed_media')
|
|
||||||
op.drop_table('patreon_failed_media')
|
|
||||||
op.drop_table('tag_head')
|
|
||||||
op.drop_index(op.f('ix_tag_alias_canonical_tag_id'), table_name='tag_alias')
|
|
||||||
op.drop_table('tag_alias')
|
|
||||||
op.drop_index(op.f('ix_source_error_type'), table_name='source')
|
|
||||||
op.drop_index(op.f('ix_source_artist_id'), table_name='source')
|
|
||||||
op.drop_table('source')
|
|
||||||
op.drop_index(op.f('ix_head_metrics_snapshot_tag_id'), table_name='head_metrics_snapshot')
|
|
||||||
op.drop_index(op.f('ix_head_metrics_snapshot_snapshot_at'), table_name='head_metrics_snapshot')
|
|
||||||
op.drop_table('head_metrics_snapshot')
|
|
||||||
op.drop_table('head_metric')
|
|
||||||
op.drop_table('ccip_prototype_state')
|
|
||||||
op.drop_table('artist_visit')
|
|
||||||
op.drop_index(op.f('ix_task_run_task_name'), table_name='task_run')
|
|
||||||
op.drop_index(op.f('ix_task_run_status'), table_name='task_run')
|
|
||||||
op.drop_index(op.f('ix_task_run_started_at'), table_name='task_run')
|
|
||||||
op.drop_index(op.f('ix_task_run_queue'), table_name='task_run')
|
|
||||||
op.drop_index(op.f('ix_task_run_finished_at'), table_name='task_run')
|
|
||||||
op.drop_index(op.f('ix_task_run_celery_task_id'), table_name='task_run')
|
|
||||||
op.drop_table('task_run')
|
|
||||||
op.drop_index(op.f('ix_tag_fandom_id'), table_name='tag')
|
|
||||||
op.drop_table('tag')
|
|
||||||
op.drop_table('ml_settings')
|
|
||||||
op.drop_index(op.f('ix_library_audit_run_status'), table_name='library_audit_run')
|
|
||||||
op.drop_index(op.f('ix_library_audit_run_rule'), table_name='library_audit_run')
|
|
||||||
op.drop_table('library_audit_run')
|
|
||||||
op.drop_table('import_settings')
|
|
||||||
op.drop_index(op.f('ix_import_batch_status'), table_name='import_batch')
|
|
||||||
op.drop_table('import_batch')
|
|
||||||
op.drop_index(op.f('ix_head_training_run_status'), table_name='head_training_run')
|
|
||||||
op.drop_table('head_training_run')
|
|
||||||
op.drop_index(op.f('ix_head_auto_apply_run_status'), table_name='head_auto_apply_run')
|
|
||||||
op.drop_table('head_auto_apply_run')
|
|
||||||
op.drop_table('credential')
|
|
||||||
op.drop_index(op.f('ix_backup_run_tag'), table_name='backup_run')
|
|
||||||
op.drop_index(op.f('ix_backup_run_status'), table_name='backup_run')
|
|
||||||
op.drop_index(op.f('ix_backup_run_started_at'), table_name='backup_run')
|
|
||||||
op.drop_index(op.f('ix_backup_run_kind'), table_name='backup_run')
|
|
||||||
op.drop_index(op.f('ix_backup_run_finished_at'), table_name='backup_run')
|
|
||||||
op.drop_table('backup_run')
|
|
||||||
op.drop_table('artist')
|
|
||||||
op.drop_table('app_setting')
|
|
||||||
@@ -0,0 +1,64 @@
|
|||||||
|
"""service_seen — the learned roster that makes a stopped part observable.
|
||||||
|
|
||||||
|
Milestone 365. Nothing in FabledCurator knew what was SUPPOSED to be running:
|
||||||
|
`celery inspect` reports the workers that answer, so a dead worker was a
|
||||||
|
shorter list rather than a red light, and the only surface that could tell an
|
||||||
|
operator otherwise was Portainer. This table is the memory that turns an
|
||||||
|
absence into something the app can see.
|
||||||
|
|
||||||
|
Keyed on the queue set for a celery role and on agent_id for the GPU agent —
|
||||||
|
NOT on the celery worker name, which here is `celery@<container id>` and is
|
||||||
|
minted fresh on every deploy. See the model docstring for why that choice is
|
||||||
|
the whole design.
|
||||||
|
|
||||||
|
## First migration on the collapsed baseline
|
||||||
|
|
||||||
|
0089 is the single generated baseline that replaced revisions 0001..0089
|
||||||
|
(milestone 328). This is the first revision written on top of it, so it is
|
||||||
|
also the first evidence that the chain steps forward from the collapse rather
|
||||||
|
than merely reproducing the schema — which nothing had demonstrated yet.
|
||||||
|
|
||||||
|
An existing install is at 0089 because it ran the real 0089; a fresh one is at
|
||||||
|
0089 because it ran the baseline. Both arrive here identically, which was the
|
||||||
|
property the collapse was designed around.
|
||||||
|
|
||||||
|
Revision ID: 0090
|
||||||
|
Revises: 0089
|
||||||
|
Create Date: 2026-09-02
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
revision: str = "0090"
|
||||||
|
down_revision: Union[str, None] = "0089"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"service_seen",
|
||||||
|
sa.Column("key", sa.String(length=128), nullable=False),
|
||||||
|
sa.Column("kind", sa.String(length=16), nullable=False),
|
||||||
|
sa.Column("display_name", sa.String(length=64), nullable=False),
|
||||||
|
sa.Column(
|
||||||
|
"first_seen_at", sa.DateTime(timezone=True),
|
||||||
|
server_default=sa.text("now()"), nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column(
|
||||||
|
"last_seen_at", sa.DateTime(timezone=True),
|
||||||
|
server_default=sa.text("now()"), nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column("details", sa.JSON(), nullable=False),
|
||||||
|
sa.PrimaryKeyConstraint("key", name=op.f("pk_service_seen")),
|
||||||
|
)
|
||||||
|
# No secondary indexes, deliberately: one row per moving part means every
|
||||||
|
# read is a handful of rows and an index would be write cost buying
|
||||||
|
# nothing (#3301 removed seven of exactly that shape).
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_table("service_seen")
|
||||||
@@ -38,6 +38,7 @@ def all_blueprints() -> list[Blueprint]:
|
|||||||
from .suggestions import suggestions_bp
|
from .suggestions import suggestions_bp
|
||||||
from .system_activity import system_activity_bp
|
from .system_activity import system_activity_bp
|
||||||
from .system_backup import system_backup_bp
|
from .system_backup import system_backup_bp
|
||||||
|
from .system_health import system_health_bp
|
||||||
from .tags import tags_bp
|
from .tags import tags_bp
|
||||||
from .thumbnails import thumbnails_bp
|
from .thumbnails import thumbnails_bp
|
||||||
return [
|
return [
|
||||||
@@ -51,6 +52,7 @@ def all_blueprints() -> list[Blueprint]:
|
|||||||
showcase_bp,
|
showcase_bp,
|
||||||
settings_bp,
|
settings_bp,
|
||||||
system_activity_bp,
|
system_activity_bp,
|
||||||
|
system_health_bp,
|
||||||
system_backup_bp,
|
system_backup_bp,
|
||||||
admin_bp,
|
admin_bp,
|
||||||
cleanup_bp,
|
cleanup_bp,
|
||||||
|
|||||||
@@ -21,6 +21,7 @@ from ..services.gallery_service import image_url
|
|||||||
from ..services.ml.gpu_jobs import GpuJobService, error_dedupe_statements
|
from ..services.ml.gpu_jobs import GpuJobService, error_dedupe_statements
|
||||||
from ..services.ml.gpu_triage import classify_reason, recover_defective_image
|
from ..services.ml.gpu_triage import classify_reason, recover_defective_image
|
||||||
from ..services.ml.regions import RegionService
|
from ..services.ml.regions import RegionService
|
||||||
|
from ..services.service_roster import touch_service
|
||||||
|
|
||||||
gpu_bp = Blueprint("gpu", __name__, url_prefix="/api/gpu")
|
gpu_bp = Blueprint("gpu", __name__, url_prefix="/api/gpu")
|
||||||
|
|
||||||
@@ -256,6 +257,18 @@ async def lease():
|
|||||||
if not await _agent_authed(session):
|
if not await _agent_authed(session):
|
||||||
return jsonify({"error": "unauthorized"}), 401
|
return jsonify({"error": "unauthorized"}), 401
|
||||||
jobs = await GpuJobService(session).lease(agent_id, batch_size=batch)
|
jobs = await GpuJobService(session).lease(agent_id, batch_size=batch)
|
||||||
|
# The agent cannot be polled — it is HTTP-only and pulls from here, so
|
||||||
|
# web never dials it. A lease IS the check-in, and until milestone 365
|
||||||
|
# it was thrown away: an agent sitting idle with nothing to lease left
|
||||||
|
# no trace at all and was indistinguishable from one switched off a
|
||||||
|
# week ago. Recorded on the call that was already happening.
|
||||||
|
await touch_service(
|
||||||
|
session,
|
||||||
|
key=f"agent:{agent_id}",
|
||||||
|
kind="agent",
|
||||||
|
display_name="GPU agent" if agent_id == "agent" else f"GPU agent ({agent_id})",
|
||||||
|
details={"agent_id": agent_id, "last_call": "lease", "leased": len(jobs)},
|
||||||
|
)
|
||||||
ml = await MLSettings.load(session)
|
ml = await MLSettings.load(session)
|
||||||
# image rows for url/mime in one shot
|
# image rows for url/mime in one shot
|
||||||
ids = [j.image_record_id for j in jobs]
|
ids = [j.image_record_id for j in jobs]
|
||||||
@@ -329,6 +342,13 @@ async def heartbeat():
|
|||||||
if not await _agent_authed(session):
|
if not await _agent_authed(session):
|
||||||
return jsonify({"error": "unauthorized"}), 401
|
return jsonify({"error": "unauthorized"}), 401
|
||||||
n = await GpuJobService(session).heartbeat(agent_id, job_ids)
|
n = await GpuJobService(session).heartbeat(agent_id, job_ids)
|
||||||
|
await touch_service(
|
||||||
|
session,
|
||||||
|
key=f"agent:{agent_id}",
|
||||||
|
kind="agent",
|
||||||
|
display_name="GPU agent" if agent_id == "agent" else f"GPU agent ({agent_id})",
|
||||||
|
details={"agent_id": agent_id, "last_call": "heartbeat", "extended": n},
|
||||||
|
)
|
||||||
await session.commit()
|
await session.commit()
|
||||||
return jsonify({"extended": n})
|
return jsonify({"extended": n})
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,192 @@
|
|||||||
|
"""Is every part of FabledCurator running? One verdict, one endpoint.
|
||||||
|
|
||||||
|
Milestone 365. The nav indicator and the System page both read this and
|
||||||
|
nothing else — composing a verdict is this module's job, not the UI's.
|
||||||
|
|
||||||
|
## Two kinds of part, answered two different ways
|
||||||
|
|
||||||
|
**Learned** — celery roles and the GPU agent, from `service_seen`. The
|
||||||
|
question is "how long since it checked in", and these are the parts that can
|
||||||
|
be ABSENT, which is the whole point: `celery inspect` alone reports presence,
|
||||||
|
so a dead worker is a shorter list rather than a red light.
|
||||||
|
|
||||||
|
**Probed live** — Postgres and Redis. Always expected, never learned, and a
|
||||||
|
last-seen for them would be actively misleading: that Redis answered thirty
|
||||||
|
seconds ago says nothing about now.
|
||||||
|
|
||||||
|
## This endpoint must never fail because something it checks has failed
|
||||||
|
|
||||||
|
The inversion is easy to write by accident and it destroys the feature exactly
|
||||||
|
when it is needed — a 500 when Redis is down, instead of `redis: down`. Every
|
||||||
|
probe is wrapped, every wait has a deadline (rule 156), and the roster refresh
|
||||||
|
swallows its own errors. The worst case is a part reported `unknown`, which is
|
||||||
|
a true statement.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import logging
|
||||||
|
import time
|
||||||
|
from datetime import UTC, datetime
|
||||||
|
|
||||||
|
from quart import Blueprint, jsonify
|
||||||
|
from sqlalchemy import select, text
|
||||||
|
|
||||||
|
from ..config import get_config
|
||||||
|
from ..extensions import get_session
|
||||||
|
from ..models import ServiceSeen
|
||||||
|
from ..services.service_roster import refresh_if_stale
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
system_health_bp = Blueprint("system_health", __name__, url_prefix="/api/system")
|
||||||
|
|
||||||
|
# How long a learned part may go quiet before it is doubted, then disbelieved.
|
||||||
|
#
|
||||||
|
# These are deliberately generous, and the reason is a deploy rather than a
|
||||||
|
# worker: `docker compose up -d` rolls start-first, so a role is briefly served
|
||||||
|
# by two containers and then by neither while the old one drains. Thresholds
|
||||||
|
# tight enough to catch a crash in seconds would paint the page red every time
|
||||||
|
# the stack is updated, and an alarm that cries wolf on every deploy is one
|
||||||
|
# nobody reads. Tune down only after watching a real deploy pass through.
|
||||||
|
STALE_AFTER_SECONDS = 90
|
||||||
|
DOWN_AFTER_SECONDS = 300
|
||||||
|
|
||||||
|
# Probes cross a process boundary, so they carry deadlines. A hung Postgres
|
||||||
|
# must make this endpoint say "postgres: down", not hang alongside it.
|
||||||
|
PROBE_TIMEOUT_SECONDS = 2.0
|
||||||
|
|
||||||
|
_OK, _STALE, _DOWN, _UNKNOWN = "ok", "stale", "down", "unknown"
|
||||||
|
|
||||||
|
# Worst-first, so an overall verdict is just the max.
|
||||||
|
_SEVERITY = {_OK: 0, _UNKNOWN: 1, _STALE: 2, _DOWN: 3}
|
||||||
|
|
||||||
|
|
||||||
|
def _age_state(age_seconds: float) -> str:
|
||||||
|
if age_seconds >= DOWN_AFTER_SECONDS:
|
||||||
|
return _DOWN
|
||||||
|
if age_seconds >= STALE_AFTER_SECONDS:
|
||||||
|
return _STALE
|
||||||
|
return _OK
|
||||||
|
|
||||||
|
|
||||||
|
def _describe_learned(name: str, state: str, age: float, details: dict) -> str:
|
||||||
|
"""Say what the state MEANS. A red chip tells an operator less than a
|
||||||
|
sentence does at the moment they are deciding whether to go and look."""
|
||||||
|
if state == _OK:
|
||||||
|
replicas = details.get("replicas")
|
||||||
|
if replicas and replicas > 1:
|
||||||
|
return f"{name} is running ({replicas} replicas)"
|
||||||
|
return f"{name} is running"
|
||||||
|
mins = int(age // 60)
|
||||||
|
ago = f"{mins} min" if mins else f"{int(age)}s"
|
||||||
|
if state == _STALE:
|
||||||
|
return f"{name} has not checked in for {ago}"
|
||||||
|
return f"{name} has not checked in for {ago} — treat it as stopped"
|
||||||
|
|
||||||
|
|
||||||
|
async def _probe_postgres(session) -> dict:
|
||||||
|
started = time.monotonic()
|
||||||
|
try:
|
||||||
|
await asyncio.wait_for(
|
||||||
|
session.execute(text("SELECT 1")), timeout=PROBE_TIMEOUT_SECONDS
|
||||||
|
)
|
||||||
|
except Exception as exc: # noqa: BLE001 — a probe reports, it never raises
|
||||||
|
return {
|
||||||
|
"key": "postgres", "kind": "datastore", "name": "PostgreSQL",
|
||||||
|
"state": _DOWN, "detail": f"not answering: {type(exc).__name__}",
|
||||||
|
}
|
||||||
|
return {
|
||||||
|
"key": "postgres", "kind": "datastore", "name": "PostgreSQL", "state": _OK,
|
||||||
|
"detail": "answering", "latency_ms": round((time.monotonic() - started) * 1000, 1),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _ping_redis_sync() -> None:
|
||||||
|
import redis # local import; mirrors system_activity's pattern
|
||||||
|
|
||||||
|
client = redis.Redis.from_url(
|
||||||
|
get_config().celery_broker_url,
|
||||||
|
socket_connect_timeout=PROBE_TIMEOUT_SECONDS,
|
||||||
|
socket_timeout=PROBE_TIMEOUT_SECONDS,
|
||||||
|
)
|
||||||
|
client.ping()
|
||||||
|
|
||||||
|
|
||||||
|
async def _probe_redis() -> dict:
|
||||||
|
started = time.monotonic()
|
||||||
|
try:
|
||||||
|
await asyncio.wait_for(
|
||||||
|
asyncio.to_thread(_ping_redis_sync), timeout=PROBE_TIMEOUT_SECONDS * 2
|
||||||
|
)
|
||||||
|
except Exception as exc: # noqa: BLE001
|
||||||
|
return {
|
||||||
|
"key": "redis", "kind": "datastore", "name": "Redis",
|
||||||
|
"state": _DOWN,
|
||||||
|
"detail": f"not answering: {type(exc).__name__} — queues and workers "
|
||||||
|
f"cannot be reached either",
|
||||||
|
}
|
||||||
|
return {
|
||||||
|
"key": "redis", "kind": "datastore", "name": "Redis", "state": _OK,
|
||||||
|
"detail": "answering", "latency_ms": round((time.monotonic() - started) * 1000, 1),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@system_health_bp.route("/health", methods=["GET"])
|
||||||
|
async def system_health():
|
||||||
|
"""Every part, its state, and one overall verdict.
|
||||||
|
|
||||||
|
Response: {overall, parts: [{key, kind, name, state, detail, last_seen_at,
|
||||||
|
…}], checked_at}
|
||||||
|
"""
|
||||||
|
parts: list[dict] = []
|
||||||
|
now = datetime.now(UTC)
|
||||||
|
|
||||||
|
async with get_session() as session:
|
||||||
|
# Postgres first, and if it is unreachable nothing else can be read —
|
||||||
|
# say so rather than failing, because "the database is down" is the
|
||||||
|
# single most useful thing this endpoint can ever report.
|
||||||
|
pg = await _probe_postgres(session)
|
||||||
|
parts.append(pg)
|
||||||
|
|
||||||
|
if pg["state"] == _OK:
|
||||||
|
# Rate-limited inside; see service_roster on why the web process
|
||||||
|
# is the right observer.
|
||||||
|
try:
|
||||||
|
await refresh_if_stale(session)
|
||||||
|
await session.commit()
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
log.warning("system health: roster refresh failed", exc_info=True)
|
||||||
|
|
||||||
|
rows = (
|
||||||
|
await session.execute(select(ServiceSeen).order_by(ServiceSeen.display_name))
|
||||||
|
).scalars().all()
|
||||||
|
for row in rows:
|
||||||
|
age = (now - row.last_seen_at).total_seconds()
|
||||||
|
state = _age_state(age)
|
||||||
|
parts.append({
|
||||||
|
"key": row.key,
|
||||||
|
"kind": row.kind,
|
||||||
|
"name": row.display_name,
|
||||||
|
"state": state,
|
||||||
|
"detail": _describe_learned(row.display_name, state, age, row.details or {}),
|
||||||
|
"last_seen_at": row.last_seen_at.isoformat(),
|
||||||
|
"first_seen_at": row.first_seen_at.isoformat(),
|
||||||
|
**{k: v for k, v in (row.details or {}).items() if k != "agent_id"},
|
||||||
|
})
|
||||||
|
|
||||||
|
parts.append(await _probe_redis())
|
||||||
|
|
||||||
|
overall = max((p["state"] for p in parts), key=lambda s: _SEVERITY[s], default=_UNKNOWN)
|
||||||
|
return jsonify({
|
||||||
|
"overall": overall,
|
||||||
|
"parts": sorted(parts, key=lambda p: (-_SEVERITY[p["state"]], p["name"])),
|
||||||
|
"checked_at": now.isoformat(),
|
||||||
|
# So the UI can explain a `stale` without hard-coding the same numbers
|
||||||
|
# in a second place.
|
||||||
|
"thresholds": {
|
||||||
|
"stale_after_seconds": STALE_AFTER_SECONDS,
|
||||||
|
"down_after_seconds": DOWN_AFTER_SECONDS,
|
||||||
|
},
|
||||||
|
})
|
||||||
@@ -16,8 +16,11 @@ class Config:
|
|||||||
celery_broker_url: str
|
celery_broker_url: str
|
||||||
celery_result_backend: str
|
celery_result_backend: str
|
||||||
|
|
||||||
|
# Sets Quart's app.secret_key. Nothing signs a cookie today (FC has no
|
||||||
|
# login and no session use), so this currently protects nothing — it is
|
||||||
|
# required rather than defaulted so that the day something session-backed
|
||||||
|
# does land, no instance is already running on a value we published.
|
||||||
secret_key: str
|
secret_key: str
|
||||||
extension_api_key: str # used by the Firefox extension; lands in FC-3 but read here
|
|
||||||
log_level: str
|
log_level: str
|
||||||
|
|
||||||
@property
|
@property
|
||||||
@@ -47,6 +50,5 @@ def get_config() -> Config:
|
|||||||
celery_broker_url=os.environ.get("CELERY_BROKER_URL", "redis://redis:6379/0"),
|
celery_broker_url=os.environ.get("CELERY_BROKER_URL", "redis://redis:6379/0"),
|
||||||
celery_result_backend=os.environ.get("CELERY_RESULT_BACKEND", "redis://redis:6379/0"),
|
celery_result_backend=os.environ.get("CELERY_RESULT_BACKEND", "redis://redis:6379/0"),
|
||||||
secret_key=os.environ["SECRET_KEY"],
|
secret_key=os.environ["SECRET_KEY"],
|
||||||
extension_api_key=os.environ.get("EXTENSION_API_KEY", ""),
|
|
||||||
log_level=os.environ.get("LOG_LEVEL", "INFO"),
|
log_level=os.environ.get("LOG_LEVEL", "INFO"),
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -32,6 +32,7 @@ from .presentation_review import PresentationReview
|
|||||||
from .series_chapter import SeriesChapter
|
from .series_chapter import SeriesChapter
|
||||||
from .series_page import SeriesPage
|
from .series_page import SeriesPage
|
||||||
from .series_suggestion import SeriesSuggestion
|
from .series_suggestion import SeriesSuggestion
|
||||||
|
from .service_seen import ServiceSeen
|
||||||
from .source import Source
|
from .source import Source
|
||||||
from .subscribestar_failed_media import SubscribeStarFailedMedia
|
from .subscribestar_failed_media import SubscribeStarFailedMedia
|
||||||
from .subscribestar_seen_media import SubscribeStarSeenMedia
|
from .subscribestar_seen_media import SubscribeStarSeenMedia
|
||||||
@@ -63,6 +64,7 @@ __all__ = [
|
|||||||
"SeriesChapter",
|
"SeriesChapter",
|
||||||
"SeriesPage",
|
"SeriesPage",
|
||||||
"SeriesSuggestion",
|
"SeriesSuggestion",
|
||||||
|
"ServiceSeen",
|
||||||
"ImageRecord",
|
"ImageRecord",
|
||||||
"ImageProvenance",
|
"ImageProvenance",
|
||||||
"ImageRegion",
|
"ImageRegion",
|
||||||
|
|||||||
@@ -27,10 +27,10 @@ class Artist(Base):
|
|||||||
notes: Mapped[str | None] = mapped_column(Text, nullable=True)
|
notes: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
|
||||||
# True once a Source is attached; flips false if all sources removed.
|
# True once a Source is attached; flips false if all sources removed.
|
||||||
is_subscription: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False)
|
is_subscription: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, server_default="false")
|
||||||
|
|
||||||
# Per-artist scheduling overrides; null means "use global default".
|
# Per-artist scheduling overrides; null means "use global default".
|
||||||
auto_check: Mapped[bool] = mapped_column(Boolean, nullable=False, default=True)
|
auto_check: Mapped[bool] = mapped_column(Boolean, nullable=False, default=True, server_default="true")
|
||||||
check_interval_seconds: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
check_interval_seconds: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
||||||
|
|
||||||
created_at: Mapped[datetime] = mapped_column(
|
created_at: Mapped[datetime] = mapped_column(
|
||||||
|
|||||||
@@ -20,7 +20,7 @@ feedback_check_existing_enums):
|
|||||||
|
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
|
||||||
from sqlalchemy import JSON, BigInteger, DateTime, ForeignKey, Integer, String, Text
|
from sqlalchemy import JSON, BigInteger, DateTime, ForeignKey, Index, Integer, String, Text, text
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
from .base import Base
|
from .base import Base
|
||||||
@@ -29,10 +29,21 @@ from .base import Base
|
|||||||
class BackupRun(Base):
|
class BackupRun(Base):
|
||||||
__tablename__ = "backup_run"
|
__tablename__ = "backup_run"
|
||||||
|
|
||||||
|
|
||||||
|
__table_args__ = (
|
||||||
|
# alembic 0017: reporting indexes, never declared on the model (#3275).
|
||||||
|
Index("ix_backup_run_kind_started", "kind", text("started_at DESC")),
|
||||||
|
Index("ix_backup_run_status_finished", "status", text("finished_at DESC")),
|
||||||
|
Index("ix_backup_run_tag_partial", "tag", postgresql_where=text("tag IS NOT NULL")),
|
||||||
|
)
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
kind: Mapped[str] = mapped_column(String(16), nullable=False, index=True)
|
# No index=True: ix_backup_run_kind_started (above) already leads with
|
||||||
|
# `kind`, so a single-column index on it was pure write cost (#3301).
|
||||||
|
kind: Mapped[str] = mapped_column(String(16), nullable=False)
|
||||||
status: Mapped[str] = mapped_column(
|
status: Mapped[str] = mapped_column(
|
||||||
String(16), nullable=False, default="pending", index=True,
|
# No index=True — ix_backup_run_status_finished leads with `status`.
|
||||||
|
String(16), nullable=False, default="pending",
|
||||||
|
server_default="pending",
|
||||||
)
|
)
|
||||||
tag: Mapped[str | None] = mapped_column(String(64), nullable=True, index=True)
|
tag: Mapped[str | None] = mapped_column(String(64), nullable=True, index=True)
|
||||||
triggered_by: Mapped[str] = mapped_column(String(32), nullable=False)
|
triggered_by: Mapped[str] = mapped_column(String(32), nullable=False)
|
||||||
@@ -49,7 +60,9 @@ class BackupRun(Base):
|
|||||||
manifest: Mapped[dict] = mapped_column(
|
manifest: Mapped[dict] = mapped_column(
|
||||||
JSON, nullable=False, default=dict, server_default="{}",
|
JSON, nullable=False, default=dict, server_default="{}",
|
||||||
)
|
)
|
||||||
|
# Self-referential FK, unindexed until 0089 (#3300): SET NULL has to find
|
||||||
|
# the rows pointing at a deleted run before it can null them.
|
||||||
restored_from_id: Mapped[int | None] = mapped_column(
|
restored_from_id: Mapped[int | None] = mapped_column(
|
||||||
ForeignKey("backup_run.id", ondelete="SET NULL"),
|
ForeignKey("backup_run.id", ondelete="SET NULL"),
|
||||||
nullable=True,
|
nullable=True, index=True,
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -40,8 +40,10 @@ class CharacterPrototype(Base):
|
|||||||
)
|
)
|
||||||
# Provenance: the region this vector was copied from. SET NULL so pruning a
|
# Provenance: the region this vector was copied from. SET NULL so pruning a
|
||||||
# region doesn't delete the prototype mid-cycle (the next refresh reconciles).
|
# region doesn't delete the prototype mid-cycle (the next refresh reconciles).
|
||||||
|
# index=True added in 0089 — the FK was unindexed (#3300).
|
||||||
region_id: Mapped[int | None] = mapped_column(
|
region_id: Mapped[int | None] = mapped_column(
|
||||||
ForeignKey("image_region.id", ondelete="SET NULL"), nullable=True
|
ForeignKey("image_region.id", ondelete="SET NULL"), nullable=True,
|
||||||
|
index=True,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -25,8 +25,8 @@ class DownloadEvent(Base):
|
|||||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
)
|
)
|
||||||
finished_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
|
finished_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
|
||||||
bytes_downloaded: Mapped[int] = mapped_column(BigInteger, nullable=False, default=0)
|
bytes_downloaded: Mapped[int] = mapped_column(BigInteger, nullable=False, default=0, server_default="0")
|
||||||
files_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
files_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
metadata_: Mapped[dict] = mapped_column(
|
metadata_: Mapped[dict] = mapped_column(
|
||||||
"metadata", JSONB, nullable=False, default=dict,
|
"metadata", JSONB, nullable=False, default=dict,
|
||||||
|
|||||||
@@ -16,6 +16,7 @@ doesn't delete the link record).
|
|||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
|
||||||
from sqlalchemy import (
|
from sqlalchemy import (
|
||||||
|
CheckConstraint,
|
||||||
DateTime,
|
DateTime,
|
||||||
Float,
|
Float,
|
||||||
ForeignKey,
|
ForeignKey,
|
||||||
@@ -38,15 +39,33 @@ STATUSES = ("pending", "downloading", "downloaded", "failed", "skipped", "dead")
|
|||||||
class ExternalLink(Base):
|
class ExternalLink(Base):
|
||||||
__tablename__ = "external_link"
|
__tablename__ = "external_link"
|
||||||
__table_args__ = (
|
__table_args__ = (
|
||||||
|
# alembic 0028 enum CHECKs. Rule 36 territory: a new host or status value
|
||||||
|
# needs its constraint swapped in the same migration (#3275).
|
||||||
|
CheckConstraint(
|
||||||
|
"host IN ('mega', 'gdrive', 'mediafire', 'dropbox', 'pixeldrain')",
|
||||||
|
# Bare name: Base.metadata's naming convention prepends
|
||||||
|
# ck_<table>_. Pre-prefixing it here doubles the prefix — see
|
||||||
|
# alembic 0088, which renames the four constraints that shipped
|
||||||
|
# that way (#3275).
|
||||||
|
name="host",
|
||||||
|
),
|
||||||
|
CheckConstraint(
|
||||||
|
"status IN ('pending', 'downloading', 'downloaded', 'failed', 'skipped', 'dead')",
|
||||||
|
name="status",
|
||||||
|
),
|
||||||
# One row per (post, url). The full url (incl. #fragment) is the identity
|
# One row per (post, url). The full url (incl. #fragment) is the identity
|
||||||
# — the same file linked twice in a post collapses to one row.
|
# — the same file linked twice in a post collapses to one row.
|
||||||
Index("uq_external_link_post_url", "post_id", "url", unique=True),
|
Index("uq_external_link_post_url", "post_id", "url", unique=True),
|
||||||
Index("ix_external_link_status", "status"),
|
Index("ix_external_link_status", "status"),
|
||||||
|
# Unindexed FK (#3300).
|
||||||
|
Index("ix_external_link_attachment_id", "attachment_id"),
|
||||||
)
|
)
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
|
# No index=True: uq_external_link_post_url (post_id, url) already leads
|
||||||
|
# with post_id (#3301).
|
||||||
post_id: Mapped[int] = mapped_column(
|
post_id: Mapped[int] = mapped_column(
|
||||||
ForeignKey("post.id", ondelete="CASCADE"), nullable=False, index=True
|
ForeignKey("post.id", ondelete="CASCADE"), nullable=False
|
||||||
)
|
)
|
||||||
artist_id: Mapped[int | None] = mapped_column(
|
artist_id: Mapped[int | None] = mapped_column(
|
||||||
ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, index=True
|
ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, index=True
|
||||||
|
|||||||
@@ -50,7 +50,8 @@ class GpuJob(Base):
|
|||||||
# What to compute, e.g. 'ccip' (detect figures + CCIP-embed) or 'siglip_region'.
|
# What to compute, e.g. 'ccip' (detect figures + CCIP-embed) or 'siglip_region'.
|
||||||
task: Mapped[str] = mapped_column(String(32), nullable=False)
|
task: Mapped[str] = mapped_column(String(32), nullable=False)
|
||||||
status: Mapped[str] = mapped_column(
|
status: Mapped[str] = mapped_column(
|
||||||
String(16), nullable=False, default="pending", index=True
|
String(16), nullable=False, default="pending", index=True,
|
||||||
|
server_default="pending",
|
||||||
)
|
)
|
||||||
# pending | leased | done | error
|
# pending | leased | done | error
|
||||||
lease_token: Mapped[str | None] = mapped_column(String(64), nullable=True)
|
lease_token: Mapped[str | None] = mapped_column(String(64), nullable=True)
|
||||||
@@ -60,7 +61,7 @@ class GpuJob(Base):
|
|||||||
lease_expires_at: Mapped[datetime | None] = mapped_column(
|
lease_expires_at: Mapped[datetime | None] = mapped_column(
|
||||||
DateTime(timezone=True), nullable=True
|
DateTime(timezone=True), nullable=True
|
||||||
)
|
)
|
||||||
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
# Triage verdict for an ERRORED job (#125): NULL = not yet probed;
|
# Triage verdict for an ERRORED job (#125): NULL = not yet probed;
|
||||||
# 'defect' = the integrity probe says the FILE itself is bad (surfaced for
|
# 'defect' = the integrity probe says the FILE itself is bad (surfaced for
|
||||||
|
|||||||
@@ -24,10 +24,11 @@ class HeadAutoApplyRun(Base):
|
|||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
# dry_run=True is a PREVIEW: scores + counts what WOULD apply, writes nothing
|
# dry_run=True is a PREVIEW: scores + counts what WOULD apply, writes nothing
|
||||||
# (preview/apply parity, rule 93).
|
# (preview/apply parity, rule 93).
|
||||||
dry_run: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False)
|
dry_run: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, server_default="false")
|
||||||
params: Mapped[dict[str, Any]] = mapped_column(JSONB, nullable=False)
|
params: Mapped[dict[str, Any]] = mapped_column(JSONB, nullable=False)
|
||||||
status: Mapped[str] = mapped_column(
|
status: Mapped[str] = mapped_column(
|
||||||
String(16), nullable=False, default="running", index=True
|
String(16), nullable=False, default="running", index=True,
|
||||||
|
server_default="running",
|
||||||
)
|
)
|
||||||
# running | ready | error
|
# running | ready | error
|
||||||
started_at: Mapped[datetime] = mapped_column(
|
started_at: Mapped[datetime] = mapped_column(
|
||||||
|
|||||||
@@ -24,9 +24,9 @@ class HeadMetric(Base):
|
|||||||
ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True
|
ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True
|
||||||
)
|
)
|
||||||
# An auto-applied (source='head_auto') tag the operator later REMOVED.
|
# An auto-applied (source='head_auto') tag the operator later REMOVED.
|
||||||
n_misfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
n_misfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
# A tag with a head that the operator added by HAND (the head missed it).
|
# A tag with a head that the operator added by HAND (the head missed it).
|
||||||
n_underfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
n_underfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
updated_at: Mapped[datetime] = mapped_column(
|
updated_at: Mapped[datetime] = mapped_column(
|
||||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -19,8 +19,14 @@ class HeadMetricsSnapshot(Base):
|
|||||||
__tablename__ = "head_metrics_snapshot"
|
__tablename__ = "head_metrics_snapshot"
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
tag_id: Mapped[int] = mapped_column(
|
# Nullable, matching alembic 0060, which declared this column without
|
||||||
ForeignKey("tag.id", ondelete="CASCADE"), index=True
|
# `nullable=False`. The model had it as `Mapped[int]` — NOT NULL — which
|
||||||
|
# was simply never true of the database (#3275). Left nullable rather than
|
||||||
|
# tightened: a snapshot of a tag that is later hard-deleted is a row worth
|
||||||
|
# keeping, and the FK is ON DELETE CASCADE, so tightening it would only
|
||||||
|
# change behaviour, not correct a bug.
|
||||||
|
tag_id: Mapped[int | None] = mapped_column(
|
||||||
|
ForeignKey("tag.id", ondelete="CASCADE"), nullable=True, index=True
|
||||||
)
|
)
|
||||||
# Denormalized so a snapshot stays readable even if the tag is later renamed.
|
# Denormalized so a snapshot stays readable even if the tag is later renamed.
|
||||||
name: Mapped[str] = mapped_column(String(255), nullable=False)
|
name: Mapped[str] = mapped_column(String(255), nullable=False)
|
||||||
@@ -28,9 +34,9 @@ class HeadMetricsSnapshot(Base):
|
|||||||
DateTime(timezone=True), nullable=False, server_default=func.now(), index=True
|
DateTime(timezone=True), nullable=False, server_default=func.now(), index=True
|
||||||
)
|
)
|
||||||
# Current count of source='head_auto' applications still standing.
|
# Current count of source='head_auto' applications still standing.
|
||||||
n_auto_applied: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
n_auto_applied: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
n_misfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
n_misfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
n_underfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
n_underfires: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
# The head's measured quality at snapshot time (null if no head exists).
|
# The head's measured quality at snapshot time (null if no head exists).
|
||||||
ap: Mapped[float | None] = mapped_column(Float, nullable=True)
|
ap: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
precision_cv: Mapped[float | None] = mapped_column(Float, nullable=True)
|
precision_cv: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
|||||||
@@ -24,7 +24,8 @@ class HeadTrainingRun(Base):
|
|||||||
# Training parameters: {min_positives, neg_ratio, precision_target, ...}.
|
# Training parameters: {min_positives, neg_ratio, precision_target, ...}.
|
||||||
params: Mapped[dict[str, Any]] = mapped_column(JSONB, nullable=False)
|
params: Mapped[dict[str, Any]] = mapped_column(JSONB, nullable=False)
|
||||||
status: Mapped[str] = mapped_column(
|
status: Mapped[str] = mapped_column(
|
||||||
String(16), nullable=False, default="running", index=True
|
String(16), nullable=False, default="running", index=True,
|
||||||
|
server_default="running",
|
||||||
)
|
)
|
||||||
# running | ready | error
|
# running | ready | error
|
||||||
started_at: Mapped[datetime] = mapped_column(
|
started_at: Mapped[datetime] = mapped_column(
|
||||||
|
|||||||
@@ -47,8 +47,15 @@ class ImageProvenance(Base):
|
|||||||
# attachment on the post. NULL for loose downloads and pre-backfill rows.
|
# attachment on the post. NULL for loose downloads and pre-backfill rows.
|
||||||
# SET NULL so deleting the archive attachment never destroys the (image,
|
# SET NULL so deleting the archive attachment never destroys the (image,
|
||||||
# post) edge — it just forgets which archive it came from.
|
# post) edge — it just forgets which archive it came from.
|
||||||
|
# FK named explicitly: the convention renders this
|
||||||
|
# `fk_image_provenance_from_attachment_id_post_attachment`, but alembic
|
||||||
|
# 0055 created it as `fk_image_provenance_from_attachment` (#3275).
|
||||||
from_attachment_id: Mapped[int | None] = mapped_column(
|
from_attachment_id: Mapped[int | None] = mapped_column(
|
||||||
ForeignKey("post_attachment.id", ondelete="SET NULL"),
|
ForeignKey(
|
||||||
|
"post_attachment.id",
|
||||||
|
ondelete="SET NULL",
|
||||||
|
name="fk_image_provenance_from_attachment",
|
||||||
|
),
|
||||||
nullable=True, index=True,
|
nullable=True, index=True,
|
||||||
)
|
)
|
||||||
captured_metadata: Mapped[dict | None] = mapped_column(JSON, nullable=True)
|
captured_metadata: Mapped[dict | None] = mapped_column(JSON, nullable=True)
|
||||||
|
|||||||
@@ -14,10 +14,13 @@ from sqlalchemy import (
|
|||||||
Enum,
|
Enum,
|
||||||
Float,
|
Float,
|
||||||
ForeignKey,
|
ForeignKey,
|
||||||
|
Index,
|
||||||
Integer,
|
Integer,
|
||||||
String,
|
String,
|
||||||
Text,
|
Text,
|
||||||
|
UniqueConstraint,
|
||||||
func,
|
func,
|
||||||
|
text,
|
||||||
)
|
)
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
@@ -29,11 +32,38 @@ ORIGIN_CHOICES = ("downloaded", "imported_filesystem", "uploaded")
|
|||||||
class ImageRecord(Base):
|
class ImageRecord(Base):
|
||||||
__tablename__ = "image_record"
|
__tablename__ = "image_record"
|
||||||
|
|
||||||
|
|
||||||
|
__table_args__ = (
|
||||||
|
# alembic 0001. The database enforces sha256 uniqueness with a
|
||||||
|
# CONSTRAINT and carries a SEPARATE non-unique btree index; the model
|
||||||
|
# said `unique=True, index=True`, which collapses both into a single
|
||||||
|
# UNIQUE index under a different name. Same guarantee either way, but
|
||||||
|
# not the same objects, so autogenerate saw a drop and an add (#3275).
|
||||||
|
UniqueConstraint("sha256", name="uq_image_record_sha256"),
|
||||||
|
# alembic 0036, and the last thing in this schema that lived only in a
|
||||||
|
# migration. SQLAlchemy CAN express an hnsw index with an operator
|
||||||
|
# class, so there is no reason for it to be invisible to the models —
|
||||||
|
# and its absence was the quietest failure of the lot: everything
|
||||||
|
# works, similarity search just silently stops using an index.
|
||||||
|
Index(
|
||||||
|
"ix_image_record_siglip_hnsw",
|
||||||
|
"siglip_embedding",
|
||||||
|
postgresql_using="hnsw",
|
||||||
|
postgresql_ops={"siglip_embedding": "vector_cosine_ops"},
|
||||||
|
),
|
||||||
|
# alembic 0035/0071: the date-ordered browse indexes (#3275).
|
||||||
|
Index("ix_image_record_effective_date", text("effective_date DESC"), text("id DESC")),
|
||||||
|
Index("ix_image_record_earliest_post_date", text("earliest_post_date DESC"), text("id DESC")),
|
||||||
|
)
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
|
|
||||||
# On-disk identity
|
# On-disk identity
|
||||||
path: Mapped[str] = mapped_column(Text, nullable=False, unique=True)
|
path: Mapped[str] = mapped_column(Text, nullable=False, unique=True)
|
||||||
sha256: Mapped[str] = mapped_column(String(64), nullable=False, unique=True, index=True)
|
# Neither unique= nor index=: uq_image_record_sha256 in __table_args__
|
||||||
|
# above creates its own index, and the separate ix_image_record_sha256
|
||||||
|
# that 0001 also built was an exact duplicate of it — dropped in 0089
|
||||||
|
# (#3301). Lookups by sha256 use the constraint's index.
|
||||||
|
sha256: Mapped[str] = mapped_column(String(64), nullable=False)
|
||||||
phash: Mapped[str | None] = mapped_column(String(32), nullable=True, index=True)
|
phash: Mapped[str | None] = mapped_column(String(32), nullable=True, index=True)
|
||||||
size_bytes: Mapped[int] = mapped_column(BigInteger, nullable=False)
|
size_bytes: Mapped[int] = mapped_column(BigInteger, nullable=False)
|
||||||
mime: Mapped[str] = mapped_column(String(64), nullable=False)
|
mime: Mapped[str] = mapped_column(String(64), nullable=False)
|
||||||
@@ -47,7 +77,8 @@ class ImageRecord(Base):
|
|||||||
# Integrity verification status. FC-2e populates this; FC-2a leaves rows at 'unknown'.
|
# Integrity verification status. FC-2e populates this; FC-2a leaves rows at 'unknown'.
|
||||||
# Values: 'unknown' (default), 'ok', 'corrupt', 'failed_verification'.
|
# Values: 'unknown' (default), 'ok', 'corrupt', 'failed_verification'.
|
||||||
integrity_status: Mapped[str] = mapped_column(
|
integrity_status: Mapped[str] = mapped_column(
|
||||||
String(24), nullable=False, default="unknown", index=True
|
String(24), nullable=False, default="unknown", index=True,
|
||||||
|
server_default="unknown",
|
||||||
)
|
)
|
||||||
|
|
||||||
# Thumbnail (populated by FC-2)
|
# Thumbnail (populated by FC-2)
|
||||||
@@ -72,8 +103,15 @@ class ImageRecord(Base):
|
|||||||
)
|
)
|
||||||
# FC-2d-vii-c: canonical per-image artist (the single source of truth
|
# FC-2d-vii-c: canonical per-image artist (the single source of truth
|
||||||
# for attribution; provenance posts remain lineage detail).
|
# for attribution; provenance posts remain lineage detail).
|
||||||
|
# FK named explicitly: the naming convention renders this
|
||||||
|
# `fk_image_record_artist_id_artist`, but alembic 0008 created it as
|
||||||
|
# `fk_image_record_artist_id` (#3275).
|
||||||
artist_id: Mapped[int | None] = mapped_column(
|
artist_id: Mapped[int | None] = mapped_column(
|
||||||
ForeignKey("artist.id", ondelete="SET NULL"), nullable=True, index=True
|
ForeignKey(
|
||||||
|
"artist.id", ondelete="SET NULL", name="fk_image_record_artist_id"
|
||||||
|
),
|
||||||
|
nullable=True,
|
||||||
|
index=True,
|
||||||
)
|
)
|
||||||
|
|
||||||
# ML fields (populated by the ml-worker / GPU agent). 1152 = SigLIP-so400m
|
# ML fields (populated by the ml-worker / GPU agent). 1152 = SigLIP-so400m
|
||||||
|
|||||||
@@ -21,17 +21,17 @@ class ImportBatch(Base):
|
|||||||
)
|
)
|
||||||
finished_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
|
finished_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
|
||||||
|
|
||||||
total_files: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
total_files: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
imported: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
imported: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
skipped: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
skipped: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
failed: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
failed: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
attachments: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
attachments: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
# Deep-scan only: count of already-imported files whose sidecar metadata
|
# Deep-scan only: count of already-imported files whose sidecar metadata
|
||||||
# got re-applied this run (post/source/provenance upsert). Stays 0 on
|
# got re-applied this run (post/source/provenance upsert). Stays 0 on
|
||||||
# quick-scan batches. See `Importer.import_one(deep_scan=True)`.
|
# quick-scan batches. See `Importer.import_one(deep_scan=True)`.
|
||||||
refreshed: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
refreshed: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
|
|
||||||
status: Mapped[str] = mapped_column(String(16), nullable=False, default="running", index=True)
|
status: Mapped[str] = mapped_column(String(16), nullable=False, default="running", index=True, server_default="running")
|
||||||
# running | complete | cancelled
|
# running | complete | cancelled
|
||||||
|
|
||||||
tasks = relationship("ImportTask", back_populates="batch", cascade="all, delete-orphan")
|
tasks = relationship("ImportTask", back_populates="batch", cascade="all, delete-orphan")
|
||||||
|
|||||||
@@ -4,7 +4,15 @@ Enforced as a single row via a CHECK (id = 1) constraint. The application
|
|||||||
always SELECTs id=1 and never inserts/deletes after the initial migration.
|
always SELECTs id=1 and never inserts/deletes after the initial migration.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from sqlalchemy import Boolean, CheckConstraint, Float, Integer, Text, select
|
from sqlalchemy import (
|
||||||
|
Boolean,
|
||||||
|
CheckConstraint,
|
||||||
|
Float,
|
||||||
|
Integer,
|
||||||
|
Text,
|
||||||
|
select,
|
||||||
|
text,
|
||||||
|
)
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
from .base import Base
|
from .base import Base
|
||||||
@@ -14,63 +22,79 @@ class ImportSettings(Base):
|
|||||||
__tablename__ = "import_settings"
|
__tablename__ = "import_settings"
|
||||||
# Bare constraint name — Base.metadata's naming convention applies the
|
# Bare constraint name — Base.metadata's naming convention applies the
|
||||||
# ck_<table>_<name> prefix, producing the final ck_import_settings_singleton.
|
# ck_<table>_<name> prefix, producing the final ck_import_settings_singleton.
|
||||||
|
# Bare name — Base.metadata's naming convention prepends ck_<table>_,
|
||||||
|
# producing ck_import_settings_singleton. The chain shipped the DOUBLED
|
||||||
|
# ck_import_settings_ck_import_settings_singleton, because the migration
|
||||||
|
# pre-prefixed the name and the convention prefixed it again; alembic
|
||||||
|
# 0088 renames it to what this line has always produced (#3275).
|
||||||
__table_args__ = (CheckConstraint("id = 1", name="singleton"),)
|
__table_args__ = (CheckConstraint("id = 1", name="singleton"),)
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
import_scan_path: Mapped[str] = mapped_column(Text, nullable=False, default="/import")
|
import_scan_path: Mapped[str] = mapped_column(Text, nullable=False, default="/import", server_default="/import")
|
||||||
|
|
||||||
min_width: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
min_width: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
min_height: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
min_height: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
|
|
||||||
skip_transparent: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False)
|
skip_transparent: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, server_default="false")
|
||||||
transparency_threshold: Mapped[float] = mapped_column(Float, nullable=False, default=0.9)
|
transparency_threshold: Mapped[float] = mapped_column(Float, nullable=False, default=0.9, server_default="0.9")
|
||||||
|
|
||||||
skip_single_color: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False)
|
skip_single_color: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, server_default="false")
|
||||||
single_color_threshold: Mapped[float] = mapped_column(Float, nullable=False, default=0.95)
|
single_color_threshold: Mapped[float] = mapped_column(Float, nullable=False, default=0.95, server_default="0.95")
|
||||||
single_color_tolerance: Mapped[int] = mapped_column(Integer, nullable=False, default=30)
|
single_color_tolerance: Mapped[int] = mapped_column(Integer, nullable=False, default=30, server_default="30")
|
||||||
|
|
||||||
phash_threshold: Mapped[int] = mapped_column(Integer, nullable=False, default=10)
|
phash_threshold: Mapped[int] = mapped_column(Integer, nullable=False, default=10, server_default="10")
|
||||||
|
|
||||||
# FC-3c downloader knobs
|
# FC-3c downloader knobs
|
||||||
download_rate_limit_seconds: Mapped[float] = mapped_column(
|
download_rate_limit_seconds: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=3.0
|
Float, nullable=False, default=3.0,
|
||||||
|
server_default="3",
|
||||||
)
|
)
|
||||||
download_validate_files: Mapped[bool] = mapped_column(
|
download_validate_files: Mapped[bool] = mapped_column(
|
||||||
Boolean, nullable=False, default=True
|
Boolean, nullable=False, default=True,
|
||||||
|
server_default="true",
|
||||||
)
|
)
|
||||||
|
|
||||||
# FC-3d scheduling knobs
|
# FC-3d scheduling knobs
|
||||||
download_schedule_default_seconds: Mapped[int] = mapped_column(
|
download_schedule_default_seconds: Mapped[int] = mapped_column(
|
||||||
Integer, nullable=False, default=28800
|
Integer, nullable=False, default=28800,
|
||||||
|
server_default="28800",
|
||||||
)
|
)
|
||||||
download_event_retention_days: Mapped[int] = mapped_column(
|
download_event_retention_days: Mapped[int] = mapped_column(
|
||||||
Integer, nullable=False, default=90
|
Integer, nullable=False, default=90,
|
||||||
|
server_default="90",
|
||||||
)
|
)
|
||||||
download_failure_warning_threshold: Mapped[int] = mapped_column(
|
download_failure_warning_threshold: Mapped[int] = mapped_column(
|
||||||
Integer, nullable=False, default=5
|
Integer, nullable=False, default=5,
|
||||||
|
server_default="5",
|
||||||
)
|
)
|
||||||
|
|
||||||
# FC-3h backup knobs.
|
# FC-3h backup knobs.
|
||||||
backup_db_nightly_enabled: Mapped[bool] = mapped_column(
|
backup_db_nightly_enabled: Mapped[bool] = mapped_column(
|
||||||
Boolean, nullable=False, default=False,
|
Boolean, nullable=False, default=False,
|
||||||
|
server_default="false",
|
||||||
)
|
)
|
||||||
backup_db_nightly_hour_utc: Mapped[int] = mapped_column(
|
backup_db_nightly_hour_utc: Mapped[int] = mapped_column(
|
||||||
Integer, nullable=False, default=3,
|
Integer, nullable=False, default=3,
|
||||||
|
server_default="3",
|
||||||
)
|
)
|
||||||
backup_db_keep_last_n: Mapped[int] = mapped_column(
|
backup_db_keep_last_n: Mapped[int] = mapped_column(
|
||||||
Integer, nullable=False, default=14,
|
Integer, nullable=False, default=14,
|
||||||
|
server_default="14",
|
||||||
)
|
)
|
||||||
backup_images_keep_last_n: Mapped[int] = mapped_column(
|
backup_images_keep_last_n: Mapped[int] = mapped_column(
|
||||||
Integer, nullable=False, default=3,
|
Integer, nullable=False, default=3,
|
||||||
|
server_default="3",
|
||||||
)
|
)
|
||||||
|
|
||||||
# FC-6.3 series continuation matcher. enabled gates the rescan; threshold is
|
# FC-6.3 series continuation matcher. enabled gates the rescan; threshold is
|
||||||
# the weighted-score cut-off (0..1) above which a pending suggestion is made.
|
# the weighted-score cut-off (0..1) above which a pending suggestion is made.
|
||||||
series_suggest_enabled: Mapped[bool] = mapped_column(
|
series_suggest_enabled: Mapped[bool] = mapped_column(
|
||||||
Boolean, nullable=False, default=True,
|
Boolean, nullable=False, default=True,
|
||||||
|
server_default="true",
|
||||||
)
|
)
|
||||||
series_suggest_threshold: Mapped[float] = mapped_column(
|
series_suggest_threshold: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=0.5,
|
Float, nullable=False, default=0.5,
|
||||||
|
server_default="0.5",
|
||||||
)
|
)
|
||||||
|
|
||||||
# #830 off-platform file-host downloads — per-host enable lever (default on,
|
# #830 off-platform file-host downloads — per-host enable lever (default on,
|
||||||
@@ -113,7 +137,9 @@ class ImportSettings(Base):
|
|||||||
# English (e.g. "… WIP Part 1") as a European language at ~0.86. CJK stays
|
# English (e.g. "… WIP Part 1") as a European language at ~0.86. CJK stays
|
||||||
# trusted regardless (script-detected). Per-post overrides handle the misses.
|
# trusted regardless (script-detected). Per-post overrides handle the misses.
|
||||||
translation_min_confidence: Mapped[float] = mapped_column(
|
translation_min_confidence: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=0.9, server_default="0.9",
|
# text() because alembic 0084 used sa.text(); see ml_settings for why
|
||||||
|
# the form matters and why it is per-column (#3275).
|
||||||
|
Float, nullable=False, default=0.9, server_default=text("0.9"),
|
||||||
)
|
)
|
||||||
|
|
||||||
# Title-based WIP auto-tagging (task #1458). When a freshly-imported post's
|
# Title-based WIP auto-tagging (task #1458). When a freshly-imported post's
|
||||||
|
|||||||
@@ -13,10 +13,12 @@ from sqlalchemy import (
|
|||||||
Boolean,
|
Boolean,
|
||||||
DateTime,
|
DateTime,
|
||||||
ForeignKey,
|
ForeignKey,
|
||||||
|
Index,
|
||||||
Integer,
|
Integer,
|
||||||
String,
|
String,
|
||||||
Text,
|
Text,
|
||||||
func,
|
func,
|
||||||
|
text,
|
||||||
)
|
)
|
||||||
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
||||||
|
|
||||||
@@ -26,6 +28,12 @@ from .base import Base
|
|||||||
class ImportTask(Base):
|
class ImportTask(Base):
|
||||||
__tablename__ = "import_task"
|
__tablename__ = "import_task"
|
||||||
|
|
||||||
|
|
||||||
|
__table_args__ = (
|
||||||
|
Index("ix_import_task_created_at_desc", text("created_at DESC")),
|
||||||
|
# Unindexed FK (#3300).
|
||||||
|
Index("ix_import_task_result_image_id", "result_image_id"),
|
||||||
|
)
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
batch_id: Mapped[int] = mapped_column(
|
batch_id: Mapped[int] = mapped_column(
|
||||||
ForeignKey("import_batch.id", ondelete="CASCADE"), nullable=False, index=True
|
ForeignKey("import_batch.id", ondelete="CASCADE"), nullable=False, index=True
|
||||||
@@ -33,14 +41,14 @@ class ImportTask(Base):
|
|||||||
|
|
||||||
source_path: Mapped[str] = mapped_column(Text, nullable=False)
|
source_path: Mapped[str] = mapped_column(Text, nullable=False)
|
||||||
task_type: Mapped[str] = mapped_column(String(16), nullable=False) # media|archive
|
task_type: Mapped[str] = mapped_column(String(16), nullable=False) # media|archive
|
||||||
status: Mapped[str] = mapped_column(String(16), nullable=False, default="pending", index=True)
|
status: Mapped[str] = mapped_column(String(16), nullable=False, default="pending", index=True, server_default="pending")
|
||||||
|
|
||||||
# Poison-pill circuit breaker (alembic 0026). recovery_count tracks
|
# Poison-pill circuit breaker (alembic 0026). recovery_count tracks
|
||||||
# how many times the stuck-task sweep has re-queued this row; after
|
# how many times the stuck-task sweep has re-queued this row; after
|
||||||
# the cap it's failed with a diagnostic instead of looping. refetched
|
# the cap it's failed with a diagnostic instead of looping. refetched
|
||||||
# bounds the one-shot re-download remediation to a single attempt.
|
# bounds the one-shot re-download remediation to a single attempt.
|
||||||
recovery_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
recovery_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
refetched: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False)
|
refetched: Mapped[bool] = mapped_column(Boolean, nullable=False, default=False, server_default="false")
|
||||||
|
|
||||||
result_image_id: Mapped[int | None] = mapped_column(
|
result_image_id: Mapped[int | None] = mapped_column(
|
||||||
ForeignKey("image_record.id", ondelete="SET NULL"), nullable=True
|
ForeignKey("image_record.id", ondelete="SET NULL"), nullable=True
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ reads it and routes through cleanup_service.delete_images.
|
|||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
||||||
from sqlalchemy import DateTime, Integer, String, Text, func
|
from sqlalchemy import DateTime, Integer, String, Text, func, text
|
||||||
from sqlalchemy.dialects.postgresql import JSONB
|
from sqlalchemy.dialects.postgresql import JSONB
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
@@ -23,6 +23,7 @@ class LibraryAuditRun(Base):
|
|||||||
params: Mapped[dict[str, Any]] = mapped_column(JSONB, nullable=False)
|
params: Mapped[dict[str, Any]] = mapped_column(JSONB, nullable=False)
|
||||||
status: Mapped[str] = mapped_column(
|
status: Mapped[str] = mapped_column(
|
||||||
String(16), nullable=False, default="running", index=True,
|
String(16), nullable=False, default="running", index=True,
|
||||||
|
server_default="running",
|
||||||
)
|
)
|
||||||
# running | ready | applied | cancelled | error
|
# running | ready | applied | cancelled | error
|
||||||
started_at: Mapped[datetime] = mapped_column(
|
started_at: Mapped[datetime] = mapped_column(
|
||||||
@@ -31,14 +32,16 @@ class LibraryAuditRun(Base):
|
|||||||
finished_at: Mapped[datetime | None] = mapped_column(
|
finished_at: Mapped[datetime | None] = mapped_column(
|
||||||
DateTime(timezone=True), nullable=True,
|
DateTime(timezone=True), nullable=True,
|
||||||
)
|
)
|
||||||
scanned_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
scanned_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
matched_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
matched_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
matched_ids: Mapped[list[int]] = mapped_column(JSONB, nullable=False, default=list)
|
matched_ids: Mapped[list[int]] = mapped_column(
|
||||||
|
JSONB, nullable=False, default=list, server_default=text("'[]'::jsonb")
|
||||||
|
)
|
||||||
error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
# Chunked-scan state (alembic 0039): keyset cursor the next chunk resumes
|
# Chunked-scan state (alembic 0039): keyset cursor the next chunk resumes
|
||||||
# from, and the last time a chunk made progress (so the recovery sweep can
|
# from, and the last time a chunk made progress (so the recovery sweep can
|
||||||
# tell a progressing multi-chunk audit from a stuck one).
|
# tell a progressing multi-chunk audit from a stuck one).
|
||||||
resume_after_id: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
resume_after_id: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
last_progress_at: Mapped[datetime | None] = mapped_column(
|
last_progress_at: Mapped[datetime | None] = mapped_column(
|
||||||
DateTime(timezone=True), nullable=True,
|
DateTime(timezone=True), nullable=True,
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -11,6 +11,7 @@ from sqlalchemy import (
|
|||||||
String,
|
String,
|
||||||
func,
|
func,
|
||||||
select,
|
select,
|
||||||
|
text,
|
||||||
)
|
)
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
@@ -20,7 +21,10 @@ from .base import Base
|
|||||||
class MLSettings(Base):
|
class MLSettings(Base):
|
||||||
__tablename__ = "ml_settings"
|
__tablename__ = "ml_settings"
|
||||||
# Bare name — Base.metadata's naming convention prepends ck_<table>_,
|
# Bare name — Base.metadata's naming convention prepends ck_<table>_,
|
||||||
# producing the final ck_ml_settings_singleton (matches migration 0003).
|
# producing ck_ml_settings_singleton. The chain shipped the DOUBLED
|
||||||
|
# ck_ml_settings_ck_ml_settings_singleton, because the migration
|
||||||
|
# pre-prefixed the name and the convention prefixed it again; alembic
|
||||||
|
# 0088 renames it to what this line has always produced (#3275).
|
||||||
__table_args__ = (CheckConstraint("id = 1", name="singleton"),)
|
__table_args__ = (CheckConstraint("id = 1", name="singleton"),)
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
@@ -31,17 +35,20 @@ class MLSettings(Base):
|
|||||||
# queueing embed work nothing will consume (the daily GPU 'embed' backfill
|
# queueing embed work nothing will consume (the daily GPU 'embed' backfill
|
||||||
# covers those images instead).
|
# covers those images instead).
|
||||||
cpu_embed_enabled: Mapped[bool] = mapped_column(
|
cpu_embed_enabled: Mapped[bool] = mapped_column(
|
||||||
Boolean, nullable=False, default=True
|
Boolean, nullable=False, default=True,
|
||||||
|
server_default="true",
|
||||||
)
|
)
|
||||||
# Video embedding (#747). Sample one frame every N seconds (fixed CADENCE, not
|
# Video embedding (#747). Sample one frame every N seconds (fixed CADENCE, not
|
||||||
# a fixed count) so coverage reflects real screen time regardless of length;
|
# a fixed count) so coverage reflects real screen time regardless of length;
|
||||||
# cap the total so a long video can't explode into hundreds of embeds. The
|
# cap the total so a long video can't explode into hundreds of embeds. The
|
||||||
# per-frame SigLIP embeddings are mean-pooled. Operator-tunable.
|
# per-frame SigLIP embeddings are mean-pooled. Operator-tunable.
|
||||||
video_frame_interval_seconds: Mapped[float] = mapped_column(
|
video_frame_interval_seconds: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=4.0
|
Float, nullable=False, default=4.0,
|
||||||
|
server_default="4",
|
||||||
)
|
)
|
||||||
video_max_frames: Mapped[int] = mapped_column(
|
video_max_frames: Mapped[int] = mapped_column(
|
||||||
Integer, nullable=False, default=64
|
Integer, nullable=False, default=64,
|
||||||
|
server_default="64",
|
||||||
)
|
)
|
||||||
# Tagging-v2 head training (#114). The head is the suggestion source that
|
# Tagging-v2 head training (#114). The head is the suggestion source that
|
||||||
# LEARNS from the operator's tags (replacing Camie + centroid). A concept
|
# LEARNS from the operator's tags (replacing Camie + centroid). A concept
|
||||||
@@ -49,10 +56,12 @@ class MLSettings(Base):
|
|||||||
# head_auto_apply_precision is the precision bar a head must clear (at some
|
# head_auto_apply_precision is the precision bar a head must clear (at some
|
||||||
# operating point) to "graduate" into earned auto-apply. Operator-tunable.
|
# operating point) to "graduate" into earned auto-apply. Operator-tunable.
|
||||||
head_min_positives: Mapped[int] = mapped_column(
|
head_min_positives: Mapped[int] = mapped_column(
|
||||||
Integer, nullable=False, default=8
|
Integer, nullable=False, default=8,
|
||||||
|
server_default="8",
|
||||||
)
|
)
|
||||||
head_auto_apply_precision: Mapped[float] = mapped_column(
|
head_auto_apply_precision: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=0.97
|
Float, nullable=False, default=0.97,
|
||||||
|
server_default="0.97",
|
||||||
)
|
)
|
||||||
# Earned auto-apply (#114). A graduated head fires (tags images without a
|
# Earned auto-apply (#114). A graduated head fires (tags images without a
|
||||||
# human) when this master switch is on AND the head has at least
|
# human) when this master switch is on AND the head has at least
|
||||||
@@ -61,29 +70,34 @@ class MLSettings(Base):
|
|||||||
# default (operator-asked 2026-06-29: opt-OUT, not opt-in); the support +
|
# default (operator-asked 2026-06-29: opt-OUT, not opt-in); the support +
|
||||||
# measured-precision gates keep it safe, and every auto-tag is reversible.
|
# measured-precision gates keep it safe, and every auto-tag is reversible.
|
||||||
head_auto_apply_enabled: Mapped[bool] = mapped_column(
|
head_auto_apply_enabled: Mapped[bool] = mapped_column(
|
||||||
Boolean, nullable=False, default=True
|
Boolean, nullable=False, default=True,
|
||||||
|
server_default="true",
|
||||||
)
|
)
|
||||||
head_auto_apply_min_positives: Mapped[int] = mapped_column(
|
head_auto_apply_min_positives: Mapped[int] = mapped_column(
|
||||||
# Support floor raised 30→50 (operator-asked 2026-07-06): a head needs
|
# Support floor raised 30→50 (operator-asked 2026-07-06): a head needs
|
||||||
# more human labels before it may fire without a human.
|
# more human labels before it may fire without a human.
|
||||||
Integer, nullable=False, default=50
|
Integer, nullable=False, default=50,
|
||||||
|
server_default="30",
|
||||||
)
|
)
|
||||||
# CCIP character-match cosine cut (#114). 0.85 default — the v1 flat 0.75
|
# CCIP character-match cosine cut (#114). 0.85 default — the v1 flat 0.75
|
||||||
# over-fired (high-reference characters matched a scatter of images); 0.85
|
# over-fired (high-reference characters matched a scatter of images); 0.85
|
||||||
# keeps the confident single-character matches. Tunable from the agent card.
|
# keeps the confident single-character matches. Tunable from the agent card.
|
||||||
ccip_match_threshold: Mapped[float] = mapped_column(
|
ccip_match_threshold: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=0.85
|
Float, nullable=False, default=0.85,
|
||||||
|
server_default="0.85",
|
||||||
)
|
)
|
||||||
# CCIP auto-apply (#114). Confident matches (>= ccip_auto_apply_threshold,
|
# CCIP auto-apply (#114). Confident matches (>= ccip_auto_apply_threshold,
|
||||||
# above the suggest cut) auto-tag on a daily sweep. ON by default (opt-out);
|
# above the suggest cut) auto-tag on a daily sweep. ON by default (opt-out);
|
||||||
# single-character references + the high bar keep it safe, every tag reversible.
|
# single-character references + the high bar keep it safe, every tag reversible.
|
||||||
ccip_auto_apply_enabled: Mapped[bool] = mapped_column(
|
ccip_auto_apply_enabled: Mapped[bool] = mapped_column(
|
||||||
Boolean, nullable=False, default=True
|
Boolean, nullable=False, default=True,
|
||||||
|
server_default="true",
|
||||||
)
|
)
|
||||||
ccip_auto_apply_threshold: Mapped[float] = mapped_column(
|
ccip_auto_apply_threshold: Mapped[float] = mapped_column(
|
||||||
# Raised 0.92→0.95 (operator-asked 2026-07-06) so only very confident
|
# Raised 0.92→0.95 (operator-asked 2026-07-06) so only very confident
|
||||||
# character matches auto-tag.
|
# character matches auto-tag.
|
||||||
Float, nullable=False, default=0.95
|
Float, nullable=False, default=0.95,
|
||||||
|
server_default="0.92",
|
||||||
)
|
)
|
||||||
# -- Presentation chrome auto-hide (#141) -------------------------------
|
# -- Presentation chrome auto-hide (#141) -------------------------------
|
||||||
# `banner` (chrome — clusters on UI, not content) auto-applies on the sweep
|
# `banner` (chrome — clusters on UI, not content) auto-applies on the sweep
|
||||||
@@ -95,13 +109,21 @@ class MLSettings(Base):
|
|||||||
# (opt-out); every auto-tag is reversible. NOTE (#1464): `wip` + `editor
|
# (opt-out); every auto-tag is reversible. NOTE (#1464): `wip` + `editor
|
||||||
# screenshot` are no longer chrome — they went to the PROCESS path below.
|
# screenshot` are no longer chrome — they went to the PROCESS path below.
|
||||||
presentation_auto_apply_enabled: Mapped[bool] = mapped_column(
|
presentation_auto_apply_enabled: Mapped[bool] = mapped_column(
|
||||||
Boolean, nullable=False, default=True
|
Boolean, nullable=False, default=True,
|
||||||
|
server_default="true",
|
||||||
)
|
)
|
||||||
presentation_auto_apply_threshold: Mapped[float] = mapped_column(
|
presentation_auto_apply_threshold: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=0.90
|
Float, nullable=False, default=0.90,
|
||||||
|
# text(), not a string, because alembic 0082 used sa.text(): a bare
|
||||||
|
# string renders DEFAULT '0.90'::double precision while text() renders
|
||||||
|
# DEFAULT 0.90, and the chain is MIXED — some migrations used one,
|
||||||
|
# some the other. Same value, different stored expression, so each
|
||||||
|
# column here mirrors whichever form its own migration used (#3275).
|
||||||
|
server_default=text("0.90"),
|
||||||
)
|
)
|
||||||
presentation_conflict_threshold: Mapped[float] = mapped_column(
|
presentation_conflict_threshold: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=0.50
|
Float, nullable=False, default=0.50,
|
||||||
|
server_default=text("0.50"),
|
||||||
)
|
)
|
||||||
# -- Process auto-apply (#1464) ----------------------------------------
|
# -- Process auto-apply (#1464) ----------------------------------------
|
||||||
# `wip` / `editor screenshot` are PROCESS art — unfinished pieces + program
|
# `wip` / `editor screenshot` are PROCESS art — unfinished pieces + program
|
||||||
@@ -115,24 +137,29 @@ class MLSettings(Base):
|
|||||||
# (PresentationReview, mode='process') rather than silently marked. OFF by
|
# (PresentationReview, mode='process') rather than silently marked. OFF by
|
||||||
# default — a new whole-library auto-tagger is opt-in; every auto-tag reversible.
|
# default — a new whole-library auto-tagger is opt-in; every auto-tag reversible.
|
||||||
process_auto_apply_enabled: Mapped[bool] = mapped_column(
|
process_auto_apply_enabled: Mapped[bool] = mapped_column(
|
||||||
Boolean, nullable=False, default=False
|
Boolean, nullable=False, default=False,
|
||||||
|
server_default="false",
|
||||||
)
|
)
|
||||||
process_auto_apply_threshold: Mapped[float] = mapped_column(
|
process_auto_apply_threshold: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=0.90
|
Float, nullable=False, default=0.90,
|
||||||
|
server_default="0.90",
|
||||||
)
|
)
|
||||||
process_conflict_threshold: Mapped[float] = mapped_column(
|
process_conflict_threshold: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=0.50
|
Float, nullable=False, default=0.50,
|
||||||
|
server_default="0.50",
|
||||||
)
|
)
|
||||||
# Default = SigLIP 2 (so400m, 512px) for new installs (migration 0069);
|
# Default = SigLIP 2 (so400m, 512px) for new installs (migration 0069);
|
||||||
# existing libraries keep their stored value until the operator re-embeds.
|
# existing libraries keep their stored value until the operator re-embeds.
|
||||||
embedder_model_version: Mapped[str] = mapped_column(
|
embedder_model_version: Mapped[str] = mapped_column(
|
||||||
String(128), nullable=False, default="siglip2-so400m-patch16-512"
|
String(128), nullable=False, default="siglip2-so400m-patch16-512",
|
||||||
|
server_default="siglip2-so400m-patch16-512",
|
||||||
)
|
)
|
||||||
# The HF model NAME the embedder loads (server CPU embed + announced to the
|
# The HF model NAME the embedder loads (server CPU embed + announced to the
|
||||||
# GPU agent in the lease). Operator-settable so the embedder is a choice, not
|
# GPU agent in the lease). Operator-settable so the embedder is a choice, not
|
||||||
# a hardcode (#1190): set name + version together, then re-embed + retrain.
|
# a hardcode (#1190): set name + version together, then re-embed + retrain.
|
||||||
embedder_model_name: Mapped[str] = mapped_column(
|
embedder_model_name: Mapped[str] = mapped_column(
|
||||||
String(128), nullable=False, default="google/siglip2-so400m-patch16-512"
|
String(128), nullable=False, default="google/siglip2-so400m-patch16-512",
|
||||||
|
server_default="google/siglip2-so400m-patch16-512",
|
||||||
)
|
)
|
||||||
# -- Crop proposers / detectors (#1202, #134) --------------------------
|
# -- Crop proposers / detectors (#1202, #134) --------------------------
|
||||||
# WHERE-to-crop YOLO detectors feeding the crop→SigLIP bag + CCIP. Config
|
# WHERE-to-crop YOLO detectors feeding the crop→SigLIP bag + CCIP. Config
|
||||||
@@ -145,20 +172,24 @@ class MLSettings(Base):
|
|||||||
# person: general COCO figure detector for Western/realistic art the anime
|
# person: general COCO figure detector for Western/realistic art the anime
|
||||||
# person-detector misses → NMS-merged with imgutils → CCIP + concept.
|
# person-detector misses → NMS-merged with imgutils → CCIP + concept.
|
||||||
detector_person_enabled: Mapped[bool] = mapped_column(
|
detector_person_enabled: Mapped[bool] = mapped_column(
|
||||||
Boolean, nullable=False, default=True
|
Boolean, nullable=False, default=True,
|
||||||
|
server_default="true",
|
||||||
)
|
)
|
||||||
detector_person_weights: Mapped[str] = mapped_column(
|
detector_person_weights: Mapped[str] = mapped_column(
|
||||||
String(512), nullable=False, default="yolo11n.pt"
|
String(512), nullable=False, default="yolo11n.pt",
|
||||||
|
server_default="yolo11n.pt",
|
||||||
)
|
)
|
||||||
detector_person_conf: Mapped[float] = mapped_column(
|
detector_person_conf: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=0.35
|
Float, nullable=False, default=0.35,
|
||||||
|
server_default=text("0.35"),
|
||||||
)
|
)
|
||||||
# anatomy: booru_yolo anime/furry/NSFW torso components → concept crops.
|
# anatomy: booru_yolo anime/furry/NSFW torso components → concept crops.
|
||||||
# Default = yolov11m_aa22 (26 classes, best mAP50-95 0.96), committed in the
|
# Default = yolov11m_aa22 (26 classes, best mAP50-95 0.96), committed in the
|
||||||
# upstream repo so the URL resolves. License UNSTATED — fine for a private
|
# upstream repo so the URL resolves. License UNSTATED — fine for a private
|
||||||
# homelab (operator accepted #1202).
|
# homelab (operator accepted #1202).
|
||||||
detector_anatomy_enabled: Mapped[bool] = mapped_column(
|
detector_anatomy_enabled: Mapped[bool] = mapped_column(
|
||||||
Boolean, nullable=False, default=True
|
Boolean, nullable=False, default=True,
|
||||||
|
server_default="true",
|
||||||
)
|
)
|
||||||
detector_anatomy_weights: Mapped[str] = mapped_column(
|
detector_anatomy_weights: Mapped[str] = mapped_column(
|
||||||
String(512), nullable=False,
|
String(512), nullable=False,
|
||||||
@@ -166,37 +197,47 @@ class MLSettings(Base):
|
|||||||
"https://github.com/aperveyev/booru_yolo/raw/main/models/"
|
"https://github.com/aperveyev/booru_yolo/raw/main/models/"
|
||||||
"yolov11m_aa22.pt"
|
"yolov11m_aa22.pt"
|
||||||
),
|
),
|
||||||
|
server_default="https://github.com/aperveyev/booru_yolo/raw/main/models/yolov11m_aa22.pt",
|
||||||
)
|
)
|
||||||
detector_anatomy_conf: Mapped[float] = mapped_column(
|
detector_anatomy_conf: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=0.30
|
Float, nullable=False, default=0.30,
|
||||||
|
server_default=text("0.30"),
|
||||||
)
|
)
|
||||||
# panel: comic page → panel regions → concept crops (Apache-2.0, YOLOv12x).
|
# panel: comic page → panel regions → concept crops (Apache-2.0, YOLOv12x).
|
||||||
detector_panel_enabled: Mapped[bool] = mapped_column(
|
detector_panel_enabled: Mapped[bool] = mapped_column(
|
||||||
Boolean, nullable=False, default=True
|
Boolean, nullable=False, default=True,
|
||||||
|
server_default="true",
|
||||||
)
|
)
|
||||||
detector_panel_weights: Mapped[str] = mapped_column(
|
detector_panel_weights: Mapped[str] = mapped_column(
|
||||||
String(512), nullable=False,
|
String(512), nullable=False,
|
||||||
default="mosesb/best-comic-panel-detection::best.pt",
|
default="mosesb/best-comic-panel-detection::best.pt",
|
||||||
|
server_default="mosesb/best-comic-panel-detection::best.pt",
|
||||||
)
|
)
|
||||||
detector_panel_conf: Mapped[float] = mapped_column(
|
detector_panel_conf: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=0.30
|
Float, nullable=False, default=0.30,
|
||||||
|
server_default=text("0.30"),
|
||||||
)
|
)
|
||||||
# Per-frame caps bound the crop→embed explosion; max_regions is the hard
|
# Per-frame caps bound the crop→embed explosion; max_regions is the hard
|
||||||
# per-job backstop; dedupe_iou drops near-duplicate crops before the embed.
|
# per-job backstop; dedupe_iou drops near-duplicate crops before the embed.
|
||||||
detector_max_figures: Mapped[int] = mapped_column(
|
detector_max_figures: Mapped[int] = mapped_column(
|
||||||
Integer, nullable=False, default=8
|
Integer, nullable=False, default=8,
|
||||||
|
server_default="8",
|
||||||
)
|
)
|
||||||
detector_max_components: Mapped[int] = mapped_column(
|
detector_max_components: Mapped[int] = mapped_column(
|
||||||
Integer, nullable=False, default=8
|
Integer, nullable=False, default=8,
|
||||||
|
server_default="8",
|
||||||
)
|
)
|
||||||
detector_max_panels: Mapped[int] = mapped_column(
|
detector_max_panels: Mapped[int] = mapped_column(
|
||||||
Integer, nullable=False, default=8
|
Integer, nullable=False, default=8,
|
||||||
|
server_default="8",
|
||||||
)
|
)
|
||||||
detector_max_regions: Mapped[int] = mapped_column(
|
detector_max_regions: Mapped[int] = mapped_column(
|
||||||
Integer, nullable=False, default=128
|
Integer, nullable=False, default=128,
|
||||||
|
server_default="128",
|
||||||
)
|
)
|
||||||
detector_dedupe_iou: Mapped[float] = mapped_column(
|
detector_dedupe_iou: Mapped[float] = mapped_column(
|
||||||
Float, nullable=False, default=0.85
|
Float, nullable=False, default=0.85,
|
||||||
|
server_default=text("0.85"),
|
||||||
)
|
)
|
||||||
# -- CCIP character prototypes (#1317) ---------------------------------
|
# -- CCIP character prototypes (#1317) ---------------------------------
|
||||||
# The per-character reference set is precomputed + refreshed INCREMENTALLY
|
# The per-character reference set is precomputed + refreshed INCREMENTALLY
|
||||||
@@ -208,7 +249,8 @@ class MLSettings(Base):
|
|||||||
String(128), nullable=True
|
String(128), nullable=True
|
||||||
)
|
)
|
||||||
ccip_prototype_cap: Mapped[int] = mapped_column(
|
ccip_prototype_cap: Mapped[int] = mapped_column(
|
||||||
Integer, nullable=False, default=64
|
Integer, nullable=False, default=64,
|
||||||
|
server_default="64",
|
||||||
)
|
)
|
||||||
updated_at: Mapped[datetime] = mapped_column(
|
updated_at: Mapped[datetime] = mapped_column(
|
||||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
|
|||||||
@@ -35,7 +35,7 @@ class PatreonFailedMedia(Base):
|
|||||||
ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True
|
ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True
|
||||||
)
|
)
|
||||||
filehash: Mapped[str] = mapped_column(String(128), nullable=False)
|
filehash: Mapped[str] = mapped_column(String(128), nullable=False)
|
||||||
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1)
|
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1, server_default="1")
|
||||||
last_error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
last_error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
first_failed_at: Mapped[datetime] = mapped_column(
|
first_failed_at: Mapped[datetime] = mapped_column(
|
||||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
|
|||||||
@@ -35,7 +35,7 @@ class PixivFailedMedia(Base):
|
|||||||
ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True
|
ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True
|
||||||
)
|
)
|
||||||
filehash: Mapped[str] = mapped_column(String(128), nullable=False)
|
filehash: Mapped[str] = mapped_column(String(128), nullable=False)
|
||||||
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1)
|
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1, server_default="1")
|
||||||
last_error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
last_error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
first_failed_at: Mapped[datetime] = mapped_column(
|
first_failed_at: Mapped[datetime] = mapped_column(
|
||||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
|
|||||||
@@ -13,11 +13,13 @@ from sqlalchemy import (
|
|||||||
CheckConstraint,
|
CheckConstraint,
|
||||||
DateTime,
|
DateTime,
|
||||||
ForeignKey,
|
ForeignKey,
|
||||||
|
Index,
|
||||||
Integer,
|
Integer,
|
||||||
String,
|
String,
|
||||||
Text,
|
Text,
|
||||||
UniqueConstraint,
|
UniqueConstraint,
|
||||||
func,
|
func,
|
||||||
|
text,
|
||||||
)
|
)
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
@@ -27,6 +29,10 @@ from .base import Base
|
|||||||
class Post(Base):
|
class Post(Base):
|
||||||
__tablename__ = "post"
|
__tablename__ = "post"
|
||||||
__table_args__ = (
|
__table_args__ = (
|
||||||
|
# alembic 0030. The comment above described this index; nothing declared
|
||||||
|
# it, so autogenerate proposed dropping it (#3275).
|
||||||
|
Index("uq_post_artist_external_id_null_source", "artist_id", "external_post_id",
|
||||||
|
unique=True, postgresql_where=text("source_id IS NULL")),
|
||||||
# Source-bound dedup. Postgres treats NULL != NULL so rows
|
# Source-bound dedup. Postgres treats NULL != NULL so rows
|
||||||
# with source_id IS NULL aren't deduped by this constraint;
|
# with source_id IS NULL aren't deduped by this constraint;
|
||||||
# the partial unique index `uq_post_artist_external_id_null_source`
|
# the partial unique index `uq_post_artist_external_id_null_source`
|
||||||
@@ -35,7 +41,11 @@ class Post(Base):
|
|||||||
UniqueConstraint("source_id", "external_post_id", name="uq_post_source_external_id"),
|
UniqueConstraint("source_id", "external_post_id", name="uq_post_source_external_id"),
|
||||||
CheckConstraint(
|
CheckConstraint(
|
||||||
"translation_override IN ('auto', 'force', 'original')",
|
"translation_override IN ('auto', 'force', 'original')",
|
||||||
name="ck_post_translation_override",
|
# Bare name: Base.metadata's naming convention prepends
|
||||||
|
# ck_<table>_. Pre-prefixing it here doubles the prefix — see
|
||||||
|
# alembic 0088, which renames the four constraints that shipped
|
||||||
|
# that way (#3275).
|
||||||
|
name="translation_override",
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ are pruned by retention.
|
|||||||
|
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
|
||||||
from sqlalchemy import DateTime, Float, ForeignKey, String, func
|
from sqlalchemy import DateTime, Float, ForeignKey, Index, String, func
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
from .base import Base
|
from .base import Base
|
||||||
@@ -20,6 +20,14 @@ from .base import Base
|
|||||||
class PresentationReview(Base):
|
class PresentationReview(Base):
|
||||||
__tablename__ = "presentation_review"
|
__tablename__ = "presentation_review"
|
||||||
|
|
||||||
|
|
||||||
|
__table_args__ = (
|
||||||
|
Index("ix_presentation_review_resolved_at", "resolved_at"),
|
||||||
|
# Both FKs to tag were unindexed (#3300); tag_id CASCADEs, so a tag
|
||||||
|
# delete had to scan this table to find its rows.
|
||||||
|
Index("ix_presentation_review_tag_id", "tag_id"),
|
||||||
|
Index("ix_presentation_review_conflict_tag_id", "conflict_tag_id"),
|
||||||
|
)
|
||||||
image_record_id: Mapped[int] = mapped_column(
|
image_record_id: Mapped[int] = mapped_column(
|
||||||
ForeignKey("image_record.id", ondelete="CASCADE"), primary_key=True
|
ForeignKey("image_record.id", ondelete="CASCADE"), primary_key=True
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -16,7 +16,14 @@ title is the optional chapter name; stated_part is the optional operator-facing
|
|||||||
|
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
|
||||||
from sqlalchemy import DateTime, ForeignKey, Integer, Text, func
|
from sqlalchemy import (
|
||||||
|
DateTime,
|
||||||
|
ForeignKey,
|
||||||
|
Integer,
|
||||||
|
Text,
|
||||||
|
UniqueConstraint,
|
||||||
|
func,
|
||||||
|
)
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
from .base import Base
|
from .base import Base
|
||||||
@@ -25,14 +32,26 @@ from .base import Base
|
|||||||
class SeriesChapter(Base):
|
class SeriesChapter(Base):
|
||||||
__tablename__ = "series_chapter"
|
__tablename__ = "series_chapter"
|
||||||
|
|
||||||
|
__table_args__ = (
|
||||||
|
# alembic 0047 named the UNIQUE `uq_series_chapter_anchor_page`, not
|
||||||
|
# the `uq_series_chapter_anchor_page_id` a bare `unique=True` would
|
||||||
|
# render (#3275).
|
||||||
|
UniqueConstraint("anchor_page_id", name="uq_series_chapter_anchor_page"),
|
||||||
|
)
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
series_tag_id: Mapped[int] = mapped_column(
|
series_tag_id: Mapped[int] = mapped_column(
|
||||||
ForeignKey("tag.id", ondelete="CASCADE"), nullable=False, index=True
|
ForeignKey("tag.id", ondelete="CASCADE"), nullable=False, index=True
|
||||||
)
|
)
|
||||||
|
# Both the UNIQUE (above) and the FK carry the names 0047 gave them; the
|
||||||
|
# convention would render the FK `fk_series_chapter_anchor_page_id_series_page`.
|
||||||
anchor_page_id: Mapped[int] = mapped_column(
|
anchor_page_id: Mapped[int] = mapped_column(
|
||||||
ForeignKey("series_page.id", ondelete="CASCADE"),
|
ForeignKey(
|
||||||
|
"series_page.id",
|
||||||
|
ondelete="CASCADE",
|
||||||
|
name="fk_series_chapter_anchor_page",
|
||||||
|
),
|
||||||
nullable=False,
|
nullable=False,
|
||||||
unique=True,
|
|
||||||
)
|
)
|
||||||
title: Mapped[str | None] = mapped_column(Text, nullable=True)
|
title: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
stated_part: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
stated_part: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
||||||
|
|||||||
@@ -14,7 +14,14 @@ number parsed from the source post, nullable when unknown.
|
|||||||
|
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
|
||||||
from sqlalchemy import DateTime, ForeignKey, Integer, String, func
|
from sqlalchemy import (
|
||||||
|
DateTime,
|
||||||
|
ForeignKey,
|
||||||
|
Integer,
|
||||||
|
String,
|
||||||
|
UniqueConstraint,
|
||||||
|
func,
|
||||||
|
)
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
from .base import Base
|
from .base import Base
|
||||||
@@ -23,14 +30,22 @@ from .base import Base
|
|||||||
class SeriesPage(Base):
|
class SeriesPage(Base):
|
||||||
__tablename__ = "series_page"
|
__tablename__ = "series_page"
|
||||||
|
|
||||||
|
__table_args__ = (
|
||||||
|
# alembic 0005 named this `uq_series_page_image`; a bare `unique=True`
|
||||||
|
# on the column renders `uq_series_page_image_id` under the naming
|
||||||
|
# convention, which is a different object from the one the database
|
||||||
|
# has (#3275).
|
||||||
|
UniqueConstraint("image_id", name="uq_series_page_image"),
|
||||||
|
)
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
series_tag_id: Mapped[int] = mapped_column(
|
series_tag_id: Mapped[int] = mapped_column(
|
||||||
ForeignKey("tag.id", ondelete="CASCADE"), nullable=False, index=True
|
ForeignKey("tag.id", ondelete="CASCADE"), nullable=False, index=True
|
||||||
)
|
)
|
||||||
|
# UNIQUE lives in __table_args__ above, under the name 0005 gave it.
|
||||||
image_id: Mapped[int] = mapped_column(
|
image_id: Mapped[int] = mapped_column(
|
||||||
ForeignKey("image_record.id", ondelete="CASCADE"),
|
ForeignKey("image_record.id", ondelete="CASCADE"),
|
||||||
nullable=False,
|
nullable=False,
|
||||||
unique=True,
|
|
||||||
)
|
)
|
||||||
# 'placed' = in the series-global run (page_number set); 'pending' = staged
|
# 'placed' = in the series-global run (page_number set); 'pending' = staged
|
||||||
# from a post awaiting the operator's sort (page_number NULL). (#789 P2)
|
# from a post awaiting the operator's sort (page_number NULL). (#789 P2)
|
||||||
|
|||||||
@@ -0,0 +1,88 @@
|
|||||||
|
"""service_seen — the learned roster of FabledCurator's own moving parts.
|
||||||
|
|
||||||
|
Nothing else in this application knows what is SUPPOSED to be running.
|
||||||
|
`celery inspect` reports the workers that answer, so a stopped worker is a
|
||||||
|
shorter list rather than a red light, and Postgres and Redis have no
|
||||||
|
representation at all. That is why the only place an operator could see a
|
||||||
|
dead service was Portainer, which knows the intended set (milestone 365).
|
||||||
|
|
||||||
|
This table is the memory that makes an absence observable: every part that
|
||||||
|
has ever checked in, and when it last did. A row that stops advancing is a
|
||||||
|
part that stopped.
|
||||||
|
|
||||||
|
## Why the key is not the hostname
|
||||||
|
|
||||||
|
`_read_workers_sync()` returns celery's worker names, which here are
|
||||||
|
`celery@<container id>`. Those are minted fresh on every deploy. Keyed on
|
||||||
|
them, this table would record a death and a birth every time the stack is
|
||||||
|
updated — and a status page that goes red on every deploy is a status page
|
||||||
|
nobody reads, which is worse than not having one.
|
||||||
|
|
||||||
|
So a celery role is keyed on its **queue set**, which is assigned per role in
|
||||||
|
docker-compose.yml (`CELERY_QUEUES`) and survives container replacement:
|
||||||
|
|
||||||
|
default,import,thumbnail,download -> worker
|
||||||
|
maintenance,scan -> scheduler (celery worker --beat)
|
||||||
|
ml -> ml-worker
|
||||||
|
|
||||||
|
Two replicas of one role share a queue set and are therefore ONE row — which
|
||||||
|
is right, because the question being answered is "is that role being served",
|
||||||
|
not "how many containers exist". The replica count and their hostnames go in
|
||||||
|
`details`, where they can change without the identity changing.
|
||||||
|
|
||||||
|
The GPU agent is keyed on its `agent_id`, the identity its lease protocol
|
||||||
|
already uses (`api/gpu.py`).
|
||||||
|
|
||||||
|
## What is NOT in here
|
||||||
|
|
||||||
|
Postgres and Redis. They are always expected and never learned, and a
|
||||||
|
last-seen for them would be actively misleading — that one answered thirty
|
||||||
|
seconds ago says nothing about now. They are probed live at request time.
|
||||||
|
|
||||||
|
## kind
|
||||||
|
|
||||||
|
Plain `String`, not a Postgres ENUM and not CHECK-gated, matching
|
||||||
|
`gpu_job.status` and `backup_run.status`. The value set here is expected to
|
||||||
|
grow as parts are added, and a constraint swap per new kind (rule 36) would
|
||||||
|
be cost with no invariant behind it.
|
||||||
|
|
||||||
|
celery — a worker role, keyed on its queue set
|
||||||
|
agent — a GPU agent, keyed on its agent_id
|
||||||
|
"""
|
||||||
|
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
from sqlalchemy import JSON, DateTime, String, func
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
|
from .base import Base
|
||||||
|
|
||||||
|
|
||||||
|
class ServiceSeen(Base):
|
||||||
|
__tablename__ = "service_seen"
|
||||||
|
|
||||||
|
# No indexes beyond the primary key, deliberately. This table holds one row
|
||||||
|
# per moving part — a handful, forever — so every query against it is a
|
||||||
|
# full read of a few rows and an index would be write cost buying nothing
|
||||||
|
# (the lesson of #3301, which removed seven redundant ones).
|
||||||
|
key: Mapped[str] = mapped_column(String(128), primary_key=True)
|
||||||
|
kind: Mapped[str] = mapped_column(String(16), nullable=False)
|
||||||
|
|
||||||
|
# What to call it in the UI. Derived from the queue set where it is
|
||||||
|
# recognised, and falling back to the raw queue list where it is not — a
|
||||||
|
# deployment that slices its queues differently should still show something
|
||||||
|
# true rather than a name this code invented for it.
|
||||||
|
display_name: Mapped[str] = mapped_column(String(64), nullable=False)
|
||||||
|
|
||||||
|
first_seen_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=False, server_default=func.now(),
|
||||||
|
)
|
||||||
|
last_seen_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=False, server_default=func.now(),
|
||||||
|
)
|
||||||
|
|
||||||
|
# The parts that change without changing identity: replica hostnames,
|
||||||
|
# active task counts, the queues actually being served. Kept as a blob
|
||||||
|
# because it is displayed and never queried — giving it columns would
|
||||||
|
# invite filtering on it, which is what the activity endpoints are for.
|
||||||
|
details: Mapped[dict] = mapped_column(JSON, nullable=False, default=dict)
|
||||||
@@ -5,7 +5,16 @@ Multiple sources per artist support creators with cross-platform presence.
|
|||||||
|
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
|
||||||
from sqlalchemy import JSON, Boolean, DateTime, ForeignKey, Integer, String, Text
|
from sqlalchemy import (
|
||||||
|
JSON,
|
||||||
|
Boolean,
|
||||||
|
DateTime,
|
||||||
|
ForeignKey,
|
||||||
|
Integer,
|
||||||
|
String,
|
||||||
|
Text,
|
||||||
|
UniqueConstraint,
|
||||||
|
)
|
||||||
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
||||||
|
|
||||||
from .base import Base
|
from .base import Base
|
||||||
@@ -14,13 +23,27 @@ from .base import Base
|
|||||||
class Source(Base):
|
class Source(Base):
|
||||||
__tablename__ = "source"
|
__tablename__ = "source"
|
||||||
|
|
||||||
|
__table_args__ = (
|
||||||
|
# alembic 0010. One row per (artist, platform, url): re-adding a source
|
||||||
|
# the artist already has is an update, not a second row. The model had
|
||||||
|
# never declared it (#3275), so autogenerate would have proposed
|
||||||
|
# DROPPING it — the guarantee existed only in the migration chain.
|
||||||
|
#
|
||||||
|
# Named explicitly because the naming convention would render this
|
||||||
|
# `uq_source_artist_id` (uq keys off column_0_name), which is both
|
||||||
|
# wrong about the shape and not what the database actually has.
|
||||||
|
UniqueConstraint(
|
||||||
|
"artist_id", "platform", "url", name="uq_source_artist_platform_url"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
artist_id: Mapped[int] = mapped_column(
|
artist_id: Mapped[int] = mapped_column(
|
||||||
ForeignKey("artist.id", ondelete="CASCADE"), nullable=False, index=True
|
ForeignKey("artist.id", ondelete="CASCADE"), nullable=False, index=True
|
||||||
)
|
)
|
||||||
platform: Mapped[str] = mapped_column(String(64), nullable=False)
|
platform: Mapped[str] = mapped_column(String(64), nullable=False)
|
||||||
url: Mapped[str] = mapped_column(Text, nullable=False)
|
url: Mapped[str] = mapped_column(Text, nullable=False)
|
||||||
enabled: Mapped[bool] = mapped_column(Boolean, nullable=False, default=True)
|
enabled: Mapped[bool] = mapped_column(Boolean, nullable=False, default=True, server_default="true")
|
||||||
|
|
||||||
config_overrides: Mapped[dict | None] = mapped_column(JSON, nullable=True)
|
config_overrides: Mapped[dict | None] = mapped_column(JSON, nullable=True)
|
||||||
|
|
||||||
@@ -32,7 +55,7 @@ class Source(Base):
|
|||||||
# by _update_source_health alongside last_error; cleared on 'ok'.
|
# by _update_source_health alongside last_error; cleared on 'ok'.
|
||||||
error_type: Mapped[str | None] = mapped_column(String(32), nullable=True, index=True)
|
error_type: Mapped[str | None] = mapped_column(String(32), nullable=True, index=True)
|
||||||
check_interval_override: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
check_interval_override: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
||||||
consecutive_failures: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
consecutive_failures: Mapped[int] = mapped_column(Integer, nullable=False, default=0, server_default="0")
|
||||||
|
|
||||||
# alembic 0031: sticky deep-scan budget. When > 0, the next N download
|
# alembic 0031: sticky deep-scan budget. When > 0, the next N download
|
||||||
# runs use gallery-dl's full-walk config (skip: True + 1800s timeout);
|
# runs use gallery-dl's full-walk config (skip: True + 1800s timeout);
|
||||||
|
|||||||
@@ -34,7 +34,7 @@ class SubscribeStarFailedMedia(Base):
|
|||||||
ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True
|
ForeignKey("source.id", ondelete="CASCADE"), nullable=False, index=True
|
||||||
)
|
)
|
||||||
filehash: Mapped[str] = mapped_column(String(128), nullable=False)
|
filehash: Mapped[str] = mapped_column(String(128), nullable=False)
|
||||||
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1)
|
attempts: Mapped[int] = mapped_column(Integer, nullable=False, default=1, server_default="1")
|
||||||
last_error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
last_error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
first_failed_at: Mapped[datetime] = mapped_column(
|
first_failed_at: Mapped[datetime] = mapped_column(
|
||||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
|
|||||||
@@ -15,11 +15,13 @@ from sqlalchemy import (
|
|||||||
Column,
|
Column,
|
||||||
DateTime,
|
DateTime,
|
||||||
ForeignKey,
|
ForeignKey,
|
||||||
|
Index,
|
||||||
Integer,
|
Integer,
|
||||||
String,
|
String,
|
||||||
Table,
|
Table,
|
||||||
false,
|
false,
|
||||||
func,
|
func,
|
||||||
|
text,
|
||||||
)
|
)
|
||||||
from sqlalchemy import (
|
from sqlalchemy import (
|
||||||
Enum as SQLEnum,
|
Enum as SQLEnum,
|
||||||
@@ -67,17 +69,31 @@ image_tag = Table(
|
|||||||
primary_key=True,
|
primary_key=True,
|
||||||
),
|
),
|
||||||
Column("tag_id", ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True),
|
Column("tag_id", ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True),
|
||||||
Column("source", String(32), nullable=False, default="manual"),
|
Column("source", String(32), nullable=False, default="manual", server_default="manual"),
|
||||||
Column("created_at", DateTime(timezone=True), nullable=False, server_default=func.now()),
|
Column("created_at", DateTime(timezone=True), nullable=False, server_default=func.now()),
|
||||||
|
# The PK is (image_record_id, tag_id), which leads with the WRONG column
|
||||||
|
# for the two things that matter most here (#3300): the gallery's tag
|
||||||
|
# filter (tag_query.py builds `image_tag.c.tag_id == tid`) and the
|
||||||
|
# ON DELETE CASCADE from tag, which has to find a tag's rows to remove
|
||||||
|
# them. Without this index both scan the largest table in the schema.
|
||||||
|
Index("ix_image_tag_tag_id", "tag_id"),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
class Tag(Base):
|
class Tag(Base):
|
||||||
__tablename__ = "tag"
|
__tablename__ = "tag"
|
||||||
__table_args__ = (
|
__table_args__ = (
|
||||||
|
# alembic 0002. An EXPRESSION index — COALESCE cannot be expressed as a
|
||||||
|
# UniqueConstraint, which is why it only ever existed in a migration (#3275).
|
||||||
|
Index("uq_tag_name_kind_fandom", "name", "kind", text("COALESCE(fandom_id, 0)"),
|
||||||
|
unique=True),
|
||||||
CheckConstraint(
|
CheckConstraint(
|
||||||
"(fandom_id IS NULL) OR (kind = 'character')",
|
"(fandom_id IS NULL) OR (kind = 'character')",
|
||||||
name="ck_tag_fandom_requires_character",
|
# Bare name: Base.metadata's naming convention prepends
|
||||||
|
# ck_<table>_. Pre-prefixing it here doubles the prefix — see
|
||||||
|
# alembic 0088, which renames the four constraints that shipped
|
||||||
|
# that way (#3275).
|
||||||
|
name="fandom_requires_character",
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -87,6 +103,7 @@ class Tag(Base):
|
|||||||
SQLEnum(TagKind, name="tag_kind", values_callable=lambda e: [m.value for m in e]),
|
SQLEnum(TagKind, name="tag_kind", values_callable=lambda e: [m.value for m in e]),
|
||||||
nullable=False,
|
nullable=False,
|
||||||
default=TagKind.general,
|
default=TagKind.general,
|
||||||
|
server_default="general",
|
||||||
)
|
)
|
||||||
fandom_id: Mapped[int | None] = mapped_column(
|
fandom_id: Mapped[int | None] = mapped_column(
|
||||||
ForeignKey("tag.id", ondelete="SET NULL"), nullable=True, index=True
|
ForeignKey("tag.id", ondelete="SET NULL"), nullable=True, index=True
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ in image_prediction stay unmolested.
|
|||||||
|
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
|
||||||
from sqlalchemy import DateTime, ForeignKey, String, func
|
from sqlalchemy import DateTime, ForeignKey, Index, String, func
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
from .base import Base
|
from .base import Base
|
||||||
@@ -14,10 +14,17 @@ from .base import Base
|
|||||||
class TagAlias(Base):
|
class TagAlias(Base):
|
||||||
__tablename__ = "tag_alias"
|
__tablename__ = "tag_alias"
|
||||||
|
|
||||||
|
|
||||||
|
__table_args__ = (
|
||||||
|
# Named explicitly: the database calls this ix_tag_alias_canonical, while
|
||||||
|
# a bare index=True on the column would generate ix_tag_alias_canonical_tag_id
|
||||||
|
# and silently propose a drop+create on the next autogenerate (#3275).
|
||||||
|
Index("ix_tag_alias_canonical", "canonical_tag_id"),
|
||||||
|
)
|
||||||
alias_string: Mapped[str] = mapped_column(String(255), primary_key=True)
|
alias_string: Mapped[str] = mapped_column(String(255), primary_key=True)
|
||||||
alias_category: Mapped[str] = mapped_column(String(32), primary_key=True)
|
alias_category: Mapped[str] = mapped_column(String(32), primary_key=True)
|
||||||
canonical_tag_id: Mapped[int] = mapped_column(
|
canonical_tag_id: Mapped[int] = mapped_column(
|
||||||
ForeignKey("tag.id", ondelete="CASCADE"), nullable=False, index=True
|
ForeignKey("tag.id", ondelete="CASCADE"), nullable=False
|
||||||
)
|
)
|
||||||
created_at: Mapped[datetime] = mapped_column(
|
created_at: Mapped[datetime] = mapped_column(
|
||||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ Prevents re-suggestion AND prevents allowlist auto-apply on that image.
|
|||||||
|
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
|
||||||
from sqlalchemy import DateTime, ForeignKey, func
|
from sqlalchemy import DateTime, ForeignKey, Index, func
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
from .base import Base
|
from .base import Base
|
||||||
@@ -14,11 +14,24 @@ from .base import Base
|
|||||||
class TagSuggestionRejection(Base):
|
class TagSuggestionRejection(Base):
|
||||||
__tablename__ = "tag_suggestion_rejection"
|
__tablename__ = "tag_suggestion_rejection"
|
||||||
|
|
||||||
|
|
||||||
|
__table_args__ = (
|
||||||
|
# Named explicitly; see tag_alias for why (#3275).
|
||||||
|
Index("ix_tag_suggestion_rejection_tag", "tag_id"),
|
||||||
|
)
|
||||||
|
# Both FKs named explicitly. alembic 0003 used a hand-shortened `tsr`
|
||||||
|
# prefix; the convention would render the full table name (#3275).
|
||||||
image_record_id: Mapped[int] = mapped_column(
|
image_record_id: Mapped[int] = mapped_column(
|
||||||
ForeignKey("image_record.id", ondelete="CASCADE"), primary_key=True
|
ForeignKey(
|
||||||
|
"image_record.id",
|
||||||
|
ondelete="CASCADE",
|
||||||
|
name="fk_tsr_image_record_id_image_record",
|
||||||
|
),
|
||||||
|
primary_key=True,
|
||||||
)
|
)
|
||||||
tag_id: Mapped[int] = mapped_column(
|
tag_id: Mapped[int] = mapped_column(
|
||||||
ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True, index=True
|
ForeignKey("tag.id", ondelete="CASCADE", name="fk_tsr_tag_id_tag"),
|
||||||
|
primary_key=True,
|
||||||
)
|
)
|
||||||
rejected_at: Mapped[datetime] = mapped_column(
|
rejected_at: Mapped[datetime] = mapped_column(
|
||||||
DateTime(timezone=True), nullable=False, server_default=func.now()
|
DateTime(timezone=True), nullable=False, server_default=func.now()
|
||||||
|
|||||||
@@ -15,7 +15,7 @@ backend.app.tasks.maintenance.recover_stalled_task_runs (Beat 5 min).
|
|||||||
|
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
|
||||||
from sqlalchemy import DateTime, Integer, String, Text
|
from sqlalchemy import DateTime, Index, Integer, String, Text, text
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
from .base import Base
|
from .base import Base
|
||||||
@@ -24,12 +24,21 @@ from .base import Base
|
|||||||
class TaskRun(Base):
|
class TaskRun(Base):
|
||||||
__tablename__ = "task_run"
|
__tablename__ = "task_run"
|
||||||
|
|
||||||
|
|
||||||
|
__table_args__ = (
|
||||||
|
# alembic 0016: the three task-history indexes (#3275).
|
||||||
|
Index("ix_task_run_name_started", "task_name", text("started_at DESC")),
|
||||||
|
Index("ix_task_run_queue_started", "queue", text("started_at DESC")),
|
||||||
|
Index("ix_task_run_status_started", "status", text("started_at DESC")),
|
||||||
|
)
|
||||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||||
celery_task_id: Mapped[str] = mapped_column(
|
celery_task_id: Mapped[str] = mapped_column(
|
||||||
String(64), nullable=False, index=True,
|
String(64), nullable=False, index=True,
|
||||||
)
|
)
|
||||||
queue: Mapped[str] = mapped_column(String(32), nullable=False, index=True)
|
# Neither carries index=True: ix_task_run_queue_started and
|
||||||
task_name: Mapped[str] = mapped_column(String(128), nullable=False, index=True)
|
# ix_task_run_name_started already lead with these columns (#3301).
|
||||||
|
queue: Mapped[str] = mapped_column(String(32), nullable=False)
|
||||||
|
task_name: Mapped[str] = mapped_column(String(128), nullable=False)
|
||||||
target_id: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
target_id: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
||||||
started_at: Mapped[datetime] = mapped_column(
|
started_at: Mapped[datetime] = mapped_column(
|
||||||
DateTime(timezone=True), nullable=False, index=True,
|
DateTime(timezone=True), nullable=False, index=True,
|
||||||
@@ -39,7 +48,9 @@ class TaskRun(Base):
|
|||||||
)
|
)
|
||||||
duration_ms: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
duration_ms: Mapped[int | None] = mapped_column(Integer, nullable=True)
|
||||||
status: Mapped[str] = mapped_column(
|
status: Mapped[str] = mapped_column(
|
||||||
String(16), nullable=False, default="running", index=True,
|
# No index=True — ix_task_run_status_started leads with `status`.
|
||||||
|
String(16), nullable=False, default="running",
|
||||||
|
server_default="running",
|
||||||
)
|
)
|
||||||
error_type: Mapped[str | None] = mapped_column(String(128), nullable=True)
|
error_type: Mapped[str | None] = mapped_column(String(128), nullable=True)
|
||||||
error_message: Mapped[str | None] = mapped_column(Text, nullable=True)
|
error_message: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
|||||||
@@ -0,0 +1,175 @@
|
|||||||
|
"""The learned roster: which of FabledCurator's parts have checked in, and when.
|
||||||
|
|
||||||
|
Milestone 365. `celery inspect` answers "who is here"; this answers "who is
|
||||||
|
missing", which nothing in the application could do before — see
|
||||||
|
`models/service_seen.py` for why the identity is a queue set and not a
|
||||||
|
worker hostname.
|
||||||
|
|
||||||
|
## Who does the observing, and why it is the web process
|
||||||
|
|
||||||
|
Three candidates, and the choice matters more than the code:
|
||||||
|
|
||||||
|
* **A celery beat sweep.** Rejected. If the scheduler dies, the sweep stops,
|
||||||
|
every row goes stale, and the page reports that everything is down when one
|
||||||
|
thing is. An alarm that cannot distinguish "one part died" from "the
|
||||||
|
observer died" is worse than no alarm.
|
||||||
|
* **A background task in web.** Rejected on a detail of how this deploys:
|
||||||
|
hypercorn runs `--workers 4`, so a `before_serving` loop would be FOUR
|
||||||
|
concurrent inspect loops hammering the broker, forever, per container.
|
||||||
|
* **Refresh on demand, rate-limited by the data itself.** Taken. Whichever web
|
||||||
|
process happens to serve a health request refreshes the roster if it is
|
||||||
|
older than REFRESH_TTL, and otherwise reads what is already there.
|
||||||
|
|
||||||
|
The third has the property the other two lack: **the observer is the thing
|
||||||
|
serving the page.** If web is down you get a browser error rather than a
|
||||||
|
confidently green page, which is the honest failure. It also self-limits
|
||||||
|
without coordination — the TTL lives in the row everybody can see.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import logging
|
||||||
|
|
||||||
|
from sqlalchemy import func, select
|
||||||
|
from sqlalchemy.dialects.postgresql import insert as pg_insert
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from ..models import ServiceSeen
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# How stale the roster may be before a health request refreshes it. Comfortably
|
||||||
|
# under the staleness thresholds that decide a service is missing, so the
|
||||||
|
# verdict is never limited by how often anyone looked.
|
||||||
|
REFRESH_TTL_SECONDS = 20.0
|
||||||
|
|
||||||
|
# celery inspect is a broker round trip and this sits on a request path, so it
|
||||||
|
# gets a deadline (rule 156). A broker that has stopped answering must make the
|
||||||
|
# roster stale — which is a true statement about the system — not hang the one
|
||||||
|
# page that exists to explain it.
|
||||||
|
INSPECT_TIMEOUT_SECONDS = 2.0
|
||||||
|
|
||||||
|
# Queue set -> the name an operator recognises. Sorted-tuple keys, because the
|
||||||
|
# order celery reports them in is not guaranteed.
|
||||||
|
#
|
||||||
|
# A deployment that slices CELERY_QUEUES differently falls through to the raw
|
||||||
|
# queue list rather than being given a name this table invented for it: a
|
||||||
|
# wrong-but-confident label on a status page is worse than an ugly true one.
|
||||||
|
ROLE_NAMES: dict[tuple[str, ...], str] = {
|
||||||
|
("default", "download", "import", "thumbnail"): "Worker",
|
||||||
|
("maintenance", "scan"): "Scheduler",
|
||||||
|
("ml",): "ML worker",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def role_display_name(queues: tuple[str, ...]) -> str:
|
||||||
|
known = ROLE_NAMES.get(queues)
|
||||||
|
if known:
|
||||||
|
return known
|
||||||
|
return "Worker (" + ", ".join(queues) + ")"
|
||||||
|
|
||||||
|
|
||||||
|
def _inspect_celery_sync() -> dict[tuple[str, ...], dict]:
|
||||||
|
"""celery inspect, grouped by queue set rather than by worker.
|
||||||
|
|
||||||
|
Returns {queue_set: {"hostnames": [...], "active": int}}. Two replicas of
|
||||||
|
one role collapse into one entry on purpose — the question is whether the
|
||||||
|
role is being served, not how many containers exist.
|
||||||
|
"""
|
||||||
|
from ..celery_app import celery as celery_app
|
||||||
|
|
||||||
|
insp = celery_app.control.inspect(timeout=INSPECT_TIMEOUT_SECONDS)
|
||||||
|
active_queues = insp.active_queues() or {}
|
||||||
|
active_tasks = insp.active() or {}
|
||||||
|
|
||||||
|
grouped: dict[tuple[str, ...], dict] = {}
|
||||||
|
for hostname, queues in active_queues.items():
|
||||||
|
key = tuple(sorted({q["name"] for q in queues}))
|
||||||
|
entry = grouped.setdefault(key, {"hostnames": [], "active": 0})
|
||||||
|
entry["hostnames"].append(hostname)
|
||||||
|
entry["active"] += len(active_tasks.get(hostname, []))
|
||||||
|
for entry in grouped.values():
|
||||||
|
entry["hostnames"].sort()
|
||||||
|
return grouped
|
||||||
|
|
||||||
|
|
||||||
|
async def touch_service(
|
||||||
|
session: AsyncSession, *, key: str, kind: str, display_name: str, details: dict
|
||||||
|
) -> None:
|
||||||
|
"""Record that a part checked in just now.
|
||||||
|
|
||||||
|
Upsert rather than read-modify-write: several web processes and several
|
||||||
|
agents can be doing this at once, and the last writer is simply the most
|
||||||
|
recent sighting. `first_seen_at` is deliberately NOT updated — it is the
|
||||||
|
one field that answers "has this ever run", which the learned-roster design
|
||||||
|
depends on.
|
||||||
|
"""
|
||||||
|
stmt = pg_insert(ServiceSeen).values(
|
||||||
|
key=key, kind=kind, display_name=display_name, details=details,
|
||||||
|
)
|
||||||
|
stmt = stmt.on_conflict_do_update(
|
||||||
|
index_elements=[ServiceSeen.key],
|
||||||
|
set_={
|
||||||
|
"kind": stmt.excluded.kind,
|
||||||
|
"display_name": stmt.excluded.display_name,
|
||||||
|
"details": stmt.excluded.details,
|
||||||
|
"last_seen_at": func.now(),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
await session.execute(stmt)
|
||||||
|
|
||||||
|
|
||||||
|
async def refresh_celery_roster(session: AsyncSession) -> None:
|
||||||
|
"""Inspect the broker and record what answered. Never raises.
|
||||||
|
|
||||||
|
A failure here means the roster does not advance, and the rows going stale
|
||||||
|
is then a TRUE report about a broker nobody can reach. Letting the
|
||||||
|
exception out would instead break the health endpoint, which is the one
|
||||||
|
thing that must keep answering when the stack is unwell.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
grouped = await asyncio.wait_for(
|
||||||
|
asyncio.to_thread(_inspect_celery_sync),
|
||||||
|
timeout=INSPECT_TIMEOUT_SECONDS * 2,
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
log.warning("service roster: celery inspect failed; roster not refreshed", exc_info=True)
|
||||||
|
return
|
||||||
|
|
||||||
|
for queues, entry in grouped.items():
|
||||||
|
await touch_service(
|
||||||
|
session,
|
||||||
|
key="celery:" + ",".join(queues),
|
||||||
|
kind="celery",
|
||||||
|
display_name=role_display_name(queues),
|
||||||
|
details={
|
||||||
|
"queues": list(queues),
|
||||||
|
"hostnames": entry["hostnames"],
|
||||||
|
"replicas": len(entry["hostnames"]),
|
||||||
|
"active": entry["active"],
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def refresh_if_stale(session: AsyncSession) -> None:
|
||||||
|
"""Refresh the celery roster if nobody has for REFRESH_TTL_SECONDS.
|
||||||
|
|
||||||
|
Rate-limited by the data rather than by a lock: the gate is the newest
|
||||||
|
last_seen_at across the celery rows, which every web process can see. Two
|
||||||
|
processes racing through the gate costs one redundant inspect and writes
|
||||||
|
the same values twice, so the benign outcome needs no coordination to
|
||||||
|
prevent.
|
||||||
|
"""
|
||||||
|
newest = (
|
||||||
|
await session.execute(
|
||||||
|
select(func.max(ServiceSeen.last_seen_at)).where(ServiceSeen.kind == "celery")
|
||||||
|
)
|
||||||
|
).scalar_one_or_none()
|
||||||
|
|
||||||
|
if newest is not None:
|
||||||
|
age = (await session.execute(select(func.now()))).scalar_one() - newest
|
||||||
|
if age.total_seconds() < REFRESH_TTL_SECONDS:
|
||||||
|
return
|
||||||
|
|
||||||
|
await refresh_celery_roster(session)
|
||||||
+15
-8
@@ -198,14 +198,21 @@ per `docs/process.md`'s "add deps to the image when used by >1 project".
|
|||||||
refresh from being undone.
|
refresh from being undone.
|
||||||
- **`pull: true` on the scheduled path only** is the mechanism: a moved base
|
- **`pull: true` on the scheduled path only** is the mechanism: a moved base
|
||||||
tag changes the `FROM` layer's cache key and everything above it rebuilds.
|
tag changes the `FROM` layer's cache key and everything above it rebuilds.
|
||||||
**It does not currently make the unmoved case free.** Measured on the first
|
It did not always make the unmoved case free. Measured on the first real
|
||||||
real fire (run 4934, 2026-08-30): every content step reported `CACHED` and
|
fire (run 4934, 2026-08-30): every content step reported `CACHED` and the
|
||||||
the bases resolved to unchanged digests, yet all three `:latest` tags got a
|
bases resolved to unchanged digests, yet all three `:latest` tags got a NEW
|
||||||
NEW manifest digest, because buildkit mints a fresh image config per run and
|
manifest digest, because buildkit stamps a fresh image config per run and
|
||||||
republishes identical layers under it. So `:latest` is rewritten weekly
|
republishes identical layers under it — so `:latest` was rewritten weekly
|
||||||
whether or not anything changed, and `:c-<sha>` is handed a new manifest to
|
whether or not anything changed, and a digest change stopped meaning
|
||||||
diverge from on the same cadence — a digest change stops meaning anything.
|
anything (#3265).
|
||||||
Tracked as #3265; the likely fix is a deterministic `SOURCE_DATE_EPOCH`.
|
- **`SOURCE_DATE_EPOCH` is what makes it free.** Set on each build step from
|
||||||
|
`artifacts.sh epoch <artifact>` — the unix timestamp of the same commit
|
||||||
|
`revision` and `version` name, so all three are views of one `newest()`
|
||||||
|
lookup and cannot drift into disagreeing. With the config's `created` field
|
||||||
|
and history timestamps pinned to the content rather than to the wall clock,
|
||||||
|
identical source produces an identical manifest digest and the push is a
|
||||||
|
registry no-op. That restores the property the whole scheme rests on: a
|
||||||
|
channel tag's digest changes when, and only when, its content does.
|
||||||
Separately not caught: a Debian package update inside the `apt-get install`
|
Separately not caught: a Debian package update inside the `apt-get install`
|
||||||
layer while the base tag stands still — a lag rather than a hole, since the
|
layer while the base tag stands still — a lag rather than a hole, since the
|
||||||
official python/cuda images rebuild with those updates baked in.
|
official python/cuda images rebuild with those updates baked in.
|
||||||
|
|||||||
+69
-20
@@ -1,12 +1,21 @@
|
|||||||
# Base compose stack. Uses ${VAR:-default} interpolation throughout so the
|
# Base compose stack, and the install path. Uses ${VAR:-default} throughout so
|
||||||
# stack boots with zero config — sane dev defaults baked in. For production
|
# the stack boots with zero config — but those defaults are DEV defaults, and
|
||||||
# deployments, override the defaults via shell env vars or a .env file:
|
# two of them (DB_PASSWORD, SECRET_KEY) are published in this file. Copy
|
||||||
|
# .env.example to .env and set them before running this anywhere real.
|
||||||
#
|
#
|
||||||
# DB_PASSWORD=...real... SECRET_KEY=...real... docker compose up
|
# To run FabledCurator:
|
||||||
#
|
#
|
||||||
# The dev override (docker-compose.override.yml) is auto-merged when you
|
# docker compose -f docker-compose.yml up -d
|
||||||
# run `docker compose up` from this directory and switches images to
|
#
|
||||||
# local builds + DEBUG logging.
|
# The -f is load-bearing. Without it Compose auto-merges
|
||||||
|
# docker-compose.override.yml, which replaces every image: with a local
|
||||||
|
# build: and turns on DEBUG logging — the contributor path. Naming this file
|
||||||
|
# explicitly skips the override and pulls the published :latest images.
|
||||||
|
#
|
||||||
|
# FabledCurator has no authentication. Whatever can reach ${PORT} is an
|
||||||
|
# administrator, including over the stored Patreon/SubscribeStar/Pixiv session
|
||||||
|
# cookies. Do not publish this port beyond a network you trust — see
|
||||||
|
# "Before you expose it" in README.md.
|
||||||
|
|
||||||
# Rolling-deploy safety (Swarm / `docker stack deploy`): update one task at a
|
# Rolling-deploy safety (Swarm / `docker stack deploy`): update one task at a
|
||||||
# time, START the new task before stopping the old (zero-downtime via the ingress
|
# time, START the new task before stopping the old (zero-downtime via the ingress
|
||||||
@@ -74,7 +83,25 @@ services:
|
|||||||
retries: 5
|
retries: 5
|
||||||
|
|
||||||
web:
|
web:
|
||||||
image: git.fabledsword.com/bvandeusen/fabledcurator:dev
|
# :latest, NOT :dev — this file IS the install path.
|
||||||
|
#
|
||||||
|
# `docker compose up -d` merges docker-compose.override.yml, which sets
|
||||||
|
# build: for all five app services, and a build: wins over image:. So a
|
||||||
|
# contributor never pulls this tag and is unaffected by what it says.
|
||||||
|
#
|
||||||
|
# The tag is consulted only on `docker compose -f docker-compose.yml up -d`
|
||||||
|
# — the documented production path, which skips the override. That is a
|
||||||
|
# stranger installing the product, and they must land on the stable channel.
|
||||||
|
#
|
||||||
|
# :latest is main, which IS production (rule 147). :dev is the rolling
|
||||||
|
# bleeding-edge channel we work out of, republished several times a day with
|
||||||
|
# no stability promise. This file pinned :dev on all five services until
|
||||||
|
# 2026-08-31 (#3270), so the documented install shipped development builds.
|
||||||
|
# It went unnoticed because nobody who works on the project takes this path:
|
||||||
|
# the operator deploys from a swarm stack file, contributors get the
|
||||||
|
# override. Do not "fix" this back to :dev while debugging — use the
|
||||||
|
# override, or -f with an explicit tag on the command line.
|
||||||
|
image: git.fabledsword.com/bvandeusen/fabledcurator:latest
|
||||||
command: ["web"]
|
command: ["web"]
|
||||||
# Graceful shutdown: give the container time to drain in-flight work on a
|
# Graceful shutdown: give the container time to drain in-flight work on a
|
||||||
# deploy (docker SIGTERMs, then SIGKILLs after this window — default is only
|
# deploy (docker SIGTERMs, then SIGKILLs after this window — default is only
|
||||||
@@ -105,24 +132,46 @@ services:
|
|||||||
CELERY_BROKER_URL: redis://redis:6379/0
|
CELERY_BROKER_URL: redis://redis:6379/0
|
||||||
CELERY_RESULT_BACKEND: redis://redis:6379/0
|
CELERY_RESULT_BACKEND: redis://redis:6379/0
|
||||||
SECRET_KEY: ${SECRET_KEY:-dev_secret_key_not_for_production_change_me}
|
SECRET_KEY: ${SECRET_KEY:-dev_secret_key_not_for_production_change_me}
|
||||||
EXTENSION_API_KEY: ${EXTENSION_API_KEY:-}
|
|
||||||
LOG_LEVEL: ${LOG_LEVEL:-INFO}
|
LOG_LEVEL: ${LOG_LEVEL:-INFO}
|
||||||
|
# First boot only. FabledCurator refuses to start until the credential
|
||||||
|
# encryption key at /images/secrets/credential_key.b64 exists, and
|
||||||
|
# refuses to create one unless told to — auto-creating is
|
||||||
|
# indistinguishable from a restore that lost ./images/secrets, where it
|
||||||
|
# would mint a key that decrypts nothing and leave an instance that looks
|
||||||
|
# healthy while every paywalled download fails.
|
||||||
|
#
|
||||||
|
# Passed through EXPLICITLY because a variable in `.env` is only used for
|
||||||
|
# ${...} interpolation; it does not reach the container unless it is
|
||||||
|
# named here. Defaulted to empty so the refusal stands for everyone who
|
||||||
|
# has not opted in — the app tests for exactly "1".
|
||||||
|
#
|
||||||
|
# Set it in .env for one `up`, then remove it. See .env.example.
|
||||||
|
CURATOR_BOOTSTRAP_NEW_KEY: ${CURATOR_BOOTSTRAP_NEW_KEY:-}
|
||||||
volumes:
|
volumes:
|
||||||
- ./images:/images
|
- ./images:/images
|
||||||
- ./import:/import
|
- ./import:/import
|
||||||
# FC-5 legacy migration: bind-mount the host's ImageRepo images dir
|
# /import is a staging area for scripting a one-off ingest of a library
|
||||||
# under /import (FC's existing filesystem scan picks them up). Read-only
|
# you already have on disk. Drop files in ./import, or bind-mount an
|
||||||
# is sufficient — FC copies into /images during the scan. The worker +
|
# existing directory under it as below, then trigger the scan:
|
||||||
# scheduler services see the same /import via their own mounts below
|
#
|
||||||
# because of /import volume reuse. Edit the host path to match your
|
# curl -X POST http://localhost:8080/api/import/trigger
|
||||||
# install before running Settings → Maintenance → Legacy migration.
|
#
|
||||||
# - /var/lib/imagerepo/images:/import/imagerepo:ro
|
# Read-only is sufficient — FC copies into /images during the scan. The
|
||||||
|
# worker + scheduler services mount the same /import so the scan can run
|
||||||
|
# on whichever lane picks it up.
|
||||||
|
#
|
||||||
|
# Deliberately has no UI. The manual-scan surface was retired 2026-07-02
|
||||||
|
# once imports arrived via subscriptions + the extension, and the call
|
||||||
|
# not to restore it stands (operator, 2026-09-02): folder ingestion
|
||||||
|
# brings complexity the product does not need. The endpoint stays as an
|
||||||
|
# unsupported escape hatch; the supported way in is Subscriptions.
|
||||||
|
# - /srv/media/my-library:/import/my-library:ro
|
||||||
depends_on:
|
depends_on:
|
||||||
postgres: { condition: service_healthy }
|
postgres: { condition: service_healthy }
|
||||||
redis: { condition: service_healthy }
|
redis: { condition: service_healthy }
|
||||||
|
|
||||||
worker:
|
worker:
|
||||||
image: git.fabledsword.com/bvandeusen/fabledcurator:dev
|
image: git.fabledsword.com/bvandeusen/fabledcurator:latest
|
||||||
command: ["worker"]
|
command: ["worker"]
|
||||||
# Drain in-flight import/thumbnail/download tasks before SIGKILL on deploy.
|
# Drain in-flight import/thumbnail/download tasks before SIGKILL on deploy.
|
||||||
stop_grace_period: 90s
|
stop_grace_period: 90s
|
||||||
@@ -142,7 +191,7 @@ services:
|
|||||||
redis: { condition: service_healthy }
|
redis: { condition: service_healthy }
|
||||||
|
|
||||||
scheduler:
|
scheduler:
|
||||||
image: git.fabledsword.com/bvandeusen/fabledcurator:dev
|
image: git.fabledsword.com/bvandeusen/fabledcurator:latest
|
||||||
command: ["scheduler"]
|
command: ["scheduler"]
|
||||||
# Quick maintenance/scan lane + beat — short tasks, modest drain window.
|
# Quick maintenance/scan lane + beat — short tasks, modest drain window.
|
||||||
stop_grace_period: 60s
|
stop_grace_period: 60s
|
||||||
@@ -163,7 +212,7 @@ services:
|
|||||||
# 30-min backup or a multi-chunk audit can never starve the 5-min recovery
|
# 30-min backup or a multi-chunk audit can never starve the 5-min recovery
|
||||||
# sweeps / vacuum (operator-flagged 2026-06-07). One slot — these are heavy.
|
# sweeps / vacuum (operator-flagged 2026-06-07). One slot — these are heavy.
|
||||||
maintenance-long:
|
maintenance-long:
|
||||||
image: git.fabledsword.com/bvandeusen/fabledcurator:dev
|
image: git.fabledsword.com/bvandeusen/fabledcurator:latest
|
||||||
command: ["worker"]
|
command: ["worker"]
|
||||||
# Longest lane (DB backups, library audits, translation backfill) — give it
|
# Longest lane (DB backups, library audits, translation backfill) — give it
|
||||||
# the most room to finish a chunk gracefully. Chunked + idempotent, so a job
|
# the most room to finish a chunk gracefully. Chunked + idempotent, so a job
|
||||||
@@ -184,7 +233,7 @@ services:
|
|||||||
redis: { condition: service_healthy }
|
redis: { condition: service_healthy }
|
||||||
|
|
||||||
ml-worker:
|
ml-worker:
|
||||||
image: git.fabledsword.com/bvandeusen/fabledcurator-ml:dev
|
image: git.fabledsword.com/bvandeusen/fabledcurator-ml:latest
|
||||||
command: ["ml-worker"]
|
command: ["ml-worker"]
|
||||||
# A single GPU inference pass can run tens of seconds — let it finish.
|
# A single GPU inference pass can run tens of seconds — let it finish.
|
||||||
stop_grace_period: 120s
|
stop_grace_period: 120s
|
||||||
|
|||||||
@@ -5,9 +5,13 @@
|
|||||||
<img src="/favicon.svg" alt="" class="fc-brand__glyph" width="22" height="22" />
|
<img src="/favicon.svg" alt="" class="fc-brand__glyph" width="22" height="22" />
|
||||||
<span class="fc-brand__text">FabledCurator</span>
|
<span class="fc-brand__text">FabledCurator</span>
|
||||||
</RouterLink>
|
</RouterLink>
|
||||||
<span class="fc-health" :title="health.label">
|
<RouterLink
|
||||||
|
:to="{ name: 'settings', query: { tab: 'system' } }"
|
||||||
|
class="fc-health" :title="health.label"
|
||||||
|
:aria-label="`System health: ${health.label}`"
|
||||||
|
>
|
||||||
<v-icon size="x-small" :color="health.color">{{ health.icon }}</v-icon>
|
<v-icon size="x-small" :color="health.color">{{ health.icon }}</v-icon>
|
||||||
</span>
|
</RouterLink>
|
||||||
<PipelineStatusChip />
|
<PipelineStatusChip />
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
@@ -64,13 +68,15 @@
|
|||||||
</template>
|
</template>
|
||||||
|
|
||||||
<script setup>
|
<script setup>
|
||||||
import { computed, onBeforeUnmount, onMounted, ref } from 'vue'
|
import { computed, onBeforeUnmount, onMounted, onUnmounted, ref } from 'vue'
|
||||||
import { useRoute } from 'vue-router'
|
import { useRoute } from 'vue-router'
|
||||||
import router, { FRONT_DOOR } from '../router.js'
|
import router, { FRONT_DOOR } from '../router.js'
|
||||||
import { useSystemStore } from '../stores/system.js'
|
import { useSystemStore } from '../stores/system.js'
|
||||||
|
import { useSystemHealthStore } from '../stores/systemHealth.js'
|
||||||
import PipelineStatusChip from './PipelineStatusChip.vue'
|
import PipelineStatusChip from './PipelineStatusChip.vue'
|
||||||
|
|
||||||
const system = useSystemStore()
|
const system = useSystemStore()
|
||||||
|
const healthStore = useSystemHealthStore()
|
||||||
|
|
||||||
// Publish the nav's REAL height as --fc-nav-h so full-height workspaces
|
// Publish the nav's REAL height as --fc-nav-h so full-height workspaces
|
||||||
// (Explore/Subscriptions) and sticky sub-headers pin to it exactly instead of a
|
// (Explore/Subscriptions) and sticky sub-headers pin to it exactly instead of a
|
||||||
@@ -116,15 +122,55 @@ const settingsRoute = computed(() =>
|
|||||||
navRoutes.value.find(r => r.name === 'settings') || null
|
navRoutes.value.find(r => r.name === 'settings') || null
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// The dot beside the brand, and the only ambient signal that something in the
|
||||||
|
// stack has stopped (milestone 365).
|
||||||
|
//
|
||||||
|
// It used to read /api/health — a no-DB liveness check that proves the WEB
|
||||||
|
// container is serving and nothing else. Green there while the worker was dead
|
||||||
|
// is exactly what it looked like, and a green dot next to the product name is
|
||||||
|
// read as "everything is fine". It now reflects the whole-stack verdict.
|
||||||
|
//
|
||||||
|
// Deliberately re-using this element rather than adding a second indicator:
|
||||||
|
// there were already three partial surfaces (this, the pipeline chip, the
|
||||||
|
// Settings Activity tab) and a fourth would have made the question harder to
|
||||||
|
// answer, not easier. This is the one that already occupied the slot.
|
||||||
const health = computed(() => {
|
const health = computed(() => {
|
||||||
if (system.healthy === null) {
|
const overall = healthStore.overall
|
||||||
|
if (overall === null) {
|
||||||
return { icon: 'mdi-circle-outline', color: 'on-surface', label: 'checking…' }
|
return { icon: 'mdi-circle-outline', color: 'on-surface', label: 'checking…' }
|
||||||
}
|
}
|
||||||
if (system.healthy === true) {
|
if (overall === 'ok') {
|
||||||
return { icon: 'mdi-circle', color: 'success', label: 'healthy' }
|
return { icon: 'mdi-circle', color: 'success', label: 'All parts running' }
|
||||||
}
|
}
|
||||||
return { icon: 'mdi-alert-circle', color: 'error', label: 'unreachable' }
|
// Name what is wrong in the tooltip. "Something is unhealthy" sends someone
|
||||||
|
// hunting; "Scheduler has not checked in for 6 min" does not.
|
||||||
|
const worst = healthStore.problems[0]
|
||||||
|
const others = healthStore.problems.length - 1
|
||||||
|
const suffix = others > 0 ? ` (+${others} more)` : ''
|
||||||
|
if (overall === 'down') {
|
||||||
|
return {
|
||||||
|
icon: 'mdi-alert-circle', color: 'error',
|
||||||
|
label: (worst?.detail || 'A part has stopped') + suffix,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (overall === 'stale') {
|
||||||
|
return {
|
||||||
|
icon: 'mdi-alert', color: 'warning',
|
||||||
|
label: (worst?.detail || 'A part is quiet') + suffix,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return { icon: 'mdi-help-circle-outline', color: 'on-surface', label: 'Health unknown' }
|
||||||
})
|
})
|
||||||
|
|
||||||
|
const HEALTH_POLL_MS = 15_000
|
||||||
|
let healthTimer = null
|
||||||
|
onMounted(() => {
|
||||||
|
healthStore.refresh()
|
||||||
|
healthTimer = setInterval(() => {
|
||||||
|
if (!document.hidden) healthStore.refresh()
|
||||||
|
}, HEALTH_POLL_MS)
|
||||||
|
})
|
||||||
|
onUnmounted(() => { if (healthTimer) clearInterval(healthTimer) })
|
||||||
</script>
|
</script>
|
||||||
|
|
||||||
<style scoped>
|
<style scoped>
|
||||||
@@ -237,7 +283,14 @@ const health = computed(() => {
|
|||||||
display: flex;
|
display: flex;
|
||||||
align-items: center;
|
align-items: center;
|
||||||
flex-shrink: 0;
|
flex-shrink: 0;
|
||||||
|
/* A RouterLink since milestone 365 — it is the path to the Settings System
|
||||||
|
tab, not just an indicator. Reset the anchor so turning a span into a
|
||||||
|
link changed nothing about how the nav reads. */
|
||||||
|
text-decoration: none;
|
||||||
|
color: inherit;
|
||||||
|
border-radius: 50%;
|
||||||
}
|
}
|
||||||
|
.fc-health:hover { background: rgb(var(--v-theme-on-surface) / 0.12); }
|
||||||
.fc-nav-right {
|
.fc-nav-right {
|
||||||
flex: 1 1 0;
|
flex: 1 1 0;
|
||||||
min-width: 0;
|
min-width: 0;
|
||||||
|
|||||||
@@ -0,0 +1,128 @@
|
|||||||
|
<template>
|
||||||
|
<!-- A Settings tab, not a page of its own (operator 2026-09-02): the first
|
||||||
|
cut hung this off the health dot alone, which is a target you have to
|
||||||
|
already suspect something to look for. Settings is where someone goes
|
||||||
|
to ask the instance about itself, so it lives beside Activity. -->
|
||||||
|
<div>
|
||||||
|
<div class="d-flex align-center mb-1">
|
||||||
|
<v-spacer />
|
||||||
|
<span class="fc-sys__checked">
|
||||||
|
{{ store.checkedAt ? `checked ${formatRelative(store.checkedAt)}` : 'checking…' }}
|
||||||
|
</span>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<p class="fc-sys__lede text-body-2 mb-5">
|
||||||
|
Every moving part of FabledCurator and whether it is still checking in.
|
||||||
|
Parts are learned as they appear, so anything that has run at least once
|
||||||
|
stays listed — that is what lets a stopped one be noticed rather than
|
||||||
|
simply vanishing.
|
||||||
|
</p>
|
||||||
|
|
||||||
|
<v-alert
|
||||||
|
v-if="store.lastError" type="error" variant="tonal" density="compact" class="mb-4"
|
||||||
|
>
|
||||||
|
Could not reach FabledCurator: {{ store.lastError }}
|
||||||
|
</v-alert>
|
||||||
|
|
||||||
|
<v-card v-else variant="flat" class="fc-sys__card">
|
||||||
|
<div v-if="!store.parts.length" class="pa-6 text-center fc-sys__muted">
|
||||||
|
Still gathering — this fills in on the first check.
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<div
|
||||||
|
v-for="part in store.parts" :key="part.key"
|
||||||
|
class="fc-sys__row" :class="`fc-sys__row--${part.state}`"
|
||||||
|
>
|
||||||
|
<span class="fc-sys__dot" :class="`fc-sys__dot--${part.state}`" />
|
||||||
|
|
||||||
|
<div class="fc-sys__body">
|
||||||
|
<div class="fc-sys__name">
|
||||||
|
{{ part.name }}
|
||||||
|
<span class="fc-sys__kind">{{ kindLabel(part.kind) }}</span>
|
||||||
|
</div>
|
||||||
|
<!-- The sentence, not just a chip. At the moment someone is deciding
|
||||||
|
whether to go and open Portainer, "has not checked in for 6 min"
|
||||||
|
is the thing that answers them. -->
|
||||||
|
<div class="fc-sys__detail">{{ part.detail }}</div>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<div class="fc-sys__meta">
|
||||||
|
<div v-if="part.last_seen_at" :title="part.last_seen_at">
|
||||||
|
seen {{ formatRelative(part.last_seen_at) }}
|
||||||
|
</div>
|
||||||
|
<div v-if="part.latency_ms != null">{{ part.latency_ms }} ms</div>
|
||||||
|
<div v-if="part.queues?.length" class="fc-sys__queues">{{ part.queues.join(', ') }}</div>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
</v-card>
|
||||||
|
|
||||||
|
<p v-if="store.thresholds" class="fc-sys__foot text-caption mt-4">
|
||||||
|
A part is called stale after
|
||||||
|
{{ Math.round(store.thresholds.stale_after_seconds / 60) }} min without a
|
||||||
|
check-in and treated as stopped after
|
||||||
|
{{ Math.round(store.thresholds.down_after_seconds / 60) }} min. The window
|
||||||
|
is deliberately wide: a rolling deploy briefly runs two of a service and
|
||||||
|
then neither, and an indicator that reddened on every update would stop
|
||||||
|
being read.
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
</template>
|
||||||
|
|
||||||
|
<script setup>
|
||||||
|
import { onMounted } from 'vue'
|
||||||
|
|
||||||
|
import { useSystemHealthStore } from '../../stores/systemHealth.js'
|
||||||
|
import { formatRelative } from '../../utils/date.js'
|
||||||
|
|
||||||
|
const store = useSystemHealthStore()
|
||||||
|
|
||||||
|
function kindLabel(kind) {
|
||||||
|
if (kind === 'celery') return 'background worker'
|
||||||
|
if (kind === 'agent') return 'GPU agent'
|
||||||
|
if (kind === 'datastore') return 'datastore'
|
||||||
|
return kind
|
||||||
|
}
|
||||||
|
|
||||||
|
// No timer of its own. TopNav already polls this same pinia store every 15s
|
||||||
|
// for the health dot, and it is mounted on every route this tab is reachable
|
||||||
|
// from — a second interval here would just double the request rate for a 5s
|
||||||
|
// freshness gain. v-window keeps a visited item MOUNTED (hidden, not
|
||||||
|
// destroyed), so a local timer would also have kept firing behind Maintenance.
|
||||||
|
// One refresh on open, so arriving at the tab doesn't wait out the nav's tick.
|
||||||
|
onMounted(() => { store.refresh() })
|
||||||
|
</script>
|
||||||
|
|
||||||
|
<style scoped>
|
||||||
|
.fc-sys__lede, .fc-sys__muted, .fc-sys__checked, .fc-sys__foot {
|
||||||
|
color: rgb(var(--v-theme-on-surface) / 0.66);
|
||||||
|
}
|
||||||
|
.fc-sys__checked { font-size: 0.78rem; }
|
||||||
|
.fc-sys__card { background: rgb(var(--v-theme-on-surface) / 0.04); }
|
||||||
|
|
||||||
|
.fc-sys__row {
|
||||||
|
display: flex; align-items: center; gap: 12px;
|
||||||
|
padding: 12px 16px;
|
||||||
|
border-bottom: 1px solid rgb(var(--v-theme-on-surface) / 0.08);
|
||||||
|
}
|
||||||
|
.fc-sys__row:last-child { border-bottom: 0; }
|
||||||
|
|
||||||
|
.fc-sys__dot { width: 9px; height: 9px; border-radius: 50%; flex: 0 0 auto; }
|
||||||
|
.fc-sys__dot--ok { background: rgb(var(--v-theme-success)); }
|
||||||
|
.fc-sys__dot--stale { background: rgb(var(--v-theme-warning)); }
|
||||||
|
.fc-sys__dot--down { background: rgb(var(--v-theme-error)); }
|
||||||
|
.fc-sys__dot--unknown { background: rgb(var(--v-theme-on-surface) / 0.35); }
|
||||||
|
|
||||||
|
.fc-sys__body { min-width: 0; flex: 1 1 auto; }
|
||||||
|
.fc-sys__name { font-weight: 600; }
|
||||||
|
.fc-sys__kind {
|
||||||
|
margin-left: 8px; font-weight: 400; font-size: 0.72rem; text-transform: uppercase;
|
||||||
|
letter-spacing: 0.04em; color: rgb(var(--v-theme-on-surface) / 0.5);
|
||||||
|
}
|
||||||
|
.fc-sys__detail { font-size: 0.82rem; color: rgb(var(--v-theme-on-surface) / 0.72); }
|
||||||
|
|
||||||
|
.fc-sys__meta {
|
||||||
|
text-align: right; font-size: 0.75rem; flex: 0 0 auto;
|
||||||
|
font-variant-numeric: tabular-nums; color: rgb(var(--v-theme-on-surface) / 0.6);
|
||||||
|
}
|
||||||
|
.fc-sys__queues { opacity: 0.75; }
|
||||||
|
</style>
|
||||||
@@ -45,6 +45,13 @@ const routes = [
|
|||||||
|
|
||||||
// Settings — config, pinned to the right of the nav (TopNav special-cases it).
|
// Settings — config, pinned to the right of the nav (TopNav special-cases it).
|
||||||
{ path: '/settings', name: 'settings', component: SettingsView, meta: { title: 'Settings', stickyChrome: true } },
|
{ path: '/settings', name: 'settings', component: SettingsView, meta: { title: 'Settings', stickyChrome: true } },
|
||||||
|
// System health is a Settings TAB, not a route of its own (operator
|
||||||
|
// 2026-09-02). It first shipped as /system reachable only from the health
|
||||||
|
// dot, which is a target you have to already suspect something to go
|
||||||
|
// looking for. Settings is where someone goes to ask the instance about
|
||||||
|
// itself. The path stays as a redirect so the dot's old link, and any
|
||||||
|
// bookmark from that build, still land somewhere real.
|
||||||
|
{ path: '/system', name: 'system', redirect: () => ({ name: 'settings', query: { tab: 'system' } }) },
|
||||||
|
|
||||||
// The old standalone paths now redirect into the Browse hub, preserving any
|
// The old standalone paths now redirect into the Browse hub, preserving any
|
||||||
// deep-link query (e.g. /posts?post_id=N → /browse?tab=posts&post_id=N). The
|
// deep-link query (e.g. /posts?post_id=N → /browse?tab=posts&post_id=N). The
|
||||||
|
|||||||
@@ -4,7 +4,12 @@ import { useApi } from '../composables/useApi.js'
|
|||||||
|
|
||||||
export const useSystemStore = defineStore('system', () => {
|
export const useSystemStore = defineStore('system', () => {
|
||||||
const api = useApi()
|
const api = useApi()
|
||||||
const healthy = ref(null) // null=unknown, true=ok, false=down
|
// NOT what the nav dot reads any more (milestone 365): that is the
|
||||||
|
// whole-stack verdict in systemHealth.js. /api/health only proves the web
|
||||||
|
// container is serving, which is why a green dot here sat happily beside a
|
||||||
|
// dead worker. refreshHealth() is still called — it is also how build/version
|
||||||
|
// info arrives — so this stays as its by-product rather than its purpose.
|
||||||
|
const healthy = ref(null)
|
||||||
// What the instance says it is. Since milestone 318 stopped publishing
|
// What the instance says it is. Since milestone 318 stopped publishing
|
||||||
// version image tags, this is the only answer to "which build is this?" —
|
// version image tags, this is the only answer to "which build is this?" —
|
||||||
// there is no registry name left to check it against.
|
// there is no registry name left to check it against.
|
||||||
|
|||||||
@@ -0,0 +1,49 @@
|
|||||||
|
import { defineStore } from 'pinia'
|
||||||
|
import { computed, ref } from 'vue'
|
||||||
|
|
||||||
|
import { useApi } from '../composables/useApi.js'
|
||||||
|
|
||||||
|
// Whole-stack health: is every part of FabledCurator running (milestone 365)?
|
||||||
|
//
|
||||||
|
// Distinct from `system.js`, which polls /api/health — a no-DB liveness check
|
||||||
|
// that only proves the web container is serving. That endpoint answers "can I
|
||||||
|
// reach the API"; this one answers "is anything broken", which is the question
|
||||||
|
// a green dot beside the brand was already being read as answering.
|
||||||
|
//
|
||||||
|
// Also distinct from `systemActivity.js`, which is about what the pipeline is
|
||||||
|
// DOING — queue depths, running tasks, failures. Running and alive are
|
||||||
|
// different questions and they fail independently: a perfectly idle stack with
|
||||||
|
// a dead worker looks identical to a healthy one on the activity surfaces.
|
||||||
|
export const useSystemHealthStore = defineStore('systemHealth', () => {
|
||||||
|
const api = useApi()
|
||||||
|
|
||||||
|
const overall = ref(null) // null until the first answer: unknown ≠ ok
|
||||||
|
const parts = ref([])
|
||||||
|
const checkedAt = ref(null)
|
||||||
|
const thresholds = ref(null) // server-owned, so the UI keeps no second copy
|
||||||
|
const lastError = ref(null)
|
||||||
|
|
||||||
|
async function refresh() {
|
||||||
|
try {
|
||||||
|
const body = await api.get('/api/system/health')
|
||||||
|
overall.value = body.overall
|
||||||
|
parts.value = body.parts || []
|
||||||
|
checkedAt.value = body.checked_at
|
||||||
|
thresholds.value = body.thresholds || null
|
||||||
|
lastError.value = null
|
||||||
|
} catch (e) {
|
||||||
|
// The endpoint is built never to fail because a dependency failed, so a
|
||||||
|
// throw here means the API itself is unreachable — which is its own kind
|
||||||
|
// of unhealthy and must not be shown as "ok".
|
||||||
|
lastError.value = e.message
|
||||||
|
overall.value = 'unknown'
|
||||||
|
}
|
||||||
|
return overall.value
|
||||||
|
}
|
||||||
|
|
||||||
|
// The parts worth naming in a tooltip — everything that is not ok, worst
|
||||||
|
// first. The endpoint already sorts that way.
|
||||||
|
const problems = computed(() => parts.value.filter(p => p.state !== 'ok'))
|
||||||
|
|
||||||
|
return { overall, parts, checkedAt, thresholds, lastError, problems, refresh }
|
||||||
|
})
|
||||||
@@ -14,6 +14,7 @@
|
|||||||
style="position: sticky; top: var(--fc-nav-h, 64px); z-index: 4;"
|
style="position: sticky; top: var(--fc-nav-h, 64px); z-index: 4;"
|
||||||
>
|
>
|
||||||
<v-tab value="overview">Overview</v-tab>
|
<v-tab value="overview">Overview</v-tab>
|
||||||
|
<v-tab value="system">System</v-tab>
|
||||||
<v-tab value="activity">Activity</v-tab>
|
<v-tab value="activity">Activity</v-tab>
|
||||||
<v-tab value="cleanup">Cleanup</v-tab>
|
<v-tab value="cleanup">Cleanup</v-tab>
|
||||||
<v-tab value="maintenance">Maintenance</v-tab>
|
<v-tab value="maintenance">Maintenance</v-tab>
|
||||||
@@ -42,6 +43,13 @@
|
|||||||
</v-alert>
|
</v-alert>
|
||||||
</v-window-item>
|
</v-window-item>
|
||||||
|
|
||||||
|
<!-- Is every part of the stack still running (milestone 365). Sits
|
||||||
|
beside Activity deliberately: Activity answers "what is the queue
|
||||||
|
doing", this answers "is anything left to do it". -->
|
||||||
|
<v-window-item value="system">
|
||||||
|
<SystemHealthTab />
|
||||||
|
</v-window-item>
|
||||||
|
|
||||||
<v-window-item value="activity">
|
<v-window-item value="activity">
|
||||||
<SystemActivityTab @open-maintenance="tab = 'maintenance'" />
|
<SystemActivityTab @open-maintenance="tab = 'maintenance'" />
|
||||||
</v-window-item>
|
</v-window-item>
|
||||||
@@ -73,18 +81,24 @@
|
|||||||
</template>
|
</template>
|
||||||
|
|
||||||
<script setup>
|
<script setup>
|
||||||
import { onMounted, onUnmounted, ref, watch } from 'vue'
|
import { onMounted, onUnmounted, watch } from 'vue'
|
||||||
import { useSystemStore } from '../stores/system.js'
|
import { useSystemStore } from '../stores/system.js'
|
||||||
import SystemStatsCards from '../components/settings/SystemStatsCards.vue'
|
import SystemStatsCards from '../components/settings/SystemStatsCards.vue'
|
||||||
import SystemActivitySummary from '../components/settings/SystemActivitySummary.vue'
|
import SystemActivitySummary from '../components/settings/SystemActivitySummary.vue'
|
||||||
import SystemActivityTab from '../components/settings/SystemActivityTab.vue'
|
import SystemActivityTab from '../components/settings/SystemActivityTab.vue'
|
||||||
|
import SystemHealthTab from '../components/settings/SystemHealthTab.vue'
|
||||||
import GpuActivityPanel from '../components/settings/GpuActivityPanel.vue'
|
import GpuActivityPanel from '../components/settings/GpuActivityPanel.vue'
|
||||||
import DownloadsActivityPanel from '../components/settings/DownloadsActivityPanel.vue'
|
import DownloadsActivityPanel from '../components/settings/DownloadsActivityPanel.vue'
|
||||||
import MaintenancePanel from '../components/settings/MaintenancePanel.vue'
|
import MaintenancePanel from '../components/settings/MaintenancePanel.vue'
|
||||||
import CleanupView from './CleanupView.vue'
|
import CleanupView from './CleanupView.vue'
|
||||||
|
import { useTabQuery } from '../composables/useTabQuery.js'
|
||||||
import { useMLStore } from '../stores/ml.js'
|
import { useMLStore } from '../stores/ml.js'
|
||||||
|
|
||||||
const tab = ref('overview')
|
// ?tab= sync (the same composable Browse/Subscriptions use) so a tab can be
|
||||||
|
// linked TO — the health dot beside the brand points at ?tab=system, and the
|
||||||
|
// old /system path redirects there.
|
||||||
|
const VALID_TABS = ['overview', 'system', 'activity', 'cleanup', 'maintenance']
|
||||||
|
const { tab } = useTabQuery(VALID_TABS, 'overview')
|
||||||
const system = useSystemStore()
|
const system = useSystemStore()
|
||||||
const mlStore = useMLStore()
|
const mlStore = useMLStore()
|
||||||
|
|
||||||
|
|||||||
@@ -40,6 +40,14 @@ describe('router', () => {
|
|||||||
expect(router.currentRoute.value.query.post_id).toBe('7')
|
expect(router.currentRoute.value.query.post_id).toBe('7')
|
||||||
})
|
})
|
||||||
|
|
||||||
|
it('/system redirects into the Settings System tab', async () => {
|
||||||
|
// It shipped as a standalone page for one build; the health dot and any
|
||||||
|
// bookmark from it must still land on the surface, which is now a tab.
|
||||||
|
await router.push('/system')
|
||||||
|
expect(router.currentRoute.value.name).toBe('settings')
|
||||||
|
expect(router.currentRoute.value.query.tab).toBe('system')
|
||||||
|
})
|
||||||
|
|
||||||
it('series-read is an immersive route', () => {
|
it('series-read is an immersive route', () => {
|
||||||
const r = router.resolve('/series/5/read')
|
const r = router.resolve('/series/5/read')
|
||||||
expect(r.name).toBe('series-read')
|
expect(r.name).toBe('series-read')
|
||||||
|
|||||||
+22
-1
@@ -90,7 +90,7 @@ DERIVER='scripts/artifacts.sh'
|
|||||||
|
|
||||||
|
|
||||||
usage() {
|
usage() {
|
||||||
echo "usage: artifacts.sh {paths|revision|version} {web|ml|agent|extension}" >&2
|
echo "usage: artifacts.sh {paths|revision|version|epoch} {web|ml|agent|extension}" >&2
|
||||||
exit 2
|
exit 2
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -149,6 +149,26 @@ cmd_revision() {
|
|||||||
echo "$(newest "$1")" | cut -d' ' -f2 | cut -c1-12
|
echo "$(newest "$1")" | cut -d' ' -f2 | cut -c1-12
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# The BUILD CLOCK: the same commit's unix timestamp, for SOURCE_DATE_EPOCH.
|
||||||
|
#
|
||||||
|
# buildkit stamps the image config's `created` field and every history entry
|
||||||
|
# with the wall clock of the build unless this is set, so two builds of
|
||||||
|
# identical source produce different config blobs and therefore different
|
||||||
|
# manifest digests. That is #3265: the weekly refresh republished all three
|
||||||
|
# `:latest` tags on 2026-08-30 with every content step CACHED and the bases
|
||||||
|
# resolved to unchanged digests — nothing was different, and the digest moved
|
||||||
|
# anyway. A digest that changes on a calendar cannot also mean "the content
|
||||||
|
# changed", which is the only thing anyone wants it for.
|
||||||
|
#
|
||||||
|
# It is the same commit `revision` and `version` name — deliberately, and this
|
||||||
|
# is the point of routing it through `newest()` rather than taking git's word
|
||||||
|
# separately. Three values derived from three lookups can disagree; three
|
||||||
|
# views of one lookup cannot. Note #3127 §2 is the record of what a second
|
||||||
|
# clock costs.
|
||||||
|
cmd_epoch() {
|
||||||
|
echo "$(newest "$1")" | cut -d' ' -f1
|
||||||
|
}
|
||||||
|
|
||||||
# The VERSION: `YYYY.MM.DD.HHMM`, zero-padded, UTC. One shape across the whole
|
# The VERSION: `YYYY.MM.DD.HHMM`, zero-padded, UTC. One shape across the whole
|
||||||
# family (note #3127 §1, rule 148) — the number an instance reports about
|
# family (note #3127 §1, rule 148) — the number an instance reports about
|
||||||
# itself, and, with a `v` in front, the release tag naming the same build.
|
# itself, and, with a `v` in front, the release tag naming the same build.
|
||||||
@@ -197,5 +217,6 @@ case "$1" in
|
|||||||
paths) cmd_paths "$2" ;;
|
paths) cmd_paths "$2" ;;
|
||||||
revision) cmd_revision "$2" ;;
|
revision) cmd_revision "$2" ;;
|
||||||
version) cmd_version "$2" ;;
|
version) cmd_version "$2" ;;
|
||||||
|
epoch) cmd_epoch "$2" ;;
|
||||||
*) usage ;;
|
*) usage ;;
|
||||||
esac
|
esac
|
||||||
|
|||||||
+118
-14
@@ -33,6 +33,32 @@ history. Ancestry is immune to the shape change, and it is also the more honest
|
|||||||
question: "what is in this that was not in the last one" IS a reachability
|
question: "what is in this that was not in the last one" IS a reachability
|
||||||
question.
|
question.
|
||||||
|
|
||||||
|
Ancestry alone is not enough, though, and milestone 328 is where that showed.
|
||||||
|
The 28 `v26.*` tags are still in the repo — the operator kept them as history
|
||||||
|
when their releases were deleted — so `--match v*` walks straight back to
|
||||||
|
`v26.06.04.0` and reports 533 commits. That span is not a changelog: nobody has
|
||||||
|
run `v26.06.04.0`, its release page no longer exists to compare against, and
|
||||||
|
the 200 lines that survive truncation are precisely the internal build-out that
|
||||||
|
milestone 328 exists to stop shipping. So the match is `v[0-9][0-9][0-9][0-9].*`
|
||||||
|
— rule 148's four-digit-year shape — which is exactly the set of tags that name
|
||||||
|
a release a reader could have been running. A pre-convention tag is history,
|
||||||
|
not a predecessor.
|
||||||
|
|
||||||
|
## The first release has no changelog, and should not pretend to
|
||||||
|
|
||||||
|
Once the match is narrowed, the first rule-148 tag reaches no predecessor at
|
||||||
|
all, and the old fallback — diff against the whole history — is worse than the
|
||||||
|
problem it replaced. The honest content for a release nobody has a previous
|
||||||
|
version of is what the thing IS.
|
||||||
|
|
||||||
|
So a release with no reachable predecessor renders the product overview instead
|
||||||
|
of a commit list. It is read out of README.md between `<!-- overview:start -->`
|
||||||
|
and `<!-- overview:end -->` rather than written here, for the same reason the
|
||||||
|
changelog is derived: two hand-maintained descriptions of one product drift,
|
||||||
|
and nothing ever catches it. The release page and the repo front page are one
|
||||||
|
source. Every later release goes back to being a changelog, which is what §5 of
|
||||||
|
note #3127 says a release is for.
|
||||||
|
|
||||||
## Re-runs update, they do not fall through
|
## Re-runs update, they do not fall through
|
||||||
|
|
||||||
Note #3127 §6.7: a publisher that POSTs and recovers the id from a `409` never
|
Note #3127 §6.7: a publisher that POSTs and recovers the id from a `409` never
|
||||||
@@ -91,18 +117,45 @@ def git_ok(*args: str) -> str | None:
|
|||||||
|
|
||||||
|
|
||||||
def previous_tag(ref: str, tag: str | None) -> str | None:
|
def previous_tag(ref: str, tag: str | None) -> str | None:
|
||||||
"""The most recent `v*` tag reachable from `ref`, excluding `tag` itself.
|
"""The most recent rule-148 tag reachable from `ref`, excluding `tag` itself.
|
||||||
|
|
||||||
`--exclude` rather than `<ref>^` so this is the same call whether or not
|
`--exclude` rather than `<ref>^` so this is the same call whether or not
|
||||||
`ref` is the tag being released — and so it does not blow up on a root
|
`ref` is the tag being released — and so it does not blow up on a root
|
||||||
commit that has no parent to walk to.
|
commit that has no parent to walk to.
|
||||||
|
|
||||||
|
The glob deliberately does NOT match the old `v26.*` tags. They are kept as
|
||||||
|
history and their releases are gone, so naming one as the predecessor emits
|
||||||
|
a span nobody can look up. See the module docstring.
|
||||||
"""
|
"""
|
||||||
args = ["describe", "--tags", "--abbrev=0", "--match", "v*"]
|
args = ["describe", "--tags", "--abbrev=0", "--match", "v[0-9][0-9][0-9][0-9].*"]
|
||||||
if tag:
|
if tag:
|
||||||
args += ["--exclude", tag]
|
args += ["--exclude", tag]
|
||||||
return git_ok(*args, ref)
|
return git_ok(*args, ref)
|
||||||
|
|
||||||
|
|
||||||
|
def product_overview() -> str | None:
|
||||||
|
"""The product description, lifted verbatim from README.md.
|
||||||
|
|
||||||
|
Returns None if the markers are absent or empty — a missing overview is
|
||||||
|
reported as a note and the release still publishes, on the same reasoning
|
||||||
|
as cross_checks(): the release is the useful object even when one part of
|
||||||
|
the derivation could not run.
|
||||||
|
"""
|
||||||
|
root = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||||
|
try:
|
||||||
|
with open(os.path.join(root, "README.md"), encoding="utf-8") as fh:
|
||||||
|
readme = fh.read()
|
||||||
|
except OSError:
|
||||||
|
return None
|
||||||
|
match = re.search(
|
||||||
|
r"<!--\s*overview:start\s*-->(.*?)<!--\s*overview:end\s*-->",
|
||||||
|
readme, re.S,
|
||||||
|
)
|
||||||
|
if not match:
|
||||||
|
return None
|
||||||
|
return match.group(1).strip() or None
|
||||||
|
|
||||||
|
|
||||||
def commits(previous: str | None, ref: str) -> list[str]:
|
def commits(previous: str | None, ref: str) -> list[str]:
|
||||||
"""The subjects between the previous release and this one.
|
"""The subjects between the previous release and this one.
|
||||||
|
|
||||||
@@ -125,7 +178,10 @@ def truncate(log: list[str]) -> tuple[list[str], str | None]:
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def render(tag: str, sha: str, previous: str | None, log: list[str], notes: list[str]) -> str:
|
def render(
|
||||||
|
tag: str, sha: str, previous: str | None, log: list[str], notes: list[str],
|
||||||
|
overview: str | None,
|
||||||
|
) -> str:
|
||||||
short = sha[:7]
|
short = sha[:7]
|
||||||
parts = []
|
parts = []
|
||||||
|
|
||||||
@@ -135,6 +191,25 @@ def render(tag: str, sha: str, previous: str | None, log: list[str], notes: list
|
|||||||
# is the failure this whole milestone is about.
|
# is the failure this whole milestone is about.
|
||||||
parts.append("\n".join(f"> **Note:** {n}" for n in notes))
|
parts.append("\n".join(f"> **Note:** {n}" for n in notes))
|
||||||
|
|
||||||
|
# No predecessor means nobody reading this has run an earlier one, so the
|
||||||
|
# release describes the product rather than a diff. The overview is
|
||||||
|
# README.md's own words — see the module docstring on why it is not
|
||||||
|
# written here.
|
||||||
|
if previous is None and overview:
|
||||||
|
parts.append(overview)
|
||||||
|
parts.append(
|
||||||
|
"## Installing\n\n"
|
||||||
|
"```\ncurl -O https://git.fabledsword.com/bvandeusen/FabledCurator/raw/"
|
||||||
|
f"tag/{tag}/docker-compose.yml\ncurl -O https://git.fabledsword.com/"
|
||||||
|
f"bvandeusen/FabledCurator/raw/tag/{tag}/.env.example\n"
|
||||||
|
"mv .env.example .env # then set SECRET_KEY, DB_PASSWORD\n"
|
||||||
|
"docker compose -f docker-compose.yml up -d\n```\n\n"
|
||||||
|
"**Read \"Before you expose it\" in the README first.** FabledCurator "
|
||||||
|
"has no login, and it stores live platform session cookies for "
|
||||||
|
"accounts that usually have a payment method attached. Bind it to a "
|
||||||
|
"network you trust."
|
||||||
|
)
|
||||||
|
|
||||||
parts.append(
|
parts.append(
|
||||||
f"Built from `{short}`. The rollback unit is the immutable `:c-` tag "
|
f"Built from `{short}`. The rollback unit is the immutable `:c-` tag "
|
||||||
f"(rule 145) — these three move together:\n\n```\n"
|
f"(rule 145) — these three move together:\n\n```\n"
|
||||||
@@ -142,7 +217,19 @@ def render(tag: str, sha: str, previous: str | None, log: list[str], notes: list
|
|||||||
+ "\n```"
|
+ "\n```"
|
||||||
)
|
)
|
||||||
|
|
||||||
heading = f"## Changes since {previous}" if previous else "## Changes"
|
if previous is None:
|
||||||
|
# Deliberately NOT a commit list. The alternative is the whole history
|
||||||
|
# truncated to MAX_COMMITS, which is 200 lines of internal build-out
|
||||||
|
# presented to someone who has never seen this project.
|
||||||
|
parts.append(
|
||||||
|
"---\n\n_First release under rule 148's `vYYYY.MM.DD.HHMM` shape, so "
|
||||||
|
"there is no predecessor to diff against and no changelog to derive. "
|
||||||
|
"The description above is README.md's, quoted at publish time. Later "
|
||||||
|
"releases carry the commits since the previous one._"
|
||||||
|
)
|
||||||
|
return "\n\n".join(parts)
|
||||||
|
|
||||||
|
heading = f"## Changes since {previous}"
|
||||||
if log:
|
if log:
|
||||||
parts.append(heading + "\n\n" + "\n".join(f"- {line}" for line in log))
|
parts.append(heading + "\n\n" + "\n".join(f"- {line}" for line in log))
|
||||||
else:
|
else:
|
||||||
@@ -152,10 +239,9 @@ def render(tag: str, sha: str, previous: str | None, log: list[str], notes: list
|
|||||||
"names the same source under a new name._"
|
"names the same source under a new name._"
|
||||||
)
|
)
|
||||||
|
|
||||||
span = f"{previous}..{tag}" if previous else tag
|
|
||||||
parts.append(
|
parts.append(
|
||||||
f"---\n\n_Derived at publish time from `git log --no-merges {span}`. "
|
f"---\n\n_Derived at publish time from "
|
||||||
f"Nothing here is hand-maintained._"
|
f"`git log --no-merges {previous}..{tag}`. Nothing here is hand-maintained._"
|
||||||
)
|
)
|
||||||
return "\n\n".join(parts)
|
return "\n\n".join(parts)
|
||||||
|
|
||||||
@@ -289,13 +375,31 @@ def main() -> None:
|
|||||||
for note in notes:
|
for note in notes:
|
||||||
print(f"release: NOTE {note}")
|
print(f"release: NOTE {note}")
|
||||||
|
|
||||||
log = commits(previous, ref)
|
# A first release renders the overview instead of a changelog, so the
|
||||||
print(f"release: {len(log)} non-merge commits in the span")
|
# commit walk is skipped entirely rather than computed and discarded —
|
||||||
log, overflow = truncate(log)
|
# `commits(None, ref)` is the whole history and there is no reason to ask
|
||||||
if overflow:
|
# for it.
|
||||||
print(f"release: NOTE {overflow}")
|
overview = None
|
||||||
notes.append(overflow)
|
log: list[str] = []
|
||||||
body = render(tag or ref, sha, previous, log, notes)
|
if previous is None:
|
||||||
|
overview = product_overview()
|
||||||
|
if overview is None:
|
||||||
|
note = (
|
||||||
|
"No `<!-- overview:start -->` block found in README.md, so this "
|
||||||
|
"first release has no product description. Published anyway; add "
|
||||||
|
"the markers and re-run the workflow to fill it in."
|
||||||
|
)
|
||||||
|
print(f"release: NOTE {note}")
|
||||||
|
notes.append(note)
|
||||||
|
print("release: no rule-148 predecessor — rendering the product overview")
|
||||||
|
else:
|
||||||
|
log = commits(previous, ref)
|
||||||
|
print(f"release: {len(log)} non-merge commits in the span")
|
||||||
|
log, overflow = truncate(log)
|
||||||
|
if overflow:
|
||||||
|
print(f"release: NOTE {overflow}")
|
||||||
|
notes.append(overflow)
|
||||||
|
body = render(tag or ref, sha, previous, log, notes, overview)
|
||||||
|
|
||||||
if args.dry_run or not tag:
|
if args.dry_run or not tag:
|
||||||
print("--- body ---")
|
print("--- body ---")
|
||||||
|
|||||||
@@ -0,0 +1,155 @@
|
|||||||
|
"""Prove a freshly built image can still do the things its OS packages provide.
|
||||||
|
|
||||||
|
Run INSIDE the image, not against the source tree. That distinction is the
|
||||||
|
entire reason this file exists.
|
||||||
|
|
||||||
|
`ci.yml`'s lanes run on `ci-python:3.14` and install `requirements.txt`. A base
|
||||||
|
refresh changes neither, so all five lanes stay green through a base bump that
|
||||||
|
breaks the product. What a refresh actually re-resolves is this, from the
|
||||||
|
Dockerfile:
|
||||||
|
|
||||||
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
|
ffmpeg unar libpq5 postgresql-client zstd megatools \
|
||||||
|
libjpeg62-turbo libwebp7 libpng16-16 ca-certificates
|
||||||
|
|
||||||
|
Unpinned, every build. Nothing else in this repo looks at it.
|
||||||
|
|
||||||
|
So the checks below run the APPLICATION'S OWN code — `Thumbnailer`, which needs
|
||||||
|
no database and no app context — against whatever Pillow and ffmpeg have
|
||||||
|
become. `ffmpeg -version` exiting 0 would pass while a codec removal or an
|
||||||
|
soname bump broke every thumbnail in the library; producing a thumbnail would
|
||||||
|
not.
|
||||||
|
|
||||||
|
Every failure names the package it implicates. This fires on a Sunday,
|
||||||
|
unattended, about a change nobody made deliberately — "assertion failed" a week
|
||||||
|
later teaches nobody anything.
|
||||||
|
|
||||||
|
Usage: docker run --rm -i <image> shell -c 'python3 -' < scripts/smoke_image.py
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import shutil
|
||||||
|
import subprocess
|
||||||
|
import tempfile
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
try:
|
||||||
|
from PIL import Image
|
||||||
|
|
||||||
|
from backend.app.services.thumbnailer import Thumbnailer
|
||||||
|
except Exception as exc: # noqa: BLE001 — a smoke test reports, it never raises
|
||||||
|
print(f"smoke: FAILED — could not import the thumbnail path at all: {exc}")
|
||||||
|
print(" Implicates Pillow or its shared libraries (libjpeg62-turbo,")
|
||||||
|
print(" libpng16-16, libwebp7), or the python base image itself.")
|
||||||
|
raise SystemExit(1) from exc
|
||||||
|
|
||||||
|
|
||||||
|
# Binary → what stops working without it. Listed individually because
|
||||||
|
# `--no-install-recommends` means any one of them can vanish on its own when a
|
||||||
|
# dependency chain higher up changes.
|
||||||
|
REQUIRED_BINARIES = {
|
||||||
|
"ffmpeg": "video thumbnails and transcoding (Dockerfile: ffmpeg)",
|
||||||
|
"unar": "archive import — cbz/zip/rar members (Dockerfile: unar)",
|
||||||
|
"pg_dump": "database backup (Dockerfile: postgresql-client)",
|
||||||
|
"zstd": "backup compression, pg_dump | tar --zstd (Dockerfile: zstd)",
|
||||||
|
"megatools": "mega.nz public-link downloads, #830 (Dockerfile: megatools)",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def check_jpeg(thumbs: Thumbnailer, src: Path) -> None:
|
||||||
|
path = src / "flat.jpg"
|
||||||
|
Image.new("RGB", (900, 400), (30, 90, 160)).save(path, "JPEG")
|
||||||
|
result = thumbs.generate_image_thumbnail(path, "a" * 64)
|
||||||
|
assert result.mime == "image/jpeg", f"mime was {result.mime}"
|
||||||
|
assert result.path.stat().st_size > 0, "no bytes written"
|
||||||
|
# Re-open it. A file that writes but cannot be read back is the shape a
|
||||||
|
# half-broken codec produces, and size alone would not catch it.
|
||||||
|
with Image.open(result.path) as im:
|
||||||
|
im.load()
|
||||||
|
|
||||||
|
|
||||||
|
def check_png_alpha(thumbs: Thumbnailer, src: Path) -> None:
|
||||||
|
path = src / "alpha.png"
|
||||||
|
Image.new("RGBA", (400, 900), (200, 40, 40, 128)).save(path, "PNG")
|
||||||
|
result = thumbs.generate_image_thumbnail(path, "b" * 64)
|
||||||
|
assert result.mime == "image/png", f"mime was {result.mime}"
|
||||||
|
with Image.open(result.path) as im:
|
||||||
|
im.load()
|
||||||
|
assert im.mode in ("RGBA", "LA", "P"), f"alpha lost, mode={im.mode}"
|
||||||
|
|
||||||
|
|
||||||
|
def check_webp(thumbs: Thumbnailer, src: Path) -> None:
|
||||||
|
path = src / "sample.webp"
|
||||||
|
Image.new("RGB", (500, 500), (10, 140, 70)).save(path, "WEBP")
|
||||||
|
result = thumbs.generate_image_thumbnail(path, "c" * 64)
|
||||||
|
assert result.path.stat().st_size > 0, "no bytes written"
|
||||||
|
|
||||||
|
|
||||||
|
def check_video(thumbs: Thumbnailer, src: Path) -> None:
|
||||||
|
# Synthesised rather than committed as a fixture: a checked-in video is a
|
||||||
|
# binary blob nobody can review, and lavfi ships with every ffmpeg build.
|
||||||
|
#
|
||||||
|
# 3 seconds, not 2. The seek lands at max(1.0, duration * 0.05) = 1.0s, and
|
||||||
|
# a clip barely longer than its own seek is how #1231 produced zero frames.
|
||||||
|
# This check exists to exercise ffmpeg, not to re-litigate that edge.
|
||||||
|
clip = src / "clip.mp4"
|
||||||
|
subprocess.run(
|
||||||
|
["ffmpeg", "-nostdin", "-f", "lavfi", "-i", "testsrc=size=640x360:rate=10",
|
||||||
|
"-t", "3", "-pix_fmt", "yuv420p", "-y", str(clip)],
|
||||||
|
check=True, capture_output=True, timeout=120,
|
||||||
|
)
|
||||||
|
result = thumbs.generate_video_thumbnail(clip, "d" * 64, duration_seconds=3.0)
|
||||||
|
assert result.path.stat().st_size > 0, "no bytes written"
|
||||||
|
with Image.open(result.path) as im:
|
||||||
|
im.load()
|
||||||
|
|
||||||
|
|
||||||
|
CHECKS = (
|
||||||
|
("JPEG thumbnail", "libjpeg62-turbo / Pillow", check_jpeg),
|
||||||
|
("PNG thumbnail (alpha)", "libpng16-16 / Pillow", check_png_alpha),
|
||||||
|
("WebP decode", "libwebp7 / Pillow", check_webp),
|
||||||
|
("video thumbnail", "ffmpeg", check_video),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
failures: list[str] = []
|
||||||
|
|
||||||
|
print("smoke: binaries the apt layer provides")
|
||||||
|
for binary, purpose in REQUIRED_BINARIES.items():
|
||||||
|
if shutil.which(binary) is None:
|
||||||
|
print(f" FAIL {binary}: not on PATH")
|
||||||
|
failures.append(f"{binary} — {purpose}")
|
||||||
|
else:
|
||||||
|
print(f" ok {binary}")
|
||||||
|
|
||||||
|
print("smoke: the application's own thumbnail path, against this image's libraries")
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
root = Path(tmp)
|
||||||
|
src = root / "src"
|
||||||
|
src.mkdir()
|
||||||
|
thumbs = Thumbnailer(root)
|
||||||
|
for name, implicates, fn in CHECKS:
|
||||||
|
try:
|
||||||
|
fn(thumbs, src)
|
||||||
|
print(f" ok {name}")
|
||||||
|
except Exception as exc: # noqa: BLE001 — report every check, then fail once
|
||||||
|
print(f" FAIL {name}: {exc}")
|
||||||
|
failures.append(f"{name} — {implicates}")
|
||||||
|
|
||||||
|
if failures:
|
||||||
|
print(f"\nsmoke: FAILED — {len(failures)} check(s)")
|
||||||
|
for failure in failures:
|
||||||
|
print(f" - {failure}")
|
||||||
|
print("\nThis image was built against freshly resolved base layers. The")
|
||||||
|
print("named packages are where to look: compare this build's apt versions")
|
||||||
|
print("against the previous :latest before assuming the app changed.")
|
||||||
|
return 1
|
||||||
|
|
||||||
|
print("\nsmoke: all checks passed")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
@@ -234,3 +234,56 @@ def test_version_and_revision_describe_the_same_commit(artifact):
|
|||||||
f"in AMO_UNPADDED may differ here."
|
f"in AMO_UNPADDED may differ here."
|
||||||
)
|
)
|
||||||
assert sha.startswith(revision(artifact))
|
assert sha.startswith(revision(artifact))
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("artifact", ARTIFACTS)
|
||||||
|
def test_epoch_is_the_same_commit_the_version_names(artifact):
|
||||||
|
"""The build clock and the version must be one lookup, not two.
|
||||||
|
|
||||||
|
`epoch` feeds SOURCE_DATE_EPOCH, which decides the image config's bytes and
|
||||||
|
therefore the manifest digest; `version` is what the instance reports about
|
||||||
|
itself. If they could name different commits, an image would be stamped
|
||||||
|
reproducibly against one commit while claiming to be another — and both
|
||||||
|
values would look perfectly well-formed, exactly like the divergence the
|
||||||
|
test above guards.
|
||||||
|
|
||||||
|
They cannot, because `cmd_epoch` and `cmd_version` are two fields of one
|
||||||
|
`newest()` result. This pins that they stay that way: a future refactor
|
||||||
|
that gave epoch its own `git log` would pass every other test here.
|
||||||
|
"""
|
||||||
|
epoch = artifacts("epoch", artifact).strip()
|
||||||
|
assert epoch.isdigit(), f"{artifact} epoch is {epoch!r}, not a unix timestamp"
|
||||||
|
|
||||||
|
sha = newest_by_commit_time(artifact)
|
||||||
|
committed = subprocess.run(
|
||||||
|
["git", "show", "-s", "--format=%ct", sha],
|
||||||
|
capture_output=True, text=True, check=True, cwd=ROOT,
|
||||||
|
).stdout.strip()
|
||||||
|
assert epoch == committed, (
|
||||||
|
f"{artifact} derives epoch {epoch}, but its newest shipped commit "
|
||||||
|
f"{sha[:12]} was committed at {committed}. SOURCE_DATE_EPOCH would "
|
||||||
|
f"pin the image config to a commit the version does not name."
|
||||||
|
)
|
||||||
|
|
||||||
|
# And the two renderings must agree, which is the property that actually
|
||||||
|
# matters at build time: same commit in, same digest and same reported
|
||||||
|
# version out.
|
||||||
|
rendered = subprocess.run(
|
||||||
|
["git", "show", "-s", "--format=%cd", "--date=format-local:%Y.%m.%d.%H%M", sha],
|
||||||
|
capture_output=True, text=True, check=True, cwd=ROOT,
|
||||||
|
env={"TZ": "UTC", "PATH": os.environ.get("PATH", "")},
|
||||||
|
).stdout.strip()
|
||||||
|
assert segments(artifacts("version", artifact).strip()) == segments(rendered)
|
||||||
|
|
||||||
|
|
||||||
|
def test_epoch_is_stable_across_calls():
|
||||||
|
"""SOURCE_DATE_EPOCH's entire job is to be the same on the next build.
|
||||||
|
|
||||||
|
A value that moved between two invocations on one unchanged checkout would
|
||||||
|
reintroduce #3265 through the very mechanism meant to close it, and the
|
||||||
|
symptom would be indistinguishable: a digest that changes for no reason.
|
||||||
|
"""
|
||||||
|
for artifact in ARTIFACTS:
|
||||||
|
first = artifacts("epoch", artifact).strip()
|
||||||
|
second = artifacts("epoch", artifact).strip()
|
||||||
|
assert first == second, f"{artifact} epoch moved: {first} then {second}"
|
||||||
|
|||||||
+101
-17
@@ -24,12 +24,36 @@ SCRIPT = ROOT / "scripts" / "release_notes.py"
|
|||||||
|
|
||||||
|
|
||||||
def notes(*args: str, cwd: Path | None = None) -> str:
|
def notes(*args: str, cwd: Path | None = None) -> str:
|
||||||
|
"""Run the script the way release.yml does.
|
||||||
|
|
||||||
|
A synthetic repo runs its OWN copy of the script, because the overview is
|
||||||
|
read relative to `__file__` rather than to the cwd — which is right in
|
||||||
|
production (release.yml checks out the tag, so the script IS the tagged
|
||||||
|
tree's copy) and would otherwise make every synthetic repo silently quote
|
||||||
|
FabledCurator's real README.
|
||||||
|
"""
|
||||||
|
root = cwd or ROOT
|
||||||
|
script = root / "scripts" / "release_notes.py"
|
||||||
return subprocess.run(
|
return subprocess.run(
|
||||||
["python3", str(SCRIPT), "--dry-run", *args],
|
["python3", str(script if script.exists() else SCRIPT), "--dry-run", *args],
|
||||||
capture_output=True, text=True, check=True, cwd=cwd or ROOT,
|
capture_output=True, text=True, check=True, cwd=root,
|
||||||
).stdout
|
).stdout
|
||||||
|
|
||||||
|
|
||||||
|
OVERVIEW_TEXT = "A synthetic product, described once."
|
||||||
|
|
||||||
|
|
||||||
|
def install_script(repo: Path, *, overview: bool = True) -> None:
|
||||||
|
"""Give a synthetic repo the script and a README to quote."""
|
||||||
|
(repo / "scripts").mkdir(exist_ok=True)
|
||||||
|
(repo / "scripts" / "release_notes.py").write_text(SCRIPT.read_text())
|
||||||
|
(repo / "scripts" / "artifacts.sh").write_text("#!/bin/sh\nexit 1\n")
|
||||||
|
readme = "# Synthetic\n\n"
|
||||||
|
if overview:
|
||||||
|
readme += f"<!-- overview:start -->\n{OVERVIEW_TEXT}\n<!-- overview:end -->\n"
|
||||||
|
(repo / "README.md").write_text(readme)
|
||||||
|
|
||||||
|
|
||||||
def body_of(out: str) -> str:
|
def body_of(out: str) -> str:
|
||||||
assert "--- body ---" in out, f"no body was rendered:\n{out}"
|
assert "--- body ---" in out, f"no body was rendered:\n{out}"
|
||||||
return out.split("--- body ---", 1)[1]
|
return out.split("--- body ---", 1)[1]
|
||||||
@@ -56,9 +80,10 @@ def shaped_history(tmp_path: Path) -> Path:
|
|||||||
repo = tmp_path / "shaped"
|
repo = tmp_path / "shaped"
|
||||||
repo.mkdir()
|
repo.mkdir()
|
||||||
git(repo, "init", "-q", "-b", "main")
|
git(repo, "init", "-q", "-b", "main")
|
||||||
|
install_script(repo)
|
||||||
for i, tag in enumerate(("v26.06.04.0", "v2026.08.28.2208", "v2026.08.29.1000")):
|
for i, tag in enumerate(("v26.06.04.0", "v2026.08.28.2208", "v2026.08.29.1000")):
|
||||||
(repo / "f.txt").write_text(f"{i}\n")
|
(repo / "f.txt").write_text(f"{i}\n")
|
||||||
git(repo, "add", "f.txt")
|
git(repo, "add", "-A")
|
||||||
git(repo, "commit", "-q", "-m", f"work landing in {tag}")
|
git(repo, "commit", "-q", "-m", f"work landing in {tag}")
|
||||||
git(repo, "tag", tag)
|
git(repo, "tag", tag)
|
||||||
# One more commit and a merge, so the merge-exclusion test has something to
|
# One more commit and a merge, so the merge-exclusion test has something to
|
||||||
@@ -109,12 +134,57 @@ def test_merges_are_excluded_so_the_list_is_the_work(shaped_history):
|
|||||||
assert "Merge pull request #999" not in body
|
assert "Merge pull request #999" not in body
|
||||||
|
|
||||||
|
|
||||||
def test_the_first_release_still_renders_with_nothing_behind_it(shaped_history):
|
def test_a_pre_convention_tag_is_history_not_a_predecessor(shaped_history):
|
||||||
"""No previous tag is reachable from the oldest one. That is a real state,
|
"""The defect milestone 328 hit, and the reason the match glob narrowed.
|
||||||
not an error, and it must not take the release down with it."""
|
|
||||||
out = notes("v26.06.04.0", cwd=shaped_history)
|
The 28 `v26.*` tags are kept as history while their releases were deleted.
|
||||||
|
Ancestry alone happily names `v26.06.04.0` as the predecessor of the first
|
||||||
|
rule-148 tag — and then the body offers "changes since" a release that no
|
||||||
|
longer exists, over a span (533 commits in the real repo) that is the
|
||||||
|
internal build-out this milestone exists to stop publishing.
|
||||||
|
|
||||||
|
Reachable is not the same as comparable. Only a `vYYYY.` tag names a
|
||||||
|
release a reader could have been running.
|
||||||
|
"""
|
||||||
|
out = notes("v2026.08.28.2208", cwd=shaped_history)
|
||||||
assert "previous=<none>" in out
|
assert "previous=<none>" in out
|
||||||
assert "## Changes" in body_of(out)
|
assert "v26.06.04.0" not in body_of(out)
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_first_release_describes_the_product_instead_of_diffing(shaped_history):
|
||||||
|
"""No predecessor means nobody reading has run an earlier version, so a
|
||||||
|
changelog has no referent. The alternative the script used to take — diff
|
||||||
|
against the whole history, truncated — puts 200 lines of internal build-out
|
||||||
|
in front of someone meeting the project for the first time."""
|
||||||
|
body = body_of(notes("v2026.08.28.2208", cwd=shaped_history))
|
||||||
|
assert OVERVIEW_TEXT in body
|
||||||
|
assert "## Changes" not in body
|
||||||
|
assert not [ln for ln in body.split("\n") if ln.startswith("- work landing")]
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_overview_is_readmes_words_not_a_second_copy(shaped_history):
|
||||||
|
"""Two hand-maintained descriptions of one product drift and nothing
|
||||||
|
catches it. The release page quotes README.md so there is one source."""
|
||||||
|
readme = (shaped_history / "README.md").read_text()
|
||||||
|
assert OVERVIEW_TEXT in readme
|
||||||
|
assert OVERVIEW_TEXT in body_of(notes("v2026.08.28.2208", cwd=shaped_history))
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_missing_overview_block_is_reported_and_still_publishes(tmp_path):
|
||||||
|
"""Same reasoning as cross_checks(): the release is the useful object even
|
||||||
|
when part of the derivation could not run. Say what is missing, publish
|
||||||
|
anyway — do not leave the operator with a tag and no release."""
|
||||||
|
repo = tmp_path / "no-markers"
|
||||||
|
repo.mkdir()
|
||||||
|
git(repo, "init", "-q", "-b", "main")
|
||||||
|
install_script(repo, overview=False)
|
||||||
|
git(repo, "add", "-A")
|
||||||
|
git(repo, "commit", "-q", "-m", "first")
|
||||||
|
git(repo, "tag", "v2026.09.01.1200")
|
||||||
|
|
||||||
|
out = notes("v2026.09.01.1200", cwd=repo)
|
||||||
|
assert "No `<!-- overview:start -->` block found" in out
|
||||||
|
assert "No `<!-- overview:start -->` block found" in body_of(out)
|
||||||
|
|
||||||
|
|
||||||
def test_a_non_tag_ref_renders_but_refuses_to_claim_it_published():
|
def test_a_non_tag_ref_renders_but_refuses_to_claim_it_published():
|
||||||
@@ -135,15 +205,29 @@ def test_the_rollback_refs_name_all_three_images():
|
|||||||
assert f"bvandeusen/{image}:c-" in body, f"{image} missing from the rollback refs"
|
assert f"bvandeusen/{image}:c-" in body, f"{image} missing from the rollback refs"
|
||||||
|
|
||||||
|
|
||||||
def test_an_unbounded_span_is_truncated_and_says_so():
|
def test_a_long_span_between_two_releases_is_truncated_and_says_so(tmp_path):
|
||||||
"""With no reachable previous tag the span is the whole history. Emitting
|
"""The cap is still reachable, just not by the route it used to be.
|
||||||
eleven hundred lines would bury the one line explaining why there are
|
|
||||||
eleven hundred of them, so the cap is part of the message, not a silent
|
It no longer fires on "no predecessor" — that renders the overview now.
|
||||||
slice."""
|
What it still guards is two real releases far enough apart that the list
|
||||||
out = notes("HEAD")
|
stops being something anyone reads, which is the ordinary case for a
|
||||||
if "previous=<none>" not in out:
|
project that cuts a bookmark twice a year. The cap is part of the message,
|
||||||
pytest.skip("a previous tag is reachable from HEAD in this checkout")
|
not a silent slice.
|
||||||
|
"""
|
||||||
|
repo = tmp_path / "long"
|
||||||
|
repo.mkdir()
|
||||||
|
git(repo, "init", "-q", "-b", "main")
|
||||||
|
install_script(repo)
|
||||||
|
git(repo, "add", "-A")
|
||||||
|
git(repo, "commit", "-q", "-m", "scaffold")
|
||||||
|
git(repo, "tag", "v2026.01.01.0000")
|
||||||
|
for i in range(205):
|
||||||
|
git(repo, "commit", "-q", "--allow-empty", "-m", f"fix: change {i}")
|
||||||
|
git(repo, "tag", "v2026.07.01.0000")
|
||||||
|
|
||||||
|
out = notes("v2026.07.01.0000", cwd=repo)
|
||||||
|
assert "previous=v2026.01.01.0000" in out
|
||||||
body = body_of(out)
|
body = body_of(out)
|
||||||
listed = [ln for ln in body.split("\n") if ln.startswith("- ")]
|
listed = [ln for ln in body.split("\n") if ln.startswith("- ")]
|
||||||
assert len(listed) <= 200
|
assert len(listed) == 200
|
||||||
assert "more than a changelog is for" in body
|
assert "more than a changelog is for" in body
|
||||||
|
|||||||
Reference in New Issue
Block a user