Compare commits
57
Commits
ext-1.0.3501409
...
dev
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ad8392b790 | ||
|
|
5084ba666b | ||
|
|
fe4e0f2b71 | ||
|
|
dc8af8b1a7 | ||
|
|
131237143b | ||
|
|
59d27ef76e | ||
|
|
f630e50e75 | ||
|
|
86abaf0b94 | ||
|
|
4815040d74 | ||
|
|
81b7b6f308 | ||
|
|
bfa9fd678b | ||
|
|
24a2b70a5a | ||
|
|
2c88ad3efb | ||
|
|
bfc4f9cec9 | ||
|
|
b590d25f8f | ||
|
|
635138b0d1 | ||
|
|
c0370069e0 | ||
|
|
3590c478f5 | ||
|
|
8a4af589f1 | ||
|
|
aa71cbbdbf | ||
|
|
973db73221 | ||
|
|
bc4eba636d | ||
|
|
dbc4e8b0c6 | ||
|
|
08418d54a3 | ||
|
|
b1bd2531ad | ||
|
|
389afe2f7b | ||
|
|
b979062dd7 | ||
|
|
573228b9da | ||
|
|
d044e93bdb | ||
|
|
ed2b1adc2e | ||
|
|
5e1996e77f | ||
|
|
98b56330d0 | ||
|
|
6959e1220c | ||
|
|
2529b516e6 | ||
|
|
8f1ac0c96a | ||
|
|
5fd171a544 | ||
|
|
62583791d8 | ||
|
|
0a5bbe81dc | ||
|
|
6663e06aa6 | ||
|
|
63e0a423d7 | ||
|
|
5e72076298 | ||
|
|
e21c9fdd34 | ||
|
|
6b3ec98fa8 | ||
|
|
1a941e900b | ||
|
|
2e01242381 | ||
|
|
41f2bec3af | ||
|
|
d38585ed94 | ||
|
|
a3071a7549 | ||
|
|
b6b9fd8287 | ||
|
|
bce894ba24 | ||
|
|
5771fd5770 | ||
|
|
b3989d0224 | ||
|
|
cd0b0ff04a | ||
|
|
454eb3f973 | ||
|
|
7e065fed70 | ||
|
|
dee93faa37 | ||
|
|
d9aa5aa832 |
+97
-18
@@ -1,24 +1,103 @@
|
||||
# Database
|
||||
DB_USER=fabledcurator
|
||||
DB_PASSWORD=changeme_use_a_real_password
|
||||
DB_HOST=postgres
|
||||
DB_PORT=5432
|
||||
DB_NAME=fabledcurator
|
||||
# FabledCurator configuration.
|
||||
#
|
||||
# Copy to `.env` and edit before your first production start:
|
||||
#
|
||||
# cp .env.example .env
|
||||
#
|
||||
# Only the two values under CHANGE THESE actually need your attention. The
|
||||
# rest have working defaults baked into docker-compose.yml and are listed
|
||||
# here so you know they exist, not because you have to set them.
|
||||
#
|
||||
# Almost nothing else lives here on purpose. FabledCurator is configured from
|
||||
# its own Settings UI, backed by the database — no restart, no YAML. If you
|
||||
# are looking for where to set an import path, a download schedule or an ML
|
||||
# threshold, it is in the app, not in this file.
|
||||
|
||||
# Redis / Celery
|
||||
CELERY_BROKER_URL=redis://redis:6379/0
|
||||
CELERY_RESULT_BACKEND=redis://redis:6379/0
|
||||
|
||||
# App
|
||||
# Generate with: openssl rand -hex 32
|
||||
SECRET_KEY=changeme_32_byte_hex_secret
|
||||
# ---------------------------------------------------------------------------
|
||||
# CHANGE THESE
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Extension API key — used in FC-3, lands later but reserved now
|
||||
# Generate with: openssl rand -hex 32
|
||||
EXTENSION_API_KEY=
|
||||
# The Postgres password. docker-compose.yml falls back to a published default
|
||||
# (`fabledcurator_dev`) so that `docker compose up` works with no config at
|
||||
# all — which is exactly why you must not leave it at that on a real install.
|
||||
# It is the credential protecting your stored platform session cookies.
|
||||
DB_PASSWORD=
|
||||
|
||||
# Logging
|
||||
# Sets Quart's app.secret_key. Today it signs nothing: FabledCurator has no
|
||||
# login and uses no session cookies, so no value here is protecting anything
|
||||
# right now. Set it anyway. It is required at boot rather than defaulted so
|
||||
# that the day something session-backed does land, no instance is already
|
||||
# running on a value published in this file.
|
||||
#
|
||||
# openssl rand -hex 32
|
||||
SECRET_KEY=
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# FIRST BOOT ONLY — then delete this line
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# FabledCurator encrypts your stored platform credentials with a Fernet key it
|
||||
# keeps at /images/secrets/credential_key.b64 — inside the ./images bind mount,
|
||||
# so it outlives the container. On a brand-new install that file does not exist
|
||||
# yet, and the app REFUSES TO START rather than quietly create one:
|
||||
#
|
||||
# MissingCredentialKey: Fernet key file not found at
|
||||
# /images/secrets/credential_key.b64
|
||||
#
|
||||
# That refusal is deliberate. Auto-creating a key is indistinguishable from the
|
||||
# disaster case — a restore that brought the database back but lost
|
||||
# ./images/secrets — and there it would mint a key that cannot decrypt anything,
|
||||
# leaving an instance that looks healthy while every paywalled download fails.
|
||||
# So the choice is yours to make explicitly, once.
|
||||
#
|
||||
# Set this for your first `up`, watch the container come up, then DELETE THE
|
||||
# LINE. Leaving it set disarms the protection permanently, on an instance that
|
||||
# by then has credentials worth protecting.
|
||||
#
|
||||
# BACK UP ./images/secrets/ ALONGSIDE YOUR DATABASE. The key is the only thing
|
||||
# that can read your stored credentials; a database restored without it needs
|
||||
# every credential re-entered by hand.
|
||||
CURATOR_BOOTSTRAP_NEW_KEY=1
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Optional — defaults are fine
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Host port the UI is published on. The container always listens on 8080;
|
||||
# this is only the left-hand side of the port mapping.
|
||||
PORT=8080
|
||||
|
||||
# DEBUG | INFO | WARNING | ERROR
|
||||
LOG_LEVEL=INFO
|
||||
|
||||
# Deployment posture: plain HTTP (no TLS in the app; reverse proxy if needed)
|
||||
# See docs/superpowers/specs/2026-05-13-fabledcurator-merge-design.md §2.1
|
||||
# Postgres identity. Change these only if you are pointing at a database you
|
||||
# manage yourself — the bundled postgres service is created with whatever is
|
||||
# set here, so changing them after the first start will not rename anything.
|
||||
DB_USER=fabledcurator
|
||||
DB_NAME=fabledcurator
|
||||
|
||||
# Set by docker-compose.yml to reach the bundled services. Override only when
|
||||
# running Postgres or Redis outside this stack.
|
||||
# DB_HOST=postgres
|
||||
# DB_PORT=5432
|
||||
# CELERY_BROKER_URL=redis://redis:6379/0
|
||||
# CELERY_RESULT_BACKEND=redis://redis:6379/0
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# There is no authentication variable here, and that is not an omission
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# FabledCurator has no login, no accounts and no permission model. Anything
|
||||
# that can reach PORT is an administrator and can read the platform session
|
||||
# cookies the app stores for Patreon, SubscribeStar and Pixiv.
|
||||
#
|
||||
# Bind it to a trusted network. See "Before you expose it" in README.md and
|
||||
# the deployment posture section of SECURITY.md.
|
||||
#
|
||||
# The Firefox extension's API key is NOT configured here — it is generated
|
||||
# automatically on first use and shown under Settings → Maintenance, where you
|
||||
# can also rotate it.
|
||||
|
||||
@@ -0,0 +1,349 @@
|
||||
|
||||
# TEMPORARY — milestone 328. Delete once the baseline has shipped and settled.
|
||||
#
|
||||
# Collapsing 89 alembic revisions into one baseline has exactly one dangerous
|
||||
# failure: the baseline does not reproduce the schema the chain produced, and
|
||||
# the divergence surfaces later, on the operator's live data, in whatever
|
||||
# migration comes next.
|
||||
#
|
||||
# So the comparison happens in CI, against a throwaway pgvector Postgres, where
|
||||
# nothing is at risk. It answers one question: does `upgrade head` on the
|
||||
# collapsed tree produce the same schema as `upgrade head` on the full chain?
|
||||
#
|
||||
# The chain is read out of GIT, not the working tree, which is what lets this
|
||||
# keep working now that the revisions are deleted — `chain_ref` names a commit
|
||||
# that still carries 0001..0089. That is the whole reason this is a workflow
|
||||
# rather than a script someone ran once.
|
||||
#
|
||||
# WHAT THIS CANNOT SEE, and it matters: the comparison is of SCHEMA. Migrations
|
||||
# 0002 and 0003 also INSERTED rows (the import_settings and ml_settings
|
||||
# singletons), and the application reads those with scalar_one(), which raises
|
||||
# on an empty result. A baseline that omitted them would produce an identical
|
||||
# schema, pass this check with a perfect diff, and crash a fresh install on its
|
||||
# first settings access. Only running the app against a new database finds
|
||||
# that class of defect. Do not read a green run here as "the baseline is
|
||||
# correct" — read it as "the schema is correct".
|
||||
#
|
||||
# Autogenerate now emits nearly all of the baseline unaided, which was NOT true
|
||||
# before #3275 put the previously migration-only objects onto the models — the
|
||||
# HNSW index with its opclass, the COALESCE expression index, the partial
|
||||
# unique indexes, 107 server_defaults, the enum CHECKs. An earlier attempt at
|
||||
# this squash was reverted precisely because the generator dropped them all
|
||||
# silently. What still needs hand-adding is only what cannot live in a model:
|
||||
# the two CREATE EXTENSION statements, the two seed rows, and the pgvector
|
||||
# import the generator forgets to write.
|
||||
name: Alembic baseline
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
chain_ref:
|
||||
description: 'Commit/tag carrying the full 0001..0089 chain (pinned: the tree no longer has it)'
|
||||
type: string
|
||||
default: '725bf15'
|
||||
mode:
|
||||
description: 'chain = compare against this tree''s migrations; models = compare against a schema built from the MODELS'
|
||||
type: string
|
||||
default: 'chain'
|
||||
|
||||
jobs:
|
||||
compare:
|
||||
runs-on: python-ci
|
||||
container:
|
||||
image: git.fabledsword.com/bvandeusen/ci-python:3.14
|
||||
env:
|
||||
DB_USER: fabledcurator
|
||||
DB_PASSWORD: ci_integration
|
||||
DB_PORT: "5432"
|
||||
DB_NAME: fabledcurator_test
|
||||
SECRET_KEY: ci_integration_placeholder
|
||||
services:
|
||||
postgres:
|
||||
image: pgvector/pgvector:pg16
|
||||
env:
|
||||
POSTGRES_USER: fabledcurator
|
||||
POSTGRES_PASSWORD: ci_integration
|
||||
POSTGRES_DB: fabledcurator_test
|
||||
options: >-
|
||||
--health-cmd "pg_isready -U fabledcurator"
|
||||
--health-interval 10s
|
||||
--health-timeout 5s
|
||||
--health-retries 10
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
# Full history is the point: `chain_ref` is read out of git, so a
|
||||
# shallow clone would not have the revisions to compare against.
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Resolve the Postgres service and install deps
|
||||
run: |
|
||||
set -eux
|
||||
# Same service-IP dance as ci.yml's integration job; see the long
|
||||
# comment there for why the job name must stay separator-free.
|
||||
PG=$(docker ps --filter "name=compare" --filter "ancestor=pgvector/pgvector:pg16" -q | head -n1)
|
||||
test -n "$PG"
|
||||
PG_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$PG")
|
||||
test -n "$PG_IP"
|
||||
echo "PG_CONTAINER=$PG" >> "$GITHUB_ENV"
|
||||
echo "DB_HOST=$PG_IP" >> "$GITHUB_ENV"
|
||||
# Socket probe in python, not bash's /dev/tcp — these steps run under
|
||||
# `sh -e`, where that path does not exist. Same fix and same reasoning
|
||||
# as ci.yml's integration job; see the comment there.
|
||||
pg_ready=""
|
||||
for i in $(seq 1 60); do
|
||||
if python -c "import socket,sys; s=socket.socket(); s.settimeout(2); sys.exit(0 if s.connect_ex(('$PG_IP', 5432)) == 0 else 1)"; then
|
||||
pg_ready=1
|
||||
break
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
if [ -z "$pg_ready" ]; then
|
||||
echo "postgres at $PG_IP:5432 did not accept a connection within 120s"
|
||||
exit 1
|
||||
fi
|
||||
if command -v uv >/dev/null 2>&1; then
|
||||
uv pip install --system -r requirements.txt
|
||||
else
|
||||
pip install -r requirements.txt
|
||||
fi
|
||||
|
||||
# DB 1: the 87-revision chain, read out of git at `chain_ref`.
|
||||
#
|
||||
# A git worktree rather than a checkout, so the current tree — which is
|
||||
# what we are testing — is left completely alone.
|
||||
- name: Build the schema the OLD chain produces
|
||||
env:
|
||||
CHAIN_REF: ${{ github.event.inputs.chain_ref }}
|
||||
THIS_SHA: ${{ github.sha }}
|
||||
run: |
|
||||
set -eux
|
||||
docker exec "$PG_CONTAINER" createdb -U fabledcurator fc_chain
|
||||
# Blank means "the chain in this ref", which is what you want while
|
||||
# the chain is still intact — comparing the models against a PINNED
|
||||
# older commit reports every migration written since as a difference.
|
||||
# Pin it only after the collapse, when the tree no longer has them.
|
||||
git worktree add /tmp/chain "${CHAIN_REF:-$THIS_SHA}"
|
||||
ls /tmp/chain/alembic/versions/*.py | wc -l
|
||||
cd /tmp/chain
|
||||
DB_NAME=fc_chain alembic upgrade head
|
||||
cd -
|
||||
docker exec "$PG_CONTAINER" pg_dump -U fabledcurator --schema-only \
|
||||
--no-owner --no-privileges -d fc_chain > chain.sql
|
||||
wc -l chain.sql
|
||||
# Emit the dump itself, checksummed, for local analysis. Reconciling
|
||||
# the models against the deployed schema (#3275) needs the ACTUAL
|
||||
# schema, not an inference from a diff — parsing table context out of
|
||||
# unified-diff hunks drops every table whose CREATE TABLE line falls
|
||||
# outside a hunk, which silently under-reports.
|
||||
#
|
||||
# base64 + sha256 for the same reason as the candidate: a plain cat
|
||||
# of a file this size was truncated mid-line by the runner with the
|
||||
# step still green (run 4964).
|
||||
set +x
|
||||
B64=$(base64 -w 120 chain.sql)
|
||||
echo "===== BEGIN CHAIN SCHEMA (base64) ====="
|
||||
echo "$B64"
|
||||
echo "===== END CHAIN SCHEMA ====="
|
||||
echo "chain-sha256: $(sha256sum chain.sql | cut -d' ' -f1)"
|
||||
echo "chain-bytes: $(wc -c < chain.sql)"
|
||||
set -x
|
||||
|
||||
# A candidate baseline, autogenerated from the models against an EMPTY
|
||||
# database so every table shows up as a create. Printed for a human to
|
||||
# finish — it will be missing the three raw-SQL items named at the top.
|
||||
#
|
||||
# Gated on the TREE, not on a workflow input. A `type: boolean` input
|
||||
# read back as `github.event.inputs.generate == 'true'` silently
|
||||
# evaluated false on this runner (run 4960 skipped this step entirely
|
||||
# with no diagnostic) — the same `github.event.inputs` typing quirk
|
||||
# build.yml already works around. The file count is the real question
|
||||
# anyway: there is nothing to generate once the chain is collapsed.
|
||||
- name: Autogenerate a candidate baseline
|
||||
run: |
|
||||
set -eux
|
||||
if [ "$(ls alembic/versions/*.py | wc -l)" -le 1 ]; then
|
||||
echo "already collapsed — nothing to generate"
|
||||
exit 0
|
||||
fi
|
||||
docker exec "$PG_CONTAINER" createdb -U fabledcurator fc_gen
|
||||
# Hide the existing revisions so alembic sees an empty history and
|
||||
# emits the whole schema rather than a delta.
|
||||
mkdir -p /tmp/versions_held
|
||||
mv alembic/versions/*.py /tmp/versions_held/ 2>/dev/null || true
|
||||
DB_NAME=fc_gen alembic revision --autogenerate -m "baseline" || true
|
||||
# Printed rather than uploaded: ci-requirements.md records that this
|
||||
# runner cannot do actions/upload-artifact@v4+, and the repo dropped
|
||||
# the action entirely in 2026-05, so the job log is the retrieval
|
||||
# channel actually proven here.
|
||||
#
|
||||
# base64, not the raw file. A plain `cat` of the ~33KB candidate was
|
||||
# TRUNCATED MID-LINE by the runner on run 4964 — it stopped inside
|
||||
# `sa.Column('mime', sa.String(length=128)` and carried straight on
|
||||
# to the next traced command, with the step still green. A silent
|
||||
# cut in the middle of a schema definition is the worst possible
|
||||
# failure here, because the truncated text still looks like a
|
||||
# plausible file.
|
||||
#
|
||||
# base64 at a fixed narrow width gives many short lines instead of
|
||||
# few long ones, and — the actual point — a checksum and a line
|
||||
# count that make truncation DETECTABLE rather than invisible.
|
||||
set +x
|
||||
F=$(ls alembic/versions/*.py | head -1)
|
||||
B64=$(base64 -w 120 "$F")
|
||||
echo "===== BEGIN CANDIDATE BASELINE (base64) ====="
|
||||
echo "$B64"
|
||||
echo "===== END CANDIDATE BASELINE ====="
|
||||
echo "candidate-sha256: $(sha256sum "$F" | cut -d' ' -f1)"
|
||||
echo "candidate-bytes: $(wc -c < "$F")"
|
||||
echo "candidate-b64-lines: $(echo "$B64" | wc -l)"
|
||||
set -x
|
||||
mkdir -p /tmp/candidate
|
||||
cp alembic/versions/*.py /tmp/candidate/
|
||||
# Put the tree back exactly as it was; this job never mutates state.
|
||||
rm -f alembic/versions/*.py
|
||||
mv /tmp/versions_held/*.py alembic/versions/ 2>/dev/null || true
|
||||
|
||||
# DB 2: what the CURRENT tree produces.
|
||||
#
|
||||
# `mode: models` applies the candidate autogenerated from the MODELS
|
||||
# instead, which is what answers "do the models describe the schema?" —
|
||||
# the question #3275 exists because nobody had ever asked it. Under that
|
||||
# mode a clean diff means autogenerate is trustworthy again.
|
||||
#
|
||||
# The two extensions are created by hand first. They are database
|
||||
# objects, not table metadata, so no model can carry them and their
|
||||
# absence is not a model defect — it is simply outside what this
|
||||
# comparison is asking about.
|
||||
- name: Build the schema the CURRENT tree produces
|
||||
env:
|
||||
MODE: ${{ github.event.inputs.mode }}
|
||||
run: |
|
||||
set -eux
|
||||
docker exec "$PG_CONTAINER" createdb -U fabledcurator fc_base
|
||||
if [ "${MODE:-chain}" = "models" ]; then
|
||||
docker exec "$PG_CONTAINER" psql -U fabledcurator -d fc_base \
|
||||
-c "CREATE EXTENSION IF NOT EXISTS vector" \
|
||||
-c "CREATE EXTENSION IF NOT EXISTS tsm_system_rows"
|
||||
mkdir -p /tmp/held
|
||||
mv alembic/versions/*.py /tmp/held/
|
||||
cp /tmp/candidate/*.py alembic/versions/
|
||||
# Autogenerate EMITS pgvector.sqlalchemy.vector.VECTOR(...) without
|
||||
# importing it, so the file it writes cannot run:
|
||||
# NameError: name 'pgvector' is not defined
|
||||
# Observed on run 4988, which is the proof rather than the theory.
|
||||
# This is a defect in the GENERATOR, not in the models, so it is
|
||||
# repaired here rather than counted as a schema difference — the
|
||||
# comparison is about whether the models describe the schema.
|
||||
sed -i '0,/^import sqlalchemy as sa$/s//import sqlalchemy as sa\nimport pgvector.sqlalchemy.vector/' alembic/versions/*.py
|
||||
grep -n 'import pgvector' alembic/versions/*.py
|
||||
ls alembic/versions/*.py
|
||||
DB_NAME=fc_base alembic upgrade head
|
||||
rm -f alembic/versions/*.py
|
||||
mv /tmp/held/*.py alembic/versions/
|
||||
else
|
||||
ls alembic/versions/*.py | wc -l
|
||||
DB_NAME=fc_base alembic upgrade head
|
||||
fi
|
||||
docker exec "$PG_CONTAINER" pg_dump -U fabledcurator --schema-only \
|
||||
--no-owner --no-privileges -d fc_base > baseline.sql
|
||||
wc -l baseline.sql
|
||||
|
||||
# The verdict.
|
||||
#
|
||||
# pg_dump orders dumpable objects by name within type, not by creation
|
||||
# order, so two schemas built by different routes are directly
|
||||
# comparable. Normalisation is deliberately minimal, because a filter
|
||||
# that hides a real difference is the one way this check passes when it
|
||||
# should fail — blank lines, SQL comments, trailing whitespace, and:
|
||||
#
|
||||
# \restrict / \unrestrict — a per-invocation RANDOM NONCE that newer
|
||||
# pg_dump emits to fence the dump against injection during restore. It
|
||||
# differs on every run by construction, so it is noise by definition,
|
||||
# not a schema difference. Measured on run 4960, the control: two dumps
|
||||
# of the SAME schema came back 1123 lines each and differed on exactly
|
||||
# these two lines and nothing else. That control is what licenses this
|
||||
# filter — it was observed to be the only false positive, rather than
|
||||
# assumed to be one.
|
||||
# Column ORDER inside a CREATE TABLE is compared separately from column
|
||||
# CONTENT, and only content is fatal.
|
||||
#
|
||||
# A table built by 87 migrations has its columns in ADD COLUMN order; the
|
||||
# same table built in one shot has them in declaration order. That is a
|
||||
# real and permanent difference which no baseline can erase — the
|
||||
# operator's existing database keeps chain order forever, a fresh install
|
||||
# gets model order — so a check that fails on it would never pass and
|
||||
# would teach nothing. FC reaches every column through the ORM by name,
|
||||
# and `SELECT *` ordering is not depended on anywhere.
|
||||
#
|
||||
# So the second pass SORTS the column lines within each CREATE TABLE
|
||||
# rather than DELETING them. That distinction is the whole point: sorting
|
||||
# cannot hide a column that exists on one side only, or one whose type,
|
||||
# nullability or default differs — those still land in the diff. A filter
|
||||
# could have hidden all three.
|
||||
#
|
||||
# Both diffs are reported. The ordered one is informational; the
|
||||
# order-insensitive one is the verdict.
|
||||
- name: Diff
|
||||
run: |
|
||||
set -eu
|
||||
norm() {
|
||||
grep -vE '^\s*(--|$)' "$1" \
|
||||
| grep -vE '^\\(un)?restrict ' \
|
||||
| sed 's/[[:space:]]*$//'
|
||||
}
|
||||
norm chain.sql > a.txt
|
||||
norm baseline.sql > b.txt
|
||||
echo "normalised: chain=$(wc -l < a.txt) lines, current=$(wc -l < b.txt) lines"
|
||||
|
||||
sort_table_columns() {
|
||||
python3 - "$1" <<'PYEOF'
|
||||
import re, sys
|
||||
|
||||
lines = open(sys.argv[1]).read().splitlines()
|
||||
out, block = [], None
|
||||
for line in lines:
|
||||
if block is not None:
|
||||
# ');' on its own closes the CREATE TABLE body.
|
||||
if line.strip() == ");":
|
||||
out.extend(sorted(block))
|
||||
out.append(line)
|
||||
block = None
|
||||
else:
|
||||
# Drop the list comma before sorting. Only the LAST
|
||||
# column lacks one, so keeping it would make every
|
||||
# reordering look like a content change as well — the
|
||||
# comma is punctuation, and carries no schema meaning.
|
||||
block.append(line.rstrip().rstrip(","))
|
||||
continue
|
||||
out.append(line)
|
||||
if re.match(r"CREATE TABLE .*\($", line):
|
||||
block = []
|
||||
if block is not None: # unterminated body: emit it rather than drop it
|
||||
out.extend(block)
|
||||
print("\n".join(out))
|
||||
PYEOF
|
||||
}
|
||||
sort_table_columns a.txt > a.sorted.txt
|
||||
sort_table_columns b.txt > b.sorted.txt
|
||||
test "$(wc -l < a.sorted.txt)" = "$(wc -l < a.txt)"
|
||||
test "$(wc -l < b.sorted.txt)" = "$(wc -l < b.txt)"
|
||||
|
||||
if diff -u a.txt b.txt > schema.diff; then
|
||||
echo "ORDERED DIFF: identical, column order included."
|
||||
else
|
||||
echo "ORDERED DIFF: $(grep -cE '^[+-]' schema.diff) changed lines (informational):"
|
||||
cat schema.diff
|
||||
fi
|
||||
echo
|
||||
echo "================================================================"
|
||||
echo
|
||||
if diff -u a.sorted.txt b.sorted.txt > sorted.diff; then
|
||||
echo "SCHEMAS MATCH — every difference above is column ORDER alone."
|
||||
else
|
||||
echo "SCHEMAS DIFFER — $(grep -cE '^[+-]' sorted.diff) changed lines that are NOT ordering:"
|
||||
cat sorted.diff
|
||||
echo
|
||||
echo "The baseline is wrong, not the database. Do not stamp."
|
||||
exit 1
|
||||
fi
|
||||
+1516
-345
File diff suppressed because it is too large
Load Diff
+57
-23
@@ -2,7 +2,7 @@ name: CI
|
||||
|
||||
# CI lanes per FabledRulebook/forgejo.md "CI philosophy":
|
||||
# - lint: ruff only, no dep install — fast-fail for the common lint bounce.
|
||||
# - extension-version: the derived version resolves and MAJOR.MINOR agrees.
|
||||
# - extension-version: the derived version resolves and is a shape AMO takes.
|
||||
# - backend-lint-and-test: `pytest -m "not integration"`, no service containers.
|
||||
# - frontend-build: vitest unit + vite build.
|
||||
# - integration: pgvector + redis service containers; alembic + `pytest -m integration`.
|
||||
@@ -35,7 +35,10 @@ jobs:
|
||||
- name: Ruff lint
|
||||
# agent/ included so the GPU-agent is linted before its image is built
|
||||
# (build.yml only `docker build`s it — this is where it gets checked).
|
||||
run: ruff check backend/ tests/ alembic/ agent/
|
||||
# scripts/ likewise: release_notes.py runs only on a tag push, so a
|
||||
# syntax or import error there would otherwise surface at the one
|
||||
# moment nobody wants to debug a workflow.
|
||||
run: ruff check backend/ tests/ alembic/ agent/ scripts/
|
||||
- name: Agent syntax check
|
||||
# The agent's runtime deps (torch/transformers/ultralytics) aren't in the
|
||||
# CI image, so we can't import it — but compileall parses every module,
|
||||
@@ -55,9 +58,12 @@ jobs:
|
||||
# the extension.yml suite runs on node:24-slim, which is exactly why
|
||||
# version.spec.js sticks to packaging.sh's git-free subcommands.
|
||||
# 1. the derivation actually resolves on this commit
|
||||
# 2. MAJOR.MINOR agrees between the two files — the one part still hand-set,
|
||||
# and packaging.sh reads it from manifest.json ALONE, so a divergence
|
||||
# ships a version package.json disagrees with
|
||||
# 2. the derived string is one AMO will accept, checked against Mozilla's
|
||||
# own published grammar rather than a loose "digits and dots"
|
||||
#
|
||||
# The MAJOR.MINOR-agreement check that used to be (2) is gone with milestone
|
||||
# 318 step 8: the committed version no longer seeds anything, so there is no
|
||||
# hand-set part left for the two files to disagree about.
|
||||
#
|
||||
# Deliberately NOT checked here: that the derived value beats what has already
|
||||
# been signed. That guard belongs in build.yml, where it compares against the
|
||||
@@ -82,27 +88,38 @@ jobs:
|
||||
# busybox sh on the act_runner — no bashisms (family rule).
|
||||
VERSION=$(sh extension/scripts/packaging.sh version)
|
||||
echo "derived: $VERSION"
|
||||
# The shape AMO accepts, and the shape build.yml will stamp.
|
||||
if ! echo "$VERSION" | grep -qE '^[0-9]+(\.[0-9]+)*$'; then
|
||||
echo "ERROR: derived version '$VERSION' is not plain dotted-numeric."
|
||||
echo "AMO would reject it, and build.yml stamps it verbatim."
|
||||
|
||||
# Mozilla's published grammar for AMO, transcribed verbatim from
|
||||
# MDN's manifest.json/version page:
|
||||
#
|
||||
# ^(0|[1-9][0-9]{0,8})([.](0|[1-9][0-9]{0,8})){0,3}$
|
||||
#
|
||||
# Not the looser `^[0-9]+(\.[0-9]+)*$` this lane used to carry. That
|
||||
# one passes `2026.08.29.0201`, which AMO REJECTS — a segment must be
|
||||
# the single digit 0 or start 1-9 — and it also passes five segments,
|
||||
# where AMO allows four. Both would surface as a failed sign with the
|
||||
# version already burned: AMO 409s on re-signing, so a rejected value
|
||||
# cannot be reclaimed and cannot be reused. This lane is the cheap
|
||||
# place to find out. (#3138, milestone 318 step 8.)
|
||||
if ! echo "$VERSION" | grep -qE '^(0|[1-9][0-9]{0,8})(\.(0|[1-9][0-9]{0,8})){0,3}$'; then
|
||||
echo "ERROR: derived version '$VERSION' is not a version AMO accepts."
|
||||
echo "AMO's grammar: ^(0|[1-9][0-9]{0,8})([.](0|[1-9][0-9]{0,8})){0,3}$"
|
||||
echo "Most likely cause: a zero-padded segment (08, 0201). The rest"
|
||||
echo "of the family pads; the extension must not — see packaging.sh."
|
||||
exit 1
|
||||
fi
|
||||
mm() { grep -E '"version"' "$1" | head -1 | sed -E 's/.*"version"[[:space:]]*:[[:space:]]*"([0-9]+\.[0-9]+).*/\1/'; }
|
||||
MAN=$(mm extension/manifest.json)
|
||||
PKG=$(mm extension/package.json)
|
||||
test -n "$MAN" || { echo "ERROR: no parseable version in extension/manifest.json"; exit 1; }
|
||||
test -n "$PKG" || { echo "ERROR: no parseable version in extension/package.json"; exit 1; }
|
||||
if [ "$MAN" != "$PKG" ]; then
|
||||
echo "ERROR: MAJOR.MINOR disagrees between the two files."
|
||||
echo " extension/manifest.json = $MAN <- packaging.sh reads MAJOR.MINOR from here"
|
||||
echo " extension/package.json = $PKG"
|
||||
echo "Only MAJOR.MINOR is hand-set. The patch component is derived from"
|
||||
echo "commit time and overwritten at build time, so the committed patch"
|
||||
echo "numbers are inert — but MAJOR.MINOR still ships. Set both the same."
|
||||
|
||||
# ...and the shape this project actually derives. AMO would happily
|
||||
# take `1.0.3500147` too, so the grammar check alone would not notice
|
||||
# a regression to the pre-318 shape — which orders BELOW everything
|
||||
# signed since, and is unrecoverable once Firefox has the higher one.
|
||||
if ! echo "$VERSION" | grep -qE '^20[0-9][0-9]\.[0-9]{1,2}\.[0-9]{1,2}\.[0-9]{1,4}$'; then
|
||||
echo "ERROR: derived version '$VERSION' is not YYYY.M.D.HHMM."
|
||||
echo "Rule 148's CalVer is what build.yml signs; the old"
|
||||
echo "1.0.<minutes> shape would order below every ext-2026.* release."
|
||||
exit 1
|
||||
fi
|
||||
echo "OK: MAJOR.MINOR $MAN, derived version $VERSION"
|
||||
echo "OK: derived version $VERSION"
|
||||
|
||||
backend-lint-and-test:
|
||||
runs-on: python-ci
|
||||
@@ -238,10 +255,27 @@ jobs:
|
||||
export DB_HOST="$PG_IP"
|
||||
export CELERY_BROKER_URL="redis://$RD_IP:6379/0"
|
||||
export CELERY_RESULT_BACKEND="redis://$RD_IP:6379/0"
|
||||
# These steps run under `sh -e`, not bash, so bash's /dev/tcp magic
|
||||
# path does not exist here — the probe this loop used to run could
|
||||
# never succeed and simply burned the full 120s on every run, green
|
||||
# or red, then continued without having established anything. Python
|
||||
# is in the image and needs no installed package for a socket
|
||||
# connect, so it is the probe. Exhausting the budget is now a named
|
||||
# failure rather than a silent fall-through (rule 156): if Postgres
|
||||
# is genuinely not up, that is what the log should say, instead of
|
||||
# whatever the first query happens to raise two minutes later.
|
||||
pg_ready=""
|
||||
for i in $(seq 1 60); do
|
||||
(echo > "/dev/tcp/$PG_IP/5432") >/dev/null 2>&1 && break
|
||||
if python -c "import socket,sys; s=socket.socket(); s.settimeout(2); sys.exit(0 if s.connect_ex(('$PG_IP', 5432)) == 0 else 1)"; then
|
||||
pg_ready=1
|
||||
break
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
if [ -z "$pg_ready" ]; then
|
||||
echo "postgres at $PG_IP:5432 did not accept a connection within 120s"
|
||||
exit 1
|
||||
fi
|
||||
if command -v uv >/dev/null 2>&1; then
|
||||
uv pip install --system -r requirements.txt pytest pytest-asyncio
|
||||
else
|
||||
|
||||
@@ -0,0 +1,80 @@
|
||||
name: Release
|
||||
|
||||
# A `v*` tag publishes a changelog. It does NOT build anything.
|
||||
#
|
||||
# Milestone 318 step 2 removed the tag trigger from build.yml: by the time
|
||||
# anyone tags a commit, `main` has already built and published it, and a
|
||||
# rebuild would re-push `:c-<sha>` — which rule 145 forbids even when the
|
||||
# source matches, since image configs carry timestamps and "same source" does
|
||||
# not mean "same manifest". That left the tag with no consequence at all.
|
||||
#
|
||||
# This is the consequence it has instead. Step 6 put the derived version in the
|
||||
# Settings footer, so an operator can say WHICH build they are running; this
|
||||
# says what is IN it that was not in the one they ran last month. Both halves
|
||||
# of one question (note #3127 §5).
|
||||
#
|
||||
# Nothing here runs on a schedule and nothing auto-tags on merge. Release tags
|
||||
# are bookmarks — cut one when you will want to point at that day by name,
|
||||
# otherwise don't (note #3127 §0). FC went twelve weeks between v26.06.04.0 and
|
||||
# the next one and nothing was wrong. A schedule would turn an optional
|
||||
# bookmark back into ceremony, which is the thing this milestone is removing.
|
||||
#
|
||||
# Cutting the tag is an explicit operator action under rule 2 ("`main` — never
|
||||
# without explicit request", which since 2026-08-28 covers PR, merge and tag
|
||||
# alike). This lane only decides what happens once they do.
|
||||
#
|
||||
# Requires repo secret RELEASE_TOKEN with the `write:release` scope — the same
|
||||
# PAT build.yml uses for the ext-<version> XPI asset cache.
|
||||
|
||||
on:
|
||||
push:
|
||||
tags: ['v*']
|
||||
# So a release body can be regenerated after the fact — the publisher PATCHes
|
||||
# an existing release rather than falling through on a conflict, so re-running
|
||||
# this on a tag rewrites the body instead of silently keeping the first one
|
||||
# (note #3127 §6.7).
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
tag:
|
||||
description: 'Tag to (re)publish notes for'
|
||||
required: true
|
||||
|
||||
jobs:
|
||||
changelog:
|
||||
runs-on: python-ci
|
||||
container:
|
||||
image: git.fabledsword.com/bvandeusen/ci-python:3.14
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
# Load-bearing twice over: the previous release is found by walking
|
||||
# ancestry back through the tag graph, and the cross-check against
|
||||
# the derived web version calls artifacts.sh, which reads commit
|
||||
# times. A shallow clone would find no previous tag and emit the
|
||||
# entire history as the changelog — plausible-looking and wrong.
|
||||
fetch-depth: 0
|
||||
ref: ${{ github.event.inputs.tag || github.ref }}
|
||||
|
||||
# The `:c-<sha>` rollback refs are only real if `main` built this commit.
|
||||
# The script checks that against origin/main and downgrades the claim to
|
||||
# "unverified" when it cannot resolve one; fetching it here means that
|
||||
# downgrade stays an actual signal instead of firing on every release.
|
||||
- name: Make main's history resolvable
|
||||
run: git fetch --no-tags --quiet origin +main:refs/remotes/origin/main || true
|
||||
|
||||
# TAG goes through the environment, not through `${{ }}` inside the
|
||||
# run block. The value is operator-supplied, and an expression expanded
|
||||
# into a shell line is expanded BEFORE the shell sees it — there is no
|
||||
# quoting that makes that safe. On a tag push it is empty and the script
|
||||
# falls back to GITHUB_REF.
|
||||
- name: Publish the derived changelog
|
||||
env:
|
||||
RELEASE_TOKEN: ${{ secrets.RELEASE_TOKEN }}
|
||||
TAG: ${{ github.event.inputs.tag }}
|
||||
run: |
|
||||
set -eu
|
||||
if [ -n "${TAG:-}" ]; then
|
||||
python3 scripts/release_notes.py "$TAG"
|
||||
else
|
||||
python3 scripts/release_notes.py
|
||||
fi
|
||||
@@ -0,0 +1,77 @@
|
||||
# Contributing
|
||||
|
||||
FabledCurator is developed by a single maintainer for their own use, and
|
||||
published because it may be useful to others. That shapes what contribution
|
||||
looks like here.
|
||||
|
||||
**Issues are welcome** — bug reports, and questions about running it, are
|
||||
genuinely useful and often the fastest way to find out that something is
|
||||
broken outside the one environment it was built in.
|
||||
|
||||
**Open an issue before writing a pull request.** Not as a formality: the
|
||||
project has opinions that are not obvious from the code, and it is unpleasant
|
||||
for everyone when a finished patch turns out to conflict with one. A short
|
||||
issue first costs you nothing and may save you an evening.
|
||||
|
||||
**Contributions are licensed under the AGPL-3.0**, like the rest of the
|
||||
project. By submitting one you agree it ships under that licence. There is no
|
||||
CLA and no copyright assignment.
|
||||
|
||||
## Running it for development
|
||||
|
||||
```bash
|
||||
docker compose up -d # UI on http://localhost:8080
|
||||
```
|
||||
|
||||
The dev override (`docker-compose.override.yml`) is auto-merged and builds the
|
||||
app images locally from source, so this needs no `.env` and no registry
|
||||
access. Postgres and Redis ports are exposed on the host.
|
||||
|
||||
## What CI checks
|
||||
|
||||
Every push runs these, and they are the definition of done for a change:
|
||||
|
||||
```bash
|
||||
ruff check backend/ tests/ alembic/ agent/ scripts/ # lint (and import order)
|
||||
pytest tests/ -m "not integration" # backend unit tests
|
||||
pytest tests/ -m integration # needs pgvector + redis
|
||||
cd frontend && npm run test:unit && npm run build # frontend
|
||||
```
|
||||
|
||||
The integration lane builds its schema by running the real migrations
|
||||
(`alembic upgrade head`), never from ORM metadata — so a migration that does
|
||||
not apply cleanly fails CI rather than being discovered later.
|
||||
|
||||
Note for the linter: ruff's isort runs with `order-by-type`, which sorts
|
||||
ALL-CAPS names ahead of CamelCase. `from sqlalchemy import JSON, DateTime, ...`
|
||||
is correct; putting `JSON` alphabetically between `Integer` and `String` is
|
||||
not. This catches people out.
|
||||
|
||||
## Database changes
|
||||
|
||||
The ORM models and the migration chain must agree. This is enforced, and it is
|
||||
enforced because they silently diverged for a long time and nobody noticed
|
||||
until they were compared: the models were missing indexes, defaults and
|
||||
uniqueness guarantees that only ever existed inside a migration, which made
|
||||
`alembic revision --autogenerate` actively unsafe to run.
|
||||
|
||||
So: if you change a model, write the migration; if you write a migration,
|
||||
change the model to match. Both, in the same commit.
|
||||
|
||||
Adding a value to a CHECK-constrained column means swapping the constraint in
|
||||
the same change — the constraint is not documentation, and a new value without
|
||||
it fails at insert time.
|
||||
|
||||
## Branch model
|
||||
|
||||
`dev` is where work happens. `main` is production and is only reached by a
|
||||
merge from `dev`, never pushed to directly. If you are sending a pull request,
|
||||
target `dev`.
|
||||
|
||||
## Style
|
||||
|
||||
Match the surrounding code. The one convention worth stating explicitly is
|
||||
that comments here explain *why*, especially where a choice looks wrong at a
|
||||
glance — a comment recording which migration a constraint came from, or why a
|
||||
default is a `text()` rather than a string, is the kind that has repeatedly
|
||||
turned out to be worth its space.
|
||||
+9
-3
@@ -58,11 +58,17 @@ COPY --from=frontend-builder /build/dist ./frontend/dist
|
||||
# exactly the shape every reader already has to handle.
|
||||
#
|
||||
# Declared LAST on purpose. An ARG/ENV invalidates every layer below it, and
|
||||
# this is the one value that differs between the dev and main builds of
|
||||
# identical source — put it any earlier and the two channels could never share
|
||||
# a cached pip install.
|
||||
# these are the values that differ between builds of otherwise identical
|
||||
# source — put them any earlier and the two channels could never share a
|
||||
# cached pip install.
|
||||
#
|
||||
# FC_VERSION is what the instance reports about itself in the UI. Since
|
||||
# milestone 318 stopped publishing version image tags, that self-report is
|
||||
# the only answer to "which build is this?" — nothing else names it.
|
||||
ARG FC_CHANNEL=""
|
||||
ENV FC_CHANNEL=${FC_CHANNEL}
|
||||
ARG FC_VERSION=""
|
||||
ENV FC_VERSION=${FC_VERSION}
|
||||
|
||||
EXPOSE 8080
|
||||
|
||||
|
||||
@@ -0,0 +1,661 @@
|
||||
GNU AFFERO GENERAL PUBLIC LICENSE
|
||||
Version 3, 19 November 2007
|
||||
|
||||
Copyright (C) 2007 Free Software Foundation, Inc. <https://fsf.org/>
|
||||
Everyone is permitted to copy and distribute verbatim copies
|
||||
of this license document, but changing it is not allowed.
|
||||
|
||||
Preamble
|
||||
|
||||
The GNU Affero General Public License is a free, copyleft license for
|
||||
software and other kinds of works, specifically designed to ensure
|
||||
cooperation with the community in the case of network server software.
|
||||
|
||||
The licenses for most software and other practical works are designed
|
||||
to take away your freedom to share and change the works. By contrast,
|
||||
our General Public Licenses are intended to guarantee your freedom to
|
||||
share and change all versions of a program--to make sure it remains free
|
||||
software for all its users.
|
||||
|
||||
When we speak of free software, we are referring to freedom, not
|
||||
price. Our General Public Licenses are designed to make sure that you
|
||||
have the freedom to distribute copies of free software (and charge for
|
||||
them if you wish), that you receive source code or can get it if you
|
||||
want it, that you can change the software or use pieces of it in new
|
||||
free programs, and that you know you can do these things.
|
||||
|
||||
Developers that use our General Public Licenses protect your rights
|
||||
with two steps: (1) assert copyright on the software, and (2) offer
|
||||
you this License which gives you legal permission to copy, distribute
|
||||
and/or modify the software.
|
||||
|
||||
A secondary benefit of defending all users' freedom is that
|
||||
improvements made in alternate versions of the program, if they
|
||||
receive widespread use, become available for other developers to
|
||||
incorporate. Many developers of free software are heartened and
|
||||
encouraged by the resulting cooperation. However, in the case of
|
||||
software used on network servers, this result may fail to come about.
|
||||
The GNU General Public License permits making a modified version and
|
||||
letting the public access it on a server without ever releasing its
|
||||
source code to the public.
|
||||
|
||||
The GNU Affero General Public License is designed specifically to
|
||||
ensure that, in such cases, the modified source code becomes available
|
||||
to the community. It requires the operator of a network server to
|
||||
provide the source code of the modified version running there to the
|
||||
users of that server. Therefore, public use of a modified version, on
|
||||
a publicly accessible server, gives the public access to the source
|
||||
code of the modified version.
|
||||
|
||||
An older license, called the Affero General Public License and
|
||||
published by Affero, was designed to accomplish similar goals. This is
|
||||
a different license, not a version of the Affero GPL, but Affero has
|
||||
released a new version of the Affero GPL which permits relicensing under
|
||||
this license.
|
||||
|
||||
The precise terms and conditions for copying, distribution and
|
||||
modification follow.
|
||||
|
||||
TERMS AND CONDITIONS
|
||||
|
||||
0. Definitions.
|
||||
|
||||
"This License" refers to version 3 of the GNU Affero General Public License.
|
||||
|
||||
"Copyright" also means copyright-like laws that apply to other kinds of
|
||||
works, such as semiconductor masks.
|
||||
|
||||
"The Program" refers to any copyrightable work licensed under this
|
||||
License. Each licensee is addressed as "you". "Licensees" and
|
||||
"recipients" may be individuals or organizations.
|
||||
|
||||
To "modify" a work means to copy from or adapt all or part of the work
|
||||
in a fashion requiring copyright permission, other than the making of an
|
||||
exact copy. The resulting work is called a "modified version" of the
|
||||
earlier work or a work "based on" the earlier work.
|
||||
|
||||
A "covered work" means either the unmodified Program or a work based
|
||||
on the Program.
|
||||
|
||||
To "propagate" a work means to do anything with it that, without
|
||||
permission, would make you directly or secondarily liable for
|
||||
infringement under applicable copyright law, except executing it on a
|
||||
computer or modifying a private copy. Propagation includes copying,
|
||||
distribution (with or without modification), making available to the
|
||||
public, and in some countries other activities as well.
|
||||
|
||||
To "convey" a work means any kind of propagation that enables other
|
||||
parties to make or receive copies. Mere interaction with a user through
|
||||
a computer network, with no transfer of a copy, is not conveying.
|
||||
|
||||
An interactive user interface displays "Appropriate Legal Notices"
|
||||
to the extent that it includes a convenient and prominently visible
|
||||
feature that (1) displays an appropriate copyright notice, and (2)
|
||||
tells the user that there is no warranty for the work (except to the
|
||||
extent that warranties are provided), that licensees may convey the
|
||||
work under this License, and how to view a copy of this License. If
|
||||
the interface presents a list of user commands or options, such as a
|
||||
menu, a prominent item in the list meets this criterion.
|
||||
|
||||
1. Source Code.
|
||||
|
||||
The "source code" for a work means the preferred form of the work
|
||||
for making modifications to it. "Object code" means any non-source
|
||||
form of a work.
|
||||
|
||||
A "Standard Interface" means an interface that either is an official
|
||||
standard defined by a recognized standards body, or, in the case of
|
||||
interfaces specified for a particular programming language, one that
|
||||
is widely used among developers working in that language.
|
||||
|
||||
The "System Libraries" of an executable work include anything, other
|
||||
than the work as a whole, that (a) is included in the normal form of
|
||||
packaging a Major Component, but which is not part of that Major
|
||||
Component, and (b) serves only to enable use of the work with that
|
||||
Major Component, or to implement a Standard Interface for which an
|
||||
implementation is available to the public in source code form. A
|
||||
"Major Component", in this context, means a major essential component
|
||||
(kernel, window system, and so on) of the specific operating system
|
||||
(if any) on which the executable work runs, or a compiler used to
|
||||
produce the work, or an object code interpreter used to run it.
|
||||
|
||||
The "Corresponding Source" for a work in object code form means all
|
||||
the source code needed to generate, install, and (for an executable
|
||||
work) run the object code and to modify the work, including scripts to
|
||||
control those activities. However, it does not include the work's
|
||||
System Libraries, or general-purpose tools or generally available free
|
||||
programs which are used unmodified in performing those activities but
|
||||
which are not part of the work. For example, Corresponding Source
|
||||
includes interface definition files associated with source files for
|
||||
the work, and the source code for shared libraries and dynamically
|
||||
linked subprograms that the work is specifically designed to require,
|
||||
such as by intimate data communication or control flow between those
|
||||
subprograms and other parts of the work.
|
||||
|
||||
The Corresponding Source need not include anything that users
|
||||
can regenerate automatically from other parts of the Corresponding
|
||||
Source.
|
||||
|
||||
The Corresponding Source for a work in source code form is that
|
||||
same work.
|
||||
|
||||
2. Basic Permissions.
|
||||
|
||||
All rights granted under this License are granted for the term of
|
||||
copyright on the Program, and are irrevocable provided the stated
|
||||
conditions are met. This License explicitly affirms your unlimited
|
||||
permission to run the unmodified Program. The output from running a
|
||||
covered work is covered by this License only if the output, given its
|
||||
content, constitutes a covered work. This License acknowledges your
|
||||
rights of fair use or other equivalent, as provided by copyright law.
|
||||
|
||||
You may make, run and propagate covered works that you do not
|
||||
convey, without conditions so long as your license otherwise remains
|
||||
in force. You may convey covered works to others for the sole purpose
|
||||
of having them make modifications exclusively for you, or provide you
|
||||
with facilities for running those works, provided that you comply with
|
||||
the terms of this License in conveying all material for which you do
|
||||
not control copyright. Those thus making or running the covered works
|
||||
for you must do so exclusively on your behalf, under your direction
|
||||
and control, on terms that prohibit them from making any copies of
|
||||
your copyrighted material outside their relationship with you.
|
||||
|
||||
Conveying under any other circumstances is permitted solely under
|
||||
the conditions stated below. Sublicensing is not allowed; section 10
|
||||
makes it unnecessary.
|
||||
|
||||
3. Protecting Users' Legal Rights From Anti-Circumvention Law.
|
||||
|
||||
No covered work shall be deemed part of an effective technological
|
||||
measure under any applicable law fulfilling obligations under article
|
||||
11 of the WIPO copyright treaty adopted on 20 December 1996, or
|
||||
similar laws prohibiting or restricting circumvention of such
|
||||
measures.
|
||||
|
||||
When you convey a covered work, you waive any legal power to forbid
|
||||
circumvention of technological measures to the extent such circumvention
|
||||
is effected by exercising rights under this License with respect to
|
||||
the covered work, and you disclaim any intention to limit operation or
|
||||
modification of the work as a means of enforcing, against the work's
|
||||
users, your or third parties' legal rights to forbid circumvention of
|
||||
technological measures.
|
||||
|
||||
4. Conveying Verbatim Copies.
|
||||
|
||||
You may convey verbatim copies of the Program's source code as you
|
||||
receive it, in any medium, provided that you conspicuously and
|
||||
appropriately publish on each copy an appropriate copyright notice;
|
||||
keep intact all notices stating that this License and any
|
||||
non-permissive terms added in accord with section 7 apply to the code;
|
||||
keep intact all notices of the absence of any warranty; and give all
|
||||
recipients a copy of this License along with the Program.
|
||||
|
||||
You may charge any price or no price for each copy that you convey,
|
||||
and you may offer support or warranty protection for a fee.
|
||||
|
||||
5. Conveying Modified Source Versions.
|
||||
|
||||
You may convey a work based on the Program, or the modifications to
|
||||
produce it from the Program, in the form of source code under the
|
||||
terms of section 4, provided that you also meet all of these conditions:
|
||||
|
||||
a) The work must carry prominent notices stating that you modified
|
||||
it, and giving a relevant date.
|
||||
|
||||
b) The work must carry prominent notices stating that it is
|
||||
released under this License and any conditions added under section
|
||||
7. This requirement modifies the requirement in section 4 to
|
||||
"keep intact all notices".
|
||||
|
||||
c) You must license the entire work, as a whole, under this
|
||||
License to anyone who comes into possession of a copy. This
|
||||
License will therefore apply, along with any applicable section 7
|
||||
additional terms, to the whole of the work, and all its parts,
|
||||
regardless of how they are packaged. This License gives no
|
||||
permission to license the work in any other way, but it does not
|
||||
invalidate such permission if you have separately received it.
|
||||
|
||||
d) If the work has interactive user interfaces, each must display
|
||||
Appropriate Legal Notices; however, if the Program has interactive
|
||||
interfaces that do not display Appropriate Legal Notices, your
|
||||
work need not make them do so.
|
||||
|
||||
A compilation of a covered work with other separate and independent
|
||||
works, which are not by their nature extensions of the covered work,
|
||||
and which are not combined with it such as to form a larger program,
|
||||
in or on a volume of a storage or distribution medium, is called an
|
||||
"aggregate" if the compilation and its resulting copyright are not
|
||||
used to limit the access or legal rights of the compilation's users
|
||||
beyond what the individual works permit. Inclusion of a covered work
|
||||
in an aggregate does not cause this License to apply to the other
|
||||
parts of the aggregate.
|
||||
|
||||
6. Conveying Non-Source Forms.
|
||||
|
||||
You may convey a covered work in object code form under the terms
|
||||
of sections 4 and 5, provided that you also convey the
|
||||
machine-readable Corresponding Source under the terms of this License,
|
||||
in one of these ways:
|
||||
|
||||
a) Convey the object code in, or embodied in, a physical product
|
||||
(including a physical distribution medium), accompanied by the
|
||||
Corresponding Source fixed on a durable physical medium
|
||||
customarily used for software interchange.
|
||||
|
||||
b) Convey the object code in, or embodied in, a physical product
|
||||
(including a physical distribution medium), accompanied by a
|
||||
written offer, valid for at least three years and valid for as
|
||||
long as you offer spare parts or customer support for that product
|
||||
model, to give anyone who possesses the object code either (1) a
|
||||
copy of the Corresponding Source for all the software in the
|
||||
product that is covered by this License, on a durable physical
|
||||
medium customarily used for software interchange, for a price no
|
||||
more than your reasonable cost of physically performing this
|
||||
conveying of source, or (2) access to copy the
|
||||
Corresponding Source from a network server at no charge.
|
||||
|
||||
c) Convey individual copies of the object code with a copy of the
|
||||
written offer to provide the Corresponding Source. This
|
||||
alternative is allowed only occasionally and noncommercially, and
|
||||
only if you received the object code with such an offer, in accord
|
||||
with subsection 6b.
|
||||
|
||||
d) Convey the object code by offering access from a designated
|
||||
place (gratis or for a charge), and offer equivalent access to the
|
||||
Corresponding Source in the same way through the same place at no
|
||||
further charge. You need not require recipients to copy the
|
||||
Corresponding Source along with the object code. If the place to
|
||||
copy the object code is a network server, the Corresponding Source
|
||||
may be on a different server (operated by you or a third party)
|
||||
that supports equivalent copying facilities, provided you maintain
|
||||
clear directions next to the object code saying where to find the
|
||||
Corresponding Source. Regardless of what server hosts the
|
||||
Corresponding Source, you remain obligated to ensure that it is
|
||||
available for as long as needed to satisfy these requirements.
|
||||
|
||||
e) Convey the object code using peer-to-peer transmission, provided
|
||||
you inform other peers where the object code and Corresponding
|
||||
Source of the work are being offered to the general public at no
|
||||
charge under subsection 6d.
|
||||
|
||||
A separable portion of the object code, whose source code is excluded
|
||||
from the Corresponding Source as a System Library, need not be
|
||||
included in conveying the object code work.
|
||||
|
||||
A "User Product" is either (1) a "consumer product", which means any
|
||||
tangible personal property which is normally used for personal, family,
|
||||
or household purposes, or (2) anything designed or sold for incorporation
|
||||
into a dwelling. In determining whether a product is a consumer product,
|
||||
doubtful cases shall be resolved in favor of coverage. For a particular
|
||||
product received by a particular user, "normally used" refers to a
|
||||
typical or common use of that class of product, regardless of the status
|
||||
of the particular user or of the way in which the particular user
|
||||
actually uses, or expects or is expected to use, the product. A product
|
||||
is a consumer product regardless of whether the product has substantial
|
||||
commercial, industrial or non-consumer uses, unless such uses represent
|
||||
the only significant mode of use of the product.
|
||||
|
||||
"Installation Information" for a User Product means any methods,
|
||||
procedures, authorization keys, or other information required to install
|
||||
and execute modified versions of a covered work in that User Product from
|
||||
a modified version of its Corresponding Source. The information must
|
||||
suffice to ensure that the continued functioning of the modified object
|
||||
code is in no case prevented or interfered with solely because
|
||||
modification has been made.
|
||||
|
||||
If you convey an object code work under this section in, or with, or
|
||||
specifically for use in, a User Product, and the conveying occurs as
|
||||
part of a transaction in which the right of possession and use of the
|
||||
User Product is transferred to the recipient in perpetuity or for a
|
||||
fixed term (regardless of how the transaction is characterized), the
|
||||
Corresponding Source conveyed under this section must be accompanied
|
||||
by the Installation Information. But this requirement does not apply
|
||||
if neither you nor any third party retains the ability to install
|
||||
modified object code on the User Product (for example, the work has
|
||||
been installed in ROM).
|
||||
|
||||
The requirement to provide Installation Information does not include a
|
||||
requirement to continue to provide support service, warranty, or updates
|
||||
for a work that has been modified or installed by the recipient, or for
|
||||
the User Product in which it has been modified or installed. Access to a
|
||||
network may be denied when the modification itself materially and
|
||||
adversely affects the operation of the network or violates the rules and
|
||||
protocols for communication across the network.
|
||||
|
||||
Corresponding Source conveyed, and Installation Information provided,
|
||||
in accord with this section must be in a format that is publicly
|
||||
documented (and with an implementation available to the public in
|
||||
source code form), and must require no special password or key for
|
||||
unpacking, reading or copying.
|
||||
|
||||
7. Additional Terms.
|
||||
|
||||
"Additional permissions" are terms that supplement the terms of this
|
||||
License by making exceptions from one or more of its conditions.
|
||||
Additional permissions that are applicable to the entire Program shall
|
||||
be treated as though they were included in this License, to the extent
|
||||
that they are valid under applicable law. If additional permissions
|
||||
apply only to part of the Program, that part may be used separately
|
||||
under those permissions, but the entire Program remains governed by
|
||||
this License without regard to the additional permissions.
|
||||
|
||||
When you convey a copy of a covered work, you may at your option
|
||||
remove any additional permissions from that copy, or from any part of
|
||||
it. (Additional permissions may be written to require their own
|
||||
removal in certain cases when you modify the work.) You may place
|
||||
additional permissions on material, added by you to a covered work,
|
||||
for which you have or can give appropriate copyright permission.
|
||||
|
||||
Notwithstanding any other provision of this License, for material you
|
||||
add to a covered work, you may (if authorized by the copyright holders of
|
||||
that material) supplement the terms of this License with terms:
|
||||
|
||||
a) Disclaiming warranty or limiting liability differently from the
|
||||
terms of sections 15 and 16 of this License; or
|
||||
|
||||
b) Requiring preservation of specified reasonable legal notices or
|
||||
author attributions in that material or in the Appropriate Legal
|
||||
Notices displayed by works containing it; or
|
||||
|
||||
c) Prohibiting misrepresentation of the origin of that material, or
|
||||
requiring that modified versions of such material be marked in
|
||||
reasonable ways as different from the original version; or
|
||||
|
||||
d) Limiting the use for publicity purposes of names of licensors or
|
||||
authors of the material; or
|
||||
|
||||
e) Declining to grant rights under trademark law for use of some
|
||||
trade names, trademarks, or service marks; or
|
||||
|
||||
f) Requiring indemnification of licensors and authors of that
|
||||
material by anyone who conveys the material (or modified versions of
|
||||
it) with contractual assumptions of liability to the recipient, for
|
||||
any liability that these contractual assumptions directly impose on
|
||||
those licensors and authors.
|
||||
|
||||
All other non-permissive additional terms are considered "further
|
||||
restrictions" within the meaning of section 10. If the Program as you
|
||||
received it, or any part of it, contains a notice stating that it is
|
||||
governed by this License along with a term that is a further
|
||||
restriction, you may remove that term. If a license document contains
|
||||
a further restriction but permits relicensing or conveying under this
|
||||
License, you may add to a covered work material governed by the terms
|
||||
of that license document, provided that the further restriction does
|
||||
not survive such relicensing or conveying.
|
||||
|
||||
If you add terms to a covered work in accord with this section, you
|
||||
must place, in the relevant source files, a statement of the
|
||||
additional terms that apply to those files, or a notice indicating
|
||||
where to find the applicable terms.
|
||||
|
||||
Additional terms, permissive or non-permissive, may be stated in the
|
||||
form of a separately written license, or stated as exceptions;
|
||||
the above requirements apply either way.
|
||||
|
||||
8. Termination.
|
||||
|
||||
You may not propagate or modify a covered work except as expressly
|
||||
provided under this License. Any attempt otherwise to propagate or
|
||||
modify it is void, and will automatically terminate your rights under
|
||||
this License (including any patent licenses granted under the third
|
||||
paragraph of section 11).
|
||||
|
||||
However, if you cease all violation of this License, then your
|
||||
license from a particular copyright holder is reinstated (a)
|
||||
provisionally, unless and until the copyright holder explicitly and
|
||||
finally terminates your license, and (b) permanently, if the copyright
|
||||
holder fails to notify you of the violation by some reasonable means
|
||||
prior to 60 days after the cessation.
|
||||
|
||||
Moreover, your license from a particular copyright holder is
|
||||
reinstated permanently if the copyright holder notifies you of the
|
||||
violation by some reasonable means, this is the first time you have
|
||||
received notice of violation of this License (for any work) from that
|
||||
copyright holder, and you cure the violation prior to 30 days after
|
||||
your receipt of the notice.
|
||||
|
||||
Termination of your rights under this section does not terminate the
|
||||
licenses of parties who have received copies or rights from you under
|
||||
this License. If your rights have been terminated and not permanently
|
||||
reinstated, you do not qualify to receive new licenses for the same
|
||||
material under section 10.
|
||||
|
||||
9. Acceptance Not Required for Having Copies.
|
||||
|
||||
You are not required to accept this License in order to receive or
|
||||
run a copy of the Program. Ancillary propagation of a covered work
|
||||
occurring solely as a consequence of using peer-to-peer transmission
|
||||
to receive a copy likewise does not require acceptance. However,
|
||||
nothing other than this License grants you permission to propagate or
|
||||
modify any covered work. These actions infringe copyright if you do
|
||||
not accept this License. Therefore, by modifying or propagating a
|
||||
covered work, you indicate your acceptance of this License to do so.
|
||||
|
||||
10. Automatic Licensing of Downstream Recipients.
|
||||
|
||||
Each time you convey a covered work, the recipient automatically
|
||||
receives a license from the original licensors, to run, modify and
|
||||
propagate that work, subject to this License. You are not responsible
|
||||
for enforcing compliance by third parties with this License.
|
||||
|
||||
An "entity transaction" is a transaction transferring control of an
|
||||
organization, or substantially all assets of one, or subdividing an
|
||||
organization, or merging organizations. If propagation of a covered
|
||||
work results from an entity transaction, each party to that
|
||||
transaction who receives a copy of the work also receives whatever
|
||||
licenses to the work the party's predecessor in interest had or could
|
||||
give under the previous paragraph, plus a right to possession of the
|
||||
Corresponding Source of the work from the predecessor in interest, if
|
||||
the predecessor has it or can get it with reasonable efforts.
|
||||
|
||||
You may not impose any further restrictions on the exercise of the
|
||||
rights granted or affirmed under this License. For example, you may
|
||||
not impose a license fee, royalty, or other charge for exercise of
|
||||
rights granted under this License, and you may not initiate litigation
|
||||
(including a cross-claim or counterclaim in a lawsuit) alleging that
|
||||
any patent claim is infringed by making, using, selling, offering for
|
||||
sale, or importing the Program or any portion of it.
|
||||
|
||||
11. Patents.
|
||||
|
||||
A "contributor" is a copyright holder who authorizes use under this
|
||||
License of the Program or a work on which the Program is based. The
|
||||
work thus licensed is called the contributor's "contributor version".
|
||||
|
||||
A contributor's "essential patent claims" are all patent claims
|
||||
owned or controlled by the contributor, whether already acquired or
|
||||
hereafter acquired, that would be infringed by some manner, permitted
|
||||
by this License, of making, using, or selling its contributor version,
|
||||
but do not include claims that would be infringed only as a
|
||||
consequence of further modification of the contributor version. For
|
||||
purposes of this definition, "control" includes the right to grant
|
||||
patent sublicenses in a manner consistent with the requirements of
|
||||
this License.
|
||||
|
||||
Each contributor grants you a non-exclusive, worldwide, royalty-free
|
||||
patent license under the contributor's essential patent claims, to
|
||||
make, use, sell, offer for sale, import and otherwise run, modify and
|
||||
propagate the contents of its contributor version.
|
||||
|
||||
In the following three paragraphs, a "patent license" is any express
|
||||
agreement or commitment, however denominated, not to enforce a patent
|
||||
(such as an express permission to practice a patent or covenant not to
|
||||
sue for patent infringement). To "grant" such a patent license to a
|
||||
party means to make such an agreement or commitment not to enforce a
|
||||
patent against the party.
|
||||
|
||||
If you convey a covered work, knowingly relying on a patent license,
|
||||
and the Corresponding Source of the work is not available for anyone
|
||||
to copy, free of charge and under the terms of this License, through a
|
||||
publicly available network server or other readily accessible means,
|
||||
then you must either (1) cause the Corresponding Source to be so
|
||||
available, or (2) arrange to deprive yourself of the benefit of the
|
||||
patent license for this particular work, or (3) arrange, in a manner
|
||||
consistent with the requirements of this License, to extend the patent
|
||||
license to downstream recipients. "Knowingly relying" means you have
|
||||
actual knowledge that, but for the patent license, your conveying the
|
||||
covered work in a country, or your recipient's use of the covered work
|
||||
in a country, would infringe one or more identifiable patents in that
|
||||
country that you have reason to believe are valid.
|
||||
|
||||
If, pursuant to or in connection with a single transaction or
|
||||
arrangement, you convey, or propagate by procuring conveyance of, a
|
||||
covered work, and grant a patent license to some of the parties
|
||||
receiving the covered work authorizing them to use, propagate, modify
|
||||
or convey a specific copy of the covered work, then the patent license
|
||||
you grant is automatically extended to all recipients of the covered
|
||||
work and works based on it.
|
||||
|
||||
A patent license is "discriminatory" if it does not include within
|
||||
the scope of its coverage, prohibits the exercise of, or is
|
||||
conditioned on the non-exercise of one or more of the rights that are
|
||||
specifically granted under this License. You may not convey a covered
|
||||
work if you are a party to an arrangement with a third party that is
|
||||
in the business of distributing software, under which you make payment
|
||||
to the third party based on the extent of your activity of conveying
|
||||
the work, and under which the third party grants, to any of the
|
||||
parties who would receive the covered work from you, a discriminatory
|
||||
patent license (a) in connection with copies of the covered work
|
||||
conveyed by you (or copies made from those copies), or (b) primarily
|
||||
for and in connection with specific products or compilations that
|
||||
contain the covered work, unless you entered into that arrangement,
|
||||
or that patent license was granted, prior to 28 March 2007.
|
||||
|
||||
Nothing in this License shall be construed as excluding or limiting
|
||||
any implied license or other defenses to infringement that may
|
||||
otherwise be available to you under applicable patent law.
|
||||
|
||||
12. No Surrender of Others' Freedom.
|
||||
|
||||
If conditions are imposed on you (whether by court order, agreement or
|
||||
otherwise) that contradict the conditions of this License, they do not
|
||||
excuse you from the conditions of this License. If you cannot convey a
|
||||
covered work so as to satisfy simultaneously your obligations under this
|
||||
License and any other pertinent obligations, then as a consequence you may
|
||||
not convey it at all. For example, if you agree to terms that obligate you
|
||||
to collect a royalty for further conveying from those to whom you convey
|
||||
the Program, the only way you could satisfy both those terms and this
|
||||
License would be to refrain entirely from conveying the Program.
|
||||
|
||||
13. Remote Network Interaction; Use with the GNU General Public License.
|
||||
|
||||
Notwithstanding any other provision of this License, if you modify the
|
||||
Program, your modified version must prominently offer all users
|
||||
interacting with it remotely through a computer network (if your version
|
||||
supports such interaction) an opportunity to receive the Corresponding
|
||||
Source of your version by providing access to the Corresponding Source
|
||||
from a network server at no charge, through some standard or customary
|
||||
means of facilitating copying of software. This Corresponding Source
|
||||
shall include the Corresponding Source for any work covered by version 3
|
||||
of the GNU General Public License that is incorporated pursuant to the
|
||||
following paragraph.
|
||||
|
||||
Notwithstanding any other provision of this License, you have
|
||||
permission to link or combine any covered work with a work licensed
|
||||
under version 3 of the GNU General Public License into a single
|
||||
combined work, and to convey the resulting work. The terms of this
|
||||
License will continue to apply to the part which is the covered work,
|
||||
but the work with which it is combined will remain governed by version
|
||||
3 of the GNU General Public License.
|
||||
|
||||
14. Revised Versions of this License.
|
||||
|
||||
The Free Software Foundation may publish revised and/or new versions of
|
||||
the GNU Affero General Public License from time to time. Such new versions
|
||||
will be similar in spirit to the present version, but may differ in detail to
|
||||
address new problems or concerns.
|
||||
|
||||
Each version is given a distinguishing version number. If the
|
||||
Program specifies that a certain numbered version of the GNU Affero General
|
||||
Public License "or any later version" applies to it, you have the
|
||||
option of following the terms and conditions either of that numbered
|
||||
version or of any later version published by the Free Software
|
||||
Foundation. If the Program does not specify a version number of the
|
||||
GNU Affero General Public License, you may choose any version ever published
|
||||
by the Free Software Foundation.
|
||||
|
||||
If the Program specifies that a proxy can decide which future
|
||||
versions of the GNU Affero General Public License can be used, that proxy's
|
||||
public statement of acceptance of a version permanently authorizes you
|
||||
to choose that version for the Program.
|
||||
|
||||
Later license versions may give you additional or different
|
||||
permissions. However, no additional obligations are imposed on any
|
||||
author or copyright holder as a result of your choosing to follow a
|
||||
later version.
|
||||
|
||||
15. Disclaimer of Warranty.
|
||||
|
||||
THERE IS NO WARRANTY FOR THE PROGRAM, TO THE EXTENT PERMITTED BY
|
||||
APPLICABLE LAW. EXCEPT WHEN OTHERWISE STATED IN WRITING THE COPYRIGHT
|
||||
HOLDERS AND/OR OTHER PARTIES PROVIDE THE PROGRAM "AS IS" WITHOUT WARRANTY
|
||||
OF ANY KIND, EITHER EXPRESSED OR IMPLIED, INCLUDING, BUT NOT LIMITED TO,
|
||||
THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
PURPOSE. THE ENTIRE RISK AS TO THE QUALITY AND PERFORMANCE OF THE PROGRAM
|
||||
IS WITH YOU. SHOULD THE PROGRAM PROVE DEFECTIVE, YOU ASSUME THE COST OF
|
||||
ALL NECESSARY SERVICING, REPAIR OR CORRECTION.
|
||||
|
||||
16. Limitation of Liability.
|
||||
|
||||
IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING
|
||||
WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MODIFIES AND/OR CONVEYS
|
||||
THE PROGRAM AS PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY
|
||||
GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING OUT OF THE
|
||||
USE OR INABILITY TO USE THE PROGRAM (INCLUDING BUT NOT LIMITED TO LOSS OF
|
||||
DATA OR DATA BEING RENDERED INACCURATE OR LOSSES SUSTAINED BY YOU OR THIRD
|
||||
PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE WITH ANY OTHER PROGRAMS),
|
||||
EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE POSSIBILITY OF
|
||||
SUCH DAMAGES.
|
||||
|
||||
17. Interpretation of Sections 15 and 16.
|
||||
|
||||
If the disclaimer of warranty and limitation of liability provided
|
||||
above cannot be given local legal effect according to their terms,
|
||||
reviewing courts shall apply local law that most closely approximates
|
||||
an absolute waiver of all civil liability in connection with the
|
||||
Program, unless a warranty or assumption of liability accompanies a
|
||||
copy of the Program in return for a fee.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
How to Apply These Terms to Your New Programs
|
||||
|
||||
If you develop a new program, and you want it to be of the greatest
|
||||
possible use to the public, the best way to achieve this is to make it
|
||||
free software which everyone can redistribute and change under these terms.
|
||||
|
||||
To do so, attach the following notices to the program. It is safest
|
||||
to attach them to the start of each source file to most effectively
|
||||
state the exclusion of warranty; and each file should have at least
|
||||
the "copyright" line and a pointer to where the full notice is found.
|
||||
|
||||
<one line to give the program's name and a brief idea of what it does.>
|
||||
Copyright (C) <year> <name of author>
|
||||
|
||||
This program is free software: you can redistribute it and/or modify
|
||||
it under the terms of the GNU Affero General Public License as published by
|
||||
the Free Software Foundation, either version 3 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU Affero General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU Affero General Public License
|
||||
along with this program. If not, see <https://www.gnu.org/licenses/>.
|
||||
|
||||
Also add information on how to contact you by electronic and paper mail.
|
||||
|
||||
If your software can interact with users remotely through a computer
|
||||
network, you should also make sure that it provides a way for users to
|
||||
get its source. For example, if your program is a web application, its
|
||||
interface could display a "Source" link that leads users to an archive
|
||||
of the code. There are many ways you could offer source, and different
|
||||
solutions will be better for different programs; see section 13 for the
|
||||
specific requirements.
|
||||
|
||||
You should also get your employer (if you work as a programmer) or school,
|
||||
if any, to sign a "copyright disclaimer" for the program, if necessary.
|
||||
For more information on this, and how to apply and follow the GNU AGPL, see
|
||||
<https://www.gnu.org/licenses/>.
|
||||
@@ -1,14 +1,259 @@
|
||||
# FabledCurator
|
||||
|
||||
Self-hosted media curation — gallery, ML tagging, and subscription-driven downloading in one app. Part of the FabledSword family.
|
||||
<!-- overview:start -->
|
||||
Self-hosted media curation — a gallery, ML auto-tagging, and subscription-driven
|
||||
downloading in one application. Part of the FabledSword family.
|
||||
|
||||
Combines what was [ImageRepo](https://git.fabledsword.com/bvandeusen/ImageRepo) (gallery, ML, importer) and [GallerySubscriber](https://git.fabledsword.com/bvandeusen/GallerySubscriber) (gallery-dl wrapper, subscriptions, credential capture) into a single product.
|
||||
## What it does
|
||||
|
||||
## Status
|
||||
You point it at creators you follow. It downloads what they post, files it,
|
||||
tags it, and gives you something better than a folder full of images to look
|
||||
through afterwards.
|
||||
|
||||
In production. `main` is continuously deployed — every merge to `main` builds
|
||||
and publishes `:latest` images, so whatever is on `main` is what is running.
|
||||
Day-to-day work happens on `dev`, which publishes `:dev` images.
|
||||
- **Gallery and browsing.** Images, videos and multi-page works, organised by
|
||||
artist, tag, post and series. A Showcase front page, a filterable gallery, a
|
||||
similarity-driven Explore view, and a page-turning reader for series.
|
||||
- **Subscriptions.** Follows creators on Patreon, SubscribeStar, Pixiv and
|
||||
anything `gallery-dl` supports, on a schedule. Handles paywalled posts using
|
||||
your own logged-in session.
|
||||
- **ML tagging.** Runs image models in-container to suggest tags, group
|
||||
characters, find near-duplicates and power similarity search. Suggestions are
|
||||
reviewable — it proposes, you confirm, and it learns which proposals you keep
|
||||
rejecting.
|
||||
- **Deduplication and provenance.** Everything that arrives is hashed and
|
||||
deduplicated by content, metadata sidecars are read wherever the source
|
||||
writes them, and every file keeps a record of where it came from.
|
||||
- **Maintenance.** Backups, library audits, thumbnail and embedding backfills,
|
||||
orphan cleanup — all from the UI, all as background jobs you can watch.
|
||||
|
||||
Everything is configured from the Settings UI and stored in the database. There
|
||||
is no config file to edit beyond a handful of bootstrap environment variables.
|
||||
<!-- overview:end -->
|
||||
|
||||
## Before you expose it
|
||||
|
||||
**FabledCurator has no login.** There are no user accounts, no passwords and no
|
||||
permission model. Anything that can reach the port is an administrator.
|
||||
|
||||
That matters more here than it would in most self-hosted apps, because of what
|
||||
this one stores: **live platform session cookies for Patreon, SubscribeStar and
|
||||
Pixiv** — accounts that usually have a payment method attached. Whoever reaches
|
||||
the port can read them, alongside your entire library.
|
||||
|
||||
So:
|
||||
|
||||
- Bind it to a LAN, a VPN, or a tunnel you control.
|
||||
- Do not port-forward it. Do not put it on a public hostname.
|
||||
- A reverse proxy that adds TLS but no authentication **does not help**. If you
|
||||
want it reachable from outside, put an authenticating proxy in front of it —
|
||||
a forward-auth provider, HTTP basic auth, an identity-aware tunnel — and treat
|
||||
that layer as the only thing standing between the internet and your accounts.
|
||||
|
||||
This is a deliberate design decision for a single-operator tool on a trusted
|
||||
network, not a bug and not an oversight. It is stated here because it decides
|
||||
how you are allowed to deploy it. [SECURITY.md](SECURITY.md) covers the rest of
|
||||
the threat model.
|
||||
|
||||
## Requirements
|
||||
|
||||
- **Docker** with Compose v2.
|
||||
- **~4 GB RAM** for the app, plus whatever Postgres needs for your library size.
|
||||
- **Disk** for your media, plus several GB for ML model weights.
|
||||
- **No GPU required.** The ML worker runs on CPU — tagging and embedding are
|
||||
slower, and that is the whole difference. A GPU is only involved if you
|
||||
separately run the optional agent (below), which is a different machine's job.
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
git clone https://git.fabledsword.com/bvandeusen/FabledCurator.git
|
||||
cd FabledCurator
|
||||
|
||||
cp .env.example .env
|
||||
$EDITOR .env # set DB_PASSWORD and SECRET_KEY
|
||||
|
||||
docker compose -f docker-compose.yml up -d
|
||||
```
|
||||
|
||||
Then open <http://localhost:8080>.
|
||||
|
||||
**The `-f docker-compose.yml` is required, not decoration.** Compose
|
||||
auto-merges `docker-compose.override.yml` when you leave it off, and that
|
||||
override builds the images locally from source — the contributor path, not
|
||||
yours. Naming the file explicitly skips the override and pulls the published
|
||||
`:latest` images, which is the stable channel built from `main`.
|
||||
|
||||
If you forget it, the symptom is a long build instead of a quick pull.
|
||||
|
||||
## First run
|
||||
|
||||
The database schema is created automatically on first start — the web container
|
||||
runs its migrations before serving. Nothing to initialise by hand.
|
||||
|
||||
**One thing does need a deliberate act, and the app will not start without it.**
|
||||
FabledCurator encrypts your stored platform credentials with a key it keeps at
|
||||
`./images/secrets/credential_key.b64`. On a brand-new install that file does not
|
||||
exist, and rather than quietly creating one the app stops:
|
||||
|
||||
```
|
||||
MissingCredentialKey: Fernet key file not found at /images/secrets/credential_key.b64
|
||||
```
|
||||
|
||||
Set `CURATOR_BOOTSTRAP_NEW_KEY=1` in your `.env` for the first `up`, then delete
|
||||
the line once the container is running. `.env.example` ships it with that
|
||||
instruction attached.
|
||||
|
||||
The refusal is deliberate, and worth understanding rather than working around:
|
||||
auto-creating a key is indistinguishable from the disaster case — a restore that
|
||||
brought the database back but lost `./images/secrets` — where it would mint a key
|
||||
that cannot decrypt anything, leaving an instance that looks healthy while every
|
||||
paywalled download fails. Making you say so once, on an empty install, is the
|
||||
price of that not happening silently later.
|
||||
|
||||
**Which means: back up `./images/secrets/` alongside your database.** It is the
|
||||
only thing that can read your stored credentials. A database restored without it
|
||||
needs every credential entered again by hand.
|
||||
|
||||
A few other things are worth knowing about the first few minutes:
|
||||
|
||||
- **The ML worker downloads its model weights on first boot**, several GB from
|
||||
HuggingFace into `./models`. Until that finishes, tagging is queued rather
|
||||
than broken. It is idempotent — a restart resumes rather than refetches.
|
||||
- **The gallery starts empty**, and that is the expected state. Add a creator
|
||||
under **Subscriptions** and it fills as posts come down.
|
||||
- **If you already have a library on disk**, there is no screen that imports
|
||||
it, and there is not going to be one. Folder ingestion had a UI until July
|
||||
2026; it was retired once posts began arriving entirely through
|
||||
subscriptions and the browser extension, and the decision to leave it
|
||||
retired is deliberate — the folder path carries complexity the product does
|
||||
not need in order to do its job. The supported way to fill a new install is
|
||||
to add the creators you follow under **Subscriptions** and let it pull.
|
||||
|
||||
The `/api/import/trigger` endpoint is still wired up for anyone who wants to
|
||||
script a one-off against a folder mounted at `./import`, and its progress
|
||||
shows under **Settings → Activity**. Treat it as an unsupported escape
|
||||
hatch rather than a feature: nothing in the UI drives it and nothing else
|
||||
in this README depends on it.
|
||||
- **To download from a paywalled account**, FabledCurator needs that account's
|
||||
session — see the browser extension below. Without one it can still fetch
|
||||
public posts.
|
||||
- **Check Settings → Overview** to confirm the workers are alive. Every long
|
||||
operation in FabledCurator is a background job, so if the queues are not
|
||||
running, the UI will look like it is ignoring you rather than like it is
|
||||
broken.
|
||||
|
||||
## The browser extension
|
||||
|
||||
A Firefox extension does two jobs: it hands your logged-in platform sessions to
|
||||
FabledCurator so it can download on your behalf, and it adds a creator as a
|
||||
subscription in one click from their page.
|
||||
|
||||
It ships **inside the web image** — there is no add-on store listing to find.
|
||||
Go to **Subscriptions → Settings**, find the *Browser extension* card, and click
|
||||
**Install Firefox extension**. The XPI is Mozilla-signed, so Firefox installs it
|
||||
like any other add-on; the button serves it directly rather than making you
|
||||
download and side-load a file.
|
||||
|
||||
It pairs with your instance using an API key generated automatically on first
|
||||
use. The bar directly under that card shows the key and can rotate it.
|
||||
|
||||
See [extension/README.md](extension/README.md) for what it does in detail.
|
||||
|
||||
## The GPU agent
|
||||
|
||||
Optional, and separate. If you have a desktop with a graphics card, you can run
|
||||
an agent on it that leases ML jobs from FabledCurator over HTTP, does them on
|
||||
the GPU, and hands the results back. It never touches the database or Redis, so
|
||||
it is safe to run somewhere the rest of the stack is not.
|
||||
|
||||
Run it for a burst of tagging, stop it to get your card back. It deploys from
|
||||
`agent/docker-compose.yml`, not the main stack — see
|
||||
[agent/README.md](agent/README.md).
|
||||
|
||||
## Upgrading
|
||||
|
||||
```bash
|
||||
docker compose -f docker-compose.yml pull
|
||||
docker compose -f docker-compose.yml up -d
|
||||
```
|
||||
|
||||
Migrations run automatically on start. Take a database backup first — Settings →
|
||||
Maintenance has one — because the schema moves forward and does not move back.
|
||||
|
||||
## Deployment posture
|
||||
|
||||
FabledCurator is built to run inside a homelab over plain HTTP. It does not
|
||||
generate certificates, redirect to HTTPS, or set HSTS. If you want TLS,
|
||||
terminate it at your reverse proxy. See [Before you expose it](#before-you-expose-it)
|
||||
for why TLS alone is not enough.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
**The UI loads but nothing ever finishes.** The web container is up and the
|
||||
workers are not. `docker compose -f docker-compose.yml ps` — check `worker`,
|
||||
`scheduler` and `ml-worker` are healthy, not restarting.
|
||||
|
||||
**`docker compose up` started building instead of pulling.** You left off
|
||||
`-f docker-compose.yml`, so the dev override took over. See [Install](#install).
|
||||
|
||||
**Downloads fail with an auth error.** The stored session for that platform has
|
||||
expired. Re-capture it with the extension; sessions do not last forever.
|
||||
|
||||
**Which build am I running?** The foot of Settings shows a version and a
|
||||
channel, and `/api/health` returns the same two fields. There are no version
|
||||
tags on the images, so this is the authoritative answer.
|
||||
|
||||
---
|
||||
|
||||
# Developing FabledCurator
|
||||
|
||||
Everything below is about working on FabledCurator rather than running it. If
|
||||
you are installing it, you are done — see [CONTRIBUTING.md](CONTRIBUTING.md) if
|
||||
you want to send a patch.
|
||||
|
||||
## Status and channels
|
||||
|
||||
In production. `main` is continuously deployed — every merge builds and
|
||||
publishes `:latest`, so whatever is on `main` is what is running. Day-to-day
|
||||
work happens on `dev`, which publishes `:dev`.
|
||||
|
||||
For local development, the dev override handles everything:
|
||||
|
||||
```bash
|
||||
docker compose up -d # note: no -f, so the override applies
|
||||
```
|
||||
|
||||
That builds the images from source, turns on DEBUG logging, and exposes
|
||||
Postgres and Redis on the host. No `.env` required.
|
||||
|
||||
## Versions and tags
|
||||
|
||||
Three image tags exist, and no others:
|
||||
|
||||
| Tag | Branch | Meaning |
|
||||
| --- | --- | --- |
|
||||
| `:latest` | `main` | Production. Moves on every merge. |
|
||||
| `:c-<sha>` | `main` | Immutable — the rollback unit, all three images together. |
|
||||
| `:dev` | `dev` | The rolling test channel. Moves on every push. |
|
||||
|
||||
There are deliberately **no version tags**. Nothing pins one, and a per-build
|
||||
name nobody reads is upkeep for a model FC does not run (family rule 145; the
|
||||
reasoning is note #3127 §5). Rolling back is `docker pull …:c-<sha>`.
|
||||
|
||||
Each artifact still has a version, derived rather than chosen: the commit time
|
||||
of the newest change to that artifact's *own* shipped files, as
|
||||
`YYYY.MM.DD.HHMM` UTC (rule 148). Four artifacts, four independent versions —
|
||||
a push touching only `agent/` re-versions the agent and leaves web and ml
|
||||
alone, and CI skips the builds whose content did not move.
|
||||
|
||||
Because no registry name carries it, the running instance's own report is the
|
||||
only answer to "which build is this?". The foot of Settings shows
|
||||
`FabledCurator 2026.08.29.0201 · dev`, and `/api/health` returns the same two
|
||||
fields.
|
||||
|
||||
Release tags are optional bookmarks — FC went twelve weeks without one and
|
||||
nothing was wrong. Pushing `v<version>` publishes a Forgejo release listing the
|
||||
commits since the previous tag; it builds no image.
|
||||
|
||||
## What's in here
|
||||
|
||||
@@ -18,43 +263,16 @@ Five deployable pieces, built by `.forgejo/workflows/build.yml`:
|
||||
| --- | --- | --- | --- |
|
||||
| **Web / workers** | `Dockerfile` | `fabledcurator` | Quart API + the built Vue SPA in one image. `entrypoint.sh` picks the role: `web`, `worker`, `scheduler`. The `maintenance-long` service is a second `worker` pinned to the long-running maintenance queue. |
|
||||
| **ML worker** | `Dockerfile.ml` | `fabledcurator-ml` | Same app, plus `requirements-ml.txt` — tagging and embedding models that run in-container. |
|
||||
| **GPU agent** | `agent/Dockerfile` | `fabledcurator-agent` | Optional desktop-GPU worker (`agent/`). Leases jobs over **HTTP only** — never touches the database or Redis. Run it for a burst, stop it to reclaim the card. See `agent/README.md`. |
|
||||
| **GPU agent** | `agent/Dockerfile` | `fabledcurator-agent` | Optional desktop-GPU worker (`agent/`). Leases jobs over **HTTP only** — never touches the database or Redis. See `agent/README.md`. |
|
||||
| **Firefox extension** | `extension/` | signed XPI | MV3 extension: pushes platform session cookies into FC and adds a creator as a Source in one click. AMO-signed on both `dev` and `main` (one signature per extension change, shared by the two channels), bundled into that channel's web image and served from Settings → Maintenance. See `extension/README.md`. |
|
||||
| **Data** | — | `pgvector/pgvector:pg16`, `redis:7-alpine` | Postgres with pgvector for embeddings; Redis as the Celery broker. |
|
||||
|
||||
## Quick start
|
||||
|
||||
For local development and testing, just:
|
||||
|
||||
```bash
|
||||
docker compose up -d
|
||||
# UI: http://localhost:8080
|
||||
```
|
||||
|
||||
That uses sane dev defaults baked into `docker-compose.yml` and the dev
|
||||
override (`docker-compose.override.yml`, auto-merged) — local builds, DEBUG
|
||||
logging, exposed Postgres + Redis ports on the host. No `.env` required.
|
||||
|
||||
For a production-like deployment, override the dev defaults via shell env
|
||||
or a `.env` file (see `.env.example` for the variable names) and use:
|
||||
|
||||
```bash
|
||||
docker compose -f docker-compose.yml up -d
|
||||
# (skips the override so containers pull registry images)
|
||||
```
|
||||
|
||||
The GPU agent is deployed separately, on the machine with the card —
|
||||
`agent/docker-compose.yml`, not this stack.
|
||||
|
||||
## Deployment posture
|
||||
|
||||
FabledCurator is designed to run inside a self-hosted homelab environment over plain HTTP. If you want TLS, terminate it at your reverse proxy. The app does not generate certificates, redirect to HTTPS, or set HSTS.
|
||||
|
||||
## CI / Forgejo setup
|
||||
|
||||
Three workflows: `ci.yml` (lint, extension-version check, backend unit tests,
|
||||
Four workflows: `ci.yml` (lint, extension-version check, backend unit tests,
|
||||
frontend build, integration), `extension.yml` (extension lint, vitest, XPI
|
||||
content verification), and `build.yml` (sign + publish).
|
||||
content verification), `build.yml` (sign + publish), and `release.yml`, which
|
||||
runs only on a `v*` tag and publishes a changelog without building anything.
|
||||
|
||||
**The toolchain each job runs in is its `container.image`, not its `runs-on`
|
||||
label.** `runs-on: python-ci` only schedules the job onto a runner; every job
|
||||
@@ -71,9 +289,36 @@ The repo expects one secret:
|
||||
|
||||
Generate at https://git.fabledsword.com/user/settings/applications. The injected `GITHUB_TOKEN` cannot be used because it lacks `write:package`.
|
||||
|
||||
AMO signing additionally needs `MOZILLA_AMO_JWT_KEY` / `MOZILLA_AMO_JWT_SECRET`; it runs on
|
||||
`main` only and is cached per version, since AMO rejects a re-signed version.
|
||||
AMO signing additionally needs `MOZILLA_AMO_JWT_KEY` / `MOZILLA_AMO_JWT_SECRET`.
|
||||
It runs on **both** channels and is cached per version: because the version is
|
||||
derived from commit time, `dev` and `main` derive the same number for the same
|
||||
source, so `main` finds `dev`'s signature already cached and makes no second AMO
|
||||
call. That cache is why signing must be one-shot — AMO rejects a re-signed
|
||||
version.
|
||||
|
||||
## History
|
||||
|
||||
FabledCurator combines what was
|
||||
[ImageRepo](https://git.fabledsword.com/bvandeusen/ImageRepo) (gallery, ML,
|
||||
importer) and
|
||||
[GallerySubscriber](https://git.fabledsword.com/bvandeusen/GallerySubscriber)
|
||||
(gallery-dl wrapper, subscriptions, credential capture) into a single product.
|
||||
Both are superseded; neither is maintained.
|
||||
|
||||
## License
|
||||
|
||||
Personal project; use at your own discretion.
|
||||
**GNU Affero General Public License v3.0** — see [LICENSE](LICENSE).
|
||||
|
||||
You may run, study, modify and redistribute this software. The condition is
|
||||
reciprocity: if you distribute a modified version, or **run one as a network
|
||||
service that other people use**, you must offer those users the corresponding
|
||||
source under the same licence. That second clause (AGPL §13) is the reason this
|
||||
licence rather than the GPL — for a self-hosted web application, "distribution"
|
||||
otherwise never happens, and the obligation would never bite.
|
||||
|
||||
Running an unmodified copy for yourself, your household or your organisation
|
||||
carries no obligation at all. Neither does modifying it privately. The licence
|
||||
asks something of you only when you hand your modified version to others.
|
||||
|
||||
Contributions ship under the same licence — see [CONTRIBUTING](CONTRIBUTING.md).
|
||||
Security reports: [SECURITY.md](SECURITY.md).
|
||||
|
||||
+77
@@ -0,0 +1,77 @@
|
||||
# Security Policy
|
||||
|
||||
## Reporting a vulnerability
|
||||
|
||||
**Please do not put vulnerability details in a public issue.**
|
||||
|
||||
This project has no private disclosure channel yet. Until it does, open an
|
||||
issue on the repository that says only that you have a security report — no
|
||||
reproduction steps, no affected endpoint, no payload — and a maintainer will
|
||||
reply with a private contact to send the details to.
|
||||
|
||||
That is a deliberately awkward first step, and it exists because the
|
||||
alternative is worse: an issue tracker is public the moment it is written to,
|
||||
and every self-hosted instance stays vulnerable until its operator has had a
|
||||
chance to update.
|
||||
|
||||
Please include, once you have a private channel:
|
||||
|
||||
- what an attacker can do, and what access they need to start
|
||||
- the version or commit you tested
|
||||
- reproduction steps
|
||||
|
||||
## Scope — what this software actually handles
|
||||
|
||||
FabledCurator is self-hosted and holds things worth stating plainly, because
|
||||
they shape what counts as a serious bug here:
|
||||
|
||||
- **Platform credentials.** The app captures and stores session cookies for
|
||||
third-party subscription sites (Patreon, SubscribeStar, Pixiv) so it can
|
||||
download on the operator's behalf. These are live credentials for accounts
|
||||
that usually carry a payment method. Anything that discloses them, decrypts
|
||||
them, or lets one user of a shared instance read another's is high severity.
|
||||
- **An extension API key.** The Firefox extension authenticates to the backend
|
||||
with a shared key. Anything that leaks it or lets it be bypassed is a way in.
|
||||
- **No authentication of its own.** This is the most important thing on this
|
||||
page. FabledCurator has no login, no user accounts and no permission model —
|
||||
there is no `User` table and no session auth anywhere in the backend. Every
|
||||
HTTP client that can reach the port is the administrator, with full read and
|
||||
write access to everything above, including the stored platform credentials.
|
||||
Access control is entirely the operator's job, done at the network layer.
|
||||
Reports that an unauthenticated caller can reach an endpoint are therefore
|
||||
describing the design; reports that something *crosses the network boundary
|
||||
the operator drew* — an SSRF, a request forgery that rides a browser the
|
||||
operator already has open, a path that leaks state to an origin the operator
|
||||
did not authorise — are in scope and are serious.
|
||||
- **Arbitrary media from the internet.** Downloaded files are decoded, hashed,
|
||||
thumbnailed and fed to ML models. Anything that turns a hostile file into
|
||||
code execution is in scope.
|
||||
|
||||
## Deployment posture — read this before reporting
|
||||
|
||||
FabledCurator is designed to run **inside a private network, over plain HTTP,
|
||||
reachable only by its operator**. It does not terminate TLS, redirect to
|
||||
HTTPS, or set HSTS; if you want transport security, terminate it at your
|
||||
reverse proxy. It also does not authenticate anyone — see above. These are
|
||||
documented design decisions, not oversights.
|
||||
|
||||
Putting this on the public internet, with or without TLS, hands whoever finds
|
||||
it your Patreon, SubscribeStar and Pixiv sessions. A reverse proxy that adds
|
||||
TLS but not an authentication layer does not change that.
|
||||
|
||||
Reports that reduce to "the application is served over HTTP", "there is no
|
||||
HSTS header", or "the API needs no credentials" describe those decisions
|
||||
rather than vulnerabilities. Reports that the operator can cause the software
|
||||
to do something destructive are usually also by design — the operator is the
|
||||
administrator of their own instance.
|
||||
|
||||
What remains in scope is everything that crosses a boundary the software is
|
||||
actually supposed to hold: between untrusted downloaded content and the host,
|
||||
between a third-party origin and an operator's open browser session, and
|
||||
between the credentials at rest and anything that is not the operator.
|
||||
|
||||
## Supported versions
|
||||
|
||||
Fixes land on the `main` branch and reach the `:latest` image. There are no
|
||||
maintained release branches — the supported version is the current one, and
|
||||
the remedy for a security issue is to update.
|
||||
@@ -1,277 +0,0 @@
|
||||
"""initial unified schema
|
||||
|
||||
Revision ID: 0001
|
||||
Revises:
|
||||
Create Date: 2026-05-13
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from pgvector.sqlalchemy import Vector
|
||||
|
||||
revision: str = "0001"
|
||||
down_revision: Union[str, None] = None
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.execute("CREATE EXTENSION IF NOT EXISTS vector")
|
||||
|
||||
op.create_table(
|
||||
"artist",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("name", sa.String(length=255), nullable=False),
|
||||
sa.Column("slug", sa.String(length=255), nullable=False),
|
||||
sa.Column("notes", sa.Text(), nullable=True),
|
||||
sa.Column("is_subscription", sa.Boolean(), nullable=False, server_default=sa.false()),
|
||||
sa.Column("auto_check", sa.Boolean(), nullable=False, server_default=sa.true()),
|
||||
sa.Column("check_interval_seconds", sa.Integer(), nullable=True),
|
||||
sa.Column(
|
||||
"created_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name="pk_artist"),
|
||||
sa.UniqueConstraint("name", name="uq_artist_name"),
|
||||
sa.UniqueConstraint("slug", name="uq_artist_slug"),
|
||||
)
|
||||
|
||||
op.create_table(
|
||||
"source",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("artist_id", sa.Integer(), nullable=False),
|
||||
sa.Column("platform", sa.String(length=64), nullable=False),
|
||||
sa.Column("url", sa.Text(), nullable=False),
|
||||
sa.Column("enabled", sa.Boolean(), nullable=False, server_default=sa.true()),
|
||||
sa.Column("config_overrides", sa.JSON(), nullable=True),
|
||||
sa.Column("last_checked_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column("last_error", sa.Text(), nullable=True),
|
||||
sa.Column("check_interval_override", sa.Integer(), nullable=True),
|
||||
sa.ForeignKeyConstraint(
|
||||
["artist_id"], ["artist.id"], name="fk_source_artist_id_artist", ondelete="CASCADE"
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name="pk_source"),
|
||||
)
|
||||
op.create_index("ix_source_artist_id", "source", ["artist_id"])
|
||||
|
||||
op.create_table(
|
||||
"credential",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("platform", sa.String(length=64), nullable=False),
|
||||
sa.Column("kind", sa.String(length=32), nullable=False),
|
||||
sa.Column("encrypted_blob", sa.LargeBinary(), nullable=False),
|
||||
sa.Column("status", sa.String(length=32), nullable=False, server_default="active"),
|
||||
sa.Column(
|
||||
"captured_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column("expires_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.PrimaryKeyConstraint("id", name="pk_credential"),
|
||||
sa.UniqueConstraint("platform", name="uq_credential_platform"),
|
||||
)
|
||||
|
||||
op.create_table(
|
||||
"post",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("source_id", sa.Integer(), nullable=False),
|
||||
sa.Column("external_post_id", sa.String(length=128), nullable=False),
|
||||
sa.Column("post_url", sa.Text(), nullable=True),
|
||||
sa.Column("post_title", sa.Text(), nullable=True),
|
||||
sa.Column("post_date", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column("raw_metadata", sa.JSON(), nullable=True),
|
||||
sa.Column(
|
||||
"downloaded_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["source_id"], ["source.id"], name="fk_post_source_id_source", ondelete="CASCADE"
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name="pk_post"),
|
||||
sa.UniqueConstraint("source_id", "external_post_id", name="uq_post_source_external_id"),
|
||||
)
|
||||
op.create_index("ix_post_source_id", "post", ["source_id"])
|
||||
|
||||
op.create_table(
|
||||
"image_record",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("path", sa.Text(), nullable=False),
|
||||
sa.Column("sha256", sa.String(length=64), nullable=False),
|
||||
sa.Column("phash", sa.String(length=32), nullable=True),
|
||||
sa.Column("size_bytes", sa.BigInteger(), nullable=False),
|
||||
sa.Column("mime", sa.String(length=64), nullable=False),
|
||||
sa.Column("width", sa.Integer(), nullable=True),
|
||||
sa.Column("height", sa.Integer(), nullable=True),
|
||||
sa.Column("thumbnail_path", sa.Text(), nullable=True),
|
||||
sa.Column(
|
||||
"origin",
|
||||
sa.Enum(
|
||||
"downloaded",
|
||||
"imported_filesystem",
|
||||
"uploaded",
|
||||
name="origin_enum",
|
||||
),
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column("primary_post_id", sa.Integer(), nullable=True),
|
||||
sa.Column("wd14_predictions", sa.JSON(), nullable=True),
|
||||
sa.Column("wd14_model_version", sa.String(length=128), nullable=True),
|
||||
sa.Column("siglip_embedding", Vector(1152), nullable=True),
|
||||
sa.Column("siglip_model_version", sa.String(length=128), nullable=True),
|
||||
sa.Column("centroid_scores", sa.JSON(), nullable=True),
|
||||
sa.Column(
|
||||
"created_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column(
|
||||
"updated_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["primary_post_id"],
|
||||
["post.id"],
|
||||
name="fk_image_record_primary_post_id_post",
|
||||
ondelete="SET NULL",
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name="pk_image_record"),
|
||||
sa.UniqueConstraint("path", name="uq_image_record_path"),
|
||||
sa.UniqueConstraint("sha256", name="uq_image_record_sha256"),
|
||||
)
|
||||
op.create_index("ix_image_record_sha256", "image_record", ["sha256"])
|
||||
op.create_index("ix_image_record_phash", "image_record", ["phash"])
|
||||
op.create_index("ix_image_record_primary_post_id", "image_record", ["primary_post_id"])
|
||||
|
||||
op.create_table(
|
||||
"image_provenance",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("image_record_id", sa.Integer(), nullable=False),
|
||||
sa.Column("post_id", sa.Integer(), nullable=False),
|
||||
sa.Column("source_id", sa.Integer(), nullable=False),
|
||||
sa.Column("captured_metadata", sa.JSON(), nullable=True),
|
||||
sa.Column(
|
||||
"captured_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["image_record_id"],
|
||||
["image_record.id"],
|
||||
name="fk_image_provenance_image_record_id_image_record",
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["post_id"],
|
||||
["post.id"],
|
||||
name="fk_image_provenance_post_id_post",
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["source_id"],
|
||||
["source.id"],
|
||||
name="fk_image_provenance_source_id_source",
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name="pk_image_provenance"),
|
||||
)
|
||||
op.create_index("ix_image_provenance_image_record_id", "image_provenance", ["image_record_id"])
|
||||
op.create_index("ix_image_provenance_post_id", "image_provenance", ["post_id"])
|
||||
op.create_index("ix_image_provenance_source_id", "image_provenance", ["source_id"])
|
||||
|
||||
op.create_table(
|
||||
"tag",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("name", sa.String(length=255), nullable=False),
|
||||
sa.Column("namespace", sa.String(length=64), nullable=True),
|
||||
sa.Column(
|
||||
"created_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name="pk_tag"),
|
||||
sa.UniqueConstraint("name", name="uq_tag_name"),
|
||||
)
|
||||
op.create_index("ix_tag_name", "tag", ["name"])
|
||||
op.create_index("ix_tag_namespace", "tag", ["namespace"])
|
||||
|
||||
op.create_table(
|
||||
"image_tag",
|
||||
sa.Column("image_record_id", sa.Integer(), nullable=False),
|
||||
sa.Column("tag_id", sa.Integer(), nullable=False),
|
||||
sa.Column("source", sa.String(length=32), nullable=False, server_default="manual"),
|
||||
sa.Column(
|
||||
"created_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["image_record_id"],
|
||||
["image_record.id"],
|
||||
name="fk_image_tag_image_record_id_image_record",
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["tag_id"], ["tag.id"], name="fk_image_tag_tag_id_tag", ondelete="CASCADE"
|
||||
),
|
||||
sa.PrimaryKeyConstraint("image_record_id", "tag_id", name="pk_image_tag"),
|
||||
)
|
||||
|
||||
op.create_table(
|
||||
"download_event",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("source_id", sa.Integer(), nullable=False),
|
||||
sa.Column("post_id", sa.Integer(), nullable=True),
|
||||
sa.Column("status", sa.String(length=32), nullable=False),
|
||||
sa.Column(
|
||||
"started_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column("bytes_downloaded", sa.BigInteger(), nullable=False, server_default="0"),
|
||||
sa.Column("files_count", sa.Integer(), nullable=False, server_default="0"),
|
||||
sa.Column("error", sa.Text(), nullable=True),
|
||||
sa.ForeignKeyConstraint(
|
||||
["source_id"],
|
||||
["source.id"],
|
||||
name="fk_download_event_source_id_source",
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["post_id"],
|
||||
["post.id"],
|
||||
name="fk_download_event_post_id_post",
|
||||
ondelete="SET NULL",
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name="pk_download_event"),
|
||||
)
|
||||
op.create_index("ix_download_event_source_id", "download_event", ["source_id"])
|
||||
op.create_index("ix_download_event_post_id", "download_event", ["post_id"])
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_table("download_event")
|
||||
op.drop_table("image_tag")
|
||||
op.drop_table("tag")
|
||||
op.drop_table("image_provenance")
|
||||
op.drop_table("image_record")
|
||||
op.execute("DROP TYPE IF EXISTS origin_enum")
|
||||
op.drop_table("post")
|
||||
op.drop_table("credential")
|
||||
op.drop_table("source")
|
||||
op.drop_table("artist")
|
||||
op.execute("DROP EXTENSION IF EXISTS vector")
|
||||
@@ -1,208 +0,0 @@
|
||||
"""fc2a: tag kinds, import_task, import_batch, integrity_status
|
||||
|
||||
Revision ID: 0002
|
||||
Revises: 0001
|
||||
Create Date: 2026-05-14
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0002"
|
||||
down_revision: Union[str, None] = "0001"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
TAG_KINDS = (
|
||||
"artist",
|
||||
"character",
|
||||
"fandom",
|
||||
"general",
|
||||
"series",
|
||||
"archive",
|
||||
"post",
|
||||
"meta",
|
||||
"rating",
|
||||
)
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
# --- Tag kind enum + fandom_id ---
|
||||
tag_kind = sa.Enum(*TAG_KINDS, name="tag_kind")
|
||||
tag_kind.create(op.get_bind(), checkfirst=True)
|
||||
|
||||
op.add_column(
|
||||
"tag",
|
||||
sa.Column("kind", tag_kind, nullable=False, server_default="general"),
|
||||
)
|
||||
op.add_column(
|
||||
"tag",
|
||||
sa.Column("fandom_id", sa.Integer(), nullable=True),
|
||||
)
|
||||
op.create_foreign_key(
|
||||
"fk_tag_fandom_id_tag",
|
||||
"tag",
|
||||
"tag",
|
||||
["fandom_id"],
|
||||
["id"],
|
||||
ondelete="SET NULL",
|
||||
)
|
||||
|
||||
# Drop the old global uniqueness on name; add kind+fandom-aware uniqueness.
|
||||
op.drop_constraint("uq_tag_name", "tag", type_="unique")
|
||||
op.drop_index("ix_tag_name", table_name="tag")
|
||||
op.execute(
|
||||
"""
|
||||
CREATE UNIQUE INDEX uq_tag_name_kind_fandom
|
||||
ON tag (name, kind, COALESCE(fandom_id, 0))
|
||||
"""
|
||||
)
|
||||
|
||||
# CHECK: fandom_id is only allowed for character kind.
|
||||
op.create_check_constraint(
|
||||
"ck_tag_fandom_requires_character",
|
||||
"tag",
|
||||
"(fandom_id IS NULL) OR (kind = 'character')",
|
||||
)
|
||||
|
||||
# Drop the old namespace column — superseded by kind.
|
||||
op.drop_index("ix_tag_namespace", table_name="tag")
|
||||
op.drop_column("tag", "namespace")
|
||||
|
||||
# --- ImportBatch ---
|
||||
op.create_table(
|
||||
"import_batch",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("triggered_by", sa.String(length=32), nullable=False),
|
||||
sa.Column("source_path", sa.Text(), nullable=False),
|
||||
sa.Column("scan_mode", sa.String(length=16), nullable=False),
|
||||
sa.Column(
|
||||
"started_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column("total_files", sa.Integer(), nullable=False, server_default="0"),
|
||||
sa.Column("imported", sa.Integer(), nullable=False, server_default="0"),
|
||||
sa.Column("skipped", sa.Integer(), nullable=False, server_default="0"),
|
||||
sa.Column("failed", sa.Integer(), nullable=False, server_default="0"),
|
||||
sa.Column("status", sa.String(length=16), nullable=False, server_default="running"),
|
||||
sa.PrimaryKeyConstraint("id", name="pk_import_batch"),
|
||||
)
|
||||
op.create_index("ix_import_batch_status", "import_batch", ["status"])
|
||||
|
||||
# --- ImportTask ---
|
||||
op.create_table(
|
||||
"import_task",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("batch_id", sa.Integer(), nullable=False),
|
||||
sa.Column("source_path", sa.Text(), nullable=False),
|
||||
sa.Column("task_type", sa.String(length=16), nullable=False),
|
||||
sa.Column("status", sa.String(length=16), nullable=False, server_default="pending"),
|
||||
sa.Column("result_image_id", sa.Integer(), nullable=True),
|
||||
sa.Column("error", sa.Text(), nullable=True),
|
||||
sa.Column("size_bytes", sa.BigInteger(), nullable=True),
|
||||
sa.Column(
|
||||
"created_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column("started_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.ForeignKeyConstraint(
|
||||
["batch_id"],
|
||||
["import_batch.id"],
|
||||
name="fk_import_task_batch_id_import_batch",
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["result_image_id"],
|
||||
["image_record.id"],
|
||||
name="fk_import_task_result_image_id_image_record",
|
||||
ondelete="SET NULL",
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name="pk_import_task"),
|
||||
)
|
||||
op.create_index("ix_import_task_batch_id", "import_task", ["batch_id"])
|
||||
op.create_index("ix_import_task_status", "import_task", ["status"])
|
||||
op.create_index(
|
||||
"ix_import_task_created_at_desc",
|
||||
"import_task",
|
||||
[sa.text("created_at DESC")],
|
||||
)
|
||||
|
||||
# --- ImportSettings (single-row table) ---
|
||||
op.create_table(
|
||||
"import_settings",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("import_scan_path", sa.Text(), nullable=False, server_default="/import"),
|
||||
sa.Column("min_width", sa.Integer(), nullable=False, server_default="0"),
|
||||
sa.Column("min_height", sa.Integer(), nullable=False, server_default="0"),
|
||||
sa.Column(
|
||||
"skip_transparent", sa.Boolean(), nullable=False, server_default=sa.false()
|
||||
),
|
||||
sa.Column(
|
||||
"transparency_threshold",
|
||||
sa.Float(),
|
||||
nullable=False,
|
||||
server_default="0.9",
|
||||
),
|
||||
sa.Column(
|
||||
"skip_single_color", sa.Boolean(), nullable=False, server_default=sa.false()
|
||||
),
|
||||
sa.Column(
|
||||
"single_color_threshold",
|
||||
sa.Float(),
|
||||
nullable=False,
|
||||
server_default="0.95",
|
||||
),
|
||||
sa.Column("single_color_tolerance", sa.Integer(), nullable=False, server_default="30"),
|
||||
sa.PrimaryKeyConstraint("id", name="pk_import_settings"),
|
||||
sa.CheckConstraint("id = 1", name="ck_import_settings_singleton"),
|
||||
)
|
||||
# Seed the single row immediately so callers can always SELECT id=1.
|
||||
op.execute("INSERT INTO import_settings (id) VALUES (1)")
|
||||
|
||||
# --- ImageRecord additions ---
|
||||
op.add_column(
|
||||
"image_record",
|
||||
sa.Column(
|
||||
"integrity_status",
|
||||
sa.String(length=24),
|
||||
nullable=False,
|
||||
server_default="unknown",
|
||||
),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_image_record_integrity_status",
|
||||
"image_record",
|
||||
["integrity_status"],
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_image_record_integrity_status", table_name="image_record")
|
||||
op.drop_column("image_record", "integrity_status")
|
||||
|
||||
op.drop_table("import_settings")
|
||||
op.drop_index("ix_import_task_created_at_desc", table_name="import_task")
|
||||
op.drop_index("ix_import_task_status", table_name="import_task")
|
||||
op.drop_index("ix_import_task_batch_id", table_name="import_task")
|
||||
op.drop_table("import_task")
|
||||
op.drop_index("ix_import_batch_status", table_name="import_batch")
|
||||
op.drop_table("import_batch")
|
||||
|
||||
op.drop_constraint("ck_tag_fandom_requires_character", "tag", type_="check")
|
||||
op.execute("DROP INDEX uq_tag_name_kind_fandom")
|
||||
op.add_column("tag", sa.Column("namespace", sa.String(length=64), nullable=True))
|
||||
op.create_index("ix_tag_namespace", "tag", ["namespace"])
|
||||
op.create_index("ix_tag_name", "tag", ["name"], unique=False)
|
||||
op.create_unique_constraint("uq_tag_name", "tag", ["name"])
|
||||
op.drop_constraint("fk_tag_fandom_id_tag", "tag", type_="foreignkey")
|
||||
op.drop_column("tag", "fandom_id")
|
||||
op.drop_column("tag", "kind")
|
||||
sa.Enum(name="tag_kind").drop(op.get_bind(), checkfirst=True)
|
||||
@@ -1,172 +0,0 @@
|
||||
"""fc2b: ML pipeline — allowlist, aliases, centroids, ml_settings
|
||||
|
||||
Revision ID: 0003
|
||||
Revises: 0002
|
||||
Create Date: 2026-05-15
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from pgvector.sqlalchemy import Vector
|
||||
|
||||
revision: str = "0003"
|
||||
down_revision: Union[str, None] = "0002"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
# 3.1 rename wd14_* -> tagger_*
|
||||
op.alter_column("image_record", "wd14_predictions", new_column_name="tagger_predictions")
|
||||
op.alter_column(
|
||||
"image_record", "wd14_model_version", new_column_name="tagger_model_version"
|
||||
)
|
||||
|
||||
# 3.2 tag_allowlist
|
||||
op.create_table(
|
||||
"tag_allowlist",
|
||||
sa.Column("tag_id", sa.Integer(), nullable=False),
|
||||
sa.Column(
|
||||
"min_confidence", sa.Float(), nullable=False, server_default="0.95"
|
||||
),
|
||||
sa.Column(
|
||||
"added_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["tag_id"], ["tag.id"], name="fk_tag_allowlist_tag_id_tag",
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.PrimaryKeyConstraint("tag_id", name="pk_tag_allowlist"),
|
||||
sa.CheckConstraint(
|
||||
"min_confidence > 0 AND min_confidence <= 1",
|
||||
name="ck_tag_allowlist_confidence_range",
|
||||
),
|
||||
)
|
||||
|
||||
# 3.3 tag_suggestion_rejection
|
||||
op.create_table(
|
||||
"tag_suggestion_rejection",
|
||||
sa.Column("image_record_id", sa.Integer(), nullable=False),
|
||||
sa.Column("tag_id", sa.Integer(), nullable=False),
|
||||
sa.Column(
|
||||
"rejected_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["image_record_id"], ["image_record.id"],
|
||||
name="fk_tsr_image_record_id_image_record", ondelete="CASCADE",
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["tag_id"], ["tag.id"], name="fk_tsr_tag_id_tag", ondelete="CASCADE",
|
||||
),
|
||||
sa.PrimaryKeyConstraint(
|
||||
"image_record_id", "tag_id", name="pk_tag_suggestion_rejection"
|
||||
),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_tag_suggestion_rejection_tag", "tag_suggestion_rejection", ["tag_id"]
|
||||
)
|
||||
|
||||
# 3.4 tag_alias
|
||||
op.create_table(
|
||||
"tag_alias",
|
||||
sa.Column("alias_string", sa.String(length=255), nullable=False),
|
||||
sa.Column("alias_category", sa.String(length=32), nullable=False),
|
||||
sa.Column("canonical_tag_id", sa.Integer(), nullable=False),
|
||||
sa.Column(
|
||||
"created_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["canonical_tag_id"], ["tag.id"],
|
||||
name="fk_tag_alias_canonical_tag_id_tag", ondelete="CASCADE",
|
||||
),
|
||||
sa.PrimaryKeyConstraint(
|
||||
"alias_string", "alias_category", name="pk_tag_alias"
|
||||
),
|
||||
)
|
||||
op.create_index("ix_tag_alias_canonical", "tag_alias", ["canonical_tag_id"])
|
||||
|
||||
# 3.5 tag_reference_embedding (centroids)
|
||||
op.create_table(
|
||||
"tag_reference_embedding",
|
||||
sa.Column("tag_id", sa.Integer(), nullable=False),
|
||||
sa.Column("embedding", Vector(1152), nullable=False),
|
||||
sa.Column("reference_count", sa.Integer(), nullable=False),
|
||||
sa.Column("model_version", sa.String(length=128), nullable=False),
|
||||
sa.Column(
|
||||
"updated_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["tag_id"], ["tag.id"],
|
||||
name="fk_tag_reference_embedding_tag_id_tag", ondelete="CASCADE",
|
||||
),
|
||||
sa.PrimaryKeyConstraint("tag_id", name="pk_tag_reference_embedding"),
|
||||
)
|
||||
|
||||
# 3.6 ml_settings singleton
|
||||
op.create_table(
|
||||
"ml_settings",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column(
|
||||
"suggestion_threshold_artist", sa.Float(), nullable=False,
|
||||
server_default="0.30",
|
||||
),
|
||||
sa.Column(
|
||||
"suggestion_threshold_character", sa.Float(), nullable=False,
|
||||
server_default="0.50",
|
||||
),
|
||||
sa.Column(
|
||||
"suggestion_threshold_copyright", sa.Float(), nullable=False,
|
||||
server_default="0.50",
|
||||
),
|
||||
sa.Column(
|
||||
"suggestion_threshold_general", sa.Float(), nullable=False,
|
||||
server_default="0.95",
|
||||
),
|
||||
sa.Column(
|
||||
"centroid_similarity_threshold", sa.Float(), nullable=False,
|
||||
server_default="0.55",
|
||||
),
|
||||
sa.Column(
|
||||
"min_reference_images", sa.Integer(), nullable=False,
|
||||
server_default="5",
|
||||
),
|
||||
sa.Column(
|
||||
"tagger_model_version", sa.String(length=128), nullable=False,
|
||||
server_default="camie-tagger-v2",
|
||||
),
|
||||
sa.Column(
|
||||
"embedder_model_version", sa.String(length=128), nullable=False,
|
||||
server_default="siglip-so400m-patch14-384",
|
||||
),
|
||||
sa.Column(
|
||||
"updated_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name="pk_ml_settings"),
|
||||
sa.CheckConstraint("id = 1", name="ck_ml_settings_singleton"),
|
||||
)
|
||||
op.execute("INSERT INTO ml_settings (id) VALUES (1)")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_table("ml_settings")
|
||||
op.drop_table("tag_reference_embedding")
|
||||
op.drop_index("ix_tag_alias_canonical", table_name="tag_alias")
|
||||
op.drop_table("tag_alias")
|
||||
op.drop_index(
|
||||
"ix_tag_suggestion_rejection_tag", table_name="tag_suggestion_rejection"
|
||||
)
|
||||
op.drop_table("tag_suggestion_rejection")
|
||||
op.drop_table("tag_allowlist")
|
||||
op.alter_column(
|
||||
"image_record", "tagger_model_version", new_column_name="wd14_model_version"
|
||||
)
|
||||
op.alter_column(
|
||||
"image_record", "tagger_predictions", new_column_name="wd14_predictions"
|
||||
)
|
||||
@@ -1,23 +0,0 @@
|
||||
"""fc2c-i: enable tsm_system_rows for scalable random sampling
|
||||
|
||||
Revision ID: 0004
|
||||
Revises: 0003
|
||||
Create Date: 2026-05-15
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0004"
|
||||
down_revision: Union[str, None] = "0003"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.execute("CREATE EXTENSION IF NOT EXISTS tsm_system_rows")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.execute("DROP EXTENSION IF EXISTS tsm_system_rows")
|
||||
@@ -1,50 +0,0 @@
|
||||
"""fc2c-iii-a: series_page ordered membership
|
||||
|
||||
Revision ID: 0005
|
||||
Revises: 0004
|
||||
Create Date: 2026-05-16
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0005"
|
||||
down_revision: Union[str, None] = "0004"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"series_page",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("series_tag_id", sa.Integer(), nullable=False),
|
||||
sa.Column("image_id", sa.Integer(), nullable=False),
|
||||
sa.Column("page_number", sa.Integer(), nullable=False),
|
||||
sa.Column(
|
||||
"created_at", sa.DateTime(timezone=True),
|
||||
nullable=False, server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column(
|
||||
"updated_at", sa.DateTime(timezone=True),
|
||||
nullable=False, server_default=sa.func.now(),
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["series_tag_id"], ["tag.id"], ondelete="CASCADE"
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["image_id"], ["image_record.id"], ondelete="CASCADE"
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id"),
|
||||
sa.UniqueConstraint("image_id", name="uq_series_page_image"),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_series_page_series_tag_id", "series_page", ["series_tag_id"]
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_series_page_series_tag_id", table_name="series_page")
|
||||
op.drop_table("series_page")
|
||||
@@ -1,30 +0,0 @@
|
||||
"""fc2d: import_settings.phash_threshold
|
||||
|
||||
Revision ID: 0006
|
||||
Revises: 0005
|
||||
Create Date: 2026-05-17
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0006"
|
||||
down_revision: Union[str, None] = "0005"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"phash_threshold", sa.Integer(),
|
||||
nullable=False, server_default="10",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("import_settings", "phash_threshold")
|
||||
@@ -1,31 +0,0 @@
|
||||
"""fc2d-iv: post.description + post.attachment_count
|
||||
|
||||
Revision ID: 0007
|
||||
Revises: 0006
|
||||
Create Date: 2026-05-18
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0007"
|
||||
down_revision: Union[str, None] = "0006"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"post", sa.Column("description", sa.Text(), nullable=True)
|
||||
)
|
||||
op.add_column(
|
||||
"post",
|
||||
sa.Column("attachment_count", sa.Integer(), nullable=True),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("post", "attachment_count")
|
||||
op.drop_column("post", "description")
|
||||
@@ -1,52 +0,0 @@
|
||||
"""fc2d-vii-c: image_record.artist_id + backfill + drop artist tags
|
||||
|
||||
Revision ID: 0008
|
||||
Revises: 0007
|
||||
Create Date: 2026-05-18
|
||||
|
||||
Internal forward-correctness migration (the big legacy-import migration
|
||||
stays deferred). downgrade() does NOT recreate deleted artist tags;
|
||||
downgrade is dev-only and the data is reconstructable by re-import.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
from backend.app.utils.artist_backfill import (
|
||||
BACKFILL_PRIMARY_SQL,
|
||||
BACKFILL_PROVENANCE_SQL,
|
||||
BACKFILL_TAG_SQL,
|
||||
DELETE_ARTIST_TAGS_SQL,
|
||||
)
|
||||
|
||||
revision: str = "0008"
|
||||
down_revision: Union[str, None] = "0007"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"image_record",
|
||||
sa.Column("artist_id", sa.Integer(), nullable=True),
|
||||
)
|
||||
op.create_foreign_key(
|
||||
"fk_image_record_artist_id", "image_record", "artist",
|
||||
["artist_id"], ["id"], ondelete="SET NULL",
|
||||
)
|
||||
op.create_index(
|
||||
"ix_image_record_artist_id", "image_record", ["artist_id"],
|
||||
)
|
||||
op.execute(BACKFILL_PRIMARY_SQL)
|
||||
op.execute(BACKFILL_PROVENANCE_SQL)
|
||||
op.execute(BACKFILL_TAG_SQL)
|
||||
op.execute(DELETE_ARTIST_TAGS_SQL)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_image_record_artist_id", table_name="image_record")
|
||||
op.drop_constraint(
|
||||
"fk_image_record_artist_id", "image_record", type_="foreignkey"
|
||||
)
|
||||
op.drop_column("image_record", "artist_id")
|
||||
@@ -1,68 +0,0 @@
|
||||
"""fc2d-iii: post_attachment + import_batch.attachments
|
||||
|
||||
Revision ID: 0009
|
||||
Revises: 0008
|
||||
Create Date: 2026-05-19
|
||||
|
||||
Internal forward-correctness migration (big legacy-import migration
|
||||
stays deferred). No backfill — no attachments exist yet.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0009"
|
||||
down_revision: Union[str, None] = "0008"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"post_attachment",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column(
|
||||
"post_id", sa.Integer(),
|
||||
sa.ForeignKey("post.id", ondelete="SET NULL"), nullable=True,
|
||||
),
|
||||
sa.Column(
|
||||
"artist_id", sa.Integer(),
|
||||
sa.ForeignKey("artist.id", ondelete="SET NULL"), nullable=True,
|
||||
),
|
||||
sa.Column("sha256", sa.String(64), nullable=False),
|
||||
sa.Column("path", sa.Text(), nullable=False),
|
||||
sa.Column("original_filename", sa.Text(), nullable=False),
|
||||
sa.Column("ext", sa.String(32), nullable=False),
|
||||
sa.Column("mime", sa.String(128), nullable=True),
|
||||
sa.Column("size_bytes", sa.BigInteger(), nullable=False),
|
||||
sa.Column(
|
||||
"captured_at", sa.DateTime(timezone=True),
|
||||
server_default=sa.func.now(), nullable=False,
|
||||
),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_post_attachment_sha256", "post_attachment", ["sha256"],
|
||||
unique=True,
|
||||
)
|
||||
op.create_index(
|
||||
"ix_post_attachment_post_id", "post_attachment", ["post_id"],
|
||||
)
|
||||
op.create_index(
|
||||
"ix_post_attachment_artist_id", "post_attachment", ["artist_id"],
|
||||
)
|
||||
op.add_column(
|
||||
"import_batch",
|
||||
sa.Column(
|
||||
"attachments", sa.Integer(), nullable=False,
|
||||
server_default="0",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("import_batch", "attachments")
|
||||
op.drop_index("ix_post_attachment_artist_id", table_name="post_attachment")
|
||||
op.drop_index("ix_post_attachment_post_id", table_name="post_attachment")
|
||||
op.drop_index("ix_post_attachment_sha256", table_name="post_attachment")
|
||||
op.drop_table("post_attachment")
|
||||
@@ -1,32 +0,0 @@
|
||||
"""fc3a: unique(source.artist_id, source.platform, source.url)
|
||||
|
||||
Revision ID: 0010
|
||||
Revises: 0009
|
||||
Create Date: 2026-05-20
|
||||
|
||||
Enforces FC-3a's dedup invariant at the DB level. No backfill — no
|
||||
existing rows are expected to collide; if they do the migration will
|
||||
fail loudly (intended).
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0010"
|
||||
down_revision: Union[str, None] = "0009"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_unique_constraint(
|
||||
"uq_source_artist_platform_url",
|
||||
"source",
|
||||
["artist_id", "platform", "url"],
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_constraint(
|
||||
"uq_source_artist_platform_url", "source", type_="unique"
|
||||
)
|
||||
@@ -1,41 +0,0 @@
|
||||
"""fc3b: rename credential.kind -> credential_type, drop status, add last_verified
|
||||
|
||||
Revision ID: 0011
|
||||
Revises: 0010
|
||||
Create Date: 2026-05-20
|
||||
|
||||
Aligns the credential table with the GallerySubscriber wire-field names
|
||||
so the existing browser extension can POST to FC unmodified. Greenfield —
|
||||
no rows exist in production yet, so no data preservation logic is
|
||||
needed; the rename uses ALTER COLUMN rather than copy-then-drop.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0011"
|
||||
down_revision: Union[str, None] = "0010"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.alter_column("credential", "kind", new_column_name="credential_type")
|
||||
op.drop_column("credential", "status")
|
||||
op.add_column(
|
||||
"credential",
|
||||
sa.Column("last_verified", sa.DateTime(timezone=True), nullable=True),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("credential", "last_verified")
|
||||
op.add_column(
|
||||
"credential",
|
||||
sa.Column(
|
||||
"status", sa.String(length=32), nullable=False,
|
||||
server_default="active",
|
||||
),
|
||||
)
|
||||
op.alter_column("credential", "credential_type", new_column_name="kind")
|
||||
@@ -1,36 +0,0 @@
|
||||
"""fc3b: app_setting key/value table
|
||||
|
||||
Revision ID: 0012
|
||||
Revises: 0011
|
||||
Create Date: 2026-05-20
|
||||
|
||||
A simple key/value table for small app settings that don't fit
|
||||
ImportSettings. Initially seeds only `extension_api_key` (done in
|
||||
create_app on first boot — not in the migration, to keep it
|
||||
deterministic and independent of randomness).
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0012"
|
||||
down_revision: Union[str, None] = "0011"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"app_setting",
|
||||
sa.Column("key", sa.String(length=64), primary_key=True),
|
||||
sa.Column("value", sa.Text(), nullable=False),
|
||||
sa.Column(
|
||||
"updated_at", sa.DateTime(timezone=True),
|
||||
nullable=False, server_default=sa.func.now(),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_table("app_setting")
|
||||
@@ -1,52 +0,0 @@
|
||||
"""fc3c: download_event.metadata + import_settings downloader fields
|
||||
|
||||
Revision ID: 0013
|
||||
Revises: 0012
|
||||
Create Date: 2026-05-20
|
||||
|
||||
Additive only. download_event.metadata is the rich JSONB blob FC-3c
|
||||
populates per run (run_stats, stdout/stderr, quarantined paths, import
|
||||
summary). import_settings gains two operator-tunable downloader knobs:
|
||||
download_rate_limit_seconds (gallery-dl extractor.sleep) and
|
||||
download_validate_files (toggle the magic-byte validator).
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from sqlalchemy.dialects import postgresql
|
||||
|
||||
revision: str = "0013"
|
||||
down_revision: Union[str, None] = "0012"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"download_event",
|
||||
sa.Column(
|
||||
"metadata", postgresql.JSONB,
|
||||
nullable=False, server_default=sa.text("'{}'::jsonb"),
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"download_rate_limit_seconds", sa.Float(),
|
||||
nullable=False, server_default="3.0",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"download_validate_files", sa.Boolean(),
|
||||
nullable=False, server_default=sa.true(),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("import_settings", "download_validate_files")
|
||||
op.drop_column("import_settings", "download_rate_limit_seconds")
|
||||
op.drop_column("download_event", "metadata")
|
||||
@@ -1,58 +0,0 @@
|
||||
"""fc3d: scheduling + source health columns
|
||||
|
||||
Revision ID: 0014
|
||||
Revises: 0013
|
||||
Create Date: 2026-05-21
|
||||
|
||||
Additive only. source.consecutive_failures (default 0, DownloadService
|
||||
finalize hook owns the writes). import_settings gains the three
|
||||
scheduling knobs (global default interval, event retention, failure
|
||||
warning threshold).
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0014"
|
||||
down_revision: Union[str, None] = "0013"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"source",
|
||||
sa.Column(
|
||||
"consecutive_failures", sa.Integer(),
|
||||
nullable=False, server_default="0",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"download_schedule_default_seconds", sa.Integer(),
|
||||
nullable=False, server_default="28800",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"download_event_retention_days", sa.Integer(),
|
||||
nullable=False, server_default="90",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"download_failure_warning_threshold", sa.Integer(),
|
||||
nullable=False, server_default="5",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("import_settings", "download_failure_warning_threshold")
|
||||
op.drop_column("import_settings", "download_event_retention_days")
|
||||
op.drop_column("import_settings", "download_schedule_default_seconds")
|
||||
op.drop_column("source", "consecutive_failures")
|
||||
@@ -1,51 +0,0 @@
|
||||
"""fc5: migration_run table
|
||||
|
||||
Revision ID: 0015
|
||||
Revises: 0014
|
||||
Create Date: 2026-05-22
|
||||
|
||||
Additive only. New table tracks each invocation of the FC-5 migration
|
||||
tooling (backup, gs, ir, ml_queue, verify, rollback). kind/status are
|
||||
plain String(32) — values validated at the API layer per the spec, not
|
||||
a Postgres ENUM (so adding kinds later doesn't need a schema migration).
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from sqlalchemy.dialects import postgresql
|
||||
|
||||
revision: str = "0015"
|
||||
down_revision: Union[str, None] = "0014"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"migration_run",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column("kind", sa.String(32), nullable=False, index=True),
|
||||
sa.Column("status", sa.String(32), nullable=False, index=True),
|
||||
sa.Column(
|
||||
"dry_run", sa.Boolean(), nullable=False, server_default=sa.false(),
|
||||
),
|
||||
sa.Column(
|
||||
"started_at", sa.DateTime(timezone=True),
|
||||
nullable=False, server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column(
|
||||
"counts", postgresql.JSONB,
|
||||
nullable=False, server_default=sa.text("'{}'::jsonb"),
|
||||
),
|
||||
sa.Column("error", sa.Text(), nullable=True),
|
||||
sa.Column(
|
||||
"metadata", postgresql.JSONB,
|
||||
nullable=False, server_default=sa.text("'{}'::jsonb"),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_table("migration_run")
|
||||
@@ -1,86 +0,0 @@
|
||||
"""fc3i: task_run table
|
||||
|
||||
Revision ID: 0016
|
||||
Revises: 0015
|
||||
Create Date: 2026-05-24
|
||||
|
||||
Additive only. New table records every Celery task attempt via signal
|
||||
handlers (backend.app.celery_signals). Status is plain String(16) not
|
||||
Postgres ENUM (per feedback_check_existing_enums: ENUM columns hard-
|
||||
fail at INSERT, String columns extend cleanly).
|
||||
|
||||
Composite indexes anticipate the three dashboard panes:
|
||||
- (queue, started_at desc) — per-lane recent activity
|
||||
- (status, started_at desc) — recent failures pane
|
||||
- (task_name, started_at desc) — drill-down by task
|
||||
|
||||
Indexed columns get individual indexes via `index=True` on the model;
|
||||
the composites below cover the multi-column lookups.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0016"
|
||||
down_revision: Union[str, None] = "0015"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"task_run",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column("celery_task_id", sa.String(length=64), nullable=False),
|
||||
sa.Column("queue", sa.String(length=32), nullable=False),
|
||||
sa.Column("task_name", sa.String(length=128), nullable=False),
|
||||
sa.Column("target_id", sa.Integer(), nullable=True),
|
||||
sa.Column("started_at", sa.DateTime(timezone=True), nullable=False),
|
||||
sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column("duration_ms", sa.Integer(), nullable=True),
|
||||
sa.Column(
|
||||
"status", sa.String(length=16), nullable=False,
|
||||
server_default="running",
|
||||
),
|
||||
sa.Column("error_type", sa.String(length=128), nullable=True),
|
||||
sa.Column("error_message", sa.Text(), nullable=True),
|
||||
sa.Column("retry_count", sa.Integer(), nullable=True),
|
||||
sa.Column("worker_hostname", sa.String(length=128), nullable=True),
|
||||
sa.Column("args_summary", sa.String(length=255), nullable=True),
|
||||
)
|
||||
|
||||
# Single-column indexes (matches Mapped[...].index=True on model).
|
||||
op.create_index("ix_task_run_celery_task_id", "task_run", ["celery_task_id"])
|
||||
op.create_index("ix_task_run_queue", "task_run", ["queue"])
|
||||
op.create_index("ix_task_run_task_name", "task_run", ["task_name"])
|
||||
op.create_index("ix_task_run_started_at", "task_run", ["started_at"])
|
||||
op.create_index("ix_task_run_finished_at", "task_run", ["finished_at"])
|
||||
op.create_index("ix_task_run_status", "task_run", ["status"])
|
||||
|
||||
# Composite indexes for dashboard query patterns.
|
||||
op.create_index(
|
||||
"ix_task_run_queue_started",
|
||||
"task_run", ["queue", sa.text("started_at DESC")],
|
||||
)
|
||||
op.create_index(
|
||||
"ix_task_run_status_started",
|
||||
"task_run", ["status", sa.text("started_at DESC")],
|
||||
)
|
||||
op.create_index(
|
||||
"ix_task_run_name_started",
|
||||
"task_run", ["task_name", sa.text("started_at DESC")],
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_task_run_name_started", table_name="task_run")
|
||||
op.drop_index("ix_task_run_status_started", table_name="task_run")
|
||||
op.drop_index("ix_task_run_queue_started", table_name="task_run")
|
||||
op.drop_index("ix_task_run_status", table_name="task_run")
|
||||
op.drop_index("ix_task_run_finished_at", table_name="task_run")
|
||||
op.drop_index("ix_task_run_started_at", table_name="task_run")
|
||||
op.drop_index("ix_task_run_task_name", table_name="task_run")
|
||||
op.drop_index("ix_task_run_queue", table_name="task_run")
|
||||
op.drop_index("ix_task_run_celery_task_id", table_name="task_run")
|
||||
op.drop_table("task_run")
|
||||
@@ -1,82 +0,0 @@
|
||||
"""fc3h: backup_run table
|
||||
|
||||
Revision ID: 0017
|
||||
Revises: 0016
|
||||
Create Date: 2026-05-24
|
||||
|
||||
Additive. New table records every backup/restore attempt with artifact
|
||||
metadata. Lifecycle tracking lives in task_run from FC-3i; this is
|
||||
artifact-only (paths, sizes, tag, restore lineage).
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0017"
|
||||
down_revision: Union[str, None] = "0016"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"backup_run",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column("kind", sa.String(length=16), nullable=False),
|
||||
sa.Column(
|
||||
"status", sa.String(length=16), nullable=False,
|
||||
server_default="pending",
|
||||
),
|
||||
sa.Column("tag", sa.String(length=64), nullable=True),
|
||||
sa.Column("triggered_by", sa.String(length=32), nullable=False),
|
||||
sa.Column("started_at", sa.DateTime(timezone=True), nullable=False),
|
||||
sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column("sql_path", sa.Text(), nullable=True),
|
||||
sa.Column("tar_path", sa.Text(), nullable=True),
|
||||
sa.Column("size_bytes", sa.BigInteger(), nullable=True),
|
||||
sa.Column("error", sa.Text(), nullable=True),
|
||||
sa.Column(
|
||||
"manifest", sa.JSON(), nullable=False, server_default="{}",
|
||||
),
|
||||
sa.Column(
|
||||
"restored_from_id", sa.Integer(),
|
||||
sa.ForeignKey("backup_run.id", ondelete="SET NULL"),
|
||||
nullable=True,
|
||||
),
|
||||
)
|
||||
|
||||
# Single-column indexes (matches Mapped[...].index=True).
|
||||
op.create_index("ix_backup_run_kind", "backup_run", ["kind"])
|
||||
op.create_index("ix_backup_run_status", "backup_run", ["status"])
|
||||
op.create_index("ix_backup_run_tag", "backup_run", ["tag"])
|
||||
op.create_index("ix_backup_run_started_at", "backup_run", ["started_at"])
|
||||
op.create_index("ix_backup_run_finished_at", "backup_run", ["finished_at"])
|
||||
|
||||
# Composite indexes for dashboard query patterns.
|
||||
op.create_index(
|
||||
"ix_backup_run_kind_started",
|
||||
"backup_run", ["kind", sa.text("started_at DESC")],
|
||||
)
|
||||
op.create_index(
|
||||
"ix_backup_run_status_finished",
|
||||
"backup_run", ["status", sa.text("finished_at DESC")],
|
||||
)
|
||||
# Partial index: only tagged rows participate in retention-exempt query.
|
||||
op.create_index(
|
||||
"ix_backup_run_tag_partial",
|
||||
"backup_run", ["tag"],
|
||||
postgresql_where=sa.text("tag IS NOT NULL"),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_backup_run_tag_partial", table_name="backup_run")
|
||||
op.drop_index("ix_backup_run_status_finished", table_name="backup_run")
|
||||
op.drop_index("ix_backup_run_kind_started", table_name="backup_run")
|
||||
op.drop_index("ix_backup_run_finished_at", table_name="backup_run")
|
||||
op.drop_index("ix_backup_run_started_at", table_name="backup_run")
|
||||
op.drop_index("ix_backup_run_tag", table_name="backup_run")
|
||||
op.drop_index("ix_backup_run_status", table_name="backup_run")
|
||||
op.drop_index("ix_backup_run_kind", table_name="backup_run")
|
||||
op.drop_table("backup_run")
|
||||
@@ -1,62 +0,0 @@
|
||||
"""fc3h: backup_* knobs on import_settings
|
||||
|
||||
Revision ID: 0018
|
||||
Revises: 0017
|
||||
Create Date: 2026-05-24
|
||||
|
||||
Adds four columns to the singleton import_settings row:
|
||||
- backup_db_nightly_enabled (default False — opt-in)
|
||||
- backup_db_nightly_hour_utc (default 3)
|
||||
- backup_db_keep_last_n (default 14)
|
||||
- backup_images_keep_last_n (default 3)
|
||||
|
||||
server_default ensures the singleton row is backfilled in place
|
||||
without an UPDATE statement.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0018"
|
||||
down_revision: Union[str, None] = "0017"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"backup_db_nightly_enabled", sa.Boolean(),
|
||||
nullable=False, server_default=sa.false(),
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"backup_db_nightly_hour_utc", sa.Integer(),
|
||||
nullable=False, server_default="3",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"backup_db_keep_last_n", sa.Integer(),
|
||||
nullable=False, server_default="14",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"backup_images_keep_last_n", sa.Integer(),
|
||||
nullable=False, server_default="3",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("import_settings", "backup_images_keep_last_n")
|
||||
op.drop_column("import_settings", "backup_db_keep_last_n")
|
||||
op.drop_column("import_settings", "backup_db_nightly_hour_utc")
|
||||
op.drop_column("import_settings", "backup_db_nightly_enabled")
|
||||
@@ -1,38 +0,0 @@
|
||||
"""import_batch.refreshed counter for deep-scan sidecar re-application
|
||||
|
||||
Revision ID: 0019
|
||||
Revises: 0018
|
||||
Create Date: 2026-05-25
|
||||
|
||||
Adds a `refreshed` counter to `import_batch`, mirroring the existing
|
||||
`imported`/`skipped`/`failed`/`attachments` columns. Deep scan now
|
||||
re-applies sidecar metadata to already-imported files (the IR feature
|
||||
that didn't make the FC port the first time); a "refreshed" outcome
|
||||
increments this counter so the UI can surface "X new, Y refreshed"
|
||||
instead of the misleading "Scan complete — no new files" message.
|
||||
|
||||
server_default=0 backfills existing rows in place — no UPDATE needed.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0019"
|
||||
down_revision: Union[str, None] = "0018"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"import_batch",
|
||||
sa.Column(
|
||||
"refreshed", sa.Integer(),
|
||||
nullable=False, server_default=sa.text("0"),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("import_batch", "refreshed")
|
||||
@@ -1,65 +0,0 @@
|
||||
"""fc-cleanup: library_audit_run table for async transparency/single_color audits
|
||||
|
||||
Revision ID: 0020
|
||||
Revises: 0019
|
||||
Create Date: 2026-05-26
|
||||
|
||||
The table backs the async audit lifecycle: rule + params snapshot, status
|
||||
state machine ('running' → 'ready' → 'applied'/'cancelled'/'error'), and
|
||||
the matched_ids JSONB array that the apply step deletes. Capped at 50k IDs
|
||||
per row by the scan task (oversize = rule too aggressive, operator narrows
|
||||
before re-running).
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from sqlalchemy.dialects import postgresql
|
||||
|
||||
revision: str = "0020"
|
||||
down_revision: Union[str, None] = "0019"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"library_audit_run",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column("rule", sa.String(32), nullable=False),
|
||||
sa.Column("params", postgresql.JSONB(astext_type=sa.Text()), nullable=False),
|
||||
sa.Column(
|
||||
"status", sa.String(16),
|
||||
nullable=False, server_default="running",
|
||||
),
|
||||
sa.Column(
|
||||
"started_at", sa.DateTime(timezone=True),
|
||||
nullable=False, server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column(
|
||||
"scanned_count", sa.Integer(),
|
||||
nullable=False, server_default="0",
|
||||
),
|
||||
sa.Column(
|
||||
"matched_count", sa.Integer(),
|
||||
nullable=False, server_default="0",
|
||||
),
|
||||
sa.Column(
|
||||
"matched_ids", postgresql.JSONB(astext_type=sa.Text()),
|
||||
nullable=False, server_default=sa.text("'[]'::jsonb"),
|
||||
),
|
||||
sa.Column("error", sa.Text(), nullable=True),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_library_audit_run_rule", "library_audit_run", ["rule"],
|
||||
)
|
||||
op.create_index(
|
||||
"ix_library_audit_run_status", "library_audit_run", ["status"],
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_library_audit_run_status", table_name="library_audit_run")
|
||||
op.drop_index("ix_library_audit_run_rule", table_name="library_audit_run")
|
||||
op.drop_table("library_audit_run")
|
||||
@@ -1,54 +0,0 @@
|
||||
"""provenance-race: dedupe + UNIQUE(image_record_id, post_id) on image_provenance
|
||||
|
||||
Revision ID: 0021
|
||||
Revises: 0020
|
||||
Create Date: 2026-05-26
|
||||
|
||||
Closes the race in Importer._apply_sidecar's existence-check + INSERT pattern.
|
||||
Two workers writing for the same (image, post) pair both saw no existing row
|
||||
and both inserted, leaving duplicates that then broke .scalar_one_or_none()
|
||||
on every subsequent deep-scan rederive against those images
|
||||
(MultipleResultsFound). Most plausibly seeded when the 5-min recovery sweep
|
||||
re-enqueued a still-running long-import task and the second worker collided
|
||||
with the first inside _apply_sidecar.
|
||||
|
||||
Migration steps:
|
||||
1. DELETE all but min(id) per (image_record_id, post_id) pair. Operator's
|
||||
DB had 2 affected pairs at write-time; harmless no-op if zero.
|
||||
2. Add UNIQUE constraint so the importer's new savepoint+IntegrityError
|
||||
recovery path can trip on collision and re-select, mirroring
|
||||
uq_source_artist_platform_url and uq_post_source_external_id.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0021"
|
||||
down_revision: Union[str, None] = "0020"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.execute(
|
||||
"""
|
||||
DELETE FROM image_provenance ip1
|
||||
USING image_provenance ip2
|
||||
WHERE ip1.image_record_id = ip2.image_record_id
|
||||
AND ip1.post_id = ip2.post_id
|
||||
AND ip1.id > ip2.id
|
||||
"""
|
||||
)
|
||||
op.create_unique_constraint(
|
||||
"uq_image_provenance_image_post",
|
||||
"image_provenance",
|
||||
["image_record_id", "post_id"],
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_constraint(
|
||||
"uq_image_provenance_image_post",
|
||||
"image_provenance",
|
||||
type_="unique",
|
||||
)
|
||||
@@ -1,223 +0,0 @@
|
||||
"""source-collapse: one Source per (artist, platform) — consolidate junk per-post Sources
|
||||
|
||||
Revision ID: 0022
|
||||
Revises: 0021
|
||||
Create Date: 2026-05-26
|
||||
|
||||
Closes the operator-flagged 2026-05-26 issue where the filesystem importer
|
||||
called _find_or_create_source(url=sd.post_url), creating one Source row per
|
||||
imported post URL. Operator's Atole artist had 406 Source rows where there
|
||||
should have been 1 (the /cw/Atole subscription Source).
|
||||
|
||||
Source represents a subscription feed (one per artist+platform — the
|
||||
gallery-dl URL polled by the FC-3 downloader). Posts hang off it. The
|
||||
filesystem importer was misusing Source as a per-post key.
|
||||
|
||||
Migration steps per (artist_id, platform) group with >1 Source:
|
||||
1. Pick canonical — prefer a URL NOT matching '/posts/<id>$' (real
|
||||
campaign URL like /cw/Atole); else min(id).
|
||||
2. PRE-merge any Posts under non-canonical sources whose
|
||||
external_post_id ALREADY exists under the canonical source. (Same
|
||||
gallery-dl post imported via two different sidecar paths can plant
|
||||
two Post rows with identical external_post_id under different
|
||||
Sources for the same artist.) Repoint ImageProvenance +
|
||||
ImageRecord.primary_post_id to the canonical-side Post, dedupe
|
||||
ImageProvenance against alembic 0021's uq, then delete the
|
||||
non-canonical-side Post. This MUST happen before step 3 — Postgres
|
||||
fires uq_post_source_external_id row-by-row during the bulk UPDATE
|
||||
and the merge-after-reparent ordering 500s on first collision
|
||||
(operator-hit during v26.05.26.1 deploy, 2026-05-26).
|
||||
3. Reparent remaining Posts onto canonical (no collisions possible now).
|
||||
4. Reparent ImageProvenance.source_id off the non-canonical sources.
|
||||
5. Delete the orphan Source rows.
|
||||
6. If the canonical Source's URL still looks like a per-post URL (no
|
||||
campaign URL existed among candidates), rewrite it to
|
||||
'sidecar:<platform>:<artist_slug>' so the artist detail page shows
|
||||
something readable.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
from sqlalchemy import text
|
||||
|
||||
revision: str = "0022"
|
||||
down_revision: Union[str, None] = "0021"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
_POST_URL_RE = r"/posts/[^/]+$"
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
conn = op.get_bind()
|
||||
|
||||
# Find (artist_id, platform) groups with > 1 Source row.
|
||||
groups = conn.execute(text("""
|
||||
SELECT artist_id, platform
|
||||
FROM source
|
||||
GROUP BY artist_id, platform
|
||||
HAVING COUNT(*) > 1
|
||||
""")).fetchall()
|
||||
|
||||
for artist_id, platform in groups:
|
||||
rows = conn.execute(
|
||||
text("""
|
||||
SELECT id, url FROM source
|
||||
WHERE artist_id = :a AND platform = :p
|
||||
ORDER BY id ASC
|
||||
"""),
|
||||
{"a": artist_id, "p": platform},
|
||||
).fetchall()
|
||||
|
||||
# Canonical: first row whose URL doesn't look like a per-post URL;
|
||||
# else min(id).
|
||||
canonical_id = None
|
||||
for sid, url in rows:
|
||||
if not _matches_post_url(url):
|
||||
canonical_id = sid
|
||||
break
|
||||
if canonical_id is None:
|
||||
canonical_id = rows[0][0]
|
||||
|
||||
other_ids = [sid for sid, _ in rows if sid != canonical_id]
|
||||
if not other_ids:
|
||||
continue
|
||||
|
||||
# STEP 2: PRE-merge ALL Posts with duplicate external_post_id
|
||||
# across the entire (canonical + others) group, BEFORE the bulk
|
||||
# reparent. Two cases must both be handled:
|
||||
# (A) canonical has Post X with epid=N; an "other" source has
|
||||
# Post Y with epid=N → after bulk UPDATE, (canonical, N)
|
||||
# collides with itself.
|
||||
# (B) two different "other" sources each have a Post with
|
||||
# epid=N; canonical has none → after bulk UPDATE, both
|
||||
# are repointed to (canonical, N) and the second collides.
|
||||
# The earlier version of this migration only handled (A); the
|
||||
# operator's deploy 2026-05-26 tripped (B) at line 139.
|
||||
# Fix: group ALL Posts in the (artist, platform) by epid; for
|
||||
# any group with count>1, pick the keep (prefer one already
|
||||
# under canonical; else lowest id) and merge the rest into it.
|
||||
all_posts = conn.execute(
|
||||
text("""
|
||||
SELECT external_post_id, id, source_id
|
||||
FROM post
|
||||
WHERE source_id = :canonical OR source_id = ANY(:others)
|
||||
ORDER BY external_post_id, id
|
||||
"""),
|
||||
{"canonical": canonical_id, "others": other_ids},
|
||||
).fetchall()
|
||||
by_epid: dict = {}
|
||||
for epid, post_id, src_id in all_posts:
|
||||
by_epid.setdefault(epid, []).append((post_id, src_id))
|
||||
for _epid, posts in by_epid.items():
|
||||
if len(posts) <= 1:
|
||||
continue
|
||||
# Prefer a Post already under canonical as the keep.
|
||||
canonical_posts = [p for p in posts if p[1] == canonical_id]
|
||||
if canonical_posts:
|
||||
keep_id = canonical_posts[0][0]
|
||||
else:
|
||||
keep_id = posts[0][0] # already sorted by id ASC
|
||||
drop_ids = [p[0] for p in posts if p[0] != keep_id]
|
||||
for drop_id in drop_ids:
|
||||
# Pre-delete image_provenance rows under drop_ whose
|
||||
# image_record_id ALREADY has a provenance under keep —
|
||||
# the UPDATE below would otherwise repoint them and
|
||||
# trip uq_image_provenance_image_post (alembic 0021)
|
||||
# row-by-row before any after-the-fact dedupe could
|
||||
# run. Operator's v26.05.26.3 deploy 2026-05-26 tripped
|
||||
# this at line 123.
|
||||
conn.execute(
|
||||
text("""
|
||||
DELETE FROM image_provenance
|
||||
WHERE post_id = :drop_
|
||||
AND image_record_id IN (
|
||||
SELECT image_record_id FROM image_provenance
|
||||
WHERE post_id = :keep
|
||||
)
|
||||
"""),
|
||||
{"keep": keep_id, "drop_": drop_id},
|
||||
)
|
||||
# Now safe to repoint the survivors.
|
||||
conn.execute(
|
||||
text("""
|
||||
UPDATE image_provenance SET post_id = :keep
|
||||
WHERE post_id = :drop_
|
||||
"""),
|
||||
{"keep": keep_id, "drop_": drop_id},
|
||||
)
|
||||
conn.execute(
|
||||
text("""
|
||||
UPDATE image_record SET primary_post_id = :keep
|
||||
WHERE primary_post_id = :drop_
|
||||
"""),
|
||||
{"keep": keep_id, "drop_": drop_id},
|
||||
)
|
||||
conn.execute(
|
||||
text("DELETE FROM post WHERE id = :drop_"),
|
||||
{"drop_": drop_id},
|
||||
)
|
||||
|
||||
# STEP 3: Bulk reparent the remaining Posts off the other
|
||||
# Sources. After step 2, no collisions on
|
||||
# (canonical, external_post_id) are possible.
|
||||
conn.execute(
|
||||
text("""
|
||||
UPDATE post SET source_id = :canonical
|
||||
WHERE source_id = ANY(:others)
|
||||
"""),
|
||||
{"canonical": canonical_id, "others": other_ids},
|
||||
)
|
||||
|
||||
# STEP 4: Reparent ImageProvenance.source_id (denormalized FK).
|
||||
# No UNIQUE on source_id; safe bulk update.
|
||||
conn.execute(
|
||||
text("""
|
||||
UPDATE image_provenance SET source_id = :canonical
|
||||
WHERE source_id = ANY(:others)
|
||||
"""),
|
||||
{"canonical": canonical_id, "others": other_ids},
|
||||
)
|
||||
|
||||
# STEP 5: Drop the orphan Sources.
|
||||
conn.execute(
|
||||
text("DELETE FROM source WHERE id = ANY(:others)"),
|
||||
{"others": other_ids},
|
||||
)
|
||||
|
||||
# If the canonical's URL still looks per-post (no campaign URL
|
||||
# existed among the candidates), rewrite to a synthetic anchor so
|
||||
# the artist detail page renders something readable.
|
||||
canonical_url = conn.execute(
|
||||
text("SELECT url FROM source WHERE id = :id"),
|
||||
{"id": canonical_id},
|
||||
).scalar_one()
|
||||
if _matches_post_url(canonical_url):
|
||||
slug = conn.execute(
|
||||
text("SELECT slug FROM artist WHERE id = :id"),
|
||||
{"id": artist_id},
|
||||
).scalar_one()
|
||||
conn.execute(
|
||||
text("""
|
||||
UPDATE source
|
||||
SET url = :new_url, enabled = false
|
||||
WHERE id = :id
|
||||
"""),
|
||||
{
|
||||
"id": canonical_id,
|
||||
"new_url": f"sidecar:{platform}:{slug}",
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# Lossy migration — orphan Sources deleted, Posts reparented, Posts
|
||||
# merged. No safe downgrade. If you need to roll back the schema
|
||||
# invariant, fork from 0021 and re-run filesystem imports.
|
||||
pass
|
||||
|
||||
|
||||
def _matches_post_url(url: str) -> bool:
|
||||
"""True if url ends with /posts/<token> (gallery-dl-style per-post URL)."""
|
||||
import re
|
||||
return bool(re.search(_POST_URL_RE, url or ""))
|
||||
@@ -1,99 +0,0 @@
|
||||
"""drop meta + rating tag kinds — operator-retired 2026-05-26
|
||||
|
||||
Revision ID: 0023
|
||||
Revises: 0022
|
||||
Create Date: 2026-05-26
|
||||
|
||||
Operator decided meta + rating aren't valid tag kinds for FC. Per-row
|
||||
behavior: DELETE existing rows (operator chose "clean break" over
|
||||
"convert to general"). All cascading FKs (image_tag, tag_alias,
|
||||
tag_allowlist, tag_reference_embedding, tag_suggestion_rejection,
|
||||
series_page) use ondelete="CASCADE" so a single DELETE on tag cleans
|
||||
the related rows in one go.
|
||||
|
||||
After the data cleanup, recreate the tag_kind ENUM without 'meta' /
|
||||
'rating' (Postgres has no `ALTER TYPE ... DROP VALUE`; standard
|
||||
rename-create-cast-drop dance). The server default 'general' is
|
||||
dropped before the type swap and restored after.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0023"
|
||||
down_revision: Union[str, None] = "0022"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
# 1. Delete tags of the retired kinds. CASCADE handles related tables.
|
||||
op.execute("DELETE FROM tag WHERE kind IN ('meta', 'rating')")
|
||||
|
||||
# 2. Drop the CHECK constraint that references the enum's literal
|
||||
# values. Postgres can't resolve `kind = 'character'` across the
|
||||
# type swap below — the literal would bind to the new tag_kind
|
||||
# but the column is on tag_kind_old, producing
|
||||
# "operator does not exist: tag_kind = tag_kind_old".
|
||||
# (Operator-hit during the v26.05.26.5 deploy attempt; ck was
|
||||
# originally added by alembic 0002.) Recreated post-swap.
|
||||
op.drop_constraint(
|
||||
"ck_tag_fandom_requires_character", "tag", type_="check"
|
||||
)
|
||||
|
||||
# 3. Drop the server default — ALTER COLUMN TYPE can't carry it
|
||||
# across the type swap below.
|
||||
op.execute("ALTER TABLE tag ALTER COLUMN kind DROP DEFAULT")
|
||||
|
||||
# 4. Recreate the tag_kind enum without meta/rating.
|
||||
op.execute("ALTER TYPE tag_kind RENAME TO tag_kind_old")
|
||||
op.execute(
|
||||
"CREATE TYPE tag_kind AS ENUM ("
|
||||
"'artist', 'character', 'fandom', 'general', "
|
||||
"'series', 'archive', 'post'"
|
||||
")"
|
||||
)
|
||||
op.execute(
|
||||
"ALTER TABLE tag "
|
||||
"ALTER COLUMN kind TYPE tag_kind "
|
||||
"USING kind::text::tag_kind"
|
||||
)
|
||||
op.execute("DROP TYPE tag_kind_old")
|
||||
|
||||
# 5. Restore the server default.
|
||||
op.execute("ALTER TABLE tag ALTER COLUMN kind SET DEFAULT 'general'")
|
||||
|
||||
# 6. Restore the CHECK constraint (now bound to the new tag_kind).
|
||||
op.create_check_constraint(
|
||||
"ck_tag_fandom_requires_character",
|
||||
"tag",
|
||||
"(fandom_id IS NULL) OR (kind = 'character')",
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# Add the values back to the enum so old code can boot. The deleted
|
||||
# tag rows are gone permanently — no safe restore.
|
||||
op.drop_constraint(
|
||||
"ck_tag_fandom_requires_character", "tag", type_="check"
|
||||
)
|
||||
op.execute("ALTER TABLE tag ALTER COLUMN kind DROP DEFAULT")
|
||||
op.execute("ALTER TYPE tag_kind RENAME TO tag_kind_old")
|
||||
op.execute(
|
||||
"CREATE TYPE tag_kind AS ENUM ("
|
||||
"'artist', 'character', 'fandom', 'general', "
|
||||
"'series', 'archive', 'post', 'meta', 'rating'"
|
||||
")"
|
||||
)
|
||||
op.execute(
|
||||
"ALTER TABLE tag "
|
||||
"ALTER COLUMN kind TYPE tag_kind "
|
||||
"USING kind::text::tag_kind"
|
||||
)
|
||||
op.execute("DROP TYPE tag_kind_old")
|
||||
op.execute("ALTER TABLE tag ALTER COLUMN kind SET DEFAULT 'general'")
|
||||
op.create_check_constraint(
|
||||
"ck_tag_fandom_requires_character",
|
||||
"tag",
|
||||
"(fandom_id IS NULL) OR (kind = 'character')",
|
||||
)
|
||||
@@ -1,80 +0,0 @@
|
||||
"""backfill post.post_title from description first-line — 2026-05-27
|
||||
|
||||
Revision ID: 0024
|
||||
Revises: 0023
|
||||
Create Date: 2026-05-27
|
||||
|
||||
SubscribeStar gallery-dl always writes `title: ""` and embeds the leading
|
||||
sentence inside `content` HTML. FC's sidecar parser was leaving
|
||||
post_title NULL for every SubscribeStar post since FC-3 shipped. The
|
||||
parser fix (sidecar._first_line_text fallback) now synthesizes a title
|
||||
at parse time; this migration applies the same logic retroactively to
|
||||
existing rows.
|
||||
|
||||
Operator-flagged 2026-05-27 after inspecting
|
||||
/mnt/Data/Patreon/Cheunart/subscribestar/ sidecars.
|
||||
|
||||
Idempotent: only touches rows where post_title IS NULL or empty AND
|
||||
description IS NOT NULL. Re-running the migration is a no-op.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
from sqlalchemy import text
|
||||
|
||||
revision: str = "0024"
|
||||
down_revision: Union[str, None] = "0023"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
_TAG_RE = re.compile(r"<[^>]+>")
|
||||
_WS_RE = re.compile(r"\s+")
|
||||
|
||||
|
||||
def _first_line_text(body: str, limit: int = 120) -> str | None:
|
||||
"""Mirror of sidecar._first_line_text. Kept inline so the migration
|
||||
doesn't carry a runtime import dependency from app code that may
|
||||
have moved by the time the migration is replayed years from now."""
|
||||
if not body:
|
||||
return None
|
||||
text_ = _TAG_RE.sub(" ", body)
|
||||
text_ = text_.replace("\xa0", " ")
|
||||
for line in text_.splitlines():
|
||||
line = _WS_RE.sub(" ", line).strip()
|
||||
if line:
|
||||
if len(line) > limit:
|
||||
return line[: limit - 1].rstrip() + "…"
|
||||
return line
|
||||
return None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
bind = op.get_bind()
|
||||
rows = bind.execute(
|
||||
text(
|
||||
"SELECT id, description FROM post "
|
||||
"WHERE (post_title IS NULL OR post_title = '') "
|
||||
"AND description IS NOT NULL AND description <> ''"
|
||||
)
|
||||
).fetchall()
|
||||
updated = 0
|
||||
for row in rows:
|
||||
derived = _first_line_text(row.description)
|
||||
if not derived:
|
||||
continue
|
||||
bind.execute(
|
||||
text("UPDATE post SET post_title = :t WHERE id = :id"),
|
||||
{"t": derived, "id": row.id},
|
||||
)
|
||||
updated += 1
|
||||
print(f"0024: backfilled post_title on {updated} row(s)")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# No safe restore — we can't tell which post_titles were derived vs
|
||||
# genuinely present. Leave the column alone on rollback.
|
||||
pass
|
||||
@@ -1,288 +0,0 @@
|
||||
"""sidecar-audit followup: correct external_post_id + post_url across all platforms
|
||||
|
||||
Revision ID: 0025
|
||||
Revises: 0024
|
||||
Create Date: 2026-05-27
|
||||
|
||||
Closes the operator-flagged 2026-05-27 sidecar audit findings. Three
|
||||
data-correctness bugs across non-Patreon platforms had been silently
|
||||
corrupting Posts since FC-3 shipped; the parser fix (sidecar.py, same
|
||||
commit) addresses new imports. This migration cleans up existing rows.
|
||||
|
||||
Per-platform actions:
|
||||
|
||||
subscribestar — gallery-dl wrote the per-attachment id in `id` and
|
||||
the actual post id in `post_id`. FC's parser picked `id`, so every
|
||||
multi-image SubscribeStar post was fragmented into N Post rows.
|
||||
1. For each SubscribeStar Post, read its sidecar (via the related
|
||||
ImageRecord's on-disk path), pull `post_id`, overwrite
|
||||
external_post_id and post_url.
|
||||
2. Merge groups of Posts under one source that now share an
|
||||
external_post_id (fragments of the same actual post). Same
|
||||
ImageProvenance pre-delete + repoint dance as alembic 0022.
|
||||
|
||||
hentaifoundry — sidecars have NO `url` field; `src` is the image
|
||||
URL. FC's parser stored post_url=NULL. Read each HF Post's sidecar
|
||||
for `user` + `index`, derive the canonical /pictures/user/<u>/<i>
|
||||
permalink. external_post_id (= `index`) was already correct.
|
||||
|
||||
discord — gallery-dl wrote the CDN attachment URL in `url`. FC's
|
||||
parser stored that as post_url. Read each Discord Post's sidecar
|
||||
for the server/channel/message triple, derive the proper
|
||||
discord.com/channels/.../<message> permalink. external_post_id (=
|
||||
`message_id`) was already correct.
|
||||
|
||||
pixiv — pure-SQL backfill: replace any `i.pximg.net`-style URL on
|
||||
Post.post_url with the derived `/artworks/<id>` permalink. Pixiv
|
||||
external_post_id (= `id`) was already correct; no sidecar IO
|
||||
needed.
|
||||
|
||||
Idempotent: re-running on already-corrected data is a no-op (skips
|
||||
rows whose derived value matches what's already stored).
|
||||
|
||||
Posts whose related ImageRecord paths don't resolve on disk (orphaned
|
||||
filesystem state) are skipped with a count in the migration output —
|
||||
those will be picked up by a future deep-scan.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
from sqlalchemy import text
|
||||
|
||||
revision: str = "0025"
|
||||
down_revision: Union[str, None] = "0024"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
# Mirror of sidecar._NUMBERING_PREFIX. Kept inline so the migration is
|
||||
# self-contained (the operator's banked rule:
|
||||
# reference_postgres_enum_swap_drop_checks.md says migrations shouldn't
|
||||
# import from runtime app code).
|
||||
_NUMBERING_PREFIX = re.compile(r"^\d+_(.+)$")
|
||||
|
||||
|
||||
def _find_sidecar(media_path: Path) -> Path | None:
|
||||
"""gallery-dl writes the sidecar under the unprefixed stem
|
||||
(`HOLLOW-ICHIGO.json`) while the media file gets a NN_ ordering
|
||||
prefix (`01_HOLLOW-ICHIGO.png`). Try in order:
|
||||
1. <stem>.json next to the media
|
||||
2. <media>.json next to the media (full-name variant)
|
||||
3. strip the NN_ prefix from the stem, then <stripped>.json
|
||||
"""
|
||||
if not media_path:
|
||||
return None
|
||||
cand = media_path.with_suffix(".json")
|
||||
if cand.is_file():
|
||||
return cand
|
||||
cand = media_path.parent / f"{media_path.name}.json"
|
||||
if cand.is_file():
|
||||
return cand
|
||||
m = _NUMBERING_PREFIX.match(media_path.stem)
|
||||
if m:
|
||||
cand = media_path.parent / f"{m.group(1)}.json"
|
||||
if cand.is_file():
|
||||
return cand
|
||||
return None
|
||||
|
||||
|
||||
def _str_id(v) -> str | None:
|
||||
"""str() a JSON scalar id; reject bool (JSON booleans are ints in
|
||||
Python's eyes but they aren't valid sidecar ids)."""
|
||||
if isinstance(v, bool):
|
||||
return None
|
||||
if isinstance(v, (str, int)) and str(v).strip():
|
||||
return str(v).strip()
|
||||
return None
|
||||
|
||||
|
||||
def _str_field(v) -> str | None:
|
||||
if isinstance(v, str) and v.strip():
|
||||
return v.strip()
|
||||
return None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
conn = op.get_bind()
|
||||
|
||||
# ── PART 1: Per-platform corrections requiring filesystem IO ─────
|
||||
# SubscribeStar, HentaiFoundry, Discord all need fields from the
|
||||
# sidecar to construct the right post_url. We walk each Post's
|
||||
# related ImageRecord.path to find the sidecar, read it, derive,
|
||||
# and update.
|
||||
targets = conn.execute(text("""
|
||||
SELECT p.id, p.external_post_id, p.post_url, s.platform
|
||||
FROM post p
|
||||
JOIN source s ON s.id = p.source_id
|
||||
WHERE s.platform IN ('subscribestar', 'hentaifoundry', 'discord')
|
||||
""")).fetchall()
|
||||
|
||||
stats: dict[str, dict[str, int]] = {
|
||||
plat: {"read": 0, "updated": 0, "no_sidecar": 0}
|
||||
for plat in ("subscribestar", "hentaifoundry", "discord")
|
||||
}
|
||||
for post_row in targets:
|
||||
plat = post_row.platform
|
||||
path = _first_attachment_path(conn, post_row.id)
|
||||
if not path:
|
||||
stats[plat]["no_sidecar"] += 1
|
||||
continue
|
||||
sidecar = _find_sidecar(Path(path))
|
||||
if sidecar is None:
|
||||
stats[plat]["no_sidecar"] += 1
|
||||
continue
|
||||
try:
|
||||
data = json.loads(sidecar.read_text(encoding="utf-8"))
|
||||
except (OSError, json.JSONDecodeError):
|
||||
stats[plat]["no_sidecar"] += 1
|
||||
continue
|
||||
stats[plat]["read"] += 1
|
||||
|
||||
new_epid = post_row.external_post_id
|
||||
new_url = None
|
||||
if plat == "subscribestar":
|
||||
pid = _str_id(data.get("post_id"))
|
||||
if pid:
|
||||
new_epid = pid
|
||||
new_url = f"https://www.subscribestar.com/posts/{pid}"
|
||||
elif plat == "hentaifoundry":
|
||||
user = _str_field(data.get("user")) or _str_field(data.get("artist"))
|
||||
idx = _str_id(data.get("index"))
|
||||
if user and idx:
|
||||
new_url = f"https://www.hentai-foundry.com/pictures/user/{user}/{idx}"
|
||||
elif plat == "discord":
|
||||
sid = _str_id(data.get("server_id"))
|
||||
cid = _str_id(data.get("channel_id"))
|
||||
mid = _str_id(data.get("message_id"))
|
||||
if sid and cid and mid:
|
||||
new_url = f"https://discord.com/channels/{sid}/{cid}/{mid}"
|
||||
|
||||
# Idempotent: skip if nothing changed.
|
||||
if new_epid == post_row.external_post_id and new_url == post_row.post_url:
|
||||
continue
|
||||
conn.execute(
|
||||
text("""
|
||||
UPDATE post
|
||||
SET external_post_id = :epid, post_url = :url
|
||||
WHERE id = :id
|
||||
"""),
|
||||
{"epid": new_epid, "url": new_url, "id": post_row.id},
|
||||
)
|
||||
stats[plat]["updated"] += 1
|
||||
|
||||
for plat, s in stats.items():
|
||||
print(
|
||||
f"0025: {plat} — read {s['read']} sidecars, "
|
||||
f"updated {s['updated']} Posts, "
|
||||
f"{s['no_sidecar']} Posts had no resolvable sidecar"
|
||||
)
|
||||
|
||||
# ── PART 2: Merge SubscribeStar fragments now sharing epid ───────
|
||||
# After Part 1, each group of Posts under one source with the SAME
|
||||
# new external_post_id is a fragment-set of the same actual post.
|
||||
# Merge to one canonical row. Pre-handle the same ImageProvenance
|
||||
# collision pattern as alembic 0022 (uq_image_provenance_image_post).
|
||||
fragment_groups = conn.execute(text("""
|
||||
SELECT p.source_id, p.external_post_id,
|
||||
ARRAY_AGG(p.id ORDER BY p.id ASC) AS post_ids
|
||||
FROM post p
|
||||
JOIN source s ON s.id = p.source_id
|
||||
WHERE s.platform = 'subscribestar'
|
||||
AND p.external_post_id IS NOT NULL
|
||||
GROUP BY p.source_id, p.external_post_id
|
||||
HAVING COUNT(*) > 1
|
||||
""")).fetchall()
|
||||
|
||||
merged = 0
|
||||
for grp in fragment_groups:
|
||||
post_ids = list(grp.post_ids)
|
||||
keep_id, *drop_ids = post_ids
|
||||
for drop_id in drop_ids:
|
||||
# Pre-DELETE colliding ImageProvenance under drop_ that
|
||||
# already exist under keep (alembic 0022 banked the pattern).
|
||||
conn.execute(
|
||||
text("""
|
||||
DELETE FROM image_provenance
|
||||
WHERE post_id = :drop_
|
||||
AND image_record_id IN (
|
||||
SELECT image_record_id FROM image_provenance
|
||||
WHERE post_id = :keep
|
||||
)
|
||||
"""),
|
||||
{"keep": keep_id, "drop_": drop_id},
|
||||
)
|
||||
conn.execute(
|
||||
text("""
|
||||
UPDATE image_provenance SET post_id = :keep
|
||||
WHERE post_id = :drop_
|
||||
"""),
|
||||
{"keep": keep_id, "drop_": drop_id},
|
||||
)
|
||||
conn.execute(
|
||||
text("""
|
||||
UPDATE image_record SET primary_post_id = :keep
|
||||
WHERE primary_post_id = :drop_
|
||||
"""),
|
||||
{"keep": keep_id, "drop_": drop_id},
|
||||
)
|
||||
conn.execute(
|
||||
text("""
|
||||
UPDATE post_attachment SET post_id = :keep
|
||||
WHERE post_id = :drop_
|
||||
"""),
|
||||
{"keep": keep_id, "drop_": drop_id},
|
||||
)
|
||||
conn.execute(
|
||||
text("DELETE FROM post WHERE id = :drop_"),
|
||||
{"drop_": drop_id},
|
||||
)
|
||||
merged += 1
|
||||
print(f"0025: subscribestar — merged {merged} duplicate Post fragments")
|
||||
|
||||
# ── PART 3: Pixiv post_url backfill (pure SQL) ───────────────────
|
||||
# Pixiv's external_post_id is already correct (gallery-dl's `id` is
|
||||
# the post id). Only post_url needs derivation: replace anything
|
||||
# under i.pximg.net (the file URL) with the /artworks/<id> permalink.
|
||||
pixiv_updated = conn.execute(text("""
|
||||
UPDATE post p
|
||||
SET post_url = 'https://www.pixiv.net/artworks/' || p.external_post_id
|
||||
FROM source s
|
||||
WHERE p.source_id = s.id
|
||||
AND s.platform = 'pixiv'
|
||||
AND p.external_post_id IS NOT NULL
|
||||
AND (p.post_url IS NULL
|
||||
OR p.post_url LIKE 'https://i.pximg.net/%'
|
||||
OR p.post_url LIKE 'http://i.pximg.net/%')
|
||||
""")).rowcount
|
||||
print(f"0025: pixiv — backfilled post_url on {pixiv_updated} Posts")
|
||||
|
||||
|
||||
def _first_attachment_path(conn, post_id: int) -> str | None:
|
||||
"""Return any ImageRecord.path attached to this post (via
|
||||
ImageProvenance). Lowest-id row keeps the migration deterministic
|
||||
so re-running on the same DB picks the same sidecar."""
|
||||
row = conn.execute(
|
||||
text("""
|
||||
SELECT ir.path
|
||||
FROM image_provenance ip
|
||||
JOIN image_record ir ON ir.id = ip.image_record_id
|
||||
WHERE ip.post_id = :pid
|
||||
ORDER BY ip.id ASC
|
||||
LIMIT 1
|
||||
"""),
|
||||
{"pid": post_id},
|
||||
).first()
|
||||
return row[0] if row else None
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# Lossy: external_post_id values were overwritten with the correct
|
||||
# post_id; original per-attachment ids weren't preserved. Post-merge
|
||||
# also deleted drop rows. No safe restore. To roll back the schema
|
||||
# invariant, fork from 0024 and re-run sidecar imports.
|
||||
pass
|
||||
@@ -1,53 +0,0 @@
|
||||
"""import_task.recovery_count + refetched — poison-pill circuit breaker
|
||||
|
||||
Revision ID: 0026
|
||||
Revises: 0025
|
||||
Create Date: 2026-05-28
|
||||
|
||||
Backs the import-task resilience work (operator-flagged 2026-05-28):
|
||||
|
||||
- recovery_count: how many times recover_interrupted_tasks has
|
||||
re-queued this row from a stuck 'processing' state. A row that
|
||||
hard-crashes the worker (OOM / segfault on a corrupt or oversized
|
||||
input) leaves no terminal flip, so the sweep re-queues it — and
|
||||
without a cap it would loop forever, re-crashing the worker each
|
||||
time. After MAX_RECOVERY_ATTEMPTS the sweep marks it 'failed' with a
|
||||
diagnostic instead.
|
||||
|
||||
- refetched: whether a one-shot re-download has already been attempted
|
||||
for this task's file. Bounds the Layer-2 re-fetch remediation to a
|
||||
single attempt so source-side corruption doesn't loop.
|
||||
|
||||
Both default to 0 / false; additive, no backfill needed.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0026"
|
||||
down_revision: Union[str, None] = "0025"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"import_task",
|
||||
sa.Column(
|
||||
"recovery_count", sa.Integer(), nullable=False,
|
||||
server_default="0",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"import_task",
|
||||
sa.Column(
|
||||
"refetched", sa.Boolean(), nullable=False,
|
||||
server_default=sa.false(),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("import_task", "refetched")
|
||||
op.drop_column("import_task", "recovery_count")
|
||||
@@ -1,50 +0,0 @@
|
||||
"""drop migration_run — one-and-done GS/IR migration tooling removed
|
||||
|
||||
Revision ID: 0027
|
||||
Revises: 0026
|
||||
Create Date: 2026-05-29
|
||||
|
||||
The GS/IR migration tooling (services/migrators, /api/migrate, the
|
||||
run_migration task, LegacyMigrationCard, and the MigrationRun model) was
|
||||
removed after the migration cutover completed. This drops its now-orphaned
|
||||
run-log table. Downgrade recreates the table (mirrors the old model) so the
|
||||
migration is reversible.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from sqlalchemy.dialects.postgresql import JSONB
|
||||
|
||||
revision: str = "0027"
|
||||
down_revision: Union[str, None] = "0026"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.drop_table("migration_run")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.create_table(
|
||||
"migration_run",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column("kind", sa.String(length=32), nullable=False),
|
||||
sa.Column("status", sa.String(length=32), nullable=False),
|
||||
sa.Column("dry_run", sa.Boolean(), nullable=False, server_default=sa.false()),
|
||||
sa.Column(
|
||||
"started_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column(
|
||||
"counts", JSONB(), nullable=False, server_default=sa.text("'{}'::jsonb"),
|
||||
),
|
||||
sa.Column("error", sa.Text(), nullable=True),
|
||||
sa.Column(
|
||||
"metadata", JSONB(), nullable=False, server_default=sa.text("'{}'::jsonb"),
|
||||
),
|
||||
)
|
||||
op.create_index("ix_migration_run_kind", "migration_run", ["kind"])
|
||||
op.create_index("ix_migration_run_status", "migration_run", ["status"])
|
||||
@@ -1,190 +0,0 @@
|
||||
"""collapse-sidecar-synthetic: repoint Posts/ImageProvenance/DownloadEvents
|
||||
from `sidecar:<platform>:<slug>` synthetic Source anchors onto the real
|
||||
Source for the same (artist, platform) when one exists, then delete the
|
||||
synthetic.
|
||||
|
||||
Revision ID: 0028
|
||||
Revises: 0027
|
||||
Create Date: 2026-05-31
|
||||
|
||||
Background: alembic 0022 (2026-05-26) consolidated the old per-post-URL
|
||||
Source rows into one canonical Source per (artist, platform). When NO
|
||||
real campaign URL was salvageable among the candidates, it rewrote the
|
||||
canonical row to url='sidecar:<platform>:<slug>' enabled=false as a
|
||||
disabled anchor for any Posts already attached.
|
||||
|
||||
That was fine while it was the only Source for that artist+platform.
|
||||
But: the unique constraint on Source is (artist_id, platform, url), not
|
||||
(artist_id, platform). When the operator later added the real
|
||||
subscription via the UI / extension / etc., a SECOND row landed —
|
||||
the real one — with id > the synthetic. Both coexisted.
|
||||
|
||||
Two follow-on problems surfaced 2026-05-31:
|
||||
|
||||
1. The Subscriptions UI listed both rows. The synthetic was disabled
|
||||
so the scheduler never polled it, but it looked like a phantom
|
||||
subscription. (Fixed in same commit by SourceService.list filter.)
|
||||
2. importer._source_for_sidecar picked Source by `ORDER BY id ASC
|
||||
LIMIT 1`, so EVERY gallery-dl download since the real Source was
|
||||
added attached its Post to the SYNTHETIC anchor, not the real
|
||||
Source. (Fixed in same commit by preferring non-sidecar URLs.)
|
||||
|
||||
This migration is the data half of the cleanup: for every (artist,
|
||||
platform) with both a synthetic AND a real Source, repoint the
|
||||
synthetic's children (Posts, ImageProvenance, DownloadEvents) onto the
|
||||
real Source and delete the synthetic. Reuses the same epid/provenance
|
||||
collision dance from alembic 0022 because the same uniqueness
|
||||
constraints fire row-by-row during bulk UPDATEs.
|
||||
|
||||
Lone synthetic anchors — those where no real Source for the same
|
||||
(artist, platform) exists (e.g., filesystem-imported artist with no
|
||||
subscription added) — are LEFT INTACT. They anchor real imported
|
||||
content; deleting them would CASCADE-delete the Posts the operator
|
||||
imported. The SourceService.list filter hides them from the UI; the
|
||||
operator can delete them by hand if they want the underlying imports
|
||||
gone.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
from sqlalchemy import text
|
||||
|
||||
revision: str = "0028"
|
||||
down_revision: Union[str, None] = "0027"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
conn = op.get_bind()
|
||||
|
||||
# Find (artist_id, platform) groups where BOTH a sidecar synthetic
|
||||
# and at least one real Source exist.
|
||||
groups = conn.execute(text("""
|
||||
SELECT artist_id, platform
|
||||
FROM source
|
||||
GROUP BY artist_id, platform
|
||||
HAVING bool_or(url LIKE 'sidecar:%')
|
||||
AND bool_or(url NOT LIKE 'sidecar:%')
|
||||
""")).fetchall()
|
||||
|
||||
for artist_id, platform in groups:
|
||||
rows = conn.execute(
|
||||
text("""
|
||||
SELECT id, url FROM source
|
||||
WHERE artist_id = :a AND platform = :p
|
||||
ORDER BY id ASC
|
||||
"""),
|
||||
{"a": artist_id, "p": platform},
|
||||
).fetchall()
|
||||
|
||||
synthetic_ids = [sid for sid, url in rows if url.startswith("sidecar:")]
|
||||
real_rows = [(sid, url) for sid, url in rows if not url.startswith("sidecar:")]
|
||||
if not synthetic_ids or not real_rows:
|
||||
continue # belt+suspenders; the GROUP BY already filtered
|
||||
|
||||
# Canonical real: lowest-id non-sidecar Source.
|
||||
canonical_id = real_rows[0][0]
|
||||
|
||||
# STEP A: PRE-merge Post collisions on (canonical, external_post_id).
|
||||
# Mirror alembic 0022's pre-merge logic — when synth has Post X
|
||||
# epid=N and real has Post Y epid=N, the bulk UPDATE below would
|
||||
# trip uq_post_source_external_id row-by-row. Group all Posts
|
||||
# under (canonical + synthetics) by epid; for any group >1,
|
||||
# pick a keep (prefer one already under canonical, else lowest
|
||||
# id) and merge the rest into it.
|
||||
all_posts = conn.execute(
|
||||
text("""
|
||||
SELECT external_post_id, id, source_id
|
||||
FROM post
|
||||
WHERE source_id = :canonical OR source_id = ANY(:synths)
|
||||
ORDER BY external_post_id, id
|
||||
"""),
|
||||
{"canonical": canonical_id, "synths": synthetic_ids},
|
||||
).fetchall()
|
||||
by_epid: dict = {}
|
||||
for epid, post_id, src_id in all_posts:
|
||||
by_epid.setdefault(epid, []).append((post_id, src_id))
|
||||
for _epid, posts in by_epid.items():
|
||||
if len(posts) <= 1:
|
||||
continue
|
||||
canonical_side = [p for p in posts if p[1] == canonical_id]
|
||||
keep_id = canonical_side[0][0] if canonical_side else posts[0][0]
|
||||
drop_ids = [p[0] for p in posts if p[0] != keep_id]
|
||||
for drop_id in drop_ids:
|
||||
# Pre-delete image_provenance rows under drop_ whose
|
||||
# image_record_id already has provenance under keep —
|
||||
# avoids tripping uq_image_provenance_image_post (0021)
|
||||
# row-by-row during the repoint UPDATE.
|
||||
conn.execute(
|
||||
text("""
|
||||
DELETE FROM image_provenance
|
||||
WHERE post_id = :drop_
|
||||
AND image_record_id IN (
|
||||
SELECT image_record_id FROM image_provenance
|
||||
WHERE post_id = :keep
|
||||
)
|
||||
"""),
|
||||
{"keep": keep_id, "drop_": drop_id},
|
||||
)
|
||||
conn.execute(
|
||||
text("""
|
||||
UPDATE image_provenance SET post_id = :keep
|
||||
WHERE post_id = :drop_
|
||||
"""),
|
||||
{"keep": keep_id, "drop_": drop_id},
|
||||
)
|
||||
conn.execute(
|
||||
text("""
|
||||
UPDATE image_record SET primary_post_id = :keep
|
||||
WHERE primary_post_id = :drop_
|
||||
"""),
|
||||
{"keep": keep_id, "drop_": drop_id},
|
||||
)
|
||||
conn.execute(
|
||||
text("DELETE FROM post WHERE id = :drop_"),
|
||||
{"drop_": drop_id},
|
||||
)
|
||||
|
||||
# STEP B: Bulk reparent the remaining Posts off the synthetics.
|
||||
conn.execute(
|
||||
text("""
|
||||
UPDATE post SET source_id = :canonical
|
||||
WHERE source_id = ANY(:synths)
|
||||
"""),
|
||||
{"canonical": canonical_id, "synths": synthetic_ids},
|
||||
)
|
||||
|
||||
# STEP C: Reparent ImageProvenance.source_id (denormalized FK;
|
||||
# no UNIQUE on source_id, safe bulk).
|
||||
conn.execute(
|
||||
text("""
|
||||
UPDATE image_provenance SET source_id = :canonical
|
||||
WHERE source_id = ANY(:synths)
|
||||
"""),
|
||||
{"canonical": canonical_id, "synths": synthetic_ids},
|
||||
)
|
||||
|
||||
# STEP D: Reparent any DownloadEvent.source_id. Synthetics are
|
||||
# enabled=false so the scheduler never created events for them;
|
||||
# this is belt+suspenders for any rows planted by manual force
|
||||
# or older code paths.
|
||||
conn.execute(
|
||||
text("""
|
||||
UPDATE download_event SET source_id = :canonical
|
||||
WHERE source_id = ANY(:synths)
|
||||
"""),
|
||||
{"canonical": canonical_id, "synths": synthetic_ids},
|
||||
)
|
||||
|
||||
# STEP E: Drop the now-empty synthetics.
|
||||
conn.execute(
|
||||
text("DELETE FROM source WHERE id = ANY(:synths)"),
|
||||
{"synths": synthetic_ids},
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# Lossy migration — synthetic Sources deleted, Posts repointed and
|
||||
# potentially merged. No safe downgrade.
|
||||
pass
|
||||
@@ -1,71 +0,0 @@
|
||||
"""drop artist + copyright ml thresholds; lower general default to 0.50
|
||||
|
||||
Revision ID: 0029
|
||||
Revises: 0028
|
||||
Create Date: 2026-06-01
|
||||
|
||||
Operator-flagged 2026-06-01: the view modal's Suggestions panel hides
|
||||
most general-category predictions because the default threshold is
|
||||
0.95. Lowering the default to 0.50 (matches character) so general
|
||||
suggestions surface more aggressively; the value remains tunable in
|
||||
Settings → ML.
|
||||
|
||||
Same change retires two ML suggestion categories whose Tag.kind
|
||||
surfaces are unused:
|
||||
|
||||
- `artist`: retired in FC-2d-vii-c — artist identity is acquisition-
|
||||
derived (image_record.artist_id), never ML-inferred. The threshold
|
||||
column was a leftover from before that retirement.
|
||||
- `copyright`: retired 2026-06-01 — the app uses `fandom` for the
|
||||
franchise/copyright concept (per TagsView.vue's doc comment); no
|
||||
Tag rows of kind=copyright exist, and the threshold column never
|
||||
fed anything user-visible.
|
||||
|
||||
Both columns are dropped from ml_settings; the existing row's
|
||||
suggestion_threshold_general value is bumped from 0.95 to 0.50 iff
|
||||
it's still at the old default, so deployed installs pick up the new
|
||||
UX without overriding any operator tuning.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
from sqlalchemy import text
|
||||
|
||||
revision: str = "0029"
|
||||
down_revision: Union[str, None] = "0028"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
# Bump the general threshold for installs still at the old default.
|
||||
op.execute(text(
|
||||
"UPDATE ml_settings "
|
||||
"SET suggestion_threshold_general = 0.50 "
|
||||
"WHERE id = 1 AND suggestion_threshold_general = 0.95"
|
||||
))
|
||||
op.drop_column("ml_settings", "suggestion_threshold_artist")
|
||||
op.drop_column("ml_settings", "suggestion_threshold_copyright")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# Restore the columns with their prior defaults. The bump from
|
||||
# 0.95 → 0.50 isn't reversible without remembering whether the
|
||||
# operator had explicitly set 0.95 (unlikely — that was just the
|
||||
# default) so we leave the current general value as-is.
|
||||
from sqlalchemy import Column, Float
|
||||
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
Column(
|
||||
"suggestion_threshold_artist",
|
||||
Float, nullable=False, server_default="0.30",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
Column(
|
||||
"suggestion_threshold_copyright",
|
||||
Float, nullable=False, server_default="0.50",
|
||||
),
|
||||
)
|
||||
@@ -1,145 +0,0 @@
|
||||
"""nullable post.source_id + denormalized post.artist_id; retire sidecar synthetics
|
||||
|
||||
Revision ID: 0030
|
||||
Revises: 0029
|
||||
Create Date: 2026-06-01
|
||||
|
||||
Operator-asked 2026-06-01 after the Dymkens orphan investigation: the
|
||||
sidecar synthetic Source pattern (`sidecar:<platform>:<slug>` rows
|
||||
with enabled=false) was technically correct but misled the operator
|
||||
into thinking they had phantom subscriptions. The synthetics existed
|
||||
solely to satisfy `Post.source_id NOT NULL` for filesystem-imported
|
||||
content with no real subscription.
|
||||
|
||||
This migration makes the data model honest:
|
||||
|
||||
1. **Post gets a denormalized `artist_id` column** so artist filters
|
||||
work without traversing `Post → Source.artist_id`. Backfilled from
|
||||
the existing Source linkage, then NOT NULL'd.
|
||||
2. **`Post.source_id` becomes nullable**, FK ondelete `CASCADE` → `SET
|
||||
NULL`. Deleting a Source detaches its Posts instead of destroying
|
||||
imported content (semantically: subscription ends, archive stays).
|
||||
3. **`ImageProvenance.source_id` becomes nullable** with the same FK
|
||||
semantic change.
|
||||
4. **Sidecar synthetic Sources are deleted** — first NULL out the
|
||||
FKs from Post + ImageProvenance pointing at them (so the implicit
|
||||
CASCADE doesn't fire), then delete. DownloadEvent FK is unchanged
|
||||
(still CASCADE'd, NOT NULL'd) — synthetics have `enabled=false`
|
||||
so no events exist for them.
|
||||
|
||||
Uniqueness handling: the existing `uq_post_source_external_id`
|
||||
(source_id, external_post_id) keeps working for source-bound Posts
|
||||
(Postgres treats NULL != NULL so NULL-source rows aren't deduped by
|
||||
it). A second partial unique index covers the NULL-source case on
|
||||
(artist_id, external_post_id) so filesystem-imported posts still
|
||||
dedupe within an artist.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from sqlalchemy import text
|
||||
|
||||
revision: str = "0030"
|
||||
down_revision: Union[str, None] = "0029"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
conn = op.get_bind()
|
||||
|
||||
# Step 1: add Post.artist_id, initially nullable for backfill.
|
||||
# FK naming follows the Base.metadata naming_convention
|
||||
# (fk_<table>_<column>_<referred_table>) — alembic 0001 set this up.
|
||||
op.add_column(
|
||||
"post",
|
||||
sa.Column("artist_id", sa.Integer, nullable=True),
|
||||
)
|
||||
op.create_foreign_key(
|
||||
"fk_post_artist_id_artist", "post", "artist",
|
||||
["artist_id"], ["id"], ondelete="CASCADE",
|
||||
)
|
||||
|
||||
# Step 2: backfill from Source.artist_id (every existing Post has a
|
||||
# Source today, so every row gets populated).
|
||||
conn.execute(text("""
|
||||
UPDATE post p
|
||||
SET artist_id = s.artist_id
|
||||
FROM source s
|
||||
WHERE p.source_id = s.id AND p.artist_id IS NULL
|
||||
"""))
|
||||
|
||||
# Sanity: count any remaining NULLs. Should be zero pre-this-migration.
|
||||
remaining = conn.execute(text(
|
||||
"SELECT COUNT(*) FROM post WHERE artist_id IS NULL"
|
||||
)).scalar_one()
|
||||
if remaining:
|
||||
raise RuntimeError(
|
||||
f"alembic 0030: {remaining} post rows have no resolvable "
|
||||
f"artist_id after backfill. Investigate before continuing."
|
||||
)
|
||||
|
||||
# Step 3: enforce NOT NULL + add index for artist-filter queries.
|
||||
op.alter_column("post", "artist_id", nullable=False)
|
||||
op.create_index("ix_post_artist_id", "post", ["artist_id"])
|
||||
|
||||
# Step 4: relax post.source_id + flip FK to SET NULL. The original FK
|
||||
# name from alembic 0001 is `fk_post_source_id_source` per the
|
||||
# NAMING_CONVENTION in models/base.py.
|
||||
op.alter_column("post", "source_id", nullable=True)
|
||||
op.drop_constraint("fk_post_source_id_source", "post", type_="foreignkey")
|
||||
op.create_foreign_key(
|
||||
"fk_post_source_id_source", "post", "source",
|
||||
["source_id"], ["id"], ondelete="SET NULL",
|
||||
)
|
||||
|
||||
# Step 5: relax image_provenance.source_id + flip FK to SET NULL.
|
||||
op.alter_column("image_provenance", "source_id", nullable=True)
|
||||
op.drop_constraint(
|
||||
"fk_image_provenance_source_id_source", "image_provenance",
|
||||
type_="foreignkey",
|
||||
)
|
||||
op.create_foreign_key(
|
||||
"fk_image_provenance_source_id_source", "image_provenance", "source",
|
||||
["source_id"], ["id"], ondelete="SET NULL",
|
||||
)
|
||||
|
||||
# Step 6: partial unique index on (artist_id, external_post_id) for
|
||||
# NULL-source Posts. The existing uq_post_source_external_id keeps
|
||||
# guarding source-bound rows; NULL-source rows now dedupe within
|
||||
# an artist.
|
||||
op.execute(
|
||||
"CREATE UNIQUE INDEX uq_post_artist_external_id_null_source "
|
||||
"ON post (artist_id, external_post_id) "
|
||||
"WHERE source_id IS NULL"
|
||||
)
|
||||
|
||||
# Step 7: retire sidecar synthetic Sources. NULL out the references
|
||||
# FIRST (the new FK is SET NULL so CASCADE wouldn't fire anyway, but
|
||||
# being explicit makes the intent clear). Then delete the synthetic
|
||||
# source rows. Any DownloadEvent rows under synthetics CASCADE-die
|
||||
# with the source — synthetics have enabled=false so there shouldn't
|
||||
# be any in practice.
|
||||
conn.execute(text("""
|
||||
UPDATE post
|
||||
SET source_id = NULL
|
||||
WHERE source_id IN (SELECT id FROM source WHERE url LIKE 'sidecar:%')
|
||||
"""))
|
||||
conn.execute(text("""
|
||||
UPDATE image_provenance
|
||||
SET source_id = NULL
|
||||
WHERE source_id IN (SELECT id FROM source WHERE url LIKE 'sidecar:%')
|
||||
"""))
|
||||
deleted = conn.execute(text(
|
||||
"DELETE FROM source WHERE url LIKE 'sidecar:%' RETURNING id"
|
||||
)).rowcount
|
||||
print(f"alembic 0030: deleted {deleted} sidecar synthetic source rows")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# Lossy migration — the deleted sidecar synthetics can't be
|
||||
# restored from the orphan post.source_id / image_provenance.source_id
|
||||
# values, and the partial unique index encodes a constraint that
|
||||
# NULL-source Posts may now exist. No safe downgrade.
|
||||
pass
|
||||
@@ -1,45 +0,0 @@
|
||||
"""source.backfill_runs_remaining: sticky deep-scan mode
|
||||
|
||||
Revision ID: 0031
|
||||
Revises: 0030
|
||||
Create Date: 2026-06-01
|
||||
|
||||
Tick vs backfill mode for subscription downloads. When
|
||||
`backfill_runs_remaining > 0`, the next N download runs use
|
||||
`skip: True` + 30-min timeout (walk full history). When 0, runs use
|
||||
`skip: "exit:20"` + 14.5-min timeout (catch-up mode, exits early once
|
||||
20 contiguous archived items are seen).
|
||||
|
||||
Operator-flagged 2026-06-01 (Knuxy run #38887): a creator with ~550
|
||||
archived posts saturates the 870s catch-up timeout even when there is
|
||||
no new content, because gallery-dl's default `skip: True` keeps walking.
|
||||
Tick mode short-circuits that; backfill mode is the explicit opt-in for
|
||||
deep history scans.
|
||||
|
||||
Default 0 (all existing subscriptions start in tick mode).
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0031"
|
||||
down_revision: Union[str, None] = "0030"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"source",
|
||||
sa.Column(
|
||||
"backfill_runs_remaining",
|
||||
sa.Integer,
|
||||
nullable=False,
|
||||
server_default="0",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("source", "backfill_runs_remaining")
|
||||
@@ -1,41 +0,0 @@
|
||||
"""source.error_type: surface ErrorType taxonomy in FailingSourcesCard
|
||||
|
||||
Revision ID: 0032
|
||||
Revises: 0031
|
||||
Create Date: 2026-06-02
|
||||
|
||||
Audit 2026-06-02: the backend computes 13 ErrorType categories (auth_error,
|
||||
rate_limited, not_found, access_denied, validation_failed, etc.) and
|
||||
stamps each one on DownloadEvent.metadata, but the Source row only carried
|
||||
the free-text last_error. Operators couldn't bulk-triage failing sources
|
||||
("all auth_error → rotate cookies, all rate_limited → just wait") without
|
||||
opening Logs per row.
|
||||
|
||||
This column receives the last error_type from _update_source_health
|
||||
and gets cleared on a successful run. Nullable + indexed so the failing-
|
||||
sources rollup can filter/group cheaply.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0032"
|
||||
down_revision: Union[str, None] = "0031"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"source",
|
||||
sa.Column("error_type", sa.String(length=32), nullable=True),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_source_error_type", "source", ["error_type"],
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_source_error_type", table_name="source")
|
||||
op.drop_column("source", "error_type")
|
||||
@@ -1,48 +0,0 @@
|
||||
"""suggestion_threshold default 0.50 → 0.70
|
||||
|
||||
Revision ID: 0033
|
||||
Revises: 0032
|
||||
Create Date: 2026-06-02
|
||||
|
||||
Operator-flagged 2026-06-02 — the 0.50 default (set on 2026-06-01) is
|
||||
too noisy in practice; raise to 0.70 for both suggestion categories.
|
||||
|
||||
Only conditionally updates singletons whose current value is still the
|
||||
2026-06-01 default (0.50). Operators who deliberately tuned their row
|
||||
to some other value (0.55, 0.65, 0.80, etc. via the Settings UI) keep
|
||||
their pick — the migration only catches the unchanged-default case.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0033"
|
||||
down_revision: Union[str, None] = "0032"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.execute(
|
||||
"UPDATE ml_settings "
|
||||
"SET suggestion_threshold_character = 0.70 "
|
||||
"WHERE id = 1 AND suggestion_threshold_character = 0.50"
|
||||
)
|
||||
op.execute(
|
||||
"UPDATE ml_settings "
|
||||
"SET suggestion_threshold_general = 0.70 "
|
||||
"WHERE id = 1 AND suggestion_threshold_general = 0.50"
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.execute(
|
||||
"UPDATE ml_settings "
|
||||
"SET suggestion_threshold_character = 0.50 "
|
||||
"WHERE id = 1 AND suggestion_threshold_character = 0.70"
|
||||
)
|
||||
op.execute(
|
||||
"UPDATE ml_settings "
|
||||
"SET suggestion_threshold_general = 0.50 "
|
||||
"WHERE id = 1 AND suggestion_threshold_general = 0.70"
|
||||
)
|
||||
@@ -1,53 +0,0 @@
|
||||
"""artist_visit: per-artist last-viewed timestamp for the "+N new" badge
|
||||
|
||||
Revision ID: 0034
|
||||
Revises: 0033
|
||||
Create Date: 2026-06-03
|
||||
|
||||
Powers the artists-directory "+N new since last visit" badge + ArtistView
|
||||
banner. Single row per artist (no user_id yet — rule #47 multi-user ACL
|
||||
is aspirational; widens to (user_id, artist_id) PK when User lands).
|
||||
|
||||
Seed every existing artist with `last_viewed_at = NOW()` so the badge
|
||||
starts at 0 across the board — no noisy "you have 5000 unseen images"
|
||||
on first deploy. New artists auto-get a row via
|
||||
`ArtistService.find_or_create`.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0034"
|
||||
down_revision: Union[str, None] = "0033"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"artist_visit",
|
||||
sa.Column(
|
||||
"artist_id",
|
||||
sa.Integer,
|
||||
sa.ForeignKey("artist.id", ondelete="CASCADE"),
|
||||
primary_key=True,
|
||||
),
|
||||
sa.Column(
|
||||
"last_viewed_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.text("NOW()"),
|
||||
),
|
||||
)
|
||||
# Seed: every existing artist starts "fully caught up". Without this,
|
||||
# every operator with N artists would see N badges (worth of every
|
||||
# image ever imported) on first deploy.
|
||||
op.execute(
|
||||
"INSERT INTO artist_visit (artist_id, last_viewed_at) "
|
||||
"SELECT id, NOW() FROM artist"
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_table("artist_visit")
|
||||
@@ -1,70 +0,0 @@
|
||||
"""image_record.effective_date: materialized gallery sort key + index
|
||||
|
||||
Revision ID: 0035
|
||||
Revises: 0034
|
||||
Create Date: 2026-06-04
|
||||
|
||||
The gallery ordered/cursored on COALESCE(post.post_date,
|
||||
image_record.created_at) across the Post outer join. That expression spans
|
||||
two tables, so no index can serve it — every /scroll sorted a large slice
|
||||
of the library, and the frontend fired ten of them serially per initial
|
||||
load. Materialize the value into image_record.effective_date and index
|
||||
(effective_date DESC, id DESC) so the cursor scroll is an index range scan.
|
||||
|
||||
Backfill = COALESCE(primary post's post_date, created_at) so existing rows
|
||||
keep their exact ordering. New rows get the created_at-equivalent server
|
||||
default; services/importer.py overrides it with the post's date when a
|
||||
primary post with a date is linked.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0035"
|
||||
down_revision: Union[str, None] = "0034"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
# Add nullable first so the backfill can populate before NOT NULL.
|
||||
op.add_column(
|
||||
"image_record",
|
||||
sa.Column("effective_date", sa.DateTime(timezone=True), nullable=True),
|
||||
)
|
||||
# Pure set-based UPDATEs (no per-row params) — immune to the 65535
|
||||
# bind-parameter ceiling regardless of library size.
|
||||
op.execute(
|
||||
"""
|
||||
UPDATE image_record AS ir
|
||||
SET effective_date = COALESCE(p.post_date, ir.created_at)
|
||||
FROM post AS p
|
||||
WHERE ir.primary_post_id = p.id
|
||||
"""
|
||||
)
|
||||
op.execute(
|
||||
"""
|
||||
UPDATE image_record
|
||||
SET effective_date = created_at
|
||||
WHERE effective_date IS NULL
|
||||
"""
|
||||
)
|
||||
op.alter_column(
|
||||
"image_record",
|
||||
"effective_date",
|
||||
nullable=False,
|
||||
server_default=sa.text("now()"),
|
||||
)
|
||||
# DESC/DESC matches the gallery's ORDER BY effective_date DESC, id DESC
|
||||
# so the scroll is a forward index scan; raw SQL because alembic's
|
||||
# column list doesn't express per-column DESC cleanly.
|
||||
op.execute(
|
||||
"CREATE INDEX ix_image_record_effective_date "
|
||||
"ON image_record (effective_date DESC, id DESC)"
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_image_record_effective_date", table_name="image_record")
|
||||
op.drop_column("image_record", "effective_date")
|
||||
@@ -1,41 +0,0 @@
|
||||
"""image_record.siglip_embedding: HNSW cosine index for "more like this"
|
||||
|
||||
Revision ID: 0036
|
||||
Revises: 0035
|
||||
Create Date: 2026-06-04
|
||||
|
||||
Gallery Phase 3 (visual similarity search) ranks images by
|
||||
`siglip_embedding.cosine_distance(source_embedding)`. Without an index that's
|
||||
a sequential scan computing a 1152-dim distance for every row — fine at small
|
||||
scale, but it grows linearly with the library. Add an HNSW index with
|
||||
`vector_cosine_ops` so the top-N nearest search is sub-50ms ANN.
|
||||
|
||||
1152 dims is under pgvector's 2000-dim HNSW limit, so HNSW (no training,
|
||||
better recall than IVFFlat) is the right choice. ONE-TIME COST: building the
|
||||
index over the existing embeddings (~57k vectors on the operator's library)
|
||||
locks image_record for ~30-60s during this migration on deploy — acceptable
|
||||
for a single-operator homelab. NULL embeddings (videos / not-yet-embedded
|
||||
rows) are simply not indexed.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0036"
|
||||
down_revision: Union[str, None] = "0035"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
# Raw SQL: alembic's create_index doesn't express the `USING hnsw (...
|
||||
# vector_cosine_ops)` access-method + opclass cleanly. Must match the
|
||||
# query's cosine_distance operator class to be usable by the planner.
|
||||
op.execute(
|
||||
"CREATE INDEX ix_image_record_siglip_hnsw "
|
||||
"ON image_record USING hnsw (siglip_embedding vector_cosine_ops)"
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_image_record_siglip_hnsw", table_name="image_record")
|
||||
@@ -1,53 +0,0 @@
|
||||
"""patreon_seen_media: per-source ledger of already-ingested Patreon media
|
||||
|
||||
Revision ID: 0037
|
||||
Revises: 0036
|
||||
Create Date: 2026-06-05
|
||||
|
||||
Native Patreon ingester (build step 2a). Replaces gallery-dl's
|
||||
archive.sqlite3 with our own queryable table. The downloader upserts one
|
||||
row per (source, media) so routine walks skip media we've already
|
||||
processed; a future "recovery" mode bypasses the ledger to re-walk.
|
||||
|
||||
`filehash` is a 32-hex Patreon CDN MD5, OR a video sentinel of the form
|
||||
``video:<post_id>:<media_id>`` — hence String(128). The unique
|
||||
constraint on (source_id, filehash) is the dedup upsert key.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0037"
|
||||
down_revision: Union[str, None] = "0036"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"patreon_seen_media",
|
||||
sa.Column("id", sa.Integer, primary_key=True),
|
||||
sa.Column(
|
||||
"source_id",
|
||||
sa.Integer,
|
||||
sa.ForeignKey("source.id", ondelete="CASCADE"),
|
||||
nullable=False,
|
||||
index=True,
|
||||
),
|
||||
sa.Column("filehash", sa.String(128), nullable=False),
|
||||
sa.Column("post_id", sa.String(64), nullable=True),
|
||||
sa.Column(
|
||||
"seen_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.text("NOW()"),
|
||||
),
|
||||
sa.UniqueConstraint(
|
||||
"source_id", "filehash", name="uq_patreon_seen_media_source_id"
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_table("patreon_seen_media")
|
||||
@@ -1,58 +0,0 @@
|
||||
"""patreon_failed_media: per-source dead-letter ledger for failing Patreon media
|
||||
|
||||
Revision ID: 0038
|
||||
Revises: 0037
|
||||
Create Date: 2026-06-06
|
||||
|
||||
Plan #705 (#7). Media that keeps failing to download/validate (404'd CDN,
|
||||
deleted post, geo-blocked Mux, persistently-corrupt bytes) gets recorded here
|
||||
with an attempt counter; once it crosses the dead-letter threshold the ingester
|
||||
skips it on routine walks (recovery still re-attempts). A clean download clears
|
||||
the row. UNIQUE (source_id, filehash) is the upsert key (same media key the
|
||||
seen-ledger uses).
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0038"
|
||||
down_revision: Union[str, None] = "0037"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"patreon_failed_media",
|
||||
sa.Column("id", sa.Integer, primary_key=True),
|
||||
sa.Column(
|
||||
"source_id",
|
||||
sa.Integer,
|
||||
sa.ForeignKey("source.id", ondelete="CASCADE"),
|
||||
nullable=False,
|
||||
index=True,
|
||||
),
|
||||
sa.Column("filehash", sa.String(128), nullable=False),
|
||||
sa.Column("attempts", sa.Integer, nullable=False, server_default="1"),
|
||||
sa.Column("last_error", sa.Text, nullable=True),
|
||||
sa.Column(
|
||||
"first_failed_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.text("NOW()"),
|
||||
),
|
||||
sa.Column(
|
||||
"last_failed_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.text("NOW()"),
|
||||
),
|
||||
sa.UniqueConstraint(
|
||||
"source_id", "filehash", name="uq_patreon_failed_media_source_id"
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_table("patreon_failed_media")
|
||||
@@ -1,40 +0,0 @@
|
||||
"""library_audit_run: resume cursor + progress timestamp for chunked scans
|
||||
|
||||
Revision ID: 0039
|
||||
Revises: 0038
|
||||
Create Date: 2026-06-07
|
||||
|
||||
scan_library_for_rule used to run one 2h pass that timed out on large libraries
|
||||
and monopolized the concurrency-1 maintenance queue (operator-flagged). It now
|
||||
runs short time-boxed chunks that re-enqueue: `resume_after_id` persists the
|
||||
keyset cursor so the next chunk continues where it left off, and
|
||||
`last_progress_at` lets the recovery sweep tell a progressing multi-chunk audit
|
||||
from a genuinely stuck one.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0039"
|
||||
down_revision: Union[str, None] = "0038"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"library_audit_run",
|
||||
sa.Column(
|
||||
"resume_after_id", sa.Integer, nullable=False, server_default="0"
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"library_audit_run",
|
||||
sa.Column("last_progress_at", sa.DateTime(timezone=True), nullable=True),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("library_audit_run", "last_progress_at")
|
||||
op.drop_column("library_audit_run", "resume_after_id")
|
||||
@@ -1,108 +0,0 @@
|
||||
"""series chapters: chapter layer over series_page (FC-6.1)
|
||||
|
||||
Revision ID: 0040
|
||||
Revises: 0039
|
||||
Create Date: 2026-06-07
|
||||
|
||||
A series (Tag kind='series') gains an ordered chapter layer. Reading order
|
||||
becomes (series_chapter.chapter_number, series_page.page_number). Every existing
|
||||
series is backfilled into a single auto-chapter (chapter_number=1) holding its
|
||||
current flat pages, so no data is lost and the old flat ordering is preserved.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0040"
|
||||
down_revision: Union[str, None] = "0039"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"series_chapter",
|
||||
sa.Column("id", sa.Integer, primary_key=True),
|
||||
sa.Column(
|
||||
"series_tag_id",
|
||||
sa.Integer,
|
||||
sa.ForeignKey("tag.id", ondelete="CASCADE"),
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column("chapter_number", sa.Integer, nullable=False),
|
||||
sa.Column("title", sa.Text, nullable=True),
|
||||
sa.Column(
|
||||
"is_placeholder", sa.Boolean, nullable=False, server_default="false"
|
||||
),
|
||||
sa.Column("stated_page_start", sa.Integer, nullable=True),
|
||||
sa.Column("stated_page_end", sa.Integer, nullable=True),
|
||||
sa.Column(
|
||||
"created_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.text("now()"),
|
||||
),
|
||||
sa.Column(
|
||||
"updated_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.text("now()"),
|
||||
),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_series_chapter_series_tag_id", "series_chapter", ["series_tag_id"]
|
||||
)
|
||||
|
||||
# New columns on series_page; chapter_id starts nullable so we can backfill.
|
||||
op.add_column(
|
||||
"series_page", sa.Column("chapter_id", sa.Integer, nullable=True)
|
||||
)
|
||||
op.add_column(
|
||||
"series_page", sa.Column("stated_page", sa.Integer, nullable=True)
|
||||
)
|
||||
|
||||
conn = op.get_bind()
|
||||
# One auto-chapter per existing series (any series_tag_id present in pages).
|
||||
conn.execute(
|
||||
sa.text(
|
||||
"INSERT INTO series_chapter "
|
||||
"(series_tag_id, chapter_number, is_placeholder, created_at, updated_at) "
|
||||
"SELECT DISTINCT series_tag_id, 1, false, now(), now() "
|
||||
"FROM series_page"
|
||||
)
|
||||
)
|
||||
# Point every existing page at its series' auto-chapter.
|
||||
conn.execute(
|
||||
sa.text(
|
||||
"UPDATE series_page sp "
|
||||
"SET chapter_id = sc.id "
|
||||
"FROM series_chapter sc "
|
||||
"WHERE sc.series_tag_id = sp.series_tag_id"
|
||||
)
|
||||
)
|
||||
|
||||
# Now lock chapter_id down: NOT NULL + FK (cascade) + index.
|
||||
op.alter_column("series_page", "chapter_id", nullable=False)
|
||||
op.create_foreign_key(
|
||||
"fk_series_page_chapter_id",
|
||||
"series_page",
|
||||
"series_chapter",
|
||||
["chapter_id"],
|
||||
["id"],
|
||||
ondelete="CASCADE",
|
||||
)
|
||||
op.create_index(
|
||||
"ix_series_page_chapter_id", "series_page", ["chapter_id"]
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_series_page_chapter_id", table_name="series_page")
|
||||
op.drop_constraint(
|
||||
"fk_series_page_chapter_id", "series_page", type_="foreignkey"
|
||||
)
|
||||
op.drop_column("series_page", "stated_page")
|
||||
op.drop_column("series_page", "chapter_id")
|
||||
op.drop_index("ix_series_chapter_series_tag_id", table_name="series_chapter")
|
||||
op.drop_table("series_chapter")
|
||||
@@ -1,98 +0,0 @@
|
||||
"""series suggestions: assisted-continuation matcher (FC-6.3)
|
||||
|
||||
Revision ID: 0041
|
||||
Revises: 0040
|
||||
Create Date: 2026-06-07
|
||||
|
||||
A confirm-only queue of "this post may continue this series" hints, plus two
|
||||
import_settings knobs (enable + score threshold) for the matcher.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0041"
|
||||
down_revision: Union[str, None] = "0040"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"series_suggestion",
|
||||
sa.Column("id", sa.Integer, primary_key=True),
|
||||
sa.Column(
|
||||
"post_id",
|
||||
sa.Integer,
|
||||
sa.ForeignKey("post.id", ondelete="CASCADE"),
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column(
|
||||
"series_tag_id",
|
||||
sa.Integer,
|
||||
sa.ForeignKey("tag.id", ondelete="CASCADE"),
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column("score", sa.Float, nullable=False),
|
||||
sa.Column("signals", sa.JSON, nullable=True),
|
||||
sa.Column(
|
||||
"status", sa.String(16), nullable=False, server_default="pending"
|
||||
),
|
||||
sa.Column(
|
||||
"created_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.text("now()"),
|
||||
),
|
||||
sa.Column(
|
||||
"updated_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.text("now()"),
|
||||
),
|
||||
sa.UniqueConstraint(
|
||||
"post_id", "series_tag_id", name="uq_series_suggestion_post_series"
|
||||
),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_series_suggestion_post_id", "series_suggestion", ["post_id"]
|
||||
)
|
||||
op.create_index(
|
||||
"ix_series_suggestion_series_tag_id",
|
||||
"series_suggestion",
|
||||
["series_tag_id"],
|
||||
)
|
||||
op.create_index(
|
||||
"ix_series_suggestion_status", "series_suggestion", ["status"]
|
||||
)
|
||||
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"series_suggest_enabled",
|
||||
sa.Boolean,
|
||||
nullable=False,
|
||||
server_default=sa.true(),
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"series_suggest_threshold",
|
||||
sa.Float,
|
||||
nullable=False,
|
||||
server_default="0.5",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("import_settings", "series_suggest_threshold")
|
||||
op.drop_column("import_settings", "series_suggest_enabled")
|
||||
op.drop_index("ix_series_suggestion_status", table_name="series_suggestion")
|
||||
op.drop_index(
|
||||
"ix_series_suggestion_series_tag_id", table_name="series_suggestion"
|
||||
)
|
||||
op.drop_index("ix_series_suggestion_post_id", table_name="series_suggestion")
|
||||
op.drop_table("series_suggestion")
|
||||
@@ -1,32 +0,0 @@
|
||||
"""series chapter stated_part: operator-facing Part N label (FC-6.4)
|
||||
|
||||
Revision ID: 0042
|
||||
Revises: 0041
|
||||
Create Date: 2026-06-07
|
||||
|
||||
A chapter's positional chapter_number is auto-managed (rewritten 1..N on
|
||||
reorder/delete), so it can't double as the installment number the operator wants
|
||||
to type (e.g. a series authored from a post that is Part 2). Add a nullable
|
||||
stated_part alongside it — the same split as series_page.page_number (order) vs
|
||||
series_page.stated_page (printed number). Nullable; the UI falls back to
|
||||
chapter_number when unset.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0042"
|
||||
down_revision: Union[str, None] = "0041"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"series_chapter", sa.Column("stated_part", sa.Integer, nullable=True)
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("series_chapter", "stated_part")
|
||||
@@ -1,62 +0,0 @@
|
||||
"""post_attachment: per-post sha uniqueness (empty-post flood fix)
|
||||
|
||||
Revision ID: 0043
|
||||
Revises: 0042
|
||||
Create Date: 2026-06-08
|
||||
|
||||
PostAttachment.sha256 was GLOBALLY unique, so a non-art file the creator attaches
|
||||
to many posts (a standard pdf/zip/link-card) only ever got ONE row — on the first
|
||||
post — leaving every later post a bare shell (no image, no attachment). The native
|
||||
Patreon backfill of Anduo surfaced 1589 such shells (operator-flagged 2026-06-08).
|
||||
|
||||
Switch to PER-POST uniqueness: the on-disk blob stays sha-deduped, but each post
|
||||
gets its own row. Replace the unique sha256 index with a plain lookup index plus
|
||||
two partial uniques — (post_id, sha256) for real posts and (sha256) for the
|
||||
NULL-post filesystem case (still one row per file there).
|
||||
|
||||
Existing data has ≤1 row per sha (the old global unique), so the new partial
|
||||
uniques can't be violated on upgrade — no data backfill needed here. The bare-post
|
||||
shells themselves are removed by the separate prune-empty-posts cleanup tool.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0043"
|
||||
down_revision: Union[str, None] = "0042"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
# Drop the global unique index; recreate it as a plain (non-unique) lookup
|
||||
# index so sha-based reads keep their index (matches the model's index=True).
|
||||
op.drop_index("ix_post_attachment_sha256", table_name="post_attachment")
|
||||
op.create_index(
|
||||
"ix_post_attachment_sha256", "post_attachment", ["sha256"],
|
||||
)
|
||||
op.create_index(
|
||||
"uq_post_attachment_post_sha", "post_attachment",
|
||||
["post_id", "sha256"], unique=True,
|
||||
postgresql_where=sa.text("post_id IS NOT NULL"),
|
||||
)
|
||||
op.create_index(
|
||||
"uq_post_attachment_null_post_sha", "post_attachment",
|
||||
["sha256"], unique=True,
|
||||
postgresql_where=sa.text("post_id IS NULL"),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index(
|
||||
"uq_post_attachment_null_post_sha", table_name="post_attachment"
|
||||
)
|
||||
op.drop_index(
|
||||
"uq_post_attachment_post_sha", table_name="post_attachment"
|
||||
)
|
||||
op.drop_index("ix_post_attachment_sha256", table_name="post_attachment")
|
||||
op.create_index(
|
||||
"ix_post_attachment_sha256", "post_attachment", ["sha256"],
|
||||
unique=True,
|
||||
)
|
||||
@@ -1,37 +0,0 @@
|
||||
"""ml_settings.tagger_store_floor
|
||||
|
||||
The ingest confidence floor below which tagger predictions are not stored,
|
||||
promoted from the TAGGER_STORE_FLOOR env var to a DB-backed, UI-tunable
|
||||
setting. Default 0.70 (was an env default of 0.05): the suggestion path
|
||||
already filters at 0.70 and the centroid/learned path covers low-confidence
|
||||
preferred tags, so the sub-0.70 tail was redundant weight — it had grown
|
||||
image_record's TOAST to ~100 GB. See plan-task #764.
|
||||
|
||||
Revision ID: 0044
|
||||
Revises: 0043
|
||||
Create Date: 2026-06-10
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0044"
|
||||
down_revision: Union[str, None] = "0043"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"tagger_store_floor", sa.Float(),
|
||||
nullable=False, server_default="0.7",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("ml_settings", "tagger_store_floor")
|
||||
@@ -1,69 +0,0 @@
|
||||
"""image_prediction table (DDL only — backfill runs as a background task)
|
||||
|
||||
Normalizes the per-image tagger predictions out of the JSON blob into a
|
||||
queryable table (#768). This migration creates ONLY the table + indexes — it
|
||||
is pure DDL and commits instantly, so web boots immediately.
|
||||
|
||||
The data backfill from the existing image_record.tagger_predictions JSON is
|
||||
deliberately NOT done here. Doing it inline made the whole migration one
|
||||
transaction over the ~100 GB TOAST: nothing committed until the very end, it
|
||||
was invisible/unmonitorable mid-run, and an early MATERIALIZED-CTE form spilled
|
||||
the full 100 GB to temp. Instead the backfill is the
|
||||
backend.app.tasks.admin.backfill_image_predictions_task — batched by id window,
|
||||
committed per chunk (visible progress + resumable), idempotent
|
||||
(ON CONFLICT DO NOTHING). Trigger it from Settings → Maintenance once web is up.
|
||||
|
||||
The old image_record.tagger_predictions column is left in place (vestigial) and
|
||||
dropped in a follow-up once the backfill + code cutover are verified — dropping
|
||||
it needs an ACCESS EXCLUSIVE lock on the hot image_record table (the 0044 lock
|
||||
class), so it's deferred to a quiesced-worker window.
|
||||
|
||||
Revision ID: 0045
|
||||
Revises: 0044
|
||||
Create Date: 2026-06-10
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0045"
|
||||
down_revision: Union[str, None] = "0044"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"image_prediction",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column(
|
||||
"image_record_id", sa.Integer(),
|
||||
sa.ForeignKey("image_record.id", ondelete="CASCADE"),
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column("raw_name", sa.String(length=255), nullable=False),
|
||||
sa.Column("category", sa.String(length=64), nullable=False),
|
||||
sa.Column("score", sa.Float(), nullable=False),
|
||||
sa.UniqueConstraint(
|
||||
"image_record_id", "raw_name", name="image_raw_name",
|
||||
),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_image_prediction_image", "image_prediction", ["image_record_id"],
|
||||
)
|
||||
op.create_index(
|
||||
"ix_image_prediction_name_score", "image_prediction",
|
||||
["raw_name", "score"],
|
||||
)
|
||||
# No data backfill here — see the module docstring. The one-time copy from
|
||||
# image_record.tagger_predictions runs as backfill_image_predictions_task
|
||||
# (batched, resumable, idempotent), kept out of this transaction so web boots
|
||||
# without waiting on a ~100 GB pass.
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_image_prediction_name_score", "image_prediction")
|
||||
op.drop_index("ix_image_prediction_image", "image_prediction")
|
||||
op.drop_table("image_prediction")
|
||||
@@ -1,43 +0,0 @@
|
||||
"""drop image_record.tagger_predictions (predictions normalized to image_prediction)
|
||||
|
||||
Final step of #768. The per-tag predictions now live in the image_prediction
|
||||
table (backfilled from the JSON, read by suggestions + allowlist, written by
|
||||
tag_and_embed). The old JSON column is dead weight — and it's the ~100 GB of
|
||||
sub-0.70 score tail that bloated image_record's TOAST and broke DB backups
|
||||
(#739). Dropping it is a fast catalog change; it does NOT reclaim the disk on
|
||||
its own — run `VACUUM FULL image_record` (or pg_repack) afterward, off-hours,
|
||||
to return the space to the OS so backups go small.
|
||||
|
||||
DROP COLUMN needs a brief ACCESS EXCLUSIVE lock on image_record; env.py's
|
||||
lock_timeout guards it, so quiesce the ml-worker if a tagging run is in flight
|
||||
(see the migration-lock reference). tagger_model_version is kept — it's the
|
||||
"has this been tagged / is it current?" signal the backfill sweep reads.
|
||||
|
||||
Revision ID: 0046
|
||||
Revises: 0045
|
||||
Create Date: 2026-06-11
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0046"
|
||||
down_revision: Union[str, None] = "0045"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.drop_column("image_record", "tagger_predictions")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# Re-add the column empty. The JSON data is not restored (it lived only in
|
||||
# this column); a downgrade would re-tag or backfill from image_prediction
|
||||
# separately if ever needed.
|
||||
op.add_column(
|
||||
"image_record",
|
||||
sa.Column("tagger_predictions", sa.JSON(), nullable=True),
|
||||
)
|
||||
@@ -1,175 +0,0 @@
|
||||
"""series chapters become cosmetic dividers; pages become one series-global run
|
||||
|
||||
FC-6.x reframe (#789). A series is now ONE flat, series-global ordered run of
|
||||
pages; chapters stop owning pages and become labeled dividers anchored to the
|
||||
page that begins them.
|
||||
|
||||
Migration (order matters — series_page.chapter_id cascades, so it must be
|
||||
dropped BEFORE any chapter row is deleted, or pages would cascade away):
|
||||
a. Renumber series_page.page_number to a series-global 1..N (ordered by the
|
||||
OLD (chapter_number, page_number)).
|
||||
b. Add series_chapter.anchor_page_id and populate it with each chapter's first
|
||||
page (lowest new page_number).
|
||||
c. Drop series_page.chapter_id (severs the cascade link).
|
||||
d. Prune chapters that shouldn't become dividers: empty/placeholder ones (no
|
||||
anchor) and the redundant unlabeled chapter that would sit at page 1.
|
||||
e. Reshape series_chapter into the divider: drop chapter_number,
|
||||
is_placeholder, stated_page_start/end; make anchor_page_id NOT NULL +
|
||||
UNIQUE + FK→series_page ON DELETE CASCADE.
|
||||
|
||||
Revision ID: 0047
|
||||
Revises: 0046
|
||||
Create Date: 2026-06-11
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0047"
|
||||
down_revision: Union[str, None] = "0046"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
# a. series-global page numbering, preserving the old reading order.
|
||||
op.execute(
|
||||
"""
|
||||
WITH ordered AS (
|
||||
SELECT sp.id,
|
||||
ROW_NUMBER() OVER (
|
||||
PARTITION BY sp.series_tag_id
|
||||
ORDER BY sc.chapter_number, sp.page_number, sp.id
|
||||
) AS rn
|
||||
FROM series_page sp
|
||||
JOIN series_chapter sc ON sc.id = sp.chapter_id
|
||||
)
|
||||
UPDATE series_page sp
|
||||
SET page_number = ordered.rn
|
||||
FROM ordered
|
||||
WHERE sp.id = ordered.id
|
||||
"""
|
||||
)
|
||||
|
||||
# b. anchor each existing chapter at its first page (lowest new page_number).
|
||||
op.add_column(
|
||||
"series_chapter",
|
||||
sa.Column("anchor_page_id", sa.Integer(), nullable=True),
|
||||
)
|
||||
op.execute(
|
||||
"""
|
||||
WITH firsts AS (
|
||||
SELECT DISTINCT ON (sp.chapter_id)
|
||||
sp.chapter_id, sp.id AS page_id
|
||||
FROM series_page sp
|
||||
ORDER BY sp.chapter_id, sp.page_number, sp.id
|
||||
)
|
||||
UPDATE series_chapter sc
|
||||
SET anchor_page_id = firsts.page_id
|
||||
FROM firsts
|
||||
WHERE firsts.chapter_id = sc.id
|
||||
"""
|
||||
)
|
||||
|
||||
# c. sever the ownership link (drops the FK + index with the column) BEFORE
|
||||
# pruning chapters, so deleting a chapter can't cascade-delete its pages.
|
||||
op.drop_column("series_page", "chapter_id")
|
||||
|
||||
# d. prune chapters that don't become dividers: placeholders / empty ones
|
||||
# (no anchor), and the unlabeled chapter that would land redundantly at
|
||||
# page 1 (the series just starts — no divider needed there).
|
||||
op.execute(
|
||||
"""
|
||||
DELETE FROM series_chapter sc
|
||||
USING (
|
||||
SELECT sc2.id
|
||||
FROM series_chapter sc2
|
||||
LEFT JOIN series_page sp ON sp.id = sc2.anchor_page_id
|
||||
WHERE sc2.anchor_page_id IS NULL
|
||||
OR (sp.page_number = 1
|
||||
AND sc2.title IS NULL
|
||||
AND sc2.stated_part IS NULL)
|
||||
) gone
|
||||
WHERE sc.id = gone.id
|
||||
"""
|
||||
)
|
||||
|
||||
# e. reshape into the divider model.
|
||||
op.drop_column("series_chapter", "chapter_number")
|
||||
op.drop_column("series_chapter", "is_placeholder")
|
||||
op.drop_column("series_chapter", "stated_page_start")
|
||||
op.drop_column("series_chapter", "stated_page_end")
|
||||
op.alter_column("series_chapter", "anchor_page_id", nullable=False)
|
||||
op.create_unique_constraint(
|
||||
"uq_series_chapter_anchor_page", "series_chapter", ["anchor_page_id"]
|
||||
)
|
||||
op.create_foreign_key(
|
||||
"fk_series_chapter_anchor_page",
|
||||
"series_chapter",
|
||||
"series_page",
|
||||
["anchor_page_id"],
|
||||
["id"],
|
||||
ondelete="CASCADE",
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# Lossy: dividers can't be reconstructed as owning chapters. Collapse back to
|
||||
# exactly one chapter per series that owns all its pages in order.
|
||||
op.add_column(
|
||||
"series_page", sa.Column("chapter_id", sa.Integer(), nullable=True)
|
||||
)
|
||||
op.drop_constraint(
|
||||
"fk_series_chapter_anchor_page", "series_chapter", type_="foreignkey"
|
||||
)
|
||||
op.drop_constraint(
|
||||
"uq_series_chapter_anchor_page", "series_chapter", type_="unique"
|
||||
)
|
||||
op.drop_column("series_chapter", "anchor_page_id")
|
||||
op.add_column(
|
||||
"series_chapter",
|
||||
sa.Column(
|
||||
"chapter_number", sa.Integer(), nullable=False, server_default="1"
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"series_chapter",
|
||||
sa.Column(
|
||||
"is_placeholder", sa.Boolean(), nullable=False,
|
||||
server_default="false",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"series_chapter",
|
||||
sa.Column("stated_page_start", sa.Integer(), nullable=True),
|
||||
)
|
||||
op.add_column(
|
||||
"series_chapter",
|
||||
sa.Column("stated_page_end", sa.Integer(), nullable=True),
|
||||
)
|
||||
op.execute("DELETE FROM series_chapter")
|
||||
op.execute(
|
||||
"""
|
||||
INSERT INTO series_chapter (series_tag_id, chapter_number)
|
||||
SELECT DISTINCT series_tag_id, 1 FROM series_page
|
||||
"""
|
||||
)
|
||||
op.execute(
|
||||
"""
|
||||
UPDATE series_page sp
|
||||
SET chapter_id = sc.id
|
||||
FROM series_chapter sc
|
||||
WHERE sc.series_tag_id = sp.series_tag_id
|
||||
"""
|
||||
)
|
||||
op.alter_column("series_page", "chapter_id", nullable=False)
|
||||
op.create_foreign_key(
|
||||
"fk_series_page_chapter",
|
||||
"series_page",
|
||||
"series_chapter",
|
||||
["chapter_id"],
|
||||
["id"],
|
||||
ondelete="CASCADE",
|
||||
)
|
||||
@@ -1,45 +0,0 @@
|
||||
"""series_page pending staging: status + nullable page_number (#789 Phase 2)
|
||||
|
||||
Pages added from a post no longer append straight into the run — they land
|
||||
'pending' with a NULL page_number, staged grouped by their source post so the
|
||||
operator can drop junk (text-free alts, bumpers) and place the keepers into the
|
||||
sequence. A page only gets a series-global page_number once it's 'placed'.
|
||||
|
||||
Revision ID: 0048
|
||||
Revises: 0047
|
||||
Create Date: 2026-06-11
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0048"
|
||||
down_revision: Union[str, None] = "0047"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"series_page",
|
||||
sa.Column(
|
||||
"status", sa.String(length=16), nullable=False,
|
||||
server_default="placed",
|
||||
),
|
||||
)
|
||||
op.alter_column(
|
||||
"series_page", "page_number",
|
||||
existing_type=sa.Integer(), nullable=True,
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# Lossy: pending pages are unsorted staging rows with no order — drop them.
|
||||
op.execute("DELETE FROM series_page WHERE status = 'pending'")
|
||||
op.alter_column(
|
||||
"series_page", "page_number",
|
||||
existing_type=sa.Integer(), nullable=False,
|
||||
)
|
||||
op.drop_column("series_page", "status")
|
||||
@@ -1,90 +0,0 @@
|
||||
"""external_link table — off-platform file-host links found in post bodies
|
||||
|
||||
Creators host the real files on mega.nz / Google Drive / MediaFire / Dropbox /
|
||||
Pixeldrain and link them in the post text. This table records each such link
|
||||
(so nothing is silently dropped), and doubles as the dedup + dead-letter ledger
|
||||
the download worker (a later slice) walks. `url` keeps the FULL link including
|
||||
the `#fragment` — mega.nz's decryption key lives there; truncating it makes the
|
||||
file undownloadable.
|
||||
|
||||
CHECK whitelists for host + status include the full enum up front (incl. the
|
||||
download-worker statuses) so the worker slice needs no constraint migration.
|
||||
|
||||
Revision ID: 0049
|
||||
Revises: 0048
|
||||
Create Date: 2026-06-14
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0049"
|
||||
down_revision: Union[str, None] = "0048"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"external_link",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column(
|
||||
"post_id", sa.Integer(),
|
||||
sa.ForeignKey("post.id", ondelete="CASCADE"), nullable=False,
|
||||
),
|
||||
sa.Column(
|
||||
"artist_id", sa.Integer(),
|
||||
sa.ForeignKey("artist.id", ondelete="SET NULL"), nullable=True,
|
||||
),
|
||||
sa.Column("host", sa.String(length=16), nullable=False),
|
||||
sa.Column("url", sa.Text(), nullable=False),
|
||||
sa.Column("label", sa.Text(), nullable=True),
|
||||
sa.Column(
|
||||
"status", sa.String(length=16), nullable=False,
|
||||
server_default="pending",
|
||||
),
|
||||
sa.Column("attempts", sa.Integer(), nullable=False, server_default="0"),
|
||||
sa.Column("last_error", sa.Text(), nullable=True),
|
||||
sa.Column(
|
||||
"attachment_id", sa.Integer(),
|
||||
sa.ForeignKey("post_attachment.id", ondelete="SET NULL"),
|
||||
nullable=True,
|
||||
),
|
||||
sa.Column(
|
||||
"created_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column("completed_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column("duration_seconds", sa.Float(), nullable=True),
|
||||
sa.CheckConstraint(
|
||||
"host IN ('mega','gdrive','mediafire','dropbox','pixeldrain')",
|
||||
name="ck_external_link_host",
|
||||
),
|
||||
sa.CheckConstraint(
|
||||
"status IN ('pending','downloading','downloaded','failed',"
|
||||
"'skipped','dead')",
|
||||
name="ck_external_link_status",
|
||||
),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_external_link_post_id", "external_link", ["post_id"],
|
||||
)
|
||||
op.create_index(
|
||||
"ix_external_link_artist_id", "external_link", ["artist_id"],
|
||||
)
|
||||
op.create_index(
|
||||
"ix_external_link_status", "external_link", ["status"],
|
||||
)
|
||||
op.create_index(
|
||||
"uq_external_link_post_url", "external_link", ["post_id", "url"],
|
||||
unique=True,
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("uq_external_link_post_url", table_name="external_link")
|
||||
op.drop_index("ix_external_link_status", table_name="external_link")
|
||||
op.drop_index("ix_external_link_artist_id", table_name="external_link")
|
||||
op.drop_index("ix_external_link_post_id", table_name="external_link")
|
||||
op.drop_table("external_link")
|
||||
@@ -1,38 +0,0 @@
|
||||
"""import_settings: per-host enable toggles for external file-host downloads
|
||||
|
||||
Operator levers (#830): disable a single host (e.g. mega.nz when it's
|
||||
rate-limiting/banning) without touching the others. The worker reads these via
|
||||
getattr and defaults to enabled, so the toggles default TRUE (works out of the
|
||||
box, rule #26).
|
||||
|
||||
Revision ID: 0050
|
||||
Revises: 0049
|
||||
Create Date: 2026-06-14
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0050"
|
||||
down_revision: Union[str, None] = "0049"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
_HOSTS = ("mega", "gdrive", "mediafire", "dropbox", "pixeldrain")
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
for host in _HOSTS:
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
f"extdl_{host}_enabled", sa.Boolean(), nullable=False,
|
||||
server_default=sa.true(),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
for host in _HOSTS:
|
||||
op.drop_column("import_settings", f"extdl_{host}_enabled")
|
||||
@@ -1,38 +0,0 @@
|
||||
"""image_record: source_url + source_filehash (inline-image localization)
|
||||
|
||||
#830 Phase 2. To render a post body faithfully we serve LOCAL copies of inline
|
||||
images instead of hotlinking the public CDN. The join key between a body
|
||||
`<img src=CDN>` and the local file is the CDN's 32-hex filehash (the same
|
||||
identity extract_media dedups by). Persist it (indexed) plus the full source
|
||||
URL for provenance/debugging. Both NULL for filesystem-imported / pre-existing
|
||||
rows — those fall back to hotlinking until re-downloaded.
|
||||
|
||||
Revision ID: 0051
|
||||
Revises: 0050
|
||||
Create Date: 2026-06-14
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0051"
|
||||
down_revision: Union[str, None] = "0050"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column("image_record", sa.Column("source_url", sa.Text(), nullable=True))
|
||||
op.add_column(
|
||||
"image_record", sa.Column("source_filehash", sa.String(length=32), nullable=True)
|
||||
)
|
||||
op.create_index(
|
||||
"ix_image_record_source_filehash", "image_record", ["source_filehash"]
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_image_record_source_filehash", table_name="image_record")
|
||||
op.drop_column("image_record", "source_filehash")
|
||||
op.drop_column("image_record", "source_url")
|
||||
@@ -1,32 +0,0 @@
|
||||
"""image_record: duration_seconds (Tier-1 video near-dup key)
|
||||
|
||||
#871. Videos previously deduped on sha256 only (pHash is images-only), so a
|
||||
different encode/remux of the same video imported as a distinct record. Persist
|
||||
the container duration so the importer can treat same-artist videos with matching
|
||||
duration (+ aspect ratio) as the same content and dedup/supersede like images.
|
||||
NULL for images and for video rows imported before this column existed (a
|
||||
backfill re-probes those so they participate in dedup).
|
||||
|
||||
Revision ID: 0052
|
||||
Revises: 0051
|
||||
Create Date: 2026-06-16
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0052"
|
||||
down_revision: Union[str, None] = "0051"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"image_record", sa.Column("duration_seconds", sa.Float(), nullable=True)
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("image_record", "duration_seconds")
|
||||
@@ -1,49 +0,0 @@
|
||||
"""ml_settings: video tagging knobs (cadence sampling + noise floor)
|
||||
|
||||
#747. Video tag quality/perf: sample frames at a fixed cadence (interval) so a
|
||||
tag's frame-presence reflects real screen time, cap total frames so long videos
|
||||
stay bounded, and keep a tag only if it appears in >= min_tag_frames sampled
|
||||
frames. Operator-tunable via Settings → ML (replaces the VIDEO_ML_FRAMES env var).
|
||||
|
||||
Revision ID: 0053
|
||||
Revises: 0052
|
||||
Create Date: 2026-06-16
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0053"
|
||||
down_revision: Union[str, None] = "0052"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"video_frame_interval_seconds", sa.Float(), nullable=False,
|
||||
server_default="4.0",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"video_max_frames", sa.Integer(), nullable=False, server_default="64",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"video_min_tag_frames", sa.Integer(), nullable=False,
|
||||
server_default="3",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("ml_settings", "video_min_tag_frames")
|
||||
op.drop_column("ml_settings", "video_max_frames")
|
||||
op.drop_column("ml_settings", "video_frame_interval_seconds")
|
||||
@@ -1,82 +0,0 @@
|
||||
"""subscribestar_seen_media + subscribestar_failed_media: per-source ledgers
|
||||
|
||||
Revision ID: 0054
|
||||
Revises: 0053
|
||||
Create Date: 2026-06-17
|
||||
|
||||
SubscribeStar native ingester (phase 1 of the gallery-dl → native-core
|
||||
migration). Mirrors the Patreon ledger tables (0037/0038): a seen-ledger so
|
||||
routine walks skip already-ingested media (recovery bypasses it) and a
|
||||
dead-letter ledger so persistently-failing media stops re-burning backfill
|
||||
chunks. `filehash` is a CDN content hash when present, else a synthesized
|
||||
``<post_id>:<filename>`` key — hence String(128). UNIQUE (source_id, filehash)
|
||||
is the upsert key on each.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0054"
|
||||
down_revision: Union[str, None] = "0053"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"subscribestar_seen_media",
|
||||
sa.Column("id", sa.Integer, primary_key=True),
|
||||
sa.Column(
|
||||
"source_id",
|
||||
sa.Integer,
|
||||
sa.ForeignKey("source.id", ondelete="CASCADE"),
|
||||
nullable=False,
|
||||
index=True,
|
||||
),
|
||||
sa.Column("filehash", sa.String(128), nullable=False),
|
||||
sa.Column("post_id", sa.String(64), nullable=True),
|
||||
sa.Column(
|
||||
"seen_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.text("NOW()"),
|
||||
),
|
||||
sa.UniqueConstraint(
|
||||
"source_id", "filehash", name="uq_subscribestar_seen_media_source_id"
|
||||
),
|
||||
)
|
||||
op.create_table(
|
||||
"subscribestar_failed_media",
|
||||
sa.Column("id", sa.Integer, primary_key=True),
|
||||
sa.Column(
|
||||
"source_id",
|
||||
sa.Integer,
|
||||
sa.ForeignKey("source.id", ondelete="CASCADE"),
|
||||
nullable=False,
|
||||
index=True,
|
||||
),
|
||||
sa.Column("filehash", sa.String(128), nullable=False),
|
||||
sa.Column("attempts", sa.Integer, nullable=False, server_default="1"),
|
||||
sa.Column("last_error", sa.Text, nullable=True),
|
||||
sa.Column(
|
||||
"first_failed_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.text("NOW()"),
|
||||
),
|
||||
sa.Column(
|
||||
"last_failed_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.text("NOW()"),
|
||||
),
|
||||
sa.UniqueConstraint(
|
||||
"source_id", "filehash", name="uq_subscribestar_failed_media_source_id"
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_table("subscribestar_failed_media")
|
||||
op.drop_table("subscribestar_seen_media")
|
||||
@@ -1,55 +0,0 @@
|
||||
"""image_provenance: from_attachment_id (which archive an image was extracted from)
|
||||
|
||||
Milestone #87. When an image is pulled out of a .zip/.rar, record WHICH archive
|
||||
PostAttachment it came from, so the provenance UI can show the single archive a
|
||||
file lives inside instead of every attachment on the post. Nullable FK with
|
||||
ON DELETE SET NULL — a loose (non-archive) download leaves it NULL, and deleting
|
||||
the archive attachment forgets the linkage without destroying the (image, post)
|
||||
provenance edge. Existing rows are NULL until the reextract backfill stamps them.
|
||||
|
||||
Revision ID: 0055
|
||||
Revises: 0054
|
||||
Create Date: 2026-06-22
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0055"
|
||||
down_revision: Union[str, None] = "0054"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"image_provenance",
|
||||
sa.Column("from_attachment_id", sa.Integer(), nullable=True),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_image_provenance_from_attachment_id",
|
||||
"image_provenance",
|
||||
["from_attachment_id"],
|
||||
)
|
||||
op.create_foreign_key(
|
||||
"fk_image_provenance_from_attachment",
|
||||
"image_provenance",
|
||||
"post_attachment",
|
||||
["from_attachment_id"],
|
||||
["id"],
|
||||
ondelete="SET NULL",
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_constraint(
|
||||
"fk_image_provenance_from_attachment",
|
||||
"image_provenance",
|
||||
type_="foreignkey",
|
||||
)
|
||||
op.drop_index(
|
||||
"ix_image_provenance_from_attachment_id",
|
||||
table_name="image_provenance",
|
||||
)
|
||||
op.drop_column("image_provenance", "from_attachment_id")
|
||||
@@ -1,43 +0,0 @@
|
||||
"""tag_eval_run: persisted head-vs-centroid tagging eval runs (#1130)
|
||||
|
||||
Milestone #114 slice 1. A long ml-queue eval whose full report must SURVIVE
|
||||
navigation, so the run + report live in a row the admin card rehydrates from
|
||||
(mirrors library_audit_run). running -> ready / error.
|
||||
|
||||
Revision ID: 0056
|
||||
Revises: 0055
|
||||
Create Date: 2026-06-28
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from sqlalchemy.dialects.postgresql import JSONB
|
||||
|
||||
revision: str = "0056"
|
||||
down_revision: Union[str, None] = "0055"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"tag_eval_run",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column("params", JSONB(), nullable=False),
|
||||
sa.Column("status", sa.String(length=16), nullable=False, server_default="running"),
|
||||
sa.Column(
|
||||
"started_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column("report", JSONB(), nullable=True),
|
||||
sa.Column("error", sa.Text(), nullable=True),
|
||||
sa.Column("last_progress_at", sa.DateTime(timezone=True), nullable=True),
|
||||
)
|
||||
op.create_index("ix_tag_eval_run_status", "tag_eval_run", ["status"])
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_tag_eval_run_status", table_name="tag_eval_run")
|
||||
op.drop_table("tag_eval_run")
|
||||
@@ -1,40 +0,0 @@
|
||||
"""tag_positive_confirmation: operator-affirmed correct positives (#1130)
|
||||
|
||||
Mirror of tag_suggestion_rejection. "Keep" on a doubted positive records here so
|
||||
the eval's doubts list stops resurfacing confirmed-correct images every run.
|
||||
|
||||
Revision ID: 0057
|
||||
Revises: 0056
|
||||
Create Date: 2026-06-28
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0057"
|
||||
down_revision: Union[str, None] = "0056"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"tag_positive_confirmation",
|
||||
sa.Column(
|
||||
"image_record_id", sa.Integer(),
|
||||
sa.ForeignKey("image_record.id", ondelete="CASCADE"), primary_key=True,
|
||||
),
|
||||
sa.Column(
|
||||
"tag_id", sa.Integer(),
|
||||
sa.ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True, index=True,
|
||||
),
|
||||
sa.Column(
|
||||
"confirmed_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_table("tag_positive_confirmation")
|
||||
@@ -1,95 +0,0 @@
|
||||
"""tag_head + head_training_run: production heads that learn from tags (#114)
|
||||
|
||||
The eval (#1130) proved the frozen-embedding + trained-head spine; this lands its
|
||||
production form. tag_head stores one logistic-regression head per concept (the
|
||||
new suggestion source, replacing Camie + centroid); head_training_run tracks the
|
||||
batch that (re)trains them. Adds two head-training tunables to ml_settings.
|
||||
|
||||
Revision ID: 0058
|
||||
Revises: 0057
|
||||
Create Date: 2026-06-28
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from pgvector.sqlalchemy import Vector
|
||||
from sqlalchemy.dialects.postgresql import JSONB
|
||||
|
||||
revision: str = "0058"
|
||||
down_revision: Union[str, None] = "0057"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
_HEAD_DIM = 1152
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"tag_head",
|
||||
sa.Column(
|
||||
"tag_id", sa.Integer(),
|
||||
sa.ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True,
|
||||
),
|
||||
sa.Column("embedding_version", sa.String(length=128), nullable=False),
|
||||
sa.Column("weights", Vector(_HEAD_DIM), nullable=False),
|
||||
sa.Column("bias", sa.Float(), nullable=False),
|
||||
sa.Column("suggest_threshold", sa.Float(), nullable=False),
|
||||
sa.Column("auto_apply_threshold", sa.Float(), nullable=True),
|
||||
sa.Column("n_pos", sa.Integer(), nullable=False),
|
||||
sa.Column("n_neg", sa.Integer(), nullable=False),
|
||||
sa.Column("ap", sa.Float(), nullable=False),
|
||||
sa.Column("precision_cv", sa.Float(), nullable=False),
|
||||
sa.Column("recall", sa.Float(), nullable=False),
|
||||
sa.Column(
|
||||
"trained_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column("metrics", JSONB(), nullable=True),
|
||||
)
|
||||
|
||||
op.create_table(
|
||||
"head_training_run",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column("params", JSONB(), nullable=False),
|
||||
sa.Column(
|
||||
"status", sa.String(length=16), nullable=False,
|
||||
server_default="running",
|
||||
),
|
||||
sa.Column(
|
||||
"started_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column("n_trained", sa.Integer(), nullable=True),
|
||||
sa.Column("n_skipped", sa.Integer(), nullable=True),
|
||||
sa.Column("error", sa.Text(), nullable=True),
|
||||
sa.Column("last_progress_at", sa.DateTime(timezone=True), nullable=True),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_head_training_run_status", "head_training_run", ["status"],
|
||||
)
|
||||
|
||||
# Head-training tunables on the ml_settings singleton.
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"head_min_positives", sa.Integer(), nullable=False,
|
||||
server_default="8",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"head_auto_apply_precision", sa.Float(), nullable=False,
|
||||
server_default="0.97",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("ml_settings", "head_auto_apply_precision")
|
||||
op.drop_column("ml_settings", "head_min_positives")
|
||||
op.drop_index("ix_head_training_run_status", table_name="head_training_run")
|
||||
op.drop_table("head_training_run")
|
||||
op.drop_table("tag_head")
|
||||
@@ -1,70 +0,0 @@
|
||||
"""head_auto_apply_run + earned-auto-apply settings (#114)
|
||||
|
||||
A graduated head can apply its tag without a human, gated by a master switch +
|
||||
a support floor. head_auto_apply_run tracks each sweep / dry-run preview.
|
||||
|
||||
Revision ID: 0059
|
||||
Revises: 0058
|
||||
Create Date: 2026-06-29
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from sqlalchemy.dialects.postgresql import JSONB
|
||||
|
||||
revision: str = "0059"
|
||||
down_revision: Union[str, None] = "0058"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"head_auto_apply_run",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column(
|
||||
"dry_run", sa.Boolean(), nullable=False, server_default=sa.false()
|
||||
),
|
||||
sa.Column("params", JSONB(), nullable=False),
|
||||
sa.Column(
|
||||
"status", sa.String(length=16), nullable=False,
|
||||
server_default="running",
|
||||
),
|
||||
sa.Column(
|
||||
"started_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column("n_applied", sa.Integer(), nullable=True),
|
||||
sa.Column("report", JSONB(), nullable=True),
|
||||
sa.Column("error", sa.Text(), nullable=True),
|
||||
sa.Column("last_progress_at", sa.DateTime(timezone=True), nullable=True),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_head_auto_apply_run_status", "head_auto_apply_run", ["status"],
|
||||
)
|
||||
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"head_auto_apply_enabled", sa.Boolean(), nullable=False,
|
||||
server_default=sa.true(), # opt-out: on by default (operator-asked)
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"head_auto_apply_min_positives", sa.Integer(), nullable=False,
|
||||
server_default="30",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("ml_settings", "head_auto_apply_min_positives")
|
||||
op.drop_column("ml_settings", "head_auto_apply_enabled")
|
||||
op.drop_index(
|
||||
"ix_head_auto_apply_run_status", table_name="head_auto_apply_run"
|
||||
)
|
||||
op.drop_table("head_auto_apply_run")
|
||||
@@ -1,74 +0,0 @@
|
||||
"""head_metric + head_metrics_snapshot: auto-apply observability (#114)
|
||||
|
||||
Running misfire/under-fire counters per concept (captured at correction time,
|
||||
since image_tag.source is lost on delete) + a daily per-concept time-series so
|
||||
the operator can tune the precision target + support floor from real data.
|
||||
|
||||
Revision ID: 0060
|
||||
Revises: 0059
|
||||
Create Date: 2026-06-29
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0060"
|
||||
down_revision: Union[str, None] = "0059"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"head_metric",
|
||||
sa.Column(
|
||||
"tag_id", sa.Integer(),
|
||||
sa.ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True,
|
||||
),
|
||||
sa.Column("n_misfires", sa.Integer(), nullable=False, server_default="0"),
|
||||
sa.Column("n_underfires", sa.Integer(), nullable=False, server_default="0"),
|
||||
sa.Column(
|
||||
"updated_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
)
|
||||
|
||||
op.create_table(
|
||||
"head_metrics_snapshot",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column(
|
||||
"tag_id", sa.Integer(),
|
||||
sa.ForeignKey("tag.id", ondelete="CASCADE"),
|
||||
),
|
||||
sa.Column("name", sa.String(length=255), nullable=False),
|
||||
sa.Column(
|
||||
"snapshot_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column("n_auto_applied", sa.Integer(), nullable=False, server_default="0"),
|
||||
sa.Column("n_misfires", sa.Integer(), nullable=False, server_default="0"),
|
||||
sa.Column("n_underfires", sa.Integer(), nullable=False, server_default="0"),
|
||||
sa.Column("ap", sa.Float(), nullable=True),
|
||||
sa.Column("precision_cv", sa.Float(), nullable=True),
|
||||
sa.Column("recall", sa.Float(), nullable=True),
|
||||
sa.Column("n_pos", sa.Integer(), nullable=True),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_head_metrics_snapshot_tag_id", "head_metrics_snapshot", ["tag_id"],
|
||||
)
|
||||
op.create_index(
|
||||
"ix_head_metrics_snapshot_snapshot_at", "head_metrics_snapshot",
|
||||
["snapshot_at"],
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index(
|
||||
"ix_head_metrics_snapshot_snapshot_at", table_name="head_metrics_snapshot"
|
||||
)
|
||||
op.drop_index(
|
||||
"ix_head_metrics_snapshot_tag_id", table_name="head_metrics_snapshot"
|
||||
)
|
||||
op.drop_table("head_metrics_snapshot")
|
||||
op.drop_table("head_metric")
|
||||
@@ -1,59 +0,0 @@
|
||||
"""image_region: detected/proposed regions + their crop embeddings (#114)
|
||||
|
||||
Storage backbone of the crop pipeline. A region = normalized bbox + the crop's
|
||||
embedding (CCIP for face/figure → character id; SigLIP for concept regions →
|
||||
head bag-of-embeddings). Also serves as grounded-tag bbox provenance.
|
||||
|
||||
Revision ID: 0061
|
||||
Revises: 0060
|
||||
Create Date: 2026-06-29
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from pgvector.sqlalchemy import Vector
|
||||
|
||||
revision: str = "0061"
|
||||
down_revision: Union[str, None] = "0060"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
_CCIP_DIM = 768
|
||||
_SIGLIP_DIM = 1152
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"image_region",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column(
|
||||
"image_record_id", sa.Integer(),
|
||||
sa.ForeignKey("image_record.id", ondelete="CASCADE"), nullable=False,
|
||||
),
|
||||
sa.Column("kind", sa.String(length=16), nullable=False),
|
||||
# Video/animated: source frame timestamp (seconds); NULL for stills.
|
||||
sa.Column("frame_time", sa.Float(), nullable=True),
|
||||
sa.Column("rx", sa.Float(), nullable=False),
|
||||
sa.Column("ry", sa.Float(), nullable=False),
|
||||
sa.Column("rw", sa.Float(), nullable=False),
|
||||
sa.Column("rh", sa.Float(), nullable=False),
|
||||
sa.Column("score", sa.Float(), nullable=True),
|
||||
sa.Column("detector_version", sa.String(length=64), nullable=True),
|
||||
sa.Column("crop_version", sa.String(length=64), nullable=True),
|
||||
sa.Column("embedding_version", sa.String(length=128), nullable=True),
|
||||
sa.Column("ccip_embedding", Vector(_CCIP_DIM), nullable=True),
|
||||
sa.Column("siglip_embedding", Vector(_SIGLIP_DIM), nullable=True),
|
||||
sa.Column(
|
||||
"created_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_image_region_image_record_id", "image_region", ["image_record_id"],
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_image_region_image_record_id", table_name="image_region")
|
||||
op.drop_table("image_region")
|
||||
@@ -1,55 +0,0 @@
|
||||
"""gpu_job: the HTTP-leased GPU work queue for the desktop agent (#114)
|
||||
|
||||
The agent stays HTTP-only — the server enqueues per-(image, task) jobs here and
|
||||
the agent leases/submits over the web API; Redis/Postgres stay private.
|
||||
|
||||
Revision ID: 0062
|
||||
Revises: 0061
|
||||
Create Date: 2026-06-29
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0062"
|
||||
down_revision: Union[str, None] = "0061"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"gpu_job",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column(
|
||||
"image_record_id", sa.Integer(),
|
||||
sa.ForeignKey("image_record.id", ondelete="CASCADE"), nullable=False,
|
||||
),
|
||||
sa.Column("task", sa.String(length=32), nullable=False),
|
||||
sa.Column(
|
||||
"status", sa.String(length=16), nullable=False,
|
||||
server_default="pending",
|
||||
),
|
||||
sa.Column("lease_token", sa.String(length=64), nullable=True),
|
||||
sa.Column("leased_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column("lease_expires_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column("attempts", sa.Integer(), nullable=False, server_default="0"),
|
||||
sa.Column("error", sa.Text(), nullable=True),
|
||||
sa.Column(
|
||||
"created_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column(
|
||||
"updated_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
)
|
||||
op.create_index("ix_gpu_job_image_record_id", "gpu_job", ["image_record_id"])
|
||||
op.create_index("ix_gpu_job_status", "gpu_job", ["status"])
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_gpu_job_status", table_name="gpu_job")
|
||||
op.drop_index("ix_gpu_job_image_record_id", table_name="gpu_job")
|
||||
op.drop_table("gpu_job")
|
||||
@@ -1,33 +0,0 @@
|
||||
"""ml_settings.ccip_match_threshold — tunable CCIP character-match cut (#114)
|
||||
|
||||
The v1 matcher used a flat 0.75 cosine; live data showed that over-fires (a
|
||||
high-reference character matched a scatter of images). 0.85 keeps the confident
|
||||
single-character matches and drops the noise. Tunable from the GPU agent card.
|
||||
|
||||
Revision ID: 0063
|
||||
Revises: 0062
|
||||
Create Date: 2026-06-29
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0063"
|
||||
down_revision: Union[str, None] = "0062"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"ccip_match_threshold", sa.Float(), nullable=False,
|
||||
server_default="0.85",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("ml_settings", "ccip_match_threshold")
|
||||
@@ -1,42 +0,0 @@
|
||||
"""ml_settings: CCIP auto-apply switch + threshold (#114)
|
||||
|
||||
Confident CCIP character matches auto-tag (source='ccip_auto') on a daily sweep,
|
||||
so identity tags keep flowing without pressing a button. ON by default (opt-out,
|
||||
like head auto-apply); the high threshold (0.92, above the 0.85 suggest cut) +
|
||||
single-character references keep it safe, and every auto-tag is reversible.
|
||||
|
||||
Revision ID: 0064
|
||||
Revises: 0063
|
||||
Create Date: 2026-06-30
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0064"
|
||||
down_revision: Union[str, None] = "0063"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"ccip_auto_apply_enabled", sa.Boolean(), nullable=False,
|
||||
server_default=sa.true(),
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"ccip_auto_apply_threshold", sa.Float(), nullable=False,
|
||||
server_default="0.92",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("ml_settings", "ccip_auto_apply_threshold")
|
||||
op.drop_column("ml_settings", "ccip_auto_apply_enabled")
|
||||
@@ -1,35 +0,0 @@
|
||||
"""ml_settings: embedder_model_name (#1190 operator model swap)
|
||||
|
||||
The embedder MODEL VERSION was already a setting (and stamps image_record.
|
||||
siglip_model_version); the HF model NAME was env-only, so an operator couldn't
|
||||
actually point the pipeline at a different embedder. Storing the name as a
|
||||
setting makes the model an operator choice: set name + version → re-embed (the
|
||||
GPU agent) → retrain heads. Default = the current SigLIP so400m.
|
||||
|
||||
Revision ID: 0065
|
||||
Revises: 0064
|
||||
Create Date: 2026-06-30
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0065"
|
||||
down_revision: Union[str, None] = "0064"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"embedder_model_name", sa.String(length=128), nullable=False,
|
||||
server_default="google/siglip-so400m-patch14-384",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("ml_settings", "embedder_model_name")
|
||||
@@ -1,57 +0,0 @@
|
||||
"""drop the dead per-tag centroid subsystem (#1189 cleanup)
|
||||
|
||||
The v2 pivot replaced per-tag SigLIP centroids with learned heads + CCIP.
|
||||
Nothing read the centroids anymore — they were recomputed (on merge + a daily
|
||||
beat) but never consumed for suggestions or auto-apply. Remove the storage +
|
||||
its two now-unused settings columns. (The recompute tasks, beat, endpoint,
|
||||
service, and UI card are removed in the same change.)
|
||||
|
||||
Revision ID: 0066
|
||||
Revises: 0065
|
||||
Create Date: 2026-06-30
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0066"
|
||||
down_revision: Union[str, None] = "0065"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.drop_table("tag_reference_embedding")
|
||||
op.drop_column("ml_settings", "centroid_similarity_threshold")
|
||||
op.drop_column("ml_settings", "min_reference_images")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"min_reference_images", sa.Integer(), nullable=False,
|
||||
server_default="5",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"centroid_similarity_threshold", sa.Float(), nullable=False,
|
||||
server_default="0.55",
|
||||
),
|
||||
)
|
||||
op.create_table(
|
||||
"tag_reference_embedding",
|
||||
sa.Column("tag_id", sa.Integer(), nullable=False),
|
||||
sa.Column("embedding", sa.LargeBinary(), nullable=False),
|
||||
sa.Column("reference_count", sa.Integer(), nullable=False),
|
||||
sa.Column("model_version", sa.String(length=128), nullable=False),
|
||||
sa.Column(
|
||||
"updated_at", sa.DateTime(timezone=True),
|
||||
server_default=sa.func.now(), nullable=False,
|
||||
),
|
||||
sa.ForeignKeyConstraint(["tag_id"], ["tag.id"], ondelete="CASCADE"),
|
||||
sa.PrimaryKeyConstraint("tag_id"),
|
||||
)
|
||||
@@ -1,66 +0,0 @@
|
||||
"""retire the Camie tagger + allowlist bulk-apply (#1189)
|
||||
|
||||
The v2 pivot made heads + CCIP the tag source and head auto-apply the earned
|
||||
propagation. The Camie tagger ran only to feed the allowlist bulk-apply (its
|
||||
predictions had no other consumer), and the allowlist was a second, un-earned
|
||||
auto-apply path parallel to heads. Both are retired — drop their storage.
|
||||
|
||||
(image_prediction = Camie's per-image predictions; tag_allowlist = the bulk-
|
||||
apply allowlist. Nothing references INTO these tables, so the drop is clean.)
|
||||
|
||||
Revision ID: 0067
|
||||
Revises: 0066
|
||||
Create Date: 2026-06-30
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0067"
|
||||
down_revision: Union[str, None] = "0066"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.drop_table("image_prediction")
|
||||
op.drop_table("tag_allowlist")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.create_table(
|
||||
"tag_allowlist",
|
||||
sa.Column("tag_id", sa.Integer(), nullable=False),
|
||||
sa.Column(
|
||||
"min_confidence", sa.Float(), nullable=False, server_default="0.9"
|
||||
),
|
||||
sa.Column(
|
||||
"created_at", sa.DateTime(timezone=True),
|
||||
server_default=sa.func.now(), nullable=False,
|
||||
),
|
||||
sa.ForeignKeyConstraint(["tag_id"], ["tag.id"], ondelete="CASCADE"),
|
||||
sa.PrimaryKeyConstraint("tag_id"),
|
||||
sa.CheckConstraint(
|
||||
"min_confidence >= 0 AND min_confidence <= 1",
|
||||
name="ck_tag_allowlist_confidence_range",
|
||||
),
|
||||
)
|
||||
op.create_table(
|
||||
"image_prediction",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column("image_record_id", sa.Integer(), nullable=False),
|
||||
sa.Column("raw_name", sa.String(length=255), nullable=False),
|
||||
sa.Column("category", sa.String(length=32), nullable=False),
|
||||
sa.Column("score", sa.Float(), nullable=False),
|
||||
sa.ForeignKeyConstraint(
|
||||
["image_record_id"], ["image_record.id"], ondelete="CASCADE"
|
||||
),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_image_prediction_image", "image_prediction", ["image_record_id"]
|
||||
)
|
||||
op.create_index(
|
||||
"ix_image_prediction_name_score", "image_prediction",
|
||||
["raw_name", "score"],
|
||||
)
|
||||
@@ -1,80 +0,0 @@
|
||||
"""drop dead tagger/suggestion settings + columns left after Camie retirement (#1199)
|
||||
|
||||
Hygiene follow-up to #1189. These were left inert to bound that change; nothing
|
||||
reads them now:
|
||||
- ml_settings: tagger_store_floor + tagger_model_version (only the deleted Camie
|
||||
tagger used them), suggestion_threshold_character/general (already dead pre-
|
||||
retirement — scoring uses per-head thresholds), video_min_tag_frames (only the
|
||||
deleted video-prediction aggregator used it).
|
||||
- image_record: tagger_model_version (no writer now), centroid_scores (long-dead
|
||||
JSON cache, no reader).
|
||||
|
||||
Revision ID: 0068
|
||||
Revises: 0067
|
||||
Create Date: 2026-06-30
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0068"
|
||||
down_revision: Union[str, None] = "0067"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.drop_column("ml_settings", "suggestion_threshold_character")
|
||||
op.drop_column("ml_settings", "suggestion_threshold_general")
|
||||
op.drop_column("ml_settings", "tagger_store_floor")
|
||||
op.drop_column("ml_settings", "video_min_tag_frames")
|
||||
op.drop_column("ml_settings", "tagger_model_version")
|
||||
op.drop_column("image_record", "tagger_model_version")
|
||||
op.drop_column("image_record", "centroid_scores")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.add_column(
|
||||
"image_record",
|
||||
sa.Column("centroid_scores", sa.JSON(), nullable=True),
|
||||
)
|
||||
op.add_column(
|
||||
"image_record",
|
||||
sa.Column("tagger_model_version", sa.String(length=128), nullable=True),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"tagger_model_version", sa.String(length=128), nullable=False,
|
||||
server_default="camie-tagger-v2",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"video_min_tag_frames", sa.Integer(), nullable=False,
|
||||
server_default="3",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"tagger_store_floor", sa.Float(), nullable=False,
|
||||
server_default="0.7",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"suggestion_threshold_general", sa.Float(), nullable=False,
|
||||
server_default="0.7",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"suggestion_threshold_character", sa.Float(), nullable=False,
|
||||
server_default="0.7",
|
||||
),
|
||||
)
|
||||
@@ -1,51 +0,0 @@
|
||||
"""default the embedder to SigLIP 2 — for FRESH installs only (#1203)
|
||||
|
||||
Make SigLIP 2 (so400m, 512px; a 1152-d drop-in) the default embedder. New
|
||||
installs start on it. An EXISTING library is NOT touched: flipping its stored
|
||||
embedder version would mark every embedding stale (the scorer is version-gated)
|
||||
and kill suggestions until a full re-embed+retrain — so an existing instance
|
||||
switches deliberately via Settings → GPU agent → Embedding model → Re-embed →
|
||||
Retrain. We detect "fresh" by the absence of any embedded image.
|
||||
|
||||
Revision ID: 0069
|
||||
Revises: 0068
|
||||
Create Date: 2026-06-30
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0069"
|
||||
down_revision: Union[str, None] = "0068"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
_NEW_NAME = "google/siglip2-so400m-patch16-512"
|
||||
_NEW_VERSION = "siglip2-so400m-patch16-512"
|
||||
_OLD_NAME = "google/siglip-so400m-patch14-384"
|
||||
_OLD_VERSION = "siglip-so400m-patch14-384"
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
# Fresh install (nothing embedded yet) → adopt SigLIP 2.
|
||||
op.execute(
|
||||
f"""
|
||||
UPDATE ml_settings SET
|
||||
embedder_model_name = '{_NEW_NAME}',
|
||||
embedder_model_version = '{_NEW_VERSION}'
|
||||
WHERE NOT EXISTS (
|
||||
SELECT 1 FROM image_record WHERE siglip_embedding IS NOT NULL
|
||||
)
|
||||
"""
|
||||
)
|
||||
op.alter_column("ml_settings", "embedder_model_name", server_default=_NEW_NAME)
|
||||
op.alter_column(
|
||||
"ml_settings", "embedder_model_version", server_default=_NEW_VERSION
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.alter_column("ml_settings", "embedder_model_name", server_default=_OLD_NAME)
|
||||
op.alter_column(
|
||||
"ml_settings", "embedder_model_version", server_default=_OLD_VERSION
|
||||
)
|
||||
@@ -1,44 +0,0 @@
|
||||
"""partial indexes so GPU-job leasing stays O(batch), not O(completed)
|
||||
|
||||
The lease claims the lowest-id pending (or expired-leased) jobs. With only a
|
||||
plain `status` index, `... ORDER BY id LIMIT n` walked the primary-key index from
|
||||
the start, skipping the entire prefix of already-done/error rows before reaching
|
||||
pending ones — so leasing slowed to a crawl as `done` piled up (the whole reason
|
||||
throughput fell off a cliff mid-run and /status stalled). Two partial indexes fix
|
||||
it: the pending one is id-ordered so the hot path reads just the first n entries,
|
||||
and the leased-expiry one keeps the crash-recovery reclaim + the orphan sweep
|
||||
cheap. They cover only the small live slice of the table, so they stay tiny even
|
||||
as the done/error history grows to millions.
|
||||
|
||||
Revision ID: 0070
|
||||
Revises: 0069
|
||||
Create Date: 2026-06-30
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0070"
|
||||
down_revision: Union[str, None] = "0069"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
# Hot path: lowest-id pending jobs. Index on id, restricted to pending, so
|
||||
# `WHERE status='pending' ORDER BY id LIMIT n` is a short index-order scan.
|
||||
op.create_index(
|
||||
"ix_gpu_job_pending", "gpu_job", ["id"],
|
||||
postgresql_where=sa.text("status = 'pending'"),
|
||||
)
|
||||
# Crash-recovery: expired leases, for the lease backstop + recover_orphaned.
|
||||
op.create_index(
|
||||
"ix_gpu_job_leased_expires", "gpu_job", ["lease_expires_at"],
|
||||
postgresql_where=sa.text("status = 'leased'"),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_gpu_job_leased_expires", table_name="gpu_job")
|
||||
op.drop_index("ix_gpu_job_pending", table_name="gpu_job")
|
||||
@@ -1,80 +0,0 @@
|
||||
"""image_record.earliest_post_date: original-publish gallery sort key + index
|
||||
|
||||
Revision ID: 0071
|
||||
Revises: 0070
|
||||
Create Date: 2026-07-01
|
||||
|
||||
effective_date (0035) keys off the PRIMARY post — which is often the repost /
|
||||
download the file actually came from — and falls back to created_at, so the
|
||||
gallery's default order surfaces download dates rather than when content was
|
||||
first posted (operator-flagged 2026-07-01). Materialize a second sort key,
|
||||
earliest_post_date = MIN(post_date) across ALL of an image's provenance posts
|
||||
(every post it appears in), falling back to created_at only when no linked post
|
||||
carries a date. Indexed (DESC, id DESC) so the "post date" gallery sort is an
|
||||
index range scan just like effective_date.
|
||||
|
||||
Backfill mirrors 0035: created_at baseline, then override with the MIN over
|
||||
image_provenance ⋈ post. New rows get the created_at-equivalent server default;
|
||||
services/importer.py recomputes it whenever a dated post is linked.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0071"
|
||||
down_revision: Union[str, None] = "0070"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
# Add nullable first so the backfill can populate before NOT NULL.
|
||||
op.add_column(
|
||||
"image_record",
|
||||
sa.Column("earliest_post_date", sa.DateTime(timezone=True), nullable=True),
|
||||
)
|
||||
# Baseline: download date. Set-based (no per-row binds) → immune to the
|
||||
# 65535 bind-parameter ceiling regardless of library size.
|
||||
op.execute(
|
||||
"""
|
||||
UPDATE image_record
|
||||
SET earliest_post_date = created_at
|
||||
"""
|
||||
)
|
||||
# Override with the earliest post_date across EVERY post the image appears
|
||||
# in (image_provenance is the many-to-many edge; ignore posts with no date).
|
||||
op.execute(
|
||||
"""
|
||||
UPDATE image_record AS ir
|
||||
SET earliest_post_date = sub.min_date
|
||||
FROM (
|
||||
SELECT ip.image_record_id AS iid, MIN(p.post_date) AS min_date
|
||||
FROM image_provenance AS ip
|
||||
JOIN post AS p ON p.id = ip.post_id
|
||||
WHERE p.post_date IS NOT NULL
|
||||
GROUP BY ip.image_record_id
|
||||
) AS sub
|
||||
WHERE ir.id = sub.iid
|
||||
"""
|
||||
)
|
||||
op.alter_column(
|
||||
"image_record",
|
||||
"earliest_post_date",
|
||||
nullable=False,
|
||||
server_default=sa.text("now()"),
|
||||
)
|
||||
# DESC/DESC matches the gallery's ORDER BY earliest_post_date DESC, id DESC
|
||||
# so the "post date" scroll is a forward index scan; raw SQL because
|
||||
# alembic's column list doesn't express per-column DESC cleanly.
|
||||
op.execute(
|
||||
"CREATE INDEX ix_image_record_earliest_post_date "
|
||||
"ON image_record (earliest_post_date DESC, id DESC)"
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index(
|
||||
"ix_image_record_earliest_post_date", table_name="image_record"
|
||||
)
|
||||
op.drop_column("image_record", "earliest_post_date")
|
||||
@@ -1,32 +0,0 @@
|
||||
"""gpu_job.triage_status — the probe's verdict on an errored job's FILE
|
||||
|
||||
Failure triage (#125): a periodic sweep probes each errored image's file
|
||||
(sha256 + decode, verify_integrity's machinery) exactly once and stores the
|
||||
verdict here — 'defect' (the file is bad: recovery material, excluded from
|
||||
/retry_errors) or 'file_ok' (failure was operational, safe to retry). NULL
|
||||
means not yet probed; selecting on NULL is what makes the sweep resumable.
|
||||
No index: the errored slice the sweep scans is tiny by design (tombstones).
|
||||
|
||||
Revision ID: 0072
|
||||
Revises: 0071
|
||||
Create Date: 2026-07-02
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0072"
|
||||
down_revision: Union[str, None] = "0071"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"gpu_job", sa.Column("triage_status", sa.String(16), nullable=True)
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("gpu_job", "triage_status")
|
||||
@@ -1,46 +0,0 @@
|
||||
"""drop tag_eval_run — the head-vs-centroid eval harness is retired
|
||||
|
||||
The eval (#1130) existed to prove the heads tagging spine on the operator's own
|
||||
data. It did; the operator accepted the system and retired the harness
|
||||
(2026-07-02) — card, API, task, model and this table all go. The eval's data
|
||||
loaders + metric helpers live on in services/ml/training_data.py, where the
|
||||
production heads trainer uses them nightly.
|
||||
|
||||
Revision ID: 0073
|
||||
Revises: 0072
|
||||
Create Date: 2026-07-02
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from sqlalchemy.dialects import postgresql
|
||||
|
||||
revision: str = "0073"
|
||||
down_revision: Union[str, None] = "0072"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.drop_index("ix_tag_eval_run_status", table_name="tag_eval_run")
|
||||
op.drop_table("tag_eval_run")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# Recreates the shape from 0056 (data is not restorable).
|
||||
op.create_table(
|
||||
"tag_eval_run",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column("params", postgresql.JSONB(), nullable=False),
|
||||
sa.Column("status", sa.String(length=16), nullable=False,
|
||||
server_default="running"),
|
||||
sa.Column("started_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now()),
|
||||
sa.Column("finished_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column("report", postgresql.JSONB(), nullable=True),
|
||||
sa.Column("error", sa.Text(), nullable=True),
|
||||
sa.Column("last_progress_at", sa.DateTime(timezone=True),
|
||||
nullable=True),
|
||||
)
|
||||
op.create_index("ix_tag_eval_run_status", "tag_eval_run", ["status"])
|
||||
@@ -1,35 +0,0 @@
|
||||
"""ml_settings.cpu_embed_enabled — the CPU embed fallback becomes a switch
|
||||
|
||||
B3 (operator 2026-07-02): the ml-worker's only processing role is the CPU
|
||||
whole-image embed for stacks without a GPU agent. ON by default (a fresh
|
||||
install works agent-less); agent-equipped stacks that drop the ml-worker
|
||||
container turn it off so import hooks stop queueing embed work into a queue
|
||||
nothing consumes — the daily GPU 'embed' backfill covers those images.
|
||||
|
||||
Revision ID: 0074
|
||||
Revises: 0073
|
||||
Create Date: 2026-07-02
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0074"
|
||||
down_revision: Union[str, None] = "0073"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"cpu_embed_enabled", sa.Boolean(), nullable=False,
|
||||
server_default=sa.true(),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("ml_settings", "cpu_embed_enabled")
|
||||
@@ -1,60 +0,0 @@
|
||||
"""tag.is_system + seed the three hygiene system tags
|
||||
|
||||
Training hygiene (operator 2026-07-03, milestone #128): rough WIPs tagged as a
|
||||
character poison that character's head and CCIP references; banners/editor
|
||||
screenshots pollute whole-image similarity. The fix keys on SYSTEM tags the
|
||||
product ships — not operator configuration — so the seed lives here.
|
||||
|
||||
Seeding ADOPTS an existing same-(name, kind=general) tag (case-insensitive,
|
||||
matching TagService.rename's collision stance) instead of inserting a
|
||||
duplicate, so an operator who already tagged `wip` keeps their applications.
|
||||
|
||||
Revision ID: 0075
|
||||
Revises: 0074
|
||||
Create Date: 2026-07-03
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0075"
|
||||
down_revision: Union[str, None] = "0074"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
SYSTEM_TAG_NAMES = ("wip", "banner", "editor screenshot")
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"tag",
|
||||
sa.Column(
|
||||
"is_system", sa.Boolean(), nullable=False,
|
||||
server_default=sa.false(),
|
||||
),
|
||||
)
|
||||
conn = op.get_bind()
|
||||
for name in SYSTEM_TAG_NAMES:
|
||||
adopted = conn.execute(
|
||||
sa.text(
|
||||
"UPDATE tag SET is_system = true "
|
||||
"WHERE lower(name) = lower(:name) AND kind = 'general'"
|
||||
),
|
||||
{"name": name},
|
||||
)
|
||||
if adopted.rowcount == 0:
|
||||
conn.execute(
|
||||
sa.text(
|
||||
"INSERT INTO tag (name, kind, is_system) "
|
||||
"VALUES (:name, 'general', true)"
|
||||
),
|
||||
{"name": name},
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# The seeded rows survive as ordinary general tags — dropping the flag is
|
||||
# enough to disarm the mechanism, and deleting rows would orphan any
|
||||
# operator applications made while the flag existed.
|
||||
op.drop_column("tag", "is_system")
|
||||
@@ -1,82 +0,0 @@
|
||||
"""pixiv_seen_media + pixiv_failed_media: per-source ledgers
|
||||
|
||||
Revision ID: 0076
|
||||
Revises: 0075
|
||||
Create Date: 2026-07-03
|
||||
|
||||
Pixiv native ingester (milestone #129, gallery-dl → native-core migration).
|
||||
Mirrors the Patreon (0037/0038) and SubscribeStar (0054) ledger tables: a
|
||||
seen-ledger so routine walks skip already-ingested media (recovery bypasses
|
||||
it) and a dead-letter ledger so persistently-failing media stops re-burning
|
||||
backfill chunks. Pixiv URLs carry no content hash, so `filehash` is always the
|
||||
synthesized ``<illust_id>:p<num>`` / ``<illust_id>:ugoira`` key — String(128)
|
||||
matches the siblings. UNIQUE (source_id, filehash) is the upsert key on each.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0076"
|
||||
down_revision: Union[str, None] = "0075"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"pixiv_seen_media",
|
||||
sa.Column("id", sa.Integer, primary_key=True),
|
||||
sa.Column(
|
||||
"source_id",
|
||||
sa.Integer,
|
||||
sa.ForeignKey("source.id", ondelete="CASCADE"),
|
||||
nullable=False,
|
||||
index=True,
|
||||
),
|
||||
sa.Column("filehash", sa.String(128), nullable=False),
|
||||
sa.Column("post_id", sa.String(64), nullable=True),
|
||||
sa.Column(
|
||||
"seen_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.text("NOW()"),
|
||||
),
|
||||
sa.UniqueConstraint(
|
||||
"source_id", "filehash", name="uq_pixiv_seen_media_source_id"
|
||||
),
|
||||
)
|
||||
op.create_table(
|
||||
"pixiv_failed_media",
|
||||
sa.Column("id", sa.Integer, primary_key=True),
|
||||
sa.Column(
|
||||
"source_id",
|
||||
sa.Integer,
|
||||
sa.ForeignKey("source.id", ondelete="CASCADE"),
|
||||
nullable=False,
|
||||
index=True,
|
||||
),
|
||||
sa.Column("filehash", sa.String(128), nullable=False),
|
||||
sa.Column("attempts", sa.Integer, nullable=False, server_default="1"),
|
||||
sa.Column("last_error", sa.Text, nullable=True),
|
||||
sa.Column(
|
||||
"first_failed_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.text("NOW()"),
|
||||
),
|
||||
sa.Column(
|
||||
"last_failed_at",
|
||||
sa.DateTime(timezone=True),
|
||||
nullable=False,
|
||||
server_default=sa.text("NOW()"),
|
||||
),
|
||||
sa.UniqueConstraint(
|
||||
"source_id", "filehash", name="uq_pixiv_failed_media_source_id"
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_table("pixiv_failed_media")
|
||||
op.drop_table("pixiv_seen_media")
|
||||
@@ -1,32 +0,0 @@
|
||||
"""drop uq_artist_name — decouple display name from identity/storage
|
||||
|
||||
Revision ID: 0077
|
||||
Revises: 0076
|
||||
Create Date: 2026-07-04
|
||||
|
||||
Artist model fragility fix (milestone #130). One `slug` column was doing
|
||||
identity + storage-path + display, and BOTH `name` and `slug` were UNIQUE, so
|
||||
the display name couldn't be edited freely and two genuinely different creators
|
||||
collided. Decouple: `slug` stays the immutable, unique storage/identity key (the
|
||||
on-disk path component — untouched here); `name` becomes freely editable, NON-
|
||||
unique display text. This migration only drops the `uq_artist_name` constraint;
|
||||
no data moves and no path changes.
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0077"
|
||||
down_revision: Union[str, None] = "0076"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.drop_constraint("uq_artist_name", "artist", type_="unique")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
# Re-adding the UNIQUE would fail if duplicate names now exist; callers that
|
||||
# need to reverse this must dedupe names first.
|
||||
op.create_unique_constraint("uq_artist_name", "artist", ["name"])
|
||||
@@ -1,83 +0,0 @@
|
||||
"""ml_settings crop-proposer / detector config (#134)
|
||||
|
||||
Move the WHERE-to-crop detector config (per-proposer enable + weights + conf,
|
||||
plus caps + dedupe IoU) into the DB so it's UI-tunable and announced to the GPU
|
||||
agent in the lease (like the embedder model) — no restart, agent env is now
|
||||
bootstrap-only. All server_defaults are the working values so existing rows +
|
||||
fresh installs crop out-of-the-box with all three proposers ON.
|
||||
|
||||
Revision ID: 0078
|
||||
Revises: 0077
|
||||
Create Date: 2026-07-05
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0078"
|
||||
down_revision: Union[str, None] = "0077"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
_ANATOMY_DEFAULT = (
|
||||
"https://github.com/aperveyev/booru_yolo/raw/main/models/yolov11m_aa22.pt"
|
||||
)
|
||||
_PANEL_DEFAULT = "mosesb/best-comic-panel-detection::best.pt"
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column("ml_settings", sa.Column(
|
||||
"detector_person_enabled", sa.Boolean(), nullable=False,
|
||||
server_default=sa.true()))
|
||||
op.add_column("ml_settings", sa.Column(
|
||||
"detector_person_weights", sa.String(512), nullable=False,
|
||||
server_default="yolo11n.pt"))
|
||||
op.add_column("ml_settings", sa.Column(
|
||||
"detector_person_conf", sa.Float(), nullable=False,
|
||||
server_default=sa.text("0.35")))
|
||||
op.add_column("ml_settings", sa.Column(
|
||||
"detector_anatomy_enabled", sa.Boolean(), nullable=False,
|
||||
server_default=sa.true()))
|
||||
op.add_column("ml_settings", sa.Column(
|
||||
"detector_anatomy_weights", sa.String(512), nullable=False,
|
||||
server_default=_ANATOMY_DEFAULT))
|
||||
op.add_column("ml_settings", sa.Column(
|
||||
"detector_anatomy_conf", sa.Float(), nullable=False,
|
||||
server_default=sa.text("0.30")))
|
||||
op.add_column("ml_settings", sa.Column(
|
||||
"detector_panel_enabled", sa.Boolean(), nullable=False,
|
||||
server_default=sa.true()))
|
||||
op.add_column("ml_settings", sa.Column(
|
||||
"detector_panel_weights", sa.String(512), nullable=False,
|
||||
server_default=_PANEL_DEFAULT))
|
||||
op.add_column("ml_settings", sa.Column(
|
||||
"detector_panel_conf", sa.Float(), nullable=False,
|
||||
server_default=sa.text("0.30")))
|
||||
op.add_column("ml_settings", sa.Column(
|
||||
"detector_max_figures", sa.Integer(), nullable=False,
|
||||
server_default=sa.text("8")))
|
||||
op.add_column("ml_settings", sa.Column(
|
||||
"detector_max_components", sa.Integer(), nullable=False,
|
||||
server_default=sa.text("8")))
|
||||
op.add_column("ml_settings", sa.Column(
|
||||
"detector_max_panels", sa.Integer(), nullable=False,
|
||||
server_default=sa.text("8")))
|
||||
op.add_column("ml_settings", sa.Column(
|
||||
"detector_max_regions", sa.Integer(), nullable=False,
|
||||
server_default=sa.text("128")))
|
||||
op.add_column("ml_settings", sa.Column(
|
||||
"detector_dedupe_iou", sa.Float(), nullable=False,
|
||||
server_default=sa.text("0.85")))
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
for col in (
|
||||
"detector_person_enabled", "detector_person_weights", "detector_person_conf",
|
||||
"detector_anatomy_enabled", "detector_anatomy_weights", "detector_anatomy_conf",
|
||||
"detector_panel_enabled", "detector_panel_weights", "detector_panel_conf",
|
||||
"detector_max_figures", "detector_max_components", "detector_max_panels",
|
||||
"detector_max_regions", "detector_dedupe_iou",
|
||||
):
|
||||
op.drop_column("ml_settings", col)
|
||||
@@ -1,77 +0,0 @@
|
||||
"""character prototype store (#1317) — precomputed, incremental CCIP references
|
||||
|
||||
New tables character_prototype + ccip_prototype_state, plus MLSettings columns
|
||||
ccip_ref_signature (cheap global change gate) + ccip_prototype_cap (per-character
|
||||
reference cap). The reference set the CCIP matcher uses becomes a precomputed
|
||||
artifact refreshed incrementally off the request path. See milestone 138 /
|
||||
backend.app.services.ml.character_prototypes.
|
||||
|
||||
Revision ID: 0079
|
||||
Revises: 0078
|
||||
Create Date: 2026-07-06
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from pgvector.sqlalchemy import Vector
|
||||
|
||||
revision: str = "0079"
|
||||
down_revision: Union[str, None] = "0078"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
# Matches models.image_region.CCIP_DIM (the CCIP figure-embedding width).
|
||||
_CCIP_DIM = 768
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"character_prototype",
|
||||
sa.Column("id", sa.Integer(), primary_key=True),
|
||||
sa.Column(
|
||||
"tag_id", sa.Integer(),
|
||||
sa.ForeignKey("tag.id", ondelete="CASCADE"), nullable=False,
|
||||
),
|
||||
sa.Column("ccip_embedding", Vector(_CCIP_DIM), nullable=False),
|
||||
sa.Column(
|
||||
"region_id", sa.Integer(),
|
||||
sa.ForeignKey("image_region.id", ondelete="SET NULL"), nullable=True,
|
||||
),
|
||||
)
|
||||
op.create_index(
|
||||
"ix_character_prototype_tag_id", "character_prototype", ["tag_id"]
|
||||
)
|
||||
op.create_table(
|
||||
"ccip_prototype_state",
|
||||
sa.Column(
|
||||
"tag_id", sa.Integer(),
|
||||
sa.ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True,
|
||||
),
|
||||
sa.Column("fingerprint", sa.String(64), nullable=False),
|
||||
sa.Column(
|
||||
"updated_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column("ccip_ref_signature", sa.String(128), nullable=True),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"ccip_prototype_cap", sa.Integer(), nullable=False,
|
||||
server_default=sa.text("64"),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("ml_settings", "ccip_prototype_cap")
|
||||
op.drop_column("ml_settings", "ccip_ref_signature")
|
||||
op.drop_table("ccip_prototype_state")
|
||||
op.drop_index(
|
||||
"ix_character_prototype_tag_id", table_name="character_prototype"
|
||||
)
|
||||
op.drop_table("character_prototype")
|
||||
@@ -1,31 +0,0 @@
|
||||
"""tag_head.train_fingerprint (#1317 phase 2) — incremental head retraining
|
||||
|
||||
A per-head training-data fingerprint (positive + rejection count/latest-timestamp)
|
||||
so a manual Retrain refits only the tags whose data changed; the nightly run
|
||||
ignores it (full reconcile). Nullable — a NULL fingerprint (existing heads) forces
|
||||
a refit on the first incremental run, then it's stamped.
|
||||
|
||||
Revision ID: 0080
|
||||
Revises: 0079
|
||||
Create Date: 2026-07-06
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0080"
|
||||
down_revision: Union[str, None] = "0079"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"tag_head",
|
||||
sa.Column("train_fingerprint", sa.String(128), nullable=True),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("tag_head", "train_fingerprint")
|
||||
@@ -1,43 +0,0 @@
|
||||
"""stricter auto-apply defaults (milestone 139) — cut auto-apply misfires
|
||||
|
||||
head_auto_apply_min_positives 30→50 and ccip_auto_apply_threshold 0.92→0.95
|
||||
(operator-asked 2026-07-06). The head graduation precision bar stays 0.97 — the
|
||||
operator confirmed the general-tag confidence was already well tuned; only the
|
||||
support floor + the CCIP match confidence are raised. The model defaults change
|
||||
for fresh installs; here we bump the existing singleton row IFF it is still at
|
||||
the previous default, so a deliberate operator change is NOT clobbered.
|
||||
|
||||
Revision ID: 0081
|
||||
Revises: 0080
|
||||
Create Date: 2026-07-06
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0081"
|
||||
down_revision: Union[str, None] = "0080"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.execute(
|
||||
"UPDATE ml_settings SET head_auto_apply_min_positives = 50 "
|
||||
"WHERE head_auto_apply_min_positives = 30"
|
||||
)
|
||||
op.execute(
|
||||
"UPDATE ml_settings SET ccip_auto_apply_threshold = 0.95 "
|
||||
"WHERE ccip_auto_apply_threshold = 0.92"
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.execute(
|
||||
"UPDATE ml_settings SET head_auto_apply_min_positives = 30 "
|
||||
"WHERE head_auto_apply_min_positives = 50"
|
||||
)
|
||||
op.execute(
|
||||
"UPDATE ml_settings SET ccip_auto_apply_threshold = 0.92 "
|
||||
"WHERE ccip_auto_apply_threshold = 0.95"
|
||||
)
|
||||
@@ -1,85 +0,0 @@
|
||||
"""presentation-chrome auto-hide (#141) — settings knobs + review table
|
||||
|
||||
MLSettings gains presentation_auto_apply_enabled / _threshold and
|
||||
presentation_conflict_threshold: banner + editor-screenshot auto-hide on the
|
||||
sweep with a FLAT threshold (decoupled from content-head graduation), and a
|
||||
conflict threshold that flags an auto-hide that "also looks like content".
|
||||
|
||||
New table presentation_review records an auto-hidden chrome image that also
|
||||
scored high on a content head, surfaced in the Hidden view for a keep-hidden /
|
||||
un-hide decision. Resolved rows are pruned by retention.
|
||||
|
||||
Revision ID: 0082
|
||||
Revises: 0081
|
||||
Create Date: 2026-07-07
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0082"
|
||||
down_revision: Union[str, None] = "0081"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"presentation_auto_apply_enabled", sa.Boolean(), nullable=False,
|
||||
server_default=sa.text("true"),
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"presentation_auto_apply_threshold", sa.Float(), nullable=False,
|
||||
server_default=sa.text("0.90"),
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"presentation_conflict_threshold", sa.Float(), nullable=False,
|
||||
server_default=sa.text("0.50"),
|
||||
),
|
||||
)
|
||||
op.create_table(
|
||||
"presentation_review",
|
||||
sa.Column(
|
||||
"image_record_id", sa.Integer(),
|
||||
sa.ForeignKey("image_record.id", ondelete="CASCADE"),
|
||||
primary_key=True,
|
||||
),
|
||||
sa.Column(
|
||||
"tag_id", sa.Integer(),
|
||||
sa.ForeignKey("tag.id", ondelete="CASCADE"), primary_key=True,
|
||||
),
|
||||
sa.Column(
|
||||
"conflict_tag_id", sa.Integer(),
|
||||
sa.ForeignKey("tag.id", ondelete="SET NULL"), nullable=True,
|
||||
),
|
||||
sa.Column("conflict_score", sa.Float(), nullable=False),
|
||||
sa.Column(
|
||||
"created_at", sa.DateTime(timezone=True), nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
sa.Column("resolved_at", sa.DateTime(timezone=True), nullable=True),
|
||||
)
|
||||
# The review list queries the unresolved flags (resolved_at IS NULL).
|
||||
op.create_index(
|
||||
"ix_presentation_review_resolved_at", "presentation_review",
|
||||
["resolved_at"],
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index(
|
||||
"ix_presentation_review_resolved_at", table_name="presentation_review"
|
||||
)
|
||||
op.drop_table("presentation_review")
|
||||
op.drop_column("ml_settings", "presentation_conflict_threshold")
|
||||
op.drop_column("ml_settings", "presentation_auto_apply_threshold")
|
||||
op.drop_column("ml_settings", "presentation_auto_apply_enabled")
|
||||
@@ -1,73 +0,0 @@
|
||||
"""post-text translation via Interpreter (milestone 143) — Post columns + settings
|
||||
|
||||
Post gains the translated title/description + the detected source language,
|
||||
Interpreter engine_version (cache key), and translated_at — filled by the
|
||||
translate sweep. ImportSettings gains translation_enabled (OFF by default),
|
||||
interpreter_base_url (EMPTY — the operator sets their own, behind a reverse
|
||||
proxy), and translation_target_lang (en).
|
||||
|
||||
Revision ID: 0083
|
||||
Revises: 0082
|
||||
Create Date: 2026-07-07
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0083"
|
||||
down_revision: Union[str, None] = "0082"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"post", sa.Column("post_title_translated", sa.Text(), nullable=True)
|
||||
)
|
||||
op.add_column(
|
||||
"post", sa.Column("description_translated", sa.Text(), nullable=True)
|
||||
)
|
||||
op.add_column(
|
||||
"post",
|
||||
sa.Column("translated_source_lang", sa.String(8), nullable=True),
|
||||
)
|
||||
op.add_column(
|
||||
"post",
|
||||
sa.Column("translation_engine_version", sa.String(128), nullable=True),
|
||||
)
|
||||
op.add_column(
|
||||
"post",
|
||||
sa.Column("translated_at", sa.DateTime(timezone=True), nullable=True),
|
||||
)
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"translation_enabled", sa.Boolean(), nullable=False,
|
||||
server_default=sa.text("false"),
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"interpreter_base_url", sa.Text(), nullable=False, server_default="",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"translation_target_lang", sa.Text(), nullable=False,
|
||||
server_default="en",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("import_settings", "translation_target_lang")
|
||||
op.drop_column("import_settings", "interpreter_base_url")
|
||||
op.drop_column("import_settings", "translation_enabled")
|
||||
op.drop_column("post", "translated_at")
|
||||
op.drop_column("post", "translation_engine_version")
|
||||
op.drop_column("post", "translated_source_lang")
|
||||
op.drop_column("post", "description_translated")
|
||||
op.drop_column("post", "post_title_translated")
|
||||
@@ -1,51 +0,0 @@
|
||||
"""translation strictness setting + per-post translation override (milestone 155)
|
||||
|
||||
ImportSettings gains ``translation_min_confidence`` (the latin-script acceptance
|
||||
floor, now operator-tunable in the UI; default 0.9 — stricter than the old
|
||||
hardcoded 0.8, since Interpreter confidently mis-detects short ASCII English at
|
||||
~0.86). Post gains ``translation_override`` — a sticky per-post choice of
|
||||
auto / force / original so the operator can force a skipped translation on, or
|
||||
knock a wrongly-translated one back to the original, and have it survive a
|
||||
Re-translate-all.
|
||||
|
||||
Revision ID: 0084
|
||||
Revises: 0083
|
||||
Create Date: 2026-07-10
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0084"
|
||||
down_revision: Union[str, None] = "0083"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"translation_min_confidence", sa.Float(), nullable=False,
|
||||
server_default=sa.text("0.9"),
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"post",
|
||||
sa.Column(
|
||||
"translation_override", sa.String(16), nullable=False,
|
||||
server_default="auto",
|
||||
),
|
||||
)
|
||||
op.create_check_constraint(
|
||||
"ck_post_translation_override",
|
||||
"post",
|
||||
"translation_override IN ('auto', 'force', 'original')",
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_constraint("ck_post_translation_override", "post", type_="check")
|
||||
op.drop_column("post", "translation_override")
|
||||
op.drop_column("import_settings", "translation_min_confidence")
|
||||
@@ -1,35 +0,0 @@
|
||||
"""title-based WIP auto-tagging (task #1458) — ImportSettings toggle
|
||||
|
||||
ImportSettings gains wip_title_tagging_enabled (ON by default): when a freshly
|
||||
imported post's title explicitly declares work-in-progress ("WIP" / "work in
|
||||
progress"), the importer applies the `wip` system tag to its images. No new
|
||||
table — the tag itself is the seeded `wip` system tag (migration 0075) and the
|
||||
application reuses image_tag with source='wip_title'.
|
||||
|
||||
Revision ID: 0085
|
||||
Revises: 0084
|
||||
Create Date: 2026-07-12
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0085"
|
||||
down_revision: Union[str, None] = "0084"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"wip_title_tagging_enabled", sa.Boolean(), nullable=False,
|
||||
server_default=sa.text("true"),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("import_settings", "wip_title_tagging_enabled")
|
||||
@@ -1,61 +0,0 @@
|
||||
"""process auto-apply settings + review mode (#1464) — system-tag refactor
|
||||
|
||||
The system-tag behavior refactor gives `wip` / `editor screenshot` (the PROCESS
|
||||
group) their own provisional auto-apply, parallel to the presentation (chrome)
|
||||
sweep. MLSettings gains three knobs: enabled (OFF by default — a new whole-library
|
||||
auto-tagger is opt-in), the flat apply threshold, and the ring-loud conflict
|
||||
threshold. presentation_review gains a `mode` column so one review surface serves
|
||||
both chrome and process flags (existing rows backfill 'chrome'). server_defaults
|
||||
so the existing rows fill cleanly.
|
||||
|
||||
Revision ID: 0086
|
||||
Revises: 0085
|
||||
Create Date: 2026-07-13
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0086"
|
||||
down_revision: Union[str, None] = "0085"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"process_auto_apply_enabled", sa.Boolean(), nullable=False,
|
||||
server_default=sa.text("false"),
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"process_auto_apply_threshold", sa.Float(), nullable=False,
|
||||
server_default="0.90",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"ml_settings",
|
||||
sa.Column(
|
||||
"process_conflict_threshold", sa.Float(), nullable=False,
|
||||
server_default="0.50",
|
||||
),
|
||||
)
|
||||
op.add_column(
|
||||
"presentation_review",
|
||||
sa.Column(
|
||||
"mode", sa.String(16), nullable=False,
|
||||
server_default="chrome",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("presentation_review", "mode")
|
||||
op.drop_column("ml_settings", "process_conflict_threshold")
|
||||
op.drop_column("ml_settings", "process_auto_apply_threshold")
|
||||
op.drop_column("ml_settings", "process_auto_apply_enabled")
|
||||
@@ -1,33 +0,0 @@
|
||||
"""soft WIP title tier toggle (#1474) — ImportSettings.wip_soft_title_tagging_enabled
|
||||
|
||||
The soft tier also tags sketch/doodle/scribble titles, but with a provisional source
|
||||
that never trains the head. OFF by default (a lower-precision tier is opt-in).
|
||||
server_default so the existing singleton row (id=1) fills cleanly.
|
||||
|
||||
Revision ID: 0087
|
||||
Revises: 0086
|
||||
Create Date: 2026-07-13
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0087"
|
||||
down_revision: Union[str, None] = "0086"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"import_settings",
|
||||
sa.Column(
|
||||
"wip_soft_title_tagging_enabled", sa.Boolean(), nullable=False,
|
||||
server_default=sa.text("false"),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("import_settings", "wip_soft_title_tagging_enabled")
|
||||
@@ -0,0 +1,836 @@
|
||||
"""The whole schema, in one migration.
|
||||
|
||||
This replaces alembic revisions 0001..0089 — the entire build-out of the
|
||||
project, 89 files and ~6,000 lines that a new installation used to replay in
|
||||
order to arrive at a schema this file creates in one pass. Nothing about the
|
||||
resulting database changes; what goes away is the requirement that a stranger
|
||||
re-run our development history to get it.
|
||||
|
||||
## Why the revision id is 0089
|
||||
|
||||
`revision = "0089"` and `down_revision = None` are both deliberate, and the
|
||||
combination is the entire migration strategy for existing installations.
|
||||
|
||||
An already-deployed database has `alembic_version = '0089'`, because it ran the
|
||||
real 0089. This file claims that same id, so alembic reads the version table,
|
||||
sees head already reached, and does nothing at all. No stamp is needed — which
|
||||
matters because `alembic stamp` writes a version string without validating
|
||||
anything about the schema it is writing it against, and a stamp that is wrong
|
||||
is indistinguishable from one that is right until the next migration fails.
|
||||
|
||||
An empty database has no version row, so alembic runs this file and then
|
||||
records `0089`. Both paths converge on the same schema and the same version,
|
||||
and neither requires anyone to assert anything by hand.
|
||||
|
||||
The next migration written after this one is `0090`, exactly as it would have
|
||||
been. The numbering is continuous across the collapse on purpose.
|
||||
|
||||
## What was added to the generated output, and why
|
||||
|
||||
`alembic revision --autogenerate` produced almost all of this from the models,
|
||||
which is only true because #3275 first made the models actually describe the
|
||||
schema. Before that reconciliation the generator silently omitted eleven
|
||||
indexes and three uniqueness guarantees, and an earlier attempt at this squash
|
||||
had to be reverted for exactly that reason.
|
||||
|
||||
Four things still had to be added by hand, because they are not in the models:
|
||||
|
||||
1. **`CREATE EXTENSION vector`** (from 0001) and **`tsm_system_rows`** (0004).
|
||||
Extensions are database objects, not table metadata, so no model can carry
|
||||
them. `IF NOT EXISTS` because a re-run must not fail.
|
||||
|
||||
2. **Three seed inserts** — the two settings singletons (0002, 0003) and the
|
||||
three hygiene system tags (0075). Some migrations did not only build schema;
|
||||
they inserted rows the product needs in order to function, and nothing in
|
||||
the application ever creates them. Every consumer reads them with
|
||||
`scalar_one()`, which RAISES `NoResultFound` on an empty result rather than
|
||||
returning None, so their absence is a crash and not a degradation.
|
||||
|
||||
Distinguishing these from the other data statements in the chain is the
|
||||
whole trick, and the rule turns out to be mechanical:
|
||||
|
||||
* `INSERT ... VALUES (...)` with literal values is a SEED. It creates
|
||||
something the product ships. It must be carried.
|
||||
* `INSERT ... SELECT ... FROM <table>` is a BACKFILL. It derives rows
|
||||
from rows that already exist, so on an empty database it inserts
|
||||
nothing and carrying it would be pointless. 0034 (artist_visit), 0040
|
||||
and 0047 (series_chapter) are all of this shape and are correctly
|
||||
absent here.
|
||||
|
||||
This category is invisible to every automated check this project has:
|
||||
`baseline.yml` compares SCHEMA, and a baseline missing all three seeds still
|
||||
produces a byte-identical schema and a perfectly green diff. What caught the
|
||||
system tags was the integration suite — 36 tests failing on
|
||||
`NoResultFound` — after a first version of this file shipped with only the
|
||||
two settings rows. A first-run check against the real application is the
|
||||
only thing that finds this class of defect.
|
||||
|
||||
3. **The `pgvector` import.** Autogenerate emits qualified
|
||||
`pgvector.sqlalchemy.vector.VECTOR(...)` references without importing the
|
||||
package, so the file it writes cannot execute — `NameError: name 'pgvector'
|
||||
is not defined`, observed on run 4988.
|
||||
|
||||
The other data statements in the old chain were deliberately NOT carried over.
|
||||
0023's `DELETE FROM tag WHERE kind IN (...)`, and 0047's `series_page` /
|
||||
`series_chapter` deletes, are historical cleanups that operate on rows an empty
|
||||
database does not have.
|
||||
|
||||
## Downgrade
|
||||
|
||||
There is none. A baseline's downgrade would be "drop the entire schema", which
|
||||
is not a migration but a data-loss event wearing one as a disguise. Restore
|
||||
from a backup instead — that is what backup_run exists for.
|
||||
|
||||
Revision ID: 0089
|
||||
Revises:
|
||||
Create Date: 2026-09-01
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
import sqlalchemy as sa
|
||||
import pgvector.sqlalchemy.vector
|
||||
from sqlalchemy.dialects import postgresql
|
||||
|
||||
revision: str = "0089"
|
||||
down_revision: Union[str, None] = None
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
# Extensions first: image_record.siglip_embedding is a vector column and
|
||||
# cannot be created before the type exists. From 0001 and 0004.
|
||||
op.execute("CREATE EXTENSION IF NOT EXISTS vector")
|
||||
op.execute("CREATE EXTENSION IF NOT EXISTS tsm_system_rows")
|
||||
|
||||
op.create_table('app_setting',
|
||||
sa.Column('key', sa.String(length=64), nullable=False),
|
||||
sa.Column('value', sa.Text(), nullable=False),
|
||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.PrimaryKeyConstraint('key', name=op.f('pk_app_setting'))
|
||||
)
|
||||
op.create_table('artist',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('name', sa.String(length=255), nullable=False),
|
||||
sa.Column('slug', sa.String(length=255), nullable=False),
|
||||
sa.Column('notes', sa.Text(), nullable=True),
|
||||
sa.Column('is_subscription', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('auto_check', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('check_interval_seconds', sa.Integer(), nullable=True),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_artist')),
|
||||
sa.UniqueConstraint('slug', name=op.f('uq_artist_slug'))
|
||||
)
|
||||
op.create_table('backup_run',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('kind', sa.String(length=16), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='pending', nullable=False),
|
||||
sa.Column('tag', sa.String(length=64), nullable=True),
|
||||
sa.Column('triggered_by', sa.String(length=32), nullable=False),
|
||||
sa.Column('started_at', sa.DateTime(timezone=True), nullable=False),
|
||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('sql_path', sa.Text(), nullable=True),
|
||||
sa.Column('tar_path', sa.Text(), nullable=True),
|
||||
sa.Column('size_bytes', sa.BigInteger(), nullable=True),
|
||||
sa.Column('error', sa.Text(), nullable=True),
|
||||
sa.Column('manifest', sa.JSON(), server_default='{}', nullable=False),
|
||||
sa.Column('restored_from_id', sa.Integer(), nullable=True),
|
||||
sa.ForeignKeyConstraint(['restored_from_id'], ['backup_run.id'], name=op.f('fk_backup_run_restored_from_id_backup_run'), ondelete='SET NULL'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_backup_run'))
|
||||
)
|
||||
op.create_index(op.f('ix_backup_run_finished_at'), 'backup_run', ['finished_at'], unique=False)
|
||||
op.create_index('ix_backup_run_kind_started', 'backup_run', ['kind', sa.literal_column('started_at DESC')], unique=False)
|
||||
op.create_index(op.f('ix_backup_run_restored_from_id'), 'backup_run', ['restored_from_id'], unique=False)
|
||||
op.create_index(op.f('ix_backup_run_started_at'), 'backup_run', ['started_at'], unique=False)
|
||||
op.create_index('ix_backup_run_status_finished', 'backup_run', ['status', sa.literal_column('finished_at DESC')], unique=False)
|
||||
op.create_index(op.f('ix_backup_run_tag'), 'backup_run', ['tag'], unique=False)
|
||||
op.create_index('ix_backup_run_tag_partial', 'backup_run', ['tag'], unique=False, postgresql_where=sa.text('tag IS NOT NULL'))
|
||||
op.create_table('credential',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('platform', sa.String(length=64), nullable=False),
|
||||
sa.Column('credential_type', sa.String(length=32), nullable=False),
|
||||
sa.Column('encrypted_blob', sa.LargeBinary(), nullable=False),
|
||||
sa.Column('captured_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('expires_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('last_verified', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_credential')),
|
||||
sa.UniqueConstraint('platform', name=op.f('uq_credential_platform'))
|
||||
)
|
||||
op.create_table('head_auto_apply_run',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('dry_run', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='running', nullable=False),
|
||||
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('n_applied', sa.Integer(), nullable=True),
|
||||
sa.Column('report', postgresql.JSONB(astext_type=sa.Text()), nullable=True),
|
||||
sa.Column('error', sa.Text(), nullable=True),
|
||||
sa.Column('last_progress_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_head_auto_apply_run'))
|
||||
)
|
||||
op.create_index(op.f('ix_head_auto_apply_run_status'), 'head_auto_apply_run', ['status'], unique=False)
|
||||
op.create_table('head_training_run',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='running', nullable=False),
|
||||
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('n_trained', sa.Integer(), nullable=True),
|
||||
sa.Column('n_skipped', sa.Integer(), nullable=True),
|
||||
sa.Column('error', sa.Text(), nullable=True),
|
||||
sa.Column('last_progress_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_head_training_run'))
|
||||
)
|
||||
op.create_index(op.f('ix_head_training_run_status'), 'head_training_run', ['status'], unique=False)
|
||||
op.create_table('import_batch',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('triggered_by', sa.String(length=32), nullable=False),
|
||||
sa.Column('source_path', sa.Text(), nullable=False),
|
||||
sa.Column('scan_mode', sa.String(length=16), nullable=False),
|
||||
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('total_files', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('imported', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('skipped', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('failed', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('attachments', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('refreshed', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='running', nullable=False),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_import_batch'))
|
||||
)
|
||||
op.create_index(op.f('ix_import_batch_status'), 'import_batch', ['status'], unique=False)
|
||||
op.create_table('import_settings',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('import_scan_path', sa.Text(), server_default='/import', nullable=False),
|
||||
sa.Column('min_width', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('min_height', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('skip_transparent', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('transparency_threshold', sa.Float(), server_default='0.9', nullable=False),
|
||||
sa.Column('skip_single_color', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('single_color_threshold', sa.Float(), server_default='0.95', nullable=False),
|
||||
sa.Column('single_color_tolerance', sa.Integer(), server_default='30', nullable=False),
|
||||
sa.Column('phash_threshold', sa.Integer(), server_default='10', nullable=False),
|
||||
sa.Column('download_rate_limit_seconds', sa.Float(), server_default='3', nullable=False),
|
||||
sa.Column('download_validate_files', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('download_schedule_default_seconds', sa.Integer(), server_default='28800', nullable=False),
|
||||
sa.Column('download_event_retention_days', sa.Integer(), server_default='90', nullable=False),
|
||||
sa.Column('download_failure_warning_threshold', sa.Integer(), server_default='5', nullable=False),
|
||||
sa.Column('backup_db_nightly_enabled', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('backup_db_nightly_hour_utc', sa.Integer(), server_default='3', nullable=False),
|
||||
sa.Column('backup_db_keep_last_n', sa.Integer(), server_default='14', nullable=False),
|
||||
sa.Column('backup_images_keep_last_n', sa.Integer(), server_default='3', nullable=False),
|
||||
sa.Column('series_suggest_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('series_suggest_threshold', sa.Float(), server_default='0.5', nullable=False),
|
||||
sa.Column('extdl_mega_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('extdl_gdrive_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('extdl_mediafire_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('extdl_dropbox_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('extdl_pixeldrain_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('translation_enabled', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('interpreter_base_url', sa.Text(), server_default='', nullable=False),
|
||||
sa.Column('translation_target_lang', sa.Text(), server_default='en', nullable=False),
|
||||
sa.Column('translation_min_confidence', sa.Float(), server_default=sa.text('0.9'), nullable=False),
|
||||
sa.Column('wip_title_tagging_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('wip_soft_title_tagging_enabled', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.CheckConstraint('id = 1', name=op.f('ck_import_settings_singleton')),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_import_settings'))
|
||||
)
|
||||
op.create_table('library_audit_run',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('rule', sa.String(length=32), nullable=False),
|
||||
sa.Column('params', postgresql.JSONB(astext_type=sa.Text()), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='running', nullable=False),
|
||||
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('scanned_count', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('matched_count', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('matched_ids', postgresql.JSONB(astext_type=sa.Text()), server_default=sa.text("'[]'::jsonb"), nullable=False),
|
||||
sa.Column('error', sa.Text(), nullable=True),
|
||||
sa.Column('resume_after_id', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('last_progress_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_library_audit_run'))
|
||||
)
|
||||
op.create_index(op.f('ix_library_audit_run_rule'), 'library_audit_run', ['rule'], unique=False)
|
||||
op.create_index(op.f('ix_library_audit_run_status'), 'library_audit_run', ['status'], unique=False)
|
||||
op.create_table('ml_settings',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('cpu_embed_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('video_frame_interval_seconds', sa.Float(), server_default='4', nullable=False),
|
||||
sa.Column('video_max_frames', sa.Integer(), server_default='64', nullable=False),
|
||||
sa.Column('head_min_positives', sa.Integer(), server_default='8', nullable=False),
|
||||
sa.Column('head_auto_apply_precision', sa.Float(), server_default='0.97', nullable=False),
|
||||
sa.Column('head_auto_apply_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('head_auto_apply_min_positives', sa.Integer(), server_default='30', nullable=False),
|
||||
sa.Column('ccip_match_threshold', sa.Float(), server_default='0.85', nullable=False),
|
||||
sa.Column('ccip_auto_apply_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('ccip_auto_apply_threshold', sa.Float(), server_default='0.92', nullable=False),
|
||||
sa.Column('presentation_auto_apply_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('presentation_auto_apply_threshold', sa.Float(), server_default=sa.text('0.90'), nullable=False),
|
||||
sa.Column('presentation_conflict_threshold', sa.Float(), server_default=sa.text('0.50'), nullable=False),
|
||||
sa.Column('process_auto_apply_enabled', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('process_auto_apply_threshold', sa.Float(), server_default='0.90', nullable=False),
|
||||
sa.Column('process_conflict_threshold', sa.Float(), server_default='0.50', nullable=False),
|
||||
sa.Column('embedder_model_version', sa.String(length=128), server_default='siglip2-so400m-patch16-512', nullable=False),
|
||||
sa.Column('embedder_model_name', sa.String(length=128), server_default='google/siglip2-so400m-patch16-512', nullable=False),
|
||||
sa.Column('detector_person_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('detector_person_weights', sa.String(length=512), server_default='yolo11n.pt', nullable=False),
|
||||
sa.Column('detector_person_conf', sa.Float(), server_default=sa.text('0.35'), nullable=False),
|
||||
sa.Column('detector_anatomy_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('detector_anatomy_weights', sa.String(length=512), server_default='https://github.com/aperveyev/booru_yolo/raw/main/models/yolov11m_aa22.pt', nullable=False),
|
||||
sa.Column('detector_anatomy_conf', sa.Float(), server_default=sa.text('0.30'), nullable=False),
|
||||
sa.Column('detector_panel_enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('detector_panel_weights', sa.String(length=512), server_default='mosesb/best-comic-panel-detection::best.pt', nullable=False),
|
||||
sa.Column('detector_panel_conf', sa.Float(), server_default=sa.text('0.30'), nullable=False),
|
||||
sa.Column('detector_max_figures', sa.Integer(), server_default='8', nullable=False),
|
||||
sa.Column('detector_max_components', sa.Integer(), server_default='8', nullable=False),
|
||||
sa.Column('detector_max_panels', sa.Integer(), server_default='8', nullable=False),
|
||||
sa.Column('detector_max_regions', sa.Integer(), server_default='128', nullable=False),
|
||||
sa.Column('detector_dedupe_iou', sa.Float(), server_default=sa.text('0.85'), nullable=False),
|
||||
sa.Column('ccip_ref_signature', sa.String(length=128), nullable=True),
|
||||
sa.Column('ccip_prototype_cap', sa.Integer(), server_default='64', nullable=False),
|
||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.CheckConstraint('id = 1', name=op.f('ck_ml_settings_singleton')),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_ml_settings'))
|
||||
)
|
||||
op.create_table('tag',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('name', sa.String(length=255), nullable=False),
|
||||
sa.Column('kind', sa.Enum('artist', 'character', 'fandom', 'general', 'series', 'archive', 'post', name='tag_kind'), server_default='general', nullable=False),
|
||||
sa.Column('fandom_id', sa.Integer(), nullable=True),
|
||||
sa.Column('is_system', sa.Boolean(), server_default=sa.text('false'), nullable=False),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.CheckConstraint("(fandom_id IS NULL) OR (kind = 'character')", name=op.f('ck_tag_fandom_requires_character')),
|
||||
sa.ForeignKeyConstraint(['fandom_id'], ['tag.id'], name=op.f('fk_tag_fandom_id_tag'), ondelete='SET NULL'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_tag'))
|
||||
)
|
||||
op.create_index(op.f('ix_tag_fandom_id'), 'tag', ['fandom_id'], unique=False)
|
||||
op.create_index('uq_tag_name_kind_fandom', 'tag', ['name', 'kind', sa.literal_column('COALESCE(fandom_id, 0)')], unique=True)
|
||||
op.create_table('task_run',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('celery_task_id', sa.String(length=64), nullable=False),
|
||||
sa.Column('queue', sa.String(length=32), nullable=False),
|
||||
sa.Column('task_name', sa.String(length=128), nullable=False),
|
||||
sa.Column('target_id', sa.Integer(), nullable=True),
|
||||
sa.Column('started_at', sa.DateTime(timezone=True), nullable=False),
|
||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('duration_ms', sa.Integer(), nullable=True),
|
||||
sa.Column('status', sa.String(length=16), server_default='running', nullable=False),
|
||||
sa.Column('error_type', sa.String(length=128), nullable=True),
|
||||
sa.Column('error_message', sa.Text(), nullable=True),
|
||||
sa.Column('retry_count', sa.Integer(), nullable=True),
|
||||
sa.Column('worker_hostname', sa.String(length=128), nullable=True),
|
||||
sa.Column('args_summary', sa.String(length=255), nullable=True),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_task_run'))
|
||||
)
|
||||
op.create_index(op.f('ix_task_run_celery_task_id'), 'task_run', ['celery_task_id'], unique=False)
|
||||
op.create_index(op.f('ix_task_run_finished_at'), 'task_run', ['finished_at'], unique=False)
|
||||
op.create_index('ix_task_run_name_started', 'task_run', ['task_name', sa.literal_column('started_at DESC')], unique=False)
|
||||
op.create_index('ix_task_run_queue_started', 'task_run', ['queue', sa.literal_column('started_at DESC')], unique=False)
|
||||
op.create_index(op.f('ix_task_run_started_at'), 'task_run', ['started_at'], unique=False)
|
||||
op.create_index('ix_task_run_status_started', 'task_run', ['status', sa.literal_column('started_at DESC')], unique=False)
|
||||
op.create_table('artist_visit',
|
||||
sa.Column('artist_id', sa.Integer(), nullable=False),
|
||||
sa.Column('last_viewed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_artist_visit_artist_id_artist'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('artist_id', name=op.f('pk_artist_visit'))
|
||||
)
|
||||
op.create_table('ccip_prototype_state',
|
||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('fingerprint', sa.String(length=64), nullable=False),
|
||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_ccip_prototype_state_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('tag_id', name=op.f('pk_ccip_prototype_state'))
|
||||
)
|
||||
op.create_table('head_metric',
|
||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('n_misfires', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('n_underfires', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_head_metric_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('tag_id', name=op.f('pk_head_metric'))
|
||||
)
|
||||
op.create_table('head_metrics_snapshot',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('tag_id', sa.Integer(), nullable=True),
|
||||
sa.Column('name', sa.String(length=255), nullable=False),
|
||||
sa.Column('snapshot_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('n_auto_applied', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('n_misfires', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('n_underfires', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('ap', sa.Float(), nullable=True),
|
||||
sa.Column('precision_cv', sa.Float(), nullable=True),
|
||||
sa.Column('recall', sa.Float(), nullable=True),
|
||||
sa.Column('n_pos', sa.Integer(), nullable=True),
|
||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_head_metrics_snapshot_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_head_metrics_snapshot'))
|
||||
)
|
||||
op.create_index(op.f('ix_head_metrics_snapshot_snapshot_at'), 'head_metrics_snapshot', ['snapshot_at'], unique=False)
|
||||
op.create_index(op.f('ix_head_metrics_snapshot_tag_id'), 'head_metrics_snapshot', ['tag_id'], unique=False)
|
||||
op.create_table('source',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('artist_id', sa.Integer(), nullable=False),
|
||||
sa.Column('platform', sa.String(length=64), nullable=False),
|
||||
sa.Column('url', sa.Text(), nullable=False),
|
||||
sa.Column('enabled', sa.Boolean(), server_default='true', nullable=False),
|
||||
sa.Column('config_overrides', sa.JSON(), nullable=True),
|
||||
sa.Column('last_checked_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('last_error', sa.Text(), nullable=True),
|
||||
sa.Column('error_type', sa.String(length=32), nullable=True),
|
||||
sa.Column('check_interval_override', sa.Integer(), nullable=True),
|
||||
sa.Column('consecutive_failures', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('backfill_runs_remaining', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_source_artist_id_artist'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_source')),
|
||||
sa.UniqueConstraint('artist_id', 'platform', 'url', name='uq_source_artist_platform_url')
|
||||
)
|
||||
op.create_index(op.f('ix_source_artist_id'), 'source', ['artist_id'], unique=False)
|
||||
op.create_index(op.f('ix_source_error_type'), 'source', ['error_type'], unique=False)
|
||||
op.create_table('tag_alias',
|
||||
sa.Column('alias_string', sa.String(length=255), nullable=False),
|
||||
sa.Column('alias_category', sa.String(length=32), nullable=False),
|
||||
sa.Column('canonical_tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['canonical_tag_id'], ['tag.id'], name=op.f('fk_tag_alias_canonical_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('alias_string', 'alias_category', name=op.f('pk_tag_alias'))
|
||||
)
|
||||
op.create_index('ix_tag_alias_canonical', 'tag_alias', ['canonical_tag_id'], unique=False)
|
||||
op.create_table('tag_head',
|
||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('embedding_version', sa.String(length=128), nullable=False),
|
||||
sa.Column('weights', pgvector.sqlalchemy.vector.VECTOR(dim=1152), nullable=False),
|
||||
sa.Column('bias', sa.Float(), nullable=False),
|
||||
sa.Column('suggest_threshold', sa.Float(), nullable=False),
|
||||
sa.Column('auto_apply_threshold', sa.Float(), nullable=True),
|
||||
sa.Column('n_pos', sa.Integer(), nullable=False),
|
||||
sa.Column('n_neg', sa.Integer(), nullable=False),
|
||||
sa.Column('ap', sa.Float(), nullable=False),
|
||||
sa.Column('precision_cv', sa.Float(), nullable=False),
|
||||
sa.Column('recall', sa.Float(), nullable=False),
|
||||
sa.Column('trained_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('train_fingerprint', sa.String(length=128), nullable=True),
|
||||
sa.Column('metrics', postgresql.JSONB(astext_type=sa.Text()), nullable=True),
|
||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_tag_head_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('tag_id', name=op.f('pk_tag_head'))
|
||||
)
|
||||
op.create_table('patreon_failed_media',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('source_id', sa.Integer(), nullable=False),
|
||||
sa.Column('filehash', sa.String(length=128), nullable=False),
|
||||
sa.Column('attempts', sa.Integer(), server_default='1', nullable=False),
|
||||
sa.Column('last_error', sa.Text(), nullable=True),
|
||||
sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_patreon_failed_media_source_id_source'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_patreon_failed_media')),
|
||||
sa.UniqueConstraint('source_id', 'filehash', name='uq_patreon_failed_media_source_id')
|
||||
)
|
||||
op.create_index(op.f('ix_patreon_failed_media_source_id'), 'patreon_failed_media', ['source_id'], unique=False)
|
||||
op.create_table('patreon_seen_media',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('source_id', sa.Integer(), nullable=False),
|
||||
sa.Column('filehash', sa.String(length=128), nullable=False),
|
||||
sa.Column('post_id', sa.String(length=64), nullable=True),
|
||||
sa.Column('seen_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_patreon_seen_media_source_id_source'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_patreon_seen_media')),
|
||||
sa.UniqueConstraint('source_id', 'filehash', name='uq_patreon_seen_media_source_id')
|
||||
)
|
||||
op.create_index(op.f('ix_patreon_seen_media_source_id'), 'patreon_seen_media', ['source_id'], unique=False)
|
||||
op.create_table('pixiv_failed_media',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('source_id', sa.Integer(), nullable=False),
|
||||
sa.Column('filehash', sa.String(length=128), nullable=False),
|
||||
sa.Column('attempts', sa.Integer(), server_default='1', nullable=False),
|
||||
sa.Column('last_error', sa.Text(), nullable=True),
|
||||
sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_pixiv_failed_media_source_id_source'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_pixiv_failed_media')),
|
||||
sa.UniqueConstraint('source_id', 'filehash', name='uq_pixiv_failed_media_source_id')
|
||||
)
|
||||
op.create_index(op.f('ix_pixiv_failed_media_source_id'), 'pixiv_failed_media', ['source_id'], unique=False)
|
||||
op.create_table('pixiv_seen_media',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('source_id', sa.Integer(), nullable=False),
|
||||
sa.Column('filehash', sa.String(length=128), nullable=False),
|
||||
sa.Column('post_id', sa.String(length=64), nullable=True),
|
||||
sa.Column('seen_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_pixiv_seen_media_source_id_source'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_pixiv_seen_media')),
|
||||
sa.UniqueConstraint('source_id', 'filehash', name='uq_pixiv_seen_media_source_id')
|
||||
)
|
||||
op.create_index(op.f('ix_pixiv_seen_media_source_id'), 'pixiv_seen_media', ['source_id'], unique=False)
|
||||
op.create_table('post',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('source_id', sa.Integer(), nullable=True),
|
||||
sa.Column('artist_id', sa.Integer(), nullable=False),
|
||||
sa.Column('external_post_id', sa.String(length=128), nullable=False),
|
||||
sa.Column('post_url', sa.Text(), nullable=True),
|
||||
sa.Column('post_title', sa.Text(), nullable=True),
|
||||
sa.Column('post_date', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('raw_metadata', sa.JSON(), nullable=True),
|
||||
sa.Column('description', sa.Text(), nullable=True),
|
||||
sa.Column('attachment_count', sa.Integer(), nullable=True),
|
||||
sa.Column('post_title_translated', sa.Text(), nullable=True),
|
||||
sa.Column('description_translated', sa.Text(), nullable=True),
|
||||
sa.Column('translated_source_lang', sa.String(length=8), nullable=True),
|
||||
sa.Column('translation_engine_version', sa.String(length=128), nullable=True),
|
||||
sa.Column('translated_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('translation_override', sa.String(length=16), server_default='auto', nullable=False),
|
||||
sa.Column('downloaded_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.CheckConstraint("translation_override IN ('auto', 'force', 'original')", name=op.f('ck_post_translation_override')),
|
||||
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_post_artist_id_artist'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_post_source_id_source'), ondelete='SET NULL'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_post')),
|
||||
sa.UniqueConstraint('source_id', 'external_post_id', name='uq_post_source_external_id')
|
||||
)
|
||||
op.create_index(op.f('ix_post_artist_id'), 'post', ['artist_id'], unique=False)
|
||||
op.create_index(op.f('ix_post_source_id'), 'post', ['source_id'], unique=False)
|
||||
op.create_index('uq_post_artist_external_id_null_source', 'post', ['artist_id', 'external_post_id'], unique=True, postgresql_where=sa.text('source_id IS NULL'))
|
||||
op.create_table('subscribestar_failed_media',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('source_id', sa.Integer(), nullable=False),
|
||||
sa.Column('filehash', sa.String(length=128), nullable=False),
|
||||
sa.Column('attempts', sa.Integer(), server_default='1', nullable=False),
|
||||
sa.Column('last_error', sa.Text(), nullable=True),
|
||||
sa.Column('first_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('last_failed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_subscribestar_failed_media_source_id_source'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_subscribestar_failed_media')),
|
||||
sa.UniqueConstraint('source_id', 'filehash', name='uq_subscribestar_failed_media_source_id')
|
||||
)
|
||||
op.create_index(op.f('ix_subscribestar_failed_media_source_id'), 'subscribestar_failed_media', ['source_id'], unique=False)
|
||||
op.create_table('subscribestar_seen_media',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('source_id', sa.Integer(), nullable=False),
|
||||
sa.Column('filehash', sa.String(length=128), nullable=False),
|
||||
sa.Column('post_id', sa.String(length=64), nullable=True),
|
||||
sa.Column('seen_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_subscribestar_seen_media_source_id_source'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_subscribestar_seen_media')),
|
||||
sa.UniqueConstraint('source_id', 'filehash', name='uq_subscribestar_seen_media_source_id')
|
||||
)
|
||||
op.create_index(op.f('ix_subscribestar_seen_media_source_id'), 'subscribestar_seen_media', ['source_id'], unique=False)
|
||||
op.create_table('download_event',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('source_id', sa.Integer(), nullable=False),
|
||||
sa.Column('post_id', sa.Integer(), nullable=True),
|
||||
sa.Column('status', sa.String(length=32), nullable=False),
|
||||
sa.Column('started_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('bytes_downloaded', sa.BigInteger(), server_default='0', nullable=False),
|
||||
sa.Column('files_count', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('error', sa.Text(), nullable=True),
|
||||
sa.Column('metadata', postgresql.JSONB(astext_type=sa.Text()), server_default=sa.text("'{}'::jsonb"), nullable=False),
|
||||
sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_download_event_post_id_post'), ondelete='SET NULL'),
|
||||
sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_download_event_source_id_source'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_download_event'))
|
||||
)
|
||||
op.create_index(op.f('ix_download_event_post_id'), 'download_event', ['post_id'], unique=False)
|
||||
op.create_index(op.f('ix_download_event_source_id'), 'download_event', ['source_id'], unique=False)
|
||||
op.create_table('image_record',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('path', sa.Text(), nullable=False),
|
||||
sa.Column('sha256', sa.String(length=64), nullable=False),
|
||||
sa.Column('phash', sa.String(length=32), nullable=True),
|
||||
sa.Column('size_bytes', sa.BigInteger(), nullable=False),
|
||||
sa.Column('mime', sa.String(length=64), nullable=False),
|
||||
sa.Column('width', sa.Integer(), nullable=True),
|
||||
sa.Column('height', sa.Integer(), nullable=True),
|
||||
sa.Column('duration_seconds', sa.Float(), nullable=True),
|
||||
sa.Column('integrity_status', sa.String(length=24), server_default='unknown', nullable=False),
|
||||
sa.Column('thumbnail_path', sa.Text(), nullable=True),
|
||||
sa.Column('source_url', sa.Text(), nullable=True),
|
||||
sa.Column('source_filehash', sa.String(length=32), nullable=True),
|
||||
sa.Column('origin', sa.Enum('downloaded', 'imported_filesystem', 'uploaded', name='origin_enum'), nullable=False),
|
||||
sa.Column('primary_post_id', sa.Integer(), nullable=True),
|
||||
sa.Column('artist_id', sa.Integer(), nullable=True),
|
||||
sa.Column('siglip_embedding', pgvector.sqlalchemy.vector.VECTOR(dim=1152), nullable=True),
|
||||
sa.Column('siglip_model_version', sa.String(length=128), nullable=True),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('effective_date', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('earliest_post_date', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name='fk_image_record_artist_id', ondelete='SET NULL'),
|
||||
sa.ForeignKeyConstraint(['primary_post_id'], ['post.id'], name=op.f('fk_image_record_primary_post_id_post'), ondelete='SET NULL'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_image_record')),
|
||||
sa.UniqueConstraint('path', name=op.f('uq_image_record_path')),
|
||||
sa.UniqueConstraint('sha256', name='uq_image_record_sha256')
|
||||
)
|
||||
op.create_index(op.f('ix_image_record_artist_id'), 'image_record', ['artist_id'], unique=False)
|
||||
op.create_index('ix_image_record_earliest_post_date', 'image_record', [sa.literal_column('earliest_post_date DESC'), sa.literal_column('id DESC')], unique=False)
|
||||
op.create_index('ix_image_record_effective_date', 'image_record', [sa.literal_column('effective_date DESC'), sa.literal_column('id DESC')], unique=False)
|
||||
op.create_index(op.f('ix_image_record_integrity_status'), 'image_record', ['integrity_status'], unique=False)
|
||||
op.create_index(op.f('ix_image_record_phash'), 'image_record', ['phash'], unique=False)
|
||||
op.create_index(op.f('ix_image_record_primary_post_id'), 'image_record', ['primary_post_id'], unique=False)
|
||||
op.create_index('ix_image_record_siglip_hnsw', 'image_record', ['siglip_embedding'], unique=False, postgresql_using='hnsw', postgresql_ops={'siglip_embedding': 'vector_cosine_ops'})
|
||||
op.create_index(op.f('ix_image_record_source_filehash'), 'image_record', ['source_filehash'], unique=False)
|
||||
op.create_table('post_attachment',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('post_id', sa.Integer(), nullable=True),
|
||||
sa.Column('artist_id', sa.Integer(), nullable=True),
|
||||
sa.Column('sha256', sa.String(length=64), nullable=False),
|
||||
sa.Column('path', sa.Text(), nullable=False),
|
||||
sa.Column('original_filename', sa.Text(), nullable=False),
|
||||
sa.Column('ext', sa.String(length=32), nullable=False),
|
||||
sa.Column('mime', sa.String(length=128), nullable=True),
|
||||
sa.Column('size_bytes', sa.BigInteger(), nullable=False),
|
||||
sa.Column('captured_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_post_attachment_artist_id_artist'), ondelete='SET NULL'),
|
||||
sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_post_attachment_post_id_post'), ondelete='SET NULL'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_post_attachment'))
|
||||
)
|
||||
op.create_index(op.f('ix_post_attachment_artist_id'), 'post_attachment', ['artist_id'], unique=False)
|
||||
op.create_index(op.f('ix_post_attachment_post_id'), 'post_attachment', ['post_id'], unique=False)
|
||||
op.create_index(op.f('ix_post_attachment_sha256'), 'post_attachment', ['sha256'], unique=False)
|
||||
op.create_index('uq_post_attachment_null_post_sha', 'post_attachment', ['sha256'], unique=True, postgresql_where=sa.text('post_id IS NULL'))
|
||||
op.create_index('uq_post_attachment_post_sha', 'post_attachment', ['post_id', 'sha256'], unique=True, postgresql_where=sa.text('post_id IS NOT NULL'))
|
||||
op.create_table('series_suggestion',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('post_id', sa.Integer(), nullable=False),
|
||||
sa.Column('series_tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('score', sa.Float(), nullable=False),
|
||||
sa.Column('signals', sa.JSON(), nullable=True),
|
||||
sa.Column('status', sa.String(length=16), server_default='pending', nullable=False),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_series_suggestion_post_id_post'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['series_tag_id'], ['tag.id'], name=op.f('fk_series_suggestion_series_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_series_suggestion')),
|
||||
sa.UniqueConstraint('post_id', 'series_tag_id', name='uq_series_suggestion_post_series')
|
||||
)
|
||||
op.create_index(op.f('ix_series_suggestion_post_id'), 'series_suggestion', ['post_id'], unique=False)
|
||||
op.create_index(op.f('ix_series_suggestion_series_tag_id'), 'series_suggestion', ['series_tag_id'], unique=False)
|
||||
op.create_index(op.f('ix_series_suggestion_status'), 'series_suggestion', ['status'], unique=False)
|
||||
op.create_table('external_link',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('post_id', sa.Integer(), nullable=False),
|
||||
sa.Column('artist_id', sa.Integer(), nullable=True),
|
||||
sa.Column('host', sa.String(length=16), nullable=False),
|
||||
sa.Column('url', sa.Text(), nullable=False),
|
||||
sa.Column('label', sa.Text(), nullable=True),
|
||||
sa.Column('status', sa.String(length=16), server_default='pending', nullable=False),
|
||||
sa.Column('attempts', sa.Integer(), server_default=sa.text('0'), nullable=False),
|
||||
sa.Column('last_error', sa.Text(), nullable=True),
|
||||
sa.Column('attachment_id', sa.Integer(), nullable=True),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('completed_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('duration_seconds', sa.Float(), nullable=True),
|
||||
sa.CheckConstraint("host IN ('mega', 'gdrive', 'mediafire', 'dropbox', 'pixeldrain')", name=op.f('ck_external_link_host')),
|
||||
sa.CheckConstraint("status IN ('pending', 'downloading', 'downloaded', 'failed', 'skipped', 'dead')", name=op.f('ck_external_link_status')),
|
||||
sa.ForeignKeyConstraint(['artist_id'], ['artist.id'], name=op.f('fk_external_link_artist_id_artist'), ondelete='SET NULL'),
|
||||
sa.ForeignKeyConstraint(['attachment_id'], ['post_attachment.id'], name=op.f('fk_external_link_attachment_id_post_attachment'), ondelete='SET NULL'),
|
||||
sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_external_link_post_id_post'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_external_link'))
|
||||
)
|
||||
op.create_index(op.f('ix_external_link_artist_id'), 'external_link', ['artist_id'], unique=False)
|
||||
op.create_index('ix_external_link_attachment_id', 'external_link', ['attachment_id'], unique=False)
|
||||
op.create_index('ix_external_link_status', 'external_link', ['status'], unique=False)
|
||||
op.create_index('uq_external_link_post_url', 'external_link', ['post_id', 'url'], unique=True)
|
||||
op.create_table('gpu_job',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
||||
sa.Column('task', sa.String(length=32), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='pending', nullable=False),
|
||||
sa.Column('lease_token', sa.String(length=64), nullable=True),
|
||||
sa.Column('leased_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('lease_expires_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('attempts', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('error', sa.Text(), nullable=True),
|
||||
sa.Column('triage_status', sa.String(length=16), nullable=True),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_gpu_job_image_record_id_image_record'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_gpu_job'))
|
||||
)
|
||||
op.create_index(op.f('ix_gpu_job_image_record_id'), 'gpu_job', ['image_record_id'], unique=False)
|
||||
op.create_index('ix_gpu_job_leased_expires', 'gpu_job', ['lease_expires_at'], unique=False, postgresql_where=sa.text("status = 'leased'"))
|
||||
op.create_index('ix_gpu_job_pending', 'gpu_job', ['id'], unique=False, postgresql_where=sa.text("status = 'pending'"))
|
||||
op.create_index(op.f('ix_gpu_job_status'), 'gpu_job', ['status'], unique=False)
|
||||
op.create_table('image_provenance',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
||||
sa.Column('post_id', sa.Integer(), nullable=False),
|
||||
sa.Column('source_id', sa.Integer(), nullable=True),
|
||||
sa.Column('from_attachment_id', sa.Integer(), nullable=True),
|
||||
sa.Column('captured_metadata', sa.JSON(), nullable=True),
|
||||
sa.Column('captured_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['from_attachment_id'], ['post_attachment.id'], name='fk_image_provenance_from_attachment', ondelete='SET NULL'),
|
||||
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_image_provenance_image_record_id_image_record'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['post_id'], ['post.id'], name=op.f('fk_image_provenance_post_id_post'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['source_id'], ['source.id'], name=op.f('fk_image_provenance_source_id_source'), ondelete='SET NULL'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_image_provenance')),
|
||||
sa.UniqueConstraint('image_record_id', 'post_id', name='uq_image_provenance_image_post')
|
||||
)
|
||||
op.create_index(op.f('ix_image_provenance_from_attachment_id'), 'image_provenance', ['from_attachment_id'], unique=False)
|
||||
op.create_index(op.f('ix_image_provenance_image_record_id'), 'image_provenance', ['image_record_id'], unique=False)
|
||||
op.create_index(op.f('ix_image_provenance_post_id'), 'image_provenance', ['post_id'], unique=False)
|
||||
op.create_index(op.f('ix_image_provenance_source_id'), 'image_provenance', ['source_id'], unique=False)
|
||||
op.create_table('image_region',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
||||
sa.Column('kind', sa.String(length=16), nullable=False),
|
||||
sa.Column('frame_time', sa.Float(), nullable=True),
|
||||
sa.Column('rx', sa.Float(), nullable=False),
|
||||
sa.Column('ry', sa.Float(), nullable=False),
|
||||
sa.Column('rw', sa.Float(), nullable=False),
|
||||
sa.Column('rh', sa.Float(), nullable=False),
|
||||
sa.Column('score', sa.Float(), nullable=True),
|
||||
sa.Column('detector_version', sa.String(length=64), nullable=True),
|
||||
sa.Column('crop_version', sa.String(length=64), nullable=True),
|
||||
sa.Column('embedding_version', sa.String(length=128), nullable=True),
|
||||
sa.Column('ccip_embedding', pgvector.sqlalchemy.vector.VECTOR(dim=768), nullable=True),
|
||||
sa.Column('siglip_embedding', pgvector.sqlalchemy.vector.VECTOR(dim=1152), nullable=True),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_image_region_image_record_id_image_record'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_image_region'))
|
||||
)
|
||||
op.create_index(op.f('ix_image_region_image_record_id'), 'image_region', ['image_record_id'], unique=False)
|
||||
op.create_table('image_tag',
|
||||
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('source', sa.String(length=32), server_default='manual', nullable=False),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_image_tag_image_record_id_image_record'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_image_tag_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_image_tag'))
|
||||
)
|
||||
op.create_index('ix_image_tag_tag_id', 'image_tag', ['tag_id'], unique=False)
|
||||
op.create_table('import_task',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('batch_id', sa.Integer(), nullable=False),
|
||||
sa.Column('source_path', sa.Text(), nullable=False),
|
||||
sa.Column('task_type', sa.String(length=16), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='pending', nullable=False),
|
||||
sa.Column('recovery_count', sa.Integer(), server_default='0', nullable=False),
|
||||
sa.Column('refetched', sa.Boolean(), server_default='false', nullable=False),
|
||||
sa.Column('result_image_id', sa.Integer(), nullable=True),
|
||||
sa.Column('error', sa.Text(), nullable=True),
|
||||
sa.Column('size_bytes', sa.BigInteger(), nullable=True),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('started_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column('finished_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.ForeignKeyConstraint(['batch_id'], ['import_batch.id'], name=op.f('fk_import_task_batch_id_import_batch'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['result_image_id'], ['image_record.id'], name=op.f('fk_import_task_result_image_id_image_record'), ondelete='SET NULL'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_import_task'))
|
||||
)
|
||||
op.create_index(op.f('ix_import_task_batch_id'), 'import_task', ['batch_id'], unique=False)
|
||||
op.create_index('ix_import_task_created_at_desc', 'import_task', [sa.literal_column('created_at DESC')], unique=False)
|
||||
op.create_index('ix_import_task_result_image_id', 'import_task', ['result_image_id'], unique=False)
|
||||
op.create_index(op.f('ix_import_task_status'), 'import_task', ['status'], unique=False)
|
||||
op.create_table('presentation_review',
|
||||
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('conflict_tag_id', sa.Integer(), nullable=True),
|
||||
sa.Column('conflict_score', sa.Float(), nullable=False),
|
||||
sa.Column('mode', sa.String(length=16), server_default='chrome', nullable=False),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('resolved_at', sa.DateTime(timezone=True), nullable=True),
|
||||
sa.ForeignKeyConstraint(['conflict_tag_id'], ['tag.id'], name=op.f('fk_presentation_review_conflict_tag_id_tag'), ondelete='SET NULL'),
|
||||
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_presentation_review_image_record_id_image_record'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_presentation_review_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_presentation_review'))
|
||||
)
|
||||
op.create_index('ix_presentation_review_conflict_tag_id', 'presentation_review', ['conflict_tag_id'], unique=False)
|
||||
op.create_index('ix_presentation_review_resolved_at', 'presentation_review', ['resolved_at'], unique=False)
|
||||
op.create_index('ix_presentation_review_tag_id', 'presentation_review', ['tag_id'], unique=False)
|
||||
op.create_table('series_page',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('series_tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('image_id', sa.Integer(), nullable=False),
|
||||
sa.Column('status', sa.String(length=16), server_default='placed', nullable=False),
|
||||
sa.Column('page_number', sa.Integer(), nullable=True),
|
||||
sa.Column('stated_page', sa.Integer(), nullable=True),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['image_id'], ['image_record.id'], name=op.f('fk_series_page_image_id_image_record'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['series_tag_id'], ['tag.id'], name=op.f('fk_series_page_series_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_series_page')),
|
||||
sa.UniqueConstraint('image_id', name='uq_series_page_image')
|
||||
)
|
||||
op.create_index(op.f('ix_series_page_series_tag_id'), 'series_page', ['series_tag_id'], unique=False)
|
||||
op.create_table('tag_positive_confirmation',
|
||||
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('confirmed_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name=op.f('fk_tag_positive_confirmation_image_record_id_image_record'), ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_tag_positive_confirmation_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_tag_positive_confirmation'))
|
||||
)
|
||||
op.create_index(op.f('ix_tag_positive_confirmation_tag_id'), 'tag_positive_confirmation', ['tag_id'], unique=False)
|
||||
op.create_table('tag_suggestion_rejection',
|
||||
sa.Column('image_record_id', sa.Integer(), nullable=False),
|
||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('rejected_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['image_record_id'], ['image_record.id'], name='fk_tsr_image_record_id_image_record', ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name='fk_tsr_tag_id_tag', ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('image_record_id', 'tag_id', name=op.f('pk_tag_suggestion_rejection'))
|
||||
)
|
||||
op.create_index('ix_tag_suggestion_rejection_tag', 'tag_suggestion_rejection', ['tag_id'], unique=False)
|
||||
op.create_table('character_prototype',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('ccip_embedding', pgvector.sqlalchemy.vector.VECTOR(dim=768), nullable=False),
|
||||
sa.Column('region_id', sa.Integer(), nullable=True),
|
||||
sa.ForeignKeyConstraint(['region_id'], ['image_region.id'], name=op.f('fk_character_prototype_region_id_image_region'), ondelete='SET NULL'),
|
||||
sa.ForeignKeyConstraint(['tag_id'], ['tag.id'], name=op.f('fk_character_prototype_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_character_prototype'))
|
||||
)
|
||||
op.create_index(op.f('ix_character_prototype_region_id'), 'character_prototype', ['region_id'], unique=False)
|
||||
op.create_index(op.f('ix_character_prototype_tag_id'), 'character_prototype', ['tag_id'], unique=False)
|
||||
op.create_table('series_chapter',
|
||||
sa.Column('id', sa.Integer(), nullable=False),
|
||||
sa.Column('series_tag_id', sa.Integer(), nullable=False),
|
||||
sa.Column('anchor_page_id', sa.Integer(), nullable=False),
|
||||
sa.Column('title', sa.Text(), nullable=True),
|
||||
sa.Column('stated_part', sa.Integer(), nullable=True),
|
||||
sa.Column('created_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.Column('updated_at', sa.DateTime(timezone=True), server_default=sa.text('now()'), nullable=False),
|
||||
sa.ForeignKeyConstraint(['anchor_page_id'], ['series_page.id'], name='fk_series_chapter_anchor_page', ondelete='CASCADE'),
|
||||
sa.ForeignKeyConstraint(['series_tag_id'], ['tag.id'], name=op.f('fk_series_chapter_series_tag_id_tag'), ondelete='CASCADE'),
|
||||
sa.PrimaryKeyConstraint('id', name=op.f('pk_series_chapter')),
|
||||
sa.UniqueConstraint('anchor_page_id', name='uq_series_chapter_anchor_page')
|
||||
)
|
||||
op.create_index(op.f('ix_series_chapter_series_tag_id'), 'series_chapter', ['series_tag_id'], unique=False)
|
||||
|
||||
# The singleton settings rows. NOT schema — see the note above; the app
|
||||
# reads these with scalar_one() and never creates them, so a fresh
|
||||
# install without these two rows raises NoResultFound on first use.
|
||||
# From 0002 and 0003.
|
||||
op.execute("INSERT INTO import_settings (id) VALUES (1)")
|
||||
op.execute("INSERT INTO ml_settings (id) VALUES (1)")
|
||||
|
||||
# The three hygiene system tags, from 0075. These are PRODUCT data, not
|
||||
# operator configuration — 0075's own docstring says so: "the fix keys on
|
||||
# SYSTEM tags the product ships". The presentation and process auto-apply
|
||||
# sweeps look them up with scalar_one(), so without these rows those
|
||||
# features raise NoResultFound rather than degrading.
|
||||
#
|
||||
# 0075 adopted an existing same-name general tag before inserting, because
|
||||
# an operator might already have tagged `wip` by hand. That cannot happen
|
||||
# on the empty database this file runs against, but the guard is kept: it
|
||||
# costs nothing and makes the statement safe to re-run.
|
||||
for _name in ("wip", "banner", "editor screenshot"):
|
||||
op.execute(
|
||||
sa.text(
|
||||
"INSERT INTO tag (name, kind, is_system) "
|
||||
"SELECT :name, 'general', true WHERE NOT EXISTS ("
|
||||
" SELECT 1 FROM tag WHERE lower(name) = lower(:name)"
|
||||
")"
|
||||
).bindparams(name=_name)
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Deliberately not implemented.
|
||||
|
||||
Downgrading a baseline means dropping every table in the database. That is
|
||||
not a migration, and offering it as one invites someone to run it. Restore
|
||||
from a backup instead.
|
||||
"""
|
||||
raise NotImplementedError(
|
||||
"0089 is the baseline; there is nothing below it. Restore from a backup."
|
||||
)
|
||||
@@ -0,0 +1,64 @@
|
||||
"""service_seen — the learned roster that makes a stopped part observable.
|
||||
|
||||
Milestone 365. Nothing in FabledCurator knew what was SUPPOSED to be running:
|
||||
`celery inspect` reports the workers that answer, so a dead worker was a
|
||||
shorter list rather than a red light, and the only surface that could tell an
|
||||
operator otherwise was Portainer. This table is the memory that turns an
|
||||
absence into something the app can see.
|
||||
|
||||
Keyed on the queue set for a celery role and on agent_id for the GPU agent —
|
||||
NOT on the celery worker name, which here is `celery@<container id>` and is
|
||||
minted fresh on every deploy. See the model docstring for why that choice is
|
||||
the whole design.
|
||||
|
||||
## First migration on the collapsed baseline
|
||||
|
||||
0089 is the single generated baseline that replaced revisions 0001..0089
|
||||
(milestone 328). This is the first revision written on top of it, so it is
|
||||
also the first evidence that the chain steps forward from the collapse rather
|
||||
than merely reproducing the schema — which nothing had demonstrated yet.
|
||||
|
||||
An existing install is at 0089 because it ran the real 0089; a fresh one is at
|
||||
0089 because it ran the baseline. Both arrive here identically, which was the
|
||||
property the collapse was designed around.
|
||||
|
||||
Revision ID: 0090
|
||||
Revises: 0089
|
||||
Create Date: 2026-09-02
|
||||
|
||||
"""
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision: str = "0090"
|
||||
down_revision: Union[str, None] = "0089"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"service_seen",
|
||||
sa.Column("key", sa.String(length=128), nullable=False),
|
||||
sa.Column("kind", sa.String(length=16), nullable=False),
|
||||
sa.Column("display_name", sa.String(length=64), nullable=False),
|
||||
sa.Column(
|
||||
"first_seen_at", sa.DateTime(timezone=True),
|
||||
server_default=sa.text("now()"), nullable=False,
|
||||
),
|
||||
sa.Column(
|
||||
"last_seen_at", sa.DateTime(timezone=True),
|
||||
server_default=sa.text("now()"), nullable=False,
|
||||
),
|
||||
sa.Column("details", sa.JSON(), nullable=False),
|
||||
sa.PrimaryKeyConstraint("key", name=op.f("pk_service_seen")),
|
||||
)
|
||||
# No secondary indexes, deliberately: one row per moving part means every
|
||||
# read is a handful of rows and an index would be write cost buying
|
||||
# nothing (#3301 removed seven of exactly that shape).
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_table("service_seen")
|
||||
@@ -38,6 +38,7 @@ def all_blueprints() -> list[Blueprint]:
|
||||
from .suggestions import suggestions_bp
|
||||
from .system_activity import system_activity_bp
|
||||
from .system_backup import system_backup_bp
|
||||
from .system_health import system_health_bp
|
||||
from .tags import tags_bp
|
||||
from .thumbnails import thumbnails_bp
|
||||
return [
|
||||
@@ -51,6 +52,7 @@ def all_blueprints() -> list[Blueprint]:
|
||||
showcase_bp,
|
||||
settings_bp,
|
||||
system_activity_bp,
|
||||
system_health_bp,
|
||||
system_backup_bp,
|
||||
admin_bp,
|
||||
cleanup_bp,
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user